Compare commits
425
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
00b6f76ee0 | ||
|
|
f9edf4d6ad | ||
|
|
56684ff147 | ||
|
|
b6bbe6b604 | ||
|
|
d279ee06a2 | ||
|
|
c033660d0d | ||
|
|
5e9160f82c | ||
|
|
bce07e9cb9 | ||
|
|
d0d5347cd9 | ||
|
|
31efac1ae1 | ||
|
|
3eca5d8334 | ||
|
|
5eae9ee60b | ||
|
|
d3d8d7ded3 | ||
|
|
c27a478e44 | ||
|
|
a48cb9f1a4 | ||
|
|
3100f0143b | ||
|
|
2bffd6b622 | ||
|
|
c53c43b14f | ||
|
|
14790487a7 | ||
|
|
3c557c9f0a | ||
|
|
419cb52287 | ||
|
|
cab952bb00 | ||
|
|
999cb86071 | ||
|
|
b55b24b7ba | ||
|
|
b660952421 | ||
|
|
b0b4018919 | ||
|
|
a90373ca84 | ||
|
|
4d1a43fd7a | ||
|
|
38107b90ed | ||
|
|
24cc9d14d8 | ||
|
|
ac0020594b | ||
|
|
06d8e2ee45 | ||
|
|
798331ddbc | ||
|
|
9b5803f587 | ||
|
|
694ee0a090 | ||
|
|
bf5a163dcf | ||
|
|
4f90ce02a7 | ||
|
|
5b45c60c9c | ||
|
|
2e5b059530 | ||
|
|
761bdbf24f | ||
|
|
6f5d14fd62 | ||
|
|
ca81c3ebfa | ||
|
|
05e1136027 | ||
|
|
ff0b2f8a4e | ||
|
|
96086add99 | ||
|
|
5c44630ea2 | ||
|
|
425585ee71 | ||
|
|
7179006aee | ||
|
|
b39a705e77 | ||
|
|
d239292655 | ||
|
|
7a8fae0357 | ||
|
|
175d3a1f50 | ||
|
|
4ad80736f4 | ||
|
|
fcf45b6845 | ||
|
|
e80b12521e | ||
|
|
5d42b4241c | ||
|
|
527a1a7ac6 | ||
|
|
18e9be0af7 | ||
|
|
7a4acbce12 | ||
|
|
2379e6163e | ||
|
|
581c6ddef8 | ||
|
|
d09e55e229 | ||
|
|
e553153e48 | ||
|
|
b0930708b6 | ||
|
|
173de3e0d2 | ||
|
|
39f25e5d4e | ||
|
|
bdf584abb2 | ||
|
|
1f7651644b | ||
|
|
1d065adf8b | ||
|
|
2e906ffe3a | ||
|
|
dbafa952ac | ||
|
|
8ef7e80473 | ||
|
|
4467d9dd32 | ||
|
|
ae6b960453 | ||
|
|
92fb0830e0 | ||
|
|
3f45755d58 | ||
|
|
21f104bb77 | ||
|
|
f825a89e23 | ||
|
|
b8c85f7627 | ||
|
|
5107583b97 | ||
|
|
910d81904c | ||
|
|
076feb089a | ||
|
|
e397376820 | ||
|
|
13b36e2bd7 | ||
|
|
a4c2aced55 | ||
|
|
ef428d756c | ||
|
|
4313917b4d | ||
|
|
011e0dbb96 | ||
|
|
f37e7ae326 | ||
|
|
7447dce121 | ||
|
|
93e2d5f365 | ||
|
|
2893b6c974 | ||
|
|
ea0508aaa5 | ||
|
|
75444950f3 | ||
|
|
8236b0e30a | ||
|
|
c2ae7846c9 | ||
|
|
a69c5be8b2 | ||
|
|
930921e8cb | ||
|
|
159e588550 | ||
|
|
6185874f9c | ||
|
|
89e7977943 | ||
|
|
67e72b30d7 | ||
|
|
0e98ffc498 | ||
|
|
efb88f94e3 | ||
|
|
ef480746da | ||
|
|
c5b2afbc35 | ||
|
|
b086dc3c2b | ||
|
|
30a1ed6b9c | ||
|
|
61e34927dc | ||
|
|
680c90b3a8 | ||
|
|
7d629f49e3 | ||
|
|
c04e34620e | ||
|
|
8df5b209a7 | ||
|
|
e8aaf050be | ||
|
|
5062b907bd | ||
|
|
4f5697fdd9 | ||
|
|
955dd1c691 | ||
|
|
ebe51f8e97 | ||
|
|
4e8109770d | ||
|
|
c513f7e6d7 | ||
|
|
a4f586e657 | ||
|
|
c54c64cc9b | ||
|
|
4ff3e40fea | ||
|
|
0aca0eb724 | ||
|
|
db2554dd81 | ||
|
|
955fdb660d | ||
|
|
304aed5813 | ||
|
|
1ffd013de9 | ||
|
|
dc9cfba6bb | ||
|
|
773f427f16 | ||
|
|
7e5e920c72 | ||
|
|
f191dc09d5 | ||
|
|
1c3ef98828 | ||
|
|
e9c71e5d2e | ||
|
|
17201e279d | ||
|
|
3fa5ed1dda | ||
|
|
42894bf93b | ||
|
|
8f59b2e1c2 | ||
|
|
0645dcf173 | ||
|
|
cadd27df5b | ||
|
|
b49ec39aff | ||
|
|
8fadb9f424 | ||
|
|
c233fbca6e | ||
|
|
437e81cfff | ||
|
|
234dd3e36c | ||
|
|
0e8522cfad | ||
|
|
895c79a2fe | ||
|
|
76c97f6c94 | ||
|
|
e5359354b7 | ||
|
|
052098bf36 | ||
|
|
fe377266e1 | ||
|
|
b668878129 | ||
|
|
485bea0f4f | ||
|
|
1ea9132e10 | ||
|
|
04a7f6f6c7 | ||
|
|
5b3d32b37d | ||
|
|
f7e2ab12f2 | ||
|
|
b6cbd2319f | ||
|
|
92c8285549 | ||
|
|
d2b25f154f | ||
|
|
85efde0b4a | ||
|
|
677dc5ec7c | ||
|
|
3c89a31df0 | ||
|
|
1b4a93f65a | ||
|
|
b41583113a | ||
|
|
02e89c1d2d | ||
|
|
5d17712adb | ||
|
|
476960f4b8 | ||
|
|
24f0c71939 | ||
|
|
5705866d40 | ||
|
|
2b7065998a | ||
|
|
6a535e2651 | ||
|
|
a65a2b7f18 | ||
|
|
2c292404d2 | ||
|
|
ba3f476be6 | ||
|
|
ff6d644391 | ||
|
|
cf2b408a63 | ||
|
|
d97b3d703a | ||
|
|
23a4784e72 | ||
|
|
a0160730f2 | ||
|
|
24cbf12f16 | ||
|
|
bcf3ae4856 | ||
|
|
06625b7470 | ||
|
|
aab7ea9e8f | ||
|
|
6a9bb02f37 | ||
|
|
0d908facd3 | ||
|
|
cd828725c7 | ||
|
|
512a6a753f | ||
|
|
6a4707d791 | ||
|
|
6248b411f0 | ||
|
|
479d8b47e0 | ||
|
|
f2ff2c424f | ||
|
|
c5334b1c97 | ||
|
|
d0e3beb3aa | ||
|
|
9a73299594 | ||
|
|
4b02e7d068 | ||
|
|
d7f07fa5c1 | ||
|
|
bdb2c0e36b | ||
|
|
55e0e7e9cf | ||
|
|
00f94d57ed | ||
|
|
4c01267b76 | ||
|
|
d493d4792e | ||
|
|
6559a91495 | ||
|
|
60502593b7 | ||
|
|
a6ed3a5c7d | ||
|
|
3da118d2ee | ||
|
|
7515e5dcbd | ||
|
|
742ed4dfb8 | ||
|
|
2ba4bc97d8 | ||
|
|
d9e4dfb6e6 | ||
|
|
22dc87b07c | ||
|
|
e05530a805 | ||
|
|
989335b67b | ||
|
|
bf4aefcd00 | ||
|
|
378afa1584 | ||
|
|
67958b08d9 | ||
|
|
f512bf3d09 | ||
|
|
8295d01614 | ||
|
|
94b6df986c | ||
|
|
6b3d003950 | ||
|
|
d110b1d945 | ||
|
|
b3058ca46e | ||
|
|
d3d73676c0 | ||
|
|
56abaec75e | ||
|
|
7334e21c93 | ||
|
|
b22b15f00a | ||
|
|
9e608b975c | ||
|
|
9e9b849dd7 | ||
|
|
193a5f8a82 | ||
|
|
3aab433edb | ||
|
|
1c1af460b6 | ||
|
|
c5cd14c2b2 | ||
|
|
16b7359485 | ||
|
|
1207df5189 | ||
|
|
f0db817678 | ||
|
|
9e59499c56 | ||
|
|
d63c76e7ab | ||
|
|
cc1c872a93 | ||
|
|
7acfb79584 | ||
|
|
de2a53f613 | ||
|
|
dda28d6c72 | ||
|
|
408f69ec1d | ||
|
|
73a01f1256 | ||
|
|
956e55c76a | ||
|
|
846c35455d | ||
|
|
ca779b2864 | ||
|
|
20bd381c87 | ||
|
|
5a202f3791 | ||
|
|
8dcce084ca | ||
|
|
45d617c39e | ||
|
|
d345ffbf80 | ||
|
|
17edfe2cf0 | ||
|
|
546fdb84fa | ||
|
|
05b0192a60 | ||
|
|
8bcae3c78e | ||
|
|
f0ecae38b6 | ||
|
|
400e3a9fec | ||
|
|
bd1d8f1a59 | ||
|
|
751edeb7e6 | ||
|
|
b43bd2e67f | ||
|
|
81a0e8685d | ||
|
|
41b7837d0a | ||
|
|
8c51b05b9c | ||
|
|
24412a0e59 | ||
|
|
3bcd443e63 | ||
|
|
37770f594a | ||
|
|
a5e41c1a53 | ||
|
|
b0a1e4f9a6 | ||
|
|
bd36fe883b | ||
|
|
d102c06306 | ||
|
|
6e8421a81e | ||
|
|
8ce6eca34d | ||
|
|
c3850a0b66 | ||
|
|
2bc4cb46a6 | ||
|
|
f7d88bb4fb | ||
|
|
f99587c27d | ||
|
|
2d4b211523 | ||
|
|
10da8f0d09 | ||
|
|
83cda847dd | ||
|
|
78c769f179 | ||
|
|
006bf3b131 | ||
|
|
8cbbef3fae | ||
|
|
55309dd242 | ||
|
|
63648c7000 | ||
|
|
91644d8aaf | ||
|
|
72306c6013 | ||
|
|
e60bde3579 | ||
|
|
c85a8222cc | ||
|
|
2b68791f6a | ||
|
|
f7c362cef5 | ||
|
|
13c095a3da | ||
|
|
591aa71d12 | ||
|
|
b9a2ce3077 | ||
|
|
743c32b512 | ||
|
|
993214723e | ||
|
|
3938f7f8a2 | ||
|
|
dd40bea467 | ||
|
|
afae86f3ea | ||
|
|
f713847e65 | ||
|
|
17fc8b1964 | ||
|
|
9238605661 | ||
|
|
738b9491b2 | ||
|
|
e10df68ed8 | ||
|
|
c4d96c1390 | ||
|
|
b8492bd28d | ||
|
|
a5bd70216c | ||
|
|
17f09375ad | ||
|
|
386bd1d41e | ||
|
|
b5e43bacd7 | ||
|
|
a14ccc36bf | ||
|
|
7f52a6f3ba | ||
|
|
f325d111f3 | ||
|
|
699ee9c447 | ||
|
|
9416c58723 | ||
|
|
6a8ee3ec7f | ||
|
|
a59d83d47d | ||
|
|
bb39be7f24 | ||
|
|
845a9d0125 | ||
|
|
7d7a7e75d4 | ||
|
|
0685037593 | ||
|
|
e92faa23a6 | ||
|
|
40968b3578 | ||
|
|
310448bfcb | ||
|
|
e73ac2af09 | ||
|
|
e01160299a | ||
|
|
5461a13984 | ||
|
|
056092b082 | ||
|
|
a5ca970015 | ||
|
|
e7a7951f1e | ||
|
|
1f71f3bcbc | ||
|
|
3cf8cd86f2 | ||
|
|
34987ec194 | ||
|
|
b58d61cfb7 | ||
|
|
1abd93e0f8 | ||
|
|
6dfd239011 | ||
|
|
07094e34a9 | ||
|
|
f4dee1cd08 | ||
|
|
e38f9123db | ||
|
|
a42b646689 | ||
|
|
74f9f50086 | ||
|
|
d16544b928 | ||
|
|
3b24e6753b | ||
|
|
e815eb922f | ||
|
|
c9c5337a62 | ||
|
|
bb78d70b99 | ||
|
|
a7de15534c | ||
|
|
10d1029ead | ||
|
|
883980f2bd | ||
|
|
d6e426e6d5 | ||
|
|
256e7b89e4 | ||
|
|
f2e704abf3 | ||
|
|
61f36516d7 | ||
|
|
45720fe5a6 | ||
|
|
adf961c883 | ||
|
|
b4a44a2e66 | ||
|
|
17fa783dce | ||
|
|
90e050944f | ||
|
|
a6e90f3ee3 | ||
|
|
0555794850 | ||
|
|
efc2dc53c9 | ||
|
|
5c2f656fe7 | ||
|
|
945b13a1f1 | ||
|
|
9179aa356e | ||
|
|
e94a52a88b | ||
|
|
d54a0f4737 | ||
|
|
aadfd18d4c | ||
|
|
2c6c6c176e | ||
|
|
190918a478 | ||
|
|
38d0d4de02 | ||
|
|
1c85986079 | ||
|
|
c7092722aa | ||
|
|
36356ba8a1 | ||
|
|
85eb7f5ce2 | ||
|
|
8196fab72a | ||
|
|
8ebd488d9e | ||
|
|
42b81d9f1c | ||
|
|
72b9cfb1e1 | ||
|
|
650f355219 | ||
|
|
7f5cfee281 | ||
|
|
c5302e587e | ||
|
|
e1115bc92a | ||
|
|
36d7a6f234 | ||
|
|
4b23ad697c | ||
|
|
e7f2d8575d | ||
|
|
7c1968a34a | ||
|
|
6db13c60b8 | ||
|
|
d99426be94 | ||
|
|
f5505fb03d | ||
|
|
95dcb04454 | ||
|
|
1dba7b465a | ||
|
|
57e938c438 | ||
|
|
3000b40cf3 | ||
|
|
bc820fbd8c | ||
|
|
9066d34eaa | ||
|
|
540fa08907 | ||
|
|
14876b8ae5 | ||
|
|
5935e13866 | ||
|
|
8c3ef996ea | ||
|
|
74fdf0582b | ||
|
|
44f5f8b5c5 | ||
|
|
2f252df084 | ||
|
|
c8c2930fc0 | ||
|
|
417c9516ca | ||
|
|
53dbddb07b | ||
|
|
081341b433 | ||
|
|
d074385944 | ||
|
|
4a1876faf2 | ||
|
|
be88e3fec7 | ||
|
|
e162c013fd | ||
|
|
06dda26d85 | ||
|
|
585e14d5e2 | ||
|
|
183d96ee26 | ||
|
|
bba1560416 | ||
|
|
b36998ef01 | ||
|
|
aef8e766ae | ||
|
|
9ea44d473d | ||
|
|
46203ea761 | ||
|
|
75bdb53342 | ||
|
|
79dfa78e8f | ||
|
|
bdadf3447c | ||
|
|
7706697feb | ||
|
|
c0f704c381 | ||
|
|
a7920bd4b3 | ||
|
|
7e43b5366c | ||
|
|
73bb068264 |
+14
-3
@@ -22,6 +22,9 @@ jobs:
|
|||||||
run: rustup component add rustfmt clippy
|
run: rustup component add rustfmt clippy
|
||||||
- name: Install thumbv7em-none-eabihf target
|
- name: Install thumbv7em-none-eabihf target
|
||||||
run: rustup target add thumbv7em-none-eabihf
|
run: rustup target add thumbv7em-none-eabihf
|
||||||
|
- name: Install wasm32-unknown-unknown target
|
||||||
|
# ci-test.sh builds the reader and clawhdf5-wasm for the browser.
|
||||||
|
run: rustup target add wasm32-unknown-unknown
|
||||||
- name: Install Python interop dependencies
|
- name: Install Python interop dependencies
|
||||||
# The interop suites used to skip silently when python3/h5py were
|
# The interop suites used to skip silently when python3/h5py were
|
||||||
# missing, so they never ran in CI. Install them and make a missing
|
# missing, so they never ran in CI. Install them and make a missing
|
||||||
@@ -31,12 +34,20 @@ jobs:
|
|||||||
# cmake builds libz-ng-sys for the opt-in `fast-deflate` (zlib-ng)
|
# cmake builds libz-ng-sys for the opt-in `fast-deflate` (zlib-ng)
|
||||||
# steps in ci-test.sh; rust:latest does not ship it. The default
|
# steps in ci-test.sh; rust:latest does not ship it. The default
|
||||||
# build (pure-Rust zlib-rs) does not need it.
|
# build (pure-Rust zlib-rs) does not need it.
|
||||||
apt-get install -y --no-install-recommends python3 python3-venv cmake
|
# hdf5-tools: h5ls/h5stat/h5dump/h5diff, which the h5rs
|
||||||
|
# (clawhdf5-tools) interop tests compare against.
|
||||||
|
apt-get install -y --no-install-recommends python3 python3-venv cmake hdf5-tools
|
||||||
python3 -m venv /opt/interop
|
python3 -m venv /opt/interop
|
||||||
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray
|
# maturin + pytest: ci-test.sh builds the Python package
|
||||||
|
# (crates/clawhdf5-py) and runs its tests against h5py.
|
||||||
|
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray h5netcdf hdf5plugin maturin pytest
|
||||||
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
||||||
- name: Show interop library versions
|
- name: Show interop library versions
|
||||||
run: /opt/interop/bin/python -c "import h5py, netCDF4; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__)"
|
# h5dump's version too: the h5rs dump test requires its exact output
|
||||||
|
# (checked against Debian's 1.14.5 in rust:latest and 1.14.6).
|
||||||
|
run: |
|
||||||
|
/opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)"
|
||||||
|
h5dump --version
|
||||||
- name: Run CI script
|
- name: Run CI script
|
||||||
env:
|
env:
|
||||||
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
||||||
|
|||||||
@@ -0,0 +1,58 @@
|
|||||||
|
name: Conformance
|
||||||
|
# Nightly: read every file of the pinned public HDF5 corpora with clawhdf5 and
|
||||||
|
# with h5py/libhdf5 and compare (conformance/run.sh; CONFORMANCE.md explains
|
||||||
|
# the method). Fails on any panic, hang, crash or out-of-memory in clawhdf5,
|
||||||
|
# and when the ok count drops below conformance/baseline.json or a file the
|
||||||
|
# baseline lists as ok stops being ok. The report is printed into the job log;
|
||||||
|
# nothing is uploaded (artifact actions are JavaScript, which rust:latest
|
||||||
|
# cannot run — see CLAUDE.md).
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "17 3 * * *"
|
||||||
|
workflow_dispatch:
|
||||||
|
jobs:
|
||||||
|
conformance:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
container: rust:latest
|
||||||
|
timeout-minutes: 60
|
||||||
|
env:
|
||||||
|
CARGO_NET_RETRY: "10"
|
||||||
|
steps:
|
||||||
|
# Plain git, not actions/checkout (a JavaScript action; see ci.yml).
|
||||||
|
- name: Check out
|
||||||
|
run: |
|
||||||
|
git init -q .
|
||||||
|
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||||
|
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||||
|
git checkout -q FETCH_HEAD
|
||||||
|
- name: Install h5py, h5dump and the probe's codec libraries
|
||||||
|
# hdf5-tools: h5dump for the CVE-corpus comparison. libaec-dev and
|
||||||
|
# pkg-config: the probe builds clawhdf5-format with `szip` (the core
|
||||||
|
# crates' default build needs neither).
|
||||||
|
run: |
|
||||||
|
apt-get update
|
||||||
|
apt-get install -y --no-install-recommends python3 python3-venv hdf5-tools libaec-dev pkg-config
|
||||||
|
python3 -m venv /opt/conformance
|
||||||
|
/opt/conformance/bin/pip install --no-cache-dir -r conformance/requirements.txt
|
||||||
|
/opt/conformance/bin/python -c "import h5py, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'hdf5plugin', hdf5plugin.version)"
|
||||||
|
h5dump --version
|
||||||
|
- name: Probe unit tests
|
||||||
|
run: cargo test --release --manifest-path conformance/probe/Cargo.toml
|
||||||
|
env:
|
||||||
|
CARGO_TARGET_DIR: conformance/.cache/target
|
||||||
|
- name: Reference-side tests
|
||||||
|
run: /opt/conformance/bin/python conformance/test_ref.py
|
||||||
|
- name: Sweep
|
||||||
|
# The corpora come from GitHub (pinned commits, conformance/corpus.txt),
|
||||||
|
# so this job needs a runner that reaches github.com.
|
||||||
|
env:
|
||||||
|
CLAWHDF5_PYTHON: /opt/conformance/bin/python
|
||||||
|
run: bash conformance/run.sh
|
||||||
|
- name: Report
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
if [ -f CONFORMANCE.md ]; then cat CONFORMANCE.md; else echo "no report was generated"; fi
|
||||||
|
if [ -f conformance/.cache/results/summary.md ]; then
|
||||||
|
echo; echo "---- per-file detail (conformance/.cache/results/summary.md) ----"
|
||||||
|
cat conformance/.cache/results/summary.md
|
||||||
|
fi
|
||||||
@@ -5,3 +5,5 @@ benchmarks/longmemeval/*.json
|
|||||||
# Local model weights (MiniLM etc.) — large, not committed
|
# Local model weights (MiniLM etc.) — large, not committed
|
||||||
weights/
|
weights/
|
||||||
.venv
|
.venv
|
||||||
|
__pycache__/
|
||||||
|
.pytest_cache/
|
||||||
|
|||||||
+536
-31
@@ -30,12 +30,15 @@ target: Criterion stretched it where 5 s could not hold the samples it needed
|
|||||||
> and in memory), Consolidation Efficiency, Ephemeral Tier, Multi-modal Search,
|
> and in memory), Consolidation Efficiency, Ephemeral Tier, Multi-modal Search,
|
||||||
> the Search and Read harnesses, and the "h5bench-Equivalent I/O Benchmarks"
|
> the Search and Read harnesses, and the "h5bench-Equivalent I/O Benchmarks"
|
||||||
> and "Independent Validation: tank" sections. What does not yet meet that bar:
|
> and "Independent Validation: tank" sections. What does not yet meet that bar:
|
||||||
> the LongMemEval rows that need real embeddings (not re-run here, except the
|
> the Consolidation Efficiency 100K cycle row and
|
||||||
> dated float16 comparison), the Consolidation Efficiency 100K cycle row and
|
|
||||||
> memory-reduction part (the 2026-09-24 run was stopped before it produced
|
> memory-reduction part (the 2026-09-24 run was stopped before it produced
|
||||||
> them), the int8 side of "Quantising the index copy" (not re-run), and the
|
> them), the int8 side of "Quantising the index copy" (not re-run), and the
|
||||||
> i7-12650H and macOS M3 Max rows under Cross-Platform Notes. That is a
|
> i7-12650H and macOS M3 Max rows under Cross-Platform Notes. That is a
|
||||||
> known, tracked documentation gap, not a claim that those numbers are wrong.
|
> known, tracked documentation gap, not a claim that those numbers are wrong.
|
||||||
|
> The LongMemEval rows that need real embeddings (vector-only, hybrid, RRF,
|
||||||
|
> stemmed hybrid, re-ranking, the weight sweep, the oracle variant) were
|
||||||
|
> re-run on 2026-09-27 on tank; see "Re-run with real embeddings" under
|
||||||
|
> LongMemEval Results.
|
||||||
>
|
>
|
||||||
> **Correctness note (2026-08-06).** Being dated and reproducible is necessary but
|
> **Correctness note (2026-08-06).** Being dated and reproducible is necessary but
|
||||||
> not sufficient — a number can be perfectly reproducible and still measure the
|
> not sufficient — a number can be perfectly reproducible and still measure the
|
||||||
@@ -48,6 +51,31 @@ target: Criterion stretched it where 5 s could not hold the samples it needed
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
## Current headline numbers
|
||||||
|
|
||||||
|
The newest dated measurement of each headline figure, as of 2026-09-28.
|
||||||
|
Everything below this section is the dated record behind them; sections whose
|
||||||
|
figures a later run replaced are marked *Superseded*. Machine "tank" is an AMD
|
||||||
|
Ryzen 7 7800X3D (8C/16T); rows marked idle were run with the 1-minute load
|
||||||
|
average below 2.
|
||||||
|
|
||||||
|
| Figure | Value | Measured | Command | Details |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| Agent memory search, `HDF5Memory::hybrid_search` p50 | 0.49 ms at 10K, 4.69 ms at 100K records | 2026-09-24, tank, `5c8323c` | `cargo run --release -p clawhdf5-bench --bin search_harness -- --full` | [Current: search harness](#current-search-harness-2026-09-24) |
|
||||||
|
| LongMemEval `longmemeval_s` (full haystack), default hybrid 0.4/0.6, turn-level retrieval Hit@5 (not QA accuracy) | 81.4% | 2026-09-27, tank, search code of `7a8fae0` | `longmemeval_bench … --embeddings weights/all-minilm-l6-v2` | [Re-run with real embeddings](#re-run-with-real-embeddings-2026-09-27-tank), [Fusion method](#fusion-method--weighted-vs-rrf-full-haystack-n500) |
|
||||||
|
| Loaded store memory, 100K × 384 | 399 MiB (2.72x raw) with the `f32` index; 256 MiB (1.74x) with the int8 index (int8 side not re-run since it was first measured) | `f32`: 2026-09-24, tank, `5c8323c`; int8: 2026-09-19 (`c0a9206`), machine not recorded | `search_harness -- --footprint --full [--int8]` | [Memory footprint](#memory-footprint), [Quantising the index copy](#quantising-the-index-copy-quantized_index) |
|
||||||
|
| int8 index vs `f32` index, QPS at equal recall | 1.63x (x86-64 AVX2), 1.18x (Raspberry Pi 5, `SDOT`) | x86: 2026-09-20 (`dea02f5`), machine not recorded; Pi 5: 2026-09-21 (`114a2df`); not re-checked against the 2026-09-24 `f32` figure | `search_harness -- --full` | [Quantising the index copy](#quantising-the-index-copy-quantized_index), [On ARM](#on-arm-raspberry-pi-5-cortex-a76) |
|
||||||
|
| `float16` store file size, 100K × 384 | 80.8 MiB vs 154.0 MiB `f32` (48% smaller) | 2026-09-23, tank | `search_harness -- --float16-study --full` | [float16 embedding storage](#float16-embedding-storage-memoryconfigfloat16) |
|
||||||
|
| Full reads of chunked deflate data, 16 threads on one `File` | 4944 MB/s, 1.58x 16 h5py processes (noisy run: compare ratios, not MB/s) | 2026-09-26, tank, `c5334b1` | `concurrent_read` + `concurrent_read_h5py.py` | [Results after in-place chunk decoding](#results-after-in-place-chunk-decoding-2026-09-26-tank-c5334b1) |
|
||||||
|
| Same, clawhdf5 only, against the build before range-read M2/M3 | 8525 MB/s vs 6258 (+36%); contiguous and metadata reads at parity | 2026-09-27, tank (idle), `7a8fae0` vs `8f59b2e` | `concurrent_read --decode-threads 1 --reps 3` | [Local metadata and data reads after range-read M2/M3](#local-metadata-and-data-reads-after-range-read-m2m3-2026-09-27-tank) |
|
||||||
|
| `ObjectHeader::parse` (401 headers) | 23.5–23.6 µs, 1.0–2.6% below `8f59b2e` | 2026-09-27, tank (idle), `96086ad` | `cargo bench -p clawhdf5 --bench local_metadata_bench` | [`ObjectHeader::parse` back at 8f59b2e's speed](#objectheaderparse-back-at-8f59b2es-speed-2026-09-27-tank) |
|
||||||
|
| Selection reads, 64 MB chunked + deflate `f64` | full 63.2 ms; one 64 × 64 window 0.18 ms | 2026-09-24, tank, `5c8323c` | `cargo run --release -p clawhdf5-bench --bin read_harness` | [Current: read harness](#current-read-harness-2026-09-24) |
|
||||||
|
| Deflate backend, zlib-rs (default) vs zlib-ng | within 6% on every HDF5 read/write path | 2026-09-23, tank | `cargo bench -p clawhdf5-filters --bench deflate_bench` (and the two commands with it) | [Deflate backend](#deflate-backend-zlib-rs-vs-zlib-ng) |
|
||||||
|
| vs libhdf5 1.14.6: chunked deflate-6 write 512×512 / 128 attributes / 64 groups | 35x (1.46 vs 51.4 ms, pure-Rust deflate) / 10.3x / 10.6x | write 2026-09-23, tank; attributes and groups 2026-08-03, tank | `cargo bench -p clawhdf5-bench --bench h5bench_write --features libhdf5-compare -- '^write_2d_chunked/'`; `cargo bench -p clawhdf5-bench --features libhdf5-compare` | [Deflate backend](#deflate-backend-zlib-rs-vs-zlib-ng), [Independent Validation: tank](#independent-validation-tank-ryzen-7-7800x3d-2026-08-03) |
|
||||||
|
| Signed checkpoints | about 20% of a checkpoint (598 vs 495 ms at 100K) | 2026-09-25, tank | `search_harness -- --signing-study --full` | [Signed checkpoints](#signed-checkpoints) |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## Memory footprint
|
## Memory footprint
|
||||||
|
|
||||||
`cargo run --release -p clawhdf5-bench --bin search_harness -- --footprint --full`,
|
`cargo run --release -p clawhdf5-bench --bin search_harness -- --footprint --full`,
|
||||||
@@ -61,6 +89,10 @@ change at all. Measured that way a store holding the corpus twice and one
|
|||||||
holding it once came out *identical* (1.00x both), which is how the first
|
holding it once came out *identical* (1.00x both), which is how the first
|
||||||
attempt at this measurement went.
|
attempt at this measurement went.
|
||||||
|
|
||||||
|
> *Superseded* by the current figures below (2026-09-24): this table is the
|
||||||
|
> record of the double-copy fix (commit 2e7e045, undated); the store measured
|
||||||
|
> 2.72x, not 2.43x, by the time the int8 index landed.
|
||||||
|
|
||||||
| N | vectors (raw) | reopened, before | reopened, after |
|
| N | vectors (raw) | reopened, before | reopened, after |
|
||||||
|---:|---:|---:|---:|
|
|---:|---:|---:|---:|
|
||||||
| 1 000 | 1 MiB | 5 MiB (3.41x) | 4 MiB (2.39x) |
|
| 1 000 | 1 MiB | 5 MiB (3.41x) | 4 MiB (2.39x) |
|
||||||
@@ -325,6 +357,19 @@ which of two gold sessions ranks first, out of ~320. Those flips show the
|
|||||||
half-precision path was in effect; they do not change a single hit. The f32
|
half-precision path was in effect; they do not change a single hit. The f32
|
||||||
run reproduces the published hybrid numbers exactly.
|
run reproduces the published hybrid numbers exactly.
|
||||||
|
|
||||||
|
**Re-checked 2026-09-27** (tank, commit 7a8fae0, the same pair of runs on
|
||||||
|
`longmemeval_s_cleaned.json`, which is the same file; the machine was not
|
||||||
|
idle, which does not affect recall): the result is the same. The table above
|
||||||
|
reproduced exactly. Every Hit@k and MRR of the eight modes matched between f32
|
||||||
|
and float16 at both levels, with two exceptions: RRF's session MRR (0.9253 vs
|
||||||
|
0.9254) and two per-type session MRRs in the fourth decimal. Three modes
|
||||||
|
differed by one question in the recency count. That re-run also corrects the
|
||||||
|
sentence above: two f32 runs on the same day differed by one question in
|
||||||
|
recency as well, so those flips are run-to-run variation and do not show that
|
||||||
|
the half-precision path was in effect. `--float16` is what shows that: the
|
||||||
|
harness prints "Stores use MemoryConfig::float16" and `MemoryConfig::float16`
|
||||||
|
is set on every store.
|
||||||
|
|
||||||
### Opening a store (`read_from_disk`)
|
### Opening a store (`read_from_disk`)
|
||||||
|
|
||||||
`HDF5Memory::open` memory-mapped the file, copied the whole mapping into a
|
`HDF5Memory::open` memory-mapped the file, copied the whole mapping into a
|
||||||
@@ -363,6 +408,9 @@ point: does a selection cost what the *selection* costs?
|
|||||||
|
|
||||||
### Baseline (v2.4.0): every selection decodes the whole dataset
|
### Baseline (v2.4.0): every selection decodes the whole dataset
|
||||||
|
|
||||||
|
> *Superseded* by [Current: read harness](#current-read-harness-2026-09-24)
|
||||||
|
> (2026-09-24). Kept as the before picture.
|
||||||
|
|
||||||
4096 x 2048 f64 (64 MB per dataset), chunks 256 x 256, file 129 MB
|
4096 x 2048 f64 (64 MB per dataset), chunks 256 x 256, file 129 MB
|
||||||
|
|
||||||
| layout | read | selected | time ms | MB/s of selection | vs full read |
|
| layout | read | selected | time ms | MB/s of selection | vs full read |
|
||||||
@@ -388,6 +436,9 @@ point: does a selection cost what the *selection* costs?
|
|||||||
|
|
||||||
### After: partial reads
|
### After: partial reads
|
||||||
|
|
||||||
|
> *Superseded* by [Current: read harness](#current-read-harness-2026-09-24)
|
||||||
|
> (2026-09-24).
|
||||||
|
|
||||||
Only the rows of a contiguous dataset, or the chunks, that overlap the
|
Only the rows of a contiguous dataset, or the chunks, that overlap the
|
||||||
selection's bounding box are read/decoded. A 64 x 64 window of the compressed
|
selection's bounding box are read/decoded. A 64 x 64 window of the compressed
|
||||||
dataset: **105 -> 0.39 ms**; one row: **106 -> 2.7 ms**; one column:
|
dataset: **105 -> 0.39 ms**; one row: **106 -> 2.7 ms**; one column:
|
||||||
@@ -419,6 +470,9 @@ because the machine's speed drifted; compare the *vs full read* column.)
|
|||||||
|
|
||||||
### After: parallel cached decode, fewer copies (full reads)
|
### After: parallel cached decode, fewer copies (full reads)
|
||||||
|
|
||||||
|
> *Superseded* by [Current: read harness](#current-read-harness-2026-09-24)
|
||||||
|
> (2026-09-24).
|
||||||
|
|
||||||
Full-read times, old and new binaries run alternately at the same moment (this
|
Full-read times, old and new binaries run alternately at the same moment (this
|
||||||
machine's absolute speed drifts over a long session, so only same-moment
|
machine's absolute speed drifts over a long session, so only same-moment
|
||||||
comparisons mean anything):
|
comparisons mean anything):
|
||||||
@@ -482,8 +536,338 @@ The rows and columns of the uncompressed layouts are within 20% (chunked
|
|||||||
column 0.45 -> 0.49 ms, contiguous column 2.55 -> 2.61 ms). This run does not
|
column 0.45 -> 0.49 ms, contiguous column 2.55 -> 2.61 ms). This run does not
|
||||||
explain the slower windows.
|
explain the slower windows.
|
||||||
|
|
||||||
|
## Local file speed after range reads
|
||||||
|
|
||||||
|
### `ObjectHeader::parse` back at 8f59b2e's speed (2026-09-27, tank)
|
||||||
|
|
||||||
|
The remaining 4% (below) was the call to the version-1 message loop, which
|
||||||
|
`4313917` kept out of line with `#[inline(never)]`. Found with A/B builds
|
||||||
|
changing one piece at a time (perf is not available: `perf_event_paranoid`
|
||||||
|
4): `#[inline]` on `parse_v1_messages` alone brought
|
||||||
|
`object_header_parse_x401` from about 24.5–24.9 µs to 23.6–24.0 µs against
|
||||||
|
8f59b2e's 23.6–24.1 µs (short 4-second rounds); no attribute measured like
|
||||||
|
`#[inline(never)]`;
|
||||||
|
creating the chunk list only when a continuation is found measured no
|
||||||
|
faster on top and was not kept.
|
||||||
|
|
||||||
|
Same method as below: `8f59b2e` built in its own worktree and target
|
||||||
|
directory, separate binaries alternating, `taskset -c 5
|
||||||
|
local_metadata_bench --bench --warm-up-time 3 --measurement-time 10`, every
|
||||||
|
binary started with the 1-minute load average below 2 (0.19–1.86) and no
|
||||||
|
`rustc` running. Candidate: `96086ad` (this change). Median (range) of 3
|
||||||
|
rounds; run 2 also alternated `main` `425585e`.
|
||||||
|
|
||||||
|
| function | 8f59b2e | 425585e (main) | 96086ad | vs 8f59b2e |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| run 1: `object_header_parse_x401` | 23.81 µs (23.76–24.00) | | 23.57 µs (23.23–23.89) | **−1.0%** |
|
||||||
|
| run 1: `snod_parse_all` | 1.840 µs (1.837–1.854) | | 1.839 µs (1.837–1.873) | 0.0% |
|
||||||
|
| run 1: `btree_v1_walk` | 343 ns (338–349) | | 352 ns (344–363) | +2.7% |
|
||||||
|
| run 1: `facade_list_400_groups` | 8.03 ms (8.01–8.10) | | 8.05 ms (8.01–8.15) | +0.2% |
|
||||||
|
| run 2: `object_header_parse_x401` | 24.15 µs (23.74–24.41) | 24.93 µs (24.72–25.05) | 23.52 µs (23.34–23.68) | **−2.6%** |
|
||||||
|
| run 2: `snod_parse_all` | 1.843 µs (1.830–1.847) | 1.856 µs (1.850–1.865) | 1.861 µs (1.858–1.869) | +1.0% |
|
||||||
|
| run 2: `btree_v1_walk` | 346 ns (339–366) | 356 ns (347–356) | 357 ns (350–359) | +3.2% |
|
||||||
|
| run 2: `facade_list_400_groups` | 8.09 ms (8.02–8.10) | 8.12 ms (7.97–8.19) | 8.08 ms (7.98–8.12) | −0.1% |
|
||||||
|
|
||||||
|
- `ObjectHeader::parse` is at or below 8f59b2e (−1.0%, −2.6%) and 5.6%
|
||||||
|
faster than `main` in the same run.
|
||||||
|
- `btree_v1_walk` (one walk of a 350 ns B-tree) is 3% above 8f59b2e in
|
||||||
|
both runs, with overlapping ranges, and is the same on `main` (+0.2%
|
||||||
|
between `main` and this change): not from this change. The walk's code
|
||||||
|
changed in `e553153` (after a failed child the siblings are only read,
|
||||||
|
so the error returns after them; the fixture never takes that path, but
|
||||||
|
the loop carries the extra state); left as is.
|
||||||
|
- `snod_parse_all` and the facade listing are within noise.
|
||||||
|
|
||||||
|
### Local metadata and data reads after range-read M2/M3 (2026-09-27, tank)
|
||||||
|
|
||||||
|
> The `object_header_parse_x401` row (+4.2%) is *superseded* by
|
||||||
|
> [`ObjectHeader::parse` back at 8f59b2e's speed](#objectheaderparse-back-at-8f59b2es-speed-2026-09-27-tank)
|
||||||
|
> (2026-09-27, `96086ad`); the other rows are current.
|
||||||
|
|
||||||
|
`main` just before range-read M2/M3 (`8f59b2e`, PR #17) against `main`
|
||||||
|
`7a8fae0` (PRs #18 and #19), each built in its own worktree and run as
|
||||||
|
separate binaries, alternating base and candidate. Machine: tank (AMD Ryzen
|
||||||
|
7 7800X3D, 16 threads). **Idle:** every round started with the 1-minute load
|
||||||
|
average below 2 (1.05–1.98; `target/ab-results2/load.log`). Criterion:
|
||||||
|
`taskset -c 5 local_metadata_bench --bench --warm-up-time 3
|
||||||
|
--measurement-time 10`, 3 rounds each. Reads: `concurrent_read --dir
|
||||||
|
~/.cache/concurrent-read --decode-threads 1 --reps 3`, 3 rounds each.
|
||||||
|
Median (range) over the rounds.
|
||||||
|
|
||||||
|
`local_metadata_bench` (the 400-group v1 fixture):
|
||||||
|
|
||||||
|
| function | 8f59b2e | 7a8fae0 | change |
|
||||||
|
|---|---:|---:|---:|
|
||||||
|
| `object_header_parse_x401` | 23.86 µs (23.79–23.96) | 24.86 µs (24.69–24.98) | **+4.2%** |
|
||||||
|
| `snod_parse_all` | 1.842 µs (1.835–1.847) | 1.833 µs (1.833–1.854) | −0.5% |
|
||||||
|
| `btree_v1_walk` | 344 ns (338–346) | 348 ns (337–356) | +1.1% |
|
||||||
|
| `facade_list_400_groups` | 8.24 ms (8.04–8.28) | 8.10 ms (8.04–8.13) | −1.7% |
|
||||||
|
|
||||||
|
`concurrent_read`, MB/s (64 datasets of 64 MiB `f32`; deflate chunks
|
||||||
|
256 x 256, level 4):
|
||||||
|
|
||||||
|
| layout | mode | threads | 8f59b2e | 7a8fae0 | change |
|
||||||
|
|---|---|---:|---:|---:|---:|
|
||||||
|
| deflate | distinct | 1 | 891 (880–892) | 907 (906–908) | +1.7% |
|
||||||
|
| deflate | distinct | 2 | 1692 (1680–1699) | 1777 (1767–1779) | +5.0% |
|
||||||
|
| deflate | distinct | 4 | 3102 (3099–3102) | 3348 (3347–3350) | +8.0% |
|
||||||
|
| deflate | distinct | 8 | 5306 (5168–5345) | 6240 (6226–6248) | +17.6% |
|
||||||
|
| deflate | distinct | 16 | 6258 (6109–6442) | 8525 (8513–8561) | **+36.2%** |
|
||||||
|
| deflate | same | 1 | 234 (234–235) | 233 (233–234) | −0.5% |
|
||||||
|
| deflate | same | 16 | 2443 (2038–2444) | 2468 (2424–2473) | +1.0% |
|
||||||
|
| contiguous | distinct | 1 | 13477 (12949–13874) | 13302 (13287–13578) | −1.3% |
|
||||||
|
| contiguous | distinct | 16 | 12501 (12484–12530) | 12566 (12449–12567) | +0.5% |
|
||||||
|
| contiguous | same | 1 | 29866 (28832–30207) | 29364 (28838–29780) | −1.7% |
|
||||||
|
| contiguous | same | 16 | 163415 (161097–229146) | 233280 (159377–238440) | (noise) |
|
||||||
|
|
||||||
|
What this shows:
|
||||||
|
- **Local metadata reads are at parity or faster.** Listing the 400-group
|
||||||
|
file through the facade is 1.7% faster than before M2/M3; the +7–10%
|
||||||
|
listing regression found while merging #18 is gone.
|
||||||
|
- **`ObjectHeader::parse` alone is 4.2% slower** (about 2.5 ns per header;
|
||||||
|
the base and candidate ranges do not overlap). It is the cost of reading
|
||||||
|
continuation chunks from a bounded queue (the fix for unbounded reads on
|
||||||
|
crafted headers) and does not show in the listing. (Fixed later the
|
||||||
|
same day; see the section above and `docs/known-issues.md`.)
|
||||||
|
- **Full reads of deflate data got faster** after #18 (in-place chunk
|
||||||
|
decoding into the typed output and per-thread scratch buffers): +1.7% on
|
||||||
|
one thread, +36% at 16.
|
||||||
|
- Single-thread contiguous hyperslabs are within noise (−1.7%, overlapping
|
||||||
|
ranges). The multi-thread `contiguous same` rows read one 64 MiB dataset
|
||||||
|
out of the CPU caches and swing widely between rounds of the same build.
|
||||||
|
|
||||||
|
An earlier run the same day at load 2.3–3.3 (two orphaned h5py processes,
|
||||||
|
since stopped, each using a core) reported that single-thread contiguous
|
||||||
|
hyperslab row as −5.6%; the idle rerun above does not reproduce it.
|
||||||
|
|
||||||
|
### Results after in-place chunk decoding (2026-09-26, tank, `c5334b1`)
|
||||||
|
|
||||||
|
Same machine, files and commands, re-run after chunked reads started
|
||||||
|
decoding into reusable per-thread buffers straight into the (typed) output,
|
||||||
|
with the calling thread decoding alongside the pool. Load average 1.78 at
|
||||||
|
the start; it rose to 6-9 during the runs (the clawhdf5 runs' own threads,
|
||||||
|
and it stayed around 5-6 through the h5py runs, so something else was
|
||||||
|
active). **This run was noisier than the previous one: h5py's own contiguous
|
||||||
|
figures are about 40% lower than in the run below, and ours dropped
|
||||||
|
similarly, so compare ratios within a run rather than MB/s across runs.**
|
||||||
|
h5py was re-run in the same session.
|
||||||
|
|
||||||
|
Each read decoding on its calling thread (`--decode-threads 1`, like h5py):
|
||||||
|
|
||||||
|
| layout | mode | threads | clawhdf5 MB/s (eff) | h5py threads MB/s (eff) | h5py processes MB/s (eff) | vs h5py processes |
|
||||||
|
|---|---|---:|---:|---:|---:|---:|
|
||||||
|
| deflate | distinct | 1 | 670 (1.00) | 410 (1.00) | 397 (1.00) | 1.69x |
|
||||||
|
| deflate | distinct | 4 | 2434 (0.91) | 406 (0.25) | 1470 (0.93) | 1.66x |
|
||||||
|
| deflate | distinct | 8 | 3749 (0.70) | 406 (0.12) | 2398 (0.76) | 1.56x |
|
||||||
|
| deflate | distinct | 16 | 4944 (0.46) | 390 (0.06) | 3135 (0.49) | 1.58x |
|
||||||
|
| deflate | same | 1 | 211 (1.00) | 125 (1.00) | 124 (1.00) | 1.70x |
|
||||||
|
| deflate | same | 16 | 1835 (0.54) | 122 (0.06) | 961 (0.48) | 1.91x |
|
||||||
|
| contiguous | distinct | 1 | 6718 (1.00) | 5545 (1.00) | 5200 (1.00) | 1.29x |
|
||||||
|
| contiguous | distinct | 16 | 11035 (0.10) | 4950 (0.06) | 10558 (0.13) | 1.05x |
|
||||||
|
| contiguous | same | 1 | 14483 (1.00) | 2593 (1.00) | 2737 (1.00) | 5.29x |
|
||||||
|
| contiguous | same | 16 | 132175 (0.57) | 2224 (0.05) | 14809 (0.34) | 8.93x |
|
||||||
|
|
||||||
|
With the default rayon pool, deflate `distinct` reads 6143 MB/s from a single
|
||||||
|
thread (15x h5py's 410 on one call) and 4556 MB/s at 16 threads (1.45x h5py
|
||||||
|
processes); the other rows are within the noise of the table above.
|
||||||
|
|
||||||
|
What changed: full reads of chunked datasets were 0.69x-0.76x of h5py
|
||||||
|
processes at 16 threads in the run below, and are 1.58x here; with one
|
||||||
|
thread they were 1.44x and are 1.69x. Minor page faults for the 16-thread
|
||||||
|
run fell from about 4.6M to 0.2M (`/usr/bin/time -v`, provisional, loaded
|
||||||
|
machine). clawhdf5 now reads faster than 16 h5py processes in every row of
|
||||||
|
this benchmark except contiguous full reads at 16 threads, where both
|
||||||
|
saturate memory bandwidth (1.05x).
|
||||||
|
|
||||||
|
### Results after the read fixes (2026-09-26, tank, `408f69e`)
|
||||||
|
|
||||||
|
> *Superseded* by [Results after in-place chunk decoding](#results-after-in-place-chunk-decoding-2026-09-26-tank-c5334b1)
|
||||||
|
> (2026-09-26, `c5334b1`), which closed the 16-thread gap listed at the end
|
||||||
|
> of this section.
|
||||||
|
|
||||||
|
Same machine, files and commands as the first run below, re-run on an idle
|
||||||
|
tank (load average 1.60 at the start; the 1-minute figure rose to about 5
|
||||||
|
during the clawhdf5 runs, mostly their own threads) after two fixes:
|
||||||
|
contiguous reads back their output with transparent huge pages and copy
|
||||||
|
hyperslabs run by run, and full chunked reads no longer queue behind a
|
||||||
|
one-thread rayon pool. h5py was re-run in the same session.
|
||||||
|
|
||||||
|
Each read decoding on its calling thread (`--decode-threads 1`, like h5py):
|
||||||
|
|
||||||
|
| layout | mode | threads | clawhdf5 MB/s (eff) | h5py threads MB/s (eff) | h5py processes MB/s (eff) |
|
||||||
|
|---|---|---:|---:|---:|---:|
|
||||||
|
| deflate | distinct | 1 | 606 (1.00) | 432 (1.00) | 421 (1.00) |
|
||||||
|
| deflate | distinct | 4 | 1816 (0.75) | 428 (0.25) | 1654 (0.98) |
|
||||||
|
| deflate | distinct | 8 | 2943 (0.61) | 428 (0.12) | 3042 (0.90) |
|
||||||
|
| deflate | distinct | 16 | 2142 (0.22) | 375 (0.05) | 3083 (0.46) |
|
||||||
|
| deflate | same | 1 | 154 (1.00) | 130 (1.00) | 129 (1.00) |
|
||||||
|
| deflate | same | 4 | 599 (0.98) | 129 (0.25) | 499 (0.97) |
|
||||||
|
| deflate | same | 16 | 1592 (0.65) | 128 (0.06) | 1399 (0.68) |
|
||||||
|
| contiguous | distinct | 1 | 13665 (1.00) | 9490 (1.00) | 8781 (1.00) |
|
||||||
|
| contiguous | distinct | 16 | 12674 (0.06) | 2285 (0.02) | 6942 (0.05) |
|
||||||
|
| contiguous | same | 1 | 31991 (1.00) | 5087 (1.00) | 5078 (1.00) |
|
||||||
|
| contiguous | same | 16 | 237151 (0.46) | 4304 (0.05) | 35772 (0.44) |
|
||||||
|
|
||||||
|
With the default rayon pool: deflate `distinct` 2117 MB/s at 1 thread (4.9x
|
||||||
|
h5py), 3163 at 4, 2341 at 16 (0.76x h5py processes); deflate `same` 1439 MB/s
|
||||||
|
at 16; contiguous as above within a few percent.
|
||||||
|
|
||||||
|
Before -> after for clawhdf5 (`--decode-threads 1` unless noted):
|
||||||
|
contiguous full read at 1 thread 2495 -> 13665 MB/s (0.25x -> 1.44x h5py);
|
||||||
|
contiguous 256 x 256 hyperslabs at 1 thread 624 -> 31991 MB/s (0.12x ->
|
||||||
|
6.3x); deflate full reads at 8 threads 887 -> 2943 MB/s; deflate
|
||||||
|
hyperslabs at 16 threads 1244 -> 1592 MB/s.
|
||||||
|
|
||||||
|
Read with care:
|
||||||
|
- `contiguous same` reads 1024 slabs of one 64 MiB dataset over and over, so
|
||||||
|
it mostly measures copies out of the CPU's caches (the 7800X3D has 96 MiB
|
||||||
|
of L3); the per-call overhead is what differs (h5py's is about 50 us).
|
||||||
|
- At 16 threads every tool dropped in this run (h5py threads on contiguous
|
||||||
|
data from 8002 to 2285 MB/s, processes from 12846 to 6942), so the
|
||||||
|
16-thread rows are noisier than the others.
|
||||||
|
- Still behind at this commit: full reads of chunked data at 16 threads
|
||||||
|
(0.69x-0.76x h5py processes); fixed by `c5334b1` (above), recorded as
|
||||||
|
fixed in `docs/known-issues.md`.
|
||||||
|
|
||||||
|
### First run, before the read fixes (2026-09-26, tank, `91644d8`)
|
||||||
|
|
||||||
|
> *Superseded* results: the tables and "What this shows" are the before
|
||||||
|
> picture for [Results after in-place chunk decoding](#results-after-in-place-chunk-decoding-2026-09-26-tank-c5334b1)
|
||||||
|
> (2026-09-26). The workload description and the **Run** box below are
|
||||||
|
> still how every `concurrent_read` figure in this file is produced.
|
||||||
|
|
||||||
|
Measured on tank (AMD Ryzen 7 7800X3D, 8 cores / 16 threads, 61 GiB, Linux
|
||||||
|
7.0) at commit `91644d8`, load average 1.84 when the run started (the
|
||||||
|
1-minute figure rose to 3.7 during the runs; that is mostly the benchmark's
|
||||||
|
own threads). Warm page cache. clawhdf5 2.7.0 (workspace), h5py 3.16.0 on
|
||||||
|
HDF5 2.0.0. Commands exactly as in the **Run** box below; files at their
|
||||||
|
defaults (64 datasets of 16384 x 1024 `f32`, 64 MiB each; deflate chunks
|
||||||
|
256 x 256, level 4). MB/s is decoded data, the median of the repetitions;
|
||||||
|
eff is scaling efficiency against the same tool's 1-thread row.
|
||||||
|
|
||||||
|
Each read decoding on its calling thread (`--decode-threads 1`, like h5py):
|
||||||
|
|
||||||
|
| layout | mode | threads | clawhdf5 MB/s (eff) | h5py threads MB/s (eff) | h5py processes MB/s (eff) |
|
||||||
|
|---|---|---:|---:|---:|---:|
|
||||||
|
| deflate | distinct | 1 | 421 (1.00) | 433 (1.00) | 421 (1.00) |
|
||||||
|
| deflate | distinct | 4 | 890 (0.53) | 428 (0.25) | 1651 (0.98) |
|
||||||
|
| deflate | distinct | 16 | 880 (0.13) | 427 (0.06) | 4424 (0.66) |
|
||||||
|
| deflate | same | 1 | 151 (1.00) | 130 (1.00) | 129 (1.00) |
|
||||||
|
| deflate | same | 4 | 490 (0.81) | 129 (0.25) | 497 (0.96) |
|
||||||
|
| deflate | same | 16 | 1244 (0.52) | 128 (0.06) | 1402 (0.68) |
|
||||||
|
| contiguous | distinct | 1 | 2495 (1.00) | 9789 (1.00) | 9169 (1.00) |
|
||||||
|
| contiguous | distinct | 16 | 8083 (0.20) | 8096 (0.05) | 12272 (0.08) |
|
||||||
|
| contiguous | same | 1 | 624 (1.00) | 5022 (1.00) | 5172 (1.00) |
|
||||||
|
| contiguous | same | 16 | 4778 (0.48) | 4411 (0.05) | 37138 (0.45) |
|
||||||
|
|
||||||
|
With the default rayon pool decoding inside each read, deflate `distinct`
|
||||||
|
is 912 MB/s at 1 thread (2.1x h5py) and 2824 MB/s at 16 (6.6x h5py threads,
|
||||||
|
0.64x h5py processes); the other rows are within a few percent of the table
|
||||||
|
above. Full tables (2, 4, 8 threads, both decode modes) come from
|
||||||
|
`compare_concurrent_read.py` on the JSON files.
|
||||||
|
|
||||||
|
What this shows:
|
||||||
|
- **h5py threads do not scale** (flat at about 430 MB/s on deflate, every
|
||||||
|
thread count): libhdf5's global lock.
|
||||||
|
- **clawhdf5 threads on one `File` do, for hyperslab reads of compressed
|
||||||
|
data:** 1244 MB/s at 16 threads, 9.7x h5py threads and 0.89x h5py
|
||||||
|
processes, without a process pool.
|
||||||
|
- **Where clawhdf5 was behind** at `91644d8` (both since fixed; see
|
||||||
|
`docs/known-issues.md`, "Concurrent and contiguous read performance"):
|
||||||
|
- *Full reads of chunked datasets stop scaling at about 4 threads*
|
||||||
|
(about 880 MB/s) while h5py processes reach 4424 MB/s. Hyperslab
|
||||||
|
reads, which bypass the `File`'s chunk cache, keep scaling, so the
|
||||||
|
cache (one mutex and one 16 MiB budget per `File`, thrashed by 64 MiB
|
||||||
|
datasets) is the suspect. The cause of the `--decode-threads 1`
|
||||||
|
ceiling was not the cache: every full read queued its chunks for the
|
||||||
|
pool's single rayon worker. That case was fixed after these
|
||||||
|
measurements (2026-09-26, not yet re-measured here). With the default
|
||||||
|
pool the gap to h5py processes remains (see `docs/known-issues.md`).
|
||||||
|
- *Contiguous reads are slow*: 2.5 GB/s for a single-threaded full read
|
||||||
|
against h5py's 9.8 GB/s (0.25x), and 0.12x for 256 x 256 hyperslabs.
|
||||||
|
Threads close the gap (about 1.0x h5py at 16), but single-thread
|
||||||
|
contiguous I/O is a real deficit.
|
||||||
|
|
||||||
|
The question: libhdf5's threadsafe build serialises every API call under one
|
||||||
|
global mutex, and h5py holds a global lock around every call too, so threads
|
||||||
|
reading through h5py cannot decode in parallel; h5py users scale with
|
||||||
|
processes. A clawhdf5 `File` is `Send + Sync`, and nothing on the read paths
|
||||||
|
this harness uses (`read_f32`, `read_f32_selection`) takes a library-wide
|
||||||
|
lock: the one mutex is the `File`'s chunk cache (keyed per dataset), taken by
|
||||||
|
full reads of chunked datasets for each chunk's O(1) lookup and insert, never
|
||||||
|
across a decode; hyperslab reads do not use the cache. How does
|
||||||
|
decoded throughput scale with threads on one open file, against h5py threads
|
||||||
|
and h5py processes on the same files?
|
||||||
|
|
||||||
|
Workload (`crates/clawhdf5-bench/src/bin/concurrent_read.rs`; the h5py script
|
||||||
|
mirrors it): `<dir>/deflate.h5` and `<dir>/contiguous.h5`, each with 64 `f32`
|
||||||
|
datasets of 64 MiB decoded (`[16384, 1024]`; the deflate file chunked
|
||||||
|
`256 x 256`, level 4), written by clawhdf5 on first use and reused while
|
||||||
|
`manifest.json` matches. The data is a slowly varying ramp plus 8 bits of
|
||||||
|
noise per element, every value exact in `f32`, so both harnesses check what
|
||||||
|
they read; it deflates about 3.1x (128 MiB -> 40.7 MiB for two 64 MiB
|
||||||
|
datasets). For each layout and thread count
|
||||||
|
(1, 2, 4, 8, 16; fixed total work per repetition, split among the threads):
|
||||||
|
|
||||||
|
- `distinct`: every dataset read in full once, thread `t` taking datasets
|
||||||
|
`t, t + T, ...`;
|
||||||
|
- `same`: 1024 random `256 x 256` hyperslabs of `d00` in total, from a seeded
|
||||||
|
splitmix64 stream that both harnesses generate identically.
|
||||||
|
|
||||||
|
Reported per row: MB/s of decoded (selected) data from the median of the
|
||||||
|
repetitions, and scaling efficiency `MB/s(T) / (T x MB/s(1))`. Each worker
|
||||||
|
times itself from a start barrier; a repetition spans the earliest start to
|
||||||
|
the latest finish. Page cache: warm by default (each file is read once before
|
||||||
|
timing); `--cold` evicts the files with `posix_fadvise(POSIX_FADV_DONTNEED)`
|
||||||
|
before every repetition (no root needed; best effort). clawhdf5 opens one
|
||||||
|
`File` per repetition, shared by all threads; h5py threads share one
|
||||||
|
`h5py.File`; h5py processes (spawned before timing) each open the file inside
|
||||||
|
the timed region.
|
||||||
|
|
||||||
|
Decode inside a single clawhdf5 read is itself parallel in this binary
|
||||||
|
(clawhdf5-format's `parallel` feature, enabled here through clawhdf5-agent;
|
||||||
|
it is off in the facade's default features), so a 1-thread clawhdf5 full read
|
||||||
|
of the deflate file already uses the whole rayon pool. Run both
|
||||||
|
`--decode-threads 1` (each read decodes on its calling thread, like h5py —
|
||||||
|
this isolates the API's own scaling) and the default pool.
|
||||||
|
|
||||||
|
> **Run** (from the repository root). The default files take about 5.4 GiB
|
||||||
|
> of disk (4 GiB contiguous + about 1.3 GiB deflate). Generating them is
|
||||||
|
> memory-hungry because `FileBuilder` holds a whole file in memory: peak RSS
|
||||||
|
> was 676 MB for `--datasets 2 --mib 64` (2026-09-25, tank,
|
||||||
|
> `/usr/bin/time -f %M`), about 5x one file's decoded size, so expect about
|
||||||
|
> 21 GB at the defaults (once; later runs reuse the files). Put `--dir` on a
|
||||||
|
> real disk, not tmpfs, if `--cold` is to mean anything.
|
||||||
|
>
|
||||||
|
> ```bash
|
||||||
|
> DIR=/path/on/disk/concurrent-read
|
||||||
|
> BENCH=crates/clawhdf5-bench/scripts
|
||||||
|
> PY=.venv/bin/python # h5py 3.16 / HDF5 2.0 in this repo
|
||||||
|
> cargo build --release -p clawhdf5-bench --bin concurrent_read
|
||||||
|
> B=target/release/concurrent_read
|
||||||
|
> $B --dir $DIR --json claw-pool.json # generates on first run
|
||||||
|
> $B --dir $DIR --decode-threads 1 --json claw-1.json
|
||||||
|
> $PY $BENCH/concurrent_read_h5py.py --dir $DIR --executor threads --json h5py-threads.json
|
||||||
|
> $PY $BENCH/concurrent_read_h5py.py --dir $DIR --executor processes --json h5py-procs.json
|
||||||
|
> $PY $BENCH/compare_concurrent_read.py claw-1.json h5py-threads.json h5py-procs.json
|
||||||
|
> $PY $BENCH/compare_concurrent_read.py claw-pool.json h5py-threads.json h5py-procs.json
|
||||||
|
> ```
|
||||||
|
>
|
||||||
|
> Cold page cache: add `--cold` to every harness command. Smoke test (seconds):
|
||||||
|
> `$B --dir /tmp/cr --datasets 4 --mib 1 --threads 1,2,4 --slabs 16 --reps 1`
|
||||||
|
> and the same `--threads/--slabs/--reps` to the h5py script.
|
||||||
|
|
||||||
|
Other flags (both harnesses): `--threads`, `--reps`, `--slab`, `--slabs`,
|
||||||
|
`--seed`, `--modes distinct,same`, `--layouts deflate,contiguous`; sizes
|
||||||
|
(`--datasets`, `--mib`) only on the Rust harness, which writes the files.
|
||||||
|
|
||||||
## Search harness baseline (v2.3.0)
|
## Search harness baseline (v2.3.0)
|
||||||
|
|
||||||
|
> *Historical.* This baseline and the "After: …" subsections that follow
|
||||||
|
> record each step of the search work; they are *superseded* by
|
||||||
|
> [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24),
|
||||||
|
> the last subsection of this part.
|
||||||
|
|
||||||
Produced by `cargo run --release -p clawhdf5-bench --bin search_harness -- --full`
|
Produced by `cargo run --release -p clawhdf5-bench --bin search_harness -- --full`
|
||||||
on deterministic **clustered** synthetic data (384-dim, unit-normalised; points =
|
on deterministic **clustered** synthetic data (384-dim, unit-normalised; points =
|
||||||
cluster centre + noise — uniform random vectors are nearly equidistant in high
|
cluster centre + noise — uniform random vectors are nearly equidistant in high
|
||||||
@@ -546,10 +930,11 @@ build: 9752.6 ms (10254 vectors/s) · exact scan: 40 QPS, p50 24648 µs
|
|||||||
| 1000 | 11 | 3.9 | 0.9 | 68.1 | 5.48 | 5.57 | 182.5 |
|
| 1000 | 11 | 3.9 | 0.9 | 68.1 | 5.48 | 5.57 | 182.5 |
|
||||||
| 10000 | 114 | 32.2 | 10.9 | 845.0 | 48.56 | 78.65 | 19.8 |
|
| 10000 | 114 | 32.2 | 10.9 | 845.0 | 48.56 | 78.65 | 19.8 |
|
||||||
| 100000 | 1486 | 713.0 | 354.5 | 10486.5 | 883.51 | 975.23 | 1.1 |
|
| 100000 | 1486 | 713.0 | 354.5 | 10486.5 | 883.51 | 975.23 | 1.1 |
|
||||||
wrote /tmp/claude-1000/-home-osobh-projects-clawhdf5/422f755e-dd25-4c35-8613-5439087e3aaa/scratchpad/baseline_full.json
|
|
||||||
|
|
||||||
### After: HNSW neighbour-selection heuristic
|
### After: HNSW neighbour-selection heuristic
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
Same harness, same data, after replacing closest-M neighbour selection with the
|
Same harness, same data, after replacing closest-M neighbour selection with the
|
||||||
HNSW paper's diversity heuristic (Algorithm 4, keeping pruned connections) for
|
HNSW paper's diversity heuristic (Algorithm 4, keeping pruned connections) for
|
||||||
both new links and back-link pruning. Recall@10 at `ef = 64`: **0.87 → 1.00**
|
both new links and back-link pruning. Recall@10 at `ef = 64`: **0.87 → 1.00**
|
||||||
@@ -595,6 +980,8 @@ build: 36472.8 ms (2742 vectors/s) · exact scan: 40 QPS, p50 24644 µs
|
|||||||
|
|
||||||
### After: persistent keyword index, no store rewrite per query
|
### After: persistent keyword index, no store rewrite per query
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
`hybrid_search` used to rebuild the BM25 index from scratch (re-tokenising every
|
`hybrid_search` used to rebuild the BM25 index from scratch (re-tokenising every
|
||||||
record) and rewrite the whole `.h5` file on **every query**. The index is now
|
record) and rewrite the whole `.h5` file on **every query**. The index is now
|
||||||
kept for the life of the store and updated incrementally, and activation boosts
|
kept for the life of the store and updated incrementally, and activation boosts
|
||||||
@@ -615,6 +1002,8 @@ index removes that.
|
|||||||
|
|
||||||
### After: vector index persisted with the checkpoint
|
### After: vector index persisted with the checkpoint
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
The HNSW graph (not the vectors, which the store already holds) is saved to
|
The HNSW graph (not the vectors, which the store already holds) is saved to
|
||||||
`<store>.h5.ann` at each checkpoint and reloaded by `open()`, tied to that
|
`<store>.h5.ann` at each checkpoint and reloaded by `open()`, tied to that
|
||||||
checkpoint by a generation id. The index is now built once per store (the *cold
|
checkpoint by a generation id. The index is now built once per store (the *cold
|
||||||
@@ -632,6 +1021,8 @@ index incrementally.
|
|||||||
|
|
||||||
### After: unit-vector dot product, reusable visited set
|
### After: unit-vector dot product, reusable visited set
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
Cosine distance recomputed both vector norms on every evaluation; the index now
|
Cosine distance recomputed both vector norms on every evaluation; the index now
|
||||||
stores unit vectors and uses a plain dot product. The per-call `HashSet` of
|
stores unit vectors and uses a plain dot product. The per-call `HashSet` of
|
||||||
visited nodes became a reusable epoch-stamped array. Recall is unchanged.
|
visited nodes became a reusable epoch-stamped array. Recall is unchanged.
|
||||||
@@ -677,6 +1068,8 @@ build: 21084.6 ms (4743 vectors/s) · exact scan: 39 QPS, p50 24739 µs
|
|||||||
|
|
||||||
### After: unranked keyword scores, top-k merge (rankings unchanged)
|
### After: unranked keyword scores, top-k merge (rankings unchanged)
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
A fusion study (`search_harness --fusion-study`) showed that capping the
|
A fusion study (`search_harness --fusion-study`) showed that capping the
|
||||||
keyword candidate pool is **not** a safe optimisation: against the current
|
keyword candidate pool is **not** a safe optimisation: against the current
|
||||||
full-corpus normalisation the final top-10 overlap is only 0.83-0.92 and the
|
full-corpus normalisation the final top-10 overlap is only 0.83-0.92 and the
|
||||||
@@ -699,6 +1092,8 @@ results.
|
|||||||
|
|
||||||
### After: batched bulk build (optionally parallel); deletions handled in search
|
### After: batched bulk build (optionally parallel); deletions handled in search
|
||||||
|
|
||||||
|
> *Superseded* by [Current: search harness](#current-search-harness-2026-09-24) (2026-09-24).
|
||||||
|
|
||||||
Profiling showed **90% of a build's distance evaluations are in back-link
|
Profiling showed **90% of a build's distance evaluations are in back-link
|
||||||
pruning**. The bulk build now inserts in batches: plan each node's neighbours
|
pruning**. The bulk build now inserts in batches: plan each node's neighbours
|
||||||
against the graph as it stood at the start of the batch, link, then prune every
|
against the graph as it stood at the start of the batch, link, then prune every
|
||||||
@@ -1197,10 +1592,75 @@ turn-level row plus its session Hit@1 in the tokenizer table. The run also
|
|||||||
produced figures this document does not publish (stemmed session Hit@5,
|
produced figures this document does not publish (stemmed session Hit@5,
|
||||||
Hit@10 and MRR, and per-type Hit@5/Hit@10/MRR for both modes), so there was
|
Hit@10 and MRR, and per-type Hit@5/Hit@10/MRR for both modes), so there was
|
||||||
nothing to compare them with. Rows that need real embeddings (vector-only,
|
nothing to compare them with. Rows that need real embeddings (vector-only,
|
||||||
hybrid, RRF, re-ranking, the weight sweep) were not re-run.
|
hybrid, RRF, re-ranking, the weight sweep) were not re-run then; they were on
|
||||||
|
2026-09-27 (next section).
|
||||||
|
|
||||||
|
### Re-run with real embeddings (2026-09-27, tank)
|
||||||
|
|
||||||
|
Every recall row in this section that needs real embeddings was measured
|
||||||
|
again on 2026-09-27 on tank (AMD Ryzen 7 7800X3D, 16 threads; MiniLM
|
||||||
|
embeddings on an RTX 5060 Ti, retrieval on the CPU), with the search code of
|
||||||
|
commit 7a8fae0. Six runs:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo build --release -p clawhdf5-bench --bin longmemeval_bench --features embeddings-cuda
|
||||||
|
B=target/release/longmemeval_bench W=weights/all-minilm-l6-v2
|
||||||
|
$B benchmarks/longmemeval/longmemeval_oracle.json --embeddings $W
|
||||||
|
$B benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings $W
|
||||||
|
$B benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings $W --float16
|
||||||
|
$B benchmarks/longmemeval/longmemeval_oracle.json --embeddings $W --sweep
|
||||||
|
$B benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings $W --sweep
|
||||||
|
$B benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings $W --rerank-sweep
|
||||||
|
```
|
||||||
|
|
||||||
|
(`longmemeval_s.json`, used by the commands elsewhere in this section, is the
|
||||||
|
same file as `longmemeval_s_cleaned.json`.)
|
||||||
|
|
||||||
|
**The machine was not idle.** The 1-minute load never fell below 2 in a
|
||||||
|
2-hour wait, because two stray test processes were each holding a core; it
|
||||||
|
was 2.1–2.7 when each run started and up to 8.4 while the runs were going.
|
||||||
|
Recall does not depend on load. Latency does, so no latency figure in this
|
||||||
|
file was updated from these runs.
|
||||||
|
|
||||||
|
**What reproduced exactly:** every published full-haystack Hit@1/5/10 and MRR
|
||||||
|
at turn and session level for BM25, vector-only, hybrid 0.4/0.6, both stemmed
|
||||||
|
modes, and every re-ranking row; RRF except as below; the float16 table; 4 of
|
||||||
|
the 11 weight-sweep rows (0.0, 0.1, 0.3 and 0.9); and the oracle vector-only
|
||||||
|
figure. The headline,
|
||||||
|
hybrid 0.4/0.6 turn Hit@5 **81.4%**, is unchanged.
|
||||||
|
|
||||||
|
**What changed.** Old values are kept here; the tables below show the new ones.
|
||||||
|
|
||||||
|
| Figure | Published | 2026-09-27 | Why |
|
||||||
|
|---|---:|---:|---|
|
||||||
|
| Oracle, hybrid turn Hit@5 | 85.2% (2026-08-07, c913cd1) | **86.8%** | Weights. 85.2% was measured at 0.7/0.3, the default then. The default has been 0.4/0.6 since 29baabb. Today's oracle sweep gives 85.2% at 0.7/0.3 and 86.8% at 0.4/0.6. |
|
||||||
|
| Oracle, BM25-only turn Hit@5 with embeddings | 84.2% (2026-08-07) | 84.4% | Now equal to the zero-embedding figure, as it already was on the full haystack. At c913cd1, tied candidates were ordered by `HashMap` iteration. Since 3ed0489 (2026-09-19) they are broken by index. 3ed0489 is the likely cause; it was not bisected. |
|
||||||
|
| Ablation / sweep, hybrid 0.7/0.3, turn | 44.4% / 79.2% / 86.0% / 0.5868 | 44.2% / 79.2% / 85.8% / 0.5856 | The same tie-breaking change. 29baabb's re-run on 2026-09-19, made after 3ed0489 that same day, already had 44.2% / 85.8% / 0.5856. |
|
||||||
|
| Ablation, hybrid 0.7/0.3, session | 88.2% / 95.8% / 97.8% / 0.9158 | 88.0% / 95.8% / 97.6% / 0.9146 | Same. |
|
||||||
|
| Ablation / sweep, vector-only turn MRR | 0.5027 | 0.5031 | Same. The fusion and float16 tables already had 0.5031. |
|
||||||
|
| Sweep 0.2/0.8, turn | 53.6% / 78.2% / 0.6440 | 53.4% / 78.0% / 0.6429 | Same (the sweep was measured 2026-08-07, 1537a94). |
|
||||||
|
| Sweep 0.4/0.6, 0.5/0.5, 0.8/0.2 turn MRR | 0.6429 / 0.6234 / 0.5571 | 0.6430 / 0.6232 / 0.5574 | Same. |
|
||||||
|
| Sweep 0.6/0.4 | Hit@5 79.8%, MRR 0.6069, session Hit@5 96.6% | 79.6%, 0.6067, 96.4% | Same. |
|
||||||
|
| RRF turn MRR | 0.5967 (2026-09-19, aa92fef) | 0.5969 | Not explained. It may be a later search-path change, such as the int8 index becoming the default in 8b85d93. It may also be the run-to-run variation described next. It was not bisected. |
|
||||||
|
|
||||||
|
**Not exactly deterministic.** Between two f32 runs today, the recency count
|
||||||
|
(the share of `knowledge-update` questions where the newest gold session
|
||||||
|
ranks first) differed by one question in two modes: hybrid 0.4/0.6 was
|
||||||
|
144/320 in one run and 145/320 in the other, and re-rank with a 1-day
|
||||||
|
half-life was 165/319 and 166/319. Each question gets a fresh store, so this
|
||||||
|
is not state carried between modes. The cause was not found; a candidate is
|
||||||
|
that the GPU embeddings are not bit-for-bit identical from run to run. Hit@k
|
||||||
|
and MRR agreed in every run that repeated a mode (hybrid 0.4/0.6 was measured
|
||||||
|
four times: three f32 runs and one float16 run), but a last-digit change in an
|
||||||
|
MRR, or a one-question change in recency, is within this variation.
|
||||||
|
|
||||||
### Full haystack — `longmemeval_s`, n=500 (the number to cite)
|
### Full haystack — `longmemeval_s`, n=500 (the number to cite)
|
||||||
|
|
||||||
|
This table is **BM25-only** (zero embeddings). With real embeddings and the
|
||||||
|
default hybrid 0.4/0.6 the same corpus gives turn Hit@5 **81.4%** (2026-09-27;
|
||||||
|
see [Fusion method](#fusion-method--weighted-vs-rrf-full-haystack-n500)),
|
||||||
|
which is the headline figure.
|
||||||
|
|
||||||
47.7 sessions and 493.5 turns per question; 4.0% of haystack sessions are evidence
|
47.7 sessions and 493.5 turns per question; 4.0% of haystack sessions are evidence
|
||||||
sessions, so retrieval has to actually discriminate.
|
sessions, so retrieval has to actually discriminate.
|
||||||
|
|
||||||
@@ -1227,13 +1687,16 @@ the question.
|
|||||||
|
|
||||||
Real 384-d `all-MiniLM-L6-v2` embeddings, 190,015 unique texts encoded once on an
|
Real 384-d `all-MiniLM-L6-v2` embeddings, 190,015 unique texts encoded once on an
|
||||||
RTX 5060 Ti (~13 min; the same work on the 8-core CPU was still unfinished after
|
RTX 5060 Ti (~13 min; the same work on the 8-core CPU was still unfinished after
|
||||||
30 minutes, so the GPU path is not a convenience here). Turn-level:
|
30 minutes, so the GPU path is not a convenience here). First measured
|
||||||
|
2026-08-07 (c913cd1); the values below are the 2026-09-27 re-run, which moved
|
||||||
|
the vector-only MRR and the `0.7/0.3` row (see the re-run section above).
|
||||||
|
Turn-level:
|
||||||
|
|
||||||
| Mode | Hit@1 | Hit@5 | Hit@10 | MRR |
|
| Mode | Hit@1 | Hit@5 | Hit@10 | MRR |
|
||||||
|------|-------|-------|--------|-----|
|
|------|-------|-------|--------|-----|
|
||||||
| BM25 only (`0.0`/`1.0`) | **53.8%** | 75.0% | 81.6% | **0.6320** |
|
| BM25 only (`0.0`/`1.0`) | **53.8%** | 75.0% | 81.6% | **0.6320** |
|
||||||
| Vector only (`1.0`/`0.0`) | 36.0% | 71.8% | 81.6% | 0.5027 |
|
| Vector only (`1.0`/`0.0`) | 36.0% | 71.8% | 81.6% | 0.5031 |
|
||||||
| Hybrid (`0.7`/`0.3`) | 44.4% | **79.2%** | **86.0%** | 0.5868 |
|
| Hybrid (`0.7`/`0.3`) | 44.2% | **79.2%** | **85.8%** | 0.5856 |
|
||||||
|
|
||||||
Session-level:
|
Session-level:
|
||||||
|
|
||||||
@@ -1241,7 +1704,7 @@ Session-level:
|
|||||||
|------|-------|-------|--------|-----|
|
|------|-------|-------|--------|-----|
|
||||||
| BM25 only | 86.2% | 93.6% | 96.6% | 0.8948 |
|
| BM25 only | 86.2% | 93.6% | 96.6% | 0.8948 |
|
||||||
| Vector only | 85.4% | 94.2% | 96.6% | 0.8901 |
|
| Vector only | 85.4% | 94.2% | 96.6% | 0.8901 |
|
||||||
| Hybrid | **88.2%** | **95.8%** | **97.8%** | **0.9158** |
|
| Hybrid | **88.0%** | **95.8%** | **97.6%** | **0.9146** |
|
||||||
|
|
||||||
### Fusion method — weighted vs. RRF, full haystack, n=500
|
### Fusion method — weighted vs. RRF, full haystack, n=500
|
||||||
|
|
||||||
@@ -1255,7 +1718,7 @@ takes a `Fusion`, and both run over the same HNSW + BM25 candidates:
|
|||||||
| BM25 only | **53.8%** | 75.0% | 81.6% | 0.6320 | 86.2% | 0.8948 |
|
| BM25 only | **53.8%** | 75.0% | 81.6% | 0.6320 | 86.2% | 0.8948 |
|
||||||
| Vector only | 36.0% | 71.8% | 81.6% | 0.5031 | 85.4% | 0.8901 |
|
| Vector only | 36.0% | 71.8% | 81.6% | 0.5031 | 85.4% | 0.8901 |
|
||||||
| **Weighted 0.4 / 0.6** | 51.6% | **81.4%** | **87.8%** | **0.6430** | **91.0%** | **0.9347** |
|
| **Weighted 0.4 / 0.6** | 51.6% | **81.4%** | **87.8%** | **0.6430** | **91.0%** | **0.9347** |
|
||||||
| RRF (k=60) | 45.0% | 78.8% | 87.6% | 0.5967 | 89.6% | 0.9253 |
|
| RRF (k=60) | 45.0% | 78.8% | 87.6% | 0.5969 | 89.6% | 0.9253 |
|
||||||
|
|
||||||
**RRF loses to the tuned weighted sum** — 6.6pp of turn Hit@1 and 0.046 of MRR
|
**RRF loses to the tuned weighted sum** — 6.6pp of turn Hit@1 and 0.046 of MRR
|
||||||
— and lands almost exactly where the old `0.7/0.3` weighting did (44.2% /
|
— and lands almost exactly where the old `0.7/0.3` weighting did (44.2% /
|
||||||
@@ -1305,7 +1768,9 @@ over rank-1 precision.
|
|||||||
activation. Until now its combined score contained **no relevance term at
|
activation. Until now its combined score contained **no relevance term at
|
||||||
all** — `RerankInput` did not carry the retrieval score — so a caller that
|
all** — `RerankInput` did not carry the retrieval score — so a caller that
|
||||||
re-ranked its candidates threw the retriever's ordering away and returned them
|
re-ranked its candidates threw the retriever's ordering away and returned them
|
||||||
ordered by age. The OpenClaw backend did exactly that on every search.
|
ordered by age. `ClawhdfBackend` (the `openclaw` module) did exactly that on
|
||||||
|
every search. (OpenClaw itself never integrated clawhdf5; see
|
||||||
|
`docs/openclaw.md`.)
|
||||||
|
|
||||||
Measuring that is unambiguous. "Recency" below is the share of
|
Measuring that is unambiguous. "Recency" below is the share of
|
||||||
`knowledge-update` questions where the newest gold session outranked the stale
|
`knowledge-update` questions where the newest gold session outranked the stale
|
||||||
@@ -1320,6 +1785,12 @@ one (see `newest_gold_first`); ~45% is chance.
|
|||||||
| + re-rank, relevance-led, half-life 30 days | 51.8% | 81.0% | 87.8% | 0.6427 | 51.4% |
|
| + re-rank, relevance-led, half-life 30 days | 51.8% | 81.0% | 87.8% | 0.6427 | 51.4% |
|
||||||
| + re-rank, relevance-led, half-life 90 days | **52.0%** | 80.4% | 87.8% | 0.6425 | 50.8% |
|
| + re-rank, relevance-led, half-life 90 days | **52.0%** | 80.4% | 87.8% | 0.6425 | 50.8% |
|
||||||
|
|
||||||
|
Re-run on 2026-09-27 (`--rerank-sweep`, tank, commit 7a8fae0): every Hit@k and
|
||||||
|
MRR above reproduced exactly. The recency column came out 45.3%, 87.5%,
|
||||||
|
52.0%, 52.2%, 51.7% and 50.5%. Each of those is within one question of the
|
||||||
|
value in the table, which is the run-to-run variation described under
|
||||||
|
"Re-run with real embeddings" above, so the table was left as it was.
|
||||||
|
|
||||||
**The pre-fix row is the finding.** Ordering candidates by recency alone costs
|
**The pre-fix row is the finding.** Ordering candidates by recency alone costs
|
||||||
40.6pp of Hit@1 and two thirds of MRR: the results are the newest memories in
|
40.6pp of Hit@1 and two thirds of MRR: the results are the newest memories in
|
||||||
the pool rather than the ones that answer the question. It does ace the recency
|
the pool rather than the ones that answer the question. It does ace the recency
|
||||||
@@ -1333,9 +1804,9 @@ cannot reach the 87.5% the degenerate ordering gets. Those two rows are the
|
|||||||
ends of a trade-off, and the default sits deliberately near the relevance end.
|
ends of a trade-off, and the default sits deliberately near the relevance end.
|
||||||
|
|
||||||
**Half-life is not a sensitive knob.** Across 1, 7, 30 and 90 days recency
|
**Half-life is not a sensitive knob.** Across 1, 7, 30 and 90 days recency
|
||||||
moves 1.4pp and MRR 0.003 — inside the noise of a 500-question run — because
|
moves 1.4pp (1.7pp in the 2026-09-27 re-run) and MRR 0.003 — inside the
|
||||||
the temporal term is capped by its weight (0.3) while relevance differences
|
noise of a 500-question run — because the temporal term is capped by its
|
||||||
between candidates are larger. The 24-hour default is kept; there is no
|
weight (0.3) while relevance differences between candidates are larger. The 24-hour default is kept; there is no
|
||||||
measured reason to change it, and a corpus-matched value is not the lever it
|
measured reason to change it, and a corpus-matched value is not the lever it
|
||||||
looks like.
|
looks like.
|
||||||
|
|
||||||
@@ -1343,24 +1814,28 @@ looks like.
|
|||||||
|
|
||||||
`0.7/0.3` was a documented default, never a searched one. Sweeping
|
`0.7/0.3` was a documented default, never a searched one. Sweeping
|
||||||
`vector_weight` from 0.0 to 1.0 (`--sweep`, reusing the one-time embedding
|
`vector_weight` from 0.0 to 1.0 (`--sweep`, reusing the one-time embedding
|
||||||
table) shows it is not merely suboptimal but **strictly dominated**:
|
table) shows it is not merely suboptimal but **strictly dominated**. First
|
||||||
|
measured 2026-08-07 (1537a94); the values below are the 2026-09-27 re-run,
|
||||||
|
which changed the 0.2, 0.4, 0.5, 0.6, 0.7, 0.8 and 1.0 rows in the last digit
|
||||||
|
or by one or two questions (see the re-run section above):
|
||||||
|
|
||||||
| vector / keyword | Hit@1 | Hit@5 | Hit@10 | MRR | session Hit@5 |
|
| vector / keyword | Hit@1 | Hit@5 | Hit@10 | MRR | session Hit@5 |
|
||||||
|---|---|---|---|---|---|
|
|---|---|---|---|---|---|
|
||||||
| 0.0 / 1.0 (BM25) | **53.8%** | 75.0% | 81.6% | 0.6320 | 93.6% |
|
| 0.0 / 1.0 (BM25) | **53.8%** | 75.0% | 81.6% | 0.6320 | 93.6% |
|
||||||
| 0.1 / 0.9 | 53.2% | 77.4% | 83.8% | 0.6374 | 95.0% |
|
| 0.1 / 0.9 | 53.2% | 77.4% | 83.8% | 0.6374 | 95.0% |
|
||||||
| 0.2 / 0.8 | 53.6% | 78.2% | 85.6% | 0.6440 | 95.4% |
|
| 0.2 / 0.8 | 53.4% | 78.0% | 85.6% | 0.6429 | 95.4% |
|
||||||
| 0.3 / 0.7 | 53.2% | 78.8% | 87.2% | **0.6463** | 96.0% |
|
| 0.3 / 0.7 | 53.2% | 78.8% | 87.2% | **0.6463** | 96.0% |
|
||||||
| **0.4 / 0.6** | 51.6% | **81.4%** | 87.8% | 0.6429 | 96.8% |
|
| **0.4 / 0.6** | 51.6% | **81.4%** | 87.8% | 0.6430 | 96.8% |
|
||||||
| 0.5 / 0.5 | 48.2% | **81.4%** | **88.2%** | 0.6234 | **97.4%** |
|
| 0.5 / 0.5 | 48.2% | **81.4%** | **88.2%** | 0.6232 | **97.4%** |
|
||||||
| 0.6 / 0.4 | 46.6% | 79.8% | 87.4% | 0.6069 | 96.6% |
|
| 0.6 / 0.4 | 46.6% | 79.6% | 87.4% | 0.6067 | 96.4% |
|
||||||
| 0.7 / 0.3 *(old default)* | 44.4% | 79.2% | 86.0% | 0.5868 | 95.8% |
|
| 0.7 / 0.3 *(old default)* | 44.2% | 79.2% | 85.8% | 0.5856 | 95.8% |
|
||||||
| 0.8 / 0.2 | 40.6% | 76.2% | 85.4% | 0.5571 | 95.2% |
|
| 0.8 / 0.2 | 40.6% | 76.2% | 85.4% | 0.5574 | 95.2% |
|
||||||
| 0.9 / 0.1 | 37.8% | 73.4% | 84.6% | 0.5289 | 94.2% |
|
| 0.9 / 0.1 | 37.8% | 73.4% | 84.6% | 0.5289 | 94.2% |
|
||||||
| 1.0 / 0.0 (vector) | 36.0% | 71.8% | 81.6% | 0.5027 | 94.2% |
|
| 1.0 / 0.0 (vector) | 36.0% | 71.8% | 81.6% | 0.5031 | 94.2% |
|
||||||
|
|
||||||
**`0.4/0.6` beats `0.7/0.3` on every metric at both granularities** — Hit@1
|
**`0.4/0.6` beats `0.7/0.3` on every metric at both granularities** — Hit@1
|
||||||
+7.2pp, Hit@5 +2.2, Hit@10 +1.8, MRR +0.056. There is no trade being made; the
|
+7.4pp, Hit@5 +2.2, Hit@10 +2.0, MRR +0.057 (2026-09-27 figures; +7.2pp,
|
||||||
|
+2.2, +1.8 and +0.056 as measured on 2026-08-07). There is no trade being made; the
|
||||||
old default was simply on the wrong side of the peak. **`0.4/0.6` is the
|
old default was simply on the wrong side of the peak. **`0.4/0.6` is the
|
||||||
recommended setting**, with `0.3/0.7` preferable if rank-1 precision matters
|
recommended setting**, with `0.3/0.7` preferable if rank-1 precision matters
|
||||||
most (it takes the best MRR in the sweep and gives up only 0.6pp of Hit@1
|
most (it takes the best MRR in the sweep and gives up only 0.6pp of Hit@1
|
||||||
@@ -1393,7 +1868,7 @@ worth stating plainly rather than hiding: LongMemEval questions share substantia
|
|||||||
vocabulary with their evidence turns, which is close to the best case for lexical
|
vocabulary with their evidence turns, which is close to the best case for lexical
|
||||||
matching, and MiniLM at 384 dimensions is a small embedding model.
|
matching, and MiniLM at 384 dimensions is a small embedding model.
|
||||||
|
|
||||||
> **Run:** `cargo run --release --bin longmemeval_bench --features embeddings -- \
|
> **Run:** `cargo run --release -p clawhdf5-bench --bin longmemeval_bench --features embeddings -- \
|
||||||
> benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings weights/all-minilm-l6-v2`
|
> benchmarks/longmemeval/longmemeval_s_cleaned.json --embeddings weights/all-minilm-l6-v2`
|
||||||
> For the GPU path use `--features embeddings-cuda`. That requires `nvcc` on
|
> For the GPU path use `--features embeddings-cuda`. That requires `nvcc` on
|
||||||
> `PATH` at *build* time — cudarc's build script shells out to it. The toolkit
|
> `PATH` at *build* time — cudarc's build script shells out to it. The toolkit
|
||||||
@@ -1421,11 +1896,27 @@ price of the harder corpus, and is the reason oracle-only numbers should not be
|
|||||||
presented as LongMemEval results. Session-level figures on this variant are
|
presented as LongMemEval results. Session-level figures on this variant are
|
||||||
degenerate — see below.
|
degenerate — see below.
|
||||||
|
|
||||||
With real embeddings the same oracle corpus gives BM25-only 84.2% / vector-only
|
With real embeddings, the same oracle corpus gives these turn-level figures
|
||||||
80.4% / hybrid **85.2%** Hit@5 turn-level — hybrid ahead at Hit@5 and Hit@10 and
|
(2026-09-27, tank, commit 7a8fae0; command and load in "Re-run with real
|
||||||
behind at Hit@1, matching the full-haystack pattern above. (BM25-only reads 84.2%
|
embeddings" above):
|
||||||
here against 84.4% with zero embedding vectors: one question of 500 changes rank,
|
|
||||||
with MRR identical at 0.6597. On the full haystack the two agree exactly.)
|
| Mode | Hit@1 | Hit@5 | Hit@10 | MRR |
|
||||||
|
|---|---:|---:|---:|---:|
|
||||||
|
| BM25 only | **52.6%** | 84.4% | 90.4% | 0.6597 |
|
||||||
|
| Vector only | 39.6% | 80.4% | 91.0% | 0.5605 |
|
||||||
|
| Hybrid 0.4 / 0.6 (default) | 52.4% | **86.8%** | **92.4%** | **0.6678** |
|
||||||
|
| Hybrid 0.7 / 0.3 (old default) | 48.4% | 85.2% | 92.2% | 0.6382 |
|
||||||
|
|
||||||
|
Hybrid 0.4/0.6 leads at Hit@5, Hit@10 and MRR and is 0.2pp (one question)
|
||||||
|
behind BM25 at Hit@1, as on the full haystack.
|
||||||
|
|
||||||
|
Until 2026-09-27 this paragraph gave BM25-only 84.2%, vector-only 80.4% and
|
||||||
|
hybrid **85.2%** Hit@5. Those were measured on 2026-08-07 (c913cd1), when the
|
||||||
|
hybrid default was 0.7/0.3; today's 0.7/0.3 row reproduces the 85.2%. The
|
||||||
|
0.4/0.6 default (29baabb) is what moves hybrid to 86.8%. The earlier BM25-only
|
||||||
|
84.2% with embeddings, one question below the zero-embedding 84.4%, predates
|
||||||
|
the index tie-break of 3ed0489; the two now agree, as they always did on the
|
||||||
|
full haystack.
|
||||||
|
|
||||||
### Retracted: session-level recall and the MemX comparison
|
### Retracted: session-level recall and the MemX comparison
|
||||||
|
|
||||||
@@ -1796,7 +2287,8 @@ The tank row was measured 2026-09-24 on tank (AMD Ryzen 7 7800X3D), commit
|
|||||||
### Reproducibility
|
### Reproducibility
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
rustup override set nightly
|
# Any stable toolchain at or above the MSRV (1.92) works; the original
|
||||||
|
# 2026-07-01 run used a nightly, later runs stable.
|
||||||
|
|
||||||
# Latency benchmarks (Criterion)
|
# Latency benchmarks (Criterion)
|
||||||
cargo bench -p clawhdf5-agent
|
cargo bench -p clawhdf5-agent
|
||||||
@@ -1899,7 +2391,9 @@ libhdf5 reads from a temp file including `open` + `read` + `close` overhead.
|
|||||||
| clawhdf5 hyperslab (f64, 10% slice) | — | 4.09 µs / **1.8 GiB/s** | 50.1 µs / **1.5 GiB/s** |
|
| clawhdf5 hyperslab (f64, 10% slice) | — | 4.09 µs / **1.8 GiB/s** | 50.1 µs / **1.5 GiB/s** |
|
||||||
|
|
||||||
libhdf5 f64 comparison excluded — clawhdf5's datatype encoding differs from libhdf5's (known
|
libhdf5 f64 comparison excluded — clawhdf5's datatype encoding differs from libhdf5's (known
|
||||||
gap), making cross-format reads unreliable for comparison.
|
gap), making cross-format reads unreliable for comparison. (That gap was the float sign-bit
|
||||||
|
bug, fixed 2026-09-23: `docs/known-issues.md`, "Every `f32` dataset we wrote was unreadable by
|
||||||
|
h5py / libhdf5". The comparison has not been re-run since.)
|
||||||
|
|
||||||
### Chunked Read Throughput
|
### Chunked Read Throughput
|
||||||
|
|
||||||
@@ -2007,6 +2501,13 @@ global file mutex and flushes to disk on every attribute write or group creation
|
|||||||
|
|
||||||
## vs libhdf5 Summary
|
## vs libhdf5 Summary
|
||||||
|
|
||||||
|
> Measured on the original i7-12650H (clawhdf5 2026-07-01, libhdf5
|
||||||
|
> 2026-06-30). The newest run of this table is
|
||||||
|
> [Independent Validation: tank](#independent-validation-tank-ryzen-7-7800x3d-2026-08-03)
|
||||||
|
> (2026-08-03), which reproduces every row within ~15% except chunked
|
||||||
|
> write (45.3x on tank; 35x on 2026-09-23 with the pure-Rust deflate, see
|
||||||
|
> [Deflate backend](#deflate-backend-zlib-rs-vs-zlib-ng)).
|
||||||
|
|
||||||
| Workload | clawhdf5 | libhdf5 | Speedup |
|
| Workload | clawhdf5 | libhdf5 | Speedup |
|
||||||
|----------|----------|---------|---------|
|
|----------|----------|---------|---------|
|
||||||
| Sequential read, 1K f32 | 634 ns | 45.2 µs | **71×** |
|
| Sequential read, 1K f32 | 634 ns | 45.2 µs | **71×** |
|
||||||
@@ -2038,7 +2539,7 @@ to the page cache. There is no algorithmic headroom above ~1.7 GiB/s on this har
|
|||||||
|
|
||||||
### Caveats
|
### Caveats
|
||||||
|
|
||||||
- libhdf5 f64 read comparison excluded — clawhdf5's f32 datatype encoding differs from libhdf5's (known compatibility gap). f64 results are clawhdf5-only.
|
- libhdf5 f64 read comparison excluded — clawhdf5's f32 datatype encoding differs from libhdf5's (known compatibility gap at the time; fixed 2026-09-23, see [Sequential Read Throughput](#sequential-read-throughput)). f64 results are clawhdf5-only.
|
||||||
- Serial benchmarks. clawhdf5 uses Rayon for chunk compression when > 2 chunks; that parallelism is already reflected in the chunked write numbers.
|
- Serial benchmarks. clawhdf5 uses Rayon for chunk compression when > 2 chunks; that parallelism is already reflected in the chunked write numbers.
|
||||||
- clawhdf5 reads from `Vec<u8>` (zero-copy from mmap in production); libhdf5 reads from a temp file. This gives clawhdf5 a structural read advantage that reflects realistic API usage.
|
- clawhdf5 reads from `Vec<u8>` (zero-copy from mmap in production); libhdf5 reads from a temp file. This gives clawhdf5 a structural read advantage that reflects realistic API usage.
|
||||||
|
|
||||||
@@ -2235,6 +2736,10 @@ Same not-like-for-like caveat as the "Comparison to MemX" section at the top of
|
|||||||
file applies — MemX's figure is end-to-end, these are a single component. Ratios are
|
file applies — MemX's figure is end-to-end, these are a single component. Ratios are
|
||||||
an order-of-magnitude indication, not a benchmark result.
|
an order-of-magnitude indication, not a benchmark result.
|
||||||
|
|
||||||
|
> The Ratio column below was retracted afterwards: see
|
||||||
|
> [Comparison to MemX](#comparison-to-memx-arxiv260316171). Kept as recorded
|
||||||
|
> on 2026-08-05; do not cite it.
|
||||||
|
|
||||||
| Metric | MemX (claimed, end-to-end) | ClawhDF5 (tank, component only) | Ratio |
|
| Metric | MemX (claimed, end-to-end) | ClawhDF5 (tank, component only) | Ratio |
|
||||||
|--------|----------------------------|----------------------------------|-------|
|
|--------|----------------------------|----------------------------------|-------|
|
||||||
| 100K flat search | <90 ms | 6.60 ms | ~14x |
|
| 100K flat search | <90 ms | 6.60 ms | ~14x |
|
||||||
|
|||||||
+2246
File diff suppressed because it is too large
Load Diff
@@ -1,198 +1,267 @@
|
|||||||
# clawhdf5
|
# clawhdf5
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated I/O. A standalone library. Its one verified consumer is ClawBrainHub (`.brain` files); no agent framework integrates it (OpenClaw and ZeroClaw claims were withdrawn on 2026-09-25 — neither was ever true).
|
Pure-Rust HDF5 implementation (read, write, in-place edit, remote and browser
|
||||||
|
reads) plus agent memory on top of it: HNSW vector search, a WAL-backed store,
|
||||||
|
and GPU vector distances. A standalone library. Its one verified consumer is
|
||||||
|
ClawBrainHub (`.brain` files); no agent framework integrates it (see
|
||||||
|
*Standing rules*).
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal FFI bindings crate for the optional `szip` feature):
|
Cargo workspace, 19 crates under `crates/` (plus `libaec-sys`, the FFI crate
|
||||||
|
behind the optional `szip` feature). MSRV 1.92 (`rust-version`, checked in CI).
|
||||||
|
|
||||||
| Crate | Role |
|
| Crate | Role |
|
||||||
|-------|------|
|
|-------|------|
|
||||||
| `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants |
|
| `clawhdf5-format` | The HDF5 format: parsers and writer (superblock, headers, B-trees, heaps, chunk indexes), the `Storage` trait, the filter pipeline and registry (`filter_registry`), every codec except deflate (LZ4, Zstd, SZIP, N-Bit, scale-offset, pcodec; pure-Rust LZF, bitshuffle, bzip2, Blosc 1; Blosc2 and ZFP read-only), `float16`, `checksum` |
|
||||||
| `clawhdf5-io` | Read/write implementation |
|
| `clawhdf5-filters` | Deflate backends (zlib-rs default, zlib-ng, Apple Compression) |
|
||||||
| `clawhdf5-filters` | Compression filters (gzip, LZ4, Zstd, Blosc) |
|
| `clawhdf5-io` | I/O adapters (buffers, mmap, prefetch) |
|
||||||
| `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs |
|
| `clawhdf5-derive` | `#[derive(H5Type)]` for compound types |
|
||||||
| `clawhdf5` | Main facade crate |
|
| `clawhdf5` | Facade: `File`, `FileBuilder`, `Dataset`, `FileEditor` (`src/edit/`), SWMR reading (`src/swmr.rs`) |
|
||||||
| `clawhdf5-netcdf4` | NetCDF-4 compatibility layer |
|
| `clawhdf5-netcdf4` | NetCDF-4 read support |
|
||||||
| `clawhdf5-ann` | HNSW approximate nearest-neighbor vector index |
|
| `clawhdf5-remote` | `open_url`: HTTP(S) range requests and object stores (S3, GCS, Azure) through `BlockCache` |
|
||||||
| `clawhdf5-agent` | Agent memory, session history, knowledge graph storage |
|
| `clawhdf5-tools` | `h5rs`: `ls`, `dump` (DDL / hdf5-json), `stat`, `diff`, `check` |
|
||||||
| `clawhdf5-gpu` | GPU-accelerated I/O via wgpu (hand-written WGSL compute shaders) |
|
| `clawhdf5-py` | PyO3 bindings (h5py-like API, remote files, `'r+'` editing) |
|
||||||
| `clawhdf5-accel` | CPU SIMD acceleration path |
|
| `clawhdf5-wasm` | wasm-bindgen browser reader (`open(bytes)`, `openUrl(url)`); demo in `examples/wasm-viewer/` |
|
||||||
| `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration |
|
| `clawhdf5-ann` | HNSW index |
|
||||||
|
| `clawhdf5-agent` | Agent memory store (`HDF5Memory`), sessions, knowledge graph, BM25 |
|
||||||
|
| `clawhdf5-accel` | CPU SIMD kernels (AVX2, NEON) |
|
||||||
|
| `clawhdf5-gpu` | wgpu vector distances (WGSL) — not dataset I/O; HDF5 I/O is CPU-only |
|
||||||
|
| `clawhdf5-migrate` | SQLite → agent store migration |
|
||||||
|
| `clawhdf5-cli` | Agent-memory CLI |
|
||||||
|
| `clawhdf5-napi` | Node.js addon (the `packages/clawhdf5-node` wrapper is broken; `docs/known-issues.md`) |
|
||||||
| `clawhdf5-android` | Android JNI bindings |
|
| `clawhdf5-android` | Android JNI bindings |
|
||||||
| `clawhdf5-cli` | Command-line interface |
|
| `clawhdf5-bench` | Benchmarks and harnesses (`search_harness`, `read_harness`, `concurrent_read`, `longmemeval_bench`, …) |
|
||||||
| `clawhdf5-napi` | Node.js native addon bindings |
|
|
||||||
| `clawhdf5-py` | PyO3 Python bindings |
|
|
||||||
| `clawhdf5-bench` | Benchmark suite |
|
|
||||||
|
|
||||||
## Key Features
|
Reference docs: `docs/known-issues.md` (open issues table first — check it
|
||||||
- Zero-C-dependency HDF5 read/write: no libhdf5, and deflate defaults to
|
before calling something a bug or a feature), `BENCHMARKS.md` (headline
|
||||||
pure-Rust zlib-rs (`fast-deflate` opts into zlib-ng, which needs cmake).
|
numbers first), `CONFORMANCE.md` (generated), `docs/design/range-reads.md`
|
||||||
`ci-test.sh` fails if a C-building crate enters the core crates' default
|
and `docs/design/swmr.md`, `CHANGELOG.md` (full detail of every fix).
|
||||||
tree. flate2 must keep `runtime_detection` with zlib-rs — without it zlib-rs
|
|
||||||
loses SIMD and inflates 3.5x slower. MSRV is 1.92 (`rust-version`, checked
|
## Standing rules
|
||||||
in CI).
|
|
||||||
- HNSW vector index for semantic similarity search over agent memories — the
|
- **No C in the default build.** No libhdf5; deflate defaults to pure-Rust
|
||||||
`clawhdf5-agent` `hnsw` feature is **on by default**, so `hybrid_search` uses
|
zlib-rs (`fast-deflate` opts into zlib-ng, which needs cmake). `ci-test.sh`
|
||||||
the approximate `clawhdf5-ann` index for the vector stage (the index mirrors
|
fails if a C-building crate enters the core crates' default tree. Zstd,
|
||||||
the cache and self-heals on drift). Build the agent with
|
SZIP, `https` (ring) and `s3`/`gcs`/`azure` (aws-lc-rs) are opt-in. flate2
|
||||||
`--no-default-features --features float16` to force the exact linear cosine scan.
|
must keep `runtime_detection` with zlib-rs — without it zlib-rs loses SIMD
|
||||||
The agent's `parallel` feature (also default) builds the index on a thread
|
and inflates 3.5x slower.
|
||||||
pool; the graph is identical with or without it.
|
- **Every file we write must open in h5py/libhdf5.** Interop tests compare
|
||||||
The index uses the HNSW paper's diversity heuristic for neighbour selection
|
against h5py and h5dump; `f32` and empty datasets did not open until
|
||||||
(plain closest-M capped recall on clustered data: 0.31 recall@10 at 100K). Its
|
2026-09-23.
|
||||||
graph is saved to `<store>.h5.ann` at each checkpoint and reloaded by `open()`
|
- **float16 has one implementation:** `clawhdf5_format::float16`.
|
||||||
(tied to the checkpoint by a generation id; stale/damaged sidecars are
|
- **Claims need evidence.** Performance and integration claims in docs must
|
||||||
ignored and the index rebuilt). `MemoryConfig::quantized_index` (**on by
|
be measured, dated (with machine and command), or withdrawn. Benchmark
|
||||||
default** for new stores, persisted; stores predating the setting load as
|
numbers are dated records: never edit a measured value, add a new dated
|
||||||
`false` and keep their f32 index — guarded by
|
section and mark the old one superseded.
|
||||||
`tests/fixtures/store_v2_5_0.h5`; CLI opt-out is `create --f32-index`)
|
- **OpenClaw is not supported** (decided 2026-09-25): clawhdf5 is not and
|
||||||
stores the index's own copy of the embeddings as `i8`,
|
never was an OpenClaw memory plugin; the old `memory.backend = "clawhdf5"`
|
||||||
which roughly halves a loaded store's memory (2.72x -> 1.74x the raw vectors
|
config was never valid. `docs/openclaw.md` records what a real plugin would
|
||||||
at 100K); because quantised distances are approximate and `ef` cannot
|
need. The `openclaw` module's `ClawhdfBackend` is just `search` with
|
||||||
compensate, the query path then re-scores the candidate pool against the
|
re-rank + confidence on.
|
||||||
exact embeddings, which holds recall at the f32 index's level. It is also
|
- **ZeroClaw does not use clawhdf5** (checked 2026-09-25 against upstream
|
||||||
faster at equal recall: 1.63x the QPS on x86-64 (AVX2) and 1.18x on a
|
v0.8.5 and the `osobh/zeroclaw` fork and their history): its memory
|
||||||
Raspberry Pi 5 (`clawhdf5_accel::dot_i8`, NEON `SDOT` via inline asm since
|
backends are its own; `clawhdf5-migrate`'s default SQLite layout is not
|
||||||
the intrinsic is unstable; plain NEON on pre-dotprod cores). The aarch64
|
ZeroClaw's schema. Don't reintroduce integration claims without an
|
||||||
code is `cfg`'d out on x86, so x86 CI never compiles or lints it — test it
|
integration and a test against the real consumer.
|
||||||
on real ARM (`rpivision02`, 10.0.2.3, is a Pi 5). `hybrid_search` keeps one incremental BM25
|
- **known-issues.md:** one entry per bug; when fixed, record it in
|
||||||
index for the life of the store and never writes the store: Hebbian
|
`CHANGELOG.md` and move the entry to *Fixed (history)* with date, PR,
|
||||||
activation boosts are persisted by the next checkpoint (or on drop), not per
|
affected releases and what users must do — never delete it.
|
||||||
query. Measure any search-path change with
|
|
||||||
`cargo run --release -p clawhdf5-bench --bin search_harness` (baselines in
|
## HDF5 library: invariants and gotchas
|
||||||
`BENCHMARKS.md`).
|
|
||||||
- WAL (write-ahead log) for crash-safe persistence, with a chained CRC32
|
- **Remote/range reads** (`docs/design/range-reads.md`, M0-M5 merged in PRs
|
||||||
trailer per entry (each entry's CRC folds in the previous entry's CRC) so a
|
#17-#19, M4 listing costs cut in #21): every format-crate read path goes through `Storage`
|
||||||
corrupted, reordered, duplicated, or spliced entry stops replay cleanly
|
(`read_at`/`read_ranges`/`hint`). `File::open_storage` takes any
|
||||||
instead of loading bad or tampered data. The pre-chaining per-entry-CRC
|
`Storage`; `clawhdf5_remote::open_url` wraps HTTP (`HttpStorage`, ureq) or
|
||||||
format (v2) is still fully readable; the oldest no-CRC format (v1) is only
|
`ObjectStoreStorage` in `BlockCache` (1 MiB blocks, LRU budget, in-flight
|
||||||
reachable through the one-time migration path in `HDF5Memory::open`, not
|
dedup, coalesced runs). Remote files are pinned by ETag/Last-Modified and
|
||||||
through the public `WalFile::read_entries`.
|
length (`RemoteError::FileChanged`). Zero-copy APIs and `File::as_bytes`
|
||||||
**What the WAL guarantees:** integrity, ordering, and recovery from a
|
need an in-memory file. Parse through `File::storage()` and the `*_in`
|
||||||
*process* crash at any point — including between a checkpoint and the WAL
|
functions, not `as_bytes`, in new code (the Python bindings do).
|
||||||
truncate (each checkpoint records a `WalMark` in `/meta`, and `open()` skips
|
`ObjectStoreStorage` runs reads on its own small tokio runtime, so it
|
||||||
the WAL prefix the `.h5` already contains, so entries are never applied
|
works from any thread.
|
||||||
twice). Checkpoints and snapshots are made durable as a unit (temp file
|
- **SWMR** (`docs/design/swmr.md`): `File::open_swmr` reads a file a libhdf5
|
||||||
synced, renamed, directory synced). **What it does not guarantee:**
|
SWMR writer is appending to — positioned reads, no chunk cache, bounded
|
||||||
individual WAL appends are *not* fsynced (a deliberate latency trade-off), so
|
retries (100), `Dataset::refresh()`. clawhdf5 has no SWMR writer; remote
|
||||||
saves made since the last checkpoint can be lost on power failure or kernel
|
SWMR is out of scope.
|
||||||
panic. Current header version is 4 (adds the `Update` record used by
|
- **Browser** (`clawhdf5-wasm`, read-only, no Zstd/SZIP): `openUrl` reads
|
||||||
`save_or_update`); v3 files are read and upgraded in place.
|
through the restartable "NeedBytes" cache (`src/lazy.rs`: a call is re-run
|
||||||
- A store has a **single writer**: `HDF5Memory::create`/`open` hold an exclusive
|
after each wave of misses; no block is evicted while a call runs); the HTTP
|
||||||
advisory lock on `<store>.h5.lock` and a second opener gets
|
is JavaScript (`js/remote.js`).
|
||||||
`MemoryError::Locked`. Use `HDF5Memory::open_read_only` for a lock-free,
|
- **In-place editing** (`clawhdf5::FileEditor`): overwrites values, grows and
|
||||||
never-writing point-in-time view (the CLI's `recall`/`stats`/`agents-md`/
|
shrinks chunked datasets (every chunk index) and sets attributes (compact
|
||||||
`export` do). An unreadable WAL (torn header, bad magic) is quarantined to
|
and dense) without rewriting the file, changing indexes and heaps as
|
||||||
`<store>.h5.wal.corrupt-<ts>` rather than blocking `open()`; a WAL with an
|
libhdf5 does; freed space is reused within one editor. Anything it cannot
|
||||||
unknown *newer* version still fails and is left untouched.
|
do safely is `Error::Unsupported` before any write (limits in
|
||||||
- `MemoryConfig::float16` (**on by default** for new stores, persisted;
|
`docs/known-issues.md`). The algorithms follow libhdf5 `hdf5_1_14_6`
|
||||||
existing stores keep their recorded `false` — guarded by the v2.5.0
|
(github.com/HDFGroup/hdf5). Test changes with `cargo test -p
|
||||||
fixture in `tests/float16_store.rs`; CLI opt-out is `create --f32`) writes
|
clawhdf5-tools --test edit_interop --test edit_coverage_interop`.
|
||||||
`/memory/embeddings` as IEEE half precision (48% smaller file at 100K;
|
- **Provenance:** `Dataset::verify_provenance()` (facade `provenance`
|
||||||
LongMemEval with real MiniLM embeddings identical to f32).
|
feature, default) re-hashes a dataset against its `_provenance_sha256`
|
||||||
`MemoryCache::half_precision` rounds each embedding as it enters the cache (push, update, WAL replay, and on load of a store still
|
attribute (`DatasetBuilder::with_provenance`). Opt-in per call; unkeyed
|
||||||
`f32` on disk), so memory and file agree bit for bit; the conversions live
|
hash — tamper-evident, not tamper-proof.
|
||||||
in `clawhdf5_format::float16` and must stay the single implementation.
|
|
||||||
Values beyond ±65504 are `MemoryError::InvalidEntry`. Interop: every file
|
## Agent memory: invariants and gotchas
|
||||||
must open in h5py — `f32` datasets and empty datasets did not until
|
|
||||||
2026-09-23 (see `docs/known-issues.md`); the agent's `h5py_interop` test
|
- **Search.** `HDF5Memory::search(query_emb, text, &SearchOptions)` is the
|
||||||
guards a whole store.
|
full path: optional source-channel filter (before ranking; exact scan of
|
||||||
- `HDF5Memory::search(query_emb, text, &SearchOptions)` is the full search
|
the allowed records when cheaper than `pool × M` index distance
|
||||||
path: optional source-channel filter (applied before ranking; exact scan of
|
|
||||||
the allowed records whenever cheaper than `pool × M` index distance
|
|
||||||
evaluations, and as the fallback when the pool comes back short), fusion,
|
evaluations, and as the fallback when the pool comes back short), fusion,
|
||||||
activation scaling, optional re-ranking and confidence rejection.
|
activation scaling, optional re-ranking and confidence rejection.
|
||||||
`hybrid_search`/`hybrid_search_with` are thin wrappers; `ClawhdfBackend`
|
`hybrid_search`/`hybrid_search_with` are thin wrappers. It keeps one
|
||||||
(the `openclaw` module) is `search` with re-rank + confidence on.
|
incremental BM25 index for the life of the store and never writes the
|
||||||
- **OpenClaw is not supported** (decided 2026-09-25): clawhdf5 is not an
|
store: Hebbian activation boosts are persisted by the next checkpoint (or
|
||||||
OpenClaw memory plugin and never was — the old `memory.backend = "clawhdf5"`
|
on drop).
|
||||||
config was never valid. Don't reintroduce OpenClaw claims; `docs/openclaw.md`
|
- **HNSW** (`hnsw` feature, default): the approximate `clawhdf5-ann` index
|
||||||
records what a real plugin would need.
|
mirrors the cache and self-heals on drift; build the agent with
|
||||||
- **ZeroClaw does not use clawhdf5** (checked 2026-09-25 against upstream
|
`--no-default-features --features float16` for the exact linear scan.
|
||||||
v0.8.5 and the `osobh/zeroclaw` fork, and their full history): no
|
`parallel` (default) builds it on a thread pool with an identical graph.
|
||||||
`clawhdf5` feature or backend exists; ZeroClaw's memory backends are
|
Neighbour selection uses the HNSW paper's diversity heuristic (closest-M
|
||||||
sqlite/lucid/postgres/qdrant/markdown/none behind its own `Memory` trait.
|
capped recall at 0.31 recall@10 at 100K on clustered data). The graph is
|
||||||
`clawhdf5-migrate`'s default SQLite layout (`memory_chunks`, `sessions`,
|
saved to `<store>.h5.ann` at each checkpoint, tied to it by a generation
|
||||||
`entities`, `relations`) is not ZeroClaw's schema either (ZeroClaw's is a
|
id; a stale or damaged sidecar is ignored and the index rebuilt.
|
||||||
`memories` table). Don't reintroduce integration claims without an
|
- **`MemoryConfig::quantized_index`** (default on for new stores, persisted;
|
||||||
integration and a test against the real consumer. Measure changes with
|
older stores load as `false` — guarded by `tests/fixtures/store_v2_5_0.h5`;
|
||||||
`search_harness --options-study`.
|
CLI `create --f32-index`): the index's copy of the embeddings is `i8`, and
|
||||||
- `MemoryConfig::compression` is off by default; when on, embeddings are
|
the query path re-scores candidates against the exact embeddings. The
|
||||||
deflate-compressed, or Zstd with the agent's `zstd` feature (links libzstd).
|
aarch64 kernels (`clawhdf5_accel::dot_i8`, NEON `SDOT` via inline asm) are
|
||||||
- Signed checkpoints (`clawhdf5-agent` `signing` module): with
|
`cfg`'d out on x86, so x86 CI never compiles them — test on real ARM
|
||||||
`HDF5Memory::set_signing_key` every checkpoint stores an Ed25519-signed
|
(`rpivision02`, 10.0.2.3, a Pi 5) or rely on the `test-arm64` job.
|
||||||
manifest (SHA-256 per record in a Merkle tree + settings/sessions/graph
|
- **`MemoryConfig::float16`** (default on for new stores, persisted; older
|
||||||
hashes; per-record hashes in `/integrity/record_hashes`);
|
stores keep `false` — guarded in `tests/float16_store.rs`; CLI `create
|
||||||
`HDF5Memory::verify(path, &pk)` locates edits. The hashes must cover exactly
|
--f32`): `/memory/embeddings` is IEEE half. `MemoryCache::half_precision`
|
||||||
what the file persists in the form the loader returns it (strings lose
|
rounds each embedding as it enters the cache (push, update, WAL replay, and
|
||||||
trailing NULs; an empty WAL mark is not written) or untouched stores stop
|
load of a store still `f32` on disk) so memory and file agree bit for bit.
|
||||||
verifying — `tests/signed_store.rs` round-trips awkward strings. The key is
|
Values beyond ±65504 are `MemoryError::InvalidEntry`. The agent's
|
||||||
never persisted; a signed store refuses to checkpoint without it
|
`h5py_interop` test guards that a whole store opens in h5py.
|
||||||
(`MemoryError::SigningKeyRequired`, and `MemoryError` is `#[non_exhaustive]`).
|
- **WAL.** Chained CRC32 per entry (a corrupted, reordered, duplicated or
|
||||||
|
spliced entry stops replay cleanly). Header version 4 (`Update` record for
|
||||||
|
`save_or_update`); v3 is upgraded in place, v2 read, v1 only through the
|
||||||
|
one-time migration in `HDF5Memory::open`. Each checkpoint records a
|
||||||
|
`WalMark` in `/meta` so `open()` never applies an entry twice; checkpoints
|
||||||
|
and snapshots are durable as a unit (temp file synced, renamed, directory
|
||||||
|
synced). Individual WAL appends are **not** fsynced (deliberate): saves
|
||||||
|
since the last checkpoint can be lost on power failure or kernel panic.
|
||||||
|
- **Single writer.** `create`/`open` hold an exclusive lock on
|
||||||
|
`<store>.h5.lock` (`MemoryError::Locked` for a second opener);
|
||||||
|
`open_read_only` is a lock-free point-in-time view (CLI `recall`/`stats`/
|
||||||
|
`agents-md`/`export`). An unreadable WAL is quarantined to
|
||||||
|
`<store>.h5.wal.corrupt-<ts>`; a WAL of an unknown newer version fails and
|
||||||
|
is left untouched.
|
||||||
|
- **Signed checkpoints** (`signing` module): with `set_signing_key` each
|
||||||
|
checkpoint stores an Ed25519-signed manifest (per-record SHA-256 in a
|
||||||
|
Merkle tree plus settings/sessions/graph hashes; `/integrity/record_hashes`);
|
||||||
|
`HDF5Memory::verify(path, &pk)` locates edits. The hashes must cover
|
||||||
|
exactly what the file persists in the form the loader returns it (strings
|
||||||
|
lose trailing NULs; an empty WAL mark is not written) —
|
||||||
|
`tests/signed_store.rs` round-trips awkward strings. The key is never
|
||||||
|
persisted; a signed store refuses to checkpoint without it
|
||||||
|
(`MemoryError::SigningKeyRequired`; `MemoryError` is `#[non_exhaustive]`).
|
||||||
WAL entries after the checkpoint are not covered.
|
WAL entries after the checkpoint are not covered.
|
||||||
- `Dataset::verify_provenance()` (clawhdf5 facade, `provenance` feature, on by
|
- **Write bookkeeping.** `save`/`save_batch`/`save_or_update` feed an
|
||||||
default) recomputes a dataset's SHA-256 and compares it against the
|
in-memory, session-scoped provenance ledger and anomaly detector
|
||||||
`_provenance_sha256` attribute written automatically on save when
|
(`provenance.rs`, `anomaly.rs`); alerts never block a save
|
||||||
`DatasetBuilder::with_provenance` is used. It's opt-in per call, not run
|
(`take_anomaly_alerts`). `MemorySource` is inferred from the caller's
|
||||||
automatically on open — it decodes and hashes the whole dataset. The hash
|
`source_channel` string — a heuristic, not a trust boundary.
|
||||||
is unkeyed (tamper-*evident*, not tamper-*proof*): it detects accidental
|
- `MemoryConfig::compression` is off by default (deflate, or Zstd with the
|
||||||
corruption, not a deliberate actor able to modify both the data and the
|
agent's `zstd` feature, which links libzstd).
|
||||||
stored hash.
|
|
||||||
- `clawhdf5-agent`'s `HDF5Memory::save`/`save_batch`/`save_or_update` run every
|
|
||||||
write through an in-memory (session-scoped, not persisted to disk)
|
|
||||||
provenance ledger and write-anomaly detector: a content hash per record
|
|
||||||
(`provenance.rs`) for detecting accidental mid-session corruption, plus
|
|
||||||
rate-limit/injection-pattern/source-distribution checks (`anomaly.rs`).
|
|
||||||
Alerts never block a save — drain them with `HDF5Memory::take_anomaly_alerts`.
|
|
||||||
`MemorySource` for this bookkeeping is inferred from the caller-supplied
|
|
||||||
`source_channel` string (a heuristic, not an authenticated trust boundary).
|
|
||||||
- GPU-accelerated batch I/O for large dataset processing
|
|
||||||
- Python and Node.js bindings for cross-language use
|
|
||||||
- NetCDF-4 compatibility for scientific data interop
|
|
||||||
|
|
||||||
## Workflows
|
## Workflows
|
||||||
|
|
||||||
### Build
|
Put `$HOME/.cargo/bin` on `PATH`. The h5py/netCDF4 interop tests find their
|
||||||
|
Python through `CLAWHDF5_PYTHON` (or `.venv/bin/python`); create it with
|
||||||
|
`python3 -m venv .venv && .venv/bin/pip install h5py numpy netCDF4 hdf5plugin`.
|
||||||
|
Set `CLAWHDF5_REQUIRE_INTEROP=1` to make a missing interpreter a failure.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
cargo build --release
|
cargo build --release
|
||||||
```
|
|
||||||
|
|
||||||
### Test
|
|
||||||
```bash
|
|
||||||
cargo test --workspace
|
cargo test --workspace
|
||||||
|
bash scripts/ci-test.sh # everything CI runs (see below)
|
||||||
```
|
```
|
||||||
|
|
||||||
### CI
|
### CI (`.gitea/workflows/`)
|
||||||
`.gitea/workflows/ci.yml` has two jobs, both green as of 2026-09-22:
|
- **`ci.yml` `test`** (`ubuntu-latest`, `rust:latest` container; runners
|
||||||
- **`test`** (`ubuntu-latest`, in `rust:latest`) runs `scripts/ci-test.sh` with
|
`tank`, `architect`): installs h5py/netCDF4/xarray/hdf5plugin/maturin/pytest,
|
||||||
the h5py/netCDF4 interop suites required (`CLAWHDF5_REQUIRE_INTEROP=1`).
|
`hdf5-tools` and `cmake`, then runs `scripts/ci-test.sh` with
|
||||||
Served by the `tank` and `architect` runners.
|
`CLAWHDF5_REQUIRE_INTEROP=1`. The script runs: fmt; clippy (workspace, the
|
||||||
- **`test-arm64`** (`linux_arm64`) lints and tests the aarch64 code — the NEON
|
format feature matrix, each plugin filter alone, parallel, fast-deflate,
|
||||||
kernels are `cfg`'d out on x86, so this is the only place they are built.
|
remote with all backends, h5rs remote); "no C in the default build";
|
||||||
Served by `vision-01` (host mode) and `vision-02` (Docker), so steps must
|
wasm32 build and clippy; `check-32bit-casts.sh`; the wasm package under
|
||||||
work in both.
|
Node when `node` and `wasm-bindgen` exist (not in CI); the MSRV check;
|
||||||
|
`cargo test` (workspace plus feature variants: format matrix, parallel,
|
||||||
|
remote/object_store, h5rs URLs, ann parallel, fast-deflate); the h5py
|
||||||
|
interop suites (`writer_h5py_tests --include-ignored`, plugin filters,
|
||||||
|
ZFP); the Python package (clippy, `maturin build`, pytest vs h5py);
|
||||||
|
`cargo bench --no-run`; `check-nostd.sh`; an optional fuzz smoke run
|
||||||
|
(`CLAWHDF5_FUZZ_SECONDS`).
|
||||||
|
- **`ci.yml` `test-arm64`** (`linux_arm64`; `vision-01` host mode,
|
||||||
|
`vision-02` Docker — steps must work in both): clippy of
|
||||||
|
`clawhdf5-accel`, tests of `-accel`, `-ann`, `-format`; the only place the NEON kernels build.
|
||||||
|
- **`conformance.yml`** (nightly 03:17 UTC and manual): probe unit tests,
|
||||||
|
`conformance/test_ref.py`, then `conformance/run.sh` (gate:
|
||||||
|
`conformance/check.py` against `baseline.json`).
|
||||||
|
|
||||||
Keep workflows free of JavaScript actions (`actions/checkout`, `actions/cache`,
|
Keep workflows free of JavaScript actions (`actions/checkout`,
|
||||||
…): `rust:latest` has no `node`, and not every runner reaches GitHub, where
|
`actions/cache`, …): `rust:latest` has no `node` and not every runner reaches
|
||||||
they are fetched from. Check out with plain `git` instead. The `test` job
|
GitHub. Check out with plain `git`. Runners are `gitea-runner` 3.5.0 from
|
||||||
installs `cmake` for the opt-in `fast-deflate` (zlib-ng) steps; the default
|
`docker.gitea.com/act_runner` (`gitea/act_runner:latest` on Docker Hub is
|
||||||
build needs no C toolchain, so `test-arm64` does not.
|
frozen at 0.6.1).
|
||||||
All runners are on `gitea-runner` 3.5.0, from `docker.gitea.com/act_runner`
|
|
||||||
— `gitea/act_runner:latest` on Docker Hub is frozen at 0.6.1.
|
### Conformance
|
||||||
|
```bash
|
||||||
|
CLAWHDF5_PYTHON=.venv/bin/python bash conformance/run.sh --no-fetch # writes CONFORMANCE.md
|
||||||
|
```
|
||||||
|
Reads 697 files of eight pinned corpora with clawhdf5 and h5py and compares
|
||||||
|
them object by object (602 ok in the run of 2026-09-28). `CONFORMANCE.md` is
|
||||||
|
generated — never hand-edit it (its wording lives in `conformance/report.py`).
|
||||||
|
Use `--update-baseline` only after an intended change in results.
|
||||||
|
`CONFORMANCE_CACHE` points at an existing corpus cache (`conformance/.cache`,
|
||||||
|
about 450 MB). See `conformance/README.md`.
|
||||||
|
|
||||||
|
### HDF5 tools (`h5rs`)
|
||||||
|
```bash
|
||||||
|
cargo run -p clawhdf5-tools -- ls -r file.h5 # also dump [--json], stat, diff, check
|
||||||
|
bash scripts/h5rs-fuzz.sh # every subcommand over the CVE corpus: no panic/crash/hang
|
||||||
|
bash scripts/h5rs-check-ok-files.sh --data # check passes every fully-read conformance file
|
||||||
|
```
|
||||||
|
Interop tests compare against h5ls/h5stat/h5dump/h5diff (Debian `hdf5-tools`);
|
||||||
|
`dump` must stay byte-identical to h5dump on the test files.
|
||||||
|
|
||||||
|
### Remote and browser tests
|
||||||
|
- `clawhdf5-remote` tests run a std-only HTTP server
|
||||||
|
(`tests/common/server.rs`, also the `range_server` example);
|
||||||
|
`CLAWHDF5_REMOTE_CORPUS=conformance/.cache/corpus` compares every corpus
|
||||||
|
file over HTTP with `File::open`.
|
||||||
|
- wasm: `bash examples/wasm-viewer/test/run.sh` builds the package (needs the
|
||||||
|
`wasm-bindgen` CLI at the crate's exact version) and tests it under Node and
|
||||||
|
headless Chromium (Playwright's download in `~/.cache/ms-playwright` on
|
||||||
|
tank) against `test/serve.py` (range server with request counts). CI has
|
||||||
|
neither, so it runs the native `h5py_interop` and `lazy` tests
|
||||||
|
(`CLAWHDF5_WASM_CORPUS=conformance/.cache/corpus` for the corpus).
|
||||||
|
|
||||||
|
### Python bindings
|
||||||
|
```bash
|
||||||
|
cd crates/clawhdf5-py && maturin develop
|
||||||
|
python -m pytest crates/clawhdf5-py/tests # compares with h5py; editing tests want CLAWHDF5_H5RS=<path to h5rs>
|
||||||
|
```
|
||||||
|
|
||||||
|
### Benchmarks
|
||||||
|
- Search path: `cargo run --release -p clawhdf5-bench --bin search_harness`
|
||||||
|
(`--full`, `--options-study`, `--footprint`, …); reads: `read_harness`,
|
||||||
|
`concurrent_read`; criterion benches with `cargo bench -p <crate>`.
|
||||||
|
- Run on an idle machine (1-minute load average below 2; wait otherwise),
|
||||||
|
alternate base and candidate binaries for A/B comparisons, and record date,
|
||||||
|
machine, commit and command with every number in `BENCHMARKS.md`.
|
||||||
|
- `BENCHMARKS.md` is written by hand from dated runs; no script regenerates
|
||||||
|
it (the old `scripts/run-benchmarks.sh`, which benchmarked the pre-rename
|
||||||
|
`rustyhdf5-format` and overwrote the file, was removed on 2026-09-28).
|
||||||
|
|
||||||
### CLI
|
### CLI
|
||||||
```bash
|
```bash
|
||||||
cargo run -p clawhdf5-cli -- --help
|
cargo run -p clawhdf5-cli -- --help
|
||||||
# create, save, search, recall, stats, flush-wal, agents-md, export, snapshot subcommands
|
# create, save, search, recall, stats, flush-wal, agents-md, export, snapshot, keygen, verify
|
||||||
```
|
|
||||||
|
|
||||||
### Python bindings
|
|
||||||
```bash
|
|
||||||
cd crates/clawhdf5-py
|
|
||||||
maturin develop
|
|
||||||
python -c "import clawhdf5; print(clawhdf5.__version__)"
|
|
||||||
```
|
```
|
||||||
|
|
||||||
## Integration
|
## Integration
|
||||||
@@ -200,8 +269,7 @@ python -c "import clawhdf5; print(clawhdf5.__version__)"
|
|||||||
verified consumer: `cbh-core` reads and writes `.brain` files through the
|
verified consumer: `cbh-core` reads and writes `.brain` files through the
|
||||||
facade (`File`, `FileBuilder`, `AttrValue`, `Selection`), `cbh-scanner`
|
facade (`File`, `FileBuilder`, `AttrValue`, `Selection`), `cbh-scanner`
|
||||||
uses the facade, and `cbh-cli` uses `clawhdf5_agent::bm25::BM25Index`. It
|
uses the facade, and `cbh-cli` uses `clawhdf5_agent::bm25::BM25Index`. It
|
||||||
depends on this repo by path (`../clawhdf5`), so it builds against whatever
|
depends on this repo by path (`../clawhdf5`), so changes to those APIs
|
||||||
is checked out — changes to those APIs reach it directly. Verified
|
reach it directly. Verified 2026-09-25 against main: builds, and its 204
|
||||||
2026-09-25 against main: builds, and its 204 tests pass.
|
tests pass.
|
||||||
- OpenClaw and ZeroClaw were both described as consumers; neither integrates
|
- OpenClaw and ZeroClaw integrate nothing (see *Standing rules*).
|
||||||
clawhdf5 (see Key Features and `docs/openclaw.md`).
|
|
||||||
|
|||||||
+305
@@ -0,0 +1,305 @@
|
|||||||
|
# clawhdf5 conformance report
|
||||||
|
|
||||||
|
Every HDF5 file of eight public corpora (pinned by commit) is read twice — by
|
||||||
|
clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade
|
||||||
|
makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are
|
||||||
|
compared object by object: the set of hard-linked objects, each dataset's and
|
||||||
|
attribute's shape, and a SHA-256 of its values in a canonical encoding. The
|
||||||
|
CVE corpus is also run through `h5dump`. Each side runs under a timeout and an
|
||||||
|
address-space limit, so a hang, crash or runaway allocation is recorded, not
|
||||||
|
fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.
|
||||||
|
|
||||||
|
## Run
|
||||||
|
|
||||||
|
| | |
|
||||||
|
|---|---|
|
||||||
|
| date | 2026-09-28 04:29 UTC |
|
||||||
|
| clawhdf5 commit | `bf5a163dcf7fe28d651ada6545d8136ffeffc825` |
|
||||||
|
| machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 |
|
||||||
|
| command | `conformance/run.sh --no-fetch --update-baseline` |
|
||||||
|
| rustc | rustc 1.98.1 (48a229cea 2026-09-01) |
|
||||||
|
| reference | h5py 3.16.0, HDF5 2.0.0, numpy 2.5.3, hdf5plugin 7.1.0, Python 3.14.4 |
|
||||||
|
| h5dump | Version 1.14.6 (CVE corpus only) |
|
||||||
|
| limits | 20 s timeout (SIGKILL), 4096 MiB address space, per process; 16 files in parallel |
|
||||||
|
| runtime | 20 s probing + comparing (5 s fetch/build before it) |
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
A file's class is the first that applies:
|
||||||
|
|
||||||
|
- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.
|
||||||
|
- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.
|
||||||
|
- **ref-bug** — every difference is an object clawhdf5 refuses that h5py reads only through a libhdf5 bug: the values h5py returns for it change with the reading process's heap, re-checked in every run (see *Reference bugs*).
|
||||||
|
- **our-error** — clawhdf5 returned an error for something h5py reads.
|
||||||
|
- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.
|
||||||
|
- **ok** — every object h5py reads, clawhdf5 reads identically.
|
||||||
|
|
||||||
|
| corpus | files | ok | our-error | mismatch | h5py-cannot-read | ref-bug | panic | hang | crash | oom |
|
||||||
|
|---|---|---|---|---|---|---|---|---|---|---|
|
||||||
|
| NCAS-CMS_pyfive | 33 | 33 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| cve_hdf5 | 147 | 113 | 0 | 0 | 32 | 2 | 0 | 0 | 0 | 0 |
|
||||||
|
| h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| hdf5 | 466 | 405 | 1 | 0 | 60 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| **all** | **697** | **602** | **1** | **0** | **92** | **2** | **0** | **0** | **0** | **0** |
|
||||||
|
|
||||||
|
**Our errors and mismatches: 1.** Files not ok: 1 our-error, 92 h5py-cannot-read, 2 ref-bug. 2 object(s) were compared against h5py's values corrected for a known h5py bug (2 identical to clawhdf5's; see *Reference bugs*).
|
||||||
|
|
||||||
|
Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):
|
||||||
|
|
||||||
|
| corpus | source | commit |
|
||||||
|
|---|---|---|
|
||||||
|
| hdf5 | https://github.com/HDFGroup/hdf5 | `a3cf1ea82cc7` |
|
||||||
|
| cve_hdf5 | https://github.com/HDFGroup/cve_hdf5 | `3fd1f5ae3869` |
|
||||||
|
| netcdf-c | https://github.com/Unidata/netcdf-c | `beb7b9585273` |
|
||||||
|
| NCAS-CMS_pyfive | https://github.com/NCAS-CMS/pyfive | `8cf07b874913` |
|
||||||
|
| usnistgov_h5wasm | https://github.com/usnistgov/h5wasm | `02f6336527d2` |
|
||||||
|
| netcdf4-python | https://github.com/Unidata/netcdf4-python | `6e67576d39ae` |
|
||||||
|
| xarray-data | https://github.com/pydata/xarray-data | `a35297e9da2c` |
|
||||||
|
| h5py_data | https://github.com/h5py/h5py (`h5py/tests/data_files`) | `b2f0347c4200` |
|
||||||
|
|
||||||
|
## Panics, hangs, crashes, out-of-memory
|
||||||
|
|
||||||
|
None.
|
||||||
|
|
||||||
|
## Our-error root causes
|
||||||
|
|
||||||
|
Grouped by normalised error message. *files* counts files whose class this cause affects.
|
||||||
|
|
||||||
|
| files | objects | error | examples |
|
||||||
|
|---:|---:|---|---|
|
||||||
|
| 1 | 1 | `ChunkedReadError("…")` | `hdf5/test/testfiles/bad_nbit_parms_walk.h5` |
|
||||||
|
|
||||||
|
## Mismatch root causes
|
||||||
|
|
||||||
|
None.
|
||||||
|
|
||||||
|
## CVE corpus: clawhdf5 vs h5dump vs h5py
|
||||||
|
|
||||||
|
The 147 files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for
|
||||||
|
published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object
|
||||||
|
errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so
|
||||||
|
its read/error split is not comparable with the other two rows; the panic, crash, hang and oom
|
||||||
|
columns are.
|
||||||
|
|
||||||
|
| tool | read | error | panic | crash | hang | oom |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
| clawhdf5 | 121 | 26 | 0 | 0 | 0 | 0 |
|
||||||
|
| h5dump 1.14.6 | 16 | 129 | 0 | 2 | 0 | 0 |
|
||||||
|
| h5py 3.16.0 / HDF5 2.0.0 | 115 | 31 | 0 | 1 | 0 | 0 |
|
||||||
|
|
||||||
|
<details><summary>Per-file outcomes</summary>
|
||||||
|
|
||||||
|
| file | h5dump | h5py | clawhdf5 | class |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| cvefiles/cve-2016-4330.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4331.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-mtime-new.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-mtime.h5 | error exit | read 4 obj, 3 errors | read 4 obj, 3 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-stab.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2016-4333.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17505.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17506.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17507.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17508.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17509.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11202.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11203.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11204.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11205.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11206-new.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11206-old.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11207.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13866.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-13867.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13868.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13869.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13870.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13871.h5 | error exit | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2018-13872.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13873.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13874.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-13875.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13876.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-14031.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14033.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14034.h5 | error exit | read 1 obj, 2 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-14035.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14460.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2018-15671.h5 | ok | read 1 obj | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-15672.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-16438.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-17233.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17234.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17237.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17432.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17433 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-17434.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17435.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17436 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-17437.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17438 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17439 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8396.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8397.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8398.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2019-9151.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2019-9152.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2020-10809 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-10810.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-10811.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2020-10812.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-18232.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2020-18494.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2021-36977.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-37501.h5 | error exit | read 18 obj, 1 errors | read 18 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-45829.h5 | error exit | read 1 obj, 2 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-45830.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2021-45833.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-46242.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2021-46243.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-46244.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29157.h5 | error exit | read 4 obj, 7 errors | read 4 obj, 7 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29158.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29159.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29160.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29161.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29162.h5 | error exit | read 17 obj, 4 errors | read 17 obj, 4 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29163.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29164.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-29165.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29166.h5 | error exit | read 17 obj, 2 errors | read 17 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32605.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32606.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32607-1.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32607-2.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32608.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32609.h5 | error exit | SIGSEGV | read 3 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2024-32610.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32611.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32612.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32613.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32614.h5 | error exit | read 25 obj, 2 errors | read 25 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32615.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32616.h5 | error exit | read 10 obj, 7 errors | read 10 obj, 6 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32617.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32618.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32619.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32620.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32621.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32622.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32623.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32624.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33873.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33874.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33875.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2024-33876.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33877.h5 | error exit | read 8 obj, 1 errors | read 8 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2153.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2308.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | ref-bug |
|
||||||
|
| cvefiles/cve-2025-2309.h5 | ok | read 6 obj, 1 errors | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2025-2310.h5 | error exit | read 24 obj, 8 errors | read 24 obj, 8 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2912.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2913.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2914.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2915.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2923.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2924.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2925.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2926.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-44904.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | ref-bug |
|
||||||
|
| cvefiles/cve-2025-44905.h5 | error exit | read 25 obj, 3 errors | read 25 obj, 3 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-1.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-2.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-3.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-4.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6270-1.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6270-2.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6270-3.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6516.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6750.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6816.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6817.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6818.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6856.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6857.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6858.h5 | SIGSEGV | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-7067.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-7068.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-7069.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2026-26200.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2026-34734.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2026-92627.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/unknown-1.h5 | error exit | read 11 obj, 1 errors | read 11 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4431-poc-03.h5 | error exit | read 1 obj | read 1 obj | ok |
|
||||||
|
| fuzzerfiles/gh-4432-poc-05.h5 | SIGSEGV | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4433-poc-08.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4434-poc-09.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| fuzzerfiles/gh-4435-poc-10.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4585.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| fuzzerfiles/gh_2649_flawed.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh_2649_plain_model.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
## Reference bugs
|
||||||
|
|
||||||
|
### Objects h5py reads only through a libhdf5 bug (*ref-bug*)
|
||||||
|
|
||||||
|
clawhdf5 refuses these objects; h5py 3.16 / HDF5 2.0 returns values for them. `conformance/ref_bugs.py`
|
||||||
|
re-reads each with h5py in six fresh processes whose heaps differ (h5py imported before numpy, three
|
||||||
|
times and twice more with `MALLOC_PERTURB_`, and numpy imported first). Values the file determines
|
||||||
|
come out the same every time; these do not, so they are memory libhdf5 over-reads, not the file's
|
||||||
|
data. A file is *ref-bug* only while every one of its differences is such an object confirmed in
|
||||||
|
the same run; an object that reads the same every time goes back to *our-error*. Reproducer:
|
||||||
|
`python conformance/ref_bugs.py conformance/.cache/corpus` (prints every read's outcome).
|
||||||
|
|
||||||
|
| file | object | distinct results in 6 reads | confirmed | what goes wrong |
|
||||||
|
|---|---|---:|---|---|
|
||||||
|
| `cve_hdf5/cvefiles/cve-2025-2308.h5` | `/Scale_offset_long_long_data_le` | 6 | yes | the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the 26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads past its buffer, and develop refuses the chunk ("Buffer too short") |
|
||||||
|
| `cve_hdf5/cvefiles/cve-2025-44904.h5` | `/Scale_offset_float_data_le` | 6 | yes | unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the stored bytes into a buffer of that size and use it as the whole chunk (H5D__chunk_lock), so the rest is heap memory; develop refuses them ("incorrect chunk size returned from index for unfiltered chunk") |
|
||||||
|
| `hdf5/test/testfiles/bad_nbit_parms_walk.h5` | `/Nbit_int_data_le` | 1 | **no** | the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test (`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail |
|
||||||
|
|
||||||
|
### Values corrected for a known h5py bug
|
||||||
|
|
||||||
|
- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence
|
||||||
|
whose base type is big-endian with the file's big-endian bytes but a native (little-endian)
|
||||||
|
numpy dtype: a `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` reads back as
|
||||||
|
`[4.6e-41, 9.0e-44]`; `h5dump` prints the file's values. `ref.py` checks that the installed
|
||||||
|
h5py still does this (by writing and reading exactly that dataset in memory) and, if so,
|
||||||
|
relabels such elements with the file's byte order before hashing, so the values are still
|
||||||
|
compared. Corrected objects: `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` `/@vlen_uint64` (same as clawhdf5), `hdf5/tools/test/testfiles/tcomplex_be.h5` `/VariableLengthDatasetFloatComplex` (same as clawhdf5).
|
||||||
|
|
||||||
|
## Other comparison rules
|
||||||
|
|
||||||
|
- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose
|
||||||
|
bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a
|
||||||
|
bit offset / reduced precision into the plain numpy type of the same size. The probe
|
||||||
|
compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared
|
||||||
|
raw bytes, which reported every N-Bit float dataset as a mismatch).
|
||||||
|
- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size
|
||||||
|
(FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not
|
||||||
|
compared (shape and presence still are): dataset file type size 1 -> numpy float16 (2) (15x), attr file type size 1 -> numpy float16 (2) (15x), dataset file type size 2 -> numpy float32 (4) (2x), dataset file type size 8 -> numpy float128 (16) (1x), dataset file type size 12 -> numpy float128 (16) (1x), attr file type size 2 -> numpy float32 (4) (1x), dataset file type size 2 -> numpy >f4 (4) (1x), attr file type size 2 -> numpy >f4 (4) (1x).
|
||||||
|
- **References** are compared by presence only (`R`), not by target.
|
||||||
|
|
||||||
|
## Objects h5py fails on but clawhdf5 reads
|
||||||
|
|
||||||
|
- 19 x `OSError: Can't synchronously read data (no appropriate function for conversion path)`
|
||||||
|
- 1 x `TypeError: unhandled dtype kind M (dtype('…'))`
|
||||||
|
- 1 x `TypeError: No NumPy equivalent for TypeTimeID exists`
|
||||||
|
- 1 x `ValueError: Insufficient precision in available types to represent (N, N, N, N, N)`
|
||||||
|
|
||||||
|
## Reproduce
|
||||||
|
|
||||||
|
```sh
|
||||||
|
# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git
|
||||||
|
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for
|
||||||
|
every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.
|
||||||
|
`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)
|
||||||
|
must keep; `conformance/run.sh --update-baseline` rewrites it.
|
||||||
+12
@@ -16,6 +16,9 @@ members = [
|
|||||||
"crates/clawhdf5-cli",
|
"crates/clawhdf5-cli",
|
||||||
"crates/clawhdf5-napi",
|
"crates/clawhdf5-napi",
|
||||||
"crates/clawhdf5-bench",
|
"crates/clawhdf5-bench",
|
||||||
|
"crates/clawhdf5-tools",
|
||||||
|
"crates/clawhdf5-wasm",
|
||||||
|
"crates/clawhdf5-remote",
|
||||||
"crates/libaec-sys",
|
"crates/libaec-sys",
|
||||||
]
|
]
|
||||||
resolver = "2"
|
resolver = "2"
|
||||||
@@ -34,3 +37,12 @@ tempfile = "3"
|
|||||||
criterion = { version = "0.5", features = ["html_reports"] }
|
criterion = { version = "0.5", features = ["html_reports"] }
|
||||||
half = "2.7"
|
half = "2.7"
|
||||||
serde = { version = "1", features = ["derive"] }
|
serde = { version = "1", features = ["derive"] }
|
||||||
|
|
||||||
|
# The browser build of clawhdf5-wasm (examples/wasm-viewer/build.sh): size
|
||||||
|
# over speed, whole-program optimisation. Native profiles are unaffected.
|
||||||
|
[profile.wasm-release]
|
||||||
|
inherits = "release"
|
||||||
|
opt-level = "s"
|
||||||
|
lto = true
|
||||||
|
codegen-units = 1
|
||||||
|
panic = "abort"
|
||||||
|
|||||||
+144
-177
@@ -1,193 +1,160 @@
|
|||||||
# ClawhDF5 Roadmap — Agent Memory Evolution
|
# clawhdf5 roadmap
|
||||||
|
|
||||||
> Making clawhdf5 the defacto agentic memory solution.
|
What has shipped, and what is genuinely next. Everything here is checked
|
||||||
> Single file. Pure Rust. Zero dependencies. Trusted everywhere.
|
against `CHANGELOG.md`, `git log` and [`docs/known-issues.md`](docs/known-issues.md);
|
||||||
|
dates are merge dates on `main`. Nothing after v2.7.0 has been released:
|
||||||
|
the work since then is on `main` under `CHANGELOG.md` "Unreleased".
|
||||||
|
|
||||||
|
_Last updated: 2026-09-28 (at `9b5803f`, PR #21)._
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Track 1: Knowledge Graph in HDF5
|
## Done
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** Critical
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **1.1** Entity storage — entities with properties, embeddings, timestamps (created_at/updated_at)
|
### Releases
|
||||||
- [x] **1.2** Relation storage — typed edges with RelationType enum (Temporal/Causal/Associative/Hierarchical/Custom), metadata, timestamps
|
|
||||||
- [x] **1.3** Entity extraction helpers — rule-based extraction (Person, Org, Location, Date, Technology, Project) with extract_and_store_entities() integration
|
|
||||||
- [x] **1.4** Entity resolution — fuzzy name matching (Levenshtein distance) via resolve_or_create()
|
|
||||||
- [x] **1.5** Graph traversal queries — BFS neighbors with depth, subgraph extraction from seeds
|
|
||||||
- [x] **1.6** Spreading activation — weighted activation propagation with configurable decay
|
|
||||||
- [x] **1.7** Graph-aware retrieval — get_entity_context() for formatted context injection
|
|
||||||
- [x] **1.8** Tests — comprehensive tests for all new features
|
|
||||||
|
|
||||||
**Research:** Graph-Native Cognitive Memory (2026), Graph-based Agent Memory survey (2026), SYNAPSE (2025)
|
| Version | Date | Headline |
|
||||||
|
|---|---|---|
|
||||||
|
| v2.0.0 | 2026-03-19 | rustyhdf5 (11 crates) and edgehdf5 (4 crates) unified into one workspace as `clawhdf5-*` |
|
||||||
|
| v2.1.0 | 2026-06-03 | HNSW backs the agent's vector search by default; live, mutable HNSW index |
|
||||||
|
| v2.2.0 – v2.7.0 | 2026-09-18 – 2026-09-20 | bounded decompression and read-path bounds checks, single-writer store locking, WAL v4, HNSW recall fix (0.31 -> 0.98 recall@10 at 100K), fusion weights tuned on LongMemEval, int8 index, Extensible Array read fix and chunk-index checksums |
|
||||||
|
|
||||||
|
Details per release: [`CHANGELOG.md`](CHANGELOG.md).
|
||||||
|
|
||||||
|
### Since v2.7.0 (unreleased, on `main`)
|
||||||
|
|
||||||
|
| PR | Merged | What |
|
||||||
|
|---|---|---|
|
||||||
|
| #3 | 2026-09-23 | pure-Rust deflate (zlib-rs) by default, no C in the core crates' default build (checked in CI), MSRV 1.92 |
|
||||||
|
| #4 | 2026-09-25 | files open in h5py again (every `f32` and every empty dataset clawhdf5 wrote was unreadable by libhdf5); float16 embedding storage |
|
||||||
|
| #5 | 2026-09-25 | `HDF5Memory::search` with `SearchOptions` (source filters, re-ranking, confidence); float16 on by default |
|
||||||
|
| #6 | 2026-09-25 | `clawhdf5-migrate` writes real agent stores; knowledge-graph fix; dated benchmark re-run |
|
||||||
|
| #7 | 2026-09-25 | consolidation benchmark completed (cheaper novelty scoring) |
|
||||||
|
| #8 | 2026-09-25 | Ed25519-signed checkpoints (`HDF5Memory::verify`) |
|
||||||
|
| #9, #10 | 2026-09-25 | OpenClaw and ZeroClaw integration claims withdrawn — neither ever integrated clawhdf5 |
|
||||||
|
| #11 | 2026-09-26 | silent wrong data and libhdf5 interop bugs found by the HDF5 audit fixed |
|
||||||
|
| #12 | 2026-09-26 | reproducible conformance sweep over eight public corpora, nightly CI job ([`CONFORMANCE.md`](CONFORMANCE.md)) |
|
||||||
|
| #13 | 2026-09-26 | reads HDF5 1.6-era layouts, user blocks, virtual datasets, dense attributes, very large groups |
|
||||||
|
| #14 | 2026-09-26 | `h5rs` tools (`ls`, `dump`, `stat`, `diff`, `check`), the browser reader (`clawhdf5-wasm`), libhdf5's header checks, plugin filters (LZF, bitshuffle, bzip2, Blosc), concurrency benchmark |
|
||||||
|
| #15 | 2026-09-26 | fast contiguous and concurrent reads, variable-length data, nested groups and links in the writer, Python bindings |
|
||||||
|
| #16 | 2026-09-26 | chunked full reads faster than an h5py process pool, writer B-trees of any size, Blosc2 (read), 599/697 conformance |
|
||||||
|
| #17 | 2026-09-26 | range reads M0/M1 (indexed name lookups, the `Storage` trait), ZFP (read), in-place editing (`FileEditor`) |
|
||||||
|
| #18 | 2026-09-27 | range reads M2/M3 (`File::open_storage`; `clawhdf5-remote`: HTTP(S), S3, GCS, Azure), in-place editing of every chunk index, shrinking, dense attributes |
|
||||||
|
| #19 | 2026-09-27 | remote files in the browser (`openUrl`, M4), SWMR reader (`File::open_swmr`, M5), Python remote reads and `'r+'` editing |
|
||||||
|
| #20 | 2026-09-28 | benchmarks re-measured: LongMemEval with real MiniLM embeddings, local reads on an idle machine |
|
||||||
|
| #21 | 2026-09-28 | remote files open in a few requests (group lookups down the B-tree, `Storage::hint`), `ObjectHeader::parse` back to its earlier speed, the last conformance mismatches resolved: 602/697 ok, 0 mismatch (the run of 2026-09-28 in [`CONFORMANCE.md`](CONFORMANCE.md) still counts 1 our-error, a corrupt N-Bit file libhdf5's own tests refuse) |
|
||||||
|
|
||||||
|
### Range reads (design: [`docs/design/range-reads.md`](docs/design/range-reads.md))
|
||||||
|
|
||||||
|
- [x] M0 — indexed name lookups (#17)
|
||||||
|
- [x] M1 — metadata parsed through the `Storage` trait (#17)
|
||||||
|
- [x] M2 — raw data through `Storage`, `File::open_storage` (#18)
|
||||||
|
- [x] M3 — `clawhdf5-remote`: HTTP(S) range requests and object stores through a block cache; `h5rs` URLs (#18); Python URLs (#19)
|
||||||
|
- [x] M4 — `openUrl` in the browser, restartable "NeedBytes" cache (#19; fewer round trips in #21)
|
||||||
|
- [x] M5 — reading files a SWMR writer is appending to ([`docs/design/swmr.md`](docs/design/swmr.md), #19)
|
||||||
|
|
||||||
|
### Agent memory (`clawhdf5-agent`)
|
||||||
|
|
||||||
|
Shipped before and during the v2 releases, and kept current since:
|
||||||
|
knowledge graph with entity extraction and resolution; three-tier
|
||||||
|
consolidation with decay; hybrid retrieval (HNSW + BM25, weighted or RRF
|
||||||
|
fusion, re-ranking, confidence rejection, query expansion); temporal index
|
||||||
|
and session DAG; per-save provenance ledger and write-anomaly detection;
|
||||||
|
multi-modal embeddings; WAL with chained CRC32; single-writer locking;
|
||||||
|
signed checkpoints. Retrieval is measured, not claimed: see
|
||||||
|
[`BENCHMARKS.md`](BENCHMARKS.md) ("LongMemEval Results" reports retrieval
|
||||||
|
recall, not QA accuracy; earlier headline numbers that compared different
|
||||||
|
granularities were retracted there).
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Track 2: Memory Consolidation Engine
|
## Next
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** Critical
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **2.1** Importance scoring — surprise (novelty), correction boost, length scoring with configurable weights
|
Not scheduled; listed roughly by how much they unblock. None has a date.
|
||||||
- [x] **2.2** Three-tier memory model — Working → Episodic → Semantic with bounded capacities
|
|
||||||
- [x] **2.3** Time-decay with reactivation — exponential decay with configurable half-life, access resets timestamp
|
|
||||||
- [x] **2.4** Bounded memory with graceful degradation — evict lowest-decay entries when over capacity
|
|
||||||
- [x] **2.5** Consolidation cycles — promote/evict across tiers based on importance and access thresholds
|
|
||||||
- [x] **2.6** Memory statistics — ConsolidationStats with per-tier counts, eviction/promotion tracking
|
|
||||||
- [x] **2.7** Tests — comprehensive tests for all features
|
|
||||||
|
|
||||||
**Research:** CraniMem (2026), D-MEM (2026), AI Hippocampus survey (2026)
|
### Distribution
|
||||||
|
|
||||||
|
- [ ] **Publish the crates to crates.io.** Nothing is published; the READMEs
|
||||||
|
say to depend on git. Before publishing: no `publish` settings exist
|
||||||
|
(only `clawhdf5-wasm` has `publish = false`).
|
||||||
|
- [ ] **Publish Python wheels to PyPI.** `crates/clawhdf5-py` builds with
|
||||||
|
maturin and is tested in CI, but no wheel is published. The default wheel
|
||||||
|
reads plain `http://` only; `https`/`s3`/`gcs`/`azure` wheels compile C
|
||||||
|
(ring, aws-lc-rs).
|
||||||
|
- [ ] **The Node.js package** (`packages/clawhdf5-node` over
|
||||||
|
`clawhdf5-napi`) has never worked and is not in CI: fix it and add CI, or
|
||||||
|
remove it ([known issue](docs/known-issues.md)).
|
||||||
|
|
||||||
|
### HDF5 features
|
||||||
|
|
||||||
|
- [ ] **SWMR writing.** The reader is done (M5); writing a file while
|
||||||
|
libhdf5 readers follow it is not. Also not covered: remote SWMR (a remote
|
||||||
|
file is pinned at open), `MmapFile`/`LazyFile` SWMR reads, refreshing
|
||||||
|
groups or attributes.
|
||||||
|
- [ ] **MPI collective I/O.** `clawhdf5-io`'s `MpiVol` (`mpi-io`) is
|
||||||
|
root-read + broadcast and gather-to-root writes, not collective MPI-IO
|
||||||
|
(`MPI_File_read_at_all`/`write_at_all`).
|
||||||
|
- [ ] **Paged-metadata single-request reads.** Files written with paged
|
||||||
|
aggregation (`H5Pset_file_space_strategy(PAGE)`, `h5repack -S PAGE`)
|
||||||
|
keep their metadata in a few pages; range reads could fetch those in one
|
||||||
|
request and use the file's page size as the block size. Today the block
|
||||||
|
size is fixed (1 MiB) and only the first block is read ahead
|
||||||
|
(range-reads design, option (c) as a policy).
|
||||||
|
- [ ] **Blosc2 and ZFP encoders.** Both filters are read-only; the other
|
||||||
|
plugin filters (LZF, bitshuffle, bzip2, Blosc 1) read and write.
|
||||||
|
- [ ] **External links and external raw data** are explicit errors, not
|
||||||
|
followed.
|
||||||
|
- [ ] **Virtual datasets:** the "first missing" view and printf gaps other
|
||||||
|
than 0, source-to-virtual type conversion other than a byte swap, nested
|
||||||
|
virtual sources, source files outside the virtual file's directory.
|
||||||
|
- [ ] **Datatypes:** x87 long double and binary128 are refused.
|
||||||
|
- [ ] **Writer:** one attribute or link message over 65 515 bytes in dense
|
||||||
|
storage is an error (huge fractal-heap objects); no option to write
|
||||||
|
files HDF5 1.8 can read.
|
||||||
|
- [ ] **`FileEditor`:** new chunks in implicit indexes, variable-length and
|
||||||
|
reference data, filters it cannot encode (scale-offset, N-Bit, SZIP),
|
||||||
|
some dense-attribute heap layouts, creating or deleting objects and
|
||||||
|
attributes (also from Python `'r+'`), and no journal (a crash mid-edit
|
||||||
|
can leave the file inconsistent). Freed space is reused only within one
|
||||||
|
editor.
|
||||||
|
- [ ] **Selection reads** decode the whole dataset when the selection's
|
||||||
|
bounding box covers more than half of it (a strided `ds[::100]`), and
|
||||||
|
for compact/virtual datasets or a non-default fill value: correct, but
|
||||||
|
more work than needed.
|
||||||
|
- [ ] **Readers:** `LazyFile` and `MmapFile` still need the whole file;
|
||||||
|
the zero-copy methods need the file in memory.
|
||||||
|
|
||||||
|
### Remote and browser
|
||||||
|
|
||||||
|
- [ ] Run the `s3`/`gcs`/`azure` backends against real buckets (only built
|
||||||
|
and URL-parsing-tested so far).
|
||||||
|
- [ ] `h5rs` options for request headers and cache settings.
|
||||||
|
- [ ] Browser limits in [`docs/known-issues.md`](docs/known-issues.md)
|
||||||
|
("`clawhdf5-wasm` (browser) limits"): files of 4 GiB or more (wasm32),
|
||||||
|
compound/reference/opaque datasets, round trips per index level. The
|
||||||
|
package doubled in size with `openUrl`
|
||||||
|
([size table](examples/wasm-viewer/README.md#size)); dropping the
|
||||||
|
function-name section would take a third off the raw size (13% gzipped).
|
||||||
|
|
||||||
|
### Quality
|
||||||
|
|
||||||
|
- [ ] Scheduled fuzz campaigns: the cargo-fuzz targets
|
||||||
|
([`crates/clawhdf5-format/fuzz`](crates/clawhdf5-format/fuzz/README.md),
|
||||||
|
and the agent's WAL target) run only by hand or with
|
||||||
|
`CLAWHDF5_FUZZ_SECONDS`.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
## Track 3: Hybrid Retrieval Pipeline
|
## Withdrawn
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** High
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **3.1** Reciprocal Rank Fusion (RRF) — rrf_hybrid_search() with k=60 constant
|
- **OpenClaw integration** (withdrawn 2026-09-25, PR #9). clawhdf5 was
|
||||||
- [x] **3.2** Multi-factor re-ranking — temporal decay, source authority hierarchy, activation scores (reranker.rs)
|
never an OpenClaw memory plugin; the documented
|
||||||
- [x] **3.3** Low-confidence rejection — min_score threshold, gap filtering, max_results (confidence.rs)
|
`memory.backend = "clawhdf5"` was never valid. The Rust `ClawhdfBackend`
|
||||||
- [x] **3.4** Query expansion — synonyms, acronyms, temporal rewrites, morphological variants, knowledge graph aliases + expanded_search() with RRF merge
|
remains as a library API. [`docs/openclaw.md`](docs/openclaw.md) records
|
||||||
- [x] **3.5** Result explanation — ReRankResult with full score breakdown per factor
|
what a real plugin would need.
|
||||||
- [x] **3.6** Configurable pipeline — ReRankConfig + ConfidenceConfig with tunable weights/thresholds
|
- **ZeroClaw integration** (withdrawn 2026-09-25, PR #10). ZeroClaw has no
|
||||||
- [x] **3.7** Tests + MemX-comparable benchmarks — 5 integration tests (Hit@1≥90%, search<500ms@100K, BM25<200ms@100K, hybrid<50ms@10K, compact<200ms@10K)
|
clawhdf5 backend, and `clawhdf5-migrate`'s SQLite layout is not
|
||||||
|
ZeroClaw's schema.
|
||||||
|
|
||||||
**Research:** MemX (2026), SwiftMem (2026)
|
The old track-by-track tracker this file used to be (agent-memory
|
||||||
|
Tracks 1–8, mid-2026) is in git history (`git log -- ROADMAP.md`).
|
||||||
---
|
|
||||||
|
|
||||||
## Track 4: Temporal Reasoning
|
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** High
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **4.1** Temporal index — sorted timestamp index with binary search, insert/remove
|
|
||||||
- [x] **4.2** Time-range queries — range_query, before, after, latest, earliest
|
|
||||||
- [x] **4.3** Session DAG — parent/child linking, chain walking, time-range overlap queries
|
|
||||||
- [x] **4.4** Temporal re-ranking — query hint enum (Latest/Earliest/Around/Between/None) with boost scoring
|
|
||||||
- [x] **4.5** Temporal entity tracking — EntityTimeline with state change history + point-in-time reconstruction
|
|
||||||
- [x] **4.6** Tests — comprehensive tests for all features
|
|
||||||
|
|
||||||
**Research:** MemX temporal gaps (≤43.6% Hit@5), MemoryArena multi-session tasks (2026)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Track 5: Memory Security & Provenance
|
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** Medium-High
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **5.1** Source attribution — MemoryProvenance with source, creator, session, FNV-1a content hash
|
|
||||||
- [x] **5.2** Write anomaly detection — rate limiting, 15 injection patterns, source distribution analysis
|
|
||||||
- [x] **5.3** Source isolation — per-MemorySource sub-stores preventing cross-contamination
|
|
||||||
- [x] **5.4** Memory integrity verification — content hash comparison via verify_integrity()
|
|
||||||
- [x] **5.5** Poisoning resistance — pattern detection for prompt injection attempts
|
|
||||||
- [x] **5.6** Tests — comprehensive tests including adversarial patterns
|
|
||||||
|
|
||||||
**Research:** MemoryGraft (2025), SSGM Framework (2026)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Track 6: Multi-Modal Memory
|
|
||||||
**Status:** 🟢 Phase 1 Complete
|
|
||||||
**Priority:** Medium
|
|
||||||
**Crate:** `clawhdf5-agent`
|
|
||||||
|
|
||||||
- [x] **6.1** Image embedding storage — ModalEmbedding with model provenance (CLIP, SigLIP, etc.)
|
|
||||||
- [x] **6.2** Audio fingerprints — Audio modality with embedding storage
|
|
||||||
- [x] **6.3** Multi-modal search — search_by_modality (filtered) + search_cross_modal (all embeddings)
|
|
||||||
- [x] **6.4** Observation records — raw perception vs interpretation with confidence scoring
|
|
||||||
- [x] **6.5** Media reference storage — MediaRef with Path/Url/Inline, MIME types, FNV-1a checksums
|
|
||||||
- [x] **6.6** Tests — 35 comprehensive tests
|
|
||||||
|
|
||||||
**Research:** Neuro-Symbolic Memory (2026), RAGdb multi-modal RAG (2025)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Track 7: OpenClaw Integration — withdrawn (2026-09-25)
|
|
||||||
**Status:** ⚪ Withdrawn (the items below were library work; no OpenClaw integration shipped)
|
|
||||||
**Priority:** Critical (for adoption)
|
|
||||||
**Crates:** `clawhdf5-agent`, `clawhdf5-napi`
|
|
||||||
|
|
||||||
- [x] **7.1** Memory backend trait — MemoryBackend with search/get/write/ingest/export/stats
|
|
||||||
- [x] **7.2** Hybrid retrieval pipeline — ClawhdfBackend wires RRF → reranker → confidence rejection
|
|
||||||
- [x] **7.3** Markdown import/export — MarkdownParser + MarkdownExporter with line tracking + metadata
|
|
||||||
- [x] **7.4** `search()` — backed by the full hybrid retrieval pipeline (a Rust method; no OpenClaw tool was ever registered)
|
|
||||||
- [x] **7.5** `get()` — read back by path, with a line slice (not an OpenClaw tool either)
|
|
||||||
- [x] **7.6** Compaction integration — run_compaction() (decay + compact + WAL flush), run_consolidation() (hippocampal engine), tick_session(), flush_wal()
|
|
||||||
- [ ] **7.7** ~~Config surface — `memory.backend = "clawhdf5"`~~ — never valid OpenClaw config; docs removed
|
|
||||||
- [ ] **7.8** ~~Documentation + migration guide~~ — removed: they described an integration that never worked
|
|
||||||
|
|
||||||
**Node.js bridge:** `clawhdf5-napi` (napi-rs) and a TypeScript wrapper in `packages/clawhdf5-node` exist but are unpublished, untested in CI and known to be broken (docs/known-issues.md).
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
> **Withdrawn.** None of this track produced a working OpenClaw integration: no
|
|
||||||
> plugin was built, the documented `memory.backend = "clawhdf5"` config was never
|
|
||||||
> valid in any OpenClaw release, and the Node package was never published. The
|
|
||||||
> Rust `ClawhdfBackend` remains as a library API. Not pursued for now; see
|
|
||||||
> [docs/openclaw.md](docs/openclaw.md) for what a plugin would need today.
|
|
||||||
|
|
||||||
## Track 8: Benchmarking & Validation
|
|
||||||
**Status:** 🟢 Complete
|
|
||||||
**Priority:** High
|
|
||||||
**Crates:** `clawhdf5-agent`, `clawhdf5-bench`
|
|
||||||
|
|
||||||
- [x] **8.1** MemoryArena benchmark — 35 queries, 50 sessions, Hit@10=91.4%, MRR=0.547
|
|
||||||
- [x] **8.2** LongMemEval benchmark — 500 queries, session Hit@1=100%, turn Hit@5=84.4% (beats MemX 51.6%), MRR=0.660
|
|
||||||
- [x] **8.3** Latency benchmarks — vector search at 1K/10K/100K, hybrid/RRF, graph traversal, consolidation, temporal
|
|
||||||
- [x] **8.4** Memory footprint — 1.7 KB/record uncompressed, 282 B compressed (6.2x ratio), 100K+ rec/s ingestion
|
|
||||||
- [x] **8.5** Consolidation efficiency — 8.8x search speedup, 90% noise eviction, zero quality loss
|
|
||||||
- [x] **8.6** Cross-platform benchmarks — x86 measured, ARM estimated, cross_platform.sh script
|
|
||||||
- [x] **8.7** Published results in BENCHMARKS.md with ephemeral tier Redis comparison (70-140x faster)
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## Implementation Order
|
|
||||||
|
|
||||||
**Phase 1:** ~~Tracks 1, 2, 3 — core memory intelligence~~ 🟢 Complete
|
|
||||||
**Phase 2:** ~~Track 4 (temporal) + Track 5 (security)~~ 🟢 Complete
|
|
||||||
**Phase 3:** ~~Track 6 (multi-modal)~~ 🟢 Complete; Track 7 (OpenClaw integration) withdrawn
|
|
||||||
**Phase 4:** ~~Track 8 (benchmarking + validation)~~ 🟢 Complete
|
|
||||||
|
|
||||||
All 8 tracks delivered. 1,650+ tests passing, zero clippy warnings.
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
## What's Next
|
|
||||||
|
|
||||||
Verified against current repo state on 2026-08-05 (see also `docs/superpowers/plans/` for the filter-codec/format-write/MPI-IO work, now shipped):
|
|
||||||
|
|
||||||
- [ ] TypeScript bridge not wired into CI — `packages/clawhdf5-node/` already has a complete, working napi-rs package (package.json, tsconfig, hand-written TS wrapper matching all 21 `#[napi]` items, Jest test suite, README); it isn't published to npm and has no committed lockfile
|
|
||||||
- [ ] Publish crates to crates.io — no `publish` config anywhere in the workspace yet
|
|
||||||
- [ ] Python wheel distribution via maturin — `crates/clawhdf5-py/pyproject.toml` exists (maturin-buildable locally) but wheels aren't published anywhere
|
|
||||||
- [ ] `chunked_read.rs`/`data_read.rs` full bounds-check audit + scheduled fuzz campaigns (the new `fuzz_dataset_read` target covers the two files' main entry points; a full manual audit of every indexing site is still open) — see Tier 4 below
|
|
||||||
- [ ] WAL per-entry checksum landed as CRC32 (see below); a stronger per-entry format (explicit length prefix, avoiding the read-then-verify restructuring) could still be revisited if profiling shows it matters
|
|
||||||
- [ ] HNSW build parallelism is still narrow (only `prune_connections`); the correctness-sensitive outer insert loop needs its own dedicated design pass before parallelizing
|
|
||||||
|
|
||||||
### Recently closed out (2026-08-05, Tier 3–4 hardening pass)
|
|
||||||
|
|
||||||
- [x] Academic benchmark cross-validation — LongMemEval reproduced against MemX on tank (Ryzen 7 7800X3D): turn-level Hit@5 84.4% vs MemX's 51.6%; recall numbers are deterministic and reproduce exactly across machines. SIMD/Parallelism and Vector Search sections also re-run and dated. See [BENCHMARKS.md § Independent Validation: tank — LongMemEval & Vector Search](BENCHMARKS.md#independent-validation-tank--longmemeval--vector-search-ryzen-7-7800x3d-2026-08-05)
|
|
||||||
- [x] Android JNI (`clawhdf5-android`): validate `embedding_len`/`query_embedding_len` against the handle's configured `embedding_dim` before constructing a slice from a raw pointer
|
|
||||||
- [x] `clawhdf5-py`: bumped pyo3/numpy 0.28 → 0.29, clearing two RUSTSEC advisories
|
|
||||||
- [x] WAL (`clawhdf5-agent`): length-prefix caps (`MAX_WAL_FIELD_LEN`) to reject a corrupted length claim before allocating, then a full per-entry CRC32 trailer (`WAL_VERSION` 2) so a bit-flip stops replay cleanly instead of loading corrupted data; old-format WAL files still read correctly and are migrated on next open
|
|
||||||
- [x] `chunked_read.rs`/`data_read.rs`/`local_heap.rs` bounds-check audit: added `ensure_len` overflow guards, a recursion-depth guard against cyclic B-trees, and a fix for an unguarded compound-datatype byte-offset overrun. Added a new `fuzz_dataset_read` cargo-fuzz target exercising the contiguous/chunked/compact read paths — it found and we fixed 3 real crash bugs (integer-overflow panics) within the first few runs
|
|
||||||
- [x] `clawhdf5-ann`: optional `parallel` feature (rayon) for HNSW's `prune_connections` neighbor-distance computation
|
|
||||||
- [x] `[workspace.dependencies]` added for `tempfile`/`criterion`/`half`/`serde`, fixing a real version skew on `half` (2 vs 2.7)
|
|
||||||
|
|
||||||
### Recently closed out (2026-08-05 hardening pass)
|
|
||||||
|
|
||||||
- [x] CI/CD pipeline — `.gitea/workflows/ci.yml` now runs `scripts/ci-test.sh` (fmt, clippy, tests, no_std check) on push/PR to `main`
|
|
||||||
- [x] Fixed no_std build breakage in `clawhdf5-format` (missing alloc imports, `AtomicU64` unsupported on thumbv7em, `f64::powi` requiring std/libm)
|
|
||||||
- [x] Fixed version skew: `clawhdf5-py` (pyproject.toml) and `packages/clawhdf5-node` (package.json) were both behind the actual crate version
|
|
||||||
|
|
||||||
### Recently closed out (2026-08-03 cleanup pass)
|
|
||||||
|
|
||||||
- [x] Removed `clawhdf5-types` — it was an empty 1-line stub crate; shared type definitions already live in `clawhdf5-format`, so CLAUDE.md and the workspace manifest were corrected instead of filling it in
|
|
||||||
- [x] Superblock v4 (page-buffer mode) read/write — the only unimplemented task from `docs/superpowers/plans/2026-06-29-format-write-extensions.md`; now done (`Superblock::parse_v4`/`serialize`, `FileWriter::with_page_size`)
|
|
||||||
- [x] Reconciled the three `docs/superpowers/plans/*.md` docs against actual shipped code — they were pre-work plans for `d6c4d4f` (2026-06-30), committed to git late; checkboxes now reflect reality
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
_Last updated: 2026-08-05_
|
|
||||||
|
|||||||
@@ -29,7 +29,8 @@
|
|||||||
# - Use wasm-pack with a custom bench harness
|
# - Use wasm-pack with a custom bench harness
|
||||||
# - Replace std::time::Instant with web_sys::Performance::now()
|
# - Replace std::time::Instant with web_sys::Performance::now()
|
||||||
# - Replace TempDir/HDF5 I/O with an in-memory backend (separate effort)
|
# - Replace TempDir/HDF5 I/O with an in-memory backend (separate effort)
|
||||||
# See ROADMAP.md §WASM for the full scope.
|
# Browser reads are tested (not benchmarked) by
|
||||||
|
# examples/wasm-viewer/test/run.sh; see examples/wasm-viewer/README.md.
|
||||||
|
|
||||||
set -euo pipefail
|
set -euo pipefail
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
/.cache/
|
||||||
|
# pin the probe's dependencies (the workspace lock is not committed)
|
||||||
|
!/probe/Cargo.lock
|
||||||
@@ -0,0 +1,70 @@
|
|||||||
|
# Conformance sweep
|
||||||
|
|
||||||
|
Reads every HDF5 file of eight public corpora with clawhdf5 and with
|
||||||
|
h5py/libhdf5, compares the two readings object by object, and writes
|
||||||
|
[`CONFORMANCE.md`](../CONFORMANCE.md).
|
||||||
|
|
||||||
|
```sh
|
||||||
|
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh # ~30 s once the corpus is cached
|
||||||
|
conformance/run.sh --no-fetch # use the cached corpus as is
|
||||||
|
conformance/run.sh --update-baseline # after an intended change in results
|
||||||
|
```
|
||||||
|
|
||||||
|
Latest result (tank, 2026-09-28 04:29 UTC, `conformance/run.sh --no-fetch
|
||||||
|
--update-baseline`): 602 of 697 files ok, 1 our-error, 0 mismatch, 2
|
||||||
|
ref-bug, 92 h5py-cannot-read, and no panic, hang, crash or out-of-memory.
|
||||||
|
The our-error file is `bad_nbit_parms_walk.h5`, which flips between ref-bug
|
||||||
|
and our-error from run to run (see `docs/known-issues.md`). The report with every file is
|
||||||
|
[`CONFORMANCE.md`](../CONFORMANCE.md).
|
||||||
|
|
||||||
|
## Classes
|
||||||
|
|
||||||
|
`compare.py` puts each file in one class:
|
||||||
|
|
||||||
|
| class | meaning |
|
||||||
|
|---|---|
|
||||||
|
| **ok** | clawhdf5 and h5py read the same objects with the same values |
|
||||||
|
| **our-error** | h5py reads something clawhdf5 refuses |
|
||||||
|
| **mismatch** | both read it, with different values or structure |
|
||||||
|
| **h5py-cannot-read** | h5py (libhdf5) cannot read the file; not compared |
|
||||||
|
| **ref-bug** | h5py reads an object clawhdf5 refuses, but only through a libhdf5 over-read: `ref_bugs.py` re-reads it in six processes with different heaps (import order, `MALLOC_PERTURB_`) and its values change. The file is ref-bug only while that is confirmed in the same run; if the values become stable it counts as our-error again |
|
||||||
|
| **panic / hang / crash / oom** | a clawhdf5 failure under the timeout and address-space limit; the gate fails on any |
|
||||||
|
|
||||||
|
Where h5py itself returns wrong values through a known h5py bug (the
|
||||||
|
big-endian variable-length bug: elements returned with the file's bytes
|
||||||
|
under a little-endian dtype), `ref.py` checks that the installed h5py has
|
||||||
|
the bug, corrects the values before hashing and marks them `ref_fix`, so
|
||||||
|
those objects are still compared. The evidence for the three remaining
|
||||||
|
non-ok files (ref-bug or, for one, our-error) is under "Conformance: the last non-ok files" in
|
||||||
|
[`docs/known-issues.md`](../docs/known-issues.md).
|
||||||
|
|
||||||
|
Needs Rust, `git`, `h5dump` (Debian/Ubuntu `hdf5-tools`), `libaec` (for the
|
||||||
|
probe's `szip` feature; `libaec-dev`), and a Python with the packages in
|
||||||
|
`requirements.txt`. The first run downloads about 450 MB of sparse checkouts.
|
||||||
|
|
||||||
|
| file | role |
|
||||||
|
|---|---|
|
||||||
|
| `corpus.txt` | the corpora: git URL, pinned commit, swept root, sparse-checkout patterns |
|
||||||
|
| `fetch-corpus.sh` | shallow, sparse, blob-filtered checkout of each pinned commit into `.cache/src/` (gitignored); no-op when already there |
|
||||||
|
| `list_files.py` | which files are probed (HDF5/netCDF-4 extensions minus netCDF classic, plus the CVE reproducers) |
|
||||||
|
| `probe/` | the clawhdf5 side: a standalone crate (outside the workspace, so `cargo test --workspace` never builds it) that walks a file with `clawhdf5-format` and prints canonical JSON |
|
||||||
|
| `ref.py` | the h5py side: the same JSON from h5py (values corrected for a known h5py bug are marked `ref_fix`) |
|
||||||
|
| `ref_bugs.py` | re-reads the objects h5py reads only through a libhdf5 bug in six differently-set-up processes; an object whose values change is confirmed as a libhdf5 over-read |
|
||||||
|
| `test_ref.py` | tests of `ref.py`'s correction and `ref_bugs.py`'s confirmation (`python conformance/test_ref.py`) |
|
||||||
|
| `run_one.sh` | runs both sides on one file (and `h5dump` on the CVE corpus) under a timeout and an address-space limit |
|
||||||
|
| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / ref-bug / panic / hang / crash / oom) and groups root causes |
|
||||||
|
| `report.py` | writes `CONFORMANCE.md` |
|
||||||
|
| `check.py` | the gate: fails on any panic/hang/crash/oom, on an ok count below `baseline.json`, or on a baseline-ok file that is no longer ok |
|
||||||
|
| `baseline.json` | the ok files the gate holds the line on |
|
||||||
|
| `requirements.txt` | pinned h5py / numpy / hdf5plugin / netCDF4 |
|
||||||
|
|
||||||
|
Results for every file (both sides' JSON and stderr, `results.csv`,
|
||||||
|
`results.json`, `summary.md`) are left in `.cache/results/`.
|
||||||
|
|
||||||
|
The nightly job is `.gitea/workflows/conformance.yml`; it prints the report
|
||||||
|
into the job log.
|
||||||
|
|
||||||
|
The canonical value encoding both sides hash is documented at the top of
|
||||||
|
`probe/src/main.rs`. Values are compared as libhdf5 presents them: a float
|
||||||
|
with a non-IEEE bit layout (N-Bit) or an integer with a bit offset is compared
|
||||||
|
as the converted number, not as raw file bytes.
|
||||||
@@ -0,0 +1,648 @@
|
|||||||
|
{
|
||||||
|
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||||
|
"commit": "bf5a163dcf7fe28d651ada6545d8136ffeffc825",
|
||||||
|
"date": "2026-09-28 04:29 UTC",
|
||||||
|
"reference": "h5py 3.16.0 / HDF5 2.0.0",
|
||||||
|
"files": 697,
|
||||||
|
"ok": 602,
|
||||||
|
"counts": {
|
||||||
|
"h5py-cannot-read": 92,
|
||||||
|
"ok": 602,
|
||||||
|
"our-error": 1,
|
||||||
|
"ref-bug": 2
|
||||||
|
},
|
||||||
|
"per_corpus": {
|
||||||
|
"NCAS-CMS_pyfive": {
|
||||||
|
"ok": 33
|
||||||
|
},
|
||||||
|
"cve_hdf5": {
|
||||||
|
"h5py-cannot-read": 32,
|
||||||
|
"ok": 113,
|
||||||
|
"ref-bug": 2
|
||||||
|
},
|
||||||
|
"h5py_data": {
|
||||||
|
"ok": 4
|
||||||
|
},
|
||||||
|
"hdf5": {
|
||||||
|
"h5py-cannot-read": 60,
|
||||||
|
"ok": 405,
|
||||||
|
"our-error": 1
|
||||||
|
},
|
||||||
|
"netcdf-c": {
|
||||||
|
"ok": 20
|
||||||
|
},
|
||||||
|
"netcdf4-python": {
|
||||||
|
"ok": 18
|
||||||
|
},
|
||||||
|
"usnistgov_h5wasm": {
|
||||||
|
"ok": 5
|
||||||
|
},
|
||||||
|
"xarray-data": {
|
||||||
|
"ok": 4
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"ok_files": [
|
||||||
|
"NCAS-CMS_pyfive/tests/compact.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/btreev2.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/chunked.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/cmip_bad_eg.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/compressed.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/compressed_v1.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dataset_datatypes.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dataset_multidim.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dim_scales.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/earliest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_h5variable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_variable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_variable.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enums_from_netcdf.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fillvalue_earliest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fillvalue_latest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/filter_pipeline_v2.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fletcher32.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fractal_heap_no_mci_rlat.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/groups.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/h5netcdf_test.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_A.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_A_contiguous.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_B.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/latest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/netcdf4_classic.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/new_style_groups.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/noy_AERmonZ_UKESM1-0-LL_piControl_r1i1p1f2_gnz_200001-200012.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/references.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/resizable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/opaque_datetime.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/opaque_fixed.hdf5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4330.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4331.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4332-mtime-new.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4332-mtime.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4333.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17505.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17506.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17507.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17508.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17509.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11202.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11203.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11204.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11205.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11206-new.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11206-old.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11207.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13867.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13868.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13869.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13870.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13871.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13872.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13873.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13875.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14031.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14033.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14034.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14035.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14460.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-15671.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-15672.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-16438.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17233.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17234.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17237.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17432.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17434.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17435.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17437.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17438",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17439",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8396.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8397.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8398.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-9151.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-9152.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-10811.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-18232.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-18494.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-36977.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-37501.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-45829.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-45833.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-46243.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-46244.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29157.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29158.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29159.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29160.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29161.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29162.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29163.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29164.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29165.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29166.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32605.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32606.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32607-1.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32607-2.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32608.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32610.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32611.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32612.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32613.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32614.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32615.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32616.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32617.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32618.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32619.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32620.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32621.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32622.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32623.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32624.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33873.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33874.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33875.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33876.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33877.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2309.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2310.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2924.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2925.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-44905.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-1.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-2.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-3.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-4.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6516.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6857.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-7067.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-26200.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-34734.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-92627.h5",
|
||||||
|
"cve_hdf5/cvefiles/unknown-1.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4431-poc-03.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4432-poc-05.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4433-poc-08.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4435-poc-10.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh_2649_flawed.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh_2649_plain_model.h5",
|
||||||
|
"h5py_data/compound-dtype-complex.h5",
|
||||||
|
"h5py_data/vlen_string_dset.h5",
|
||||||
|
"h5py_data/vlen_string_dset_utc.h5",
|
||||||
|
"h5py_data/vlen_string_s390x.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bitgroom.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc2.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bshuf.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bzip2.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_granularbr.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_jpeg.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lzf.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zfp.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zstd.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/c++/test/th5s.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be_new_ref-32bit.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be_new_ref.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_le.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_le_new_ref.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ld.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_be.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_cray.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_le.h5",
|
||||||
|
"hdf5/test/testfiles/aggr.h5",
|
||||||
|
"hdf5/test/testfiles/bad_chunk_ndims.h5",
|
||||||
|
"hdf5/test/testfiles/bad_compound.h5",
|
||||||
|
"hdf5/test/testfiles/bad_offset.h5",
|
||||||
|
"hdf5/test/testfiles/be_data.h5",
|
||||||
|
"hdf5/test/testfiles/be_extlink1.h5",
|
||||||
|
"hdf5/test/testfiles/be_extlink2.h5",
|
||||||
|
"hdf5/test/testfiles/btree_idx_1_6.h5",
|
||||||
|
"hdf5/test/testfiles/btree_idx_1_8.h5",
|
||||||
|
"hdf5/test/testfiles/charsets.h5",
|
||||||
|
"hdf5/test/testfiles/corrupt_stab_msg.h5",
|
||||||
|
"hdf5/test/testfiles/deflate.h5",
|
||||||
|
"hdf5/test/testfiles/file_image_core_test.h5",
|
||||||
|
"hdf5/test/testfiles/filespace_1_6.h5",
|
||||||
|
"hdf5/test/testfiles/filespace_1_8.h5",
|
||||||
|
"hdf5/test/testfiles/fill18.h5",
|
||||||
|
"hdf5/test/testfiles/fill_old.h5",
|
||||||
|
"hdf5/test/testfiles/filter_error.h5",
|
||||||
|
"hdf5/test/testfiles/fsm_aggr_nopersist.h5",
|
||||||
|
"hdf5/test/testfiles/fsm_aggr_persist.h5",
|
||||||
|
"hdf5/test/testfiles/group_old.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext1_f.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext1_i.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext2_if.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext_none.h5",
|
||||||
|
"hdf5/test/testfiles/le_data.h5",
|
||||||
|
"hdf5/test/testfiles/le_extlink1.h5",
|
||||||
|
"hdf5/test/testfiles/le_extlink2.h5",
|
||||||
|
"hdf5/test/testfiles/memleak_H5O_dtype_decode_helper_H5Odtype.h5",
|
||||||
|
"hdf5/test/testfiles/mergemsg.h5",
|
||||||
|
"hdf5/test/testfiles/noencoder.h5",
|
||||||
|
"hdf5/test/testfiles/none.h5",
|
||||||
|
"hdf5/test/testfiles/paged_nopersist.h5",
|
||||||
|
"hdf5/test/testfiles/paged_persist.h5",
|
||||||
|
"hdf5/test/testfiles/specmetaread.h5",
|
||||||
|
"hdf5/test/testfiles/tarrold.h5",
|
||||||
|
"hdf5/test/testfiles/tbad_msg_count.h5",
|
||||||
|
"hdf5/test/testfiles/tbogus.h5",
|
||||||
|
"hdf5/test/testfiles/test_filters_be.h5",
|
||||||
|
"hdf5/test/testfiles/test_filters_le.h5",
|
||||||
|
"hdf5/test/testfiles/th5s.h5",
|
||||||
|
"hdf5/test/testfiles/tlayouto.h5",
|
||||||
|
"hdf5/test/testfiles/tmisc38a.h5",
|
||||||
|
"hdf5/test/testfiles/tmisc38b.h5",
|
||||||
|
"hdf5/test/testfiles/tmtimen.h5",
|
||||||
|
"hdf5/test/testfiles/tmtimeo.h5",
|
||||||
|
"hdf5/test/testfiles/tnullspace.h5",
|
||||||
|
"hdf5/test/testfiles/tsizeslheap.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bigendian/tall.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bigendian/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binfp64.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin8w.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binuin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binuin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bounds_latest_latest.h5",
|
||||||
|
"hdf5/tools/test/testfiles/charsets.h5",
|
||||||
|
"hdf5/tools/test/testfiles/compounds_array_vlen1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/compounds_array_vlen2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/err_attr_dspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/file_space.h5",
|
||||||
|
"hdf5/tools/test/testfiles/filter_fail.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_equal.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_less.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_noclose.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_equal.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_less.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_mdc_image.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_sec2_v0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_sec2_v2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_extlinks_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_extlinks_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_ref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copytst.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copytst_new.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr_v_level1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr_v_level2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_basic1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_basic2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_comp_vl_strs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_danglelinks1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_danglelinks2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dtypes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_empty.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_enum_invalid_values.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_eps1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_eps2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude1-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude1-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude2-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude2-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude3-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude3-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_ext2softlink_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_ext2softlink_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_extlink_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_extlink_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_hyper1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_hyper2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_linked_softlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_links.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_dset_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_dset_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_softlinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_strings1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_strings2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_types.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_edge_v3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_err_level.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_i.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_s.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_if.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_is.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_non_v3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_CVE-2018-14460.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_CVE-2018-17432.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_aggr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_attr_refs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_deflate.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_early.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_f32le.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_f32le_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fill.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_filters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fletcher.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_nopersist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_persist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_hlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_1d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_2d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_2d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_3d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_3d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout.UD.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layouto.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_named_dtypes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nbit.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum_deflated.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_paged_nopersist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_paged_persist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_refs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_shuffle.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_soffset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_szip.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_uint8be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_uint8be_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_old_fill.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_old_layout.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_refcount.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_filters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_idx.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_newgrat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_threshold.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_tsohm.h5",
|
||||||
|
"hdf5/tools/test/testfiles/mod_h5clear_mdc_image.h5",
|
||||||
|
"hdf5/tools/test/testfiles/non_comparables1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/non_comparables2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_i.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_s.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_if.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_is.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/packedbits.h5",
|
||||||
|
"hdf5/tools/test/testfiles/t128bit_float.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE-2021-37501_attr_decode.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_new.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_old.h5",
|
||||||
|
"hdf5/tools/test/testfiles/taindices.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tall.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray1_big.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray5.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr4_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattrreg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbfloat16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbfloat16_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbigdims.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbinary.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbitnopaque.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tchar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdintarray.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdints.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcomplex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcomplex_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound_complex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound_complex2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdatareg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset_idx.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tempty.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinkfar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinksrc.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinktar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textpfe.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfcontents1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfcontents2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfilters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat16_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat6.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloatsattrs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfpformat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfvalues.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgroup.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgrp_comments.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgrpnullspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/thlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/thyperslab.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintascii.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tints4dims.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintsattrs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintsnodata.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tlarge_objname.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tldouble.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tldouble_scalar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tlonglinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tloop.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnamed_dtype_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnestedcmpddt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnestedcomp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tno-subset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnullspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/torderattr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tordergr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_compat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_ext1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_ext2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_grp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_obj.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_obj_del.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_param.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_reg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_reg_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tsaf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarintattrsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarstring.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tslink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tsoftlinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_dset_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_dset_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudfilter.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudfilter2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes5.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvlenstr_array.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvlstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvms.h5",
|
||||||
|
"hdf5/tools/test/testfiles/twithub.h5",
|
||||||
|
"hdf5/tools/test/testfiles/twithub513.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtfp32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtfp64.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtuin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtuin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_e.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_e.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/3_1_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/3_2_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/f-0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/f-3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/vds-eiger.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/vds-percival-unlim-maxmin.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tbitfields.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tcompound2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tenum.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/test35.nc",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tloop2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tmany.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-amp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-apos.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-gt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-lt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-quot.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-sp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tnodata.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tobjref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/topaque.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref-escapes-at.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref-escapes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tstring-at.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tstring.h5",
|
||||||
|
"hdf5/tools/test/testfiles/zerodim.h5",
|
||||||
|
"netcdf-c/h5_test/ref_tst_h_compounds.h5",
|
||||||
|
"netcdf-c/h5_test/ref_tst_h_compounds2.h5",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat1.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat2.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat3.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_szip.h5",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_compounds.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_dims.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_interops4.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_xplatform2_1.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_xplatform2_2.nc",
|
||||||
|
"netcdf-c/nc_test4/tdset.h5",
|
||||||
|
"netcdf-c/ncdump/ref_nc_test_netcdf4_4_0.nc",
|
||||||
|
"netcdf-c/ncdump/ref_no_ncproperty.nc",
|
||||||
|
"netcdf-c/ncdump/ref_provenance_v1.nc",
|
||||||
|
"netcdf-c/ncdump/ref_test_corrupt_magic.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds2.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds3.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds4.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_irish_rover.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2000.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2001.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2002.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2003.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2004.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2005.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2006.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2007.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2008.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2009.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2010.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2011.nc",
|
||||||
|
"netcdf4-python/examples/data/rtofs_glo_3dz_f006_6hrly_reg3.nc",
|
||||||
|
"netcdf4-python/test/20171025_2056.Cloud_Top_Height.nc",
|
||||||
|
"netcdf4-python/test/issue1152.nc",
|
||||||
|
"netcdf4-python/test/issue671.nc",
|
||||||
|
"netcdf4-python/test/issue672.nc",
|
||||||
|
"netcdf4-python/test/test_gold.nc",
|
||||||
|
"usnistgov_h5wasm/test/array.h5",
|
||||||
|
"usnistgov_h5wasm/test/compressed.h5",
|
||||||
|
"usnistgov_h5wasm/test/empty.h5",
|
||||||
|
"usnistgov_h5wasm/test/float16.h5",
|
||||||
|
"usnistgov_h5wasm/test/vlen.h5",
|
||||||
|
"xarray-data/ROMS_example.nc",
|
||||||
|
"xarray-data/basin_mask.nc",
|
||||||
|
"xarray-data/imerghh_730.hdf5",
|
||||||
|
"xarray-data/precipitation.nc4"
|
||||||
|
]
|
||||||
|
}
|
||||||
Executable
+88
@@ -0,0 +1,88 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""check.py <results_dir> <baseline.json> [--update]
|
||||||
|
|
||||||
|
The conformance gate. Fails (exit 1) when
|
||||||
|
* clawhdf5 panicked, hung, crashed or ran out of memory on any file, or
|
||||||
|
* the ok count fell below the baseline's, or
|
||||||
|
* a file the baseline lists as ok is no longer ok (even if another file
|
||||||
|
became ok and the total held).
|
||||||
|
New ok files are reported so the baseline can be raised (--update rewrites it
|
||||||
|
from the results).
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
FATAL = ("panic", "hang", "crash", "oom")
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||||
|
update = "--update" in sys.argv
|
||||||
|
res_dir, base_path = args
|
||||||
|
res = json.load(open(os.path.join(res_dir, "results.json")))
|
||||||
|
rows = res["rows"]
|
||||||
|
counts = {}
|
||||||
|
per_corpus = {}
|
||||||
|
for r in rows:
|
||||||
|
counts[r["class"]] = counts.get(r["class"], 0) + 1
|
||||||
|
pc = per_corpus.setdefault(r["corpus"], {})
|
||||||
|
pc[r["class"]] = pc.get(r["class"], 0) + 1
|
||||||
|
ok_files = sorted(r["file"] for r in rows if r["class"] == "ok")
|
||||||
|
|
||||||
|
if update:
|
||||||
|
meta = {}
|
||||||
|
mp = os.path.join(res_dir, "report-meta.json")
|
||||||
|
if os.path.exists(mp):
|
||||||
|
meta = json.load(open(mp))
|
||||||
|
base = {
|
||||||
|
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. "
|
||||||
|
"Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||||
|
"commit": meta.get("commit", ""),
|
||||||
|
"date": meta.get("date", ""),
|
||||||
|
"reference": meta.get("reference", ""),
|
||||||
|
"files": len(rows),
|
||||||
|
"ok": len(ok_files),
|
||||||
|
"counts": dict(sorted(counts.items())),
|
||||||
|
"per_corpus": {k: dict(sorted(v.items())) for k, v in sorted(per_corpus.items())},
|
||||||
|
"ok_files": ok_files,
|
||||||
|
}
|
||||||
|
with open(base_path, "w") as fh:
|
||||||
|
json.dump(base, fh, indent=1)
|
||||||
|
fh.write("\n")
|
||||||
|
print(f"baseline updated: {len(ok_files)} ok of {len(rows)} files -> {base_path}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
base = json.load(open(base_path))
|
||||||
|
failures = []
|
||||||
|
fatal = [r for r in rows if r["class"] in FATAL]
|
||||||
|
for r in fatal:
|
||||||
|
failures.append(f"{r['class']}: {r['file']}: {r['ours_detail'][:200]}")
|
||||||
|
if len(ok_files) < base["ok"]:
|
||||||
|
failures.append(f"ok count dropped: {len(ok_files)} < baseline {base['ok']}")
|
||||||
|
now_ok = set(ok_files)
|
||||||
|
by_file = {r["file"]: r for r in rows}
|
||||||
|
for f in base["ok_files"]:
|
||||||
|
if f not in now_ok:
|
||||||
|
r = by_file.get(f)
|
||||||
|
why = f"now {r['class']}: {(r['ours_detail'] or r['first_issue'])[:200]}" if r else "no longer in the corpus"
|
||||||
|
failures.append(f"regressed: {f}: {why}")
|
||||||
|
gained = sorted(now_ok - set(base["ok_files"]))
|
||||||
|
|
||||||
|
print(f"conformance: {len(ok_files)} ok of {len(rows)} files (baseline {base['ok']} of {base['files']}); "
|
||||||
|
+ ", ".join(f"{k} {v}" for k, v in sorted(counts.items())))
|
||||||
|
if gained:
|
||||||
|
print(f"{len(gained)} file(s) newly ok — raise the baseline with `conformance/run.sh --update-baseline`:")
|
||||||
|
for f in gained:
|
||||||
|
print(f" + {f}")
|
||||||
|
if failures:
|
||||||
|
print(f"CONFORMANCE GATE FAILED ({len(failures)}):")
|
||||||
|
for f in failures:
|
||||||
|
print(f" - {f}")
|
||||||
|
return 1
|
||||||
|
print("conformance gate passed")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
Executable
+322
@@ -0,0 +1,322 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""compare.py <results_dir>: classify each file and group failures by root cause.
|
||||||
|
|
||||||
|
Writes <results_dir>/results.csv, results.json and summary.md.
|
||||||
|
File classes (first match wins):
|
||||||
|
hang, oom, crash, panic ours: timeout / allocation failure / signal / any panic (caught or not)
|
||||||
|
h5py-cannot-read libhdf5/h5py failed to open the file (or crashed/hung)
|
||||||
|
ref-bug every issue is an object we refuse that h5py reads only through a
|
||||||
|
libhdf5 bug, confirmed in this run by ref_bugs.py (its values
|
||||||
|
change with the reading process's heap)
|
||||||
|
our-error we fail to open, list, or read something h5py reads
|
||||||
|
mismatch we read something with different shape/values, or a different object set
|
||||||
|
ok
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import csv
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
R = sys.argv[1]
|
||||||
|
RUNS = os.path.join(R, "runs")
|
||||||
|
|
||||||
|
# Objects ref_bugs.py confirmed in this run: h5py's values for them come from
|
||||||
|
# libhdf5 reading memory the file does not determine.
|
||||||
|
try:
|
||||||
|
REF_BUGS = {(b["file"], b["object"])
|
||||||
|
for b in json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"] if b.get("confirmed")}
|
||||||
|
except (OSError, ValueError, KeyError):
|
||||||
|
REF_BUGS = set()
|
||||||
|
|
||||||
|
|
||||||
|
def is_ref_bug(rel, issue):
|
||||||
|
"""An our-error on reading an object that ref_bugs.py confirmed."""
|
||||||
|
kind, detail = issue[0], issue[1]
|
||||||
|
return kind == "our-error" and any(f == rel and detail.startswith(obj + ": error: ") for f, obj in REF_BUGS)
|
||||||
|
|
||||||
|
|
||||||
|
def load(d, name):
|
||||||
|
rc_p = os.path.join(d, name + ".rc")
|
||||||
|
if not os.path.exists(rc_p):
|
||||||
|
return None
|
||||||
|
rc = int(open(rc_p).read().strip() or -1)
|
||||||
|
err = open(os.path.join(d, name + ".err"), errors="replace").read()
|
||||||
|
js = None
|
||||||
|
try:
|
||||||
|
js = json.load(open(os.path.join(d, name + ".json")))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
pass
|
||||||
|
return {"rc": rc, "err": err, "json": js}
|
||||||
|
|
||||||
|
|
||||||
|
def proc_status(p):
|
||||||
|
"""-> (status, detail)"""
|
||||||
|
if p is None:
|
||||||
|
return "missing", ""
|
||||||
|
rc, err = p["rc"], p["err"]
|
||||||
|
first_panic = next((ln for ln in err.splitlines() if ln.startswith("PANIC:") or "panicked at" in ln), "")
|
||||||
|
if rc == 0 and p["json"] is not None:
|
||||||
|
return "ok", ""
|
||||||
|
if rc == 137 or rc == 124:
|
||||||
|
return "hang", f"timeout ({os.environ.get('TMO', '20')} s)"
|
||||||
|
if "memory allocation of" in err or "MemoryError" in err or "std::bad_alloc" in err:
|
||||||
|
m = re.search(r"memory allocation of \d+ bytes failed", err)
|
||||||
|
return "oom", m.group(0) if m else "allocation failure"
|
||||||
|
if "overflowed its stack" in err:
|
||||||
|
return "crash", "stack overflow"
|
||||||
|
if rc == 101:
|
||||||
|
return "panic", first_panic or (err.strip().splitlines() or [""])[-1]
|
||||||
|
if rc in (134, 139, 136, 135, 132) or rc > 128:
|
||||||
|
sig = {134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS", 132: "SIGILL"}.get(rc, f"signal {rc - 128}")
|
||||||
|
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||||
|
return "crash", f"{sig}: {tail[0][:200] if tail else ''}"
|
||||||
|
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||||
|
return "crash", f"rc={rc}: {tail[0][:200] if tail else ''}"
|
||||||
|
|
||||||
|
|
||||||
|
def norm(msg):
|
||||||
|
m = msg.split("\n")[0]
|
||||||
|
m = re.sub(r"0x[0-9a-fA-F]+", "X", m)
|
||||||
|
m = re.sub(r'"[^"]*"', '"…"', m)
|
||||||
|
m = re.sub(r"'[^']*'", "'…'", m)
|
||||||
|
m = re.sub(r"\d+", "N", m)
|
||||||
|
return m[:160]
|
||||||
|
|
||||||
|
|
||||||
|
def panic_head(msg):
|
||||||
|
"""First line + first clawhdf5 frame of a PANIC record."""
|
||||||
|
lines = msg.split("\n")
|
||||||
|
frame = next((ln.strip() for ln in lines[1:] if "clawhdf5_format" in ln), "")
|
||||||
|
return lines[0][:300], frame[:300]
|
||||||
|
|
||||||
|
|
||||||
|
def eq_shape(a, b):
|
||||||
|
return a == b
|
||||||
|
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
issues_by_file = {}
|
||||||
|
root_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||||
|
mismatch_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||||
|
panics = []
|
||||||
|
ref_only_errors = collections.Counter()
|
||||||
|
incomparable = collections.Counter()
|
||||||
|
# Values ref.py corrected for a known h5py bug: (file, object, fixes, same as ours)
|
||||||
|
ref_fixes = []
|
||||||
|
|
||||||
|
|
||||||
|
def add(bucket, key, file, example):
|
||||||
|
b = bucket[key]
|
||||||
|
b["count"] += 1
|
||||||
|
if file not in b["files"] and len(b["examples"]) < 6:
|
||||||
|
b["examples"].append(example)
|
||||||
|
b["files"].add(file)
|
||||||
|
|
||||||
|
|
||||||
|
files = [ln.strip() for ln in open(os.path.join(R, "files.txt")) if ln.strip()]
|
||||||
|
for rel in files:
|
||||||
|
d = os.path.join(RUNS, rel.replace("/", "__"))
|
||||||
|
corpus = rel.split("/")[0]
|
||||||
|
ours, ref = load(d, "ours"), load(d, "ref")
|
||||||
|
h5dump = load(d, "h5dump")
|
||||||
|
os_, od = proc_status(ours)
|
||||||
|
rs, rd = proc_status(ref)
|
||||||
|
oj = ours["json"] if ours else None
|
||||||
|
rj = ref["json"] if ref else None
|
||||||
|
issues = [] # (kind, detail)
|
||||||
|
caught_panics = []
|
||||||
|
|
||||||
|
def scan_err(path, what, msg):
|
||||||
|
if msg.startswith("PANIC:"):
|
||||||
|
caught_panics.append((path, what, msg))
|
||||||
|
|
||||||
|
if oj:
|
||||||
|
for o in oj.get("objects", []):
|
||||||
|
for k in ("error", "attrs_error", "list_error"):
|
||||||
|
if k in o:
|
||||||
|
scan_err(o["path"], k, o[k])
|
||||||
|
for an, av in (o.get("attrs") or {}).items():
|
||||||
|
if "error" in av:
|
||||||
|
scan_err(o["path"], f"attr {an}", av["error"])
|
||||||
|
if oj.get("open_error", "").startswith("PANIC:"):
|
||||||
|
caught_panics.append(("<open>", "open", oj["open_error"]))
|
||||||
|
|
||||||
|
ref_open_fail = rs != "ok" or (rj is not None and "open_error" in rj)
|
||||||
|
ours_open_err = oj.get("open_error") if oj else None
|
||||||
|
n_obj = n_ok = 0
|
||||||
|
if os_ == "ok" and rj and not ref_open_fail and not ours_open_err:
|
||||||
|
ro = {x["path"]: x for x in rj.get("objects", [])}
|
||||||
|
oo = {x["path"]: x for x in oj.get("objects", [])}
|
||||||
|
our_list_errors = [x for x in oo.values() if "list_error" in x]
|
||||||
|
for p in sorted(set(ro) | set(oo)):
|
||||||
|
a, b = ro.get(p), oo.get(p)
|
||||||
|
n_obj += 1
|
||||||
|
if a is None:
|
||||||
|
issues.append(("mismatch", f"extra object {p} (kind={b.get('kind')})", "extra-object", b))
|
||||||
|
continue
|
||||||
|
if b is None:
|
||||||
|
if our_list_errors:
|
||||||
|
continue # accounted for by the list_error
|
||||||
|
issues.append(("mismatch", f"missing object {p} (kind={a.get('kind')})", "missing-object", a))
|
||||||
|
continue
|
||||||
|
ok = True
|
||||||
|
if a.get("kind") != b.get("kind") and "error" not in b and "error" not in a:
|
||||||
|
issues.append(("mismatch", f"{p}: kind {a.get('kind')} vs ours {b.get('kind')}", "kind", b))
|
||||||
|
ok = False
|
||||||
|
# h5py could not open the object at all: it read none of its
|
||||||
|
# attributes or links, so there is nothing to compare ours with
|
||||||
|
# (the object's own error is compared above and below).
|
||||||
|
ref_unopened = a.get("kind") == "unknown" and "error" in a
|
||||||
|
for k in ("error", "list_error", "attrs_error"):
|
||||||
|
if ref_unopened and k != "error":
|
||||||
|
continue
|
||||||
|
if k in b and k not in a:
|
||||||
|
issues.append(("our-error", f"{p}: {k}: {b[k]}", b[k], b))
|
||||||
|
ok = False
|
||||||
|
elif k in a and k not in b and k == "error":
|
||||||
|
ref_only_errors[norm(a[k])] += 1
|
||||||
|
if a.get("kind") == "dataset" and "error" not in a and "error" not in b:
|
||||||
|
if "skipped" in a or "skipped" in b:
|
||||||
|
pass
|
||||||
|
elif a.get("converted"):
|
||||||
|
incomparable[f"dataset {a['converted']}"] += 1
|
||||||
|
elif a.get("shape") != b.get("shape"):
|
||||||
|
issues.append(("mismatch", f"{p}: shape {a.get('shape')} vs ours {b.get('shape')}", "shape", b))
|
||||||
|
ok = False
|
||||||
|
elif a.get("hash") != b.get("hash"):
|
||||||
|
issues.append(("mismatch", f"{p}: values differ (h5py {a.get('dtype')} vs ours {b.get('dtype')})", "values", b | {"ref_head": a.get("head"), "ref_dtype": a.get("dtype")}))
|
||||||
|
ok = False
|
||||||
|
if a.get("ref_fix") and "hash" in b:
|
||||||
|
ref_fixes.append((rel, p, a["ref_fix"], a.get("hash") == b.get("hash")))
|
||||||
|
ra, oa = a.get("attrs") or {}, b.get("attrs") or {}
|
||||||
|
if "attrs_error" not in b and "attrs_error" not in a and not ref_unopened:
|
||||||
|
for an in sorted(set(ra) | set(oa)):
|
||||||
|
x, y = ra.get(an), oa.get(an)
|
||||||
|
if x is None:
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: extra attribute", "extra-attr", y or {}))
|
||||||
|
elif y is None:
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: missing attribute", "missing-attr", x))
|
||||||
|
elif "error" in y and "error" not in x:
|
||||||
|
issues.append(("our-error", f"{p}@{an}: {y['error']}", y["error"], y))
|
||||||
|
elif "error" in x:
|
||||||
|
continue
|
||||||
|
elif x.get("converted"):
|
||||||
|
incomparable[f"attr {x['converted']}"] += 1
|
||||||
|
elif x.get("shape") != y.get("shape"):
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: attr shape {x.get('shape')} vs ours {y.get('shape')}", "attr-shape", y | {"ref_dtype": x.get("dtype")}))
|
||||||
|
elif x.get("hash") != y.get("hash"):
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: attr values differ (h5py {x.get('dtype')} vs ours {y.get('dtype')})", "attr-values", y | {"ref_head": x.get("head"), "ref_dtype": x.get("dtype")}))
|
||||||
|
if x.get("ref_fix") and "hash" in y:
|
||||||
|
ref_fixes.append((rel, f"{p}@{an}", x["ref_fix"], x.get("hash") == y.get("hash")))
|
||||||
|
if ok:
|
||||||
|
n_ok += 1
|
||||||
|
|
||||||
|
# classify
|
||||||
|
if os_ in ("hang", "oom", "crash", "panic"):
|
||||||
|
cls = os_
|
||||||
|
elif caught_panics:
|
||||||
|
cls = "panic"
|
||||||
|
elif ref_open_fail:
|
||||||
|
cls = "h5py-cannot-read"
|
||||||
|
elif issues and all(is_ref_bug(rel, i) for i in issues):
|
||||||
|
cls = "ref-bug"
|
||||||
|
elif ours_open_err:
|
||||||
|
cls = "our-error"
|
||||||
|
issues.append(("our-error", f"open: {ours_open_err}", ours_open_err, {}))
|
||||||
|
elif any(i[0] == "our-error" for i in issues):
|
||||||
|
cls = "our-error"
|
||||||
|
elif issues:
|
||||||
|
cls = "mismatch"
|
||||||
|
else:
|
||||||
|
cls = "ok"
|
||||||
|
|
||||||
|
if os_ in ("hang", "oom", "crash", "panic") or caught_panics:
|
||||||
|
panics.append({
|
||||||
|
"file": rel, "class": cls, "detail": od,
|
||||||
|
"stderr": (ours["err"] if ours else "")[:3000],
|
||||||
|
"caught": [(p, w, m[:2500]) for p, w, m in caught_panics[:3]],
|
||||||
|
"n_caught": len(caught_panics),
|
||||||
|
})
|
||||||
|
# A ref-bug file's differences are listed with the evidence instead.
|
||||||
|
for kind, detail, key, rec in (issues if cls != "ref-bug" else []):
|
||||||
|
if kind == "our-error":
|
||||||
|
add(root_causes, norm(key), rel, detail[:300])
|
||||||
|
else:
|
||||||
|
if key in ("values", "attr-values", "shape", "attr-shape"):
|
||||||
|
mk = f"{key}: ours={rec.get('dtype')} h5py={rec.get('ref_dtype')} layout={rec.get('layout','-')} filters={rec.get('filters','-')}"
|
||||||
|
else:
|
||||||
|
mk = key
|
||||||
|
add(mismatch_causes, mk, rel, detail[:300] + (f" | ref_head={rec.get('ref_head')} our_head={rec.get('head')}" if rec.get("ref_head") else ""))
|
||||||
|
ref_detail = rd if rs != "ok" else ((rj or {}).get("open_error") or "")
|
||||||
|
h5d = ""
|
||||||
|
if h5dump:
|
||||||
|
rc = h5dump["rc"]
|
||||||
|
h5d = {0: "ok", 1: "error", 137: "hang", 124: "hang", 134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS"}.get(rc, f"rc={rc}")
|
||||||
|
if "memory allocation" in h5dump["err"] or "Cannot allocate" in h5dump["err"]:
|
||||||
|
h5d += "(oom)"
|
||||||
|
rows.append({
|
||||||
|
"file": rel, "corpus": corpus, "class": cls,
|
||||||
|
"ours": os_ if os_ != "ok" else ("open-error" if ours_open_err else ("panic" if caught_panics else "ok")),
|
||||||
|
"ours_detail": (od or ours_open_err or (caught_panics[0][2].split("\n")[0] if caught_panics else ""))[:300],
|
||||||
|
"ref": rs if rs != "ok" else ("open-error" if (rj or {}).get("open_error") else "ok"),
|
||||||
|
"ref_detail": ref_detail[:300],
|
||||||
|
"h5dump_1_14_6": h5d,
|
||||||
|
"h5dump_detail": ([ln for ln in h5dump["err"].splitlines() if ln.strip()][-1:] or [""])[0][:200] if h5dump else "",
|
||||||
|
"objects": n_obj, "objects_ok": n_ok,
|
||||||
|
"issues": len(issues), "first_issue": issues[0][1][:300] if issues else "",
|
||||||
|
"superblock": (oj or {}).get("superblock_version", ""),
|
||||||
|
})
|
||||||
|
# the first issues of each file, for report.py's known-cause matching
|
||||||
|
issues_by_file[rel] = [
|
||||||
|
{"kind": k, "key": key, "detail": det[:300], "ours_dtype": rec.get("dtype"), "ref_dtype": rec.get("ref_dtype")}
|
||||||
|
for k, det, key, rec in issues[:50]
|
||||||
|
]
|
||||||
|
|
||||||
|
with open(os.path.join(R, "results.csv"), "w", newline="") as fh:
|
||||||
|
w = csv.DictWriter(fh, fieldnames=list(rows[0].keys()))
|
||||||
|
w.writeheader()
|
||||||
|
w.writerows(rows)
|
||||||
|
|
||||||
|
|
||||||
|
def ser(b):
|
||||||
|
return {k: {"files": len(v["files"]), "count": v["count"], "examples": v["examples"], "file_list": sorted(v["files"])} for k, v in sorted(b.items(), key=lambda kv: -len(kv[1]["files"]))}
|
||||||
|
|
||||||
|
|
||||||
|
json.dump({"rows": rows, "issues": issues_by_file, "root_causes": ser(root_causes), "mismatch_causes": ser(mismatch_causes),
|
||||||
|
"panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common(),
|
||||||
|
"ref_fixes": ref_fixes, "ref_bugs_confirmed": sorted(REF_BUGS)},
|
||||||
|
open(os.path.join(R, "results.json"), "w"), indent=1)
|
||||||
|
|
||||||
|
classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "hang", "panic", "crash", "oom"]
|
||||||
|
by_corpus = collections.defaultdict(collections.Counter)
|
||||||
|
for r in rows:
|
||||||
|
by_corpus[r["corpus"]][r["class"]] += 1
|
||||||
|
by_corpus["ALL"][r["class"]] += 1
|
||||||
|
lines = ["# Conformance sweep summary", "", "| corpus | files | " + " | ".join(classes) + " |", "|---" * (len(classes) + 2) + "|"]
|
||||||
|
for c in sorted(by_corpus, key=lambda k: (k == "ALL", k)):
|
||||||
|
cnt = by_corpus[c]
|
||||||
|
lines.append(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in classes) + " |")
|
||||||
|
lines += ["", "## Panics / hangs / crashes / OOM", ""]
|
||||||
|
for p in panics:
|
||||||
|
lines.append(f"- **{p['file']}** [{p['class']}] {p['detail']}")
|
||||||
|
for path, what, m in p["caught"][:1]:
|
||||||
|
lines.append(" ```\n " + f"{path} ({what}): " + m.replace("\n", "\n ")[:1500] + "\n ```")
|
||||||
|
if not p["caught"] and p["stderr"]:
|
||||||
|
lines.append(" ```\n " + p["stderr"].strip()[:1500].replace("\n", "\n ") + "\n ```")
|
||||||
|
lines += ["", "## Our-error root causes (files affected)", ""]
|
||||||
|
for k, v in ser(root_causes).items():
|
||||||
|
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||||
|
for ex in v["examples"][:3]:
|
||||||
|
lines.append(f" - {ex}")
|
||||||
|
lines += ["", "## Mismatch root causes", ""]
|
||||||
|
for k, v in ser(mismatch_causes).items():
|
||||||
|
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||||
|
for ex in v["examples"][:3]:
|
||||||
|
lines.append(f" - {ex}")
|
||||||
|
lines += ["", "## Objects h5py fails on but we read (top)", ""]
|
||||||
|
for k, n in ref_only_errors.most_common(15):
|
||||||
|
lines.append(f"- {n} x `{k}`")
|
||||||
|
open(os.path.join(R, "summary.md"), "w").write("\n".join(lines) + "\n")
|
||||||
|
print("\n".join(lines[:4 + len(by_corpus)]))
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
# Conformance corpora, pinned by commit. fetch-corpus.sh reads this file.
|
||||||
|
#
|
||||||
|
# name git-url commit root [sparse-checkout patterns...]
|
||||||
|
#
|
||||||
|
# `root` is the directory inside the checkout that is swept ("." = all of it).
|
||||||
|
# Patterns are git non-cone sparse-checkout patterns; none = whole repository.
|
||||||
|
# Every file under <root> with an HDF5/netCDF-4 extension is probed; for
|
||||||
|
# cve_hdf5 the extension-less files in cvefiles/ and fuzzerfiles/ are too.
|
||||||
|
# Licences: each corpus keeps its upstream licence; nothing here is committed
|
||||||
|
# to this repository — the files are downloaded into the gitignored cache.
|
||||||
|
hdf5 https://github.com/HDFGroup/hdf5.git a3cf1ea82cc7a66e50029a688121e1b105a7ce88 . *.h5 *.he5 *.nc *.hdf5 *.h5f
|
||||||
|
cve_hdf5 https://github.com/HDFGroup/cve_hdf5.git 3fd1f5ae3869e01b8ae02b41d7108de7ffb1a374 .
|
||||||
|
netcdf-c https://github.com/Unidata/netcdf-c.git beb7b9585273c1548386231a59b809d906359033 . /nc_test4/*.nc /ncdump/*.nc /nc_test4/*.h5 /ncdump/*.h5 /h5_test/*.h5 /hdf5_test/*.h5
|
||||||
|
NCAS-CMS_pyfive https://github.com/NCAS-CMS/pyfive.git 8cf07b8749133f41c5e30b8a4c604486f687fe74 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||||
|
usnistgov_h5wasm https://github.com/usnistgov/h5wasm.git 02f6336527d2812783fcedabfbf42127ec8d06d2 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||||
|
netcdf4-python https://github.com/Unidata/netcdf4-python.git 6e67576d39aef8091fb20bd767b4f1a52ddc1bec . *.nc *.h5
|
||||||
|
xarray-data https://github.com/pydata/xarray-data.git a35297e9da2cc99c811014f0c8a4297345a5c28d . /basin_mask.nc /precipitation.nc4 /imerghh_730.hdf5 /eraint_uvz.nc /ROMS_example.nc /tiny.nc
|
||||||
|
# h5py 3.16.0 (tag 3.16.0), its test data files.
|
||||||
|
h5py_data https://github.com/h5py/h5py.git b2f0347c4200333acd89b43733f1caa0c115162f h5py/tests/data_files /h5py/tests/data_files/*
|
||||||
Executable
+39
@@ -0,0 +1,39 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# fetch-corpus.sh [cache_dir]
|
||||||
|
#
|
||||||
|
# Download the corpora pinned in conformance/corpus.txt into the (gitignored)
|
||||||
|
# cache: <cache>/src/<name> is a shallow, sparse, blob-filtered checkout of the
|
||||||
|
# pinned commit and <cache>/corpus/<name> links to the swept root inside it.
|
||||||
|
# A corpus already checked out at its pinned commit is left alone, so a second
|
||||||
|
# run costs nothing and needs no network.
|
||||||
|
set -euo pipefail
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
CACHE="${1:-${CONFORMANCE_CACHE:-$HERE/.cache}}"
|
||||||
|
mkdir -p "$CACHE/src" "$CACHE/corpus"
|
||||||
|
CACHE="$(cd "$CACHE" && pwd)"
|
||||||
|
|
||||||
|
retry() { local i; for i in 1 2 3 4; do "$@" && return 0; sleep $((i * 5)); done; return 1; }
|
||||||
|
|
||||||
|
grep -v '^[[:space:]]*\(#\|$\)' "$HERE/corpus.txt" | while read -r name url commit root patterns; do
|
||||||
|
src="$CACHE/src/$name"
|
||||||
|
if [ -d "$src/.git" ] && [ "$(git -C "$src" rev-parse HEAD 2>/dev/null)" = "$commit" ]; then
|
||||||
|
echo "cached $name @ ${commit:0:12}"
|
||||||
|
else
|
||||||
|
echo "fetching $name @ ${commit:0:12} from $url"
|
||||||
|
rm -rf "$src"
|
||||||
|
git init -q "$src"
|
||||||
|
git -C "$src" remote add origin "$url"
|
||||||
|
git -C "$src" config advice.detachedHead false
|
||||||
|
if [ -n "$patterns" ]; then
|
||||||
|
git -C "$src" config core.sparseCheckout true
|
||||||
|
# no-cone patterns (globs); `set -f` keeps the shell from expanding them
|
||||||
|
(set -f; printf '%s\n' $patterns) > "$src/.git/info/sparse-checkout"
|
||||||
|
fi
|
||||||
|
retry git -C "$src" fetch -q --depth 1 --filter=blob:none origin "$commit"
|
||||||
|
retry git -C "$src" checkout -q FETCH_HEAD
|
||||||
|
got="$(git -C "$src" rev-parse HEAD)"
|
||||||
|
[ "$got" = "$commit" ] || { echo "error: $name checked out $got, expected $commit" >&2; exit 1; }
|
||||||
|
fi
|
||||||
|
ln -sfn "$src/$root" "$CACHE/corpus/$name"
|
||||||
|
done
|
||||||
|
echo "corpus ready in $CACHE/corpus"
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""list_files.py <corpus_dir>: print the files the sweep probes, one per line,
|
||||||
|
as <corpus>/<path> in byte order.
|
||||||
|
|
||||||
|
* every file named *.h5 *.hdf5 *.he5 *.nc *.nc4 *.hdf *.h5f in each corpus,
|
||||||
|
except netCDF classic / 64-bit-offset / CDF5 files (magic "CDF"): they are
|
||||||
|
not HDF5, so neither side can read them and they say nothing;
|
||||||
|
* plus, for cve_hdf5, every file in cvefiles/ and fuzzerfiles/ except
|
||||||
|
.md/.c sources — the reproducers are mostly extension-less, and they are
|
||||||
|
kept whatever their bytes look like (that is their point).
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
EXTS = (".h5", ".hdf5", ".he5", ".nc", ".nc4", ".hdf", ".h5f")
|
||||||
|
|
||||||
|
|
||||||
|
def walk(top):
|
||||||
|
for dirpath, dirnames, filenames in os.walk(top):
|
||||||
|
dirnames[:] = [d for d in dirnames if d != ".git"]
|
||||||
|
for fn in filenames:
|
||||||
|
p = os.path.join(dirpath, fn)
|
||||||
|
if os.path.isfile(p) and not os.path.islink(p):
|
||||||
|
yield os.path.relpath(p, top)
|
||||||
|
|
||||||
|
|
||||||
|
def main(root):
|
||||||
|
out = set()
|
||||||
|
for corpus in sorted(os.listdir(root)):
|
||||||
|
top = os.path.join(root, corpus)
|
||||||
|
if not os.path.isdir(top):
|
||||||
|
continue
|
||||||
|
for rel in walk(top):
|
||||||
|
path = os.path.join(top, rel)
|
||||||
|
if rel.lower().endswith(EXTS):
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
if fh.read(3) == b"CDF":
|
||||||
|
continue
|
||||||
|
out.add(f"{corpus}/{rel}")
|
||||||
|
elif corpus == "cve_hdf5" and rel.split(os.sep)[0] in ("cvefiles", "fuzzerfiles") \
|
||||||
|
and not rel.endswith((".md", ".c")):
|
||||||
|
out.add(f"{corpus}/{rel}")
|
||||||
|
for f in sorted(out, key=lambda s: s.encode()):
|
||||||
|
print(f)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1])
|
||||||
Generated
+492
@@ -0,0 +1,492 @@
|
|||||||
|
# This file is automatically @generated by Cargo.
|
||||||
|
# It is not intended for manual editing.
|
||||||
|
version = 4
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "adler2"
|
||||||
|
version = "2.0.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "better_io"
|
||||||
|
version = "0.2.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ef0a3155e943e341e557863e69a708999c94ede624e37865c8e2a91b94efa78f"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "block-buffer"
|
||||||
|
version = "0.10.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
|
||||||
|
dependencies = [
|
||||||
|
"generic-array",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "byteorder"
|
||||||
|
version = "1.5.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "bzip2"
|
||||||
|
version = "0.6.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f3a53fac24f34a81bc9954b5d6cfce0c21e18ec6959f44f56e8e90e4bb7c346c"
|
||||||
|
dependencies = [
|
||||||
|
"libbz2-rs-sys",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cc"
|
||||||
|
version = "1.5.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
|
||||||
|
dependencies = [
|
||||||
|
"find-msvc-tools",
|
||||||
|
"jobserver",
|
||||||
|
"libc",
|
||||||
|
"shlex",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cfg-if"
|
||||||
|
version = "1.0.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "clawhdf5-format"
|
||||||
|
version = "2.7.0"
|
||||||
|
dependencies = [
|
||||||
|
"byteorder",
|
||||||
|
"bzip2",
|
||||||
|
"flate2",
|
||||||
|
"libaec-sys",
|
||||||
|
"libc",
|
||||||
|
"lz4_flex",
|
||||||
|
"pco",
|
||||||
|
"portable-atomic",
|
||||||
|
"ruzstd",
|
||||||
|
"sha2",
|
||||||
|
"snap",
|
||||||
|
"zstd",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "conformance-probe"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"clawhdf5-format",
|
||||||
|
"serde_json",
|
||||||
|
"sha2",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cpufeatures"
|
||||||
|
version = "0.2.17"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280"
|
||||||
|
dependencies = [
|
||||||
|
"libc",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crc32fast"
|
||||||
|
version = "1.5.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crunchy"
|
||||||
|
version = "0.2.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crypto-common"
|
||||||
|
version = "0.1.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
|
||||||
|
dependencies = [
|
||||||
|
"generic-array",
|
||||||
|
"typenum",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "digest"
|
||||||
|
version = "0.10.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
||||||
|
dependencies = [
|
||||||
|
"block-buffer",
|
||||||
|
"crypto-common",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "dtype_dispatch"
|
||||||
|
version = "0.2.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ab23e69df104e2fd85ee63a533a22d2132ef5975dc6b36f9f3e5a7305e4a8ed7"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "find-msvc-tools"
|
||||||
|
version = "0.1.14"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "flate2"
|
||||||
|
version = "1.1.10"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb"
|
||||||
|
dependencies = [
|
||||||
|
"crc32fast",
|
||||||
|
"miniz_oxide",
|
||||||
|
"zlib-rs",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "generic-array"
|
||||||
|
version = "0.14.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
|
||||||
|
dependencies = [
|
||||||
|
"typenum",
|
||||||
|
"version_check",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "getrandom"
|
||||||
|
version = "0.4.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"libc",
|
||||||
|
"r-efi",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "half"
|
||||||
|
version = "2.7.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"crunchy",
|
||||||
|
"zerocopy",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "itoa"
|
||||||
|
version = "1.0.18"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "jobserver"
|
||||||
|
version = "0.1.35"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3"
|
||||||
|
dependencies = [
|
||||||
|
"getrandom",
|
||||||
|
"libc",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libaec-sys"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libbz2-rs-sys"
|
||||||
|
version = "0.2.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "34b357333733e8260735ba5894eb928c02ecc69c78715f01a8019e7fa7f2db4c"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libc"
|
||||||
|
version = "0.2.189"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "lz4_flex"
|
||||||
|
version = "0.11.6"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a"
|
||||||
|
dependencies = [
|
||||||
|
"twox-hash",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "memchr"
|
||||||
|
version = "2.8.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "miniz_oxide"
|
||||||
|
version = "0.9.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c"
|
||||||
|
dependencies = [
|
||||||
|
"adler2",
|
||||||
|
"simd-adler32",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pco"
|
||||||
|
version = "1.0.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "386342cad4c6e97f081568e5d910ea7d871314c843aa8fc564f2a6b64cab9456"
|
||||||
|
dependencies = [
|
||||||
|
"better_io",
|
||||||
|
"dtype_dispatch",
|
||||||
|
"half",
|
||||||
|
"rand_xoshiro",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pkg-config"
|
||||||
|
version = "0.3.34"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "portable-atomic"
|
||||||
|
version = "1.15.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "proc-macro2"
|
||||||
|
version = "1.0.107"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
|
||||||
|
dependencies = [
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "quote"
|
||||||
|
version = "1.0.47"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "r-efi"
|
||||||
|
version = "6.0.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rand_core"
|
||||||
|
version = "0.6.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rand_xoshiro"
|
||||||
|
version = "0.6.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6f97cdb2a36ed4183de61b2f824cc45c9f1037f28afe0a322e9fff4c108b5aaa"
|
||||||
|
dependencies = [
|
||||||
|
"rand_core",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "ruzstd"
|
||||||
|
version = "0.9.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "a252f5e20f038fe7b4ea53e073e65398d652c864cc162fc77c56c2f13717b888"
|
||||||
|
dependencies = [
|
||||||
|
"twox-hash",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||||
|
dependencies = [
|
||||||
|
"serde_core",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_core"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||||
|
dependencies = [
|
||||||
|
"serde_derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_derive"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn 3.0.6",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_json"
|
||||||
|
version = "1.0.151"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||||
|
dependencies = [
|
||||||
|
"itoa",
|
||||||
|
"memchr",
|
||||||
|
"serde",
|
||||||
|
"serde_core",
|
||||||
|
"zmij",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "sha2"
|
||||||
|
version = "0.10.9"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"cpufeatures",
|
||||||
|
"digest",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "shlex"
|
||||||
|
version = "2.0.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "simd-adler32"
|
||||||
|
version = "0.3.10"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "snap"
|
||||||
|
version = "1.1.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "2.0.119"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "3.0.6"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "twox-hash"
|
||||||
|
version = "2.1.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "typenum"
|
||||||
|
version = "1.20.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "unicode-ident"
|
||||||
|
version = "1.0.26"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "version_check"
|
||||||
|
version = "0.9.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zerocopy"
|
||||||
|
version = "0.8.59"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6df92bf3d9227be3d53173901ddbffac2babc27ae50f397776ffd6dc33f800cb"
|
||||||
|
dependencies = [
|
||||||
|
"zerocopy-derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zerocopy-derive"
|
||||||
|
version = "0.8.59"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ac4f328cf2f05d084e496c3e9c3f33ed0a183656a16e1fcec4d464d8373aec82"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn 2.0.119",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zlib-rs"
|
||||||
|
version = "0.6.8"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zmij"
|
||||||
|
version = "1.0.23"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd"
|
||||||
|
version = "0.13.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-safe",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-safe"
|
||||||
|
version = "7.3.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-sys",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-sys"
|
||||||
|
version = "2.1.0+zstd.1.5.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0"
|
||||||
|
dependencies = [
|
||||||
|
"cc",
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
[package]
|
||||||
|
name = "conformance-probe"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition = "2024"
|
||||||
|
rust-version = "1.92"
|
||||||
|
publish = false
|
||||||
|
description = "Walks an HDF5 file with clawhdf5-format and prints a canonical JSON description (see conformance/README.md)"
|
||||||
|
|
||||||
|
# Deliberately outside the main workspace: `cargo test --workspace` never
|
||||||
|
# builds it, and it links the optional C codecs (zstd, libaec) that the core
|
||||||
|
# crates' default build must not.
|
||||||
|
[workspace]
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-format = { path = "../../crates/clawhdf5-format", features = ["lz4", "zstd", "szip", "pcodec", "plugin-filters"] }
|
||||||
|
serde_json = "1"
|
||||||
|
sha2 = "0.10"
|
||||||
|
|
||||||
|
[profile.release]
|
||||||
|
# Keep panics catchable (the probe records them per object) and turn integer
|
||||||
|
# overflow into a reported panic instead of silent wraparound.
|
||||||
|
debug = 1
|
||||||
|
overflow-checks = true
|
||||||
|
debug-assertions = true
|
||||||
|
panic = "unwind"
|
||||||
@@ -0,0 +1,960 @@
|
|||||||
|
//! Conformance probe: walks an HDF5 file with clawhdf5-format (the same calls
|
||||||
|
//! the `clawhdf5` facade makes) and prints a canonical JSON description:
|
||||||
|
//! every hard-linked object (sorted-name DFS, deduplicated by header address),
|
||||||
|
//! and for each dataset / attribute its shape plus the SHA-256 of its values
|
||||||
|
//! in a canonical encoding shared with `ref.py`.
|
||||||
|
//!
|
||||||
|
//! Canonical value encoding (per element, concatenated, row-major):
|
||||||
|
//! int / float / bitfield / enum / time : element bytes, little-endian
|
||||||
|
//! non-IEEE-layout float (e.g. N-Bit) : the IEEE float of the same size it converts to
|
||||||
|
//! int with bit offset / short precision: the full-width integer it converts to
|
||||||
|
//! opaque : raw bytes
|
||||||
|
//! compound : members in declaration order (padding dropped)
|
||||||
|
//! array : base elements row-major
|
||||||
|
//! string (fixed or VL) : b'S' + u32le len + bytes (cut at first NUL, trailing spaces stripped)
|
||||||
|
//! VL sequence : b'V' + u32le count + base elements
|
||||||
|
//! reference : b'R' (payload not compared)
|
||||||
|
//!
|
||||||
|
//! Every object is processed inside catch_unwind; a caught panic is recorded
|
||||||
|
//! with its message, location and the clawhdf5 frames of its backtrace.
|
||||||
|
|
||||||
|
use std::cell::RefCell;
|
||||||
|
use std::collections::HashSet;
|
||||||
|
use std::panic::{self, AssertUnwindSafe};
|
||||||
|
|
||||||
|
use clawhdf5_format::attribute::extract_attributes_full;
|
||||||
|
use clawhdf5_format::data_layout::DataLayout;
|
||||||
|
use clawhdf5_format::data_read;
|
||||||
|
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
||||||
|
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||||
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||||
|
use clawhdf5_format::group_v1::{self, GroupEntry};
|
||||||
|
use clawhdf5_format::group_v2;
|
||||||
|
use clawhdf5_format::message_type::MessageType;
|
||||||
|
use clawhdf5_format::object_header::{ObjectClass, ObjectHeader};
|
||||||
|
use clawhdf5_format::signature;
|
||||||
|
use clawhdf5_format::superblock::Superblock;
|
||||||
|
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
||||||
|
use clawhdf5_format::vl_data::{VlResolver, check_element_size};
|
||||||
|
use serde_json::{Map, Value, json};
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
const MAX_BYTES: u64 = 200 * 1024 * 1024;
|
||||||
|
const MAX_OBJECTS: usize = 200_000;
|
||||||
|
|
||||||
|
thread_local! {
|
||||||
|
static LAST_PANIC: RefCell<Option<String>> = const { RefCell::new(None) };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn install_hook() {
|
||||||
|
panic::set_hook(Box::new(|info| {
|
||||||
|
let msg = if let Some(s) = info.payload().downcast_ref::<&str>() {
|
||||||
|
s.to_string()
|
||||||
|
} else if let Some(s) = info.payload().downcast_ref::<String>() {
|
||||||
|
s.clone()
|
||||||
|
} else {
|
||||||
|
"<non-string panic>".into()
|
||||||
|
};
|
||||||
|
let loc = info
|
||||||
|
.location()
|
||||||
|
.map(|l| format!("{}:{}", l.file(), l.line()))
|
||||||
|
.unwrap_or_default();
|
||||||
|
let bt = std::backtrace::Backtrace::force_capture().to_string();
|
||||||
|
// keep only frames from clawhdf5 code
|
||||||
|
let mut frames = Vec::new();
|
||||||
|
let lines: Vec<&str> = bt.lines().collect();
|
||||||
|
for (i, l) in lines.iter().enumerate() {
|
||||||
|
let t = l.trim();
|
||||||
|
if t.contains("clawhdf5_format::") || t.contains("conformance_probe::") {
|
||||||
|
let at = lines
|
||||||
|
.get(i + 1)
|
||||||
|
.map(|n| n.trim())
|
||||||
|
.filter(|n| n.starts_with("at "))
|
||||||
|
.map(|n| {
|
||||||
|
let n = n.trim_start_matches("at ");
|
||||||
|
match n.find("/crates/") {
|
||||||
|
Some(p) => n[p + 1..].to_string(),
|
||||||
|
None => n.to_string(),
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.unwrap_or_default();
|
||||||
|
let name = t.split_once(": ").map(|x| x.1).unwrap_or(t);
|
||||||
|
frames.push(format!("{name} ({at})"));
|
||||||
|
if frames.len() >= 12 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let full = format!("PANIC: {msg} @ {loc}\n {}", frames.join("\n "));
|
||||||
|
eprintln!("{full}");
|
||||||
|
LAST_PANIC.with(|p| *p.borrow_mut() = Some(full));
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run `f`, turning a panic into Err("PANIC: ...").
|
||||||
|
fn guarded<T>(f: impl FnOnce() -> Result<T, String>) -> Result<T, String> {
|
||||||
|
match panic::catch_unwind(AssertUnwindSafe(f)) {
|
||||||
|
Ok(r) => r,
|
||||||
|
Err(_) => Err(LAST_PANIC
|
||||||
|
.with(|p| p.borrow_mut().take())
|
||||||
|
.unwrap_or_else(|| "PANIC: <unknown>".into())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn e<E: std::fmt::Debug>(x: E) -> String {
|
||||||
|
format!("{x:?}")
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Ctx<'a> {
|
||||||
|
data: &'a [u8],
|
||||||
|
os: u8,
|
||||||
|
ls: u8,
|
||||||
|
base_dir: std::path::PathBuf,
|
||||||
|
/// Resolves variable-length elements as the library does (null
|
||||||
|
/// elements, strings cut at a NUL, heap objects of the wrong size
|
||||||
|
/// refused), caching each heap collection.
|
||||||
|
vl: RefCell<VlResolver<'a>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'a> Ctx<'a> {
|
||||||
|
fn header(&self, addr: u64) -> Result<ObjectHeader, String> {
|
||||||
|
ObjectHeader::parse(self.data, addr as usize, self.os, self.ls).map_err(e)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result<Option<Vec<u8>>, String> {
|
||||||
|
match h.messages.iter().find(|m| m.msg_type == t) {
|
||||||
|
None => Ok(None),
|
||||||
|
Some(m) => {
|
||||||
|
clawhdf5_format::shared_message::message_data(self.data, m, self.os, self.ls)
|
||||||
|
.map(|c| Some(c.into_owned()))
|
||||||
|
.map_err(e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon(&self, dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let size = dt.type_size() as usize;
|
||||||
|
if b.len() < size {
|
||||||
|
return Err(format!(
|
||||||
|
"canon: element slice {} < type size {size}",
|
||||||
|
b.len()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
match dt {
|
||||||
|
Datatype::FloatingPoint { .. } if !ieee_layout(dt) => {
|
||||||
|
canon_custom_float(dt, &b[..size], out)?
|
||||||
|
}
|
||||||
|
Datatype::FixedPoint { .. } if partial_int(dt) => {
|
||||||
|
canon_partial_int(dt, &b[..size], out)?
|
||||||
|
}
|
||||||
|
Datatype::FixedPoint { byte_order, .. }
|
||||||
|
| Datatype::BitField { byte_order, .. }
|
||||||
|
| Datatype::FloatingPoint { byte_order, .. } => match byte_order {
|
||||||
|
DatatypeByteOrder::LittleEndian => out.extend_from_slice(&b[..size]),
|
||||||
|
DatatypeByteOrder::BigEndian => out.extend(b[..size].iter().rev()),
|
||||||
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||||
|
},
|
||||||
|
Datatype::Time { .. } | Datatype::Opaque { .. } => out.extend_from_slice(&b[..size]),
|
||||||
|
Datatype::String { .. } => canon_str(&b[..size], out),
|
||||||
|
Datatype::Compound { members, .. } => {
|
||||||
|
for m in members {
|
||||||
|
let off = m.byte_offset as usize;
|
||||||
|
let ms = m.datatype.type_size() as usize;
|
||||||
|
if off.checked_add(ms).is_none_or(|end| end > size) {
|
||||||
|
return Err(format!("canon: member {} out of bounds", m.name));
|
||||||
|
}
|
||||||
|
self.canon(&m.datatype, &b[off..off + ms], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Datatype::Reference { .. } => out.push(b'R'),
|
||||||
|
Datatype::Enumeration { base_type, .. } => self.canon(base_type, b, out)?,
|
||||||
|
Datatype::Array {
|
||||||
|
base_type,
|
||||||
|
dimensions,
|
||||||
|
} => {
|
||||||
|
let n: usize = dimensions.iter().map(|d| *d as usize).product();
|
||||||
|
let bs = base_type.type_size() as usize;
|
||||||
|
for i in 0..n {
|
||||||
|
self.canon(base_type, &b[i * bs..], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Datatype::VariableLength {
|
||||||
|
size: vl_size,
|
||||||
|
is_string,
|
||||||
|
base_type,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
check_element_size(*vl_size, self.os).map_err(e)?;
|
||||||
|
let el = &b[..size];
|
||||||
|
if *is_string {
|
||||||
|
let s = self.vl.borrow_mut().string_bytes(el).map_err(e)?;
|
||||||
|
canon_str(&s[0], out);
|
||||||
|
} else {
|
||||||
|
let bs = base_type.type_size() as usize;
|
||||||
|
// The borrow ends here: the base type may itself be
|
||||||
|
// variable-length.
|
||||||
|
let seq = self.vl.borrow_mut().sequences(el, bs).map_err(e)?;
|
||||||
|
let seq = &seq[0];
|
||||||
|
let len = seq.len() / bs;
|
||||||
|
out.push(b'V');
|
||||||
|
out.extend_from_slice(&(len as u32).to_le_bytes());
|
||||||
|
for i in 0..len {
|
||||||
|
self.canon(base_type, &seq[i * bs..], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns (shape json, n_elements)
|
||||||
|
fn shape(ds: &Dataspace) -> (Value, u64) {
|
||||||
|
match ds.space_type {
|
||||||
|
DataspaceType::Null => (Value::String("null".into()), 0),
|
||||||
|
DataspaceType::Scalar => (json!([]), 1),
|
||||||
|
DataspaceType::Simple => {
|
||||||
|
let n = ds.dimensions.iter().fold(1u64, |a, d| a.saturating_mul(*d));
|
||||||
|
(json!(ds.dimensions), n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hash_values(
|
||||||
|
&self,
|
||||||
|
dt: &Datatype,
|
||||||
|
raw: &[u8],
|
||||||
|
n: u64,
|
||||||
|
rec: &mut Map<String, Value>,
|
||||||
|
) -> Result<(), String> {
|
||||||
|
let size = dt.type_size() as usize;
|
||||||
|
let need = (n as usize).checked_mul(size).ok_or("n*size overflow")?;
|
||||||
|
if raw.len() != need {
|
||||||
|
return Err(format!(
|
||||||
|
"raw length {} != n_elements {n} * type_size {size}",
|
||||||
|
raw.len()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let mut canon = Vec::with_capacity(need);
|
||||||
|
for i in 0..n as usize {
|
||||||
|
self.canon(dt, &raw[i * size..(i + 1) * size], &mut canon)?;
|
||||||
|
}
|
||||||
|
let h = Sha256::digest(&canon);
|
||||||
|
rec.insert("hash".into(), Value::String(hex(&h)));
|
||||||
|
rec.insert(
|
||||||
|
"head".into(),
|
||||||
|
Value::String(hex(&canon[..canon.len().min(48)])),
|
||||||
|
);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// VDS source files resolve next to the virtual file; like the library,
|
||||||
|
/// refuse absolute paths and `..`.
|
||||||
|
fn vds_resolver(
|
||||||
|
&self,
|
||||||
|
) -> impl Fn(&str) -> Result<Option<Vec<u8>>, clawhdf5_format::error::FormatError> + use<> {
|
||||||
|
let base = self.base_dir.clone();
|
||||||
|
move |name: &str| {
|
||||||
|
use clawhdf5_format::error::FormatError;
|
||||||
|
let p = std::path::Path::new(name);
|
||||||
|
if p.is_absolute()
|
||||||
|
|| p.components()
|
||||||
|
.any(|c| matches!(c, std::path::Component::ParentDir))
|
||||||
|
{
|
||||||
|
return Err(FormatError::ChunkedReadError(format!("refused {name}")));
|
||||||
|
}
|
||||||
|
match std::fs::read(base.join(p)) {
|
||||||
|
Ok(b) => Ok(Some(b)),
|
||||||
|
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
||||||
|
Err(err) => Err(FormatError::ChunkedReadError(err.to_string())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_named_datatype(&self, h: &ObjectHeader) -> Result<(), String> {
|
||||||
|
let dtb = self
|
||||||
|
.payload(h, MessageType::Datatype)?
|
||||||
|
.ok_or("MissingMessage(Datatype)")?;
|
||||||
|
Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map<String, Value>) -> Result<(), String> {
|
||||||
|
let dtb = self
|
||||||
|
.payload(h, MessageType::Datatype)?
|
||||||
|
.ok_or("MissingMessage(Datatype)")?;
|
||||||
|
let (dt, _) = Datatype::parse_in_header(&dtb, h.version).map_err(e)?;
|
||||||
|
rec.insert("dtype".into(), Value::String(dtype_str(&dt)));
|
||||||
|
let dsb = self
|
||||||
|
.payload(h, MessageType::Dataspace)?
|
||||||
|
.ok_or("MissingMessage(Dataspace)")?;
|
||||||
|
let mut ds = Dataspace::parse(&dsb, self.ls).map_err(e)?;
|
||||||
|
// A virtual dataset's extent can come from its sources (unlimited /
|
||||||
|
// printf mappings), as h5py reports it, rather than the stored one.
|
||||||
|
if let Some(lm) = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
&& let Ok(dl @ DataLayout::Virtual { .. }) =
|
||||||
|
DataLayout::parse(&lm.data, self.os, self.ls)
|
||||||
|
{
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
Some(&resolver),
|
||||||
|
)
|
||||||
|
.map_err(e)?;
|
||||||
|
}
|
||||||
|
let lm = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
.ok_or("MissingMessage(DataLayout)")?;
|
||||||
|
let dl = DataLayout::parse(&lm.data, self.os, self.ls).map_err(e)?;
|
||||||
|
// What libhdf5 checks when it opens the dataset (as File::dataset).
|
||||||
|
data_read::check_dataset_storage(&dl, &ds, &dt, self.data.len() as u64).map_err(e)?;
|
||||||
|
let (shape, n) = Self::shape(&ds);
|
||||||
|
rec.insert("shape".into(), shape);
|
||||||
|
if n.saturating_mul(dt.type_size() as u64) > MAX_BYTES {
|
||||||
|
rec.insert("skipped".into(), Value::String("too large".into()));
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
rec.insert(
|
||||||
|
"layout".into(),
|
||||||
|
Value::String(
|
||||||
|
match &dl {
|
||||||
|
DataLayout::Compact { .. } => "compact",
|
||||||
|
DataLayout::Contiguous { .. } => "contiguous",
|
||||||
|
DataLayout::Chunked { .. } => "chunked",
|
||||||
|
DataLayout::Virtual { .. } => "virtual",
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
),
|
||||||
|
);
|
||||||
|
let pipeline = match self.payload(h, MessageType::FilterPipeline)? {
|
||||||
|
Some(p) => Some(FilterPipeline::parse(&p).map_err(e)?),
|
||||||
|
None => None,
|
||||||
|
};
|
||||||
|
if let Some(p) = &pipeline {
|
||||||
|
rec.insert(
|
||||||
|
"filters".into(),
|
||||||
|
json!(p.filters.iter().map(|f| f.filter_id).collect::<Vec<_>>()),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let raw = if matches!(dl, DataLayout::Virtual { .. }) {
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||||
|
self.data,
|
||||||
|
&h.messages,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
)
|
||||||
|
.map_err(e)?;
|
||||||
|
clawhdf5_format::vds::read_virtual_dataset(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
fill.as_deref(),
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
Some(&resolver),
|
||||||
|
)
|
||||||
|
.map_err(e)?
|
||||||
|
.data
|
||||||
|
} else {
|
||||||
|
let cache = clawhdf5_format::chunk_cache::ChunkCache::new();
|
||||||
|
clawhdf5_format::fill_value::read_full_with_fill::<clawhdf5_format::error::FormatError>(
|
||||||
|
&h.messages,
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
dt.type_size() as usize,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
|| {
|
||||||
|
data_read::read_raw_data_cached(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
pipeline.as_ref(),
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
&cache,
|
||||||
|
)
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.map_err(e)?
|
||||||
|
};
|
||||||
|
self.hash_values(&dt, &raw, n, rec)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn attrs(&self, h: &ObjectHeader) -> Result<Map<String, Value>, String> {
|
||||||
|
let msgs = extract_attributes_full(self.data, h, self.os, self.ls).map_err(e)?;
|
||||||
|
let mut out = Map::new();
|
||||||
|
for a in &msgs {
|
||||||
|
let r = guarded(|| {
|
||||||
|
let mut rec = Map::new();
|
||||||
|
rec.insert("dtype".into(), Value::String(dtype_str(&a.datatype)));
|
||||||
|
let (shape, n) = Self::shape(&a.dataspace);
|
||||||
|
rec.insert("shape".into(), shape);
|
||||||
|
self.hash_values(&a.datatype, &a.raw_data, n, &mut rec)?;
|
||||||
|
Ok(rec)
|
||||||
|
});
|
||||||
|
let v = match r {
|
||||||
|
Ok(rec) => Value::Object(rec),
|
||||||
|
Err(msg) => json!({ "error": msg }),
|
||||||
|
};
|
||||||
|
out.insert(a.name.clone(), v);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entries(&self, h: &ObjectHeader) -> Result<Vec<GroupEntry>, String> {
|
||||||
|
let v1 = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::SymbolTable);
|
||||||
|
if let Some(m) = v1 {
|
||||||
|
let stm = SymbolTableMessage::parse(&m.data, self.os).map_err(e)?;
|
||||||
|
group_v1::resolve_v1_group_entries(self.data, &stm, self.os, self.ls).map_err(e)
|
||||||
|
} else if h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link)
|
||||||
|
{
|
||||||
|
group_v2::resolve_v2_group_entries(self.data, h, self.os, self.ls).map_err(e)
|
||||||
|
} else {
|
||||||
|
Ok(Vec::new())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Element bytes as an unsigned integer (at most 16 bytes), honouring byte order.
|
||||||
|
fn element_bits(b: &[u8], byte_order: &DatatypeByteOrder) -> Result<u128, String> {
|
||||||
|
if b.len() > 16 {
|
||||||
|
return Err(format!("canon: {}-byte numeric element", b.len()));
|
||||||
|
}
|
||||||
|
let mut v = 0u128;
|
||||||
|
match byte_order {
|
||||||
|
DatatypeByteOrder::LittleEndian => {
|
||||||
|
for (i, x) in b.iter().enumerate() {
|
||||||
|
v |= u128::from(*x) << (8 * i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
DatatypeByteOrder::BigEndian => {
|
||||||
|
for x in b {
|
||||||
|
v = (v << 8) | u128::from(*x);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||||
|
}
|
||||||
|
Ok(v)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn field(v: u128, pos: u32, len: u32) -> u128 {
|
||||||
|
if len == 0 || pos >= 128 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let v = v >> pos;
|
||||||
|
if len >= 128 {
|
||||||
|
v
|
||||||
|
} else {
|
||||||
|
v & ((1u128 << len) - 1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True when a float's bit fields are exactly IEEE 754 binary16/32/64 for its
|
||||||
|
/// size. h5py hands back such a type's bytes untouched; any other layout (an
|
||||||
|
/// N-Bit `H5Tset_precision` float, say) is *converted* by libhdf5 into the
|
||||||
|
/// numpy float of the same size, so comparing raw bytes would be meaningless.
|
||||||
|
fn ieee_layout(dt: &Datatype) -> bool {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
..
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
return true;
|
||||||
|
};
|
||||||
|
let std = match size {
|
||||||
|
2 => (16, 10, 5, 10, 15),
|
||||||
|
4 => (32, 23, 8, 23, 127),
|
||||||
|
8 => (64, 52, 11, 52, 1023),
|
||||||
|
_ => return true, // no same-size numpy float to convert to: compare raw
|
||||||
|
};
|
||||||
|
*bit_offset == 0
|
||||||
|
&& (
|
||||||
|
*bit_precision,
|
||||||
|
*exponent_location,
|
||||||
|
*exponent_size,
|
||||||
|
*mantissa_size,
|
||||||
|
*exponent_bias,
|
||||||
|
) == (std.0, std.1, std.2, std.3, std.4)
|
||||||
|
&& *mantissa_location == 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Canonicalise a non-IEEE-layout float the way libhdf5's float->float
|
||||||
|
/// conversion presents it to h5py: as the IEEE float of the same size.
|
||||||
|
/// Assumes the implied-leading-one normalisation and the sign bit at the top
|
||||||
|
/// of the precision (what `H5Tset_precision` produces; the parser does not
|
||||||
|
/// keep either field).
|
||||||
|
fn canon_custom_float(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
unreachable!()
|
||||||
|
};
|
||||||
|
let (esize, msize) = (u32::from(*exponent_size), u32::from(*mantissa_size));
|
||||||
|
if esize == 0 || esize > 30 || msize > 64 {
|
||||||
|
return Err(format!("canon: unsupported float layout e{esize} m{msize}"));
|
||||||
|
}
|
||||||
|
let v = element_bits(b, byte_order)?;
|
||||||
|
let sign_pos = (u32::from(*bit_offset) + u32::from(*bit_precision)).saturating_sub(1);
|
||||||
|
let neg = field(v, sign_pos, 1) == 1;
|
||||||
|
let e = field(v, u32::from(*exponent_location), esize) as i64;
|
||||||
|
let m = field(v, u32::from(*mantissa_location), msize);
|
||||||
|
let emax = (1i64 << esize) - 1;
|
||||||
|
let bias = i64::from(*exponent_bias);
|
||||||
|
let mag = if e == emax {
|
||||||
|
if m == 0 { f64::INFINITY } else { f64::NAN }
|
||||||
|
} else if e == 0 {
|
||||||
|
(m as f64) * 2f64.powi((1 - bias - msize as i64) as i32)
|
||||||
|
} else {
|
||||||
|
((1u128 << msize) as f64 + m as f64) * 2f64.powi((e - bias - msize as i64) as i32)
|
||||||
|
};
|
||||||
|
let x = if neg { -mag } else { mag };
|
||||||
|
match size {
|
||||||
|
2 => out
|
||||||
|
.extend_from_slice(&clawhdf5_format::float16::f32_to_f16_bits(x as f32).to_le_bytes()),
|
||||||
|
4 => out.extend_from_slice(&(x as f32).to_le_bytes()),
|
||||||
|
8 => out.extend_from_slice(&x.to_le_bytes()),
|
||||||
|
_ => unreachable!("ieee_layout keeps other sizes raw"),
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Integers stored with a bit offset or reduced precision (N-Bit): libhdf5
|
||||||
|
/// converts them to the full-width integer of the same size, shifting the
|
||||||
|
/// value down and sign-extending from the top precision bit.
|
||||||
|
fn canon_partial_int(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
signed,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
unreachable!()
|
||||||
|
};
|
||||||
|
let prec = u32::from(*bit_precision);
|
||||||
|
let v = element_bits(b, byte_order)?;
|
||||||
|
let mut x = field(v, u32::from(*bit_offset), prec);
|
||||||
|
if *signed && prec > 0 && prec < 128 && field(x, prec - 1, 1) == 1 {
|
||||||
|
x |= !0u128 << prec;
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&x.to_le_bytes()[..*size as usize]);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn partial_int(dt: &Datatype) -> bool {
|
||||||
|
matches!(dt, Datatype::FixedPoint { size, bit_offset, bit_precision, .. }
|
||||||
|
if *bit_offset != 0 || u32::from(*bit_precision) != size * 8)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon_str(b: &[u8], out: &mut Vec<u8>) {
|
||||||
|
let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len());
|
||||||
|
let mut s = &b[..cut];
|
||||||
|
while let [rest @ .., b' '] = s {
|
||||||
|
s = rest;
|
||||||
|
}
|
||||||
|
out.push(b'S');
|
||||||
|
out.extend_from_slice(&(s.len() as u32).to_le_bytes());
|
||||||
|
out.extend_from_slice(s);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hex(b: &[u8]) -> String {
|
||||||
|
b.iter().map(|x| format!("{x:02x}")).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dtype_str(dt: &Datatype) -> String {
|
||||||
|
match dt {
|
||||||
|
Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
signed,
|
||||||
|
byte_order,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
format!(
|
||||||
|
"{}{}{}",
|
||||||
|
bo(byte_order),
|
||||||
|
if *signed { "i" } else { "u" },
|
||||||
|
size
|
||||||
|
)
|
||||||
|
}
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
size, byte_order, ..
|
||||||
|
} => format!("{}f{}", bo(byte_order), size),
|
||||||
|
Datatype::BitField {
|
||||||
|
size, byte_order, ..
|
||||||
|
} => format!("{}b{}", bo(byte_order), size),
|
||||||
|
Datatype::Time { size, .. } => format!("time{size}"),
|
||||||
|
Datatype::String { size, .. } => format!("S{size}"),
|
||||||
|
Datatype::Opaque { size, .. } => format!("V{size}"),
|
||||||
|
Datatype::Compound { size, members } => format!(
|
||||||
|
"{{{}}}{size}",
|
||||||
|
members
|
||||||
|
.iter()
|
||||||
|
.map(|m| format!("{}:{}", m.name, dtype_str(&m.datatype)))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(",")
|
||||||
|
),
|
||||||
|
Datatype::Reference { ref_type, .. } => format!("ref({ref_type:?})"),
|
||||||
|
Datatype::Enumeration { base_type, .. } => format!("enum({})", dtype_str(base_type)),
|
||||||
|
Datatype::VariableLength {
|
||||||
|
is_string: true, ..
|
||||||
|
} => "vlstr".into(),
|
||||||
|
Datatype::VariableLength { base_type, .. } => format!("vlen({})", dtype_str(base_type)),
|
||||||
|
Datatype::Array {
|
||||||
|
base_type,
|
||||||
|
dimensions,
|
||||||
|
} => format!("({}){dimensions:?}", dtype_str(base_type)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bo(b: &DatatypeByteOrder) -> &'static str {
|
||||||
|
match b {
|
||||||
|
DatatypeByteOrder::LittleEndian => "<",
|
||||||
|
DatatypeByteOrder::BigEndian => ">",
|
||||||
|
DatatypeByteOrder::Vax => "vax",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn is_group(h: &ObjectHeader) -> bool {
|
||||||
|
h.messages.iter().any(|m| {
|
||||||
|
matches!(
|
||||||
|
m.msg_type,
|
||||||
|
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The probe's kind for an object header: libhdf5's object class
|
||||||
|
/// ([`ObjectHeader::object_class`]: group, then dataset — a datatype *and* a
|
||||||
|
/// dataspace — then named datatype), which is what h5py opens the object as.
|
||||||
|
/// The root group, and a header with only link messages, count as groups.
|
||||||
|
fn kind_of(h: &ObjectHeader, is_root: bool) -> &'static str {
|
||||||
|
match h.object_class() {
|
||||||
|
Some(ObjectClass::Group) => "group",
|
||||||
|
Some(ObjectClass::Dataset) => "dataset",
|
||||||
|
_ if is_root || is_group(h) => "group",
|
||||||
|
Some(ObjectClass::NamedDatatype) => "datatype",
|
||||||
|
None => "unknown",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
install_hook();
|
||||||
|
let path = std::env::args().nth(1).expect("usage: probe <file>");
|
||||||
|
let mut top = Map::new();
|
||||||
|
top.insert("file".into(), Value::String(path.clone()));
|
||||||
|
let data = match std::fs::read(&path) {
|
||||||
|
Ok(d) => d,
|
||||||
|
Err(err) => {
|
||||||
|
top.insert("open_error".into(), Value::String(format!("Io({err})")));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// Every address is relative to the superblock: look at the file from
|
||||||
|
// there on (past any user block), as libhdf5 does.
|
||||||
|
let hdf5: &[u8] = match signature::find_signature(&data) {
|
||||||
|
Ok(off) => &data[off..],
|
||||||
|
Err(_) => &data,
|
||||||
|
};
|
||||||
|
let sb = guarded(|| Superblock::parse(hdf5, 0).map_err(e));
|
||||||
|
let sb = match sb {
|
||||||
|
Ok(sb) => sb,
|
||||||
|
Err(msg) => {
|
||||||
|
top.insert("open_error".into(), Value::String(msg));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// libhdf5 refuses a truncated file and reads nothing past the recorded
|
||||||
|
// end of file.
|
||||||
|
let base = (data.len() - hdf5.len()) as u64;
|
||||||
|
let hdf5 = match sb.data_end(base, data.len() as u64) {
|
||||||
|
Ok(end) => &hdf5[..end as usize],
|
||||||
|
Err(err) => {
|
||||||
|
top.insert("open_error".into(), Value::String(e(err)));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// libhdf5 decodes the superblock extension at open (an error refuses
|
||||||
|
// the file), and loads a metadata cache image over the file's own
|
||||||
|
// metadata. It loads the image only when it first reads metadata — the
|
||||||
|
// root group — so a file whose image it cannot load still opens and
|
||||||
|
// that read fails. The library decides all three cases with the same
|
||||||
|
// `cache_image_state`: `File` and `MmapFile` open such a file and fail
|
||||||
|
// every object lookup with the image's error, which is what the probe
|
||||||
|
// records here (on the root group, where libhdf5 reports it).
|
||||||
|
use clawhdf5_format::superblock_ext::{self, CacheImageState};
|
||||||
|
let state = match guarded(|| superblock_ext::cache_image_state(hdf5, &sb).map_err(e)) {
|
||||||
|
Ok(x) => x,
|
||||||
|
Err(msg) => {
|
||||||
|
top.insert("open_error".into(), Value::String(msg));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let mut image_error = None;
|
||||||
|
let view = match state {
|
||||||
|
CacheImageState::Absent => None,
|
||||||
|
CacheImageState::Unloadable(err) => {
|
||||||
|
image_error = Some(e(err));
|
||||||
|
None
|
||||||
|
}
|
||||||
|
CacheImageState::Loaded(image) => {
|
||||||
|
let mut v = hdf5.to_vec();
|
||||||
|
match image.block(hdf5).and_then(|b| image.apply(b, &mut v)) {
|
||||||
|
Ok(()) => Some(v),
|
||||||
|
Err(err) => {
|
||||||
|
image_error = Some(e(err));
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let hdf5: &[u8] = view.as_deref().unwrap_or(hdf5);
|
||||||
|
top.insert("superblock_version".into(), json!(sb.version));
|
||||||
|
let ctx = Ctx {
|
||||||
|
data: hdf5,
|
||||||
|
os: sb.offset_size,
|
||||||
|
ls: sb.length_size,
|
||||||
|
base_dir: std::path::Path::new(&path)
|
||||||
|
.parent()
|
||||||
|
.map(|p| p.to_path_buf())
|
||||||
|
.unwrap_or_default(),
|
||||||
|
vl: RefCell::new(VlResolver::new(hdf5, sb.offset_size, sb.length_size)),
|
||||||
|
};
|
||||||
|
let mut objects: Vec<Value> = Vec::new();
|
||||||
|
let mut visited = HashSet::new();
|
||||||
|
let mut soft_v1 = 0u64;
|
||||||
|
// explicit DFS stack: (address, path)
|
||||||
|
let mut stack: Vec<(u64, String)> = vec![(sb.root_group_address, "/".to_string())];
|
||||||
|
while let Some((addr, p)) = stack.pop() {
|
||||||
|
if objects.len() >= MAX_OBJECTS {
|
||||||
|
top.insert("truncated".into(), json!(true));
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if !visited.insert(addr) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mut rec = Map::new();
|
||||||
|
rec.insert("path".into(), Value::String(p.clone()));
|
||||||
|
let r = guarded(|| {
|
||||||
|
if let Some(msg) = &image_error {
|
||||||
|
return Err(msg.clone());
|
||||||
|
}
|
||||||
|
let h = ctx.header(addr)?;
|
||||||
|
Ok(h)
|
||||||
|
});
|
||||||
|
let h = match r {
|
||||||
|
Ok(h) => h,
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("kind".into(), Value::String("unknown".into()));
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
objects.push(Value::Object(rec));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let kind = kind_of(&h, addr == sb.root_group_address);
|
||||||
|
rec.insert("kind".into(), Value::String(kind.into()));
|
||||||
|
if kind == "dataset"
|
||||||
|
&& let Err(msg) = guarded(|| ctx.read_dataset(&h, &mut rec))
|
||||||
|
{
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
// Opening a committed datatype decodes it (h5py's `f[name]` fails on
|
||||||
|
// one libhdf5 cannot decode), so decode it here too.
|
||||||
|
if kind == "datatype"
|
||||||
|
&& let Err(msg) = guarded(|| ctx.read_named_datatype(&h))
|
||||||
|
{
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
if kind != "datatype" {
|
||||||
|
match guarded(|| ctx.attrs(&h)) {
|
||||||
|
Ok(m) => {
|
||||||
|
rec.insert("attrs".into(), Value::Object(m));
|
||||||
|
}
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("attrs_error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if kind == "group" {
|
||||||
|
match guarded(|| ctx.entries(&h)) {
|
||||||
|
Ok(mut ents) => {
|
||||||
|
ents.retain(|en| {
|
||||||
|
if en.cache_type == 2 {
|
||||||
|
soft_v1 += 1;
|
||||||
|
false
|
||||||
|
} else {
|
||||||
|
true
|
||||||
|
}
|
||||||
|
});
|
||||||
|
ents.sort_by(|a, b| a.name.cmp(&b.name));
|
||||||
|
let base = if p == "/" { String::new() } else { p.clone() };
|
||||||
|
for en in ents.into_iter().rev() {
|
||||||
|
stack.push((en.object_header_address, format!("{base}/{}", en.name)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("list_error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
objects.push(Value::Object(rec));
|
||||||
|
}
|
||||||
|
if soft_v1 > 0 {
|
||||||
|
top.insert("v1_soft_link_entries".into(), json!(soft_v1));
|
||||||
|
}
|
||||||
|
top.insert("objects".into(), Value::Array(objects));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The N-Bit float of libhdf5's `test/testfiles/le_data.h5`
|
||||||
|
/// (`Nbit_float_data_le`): offset 7, precision 20, sign bit 26, exponent
|
||||||
|
/// 20+6 (bias 31), mantissa 7+13.
|
||||||
|
fn nbit_f32(byte_order: DatatypeByteOrder) -> Datatype {
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order,
|
||||||
|
bit_offset: 7,
|
||||||
|
bit_precision: 20,
|
||||||
|
exponent_location: 20,
|
||||||
|
exponent_size: 6,
|
||||||
|
mantissa_location: 7,
|
||||||
|
mantissa_size: 13,
|
||||||
|
exponent_bias: 31,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon_one(dt: &Datatype, bytes: &[u8]) -> Vec<u8> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
canon_custom_float(dt, bytes, &mut out).unwrap();
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn nbit_float_canonicalises_to_the_value_libhdf5_returns() {
|
||||||
|
let le = nbit_f32(DatatypeByteOrder::LittleEndian);
|
||||||
|
let be = nbit_f32(DatatypeByteOrder::BigEndian);
|
||||||
|
assert!(!ieee_layout(&le));
|
||||||
|
// 1.0: exponent = bias, mantissa 0
|
||||||
|
let one: u32 = 31 << 20;
|
||||||
|
assert_eq!(canon_one(&le, &one.to_le_bytes()), 1.0f32.to_le_bytes());
|
||||||
|
assert_eq!(canon_one(&be, &one.to_be_bytes()), 1.0f32.to_le_bytes());
|
||||||
|
// -2.1999512 (h5py's reading of the file's -2.2): sign, e = 32, m = 819
|
||||||
|
let v: u32 = (1 << 26) | (32 << 20) | (819 << 7);
|
||||||
|
assert_eq!(
|
||||||
|
canon_one(&le, &v.to_le_bytes()),
|
||||||
|
(-2.199_951_2f32).to_le_bytes()
|
||||||
|
);
|
||||||
|
assert_eq!(canon_one(&le, &[0; 4]), 0.0f32.to_le_bytes());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn ieee_floats_keep_their_raw_bytes() {
|
||||||
|
let f32le = Datatype::FloatingPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 32,
|
||||||
|
exponent_location: 23,
|
||||||
|
exponent_size: 8,
|
||||||
|
mantissa_location: 0,
|
||||||
|
mantissa_size: 23,
|
||||||
|
exponent_bias: 127,
|
||||||
|
};
|
||||||
|
assert!(ieee_layout(&f32le));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn kind_follows_libhdf5_object_class() {
|
||||||
|
use clawhdf5_format::object_header::HeaderMessage;
|
||||||
|
let header = |types: &[MessageType]| ObjectHeader {
|
||||||
|
version: 2,
|
||||||
|
messages: types
|
||||||
|
.iter()
|
||||||
|
.map(|&msg_type| HeaderMessage {
|
||||||
|
msg_type,
|
||||||
|
size: 0,
|
||||||
|
flags: 0,
|
||||||
|
creation_order: None,
|
||||||
|
data: Vec::new(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
reference_count: None,
|
||||||
|
flags: 0,
|
||||||
|
access_time: None,
|
||||||
|
modification_time: None,
|
||||||
|
change_time: None,
|
||||||
|
birth_time: None,
|
||||||
|
};
|
||||||
|
use MessageType::*;
|
||||||
|
// cve-2024-33874 `/Dset1`: a datatype and a layout but no dataspace
|
||||||
|
// is a named datatype to libhdf5 (h5py opens it as one).
|
||||||
|
assert_eq!(kind_of(&header(&[Datatype, DataLayout]), false), "datatype");
|
||||||
|
assert_eq!(
|
||||||
|
kind_of(&header(&[Datatype, Dataspace, DataLayout]), false),
|
||||||
|
"dataset"
|
||||||
|
);
|
||||||
|
assert_eq!(kind_of(&header(&[SymbolTable]), false), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[Link]), false), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[]), true), "group");
|
||||||
|
assert_eq!(kind_of(&header(&[]), false), "unknown");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn partial_precision_int_is_shifted_and_sign_extended() {
|
||||||
|
let dt = Datatype::FixedPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order: DatatypeByteOrder::BigEndian,
|
||||||
|
signed: true,
|
||||||
|
bit_offset: 4,
|
||||||
|
bit_precision: 17,
|
||||||
|
};
|
||||||
|
assert!(partial_int(&dt));
|
||||||
|
let stored = (((-5i32) as u32) & 0x1_FFFF) << 4;
|
||||||
|
let mut out = Vec::new();
|
||||||
|
canon_partial_int(&dt, &stored.to_be_bytes(), &mut out).unwrap();
|
||||||
|
assert_eq!(out, (-5i32).to_le_bytes());
|
||||||
|
}
|
||||||
|
}
|
||||||
Executable
+325
@@ -0,0 +1,325 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Reference probe: same JSON as the Rust `conformance-probe`, produced with h5py.
|
||||||
|
|
||||||
|
Walk: iterative DFS from '/', children in sorted (UTF-8 byte) name order, hard
|
||||||
|
links only, each object once (first path wins, deduplicated by object identity).
|
||||||
|
Canonical value encoding: see harness/src/main.rs.
|
||||||
|
"""
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import struct
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import h5py
|
||||||
|
|
||||||
|
try:
|
||||||
|
import hdf5plugin # noqa: F401 registers blosc/lz4/zstd/bzip2/... filters
|
||||||
|
except Exception: # pragma: no cover
|
||||||
|
pass
|
||||||
|
|
||||||
|
MAX_BYTES = 200 * 1024 * 1024
|
||||||
|
MAX_OBJECTS = 200_000
|
||||||
|
|
||||||
|
|
||||||
|
def canon_str(b, out):
|
||||||
|
if isinstance(b, str):
|
||||||
|
b = b.encode("utf-8", "surrogateescape")
|
||||||
|
b = bytes(b)
|
||||||
|
cut = b.find(b"\x00")
|
||||||
|
if cut >= 0:
|
||||||
|
b = b[:cut]
|
||||||
|
b = b.rstrip(b" ")
|
||||||
|
out += b"S" + struct.pack("<I", len(b)) + b
|
||||||
|
|
||||||
|
|
||||||
|
def simple(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return all(simple(dt.fields[n][0]) for n in dt.names)
|
||||||
|
if dt.subdtype:
|
||||||
|
return simple(dt.subdtype[0])
|
||||||
|
return dt.kind in "iufcbV"
|
||||||
|
|
||||||
|
|
||||||
|
def packed(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return np.dtype([(n, packed(dt.fields[n][0])) for n in dt.names])
|
||||||
|
if dt.subdtype:
|
||||||
|
base, shape = dt.subdtype
|
||||||
|
return np.dtype((packed(base), shape))
|
||||||
|
if dt.kind in "iufcb":
|
||||||
|
return dt.newbyteorder("<")
|
||||||
|
return dt
|
||||||
|
|
||||||
|
|
||||||
|
# --- reference corrections ---------------------------------------------------
|
||||||
|
# Where h5py is known to return values the file does not hold, and the right
|
||||||
|
# values follow from what it returned, ref.py corrects them and records the
|
||||||
|
# correction on the object ("ref_fix"), so the comparison is still a real
|
||||||
|
# comparison and CONFORMANCE.md lists every corrected object. Each correction
|
||||||
|
# first checks that the installed h5py still has the bug.
|
||||||
|
|
||||||
|
# Corrections applied while encoding the current object.
|
||||||
|
FIXES = set()
|
||||||
|
_BE_VLEN_BUG = None
|
||||||
|
|
||||||
|
|
||||||
|
def be_vlen_bug():
|
||||||
|
"""h5py (3.16 / HDF5 2.0 at least) returns the elements of a
|
||||||
|
variable-length sequence whose base type is big-endian with the file's
|
||||||
|
big-endian bytes under a native (little-endian) dtype: a
|
||||||
|
`vlen_dtype('>f4')` dataset holding [1.0, 2.0] reads back as
|
||||||
|
[4.6e-41, 9.0e-44]. `h5dump` prints the file's values. Checked once per
|
||||||
|
process by writing and reading exactly that dataset in memory."""
|
||||||
|
global _BE_VLEN_BUG
|
||||||
|
if _BE_VLEN_BUG is None:
|
||||||
|
import io
|
||||||
|
try:
|
||||||
|
bio = io.BytesIO()
|
||||||
|
with h5py.File(bio, "w") as f:
|
||||||
|
d = f.create_dataset("v", (1,), dtype=h5py.vlen_dtype(np.dtype(">f4")))
|
||||||
|
d[0] = np.array([1.0, 2.0], dtype=">f4")
|
||||||
|
with h5py.File(bio, "r") as f:
|
||||||
|
got = np.asarray(f["v"][0])
|
||||||
|
_BE_VLEN_BUG = (got.dtype == np.dtype("<f4")
|
||||||
|
and got.view(">f4").tolist() == [1.0, 2.0]
|
||||||
|
and got.tolist() != [1.0, 2.0])
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
_BE_VLEN_BUG = False
|
||||||
|
return _BE_VLEN_BUG
|
||||||
|
|
||||||
|
|
||||||
|
def unswapped(got, base):
|
||||||
|
"""`got` is `base` (big-endian somewhere) with every field in native
|
||||||
|
little-endian order instead: the shape of h5py's big-endian VL bug."""
|
||||||
|
return base.newbyteorder("<") == got and base != got
|
||||||
|
|
||||||
|
|
||||||
|
def canon_el(dt, val, out):
|
||||||
|
if dt.fields:
|
||||||
|
for n in dt.names:
|
||||||
|
canon_el(dt.fields[n][0], val[n], out)
|
||||||
|
return
|
||||||
|
if dt.subdtype:
|
||||||
|
base, _ = dt.subdtype
|
||||||
|
for x in np.asarray(val).reshape(-1):
|
||||||
|
canon_el(base, x, out)
|
||||||
|
return
|
||||||
|
k = dt.kind
|
||||||
|
if k in "iufcb":
|
||||||
|
out += np.asarray(val, dtype=dt).astype(dt.newbyteorder("<")).tobytes()
|
||||||
|
elif k == "V":
|
||||||
|
out += np.asarray(val, dtype=dt).tobytes()
|
||||||
|
elif k == "S":
|
||||||
|
canon_str(val, out)
|
||||||
|
elif k == "O":
|
||||||
|
if h5py.check_string_dtype(dt) is not None:
|
||||||
|
canon_str(val if val is not None else b"", out)
|
||||||
|
elif h5py.check_ref_dtype(dt) is not None:
|
||||||
|
out += b"R"
|
||||||
|
else:
|
||||||
|
base = h5py.check_vlen_dtype(dt)
|
||||||
|
if base is None:
|
||||||
|
raise TypeError(f"unhandled object dtype {dt!r}")
|
||||||
|
arr = np.asarray(val if val is not None else [])
|
||||||
|
if arr.dtype != base and be_vlen_bug() and unswapped(arr.dtype, base):
|
||||||
|
# h5py's big-endian VL bug (see be_vlen_bug): the bytes are
|
||||||
|
# the file's, the dtype label is wrong. Relabel, don't convert.
|
||||||
|
arr = arr.view(base)
|
||||||
|
FIXES.add("h5py-be-vlen")
|
||||||
|
arr = np.asarray(arr, dtype=base).reshape(-1)
|
||||||
|
out += b"V" + struct.pack("<I", arr.shape[0])
|
||||||
|
if simple(base):
|
||||||
|
out += arr.astype(packed(base)).tobytes()
|
||||||
|
else:
|
||||||
|
for x in arr:
|
||||||
|
canon_el(base, x, out)
|
||||||
|
elif k == "U":
|
||||||
|
canon_str(str(val), out)
|
||||||
|
else:
|
||||||
|
raise TypeError(f"unhandled dtype kind {k} ({dt!r})")
|
||||||
|
|
||||||
|
|
||||||
|
def has_obj(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return any(has_obj(dt.fields[n][0]) for n in dt.names)
|
||||||
|
if dt.subdtype:
|
||||||
|
return has_obj(dt.subdtype[0])
|
||||||
|
return dt.kind == "O"
|
||||||
|
|
||||||
|
|
||||||
|
def note_conversion(tid, dt, rec):
|
||||||
|
"""h5py converts some file types (FP8, bfloat16, x87 long double, ...) to a
|
||||||
|
different-sized numpy type; then value bytes are not comparable."""
|
||||||
|
try:
|
||||||
|
if not has_obj(dt) and tid.get_size() != dt.itemsize:
|
||||||
|
rec["converted"] = f"file type size {tid.get_size()} -> numpy {dt} ({dt.itemsize})"
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def hash_values(arr, dt, rec):
|
||||||
|
# h5py expands an HDF5 array element type into trailing array dims, a
|
||||||
|
# nested array type (an array of arrays) into all of them. Converting the
|
||||||
|
# expanded array back to the inner subarray type would broadcast every
|
||||||
|
# element into a whole subarray, so strip every level.
|
||||||
|
while dt.subdtype is not None:
|
||||||
|
dt = dt.subdtype[0]
|
||||||
|
arr = np.asarray(arr, dtype=dt)
|
||||||
|
FIXES.clear()
|
||||||
|
if simple(dt):
|
||||||
|
c = np.ascontiguousarray(arr).astype(packed(dt)).tobytes()
|
||||||
|
else:
|
||||||
|
out = bytearray()
|
||||||
|
for x in arr.reshape(-1):
|
||||||
|
canon_el(dt, x, out)
|
||||||
|
c = bytes(out)
|
||||||
|
if FIXES:
|
||||||
|
rec["ref_fix"] = sorted(FIXES)
|
||||||
|
rec["hash"] = hashlib.sha256(c).hexdigest()
|
||||||
|
rec["head"] = c[:48].hex()
|
||||||
|
|
||||||
|
|
||||||
|
def err(e):
|
||||||
|
s = f"{type(e).__name__}: {e}"
|
||||||
|
return s.splitlines()[0][:400] if s else type(e).__name__
|
||||||
|
|
||||||
|
|
||||||
|
def shape_of(s):
|
||||||
|
return "null" if s is None else list(s)
|
||||||
|
|
||||||
|
|
||||||
|
def n_bytes(shape, tid):
|
||||||
|
n = 1
|
||||||
|
for d in shape or ():
|
||||||
|
n *= d
|
||||||
|
return n * tid.get_size()
|
||||||
|
|
||||||
|
|
||||||
|
def read_attrs(obj):
|
||||||
|
out = {}
|
||||||
|
names = sorted(obj.attrs.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||||
|
for name in names:
|
||||||
|
rec = {}
|
||||||
|
try:
|
||||||
|
aid = obj.attrs.get_id(name)
|
||||||
|
rec["dtype"] = str(aid.dtype)
|
||||||
|
rec["shape"] = shape_of(aid.shape)
|
||||||
|
note_conversion(aid.get_type(), aid.dtype, rec)
|
||||||
|
if aid.shape is None:
|
||||||
|
hash_values(np.empty((0,), dtype=aid.dtype), aid.dtype, rec)
|
||||||
|
else:
|
||||||
|
val = obj.attrs[name]
|
||||||
|
hash_values(val, aid.dtype, rec)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec = {"error": err(e)}
|
||||||
|
out[name] = rec
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def main(path):
|
||||||
|
top = {"file": path}
|
||||||
|
try:
|
||||||
|
f = h5py.File(path, "r")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
top["open_error"] = err(e)
|
||||||
|
print(json.dumps(top))
|
||||||
|
return
|
||||||
|
objects = []
|
||||||
|
seen = set()
|
||||||
|
# Objects h5py cannot open have no ObjectID to deduplicate by; they are
|
||||||
|
# deduplicated by the address their hard link points at instead, as the
|
||||||
|
# probe deduplicates every object by header address.
|
||||||
|
seen_unopenable = set()
|
||||||
|
stack = [("/", None, None)]
|
||||||
|
while stack:
|
||||||
|
p, obj, link_addr = stack.pop()
|
||||||
|
if len(objects) >= MAX_OBJECTS:
|
||||||
|
top["truncated"] = True
|
||||||
|
break
|
||||||
|
rec = {"path": p}
|
||||||
|
try:
|
||||||
|
if obj is None:
|
||||||
|
obj = f[p]
|
||||||
|
key = hash(obj.id) # h5py ObjectID hash = (fileno, object address/token)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
if link_addr is not None:
|
||||||
|
if link_addr in seen_unopenable:
|
||||||
|
continue
|
||||||
|
seen_unopenable.add(link_addr)
|
||||||
|
rec["kind"] = "unknown"
|
||||||
|
rec["error"] = err(e)
|
||||||
|
objects.append(rec)
|
||||||
|
continue
|
||||||
|
if key in seen:
|
||||||
|
continue
|
||||||
|
seen.add(key)
|
||||||
|
if isinstance(obj, h5py.Dataset):
|
||||||
|
kind = "dataset"
|
||||||
|
elif isinstance(obj, h5py.Group):
|
||||||
|
kind = "group"
|
||||||
|
elif isinstance(obj, h5py.Datatype):
|
||||||
|
kind = "datatype"
|
||||||
|
else:
|
||||||
|
kind = "unknown"
|
||||||
|
rec["kind"] = kind
|
||||||
|
if kind == "dataset":
|
||||||
|
try:
|
||||||
|
dt = obj.dtype
|
||||||
|
rec["dtype"] = str(dt)
|
||||||
|
rec["shape"] = shape_of(obj.shape)
|
||||||
|
note_conversion(obj.id.get_type(), dt, rec)
|
||||||
|
if obj.shape is None:
|
||||||
|
hash_values(np.empty((0,), dtype=dt), dt, rec)
|
||||||
|
elif n_bytes(obj.shape, obj.id.get_type()) > MAX_BYTES:
|
||||||
|
rec["skipped"] = "too large"
|
||||||
|
else:
|
||||||
|
arr = np.empty(obj.shape, dtype=dt)
|
||||||
|
if arr.size:
|
||||||
|
try:
|
||||||
|
obj.read_direct(arr)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
arr = obj[()]
|
||||||
|
hash_values(arr, dt, rec)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["error"] = err(e)
|
||||||
|
if kind != "datatype":
|
||||||
|
try:
|
||||||
|
rec["attrs"] = read_attrs(obj)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["attrs_error"] = err(e)
|
||||||
|
if kind == "group":
|
||||||
|
try:
|
||||||
|
names = sorted(obj.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||||
|
base = "" if p == "/" else p
|
||||||
|
kids = []
|
||||||
|
for n in names:
|
||||||
|
# The link's own type: `obj.get(n, getlink=True)` reports
|
||||||
|
# a user-defined link (type 64-255) as a HardLink.
|
||||||
|
try:
|
||||||
|
info = obj.id.links.get_info(n.encode("utf-8", "surrogateescape"))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
info = None
|
||||||
|
if info is not None and info.type != h5py.h5l.TYPE_HARD:
|
||||||
|
continue
|
||||||
|
addr = info.u if info is not None else None
|
||||||
|
kids.append((f"{base}/{n}", addr))
|
||||||
|
for k, addr in reversed(kids):
|
||||||
|
stack.append((k, None, addr))
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["list_error"] = err(e)
|
||||||
|
objects.append(rec)
|
||||||
|
top["objects"] = objects
|
||||||
|
print(json.dumps(top), flush=True)
|
||||||
|
# Exit without tearing down the h5py objects: freeing them for some files
|
||||||
|
# that hold references (hdf5's h5repack_attr_refs.h5, cve-2024-32623.h5)
|
||||||
|
# makes libhdf5 2.0 abort with "free(): chunks in smallbin corrupted"
|
||||||
|
# about half the time. That happens after the reading is done, so it says
|
||||||
|
# nothing about what h5py read, but it flipped those files between ok and
|
||||||
|
# h5py-cannot-read from one run to the next.
|
||||||
|
os._exit(0)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1])
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""ref_bugs.py <corpus_dir>: re-check the objects h5py reads only through a
|
||||||
|
libhdf5 bug.
|
||||||
|
|
||||||
|
For each object of READ_BUGS (below), h5py reads it in several fresh
|
||||||
|
processes whose heaps differ: h5py imported before numpy (three runs, plus
|
||||||
|
two with glibc's MALLOC_PERTURB_, which fills newly allocated and freed heap
|
||||||
|
blocks with a byte pattern) and numpy imported first. Values the file
|
||||||
|
determines come out the same every time. An object whose values differ
|
||||||
|
between those runs is read from memory the file does not determine — an
|
||||||
|
over-read or an uninitialised buffer in libhdf5 — so the values h5py reports
|
||||||
|
for it are not the file's, and clawhdf5 refusing the object is not a
|
||||||
|
clawhdf5 error. compare.py classifies a file as `ref-bug` only on objects
|
||||||
|
confirmed that way in the same run (`$OUT/ref_bugs.json`); an object whose
|
||||||
|
reading turns out stable stays an our-error.
|
||||||
|
|
||||||
|
Run by conformance/run.sh; on its own it is the reproducer (JSON on stdout).
|
||||||
|
"""
|
||||||
|
import concurrent.futures
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
# (file, object) -> what goes wrong. Checked 2026-09-27 against HDF5 2.0.0
|
||||||
|
# (h5py 3.16), h5dump 1.14.6 and the HDFGroup/hdf5 sources (tag hdf5_1_14_6
|
||||||
|
# and develop); see docs/known-issues.md, "Conformance: the last non-ok files".
|
||||||
|
READ_BUGS = {
|
||||||
|
("cve_hdf5/cvefiles/cve-2025-2308.h5", "/Scale_offset_long_long_data_le"):
|
||||||
|
"the first chunk records minbits 11: its 12 values need 17 bytes of codes, and the "
|
||||||
|
"26-byte chunk holds 5 after its 21-byte header; libhdf5's scale-offset decoder reads "
|
||||||
|
"past its buffer, and develop refuses the chunk (\"Buffer too short\")",
|
||||||
|
("cve_hdf5/cvefiles/cve-2025-44904.h5", "/Scale_offset_float_data_le"):
|
||||||
|
"unfiltered chunks stored as 38 and 37 bytes for 48-byte chunks: 1.14/2.0 read the "
|
||||||
|
"stored bytes into a buffer of that size and use it as the whole chunk "
|
||||||
|
"(H5D__chunk_lock), so the rest is heap memory; develop refuses them (\"incorrect chunk "
|
||||||
|
"size returned from index for unfiltered chunk\")",
|
||||||
|
("hdf5/test/testfiles/bad_nbit_parms_walk.h5", "/Nbit_int_data_le"):
|
||||||
|
"the N-Bit parameter list holds 7 values (cd_values[0] = 7) where an integer needs 8: "
|
||||||
|
"the decoder takes the bit offset from cd_values[7], past the list; libhdf5's own test "
|
||||||
|
"(`test_filter_bad_params`, test/dsets.c on develop) requires the read to fail",
|
||||||
|
}
|
||||||
|
|
||||||
|
# (which module is imported first, MALLOC_PERTURB_)
|
||||||
|
RUNS = [("h5py", None), ("h5py", None), ("h5py", None), ("h5py", "170"), ("h5py", "255"),
|
||||||
|
("numpy", None)]
|
||||||
|
|
||||||
|
READ = r"""
|
||||||
|
import hashlib, sys
|
||||||
|
if sys.argv[3] == "h5py":
|
||||||
|
import h5py, numpy as np
|
||||||
|
else:
|
||||||
|
import numpy as np, h5py
|
||||||
|
try:
|
||||||
|
import hdf5plugin # noqa: F401
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
with h5py.File(sys.argv[1], "r") as f:
|
||||||
|
a = np.ascontiguousarray(f[sys.argv[2]][()])
|
||||||
|
print("values " + hashlib.sha256(a.tobytes()).hexdigest()[:16])
|
||||||
|
except Exception as e:
|
||||||
|
print("error " + (str(e).splitlines() or [type(e).__name__])[0][:120])
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def read_once(path, obj, first, perturb):
|
||||||
|
env = dict(os.environ)
|
||||||
|
env.pop("MALLOC_PERTURB_", None)
|
||||||
|
if perturb:
|
||||||
|
env["MALLOC_PERTURB_"] = perturb
|
||||||
|
try:
|
||||||
|
p = subprocess.run([sys.executable, "-c", READ, path, obj, first], env=env,
|
||||||
|
capture_output=True, text=True, timeout=60)
|
||||||
|
out = p.stdout.strip().splitlines()
|
||||||
|
return out[-1] if out else f"exit {p.returncode}"
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
return "timeout"
|
||||||
|
|
||||||
|
|
||||||
|
def check(corpus, key):
|
||||||
|
f, obj = key
|
||||||
|
path = os.path.join(corpus, f)
|
||||||
|
rec = {"file": f, "object": obj, "why": READ_BUGS[key]}
|
||||||
|
if not os.path.exists(path):
|
||||||
|
return rec | {"missing": True, "confirmed": False}
|
||||||
|
runs = [{"first": a, "malloc_perturb": p, "outcome": read_once(path, obj, a, p)} for a, p in RUNS]
|
||||||
|
distinct = sorted({r["outcome"] for r in runs})
|
||||||
|
return rec | {
|
||||||
|
"runs": runs,
|
||||||
|
"distinct": len(distinct),
|
||||||
|
"confirmed": len(distinct) > 1 and any(o.startswith("values ") for o in distinct),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
corpus = sys.argv[1]
|
||||||
|
keys = list(READ_BUGS)
|
||||||
|
with concurrent.futures.ThreadPoolExecutor(max_workers=len(keys)) as ex:
|
||||||
|
out = list(ex.map(lambda k: check(corpus, k), keys))
|
||||||
|
print(json.dumps({"read_bugs": out}, indent=1))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,361 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""report.py <results_dir> <CONFORMANCE.md> <corpus_dir>
|
||||||
|
|
||||||
|
Render the sweep's results (compare.py's results.json plus the raw per-side
|
||||||
|
runs) as CONFORMANCE.md, and write <results_dir>/report-meta.json (commit,
|
||||||
|
date, versions) for check.py --update.
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import datetime
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy
|
||||||
|
|
||||||
|
try:
|
||||||
|
import hdf5plugin
|
||||||
|
HDF5PLUGIN = hdf5plugin.version
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
HDF5PLUGIN = "not installed"
|
||||||
|
|
||||||
|
R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||||
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
ROOT = os.path.dirname(HERE)
|
||||||
|
CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "ref-bug", "panic", "hang", "crash", "oom"]
|
||||||
|
|
||||||
|
|
||||||
|
def sh(*cmd, cwd=ROOT):
|
||||||
|
try:
|
||||||
|
return subprocess.run(cmd, cwd=cwd, capture_output=True, text=True, timeout=30).stdout.strip()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def cpu_model():
|
||||||
|
try:
|
||||||
|
for ln in open("/proc/cpuinfo"):
|
||||||
|
if ln.startswith(("model name", "Model")):
|
||||||
|
return ln.split(":", 1)[1].strip()
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return platform.processor() or "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def mem_gib():
|
||||||
|
try:
|
||||||
|
for ln in open("/proc/meminfo"):
|
||||||
|
if ln.startswith("MemTotal:"):
|
||||||
|
return f"{int(ln.split()[1]) / 1048576:.0f} GiB"
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return "?"
|
||||||
|
|
||||||
|
|
||||||
|
res = json.load(open(os.path.join(R, "results.json")))
|
||||||
|
meta_run = json.load(open(os.path.join(R, "meta.json"))) if os.path.exists(os.path.join(R, "meta.json")) else {}
|
||||||
|
rows = res["rows"]
|
||||||
|
issues = res.get("issues", {})
|
||||||
|
|
||||||
|
# safe.directory: a checkout owned by another user (a container) is still ours to read
|
||||||
|
commit = sh("git", "-c", "safe.directory=*", "rev-parse", "HEAD") or os.environ.get("GITHUB_SHA", "unknown")
|
||||||
|
lib_dirty = sh("git", "-c", "safe.directory=*", "status", "--porcelain", "--", "crates", "Cargo.toml")
|
||||||
|
h5dump_v = sh("h5dump", "--version").replace("h5dump: ", "")
|
||||||
|
meta = {
|
||||||
|
"date": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC"),
|
||||||
|
"commit": commit + (" (library sources modified)" if lib_dirty else ""),
|
||||||
|
"reference": f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}",
|
||||||
|
}
|
||||||
|
json.dump(meta, open(os.path.join(R, "report-meta.json"), "w"), indent=1)
|
||||||
|
|
||||||
|
pins = []
|
||||||
|
for ln in open(os.path.join(HERE, "corpus.txt")):
|
||||||
|
if ln.strip() and not ln.lstrip().startswith("#"):
|
||||||
|
name, url, rev, root, *_ = ln.split()
|
||||||
|
pins.append((name, url, rev, root))
|
||||||
|
|
||||||
|
by_corpus = collections.defaultdict(collections.Counter)
|
||||||
|
for r in rows:
|
||||||
|
by_corpus[r["corpus"]][r["class"]] += 1
|
||||||
|
total = collections.Counter(r["class"] for r in rows)
|
||||||
|
|
||||||
|
|
||||||
|
def ex_list(files, n=3):
|
||||||
|
s = ", ".join(f"`{f}`" for f in files[:n])
|
||||||
|
return s + (f" (+{len(files) - n} more)" if len(files) > n else "")
|
||||||
|
|
||||||
|
|
||||||
|
# --- reference bugs ---------------------------------------------------------
|
||||||
|
# ref_bugs.py's re-check of the objects h5py reads only through a libhdf5 bug
|
||||||
|
# (compare.py classifies on the confirmed ones), and the objects whose h5py
|
||||||
|
# values ref.py corrected (compare.py's ref_fixes).
|
||||||
|
try:
|
||||||
|
ref_bugs = json.load(open(os.path.join(R, "ref_bugs.json")))["read_bugs"]
|
||||||
|
except (OSError, ValueError, KeyError):
|
||||||
|
ref_bugs = []
|
||||||
|
ref_fixes = res.get("ref_fixes", [])
|
||||||
|
|
||||||
|
|
||||||
|
# --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------
|
||||||
|
def side(run, name):
|
||||||
|
p = os.path.join(R, "runs", run, name)
|
||||||
|
if not os.path.exists(p + ".rc"):
|
||||||
|
return None
|
||||||
|
rc = int(open(p + ".rc").read().strip() or -1)
|
||||||
|
err = open(p + ".err", errors="replace").read()
|
||||||
|
try:
|
||||||
|
j = json.load(open(p + ".json"))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
j = None
|
||||||
|
return rc, err, j
|
||||||
|
|
||||||
|
|
||||||
|
def outcome(s, rust=False):
|
||||||
|
"""-> (bucket, text). bucket in read / error / panic / crash / hang / oom."""
|
||||||
|
if s is None:
|
||||||
|
return "missing", "not run"
|
||||||
|
rc, err, j = s
|
||||||
|
if rc in (137, 124):
|
||||||
|
return "hang", "hang (killed at timeout)"
|
||||||
|
if "memory allocation of" in err or "MemoryError" in err or "bad_alloc" in err or "Cannot allocate" in err:
|
||||||
|
return "oom", "out of memory"
|
||||||
|
if rust and (rc == 101 or "PANIC:" in err):
|
||||||
|
return "panic", "panic"
|
||||||
|
if "overflowed its stack" in err:
|
||||||
|
return "crash", "stack overflow"
|
||||||
|
if rc == 139:
|
||||||
|
return "crash", "SIGSEGV"
|
||||||
|
if rc == 134:
|
||||||
|
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||||
|
if rc > 128:
|
||||||
|
return "crash", f"signal {rc - 128}"
|
||||||
|
if j is None:
|
||||||
|
return ("error", "error exit") if rc in (0, 1) else ("crash", f"exit {rc}")
|
||||||
|
if "open_error" in j:
|
||||||
|
return "error", "open error"
|
||||||
|
objs = j.get("objects", [])
|
||||||
|
ne = sum(1 for o in objs for k in ("error", "attrs_error", "list_error") if k in o)
|
||||||
|
ne += sum(1 for o in objs for a in (o.get("attrs") or {}).values() if "error" in a)
|
||||||
|
return "read", f"read {len(objs)} obj" + (f", {ne} errors" if ne else "")
|
||||||
|
|
||||||
|
|
||||||
|
def h5dump_outcome(s):
|
||||||
|
if s is None:
|
||||||
|
return "missing", "not run"
|
||||||
|
rc, err, _ = s
|
||||||
|
if rc in (137, 124):
|
||||||
|
return "hang", "hang (killed at timeout)"
|
||||||
|
if "memory allocation" in err or "Cannot allocate" in err:
|
||||||
|
return "oom", "out of memory"
|
||||||
|
if rc == 139:
|
||||||
|
return "crash", "SIGSEGV"
|
||||||
|
if rc == 134:
|
||||||
|
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||||
|
if rc > 128:
|
||||||
|
return "crash", f"signal {rc - 128}"
|
||||||
|
return ("read", "ok") if rc == 0 else ("error", "error exit")
|
||||||
|
|
||||||
|
|
||||||
|
cve_rows = []
|
||||||
|
buckets = {"clawhdf5": collections.Counter(), "h5dump": collections.Counter(), "h5py": collections.Counter()}
|
||||||
|
ours_panic = {r["file"] for r in rows if r["class"] == "panic"}
|
||||||
|
for r in rows:
|
||||||
|
if r["corpus"] != "cve_hdf5":
|
||||||
|
continue
|
||||||
|
run = r["file"].replace("/", "__")
|
||||||
|
o = outcome(side(run, "ours"), rust=True)
|
||||||
|
if o[0] == "read" and r["file"] in ours_panic:
|
||||||
|
o = ("panic", "caught panic")
|
||||||
|
p = outcome(side(run, "ref"))
|
||||||
|
d = h5dump_outcome(side(run, "h5dump"))
|
||||||
|
buckets["clawhdf5"][o[0]] += 1
|
||||||
|
buckets["h5py"][p[0]] += 1
|
||||||
|
buckets["h5dump"][d[0]] += 1
|
||||||
|
cve_rows.append((r["file"].split("/", 1)[1], d[1], p[1], o[1], r["class"]))
|
||||||
|
|
||||||
|
# --- render -----------------------------------------------------------------
|
||||||
|
L = []
|
||||||
|
w = L.append
|
||||||
|
w("# clawhdf5 conformance report")
|
||||||
|
w("")
|
||||||
|
w("Every HDF5 file of eight public corpora (pinned by commit) is read twice — by")
|
||||||
|
w("clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade")
|
||||||
|
w("makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are")
|
||||||
|
w("compared object by object: the set of hard-linked objects, each dataset's and")
|
||||||
|
w("attribute's shape, and a SHA-256 of its values in a canonical encoding. The")
|
||||||
|
w("CVE corpus is also run through `h5dump`. Each side runs under a timeout and an")
|
||||||
|
w("address-space limit, so a hang, crash or runaway allocation is recorded, not")
|
||||||
|
w("fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.")
|
||||||
|
w("")
|
||||||
|
w("## Run")
|
||||||
|
w("")
|
||||||
|
w("| | |")
|
||||||
|
w("|---|---|")
|
||||||
|
w(f"| date | {meta['date']} |")
|
||||||
|
w(f"| clawhdf5 commit | `{meta['commit']}` |")
|
||||||
|
w(f"| machine | `{platform.node()}`: {cpu_model()}, {os.cpu_count()} CPUs, {mem_gib()}, {platform.system()} {platform.release()} {platform.machine()} |")
|
||||||
|
w(f"| command | `{os.environ.get('CONFORMANCE_CMD', 'conformance/run.sh')}` |")
|
||||||
|
w(f"| rustc | {sh('rustc', '-V')} |")
|
||||||
|
w(f"| reference | h5py {h5py.__version__}, HDF5 {h5py.version.hdf5_version}, numpy {numpy.__version__}, hdf5plugin {HDF5PLUGIN}, Python {platform.python_version()} |")
|
||||||
|
w(f"| h5dump | {h5dump_v} (CVE corpus only) |")
|
||||||
|
if meta_run:
|
||||||
|
w(f"| limits | {meta_run.get('timeout_s')} s timeout (SIGKILL), {int(meta_run.get('mem_kb', 0)) // 1024} MiB address space, per process; {meta_run.get('jobs')} files in parallel |")
|
||||||
|
w(f"| runtime | {meta_run.get('probe_seconds')} s probing + comparing ({meta_run.get('build_seconds')} s fetch/build before it) |")
|
||||||
|
w("")
|
||||||
|
w("## Results")
|
||||||
|
w("")
|
||||||
|
w("A file's class is the first that applies:")
|
||||||
|
w("")
|
||||||
|
w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.")
|
||||||
|
w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.")
|
||||||
|
w("- **ref-bug** — every difference is an object clawhdf5 refuses that h5py reads only through a libhdf5 bug: the values h5py returns for it change with the reading process's heap, re-checked in every run (see *Reference bugs*).")
|
||||||
|
w("- **our-error** — clawhdf5 returned an error for something h5py reads.")
|
||||||
|
w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.")
|
||||||
|
w("- **ok** — every object h5py reads, clawhdf5 reads identically.")
|
||||||
|
w("")
|
||||||
|
w("| corpus | files | " + " | ".join(CLASSES) + " |")
|
||||||
|
w("|---" * (len(CLASSES) + 2) + "|")
|
||||||
|
for c in sorted(by_corpus):
|
||||||
|
cnt = by_corpus[c]
|
||||||
|
w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |")
|
||||||
|
w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |")
|
||||||
|
w("")
|
||||||
|
nonok = total.get("our-error", 0) + total.get("mismatch", 0)
|
||||||
|
w(f"**Our errors and mismatches: {nonok}.** Files not ok: "
|
||||||
|
+ (", ".join(f"{total[c]} {c}" for c in CLASSES if c != "ok" and total.get(c)) or "none") + "."
|
||||||
|
+ (f" {len(ref_fixes)} object(s) were compared against h5py's values corrected for a known h5py bug"
|
||||||
|
f" ({sum(1 for x in ref_fixes if x[3])} identical to clawhdf5's; see *Reference bugs*)." if ref_fixes else ""))
|
||||||
|
w("")
|
||||||
|
w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):")
|
||||||
|
w("")
|
||||||
|
w("| corpus | source | commit |")
|
||||||
|
w("|---|---|---|")
|
||||||
|
for name, url, rev, root in pins:
|
||||||
|
w(f"| {name} | {url.removesuffix('.git')}" + ("" if root == "." else f" (`{root}`)") + f" | `{rev[:12]}` |")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Panics, hangs, crashes, out-of-memory")
|
||||||
|
w("")
|
||||||
|
if not res["panics"]:
|
||||||
|
w("None.")
|
||||||
|
else:
|
||||||
|
for p in res["panics"]:
|
||||||
|
w(f"- `{p['file']}` [{p['class']}] {p['detail']}")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Our-error root causes")
|
||||||
|
w("")
|
||||||
|
if res["root_causes"]:
|
||||||
|
w("Grouped by normalised error message. *files* counts files whose class this cause affects.")
|
||||||
|
w("")
|
||||||
|
w("| files | objects | error | examples |")
|
||||||
|
w("|---:|---:|---|---|")
|
||||||
|
for k, v in res["root_causes"].items():
|
||||||
|
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||||
|
else:
|
||||||
|
w("None.")
|
||||||
|
w("")
|
||||||
|
w("## Mismatch root causes")
|
||||||
|
w("")
|
||||||
|
if res["mismatch_causes"]:
|
||||||
|
w("| files | objects | cause | examples |")
|
||||||
|
w("|---:|---:|---|---|")
|
||||||
|
for k, v in res["mismatch_causes"].items():
|
||||||
|
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||||
|
else:
|
||||||
|
w("None.")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## CVE corpus: clawhdf5 vs h5dump vs h5py")
|
||||||
|
w("")
|
||||||
|
w(f"The {len(cve_rows)} files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for")
|
||||||
|
w("published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object")
|
||||||
|
w("errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so")
|
||||||
|
w("its read/error split is not comparable with the other two rows; the panic, crash, hang and oom")
|
||||||
|
w("columns are.")
|
||||||
|
w("")
|
||||||
|
w("| tool | read | error | panic | crash | hang | oom |")
|
||||||
|
w("|---|---:|---:|---:|---:|---:|---:|")
|
||||||
|
for tool, label in (("clawhdf5", "clawhdf5"), ("h5dump", f"h5dump {h5dump_v.split()[-1] if h5dump_v else ''}"),
|
||||||
|
("h5py", f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}")):
|
||||||
|
b = buckets[tool]
|
||||||
|
w(f"| {label} | " + " | ".join(str(b.get(k, 0)) for k in ("read", "error", "panic", "crash", "hang", "oom")) + " |")
|
||||||
|
w("")
|
||||||
|
w("<details><summary>Per-file outcomes</summary>")
|
||||||
|
w("")
|
||||||
|
w("| file | h5dump | h5py | clawhdf5 | class |")
|
||||||
|
w("|---|---|---|---|---|")
|
||||||
|
for f, d, p, o, cls in cve_rows:
|
||||||
|
w(f"| {f} | {d} | {p} | {o} | {cls} |")
|
||||||
|
w("")
|
||||||
|
w("</details>")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Reference bugs")
|
||||||
|
w("")
|
||||||
|
w("### Objects h5py reads only through a libhdf5 bug (*ref-bug*)")
|
||||||
|
w("")
|
||||||
|
w("clawhdf5 refuses these objects; h5py 3.16 / HDF5 2.0 returns values for them. `conformance/ref_bugs.py`")
|
||||||
|
w("re-reads each with h5py in six fresh processes whose heaps differ (h5py imported before numpy, three")
|
||||||
|
w("times and twice more with `MALLOC_PERTURB_`, and numpy imported first). Values the file determines")
|
||||||
|
w("come out the same every time; these do not, so they are memory libhdf5 over-reads, not the file's")
|
||||||
|
w("data. A file is *ref-bug* only while every one of its differences is such an object confirmed in")
|
||||||
|
w("the same run; an object that reads the same every time goes back to *our-error*. Reproducer:")
|
||||||
|
w("`python conformance/ref_bugs.py conformance/.cache/corpus` (prints every read's outcome).")
|
||||||
|
w("")
|
||||||
|
w("| file | object | distinct results in 6 reads | confirmed | what goes wrong |")
|
||||||
|
w("|---|---|---:|---|---|")
|
||||||
|
for b in ref_bugs:
|
||||||
|
n = "missing" if b.get("missing") else b.get("distinct", "?")
|
||||||
|
w(f"| `{b['file']}` | `{b['object']}` | {n} | {'yes' if b.get('confirmed') else '**no**'} | {b['why']} |")
|
||||||
|
w("")
|
||||||
|
w("### Values corrected for a known h5py bug")
|
||||||
|
w("")
|
||||||
|
w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence")
|
||||||
|
w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)")
|
||||||
|
w(" numpy dtype: a `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]` reads back as")
|
||||||
|
w(" `[4.6e-41, 9.0e-44]`; `h5dump` prints the file's values. `ref.py` checks that the installed")
|
||||||
|
w(" h5py still does this (by writing and reading exactly that dataset in memory) and, if so,")
|
||||||
|
w(" relabels such elements with the file's byte order before hashing, so the values are still")
|
||||||
|
w(" compared. Corrected objects: "
|
||||||
|
+ (", ".join(f"`{f}` `{p}` ({'same as clawhdf5' if same else '**differs from clawhdf5**'})"
|
||||||
|
for f, p, _, same in ref_fixes) if ref_fixes else "none") + ".")
|
||||||
|
w("")
|
||||||
|
w("## Other comparison rules")
|
||||||
|
w("")
|
||||||
|
w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose")
|
||||||
|
w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a")
|
||||||
|
w(" bit offset / reduced precision into the plain numpy type of the same size. The probe")
|
||||||
|
w(" compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared")
|
||||||
|
w(" raw bytes, which reported every N-Bit float dataset as a mismatch).")
|
||||||
|
if res["incomparable"]:
|
||||||
|
w("- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size")
|
||||||
|
w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not")
|
||||||
|
w(" compared (shape and presence still are): "
|
||||||
|
+ ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".")
|
||||||
|
w("- **References** are compared by presence only (`R`), not by target.")
|
||||||
|
w("")
|
||||||
|
if res.get("ref_only_errors"):
|
||||||
|
w("## Objects h5py fails on but clawhdf5 reads")
|
||||||
|
w("")
|
||||||
|
for k, n in res["ref_only_errors"][:15]:
|
||||||
|
w(f"- {n} x `{k}`")
|
||||||
|
w("")
|
||||||
|
w("## Reproduce")
|
||||||
|
w("")
|
||||||
|
w("```sh")
|
||||||
|
w("# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git")
|
||||||
|
w("CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh")
|
||||||
|
w("```")
|
||||||
|
w("")
|
||||||
|
w("The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for")
|
||||||
|
w("every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.")
|
||||||
|
w("`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)")
|
||||||
|
w("must keep; `conformance/run.sh --update-baseline` rewrites it.")
|
||||||
|
|
||||||
|
with open(OUT_MD, "w") as fh:
|
||||||
|
fh.write("\n".join(L) + "\n")
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
# The reference side of the conformance sweep. Pinned so the nightly job and a
|
||||||
|
# local run compare against the same libhdf5 (h5py wheels bundle it).
|
||||||
|
h5py==3.16.0
|
||||||
|
numpy==2.5.3
|
||||||
|
hdf5plugin==7.1.0
|
||||||
|
netCDF4==1.7.4
|
||||||
Executable
+90
@@ -0,0 +1,90 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# conformance/run.sh — the clawhdf5 conformance sweep, end to end.
|
||||||
|
#
|
||||||
|
# fetch the pinned corpora (cached) -> build the probe -> probe every file
|
||||||
|
# with clawhdf5 and with h5py (and h5dump for the CVE corpus), each under a
|
||||||
|
# timeout and a memory limit -> compare -> write CONFORMANCE.md -> check the
|
||||||
|
# result against conformance/baseline.json.
|
||||||
|
#
|
||||||
|
# Usage: conformance/run.sh [--no-fetch] [--no-report] [--update-baseline]
|
||||||
|
#
|
||||||
|
# Environment:
|
||||||
|
# CLAWHDF5_PYTHON python with h5py, numpy, hdf5plugin (default: repo .venv, then python3)
|
||||||
|
# CONFORMANCE_CACHE corpus / build / results cache (default: conformance/.cache)
|
||||||
|
# CONFORMANCE_OUT results directory (default: $CONFORMANCE_CACHE/results)
|
||||||
|
# CONFORMANCE_REPORT report path (default: CONFORMANCE.md at the repo root)
|
||||||
|
# JOBS parallel files (default: nproc)
|
||||||
|
# CONFORMANCE_PROBE use this prebuilt probe binary instead of building one
|
||||||
|
# TMO / MEM_KB per-process timeout in seconds (20) / address-space limit in KiB (4 GiB)
|
||||||
|
#
|
||||||
|
# Exit status: 0 = gate passed; 1 = a panic/hang/crash/oom in clawhdf5, or the
|
||||||
|
# ok count fell below the baseline, or a baseline-ok file regressed; 2 = setup error.
|
||||||
|
set -euo pipefail
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
ROOT="$(cd "$HERE/.." && pwd)"
|
||||||
|
FETCH=1 REPORT=1 UPDATE=0
|
||||||
|
for a in "$@"; do
|
||||||
|
case "$a" in
|
||||||
|
--no-fetch) FETCH=0 ;;
|
||||||
|
--no-report) REPORT=0 ;;
|
||||||
|
--update-baseline) UPDATE=1 ;;
|
||||||
|
-h|--help) sed -n '2,23p' "$0"; exit 0 ;;
|
||||||
|
*) echo "unknown argument: $a" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
export PATH="$HOME/.cargo/bin:$PATH"
|
||||||
|
CACHE="${CONFORMANCE_CACHE:-$HERE/.cache}"
|
||||||
|
mkdir -p "$CACHE"; CACHE="$(cd "$CACHE" && pwd)"
|
||||||
|
OUT="${CONFORMANCE_OUT:-$CACHE/results}"
|
||||||
|
REPORT_PATH="${CONFORMANCE_REPORT:-$ROOT/CONFORMANCE.md}"
|
||||||
|
JOBS="${JOBS:-$(nproc 2>/dev/null || echo 4)}"
|
||||||
|
if [ -n "${CLAWHDF5_PYTHON:-}" ]; then PY="$CLAWHDF5_PYTHON"
|
||||||
|
elif [ -x "$ROOT/.venv/bin/python" ]; then PY="$ROOT/.venv/bin/python"
|
||||||
|
else PY="$(command -v python3)"; fi
|
||||||
|
export PY TMO="${TMO:-20}" MEM_KB="${MEM_KB:-4194304}"
|
||||||
|
command -v h5dump >/dev/null || { echo "error: h5dump not found (install hdf5-tools)" >&2; exit 2; }
|
||||||
|
"$PY" -c 'import h5py, numpy, hdf5plugin' || { echo "error: $PY lacks h5py/numpy/hdf5plugin" >&2; exit 2; }
|
||||||
|
|
||||||
|
t0=$(date +%s)
|
||||||
|
[ "$FETCH" = 1 ] && bash "$HERE/fetch-corpus.sh" "$CACHE"
|
||||||
|
C="$CACHE/corpus"
|
||||||
|
[ -d "$C" ] || { echo "error: no corpus in $C (run without --no-fetch)" >&2; exit 2; }
|
||||||
|
|
||||||
|
if [ -n "${CONFORMANCE_PROBE:-}" ]; then
|
||||||
|
export PROBE="$CONFORMANCE_PROBE" # a prebuilt probe, e.g. an older one for a before/after
|
||||||
|
else
|
||||||
|
echo "== building the probe"
|
||||||
|
CARGO_TARGET_DIR="${CARGO_TARGET_DIR:-$CACHE/target}" \
|
||||||
|
cargo build -q --release --manifest-path "$HERE/probe/Cargo.toml"
|
||||||
|
export PROBE="${CARGO_TARGET_DIR:-$CACHE/target}/release/conformance-probe"
|
||||||
|
fi
|
||||||
|
t1=$(date +%s)
|
||||||
|
|
||||||
|
rm -rf "$OUT"; mkdir -p "$OUT"
|
||||||
|
"$PY" "$HERE/list_files.py" "$C" > "$OUT/files.txt"
|
||||||
|
echo "== probing $(wc -l <"$OUT/files.txt") files, $JOBS at a time (timeout ${TMO}s, limit $((MEM_KB / 1024)) MiB)"
|
||||||
|
export C OUT HERE
|
||||||
|
# The shell's "Segmentation fault (core dumped)" notices go to probe.log; the
|
||||||
|
# signals themselves are recorded in each side's .rc.
|
||||||
|
xargs -a "$OUT/files.txt" -d '\n' -P "$JOBS" -I{} bash -c '
|
||||||
|
f="$1"; d="$OUT/runs/${f//\//__}"
|
||||||
|
case "$f" in cve_hdf5/*) export WITH_H5DUMP=1 ;; esac
|
||||||
|
"$HERE/run_one.sh" "$C/$f" "$d"' _ {} 2>"$OUT/probe.log"
|
||||||
|
echo "== re-checking the objects h5py reads only through a libhdf5 bug"
|
||||||
|
"$PY" "$HERE/ref_bugs.py" "$C" > "$OUT/ref_bugs.json" 2> "$OUT/ref_bugs.err" || true
|
||||||
|
echo "== comparing"
|
||||||
|
"$PY" "$HERE/compare.py" "$OUT" >/dev/null
|
||||||
|
t2=$(date +%s)
|
||||||
|
cat > "$OUT/meta.json" <<EOF
|
||||||
|
{"build_seconds": $((t1 - t0)), "probe_seconds": $((t2 - t1)), "jobs": $JOBS, "timeout_s": $TMO, "mem_kb": $MEM_KB}
|
||||||
|
EOF
|
||||||
|
export CONFORMANCE_CMD="${CONFORMANCE_CMD:-conformance/run.sh${*:+ $*}}"
|
||||||
|
if [ "$REPORT" = 1 ]; then
|
||||||
|
"$PY" "$HERE/report.py" "$OUT" "$REPORT_PATH" "$C"
|
||||||
|
echo "== wrote $REPORT_PATH"
|
||||||
|
fi
|
||||||
|
if [ "$UPDATE" = 1 ]; then
|
||||||
|
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json" --update
|
||||||
|
fi
|
||||||
|
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json"
|
||||||
Executable
+27
@@ -0,0 +1,27 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# run_one.sh <file> <outdir>
|
||||||
|
#
|
||||||
|
# Probe one file with clawhdf5 (PROBE) and with h5py (PY ref.py), and with
|
||||||
|
# h5dump too when WITH_H5DUMP is set. Each side runs under a timeout (TMO
|
||||||
|
# seconds, SIGKILL) and an address-space limit (MEM_KB), with core dumps off.
|
||||||
|
# Writes <outdir>/<side>.{json,err,rc}; rc 137 = killed by the timeout.
|
||||||
|
set -u
|
||||||
|
f="$1"; out="$2"; mkdir -p "$out"
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
: "${PROBE:?PROBE must name the conformance-probe binary}"
|
||||||
|
: "${PY:?PY must name a python with h5py}"
|
||||||
|
TMO="${TMO:-20}"
|
||||||
|
MEM_KB="${MEM_KB:-4194304}"
|
||||||
|
run() { # name cmd...
|
||||||
|
local name=$1; shift
|
||||||
|
( ulimit -v "$MEM_KB"; ulimit -c 0; RUST_BACKTRACE=1 exec timeout -s KILL "$TMO" "$@" ) \
|
||||||
|
>"$out/$name.json" 2>"$out/$name.err"
|
||||||
|
echo $? >"$out/$name.rc"
|
||||||
|
}
|
||||||
|
run ours "$PROBE" "$f"
|
||||||
|
run ref "$PY" "$HERE/ref.py" "$f"
|
||||||
|
if [ -n "${WITH_H5DUMP:-}" ]; then
|
||||||
|
run h5dump h5dump "$f"
|
||||||
|
: >"$out/h5dump.json" # h5dump's text dump is not compared, only its exit status
|
||||||
|
fi
|
||||||
|
exit 0
|
||||||
@@ -0,0 +1,92 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Tests of the reference side's corrections: `python conformance/test_ref.py`.
|
||||||
|
|
||||||
|
- ref.py compares a big-endian VL sequence by the file's values even though
|
||||||
|
h5py returns them byte-swapped (and records that it corrected them);
|
||||||
|
- ref_bugs.py confirms an object only when its reads disagree.
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
sys.path.insert(0, HERE)
|
||||||
|
import ref_bugs # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
def ref_objects(path):
|
||||||
|
out = subprocess.run([sys.executable, os.path.join(HERE, "ref.py"), path],
|
||||||
|
capture_output=True, text=True, check=True).stdout
|
||||||
|
return {o["path"]: o for o in json.loads(out)["objects"]}
|
||||||
|
|
||||||
|
|
||||||
|
class BigEndianVlen(unittest.TestCase):
|
||||||
|
def test_be_vlen_compared_by_file_values(self):
|
||||||
|
with tempfile.TemporaryDirectory() as d:
|
||||||
|
path = os.path.join(d, "v.h5")
|
||||||
|
with h5py.File(path, "w") as f:
|
||||||
|
for name, order in (("be", ">"), ("le", "<")):
|
||||||
|
t = np.dtype(order + "f4")
|
||||||
|
ds = f.create_dataset(name, (2,), dtype=h5py.vlen_dtype(t))
|
||||||
|
ds[0] = np.array([1.0, 2.0], dtype=t)
|
||||||
|
ds[1] = np.array([3.0], dtype=t)
|
||||||
|
u = np.dtype(order + "u8")
|
||||||
|
f.attrs.create(name, [np.array([1, 2], dtype=u), np.array([42], dtype=u)],
|
||||||
|
dtype=h5py.vlen_dtype(u))
|
||||||
|
objs = ref_objects(path)
|
||||||
|
be, le = objs["/be"], objs["/le"]
|
||||||
|
# Same values, so the same canonical hash whatever the file's byte order.
|
||||||
|
self.assertEqual(be["hash"], le["hash"])
|
||||||
|
self.assertEqual(objs["/"]["attrs"]["be"]["hash"], objs["/"]["attrs"]["le"]["hash"])
|
||||||
|
self.assertNotIn("ref_fix", le)
|
||||||
|
# And the correction is recorded wherever h5py needed it.
|
||||||
|
import ref
|
||||||
|
if ref.be_vlen_bug():
|
||||||
|
self.assertEqual(be.get("ref_fix"), ["h5py-be-vlen"])
|
||||||
|
self.assertEqual(objs["/"]["attrs"]["be"].get("ref_fix"), ["h5py-be-vlen"])
|
||||||
|
|
||||||
|
|
||||||
|
class RefBugsConfirmation(unittest.TestCase):
|
||||||
|
def run_check(self, outcomes):
|
||||||
|
seq = iter(outcomes)
|
||||||
|
saved = ref_bugs.read_once
|
||||||
|
ref_bugs.read_once = lambda *a: next(seq)
|
||||||
|
try:
|
||||||
|
key = next(iter(ref_bugs.READ_BUGS))
|
||||||
|
with tempfile.TemporaryDirectory() as d:
|
||||||
|
p = os.path.join(d, key[0])
|
||||||
|
os.makedirs(os.path.dirname(p))
|
||||||
|
open(p, "wb").close()
|
||||||
|
return ref_bugs.check(d, key)
|
||||||
|
finally:
|
||||||
|
ref_bugs.read_once = saved
|
||||||
|
|
||||||
|
def test_stable_values_are_not_confirmed(self):
|
||||||
|
r = self.run_check(["values a"] * len(ref_bugs.RUNS))
|
||||||
|
self.assertFalse(r["confirmed"])
|
||||||
|
|
||||||
|
def test_changing_values_are_confirmed(self):
|
||||||
|
r = self.run_check(["values a"] * (len(ref_bugs.RUNS) - 1) + ["values b"])
|
||||||
|
self.assertTrue(r["confirmed"])
|
||||||
|
r = self.run_check(["values a"] * (len(ref_bugs.RUNS) - 1) + ["error filter failed"])
|
||||||
|
self.assertTrue(r["confirmed"])
|
||||||
|
|
||||||
|
def test_errors_only_are_not_confirmed(self):
|
||||||
|
# h5py cannot read it at all: nothing it reads, nothing to excuse.
|
||||||
|
r = self.run_check(["error x"] * (len(ref_bugs.RUNS) - 1) + ["error y"])
|
||||||
|
self.assertFalse(r["confirmed"])
|
||||||
|
|
||||||
|
def test_missing_file_is_not_confirmed(self):
|
||||||
|
key = next(iter(ref_bugs.READ_BUGS))
|
||||||
|
with tempfile.TemporaryDirectory() as d:
|
||||||
|
self.assertFalse(ref_bugs.check(d, key)["confirmed"])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -3,7 +3,7 @@ name = "clawhdf5-accel"
|
|||||||
version = "2.7.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
rust-version.workspace = true
|
rust-version.workspace = true
|
||||||
description = "SIMD-accelerated operations for rustyhdf5"
|
description = "SIMD kernels (AVX2, NEON) used by clawhdf5 — pure Rust"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
|
|||||||
@@ -1,24 +1,62 @@
|
|||||||
# clawhdf5-accel
|
# clawhdf5-accel
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-accel)
|
CPU SIMD kernels for vector search: dot products, cosine similarity, L2
|
||||||
[](https://docs.rs/clawhdf5-accel)
|
distance, norms and int8 dot products, dispatched at run time to the best
|
||||||
|
backend the CPU has, with a portable scalar fallback for every operation.
|
||||||
|
[`clawhdf5-ann`](../clawhdf5-ann/README.md) and
|
||||||
|
[`clawhdf5-agent`](../clawhdf5-agent/README.md) use it in their distance
|
||||||
|
loops; it has nothing to do with HDF5 file I/O.
|
||||||
|
|
||||||
SIMD-accelerated operations for clawhdf5.
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
|
```toml
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-accel = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
```
|
||||||
|
|
||||||
|
## API
|
||||||
|
|
||||||
|
```rust
|
||||||
|
use clawhdf5_accel::{cosine_similarity, detect_backend, dot_i8, dot_product, l2_distance};
|
||||||
|
|
||||||
|
let a = [1.0f32, 2.0, 3.0, 4.0];
|
||||||
|
let b = [4.0f32, 3.0, 2.0, 1.0];
|
||||||
|
assert_eq!(dot_product(&a, &b), 20.0);
|
||||||
|
let _cos = cosine_similarity(&a, &b);
|
||||||
|
let _l2 = l2_distance(&a, &b);
|
||||||
|
assert_eq!(dot_i8(&[1, -2, 3], &[4, 5, -6]), -24);
|
||||||
|
println!("{:?}", detect_backend()); // e.g. Avx2 on x86-64, Neon on aarch64
|
||||||
|
```
|
||||||
|
|
||||||
|
Also `vector_norm`, `batch_norms`, `batch_cosine`, `batch_cosine_prenorm`,
|
||||||
|
`f16_to_f32_batch`, `checksum_fletcher32` and `align_to_cache_line`.
|
||||||
|
|
||||||
|
## Backends
|
||||||
|
|
||||||
|
`detect_backend()` picks once per process: `Avx512` (with the `avx512`
|
||||||
|
feature), `Avx2` (AVX2 + FMA), `Neon` (every aarch64 CPU), or `Scalar`.
|
||||||
|
`Sse4` and `WasmSimd128` are reported when detected but run the scalar
|
||||||
|
kernels.
|
||||||
|
`dot_i8`, used by the agent's quantised (int8) HNSW index, runs on
|
||||||
|
AVX2 and on NEON — with the `SDOT` instruction (through inline assembly,
|
||||||
|
since the intrinsic is unstable) on cores that have dotprod, such as the
|
||||||
|
Raspberry Pi 5, and plain NEON on older ones. At equal recall the int8
|
||||||
|
index answers 1.63x the queries per second of the f32 one on x86-64
|
||||||
|
(AVX2; 2026-09-20, machine not recorded, not re-run) and 1.18x on a
|
||||||
|
Raspberry Pi 5 (2026-09-21) ([`BENCHMARKS.md` § Quantising the index copy](../../BENCHMARKS.md#quantising-the-index-copy-quantized_index)).
|
||||||
|
|
||||||
|
The aarch64 code is compiled out on x86, so only the `test-arm64` CI job
|
||||||
|
builds and tests it.
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
- AVX2 and NEON SIMD acceleration
|
| Feature | Default | What | Builds C |
|
||||||
- AVX-512 support (`avx512` feature)
|
|---|---|---|---|
|
||||||
- Float16 conversion (`float16` feature)
|
| `avx512` | no | AVX-512F kernels | no |
|
||||||
- CRC32 checksum acceleration
|
| `float16` | no | `f16_to_f32_batch` through the `half` crate (a software conversion otherwise) | no |
|
||||||
|
|
||||||
## Usage
|
The half-precision conversion used for stored embeddings is
|
||||||
|
`clawhdf5_format::float16`, not this crate's.
|
||||||
```rust
|
|
||||||
use clawhdf5_accel::checksum::crc32_simd;
|
|
||||||
|
|
||||||
let crc = crc32_simd(&data);
|
|
||||||
```
|
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
|
|||||||
@@ -237,7 +237,10 @@ pub fn f16_to_f32_batch(input: &[u16], output: &mut [f32]) {
|
|||||||
convert::f16_to_f32_batch(input, output);
|
convert::f16_to_f32_batch(input, output);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compute Fletcher-32 checksum.
|
/// Compute a textbook Fletcher-32 checksum (both sums start at 0xffff).
|
||||||
|
///
|
||||||
|
/// This is not HDF5's checksum; the Fletcher-32 I/O filter uses
|
||||||
|
/// `clawhdf5_format::checksum::fletcher32`.
|
||||||
pub fn checksum_fletcher32(data: &[u8]) -> u32 {
|
pub fn checksum_fletcher32(data: &[u8]) -> u32 {
|
||||||
checksum::checksum_fletcher32(data)
|
checksum::checksum_fletcher32(data)
|
||||||
}
|
}
|
||||||
|
|||||||
+112
-16
@@ -1,28 +1,124 @@
|
|||||||
# clawhdf5-agent
|
# clawhdf5-agent
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-agent)
|
Persistent memory for AI agents in a single HDF5 file: text chunks with
|
||||||
[](https://docs.rs/clawhdf5-agent)
|
embeddings and metadata, hybrid search (HNSW vector search + BM25 keyword
|
||||||
|
search, fused), sessions, a knowledge graph, a write-ahead log for crash
|
||||||
|
safety, and optionally Ed25519-signed checkpoints. Stores open in h5py like
|
||||||
|
any other HDF5 file. Built on [`clawhdf5`](../clawhdf5/README.md),
|
||||||
|
[`clawhdf5-ann`](../clawhdf5-ann/README.md) and
|
||||||
|
[`clawhdf5-accel`](../clawhdf5-accel/README.md).
|
||||||
|
|
||||||
HDF5-backed persistent memory store for on-device AI agents.
|
It is a library: no agent framework integrates it (OpenClaw and ZeroClaw
|
||||||
|
integration claims were withdrawn on 2026-09-25; see
|
||||||
|
[`docs/openclaw.md`](../../docs/openclaw.md)). The command-line front end
|
||||||
|
is [`clawhdf5-cli`](../clawhdf5-cli/README.md).
|
||||||
|
|
||||||
Built on [clawhdf5](https://crates.io/crates/clawhdf5), clawhdf5-agent provides a vector-searchable memory backend optimized for edge AI workloads. Store embeddings, text chunks, and metadata in a single HDF5 file with SIMD-accelerated similarity search.
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
## Features
|
|
||||||
|
|
||||||
- Persistent vector store in HDF5 format
|
|
||||||
- Cosine similarity and L2 distance search
|
|
||||||
- SIMD-accelerated via clawhdf5-accel (AVX2, NEON)
|
|
||||||
- Optional GPU acceleration via clawhdf5-gpu
|
|
||||||
- Memory-mapped access for large stores
|
|
||||||
- f16 storage support for compact embeddings
|
|
||||||
|
|
||||||
## Usage
|
|
||||||
|
|
||||||
```toml
|
```toml
|
||||||
[dependencies]
|
[dependencies]
|
||||||
clawhdf5-agent = "2.1.0"
|
clawhdf5-agent = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
```
|
```
|
||||||
|
|
||||||
|
## Usage
|
||||||
|
|
||||||
|
```rust,no_run
|
||||||
|
use std::path::PathBuf;
|
||||||
|
use clawhdf5_agent::{AgentMemory, HDF5Memory, MemoryConfig, MemoryEntry, SearchOptions};
|
||||||
|
|
||||||
|
let config = MemoryConfig::new(PathBuf::from("agent.h5"), "my-agent", 384);
|
||||||
|
let mut mem = HDF5Memory::create(config)?;
|
||||||
|
|
||||||
|
mem.save(MemoryEntry {
|
||||||
|
chunk: "The deploy key rotates every Monday.".into(),
|
||||||
|
embedding: vec![0.01; 384], // from your embedding model
|
||||||
|
source_channel: "chat".into(),
|
||||||
|
timestamp: 1_790_000_000.0,
|
||||||
|
session_id: "s1".into(),
|
||||||
|
tags: "ops".into(),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
let query = vec![0.01f32; 384];
|
||||||
|
let hits = mem.search(&query, "deploy key", &SearchOptions::new(5).with_sources(["chat"]));
|
||||||
|
for h in &hits {
|
||||||
|
println!("{:.3} {}", h.score, h.chunk);
|
||||||
|
}
|
||||||
|
mem.flush_wal()?; // checkpoint now; otherwise one is made once the WAL holds more than 500 entries (wal_max_entries)
|
||||||
|
# Ok::<(), clawhdf5_agent::MemoryError>(())
|
||||||
|
```
|
||||||
|
|
||||||
|
## What is in it
|
||||||
|
|
||||||
|
- **`HDF5Memory`** — `create`, `open` (single writer: an exclusive lock on
|
||||||
|
`<store>.h5.lock`, a second opener gets `MemoryError::Locked`),
|
||||||
|
`open_read_only` (no lock, never writes). Through the `AgentMemory`
|
||||||
|
trait: `save`, `save_batch`, `delete`, `compact`, `count`, `snapshot`,
|
||||||
|
sessions; also `save_or_update`, `delete_batch`, `flush_wal`.
|
||||||
|
- **Search** — `search(query_embedding, text, &SearchOptions)`: optional
|
||||||
|
source-channel filter applied before ranking, vector + BM25 fusion
|
||||||
|
(weighted or RRF), Hebbian activation scaling, optional re-ranking
|
||||||
|
(`reranker::ReRankConfig`) and confidence rejection
|
||||||
|
(`confidence::ConfidenceConfig`). `hybrid_search` and
|
||||||
|
`hybrid_search_with` are thin wrappers. The vector stage uses the HNSW
|
||||||
|
index (`hnsw` feature); its graph is saved to `<store>.h5.ann` at each
|
||||||
|
checkpoint and reloaded on open (rebuilt if stale or damaged).
|
||||||
|
- **Storage settings** (`MemoryConfig`, persisted with the store):
|
||||||
|
`float16` embeddings (on by default for new stores; 48% smaller file at
|
||||||
|
100K records, same retrieval on LongMemEval), `quantized_index` (int8
|
||||||
|
copy of the vectors in the index, on by default; re-scored against the
|
||||||
|
exact embeddings), `compression` (off by default), HNSW `m`/`ef`
|
||||||
|
parameters, WAL settings (`wal_enabled`, on by default; `wal_max_entries`,
|
||||||
|
500: the WAL is checkpointed into the `.h5` once it holds more).
|
||||||
|
- **WAL** (`wal`) — every write is appended to `<store>.h5.wal` with a
|
||||||
|
chained CRC32 per entry, so a corrupted, reordered or spliced entry stops
|
||||||
|
replay. Recovers from a process crash at any point, including between a
|
||||||
|
checkpoint and the WAL truncate. WAL appends are not fsynced: saves since
|
||||||
|
the last checkpoint can be lost on power failure. An unreadable WAL is
|
||||||
|
quarantined to `<store>.h5.wal.corrupt-<ts>`.
|
||||||
|
- **Signed checkpoints** (`signing`) — `set_signing_key` signs a manifest
|
||||||
|
(SHA-256 Merkle tree over records, plus settings, sessions and graph) at
|
||||||
|
every checkpoint; `HDF5Memory::verify(path, &public_key)` checks it and
|
||||||
|
locates edits. WAL entries after the checkpoint are not covered.
|
||||||
|
- **Knowledge graph** (`knowledge`, `entity_extract`) — `add_entity`,
|
||||||
|
`add_entity_alias`, `add_relation`, `extract_and_store_entities`,
|
||||||
|
traversal and spreading activation.
|
||||||
|
- **Also:** sessions (`session`), temporal index (`temporal`),
|
||||||
|
consolidation tiers (`consolidation`), an in-memory TTL tier
|
||||||
|
(`ephemeral`), multi-modal embeddings (`multimodal`), `AGENTS.md`
|
||||||
|
generation (`agents_md`), query expansion, and a session-scoped
|
||||||
|
provenance ledger and write-anomaly detector on every save
|
||||||
|
(`take_anomaly_alerts`; alerts never block a save, and the source is
|
||||||
|
inferred from `source_channel`, not authenticated).
|
||||||
|
- `openclaw::ClawhdfBackend` is `search` with re-ranking and confidence
|
||||||
|
on, plus Markdown import/export. The module name is historical: it is not
|
||||||
|
an OpenClaw plugin.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
| Feature | Default | What | Builds C |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `hnsw` | yes | HNSW vector index (`clawhdf5-ann`); without it the vector stage is an exact linear cosine scan | no |
|
||||||
|
| `parallel` | yes | build the HNSW index on a rayon pool (same graph either way) | no |
|
||||||
|
| `float16` | yes | f16 helpers in `vector_search` (`half`). Stores' `MemoryConfig::float16` works without it. | no |
|
||||||
|
| `fast-math` | no | `matrixmultiply` batch distances in `strategy` | no |
|
||||||
|
| `accelerate` | no | Apple Accelerate BLAS in `strategy` (macOS) | links a system framework |
|
||||||
|
| `openblas` | no | OpenBLAS in `strategy` | yes (`openblas-src`) |
|
||||||
|
| `gpu` | no | `gpu_search` through [`clawhdf5-gpu`](../clawhdf5-gpu/README.md) (wgpu), used by `strategy`, not by `HDF5Memory::search` | no, but needs GPU drivers |
|
||||||
|
| `zstd` | no | Zstd instead of deflate when `MemoryConfig::compression` is on | yes (libzstd) |
|
||||||
|
| `async` | no | `async_memory` wrapper on tokio | no |
|
||||||
|
|
||||||
|
`--no-default-features --features float16` forces the exact linear scan.
|
||||||
|
|
||||||
|
## Measurements and limits
|
||||||
|
|
||||||
|
- Search recall and latency, file size, LongMemEval and MemoryArena
|
||||||
|
retrieval numbers: [`BENCHMARKS.md`](../../BENCHMARKS.md), measured with
|
||||||
|
the `clawhdf5-bench` binaries (`search_harness`, `longmemeval_bench`,
|
||||||
|
`footprint_bench`, ...).
|
||||||
|
- Known issues and their history: [`docs/known-issues.md`](../../docs/known-issues.md).
|
||||||
|
- Migrating a SQLite memory database:
|
||||||
|
[`clawhdf5-migrate`](../clawhdf5-migrate/README.md).
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT
|
MIT
|
||||||
|
|||||||
@@ -477,10 +477,34 @@ fn write_string_dataset(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `/meta`'s attributes, failing if any of them cannot be read.
|
||||||
|
///
|
||||||
|
/// `Group::attrs` leaves out an attribute it cannot decode. For the store's
|
||||||
|
/// settings that would silently fall back to defaults (e.g. `float16`, the
|
||||||
|
/// WAL mark), so an unreadable attribute is an error here, as it was before
|
||||||
|
/// `attrs` became tolerant.
|
||||||
|
fn meta_attrs(
|
||||||
|
file: &clawhdf5::File,
|
||||||
|
) -> Result<std::collections::HashMap<String, AttrValue>, MemoryError> {
|
||||||
|
let meta = file
|
||||||
|
.group("meta")
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
||||||
|
let (attrs, errors) = meta
|
||||||
|
.attrs_with_errors()
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
||||||
|
if let Some(e) = errors.first() {
|
||||||
|
return Err(MemoryError::Schema(format!(
|
||||||
|
"cannot read /meta attrs: {} unreadable, first: {e}",
|
||||||
|
errors.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(attrs)
|
||||||
|
}
|
||||||
|
|
||||||
/// Validate an HDF5 file has the correct schema and load all data.
|
/// Validate an HDF5 file has the correct schema and load all data.
|
||||||
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
||||||
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
||||||
let attrs = file.group("meta").ok()?.attrs().ok()?;
|
let attrs = meta_attrs(file).ok()?;
|
||||||
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
||||||
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
||||||
_ => return None,
|
_ => return None,
|
||||||
@@ -498,10 +522,7 @@ pub fn read_signature(
|
|||||||
file: &clawhdf5::File,
|
file: &clawhdf5::File,
|
||||||
) -> Result<Option<crate::signing::StoredSignature>, MemoryError> {
|
) -> Result<Option<crate::signing::StoredSignature>, MemoryError> {
|
||||||
use crate::signing::{Manifest, StoredSignature, from_hex};
|
use crate::signing::{Manifest, StoredSignature, from_hex};
|
||||||
let attrs = file
|
let attrs = meta_attrs(file)?;
|
||||||
.group("meta")
|
|
||||||
.and_then(|g| g.attrs())
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
|
||||||
let version = match attrs.get(SIG_VERSION_ATTR) {
|
let version = match attrs.get(SIG_VERSION_ATTR) {
|
||||||
None => return Ok(None),
|
None => return Ok(None),
|
||||||
Some(AttrValue::I64(v)) => *v,
|
Some(AttrValue::I64(v)) => *v,
|
||||||
@@ -552,18 +573,14 @@ pub fn read_signature(
|
|||||||
|
|
||||||
/// Read the checkpoint bookkeeping from `/meta`.
|
/// Read the checkpoint bookkeeping from `/meta`.
|
||||||
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
||||||
let ann_generation = file
|
let ann_generation =
|
||||||
.group("meta")
|
meta_attrs(file)
|
||||||
.ok()
|
.ok()
|
||||||
.and_then(|g| g.attrs().ok())
|
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
||||||
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
Some(AttrValue::I64(v)) => Some(*v as u64),
|
||||||
Some(AttrValue::I64(v)) => Some(*v as u64),
|
_ => None,
|
||||||
_ => None,
|
});
|
||||||
});
|
let signed = meta_attrs(file).is_ok_and(|attrs| attrs.contains_key(SIG_VERSION_ATTR));
|
||||||
let signed = file
|
|
||||||
.group("meta")
|
|
||||||
.and_then(|g| g.attrs())
|
|
||||||
.is_ok_and(|attrs| attrs.contains_key(SIG_VERSION_ATTR));
|
|
||||||
CheckpointMeta {
|
CheckpointMeta {
|
||||||
wal_applied: read_wal_mark(file),
|
wal_applied: read_wal_mark(file),
|
||||||
ann_generation,
|
ann_generation,
|
||||||
@@ -575,12 +592,7 @@ pub fn validate_and_load(
|
|||||||
file: &clawhdf5::File,
|
file: &clawhdf5::File,
|
||||||
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
||||||
// Read /meta group attributes
|
// Read /meta group attributes
|
||||||
let meta = file
|
let attrs = meta_attrs(file)?;
|
||||||
.group("meta")
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
|
||||||
let attrs = meta
|
|
||||||
.attrs()
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
|
||||||
|
|
||||||
let schema_version = match attrs.get("schema_version") {
|
let schema_version = match attrs.get("schema_version") {
|
||||||
Some(AttrValue::String(s)) => s.clone(),
|
Some(AttrValue::String(s)) => s.clone(),
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ const LOCK_RETRY_DELAY: std::time::Duration = std::time::Duration::from_millis(1
|
|||||||
/// never leaves a stale lock behind; the empty lock file itself is harmless).
|
/// never leaves a stale lock behind; the empty lock file itself is harmless).
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
pub(crate) struct StoreLock {
|
pub(crate) struct StoreLock {
|
||||||
_file: File,
|
file: File,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl StoreLock {
|
impl StoreLock {
|
||||||
@@ -42,7 +42,7 @@ impl StoreLock {
|
|||||||
let mut attempts_left = LOCK_RETRIES;
|
let mut attempts_left = LOCK_RETRIES;
|
||||||
loop {
|
loop {
|
||||||
match file.try_lock() {
|
match file.try_lock() {
|
||||||
Ok(()) => return Ok(Self { _file: file }),
|
Ok(()) => return Ok(Self { file }),
|
||||||
Err(TryLockError::WouldBlock) if attempts_left > 0 => {
|
Err(TryLockError::WouldBlock) if attempts_left > 0 => {
|
||||||
attempts_left -= 1;
|
attempts_left -= 1;
|
||||||
std::thread::sleep(LOCK_RETRY_DELAY);
|
std::thread::sleep(LOCK_RETRY_DELAY);
|
||||||
@@ -60,6 +60,16 @@ impl StoreLock {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl Drop for StoreLock {
|
||||||
|
/// Unlocks before the file is closed: a process another thread forks
|
||||||
|
/// inherits the descriptor until it execs, and a `flock` lasts while any
|
||||||
|
/// descriptor of the open file does, so closing alone could keep the
|
||||||
|
/// store locked for a moment after the drop (see `FileEditor`'s `Drop`).
|
||||||
|
fn drop(&mut self) {
|
||||||
|
let _ = self.file.unlock();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|||||||
@@ -258,3 +258,43 @@ fn an_existing_f32_store_stays_f32() {
|
|||||||
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
||||||
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `Group::attrs` leaves out an attribute it cannot decode. A store whose
|
||||||
|
/// `float16` setting is unreadable must not open as `float16 = false` (or with
|
||||||
|
/// any other default in place of a setting it has): it is an error.
|
||||||
|
#[test]
|
||||||
|
fn unreadable_meta_attribute_fails_open_instead_of_defaulting() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("store.h5");
|
||||||
|
{
|
||||||
|
let mut m = HDF5Memory::create(config(&dir, "store.h5", true)).unwrap();
|
||||||
|
m.save(entry(1)).unwrap();
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
}
|
||||||
|
assert!(HDF5Memory::open_read_only(&path).is_ok());
|
||||||
|
|
||||||
|
// Give the `float16` attribute message an unknown version (the name is
|
||||||
|
// at +8 in a version-1 message and +9 in a version-3 one).
|
||||||
|
let mut bytes = std::fs::read(&path).unwrap();
|
||||||
|
let name = b"float16\0";
|
||||||
|
let mut hit = false;
|
||||||
|
let positions: Vec<usize> = (9..bytes.len() - name.len())
|
||||||
|
.filter(|&p| &bytes[p..p + name.len()] == name)
|
||||||
|
.collect();
|
||||||
|
for pos in positions {
|
||||||
|
for (back, version) in [(8, 1u8), (9, 3u8)] {
|
||||||
|
if bytes[pos - back] == version {
|
||||||
|
bytes[pos - back] = 0x7f;
|
||||||
|
hit = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(hit, "float16 attribute message not found");
|
||||||
|
std::fs::write(&path, &bytes).unwrap();
|
||||||
|
|
||||||
|
match HDF5Memory::open_read_only(&path) {
|
||||||
|
Err(MemoryError::Schema(msg)) => assert!(msg.contains("/meta"), "{msg}"),
|
||||||
|
Err(e) => panic!("unexpected error: {e}"),
|
||||||
|
Ok(_) => panic!("store opened with an unreadable float16 setting"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -3,7 +3,7 @@ name = "clawhdf5-android"
|
|||||||
version = "2.7.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
rust-version.workspace = true
|
rust-version.workspace = true
|
||||||
description = "Android JNI bridge for edgehdf5-memory HDF5 backend"
|
description = "Android JNI bindings for clawhdf5 agent memory"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
|
|
||||||
[lib]
|
[lib]
|
||||||
|
|||||||
@@ -0,0 +1,44 @@
|
|||||||
|
# clawhdf5-android
|
||||||
|
|
||||||
|
A C ABI over [`clawhdf5-agent`](../clawhdf5-agent/README.md) for Android
|
||||||
|
apps: a `cdylib` exporting `extern "C"` functions (`edgehdf5_*`, a name
|
||||||
|
kept from the project's earlier "edgehdf5" days) that manage an
|
||||||
|
`HDF5Memory` through an opaque handle.
|
||||||
|
|
||||||
|
The functions are plain C symbols, not JNI-mangled `Java_...` entry points:
|
||||||
|
a Kotlin/Java app calls them through a thin JNI shim or JNA of its own. No
|
||||||
|
such shim, Gradle project or AAR is in this repository, and the crate is
|
||||||
|
not built for an Android target in CI (only its host-side unit tests run
|
||||||
|
with the workspace).
|
||||||
|
|
||||||
|
## Functions
|
||||||
|
|
||||||
|
| Function | What |
|
||||||
|
|---|---|
|
||||||
|
| `edgehdf5_create(path, agent_id, embedding_dim)` / `edgehdf5_open(path)` | a handle, or null on failure |
|
||||||
|
| `edgehdf5_close(handle)` | drop the store; what is not yet checkpointed stays in its WAL, as with any `HDF5Memory` |
|
||||||
|
| `edgehdf5_save(handle, ...)` | save one entry; the embedding length is checked against the store's dimension before the pointer is read |
|
||||||
|
| `edgehdf5_delete`, `edgehdf5_count`, `edgehdf5_count_active` | |
|
||||||
|
| `edgehdf5_hybrid_search(handle, query, len, text, vector_weight, keyword_weight, max_results, out_indices, out_scores, out_chunks)` | results into caller-provided arrays; returns the number written |
|
||||||
|
| `edgehdf5_add_session`, `edgehdf5_get_session_summary` | sessions |
|
||||||
|
| `edgehdf5_add_entity`, `edgehdf5_add_relation` | knowledge graph |
|
||||||
|
| `edgehdf5_free_string` | free a string this library returned |
|
||||||
|
|
||||||
|
Every function is `unsafe`: the caller guarantees valid, NUL-terminated
|
||||||
|
strings and correctly sized buffers (see each function's `# Safety`
|
||||||
|
section), and serialises access to a handle; separate handles are
|
||||||
|
independent.
|
||||||
|
|
||||||
|
## Build
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo build --release -p clawhdf5-android # host build; for a device, add --target aarch64-linux-android with the NDK's linker configured
|
||||||
|
```
|
||||||
|
|
||||||
|
It depends on `clawhdf5-agent` with **default features off**, so there is
|
||||||
|
no HNSW index (the vector stage is an exact linear scan) and no rayon
|
||||||
|
pool. No C is compiled.
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
MIT
|
||||||
@@ -1,25 +1,70 @@
|
|||||||
# clawhdf5-ann
|
# clawhdf5-ann
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-ann)
|
An HNSW (Hierarchical Navigable Small World) approximate nearest-neighbour
|
||||||
[](https://docs.rs/clawhdf5-ann)
|
index in pure Rust, with cosine or L2 distance, optional int8 storage of
|
||||||
|
the vectors, deletions, and persistence as an HDF5 file. It is the vector
|
||||||
|
stage of [`clawhdf5-agent`](../clawhdf5-agent/README.md)'s search (the
|
||||||
|
agent's `hnsw` feature, on by default); distances run on
|
||||||
|
[`clawhdf5-accel`](../clawhdf5-accel/README.md)'s SIMD kernels.
|
||||||
|
|
||||||
HNSW approximate nearest neighbor index stored as HDF5.
|
Neighbours are chosen with the HNSW paper's diversity heuristic, not plain
|
||||||
|
closest-M (which capped recall on clustered data at 0.31 recall@10 at 100K
|
||||||
|
vectors).
|
||||||
|
|
||||||
## Features
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
- Build and query HNSW indexes persisted in HDF5 format
|
```toml
|
||||||
- Pure Rust, no C dependencies
|
[dependencies]
|
||||||
- Efficient similarity search for high-dimensional vectors
|
clawhdf5-ann = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
```
|
||||||
|
|
||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
```rust
|
```rust
|
||||||
use clawhdf5_ann::HnswIndex;
|
use clawhdf5_ann::{DistanceMetric, HnswIndex, Storage};
|
||||||
|
|
||||||
let index = HnswIndex::from_hdf5("vectors.h5").unwrap();
|
let vectors: Vec<Vec<f32>> = (0..500)
|
||||||
let neighbors = index.search(&query, 10);
|
.map(|i| (0..16).map(|j| ((i * 31 + j * 7) % 97) as f32 / 97.0).collect())
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// m = 16 connections per node, ef_construction = 200
|
||||||
|
let mut index = HnswIndex::build_with(&vectors, 16, 200, DistanceMetric::Cosine, Storage::Int8);
|
||||||
|
let hits = index.search(&vectors[42], 10, 64); // (id, distance), closest first; ef >= k
|
||||||
|
assert!(hits[0].1 < 1e-3); // vector 42 itself (or an identical one)
|
||||||
|
|
||||||
|
let id = index.insert(vec![0.5; 16]);
|
||||||
|
index.mark_deleted(id);
|
||||||
|
|
||||||
|
// Persist as HDF5 (a self-contained file: graph and vectors) and load it back
|
||||||
|
let bytes = index.to_hdf5_bytes().unwrap();
|
||||||
|
let loaded = HnswIndex::load_from_hdf5(&bytes).unwrap();
|
||||||
|
assert_eq!(loaded.len(), index.len());
|
||||||
```
|
```
|
||||||
|
|
||||||
|
- `HnswIndex::build` (L2), `build_with_metric`, `build_with` (metric and
|
||||||
|
storage); `new`/`new_with` plus `insert` for an index built
|
||||||
|
incrementally.
|
||||||
|
- `Storage::Int8` keeps each vector as `i8`, a quarter of the memory; it
|
||||||
|
applies to `Cosine` only (an L2 index keeps `Float32`). Distances are then
|
||||||
|
approximate, so a caller that needs exact ranking re-scores the
|
||||||
|
candidates, as the agent does.
|
||||||
|
- `mark_deleted`, `is_deleted`, `deleted_count`, `active_len`, `compact`
|
||||||
|
(returns the old-to-new id map).
|
||||||
|
- `save_to_hdf5(&mut writer)` / `to_hdf5_bytes` / `load_from_hdf5` store
|
||||||
|
the whole index; `graph_to_bytes` / `from_graph_bytes` store only the
|
||||||
|
graph (with a CRC32) for a caller that keeps the vectors elsewhere — the
|
||||||
|
agent's `<store>.h5.ann` sidecar.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
| Feature | Default | What | Builds C |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `parallel` | no | build the graph on a rayon pool; the graph is identical with or without it | no |
|
||||||
|
|
||||||
|
Recall and speed against exact search, for the index alone and in the
|
||||||
|
agent: [`BENCHMARKS.md`](../../BENCHMARKS.md), measured with
|
||||||
|
`cargo run --release -p clawhdf5-bench --bin search_harness`.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT
|
MIT
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ use clawhdf5_format::filter_pipeline::FilterPipeline;
|
|||||||
use clawhdf5_format::group_v2::resolve_path_any;
|
use clawhdf5_format::group_v2::resolve_path_any;
|
||||||
use clawhdf5_format::message_type::MessageType;
|
use clawhdf5_format::message_type::MessageType;
|
||||||
use clawhdf5_format::object_header::ObjectHeader;
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
use clawhdf5_format::signature::find_signature;
|
use clawhdf5_format::signature::split_user_block;
|
||||||
use clawhdf5_format::superblock::Superblock;
|
use clawhdf5_format::superblock::Superblock;
|
||||||
use clawhdf5_io::FileWriter as IoFileWriter;
|
use clawhdf5_io::FileWriter as IoFileWriter;
|
||||||
|
|
||||||
@@ -861,8 +861,9 @@ impl HnswIndex {
|
|||||||
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
||||||
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
||||||
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
||||||
let sig_offset = find_signature(data)?;
|
// Addresses are relative to the superblock: skip any user block.
|
||||||
let sb = Superblock::parse(data, sig_offset)?;
|
let (_, data) = split_user_block(data)?;
|
||||||
|
let sb = Superblock::parse(data, 0)?;
|
||||||
|
|
||||||
// Read config dataset and its attributes
|
// Read config dataset and its attributes
|
||||||
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
||||||
|
|||||||
@@ -34,6 +34,10 @@ path = "src/bin/consolidation_efficiency.rs"
|
|||||||
name = "ephemeral_perf"
|
name = "ephemeral_perf"
|
||||||
path = "src/bin/ephemeral_perf.rs"
|
path = "src/bin/ephemeral_perf.rs"
|
||||||
|
|
||||||
|
[[bin]]
|
||||||
|
name = "concurrent_read"
|
||||||
|
path = "src/bin/concurrent_read.rs"
|
||||||
|
|
||||||
[[bin]]
|
[[bin]]
|
||||||
name = "mpi_io_bench"
|
name = "mpi_io_bench"
|
||||||
path = "src/bin/mpi_io_bench.rs"
|
path = "src/bin/mpi_io_bench.rs"
|
||||||
@@ -64,6 +68,10 @@ clawhdf5-io = { path = "../clawhdf5-io" }
|
|||||||
mpi = { version = "0.8", optional = true }
|
mpi = { version = "0.8", optional = true }
|
||||||
serde = { workspace = true }
|
serde = { workspace = true }
|
||||||
serde_json = "1"
|
serde_json = "1"
|
||||||
|
# concurrent_read: size the decode pool (--decode-threads) and evict files
|
||||||
|
# from the page cache (--cold, posix_fadvise). Both pure Rust / bindings only.
|
||||||
|
rayon = "1"
|
||||||
|
libc = "0.2"
|
||||||
tempfile = { workspace = true }
|
tempfile = { workspace = true }
|
||||||
# Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5).
|
# Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5).
|
||||||
# Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare
|
# Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# clawhdf5-bench
|
||||||
|
|
||||||
|
The measurement harnesses behind [`BENCHMARKS.md`](../../BENCHMARKS.md):
|
||||||
|
HDF5 read and write speed (against libhdf5 and h5py where noted) and the
|
||||||
|
agent store's search, footprint and retrieval quality. Not meant for
|
||||||
|
publishing; nothing else in the workspace depends on it. Run everything with
|
||||||
|
`--release`, and quote numbers with the machine, date and command, as
|
||||||
|
`BENCHMARKS.md` does.
|
||||||
|
|
||||||
|
## Binaries
|
||||||
|
|
||||||
|
| Binary | Measures |
|
||||||
|
|---|---|
|
||||||
|
| `read_harness` | full reads vs hyperslab selections of a chunked 2-D dataset (compressed and not) and a contiguous one: does a selection cost scale with the selection or the dataset? (`-- --large` for 512 MB) |
|
||||||
|
| `concurrent_read` | decoded read throughput vs threads on one open `File`; `scripts/concurrent_read_h5py.py` runs the same workload with h5py (threads and processes) and `scripts/compare_concurrent_read.py` tabulates both |
|
||||||
|
| `search_harness` | HNSW recall@10 vs exact search, QPS and latency per `ef`, and end-to-end `HDF5Memory` ingest/checkpoint/open/search at 1K–100K (`--full`); studies: `--float16-study`, `--options-study`, `--signing-study`, `--ann-only --uniform` |
|
||||||
|
| `longmemeval_bench` | LongMemEval retrieval recall (turn and session Hit@k, MRR) — **retrieval, not QA accuracy**. Oracle or full `longmemeval_s` haystack; `--features embeddings` (or `embeddings-cuda`) embeds with MiniLM, otherwise the vector stage is inert and the run is BM25-only |
|
||||||
|
| `memory_arena` | a deterministic multi-session retrieval benchmark (BM25-only) |
|
||||||
|
| `footprint_bench` | file size and bytes per record at 100–100K records, float16 or `--f32`, WAL on/off, compressed or not |
|
||||||
|
| `consolidation_efficiency` | retrieval before and after consolidation on signal + noise records |
|
||||||
|
| `ephemeral_perf` | the in-memory ephemeral tier's set/get latency |
|
||||||
|
| `mpi_io_bench` | `clawhdf5-io`'s `MpiVol` (root-read + broadcast, not collective I/O); needs `--features mpi-io` and `mpirun` |
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo run --release -p clawhdf5-bench --bin search_harness -- --full
|
||||||
|
cargo run --release -p clawhdf5-bench --bin read_harness
|
||||||
|
```
|
||||||
|
|
||||||
|
## Criterion benches and example
|
||||||
|
|
||||||
|
- `cargo bench -p clawhdf5-bench` runs `h5bench_write`, `h5bench_read` and
|
||||||
|
`h5bench_meta` (h5bench-style sequential, chunked, strided and metadata
|
||||||
|
workloads). `--features libhdf5-compare` adds the same workloads through
|
||||||
|
libhdf5 (the `hdf5-metno` crate; needs a system libhdf5 1.14).
|
||||||
|
- `examples/worldmodel_sampling.rs`: shuffled per-frame reads of a
|
||||||
|
`(N, H, W, C)` `uint8` dataset, clawhdf5 against h5py on the same file.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
| Feature | What | Builds C |
|
||||||
|
|---|---|---|
|
||||||
|
| `libhdf5-compare` | libhdf5 variants of the Criterion benches | links the system libhdf5 |
|
||||||
|
| `mpi-io` | `mpi_io_bench` | yes (`mpi-sys`; needs an MPI installation) |
|
||||||
|
| `embeddings` | MiniLM embeddings for `longmemeval_bench` (candle) | yes (a `cc` build dependency in the candle/tokenizers tree) |
|
||||||
|
| `embeddings-cuda` | the same on a CUDA GPU (minutes instead of hours on the full haystack) | yes (CUDA) |
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
MIT
|
||||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,70 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Tabulate concurrent_read JSON results (clawhdf5, h5py threads/processes).
|
||||||
|
|
||||||
|
python compare_concurrent_read.py clawhdf5.json h5py-threads.json h5py-procs.json
|
||||||
|
|
||||||
|
Prints one Markdown table: for each layout, mode and thread count, every
|
||||||
|
tool's MB/s and scaling efficiency, and the first file's MB/s relative to each
|
||||||
|
of the others. Refuses to compare runs whose workload parameters differ.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
|
||||||
|
COMPARED = ("datasets", "rows", "cols", "chunk", "deflate_level", "slab", "slabs", "seed")
|
||||||
|
|
||||||
|
|
||||||
|
def main(paths):
|
||||||
|
if len(paths) < 2:
|
||||||
|
sys.exit(__doc__)
|
||||||
|
docs = []
|
||||||
|
for p in paths:
|
||||||
|
with open(p) as fh:
|
||||||
|
docs.append(json.load(fh))
|
||||||
|
ref = docs[0]
|
||||||
|
for d, p in zip(docs[1:], paths[1:]):
|
||||||
|
diff = [k for k in COMPARED if d["params"].get(k) != ref["params"].get(k)]
|
||||||
|
if diff:
|
||||||
|
sys.exit(f"{p}: workload differs from {paths[0]} in {', '.join(diff)}")
|
||||||
|
if d["cache"] != ref["cache"]:
|
||||||
|
print(f"warning: {p} ran {d['cache']!r}, {paths[0]} ran {ref['cache']!r}",
|
||||||
|
file=sys.stderr)
|
||||||
|
if d.get("host") != ref.get("host"):
|
||||||
|
print(f"warning: {p} ran on {d.get('host')}, {paths[0]} on {ref.get('host')}",
|
||||||
|
file=sys.stderr)
|
||||||
|
|
||||||
|
names = [d["tool"] for d in docs]
|
||||||
|
for d in docs:
|
||||||
|
extra = f", HDF5 {d['hdf5_version']}" if "hdf5_version" in d else ""
|
||||||
|
print(f"- {d['tool']} {d['version']}{extra}: host {d.get('host')}, "
|
||||||
|
f"{d.get('cpus')} CPUs, cache {d['cache']}, decode threads per read "
|
||||||
|
f"{d.get('decode_threads')}")
|
||||||
|
p = ref["params"]
|
||||||
|
print(f"\n{p['datasets']} datasets of {p['rows']} x {p['cols']} f32, chunks "
|
||||||
|
f"{p['chunk'][0]} x {p['chunk'][1]} (deflate {p['deflate_level']}); "
|
||||||
|
f"`same`: {p['slabs']} slabs of {p['slab']} x {p['slab']}\n")
|
||||||
|
|
||||||
|
index = [{(r["layout"], r["mode"], r["threads"]): r for r in d["results"]} for d in docs]
|
||||||
|
keys = [(r["layout"], r["mode"], r["threads"]) for r in ref["results"]]
|
||||||
|
|
||||||
|
head = ["layout", "mode", "threads"]
|
||||||
|
head += [f"{n} MB/s (eff)" for n in names]
|
||||||
|
head += [f"{names[0]} / {n}" for n in names[1:]]
|
||||||
|
print("| " + " | ".join(head) + " |")
|
||||||
|
print("|---|---|" + "---:|" * (len(head) - 2))
|
||||||
|
for key in keys:
|
||||||
|
cells = [key[0], key[1], str(key[2])]
|
||||||
|
rs = [ix.get(key) for ix in index]
|
||||||
|
for r in rs:
|
||||||
|
if r is None:
|
||||||
|
cells.append("-")
|
||||||
|
else:
|
||||||
|
eff = "-" if r["efficiency"] is None else f"{r['efficiency']:.2f}"
|
||||||
|
cells.append(f"{r['mb_s']:.0f} ({eff})")
|
||||||
|
for r in rs[1:]:
|
||||||
|
cells.append("-" if r is None else f"{rs[0]['mb_s'] / r['mb_s']:.2f}x")
|
||||||
|
print("| " + " | ".join(cells) + " |")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1:])
|
||||||
@@ -0,0 +1,265 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""The concurrent_read workload with h5py, on the files concurrent_read wrote.
|
||||||
|
|
||||||
|
libhdf5 serialises every API call under one global lock, and h5py holds its
|
||||||
|
own global lock around every call as well, so h5py *threads* cannot decode in
|
||||||
|
parallel. h5py users scale with *processes* instead; ``--executor processes``
|
||||||
|
measures that (each worker opens the file itself).
|
||||||
|
|
||||||
|
The workload mirrors ``crates/clawhdf5-bench/src/bin/concurrent_read.rs``:
|
||||||
|
|
||||||
|
* ``distinct``: every dataset read in full once per repetition; worker ``t``
|
||||||
|
of ``T`` reads datasets ``t, t + T, ...``.
|
||||||
|
* ``same``: ``--slabs`` random ``--slab`` x ``--slab`` hyperslabs of ``d00``
|
||||||
|
(slab ``j`` to worker ``j % T``), offsets from the same splitmix64 stream.
|
||||||
|
|
||||||
|
Each worker times itself from a start barrier; a repetition spans the earliest
|
||||||
|
start to the latest finish (CLOCK_MONOTONIC, comparable across processes).
|
||||||
|
Threads share one ``h5py.File`` per repetition; process workers open the file
|
||||||
|
inside the timed region (a few ms against reads of many MiB).
|
||||||
|
|
||||||
|
Generate the files first with the Rust harness (it writes ``manifest.json``),
|
||||||
|
then, for example::
|
||||||
|
|
||||||
|
python concurrent_read_h5py.py --dir DIR --executor threads --json h5py-threads.json
|
||||||
|
python concurrent_read_h5py.py --dir DIR --executor processes --json h5py-procs.json
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import multiprocessing as mp
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
import socket
|
||||||
|
import sys
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
M64 = (1 << 64) - 1
|
||||||
|
|
||||||
|
|
||||||
|
def splitmix64(state):
|
||||||
|
"""Return (new_state, value); the same stream as the Rust harness."""
|
||||||
|
state = (state + 0x9E3779B97F4A7C15) & M64
|
||||||
|
z = state
|
||||||
|
z = ((z ^ (z >> 30)) * 0xBF58476D1CE4E5B9) & M64
|
||||||
|
z = ((z ^ (z >> 27)) * 0x94D049BB133111EB) & M64
|
||||||
|
return state, z ^ (z >> 31)
|
||||||
|
|
||||||
|
|
||||||
|
def value(k, i):
|
||||||
|
"""Element i (row-major) of dataset k, exactly as concurrent_read writes it."""
|
||||||
|
_, noise = splitmix64(i ^ (k << 40))
|
||||||
|
return np.float32((((i >> 6) % 16384) + k) + (noise & 0xFF) / 256.0)
|
||||||
|
|
||||||
|
|
||||||
|
def slab_offsets(seed, count, rows, cols, slab):
|
||||||
|
s = seed
|
||||||
|
out = []
|
||||||
|
for _ in range(count):
|
||||||
|
s, r = splitmix64(s)
|
||||||
|
s, c = splitmix64(s)
|
||||||
|
out.append((r % (rows - slab + 1), c % (cols - slab + 1)))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def now():
|
||||||
|
return time.clock_gettime(time.CLOCK_MONOTONIC)
|
||||||
|
|
||||||
|
|
||||||
|
def work(f, mode, t, threads, m, slabs, slab, verify):
|
||||||
|
"""Worker t's share of one repetition on an open h5py.File."""
|
||||||
|
n = m["rows"] * m["cols"]
|
||||||
|
if mode == "distinct":
|
||||||
|
for k in range(t, m["datasets"], threads):
|
||||||
|
got = f[f"d{k:02d}"][...]
|
||||||
|
assert got.size == n
|
||||||
|
if verify:
|
||||||
|
flat = got.reshape(-1)
|
||||||
|
for i in (0, n // 3, n - 1):
|
||||||
|
assert flat[i] == value(k, i), f"d{k:02d}[{i}]"
|
||||||
|
else:
|
||||||
|
ds = f["d00"]
|
||||||
|
cols = m["cols"]
|
||||||
|
for r, c in slabs[t::threads]:
|
||||||
|
got = ds[r : r + slab, c : c + slab]
|
||||||
|
assert got.shape == (slab, slab)
|
||||||
|
if verify:
|
||||||
|
assert got[0, 0] == value(0, r * cols + c)
|
||||||
|
last = (r + slab - 1) * cols + c + slab - 1
|
||||||
|
assert got[-1, -1] == value(0, last)
|
||||||
|
|
||||||
|
|
||||||
|
# ----- process workers ------------------------------------------------------
|
||||||
|
|
||||||
|
_barrier = None
|
||||||
|
|
||||||
|
|
||||||
|
def _init(barrier):
|
||||||
|
global _barrier
|
||||||
|
_barrier = barrier
|
||||||
|
|
||||||
|
|
||||||
|
def _proc_task(task):
|
||||||
|
path, mode, t, threads, m, slabs, slab = task
|
||||||
|
_barrier.wait()
|
||||||
|
start = now()
|
||||||
|
with h5py.File(path, "r") as f:
|
||||||
|
work(f, mode, t, threads, m, slabs, slab, False)
|
||||||
|
return start, now()
|
||||||
|
|
||||||
|
|
||||||
|
def _noop(_):
|
||||||
|
return os.getpid()
|
||||||
|
|
||||||
|
|
||||||
|
def run_threads(path, mode, threads, m, slabs, slab):
|
||||||
|
spans = [None] * threads
|
||||||
|
barrier = threading.Barrier(threads)
|
||||||
|
with h5py.File(path, "r") as f:
|
||||||
|
|
||||||
|
def body(t):
|
||||||
|
barrier.wait()
|
||||||
|
start = now()
|
||||||
|
work(f, mode, t, threads, m, slabs, slab, False)
|
||||||
|
spans[t] = (start, now())
|
||||||
|
|
||||||
|
ts = [threading.Thread(target=body, args=(t,)) for t in range(threads)]
|
||||||
|
for th in ts:
|
||||||
|
th.start()
|
||||||
|
for th in ts:
|
||||||
|
th.join()
|
||||||
|
return max(e for _, e in spans) - min(s for s, _ in spans)
|
||||||
|
|
||||||
|
|
||||||
|
def run_processes(pool, path, mode, threads, m, slabs, slab):
|
||||||
|
tasks = [(path, mode, t, threads, m, slabs, slab) for t in range(threads)]
|
||||||
|
# One task per worker: each blocks in the barrier until all T have
|
||||||
|
# started, so no worker can take a second task.
|
||||||
|
spans = pool.map(_proc_task, tasks, chunksize=1)
|
||||||
|
return max(e for _, e in spans) - min(s for s, _ in spans)
|
||||||
|
|
||||||
|
|
||||||
|
def warm(path):
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
while fh.read(1 << 24):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def evict(path):
|
||||||
|
fd = os.open(path, os.O_RDONLY)
|
||||||
|
try:
|
||||||
|
os.posix_fadvise(fd, 0, 0, os.POSIX_FADV_DONTNEED)
|
||||||
|
finally:
|
||||||
|
os.close(fd)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
|
||||||
|
ap.add_argument("--dir", default="concurrent-read-data")
|
||||||
|
ap.add_argument("--executor", choices=["threads", "processes"], default="threads")
|
||||||
|
ap.add_argument("--threads", default="1,2,4,8,16")
|
||||||
|
ap.add_argument("--reps", type=int, default=3)
|
||||||
|
ap.add_argument("--slab", type=int, default=256)
|
||||||
|
ap.add_argument("--slabs", type=int, default=1024)
|
||||||
|
ap.add_argument("--seed", type=int, default=42)
|
||||||
|
ap.add_argument("--cold", action="store_true")
|
||||||
|
ap.add_argument("--modes", default="distinct,same")
|
||||||
|
ap.add_argument("--layouts", default="deflate,contiguous")
|
||||||
|
ap.add_argument("--json")
|
||||||
|
a = ap.parse_args()
|
||||||
|
|
||||||
|
# The Rust harness pins this value (splitmix64_reference).
|
||||||
|
assert splitmix64(42)[1] == 0xBDD732262FEB6E95, "splitmix64 port is wrong"
|
||||||
|
|
||||||
|
try:
|
||||||
|
with open(os.path.join(a.dir, "manifest.json")) as fh:
|
||||||
|
m = json.load(fh)
|
||||||
|
except FileNotFoundError:
|
||||||
|
sys.exit(f"{a.dir}/manifest.json not found: generate the files with "
|
||||||
|
"`cargo run --release -p clawhdf5-bench --bin concurrent_read -- --dir ...` first")
|
||||||
|
threads_list = [int(x) for x in a.threads.split(",")]
|
||||||
|
modes = a.modes.split(",")
|
||||||
|
layouts = a.layouts.split(",")
|
||||||
|
if a.slab < 1 or a.slab > min(m["rows"], m["cols"]):
|
||||||
|
sys.exit(f"--slab must be 1..={min(m['rows'], m['cols'])}")
|
||||||
|
files = dict(m["files"])
|
||||||
|
slabs = slab_offsets(a.seed, a.slabs, m["rows"], m["cols"], a.slab)
|
||||||
|
dataset_bytes = m["rows"] * m["cols"] * 4
|
||||||
|
tool = f"h5py-{a.executor}"
|
||||||
|
|
||||||
|
ctx = mp.get_context("spawn") # never fork a process holding HDF5 state
|
||||||
|
pools = {}
|
||||||
|
if a.executor == "processes":
|
||||||
|
for t in threads_list:
|
||||||
|
pool = ctx.Pool(t, initializer=_init, initargs=(ctx.Barrier(t),))
|
||||||
|
pool.map(_noop, range(t)) # start the workers outside the timing
|
||||||
|
pools[t] = pool
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
print("| layout | mode | threads | MB/s | efficiency | median s |")
|
||||||
|
print("|---|---|---:|---:|---:|---:|")
|
||||||
|
try:
|
||||||
|
for layout in layouts:
|
||||||
|
path = os.path.join(a.dir, files[layout])
|
||||||
|
if not a.cold:
|
||||||
|
warm(path)
|
||||||
|
for mode in modes:
|
||||||
|
with h5py.File(path, "r") as f: # untimed, checked pass
|
||||||
|
work(f, mode, 0, 1, m, slabs, a.slab, True)
|
||||||
|
nbytes = (dataset_bytes * m["datasets"] if mode == "distinct"
|
||||||
|
else a.slab * a.slab * 4 * a.slabs)
|
||||||
|
base = None
|
||||||
|
for t in threads_list:
|
||||||
|
times = []
|
||||||
|
for _ in range(a.reps):
|
||||||
|
if a.cold:
|
||||||
|
evict(path)
|
||||||
|
if a.executor == "threads":
|
||||||
|
times.append(run_threads(path, mode, t, m, slabs, a.slab))
|
||||||
|
else:
|
||||||
|
times.append(run_processes(pools[t], path, mode, t, m, slabs, a.slab))
|
||||||
|
med = sorted(times)[len(times) // 2]
|
||||||
|
mb_s = nbytes / (1 << 20) / med
|
||||||
|
if t == 1:
|
||||||
|
base = mb_s
|
||||||
|
eff = mb_s / (t * base) if base else None
|
||||||
|
print(f"| {layout} | {mode} | {t} | {mb_s:.0f} | "
|
||||||
|
f"{'-' if eff is None else f'{eff:.2f}'} | {med:.4f} |")
|
||||||
|
rows.append({
|
||||||
|
"layout": layout, "mode": mode, "threads": t, "bytes": nbytes,
|
||||||
|
"times_s": times, "median_s": med, "mb_s": mb_s, "efficiency": eff,
|
||||||
|
})
|
||||||
|
finally:
|
||||||
|
for pool in pools.values():
|
||||||
|
pool.terminate()
|
||||||
|
|
||||||
|
if a.json:
|
||||||
|
doc = {
|
||||||
|
"tool": tool,
|
||||||
|
"version": h5py.__version__,
|
||||||
|
"hdf5_version": h5py.version.hdf5_version,
|
||||||
|
"python": platform.python_version(),
|
||||||
|
"host": socket.gethostname(),
|
||||||
|
"cpus": os.cpu_count(),
|
||||||
|
"unix_time": int(time.time()),
|
||||||
|
"cache": ("cold (posix_fadvise DONTNEED before each repetition)"
|
||||||
|
if a.cold else "warm"),
|
||||||
|
"decode_threads": 1,
|
||||||
|
"params": {
|
||||||
|
"datasets": m["datasets"], "rows": m["rows"], "cols": m["cols"],
|
||||||
|
"chunk": m["chunk"], "deflate_level": m["deflate_level"],
|
||||||
|
"mib": dataset_bytes // (1 << 20), "slab": a.slab, "slabs": a.slabs,
|
||||||
|
"seed": a.seed, "reps": a.reps, "dir": a.dir,
|
||||||
|
},
|
||||||
|
"results": rows,
|
||||||
|
}
|
||||||
|
with open(a.json, "w") as fh:
|
||||||
|
json.dump(doc, fh, indent=2)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -0,0 +1,523 @@
|
|||||||
|
//! Concurrent-read harness: how does decoded read throughput scale with the
|
||||||
|
//! number of threads reading one open file?
|
||||||
|
//!
|
||||||
|
//! libhdf5 (threadsafe build) serialises every API call under one global
|
||||||
|
//! mutex, and h5py holds it too, so threads cannot decode in parallel there.
|
||||||
|
//! A clawhdf5 [`File`] is `Send + Sync`; this harness measures what that buys.
|
||||||
|
//! `crates/clawhdf5-bench/scripts/concurrent_read_h5py.py` runs the same
|
||||||
|
//! workload on the same files with h5py (threads, and processes), and
|
||||||
|
//! `compare_concurrent_read.py` tabulates the JSON both write.
|
||||||
|
//!
|
||||||
|
//! Files (generated on first use, reused while `manifest.json` matches):
|
||||||
|
//!
|
||||||
|
//! * `<dir>/deflate.h5`: `--datasets` datasets `d00`, `d01`, ... of `f32`,
|
||||||
|
//! `--mib` MiB decoded each, shape `[mib * 256, 1024]`, chunks `256 x 256`,
|
||||||
|
//! deflate level 4.
|
||||||
|
//! * `<dir>/contiguous.h5`: the same datasets, contiguous.
|
||||||
|
//!
|
||||||
|
//! Modes, for each layout and each thread count `T` (strong scaling: the total
|
||||||
|
//! work per repetition is fixed, split among the threads):
|
||||||
|
//!
|
||||||
|
//! * `distinct`: every dataset is read in full once; thread `t` reads datasets
|
||||||
|
//! `t, t + T, t + 2T, ...`.
|
||||||
|
//! * `same`: all threads read `d00`, `--slabs` random `--slab` x `--slab`
|
||||||
|
//! hyperslabs in total (slab `j` goes to thread `j % T`). The offsets come
|
||||||
|
//! from a splitmix64 stream seeded with `--seed`, identical in the h5py
|
||||||
|
//! script.
|
||||||
|
//!
|
||||||
|
//! One `File` per layout per repetition is shared by all threads (opened
|
||||||
|
//! fresh each repetition, so no chunk cache carries over). Page cache:
|
||||||
|
//! `warm` (default) reads every file once before timing; `--cold` evicts the
|
||||||
|
//! files from the page cache with `posix_fadvise(POSIX_FADV_DONTNEED)` before
|
||||||
|
//! every repetition (no root needed; it only evicts clean, unmapped pages, so
|
||||||
|
//! it is best effort — the JSON says which was used).
|
||||||
|
//!
|
||||||
|
//! Decode inside one read is itself parallel when clawhdf5-format's `parallel`
|
||||||
|
//! feature is on (it is in this binary, via clawhdf5-agent). `--decode-threads
|
||||||
|
//! N` sizes that rayon pool; `--decode-threads 1` measures the API's own
|
||||||
|
//! thread scaling, comparable with h5py where each call decodes on the
|
||||||
|
//! calling thread.
|
||||||
|
//!
|
||||||
|
//! ```text
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \
|
||||||
|
//! --dir /data/concurrent-read --json clawhdf5.json
|
||||||
|
//! cargo run --release -p clawhdf5-bench --bin concurrent_read -- \
|
||||||
|
//! --dir /tmp/cr --datasets 4 --mib 1 --threads 1,2 --slabs 16 --reps 1 # smoke
|
||||||
|
//! ```
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::sync::Barrier;
|
||||||
|
use std::time::Instant;
|
||||||
|
|
||||||
|
use clawhdf5::{File, FileBuilder, Selection};
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
|
||||||
|
const COLS: u64 = 1024;
|
||||||
|
const ROWS_PER_MIB: u64 = 256; // 256 rows x 1024 cols x 4 bytes = 1 MiB
|
||||||
|
const CHUNK: u64 = 256;
|
||||||
|
const DEFLATE_LEVEL: u32 = 4;
|
||||||
|
const LAYOUTS: [&str; 2] = ["deflate", "contiguous"];
|
||||||
|
const MANIFEST_VERSION: u32 = 1;
|
||||||
|
|
||||||
|
/// splitmix64 — shared with the h5py script, which must produce the same
|
||||||
|
/// stream (both the data and the hyperslab offsets depend on it).
|
||||||
|
fn splitmix64(state: &mut u64) -> u64 {
|
||||||
|
*state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||||
|
let mut z = *state;
|
||||||
|
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||||
|
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||||
|
z ^ (z >> 31)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Element `i` (row-major) of dataset `k`: a slowly varying integer part plus
|
||||||
|
/// 8 bits of noise, so deflate has real work to do (about 3.1x) and every value
|
||||||
|
/// is exact in `f32` (< 2^15 with 8 fraction bits), which lets both harnesses
|
||||||
|
/// check what they read against this formula.
|
||||||
|
fn value(k: u64, i: u64) -> f32 {
|
||||||
|
let mut s = i ^ (k << 40);
|
||||||
|
let noise = splitmix64(&mut s) & 0xff;
|
||||||
|
(((i >> 6) % 16384) + k) as f32 + noise as f32 / 256.0
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize, Deserialize, PartialEq, Debug, Clone)]
|
||||||
|
struct Manifest {
|
||||||
|
version: u32,
|
||||||
|
datasets: u64,
|
||||||
|
rows: u64,
|
||||||
|
cols: u64,
|
||||||
|
chunk: [u64; 2],
|
||||||
|
deflate_level: u32,
|
||||||
|
files: Vec<(String, String)>, // (layout, file name)
|
||||||
|
writer: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn manifest_for(datasets: u64, mib: u64) -> Manifest {
|
||||||
|
Manifest {
|
||||||
|
version: MANIFEST_VERSION,
|
||||||
|
datasets,
|
||||||
|
rows: mib * ROWS_PER_MIB,
|
||||||
|
cols: COLS,
|
||||||
|
chunk: [CHUNK, CHUNK],
|
||||||
|
deflate_level: DEFLATE_LEVEL,
|
||||||
|
files: LAYOUTS
|
||||||
|
.iter()
|
||||||
|
.map(|l| (l.to_string(), format!("{l}.h5")))
|
||||||
|
.collect(),
|
||||||
|
writer: format!("clawhdf5 {}", env!("CARGO_PKG_VERSION")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dataset_values(k: u64, n: u64) -> Vec<f32> {
|
||||||
|
(0..n).map(|i| value(k, i)).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the files unless `dir` already holds ones matching `want`.
|
||||||
|
fn ensure_files(dir: &Path, want: &Manifest) -> std::io::Result<bool> {
|
||||||
|
let manifest_path = dir.join("manifest.json");
|
||||||
|
if let Ok(text) = std::fs::read_to_string(&manifest_path)
|
||||||
|
&& let Ok(have) = serde_json::from_str::<Manifest>(&text)
|
||||||
|
&& have.version == want.version
|
||||||
|
&& have.datasets == want.datasets
|
||||||
|
&& have.rows == want.rows
|
||||||
|
&& have.cols == want.cols
|
||||||
|
&& have.chunk == want.chunk
|
||||||
|
&& have.deflate_level == want.deflate_level
|
||||||
|
&& have.files == want.files
|
||||||
|
&& want.files.iter().all(|(_, f)| dir.join(f).exists())
|
||||||
|
{
|
||||||
|
return Ok(false);
|
||||||
|
}
|
||||||
|
std::fs::create_dir_all(dir)?;
|
||||||
|
// A stale manifest must not survive a half-written regeneration.
|
||||||
|
let _ = std::fs::remove_file(&manifest_path);
|
||||||
|
let n = want.rows * want.cols;
|
||||||
|
for (layout, file) in &want.files {
|
||||||
|
// One layout at a time keeps the peak memory to about twice one
|
||||||
|
// file's decoded size.
|
||||||
|
let mut b = FileBuilder::new();
|
||||||
|
for k in 0..want.datasets {
|
||||||
|
let ds = b.create_dataset(&format!("d{k:02}"));
|
||||||
|
ds.with_f32_data(&dataset_values(k, n))
|
||||||
|
.with_shape(&[want.rows, want.cols]);
|
||||||
|
if layout == "deflate" {
|
||||||
|
ds.with_chunks(&[CHUNK.min(want.rows), CHUNK])
|
||||||
|
.with_deflate(DEFLATE_LEVEL);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
b.write(dir.join(file)).map_err(std::io::Error::other)?;
|
||||||
|
}
|
||||||
|
std::fs::write(
|
||||||
|
&manifest_path,
|
||||||
|
serde_json::to_string_pretty(want).map_err(std::io::Error::other)?,
|
||||||
|
)?;
|
||||||
|
Ok(true)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn slab_offsets(seed: u64, count: usize, rows: u64, cols: u64, slab: u64) -> Vec<(u64, u64)> {
|
||||||
|
let mut s = seed;
|
||||||
|
(0..count)
|
||||||
|
.map(|_| {
|
||||||
|
let r = splitmix64(&mut s) % (rows - slab + 1);
|
||||||
|
let c = splitmix64(&mut s) % (cols - slab + 1);
|
||||||
|
(r, c)
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Warm the page cache by reading every byte of `path`.
|
||||||
|
fn warm(path: &Path) -> std::io::Result<()> {
|
||||||
|
let mut f = std::fs::File::open(path)?;
|
||||||
|
std::io::copy(&mut f, &mut std::io::sink())?;
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the kernel to drop `path`'s pages from the page cache.
|
||||||
|
fn evict(path: &Path) -> std::io::Result<()> {
|
||||||
|
use std::os::fd::AsRawFd;
|
||||||
|
let f = std::fs::File::open(path)?;
|
||||||
|
// SAFETY: plain syscall on a valid, open file descriptor.
|
||||||
|
let rc = unsafe { libc::posix_fadvise(f.as_raw_fd(), 0, 0, libc::POSIX_FADV_DONTNEED) };
|
||||||
|
if rc != 0 {
|
||||||
|
return Err(std::io::Error::from_raw_os_error(rc));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize)]
|
||||||
|
struct Row {
|
||||||
|
layout: String,
|
||||||
|
mode: String,
|
||||||
|
threads: usize,
|
||||||
|
/// Decoded (selected) bytes read per repetition.
|
||||||
|
bytes: u64,
|
||||||
|
times_s: Vec<f64>,
|
||||||
|
median_s: f64,
|
||||||
|
mb_s: f64,
|
||||||
|
/// `mb_s / (threads * mb_s at threads = 1)`; null without a 1-thread row.
|
||||||
|
efficiency: Option<f64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Args {
|
||||||
|
dir: PathBuf,
|
||||||
|
datasets: u64,
|
||||||
|
mib: u64,
|
||||||
|
threads: Vec<usize>,
|
||||||
|
reps: usize,
|
||||||
|
slab: u64,
|
||||||
|
slabs: usize,
|
||||||
|
seed: u64,
|
||||||
|
cold: bool,
|
||||||
|
decode_threads: usize,
|
||||||
|
modes: Vec<String>,
|
||||||
|
layouts: Vec<String>,
|
||||||
|
json: Option<PathBuf>,
|
||||||
|
}
|
||||||
|
|
||||||
|
const USAGE: &str = "\
|
||||||
|
usage: concurrent_read [--dir DIR] [--datasets N] [--mib N] [--threads 1,2,4,8,16]
|
||||||
|
[--reps N] [--slab N] [--slabs N] [--seed N] [--cold]
|
||||||
|
[--decode-threads N] [--modes distinct,same]
|
||||||
|
[--layouts deflate,contiguous] [--json FILE]";
|
||||||
|
|
||||||
|
fn parse_list<T: std::str::FromStr>(s: &str) -> Result<Vec<T>, String> {
|
||||||
|
s.split(',')
|
||||||
|
.map(|x| x.trim().parse().map_err(|_| format!("bad list item {x:?}")))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_args() -> Result<Args, String> {
|
||||||
|
let mut a = Args {
|
||||||
|
dir: PathBuf::from("concurrent-read-data"),
|
||||||
|
datasets: 64,
|
||||||
|
mib: 64,
|
||||||
|
threads: vec![1, 2, 4, 8, 16],
|
||||||
|
reps: 3,
|
||||||
|
slab: 256,
|
||||||
|
slabs: 1024,
|
||||||
|
seed: 42,
|
||||||
|
cold: false,
|
||||||
|
decode_threads: 0,
|
||||||
|
modes: vec!["distinct".into(), "same".into()],
|
||||||
|
layouts: LAYOUTS.iter().map(|s| s.to_string()).collect(),
|
||||||
|
json: None,
|
||||||
|
};
|
||||||
|
let mut it = std::env::args().skip(1);
|
||||||
|
while let Some(flag) = it.next() {
|
||||||
|
if flag == "--cold" {
|
||||||
|
a.cold = true;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if flag == "-h" || flag == "--help" {
|
||||||
|
return Err(USAGE.into());
|
||||||
|
}
|
||||||
|
let v = it.next().ok_or(format!("{flag} needs a value\n{USAGE}"))?;
|
||||||
|
let num = |v: &str| {
|
||||||
|
v.parse::<u64>()
|
||||||
|
.map_err(|_| format!("{flag}: bad number {v:?}"))
|
||||||
|
};
|
||||||
|
match flag.as_str() {
|
||||||
|
"--dir" => a.dir = v.into(),
|
||||||
|
"--datasets" => a.datasets = num(&v)?,
|
||||||
|
"--mib" => a.mib = num(&v)?,
|
||||||
|
"--threads" => a.threads = parse_list(&v)?,
|
||||||
|
"--reps" => a.reps = num(&v)? as usize,
|
||||||
|
"--slab" => a.slab = num(&v)?,
|
||||||
|
"--slabs" => a.slabs = num(&v)? as usize,
|
||||||
|
"--seed" => a.seed = num(&v)?,
|
||||||
|
"--decode-threads" => a.decode_threads = num(&v)? as usize,
|
||||||
|
"--modes" => a.modes = parse_list(&v)?,
|
||||||
|
"--layouts" => a.layouts = parse_list(&v)?,
|
||||||
|
"--json" => a.json = Some(v.into()),
|
||||||
|
_ => return Err(format!("unknown flag {flag}\n{USAGE}")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if a.datasets == 0 || a.datasets > 100 {
|
||||||
|
return Err("--datasets must be 1..=100".into());
|
||||||
|
}
|
||||||
|
if a.mib == 0 || a.reps == 0 || a.slabs == 0 || a.threads.contains(&0) {
|
||||||
|
return Err("--mib, --reps, --slabs and every --threads value must be > 0".into());
|
||||||
|
}
|
||||||
|
if a.slab == 0 || a.slab > COLS || a.slab > a.mib * ROWS_PER_MIB {
|
||||||
|
return Err(format!(
|
||||||
|
"--slab must be 1..={}",
|
||||||
|
COLS.min(a.mib * ROWS_PER_MIB)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for m in &a.modes {
|
||||||
|
if m != "distinct" && m != "same" {
|
||||||
|
return Err(format!("unknown mode {m:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for l in &a.layouts {
|
||||||
|
if !LAYOUTS.contains(&l.as_str()) {
|
||||||
|
return Err(format!("unknown layout {l:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(a)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One timed repetition: `T` threads on one shared `File`. Returns seconds.
|
||||||
|
fn run_once(
|
||||||
|
path: &Path,
|
||||||
|
mode: &str,
|
||||||
|
threads: usize,
|
||||||
|
m: &Manifest,
|
||||||
|
slabs: &[(u64, u64)],
|
||||||
|
slab: u64,
|
||||||
|
verify: bool,
|
||||||
|
) -> f64 {
|
||||||
|
let file = File::open(path).expect("open");
|
||||||
|
let barrier = Barrier::new(threads + 1); // + the spawning thread
|
||||||
|
let n = m.rows * m.cols;
|
||||||
|
// Each thread times itself from the barrier; the repetition spans the
|
||||||
|
// earliest start to the latest finish (timing on the spawning thread
|
||||||
|
// instead undercounts whenever it is scheduled after the workers ran).
|
||||||
|
let spans: Vec<(Instant, Instant)> = std::thread::scope(|s| {
|
||||||
|
let handles: Vec<_> = (0..threads)
|
||||||
|
.map(|t| {
|
||||||
|
let (file, barrier) = (&file, &barrier);
|
||||||
|
s.spawn(move || {
|
||||||
|
barrier.wait();
|
||||||
|
let start = Instant::now();
|
||||||
|
match mode {
|
||||||
|
"distinct" => {
|
||||||
|
for k in (t as u64..m.datasets).step_by(threads) {
|
||||||
|
let got = file.dataset(&format!("d{k:02}")).unwrap().read_f32();
|
||||||
|
let got = got.unwrap();
|
||||||
|
assert_eq!(got.len() as u64, n);
|
||||||
|
if verify {
|
||||||
|
for i in [0, n / 3, n - 1] {
|
||||||
|
assert_eq!(got[i as usize], value(k, i), "d{k:02}[{i}]");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
std::hint::black_box(got);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
let ds = file.dataset("d00").unwrap();
|
||||||
|
for &(r, c) in slabs.iter().skip(t).step_by(threads) {
|
||||||
|
let sel = Selection::Hyperslab {
|
||||||
|
start: vec![r, c],
|
||||||
|
stride: vec![1, 1],
|
||||||
|
count: vec![slab, slab],
|
||||||
|
block: vec![1, 1],
|
||||||
|
};
|
||||||
|
let got = ds.read_f32_selection(&sel).unwrap();
|
||||||
|
assert_eq!(got.len() as u64, slab * slab);
|
||||||
|
if verify {
|
||||||
|
let last = (r + slab - 1) * m.cols + c + slab - 1;
|
||||||
|
assert_eq!(got[0], value(0, r * m.cols + c));
|
||||||
|
assert_eq!(*got.last().unwrap(), value(0, last));
|
||||||
|
}
|
||||||
|
std::hint::black_box(got);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
(start, Instant::now())
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
barrier.wait();
|
||||||
|
handles.into_iter().map(|h| h.join().unwrap()).collect()
|
||||||
|
});
|
||||||
|
let start = spans.iter().map(|s| s.0).min().unwrap();
|
||||||
|
let end = spans.iter().map(|s| s.1).max().unwrap();
|
||||||
|
(end - start).as_secs_f64()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn median(v: &[f64]) -> f64 {
|
||||||
|
let mut s = v.to_vec();
|
||||||
|
s.sort_by(f64::total_cmp);
|
||||||
|
s[s.len() / 2]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hostname() -> String {
|
||||||
|
std::fs::read_to_string("/proc/sys/kernel/hostname")
|
||||||
|
.map(|s| s.trim().to_string())
|
||||||
|
.unwrap_or_else(|_| "unknown".into())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
let args = match parse_args() {
|
||||||
|
Ok(a) => a,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("{e}");
|
||||||
|
std::process::exit(2);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
if cfg!(debug_assertions) {
|
||||||
|
eprintln!("warning: debug build — numbers are meaningless. Use --release.");
|
||||||
|
}
|
||||||
|
if args.decode_threads > 0 {
|
||||||
|
rayon::ThreadPoolBuilder::new()
|
||||||
|
.num_threads(args.decode_threads)
|
||||||
|
.build_global()
|
||||||
|
.expect("configure rayon pool");
|
||||||
|
}
|
||||||
|
|
||||||
|
let manifest = manifest_for(args.datasets, args.mib);
|
||||||
|
let t = Instant::now();
|
||||||
|
match ensure_files(&args.dir, &manifest) {
|
||||||
|
Ok(true) => eprintln!(
|
||||||
|
"generated {} in {:.1} s",
|
||||||
|
args.dir.display(),
|
||||||
|
t.elapsed().as_secs_f64()
|
||||||
|
),
|
||||||
|
Ok(false) => eprintln!("reusing {}", args.dir.display()),
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("cannot write test files in {}: {e}", args.dir.display());
|
||||||
|
std::process::exit(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let path_of = |layout: &str| args.dir.join(format!("{layout}.h5"));
|
||||||
|
let slabs = slab_offsets(
|
||||||
|
args.seed,
|
||||||
|
args.slabs,
|
||||||
|
manifest.rows,
|
||||||
|
manifest.cols,
|
||||||
|
args.slab,
|
||||||
|
);
|
||||||
|
let dataset_bytes = manifest.rows * manifest.cols * 4;
|
||||||
|
|
||||||
|
let mut rows: Vec<Row> = Vec::new();
|
||||||
|
println!("| layout | mode | threads | MB/s | efficiency | median s |");
|
||||||
|
println!("|---|---|---:|---:|---:|---:|");
|
||||||
|
for layout in &args.layouts {
|
||||||
|
let path = path_of(layout);
|
||||||
|
// Untimed pass: page cache warm (unless --cold), results checked.
|
||||||
|
if !args.cold {
|
||||||
|
warm(&path).expect("warm page cache");
|
||||||
|
}
|
||||||
|
for mode in &args.modes {
|
||||||
|
run_once(&path, mode, 1, &manifest, &slabs, args.slab, true);
|
||||||
|
let bytes = match mode.as_str() {
|
||||||
|
"distinct" => dataset_bytes * manifest.datasets,
|
||||||
|
_ => args.slab * args.slab * 4 * args.slabs as u64,
|
||||||
|
};
|
||||||
|
let mut base: Option<f64> = None;
|
||||||
|
for &threads in &args.threads {
|
||||||
|
let times: Vec<f64> = (0..args.reps)
|
||||||
|
.map(|_| {
|
||||||
|
if args.cold {
|
||||||
|
evict(&path).expect("posix_fadvise");
|
||||||
|
}
|
||||||
|
run_once(&path, mode, threads, &manifest, &slabs, args.slab, false)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let med = median(×);
|
||||||
|
let mb_s = bytes as f64 / (1 << 20) as f64 / med;
|
||||||
|
if threads == 1 {
|
||||||
|
base = Some(mb_s);
|
||||||
|
}
|
||||||
|
let efficiency = base.map(|b| mb_s / (threads as f64 * b));
|
||||||
|
println!(
|
||||||
|
"| {layout} | {mode} | {threads} | {mb_s:.0} | {} | {med:.4} |",
|
||||||
|
efficiency.map_or("-".into(), |e| format!("{e:.2}"))
|
||||||
|
);
|
||||||
|
rows.push(Row {
|
||||||
|
layout: layout.clone(),
|
||||||
|
mode: mode.clone(),
|
||||||
|
threads,
|
||||||
|
bytes,
|
||||||
|
times_s: times,
|
||||||
|
median_s: med,
|
||||||
|
mb_s,
|
||||||
|
efficiency,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if let Some(out) = &args.json {
|
||||||
|
let doc = serde_json::json!({
|
||||||
|
"tool": "clawhdf5",
|
||||||
|
"version": env!("CARGO_PKG_VERSION"),
|
||||||
|
"host": hostname(),
|
||||||
|
"cpus": std::thread::available_parallelism().map_or(0, |n| n.get()),
|
||||||
|
"unix_time": std::time::SystemTime::now()
|
||||||
|
.duration_since(std::time::UNIX_EPOCH)
|
||||||
|
.map_or(0, |d| d.as_secs()),
|
||||||
|
"cache": if args.cold { "cold (posix_fadvise DONTNEED before each repetition)" } else { "warm" },
|
||||||
|
"decode_threads": rayon::current_num_threads(),
|
||||||
|
"params": {
|
||||||
|
"datasets": manifest.datasets,
|
||||||
|
"mib": args.mib,
|
||||||
|
"rows": manifest.rows,
|
||||||
|
"cols": manifest.cols,
|
||||||
|
"chunk": manifest.chunk,
|
||||||
|
"deflate_level": manifest.deflate_level,
|
||||||
|
"slab": args.slab,
|
||||||
|
"slabs": args.slabs,
|
||||||
|
"seed": args.seed,
|
||||||
|
"reps": args.reps,
|
||||||
|
"dir": args.dir,
|
||||||
|
},
|
||||||
|
"results": rows,
|
||||||
|
});
|
||||||
|
std::fs::write(out, serde_json::to_string_pretty(&doc).unwrap()).expect("write json");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_are_exact_in_f32() {
|
||||||
|
for k in [0, 7, 63] {
|
||||||
|
for i in [0u64, 1, 4095, 1 << 20, (1 << 24) - 1] {
|
||||||
|
let v = value(k, i);
|
||||||
|
assert_eq!(v, (v as f64) as f32);
|
||||||
|
assert!(v < 32768.0);
|
||||||
|
assert_eq!((v * 256.0).fract(), 0.0);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The h5py script hard-codes this vector to check its splitmix64 port.
|
||||||
|
#[test]
|
||||||
|
fn splitmix64_reference() {
|
||||||
|
let mut s = 42;
|
||||||
|
assert_eq!(splitmix64(&mut s), 0xBDD7_3226_2FEB_6E95);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,148 @@
|
|||||||
|
//! Keeps the concurrent-read harnesses working: runs `concurrent_read`, the
|
||||||
|
//! h5py script (threads and processes) and the comparison script end to end
|
||||||
|
//! on tiny files. h5py reading the files also checks, element by element at
|
||||||
|
//! spot positions, that both harnesses generate the same data and slabs.
|
||||||
|
//!
|
||||||
|
//! The h5py half is skipped when python3 with h5py is unavailable, unless
|
||||||
|
//! `CLAWHDF5_REQUIRE_INTEROP=1`; `CLAWHDF5_PYTHON` picks the interpreter.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::process::Command;
|
||||||
|
|
||||||
|
fn python() -> String {
|
||||||
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn interop_required() -> bool {
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn python_available() -> bool {
|
||||||
|
Command::new(python())
|
||||||
|
.args(["-c", "import h5py, numpy"])
|
||||||
|
.output()
|
||||||
|
.map(|o| o.status.success())
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn scripts() -> PathBuf {
|
||||||
|
Path::new(env!("CARGO_MANIFEST_DIR")).join("scripts")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run(cmd: &mut Command) -> String {
|
||||||
|
let out = cmd.output().expect("spawn");
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"{cmd:?} failed\nSTDOUT:\n{}\nSTDERR:\n{}",
|
||||||
|
String::from_utf8_lossy(&out.stdout),
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
String::from_utf8_lossy(&out.stdout).into_owned()
|
||||||
|
}
|
||||||
|
|
||||||
|
const SMALL: [&str; 8] = [
|
||||||
|
"--threads",
|
||||||
|
"1,2",
|
||||||
|
"--slabs",
|
||||||
|
"8",
|
||||||
|
"--reps",
|
||||||
|
"1",
|
||||||
|
"--slab",
|
||||||
|
"64",
|
||||||
|
];
|
||||||
|
|
||||||
|
fn results(path: &Path) -> serde_json::Value {
|
||||||
|
serde_json::from_str(&std::fs::read_to_string(path).unwrap()).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn harnesses_run_end_to_end_on_tiny_files() {
|
||||||
|
let dir = tempfile::TempDir::new().unwrap();
|
||||||
|
let data = dir.path().join("data");
|
||||||
|
let claw = dir.path().join("claw.json");
|
||||||
|
|
||||||
|
let bin = env!("CARGO_BIN_EXE_concurrent_read");
|
||||||
|
run(Command::new(bin)
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--datasets", "3", "--mib", "1"])
|
||||||
|
.args(SMALL)
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&claw));
|
||||||
|
// Second run reuses the files (and exercises --cold).
|
||||||
|
let out = Command::new(bin)
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--datasets", "3", "--mib", "1", "--cold"])
|
||||||
|
.args(SMALL)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(out.status.success());
|
||||||
|
assert!(String::from_utf8_lossy(&out.stderr).contains("reusing"));
|
||||||
|
|
||||||
|
let doc = results(&claw);
|
||||||
|
assert_eq!(doc["tool"], "clawhdf5");
|
||||||
|
// 2 layouts x 2 modes x 2 thread counts.
|
||||||
|
assert_eq!(doc["results"].as_array().unwrap().len(), 8);
|
||||||
|
for r in doc["results"].as_array().unwrap() {
|
||||||
|
assert!(r["mb_s"].as_f64().unwrap() > 0.0, "{r}");
|
||||||
|
}
|
||||||
|
|
||||||
|
if !python_available() {
|
||||||
|
assert!(
|
||||||
|
!interop_required(),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but {} has no h5py",
|
||||||
|
python()
|
||||||
|
);
|
||||||
|
eprintln!("skipping the h5py half: no h5py in {}", python());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let mut jsons = vec![claw];
|
||||||
|
for executor in ["threads", "processes"] {
|
||||||
|
let out = dir.path().join(format!("h5py-{executor}.json"));
|
||||||
|
run(Command::new(python())
|
||||||
|
.arg(scripts().join("concurrent_read_h5py.py"))
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args(["--executor", executor])
|
||||||
|
.args(SMALL)
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&out));
|
||||||
|
let doc = results(&out);
|
||||||
|
assert_eq!(doc["tool"], format!("h5py-{executor}"));
|
||||||
|
assert_eq!(doc["results"].as_array().unwrap().len(), 8);
|
||||||
|
jsons.push(out);
|
||||||
|
}
|
||||||
|
let table = run(Command::new(python())
|
||||||
|
.arg(scripts().join("compare_concurrent_read.py"))
|
||||||
|
.args(&jsons));
|
||||||
|
assert!(table.contains("| deflate | same | 2 |"), "{table}");
|
||||||
|
assert!(table.contains("clawhdf5 / h5py-processes"), "{table}");
|
||||||
|
|
||||||
|
// A different workload must not be compared.
|
||||||
|
let other = dir.path().join("other.json");
|
||||||
|
run(Command::new(python())
|
||||||
|
.arg(scripts().join("concurrent_read_h5py.py"))
|
||||||
|
.arg("--dir")
|
||||||
|
.arg(&data)
|
||||||
|
.args([
|
||||||
|
"--threads",
|
||||||
|
"1",
|
||||||
|
"--slabs",
|
||||||
|
"4",
|
||||||
|
"--reps",
|
||||||
|
"1",
|
||||||
|
"--slab",
|
||||||
|
"64",
|
||||||
|
])
|
||||||
|
.arg("--json")
|
||||||
|
.arg(&other));
|
||||||
|
let out = Command::new(python())
|
||||||
|
.arg(scripts().join("compare_concurrent_read.py"))
|
||||||
|
.arg(&jsons[0])
|
||||||
|
.arg(&other)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(!out.status.success());
|
||||||
|
assert!(String::from_utf8_lossy(&out.stderr).contains("slabs"));
|
||||||
|
}
|
||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# clawhdf5-cli
|
||||||
|
|
||||||
|
The `clawhdf5` command: create, fill, search and inspect a
|
||||||
|
[`clawhdf5-agent`](../clawhdf5-agent/README.md) memory store from the
|
||||||
|
shell. Output is JSON. (For general HDF5 files use `h5rs` from
|
||||||
|
[`clawhdf5-tools`](../clawhdf5-tools/README.md).)
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cargo install --path crates/clawhdf5-cli # installs `clawhdf5`; not on crates.io yet
|
||||||
|
# or: cargo run -p clawhdf5-cli -- --help
|
||||||
|
```
|
||||||
|
|
||||||
|
No C is compiled.
|
||||||
|
|
||||||
|
## Commands
|
||||||
|
|
||||||
|
The store is `--path FILE` (or `CLAWHDF5_PATH`) before the subcommand.
|
||||||
|
|
||||||
|
| Command | What |
|
||||||
|
|---|---|
|
||||||
|
| `create [--agent-id ID] [--dim N] [--wal] [--f32] [--f32-index]` | a new store (dimension 384 by default); float16 embeddings and an int8 index copy unless `--f32` / `--f32-index`. The WAL is off unless `--wal` (the library's default is on), so each save is checkpointed at once |
|
||||||
|
| `save [--json '{...}']` | save one entry, from `--json` or stdin: `{"chunk", "embedding", "source_channel", "timestamp", "session_id", "tags"}` |
|
||||||
|
| `search --embedding '[...]' [--query TEXT] [-k N] [--vector-weight W] [--keyword-weight W]` | hybrid search (defaults 5 results, weights 0.7 / 0.3) |
|
||||||
|
| `recall INDEX` | one entry by index |
|
||||||
|
| `stats` | counts and configuration |
|
||||||
|
| `flush-wal` | checkpoint the WAL into the `.h5` |
|
||||||
|
| `agents-md [--output FILE]` | generate an `AGENTS.md` from the store |
|
||||||
|
| `export` | every entry as JSON lines |
|
||||||
|
| `snapshot DEST` | a copy of the store's `.h5` file |
|
||||||
|
| `keygen --out FILE` | a new Ed25519 signing key (64 hex characters, created owner-only on Unix) |
|
||||||
|
| `verify --public-key HEX_OR_FILE` | check a signed store; exit status 2 if it does not verify |
|
||||||
|
|
||||||
|
`recall`, `stats`, `agents-md` and `export` open the store read-only
|
||||||
|
(no lock, nothing written), so they work while another process has it
|
||||||
|
open. `save`, `search` (which records activation boosts) and `flush-wal`
|
||||||
|
open it for writing and take the store's lock. With
|
||||||
|
`--signing-key FILE` (or `CLAWHDF5_SIGNING_KEY`) every checkpoint a command
|
||||||
|
makes is signed; a signed store refuses to checkpoint without the key.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
clawhdf5 --path mem.h5 create --agent-id demo --dim 3
|
||||||
|
echo '{"chunk":"hello","embedding":[0.1,0.2,0.3],"source_channel":"cli","timestamp":0,"session_id":"s1","tags":""}' \
|
||||||
|
| clawhdf5 --path mem.h5 save
|
||||||
|
clawhdf5 --path mem.h5 search --embedding '[0.1,0.2,0.3]' --query hello -k 3
|
||||||
|
```
|
||||||
|
|
||||||
|
## License
|
||||||
|
|
||||||
|
MIT
|
||||||
@@ -3,7 +3,7 @@ name = "clawhdf5-derive"
|
|||||||
version = "2.7.0"
|
version = "2.7.0"
|
||||||
edition = "2024"
|
edition = "2024"
|
||||||
rust-version.workspace = true
|
rust-version.workspace = true
|
||||||
description = "Derive macros for rustyhdf5 HDF5 traits"
|
description = "Derive macro (H5Type) for clawhdf5 compound types"
|
||||||
license = "MIT"
|
license = "MIT"
|
||||||
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
repository = "https://git.redclaw.dev/quantumclaw/clawhdf5"
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
|
|||||||
@@ -1,28 +1,50 @@
|
|||||||
# clawhdf5-derive
|
# clawhdf5-derive
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-derive)
|
`#[derive(H5Type)]`: maps a Rust struct with named fields to an HDF5
|
||||||
[](https://docs.rs/clawhdf5-derive)
|
compound datatype. The derive generates three inherent methods:
|
||||||
|
|
||||||
Derive macros for clawhdf5 HDF5 traits.
|
- `hdf5_datatype() -> clawhdf5_format::datatype::Datatype` — the
|
||||||
|
`Datatype::Compound` (members in field order, packed, little-endian);
|
||||||
|
- `to_bytes(&self) -> Vec<u8>` — one element in that layout;
|
||||||
|
- `from_bytes(&[u8]) -> Self` — the reverse (panics if the slice is shorter
|
||||||
|
than the compound).
|
||||||
|
|
||||||
## Features
|
Supported field types: `f32`, `f64`, `i8`–`i64`, `u8`–`u64`, `bool`
|
||||||
|
(stored as `u8`) and fixed-size arrays `[T; N]` of those numeric types.
|
||||||
|
Tuple structs, enums and nested structs are refused at compile time.
|
||||||
|
|
||||||
- `#[derive(HDF5Type)]` for automatic HDF5 datatype mapping
|
The generated code names `clawhdf5_format`, so the crate using the derive
|
||||||
- Struct-to-compound-type derivation
|
must depend on [`clawhdf5-format`](../clawhdf5-format/README.md) too. Not
|
||||||
|
on crates.io yet:
|
||||||
|
|
||||||
## Usage
|
```toml
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-derive = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
clawhdf5-format = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
```
|
||||||
|
|
||||||
|
## Example
|
||||||
|
|
||||||
```rust
|
```rust
|
||||||
use clawhdf5_derive::HDF5Type;
|
use clawhdf5_derive::H5Type;
|
||||||
|
use clawhdf5_format::datatype::Datatype;
|
||||||
|
|
||||||
#[derive(HDF5Type)]
|
#[derive(H5Type, Debug, PartialEq)]
|
||||||
struct Point {
|
struct Point {
|
||||||
x: f64,
|
id: u32,
|
||||||
y: f64,
|
pos: [f64; 3],
|
||||||
z: f64,
|
valid: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
let p = Point { id: 7, pos: [1.0, 2.0, 3.0], valid: true };
|
||||||
|
let bytes = p.to_bytes();
|
||||||
|
assert_eq!(bytes.len(), 4 + 24 + 1);
|
||||||
|
assert_eq!(Point::from_bytes(&bytes), p);
|
||||||
|
assert!(matches!(Point::hdf5_datatype(), Datatype::Compound { size: 29, .. }));
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Tests: `crates/clawhdf5-format/tests/derive_tests.rs`.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT
|
MIT
|
||||||
|
|||||||
@@ -1,27 +1,56 @@
|
|||||||
# clawhdf5-filters
|
# clawhdf5-filters
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-filters)
|
Standalone deflate (zlib) compression and decompression with a choice of
|
||||||
[](https://docs.rs/clawhdf5-filters)
|
backend: pure-Rust zlib-rs (default), zlib-ng, Apple's Compression
|
||||||
|
framework, or miniz_oxide.
|
||||||
|
|
||||||
Filter and compression pipeline for clawhdf5.
|
This crate holds **deflate backends only**. The HDF5 filter pipeline, the
|
||||||
|
filter registry and every other codec (shuffle, Fletcher-32, N-Bit,
|
||||||
|
scale-offset, LZ4, Zstd, SZIP, pcodec, LZF, bitshuffle, bzip2, Blosc,
|
||||||
|
Blosc2, ZFP) live in [`clawhdf5-format`](../clawhdf5-format/README.md),
|
||||||
|
which calls flate2 itself and selects its deflate backend with its own
|
||||||
|
features. No library crate of the workspace depends on this one (the
|
||||||
|
`clawhdf5` facade uses it only in tests).
|
||||||
|
|
||||||
## Features
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
- DEFLATE compression/decompression
|
```toml
|
||||||
- Pure-Rust deflate via zlib-rs (default, `zlib-rs` feature)
|
[dependencies]
|
||||||
- zlib-ng instead, if you want it (`fast-deflate` feature; C, needs cmake)
|
clawhdf5-filters = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
- Apple Compression framework support (`apple-compression` feature)
|
```
|
||||||
|
|
||||||
## Usage
|
## API
|
||||||
|
|
||||||
```rust
|
```rust
|
||||||
use clawhdf5_filters::{deflate_compress, deflate_decompress};
|
use clawhdf5_filters::{deflate_backend, deflate_compress, deflate_decompress};
|
||||||
|
|
||||||
|
let data: Vec<u8> = (0..10_000u32).map(|i| (i % 251) as u8).collect();
|
||||||
let compressed = deflate_compress(&data, 6).unwrap();
|
let compressed = deflate_compress(&data, 6).unwrap();
|
||||||
// The second argument bounds the output: the expected decompressed size.
|
// The second argument bounds the output: the expected decompressed size.
|
||||||
let decompressed = deflate_decompress(&compressed, data.len()).unwrap();
|
let decompressed = deflate_decompress(&compressed, data.len()).unwrap();
|
||||||
|
assert_eq!(decompressed, data);
|
||||||
|
println!("backend: {}", deflate_backend()); // "zlib-rs" by default
|
||||||
```
|
```
|
||||||
|
|
||||||
|
Also `deflate_compress_miniz`/`deflate_decompress_miniz` (always
|
||||||
|
miniz_oxide) and `fast_deflate::{compress, decompress, active_backend}`.
|
||||||
|
|
||||||
|
## Features
|
||||||
|
|
||||||
|
Backend priority: `apple-compression` (macOS only) > zlib-ng > zlib-rs >
|
||||||
|
miniz_oxide (with none enabled).
|
||||||
|
|
||||||
|
| Feature | Default | Backend | Builds C |
|
||||||
|
|---|---|---|---|
|
||||||
|
| `zlib-rs` | yes | zlib-rs through flate2, with `runtime_detection` (needed for its SIMD) | no |
|
||||||
|
| `fast-deflate` | no | zlib-ng through flate2 | yes (cmake) |
|
||||||
|
| `system-zlib` | no | the system zlib through flate2 | yes (`libz-sys`) |
|
||||||
|
| `apple-compression` | no | Apple Compression framework, macOS only (ignored elsewhere) | no (links a system framework) |
|
||||||
|
|
||||||
|
zlib-rs matches zlib-ng on HDF5 reads and writes and produces
|
||||||
|
byte-identical output: see "Deflate backend" in
|
||||||
|
[`BENCHMARKS.md`](../../BENCHMARKS.md).
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
MIT
|
MIT
|
||||||
|
|||||||
@@ -22,6 +22,17 @@ zstd = { version = "0.13", optional = true }
|
|||||||
blake3 = { version = "1", optional = true }
|
blake3 = { version = "1", optional = true }
|
||||||
libaec-sys = { path = "../libaec-sys", version = "0.1", optional = true }
|
libaec-sys = { path = "../libaec-sys", version = "0.1", optional = true }
|
||||||
pco = { version = "1.0", optional = true }
|
pco = { version = "1.0", optional = true }
|
||||||
|
# Pure-Rust Zstandard, for the plugin filters that embed zstd (bitshuffle,
|
||||||
|
# blosc). The `zstd` feature (filter 32015) links libzstd instead.
|
||||||
|
ruzstd = { version = "0.9", optional = true }
|
||||||
|
# bzip2 with its default backend, libbz2-rs-sys: a pure-Rust port of
|
||||||
|
# libbzip2 (no C is compiled, despite the -sys name).
|
||||||
|
bzip2 = { version = "0.6", optional = true }
|
||||||
|
snap = { version = "1", optional = true }
|
||||||
|
|
||||||
|
[target.'cfg(target_os = "linux")'.dependencies]
|
||||||
|
# madvise(MADV_HUGEPAGE) for large read buffers (see src/bulk_alloc.rs).
|
||||||
|
libc = { version = "0.2", default-features = false }
|
||||||
|
|
||||||
[dev-dependencies]
|
[dev-dependencies]
|
||||||
half = { workspace = true }
|
half = { workspace = true }
|
||||||
@@ -37,7 +48,7 @@ harness = false
|
|||||||
# Deflate backend: `zlib-rs` (pure Rust) by default. `fast-deflate` selects
|
# Deflate backend: `zlib-rs` (pure Rust) by default. `fast-deflate` selects
|
||||||
# zlib-ng instead (C, built with cmake); flate2 prefers a C zlib whenever one
|
# zlib-ng instead (C, built with cmake); flate2 prefers a C zlib whenever one
|
||||||
# is enabled, so turning it on anywhere in the build overrides the default.
|
# is enabled, so turning it on anywhere in the build overrides the default.
|
||||||
default = ["std", "checksum", "deflate", "provenance", "zlib-rs", "system-zlib-decompress"]
|
default = ["std", "checksum", "deflate", "provenance", "zlib-rs", "system-zlib-decompress", "lzf"]
|
||||||
std = []
|
std = []
|
||||||
checksum = []
|
checksum = []
|
||||||
deflate = ["flate2"]
|
deflate = ["flate2"]
|
||||||
@@ -56,6 +67,25 @@ zstd = ["dep:zstd"]
|
|||||||
blake3_hash = ["blake3"]
|
blake3_hash = ["blake3"]
|
||||||
szip = ["libaec-sys"]
|
szip = ["libaec-sys"]
|
||||||
pcodec = ["dep:pco"]
|
pcodec = ["dep:pco"]
|
||||||
|
# Plugin filters, pure Rust. LZF (32000) is h5py's built-in compression; it
|
||||||
|
# has no dependencies, so it is on by default.
|
||||||
|
lzf = []
|
||||||
|
# Bitshuffle (32008), with its LZ4 and Zstandard modes.
|
||||||
|
bitshuffle = ["lz4_flex", "ruzstd"]
|
||||||
|
# bzip2 (307).
|
||||||
|
bzip2 = ["dep:bzip2", "std"]
|
||||||
|
# Blosc 1 (32001) with its BloscLZ, LZ4, Snappy, Zlib and Zstandard codecs.
|
||||||
|
blosc = ["lz4_flex", "ruzstd", "snap", "deflate", "std"]
|
||||||
|
# Blosc2 (32026), read-only: frames, B2ND arrays, and the Blosc codecs above.
|
||||||
|
blosc2 = ["blosc"]
|
||||||
|
# ZFP (32013, H5Z-ZFP), read-only: every mode, for int32, int64, float and
|
||||||
|
# double fields of 1 to 4 dimensions.
|
||||||
|
zfp = []
|
||||||
|
# Every plugin filter above.
|
||||||
|
plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc", "blosc2", "zfp"]
|
||||||
|
# Test instrumentation: per-thread counts of heap objects read (see
|
||||||
|
# `lookup_stats`), so tests can bound the cost of a name lookup.
|
||||||
|
lookup-stats = ["std"]
|
||||||
|
|
||||||
[[bench]]
|
[[bench]]
|
||||||
name = "parallel_decompress_bench"
|
name = "parallel_decompress_bench"
|
||||||
|
|||||||
@@ -1,27 +1,106 @@
|
|||||||
# clawhdf5-format
|
# clawhdf5-format
|
||||||
|
|
||||||
[](https://crates.io/crates/clawhdf5-format)
|
The HDF5 file format in pure Rust: parsers and writers for every on-disk
|
||||||
[](https://docs.rs/clawhdf5-format)
|
structure, the filter pipeline and its codecs, and the shared type
|
||||||
|
definitions the other crates use. Most users want the
|
||||||
|
[`clawhdf5`](../clawhdf5/README.md) facade, which wraps this crate in an
|
||||||
|
h5py-like API; use this one directly for low-level access or in `no_std`
|
||||||
|
code.
|
||||||
|
|
||||||
Pure-Rust HDF5 binary format parsing and writing — no C dependencies.
|
Not on crates.io yet; depend on it from git:
|
||||||
|
|
||||||
|
```toml
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-format = { git = "https://git.redclaw.dev/quantumclaw/clawhdf5" }
|
||||||
|
```
|
||||||
|
|
||||||
|
## What is in it
|
||||||
|
|
||||||
|
- **Parsing:** superblock v0–v3 (`superblock`, with the superblock
|
||||||
|
extension and metadata cache images, `superblock_ext`), object headers v1
|
||||||
|
and v2 (`object_header`), every header message the readers use
|
||||||
|
(`datatype`, `dataspace`, `data_layout` v1–v4 including virtual datasets,
|
||||||
|
`fill_value`, `attribute`, `link_message`, `shared_message`, ...), groups
|
||||||
|
old and new (`group_v1` symbol tables with local heaps, `group_v2` with
|
||||||
|
fractal heaps and v2 B-trees), and every chunk index (v1 B-tree, single
|
||||||
|
chunk, implicit, fixed array, extensible array, v2 B-tree).
|
||||||
|
- **Reading data:** `data_read` (contiguous, compact, chunked),
|
||||||
|
`partial_read` and `selection` (hyperslabs and points), `vl_data`
|
||||||
|
(variable-length strings and sequences through the global heap),
|
||||||
|
`chunk_cache`.
|
||||||
|
- **Storage:** the `storage::Storage` trait (`read_at`, `read_ranges`,
|
||||||
|
`len`, `hint`) that every read path goes through, so a file can be read
|
||||||
|
from memory, a file handle or a remote backend
|
||||||
|
([`clawhdf5-remote`](../clawhdf5-remote/README.md)).
|
||||||
|
- **Writing:** `file_writer::FileWriter` and the builders in
|
||||||
|
`type_builders` (datasets, groups, attributes, compound and enum types,
|
||||||
|
links, virtual datasets, creation-order tracking); chunk indexes and
|
||||||
|
dense-storage B-trees of any size (`chunked_write`, `btree_v2_write`,
|
||||||
|
`ea_writer`). Output is read by h5py and h5dump.
|
||||||
|
- **Filters:** `filter_pipeline` and `filter_registry` (look up by ID; other
|
||||||
|
IDs can be registered at run time with `register_filter`). Built in:
|
||||||
|
deflate, shuffle, Fletcher-32, N-Bit, scale-offset; behind features LZ4,
|
||||||
|
Zstd, SZIP (decode), pcodec, and the plugin filters LZF, bitshuffle,
|
||||||
|
bzip2, Blosc 1 (read and write), Blosc2 and ZFP (read only).
|
||||||
|
- **Shared pieces:** `float16` (the one IEEE half-precision conversion the
|
||||||
|
workspace uses), `provenance` (SHA-256 dataset hashes), `checksum`
|
||||||
|
(Jenkins lookup3 for v2+ structures).
|
||||||
|
|
||||||
|
## Example
|
||||||
|
|
||||||
|
```rust
|
||||||
|
use clawhdf5_format::file_writer::{AttrValue, FileWriter};
|
||||||
|
use clawhdf5_format::{group_v2, object_header, signature, superblock};
|
||||||
|
|
||||||
|
// Write a file to memory
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.create_dataset("data")
|
||||||
|
.with_f64_data(&[1.0, 2.0, 3.0])
|
||||||
|
.with_shape(&[3])
|
||||||
|
.set_attr("unit", AttrValue::String("m/s".into()));
|
||||||
|
let bytes = fw.finish().unwrap();
|
||||||
|
|
||||||
|
// Parse it back: superblock -> path -> object header
|
||||||
|
let (_user_block, file) = signature::split_user_block(&bytes).unwrap();
|
||||||
|
let sb = superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let addr = group_v2::resolve_path_any(file, &sb, "data").unwrap();
|
||||||
|
let hdr = object_header::ObjectHeader::parse(file, addr as usize, sb.offset_size, sb.length_size)
|
||||||
|
.unwrap();
|
||||||
|
assert!(!hdr.messages.is_empty());
|
||||||
|
```
|
||||||
|
|
||||||
## Features
|
## Features
|
||||||
|
|
||||||
- Zero-copy superblock, object header, and B-tree parsing
|
| Feature | Default | What | Builds C |
|
||||||
- Chunked dataset read/write with filter pipelines
|
|---|---|---|---|
|
||||||
- `no_std` support (disable `std` feature)
|
| `std` | yes | standard library; without it the crate is `no_std` + `alloc` (CI builds it for `thumbv7em-none-eabihf`) | no |
|
||||||
- Optional parallel reads via Rayon
|
| `checksum` | yes | verify Jenkins lookup3 checksums | no |
|
||||||
- SHA-256 provenance tracking
|
| `deflate` | yes | deflate through flate2 | no |
|
||||||
|
| `zlib-rs` | yes | flate2's pure-Rust zlib-rs backend, with `runtime_detection` (without it zlib-rs loses SIMD and inflates 3.5x slower) | no |
|
||||||
|
| `system-zlib-decompress` | yes | macOS only: inflate with the system libz first, falling back to flate2; no effect elsewhere | no (links the system libz on macOS) |
|
||||||
|
| `provenance` | yes | SHA-256 provenance hashes | no |
|
||||||
|
| `lzf` | yes | LZF (32000) | no |
|
||||||
|
| `parallel` | no | rayon-parallel chunk decoding | no |
|
||||||
|
| `fast-checksum` | no | hardware CRC32 through `crc32fast` | no |
|
||||||
|
| `lz4` | no | LZ4 (32004) | no |
|
||||||
|
| `pcodec` | no | pcodec | no |
|
||||||
|
| `bitshuffle`, `bzip2`, `blosc` | no | 32008, 307, 32001, read and write | no |
|
||||||
|
| `blosc2`, `zfp` | no | 32026, 32013, read only | no |
|
||||||
|
| `plugin-filters` | no | all six plugin filters above | no |
|
||||||
|
| `lookup-stats` | no | counters for name-lookup benchmarks | no |
|
||||||
|
| `zstd` | no | Zstandard (32015) | yes (libzstd) |
|
||||||
|
| `szip` | no | SZIP (4) decoding | links the system libaec (`libaec-dev`) |
|
||||||
|
| `fast-deflate` | no | zlib-ng | yes (cmake) |
|
||||||
|
| `system-zlib` | no | the system zlib | yes (`libz-sys`) |
|
||||||
|
| `blake3_hash` | no | `provenance::blake3_hash` | yes (`cc`) |
|
||||||
|
|
||||||
## Usage
|
## Robustness
|
||||||
|
|
||||||
```rust
|
Every parser is meant to return an error, never panic, on hostile input:
|
||||||
use clawhdf5_format::Superblock;
|
nine cargo-fuzz targets live in [`fuzz/`](fuzz/README.md), the conformance
|
||||||
|
sweep includes the HDF Group's CVE corpus
|
||||||
let data = std::fs::read("data.h5").unwrap();
|
([`CONFORMANCE.md`](../../CONFORMANCE.md)), and header checks follow
|
||||||
let sb = Superblock::from_bytes(&data).unwrap();
|
libhdf5's. Open gaps are in [`docs/known-issues.md`](../../docs/known-issues.md).
|
||||||
println!("HDF5 version {}.{}", sb.version_major(), sb.version_minor());
|
|
||||||
```
|
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
|
|||||||
@@ -51,10 +51,18 @@ done
|
|||||||
|
|
||||||
## CI
|
## CI
|
||||||
|
|
||||||
These targets are **not** run in CI (`.gitea/workflows/ci.yml`) — cargo-fuzz
|
These targets are **not** run by the CI workflows (`.gitea/workflows/ci.yml`)
|
||||||
requires nightly and each meaningful run takes minutes, which doesn't fit a
|
— cargo-fuzz requires nightly and each meaningful run takes minutes, which
|
||||||
per-PR gate. Run them manually on a schedule (e.g. before a release, or after
|
doesn't fit a per-PR gate. Run them by hand before a release or after
|
||||||
touching parser code) instead.
|
touching parser code. `scripts/ci-test.sh` has an opt-in smoke run: with
|
||||||
|
`CLAWHDF5_FUZZ_SECONDS=N` it runs every target of this crate and of
|
||||||
|
`crates/clawhdf5-agent/fuzz` (the WAL parser) for N seconds each.
|
||||||
|
|
||||||
|
Other robustness checks that do run: the nightly conformance sweep reads
|
||||||
|
the HDF Group's CVE reproducers and fails on any panic, hang, crash or
|
||||||
|
out-of-memory ([`conformance/README.md`](../../../conformance/README.md)),
|
||||||
|
and `scripts/h5rs-fuzz.sh` runs every `h5rs` subcommand over them, optionally
|
||||||
|
on byte-flipped copies.
|
||||||
|
|
||||||
## Reproducing Crashes
|
## Reproducing Crashes
|
||||||
|
|
||||||
|
|||||||
Binary file not shown.
@@ -0,0 +1,121 @@
|
|||||||
|
//! File address and length → in-memory index conversion.
|
||||||
|
//!
|
||||||
|
//! HDF5 addresses and lengths are 64-bit; the file is parsed through a
|
||||||
|
//! `&[u8]` indexed by `usize`. On a 64-bit target every `u64` fits, but on a
|
||||||
|
//! 32-bit one (`wasm32`, `i686`, `thumbv7em`) an address past `usize::MAX`
|
||||||
|
//! used to be truncated by an `as usize` cast — silently pointing at another
|
||||||
|
//! part of the file — or to panic. [`to_usize`] is the one conversion the
|
||||||
|
//! parsers use instead: such an address is a clean
|
||||||
|
//! [`FormatError::Overflow`]. It cannot be inside the data anyway: no slice
|
||||||
|
//! is longer than `isize::MAX` bytes.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::format;
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// A file address, offset or length from the file as a `usize` index.
|
||||||
|
///
|
||||||
|
/// Fails with [`FormatError::Overflow`] when the value does not fit this
|
||||||
|
/// platform's `usize` (only possible on targets narrower than 64 bits).
|
||||||
|
#[inline]
|
||||||
|
pub fn to_usize(value: u64) -> Result<usize, FormatError> {
|
||||||
|
to_index::<usize>(value)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A file address for a [`crate::storage::Storage`] read, checked as
|
||||||
|
/// [`to_usize`] checks it: the parsers read through 64-bit offsets, but an
|
||||||
|
/// address that could not index an in-memory file on this platform is the
|
||||||
|
/// same [`FormatError::Overflow`] the slice parsers gave for it.
|
||||||
|
#[inline]
|
||||||
|
pub fn checked_addr(value: u64) -> Result<u64, FormatError> {
|
||||||
|
to_usize(value).map(|_| value)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`to_usize`] for an index type of any width. `usize` is 64 bits wide on
|
||||||
|
/// the hosts CI tests on, where the error path cannot be reached through
|
||||||
|
/// `usize`; tests run the same code with `u32` in its place, as on a 32-bit
|
||||||
|
/// target.
|
||||||
|
#[inline]
|
||||||
|
fn to_index<T: TryFrom<u64>>(value: u64) -> Result<T, FormatError> {
|
||||||
|
T::try_from(value).map_err(|_| too_large(value))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A count or offset into an in-memory buffer (a codec's progress counter,
|
||||||
|
/// a size the writer computed from data it holds) as a `usize`, saturating
|
||||||
|
/// at `usize::MAX` instead of truncating.
|
||||||
|
///
|
||||||
|
/// For values that are bounded by the length of something in memory, so
|
||||||
|
/// always fit; if one ever did not, a saturated index fails its bounds check
|
||||||
|
/// or allocation instead of silently addressing the wrong bytes. A value
|
||||||
|
/// read from the file uses [`to_usize`].
|
||||||
|
#[inline]
|
||||||
|
pub fn saturating_usize(value: u64) -> usize {
|
||||||
|
saturating_index(value, usize::MAX)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`saturating_usize`] for an index type of any width, whose largest
|
||||||
|
/// value is `max` (see [`to_index`]).
|
||||||
|
#[inline]
|
||||||
|
fn saturating_index<T: TryFrom<u64>>(value: u64, max: T) -> T {
|
||||||
|
T::try_from(value).unwrap_or(max)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cold]
|
||||||
|
#[inline(never)]
|
||||||
|
fn too_large(value: u64) -> FormatError {
|
||||||
|
FormatError::Overflow(format!(
|
||||||
|
"file address or length {value:#x} exceeds this platform's address space"
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_that_fit_convert_exactly() {
|
||||||
|
assert_eq!(to_usize(0), Ok(0));
|
||||||
|
assert_eq!(to_usize(0x1234), Ok(0x1234));
|
||||||
|
assert_eq!(to_usize(usize::MAX as u64), Ok(usize::MAX));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn saturating_conversion_never_wraps() {
|
||||||
|
assert_eq!(saturating_usize(0), 0);
|
||||||
|
assert_eq!(saturating_usize(0x1234), 0x1234);
|
||||||
|
assert_eq!(saturating_usize(usize::MAX as u64), usize::MAX);
|
||||||
|
// Past usize::MAX (32-bit targets) or at u64::MAX: saturates.
|
||||||
|
assert_eq!(saturating_usize(u64::MAX), usize::MAX);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn values_past_usize_max_are_an_error_not_truncated() {
|
||||||
|
// Reachable through `usize` only where it is narrower than u64 (no
|
||||||
|
// such target runs tests in CI), so the same conversion is run with
|
||||||
|
// u32 standing in for a 32-bit usize.
|
||||||
|
let max = u64::from(u32::MAX);
|
||||||
|
assert_eq!(to_index::<u32>(max), Ok(u32::MAX));
|
||||||
|
for past in [max + 1, max + 0x10, 0x1_0000_1234, u64::MAX] {
|
||||||
|
let err = to_index::<u32>(past).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(err, FormatError::Overflow(_)),
|
||||||
|
"{past:#x}: {err:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Where an `as` cast would have wrapped to a small, valid-looking
|
||||||
|
// index, it is not returned.
|
||||||
|
assert_eq!(0x1_0000_1234_u64 as u32, 0x1234);
|
||||||
|
assert!(to_index::<u32>(0x1_0000_1234).is_err());
|
||||||
|
|
||||||
|
assert_eq!(saturating_index(max + 1, u32::MAX), u32::MAX);
|
||||||
|
assert_eq!(saturating_index(0x1_0000_1234, u32::MAX), u32::MAX);
|
||||||
|
assert_eq!(saturating_index(0x1234, u32::MAX), 0x1234);
|
||||||
|
|
||||||
|
// And through `usize` itself, whichever width it has here.
|
||||||
|
match (usize::MAX as u64).checked_add(1) {
|
||||||
|
Some(past) => assert!(matches!(to_usize(past), Err(FormatError::Overflow(_)))),
|
||||||
|
None => assert_eq!(to_usize(u64::MAX), Ok(usize::MAX)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -5,8 +5,10 @@ use alloc::{borrow::Cow, string::String, vec::Vec};
|
|||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::borrow::Cow;
|
use std::borrow::Cow;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::attribute_info::AttributeInfoMessage;
|
use crate::attribute_info::AttributeInfoMessage;
|
||||||
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records_in, find_btree_v2_records_in};
|
||||||
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::data_read;
|
use crate::data_read;
|
||||||
use crate::dataspace::Dataspace;
|
use crate::dataspace::Dataspace;
|
||||||
use crate::datatype::Datatype;
|
use crate::datatype::Datatype;
|
||||||
@@ -15,6 +17,7 @@ use crate::fractal_heap::FractalHeapHeader;
|
|||||||
use crate::message_type::MessageType;
|
use crate::message_type::MessageType;
|
||||||
use crate::object_header::ObjectHeader;
|
use crate::object_header::ObjectHeader;
|
||||||
use crate::shared_message;
|
use crate::shared_message;
|
||||||
|
use crate::storage::Storage;
|
||||||
use crate::vl_data;
|
use crate::vl_data;
|
||||||
|
|
||||||
/// A parsed HDF5 attribute message.
|
/// A parsed HDF5 attribute message.
|
||||||
@@ -50,7 +53,7 @@ impl AttributeMessage {
|
|||||||
///
|
///
|
||||||
/// `length_size` is needed for dataspace dimension parsing.
|
/// `length_size` is needed for dataspace dimension parsing.
|
||||||
pub fn parse(data: &[u8], length_size: u8) -> Result<AttributeMessage, FormatError> {
|
pub fn parse(data: &[u8], length_size: u8) -> Result<AttributeMessage, FormatError> {
|
||||||
Self::parse_impl(data, length_size, None)
|
Self::parse_impl(data, length_size, None::<(&[u8], u8)>)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// [`AttributeMessage::parse`] with access to the rest of the file, which
|
/// [`AttributeMessage::parse`] with access to the rest of the file, which
|
||||||
@@ -65,13 +68,24 @@ impl AttributeMessage {
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
Self::parse_impl(data, length_size, Some((file_data, offset_size)))
|
Self::parse_in_storage(data, file_data, offset_size, length_size)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_impl(
|
/// [`AttributeMessage::parse_in_file`] with the file behind any
|
||||||
|
/// [`Storage`].
|
||||||
|
pub fn parse_in_storage<S: Storage + ?Sized>(
|
||||||
|
data: &[u8],
|
||||||
|
file: &S,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
|
Self::parse_impl(data, length_size, Some((file, offset_size)))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn parse_impl<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
ensure_len(data, 0, 2)?;
|
ensure_len(data, 0, 2)?;
|
||||||
let version = data[0];
|
let version = data[0];
|
||||||
@@ -86,19 +100,19 @@ impl AttributeMessage {
|
|||||||
|
|
||||||
/// The bytes of an embedded datatype/dataspace message, following the
|
/// The bytes of an embedded datatype/dataspace message, following the
|
||||||
/// shared-message reference when `shared` is set.
|
/// shared-message reference when `shared` is set.
|
||||||
fn embedded_message<'a>(
|
fn embedded_message<'a, S: Storage + ?Sized>(
|
||||||
bytes: &'a [u8],
|
bytes: &'a [u8],
|
||||||
shared: bool,
|
shared: bool,
|
||||||
msg_type: MessageType,
|
msg_type: MessageType,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<Cow<'a, [u8]>, FormatError> {
|
) -> Result<Cow<'a, [u8]>, FormatError> {
|
||||||
if !shared {
|
if !shared {
|
||||||
return Ok(Cow::Borrowed(bytes));
|
return Ok(Cow::Borrowed(bytes));
|
||||||
}
|
}
|
||||||
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
||||||
let shared_ref = shared_message::parse_shared_ref(bytes, offset_size)?;
|
let shared_ref = shared_message::parse_shared_ref_sized(bytes, offset_size, length_size)?;
|
||||||
shared_message::resolve_shared_message(
|
shared_message::resolve_shared_message_in(
|
||||||
file_data,
|
file_data,
|
||||||
&shared_ref,
|
&shared_ref,
|
||||||
msg_type,
|
msg_type,
|
||||||
@@ -143,10 +157,10 @@ impl AttributeMessage {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_v2(
|
fn parse_v2<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
||||||
let flags = data.get(1).copied().unwrap_or(0);
|
let flags = data.get(1).copied().unwrap_or(0);
|
||||||
@@ -197,10 +211,10 @@ impl AttributeMessage {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn parse_v3(
|
fn parse_v3<S: Storage + ?Sized>(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
file: Option<(&[u8], u8)>,
|
file: Option<(&S, u8)>,
|
||||||
) -> Result<AttributeMessage, FormatError> {
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
// Flags: bit 0 = datatype is shared, bit 1 = dataspace is shared.
|
||||||
let flags = data.get(1).copied().unwrap_or(0);
|
let flags = data.get(1).copied().unwrap_or(0);
|
||||||
@@ -322,9 +336,19 @@ impl AttributeMessage {
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<String>, FormatError> {
|
||||||
|
self.read_vl_strings_in(file_data, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::read_vl_strings`] over any [`Storage`].
|
||||||
|
pub fn read_vl_strings_in<S: Storage + ?Sized>(
|
||||||
|
&self,
|
||||||
|
file_data: &S,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Vec<String>, FormatError> {
|
) -> Result<Vec<String>, FormatError> {
|
||||||
let num_elements = self.dataspace.num_elements();
|
let num_elements = self.dataspace.num_elements();
|
||||||
vl_data::read_vl_strings(
|
vl_data::read_vl_strings_in(
|
||||||
file_data,
|
file_data,
|
||||||
&self.raw_data,
|
&self.raw_data,
|
||||||
num_elements,
|
num_elements,
|
||||||
@@ -341,7 +365,8 @@ fn compute_raw_data(
|
|||||||
dataspace: &Dataspace,
|
dataspace: &Dataspace,
|
||||||
datatype: &Datatype,
|
datatype: &Datatype,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let num_elements = dataspace.num_elements() as usize;
|
// Saturating, like the product: the size is capped at what is there.
|
||||||
|
let num_elements = usize::try_from(dataspace.num_elements()).unwrap_or(usize::MAX);
|
||||||
let elem_size = datatype.type_size() as usize;
|
let elem_size = datatype.type_size() as usize;
|
||||||
let expected_size = num_elements.saturating_mul(elem_size);
|
let expected_size = num_elements.saturating_mul(elem_size);
|
||||||
let available = data.len().saturating_sub(pos);
|
let available = data.len().saturating_sub(pos);
|
||||||
@@ -362,6 +387,18 @@ fn extract_name(bytes: &[u8]) -> String {
|
|||||||
String::from_utf8_lossy(&bytes[..end]).into_owned()
|
String::from_utf8_lossy(&bytes[..end]).into_owned()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// An attribute's datatype gets libhdf5's extra check for a header without
|
||||||
|
/// a checksum (see [`Datatype::check_unused_bits`]).
|
||||||
|
fn check_in_header(
|
||||||
|
attr: AttributeMessage,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
) -> Result<AttributeMessage, FormatError> {
|
||||||
|
if header.version == 1 {
|
||||||
|
attr.datatype.check_unused_bits()?;
|
||||||
|
}
|
||||||
|
Ok(attr)
|
||||||
|
}
|
||||||
|
|
||||||
/// Extract all attribute messages from an object header.
|
/// Extract all attribute messages from an object header.
|
||||||
pub fn extract_attributes(
|
pub fn extract_attributes(
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
@@ -371,7 +408,7 @@ pub fn extract_attributes(
|
|||||||
for msg in &header.messages {
|
for msg in &header.messages {
|
||||||
if msg.msg_type == MessageType::Attribute {
|
if msg.msg_type == MessageType::Attribute {
|
||||||
let attr = AttributeMessage::parse(&msg.data, length_size)?;
|
let attr = AttributeMessage::parse(&msg.data, length_size)?;
|
||||||
attrs.push(attr);
|
attrs.push(check_in_header(attr, header)?);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Ok(attrs)
|
Ok(attrs)
|
||||||
@@ -394,59 +431,339 @@ pub fn find_attribute<'a>(
|
|||||||
///
|
///
|
||||||
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
||||||
/// (e.g., objects with many attributes, typically >8).
|
/// (e.g., objects with many attributes, typically >8).
|
||||||
|
///
|
||||||
|
/// Fails if any attribute cannot be read; see [`extract_attributes_tolerant`]
|
||||||
|
/// to read the others.
|
||||||
pub fn extract_attributes_full(
|
pub fn extract_attributes_full(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
let mut attrs = Vec::new();
|
extract_attributes_full_in(file_data, header, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
// Collect compact attributes (inline in OH)
|
/// [`extract_attributes_full`] over any [`Storage`]. Dense attribute
|
||||||
for msg in &header.messages {
|
/// storage is indexed by a v2 B-tree, which is not read over [`Storage`]
|
||||||
if msg.msg_type == MessageType::Attribute {
|
/// yet: on a backend without the whole file in memory an object with dense
|
||||||
if shared_message::is_shared(msg.flags) {
|
/// attributes is [`FormatError::ContiguousStorageRequired`].
|
||||||
// Shared attribute: resolve the reference to get actual attribute data
|
pub fn extract_attributes_full_in<S: Storage + ?Sized>(
|
||||||
let shared_ref = shared_message::parse_shared_ref(&msg.data, offset_size)?;
|
file: &S,
|
||||||
let resolved_data = shared_message::resolve_shared_message(
|
header: &ObjectHeader,
|
||||||
file_data,
|
offset_size: u8,
|
||||||
&shared_ref,
|
length_size: u8,
|
||||||
MessageType::Attribute,
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
offset_size,
|
extract_attributes_with(file, header, offset_size, length_size, &mut Err)
|
||||||
length_size,
|
}
|
||||||
)?;
|
|
||||||
let attr = AttributeMessage::parse_in_file(
|
/// Like [`extract_attributes_full`], but an attribute that cannot be read
|
||||||
&resolved_data,
|
/// (a corrupt or unsupported attribute message, or a heap object that cannot
|
||||||
file_data,
|
/// be located) is left out and its error returned alongside the attributes
|
||||||
offset_size,
|
/// that could be read, instead of failing them all.
|
||||||
length_size,
|
///
|
||||||
)?;
|
/// Errors in the structures that index the attributes (the Attribute Info
|
||||||
attrs.push(attr);
|
/// message, the dense-storage heap header or B-tree) still fail the call:
|
||||||
} else {
|
/// then it is unknown which attributes exist at all.
|
||||||
let attr = AttributeMessage::parse_in_file(
|
pub fn extract_attributes_tolerant(
|
||||||
&msg.data,
|
file_data: &[u8],
|
||||||
file_data,
|
header: &ObjectHeader,
|
||||||
offset_size,
|
offset_size: u8,
|
||||||
length_size,
|
length_size: u8,
|
||||||
)?;
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
attrs.push(attr);
|
extract_attributes_tolerant_core(file_data, header, offset_size, length_size)
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
/// [`extract_attributes_tolerant`] over any [`Storage`] (see
|
||||||
|
/// [`extract_attributes_full_in`] for dense storage). One with the whole
|
||||||
|
/// file in memory is read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn extract_attributes_tolerant_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => extract_attributes_tolerant(all, header, offset_size, length_size),
|
||||||
|
None => extract_attributes_tolerant_core(file_data, header, offset_size, length_size),
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn extract_attributes_tolerant_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
let mut errors = Vec::new();
|
||||||
|
let attrs = extract_attributes_with(file_data, header, offset_size, length_size, &mut |e| {
|
||||||
|
errors.push(e);
|
||||||
|
Ok(())
|
||||||
|
})?;
|
||||||
|
Ok((attrs, errors))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read every attribute; each one that fails goes to `on_error`, which
|
||||||
|
/// either stops the read (returns the error) or skips that attribute.
|
||||||
|
fn extract_attributes_with<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
|
let mut attrs = Vec::new();
|
||||||
|
// Each attribute's creation order, where the file records one.
|
||||||
|
let mut orders: Vec<u32> = Vec::new();
|
||||||
|
|
||||||
|
extract_compact_attributes(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut attrs,
|
||||||
|
&mut orders,
|
||||||
|
on_error,
|
||||||
|
)?;
|
||||||
|
|
||||||
// Check for dense attributes via AttributeInfo message
|
// Check for dense attributes via AttributeInfo message
|
||||||
let attr_info = find_attribute_info(header, offset_size)?;
|
let attr_info = find_attribute_info(header, offset_size)?;
|
||||||
if let Some(info) = attr_info
|
if let Some(info) = &attr_info
|
||||||
&& let Some(fh_addr) = info.fractal_heap_address
|
&& let Some(fh_addr) = info.fractal_heap_address
|
||||||
{
|
{
|
||||||
let dense_attrs =
|
extract_dense_attributes(
|
||||||
extract_dense_attributes(file_data, &info, fh_addr, offset_size, length_size)?;
|
file_data,
|
||||||
attrs.extend(dense_attrs);
|
info,
|
||||||
|
fh_addr,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut attrs,
|
||||||
|
&mut orders,
|
||||||
|
on_error,
|
||||||
|
)?;
|
||||||
|
}
|
||||||
|
|
||||||
|
// An object that tracks attribute creation order lists its attributes
|
||||||
|
// in that order (h5py's `track_order=True`), as libhdf5 does; otherwise
|
||||||
|
// they come in storage order.
|
||||||
|
if attr_info.is_some_and(|i| i.max_creation_index.is_some()) {
|
||||||
|
let mut paired: Vec<(u32, AttributeMessage)> = orders.into_iter().zip(attrs).collect();
|
||||||
|
paired.sort_by_key(|(o, _)| *o);
|
||||||
|
attrs = paired.into_iter().map(|(_, a)| a).collect();
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(attrs)
|
Ok(attrs)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// B-tree v2 record type of dense attribute storage's name index.
|
||||||
|
const ATTRIBUTE_NAME_INDEX: u8 = 8;
|
||||||
|
|
||||||
|
/// The attribute called `name` on the object with header `header`: the
|
||||||
|
/// first one [`extract_attributes_tolerant`] returns under that name, or
|
||||||
|
/// `None` if it returns none (an attribute that cannot be read is not
|
||||||
|
/// returned there either).
|
||||||
|
///
|
||||||
|
/// Compact attributes are in the header and are scanned. Dense attributes
|
||||||
|
/// are found through the name index (a v2 B-tree of lookup3 name hashes,
|
||||||
|
/// record type 8): only the attributes whose names hash like `name` are read
|
||||||
|
/// from the heap, O(log n) instead of all of them. Errors in the structures
|
||||||
|
/// that index the attributes fail the call, as they fail a listing.
|
||||||
|
pub fn find_attribute_in_file(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
find_attribute_core(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
name,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut Vec::new(),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_attribute_in_file`] over any [`Storage`] (see
|
||||||
|
/// [`extract_attributes_full_in`] for dense storage, whose name index still
|
||||||
|
/// needs the whole file in memory). One with the whole file in memory is
|
||||||
|
/// read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn find_attribute_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => find_attribute_in_file(all, header, name, offset_size, length_size),
|
||||||
|
None => find_attribute_core(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
name,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut Vec::new(),
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_attribute_in`], also returning the errors of the attributes it
|
||||||
|
/// could not read on the way (which it leaves out rather than failing
|
||||||
|
/// the call): the attribute asked for may be one of them. A reader of a
|
||||||
|
/// file that is being written uses them to tell a read that raced the
|
||||||
|
/// writer from an absent attribute.
|
||||||
|
pub fn find_attribute_reporting_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Option<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
let mut errors = Vec::new();
|
||||||
|
let found = find_attribute_core(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
name,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut errors,
|
||||||
|
)?;
|
||||||
|
Ok((found, errors))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn find_attribute_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
errors: &mut Vec<FormatError>,
|
||||||
|
) -> Result<Option<AttributeMessage>, FormatError> {
|
||||||
|
let attr_info = find_attribute_info(header, offset_size)?;
|
||||||
|
let dense = attr_info
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|i| Some((i.fractal_heap_address?, i.btree_name_index_address?)));
|
||||||
|
let Some((fh_addr, btree_addr)) = dense else {
|
||||||
|
// Compact only (or dense storage without a name index, which a
|
||||||
|
// listing reports): as a listing finds it.
|
||||||
|
let (attrs, errs) =
|
||||||
|
extract_attributes_tolerant_in(file_data, header, offset_size, length_size)?;
|
||||||
|
errors.extend(errs);
|
||||||
|
return Ok(attrs.into_iter().find(|a| a.name == name));
|
||||||
|
};
|
||||||
|
let btree_hdr = BTreeV2Header::parse_in(
|
||||||
|
file_data,
|
||||||
|
to_usize(btree_addr)? as u64,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
let fh = FractalHeapHeader::parse_in(file_data, fh_addr, offset_size, length_size)?;
|
||||||
|
if btree_hdr.tree_type != ATTRIBUTE_NAME_INDEX || btree_hdr.record_size < 4 {
|
||||||
|
let (attrs, errs) =
|
||||||
|
extract_attributes_tolerant_in(file_data, header, offset_size, length_size)?;
|
||||||
|
errors.extend(errs);
|
||||||
|
return Ok(attrs.into_iter().find(|a| a.name == name));
|
||||||
|
}
|
||||||
|
|
||||||
|
// A listing has the compact attributes first.
|
||||||
|
let mut compact = Vec::new();
|
||||||
|
extract_compact_attributes(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut compact,
|
||||||
|
&mut Vec::new(),
|
||||||
|
&mut |_| Ok(()),
|
||||||
|
)?;
|
||||||
|
if let Some(a) = compact.into_iter().find(|a| a.name == name) {
|
||||||
|
return Ok(Some(a));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Record: heap ID + message flags(1) + creation order(4) + hash(4); the
|
||||||
|
// hash is the last field.
|
||||||
|
let hash = jenkins_lookup3(name.as_bytes());
|
||||||
|
let hash_at = usize::from(btree_hdr.record_size) - 4;
|
||||||
|
let records = find_btree_v2_records_in(file_data, &btree_hdr, offset_size, &mut |r| match r
|
||||||
|
.get(hash_at..hash_at + 4)
|
||||||
|
{
|
||||||
|
Some(h) => u32::from_le_bytes([h[0], h[1], h[2], h[3]]).cmp(&hash),
|
||||||
|
None => core::cmp::Ordering::Less,
|
||||||
|
})?;
|
||||||
|
let id_len = usize::from(fh.heap_id_length);
|
||||||
|
for record in &records {
|
||||||
|
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let attr = fh
|
||||||
|
.read_managed_object_in(file_data, id_bytes, offset_size)
|
||||||
|
.and_then(|d| {
|
||||||
|
AttributeMessage::parse_in_storage(&d, file_data, offset_size, length_size)
|
||||||
|
});
|
||||||
|
// One that cannot be read is left out, as from a listing.
|
||||||
|
match attr {
|
||||||
|
Ok(attr) if attr.name == name => return Ok(Some(attr)),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => errors.push(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The attributes stored in the object header itself (compact storage), and
|
||||||
|
/// each one's creation order into `orders`.
|
||||||
|
fn extract_compact_attributes<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
attrs: &mut Vec<AttributeMessage>,
|
||||||
|
orders: &mut Vec<u32>,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
for msg in &header.messages {
|
||||||
|
if msg.msg_type == MessageType::Attribute {
|
||||||
|
let attr = if shared_message::is_shared(msg.flags) {
|
||||||
|
// Shared attribute: resolve the reference to get actual attribute data
|
||||||
|
shared_message::parse_shared_ref_sized(&msg.data, offset_size, length_size)
|
||||||
|
.and_then(|shared_ref| {
|
||||||
|
shared_message::resolve_shared_message_in(
|
||||||
|
file_data,
|
||||||
|
&shared_ref,
|
||||||
|
MessageType::Attribute,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.and_then(|resolved| {
|
||||||
|
AttributeMessage::parse_in_storage(
|
||||||
|
&resolved,
|
||||||
|
file_data,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
} else {
|
||||||
|
AttributeMessage::parse_in_storage(&msg.data, file_data, offset_size, length_size)
|
||||||
|
};
|
||||||
|
let attr = attr.and_then(|a| check_in_header(a, header));
|
||||||
|
match attr {
|
||||||
|
Ok(attr) => {
|
||||||
|
attrs.push(attr);
|
||||||
|
orders.push(msg.creation_order.map_or(0, u32::from));
|
||||||
|
}
|
||||||
|
Err(e) => on_error(e)?,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Find and parse the Attribute Info message from an object header.
|
/// Find and parse the Attribute Info message from an object header.
|
||||||
fn find_attribute_info(
|
fn find_attribute_info(
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
@@ -461,16 +778,21 @@ fn find_attribute_info(
|
|||||||
Ok(None)
|
Ok(None)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract attributes from dense storage (fractal heap + B-tree v2).
|
/// Extract attributes from dense storage (fractal heap + B-tree v2), and
|
||||||
fn extract_dense_attributes(
|
/// each one's creation order into `orders`.
|
||||||
file_data: &[u8],
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn extract_dense_attributes<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
attr_info: &AttributeInfoMessage,
|
attr_info: &AttributeInfoMessage,
|
||||||
fh_addr: u64,
|
fh_addr: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
attrs: &mut Vec<AttributeMessage>,
|
||||||
|
orders: &mut Vec<u32>,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
// Parse fractal heap
|
// Parse fractal heap
|
||||||
let fh = FractalHeapHeader::parse(file_data, fh_addr as usize, offset_size, length_size)?;
|
let fh = FractalHeapHeader::parse_in(file_data, fh_addr, offset_size, length_size)?;
|
||||||
|
|
||||||
// Parse B-tree v2 for name index (type 8)
|
// Parse B-tree v2 for name index (type 8)
|
||||||
let btree_addr = attr_info
|
let btree_addr = attr_info
|
||||||
@@ -479,31 +801,47 @@ fn extract_dense_attributes(
|
|||||||
expected: 1,
|
expected: 1,
|
||||||
available: 0,
|
available: 0,
|
||||||
})?;
|
})?;
|
||||||
let btree_hdr = BTreeV2Header::parse(file_data, btree_addr as usize, offset_size, length_size)?;
|
let btree_hdr = BTreeV2Header::parse_in(
|
||||||
let records = collect_btree_v2_records(file_data, &btree_hdr, offset_size, length_size)?;
|
file_data,
|
||||||
|
to_usize(btree_addr)? as u64,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
let records = collect_btree_v2_records_in(file_data, &btree_hdr, offset_size, length_size)?;
|
||||||
|
|
||||||
let mut attrs = Vec::new();
|
|
||||||
for record in &records {
|
for record in &records {
|
||||||
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
||||||
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
||||||
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
||||||
let id_offset = 0;
|
let id_len = fh.heap_id_length as usize;
|
||||||
|
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||||
if record.data.len() < id_offset + fh.heap_id_length as usize {
|
on_error(FormatError::UnexpectedEof {
|
||||||
|
expected: id_len,
|
||||||
|
available: record.data.len(),
|
||||||
|
})?;
|
||||||
continue;
|
continue;
|
||||||
}
|
};
|
||||||
let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize];
|
|
||||||
|
|
||||||
// Read attribute message from fractal heap
|
|
||||||
let attr_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
|
||||||
|
|
||||||
// The data in the heap is a complete attribute message
|
// The data in the heap is a complete attribute message
|
||||||
let attr =
|
let attr = fh
|
||||||
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)?;
|
.read_managed_object_in(file_data, id_bytes, offset_size)
|
||||||
attrs.push(attr);
|
.and_then(|attr_data| {
|
||||||
|
AttributeMessage::parse_in_storage(&attr_data, file_data, offset_size, length_size)
|
||||||
|
});
|
||||||
|
match attr {
|
||||||
|
Ok(attr) => {
|
||||||
|
attrs.push(attr);
|
||||||
|
let order = record
|
||||||
|
.data
|
||||||
|
.get(id_len + 1..id_len + 5)
|
||||||
|
.map_or(0, |b| u32::from_le_bytes([b[0], b[1], b[2], b[3]]));
|
||||||
|
orders.push(order);
|
||||||
|
}
|
||||||
|
Err(e) => on_error(e)?,
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(attrs)
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -523,7 +861,8 @@ mod tests {
|
|||||||
|
|
||||||
/// Build an f64 LE datatype message.
|
/// Build an f64 LE datatype message.
|
||||||
fn build_f64_dt() -> Vec<u8> {
|
fn build_f64_dt() -> Vec<u8> {
|
||||||
let mut buf = build_dt_header(1, 1, [0x00, 0x00, 0x02], 8);
|
// Sign bit 63 (bits 8-15 of the class bits).
|
||||||
|
let mut buf = build_dt_header(1, 1, [0x20, 63, 0x00], 8);
|
||||||
let mut props = [0u8; 12];
|
let mut props = [0u8; 12];
|
||||||
props[2..4].copy_from_slice(&64u16.to_le_bytes()); // bit_precision
|
props[2..4].copy_from_slice(&64u16.to_le_bytes()); // bit_precision
|
||||||
props[4] = 52; // exp_location
|
props[4] = 52; // exp_location
|
||||||
@@ -897,4 +1236,73 @@ mod tests {
|
|||||||
let strs = attr.read_as_strings().unwrap();
|
let strs = attr.read_as_strings().unwrap();
|
||||||
assert_eq!(strs, vec!["abcd", "EFGH"]);
|
assert_eq!(strs, vec!["abcd", "EFGH"]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Every object's attributes in h5py-written files read identically
|
||||||
|
/// through a read_at-only CountingStorage — compact ones, shared ones,
|
||||||
|
/// those behind an Attribute Info message and dense storage (its v2
|
||||||
|
/// B-tree name index included) — and through a slice as Storage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let files: [(&str, &[u8]); 5] = [
|
||||||
|
("attrs", include_bytes!("../tests/fixtures/attrs.h5")),
|
||||||
|
(
|
||||||
|
"mixed_attrs",
|
||||||
|
include_bytes!("../tests/fixtures/mixed_attrs.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"dense_attrs",
|
||||||
|
include_bytes!("../tests/fixtures/dense_attrs.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"dense_attrs_root",
|
||||||
|
include_bytes!("../tests/fixtures/dense_attrs_root.h5"),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"shared_fill_value",
|
||||||
|
include_bytes!("../tests/fixtures/shared_fill_value.h5"),
|
||||||
|
),
|
||||||
|
];
|
||||||
|
let (mut same, mut dense, mut attrs) = (0, 0, 0);
|
||||||
|
for (name, file) in files {
|
||||||
|
let sb = crate::superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let (os, ls) = (sb.offset_size, sb.length_size);
|
||||||
|
let mut addrs = vec![sb.root_group_address];
|
||||||
|
addrs.extend(
|
||||||
|
crate::group_v2::resolve_group_children(file, &sb, sb.root_group_address)
|
||||||
|
.unwrap()
|
||||||
|
.iter()
|
||||||
|
.map(|e| e.object_header_address),
|
||||||
|
);
|
||||||
|
let storage = CountingStorage::new(file.to_vec());
|
||||||
|
for addr in addrs {
|
||||||
|
let header = ObjectHeader::parse(file, addr as usize, os, ls).unwrap();
|
||||||
|
let want = extract_attributes_full(file, &header, os, ls);
|
||||||
|
let slice_storage = extract_attributes_full_in(&file, &header, os, ls);
|
||||||
|
assert_eq!(format!("{slice_storage:?}"), format!("{want:?}"));
|
||||||
|
let got = extract_attributes_full_in(&storage, &header, os, ls);
|
||||||
|
let got_t = extract_attributes_tolerant_in(&storage, &header, os, ls);
|
||||||
|
let is_dense = find_attribute_info(&header, os)
|
||||||
|
.unwrap()
|
||||||
|
.is_some_and(|i| i.fractal_heap_address.is_some());
|
||||||
|
if is_dense {
|
||||||
|
dense += 1;
|
||||||
|
}
|
||||||
|
attrs += want.as_ref().map_or(0, Vec::len);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"), "{name}");
|
||||||
|
let want_t = extract_attributes_tolerant(file, &header, os, ls);
|
||||||
|
assert_eq!(format!("{got_t:?}"), format!("{want_t:?}"), "{name}");
|
||||||
|
same += 1;
|
||||||
|
for a in want.iter().flatten() {
|
||||||
|
let one = find_attribute_in(&storage, &header, &a.name, os, ls);
|
||||||
|
let want_one = find_attribute_in_file(file, &header, &a.name, os, ls);
|
||||||
|
assert_eq!(format!("{one:?}"), format!("{want_one:?}"), "{name}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
same >= 5 && dense >= 2 && attrs >= 5,
|
||||||
|
"{same} {dense} {attrs}"
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{Storage, read_exact_at};
|
||||||
|
|
||||||
/// A parsed B-tree v1 node.
|
/// A parsed B-tree v1 node.
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -74,13 +75,30 @@ impl BTreeV1Node {
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
offset: usize,
|
offset: usize,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<BTreeV1Node, FormatError> {
|
||||||
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one read of the node's header,
|
||||||
|
/// one of its keys and children.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<BTreeV1Node, FormatError> {
|
) -> Result<BTreeV1Node, FormatError> {
|
||||||
// signature(4) + node_type(1) + node_level(1) + entries_used(2) = 8
|
// signature(4) + node_type(1) + node_level(1) + entries_used(2) = 8
|
||||||
// + left_sibling(offset_size) + right_sibling(offset_size)
|
// + left_sibling(offset_size) + right_sibling(offset_size)
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let header_size = 8 + os * 2;
|
let header_size = 8 + os * 2;
|
||||||
ensure_len(file_data, offset, header_size)?;
|
// The body is read once the header says how long it is.
|
||||||
|
file.hint(offset, NODE_HINT_LEN);
|
||||||
|
let header = read_exact_at(file, offset, header_size)?;
|
||||||
|
let file_data: &[u8] = &header;
|
||||||
|
// The header's read checked that `offset + header_size` fits.
|
||||||
|
let body_start = offset + header_size as u64;
|
||||||
|
let offset = 0usize;
|
||||||
|
|
||||||
if &file_data[offset..offset + 4] != b"TREE" {
|
if &file_data[offset..offset + 4] != b"TREE" {
|
||||||
return Err(FormatError::InvalidBTreeSignature);
|
return Err(FormatError::InvalidBTreeSignature);
|
||||||
@@ -102,31 +120,30 @@ impl BTreeV1Node {
|
|||||||
} else {
|
} else {
|
||||||
Some(read_offset(file_data, pos, offset_size)?)
|
Some(read_offset(file_data, pos, offset_size)?)
|
||||||
};
|
};
|
||||||
pos += os;
|
|
||||||
|
|
||||||
// For type 0: keys are offset_size bytes, children are offset_size bytes
|
// For type 0: keys are offset_size bytes, children are offset_size bytes
|
||||||
// Layout: key[0], child[0], key[1], child[1], ..., key[N-1], child[N-1], key[N]
|
// Layout: key[0], child[0], key[1], child[1], ..., key[N-1], child[N-1], key[N]
|
||||||
let eu = entries_used as usize;
|
let eu = entries_used as usize;
|
||||||
let key_size = os; // For type 0, key = offset_size
|
let key_size = os; // For type 0, key = offset_size
|
||||||
let needed = eu * (key_size + os) + key_size; // eu children + (eu+1) keys
|
let needed = eu * (key_size + os) + key_size; // eu children + (eu+1) keys
|
||||||
ensure_len(file_data, pos, needed)?;
|
let body = read_exact_at(file, body_start, needed)?;
|
||||||
|
let file_data: &[u8] = &body;
|
||||||
|
|
||||||
let mut keys = Vec::with_capacity(eu + 1);
|
let mut keys = Vec::with_capacity(eu + 1);
|
||||||
let mut children = Vec::with_capacity(eu);
|
let mut children = Vec::with_capacity(eu);
|
||||||
|
|
||||||
for _i in 0..eu {
|
if os == 0 {
|
||||||
// key[i]
|
// What reading the first key reports (and keeps `chunks_exact`
|
||||||
let key = read_offset(file_data, pos, offset_size)?;
|
// below from being given a zero size).
|
||||||
keys.push(key);
|
return Err(FormatError::InvalidOffsetSize(offset_size));
|
||||||
pos += key_size;
|
|
||||||
// child[i]
|
|
||||||
let child = read_offset(file_data, pos, offset_size)?;
|
|
||||||
children.push(child);
|
|
||||||
pos += os;
|
|
||||||
}
|
}
|
||||||
// final key
|
// `needed` bytes: key[0], child[0], ..., child[eu - 1], key[eu].
|
||||||
let key = read_offset(file_data, pos, offset_size)?;
|
let (pairs, last) = file_data.split_at(eu * (key_size + os));
|
||||||
keys.push(key);
|
for pair in pairs.chunks_exact(key_size + os) {
|
||||||
|
keys.push(read_offset(pair, 0, offset_size)?);
|
||||||
|
children.push(read_offset(pair, key_size, offset_size)?);
|
||||||
|
}
|
||||||
|
keys.push(read_offset(last, 0, offset_size)?);
|
||||||
|
|
||||||
Ok(BTreeV1Node {
|
Ok(BTreeV1Node {
|
||||||
node_type,
|
node_type,
|
||||||
@@ -141,7 +158,18 @@ impl BTreeV1Node {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Maximum recursion depth for B-tree traversal (malformed data protection).
|
/// Maximum recursion depth for B-tree traversal (malformed data protection).
|
||||||
const MAX_BTREE_DEPTH: usize = 64;
|
pub(crate) const MAX_BTREE_DEPTH: usize = 64;
|
||||||
|
|
||||||
|
/// What a symbol table node takes with libhdf5's default group leaf K (4):
|
||||||
|
/// its 8-byte header and 2K entries of 40 bytes (8-byte offsets). Hinted
|
||||||
|
/// before one is read ([`Storage::hint`]); a node of another size is read
|
||||||
|
/// all the same.
|
||||||
|
const SNOD_HINT_LEN: usize = 8 + 8 * 40;
|
||||||
|
|
||||||
|
/// What a group B-tree node takes with libhdf5's default internal K (16):
|
||||||
|
/// its header (24 bytes with 8-byte offsets), 2K + 1 keys and 2K children
|
||||||
|
/// of 8 bytes. Hinted before one is read.
|
||||||
|
const NODE_HINT_LEN: usize = 24 + (2 * 16 + 1 + 2 * 16) * 8;
|
||||||
|
|
||||||
/// Collect all leaf-level child addresses (SNOD addresses) by traversing the B-tree.
|
/// Collect all leaf-level child addresses (SNOD addresses) by traversing the B-tree.
|
||||||
pub fn collect_symbol_table_nodes(
|
pub fn collect_symbol_table_nodes(
|
||||||
@@ -150,11 +178,21 @@ pub fn collect_symbol_table_nodes(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<u64>, FormatError> {
|
) -> Result<Vec<u64>, FormatError> {
|
||||||
collect_symbol_table_nodes_inner(file_data, btree_address, offset_size, length_size, 0)
|
collect_symbol_table_nodes_in(file_data, btree_address, offset_size, length_size)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn collect_symbol_table_nodes_inner(
|
/// [`collect_symbol_table_nodes`] over any [`Storage`]: two reads per node.
|
||||||
file_data: &[u8],
|
pub fn collect_symbol_table_nodes_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
btree_address: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<u64>, FormatError> {
|
||||||
|
collect_symbol_table_nodes_inner(file, btree_address, offset_size, length_size, 0)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn collect_symbol_table_nodes_inner<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
btree_address: u64,
|
btree_address: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
@@ -164,29 +202,47 @@ fn collect_symbol_table_nodes_inner(
|
|||||||
return Err(FormatError::NestingDepthExceeded);
|
return Err(FormatError::NestingDepthExceeded);
|
||||||
}
|
}
|
||||||
|
|
||||||
let node = BTreeV1Node::parse(file_data, btree_address as usize, offset_size, length_size)?;
|
let node = BTreeV1Node::parse_in(file, btree_address, offset_size, length_size)?;
|
||||||
|
|
||||||
if node.node_type != 0 {
|
if node.node_type != 0 {
|
||||||
return Err(FormatError::InvalidBTreeNodeType(node.node_type));
|
return Err(FormatError::InvalidBTreeNodeType(node.node_type));
|
||||||
}
|
}
|
||||||
|
|
||||||
if node.node_level == 0 {
|
if node.node_level == 0 {
|
||||||
// Leaf: children are SNOD addresses
|
// Leaf: children are SNOD addresses, read next (see
|
||||||
|
// `Storage::hint`).
|
||||||
|
for &snod in &node.children {
|
||||||
|
file.hint(snod, SNOD_HINT_LEN);
|
||||||
|
}
|
||||||
Ok(node.children)
|
Ok(node.children)
|
||||||
} else {
|
} else {
|
||||||
// Internal: recurse into children
|
// Internal: recurse into children. A child that fails does not
|
||||||
|
// stop the walk: the others are still descended into (reading, not
|
||||||
|
// using, what they hold), then the first error is returned. The
|
||||||
|
// result and the error are those of stopping at the first failure;
|
||||||
|
// a storage that records what it lacks (see `storage::touch`)
|
||||||
|
// learns every node the walk can reach in one attempt.
|
||||||
let mut result = Vec::new();
|
let mut result = Vec::new();
|
||||||
|
let mut failed = None;
|
||||||
for &child_addr in &node.children {
|
for &child_addr in &node.children {
|
||||||
let child_snods = collect_symbol_table_nodes_inner(
|
match collect_symbol_table_nodes_inner(
|
||||||
file_data,
|
file,
|
||||||
child_addr,
|
child_addr,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
depth + 1,
|
depth + 1,
|
||||||
)?;
|
) {
|
||||||
result.extend(child_snods);
|
Ok(child_snods) if failed.is_none() => result.extend(child_snods),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => {
|
||||||
|
failed.get_or_insert(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
match failed {
|
||||||
|
Some(e) => Err(e),
|
||||||
|
None => Ok(result),
|
||||||
}
|
}
|
||||||
Ok(result)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -317,4 +373,48 @@ mod tests {
|
|||||||
assert_eq!(node.entries_used, 1);
|
assert_eq!(node.entries_used, 1);
|
||||||
assert_eq!(node.children, vec![0x50]);
|
assert_eq!(node.children, vec![0x50]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Nodes and trees, cut at every length, parse identically through a
|
||||||
|
/// `read_at`-only storage.
|
||||||
|
#[test]
|
||||||
|
fn storage_parse_matches_slice_parse() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let nodes = [
|
||||||
|
build_btree_node(0, 0, &[0, 5, 10], &[0x100, 0x200], None, None, 8),
|
||||||
|
build_btree_node(0, 0, &[0, 5], &[0x100], Some(0x40), Some(0x80), 4),
|
||||||
|
build_btree_node(1, 2, &[0, 5], &[0x100], None, Some(0x80), 8),
|
||||||
|
];
|
||||||
|
for (n, node) in nodes.iter().enumerate() {
|
||||||
|
let os = if n == 1 { 4 } else { 8 };
|
||||||
|
for cut in 0..=node.len() {
|
||||||
|
let f = &node[..cut];
|
||||||
|
let storage = CountingStorage::new(f.to_vec());
|
||||||
|
let want = BTreeV1Node::parse(f, 0, os, 8);
|
||||||
|
let got = BTreeV1Node::parse_in(&storage, 0, os, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let leaf1 = build_btree_node(0, 0, &[0, 5], &[0xA00], None, None, 8);
|
||||||
|
let leaf2 = build_btree_node(0, 0, &[5, 10], &[0xB00], None, None, 8);
|
||||||
|
let internal = build_btree_node(0, 1, &[0, 5, 10], &[0, 256], None, None, 8);
|
||||||
|
let mut file = vec![0u8; 512 + internal.len()];
|
||||||
|
file[..leaf1.len()].copy_from_slice(&leaf1);
|
||||||
|
file[256..256 + leaf2.len()].copy_from_slice(&leaf2);
|
||||||
|
file[512..].copy_from_slice(&internal);
|
||||||
|
for cut in [file.len(), 300, 260, 100, 10] {
|
||||||
|
let mut f = file.clone();
|
||||||
|
if cut < 512 {
|
||||||
|
// Truncate the leaves, keep the root.
|
||||||
|
f[cut..512].fill(0);
|
||||||
|
}
|
||||||
|
let storage = CountingStorage::new(f.clone());
|
||||||
|
assert_eq!(
|
||||||
|
collect_symbol_table_nodes_in(&storage, 512, 8, 8),
|
||||||
|
collect_symbol_table_nodes(&f, 512, 8, 8)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let storage = CountingStorage::new(file);
|
||||||
|
collect_symbol_table_nodes_in(&storage, 512, 8, 8).unwrap();
|
||||||
|
assert_eq!(storage.reads(), 6);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,11 +2,14 @@
|
|||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
use core::cmp::Ordering;
|
||||||
|
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
use byteorder::{ByteOrder, LittleEndian};
|
use byteorder::{ByteOrder, LittleEndian};
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{Storage, Window, len_usize};
|
||||||
|
|
||||||
/// Parsed B-tree v2 header (signature "BTHD").
|
/// Parsed B-tree v2 header (signature "BTHD").
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -71,7 +74,7 @@ fn ensure_len(data: &[u8], pos: usize, needed: usize) -> Result<(), FormatError>
|
|||||||
|
|
||||||
/// Compute the number of bytes needed to represent a count, using variable-width encoding.
|
/// Compute the number of bytes needed to represent a count, using variable-width encoding.
|
||||||
/// B-tree v2 uses this for the number of records fields in internal nodes.
|
/// B-tree v2 uses this for the number of records fields in internal nodes.
|
||||||
fn bytes_for_max_records(max_nrec: u64) -> usize {
|
pub(crate) fn bytes_for_max_records(max_nrec: u64) -> usize {
|
||||||
if max_nrec == 0 {
|
if max_nrec == 0 {
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
@@ -97,38 +100,52 @@ impl BTreeV2Header {
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<BTreeV2Header, FormatError> {
|
) -> Result<BTreeV2Header, FormatError> {
|
||||||
ensure_len(file_data, offset, 4)?;
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
if &file_data[offset..offset + 4] != b"BTHD" {
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one bounded read of the
|
||||||
|
/// header.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<BTreeV2Header, FormatError> {
|
||||||
|
// Every field and the checksum; the window holds all of it or ends
|
||||||
|
// at the end of the file, so its bounds checks are the whole-file
|
||||||
|
// ones.
|
||||||
|
let full = 16 + usize::from(offset_size) + 2 + usize::from(length_size) + 4;
|
||||||
|
let w = Window::read(file, offset, full)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
w.ensure(0, 4)?;
|
||||||
|
if &d[..4] != b"BTHD" {
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
return Err(FormatError::InvalidBTreeV2Signature);
|
||||||
}
|
}
|
||||||
|
|
||||||
ensure_len(file_data, offset, 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1)?;
|
w.ensure(0, 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1)?;
|
||||||
let version = file_data[offset + 4];
|
let version = d[4];
|
||||||
if version != 0 {
|
if version != 0 {
|
||||||
return Err(FormatError::InvalidBTreeV2Version(version));
|
return Err(FormatError::InvalidBTreeV2Version(version));
|
||||||
}
|
}
|
||||||
|
|
||||||
let tree_type = file_data[offset + 5];
|
let tree_type = d[5];
|
||||||
let node_size = u32::from_le_bytes([
|
let node_size = u32::from_le_bytes([d[6], d[7], d[8], d[9]]);
|
||||||
file_data[offset + 6],
|
let record_size = u16::from_le_bytes([d[10], d[11]]);
|
||||||
file_data[offset + 7],
|
let depth = u16::from_le_bytes([d[12], d[13]]);
|
||||||
file_data[offset + 8],
|
let _split_percent = d[14];
|
||||||
file_data[offset + 9],
|
let _merge_percent = d[15];
|
||||||
]);
|
|
||||||
let record_size = u16::from_le_bytes([file_data[offset + 10], file_data[offset + 11]]);
|
|
||||||
let depth = u16::from_le_bytes([file_data[offset + 12], file_data[offset + 13]]);
|
|
||||||
let _split_percent = file_data[offset + 14];
|
|
||||||
let _merge_percent = file_data[offset + 15];
|
|
||||||
|
|
||||||
let mut pos = offset + 16;
|
let mut pos = 16;
|
||||||
let root_node_address = read_offset(file_data, pos, offset_size)?;
|
w.ensure(pos, usize::from(offset_size))?;
|
||||||
|
let root_node_address = read_offset(d, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
|
|
||||||
ensure_len(file_data, pos, 2)?;
|
w.ensure(pos, 2)?;
|
||||||
let num_records_in_root = u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
let num_records_in_root = u16::from_le_bytes([d[pos], d[pos + 1]]);
|
||||||
pos += 2;
|
pos += 2;
|
||||||
|
|
||||||
let total_records = read_offset(file_data, pos, length_size)?;
|
w.ensure(pos, usize::from(length_size))?;
|
||||||
|
let total_records = read_offset(d, pos, length_size)?;
|
||||||
#[allow(unused_assignments)]
|
#[allow(unused_assignments)]
|
||||||
{
|
{
|
||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
@@ -137,9 +154,9 @@ impl BTreeV2Header {
|
|||||||
// Validate header checksum
|
// Validate header checksum
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
{
|
{
|
||||||
ensure_len(file_data, pos, 4)?;
|
w.ensure(pos, 4)?;
|
||||||
let stored = LittleEndian::read_u32(&file_data[pos..pos + 4]);
|
let stored = LittleEndian::read_u32(&d[pos..pos + 4]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&file_data[offset..pos]);
|
let computed = crate::checksum::jenkins_lookup3(&d[..pos]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
return Err(FormatError::ChecksumMismatch {
|
return Err(FormatError::ChecksumMismatch {
|
||||||
expected: stored,
|
expected: stored,
|
||||||
@@ -163,7 +180,7 @@ impl BTreeV2Header {
|
|||||||
/// Compute maximum records per node for a given depth level.
|
/// Compute maximum records per node for a given depth level.
|
||||||
/// leaf: (node_size - overhead) / record_size
|
/// leaf: (node_size - overhead) / record_size
|
||||||
/// internal: depends on pointers
|
/// internal: depends on pointers
|
||||||
fn max_records_leaf(node_size: u32, record_size: u16) -> u64 {
|
pub(crate) fn max_records_leaf(node_size: u32, record_size: u16) -> u64 {
|
||||||
// Leaf overhead: signature(4) + version(1) + type(1) + checksum(4) = 10
|
// Leaf overhead: signature(4) + version(1) + type(1) + checksum(4) = 10
|
||||||
let overhead = 10u32;
|
let overhead = 10u32;
|
||||||
if node_size <= overhead || record_size == 0 {
|
if node_size <= overhead || record_size == 0 {
|
||||||
@@ -177,10 +194,17 @@ const MAX_DEPTH: u16 = 64;
|
|||||||
|
|
||||||
/// Take `n` records from the traversal's budget, or refuse the tree.
|
/// Take `n` records from the traversal's budget, or refuse the tree.
|
||||||
fn spend(budget: &mut usize, n: usize) -> Result<(), FormatError> {
|
fn spend(budget: &mut usize, n: usize) -> Result<(), FormatError> {
|
||||||
*budget = budget
|
match budget.checked_sub(n) {
|
||||||
.checked_sub(n)
|
Some(left) => {
|
||||||
.ok_or(FormatError::NestingDepthExceeded)?;
|
*budget = left;
|
||||||
Ok(())
|
Ok(())
|
||||||
|
}
|
||||||
|
None => {
|
||||||
|
// Spent: a walk that goes on after a failure stops here.
|
||||||
|
*budget = 0;
|
||||||
|
Err(FormatError::NestingDepthExceeded)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Collect all records from a B-tree v2 by traversing from the root.
|
/// Collect all records from a B-tree v2 by traversing from the root.
|
||||||
@@ -189,6 +213,17 @@ pub fn collect_btree_v2_records(
|
|||||||
header: &BTreeV2Header,
|
header: &BTreeV2Header,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
collect_btree_v2_records_in(file_data, header, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`collect_btree_v2_records`] over any [`Storage`]: one bounded read per
|
||||||
|
/// node.
|
||||||
|
pub fn collect_btree_v2_records_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
if header.total_records == 0 || header.num_records_in_root == 0 {
|
if header.total_records == 0 || header.num_records_in_root == 0 {
|
||||||
return Ok(Vec::new());
|
return Ok(Vec::new());
|
||||||
@@ -208,24 +243,25 @@ pub fn collect_btree_v2_records(
|
|||||||
// millions of records from a few kilobytes. Counting against what the
|
// millions of records from a few kilobytes. Counting against what the
|
||||||
// file could physically contain bounds that without trusting the
|
// file could physically contain bounds that without trusting the
|
||||||
// header's own `total_records`.
|
// header's own `total_records`.
|
||||||
let mut budget = file_data.len() / usize::from(header.record_size.max(1));
|
let mut budget = len_usize(file) / usize::from(header.record_size.max(1));
|
||||||
|
|
||||||
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
||||||
|
|
||||||
if header.depth == 0 {
|
if header.depth == 0 {
|
||||||
// Root is a leaf
|
// Root is a leaf
|
||||||
parse_leaf_records(
|
parse_leaf_records(
|
||||||
file_data,
|
file,
|
||||||
header.root_node_address as usize,
|
to_usize(header.root_node_address)?,
|
||||||
header.num_records_in_root,
|
header.num_records_in_root,
|
||||||
header.record_size,
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
// Root is internal; traverse recursively
|
// Root is internal; traverse recursively
|
||||||
let mut records = Vec::new();
|
let mut records = Vec::new();
|
||||||
collect_internal_records(
|
collect_internal_records(
|
||||||
file_data,
|
file,
|
||||||
header.root_node_address as usize,
|
to_usize(header.root_node_address)?,
|
||||||
header.num_records_in_root,
|
header.num_records_in_root,
|
||||||
header.depth,
|
header.depth,
|
||||||
header.record_size,
|
header.record_size,
|
||||||
@@ -240,36 +276,72 @@ pub fn collect_btree_v2_records(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A node's bytes: `want` bytes at `offset` (fewer only at the end of the
|
||||||
|
/// file), after checking its 4-byte signature. A node is read in one piece
|
||||||
|
/// when it fits in `node_size` (every valid node does); a larger claimed
|
||||||
|
/// extent — record counts from a damaged parent — is first checked against
|
||||||
|
/// the end of the file, so it costs a read only of bytes the file has.
|
||||||
|
/// Bounds errors are the whole-file ones: the signature check needs the
|
||||||
|
/// first 6 bytes, then `checks` — `(position, length)` pairs relative to
|
||||||
|
/// the node, in the order the parser checks them — must lie in the file.
|
||||||
|
fn read_node<'a, S: Storage + ?Sized>(
|
||||||
|
file: &'a S,
|
||||||
|
offset: usize,
|
||||||
|
want: usize,
|
||||||
|
node_size: u32,
|
||||||
|
signature: &[u8; 4],
|
||||||
|
checks: &[(usize, usize)],
|
||||||
|
) -> Result<Window<'a>, FormatError> {
|
||||||
|
let one_read = usize::try_from(node_size).unwrap_or(usize::MAX).max(6);
|
||||||
|
let w = Window::read(file, offset as u64, want.min(one_read))?;
|
||||||
|
w.ensure(0, 6)?;
|
||||||
|
if &w.bytes[..4] != signature {
|
||||||
|
return Err(FormatError::InvalidBTreeV2Signature);
|
||||||
|
}
|
||||||
|
if want <= one_read {
|
||||||
|
return Ok(w);
|
||||||
|
}
|
||||||
|
for &(rel, len) in checks {
|
||||||
|
Window::check_extent(file, offset as u64, rel, len)?;
|
||||||
|
}
|
||||||
|
Window::read(file, offset as u64, want)
|
||||||
|
}
|
||||||
|
|
||||||
/// Parse records from a leaf node (signature "BTLF").
|
/// Parse records from a leaf node (signature "BTLF").
|
||||||
fn parse_leaf_records(
|
fn parse_leaf_records<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
offset: usize,
|
offset: usize,
|
||||||
num_records: u16,
|
num_records: u16,
|
||||||
record_size: u16,
|
record_size: u16,
|
||||||
|
node_size: u32,
|
||||||
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
// signature(4) + version(1) + type(1) = 6 bytes header
|
// signature(4) + version(1) + type(1) = 6 bytes header
|
||||||
ensure_len(file_data, offset, 6)?;
|
let pos = 6;
|
||||||
if &file_data[offset..offset + 4] != b"BTLF" {
|
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
|
||||||
}
|
|
||||||
|
|
||||||
let pos = offset + 6;
|
|
||||||
let rs = record_size as usize;
|
let rs = record_size as usize;
|
||||||
let total = (num_records as usize)
|
let total = (num_records as usize)
|
||||||
.checked_mul(rs)
|
.checked_mul(rs)
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
.ok_or(FormatError::UnexpectedEof {
|
||||||
expected: usize::MAX,
|
expected: usize::MAX,
|
||||||
available: file_data.len(),
|
available: len_usize(file),
|
||||||
})?;
|
})?;
|
||||||
ensure_len(file_data, pos, total)?;
|
let w = read_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
pos + total + 4,
|
||||||
|
node_size,
|
||||||
|
b"BTLF",
|
||||||
|
&[(pos, total)],
|
||||||
|
)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
w.ensure(pos, total)?;
|
||||||
|
|
||||||
// Validate checksum: 4 bytes after records + padding
|
// Validate checksum: 4 bytes after records + padding
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
{
|
{
|
||||||
let checksum_pos = pos + total;
|
let checksum_pos = pos + total;
|
||||||
if file_data.len() >= checksum_pos + 4 {
|
if d.len() >= checksum_pos + 4 {
|
||||||
let stored = LittleEndian::read_u32(&file_data[checksum_pos..checksum_pos + 4]);
|
let stored = LittleEndian::read_u32(&d[checksum_pos..checksum_pos + 4]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&file_data[offset..checksum_pos]);
|
let computed = crate::checksum::jenkins_lookup3(&d[..checksum_pos]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
return Err(FormatError::ChecksumMismatch {
|
return Err(FormatError::ChecksumMismatch {
|
||||||
expected: stored,
|
expected: stored,
|
||||||
@@ -283,16 +355,135 @@ fn parse_leaf_records(
|
|||||||
for i in 0..num_records as usize {
|
for i in 0..num_records as usize {
|
||||||
let start = pos + i * rs;
|
let start = pos + i * rs;
|
||||||
records.push(BTreeV2Record {
|
records.push(BTreeV2Record {
|
||||||
data: file_data[start..start + rs].to_vec(),
|
data: d[start..start + rs].to_vec(),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
Ok(records)
|
Ok(records)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// An internal node read from the file: its bytes (from the signature on),
|
||||||
|
/// where its records start, and its children as `(address, record count)`.
|
||||||
|
struct InternalNode<'a> {
|
||||||
|
node: Window<'a>,
|
||||||
|
records_start: usize,
|
||||||
|
children: Vec<(u64, u16)>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl InternalNode<'_> {
|
||||||
|
/// Record `i`, `rs` bytes long.
|
||||||
|
fn record(&self, i: usize, rs: usize) -> Result<&[u8], FormatError> {
|
||||||
|
let overflow = || FormatError::UnexpectedEof {
|
||||||
|
expected: usize::MAX,
|
||||||
|
available: usize::MAX,
|
||||||
|
};
|
||||||
|
let rec_start = i
|
||||||
|
.checked_mul(rs)
|
||||||
|
.and_then(|o| self.records_start.checked_add(o))
|
||||||
|
.ok_or_else(overflow)?;
|
||||||
|
self.node.ensure(rec_start, rs)?;
|
||||||
|
Ok(&self.node.bytes[rec_start..rec_start + rs])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An internal node's layout: where its records start, and its children as
|
||||||
|
/// `(address, record count)`.
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn read_internal_node<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: usize,
|
||||||
|
num_records: u16,
|
||||||
|
depth: u16,
|
||||||
|
record_size: u16,
|
||||||
|
node_size: u32,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
) -> Result<InternalNode<'_>, FormatError> {
|
||||||
|
let nr = num_records as usize;
|
||||||
|
let rs = record_size as usize;
|
||||||
|
|
||||||
|
// Records first
|
||||||
|
let records_total = nr.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: usize::MAX,
|
||||||
|
available: len_usize(file),
|
||||||
|
})?;
|
||||||
|
|
||||||
|
// Child pointer layout, as libhdf5 computes it (H5B2__hdr_init): the
|
||||||
|
// child's record count is always encoded in the width needed for a
|
||||||
|
// *leaf's* maximum, and — below the first internal level — the child
|
||||||
|
// subtree's total record count in the width needed for the most records
|
||||||
|
// a subtree of that depth can hold.
|
||||||
|
let child_depth = depth - 1;
|
||||||
|
let nrec_width = bytes_for_max_records(max_leaf_nrec);
|
||||||
|
let total_nrec_width = if depth > 1 {
|
||||||
|
bytes_for_max_records(cum_max_records(
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
child_depth,
|
||||||
|
))
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
|
||||||
|
let num_children = nr + 1;
|
||||||
|
let child_ptr_size = offset_size as usize + nrec_width + total_nrec_width;
|
||||||
|
let pointers = num_children * child_ptr_size;
|
||||||
|
|
||||||
|
// signature(4) + version(1) + type(1) = 6, records, pointers, checksum.
|
||||||
|
let w = read_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
6 + records_total + pointers + 4,
|
||||||
|
node_size,
|
||||||
|
b"BTIN",
|
||||||
|
&[(6, records_total), (6 + records_total, pointers)],
|
||||||
|
)?;
|
||||||
|
let d = &w.bytes;
|
||||||
|
let mut pos = 6;
|
||||||
|
w.ensure(pos, records_total)?;
|
||||||
|
let records_start = pos;
|
||||||
|
pos += records_total;
|
||||||
|
|
||||||
|
w.ensure(pos, pointers)?;
|
||||||
|
|
||||||
|
let mut children = Vec::with_capacity(num_children);
|
||||||
|
for _ in 0..num_children {
|
||||||
|
let addr = read_offset(d, pos, offset_size)?;
|
||||||
|
pos += offset_size as usize;
|
||||||
|
let child_nrec = read_var_uint(d, pos, nrec_width)? as u16;
|
||||||
|
pos += nrec_width;
|
||||||
|
pos += total_nrec_width; // skip total records in subtree
|
||||||
|
children.push((addr, child_nrec));
|
||||||
|
}
|
||||||
|
|
||||||
|
// The checksum follows the child pointers and covers the node up to it.
|
||||||
|
// Lookups prune children by the keys in this node, so an unverified
|
||||||
|
// internal node could hide a record without any error: libhdf5 refuses
|
||||||
|
// a mismatch here, and so does this.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
{
|
||||||
|
w.ensure(pos, 4)?;
|
||||||
|
let stored = LittleEndian::read_u32(&d[pos..pos + 4]);
|
||||||
|
let computed = crate::checksum::jenkins_lookup3(&d[..pos]);
|
||||||
|
if computed != stored {
|
||||||
|
return Err(FormatError::ChecksumMismatch {
|
||||||
|
expected: stored,
|
||||||
|
computed,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(InternalNode {
|
||||||
|
node: w,
|
||||||
|
records_start,
|
||||||
|
children,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
/// Recursively collect records from an internal node.
|
/// Recursively collect records from an internal node.
|
||||||
#[allow(clippy::too_many_arguments, clippy::only_used_in_recursion)]
|
#[allow(clippy::too_many_arguments, clippy::only_used_in_recursion)]
|
||||||
fn collect_internal_records(
|
fn collect_internal_records<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
offset: usize,
|
offset: usize,
|
||||||
num_records: u16,
|
num_records: u16,
|
||||||
depth: u16,
|
depth: u16,
|
||||||
@@ -304,145 +495,295 @@ fn collect_internal_records(
|
|||||||
budget: &mut usize,
|
budget: &mut usize,
|
||||||
out: &mut Vec<BTreeV2Record>,
|
out: &mut Vec<BTreeV2Record>,
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
// signature(4) + version(1) + type(1) = 6
|
|
||||||
ensure_len(file_data, offset, 6)?;
|
|
||||||
if &file_data[offset..offset + 4] != b"BTIN" {
|
|
||||||
return Err(FormatError::InvalidBTreeV2Signature);
|
|
||||||
}
|
|
||||||
|
|
||||||
let nr = num_records as usize;
|
let nr = num_records as usize;
|
||||||
let rs = record_size as usize;
|
let rs = record_size as usize;
|
||||||
let mut pos = offset + 6;
|
let node = read_internal_node(
|
||||||
|
file,
|
||||||
// Read all records first
|
offset,
|
||||||
let records_total = nr.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
num_records,
|
||||||
expected: usize::MAX,
|
depth,
|
||||||
available: file_data.len(),
|
record_size,
|
||||||
})?;
|
node_size,
|
||||||
ensure_len(file_data, pos, records_total)?;
|
offset_size,
|
||||||
let records_start = pos;
|
max_leaf_nrec,
|
||||||
pos += records_total;
|
)?;
|
||||||
|
|
||||||
// Compute sizes for child pointers
|
|
||||||
// max_records at child depth - for variable-width nrec encoding
|
|
||||||
let child_depth = depth - 1;
|
let child_depth = depth - 1;
|
||||||
let max_nrec_child = if child_depth == 0 {
|
|
||||||
max_leaf_nrec
|
|
||||||
} else {
|
|
||||||
// For internal nodes at child_depth, the true max_nrec depends on the
|
|
||||||
// node size, record size, and the recursive width of child pointer
|
|
||||||
// entries (which themselves depend on max_nrec at deeper levels).
|
|
||||||
// Computing the exact value requires iterating from the leaf level
|
|
||||||
// upward, as described in the HDF5 spec (III.A.2 "Computing the Size
|
|
||||||
// of B-tree Nodes").
|
|
||||||
//
|
|
||||||
// We use `max_leaf_nrec * 2` as a conservative upper bound. This
|
|
||||||
// over-estimates the nrec encoding width, which means we may read
|
|
||||||
// slightly more bytes per child pointer than strictly necessary, but
|
|
||||||
// never fewer. The over-read bytes are harmless because we only
|
|
||||||
// decode `num_records` entries (the actual count from the node header).
|
|
||||||
//
|
|
||||||
// Known limitation: for very deep trees (depth > 3) with small record
|
|
||||||
// sizes, the true max could exceed this estimate, causing us to
|
|
||||||
// under-allocate the nrec encoding width and misparse child pointers.
|
|
||||||
// In practice, HDF5 B-tree v2 depths rarely exceed 2-3.
|
|
||||||
max_leaf_nrec * 2
|
|
||||||
};
|
|
||||||
let nrec_width = bytes_for_max_records(max_nrec_child);
|
|
||||||
|
|
||||||
// Total records in subtree width (only if depth > 1)
|
|
||||||
let total_nrec_width = if depth > 1 {
|
|
||||||
// Width to hold total records in a subtree
|
|
||||||
// We compute max possible total records at this subtree depth
|
|
||||||
let max_total = header_max_total_records(max_leaf_nrec, depth - 1);
|
|
||||||
bytes_for_max_records(max_total)
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
|
|
||||||
let num_children = nr + 1;
|
|
||||||
let child_ptr_size = offset_size as usize + nrec_width + total_nrec_width;
|
|
||||||
ensure_len(file_data, pos, num_children * child_ptr_size)?;
|
|
||||||
|
|
||||||
// Read child pointers
|
|
||||||
let mut children = Vec::with_capacity(num_children);
|
|
||||||
for _ in 0..num_children {
|
|
||||||
let addr = read_offset(file_data, pos, offset_size)?;
|
|
||||||
pos += offset_size as usize;
|
|
||||||
let child_nrec = read_var_uint(file_data, pos, nrec_width)? as u16;
|
|
||||||
pos += nrec_width;
|
|
||||||
pos += total_nrec_width; // skip total records in subtree
|
|
||||||
children.push((addr, child_nrec));
|
|
||||||
}
|
|
||||||
|
|
||||||
// Interleave: child[0], record[0], child[1], record[1], ..., child[nr]
|
// Interleave: child[0], record[0], child[1], record[1], ..., child[nr]
|
||||||
// We collect child[0] records, then record[0], then child[1], etc.
|
// We collect child[0] records, then record[0], then child[1], etc.
|
||||||
for (i, &(child_addr, child_nrec)) in children.iter().enumerate() {
|
// A child that fails does not stop the walk: the others are still
|
||||||
if child_depth == 0 {
|
// descended into (their records are dropped with the result), then the
|
||||||
// Before parsing, so a refused tree is not also a large allocation.
|
// first error is returned, as when stopping there. A storage that
|
||||||
spend(budget, usize::from(child_nrec))?;
|
// records what it lacks (see `storage::touch`) so learns every node the
|
||||||
let leaf_recs =
|
// walk can reach in one attempt. The record budget is spent as before,
|
||||||
parse_leaf_records(file_data, child_addr as usize, child_nrec, record_size)?;
|
// so the walk is no longer than a successful one.
|
||||||
out.extend(leaf_recs);
|
let mut failed = None;
|
||||||
} else {
|
for (i, &(child_addr, child_nrec)) in node.children.iter().enumerate() {
|
||||||
collect_internal_records(
|
if failed.is_some() && *budget == 0 {
|
||||||
file_data,
|
// The record budget is spent: the tree is refused, and a walk
|
||||||
child_addr as usize,
|
// over what is left could be as long as the one it bounds.
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if let Err(e) = (|| -> Result<(), FormatError> {
|
||||||
|
if child_depth == 0 {
|
||||||
|
// Before parsing, so a refused tree is not also a large allocation.
|
||||||
|
spend(budget, usize::from(child_nrec))?;
|
||||||
|
let leaf_recs = parse_leaf_records(
|
||||||
|
file,
|
||||||
|
to_usize(child_addr)?,
|
||||||
|
child_nrec,
|
||||||
|
record_size,
|
||||||
|
node_size,
|
||||||
|
)?;
|
||||||
|
out.extend(leaf_recs);
|
||||||
|
} else {
|
||||||
|
collect_internal_records(
|
||||||
|
file,
|
||||||
|
to_usize(child_addr)?,
|
||||||
|
child_nrec,
|
||||||
|
child_depth,
|
||||||
|
record_size,
|
||||||
|
node_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
budget,
|
||||||
|
out,
|
||||||
|
)?;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Add record[i] (except after the last child)
|
||||||
|
if i < nr {
|
||||||
|
let data = node.record(i, rs)?;
|
||||||
|
spend(budget, 1)?;
|
||||||
|
out.push(BTreeV2Record {
|
||||||
|
data: data.to_vec(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
})() {
|
||||||
|
failed.get_or_insert(e);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
match failed {
|
||||||
|
Some(e) => Err(e),
|
||||||
|
None => Ok(()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The records of a B-tree v2 that fall in one key range, found by
|
||||||
|
/// descending the tree instead of reading all of it.
|
||||||
|
///
|
||||||
|
/// `cmp` places a record relative to the range: `Less` if the record sorts
|
||||||
|
/// before it, `Greater` if after, `Equal` if the record is in it. The tree
|
||||||
|
/// must be ordered consistently with `cmp`, as libhdf5 orders it (a link or
|
||||||
|
/// attribute name index by name hash, so all records with one hash form a
|
||||||
|
/// range whatever order their names are in). Only the nodes whose key
|
||||||
|
/// interval overlaps the range are read: O(depth) nodes plus those holding
|
||||||
|
/// the matches. Matches come in tree order.
|
||||||
|
pub fn find_btree_v2_records(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
find_btree_v2_records_in(file_data, header, offset_size, cmp)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_btree_v2_records`] over any [`Storage`]: one bounded read per
|
||||||
|
/// node visited.
|
||||||
|
pub fn find_btree_v2_records_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset_size: u8,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||||
|
if header.total_records == 0 || header.num_records_in_root == 0 {
|
||||||
|
return Ok(Vec::new());
|
||||||
|
}
|
||||||
|
if header.depth > MAX_DEPTH {
|
||||||
|
return Err(FormatError::NestingDepthExceeded);
|
||||||
|
}
|
||||||
|
// As in `collect_btree_v2_records`: a valid tree cannot hold more
|
||||||
|
// records than the file has room for, however its children are shared.
|
||||||
|
let mut budget = len_usize(file) / usize::from(header.record_size.max(1));
|
||||||
|
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
||||||
|
let mut out = Vec::new();
|
||||||
|
find_in_node(
|
||||||
|
file,
|
||||||
|
header,
|
||||||
|
to_usize(header.root_node_address)?,
|
||||||
|
header.num_records_in_root,
|
||||||
|
header.depth,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
cmp,
|
||||||
|
&mut budget,
|
||||||
|
&mut out,
|
||||||
|
)?;
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
fn find_in_node<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &BTreeV2Header,
|
||||||
|
offset: usize,
|
||||||
|
num_records: u16,
|
||||||
|
depth: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||||
|
budget: &mut usize,
|
||||||
|
out: &mut Vec<BTreeV2Record>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
spend(budget, usize::from(num_records))?;
|
||||||
|
if depth == 0 {
|
||||||
|
let records = parse_leaf_records(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
num_records,
|
||||||
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
|
)?;
|
||||||
|
out.extend(
|
||||||
|
records
|
||||||
|
.into_iter()
|
||||||
|
.filter(|r| cmp(&r.data) == Ordering::Equal),
|
||||||
|
);
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let rs = usize::from(header.record_size);
|
||||||
|
let node = read_internal_node(
|
||||||
|
file,
|
||||||
|
offset,
|
||||||
|
num_records,
|
||||||
|
depth,
|
||||||
|
header.record_size,
|
||||||
|
header.node_size,
|
||||||
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
)?;
|
||||||
|
let nr = usize::from(num_records);
|
||||||
|
let mut order = Vec::with_capacity(nr);
|
||||||
|
for i in 0..nr {
|
||||||
|
order.push(cmp(node.record(i, rs)?));
|
||||||
|
}
|
||||||
|
// Child `i` holds the keys between record `i - 1` and record `i`: it can
|
||||||
|
// hold a match unless the record before it is already past the range or
|
||||||
|
// the record after it is still before it.
|
||||||
|
for (i, &(child_addr, child_nrec)) in node.children.iter().enumerate() {
|
||||||
|
let after_left = i == 0 || order[i - 1] != Ordering::Greater;
|
||||||
|
let before_right = i == nr || order[i] != Ordering::Less;
|
||||||
|
if after_left && before_right {
|
||||||
|
find_in_node(
|
||||||
|
file,
|
||||||
|
header,
|
||||||
|
to_usize(child_addr)?,
|
||||||
child_nrec,
|
child_nrec,
|
||||||
child_depth,
|
depth - 1,
|
||||||
record_size,
|
|
||||||
node_size,
|
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
|
||||||
max_leaf_nrec,
|
max_leaf_nrec,
|
||||||
|
cmp,
|
||||||
budget,
|
budget,
|
||||||
out,
|
out,
|
||||||
)?;
|
)?;
|
||||||
}
|
}
|
||||||
|
if i < nr && order[i] == Ordering::Equal {
|
||||||
// Add record[i] (except after the last child)
|
|
||||||
if i < nr {
|
|
||||||
let rec_offset = i.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
|
||||||
expected: usize::MAX,
|
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
let rec_start =
|
|
||||||
records_start
|
|
||||||
.checked_add(rec_offset)
|
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
|
||||||
expected: usize::MAX,
|
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
let rec_end = rec_start
|
|
||||||
.checked_add(rs)
|
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
|
||||||
expected: usize::MAX,
|
|
||||||
available: file_data.len(),
|
|
||||||
})?;
|
|
||||||
if rec_end > file_data.len() {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: rec_end,
|
|
||||||
available: file_data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
spend(budget, 1)?;
|
|
||||||
out.push(BTreeV2Record {
|
out.push(BTreeV2Record {
|
||||||
data: file_data[rec_start..rec_end].to_vec(),
|
data: node.record(i, rs)?.to_vec(),
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Estimate maximum total records at a given depth (for variable-width encoding).
|
/// Most records a subtree whose root is at `depth` can hold (libhdf5's
|
||||||
fn header_max_total_records(max_leaf_nrec: u64, depth: u16) -> u64 {
|
/// `cum_max_nrec`). See [`node_info`].
|
||||||
// Conservative: branching factor * max_leaf at each level
|
fn cum_max_records(
|
||||||
let mut total = max_leaf_nrec;
|
node_size: u32,
|
||||||
for _ in 0..depth {
|
record_size: u16,
|
||||||
total = total.saturating_mul(max_leaf_nrec.max(2));
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
depth: u16,
|
||||||
|
) -> u64 {
|
||||||
|
node_info_from_leaf(node_size, record_size, offset_size, max_leaf_nrec, depth)
|
||||||
|
.last()
|
||||||
|
.map_or(max_leaf_nrec, |n| n.cum_max_nrec)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Capacity of a B-tree v2 node at one depth, as libhdf5 computes it
|
||||||
|
/// (`H5B2__hdr_init`'s `node_info`).
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub(crate) struct NodeInfo {
|
||||||
|
/// Most records one node at this depth holds.
|
||||||
|
pub(crate) max_nrec: u64,
|
||||||
|
/// Most records a subtree rooted at this depth holds.
|
||||||
|
pub(crate) cum_max_nrec: u64,
|
||||||
|
/// Bytes a subtree's total record count takes in a pointer to a node
|
||||||
|
/// at this depth (0 for a leaf, whose count is its own).
|
||||||
|
pub(crate) cum_max_nrec_size: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Node capacities for depths `0..=depth` (entry `d` for depth `d`): a leaf
|
||||||
|
/// holds `max_nrec(0)` records; an internal node at depth `d` holds
|
||||||
|
/// `max_nrec(d)` records and `max_nrec(d) + 1` subtrees of depth `d - 1`,
|
||||||
|
/// where `max_nrec(d)` is what fits in a node once each record is paired
|
||||||
|
/// with a child pointer of the width depth `d` needs (address, the child's
|
||||||
|
/// record count in the width a *leaf's* maximum needs, and below the first
|
||||||
|
/// internal level the child subtree's total in the width its maximum
|
||||||
|
/// needs), with one pointer more than records.
|
||||||
|
pub(crate) fn node_info(
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
depth: u16,
|
||||||
|
) -> Vec<NodeInfo> {
|
||||||
|
let max_leaf = max_records_leaf(node_size, record_size);
|
||||||
|
node_info_from_leaf(node_size, record_size, offset_size, max_leaf, depth)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn node_info_from_leaf(
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
depth: u16,
|
||||||
|
) -> Vec<NodeInfo> {
|
||||||
|
// Internal node overhead: signature(4) + version(1) + type(1) + checksum(4).
|
||||||
|
const PREFIX: u64 = 10;
|
||||||
|
let nrec_width = bytes_for_max_records(max_leaf_nrec) as u64;
|
||||||
|
let mut info = Vec::with_capacity(usize::from(depth) + 1);
|
||||||
|
info.push(NodeInfo {
|
||||||
|
max_nrec: max_leaf_nrec,
|
||||||
|
cum_max_nrec: max_leaf_nrec,
|
||||||
|
cum_max_nrec_size: 0,
|
||||||
|
});
|
||||||
|
for d in 1..=depth {
|
||||||
|
let below = info[usize::from(d) - 1];
|
||||||
|
let ptr = u64::from(offset_size)
|
||||||
|
+ nrec_width
|
||||||
|
+ if d > 1 {
|
||||||
|
below.cum_max_nrec_size as u64
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
let max_nrec = u64::from(node_size)
|
||||||
|
.saturating_sub(PREFIX)
|
||||||
|
.saturating_sub(ptr)
|
||||||
|
/ (u64::from(record_size) + ptr).max(1);
|
||||||
|
let cum = max_nrec
|
||||||
|
.saturating_add(1)
|
||||||
|
.saturating_mul(below.cum_max_nrec)
|
||||||
|
.saturating_add(max_nrec);
|
||||||
|
info.push(NodeInfo {
|
||||||
|
max_nrec,
|
||||||
|
cum_max_nrec: cum,
|
||||||
|
cum_max_nrec_size: bytes_for_max_records(cum),
|
||||||
|
});
|
||||||
}
|
}
|
||||||
total
|
info
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -512,9 +853,15 @@ mod tests {
|
|||||||
child_nrec: u64,
|
child_nrec: u64,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let max_leaf = max_records_leaf(node_size, record_size);
|
let max_leaf = max_records_leaf(node_size, record_size);
|
||||||
let nrec_width = bytes_for_max_records(if depth == 1 { max_leaf } else { max_leaf * 2 });
|
let nrec_width = bytes_for_max_records(max_leaf);
|
||||||
let total_width = if depth > 1 {
|
let total_width = if depth > 1 {
|
||||||
bytes_for_max_records(header_max_total_records(max_leaf, depth - 1))
|
bytes_for_max_records(cum_max_records(
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
8,
|
||||||
|
max_leaf,
|
||||||
|
depth - 1,
|
||||||
|
))
|
||||||
} else {
|
} else {
|
||||||
0
|
0
|
||||||
};
|
};
|
||||||
@@ -526,6 +873,8 @@ mod tests {
|
|||||||
buf.extend_from_slice(&child_nrec.to_le_bytes()[..nrec_width]);
|
buf.extend_from_slice(&child_nrec.to_le_bytes()[..nrec_width]);
|
||||||
buf.resize(buf.len() + total_width, 0);
|
buf.resize(buf.len() + total_width, 0);
|
||||||
}
|
}
|
||||||
|
let sum = crate::checksum::jenkins_lookup3(&buf);
|
||||||
|
buf.extend_from_slice(&sum.to_le_bytes());
|
||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -673,4 +1022,18 @@ mod tests {
|
|||||||
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
||||||
assert!(records.is_empty());
|
assert!(records.is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn subtree_capacity_matches_libhdf5() {
|
||||||
|
// A link-name index (11-byte records, 512-byte nodes, 8-byte
|
||||||
|
// addresses): libhdf5's H5B2__hdr_init gives 45 records per leaf,
|
||||||
|
// then cum_max_nrec 1 149 at depth 1 and 26 449 at depth 2 — two
|
||||||
|
// bytes of subtree count in a depth-3 root's child pointers, where
|
||||||
|
// leaf_max^3 = 91 125 would need three.
|
||||||
|
let leaf = max_records_leaf(512, 11);
|
||||||
|
assert_eq!(leaf, 45);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 0), 45);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 1), 1_149);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 2), 26_449);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,503 @@
|
|||||||
|
//! Writing version-2 B-trees: a header (`BTHD`) and its nodes, leaves
|
||||||
|
//! (`BTLF`) and, for more records than one leaf holds, internal nodes
|
||||||
|
//! (`BTIN`) to any depth.
|
||||||
|
//!
|
||||||
|
//! Node capacities come from [`crate::btree_v2::node_info`], the arithmetic
|
||||||
|
//! libhdf5 uses (`H5B2__hdr_init`) and the reader decodes pointers with, so
|
||||||
|
//! the pointer widths the writer encodes are the ones every reader expects.
|
||||||
|
|
||||||
|
use crate::addr::saturating_usize;
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::btree_v2::{NodeInfo, bytes_for_max_records, node_info};
|
||||||
|
use crate::checksum::jenkins_lookup3;
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// How a B-tree is laid out: its record type and node geometry, as the
|
||||||
|
/// header records them.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub(crate) struct BTreeV2Params {
|
||||||
|
/// Record type (5: link names, 6: link creation order, 8: attribute
|
||||||
|
/// names, 9: attribute creation order, 10/11: chunks).
|
||||||
|
pub(crate) tree_type: u8,
|
||||||
|
/// Bytes per node.
|
||||||
|
pub(crate) node_size: u32,
|
||||||
|
/// Bytes per record.
|
||||||
|
pub(crate) record_size: u16,
|
||||||
|
/// Split and merge percentages. The writer fills nodes itself; these
|
||||||
|
/// only tell libhdf5 when to split and merge as it modifies the tree.
|
||||||
|
pub(crate) split_percent: u8,
|
||||||
|
pub(crate) merge_percent: u8,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Size of a B-tree v2 header.
|
||||||
|
pub(crate) fn header_size(offset_size: u8, length_size: u8) -> usize {
|
||||||
|
4 + 1 + 1 + 4 + 2 + 2 + 1 + 1 + offset_size as usize + 2 + length_size as usize + 4
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Deepest tree the writer builds. Even at the smallest fan-out libhdf5's
|
||||||
|
/// arithmetic allows, a few levels hold more records than any file could.
|
||||||
|
const MAX_WRITE_DEPTH: u16 = 32;
|
||||||
|
|
||||||
|
/// Write a B-tree v2 holding `records` (`record_size` bytes each,
|
||||||
|
/// concatenated, already in the tree's key order) at `addr`: the header,
|
||||||
|
/// then its nodes, each `node_size` bytes. No records gives a header with
|
||||||
|
/// an undefined root.
|
||||||
|
///
|
||||||
|
/// The tree is as shallow as the node size allows: a single leaf when the
|
||||||
|
/// records fit one, otherwise internal nodes above leaves. Records are
|
||||||
|
/// spread evenly over each node's children, so every node but the root is
|
||||||
|
/// at least about half full (above libhdf5's merge threshold, which is below
|
||||||
|
/// half), and each node holds at most its depth's maximum.
|
||||||
|
pub(crate) fn build_btree_v2(
|
||||||
|
p: BTreeV2Params,
|
||||||
|
records: &[u8],
|
||||||
|
addr: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let rs = usize::from(p.record_size);
|
||||||
|
if rs == 0 || !records.len().is_multiple_of(rs) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"B-tree v2 records are {} bytes, not a multiple of the record size {rs}",
|
||||||
|
records.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let n = (records.len() / rs) as u64;
|
||||||
|
let hdr_len = header_size(offset_size, length_size);
|
||||||
|
|
||||||
|
// The shallowest depth whose subtree can hold every record.
|
||||||
|
let mut info = node_info(p.node_size, p.record_size, offset_size, 0);
|
||||||
|
let max_leaf = info[0].max_nrec;
|
||||||
|
if max_leaf == 0 || max_leaf > u64::from(u16::MAX) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"a {}-byte B-tree v2 node holds {max_leaf} {}-byte records; \
|
||||||
|
a node holds 1 to 65535",
|
||||||
|
p.node_size, p.record_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let mut depth = 0u16;
|
||||||
|
while info[usize::from(depth)].cum_max_nrec < n {
|
||||||
|
depth += 1;
|
||||||
|
if depth > MAX_WRITE_DEPTH {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"{n} records do not fit a B-tree v2 of {}-byte nodes",
|
||||||
|
p.node_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
info = node_info(p.node_size, p.record_size, offset_size, depth);
|
||||||
|
let max = info[usize::from(depth)].max_nrec;
|
||||||
|
if max == 0 || max > u64::from(u16::MAX) {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"a {}-byte B-tree v2 internal node holds {max} records; \
|
||||||
|
a node holds 1 to 65535",
|
||||||
|
p.node_size
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut w = TreeWriter {
|
||||||
|
p,
|
||||||
|
records,
|
||||||
|
info: &info,
|
||||||
|
nrec_width: bytes_for_max_records(max_leaf),
|
||||||
|
offset_size,
|
||||||
|
first_node: addr + hdr_len as u64,
|
||||||
|
nodes: Vec::new(),
|
||||||
|
};
|
||||||
|
let root = (n > 0)
|
||||||
|
.then(|| w.node(depth, 0, saturating_usize(n)))
|
||||||
|
.transpose()?;
|
||||||
|
|
||||||
|
let mut out = Vec::with_capacity(hdr_len + w.nodes.len() * p.node_size as usize);
|
||||||
|
out.extend_from_slice(b"BTHD");
|
||||||
|
out.push(0); // version
|
||||||
|
out.push(p.tree_type);
|
||||||
|
out.extend_from_slice(&p.node_size.to_le_bytes());
|
||||||
|
out.extend_from_slice(&p.record_size.to_le_bytes());
|
||||||
|
out.extend_from_slice(&depth.to_le_bytes());
|
||||||
|
out.push(p.split_percent);
|
||||||
|
out.push(p.merge_percent);
|
||||||
|
match root {
|
||||||
|
Some(r) => push_uint(&mut out, r.addr, offset_size as usize),
|
||||||
|
None => out.extend(core::iter::repeat_n(0xFF, offset_size as usize)),
|
||||||
|
}
|
||||||
|
let root_nrec = root.map_or(0, |r| r.nrec);
|
||||||
|
out.extend_from_slice(&(root_nrec as u16).to_le_bytes());
|
||||||
|
push_uint(&mut out, n, length_size as usize);
|
||||||
|
let sum = jenkins_lookup3(&out);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len(), hdr_len);
|
||||||
|
for node in &w.nodes {
|
||||||
|
out.extend_from_slice(node);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A written node, as its parent points at it.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
struct NodeRef {
|
||||||
|
addr: u64,
|
||||||
|
/// Records in the node itself.
|
||||||
|
nrec: u64,
|
||||||
|
/// Records in the subtree it roots.
|
||||||
|
all_nrec: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
struct TreeWriter<'a> {
|
||||||
|
p: BTreeV2Params,
|
||||||
|
records: &'a [u8],
|
||||||
|
info: &'a [NodeInfo],
|
||||||
|
/// Width of a child's record count: what a leaf's maximum needs.
|
||||||
|
nrec_width: usize,
|
||||||
|
offset_size: u8,
|
||||||
|
/// Address of the first node (right after the header).
|
||||||
|
first_node: u64,
|
||||||
|
/// Nodes in file order (children before their parent).
|
||||||
|
nodes: Vec<Vec<u8>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TreeWriter<'_> {
|
||||||
|
fn record(&self, i: usize) -> &[u8] {
|
||||||
|
let rs = usize::from(self.p.record_size);
|
||||||
|
&self.records[i * rs..(i + 1) * rs]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn push_node(&mut self, mut node: Vec<u8>) -> u64 {
|
||||||
|
// The checksum covers the node up to it, not the padding after.
|
||||||
|
let sum = jenkins_lookup3(&node);
|
||||||
|
node.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert!(node.len() <= self.p.node_size as usize);
|
||||||
|
node.resize(self.p.node_size as usize, 0);
|
||||||
|
let addr = self.first_node + self.nodes.len() as u64 * u64::from(self.p.node_size);
|
||||||
|
self.nodes.push(node);
|
||||||
|
addr
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the subtree of `depth` holding records `first..first + n`.
|
||||||
|
fn node(&mut self, depth: u16, first: usize, n: usize) -> Result<NodeRef, FormatError> {
|
||||||
|
let rs = usize::from(self.p.record_size);
|
||||||
|
let mut node = Vec::with_capacity(self.p.node_size as usize);
|
||||||
|
if depth == 0 {
|
||||||
|
debug_assert!(n as u64 <= self.info[0].max_nrec);
|
||||||
|
node.extend_from_slice(b"BTLF");
|
||||||
|
node.push(0); // version
|
||||||
|
node.push(self.p.tree_type);
|
||||||
|
node.extend_from_slice(&self.records[first * rs..(first + n) * rs]);
|
||||||
|
let addr = self.push_node(node);
|
||||||
|
return Ok(NodeRef {
|
||||||
|
addr,
|
||||||
|
nrec: n as u64,
|
||||||
|
all_nrec: n as u64,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
// As few children as hold the records, at least two, with the
|
||||||
|
// records spread evenly: `k` children and `k - 1` records between
|
||||||
|
// them.
|
||||||
|
let below = self.info[usize::from(depth) - 1].cum_max_nrec;
|
||||||
|
let k = (n as u64 + 1).div_ceil(below + 1).max(2);
|
||||||
|
let max = self.info[usize::from(depth)].max_nrec;
|
||||||
|
if k - 1 > max || (n as u64) < k - 1 + k {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"cannot spread {n} B-tree v2 records over {k} children at depth {depth}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let k = saturating_usize(k);
|
||||||
|
let in_children = n - (k - 1);
|
||||||
|
let (base, extra) = (in_children / k, in_children % k);
|
||||||
|
|
||||||
|
let mut children = Vec::with_capacity(k);
|
||||||
|
let mut separators = Vec::with_capacity(k - 1);
|
||||||
|
let mut next = first;
|
||||||
|
for c in 0..k {
|
||||||
|
let m = base + usize::from(c < extra);
|
||||||
|
children.push(self.node(depth - 1, next, m)?);
|
||||||
|
next += m;
|
||||||
|
if c + 1 < k {
|
||||||
|
separators.push(next);
|
||||||
|
next += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
debug_assert_eq!(next, first + n);
|
||||||
|
|
||||||
|
node.extend_from_slice(b"BTIN");
|
||||||
|
node.push(0); // version
|
||||||
|
node.push(self.p.tree_type);
|
||||||
|
for &s in &separators {
|
||||||
|
node.extend_from_slice(self.record(s));
|
||||||
|
}
|
||||||
|
let total_width = if depth > 1 {
|
||||||
|
self.info[usize::from(depth) - 1].cum_max_nrec_size
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
for c in &children {
|
||||||
|
push_uint(&mut node, c.addr, self.offset_size as usize);
|
||||||
|
push_uint(&mut node, c.nrec, self.nrec_width);
|
||||||
|
if depth > 1 {
|
||||||
|
push_uint(&mut node, c.all_nrec, total_width);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let addr = self.push_node(node);
|
||||||
|
Ok(NodeRef {
|
||||||
|
addr,
|
||||||
|
nrec: (k - 1) as u64,
|
||||||
|
all_nrec: n as u64,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append `v` as a `width`-byte little-endian integer.
|
||||||
|
fn push_uint(buf: &mut Vec<u8>, v: u64, width: usize) {
|
||||||
|
let bytes = v.to_le_bytes();
|
||||||
|
buf.extend_from_slice(&bytes[..width.min(8)]);
|
||||||
|
buf.extend(vec![0u8; width.saturating_sub(8)]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||||
|
|
||||||
|
fn params(node_size: u32, record_size: u16) -> BTreeV2Params {
|
||||||
|
BTreeV2Params {
|
||||||
|
tree_type: 5,
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
split_percent: 100,
|
||||||
|
merge_percent: 40,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `n` 11-byte records: a big-endian counter, so byte order is key order.
|
||||||
|
fn records(n: usize, rs: usize) -> Vec<u8> {
|
||||||
|
let mut out = Vec::with_capacity(n * rs);
|
||||||
|
for i in 0..n {
|
||||||
|
let mut r = vec![0u8; rs];
|
||||||
|
r[..8].copy_from_slice(&(i as u64).to_be_bytes());
|
||||||
|
out.extend_from_slice(&r);
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
fn roundtrip(node_size: u32, rs: u16, n: usize, os: u8, ls: u8) -> BTreeV2Header {
|
||||||
|
let recs = records(n, usize::from(rs));
|
||||||
|
let base = 4096u64;
|
||||||
|
let tree = build_btree_v2(params(node_size, rs), &recs, base, os, ls).unwrap();
|
||||||
|
let mut file = vec![0u8; base as usize];
|
||||||
|
file.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&file, base as usize, os, ls).unwrap();
|
||||||
|
assert_eq!(hdr.total_records, n as u64);
|
||||||
|
let got = collect_btree_v2_records(&file, &hdr, os, ls).unwrap();
|
||||||
|
assert_eq!(got.len(), n);
|
||||||
|
let flat: Vec<u8> = got.into_iter().flat_map(|r| r.data).collect();
|
||||||
|
assert_eq!(flat, recs, "node {node_size} rs {rs} n {n}");
|
||||||
|
hdr
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn one_leaf_then_deeper_trees_read_back_in_order() {
|
||||||
|
// 512-byte nodes of 11-byte records: 45 per leaf, 1149 at depth 1,
|
||||||
|
// 26 449 at depth 2.
|
||||||
|
let info = node_info(512, 11, 8, 3);
|
||||||
|
assert_eq!(
|
||||||
|
info.iter().map(|i| i.cum_max_nrec).collect::<Vec<_>>(),
|
||||||
|
[45, 1149, 26_449, 608_349]
|
||||||
|
);
|
||||||
|
for (n, depth) in [
|
||||||
|
(0, 0),
|
||||||
|
(1, 0),
|
||||||
|
(45, 0),
|
||||||
|
(46, 1),
|
||||||
|
(1149, 1),
|
||||||
|
(1150, 2),
|
||||||
|
(26_449, 2),
|
||||||
|
(26_450, 3),
|
||||||
|
(100_000, 3),
|
||||||
|
] {
|
||||||
|
let hdr = roundtrip(512, 11, n, 8, 8);
|
||||||
|
assert_eq!(hdr.depth, depth, "{n} records");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pointer_widths_follow_the_offset_and_length_sizes() {
|
||||||
|
for (os, ls) in [(4, 4), (8, 4), (4, 8), (2, 2)] {
|
||||||
|
roundtrip(512, 11, 5000, os, ls);
|
||||||
|
}
|
||||||
|
// Wide counts: a leaf of 2048 bytes / 9-byte records (226, one byte)
|
||||||
|
// and deeper subtree totals of three bytes.
|
||||||
|
roundtrip(2048, 9, 300_000, 8, 8);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn every_node_is_within_its_capacity_and_above_the_merge_threshold() {
|
||||||
|
let rs = 17u16;
|
||||||
|
let n = 70_000usize;
|
||||||
|
let info = node_info(512, rs, 8, 3);
|
||||||
|
let recs = records(n, usize::from(rs));
|
||||||
|
let tree = build_btree_v2(params(512, rs), &recs, 0, 8, 8).unwrap();
|
||||||
|
let hdr_len = header_size(8, 8);
|
||||||
|
let nodes = (tree.len() - hdr_len) / 512;
|
||||||
|
for i in 0..nodes {
|
||||||
|
let node = &tree[hdr_len + i * 512..hdr_len + (i + 1) * 512];
|
||||||
|
let sig = &node[..4];
|
||||||
|
if sig == b"BTLF" {
|
||||||
|
continue; // counts checked through the parents below
|
||||||
|
}
|
||||||
|
assert_eq!(sig, b"BTIN");
|
||||||
|
}
|
||||||
|
// Walk from the header: each child's count within [40%, 100%].
|
||||||
|
let hdr = BTreeV2Header::parse(&tree, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(hdr.depth, 3);
|
||||||
|
assert!(u64::from(hdr.num_records_in_root) <= info[3].max_nrec);
|
||||||
|
fn walk(tree: &[u8], addr: usize, nrec: usize, depth: usize, info: &[NodeInfo], rs: usize) {
|
||||||
|
if depth == 0 {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let nrec_w = bytes_for_max_records(info[0].max_nrec);
|
||||||
|
let tot_w = if depth > 1 {
|
||||||
|
info[depth - 1].cum_max_nrec_size
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
let mut pos = addr + 6 + nrec * rs;
|
||||||
|
for _ in 0..=nrec {
|
||||||
|
let a = u64::from_le_bytes(tree[pos..pos + 8].try_into().unwrap()) as usize;
|
||||||
|
pos += 8;
|
||||||
|
let mut c = 0usize;
|
||||||
|
for b in 0..nrec_w {
|
||||||
|
c |= usize::from(tree[pos + b]) << (8 * b);
|
||||||
|
}
|
||||||
|
pos += nrec_w + tot_w;
|
||||||
|
let max = info[depth - 1].max_nrec as usize;
|
||||||
|
assert!(c <= max && c * 100 > max * 40, "{c} of {max}");
|
||||||
|
walk(tree, a, c, depth - 1, info, rs);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
walk(
|
||||||
|
&tree,
|
||||||
|
hdr.root_node_address as usize,
|
||||||
|
usize::from(hdr.num_records_in_root),
|
||||||
|
3,
|
||||||
|
&info,
|
||||||
|
usize::from(rs),
|
||||||
|
);
|
||||||
|
assert!(nodes > 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Descending to a key range finds exactly the records a full read
|
||||||
|
/// holds in it — runs of equal keys that straddle node boundaries
|
||||||
|
/// included — at every depth, and nothing for keys not in the tree.
|
||||||
|
#[test]
|
||||||
|
fn a_key_range_search_matches_a_full_scan() {
|
||||||
|
use crate::btree_v2::find_btree_v2_records;
|
||||||
|
use core::cmp::Ordering;
|
||||||
|
let rs = 11usize;
|
||||||
|
// Keys 0, 0, 0, 2, 2, 2, 4, ...: runs of three, odd keys missing.
|
||||||
|
for n in [1usize, 45, 46, 1150, 30_000] {
|
||||||
|
let mut recs = Vec::with_capacity(n * rs);
|
||||||
|
for i in 0..n {
|
||||||
|
let mut r = vec![0u8; rs];
|
||||||
|
r[..8].copy_from_slice(&((i / 3 * 2) as u64).to_be_bytes());
|
||||||
|
r[8..].copy_from_slice(&[(i % 3) as u8, 0, 0]);
|
||||||
|
recs.extend_from_slice(&r);
|
||||||
|
}
|
||||||
|
let base = 4096u64;
|
||||||
|
let tree = build_btree_v2(params(512, 11), &recs, base, 8, 8).unwrap();
|
||||||
|
let mut file = vec![0u8; base as usize];
|
||||||
|
file.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&file, base as usize, 8, 8).unwrap();
|
||||||
|
let all = collect_btree_v2_records(&file, &hdr, 8, 8).unwrap();
|
||||||
|
let key = |r: &[u8]| u64::from_be_bytes(r[..8].try_into().unwrap());
|
||||||
|
let last = key(&all[n - 1].data);
|
||||||
|
let probes = (0..=last + 1).step_by(if n > 1000 { 37 } else { 1 });
|
||||||
|
for k in probes.chain([last, last + 1, u64::MAX]) {
|
||||||
|
let found =
|
||||||
|
find_btree_v2_records(&file, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k)).unwrap();
|
||||||
|
let want: Vec<&[u8]> = all
|
||||||
|
.iter()
|
||||||
|
.map(|r| r.data.as_slice())
|
||||||
|
.filter(|r| key(r) == k)
|
||||||
|
.collect();
|
||||||
|
let got: Vec<&[u8]> = found.iter().map(|r| r.data.as_slice()).collect();
|
||||||
|
assert_eq!(got, want, "n {n} key {k}");
|
||||||
|
assert_eq!(
|
||||||
|
got.len(),
|
||||||
|
if k % 2 == 0 && k <= last {
|
||||||
|
want.len()
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Every record, or none, when the whole tree is in or out of range.
|
||||||
|
let every = find_btree_v2_records(&file, &hdr, 8, &mut |_| Ordering::Equal).unwrap();
|
||||||
|
assert_eq!(every.len(), n);
|
||||||
|
let none = find_btree_v2_records(&file, &hdr, 8, &mut |_| Ordering::Less).unwrap();
|
||||||
|
assert!(none.is_empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A two-level tree read through a `read_at`-only storage gives what
|
||||||
|
/// the slice gives — records, descents and errors — whole, truncated
|
||||||
|
/// at every length, and with each byte of its nodes flipped, and each
|
||||||
|
/// node costs one read.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::btree_v2::{
|
||||||
|
collect_btree_v2_records_in, find_btree_v2_records, find_btree_v2_records_in,
|
||||||
|
};
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let (rs, n, base) = (11usize, 120usize, 64usize);
|
||||||
|
let recs = records(n, rs);
|
||||||
|
let tree = build_btree_v2(params(128, 11), &recs, base as u64, 8, 8).unwrap();
|
||||||
|
let mut whole = vec![0u8; base];
|
||||||
|
whole.extend_from_slice(&tree);
|
||||||
|
let hdr = BTreeV2Header::parse(&whole, base, 8, 8).unwrap();
|
||||||
|
assert!(hdr.depth >= 1, "{hdr:?}");
|
||||||
|
let key = |r: &[u8]| u64::from_be_bytes(r[..8].try_into().unwrap());
|
||||||
|
let mut files = Vec::new();
|
||||||
|
for cut in base..=whole.len() {
|
||||||
|
files.push(whole[..cut].to_vec());
|
||||||
|
}
|
||||||
|
for at in base..whole.len() {
|
||||||
|
let mut bad = whole.clone();
|
||||||
|
bad[at] ^= 0x5a;
|
||||||
|
files.push(bad);
|
||||||
|
}
|
||||||
|
let mut ok = 0;
|
||||||
|
for f in &files {
|
||||||
|
let st = CountingStorage::new(f.clone());
|
||||||
|
let want_h = BTreeV2Header::parse(f, base, 8, 8);
|
||||||
|
let got_h = BTreeV2Header::parse_in(&st, base as u64, 8, 8);
|
||||||
|
assert_eq!(format!("{got_h:?}"), format!("{want_h:?}"));
|
||||||
|
// The nodes of the intact header, over each damaged file.
|
||||||
|
let want = collect_btree_v2_records(f, &hdr, 8, 8);
|
||||||
|
st.reset();
|
||||||
|
let got = collect_btree_v2_records_in(&st, &hdr, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
if want.is_ok() {
|
||||||
|
ok += 1;
|
||||||
|
assert!(st.reads() <= 1 + n as u64 / 3, "{} reads", st.reads());
|
||||||
|
}
|
||||||
|
for k in [0u64, 7, 60, 119, 500] {
|
||||||
|
let want = find_btree_v2_records(f, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k));
|
||||||
|
let got = find_btree_v2_records_in(&st, &hdr, 8, &mut |r: &[u8]| key(r).cmp(&k));
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(ok > 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_node_too_small_or_too_big_is_an_error() {
|
||||||
|
assert!(build_btree_v2(params(16, 11), &records(1, 11), 0, 8, 8).is_err());
|
||||||
|
// A leaf with room for more than 65 535 records.
|
||||||
|
assert!(build_btree_v2(params(1 << 20, 11), &records(1, 11), 0, 8, 8).is_err());
|
||||||
|
// Records that are not whole.
|
||||||
|
assert!(build_btree_v2(params(512, 11), &[0u8; 12], 0, 8, 8).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,79 @@
|
|||||||
|
//! Large output buffers backed by transparent huge pages where the OS offers
|
||||||
|
//! them.
|
||||||
|
//!
|
||||||
|
//! A fresh multi-megabyte `Vec` is mapped lazily by the kernel: the first
|
||||||
|
//! write to each 4 KiB page takes a page fault, and the kernel zeroes the page
|
||||||
|
//! before handing it over. For a 64 MiB read that is 16384 faults, and they
|
||||||
|
//! cost far more than the copy that fills the buffer — single-threaded
|
||||||
|
//! contiguous reads ran at about a quarter of h5py's speed because of them.
|
||||||
|
//! numpy (so h5py) avoids this by asking for transparent huge pages
|
||||||
|
//! (`madvise(MADV_HUGEPAGE)`) on every allocation of 4 MiB or more, which
|
||||||
|
//! turns 512 faults into one; this module does the same.
|
||||||
|
//!
|
||||||
|
//! The advice only changes how the pages are backed, never their contents, so
|
||||||
|
//! it is harmless when it cannot be honoured (THP disabled, not Linux, a
|
||||||
|
//! region that is part of the heap): the buffer is then exactly what it would
|
||||||
|
//! have been without it.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
|
/// Buffers smaller than this are left alone (numpy uses the same threshold).
|
||||||
|
#[cfg(any(target_os = "linux", test))]
|
||||||
|
pub(crate) const HUGE_PAGE_THRESHOLD: usize = 4 << 20;
|
||||||
|
|
||||||
|
/// Advise the kernel to back `[ptr, ptr + len)` with transparent huge pages,
|
||||||
|
/// when `len` is large enough to benefit. Call it before the first write so
|
||||||
|
/// the faults happen at huge-page granularity.
|
||||||
|
#[inline]
|
||||||
|
pub(crate) fn advise_huge_pages(ptr: *const u8, len: usize) {
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
if len >= HUGE_PAGE_THRESHOLD {
|
||||||
|
const PAGE: usize = 4096;
|
||||||
|
let start = (ptr as usize).next_multiple_of(PAGE);
|
||||||
|
let end = (ptr as usize + len) & !(PAGE - 1);
|
||||||
|
if end > start {
|
||||||
|
// SAFETY: `[start, end)` lies inside an allocation of `len` bytes
|
||||||
|
// at `ptr` that the caller owns, and is page aligned as madvise
|
||||||
|
// requires. MADV_HUGEPAGE does not change the memory's contents or
|
||||||
|
// validity; on failure (EINVAL when THP is compiled out, etc.) the
|
||||||
|
// region is simply left as it was, so the result is ignored.
|
||||||
|
unsafe {
|
||||||
|
libc::madvise(start as *mut libc::c_void, end - start, libc::MADV_HUGEPAGE);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
#[cfg(not(target_os = "linux"))]
|
||||||
|
let _ = (ptr, len);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `Vec::with_capacity(count)` for a buffer about to be filled in bulk, with
|
||||||
|
/// huge-page advice when it is large (see the module docs).
|
||||||
|
#[inline]
|
||||||
|
pub(crate) fn vec_for_bulk<T>(count: usize) -> Vec<T> {
|
||||||
|
let v: Vec<T> = Vec::with_capacity(count);
|
||||||
|
advise_huge_pages(
|
||||||
|
v.as_ptr().cast::<u8>(),
|
||||||
|
v.capacity().saturating_mul(core::mem::size_of::<T>()),
|
||||||
|
);
|
||||||
|
v
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bulk_vec_is_an_ordinary_vec() {
|
||||||
|
for count in [0usize, 1, 1000, HUGE_PAGE_THRESHOLD / 4 + 3] {
|
||||||
|
let mut v: Vec<u32> = vec_for_bulk(count);
|
||||||
|
assert!(v.capacity() >= count);
|
||||||
|
v.extend((0..count as u32).map(|i| i.wrapping_mul(2654435761)));
|
||||||
|
assert!(
|
||||||
|
v.iter()
|
||||||
|
.enumerate()
|
||||||
|
.all(|(i, &x)| x == (i as u32).wrapping_mul(2654435761))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -14,6 +14,45 @@ pub fn jenkins_lookup3(data: &[u8]) -> u32 {
|
|||||||
hashlittle(data, 0)
|
hashlittle(data, 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// HDF5's Fletcher-32 checksum, as the Fletcher-32 I/O filter (filter id 3)
|
||||||
|
/// stores it after each chunk.
|
||||||
|
///
|
||||||
|
/// A line-for-line port of `H5_checksum_fletcher32` (H5checksum.c, libhdf5
|
||||||
|
/// 1.8 through 1.14): big-endian 16-bit words summed in blocks of 360, each
|
||||||
|
/// sum reduced after a block by the ones'-complement fold
|
||||||
|
/// `(s & 0xffff) + (s >> 16)` rather than `% 65535`, an odd trailing byte
|
||||||
|
/// taken as the high byte of a last word, and a final fold of both sums.
|
||||||
|
/// The fold and `% 65535` differ whenever a sum is a non-zero multiple of
|
||||||
|
/// 65535: the fold leaves 0xffff where the modulo gives 0, so the two
|
||||||
|
/// disagree on about one chunk in 32768 and libhdf5 rejects the other's
|
||||||
|
/// checksum. This must stay the only implementation.
|
||||||
|
pub fn fletcher32(data: &[u8]) -> u32 {
|
||||||
|
let mut sum1: u32 = 0;
|
||||||
|
let mut sum2: u32 = 0;
|
||||||
|
// 360 words keep both sums inside 32 bits between folds (the bound
|
||||||
|
// libhdf5 uses: after a fold sum1 < 0x10200, so sum2 stays below
|
||||||
|
// 360 * 361 / 2 * 0xffff + 360 * 0x10200 + 0x1fffe < 2^32). The adds wrap
|
||||||
|
// like the C unsigned arithmetic all the same.
|
||||||
|
let (words, odd) = data.as_chunks::<2>();
|
||||||
|
for block in words.chunks(360) {
|
||||||
|
for w in block {
|
||||||
|
sum1 = sum1.wrapping_add((u32::from(w[0]) << 8) | u32::from(w[1]));
|
||||||
|
sum2 = sum2.wrapping_add(sum1);
|
||||||
|
}
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
}
|
||||||
|
if let [last] = odd {
|
||||||
|
sum1 = sum1.wrapping_add(u32::from(*last) << 8);
|
||||||
|
sum2 = sum2.wrapping_add(sum1);
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
}
|
||||||
|
sum1 = (sum1 & 0xffff) + (sum1 >> 16);
|
||||||
|
sum2 = (sum2 & 0xffff) + (sum2 >> 16);
|
||||||
|
(sum2 << 16) | sum1
|
||||||
|
}
|
||||||
|
|
||||||
/// Compute CRC32 (IEEE / ISO 3309) over data.
|
/// Compute CRC32 (IEEE / ISO 3309) over data.
|
||||||
///
|
///
|
||||||
/// When the `fast-checksum` feature is enabled, this uses hardware CRC32
|
/// When the `fast-checksum` feature is enabled, this uses hardware CRC32
|
||||||
@@ -207,6 +246,19 @@ fn hashlittle(data: &[u8], initval: u32) -> u32 {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Values of libhdf5's `H5_checksum_fletcher32` (h5py 3.x's bundled
|
||||||
|
/// libhdf5, called through ctypes). The first three are sums that are
|
||||||
|
/// multiples of 65535, where `% 65535` gave 0 instead of 0xffff.
|
||||||
|
#[test]
|
||||||
|
fn fletcher32_matches_libhdf5() {
|
||||||
|
assert_eq!(fletcher32(&[0x00, 0x01, 0xff, 0xfe]), 0x0001_ffff);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 720]), 0xffff_ffff);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 721]), 0xff00_ff00);
|
||||||
|
assert_eq!(fletcher32(&[0xff; 1441]), 0xff00_ff00);
|
||||||
|
assert_eq!(fletcher32(&[]), 0);
|
||||||
|
assert_eq!(fletcher32(&[7]), 0x0700_0700);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn empty_input() {
|
fn empty_input() {
|
||||||
// Empty input should return the initial state after no mixing
|
// Empty input should return the initial state after no mixing
|
||||||
|
|||||||
@@ -223,13 +223,32 @@ pub const DEFAULT_CACHE_BYTES: usize = 16 * 1024 * 1024; // 16 MiB
|
|||||||
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
||||||
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
||||||
|
|
||||||
|
/// Most datasets whose chunk index a [`ChunkCache`] keeps at once.
|
||||||
|
pub const MAX_INDEXED_DATASETS: usize = 64;
|
||||||
|
|
||||||
|
/// Most chunk-index entries, summed over all datasets, a [`ChunkCache`] keeps.
|
||||||
|
/// Least-recently-used datasets' indexes are dropped past this (the dataset
|
||||||
|
/// being read is always kept), so a file with many or huge chunked datasets
|
||||||
|
/// cannot grow the cache without bound.
|
||||||
|
pub const MAX_INDEXED_CHUNKS: usize = 1 << 20;
|
||||||
|
|
||||||
|
/// The dataset key the address-less (legacy) methods use when
|
||||||
|
/// [`ChunkCache::ensure_dataset`] has not been called.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
const UNBOUND_DATASET: u64 = u64::MAX;
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// LRU entry
|
// LRU entry
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Decompressed chunks are keyed by dataset *and* coordinate: every chunked
|
||||||
|
/// dataset has a chunk at (0, 0, ...), so the coordinate alone is ambiguous.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
type SlotKey = (u64, ChunkCoord);
|
||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
struct CachedChunk {
|
struct CachedChunk {
|
||||||
coord: ChunkCoord,
|
key: SlotKey,
|
||||||
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
||||||
/// (potentially large) decompressed chunk.
|
/// (potentially large) decompressed chunk.
|
||||||
data: Arc<CacheAlignedBuffer>,
|
data: Arc<CacheAlignedBuffer>,
|
||||||
@@ -237,21 +256,54 @@ struct CachedChunk {
|
|||||||
last_access: u64,
|
last_access: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Per-dataset index state.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
#[derive(Default)]
|
||||||
|
struct DatasetEntry {
|
||||||
|
/// Chunk coordinate -> ChunkInfo (offset + size in file).
|
||||||
|
index: Option<Arc<HashMap<ChunkCoord, ChunkInfo>>>,
|
||||||
|
/// The same chunks in the order the chunk index lists them: what
|
||||||
|
/// [`ChunkCache::chunks_for`] returns, so a cached read walks (and, on a
|
||||||
|
/// damaged file, fails at) the chunks in the same order as an uncached
|
||||||
|
/// one, rather than in hash-map order.
|
||||||
|
ordered: Option<Arc<Vec<ChunkInfo>>>,
|
||||||
|
/// Pre-built chunk index for O(1) coordinate lookups.
|
||||||
|
chunk_index: Option<Arc<ChunkIndex>>,
|
||||||
|
/// Pre-computed chunk layout for fast assembly.
|
||||||
|
chunk_layout: Option<Arc<ChunkLayout>>,
|
||||||
|
/// Tick of the last use, for dropping the least recently used dataset.
|
||||||
|
last_used: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
impl DatasetEntry {
|
||||||
|
fn weight(&self) -> usize {
|
||||||
|
self.index.as_ref().map_or(0, |m| m.len())
|
||||||
|
+ self.ordered.as_ref().map_or(0, |o| o.len())
|
||||||
|
+ self.chunk_index.as_ref().map_or(0, |c| c.num_chunks())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// ChunkCache
|
// ChunkCache
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
/// A per-dataset chunk cache with hash-based index and LRU eviction.
|
/// A per-file chunk cache: chunk indexes per dataset, plus an LRU of
|
||||||
|
/// decompressed chunks, all keyed by dataset.
|
||||||
///
|
///
|
||||||
/// # Usage
|
/// A dataset is identified by the address of its chunk index (B-tree, fixed
|
||||||
|
/// or extensible array, ...), which is unique within a file. Every method
|
||||||
|
/// that takes an `addr` works on that dataset only, so threads reading
|
||||||
|
/// different datasets through one shared cache never see each other's
|
||||||
|
/// chunks. The address-less methods (`has_index`, `populate_index`,
|
||||||
|
/// `get_decompressed`, ...) act on the dataset last bound with
|
||||||
|
/// [`Self::ensure_dataset`]; that binding is shared state, so concurrent
|
||||||
|
/// readers must use the `*_in` / `*_for` methods instead (the chunked
|
||||||
|
/// readers in [`crate::chunked_read`] do).
|
||||||
///
|
///
|
||||||
/// ```ignore
|
/// Memory is bounded: decompressed data by `max_bytes`/`max_slots` across
|
||||||
/// let cache = ChunkCache::new();
|
/// all datasets, indexes by [`MAX_INDEXED_DATASETS`] and
|
||||||
/// // Pass &cache to read_chunked_data — it will populate the index lazily.
|
/// [`MAX_INDEXED_CHUNKS`].
|
||||||
/// ```
|
|
||||||
///
|
|
||||||
/// The cache is wrapped in `Mutex` internally so it can be mutated through
|
|
||||||
/// shared references (thread-safe).
|
|
||||||
///
|
///
|
||||||
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
@@ -261,26 +313,20 @@ pub struct ChunkCache {
|
|||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
struct CacheInner {
|
struct CacheInner {
|
||||||
/// Hash index: chunk coordinate -> ChunkInfo (offset + size in file).
|
/// Per-dataset chunk indexes, keyed by chunk-index address.
|
||||||
/// Populated once per dataset on first access.
|
datasets: HashMap<u64, DatasetEntry>,
|
||||||
index: Option<HashMap<ChunkCoord, ChunkInfo>>,
|
|
||||||
|
|
||||||
/// Address of the dataset (its chunk-index base address) that the cached
|
/// Dataset the address-less methods act on (see `ensure_dataset`).
|
||||||
/// index, chunk index, layout, and decompressed slots currently belong to.
|
current: Option<u64>,
|
||||||
/// The cache is shared per file across datasets, so every cached-read entry
|
|
||||||
/// checks this and resets the per-dataset state when the dataset changes —
|
|
||||||
/// otherwise one dataset's chunk index (with its own rank) would be reused
|
|
||||||
/// for another, corrupting reads.
|
|
||||||
index_addr: Option<u64>,
|
|
||||||
|
|
||||||
/// LRU cache of decompressed chunk data.
|
/// LRU cache of decompressed chunk data.
|
||||||
slots: Vec<CachedChunk>,
|
slots: Vec<CachedChunk>,
|
||||||
|
|
||||||
/// Coordinate -> index into `slots`, for O(1) lookup instead of a linear
|
/// Key -> index into `slots`, for O(1) lookup instead of a linear
|
||||||
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
||||||
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
||||||
/// `i`, so the moved element's index entry must be updated too.
|
/// `i`, so the moved element's index entry must be updated too.
|
||||||
slot_index: HashMap<ChunkCoord, usize>,
|
slot_index: HashMap<SlotKey, usize>,
|
||||||
|
|
||||||
/// Current total bytes of cached decompressed data.
|
/// Current total bytes of cached decompressed data.
|
||||||
current_bytes: usize,
|
current_bytes: usize,
|
||||||
@@ -294,17 +340,145 @@ struct CacheInner {
|
|||||||
/// Monotonic counter for LRU ordering.
|
/// Monotonic counter for LRU ordering.
|
||||||
tick: u64,
|
tick: u64,
|
||||||
|
|
||||||
/// Last accessed chunk coordinate (for sequential detection).
|
/// Last accessed chunk (for sequential detection).
|
||||||
last_coord: Option<ChunkCoord>,
|
last_coord: Option<SlotKey>,
|
||||||
|
|
||||||
/// Access pattern statistics.
|
/// Access pattern statistics.
|
||||||
stats: AccessStats,
|
stats: AccessStats,
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-built chunk index for O(1) coordinate lookups.
|
#[cfg(feature = "std")]
|
||||||
chunk_index: Option<ChunkIndex>,
|
impl CacheInner {
|
||||||
|
fn current(&self) -> u64 {
|
||||||
|
self.current.unwrap_or(UNBOUND_DATASET)
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-computed chunk layout for fast assembly.
|
fn touch(&mut self, addr: u64) -> &mut DatasetEntry {
|
||||||
chunk_layout: Option<ChunkLayout>,
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
let entry = self.datasets.entry(addr).or_default();
|
||||||
|
entry.last_used = tick;
|
||||||
|
entry
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entry(&self, addr: u64) -> Option<&DatasetEntry> {
|
||||||
|
self.datasets.get(&addr)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop least-recently-used datasets' indexes (never `keep`'s) until the
|
||||||
|
/// dataset and chunk-entry budgets hold.
|
||||||
|
fn trim_datasets(&mut self, keep: u64) {
|
||||||
|
loop {
|
||||||
|
let total: usize = self.datasets.values().map(DatasetEntry::weight).sum();
|
||||||
|
if self.datasets.len() <= MAX_INDEXED_DATASETS && total <= MAX_INDEXED_CHUNKS {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let victim = self
|
||||||
|
.datasets
|
||||||
|
.iter()
|
||||||
|
.filter(|(a, _)| **a != keep)
|
||||||
|
.min_by_key(|(_, e)| e.last_used)
|
||||||
|
.map(|(a, _)| *a);
|
||||||
|
match victim {
|
||||||
|
Some(a) => {
|
||||||
|
self.datasets.remove(&a);
|
||||||
|
}
|
||||||
|
None => return,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn get_decompressed(&mut self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
|
||||||
|
// Track sequential vs random access
|
||||||
|
let is_sequential = self.last_coord.as_ref().is_some_and(|(prev_addr, prev)| {
|
||||||
|
// Sequential if exactly one dimension changed
|
||||||
|
let changes: usize = prev
|
||||||
|
.iter()
|
||||||
|
.zip(coord.iter())
|
||||||
|
.filter(|(a, b)| a != b)
|
||||||
|
.count();
|
||||||
|
*prev_addr == addr && changes <= 1
|
||||||
|
});
|
||||||
|
if is_sequential {
|
||||||
|
self.stats.sequential_count += 1;
|
||||||
|
} else if self.last_coord.is_some() {
|
||||||
|
self.stats.random_count += 1;
|
||||||
|
}
|
||||||
|
let key: SlotKey = (addr, coord.to_vec());
|
||||||
|
let found = if let Some(&idx) = self.slot_index.get(&key) {
|
||||||
|
self.slots[idx].last_access = tick;
|
||||||
|
Some(Arc::clone(&self.slots[idx].data))
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
self.last_coord = Some(key);
|
||||||
|
if let Some(ref data) = found {
|
||||||
|
self.stats.hits += 1;
|
||||||
|
self.stats.bytes_read += data.len() as u64;
|
||||||
|
} else {
|
||||||
|
self.stats.misses += 1;
|
||||||
|
}
|
||||||
|
found
|
||||||
|
}
|
||||||
|
|
||||||
|
fn put_decompressed(
|
||||||
|
&mut self,
|
||||||
|
key: SlotKey,
|
||||||
|
data: Arc<CacheAlignedBuffer>,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
let data_len = data.len();
|
||||||
|
|
||||||
|
// Don't cache if single chunk exceeds budget — still return the data
|
||||||
|
// to the caller, just don't retain it.
|
||||||
|
if data_len > self.max_bytes {
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check if already present
|
||||||
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
if let Some(&idx) = self.slot_index.get(&key) {
|
||||||
|
self.slots[idx].last_access = tick;
|
||||||
|
return Arc::clone(&self.slots[idx].data); // already cached
|
||||||
|
}
|
||||||
|
|
||||||
|
// Evict until we have room
|
||||||
|
while self.slots.len() >= self.max_slots
|
||||||
|
|| (self.current_bytes + data_len > self.max_bytes && !self.slots.is_empty())
|
||||||
|
{
|
||||||
|
// Find LRU slot
|
||||||
|
let lru_idx = self
|
||||||
|
.slots
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.min_by_key(|(_, s)| s.last_access)
|
||||||
|
.map(|(i, _)| i)
|
||||||
|
.unwrap();
|
||||||
|
let removed = self.slots.swap_remove(lru_idx);
|
||||||
|
self.slot_index.remove(&removed.key);
|
||||||
|
// swap_remove moved the former last element into `lru_idx` (unless
|
||||||
|
// it *was* the last element) — fix up that element's index entry.
|
||||||
|
if lru_idx < self.slots.len() {
|
||||||
|
let moved_key = self.slots[lru_idx].key.clone();
|
||||||
|
self.slot_index.insert(moved_key, lru_idx);
|
||||||
|
}
|
||||||
|
self.current_bytes -= removed.data.len();
|
||||||
|
self.stats.evictions += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
self.current_bytes += data_len;
|
||||||
|
let new_idx = self.slots.len();
|
||||||
|
self.slot_index.insert(key.clone(), new_idx);
|
||||||
|
self.slots.push(CachedChunk {
|
||||||
|
key,
|
||||||
|
data: Arc::clone(&data),
|
||||||
|
last_access: tick,
|
||||||
|
});
|
||||||
|
data
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Access pattern statistics tracked by the chunk cache.
|
/// Access pattern statistics tracked by the chunk cache.
|
||||||
@@ -356,8 +530,8 @@ impl ChunkCache {
|
|||||||
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
||||||
Self {
|
Self {
|
||||||
inner: std::sync::Mutex::new(CacheInner {
|
inner: std::sync::Mutex::new(CacheInner {
|
||||||
index: None,
|
datasets: HashMap::new(),
|
||||||
index_addr: None,
|
current: None,
|
||||||
slots: Vec::with_capacity(max_slots.min(64)),
|
slots: Vec::with_capacity(max_slots.min(64)),
|
||||||
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
||||||
current_bytes: 0,
|
current_bytes: 0,
|
||||||
@@ -366,340 +540,345 @@ impl ChunkCache {
|
|||||||
tick: 0,
|
tick: 0,
|
||||||
last_coord: None,
|
last_coord: None,
|
||||||
stats: AccessStats::default(),
|
stats: AccessStats::default(),
|
||||||
chunk_index: None,
|
|
||||||
chunk_layout: None,
|
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Index operations -----
|
fn lock(&self) -> std::sync::MutexGuard<'_, CacheInner> {
|
||||||
|
self.inner.lock().unwrap_or_else(|e| e.into_inner())
|
||||||
|
}
|
||||||
|
|
||||||
/// The most decompressed bytes this cache will hold.
|
/// The most decompressed bytes this cache will hold.
|
||||||
pub fn max_bytes(&self) -> usize {
|
pub fn max_bytes(&self) -> usize {
|
||||||
self.inner.lock().map(|g| g.max_bytes).unwrap_or(0)
|
self.lock().max_bytes
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Bind the cache to the dataset at chunk-index address `addr`.
|
// ----- Dataset-keyed operations (safe to use concurrently) -----
|
||||||
|
|
||||||
|
/// The chunk list of the dataset whose chunk index is at `addr`.
|
||||||
///
|
///
|
||||||
/// The cache is shared per file across all of its datasets. If the cache
|
/// On the first call for a dataset, `build` scans its chunk index; the
|
||||||
/// currently holds state for a different dataset, all per-dataset state
|
/// result is kept (offsets truncated to `rank` for the lookup key), so
|
||||||
/// (chunk index, chunk-index map, layout, and decompressed slots) is
|
/// later calls skip the scan. `build` runs without the cache lock held;
|
||||||
/// dropped so the next access rebuilds it for this dataset. Reading the
|
/// if two threads race to build the same dataset's index, the first
|
||||||
/// same dataset again is a no-op, preserving the cache's benefit for
|
/// stored one wins and both return equivalent lists.
|
||||||
/// repeated/sequential access. Returns `true` if a reset occurred.
|
pub fn chunks_for<E>(
|
||||||
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
&self,
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
addr: u64,
|
||||||
if inner.index_addr == Some(addr) {
|
rank: usize,
|
||||||
return false;
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
) -> Result<Vec<ChunkInfo>, E> {
|
||||||
|
if let Some(ordered) = self.lock().touch(addr).ordered.clone() {
|
||||||
|
return Ok(ordered.as_ref().clone());
|
||||||
}
|
}
|
||||||
inner.index = None;
|
let chunks = build()?;
|
||||||
inner.chunk_index = None;
|
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
|
||||||
inner.chunk_layout = None;
|
.iter()
|
||||||
inner.slots.clear();
|
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci.clone()))
|
||||||
inner.slot_index.clear();
|
.collect();
|
||||||
inner.current_bytes = 0;
|
let mut inner = self.lock();
|
||||||
inner.last_coord = None;
|
let entry = inner.touch(addr);
|
||||||
inner.index_addr = Some(addr);
|
// Another thread may have built this dataset's index meanwhile: keep
|
||||||
true
|
// the first one, so every reader sees the same order.
|
||||||
|
let ordered = Arc::clone(entry.ordered.get_or_insert_with(|| Arc::new(chunks)));
|
||||||
|
entry.index.get_or_insert_with(|| Arc::new(map));
|
||||||
|
inner.trim_datasets(addr);
|
||||||
|
Ok(ordered.as_ref().clone())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns `true` if the chunk index has been built.
|
fn index_for<E>(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
rank: usize,
|
||||||
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
) -> Result<Arc<HashMap<ChunkCoord, ChunkInfo>>, E> {
|
||||||
|
if let Some(index) = self.lock().touch(addr).index.clone() {
|
||||||
|
return Ok(index);
|
||||||
|
}
|
||||||
|
let chunks = build()?;
|
||||||
|
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
|
||||||
|
.iter()
|
||||||
|
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci.clone()))
|
||||||
|
.collect();
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
if entry.index.is_none() {
|
||||||
|
entry.ordered = Some(Arc::new(chunks));
|
||||||
|
}
|
||||||
|
let index = Arc::clone(entry.index.get_or_insert_with(|| Arc::new(map)));
|
||||||
|
inner.trim_datasets(addr);
|
||||||
|
Ok(index)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The pre-computed assembly layout of the dataset at `addr`, building
|
||||||
|
/// its chunk index (via `build`, as in [`Self::chunks_for`]) and layout on
|
||||||
|
/// first use.
|
||||||
|
pub fn chunk_layout_for<E>(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
rank: usize,
|
||||||
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
ds_dims: &[usize],
|
||||||
|
chunk_dims: &[usize],
|
||||||
|
elem_size: usize,
|
||||||
|
) -> Result<Arc<ChunkLayout>, E> {
|
||||||
|
let (layout, chunk_index) = {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
(entry.chunk_layout.clone(), entry.chunk_index.clone())
|
||||||
|
};
|
||||||
|
if let Some(layout) = layout {
|
||||||
|
return Ok(layout);
|
||||||
|
}
|
||||||
|
let chunk_index = match chunk_index {
|
||||||
|
Some(ci) => ci,
|
||||||
|
None => {
|
||||||
|
let index = self.index_for(addr, rank, build)?;
|
||||||
|
let chunks: Vec<ChunkInfo> = index.values().cloned().collect();
|
||||||
|
Arc::new(ChunkIndex::build(&chunks, rank))
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let layout = ChunkLayout::build(&chunk_index, ds_dims, chunk_dims, elem_size);
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
entry.chunk_index.get_or_insert(chunk_index);
|
||||||
|
let layout = Arc::clone(entry.chunk_layout.get_or_insert_with(|| Arc::new(layout)));
|
||||||
|
inner.trim_datasets(addr);
|
||||||
|
Ok(layout)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cached decompressed chunk at `coord` of the dataset at `addr`.
|
||||||
|
///
|
||||||
|
/// O(1) lookup; the clone is an `Arc` refcount bump, not a copy of the
|
||||||
|
/// underlying decompressed data.
|
||||||
|
pub fn get_decompressed_in(&self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
|
self.lock().get_decompressed(addr, coord)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cache decompressed chunk data for `coord` of the dataset at `addr`.
|
||||||
|
/// Returns the `Arc`-shared buffer now cached (or already cached).
|
||||||
|
pub fn put_decompressed_in(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
coord: ChunkCoord,
|
||||||
|
data: Vec<u8>,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
self.put_decompressed_aligned_in(addr, coord, CacheAlignedBuffer::from_vec(data))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::put_decompressed_in`] for an already-aligned buffer.
|
||||||
|
pub fn put_decompressed_aligned_in(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
coord: ChunkCoord,
|
||||||
|
data: CacheAlignedBuffer,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
let data = Arc::new(data);
|
||||||
|
self.lock().put_decompressed((addr, coord), data)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record that the given chunk coordinates of the dataset at `addr` are
|
||||||
|
/// predicted to be accessed soon (bookkeeping only).
|
||||||
|
///
|
||||||
|
/// This does **not** prefetch or pre-decompress anything — it only
|
||||||
|
/// checks whether each coordinate is already in the chunk index and
|
||||||
|
/// updates access-pattern stats accordingly.
|
||||||
|
pub fn prefetch_hint_in(&self, addr: u64, next_coords: &[ChunkCoord]) {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let Some(index) = inner.entry(addr).and_then(|e| e.index.clone()) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let known = next_coords
|
||||||
|
.iter()
|
||||||
|
.filter(|c| index.contains_key(*c))
|
||||||
|
.count();
|
||||||
|
inner.stats.sequential_count += known as u64;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- Address-less operations on the bound dataset -----
|
||||||
|
|
||||||
|
/// Bind the address-less methods to the dataset at chunk-index address
|
||||||
|
/// `addr`. Returns `true` if this changed the bound dataset.
|
||||||
|
///
|
||||||
|
/// Each dataset's state is kept separately, so switching loses nothing
|
||||||
|
/// and never exposes one dataset's index or chunks to another. The
|
||||||
|
/// binding itself is shared, though: concurrent readers should use the
|
||||||
|
/// `addr`-taking methods rather than bind and then call these.
|
||||||
|
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let changed = inner.current != Some(addr);
|
||||||
|
inner.current = Some(addr);
|
||||||
|
changed
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns `true` if the bound dataset's chunk index has been built.
|
||||||
pub fn has_index(&self) -> bool {
|
pub fn has_index(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.index
|
.is_some_and(|e| e.index.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the chunk index from a pre-collected list of `ChunkInfo`.
|
/// Build the bound dataset's chunk index from a pre-collected list of
|
||||||
|
/// `ChunkInfo`.
|
||||||
///
|
///
|
||||||
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
||||||
/// (B-tree v1 stores rank+1 offsets).
|
/// (B-tree v1 stores rank+1 offsets).
|
||||||
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let addr = self.lock().current();
|
||||||
if inner.index.is_some() {
|
let _ = self.index_for::<core::convert::Infallible>(addr, rank, || Ok(chunks.to_vec()));
|
||||||
return; // already populated
|
|
||||||
}
|
|
||||||
let mut map = HashMap::with_capacity(chunks.len());
|
|
||||||
|
|
||||||
for ci in chunks {
|
|
||||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
|
||||||
map.insert(coord, ci.clone());
|
|
||||||
}
|
|
||||||
inner.index = Some(map);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Look up a chunk by its spatial coordinate in the index.
|
/// Look up a chunk by its spatial coordinate in the bound dataset's index.
|
||||||
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let inner = self.lock();
|
||||||
inner.index.as_ref()?.get(coord).cloned()
|
inner
|
||||||
|
.entry(inner.current())?
|
||||||
|
.index
|
||||||
|
.as_ref()?
|
||||||
|
.get(coord)
|
||||||
|
.cloned()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return all indexed chunks as a `Vec<ChunkInfo>` (order unspecified).
|
/// Return all of the bound dataset's indexed chunks (order unspecified).
|
||||||
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let inner = self.lock();
|
||||||
inner.index.as_ref().map(|m| m.values().cloned().collect())
|
let index = inner.entry(inner.current())?.index.as_ref()?;
|
||||||
|
Some(index.values().cloned().collect())
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Chunk index (pre-built coordinate → ChunkInfo map) -----
|
/// Returns `true` if the bound dataset's `ChunkIndex` has been built.
|
||||||
|
|
||||||
/// Returns `true` if the chunk B-tree index has been built.
|
|
||||||
pub fn has_chunk_index(&self) -> bool {
|
pub fn has_chunk_index(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.chunk_index
|
.is_some_and(|e| e.chunk_index.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build and store the chunk B-tree index from a pre-collected list of `ChunkInfo`.
|
/// Build and store the bound dataset's `ChunkIndex`.
|
||||||
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let built = Arc::new(ChunkIndex::build(chunks, rank));
|
||||||
if inner.chunk_index.is_some() {
|
let mut inner = self.lock();
|
||||||
return;
|
let addr = inner.current();
|
||||||
}
|
inner.touch(addr).chunk_index.get_or_insert(built);
|
||||||
inner.chunk_index = Some(ChunkIndex::build(chunks, rank));
|
inner.trim_datasets(addr);
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Chunk layout (pre-computed assembly plan) -----
|
/// Returns `true` if the bound dataset's chunk layout has been computed.
|
||||||
|
|
||||||
/// Returns `true` if the chunk layout has been computed.
|
|
||||||
pub fn has_chunk_layout(&self) -> bool {
|
pub fn has_chunk_layout(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.chunk_layout
|
.is_some_and(|e| e.chunk_layout.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build and store the pre-computed chunk layout for fast assembly.
|
/// Build and store the bound dataset's chunk layout (needs its
|
||||||
|
/// `ChunkIndex`; does nothing without one).
|
||||||
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
if inner.chunk_layout.is_some() {
|
let addr = inner.current();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
if entry.chunk_layout.is_some() {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if let Some(ref idx) = inner.chunk_index {
|
if let Some(idx) = entry.chunk_index.clone() {
|
||||||
inner.chunk_layout = Some(ChunkLayout::build(idx, ds_dims, chunk_dims, elem_size));
|
entry.chunk_layout = Some(Arc::new(ChunkLayout::build(
|
||||||
|
&idx, ds_dims, chunk_dims, elem_size,
|
||||||
|
)));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Execute a function with a reference to the chunk layout.
|
/// Execute a function with a reference to the bound dataset's chunk
|
||||||
///
|
/// layout. Returns `None` if the layout hasn't been computed yet.
|
||||||
/// Returns `None` if the layout hasn't been computed yet.
|
|
||||||
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
||||||
where
|
where
|
||||||
F: FnOnce(&ChunkLayout) -> R,
|
F: FnOnce(&ChunkLayout) -> R,
|
||||||
{
|
{
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let layout = {
|
||||||
inner.chunk_layout.as_ref().map(f)
|
let inner = self.lock();
|
||||||
|
inner.entry(inner.current())?.chunk_layout.clone()?
|
||||||
|
};
|
||||||
|
Some(f(&layout))
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Decompressed data cache (LRU) -----
|
/// Try to get cached decompressed data for a chunk of the bound dataset.
|
||||||
|
|
||||||
/// Try to get cached decompressed data for a chunk coordinate.
|
|
||||||
///
|
///
|
||||||
/// O(1) lookup. Returns an owned copy for API compatibility with callers
|
/// Returns an owned copy; prefer [`Self::get_decompressed_aligned`] when
|
||||||
/// that need a `Vec<u8>`; prefer [`Self::get_decompressed_aligned`] when
|
/// an `Arc`-shared buffer works for the caller.
|
||||||
/// an `Arc`-shared buffer works for the caller, since that avoids the
|
|
||||||
/// copy entirely.
|
|
||||||
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
||||||
self.get_decompressed_aligned(coord)
|
self.get_decompressed_aligned(coord)
|
||||||
.map(|arc| arc.as_slice().to_vec())
|
.map(|arc| arc.as_slice().to_vec())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Try to get a reference-counted clone of the aligned buffer for a chunk.
|
/// Reference-counted cached buffer for a chunk of the bound dataset.
|
||||||
///
|
|
||||||
/// O(1) index lookup; the clone is an `Arc` refcount bump, not a copy of
|
|
||||||
/// the underlying decompressed data.
|
|
||||||
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
inner.tick += 1;
|
let addr = inner.current();
|
||||||
let tick = inner.tick;
|
inner.get_decompressed(addr, coord)
|
||||||
|
|
||||||
// Track sequential vs random access
|
|
||||||
let is_sequential = inner.last_coord.as_ref().is_some_and(|prev| {
|
|
||||||
// Sequential if exactly one dimension changed
|
|
||||||
let changes: usize = prev
|
|
||||||
.iter()
|
|
||||||
.zip(coord.iter())
|
|
||||||
.filter(|(a, b)| a != b)
|
|
||||||
.count();
|
|
||||||
changes <= 1
|
|
||||||
});
|
|
||||||
if is_sequential {
|
|
||||||
inner.stats.sequential_count += 1;
|
|
||||||
} else if inner.last_coord.is_some() {
|
|
||||||
inner.stats.random_count += 1;
|
|
||||||
}
|
|
||||||
inner.last_coord = Some(coord.to_vec());
|
|
||||||
|
|
||||||
let found = if let Some(&idx) = inner.slot_index.get(coord) {
|
|
||||||
inner.slots[idx].last_access = tick;
|
|
||||||
Some(Arc::clone(&inner.slots[idx].data))
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
};
|
|
||||||
if let Some(ref data) = found {
|
|
||||||
inner.stats.hits += 1;
|
|
||||||
inner.stats.bytes_read += data.len() as u64;
|
|
||||||
} else {
|
|
||||||
inner.stats.misses += 1;
|
|
||||||
}
|
|
||||||
found
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert decompressed chunk data into the LRU cache.
|
/// Insert decompressed chunk data for the bound dataset into the LRU
|
||||||
///
|
/// cache, returning the `Arc`-shared buffer now cached.
|
||||||
/// The data is stored in a [`CacheAlignedBuffer`] so subsequent reads
|
|
||||||
/// return cache-line-aligned memory. Returns the `Arc`-shared buffer that
|
|
||||||
/// is now cached (or already was), so the caller can reuse it directly
|
|
||||||
/// instead of holding a separate copy of the same data.
|
|
||||||
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
||||||
let aligned = CacheAlignedBuffer::from_vec(data);
|
self.put_decompressed_aligned(coord, CacheAlignedBuffer::from_vec(data))
|
||||||
self.put_decompressed_aligned(coord, aligned)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert an already-aligned buffer into the LRU cache.
|
/// Insert an already-aligned buffer for the bound dataset.
|
||||||
///
|
|
||||||
/// Returns the `Arc`-shared buffer now held by the cache (the one just
|
|
||||||
/// inserted, or the existing cached copy if `coord` was already present).
|
|
||||||
pub fn put_decompressed_aligned(
|
pub fn put_decompressed_aligned(
|
||||||
&self,
|
&self,
|
||||||
coord: ChunkCoord,
|
coord: ChunkCoord,
|
||||||
data: CacheAlignedBuffer,
|
data: CacheAlignedBuffer,
|
||||||
) -> Arc<CacheAlignedBuffer> {
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
let data = Arc::new(data);
|
let data = Arc::new(data);
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
let data_len = data.len();
|
let addr = inner.current();
|
||||||
|
inner.put_decompressed((addr, coord), data)
|
||||||
// Don't cache if single chunk exceeds budget — still return the data
|
|
||||||
// to the caller, just don't retain it.
|
|
||||||
if data_len > inner.max_bytes {
|
|
||||||
return data;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check if already present
|
|
||||||
inner.tick += 1;
|
|
||||||
let tick = inner.tick;
|
|
||||||
if let Some(&idx) = inner.slot_index.get(&coord) {
|
|
||||||
inner.slots[idx].last_access = tick;
|
|
||||||
return Arc::clone(&inner.slots[idx].data); // already cached
|
|
||||||
}
|
|
||||||
|
|
||||||
// Evict until we have room
|
|
||||||
while inner.slots.len() >= inner.max_slots
|
|
||||||
|| (inner.current_bytes + data_len > inner.max_bytes && !inner.slots.is_empty())
|
|
||||||
{
|
|
||||||
// Find LRU slot
|
|
||||||
let lru_idx = inner
|
|
||||||
.slots
|
|
||||||
.iter()
|
|
||||||
.enumerate()
|
|
||||||
.min_by_key(|(_, s)| s.last_access)
|
|
||||||
.map(|(i, _)| i)
|
|
||||||
.unwrap();
|
|
||||||
let removed = inner.slots.swap_remove(lru_idx);
|
|
||||||
inner.slot_index.remove(&removed.coord);
|
|
||||||
// swap_remove moved the former last element into `lru_idx` (unless
|
|
||||||
// it *was* the last element) — fix up that element's index entry.
|
|
||||||
if lru_idx < inner.slots.len() {
|
|
||||||
let moved_coord = inner.slots[lru_idx].coord.clone();
|
|
||||||
inner.slot_index.insert(moved_coord, lru_idx);
|
|
||||||
}
|
|
||||||
inner.current_bytes -= removed.data.len();
|
|
||||||
inner.stats.evictions += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
inner.current_bytes += data_len;
|
|
||||||
let new_idx = inner.slots.len();
|
|
||||||
inner.slot_index.insert(coord.clone(), new_idx);
|
|
||||||
inner.slots.push(CachedChunk {
|
|
||||||
coord,
|
|
||||||
data: Arc::clone(&data),
|
|
||||||
last_access: tick,
|
|
||||||
});
|
|
||||||
data
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Clear the entire cache (index + decompressed data).
|
/// [`Self::prefetch_hint_in`] for the bound dataset.
|
||||||
|
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
||||||
|
let addr = self.lock().current();
|
||||||
|
self.prefetch_hint_in(addr, next_coords);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- Whole-cache operations -----
|
||||||
|
|
||||||
|
/// Clear the entire cache (indexes + decompressed data + stats).
|
||||||
pub fn clear(&self) {
|
pub fn clear(&self) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
inner.index = None;
|
inner.datasets.clear();
|
||||||
inner.index_addr = None;
|
inner.current = None;
|
||||||
inner.slots.clear();
|
inner.slots.clear();
|
||||||
inner.slot_index.clear();
|
inner.slot_index.clear();
|
||||||
inner.current_bytes = 0;
|
inner.current_bytes = 0;
|
||||||
inner.tick = 0;
|
inner.tick = 0;
|
||||||
inner.last_coord = None;
|
inner.last_coord = None;
|
||||||
inner.stats = AccessStats::default();
|
inner.stats = AccessStats::default();
|
||||||
inner.chunk_index = None;
|
|
||||||
inner.chunk_layout = None;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Record that the given chunk coordinates are predicted to be accessed
|
|
||||||
/// soon (bookkeeping only).
|
|
||||||
///
|
|
||||||
/// This does **not** prefetch or pre-decompress anything — it only
|
|
||||||
/// checks whether each coordinate is already in the chunk index and
|
|
||||||
/// updates access-pattern stats accordingly. Real prefetching (e.g.
|
|
||||||
/// background pre-decompression) is not implemented.
|
|
||||||
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
|
||||||
if inner.index.is_none() {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
drop(inner);
|
|
||||||
// For each predicted coordinate, verify it exists in the index.
|
|
||||||
// The index is already populated, so this is a no-op for known chunks.
|
|
||||||
// The purpose is to signal intent — callers can pre-decompress if needed.
|
|
||||||
// We touch the stats to record that prefetch hints were issued.
|
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
|
||||||
for coord in next_coords {
|
|
||||||
let exists = inner
|
|
||||||
.index
|
|
||||||
.as_ref()
|
|
||||||
.map(|idx| idx.contains_key(coord))
|
|
||||||
.unwrap_or(false);
|
|
||||||
if exists {
|
|
||||||
inner.stats.sequential_count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return the current access pattern statistics.
|
/// Return the current access pattern statistics.
|
||||||
pub fn access_stats(&self) -> AccessStats {
|
pub fn access_stats(&self) -> AccessStats {
|
||||||
self.inner
|
self.lock().stats.clone()
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.stats
|
|
||||||
.clone()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Update the sweep direction label in the access stats.
|
/// Update the sweep direction label in the access stats.
|
||||||
pub fn set_sweep_direction(&self, direction: &'static str) {
|
pub fn set_sweep_direction(&self, direction: &'static str) {
|
||||||
self.inner
|
self.lock().stats.sweep_direction = Some(direction);
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.stats
|
|
||||||
.sweep_direction = Some(direction);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Number of decompressed chunks currently cached.
|
/// Number of decompressed chunks currently cached (all datasets).
|
||||||
pub fn cached_chunk_count(&self) -> usize {
|
pub fn cached_chunk_count(&self) -> usize {
|
||||||
self.inner
|
self.lock().slots.len()
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.slots
|
|
||||||
.len()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Total bytes of decompressed data currently cached.
|
/// Total bytes of decompressed data currently cached (all datasets).
|
||||||
pub fn cached_bytes(&self) -> usize {
|
pub fn cached_bytes(&self) -> usize {
|
||||||
self.inner
|
self.lock().current_bytes
|
||||||
.lock()
|
}
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.current_bytes
|
/// Number of datasets whose chunk index is currently kept.
|
||||||
|
pub fn indexed_dataset_count(&self) -> usize {
|
||||||
|
self.lock().datasets.len()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -808,6 +987,92 @@ mod tests {
|
|||||||
assert_eq!(cache.cached_bytes(), 0);
|
assert_eq!(cache.cached_bytes(), 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn datasets_sharing_coordinates_stay_separate() {
|
||||||
|
let cache = ChunkCache::new();
|
||||||
|
let a = vec![make_chunk(vec![0, 0], 0x100, 8)];
|
||||||
|
let b = vec![make_chunk(vec![0, 0], 0x900, 8)];
|
||||||
|
let got_a = cache.chunks_for::<()>(1, 1, || Ok(a.clone())).unwrap();
|
||||||
|
let got_b = cache.chunks_for::<()>(2, 1, || Ok(b.clone())).unwrap();
|
||||||
|
assert_eq!(got_a[0].address, 0x100);
|
||||||
|
assert_eq!(got_b[0].address, 0x900);
|
||||||
|
// Built once per dataset: a second lookup doesn't call the builder.
|
||||||
|
let again = cache
|
||||||
|
.chunks_for::<()>(1, 1, || panic!("index rebuilt"))
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(again[0].address, 0x100);
|
||||||
|
|
||||||
|
cache.put_decompressed_in(1, vec![0], vec![1; 4]);
|
||||||
|
cache.put_decompressed_in(2, vec![0], vec![2; 4]);
|
||||||
|
assert_eq!(
|
||||||
|
cache.get_decompressed_in(1, &[0]).unwrap().as_slice(),
|
||||||
|
&[1; 4]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
cache.get_decompressed_in(2, &[0]).unwrap().as_slice(),
|
||||||
|
&[2; 4]
|
||||||
|
);
|
||||||
|
assert!(cache.get_decompressed_in(3, &[0]).is_none());
|
||||||
|
assert_eq!(cache.cached_chunk_count(), 2);
|
||||||
|
|
||||||
|
// The bound-dataset methods see only the bound dataset.
|
||||||
|
cache.ensure_dataset(2);
|
||||||
|
assert_eq!(cache.lookup_index(&[0]).unwrap().address, 0x900);
|
||||||
|
assert_eq!(cache.get_decompressed(&[0]).unwrap(), vec![2; 4]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dataset_indexes_are_bounded() {
|
||||||
|
let cache = ChunkCache::new();
|
||||||
|
for addr in 0..(MAX_INDEXED_DATASETS as u64 + 10) {
|
||||||
|
cache
|
||||||
|
.chunks_for::<()>(addr, 1, || Ok(vec![make_chunk(vec![0], addr, 8)]))
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), MAX_INDEXED_DATASETS);
|
||||||
|
|
||||||
|
// One huge index evicts the others but is itself kept.
|
||||||
|
let huge: Vec<ChunkInfo> = (0..MAX_INDEXED_CHUNKS as u64)
|
||||||
|
.map(|i| make_chunk(vec![i], i, 8))
|
||||||
|
.collect();
|
||||||
|
let got = cache.chunks_for::<()>(9999, 1, || Ok(huge)).unwrap();
|
||||||
|
assert_eq!(got.len(), MAX_INDEXED_CHUNKS);
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn concurrent_readers_of_different_datasets_see_their_own_chunks() {
|
||||||
|
let cache = std::sync::Arc::new(ChunkCache::with_capacity(1 << 20, 64));
|
||||||
|
let handles: Vec<_> = (0..8u64)
|
||||||
|
.map(|t| {
|
||||||
|
let cache = std::sync::Arc::clone(&cache);
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
for round in 0..500u64 {
|
||||||
|
let addr = (t + round) % 16;
|
||||||
|
let coord = vec![round % 4];
|
||||||
|
let chunks = cache
|
||||||
|
.chunks_for::<()>(addr, 1, || {
|
||||||
|
Ok((0..4).map(|c| make_chunk(vec![c], addr, 8)).collect())
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
assert!(chunks.iter().all(|c| c.address == addr));
|
||||||
|
let want = vec![addr as u8; 8];
|
||||||
|
let got = match cache.get_decompressed_in(addr, &coord) {
|
||||||
|
Some(hit) => hit.to_vec(),
|
||||||
|
None => cache
|
||||||
|
.put_decompressed_in(addr, coord, want.clone())
|
||||||
|
.to_vec(),
|
||||||
|
};
|
||||||
|
assert_eq!(got, want);
|
||||||
|
}
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
for h in handles {
|
||||||
|
h.join().unwrap();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn duplicate_insert_is_noop() {
|
fn duplicate_insert_is_noop() {
|
||||||
let cache = ChunkCache::new();
|
let cache = ChunkCache::new();
|
||||||
|
|||||||
@@ -0,0 +1,228 @@
|
|||||||
|
//! Chunk-index linearisation shared by the Fixed Array and Extensible Array
|
||||||
|
//! chunk indexes (reader and writer).
|
||||||
|
//!
|
||||||
|
//! Both indexes store one element per chunk at a *linear* index, and the
|
||||||
|
//! library derives that index from the chunk's scaled coordinates
|
||||||
|
//! (`offset / chunk_dim`) using the dataset's **maximum** dimensions, not its
|
||||||
|
//! current ones (`H5D__farray_idx_get_addr` / `H5D__earray_idx_get_addr`,
|
||||||
|
//! via `layout->max_down_chunks`). A dataset whose current shape is smaller
|
||||||
|
//! than its maxshape therefore has gaps in the index, and laying it out by the
|
||||||
|
//! current shape puts every chunk after the first row in the wrong place.
|
||||||
|
//!
|
||||||
|
//! The Extensible Array adds one more step: its one unlimited dimension has no
|
||||||
|
//! finite chunk count, so the library *swizzles* the coordinates to make that
|
||||||
|
//! dimension the slowest-varying one (`H5VM_swizzle_coords`, which moves
|
||||||
|
//! `coords[unlim_dim]` to the front and shifts the dimensions before it right
|
||||||
|
//! by one) before linearising with `swizzled_max_down_chunks`. When the
|
||||||
|
//! unlimited dimension is already dimension 0 no swizzle happens.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// How a chunk index maps linear element indexes to chunk coordinates.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub(crate) struct ChunkGrid {
|
||||||
|
/// Spatial chunk dimensions, in dataset order.
|
||||||
|
chunk_dims: Vec<u64>,
|
||||||
|
/// Chunks per dimension covering the *current* extent, in dataset order.
|
||||||
|
cur_chunks: Vec<u64>,
|
||||||
|
/// Dataset dimension stored at each linearisation position (slowest
|
||||||
|
/// first). The identity except for a swizzled Extensible Array.
|
||||||
|
order: Vec<usize>,
|
||||||
|
/// Linear stride of each linearisation position.
|
||||||
|
down: Vec<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ChunkGrid {
|
||||||
|
/// Grid for a Fixed Array index: row-major over the chunk counts of the
|
||||||
|
/// maximum dimensions (`max_dims`, falling back to the current dimensions
|
||||||
|
/// when the dataspace records none).
|
||||||
|
pub(crate) fn fixed_array(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
Self::build(cur_dims, max_dims, chunk_dims, None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Grid for an Extensible Array index: like the Fixed Array, but the
|
||||||
|
/// unlimited dimension (the one whose maximum is `H5S_UNLIMITED`) is moved
|
||||||
|
/// to the slowest-varying position first.
|
||||||
|
pub(crate) fn extensible_array(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let unlim = max_dims.and_then(|m| m.iter().position(|&d| d == u64::MAX));
|
||||||
|
Self::build(cur_dims, max_dims, chunk_dims, unlim)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
unlim: Option<usize>,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let rank = chunk_dims.len();
|
||||||
|
if cur_dims.len() != rank || max_dims.is_some_and(|m| m.len() != rank) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"chunk index rank does not match the dataspace".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if chunk_dims.contains(&0) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"chunk dimension is zero".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let cur_chunks: Vec<u64> = cur_dims
|
||||||
|
.iter()
|
||||||
|
.zip(chunk_dims)
|
||||||
|
.map(|(&d, &c)| d.div_ceil(c))
|
||||||
|
.collect();
|
||||||
|
// Chunk counts of the maximum extent. An unlimited dimension has no
|
||||||
|
// finite count; it only ever sits in the slowest position, where its
|
||||||
|
// count never enters a stride. A (corrupt) maximum smaller than the
|
||||||
|
// current extent is widened so no allocated chunk becomes unreachable.
|
||||||
|
let max_chunks: Vec<u64> = (0..rank)
|
||||||
|
.map(|d| {
|
||||||
|
let max = max_dims.map_or(cur_dims[d], |m| m[d]);
|
||||||
|
if max == u64::MAX {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
max.div_ceil(chunk_dims[d]).max(cur_chunks[d])
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let mut order: Vec<usize> = (0..rank).collect();
|
||||||
|
if let Some(u) = unlim {
|
||||||
|
order.remove(u);
|
||||||
|
order.insert(0, u);
|
||||||
|
}
|
||||||
|
let mut down = vec![1u64; rank];
|
||||||
|
for p in (0..rank.saturating_sub(1)).rev() {
|
||||||
|
let next = max_chunks[order[p + 1]];
|
||||||
|
if next == u64::MAX {
|
||||||
|
// Only reachable with more than one unlimited dimension, which
|
||||||
|
// neither index type can describe.
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"array chunk index with more than one unlimited dimension".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
down[p] = down[p + 1].checked_mul(next).ok_or_else(|| {
|
||||||
|
FormatError::Overflow("chunk index linear stride overflows u64".into())
|
||||||
|
})?;
|
||||||
|
}
|
||||||
|
Ok(Self {
|
||||||
|
chunk_dims: chunk_dims.to_vec(),
|
||||||
|
cur_chunks,
|
||||||
|
order,
|
||||||
|
down,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dataset-space offsets of the chunk stored at linear `index`, or `None`
|
||||||
|
/// when that chunk lies outside the current extent (the index still has a
|
||||||
|
/// slot for it; the library ignores such chunks on read).
|
||||||
|
pub(crate) fn offsets(&self, index: u64) -> Option<Vec<u64>> {
|
||||||
|
let rank = self.chunk_dims.len();
|
||||||
|
let mut offsets = vec![0u64; rank];
|
||||||
|
let mut rem = index;
|
||||||
|
for p in 0..rank {
|
||||||
|
let d = self.order[p];
|
||||||
|
// A zero stride: a later dimension has no chunks (its maximum,
|
||||||
|
// or with none recorded its current extent, is 0), so no slot of
|
||||||
|
// the index is a chunk of the dataset.
|
||||||
|
if self.down[p] == 0 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let scaled = rem / self.down[p];
|
||||||
|
rem %= self.down[p];
|
||||||
|
if scaled >= self.cur_chunks[d] {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
offsets[d] = scaled * self.chunk_dims[d];
|
||||||
|
}
|
||||||
|
Some(offsets)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Linear index of the chunk with scaled coordinates `scaled`
|
||||||
|
/// (`offset / chunk_dim` per dimension, in dataset order).
|
||||||
|
pub(crate) fn linear_index(&self, scaled: &[u64]) -> u64 {
|
||||||
|
self.order
|
||||||
|
.iter()
|
||||||
|
.zip(&self.down)
|
||||||
|
.map(|(&d, &stride)| scaled[d] * stride)
|
||||||
|
.sum()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn fixed_array_uses_max_dims() {
|
||||||
|
// shape (4, 6), chunks (2, 3), maxshape (20, 10): 10 x 4 chunk grid.
|
||||||
|
let g = ChunkGrid::fixed_array(&[4, 6], Some(&[20, 10]), &[2, 3]).unwrap();
|
||||||
|
assert_eq!(g.offsets(0), Some(vec![0, 0]));
|
||||||
|
assert_eq!(g.offsets(1), Some(vec![0, 3]));
|
||||||
|
assert_eq!(g.offsets(2), None); // column chunk 2 is beyond the extent
|
||||||
|
assert_eq!(g.offsets(4), Some(vec![2, 0]));
|
||||||
|
assert_eq!(g.offsets(5), Some(vec![2, 3]));
|
||||||
|
assert_eq!(g.offsets(8), None); // row chunk 2 is beyond the extent
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 5);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extensible_array_swizzles_unlimited_dim() {
|
||||||
|
// maxshape (10, None): dim 1 is unlimited and becomes slowest.
|
||||||
|
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[10, u64::MAX]), &[2, 3]).unwrap();
|
||||||
|
// max chunks of dim 0 = 5, so index = c1 * 5 + c0.
|
||||||
|
assert_eq!(g.linear_index(&[1, 0]), 1);
|
||||||
|
assert_eq!(g.linear_index(&[0, 1]), 5);
|
||||||
|
assert_eq!(g.offsets(5), Some(vec![0, 3]));
|
||||||
|
assert_eq!(g.offsets(6), Some(vec![2, 3]));
|
||||||
|
assert_eq!(g.offsets(2), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extensible_array_unlimited_first_is_row_major() {
|
||||||
|
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[u64::MAX, 30]), &[2, 3]).unwrap();
|
||||||
|
// max chunks of dim 1 = 10.
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 11);
|
||||||
|
assert_eq!(g.offsets(11), Some(vec![2, 3]));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn zero_extent_has_no_chunks() {
|
||||||
|
// No maximum recorded and a zero current dimension: every stride
|
||||||
|
// before it is 0 (this divided by zero).
|
||||||
|
let g = ChunkGrid::fixed_array(&[1, 0], None, &[6, 6]).unwrap();
|
||||||
|
for i in 0..16 {
|
||||||
|
assert_eq!(g.offsets(i), None);
|
||||||
|
}
|
||||||
|
let g = ChunkGrid::fixed_array(&[0, 0, 3], Some(&[4, 0, 3]), &[2, 2, 3]).unwrap();
|
||||||
|
for i in 0..16 {
|
||||||
|
assert_eq!(g.offsets(i), None);
|
||||||
|
}
|
||||||
|
let g = ChunkGrid::extensible_array(&[0, 5], Some(&[u64::MAX, 0]), &[2, 2]).unwrap();
|
||||||
|
for i in 0..16 {
|
||||||
|
assert_eq!(g.offsets(i), None);
|
||||||
|
}
|
||||||
|
// A zero last dimension leaves the other strides alone.
|
||||||
|
let g = ChunkGrid::fixed_array(&[4, 0], Some(&[4, 6]), &[2, 3]).unwrap();
|
||||||
|
assert_eq!(g.offsets(0), None);
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 3);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rejects_two_unlimited_dims_after_the_first() {
|
||||||
|
assert!(ChunkGrid::fixed_array(&[4, 6], Some(&[u64::MAX, u64::MAX]), &[2, 3]).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -18,6 +18,7 @@ use alloc::collections::BTreeMap;
|
|||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::chunk_cache::ChunkCoord;
|
use crate::chunk_cache::ChunkCoord;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
|
|
||||||
@@ -167,7 +168,15 @@ impl ChunkLayout {
|
|||||||
|
|
||||||
for (_coord, ci) in index.iter() {
|
for (_coord, ci) in index.iter() {
|
||||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
||||||
let chunk_offsets: Vec<usize> = coord.iter().map(|&o| o as usize).collect();
|
// `ds_dims` are `usize`: a chunk at an offset past `usize::MAX`
|
||||||
|
// (only on a 32-bit target) lies outside the dataset.
|
||||||
|
let Ok(chunk_offsets) = coord
|
||||||
|
.iter()
|
||||||
|
.map(|&o| to_usize(o))
|
||||||
|
.collect::<Result<Vec<usize>, _>>()
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
|
||||||
let copies = if rank == 0 {
|
let copies = if rank == 0 {
|
||||||
// Scalar dataset — single copy
|
// Scalar dataset — single copy
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,12 +1,14 @@
|
|||||||
//! HDF5 Data Layout message parsing (message type 0x0008).
|
//! HDF5 Data Layout message parsing (message type 0x0008).
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{string::String, vec::Vec};
|
use alloc::{format, string::String, vec::Vec};
|
||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::string::String;
|
use std::string::String;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::Storage;
|
||||||
|
|
||||||
/// A single VDS (Virtual Dataset) source mapping.
|
/// A single VDS (Virtual Dataset) source mapping.
|
||||||
///
|
///
|
||||||
@@ -24,6 +26,34 @@ pub struct VdsMapping {
|
|||||||
pub virtual_selection: Vec<u8>,
|
pub virtual_selection: Vec<u8>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Most dimensions a layout message can list (libhdf5 `H5O_LAYOUT_NDIMS`):
|
||||||
|
/// 32 dataspace dimensions plus the element size.
|
||||||
|
const MAX_LAYOUT_NDIMS: usize = 33;
|
||||||
|
|
||||||
|
/// libhdf5's checks on a chunked layout message's dimensions
|
||||||
|
/// (`H5O__layout_decode`): at most [`MAX_LAYOUT_NDIMS`], no dimension 0, and
|
||||||
|
/// before version 4 at least one dataspace dimension plus the element size.
|
||||||
|
/// A zero chunk dimension used to read the dataset as all fill values.
|
||||||
|
fn check_chunk_dims(dims: Vec<u32>, layout_version: u8) -> Result<Vec<u32>, FormatError> {
|
||||||
|
if dims.len() > MAX_LAYOUT_NDIMS {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"dimensionality is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if layout_version < 4 && dims.len() < 2 {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"bad dimensions for chunked storage".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if let Some(u) = dims.iter().position(|&d| d == 0) {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(format!(
|
||||||
|
"bad chunk dimension value when parsing layout message - chunk dimension must be \
|
||||||
|
positive: mesg->u.chunk.dim[{u}] = 0"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(dims)
|
||||||
|
}
|
||||||
|
|
||||||
/// Parsed HDF5 data layout message.
|
/// Parsed HDF5 data layout message.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
pub enum DataLayout {
|
pub enum DataLayout {
|
||||||
@@ -45,7 +75,9 @@ pub enum DataLayout {
|
|||||||
chunk_dimensions: Vec<u32>,
|
chunk_dimensions: Vec<u32>,
|
||||||
/// B-tree address, or `None` if undefined.
|
/// B-tree address, or `None` if undefined.
|
||||||
btree_address: Option<u64>,
|
btree_address: Option<u64>,
|
||||||
/// Layout version (3 or 4).
|
/// Layout version (3 or 4). Version 1/2 messages (HDF5 1.4/1.6-era)
|
||||||
|
/// use the same version-1 B-tree chunk index as version 3 and are
|
||||||
|
/// reported as 3.
|
||||||
version: u8,
|
version: u8,
|
||||||
/// Chunk index type (v4 only).
|
/// Chunk index type (v4 only).
|
||||||
chunk_index_type: Option<u8>,
|
chunk_index_type: Option<u8>,
|
||||||
@@ -53,6 +85,11 @@ pub enum DataLayout {
|
|||||||
single_chunk_filtered_size: Option<u64>,
|
single_chunk_filtered_size: Option<u64>,
|
||||||
/// Filter mask for v4 single chunk with filters.
|
/// Filter mask for v4 single chunk with filters.
|
||||||
single_chunk_filter_mask: Option<u32>,
|
single_chunk_filter_mask: Option<u32>,
|
||||||
|
/// Layout v4 flag bit 0 (`H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS`):
|
||||||
|
/// partial edge chunks — those extending past the dataset's current
|
||||||
|
/// extent in some dimension — are stored without the filter pipeline,
|
||||||
|
/// even though their filter mask is 0. Always `false` for v3.
|
||||||
|
dont_filter_partial_edge_chunks: bool,
|
||||||
},
|
},
|
||||||
/// Virtual dataset layout (v4 only).
|
/// Virtual dataset layout (v4 only).
|
||||||
Virtual {
|
Virtual {
|
||||||
@@ -67,21 +104,33 @@ pub enum DataLayout {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Version-1 VDS mapping flag: the source file name is stored by an earlier
|
||||||
|
/// entry, whose index follows in place of the name.
|
||||||
|
const VDS_SOURCE_FILE_SHARED: u8 = 0x01;
|
||||||
|
/// Version-1 VDS mapping flag: likewise for the source dataset name.
|
||||||
|
const VDS_SOURCE_DSET_SHARED: u8 = 0x02;
|
||||||
|
/// Version-1 VDS mapping flag: the source is in the virtual file itself
|
||||||
|
/// (`"."`); no file name is stored.
|
||||||
|
const VDS_SOURCE_SAME_FILE: u8 = 0x04;
|
||||||
|
const VDS_ALL_FLAGS: u8 = VDS_SOURCE_FILE_SHARED | VDS_SOURCE_DSET_SHARED | VDS_SOURCE_SAME_FILE;
|
||||||
|
|
||||||
/// Parse VDS mappings from global-heap object data.
|
/// Parse VDS mappings from global-heap object data.
|
||||||
///
|
///
|
||||||
/// The global-heap block holding a VDS mapping list is laid out as
|
/// The global-heap block holding a VDS mapping list is laid out as
|
||||||
/// (reverse-engineered and validated against HDF5 2.0):
|
/// (`H5D__virtual_store_layout` / `H5D__virtual_load_layout` in libhdf5):
|
||||||
///
|
///
|
||||||
/// ```text
|
/// ```text
|
||||||
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
||||||
/// ```
|
/// ```
|
||||||
///
|
///
|
||||||
/// Each entry is:
|
/// Each entry is:
|
||||||
/// - source file name — a null-terminated string in **block version 0**; in
|
/// - **block version 1 only:** a flags byte. `0x04`: the source is in the
|
||||||
/// **block version 1** a same-file reference is encoded as a single `0x04`
|
/// virtual file itself and no file name is stored; `0x01`/`0x02`: the
|
||||||
/// marker byte (the source file is the virtual file itself) in place of the
|
/// source file/dataset name is that of an earlier entry, whose index
|
||||||
/// name;
|
/// (`length_size` bytes) is stored instead of the name. libhdf5 2.0 writes
|
||||||
/// - source dataset name (null-terminated string);
|
/// version 1 when the file's low version bound is 2.0 and it saves space;
|
||||||
|
/// - source file name (null-terminated string, unless flagged above);
|
||||||
|
/// - source dataset name (null-terminated string, unless flagged above);
|
||||||
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
||||||
/// in length);
|
/// in length);
|
||||||
/// - virtual selection (serialized `H5S` dataspace selection).
|
/// - virtual selection (serialized `H5S` dataspace selection).
|
||||||
@@ -107,7 +156,7 @@ pub fn parse_vds_mappings(
|
|||||||
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
||||||
// least a few bytes, so the loop is naturally bounded by the heap data and
|
// least a few bytes, so the loop is naturally bounded by the heap data and
|
||||||
// a bogus `nused` simply errors out on the first short read.
|
// a bogus `nused` simply errors out on the first short read.
|
||||||
let mut mappings = Vec::new();
|
let mut mappings: Vec<VdsMapping> = Vec::new();
|
||||||
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
||||||
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
||||||
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
||||||
@@ -127,17 +176,57 @@ pub fn parse_vds_mappings(
|
|||||||
Ok(bytes)
|
Ok(bytes)
|
||||||
};
|
};
|
||||||
|
|
||||||
for _ in 0..nused {
|
if version > 1 {
|
||||||
// Source file name (with the version-1 same-file marker handled).
|
return Err(FormatError::ChunkedReadError(
|
||||||
let source_file = if version >= 1 && heap_data.get(pos) == Some(&0x04) {
|
"unsupported VDS mapping block version".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for i in 0..nused {
|
||||||
|
// Version 1 prefixes each entry with a flags byte; a name may then be
|
||||||
|
// omitted (same file) or replaced by the index of an earlier entry
|
||||||
|
// holding the same name (`H5D__virtual_load_layout`).
|
||||||
|
let flags = if version >= 1 {
|
||||||
|
let f = *heap_data.get(pos).ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: pos + 1,
|
||||||
|
available: heap_data.len(),
|
||||||
|
})?;
|
||||||
pos += 1;
|
pos += 1;
|
||||||
|
if f & !VDS_ALL_FLAGS != 0 {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"unknown VDS mapping flags".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
f
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
// Index of an earlier entry, for a shared name.
|
||||||
|
let earlier = |pos: &mut usize| -> Result<usize, FormatError> {
|
||||||
|
let idx = read_length(heap_data, *pos, length_size)?;
|
||||||
|
*pos += ls;
|
||||||
|
if idx >= i {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"VDS mapping shares a name with a later entry".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
to_usize(idx)
|
||||||
|
};
|
||||||
|
|
||||||
|
let source_file = if flags & VDS_SOURCE_SAME_FILE != 0 {
|
||||||
String::from(".")
|
String::from(".")
|
||||||
|
} else if flags & VDS_SOURCE_FILE_SHARED != 0 {
|
||||||
|
let idx = earlier(&mut pos)?;
|
||||||
|
mappings[idx].source_file.clone()
|
||||||
} else {
|
} else {
|
||||||
read_null_terminated_string(heap_data, &mut pos)?
|
read_null_terminated_string(heap_data, &mut pos)?
|
||||||
};
|
};
|
||||||
|
|
||||||
// Source dataset name.
|
let source_dataset = if flags & VDS_SOURCE_DSET_SHARED != 0 {
|
||||||
let source_dataset = read_null_terminated_string(heap_data, &mut pos)?;
|
let idx = earlier(&mut pos)?;
|
||||||
|
mappings[idx].source_dataset.clone()
|
||||||
|
} else {
|
||||||
|
read_null_terminated_string(heap_data, &mut pos)?
|
||||||
|
};
|
||||||
|
|
||||||
// Source selection, then virtual selection (both self-describing length).
|
// Source selection, then virtual selection (both self-describing length).
|
||||||
let source_selection = read_selection(heap_data, &mut pos)?;
|
let source_selection = read_selection(heap_data, &mut pos)?;
|
||||||
@@ -222,6 +311,16 @@ impl DataLayout {
|
|||||||
&mut self,
|
&mut self,
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
self.resolve_vds_mappings_in(file_data, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::resolve_vds_mappings`] over any [`Storage`]: one read of the
|
||||||
|
/// global heap collection holding the mappings.
|
||||||
|
pub fn resolve_vds_mappings_in<S: Storage + ?Sized>(
|
||||||
|
&mut self,
|
||||||
|
file_data: &S,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
if let DataLayout::Virtual {
|
if let DataLayout::Virtual {
|
||||||
global_heap_address,
|
global_heap_address,
|
||||||
@@ -231,11 +330,8 @@ impl DataLayout {
|
|||||||
} = self
|
} = self
|
||||||
&& let Some(addr) = *global_heap_address
|
&& let Some(addr) = *global_heap_address
|
||||||
{
|
{
|
||||||
let coll = crate::global_heap::GlobalHeapCollection::parse(
|
let coll =
|
||||||
file_data,
|
crate::global_heap::GlobalHeapCollection::parse_in(file_data, addr, length_size)?;
|
||||||
addr as usize,
|
|
||||||
length_size,
|
|
||||||
)?;
|
|
||||||
let obj = coll.get_object(*global_heap_index as u16).ok_or(
|
let obj = coll.get_object(*global_heap_index as u16).ok_or(
|
||||||
FormatError::GlobalHeapObjectNotFound {
|
FormatError::GlobalHeapObjectNotFound {
|
||||||
collection_address: addr,
|
collection_address: addr,
|
||||||
@@ -256,6 +352,7 @@ impl DataLayout {
|
|||||||
let layout_class = data[1];
|
let layout_class = data[1];
|
||||||
|
|
||||||
match version {
|
match version {
|
||||||
|
1 | 2 => Self::parse_v1_v2(data, offset_size),
|
||||||
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
||||||
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
||||||
// message structure as v4 — only the version number was bumped.
|
// message structure as v4 — only the version number was bumped.
|
||||||
@@ -264,6 +361,87 @@ impl DataLayout {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Layout message versions 1 and 2 (HDF5 before 1.6.3):
|
||||||
|
///
|
||||||
|
/// ```text
|
||||||
|
/// version(1) · dimensionality(1) · layout class(1) · reserved(5)
|
||||||
|
/// · address(offset_size) — contiguous and chunked only
|
||||||
|
/// · dimension sizes(4 × dimensionality)
|
||||||
|
/// · compact data size(4) · compact raw data — compact only
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// The dimension sizes are the dataset's (contiguous/compact) or the
|
||||||
|
/// chunk's (chunked) extent plus a trailing element-size dimension, as in
|
||||||
|
/// version 3's chunked form. libhdf5 ignores them for contiguous storage
|
||||||
|
/// and sizes the data from the dataspace; the product of the stored
|
||||||
|
/// dimensions is that same size, and a disagreement (a dimension that was
|
||||||
|
/// truncated to 32 bits) is caught by the reader's size check rather than
|
||||||
|
/// returning wrong data.
|
||||||
|
fn parse_v1_v2(data: &[u8], offset_size: u8) -> Result<DataLayout, FormatError> {
|
||||||
|
ensure_len(data, 0, 8)?;
|
||||||
|
let dimensionality = data[1] as usize;
|
||||||
|
let layout_class = data[2];
|
||||||
|
// H5O_LAYOUT_NDIMS: 32 dataspace dimensions + the element-size one.
|
||||||
|
if dimensionality > 33 {
|
||||||
|
return Err(FormatError::Overflow(format!(
|
||||||
|
"data layout dimensionality {dimensionality} exceeds 33"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let mut p = 8;
|
||||||
|
let os = offset_size as usize;
|
||||||
|
let address = match layout_class {
|
||||||
|
1 | 2 => {
|
||||||
|
ensure_len(data, p, os)?;
|
||||||
|
let a = if is_undefined(data, p, offset_size) {
|
||||||
|
None
|
||||||
|
} else {
|
||||||
|
Some(read_offset(data, p, offset_size)?)
|
||||||
|
};
|
||||||
|
p += os;
|
||||||
|
a
|
||||||
|
}
|
||||||
|
0 => None,
|
||||||
|
_ => return Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||||
|
};
|
||||||
|
ensure_len(data, p, dimensionality * 4)?;
|
||||||
|
let dims: Vec<u32> = data[p..p + dimensionality * 4]
|
||||||
|
.as_chunks::<4>()
|
||||||
|
.0
|
||||||
|
.iter()
|
||||||
|
.map(|c| u32::from_le_bytes(*c))
|
||||||
|
.collect();
|
||||||
|
p += dimensionality * 4;
|
||||||
|
match layout_class {
|
||||||
|
0 => {
|
||||||
|
ensure_len(data, p, 4)?;
|
||||||
|
let size =
|
||||||
|
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]) as usize;
|
||||||
|
ensure_len(data, p + 4, size)?;
|
||||||
|
Ok(DataLayout::Compact {
|
||||||
|
data: data[p + 4..p + 4 + size].to_vec(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
1 => {
|
||||||
|
let size = dims
|
||||||
|
.iter()
|
||||||
|
.try_fold(1u64, |acc, &d| acc.checked_mul(d as u64))
|
||||||
|
.ok_or_else(|| {
|
||||||
|
FormatError::Overflow(format!("contiguous layout size {dims:?}"))
|
||||||
|
})?;
|
||||||
|
Ok(DataLayout::Contiguous { address, size })
|
||||||
|
}
|
||||||
|
_ => Ok(DataLayout::Chunked {
|
||||||
|
chunk_dimensions: check_chunk_dims(dims, 2)?,
|
||||||
|
btree_address: address,
|
||||||
|
version: 3,
|
||||||
|
chunk_index_type: None,
|
||||||
|
single_chunk_filtered_size: None,
|
||||||
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn parse_v3(
|
fn parse_v3(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
layout_class: u8,
|
layout_class: u8,
|
||||||
@@ -316,12 +494,13 @@ impl DataLayout {
|
|||||||
p += 4;
|
p += 4;
|
||||||
}
|
}
|
||||||
Ok(DataLayout::Chunked {
|
Ok(DataLayout::Chunked {
|
||||||
chunk_dimensions,
|
chunk_dimensions: check_chunk_dims(chunk_dimensions, 3)?,
|
||||||
btree_address,
|
btree_address,
|
||||||
version: 3,
|
version: 3,
|
||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||||
@@ -364,47 +543,40 @@ impl DataLayout {
|
|||||||
let dimensionality = data[pos + 1] as usize;
|
let dimensionality = data[pos + 1] as usize;
|
||||||
let dim_size_encoded_length = data[pos + 2] as usize;
|
let dim_size_encoded_length = data[pos + 2] as usize;
|
||||||
let mut p = pos + 3;
|
let mut p = pos + 3;
|
||||||
|
if dimensionality > MAX_LAYOUT_NDIMS {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"dimensionality is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
// dimension sizes
|
// Each dimension takes 1 to 8 bytes (libhdf5 writes the
|
||||||
|
// fewest that hold the largest one, so 3, 5, 6 and 7 occur:
|
||||||
|
// a chunk dimension of 70 000 takes 3). libhdf5 refuses 0
|
||||||
|
// and more than 8.
|
||||||
|
if dim_size_encoded_length == 0 || dim_size_encoded_length > 8 {
|
||||||
|
return Err(FormatError::InvalidChunkDimensions(
|
||||||
|
"encoded chunk dimension size is too large".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
ensure_len(data, p, dimensionality * dim_size_encoded_length)?;
|
||||||
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
let mut chunk_dimensions = Vec::with_capacity(dimensionality);
|
||||||
for _ in 0..dimensionality {
|
for _ in 0..dimensionality {
|
||||||
let val = match dim_size_encoded_length {
|
let val = data[p..p + dim_size_encoded_length]
|
||||||
1 => data[p] as u32,
|
.iter()
|
||||||
2 => u16::from_le_bytes([data[p], data[p + 1]]) as u32,
|
.rev()
|
||||||
4 => u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]),
|
.fold(0u64, |acc, &b| (acc << 8) | u64::from(b));
|
||||||
8 => {
|
// Chunk dimensions are held as u32; HDF5 2.0 can write
|
||||||
// V4 chunked encodes dimension sizes as 8 bytes, but
|
// larger ones (layout version 5), which are refused
|
||||||
// our ChunkedStorageV4 stores them as u32. We read only
|
// rather than truncated.
|
||||||
// the low 4 bytes (little-endian). This silently
|
let val = u32::try_from(val).map_err(|_| {
|
||||||
// truncates dimensions > 4 GiB, which are not expected
|
FormatError::InvalidChunkDimensions(format!(
|
||||||
// in practice (HDF5 chunk dimensions are always small).
|
"chunk dimension {val} is larger than 2^32 - 1, which is not supported"
|
||||||
// If the high bytes are non-zero, the file is malformed
|
))
|
||||||
// or uses dimensions we cannot represent.
|
})?;
|
||||||
let high = u32::from_le_bytes([
|
|
||||||
data[p + 4],
|
|
||||||
data[p + 5],
|
|
||||||
data[p + 6],
|
|
||||||
data[p + 7],
|
|
||||||
]);
|
|
||||||
if high != 0 {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: p + 8,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]])
|
|
||||||
}
|
|
||||||
_ => {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: p + dim_size_encoded_length,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
};
|
|
||||||
chunk_dimensions.push(val);
|
chunk_dimensions.push(val);
|
||||||
p += dim_size_encoded_length;
|
p += dim_size_encoded_length;
|
||||||
}
|
}
|
||||||
|
let chunk_dimensions = check_chunk_dims(chunk_dimensions, 4)?;
|
||||||
|
|
||||||
// chunk index type
|
// chunk index type
|
||||||
ensure_len(data, p, 1)?;
|
ensure_len(data, p, 1)?;
|
||||||
@@ -505,6 +677,7 @@ impl DataLayout {
|
|||||||
chunk_index_type: Some(chunk_index_type),
|
chunk_index_type: Some(chunk_index_type),
|
||||||
single_chunk_filtered_size,
|
single_chunk_filtered_size,
|
||||||
single_chunk_filter_mask,
|
single_chunk_filter_mask,
|
||||||
|
dont_filter_partial_edge_chunks: flags & 0x01 != 0,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
3 => {
|
3 => {
|
||||||
@@ -539,6 +712,202 @@ impl DataLayout {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Version 1/2 header: version, dimensionality, class, reserved(5).
|
||||||
|
fn v1v2_header(version: u8, ndims: u8, class: u8) -> Vec<u8> {
|
||||||
|
vec![version, ndims, class, 0, 0, 0, 0, 0]
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v2_compact() {
|
||||||
|
let mut buf = v1v2_header(2, 2, 0);
|
||||||
|
// dims (3 elements of 2 bytes) — no address for compact
|
||||||
|
buf.extend_from_slice(&3u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&2u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&6u32.to_le_bytes()); // compact size (u32 in v1/v2)
|
||||||
|
buf.extend_from_slice(&[1, 0, 2, 0, 3, 0]);
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Compact {
|
||||||
|
data: vec![1, 0, 2, 0, 3, 0]
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_contiguous_size_from_dimensions() {
|
||||||
|
let mut buf = v1v2_header(1, 3, 1);
|
||||||
|
buf.extend_from_slice(&0x800u32.to_le_bytes()); // 4-byte address
|
||||||
|
for d in [10u32, 20, 4] {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 4, 4).unwrap(),
|
||||||
|
DataLayout::Contiguous {
|
||||||
|
address: Some(0x800),
|
||||||
|
size: 800,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_contiguous_undefined_address() {
|
||||||
|
let mut buf = v1v2_header(1, 2, 1);
|
||||||
|
buf.extend_from_slice(&[0xFF; 8]);
|
||||||
|
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Contiguous {
|
||||||
|
address: None,
|
||||||
|
size: 40,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_chunked_maps_to_btree_v1_index() {
|
||||||
|
let mut buf = v1v2_header(1, 3, 2);
|
||||||
|
buf.extend_from_slice(&0x1234u64.to_le_bytes());
|
||||||
|
for d in [50u32, 50, 4] {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Chunked {
|
||||||
|
chunk_dimensions: vec![50, 50, 4],
|
||||||
|
btree_address: Some(0x1234),
|
||||||
|
version: 3,
|
||||||
|
chunk_index_type: None,
|
||||||
|
single_chunk_filtered_size: None,
|
||||||
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A v3 chunked layout message with these dims (element size last).
|
||||||
|
fn v3_chunked_msg(dims: &[u32]) -> Vec<u8> {
|
||||||
|
let mut buf = vec![3u8, 2, dims.len() as u8];
|
||||||
|
buf.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
for d in dims {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn chunk_dimensions_are_checked_when_the_layout_is_parsed() {
|
||||||
|
assert!(DataLayout::parse(&v3_chunked_msg(&[4, 4, 8]), 8, 8).is_ok());
|
||||||
|
// A zero chunk dimension used to read as all fill values.
|
||||||
|
let err = DataLayout::parse(&v3_chunked_msg(&[4, 0, 8]), 8, 8).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(&err, FormatError::InvalidChunkDimensions(m) if m.contains("dim[1] = 0")),
|
||||||
|
"{err:?}"
|
||||||
|
);
|
||||||
|
// Only the element-size dimension: libhdf5 "bad dimensions".
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v3_chunked_msg(&[8]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions("bad dimensions for chunked storage".into())
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v3_chunked_msg(&[1; 34]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions("dimensionality is too large".into())
|
||||||
|
);
|
||||||
|
// v1/v2 and v4 messages get the zero check too.
|
||||||
|
let mut v1 = v1v2_header(1, 2, 2);
|
||||||
|
v1.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
v1.extend_from_slice(&0u32.to_le_bytes());
|
||||||
|
v1.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v1, 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(_))
|
||||||
|
));
|
||||||
|
let mut v4 = vec![4u8, 2, 0, 2, 4];
|
||||||
|
v4.extend_from_slice(&0u32.to_le_bytes());
|
||||||
|
v4.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
v4.push(3); // fixed array index
|
||||||
|
v4.push(0); // page bits
|
||||||
|
v4.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v4, 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(_))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A v4 chunked layout (fixed array index) whose `dims` are each
|
||||||
|
/// encoded in `width` bytes.
|
||||||
|
fn v4_chunked_msg(width: u8, dims: &[u64]) -> Vec<u8> {
|
||||||
|
let mut m = vec![4u8, 2, 0, dims.len() as u8, width];
|
||||||
|
for &d in dims {
|
||||||
|
m.extend_from_slice(&d.to_le_bytes()[..width.min(8) as usize]);
|
||||||
|
}
|
||||||
|
m.push(3); // fixed array index
|
||||||
|
m.push(0); // page bits
|
||||||
|
m.extend_from_slice(&0x1000u64.to_le_bytes());
|
||||||
|
m
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_chunk_dimensions_take_1_to_8_bytes() {
|
||||||
|
// libhdf5 encodes each dimension in the fewest bytes that hold the
|
||||||
|
// largest: a chunk dimension of 70 000 takes 3, and 3, 5, 6 and 7
|
||||||
|
// were refused ("UnexpectedEof").
|
||||||
|
for width in 1..=8u8 {
|
||||||
|
let dims = [if width >= 3 { 70_000 } else { 200 }, 8];
|
||||||
|
let layout = DataLayout::parse(&v4_chunked_msg(width, &dims), 8, 8)
|
||||||
|
.unwrap_or_else(|e| panic!("width {width}: {e:?}"));
|
||||||
|
assert!(
|
||||||
|
matches!(&layout, DataLayout::Chunked { chunk_dimensions, .. }
|
||||||
|
if chunk_dimensions.iter().map(|&d| u64::from(d)).eq(dims)),
|
||||||
|
"width {width}: {layout:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// libhdf5 refuses 0 and more than 8 bytes.
|
||||||
|
for width in [0u8, 9] {
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v4_chunked_msg(width, &[4, 8]), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidChunkDimensions(
|
||||||
|
"encoded chunk dimension size is too large".into()
|
||||||
|
)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// A dimension past u32 cannot be represented and is refused, not
|
||||||
|
// truncated.
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v4_chunked_msg(5, &[1 << 32, 8]), 8, 8),
|
||||||
|
Err(FormatError::InvalidChunkDimensions(m)) if m.contains("2^32")
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1v2_rejects_bad_class_dimensionality_and_truncation() {
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v1v2_header(1, 1, 3), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidLayoutClass(3)
|
||||||
|
);
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v1v2_header(2, 34, 1), 8, 8).unwrap_err(),
|
||||||
|
FormatError::Overflow(_)
|
||||||
|
));
|
||||||
|
// Chunked, dims cut short.
|
||||||
|
let mut buf = v1v2_header(1, 2, 2);
|
||||||
|
buf.extend_from_slice(&0x10u64.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&7u32.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnexpectedEof { .. }
|
||||||
|
));
|
||||||
|
// Compact, raw data shorter than its declared size.
|
||||||
|
let mut buf = v1v2_header(2, 1, 0);
|
||||||
|
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&100u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&[0; 4]);
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnexpectedEof { .. }
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v3_compact() {
|
fn v3_compact() {
|
||||||
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
||||||
@@ -602,6 +971,7 @@ mod tests {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -679,10 +1049,35 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_chunked_dont_filter_partial_edge_chunks_flag() {
|
||||||
|
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||||
|
buf.push(0x01); // flags bit 0 = don't filter partial edge chunks
|
||||||
|
buf.push(2); // dimensionality=2
|
||||||
|
buf.push(4); // dim_size_encoded_length=4
|
||||||
|
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||||
|
buf.push(3); // Fixed Array
|
||||||
|
buf.push(10); // max_dblk_page_nelmts_bits
|
||||||
|
buf.extend_from_slice(&0x3000u64.to_le_bytes());
|
||||||
|
match DataLayout::parse(&buf, 8, 8).unwrap() {
|
||||||
|
DataLayout::Chunked {
|
||||||
|
dont_filter_partial_edge_chunks,
|
||||||
|
btree_address,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
assert!(dont_filter_partial_edge_chunks);
|
||||||
|
assert_eq!(btree_address, Some(0x3000));
|
||||||
|
}
|
||||||
|
other => panic!("expected Chunked, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v4_chunked_single_chunk_with_filters() {
|
fn v4_chunked_single_chunk_with_filters() {
|
||||||
let mut buf = vec![4u8, 2]; // version=4, class=2
|
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||||
@@ -705,6 +1100,7 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: Some(1024),
|
single_chunk_filtered_size: Some(1024),
|
||||||
single_chunk_filter_mask: Some(0),
|
single_chunk_filter_mask: Some(0),
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -815,6 +1211,62 @@ mod tests {
|
|||||||
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_vds_mappings_v1_shared_names() {
|
||||||
|
// Written by HDF5 2.0 (h5py, libver=("v200", "v200")) for three
|
||||||
|
// mappings from `a_rather_long_source_file.h5:a_rather_long_dataset_name`
|
||||||
|
// and one from the same file: the entries carry flags 0x00, 0x03, 0x03
|
||||||
|
// and 0x06, so names after the first are stored as entry indices.
|
||||||
|
let blob: &[u8] = &[
|
||||||
|
0x01, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x61, 0x5f, 0x72, 0x61,
|
||||||
|
0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x73, 0x6f, 0x75, 0x72,
|
||||||
|
0x63, 0x65, 0x5f, 0x66, 0x69, 0x6c, 0x65, 0x2e, 0x68, 0x35, 0x00, 0x61, 0x5f, 0x72,
|
||||||
|
0x61, 0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x64, 0x61, 0x74,
|
||||||
|
0x61, 0x73, 0x65, 0x74, 0x5f, 0x6e, 0x61, 0x6d, 0x65, 0x00, 0x02, 0x00, 0x00, 0x00,
|
||||||
|
0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02,
|
||||||
|
0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00,
|
||||||
|
0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x04, 0x00, 0x01, 0x00, 0x01,
|
||||||
|
0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01,
|
||||||
|
0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00,
|
||||||
|
0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x08, 0x00, 0x01, 0x00, 0x01, 0x00,
|
||||||
|
0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02, 0x00,
|
||||||
|
0x00, 0x00, 0x02, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x01, 0x00, 0x04, 0x00, 0x06, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02,
|
||||||
|
0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00,
|
||||||
|
0x00, 0x01, 0x02, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x8e, 0xa7, 0xea, 0x7a,
|
||||||
|
];
|
||||||
|
let mappings = parse_vds_mappings(blob, 8).unwrap();
|
||||||
|
let names: Vec<(&str, &str)> = mappings
|
||||||
|
.iter()
|
||||||
|
.map(|m| (m.source_file.as_str(), m.source_dataset.as_str()))
|
||||||
|
.collect();
|
||||||
|
let (file, dset) = ("a_rather_long_source_file.h5", "a_rather_long_dataset_name");
|
||||||
|
assert_eq!(
|
||||||
|
names,
|
||||||
|
vec![(file, dset), (file, dset), (file, dset), (".", dset)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_vds_mappings_v1_forward_reference_is_error() {
|
||||||
|
// Entry 0 claiming to share entry 0's file name must not index past
|
||||||
|
// the entries decoded so far.
|
||||||
|
let mut blob = vec![0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x01];
|
||||||
|
blob.extend_from_slice(&[0u8; 8]);
|
||||||
|
blob.extend_from_slice(b"d\0");
|
||||||
|
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||||
|
// Unknown flag bits are refused.
|
||||||
|
let blob = [0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x08, b'd', 0];
|
||||||
|
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_vds_mappings_external_v0() {
|
fn parse_vds_mappings_external_v0() {
|
||||||
// Block version 0 with an explicit (external) source file name.
|
// Block version 0 with an explicit (external) source file name.
|
||||||
@@ -862,4 +1314,44 @@ mod tests {
|
|||||||
let blob = [0x01u8, 0, 0, 0, 0, 0, 0, 0, 0];
|
let blob = [0x01u8, 0, 0, 0, 0, 0, 0, 0, 0];
|
||||||
assert!(parse_vds_mappings(&blob, 8).unwrap().is_empty());
|
assert!(parse_vds_mappings(&blob, 8).unwrap().is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A virtual dataset's mappings resolve identically through a
|
||||||
|
/// read_at-only CountingStorage, in two reads of the global heap.
|
||||||
|
#[test]
|
||||||
|
fn vds_mappings_through_storage_match_slice() {
|
||||||
|
use crate::message_type::MessageType;
|
||||||
|
use crate::object_header::ObjectHeader;
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/vds_same_file.h5");
|
||||||
|
let sb = crate::superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let (os, ls) = (sb.offset_size, sb.length_size);
|
||||||
|
let storage = CountingStorage::new(file.to_vec());
|
||||||
|
let mut virtuals = 0;
|
||||||
|
for child in
|
||||||
|
crate::group_v2::resolve_group_children(file, &sb, sb.root_group_address).unwrap()
|
||||||
|
{
|
||||||
|
let h =
|
||||||
|
ObjectHeader::parse(file, child.object_header_address as usize, os, ls).unwrap();
|
||||||
|
let Some(msg) = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let mut want = DataLayout::parse(&msg.data, os, ls).unwrap();
|
||||||
|
if !matches!(want, DataLayout::Virtual { .. }) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mut got = want.clone();
|
||||||
|
want.resolve_vds_mappings(file, ls).unwrap();
|
||||||
|
storage.reset();
|
||||||
|
got.resolve_vds_mappings_in(&storage, ls).unwrap();
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
assert!(matches!(&got, DataLayout::Virtual { mappings, .. } if !mappings.is_empty()));
|
||||||
|
assert_eq!(storage.reads(), 2);
|
||||||
|
virtuals += 1;
|
||||||
|
}
|
||||||
|
assert!(virtuals >= 1);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+1155
-492
File diff suppressed because it is too large
Load Diff
@@ -7,6 +7,9 @@ use alloc::vec::Vec;
|
|||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// Most dimensions a dataspace can have (`H5S_MAX_RANK`).
|
||||||
|
pub const MAX_RANK: u8 = 32;
|
||||||
|
|
||||||
/// Type of dataspace.
|
/// Type of dataspace.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
pub enum DataspaceType {
|
pub enum DataspaceType {
|
||||||
@@ -67,6 +70,12 @@ impl Dataspace {
|
|||||||
let version = data[0];
|
let version = data[0];
|
||||||
let rank = data[1];
|
let rank = data[1];
|
||||||
let flags = data[2];
|
let flags = data[2];
|
||||||
|
// H5O__sdspace_decode's checks.
|
||||||
|
if rank > MAX_RANK {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"simple dataspace dimensionality is too large",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
let (space_type, header_size) = match version {
|
let (space_type, header_size) = match version {
|
||||||
1 => {
|
1 => {
|
||||||
@@ -88,6 +97,11 @@ impl Dataspace {
|
|||||||
2 => DataspaceType::Null,
|
2 => DataspaceType::Null,
|
||||||
_ => return Err(FormatError::InvalidDataspaceType(type_byte)),
|
_ => return Err(FormatError::InvalidDataspaceType(type_byte)),
|
||||||
};
|
};
|
||||||
|
if st != DataspaceType::Simple && rank > 0 {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"invalid rank for scalar or NULL dataspace",
|
||||||
|
));
|
||||||
|
}
|
||||||
(st, 4usize)
|
(st, 4usize)
|
||||||
}
|
}
|
||||||
_ => return Err(FormatError::InvalidDataspaceVersion(version)),
|
_ => return Err(FormatError::InvalidDataspaceVersion(version)),
|
||||||
@@ -107,8 +121,13 @@ impl Dataspace {
|
|||||||
// Read max dimensions if flags bit 0 is set
|
// Read max dimensions if flags bit 0 is set
|
||||||
let max_dimensions = if flags & 0x01 != 0 {
|
let max_dimensions = if flags & 0x01 != 0 {
|
||||||
let mut max_dims = Vec::with_capacity(rank as usize);
|
let mut max_dims = Vec::with_capacity(rank as usize);
|
||||||
for _ in 0..rank {
|
for &dim in &dimensions {
|
||||||
let val = read_length(data, pos, length_size)?;
|
let val = read_length(data, pos, length_size)?;
|
||||||
|
if dim > val {
|
||||||
|
return Err(FormatError::InvalidDataspace(
|
||||||
|
"dataspace dimension size is greater than its maximum size",
|
||||||
|
));
|
||||||
|
}
|
||||||
max_dims.push(val);
|
max_dims.push(val);
|
||||||
pos += ls;
|
pos += ls;
|
||||||
}
|
}
|
||||||
@@ -176,7 +195,6 @@ impl Dataspace {
|
|||||||
match self.space_type {
|
match self.space_type {
|
||||||
DataspaceType::Null => Ok(0),
|
DataspaceType::Null => Ok(0),
|
||||||
DataspaceType::Scalar => Ok(1),
|
DataspaceType::Scalar => Ok(1),
|
||||||
DataspaceType::Simple if self.dimensions.is_empty() => Ok(0),
|
|
||||||
DataspaceType::Simple => self
|
DataspaceType::Simple => self
|
||||||
.dimensions
|
.dimensions
|
||||||
.iter()
|
.iter()
|
||||||
@@ -195,18 +213,14 @@ impl Dataspace {
|
|||||||
match self.space_type {
|
match self.space_type {
|
||||||
DataspaceType::Null => 0,
|
DataspaceType::Null => 0,
|
||||||
DataspaceType::Scalar => 1,
|
DataspaceType::Scalar => 1,
|
||||||
DataspaceType::Simple => {
|
// A simple dataspace of rank 0 holds one element, as in libhdf5
|
||||||
if self.dimensions.is_empty() {
|
// (the product of no dimensions). Saturate rather than wrap: a
|
||||||
0
|
// wrapped product could under-size a buffer. Size-critical
|
||||||
} else {
|
// callers use `checked_num_elements`.
|
||||||
// Saturate rather than wrap: a wrapped product could
|
DataspaceType::Simple => self
|
||||||
// under-size a buffer. Size-critical callers use
|
.dimensions
|
||||||
// `checked_num_elements`.
|
.iter()
|
||||||
self.dimensions
|
.fold(1u64, |acc, &d| acc.saturating_mul(d)),
|
||||||
.iter()
|
|
||||||
.fold(1u64, |acc, &d| acc.saturating_mul(d))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -352,4 +366,44 @@ mod tests {
|
|||||||
let ds = Dataspace::parse(&data, 8).unwrap();
|
let ds = Dataspace::parse(&data, 8).unwrap();
|
||||||
assert_eq!(ds.max_dimensions, Some(vec![10]));
|
assert_eq!(ds.max_dimensions, Some(vec![10]));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A simple dataspace of rank 0 (cve-2020-18494's `/dset1`) holds one
|
||||||
|
/// element in libhdf5, which h5py reads as shape `()`. It was 0.
|
||||||
|
#[test]
|
||||||
|
fn simple_rank_zero_holds_one_element() {
|
||||||
|
let data = build_v2_dataspace(0, 0, 1, &[], None);
|
||||||
|
let ds = Dataspace::parse(&data, 8).unwrap();
|
||||||
|
assert_eq!(ds.space_type, DataspaceType::Simple);
|
||||||
|
assert_eq!(ds.num_elements(), 1);
|
||||||
|
assert_eq!(ds.checked_num_elements().unwrap(), 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `H5O__sdspace_decode`'s checks.
|
||||||
|
#[test]
|
||||||
|
fn refuses_what_libhdf5_refuses() {
|
||||||
|
let too_many = build_v2_dataspace(33, 0, 1, &[1; 33], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&too_many, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let scalar_with_rank = build_v2_dataspace(1, 0, 0, &[4], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&scalar_with_rank, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let null_with_rank = build_v2_dataspace(1, 0, 2, &[4], None);
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&null_with_rank, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
let over_max = build_v1_dataspace(2, 0x01, &[5, 20], Some(&[10, 10]));
|
||||||
|
assert!(matches!(
|
||||||
|
Dataspace::parse(&over_max, 8),
|
||||||
|
Err(FormatError::InvalidDataspace(_))
|
||||||
|
));
|
||||||
|
// 32 dimensions, and a size equal to the maximum or unlimited, are fine.
|
||||||
|
assert!(Dataspace::parse(&build_v2_dataspace(32, 0, 1, &[1; 32], None), 8).is_ok());
|
||||||
|
let at_max = build_v1_dataspace(2, 0x01, &[10, 20], Some(&[10, u64::MAX]));
|
||||||
|
assert!(Dataspace::parse(&at_max, 8).is_ok());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -3,11 +3,14 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
extern crate alloc;
|
extern crate alloc;
|
||||||
|
|
||||||
|
use crate::addr::saturating_usize;
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{vec, vec::Vec};
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
use crate::checksum::jenkins_lookup3;
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::chunked_write::WrittenChunk;
|
use crate::chunked_write::{
|
||||||
|
WrittenChunk, filtered_chunk_size_len, push_addr, push_index_element, push_v4_chunk_dims,
|
||||||
|
};
|
||||||
|
|
||||||
/// Serialize a v4 Extensible Array layout message.
|
/// Serialize a v4 Extensible Array layout message.
|
||||||
pub(crate) fn serialize_v4_extensible_array(
|
pub(crate) fn serialize_v4_extensible_array(
|
||||||
@@ -24,45 +27,17 @@ pub(crate) fn serialize_v4_extensible_array(
|
|||||||
let ndims = chunk_dims.len() as u8 + 1;
|
let ndims = chunk_dims.len() as u8 + 1;
|
||||||
buf.push(ndims);
|
buf.push(ndims);
|
||||||
|
|
||||||
let max_dim = chunk_dims
|
push_v4_chunk_dims(&mut buf, chunk_dims, element_size);
|
||||||
.iter()
|
|
||||||
.map(|&d| d as u64)
|
|
||||||
.chain(core::iter::once(element_size as u64))
|
|
||||||
.max()
|
|
||||||
.unwrap_or(1);
|
|
||||||
let dim_encoded_len: u8 = if max_dim <= 0xFF {
|
|
||||||
1
|
|
||||||
} else if max_dim <= 0xFFFF {
|
|
||||||
2
|
|
||||||
} else {
|
|
||||||
4
|
|
||||||
};
|
|
||||||
buf.push(dim_encoded_len);
|
|
||||||
|
|
||||||
for &d in chunk_dims {
|
|
||||||
match dim_encoded_len {
|
|
||||||
1 => buf.push(d as u8),
|
|
||||||
2 => buf.extend_from_slice(&(d as u16).to_le_bytes()),
|
|
||||||
4 => buf.extend_from_slice(&d.to_le_bytes()),
|
|
||||||
_ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
match dim_encoded_len {
|
|
||||||
1 => buf.push(element_size as u8),
|
|
||||||
2 => buf.extend_from_slice(&(element_size as u16).to_le_bytes()),
|
|
||||||
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
|
||||||
_ => unreachable!("unexpected dim_encoded_len: {dim_encoded_len}"),
|
|
||||||
}
|
|
||||||
|
|
||||||
// chunk index type = 4 (Extensible Array)
|
// chunk index type = 4 (Extensible Array)
|
||||||
buf.push(4);
|
buf.push(4);
|
||||||
|
|
||||||
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
||||||
buf.push(32); // max_nelmts_bits
|
buf.push(MAX_NELMTS_BITS);
|
||||||
buf.push(4); // idx_blk_elmts
|
buf.push(IDX_BLK_ELMTS);
|
||||||
buf.push(4); // super_blk_min_data_ptrs
|
buf.push(SUP_BLK_MIN_DATA_PTRS);
|
||||||
buf.push(16); // data_blk_min_elmts
|
buf.push(DATA_BLK_MIN_ELMTS);
|
||||||
buf.push(10); // max_dblk_page_nelmts_bits
|
buf.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||||
|
|
||||||
// EA header address
|
// EA header address
|
||||||
match offset_size {
|
match offset_size {
|
||||||
@@ -74,304 +49,281 @@ pub(crate) fn serialize_v4_extensible_array(
|
|||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// EA creation parameters — the HDF5 library's defaults for chunk indexes
|
||||||
|
// (`H5D_EARRAY_*`); the layout message above and the header must agree.
|
||||||
|
const MAX_NELMTS_BITS: u8 = 32;
|
||||||
|
const IDX_BLK_ELMTS: u8 = 4;
|
||||||
|
const SUP_BLK_MIN_DATA_PTRS: u8 = 4;
|
||||||
|
const DATA_BLK_MIN_ELMTS: u8 = 16;
|
||||||
|
const MAX_DBLK_PAGE_NELMTS_BITS: u8 = 10;
|
||||||
|
|
||||||
|
/// One data block of the array: its first element (relative to the end of
|
||||||
|
/// the index block's own elements), element count, and address when it is
|
||||||
|
/// allocated.
|
||||||
|
struct DataBlock {
|
||||||
|
start: usize,
|
||||||
|
nelmts: usize,
|
||||||
|
addr: Option<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
/// Build a complete Extensible Array at a known absolute address.
|
/// Build a complete Extensible Array at a known absolute address.
|
||||||
///
|
///
|
||||||
/// For simplicity, we put all elements inline in the index block when the
|
/// `slots[i]` is the element at linear index `i` (see `chunk_grid`); `None`
|
||||||
/// number of chunks is small (up to idx_blk_elmts), otherwise use inline +
|
/// marks an unallocated chunk. The first `IDX_BLK_ELMTS` elements live in
|
||||||
/// direct data blocks.
|
/// the index block, the rest in data blocks grouped by super block level
|
||||||
|
/// exactly as `H5EA__hdr_init` sizes them: level `u` has `2^(u/2)` data
|
||||||
|
/// blocks of `DATA_BLK_MIN_ELMTS * 2^ceil(u/2)` elements. The data blocks of
|
||||||
|
/// the first levels are addressed straight from the index block; later
|
||||||
|
/// levels go through a super block (EASB). Data blocks larger than a page
|
||||||
|
/// (`2^MAX_DBLK_PAGE_NELMTS_BITS` elements) are paged, with their page-init
|
||||||
|
/// bits kept in the owning super block. Only blocks holding a defined element
|
||||||
|
/// are allocated; the rest keep the undefined address, as in a file the
|
||||||
|
/// library wrote.
|
||||||
pub fn build_extensible_array_at(
|
pub fn build_extensible_array_at(
|
||||||
chunks: &[WrittenChunk],
|
slots: &[Option<WrittenChunk>],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
has_filters: bool,
|
has_filters: bool,
|
||||||
ea_base_address: u64,
|
ea_base_address: u64,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let num_elements = chunks.len();
|
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||||
|
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||||
// Compute element encoding size (same logic as Fixed Array)
|
|
||||||
let chunk_size_bytes: usize = if has_filters {
|
|
||||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
|
||||||
let log2_val = if max_raw <= 1 {
|
|
||||||
0
|
|
||||||
} else {
|
|
||||||
63 - max_raw.leading_zeros()
|
|
||||||
};
|
|
||||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
|
||||||
len.min(8)
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
|
|
||||||
let elem_size = if has_filters {
|
|
||||||
os + chunk_size_bytes + 4
|
|
||||||
} else {
|
|
||||||
os
|
|
||||||
};
|
|
||||||
|
|
||||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||||
|
let arr_off_size = (MAX_NELMTS_BITS as usize).div_ceil(8);
|
||||||
|
let page_nelmts = 1usize << MAX_DBLK_PAGE_NELMTS_BITS;
|
||||||
|
let idx_blk = IDX_BLK_ELMTS as usize;
|
||||||
|
|
||||||
// EA creation parameters — must match HDF5 C library defaults exactly
|
// Elements past the last defined one are never realised
|
||||||
let max_nelmts_bits: u8 = 32;
|
// (`max_idx_set` is one past the highest index ever set).
|
||||||
let idx_blk_elmts: u8 = 4;
|
let max_idx_set = slots.iter().rposition(Option::is_some).map_or(0, |i| i + 1);
|
||||||
let min_dblk_nelmts: u8 = 16;
|
let slots = &slots[..max_idx_set];
|
||||||
let super_blk_min_nelmts: u8 = 4;
|
let defined_in = |start: usize, n: usize| -> bool {
|
||||||
let max_dblk_nelmts_bits: u8 = 10;
|
let lo = idx_blk.saturating_add(start).min(slots.len());
|
||||||
|
let hi = idx_blk
|
||||||
|
.saturating_add(start)
|
||||||
|
.saturating_add(n)
|
||||||
|
.min(slots.len());
|
||||||
|
slots[lo..hi].iter().any(Option::is_some)
|
||||||
|
};
|
||||||
|
|
||||||
// EAHD size: fixed(12) + 6 stats(6*length_size) + addr(offset_size) + checksum(4)
|
// Super block levels: (ndblks, dblk_nelmts, first element).
|
||||||
|
let log2_dmin = (DATA_BLK_MIN_ELMTS as u32).trailing_zeros() as usize;
|
||||||
|
let nsblks = 1 + MAX_NELMTS_BITS as usize - log2_dmin;
|
||||||
|
let ndblk_addrs = 2 * (SUP_BLK_MIN_DATA_PTRS as usize - 1);
|
||||||
|
let mut levels: Vec<(usize, usize, usize)> = Vec::with_capacity(nsblks);
|
||||||
|
let mut start = 0usize;
|
||||||
|
for u in 0..nsblks {
|
||||||
|
let ndblks = 1usize << (u / 2);
|
||||||
|
let nelmts = (DATA_BLK_MIN_ELMTS as usize) << u.div_ceil(2);
|
||||||
|
levels.push((ndblks, nelmts, start));
|
||||||
|
// Saturate: on 32-bit targets the last levels only need to compare
|
||||||
|
// as "beyond the end".
|
||||||
|
start = start.saturating_add(ndblks.saturating_mul(nelmts));
|
||||||
|
}
|
||||||
|
// Levels whose data blocks the index block addresses directly.
|
||||||
|
let mut direct_levels = 0;
|
||||||
|
let mut n = 0;
|
||||||
|
while n < ndblk_addrs {
|
||||||
|
n += levels[direct_levels].0;
|
||||||
|
direct_levels += 1;
|
||||||
|
}
|
||||||
|
let nsblk_addrs = nsblks - direct_levels;
|
||||||
|
|
||||||
|
let dblk_size = |nelmts: usize| -> usize {
|
||||||
|
let prefix = 4 + 1 + 1 + os + arr_off_size + 4;
|
||||||
|
if nelmts > page_nelmts {
|
||||||
|
prefix + (nelmts / page_nelmts) * (page_nelmts * elem_size + 4)
|
||||||
|
} else {
|
||||||
|
prefix + nelmts * elem_size
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let sblk_bitmap_len = |ndblks: usize, nelmts: usize| -> usize {
|
||||||
|
if nelmts > page_nelmts {
|
||||||
|
ndblks * (nelmts / page_nelmts).div_ceil(8)
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Plan addresses: header, index block, the direct data blocks, then each
|
||||||
|
// allocated super block followed by its allocated data blocks.
|
||||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
||||||
let aeib_address = ea_base_address + aehd_size as u64;
|
let aeib_address = ea_base_address + aehd_size as u64;
|
||||||
|
let aeib_size = 4 + 1 + 1 + os + idx_blk * elem_size + ndblk_addrs * os + nsblk_addrs * os + 4;
|
||||||
|
let mut cursor = aeib_address + aeib_size as u64;
|
||||||
|
|
||||||
// Determine how many elements go inline vs data blocks
|
let mut ndata_blks = 0u64;
|
||||||
let n_inline = (idx_blk_elmts as usize).min(num_elements);
|
let mut data_blk_size = 0u64;
|
||||||
let remaining_after_inline = num_elements.saturating_sub(n_inline);
|
let mut nsuper_blks = 0u64;
|
||||||
|
let mut super_blk_size = 0u64;
|
||||||
|
let mut realized = idx_blk as u64;
|
||||||
|
|
||||||
// Compute super block layout per HDF5 spec
|
let mut plan_dblk = |cursor: &mut u64, start: usize, nelmts: usize| -> DataBlock {
|
||||||
let sblk_min = super_blk_min_nelmts as usize;
|
let addr = defined_in(start, nelmts).then(|| {
|
||||||
let log2_dblk_min = if min_dblk_nelmts <= 1 {
|
let a = *cursor;
|
||||||
0
|
let size = dblk_size(nelmts) as u64;
|
||||||
} else {
|
*cursor += size;
|
||||||
(min_dblk_nelmts as u32).trailing_zeros() as usize
|
ndata_blks += 1;
|
||||||
|
data_blk_size += size;
|
||||||
|
realized += nelmts as u64;
|
||||||
|
a
|
||||||
|
});
|
||||||
|
DataBlock {
|
||||||
|
start,
|
||||||
|
nelmts,
|
||||||
|
addr,
|
||||||
|
}
|
||||||
};
|
};
|
||||||
let nsblks = (max_nelmts_bits as usize).saturating_sub(log2_dblk_min) + 1;
|
|
||||||
|
|
||||||
// Direct data block addresses (from super blocks 0..sblk_min-1)
|
let mut direct: Vec<DataBlock> = Vec::with_capacity(ndblk_addrs);
|
||||||
let mut dblk_sizes: Vec<usize> = Vec::new();
|
for &(ndblks, nelmts, first) in &levels[..direct_levels] {
|
||||||
for sblk_idx in 0..sblk_min.min(nsblks) {
|
for k in 0..ndblks {
|
||||||
let ndblks = 1usize << (sblk_idx / 2);
|
direct.push(plan_dblk(&mut cursor, first + k * nelmts, nelmts));
|
||||||
let dblk_nelmts = (min_dblk_nelmts as usize) * (1 << sblk_idx.div_ceil(2));
|
|
||||||
for _ in 0..ndblks {
|
|
||||||
dblk_sizes.push(dblk_nelmts);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
let n_direct_dblks = dblk_sizes.len();
|
// (super block address, level, its data blocks)
|
||||||
|
let mut supers: Vec<(Option<u64>, usize, Vec<DataBlock>)> = Vec::with_capacity(nsblk_addrs);
|
||||||
// Super block addresses (for super blocks sblk_min..nsblks-1)
|
for (u, &(ndblks, nelmts, first)) in levels.iter().enumerate().skip(direct_levels) {
|
||||||
let n_sblk_addrs = nsblks.saturating_sub(sblk_min);
|
if !defined_in(first, ndblks.saturating_mul(nelmts)) {
|
||||||
|
supers.push((None, u, Vec::new()));
|
||||||
// EAIB size
|
continue;
|
||||||
let aeib_size = 4
|
|
||||||
+ 1
|
|
||||||
+ 1
|
|
||||||
+ os
|
|
||||||
+ idx_blk_elmts as usize * elem_size
|
|
||||||
+ n_direct_dblks * os
|
|
||||||
+ n_sblk_addrs * os
|
|
||||||
+ 4;
|
|
||||||
|
|
||||||
// Build AEHD
|
|
||||||
let mut aehd = Vec::with_capacity(aehd_size);
|
|
||||||
aehd.extend_from_slice(b"EAHD");
|
|
||||||
aehd.push(0); // version
|
|
||||||
aehd.push(client_id);
|
|
||||||
aehd.push(elem_size as u8);
|
|
||||||
aehd.push(max_nelmts_bits);
|
|
||||||
aehd.push(idx_blk_elmts);
|
|
||||||
aehd.push(min_dblk_nelmts);
|
|
||||||
aehd.push(super_blk_min_nelmts);
|
|
||||||
aehd.push(max_dblk_nelmts_bits);
|
|
||||||
|
|
||||||
// Count data blocks that will have chunks
|
|
||||||
let n_active_dblks: u64 = if remaining_after_inline > 0 {
|
|
||||||
let mut count = 0u64;
|
|
||||||
let mut ci = n_inline;
|
|
||||||
for &sz in &dblk_sizes {
|
|
||||||
if ci < num_elements {
|
|
||||||
count += 1;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
count
|
let sb_size =
|
||||||
} else {
|
4 + 1 + 1 + os + arr_off_size + sblk_bitmap_len(ndblks, nelmts) + ndblks * os + 4;
|
||||||
0
|
let sb_addr = cursor;
|
||||||
};
|
cursor += sb_size as u64;
|
||||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
nsuper_blks += 1;
|
||||||
let aedb_header_overhead = 4 + 1 + 1 + os + blk_off_size + 4;
|
super_blk_size += sb_size as u64;
|
||||||
let data_blk_total_size: u64 = if remaining_after_inline > 0 {
|
let dblks = (0..ndblks)
|
||||||
let mut total = 0u64;
|
.map(|k| plan_dblk(&mut cursor, first + k * nelmts, nelmts))
|
||||||
let mut ci = n_inline;
|
.collect();
|
||||||
for &sz in &dblk_sizes {
|
supers.push((Some(sb_addr), u, dblks));
|
||||||
if ci < num_elements {
|
}
|
||||||
total += (aedb_header_overhead + sz * elem_size) as u64;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
total
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
let max_idx_set: u64 = if remaining_after_inline > 0 {
|
|
||||||
let mut max_set = idx_blk_elmts as u64;
|
|
||||||
let mut ci = n_inline;
|
|
||||||
for &sz in &dblk_sizes {
|
|
||||||
if ci < num_elements {
|
|
||||||
max_set += sz as u64;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
max_set
|
|
||||||
} else {
|
|
||||||
idx_blk_elmts as u64
|
|
||||||
};
|
|
||||||
|
|
||||||
|
let slot = |i: usize| slots.get(i).and_then(Option::as_ref);
|
||||||
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
||||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
||||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
||||||
};
|
};
|
||||||
let write_addr = |buf: &mut Vec<u8>, val: u64| match offset_size {
|
let write_addr_opt = |buf: &mut Vec<u8>, addr: Option<u64>| match addr {
|
||||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
Some(a) => push_addr(buf, a, offset_size),
|
||||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
None => buf.extend(core::iter::repeat_n(0xFF, os)),
|
||||||
|
};
|
||||||
|
let block_prefix = |buf: &mut Vec<u8>, sig: &[u8; 4], block_off: usize| {
|
||||||
|
buf.extend_from_slice(sig);
|
||||||
|
buf.push(0); // version
|
||||||
|
buf.push(client_id);
|
||||||
|
push_addr(buf, ea_base_address, offset_size);
|
||||||
|
buf.extend_from_slice(&(block_off as u64).to_le_bytes()[..arr_off_size]);
|
||||||
|
};
|
||||||
|
// Serialise one data block (paged or not) onto `out`.
|
||||||
|
let write_dblk = |out: &mut Vec<u8>, db: &DataBlock| {
|
||||||
|
let at = out.len();
|
||||||
|
block_prefix(out, b"EADB", db.start);
|
||||||
|
let first = idx_blk + db.start;
|
||||||
|
if db.nelmts > page_nelmts {
|
||||||
|
// Paged: the prefix carries only its own checksum; each page
|
||||||
|
// follows with one of its own.
|
||||||
|
let sum = jenkins_lookup3(&out[at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
for p in 0..db.nelmts / page_nelmts {
|
||||||
|
let page_at = out.len();
|
||||||
|
for e in 0..page_nelmts {
|
||||||
|
let i = first + p * page_nelmts + e;
|
||||||
|
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[page_at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
for i in first..first + db.nelmts {
|
||||||
|
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
}
|
||||||
|
debug_assert_eq!(out.len() - at, dblk_size(db.nelmts));
|
||||||
};
|
};
|
||||||
|
|
||||||
write_length(&mut aehd, 0);
|
// Header (EAHD). The six statistics are, in order: super blocks, their
|
||||||
write_length(&mut aehd, 0);
|
// bytes, data blocks, their bytes, max index set, elements realised.
|
||||||
write_length(&mut aehd, n_active_dblks);
|
let mut out = Vec::with_capacity(saturating_usize(cursor - ea_base_address));
|
||||||
write_length(&mut aehd, data_blk_total_size);
|
out.extend_from_slice(b"EAHD");
|
||||||
write_length(&mut aehd, num_elements as u64);
|
out.push(0); // version
|
||||||
write_length(&mut aehd, max_idx_set);
|
out.push(client_id);
|
||||||
|
out.push(elem_size as u8);
|
||||||
|
out.push(MAX_NELMTS_BITS);
|
||||||
|
out.push(IDX_BLK_ELMTS);
|
||||||
|
out.push(DATA_BLK_MIN_ELMTS);
|
||||||
|
out.push(SUP_BLK_MIN_DATA_PTRS);
|
||||||
|
out.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||||
|
write_length(&mut out, nsuper_blks);
|
||||||
|
write_length(&mut out, super_blk_size);
|
||||||
|
write_length(&mut out, ndata_blks);
|
||||||
|
write_length(&mut out, data_blk_size);
|
||||||
|
write_length(&mut out, max_idx_set as u64);
|
||||||
|
write_length(&mut out, realized);
|
||||||
|
push_addr(&mut out, aeib_address, offset_size);
|
||||||
|
let sum = jenkins_lookup3(&out);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len(), aehd_size);
|
||||||
|
|
||||||
write_addr(&mut aehd, aeib_address);
|
// Index block (EAIB): inline elements, data block and super block
|
||||||
|
// addresses.
|
||||||
let aehd_checksum = jenkins_lookup3(&aehd);
|
let ib_start = out.len();
|
||||||
aehd.extend_from_slice(&aehd_checksum.to_le_bytes());
|
out.extend_from_slice(b"EAIB");
|
||||||
debug_assert_eq!(aehd.len(), aehd_size);
|
out.push(0);
|
||||||
|
out.push(client_id);
|
||||||
// Build AEIB
|
push_addr(&mut out, ea_base_address, offset_size);
|
||||||
let mut aeib = Vec::with_capacity(aeib_size);
|
for i in 0..idx_blk {
|
||||||
aeib.extend_from_slice(b"EAIB");
|
push_index_element(&mut out, slot(i), offset_size, chunk_size_bytes);
|
||||||
aeib.push(0);
|
|
||||||
aeib.push(client_id);
|
|
||||||
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
|
for db in &direct {
|
||||||
// Inline elements
|
write_addr_opt(&mut out, db.addr);
|
||||||
#[allow(clippy::needless_range_loop)]
|
|
||||||
for i in 0..idx_blk_elmts as usize {
|
|
||||||
if i < n_inline {
|
|
||||||
write_chunk_element(
|
|
||||||
&mut aeib,
|
|
||||||
&chunks[i],
|
|
||||||
offset_size,
|
|
||||||
has_filters,
|
|
||||||
chunk_size_bytes,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
write_undefined_element(&mut aeib, offset_size, has_filters, chunk_size_bytes);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
for (sb_addr, _, _) in &supers {
|
||||||
|
write_addr_opt(&mut out, *sb_addr);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[ib_start..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len() - ib_start, aeib_size);
|
||||||
|
|
||||||
// Data block addresses + build data blocks
|
for db in direct.iter().filter(|d| d.addr.is_some()) {
|
||||||
let mut data_blocks_buf = Vec::new();
|
write_dblk(&mut out, db);
|
||||||
let dblks_base = aeib_address + aeib_size as u64;
|
}
|
||||||
let mut dblk_cursor = dblks_base;
|
for (sb_addr, u, dblks) in &supers {
|
||||||
let mut chunk_idx = n_inline;
|
if sb_addr.is_none() {
|
||||||
|
|
||||||
for &nelmts in &dblk_sizes {
|
|
||||||
if chunk_idx >= num_elements {
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
}
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
let (ndblks, nelmts, first) = levels[*u];
|
||||||
match offset_size {
|
let sb_start = out.len();
|
||||||
4 => aeib.extend_from_slice(&(dblk_cursor as u32).to_le_bytes()),
|
block_prefix(&mut out, b"EASB", first);
|
||||||
8 => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
if nelmts > page_nelmts {
|
||||||
_ => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
// Page-init bits, `npages` per data block, packed MSB-first
|
||||||
}
|
// (`H5VM_bit_set`): every page of an allocated data block is
|
||||||
|
// written.
|
||||||
// Build EADB
|
let npages = nelmts / page_nelmts;
|
||||||
let mut aedb = Vec::new();
|
let mut bitmap = vec![0u8; sblk_bitmap_len(ndblks, nelmts)];
|
||||||
aedb.extend_from_slice(b"EADB");
|
for (k, db) in dblks.iter().enumerate() {
|
||||||
aedb.push(0);
|
if db.addr.is_some() {
|
||||||
aedb.push(client_id);
|
for p in 0..npages {
|
||||||
match offset_size {
|
let bit = k * npages + p;
|
||||||
4 => aedb.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
bitmap[bit / 8] |= 0x80 >> (bit % 8);
|
||||||
8 => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
}
|
||||||
_ => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
}
|
||||||
}
|
|
||||||
|
|
||||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
|
||||||
let blk_off_val = (chunk_idx - n_inline) as u64;
|
|
||||||
aedb.extend_from_slice(&blk_off_val.to_le_bytes()[..blk_off_size]);
|
|
||||||
|
|
||||||
for slot in 0..nelmts {
|
|
||||||
if chunk_idx + slot < num_elements {
|
|
||||||
write_chunk_element(
|
|
||||||
&mut aedb,
|
|
||||||
&chunks[chunk_idx + slot],
|
|
||||||
offset_size,
|
|
||||||
has_filters,
|
|
||||||
chunk_size_bytes,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
write_undefined_element(&mut aedb, offset_size, has_filters, chunk_size_bytes);
|
|
||||||
}
|
}
|
||||||
|
out.extend_from_slice(&bitmap);
|
||||||
}
|
}
|
||||||
|
for db in dblks {
|
||||||
let aedb_checksum = jenkins_lookup3(&aedb);
|
write_addr_opt(&mut out, db.addr);
|
||||||
aedb.extend_from_slice(&aedb_checksum.to_le_bytes());
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[sb_start..]);
|
||||||
dblk_cursor += aedb.len() as u64;
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
data_blocks_buf.extend_from_slice(&aedb);
|
for db in dblks.iter().filter(|d| d.addr.is_some()) {
|
||||||
chunk_idx += nelmts;
|
write_dblk(&mut out, db);
|
||||||
}
|
|
||||||
|
|
||||||
// Super block addresses (all undefined)
|
|
||||||
for _ in 0..n_sblk_addrs {
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
debug_assert_eq!(out.len() as u64, cursor - ea_base_address);
|
||||||
let aeib_checksum = jenkins_lookup3(&aeib);
|
out
|
||||||
aeib.extend_from_slice(&aeib_checksum.to_le_bytes());
|
|
||||||
debug_assert_eq!(aeib.len(), aeib_size);
|
|
||||||
|
|
||||||
let mut combined = aehd;
|
|
||||||
combined.extend_from_slice(&aeib);
|
|
||||||
combined.extend_from_slice(&data_blocks_buf);
|
|
||||||
combined
|
|
||||||
}
|
|
||||||
|
|
||||||
fn write_chunk_element(
|
|
||||||
buf: &mut Vec<u8>,
|
|
||||||
chunk: &WrittenChunk,
|
|
||||||
offset_size: u8,
|
|
||||||
has_filters: bool,
|
|
||||||
chunk_size_bytes: usize,
|
|
||||||
) {
|
|
||||||
match offset_size {
|
|
||||||
4 => buf.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
|
||||||
8 => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
_ => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
}
|
|
||||||
if has_filters {
|
|
||||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
|
||||||
buf.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
|
||||||
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn write_undefined_element(
|
|
||||||
buf: &mut Vec<u8>,
|
|
||||||
offset_size: u8,
|
|
||||||
has_filters: bool,
|
|
||||||
chunk_size_bytes: usize,
|
|
||||||
) {
|
|
||||||
let os = offset_size as usize;
|
|
||||||
// Use extend with repeat to avoid heap-allocating a temporary Vec on each call.
|
|
||||||
buf.extend(core::iter::repeat_n(0xFF, os));
|
|
||||||
if has_filters {
|
|
||||||
buf.extend(core::iter::repeat_n(0x00, chunk_size_bytes));
|
|
||||||
buf.extend_from_slice(&0u32.to_le_bytes());
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -12,7 +12,11 @@ use std::string::String;
|
|||||||
use core::fmt;
|
use core::fmt;
|
||||||
|
|
||||||
/// Errors that can occur when parsing HDF5 binary format structures.
|
/// Errors that can occur when parsing HDF5 binary format structures.
|
||||||
|
///
|
||||||
|
/// Non-exhaustive: new failure modes (new storage backends, new file
|
||||||
|
/// features) add variants, so a `match` needs a wildcard arm.
|
||||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
#[non_exhaustive]
|
||||||
pub enum FormatError {
|
pub enum FormatError {
|
||||||
/// The HDF5 magic signature was not found at any valid offset.
|
/// The HDF5 magic signature was not found at any valid offset.
|
||||||
SignatureNotFound,
|
SignatureNotFound,
|
||||||
@@ -80,6 +84,9 @@ pub enum FormatError {
|
|||||||
InvalidLocalHeapSignature,
|
InvalidLocalHeapSignature,
|
||||||
/// Invalid local heap version.
|
/// Invalid local heap version.
|
||||||
InvalidLocalHeapVersion(u8),
|
InvalidLocalHeapVersion(u8),
|
||||||
|
/// A local heap's free list points outside its data segment (libhdf5:
|
||||||
|
/// "bad heap free list").
|
||||||
|
InvalidLocalHeapFreeList,
|
||||||
/// Invalid B-tree v1 signature.
|
/// Invalid B-tree v1 signature.
|
||||||
InvalidBTreeSignature,
|
InvalidBTreeSignature,
|
||||||
/// Invalid B-tree node type.
|
/// Invalid B-tree node type.
|
||||||
@@ -117,6 +124,14 @@ pub enum FormatError {
|
|||||||
/// A message is marked shared but was parsed without access to the file,
|
/// A message is marked shared but was parsed without access to the file,
|
||||||
/// so the reference to the real message could not be followed.
|
/// so the reference to the real message could not be followed.
|
||||||
UnresolvedSharedMessage,
|
UnresolvedSharedMessage,
|
||||||
|
/// A shared-message reference points at an object header that holds no
|
||||||
|
/// (unshared) message of the referenced type (raw message type id).
|
||||||
|
SharedMessageTargetMissing(u16),
|
||||||
|
/// A superblock was parsed at a non-zero offset of the buffer (the file
|
||||||
|
/// has a user block of this many bytes). HDF5 addresses are relative to
|
||||||
|
/// the superblock, so the buffer must start there: see
|
||||||
|
/// `signature::split_user_block`.
|
||||||
|
UserBlockNotStripped(u64),
|
||||||
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
||||||
/// it reaches past a dimension's extent).
|
/// it reaches past a dimension's extent).
|
||||||
SelectionOutOfBounds(String),
|
SelectionOutOfBounds(String),
|
||||||
@@ -190,6 +205,55 @@ pub enum FormatError {
|
|||||||
DuplicateDatasetName(String),
|
DuplicateDatasetName(String),
|
||||||
/// Integer overflow in size computation (malformed data protection).
|
/// Integer overflow in size computation (malformed data protection).
|
||||||
Overflow(String),
|
Overflow(String),
|
||||||
|
/// An object header that libhdf5 refuses to load (the reason is
|
||||||
|
/// libhdf5's own error text): a misaligned or overrunning message, a
|
||||||
|
/// wrong message count, contradictory message flags, a message of a
|
||||||
|
/// class that cannot be shared flagged shareable, …
|
||||||
|
InvalidObjectHeader(&'static str),
|
||||||
|
/// A datatype message libhdf5 refuses to decode (the reason is
|
||||||
|
/// libhdf5's own error text): size 0, bit fields outside the type,
|
||||||
|
/// an empty enum name, a compound member outside its compound, …
|
||||||
|
InvalidDatatype(String),
|
||||||
|
/// A chunked layout whose chunk dimensions libhdf5 refuses: a zero
|
||||||
|
/// dimension, a rank that does not match the dataspace, an element size
|
||||||
|
/// that is not the datatype's, or a chunk of 4 GiB or more indexed by a
|
||||||
|
/// version-1 B-tree.
|
||||||
|
InvalidChunkDimensions(String),
|
||||||
|
/// The superblock's end-of-file address lies past the end of the file:
|
||||||
|
/// the file was truncated (libhdf5 refuses to open it).
|
||||||
|
TruncatedFile {
|
||||||
|
/// End of file recorded in the superblock (relative to byte 0).
|
||||||
|
stored_eof: u64,
|
||||||
|
/// The file's actual length in bytes.
|
||||||
|
actual_len: u64,
|
||||||
|
},
|
||||||
|
/// A link libhdf5 refuses to list: a symbol-table entry with an empty
|
||||||
|
/// name ("invalid link name"). Listing the group fails, as in libhdf5.
|
||||||
|
InvalidLinkName,
|
||||||
|
/// A dataspace message libhdf5 refuses to decode (the reason is
|
||||||
|
/// libhdf5's own error text): more than 32 dimensions, a rank on a
|
||||||
|
/// scalar or null dataspace, a dimension larger than its maximum.
|
||||||
|
InvalidDataspace(&'static str),
|
||||||
|
/// A dataset whose storage libhdf5 refuses when it opens the dataset
|
||||||
|
/// (the reason is libhdf5's own error text): an element count times
|
||||||
|
/// element size that overflows, contiguous storage past the end of the
|
||||||
|
/// file, compact data of the wrong size.
|
||||||
|
InvalidDatasetStorage(&'static str),
|
||||||
|
/// A superblock extension message libhdf5 refuses to decode when it
|
||||||
|
/// opens the file (the reason is libhdf5's own error text): a File Space
|
||||||
|
/// Info message that runs off its end or has a bad page size, a metadata
|
||||||
|
/// cache image outside the file, …
|
||||||
|
InvalidSuperblockExtension(&'static str),
|
||||||
|
/// A metadata cache image block libhdf5 refuses to load (the reason is
|
||||||
|
/// libhdf5's own error text).
|
||||||
|
InvalidCacheImage(&'static str),
|
||||||
|
/// The [`Storage`](crate::storage::Storage) backend failed to serve a
|
||||||
|
/// read (an I/O or network error, or a short read inside the file).
|
||||||
|
Storage(String),
|
||||||
|
/// The operation still needs the whole file as one slice and the
|
||||||
|
/// [`Storage`](crate::storage::Storage) backend has no contiguous view
|
||||||
|
/// (`as_contiguous()` is `None`); the text names the operation.
|
||||||
|
ContiguousStorageRequired(&'static str),
|
||||||
}
|
}
|
||||||
|
|
||||||
impl fmt::Display for FormatError {
|
impl fmt::Display for FormatError {
|
||||||
@@ -270,6 +334,9 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::InvalidLocalHeapSignature => {
|
FormatError::InvalidLocalHeapSignature => {
|
||||||
write!(f, "invalid local heap signature")
|
write!(f, "invalid local heap signature")
|
||||||
}
|
}
|
||||||
|
FormatError::InvalidLocalHeapFreeList => {
|
||||||
|
write!(f, "bad local heap free list")
|
||||||
|
}
|
||||||
FormatError::InvalidLocalHeapVersion(v) => {
|
FormatError::InvalidLocalHeapVersion(v) => {
|
||||||
write!(f, "invalid local heap version: {v}")
|
write!(f, "invalid local heap version: {v}")
|
||||||
}
|
}
|
||||||
@@ -339,6 +406,16 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::SelectionOutOfBounds(msg) => {
|
FormatError::SelectionOutOfBounds(msg) => {
|
||||||
write!(f, "selection out of bounds: {msg}")
|
write!(f, "selection out of bounds: {msg}")
|
||||||
}
|
}
|
||||||
|
FormatError::UserBlockNotStripped(n) => write!(
|
||||||
|
f,
|
||||||
|
"file has a {n}-byte user block: parse the bytes from the superblock on \
|
||||||
|
(signature::split_user_block)"
|
||||||
|
),
|
||||||
|
FormatError::SharedMessageTargetMissing(t) => write!(
|
||||||
|
f,
|
||||||
|
"shared message reference points at an object header with no message of type \
|
||||||
|
{t:#06x}"
|
||||||
|
),
|
||||||
FormatError::UnresolvedSharedMessage => write!(
|
FormatError::UnresolvedSharedMessage => write!(
|
||||||
f,
|
f,
|
||||||
"message is shared but no file data was available to resolve it"
|
"message is shared but no file data was available to resolve it"
|
||||||
@@ -382,9 +459,17 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::InvalidFilterPipelineVersion(v) => {
|
FormatError::InvalidFilterPipelineVersion(v) => {
|
||||||
write!(f, "invalid filter pipeline version: {v}")
|
write!(f, "invalid filter pipeline version: {v}")
|
||||||
}
|
}
|
||||||
FormatError::UnsupportedFilter(id) => {
|
FormatError::UnsupportedFilter(id) => match crate::filter_registry::known_filter(*id) {
|
||||||
write!(f, "unsupported filter: {id}")
|
Some((name, Some(feature))) => write!(
|
||||||
}
|
f,
|
||||||
|
"unsupported filter: {id} ({name}; this build lacks the `{feature}` feature)"
|
||||||
|
),
|
||||||
|
Some((name, None)) => write!(
|
||||||
|
f,
|
||||||
|
"unsupported filter: {id} ({name}, not implemented by clawhdf5)"
|
||||||
|
),
|
||||||
|
None => write!(f, "unsupported filter: {id}"),
|
||||||
|
},
|
||||||
FormatError::FilterError(msg) => {
|
FormatError::FilterError(msg) => {
|
||||||
write!(f, "filter error: {msg}")
|
write!(f, "filter error: {msg}")
|
||||||
}
|
}
|
||||||
@@ -421,6 +506,50 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::Overflow(msg) => {
|
FormatError::Overflow(msg) => {
|
||||||
write!(f, "integer overflow: {msg}")
|
write!(f, "integer overflow: {msg}")
|
||||||
}
|
}
|
||||||
|
FormatError::InvalidObjectHeader(why) => {
|
||||||
|
write!(f, "corrupt object header: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDatatype(why) => {
|
||||||
|
write!(f, "invalid datatype: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidChunkDimensions(why) => {
|
||||||
|
write!(f, "invalid chunk dimensions: {why}")
|
||||||
|
}
|
||||||
|
FormatError::TruncatedFile {
|
||||||
|
stored_eof,
|
||||||
|
actual_len,
|
||||||
|
} => {
|
||||||
|
write!(
|
||||||
|
f,
|
||||||
|
"truncated file: the superblock records end of file {stored_eof}, \
|
||||||
|
but the file is {actual_len} bytes"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
FormatError::InvalidLinkName => {
|
||||||
|
write!(f, "invalid link name: a group entry has an empty name")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDataspace(why) => {
|
||||||
|
write!(f, "invalid dataspace: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidDatasetStorage(why) => {
|
||||||
|
write!(f, "invalid dataset storage: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidSuperblockExtension(why) => {
|
||||||
|
write!(f, "invalid superblock extension: {why}")
|
||||||
|
}
|
||||||
|
FormatError::InvalidCacheImage(why) => {
|
||||||
|
write!(f, "invalid metadata cache image: {why}")
|
||||||
|
}
|
||||||
|
FormatError::Storage(why) => {
|
||||||
|
write!(f, "storage read failed: {why}")
|
||||||
|
}
|
||||||
|
FormatError::ContiguousStorageRequired(what) => {
|
||||||
|
write!(
|
||||||
|
f,
|
||||||
|
"{what} needs the whole file in memory, which this storage backend does \
|
||||||
|
not provide"
|
||||||
|
)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -9,18 +9,23 @@ extern crate alloc;
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{format, vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
|
use crate::chunk_grid::ChunkGrid;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{PAGED_BLOCK_ONE_READ_MAX, Storage, Window, read_exact_at};
|
||||||
|
|
||||||
/// Verify the Jenkins lookup3 checksum stored immediately after
|
/// Verify the Jenkins lookup3 checksum stored immediately after
|
||||||
/// `data[start..end]`, as every Extensible Array structure carries one.
|
/// `data[start..end]`, as every Extensible Array structure carries one. `w`
|
||||||
|
/// is a window of the file and `start`/`end` are relative to it.
|
||||||
///
|
///
|
||||||
/// A corrupt chunk index yields addresses pointing at the wrong bytes, so a
|
/// A corrupt chunk index yields addresses pointing at the wrong bytes, so a
|
||||||
/// mismatch is an error: otherwise the damage surfaces as plausible data read
|
/// mismatch is an error: otherwise the damage surfaces as plausible data read
|
||||||
/// from the wrong chunk.
|
/// from the wrong chunk.
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatError> {
|
fn verify_checksum(w: &Window<'_>, start: usize, end: usize) -> Result<(), FormatError> {
|
||||||
ensure_len(data, end, 4)?;
|
w.ensure(end, 4)?;
|
||||||
|
let data: &[u8] = &w.bytes;
|
||||||
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
@@ -33,7 +38,7 @@ fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatEr
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "checksum"))]
|
#[cfg(not(feature = "checksum"))]
|
||||||
fn verify_checksum(_data: &[u8], _start: usize, _end: usize) -> Result<(), FormatError> {
|
fn verify_checksum(_w: &Window<'_>, _start: usize, _end: usize) -> Result<(), FormatError> {
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -79,19 +84,6 @@ fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
|
||||||
if offset
|
|
||||||
.checked_add(needed)
|
|
||||||
.is_none_or(|end| end > data.len())
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: offset.saturating_add(needed),
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn is_undefined_addr(addr: u64, offset_size: u8) -> bool {
|
fn is_undefined_addr(addr: u64, offset_size: u8) -> bool {
|
||||||
match offset_size {
|
match offset_size {
|
||||||
2 => addr == 0xFFFF,
|
2 => addr == 0xFFFF,
|
||||||
@@ -129,6 +121,16 @@ impl ExtensibleArrayHeader {
|
|||||||
offset: usize,
|
offset: usize,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one read of the header.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Self, FormatError> {
|
) -> Result<Self, FormatError> {
|
||||||
// EAHD: signature(4) + version(1) + client_id(1) + element_size(1) +
|
// EAHD: signature(4) + version(1) + client_id(1) + element_size(1) +
|
||||||
// max_nelmts_bits(1) + idx_blk_elmts(1) + min_dblk_nelmts(1) +
|
// max_nelmts_bits(1) + idx_blk_elmts(1) + min_dblk_nelmts(1) +
|
||||||
@@ -136,9 +138,10 @@ impl ExtensibleArrayHeader {
|
|||||||
// 6 stats fields (each length_size) + index_block_address(offset_size) + checksum(4)
|
// 6 stats fields (each length_size) + index_block_address(offset_size) + checksum(4)
|
||||||
let min_size =
|
let min_size =
|
||||||
4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + offset_size as usize + 4;
|
4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + offset_size as usize + 4;
|
||||||
ensure_len(file_data, offset, min_size)?;
|
let w = Window::read(file, offset, min_size)?;
|
||||||
|
w.ensure(0, min_size)?;
|
||||||
|
|
||||||
let d = &file_data[offset..];
|
let d: &[u8] = &w.bytes;
|
||||||
if &d[0..4] != b"EAHD" {
|
if &d[0..4] != b"EAHD" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Extensible Array header signature".into(),
|
"invalid Extensible Array header signature".into(),
|
||||||
@@ -171,7 +174,7 @@ impl ExtensibleArrayHeader {
|
|||||||
pos += ls; // skip max_idx_set (6th stats field)
|
pos += ls; // skip max_idx_set (6th stats field)
|
||||||
let index_block_address = read_offset(d, pos, offset_size)?;
|
let index_block_address = read_offset(d, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
verify_checksum(file_data, offset, offset + pos)?;
|
verify_checksum(&w, 0, pos)?;
|
||||||
|
|
||||||
Ok(ExtensibleArrayHeader {
|
Ok(ExtensibleArrayHeader {
|
||||||
client_id,
|
client_id,
|
||||||
@@ -192,35 +195,33 @@ impl ExtensibleArrayHeader {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Read a single element from the extensible array element data.
|
/// Read a single element at offset `pos` of the window `w`.
|
||||||
/// Returns (chunk_info, bytes_consumed) or None if unallocated.
|
/// Returns (chunk_info, bytes_consumed) or None if unallocated.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn read_element(
|
fn read_element(
|
||||||
data: &[u8],
|
w: &Window<'_>,
|
||||||
pos: usize,
|
pos: usize,
|
||||||
client_id: u8,
|
client_id: u8,
|
||||||
element_size: u8,
|
element_size: u8,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
linear_index: usize,
|
linear_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Result<(Option<ChunkInfo>, usize), FormatError> {
|
) -> Result<(Option<ChunkInfo>, usize), FormatError> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
|
let data: &[u8] = &w.bytes;
|
||||||
|
|
||||||
if client_id == 0 {
|
if client_id == 0 {
|
||||||
// Non-filtered: just address
|
// Non-filtered: just address
|
||||||
if pos + os > data.len() {
|
w.ensure(pos, os)?;
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: pos + os,
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if is_undefined(data, pos, offset_size) {
|
if is_undefined(data, pos, offset_size) {
|
||||||
return Ok((None, os));
|
return Ok((None, os));
|
||||||
}
|
}
|
||||||
let address = read_offset(data, pos, offset_size)?;
|
let address = read_offset(data, pos, offset_size)?;
|
||||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
// A slot beyond the current extent is ignored, as the library does.
|
||||||
|
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||||
|
return Ok((None, os));
|
||||||
|
};
|
||||||
Ok((
|
Ok((
|
||||||
Some(ChunkInfo {
|
Some(ChunkInfo {
|
||||||
chunk_size: chunk_byte_size as u32,
|
chunk_size: chunk_byte_size as u32,
|
||||||
@@ -240,15 +241,7 @@ fn read_element(
|
|||||||
}
|
}
|
||||||
let chunk_size_bytes = es - os - 4;
|
let chunk_size_bytes = es - os - 4;
|
||||||
let elem_total = os + chunk_size_bytes + 4;
|
let elem_total = os + chunk_size_bytes + 4;
|
||||||
if pos
|
w.ensure(pos, elem_total)?;
|
||||||
.checked_add(elem_total)
|
|
||||||
.is_none_or(|end| end > data.len())
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: pos.saturating_add(elem_total),
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if is_undefined(data, pos, offset_size) {
|
if is_undefined(data, pos, offset_size) {
|
||||||
return Ok((None, elem_total));
|
return Ok((None, elem_total));
|
||||||
}
|
}
|
||||||
@@ -261,7 +254,9 @@ fn read_element(
|
|||||||
data[fm_off + 2],
|
data[fm_off + 2],
|
||||||
data[fm_off + 3],
|
data[fm_off + 3],
|
||||||
]);
|
]);
|
||||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||||
|
return Ok((None, elem_total));
|
||||||
|
};
|
||||||
Ok((
|
Ok((
|
||||||
Some(ChunkInfo {
|
Some(ChunkInfo {
|
||||||
chunk_size: chunk_size as u32,
|
chunk_size: chunk_size as u32,
|
||||||
@@ -274,27 +269,6 @@ fn read_element(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
|
||||||
fn index_to_chunk_offsets(
|
|
||||||
index: usize,
|
|
||||||
num_chunks_per_dim: &[u64],
|
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Vec<u64> {
|
|
||||||
let rank = num_chunks_per_dim.len();
|
|
||||||
let mut offsets = vec![0u64; rank];
|
|
||||||
let mut remaining = index as u64;
|
|
||||||
for d in (0..rank).rev() {
|
|
||||||
let nchunks = num_chunks_per_dim[d];
|
|
||||||
if nchunks == 0 {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
let chunk_idx = remaining % nchunks;
|
|
||||||
remaining /= nchunks;
|
|
||||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
|
||||||
}
|
|
||||||
offsets
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Collect elements from a data block at the given offset.
|
/// Collect elements from a data block at the given offset.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
/// Layout of super block `u`, per the HDF5 spec: the number of data blocks it
|
/// Layout of super block `u`, per the HDF5 spec: the number of data blocks it
|
||||||
@@ -331,37 +305,43 @@ fn page_nelmts(header: &ExtensibleArrayHeader) -> Option<usize> {
|
|||||||
/// paged. The bitmap lives in the super block, not here — a paged data block
|
/// paged. The bitmap lives in the super block, not here — a paged data block
|
||||||
/// stores only its prefix, then one slot per page.
|
/// stores only its prefix, then one slot per page.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn read_data_block_elements(
|
fn read_data_block_elements<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
db_offset: usize,
|
db_offset: u64,
|
||||||
nelmts: usize,
|
nelmts: usize,
|
||||||
header: &ExtensibleArrayHeader,
|
header: &ExtensibleArrayHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
start_index: usize,
|
start_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
page_init: &[u8],
|
page_init: &[u8],
|
||||||
first_page: usize,
|
first_page: usize,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
// EADB: signature(4) + version(1) + client_id(1) + header_address(offset_size)
|
// EADB: signature(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||||
// + block offset(arr_off_size)
|
// + block offset(arr_off_size)
|
||||||
let db_header_size = 4 + 1 + 1 + offset_size as usize + arr_off_size(header);
|
let db_header_size = 4 + 1 + 1 + offset_size as usize + arr_off_size(header);
|
||||||
ensure_len(file_data, db_offset, db_header_size)?;
|
let prefix = read_exact_at(file, db_offset, db_header_size)?;
|
||||||
|
|
||||||
if &file_data[db_offset..db_offset + 4] != b"EADB" {
|
if &prefix[0..4] != b"EADB" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Extensible Array data block signature".into(),
|
"invalid Extensible Array data block signature".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut pos = db_offset + db_header_size;
|
// Positions below are relative to the data block.
|
||||||
|
let mut pos = db_header_size;
|
||||||
let page = page_nelmts(header).ok_or_else(|| {
|
let page = page_nelmts(header).ok_or_else(|| {
|
||||||
FormatError::Overflow("Extensible Array page element count overflows usize".into())
|
FormatError::Overflow("Extensible Array page element count overflows usize".into())
|
||||||
})?;
|
})?;
|
||||||
|
let elem_bytes = if header.client_id == 0 {
|
||||||
|
offset_size as usize
|
||||||
|
} else {
|
||||||
|
header.element_size as usize
|
||||||
|
};
|
||||||
|
|
||||||
let mut chunks = Vec::new();
|
let mut chunks = Vec::new();
|
||||||
let read_run = |from: usize,
|
let read_run = |w: &Window<'_>,
|
||||||
|
from: usize,
|
||||||
count: usize,
|
count: usize,
|
||||||
first_index: usize,
|
first_index: usize,
|
||||||
chunks: &mut Vec<ChunkInfo>|
|
chunks: &mut Vec<ChunkInfo>|
|
||||||
@@ -369,15 +349,14 @@ fn read_data_block_elements(
|
|||||||
let mut p = from;
|
let mut p = from;
|
||||||
for i in 0..count {
|
for i in 0..count {
|
||||||
let (info, consumed) = read_element(
|
let (info, consumed) = read_element(
|
||||||
file_data,
|
w,
|
||||||
p,
|
p,
|
||||||
header.client_id,
|
header.client_id,
|
||||||
header.element_size,
|
header.element_size,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
first_index + i,
|
first_index + i,
|
||||||
num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?;
|
)?;
|
||||||
if let Some(ci) = info {
|
if let Some(ci) = info {
|
||||||
chunks.push(ci);
|
chunks.push(ci);
|
||||||
@@ -388,18 +367,19 @@ fn read_data_block_elements(
|
|||||||
};
|
};
|
||||||
|
|
||||||
if nelmts <= page {
|
if nelmts <= page {
|
||||||
// Prefix and elements are covered by one checksum.
|
// Prefix and elements are covered by one checksum. One window holds
|
||||||
let elem_bytes = if header.client_id == 0 {
|
// all of it (or ends at the end of the file), so its bounds checks
|
||||||
offset_size as usize
|
// are the whole-file ones.
|
||||||
} else {
|
|
||||||
header.element_size as usize
|
|
||||||
};
|
|
||||||
let end = nelmts
|
let end = nelmts
|
||||||
.checked_mul(elem_bytes)
|
.checked_mul(elem_bytes)
|
||||||
.and_then(|b| pos.checked_add(b))
|
.and_then(|b| pos.checked_add(b))
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array data block span".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array data block span".into()))?;
|
||||||
verify_checksum(file_data, db_offset, end)?;
|
// The checksum's bounds check comes first: make it before reading.
|
||||||
read_run(pos, nelmts, start_index, &mut chunks)?;
|
#[cfg(feature = "checksum")]
|
||||||
|
Window::check_extent(file, db_offset, end, 4)?;
|
||||||
|
let w = Window::read(file, db_offset, end.saturating_add(4))?;
|
||||||
|
verify_checksum(&w, 0, end)?;
|
||||||
|
read_run(&w, pos, nelmts, start_index, &mut chunks)?;
|
||||||
return Ok(chunks);
|
return Ok(chunks);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -407,18 +387,32 @@ fn read_data_block_elements(
|
|||||||
// each holding `page` elements followed by a checksum. Pages whose bit is
|
// each holding `page` elements followed by a checksum. Pages whose bit is
|
||||||
// clear were never written; their slot still occupies the file, so stride
|
// clear were never written; their slot still occupies the file, so stride
|
||||||
// over it rather than reading zeros as addresses.
|
// over it rather than reading zeros as addresses.
|
||||||
verify_checksum(file_data, db_offset, pos)?;
|
let npages = nelmts.div_ceil(page);
|
||||||
pos += 4;
|
// The whole data block in one window when it is small: every position
|
||||||
let elem_bytes = if header.client_id == 0 {
|
// checked below lies inside it (or past the end of the file). A larger
|
||||||
offset_size as usize
|
// block is read as its prefix, then each page in use on its own.
|
||||||
|
let block_len = pos
|
||||||
|
.saturating_add(4)
|
||||||
|
.saturating_add(npages.saturating_mul(page.saturating_mul(elem_bytes).saturating_add(4)));
|
||||||
|
let whole = if block_len <= PAGED_BLOCK_ONE_READ_MAX {
|
||||||
|
Some(Window::read(file, db_offset, block_len)?)
|
||||||
} else {
|
} else {
|
||||||
header.element_size as usize
|
None
|
||||||
};
|
};
|
||||||
|
let head_w;
|
||||||
|
let head = match &whole {
|
||||||
|
Some(w) => w,
|
||||||
|
None => {
|
||||||
|
head_w = Window::read(file, db_offset, pos + 4)?;
|
||||||
|
&head_w
|
||||||
|
}
|
||||||
|
};
|
||||||
|
verify_checksum(head, 0, pos)?;
|
||||||
|
pos += 4;
|
||||||
let page_stride = page
|
let page_stride = page
|
||||||
.checked_mul(elem_bytes)
|
.checked_mul(elem_bytes)
|
||||||
.and_then(|b| b.checked_add(4))
|
.and_then(|b| b.checked_add(4))
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array page stride".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array page stride".into()))?;
|
||||||
let npages = nelmts.div_ceil(page);
|
|
||||||
for p in 0..npages {
|
for p in 0..npages {
|
||||||
// One bit per page across the whole super block, packed contiguously
|
// One bit per page across the whole super block, packed contiguously
|
||||||
// and MSB-first within each byte, as H5VM_bit_get reads it.
|
// and MSB-first within each byte, as H5VM_bit_get reads it.
|
||||||
@@ -428,10 +422,20 @@ fn read_data_block_elements(
|
|||||||
.is_some_and(|byte| byte & (0x80 >> (bit % 8)) != 0);
|
.is_some_and(|byte| byte & (0x80 >> (bit % 8)) != 0);
|
||||||
if initialised {
|
if initialised {
|
||||||
let count = core::cmp::min(page, nelmts - p * page);
|
let count = core::cmp::min(page, nelmts - p * page);
|
||||||
|
// `w` holds the page from `base` on (positions below are
|
||||||
|
// relative to it, and `pos` to the data block).
|
||||||
|
let page_w;
|
||||||
|
let (w, base) = match &whole {
|
||||||
|
Some(w) => (w, 0),
|
||||||
|
None => {
|
||||||
|
page_w = Window::read(file, db_offset.saturating_add(pos as u64), page_stride)?;
|
||||||
|
(&page_w, pos)
|
||||||
|
}
|
||||||
|
};
|
||||||
// Each page carries its own checksum, over a full page's worth of
|
// Each page carries its own checksum, over a full page's worth of
|
||||||
// slots even when the last one holds fewer live elements.
|
// slots even when the last one holds fewer live elements.
|
||||||
verify_checksum(file_data, pos, pos + page * elem_bytes)?;
|
verify_checksum(w, pos - base, pos - base + page * elem_bytes)?;
|
||||||
read_run(pos, count, start_index + p * page, &mut chunks)?;
|
read_run(w, pos - base, count, start_index + p * page, &mut chunks)?;
|
||||||
}
|
}
|
||||||
pos = pos
|
pos = pos
|
||||||
.checked_add(page_stride)
|
.checked_add(page_stride)
|
||||||
@@ -449,25 +453,45 @@ pub fn read_extensible_array_chunks(
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &ExtensibleArrayHeader,
|
header: &ExtensibleArrayHeader,
|
||||||
dataset_dims: &[u64],
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dimensions: &[u32],
|
||||||
|
element_size: u32,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
|
read_extensible_array_chunks_in(
|
||||||
|
&file_data,
|
||||||
|
header,
|
||||||
|
dataset_dims,
|
||||||
|
max_dims,
|
||||||
|
chunk_dimensions,
|
||||||
|
element_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`read_extensible_array_chunks`] over any [`Storage`]: one read of the
|
||||||
|
/// index block's prefix, one of the whole index block, and the same for
|
||||||
|
/// every super block and data block it references.
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
pub fn read_extensible_array_chunks_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &ExtensibleArrayHeader,
|
||||||
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
chunk_dimensions: &[u32],
|
chunk_dimensions: &[u32],
|
||||||
element_size: u32,
|
element_size: u32,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let rank = chunk_dimensions.len();
|
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
|
|
||||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
// Linear indexes follow the maximum dimensions, with the unlimited
|
||||||
for d in 0..rank {
|
// dimension swizzled to the slowest position (see `chunk_grid`).
|
||||||
let ch_dim = chunk_dimensions[d] as u64;
|
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||||
if ch_dim == 0 {
|
let grid = ChunkGrid::extensible_array(dataset_dims, max_dims, &dims_u64)?;
|
||||||
return Err(FormatError::ChunkedReadError(
|
let grid = &grid;
|
||||||
"chunk dimension is zero".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let ds_dim = dataset_dims[d];
|
|
||||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
|
||||||
}
|
|
||||||
|
|
||||||
let chunk_byte_size: u64 =
|
let chunk_byte_size: u64 =
|
||||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||||
@@ -475,19 +499,20 @@ pub fn read_extensible_array_chunks(
|
|||||||
// Parse index block (EAIB): signature(4) + version(1) + client_id(1)
|
// Parse index block (EAIB): signature(4) + version(1) + client_id(1)
|
||||||
// + header address(offset_size), then the inline elements, then the
|
// + header address(offset_size), then the inline elements, then the
|
||||||
// direct data block addresses, then the super block addresses.
|
// direct data block addresses, then the super block addresses.
|
||||||
let ib_offset = header.index_block_address as usize;
|
// Positions below are relative to the index block.
|
||||||
|
let ib_offset = header.index_block_address;
|
||||||
let ib_header_size = 4 + 1 + 1 + os;
|
let ib_header_size = 4 + 1 + 1 + os;
|
||||||
ensure_len(file_data, ib_offset, ib_header_size)?;
|
let prefix = read_exact_at(file, ib_offset, ib_header_size)?;
|
||||||
|
|
||||||
if &file_data[ib_offset..ib_offset + 4] != b"EAIB" {
|
if &prefix[0..4] != b"EAIB" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Extensible Array index block signature".into(),
|
"invalid Extensible Array index block signature".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
let mut pos = ib_offset + ib_header_size;
|
let mut pos = ib_header_size;
|
||||||
|
|
||||||
let mut chunks = Vec::new();
|
let mut chunks = Vec::new();
|
||||||
let total_elements = header.num_elements as usize;
|
let total_elements = to_usize(header.num_elements)?;
|
||||||
|
|
||||||
let dmin = header.min_dblk_nelmts as usize;
|
let dmin = header.min_dblk_nelmts as usize;
|
||||||
if dmin == 0 || !dmin.is_power_of_two() {
|
if dmin == 0 || !dmin.is_power_of_two() {
|
||||||
@@ -544,21 +569,26 @@ pub fn read_extensible_array_chunks(
|
|||||||
.and_then(|n| n.checked_mul(os).and_then(|b| p.checked_add(b)))
|
.and_then(|n| n.checked_mul(os).and_then(|b| p.checked_add(b)))
|
||||||
})
|
})
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array index block span".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array index block span".into()))?;
|
||||||
verify_checksum(file_data, ib_offset, ib_end)?;
|
// The whole index block in one window: every position read below is
|
||||||
|
// before `ib_end`.
|
||||||
|
// The checksum's bounds check comes first: make it before reading.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
Window::check_extent(file, ib_offset, ib_end, 4)?;
|
||||||
|
let w = Window::read(file, ib_offset, ib_end.saturating_add(4))?;
|
||||||
|
verify_checksum(&w, 0, ib_end)?;
|
||||||
|
|
||||||
// 1. Elements stored inline in the index block.
|
// 1. Elements stored inline in the index block.
|
||||||
let n_inline = (header.idx_blk_elmts as usize).min(total_elements);
|
let n_inline = (header.idx_blk_elmts as usize).min(total_elements);
|
||||||
for i in 0..n_inline {
|
for i in 0..n_inline {
|
||||||
let (info, consumed) = read_element(
|
let (info, consumed) = read_element(
|
||||||
file_data,
|
&w,
|
||||||
pos,
|
pos,
|
||||||
header.client_id,
|
header.client_id,
|
||||||
header.element_size,
|
header.element_size,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
i,
|
i,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?;
|
)?;
|
||||||
if let Some(ci) = info {
|
if let Some(ci) = info {
|
||||||
chunks.push(ci);
|
chunks.push(ci);
|
||||||
@@ -575,8 +605,8 @@ pub fn read_extensible_array_chunks(
|
|||||||
if global_index >= total_elements {
|
if global_index >= total_elements {
|
||||||
return Ok(chunks);
|
return Ok(chunks);
|
||||||
}
|
}
|
||||||
ensure_len(file_data, pos, os)?;
|
w.ensure(pos, os)?;
|
||||||
let addr = read_offset(file_data, pos, offset_size)?;
|
let addr = read_offset(&w.bytes, pos, offset_size)?;
|
||||||
pos += os;
|
pos += os;
|
||||||
if !is_undefined_addr(addr, offset_size) {
|
if !is_undefined_addr(addr, offset_size) {
|
||||||
if dblk_nelmts > page_nelmts(header).unwrap_or(usize::MAX) {
|
if dblk_nelmts > page_nelmts(header).unwrap_or(usize::MAX) {
|
||||||
@@ -587,15 +617,14 @@ pub fn read_extensible_array_chunks(
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
chunks.extend(read_data_block_elements(
|
chunks.extend(read_data_block_elements(
|
||||||
file_data,
|
file,
|
||||||
addr as usize,
|
addr,
|
||||||
dblk_nelmts,
|
dblk_nelmts,
|
||||||
header,
|
header,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_index,
|
global_index,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
&[],
|
&[],
|
||||||
0,
|
0,
|
||||||
)?);
|
)?);
|
||||||
@@ -609,24 +638,23 @@ pub fn read_extensible_array_chunks(
|
|||||||
if global_index >= total_elements {
|
if global_index >= total_elements {
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
ensure_len(file_data, pos, os)?;
|
w.ensure(pos, os)?;
|
||||||
let sb_addr = read_offset(file_data, pos, offset_size)?;
|
let sb_addr = read_offset(&w.bytes, pos, offset_size)?;
|
||||||
pos += os;
|
pos += os;
|
||||||
let (ndblks, dblk_nelmts) = sblk_info(u, dmin).ok_or_else(|| {
|
let (ndblks, dblk_nelmts) = sblk_info(u, dmin).ok_or_else(|| {
|
||||||
FormatError::Overflow("Extensible Array super block layout overflows usize".into())
|
FormatError::Overflow("Extensible Array super block layout overflows usize".into())
|
||||||
})?;
|
})?;
|
||||||
if !is_undefined_addr(sb_addr, offset_size) {
|
if !is_undefined_addr(sb_addr, offset_size) {
|
||||||
chunks.extend(read_super_block(
|
chunks.extend(read_super_block(
|
||||||
file_data,
|
file,
|
||||||
sb_addr as usize,
|
sb_addr,
|
||||||
ndblks,
|
ndblks,
|
||||||
dblk_nelmts,
|
dblk_nelmts,
|
||||||
header,
|
header,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_index,
|
global_index,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?);
|
)?);
|
||||||
}
|
}
|
||||||
global_index =
|
global_index =
|
||||||
@@ -644,23 +672,22 @@ pub fn read_extensible_array_chunks(
|
|||||||
/// + block offset + the page-init bitmap for every data block it owns
|
/// + block offset + the page-init bitmap for every data block it owns
|
||||||
/// + one address per data block + checksum.
|
/// + one address per data block + checksum.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn read_super_block(
|
fn read_super_block<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file: &S,
|
||||||
sb_offset: usize,
|
sb_offset: u64,
|
||||||
ndblks: usize,
|
ndblks: usize,
|
||||||
dblk_nelmts: usize,
|
dblk_nelmts: usize,
|
||||||
header: &ExtensibleArrayHeader,
|
header: &ExtensibleArrayHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
start_index: usize,
|
start_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let sb_header_size = 4 + 1 + 1 + os + arr_off_size(header);
|
let sb_header_size = 4 + 1 + 1 + os + arr_off_size(header);
|
||||||
ensure_len(file_data, sb_offset, sb_header_size)?;
|
let prefix = read_exact_at(file, sb_offset, sb_header_size)?;
|
||||||
|
|
||||||
if &file_data[sb_offset..sb_offset + 4] != b"EASB" {
|
if &prefix[0..4] != b"EASB" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Extensible Array super block signature".into(),
|
"invalid Extensible Array super block signature".into(),
|
||||||
));
|
));
|
||||||
@@ -682,36 +709,44 @@ fn read_super_block(
|
|||||||
let bitmap_bytes = per_dblk_bitmap
|
let bitmap_bytes = per_dblk_bitmap
|
||||||
.checked_mul(ndblks)
|
.checked_mul(ndblks)
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array page bitmap size".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array page bitmap size".into()))?;
|
||||||
let bitmap_start = sb_offset + sb_header_size;
|
// Positions below are relative to the super block, whose bytes (up to
|
||||||
ensure_len(file_data, bitmap_start, bitmap_bytes)?;
|
// its checksum) are all in one window.
|
||||||
let bitmap = &file_data[bitmap_start..bitmap_start + bitmap_bytes];
|
let bitmap_start = sb_header_size;
|
||||||
|
// The bitmap's bounds check, then (with checksums) the checksum's, come
|
||||||
|
// before anything else is read from the block: make them before reading
|
||||||
|
// it, so size fields stretching it past the end of the file cost no read.
|
||||||
|
Window::check_extent(file, sb_offset, bitmap_start, bitmap_bytes)?;
|
||||||
let mut pos = bitmap_start + bitmap_bytes;
|
let mut pos = bitmap_start + bitmap_bytes;
|
||||||
let mut chunks = Vec::new();
|
|
||||||
let mut global_idx = start_index;
|
|
||||||
|
|
||||||
// One checksum covers the prefix, the bitmap and every data block address.
|
// One checksum covers the prefix, the bitmap and every data block address.
|
||||||
let sb_end = ndblks
|
let sb_end = ndblks
|
||||||
.checked_mul(os)
|
.checked_mul(os)
|
||||||
.and_then(|b| pos.checked_add(b))
|
.and_then(|b| pos.checked_add(b))
|
||||||
.ok_or_else(|| FormatError::Overflow("Extensible Array super block span".into()))?;
|
.ok_or_else(|| FormatError::Overflow("Extensible Array super block span".into()))?;
|
||||||
verify_checksum(file_data, sb_offset, sb_end)?;
|
#[cfg(feature = "checksum")]
|
||||||
|
Window::check_extent(file, sb_offset, sb_end, 4)?;
|
||||||
|
let w = Window::read(file, sb_offset, sb_end.saturating_add(4))?;
|
||||||
|
w.ensure(bitmap_start, bitmap_bytes)?;
|
||||||
|
let bitmap = &w.bytes[bitmap_start..bitmap_start + bitmap_bytes];
|
||||||
|
|
||||||
|
let mut chunks = Vec::new();
|
||||||
|
let mut global_idx = start_index;
|
||||||
|
verify_checksum(&w, 0, sb_end)?;
|
||||||
|
|
||||||
for i in 0..ndblks {
|
for i in 0..ndblks {
|
||||||
ensure_len(file_data, pos, os)?;
|
w.ensure(pos, os)?;
|
||||||
let addr = read_offset(file_data, pos, offset_size)?;
|
let addr = read_offset(&w.bytes, pos, offset_size)?;
|
||||||
pos += os;
|
pos += os;
|
||||||
if !is_undefined_addr(addr, offset_size) {
|
if !is_undefined_addr(addr, offset_size) {
|
||||||
chunks.extend(read_data_block_elements(
|
chunks.extend(read_data_block_elements(
|
||||||
file_data,
|
file,
|
||||||
addr as usize,
|
addr,
|
||||||
dblk_nelmts,
|
dblk_nelmts,
|
||||||
header,
|
header,
|
||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_idx,
|
global_idx,
|
||||||
num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
bitmap,
|
bitmap,
|
||||||
i * npages,
|
i * npages,
|
||||||
)?);
|
)?);
|
||||||
@@ -735,35 +770,18 @@ mod tests {
|
|||||||
}
|
}
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_1d() {
|
fn index_to_offsets_1d() {
|
||||||
let num_chunks = vec![5u64];
|
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||||
let chunk_dims = vec![20u32];
|
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![20]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
|
||||||
vec![80]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_2d() {
|
fn index_to_offsets_2d() {
|
||||||
let num_chunks = vec![3u64, 2];
|
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||||
let chunk_dims = vec![4u32, 3];
|
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||||
vec![0, 0]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![0, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 0]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -830,7 +848,7 @@ mod tests {
|
|||||||
index_block_address: (usize::MAX - 4) as u64,
|
index_block_address: (usize::MAX - 4) as u64,
|
||||||
};
|
};
|
||||||
let buf = vec![0u8; 64];
|
let buf = vec![0u8; 64];
|
||||||
let r = read_extensible_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_extensible_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -913,9 +931,17 @@ mod tests {
|
|||||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
||||||
let ds_dims = vec![40u64]; // 2 chunks × 20 elements
|
let ds_dims = vec![40u64]; // 2 chunks × 20 elements
|
||||||
let chunk_dims = vec![20u32];
|
let chunk_dims = vec![20u32];
|
||||||
let chunks =
|
let chunks = read_extensible_array_chunks(
|
||||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
&file_data,
|
||||||
.unwrap();
|
&header,
|
||||||
|
&ds_dims,
|
||||||
|
None,
|
||||||
|
&chunk_dims,
|
||||||
|
8,
|
||||||
|
os,
|
||||||
|
ls,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
assert_eq!(chunks.len(), 2);
|
assert_eq!(chunks.len(), 2);
|
||||||
assert_eq!(chunks[0].address, base_addr);
|
assert_eq!(chunks[0].address, base_addr);
|
||||||
@@ -925,11 +951,11 @@ mod tests {
|
|||||||
assert_eq!(chunks[1].offsets, vec![20]);
|
assert_eq!(chunks[1].offsets, vec![20]);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build a synthetic EA with inline elements + one direct data block.
|
/// A synthetic EA with inline elements + one direct data block: the
|
||||||
#[test]
|
/// file, with the header at 0x100 (8-byte offsets and lengths, 4 chunks
|
||||||
fn read_inline_plus_data_blocks() {
|
/// of 10 elements from 0x1000 on).
|
||||||
|
fn build_inline_plus_data_blocks() -> Vec<u8> {
|
||||||
let os: u8 = 8;
|
let os: u8 = 8;
|
||||||
let ls: u8 = 8;
|
|
||||||
let osv = os as usize;
|
let osv = os as usize;
|
||||||
let chunk_byte_size = 10u64 * 8; // 10 elements × 8 bytes
|
let chunk_byte_size = 10u64 * 8; // 10 elements × 8 bytes
|
||||||
let idx_blk_elmts = 2u8;
|
let idx_blk_elmts = 2u8;
|
||||||
@@ -1019,13 +1045,30 @@ mod tests {
|
|||||||
dbpos += osv;
|
dbpos += osv;
|
||||||
}
|
}
|
||||||
stamp_checksum(&mut file_data, aedb_offset, dbpos);
|
stamp_checksum(&mut file_data, aedb_offset, dbpos);
|
||||||
|
file_data
|
||||||
|
}
|
||||||
|
|
||||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
/// Build a synthetic EA with inline elements + one direct data block.
|
||||||
|
#[test]
|
||||||
|
fn read_inline_plus_data_blocks() {
|
||||||
|
let (os, ls) = (8u8, 8u8);
|
||||||
|
let chunk_byte_size = 10u64 * 8;
|
||||||
|
let base_addr = 0x1000u64;
|
||||||
|
let file_data = build_inline_plus_data_blocks();
|
||||||
|
let header = ExtensibleArrayHeader::parse(&file_data, 0x100, os, ls).unwrap();
|
||||||
let ds_dims = vec![40u64];
|
let ds_dims = vec![40u64];
|
||||||
let chunk_dims = vec![10u32];
|
let chunk_dims = vec![10u32];
|
||||||
let chunks =
|
let chunks = read_extensible_array_chunks(
|
||||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
&file_data,
|
||||||
.unwrap();
|
&header,
|
||||||
|
&ds_dims,
|
||||||
|
None,
|
||||||
|
&chunk_dims,
|
||||||
|
8,
|
||||||
|
os,
|
||||||
|
ls,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
assert_eq!(chunks.len(), 4);
|
assert_eq!(chunks.len(), 4);
|
||||||
for (i, c) in chunks.iter().enumerate() {
|
for (i, c) in chunks.iter().enumerate() {
|
||||||
@@ -1047,10 +1090,9 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn read_element_unallocated() {
|
fn read_element_unallocated() {
|
||||||
let data = vec![0xFFu8; 16];
|
let data = vec![0xFFu8; 16];
|
||||||
let num_chunks = vec![5u64];
|
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||||
let chunk_dims = vec![10u32];
|
|
||||||
let (info, consumed) =
|
let (info, consumed) =
|
||||||
read_element(&data, 0, 0, 8, 8, 80, 0, &num_chunks, &chunk_dims).unwrap();
|
read_element(&Window::whole(&data), 0, 0, 8, 8, 80, 0, &grid).unwrap();
|
||||||
assert!(info.is_none());
|
assert!(info.is_none());
|
||||||
assert_eq!(consumed, 8);
|
assert_eq!(consumed, 8);
|
||||||
}
|
}
|
||||||
@@ -1069,18 +1111,16 @@ mod tests {
|
|||||||
// Filter mask
|
// Filter mask
|
||||||
data[12..16].copy_from_slice(&0u32.to_le_bytes());
|
data[12..16].copy_from_slice(&0u32.to_le_bytes());
|
||||||
|
|
||||||
let num_chunks = vec![5u64];
|
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||||
let chunk_dims = vec![10u32];
|
|
||||||
let (info, consumed) = read_element(
|
let (info, consumed) = read_element(
|
||||||
&data,
|
&Window::whole(&data),
|
||||||
0,
|
0,
|
||||||
1,
|
1,
|
||||||
elem_size as u8,
|
elem_size as u8,
|
||||||
os,
|
os,
|
||||||
80,
|
80,
|
||||||
2,
|
2,
|
||||||
&num_chunks,
|
&grid,
|
||||||
&chunk_dims,
|
|
||||||
)
|
)
|
||||||
.unwrap();
|
.unwrap();
|
||||||
let ci = info.unwrap();
|
let ci = info.unwrap();
|
||||||
@@ -1090,4 +1130,38 @@ mod tests {
|
|||||||
assert_eq!(ci.offsets, vec![20]);
|
assert_eq!(ci.offsets, vec![20]);
|
||||||
assert_eq!(consumed, elem_size);
|
assert_eq!(consumed, elem_size);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The Storage path reads exactly what the slice path reads: the array
|
||||||
|
/// whole, cut at every length through its structures, and with a byte
|
||||||
|
/// damaged in each of them, through a read_at-only CountingStorage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let full = build_inline_plus_data_blocks();
|
||||||
|
let mut files = Vec::new();
|
||||||
|
for cut in 0x100..0x340 {
|
||||||
|
files.push(full[..cut].to_vec());
|
||||||
|
}
|
||||||
|
for at in [0x104, 0x150, 0x204, 0x216, 0x230, 0x304, 0x318] {
|
||||||
|
let mut damaged = full.clone();
|
||||||
|
damaged[at] ^= 1;
|
||||||
|
files.push(damaged);
|
||||||
|
}
|
||||||
|
files.push(full);
|
||||||
|
let mut compared = 0;
|
||||||
|
for f in files {
|
||||||
|
let storage = CountingStorage::new(f.clone());
|
||||||
|
let want = ExtensibleArrayHeader::parse(&f, 0x100, 8, 8);
|
||||||
|
let got = ExtensibleArrayHeader::parse_in(&storage, 0x100, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
let Ok(h) = want else { continue };
|
||||||
|
for dims in [&[40u64][..], &[25]] {
|
||||||
|
let want = read_extensible_array_chunks(&f, &h, dims, None, &[10], 8, 8, 8);
|
||||||
|
let got = read_extensible_array_chunks_in(&storage, &h, dims, None, &[10], 8, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"), "{} bytes", f.len());
|
||||||
|
compared += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(compared > 100);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -12,7 +12,8 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{format, vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks};
|
use crate::addr::to_usize;
|
||||||
|
use crate::chunked_read::{alloc_output, checked_byte_len, list_chunks_in};
|
||||||
use crate::data_layout::DataLayout;
|
use crate::data_layout::DataLayout;
|
||||||
use crate::dataspace::Dataspace;
|
use crate::dataspace::Dataspace;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
@@ -98,15 +99,63 @@ pub fn parse_fill_value(msg: &HeaderMessage) -> Result<Option<Vec<u8>>, FormatEr
|
|||||||
|
|
||||||
/// The fill value that applies to a dataset given its header messages. The new
|
/// The fill value that applies to a dataset given its header messages. The new
|
||||||
/// message wins over the old one when both are present.
|
/// message wins over the old one when both are present.
|
||||||
|
///
|
||||||
|
/// A *shared* fill value message holds only a reference to the real message,
|
||||||
|
/// which cannot be followed without the file: this returns
|
||||||
|
/// [`FormatError::UnresolvedSharedMessage`] for one (it used to answer "zeros").
|
||||||
|
/// Use [`dataset_fill_value_in`] when the file bytes are at hand.
|
||||||
pub fn dataset_fill_value(messages: &[HeaderMessage]) -> Result<Option<Vec<u8>>, FormatError> {
|
pub fn dataset_fill_value(messages: &[HeaderMessage]) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
|
fill_value_from(messages, |_| Err(FormatError::UnresolvedSharedMessage))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`dataset_fill_value`] for a dataset in `file_data`, following a shared
|
||||||
|
/// fill value message to where it lives: another object header, or the
|
||||||
|
/// file's shared-message (SOHM) heap, as libhdf5 writes it when the file has
|
||||||
|
/// a SOHM index for fill values.
|
||||||
|
pub fn dataset_fill_value_in(
|
||||||
|
file_data: &[u8],
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
|
dataset_fill_value_from_storage(&file_data, messages, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`dataset_fill_value_in`] with the file behind any
|
||||||
|
/// [`Storage`](crate::storage::Storage) (a `&dyn Storage` too). (The trait
|
||||||
|
/// is not imported here: its `len` would shadow the slice method in this
|
||||||
|
/// module.)
|
||||||
|
pub fn dataset_fill_value_from_storage<S: crate::storage::Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
|
fill_value_from(messages, |msg| {
|
||||||
|
crate::shared_message::message_data_with_sohm_in(file, msg, offset_size, length_size)
|
||||||
|
.map(|data| data.into_owned())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fill_value_from(
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
resolve_shared: impl Fn(&HeaderMessage) -> Result<Vec<u8>, FormatError>,
|
||||||
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
for wanted in [MessageType::FillValue, MessageType::FillValueOld] {
|
for wanted in [MessageType::FillValue, MessageType::FillValueOld] {
|
||||||
if let Some(msg) = messages.iter().find(|m| m.msg_type == wanted) {
|
if let Some(msg) = messages.iter().find(|m| m.msg_type == wanted) {
|
||||||
if crate::shared_message::is_shared(msg.flags) {
|
let value = if crate::shared_message::is_shared(msg.flags) {
|
||||||
// A shared fill value is legal but vanishingly rare; treat it
|
let data = resolve_shared(msg)?;
|
||||||
// as the default rather than misparsing the reference.
|
parse_fill_value(&HeaderMessage {
|
||||||
return Ok(None);
|
msg_type: msg.msg_type,
|
||||||
}
|
size: data.len(),
|
||||||
if let Some(value) = parse_fill_value(msg)? {
|
flags: msg.flags & !0x02,
|
||||||
|
creation_order: msg.creation_order,
|
||||||
|
data,
|
||||||
|
})?
|
||||||
|
} else {
|
||||||
|
parse_fill_value(msg)?
|
||||||
|
};
|
||||||
|
if let Some(value) = value {
|
||||||
return Ok(Some(value));
|
return Ok(Some(value));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -164,6 +213,30 @@ pub fn read_full_with_fill<E: From<FormatError>>(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
read: impl FnOnce() -> Result<Vec<u8>, E>,
|
read: impl FnOnce() -> Result<Vec<u8>, E>,
|
||||||
|
) -> Result<Vec<u8>, E> {
|
||||||
|
read_full_with_fill_in(
|
||||||
|
messages,
|
||||||
|
file_data,
|
||||||
|
layout,
|
||||||
|
dataspace,
|
||||||
|
elem_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
read,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`read_full_with_fill`] over any [`Storage`](crate::storage::Storage).
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
pub fn read_full_with_fill_in<E: From<FormatError>, S: crate::storage::Storage + ?Sized>(
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
file_data: &S,
|
||||||
|
layout: &DataLayout,
|
||||||
|
dataspace: &Dataspace,
|
||||||
|
elem_size: usize,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
read: impl FnOnce() -> Result<Vec<u8>, E>,
|
||||||
) -> Result<Vec<u8>, E> {
|
) -> Result<Vec<u8>, E> {
|
||||||
// A dataset with external raw data also has no data address in this
|
// A dataset with external raw data also has no data address in this
|
||||||
// file. It is NOT unallocated — its values live elsewhere — so it must
|
// file. It is NOT unallocated — its values live elsewhere — so it must
|
||||||
@@ -174,12 +247,12 @@ pub fn read_full_with_fill<E: From<FormatError>>(
|
|||||||
{
|
{
|
||||||
return Err(FormatError::ExternalDataFilesUnsupported.into());
|
return Err(FormatError::ExternalDataFilesUnsupported.into());
|
||||||
}
|
}
|
||||||
let fill = dataset_fill_value(messages)?;
|
let fill = dataset_fill_value_from_storage(file_data, messages, offset_size, length_size)?;
|
||||||
if !has_storage(layout) {
|
if !has_storage(layout) {
|
||||||
return Ok(filled_dataset(dataspace, elem_size, fill.as_deref())?);
|
return Ok(filled_dataset(dataspace, elem_size, fill.as_deref())?);
|
||||||
}
|
}
|
||||||
let mut output = read()?;
|
let mut output = read()?;
|
||||||
apply_to_unallocated_chunks(
|
apply_to_unallocated_chunks_in(
|
||||||
&mut output,
|
&mut output,
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
@@ -205,6 +278,30 @@ pub fn apply_to_unallocated_chunks(
|
|||||||
fill: Option<&[u8]>,
|
fill: Option<&[u8]>,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
apply_to_unallocated_chunks_in(
|
||||||
|
output,
|
||||||
|
file_data,
|
||||||
|
layout,
|
||||||
|
dataspace,
|
||||||
|
elem_size,
|
||||||
|
fill,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`apply_to_unallocated_chunks`] over any [`Storage`](crate::storage::Storage).
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
pub fn apply_to_unallocated_chunks_in<S: crate::storage::Storage + ?Sized>(
|
||||||
|
output: &mut [u8],
|
||||||
|
file_data: &S,
|
||||||
|
layout: &DataLayout,
|
||||||
|
dataspace: &Dataspace,
|
||||||
|
elem_size: usize,
|
||||||
|
fill: Option<&[u8]>,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
let Some(fill) = fill.filter(|f| f.len() == elem_size && !is_default(Some(f))) else {
|
let Some(fill) = fill.filter(|f| f.len() == elem_size && !is_default(Some(f))) else {
|
||||||
return Ok(());
|
return Ok(());
|
||||||
@@ -212,7 +309,7 @@ pub fn apply_to_unallocated_chunks(
|
|||||||
if !matches!(layout, DataLayout::Chunked { .. }) || elem_size == 0 {
|
if !matches!(layout, DataLayout::Chunked { .. }) || elem_size == 0 {
|
||||||
return Ok(());
|
return Ok(());
|
||||||
}
|
}
|
||||||
let (chunks, chunk_dims) = list_chunks(
|
let (chunks, chunk_dims) = list_chunks_in(
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
@@ -221,7 +318,11 @@ pub fn apply_to_unallocated_chunks(
|
|||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
let rank = chunk_dims.len();
|
let rank = chunk_dims.len();
|
||||||
let ds_dims: Vec<usize> = dataspace.dimensions.iter().map(|&d| d as usize).collect();
|
let ds_dims: Vec<usize> = dataspace
|
||||||
|
.dimensions
|
||||||
|
.iter()
|
||||||
|
.map(|&d| to_usize(d))
|
||||||
|
.collect::<Result<_, _>>()?;
|
||||||
if rank == 0 || ds_dims.len() != rank || chunk_dims.contains(&0) {
|
if rank == 0 || ds_dims.len() != rank || chunk_dims.contains(&0) {
|
||||||
return Ok(());
|
return Ok(());
|
||||||
}
|
}
|
||||||
@@ -253,7 +354,7 @@ pub fn apply_to_unallocated_chunks(
|
|||||||
let mut cell = 0usize;
|
let mut cell = 0usize;
|
||||||
let mut in_range = true;
|
let mut in_range = true;
|
||||||
for d in 0..rank {
|
for d in 0..rank {
|
||||||
let coord = chunk.offsets[d] as usize / chunk_dims[d];
|
let coord = to_usize(chunk.offsets[d])? / chunk_dims[d];
|
||||||
if coord >= grid[d] {
|
if coord >= grid[d] {
|
||||||
in_range = false;
|
in_range = false;
|
||||||
break;
|
break;
|
||||||
@@ -404,4 +505,37 @@ mod tests {
|
|||||||
.collect();
|
.collect();
|
||||||
assert_eq!(filled, [2, 3, 7, 8]);
|
assert_eq!(filled, [2, 3, 7, 8]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Fill values, shared ones in the SOHM heap included, resolve
|
||||||
|
/// identically through a read_at-only CountingStorage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::object_header::ObjectHeader;
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/shared_fill_value.h5");
|
||||||
|
let sb = crate::superblock::Superblock::parse(file, 0).unwrap();
|
||||||
|
let (os, ls) = (sb.offset_size, sb.length_size);
|
||||||
|
let storage = CountingStorage::new(file.to_vec());
|
||||||
|
let mut shared = 0;
|
||||||
|
let children =
|
||||||
|
crate::group_v2::resolve_group_children(file, &sb, sb.root_group_address).unwrap();
|
||||||
|
assert!(children.len() >= 3);
|
||||||
|
for child in children {
|
||||||
|
let h =
|
||||||
|
ObjectHeader::parse(file, child.object_header_address as usize, os, ls).unwrap();
|
||||||
|
shared += h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.filter(|m| {
|
||||||
|
m.msg_type == MessageType::FillValue
|
||||||
|
&& crate::shared_message::is_shared(m.flags)
|
||||||
|
})
|
||||||
|
.count();
|
||||||
|
let want = dataset_fill_value_in(file, &h.messages, os, ls);
|
||||||
|
assert_eq!(want, Ok(Some((-7i32).to_le_bytes().to_vec())));
|
||||||
|
let got = dataset_fill_value_from_storage(&storage, &h.messages, os, ls);
|
||||||
|
assert_eq!(got, want, "{}", child.name);
|
||||||
|
}
|
||||||
|
assert!(shared >= 2);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -19,8 +19,36 @@ pub const FILTER_SCALEOFFSET: u16 = 6;
|
|||||||
pub const FILTER_LZ4: u16 = 32004;
|
pub const FILTER_LZ4: u16 = 32004;
|
||||||
/// Zstandard compression.
|
/// Zstandard compression.
|
||||||
pub const FILTER_ZSTD: u16 = 32015;
|
pub const FILTER_ZSTD: u16 = 32015;
|
||||||
/// Pcodec lossless numerical codec (clawhdf5 internal; not yet HDF5-registered).
|
/// bzip2 (registered by PyTables; hdf5plugin's `BZip2`).
|
||||||
pub const FILTER_PCODEC: u16 = 32023;
|
pub const FILTER_BZIP2: u16 = 307;
|
||||||
|
/// LZF — h5py's built-in `compression="lzf"`.
|
||||||
|
pub const FILTER_LZF: u16 = 32000;
|
||||||
|
/// Blosc 1 (hdf5-blosc; hdf5plugin's `Blosc`).
|
||||||
|
pub const FILTER_BLOSC: u16 = 32001;
|
||||||
|
/// Bitshuffle, optionally with LZ4 or Zstandard (hdf5plugin's `Bitshuffle`).
|
||||||
|
pub const FILTER_BITSHUFFLE: u16 = 32008;
|
||||||
|
/// ZFP lossy (and lossless) compression of numeric arrays (H5Z-ZFP;
|
||||||
|
/// hdf5plugin's `Zfp`). Read-only, with the `zfp` feature.
|
||||||
|
pub const FILTER_ZFP: u16 = 32013;
|
||||||
|
/// Blosc 2 (hdf5plugin's `Blosc2`).
|
||||||
|
pub const FILTER_BLOSC2: u16 = 32026;
|
||||||
|
/// Pcodec lossless numerical codec — a **private, unregistered** clawhdf5
|
||||||
|
/// filter. Pcodec has no ID in the HDF Group's filter registry (checked
|
||||||
|
/// 2026-09-25, `hdf5_plugins/docs/RegisteredFilterPlugins.md`), so it uses an
|
||||||
|
/// ID from the registry's testing/private range (256–511). No libhdf5 plugin
|
||||||
|
/// decodes it: h5py/libhdf5 report the filter as unavailable. Only clawhdf5
|
||||||
|
/// (with the `pcodec` feature) reads these datasets.
|
||||||
|
pub const FILTER_PCODEC: u16 = 480;
|
||||||
|
/// Filter name written with [`FILTER_PCODEC`].
|
||||||
|
pub const FILTER_PCODEC_NAME: &str = "pcodec (clawhdf5 private)";
|
||||||
|
/// The ID clawhdf5 up to 2.7.0 wrote pcodec under. It is registered to
|
||||||
|
/// Granular BitRound (GBR), whose decode is a pass-through, so libhdf5 with
|
||||||
|
/// that plugin would have returned the compressed bytes as data. Read as
|
||||||
|
/// pcodec only when the filter is named exactly [`FILTER_PCODEC_LEGACY_NAME`],
|
||||||
|
/// the name those versions wrote; never written.
|
||||||
|
pub const FILTER_PCODEC_LEGACY: u16 = 32023;
|
||||||
|
/// The filter name clawhdf5 up to 2.7.0 wrote with [`FILTER_PCODEC_LEGACY`].
|
||||||
|
pub const FILTER_PCODEC_LEGACY_NAME: &str = "pcodec";
|
||||||
|
|
||||||
/// Description of a single filter in a pipeline.
|
/// Description of a single filter in a pipeline.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
|||||||
@@ -0,0 +1,496 @@
|
|||||||
|
//! Filter registry: every filter is looked up here by its HDF5 filter ID.
|
||||||
|
//!
|
||||||
|
//! Two tiers:
|
||||||
|
//!
|
||||||
|
//! * **Built-in filters** — a static table of the filters compiled into this
|
||||||
|
//! build: the HDF5 standard filters (deflate, shuffle, Fletcher32, szip,
|
||||||
|
//! N-Bit, scale-offset) and the plugin filters whose cargo features are
|
||||||
|
//! enabled (LZ4, Zstandard, pcodec, LZF, bitshuffle, bzip2, blosc,
|
||||||
|
//! blosc2, zfp).
|
||||||
|
//! [`builtin_filters`] lists them.
|
||||||
|
//! * **Registered filters** (`std` only) — codecs the application supplies
|
||||||
|
//! for any other ID with [`register_filter`] (a [`FilterCodec`], or just a
|
||||||
|
//! decoding closure). A registered codec cannot shadow a built-in one,
|
||||||
|
//! except under 32023: that ID belongs to Granular BitRound, and the
|
||||||
|
//! built-in entry there only reads the pcodec chunks clawhdf5 <= 2.7.0
|
||||||
|
//! wrote (filter name `"pcodec"`), so a codec registered for 32023 handles
|
||||||
|
//! every other chunk with that ID, and writes.
|
||||||
|
//!
|
||||||
|
//! An ID in neither tier fails with [`FormatError::UnsupportedFilter`], as it
|
||||||
|
//! always has.
|
||||||
|
//!
|
||||||
|
//! ```
|
||||||
|
//! # #[cfg(feature = "std")] {
|
||||||
|
//! use clawhdf5_format::filter_registry::{self, FilterContext};
|
||||||
|
//! use clawhdf5_format::error::FormatError;
|
||||||
|
//!
|
||||||
|
//! // A toy filter in the private-use range: every byte XORed with 0x5A.
|
||||||
|
//! filter_registry::register_filter(300, |input: &[u8], _ctx: &FilterContext<'_>| {
|
||||||
|
//! Ok::<_, FormatError>(input.iter().map(|b| b ^ 0x5A).collect())
|
||||||
|
//! })
|
||||||
|
//! .unwrap();
|
||||||
|
//! assert!(filter_registry::is_filter_available(300));
|
||||||
|
//! filter_registry::unregister_filter(300);
|
||||||
|
//! # }
|
||||||
|
//! ```
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_pipeline::FilterDescription;
|
||||||
|
|
||||||
|
/// What a codec is told about the filter it is applying.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub struct FilterContext<'a> {
|
||||||
|
/// The filter as recorded in the dataset's filter pipeline: its ID, name,
|
||||||
|
/// flags and client data (`cd_values`).
|
||||||
|
pub filter: &'a FilterDescription,
|
||||||
|
/// Size in bytes of one dataset element (the datatype's size).
|
||||||
|
pub element_size: usize,
|
||||||
|
/// Decoding only: the most bytes this stage may produce — what entered
|
||||||
|
/// the filter when the chunk was written. 0 means unknown; a decoder then
|
||||||
|
/// falls back to a fixed ceiling. Always 0 when encoding.
|
||||||
|
pub max_output: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl FilterContext<'_> {
|
||||||
|
/// The filter's client data (`cd_values`).
|
||||||
|
pub fn client_data(&self) -> &[u32] {
|
||||||
|
&self.filter.client_data
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The largest output a decoder should allow: [`Self::max_output`], or
|
||||||
|
/// 256 MiB when that is unknown.
|
||||||
|
pub fn output_limit(&self) -> usize {
|
||||||
|
if self.max_output != 0 {
|
||||||
|
self.max_output
|
||||||
|
} else {
|
||||||
|
crate::filters::MAX_DECOMPRESS_SIZE
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A filter implementation.
|
||||||
|
///
|
||||||
|
/// `decode` undoes the filter (the read direction). `encode` applies it (the
|
||||||
|
/// write direction); the default refuses with
|
||||||
|
/// [`FormatError::UnsupportedFilter`], which is right for a read-only codec.
|
||||||
|
pub trait FilterCodec: Send + Sync {
|
||||||
|
/// Undo the filter on one chunk. The output must not exceed
|
||||||
|
/// [`FilterContext::output_limit`]; the pipeline rejects a larger one.
|
||||||
|
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError>;
|
||||||
|
|
||||||
|
/// Apply the filter to one chunk.
|
||||||
|
fn encode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let _ = input;
|
||||||
|
Err(FormatError::UnsupportedFilter(ctx.filter.filter_id))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Any `Fn(&[u8], &FilterContext) -> Result<Vec<u8>, FormatError>` is a
|
||||||
|
/// decode-only codec.
|
||||||
|
impl<F> FilterCodec for F
|
||||||
|
where
|
||||||
|
F: Fn(&[u8], &FilterContext<'_>) -> Result<Vec<u8>, FormatError> + Send + Sync,
|
||||||
|
{
|
||||||
|
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
self(input, ctx)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Signature of a built-in filter's decoder or encoder.
|
||||||
|
pub type BuiltinFn = fn(&[u8], &FilterContext<'_>) -> Result<Vec<u8>, FormatError>;
|
||||||
|
|
||||||
|
/// A filter compiled into this build.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub struct BuiltinFilter {
|
||||||
|
/// HDF5 filter ID.
|
||||||
|
pub id: u16,
|
||||||
|
/// Human-readable name.
|
||||||
|
pub name: &'static str,
|
||||||
|
/// Decoder.
|
||||||
|
pub(crate) decode: BuiltinFn,
|
||||||
|
/// Encoder, if this build can write the filter.
|
||||||
|
pub(crate) encode: Option<BuiltinFn>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl BuiltinFilter {
|
||||||
|
/// Whether this build can write the filter as well as read it.
|
||||||
|
pub fn can_encode(&self) -> bool {
|
||||||
|
self.encode.is_some()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether the built-in entry only borrows its ID for some chunks, so a
|
||||||
|
/// registered codec may take the rest: the legacy pcodec entry under
|
||||||
|
/// Granular BitRound's 32023, which claims only chunks named `"pcodec"`.
|
||||||
|
fn is_shared(&self) -> bool {
|
||||||
|
self.id == crate::filter_pipeline::FILTER_PCODEC_LEGACY
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether this entry decodes chunks written with `filter`.
|
||||||
|
fn claims(&self, filter: &crate::filter_pipeline::FilterDescription) -> bool {
|
||||||
|
!self.is_shared()
|
||||||
|
|| filter.name.as_deref() == Some(crate::filter_pipeline::FILTER_PCODEC_LEGACY_NAME)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The filters compiled into this build, in ID order.
|
||||||
|
pub fn builtin_filters() -> &'static [BuiltinFilter] {
|
||||||
|
crate::filters::BUILTIN_FILTERS
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The built-in filter with this ID, if it is compiled in.
|
||||||
|
pub fn builtin_filter(id: u16) -> Option<&'static BuiltinFilter> {
|
||||||
|
builtin_filters().iter().find(|f| f.id == id)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a filter ID may be missing from this build: the filter's name, and
|
||||||
|
/// the cargo feature that provides it (`None`: clawhdf5 does not implement
|
||||||
|
/// it — register a codec for it with [`register_filter`]). `None` for an ID
|
||||||
|
/// clawhdf5 knows nothing about.
|
||||||
|
pub fn known_filter(id: u16) -> Option<(&'static str, Option<&'static str>)> {
|
||||||
|
Some(match id {
|
||||||
|
1 => ("deflate", Some("deflate")),
|
||||||
|
4 => ("SZIP", Some("szip")),
|
||||||
|
307 => ("bzip2", Some("bzip2")),
|
||||||
|
480 => ("pcodec", Some("pcodec")),
|
||||||
|
32000 => ("LZF", Some("lzf")),
|
||||||
|
32001 => ("Blosc", Some("blosc")),
|
||||||
|
32004 => ("LZ4", Some("lz4")),
|
||||||
|
32008 => ("bitshuffle", Some("bitshuffle")),
|
||||||
|
32013 => ("ZFP", Some("zfp")),
|
||||||
|
32015 => ("Zstandard", Some("zstd")),
|
||||||
|
32019 => ("JPEG", None),
|
||||||
|
32022 => ("BitGroom", None),
|
||||||
|
32023 => ("Granular BitRound", None),
|
||||||
|
32026 => ("Blosc2", Some("blosc2")),
|
||||||
|
_ => return None,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a chunk filtered with `id` can be decoded: a built-in filter or a
|
||||||
|
/// registered one.
|
||||||
|
pub fn is_filter_available(id: u16) -> bool {
|
||||||
|
if builtin_filter(id).is_some() {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
{
|
||||||
|
registered(id).is_some()
|
||||||
|
}
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
{
|
||||||
|
false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether chunks filtered with `id` may be decoded by a codec the
|
||||||
|
/// application registered (whose stored sizes this crate cannot bound).
|
||||||
|
pub(crate) fn may_be_registered(id: u16) -> bool {
|
||||||
|
if builtin_filter(id).is_some_and(|b| !b.is_shared()) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
{
|
||||||
|
registered(id).is_some()
|
||||||
|
}
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
{
|
||||||
|
false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
mod custom {
|
||||||
|
use super::FilterCodec;
|
||||||
|
use std::collections::BTreeMap;
|
||||||
|
use std::sync::{Arc, PoisonError, RwLock};
|
||||||
|
|
||||||
|
pub(super) type Registry = BTreeMap<u16, Arc<dyn FilterCodec>>;
|
||||||
|
|
||||||
|
static REGISTRY: RwLock<Registry> = RwLock::new(BTreeMap::new());
|
||||||
|
|
||||||
|
pub(super) fn with_read<R>(f: impl FnOnce(&Registry) -> R) -> R {
|
||||||
|
// A panic while holding the lock cannot leave the map half-updated
|
||||||
|
// (every update is a single insert/remove), so poisoning is ignored.
|
||||||
|
f(®ISTRY.read().unwrap_or_else(PoisonError::into_inner))
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn with_write<R>(f: impl FnOnce(&mut Registry) -> R) -> R {
|
||||||
|
f(&mut REGISTRY.write().unwrap_or_else(PoisonError::into_inner))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Register a codec for filter `id`, process-wide. It is used for every
|
||||||
|
/// chunk read (and, if it implements [`FilterCodec::encode`], written) with
|
||||||
|
/// that filter ID, by every file.
|
||||||
|
///
|
||||||
|
/// A plain closure `Fn(&[u8], &FilterContext) -> Result<Vec<u8>, FormatError>`
|
||||||
|
/// registers a decoder. Replaces (and returns) an earlier registration for
|
||||||
|
/// the same ID. Fails with [`FormatError::FilterError`] if `id` is a built-in
|
||||||
|
/// filter of this build: those cannot be overridden. The exception is 32023
|
||||||
|
/// (Granular BitRound): with the `pcodec` feature the built-in entry there
|
||||||
|
/// reads only chunks whose filter is named `"pcodec"` (clawhdf5 <= 2.7.0's
|
||||||
|
/// files); a codec registered for 32023 decodes every other chunk with that
|
||||||
|
/// ID and does all the writing.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
pub fn register_filter<C>(
|
||||||
|
id: u16,
|
||||||
|
codec: C,
|
||||||
|
) -> Result<Option<std::sync::Arc<dyn FilterCodec>>, FormatError>
|
||||||
|
where
|
||||||
|
C: FilterCodec + 'static,
|
||||||
|
{
|
||||||
|
if let Some(builtin) = builtin_filter(id).filter(|b| !b.is_shared()) {
|
||||||
|
return Err(FormatError::FilterError(format!(
|
||||||
|
"filter {id} ({}) is built in and cannot be re-registered",
|
||||||
|
builtin.name
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let codec: std::sync::Arc<dyn FilterCodec> = std::sync::Arc::new(codec);
|
||||||
|
Ok(custom::with_write(|r| r.insert(id, codec)))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Remove the codec registered for `id`. Returns whether one was registered.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
pub fn unregister_filter(id: u16) -> bool {
|
||||||
|
custom::with_write(|r| r.remove(&id).is_some())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The codec registered for `id`, if any.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
pub fn registered(id: u16) -> Option<std::sync::Arc<dyn FilterCodec>> {
|
||||||
|
custom::with_read(|r| r.get(&id).cloned())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Undo filter `ctx.filter` on `input`: the built-in decoder if there is one
|
||||||
|
/// that claims the chunk, else a registered one, else the built-in decoder's
|
||||||
|
/// own refusal or [`FormatError::UnsupportedFilter`].
|
||||||
|
pub(crate) fn decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let id = ctx.filter.filter_id;
|
||||||
|
let builtin = builtin_filter(id);
|
||||||
|
if let Some(builtin) = builtin.filter(|b| b.claims(ctx.filter)) {
|
||||||
|
return (builtin.decode)(input, ctx);
|
||||||
|
}
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
if let Some(codec) = registered(id) {
|
||||||
|
let out = codec.decode(input, ctx)?;
|
||||||
|
// A registered codec is outside our control: hold it to the same
|
||||||
|
// bound the built-in decoders enforce.
|
||||||
|
if out.len() > ctx.output_limit() {
|
||||||
|
return Err(FormatError::DecompressionError(format!(
|
||||||
|
"filter {id}: decoded {} bytes, more than the {} the chunk can hold",
|
||||||
|
out.len(),
|
||||||
|
ctx.output_limit()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
match builtin {
|
||||||
|
Some(builtin) => (builtin.decode)(input, ctx),
|
||||||
|
None => Err(FormatError::UnsupportedFilter(id)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Apply filter `ctx.filter` to `input`.
|
||||||
|
pub(crate) fn encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let id = ctx.filter.filter_id;
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
if builtin_filter(id).is_some_and(|b| b.is_shared())
|
||||||
|
&& let Some(codec) = registered(id)
|
||||||
|
{
|
||||||
|
return codec.encode(input, ctx);
|
||||||
|
}
|
||||||
|
if let Some(builtin) = builtin_filter(id) {
|
||||||
|
return match builtin.encode {
|
||||||
|
Some(encode) => encode(input, ctx),
|
||||||
|
None => Err(FormatError::UnsupportedFilter(id)),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
if let Some(codec) = registered(id) {
|
||||||
|
return codec.encode(input, ctx);
|
||||||
|
}
|
||||||
|
Err(FormatError::UnsupportedFilter(id))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(all(test, feature = "std"))]
|
||||||
|
pub(crate) mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::filter_pipeline::{FILTER_FLETCHER32, FILTER_SHUFFLE, FilterPipeline};
|
||||||
|
use crate::filters::{compress_chunk, decompress_chunk};
|
||||||
|
|
||||||
|
fn pipeline(id: u16) -> FilterPipeline {
|
||||||
|
FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![FilterDescription {
|
||||||
|
filter_id: id,
|
||||||
|
name: Some("test".into()),
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![7],
|
||||||
|
}],
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Xor;
|
||||||
|
impl FilterCodec for Xor {
|
||||||
|
fn decode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let k = ctx.client_data()[0] as u8;
|
||||||
|
Ok(input.iter().map(|b| b ^ k).collect())
|
||||||
|
}
|
||||||
|
fn encode(&self, input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
self.decode(input, ctx)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Each test uses its own ID: the registry is process-wide and tests run
|
||||||
|
// in parallel.
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unknown_filter_keeps_its_error() {
|
||||||
|
let err = decompress_chunk(b"abc", &pipeline(311), 3, 1).unwrap_err();
|
||||||
|
assert_eq!(err, FormatError::UnsupportedFilter(311));
|
||||||
|
let err = compress_chunk(b"abc", &pipeline(311), 1).unwrap_err();
|
||||||
|
assert_eq!(err, FormatError::UnsupportedFilter(311));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn registered_codec_round_trips_through_the_pipeline() {
|
||||||
|
assert!(!is_filter_available(312));
|
||||||
|
assert!(register_filter(312, Xor).unwrap().is_none());
|
||||||
|
assert!(is_filter_available(312));
|
||||||
|
let data = b"hello, registry".to_vec();
|
||||||
|
let enc = compress_chunk(&data, &pipeline(312), 1).unwrap();
|
||||||
|
assert_ne!(enc, data);
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(&enc, &pipeline(312), data.len(), 1).unwrap(),
|
||||||
|
data
|
||||||
|
);
|
||||||
|
assert!(unregister_filter(312));
|
||||||
|
assert!(!unregister_filter(312));
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(&enc, &pipeline(312), data.len(), 1).unwrap_err(),
|
||||||
|
FormatError::UnsupportedFilter(312)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn closure_registers_a_decoder_only() {
|
||||||
|
register_filter(313, |input: &[u8], _ctx: &FilterContext<'_>| {
|
||||||
|
Ok(input.iter().rev().copied().collect())
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(b"abc", &pipeline(313), 3, 1).unwrap(),
|
||||||
|
b"cba"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
compress_chunk(b"abc", &pipeline(313), 1).unwrap_err(),
|
||||||
|
FormatError::UnsupportedFilter(313)
|
||||||
|
);
|
||||||
|
unregister_filter(313);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn registered_decoder_output_is_bounded() {
|
||||||
|
register_filter(314, |_input: &[u8], _ctx: &FilterContext<'_>| {
|
||||||
|
Ok(vec![0u8; 1000])
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
let err = decompress_chunk(b"abc", &pipeline(314), 10, 1).unwrap_err();
|
||||||
|
assert!(matches!(err, FormatError::DecompressionError(_)), "{err:?}");
|
||||||
|
unregister_filter(314);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn builtins_cannot_be_overridden() {
|
||||||
|
for id in [FILTER_SHUFFLE, FILTER_FLETCHER32] {
|
||||||
|
let Err(err) = register_filter(id, Xor) else {
|
||||||
|
panic!("built-in filter {id} was re-registered");
|
||||||
|
};
|
||||||
|
assert!(matches!(err, FormatError::FilterError(_)), "{err:?}");
|
||||||
|
}
|
||||||
|
assert!(builtin_filter(FILTER_SHUFFLE).is_some());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Serialises the tests that register or read filter 32023 (the
|
||||||
|
/// registry is process-wide).
|
||||||
|
pub(crate) static ID_32023: std::sync::Mutex<()> = std::sync::Mutex::new(());
|
||||||
|
|
||||||
|
/// 32023 is Granular BitRound's ID; the `pcodec` build's built-in entry
|
||||||
|
/// there reads only clawhdf5 <= 2.7.0's pcodec chunks (named "pcodec"),
|
||||||
|
/// so a codec can be registered for the rest, and writes with it.
|
||||||
|
#[test]
|
||||||
|
fn a_codec_can_be_registered_for_granular_bitround() {
|
||||||
|
let _guard = ID_32023
|
||||||
|
.lock()
|
||||||
|
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||||
|
let named = |name: Option<&str>| FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![FilterDescription {
|
||||||
|
filter_id: 32023,
|
||||||
|
name: name.map(Into::into),
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![7],
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
let prev = register_filter(32023, Xor).expect("32023 must be registrable");
|
||||||
|
assert!(prev.is_none());
|
||||||
|
let data = b"granular bitround".to_vec();
|
||||||
|
for name in [None, Some("granular_bitround"), Some("test")] {
|
||||||
|
let pl = named(name);
|
||||||
|
let enc = compress_chunk(&data, &pl, 1).unwrap();
|
||||||
|
assert_ne!(enc, data);
|
||||||
|
assert_eq!(decompress_chunk(&enc, &pl, data.len(), 1).unwrap(), data);
|
||||||
|
}
|
||||||
|
// clawhdf5 <= 2.7.0's pcodec chunks still go to the built-in reader.
|
||||||
|
#[cfg(feature = "pcodec")]
|
||||||
|
{
|
||||||
|
let raw: Vec<u8> = (0..64)
|
||||||
|
.flat_map(|i| (f64::from(i) * 0.5).to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
let comp = crate::filters::pcodec_compress(&raw, 8).unwrap();
|
||||||
|
let mut pl = named(Some("pcodec"));
|
||||||
|
pl.filters[0].client_data = vec![8];
|
||||||
|
assert_eq!(decompress_chunk(&comp, &pl, raw.len(), 8).unwrap(), raw);
|
||||||
|
}
|
||||||
|
assert!(unregister_filter(32023));
|
||||||
|
let pl = named(None);
|
||||||
|
assert!(matches!(
|
||||||
|
decompress_chunk(&data, &pl, data.len(), 1),
|
||||||
|
Err(FormatError::UnsupportedFilter(32023))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unsupported_filter_error_names_the_filter() {
|
||||||
|
let msg = FormatError::UnsupportedFilter(32026).to_string();
|
||||||
|
assert!(msg.contains("Blosc2") && msg.contains("`blosc2`"), "{msg}");
|
||||||
|
let msg = FormatError::UnsupportedFilter(32013).to_string();
|
||||||
|
assert!(msg.contains("ZFP") && msg.contains("`zfp`"), "{msg}");
|
||||||
|
let msg = FormatError::UnsupportedFilter(32019).to_string();
|
||||||
|
assert!(
|
||||||
|
msg.contains("JPEG") && msg.contains("not implemented"),
|
||||||
|
"{msg}"
|
||||||
|
);
|
||||||
|
let msg = FormatError::UnsupportedFilter(32000).to_string();
|
||||||
|
assert!(msg.contains("LZF") && msg.contains("`lzf`"), "{msg}");
|
||||||
|
assert_eq!(
|
||||||
|
FormatError::UnsupportedFilter(399).to_string(),
|
||||||
|
"unsupported filter: 399"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn builtin_table_is_sorted_and_unique() {
|
||||||
|
let ids: Vec<u16> = builtin_filters().iter().map(|f| f.id).collect();
|
||||||
|
let mut sorted = ids.clone();
|
||||||
|
sorted.sort_unstable();
|
||||||
|
sorted.dedup();
|
||||||
|
assert_eq!(ids, sorted);
|
||||||
|
}
|
||||||
|
}
|
||||||
+1709
-332
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,446 @@
|
|||||||
|
//! Bitshuffle (HDF5 filter 32008) and the bit transpose it shares with blosc.
|
||||||
|
//!
|
||||||
|
//! **The transform.** A block of `n` elements (`n` a multiple of 8) of
|
||||||
|
//! `es` bytes each is viewed as an `n × 8·es` bit matrix — row *i* is
|
||||||
|
//! element *i*, column `8·j + k` is bit *k* (LSB first) of its byte *j* — and
|
||||||
|
//! transposed: the output is `8·es` rows of `n` bits, row `8·j + k` holding
|
||||||
|
//! bit *k* of byte *j* of every element in order, packed LSB first. That is
|
||||||
|
//! what `bshuf_trans_bit_elem` produces (checked against hdf5plugin's
|
||||||
|
//! library bit for bit).
|
||||||
|
//!
|
||||||
|
//! **The filter** (`bshuf_h5filter.c`). `cd_values`: `[0..2]` bitshuffle
|
||||||
|
//! version, `[2]` element size, `[3]` block size in elements (0 = default:
|
||||||
|
//! 8192 bytes' worth, rounded down to a multiple of 8, at least 128),
|
||||||
|
//! `[4]` compression (0 none, 2 LZ4, 3 Zstandard), `[5]` Zstandard level.
|
||||||
|
//! The chunk is cut into blocks of `block size` elements; the tail shorter
|
||||||
|
//! than a block is transposed as one block rounded down to a multiple of 8
|
||||||
|
//! elements, and the last `n mod 8` elements are stored as they are.
|
||||||
|
//! Uncompressed, that is the whole chunk. Compressed, the chunk starts with a
|
||||||
|
//! 12-byte header — the decoded size (u64 big-endian) and the block size in
|
||||||
|
//! bytes (u32 big-endian) — and each transposed block is stored as a u32
|
||||||
|
//! big-endian length and an LZ4 block / Zstandard frame; the untransposed
|
||||||
|
//! tail follows the last block.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
use crate::filter_registry::FilterContext;
|
||||||
|
|
||||||
|
/// Transpose an 8×8 bit matrix packed in a u64 (byte *r* = row *r*, bit *c*
|
||||||
|
/// of that byte = column *c*). An involution.
|
||||||
|
#[inline]
|
||||||
|
fn transpose8(mut x: u64) -> u64 {
|
||||||
|
let t = (x ^ (x >> 7)) & 0x00AA_00AA_00AA_00AA;
|
||||||
|
x = x ^ t ^ (t << 7);
|
||||||
|
let t = (x ^ (x >> 14)) & 0x0000_CCCC_0000_CCCC;
|
||||||
|
x = x ^ t ^ (t << 14);
|
||||||
|
let t = (x ^ (x >> 28)) & 0x0000_0000_F0F0_F0F0;
|
||||||
|
x ^ t ^ (t << 28)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bit-transpose one block: `input` and `out` are `n * es` bytes, `n` a
|
||||||
|
/// multiple of 8.
|
||||||
|
pub(crate) fn bitshuffle_block(input: &[u8], out: &mut [u8], n: usize, es: usize) {
|
||||||
|
debug_assert!(n.is_multiple_of(8) && input.len() == n * es && out.len() == n * es);
|
||||||
|
let row = n / 8;
|
||||||
|
for j in 0..es {
|
||||||
|
for g in 0..row {
|
||||||
|
let mut x = 0u64;
|
||||||
|
for t in 0..8 {
|
||||||
|
x |= u64::from(input[(8 * g + t) * es + j]) << (8 * t);
|
||||||
|
}
|
||||||
|
let y = transpose8(x);
|
||||||
|
for k in 0..8 {
|
||||||
|
out[(8 * j + k) * row + g] = (y >> (8 * k)) as u8;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Undo [`bitshuffle_block`].
|
||||||
|
pub(crate) fn bitunshuffle_block(input: &[u8], out: &mut [u8], n: usize, es: usize) {
|
||||||
|
debug_assert!(n.is_multiple_of(8) && input.len() == n * es && out.len() == n * es);
|
||||||
|
let row = n / 8;
|
||||||
|
for j in 0..es {
|
||||||
|
for g in 0..row {
|
||||||
|
let mut y = 0u64;
|
||||||
|
for k in 0..8 {
|
||||||
|
y |= u64::from(input[(8 * j + k) * row + g]) << (8 * k);
|
||||||
|
}
|
||||||
|
let x = transpose8(y);
|
||||||
|
for t in 0..8 {
|
||||||
|
out[(8 * g + t) * es + j] = (x >> (8 * t)) as u8;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `bshuf_default_block_size`: 8 KiB of elements, a multiple of 8, >= 128.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn default_block_size(es: usize) -> usize {
|
||||||
|
((8192 / es) / 8 * 8).max(128)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn err(msg: &str) -> FormatError {
|
||||||
|
FormatError::DecompressionError(format!("bitshuffle: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `cd_values[4]`: the compression bitshuffle applies after the transpose.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
enum Codec {
|
||||||
|
None,
|
||||||
|
Lz4,
|
||||||
|
Zstd,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn codec(cd: &[u32]) -> Result<Codec, FormatError> {
|
||||||
|
match cd.get(4).copied().unwrap_or(0) {
|
||||||
|
0 => Ok(Codec::None),
|
||||||
|
2 => Ok(Codec::Lz4),
|
||||||
|
3 => Ok(Codec::Zstd),
|
||||||
|
other => Err(FormatError::FilterError(format!(
|
||||||
|
"bitshuffle: unknown compression {other}"
|
||||||
|
))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The element counts of the transposed blocks for `size` elements.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn blocks(size: usize, block: usize) -> impl Iterator<Item = usize> {
|
||||||
|
let full = size / block;
|
||||||
|
let last = (size % block) / 8 * 8;
|
||||||
|
core::iter::repeat_n(block, full).chain((last > 0).then_some(last))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a bitshuffle-filtered chunk.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
pub(crate) fn bitshuffle_decode(
|
||||||
|
input: &[u8],
|
||||||
|
ctx: &FilterContext<'_>,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let cd = ctx.client_data();
|
||||||
|
let es = match cd.get(2) {
|
||||||
|
Some(&e) if e != 0 => e as usize,
|
||||||
|
_ => return Err(err("missing element size")),
|
||||||
|
};
|
||||||
|
let codec = codec(cd)?;
|
||||||
|
let limit = ctx.output_limit();
|
||||||
|
if codec == Codec::None {
|
||||||
|
if input.len() > limit {
|
||||||
|
return Err(err("output exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
let block = match cd.get(3) {
|
||||||
|
Some(&b) if b != 0 => b as usize,
|
||||||
|
_ => default_block_size(es),
|
||||||
|
};
|
||||||
|
if !block.is_multiple_of(8) {
|
||||||
|
return Err(err("block size is not a multiple of 8"));
|
||||||
|
}
|
||||||
|
if !input.len().is_multiple_of(es) {
|
||||||
|
return Err(err("chunk is not a whole number of elements"));
|
||||||
|
}
|
||||||
|
let size = input.len() / es;
|
||||||
|
let mut out = vec![0u8; input.len()];
|
||||||
|
let mut pos = 0;
|
||||||
|
for n in blocks(size, block) {
|
||||||
|
let bytes = n * es;
|
||||||
|
bitunshuffle_block(&input[pos..pos + bytes], &mut out[pos..pos + bytes], n, es);
|
||||||
|
pos += bytes;
|
||||||
|
}
|
||||||
|
out[pos..].copy_from_slice(&input[pos..]);
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
|
||||||
|
let header = input.get(..12).ok_or_else(|| err("truncated header"))?;
|
||||||
|
let total = u64::from_be_bytes(header[..8].try_into().unwrap());
|
||||||
|
let block_bytes = u32::from_be_bytes(header[8..12].try_into().unwrap()) as usize;
|
||||||
|
let total = usize::try_from(total)
|
||||||
|
.ok()
|
||||||
|
.filter(|&t| t <= limit)
|
||||||
|
.ok_or_else(|| err("decoded size exceeds the chunk size"))?;
|
||||||
|
if !total.is_multiple_of(es) {
|
||||||
|
return Err(err("chunk is not a whole number of elements"));
|
||||||
|
}
|
||||||
|
if block_bytes == 0 || !block_bytes.is_multiple_of(es) {
|
||||||
|
return Err(err("bad block size"));
|
||||||
|
}
|
||||||
|
let block = block_bytes / es;
|
||||||
|
if !block.is_multiple_of(8) {
|
||||||
|
return Err(err("block size is not a multiple of 8"));
|
||||||
|
}
|
||||||
|
let size = total / es;
|
||||||
|
let mut out = vec![0u8; total];
|
||||||
|
let mut tmp = vec![0u8; block_bytes.min(total)];
|
||||||
|
let mut ip = 12usize;
|
||||||
|
let mut op = 0usize;
|
||||||
|
let mut zstd = None;
|
||||||
|
for n in blocks(size, block) {
|
||||||
|
let bytes = n * es;
|
||||||
|
let len = input
|
||||||
|
.get(ip..ip + 4)
|
||||||
|
.map(|b| u32::from_be_bytes(b.try_into().unwrap()) as usize)
|
||||||
|
.ok_or_else(|| err("truncated block header"))?;
|
||||||
|
ip += 4;
|
||||||
|
let comp = input
|
||||||
|
.get(ip..ip.saturating_add(len))
|
||||||
|
.ok_or_else(|| err("truncated block"))?;
|
||||||
|
ip += len;
|
||||||
|
let dst = &mut tmp[..bytes];
|
||||||
|
let got = match codec {
|
||||||
|
Codec::Lz4 => lz4_flex::block::decompress_into(comp, dst)
|
||||||
|
.map_err(|e| err(&format!("lz4: {e}")))?,
|
||||||
|
Codec::Zstd => zstd_decode_into(
|
||||||
|
zstd.get_or_insert_with(ruzstd::decoding::FrameDecoder::new),
|
||||||
|
comp,
|
||||||
|
dst,
|
||||||
|
)?,
|
||||||
|
Codec::None => unreachable!(),
|
||||||
|
};
|
||||||
|
if got != bytes {
|
||||||
|
return Err(err("block decoded to the wrong size"));
|
||||||
|
}
|
||||||
|
bitunshuffle_block(dst, &mut out[op..op + bytes], n, es);
|
||||||
|
op += bytes;
|
||||||
|
}
|
||||||
|
let tail = total - op;
|
||||||
|
let rest = input
|
||||||
|
.get(ip..ip + tail)
|
||||||
|
.ok_or_else(|| err("truncated trailing elements"))?;
|
||||||
|
out[op..].copy_from_slice(rest);
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode Zstandard frames into exactly `dst`, failing if they hold more.
|
||||||
|
///
|
||||||
|
/// ruzstd reserves a frame's declared window (by default up to 100 MiB)
|
||||||
|
/// before decoding it, so the window is capped at what the output could
|
||||||
|
/// need: twice `dst` (window sizes are rounded up), and at least 128 KiB.
|
||||||
|
/// The encoders behind these filters (c-blosc, c-blosc2, bitshuffle)
|
||||||
|
/// compress each block in one call with its size known, so libzstd's
|
||||||
|
/// window never exceeds the block.
|
||||||
|
#[cfg(any(feature = "bitshuffle", feature = "blosc"))]
|
||||||
|
pub(crate) fn zstd_decode_into(
|
||||||
|
decoder: &mut ruzstd::decoding::FrameDecoder,
|
||||||
|
frames: &[u8],
|
||||||
|
dst: &mut [u8],
|
||||||
|
) -> Result<usize, FormatError> {
|
||||||
|
decoder.set_max_window_size((2 * dst.len()).max(1 << 17) as u64);
|
||||||
|
decoder
|
||||||
|
.decode_all(frames, dst)
|
||||||
|
.map_err(|e| FormatError::DecompressionError(format!("zstd: {e}")))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compress with ruzstd. It implements one level (roughly zstd's level 1),
|
||||||
|
/// so the requested level only matters to other encoders.
|
||||||
|
#[cfg(any(feature = "bitshuffle", feature = "blosc"))]
|
||||||
|
pub(crate) fn zstd_encode(data: &[u8]) -> Vec<u8> {
|
||||||
|
ruzstd::encoding::compress_to_vec(data, ruzstd::encoding::CompressionLevel::Fastest)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a chunk with the bitshuffle filter.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
pub(crate) fn bitshuffle_encode(
|
||||||
|
input: &[u8],
|
||||||
|
ctx: &FilterContext<'_>,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let cd = ctx.client_data();
|
||||||
|
let es = match cd.get(2) {
|
||||||
|
Some(&e) if e != 0 => e as usize,
|
||||||
|
_ => ctx.element_size.max(1),
|
||||||
|
};
|
||||||
|
let codec = codec(cd)?;
|
||||||
|
let block = match cd.get(3) {
|
||||||
|
Some(&b) if b != 0 => b as usize,
|
||||||
|
_ => default_block_size(es),
|
||||||
|
};
|
||||||
|
let cerr = |m: &str| FormatError::CompressionError(format!("bitshuffle: {m}"));
|
||||||
|
if !block.is_multiple_of(8) {
|
||||||
|
return Err(cerr("block size is not a multiple of 8"));
|
||||||
|
}
|
||||||
|
if !input.len().is_multiple_of(es) {
|
||||||
|
return Err(cerr("chunk is not a whole number of elements"));
|
||||||
|
}
|
||||||
|
let size = input.len() / es;
|
||||||
|
let mut out = Vec::with_capacity(input.len() + 12 + input.len() / 64);
|
||||||
|
if codec != Codec::None {
|
||||||
|
out.extend_from_slice(&(input.len() as u64).to_be_bytes());
|
||||||
|
let block_bytes =
|
||||||
|
u32::try_from(block * es).map_err(|_| cerr("block size does not fit in 32 bits"))?;
|
||||||
|
out.extend_from_slice(&block_bytes.to_be_bytes());
|
||||||
|
}
|
||||||
|
let mut tmp = vec![0u8; (block * es).min(input.len())];
|
||||||
|
let mut pos = 0;
|
||||||
|
for n in blocks(size, block) {
|
||||||
|
let bytes = n * es;
|
||||||
|
let dst = &mut tmp[..bytes];
|
||||||
|
bitshuffle_block(&input[pos..pos + bytes], dst, n, es);
|
||||||
|
match codec {
|
||||||
|
Codec::None => out.extend_from_slice(dst),
|
||||||
|
Codec::Lz4 | Codec::Zstd => {
|
||||||
|
let comp = if codec == Codec::Lz4 {
|
||||||
|
lz4_flex::block::compress(dst)
|
||||||
|
} else {
|
||||||
|
zstd_encode(dst)
|
||||||
|
};
|
||||||
|
out.extend_from_slice(&(comp.len() as u32).to_be_bytes());
|
||||||
|
out.extend_from_slice(&comp);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pos += bytes;
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&input[pos..]);
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The definition, one bit at a time.
|
||||||
|
fn naive(input: &[u8], n: usize, es: usize) -> Vec<u8> {
|
||||||
|
let mut out = vec![0u8; n * es];
|
||||||
|
for i in 0..n {
|
||||||
|
for j in 0..es {
|
||||||
|
for k in 0..8 {
|
||||||
|
if input[i * es + j] >> k & 1 == 1 {
|
||||||
|
let p = (8 * j + k) * n + i;
|
||||||
|
out[p / 8] |= 1 << (p % 8);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn transpose_matches_the_definition_and_inverts() {
|
||||||
|
for (n, es) in [(8, 1), (16, 2), (24, 4), (128, 8), (64, 3), (8, 16)] {
|
||||||
|
let input: Vec<u8> = (0..n * es)
|
||||||
|
.map(|i| (i as u32).wrapping_mul(2_654_435_761).rotate_left(7) as u8)
|
||||||
|
.collect();
|
||||||
|
let mut out = vec![0u8; n * es];
|
||||||
|
bitshuffle_block(&input, &mut out, n, es);
|
||||||
|
assert_eq!(out, naive(&input, n, es), "n={n} es={es}");
|
||||||
|
let mut back = vec![0u8; n * es];
|
||||||
|
bitunshuffle_block(&out, &mut back, n, es);
|
||||||
|
assert_eq!(back, input);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
fn ctx_for(cd: Vec<u32>) -> crate::filter_pipeline::FilterDescription {
|
||||||
|
crate::filter_pipeline::FilterDescription {
|
||||||
|
filter_id: crate::filter_pipeline::FILTER_BITSHUFFLE,
|
||||||
|
name: None,
|
||||||
|
flags: 0,
|
||||||
|
client_data: cd,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
#[test]
|
||||||
|
fn filter_round_trips_every_mode() {
|
||||||
|
for es in [1usize, 2, 4, 8] {
|
||||||
|
for n in [0usize, 1, 7, 8, 100, 1000, 5003] {
|
||||||
|
let data: Vec<u8> = (0..n * es)
|
||||||
|
.map(|i| (i % 97) as u8 ^ (i / 300) as u8)
|
||||||
|
.collect();
|
||||||
|
for (comp, block) in [(0, 0), (0, 16), (2, 0), (2, 64), (3, 0), (3, 1024)] {
|
||||||
|
let f = ctx_for(vec![0, 4, es as u32, block, comp]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: es,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = bitshuffle_encode(&data, &ctx).unwrap();
|
||||||
|
let dec = bitshuffle_decode(&enc, &ctx).unwrap();
|
||||||
|
assert_eq!(dec, data, "es={es} n={n} comp={comp} block={block}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
#[test]
|
||||||
|
fn rejects_oversized_and_truncated_chunks() {
|
||||||
|
let data = vec![5u8; 4096];
|
||||||
|
let f = ctx_for(vec![0, 4, 4, 0, 2]);
|
||||||
|
let mut ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = bitshuffle_encode(&data, &ctx).unwrap();
|
||||||
|
assert!(bitshuffle_decode(&enc[..enc.len() - 1], &ctx).is_err());
|
||||||
|
ctx.max_output = 100;
|
||||||
|
assert!(bitshuffle_decode(&enc, &ctx).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Random and mutated chunks, in every mode, and hostile `cd_values`:
|
||||||
|
/// errors are fine, panics are not.
|
||||||
|
#[cfg(feature = "bitshuffle")]
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_chunks_never_panic() {
|
||||||
|
use crate::test_fuzz::{Rng, fuzz_decoder};
|
||||||
|
let data: Vec<u8> = (0..3001u32)
|
||||||
|
.flat_map(|i| ((i / 7) as u16).to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
for (comp, block) in [(0, 0), (0, 16), (2, 0), (2, 64), (3, 0), (3, 1024)] {
|
||||||
|
let f = ctx_for(vec![0, 4, 2, block, comp]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 2,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let seeds = vec![
|
||||||
|
bitshuffle_encode(&data, &ctx).unwrap(),
|
||||||
|
bitshuffle_encode(&data[..34], &ctx).unwrap(),
|
||||||
|
bitshuffle_encode(&data[..512], &ctx).unwrap(),
|
||||||
|
];
|
||||||
|
fuzz_decoder(
|
||||||
|
0xb5 + comp as u64 * 7 + block as u64,
|
||||||
|
&seeds,
|
||||||
|
4_000,
|
||||||
|
data.len(),
|
||||||
|
|s| bitshuffle_decode(s, &ctx),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Hostile filter parameters on a valid chunk.
|
||||||
|
let mut rng = Rng::new(0xcd);
|
||||||
|
let good = ctx_for(vec![0, 4, 2, 0, 2]);
|
||||||
|
let enc = bitshuffle_encode(
|
||||||
|
&data,
|
||||||
|
&FilterContext {
|
||||||
|
filter: &good,
|
||||||
|
element_size: 2,
|
||||||
|
max_output: data.len(),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
for _ in 0..3_000 {
|
||||||
|
let cd: Vec<u32> = (0..rng.below(7))
|
||||||
|
.map(|_| match rng.below(4) {
|
||||||
|
0 => rng.below(5) as u32,
|
||||||
|
1 => u32::MAX - rng.below(4) as u32,
|
||||||
|
2 => 1 << rng.below(32),
|
||||||
|
_ => rng.next_u64() as u32,
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let f = ctx_for(cd);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 2,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let _ = bitshuffle_decode(&enc, &ctx);
|
||||||
|
let _ = bitshuffle_decode(&data, &ctx);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,711 @@
|
|||||||
|
//! Blosc 1 (HDF5 filter 32001, `hdf5-blosc`, hdf5plugin's `Blosc`), in pure
|
||||||
|
//! Rust: the Blosc 1 frame, its byte shuffle and bit shuffle, and the
|
||||||
|
//! BloscLZ, LZ4/LZ4HC, Snappy, Zlib and Zstandard codecs inside it.
|
||||||
|
//!
|
||||||
|
//! **Frame** (c-blosc 1.x, format version 2). A 16-byte header — version
|
||||||
|
//! (2), codec format version (1), flags, type size, then little-endian `u32`
|
||||||
|
//! decoded size, block size and frame size. Flags: bit 0 byte shuffle, bit
|
||||||
|
//! 1 stored raw ("memcpyed": the data follows the header), bit 2 bit
|
||||||
|
//! shuffle, bit 4 "do not split", bits 5-7 the codec (0 BloscLZ, 1 LZ4 and
|
||||||
|
//! LZ4HC, 2 Snappy, 3 Zlib, 4 Zstandard). Unless stored raw, a table of
|
||||||
|
//! `u32` block offsets follows, one per block of `block size` bytes (the
|
||||||
|
//! last one may be shorter). A block is one stream, or — when the "do not
|
||||||
|
//! split" flag is clear, the type size is at most 16, the block holds at
|
||||||
|
//! least 128 elements, and it is not the short last block — `type size`
|
||||||
|
//! streams, one per byte plane. Each stream is a `u32` length and the
|
||||||
|
//! codec's output; a length equal to the stream's decoded size means the
|
||||||
|
//! bytes are stored raw. The decoded block is then unshuffled (byte shuffle
|
||||||
|
//! for type size > 1; bit shuffle when the block holds a multiple of 8
|
||||||
|
//! elements, the trailing partial element copied as is).
|
||||||
|
//!
|
||||||
|
//! **Filter** (`blosc_filter.c`) `cd_values`: `[0]` filter revision, `[1]`
|
||||||
|
//! Blosc format version, `[2]` type size, `[3]` chunk size in bytes, `[4]`
|
||||||
|
//! compression level, `[5]` shuffle (0 none, 1 byte, 2 bit), `[6]`
|
||||||
|
//! compressor (0 blosclz, 1 lz4, 2 lz4hc, 3 snappy, 4 zlib, 5 zstd). The
|
||||||
|
//! decoder needs only the frame.
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_registry::FilterContext;
|
||||||
|
use crate::filters_bitshuffle::{bitshuffle_block, bitunshuffle_block};
|
||||||
|
|
||||||
|
const HEADER: usize = 16;
|
||||||
|
const FLAG_SHUFFLE: u8 = 0x01;
|
||||||
|
const FLAG_MEMCPYED: u8 = 0x02;
|
||||||
|
const FLAG_BITSHUFFLE: u8 = 0x04;
|
||||||
|
const FLAG_FUTURE: u8 = 0x08;
|
||||||
|
const FLAG_DONT_SPLIT: u8 = 0x10;
|
||||||
|
const MAX_SPLITS: usize = 16;
|
||||||
|
const MIN_BUFFERSIZE: usize = 128;
|
||||||
|
|
||||||
|
fn err(msg: &str) -> FormatError {
|
||||||
|
FormatError::DecompressionError(format!("blosc: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn le32(b: &[u8], at: usize) -> Result<usize, FormatError> {
|
||||||
|
b.get(at..at + 4)
|
||||||
|
.map(|s| u32::from_le_bytes(s.try_into().unwrap()) as usize)
|
||||||
|
.ok_or_else(|| err("truncated frame"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The codec inside a Blosc frame (flags bits 5-7).
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub(crate) enum Codec {
|
||||||
|
BloscLz,
|
||||||
|
Lz4,
|
||||||
|
Snappy,
|
||||||
|
Zlib,
|
||||||
|
Zstd,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Codec {
|
||||||
|
pub(crate) fn from_flags(flags: u8) -> Result<Codec, FormatError> {
|
||||||
|
match flags >> 5 {
|
||||||
|
0 => Ok(Codec::BloscLz),
|
||||||
|
1 => Ok(Codec::Lz4),
|
||||||
|
2 => Ok(Codec::Snappy),
|
||||||
|
3 => Ok(Codec::Zlib),
|
||||||
|
4 => Ok(Codec::Zstd),
|
||||||
|
other => Err(err(&format!("unknown codec {other}"))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode one codec stream into exactly `dst`.
|
||||||
|
pub(crate) fn decode_stream(
|
||||||
|
codec: Codec,
|
||||||
|
src: &[u8],
|
||||||
|
dst: &mut [u8],
|
||||||
|
zstd: &mut Option<ruzstd::decoding::FrameDecoder>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
let n = match codec {
|
||||||
|
Codec::BloscLz => blosclz_decompress(src, dst),
|
||||||
|
Codec::Lz4 => {
|
||||||
|
lz4_flex::block::decompress_into(src, dst).map_err(|e| err(&format!("lz4: {e}")))?
|
||||||
|
}
|
||||||
|
Codec::Snappy => {
|
||||||
|
let len = snap::raw::decompress_len(src).map_err(|e| err(&format!("snappy: {e}")))?;
|
||||||
|
if len != dst.len() {
|
||||||
|
return Err(err("snappy stream has the wrong size"));
|
||||||
|
}
|
||||||
|
snap::raw::Decoder::new()
|
||||||
|
.decompress(src, dst)
|
||||||
|
.map_err(|e| err(&format!("snappy: {e}")))?
|
||||||
|
}
|
||||||
|
Codec::Zlib => {
|
||||||
|
let out = crate::filters::inflate_bounded(src, dst.len(), dst.len())
|
||||||
|
.map_err(|e| err(&format!("zlib: {e}")))?;
|
||||||
|
let n = out.len();
|
||||||
|
if n == dst.len() {
|
||||||
|
dst.copy_from_slice(&out);
|
||||||
|
}
|
||||||
|
n
|
||||||
|
}
|
||||||
|
Codec::Zstd => crate::filters_bitshuffle::zstd_decode_into(
|
||||||
|
zstd.get_or_insert_with(ruzstd::decoding::FrameDecoder::new),
|
||||||
|
src,
|
||||||
|
dst,
|
||||||
|
)?,
|
||||||
|
};
|
||||||
|
if n != dst.len() {
|
||||||
|
return Err(err("stream decoded to the wrong size"));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a Blosc-filtered chunk: one Blosc 1 frame.
|
||||||
|
///
|
||||||
|
/// An HDF5 chunk is never empty, so a frame that decodes to nothing where
|
||||||
|
/// the chunk size is known is corrupt (libhdf5's filter fails it too).
|
||||||
|
pub(crate) fn blosc_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let out = blosc_decompress(input, ctx.output_limit())?;
|
||||||
|
if out.is_empty() && ctx.max_output != 0 {
|
||||||
|
return Err(err("empty frame for a non-empty chunk"));
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decompress a Blosc 1 frame, refusing more than `limit` bytes of output.
|
||||||
|
pub fn blosc_decompress(input: &[u8], limit: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
|
if input.len() < HEADER {
|
||||||
|
return Err(err("truncated header"));
|
||||||
|
}
|
||||||
|
let version = input[0];
|
||||||
|
let codec_version = input[1];
|
||||||
|
let flags = input[2];
|
||||||
|
let typesize = input[3] as usize;
|
||||||
|
let nbytes = le32(input, 4)?;
|
||||||
|
let blocksize = le32(input, 8)?;
|
||||||
|
let cbytes = le32(input, 12)?;
|
||||||
|
if version != 1 && version != 2 {
|
||||||
|
return Err(err(&format!(
|
||||||
|
"frame format version {version} is not Blosc 1 (a Blosc 2 chunk?)"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
if flags & FLAG_FUTURE != 0 {
|
||||||
|
return Err(err("unknown header flags"));
|
||||||
|
}
|
||||||
|
if nbytes > limit {
|
||||||
|
return Err(err("decoded size exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
if cbytes > input.len() {
|
||||||
|
return Err(err("frame is longer than the chunk"));
|
||||||
|
}
|
||||||
|
if cbytes < HEADER {
|
||||||
|
return Err(err("truncated frame"));
|
||||||
|
}
|
||||||
|
let src = &input[..cbytes];
|
||||||
|
if nbytes == 0 {
|
||||||
|
return Ok(Vec::new());
|
||||||
|
}
|
||||||
|
if blocksize == 0 || typesize == 0 {
|
||||||
|
return Err(err("bad block or type size"));
|
||||||
|
}
|
||||||
|
let mut out = vec![0u8; nbytes];
|
||||||
|
if flags & FLAG_MEMCPYED != 0 {
|
||||||
|
if cbytes != nbytes + HEADER {
|
||||||
|
return Err(err("stored frame has the wrong size"));
|
||||||
|
}
|
||||||
|
out.copy_from_slice(&src[HEADER..]);
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
let codec = Codec::from_flags(flags)?;
|
||||||
|
if codec_version != 1 {
|
||||||
|
return Err(err(&format!(
|
||||||
|
"unsupported {codec:?} format version {codec_version}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let nblocks = nbytes.div_ceil(blocksize);
|
||||||
|
let leftover = nbytes % blocksize;
|
||||||
|
if nblocks > (cbytes - HEADER) / 4 {
|
||||||
|
return Err(err("block table is truncated"));
|
||||||
|
}
|
||||||
|
let block_len = blocksize.min(nbytes);
|
||||||
|
let mut tmp = vec![0u8; block_len];
|
||||||
|
let mut zstd = None;
|
||||||
|
let dont_split = flags & FLAG_DONT_SPLIT != 0;
|
||||||
|
for j in 0..nblocks {
|
||||||
|
let is_leftover = j == nblocks - 1 && leftover > 0;
|
||||||
|
let bsize = if is_leftover { leftover } else { blocksize };
|
||||||
|
let nsplits = if !dont_split
|
||||||
|
&& typesize <= MAX_SPLITS
|
||||||
|
&& bsize / typesize >= MIN_BUFFERSIZE
|
||||||
|
&& !is_leftover
|
||||||
|
{
|
||||||
|
typesize
|
||||||
|
} else {
|
||||||
|
1
|
||||||
|
};
|
||||||
|
let neblock = bsize / nsplits;
|
||||||
|
let mut pos = le32(src, HEADER + 4 * j)?;
|
||||||
|
let tmp = &mut tmp[..bsize];
|
||||||
|
for s in 0..nsplits {
|
||||||
|
let clen = src
|
||||||
|
.get(pos..)
|
||||||
|
.and_then(|rest| rest.get(..4))
|
||||||
|
.map(|b| u32::from_le_bytes(b.try_into().unwrap()) as usize)
|
||||||
|
.ok_or_else(|| err("block offset out of range"))?;
|
||||||
|
pos += 4;
|
||||||
|
let stream = src
|
||||||
|
.get(pos..pos.saturating_add(clen))
|
||||||
|
.ok_or_else(|| err("stream runs past the frame"))?;
|
||||||
|
let dst = &mut tmp[s * neblock..(s + 1) * neblock];
|
||||||
|
if clen == neblock {
|
||||||
|
dst.copy_from_slice(stream);
|
||||||
|
} else {
|
||||||
|
decode_stream(codec, stream, dst, &mut zstd)?;
|
||||||
|
}
|
||||||
|
pos += clen;
|
||||||
|
}
|
||||||
|
// `bsize` is a whole number of splits by construction (`nsplits` > 1
|
||||||
|
// only for full blocks, and c-blosc sizes those in whole elements).
|
||||||
|
if nsplits * neblock != bsize {
|
||||||
|
return Err(err("block is not a whole number of streams"));
|
||||||
|
}
|
||||||
|
let dest = &mut out[j * blocksize..j * blocksize + bsize];
|
||||||
|
unshuffle_block(flags, typesize, tmp, dest);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Undo the frame's shuffle on one decoded block.
|
||||||
|
fn unshuffle_block(flags: u8, typesize: usize, src: &[u8], dest: &mut [u8]) {
|
||||||
|
let bsize = src.len();
|
||||||
|
if flags & FLAG_SHUFFLE != 0 && typesize > 1 {
|
||||||
|
let n = bsize / typesize;
|
||||||
|
for i in 0..n {
|
||||||
|
for b in 0..typesize {
|
||||||
|
dest[i * typesize + b] = src[b * n + i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
dest[n * typesize..].copy_from_slice(&src[n * typesize..]);
|
||||||
|
} else if flags & FLAG_BITSHUFFLE != 0 && bsize >= typesize {
|
||||||
|
let n = bsize / typesize;
|
||||||
|
if n.is_multiple_of(8) {
|
||||||
|
let body = n * typesize;
|
||||||
|
bitunshuffle_block(&src[..body], &mut dest[..body], n, typesize);
|
||||||
|
dest[body..].copy_from_slice(&src[body..]);
|
||||||
|
} else {
|
||||||
|
dest.copy_from_slice(src);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
dest.copy_from_slice(src);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// BloscLZ decompression (c-blosc 1.21 `blosclz_decompress`): returns the
|
||||||
|
/// number of bytes written, or 0 on malformed input — exactly as the C
|
||||||
|
/// decoder, including stopping before a match that ends the stream, so a
|
||||||
|
/// stream libblosc rejects is rejected here too.
|
||||||
|
///
|
||||||
|
/// Instructions: a control byte `ctrl`. Below 32, a literal run of
|
||||||
|
/// `ctrl + 1` bytes. Otherwise a match: length `(ctrl >> 5) + 2`, extended
|
||||||
|
/// by following bytes while they are 255 when the top three bits are all
|
||||||
|
/// set; distance `((ctrl & 31) << 8) + next byte + 1`, or — when that byte
|
||||||
|
/// is 255 and the high bits are 31 — a 16-bit big-endian distance plus 8192.
|
||||||
|
/// The first instruction is always a literal.
|
||||||
|
pub(crate) fn blosclz_decompress(input: &[u8], out: &mut [u8]) -> usize {
|
||||||
|
const MAX_DISTANCE: usize = 8191;
|
||||||
|
let limit = input.len();
|
||||||
|
if limit == 0 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let mut ip = 1usize;
|
||||||
|
let mut op = 0usize;
|
||||||
|
let mut ctrl = (input[0] & 31) as usize;
|
||||||
|
loop {
|
||||||
|
if ctrl >= 32 {
|
||||||
|
let mut len = (ctrl >> 5) - 1;
|
||||||
|
let ofs = (ctrl & 31) << 8;
|
||||||
|
if len == 6 {
|
||||||
|
loop {
|
||||||
|
if ip + 1 >= limit {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let code = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
len += code;
|
||||||
|
if code != 255 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else if ip + 1 >= limit {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let code = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
len += 3;
|
||||||
|
// The copy source is `distance` bytes back.
|
||||||
|
let mut distance = ofs + code + 1;
|
||||||
|
if code == 255 && ofs == 31 << 8 {
|
||||||
|
if ip + 1 >= limit {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let far = ((input[ip] as usize) << 8) + input[ip + 1] as usize;
|
||||||
|
ip += 2;
|
||||||
|
distance = far + MAX_DISTANCE + 1;
|
||||||
|
}
|
||||||
|
if op + len > out.len() {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
if distance > op {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
if ip >= limit {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
ctrl = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
let start = op - distance;
|
||||||
|
if distance >= len {
|
||||||
|
out.copy_within(start..start + len, op);
|
||||||
|
} else {
|
||||||
|
for k in 0..len {
|
||||||
|
out[op + k] = out[start + k];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
op += len;
|
||||||
|
} else {
|
||||||
|
let run = ctrl + 1;
|
||||||
|
if op + run > out.len() || ip + run > limit {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
out[op..op + run].copy_from_slice(&input[ip..ip + run]);
|
||||||
|
op += run;
|
||||||
|
ip += run;
|
||||||
|
if ip >= limit {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
ctrl = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
op
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The codec our encoder puts inside the frame.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub(crate) enum EncodeCodec {
|
||||||
|
Lz4,
|
||||||
|
Snappy,
|
||||||
|
Zlib,
|
||||||
|
Zstd,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl EncodeCodec {
|
||||||
|
/// From the filter's `cd_values[6]` compressor code.
|
||||||
|
fn from_cd(code: u32) -> Result<EncodeCodec, FormatError> {
|
||||||
|
match code {
|
||||||
|
1 | 2 => Ok(EncodeCodec::Lz4),
|
||||||
|
3 => Ok(EncodeCodec::Snappy),
|
||||||
|
4 => Ok(EncodeCodec::Zlib),
|
||||||
|
5 => Ok(EncodeCodec::Zstd),
|
||||||
|
0 => Err(FormatError::CompressionError(
|
||||||
|
"blosc: clawhdf5 cannot write BloscLZ; choose lz4, snappy, zlib or zstd".into(),
|
||||||
|
)),
|
||||||
|
other => Err(FormatError::CompressionError(format!(
|
||||||
|
"blosc: unknown compressor {other}"
|
||||||
|
))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn flags(self) -> u8 {
|
||||||
|
(match self {
|
||||||
|
EncodeCodec::Lz4 => 1,
|
||||||
|
EncodeCodec::Snappy => 2,
|
||||||
|
EncodeCodec::Zlib => 3,
|
||||||
|
EncodeCodec::Zstd => 4,
|
||||||
|
}) << 5
|
||||||
|
}
|
||||||
|
|
||||||
|
fn encode(self, data: &[u8], level: u32) -> Result<Vec<u8>, FormatError> {
|
||||||
|
match self {
|
||||||
|
EncodeCodec::Lz4 => Ok(lz4_flex::block::compress(data)),
|
||||||
|
EncodeCodec::Snappy => snap::raw::Encoder::new()
|
||||||
|
.compress_vec(data)
|
||||||
|
.map_err(|e| FormatError::CompressionError(format!("blosc: snappy: {e}"))),
|
||||||
|
EncodeCodec::Zlib => crate::filters::deflate_bounded(data, level.min(9))
|
||||||
|
.map_err(|e| FormatError::CompressionError(format!("blosc: zlib: {e}"))),
|
||||||
|
EncodeCodec::Zstd => Ok(crate::filters_bitshuffle::zstd_encode(data)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Block size our encoder uses: at most 256 KiB, a whole number of
|
||||||
|
/// elements (and, for bit shuffle, of 8-element groups).
|
||||||
|
fn encode_block_size(nbytes: usize, typesize: usize, bitshuffle: bool) -> usize {
|
||||||
|
let unit = if bitshuffle { 8 * typesize } else { typesize };
|
||||||
|
let target = (256 * 1024).min(nbytes);
|
||||||
|
if target < unit {
|
||||||
|
return nbytes.max(1);
|
||||||
|
}
|
||||||
|
target / unit * unit
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a chunk as one Blosc 1 frame. `cd_values` as hdf5-blosc:
|
||||||
|
/// `[2]` type size, `[4]` level (0 = store), `[5]` shuffle, `[6]` codec.
|
||||||
|
pub(crate) fn blosc_encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let cd = ctx.client_data();
|
||||||
|
let cerr = |m: &str| FormatError::CompressionError(format!("blosc: {m}"));
|
||||||
|
let typesize = match cd.get(2) {
|
||||||
|
Some(&t) if t != 0 => t as usize,
|
||||||
|
_ => ctx.element_size.max(1),
|
||||||
|
};
|
||||||
|
// Blosc records the type size in one byte; c-blosc treats larger types
|
||||||
|
// as bytes.
|
||||||
|
let typesize = if typesize > 255 { 1 } else { typesize };
|
||||||
|
let level = cd.get(4).copied().unwrap_or(5);
|
||||||
|
let shuffle = cd.get(5).copied().unwrap_or(1);
|
||||||
|
let codec = EncodeCodec::from_cd(cd.get(6).copied().unwrap_or(1))?;
|
||||||
|
let nbytes = input.len();
|
||||||
|
if nbytes > i32::MAX as usize - HEADER {
|
||||||
|
return Err(cerr("chunk too large for a Blosc frame"));
|
||||||
|
}
|
||||||
|
let mut flags = codec.flags();
|
||||||
|
match shuffle {
|
||||||
|
0 => {}
|
||||||
|
1 => flags |= FLAG_SHUFFLE,
|
||||||
|
2 => flags |= FLAG_BITSHUFFLE,
|
||||||
|
other => return Err(cerr(&format!("unknown shuffle mode {other}"))),
|
||||||
|
}
|
||||||
|
let blocksize = encode_block_size(nbytes, typesize, shuffle == 2);
|
||||||
|
let header = |flags: u8, blocksize: usize, cbytes: usize| {
|
||||||
|
let mut h = Vec::with_capacity(HEADER);
|
||||||
|
h.extend_from_slice(&[2, 1, flags, typesize as u8]);
|
||||||
|
h.extend_from_slice(&(nbytes as u32).to_le_bytes());
|
||||||
|
h.extend_from_slice(&(blocksize as u32).to_le_bytes());
|
||||||
|
h.extend_from_slice(&(cbytes as u32).to_le_bytes());
|
||||||
|
h
|
||||||
|
};
|
||||||
|
let stored = || {
|
||||||
|
let mut out = header(
|
||||||
|
FLAG_MEMCPYED | (flags & !(FLAG_SHUFFLE | FLAG_BITSHUFFLE)),
|
||||||
|
blocksize,
|
||||||
|
nbytes + HEADER,
|
||||||
|
);
|
||||||
|
out.extend_from_slice(input);
|
||||||
|
out
|
||||||
|
};
|
||||||
|
if level == 0 || nbytes == 0 {
|
||||||
|
return Ok(stored());
|
||||||
|
}
|
||||||
|
|
||||||
|
let nblocks = nbytes.div_ceil(blocksize);
|
||||||
|
let leftover = nbytes % blocksize;
|
||||||
|
let mut body = Vec::with_capacity(nbytes / 2);
|
||||||
|
let mut starts = Vec::with_capacity(nblocks);
|
||||||
|
let table_end = HEADER + 4 * nblocks;
|
||||||
|
let mut shuffled = vec![0u8; blocksize];
|
||||||
|
for j in 0..nblocks {
|
||||||
|
let is_leftover = j == nblocks - 1 && leftover > 0;
|
||||||
|
let bsize = if is_leftover { leftover } else { blocksize };
|
||||||
|
let block = &input[j * blocksize..j * blocksize + bsize];
|
||||||
|
let sh = &mut shuffled[..bsize];
|
||||||
|
shuffle_block(flags, typesize, block, sh);
|
||||||
|
starts.push(table_end + body.len());
|
||||||
|
let nsplits = if typesize <= MAX_SPLITS
|
||||||
|
&& bsize / typesize >= MIN_BUFFERSIZE
|
||||||
|
&& !is_leftover
|
||||||
|
&& bsize.is_multiple_of(typesize)
|
||||||
|
{
|
||||||
|
typesize
|
||||||
|
} else {
|
||||||
|
1
|
||||||
|
};
|
||||||
|
let neblock = bsize / nsplits;
|
||||||
|
for s in 0..nsplits {
|
||||||
|
let part = &sh[s * neblock..(s + 1) * neblock];
|
||||||
|
let comp = codec.encode(part, level)?;
|
||||||
|
if comp.len() < neblock {
|
||||||
|
body.extend_from_slice(&(comp.len() as u32).to_le_bytes());
|
||||||
|
body.extend_from_slice(&comp);
|
||||||
|
} else {
|
||||||
|
body.extend_from_slice(&(neblock as u32).to_le_bytes());
|
||||||
|
body.extend_from_slice(part);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if table_end + body.len() >= nbytes + HEADER {
|
||||||
|
// Incompressible: store instead, as c-blosc does.
|
||||||
|
return Ok(stored());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// A split block must decode as split: the decoder infers splitting from
|
||||||
|
// the same rule, which requires a whole number of elements per block.
|
||||||
|
let cbytes = table_end + body.len();
|
||||||
|
let mut out = header(flags, blocksize, cbytes);
|
||||||
|
for s in starts {
|
||||||
|
out.extend_from_slice(&(s as u32).to_le_bytes());
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&body);
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Apply the frame's shuffle to one block (the inverse of
|
||||||
|
/// [`unshuffle_block`]).
|
||||||
|
fn shuffle_block(flags: u8, typesize: usize, src: &[u8], dest: &mut [u8]) {
|
||||||
|
let bsize = src.len();
|
||||||
|
if flags & FLAG_SHUFFLE != 0 && typesize > 1 {
|
||||||
|
let n = bsize / typesize;
|
||||||
|
for i in 0..n {
|
||||||
|
for b in 0..typesize {
|
||||||
|
dest[b * n + i] = src[i * typesize + b];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
dest[n * typesize..].copy_from_slice(&src[n * typesize..]);
|
||||||
|
} else if flags & FLAG_BITSHUFFLE != 0 && bsize >= typesize {
|
||||||
|
let n = bsize / typesize;
|
||||||
|
if n.is_multiple_of(8) {
|
||||||
|
let body = n * typesize;
|
||||||
|
bitshuffle_block(&src[..body], &mut dest[..body], n, typesize);
|
||||||
|
dest[body..].copy_from_slice(&src[body..]);
|
||||||
|
} else {
|
||||||
|
dest.copy_from_slice(src);
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
dest.copy_from_slice(src);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::filter_pipeline::{FILTER_BLOSC, FilterDescription};
|
||||||
|
|
||||||
|
/// A blosclz stream: literal "abc", then a 9-byte match 3 back (a run
|
||||||
|
/// of "abc"), then literal "Z".
|
||||||
|
#[test]
|
||||||
|
fn blosclz_decodes_literals_and_overlapping_matches() {
|
||||||
|
// Match: length (ctrl >> 5) + 2 = 8, distance ofs + code + 1 = 3.
|
||||||
|
let stream = [2, b'a', b'b', b'c', (6 << 5), 2, 0, b'Z'];
|
||||||
|
let mut out = [0u8; 12];
|
||||||
|
assert_eq!(blosclz_decompress(&stream, &mut out), 12);
|
||||||
|
assert_eq!(&out, b"abcabcabcabZ");
|
||||||
|
// A stream cut inside a match is malformed.
|
||||||
|
let mut out = [0u8; 11];
|
||||||
|
assert_eq!(blosclz_decompress(&stream[..6], &mut out), 0);
|
||||||
|
// A match before the start of the output is malformed.
|
||||||
|
assert_eq!(blosclz_decompress(&[0, b'a', 32, 5, 0, b'x'], &mut out), 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn desc(cd: Vec<u32>) -> FilterDescription {
|
||||||
|
FilterDescription {
|
||||||
|
filter_id: FILTER_BLOSC,
|
||||||
|
name: None,
|
||||||
|
flags: 0,
|
||||||
|
client_data: cd,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn frame_round_trips_every_codec_and_shuffle() {
|
||||||
|
for ts in [1usize, 2, 4, 8, 3, 32] {
|
||||||
|
for n in [0usize, 5, 100, 1000, 70_000, 300_001] {
|
||||||
|
if n * ts > 1 << 20 && ts > 1 {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let data: Vec<u8> = (0..n * ts)
|
||||||
|
.map(|i| ((i / ts) % 200) as u8 ^ (i % ts) as u8)
|
||||||
|
.collect();
|
||||||
|
for codec in [1u32, 3, 4, 5] {
|
||||||
|
for shuffle in [0u32, 1, 2] {
|
||||||
|
for level in [0u32, 5] {
|
||||||
|
let f = desc(vec![2, 2, ts as u32, 0, level, shuffle, codec]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: ts,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = blosc_encode(&data, &ctx).unwrap();
|
||||||
|
let dec = blosc_decode(&enc, &ctx).unwrap_or_else(|e| {
|
||||||
|
panic!("ts={ts} n={n} codec={codec} shuffle={shuffle}: {e}")
|
||||||
|
});
|
||||||
|
assert!(
|
||||||
|
dec == data,
|
||||||
|
"ts={ts} n={n} codec={codec} shuffle={shuffle} level={level}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rejects_bad_frames() {
|
||||||
|
let data = vec![9u8; 50_000];
|
||||||
|
let f = desc(vec![2, 2, 4, 0, 5, 1, 1]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = blosc_encode(&data, &ctx).unwrap();
|
||||||
|
assert!(blosc_decode(&enc[..enc.len() - 3], &ctx).is_err());
|
||||||
|
let small = FilterContext {
|
||||||
|
max_output: 49_999,
|
||||||
|
..ctx
|
||||||
|
};
|
||||||
|
assert!(blosc_decode(&enc, &small).is_err());
|
||||||
|
let mut v3 = enc.clone();
|
||||||
|
v3[0] = 3;
|
||||||
|
assert!(blosc_decode(&v3, &ctx).is_err());
|
||||||
|
let f0 = desc(vec![2, 2, 4, 0, 5, 1, 0]);
|
||||||
|
let ctx0 = FilterContext { filter: &f0, ..ctx };
|
||||||
|
assert!(blosc_encode(&data, &ctx0).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A frame that declares no data, for a chunk that has some.
|
||||||
|
#[test]
|
||||||
|
fn empty_frame_for_a_non_empty_chunk_is_an_error() {
|
||||||
|
let mut frame = vec![2u8, 1, 0x20, 4];
|
||||||
|
for v in [0u32, 64, 16] {
|
||||||
|
frame.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(blosc_decompress(&frame, 64).unwrap(), b"");
|
||||||
|
let f = desc(vec![2, 2, 4, 64, 5, 1, 1]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: 64,
|
||||||
|
};
|
||||||
|
assert!(blosc_decode(&frame, &ctx).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A frame whose header claims a compressed size smaller than the
|
||||||
|
/// header itself, not stored raw: an error, not an arithmetic overflow
|
||||||
|
/// (it panicked in debug builds).
|
||||||
|
#[test]
|
||||||
|
fn frame_size_below_the_header_is_an_error() {
|
||||||
|
let mut frame = vec![2u8, 1, 1 << 5, 4];
|
||||||
|
for v in [64u32, 64, 8] {
|
||||||
|
frame.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
frame.extend_from_slice(&[0; 40]);
|
||||||
|
assert!(blosc_decompress(&frame, 1000).is_err());
|
||||||
|
for cbytes in 0..16u32 {
|
||||||
|
frame[12..16].copy_from_slice(&cbytes.to_le_bytes());
|
||||||
|
assert!(blosc_decompress(&frame, 1000).is_err(), "cbytes={cbytes}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A BloscLZ frame (our encoder cannot write one): a single block,
|
||||||
|
/// one stream, no shuffle.
|
||||||
|
fn blosclz_frame() -> Vec<u8> {
|
||||||
|
let stream = [2, b'a', b'b', b'c', (6 << 5), 2, 0, b'Z'];
|
||||||
|
let mut f = vec![2u8, 1, 0, 1];
|
||||||
|
for v in [12u32, 12, (HEADER + 4 + 4 + stream.len()) as u32] {
|
||||||
|
f.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
f.extend_from_slice(&((HEADER + 4) as u32).to_le_bytes());
|
||||||
|
f.extend_from_slice(&(stream.len() as u32).to_le_bytes());
|
||||||
|
f.extend_from_slice(&stream);
|
||||||
|
f
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Random and mutated frames, every codec and shuffle: errors are fine,
|
||||||
|
/// panics are not.
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_frames_never_panic() {
|
||||||
|
let limit = 6000;
|
||||||
|
let data: Vec<u8> = (0..1500u32).flat_map(|i| (i / 5).to_le_bytes()).collect();
|
||||||
|
let mut seeds = vec![blosclz_frame()];
|
||||||
|
for codec in [1u32, 3, 4, 5] {
|
||||||
|
for shuffle in [0u32, 1, 2] {
|
||||||
|
for (ts, n) in [(4usize, data.len()), (4, 520), (1, 300), (2, 4)] {
|
||||||
|
let f = desc(vec![2, 2, ts as u32, 0, 5, shuffle, codec]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: ts,
|
||||||
|
max_output: n,
|
||||||
|
};
|
||||||
|
seeds.push(blosc_encode(&data[..n], &ctx).unwrap());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Stored raw.
|
||||||
|
let f = desc(vec![2, 2, 4, 0, 0, 1, 1]);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: 64,
|
||||||
|
};
|
||||||
|
seeds.push(blosc_encode(&data[..64], &ctx).unwrap());
|
||||||
|
crate::test_fuzz::fuzz_decoder(0xb10, &seeds, 30_000, limit, |s| {
|
||||||
|
blosc_decompress(s, limit)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// BloscLZ streams on their own, random and mutated.
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_blosclz_streams_never_panic() {
|
||||||
|
let seed = blosclz_frame()[HEADER + 8..].to_vec();
|
||||||
|
let mut out = [0u8; 64];
|
||||||
|
crate::test_fuzz::fuzz_decoder(0xb11, &[seed], 30_000, 64, |s| {
|
||||||
|
let n = blosclz_decompress(s, &mut out);
|
||||||
|
if n == 0 {
|
||||||
|
Err(err("malformed"))
|
||||||
|
} else {
|
||||||
|
Ok(out[..n].to_vec())
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,133 @@
|
|||||||
|
//! bzip2 (HDF5 filter 307, PyTables' `H5Zbzip2.c`, hdf5plugin's `BZip2`).
|
||||||
|
//!
|
||||||
|
//! The chunk is one bzip2 stream; `cd_values[0]` is the block size (1-9,
|
||||||
|
//! the compression level). Decoded with the `bzip2` crate's default backend,
|
||||||
|
//! `libbz2-rs-sys`, a pure-Rust port of libbzip2.
|
||||||
|
|
||||||
|
use crate::addr::saturating_usize;
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_registry::FilterContext;
|
||||||
|
|
||||||
|
fn err(msg: &str) -> FormatError {
|
||||||
|
FormatError::DecompressionError(format!("bzip2: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode a bzip2-filtered chunk, refusing output beyond the chunk size.
|
||||||
|
pub(crate) fn bzip2_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
use bzip2::{Decompress, Status};
|
||||||
|
let limit = ctx.output_limit();
|
||||||
|
let max_capacity = limit.saturating_add(1);
|
||||||
|
let hint = if ctx.max_output != 0 {
|
||||||
|
ctx.max_output
|
||||||
|
} else {
|
||||||
|
input.len().saturating_mul(4)
|
||||||
|
};
|
||||||
|
let mut out = Vec::new();
|
||||||
|
out.try_reserve_exact(hint.clamp(1, max_capacity))
|
||||||
|
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||||
|
let mut dec = Decompress::new(false);
|
||||||
|
loop {
|
||||||
|
let (in_before, out_before) = (dec.total_in(), dec.total_out());
|
||||||
|
let status = dec
|
||||||
|
.decompress_vec(&input[saturating_usize(in_before)..], &mut out)
|
||||||
|
.map_err(|e| err(&e.to_string()))?;
|
||||||
|
if out.len() > limit {
|
||||||
|
return Err(err("output exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
if status == Status::StreamEnd {
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
if out.len() == out.capacity() {
|
||||||
|
let grow = out
|
||||||
|
.capacity()
|
||||||
|
.min(max_capacity.saturating_sub(out.capacity()))
|
||||||
|
.max(1);
|
||||||
|
out.try_reserve_exact(grow)
|
||||||
|
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||||
|
} else if saturating_usize(dec.total_in()) >= input.len()
|
||||||
|
|| (dec.total_in(), dec.total_out()) == (in_before, out_before)
|
||||||
|
{
|
||||||
|
return Err(err("truncated stream"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a chunk as one bzip2 stream at block size `cd_values[0]`
|
||||||
|
/// (default 9, as hdf5plugin).
|
||||||
|
pub(crate) fn bzip2_encode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
use bzip2::{Action, Compress, Compression, Status};
|
||||||
|
let level = ctx.client_data().first().copied().unwrap_or(9).clamp(1, 9);
|
||||||
|
let cerr = |m: String| FormatError::CompressionError(format!("bzip2: {m}"));
|
||||||
|
let mut enc = Compress::new(Compression::new(level), 0);
|
||||||
|
// bzip2's worst case is about 1% + 600 bytes over the input.
|
||||||
|
let mut out = Vec::with_capacity(input.len() + input.len() / 100 + 600);
|
||||||
|
loop {
|
||||||
|
let consumed = saturating_usize(enc.total_in());
|
||||||
|
let status = enc
|
||||||
|
.compress_vec(&input[consumed..], &mut out, Action::Finish)
|
||||||
|
.map_err(|e| cerr(e.to_string()))?;
|
||||||
|
if status == Status::StreamEnd {
|
||||||
|
return Ok(out);
|
||||||
|
}
|
||||||
|
if out.len() == out.capacity() {
|
||||||
|
out.reserve(out.capacity().max(4096));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::filter_pipeline::{FILTER_BZIP2, FilterDescription};
|
||||||
|
|
||||||
|
fn desc(level: u32) -> FilterDescription {
|
||||||
|
FilterDescription {
|
||||||
|
filter_id: FILTER_BZIP2,
|
||||||
|
name: None,
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![level],
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn round_trips_and_bounds() {
|
||||||
|
let data: Vec<u8> = (0..100_000u32)
|
||||||
|
.flat_map(|i| (i % 777).to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
for level in [1, 5, 9] {
|
||||||
|
let f = desc(level);
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let enc = bzip2_encode(&data, &ctx).unwrap();
|
||||||
|
assert!(enc.len() < data.len() / 4);
|
||||||
|
assert_eq!(bzip2_decode(&enc, &ctx).unwrap(), data);
|
||||||
|
// Truncated, and larger than the chunk: errors, not data.
|
||||||
|
assert!(bzip2_decode(&enc[..enc.len() / 2], &ctx).is_err());
|
||||||
|
let small = FilterContext {
|
||||||
|
max_output: data.len() - 1,
|
||||||
|
..ctx
|
||||||
|
};
|
||||||
|
assert!(bzip2_decode(&enc, &small).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Random and mutated streams: errors are fine, panics are not.
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_streams_never_panic() {
|
||||||
|
let f = desc(9);
|
||||||
|
let data: Vec<u8> = (0..4000u32).flat_map(|i| (i % 91).to_le_bytes()).collect();
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter: &f,
|
||||||
|
element_size: 4,
|
||||||
|
max_output: data.len(),
|
||||||
|
};
|
||||||
|
let seeds = vec![
|
||||||
|
bzip2_encode(&data, &ctx).unwrap(),
|
||||||
|
bzip2_encode(&data[..40], &ctx).unwrap(),
|
||||||
|
];
|
||||||
|
crate::test_fuzz::fuzz_decoder(0xb2, &seeds, 3_000, data.len(), |s| bzip2_decode(s, &ctx));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,260 @@
|
|||||||
|
//! LZF (HDF5 filter 32000) — h5py's built-in compression filter
|
||||||
|
//! (`compression="lzf"`), in pure Rust.
|
||||||
|
//!
|
||||||
|
//! The chunk is one raw LZF stream (liblzf 3.x format, no header). The
|
||||||
|
//! stream is a sequence of instructions, each starting with a control byte:
|
||||||
|
//!
|
||||||
|
//! * `000LLLLL` — a literal run: the next `L + 1` bytes (1..=32) are copied.
|
||||||
|
//! * `LLLOOOOO [E] OOOOOOOO` — a back reference: copy `len + 2` bytes from
|
||||||
|
//! `distance` bytes back, where `len` is the top three bits (1..=6), or
|
||||||
|
//! `7 + E` when they are all ones, and `distance` is the 13-bit offset
|
||||||
|
//! (high five bits in the control byte, low eight in the last byte) plus 1.
|
||||||
|
//!
|
||||||
|
//! h5py's filter (`lzf_filter.c`) records the chunk's size in bytes in
|
||||||
|
//! `cd_values[2]` (slots 0 and 1 hold the filter and liblzf versions) and
|
||||||
|
//! sizes its output buffer from it.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_registry::FilterContext;
|
||||||
|
|
||||||
|
/// `H5PY_FILTER_LZF_VERSION`, written to `cd_values[0]`.
|
||||||
|
pub const LZF_FILTER_VERSION: u32 = 4;
|
||||||
|
/// `LZF_VERSION` (liblzf 1.5), written to `cd_values[1]`.
|
||||||
|
pub const LZF_API_VERSION: u32 = 0x0105;
|
||||||
|
|
||||||
|
const MAX_LITERAL: usize = 32;
|
||||||
|
const MAX_OFFSET: usize = 1 << 13;
|
||||||
|
const MAX_REF: usize = (1 << 8) + (1 << 3);
|
||||||
|
const HASH_LOG: u32 = 14;
|
||||||
|
|
||||||
|
fn err(msg: &str) -> FormatError {
|
||||||
|
FormatError::DecompressionError(format!("lzf: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode an LZF-filtered chunk.
|
||||||
|
pub(crate) fn lzf_decode(input: &[u8], ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let limit = ctx.output_limit();
|
||||||
|
let hint = match ctx.client_data().get(2) {
|
||||||
|
Some(&n) if n != 0 => n as usize,
|
||||||
|
_ => input.len().saturating_mul(2),
|
||||||
|
};
|
||||||
|
lzf_decompress(input, hint.min(limit), limit)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decompress a raw LZF stream, refusing to produce more than `limit` bytes.
|
||||||
|
pub fn lzf_decompress(
|
||||||
|
input: &[u8],
|
||||||
|
size_hint: usize,
|
||||||
|
limit: usize,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let mut out: Vec<u8> = Vec::new();
|
||||||
|
out.try_reserve(size_hint)
|
||||||
|
.map_err(|_| err("cannot allocate the output buffer"))?;
|
||||||
|
let mut ip = 0usize;
|
||||||
|
while ip < input.len() {
|
||||||
|
let ctrl = input[ip] as usize;
|
||||||
|
ip += 1;
|
||||||
|
if ctrl < 32 {
|
||||||
|
let run = ctrl + 1;
|
||||||
|
let lit = input
|
||||||
|
.get(ip..ip + run)
|
||||||
|
.ok_or_else(|| err("literal run past the end of the input"))?;
|
||||||
|
if out.len() + run > limit {
|
||||||
|
return Err(err("output exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
out.extend_from_slice(lit);
|
||||||
|
ip += run;
|
||||||
|
} else {
|
||||||
|
let mut len = ctrl >> 5;
|
||||||
|
if len == 7 {
|
||||||
|
len += *input
|
||||||
|
.get(ip)
|
||||||
|
.ok_or_else(|| err("truncated back reference"))?
|
||||||
|
as usize;
|
||||||
|
ip += 1;
|
||||||
|
}
|
||||||
|
let low = *input
|
||||||
|
.get(ip)
|
||||||
|
.ok_or_else(|| err("truncated back reference"))? as usize;
|
||||||
|
ip += 1;
|
||||||
|
let distance = ((ctrl & 0x1f) << 8) + low + 1;
|
||||||
|
let len = len + 2;
|
||||||
|
if distance > out.len() {
|
||||||
|
return Err(err("back reference before the start of the output"));
|
||||||
|
}
|
||||||
|
if out.len() + len > limit {
|
||||||
|
return Err(err("output exceeds the chunk size"));
|
||||||
|
}
|
||||||
|
let start = out.len() - distance;
|
||||||
|
if distance >= len {
|
||||||
|
out.extend_from_within(start..start + len);
|
||||||
|
} else {
|
||||||
|
// Overlapping copy: repeats the last `distance` bytes.
|
||||||
|
for k in 0..len {
|
||||||
|
let b = out[start + k];
|
||||||
|
out.push(b);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Encode a chunk with the LZF filter.
|
||||||
|
pub(crate) fn lzf_encode(input: &[u8], _ctx: &FilterContext<'_>) -> Result<Vec<u8>, FormatError> {
|
||||||
|
Ok(lzf_compress(input))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hash3(b: &[u8]) -> usize {
|
||||||
|
let v = (u32::from(b[0]) << 16) | (u32::from(b[1]) << 8) | u32::from(b[2]);
|
||||||
|
(v.wrapping_mul(2_654_435_761) >> (32 - HASH_LOG)) as usize
|
||||||
|
}
|
||||||
|
|
||||||
|
fn flush_literals(out: &mut Vec<u8>, lit: &[u8]) {
|
||||||
|
for run in lit.chunks(MAX_LITERAL) {
|
||||||
|
out.push((run.len() - 1) as u8);
|
||||||
|
out.extend_from_slice(run);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compress `input` into a raw LZF stream any liblzf decoder reads.
|
||||||
|
///
|
||||||
|
/// Incompressible input grows by one byte per 32. (h5py's own filter gives
|
||||||
|
/// up on such a chunk and stores it unfiltered; storing the slightly larger
|
||||||
|
/// stream is equally readable.)
|
||||||
|
pub fn lzf_compress(input: &[u8]) -> Vec<u8> {
|
||||||
|
let n = input.len();
|
||||||
|
let mut out = Vec::with_capacity(n + n / MAX_LITERAL + 1);
|
||||||
|
let mut table = vec![0u32; 1 << HASH_LOG];
|
||||||
|
let mut lit_start = 0usize;
|
||||||
|
let mut i = 0usize;
|
||||||
|
while i + 2 < n {
|
||||||
|
let h = hash3(&input[i..]);
|
||||||
|
let cand = table[h] as usize;
|
||||||
|
table[h] = (i + 1) as u32;
|
||||||
|
if cand != 0 {
|
||||||
|
let r = cand - 1;
|
||||||
|
let distance = i - r;
|
||||||
|
if distance <= MAX_OFFSET && input[r..r + 3] == input[i..i + 3] {
|
||||||
|
let max_len = (n - i).min(MAX_REF);
|
||||||
|
let mut len = 3;
|
||||||
|
while len < max_len && input[r + len] == input[i + len] {
|
||||||
|
len += 1;
|
||||||
|
}
|
||||||
|
flush_literals(&mut out, &input[lit_start..i]);
|
||||||
|
let code = len - 2;
|
||||||
|
let off = distance - 1;
|
||||||
|
if code < 7 {
|
||||||
|
out.push(((code << 5) | (off >> 8)) as u8);
|
||||||
|
} else {
|
||||||
|
out.push(((7 << 5) | (off >> 8)) as u8);
|
||||||
|
out.push((code - 7) as u8);
|
||||||
|
}
|
||||||
|
out.push((off & 0xff) as u8);
|
||||||
|
// Index the positions the match covered so later data can
|
||||||
|
// refer back into it.
|
||||||
|
let end = i + len;
|
||||||
|
let mut j = i + 1;
|
||||||
|
while j < end && j + 2 < n {
|
||||||
|
table[hash3(&input[j..])] = (j + 1) as u32;
|
||||||
|
j += 1;
|
||||||
|
}
|
||||||
|
i = end;
|
||||||
|
lit_start = i;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
flush_literals(&mut out, &input[lit_start..]);
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn round_trip(data: &[u8]) {
|
||||||
|
let c = lzf_compress(data);
|
||||||
|
assert_eq!(lzf_decompress(&c, data.len(), data.len()).unwrap(), data);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn round_trips() {
|
||||||
|
round_trip(b"");
|
||||||
|
round_trip(b"a");
|
||||||
|
round_trip(b"abcabcabcabcabcabcabcabcabcabcabcabc");
|
||||||
|
round_trip(&[7u8; 10_000]);
|
||||||
|
let noise: Vec<u8> = (0..70_000u32)
|
||||||
|
.map(|i| (i.wrapping_mul(2_654_435_761) >> 13) as u8)
|
||||||
|
.collect();
|
||||||
|
round_trip(&noise);
|
||||||
|
let ramp: Vec<u8> = (0..100_000u32)
|
||||||
|
.flat_map(|i| (i % 1000).to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
round_trip(&ramp);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn compresses_repetitive_data() {
|
||||||
|
let data = [42u8; 4096];
|
||||||
|
assert!(lzf_compress(&data).len() < 100);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The chunk h5py 3.16's bundled liblzf writes for
|
||||||
|
/// `b"hello hello hello hello"` (read back with `read_direct_chunk`): a
|
||||||
|
/// 7-byte literal, a 14-byte back reference 6 bytes back (extended
|
||||||
|
/// length), and a 2-byte literal.
|
||||||
|
#[test]
|
||||||
|
fn decodes_liblzf_output() {
|
||||||
|
let stream = b"\x06hello h\xe0\x05\x05\x01lo";
|
||||||
|
assert_eq!(
|
||||||
|
lzf_decompress(stream, 23, 23).unwrap(),
|
||||||
|
b"hello hello hello hello"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rejects_corrupt_streams() {
|
||||||
|
// Back reference before the start.
|
||||||
|
assert!(lzf_decompress(&[0x20, 0x00], 10, 10).is_err());
|
||||||
|
// Literal run past the end.
|
||||||
|
assert!(lzf_decompress(&[0x05, 1, 2], 10, 10).is_err());
|
||||||
|
// Output over the limit.
|
||||||
|
let c = lzf_compress(&[1u8; 100]);
|
||||||
|
assert!(lzf_decompress(&c, 10, 99).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Random and mutated streams: errors are fine, panics are not.
|
||||||
|
#[test]
|
||||||
|
fn fuzzed_streams_never_panic() {
|
||||||
|
let seeds: Vec<Vec<u8>> = [
|
||||||
|
b"hello hello hello hello".to_vec(),
|
||||||
|
vec![7u8; 3000],
|
||||||
|
(0..2000u32).flat_map(|i| (i % 37).to_le_bytes()).collect(),
|
||||||
|
(0..500u32)
|
||||||
|
.map(|i| (i.wrapping_mul(2_654_435_761) >> 13) as u8)
|
||||||
|
.collect(),
|
||||||
|
]
|
||||||
|
.iter()
|
||||||
|
.map(|d| lzf_compress(d))
|
||||||
|
.collect();
|
||||||
|
for limit in [0usize, 23, 4096, 8000] {
|
||||||
|
crate::test_fuzz::fuzz_decoder(
|
||||||
|
0x1f2 + limit as u64,
|
||||||
|
&seeds[..1],
|
||||||
|
5_000,
|
||||||
|
limit.max(23),
|
||||||
|
|s| lzf_decompress(s, limit, limit.max(23)),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
crate::test_fuzz::fuzz_decoder(0x1f3, &seeds, 20_000, 8000, |s| {
|
||||||
|
lzf_decompress(s, 8000, 8000)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,19 +1,40 @@
|
|||||||
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
||||||
//!
|
//!
|
||||||
//! Gated by the `szip` feature which links against the system libaec library.
|
//! Gated by the `szip` feature which links against the system libaec library.
|
||||||
|
//!
|
||||||
|
//! libhdf5's SZIP filter (`H5Zszip.c`) prefixes each chunk with its
|
||||||
|
//! uncompressed size and hands the rest to szlib's `SZ_BufftoBuffDecompress`.
|
||||||
|
//! libaec implements that call (`sz_compat.c`) on top of `aec_buffer_decode`
|
||||||
|
//! with some reshaping — 32/64-bit samples are coded as byte planes of 8-bit
|
||||||
|
//! samples, and scanlines that are not a whole number of blocks are padded —
|
||||||
|
//! which [`szip_decompress`] reproduces so its output matches libhdf5's.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
/// Decompress SZIP-compressed data using libaec.
|
/// `SZ_MSB_OPTION_MASK`: samples are big-endian.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
const SZ_MSB_OPTION_MASK: u32 = 16;
|
||||||
|
/// `SZ_NN_OPTION_MASK`: nearest-neighbour preprocessing.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
const SZ_NN_OPTION_MASK: u32 = 32;
|
||||||
|
|
||||||
|
/// Decompress one SZIP-filtered chunk.
|
||||||
///
|
///
|
||||||
/// `cd` is the HDF5 SZIP filter client data (matches `H5Z_SZIP_PARM_*` indices):
|
/// `cd` is the HDF5 SZIP filter client data (`H5Z_SZIP_PARM_*` indices):
|
||||||
/// cd[0] = options mask (`H5_SZIP_NN_OPTION_MASK = 0x20` enables NN preprocessing)
|
/// cd[0] = options mask (`SZ_*_OPTION_MASK`: 16 = MSB byte order,
|
||||||
/// cd[1] = pixels per block (H5Z_SZIP_PARM_PPB; 8, 10, 16, or 32)
|
/// 32 = nearest-neighbour preprocessing; K13/EC/LSB/RAW bits carry
|
||||||
/// cd[2] = bits per sample (H5Z_SZIP_PARM_BPP; element bit width)
|
/// no decoding information for libaec)
|
||||||
/// cd[3] = pixels per scan line (H5Z_SZIP_PARM_PPS; informational only)
|
/// cd[1] = pixels per block
|
||||||
|
/// cd[2] = bits per pixel (sample precision, rounded up to 32 or 64 above
|
||||||
|
/// 24 by libhdf5)
|
||||||
|
/// cd[3] = pixels per scanline
|
||||||
|
///
|
||||||
|
/// The chunk is a 4-byte little-endian uncompressed size followed by the
|
||||||
|
/// szlib stream.
|
||||||
|
#[cfg_attr(not(feature = "szip"), allow(dead_code))]
|
||||||
pub(crate) fn szip_decompress(
|
pub(crate) fn szip_decompress(
|
||||||
_data: &[u8],
|
_data: &[u8],
|
||||||
_cd: &[u32],
|
_cd: &[u32],
|
||||||
@@ -33,62 +54,174 @@ pub(crate) fn szip_decompress(
|
|||||||
|
|
||||||
#[cfg(feature = "szip")]
|
#[cfg(feature = "szip")]
|
||||||
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
if cd.len() < 3 {
|
let err = |m: &str| FormatError::ChunkedReadError(format!("szip: {m}"));
|
||||||
return Err(FormatError::ChunkedReadError(
|
if cd.len() < 4 {
|
||||||
"szip: missing client data".into(),
|
return Err(err("missing client data"));
|
||||||
));
|
|
||||||
}
|
}
|
||||||
let options = cd[0];
|
let options = cd[0];
|
||||||
let pixels_per_block = cd[1];
|
let pixels_per_block = cd[1] as usize;
|
||||||
let bits_per_sample = cd[2]; // H5Z_SZIP_PARM_BPP
|
let bits_per_pixel = cd[2];
|
||||||
if bits_per_sample == 0 || bits_per_sample > 32 {
|
let pixels_per_scanline = cd[3] as usize;
|
||||||
return Err(FormatError::ChunkedReadError(
|
if !(1..=32).contains(&bits_per_pixel) && bits_per_pixel != 64 {
|
||||||
"szip: invalid bits per sample".into(),
|
return Err(err("invalid bits per sample"));
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if chunk_size == 0 {
|
if pixels_per_block == 0 || pixels_per_scanline == 0 {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(err("invalid block or scanline size"));
|
||||||
"szip: unknown output size".into(),
|
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if data.is_empty() {
|
if data.len() < 4 {
|
||||||
return Err(FormatError::ChunkedReadError("szip: empty input".into()));
|
return Err(err("chunk too short"));
|
||||||
}
|
}
|
||||||
|
// H5Zszip.c: UINT32DECODE of the uncompressed size, then the stream.
|
||||||
|
let dest_len = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||||
|
let limit = if chunk_size != 0 {
|
||||||
|
chunk_size
|
||||||
|
} else {
|
||||||
|
crate::filters::MAX_DECOMPRESS_SIZE
|
||||||
|
};
|
||||||
|
if dest_len > limit {
|
||||||
|
return Err(err("declared size exceeds chunk size"));
|
||||||
|
}
|
||||||
|
let stream = &data[4..];
|
||||||
|
|
||||||
// Map HDF5 option mask to libaec flags.
|
// --- libaec sz_compat.c: SZ_BufftoBuffDecompress ---
|
||||||
// HDF5 always stores SZIP data in MSB order, so AEC_DATA_MSB is unconditional.
|
let rsi = pixels_per_scanline.div_ceil(pixels_per_block);
|
||||||
// H5_SZIP_NN_OPTION_MASK (0x20): NN differential preprocessing.
|
let mut flags = 0;
|
||||||
let mut flags: u32 = libaec_sys::AEC_DATA_MSB;
|
if options & SZ_MSB_OPTION_MASK != 0 {
|
||||||
if options & 0x20 != 0 {
|
flags |= libaec_sys::AEC_DATA_MSB;
|
||||||
|
}
|
||||||
|
if options & SZ_NN_OPTION_MASK != 0 {
|
||||||
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
||||||
}
|
}
|
||||||
|
let pad_scanline = !pixels_per_scanline.is_multiple_of(pixels_per_block);
|
||||||
|
let deinterleave = bits_per_pixel == 32 || bits_per_pixel == 64;
|
||||||
|
let bits_per_sample = if deinterleave { 8 } else { bits_per_pixel };
|
||||||
|
let pixel_size = match bits_per_sample {
|
||||||
|
17.. => 4,
|
||||||
|
9.. => 2,
|
||||||
|
_ => 1,
|
||||||
|
};
|
||||||
|
let scanlines = (dest_len / pixel_size).div_ceil(pixels_per_scanline);
|
||||||
|
let buf_size = if pad_scanline {
|
||||||
|
rsi.checked_mul(pixels_per_block)
|
||||||
|
.and_then(|n| n.checked_mul(pixel_size))
|
||||||
|
.and_then(|n| n.checked_mul(scanlines))
|
||||||
|
.filter(|&n| n <= crate::filters::MAX_DECOMPRESS_SIZE.max(limit))
|
||||||
|
.ok_or_else(|| err("scanline padding too large"))?
|
||||||
|
} else {
|
||||||
|
dest_len
|
||||||
|
};
|
||||||
|
|
||||||
let mut out = vec![0u8; chunk_size];
|
let mut buf = vec![0u8; buf_size];
|
||||||
let mut strm = libaec_sys::AecStream::zeroed();
|
let mut strm = libaec_sys::AecStream::zeroed();
|
||||||
strm.next_in = data.as_ptr();
|
strm.next_in = stream.as_ptr();
|
||||||
strm.avail_in = data.len();
|
strm.avail_in = stream.len();
|
||||||
strm.next_out = out.as_mut_ptr();
|
strm.next_out = buf.as_mut_ptr();
|
||||||
strm.avail_out = chunk_size;
|
strm.avail_out = buf_size;
|
||||||
strm.bits_per_sample = bits_per_sample;
|
strm.bits_per_sample = bits_per_sample;
|
||||||
strm.block_size = pixels_per_block;
|
strm.block_size = pixels_per_block as u32;
|
||||||
strm.rsi = 128; // HDF5 default: 128 blocks per reference sample interval
|
strm.rsi = rsi as u32;
|
||||||
strm.flags = flags;
|
strm.flags = flags;
|
||||||
|
// SAFETY: next_in/avail_in and next_out/avail_out describe live buffers
|
||||||
|
// (`stream` and `buf`) that outlive the call.
|
||||||
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
||||||
if result != 0 {
|
if result != 0 {
|
||||||
return Err(FormatError::DecompressionError(format!(
|
return Err(FormatError::DecompressionError(format!(
|
||||||
"szip: libaec error {result}"
|
"szip: libaec error {result}"
|
||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
let decoded_len = chunk_size - strm.avail_out;
|
let mut total_out = strm.total_out;
|
||||||
out.truncate(decoded_len);
|
if pad_scanline {
|
||||||
Ok(out)
|
let line = pixels_per_scanline * pixel_size;
|
||||||
|
let padded_line = rsi * pixels_per_block * pixel_size;
|
||||||
|
// remove_padding: compact each padded line down to `line` bytes.
|
||||||
|
let mut i = line;
|
||||||
|
let mut j = padded_line;
|
||||||
|
while j < total_out {
|
||||||
|
let end = (j + line).min(buf.len());
|
||||||
|
buf.copy_within(j..end, i);
|
||||||
|
i += line;
|
||||||
|
j += padded_line;
|
||||||
|
}
|
||||||
|
total_out = scanlines * line;
|
||||||
|
}
|
||||||
|
if total_out < dest_len {
|
||||||
|
return Err(err("stream decoded to fewer bytes than declared"));
|
||||||
|
}
|
||||||
|
buf.truncate(dest_len);
|
||||||
|
if deinterleave {
|
||||||
|
// deinterleave_buffer: byte planes back into words.
|
||||||
|
let w = (bits_per_pixel / 8) as usize;
|
||||||
|
let n = dest_len / w;
|
||||||
|
let mut out = vec![0u8; dest_len];
|
||||||
|
for i in 0..n {
|
||||||
|
for j in 0..w {
|
||||||
|
out[i * w + j] = buf[j * n + i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
} else {
|
||||||
|
Ok(buf)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
fn unhex(s: &str) -> Vec<u8> {
|
||||||
|
(0..s.len())
|
||||||
|
.step_by(2)
|
||||||
|
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// SZIP chunks written by libhdf5, decoded exactly as libhdf5 decodes
|
||||||
|
/// them. Each case: fixture, chunk byte offset and size (from h5py's
|
||||||
|
/// `get_chunk_info`), the filter's cd_values, and the chunk's values as
|
||||||
|
/// h5py reads them (file byte order, hex). Before the fix every one of
|
||||||
|
/// these came back as garbage or zeros (or "invalid bits per sample" for
|
||||||
|
/// 64-bit): the 4-byte size prefix was fed to libaec, 32/64-bit samples
|
||||||
|
/// were not de-interleaved from byte planes, the reference sample
|
||||||
|
/// interval was fixed at 128 instead of derived from the scanline, padded
|
||||||
|
/// scanlines were not unpadded, and LE data was decoded as MSB.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
#[test]
|
||||||
|
fn szip_decodes_libhdf5_chunks_exactly() {
|
||||||
|
/// (name, file, chunk offset, chunk size, cd_values, decoded hex)
|
||||||
|
type Case<'a> = (&'a str, &'a [u8], usize, usize, [u32; 4], &'a str);
|
||||||
|
let noencoder: &[u8] = include_bytes!("../tests/fixtures/filters/noencoder.h5");
|
||||||
|
let le_data: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||||
|
let h5py: &[u8] = include_bytes!("../tests/fixtures/filters/szip_h5py.h5");
|
||||||
|
#[rustfmt::skip]
|
||||||
|
let cases: &[Case] = &[
|
||||||
|
// <i4, 10 px/scanline over 4 px/block: padded scanlines + byte planes.
|
||||||
|
("noencoder /noencoder_szip_dset.h5", noencoder, 6040, 16, [168, 4, 32, 10],
|
||||||
|
"00000000010000000200000003000000040000000500000006000000070000000800000009000000"),
|
||||||
|
// <f4, LSB + NN.
|
||||||
|
("le_data /Szip_float_data_le", le_data, 55224, 48, [169, 4, 32, 12],
|
||||||
|
"abaaaa3eabaa2a3f0000803fabaa2a3f0000803fabaaaa3f0000803fabaaaa3f5555d53fabaaaa3f5555d53f00000040"),
|
||||||
|
// >f4, MSB + NN.
|
||||||
|
("le_data /Szip_float_data_be", le_data, 55396, 48, [177, 4, 32, 12],
|
||||||
|
"3eaaaaab3f2aaaab3f8000003f2aaaab3f8000003faaaaab3f8000003faaaaab3fd555553faaaaab3fd5555540000000"),
|
||||||
|
// <f8 (64-bit), NN.
|
||||||
|
("szip_h5py /f8", h5py, 4016, 100, [169, 8, 64, 10],
|
||||||
|
"00000000000008c000000000000008c000000000000008c000000000000008c000000000000004c000000000000004c000000000000004c000000000000004c000000000000000c000000000000000c000000000000000c000000000000000c0000000000000f8bf000000000000f8bf000000000000f8bf000000000000f8bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000e0bf000000000000e0bf000000000000e0bf000000000000e0bf0000000000000000000000000000000000000000000000000000000000000000000000000000e03f000000000000e03f000000000000e03f000000000000e03f000000000000f03f000000000000f03f000000000000f03f000000000000f03f000000000000f83f000000000000f83f000000000000f83f000000000000f83f"),
|
||||||
|
// <i8 (64-bit), entropy coding without NN.
|
||||||
|
("szip_h5py /i8", h5py, 4188, 53, [141, 4, 64, 10],
|
||||||
|
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c00000000000000"),
|
||||||
|
// <u2, 35 px/scanline over 8 px/block: padded scanlines, 16-bit samples.
|
||||||
|
("szip_h5py /u2", h5py, 4308, 43, [169, 8, 16, 35],
|
||||||
|
"00000000000000006100610061006100c200c200c200c20023012301230123018401840184018401e501e501e501e5014602460246024602a702a702a702a702080308030803"),
|
||||||
|
];
|
||||||
|
for (name, file, off, len, cd, want) in cases {
|
||||||
|
let want = unhex(want);
|
||||||
|
let got = szip_decompress(&file[*off..off + len], cd, want.len())
|
||||||
|
.unwrap_or_else(|e| panic!("{name}: {e:?}"));
|
||||||
|
assert_eq!(got, want, "{name}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn szip_disabled_returns_unsupported() {
|
fn szip_disabled_returns_unsupported() {
|
||||||
#[cfg(not(feature = "szip"))]
|
#[cfg(not(feature = "szip"))]
|
||||||
@@ -132,6 +265,8 @@ mod tests {
|
|||||||
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
||||||
let enc_len = encoded.len() - enc.avail_out;
|
let enc_len = encoded.len() - enc.avail_out;
|
||||||
encoded.truncate(enc_len);
|
encoded.truncate(enc_len);
|
||||||
|
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||||
|
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||||
|
|
||||||
// Decode through our public interface.
|
// Decode through our public interface.
|
||||||
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
||||||
@@ -163,6 +298,8 @@ mod tests {
|
|||||||
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
||||||
let enc_len = encoded.len() - enc.avail_out;
|
let enc_len = encoded.len() - enc.avail_out;
|
||||||
encoded.truncate(enc_len);
|
encoded.truncate(enc_len);
|
||||||
|
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||||
|
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||||
|
|
||||||
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
||||||
let cd = [0x20u32, 8, 8, 1024];
|
let cd = [0x20u32, 8, 8, 1024];
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -6,18 +6,23 @@ extern crate alloc;
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{format, vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
|
use crate::chunk_grid::ChunkGrid;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{PAGED_BLOCK_ONE_READ_MAX, Storage, Window, len_usize, read_exact_at};
|
||||||
|
|
||||||
/// Verify the Jenkins lookup3 checksum stored immediately after
|
/// Verify the Jenkins lookup3 checksum stored immediately after
|
||||||
/// `data[start..end]`, as every Fixed Array structure carries one.
|
/// `data[start..end]`, as every Fixed Array structure carries one. `w` is
|
||||||
|
/// a window of the file and `start`/`end` are relative to it.
|
||||||
///
|
///
|
||||||
/// A corrupt chunk index silently yields addresses pointing at the wrong
|
/// A corrupt chunk index silently yields addresses pointing at the wrong
|
||||||
/// bytes, so a mismatch has to be an error rather than a shrug: without this
|
/// bytes, so a mismatch has to be an error rather than a shrug: without this
|
||||||
/// the damage surfaces as plausible-looking data from the wrong chunk.
|
/// the damage surfaces as plausible-looking data from the wrong chunk.
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatError> {
|
fn verify_checksum(w: &Window<'_>, start: usize, end: usize) -> Result<(), FormatError> {
|
||||||
ensure_len(data, end, 4)?;
|
w.ensure(end, 4)?;
|
||||||
|
let data = &w.bytes;
|
||||||
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
let stored = u32::from_le_bytes([data[end], data[end + 1], data[end + 2], data[end + 3]]);
|
||||||
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
let computed = crate::checksum::jenkins_lookup3(&data[start..end]);
|
||||||
if computed != stored {
|
if computed != stored {
|
||||||
@@ -30,7 +35,7 @@ fn verify_checksum(data: &[u8], start: usize, end: usize) -> Result<(), FormatEr
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "checksum"))]
|
#[cfg(not(feature = "checksum"))]
|
||||||
fn verify_checksum(_data: &[u8], _start: usize, _end: usize) -> Result<(), FormatError> {
|
fn verify_checksum(_w: &Window<'_>, _start: usize, _end: usize) -> Result<(), FormatError> {
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -72,19 +77,6 @@ fn read_length(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
|||||||
read_offset(data, pos, size)
|
read_offset(data, pos, size)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
|
||||||
if offset
|
|
||||||
.checked_add(needed)
|
|
||||||
.is_none_or(|end| end > data.len())
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: offset.saturating_add(needed),
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn is_undefined(data: &[u8], pos: usize, size: u8) -> bool {
|
fn is_undefined(data: &[u8], pos: usize, size: u8) -> bool {
|
||||||
let s = size as usize;
|
let s = size as usize;
|
||||||
if pos + s > data.len() {
|
if pos + s > data.len() {
|
||||||
@@ -100,13 +92,24 @@ impl FixedArrayHeader {
|
|||||||
offset: usize,
|
offset: usize,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one read of the header.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Self, FormatError> {
|
) -> Result<Self, FormatError> {
|
||||||
// FAHD signature(4) + version(1) + client_id(1) + element_size(1) +
|
// FAHD signature(4) + version(1) + client_id(1) + element_size(1) +
|
||||||
// max_nelmts_bits(1) + num_elements(length_size) + data_block_addr(offset_size) + checksum(4)
|
// max_nelmts_bits(1) + num_elements(length_size) + data_block_addr(offset_size) + checksum(4)
|
||||||
let min_size = 4 + 1 + 1 + 1 + 1 + length_size as usize + offset_size as usize + 4;
|
let min_size = 4 + 1 + 1 + 1 + 1 + length_size as usize + offset_size as usize + 4;
|
||||||
ensure_len(file_data, offset, min_size)?;
|
let w = Window::read(file, offset, min_size)?;
|
||||||
|
w.ensure(0, min_size)?;
|
||||||
|
|
||||||
let d = &file_data[offset..];
|
let d: &[u8] = &w.bytes;
|
||||||
if &d[0..4] != b"FAHD" {
|
if &d[0..4] != b"FAHD" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Fixed Array header signature".into(),
|
"invalid Fixed Array header signature".into(),
|
||||||
@@ -129,7 +132,7 @@ impl FixedArrayHeader {
|
|||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
let data_block_address = read_offset(d, pos, offset_size)?;
|
let data_block_address = read_offset(d, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
verify_checksum(file_data, offset, offset + pos)?;
|
verify_checksum(&w, 0, pos)?;
|
||||||
|
|
||||||
Ok(FixedArrayHeader {
|
Ok(FixedArrayHeader {
|
||||||
client_id,
|
client_id,
|
||||||
@@ -151,19 +154,43 @@ pub fn read_fixed_array_chunks(
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &FixedArrayHeader,
|
header: &FixedArrayHeader,
|
||||||
dataset_dims: &[u64],
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dimensions: &[u32],
|
||||||
|
element_size: u32,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
|
read_fixed_array_chunks_in(
|
||||||
|
&file_data,
|
||||||
|
header,
|
||||||
|
dataset_dims,
|
||||||
|
max_dims,
|
||||||
|
chunk_dimensions,
|
||||||
|
element_size,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`read_fixed_array_chunks`] over any [`Storage`]: one read of the data
|
||||||
|
/// block's prefix, one of the whole data block (pages included).
|
||||||
|
#[allow(clippy::too_many_arguments)]
|
||||||
|
pub fn read_fixed_array_chunks_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
header: &FixedArrayHeader,
|
||||||
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
chunk_dimensions: &[u32],
|
chunk_dimensions: &[u32],
|
||||||
element_size: u32,
|
element_size: u32,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let db_offset = header.data_block_address as usize;
|
let file_len = len_usize(file);
|
||||||
let rank = chunk_dimensions.len();
|
let db_offset = to_usize(header.data_block_address)?;
|
||||||
|
|
||||||
// Parse data block header: FADB(4) + version(1) + client_id(1) + header_address(offset_size)
|
// Parse data block header: FADB(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||||
let db_header_size = 4 + 1 + 1 + offset_size as usize;
|
let db_header_size = 4 + 1 + 1 + offset_size as usize;
|
||||||
ensure_len(file_data, db_offset, db_header_size)?;
|
let d = read_exact_at(file, db_offset as u64, db_header_size)?;
|
||||||
|
|
||||||
let d = &file_data[db_offset..];
|
|
||||||
if &d[0..4] != b"FADB" {
|
if &d[0..4] != b"FADB" {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"invalid Fixed Array data block signature".into(),
|
"invalid Fixed Array data block signature".into(),
|
||||||
@@ -173,11 +200,11 @@ pub fn read_fixed_array_chunks(
|
|||||||
// Elements start immediately after the data block prefix.
|
// Elements start immediately after the data block prefix.
|
||||||
let elements_start = db_offset + db_header_size;
|
let elements_start = db_offset + db_header_size;
|
||||||
|
|
||||||
let num_elements = header.num_elements as usize;
|
let num_elements = to_usize(header.num_elements)?;
|
||||||
// A chunk index cannot describe more elements than the file has bytes (each
|
// A chunk index cannot describe more elements than the file has bytes (each
|
||||||
// element occupies at least `offset_size` bytes). Reject a corrupt count
|
// element occupies at least `offset_size` bytes). Reject a corrupt count
|
||||||
// before it can drive a huge loop or overflow an offset computation.
|
// before it can drive a huge loop or overflow an offset computation.
|
||||||
if num_elements > file_data.len() {
|
if num_elements > file_len {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"Fixed Array element count exceeds file size".into(),
|
"Fixed Array element count exceeds file size".into(),
|
||||||
));
|
));
|
||||||
@@ -198,44 +225,43 @@ pub fn read_fixed_array_chunks(
|
|||||||
))
|
))
|
||||||
};
|
};
|
||||||
|
|
||||||
// Compute chunk offsets based on index.
|
// The index is laid out over the chunk grid of the *maximum* dimensions
|
||||||
// Chunks are stored in row-major order within the dataset space.
|
// (row-major), so a dataset smaller than its maxshape has gaps.
|
||||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||||
for d_idx in 0..rank {
|
let grid = ChunkGrid::fixed_array(dataset_dims, max_dims, &dims_u64)?;
|
||||||
let ch_dim = chunk_dimensions[d_idx] as u64;
|
|
||||||
if ch_dim == 0 {
|
|
||||||
return Err(FormatError::ChunkedReadError(
|
|
||||||
"chunk dimension is zero".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let ds_dim = dataset_dims[d_idx];
|
|
||||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
|
||||||
}
|
|
||||||
|
|
||||||
let chunk_byte_size: u64 =
|
let chunk_byte_size: u64 =
|
||||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||||
|
|
||||||
let mut chunks = Vec::new();
|
let mut chunks = Vec::new();
|
||||||
let push_element =
|
// `rel` is relative to the data block, whose bytes are in `w`.
|
||||||
|i: usize, abs: usize, chunks: &mut Vec<ChunkInfo>| -> Result<(), FormatError> {
|
let push_element = |w: &Window<'_>,
|
||||||
if let Some((address, chunk_size, filter_mask)) = parse_fa_element(
|
i: usize,
|
||||||
file_data,
|
rel: usize,
|
||||||
abs,
|
chunks: &mut Vec<ChunkInfo>|
|
||||||
header.client_id,
|
-> Result<(), FormatError> {
|
||||||
offset_size,
|
if let Some((address, chunk_size, filter_mask)) = parse_fa_element(
|
||||||
header.element_size,
|
w,
|
||||||
chunk_byte_size,
|
rel,
|
||||||
)? {
|
header.client_id,
|
||||||
let offsets = index_to_chunk_offsets(i, &num_chunks_per_dim, chunk_dimensions);
|
offset_size,
|
||||||
chunks.push(ChunkInfo {
|
header.element_size,
|
||||||
chunk_size,
|
chunk_byte_size,
|
||||||
filter_mask,
|
)? {
|
||||||
offsets,
|
// A slot beyond the current extent is ignored, as the
|
||||||
address,
|
// library does.
|
||||||
});
|
let Some(offsets) = grid.offsets(i as u64) else {
|
||||||
}
|
return Ok(());
|
||||||
Ok(())
|
};
|
||||||
};
|
chunks.push(ChunkInfo {
|
||||||
|
chunk_size,
|
||||||
|
filter_mask,
|
||||||
|
offsets,
|
||||||
|
address,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
};
|
||||||
|
|
||||||
// A data block is paged when it holds more elements than fit in one page.
|
// A data block is paged when it holds more elements than fit in one page.
|
||||||
// `max_nelmts_bits` is an untrusted u8; a shift >= the pointer width would
|
// `max_nelmts_bits` is an untrusted u8; a shift >= the pointer width would
|
||||||
@@ -250,10 +276,16 @@ pub fn read_fixed_array_chunks(
|
|||||||
|
|
||||||
if !is_paged {
|
if !is_paged {
|
||||||
// Non-paged: prefix, then `num_elements` elements packed directly,
|
// Non-paged: prefix, then `num_elements` elements packed directly,
|
||||||
// then a checksum over both.
|
// then a checksum over both. One window holds all of it (or ends at
|
||||||
verify_checksum(file_data, db_offset, elem_at(elements_start, num_elements)?)?;
|
// the end of the file), so its bounds checks are the whole-file ones.
|
||||||
|
let end = elem_at(elements_start, num_elements)?;
|
||||||
|
// The checksum's bounds check comes first: make it before reading.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
Window::check_extent(file, db_offset as u64, end - db_offset, 4)?;
|
||||||
|
let w = Window::read(file, db_offset as u64, end.saturating_add(4) - db_offset)?;
|
||||||
|
verify_checksum(&w, 0, end - db_offset)?;
|
||||||
for i in 0..num_elements {
|
for i in 0..num_elements {
|
||||||
push_element(i, elem_at(elements_start, i)?, &mut chunks)?;
|
push_element(&w, i, elem_at(elements_start, i)? - db_offset, &mut chunks)?;
|
||||||
}
|
}
|
||||||
return Ok(chunks);
|
return Ok(chunks);
|
||||||
}
|
}
|
||||||
@@ -276,22 +308,40 @@ pub fn read_fixed_array_chunks(
|
|||||||
.and_then(|x| x.checked_add(4))
|
.and_then(|x| x.checked_add(4))
|
||||||
.ok_or_else(stride_overflow)?;
|
.ok_or_else(stride_overflow)?;
|
||||||
|
|
||||||
if bitmap_start + bitmap_size > file_data.len() {
|
if bitmap_start + bitmap_size > file_len {
|
||||||
return Err(FormatError::UnexpectedEof {
|
return Err(FormatError::UnexpectedEof {
|
||||||
expected: bitmap_start + bitmap_size,
|
expected: bitmap_start + bitmap_size,
|
||||||
available: file_data.len(),
|
available: file_len,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
// The whole data block in one window when it is small: every page slot
|
||||||
|
// is at most `page_stride` bytes, so every position checked below lies
|
||||||
|
// inside it (or past the end of the file). A larger block is read as its
|
||||||
|
// prefix and bitmap, then each page in use on its own.
|
||||||
|
let block_len = (pages_start - db_offset).saturating_add(npages.saturating_mul(page_stride));
|
||||||
|
let whole = if block_len <= PAGED_BLOCK_ONE_READ_MAX {
|
||||||
|
Some(Window::read(file, db_offset as u64, block_len)?)
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
let head_w;
|
||||||
|
let head = match &whole {
|
||||||
|
Some(w) => w,
|
||||||
|
None => {
|
||||||
|
head_w = Window::read(file, db_offset as u64, pages_start - db_offset)?;
|
||||||
|
&head_w
|
||||||
|
}
|
||||||
|
};
|
||||||
// The prefix and page bitmap are covered by their own checksum, and each
|
// The prefix and page bitmap are covered by their own checksum, and each
|
||||||
// initialised page by one of its own.
|
// initialised page by one of its own.
|
||||||
verify_checksum(file_data, db_offset, bitmap_start + bitmap_size)?;
|
verify_checksum(head, 0, bitmap_start + bitmap_size - db_offset)?;
|
||||||
|
|
||||||
for p in 0..npages {
|
for p in 0..npages {
|
||||||
let page_first = p * page_nelmts; // < num_elements, cannot overflow
|
let page_first = p * page_nelmts; // < num_elements, cannot overflow
|
||||||
let page_count = core::cmp::min(page_nelmts, num_elements - page_first);
|
let page_count = core::cmp::min(page_nelmts, num_elements - page_first);
|
||||||
|
|
||||||
// Check the page-init bit (MSB-first within each byte).
|
// Check the page-init bit (MSB-first within each byte).
|
||||||
let bit_byte = file_data[bitmap_start + p / 8];
|
let bit_byte = head.bytes[bitmap_start + p / 8 - db_offset];
|
||||||
let bit_mask = 1u8 << (7 - (p % 8));
|
let bit_mask = 1u8 << (7 - (p % 8));
|
||||||
if bit_byte & bit_mask == 0 {
|
if bit_byte & bit_mask == 0 {
|
||||||
continue; // entire page unallocated
|
continue; // entire page unallocated
|
||||||
@@ -301,21 +351,33 @@ pub fn read_fixed_array_chunks(
|
|||||||
.checked_mul(page_stride)
|
.checked_mul(page_stride)
|
||||||
.and_then(|o| pages_start.checked_add(o))
|
.and_then(|o| pages_start.checked_add(o))
|
||||||
.ok_or_else(stride_overflow)?;
|
.ok_or_else(stride_overflow)?;
|
||||||
verify_checksum(file_data, page_off, elem_at(page_off, page_count)?)?;
|
let page_end = elem_at(page_off, page_count)?;
|
||||||
|
// `w` holds the page from `base` on (positions below are relative
|
||||||
|
// to it).
|
||||||
|
let page_w;
|
||||||
|
let (w, base) = match &whole {
|
||||||
|
Some(w) => (w, db_offset),
|
||||||
|
None => {
|
||||||
|
page_w =
|
||||||
|
Window::read(file, page_off as u64, page_end.saturating_add(4) - page_off)?;
|
||||||
|
(&page_w, page_off)
|
||||||
|
}
|
||||||
|
};
|
||||||
|
verify_checksum(w, page_off - base, page_end - base)?;
|
||||||
for e in 0..page_count {
|
for e in 0..page_count {
|
||||||
push_element(page_first + e, elem_at(page_off, e)?, &mut chunks)?;
|
push_element(w, page_first + e, elem_at(page_off, e)? - base, &mut chunks)?;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(chunks)
|
Ok(chunks)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Parse a single Fixed Array element at absolute file offset `abs`.
|
/// Parse a single Fixed Array element at offset `abs` of the window `w`.
|
||||||
///
|
///
|
||||||
/// Returns `Some((address, chunk_size, filter_mask))` for an allocated chunk, or
|
/// Returns `Some((address, chunk_size, filter_mask))` for an allocated chunk, or
|
||||||
/// `None` if the element is undefined (an unallocated chunk, address all-`0xFF`).
|
/// `None` if the element is undefined (an unallocated chunk, address all-`0xFF`).
|
||||||
fn parse_fa_element(
|
fn parse_fa_element(
|
||||||
file_data: &[u8],
|
w: &Window<'_>,
|
||||||
abs: usize,
|
abs: usize,
|
||||||
client_id: u8,
|
client_id: u8,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
@@ -325,12 +387,8 @@ fn parse_fa_element(
|
|||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
if client_id == 0 {
|
if client_id == 0 {
|
||||||
// Non-filtered: element is just the chunk address.
|
// Non-filtered: element is just the chunk address.
|
||||||
if abs + os > file_data.len() {
|
w.ensure(abs, os)?;
|
||||||
return Err(FormatError::UnexpectedEof {
|
let file_data: &[u8] = &w.bytes;
|
||||||
expected: abs + os,
|
|
||||||
available: file_data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if is_undefined(file_data, abs, offset_size) {
|
if is_undefined(file_data, abs, offset_size) {
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
}
|
||||||
@@ -345,17 +403,14 @@ fn parse_fa_element(
|
|||||||
));
|
));
|
||||||
}
|
}
|
||||||
let chunk_size_bytes = es - os - 4;
|
let chunk_size_bytes = es - os - 4;
|
||||||
if abs + es > file_data.len() {
|
w.ensure(abs, es)?;
|
||||||
return Err(FormatError::UnexpectedEof {
|
let file_data: &[u8] = &w.bytes;
|
||||||
expected: abs + es,
|
|
||||||
available: file_data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if is_undefined(file_data, abs, offset_size) {
|
if is_undefined(file_data, abs, offset_size) {
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
}
|
||||||
let address = read_offset(file_data, abs, offset_size)?;
|
let address = read_offset(file_data, abs, offset_size)?;
|
||||||
let chunk_size = read_variable_length(&file_data[abs + os..], chunk_size_bytes)?;
|
let chunk_size =
|
||||||
|
read_variable_length(&file_data[abs + os..abs + es - 4], chunk_size_bytes)?;
|
||||||
let fm_off = abs + os + chunk_size_bytes;
|
let fm_off = abs + os + chunk_size_bytes;
|
||||||
let filter_mask = u32::from_le_bytes([
|
let filter_mask = u32::from_le_bytes([
|
||||||
file_data[fm_off],
|
file_data[fm_off],
|
||||||
@@ -367,27 +422,6 @@ fn parse_fa_element(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
|
||||||
fn index_to_chunk_offsets(
|
|
||||||
index: usize,
|
|
||||||
num_chunks_per_dim: &[u64],
|
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Vec<u64> {
|
|
||||||
let rank = num_chunks_per_dim.len();
|
|
||||||
let mut offsets = vec![0u64; rank];
|
|
||||||
let mut remaining = index as u64;
|
|
||||||
for d in (0..rank).rev() {
|
|
||||||
let nchunks = num_chunks_per_dim[d];
|
|
||||||
if nchunks == 0 {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
let chunk_idx = remaining % nchunks;
|
|
||||||
remaining /= nchunks;
|
|
||||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
|
||||||
}
|
|
||||||
offsets
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read a variable-length little-endian unsigned integer.
|
/// Read a variable-length little-endian unsigned integer.
|
||||||
fn read_variable_length(data: &[u8], size: usize) -> Result<u64, FormatError> {
|
fn read_variable_length(data: &[u8], size: usize) -> Result<u64, FormatError> {
|
||||||
if size > 8 || data.len() < size {
|
if size > 8 || data.len() < size {
|
||||||
@@ -416,44 +450,21 @@ mod tests {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_1d() {
|
fn index_to_offsets_1d() {
|
||||||
let num_chunks = vec![5u64];
|
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||||
let chunk_dims = vec![20u32];
|
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![20]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
|
||||||
vec![80]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_2d() {
|
fn index_to_offsets_2d() {
|
||||||
// 10x6 dataset with 4x3 chunks => ceil(10/4)=3, ceil(6/3)=2 => 6 chunks
|
// 10x6 dataset with 4x3 chunks => ceil(10/4)=3, ceil(6/3)=2 => 6 chunks
|
||||||
let num_chunks = vec![3u64, 2];
|
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||||
let chunk_dims = vec![4u32, 3];
|
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||||
vec![0, 0]
|
assert_eq!(g.offsets(3).unwrap(), vec![4, 3]);
|
||||||
);
|
assert_eq!(g.offsets(5).unwrap(), vec![8, 3]);
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![0, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 0]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(3, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(5, &num_chunks, &chunk_dims),
|
|
||||||
vec![8, 3]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -517,7 +528,7 @@ mod tests {
|
|||||||
|
|
||||||
let read = |f: &[u8], fahd: usize| -> Result<Vec<ChunkInfo>, FormatError> {
|
let read = |f: &[u8], fahd: usize| -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let h = FixedArrayHeader::parse(f, fahd, 8, 8)?;
|
let h = FixedArrayHeader::parse(f, fahd, 8, 8)?;
|
||||||
read_fixed_array_chunks(f, &h, &[60], &[20], 8, 8, 8)
|
read_fixed_array_chunks(f, &h, &[60], None, &[20], 8, 8, 8)
|
||||||
};
|
};
|
||||||
|
|
||||||
let (clean, fahd) = build();
|
let (clean, fahd) = build();
|
||||||
@@ -562,7 +573,7 @@ mod tests {
|
|||||||
let db = 0x100usize;
|
let db = 0x100usize;
|
||||||
buf[db..db + 4].copy_from_slice(b"FADB");
|
buf[db..db + 4].copy_from_slice(b"FADB");
|
||||||
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -579,7 +590,7 @@ mod tests {
|
|||||||
stamp_checksum(&mut buf, fahd, fahd + 24);
|
stamp_checksum(&mut buf, fahd, fahd + 24);
|
||||||
buf[0x80..0x84].copy_from_slice(b"FADB");
|
buf[0x80..0x84].copy_from_slice(b"FADB");
|
||||||
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -602,7 +613,7 @@ mod tests {
|
|||||||
data_block_address: (usize::MAX - 4) as u64,
|
data_block_address: (usize::MAX - 4) as u64,
|
||||||
};
|
};
|
||||||
let buf = vec![0u8; 64];
|
let buf = vec![0u8; 64];
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -664,6 +675,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -740,6 +752,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -840,6 +853,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -858,4 +872,126 @@ mod tests {
|
|||||||
.collect();
|
.collect();
|
||||||
assert_eq!(got, expect);
|
assert_eq!(got, expect);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A fixed array (header at 0x100, data block at 0x200) of `n` chunks,
|
||||||
|
/// filtered or not, paged when `n` exceeds `1 << page_bits`; every
|
||||||
|
/// page initialised except page 1.
|
||||||
|
fn build_fixed_array(n: usize, filtered: bool, page_bits: u8) -> Vec<u8> {
|
||||||
|
let os = 8usize;
|
||||||
|
let es = if filtered { os + 4 + 4 } else { os };
|
||||||
|
let (fahd, db) = (0x100usize, 0x200usize);
|
||||||
|
let mut f = vec![0u8; 0x2000];
|
||||||
|
f[fahd..fahd + 4].copy_from_slice(b"FAHD");
|
||||||
|
f[fahd + 5] = u8::from(filtered);
|
||||||
|
f[fahd + 6] = es as u8;
|
||||||
|
f[fahd + 7] = page_bits;
|
||||||
|
f[fahd + 8..fahd + 16].copy_from_slice(&(n as u64).to_le_bytes());
|
||||||
|
f[fahd + 16..fahd + 24].copy_from_slice(&(db as u64).to_le_bytes());
|
||||||
|
stamp_checksum(&mut f, fahd, fahd + 24);
|
||||||
|
f[db..db + 4].copy_from_slice(b"FADB");
|
||||||
|
f[db + 5] = u8::from(filtered);
|
||||||
|
f[db + 6..db + 14].copy_from_slice(&(fahd as u64).to_le_bytes());
|
||||||
|
let elems = db + 6 + os;
|
||||||
|
let write = |f: &mut Vec<u8>, at: usize, i: usize| {
|
||||||
|
let addr = if i == 2 {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
0x1000 + i as u64 * 0x100
|
||||||
|
};
|
||||||
|
f[at..at + os].copy_from_slice(&addr.to_le_bytes());
|
||||||
|
if filtered {
|
||||||
|
f[at + os..at + os + 4].copy_from_slice(&(100 + i as u32).to_le_bytes());
|
||||||
|
f[at + os + 4..at + os + 8].copy_from_slice(&(i as u32 & 1).to_le_bytes());
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let page = 1usize << page_bits;
|
||||||
|
if n <= page {
|
||||||
|
for i in 0..n {
|
||||||
|
write(&mut f, elems + i * es, i);
|
||||||
|
}
|
||||||
|
stamp_checksum(&mut f, db, elems + n * es);
|
||||||
|
} else {
|
||||||
|
let npages = n.div_ceil(page);
|
||||||
|
let bitmap = npages.div_ceil(8);
|
||||||
|
for p in 0..npages {
|
||||||
|
if p != 1 {
|
||||||
|
f[elems + p / 8] |= 0x80 >> (p % 8);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
stamp_checksum(&mut f, db, elems + bitmap);
|
||||||
|
let pages_start = elems + bitmap + 4;
|
||||||
|
for p in (0..npages).filter(|&p| p != 1) {
|
||||||
|
let at = pages_start + p * (page * es + 4);
|
||||||
|
let count = page.min(n - p * page);
|
||||||
|
for e in 0..count {
|
||||||
|
write(&mut f, at + e * es, p * page + e);
|
||||||
|
}
|
||||||
|
stamp_checksum(&mut f, at, at + count * es);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
f
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Non-paged and paged, filtered and unfiltered arrays, cut at every
|
||||||
|
/// length through the data block and with a damaged byte, read
|
||||||
|
/// identically through a `read_at`-only storage.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
for (n, filtered, bits) in [(3, false, 10), (3, true, 10), (11, false, 2), (11, true, 2)] {
|
||||||
|
let full = build_fixed_array(n, filtered, bits);
|
||||||
|
let es = if filtered { 16 } else { 8 };
|
||||||
|
let dims = [n as u64 * 20];
|
||||||
|
let h = FixedArrayHeader::parse(&full, 0x100, 8, 8).unwrap();
|
||||||
|
let chunks = read_fixed_array_chunks(&full, &h, &dims, None, &[20], 8, 8, 8).unwrap();
|
||||||
|
// Chunk 2 is unallocated, and so is page 1 of a paged array.
|
||||||
|
let expect = if n > 4 { n - 1 - 4 } else { n - 1 };
|
||||||
|
assert_eq!(chunks.len(), expect);
|
||||||
|
let mut files = Vec::new();
|
||||||
|
for cut in (0x100..0x200 + 40 + n * (es + 4) + 16).step_by(3) {
|
||||||
|
files.push(full[..cut].to_vec());
|
||||||
|
}
|
||||||
|
for at in [0x104, 0x210, 0x21a, 0x230] {
|
||||||
|
let mut damaged = full.clone();
|
||||||
|
damaged[at] ^= 1;
|
||||||
|
files.push(damaged);
|
||||||
|
}
|
||||||
|
files.push(full);
|
||||||
|
for f in files {
|
||||||
|
let storage = CountingStorage::new(f.clone());
|
||||||
|
let want = FixedArrayHeader::parse(&f, 0x100, 8, 8);
|
||||||
|
let got = FixedArrayHeader::parse_in(&storage, 0x100, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
let Ok(h) = want else { continue };
|
||||||
|
let want = read_fixed_array_chunks(&f, &h, &dims, None, &[20], 8, 8, 8);
|
||||||
|
let got = read_fixed_array_chunks_in(&storage, &h, &dims, None, &[20], 8, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"), "{} bytes", f.len());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A header whose element count stretches its data block (one checksum
|
||||||
|
/// over the whole block) far past the end of a 16 MiB file: the
|
||||||
|
/// checksum's bounds check fails before the block is read, with the
|
||||||
|
/// slice read's error.
|
||||||
|
#[cfg(feature = "checksum")]
|
||||||
|
#[test]
|
||||||
|
fn oversized_block_fails_before_reading() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let mut f = build_fixed_array(3, false, 10);
|
||||||
|
f.resize(16 << 20, 0);
|
||||||
|
let mut h = FixedArrayHeader::parse(&f, 0x100, 8, 8).unwrap();
|
||||||
|
h.max_nelmts_bits = 30;
|
||||||
|
h.num_elements = 4 << 20;
|
||||||
|
let dims = [h.num_elements * 20];
|
||||||
|
let want = read_fixed_array_chunks(&f, &h, &dims, None, &[20], 8, 8, 8);
|
||||||
|
assert!(
|
||||||
|
matches!(want, Err(FormatError::UnexpectedEof { .. })),
|
||||||
|
"{want:?}"
|
||||||
|
);
|
||||||
|
let storage = CountingStorage::new(f);
|
||||||
|
let got = read_fixed_array_chunks_in(&storage, &h, &dims, None, &[20], 8, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
assert!(storage.bytes_read() < 64, "{} bytes", storage.bytes_read());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,597 @@
|
|||||||
|
//! Copying a selection out of a row-major buffer one contiguous run at a time.
|
||||||
|
//!
|
||||||
|
//! A selection's elements, in output order, fall into runs that are adjacent
|
||||||
|
//! in the source: a whole block along the last dimension, blocks that touch
|
||||||
|
//! (`stride == block`), and whole rows when the inner dimensions are selected
|
||||||
|
//! in full. Copying run by run turns a 256 x 256 hyperslab of a 1024-wide
|
||||||
|
//! dataset into 256 `memcpy`s of 1 KiB, where the old extractor recursed and
|
||||||
|
//! bounds-checked once per element.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::data_read::NativeElement;
|
||||||
|
use crate::error::FormatError;
|
||||||
|
use crate::selection::Selection;
|
||||||
|
use crate::storage::{ExtentBytes, ExtentReq, Storage, raw_batches};
|
||||||
|
|
||||||
|
/// Row-major element strides of `dims` (the last dimension has stride 1).
|
||||||
|
fn strides(dims: &[u64]) -> Vec<u64> {
|
||||||
|
let mut s = vec![1u64; dims.len()];
|
||||||
|
for d in (0..dims.len().saturating_sub(1)).rev() {
|
||||||
|
s[d] = s[d + 1].wrapping_mul(dims[d + 1]);
|
||||||
|
}
|
||||||
|
s
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Merges adjacent runs before handing them on.
|
||||||
|
struct Coalesce<F: FnMut(u64, u64)> {
|
||||||
|
start: u64,
|
||||||
|
len: u64,
|
||||||
|
emit: F,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<F: FnMut(u64, u64)> Coalesce<F> {
|
||||||
|
#[inline]
|
||||||
|
fn push(&mut self, start: u64, len: u64) {
|
||||||
|
if len == 0 {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if self.len > 0 && self.start.wrapping_add(self.len) == start {
|
||||||
|
self.len += len;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
self.flush();
|
||||||
|
self.start = start;
|
||||||
|
self.len = len;
|
||||||
|
}
|
||||||
|
|
||||||
|
fn flush(&mut self) {
|
||||||
|
if self.len > 0 {
|
||||||
|
(self.emit)(self.start, self.len);
|
||||||
|
self.len = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Call `emit(first_element, element_count)` for each run of a hyperslab's
|
||||||
|
/// elements that is contiguous in a row-major dataset of shape `dims`, in
|
||||||
|
/// the order the selection returns them. Adjacent runs are merged.
|
||||||
|
///
|
||||||
|
/// Coordinates at or past a dimension's extent are skipped, as the
|
||||||
|
/// element-wise extractor always did; callers that want them to be an error
|
||||||
|
/// validate the selection first. The four vectors must have `dims.len()`
|
||||||
|
/// entries.
|
||||||
|
pub(crate) fn hyperslab_runs(
|
||||||
|
dims: &[u64],
|
||||||
|
start: &[u64],
|
||||||
|
stride: &[u64],
|
||||||
|
count: &[u64],
|
||||||
|
block: &[u64],
|
||||||
|
emit: impl FnMut(u64, u64),
|
||||||
|
) {
|
||||||
|
let rank = dims.len();
|
||||||
|
let mut out = Coalesce {
|
||||||
|
start: 0,
|
||||||
|
len: 0,
|
||||||
|
emit,
|
||||||
|
};
|
||||||
|
if rank == 0 {
|
||||||
|
out.push(0, 1);
|
||||||
|
out.flush();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (0..rank).any(|d| count[d] == 0 || block[d] == 0) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let strides = strides(dims);
|
||||||
|
let last = rank - 1;
|
||||||
|
// Odometer over the outer dimensions: (block index, offset in block).
|
||||||
|
let mut ci = vec![0u64; last];
|
||||||
|
let mut bi = vec![0u64; last];
|
||||||
|
'outer: loop {
|
||||||
|
// Base offset of this row, or skip it if a coordinate is out of range.
|
||||||
|
let mut base = 0u64;
|
||||||
|
let mut in_range = true;
|
||||||
|
for d in 0..last {
|
||||||
|
let coord = start[d]
|
||||||
|
.saturating_add(ci[d].saturating_mul(stride[d]))
|
||||||
|
.saturating_add(bi[d]);
|
||||||
|
if coord >= dims[d] {
|
||||||
|
in_range = false;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
base = base.wrapping_add(coord.wrapping_mul(strides[d]));
|
||||||
|
}
|
||||||
|
if in_range && (stride[last] == block[last] || count[last] == 1) {
|
||||||
|
// Blocks that touch (the common unit-stride case: block 1,
|
||||||
|
// stride 1) are one range; don't split it into per-element runs.
|
||||||
|
let s = start[last];
|
||||||
|
let e = s
|
||||||
|
.saturating_add(count[last].saturating_mul(block[last]))
|
||||||
|
.min(dims[last]);
|
||||||
|
if s < e {
|
||||||
|
out.push(base.wrapping_add(s), e - s);
|
||||||
|
}
|
||||||
|
} else if in_range {
|
||||||
|
for c in 0..count[last] {
|
||||||
|
let s = start[last].saturating_add(c.saturating_mul(stride[last]));
|
||||||
|
if s >= dims[last] {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let e = s.saturating_add(block[last]).min(dims[last]);
|
||||||
|
out.push(base.wrapping_add(s), e - s);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Advance the odometer, last outer dimension fastest.
|
||||||
|
let mut d = last;
|
||||||
|
loop {
|
||||||
|
if d == 0 {
|
||||||
|
break 'outer;
|
||||||
|
}
|
||||||
|
d -= 1;
|
||||||
|
bi[d] += 1;
|
||||||
|
if bi[d] < block[d] {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
bi[d] = 0;
|
||||||
|
ci[d] += 1;
|
||||||
|
if ci[d] < count[d] {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
ci[d] = 0;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out.flush();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The selected elements of `src` — a row-major dataset of shape `dims` and
|
||||||
|
/// `elem_size`-byte elements — copied into a fresh `Vec<T>`, one `memcpy` per
|
||||||
|
/// contiguous run, with no zero-filling of the output first.
|
||||||
|
///
|
||||||
|
/// For `T` other than `u8`, `elem_size` must equal `size_of::<T>()`. The
|
||||||
|
/// selection must be a validated hyperslab, point list or `None` (`All` is the
|
||||||
|
/// caller's to handle); `src` must hold exactly the dataset. Anything that
|
||||||
|
/// would read outside `src` is an error, never a partial result.
|
||||||
|
pub(crate) fn gather<T: NativeElement>(
|
||||||
|
src: &[u8],
|
||||||
|
dims: &[u64],
|
||||||
|
elem_size: usize,
|
||||||
|
selection: &Selection,
|
||||||
|
) -> Result<Vec<T>, FormatError> {
|
||||||
|
let t_size = core::mem::size_of::<T>();
|
||||||
|
if elem_size == 0 || (t_size != 1 && t_size != elem_size) {
|
||||||
|
return Err(FormatError::DataSizeMismatch {
|
||||||
|
expected: t_size,
|
||||||
|
actual: elem_size,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
let n_elements = match selection {
|
||||||
|
Selection::None => 0,
|
||||||
|
Selection::Hyperslab { count, block, .. } => count
|
||||||
|
.iter()
|
||||||
|
.zip(block)
|
||||||
|
.try_fold(1u64, |acc, (&c, &b)| acc.checked_mul(c.checked_mul(b)?))
|
||||||
|
.ok_or_else(|| FormatError::Overflow("hyperslab count x block overflows".into()))?,
|
||||||
|
Selection::Points(points) => points.len() as u64,
|
||||||
|
Selection::All => {
|
||||||
|
return Err(FormatError::SelectionOutOfBounds(
|
||||||
|
"gather does not take Selection::All".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let out_bytes = crate::chunked_read::checked_byte_len(n_elements, elem_size)?;
|
||||||
|
let out_len = out_bytes / t_size;
|
||||||
|
let mut out: Vec<T> = crate::bulk_alloc::vec_for_bulk(out_len);
|
||||||
|
let dst = out.as_mut_ptr().cast::<u8>();
|
||||||
|
let mut written = 0usize;
|
||||||
|
let mut failed = false;
|
||||||
|
let mut copy_run = |first: u64, n: u64| {
|
||||||
|
if failed {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let range = usize::try_from(first)
|
||||||
|
.ok()
|
||||||
|
.and_then(|f| f.checked_mul(elem_size))
|
||||||
|
.zip(
|
||||||
|
usize::try_from(n)
|
||||||
|
.ok()
|
||||||
|
.and_then(|n| n.checked_mul(elem_size)),
|
||||||
|
)
|
||||||
|
.and_then(|(at, len)| Some((at, len, at.checked_add(len)?)));
|
||||||
|
match range {
|
||||||
|
Some((at, len, end)) if end <= src.len() && written + len <= out_bytes => {
|
||||||
|
// SAFETY: `src[at..end]` is in bounds (checked above), and
|
||||||
|
// `dst + written .. + len` lies within `out`'s capacity of
|
||||||
|
// `out_bytes` bytes (checked above); `out` is a fresh
|
||||||
|
// allocation, so the regions do not overlap.
|
||||||
|
unsafe {
|
||||||
|
core::ptr::copy_nonoverlapping(src.as_ptr().add(at), dst.add(written), len)
|
||||||
|
};
|
||||||
|
written += len;
|
||||||
|
}
|
||||||
|
_ => failed = true,
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let mut bad_point = false;
|
||||||
|
match selection {
|
||||||
|
Selection::Hyperslab {
|
||||||
|
start,
|
||||||
|
stride,
|
||||||
|
count,
|
||||||
|
block,
|
||||||
|
} => {
|
||||||
|
let rank = dims.len();
|
||||||
|
if [start.len(), stride.len(), count.len(), block.len()] != [rank; 4] {
|
||||||
|
return Err(FormatError::SelectionOutOfBounds(
|
||||||
|
"hyperslab rank does not match dataset rank".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
hyperslab_runs(dims, start, stride, count, block, &mut copy_run);
|
||||||
|
}
|
||||||
|
Selection::Points(points) => {
|
||||||
|
let strides = strides(dims);
|
||||||
|
let mut runs = Coalesce {
|
||||||
|
start: 0,
|
||||||
|
len: 0,
|
||||||
|
emit: &mut copy_run,
|
||||||
|
};
|
||||||
|
for p in points {
|
||||||
|
if p.len() != dims.len() || p.iter().zip(dims).any(|(c, n)| c >= n) {
|
||||||
|
bad_point = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
let at = p
|
||||||
|
.iter()
|
||||||
|
.zip(&strides)
|
||||||
|
.fold(0u64, |acc, (c, s)| acc.wrapping_add(c.wrapping_mul(*s)));
|
||||||
|
runs.push(at, 1);
|
||||||
|
}
|
||||||
|
runs.flush();
|
||||||
|
}
|
||||||
|
Selection::None | Selection::All => {}
|
||||||
|
}
|
||||||
|
if failed || bad_point || written != out_bytes {
|
||||||
|
return Err(FormatError::SelectionOutOfBounds(
|
||||||
|
"selection addresses elements outside the dataset".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
// SAFETY: all `out_bytes` bytes, i.e. `out_len` values of `T`, were
|
||||||
|
// written above, and every bit pattern is a valid `T` (`NativeElement`).
|
||||||
|
unsafe { out.set_len(out_len) };
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Largest gap between two of a selection's runs that [`gather_storage`]
|
||||||
|
/// reads through rather than asking for the runs separately: skipping a
|
||||||
|
/// few KiB costs a remote backend far less than another request (and a
|
||||||
|
/// local one less than another call and allocation).
|
||||||
|
pub(crate) const GATHER_GAP_BYTES: usize = 4 << 10;
|
||||||
|
|
||||||
|
/// Largest single read [`gather_storage`] makes of a selection's runs: runs
|
||||||
|
/// are merged into reads up to this size, and a longer run is split.
|
||||||
|
pub(crate) const GATHER_SPAN_BYTES: usize = 8 << 20;
|
||||||
|
|
||||||
|
/// Call `emit(first_element, element_count)` for each run of a validated
|
||||||
|
/// hyperslab or point selection (in output order; see [`hyperslab_runs`]),
|
||||||
|
/// or the error for a hyperslab of the wrong rank or a point outside `dims`
|
||||||
|
/// (runs before that point have been emitted).
|
||||||
|
fn selection_runs(
|
||||||
|
dims: &[u64],
|
||||||
|
selection: &Selection,
|
||||||
|
emit: &mut dyn FnMut(u64, u64),
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
match selection {
|
||||||
|
Selection::Hyperslab {
|
||||||
|
start,
|
||||||
|
stride,
|
||||||
|
count,
|
||||||
|
block,
|
||||||
|
} => {
|
||||||
|
let rank = dims.len();
|
||||||
|
if [start.len(), stride.len(), count.len(), block.len()] != [rank; 4] {
|
||||||
|
return Err(FormatError::SelectionOutOfBounds(
|
||||||
|
"hyperslab rank does not match dataset rank".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
hyperslab_runs(dims, start, stride, count, block, emit);
|
||||||
|
}
|
||||||
|
Selection::Points(points) => {
|
||||||
|
let strides = strides(dims);
|
||||||
|
let mut coalesce = Coalesce {
|
||||||
|
start: 0,
|
||||||
|
len: 0,
|
||||||
|
emit,
|
||||||
|
};
|
||||||
|
for p in points {
|
||||||
|
if p.len() != dims.len() || p.iter().zip(dims).any(|(c, n)| c >= n) {
|
||||||
|
return Err(FormatError::SelectionOutOfBounds(
|
||||||
|
"selection addresses elements outside the dataset".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let at = p
|
||||||
|
.iter()
|
||||||
|
.zip(&strides)
|
||||||
|
.fold(0u64, |acc, (c, s)| acc.wrapping_add(c.wrapping_mul(*s)));
|
||||||
|
coalesce.push(at, 1);
|
||||||
|
}
|
||||||
|
coalesce.flush();
|
||||||
|
}
|
||||||
|
Selection::None | Selection::All => {}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One read of [`gather_storage`]: bytes `[start, end)` of the dataset,
|
||||||
|
/// which hold the output's bytes up to `out_end` (from where the previous
|
||||||
|
/// span's end left off).
|
||||||
|
#[derive(Clone, Copy)]
|
||||||
|
struct Span {
|
||||||
|
start: usize,
|
||||||
|
end: usize,
|
||||||
|
out_end: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`gather`] of bytes (`T = u8`) from a dataset that is not in memory: the
|
||||||
|
/// dataset's `src_len` bytes start at `base` in `file`, which must hold all
|
||||||
|
/// of them (the caller checks). Same checks and errors as [`gather`].
|
||||||
|
///
|
||||||
|
/// The selection's runs are walked twice. The first walk checks them and
|
||||||
|
/// plans the reads: runs in increasing order with at most
|
||||||
|
/// [`GATHER_GAP_BYTES`] between them are read as one span (the gap is read
|
||||||
|
/// and dropped), up to [`GATHER_SPAN_BYTES`] per span. So a strided
|
||||||
|
/// selection is a few large reads, not one per element, and nothing is
|
||||||
|
/// allocated per run. The spans are fetched batch by batch (one
|
||||||
|
/// [`Storage::read_ranges`] call per [`crate::storage::RAW_BATCH_BYTES`])
|
||||||
|
/// while the second walk copies each run out of its span.
|
||||||
|
pub(crate) fn gather_storage<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
base: u64,
|
||||||
|
src_len: usize,
|
||||||
|
dims: &[u64],
|
||||||
|
elem_size: usize,
|
||||||
|
selection: &Selection,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
if elem_size == 0 {
|
||||||
|
return Err(FormatError::DataSizeMismatch {
|
||||||
|
expected: 1,
|
||||||
|
actual: elem_size,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
let n_elements = match selection {
|
||||||
|
Selection::None => 0,
|
||||||
|
Selection::Hyperslab { count, block, .. } => count
|
||||||
|
.iter()
|
||||||
|
.zip(block)
|
||||||
|
.try_fold(1u64, |acc, (&c, &b)| acc.checked_mul(c.checked_mul(b)?))
|
||||||
|
.ok_or_else(|| FormatError::Overflow("hyperslab count x block overflows".into()))?,
|
||||||
|
Selection::Points(points) => points.len() as u64,
|
||||||
|
Selection::All => {
|
||||||
|
return Err(FormatError::SelectionOutOfBounds(
|
||||||
|
"gather does not take Selection::All".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let out_bytes = crate::chunked_read::checked_byte_len(n_elements, elem_size)?;
|
||||||
|
let outside = || {
|
||||||
|
FormatError::SelectionOutOfBounds("selection addresses elements outside the dataset".into())
|
||||||
|
};
|
||||||
|
|
||||||
|
// First walk: check every run and plan the spans.
|
||||||
|
let mut spans: Vec<Span> = Vec::new();
|
||||||
|
let mut total = 0usize;
|
||||||
|
let mut failed = false;
|
||||||
|
selection_runs(dims, selection, &mut |first: u64, n: u64| {
|
||||||
|
if failed {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let range = usize::try_from(first)
|
||||||
|
.ok()
|
||||||
|
.and_then(|f| f.checked_mul(elem_size))
|
||||||
|
.zip(
|
||||||
|
usize::try_from(n)
|
||||||
|
.ok()
|
||||||
|
.and_then(|n| n.checked_mul(elem_size)),
|
||||||
|
)
|
||||||
|
.and_then(|(at, len)| Some((at, len, at.checked_add(len)?)));
|
||||||
|
let Some((mut at, mut len)) = range
|
||||||
|
.filter(|&(_, len, end)| end <= src_len && len <= out_bytes - total)
|
||||||
|
.map(|(at, len, _)| (at, len))
|
||||||
|
else {
|
||||||
|
failed = true;
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
while len > 0 {
|
||||||
|
let room = match spans.last_mut() {
|
||||||
|
Some(s)
|
||||||
|
if at >= s.end
|
||||||
|
&& at - s.end <= GATHER_GAP_BYTES
|
||||||
|
&& at - s.start < GATHER_SPAN_BYTES =>
|
||||||
|
{
|
||||||
|
let take = len.min(GATHER_SPAN_BYTES - (at - s.start));
|
||||||
|
s.end = at + take;
|
||||||
|
s.out_end += take;
|
||||||
|
take
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
let take = len.min(GATHER_SPAN_BYTES);
|
||||||
|
spans.push(Span {
|
||||||
|
start: at,
|
||||||
|
end: at + take,
|
||||||
|
out_end: total + take,
|
||||||
|
});
|
||||||
|
take
|
||||||
|
}
|
||||||
|
};
|
||||||
|
total += room;
|
||||||
|
at += room;
|
||||||
|
len -= room;
|
||||||
|
}
|
||||||
|
})?;
|
||||||
|
if failed || total != out_bytes {
|
||||||
|
return Err(outside());
|
||||||
|
}
|
||||||
|
|
||||||
|
// The spans' reads, and the batches they are fetched in.
|
||||||
|
let reqs: Vec<ExtentReq> = spans
|
||||||
|
.iter()
|
||||||
|
.map(|s| ExtentReq {
|
||||||
|
addr: base + s.start as u64,
|
||||||
|
len: s.end - s.start,
|
||||||
|
fetch: Some(s.end - s.start),
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let batches = raw_batches(reqs.len(), false, |i| reqs[i].len);
|
||||||
|
|
||||||
|
// Second walk: copy each run out of its span, fetching each batch of
|
||||||
|
// spans when the walk reaches it (and dropping the previous one).
|
||||||
|
let mut out = crate::bulk_alloc::vec_for_bulk(out_bytes);
|
||||||
|
let mut span = 0usize;
|
||||||
|
let mut batch = 0usize;
|
||||||
|
let mut fetched: Option<ExtentBytes<'_>> = None;
|
||||||
|
let mut error: Option<FormatError> = None;
|
||||||
|
selection_runs(dims, selection, &mut |first: u64, n: u64| {
|
||||||
|
if error.is_some() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// Checked by the first walk (these cannot saturate or wrap).
|
||||||
|
let mut at = crate::addr::saturating_usize(first).wrapping_mul(elem_size);
|
||||||
|
let mut len = crate::addr::saturating_usize(n).wrapping_mul(elem_size);
|
||||||
|
while len > 0 {
|
||||||
|
while spans.get(span).is_some_and(|s| s.out_end <= out.len()) {
|
||||||
|
span += 1;
|
||||||
|
}
|
||||||
|
if fetched.is_none() || span >= batches[batch].end {
|
||||||
|
fetched = None;
|
||||||
|
while batches.get(batch).is_some_and(|b| span >= b.end) {
|
||||||
|
batch += 1;
|
||||||
|
}
|
||||||
|
let (Some(b), Some(_)) = (batches.get(batch).cloned(), spans.get(span)) else {
|
||||||
|
// The second walk emitted more than the first.
|
||||||
|
error = Some(outside());
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
match ExtentBytes::fetch(file, &reqs[b.clone()], b.start) {
|
||||||
|
Ok(f) => fetched = Some(f),
|
||||||
|
Err(e) => {
|
||||||
|
error = Some(e);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let s = spans[span];
|
||||||
|
let take = len.min(s.out_end - out.len());
|
||||||
|
let bytes = match fetched
|
||||||
|
.as_ref()
|
||||||
|
.map(|f| f.get(span, &reqs[span]))
|
||||||
|
.unwrap_or_else(|| Err(outside()))
|
||||||
|
{
|
||||||
|
Ok(b) => b,
|
||||||
|
Err(e) => {
|
||||||
|
error = Some(e);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
match at
|
||||||
|
.checked_sub(s.start)
|
||||||
|
.and_then(|o| bytes.get(o..o.checked_add(take)?))
|
||||||
|
{
|
||||||
|
Some(b) => out.extend_from_slice(b),
|
||||||
|
None => {
|
||||||
|
error = Some(outside());
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
at += take;
|
||||||
|
len -= take;
|
||||||
|
}
|
||||||
|
})?;
|
||||||
|
if let Some(e) = error {
|
||||||
|
return Err(e);
|
||||||
|
}
|
||||||
|
if out.len() != out_bytes {
|
||||||
|
return Err(outside());
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn runs(dims: &[u64], sel: [&[u64]; 4]) -> Vec<(u64, u64)> {
|
||||||
|
let mut v = Vec::new();
|
||||||
|
hyperslab_runs(dims, sel[0], sel[1], sel[2], sel[3], |s, n| v.push((s, n)));
|
||||||
|
v
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn runs_merge_blocks_and_whole_rows() {
|
||||||
|
// A box: one run per row.
|
||||||
|
assert_eq!(
|
||||||
|
runs(&[4, 10], [&[1, 2], &[1, 1], &[2, 3], &[1, 1]]),
|
||||||
|
vec![(12, 3), (22, 3)]
|
||||||
|
);
|
||||||
|
// Whole rows: one run.
|
||||||
|
assert_eq!(
|
||||||
|
runs(&[4, 10], [&[1, 0], &[1, 1], &[3, 10], &[1, 1]]),
|
||||||
|
vec![(10, 30)]
|
||||||
|
);
|
||||||
|
// stride == block: blocks merge.
|
||||||
|
assert_eq!(
|
||||||
|
runs(&[1, 10], [&[0, 1], &[1, 2], &[1, 4], &[1, 2]]),
|
||||||
|
vec![(1, 8)]
|
||||||
|
);
|
||||||
|
// Strided with blocks along both dimensions.
|
||||||
|
assert_eq!(
|
||||||
|
runs(&[6, 10], [&[0, 1], &[3, 4], &[2, 2], &[2, 2]]),
|
||||||
|
vec![
|
||||||
|
(1, 2),
|
||||||
|
(5, 2),
|
||||||
|
(11, 2),
|
||||||
|
(15, 2),
|
||||||
|
(31, 2),
|
||||||
|
(35, 2),
|
||||||
|
(41, 2),
|
||||||
|
(45, 2)
|
||||||
|
]
|
||||||
|
);
|
||||||
|
// Empty.
|
||||||
|
assert!(runs(&[4, 10], [&[0, 0], &[1, 1], &[0, 3], &[1, 1]]).is_empty());
|
||||||
|
// Scalar.
|
||||||
|
assert_eq!(runs(&[], [&[], &[], &[], &[]]), vec![(0, 1)]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn gather_matches_element_order_and_rejects_out_of_range() {
|
||||||
|
let dims = [3u64, 4];
|
||||||
|
let src: Vec<u8> = (0..12u16).flat_map(|v| v.to_le_bytes()).collect();
|
||||||
|
let sel = Selection::Hyperslab {
|
||||||
|
start: vec![0, 1],
|
||||||
|
stride: vec![2, 2],
|
||||||
|
count: vec![2, 2],
|
||||||
|
block: vec![1, 1],
|
||||||
|
};
|
||||||
|
let got: Vec<u8> = gather(&src, &dims, 2, &sel).unwrap();
|
||||||
|
let want: Vec<u8> = [1u16, 3, 9, 11]
|
||||||
|
.iter()
|
||||||
|
.flat_map(|v| v.to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
assert_eq!(got, want);
|
||||||
|
let pts = Selection::Points(vec![vec![2, 3], vec![0, 0], vec![0, 1]]);
|
||||||
|
let got: Vec<u8> = gather(&src, &dims, 2, &pts).unwrap();
|
||||||
|
let want: Vec<u8> = [11u16, 0, 1].iter().flat_map(|v| v.to_le_bytes()).collect();
|
||||||
|
assert_eq!(got, want);
|
||||||
|
// Past the extent, or a source shorter than the dataset: an error.
|
||||||
|
let bad = Selection::Points(vec![vec![3, 0]]);
|
||||||
|
assert!(gather::<u8>(&src, &dims, 2, &bad).is_err());
|
||||||
|
let past = Selection::Hyperslab {
|
||||||
|
start: vec![2, 0],
|
||||||
|
stride: vec![1, 1],
|
||||||
|
count: vec![2, 4],
|
||||||
|
block: vec![1, 1],
|
||||||
|
};
|
||||||
|
assert!(gather::<u8>(&src, &dims, 2, &past).is_err());
|
||||||
|
assert!(gather::<u8>(&src[..20], &dims, 2, &pts).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1,9 +1,12 @@
|
|||||||
//! HDF5 Global Heap collection parsing.
|
//! HDF5 Global Heap collection parsing.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::{borrow::Cow, format, string::String, vec::Vec};
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
use std::borrow::Cow;
|
||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{Storage, len_usize, read_exact_at};
|
||||||
|
|
||||||
/// Magic signature for global heap collections.
|
/// Magic signature for global heap collections.
|
||||||
const GCOL_SIGNATURE: [u8; 4] = *b"GCOL";
|
const GCOL_SIGNATURE: [u8; 4] = *b"GCOL";
|
||||||
@@ -28,19 +31,20 @@ pub struct GlobalHeapObject {
|
|||||||
pub data: Vec<u8>,
|
pub data: Vec<u8>,
|
||||||
}
|
}
|
||||||
|
|
||||||
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
/// Checks that `[offset, offset + needed)` ends by `data_len`.
|
||||||
|
fn ensure_len(data_len: usize, offset: usize, needed: usize) -> Result<(), FormatError> {
|
||||||
match offset.checked_add(needed) {
|
match offset.checked_add(needed) {
|
||||||
Some(end) if end <= data.len() => Ok(()),
|
Some(end) if end <= data_len => Ok(()),
|
||||||
_ => Err(FormatError::UnexpectedEof {
|
_ => Err(FormatError::UnexpectedEof {
|
||||||
expected: offset.saturating_add(needed),
|
expected: offset.saturating_add(needed),
|
||||||
available: data.len(),
|
available: data_len,
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
fn read_length(data: &[u8], offset: usize, length_size: u8) -> Result<u64, FormatError> {
|
fn read_length(data: &[u8], offset: usize, length_size: u8) -> Result<u64, FormatError> {
|
||||||
let s = length_size as usize;
|
let s = length_size as usize;
|
||||||
ensure_len(data, offset, s)?;
|
ensure_len(data.len(), offset, s)?;
|
||||||
let slice = &data[offset..offset + s];
|
let slice = &data[offset..offset + s];
|
||||||
Ok(match length_size {
|
Ok(match length_size {
|
||||||
2 => u16::from_le_bytes([slice[0], slice[1]]) as u64,
|
2 => u16::from_le_bytes([slice[0], slice[1]]) as u64,
|
||||||
@@ -52,11 +56,42 @@ fn read_length(data: &[u8], offset: usize, length_size: u8) -> Result<u64, Forma
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn object_overrun_msg(index: u16, size: usize, collection_size: u64) -> String {
|
||||||
|
format!(
|
||||||
|
"global heap object {index} ({size} bytes) runs past the end of its \
|
||||||
|
{collection_size}-byte collection"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
/// Round up to next multiple of 8.
|
/// Round up to next multiple of 8.
|
||||||
fn pad8(x: usize) -> usize {
|
fn pad8(x: usize) -> usize {
|
||||||
(x + 7) & !7
|
(x + 7) & !7
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Where one object of a global heap collection lies in the file, without
|
||||||
|
/// its data: see [`GlobalHeapCollection::parse_index`].
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub struct GlobalHeapObjectRef {
|
||||||
|
/// Object index (1-based; 0 is the free space marker).
|
||||||
|
pub index: u16,
|
||||||
|
/// Reference count.
|
||||||
|
pub reference_count: u16,
|
||||||
|
/// Offset of the object's data in the file data the collection was
|
||||||
|
/// parsed from.
|
||||||
|
pub offset: usize,
|
||||||
|
/// Size of the object's data in bytes.
|
||||||
|
pub size: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A global heap collection's objects, located but not copied.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct GlobalHeapIndex {
|
||||||
|
/// Total size of this collection including header.
|
||||||
|
pub collection_size: u64,
|
||||||
|
/// The objects, in file order.
|
||||||
|
pub objects: Vec<GlobalHeapObjectRef>,
|
||||||
|
}
|
||||||
|
|
||||||
impl GlobalHeapCollection {
|
impl GlobalHeapCollection {
|
||||||
/// Parse a global heap collection at the given offset in the file data.
|
/// Parse a global heap collection at the given offset in the file data.
|
||||||
pub fn parse(
|
pub fn parse(
|
||||||
@@ -64,71 +99,153 @@ impl GlobalHeapCollection {
|
|||||||
offset: usize,
|
offset: usize,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<GlobalHeapCollection, FormatError> {
|
) -> Result<GlobalHeapCollection, FormatError> {
|
||||||
// signature(4) + version(1) + reserved(3) + collection_size(length_size)
|
Self::parse_in(file_data, offset as u64, length_size)
|
||||||
let header_size = 8 + length_size as usize;
|
}
|
||||||
ensure_len(file_data, offset, header_size)?;
|
|
||||||
|
|
||||||
if file_data[offset..offset + 4] != GCOL_SIGNATURE {
|
/// [`Self::parse`] over any [`Storage`]: one read of the header, one of
|
||||||
|
/// the collection.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<GlobalHeapCollection, FormatError> {
|
||||||
|
let (bytes, base, index) = Self::read_collection(file, offset, length_size)?;
|
||||||
|
Ok(GlobalHeapCollection {
|
||||||
|
collection_size: index.collection_size,
|
||||||
|
objects: index
|
||||||
|
.objects
|
||||||
|
.iter()
|
||||||
|
.map(|o| GlobalHeapObject {
|
||||||
|
index: o.index,
|
||||||
|
reference_count: o.reference_count,
|
||||||
|
data: bytes[o.offset - base..o.offset - base + o.size].to_vec(),
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Locate the objects of the global heap collection at `offset` without
|
||||||
|
/// copying their data, so a caller can keep many collections indexed
|
||||||
|
/// for the cost of their object headers.
|
||||||
|
///
|
||||||
|
/// The collection must lie inside `file_data`, and every object inside
|
||||||
|
/// the collection, as libhdf5 lays them out; an object that runs past
|
||||||
|
/// its collection is an error.
|
||||||
|
pub fn parse_index(
|
||||||
|
file_data: &[u8],
|
||||||
|
offset: usize,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<GlobalHeapIndex, FormatError> {
|
||||||
|
Self::parse_index_in(file_data, offset as u64, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse_index`] over any [`Storage`]: one read of the header,
|
||||||
|
/// one of the collection. The object offsets are file offsets.
|
||||||
|
pub fn parse_index_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<GlobalHeapIndex, FormatError> {
|
||||||
|
Ok(Self::read_collection(file, offset, length_size)?.2)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read the collection at `offset` and index its objects: the
|
||||||
|
/// collection's bytes, its offset as a `usize`, and the index (with
|
||||||
|
/// file offsets).
|
||||||
|
pub(crate) fn read_collection<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Cow<'_, [u8]>, usize, GlobalHeapIndex), FormatError> {
|
||||||
|
let file_len = len_usize(file);
|
||||||
|
// signature(4) + version(1) + reserved(3) + collection_size(length_size),
|
||||||
|
// padded to a multiple of 8 as libhdf5 lays it out (`H5HG_SIZEOF_HDR`).
|
||||||
|
// With 8-byte lengths the padding is 0; with 4-byte lengths it is 4,
|
||||||
|
// and reading without it put every object 4 bytes early.
|
||||||
|
let header_size = pad8(8 + length_size as usize);
|
||||||
|
let header = read_exact_at(file, offset, header_size)?;
|
||||||
|
let offset = usize::try_from(offset).map_err(|_| FormatError::UnexpectedEof {
|
||||||
|
expected: usize::MAX,
|
||||||
|
available: file_len,
|
||||||
|
})?;
|
||||||
|
|
||||||
|
if header[..4] != GCOL_SIGNATURE {
|
||||||
return Err(FormatError::InvalidGlobalHeapSignature);
|
return Err(FormatError::InvalidGlobalHeapSignature);
|
||||||
}
|
}
|
||||||
|
|
||||||
let version = file_data[offset + 4];
|
let version = header[4];
|
||||||
if version != 1 {
|
if version != 1 {
|
||||||
return Err(FormatError::InvalidGlobalHeapVersion(version));
|
return Err(FormatError::InvalidGlobalHeapVersion(version));
|
||||||
}
|
}
|
||||||
|
|
||||||
let collection_size = read_length(file_data, offset + 8, length_size)?;
|
let collection_size = read_length(&header, 8, length_size)?;
|
||||||
let collection_size_usize =
|
let collection_end = usize::try_from(collection_size)
|
||||||
usize::try_from(collection_size).map_err(|_| FormatError::UnexpectedEof {
|
.ok()
|
||||||
expected: u64::MAX as usize,
|
.and_then(|size| offset.checked_add(size))
|
||||||
available: file_data.len(),
|
.ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: usize::MAX,
|
||||||
|
available: file_len,
|
||||||
})?;
|
})?;
|
||||||
let collection_end =
|
if collection_end > file_len {
|
||||||
offset
|
return Err(FormatError::UnexpectedEof {
|
||||||
.checked_add(collection_size_usize)
|
expected: collection_end,
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
available: file_len,
|
||||||
expected: usize::MAX,
|
});
|
||||||
available: file_data.len(),
|
}
|
||||||
})?;
|
let collection = read_exact_at(file, offset as u64, collection_end - offset)?;
|
||||||
|
// Positions below are file offsets; `file_data(p)` is the byte at `p`.
|
||||||
|
let file_data = |p: usize| collection[p - offset];
|
||||||
|
|
||||||
let mut pos = offset + header_size;
|
let mut pos = offset + header_size;
|
||||||
let mut objects = Vec::new();
|
let mut objects = Vec::new();
|
||||||
|
|
||||||
// Parse objects until we hit index 0 (free space) or run out of space
|
// Parse objects until we hit index 0 (free space) or run out of space
|
||||||
while pos + 2 <= collection_end {
|
while pos + 2 <= collection_end {
|
||||||
ensure_len(file_data, pos, 2)?;
|
let object_index = u16::from_le_bytes([file_data(pos), file_data(pos + 1)]);
|
||||||
let object_index = u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
|
||||||
|
|
||||||
if object_index == 0 {
|
if object_index == 0 {
|
||||||
// Free space marker — done
|
// Free space marker — done
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
|
||||||
// object_index(2) + reference_count(2) + reserved(4) + object_size(length_size)
|
// object_index(2) + reference_count(2) + reserved(4) +
|
||||||
let obj_header_size = 8 + length_size as usize;
|
// object_size(length_size), padded to 8 (`H5HG_SIZEOF_OBJHDR`).
|
||||||
ensure_len(file_data, pos, obj_header_size)?;
|
let obj_header_size = pad8(8 + length_size as usize);
|
||||||
|
ensure_len(collection_end, pos, obj_header_size)?;
|
||||||
|
|
||||||
let reference_count = u16::from_le_bytes([file_data[pos + 2], file_data[pos + 3]]);
|
let reference_count = u16::from_le_bytes([file_data(pos + 2), file_data(pos + 3)]);
|
||||||
let object_size = read_length(file_data, pos + 8, length_size)? as usize;
|
let object_size =
|
||||||
|
usize::try_from(read_length(&collection[pos - offset..], 8, length_size)?)
|
||||||
|
.map_err(|_| FormatError::Overflow("global heap object size".into()))?;
|
||||||
|
|
||||||
pos += obj_header_size;
|
pos += obj_header_size;
|
||||||
ensure_len(file_data, pos, object_size)?;
|
if pos
|
||||||
let data = file_data[pos..pos + object_size].to_vec();
|
.checked_add(object_size)
|
||||||
|
.is_none_or(|end| end > collection_end)
|
||||||
|
{
|
||||||
|
return Err(FormatError::VlDataError(object_overrun_msg(
|
||||||
|
object_index,
|
||||||
|
object_size,
|
||||||
|
collection_size,
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
|
||||||
objects.push(GlobalHeapObject {
|
objects.push(GlobalHeapObjectRef {
|
||||||
index: object_index,
|
index: object_index,
|
||||||
reference_count,
|
reference_count,
|
||||||
data,
|
offset: pos,
|
||||||
|
size: object_size,
|
||||||
});
|
});
|
||||||
|
|
||||||
// Advance past data + padding to 8-byte boundary
|
// Advance past data + padding to 8-byte boundary
|
||||||
pos += pad8(object_size);
|
pos = pos.saturating_add(pad8(object_size));
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(GlobalHeapCollection {
|
let index = GlobalHeapIndex {
|
||||||
collection_size,
|
collection_size,
|
||||||
objects,
|
objects,
|
||||||
})
|
};
|
||||||
|
Ok((collection, offset, index))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Get an object by its index.
|
/// Get an object by its index.
|
||||||
@@ -149,11 +266,12 @@ mod tests {
|
|||||||
let ls = length_size as usize;
|
let ls = length_size as usize;
|
||||||
|
|
||||||
// Calculate total size
|
// Calculate total size
|
||||||
let header_size = 8 + ls;
|
// libhdf5 pads both headers to a multiple of 8.
|
||||||
|
let header_size = pad8(8 + ls);
|
||||||
let mut obj_size_total = 0usize;
|
let mut obj_size_total = 0usize;
|
||||||
for (_, _, data) in objects {
|
for (_, _, data) in objects {
|
||||||
let obj_header = 8 + ls;
|
let obj_header = pad8(8 + ls);
|
||||||
obj_size_total += obj_header + pad8(data.len());
|
obj_size_total += obj_header + pad8(<[u8]>::len(data));
|
||||||
}
|
}
|
||||||
// Free space marker (2 bytes for index 0)
|
// Free space marker (2 bytes for index 0)
|
||||||
obj_size_total += 2;
|
obj_size_total += 2;
|
||||||
@@ -170,6 +288,7 @@ mod tests {
|
|||||||
8 => buf.extend_from_slice(&(collection_size as u64).to_le_bytes()),
|
8 => buf.extend_from_slice(&(collection_size as u64).to_le_bytes()),
|
||||||
_ => panic!("unsupported length_size"),
|
_ => panic!("unsupported length_size"),
|
||||||
}
|
}
|
||||||
|
buf.resize(header_size, 0);
|
||||||
|
|
||||||
// Objects
|
// Objects
|
||||||
for (index, ref_count, data) in objects {
|
for (index, ref_count, data) in objects {
|
||||||
@@ -177,14 +296,17 @@ mod tests {
|
|||||||
buf.extend_from_slice(&ref_count.to_le_bytes());
|
buf.extend_from_slice(&ref_count.to_le_bytes());
|
||||||
buf.extend_from_slice(&[0u8; 4]); // reserved
|
buf.extend_from_slice(&[0u8; 4]); // reserved
|
||||||
match length_size {
|
match length_size {
|
||||||
4 => buf.extend_from_slice(&(data.len() as u32).to_le_bytes()),
|
// `<[u8]>::len`: with `Storage` in scope `data.len()` on a
|
||||||
8 => buf.extend_from_slice(&(data.len() as u64).to_le_bytes()),
|
// `&&[u8]` resolves to `Storage::len` (a `u64`).
|
||||||
|
4 => buf.extend_from_slice(&(<[u8]>::len(data) as u32).to_le_bytes()),
|
||||||
|
8 => buf.extend_from_slice(&(<[u8]>::len(data) as u64).to_le_bytes()),
|
||||||
_ => panic!("unsupported"),
|
_ => panic!("unsupported"),
|
||||||
}
|
}
|
||||||
|
buf.resize(buf.len() + (pad8(8 + ls) - (8 + ls)), 0);
|
||||||
buf.extend_from_slice(data);
|
buf.extend_from_slice(data);
|
||||||
// Pad to 8 bytes
|
// Pad to 8 bytes
|
||||||
let padded = pad8(data.len());
|
let padded = pad8(<[u8]>::len(data));
|
||||||
buf.resize(buf.len() + (padded - data.len()), 0);
|
buf.resize(buf.len() + (padded - <[u8]>::len(data)), 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Free space marker
|
// Free space marker
|
||||||
@@ -252,4 +374,44 @@ mod tests {
|
|||||||
assert_eq!(coll.objects.len(), 1);
|
assert_eq!(coll.objects.len(), 1);
|
||||||
assert_eq!(coll.objects[0].data, b"test");
|
assert_eq!(coll.objects[0].data, b"test");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Collections, and every truncation of them, index and parse
|
||||||
|
/// identically through a `read_at`-only storage: two reads each.
|
||||||
|
#[test]
|
||||||
|
fn storage_parse_matches_slice_parse() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let objs: &[(u16, u16, &[u8])] = &[(1, 1, b"hello"), (2, 3, b"a longer object")];
|
||||||
|
for ls in [4u8, 8] {
|
||||||
|
let coll = build_collection(objs, ls);
|
||||||
|
let mut corrupt = coll.clone();
|
||||||
|
corrupt[8] = 200; // collection size past the end of the file
|
||||||
|
let mut overrun = coll.clone();
|
||||||
|
let size_at = pad8(8 + ls as usize) + 8;
|
||||||
|
overrun[size_at] = 250; // first object runs past the collection
|
||||||
|
for full in [coll, corrupt, overrun] {
|
||||||
|
for at in [0usize, 5] {
|
||||||
|
for cut in 0..=full.len() {
|
||||||
|
let mut f = vec![0u8; at];
|
||||||
|
f.extend_from_slice(&full[..cut]);
|
||||||
|
let storage = CountingStorage::new(f.clone());
|
||||||
|
let want = GlobalHeapCollection::parse(&f, at, ls);
|
||||||
|
let got = GlobalHeapCollection::parse_in(&storage, at as u64, ls);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
let want = GlobalHeapCollection::parse_index(&f, at, ls);
|
||||||
|
let got = GlobalHeapCollection::parse_index_in(&storage, at as u64, ls);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let storage = CountingStorage::new(build_collection(objs, 8));
|
||||||
|
assert_eq!(
|
||||||
|
GlobalHeapCollection::parse_in(&storage, 0, 8)
|
||||||
|
.unwrap()
|
||||||
|
.objects
|
||||||
|
.len(),
|
||||||
|
2
|
||||||
|
);
|
||||||
|
assert_eq!(storage.reads(), 2);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -3,11 +3,13 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{string::String, vec::Vec};
|
use alloc::{string::String, vec::Vec};
|
||||||
|
|
||||||
use crate::btree_v1::collect_symbol_table_nodes;
|
use crate::addr::checked_addr;
|
||||||
|
use crate::btree_v1::{BTreeV1Node, collect_symbol_table_nodes_in};
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::local_heap::LocalHeap;
|
use crate::local_heap::LocalHeap;
|
||||||
use crate::message_type::MessageType;
|
use crate::message_type::MessageType;
|
||||||
use crate::object_header::ObjectHeader;
|
use crate::object_header::ObjectHeader;
|
||||||
|
use crate::storage::Storage;
|
||||||
use crate::symbol_table::{SymbolTableMessage, SymbolTableNode};
|
use crate::symbol_table::{SymbolTableMessage, SymbolTableNode};
|
||||||
|
|
||||||
/// A resolved group entry (child name + object header address).
|
/// A resolved group entry (child name + object header address).
|
||||||
@@ -21,43 +23,212 @@ pub struct GroupEntry {
|
|||||||
pub cache_type: u32,
|
pub cache_type: u32,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Given a SymbolTableMessage, resolve all group children.
|
/// Given a SymbolTableMessage, resolve all group children: the group's
|
||||||
|
/// listing.
|
||||||
|
///
|
||||||
|
/// An entry with an empty name fails the listing with
|
||||||
|
/// [`FormatError::InvalidLinkName`], as it fails libhdf5's link iteration
|
||||||
|
/// (`H5G__ent_to_link`: "invalid link name"). Looking a name up
|
||||||
|
/// ([`resolve_path`], and the path resolution in
|
||||||
|
/// [`crate::group_v2::resolve_path_any`]) still works in such a group, as it
|
||||||
|
/// does in libhdf5.
|
||||||
pub fn resolve_v1_group_entries(
|
pub fn resolve_v1_group_entries(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
sym_table_msg: &SymbolTableMessage,
|
sym_table_msg: &SymbolTableMessage,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
|
resolve_v1_group_entries_in(file_data, sym_table_msg, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`resolve_v1_group_entries`] over any [`Storage`].
|
||||||
|
pub fn resolve_v1_group_entries_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
|
let entries = v1_group_entries(file_data, sym_table_msg, offset_size, length_size, true)?;
|
||||||
|
if entries.iter().any(|e| e.name.is_empty()) {
|
||||||
|
return Err(FormatError::InvalidLinkName);
|
||||||
|
}
|
||||||
|
Ok(entries)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every entry of a v1 group, empty names included — for looking a name up,
|
||||||
|
/// which never matches an empty name.
|
||||||
|
///
|
||||||
|
/// With `hint_headers` (a listing, whose children are usually opened
|
||||||
|
/// next), each entry's object header is hinted (see
|
||||||
|
/// [`Storage::hint`]) as soon as its symbol table node is read.
|
||||||
|
pub(crate) fn v1_group_entries<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
hint_headers: bool,
|
||||||
) -> Result<Vec<GroupEntry>, FormatError> {
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
// Parse local heap
|
// Parse local heap
|
||||||
let heap = LocalHeap::parse(
|
let heap = LocalHeap::parse_in(
|
||||||
file_data,
|
file_data,
|
||||||
sym_table_msg.local_heap_address as usize,
|
checked_addr(sym_table_msg.local_heap_address)?,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
|
|
||||||
// Collect all SNOD addresses from B-tree
|
// Collect all SNOD addresses from B-tree
|
||||||
let snod_addrs = collect_symbol_table_nodes(
|
let snod_addrs = collect_symbol_table_nodes_in(
|
||||||
file_data,
|
file_data,
|
||||||
sym_table_msg.btree_address,
|
sym_table_msg.btree_address,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
|
|
||||||
|
// The names are read one by one from the heap's data segment; read
|
||||||
|
// (up to 1 MiB of) it first, so a storage that records what it lacks
|
||||||
|
// asks for it at once (see `storage::touch`).
|
||||||
|
if !snod_addrs.is_empty() {
|
||||||
|
let len = usize::try_from(heap.data_segment_size).map_or(1 << 20, |n| n.min(1 << 20));
|
||||||
|
crate::storage::touch(file_data, heap.data_segment_address, len);
|
||||||
|
}
|
||||||
|
|
||||||
let mut entries = Vec::new();
|
let mut entries = Vec::new();
|
||||||
|
let mut heap_checked = false;
|
||||||
|
// After the first node that fails, the others are only read (as
|
||||||
|
// `storage::touch` does); that error is returned.
|
||||||
|
let mut failed = None;
|
||||||
for snod_addr in snod_addrs {
|
for snod_addr in snod_addrs {
|
||||||
let snod = SymbolTableNode::parse(file_data, snod_addr as usize, offset_size)?;
|
let snod = checked_addr(snod_addr)
|
||||||
for entry in &snod.entries {
|
.and_then(|a| SymbolTableNode::parse_in(file_data, a, offset_size));
|
||||||
let name = heap.read_string(file_data, entry.link_name_offset)?;
|
if hint_headers && let Ok(snod) = &snod {
|
||||||
entries.push(GroupEntry {
|
for entry in &snod.entries {
|
||||||
name,
|
if entry.object_header_address != u64::MAX {
|
||||||
object_header_address: entry.object_header_address,
|
file_data.hint(
|
||||||
cache_type: entry.cache_type,
|
entry.object_header_address,
|
||||||
});
|
crate::object_header::OBJECT_HEADER_HINT_LEN,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if failed.is_some() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let node = || -> Result<(), FormatError> {
|
||||||
|
let snod = snod?;
|
||||||
|
for entry in &snod.entries {
|
||||||
|
// Like libhdf5, look at the heap's free list only once a name
|
||||||
|
// is needed: an empty group with a damaged heap still lists.
|
||||||
|
if !heap_checked {
|
||||||
|
heap.validate_free_list_in(file_data, length_size)?;
|
||||||
|
heap_checked = true;
|
||||||
|
}
|
||||||
|
let name = heap.read_string_in(file_data, entry.link_name_offset)?;
|
||||||
|
entries.push(GroupEntry {
|
||||||
|
name,
|
||||||
|
object_header_address: entry.object_header_address,
|
||||||
|
cache_type: entry.cache_type,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
};
|
||||||
|
if let Err(e) = node() {
|
||||||
|
failed = Some(e);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(entries)
|
match failed {
|
||||||
|
Some(e) => Err(e),
|
||||||
|
None => Ok(entries),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The entry called `name` in a v1 group, looked up as libhdf5 looks it up
|
||||||
|
/// (`H5G__stab_lookup`: `H5B_find` down the group's B-tree, then
|
||||||
|
/// `H5G__node_found` in one symbol table node) instead of by reading every
|
||||||
|
/// entry: at each node, the child whose key interval holds the name
|
||||||
|
/// (left key < name <= right key, keys being names in the local heap,
|
||||||
|
/// compared bytewise as `strcmp` does) is found by binary search. Reads
|
||||||
|
/// O(depth) nodes and names, where listing reads the whole group.
|
||||||
|
///
|
||||||
|
/// `Ok(None)` when the search does not lead to the name. In a group whose
|
||||||
|
/// B-tree is out of name order (damaged, or made by hand) that does not
|
||||||
|
/// prove it absent, so callers then fall back to reading every entry;
|
||||||
|
/// libhdf5 would report it missing.
|
||||||
|
pub(crate) fn find_v1_entry<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<GroupEntry>, FormatError> {
|
||||||
|
let heap = LocalHeap::parse_in(
|
||||||
|
file_data,
|
||||||
|
checked_addr(sym_table_msg.local_heap_address)?,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
// The keys and names are read from the heap's data segment one at a
|
||||||
|
// time: hint it (see `Storage::hint`).
|
||||||
|
file_data.hint(
|
||||||
|
heap.data_segment_address,
|
||||||
|
usize::try_from(heap.data_segment_size).map_or(1 << 20, |n| n.min(1 << 20)),
|
||||||
|
);
|
||||||
|
// As in listing: the heap's free list is checked before a name is used.
|
||||||
|
let mut heap_checked = false;
|
||||||
|
let mut name_at = |offset: u64| -> Result<String, FormatError> {
|
||||||
|
if !heap_checked {
|
||||||
|
heap.validate_free_list_in(file_data, length_size)?;
|
||||||
|
heap_checked = true;
|
||||||
|
}
|
||||||
|
heap.read_string_in(file_data, offset)
|
||||||
|
};
|
||||||
|
let want = name.as_bytes();
|
||||||
|
let mut address = sym_table_msg.btree_address;
|
||||||
|
for _ in 0..=crate::btree_v1::MAX_BTREE_DEPTH {
|
||||||
|
let node =
|
||||||
|
BTreeV1Node::parse_in(file_data, checked_addr(address)?, offset_size, length_size)?;
|
||||||
|
if node.node_type != 0 {
|
||||||
|
return Err(FormatError::InvalidBTreeNodeType(node.node_type));
|
||||||
|
}
|
||||||
|
// H5B_find's binary search with H5G__node_cmp3: go left when the
|
||||||
|
// name sorts at or before the left key, right when after the right
|
||||||
|
// key; otherwise this child holds it.
|
||||||
|
let (mut lo, mut hi) = (0usize, node.children.len());
|
||||||
|
let mut child = None;
|
||||||
|
while lo < hi {
|
||||||
|
let i = lo + (hi - lo) / 2;
|
||||||
|
let (Some(&left), Some(&right)) = (node.keys.get(i), node.keys.get(i + 1)) else {
|
||||||
|
return Ok(None);
|
||||||
|
};
|
||||||
|
if want <= name_at(left)?.as_bytes() {
|
||||||
|
hi = i;
|
||||||
|
} else if want > name_at(right)?.as_bytes() {
|
||||||
|
lo = i + 1;
|
||||||
|
} else {
|
||||||
|
child = Some(node.children[i]);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let Some(child) = child else {
|
||||||
|
return Ok(None);
|
||||||
|
};
|
||||||
|
if node.node_level > 0 {
|
||||||
|
address = child;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let snod = SymbolTableNode::parse_in(file_data, checked_addr(child)?, offset_size)?;
|
||||||
|
for entry in &snod.entries {
|
||||||
|
if name_at(entry.link_name_offset)?.as_bytes() == want {
|
||||||
|
return Ok(Some(GroupEntry {
|
||||||
|
name: String::from(name),
|
||||||
|
object_header_address: entry.object_header_address,
|
||||||
|
cache_type: entry.cache_type,
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return Ok(None);
|
||||||
|
}
|
||||||
|
Err(FormatError::NestingDepthExceeded)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Symbol table cache type for a soft link: the scratch pad's first four bytes
|
/// Symbol table cache type for a soft link: the scratch pad's first four bytes
|
||||||
@@ -73,25 +244,99 @@ pub fn find_v1_soft_link(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Option<String>, FormatError> {
|
) -> Result<Option<String>, FormatError> {
|
||||||
let heap = LocalHeap::parse(
|
find_v1_soft_link_in(file_data, sym_table_msg, name, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`find_v1_soft_link`] over any [`Storage`].
|
||||||
|
pub fn find_v1_soft_link_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<String>, FormatError> {
|
||||||
|
let mut found = None;
|
||||||
|
for_each_v1_soft_link(
|
||||||
file_data,
|
file_data,
|
||||||
sym_table_msg.local_heap_address as usize,
|
sym_table_msg,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
|link_name| link_name == name,
|
||||||
|
|_, target| {
|
||||||
|
found = Some(target);
|
||||||
|
false
|
||||||
|
},
|
||||||
|
)?;
|
||||||
|
Ok(found)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every soft link in a v1 group, as `(name, target path)`.
|
||||||
|
pub fn v1_soft_links(
|
||||||
|
file_data: &[u8],
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<(String, String)>, FormatError> {
|
||||||
|
v1_soft_links_in(file_data, sym_table_msg, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`v1_soft_links`] over any [`Storage`].
|
||||||
|
pub fn v1_soft_links_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<(String, String)>, FormatError> {
|
||||||
|
let mut links = Vec::new();
|
||||||
|
for_each_v1_soft_link(
|
||||||
|
file_data,
|
||||||
|
sym_table_msg,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
|_| true,
|
||||||
|
|name, target| {
|
||||||
|
links.push((String::from(name), target));
|
||||||
|
true
|
||||||
|
},
|
||||||
|
)?;
|
||||||
|
Ok(links)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Visit the soft links of a v1 group whose name passes `wanted`, with their
|
||||||
|
/// target paths, until `visit` returns false.
|
||||||
|
fn for_each_v1_soft_link<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
wanted: impl Fn(&str) -> bool,
|
||||||
|
mut visit: impl FnMut(&str, String) -> bool,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
let heap = LocalHeap::parse_in(
|
||||||
|
file_data,
|
||||||
|
checked_addr(sym_table_msg.local_heap_address)?,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
let snod_addrs = collect_symbol_table_nodes(
|
let snod_addrs = collect_symbol_table_nodes_in(
|
||||||
file_data,
|
file_data,
|
||||||
sym_table_msg.btree_address,
|
sym_table_msg.btree_address,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
|
let mut heap_checked = false;
|
||||||
for snod_addr in snod_addrs {
|
for snod_addr in snod_addrs {
|
||||||
let snod = SymbolTableNode::parse(file_data, snod_addr as usize, offset_size)?;
|
let snod = SymbolTableNode::parse_in(file_data, checked_addr(snod_addr)?, offset_size)?;
|
||||||
for entry in &snod.entries {
|
for entry in &snod.entries {
|
||||||
if entry.cache_type != CACHE_TYPE_SOFT_LINK {
|
if entry.cache_type != CACHE_TYPE_SOFT_LINK {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if heap.read_string(file_data, entry.link_name_offset)? != name {
|
if !heap_checked {
|
||||||
|
heap.validate_free_list_in(file_data, length_size)?;
|
||||||
|
heap_checked = true;
|
||||||
|
}
|
||||||
|
let name = heap.read_string_in(file_data, entry.link_name_offset)?;
|
||||||
|
if !wanted(&name) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let value_offset = u32::from_le_bytes([
|
let value_offset = u32::from_le_bytes([
|
||||||
@@ -100,12 +345,19 @@ pub fn find_v1_soft_link(
|
|||||||
entry.scratch_pad[2],
|
entry.scratch_pad[2],
|
||||||
entry.scratch_pad[3],
|
entry.scratch_pad[3],
|
||||||
]);
|
]);
|
||||||
return heap
|
let target = heap.read_string_in(file_data, u64::from(value_offset))?;
|
||||||
.read_string(file_data, u64::from(value_offset))
|
if !visit(&name, target) {
|
||||||
.map(Some);
|
return Ok(());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Ok(None)
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a v1 symbol-table entry is a soft link (no object header of its
|
||||||
|
/// own; its target path is in the local heap).
|
||||||
|
pub fn is_v1_soft_link(entry: &GroupEntry) -> bool {
|
||||||
|
entry.cache_type == CACHE_TYPE_SOFT_LINK
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract the SymbolTableMessage from an object header's messages.
|
/// Extract the SymbolTableMessage from an object header's messages.
|
||||||
@@ -131,6 +383,17 @@ pub fn resolve_path(
|
|||||||
path: &str,
|
path: &str,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
resolve_path_in(file_data, root_sym_table, path, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`resolve_path`] over any [`Storage`].
|
||||||
|
pub fn resolve_path_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
root_sym_table: &SymbolTableMessage,
|
||||||
|
path: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<u64, FormatError> {
|
) -> Result<u64, FormatError> {
|
||||||
let components: Vec<&str> = path.split('/').filter(|s| !s.is_empty()).collect();
|
let components: Vec<&str> = path.split('/').filter(|s| !s.is_empty()).collect();
|
||||||
if components.is_empty() {
|
if components.is_empty() {
|
||||||
@@ -140,8 +403,13 @@ pub fn resolve_path(
|
|||||||
let mut current_sym_table = root_sym_table.clone();
|
let mut current_sym_table = root_sym_table.clone();
|
||||||
|
|
||||||
for (i, component) in components.iter().enumerate() {
|
for (i, component) in components.iter().enumerate() {
|
||||||
let entries =
|
let entries = v1_group_entries(
|
||||||
resolve_v1_group_entries(file_data, ¤t_sym_table, offset_size, length_size)?;
|
file_data,
|
||||||
|
¤t_sym_table,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
false,
|
||||||
|
)?;
|
||||||
|
|
||||||
let found = entries.iter().find(|e| e.name == *component);
|
let found = entries.iter().find(|e| e.name == *component);
|
||||||
match found {
|
match found {
|
||||||
@@ -151,9 +419,9 @@ pub fn resolve_path(
|
|||||||
return Ok(entry.object_header_address);
|
return Ok(entry.object_header_address);
|
||||||
}
|
}
|
||||||
// Not last — must be a group, parse its object header to get symbol table
|
// Not last — must be a group, parse its object header to get symbol table
|
||||||
let obj_header = ObjectHeader::parse(
|
let obj_header = ObjectHeader::parse_in(
|
||||||
file_data,
|
file_data,
|
||||||
entry.object_header_address as usize,
|
checked_addr(entry.object_header_address)?,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
@@ -358,6 +626,22 @@ mod tests {
|
|||||||
assert_eq!(entries[1].object_header_address, 0x2000);
|
assert_eq!(entries[1].object_header_address, 0x2000);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// cve-2021-46244 `/BAG_root`: a symbol-table entry with an empty name.
|
||||||
|
/// libhdf5 fails the group's listing ("invalid link name"); a lookup of
|
||||||
|
/// the other names still works.
|
||||||
|
#[test]
|
||||||
|
fn empty_entry_name_fails_the_listing_not_a_lookup() {
|
||||||
|
let (file, msg) = build_synthetic_group(&[("", 0x1000, 0), ("elevation", 0x2000, 0)], 8, 8);
|
||||||
|
assert_eq!(
|
||||||
|
resolve_v1_group_entries(&file, &msg, 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidLinkName
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
resolve_path(&file, &msg, "elevation", 8, 8).unwrap(),
|
||||||
|
0x2000
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn resolve_path_single_level() {
|
fn resolve_path_single_level() {
|
||||||
let (file, msg) =
|
let (file, msg) =
|
||||||
|
|||||||
@@ -6,7 +6,14 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{string::String, vec::Vec};
|
use alloc::{string::String, vec::Vec};
|
||||||
|
|
||||||
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::collections::BTreeSet;
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
use std::collections::BTreeSet;
|
||||||
|
|
||||||
|
use crate::addr::checked_addr;
|
||||||
|
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records_in, find_btree_v2_records_in};
|
||||||
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::fractal_heap::FractalHeapHeader;
|
use crate::fractal_heap::FractalHeapHeader;
|
||||||
use crate::group_v1::{self, GroupEntry};
|
use crate::group_v1::{self, GroupEntry};
|
||||||
@@ -14,6 +21,7 @@ use crate::link_info::LinkInfoMessage;
|
|||||||
use crate::link_message::{LinkMessage, LinkTarget};
|
use crate::link_message::{LinkMessage, LinkTarget};
|
||||||
use crate::message_type::MessageType;
|
use crate::message_type::MessageType;
|
||||||
use crate::object_header::ObjectHeader;
|
use crate::object_header::ObjectHeader;
|
||||||
|
use crate::storage::Storage;
|
||||||
use crate::superblock::Superblock;
|
use crate::superblock::Superblock;
|
||||||
use crate::symbol_table::SymbolTableMessage;
|
use crate::symbol_table::SymbolTableMessage;
|
||||||
|
|
||||||
@@ -25,6 +33,16 @@ pub fn resolve_v2_group_entries(
|
|||||||
object_header: &ObjectHeader,
|
object_header: &ObjectHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
|
resolve_v2_group_entries_in(file_data, object_header, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`resolve_v2_group_entries`] over any [`Storage`].
|
||||||
|
pub fn resolve_v2_group_entries_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
object_header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<Vec<GroupEntry>, FormatError> {
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
// Look for Link Info message to determine storage type
|
// Look for Link Info message to determine storage type
|
||||||
let link_info = find_link_info(object_header, offset_size)?;
|
let link_info = find_link_info(object_header, offset_size)?;
|
||||||
@@ -38,6 +56,24 @@ pub fn resolve_v2_group_entries(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// First user-defined link type (HDF5 reserves 2-63; 64 is external).
|
||||||
|
const FIRST_USER_DEFINED_LINK_TYPE: u8 = 65;
|
||||||
|
|
||||||
|
/// Parse a Link message, or `None` for a user-defined link (type 65-255).
|
||||||
|
///
|
||||||
|
/// A user-defined link's target is only meaningful to the application that
|
||||||
|
/// registered its class, so, like libhdf5 without that class, we cannot
|
||||||
|
/// follow it. Leaving it out lets the rest of the group be listed and
|
||||||
|
/// resolved instead of one such link failing the whole group; reserved
|
||||||
|
/// types (2-63) are still an error.
|
||||||
|
fn parse_link(data: &[u8], offset_size: u8) -> Result<Option<LinkMessage>, FormatError> {
|
||||||
|
match LinkMessage::parse(data, offset_size) {
|
||||||
|
Ok(link) => Ok(Some(link)),
|
||||||
|
Err(FormatError::InvalidLinkType(t)) if t >= FIRST_USER_DEFINED_LINK_TYPE => Ok(None),
|
||||||
|
Err(e) => Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Extract link entries from Link messages directly in the object header (compact storage).
|
/// Extract link entries from Link messages directly in the object header (compact storage).
|
||||||
fn resolve_compact_entries(
|
fn resolve_compact_entries(
|
||||||
object_header: &ObjectHeader,
|
object_header: &ObjectHeader,
|
||||||
@@ -46,7 +82,9 @@ fn resolve_compact_entries(
|
|||||||
let mut entries = Vec::new();
|
let mut entries = Vec::new();
|
||||||
for msg in &object_header.messages {
|
for msg in &object_header.messages {
|
||||||
if msg.msg_type == MessageType::Link {
|
if msg.msg_type == MessageType::Link {
|
||||||
let link = LinkMessage::parse(&msg.data, offset_size)?;
|
let Some(link) = parse_link(&msg.data, offset_size)? else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
if let LinkTarget::Hard {
|
if let LinkTarget::Hard {
|
||||||
object_header_address,
|
object_header_address,
|
||||||
} = link.link_target
|
} = link.link_target
|
||||||
@@ -63,25 +101,66 @@ fn resolve_compact_entries(
|
|||||||
Ok(entries)
|
Ok(entries)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Visit every link in dense storage (fractal heap + B-tree v2 name index).
|
/// What a version-2 B-tree header takes with 8-byte offsets and lengths
|
||||||
fn for_each_dense_link(
|
/// (22 bytes of fields, the root node's address and record count, and the
|
||||||
file_data: &[u8],
|
/// checksum), rounded up: hinted before one is read.
|
||||||
|
const BTREE_V2_HEADER_HINT_LEN: usize = 64;
|
||||||
|
|
||||||
|
/// The fractal heap of a dense group. The name index's header and the
|
||||||
|
/// heap's root block are read next, whatever the lookup: they are hinted
|
||||||
|
/// (see [`Storage::hint`]) so that a storage fetching between attempts
|
||||||
|
/// gets them in the same round trip as the heap's header.
|
||||||
|
fn dense_heap<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
link_info: &LinkInfoMessage,
|
link_info: &LinkInfoMessage,
|
||||||
fh_addr: u64,
|
fh_addr: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<FractalHeapHeader, FormatError> {
|
||||||
|
if let Some(btree_addr) = link_info.btree_name_index_address {
|
||||||
|
file_data.hint(btree_addr, BTREE_V2_HEADER_HINT_LEN);
|
||||||
|
}
|
||||||
|
let fh =
|
||||||
|
FractalHeapHeader::parse_in(file_data, checked_addr(fh_addr)?, offset_size, length_size)?;
|
||||||
|
fh.hint_root_block(file_data);
|
||||||
|
Ok(fh)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Visit every link in dense storage (fractal heap + B-tree v2 name index).
|
||||||
|
///
|
||||||
|
/// With `hint_headers` (a listing, whose children are usually opened
|
||||||
|
/// next), the object header of every hard link is hinted (see
|
||||||
|
/// [`Storage::hint`]) as soon as the link is read, even after a failure.
|
||||||
|
fn for_each_dense_link<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
link_info: &LinkInfoMessage,
|
||||||
|
fh_addr: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
hint_headers: bool,
|
||||||
mut visit: impl FnMut(LinkMessage),
|
mut visit: impl FnMut(LinkMessage),
|
||||||
) -> Result<(), FormatError> {
|
) -> Result<(), FormatError> {
|
||||||
// Parse fractal heap
|
let fh = dense_heap(file_data, link_info, fh_addr, offset_size, length_size)?;
|
||||||
let fh = FractalHeapHeader::parse(file_data, fh_addr as usize, offset_size, length_size)?;
|
if hint_headers {
|
||||||
|
// Every link is read: so is every block of the heap.
|
||||||
|
fh.hint_managed_blocks(file_data);
|
||||||
|
}
|
||||||
|
|
||||||
// Parse B-tree v2 for name index
|
// Parse B-tree v2 for name index
|
||||||
let btree_addr = link_info
|
let btree_addr = link_info
|
||||||
.btree_name_index_address
|
.btree_name_index_address
|
||||||
.ok_or_else(|| FormatError::PathNotFound(String::from("no B-tree v2 name index")))?;
|
.ok_or_else(|| FormatError::PathNotFound(String::from("no B-tree v2 name index")))?;
|
||||||
let btree_hdr = BTreeV2Header::parse(file_data, btree_addr as usize, offset_size, length_size)?;
|
let btree_hdr = BTreeV2Header::parse_in(
|
||||||
let records = collect_btree_v2_records(file_data, &btree_hdr, offset_size, length_size)?;
|
file_data,
|
||||||
|
checked_addr(btree_addr)?,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
let records = collect_btree_v2_records_in(file_data, &btree_hdr, offset_size, length_size)?;
|
||||||
|
|
||||||
|
// After the first link that fails, the others are only read, not
|
||||||
|
// visited (a touch, see `storage::touch`); that error is returned.
|
||||||
|
let mut failed = None;
|
||||||
for record in &records {
|
for record in &records {
|
||||||
// For type 5 (name index): hash(4) + heap_id(heap_id_length)
|
// For type 5 (name index): hash(4) + heap_id(heap_id_length)
|
||||||
// For type 6 (creation order): creation_order(8) + heap_id(heap_id_length)
|
// For type 6 (creation order): creation_order(8) + heap_id(heap_id_length)
|
||||||
@@ -97,15 +176,41 @@ fn for_each_dense_link(
|
|||||||
let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize];
|
let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize];
|
||||||
|
|
||||||
// Read managed object from fractal heap
|
// Read managed object from fractal heap
|
||||||
let link_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
let link = fh
|
||||||
visit(LinkMessage::parse(&link_data, offset_size)?);
|
.read_managed_object_in(file_data, id_bytes, offset_size)
|
||||||
|
.and_then(|d| parse_link(&d, offset_size));
|
||||||
|
if hint_headers
|
||||||
|
&& let Ok(Some(LinkMessage {
|
||||||
|
link_target:
|
||||||
|
LinkTarget::Hard {
|
||||||
|
object_header_address,
|
||||||
|
},
|
||||||
|
..
|
||||||
|
})) = &link
|
||||||
|
{
|
||||||
|
file_data.hint(
|
||||||
|
*object_header_address,
|
||||||
|
crate::object_header::OBJECT_HEADER_HINT_LEN,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if failed.is_some() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
match link {
|
||||||
|
Ok(Some(link)) => visit(link),
|
||||||
|
Ok(None) => {}
|
||||||
|
Err(e) => failed = Some(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
match failed {
|
||||||
|
Some(e) => Err(e),
|
||||||
|
None => Ok(()),
|
||||||
}
|
}
|
||||||
Ok(())
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve entries from dense storage (fractal heap + B-tree v2).
|
/// Resolve entries from dense storage (fractal heap + B-tree v2).
|
||||||
fn resolve_dense_entries(
|
fn resolve_dense_entries<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file_data: &S,
|
||||||
link_info: &LinkInfoMessage,
|
link_info: &LinkInfoMessage,
|
||||||
fh_addr: u64,
|
fh_addr: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
@@ -118,6 +223,7 @@ fn resolve_dense_entries(
|
|||||||
fh_addr,
|
fh_addr,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
|
true,
|
||||||
|link| {
|
|link| {
|
||||||
if let LinkTarget::Hard {
|
if let LinkTarget::Hard {
|
||||||
object_header_address,
|
object_header_address,
|
||||||
@@ -134,60 +240,272 @@ fn resolve_dense_entries(
|
|||||||
Ok(entries)
|
Ok(entries)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The soft or external link called `name` in this group, if there is one.
|
/// The soft link called `name` in a v1 (symbol table) group, if there is
|
||||||
/// Hard links are what `resolve_group_entries` returns; this is consulted only
|
/// one. Hard links are what `resolve_group_entries` returns; this is
|
||||||
/// when a path component isn't among them.
|
/// consulted only when a path component isn't among them.
|
||||||
fn find_symbolic_link(
|
fn find_v1_symbolic_link<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file_data: &S,
|
||||||
object_header: &ObjectHeader,
|
object_header: &ObjectHeader,
|
||||||
name: &str,
|
name: &str,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Option<LinkTarget>, FormatError> {
|
) -> Result<Option<LinkTarget>, FormatError> {
|
||||||
if is_v1_group(object_header) {
|
let Some(sym_msg) = object_header
|
||||||
let Some(sym_msg) = object_header
|
.messages
|
||||||
.messages
|
.iter()
|
||||||
.iter()
|
.find(|m| m.msg_type == MessageType::SymbolTable)
|
||||||
.find(|m| m.msg_type == MessageType::SymbolTable)
|
else {
|
||||||
else {
|
|
||||||
return Ok(None);
|
|
||||||
};
|
|
||||||
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
|
||||||
return group_v1::find_v1_soft_link(file_data, &stm, name, offset_size, length_size)
|
|
||||||
.map(|target| target.map(|target_path| LinkTarget::Soft { target_path }));
|
|
||||||
}
|
|
||||||
if !is_v2_group(object_header) {
|
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
};
|
||||||
let is_symbolic = |t: &LinkTarget| !matches!(t, LinkTarget::Hard { .. });
|
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
||||||
|
group_v1::find_v1_soft_link_in(file_data, &stm, name, offset_size, length_size)
|
||||||
|
.map(|target| target.map(|target_path| LinkTarget::Soft { target_path }))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// B-tree v2 record type of a dense group's link name index.
|
||||||
|
const LINK_NAME_INDEX: u8 = 5;
|
||||||
|
|
||||||
|
/// The links called `name` in a v2 group (a valid group has at most one),
|
||||||
|
/// in storage order: header message order for a compact group, name index
|
||||||
|
/// order for a dense one.
|
||||||
|
///
|
||||||
|
/// In dense storage the link name index (a v2 B-tree of lookup3 name
|
||||||
|
/// hashes, record type 5) is descended to the records with the name's hash,
|
||||||
|
/// and only their links are read from the heap — O(log n) instead of every
|
||||||
|
/// link. libhdf5 orders records with equal hashes by name; all of them are
|
||||||
|
/// read and compared here, so that order does not matter. An index of
|
||||||
|
/// another type is scanned in full.
|
||||||
|
fn links_named<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
object_header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<LinkMessage>, FormatError> {
|
||||||
|
let mut found = Vec::new();
|
||||||
let link_info = find_link_info(object_header, offset_size)?;
|
let link_info = find_link_info(object_header, offset_size)?;
|
||||||
let mut found = None;
|
let Some(fh_addr) = link_info.fractal_heap_address else {
|
||||||
if let Some(fh_addr) = link_info.fractal_heap_address {
|
for msg in &object_header.messages {
|
||||||
|
if msg.msg_type == MessageType::Link
|
||||||
|
&& let Some(link) = parse_link(&msg.data, offset_size)?
|
||||||
|
&& link.name == name
|
||||||
|
{
|
||||||
|
found.push(link);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return Ok(found);
|
||||||
|
};
|
||||||
|
|
||||||
|
let fh = dense_heap(file_data, &link_info, fh_addr, offset_size, length_size)?;
|
||||||
|
let btree_addr = link_info
|
||||||
|
.btree_name_index_address
|
||||||
|
.ok_or_else(|| FormatError::PathNotFound(String::from("no B-tree v2 name index")))?;
|
||||||
|
let btree_hdr = BTreeV2Header::parse_in(
|
||||||
|
file_data,
|
||||||
|
checked_addr(btree_addr)?,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
if btree_hdr.tree_type != LINK_NAME_INDEX {
|
||||||
for_each_dense_link(
|
for_each_dense_link(
|
||||||
file_data,
|
file_data,
|
||||||
&link_info,
|
&link_info,
|
||||||
fh_addr,
|
fh_addr,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
|
false,
|
||||||
|link| {
|
|link| {
|
||||||
if link.name == name && is_symbolic(&link.link_target) {
|
if link.name == name {
|
||||||
found = Some(link.link_target);
|
found.push(link);
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
)?;
|
)?;
|
||||||
} else {
|
return Ok(found);
|
||||||
for msg in &object_header.messages {
|
}
|
||||||
if msg.msg_type == MessageType::Link {
|
|
||||||
let link = LinkMessage::parse(&msg.data, offset_size)?;
|
// Record: hash(4) + heap ID.
|
||||||
if link.name == name && is_symbolic(&link.link_target) {
|
let hash = jenkins_lookup3(name.as_bytes());
|
||||||
found = Some(link.link_target);
|
let records = find_btree_v2_records_in(file_data, &btree_hdr, offset_size, &mut |r| {
|
||||||
}
|
match r.get(..4) {
|
||||||
}
|
Some(h) => u32::from_le_bytes([h[0], h[1], h[2], h[3]]).cmp(&hash),
|
||||||
|
// Too short to hold a hash (a corrupt record size): never a match.
|
||||||
|
None => core::cmp::Ordering::Less,
|
||||||
|
}
|
||||||
|
})?;
|
||||||
|
let id_len = usize::from(fh.heap_id_length);
|
||||||
|
for record in &records {
|
||||||
|
let Some(id_bytes) = record.data.get(4..4 + id_len) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let link_data = fh.read_managed_object_in(file_data, id_bytes, offset_size)?;
|
||||||
|
if let Some(link) = parse_link(&link_data, offset_size)?
|
||||||
|
&& link.name == name
|
||||||
|
{
|
||||||
|
found.push(link);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Ok(found)
|
Ok(found)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The link called `name` in a v2 group, if any.
|
||||||
|
///
|
||||||
|
/// A valid group has at most one; libhdf5 cannot create two. If a damaged
|
||||||
|
/// or hand-made group has several, the first wins and the rest are
|
||||||
|
/// ignored, whatever their kind and even if the first cannot be followed.
|
||||||
|
/// That is libhdf5's rule for a compact group (`H5G__compact_lookup` stops
|
||||||
|
/// at the first Link message of that name; h5py then fails to open a
|
||||||
|
/// dangling first link although a later one resolves). For a dense group
|
||||||
|
/// "first" is first in name index order; libhdf5 binary-searches the index
|
||||||
|
/// and may land on another of several exact duplicates. The listing
|
||||||
|
/// ([`resolve_group_children`]), [`resolve_child`] and path resolution all
|
||||||
|
/// apply this rule, so they agree.
|
||||||
|
fn first_link_named<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
object_header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<LinkMessage>, FormatError> {
|
||||||
|
Ok(
|
||||||
|
links_named(file_data, object_header, name, offset_size, length_size)?
|
||||||
|
.into_iter()
|
||||||
|
.next(),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The link [`resolve_path_any`] follows for one path component `name` of
|
||||||
|
/// the group with header `object_header`: a hard link (as `Hard`), else a
|
||||||
|
/// soft or external link of that name, else `None`. Fails with
|
||||||
|
/// `PathNotFound` if the object is not a group.
|
||||||
|
fn lookup_link<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
object_header: &ObjectHeader,
|
||||||
|
name: &str,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<LinkTarget>, FormatError> {
|
||||||
|
if is_v1_group(object_header) {
|
||||||
|
// Down the group's B-tree, as libhdf5 looks a name up; only when
|
||||||
|
// that does not find a hard link of that name is every entry read
|
||||||
|
// (a soft link, a group whose B-tree is out of order). A storage
|
||||||
|
// error (a read a restartable storage has not fetched yet) is
|
||||||
|
// returned as is: reading every entry would not get further.
|
||||||
|
if let Some(sym_msg) = object_header
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::SymbolTable)
|
||||||
|
{
|
||||||
|
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
||||||
|
match group_v1::find_v1_entry(file_data, &stm, name, offset_size, length_size) {
|
||||||
|
Ok(Some(e)) if e.object_header_address != u64::MAX => {
|
||||||
|
return Ok(Some(LinkTarget::Hard {
|
||||||
|
object_header_address: e.object_header_address,
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
Err(e @ FormatError::Storage(_)) => return Err(e),
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let entries = resolve_group_entries(file_data, object_header, offset_size, length_size)?;
|
||||||
|
if let Some(e) = entries
|
||||||
|
.iter()
|
||||||
|
.find(|e| e.name == name && e.object_header_address != u64::MAX)
|
||||||
|
{
|
||||||
|
return Ok(Some(LinkTarget::Hard {
|
||||||
|
object_header_address: e.object_header_address,
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
return find_v1_symbolic_link(file_data, object_header, name, offset_size, length_size);
|
||||||
|
}
|
||||||
|
if !is_v2_group(object_header) {
|
||||||
|
return Err(FormatError::PathNotFound(String::from(
|
||||||
|
"object header is not a group",
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(
|
||||||
|
first_link_named(file_data, object_header, name, offset_size, length_size)?
|
||||||
|
.map(|link| link.link_target)
|
||||||
|
.filter(|t| {
|
||||||
|
!matches!(
|
||||||
|
t,
|
||||||
|
LinkTarget::Hard {
|
||||||
|
object_header_address: u64::MAX
|
||||||
|
}
|
||||||
|
)
|
||||||
|
}),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The object header address of the child called `name` of the group at
|
||||||
|
/// `group_address`: the address [`resolve_group_children`] lists under that
|
||||||
|
/// name, or `PathNotFound` if it lists none.
|
||||||
|
///
|
||||||
|
/// A dense group's child is found through its link name index (see
|
||||||
|
/// [`links_named`]) and only the named link is read and, if it is a soft
|
||||||
|
/// link, followed — not every link in the group. A v1 group is listed.
|
||||||
|
pub fn resolve_child(
|
||||||
|
file_data: &[u8],
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
name: &str,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
resolve_child_core(file_data, superblock, group_address, name)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`resolve_child`] over any [`Storage`]. One with the whole file in memory
|
||||||
|
/// is read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn resolve_child_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
name: &str,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => resolve_child(all, superblock, group_address, name),
|
||||||
|
None => resolve_child_core(file_data, superblock, group_address, name),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resolve_child_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
name: &str,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
let os = superblock.offset_size;
|
||||||
|
let ls = superblock.length_size;
|
||||||
|
let not_found = || FormatError::PathNotFound(String::from(name));
|
||||||
|
let header = ObjectHeader::parse_in(file_data, checked_addr(group_address)?, os, ls)?;
|
||||||
|
if !is_v2_group(&header) || is_v1_group(&header) {
|
||||||
|
return group_children(file_data, superblock, group_address, false)?
|
||||||
|
.into_iter()
|
||||||
|
.find(|e| e.name == name)
|
||||||
|
.map(|e| e.object_header_address)
|
||||||
|
.ok_or_else(not_found);
|
||||||
|
}
|
||||||
|
// The first link of that name only, as the listing (see
|
||||||
|
// `first_link_named`).
|
||||||
|
match first_link_named(file_data, &header, name, os, ls)?.map(|l| l.link_target) {
|
||||||
|
Some(LinkTarget::Hard {
|
||||||
|
object_header_address,
|
||||||
|
}) => Ok(object_header_address),
|
||||||
|
Some(LinkTarget::Soft { target_path }) => {
|
||||||
|
match resolve_path_from_in(file_data, superblock, group_address, &target_path) {
|
||||||
|
// Left out of the listing: dangling, cyclic, or in another file.
|
||||||
|
Err(
|
||||||
|
FormatError::PathNotFound(_)
|
||||||
|
| FormatError::NestingDepthExceeded
|
||||||
|
| FormatError::ExternalLinkUnsupported { .. },
|
||||||
|
) => Err(not_found()),
|
||||||
|
other => other,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Some(LinkTarget::External { .. }) | None => Err(not_found()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Find and parse the Link Info message from an object header.
|
/// Find and parse the Link Info message from an object header.
|
||||||
fn find_link_info(
|
fn find_link_info(
|
||||||
object_header: &ObjectHeader,
|
object_header: &ObjectHeader,
|
||||||
@@ -231,69 +549,245 @@ pub fn resolve_path_any(
|
|||||||
superblock: &Superblock,
|
superblock: &Superblock,
|
||||||
path: &str,
|
path: &str,
|
||||||
) -> Result<u64, FormatError> {
|
) -> Result<u64, FormatError> {
|
||||||
resolve_path_following_links(file_data, superblock, path, 0)
|
resolve_path_any_core(file_data, superblock, path)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`resolve_path_any`] over any [`Storage`]. One with the whole file in memory
|
||||||
|
/// is read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn resolve_path_any_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
superblock: &Superblock,
|
||||||
|
path: &str,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => resolve_path_any(all, superblock, path),
|
||||||
|
None => resolve_path_any_core(file_data, superblock, path),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resolve_path_any_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
superblock: &Superblock,
|
||||||
|
path: &str,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
resolve_path_following_links(
|
||||||
|
file_data,
|
||||||
|
superblock,
|
||||||
|
superblock.root_group_address,
|
||||||
|
path,
|
||||||
|
0,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve `path` relative to the group at `group_address` (an absolute path
|
||||||
|
/// starts at the root group instead), following soft links. This is how a
|
||||||
|
/// relative soft link's target is resolved: from the group holding the link.
|
||||||
|
pub fn resolve_path_from(
|
||||||
|
file_data: &[u8],
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
path: &str,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
resolve_path_from_in(file_data, superblock, group_address, path)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`resolve_path_from`] over any [`Storage`].
|
||||||
|
pub fn resolve_path_from_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
path: &str,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
let start = if path.starts_with('/') {
|
||||||
|
superblock.root_group_address
|
||||||
|
} else {
|
||||||
|
group_address
|
||||||
|
};
|
||||||
|
resolve_path_following_links(file_data, superblock, start, path, 0)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The children of the group at `group_address` that can be opened, as h5py
|
||||||
|
/// lists them: hard links, and soft links resolved to the object they point
|
||||||
|
/// at (under the soft link's own name). Links that cannot be followed are
|
||||||
|
/// left out rather than failing the listing — a dangling or cyclic soft link
|
||||||
|
/// (h5py lists its name but cannot open it), an external link (another
|
||||||
|
/// file), and a user-defined link. An object header that is not a group has
|
||||||
|
/// no children.
|
||||||
|
///
|
||||||
|
/// Any other error, such as a corrupt structure met while resolving a soft
|
||||||
|
/// link, is returned.
|
||||||
|
pub fn resolve_group_children(
|
||||||
|
file_data: &[u8],
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
|
resolve_group_children_core(file_data, superblock, group_address)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`resolve_group_children`] over any [`Storage`]. One with the whole file in memory
|
||||||
|
/// is read as the slice, by code compiled in this crate (see
|
||||||
|
/// [`crate::storage`], "Slice entry points").
|
||||||
|
#[inline]
|
||||||
|
pub fn resolve_group_children_in<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
|
match file_data.as_contiguous() {
|
||||||
|
Some(all) => resolve_group_children(all, superblock, group_address),
|
||||||
|
None => resolve_group_children_core(file_data, superblock, group_address),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn resolve_group_children_core<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
|
group_children(file_data, superblock, group_address, true)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`resolve_group_children`]; with `hint_headers`, every child's object
|
||||||
|
/// header is hinted (see [`Storage::hint`]) as soon as its address is
|
||||||
|
/// known, for a listing whose children are opened next.
|
||||||
|
fn group_children<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
hint_headers: bool,
|
||||||
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
|
let os = superblock.offset_size;
|
||||||
|
let ls = superblock.length_size;
|
||||||
|
let header = ObjectHeader::parse_in(file_data, checked_addr(group_address)?, os, ls)?;
|
||||||
|
|
||||||
|
let mut entries = Vec::new();
|
||||||
|
let mut soft = Vec::new();
|
||||||
|
if is_v1_group(&header) {
|
||||||
|
let sym_msg = header
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::SymbolTable)
|
||||||
|
.ok_or_else(|| FormatError::PathNotFound(String::from("no symbol table message")))?;
|
||||||
|
let stm = SymbolTableMessage::parse(&sym_msg.data, os)?;
|
||||||
|
let all = group_v1::v1_group_entries(file_data, &stm, os, ls, hint_headers)?;
|
||||||
|
if all.iter().any(|e| e.name.is_empty()) {
|
||||||
|
return Err(FormatError::InvalidLinkName);
|
||||||
|
}
|
||||||
|
if all.iter().any(group_v1::is_v1_soft_link) {
|
||||||
|
soft = group_v1::v1_soft_links_in(file_data, &stm, os, ls)?;
|
||||||
|
}
|
||||||
|
entries.extend(all.into_iter().filter(|e| !group_v1::is_v1_soft_link(e)));
|
||||||
|
} else if is_v2_group(&header) {
|
||||||
|
// Only the first link of each name counts (see `first_link_named`).
|
||||||
|
let mut seen = BTreeSet::new();
|
||||||
|
let mut visit = |link: LinkMessage| {
|
||||||
|
if !seen.insert(link.name.clone()) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
match link.link_target {
|
||||||
|
LinkTarget::Hard {
|
||||||
|
object_header_address,
|
||||||
|
} => entries.push(GroupEntry {
|
||||||
|
name: link.name,
|
||||||
|
object_header_address,
|
||||||
|
cache_type: 0,
|
||||||
|
}),
|
||||||
|
LinkTarget::Soft { target_path } => soft.push((link.name, target_path)),
|
||||||
|
LinkTarget::External { .. } => {}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let link_info = find_link_info(&header, os)?;
|
||||||
|
if let Some(fh_addr) = link_info.fractal_heap_address {
|
||||||
|
for_each_dense_link(file_data, &link_info, fh_addr, os, ls, hint_headers, visit)?;
|
||||||
|
} else {
|
||||||
|
for msg in &header.messages {
|
||||||
|
if msg.msg_type == MessageType::Link
|
||||||
|
&& let Some(link) = parse_link(&msg.data, os)?
|
||||||
|
{
|
||||||
|
visit(link);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for (name, target) in soft {
|
||||||
|
match resolve_path_from_in(file_data, superblock, group_address, &target) {
|
||||||
|
Ok(object_header_address) => entries.push(GroupEntry {
|
||||||
|
name,
|
||||||
|
object_header_address,
|
||||||
|
cache_type: 0,
|
||||||
|
}),
|
||||||
|
// Dangling, cyclic, or ending in another file: not openable here.
|
||||||
|
Err(
|
||||||
|
FormatError::PathNotFound(_)
|
||||||
|
| FormatError::NestingDepthExceeded
|
||||||
|
| FormatError::ExternalLinkUnsupported { .. },
|
||||||
|
) => {}
|
||||||
|
Err(e) => return Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(entries)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Soft links followed while resolving one path. Guards against link cycles
|
/// Soft links followed while resolving one path. Guards against link cycles
|
||||||
/// (`a -> b -> a`), which are legal to create.
|
/// (`a -> b -> a`), which are legal to create.
|
||||||
const MAX_SOFT_LINK_DEPTH: u8 = 16;
|
const MAX_SOFT_LINK_DEPTH: u8 = 16;
|
||||||
|
|
||||||
fn resolve_path_following_links(
|
/// Walk `path` from the group at `start`, following soft links.
|
||||||
file_data: &[u8],
|
fn resolve_path_following_links<S: Storage + ?Sized>(
|
||||||
|
file_data: &S,
|
||||||
superblock: &Superblock,
|
superblock: &Superblock,
|
||||||
|
start: u64,
|
||||||
path: &str,
|
path: &str,
|
||||||
depth: u8,
|
depth: u8,
|
||||||
) -> Result<u64, FormatError> {
|
) -> Result<u64, FormatError> {
|
||||||
let components: Vec<&str> = path.split('/').filter(|s| !s.is_empty()).collect();
|
let components: Vec<&str> = path
|
||||||
|
.split('/')
|
||||||
|
.filter(|s| !s.is_empty() && *s != ".")
|
||||||
|
.collect();
|
||||||
if components.is_empty() {
|
if components.is_empty() {
|
||||||
return Ok(superblock.root_group_address);
|
return Ok(start);
|
||||||
}
|
}
|
||||||
|
|
||||||
let os = superblock.offset_size;
|
let os = superblock.offset_size;
|
||||||
let ls = superblock.length_size;
|
let ls = superblock.length_size;
|
||||||
|
|
||||||
let root_header =
|
let mut current_addr = start;
|
||||||
ObjectHeader::parse(file_data, superblock.root_group_address as usize, os, ls)?;
|
let mut current_header = ObjectHeader::parse_in(file_data, checked_addr(start)?, os, ls)?;
|
||||||
|
|
||||||
let mut current_addr = superblock.root_group_address;
|
|
||||||
let mut current_header = root_header;
|
|
||||||
|
|
||||||
for (i, component) in components.iter().enumerate() {
|
for (i, component) in components.iter().enumerate() {
|
||||||
let entries = resolve_group_entries(file_data, ¤t_header, os, ls)?;
|
match lookup_link(file_data, ¤t_header, component, os, ls)? {
|
||||||
|
Some(LinkTarget::Hard {
|
||||||
let found = entries
|
object_header_address,
|
||||||
.iter()
|
}) => {
|
||||||
.find(|e| e.name == *component && e.object_header_address != u64::MAX);
|
|
||||||
match found {
|
|
||||||
Some(entry) => {
|
|
||||||
if i == components.len() - 1 {
|
if i == components.len() - 1 {
|
||||||
return Ok(entry.object_header_address);
|
return Ok(object_header_address);
|
||||||
}
|
}
|
||||||
current_addr = entry.object_header_address;
|
current_addr = object_header_address;
|
||||||
current_header = ObjectHeader::parse(file_data, current_addr as usize, os, ls)?;
|
current_header =
|
||||||
|
ObjectHeader::parse_in(file_data, checked_addr(current_addr)?, os, ls)?;
|
||||||
}
|
}
|
||||||
None => {
|
found => {
|
||||||
return match find_symbolic_link(file_data, ¤t_header, component, os, ls)? {
|
return match found {
|
||||||
Some(LinkTarget::Soft { target_path }) => {
|
Some(LinkTarget::Soft { target_path }) => {
|
||||||
if depth >= MAX_SOFT_LINK_DEPTH {
|
if depth >= MAX_SOFT_LINK_DEPTH {
|
||||||
return Err(FormatError::NestingDepthExceeded);
|
return Err(FormatError::NestingDepthExceeded);
|
||||||
}
|
}
|
||||||
// A relative target is relative to the group holding
|
// A relative target is relative to the group holding
|
||||||
// the link; then the rest of the original path.
|
// the link; then the rest of the original path.
|
||||||
let mut full = String::new();
|
let from = if target_path.starts_with('/') {
|
||||||
if !target_path.starts_with('/') {
|
superblock.root_group_address
|
||||||
for parent in &components[..i] {
|
} else {
|
||||||
full.push('/');
|
current_addr
|
||||||
full.push_str(parent);
|
};
|
||||||
}
|
let mut full = target_path;
|
||||||
}
|
|
||||||
full.push('/');
|
|
||||||
full.push_str(&target_path);
|
|
||||||
for rest in &components[i + 1..] {
|
for rest in &components[i + 1..] {
|
||||||
full.push('/');
|
full.push('/');
|
||||||
full.push_str(rest);
|
full.push_str(rest);
|
||||||
}
|
}
|
||||||
resolve_path_following_links(file_data, superblock, &full, depth + 1)
|
resolve_path_following_links(file_data, superblock, from, &full, depth + 1)
|
||||||
}
|
}
|
||||||
Some(LinkTarget::External {
|
Some(LinkTarget::External {
|
||||||
filename,
|
filename,
|
||||||
@@ -312,8 +806,8 @@ fn resolve_path_following_links(
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve group entries from an object header, auto-detecting v1 vs v2.
|
/// Resolve group entries from an object header, auto-detecting v1 vs v2.
|
||||||
fn resolve_group_entries(
|
fn resolve_group_entries<S: Storage + ?Sized>(
|
||||||
file_data: &[u8],
|
file_data: &S,
|
||||||
object_header: &ObjectHeader,
|
object_header: &ObjectHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
@@ -326,9 +820,11 @@ fn resolve_group_entries(
|
|||||||
.find(|m| m.msg_type == MessageType::SymbolTable)
|
.find(|m| m.msg_type == MessageType::SymbolTable)
|
||||||
.ok_or_else(|| FormatError::PathNotFound(String::from("no symbol table message")))?;
|
.ok_or_else(|| FormatError::PathNotFound(String::from("no symbol table message")))?;
|
||||||
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
||||||
group_v1::resolve_v1_group_entries(file_data, &stm, offset_size, length_size)
|
// A lookup: an entry with an empty name (which fails a listing) is
|
||||||
|
// skipped by the name comparison, as in libhdf5.
|
||||||
|
group_v1::v1_group_entries(file_data, &stm, offset_size, length_size, false)
|
||||||
} else if is_v2_group(object_header) {
|
} else if is_v2_group(object_header) {
|
||||||
resolve_v2_group_entries(file_data, object_header, offset_size, length_size)
|
resolve_v2_group_entries_in(file_data, object_header, offset_size, length_size)
|
||||||
} else {
|
} else {
|
||||||
Err(FormatError::PathNotFound(String::from(
|
Err(FormatError::PathNotFound(String::from(
|
||||||
"object header is not a group",
|
"object header is not a group",
|
||||||
@@ -509,6 +1005,86 @@ mod tests {
|
|||||||
assert_eq!(values, vec![22.5, 23.1, 21.8]);
|
assert_eq!(values, vec![22.5, 23.1, 21.8]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `v1_groups_400.h5` from its superblock on (it has a user block).
|
||||||
|
fn v1_groups_400() -> (Vec<u8>, Superblock) {
|
||||||
|
let all: &[u8] = include_bytes!("../tests/fixtures/v1_groups_400.h5");
|
||||||
|
let data = all[signature::find_signature(all).unwrap()..].to_vec();
|
||||||
|
let sb = Superblock::parse(&data, 0).unwrap();
|
||||||
|
(data, sb)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every child of a v1 group resolves by name, down the group's B-tree,
|
||||||
|
/// to the address the listing gives, reading a small part of what the
|
||||||
|
/// listing reads; a name the group does not hold is not found.
|
||||||
|
#[test]
|
||||||
|
fn v1_lookup_down_the_btree_agrees_with_the_listing() {
|
||||||
|
let (data, sb) = v1_groups_400();
|
||||||
|
let children = resolve_group_children(&data, &sb, sb.root_group_address).unwrap();
|
||||||
|
assert_eq!(children.len(), 401);
|
||||||
|
for c in &children {
|
||||||
|
let path = format!("/{}", c.name);
|
||||||
|
assert_eq!(
|
||||||
|
resolve_path_any(&data, &sb, &path).unwrap(),
|
||||||
|
c.object_header_address,
|
||||||
|
"{path}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
for missing in ["/g0400", "/a", "/g", "/g00000", "/zz", "/x0"] {
|
||||||
|
assert!(
|
||||||
|
matches!(
|
||||||
|
resolve_path_any(&data, &sb, missing),
|
||||||
|
Err(FormatError::PathNotFound(_))
|
||||||
|
),
|
||||||
|
"{missing}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let st = crate::storage::CountingStorage::new(data.clone());
|
||||||
|
resolve_group_children_in(&st, &sb, sb.root_group_address).unwrap();
|
||||||
|
let listing = st.bytes_read();
|
||||||
|
st.reset();
|
||||||
|
let last = children.last().unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
resolve_path_any_in(&st, &sb, &format!("/{}", last.name)).unwrap(),
|
||||||
|
last.object_header_address
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
st.bytes_read() * 8 < listing,
|
||||||
|
"lookup read {} bytes, listing {listing}",
|
||||||
|
st.bytes_read()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A v1 group whose B-tree is out of name order (a name changed in the
|
||||||
|
/// heap so that it sorts past every key) is still looked up by reading
|
||||||
|
/// every entry, as before the lookup went down the B-tree.
|
||||||
|
#[test]
|
||||||
|
fn v1_lookup_falls_back_when_the_btree_is_out_of_order() {
|
||||||
|
let (mut data, sb) = v1_groups_400();
|
||||||
|
let at: Vec<usize> = data
|
||||||
|
.windows(6)
|
||||||
|
.enumerate()
|
||||||
|
.filter(|(_, w)| *w == b"g0200\0")
|
||||||
|
.map(|(i, _)| i)
|
||||||
|
.collect();
|
||||||
|
assert_eq!(at.len(), 1, "one heap string");
|
||||||
|
data[at[0]] = b'~';
|
||||||
|
let children = resolve_group_children(&data, &sb, sb.root_group_address).unwrap();
|
||||||
|
let moved = children.iter().find(|c| c.name == "~0200").unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
resolve_path_any(&data, &sb, "/~0200").unwrap(),
|
||||||
|
moved.object_header_address
|
||||||
|
);
|
||||||
|
assert!(resolve_path_any(&data, &sb, "/g0200").is_err());
|
||||||
|
for c in &children {
|
||||||
|
let path = format!("/{}", c.name);
|
||||||
|
assert_eq!(
|
||||||
|
resolve_path_any(&data, &sb, &path).unwrap(),
|
||||||
|
c.object_header_address,
|
||||||
|
"{path}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn path_not_found_v2() {
|
fn path_not_found_v2() {
|
||||||
let file_data: &[u8] = include_bytes!("../tests/fixtures/v2_groups.h5");
|
let file_data: &[u8] = include_bytes!("../tests/fixtures/v2_groups.h5");
|
||||||
|
|||||||
@@ -112,7 +112,8 @@ pub fn partition(
|
|||||||
|
|
||||||
for idx in 0..num_items {
|
for idx in 0..num_items {
|
||||||
let h = fxhash_combine(seed, idx as u64);
|
let h = fxhash_combine(seed, idx as u64);
|
||||||
let lane = (h % num_lanes as u64) as usize;
|
// Below `num_lanes`, so it fits.
|
||||||
|
let lane = crate::addr::saturating_usize(h % num_lanes as u64);
|
||||||
lanes[lane].push(idx);
|
lanes[lane].push(idx);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -26,12 +26,13 @@
|
|||||||
//! use clawhdf5_format::{signature, superblock, object_header, group_v2,
|
//! use clawhdf5_format::{signature, superblock, object_header, group_v2,
|
||||||
//! datatype, dataspace, data_layout, data_read, message_type::MessageType};
|
//! datatype, dataspace, data_layout, data_read, message_type::MessageType};
|
||||||
//!
|
//!
|
||||||
//! let file_data = std::fs::read("output.h5").unwrap();
|
//! let bytes = std::fs::read("output.h5").unwrap();
|
||||||
//! let sig = signature::find_signature(&file_data).unwrap();
|
//! // Addresses are relative to the superblock: skip any user block.
|
||||||
//! let sb = superblock::Superblock::parse(&file_data, sig).unwrap();
|
//! let (_user_block, file_data) = signature::split_user_block(&bytes).unwrap();
|
||||||
//! let addr = group_v2::resolve_path_any(&file_data, &sb, "data").unwrap();
|
//! let sb = superblock::Superblock::parse(file_data, 0).unwrap();
|
||||||
|
//! let addr = group_v2::resolve_path_any(file_data, &sb, "data").unwrap();
|
||||||
//! let hdr = object_header::ObjectHeader::parse(
|
//! let hdr = object_header::ObjectHeader::parse(
|
||||||
//! &file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
//! file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||||
//! ```
|
//! ```
|
||||||
//!
|
//!
|
||||||
//! # Features
|
//! # Features
|
||||||
@@ -42,18 +43,30 @@
|
|||||||
//! | `checksum` | yes | Jenkins lookup3 checksum validation |
|
//! | `checksum` | yes | Jenkins lookup3 checksum validation |
|
||||||
//! | `deflate` | yes | Deflate (gzip) compression via `flate2` |
|
//! | `deflate` | yes | Deflate (gzip) compression via `flate2` |
|
||||||
//! | `provenance` | yes | SHINES provenance — SHA-256 hashing & verification |
|
//! | `provenance` | yes | SHINES provenance — SHA-256 hashing & verification |
|
||||||
|
//! | `lzf` | yes | LZF filter (32000), h5py's `compression="lzf"` |
|
||||||
|
//! | `bitshuffle` | no | Bitshuffle filter (32008), none/LZ4/Zstandard |
|
||||||
|
//! | `bzip2` | no | bzip2 filter (307) |
|
||||||
|
//! | `blosc` | no | Blosc 1 filter (32001) |
|
||||||
|
//! | `plugin-filters` | no | The four above |
|
||||||
|
//!
|
||||||
|
//! Filters are looked up by ID in [`filter_registry`], which also takes
|
||||||
|
//! codecs registered at run time for other IDs.
|
||||||
|
|
||||||
#![cfg_attr(not(feature = "std"), no_std)]
|
#![cfg_attr(not(feature = "std"), no_std)]
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
extern crate alloc;
|
extern crate alloc;
|
||||||
|
|
||||||
|
pub mod addr;
|
||||||
pub mod attribute;
|
pub mod attribute;
|
||||||
pub mod attribute_info;
|
pub mod attribute_info;
|
||||||
pub mod btree_v1;
|
pub mod btree_v1;
|
||||||
pub mod btree_v2;
|
pub mod btree_v2;
|
||||||
|
mod btree_v2_write;
|
||||||
|
mod bulk_alloc;
|
||||||
pub mod checksum;
|
pub mod checksum;
|
||||||
pub mod chunk_cache;
|
pub mod chunk_cache;
|
||||||
|
mod chunk_grid;
|
||||||
pub mod chunk_index;
|
pub mod chunk_index;
|
||||||
pub mod chunked_read;
|
pub mod chunked_read;
|
||||||
pub mod chunked_write;
|
pub mod chunked_write;
|
||||||
@@ -69,11 +82,25 @@ pub mod extensible_array;
|
|||||||
pub mod file_writer;
|
pub mod file_writer;
|
||||||
pub mod fill_value;
|
pub mod fill_value;
|
||||||
pub mod filter_pipeline;
|
pub mod filter_pipeline;
|
||||||
|
pub mod filter_registry;
|
||||||
pub mod filters;
|
pub mod filters;
|
||||||
|
#[cfg(any(feature = "bitshuffle", feature = "blosc"))]
|
||||||
|
mod filters_bitshuffle;
|
||||||
|
#[cfg(feature = "blosc")]
|
||||||
|
pub mod filters_blosc;
|
||||||
|
#[cfg(feature = "blosc2")]
|
||||||
|
pub mod filters_blosc2;
|
||||||
|
#[cfg(feature = "bzip2")]
|
||||||
|
mod filters_bzip2;
|
||||||
|
#[cfg(feature = "lzf")]
|
||||||
|
pub mod filters_lzf;
|
||||||
mod filters_szip;
|
mod filters_szip;
|
||||||
|
#[cfg(feature = "zfp")]
|
||||||
|
pub mod filters_zfp;
|
||||||
pub mod fixed_array;
|
pub mod fixed_array;
|
||||||
pub mod float16;
|
pub mod float16;
|
||||||
pub mod fractal_heap;
|
pub mod fractal_heap;
|
||||||
|
mod gather;
|
||||||
pub mod global_heap;
|
pub mod global_heap;
|
||||||
pub mod group_info;
|
pub mod group_info;
|
||||||
pub mod group_v1;
|
pub mod group_v1;
|
||||||
@@ -83,6 +110,7 @@ pub mod lane_partition;
|
|||||||
pub mod link_info;
|
pub mod link_info;
|
||||||
pub mod link_message;
|
pub mod link_message;
|
||||||
pub mod local_heap;
|
pub mod local_heap;
|
||||||
|
pub mod lookup_stats;
|
||||||
pub mod message_type;
|
pub mod message_type;
|
||||||
pub mod metadata_cache;
|
pub mod metadata_cache;
|
||||||
pub mod metadata_index;
|
pub mod metadata_index;
|
||||||
@@ -96,10 +124,24 @@ pub mod property_list;
|
|||||||
pub mod selection;
|
pub mod selection;
|
||||||
pub mod shared_message;
|
pub mod shared_message;
|
||||||
pub mod signature;
|
pub mod signature;
|
||||||
|
pub mod storage;
|
||||||
pub mod superblock;
|
pub mod superblock;
|
||||||
|
pub mod superblock_ext;
|
||||||
pub mod symbol_table;
|
pub mod symbol_table;
|
||||||
|
#[cfg(all(
|
||||||
|
test,
|
||||||
|
any(
|
||||||
|
feature = "lzf",
|
||||||
|
feature = "bitshuffle",
|
||||||
|
feature = "bzip2",
|
||||||
|
feature = "blosc"
|
||||||
|
)
|
||||||
|
))]
|
||||||
|
mod test_fuzz;
|
||||||
pub mod type_builders;
|
pub mod type_builders;
|
||||||
|
pub mod vds;
|
||||||
pub mod vl_data;
|
pub mod vl_data;
|
||||||
|
mod writer_tree;
|
||||||
|
|
||||||
#[cfg(feature = "provenance")]
|
#[cfg(feature = "provenance")]
|
||||||
pub mod provenance;
|
pub mod provenance;
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{string::String, vec::Vec};
|
use alloc::{string::String, vec::Vec};
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::datatype::CharacterSet;
|
use crate::datatype::CharacterSet;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
@@ -247,7 +248,7 @@ impl LinkMessage {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Link name length
|
// Link name length
|
||||||
let name_len = read_offset(data, pos, name_size_field_width)? as usize;
|
let name_len = to_usize(read_offset(data, pos, name_size_field_width)?)?;
|
||||||
pos += name_size_field_width as usize;
|
pos += name_size_field_width as usize;
|
||||||
|
|
||||||
// Link name
|
// Link name
|
||||||
|
|||||||
@@ -3,7 +3,9 @@
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::string::String;
|
use alloc::string::String;
|
||||||
|
|
||||||
|
use crate::addr::to_usize;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::storage::{Storage, len_usize, read_exact_at};
|
||||||
|
|
||||||
/// Parsed HDF5 Local Heap header.
|
/// Parsed HDF5 Local Heap header.
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -16,21 +18,6 @@ pub struct LocalHeap {
|
|||||||
pub data_segment_address: u64,
|
pub data_segment_address: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Checks that `[offset, offset + needed)` fits within `data`, guarding the
|
|
||||||
/// addition against `usize` overflow from a crafted near-`usize::MAX` offset.
|
|
||||||
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
|
||||||
if offset
|
|
||||||
.checked_add(needed)
|
|
||||||
.is_none_or(|end| end > data.len())
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: offset.saturating_add(needed),
|
|
||||||
available: data.len(),
|
|
||||||
});
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
||||||
let s = size as usize;
|
let s = size as usize;
|
||||||
if pos.checked_add(s).is_none_or(|end| end > data.len()) {
|
if pos.checked_add(s).is_none_or(|end| end > data.len()) {
|
||||||
@@ -50,6 +37,10 @@ fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// First read of a name on a backend without the file in memory: most link
|
||||||
|
/// names are shorter than this.
|
||||||
|
const NAME_READ_START: usize = 64;
|
||||||
|
|
||||||
impl LocalHeap {
|
impl LocalHeap {
|
||||||
/// Parse a local heap header at the given offset in the file data.
|
/// Parse a local heap header at the given offset in the file data.
|
||||||
pub fn parse(
|
pub fn parse(
|
||||||
@@ -57,12 +48,24 @@ impl LocalHeap {
|
|||||||
offset: usize,
|
offset: usize,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<LocalHeap, FormatError> {
|
||||||
|
Self::parse_in(file_data, offset as u64, offset_size, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::parse`] over any [`Storage`]: one read of the header.
|
||||||
|
pub fn parse_in<S: Storage + ?Sized>(
|
||||||
|
file: &S,
|
||||||
|
offset: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
) -> Result<LocalHeap, FormatError> {
|
) -> Result<LocalHeap, FormatError> {
|
||||||
// signature(4) + version(1) + reserved(3) = 8, then length_size*2 + offset_size
|
// signature(4) + version(1) + reserved(3) = 8, then length_size*2 + offset_size
|
||||||
let ls = length_size as usize;
|
let ls = length_size as usize;
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let total = 8 + ls * 2 + os;
|
let total = 8 + ls * 2 + os;
|
||||||
ensure_len(file_data, offset, total)?;
|
let header = read_exact_at(file, offset, total)?;
|
||||||
|
let file_data: &[u8] = &header;
|
||||||
|
let offset = 0usize;
|
||||||
|
|
||||||
if &file_data[offset..offset + 4] != b"HEAP" {
|
if &file_data[offset..offset + 4] != b"HEAP" {
|
||||||
return Err(FormatError::InvalidLocalHeapSignature);
|
return Err(FormatError::InvalidLocalHeapSignature);
|
||||||
@@ -87,45 +90,129 @@ impl LocalHeap {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Walk the free list the way libhdf5 does when it loads a heap's data
|
||||||
|
/// (`H5HL__fl_deserialize`), rejecting a heap whose free list points
|
||||||
|
/// outside the data segment. libhdf5 refuses such a heap ("bad heap free
|
||||||
|
/// list"), and names read from it would be garbage.
|
||||||
|
///
|
||||||
|
/// libhdf5 only loads a heap when it needs a name from it (an empty
|
||||||
|
/// group's broken heap goes unnoticed), so call this before the first
|
||||||
|
/// [`Self::read_string`], not on parse.
|
||||||
|
///
|
||||||
|
/// The end of the list is `H5HL_FREE_NULL` (1); an all-ones value (the
|
||||||
|
/// undefined address) is accepted as "no free list" too.
|
||||||
|
pub fn validate_free_list(&self, file_data: &[u8], length_size: u8) -> Result<(), FormatError> {
|
||||||
|
self.validate_free_list_in(file_data, length_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::validate_free_list`] over any [`Storage`]: two small reads
|
||||||
|
/// per free block.
|
||||||
|
pub fn validate_free_list_in<S: Storage + ?Sized>(
|
||||||
|
&self,
|
||||||
|
file: &S,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
|
const FREE_NULL: u64 = 1;
|
||||||
|
let ls = length_size as usize;
|
||||||
|
let undefined = if ls >= 8 {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
(1u64 << (8 * ls)) - 1
|
||||||
|
};
|
||||||
|
let size = self.data_segment_size;
|
||||||
|
let seg = self.data_segment_address;
|
||||||
|
let mut next = self.free_list_head_offset;
|
||||||
|
// Each free block holds two lengths, so a list longer than this
|
||||||
|
// revisits a block: a cycle.
|
||||||
|
let max_blocks = size / (2 * ls as u64) + 1;
|
||||||
|
let mut walked = 0u64;
|
||||||
|
while next != FREE_NULL && next != undefined {
|
||||||
|
if next >= size || walked >= max_blocks {
|
||||||
|
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||||
|
}
|
||||||
|
walked += 1;
|
||||||
|
let at = seg
|
||||||
|
.checked_add(next)
|
||||||
|
.and_then(|a| usize::try_from(a).ok())
|
||||||
|
.ok_or(FormatError::InvalidLocalHeapFreeList)?;
|
||||||
|
let block_offset = next;
|
||||||
|
next = read_offset(&read_exact_at(file, at as u64, ls)?, 0, length_size)?;
|
||||||
|
if next == 0 {
|
||||||
|
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||||
|
}
|
||||||
|
let block_size =
|
||||||
|
read_offset(&read_exact_at(file, (at + ls) as u64, ls)?, 0, length_size)?;
|
||||||
|
if block_offset
|
||||||
|
.checked_add(block_size)
|
||||||
|
.is_none_or(|end| end > size)
|
||||||
|
{
|
||||||
|
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Read a null-terminated string from the heap's data segment at the given byte offset.
|
/// Read a null-terminated string from the heap's data segment at the given byte offset.
|
||||||
pub fn read_string(&self, file_data: &[u8], string_offset: u64) -> Result<String, FormatError> {
|
pub fn read_string(&self, file_data: &[u8], string_offset: u64) -> Result<String, FormatError> {
|
||||||
let seg_addr = self.data_segment_address as usize;
|
self.read_string_in(file_data, string_offset)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::read_string`] over any [`Storage`]: one read of up to 64
|
||||||
|
/// bytes for a short name, more (each four times the last) up to the end
|
||||||
|
/// of the data segment for a longer one.
|
||||||
|
pub fn read_string_in<S: Storage + ?Sized>(
|
||||||
|
&self,
|
||||||
|
file: &S,
|
||||||
|
string_offset: u64,
|
||||||
|
) -> Result<String, FormatError> {
|
||||||
|
let file_len = len_usize(file);
|
||||||
|
let seg_addr = to_usize(self.data_segment_address)?;
|
||||||
let str_start =
|
let str_start =
|
||||||
seg_addr
|
seg_addr
|
||||||
.checked_add(string_offset as usize)
|
.checked_add(to_usize(string_offset)?)
|
||||||
.ok_or(FormatError::Overflow(
|
.ok_or(FormatError::Overflow(
|
||||||
"local heap seg_addr + string_offset overflow".into(),
|
"local heap seg_addr + string_offset overflow".into(),
|
||||||
))?;
|
))?;
|
||||||
let seg_end = seg_addr
|
let seg_end = seg_addr
|
||||||
.checked_add(self.data_segment_size as usize)
|
.checked_add(to_usize(self.data_segment_size)?)
|
||||||
.ok_or(FormatError::Overflow(
|
.ok_or(FormatError::Overflow(
|
||||||
"local heap seg_addr + data_segment_size overflow".into(),
|
"local heap seg_addr + data_segment_size overflow".into(),
|
||||||
))?;
|
))?;
|
||||||
|
|
||||||
if str_start >= file_data.len() || str_start >= seg_end {
|
if str_start >= file_len || str_start >= seg_end {
|
||||||
return Err(FormatError::UnexpectedEof {
|
return Err(FormatError::UnexpectedEof {
|
||||||
expected: str_start + 1,
|
expected: str_start + 1,
|
||||||
available: file_data.len(),
|
available: file_len,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
// Find null terminator
|
// Find the null terminator, which lies before the end of the data
|
||||||
let search_end = seg_end.min(file_data.len());
|
// segment (or of the file). In memory that is one borrowed slice;
|
||||||
let mut end = str_start;
|
// otherwise the bytes are read in growing pieces, so a name costs a
|
||||||
while end < search_end && file_data[end] != 0 {
|
// read of about its own length, not of the rest of the segment
|
||||||
end += 1;
|
// (whose size is an untrusted header field).
|
||||||
|
let search_end = seg_end.min(file_len);
|
||||||
|
let total = search_end - str_start;
|
||||||
|
let mut want = if file.as_contiguous().is_some() {
|
||||||
|
total
|
||||||
|
} else {
|
||||||
|
total.min(NAME_READ_START)
|
||||||
|
};
|
||||||
|
loop {
|
||||||
|
let rest = read_exact_at(file, str_start as u64, want)?;
|
||||||
|
if let Some(len) = rest.iter().position(|&b| b == 0) {
|
||||||
|
let s = core::str::from_utf8(&rest[..len])
|
||||||
|
.map_err(|_| FormatError::InvalidLocalHeapSignature)?;
|
||||||
|
return Ok(String::from(s));
|
||||||
|
}
|
||||||
|
if want == total {
|
||||||
|
return Err(FormatError::UnexpectedEof {
|
||||||
|
expected: search_end + 1,
|
||||||
|
available: search_end,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
want = want.saturating_mul(4).min(total);
|
||||||
}
|
}
|
||||||
|
|
||||||
if end >= search_end {
|
|
||||||
return Err(FormatError::UnexpectedEof {
|
|
||||||
expected: end + 1,
|
|
||||||
available: search_end,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
let s = core::str::from_utf8(&file_data[str_start..end])
|
|
||||||
.map_err(|_| FormatError::InvalidLocalHeapSignature)?;
|
|
||||||
Ok(String::from(s))
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -162,8 +249,8 @@ mod tests {
|
|||||||
// data_segment_size
|
// data_segment_size
|
||||||
write_val(&mut file, pos, data_seg_size as u64, length_size);
|
write_val(&mut file, pos, data_seg_size as u64, length_size);
|
||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
// free_list_head_offset
|
// free_list_head_offset: H5HL_FREE_NULL (no free space)
|
||||||
write_val(&mut file, pos, 0xFFFFFFFF, length_size);
|
write_val(&mut file, pos, 1, length_size);
|
||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
// data_segment_address
|
// data_segment_address
|
||||||
write_val(&mut file, pos, data_seg_offset as u64, offset_size);
|
write_val(&mut file, pos, data_seg_offset as u64, offset_size);
|
||||||
@@ -243,6 +330,50 @@ mod tests {
|
|||||||
assert_eq!(s, "test");
|
assert_eq!(s, "test");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Heap with data segment `[a, b, c, 0-padding]` whose free list starts
|
||||||
|
/// at `head` and has one block `(next, size)` at offset 8.
|
||||||
|
fn heap_with_free_block(head: u64, next: u64, size: u64) -> Vec<u8> {
|
||||||
|
let mut file = build_heap_file(0, 100, &["abcdefg"], 8, 8);
|
||||||
|
file.resize(200, 0);
|
||||||
|
write_val(&mut file, 8, 32, 8); // data segment size
|
||||||
|
write_val(&mut file, 16, head, 8);
|
||||||
|
write_val(&mut file, 108, next, 8);
|
||||||
|
write_val(&mut file, 116, size, 8);
|
||||||
|
file
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn free_list_inside_the_segment_is_accepted() {
|
||||||
|
let file = heap_with_free_block(8, 1, 24);
|
||||||
|
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||||
|
heap.validate_free_list(&file, 8).unwrap();
|
||||||
|
assert_eq!(heap.read_string(&file, 0).unwrap(), "abcdefg");
|
||||||
|
// An all-ones head is "no free list" too.
|
||||||
|
let file = heap_with_free_block(u64::MAX, 0, 0);
|
||||||
|
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||||
|
assert!(heap.validate_free_list(&file, 8).is_ok());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bad_free_list_is_rejected_like_libhdf5() {
|
||||||
|
for (head, next, size, why) in [
|
||||||
|
(40, 1, 8, "head past the segment"),
|
||||||
|
(8, 1, 25, "block runs past the segment"),
|
||||||
|
(8, 0, 8, "next offset of zero"),
|
||||||
|
(8, 8, 8, "cycle"),
|
||||||
|
(8, 999, 8, "next past the segment"),
|
||||||
|
] {
|
||||||
|
let file = heap_with_free_block(head, next, size);
|
||||||
|
// The header itself parses; the free list is checked on use.
|
||||||
|
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
heap.validate_free_list(&file, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidLocalHeapFreeList,
|
||||||
|
"{why}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn invalid_version() {
|
fn invalid_version() {
|
||||||
let mut file = build_heap_file(0, 100, &["x"], 8, 8);
|
let mut file = build_heap_file(0, 100, &["x"], 8, 8);
|
||||||
@@ -250,4 +381,77 @@ mod tests {
|
|||||||
let err = LocalHeap::parse(&file, 0, 8, 8).unwrap_err();
|
let err = LocalHeap::parse(&file, 0, 8, 8).unwrap_err();
|
||||||
assert_eq!(err, FormatError::InvalidLocalHeapVersion(1));
|
assert_eq!(err, FormatError::InvalidLocalHeapVersion(1));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Header, free list and strings read identically through a
|
||||||
|
/// `read_at`-only storage, for every truncation of the file.
|
||||||
|
#[test]
|
||||||
|
fn storage_reads_match_slice_reads() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let plain = build_heap_file(0, 64, &["", "alpha", "beta"], 8, 8);
|
||||||
|
// A free block of 16 bytes at segment offset 12, ending the list.
|
||||||
|
let mut free = build_heap_file(0, 64, &["", "alpha", "beta", &"x".repeat(20)], 8, 8);
|
||||||
|
free[16..24].copy_from_slice(&12u64.to_le_bytes());
|
||||||
|
free[64 + 12..64 + 20].copy_from_slice(&1u64.to_le_bytes());
|
||||||
|
free[64 + 20..64 + 28].copy_from_slice(&16u64.to_le_bytes());
|
||||||
|
let mut bad_free = free.clone();
|
||||||
|
bad_free[64 + 20..64 + 28].copy_from_slice(&99u64.to_le_bytes());
|
||||||
|
for full in [plain, free, bad_free] {
|
||||||
|
for cut in 0..=full.len() {
|
||||||
|
let f = &full[..cut];
|
||||||
|
let storage = CountingStorage::new(f.to_vec());
|
||||||
|
let want = LocalHeap::parse(f, 0, 8, 8);
|
||||||
|
let got = LocalHeap::parse_in(&storage, 0, 8, 8);
|
||||||
|
assert_eq!(format!("{got:?}"), format!("{want:?}"));
|
||||||
|
let Ok(heap) = want else { continue };
|
||||||
|
assert_eq!(
|
||||||
|
heap.validate_free_list_in(&storage, 8),
|
||||||
|
heap.validate_free_list(f, 8)
|
||||||
|
);
|
||||||
|
for off in [0u64, 1, 2, 6, 7, 11, 100] {
|
||||||
|
assert_eq!(heap.read_string_in(&storage, off), heap.read_string(f, off));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Names of every length around the first read's size, and one with no
|
||||||
|
/// terminator, read identically through a `read_at`-only storage; a
|
||||||
|
/// short name in a heap whose header claims a huge data segment costs
|
||||||
|
/// one small read, not a read of the rest of the file.
|
||||||
|
#[test]
|
||||||
|
fn long_names_and_hostile_segment_sizes() {
|
||||||
|
use crate::storage::CountingStorage;
|
||||||
|
let names: Vec<String> = [0usize, 1, 63, 64, 65, 255, 256, 257, 1000, 5000]
|
||||||
|
.iter()
|
||||||
|
.map(|&n| "n".repeat(n))
|
||||||
|
.collect();
|
||||||
|
let refs: Vec<&str> = names.iter().map(String::as_str).collect();
|
||||||
|
let mut file = build_heap_file(0, 64, &refs, 8, 8);
|
||||||
|
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||||
|
let storage = CountingStorage::new(file.clone());
|
||||||
|
let mut off = 0u64;
|
||||||
|
for name in &names {
|
||||||
|
let got = heap.read_string_in(&storage, off);
|
||||||
|
assert_eq!(got, heap.read_string(&file, off));
|
||||||
|
assert_eq!(got.unwrap(), *name);
|
||||||
|
off += name.len() as u64 + 1;
|
||||||
|
}
|
||||||
|
// The last name loses its terminator: both report the same error.
|
||||||
|
let seg_end = 64 + heap.data_segment_size as usize;
|
||||||
|
file[seg_end - 1] = b'n';
|
||||||
|
let storage = CountingStorage::new(file.clone());
|
||||||
|
let last = off - names[names.len() - 1].len() as u64 - 1;
|
||||||
|
let want = heap.read_string(&file, last);
|
||||||
|
assert!(want.is_err());
|
||||||
|
assert_eq!(heap.read_string_in(&storage, last), want);
|
||||||
|
|
||||||
|
// A 64 MiB file whose heap claims a data segment reaching its end.
|
||||||
|
let mut big = build_heap_file(0, 64, &["short", "names"], 8, 8);
|
||||||
|
big.resize(64 << 20, 0);
|
||||||
|
big[8..16].copy_from_slice(&((64u64 << 20) - 64).to_le_bytes());
|
||||||
|
let heap = LocalHeap::parse(&big, 0, 8, 8).unwrap();
|
||||||
|
let storage = CountingStorage::new(big.clone());
|
||||||
|
assert_eq!(heap.read_string_in(&storage, 6).unwrap(), "names");
|
||||||
|
assert_eq!((storage.reads(), storage.bytes_read()), (1, 64));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
//! Work counters for tests of lookup cost (feature `lookup-stats`).
|
||||||
|
//!
|
||||||
|
//! Counts fractal-heap objects read — each is one link or attribute message
|
||||||
|
//! decoded out of a dense group or dense attribute storage — so a test can
|
||||||
|
//! check that finding one name reads a handful of them, not the whole group.
|
||||||
|
//! Per thread, so tests running in parallel do not see each other's reads.
|
||||||
|
//! Without the feature the counting compiles to nothing.
|
||||||
|
|
||||||
|
#[cfg(feature = "lookup-stats")]
|
||||||
|
std::thread_local! {
|
||||||
|
static HEAP_OBJECTS: core::cell::Cell<u64> = const { core::cell::Cell::new(0) };
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record one heap object read.
|
||||||
|
#[inline(always)]
|
||||||
|
pub(crate) fn heap_object_read() {
|
||||||
|
#[cfg(feature = "lookup-stats")]
|
||||||
|
HEAP_OBJECTS.with(|c| c.set(c.get() + 1));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Heap objects read on this thread since the last [`reset`].
|
||||||
|
#[cfg(feature = "lookup-stats")]
|
||||||
|
pub fn heap_objects_read() -> u64 {
|
||||||
|
HEAP_OBJECTS.with(core::cell::Cell::get)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Zero this thread's counters.
|
||||||
|
#[cfg(feature = "lookup-stats")]
|
||||||
|
pub fn reset() {
|
||||||
|
HEAP_OBJECTS.with(|c| c.set(0));
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -1,14 +1,27 @@
|
|||||||
//! Object header writer for v2 format.
|
//! Object header writer for v2 format.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::{format, vec::Vec};
|
||||||
|
|
||||||
use crate::checksum::jenkins_lookup3;
|
use crate::checksum::jenkins_lookup3;
|
||||||
|
use crate::error::FormatError;
|
||||||
use crate::message_type::MessageType;
|
use crate::message_type::MessageType;
|
||||||
|
|
||||||
|
/// Largest message payload a v2 object header can describe: the per-message
|
||||||
|
/// size field is 2 bytes. A bigger message cannot be encoded at all — writing
|
||||||
|
/// its size truncated to 16 bits produced files libhdf5 refuses.
|
||||||
|
pub const MAX_MESSAGE_SIZE: usize = u16::MAX as usize;
|
||||||
|
|
||||||
|
/// Object header flags: attribute creation order tracked (each message
|
||||||
|
/// then carries a 2-byte creation order) and indexed.
|
||||||
|
const OHDR_ATTR_CRT_ORDER_TRACKED: u8 = 0x04;
|
||||||
|
const OHDR_ATTR_CRT_ORDER_INDEXED: u8 = 0x08;
|
||||||
|
|
||||||
/// Writer for v2 object headers with proper checksums.
|
/// Writer for v2 object headers with proper checksums.
|
||||||
pub struct ObjectHeaderWriter {
|
pub struct ObjectHeaderWriter {
|
||||||
messages: Vec<(MessageType, Vec<u8>, u8)>, // (type, data, msg_flags)
|
messages: Vec<(MessageType, Vec<u8>, u8, u16)>, // (type, data, msg_flags, creation order)
|
||||||
|
/// Attribute creation order tracked and indexed.
|
||||||
|
attr_order: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl ObjectHeaderWriter {
|
impl ObjectHeaderWriter {
|
||||||
@@ -16,26 +29,59 @@ impl ObjectHeaderWriter {
|
|||||||
pub fn new() -> Self {
|
pub fn new() -> Self {
|
||||||
Self {
|
Self {
|
||||||
messages: Vec::new(),
|
messages: Vec::new(),
|
||||||
|
attr_order: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Track and index attribute creation order, as libhdf5 does for an
|
||||||
|
/// object created with `H5P_CRT_ORDER_TRACKED | H5P_CRT_ORDER_INDEXED`
|
||||||
|
/// (h5py's `track_order=True`): the header's flags say so, and every
|
||||||
|
/// message carries a creation order (an attribute's own; 0 for the
|
||||||
|
/// others). libhdf5 reads the setting back from these flags.
|
||||||
|
pub fn track_attr_order(&mut self) {
|
||||||
|
self.attr_order = true;
|
||||||
|
}
|
||||||
|
|
||||||
/// Add a message to the header with default flags (0).
|
/// Add a message to the header with default flags (0).
|
||||||
pub fn add_message(&mut self, msg_type: MessageType, data: Vec<u8>) {
|
pub fn add_message(&mut self, msg_type: MessageType, data: Vec<u8>) {
|
||||||
self.messages.push((msg_type, data, 0));
|
self.messages.push((msg_type, data, 0, 0));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Add a message with specific flags.
|
/// Add a message with specific flags.
|
||||||
pub fn add_message_with_flags(&mut self, msg_type: MessageType, data: Vec<u8>, flags: u8) {
|
pub fn add_message_with_flags(&mut self, msg_type: MessageType, data: Vec<u8>, flags: u8) {
|
||||||
self.messages.push((msg_type, data, flags));
|
self.messages.push((msg_type, data, flags, 0));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Add a message with its creation order, which is written only when
|
||||||
|
/// attribute creation order is tracked ([`Self::track_attr_order`]).
|
||||||
|
pub fn add_message_with_order(&mut self, msg_type: MessageType, data: Vec<u8>, order: u16) {
|
||||||
|
self.messages.push((msg_type, data, 0, order));
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Serialize the complete v2 object header (OHDR + messages + checksum).
|
/// Serialize the complete v2 object header (OHDR + messages + checksum).
|
||||||
pub fn serialize(&self) -> Vec<u8> {
|
///
|
||||||
// Calculate total message bytes: each message has type(1) + size(2) + flags(1) + data
|
/// Fails with [`FormatError::SerializationError`] when a message is larger
|
||||||
|
/// than [`MAX_MESSAGE_SIZE`] (e.g. an attribute over ~64 KiB, which would
|
||||||
|
/// need dense attribute storage), rather than writing a corrupt header.
|
||||||
|
pub fn serialize(&self) -> Result<Vec<u8>, FormatError> {
|
||||||
|
if let Some((msg_type, data, _, _)) = self
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|(_, data, _, _)| data.len() > MAX_MESSAGE_SIZE)
|
||||||
|
{
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"{msg_type:?} message is {} bytes; an object header message holds at most \
|
||||||
|
{MAX_MESSAGE_SIZE} bytes",
|
||||||
|
data.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
// Calculate total message bytes: each message has type(1) + size(2) +
|
||||||
|
// flags(1) [+ creation order(2)] + data
|
||||||
|
let msg_header = if self.attr_order { 6 } else { 4 };
|
||||||
let msg_bytes_total: usize = self
|
let msg_bytes_total: usize = self
|
||||||
.messages
|
.messages
|
||||||
.iter()
|
.iter()
|
||||||
.map(|(_, data, _)| 4 + data.len())
|
.map(|(_, data, _, _)| msg_header + data.len())
|
||||||
.sum();
|
.sum();
|
||||||
|
|
||||||
// Determine chunk size field width based on msg_bytes_total
|
// Determine chunk size field width based on msg_bytes_total
|
||||||
@@ -47,6 +93,12 @@ impl ObjectHeaderWriter {
|
|||||||
(0x02u8, 4)
|
(0x02u8, 4)
|
||||||
};
|
};
|
||||||
|
|
||||||
|
let flags = if self.attr_order {
|
||||||
|
flags | OHDR_ATTR_CRT_ORDER_TRACKED | OHDR_ATTR_CRT_ORDER_INDEXED
|
||||||
|
} else {
|
||||||
|
flags
|
||||||
|
};
|
||||||
|
|
||||||
let mut buf = Vec::new();
|
let mut buf = Vec::new();
|
||||||
|
|
||||||
// OHDR signature
|
// OHDR signature
|
||||||
@@ -64,7 +116,7 @@ impl ObjectHeaderWriter {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Messages
|
// Messages
|
||||||
for (msg_type, data, msg_flags) in &self.messages {
|
for (msg_type, data, msg_flags, order) in &self.messages {
|
||||||
let type_id = msg_type.to_u16();
|
let type_id = msg_type.to_u16();
|
||||||
assert!(
|
assert!(
|
||||||
type_id <= 255,
|
type_id <= 255,
|
||||||
@@ -73,6 +125,9 @@ impl ObjectHeaderWriter {
|
|||||||
buf.push(type_id as u8); // type (1 byte in v2)
|
buf.push(type_id as u8); // type (1 byte in v2)
|
||||||
buf.extend_from_slice(&(data.len() as u16).to_le_bytes()); // size (2 bytes)
|
buf.extend_from_slice(&(data.len() as u16).to_le_bytes()); // size (2 bytes)
|
||||||
buf.push(*msg_flags); // flags
|
buf.push(*msg_flags); // flags
|
||||||
|
if self.attr_order {
|
||||||
|
buf.extend_from_slice(&order.to_le_bytes()); // creation order
|
||||||
|
}
|
||||||
buf.extend_from_slice(data);
|
buf.extend_from_slice(data);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -80,7 +135,7 @@ impl ObjectHeaderWriter {
|
|||||||
let checksum = jenkins_lookup3(&buf);
|
let checksum = jenkins_lookup3(&buf);
|
||||||
buf.extend_from_slice(&checksum.to_le_bytes());
|
buf.extend_from_slice(&checksum.to_le_bytes());
|
||||||
|
|
||||||
buf
|
Ok(buf)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -125,15 +180,22 @@ impl BatchObjectHeaderWriter {
|
|||||||
|
|
||||||
/// Compute the serialized size of each header without actually serializing.
|
/// Compute the serialized size of each header without actually serializing.
|
||||||
/// Returns sizes in the same order as headers were added.
|
/// Returns sizes in the same order as headers were added.
|
||||||
pub fn compute_sizes(&self) -> Vec<usize> {
|
pub fn compute_sizes(&self) -> Result<Vec<usize>, FormatError> {
|
||||||
self.headers.iter().map(|h| h.serialize().len()).collect()
|
self.headers
|
||||||
|
.iter()
|
||||||
|
.map(|h| h.serialize().map(|b| b.len()))
|
||||||
|
.collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Serialize all headers into a single contiguous buffer.
|
/// Serialize all headers into a single contiguous buffer.
|
||||||
/// Returns `(combined_bytes, offsets)` where `offsets[i]` is the byte
|
/// Returns `(combined_bytes, offsets)` where `offsets[i]` is the byte
|
||||||
/// offset of header `i` within the combined buffer.
|
/// offset of header `i` within the combined buffer.
|
||||||
pub fn serialize_all(&self) -> (Vec<u8>, Vec<usize>) {
|
pub fn serialize_all(&self) -> Result<(Vec<u8>, Vec<usize>), FormatError> {
|
||||||
let serialized: Vec<Vec<u8>> = self.headers.iter().map(|h| h.serialize()).collect();
|
let serialized: Vec<Vec<u8>> = self
|
||||||
|
.headers
|
||||||
|
.iter()
|
||||||
|
.map(|h| h.serialize())
|
||||||
|
.collect::<Result<_, _>>()?;
|
||||||
let total: usize = serialized.iter().map(|s| s.len()).sum();
|
let total: usize = serialized.iter().map(|s| s.len()).sum();
|
||||||
let mut buf = Vec::with_capacity(total);
|
let mut buf = Vec::with_capacity(total);
|
||||||
let mut offsets = Vec::with_capacity(serialized.len());
|
let mut offsets = Vec::with_capacity(serialized.len());
|
||||||
@@ -141,7 +203,7 @@ impl BatchObjectHeaderWriter {
|
|||||||
offsets.push(buf.len());
|
offsets.push(buf.len());
|
||||||
buf.extend_from_slice(s);
|
buf.extend_from_slice(s);
|
||||||
}
|
}
|
||||||
(buf, offsets)
|
Ok((buf, offsets))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -159,18 +221,33 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn empty_header_roundtrip() {
|
fn empty_header_roundtrip() {
|
||||||
let writer = ObjectHeaderWriter::new();
|
let writer = ObjectHeaderWriter::new();
|
||||||
let bytes = writer.serialize();
|
let bytes = writer.serialize().unwrap();
|
||||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
assert_eq!(hdr.version, 2);
|
assert_eq!(hdr.version, 2);
|
||||||
assert_eq!(hdr.messages.len(), 0);
|
assert_eq!(hdr.messages.len(), 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tracked_attribute_order_is_in_the_flags_and_every_message() {
|
||||||
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
|
writer.track_attr_order();
|
||||||
|
writer.add_message(MessageType::Dataspace, vec![1, 2, 3, 4]);
|
||||||
|
writer.add_message_with_order(MessageType::Attribute, vec![5, 6], 7);
|
||||||
|
let bytes = writer.serialize().unwrap();
|
||||||
|
assert_eq!(bytes[5] & 0x0C, 0x0C);
|
||||||
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(hdr.messages.len(), 2);
|
||||||
|
assert_eq!(hdr.messages[0].creation_order, Some(0));
|
||||||
|
assert_eq!(hdr.messages[1].creation_order, Some(7));
|
||||||
|
assert_eq!(hdr.messages[1].data, vec![5, 6]);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn two_messages_roundtrip() {
|
fn two_messages_roundtrip() {
|
||||||
let mut writer = ObjectHeaderWriter::new();
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
writer.add_message(MessageType::Dataspace, vec![1, 2, 3, 4]);
|
writer.add_message(MessageType::Dataspace, vec![1, 2, 3, 4]);
|
||||||
writer.add_message(MessageType::Datatype, vec![5, 6]);
|
writer.add_message(MessageType::Datatype, vec![5, 6]);
|
||||||
let bytes = writer.serialize();
|
let bytes = writer.serialize().unwrap();
|
||||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
assert_eq!(hdr.messages.len(), 2);
|
assert_eq!(hdr.messages.len(), 2);
|
||||||
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
|
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
|
||||||
@@ -184,12 +261,30 @@ mod tests {
|
|||||||
let mut writer = ObjectHeaderWriter::new();
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
// Add a message with >255 bytes of payload
|
// Add a message with >255 bytes of payload
|
||||||
writer.add_message(MessageType::Datatype, vec![0xAA; 300]);
|
writer.add_message(MessageType::Datatype, vec![0xAA; 300]);
|
||||||
let bytes = writer.serialize();
|
let bytes = writer.serialize().unwrap();
|
||||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
assert_eq!(hdr.messages.len(), 1);
|
assert_eq!(hdr.messages.len(), 1);
|
||||||
assert_eq!(hdr.messages[0].data.len(), 300);
|
assert_eq!(hdr.messages[0].data.len(), 300);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn oversized_message_is_an_error_not_a_truncated_size() {
|
||||||
|
// 65535 bytes is the largest encodable payload.
|
||||||
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
|
writer.add_message(MessageType::Attribute, vec![0; MAX_MESSAGE_SIZE]);
|
||||||
|
let bytes = writer.serialize().unwrap();
|
||||||
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(hdr.messages[0].data.len(), MAX_MESSAGE_SIZE);
|
||||||
|
|
||||||
|
// One byte more used to be written with its size wrapped to 0.
|
||||||
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
|
writer.add_message(MessageType::Attribute, vec![0; MAX_MESSAGE_SIZE + 1]);
|
||||||
|
assert!(matches!(
|
||||||
|
writer.serialize(),
|
||||||
|
Err(FormatError::SerializationError(_))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn batch_writer_serialize_all() {
|
fn batch_writer_serialize_all() {
|
||||||
let mut batch = BatchObjectHeaderWriter::new();
|
let mut batch = BatchObjectHeaderWriter::new();
|
||||||
@@ -204,7 +299,7 @@ mod tests {
|
|||||||
batch.add(w2);
|
batch.add(w2);
|
||||||
assert_eq!(batch.len(), 2);
|
assert_eq!(batch.len(), 2);
|
||||||
|
|
||||||
let (buf, offsets) = batch.serialize_all();
|
let (buf, offsets) = batch.serialize_all().unwrap();
|
||||||
assert_eq!(offsets.len(), 2);
|
assert_eq!(offsets.len(), 2);
|
||||||
assert_eq!(offsets[0], 0);
|
assert_eq!(offsets[0], 0);
|
||||||
|
|
||||||
@@ -222,7 +317,7 @@ mod tests {
|
|||||||
fn batch_writer_empty() {
|
fn batch_writer_empty() {
|
||||||
let batch = BatchObjectHeaderWriter::new();
|
let batch = BatchObjectHeaderWriter::new();
|
||||||
assert!(batch.is_empty());
|
assert!(batch.is_empty());
|
||||||
let (buf, offsets) = batch.serialize_all();
|
let (buf, offsets) = batch.serialize_all().unwrap();
|
||||||
assert!(buf.is_empty());
|
assert!(buf.is_empty());
|
||||||
assert!(offsets.is_empty());
|
assert!(offsets.is_empty());
|
||||||
}
|
}
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user