Compare commits
81
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bb78d70b99 | ||
|
|
a7de15534c | ||
|
|
10d1029ead | ||
|
|
883980f2bd | ||
|
|
d6e426e6d5 | ||
|
|
256e7b89e4 | ||
|
|
f2e704abf3 | ||
|
|
61f36516d7 | ||
|
|
45720fe5a6 | ||
|
|
adf961c883 | ||
|
|
b4a44a2e66 | ||
|
|
17fa783dce | ||
|
|
90e050944f | ||
|
|
a6e90f3ee3 | ||
|
|
0555794850 | ||
|
|
efc2dc53c9 | ||
|
|
5c2f656fe7 | ||
|
|
945b13a1f1 | ||
|
|
9179aa356e | ||
|
|
e94a52a88b | ||
|
|
d54a0f4737 | ||
|
|
aadfd18d4c | ||
|
|
2c6c6c176e | ||
|
|
190918a478 | ||
|
|
38d0d4de02 | ||
|
|
1c85986079 | ||
|
|
c7092722aa | ||
|
|
36356ba8a1 | ||
|
|
85eb7f5ce2 | ||
|
|
8196fab72a | ||
|
|
8ebd488d9e | ||
|
|
42b81d9f1c | ||
|
|
72b9cfb1e1 | ||
|
|
650f355219 | ||
|
|
7f5cfee281 | ||
|
|
c5302e587e | ||
|
|
e1115bc92a | ||
|
|
36d7a6f234 | ||
|
|
4b23ad697c | ||
|
|
e7f2d8575d | ||
|
|
7c1968a34a | ||
|
|
6db13c60b8 | ||
|
|
d99426be94 | ||
|
|
f5505fb03d | ||
|
|
95dcb04454 | ||
|
|
1dba7b465a | ||
|
|
57e938c438 | ||
|
|
3000b40cf3 | ||
|
|
bc820fbd8c | ||
|
|
9066d34eaa | ||
|
|
540fa08907 | ||
|
|
14876b8ae5 | ||
|
|
5935e13866 | ||
|
|
8c3ef996ea | ||
|
|
74fdf0582b | ||
|
|
44f5f8b5c5 | ||
|
|
2f252df084 | ||
|
|
c8c2930fc0 | ||
|
|
417c9516ca | ||
|
|
53dbddb07b | ||
|
|
081341b433 | ||
|
|
d074385944 | ||
|
|
4a1876faf2 | ||
|
|
be88e3fec7 | ||
|
|
e162c013fd | ||
|
|
06dda26d85 | ||
|
|
585e14d5e2 | ||
|
|
183d96ee26 | ||
|
|
bba1560416 | ||
|
|
b36998ef01 | ||
|
|
aef8e766ae | ||
|
|
9ea44d473d | ||
|
|
46203ea761 | ||
|
|
75bdb53342 | ||
|
|
79dfa78e8f | ||
|
|
bdadf3447c | ||
|
|
7706697feb | ||
|
|
c0f704c381 | ||
|
|
a7920bd4b3 | ||
|
|
7e43b5366c | ||
|
|
73bb068264 |
@@ -33,10 +33,10 @@ jobs:
|
|||||||
# build (pure-Rust zlib-rs) does not need it.
|
# build (pure-Rust zlib-rs) does not need it.
|
||||||
apt-get install -y --no-install-recommends python3 python3-venv cmake
|
apt-get install -y --no-install-recommends python3 python3-venv cmake
|
||||||
python3 -m venv /opt/interop
|
python3 -m venv /opt/interop
|
||||||
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray
|
/opt/interop/bin/pip install --no-cache-dir h5py numpy netCDF4 xarray hdf5plugin
|
||||||
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
echo "/opt/interop/bin" >> "$GITHUB_PATH"
|
||||||
- name: Show interop library versions
|
- name: Show interop library versions
|
||||||
run: /opt/interop/bin/python -c "import h5py, netCDF4; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__)"
|
run: /opt/interop/bin/python -c "import h5py, netCDF4, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'netCDF4', netCDF4.__version__, 'hdf5plugin', hdf5plugin.version)"
|
||||||
- name: Run CI script
|
- name: Run CI script
|
||||||
env:
|
env:
|
||||||
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
# Name the interpreter outright rather than relying on $GITHUB_PATH
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
name: Conformance
|
||||||
|
# Nightly: read every file of the pinned public HDF5 corpora with clawhdf5 and
|
||||||
|
# with h5py/libhdf5 and compare (conformance/run.sh; CONFORMANCE.md explains
|
||||||
|
# the method). Fails on any panic, hang, crash or out-of-memory in clawhdf5,
|
||||||
|
# and when the ok count drops below conformance/baseline.json or a file the
|
||||||
|
# baseline lists as ok stops being ok. The report is printed into the job log;
|
||||||
|
# nothing is uploaded (artifact actions are JavaScript, which rust:latest
|
||||||
|
# cannot run — see CLAUDE.md).
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "17 3 * * *"
|
||||||
|
workflow_dispatch:
|
||||||
|
jobs:
|
||||||
|
conformance:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
container: rust:latest
|
||||||
|
timeout-minutes: 60
|
||||||
|
env:
|
||||||
|
CARGO_NET_RETRY: "10"
|
||||||
|
steps:
|
||||||
|
# Plain git, not actions/checkout (a JavaScript action; see ci.yml).
|
||||||
|
- name: Check out
|
||||||
|
run: |
|
||||||
|
git init -q .
|
||||||
|
git remote add origin "${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}.git"
|
||||||
|
for i in 1 2 3; do git fetch -q --depth 1 origin "${GITHUB_SHA}" && break; sleep 5; done
|
||||||
|
git checkout -q FETCH_HEAD
|
||||||
|
- name: Install h5py, h5dump and the probe's codec libraries
|
||||||
|
# hdf5-tools: h5dump for the CVE-corpus comparison. libaec-dev and
|
||||||
|
# pkg-config: the probe builds clawhdf5-format with `szip` (the core
|
||||||
|
# crates' default build needs neither).
|
||||||
|
run: |
|
||||||
|
apt-get update
|
||||||
|
apt-get install -y --no-install-recommends python3 python3-venv hdf5-tools libaec-dev pkg-config
|
||||||
|
python3 -m venv /opt/conformance
|
||||||
|
/opt/conformance/bin/pip install --no-cache-dir -r conformance/requirements.txt
|
||||||
|
/opt/conformance/bin/python -c "import h5py, hdf5plugin; print('h5py', h5py.__version__, 'HDF5', h5py.version.hdf5_version, 'hdf5plugin', hdf5plugin.version)"
|
||||||
|
h5dump --version
|
||||||
|
- name: Probe unit tests
|
||||||
|
run: cargo test --release --manifest-path conformance/probe/Cargo.toml
|
||||||
|
env:
|
||||||
|
CARGO_TARGET_DIR: conformance/.cache/target
|
||||||
|
- name: Sweep
|
||||||
|
# The corpora come from GitHub (pinned commits, conformance/corpus.txt),
|
||||||
|
# so this job needs a runner that reaches github.com.
|
||||||
|
env:
|
||||||
|
CLAWHDF5_PYTHON: /opt/conformance/bin/python
|
||||||
|
run: bash conformance/run.sh
|
||||||
|
- name: Report
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
if [ -f CONFORMANCE.md ]; then cat CONFORMANCE.md; else echo "no report was generated"; fi
|
||||||
|
if [ -f conformance/.cache/results/summary.md ]; then
|
||||||
|
echo; echo "---- per-file detail (conformance/.cache/results/summary.md) ----"
|
||||||
|
cat conformance/.cache/results/summary.md
|
||||||
|
fi
|
||||||
+277
@@ -3,6 +3,38 @@
|
|||||||
## Unreleased
|
## Unreleased
|
||||||
|
|
||||||
### Upgrade Notes
|
### Upgrade Notes
|
||||||
|
- **HDF5 correctness audit (2026-09-25).** A sweep of 686 public files (the
|
||||||
|
libhdf5 test files, the HDF Group's CVE reproducers, pyfive, netcdf-c,
|
||||||
|
netcdf4-python, h5wasm, h5py and xarray corpora), a 567-case read matrix and
|
||||||
|
a 96-case write matrix against HDF5 1.10–2.0 found bugs that returned wrong
|
||||||
|
values with no error, and files we wrote that libhdf5 rejects. The fixes are
|
||||||
|
listed under Correctness and Interop. What changes for callers:
|
||||||
|
- **Chunked datasets whose max shape is larger than their current shape**,
|
||||||
|
or whose unlimited dimension is not the first, were indexed by the current
|
||||||
|
shape instead of the max shape, both when read and when written. Files from
|
||||||
|
libhdf5 now read correctly. Files clawhdf5 wrote with such a max shape were
|
||||||
|
laid out wrongly and now read the way libhdf5 always read them — rewrite
|
||||||
|
them. Agent stores and ClawBrainHub files have no max shape and are
|
||||||
|
unaffected.
|
||||||
|
- Integer reads (`read_i32`/`read_i64`/`read_u64`/...) of float data now
|
||||||
|
convert (truncate toward zero, saturate at the type's range, NaN reads as
|
||||||
|
0) instead of returning the IEEE bit pattern, and out-of-range integers
|
||||||
|
saturate instead of keeping the low bits.
|
||||||
|
- `FileWriter::finish()` now returns an error instead of writing a corrupt
|
||||||
|
file for: a header message over 64 KiB (e.g. an attribute larger than
|
||||||
|
~64 KiB), a group/dataset/link name that is empty, `.` or contains `/`
|
||||||
|
(nested paths were written as one literal link), a max shape smaller than
|
||||||
|
the shape, a page size outside 512 B–1 GiB, and more than 65 535 chunks in
|
||||||
|
a dataset with several unlimited dimensions.
|
||||||
|
- **Breaking (format crate):** `ObjectHeaderWriter::serialize`,
|
||||||
|
`BatchObjectHeaderWriter::compute_sizes`/`serialize_all` and
|
||||||
|
`build_chunked_data_from_precompressed` return `Result`;
|
||||||
|
`read_fixed_array_chunks`/`read_extensible_array_chunks` take `max_dims`;
|
||||||
|
`build_fixed_array_at`/`ea_writer::build_extensible_array_at` take one
|
||||||
|
`Option<WrittenChunk>` per index slot; `fill_value::dataset_fill_value`
|
||||||
|
returns `UnresolvedSharedMessage` for a shared message it cannot resolve
|
||||||
|
instead of `None`. `FillTime::default()` is `IfSet` (libhdf5's default;
|
||||||
|
default files are byte-identical).
|
||||||
- **ZeroClaw does not use clawhdf5.** The project described itself as
|
- **ZeroClaw does not use clawhdf5.** The project described itself as
|
||||||
ZeroClaw's memory backend ("imported as a `clawhdf5` Cargo feature"). Checked
|
ZeroClaw's memory backend ("imported as a `clawhdf5` Cargo feature"). Checked
|
||||||
against ZeroClaw v0.8.5 (the latest release), the `osobh/zeroclaw` fork and
|
against ZeroClaw v0.8.5 (the latest release), the `osobh/zeroclaw` fork and
|
||||||
@@ -165,6 +197,23 @@
|
|||||||
takes `--f32`; it had kept printing "f32" after the default changed.
|
takes `--f32`; it had kept printing "f32" after the default changed.
|
||||||
|
|
||||||
### Interop
|
### Interop
|
||||||
|
- **Conformance sweep in the repo** (`conformance/`, report in
|
||||||
|
`CONFORMANCE.md`). `conformance/run.sh` fetches eight public HDF5 corpora
|
||||||
|
pinned by commit (libhdf5's test files, the HDF Group's CVE reproducers,
|
||||||
|
pyfive, netcdf-c, netcdf4-python, h5wasm, h5py, xarray-data) into a
|
||||||
|
gitignored cache, reads every file with clawhdf5 and with h5py/libhdf5 (and
|
||||||
|
the CVE files with h5dump) under a timeout and memory limit, compares them
|
||||||
|
object by object and regenerates the report — about 30 s once the corpus is
|
||||||
|
cached. A nightly Gitea job (`.gitea/workflows/conformance.yml`) runs it and
|
||||||
|
fails on any panic, hang, crash or out-of-memory, or when a file in
|
||||||
|
`conformance/baseline.json` stops reading identically. First report, on
|
||||||
|
42b81d9: 467 of 697 files identical to h5py, 123 our-error, 15 mismatch
|
||||||
|
(2 of them an h5py bug), 92 that libhdf5 cannot read, no panics, hangs or
|
||||||
|
crashes. Compared with the ad-hoc audit sweep, the probe now compares
|
||||||
|
N-Bit floats (and integers with a bit offset) as the values libhdf5
|
||||||
|
converts them to rather than raw file bytes — 8 files that were reported as
|
||||||
|
mismatches read identically — and the reference side no longer flips
|
||||||
|
between runs when libhdf5 aborts while freeing h5py objects.
|
||||||
- `clawhdf5-format`: **every `f32` dataset was unreadable by h5py and
|
- `clawhdf5-format`: **every `f32` dataset was unreadable by h5py and
|
||||||
libhdf5.** The float datatype encoder hard-coded the sign bit's position to
|
libhdf5.** The float datatype encoder hard-coded the sign bit's position to
|
||||||
63, correct only for `f64`; libhdf5 validates it and refused the dataset. It
|
63, correct only for `f64`; libhdf5 validates it and refused the dataset. It
|
||||||
@@ -179,6 +228,37 @@
|
|||||||
`float16` rounding matches numpy's bit for bit on 4 020 probe values,
|
`float16` rounding matches numpy's bit for bit on 4 020 probe values,
|
||||||
including ties, subnormals and the overflow boundary), and an agent store —
|
including ties, subnormals and the overflow boundary), and an agent store —
|
||||||
`f32` and `float16` — opened by h5py with every dataset decoded.
|
`f32` and `float16` — opened by h5py with every dataset decoded.
|
||||||
|
- `clawhdf5-format` filters, checked against libhdf5 + hdf5plugin:
|
||||||
|
- **LZ4 (32004) now uses the registered HDF5 LZ4 format** (8-byte BE size,
|
||||||
|
4-byte BE block size, BE-length-prefixed blocks). Our old framing (4-byte
|
||||||
|
LE size + one block) was readable only by clawhdf5, and we could not read
|
||||||
|
libhdf5's (`h5ex_d_lz4.h5`). Old clawhdf5 LZ4 chunks still read; they are
|
||||||
|
told apart unambiguously (a registered chunk starts with four zero bytes).
|
||||||
|
- **Zstd (32015) frames now record the content size**, which libhdf5's zstd
|
||||||
|
plugin needs; h5py could not read our zstd datasets.
|
||||||
|
- **Pcodec moved from filter ID 32023 to 480.** 32023 is registered to
|
||||||
|
Granular BitRound, whose decode is a pass-through — libhdf5 with that
|
||||||
|
plugin would have returned compressed bytes as data. Pcodec has no
|
||||||
|
registered ID; 480 is in the registry's private range (256–511) and only
|
||||||
|
clawhdf5 can read it. Chunks written under 32023 with the filter name
|
||||||
|
`pcodec` (clawhdf5 ≤ 2.7.0) still read.
|
||||||
|
- **SZIP decode matches libhdf5.** It returned garbage or zeros with no
|
||||||
|
error for libhdf5-written files (the 4-byte size prefix, 32/64-bit
|
||||||
|
byte-plane interleaving, reference interval, scanline padding and byte
|
||||||
|
order were all handled wrongly) and rejected 64-bit data.
|
||||||
|
- N-Bit honours libhdf5's "need not compress" flag (multi-filter pipelines
|
||||||
|
such as `tfilters.h5` failed) and reads enum/no-op members.
|
||||||
|
- Scale-offset `float` decode uses libhdf5's single-precision arithmetic
|
||||||
|
(was 1 ULP off for some values).
|
||||||
|
- A pipeline with Fletcher32 ahead of the compressor (h5py
|
||||||
|
`set_fletcher32()` then `set_deflate()`) no longer fails with "deflate:
|
||||||
|
output exceeds size limit".
|
||||||
|
- `clawhdf5-format`: **HDF5 1.4/1.6-era files are readable.** Data Layout
|
||||||
|
message versions 1 and 2 (compact, contiguous, and chunked through the
|
||||||
|
version-1 B-tree) failed with `InvalidLayoutVersion` — 84 of the 686 files in
|
||||||
|
the 2026-09-25 audit sweep, 205 datasets. They now read as libhdf5 does;
|
||||||
|
checked byte for byte against h5py on HDF5's own test files
|
||||||
|
(`tests/legacy_format_interop.rs`).
|
||||||
|
|
||||||
### Storage
|
### Storage
|
||||||
- `clawhdf5-format`: **half-precision datasets.**
|
- `clawhdf5-format`: **half-precision datasets.**
|
||||||
@@ -216,6 +296,203 @@
|
|||||||
- CI keeps zlib-ng building and tested; the arm64 job no longer needs cmake.
|
- CI keeps zlib-ng building and tested; the arm64 job no longer needs cmake.
|
||||||
|
|
||||||
### Correctness
|
### Correctness
|
||||||
|
- `clawhdf5-format` VDS: variable-length and reference data from a source in
|
||||||
|
another file is refused. Those elements are global-heap IDs and object
|
||||||
|
addresses in the source file; copied into the virtual dataset they would
|
||||||
|
be decoded against the wrong file and name another object.
|
||||||
|
- `clawhdf5-agent`: a store whose `/meta` has an attribute that cannot be
|
||||||
|
decoded fails to open (`MemoryError::Schema`). With `attrs()` now leaving
|
||||||
|
unreadable attributes out, it would otherwise have opened with defaults in
|
||||||
|
place of its settings (`float16`, `compression`, the WAL mark, ...).
|
||||||
|
- `clawhdf5-format` reader: an old-style group whose local heap has a free
|
||||||
|
list pointing outside the heap was listed with names read from the broken
|
||||||
|
heap (garbage names on `cve-2021-36977.h5` once its user block was
|
||||||
|
applied). libhdf5 refuses such a heap ("bad heap free list"); so do we now,
|
||||||
|
with `FormatError::InvalidLocalHeapFreeList`. As in libhdf5 the free list
|
||||||
|
is checked when the first name is read (`LocalHeap::validate_free_list`,
|
||||||
|
new), so an empty group with a damaged heap still lists as empty.
|
||||||
|
- **Files with a user block** (`h5py.File(..., userblock_size=N)`, `h5jam`;
|
||||||
|
the superblock at 512, 1024, …) could not be read: every address in the
|
||||||
|
file is relative to the superblock, but it was applied from byte 0
|
||||||
|
(`InvalidObjectHeaderVersion` on the root group). `File` (mmap, buffered,
|
||||||
|
`from_bytes`), `MmapFile`, `LazyFile`, `AsyncHDF5File`, the VOL readers,
|
||||||
|
the HNSW loader and external VDS sources now view the file from the
|
||||||
|
superblock on, using the signature's position as the base address as
|
||||||
|
libhdf5 does; `user_block_size()` reports the user block (h5py's
|
||||||
|
`userblock_size`), and `as_bytes()` returns the bytes from the superblock
|
||||||
|
on. **Breaking (format crate):** `Superblock::parse` refuses a non-zero
|
||||||
|
signature offset with `FormatError::UserBlockNotStripped`, since the
|
||||||
|
addresses it returns would be applied to the wrong bytes; pass the slice
|
||||||
|
from `signature::split_user_block` (new) and parse at offset 0.
|
||||||
|
- `clawhdf5-format` reader: version-1 shared messages (HDF5 1.6-era files,
|
||||||
|
e.g. a dataset using a committed datatype in libhdf5's `tcompound.h5`)
|
||||||
|
read the heap-offset field of the embedded symbol-table entry as the
|
||||||
|
target address and failed with `InvalidObjectHeaderVersion`. The address
|
||||||
|
is now read after it, as libhdf5 does. **Breaking (format crate):**
|
||||||
|
`shared_message::parse_shared_ref` takes `length_size`. A reference whose
|
||||||
|
target header has no message of the referenced type is now
|
||||||
|
`FormatError::SharedMessageTargetMissing` instead of returning the first
|
||||||
|
other message found there (which decoded as garbage).
|
||||||
|
- `clawhdf5-format` reader: array members of version-1 compound datatypes
|
||||||
|
(HDF5 1.6-era files, e.g. libhdf5's `tcompound.h5`) were read as a single
|
||||||
|
element: a `[4] i32` member came back as one `i32`, with the wrong size.
|
||||||
|
The legacy per-member dimension fields are now decoded into an array type,
|
||||||
|
as libhdf5 does; more than four dimensions, or a zero-sized one, is an
|
||||||
|
error.
|
||||||
|
- `clawhdf5-format` virtual datasets (VDS), checked against HDF5 2.0 through
|
||||||
|
h5py (`crates/clawhdf5/tests/vds_interop.rs`):
|
||||||
|
- **Wrong data:** elements no mapping supplies — unmapped regions, and
|
||||||
|
mappings whose source file or dataset is missing — read as 0 instead of
|
||||||
|
the virtual dataset's fill value (e.g. h5py `fillvalue=-1`). Assembly moved
|
||||||
|
to the new `vds` module: `vds::read_virtual_dataset` takes the fill value
|
||||||
|
and a resolver that can refuse a name (`VdsFileResolver`), and `File`
|
||||||
|
passes the dataset's fill value. A missing source *dataset* read as an
|
||||||
|
error; it is fill now, as in libhdf5. Source datasets are read with their
|
||||||
|
own fill value for unallocated chunks, and a source whose datatype differs
|
||||||
|
from the virtual dataset's is an error (libhdf5 converts; we do not).
|
||||||
|
`File` now refuses a source name that leaves the virtual file's directory
|
||||||
|
(`../x.h5`, absolute paths), or any external source of a `File::from_bytes`
|
||||||
|
file, with an error — these used to read as fill.
|
||||||
|
**Behaviour change:** the raw-read API (`read_raw_data_full*`), which has
|
||||||
|
no fill value, now returns an error for a virtual dataset with unmapped
|
||||||
|
elements instead of zeros.
|
||||||
|
- Unlimited and printf-style mappings are supported (all 7 VDS files in the
|
||||||
|
libhdf5 test set are such mappings, e.g. Eiger/Percival detector layouts).
|
||||||
|
`%b` in a source file or dataset name is the block number and `%%` a
|
||||||
|
literal `%` (other `%` sequences are an error, as in libhdf5); block `j`
|
||||||
|
is read from the source named with `j`, probing from 0 up to the first
|
||||||
|
missing source. Unlimited source/virtual selections cover as much as the
|
||||||
|
source's current extent fills, including a partial last block. As
|
||||||
|
libhdf5 does on `H5Dget_space`, the extent is recomputed from the sources
|
||||||
|
present (default "last available" view, printf gap 0) —
|
||||||
|
`vds::virtual_dataset_extent`, used by `Dataset::shape()` — so e.g.
|
||||||
|
`vds-eiger.h5` is `[5, 10, 10]`, not its stored `[20, 10, 10]`. A source
|
||||||
|
stored in the other byte order is byte-swapped (libhdf5 converts);
|
||||||
|
other type conversions remain an error.
|
||||||
|
- Hyperslab selection versions 1 and 2 were refused ("only version-3
|
||||||
|
hyperslab selections are supported"). Version 1 is what libhdf5 writes for
|
||||||
|
every VDS created with the default format bounds (h5py's default), so
|
||||||
|
those could not be read at all; version 2 is its encoding of an unlimited
|
||||||
|
selection. Both are decoded now, as are irregular hyperslabs (a union of
|
||||||
|
blocks, read in row-major order as libhdf5 iterates them).
|
||||||
|
`SerializedSelection` exposes the raw form, including unlimited counts.
|
||||||
|
- The version-1 mapping list HDF5 2.0 writes (low version bound 2.0) was
|
||||||
|
misparsed: each entry's flags byte was read as the start of the source
|
||||||
|
file name, and names shared with an earlier entry (stored as that entry's
|
||||||
|
index) were not followed. Now decoded as `H5D__virtual_load_layout` does.
|
||||||
|
- `clawhdf5-format` reader — **values returned wrong with no error:**
|
||||||
|
- Fixed Array and Extensible Array chunk indexes were laid out by the
|
||||||
|
dataset's current shape instead of its max shape (23 libhdf5 test files,
|
||||||
|
and any h5py file with e.g. `maxshape=(10, None)` or `(20, 10)` under
|
||||||
|
`libver='latest'`).
|
||||||
|
- Files with 4-byte offsets: unfiltered chunked datasets read as zeros.
|
||||||
|
Chunk B-tree keys store offsets in 8 bytes whatever the file's offset
|
||||||
|
size.
|
||||||
|
- A chunk's filter mask skipped the whole pipeline when any bit was set;
|
||||||
|
only the flagged filters are skipped now.
|
||||||
|
- Float data read as an integer returned the bit pattern; narrowing integer
|
||||||
|
reads kept the low bits; bfloat16 was decoded as IEEE half. Floats are now
|
||||||
|
decoded from their datatype fields (bf16, FP8 E4M3/E5M2, IEEE half, single
|
||||||
|
and double).
|
||||||
|
- `vl_data::read_vl_bytes` truncated sequences of non-byte base types.
|
||||||
|
- A shared fill-value message read as zero fill; it is resolved now,
|
||||||
|
including from the file's shared-message (SOHM) table, which could never
|
||||||
|
resolve because its index version byte was skipped.
|
||||||
|
- Two threads reading two chunked datasets through one `File` could get each
|
||||||
|
other's chunks (the shared chunk cache was switched between datasets
|
||||||
|
across separate lock acquisitions). The cache is now keyed by dataset.
|
||||||
|
- Compound datatype version 1 members with legacy array dimensions (HDF5
|
||||||
|
before 1.4, which had no array class) were read as a single scalar at
|
||||||
|
the member's offset; they are now array members, as in libhdf5
|
||||||
|
(`tarrold.h5`, `tcompound.h5`). Only reachable once layout versions 1/2
|
||||||
|
were readable, since the files that use it are that old.
|
||||||
|
- `clawhdf5-format` reader — errors on valid files: a version-1 shared
|
||||||
|
message (a committed datatype in HDF5 1.4/1.6-era files) was read as if the
|
||||||
|
object header address followed the reserved bytes; it follows a link-name
|
||||||
|
offset (the reference is an old-style symbol table entry), so the reader
|
||||||
|
followed the name offset and failed with `InvalidObjectHeaderVersion`
|
||||||
|
(`tcompound.h5`). New `shared_message::parse_shared_ref_sized` takes the
|
||||||
|
superblock's length size; `parse_shared_ref` assumes it equals the offset
|
||||||
|
size.
|
||||||
|
- `clawhdf5-format` reader — errors on valid files: enum and bool datasets
|
||||||
|
through the numeric readers; the "don't filter partial edge chunks" layout
|
||||||
|
flag; Fletcher32 ahead of deflate (NetCDF-4's order). Unknown-message flags
|
||||||
|
follow libhdf5 (`tbogus.h5`): "fail if unknown" is refused, "fail if unknown
|
||||||
|
and writing" is ignored by a reader.
|
||||||
|
- `clawhdf5-format` reader — dense groups and attributes (links or
|
||||||
|
attributes kept in a fractal heap indexed by a v2 B-tree):
|
||||||
|
- A link heap larger than the root indirect block's direct rows (512 KiB
|
||||||
|
with libhdf5's defaults: a few thousand long link names, or ~20 000 short
|
||||||
|
ones) could not be listed: child indirect blocks were given the wrong
|
||||||
|
number of rows, so every link stored in one was unreachable.
|
||||||
|
- v2 B-trees of depth 3 or more (a dense group of ~22 000+ links) were
|
||||||
|
misparsed: internal-node child pointers were read with widths from an
|
||||||
|
estimate instead of libhdf5's per-depth record capacities, and the
|
||||||
|
listing failed. The same B-tree code indexes dense attributes, shared
|
||||||
|
messages and chunks.
|
||||||
|
- Fractal-heap "huge" objects (larger than the heap's managed-object
|
||||||
|
limit, 4 KiB by default — e.g. an 8 KiB dense attribute or a link with a
|
||||||
|
very long name) and "tiny" objects are now read; the ID type was taken
|
||||||
|
from the wrong bits (6-7, the version, instead of 4-5), so a huge object
|
||||||
|
failed and took every attribute on its object down with it (NetCDF-4
|
||||||
|
files such as netcdf4-python's `issue671.nc`). Huge objects are found
|
||||||
|
directly from the ID or through the huge-object v2 B-tree, filtered or
|
||||||
|
not.
|
||||||
|
- Heaps with an I/O filter pipeline (a group created with a filter on its
|
||||||
|
creation property list compresses its link heap) are now read: the
|
||||||
|
header's pipeline was skipped with the wrong size, so its checksum was
|
||||||
|
looked for in the wrong place, and filtered direct blocks were read raw.
|
||||||
|
- A user-defined link (link class 65-255, e.g. 187 in libhdf5's
|
||||||
|
`tall.h5`/`tudlink.h5`) made its whole group unlistable. Such links
|
||||||
|
cannot be followed without the application that registered the class, so
|
||||||
|
they are now left out of `datasets()`/`groups()` and path lookup, as h5py
|
||||||
|
leaves out links it cannot open; reserved link types are still an error.
|
||||||
|
- `clawhdf5` — soft links are listed, as h5py lists them: `datasets()` and
|
||||||
|
`groups()` on `Group`/`MmapGroup`/`LazyGroup` include each soft link under
|
||||||
|
its own name as the kind of object it resolves to, and `dataset(name)` /
|
||||||
|
`group(name)` open through it. Relative targets resolve from the group
|
||||||
|
holding the link. Dangling or cyclic soft links, external links and
|
||||||
|
user-defined links are left out (h5py lists their names but cannot open
|
||||||
|
them). Previously soft links were missing from the listings, and in
|
||||||
|
old-style (symbol table) groups a soft link made the listing fail. New
|
||||||
|
`group_v2::resolve_group_children` / `resolve_path_from` and
|
||||||
|
`group_v1::v1_soft_links` in `clawhdf5-format`.
|
||||||
|
- `clawhdf5` — one unreadable attribute no longer fails `attrs()` for every
|
||||||
|
attribute on its object: it is left out of the map, and the new
|
||||||
|
`attrs_with_errors()` (on every group and dataset handle) returns the map
|
||||||
|
plus one error per attribute left out. Returned values are always complete.
|
||||||
|
An error in the attribute index itself (attribute info message, dense heap
|
||||||
|
header or B-tree) still fails the call. `clawhdf5-format` gains
|
||||||
|
`attribute::extract_attributes_tolerant`; `extract_attributes_full` stays
|
||||||
|
strict.
|
||||||
|
- `clawhdf5-format` reader — files with shared object header messages
|
||||||
|
(SOHM, `H5Pset_shared_mesg_index`): a datatype, dataspace, filter pipeline
|
||||||
|
or attribute stored in the file's SOHM heap failed with "invalid shared
|
||||||
|
message version: 2" — only shared fill values loaded the SOHM table — so
|
||||||
|
such files' datasets and attributes could not be read.
|
||||||
|
`shared_message::resolve_shared_message` now loads the table when a
|
||||||
|
reference needs it (36 cases of the audit's read matrix).
|
||||||
|
- `clawhdf5-format` writer — **files libhdf5 rejects or reads wrong:**
|
||||||
|
- Extensible Array (one unlimited dimension): chunks from index 244 on were
|
||||||
|
written but never indexed and read as 0, by libhdf5 and by us.
|
||||||
|
- Fixed Array: more than 1 024 chunks gave checksum errors (data blocks
|
||||||
|
were never paged).
|
||||||
|
- A finite max shape larger than the shape gave libhdf5 "addr overflow"; an
|
||||||
|
unlimited dimension that is not the first scrambled the data; several
|
||||||
|
unlimited dimensions (`(None, None)`) broke the whole file. These now
|
||||||
|
write the index libhdf5 writes (swizzled Extensible Array, or a B-tree v2
|
||||||
|
index for several unlimited dimensions).
|
||||||
|
- Header messages over 64 KiB (the size field is 16 bits) and compact
|
||||||
|
datasets at 65 534–65 535 bytes produced corrupt files.
|
||||||
|
- Reference, Opaque, BitField and Time datatypes were written as empty
|
||||||
|
messages; they now encode as HDF5 2.0 does.
|
||||||
|
- `with_page_size` wrote a nonexistent superblock version 4; it now writes
|
||||||
|
the v3 superblock and File Space Info message libhdf5 writes.
|
||||||
|
- `FillTime` values were rotated on disk (NEVER was written as ALLOC, and so
|
||||||
|
on). New `DatasetBuilder::with_fill_value`.
|
||||||
|
- An empty-string attribute got a zero-size datatype, which made every
|
||||||
|
attribute on the object unreadable in libhdf5.
|
||||||
|
- `maxshape` equal to the shape no longer forces chunked layout.
|
||||||
- `clawhdf5-format`: **a truncated deflate chunk read back short, with no
|
- `clawhdf5-format`: **a truncated deflate chunk read back short, with no
|
||||||
error.** The deflate filter used flate2's streaming reader, which returns the
|
error.** The deflate filter used flate2's streaming reader, which returns the
|
||||||
bytes it has when the input runs out before the end-of-stream marker. It now
|
bytes it has when the input runs out before the end-of-stream marker. It now
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# clawhdf5
|
# clawhdf5
|
||||||
|
|
||||||
## Purpose
|
## Purpose
|
||||||
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated I/O. A standalone library. Its one verified consumer is ClawBrainHub (`.brain` files); no agent framework integrates it (OpenClaw and ZeroClaw claims were withdrawn on 2026-09-25 — neither was ever true).
|
Pure-Rust HDF5 format implementation with HNSW vector search, WAL-backed persistence, agent memory storage, and GPU-accelerated vector search. A standalone library. Its one verified consumer is ClawBrainHub (`.brain` files); no agent framework integrates it (OpenClaw and ZeroClaw claims were withdrawn on 2026-09-25 — neither was ever true).
|
||||||
|
|
||||||
## Architecture
|
## Architecture
|
||||||
|
|
||||||
@@ -11,13 +11,13 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
|||||||
|-------|------|
|
|-------|------|
|
||||||
| `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants |
|
| `clawhdf5-format` | HDF5 binary spec parser (superblock, B-tree, heap) — also holds shared type definitions and physical constants |
|
||||||
| `clawhdf5-io` | Read/write implementation |
|
| `clawhdf5-io` | Read/write implementation |
|
||||||
| `clawhdf5-filters` | Compression filters (gzip, LZ4, Zstd, Blosc) |
|
| `clawhdf5-filters` | Deflate backends (zlib-rs, zlib-ng, Apple Compression); the HDF5 filter pipeline and the other codecs (LZ4, Zstd, SZIP, N-Bit, scale-offset, pcodec) live in `clawhdf5-format`. No Blosc. |
|
||||||
| `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs |
|
| `clawhdf5-derive` | Proc-macro derive for HDF5-serializable structs |
|
||||||
| `clawhdf5` | Main facade crate |
|
| `clawhdf5` | Main facade crate |
|
||||||
| `clawhdf5-netcdf4` | NetCDF-4 compatibility layer |
|
| `clawhdf5-netcdf4` | NetCDF-4 compatibility layer |
|
||||||
| `clawhdf5-ann` | HNSW approximate nearest-neighbor vector index |
|
| `clawhdf5-ann` | HNSW approximate nearest-neighbor vector index |
|
||||||
| `clawhdf5-agent` | Agent memory, session history, knowledge graph storage |
|
| `clawhdf5-agent` | Agent memory, session history, knowledge graph storage |
|
||||||
| `clawhdf5-gpu` | GPU-accelerated I/O via wgpu (hand-written WGSL compute shaders) |
|
| `clawhdf5-gpu` | GPU vector distance computation via wgpu (hand-written WGSL compute shaders) — not dataset I/O |
|
||||||
| `clawhdf5-accel` | CPU SIMD acceleration path |
|
| `clawhdf5-accel` | CPU SIMD acceleration path |
|
||||||
| `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration |
|
| `clawhdf5-migrate` | SQLite → HDF5 agent-memory migration |
|
||||||
| `clawhdf5-android` | Android JNI bindings |
|
| `clawhdf5-android` | Android JNI bindings |
|
||||||
@@ -148,7 +148,7 @@ Cargo workspace with 16 crates under `crates/` (plus `libaec-sys`, an internal F
|
|||||||
Alerts never block a save — drain them with `HDF5Memory::take_anomaly_alerts`.
|
Alerts never block a save — drain them with `HDF5Memory::take_anomaly_alerts`.
|
||||||
`MemorySource` for this bookkeeping is inferred from the caller-supplied
|
`MemorySource` for this bookkeeping is inferred from the caller-supplied
|
||||||
`source_channel` string (a heuristic, not an authenticated trust boundary).
|
`source_channel` string (a heuristic, not an authenticated trust boundary).
|
||||||
- GPU-accelerated batch I/O for large dataset processing
|
- GPU-accelerated vector distance computation (`clawhdf5-gpu`, wgpu); HDF5 I/O itself is CPU-only
|
||||||
- Python and Node.js bindings for cross-language use
|
- Python and Node.js bindings for cross-language use
|
||||||
- NetCDF-4 compatibility for scientific data interop
|
- NetCDF-4 compatibility for scientific data interop
|
||||||
|
|
||||||
|
|||||||
+300
@@ -0,0 +1,300 @@
|
|||||||
|
# clawhdf5 conformance report
|
||||||
|
|
||||||
|
Every HDF5 file of eight public corpora (pinned by commit) is read twice — by
|
||||||
|
clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade
|
||||||
|
makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are
|
||||||
|
compared object by object: the set of hard-linked objects, each dataset's and
|
||||||
|
attribute's shape, and a SHA-256 of its values in a canonical encoding. The
|
||||||
|
CVE corpus is also run through `h5dump`. Each side runs under a timeout and an
|
||||||
|
address-space limit, so a hang, crash or runaway allocation is recorded, not
|
||||||
|
fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.
|
||||||
|
|
||||||
|
## Run
|
||||||
|
|
||||||
|
| | |
|
||||||
|
|---|---|
|
||||||
|
| date | 2026-09-26 03:46 UTC |
|
||||||
|
| clawhdf5 commit | `10d1029ead524e2fe64c2cd7f61b28067d9e449c` |
|
||||||
|
| machine | `tank`: AMD Ryzen 7 7800X3D 8-Core Processor, 16 CPUs, 61 GiB, Linux 7.0.0-34-generic x86_64 |
|
||||||
|
| command | `conformance/run.sh --no-fetch --update-baseline` |
|
||||||
|
| rustc | rustc 1.98.1 (48a229cea 2026-09-01) |
|
||||||
|
| reference | h5py 3.16.0, HDF5 2.0.0, numpy 2.5.3, hdf5plugin 7.1.0, Python 3.14.4 |
|
||||||
|
| h5dump | Version 1.14.6 (CVE corpus only) |
|
||||||
|
| limits | 20 s timeout (SIGKILL), 4096 MiB address space, per process; 16 files in parallel |
|
||||||
|
| runtime | 22 s probing + comparing (0 s fetch/build before it) |
|
||||||
|
|
||||||
|
## Results
|
||||||
|
|
||||||
|
A file's class is the first that applies:
|
||||||
|
|
||||||
|
- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.
|
||||||
|
- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.
|
||||||
|
- **our-error** — clawhdf5 returned an error for something h5py reads.
|
||||||
|
- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.
|
||||||
|
- **ok** — every object h5py reads, clawhdf5 reads identically.
|
||||||
|
|
||||||
|
| corpus | files | ok | our-error | mismatch | h5py-cannot-read | panic | hang | crash | oom |
|
||||||
|
|---|---|---|---|---|---|---|---|---|---|
|
||||||
|
| NCAS-CMS_pyfive | 33 | 32 | 0 | 1 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| cve_hdf5 | 147 | 100 | 6 | 9 | 32 | 0 | 0 | 0 | 0 |
|
||||||
|
| h5py_data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| hdf5 | 466 | 386 | 8 | 12 | 60 | 0 | 0 | 0 | 0 |
|
||||||
|
| netcdf-c | 20 | 20 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| netcdf4-python | 18 | 18 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| usnistgov_h5wasm | 5 | 5 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| xarray-data | 4 | 4 | 0 | 0 | 0 | 0 | 0 | 0 | 0 |
|
||||||
|
| **all** | **697** | **569** | **14** | **22** | **92** | **0** | **0** | **0** | **0** |
|
||||||
|
|
||||||
|
2 of the 22 mismatches are a known h5py bug, not ours (see *Known not-our-bug*).
|
||||||
|
|
||||||
|
Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):
|
||||||
|
|
||||||
|
| corpus | source | commit |
|
||||||
|
|---|---|---|
|
||||||
|
| hdf5 | https://github.com/HDFGroup/hdf5 | `a3cf1ea82cc7` |
|
||||||
|
| cve_hdf5 | https://github.com/HDFGroup/cve_hdf5 | `3fd1f5ae3869` |
|
||||||
|
| netcdf-c | https://github.com/Unidata/netcdf-c | `beb7b9585273` |
|
||||||
|
| NCAS-CMS_pyfive | https://github.com/NCAS-CMS/pyfive | `8cf07b874913` |
|
||||||
|
| usnistgov_h5wasm | https://github.com/usnistgov/h5wasm | `02f6336527d2` |
|
||||||
|
| netcdf4-python | https://github.com/Unidata/netcdf4-python | `6e67576d39ae` |
|
||||||
|
| xarray-data | https://github.com/pydata/xarray-data | `a35297e9da2c` |
|
||||||
|
| h5py_data | https://github.com/h5py/h5py (`h5py/tests/data_files`) | `b2f0347c4200` |
|
||||||
|
|
||||||
|
## Panics, hangs, crashes, out-of-memory
|
||||||
|
|
||||||
|
None.
|
||||||
|
|
||||||
|
## Our-error root causes
|
||||||
|
|
||||||
|
Grouped by normalised error message. *files* counts files whose class this cause affects.
|
||||||
|
|
||||||
|
| files | objects | error | examples |
|
||||||
|
|---:|---:|---|---|
|
||||||
|
| 6 | 6 | `UnsupportedFilter(N)` | `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc.h5`, `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_blosc2.h5`, `hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bshuf.h5` (+3 more) |
|
||||||
|
| 3 | 3 | `DataSizeMismatch { expected: N, actual: N }` | `cve_hdf5/cvefiles/cve-2020-18494.h5`, `cve_hdf5/cvefiles/cve-2024-32623.h5`, `cve_hdf5/cvefiles/cve-2025-2309.h5` |
|
||||||
|
| 2 | 2 | `ChunkedReadError("…")` | `cve_hdf5/cvefiles/cve-2025-2308.h5`, `hdf5/test/testfiles/bad_nbit_parms_walk.h5` |
|
||||||
|
| 1 | 1 | `UnexpectedEof { expected: N, available: N }` | `cve_hdf5/cvefiles/cve-2019-9151.h5` |
|
||||||
|
| 1 | 1 | `MissingMessage(Dataspace)` | `cve_hdf5/cvefiles/cve-2024-33874.h5` |
|
||||||
|
| 1 | 1 | `InvalidObjectHeaderVersion(N)` | `hdf5/tools/test/testfiles/h5clear_mdc_image.h5` |
|
||||||
|
|
||||||
|
## Mismatch root causes
|
||||||
|
|
||||||
|
| files | objects | cause | examples |
|
||||||
|
|---:|---:|---|---|
|
||||||
|
| 13 | 14 | `missing-object` | `cve_hdf5/cvefiles/cve-2019-8397.h5`, `cve_hdf5/cvefiles/cve-2019-8398.h5`, `cve_hdf5/cvefiles/cve-2021-46243.h5` (+10 more) |
|
||||||
|
| 3 | 7 | `extra-attr` | `cve_hdf5/cvefiles/cve-2018-17438`, `cve_hdf5/cvefiles/cve-2018-17439`, `cve_hdf5/cvefiles/cve-2024-33874.h5` |
|
||||||
|
| 3 | 6 | `extra-object` | `cve_hdf5/cvefiles/cve-2021-46244.h5`, `hdf5/tools/test/testfiles/h5clear_fsm_persist_less.h5`, `hdf5/tools/test/testfiles/h5stat_err_refcount.h5` |
|
||||||
|
| 1 | 1 | `attr-values: ours=vlen(>u8) h5py=object layout=- filters=-` | `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5` |
|
||||||
|
| 1 | 1 | `values: ours=<f4 h5py=float32 layout=chunked filters=-` | `cve_hdf5/cvefiles/cve-2025-44904.h5` |
|
||||||
|
| 1 | 1 | `values: ours=>i2 h5py=>i2 layout=chunked filters=[6]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||||
|
| 1 | 1 | `values: ours=>f4 h5py=>f4 layout=chunked filters=[2]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||||
|
| 1 | 1 | `values: ours=<f4 h5py=float32 layout=chunked filters=[2]` | `cve_hdf5/cvefiles/cve-2025-44905.h5` |
|
||||||
|
| 1 | 1 | `values: ours=((<i4)[6, 3])[4] h5py=(('<i4', (6, 3)), (4,)) layout=contiguous filters=-` | `hdf5/tools/test/testfiles/tarray3.h5` |
|
||||||
|
| 1 | 1 | `values: ours=vlen({r:>f4,i:>f4}8) h5py=object layout=contiguous filters=-` | `hdf5/tools/test/testfiles/tcomplex_be.h5` |
|
||||||
|
|
||||||
|
## CVE corpus: clawhdf5 vs h5dump vs h5py
|
||||||
|
|
||||||
|
The 147 files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for
|
||||||
|
published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object
|
||||||
|
errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so
|
||||||
|
its read/error split is not comparable with the other two rows; the panic, crash, hang and oom
|
||||||
|
columns are.
|
||||||
|
|
||||||
|
| tool | read | error | panic | crash | hang | oom |
|
||||||
|
|---|---:|---:|---:|---:|---:|---:|
|
||||||
|
| clawhdf5 | 142 | 5 | 0 | 0 | 0 | 0 |
|
||||||
|
| h5dump 1.14.6 | 16 | 129 | 0 | 2 | 0 | 0 |
|
||||||
|
| h5py 3.16.0 / HDF5 2.0.0 | 115 | 31 | 0 | 1 | 0 | 0 |
|
||||||
|
|
||||||
|
<details><summary>Per-file outcomes</summary>
|
||||||
|
|
||||||
|
| file | h5dump | h5py | clawhdf5 | class |
|
||||||
|
|---|---|---|---|---|
|
||||||
|
| cvefiles/cve-2016-4330.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4331.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2016-4332-mtime-new.h5 | error exit | read 25 obj, 1 errors | read 25 obj | ok |
|
||||||
|
| cvefiles/cve-2016-4332-mtime.h5 | error exit | read 4 obj, 3 errors | read 4 obj | ok |
|
||||||
|
| cvefiles/cve-2016-4332-stab.h5 | error exit | open error | read 65 obj | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2016-4333.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17505.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17506.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17507.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2017-17508.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2017-17509.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11202.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11203.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11204.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2018-11205.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2018-11206-new.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11206-old.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-11207.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13866.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-13867.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13868.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13869.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13870.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13871.h5 | error exit | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2018-13872.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13873.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-13874.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-13875.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-13876.h5 | error exit | open error | read 2 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-14031.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14033.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14034.h5 | error exit | read 1 obj, 2 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-14035.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-14460.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2018-15671.h5 | ok | read 1 obj | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-15672.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-16438.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||||
|
| cvefiles/cve-2018-17233.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17234.h5 | error exit | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17237.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17432.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17433 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-17434.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17435.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17436 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2018-17437.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2018-17438 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | mismatch |
|
||||||
|
| cvefiles/cve-2018-17439 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | mismatch |
|
||||||
|
| cvefiles/cve-2019-8396.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2019-8397.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||||
|
| cvefiles/cve-2019-8398.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||||
|
| cvefiles/cve-2019-9151.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 2 errors | our-error |
|
||||||
|
| cvefiles/cve-2019-9152.h5 | error exit | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2020-10809 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-10810.h5 | error exit | open error | read 2 obj | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-10811.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2020-10812.h5 | error exit | open error | read 2 obj | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2020-18232.h5 | error exit | read 3 obj, 2 errors | read 3 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2020-18494.h5 | ok | read 2 obj | read 2 obj, 1 errors | our-error |
|
||||||
|
| cvefiles/cve-2021-36977.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-37501.h5 | error exit | read 18 obj, 1 errors | read 18 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-45829.h5 | error exit | read 1 obj, 2 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-45830.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2021-45833.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2021-46242.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2021-46243.h5 | error exit | read 3 obj, 2 errors | read 2 obj, 1 errors | mismatch |
|
||||||
|
| cvefiles/cve-2021-46244.h5 | error exit | read 2 obj, 1 errors | read 6 obj, 3 errors | mismatch |
|
||||||
|
| cvefiles/cve-2024-29157.h5 | error exit | read 4 obj, 7 errors | read 4 obj, 7 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29158.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29159.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29160.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29161.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 2 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29162.h5 | error exit | read 17 obj, 4 errors | read 17 obj, 3 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29163.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29164.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-29165.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-29166.h5 | error exit | read 17 obj, 2 errors | read 17 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32605.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32606.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32607-1.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32607-2.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32608.h5 | error exit | read 6 obj, 1 errors | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32609.h5 | error exit | SIGSEGV | read 3 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2024-32610.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32611.h5 | ok | read 6 obj | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32612.h5 | ok | read 3 obj | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32613.h5 | error exit | read 7 obj, 1 errors | read 7 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32614.h5 | error exit | read 25 obj, 2 errors | read 25 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32615.h5 | error exit | read 4 obj, 1 errors | read 4 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32616.h5 | error exit | read 10 obj, 7 errors | read 10 obj, 5 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32617.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32618.h5 | error exit | read 4 obj, 2 errors | read 3 obj | mismatch |
|
||||||
|
| cvefiles/cve-2024-32619.h5 | error exit | read 3 obj, 2 errors | read 3 obj | ok |
|
||||||
|
| cvefiles/cve-2024-32620.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32621.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32622.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-32623.h5 | ok | read 6 obj | read 6 obj, 1 errors | our-error |
|
||||||
|
| cvefiles/cve-2024-32624.h5 | error exit | read 6 obj, 1 errors | read 6 obj | ok |
|
||||||
|
| cvefiles/cve-2024-33873.h5 | error exit | read 4 obj, 1 errors | read 4 obj | ok |
|
||||||
|
| cvefiles/cve-2024-33874.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | our-error |
|
||||||
|
| cvefiles/cve-2024-33875.h5 | ok | read 2 obj | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2024-33876.h5 | ok | read 3 obj, 1 errors | read 3 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2024-33877.h5 | error exit | read 8 obj, 1 errors | read 8 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2153.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2308.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 2 errors | our-error |
|
||||||
|
| cvefiles/cve-2025-2309.h5 | ok | read 6 obj, 1 errors | read 6 obj, 1 errors | our-error |
|
||||||
|
| cvefiles/cve-2025-2310.h5 | error exit | read 24 obj, 8 errors | read 24 obj, 8 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2912.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2913.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2914.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2915.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2923.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-2924.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2925.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-2926.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-44904.h5 | error exit | read 25 obj, 1 errors | read 25 obj, 1 errors | mismatch |
|
||||||
|
| cvefiles/cve-2025-44905.h5 | error exit | read 25 obj, 3 errors | read 25 obj, 3 errors | mismatch |
|
||||||
|
| cvefiles/cve-2025-6269-1.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-2.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-3.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6269-4.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6270-1.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6270-2.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6270-3.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6516.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6750.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6816.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6817.h5 | error exit | open error | read 1 obj | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6818.h5 | error exit | open error | read 1 obj | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6856.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-6857.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-6858.h5 | SIGSEGV | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-7067.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2025-7068.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2025-7069.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| cvefiles/cve-2026-26200.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| cvefiles/cve-2026-34734.h5 | error exit | read 2 obj, 1 errors | read 2 obj | ok |
|
||||||
|
| cvefiles/cve-2026-92627.h5 | error exit | read 2 obj, 1 errors | read 2 obj, 1 errors | ok |
|
||||||
|
| cvefiles/unknown-1.h5 | error exit | read 11 obj, 1 errors | read 11 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4431-poc-03.h5 | error exit | read 1 obj | read 1 obj | ok |
|
||||||
|
| fuzzerfiles/gh-4432-poc-05.h5 | SIGSEGV | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4433-poc-08.h5 | error exit | read 1 obj, 1 errors | read 1 obj | ok |
|
||||||
|
| fuzzerfiles/gh-4434-poc-09.h5 | error exit | open error | read 1 obj, 1 errors | h5py-cannot-read |
|
||||||
|
| fuzzerfiles/gh-4435-poc-10.h5 | error exit | read 1 obj, 1 errors | read 1 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh-4585.h5 | error exit | open error | open error | h5py-cannot-read |
|
||||||
|
| fuzzerfiles/gh_2649_flawed.h5 | error exit | read 9 obj, 1 errors | read 9 obj, 1 errors | ok |
|
||||||
|
| fuzzerfiles/gh_2649_plain_model.h5 | ok | read 10 obj | read 10 obj | ok |
|
||||||
|
|
||||||
|
</details>
|
||||||
|
|
||||||
|
## Known not-our-bug
|
||||||
|
|
||||||
|
- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence
|
||||||
|
whose base type is big-endian with the file's big-endian bytes but a native (little-endian)
|
||||||
|
numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values
|
||||||
|
clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`
|
||||||
|
reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: `NCAS-CMS_pyfive/tests/data/attr_datatypes.hdf5`, `hdf5/tools/test/testfiles/tcomplex_be.h5`.
|
||||||
|
- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose
|
||||||
|
bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a
|
||||||
|
bit offset / reduced precision into the plain numpy type of the same size. The probe
|
||||||
|
compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared
|
||||||
|
raw bytes, which reported every N-Bit float dataset as a mismatch).
|
||||||
|
- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size
|
||||||
|
(FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not
|
||||||
|
compared (shape and presence still are): dataset file type size 1 -> numpy float16 (2) (15x), attr file type size 1 -> numpy float16 (2) (15x), dataset file type size 2 -> numpy float32 (4) (2x), dataset file type size 8 -> numpy float128 (16) (1x), dataset file type size 12 -> numpy float128 (16) (1x), attr file type size 2 -> numpy float32 (4) (1x), dataset file type size 2 -> numpy >f4 (4) (1x), attr file type size 2 -> numpy >f4 (4) (1x).
|
||||||
|
- **References** are compared by presence only (`R`), not by target.
|
||||||
|
|
||||||
|
## Objects h5py fails on but clawhdf5 reads
|
||||||
|
|
||||||
|
- 19 x `KeyError: '…'`
|
||||||
|
- 19 x `OSError: Can't synchronously read data (no appropriate function for conversion path)`
|
||||||
|
- 1 x `TypeError: unhandled dtype kind M (dtype('…'))`
|
||||||
|
- 1 x `OSError: Can't synchronously read data (bad coordinate offset)`
|
||||||
|
- 1 x `TypeError: No NumPy equivalent for TypeTimeID exists`
|
||||||
|
- 1 x `KeyError: "…"`
|
||||||
|
- 1 x `ValueError: Insufficient precision in available types to represent (N, N, N, N, N)`
|
||||||
|
|
||||||
|
## Reproduce
|
||||||
|
|
||||||
|
```sh
|
||||||
|
# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git
|
||||||
|
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for
|
||||||
|
every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.
|
||||||
|
`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)
|
||||||
|
must keep; `conformance/run.sh --update-baseline` rewrites it.
|
||||||
@@ -696,7 +696,7 @@ stores keep their setting. Opt out with `float16 = false` or
|
|||||||
| `fast-checksum` | no | crc32fast-accelerated checksums |
|
| `fast-checksum` | no | crc32fast-accelerated checksums |
|
||||||
| `lz4` | no | LZ4 block compression filter (id 32004) |
|
| `lz4` | no | LZ4 block compression filter (id 32004) |
|
||||||
| `zstd` | no | Zstandard compression filter (id 32015) |
|
| `zstd` | no | Zstandard compression filter (id 32015) |
|
||||||
| `pcodec` | no | Pcodec lossless numerical codec (id 32023, via `pco` crate) |
|
| `pcodec` | no | Pcodec lossless numerical codec (via `pco` crate). Private, unregistered filter id 480: **only clawhdf5 can read these datasets** (h5py/libhdf5 cannot). Files from clawhdf5 <= 2.7.0 used id 32023, which is registered to Granular BitRound; they still read. |
|
||||||
| `system-zlib` | no | System zlib backend for deflate (C) |
|
| `system-zlib` | no | System zlib backend for deflate (C) |
|
||||||
| `blake3_hash` | no | BLAKE3 content hashing for provenance |
|
| `blake3_hash` | no | BLAKE3 content hashing for provenance |
|
||||||
| `szip` | no | SZIP filter (id 4) via libaec (C, through the internal `libaec-sys` crate) |
|
| `szip` | no | SZIP filter (id 4) via libaec (C, through the internal `libaec-sys` crate) |
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
/.cache/
|
||||||
|
# pin the probe's dependencies (the workspace lock is not committed)
|
||||||
|
!/probe/Cargo.lock
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
# Conformance sweep
|
||||||
|
|
||||||
|
Reads every HDF5 file of eight public corpora with clawhdf5 and with
|
||||||
|
h5py/libhdf5, compares the two readings object by object, and writes
|
||||||
|
[`CONFORMANCE.md`](../CONFORMANCE.md).
|
||||||
|
|
||||||
|
```sh
|
||||||
|
CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh # ~30 s once the corpus is cached
|
||||||
|
conformance/run.sh --update-baseline # after an intended change in results
|
||||||
|
```
|
||||||
|
|
||||||
|
Needs Rust, `git`, `h5dump` (Debian/Ubuntu `hdf5-tools`), `libaec` (for the
|
||||||
|
probe's `szip` feature; `libaec-dev`), and a Python with the packages in
|
||||||
|
`requirements.txt`. The first run downloads about 450 MB of sparse checkouts.
|
||||||
|
|
||||||
|
| file | role |
|
||||||
|
|---|---|
|
||||||
|
| `corpus.txt` | the corpora: git URL, pinned commit, swept root, sparse-checkout patterns |
|
||||||
|
| `fetch-corpus.sh` | shallow, sparse, blob-filtered checkout of each pinned commit into `.cache/src/` (gitignored); no-op when already there |
|
||||||
|
| `list_files.py` | which files are probed (HDF5/netCDF-4 extensions minus netCDF classic, plus the CVE reproducers) |
|
||||||
|
| `probe/` | the clawhdf5 side: a standalone crate (outside the workspace, so `cargo test --workspace` never builds it) that walks a file with `clawhdf5-format` and prints canonical JSON |
|
||||||
|
| `ref.py` | the h5py side: the same JSON from h5py |
|
||||||
|
| `run_one.sh` | runs both sides on one file (and `h5dump` on the CVE corpus) under a timeout and an address-space limit |
|
||||||
|
| `compare.py` | classifies each file (ok / our-error / mismatch / h5py-cannot-read / panic / hang / crash / oom) and groups root causes |
|
||||||
|
| `report.py` | writes `CONFORMANCE.md` |
|
||||||
|
| `check.py` | the gate: fails on any panic/hang/crash/oom, on an ok count below `baseline.json`, or on a baseline-ok file that is no longer ok |
|
||||||
|
| `baseline.json` | the ok files the gate holds the line on |
|
||||||
|
| `requirements.txt` | pinned h5py / numpy / hdf5plugin / netCDF4 |
|
||||||
|
|
||||||
|
Results for every file (both sides' JSON and stderr, `results.csv`,
|
||||||
|
`results.json`, `summary.md`) are left in `.cache/results/`.
|
||||||
|
|
||||||
|
The nightly job is `.gitea/workflows/conformance.yml`; it prints the report
|
||||||
|
into the job log.
|
||||||
|
|
||||||
|
The canonical value encoding both sides hash is documented at the top of
|
||||||
|
`probe/src/main.rs`. Values are compared as libhdf5 presents them: a float
|
||||||
|
with a non-IEEE bit layout (N-Bit) or an integer with a bit offset is compared
|
||||||
|
as the converted number, not as raw file bytes.
|
||||||
@@ -0,0 +1,618 @@
|
|||||||
|
{
|
||||||
|
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||||
|
"commit": "10d1029ead524e2fe64c2cd7f61b28067d9e449c",
|
||||||
|
"date": "2026-09-26 03:46 UTC",
|
||||||
|
"reference": "h5py 3.16.0 / HDF5 2.0.0",
|
||||||
|
"files": 697,
|
||||||
|
"ok": 569,
|
||||||
|
"counts": {
|
||||||
|
"h5py-cannot-read": 92,
|
||||||
|
"mismatch": 22,
|
||||||
|
"ok": 569,
|
||||||
|
"our-error": 14
|
||||||
|
},
|
||||||
|
"per_corpus": {
|
||||||
|
"NCAS-CMS_pyfive": {
|
||||||
|
"mismatch": 1,
|
||||||
|
"ok": 32
|
||||||
|
},
|
||||||
|
"cve_hdf5": {
|
||||||
|
"h5py-cannot-read": 32,
|
||||||
|
"mismatch": 9,
|
||||||
|
"ok": 100,
|
||||||
|
"our-error": 6
|
||||||
|
},
|
||||||
|
"h5py_data": {
|
||||||
|
"ok": 4
|
||||||
|
},
|
||||||
|
"hdf5": {
|
||||||
|
"h5py-cannot-read": 60,
|
||||||
|
"mismatch": 12,
|
||||||
|
"ok": 386,
|
||||||
|
"our-error": 8
|
||||||
|
},
|
||||||
|
"netcdf-c": {
|
||||||
|
"ok": 20
|
||||||
|
},
|
||||||
|
"netcdf4-python": {
|
||||||
|
"ok": 18
|
||||||
|
},
|
||||||
|
"usnistgov_h5wasm": {
|
||||||
|
"ok": 5
|
||||||
|
},
|
||||||
|
"xarray-data": {
|
||||||
|
"ok": 4
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"ok_files": [
|
||||||
|
"NCAS-CMS_pyfive/tests/compact.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/btreev2.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/chunked.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/cmip_bad_eg.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/compressed.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/compressed_v1.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dataset_datatypes.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dataset_multidim.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/dim_scales.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/earliest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_h5variable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_variable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enum_variable.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/enums_from_netcdf.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fillvalue_earliest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fillvalue_latest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/filter_pipeline_v2.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fletcher32.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/fractal_heap_no_mci_rlat.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/groups.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/h5netcdf_test.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_A.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_A_contiguous.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/issue23_B.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/latest.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/netcdf4_classic.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/new_style_groups.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/noy_AERmonZ_UKESM1-0-LL_piControl_r1i1p1f2_gnz_200001-200012.nc",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/references.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/data/resizable.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/opaque_datetime.hdf5",
|
||||||
|
"NCAS-CMS_pyfive/tests/opaque_fixed.hdf5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4330.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4331.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4332-mtime-new.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4332-mtime.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2016-4333.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17505.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17506.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17507.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17508.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2017-17509.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11202.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11203.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11204.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11205.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11206-new.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11206-old.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-11207.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13867.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13868.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13869.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13870.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13871.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13872.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13873.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-13875.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14031.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14033.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14034.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14035.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-14460.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-15671.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-15672.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-16438.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17233.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17234.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17237.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17432.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17434.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17435.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2018-17437.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-8396.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2019-9152.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-10811.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2020-18232.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-36977.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-37501.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-45829.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2021-45833.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29157.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29158.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29159.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29160.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29161.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29162.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29163.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29164.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29165.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-29166.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32605.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32606.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32607-1.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32607-2.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32608.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32610.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32611.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32612.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32613.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32614.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32615.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32616.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32617.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32619.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32620.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32621.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32622.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-32624.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33873.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33875.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33876.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2024-33877.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2310.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2924.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-2925.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-1.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-2.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-3.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6269-4.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6516.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-6857.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2025-7067.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-26200.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-34734.h5",
|
||||||
|
"cve_hdf5/cvefiles/cve-2026-92627.h5",
|
||||||
|
"cve_hdf5/cvefiles/unknown-1.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4431-poc-03.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4432-poc-05.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4433-poc-08.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh-4435-poc-10.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh_2649_flawed.h5",
|
||||||
|
"cve_hdf5/fuzzerfiles/gh_2649_plain_model.h5",
|
||||||
|
"h5py_data/compound-dtype-complex.h5",
|
||||||
|
"h5py_data/vlen_string_dset.h5",
|
||||||
|
"h5py_data/vlen_string_dset_utc.h5",
|
||||||
|
"h5py_data/vlen_string_s390x.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_bitgroom.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_granularbr.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_jpeg.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5FLT/tfiles/h5ex_d_zstd.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/16/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/C/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_traverse.h5",
|
||||||
|
"hdf5/HDF5Examples/FORTRAN/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/110/h5ex_g_visit.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_iterate.h5",
|
||||||
|
"hdf5/HDF5Examples/JAVA/compat/H5G/h5ex_g_visit.h5",
|
||||||
|
"hdf5/c++/test/th5s.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be_new_ref-32bit.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_be_new_ref.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_le.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ds_le_new_ref.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_ld.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_be.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_cray.h5",
|
||||||
|
"hdf5/hl/test/testfiles/test_table_le.h5",
|
||||||
|
"hdf5/test/testfiles/aggr.h5",
|
||||||
|
"hdf5/test/testfiles/bad_chunk_ndims.h5",
|
||||||
|
"hdf5/test/testfiles/bad_compound.h5",
|
||||||
|
"hdf5/test/testfiles/bad_offset.h5",
|
||||||
|
"hdf5/test/testfiles/be_data.h5",
|
||||||
|
"hdf5/test/testfiles/be_extlink1.h5",
|
||||||
|
"hdf5/test/testfiles/be_extlink2.h5",
|
||||||
|
"hdf5/test/testfiles/btree_idx_1_6.h5",
|
||||||
|
"hdf5/test/testfiles/btree_idx_1_8.h5",
|
||||||
|
"hdf5/test/testfiles/charsets.h5",
|
||||||
|
"hdf5/test/testfiles/corrupt_stab_msg.h5",
|
||||||
|
"hdf5/test/testfiles/deflate.h5",
|
||||||
|
"hdf5/test/testfiles/file_image_core_test.h5",
|
||||||
|
"hdf5/test/testfiles/filespace_1_6.h5",
|
||||||
|
"hdf5/test/testfiles/filespace_1_8.h5",
|
||||||
|
"hdf5/test/testfiles/fill18.h5",
|
||||||
|
"hdf5/test/testfiles/fill_old.h5",
|
||||||
|
"hdf5/test/testfiles/filter_error.h5",
|
||||||
|
"hdf5/test/testfiles/fsm_aggr_nopersist.h5",
|
||||||
|
"hdf5/test/testfiles/fsm_aggr_persist.h5",
|
||||||
|
"hdf5/test/testfiles/group_old.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext1_f.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext1_i.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext2_if.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/test/testfiles/h5fc_ext_none.h5",
|
||||||
|
"hdf5/test/testfiles/le_data.h5",
|
||||||
|
"hdf5/test/testfiles/le_extlink1.h5",
|
||||||
|
"hdf5/test/testfiles/le_extlink2.h5",
|
||||||
|
"hdf5/test/testfiles/memleak_H5O_dtype_decode_helper_H5Odtype.h5",
|
||||||
|
"hdf5/test/testfiles/mergemsg.h5",
|
||||||
|
"hdf5/test/testfiles/noencoder.h5",
|
||||||
|
"hdf5/test/testfiles/none.h5",
|
||||||
|
"hdf5/test/testfiles/paged_nopersist.h5",
|
||||||
|
"hdf5/test/testfiles/paged_persist.h5",
|
||||||
|
"hdf5/test/testfiles/specmetaread.h5",
|
||||||
|
"hdf5/test/testfiles/tarrold.h5",
|
||||||
|
"hdf5/test/testfiles/tbad_msg_count.h5",
|
||||||
|
"hdf5/test/testfiles/tbogus.h5",
|
||||||
|
"hdf5/test/testfiles/test_filters_be.h5",
|
||||||
|
"hdf5/test/testfiles/test_filters_le.h5",
|
||||||
|
"hdf5/test/testfiles/th5s.h5",
|
||||||
|
"hdf5/test/testfiles/tlayouto.h5",
|
||||||
|
"hdf5/test/testfiles/tmisc38a.h5",
|
||||||
|
"hdf5/test/testfiles/tmisc38b.h5",
|
||||||
|
"hdf5/test/testfiles/tmtimen.h5",
|
||||||
|
"hdf5/test/testfiles/tmtimeo.h5",
|
||||||
|
"hdf5/test/testfiles/tnullspace.h5",
|
||||||
|
"hdf5/test/testfiles/tsizeslheap.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bigendian/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binfp64.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binin8w.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binuin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/binuin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/bounds_latest_latest.h5",
|
||||||
|
"hdf5/tools/test/testfiles/charsets.h5",
|
||||||
|
"hdf5/tools/test/testfiles/compounds_array_vlen1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/compounds_array_vlen2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/err_attr_dspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/file_space.h5",
|
||||||
|
"hdf5/tools/test/testfiles/filter_fail.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_equal.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_noclose.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_equal.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_fsm_persist_user_less.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_sec2_v0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5clear_sec2_v2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_extlinks_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_extlinks_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copy_ref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copytst.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5copytst_new.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr_v_level1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_attr_v_level2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_basic1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_basic2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_comp_vl_strs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_danglelinks1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_danglelinks2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dset_zero_dim_size2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_dtypes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_empty.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_enum_invalid_values.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_eps1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_eps2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude1-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude1-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude2-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude2-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude3-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_exclude3-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_ext2softlink_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_ext2softlink_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_extlink_src.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_extlink_trg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_hyper1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_hyper2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_linked_softlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_links.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_dset_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_dset_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_onion_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_softlinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_strings1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5diff_strings2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_edge_v3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_err_level.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_i.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext1_s.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_if.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_is.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_ext_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5fc_non_v3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_CVE-2018-14460.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_CVE-2018-17432.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_aggr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_attr_refs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_deflate.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_early.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_f32le.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_f32le_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fill.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_filters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fletcher.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_nopersist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_fsm_aggr_persist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_hlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_1d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_2d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_2d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_3d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_int32le_3d_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout.UD.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layout3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_layouto.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_named_dtypes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nbit.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_nested_8bit_enum_deflated.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_paged_nopersist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_paged_persist.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_refs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_shuffle.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_soffset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_szip.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_uint8be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5repack_uint8be_ex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_old_fill.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_err_old_layout.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_filters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_idx.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_newgrat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_threshold.h5",
|
||||||
|
"hdf5/tools/test/testfiles/h5stat_tsohm.h5",
|
||||||
|
"hdf5/tools/test/testfiles/mod_h5clear_mdc_image.h5",
|
||||||
|
"hdf5/tools/test/testfiles/non_comparables1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/non_comparables2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_i.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext1_s.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_if.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_is.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext2_sf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext3_isf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/old_h5fc_ext_none.h5",
|
||||||
|
"hdf5/tools/test/testfiles/packedbits.h5",
|
||||||
|
"hdf5/tools/test/testfiles/t128bit_float.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE-2021-37501_attr_decode.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_new.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tCVE_2018_11206_fill_old.h5",
|
||||||
|
"hdf5/tools/test/testfiles/taindices.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray1_big.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray5.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tarray8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattr4_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tattrreg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbfloat16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbfloat16_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbigdims.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbinary.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tbitnopaque.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tchar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdintarray.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdints.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcmpdintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcomplex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound_complex.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tcompound_complex2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdatareg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tdset_idx.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tempty.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinkfar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinksrc.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textlinktar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/textpfe.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfcontents2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfilters.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat16_be.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat6.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloat8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfloatsattrs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfpformat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tfvalues.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgroup.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgrp_comments.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tgrpnullspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/thlink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/thyperslab.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintascii.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tints4dims.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintsattrs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tintsnodata.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tlarge_objname.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tldouble.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tldouble_scalar.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tlonglinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tloop.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnamed_dtype_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnestedcmpddt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnestedcomp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tno-subset.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tnullspace.h5",
|
||||||
|
"hdf5/tools/test/testfiles/torderattr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tordergr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_attr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_compat.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_ext1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_ext2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_grp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_obj.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_obj_del.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_param.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_reg.h5",
|
||||||
|
"hdf5/tools/test/testfiles/trefer_reg_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tsaf.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarattrintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarintattrsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarintsize.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tscalarstring.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tslink.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tsoftlinks.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_dset_1d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_dset_ext.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tst_onion_objs.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tstr3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudfilter.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tudfilter2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes4.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvldtypes5.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvlenstr_array.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvlstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/tvms.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtfp32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtfp64.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtin8.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtstr.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtuin16.h5",
|
||||||
|
"hdf5/tools/test/testfiles/txtuin32.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_e.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_f.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/1_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_e.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/2_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/3_1_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/3_2_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_1.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/4_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/5_vds.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/a.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/b.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/c.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/d.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/f-0.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/f-3.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/vds-eiger.h5",
|
||||||
|
"hdf5/tools/test/testfiles/vds/vds-percival-unlim-maxmin.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tbitfields.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tcompound2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tdset2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tenum.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/test35.nc",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tloop2.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-amp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-apos.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-gt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-lt.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-quot.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tname-sp.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tnodata.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tobjref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/topaque.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref-escapes-at.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref-escapes.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tref.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tstring-at.h5",
|
||||||
|
"hdf5/tools/test/testfiles/xml/tstring.h5",
|
||||||
|
"hdf5/tools/test/testfiles/zerodim.h5",
|
||||||
|
"netcdf-c/h5_test/ref_tst_h_compounds.h5",
|
||||||
|
"netcdf-c/h5_test/ref_tst_h_compounds2.h5",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat1.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat2.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_hdf5_compat3.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_szip.h5",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_compounds.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_dims.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_interops4.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_xplatform2_1.nc",
|
||||||
|
"netcdf-c/nc_test4/ref_tst_xplatform2_2.nc",
|
||||||
|
"netcdf-c/nc_test4/tdset.h5",
|
||||||
|
"netcdf-c/ncdump/ref_nc_test_netcdf4_4_0.nc",
|
||||||
|
"netcdf-c/ncdump/ref_no_ncproperty.nc",
|
||||||
|
"netcdf-c/ncdump/ref_provenance_v1.nc",
|
||||||
|
"netcdf-c/ncdump/ref_test_corrupt_magic.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds2.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds3.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_compounds4.nc",
|
||||||
|
"netcdf-c/ncdump/ref_tst_irish_rover.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2000.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2001.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2002.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2003.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2004.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2005.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2006.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2007.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2008.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2009.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2010.nc",
|
||||||
|
"netcdf4-python/examples/data/prmsl.2011.nc",
|
||||||
|
"netcdf4-python/examples/data/rtofs_glo_3dz_f006_6hrly_reg3.nc",
|
||||||
|
"netcdf4-python/test/20171025_2056.Cloud_Top_Height.nc",
|
||||||
|
"netcdf4-python/test/issue1152.nc",
|
||||||
|
"netcdf4-python/test/issue671.nc",
|
||||||
|
"netcdf4-python/test/issue672.nc",
|
||||||
|
"netcdf4-python/test/test_gold.nc",
|
||||||
|
"usnistgov_h5wasm/test/array.h5",
|
||||||
|
"usnistgov_h5wasm/test/compressed.h5",
|
||||||
|
"usnistgov_h5wasm/test/empty.h5",
|
||||||
|
"usnistgov_h5wasm/test/float16.h5",
|
||||||
|
"usnistgov_h5wasm/test/vlen.h5",
|
||||||
|
"xarray-data/ROMS_example.nc",
|
||||||
|
"xarray-data/basin_mask.nc",
|
||||||
|
"xarray-data/imerghh_730.hdf5",
|
||||||
|
"xarray-data/precipitation.nc4"
|
||||||
|
]
|
||||||
|
}
|
||||||
Executable
+88
@@ -0,0 +1,88 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""check.py <results_dir> <baseline.json> [--update]
|
||||||
|
|
||||||
|
The conformance gate. Fails (exit 1) when
|
||||||
|
* clawhdf5 panicked, hung, crashed or ran out of memory on any file, or
|
||||||
|
* the ok count fell below the baseline's, or
|
||||||
|
* a file the baseline lists as ok is no longer ok (even if another file
|
||||||
|
became ok and the total held).
|
||||||
|
New ok files are reported so the baseline can be raised (--update rewrites it
|
||||||
|
from the results).
|
||||||
|
"""
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
FATAL = ("panic", "hang", "crash", "oom")
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
args = [a for a in sys.argv[1:] if not a.startswith("--")]
|
||||||
|
update = "--update" in sys.argv
|
||||||
|
res_dir, base_path = args
|
||||||
|
res = json.load(open(os.path.join(res_dir, "results.json")))
|
||||||
|
rows = res["rows"]
|
||||||
|
counts = {}
|
||||||
|
per_corpus = {}
|
||||||
|
for r in rows:
|
||||||
|
counts[r["class"]] = counts.get(r["class"], 0) + 1
|
||||||
|
pc = per_corpus.setdefault(r["corpus"], {})
|
||||||
|
pc[r["class"]] = pc.get(r["class"], 0) + 1
|
||||||
|
ok_files = sorted(r["file"] for r in rows if r["class"] == "ok")
|
||||||
|
|
||||||
|
if update:
|
||||||
|
meta = {}
|
||||||
|
mp = os.path.join(res_dir, "report-meta.json")
|
||||||
|
if os.path.exists(mp):
|
||||||
|
meta = json.load(open(mp))
|
||||||
|
base = {
|
||||||
|
"comment": "conformance/run.sh fails if the ok count drops below `ok` or a file in `ok_files` stops being ok. "
|
||||||
|
"Regenerate with `conformance/run.sh --update-baseline` after an intended change.",
|
||||||
|
"commit": meta.get("commit", ""),
|
||||||
|
"date": meta.get("date", ""),
|
||||||
|
"reference": meta.get("reference", ""),
|
||||||
|
"files": len(rows),
|
||||||
|
"ok": len(ok_files),
|
||||||
|
"counts": dict(sorted(counts.items())),
|
||||||
|
"per_corpus": {k: dict(sorted(v.items())) for k, v in sorted(per_corpus.items())},
|
||||||
|
"ok_files": ok_files,
|
||||||
|
}
|
||||||
|
with open(base_path, "w") as fh:
|
||||||
|
json.dump(base, fh, indent=1)
|
||||||
|
fh.write("\n")
|
||||||
|
print(f"baseline updated: {len(ok_files)} ok of {len(rows)} files -> {base_path}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
base = json.load(open(base_path))
|
||||||
|
failures = []
|
||||||
|
fatal = [r for r in rows if r["class"] in FATAL]
|
||||||
|
for r in fatal:
|
||||||
|
failures.append(f"{r['class']}: {r['file']}: {r['ours_detail'][:200]}")
|
||||||
|
if len(ok_files) < base["ok"]:
|
||||||
|
failures.append(f"ok count dropped: {len(ok_files)} < baseline {base['ok']}")
|
||||||
|
now_ok = set(ok_files)
|
||||||
|
by_file = {r["file"]: r for r in rows}
|
||||||
|
for f in base["ok_files"]:
|
||||||
|
if f not in now_ok:
|
||||||
|
r = by_file.get(f)
|
||||||
|
why = f"now {r['class']}: {(r['ours_detail'] or r['first_issue'])[:200]}" if r else "no longer in the corpus"
|
||||||
|
failures.append(f"regressed: {f}: {why}")
|
||||||
|
gained = sorted(now_ok - set(base["ok_files"]))
|
||||||
|
|
||||||
|
print(f"conformance: {len(ok_files)} ok of {len(rows)} files (baseline {base['ok']} of {base['files']}); "
|
||||||
|
+ ", ".join(f"{k} {v}" for k, v in sorted(counts.items())))
|
||||||
|
if gained:
|
||||||
|
print(f"{len(gained)} file(s) newly ok — raise the baseline with `conformance/run.sh --update-baseline`:")
|
||||||
|
for f in gained:
|
||||||
|
print(f" + {f}")
|
||||||
|
if failures:
|
||||||
|
print(f"CONFORMANCE GATE FAILED ({len(failures)}):")
|
||||||
|
for f in failures:
|
||||||
|
print(f" - {f}")
|
||||||
|
return 1
|
||||||
|
print("conformance gate passed")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
Executable
+289
@@ -0,0 +1,289 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""compare.py <results_dir>: classify each file and group failures by root cause.
|
||||||
|
|
||||||
|
Writes <results_dir>/results.csv, results.json and summary.md.
|
||||||
|
File classes (first match wins):
|
||||||
|
hang, oom, crash, panic ours: timeout / allocation failure / signal / any panic (caught or not)
|
||||||
|
h5py-cannot-read libhdf5/h5py failed to open the file (or crashed/hung)
|
||||||
|
our-error we fail to open, list, or read something h5py reads
|
||||||
|
mismatch we read something with different shape/values, or a different object set
|
||||||
|
ok
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import csv
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
R = sys.argv[1]
|
||||||
|
RUNS = os.path.join(R, "runs")
|
||||||
|
|
||||||
|
|
||||||
|
def load(d, name):
|
||||||
|
rc_p = os.path.join(d, name + ".rc")
|
||||||
|
if not os.path.exists(rc_p):
|
||||||
|
return None
|
||||||
|
rc = int(open(rc_p).read().strip() or -1)
|
||||||
|
err = open(os.path.join(d, name + ".err"), errors="replace").read()
|
||||||
|
js = None
|
||||||
|
try:
|
||||||
|
js = json.load(open(os.path.join(d, name + ".json")))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
pass
|
||||||
|
return {"rc": rc, "err": err, "json": js}
|
||||||
|
|
||||||
|
|
||||||
|
def proc_status(p):
|
||||||
|
"""-> (status, detail)"""
|
||||||
|
if p is None:
|
||||||
|
return "missing", ""
|
||||||
|
rc, err = p["rc"], p["err"]
|
||||||
|
first_panic = next((ln for ln in err.splitlines() if ln.startswith("PANIC:") or "panicked at" in ln), "")
|
||||||
|
if rc == 0 and p["json"] is not None:
|
||||||
|
return "ok", ""
|
||||||
|
if rc == 137 or rc == 124:
|
||||||
|
return "hang", f"timeout ({os.environ.get('TMO', '20')} s)"
|
||||||
|
if "memory allocation of" in err or "MemoryError" in err or "std::bad_alloc" in err:
|
||||||
|
m = re.search(r"memory allocation of \d+ bytes failed", err)
|
||||||
|
return "oom", m.group(0) if m else "allocation failure"
|
||||||
|
if "overflowed its stack" in err:
|
||||||
|
return "crash", "stack overflow"
|
||||||
|
if rc == 101:
|
||||||
|
return "panic", first_panic or (err.strip().splitlines() or [""])[-1]
|
||||||
|
if rc in (134, 139, 136, 135, 132) or rc > 128:
|
||||||
|
sig = {134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS", 132: "SIGILL"}.get(rc, f"signal {rc - 128}")
|
||||||
|
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||||
|
return "crash", f"{sig}: {tail[0][:200] if tail else ''}"
|
||||||
|
tail = [ln for ln in err.strip().splitlines() if ln.strip()][-1:]
|
||||||
|
return "crash", f"rc={rc}: {tail[0][:200] if tail else ''}"
|
||||||
|
|
||||||
|
|
||||||
|
def norm(msg):
|
||||||
|
m = msg.split("\n")[0]
|
||||||
|
m = re.sub(r"0x[0-9a-fA-F]+", "X", m)
|
||||||
|
m = re.sub(r'"[^"]*"', '"…"', m)
|
||||||
|
m = re.sub(r"'[^']*'", "'…'", m)
|
||||||
|
m = re.sub(r"\d+", "N", m)
|
||||||
|
return m[:160]
|
||||||
|
|
||||||
|
|
||||||
|
def panic_head(msg):
|
||||||
|
"""First line + first clawhdf5 frame of a PANIC record."""
|
||||||
|
lines = msg.split("\n")
|
||||||
|
frame = next((ln.strip() for ln in lines[1:] if "clawhdf5_format" in ln), "")
|
||||||
|
return lines[0][:300], frame[:300]
|
||||||
|
|
||||||
|
|
||||||
|
def eq_shape(a, b):
|
||||||
|
return a == b
|
||||||
|
|
||||||
|
|
||||||
|
rows = []
|
||||||
|
issues_by_file = {}
|
||||||
|
root_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||||
|
mismatch_causes = collections.defaultdict(lambda: {"files": set(), "count": 0, "examples": []})
|
||||||
|
panics = []
|
||||||
|
ref_only_errors = collections.Counter()
|
||||||
|
incomparable = collections.Counter()
|
||||||
|
|
||||||
|
|
||||||
|
def add(bucket, key, file, example):
|
||||||
|
b = bucket[key]
|
||||||
|
b["count"] += 1
|
||||||
|
if file not in b["files"] and len(b["examples"]) < 6:
|
||||||
|
b["examples"].append(example)
|
||||||
|
b["files"].add(file)
|
||||||
|
|
||||||
|
|
||||||
|
files = [ln.strip() for ln in open(os.path.join(R, "files.txt")) if ln.strip()]
|
||||||
|
for rel in files:
|
||||||
|
d = os.path.join(RUNS, rel.replace("/", "__"))
|
||||||
|
corpus = rel.split("/")[0]
|
||||||
|
ours, ref = load(d, "ours"), load(d, "ref")
|
||||||
|
h5dump = load(d, "h5dump")
|
||||||
|
os_, od = proc_status(ours)
|
||||||
|
rs, rd = proc_status(ref)
|
||||||
|
oj = ours["json"] if ours else None
|
||||||
|
rj = ref["json"] if ref else None
|
||||||
|
issues = [] # (kind, detail)
|
||||||
|
caught_panics = []
|
||||||
|
|
||||||
|
def scan_err(path, what, msg):
|
||||||
|
if msg.startswith("PANIC:"):
|
||||||
|
caught_panics.append((path, what, msg))
|
||||||
|
|
||||||
|
if oj:
|
||||||
|
for o in oj.get("objects", []):
|
||||||
|
for k in ("error", "attrs_error", "list_error"):
|
||||||
|
if k in o:
|
||||||
|
scan_err(o["path"], k, o[k])
|
||||||
|
for an, av in (o.get("attrs") or {}).items():
|
||||||
|
if "error" in av:
|
||||||
|
scan_err(o["path"], f"attr {an}", av["error"])
|
||||||
|
if oj.get("open_error", "").startswith("PANIC:"):
|
||||||
|
caught_panics.append(("<open>", "open", oj["open_error"]))
|
||||||
|
|
||||||
|
ref_open_fail = rs != "ok" or (rj is not None and "open_error" in rj)
|
||||||
|
ours_open_err = oj.get("open_error") if oj else None
|
||||||
|
n_obj = n_ok = 0
|
||||||
|
if os_ == "ok" and rj and not ref_open_fail and not ours_open_err:
|
||||||
|
ro = {x["path"]: x for x in rj.get("objects", [])}
|
||||||
|
oo = {x["path"]: x for x in oj.get("objects", [])}
|
||||||
|
our_list_errors = [x for x in oo.values() if "list_error" in x]
|
||||||
|
for p in sorted(set(ro) | set(oo)):
|
||||||
|
a, b = ro.get(p), oo.get(p)
|
||||||
|
n_obj += 1
|
||||||
|
if a is None:
|
||||||
|
issues.append(("mismatch", f"extra object {p} (kind={b.get('kind')})", "extra-object", b))
|
||||||
|
continue
|
||||||
|
if b is None:
|
||||||
|
if our_list_errors:
|
||||||
|
continue # accounted for by the list_error
|
||||||
|
issues.append(("mismatch", f"missing object {p} (kind={a.get('kind')})", "missing-object", a))
|
||||||
|
continue
|
||||||
|
ok = True
|
||||||
|
if a.get("kind") != b.get("kind") and "error" not in b and "error" not in a:
|
||||||
|
issues.append(("mismatch", f"{p}: kind {a.get('kind')} vs ours {b.get('kind')}", "kind", b))
|
||||||
|
ok = False
|
||||||
|
for k in ("error", "list_error", "attrs_error"):
|
||||||
|
if k in b and k not in a:
|
||||||
|
issues.append(("our-error", f"{p}: {k}: {b[k]}", b[k], b))
|
||||||
|
ok = False
|
||||||
|
elif k in a and k not in b and k == "error":
|
||||||
|
ref_only_errors[norm(a[k])] += 1
|
||||||
|
if a.get("kind") == "dataset" and "error" not in a and "error" not in b:
|
||||||
|
if "skipped" in a or "skipped" in b:
|
||||||
|
pass
|
||||||
|
elif a.get("converted"):
|
||||||
|
incomparable[f"dataset {a['converted']}"] += 1
|
||||||
|
elif a.get("shape") != b.get("shape"):
|
||||||
|
issues.append(("mismatch", f"{p}: shape {a.get('shape')} vs ours {b.get('shape')}", "shape", b))
|
||||||
|
ok = False
|
||||||
|
elif a.get("hash") != b.get("hash"):
|
||||||
|
issues.append(("mismatch", f"{p}: values differ (h5py {a.get('dtype')} vs ours {b.get('dtype')})", "values", b | {"ref_head": a.get("head"), "ref_dtype": a.get("dtype")}))
|
||||||
|
ok = False
|
||||||
|
ra, oa = a.get("attrs") or {}, b.get("attrs") or {}
|
||||||
|
if "attrs_error" not in b and "attrs_error" not in a:
|
||||||
|
for an in sorted(set(ra) | set(oa)):
|
||||||
|
x, y = ra.get(an), oa.get(an)
|
||||||
|
if x is None:
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: extra attribute", "extra-attr", y or {}))
|
||||||
|
elif y is None:
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: missing attribute", "missing-attr", x))
|
||||||
|
elif "error" in y and "error" not in x:
|
||||||
|
issues.append(("our-error", f"{p}@{an}: {y['error']}", y["error"], y))
|
||||||
|
elif "error" in x:
|
||||||
|
continue
|
||||||
|
elif x.get("converted"):
|
||||||
|
incomparable[f"attr {x['converted']}"] += 1
|
||||||
|
elif x.get("shape") != y.get("shape"):
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: attr shape {x.get('shape')} vs ours {y.get('shape')}", "attr-shape", y | {"ref_dtype": x.get("dtype")}))
|
||||||
|
elif x.get("hash") != y.get("hash"):
|
||||||
|
issues.append(("mismatch", f"{p}@{an}: attr values differ (h5py {x.get('dtype')} vs ours {y.get('dtype')})", "attr-values", y | {"ref_head": x.get("head"), "ref_dtype": x.get("dtype")}))
|
||||||
|
if ok:
|
||||||
|
n_ok += 1
|
||||||
|
|
||||||
|
# classify
|
||||||
|
if os_ in ("hang", "oom", "crash", "panic"):
|
||||||
|
cls = os_
|
||||||
|
elif caught_panics:
|
||||||
|
cls = "panic"
|
||||||
|
elif ref_open_fail:
|
||||||
|
cls = "h5py-cannot-read"
|
||||||
|
elif ours_open_err:
|
||||||
|
cls = "our-error"
|
||||||
|
issues.append(("our-error", f"open: {ours_open_err}", ours_open_err, {}))
|
||||||
|
elif any(i[0] == "our-error" for i in issues):
|
||||||
|
cls = "our-error"
|
||||||
|
elif issues:
|
||||||
|
cls = "mismatch"
|
||||||
|
else:
|
||||||
|
cls = "ok"
|
||||||
|
|
||||||
|
if os_ in ("hang", "oom", "crash", "panic") or caught_panics:
|
||||||
|
panics.append({
|
||||||
|
"file": rel, "class": cls, "detail": od,
|
||||||
|
"stderr": (ours["err"] if ours else "")[:3000],
|
||||||
|
"caught": [(p, w, m[:2500]) for p, w, m in caught_panics[:3]],
|
||||||
|
"n_caught": len(caught_panics),
|
||||||
|
})
|
||||||
|
for kind, detail, key, rec in issues:
|
||||||
|
if kind == "our-error":
|
||||||
|
add(root_causes, norm(key), rel, detail[:300])
|
||||||
|
else:
|
||||||
|
if key in ("values", "attr-values", "shape", "attr-shape"):
|
||||||
|
mk = f"{key}: ours={rec.get('dtype')} h5py={rec.get('ref_dtype')} layout={rec.get('layout','-')} filters={rec.get('filters','-')}"
|
||||||
|
else:
|
||||||
|
mk = key
|
||||||
|
add(mismatch_causes, mk, rel, detail[:300] + (f" | ref_head={rec.get('ref_head')} our_head={rec.get('head')}" if rec.get("ref_head") else ""))
|
||||||
|
ref_detail = rd if rs != "ok" else ((rj or {}).get("open_error") or "")
|
||||||
|
h5d = ""
|
||||||
|
if h5dump:
|
||||||
|
rc = h5dump["rc"]
|
||||||
|
h5d = {0: "ok", 1: "error", 137: "hang", 124: "hang", 134: "SIGABRT", 139: "SIGSEGV", 136: "SIGFPE", 135: "SIGBUS"}.get(rc, f"rc={rc}")
|
||||||
|
if "memory allocation" in h5dump["err"] or "Cannot allocate" in h5dump["err"]:
|
||||||
|
h5d += "(oom)"
|
||||||
|
rows.append({
|
||||||
|
"file": rel, "corpus": corpus, "class": cls,
|
||||||
|
"ours": os_ if os_ != "ok" else ("open-error" if ours_open_err else ("panic" if caught_panics else "ok")),
|
||||||
|
"ours_detail": (od or ours_open_err or (caught_panics[0][2].split("\n")[0] if caught_panics else ""))[:300],
|
||||||
|
"ref": rs if rs != "ok" else ("open-error" if (rj or {}).get("open_error") else "ok"),
|
||||||
|
"ref_detail": ref_detail[:300],
|
||||||
|
"h5dump_1_14_6": h5d,
|
||||||
|
"h5dump_detail": ([ln for ln in h5dump["err"].splitlines() if ln.strip()][-1:] or [""])[0][:200] if h5dump else "",
|
||||||
|
"objects": n_obj, "objects_ok": n_ok,
|
||||||
|
"issues": len(issues), "first_issue": issues[0][1][:300] if issues else "",
|
||||||
|
"superblock": (oj or {}).get("superblock_version", ""),
|
||||||
|
})
|
||||||
|
# the first issues of each file, for report.py's known-cause matching
|
||||||
|
issues_by_file[rel] = [
|
||||||
|
{"kind": k, "key": key, "detail": det[:300], "ours_dtype": rec.get("dtype"), "ref_dtype": rec.get("ref_dtype")}
|
||||||
|
for k, det, key, rec in issues[:50]
|
||||||
|
]
|
||||||
|
|
||||||
|
with open(os.path.join(R, "results.csv"), "w", newline="") as fh:
|
||||||
|
w = csv.DictWriter(fh, fieldnames=list(rows[0].keys()))
|
||||||
|
w.writeheader()
|
||||||
|
w.writerows(rows)
|
||||||
|
|
||||||
|
|
||||||
|
def ser(b):
|
||||||
|
return {k: {"files": len(v["files"]), "count": v["count"], "examples": v["examples"], "file_list": sorted(v["files"])} for k, v in sorted(b.items(), key=lambda kv: -len(kv[1]["files"]))}
|
||||||
|
|
||||||
|
|
||||||
|
json.dump({"rows": rows, "issues": issues_by_file, "root_causes": ser(root_causes), "mismatch_causes": ser(mismatch_causes),
|
||||||
|
"panics": panics, "incomparable": incomparable.most_common(), "ref_only_errors": ref_only_errors.most_common()},
|
||||||
|
open(os.path.join(R, "results.json"), "w"), indent=1)
|
||||||
|
|
||||||
|
classes = ["ok", "our-error", "mismatch", "h5py-cannot-read", "hang", "panic", "crash", "oom"]
|
||||||
|
by_corpus = collections.defaultdict(collections.Counter)
|
||||||
|
for r in rows:
|
||||||
|
by_corpus[r["corpus"]][r["class"]] += 1
|
||||||
|
by_corpus["ALL"][r["class"]] += 1
|
||||||
|
lines = ["# Conformance sweep summary", "", "| corpus | files | " + " | ".join(classes) + " |", "|---" * (len(classes) + 2) + "|"]
|
||||||
|
for c in sorted(by_corpus, key=lambda k: (k == "ALL", k)):
|
||||||
|
cnt = by_corpus[c]
|
||||||
|
lines.append(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in classes) + " |")
|
||||||
|
lines += ["", "## Panics / hangs / crashes / OOM", ""]
|
||||||
|
for p in panics:
|
||||||
|
lines.append(f"- **{p['file']}** [{p['class']}] {p['detail']}")
|
||||||
|
for path, what, m in p["caught"][:1]:
|
||||||
|
lines.append(" ```\n " + f"{path} ({what}): " + m.replace("\n", "\n ")[:1500] + "\n ```")
|
||||||
|
if not p["caught"] and p["stderr"]:
|
||||||
|
lines.append(" ```\n " + p["stderr"].strip()[:1500].replace("\n", "\n ") + "\n ```")
|
||||||
|
lines += ["", "## Our-error root causes (files affected)", ""]
|
||||||
|
for k, v in ser(root_causes).items():
|
||||||
|
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||||
|
for ex in v["examples"][:3]:
|
||||||
|
lines.append(f" - {ex}")
|
||||||
|
lines += ["", "## Mismatch root causes", ""]
|
||||||
|
for k, v in ser(mismatch_causes).items():
|
||||||
|
lines.append(f"- [{v['files']} files, {v['count']} objs] `{k}`")
|
||||||
|
for ex in v["examples"][:3]:
|
||||||
|
lines.append(f" - {ex}")
|
||||||
|
lines += ["", "## Objects h5py fails on but we read (top)", ""]
|
||||||
|
for k, n in ref_only_errors.most_common(15):
|
||||||
|
lines.append(f"- {n} x `{k}`")
|
||||||
|
open(os.path.join(R, "summary.md"), "w").write("\n".join(lines) + "\n")
|
||||||
|
print("\n".join(lines[:4 + len(by_corpus)]))
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
# Conformance corpora, pinned by commit. fetch-corpus.sh reads this file.
|
||||||
|
#
|
||||||
|
# name git-url commit root [sparse-checkout patterns...]
|
||||||
|
#
|
||||||
|
# `root` is the directory inside the checkout that is swept ("." = all of it).
|
||||||
|
# Patterns are git non-cone sparse-checkout patterns; none = whole repository.
|
||||||
|
# Every file under <root> with an HDF5/netCDF-4 extension is probed; for
|
||||||
|
# cve_hdf5 the extension-less files in cvefiles/ and fuzzerfiles/ are too.
|
||||||
|
# Licences: each corpus keeps its upstream licence; nothing here is committed
|
||||||
|
# to this repository — the files are downloaded into the gitignored cache.
|
||||||
|
hdf5 https://github.com/HDFGroup/hdf5.git a3cf1ea82cc7a66e50029a688121e1b105a7ce88 . *.h5 *.he5 *.nc *.hdf5 *.h5f
|
||||||
|
cve_hdf5 https://github.com/HDFGroup/cve_hdf5.git 3fd1f5ae3869e01b8ae02b41d7108de7ffb1a374 .
|
||||||
|
netcdf-c https://github.com/Unidata/netcdf-c.git beb7b9585273c1548386231a59b809d906359033 . /nc_test4/*.nc /ncdump/*.nc /nc_test4/*.h5 /ncdump/*.h5 /h5_test/*.h5 /hdf5_test/*.h5
|
||||||
|
NCAS-CMS_pyfive https://github.com/NCAS-CMS/pyfive.git 8cf07b8749133f41c5e30b8a4c604486f687fe74 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||||
|
usnistgov_h5wasm https://github.com/usnistgov/h5wasm.git 02f6336527d2812783fcedabfbf42127ec8d06d2 . *.h5 *.hdf5 *.hdf *.nc *.he5
|
||||||
|
netcdf4-python https://github.com/Unidata/netcdf4-python.git 6e67576d39aef8091fb20bd767b4f1a52ddc1bec . *.nc *.h5
|
||||||
|
xarray-data https://github.com/pydata/xarray-data.git a35297e9da2cc99c811014f0c8a4297345a5c28d . /basin_mask.nc /precipitation.nc4 /imerghh_730.hdf5 /eraint_uvz.nc /ROMS_example.nc /tiny.nc
|
||||||
|
# h5py 3.16.0 (tag 3.16.0), its test data files.
|
||||||
|
h5py_data https://github.com/h5py/h5py.git b2f0347c4200333acd89b43733f1caa0c115162f h5py/tests/data_files /h5py/tests/data_files/*
|
||||||
Executable
+39
@@ -0,0 +1,39 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# fetch-corpus.sh [cache_dir]
|
||||||
|
#
|
||||||
|
# Download the corpora pinned in conformance/corpus.txt into the (gitignored)
|
||||||
|
# cache: <cache>/src/<name> is a shallow, sparse, blob-filtered checkout of the
|
||||||
|
# pinned commit and <cache>/corpus/<name> links to the swept root inside it.
|
||||||
|
# A corpus already checked out at its pinned commit is left alone, so a second
|
||||||
|
# run costs nothing and needs no network.
|
||||||
|
set -euo pipefail
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
CACHE="${1:-${CONFORMANCE_CACHE:-$HERE/.cache}}"
|
||||||
|
mkdir -p "$CACHE/src" "$CACHE/corpus"
|
||||||
|
CACHE="$(cd "$CACHE" && pwd)"
|
||||||
|
|
||||||
|
retry() { local i; for i in 1 2 3 4; do "$@" && return 0; sleep $((i * 5)); done; return 1; }
|
||||||
|
|
||||||
|
grep -v '^[[:space:]]*\(#\|$\)' "$HERE/corpus.txt" | while read -r name url commit root patterns; do
|
||||||
|
src="$CACHE/src/$name"
|
||||||
|
if [ -d "$src/.git" ] && [ "$(git -C "$src" rev-parse HEAD 2>/dev/null)" = "$commit" ]; then
|
||||||
|
echo "cached $name @ ${commit:0:12}"
|
||||||
|
else
|
||||||
|
echo "fetching $name @ ${commit:0:12} from $url"
|
||||||
|
rm -rf "$src"
|
||||||
|
git init -q "$src"
|
||||||
|
git -C "$src" remote add origin "$url"
|
||||||
|
git -C "$src" config advice.detachedHead false
|
||||||
|
if [ -n "$patterns" ]; then
|
||||||
|
git -C "$src" config core.sparseCheckout true
|
||||||
|
# no-cone patterns (globs); `set -f` keeps the shell from expanding them
|
||||||
|
(set -f; printf '%s\n' $patterns) > "$src/.git/info/sparse-checkout"
|
||||||
|
fi
|
||||||
|
retry git -C "$src" fetch -q --depth 1 --filter=blob:none origin "$commit"
|
||||||
|
retry git -C "$src" checkout -q FETCH_HEAD
|
||||||
|
got="$(git -C "$src" rev-parse HEAD)"
|
||||||
|
[ "$got" = "$commit" ] || { echo "error: $name checked out $got, expected $commit" >&2; exit 1; }
|
||||||
|
fi
|
||||||
|
ln -sfn "$src/$root" "$CACHE/corpus/$name"
|
||||||
|
done
|
||||||
|
echo "corpus ready in $CACHE/corpus"
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""list_files.py <corpus_dir>: print the files the sweep probes, one per line,
|
||||||
|
as <corpus>/<path> in byte order.
|
||||||
|
|
||||||
|
* every file named *.h5 *.hdf5 *.he5 *.nc *.nc4 *.hdf *.h5f in each corpus,
|
||||||
|
except netCDF classic / 64-bit-offset / CDF5 files (magic "CDF"): they are
|
||||||
|
not HDF5, so neither side can read them and they say nothing;
|
||||||
|
* plus, for cve_hdf5, every file in cvefiles/ and fuzzerfiles/ except
|
||||||
|
.md/.c sources — the reproducers are mostly extension-less, and they are
|
||||||
|
kept whatever their bytes look like (that is their point).
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
EXTS = (".h5", ".hdf5", ".he5", ".nc", ".nc4", ".hdf", ".h5f")
|
||||||
|
|
||||||
|
|
||||||
|
def walk(top):
|
||||||
|
for dirpath, dirnames, filenames in os.walk(top):
|
||||||
|
dirnames[:] = [d for d in dirnames if d != ".git"]
|
||||||
|
for fn in filenames:
|
||||||
|
p = os.path.join(dirpath, fn)
|
||||||
|
if os.path.isfile(p) and not os.path.islink(p):
|
||||||
|
yield os.path.relpath(p, top)
|
||||||
|
|
||||||
|
|
||||||
|
def main(root):
|
||||||
|
out = set()
|
||||||
|
for corpus in sorted(os.listdir(root)):
|
||||||
|
top = os.path.join(root, corpus)
|
||||||
|
if not os.path.isdir(top):
|
||||||
|
continue
|
||||||
|
for rel in walk(top):
|
||||||
|
path = os.path.join(top, rel)
|
||||||
|
if rel.lower().endswith(EXTS):
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
if fh.read(3) == b"CDF":
|
||||||
|
continue
|
||||||
|
out.add(f"{corpus}/{rel}")
|
||||||
|
elif corpus == "cve_hdf5" and rel.split(os.sep)[0] in ("cvefiles", "fuzzerfiles") \
|
||||||
|
and not rel.endswith((".md", ".c")):
|
||||||
|
out.add(f"{corpus}/{rel}")
|
||||||
|
for f in sorted(out, key=lambda s: s.encode()):
|
||||||
|
print(f)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1])
|
||||||
Generated
+458
@@ -0,0 +1,458 @@
|
|||||||
|
# This file is automatically @generated by Cargo.
|
||||||
|
# It is not intended for manual editing.
|
||||||
|
version = 4
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "adler2"
|
||||||
|
version = "2.0.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "better_io"
|
||||||
|
version = "0.2.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ef0a3155e943e341e557863e69a708999c94ede624e37865c8e2a91b94efa78f"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "block-buffer"
|
||||||
|
version = "0.10.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
|
||||||
|
dependencies = [
|
||||||
|
"generic-array",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "byteorder"
|
||||||
|
version = "1.5.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cc"
|
||||||
|
version = "1.5.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040"
|
||||||
|
dependencies = [
|
||||||
|
"find-msvc-tools",
|
||||||
|
"jobserver",
|
||||||
|
"libc",
|
||||||
|
"shlex",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cfg-if"
|
||||||
|
version = "1.0.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "4e7648175b45a9a48536d676f68d918270699102aa8dab5496df06904c914600"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "clawhdf5-format"
|
||||||
|
version = "2.7.0"
|
||||||
|
dependencies = [
|
||||||
|
"byteorder",
|
||||||
|
"flate2",
|
||||||
|
"libaec-sys",
|
||||||
|
"lz4_flex",
|
||||||
|
"pco",
|
||||||
|
"portable-atomic",
|
||||||
|
"sha2",
|
||||||
|
"zstd",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "conformance-probe"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"clawhdf5-format",
|
||||||
|
"serde_json",
|
||||||
|
"sha2",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "cpufeatures"
|
||||||
|
version = "0.2.17"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280"
|
||||||
|
dependencies = [
|
||||||
|
"libc",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crc32fast"
|
||||||
|
version = "1.5.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "01a7799fd6b852db0e61728dde9a204c423b44d689dbd432522543614b490e78"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crunchy"
|
||||||
|
version = "0.2.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "460fbee9c2c2f33933d720630a6a0bac33ba7053db5344fac858d4b8952d77d5"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "crypto-common"
|
||||||
|
version = "0.1.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
|
||||||
|
dependencies = [
|
||||||
|
"generic-array",
|
||||||
|
"typenum",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "digest"
|
||||||
|
version = "0.10.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
|
||||||
|
dependencies = [
|
||||||
|
"block-buffer",
|
||||||
|
"crypto-common",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "dtype_dispatch"
|
||||||
|
version = "0.2.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ab23e69df104e2fd85ee63a533a22d2132ef5975dc6b36f9f3e5a7305e4a8ed7"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "find-msvc-tools"
|
||||||
|
version = "0.1.14"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "aedcfb3409746eddb02b9e19ebda1c3394f759a152e48ee875a0844d1b955484"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "flate2"
|
||||||
|
version = "1.1.10"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb"
|
||||||
|
dependencies = [
|
||||||
|
"crc32fast",
|
||||||
|
"miniz_oxide",
|
||||||
|
"zlib-rs",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "generic-array"
|
||||||
|
version = "0.14.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
|
||||||
|
dependencies = [
|
||||||
|
"typenum",
|
||||||
|
"version_check",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "getrandom"
|
||||||
|
version = "0.4.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"libc",
|
||||||
|
"r-efi",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "half"
|
||||||
|
version = "2.7.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6ea2d84b969582b4b1864a92dc5d27cd2b77b622a8d79306834f1be5ba20d84b"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"crunchy",
|
||||||
|
"zerocopy",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "itoa"
|
||||||
|
version = "1.0.18"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "jobserver"
|
||||||
|
version = "0.1.35"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3"
|
||||||
|
dependencies = [
|
||||||
|
"getrandom",
|
||||||
|
"libc",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libaec-sys"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "libc"
|
||||||
|
version = "0.2.189"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "lz4_flex"
|
||||||
|
version = "0.11.6"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "373f5eceeeab7925e0c1098212f2fbc4d416adec9d35051a6ab251e824c1854a"
|
||||||
|
dependencies = [
|
||||||
|
"twox-hash",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "memchr"
|
||||||
|
version = "2.8.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "miniz_oxide"
|
||||||
|
version = "0.9.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c"
|
||||||
|
dependencies = [
|
||||||
|
"adler2",
|
||||||
|
"simd-adler32",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pco"
|
||||||
|
version = "1.0.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "386342cad4c6e97f081568e5d910ea7d871314c843aa8fc564f2a6b64cab9456"
|
||||||
|
dependencies = [
|
||||||
|
"better_io",
|
||||||
|
"dtype_dispatch",
|
||||||
|
"half",
|
||||||
|
"rand_xoshiro",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pkg-config"
|
||||||
|
version = "0.3.34"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "portable-atomic"
|
||||||
|
version = "1.15.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "proc-macro2"
|
||||||
|
version = "1.0.107"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "985e7ec9bb745e6ce6535b544d84d6cd6f7ad8bd711c398938ae983b91a766d9"
|
||||||
|
dependencies = [
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "quote"
|
||||||
|
version = "1.0.47"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1fbf4db142a473a8d80c26bbf18454ed458bf8d26c8219c331daecfdbd079001"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "r-efi"
|
||||||
|
version = "6.0.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rand_core"
|
||||||
|
version = "0.6.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rand_xoshiro"
|
||||||
|
version = "0.6.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6f97cdb2a36ed4183de61b2f824cc45c9f1037f28afe0a322e9fff4c108b5aaa"
|
||||||
|
dependencies = [
|
||||||
|
"rand_core",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba"
|
||||||
|
dependencies = [
|
||||||
|
"serde_core",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_core"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48"
|
||||||
|
dependencies = [
|
||||||
|
"serde_derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_derive"
|
||||||
|
version = "1.0.229"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn 3.0.6",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "serde_json"
|
||||||
|
version = "1.0.151"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14"
|
||||||
|
dependencies = [
|
||||||
|
"itoa",
|
||||||
|
"memchr",
|
||||||
|
"serde",
|
||||||
|
"serde_core",
|
||||||
|
"zmij",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "sha2"
|
||||||
|
version = "0.10.9"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283"
|
||||||
|
dependencies = [
|
||||||
|
"cfg-if",
|
||||||
|
"cpufeatures",
|
||||||
|
"digest",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "shlex"
|
||||||
|
version = "2.0.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "simd-adler32"
|
||||||
|
version = "0.3.10"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "2.0.119"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "872831b642d1a07999a962a351ed35b955ea2cfc8f3862091e2a240a84f17297"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "syn"
|
||||||
|
version = "3.0.6"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "8593e8e72159ed2257d083c7a454a85cbf854f37a0966d8d483aff8c8a3ebcee"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"unicode-ident",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "twox-hash"
|
||||||
|
version = "2.1.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "typenum"
|
||||||
|
version = "1.20.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "unicode-ident"
|
||||||
|
version = "1.0.26"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "d245f478577f809a851594d02313b640fb437e0bb33866753cff937863096954"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "version_check"
|
||||||
|
version = "0.9.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zerocopy"
|
||||||
|
version = "0.8.59"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6df92bf3d9227be3d53173901ddbffac2babc27ae50f397776ffd6dc33f800cb"
|
||||||
|
dependencies = [
|
||||||
|
"zerocopy-derive",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zerocopy-derive"
|
||||||
|
version = "0.8.59"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "ac4f328cf2f05d084e496c3e9c3f33ed0a183656a16e1fcec4d464d8373aec82"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
"syn 2.0.119",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zlib-rs"
|
||||||
|
version = "0.6.8"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "b268e58e7c693d7c271f93ffc4ba3b380412554231c85bf61ca7af91042a4112"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zmij"
|
||||||
|
version = "1.0.23"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd"
|
||||||
|
version = "0.13.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-safe",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-safe"
|
||||||
|
version = "7.3.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882"
|
||||||
|
dependencies = [
|
||||||
|
"zstd-sys",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "zstd-sys"
|
||||||
|
version = "2.1.0+zstd.1.5.7"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0"
|
||||||
|
dependencies = [
|
||||||
|
"cc",
|
||||||
|
"pkg-config",
|
||||||
|
]
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
[package]
|
||||||
|
name = "conformance-probe"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition = "2024"
|
||||||
|
rust-version = "1.92"
|
||||||
|
publish = false
|
||||||
|
description = "Walks an HDF5 file with clawhdf5-format and prints a canonical JSON description (see conformance/README.md)"
|
||||||
|
|
||||||
|
# Deliberately outside the main workspace: `cargo test --workspace` never
|
||||||
|
# builds it, and it links the optional C codecs (zstd, libaec) that the core
|
||||||
|
# crates' default build must not.
|
||||||
|
[workspace]
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
|
clawhdf5-format = { path = "../../crates/clawhdf5-format", features = ["lz4", "zstd", "szip", "pcodec"] }
|
||||||
|
serde_json = "1"
|
||||||
|
sha2 = "0.10"
|
||||||
|
|
||||||
|
[profile.release]
|
||||||
|
# Keep panics catchable (the probe records them per object) and turn integer
|
||||||
|
# overflow into a reported panic instead of silent wraparound.
|
||||||
|
debug = 1
|
||||||
|
overflow-checks = true
|
||||||
|
debug-assertions = true
|
||||||
|
panic = "unwind"
|
||||||
@@ -0,0 +1,898 @@
|
|||||||
|
//! Conformance probe: walks an HDF5 file with clawhdf5-format (the same calls
|
||||||
|
//! the `clawhdf5` facade makes) and prints a canonical JSON description:
|
||||||
|
//! every hard-linked object (sorted-name DFS, deduplicated by header address),
|
||||||
|
//! and for each dataset / attribute its shape plus the SHA-256 of its values
|
||||||
|
//! in a canonical encoding shared with `ref.py`.
|
||||||
|
//!
|
||||||
|
//! Canonical value encoding (per element, concatenated, row-major):
|
||||||
|
//! int / float / bitfield / enum / time : element bytes, little-endian
|
||||||
|
//! non-IEEE-layout float (e.g. N-Bit) : the IEEE float of the same size it converts to
|
||||||
|
//! int with bit offset / short precision: the full-width integer it converts to
|
||||||
|
//! opaque : raw bytes
|
||||||
|
//! compound : members in declaration order (padding dropped)
|
||||||
|
//! array : base elements row-major
|
||||||
|
//! string (fixed or VL) : b'S' + u32le len + bytes (cut at first NUL, trailing spaces stripped)
|
||||||
|
//! VL sequence : b'V' + u32le count + base elements
|
||||||
|
//! reference : b'R' (payload not compared)
|
||||||
|
//!
|
||||||
|
//! Every object is processed inside catch_unwind; a caught panic is recorded
|
||||||
|
//! with its message, location and the clawhdf5 frames of its backtrace.
|
||||||
|
|
||||||
|
use std::cell::RefCell;
|
||||||
|
use std::collections::{HashMap, HashSet};
|
||||||
|
use std::panic::{self, AssertUnwindSafe};
|
||||||
|
use std::rc::Rc;
|
||||||
|
|
||||||
|
use clawhdf5_format::attribute::extract_attributes_full;
|
||||||
|
use clawhdf5_format::data_layout::DataLayout;
|
||||||
|
use clawhdf5_format::data_read;
|
||||||
|
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
||||||
|
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||||
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||||
|
use clawhdf5_format::global_heap::GlobalHeapCollection;
|
||||||
|
use clawhdf5_format::group_v1::{self, GroupEntry};
|
||||||
|
use clawhdf5_format::group_v2;
|
||||||
|
use clawhdf5_format::message_type::MessageType;
|
||||||
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
|
use clawhdf5_format::signature;
|
||||||
|
use clawhdf5_format::superblock::Superblock;
|
||||||
|
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
||||||
|
use serde_json::{Map, Value, json};
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
|
||||||
|
const MAX_BYTES: u64 = 200 * 1024 * 1024;
|
||||||
|
const MAX_OBJECTS: usize = 200_000;
|
||||||
|
|
||||||
|
thread_local! {
|
||||||
|
static LAST_PANIC: RefCell<Option<String>> = const { RefCell::new(None) };
|
||||||
|
}
|
||||||
|
|
||||||
|
fn install_hook() {
|
||||||
|
panic::set_hook(Box::new(|info| {
|
||||||
|
let msg = if let Some(s) = info.payload().downcast_ref::<&str>() {
|
||||||
|
s.to_string()
|
||||||
|
} else if let Some(s) = info.payload().downcast_ref::<String>() {
|
||||||
|
s.clone()
|
||||||
|
} else {
|
||||||
|
"<non-string panic>".into()
|
||||||
|
};
|
||||||
|
let loc = info
|
||||||
|
.location()
|
||||||
|
.map(|l| format!("{}:{}", l.file(), l.line()))
|
||||||
|
.unwrap_or_default();
|
||||||
|
let bt = std::backtrace::Backtrace::force_capture().to_string();
|
||||||
|
// keep only frames from clawhdf5 code
|
||||||
|
let mut frames = Vec::new();
|
||||||
|
let lines: Vec<&str> = bt.lines().collect();
|
||||||
|
for (i, l) in lines.iter().enumerate() {
|
||||||
|
let t = l.trim();
|
||||||
|
if t.contains("clawhdf5_format::") || t.contains("conformance_probe::") {
|
||||||
|
let at = lines
|
||||||
|
.get(i + 1)
|
||||||
|
.map(|n| n.trim())
|
||||||
|
.filter(|n| n.starts_with("at "))
|
||||||
|
.map(|n| {
|
||||||
|
let n = n.trim_start_matches("at ");
|
||||||
|
match n.find("/crates/") {
|
||||||
|
Some(p) => n[p + 1..].to_string(),
|
||||||
|
None => n.to_string(),
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.unwrap_or_default();
|
||||||
|
let name = t.split_once(": ").map(|x| x.1).unwrap_or(t);
|
||||||
|
frames.push(format!("{name} ({at})"));
|
||||||
|
if frames.len() >= 12 {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let full = format!("PANIC: {msg} @ {loc}\n {}", frames.join("\n "));
|
||||||
|
eprintln!("{full}");
|
||||||
|
LAST_PANIC.with(|p| *p.borrow_mut() = Some(full));
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run `f`, turning a panic into Err("PANIC: ...").
|
||||||
|
fn guarded<T>(f: impl FnOnce() -> Result<T, String>) -> Result<T, String> {
|
||||||
|
match panic::catch_unwind(AssertUnwindSafe(f)) {
|
||||||
|
Ok(r) => r,
|
||||||
|
Err(_) => Err(LAST_PANIC
|
||||||
|
.with(|p| p.borrow_mut().take())
|
||||||
|
.unwrap_or_else(|| "PANIC: <unknown>".into())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn e<E: std::fmt::Debug>(x: E) -> String {
|
||||||
|
format!("{x:?}")
|
||||||
|
}
|
||||||
|
|
||||||
|
struct Ctx<'a> {
|
||||||
|
data: &'a [u8],
|
||||||
|
os: u8,
|
||||||
|
ls: u8,
|
||||||
|
base_dir: std::path::PathBuf,
|
||||||
|
heaps: RefCell<HashMap<u64, Result<Rc<GlobalHeapCollection>, String>>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'a> Ctx<'a> {
|
||||||
|
fn header(&self, addr: u64) -> Result<ObjectHeader, String> {
|
||||||
|
ObjectHeader::parse(self.data, addr as usize, self.os, self.ls).map_err(e)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn payload(&self, h: &ObjectHeader, t: MessageType) -> Result<Option<Vec<u8>>, String> {
|
||||||
|
match h.messages.iter().find(|m| m.msg_type == t) {
|
||||||
|
None => Ok(None),
|
||||||
|
Some(m) => {
|
||||||
|
clawhdf5_format::shared_message::message_data(self.data, m, self.os, self.ls)
|
||||||
|
.map(|c| Some(c.into_owned()))
|
||||||
|
.map_err(e)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn heap_obj(&self, addr: u64, idx: u32) -> Result<Vec<u8>, String> {
|
||||||
|
let coll = {
|
||||||
|
let mut cache = self.heaps.borrow_mut();
|
||||||
|
cache
|
||||||
|
.entry(addr)
|
||||||
|
.or_insert_with(|| {
|
||||||
|
GlobalHeapCollection::parse(self.data, addr as usize, self.ls)
|
||||||
|
.map(Rc::new)
|
||||||
|
.map_err(e)
|
||||||
|
})
|
||||||
|
.clone()?
|
||||||
|
};
|
||||||
|
coll.get_object(idx as u16)
|
||||||
|
.map(|o| o.data.clone())
|
||||||
|
.ok_or_else(|| {
|
||||||
|
format!("GlobalHeapObjectNotFound {{ collection_address: {addr}, index: {idx} }}")
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_offset(&self, b: &[u8]) -> u64 {
|
||||||
|
let mut v = 0u64;
|
||||||
|
for (i, x) in b.iter().take(self.os as usize).enumerate() {
|
||||||
|
v |= (*x as u64) << (8 * i);
|
||||||
|
}
|
||||||
|
v
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon(&self, dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let size = dt.type_size() as usize;
|
||||||
|
if b.len() < size {
|
||||||
|
return Err(format!(
|
||||||
|
"canon: element slice {} < type size {size}",
|
||||||
|
b.len()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
match dt {
|
||||||
|
Datatype::FloatingPoint { .. } if !ieee_layout(dt) => {
|
||||||
|
canon_custom_float(dt, &b[..size], out)?
|
||||||
|
}
|
||||||
|
Datatype::FixedPoint { .. } if partial_int(dt) => {
|
||||||
|
canon_partial_int(dt, &b[..size], out)?
|
||||||
|
}
|
||||||
|
Datatype::FixedPoint { byte_order, .. }
|
||||||
|
| Datatype::BitField { byte_order, .. }
|
||||||
|
| Datatype::FloatingPoint { byte_order, .. } => match byte_order {
|
||||||
|
DatatypeByteOrder::LittleEndian => out.extend_from_slice(&b[..size]),
|
||||||
|
DatatypeByteOrder::BigEndian => out.extend(b[..size].iter().rev()),
|
||||||
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||||
|
},
|
||||||
|
Datatype::Time { .. } | Datatype::Opaque { .. } => out.extend_from_slice(&b[..size]),
|
||||||
|
Datatype::String { .. } => canon_str(&b[..size], out),
|
||||||
|
Datatype::Compound { members, .. } => {
|
||||||
|
for m in members {
|
||||||
|
let off = m.byte_offset as usize;
|
||||||
|
let ms = m.datatype.type_size() as usize;
|
||||||
|
if off.checked_add(ms).is_none_or(|end| end > size) {
|
||||||
|
return Err(format!("canon: member {} out of bounds", m.name));
|
||||||
|
}
|
||||||
|
self.canon(&m.datatype, &b[off..off + ms], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Datatype::Reference { .. } => out.push(b'R'),
|
||||||
|
Datatype::Enumeration { base_type, .. } => self.canon(base_type, b, out)?,
|
||||||
|
Datatype::Array {
|
||||||
|
base_type,
|
||||||
|
dimensions,
|
||||||
|
} => {
|
||||||
|
let n: usize = dimensions.iter().map(|d| *d as usize).product();
|
||||||
|
let bs = base_type.type_size() as usize;
|
||||||
|
for i in 0..n {
|
||||||
|
self.canon(base_type, &b[i * bs..], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Datatype::VariableLength {
|
||||||
|
is_string,
|
||||||
|
base_type,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
let len = u32::from_le_bytes([b[0], b[1], b[2], b[3]]) as usize;
|
||||||
|
let addr = self.read_offset(&b[4..]);
|
||||||
|
let idx_off = 4 + self.os as usize;
|
||||||
|
let idx = u32::from_le_bytes([
|
||||||
|
b[idx_off],
|
||||||
|
b[idx_off + 1],
|
||||||
|
b[idx_off + 2],
|
||||||
|
b[idx_off + 3],
|
||||||
|
]);
|
||||||
|
let obj = if len == 0 || addr == 0 || addr == u64::MAX >> (64 - 8 * self.os as u32)
|
||||||
|
{
|
||||||
|
Vec::new()
|
||||||
|
} else {
|
||||||
|
self.heap_obj(addr, idx)?
|
||||||
|
};
|
||||||
|
if *is_string {
|
||||||
|
let l = len.min(obj.len());
|
||||||
|
canon_str(&obj[..l], out);
|
||||||
|
} else {
|
||||||
|
let bs = base_type.type_size() as usize;
|
||||||
|
if bs == 0 {
|
||||||
|
return Err("canon: VL base size 0".into());
|
||||||
|
}
|
||||||
|
let need = len.checked_mul(bs).ok_or("canon: VL overflow")?;
|
||||||
|
if len > 0 && obj.len() < need {
|
||||||
|
return Err(format!("canon: VL object {} < {need}", obj.len()));
|
||||||
|
}
|
||||||
|
out.push(b'V');
|
||||||
|
out.extend_from_slice(&(len as u32).to_le_bytes());
|
||||||
|
for i in 0..len {
|
||||||
|
self.canon(base_type, &obj[i * bs..], out)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns (shape json, n_elements)
|
||||||
|
fn shape(ds: &Dataspace) -> (Value, u64) {
|
||||||
|
match ds.space_type {
|
||||||
|
DataspaceType::Null => (Value::String("null".into()), 0),
|
||||||
|
DataspaceType::Scalar => (json!([]), 1),
|
||||||
|
DataspaceType::Simple => {
|
||||||
|
let n = ds.dimensions.iter().fold(1u64, |a, d| a.saturating_mul(*d));
|
||||||
|
(json!(ds.dimensions), n)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hash_values(
|
||||||
|
&self,
|
||||||
|
dt: &Datatype,
|
||||||
|
raw: &[u8],
|
||||||
|
n: u64,
|
||||||
|
rec: &mut Map<String, Value>,
|
||||||
|
) -> Result<(), String> {
|
||||||
|
let size = dt.type_size() as usize;
|
||||||
|
let need = (n as usize).checked_mul(size).ok_or("n*size overflow")?;
|
||||||
|
if raw.len() != need {
|
||||||
|
return Err(format!(
|
||||||
|
"raw length {} != n_elements {n} * type_size {size}",
|
||||||
|
raw.len()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let mut canon = Vec::with_capacity(need);
|
||||||
|
for i in 0..n as usize {
|
||||||
|
self.canon(dt, &raw[i * size..(i + 1) * size], &mut canon)?;
|
||||||
|
}
|
||||||
|
let h = Sha256::digest(&canon);
|
||||||
|
rec.insert("hash".into(), Value::String(hex(&h)));
|
||||||
|
rec.insert(
|
||||||
|
"head".into(),
|
||||||
|
Value::String(hex(&canon[..canon.len().min(48)])),
|
||||||
|
);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// VDS source files resolve next to the virtual file; like the library,
|
||||||
|
/// refuse absolute paths and `..`.
|
||||||
|
fn vds_resolver(
|
||||||
|
&self,
|
||||||
|
) -> impl Fn(&str) -> Result<Option<Vec<u8>>, clawhdf5_format::error::FormatError> + use<> {
|
||||||
|
let base = self.base_dir.clone();
|
||||||
|
move |name: &str| {
|
||||||
|
use clawhdf5_format::error::FormatError;
|
||||||
|
let p = std::path::Path::new(name);
|
||||||
|
if p.is_absolute()
|
||||||
|
|| p.components()
|
||||||
|
.any(|c| matches!(c, std::path::Component::ParentDir))
|
||||||
|
{
|
||||||
|
return Err(FormatError::ChunkedReadError(format!("refused {name}")));
|
||||||
|
}
|
||||||
|
match std::fs::read(base.join(p)) {
|
||||||
|
Ok(b) => Ok(Some(b)),
|
||||||
|
Err(err) if err.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
||||||
|
Err(err) => Err(FormatError::ChunkedReadError(err.to_string())),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn read_dataset(&self, h: &ObjectHeader, rec: &mut Map<String, Value>) -> Result<(), String> {
|
||||||
|
let dtb = self
|
||||||
|
.payload(h, MessageType::Datatype)?
|
||||||
|
.ok_or("MissingMessage(Datatype)")?;
|
||||||
|
let (dt, _) = Datatype::parse(&dtb).map_err(e)?;
|
||||||
|
rec.insert("dtype".into(), Value::String(dtype_str(&dt)));
|
||||||
|
let dsb = self
|
||||||
|
.payload(h, MessageType::Dataspace)?
|
||||||
|
.ok_or("MissingMessage(Dataspace)")?;
|
||||||
|
let mut ds = Dataspace::parse(&dsb, self.ls).map_err(e)?;
|
||||||
|
// A virtual dataset's extent can come from its sources (unlimited /
|
||||||
|
// printf mappings), as h5py reports it, rather than the stored one.
|
||||||
|
if let Some(lm) = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
&& let Ok(dl @ DataLayout::Virtual { .. }) =
|
||||||
|
DataLayout::parse(&lm.data, self.os, self.ls)
|
||||||
|
{
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
Some(&resolver),
|
||||||
|
)
|
||||||
|
.map_err(e)?;
|
||||||
|
}
|
||||||
|
let (shape, n) = Self::shape(&ds);
|
||||||
|
rec.insert("shape".into(), shape);
|
||||||
|
if n.saturating_mul(dt.type_size() as u64) > MAX_BYTES {
|
||||||
|
rec.insert("skipped".into(), Value::String("too large".into()));
|
||||||
|
return Ok(());
|
||||||
|
}
|
||||||
|
let lm = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
.ok_or("MissingMessage(DataLayout)")?;
|
||||||
|
let dl = DataLayout::parse(&lm.data, self.os, self.ls).map_err(e)?;
|
||||||
|
rec.insert(
|
||||||
|
"layout".into(),
|
||||||
|
Value::String(
|
||||||
|
match &dl {
|
||||||
|
DataLayout::Compact { .. } => "compact",
|
||||||
|
DataLayout::Contiguous { .. } => "contiguous",
|
||||||
|
DataLayout::Chunked { .. } => "chunked",
|
||||||
|
DataLayout::Virtual { .. } => "virtual",
|
||||||
|
}
|
||||||
|
.into(),
|
||||||
|
),
|
||||||
|
);
|
||||||
|
let pipeline = match self.payload(h, MessageType::FilterPipeline)? {
|
||||||
|
Some(p) => Some(FilterPipeline::parse(&p).map_err(e)?),
|
||||||
|
None => None,
|
||||||
|
};
|
||||||
|
if let Some(p) = &pipeline {
|
||||||
|
rec.insert(
|
||||||
|
"filters".into(),
|
||||||
|
json!(p.filters.iter().map(|f| f.filter_id).collect::<Vec<_>>()),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let raw = if matches!(dl, DataLayout::Virtual { .. }) {
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||||
|
self.data,
|
||||||
|
&h.messages,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
)
|
||||||
|
.map_err(e)?;
|
||||||
|
clawhdf5_format::vds::read_virtual_dataset(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
fill.as_deref(),
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
Some(&resolver),
|
||||||
|
)
|
||||||
|
.map_err(e)?
|
||||||
|
.data
|
||||||
|
} else {
|
||||||
|
let cache = clawhdf5_format::chunk_cache::ChunkCache::new();
|
||||||
|
clawhdf5_format::fill_value::read_full_with_fill::<clawhdf5_format::error::FormatError>(
|
||||||
|
&h.messages,
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
dt.type_size() as usize,
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
|| {
|
||||||
|
data_read::read_raw_data_cached(
|
||||||
|
self.data,
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
pipeline.as_ref(),
|
||||||
|
self.os,
|
||||||
|
self.ls,
|
||||||
|
&cache,
|
||||||
|
)
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.map_err(e)?
|
||||||
|
};
|
||||||
|
self.hash_values(&dt, &raw, n, rec)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn attrs(&self, h: &ObjectHeader) -> Result<Map<String, Value>, String> {
|
||||||
|
let msgs = extract_attributes_full(self.data, h, self.os, self.ls).map_err(e)?;
|
||||||
|
let mut out = Map::new();
|
||||||
|
for a in &msgs {
|
||||||
|
let r = guarded(|| {
|
||||||
|
let mut rec = Map::new();
|
||||||
|
rec.insert("dtype".into(), Value::String(dtype_str(&a.datatype)));
|
||||||
|
let (shape, n) = Self::shape(&a.dataspace);
|
||||||
|
rec.insert("shape".into(), shape);
|
||||||
|
self.hash_values(&a.datatype, &a.raw_data, n, &mut rec)?;
|
||||||
|
Ok(rec)
|
||||||
|
});
|
||||||
|
let v = match r {
|
||||||
|
Ok(rec) => Value::Object(rec),
|
||||||
|
Err(msg) => json!({ "error": msg }),
|
||||||
|
};
|
||||||
|
out.insert(a.name.clone(), v);
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entries(&self, h: &ObjectHeader) -> Result<Vec<GroupEntry>, String> {
|
||||||
|
let v1 = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::SymbolTable);
|
||||||
|
if let Some(m) = v1 {
|
||||||
|
let stm = SymbolTableMessage::parse(&m.data, self.os).map_err(e)?;
|
||||||
|
group_v1::resolve_v1_group_entries(self.data, &stm, self.os, self.ls).map_err(e)
|
||||||
|
} else if h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link)
|
||||||
|
{
|
||||||
|
group_v2::resolve_v2_group_entries(self.data, h, self.os, self.ls).map_err(e)
|
||||||
|
} else {
|
||||||
|
Ok(Vec::new())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Element bytes as an unsigned integer (at most 16 bytes), honouring byte order.
|
||||||
|
fn element_bits(b: &[u8], byte_order: &DatatypeByteOrder) -> Result<u128, String> {
|
||||||
|
if b.len() > 16 {
|
||||||
|
return Err(format!("canon: {}-byte numeric element", b.len()));
|
||||||
|
}
|
||||||
|
let mut v = 0u128;
|
||||||
|
match byte_order {
|
||||||
|
DatatypeByteOrder::LittleEndian => {
|
||||||
|
for (i, x) in b.iter().enumerate() {
|
||||||
|
v |= u128::from(*x) << (8 * i);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
DatatypeByteOrder::BigEndian => {
|
||||||
|
for x in b {
|
||||||
|
v = (v << 8) | u128::from(*x);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
DatatypeByteOrder::Vax => return Err("canon: VAX byte order".into()),
|
||||||
|
}
|
||||||
|
Ok(v)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn field(v: u128, pos: u32, len: u32) -> u128 {
|
||||||
|
if len == 0 || pos >= 128 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
let v = v >> pos;
|
||||||
|
if len >= 128 {
|
||||||
|
v
|
||||||
|
} else {
|
||||||
|
v & ((1u128 << len) - 1)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// True when a float's bit fields are exactly IEEE 754 binary16/32/64 for its
|
||||||
|
/// size. h5py hands back such a type's bytes untouched; any other layout (an
|
||||||
|
/// N-Bit `H5Tset_precision` float, say) is *converted* by libhdf5 into the
|
||||||
|
/// numpy float of the same size, so comparing raw bytes would be meaningless.
|
||||||
|
fn ieee_layout(dt: &Datatype) -> bool {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
..
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
return true;
|
||||||
|
};
|
||||||
|
let std = match size {
|
||||||
|
2 => (16, 10, 5, 10, 15),
|
||||||
|
4 => (32, 23, 8, 23, 127),
|
||||||
|
8 => (64, 52, 11, 52, 1023),
|
||||||
|
_ => return true, // no same-size numpy float to convert to: compare raw
|
||||||
|
};
|
||||||
|
*bit_offset == 0
|
||||||
|
&& (
|
||||||
|
*bit_precision,
|
||||||
|
*exponent_location,
|
||||||
|
*exponent_size,
|
||||||
|
*mantissa_size,
|
||||||
|
*exponent_bias,
|
||||||
|
) == (std.0, std.1, std.2, std.3, std.4)
|
||||||
|
&& *mantissa_location == 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Canonicalise a non-IEEE-layout float the way libhdf5's float->float
|
||||||
|
/// conversion presents it to h5py: as the IEEE float of the same size.
|
||||||
|
/// Assumes the implied-leading-one normalisation and the sign bit at the top
|
||||||
|
/// of the precision (what `H5Tset_precision` produces; the parser does not
|
||||||
|
/// keep either field).
|
||||||
|
fn canon_custom_float(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
unreachable!()
|
||||||
|
};
|
||||||
|
let (esize, msize) = (u32::from(*exponent_size), u32::from(*mantissa_size));
|
||||||
|
if esize == 0 || esize > 30 || msize > 64 {
|
||||||
|
return Err(format!("canon: unsupported float layout e{esize} m{msize}"));
|
||||||
|
}
|
||||||
|
let v = element_bits(b, byte_order)?;
|
||||||
|
let sign_pos = (u32::from(*bit_offset) + u32::from(*bit_precision)).saturating_sub(1);
|
||||||
|
let neg = field(v, sign_pos, 1) == 1;
|
||||||
|
let e = field(v, u32::from(*exponent_location), esize) as i64;
|
||||||
|
let m = field(v, u32::from(*mantissa_location), msize);
|
||||||
|
let emax = (1i64 << esize) - 1;
|
||||||
|
let bias = i64::from(*exponent_bias);
|
||||||
|
let mag = if e == emax {
|
||||||
|
if m == 0 { f64::INFINITY } else { f64::NAN }
|
||||||
|
} else if e == 0 {
|
||||||
|
(m as f64) * 2f64.powi((1 - bias - msize as i64) as i32)
|
||||||
|
} else {
|
||||||
|
((1u128 << msize) as f64 + m as f64) * 2f64.powi((e - bias - msize as i64) as i32)
|
||||||
|
};
|
||||||
|
let x = if neg { -mag } else { mag };
|
||||||
|
match size {
|
||||||
|
2 => out
|
||||||
|
.extend_from_slice(&clawhdf5_format::float16::f32_to_f16_bits(x as f32).to_le_bytes()),
|
||||||
|
4 => out.extend_from_slice(&(x as f32).to_le_bytes()),
|
||||||
|
8 => out.extend_from_slice(&x.to_le_bytes()),
|
||||||
|
_ => unreachable!("ieee_layout keeps other sizes raw"),
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Integers stored with a bit offset or reduced precision (N-Bit): libhdf5
|
||||||
|
/// converts them to the full-width integer of the same size, shifting the
|
||||||
|
/// value down and sign-extending from the top precision bit.
|
||||||
|
fn canon_partial_int(dt: &Datatype, b: &[u8], out: &mut Vec<u8>) -> Result<(), String> {
|
||||||
|
let Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
signed,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
unreachable!()
|
||||||
|
};
|
||||||
|
let prec = u32::from(*bit_precision);
|
||||||
|
let v = element_bits(b, byte_order)?;
|
||||||
|
let mut x = field(v, u32::from(*bit_offset), prec);
|
||||||
|
if *signed && prec > 0 && prec < 128 && field(x, prec - 1, 1) == 1 {
|
||||||
|
x |= !0u128 << prec;
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&x.to_le_bytes()[..*size as usize]);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn partial_int(dt: &Datatype) -> bool {
|
||||||
|
matches!(dt, Datatype::FixedPoint { size, bit_offset, bit_precision, .. }
|
||||||
|
if *bit_offset != 0 || u32::from(*bit_precision) != size * 8)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon_str(b: &[u8], out: &mut Vec<u8>) {
|
||||||
|
let cut = b.iter().position(|&c| c == 0).unwrap_or(b.len());
|
||||||
|
let mut s = &b[..cut];
|
||||||
|
while let [rest @ .., b' '] = s {
|
||||||
|
s = rest;
|
||||||
|
}
|
||||||
|
out.push(b'S');
|
||||||
|
out.extend_from_slice(&(s.len() as u32).to_le_bytes());
|
||||||
|
out.extend_from_slice(s);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hex(b: &[u8]) -> String {
|
||||||
|
b.iter().map(|x| format!("{x:02x}")).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn dtype_str(dt: &Datatype) -> String {
|
||||||
|
match dt {
|
||||||
|
Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
signed,
|
||||||
|
byte_order,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
format!(
|
||||||
|
"{}{}{}",
|
||||||
|
bo(byte_order),
|
||||||
|
if *signed { "i" } else { "u" },
|
||||||
|
size
|
||||||
|
)
|
||||||
|
}
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
size, byte_order, ..
|
||||||
|
} => format!("{}f{}", bo(byte_order), size),
|
||||||
|
Datatype::BitField {
|
||||||
|
size, byte_order, ..
|
||||||
|
} => format!("{}b{}", bo(byte_order), size),
|
||||||
|
Datatype::Time { size, .. } => format!("time{size}"),
|
||||||
|
Datatype::String { size, .. } => format!("S{size}"),
|
||||||
|
Datatype::Opaque { size, .. } => format!("V{size}"),
|
||||||
|
Datatype::Compound { size, members } => format!(
|
||||||
|
"{{{}}}{size}",
|
||||||
|
members
|
||||||
|
.iter()
|
||||||
|
.map(|m| format!("{}:{}", m.name, dtype_str(&m.datatype)))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(",")
|
||||||
|
),
|
||||||
|
Datatype::Reference { ref_type, .. } => format!("ref({ref_type:?})"),
|
||||||
|
Datatype::Enumeration { base_type, .. } => format!("enum({})", dtype_str(base_type)),
|
||||||
|
Datatype::VariableLength {
|
||||||
|
is_string: true, ..
|
||||||
|
} => "vlstr".into(),
|
||||||
|
Datatype::VariableLength { base_type, .. } => format!("vlen({})", dtype_str(base_type)),
|
||||||
|
Datatype::Array {
|
||||||
|
base_type,
|
||||||
|
dimensions,
|
||||||
|
} => format!("({}){dimensions:?}", dtype_str(base_type)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn bo(b: &DatatypeByteOrder) -> &'static str {
|
||||||
|
match b {
|
||||||
|
DatatypeByteOrder::LittleEndian => "<",
|
||||||
|
DatatypeByteOrder::BigEndian => ">",
|
||||||
|
DatatypeByteOrder::Vax => "vax",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn is_group(h: &ObjectHeader) -> bool {
|
||||||
|
h.messages.iter().any(|m| {
|
||||||
|
matches!(
|
||||||
|
m.msg_type,
|
||||||
|
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
||||||
|
)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
install_hook();
|
||||||
|
let path = std::env::args().nth(1).expect("usage: probe <file>");
|
||||||
|
let mut top = Map::new();
|
||||||
|
top.insert("file".into(), Value::String(path.clone()));
|
||||||
|
let data = match std::fs::read(&path) {
|
||||||
|
Ok(d) => d,
|
||||||
|
Err(err) => {
|
||||||
|
top.insert("open_error".into(), Value::String(format!("Io({err})")));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// Every address is relative to the superblock: look at the file from
|
||||||
|
// there on (past any user block), as libhdf5 does.
|
||||||
|
let hdf5: &[u8] = match signature::find_signature(&data) {
|
||||||
|
Ok(off) => &data[off..],
|
||||||
|
Err(_) => &data,
|
||||||
|
};
|
||||||
|
let sb = guarded(|| Superblock::parse(hdf5, 0).map_err(e));
|
||||||
|
let sb = match sb {
|
||||||
|
Ok(sb) => sb,
|
||||||
|
Err(msg) => {
|
||||||
|
top.insert("open_error".into(), Value::String(msg));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
top.insert("superblock_version".into(), json!(sb.version));
|
||||||
|
let ctx = Ctx {
|
||||||
|
data: hdf5,
|
||||||
|
os: sb.offset_size,
|
||||||
|
ls: sb.length_size,
|
||||||
|
base_dir: std::path::Path::new(&path)
|
||||||
|
.parent()
|
||||||
|
.map(|p| p.to_path_buf())
|
||||||
|
.unwrap_or_default(),
|
||||||
|
heaps: RefCell::new(HashMap::new()),
|
||||||
|
};
|
||||||
|
let mut objects: Vec<Value> = Vec::new();
|
||||||
|
let mut visited = HashSet::new();
|
||||||
|
let mut soft_v1 = 0u64;
|
||||||
|
// explicit DFS stack: (address, path)
|
||||||
|
let mut stack: Vec<(u64, String)> = vec![(sb.root_group_address, "/".to_string())];
|
||||||
|
while let Some((addr, p)) = stack.pop() {
|
||||||
|
if objects.len() >= MAX_OBJECTS {
|
||||||
|
top.insert("truncated".into(), json!(true));
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if !visited.insert(addr) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let mut rec = Map::new();
|
||||||
|
rec.insert("path".into(), Value::String(p.clone()));
|
||||||
|
let r = guarded(|| {
|
||||||
|
let h = ctx.header(addr)?;
|
||||||
|
Ok(h)
|
||||||
|
});
|
||||||
|
let h = match r {
|
||||||
|
Ok(h) => h,
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("kind".into(), Value::String("unknown".into()));
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
objects.push(Value::Object(rec));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let is_ds = h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.any(|m| m.msg_type == MessageType::DataLayout);
|
||||||
|
let kind = if is_ds {
|
||||||
|
"dataset"
|
||||||
|
} else if is_group(&h) || addr == sb.root_group_address {
|
||||||
|
"group"
|
||||||
|
} else if h
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.any(|m| m.msg_type == MessageType::Datatype)
|
||||||
|
{
|
||||||
|
"datatype"
|
||||||
|
} else {
|
||||||
|
"unknown"
|
||||||
|
};
|
||||||
|
rec.insert("kind".into(), Value::String(kind.into()));
|
||||||
|
if kind == "dataset"
|
||||||
|
&& let Err(msg) = guarded(|| ctx.read_dataset(&h, &mut rec))
|
||||||
|
{
|
||||||
|
rec.insert("error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
if kind != "datatype" {
|
||||||
|
match guarded(|| ctx.attrs(&h)) {
|
||||||
|
Ok(m) => {
|
||||||
|
rec.insert("attrs".into(), Value::Object(m));
|
||||||
|
}
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("attrs_error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if kind == "group" {
|
||||||
|
match guarded(|| ctx.entries(&h)) {
|
||||||
|
Ok(mut ents) => {
|
||||||
|
ents.retain(|en| {
|
||||||
|
if en.cache_type == 2 {
|
||||||
|
soft_v1 += 1;
|
||||||
|
false
|
||||||
|
} else {
|
||||||
|
true
|
||||||
|
}
|
||||||
|
});
|
||||||
|
ents.sort_by(|a, b| a.name.cmp(&b.name));
|
||||||
|
let base = if p == "/" { String::new() } else { p.clone() };
|
||||||
|
for en in ents.into_iter().rev() {
|
||||||
|
stack.push((en.object_header_address, format!("{base}/{}", en.name)));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(msg) => {
|
||||||
|
rec.insert("list_error".into(), Value::String(msg));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
objects.push(Value::Object(rec));
|
||||||
|
}
|
||||||
|
if soft_v1 > 0 {
|
||||||
|
top.insert("v1_soft_link_entries".into(), json!(soft_v1));
|
||||||
|
}
|
||||||
|
top.insert("objects".into(), Value::Array(objects));
|
||||||
|
println!("{}", Value::Object(top));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The N-Bit float of libhdf5's `test/testfiles/le_data.h5`
|
||||||
|
/// (`Nbit_float_data_le`): offset 7, precision 20, sign bit 26, exponent
|
||||||
|
/// 20+6 (bias 31), mantissa 7+13.
|
||||||
|
fn nbit_f32(byte_order: DatatypeByteOrder) -> Datatype {
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order,
|
||||||
|
bit_offset: 7,
|
||||||
|
bit_precision: 20,
|
||||||
|
exponent_location: 20,
|
||||||
|
exponent_size: 6,
|
||||||
|
mantissa_location: 7,
|
||||||
|
mantissa_size: 13,
|
||||||
|
exponent_bias: 31,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn canon_one(dt: &Datatype, bytes: &[u8]) -> Vec<u8> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
canon_custom_float(dt, bytes, &mut out).unwrap();
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn nbit_float_canonicalises_to_the_value_libhdf5_returns() {
|
||||||
|
let le = nbit_f32(DatatypeByteOrder::LittleEndian);
|
||||||
|
let be = nbit_f32(DatatypeByteOrder::BigEndian);
|
||||||
|
assert!(!ieee_layout(&le));
|
||||||
|
// 1.0: exponent = bias, mantissa 0
|
||||||
|
let one: u32 = 31 << 20;
|
||||||
|
assert_eq!(canon_one(&le, &one.to_le_bytes()), 1.0f32.to_le_bytes());
|
||||||
|
assert_eq!(canon_one(&be, &one.to_be_bytes()), 1.0f32.to_le_bytes());
|
||||||
|
// -2.1999512 (h5py's reading of the file's -2.2): sign, e = 32, m = 819
|
||||||
|
let v: u32 = (1 << 26) | (32 << 20) | (819 << 7);
|
||||||
|
assert_eq!(
|
||||||
|
canon_one(&le, &v.to_le_bytes()),
|
||||||
|
(-2.199_951_2f32).to_le_bytes()
|
||||||
|
);
|
||||||
|
assert_eq!(canon_one(&le, &[0; 4]), 0.0f32.to_le_bytes());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn ieee_floats_keep_their_raw_bytes() {
|
||||||
|
let f32le = Datatype::FloatingPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 32,
|
||||||
|
exponent_location: 23,
|
||||||
|
exponent_size: 8,
|
||||||
|
mantissa_location: 0,
|
||||||
|
mantissa_size: 23,
|
||||||
|
exponent_bias: 127,
|
||||||
|
};
|
||||||
|
assert!(ieee_layout(&f32le));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn partial_precision_int_is_shifted_and_sign_extended() {
|
||||||
|
let dt = Datatype::FixedPoint {
|
||||||
|
size: 4,
|
||||||
|
byte_order: DatatypeByteOrder::BigEndian,
|
||||||
|
signed: true,
|
||||||
|
bit_offset: 4,
|
||||||
|
bit_precision: 17,
|
||||||
|
};
|
||||||
|
assert!(partial_int(&dt));
|
||||||
|
let stored = (((-5i32) as u32) & 0x1_FFFF) << 4;
|
||||||
|
let mut out = Vec::new();
|
||||||
|
canon_partial_int(&dt, &stored.to_be_bytes(), &mut out).unwrap();
|
||||||
|
assert_eq!(out, (-5i32).to_le_bytes());
|
||||||
|
}
|
||||||
|
}
|
||||||
Executable
+259
@@ -0,0 +1,259 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Reference probe: same JSON as the Rust `conformance-probe`, produced with h5py.
|
||||||
|
|
||||||
|
Walk: iterative DFS from '/', children in sorted (UTF-8 byte) name order, hard
|
||||||
|
links only, each object once (first path wins, deduplicated by object identity).
|
||||||
|
Canonical value encoding: see harness/src/main.rs.
|
||||||
|
"""
|
||||||
|
import hashlib
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import struct
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
import h5py
|
||||||
|
|
||||||
|
try:
|
||||||
|
import hdf5plugin # noqa: F401 registers blosc/lz4/zstd/bzip2/... filters
|
||||||
|
except Exception: # pragma: no cover
|
||||||
|
pass
|
||||||
|
|
||||||
|
MAX_BYTES = 200 * 1024 * 1024
|
||||||
|
MAX_OBJECTS = 200_000
|
||||||
|
|
||||||
|
|
||||||
|
def canon_str(b, out):
|
||||||
|
if isinstance(b, str):
|
||||||
|
b = b.encode("utf-8", "surrogateescape")
|
||||||
|
b = bytes(b)
|
||||||
|
cut = b.find(b"\x00")
|
||||||
|
if cut >= 0:
|
||||||
|
b = b[:cut]
|
||||||
|
b = b.rstrip(b" ")
|
||||||
|
out += b"S" + struct.pack("<I", len(b)) + b
|
||||||
|
|
||||||
|
|
||||||
|
def simple(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return all(simple(dt.fields[n][0]) for n in dt.names)
|
||||||
|
if dt.subdtype:
|
||||||
|
return simple(dt.subdtype[0])
|
||||||
|
return dt.kind in "iufcbV"
|
||||||
|
|
||||||
|
|
||||||
|
def packed(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return np.dtype([(n, packed(dt.fields[n][0])) for n in dt.names])
|
||||||
|
if dt.subdtype:
|
||||||
|
base, shape = dt.subdtype
|
||||||
|
return np.dtype((packed(base), shape))
|
||||||
|
if dt.kind in "iufcb":
|
||||||
|
return dt.newbyteorder("<")
|
||||||
|
return dt
|
||||||
|
|
||||||
|
|
||||||
|
def canon_el(dt, val, out):
|
||||||
|
if dt.fields:
|
||||||
|
for n in dt.names:
|
||||||
|
canon_el(dt.fields[n][0], val[n], out)
|
||||||
|
return
|
||||||
|
if dt.subdtype:
|
||||||
|
base, _ = dt.subdtype
|
||||||
|
for x in np.asarray(val).reshape(-1):
|
||||||
|
canon_el(base, x, out)
|
||||||
|
return
|
||||||
|
k = dt.kind
|
||||||
|
if k in "iufcb":
|
||||||
|
out += np.asarray(val, dtype=dt).astype(dt.newbyteorder("<")).tobytes()
|
||||||
|
elif k == "V":
|
||||||
|
out += np.asarray(val, dtype=dt).tobytes()
|
||||||
|
elif k == "S":
|
||||||
|
canon_str(val, out)
|
||||||
|
elif k == "O":
|
||||||
|
if h5py.check_string_dtype(dt) is not None:
|
||||||
|
canon_str(val if val is not None else b"", out)
|
||||||
|
elif h5py.check_ref_dtype(dt) is not None:
|
||||||
|
out += b"R"
|
||||||
|
else:
|
||||||
|
base = h5py.check_vlen_dtype(dt)
|
||||||
|
if base is None:
|
||||||
|
raise TypeError(f"unhandled object dtype {dt!r}")
|
||||||
|
arr = np.asarray(val if val is not None else [], dtype=base).reshape(-1)
|
||||||
|
out += b"V" + struct.pack("<I", arr.shape[0])
|
||||||
|
if simple(base):
|
||||||
|
out += arr.astype(packed(base)).tobytes()
|
||||||
|
else:
|
||||||
|
for x in arr:
|
||||||
|
canon_el(base, x, out)
|
||||||
|
elif k == "U":
|
||||||
|
canon_str(str(val), out)
|
||||||
|
else:
|
||||||
|
raise TypeError(f"unhandled dtype kind {k} ({dt!r})")
|
||||||
|
|
||||||
|
|
||||||
|
def has_obj(dt):
|
||||||
|
if dt.fields:
|
||||||
|
return any(has_obj(dt.fields[n][0]) for n in dt.names)
|
||||||
|
if dt.subdtype:
|
||||||
|
return has_obj(dt.subdtype[0])
|
||||||
|
return dt.kind == "O"
|
||||||
|
|
||||||
|
|
||||||
|
def note_conversion(tid, dt, rec):
|
||||||
|
"""h5py converts some file types (FP8, bfloat16, x87 long double, ...) to a
|
||||||
|
different-sized numpy type; then value bytes are not comparable."""
|
||||||
|
try:
|
||||||
|
if not has_obj(dt) and tid.get_size() != dt.itemsize:
|
||||||
|
rec["converted"] = f"file type size {tid.get_size()} -> numpy {dt} ({dt.itemsize})"
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def hash_values(arr, dt, rec):
|
||||||
|
if dt.subdtype is not None:
|
||||||
|
# h5py expands an HDF5 array element type into trailing array dims
|
||||||
|
dt = dt.subdtype[0]
|
||||||
|
arr = np.asarray(arr, dtype=dt)
|
||||||
|
if simple(dt):
|
||||||
|
c = np.ascontiguousarray(arr).astype(packed(dt)).tobytes()
|
||||||
|
else:
|
||||||
|
out = bytearray()
|
||||||
|
for x in arr.reshape(-1):
|
||||||
|
canon_el(dt, x, out)
|
||||||
|
c = bytes(out)
|
||||||
|
rec["hash"] = hashlib.sha256(c).hexdigest()
|
||||||
|
rec["head"] = c[:48].hex()
|
||||||
|
|
||||||
|
|
||||||
|
def err(e):
|
||||||
|
s = f"{type(e).__name__}: {e}"
|
||||||
|
return s.splitlines()[0][:400] if s else type(e).__name__
|
||||||
|
|
||||||
|
|
||||||
|
def shape_of(s):
|
||||||
|
return "null" if s is None else list(s)
|
||||||
|
|
||||||
|
|
||||||
|
def n_bytes(shape, tid):
|
||||||
|
n = 1
|
||||||
|
for d in shape or ():
|
||||||
|
n *= d
|
||||||
|
return n * tid.get_size()
|
||||||
|
|
||||||
|
|
||||||
|
def read_attrs(obj):
|
||||||
|
out = {}
|
||||||
|
names = sorted(obj.attrs.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||||
|
for name in names:
|
||||||
|
rec = {}
|
||||||
|
try:
|
||||||
|
aid = obj.attrs.get_id(name)
|
||||||
|
rec["dtype"] = str(aid.dtype)
|
||||||
|
rec["shape"] = shape_of(aid.shape)
|
||||||
|
note_conversion(aid.get_type(), aid.dtype, rec)
|
||||||
|
if aid.shape is None:
|
||||||
|
hash_values(np.empty((0,), dtype=aid.dtype), aid.dtype, rec)
|
||||||
|
else:
|
||||||
|
val = obj.attrs[name]
|
||||||
|
hash_values(val, aid.dtype, rec)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec = {"error": err(e)}
|
||||||
|
out[name] = rec
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def main(path):
|
||||||
|
top = {"file": path}
|
||||||
|
try:
|
||||||
|
f = h5py.File(path, "r")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
top["open_error"] = err(e)
|
||||||
|
print(json.dumps(top))
|
||||||
|
return
|
||||||
|
objects = []
|
||||||
|
seen = set()
|
||||||
|
stack = [("/", None)]
|
||||||
|
while stack:
|
||||||
|
p, obj = stack.pop()
|
||||||
|
if len(objects) >= MAX_OBJECTS:
|
||||||
|
top["truncated"] = True
|
||||||
|
break
|
||||||
|
rec = {"path": p}
|
||||||
|
try:
|
||||||
|
if obj is None:
|
||||||
|
obj = f[p]
|
||||||
|
key = hash(obj.id) # h5py ObjectID hash = (fileno, object address/token)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["kind"] = "unknown"
|
||||||
|
rec["error"] = err(e)
|
||||||
|
objects.append(rec)
|
||||||
|
continue
|
||||||
|
if key in seen:
|
||||||
|
continue
|
||||||
|
seen.add(key)
|
||||||
|
if isinstance(obj, h5py.Dataset):
|
||||||
|
kind = "dataset"
|
||||||
|
elif isinstance(obj, h5py.Group):
|
||||||
|
kind = "group"
|
||||||
|
elif isinstance(obj, h5py.Datatype):
|
||||||
|
kind = "datatype"
|
||||||
|
else:
|
||||||
|
kind = "unknown"
|
||||||
|
rec["kind"] = kind
|
||||||
|
if kind == "dataset":
|
||||||
|
try:
|
||||||
|
dt = obj.dtype
|
||||||
|
rec["dtype"] = str(dt)
|
||||||
|
rec["shape"] = shape_of(obj.shape)
|
||||||
|
note_conversion(obj.id.get_type(), dt, rec)
|
||||||
|
if obj.shape is None:
|
||||||
|
hash_values(np.empty((0,), dtype=dt), dt, rec)
|
||||||
|
elif n_bytes(obj.shape, obj.id.get_type()) > MAX_BYTES:
|
||||||
|
rec["skipped"] = "too large"
|
||||||
|
else:
|
||||||
|
arr = np.empty(obj.shape, dtype=dt)
|
||||||
|
if arr.size:
|
||||||
|
try:
|
||||||
|
obj.read_direct(arr)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
arr = obj[()]
|
||||||
|
hash_values(arr, dt, rec)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["error"] = err(e)
|
||||||
|
if kind != "datatype":
|
||||||
|
try:
|
||||||
|
rec["attrs"] = read_attrs(obj)
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["attrs_error"] = err(e)
|
||||||
|
if kind == "group":
|
||||||
|
try:
|
||||||
|
names = sorted(obj.keys(), key=lambda s: s.encode("utf-8", "surrogateescape"))
|
||||||
|
base = "" if p == "/" else p
|
||||||
|
kids = []
|
||||||
|
for n in names:
|
||||||
|
try:
|
||||||
|
link = obj.get(n, getlink=True)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
link = None
|
||||||
|
if link is not None and not isinstance(link, h5py.HardLink):
|
||||||
|
continue
|
||||||
|
kids.append(f"{base}/{n}")
|
||||||
|
for k in reversed(kids):
|
||||||
|
stack.append((k, None))
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
rec["list_error"] = err(e)
|
||||||
|
objects.append(rec)
|
||||||
|
top["objects"] = objects
|
||||||
|
print(json.dumps(top), flush=True)
|
||||||
|
# Exit without tearing down the h5py objects: freeing them for some files
|
||||||
|
# that hold references (hdf5's h5repack_attr_refs.h5, cve-2024-32623.h5)
|
||||||
|
# makes libhdf5 2.0 abort with "free(): chunks in smallbin corrupted"
|
||||||
|
# about half the time. That happens after the reading is done, so it says
|
||||||
|
# nothing about what h5py read, but it flipped those files between ok and
|
||||||
|
# h5py-cannot-read from one run to the next.
|
||||||
|
os._exit(0)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main(sys.argv[1])
|
||||||
@@ -0,0 +1,335 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""report.py <results_dir> <CONFORMANCE.md> <corpus_dir>
|
||||||
|
|
||||||
|
Render the sweep's results (compare.py's results.json plus the raw per-side
|
||||||
|
runs) as CONFORMANCE.md, and write <results_dir>/report-meta.json (commit,
|
||||||
|
date, versions) for check.py --update.
|
||||||
|
"""
|
||||||
|
import collections
|
||||||
|
import datetime
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import platform
|
||||||
|
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy
|
||||||
|
|
||||||
|
try:
|
||||||
|
import hdf5plugin
|
||||||
|
HDF5PLUGIN = hdf5plugin.version
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
HDF5PLUGIN = "not installed"
|
||||||
|
|
||||||
|
R, OUT_MD, CORPUS = sys.argv[1], sys.argv[2], sys.argv[3]
|
||||||
|
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
ROOT = os.path.dirname(HERE)
|
||||||
|
CLASSES = ["ok", "our-error", "mismatch", "h5py-cannot-read", "panic", "hang", "crash", "oom"]
|
||||||
|
|
||||||
|
|
||||||
|
def sh(*cmd, cwd=ROOT):
|
||||||
|
try:
|
||||||
|
return subprocess.run(cmd, cwd=cwd, capture_output=True, text=True, timeout=30).stdout.strip()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def cpu_model():
|
||||||
|
try:
|
||||||
|
for ln in open("/proc/cpuinfo"):
|
||||||
|
if ln.startswith(("model name", "Model")):
|
||||||
|
return ln.split(":", 1)[1].strip()
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return platform.processor() or "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def mem_gib():
|
||||||
|
try:
|
||||||
|
for ln in open("/proc/meminfo"):
|
||||||
|
if ln.startswith("MemTotal:"):
|
||||||
|
return f"{int(ln.split()[1]) / 1048576:.0f} GiB"
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return "?"
|
||||||
|
|
||||||
|
|
||||||
|
res = json.load(open(os.path.join(R, "results.json")))
|
||||||
|
meta_run = json.load(open(os.path.join(R, "meta.json"))) if os.path.exists(os.path.join(R, "meta.json")) else {}
|
||||||
|
rows = res["rows"]
|
||||||
|
issues = res.get("issues", {})
|
||||||
|
|
||||||
|
# safe.directory: a checkout owned by another user (a container) is still ours to read
|
||||||
|
commit = sh("git", "-c", "safe.directory=*", "rev-parse", "HEAD") or os.environ.get("GITHUB_SHA", "unknown")
|
||||||
|
lib_dirty = sh("git", "-c", "safe.directory=*", "status", "--porcelain", "--", "crates", "Cargo.toml")
|
||||||
|
h5dump_v = sh("h5dump", "--version").replace("h5dump: ", "")
|
||||||
|
meta = {
|
||||||
|
"date": datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M UTC"),
|
||||||
|
"commit": commit + (" (library sources modified)" if lib_dirty else ""),
|
||||||
|
"reference": f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}",
|
||||||
|
}
|
||||||
|
json.dump(meta, open(os.path.join(R, "report-meta.json"), "w"), indent=1)
|
||||||
|
|
||||||
|
pins = []
|
||||||
|
for ln in open(os.path.join(HERE, "corpus.txt")):
|
||||||
|
if ln.strip() and not ln.lstrip().startswith("#"):
|
||||||
|
name, url, rev, root, *_ = ln.split()
|
||||||
|
pins.append((name, url, rev, root))
|
||||||
|
|
||||||
|
by_corpus = collections.defaultdict(collections.Counter)
|
||||||
|
for r in rows:
|
||||||
|
by_corpus[r["corpus"]][r["class"]] += 1
|
||||||
|
total = collections.Counter(r["class"] for r in rows)
|
||||||
|
|
||||||
|
|
||||||
|
def ex_list(files, n=3):
|
||||||
|
s = ", ".join(f"`{f}`" for f in files[:n])
|
||||||
|
return s + (f" (+{len(files) - n} more)" if len(files) > n else "")
|
||||||
|
|
||||||
|
|
||||||
|
# --- known causes that are not clawhdf5 bugs --------------------------------
|
||||||
|
def is_h5py_be_vlen(i):
|
||||||
|
"""h5py returns the elements of a VL sequence of a big-endian base type
|
||||||
|
with their file (big-endian) bytes but a native-endian dtype."""
|
||||||
|
return (i["kind"] == "mismatch" and i["key"] in ("values", "attr-values")
|
||||||
|
and (i.get("ref_dtype") == "object") and (i.get("ours_dtype") or "").startswith("vlen(")
|
||||||
|
and ">" in (i.get("ours_dtype") or ""))
|
||||||
|
|
||||||
|
|
||||||
|
known = collections.defaultdict(list)
|
||||||
|
for r in rows:
|
||||||
|
if r["class"] != "mismatch":
|
||||||
|
continue
|
||||||
|
iss = issues.get(r["file"], [])
|
||||||
|
if iss and all(is_h5py_be_vlen(i) for i in iss):
|
||||||
|
known["h5py-be-vlen"].append(r["file"])
|
||||||
|
|
||||||
|
|
||||||
|
# --- the CVE corpus: clawhdf5 vs h5dump vs h5py ------------------------------
|
||||||
|
def side(run, name):
|
||||||
|
p = os.path.join(R, "runs", run, name)
|
||||||
|
if not os.path.exists(p + ".rc"):
|
||||||
|
return None
|
||||||
|
rc = int(open(p + ".rc").read().strip() or -1)
|
||||||
|
err = open(p + ".err", errors="replace").read()
|
||||||
|
try:
|
||||||
|
j = json.load(open(p + ".json"))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
j = None
|
||||||
|
return rc, err, j
|
||||||
|
|
||||||
|
|
||||||
|
def outcome(s, rust=False):
|
||||||
|
"""-> (bucket, text). bucket in read / error / panic / crash / hang / oom."""
|
||||||
|
if s is None:
|
||||||
|
return "missing", "not run"
|
||||||
|
rc, err, j = s
|
||||||
|
if rc in (137, 124):
|
||||||
|
return "hang", "hang (killed at timeout)"
|
||||||
|
if "memory allocation of" in err or "MemoryError" in err or "bad_alloc" in err or "Cannot allocate" in err:
|
||||||
|
return "oom", "out of memory"
|
||||||
|
if rust and (rc == 101 or "PANIC:" in err):
|
||||||
|
return "panic", "panic"
|
||||||
|
if "overflowed its stack" in err:
|
||||||
|
return "crash", "stack overflow"
|
||||||
|
if rc == 139:
|
||||||
|
return "crash", "SIGSEGV"
|
||||||
|
if rc == 134:
|
||||||
|
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||||
|
if rc > 128:
|
||||||
|
return "crash", f"signal {rc - 128}"
|
||||||
|
if j is None:
|
||||||
|
return ("error", "error exit") if rc in (0, 1) else ("crash", f"exit {rc}")
|
||||||
|
if "open_error" in j:
|
||||||
|
return "error", "open error"
|
||||||
|
objs = j.get("objects", [])
|
||||||
|
ne = sum(1 for o in objs for k in ("error", "attrs_error", "list_error") if k in o)
|
||||||
|
ne += sum(1 for o in objs for a in (o.get("attrs") or {}).values() if "error" in a)
|
||||||
|
return "read", f"read {len(objs)} obj" + (f", {ne} errors" if ne else "")
|
||||||
|
|
||||||
|
|
||||||
|
def h5dump_outcome(s):
|
||||||
|
if s is None:
|
||||||
|
return "missing", "not run"
|
||||||
|
rc, err, _ = s
|
||||||
|
if rc in (137, 124):
|
||||||
|
return "hang", "hang (killed at timeout)"
|
||||||
|
if "memory allocation" in err or "Cannot allocate" in err:
|
||||||
|
return "oom", "out of memory"
|
||||||
|
if rc == 139:
|
||||||
|
return "crash", "SIGSEGV"
|
||||||
|
if rc == 134:
|
||||||
|
return "crash", "SIGABRT" + (" (heap corruption)" if ("corrupted" in err or "free()" in err) else "")
|
||||||
|
if rc > 128:
|
||||||
|
return "crash", f"signal {rc - 128}"
|
||||||
|
return ("read", "ok") if rc == 0 else ("error", "error exit")
|
||||||
|
|
||||||
|
|
||||||
|
cve_rows = []
|
||||||
|
buckets = {"clawhdf5": collections.Counter(), "h5dump": collections.Counter(), "h5py": collections.Counter()}
|
||||||
|
ours_panic = {r["file"] for r in rows if r["class"] == "panic"}
|
||||||
|
for r in rows:
|
||||||
|
if r["corpus"] != "cve_hdf5":
|
||||||
|
continue
|
||||||
|
run = r["file"].replace("/", "__")
|
||||||
|
o = outcome(side(run, "ours"), rust=True)
|
||||||
|
if o[0] == "read" and r["file"] in ours_panic:
|
||||||
|
o = ("panic", "caught panic")
|
||||||
|
p = outcome(side(run, "ref"))
|
||||||
|
d = h5dump_outcome(side(run, "h5dump"))
|
||||||
|
buckets["clawhdf5"][o[0]] += 1
|
||||||
|
buckets["h5py"][p[0]] += 1
|
||||||
|
buckets["h5dump"][d[0]] += 1
|
||||||
|
cve_rows.append((r["file"].split("/", 1)[1], d[1], p[1], o[1], r["class"]))
|
||||||
|
|
||||||
|
# --- render -----------------------------------------------------------------
|
||||||
|
L = []
|
||||||
|
w = L.append
|
||||||
|
w("# clawhdf5 conformance report")
|
||||||
|
w("")
|
||||||
|
w("Every HDF5 file of eight public corpora (pinned by commit) is read twice — by")
|
||||||
|
w("clawhdf5 (`conformance/probe`, the same `clawhdf5-format` calls the facade")
|
||||||
|
w("makes) and by h5py/libhdf5 (`conformance/ref.py`) — and the two readings are")
|
||||||
|
w("compared object by object: the set of hard-linked objects, each dataset's and")
|
||||||
|
w("attribute's shape, and a SHA-256 of its values in a canonical encoding. The")
|
||||||
|
w("CVE corpus is also run through `h5dump`. Each side runs under a timeout and an")
|
||||||
|
w("address-space limit, so a hang, crash or runaway allocation is recorded, not")
|
||||||
|
w("fatal. This file is generated by `conformance/run.sh`; do not edit it by hand.")
|
||||||
|
w("")
|
||||||
|
w("## Run")
|
||||||
|
w("")
|
||||||
|
w("| | |")
|
||||||
|
w("|---|---|")
|
||||||
|
w(f"| date | {meta['date']} |")
|
||||||
|
w(f"| clawhdf5 commit | `{meta['commit']}` |")
|
||||||
|
w(f"| machine | `{platform.node()}`: {cpu_model()}, {os.cpu_count()} CPUs, {mem_gib()}, {platform.system()} {platform.release()} {platform.machine()} |")
|
||||||
|
w(f"| command | `{os.environ.get('CONFORMANCE_CMD', 'conformance/run.sh')}` |")
|
||||||
|
w(f"| rustc | {sh('rustc', '-V')} |")
|
||||||
|
w(f"| reference | h5py {h5py.__version__}, HDF5 {h5py.version.hdf5_version}, numpy {numpy.__version__}, hdf5plugin {HDF5PLUGIN}, Python {platform.python_version()} |")
|
||||||
|
w(f"| h5dump | {h5dump_v} (CVE corpus only) |")
|
||||||
|
if meta_run:
|
||||||
|
w(f"| limits | {meta_run.get('timeout_s')} s timeout (SIGKILL), {int(meta_run.get('mem_kb', 0)) // 1024} MiB address space, per process; {meta_run.get('jobs')} files in parallel |")
|
||||||
|
w(f"| runtime | {meta_run.get('probe_seconds')} s probing + comparing ({meta_run.get('build_seconds')} s fetch/build before it) |")
|
||||||
|
w("")
|
||||||
|
w("## Results")
|
||||||
|
w("")
|
||||||
|
w("A file's class is the first that applies:")
|
||||||
|
w("")
|
||||||
|
w("- **panic / hang / crash / oom** — clawhdf5 panicked (caught per object or not), hit the timeout, died on a signal, or failed an allocation. The CI gate fails on any of these.")
|
||||||
|
w("- **h5py-cannot-read** — libhdf5 could not open the file (or itself crashed or hung). Nothing to compare against; most are the deliberately malformed CVE reproducers.")
|
||||||
|
w("- **our-error** — clawhdf5 returned an error for something h5py reads.")
|
||||||
|
w("- **mismatch** — both read it, but the shapes, values, object set or attribute set differ.")
|
||||||
|
w("- **ok** — every object h5py reads, clawhdf5 reads identically.")
|
||||||
|
w("")
|
||||||
|
w("| corpus | files | " + " | ".join(CLASSES) + " |")
|
||||||
|
w("|---" * (len(CLASSES) + 2) + "|")
|
||||||
|
for c in sorted(by_corpus):
|
||||||
|
cnt = by_corpus[c]
|
||||||
|
w(f"| {c} | {sum(cnt.values())} | " + " | ".join(str(cnt.get(k, 0)) for k in CLASSES) + " |")
|
||||||
|
w(f"| **all** | **{len(rows)}** | " + " | ".join(f"**{total.get(k, 0)}**" for k in CLASSES) + " |")
|
||||||
|
w("")
|
||||||
|
n_known = sum(len(v) for v in known.values())
|
||||||
|
if n_known:
|
||||||
|
w(f"{n_known} of the {total.get('mismatch', 0)} mismatches are a known h5py bug, not ours (see *Known not-our-bug*).")
|
||||||
|
w("")
|
||||||
|
w("Corpora (fetched by `conformance/fetch-corpus.sh` into the gitignored `conformance/.cache/`):")
|
||||||
|
w("")
|
||||||
|
w("| corpus | source | commit |")
|
||||||
|
w("|---|---|---|")
|
||||||
|
for name, url, rev, root in pins:
|
||||||
|
w(f"| {name} | {url.removesuffix('.git')}" + ("" if root == "." else f" (`{root}`)") + f" | `{rev[:12]}` |")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Panics, hangs, crashes, out-of-memory")
|
||||||
|
w("")
|
||||||
|
if not res["panics"]:
|
||||||
|
w("None.")
|
||||||
|
else:
|
||||||
|
for p in res["panics"]:
|
||||||
|
w(f"- `{p['file']}` [{p['class']}] {p['detail']}")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Our-error root causes")
|
||||||
|
w("")
|
||||||
|
w("Grouped by normalised error message. *files* counts files whose class this cause affects.")
|
||||||
|
w("")
|
||||||
|
w("| files | objects | error | examples |")
|
||||||
|
w("|---:|---:|---|---|")
|
||||||
|
for k, v in res["root_causes"].items():
|
||||||
|
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||||
|
w("")
|
||||||
|
w("## Mismatch root causes")
|
||||||
|
w("")
|
||||||
|
w("| files | objects | cause | examples |")
|
||||||
|
w("|---:|---:|---|---|")
|
||||||
|
for k, v in res["mismatch_causes"].items():
|
||||||
|
w(f"| {v['files']} | {v['count']} | `{k.replace('|', '/')}` | {ex_list(v['file_list'])} |")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## CVE corpus: clawhdf5 vs h5dump vs h5py")
|
||||||
|
w("")
|
||||||
|
w(f"The {len(cve_rows)} files of [HDFGroup/cve_hdf5](https://github.com/HDFGroup/cve_hdf5) — reproducers for")
|
||||||
|
w("published libhdf5 CVEs and fuzzer finds. *read* = produced output (possibly with per-object")
|
||||||
|
w("errors), *error* = refused cleanly. h5dump exits non-zero on any error anywhere in a file, so")
|
||||||
|
w("its read/error split is not comparable with the other two rows; the panic, crash, hang and oom")
|
||||||
|
w("columns are.")
|
||||||
|
w("")
|
||||||
|
w("| tool | read | error | panic | crash | hang | oom |")
|
||||||
|
w("|---|---:|---:|---:|---:|---:|---:|")
|
||||||
|
for tool, label in (("clawhdf5", "clawhdf5"), ("h5dump", f"h5dump {h5dump_v.split()[-1] if h5dump_v else ''}"),
|
||||||
|
("h5py", f"h5py {h5py.__version__} / HDF5 {h5py.version.hdf5_version}")):
|
||||||
|
b = buckets[tool]
|
||||||
|
w(f"| {label} | " + " | ".join(str(b.get(k, 0)) for k in ("read", "error", "panic", "crash", "hang", "oom")) + " |")
|
||||||
|
w("")
|
||||||
|
w("<details><summary>Per-file outcomes</summary>")
|
||||||
|
w("")
|
||||||
|
w("| file | h5dump | h5py | clawhdf5 | class |")
|
||||||
|
w("|---|---|---|---|---|")
|
||||||
|
for f, d, p, o, cls in cve_rows:
|
||||||
|
w(f"| {f} | {d} | {p} | {o} | {cls} |")
|
||||||
|
w("")
|
||||||
|
w("</details>")
|
||||||
|
w("")
|
||||||
|
|
||||||
|
w("## Known not-our-bug")
|
||||||
|
w("")
|
||||||
|
w("- **h5py big-endian variable-length sequences.** h5py returns the elements of a VL sequence")
|
||||||
|
w(" whose base type is big-endian with the file's big-endian bytes but a native (little-endian)")
|
||||||
|
w(" numpy dtype, so the values it reports are byte-swapped garbage; `h5dump` prints the values")
|
||||||
|
w(" clawhdf5 reads. Reproducer: `h5py.vlen_dtype(np.dtype('>f4'))` dataset holding `[1.0, 2.0]`")
|
||||||
|
w(" reads back in h5py as `[4.6e-41, 9.0e-44]`. Affected here: "
|
||||||
|
+ (ex_list(sorted(known["h5py-be-vlen"]), 10) if known["h5py-be-vlen"] else "none") + ".")
|
||||||
|
w("- **Non-IEEE floats and partial-precision integers (N-Bit).** libhdf5 converts a float whose")
|
||||||
|
w(" bit layout is not IEEE (e.g. `H5Tset_precision` for the N-Bit filter) or an integer with a")
|
||||||
|
w(" bit offset / reduced precision into the plain numpy type of the same size. The probe")
|
||||||
|
w(" compares such values as converted numbers, not raw file bytes (before 2026-09-25 it compared")
|
||||||
|
w(" raw bytes, which reported every N-Bit float dataset as a mismatch).")
|
||||||
|
if res["incomparable"]:
|
||||||
|
w("- **Types h5py widens.** Where h5py reads a type into a numpy type of a different size")
|
||||||
|
w(" (FP8 -> float16, bfloat16 -> float32, x87 long double -> float128) the values are not")
|
||||||
|
w(" compared (shape and presence still are): "
|
||||||
|
+ ", ".join(f"{k} ({n}x)" for k, n in res["incomparable"]) + ".")
|
||||||
|
w("- **References** are compared by presence only (`R`), not by target.")
|
||||||
|
w("")
|
||||||
|
if res.get("ref_only_errors"):
|
||||||
|
w("## Objects h5py fails on but clawhdf5 reads")
|
||||||
|
w("")
|
||||||
|
for k, n in res["ref_only_errors"][:15]:
|
||||||
|
w(f"- {n} x `{k}`")
|
||||||
|
w("")
|
||||||
|
w("## Reproduce")
|
||||||
|
w("")
|
||||||
|
w("```sh")
|
||||||
|
w("# needs: Rust, python3 with h5py numpy hdf5plugin (conformance/requirements.txt), h5dump (hdf5-tools), git")
|
||||||
|
w("CLAWHDF5_PYTHON=/path/to/venv/bin/python conformance/run.sh")
|
||||||
|
w("```")
|
||||||
|
w("")
|
||||||
|
w("The corpus (about 450 MB of sparse checkouts) is cached in `conformance/.cache/`; results for")
|
||||||
|
w("every file, both sides' raw JSON and stderr, are in `conformance/.cache/results/`.")
|
||||||
|
w("`conformance/baseline.json` holds the ok files the nightly CI job (`.gitea/workflows/conformance.yml`)")
|
||||||
|
w("must keep; `conformance/run.sh --update-baseline` rewrites it.")
|
||||||
|
|
||||||
|
with open(OUT_MD, "w") as fh:
|
||||||
|
fh.write("\n".join(L) + "\n")
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
# The reference side of the conformance sweep. Pinned so the nightly job and a
|
||||||
|
# local run compare against the same libhdf5 (h5py wheels bundle it).
|
||||||
|
h5py==3.16.0
|
||||||
|
numpy==2.5.3
|
||||||
|
hdf5plugin==7.1.0
|
||||||
|
netCDF4==1.7.4
|
||||||
Executable
+88
@@ -0,0 +1,88 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# conformance/run.sh — the clawhdf5 conformance sweep, end to end.
|
||||||
|
#
|
||||||
|
# fetch the pinned corpora (cached) -> build the probe -> probe every file
|
||||||
|
# with clawhdf5 and with h5py (and h5dump for the CVE corpus), each under a
|
||||||
|
# timeout and a memory limit -> compare -> write CONFORMANCE.md -> check the
|
||||||
|
# result against conformance/baseline.json.
|
||||||
|
#
|
||||||
|
# Usage: conformance/run.sh [--no-fetch] [--no-report] [--update-baseline]
|
||||||
|
#
|
||||||
|
# Environment:
|
||||||
|
# CLAWHDF5_PYTHON python with h5py, numpy, hdf5plugin (default: repo .venv, then python3)
|
||||||
|
# CONFORMANCE_CACHE corpus / build / results cache (default: conformance/.cache)
|
||||||
|
# CONFORMANCE_OUT results directory (default: $CONFORMANCE_CACHE/results)
|
||||||
|
# CONFORMANCE_REPORT report path (default: CONFORMANCE.md at the repo root)
|
||||||
|
# JOBS parallel files (default: nproc)
|
||||||
|
# CONFORMANCE_PROBE use this prebuilt probe binary instead of building one
|
||||||
|
# TMO / MEM_KB per-process timeout in seconds (20) / address-space limit in KiB (4 GiB)
|
||||||
|
#
|
||||||
|
# Exit status: 0 = gate passed; 1 = a panic/hang/crash/oom in clawhdf5, or the
|
||||||
|
# ok count fell below the baseline, or a baseline-ok file regressed; 2 = setup error.
|
||||||
|
set -euo pipefail
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
ROOT="$(cd "$HERE/.." && pwd)"
|
||||||
|
FETCH=1 REPORT=1 UPDATE=0
|
||||||
|
for a in "$@"; do
|
||||||
|
case "$a" in
|
||||||
|
--no-fetch) FETCH=0 ;;
|
||||||
|
--no-report) REPORT=0 ;;
|
||||||
|
--update-baseline) UPDATE=1 ;;
|
||||||
|
-h|--help) sed -n '2,23p' "$0"; exit 0 ;;
|
||||||
|
*) echo "unknown argument: $a" >&2; exit 2 ;;
|
||||||
|
esac
|
||||||
|
done
|
||||||
|
|
||||||
|
export PATH="$HOME/.cargo/bin:$PATH"
|
||||||
|
CACHE="${CONFORMANCE_CACHE:-$HERE/.cache}"
|
||||||
|
mkdir -p "$CACHE"; CACHE="$(cd "$CACHE" && pwd)"
|
||||||
|
OUT="${CONFORMANCE_OUT:-$CACHE/results}"
|
||||||
|
REPORT_PATH="${CONFORMANCE_REPORT:-$ROOT/CONFORMANCE.md}"
|
||||||
|
JOBS="${JOBS:-$(nproc 2>/dev/null || echo 4)}"
|
||||||
|
if [ -n "${CLAWHDF5_PYTHON:-}" ]; then PY="$CLAWHDF5_PYTHON"
|
||||||
|
elif [ -x "$ROOT/.venv/bin/python" ]; then PY="$ROOT/.venv/bin/python"
|
||||||
|
else PY="$(command -v python3)"; fi
|
||||||
|
export PY TMO="${TMO:-20}" MEM_KB="${MEM_KB:-4194304}"
|
||||||
|
command -v h5dump >/dev/null || { echo "error: h5dump not found (install hdf5-tools)" >&2; exit 2; }
|
||||||
|
"$PY" -c 'import h5py, numpy, hdf5plugin' || { echo "error: $PY lacks h5py/numpy/hdf5plugin" >&2; exit 2; }
|
||||||
|
|
||||||
|
t0=$(date +%s)
|
||||||
|
[ "$FETCH" = 1 ] && bash "$HERE/fetch-corpus.sh" "$CACHE"
|
||||||
|
C="$CACHE/corpus"
|
||||||
|
[ -d "$C" ] || { echo "error: no corpus in $C (run without --no-fetch)" >&2; exit 2; }
|
||||||
|
|
||||||
|
if [ -n "${CONFORMANCE_PROBE:-}" ]; then
|
||||||
|
export PROBE="$CONFORMANCE_PROBE" # a prebuilt probe, e.g. an older one for a before/after
|
||||||
|
else
|
||||||
|
echo "== building the probe"
|
||||||
|
CARGO_TARGET_DIR="${CARGO_TARGET_DIR:-$CACHE/target}" \
|
||||||
|
cargo build -q --release --manifest-path "$HERE/probe/Cargo.toml"
|
||||||
|
export PROBE="${CARGO_TARGET_DIR:-$CACHE/target}/release/conformance-probe"
|
||||||
|
fi
|
||||||
|
t1=$(date +%s)
|
||||||
|
|
||||||
|
rm -rf "$OUT"; mkdir -p "$OUT"
|
||||||
|
"$PY" "$HERE/list_files.py" "$C" > "$OUT/files.txt"
|
||||||
|
echo "== probing $(wc -l <"$OUT/files.txt") files, $JOBS at a time (timeout ${TMO}s, limit $((MEM_KB / 1024)) MiB)"
|
||||||
|
export C OUT HERE
|
||||||
|
# The shell's "Segmentation fault (core dumped)" notices go to probe.log; the
|
||||||
|
# signals themselves are recorded in each side's .rc.
|
||||||
|
xargs -a "$OUT/files.txt" -d '\n' -P "$JOBS" -I{} bash -c '
|
||||||
|
f="$1"; d="$OUT/runs/${f//\//__}"
|
||||||
|
case "$f" in cve_hdf5/*) export WITH_H5DUMP=1 ;; esac
|
||||||
|
"$HERE/run_one.sh" "$C/$f" "$d"' _ {} 2>"$OUT/probe.log"
|
||||||
|
echo "== comparing"
|
||||||
|
"$PY" "$HERE/compare.py" "$OUT" >/dev/null
|
||||||
|
t2=$(date +%s)
|
||||||
|
cat > "$OUT/meta.json" <<EOF
|
||||||
|
{"build_seconds": $((t1 - t0)), "probe_seconds": $((t2 - t1)), "jobs": $JOBS, "timeout_s": $TMO, "mem_kb": $MEM_KB}
|
||||||
|
EOF
|
||||||
|
export CONFORMANCE_CMD="${CONFORMANCE_CMD:-conformance/run.sh${*:+ $*}}"
|
||||||
|
if [ "$REPORT" = 1 ]; then
|
||||||
|
"$PY" "$HERE/report.py" "$OUT" "$REPORT_PATH" "$C"
|
||||||
|
echo "== wrote $REPORT_PATH"
|
||||||
|
fi
|
||||||
|
if [ "$UPDATE" = 1 ]; then
|
||||||
|
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json" --update
|
||||||
|
fi
|
||||||
|
"$PY" "$HERE/check.py" "$OUT" "$HERE/baseline.json"
|
||||||
Executable
+27
@@ -0,0 +1,27 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# run_one.sh <file> <outdir>
|
||||||
|
#
|
||||||
|
# Probe one file with clawhdf5 (PROBE) and with h5py (PY ref.py), and with
|
||||||
|
# h5dump too when WITH_H5DUMP is set. Each side runs under a timeout (TMO
|
||||||
|
# seconds, SIGKILL) and an address-space limit (MEM_KB), with core dumps off.
|
||||||
|
# Writes <outdir>/<side>.{json,err,rc}; rc 137 = killed by the timeout.
|
||||||
|
set -u
|
||||||
|
f="$1"; out="$2"; mkdir -p "$out"
|
||||||
|
HERE="$(cd "$(dirname "$0")" && pwd)"
|
||||||
|
: "${PROBE:?PROBE must name the conformance-probe binary}"
|
||||||
|
: "${PY:?PY must name a python with h5py}"
|
||||||
|
TMO="${TMO:-20}"
|
||||||
|
MEM_KB="${MEM_KB:-4194304}"
|
||||||
|
run() { # name cmd...
|
||||||
|
local name=$1; shift
|
||||||
|
( ulimit -v "$MEM_KB"; ulimit -c 0; RUST_BACKTRACE=1 exec timeout -s KILL "$TMO" "$@" ) \
|
||||||
|
>"$out/$name.json" 2>"$out/$name.err"
|
||||||
|
echo $? >"$out/$name.rc"
|
||||||
|
}
|
||||||
|
run ours "$PROBE" "$f"
|
||||||
|
run ref "$PY" "$HERE/ref.py" "$f"
|
||||||
|
if [ -n "${WITH_H5DUMP:-}" ]; then
|
||||||
|
run h5dump h5dump "$f"
|
||||||
|
: >"$out/h5dump.json" # h5dump's text dump is not compared, only its exit status
|
||||||
|
fi
|
||||||
|
exit 0
|
||||||
@@ -477,10 +477,34 @@ fn write_string_dataset(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `/meta`'s attributes, failing if any of them cannot be read.
|
||||||
|
///
|
||||||
|
/// `Group::attrs` leaves out an attribute it cannot decode. For the store's
|
||||||
|
/// settings that would silently fall back to defaults (e.g. `float16`, the
|
||||||
|
/// WAL mark), so an unreadable attribute is an error here, as it was before
|
||||||
|
/// `attrs` became tolerant.
|
||||||
|
fn meta_attrs(
|
||||||
|
file: &clawhdf5::File,
|
||||||
|
) -> Result<std::collections::HashMap<String, AttrValue>, MemoryError> {
|
||||||
|
let meta = file
|
||||||
|
.group("meta")
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
||||||
|
let (attrs, errors) = meta
|
||||||
|
.attrs_with_errors()
|
||||||
|
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
||||||
|
if let Some(e) = errors.first() {
|
||||||
|
return Err(MemoryError::Schema(format!(
|
||||||
|
"cannot read /meta attrs: {} unreadable, first: {e}",
|
||||||
|
errors.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(attrs)
|
||||||
|
}
|
||||||
|
|
||||||
/// Validate an HDF5 file has the correct schema and load all data.
|
/// Validate an HDF5 file has the correct schema and load all data.
|
||||||
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
/// Read the checkpoint's [`WalMark`] from `/meta`, if it has one.
|
||||||
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
pub fn read_wal_mark(file: &clawhdf5::File) -> Option<WalMark> {
|
||||||
let attrs = file.group("meta").ok()?.attrs().ok()?;
|
let attrs = meta_attrs(file).ok()?;
|
||||||
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
let len = match attrs.get(WAL_APPLIED_LEN_ATTR)? {
|
||||||
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
AttrValue::I64(v) => u64::try_from(*v).ok()?,
|
||||||
_ => return None,
|
_ => return None,
|
||||||
@@ -498,10 +522,7 @@ pub fn read_signature(
|
|||||||
file: &clawhdf5::File,
|
file: &clawhdf5::File,
|
||||||
) -> Result<Option<crate::signing::StoredSignature>, MemoryError> {
|
) -> Result<Option<crate::signing::StoredSignature>, MemoryError> {
|
||||||
use crate::signing::{Manifest, StoredSignature, from_hex};
|
use crate::signing::{Manifest, StoredSignature, from_hex};
|
||||||
let attrs = file
|
let attrs = meta_attrs(file)?;
|
||||||
.group("meta")
|
|
||||||
.and_then(|g| g.attrs())
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
|
||||||
let version = match attrs.get(SIG_VERSION_ATTR) {
|
let version = match attrs.get(SIG_VERSION_ATTR) {
|
||||||
None => return Ok(None),
|
None => return Ok(None),
|
||||||
Some(AttrValue::I64(v)) => *v,
|
Some(AttrValue::I64(v)) => *v,
|
||||||
@@ -552,18 +573,14 @@ pub fn read_signature(
|
|||||||
|
|
||||||
/// Read the checkpoint bookkeeping from `/meta`.
|
/// Read the checkpoint bookkeeping from `/meta`.
|
||||||
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
pub fn read_checkpoint_meta(file: &clawhdf5::File) -> CheckpointMeta {
|
||||||
let ann_generation = file
|
let ann_generation =
|
||||||
.group("meta")
|
meta_attrs(file)
|
||||||
.ok()
|
.ok()
|
||||||
.and_then(|g| g.attrs().ok())
|
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
||||||
.and_then(|attrs| match attrs.get(ANN_GENERATION_ATTR) {
|
Some(AttrValue::I64(v)) => Some(*v as u64),
|
||||||
Some(AttrValue::I64(v)) => Some(*v as u64),
|
_ => None,
|
||||||
_ => None,
|
});
|
||||||
});
|
let signed = meta_attrs(file).is_ok_and(|attrs| attrs.contains_key(SIG_VERSION_ATTR));
|
||||||
let signed = file
|
|
||||||
.group("meta")
|
|
||||||
.and_then(|g| g.attrs())
|
|
||||||
.is_ok_and(|attrs| attrs.contains_key(SIG_VERSION_ATTR));
|
|
||||||
CheckpointMeta {
|
CheckpointMeta {
|
||||||
wal_applied: read_wal_mark(file),
|
wal_applied: read_wal_mark(file),
|
||||||
ann_generation,
|
ann_generation,
|
||||||
@@ -575,12 +592,7 @@ pub fn validate_and_load(
|
|||||||
file: &clawhdf5::File,
|
file: &clawhdf5::File,
|
||||||
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
) -> Result<(MemoryConfig, MemoryCache, SessionCache, KnowledgeCache), MemoryError> {
|
||||||
// Read /meta group attributes
|
// Read /meta group attributes
|
||||||
let meta = file
|
let attrs = meta_attrs(file)?;
|
||||||
.group("meta")
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("missing /meta group: {e}")))?;
|
|
||||||
let attrs = meta
|
|
||||||
.attrs()
|
|
||||||
.map_err(|e| MemoryError::Schema(format!("cannot read /meta attrs: {e}")))?;
|
|
||||||
|
|
||||||
let schema_version = match attrs.get("schema_version") {
|
let schema_version = match attrs.get("schema_version") {
|
||||||
Some(AttrValue::String(s)) => s.clone(),
|
Some(AttrValue::String(s)) => s.clone(),
|
||||||
|
|||||||
@@ -258,3 +258,43 @@ fn an_existing_f32_store_stays_f32() {
|
|||||||
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
assert_eq!(&values[..before.1.len()], before.1.as_slice());
|
||||||
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
assert_eq!(&values[before.1.len()..], odd.as_slice());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `Group::attrs` leaves out an attribute it cannot decode. A store whose
|
||||||
|
/// `float16` setting is unreadable must not open as `float16 = false` (or with
|
||||||
|
/// any other default in place of a setting it has): it is an error.
|
||||||
|
#[test]
|
||||||
|
fn unreadable_meta_attribute_fails_open_instead_of_defaulting() {
|
||||||
|
let dir = TempDir::new().unwrap();
|
||||||
|
let path = dir.path().join("store.h5");
|
||||||
|
{
|
||||||
|
let mut m = HDF5Memory::create(config(&dir, "store.h5", true)).unwrap();
|
||||||
|
m.save(entry(1)).unwrap();
|
||||||
|
m.flush_wal().unwrap();
|
||||||
|
}
|
||||||
|
assert!(HDF5Memory::open_read_only(&path).is_ok());
|
||||||
|
|
||||||
|
// Give the `float16` attribute message an unknown version (the name is
|
||||||
|
// at +8 in a version-1 message and +9 in a version-3 one).
|
||||||
|
let mut bytes = std::fs::read(&path).unwrap();
|
||||||
|
let name = b"float16\0";
|
||||||
|
let mut hit = false;
|
||||||
|
let positions: Vec<usize> = (9..bytes.len() - name.len())
|
||||||
|
.filter(|&p| &bytes[p..p + name.len()] == name)
|
||||||
|
.collect();
|
||||||
|
for pos in positions {
|
||||||
|
for (back, version) in [(8, 1u8), (9, 3u8)] {
|
||||||
|
if bytes[pos - back] == version {
|
||||||
|
bytes[pos - back] = 0x7f;
|
||||||
|
hit = true;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(hit, "float16 attribute message not found");
|
||||||
|
std::fs::write(&path, &bytes).unwrap();
|
||||||
|
|
||||||
|
match HDF5Memory::open_read_only(&path) {
|
||||||
|
Err(MemoryError::Schema(msg)) => assert!(msg.contains("/meta"), "{msg}"),
|
||||||
|
Err(e) => panic!("unexpected error: {e}"),
|
||||||
|
Ok(_) => panic!("store opened with an unreadable float16 setting"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ use clawhdf5_format::filter_pipeline::FilterPipeline;
|
|||||||
use clawhdf5_format::group_v2::resolve_path_any;
|
use clawhdf5_format::group_v2::resolve_path_any;
|
||||||
use clawhdf5_format::message_type::MessageType;
|
use clawhdf5_format::message_type::MessageType;
|
||||||
use clawhdf5_format::object_header::ObjectHeader;
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
use clawhdf5_format::signature::find_signature;
|
use clawhdf5_format::signature::split_user_block;
|
||||||
use clawhdf5_format::superblock::Superblock;
|
use clawhdf5_format::superblock::Superblock;
|
||||||
use clawhdf5_io::FileWriter as IoFileWriter;
|
use clawhdf5_io::FileWriter as IoFileWriter;
|
||||||
|
|
||||||
@@ -861,8 +861,9 @@ impl HnswIndex {
|
|||||||
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
/// The HDF5 data must contain the `/ann/vectors`, `/ann/graph_layer_*`,
|
||||||
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
/// and `/ann/config` datasets as produced by [`to_hdf5_bytes`].
|
||||||
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
pub fn load_from_hdf5(data: &[u8]) -> Result<Self, FormatError> {
|
||||||
let sig_offset = find_signature(data)?;
|
// Addresses are relative to the superblock: skip any user block.
|
||||||
let sb = Superblock::parse(data, sig_offset)?;
|
let (_, data) = split_user_block(data)?;
|
||||||
|
let sb = Superblock::parse(data, 0)?;
|
||||||
|
|
||||||
// Read config dataset and its attributes
|
// Read config dataset and its attributes
|
||||||
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
let config_attrs = read_dataset_attrs(data, &sb, "ann/config")?;
|
||||||
|
|||||||
Binary file not shown.
@@ -97,7 +97,7 @@ impl AttributeMessage {
|
|||||||
return Ok(Cow::Borrowed(bytes));
|
return Ok(Cow::Borrowed(bytes));
|
||||||
}
|
}
|
||||||
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
let (file_data, offset_size) = file.ok_or(FormatError::UnresolvedSharedMessage)?;
|
||||||
let shared_ref = shared_message::parse_shared_ref(bytes, offset_size)?;
|
let shared_ref = shared_message::parse_shared_ref_sized(bytes, offset_size, length_size)?;
|
||||||
shared_message::resolve_shared_message(
|
shared_message::resolve_shared_message(
|
||||||
file_data,
|
file_data,
|
||||||
&shared_ref,
|
&shared_ref,
|
||||||
@@ -394,42 +394,80 @@ pub fn find_attribute<'a>(
|
|||||||
///
|
///
|
||||||
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
/// Use this instead of `extract_attributes` when reading files that may use dense storage
|
||||||
/// (e.g., objects with many attributes, typically >8).
|
/// (e.g., objects with many attributes, typically >8).
|
||||||
|
///
|
||||||
|
/// Fails if any attribute cannot be read; see [`extract_attributes_tolerant`]
|
||||||
|
/// to read the others.
|
||||||
pub fn extract_attributes_full(
|
pub fn extract_attributes_full(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &ObjectHeader,
|
header: &ObjectHeader,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
|
extract_attributes_with(file_data, header, offset_size, length_size, &mut Err)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like [`extract_attributes_full`], but an attribute that cannot be read
|
||||||
|
/// (a corrupt or unsupported attribute message, or a heap object that cannot
|
||||||
|
/// be located) is left out and its error returned alongside the attributes
|
||||||
|
/// that could be read, instead of failing them all.
|
||||||
|
///
|
||||||
|
/// Errors in the structures that index the attributes (the Attribute Info
|
||||||
|
/// message, the dense-storage heap header or B-tree) still fail the call:
|
||||||
|
/// then it is unknown which attributes exist at all.
|
||||||
|
pub fn extract_attributes_tolerant(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<(Vec<AttributeMessage>, Vec<FormatError>), FormatError> {
|
||||||
|
let mut errors = Vec::new();
|
||||||
|
let attrs = extract_attributes_with(file_data, header, offset_size, length_size, &mut |e| {
|
||||||
|
errors.push(e);
|
||||||
|
Ok(())
|
||||||
|
})?;
|
||||||
|
Ok((attrs, errors))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read every attribute; each one that fails goes to `on_error`, which
|
||||||
|
/// either stops the read (returns the error) or skips that attribute.
|
||||||
|
fn extract_attributes_with(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
) -> Result<Vec<AttributeMessage>, FormatError> {
|
||||||
let mut attrs = Vec::new();
|
let mut attrs = Vec::new();
|
||||||
|
|
||||||
// Collect compact attributes (inline in OH)
|
// Collect compact attributes (inline in OH)
|
||||||
for msg in &header.messages {
|
for msg in &header.messages {
|
||||||
if msg.msg_type == MessageType::Attribute {
|
if msg.msg_type == MessageType::Attribute {
|
||||||
if shared_message::is_shared(msg.flags) {
|
let attr = if shared_message::is_shared(msg.flags) {
|
||||||
// Shared attribute: resolve the reference to get actual attribute data
|
// Shared attribute: resolve the reference to get actual attribute data
|
||||||
let shared_ref = shared_message::parse_shared_ref(&msg.data, offset_size)?;
|
shared_message::parse_shared_ref_sized(&msg.data, offset_size, length_size)
|
||||||
let resolved_data = shared_message::resolve_shared_message(
|
.and_then(|shared_ref| {
|
||||||
file_data,
|
shared_message::resolve_shared_message(
|
||||||
&shared_ref,
|
file_data,
|
||||||
MessageType::Attribute,
|
&shared_ref,
|
||||||
offset_size,
|
MessageType::Attribute,
|
||||||
length_size,
|
offset_size,
|
||||||
)?;
|
length_size,
|
||||||
let attr = AttributeMessage::parse_in_file(
|
)
|
||||||
&resolved_data,
|
})
|
||||||
file_data,
|
.and_then(|resolved| {
|
||||||
offset_size,
|
AttributeMessage::parse_in_file(
|
||||||
length_size,
|
&resolved,
|
||||||
)?;
|
file_data,
|
||||||
attrs.push(attr);
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)
|
||||||
|
})
|
||||||
} else {
|
} else {
|
||||||
let attr = AttributeMessage::parse_in_file(
|
AttributeMessage::parse_in_file(&msg.data, file_data, offset_size, length_size)
|
||||||
&msg.data,
|
};
|
||||||
file_data,
|
match attr {
|
||||||
offset_size,
|
Ok(attr) => attrs.push(attr),
|
||||||
length_size,
|
Err(e) => on_error(e)?,
|
||||||
)?;
|
|
||||||
attrs.push(attr);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -439,9 +477,15 @@ pub fn extract_attributes_full(
|
|||||||
if let Some(info) = attr_info
|
if let Some(info) = attr_info
|
||||||
&& let Some(fh_addr) = info.fractal_heap_address
|
&& let Some(fh_addr) = info.fractal_heap_address
|
||||||
{
|
{
|
||||||
let dense_attrs =
|
extract_dense_attributes(
|
||||||
extract_dense_attributes(file_data, &info, fh_addr, offset_size, length_size)?;
|
file_data,
|
||||||
attrs.extend(dense_attrs);
|
&info,
|
||||||
|
fh_addr,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
&mut attrs,
|
||||||
|
on_error,
|
||||||
|
)?;
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(attrs)
|
Ok(attrs)
|
||||||
@@ -468,7 +512,9 @@ fn extract_dense_attributes(
|
|||||||
fh_addr: u64,
|
fh_addr: u64,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<AttributeMessage>, FormatError> {
|
attrs: &mut Vec<AttributeMessage>,
|
||||||
|
on_error: &mut dyn FnMut(FormatError) -> Result<(), FormatError>,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
// Parse fractal heap
|
// Parse fractal heap
|
||||||
let fh = FractalHeapHeader::parse(file_data, fh_addr as usize, offset_size, length_size)?;
|
let fh = FractalHeapHeader::parse(file_data, fh_addr as usize, offset_size, length_size)?;
|
||||||
|
|
||||||
@@ -482,28 +528,32 @@ fn extract_dense_attributes(
|
|||||||
let btree_hdr = BTreeV2Header::parse(file_data, btree_addr as usize, offset_size, length_size)?;
|
let btree_hdr = BTreeV2Header::parse(file_data, btree_addr as usize, offset_size, length_size)?;
|
||||||
let records = collect_btree_v2_records(file_data, &btree_hdr, offset_size, length_size)?;
|
let records = collect_btree_v2_records(file_data, &btree_hdr, offset_size, length_size)?;
|
||||||
|
|
||||||
let mut attrs = Vec::new();
|
|
||||||
for record in &records {
|
for record in &records {
|
||||||
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
// Per HDF5 spec, both type 8 and type 9 records start with heap_id:
|
||||||
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
// Type 8: heap_id(8) + msg_flags(1) + creation_order(4) + hash(4)
|
||||||
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
// Type 9: heap_id(8) + msg_flags(1) + creation_order(4)
|
||||||
let id_offset = 0;
|
let id_len = fh.heap_id_length as usize;
|
||||||
|
let Some(id_bytes) = record.data.get(..id_len) else {
|
||||||
if record.data.len() < id_offset + fh.heap_id_length as usize {
|
on_error(FormatError::UnexpectedEof {
|
||||||
|
expected: id_len,
|
||||||
|
available: record.data.len(),
|
||||||
|
})?;
|
||||||
continue;
|
continue;
|
||||||
}
|
};
|
||||||
let id_bytes = &record.data[id_offset..id_offset + fh.heap_id_length as usize];
|
|
||||||
|
|
||||||
// Read attribute message from fractal heap
|
|
||||||
let attr_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
|
||||||
|
|
||||||
// The data in the heap is a complete attribute message
|
// The data in the heap is a complete attribute message
|
||||||
let attr =
|
let attr = fh
|
||||||
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)?;
|
.read_managed_object(file_data, id_bytes, offset_size)
|
||||||
attrs.push(attr);
|
.and_then(|attr_data| {
|
||||||
|
AttributeMessage::parse_in_file(&attr_data, file_data, offset_size, length_size)
|
||||||
|
});
|
||||||
|
match attr {
|
||||||
|
Ok(attr) => attrs.push(attr),
|
||||||
|
Err(e) => on_error(e)?,
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(attrs)
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
|
|||||||
@@ -323,39 +323,21 @@ fn collect_internal_records(
|
|||||||
let records_start = pos;
|
let records_start = pos;
|
||||||
pos += records_total;
|
pos += records_total;
|
||||||
|
|
||||||
// Compute sizes for child pointers
|
// Child pointer layout, as libhdf5 computes it (H5B2__hdr_init): the
|
||||||
// max_records at child depth - for variable-width nrec encoding
|
// child's record count is always encoded in the width needed for a
|
||||||
|
// *leaf's* maximum, and — below the first internal level — the child
|
||||||
|
// subtree's total record count in the width needed for the most records
|
||||||
|
// a subtree of that depth can hold.
|
||||||
let child_depth = depth - 1;
|
let child_depth = depth - 1;
|
||||||
let max_nrec_child = if child_depth == 0 {
|
let nrec_width = bytes_for_max_records(max_leaf_nrec);
|
||||||
max_leaf_nrec
|
|
||||||
} else {
|
|
||||||
// For internal nodes at child_depth, the true max_nrec depends on the
|
|
||||||
// node size, record size, and the recursive width of child pointer
|
|
||||||
// entries (which themselves depend on max_nrec at deeper levels).
|
|
||||||
// Computing the exact value requires iterating from the leaf level
|
|
||||||
// upward, as described in the HDF5 spec (III.A.2 "Computing the Size
|
|
||||||
// of B-tree Nodes").
|
|
||||||
//
|
|
||||||
// We use `max_leaf_nrec * 2` as a conservative upper bound. This
|
|
||||||
// over-estimates the nrec encoding width, which means we may read
|
|
||||||
// slightly more bytes per child pointer than strictly necessary, but
|
|
||||||
// never fewer. The over-read bytes are harmless because we only
|
|
||||||
// decode `num_records` entries (the actual count from the node header).
|
|
||||||
//
|
|
||||||
// Known limitation: for very deep trees (depth > 3) with small record
|
|
||||||
// sizes, the true max could exceed this estimate, causing us to
|
|
||||||
// under-allocate the nrec encoding width and misparse child pointers.
|
|
||||||
// In practice, HDF5 B-tree v2 depths rarely exceed 2-3.
|
|
||||||
max_leaf_nrec * 2
|
|
||||||
};
|
|
||||||
let nrec_width = bytes_for_max_records(max_nrec_child);
|
|
||||||
|
|
||||||
// Total records in subtree width (only if depth > 1)
|
|
||||||
let total_nrec_width = if depth > 1 {
|
let total_nrec_width = if depth > 1 {
|
||||||
// Width to hold total records in a subtree
|
bytes_for_max_records(cum_max_records(
|
||||||
// We compute max possible total records at this subtree depth
|
node_size,
|
||||||
let max_total = header_max_total_records(max_leaf_nrec, depth - 1);
|
record_size,
|
||||||
bytes_for_max_records(max_total)
|
offset_size,
|
||||||
|
max_leaf_nrec,
|
||||||
|
child_depth,
|
||||||
|
))
|
||||||
} else {
|
} else {
|
||||||
0
|
0
|
||||||
};
|
};
|
||||||
@@ -435,14 +417,36 @@ fn collect_internal_records(
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Estimate maximum total records at a given depth (for variable-width encoding).
|
/// Most records a subtree whose root is at `depth` can hold (libhdf5's
|
||||||
fn header_max_total_records(max_leaf_nrec: u64, depth: u16) -> u64 {
|
/// `cum_max_nrec`): a leaf holds `max_leaf_nrec`; an internal node at depth
|
||||||
// Conservative: branching factor * max_leaf at each level
|
/// `d` holds `max_nrec(d)` records and `max_nrec(d) + 1` subtrees of depth
|
||||||
let mut total = max_leaf_nrec;
|
/// `d - 1`, where `max_nrec(d)` is what fits in a node once each record is
|
||||||
for _ in 0..depth {
|
/// paired with a child pointer of the width depth `d` needs.
|
||||||
total = total.saturating_mul(max_leaf_nrec.max(2));
|
fn cum_max_records(
|
||||||
|
node_size: u32,
|
||||||
|
record_size: u16,
|
||||||
|
offset_size: u8,
|
||||||
|
max_leaf_nrec: u64,
|
||||||
|
depth: u16,
|
||||||
|
) -> u64 {
|
||||||
|
// Internal node overhead: signature(4) + version(1) + type(1) + checksum(4).
|
||||||
|
const PREFIX: u64 = 10;
|
||||||
|
let nrec_width = bytes_for_max_records(max_leaf_nrec) as u64;
|
||||||
|
let mut cum = max_leaf_nrec;
|
||||||
|
let mut cum_width = 0u64;
|
||||||
|
for d in 1..=depth {
|
||||||
|
let ptr = u64::from(offset_size) + nrec_width + if d > 1 { cum_width } else { 0 };
|
||||||
|
let max_nrec = u64::from(node_size)
|
||||||
|
.saturating_sub(PREFIX)
|
||||||
|
.saturating_sub(ptr)
|
||||||
|
/ (u64::from(record_size) + ptr).max(1);
|
||||||
|
cum = max_nrec
|
||||||
|
.saturating_add(1)
|
||||||
|
.saturating_mul(cum)
|
||||||
|
.saturating_add(max_nrec);
|
||||||
|
cum_width = bytes_for_max_records(cum) as u64;
|
||||||
}
|
}
|
||||||
total
|
cum
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -512,9 +516,15 @@ mod tests {
|
|||||||
child_nrec: u64,
|
child_nrec: u64,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let max_leaf = max_records_leaf(node_size, record_size);
|
let max_leaf = max_records_leaf(node_size, record_size);
|
||||||
let nrec_width = bytes_for_max_records(if depth == 1 { max_leaf } else { max_leaf * 2 });
|
let nrec_width = bytes_for_max_records(max_leaf);
|
||||||
let total_width = if depth > 1 {
|
let total_width = if depth > 1 {
|
||||||
bytes_for_max_records(header_max_total_records(max_leaf, depth - 1))
|
bytes_for_max_records(cum_max_records(
|
||||||
|
node_size,
|
||||||
|
record_size,
|
||||||
|
8,
|
||||||
|
max_leaf,
|
||||||
|
depth - 1,
|
||||||
|
))
|
||||||
} else {
|
} else {
|
||||||
0
|
0
|
||||||
};
|
};
|
||||||
@@ -673,4 +683,18 @@ mod tests {
|
|||||||
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
let records = collect_btree_v2_records(&header, &hdr, 8, 8).unwrap();
|
||||||
assert!(records.is_empty());
|
assert!(records.is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn subtree_capacity_matches_libhdf5() {
|
||||||
|
// A link-name index (11-byte records, 512-byte nodes, 8-byte
|
||||||
|
// addresses): libhdf5's H5B2__hdr_init gives 45 records per leaf,
|
||||||
|
// then cum_max_nrec 1 149 at depth 1 and 26 449 at depth 2 — two
|
||||||
|
// bytes of subtree count in a depth-3 root's child pointers, where
|
||||||
|
// leaf_max^3 = 91 125 would need three.
|
||||||
|
let leaf = max_records_leaf(512, 11);
|
||||||
|
assert_eq!(leaf, 45);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 0), 45);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 1), 1_149);
|
||||||
|
assert_eq!(cum_max_records(512, 11, 8, leaf, 2), 26_449);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -223,13 +223,32 @@ pub const DEFAULT_CACHE_BYTES: usize = 16 * 1024 * 1024; // 16 MiB
|
|||||||
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
/// coordinate map and reduces collision chains compared to power-of-two sizes.
|
||||||
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
pub const DEFAULT_MAX_SLOTS: usize = 521;
|
||||||
|
|
||||||
|
/// Most datasets whose chunk index a [`ChunkCache`] keeps at once.
|
||||||
|
pub const MAX_INDEXED_DATASETS: usize = 64;
|
||||||
|
|
||||||
|
/// Most chunk-index entries, summed over all datasets, a [`ChunkCache`] keeps.
|
||||||
|
/// Least-recently-used datasets' indexes are dropped past this (the dataset
|
||||||
|
/// being read is always kept), so a file with many or huge chunked datasets
|
||||||
|
/// cannot grow the cache without bound.
|
||||||
|
pub const MAX_INDEXED_CHUNKS: usize = 1 << 20;
|
||||||
|
|
||||||
|
/// The dataset key the address-less (legacy) methods use when
|
||||||
|
/// [`ChunkCache::ensure_dataset`] has not been called.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
const UNBOUND_DATASET: u64 = u64::MAX;
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// LRU entry
|
// LRU entry
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// Decompressed chunks are keyed by dataset *and* coordinate: every chunked
|
||||||
|
/// dataset has a chunk at (0, 0, ...), so the coordinate alone is ambiguous.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
type SlotKey = (u64, ChunkCoord);
|
||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
struct CachedChunk {
|
struct CachedChunk {
|
||||||
coord: ChunkCoord,
|
key: SlotKey,
|
||||||
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
/// Shared so a cache hit is a refcount bump, not a copy of the whole
|
||||||
/// (potentially large) decompressed chunk.
|
/// (potentially large) decompressed chunk.
|
||||||
data: Arc<CacheAlignedBuffer>,
|
data: Arc<CacheAlignedBuffer>,
|
||||||
@@ -237,21 +256,48 @@ struct CachedChunk {
|
|||||||
last_access: u64,
|
last_access: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Per-dataset index state.
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
#[derive(Default)]
|
||||||
|
struct DatasetEntry {
|
||||||
|
/// Chunk coordinate -> ChunkInfo (offset + size in file).
|
||||||
|
index: Option<Arc<HashMap<ChunkCoord, ChunkInfo>>>,
|
||||||
|
/// Pre-built chunk index for O(1) coordinate lookups.
|
||||||
|
chunk_index: Option<Arc<ChunkIndex>>,
|
||||||
|
/// Pre-computed chunk layout for fast assembly.
|
||||||
|
chunk_layout: Option<Arc<ChunkLayout>>,
|
||||||
|
/// Tick of the last use, for dropping the least recently used dataset.
|
||||||
|
last_used: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "std")]
|
||||||
|
impl DatasetEntry {
|
||||||
|
fn weight(&self) -> usize {
|
||||||
|
self.index.as_ref().map_or(0, |m| m.len())
|
||||||
|
+ self.chunk_index.as_ref().map_or(0, |c| c.num_chunks())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// ChunkCache
|
// ChunkCache
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
/// A per-dataset chunk cache with hash-based index and LRU eviction.
|
/// A per-file chunk cache: chunk indexes per dataset, plus an LRU of
|
||||||
|
/// decompressed chunks, all keyed by dataset.
|
||||||
///
|
///
|
||||||
/// # Usage
|
/// A dataset is identified by the address of its chunk index (B-tree, fixed
|
||||||
|
/// or extensible array, ...), which is unique within a file. Every method
|
||||||
|
/// that takes an `addr` works on that dataset only, so threads reading
|
||||||
|
/// different datasets through one shared cache never see each other's
|
||||||
|
/// chunks. The address-less methods (`has_index`, `populate_index`,
|
||||||
|
/// `get_decompressed`, ...) act on the dataset last bound with
|
||||||
|
/// [`Self::ensure_dataset`]; that binding is shared state, so concurrent
|
||||||
|
/// readers must use the `*_in` / `*_for` methods instead (the chunked
|
||||||
|
/// readers in [`crate::chunked_read`] do).
|
||||||
///
|
///
|
||||||
/// ```ignore
|
/// Memory is bounded: decompressed data by `max_bytes`/`max_slots` across
|
||||||
/// let cache = ChunkCache::new();
|
/// all datasets, indexes by [`MAX_INDEXED_DATASETS`] and
|
||||||
/// // Pass &cache to read_chunked_data — it will populate the index lazily.
|
/// [`MAX_INDEXED_CHUNKS`].
|
||||||
/// ```
|
|
||||||
///
|
|
||||||
/// The cache is wrapped in `Mutex` internally so it can be mutated through
|
|
||||||
/// shared references (thread-safe).
|
|
||||||
///
|
///
|
||||||
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
/// Only available with the `std` feature because it requires `std::sync::Mutex`.
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
@@ -261,26 +307,20 @@ pub struct ChunkCache {
|
|||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
struct CacheInner {
|
struct CacheInner {
|
||||||
/// Hash index: chunk coordinate -> ChunkInfo (offset + size in file).
|
/// Per-dataset chunk indexes, keyed by chunk-index address.
|
||||||
/// Populated once per dataset on first access.
|
datasets: HashMap<u64, DatasetEntry>,
|
||||||
index: Option<HashMap<ChunkCoord, ChunkInfo>>,
|
|
||||||
|
|
||||||
/// Address of the dataset (its chunk-index base address) that the cached
|
/// Dataset the address-less methods act on (see `ensure_dataset`).
|
||||||
/// index, chunk index, layout, and decompressed slots currently belong to.
|
current: Option<u64>,
|
||||||
/// The cache is shared per file across datasets, so every cached-read entry
|
|
||||||
/// checks this and resets the per-dataset state when the dataset changes —
|
|
||||||
/// otherwise one dataset's chunk index (with its own rank) would be reused
|
|
||||||
/// for another, corrupting reads.
|
|
||||||
index_addr: Option<u64>,
|
|
||||||
|
|
||||||
/// LRU cache of decompressed chunk data.
|
/// LRU cache of decompressed chunk data.
|
||||||
slots: Vec<CachedChunk>,
|
slots: Vec<CachedChunk>,
|
||||||
|
|
||||||
/// Coordinate -> index into `slots`, for O(1) lookup instead of a linear
|
/// Key -> index into `slots`, for O(1) lookup instead of a linear
|
||||||
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
/// scan. Kept in sync with `slots` on every insert/evict/clear — in
|
||||||
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
/// particular, `slots.swap_remove(i)` moves the last element into slot
|
||||||
/// `i`, so the moved element's index entry must be updated too.
|
/// `i`, so the moved element's index entry must be updated too.
|
||||||
slot_index: HashMap<ChunkCoord, usize>,
|
slot_index: HashMap<SlotKey, usize>,
|
||||||
|
|
||||||
/// Current total bytes of cached decompressed data.
|
/// Current total bytes of cached decompressed data.
|
||||||
current_bytes: usize,
|
current_bytes: usize,
|
||||||
@@ -294,17 +334,145 @@ struct CacheInner {
|
|||||||
/// Monotonic counter for LRU ordering.
|
/// Monotonic counter for LRU ordering.
|
||||||
tick: u64,
|
tick: u64,
|
||||||
|
|
||||||
/// Last accessed chunk coordinate (for sequential detection).
|
/// Last accessed chunk (for sequential detection).
|
||||||
last_coord: Option<ChunkCoord>,
|
last_coord: Option<SlotKey>,
|
||||||
|
|
||||||
/// Access pattern statistics.
|
/// Access pattern statistics.
|
||||||
stats: AccessStats,
|
stats: AccessStats,
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-built chunk index for O(1) coordinate lookups.
|
#[cfg(feature = "std")]
|
||||||
chunk_index: Option<ChunkIndex>,
|
impl CacheInner {
|
||||||
|
fn current(&self) -> u64 {
|
||||||
|
self.current.unwrap_or(UNBOUND_DATASET)
|
||||||
|
}
|
||||||
|
|
||||||
/// Pre-computed chunk layout for fast assembly.
|
fn touch(&mut self, addr: u64) -> &mut DatasetEntry {
|
||||||
chunk_layout: Option<ChunkLayout>,
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
let entry = self.datasets.entry(addr).or_default();
|
||||||
|
entry.last_used = tick;
|
||||||
|
entry
|
||||||
|
}
|
||||||
|
|
||||||
|
fn entry(&self, addr: u64) -> Option<&DatasetEntry> {
|
||||||
|
self.datasets.get(&addr)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Drop least-recently-used datasets' indexes (never `keep`'s) until the
|
||||||
|
/// dataset and chunk-entry budgets hold.
|
||||||
|
fn trim_datasets(&mut self, keep: u64) {
|
||||||
|
loop {
|
||||||
|
let total: usize = self.datasets.values().map(DatasetEntry::weight).sum();
|
||||||
|
if self.datasets.len() <= MAX_INDEXED_DATASETS && total <= MAX_INDEXED_CHUNKS {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let victim = self
|
||||||
|
.datasets
|
||||||
|
.iter()
|
||||||
|
.filter(|(a, _)| **a != keep)
|
||||||
|
.min_by_key(|(_, e)| e.last_used)
|
||||||
|
.map(|(a, _)| *a);
|
||||||
|
match victim {
|
||||||
|
Some(a) => {
|
||||||
|
self.datasets.remove(&a);
|
||||||
|
}
|
||||||
|
None => return,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn get_decompressed(&mut self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
|
||||||
|
// Track sequential vs random access
|
||||||
|
let is_sequential = self.last_coord.as_ref().is_some_and(|(prev_addr, prev)| {
|
||||||
|
// Sequential if exactly one dimension changed
|
||||||
|
let changes: usize = prev
|
||||||
|
.iter()
|
||||||
|
.zip(coord.iter())
|
||||||
|
.filter(|(a, b)| a != b)
|
||||||
|
.count();
|
||||||
|
*prev_addr == addr && changes <= 1
|
||||||
|
});
|
||||||
|
if is_sequential {
|
||||||
|
self.stats.sequential_count += 1;
|
||||||
|
} else if self.last_coord.is_some() {
|
||||||
|
self.stats.random_count += 1;
|
||||||
|
}
|
||||||
|
let key: SlotKey = (addr, coord.to_vec());
|
||||||
|
let found = if let Some(&idx) = self.slot_index.get(&key) {
|
||||||
|
self.slots[idx].last_access = tick;
|
||||||
|
Some(Arc::clone(&self.slots[idx].data))
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
self.last_coord = Some(key);
|
||||||
|
if let Some(ref data) = found {
|
||||||
|
self.stats.hits += 1;
|
||||||
|
self.stats.bytes_read += data.len() as u64;
|
||||||
|
} else {
|
||||||
|
self.stats.misses += 1;
|
||||||
|
}
|
||||||
|
found
|
||||||
|
}
|
||||||
|
|
||||||
|
fn put_decompressed(
|
||||||
|
&mut self,
|
||||||
|
key: SlotKey,
|
||||||
|
data: Arc<CacheAlignedBuffer>,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
let data_len = data.len();
|
||||||
|
|
||||||
|
// Don't cache if single chunk exceeds budget — still return the data
|
||||||
|
// to the caller, just don't retain it.
|
||||||
|
if data_len > self.max_bytes {
|
||||||
|
return data;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check if already present
|
||||||
|
self.tick += 1;
|
||||||
|
let tick = self.tick;
|
||||||
|
if let Some(&idx) = self.slot_index.get(&key) {
|
||||||
|
self.slots[idx].last_access = tick;
|
||||||
|
return Arc::clone(&self.slots[idx].data); // already cached
|
||||||
|
}
|
||||||
|
|
||||||
|
// Evict until we have room
|
||||||
|
while self.slots.len() >= self.max_slots
|
||||||
|
|| (self.current_bytes + data_len > self.max_bytes && !self.slots.is_empty())
|
||||||
|
{
|
||||||
|
// Find LRU slot
|
||||||
|
let lru_idx = self
|
||||||
|
.slots
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.min_by_key(|(_, s)| s.last_access)
|
||||||
|
.map(|(i, _)| i)
|
||||||
|
.unwrap();
|
||||||
|
let removed = self.slots.swap_remove(lru_idx);
|
||||||
|
self.slot_index.remove(&removed.key);
|
||||||
|
// swap_remove moved the former last element into `lru_idx` (unless
|
||||||
|
// it *was* the last element) — fix up that element's index entry.
|
||||||
|
if lru_idx < self.slots.len() {
|
||||||
|
let moved_key = self.slots[lru_idx].key.clone();
|
||||||
|
self.slot_index.insert(moved_key, lru_idx);
|
||||||
|
}
|
||||||
|
self.current_bytes -= removed.data.len();
|
||||||
|
self.stats.evictions += 1;
|
||||||
|
}
|
||||||
|
|
||||||
|
self.current_bytes += data_len;
|
||||||
|
let new_idx = self.slots.len();
|
||||||
|
self.slot_index.insert(key.clone(), new_idx);
|
||||||
|
self.slots.push(CachedChunk {
|
||||||
|
key,
|
||||||
|
data: Arc::clone(&data),
|
||||||
|
last_access: tick,
|
||||||
|
});
|
||||||
|
data
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Access pattern statistics tracked by the chunk cache.
|
/// Access pattern statistics tracked by the chunk cache.
|
||||||
@@ -356,8 +524,8 @@ impl ChunkCache {
|
|||||||
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
pub fn with_capacity(max_bytes: usize, max_slots: usize) -> Self {
|
||||||
Self {
|
Self {
|
||||||
inner: std::sync::Mutex::new(CacheInner {
|
inner: std::sync::Mutex::new(CacheInner {
|
||||||
index: None,
|
datasets: HashMap::new(),
|
||||||
index_addr: None,
|
current: None,
|
||||||
slots: Vec::with_capacity(max_slots.min(64)),
|
slots: Vec::with_capacity(max_slots.min(64)),
|
||||||
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
slot_index: HashMap::with_capacity(max_slots.min(64)),
|
||||||
current_bytes: 0,
|
current_bytes: 0,
|
||||||
@@ -366,340 +534,331 @@ impl ChunkCache {
|
|||||||
tick: 0,
|
tick: 0,
|
||||||
last_coord: None,
|
last_coord: None,
|
||||||
stats: AccessStats::default(),
|
stats: AccessStats::default(),
|
||||||
chunk_index: None,
|
|
||||||
chunk_layout: None,
|
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Index operations -----
|
fn lock(&self) -> std::sync::MutexGuard<'_, CacheInner> {
|
||||||
|
self.inner.lock().unwrap_or_else(|e| e.into_inner())
|
||||||
|
}
|
||||||
|
|
||||||
/// The most decompressed bytes this cache will hold.
|
/// The most decompressed bytes this cache will hold.
|
||||||
pub fn max_bytes(&self) -> usize {
|
pub fn max_bytes(&self) -> usize {
|
||||||
self.inner.lock().map(|g| g.max_bytes).unwrap_or(0)
|
self.lock().max_bytes
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Bind the cache to the dataset at chunk-index address `addr`.
|
// ----- Dataset-keyed operations (safe to use concurrently) -----
|
||||||
|
|
||||||
|
/// The chunk list of the dataset whose chunk index is at `addr`.
|
||||||
///
|
///
|
||||||
/// The cache is shared per file across all of its datasets. If the cache
|
/// On the first call for a dataset, `build` scans its chunk index; the
|
||||||
/// currently holds state for a different dataset, all per-dataset state
|
/// result is kept (offsets truncated to `rank` for the lookup key), so
|
||||||
/// (chunk index, chunk-index map, layout, and decompressed slots) is
|
/// later calls skip the scan. `build` runs without the cache lock held;
|
||||||
/// dropped so the next access rebuilds it for this dataset. Reading the
|
/// if two threads race to build the same dataset's index, the first
|
||||||
/// same dataset again is a no-op, preserving the cache's benefit for
|
/// stored one wins and both return equivalent lists.
|
||||||
/// repeated/sequential access. Returns `true` if a reset occurred.
|
pub fn chunks_for<E>(
|
||||||
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
&self,
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
addr: u64,
|
||||||
if inner.index_addr == Some(addr) {
|
rank: usize,
|
||||||
return false;
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
) -> Result<Vec<ChunkInfo>, E> {
|
||||||
|
Ok(self
|
||||||
|
.index_for(addr, rank, build)?
|
||||||
|
.values()
|
||||||
|
.cloned()
|
||||||
|
.collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn index_for<E>(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
rank: usize,
|
||||||
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
) -> Result<Arc<HashMap<ChunkCoord, ChunkInfo>>, E> {
|
||||||
|
if let Some(index) = self.lock().touch(addr).index.clone() {
|
||||||
|
return Ok(index);
|
||||||
}
|
}
|
||||||
inner.index = None;
|
let chunks = build()?;
|
||||||
inner.chunk_index = None;
|
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
|
||||||
inner.chunk_layout = None;
|
.into_iter()
|
||||||
inner.slots.clear();
|
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci))
|
||||||
inner.slot_index.clear();
|
.collect();
|
||||||
inner.current_bytes = 0;
|
let mut inner = self.lock();
|
||||||
inner.last_coord = None;
|
let entry = inner.touch(addr);
|
||||||
inner.index_addr = Some(addr);
|
let index = Arc::clone(entry.index.get_or_insert_with(|| Arc::new(map)));
|
||||||
true
|
inner.trim_datasets(addr);
|
||||||
|
Ok(index)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns `true` if the chunk index has been built.
|
/// The pre-computed assembly layout of the dataset at `addr`, building
|
||||||
|
/// its chunk index (via `build`, as in [`Self::chunks_for`]) and layout on
|
||||||
|
/// first use.
|
||||||
|
pub fn chunk_layout_for<E>(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
rank: usize,
|
||||||
|
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
|
||||||
|
ds_dims: &[usize],
|
||||||
|
chunk_dims: &[usize],
|
||||||
|
elem_size: usize,
|
||||||
|
) -> Result<Arc<ChunkLayout>, E> {
|
||||||
|
let (layout, chunk_index) = {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
(entry.chunk_layout.clone(), entry.chunk_index.clone())
|
||||||
|
};
|
||||||
|
if let Some(layout) = layout {
|
||||||
|
return Ok(layout);
|
||||||
|
}
|
||||||
|
let chunk_index = match chunk_index {
|
||||||
|
Some(ci) => ci,
|
||||||
|
None => {
|
||||||
|
let index = self.index_for(addr, rank, build)?;
|
||||||
|
let chunks: Vec<ChunkInfo> = index.values().cloned().collect();
|
||||||
|
Arc::new(ChunkIndex::build(&chunks, rank))
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let layout = ChunkLayout::build(&chunk_index, ds_dims, chunk_dims, elem_size);
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
entry.chunk_index.get_or_insert(chunk_index);
|
||||||
|
let layout = Arc::clone(entry.chunk_layout.get_or_insert_with(|| Arc::new(layout)));
|
||||||
|
inner.trim_datasets(addr);
|
||||||
|
Ok(layout)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cached decompressed chunk at `coord` of the dataset at `addr`.
|
||||||
|
///
|
||||||
|
/// O(1) lookup; the clone is an `Arc` refcount bump, not a copy of the
|
||||||
|
/// underlying decompressed data.
|
||||||
|
pub fn get_decompressed_in(&self, addr: u64, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
|
self.lock().get_decompressed(addr, coord)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Cache decompressed chunk data for `coord` of the dataset at `addr`.
|
||||||
|
/// Returns the `Arc`-shared buffer now cached (or already cached).
|
||||||
|
pub fn put_decompressed_in(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
coord: ChunkCoord,
|
||||||
|
data: Vec<u8>,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
self.put_decompressed_aligned_in(addr, coord, CacheAlignedBuffer::from_vec(data))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::put_decompressed_in`] for an already-aligned buffer.
|
||||||
|
pub fn put_decompressed_aligned_in(
|
||||||
|
&self,
|
||||||
|
addr: u64,
|
||||||
|
coord: ChunkCoord,
|
||||||
|
data: CacheAlignedBuffer,
|
||||||
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
|
let data = Arc::new(data);
|
||||||
|
self.lock().put_decompressed((addr, coord), data)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record that the given chunk coordinates of the dataset at `addr` are
|
||||||
|
/// predicted to be accessed soon (bookkeeping only).
|
||||||
|
///
|
||||||
|
/// This does **not** prefetch or pre-decompress anything — it only
|
||||||
|
/// checks whether each coordinate is already in the chunk index and
|
||||||
|
/// updates access-pattern stats accordingly.
|
||||||
|
pub fn prefetch_hint_in(&self, addr: u64, next_coords: &[ChunkCoord]) {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let Some(index) = inner.entry(addr).and_then(|e| e.index.clone()) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let known = next_coords
|
||||||
|
.iter()
|
||||||
|
.filter(|c| index.contains_key(*c))
|
||||||
|
.count();
|
||||||
|
inner.stats.sequential_count += known as u64;
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- Address-less operations on the bound dataset -----
|
||||||
|
|
||||||
|
/// Bind the address-less methods to the dataset at chunk-index address
|
||||||
|
/// `addr`. Returns `true` if this changed the bound dataset.
|
||||||
|
///
|
||||||
|
/// Each dataset's state is kept separately, so switching loses nothing
|
||||||
|
/// and never exposes one dataset's index or chunks to another. The
|
||||||
|
/// binding itself is shared, though: concurrent readers should use the
|
||||||
|
/// `addr`-taking methods rather than bind and then call these.
|
||||||
|
pub fn ensure_dataset(&self, addr: u64) -> bool {
|
||||||
|
let mut inner = self.lock();
|
||||||
|
let changed = inner.current != Some(addr);
|
||||||
|
inner.current = Some(addr);
|
||||||
|
changed
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Returns `true` if the bound dataset's chunk index has been built.
|
||||||
pub fn has_index(&self) -> bool {
|
pub fn has_index(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.index
|
.is_some_and(|e| e.index.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build the chunk index from a pre-collected list of `ChunkInfo`.
|
/// Build the bound dataset's chunk index from a pre-collected list of
|
||||||
|
/// `ChunkInfo`.
|
||||||
///
|
///
|
||||||
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
/// The `rank` parameter is used to truncate offsets to spatial dims only
|
||||||
/// (B-tree v1 stores rank+1 offsets).
|
/// (B-tree v1 stores rank+1 offsets).
|
||||||
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
pub fn populate_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let addr = self.lock().current();
|
||||||
if inner.index.is_some() {
|
let _ = self.index_for::<core::convert::Infallible>(addr, rank, || Ok(chunks.to_vec()));
|
||||||
return; // already populated
|
|
||||||
}
|
|
||||||
let mut map = HashMap::with_capacity(chunks.len());
|
|
||||||
|
|
||||||
for ci in chunks {
|
|
||||||
let coord: ChunkCoord = ci.offsets.iter().take(rank).copied().collect();
|
|
||||||
map.insert(coord, ci.clone());
|
|
||||||
}
|
|
||||||
inner.index = Some(map);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Look up a chunk by its spatial coordinate in the index.
|
/// Look up a chunk by its spatial coordinate in the bound dataset's index.
|
||||||
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
pub fn lookup_index(&self, coord: &[u64]) -> Option<ChunkInfo> {
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let inner = self.lock();
|
||||||
inner.index.as_ref()?.get(coord).cloned()
|
inner
|
||||||
|
.entry(inner.current())?
|
||||||
|
.index
|
||||||
|
.as_ref()?
|
||||||
|
.get(coord)
|
||||||
|
.cloned()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return all indexed chunks as a `Vec<ChunkInfo>` (order unspecified).
|
/// Return all of the bound dataset's indexed chunks (order unspecified).
|
||||||
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
pub fn all_indexed_chunks(&self) -> Option<Vec<ChunkInfo>> {
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let inner = self.lock();
|
||||||
inner.index.as_ref().map(|m| m.values().cloned().collect())
|
let index = inner.entry(inner.current())?.index.as_ref()?;
|
||||||
|
Some(index.values().cloned().collect())
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Chunk index (pre-built coordinate → ChunkInfo map) -----
|
/// Returns `true` if the bound dataset's `ChunkIndex` has been built.
|
||||||
|
|
||||||
/// Returns `true` if the chunk B-tree index has been built.
|
|
||||||
pub fn has_chunk_index(&self) -> bool {
|
pub fn has_chunk_index(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.chunk_index
|
.is_some_and(|e| e.chunk_index.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build and store the chunk B-tree index from a pre-collected list of `ChunkInfo`.
|
/// Build and store the bound dataset's `ChunkIndex`.
|
||||||
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
pub fn populate_chunk_index(&self, chunks: &[ChunkInfo], rank: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let built = Arc::new(ChunkIndex::build(chunks, rank));
|
||||||
if inner.chunk_index.is_some() {
|
let mut inner = self.lock();
|
||||||
return;
|
let addr = inner.current();
|
||||||
}
|
inner.touch(addr).chunk_index.get_or_insert(built);
|
||||||
inner.chunk_index = Some(ChunkIndex::build(chunks, rank));
|
inner.trim_datasets(addr);
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Chunk layout (pre-computed assembly plan) -----
|
/// Returns `true` if the bound dataset's chunk layout has been computed.
|
||||||
|
|
||||||
/// Returns `true` if the chunk layout has been computed.
|
|
||||||
pub fn has_chunk_layout(&self) -> bool {
|
pub fn has_chunk_layout(&self) -> bool {
|
||||||
self.inner
|
let inner = self.lock();
|
||||||
.lock()
|
inner
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
.entry(inner.current())
|
||||||
.chunk_layout
|
.is_some_and(|e| e.chunk_layout.is_some())
|
||||||
.is_some()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Build and store the pre-computed chunk layout for fast assembly.
|
/// Build and store the bound dataset's chunk layout (needs its
|
||||||
|
/// `ChunkIndex`; does nothing without one).
|
||||||
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
pub fn populate_chunk_layout(&self, ds_dims: &[usize], chunk_dims: &[usize], elem_size: usize) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
if inner.chunk_layout.is_some() {
|
let addr = inner.current();
|
||||||
|
let entry = inner.touch(addr);
|
||||||
|
if entry.chunk_layout.is_some() {
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if let Some(ref idx) = inner.chunk_index {
|
if let Some(idx) = entry.chunk_index.clone() {
|
||||||
inner.chunk_layout = Some(ChunkLayout::build(idx, ds_dims, chunk_dims, elem_size));
|
entry.chunk_layout = Some(Arc::new(ChunkLayout::build(
|
||||||
|
&idx, ds_dims, chunk_dims, elem_size,
|
||||||
|
)));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Execute a function with a reference to the chunk layout.
|
/// Execute a function with a reference to the bound dataset's chunk
|
||||||
///
|
/// layout. Returns `None` if the layout hasn't been computed yet.
|
||||||
/// Returns `None` if the layout hasn't been computed yet.
|
|
||||||
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
pub fn with_chunk_layout<F, R>(&self, f: F) -> Option<R>
|
||||||
where
|
where
|
||||||
F: FnOnce(&ChunkLayout) -> R,
|
F: FnOnce(&ChunkLayout) -> R,
|
||||||
{
|
{
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let layout = {
|
||||||
inner.chunk_layout.as_ref().map(f)
|
let inner = self.lock();
|
||||||
|
inner.entry(inner.current())?.chunk_layout.clone()?
|
||||||
|
};
|
||||||
|
Some(f(&layout))
|
||||||
}
|
}
|
||||||
|
|
||||||
// ----- Decompressed data cache (LRU) -----
|
/// Try to get cached decompressed data for a chunk of the bound dataset.
|
||||||
|
|
||||||
/// Try to get cached decompressed data for a chunk coordinate.
|
|
||||||
///
|
///
|
||||||
/// O(1) lookup. Returns an owned copy for API compatibility with callers
|
/// Returns an owned copy; prefer [`Self::get_decompressed_aligned`] when
|
||||||
/// that need a `Vec<u8>`; prefer [`Self::get_decompressed_aligned`] when
|
/// an `Arc`-shared buffer works for the caller.
|
||||||
/// an `Arc`-shared buffer works for the caller, since that avoids the
|
|
||||||
/// copy entirely.
|
|
||||||
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
pub fn get_decompressed(&self, coord: &[u64]) -> Option<Vec<u8>> {
|
||||||
self.get_decompressed_aligned(coord)
|
self.get_decompressed_aligned(coord)
|
||||||
.map(|arc| arc.as_slice().to_vec())
|
.map(|arc| arc.as_slice().to_vec())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Try to get a reference-counted clone of the aligned buffer for a chunk.
|
/// Reference-counted cached buffer for a chunk of the bound dataset.
|
||||||
///
|
|
||||||
/// O(1) index lookup; the clone is an `Arc` refcount bump, not a copy of
|
|
||||||
/// the underlying decompressed data.
|
|
||||||
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
pub fn get_decompressed_aligned(&self, coord: &[u64]) -> Option<Arc<CacheAlignedBuffer>> {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
inner.tick += 1;
|
let addr = inner.current();
|
||||||
let tick = inner.tick;
|
inner.get_decompressed(addr, coord)
|
||||||
|
|
||||||
// Track sequential vs random access
|
|
||||||
let is_sequential = inner.last_coord.as_ref().is_some_and(|prev| {
|
|
||||||
// Sequential if exactly one dimension changed
|
|
||||||
let changes: usize = prev
|
|
||||||
.iter()
|
|
||||||
.zip(coord.iter())
|
|
||||||
.filter(|(a, b)| a != b)
|
|
||||||
.count();
|
|
||||||
changes <= 1
|
|
||||||
});
|
|
||||||
if is_sequential {
|
|
||||||
inner.stats.sequential_count += 1;
|
|
||||||
} else if inner.last_coord.is_some() {
|
|
||||||
inner.stats.random_count += 1;
|
|
||||||
}
|
|
||||||
inner.last_coord = Some(coord.to_vec());
|
|
||||||
|
|
||||||
let found = if let Some(&idx) = inner.slot_index.get(coord) {
|
|
||||||
inner.slots[idx].last_access = tick;
|
|
||||||
Some(Arc::clone(&inner.slots[idx].data))
|
|
||||||
} else {
|
|
||||||
None
|
|
||||||
};
|
|
||||||
if let Some(ref data) = found {
|
|
||||||
inner.stats.hits += 1;
|
|
||||||
inner.stats.bytes_read += data.len() as u64;
|
|
||||||
} else {
|
|
||||||
inner.stats.misses += 1;
|
|
||||||
}
|
|
||||||
found
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert decompressed chunk data into the LRU cache.
|
/// Insert decompressed chunk data for the bound dataset into the LRU
|
||||||
///
|
/// cache, returning the `Arc`-shared buffer now cached.
|
||||||
/// The data is stored in a [`CacheAlignedBuffer`] so subsequent reads
|
|
||||||
/// return cache-line-aligned memory. Returns the `Arc`-shared buffer that
|
|
||||||
/// is now cached (or already was), so the caller can reuse it directly
|
|
||||||
/// instead of holding a separate copy of the same data.
|
|
||||||
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
pub fn put_decompressed(&self, coord: ChunkCoord, data: Vec<u8>) -> Arc<CacheAlignedBuffer> {
|
||||||
let aligned = CacheAlignedBuffer::from_vec(data);
|
self.put_decompressed_aligned(coord, CacheAlignedBuffer::from_vec(data))
|
||||||
self.put_decompressed_aligned(coord, aligned)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert an already-aligned buffer into the LRU cache.
|
/// Insert an already-aligned buffer for the bound dataset.
|
||||||
///
|
|
||||||
/// Returns the `Arc`-shared buffer now held by the cache (the one just
|
|
||||||
/// inserted, or the existing cached copy if `coord` was already present).
|
|
||||||
pub fn put_decompressed_aligned(
|
pub fn put_decompressed_aligned(
|
||||||
&self,
|
&self,
|
||||||
coord: ChunkCoord,
|
coord: ChunkCoord,
|
||||||
data: CacheAlignedBuffer,
|
data: CacheAlignedBuffer,
|
||||||
) -> Arc<CacheAlignedBuffer> {
|
) -> Arc<CacheAlignedBuffer> {
|
||||||
let data = Arc::new(data);
|
let data = Arc::new(data);
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
let data_len = data.len();
|
let addr = inner.current();
|
||||||
|
inner.put_decompressed((addr, coord), data)
|
||||||
// Don't cache if single chunk exceeds budget — still return the data
|
|
||||||
// to the caller, just don't retain it.
|
|
||||||
if data_len > inner.max_bytes {
|
|
||||||
return data;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Check if already present
|
|
||||||
inner.tick += 1;
|
|
||||||
let tick = inner.tick;
|
|
||||||
if let Some(&idx) = inner.slot_index.get(&coord) {
|
|
||||||
inner.slots[idx].last_access = tick;
|
|
||||||
return Arc::clone(&inner.slots[idx].data); // already cached
|
|
||||||
}
|
|
||||||
|
|
||||||
// Evict until we have room
|
|
||||||
while inner.slots.len() >= inner.max_slots
|
|
||||||
|| (inner.current_bytes + data_len > inner.max_bytes && !inner.slots.is_empty())
|
|
||||||
{
|
|
||||||
// Find LRU slot
|
|
||||||
let lru_idx = inner
|
|
||||||
.slots
|
|
||||||
.iter()
|
|
||||||
.enumerate()
|
|
||||||
.min_by_key(|(_, s)| s.last_access)
|
|
||||||
.map(|(i, _)| i)
|
|
||||||
.unwrap();
|
|
||||||
let removed = inner.slots.swap_remove(lru_idx);
|
|
||||||
inner.slot_index.remove(&removed.coord);
|
|
||||||
// swap_remove moved the former last element into `lru_idx` (unless
|
|
||||||
// it *was* the last element) — fix up that element's index entry.
|
|
||||||
if lru_idx < inner.slots.len() {
|
|
||||||
let moved_coord = inner.slots[lru_idx].coord.clone();
|
|
||||||
inner.slot_index.insert(moved_coord, lru_idx);
|
|
||||||
}
|
|
||||||
inner.current_bytes -= removed.data.len();
|
|
||||||
inner.stats.evictions += 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
inner.current_bytes += data_len;
|
|
||||||
let new_idx = inner.slots.len();
|
|
||||||
inner.slot_index.insert(coord.clone(), new_idx);
|
|
||||||
inner.slots.push(CachedChunk {
|
|
||||||
coord,
|
|
||||||
data: Arc::clone(&data),
|
|
||||||
last_access: tick,
|
|
||||||
});
|
|
||||||
data
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Clear the entire cache (index + decompressed data).
|
/// [`Self::prefetch_hint_in`] for the bound dataset.
|
||||||
|
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
||||||
|
let addr = self.lock().current();
|
||||||
|
self.prefetch_hint_in(addr, next_coords);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ----- Whole-cache operations -----
|
||||||
|
|
||||||
|
/// Clear the entire cache (indexes + decompressed data + stats).
|
||||||
pub fn clear(&self) {
|
pub fn clear(&self) {
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
let mut inner = self.lock();
|
||||||
inner.index = None;
|
inner.datasets.clear();
|
||||||
inner.index_addr = None;
|
inner.current = None;
|
||||||
inner.slots.clear();
|
inner.slots.clear();
|
||||||
inner.slot_index.clear();
|
inner.slot_index.clear();
|
||||||
inner.current_bytes = 0;
|
inner.current_bytes = 0;
|
||||||
inner.tick = 0;
|
inner.tick = 0;
|
||||||
inner.last_coord = None;
|
inner.last_coord = None;
|
||||||
inner.stats = AccessStats::default();
|
inner.stats = AccessStats::default();
|
||||||
inner.chunk_index = None;
|
|
||||||
inner.chunk_layout = None;
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Record that the given chunk coordinates are predicted to be accessed
|
|
||||||
/// soon (bookkeeping only).
|
|
||||||
///
|
|
||||||
/// This does **not** prefetch or pre-decompress anything — it only
|
|
||||||
/// checks whether each coordinate is already in the chunk index and
|
|
||||||
/// updates access-pattern stats accordingly. Real prefetching (e.g.
|
|
||||||
/// background pre-decompression) is not implemented.
|
|
||||||
pub fn prefetch_hint(&self, next_coords: &[ChunkCoord]) {
|
|
||||||
let inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
|
||||||
if inner.index.is_none() {
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
drop(inner);
|
|
||||||
// For each predicted coordinate, verify it exists in the index.
|
|
||||||
// The index is already populated, so this is a no-op for known chunks.
|
|
||||||
// The purpose is to signal intent — callers can pre-decompress if needed.
|
|
||||||
// We touch the stats to record that prefetch hints were issued.
|
|
||||||
let mut inner = self.inner.lock().unwrap_or_else(|e| e.into_inner());
|
|
||||||
for coord in next_coords {
|
|
||||||
let exists = inner
|
|
||||||
.index
|
|
||||||
.as_ref()
|
|
||||||
.map(|idx| idx.contains_key(coord))
|
|
||||||
.unwrap_or(false);
|
|
||||||
if exists {
|
|
||||||
inner.stats.sequential_count += 1;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Return the current access pattern statistics.
|
/// Return the current access pattern statistics.
|
||||||
pub fn access_stats(&self) -> AccessStats {
|
pub fn access_stats(&self) -> AccessStats {
|
||||||
self.inner
|
self.lock().stats.clone()
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.stats
|
|
||||||
.clone()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Update the sweep direction label in the access stats.
|
/// Update the sweep direction label in the access stats.
|
||||||
pub fn set_sweep_direction(&self, direction: &'static str) {
|
pub fn set_sweep_direction(&self, direction: &'static str) {
|
||||||
self.inner
|
self.lock().stats.sweep_direction = Some(direction);
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.stats
|
|
||||||
.sweep_direction = Some(direction);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Number of decompressed chunks currently cached.
|
/// Number of decompressed chunks currently cached (all datasets).
|
||||||
pub fn cached_chunk_count(&self) -> usize {
|
pub fn cached_chunk_count(&self) -> usize {
|
||||||
self.inner
|
self.lock().slots.len()
|
||||||
.lock()
|
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.slots
|
|
||||||
.len()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Total bytes of decompressed data currently cached.
|
/// Total bytes of decompressed data currently cached (all datasets).
|
||||||
pub fn cached_bytes(&self) -> usize {
|
pub fn cached_bytes(&self) -> usize {
|
||||||
self.inner
|
self.lock().current_bytes
|
||||||
.lock()
|
}
|
||||||
.unwrap_or_else(|e| e.into_inner())
|
|
||||||
.current_bytes
|
/// Number of datasets whose chunk index is currently kept.
|
||||||
|
pub fn indexed_dataset_count(&self) -> usize {
|
||||||
|
self.lock().datasets.len()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -808,6 +967,92 @@ mod tests {
|
|||||||
assert_eq!(cache.cached_bytes(), 0);
|
assert_eq!(cache.cached_bytes(), 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn datasets_sharing_coordinates_stay_separate() {
|
||||||
|
let cache = ChunkCache::new();
|
||||||
|
let a = vec![make_chunk(vec![0, 0], 0x100, 8)];
|
||||||
|
let b = vec![make_chunk(vec![0, 0], 0x900, 8)];
|
||||||
|
let got_a = cache.chunks_for::<()>(1, 1, || Ok(a.clone())).unwrap();
|
||||||
|
let got_b = cache.chunks_for::<()>(2, 1, || Ok(b.clone())).unwrap();
|
||||||
|
assert_eq!(got_a[0].address, 0x100);
|
||||||
|
assert_eq!(got_b[0].address, 0x900);
|
||||||
|
// Built once per dataset: a second lookup doesn't call the builder.
|
||||||
|
let again = cache
|
||||||
|
.chunks_for::<()>(1, 1, || panic!("index rebuilt"))
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(again[0].address, 0x100);
|
||||||
|
|
||||||
|
cache.put_decompressed_in(1, vec![0], vec![1; 4]);
|
||||||
|
cache.put_decompressed_in(2, vec![0], vec![2; 4]);
|
||||||
|
assert_eq!(
|
||||||
|
cache.get_decompressed_in(1, &[0]).unwrap().as_slice(),
|
||||||
|
&[1; 4]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
cache.get_decompressed_in(2, &[0]).unwrap().as_slice(),
|
||||||
|
&[2; 4]
|
||||||
|
);
|
||||||
|
assert!(cache.get_decompressed_in(3, &[0]).is_none());
|
||||||
|
assert_eq!(cache.cached_chunk_count(), 2);
|
||||||
|
|
||||||
|
// The bound-dataset methods see only the bound dataset.
|
||||||
|
cache.ensure_dataset(2);
|
||||||
|
assert_eq!(cache.lookup_index(&[0]).unwrap().address, 0x900);
|
||||||
|
assert_eq!(cache.get_decompressed(&[0]).unwrap(), vec![2; 4]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dataset_indexes_are_bounded() {
|
||||||
|
let cache = ChunkCache::new();
|
||||||
|
for addr in 0..(MAX_INDEXED_DATASETS as u64 + 10) {
|
||||||
|
cache
|
||||||
|
.chunks_for::<()>(addr, 1, || Ok(vec![make_chunk(vec![0], addr, 8)]))
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), MAX_INDEXED_DATASETS);
|
||||||
|
|
||||||
|
// One huge index evicts the others but is itself kept.
|
||||||
|
let huge: Vec<ChunkInfo> = (0..MAX_INDEXED_CHUNKS as u64)
|
||||||
|
.map(|i| make_chunk(vec![i], i, 8))
|
||||||
|
.collect();
|
||||||
|
let got = cache.chunks_for::<()>(9999, 1, || Ok(huge)).unwrap();
|
||||||
|
assert_eq!(got.len(), MAX_INDEXED_CHUNKS);
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn concurrent_readers_of_different_datasets_see_their_own_chunks() {
|
||||||
|
let cache = std::sync::Arc::new(ChunkCache::with_capacity(1 << 20, 64));
|
||||||
|
let handles: Vec<_> = (0..8u64)
|
||||||
|
.map(|t| {
|
||||||
|
let cache = std::sync::Arc::clone(&cache);
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
for round in 0..500u64 {
|
||||||
|
let addr = (t + round) % 16;
|
||||||
|
let coord = vec![round % 4];
|
||||||
|
let chunks = cache
|
||||||
|
.chunks_for::<()>(addr, 1, || {
|
||||||
|
Ok((0..4).map(|c| make_chunk(vec![c], addr, 8)).collect())
|
||||||
|
})
|
||||||
|
.unwrap();
|
||||||
|
assert!(chunks.iter().all(|c| c.address == addr));
|
||||||
|
let want = vec![addr as u8; 8];
|
||||||
|
let got = match cache.get_decompressed_in(addr, &coord) {
|
||||||
|
Some(hit) => hit.to_vec(),
|
||||||
|
None => cache
|
||||||
|
.put_decompressed_in(addr, coord, want.clone())
|
||||||
|
.to_vec(),
|
||||||
|
};
|
||||||
|
assert_eq!(got, want);
|
||||||
|
}
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
for h in handles {
|
||||||
|
h.join().unwrap();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn duplicate_insert_is_noop() {
|
fn duplicate_insert_is_noop() {
|
||||||
let cache = ChunkCache::new();
|
let cache = ChunkCache::new();
|
||||||
|
|||||||
@@ -0,0 +1,200 @@
|
|||||||
|
//! Chunk-index linearisation shared by the Fixed Array and Extensible Array
|
||||||
|
//! chunk indexes (reader and writer).
|
||||||
|
//!
|
||||||
|
//! Both indexes store one element per chunk at a *linear* index, and the
|
||||||
|
//! library derives that index from the chunk's scaled coordinates
|
||||||
|
//! (`offset / chunk_dim`) using the dataset's **maximum** dimensions, not its
|
||||||
|
//! current ones (`H5D__farray_idx_get_addr` / `H5D__earray_idx_get_addr`,
|
||||||
|
//! via `layout->max_down_chunks`). A dataset whose current shape is smaller
|
||||||
|
//! than its maxshape therefore has gaps in the index, and laying it out by the
|
||||||
|
//! current shape puts every chunk after the first row in the wrong place.
|
||||||
|
//!
|
||||||
|
//! The Extensible Array adds one more step: its one unlimited dimension has no
|
||||||
|
//! finite chunk count, so the library *swizzles* the coordinates to make that
|
||||||
|
//! dimension the slowest-varying one (`H5VM_swizzle_coords`, which moves
|
||||||
|
//! `coords[unlim_dim]` to the front and shifts the dimensions before it right
|
||||||
|
//! by one) before linearising with `swizzled_max_down_chunks`. When the
|
||||||
|
//! unlimited dimension is already dimension 0 no swizzle happens.
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
extern crate alloc;
|
||||||
|
|
||||||
|
#[cfg(not(feature = "std"))]
|
||||||
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::error::FormatError;
|
||||||
|
|
||||||
|
/// How a chunk index maps linear element indexes to chunk coordinates.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub(crate) struct ChunkGrid {
|
||||||
|
/// Spatial chunk dimensions, in dataset order.
|
||||||
|
chunk_dims: Vec<u64>,
|
||||||
|
/// Chunks per dimension covering the *current* extent, in dataset order.
|
||||||
|
cur_chunks: Vec<u64>,
|
||||||
|
/// Dataset dimension stored at each linearisation position (slowest
|
||||||
|
/// first). The identity except for a swizzled Extensible Array.
|
||||||
|
order: Vec<usize>,
|
||||||
|
/// Linear stride of each linearisation position.
|
||||||
|
down: Vec<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ChunkGrid {
|
||||||
|
/// Grid for a Fixed Array index: row-major over the chunk counts of the
|
||||||
|
/// maximum dimensions (`max_dims`, falling back to the current dimensions
|
||||||
|
/// when the dataspace records none).
|
||||||
|
pub(crate) fn fixed_array(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
Self::build(cur_dims, max_dims, chunk_dims, None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Grid for an Extensible Array index: like the Fixed Array, but the
|
||||||
|
/// unlimited dimension (the one whose maximum is `H5S_UNLIMITED`) is moved
|
||||||
|
/// to the slowest-varying position first.
|
||||||
|
pub(crate) fn extensible_array(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let unlim = max_dims.and_then(|m| m.iter().position(|&d| d == u64::MAX));
|
||||||
|
Self::build(cur_dims, max_dims, chunk_dims, unlim)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build(
|
||||||
|
cur_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
unlim: Option<usize>,
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let rank = chunk_dims.len();
|
||||||
|
if cur_dims.len() != rank || max_dims.is_some_and(|m| m.len() != rank) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"chunk index rank does not match the dataspace".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if chunk_dims.contains(&0) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"chunk dimension is zero".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let cur_chunks: Vec<u64> = cur_dims
|
||||||
|
.iter()
|
||||||
|
.zip(chunk_dims)
|
||||||
|
.map(|(&d, &c)| d.div_ceil(c))
|
||||||
|
.collect();
|
||||||
|
// Chunk counts of the maximum extent. An unlimited dimension has no
|
||||||
|
// finite count; it only ever sits in the slowest position, where its
|
||||||
|
// count never enters a stride. A (corrupt) maximum smaller than the
|
||||||
|
// current extent is widened so no allocated chunk becomes unreachable.
|
||||||
|
let max_chunks: Vec<u64> = (0..rank)
|
||||||
|
.map(|d| {
|
||||||
|
let max = max_dims.map_or(cur_dims[d], |m| m[d]);
|
||||||
|
if max == u64::MAX {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
max.div_ceil(chunk_dims[d]).max(cur_chunks[d])
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let mut order: Vec<usize> = (0..rank).collect();
|
||||||
|
if let Some(u) = unlim {
|
||||||
|
order.remove(u);
|
||||||
|
order.insert(0, u);
|
||||||
|
}
|
||||||
|
let mut down = vec![1u64; rank];
|
||||||
|
for p in (0..rank.saturating_sub(1)).rev() {
|
||||||
|
let next = max_chunks[order[p + 1]];
|
||||||
|
if next == u64::MAX {
|
||||||
|
// Only reachable with more than one unlimited dimension, which
|
||||||
|
// neither index type can describe.
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"array chunk index with more than one unlimited dimension".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
down[p] = down[p + 1].checked_mul(next).ok_or_else(|| {
|
||||||
|
FormatError::Overflow("chunk index linear stride overflows u64".into())
|
||||||
|
})?;
|
||||||
|
}
|
||||||
|
Ok(Self {
|
||||||
|
chunk_dims: chunk_dims.to_vec(),
|
||||||
|
cur_chunks,
|
||||||
|
order,
|
||||||
|
down,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Dataset-space offsets of the chunk stored at linear `index`, or `None`
|
||||||
|
/// when that chunk lies outside the current extent (the index still has a
|
||||||
|
/// slot for it; the library ignores such chunks on read).
|
||||||
|
pub(crate) fn offsets(&self, index: u64) -> Option<Vec<u64>> {
|
||||||
|
let rank = self.chunk_dims.len();
|
||||||
|
let mut offsets = vec![0u64; rank];
|
||||||
|
let mut rem = index;
|
||||||
|
for p in 0..rank {
|
||||||
|
let d = self.order[p];
|
||||||
|
let scaled = rem / self.down[p];
|
||||||
|
rem %= self.down[p];
|
||||||
|
if scaled >= self.cur_chunks[d] {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
offsets[d] = scaled * self.chunk_dims[d];
|
||||||
|
}
|
||||||
|
Some(offsets)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Linear index of the chunk with scaled coordinates `scaled`
|
||||||
|
/// (`offset / chunk_dim` per dimension, in dataset order).
|
||||||
|
pub(crate) fn linear_index(&self, scaled: &[u64]) -> u64 {
|
||||||
|
self.order
|
||||||
|
.iter()
|
||||||
|
.zip(&self.down)
|
||||||
|
.map(|(&d, &stride)| scaled[d] * stride)
|
||||||
|
.sum()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn fixed_array_uses_max_dims() {
|
||||||
|
// shape (4, 6), chunks (2, 3), maxshape (20, 10): 10 x 4 chunk grid.
|
||||||
|
let g = ChunkGrid::fixed_array(&[4, 6], Some(&[20, 10]), &[2, 3]).unwrap();
|
||||||
|
assert_eq!(g.offsets(0), Some(vec![0, 0]));
|
||||||
|
assert_eq!(g.offsets(1), Some(vec![0, 3]));
|
||||||
|
assert_eq!(g.offsets(2), None); // column chunk 2 is beyond the extent
|
||||||
|
assert_eq!(g.offsets(4), Some(vec![2, 0]));
|
||||||
|
assert_eq!(g.offsets(5), Some(vec![2, 3]));
|
||||||
|
assert_eq!(g.offsets(8), None); // row chunk 2 is beyond the extent
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 5);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extensible_array_swizzles_unlimited_dim() {
|
||||||
|
// maxshape (10, None): dim 1 is unlimited and becomes slowest.
|
||||||
|
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[10, u64::MAX]), &[2, 3]).unwrap();
|
||||||
|
// max chunks of dim 0 = 5, so index = c1 * 5 + c0.
|
||||||
|
assert_eq!(g.linear_index(&[1, 0]), 1);
|
||||||
|
assert_eq!(g.linear_index(&[0, 1]), 5);
|
||||||
|
assert_eq!(g.offsets(5), Some(vec![0, 3]));
|
||||||
|
assert_eq!(g.offsets(6), Some(vec![2, 3]));
|
||||||
|
assert_eq!(g.offsets(2), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn extensible_array_unlimited_first_is_row_major() {
|
||||||
|
let g = ChunkGrid::extensible_array(&[4, 6], Some(&[u64::MAX, 30]), &[2, 3]).unwrap();
|
||||||
|
// max chunks of dim 1 = 10.
|
||||||
|
assert_eq!(g.linear_index(&[1, 1]), 11);
|
||||||
|
assert_eq!(g.offsets(11), Some(vec![2, 3]));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn rejects_two_unlimited_dims_after_the_first() {
|
||||||
|
assert!(ChunkGrid::fixed_array(&[4, 6], Some(&[u64::MAX, u64::MAX]), &[2, 3]).is_err());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -15,7 +15,7 @@ use crate::datatype::Datatype;
|
|||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::extensible_array::{ExtensibleArrayHeader, read_extensible_array_chunks};
|
use crate::extensible_array::{ExtensibleArrayHeader, read_extensible_array_chunks};
|
||||||
use crate::filter_pipeline::FilterPipeline;
|
use crate::filter_pipeline::FilterPipeline;
|
||||||
use crate::filters::decompress_chunk;
|
use crate::filters::{all_filters_skipped, decompress_chunk_masked};
|
||||||
use crate::fixed_array::{FixedArrayHeader, read_fixed_array_chunks};
|
use crate::fixed_array::{FixedArrayHeader, read_fixed_array_chunks};
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
@@ -65,11 +65,13 @@ fn decompress_all_chunks(
|
|||||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||||
|
|
||||||
let decompressed = if let Some(pl) = pipeline {
|
let decompressed = if let Some(pl) = pipeline {
|
||||||
if chunk_info.filter_mask == 0 {
|
decompress_chunk_masked(
|
||||||
decompress_chunk(raw_chunk, pl, chunk_total_bytes, element_size)?
|
raw_chunk,
|
||||||
} else {
|
pl,
|
||||||
raw_chunk.to_vec()
|
chunk_total_bytes,
|
||||||
}
|
element_size,
|
||||||
|
chunk_info.filter_mask,
|
||||||
|
)?
|
||||||
} else {
|
} else {
|
||||||
raw_chunk.to_vec()
|
raw_chunk.to_vec()
|
||||||
};
|
};
|
||||||
@@ -223,6 +225,10 @@ pub fn collect_chunk_info(
|
|||||||
collect_chunk_info_inner(file_data, btree_address, ndims, offset_size, length_size, 0)
|
collect_chunk_info_inner(file_data, btree_address, ndims, offset_size, length_size, 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Width of each chunk offset in a v1 chunk B-tree key, independent of the
|
||||||
|
/// file's size-of-offsets.
|
||||||
|
const CHUNK_KEY_OFFSET_SIZE: u8 = 8;
|
||||||
|
|
||||||
/// Maximum recursion depth for chunk B-tree traversal (malformed/cyclic data
|
/// Maximum recursion depth for chunk B-tree traversal (malformed/cyclic data
|
||||||
/// protection), matching `btree_v1.rs`'s `MAX_BTREE_DEPTH`.
|
/// protection), matching `btree_v1.rs`'s `MAX_BTREE_DEPTH`.
|
||||||
const MAX_CHUNK_BTREE_DEPTH: usize = 64;
|
const MAX_CHUNK_BTREE_DEPTH: usize = 64;
|
||||||
@@ -260,8 +266,14 @@ fn collect_chunk_info_inner(
|
|||||||
|
|
||||||
let mut pos = offset + 8 + os * 2; // skip left/right sibling
|
let mut pos = offset + 8 + os * 2; // skip left/right sibling
|
||||||
|
|
||||||
// Key size: chunk_size(4) + filter_mask(4) + ndims * offset_size
|
// Key: chunk_size(4) + filter_mask(4) + one offset per dimension. The
|
||||||
let key_size = 4 + 4 + ndims * os;
|
// offsets are always 8 bytes each — they are dataset coordinates, not file
|
||||||
|
// addresses, so they do not follow the superblock's size-of-offsets (only
|
||||||
|
// the sibling and child addresses do).
|
||||||
|
let key_size = ndims
|
||||||
|
.checked_mul(CHUNK_KEY_OFFSET_SIZE as usize)
|
||||||
|
.and_then(|n| n.checked_add(8))
|
||||||
|
.ok_or_else(|| FormatError::ChunkedReadError("chunk key too large".into()))?;
|
||||||
|
|
||||||
if node_level == 0 {
|
if node_level == 0 {
|
||||||
// Leaf node: keys and children interleaved
|
// Leaf node: keys and children interleaved
|
||||||
@@ -287,8 +299,8 @@ fn collect_chunk_info_inner(
|
|||||||
let mut offsets = Vec::with_capacity(ndims);
|
let mut offsets = Vec::with_capacity(ndims);
|
||||||
let mut kp = pos + 8;
|
let mut kp = pos + 8;
|
||||||
for _ in 0..ndims {
|
for _ in 0..ndims {
|
||||||
offsets.push(read_offset(file_data, kp, offset_size)?);
|
offsets.push(read_offset(file_data, kp, CHUNK_KEY_OFFSET_SIZE)?);
|
||||||
kp += os;
|
kp += CHUNK_KEY_OFFSET_SIZE as usize;
|
||||||
}
|
}
|
||||||
pos += key_size;
|
pos += key_size;
|
||||||
|
|
||||||
@@ -507,6 +519,7 @@ pub fn list_chunks(
|
|||||||
addr_opt,
|
addr_opt,
|
||||||
single_filtered_size,
|
single_filtered_size,
|
||||||
single_filter_mask,
|
single_filter_mask,
|
||||||
|
unfiltered_edges,
|
||||||
) = match layout {
|
) = match layout {
|
||||||
DataLayout::Chunked {
|
DataLayout::Chunked {
|
||||||
chunk_dimensions,
|
chunk_dimensions,
|
||||||
@@ -515,6 +528,7 @@ pub fn list_chunks(
|
|||||||
chunk_index_type,
|
chunk_index_type,
|
||||||
single_chunk_filtered_size,
|
single_chunk_filtered_size,
|
||||||
single_chunk_filter_mask,
|
single_chunk_filter_mask,
|
||||||
|
dont_filter_partial_edge_chunks,
|
||||||
} => (
|
} => (
|
||||||
chunk_dimensions,
|
chunk_dimensions,
|
||||||
*version,
|
*version,
|
||||||
@@ -522,6 +536,7 @@ pub fn list_chunks(
|
|||||||
*btree_address,
|
*btree_address,
|
||||||
*single_chunk_filtered_size,
|
*single_chunk_filtered_size,
|
||||||
*single_chunk_filter_mask,
|
*single_chunk_filter_mask,
|
||||||
|
*dont_filter_partial_edge_chunks,
|
||||||
),
|
),
|
||||||
_ => {
|
_ => {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
@@ -554,7 +569,7 @@ pub fn list_chunks(
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Collect chunks based on version and index type
|
// Collect chunks based on version and index type
|
||||||
let chunks = match (version, chunk_index_type) {
|
let mut chunks = match (version, chunk_index_type) {
|
||||||
(3, _) => {
|
(3, _) => {
|
||||||
let ndims = chunk_dimensions.len(); // rank+1
|
let ndims = chunk_dimensions.len(); // rank+1
|
||||||
collect_chunk_info(file_data, addr, ndims, offset_size, length_size)?
|
collect_chunk_info(file_data, addr, ndims, offset_size, length_size)?
|
||||||
@@ -593,6 +608,7 @@ pub fn list_chunks(
|
|||||||
file_data,
|
file_data,
|
||||||
&header,
|
&header,
|
||||||
&dataspace.dimensions,
|
&dataspace.dimensions,
|
||||||
|
dataspace.max_dimensions.as_deref(),
|
||||||
spatial_chunk_dims,
|
spatial_chunk_dims,
|
||||||
elem_size as u32,
|
elem_size as u32,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -608,6 +624,7 @@ pub fn list_chunks(
|
|||||||
file_data,
|
file_data,
|
||||||
&header,
|
&header,
|
||||||
&dataspace.dimensions,
|
&dataspace.dimensions,
|
||||||
|
dataspace.max_dimensions.as_deref(),
|
||||||
spatial_chunk_dims,
|
spatial_chunk_dims,
|
||||||
elem_size as u32,
|
elem_size as u32,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -633,6 +650,23 @@ pub fn list_chunks(
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// With "don't filter partial edge chunks", a chunk that extends past the
|
||||||
|
// dataset's extent is stored raw while its filter mask still reads 0.
|
||||||
|
// Mark every filter skipped so all read paths copy it as-is.
|
||||||
|
if unfiltered_edges {
|
||||||
|
for chunk in &mut chunks {
|
||||||
|
let partial = chunk
|
||||||
|
.offsets
|
||||||
|
.iter()
|
||||||
|
.zip(&chunk_dims)
|
||||||
|
.zip(&ds_dims)
|
||||||
|
.any(|((&off, &cd), &dd)| off.saturating_add(cd as u64) > dd as u64);
|
||||||
|
if partial {
|
||||||
|
chunk.filter_mask = u32::MAX;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
Ok((chunks, chunk_dims))
|
Ok((chunks, chunk_dims))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -804,24 +838,20 @@ pub fn read_chunked_data_cached(
|
|||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
|
|
||||||
// The per-file cache is shared across datasets; bind it to this one so a
|
// The per-file cache is shared across datasets (and threads); every
|
||||||
// different dataset's chunk index is never reused for this read.
|
// lookup is keyed by this dataset's chunk-index address, so another
|
||||||
cache.ensure_dataset(addr);
|
// dataset's index or chunks are never used for this read.
|
||||||
|
let chunks = cache.chunks_for(addr, rank, || {
|
||||||
// Populate chunk index on first access
|
list_chunks(
|
||||||
if !cache.has_index() {
|
|
||||||
let (chunks, _) = list_chunks(
|
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
elem_size,
|
elem_size,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)
|
||||||
cache.populate_index(&chunks, rank);
|
.map(|(chunks, _)| chunks)
|
||||||
}
|
})?;
|
||||||
|
|
||||||
let chunks = cache.all_indexed_chunks().unwrap_or_default();
|
|
||||||
|
|
||||||
// Assemble output
|
// Assemble output
|
||||||
let total_bytes = checked_byte_len(dataspace.checked_num_elements()?, elem_size)?;
|
let total_bytes = checked_byte_len(dataspace.checked_num_elements()?, elem_size)?;
|
||||||
@@ -876,10 +906,11 @@ pub fn read_chunked_data_cached(
|
|||||||
};
|
};
|
||||||
|
|
||||||
// Chunks stored as-is (no pipeline, or the filter mask says this chunk
|
// Chunks stored as-is (no pipeline, or the filter mask says this chunk
|
||||||
// skipped it) are copied straight from the file bytes: they are already in
|
// skipped every filter) are copied straight from the file bytes: they are
|
||||||
// memory, so routing them through a Vec and then an aligned cache buffer
|
// already in memory, so routing them through a Vec and then an aligned
|
||||||
// was two extra copies of the whole dataset for nothing.
|
// cache buffer was two extra copies of the whole dataset for nothing.
|
||||||
let stored_raw = |c: &ChunkInfo| pipeline.is_none() || c.filter_mask != 0;
|
let stored_raw =
|
||||||
|
|c: &ChunkInfo| pipeline.is_none_or(|pl| all_filters_skipped(pl, c.filter_mask));
|
||||||
let mut misses: Vec<&ChunkInfo> = Vec::new();
|
let mut misses: Vec<&ChunkInfo> = Vec::new();
|
||||||
for chunk_info in &chunks {
|
for chunk_info in &chunks {
|
||||||
if stored_raw(chunk_info) {
|
if stored_raw(chunk_info) {
|
||||||
@@ -887,7 +918,7 @@ pub fn read_chunked_data_cached(
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let coord: Vec<u64> = chunk_info.offsets.iter().take(rank).copied().collect();
|
let coord: Vec<u64> = chunk_info.offsets.iter().take(rank).copied().collect();
|
||||||
match cache.get_decompressed_aligned(&coord) {
|
match cache.get_decompressed_in(addr, &coord) {
|
||||||
Some(cached) => place(&cached, chunk_info),
|
Some(cached) => place(&cached, chunk_info),
|
||||||
None => misses.push(chunk_info),
|
None => misses.push(chunk_info),
|
||||||
}
|
}
|
||||||
@@ -901,7 +932,13 @@ pub fn read_chunked_data_cached(
|
|||||||
let cache_them = total_bytes <= cache.max_bytes();
|
let cache_them = total_bytes <= cache.max_bytes();
|
||||||
if let Some(pl) = pipeline {
|
if let Some(pl) = pipeline {
|
||||||
let decode = |c: &&ChunkInfo| -> Result<Vec<u8>, FormatError> {
|
let decode = |c: &&ChunkInfo| -> Result<Vec<u8>, FormatError> {
|
||||||
decompress_chunk(raw_bytes(c)?, pl, chunk_total_bytes, elem_size as u32)
|
decompress_chunk_masked(
|
||||||
|
raw_bytes(c)?,
|
||||||
|
pl,
|
||||||
|
chunk_total_bytes,
|
||||||
|
elem_size as u32,
|
||||||
|
c.filter_mask,
|
||||||
|
)
|
||||||
};
|
};
|
||||||
for batch in misses.chunks(DECODE_BATCH) {
|
for batch in misses.chunks(DECODE_BATCH) {
|
||||||
#[cfg(feature = "parallel")]
|
#[cfg(feature = "parallel")]
|
||||||
@@ -918,7 +955,7 @@ pub fn read_chunked_data_cached(
|
|||||||
let data = data?;
|
let data = data?;
|
||||||
if cache_them {
|
if cache_them {
|
||||||
let coord: Vec<u64> = chunk_info.offsets.iter().take(rank).copied().collect();
|
let coord: Vec<u64> = chunk_info.offsets.iter().take(rank).copied().collect();
|
||||||
let cached = cache.put_decompressed(coord, data);
|
let cached = cache.put_decompressed_in(addr, coord, data);
|
||||||
place(&cached, chunk_info);
|
place(&cached, chunk_info);
|
||||||
} else {
|
} else {
|
||||||
place(&data, chunk_info);
|
place(&data, chunk_info);
|
||||||
@@ -1122,24 +1159,20 @@ pub fn read_chunked_data_sweep(
|
|||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
|
|
||||||
// The per-file cache is shared across datasets; bind it to this one so a
|
// The per-file cache is shared across datasets (and threads); every
|
||||||
// different dataset's chunk index is never reused for this read.
|
// lookup is keyed by this dataset's chunk-index address, so another
|
||||||
cache.ensure_dataset(addr);
|
// dataset's index or chunks are never used for this read.
|
||||||
|
let chunks = cache.chunks_for(addr, rank, || {
|
||||||
// Populate chunk index on first access
|
list_chunks(
|
||||||
if !cache.has_index() {
|
|
||||||
let (chunks, _) = list_chunks(
|
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
elem_size,
|
elem_size,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)
|
||||||
cache.populate_index(&chunks, rank);
|
.map(|(chunks, _)| chunks)
|
||||||
}
|
})?;
|
||||||
|
|
||||||
let chunks = cache.all_indexed_chunks().unwrap_or_default();
|
|
||||||
|
|
||||||
// Assemble output
|
// Assemble output
|
||||||
let total_bytes = checked_byte_len(dataspace.checked_num_elements()?, elem_size)?;
|
let total_bytes = checked_byte_len(dataspace.checked_num_elements()?, elem_size)?;
|
||||||
@@ -1170,12 +1203,12 @@ pub fn read_chunked_data_sweep(
|
|||||||
|
|
||||||
// Issue prefetch hint for predicted next chunks
|
// Issue prefetch hint for predicted next chunks
|
||||||
if !sweep.predicted_next.is_empty() {
|
if !sweep.predicted_next.is_empty() {
|
||||||
cache.prefetch_hint(&sweep.predicted_next);
|
cache.prefetch_hint_in(addr, &sweep.predicted_next);
|
||||||
cache.set_sweep_direction(sweep.direction);
|
cache.set_sweep_direction(sweep.direction);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Try decompressed cache first
|
// Try decompressed cache first
|
||||||
let decompressed = if let Some(cached) = cache.get_decompressed_aligned(&coord) {
|
let decompressed = if let Some(cached) = cache.get_decompressed_in(addr, &coord) {
|
||||||
cached
|
cached
|
||||||
} else {
|
} else {
|
||||||
// Decompress from file
|
// Decompress from file
|
||||||
@@ -1184,15 +1217,17 @@ pub fn read_chunked_data_sweep(
|
|||||||
ensure_len(file_data, c_addr, size)?;
|
ensure_len(file_data, c_addr, size)?;
|
||||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||||
let dec = if let Some(pl) = pipeline {
|
let dec = if let Some(pl) = pipeline {
|
||||||
if chunk_info.filter_mask == 0 {
|
decompress_chunk_masked(
|
||||||
decompress_chunk(raw_chunk, pl, chunk_total_bytes, elem_size as u32)?
|
raw_chunk,
|
||||||
} else {
|
pl,
|
||||||
raw_chunk.to_vec()
|
chunk_total_bytes,
|
||||||
}
|
elem_size as u32,
|
||||||
|
chunk_info.filter_mask,
|
||||||
|
)?
|
||||||
} else {
|
} else {
|
||||||
raw_chunk.to_vec()
|
raw_chunk.to_vec()
|
||||||
};
|
};
|
||||||
cache.put_decompressed(coord, dec)
|
cache.put_decompressed_in(addr, coord, dec)
|
||||||
};
|
};
|
||||||
|
|
||||||
let chunk_offsets: Vec<usize> = chunk_info
|
let chunk_offsets: Vec<usize> = chunk_info
|
||||||
@@ -1276,48 +1311,34 @@ pub fn read_chunked_data_indexed(
|
|||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
|
|
||||||
// The per-file cache is shared across datasets; bind it to this one so a
|
// Chunk index and assembly plan for this dataset, built on first access
|
||||||
// different dataset's chunk index is never reused for this read.
|
// and kept per dataset (keyed by chunk-index address) in the shared cache.
|
||||||
cache.ensure_dataset(addr);
|
let plan = cache.chunk_layout_for(
|
||||||
|
addr,
|
||||||
// Build chunk index on first access
|
rank,
|
||||||
if !cache.has_chunk_index() {
|
|| {
|
||||||
let (chunks, _) = list_chunks(
|
list_chunks(
|
||||||
file_data,
|
file_data,
|
||||||
layout,
|
layout,
|
||||||
dataspace,
|
dataspace,
|
||||||
elem_size,
|
elem_size,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)
|
||||||
cache.populate_chunk_index(&chunks, rank);
|
.map(|(chunks, _)| chunks)
|
||||||
// Also populate the legacy index for compatibility
|
},
|
||||||
if !cache.has_index() {
|
&ds_dims,
|
||||||
cache.populate_index(&chunks, rank);
|
&chunk_dims,
|
||||||
}
|
elem_size,
|
||||||
}
|
)?;
|
||||||
|
let chunk_total_bytes = plan.chunk_total_bytes;
|
||||||
// Build chunk layout on first access
|
|
||||||
if !cache.has_chunk_layout() {
|
|
||||||
cache.populate_chunk_layout(&ds_dims, &chunk_dims, elem_size);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Get the layout info (mappings, output size, chunk total bytes)
|
|
||||||
let (mappings_info, output_bytes, chunk_total_bytes) = cache
|
|
||||||
.with_chunk_layout(|layout| {
|
|
||||||
let info: Vec<_> = layout
|
|
||||||
.mappings
|
|
||||||
.iter()
|
|
||||||
.map(|m| (m.coord.clone(), m.file_offset, m.file_size, m.filter_mask))
|
|
||||||
.collect();
|
|
||||||
(info, layout.output_bytes, layout.chunk_total_bytes)
|
|
||||||
})
|
|
||||||
.ok_or_else(|| FormatError::ChunkedReadError("chunk layout not available".into()))?;
|
|
||||||
|
|
||||||
// Decompress chunks (using LRU cache where possible)
|
// Decompress chunks (using LRU cache where possible)
|
||||||
let mut chunk_buffers: Vec<Arc<CacheAlignedBuffer>> = Vec::with_capacity(mappings_info.len());
|
let mut chunk_buffers: Vec<Arc<CacheAlignedBuffer>> = Vec::with_capacity(plan.mappings.len());
|
||||||
for (coord, file_offset, file_size, filter_mask) in &mappings_info {
|
for m in &plan.mappings {
|
||||||
if let Some(cached) = cache.get_decompressed_aligned(coord) {
|
let (coord, file_offset, file_size, filter_mask) =
|
||||||
|
(&m.coord, &m.file_offset, &m.file_size, &m.filter_mask);
|
||||||
|
if let Some(cached) = cache.get_decompressed_in(addr, coord) {
|
||||||
chunk_buffers.push(cached);
|
chunk_buffers.push(cached);
|
||||||
} else {
|
} else {
|
||||||
let c_addr = *file_offset as usize;
|
let c_addr = *file_offset as usize;
|
||||||
@@ -1325,26 +1346,26 @@ pub fn read_chunked_data_indexed(
|
|||||||
ensure_len(file_data, c_addr, size)?;
|
ensure_len(file_data, c_addr, size)?;
|
||||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||||
let decompressed = if let Some(pl) = pipeline {
|
let decompressed = if let Some(pl) = pipeline {
|
||||||
if *filter_mask == 0 {
|
decompress_chunk_masked(
|
||||||
decompress_chunk(raw_chunk, pl, chunk_total_bytes, elem_size as u32)?
|
raw_chunk,
|
||||||
} else {
|
pl,
|
||||||
raw_chunk.to_vec()
|
chunk_total_bytes,
|
||||||
}
|
elem_size as u32,
|
||||||
|
*filter_mask,
|
||||||
|
)?
|
||||||
} else {
|
} else {
|
||||||
raw_chunk.to_vec()
|
raw_chunk.to_vec()
|
||||||
};
|
};
|
||||||
let aligned = CacheAlignedBuffer::from_vec(decompressed);
|
let aligned = CacheAlignedBuffer::from_vec(decompressed);
|
||||||
let arc = cache.put_decompressed_aligned(coord.clone(), aligned);
|
let arc = cache.put_decompressed_aligned_in(addr, coord.clone(), aligned);
|
||||||
chunk_buffers.push(arc);
|
chunk_buffers.push(arc);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Assemble using pre-computed layout
|
// Assemble using pre-computed layout
|
||||||
let mut output = vec![0u8; output_bytes];
|
let mut output = vec![0u8; plan.output_bytes];
|
||||||
let data_refs: Vec<&[u8]> = chunk_buffers.iter().map(|b| b.as_slice()).collect();
|
let data_refs: Vec<&[u8]> = chunk_buffers.iter().map(|b| b.as_slice()).collect();
|
||||||
cache.with_chunk_layout(|layout| {
|
plan.assemble(&data_refs, &mut output);
|
||||||
layout.assemble(&data_refs, &mut output);
|
|
||||||
});
|
|
||||||
|
|
||||||
Ok(output)
|
Ok(output)
|
||||||
}
|
}
|
||||||
@@ -1592,7 +1613,8 @@ mod tests {
|
|||||||
} else {
|
} else {
|
||||||
0
|
0
|
||||||
};
|
};
|
||||||
write_offset(&mut buf, off, offset_size);
|
// Key offsets are always 8 bytes (they are coordinates).
|
||||||
|
write_offset(&mut buf, off, 8);
|
||||||
}
|
}
|
||||||
// Child: address
|
// Child: address
|
||||||
write_offset(&mut buf, chunk.address, offset_size);
|
write_offset(&mut buf, chunk.address, offset_size);
|
||||||
@@ -1602,7 +1624,7 @@ mod tests {
|
|||||||
buf.extend_from_slice(&0u32.to_le_bytes()); // chunk_size
|
buf.extend_from_slice(&0u32.to_le_bytes()); // chunk_size
|
||||||
buf.extend_from_slice(&0u32.to_le_bytes()); // filter_mask
|
buf.extend_from_slice(&0u32.to_le_bytes()); // filter_mask
|
||||||
for _ in 0..ndims {
|
for _ in 0..ndims {
|
||||||
write_offset(&mut buf, u64::MAX, offset_size);
|
write_offset(&mut buf, u64::MAX, 8);
|
||||||
}
|
}
|
||||||
|
|
||||||
buf
|
buf
|
||||||
@@ -1680,6 +1702,37 @@ mod tests {
|
|||||||
assert_eq!(result[2].address, 0x300);
|
assert_eq!(result[2].address, 0x300);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn collect_chunks_with_four_byte_addresses() {
|
||||||
|
// Sibling and child addresses are 4 bytes; the key offsets stay 8.
|
||||||
|
let ndims = 3;
|
||||||
|
let os: u8 = 4;
|
||||||
|
let chunks = vec![
|
||||||
|
ChunkInfo {
|
||||||
|
chunk_size: 80,
|
||||||
|
filter_mask: 2,
|
||||||
|
offsets: vec![0, 5, 0],
|
||||||
|
address: 0x1000,
|
||||||
|
},
|
||||||
|
ChunkInfo {
|
||||||
|
chunk_size: 96,
|
||||||
|
filter_mask: 0,
|
||||||
|
offsets: vec![8, 10, 0],
|
||||||
|
address: 0x2000,
|
||||||
|
},
|
||||||
|
];
|
||||||
|
let btree = build_chunk_btree_leaf(&chunks, ndims, os);
|
||||||
|
assert_eq!(btree.len(), 8 + 2 * 4 + 2 * (8 + 3 * 8 + 4) + (8 + 3 * 8));
|
||||||
|
let result = collect_chunk_info(&btree, 0, ndims, os, os).unwrap();
|
||||||
|
assert_eq!(result.len(), 2);
|
||||||
|
for (got, want) in result.iter().zip(&chunks) {
|
||||||
|
assert_eq!(got.offsets, want.offsets);
|
||||||
|
assert_eq!(got.address, want.address);
|
||||||
|
assert_eq!(got.chunk_size, want.chunk_size);
|
||||||
|
assert_eq!(got.filter_mask, want.filter_mask);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn collect_empty_btree() {
|
fn collect_empty_btree() {
|
||||||
let ndims = 2;
|
let ndims = 2;
|
||||||
@@ -1774,6 +1827,7 @@ mod tests {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
};
|
};
|
||||||
|
|
||||||
let dataspace = Dataspace {
|
let dataspace = Dataspace {
|
||||||
@@ -1797,6 +1851,7 @@ mod tests {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
};
|
};
|
||||||
let dataspace = Dataspace {
|
let dataspace = Dataspace {
|
||||||
space_type: DataspaceType::Simple,
|
space_type: DataspaceType::Simple,
|
||||||
@@ -1954,6 +2009,7 @@ mod tests {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
};
|
};
|
||||||
let dataspace = Dataspace {
|
let dataspace = Dataspace {
|
||||||
space_type: DataspaceType::Simple,
|
space_type: DataspaceType::Simple,
|
||||||
@@ -2036,6 +2092,7 @@ mod tests {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
};
|
};
|
||||||
let dataspace = Dataspace {
|
let dataspace = Dataspace {
|
||||||
space_type: DataspaceType::Simple,
|
space_type: DataspaceType::Simple,
|
||||||
@@ -2198,6 +2255,7 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
};
|
};
|
||||||
let dataspace = Dataspace {
|
let dataspace = Dataspace {
|
||||||
space_type: DataspaceType::Simple,
|
space_type: DataspaceType::Simple,
|
||||||
@@ -2227,12 +2285,12 @@ mod tests {
|
|||||||
let datatype = make_f64_type();
|
let datatype = make_f64_type();
|
||||||
let cache = ChunkCache::new();
|
let cache = ChunkCache::new();
|
||||||
|
|
||||||
assert!(!cache.has_index());
|
assert_eq!(cache.indexed_dataset_count(), 0);
|
||||||
let raw = read_chunked_data_cached(
|
let raw = read_chunked_data_cached(
|
||||||
&file_data, &layout, &dataspace, &datatype, None, 8, 8, &cache,
|
&file_data, &layout, &dataspace, &datatype, None, 8, 8, &cache,
|
||||||
)
|
)
|
||||||
.unwrap();
|
.unwrap();
|
||||||
assert!(cache.has_index());
|
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||||
assert_eq!(raw.len(), 20 * 8);
|
assert_eq!(raw.len(), 20 * 8);
|
||||||
for i in 0..20 {
|
for i in 0..20 {
|
||||||
let val = f64::from_le_bytes(raw[i * 8..(i + 1) * 8].try_into().unwrap());
|
let val = f64::from_le_bytes(raw[i * 8..(i + 1) * 8].try_into().unwrap());
|
||||||
@@ -2254,7 +2312,7 @@ mod tests {
|
|||||||
&file_data, &layout, &dataspace, &datatype, None, 8, 8, &cache,
|
&file_data, &layout, &dataspace, &datatype, None, 8, 8, &cache,
|
||||||
)
|
)
|
||||||
.unwrap();
|
.unwrap();
|
||||||
assert!(cache.has_index());
|
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||||
assert_eq!(cache.cached_chunk_count(), 0);
|
assert_eq!(cache.cached_chunk_count(), 0);
|
||||||
|
|
||||||
// Second read — reuses the cached index
|
// Second read — reuses the cached index
|
||||||
@@ -2263,6 +2321,7 @@ mod tests {
|
|||||||
)
|
)
|
||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(raw1, raw2);
|
assert_eq!(raw1, raw2);
|
||||||
|
assert_eq!(cache.indexed_dataset_count(), 1);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -4,15 +4,16 @@
|
|||||||
extern crate alloc;
|
extern crate alloc;
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
use crate::checksum::jenkins_lookup3;
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::chunk_cache::{CACHE_LINE_SIZE, align_to_cache_line};
|
use crate::chunk_cache::{CACHE_LINE_SIZE, align_to_cache_line};
|
||||||
|
use crate::chunk_grid::ChunkGrid;
|
||||||
use crate::ea_writer;
|
use crate::ea_writer;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::filter_pipeline::{
|
use crate::filter_pipeline::{
|
||||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_PCODEC, FILTER_SHUFFLE, FILTER_ZSTD,
|
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_PCODEC, FILTER_PCODEC_NAME,
|
||||||
FilterDescription, FilterPipeline,
|
FILTER_SHUFFLE, FILTER_ZSTD, FilterDescription, FilterPipeline,
|
||||||
};
|
};
|
||||||
use crate::filters::compress_chunk;
|
use crate::filters::compress_chunk;
|
||||||
/// Round a file offset up to the next cache-line boundary.
|
/// Round a file offset up to the next cache-line boundary.
|
||||||
@@ -44,7 +45,8 @@ pub struct ChunkOptions {
|
|||||||
pub lz4: bool,
|
pub lz4: bool,
|
||||||
/// Zstandard compression level (1-22), None = no zstd. Filter ID 32015.
|
/// Zstandard compression level (1-22), None = no zstd. Filter ID 32015.
|
||||||
pub zstd_level: Option<u32>,
|
pub zstd_level: Option<u32>,
|
||||||
/// Pcodec lossless numerical compression. Filter ID 32023.
|
/// Pcodec lossless numerical compression. Private, unregistered filter
|
||||||
|
/// ID [`FILTER_PCODEC`] (480): only clawhdf5 can read it.
|
||||||
pub pcodec: bool,
|
pub pcodec: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -115,7 +117,7 @@ impl ChunkOptions {
|
|||||||
if self.pcodec {
|
if self.pcodec {
|
||||||
filters.push(FilterDescription {
|
filters.push(FilterDescription {
|
||||||
filter_id: FILTER_PCODEC,
|
filter_id: FILTER_PCODEC,
|
||||||
name: Some("pcodec".into()),
|
name: Some(FILTER_PCODEC_NAME.into()),
|
||||||
flags: 0,
|
flags: 0,
|
||||||
client_data: vec![element_size],
|
client_data: vec![element_size],
|
||||||
});
|
});
|
||||||
@@ -443,6 +445,27 @@ fn serialize_v4_fixed_array(
|
|||||||
element_size: u32,
|
element_size: u32,
|
||||||
max_bits: u8,
|
max_bits: u8,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
|
let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size);
|
||||||
|
|
||||||
|
// chunk index type = 3 (Fixed Array)
|
||||||
|
buf.push(3);
|
||||||
|
|
||||||
|
// max_dblk_page_nelmts_bits — must match FAHD max_nelmts_bits
|
||||||
|
buf.push(max_bits);
|
||||||
|
|
||||||
|
// Fixed Array header address
|
||||||
|
match offset_size {
|
||||||
|
4 => buf.extend_from_slice(&(fixed_array_address as u32).to_le_bytes()),
|
||||||
|
8 => buf.extend_from_slice(&fixed_array_address.to_le_bytes()),
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The part of a v4 chunked layout message before the chunk index type:
|
||||||
|
/// version, class, flags and the chunk dimensions (plus the element size).
|
||||||
|
fn layout_v4_chunked_prefix(chunk_dims: &[u32], element_size: u32) -> Vec<u8> {
|
||||||
let mut buf = Vec::new();
|
let mut buf = Vec::new();
|
||||||
buf.push(4); // version
|
buf.push(4); // version
|
||||||
buf.push(2); // class = chunked
|
buf.push(2); // class = chunked
|
||||||
@@ -482,125 +505,143 @@ fn serialize_v4_fixed_array(
|
|||||||
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
4 => buf.extend_from_slice(&element_size.to_le_bytes()),
|
||||||
_ => {}
|
_ => {}
|
||||||
}
|
}
|
||||||
|
|
||||||
// chunk index type = 3 (Fixed Array)
|
|
||||||
buf.push(3);
|
|
||||||
|
|
||||||
// max_dblk_page_nelmts_bits — must match FAHD max_nelmts_bits
|
|
||||||
buf.push(max_bits);
|
|
||||||
|
|
||||||
// Fixed Array header address
|
|
||||||
match offset_size {
|
|
||||||
4 => buf.extend_from_slice(&(fixed_array_address as u32).to_le_bytes()),
|
|
||||||
8 => buf.extend_from_slice(&fixed_array_address.to_le_bytes()),
|
|
||||||
_ => {}
|
|
||||||
}
|
|
||||||
|
|
||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// log2 of the elements per Fixed Array data block page (the library's
|
||||||
|
/// default, `H5D_FARRAY_MAX_DBLK_PAGE_NELMTS_BITS`).
|
||||||
|
const FA_PAGE_BITS: u8 = 10;
|
||||||
|
|
||||||
|
pub(crate) fn push_addr(buf: &mut Vec<u8>, addr: u64, offset_size: u8) {
|
||||||
|
match offset_size {
|
||||||
|
4 => buf.extend_from_slice(&(addr as u32).to_le_bytes()),
|
||||||
|
_ => buf.extend_from_slice(&addr.to_le_bytes()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Width of the chunk-size field of a filtered chunk index element. Must
|
||||||
|
/// match the library's `H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` (the EA and
|
||||||
|
/// B-tree v2 indexes use the same formula):
|
||||||
|
/// `1 + ((log2(unfiltered chunk bytes) + 8) / 8)`, capped at 8.
|
||||||
|
pub(crate) fn filtered_chunk_size_len(slots: &[Option<WrittenChunk>]) -> usize {
|
||||||
|
let max_raw = slots
|
||||||
|
.iter()
|
||||||
|
.flatten()
|
||||||
|
.map(|c| c.raw_size)
|
||||||
|
.max()
|
||||||
|
.unwrap_or(1);
|
||||||
|
let log2_val = if max_raw <= 1 {
|
||||||
|
0
|
||||||
|
} else {
|
||||||
|
63 - max_raw.leading_zeros()
|
||||||
|
};
|
||||||
|
(1 + ((log2_val + 8) / 8) as usize).min(8)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Append one chunk index element: the chunk's address, plus its stored size
|
||||||
|
/// and filter mask when the dataset is filtered. `None` is an unallocated
|
||||||
|
/// chunk (undefined address, zero size and mask).
|
||||||
|
pub(crate) fn push_index_element(
|
||||||
|
buf: &mut Vec<u8>,
|
||||||
|
slot: Option<&WrittenChunk>,
|
||||||
|
offset_size: u8,
|
||||||
|
chunk_size_bytes: Option<usize>,
|
||||||
|
) {
|
||||||
|
match slot {
|
||||||
|
Some(c) => {
|
||||||
|
push_addr(buf, c.address, offset_size);
|
||||||
|
if let Some(n) = chunk_size_bytes {
|
||||||
|
buf.extend_from_slice(&c.compressed_size.to_le_bytes()[..n]);
|
||||||
|
buf.extend_from_slice(&c.filter_mask.to_le_bytes());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
None => {
|
||||||
|
buf.extend(core::iter::repeat_n(0xFF, offset_size as usize));
|
||||||
|
if let Some(n) = chunk_size_bytes {
|
||||||
|
buf.extend(core::iter::repeat_n(0x00, n + 4));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Build a complete Fixed Array at a known absolute address.
|
/// Build a complete Fixed Array at a known absolute address.
|
||||||
|
///
|
||||||
|
/// `slots` holds one entry per element of the array, i.e. per chunk of the
|
||||||
|
/// dataset's *maximum* extent in the order [`crate::chunk_grid`] defines;
|
||||||
|
/// `None` marks a chunk that is not allocated. An array with more elements
|
||||||
|
/// than fit in one page (`2^FA_PAGE_BITS`) gets a paged data block: a
|
||||||
|
/// page-init bitmap after the prefix, then one checksummed page per
|
||||||
|
/// `2^FA_PAGE_BITS` elements, the last one short (`H5FA__dblock_create`).
|
||||||
pub fn build_fixed_array_at(
|
pub fn build_fixed_array_at(
|
||||||
chunks: &[WrittenChunk],
|
slots: &[Option<WrittenChunk>],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
has_filters: bool,
|
has_filters: bool,
|
||||||
fa_base_address: u64,
|
fa_base_address: u64,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let num_elements = chunks.len();
|
let num_elements = slots.len();
|
||||||
|
|
||||||
// For filtered chunks, compute chunk_size encoding width.
|
|
||||||
// Must match the HDF5 C library's H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN macro:
|
|
||||||
// chunk_size_len = 1 + ((H5VM_log2_gen(chunk.size) + 8) / 8)
|
|
||||||
// where chunk.size is the unfiltered chunk size in bytes (product of all chunk dims).
|
|
||||||
let chunk_size_bytes: usize = if has_filters {
|
|
||||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
|
||||||
let log2_val = if max_raw <= 1 {
|
|
||||||
0
|
|
||||||
} else {
|
|
||||||
63 - max_raw.leading_zeros()
|
|
||||||
};
|
|
||||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
|
||||||
len.min(8)
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
|
|
||||||
let elem_size = if has_filters {
|
|
||||||
os + chunk_size_bytes + 4
|
|
||||||
} else {
|
|
||||||
os
|
|
||||||
};
|
|
||||||
|
|
||||||
|
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||||
|
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||||
|
|
||||||
// FAHD total size
|
// FAHD total size
|
||||||
let nelmts_field_size = length_size as usize;
|
let fahd_total_size = 4 + 1 + 1 + 1 + 1 + length_size as usize + os + 4;
|
||||||
let fahd_total_size = 4 + 1 + 1 + 1 + 1 + nelmts_field_size + os + 4;
|
|
||||||
let fadb_address = fa_base_address + fahd_total_size as u64;
|
let fadb_address = fa_base_address + fahd_total_size as u64;
|
||||||
|
|
||||||
// Build FAHD
|
|
||||||
let mut fahd = Vec::with_capacity(fahd_total_size);
|
let mut fahd = Vec::with_capacity(fahd_total_size);
|
||||||
fahd.extend_from_slice(b"FAHD");
|
fahd.extend_from_slice(b"FAHD");
|
||||||
fahd.push(0); // version
|
fahd.push(0); // version
|
||||||
fahd.push(client_id);
|
fahd.push(client_id);
|
||||||
fahd.push(elem_size as u8);
|
fahd.push(elem_size as u8);
|
||||||
|
fahd.push(FA_PAGE_BITS);
|
||||||
// max_nelmts_bits: use 10 as default (page_size = 1024), matching h5py convention
|
|
||||||
let max_bits: u8 = 10;
|
|
||||||
fahd.push(max_bits);
|
|
||||||
|
|
||||||
match length_size {
|
match length_size {
|
||||||
4 => fahd.extend_from_slice(&(num_elements as u32).to_le_bytes()),
|
4 => fahd.extend_from_slice(&(num_elements as u32).to_le_bytes()),
|
||||||
8 => fahd.extend_from_slice(&(num_elements as u64).to_le_bytes()),
|
|
||||||
_ => fahd.extend_from_slice(&(num_elements as u64).to_le_bytes()),
|
_ => fahd.extend_from_slice(&(num_elements as u64).to_le_bytes()),
|
||||||
}
|
}
|
||||||
|
push_addr(&mut fahd, fadb_address, offset_size);
|
||||||
match offset_size {
|
|
||||||
4 => fahd.extend_from_slice(&(fadb_address as u32).to_le_bytes()),
|
|
||||||
8 => fahd.extend_from_slice(&fadb_address.to_le_bytes()),
|
|
||||||
_ => fahd.extend_from_slice(&fadb_address.to_le_bytes()),
|
|
||||||
}
|
|
||||||
|
|
||||||
// Checksum
|
|
||||||
let checksum = jenkins_lookup3(&fahd);
|
let checksum = jenkins_lookup3(&fahd);
|
||||||
fahd.extend_from_slice(&checksum.to_le_bytes());
|
fahd.extend_from_slice(&checksum.to_le_bytes());
|
||||||
|
|
||||||
assert_eq!(fahd.len(), fahd_total_size);
|
assert_eq!(fahd.len(), fahd_total_size);
|
||||||
|
|
||||||
// Build FADB
|
// FADB prefix
|
||||||
let mut fadb = Vec::new();
|
let mut fadb = Vec::new();
|
||||||
fadb.extend_from_slice(b"FADB");
|
fadb.extend_from_slice(b"FADB");
|
||||||
fadb.push(0); // version
|
fadb.push(0); // version
|
||||||
fadb.push(client_id);
|
fadb.push(client_id);
|
||||||
|
push_addr(&mut fadb, fa_base_address, offset_size);
|
||||||
|
|
||||||
// header address
|
let page_nelmts = 1usize << FA_PAGE_BITS;
|
||||||
match offset_size {
|
if num_elements <= page_nelmts {
|
||||||
4 => fadb.extend_from_slice(&(fa_base_address as u32).to_le_bytes()),
|
// Unpaged: the elements follow the prefix, one checksum over both.
|
||||||
8 => fadb.extend_from_slice(&fa_base_address.to_le_bytes()),
|
for slot in slots {
|
||||||
_ => fadb.extend_from_slice(&fa_base_address.to_le_bytes()),
|
push_index_element(&mut fadb, slot.as_ref(), offset_size, chunk_size_bytes);
|
||||||
}
|
|
||||||
|
|
||||||
// Element data
|
|
||||||
for chunk in chunks {
|
|
||||||
match offset_size {
|
|
||||||
4 => fadb.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
|
||||||
8 => fadb.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
_ => fadb.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
if has_filters {
|
let fadb_checksum = jenkins_lookup3(&fadb);
|
||||||
// Write compressed size using chunk_size_bytes (variable width)
|
fadb.extend_from_slice(&fadb_checksum.to_le_bytes());
|
||||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
} else {
|
||||||
fadb.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
// Paged: every page is written, so every page-init bit is set
|
||||||
fadb.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
// (MSB-first, as `H5VM_bit_set` packs them). The prefix and bitmap
|
||||||
|
// share a checksum; each page carries its own.
|
||||||
|
let npages = num_elements.div_ceil(page_nelmts);
|
||||||
|
let mut bitmap = vec![0u8; npages.div_ceil(8)];
|
||||||
|
for p in 0..npages {
|
||||||
|
bitmap[p / 8] |= 0x80 >> (p % 8);
|
||||||
|
}
|
||||||
|
fadb.extend_from_slice(&bitmap);
|
||||||
|
let prefix_checksum = jenkins_lookup3(&fadb);
|
||||||
|
fadb.extend_from_slice(&prefix_checksum.to_le_bytes());
|
||||||
|
for page in slots.chunks(page_nelmts) {
|
||||||
|
let start = fadb.len();
|
||||||
|
for slot in page {
|
||||||
|
push_index_element(&mut fadb, slot.as_ref(), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let page_checksum = jenkins_lookup3(&fadb[start..]);
|
||||||
|
fadb.extend_from_slice(&page_checksum.to_le_bytes());
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// FADB checksum
|
|
||||||
let fadb_checksum = jenkins_lookup3(&fadb);
|
|
||||||
fadb.extend_from_slice(&fadb_checksum.to_le_bytes());
|
|
||||||
|
|
||||||
let mut combined = fahd;
|
let mut combined = fahd;
|
||||||
combined.extend_from_slice(&fadb);
|
combined.extend_from_slice(&fadb);
|
||||||
combined
|
combined
|
||||||
@@ -667,7 +708,8 @@ pub fn build_chunked_data_from_precompressed(
|
|||||||
pre: &PrecompressedChunks,
|
pre: &PrecompressedChunks,
|
||||||
base_address: u64,
|
base_address: u64,
|
||||||
maxshape: Option<&[u64]>,
|
maxshape: Option<&[u64]>,
|
||||||
) -> ChunkedDataResult {
|
) -> Result<ChunkedDataResult, FormatError> {
|
||||||
|
let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?;
|
||||||
let offset_size: u8 = 8;
|
let offset_size: u8 = 8;
|
||||||
let length_size: u8 = 8;
|
let length_size: u8 = 8;
|
||||||
let num_chunks = pre.chunks.len();
|
let num_chunks = pre.chunks.len();
|
||||||
@@ -693,71 +735,329 @@ pub fn build_chunked_data_from_precompressed(
|
|||||||
}
|
}
|
||||||
|
|
||||||
let chunk_dims_u32: Vec<u32> = pre.chunk_dims.iter().map(|&d| d as u32).collect();
|
let chunk_dims_u32: Vec<u32> = pre.chunk_dims.iter().map(|&d| d as u32).collect();
|
||||||
let use_extensible = maxshape.is_some_and(|ms| ms.contains(&u64::MAX));
|
|
||||||
|
|
||||||
let aligned_idx = align_to_cache_line(data_buf.len());
|
let aligned_idx = align_to_cache_line(data_buf.len());
|
||||||
if aligned_idx > data_buf.len() {
|
if aligned_idx > data_buf.len() {
|
||||||
data_buf.resize(aligned_idx, 0u8);
|
data_buf.resize(aligned_idx, 0u8);
|
||||||
}
|
}
|
||||||
|
|
||||||
let layout_message = if use_extensible {
|
let layout_message = match &index {
|
||||||
let ea_address = base_address + data_buf.len() as u64;
|
ChunkIndexPlan::ExtensibleArray(grid) => {
|
||||||
let ea_bytes = ea_writer::build_extensible_array_at(
|
let ea_address = base_address + data_buf.len() as u64;
|
||||||
&written_chunks,
|
let slots = index_slots(grid, &pre.shape, &pre.chunk_dims, &written_chunks, None)?;
|
||||||
offset_size,
|
let ea_bytes = ea_writer::build_extensible_array_at(
|
||||||
length_size,
|
&slots,
|
||||||
pre.has_filters,
|
offset_size,
|
||||||
ea_address,
|
length_size,
|
||||||
);
|
pre.has_filters,
|
||||||
data_buf.extend_from_slice(&ea_bytes);
|
ea_address,
|
||||||
ea_writer::serialize_v4_extensible_array(
|
);
|
||||||
&chunk_dims_u32,
|
data_buf.extend_from_slice(&ea_bytes);
|
||||||
ea_address,
|
ea_writer::serialize_v4_extensible_array(
|
||||||
offset_size,
|
&chunk_dims_u32,
|
||||||
element_size as u32,
|
ea_address,
|
||||||
)
|
offset_size,
|
||||||
} else if num_chunks == 1 {
|
element_size as u32,
|
||||||
let chunk_addr = written_chunks[0].address;
|
)
|
||||||
let filtered_size = if pre.has_filters {
|
}
|
||||||
Some(written_chunks[0].compressed_size)
|
ChunkIndexPlan::SingleChunk => {
|
||||||
} else {
|
let chunk_addr = written_chunks[0].address;
|
||||||
None
|
let filtered_size = if pre.has_filters {
|
||||||
};
|
Some(written_chunks[0].compressed_size)
|
||||||
let filter_mask = if pre.has_filters { Some(0u32) } else { None };
|
} else {
|
||||||
serialize_v4_single_chunk(
|
None
|
||||||
&chunk_dims_u32,
|
};
|
||||||
chunk_addr,
|
let filter_mask = if pre.has_filters { Some(0u32) } else { None };
|
||||||
filtered_size,
|
serialize_v4_single_chunk(
|
||||||
filter_mask,
|
&chunk_dims_u32,
|
||||||
offset_size,
|
chunk_addr,
|
||||||
element_size as u32,
|
filtered_size,
|
||||||
)
|
filter_mask,
|
||||||
} else {
|
offset_size,
|
||||||
let fa_address = base_address + data_buf.len() as u64;
|
element_size as u32,
|
||||||
let fa_bytes = build_fixed_array_at(
|
)
|
||||||
&written_chunks,
|
}
|
||||||
offset_size,
|
ChunkIndexPlan::FixedArray(grid, nslots) => {
|
||||||
length_size,
|
let fa_address = base_address + data_buf.len() as u64;
|
||||||
pre.has_filters,
|
let slots = index_slots(
|
||||||
fa_address,
|
grid,
|
||||||
);
|
&pre.shape,
|
||||||
data_buf.extend_from_slice(&fa_bytes);
|
&pre.chunk_dims,
|
||||||
serialize_v4_fixed_array(
|
&written_chunks,
|
||||||
&chunk_dims_u32,
|
Some(*nslots),
|
||||||
fa_address,
|
)?;
|
||||||
offset_size,
|
let fa_bytes = build_fixed_array_at(
|
||||||
element_size as u32,
|
&slots,
|
||||||
10, // max_nelmts_bits — matches h5py convention
|
offset_size,
|
||||||
)
|
length_size,
|
||||||
|
pre.has_filters,
|
||||||
|
fa_address,
|
||||||
|
);
|
||||||
|
data_buf.extend_from_slice(&fa_bytes);
|
||||||
|
serialize_v4_fixed_array(
|
||||||
|
&chunk_dims_u32,
|
||||||
|
fa_address,
|
||||||
|
offset_size,
|
||||||
|
element_size as u32,
|
||||||
|
FA_PAGE_BITS,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
ChunkIndexPlan::BTreeV2 => {
|
||||||
|
let bt_address = base_address + data_buf.len() as u64;
|
||||||
|
let records: Vec<(Vec<u64>, &WrittenChunk)> = written_chunks
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, c)| (scaled_coords(&pre.shape, &pre.chunk_dims, i), c))
|
||||||
|
.collect();
|
||||||
|
let (bt_bytes, node_size) = build_btree_v2_chunk_index_at(
|
||||||
|
pre.shape.len(),
|
||||||
|
&records,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
pre.has_filters,
|
||||||
|
bt_address,
|
||||||
|
)?;
|
||||||
|
data_buf.extend_from_slice(&bt_bytes);
|
||||||
|
serialize_v4_btree_v2(
|
||||||
|
&chunk_dims_u32,
|
||||||
|
bt_address,
|
||||||
|
offset_size,
|
||||||
|
element_size as u32,
|
||||||
|
node_size,
|
||||||
|
)
|
||||||
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
ChunkedDataResult {
|
Ok(ChunkedDataResult {
|
||||||
data_bytes: data_buf,
|
data_bytes: data_buf,
|
||||||
layout_message,
|
layout_message,
|
||||||
pipeline_message: pre.pipeline_message.clone(),
|
pipeline_message: pre.pipeline_message.clone(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Most slots a Fixed Array index may have before we refuse to build it: its
|
||||||
|
/// data block holds one element per chunk of the *maximum* extent, so a huge
|
||||||
|
/// finite maxshape with small chunks would otherwise exhaust memory.
|
||||||
|
const MAX_FIXED_ARRAY_SLOTS: u64 = 1 << 26;
|
||||||
|
|
||||||
|
/// Which chunk index a dataset gets, following the library's choice in
|
||||||
|
/// `H5D__layout_set_latest_indexing`: version-2 B-tree for more than one
|
||||||
|
/// unlimited dimension, Extensible Array for exactly one, Fixed Array for a
|
||||||
|
/// finite maxshape, Single Chunk when the whole maximum extent is one chunk.
|
||||||
|
enum ChunkIndexPlan {
|
||||||
|
SingleChunk,
|
||||||
|
/// The grid and the number of array elements (chunks of the max extent).
|
||||||
|
FixedArray(ChunkGrid, usize),
|
||||||
|
ExtensibleArray(ChunkGrid),
|
||||||
|
BTreeV2,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ChunkIndexPlan {
|
||||||
|
fn new(
|
||||||
|
shape: &[u64],
|
||||||
|
maxshape: Option<&[u64]>,
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
) -> Result<Self, FormatError> {
|
||||||
|
let bad = |what: &str| FormatError::ChunkedReadError(format!("maxshape: {what}"));
|
||||||
|
if let Some(ms) = maxshape {
|
||||||
|
if ms.len() != shape.len() {
|
||||||
|
return Err(bad("rank differs from the shape"));
|
||||||
|
}
|
||||||
|
if ms.iter().zip(shape).any(|(&m, &s)| m < s) {
|
||||||
|
return Err(bad("smaller than the shape"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let max = maxshape.unwrap_or(shape);
|
||||||
|
let nunlim = max.iter().filter(|&&d| d == u64::MAX).count();
|
||||||
|
match nunlim {
|
||||||
|
0 => {
|
||||||
|
let nslots = max
|
||||||
|
.iter()
|
||||||
|
.zip(chunk_dims)
|
||||||
|
.try_fold(1u64, |acc, (&m, &c)| acc.checked_mul(m.div_ceil(c.max(1))))
|
||||||
|
.filter(|&n| n <= MAX_FIXED_ARRAY_SLOTS)
|
||||||
|
.ok_or_else(|| {
|
||||||
|
bad("too many chunks for a Fixed Array index; \
|
||||||
|
use larger chunks or an unlimited dimension")
|
||||||
|
})?;
|
||||||
|
// A Single Chunk index needs that one chunk to exist; an
|
||||||
|
// empty dataset gets an all-unallocated Fixed Array instead.
|
||||||
|
let empty = shape.contains(&0);
|
||||||
|
if nslots == 1 && !empty {
|
||||||
|
Ok(Self::SingleChunk)
|
||||||
|
} else {
|
||||||
|
let grid = ChunkGrid::fixed_array(shape, Some(max), chunk_dims)?;
|
||||||
|
Ok(Self::FixedArray(grid, nslots as usize))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
1 => Ok(Self::ExtensibleArray(ChunkGrid::extensible_array(
|
||||||
|
shape,
|
||||||
|
Some(max),
|
||||||
|
chunk_dims,
|
||||||
|
)?)),
|
||||||
|
_ => Ok(Self::BTreeV2),
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Place each written chunk at its linear index in `grid`. `chunks` are in
|
||||||
|
/// row-major order over the chunks of the current extent (`split_into_chunks`).
|
||||||
|
/// `len` fixes the slot count (Fixed Array); otherwise it is one past the
|
||||||
|
/// highest index used.
|
||||||
|
fn index_slots(
|
||||||
|
grid: &ChunkGrid,
|
||||||
|
shape: &[u64],
|
||||||
|
chunk_dims: &[u64],
|
||||||
|
chunks: &[WrittenChunk],
|
||||||
|
len: Option<usize>,
|
||||||
|
) -> Result<Vec<Option<WrittenChunk>>, FormatError> {
|
||||||
|
let mut placed: Vec<(usize, &WrittenChunk)> = Vec::with_capacity(chunks.len());
|
||||||
|
for (i, chunk) in chunks.iter().enumerate() {
|
||||||
|
let scaled = scaled_coords(shape, chunk_dims, i);
|
||||||
|
let idx = usize::try_from(grid.linear_index(&scaled))
|
||||||
|
.map_err(|_| FormatError::Overflow("chunk index slot".into()))?;
|
||||||
|
placed.push((idx, chunk));
|
||||||
|
}
|
||||||
|
let n = len.unwrap_or_else(|| placed.iter().map(|&(i, _)| i + 1).max().unwrap_or(0));
|
||||||
|
let mut slots = vec![None; n];
|
||||||
|
for (idx, chunk) in placed {
|
||||||
|
*slots
|
||||||
|
.get_mut(idx)
|
||||||
|
.ok_or_else(|| FormatError::Overflow("chunk index slot".into()))? = Some(chunk.clone());
|
||||||
|
}
|
||||||
|
Ok(slots)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Scaled coordinates (`offset / chunk_dim`) of the `i`-th chunk in the
|
||||||
|
/// row-major order `split_into_chunks` produces over the current extent.
|
||||||
|
fn scaled_coords(shape: &[u64], chunk_dims: &[u64], i: usize) -> Vec<u64> {
|
||||||
|
let rank = shape.len();
|
||||||
|
let mut scaled = vec![0u64; rank];
|
||||||
|
let mut rem = i as u64;
|
||||||
|
for d in (0..rank).rev() {
|
||||||
|
let n = shape[d].div_ceil(chunk_dims[d]);
|
||||||
|
scaled[d] = rem % n;
|
||||||
|
rem /= n;
|
||||||
|
}
|
||||||
|
scaled
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Node size the library gives a chunk index B-tree (`H5D_BT2_NODE_SIZE`),
|
||||||
|
/// with its split and merge percentages.
|
||||||
|
const BT2_NODE_SIZE: u32 = 2048;
|
||||||
|
const BT2_SPLIT_PERCENT: u8 = 100;
|
||||||
|
const BT2_MERGE_PERCENT: u8 = 40;
|
||||||
|
/// B-tree v2 record types for chunk indexes (`H5B2_CDSET_ID`,
|
||||||
|
/// `H5B2_CDSET_FILT_ID`).
|
||||||
|
const BT2_CHUNK_UNFILTERED: u8 = 10;
|
||||||
|
const BT2_CHUNK_FILTERED: u8 = 11;
|
||||||
|
|
||||||
|
/// Build a version-2 B-tree chunk index (the library's index for datasets
|
||||||
|
/// with more than one unlimited dimension) at a known absolute address.
|
||||||
|
///
|
||||||
|
/// `records` are `(scaled coordinates, chunk)` in lexicographic order of the
|
||||||
|
/// coordinates, which is the order the library's comparator
|
||||||
|
/// (`H5VM_vector_cmp_u`) keeps them in. The tree is a single leaf: the
|
||||||
|
/// library's 2048-byte node when the records fit, otherwise a leaf node
|
||||||
|
/// sized to hold them all (the root's record count is 16-bit, so at most
|
||||||
|
/// 65535 chunks). Returns the bytes and the node size the layout message
|
||||||
|
/// must record.
|
||||||
|
fn build_btree_v2_chunk_index_at(
|
||||||
|
rank: usize,
|
||||||
|
records: &[(Vec<u64>, &WrittenChunk)],
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
has_filters: bool,
|
||||||
|
base_address: u64,
|
||||||
|
) -> Result<(Vec<u8>, u32), FormatError> {
|
||||||
|
let os = offset_size as usize;
|
||||||
|
let nrec = u16::try_from(records.len()).map_err(|_| {
|
||||||
|
FormatError::ChunkedReadError(
|
||||||
|
"more than 65535 chunks with more than one unlimited dimension: \
|
||||||
|
use larger chunks"
|
||||||
|
.into(),
|
||||||
|
)
|
||||||
|
})?;
|
||||||
|
let chunk_size_bytes = has_filters.then(|| {
|
||||||
|
let slots: Vec<Option<WrittenChunk>> =
|
||||||
|
records.iter().map(|(_, c)| Some((*c).clone())).collect();
|
||||||
|
filtered_chunk_size_len(&slots)
|
||||||
|
});
|
||||||
|
let record_size = os + chunk_size_bytes.map_or(0, |n| n + 4) + 8 * rank;
|
||||||
|
// Leaf: signature, version, type, records, checksum.
|
||||||
|
let leaf_len = 4 + 1 + 1 + records.len() * record_size + 4;
|
||||||
|
let node_size = u32::try_from(leaf_len)
|
||||||
|
.map_err(|_| FormatError::Overflow("B-tree v2 leaf size".into()))?
|
||||||
|
.max(BT2_NODE_SIZE);
|
||||||
|
let tree_type = if has_filters {
|
||||||
|
BT2_CHUNK_FILTERED
|
||||||
|
} else {
|
||||||
|
BT2_CHUNK_UNFILTERED
|
||||||
|
};
|
||||||
|
|
||||||
|
let hdr_len = 4 + 1 + 1 + 4 + 2 + 2 + 1 + 1 + os + 2 + length_size as usize + 4;
|
||||||
|
let leaf_address = base_address + hdr_len as u64;
|
||||||
|
|
||||||
|
let mut out = Vec::with_capacity(hdr_len + node_size as usize);
|
||||||
|
out.extend_from_slice(b"BTHD");
|
||||||
|
out.push(0); // version
|
||||||
|
out.push(tree_type);
|
||||||
|
out.extend_from_slice(&node_size.to_le_bytes());
|
||||||
|
out.extend_from_slice(&(record_size as u16).to_le_bytes());
|
||||||
|
out.extend_from_slice(&0u16.to_le_bytes()); // depth
|
||||||
|
out.push(BT2_SPLIT_PERCENT);
|
||||||
|
out.push(BT2_MERGE_PERCENT);
|
||||||
|
if records.is_empty() {
|
||||||
|
out.extend(core::iter::repeat_n(0xFF, os));
|
||||||
|
} else {
|
||||||
|
push_addr(&mut out, leaf_address, offset_size);
|
||||||
|
}
|
||||||
|
out.extend_from_slice(&nrec.to_le_bytes());
|
||||||
|
match length_size {
|
||||||
|
4 => out.extend_from_slice(&(records.len() as u32).to_le_bytes()),
|
||||||
|
_ => out.extend_from_slice(&(records.len() as u64).to_le_bytes()),
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len(), hdr_len);
|
||||||
|
if records.is_empty() {
|
||||||
|
return Ok((out, node_size));
|
||||||
|
}
|
||||||
|
|
||||||
|
let leaf_start = out.len();
|
||||||
|
out.extend_from_slice(b"BTLF");
|
||||||
|
out.push(0); // version
|
||||||
|
out.push(tree_type);
|
||||||
|
for (scaled, chunk) in records {
|
||||||
|
push_index_element(&mut out, Some(chunk), offset_size, chunk_size_bytes);
|
||||||
|
for &c in scaled {
|
||||||
|
out.extend_from_slice(&c.to_le_bytes());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[leaf_start..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
// The library reads whole nodes; pad the leaf out to the node size.
|
||||||
|
out.resize(leaf_start + node_size as usize, 0);
|
||||||
|
Ok((out, node_size))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Serialize a v4 layout message for a version-2 B-tree chunk index.
|
||||||
|
fn serialize_v4_btree_v2(
|
||||||
|
chunk_dims: &[u32],
|
||||||
|
btree_address: u64,
|
||||||
|
offset_size: u8,
|
||||||
|
element_size: u32,
|
||||||
|
node_size: u32,
|
||||||
|
) -> Vec<u8> {
|
||||||
|
let mut buf = layout_v4_chunked_prefix(chunk_dims, element_size);
|
||||||
|
buf.push(5); // chunk index type = 5 (version-2 B-tree)
|
||||||
|
buf.extend_from_slice(&node_size.to_le_bytes());
|
||||||
|
buf.push(BT2_SPLIT_PERCENT);
|
||||||
|
buf.push(BT2_MERGE_PERCENT);
|
||||||
|
push_addr(&mut buf, btree_address, offset_size);
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
|
||||||
/// Build chunked data with absolute addresses.
|
/// Build chunked data with absolute addresses.
|
||||||
/// If `maxshape` has unlimited dims, uses Extensible Array index.
|
/// If `maxshape` has unlimited dims, uses Extensible Array index.
|
||||||
pub fn build_chunked_data_at(
|
pub fn build_chunked_data_at(
|
||||||
@@ -790,11 +1090,7 @@ pub fn build_chunked_data_at_ext(
|
|||||||
maxshape: Option<&[u64]>,
|
maxshape: Option<&[u64]>,
|
||||||
) -> Result<ChunkedDataResult, FormatError> {
|
) -> Result<ChunkedDataResult, FormatError> {
|
||||||
let pre = precompress_chunks(raw_data, shape, chunk_dims, element_size, options)?;
|
let pre = precompress_chunks(raw_data, shape, chunk_dims, element_size, options)?;
|
||||||
Ok(build_chunked_data_from_precompressed(
|
build_chunked_data_from_precompressed(&pre, base_address, maxshape)
|
||||||
&pre,
|
|
||||||
base_address,
|
|
||||||
maxshape,
|
|
||||||
))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Write selected elements into an existing in-memory dataset buffer.
|
/// Write selected elements into an existing in-memory dataset buffer.
|
||||||
@@ -1314,6 +1610,7 @@ mod tests {
|
|||||||
chunk_index_type,
|
chunk_index_type,
|
||||||
single_chunk_filtered_size,
|
single_chunk_filtered_size,
|
||||||
single_chunk_filter_mask,
|
single_chunk_filter_mask,
|
||||||
|
..
|
||||||
} => {
|
} => {
|
||||||
assert_eq!(version, 4);
|
assert_eq!(version, 4);
|
||||||
assert_eq!(chunk_index_type, Some(1));
|
assert_eq!(chunk_index_type, Some(1));
|
||||||
@@ -1382,7 +1679,8 @@ mod tests {
|
|||||||
filter_mask: 0,
|
filter_mask: 0,
|
||||||
},
|
},
|
||||||
];
|
];
|
||||||
let fa = build_fixed_array_at(&chunks, 8, 8, false, 0x2000);
|
let slots: Vec<_> = chunks.into_iter().map(Some).collect();
|
||||||
|
let fa = build_fixed_array_at(&slots, 8, 8, false, 0x2000);
|
||||||
// Should start with FAHD
|
// Should start with FAHD
|
||||||
assert_eq!(&fa[0..4], b"FAHD");
|
assert_eq!(&fa[0..4], b"FAHD");
|
||||||
// FAHD size = 4+1+1+1+1+8+8+4 = 28
|
// FAHD size = 4+1+1+1+1+8+8+4 = 28
|
||||||
@@ -1429,7 +1727,8 @@ mod tests {
|
|||||||
filter_mask: 0,
|
filter_mask: 0,
|
||||||
},
|
},
|
||||||
];
|
];
|
||||||
let ea = ea_writer::build_extensible_array_at(&chunks, 8, 8, false, 0x2000);
|
let slots: Vec<_> = chunks.into_iter().map(Some).collect();
|
||||||
|
let ea = ea_writer::build_extensible_array_at(&slots, 8, 8, false, 0x2000);
|
||||||
assert_eq!(&ea[0..4], b"EAHD");
|
assert_eq!(&ea[0..4], b"EAHD");
|
||||||
// Find EAIB after EAHD: 12 fixed + 6*8 stats + 8 addr + 4 checksum = 72
|
// Find EAIB after EAHD: 12 fixed + 6*8 stats + 8 addr + 4 checksum = 72
|
||||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * 8 + 8 + 4;
|
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * 8 + 8 + 4;
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
//! HDF5 Data Layout message parsing (message type 0x0008).
|
//! HDF5 Data Layout message parsing (message type 0x0008).
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{string::String, vec::Vec};
|
use alloc::{format, string::String, vec::Vec};
|
||||||
|
|
||||||
#[cfg(feature = "std")]
|
#[cfg(feature = "std")]
|
||||||
use std::string::String;
|
use std::string::String;
|
||||||
@@ -45,7 +45,9 @@ pub enum DataLayout {
|
|||||||
chunk_dimensions: Vec<u32>,
|
chunk_dimensions: Vec<u32>,
|
||||||
/// B-tree address, or `None` if undefined.
|
/// B-tree address, or `None` if undefined.
|
||||||
btree_address: Option<u64>,
|
btree_address: Option<u64>,
|
||||||
/// Layout version (3 or 4).
|
/// Layout version (3 or 4). Version 1/2 messages (HDF5 1.4/1.6-era)
|
||||||
|
/// use the same version-1 B-tree chunk index as version 3 and are
|
||||||
|
/// reported as 3.
|
||||||
version: u8,
|
version: u8,
|
||||||
/// Chunk index type (v4 only).
|
/// Chunk index type (v4 only).
|
||||||
chunk_index_type: Option<u8>,
|
chunk_index_type: Option<u8>,
|
||||||
@@ -53,6 +55,11 @@ pub enum DataLayout {
|
|||||||
single_chunk_filtered_size: Option<u64>,
|
single_chunk_filtered_size: Option<u64>,
|
||||||
/// Filter mask for v4 single chunk with filters.
|
/// Filter mask for v4 single chunk with filters.
|
||||||
single_chunk_filter_mask: Option<u32>,
|
single_chunk_filter_mask: Option<u32>,
|
||||||
|
/// Layout v4 flag bit 0 (`H5D_CHUNK_DONT_FILTER_PARTIAL_CHUNKS`):
|
||||||
|
/// partial edge chunks — those extending past the dataset's current
|
||||||
|
/// extent in some dimension — are stored without the filter pipeline,
|
||||||
|
/// even though their filter mask is 0. Always `false` for v3.
|
||||||
|
dont_filter_partial_edge_chunks: bool,
|
||||||
},
|
},
|
||||||
/// Virtual dataset layout (v4 only).
|
/// Virtual dataset layout (v4 only).
|
||||||
Virtual {
|
Virtual {
|
||||||
@@ -67,21 +74,33 @@ pub enum DataLayout {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Version-1 VDS mapping flag: the source file name is stored by an earlier
|
||||||
|
/// entry, whose index follows in place of the name.
|
||||||
|
const VDS_SOURCE_FILE_SHARED: u8 = 0x01;
|
||||||
|
/// Version-1 VDS mapping flag: likewise for the source dataset name.
|
||||||
|
const VDS_SOURCE_DSET_SHARED: u8 = 0x02;
|
||||||
|
/// Version-1 VDS mapping flag: the source is in the virtual file itself
|
||||||
|
/// (`"."`); no file name is stored.
|
||||||
|
const VDS_SOURCE_SAME_FILE: u8 = 0x04;
|
||||||
|
const VDS_ALL_FLAGS: u8 = VDS_SOURCE_FILE_SHARED | VDS_SOURCE_DSET_SHARED | VDS_SOURCE_SAME_FILE;
|
||||||
|
|
||||||
/// Parse VDS mappings from global-heap object data.
|
/// Parse VDS mappings from global-heap object data.
|
||||||
///
|
///
|
||||||
/// The global-heap block holding a VDS mapping list is laid out as
|
/// The global-heap block holding a VDS mapping list is laid out as
|
||||||
/// (reverse-engineered and validated against HDF5 2.0):
|
/// (`H5D__virtual_store_layout` / `H5D__virtual_load_layout` in libhdf5):
|
||||||
///
|
///
|
||||||
/// ```text
|
/// ```text
|
||||||
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
/// version(1) · nused(length_size, LE) · entry[nused] · checksum(4)
|
||||||
/// ```
|
/// ```
|
||||||
///
|
///
|
||||||
/// Each entry is:
|
/// Each entry is:
|
||||||
/// - source file name — a null-terminated string in **block version 0**; in
|
/// - **block version 1 only:** a flags byte. `0x04`: the source is in the
|
||||||
/// **block version 1** a same-file reference is encoded as a single `0x04`
|
/// virtual file itself and no file name is stored; `0x01`/`0x02`: the
|
||||||
/// marker byte (the source file is the virtual file itself) in place of the
|
/// source file/dataset name is that of an earlier entry, whose index
|
||||||
/// name;
|
/// (`length_size` bytes) is stored instead of the name. libhdf5 2.0 writes
|
||||||
/// - source dataset name (null-terminated string);
|
/// version 1 when the file's low version bound is 2.0 and it saves space;
|
||||||
|
/// - source file name (null-terminated string, unless flagged above);
|
||||||
|
/// - source dataset name (null-terminated string, unless flagged above);
|
||||||
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
/// - source selection (serialized `H5S` dataspace selection — self-describing
|
||||||
/// in length);
|
/// in length);
|
||||||
/// - virtual selection (serialized `H5S` dataspace selection).
|
/// - virtual selection (serialized `H5S` dataspace selection).
|
||||||
@@ -107,7 +126,7 @@ pub fn parse_vds_mappings(
|
|||||||
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
// `nused` is untrusted; don't pre-allocate from it. Each entry consumes at
|
||||||
// least a few bytes, so the loop is naturally bounded by the heap data and
|
// least a few bytes, so the loop is naturally bounded by the heap data and
|
||||||
// a bogus `nused` simply errors out on the first short read.
|
// a bogus `nused` simply errors out on the first short read.
|
||||||
let mut mappings = Vec::new();
|
let mut mappings: Vec<VdsMapping> = Vec::new();
|
||||||
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
// Reads one self-describing selection at `pos`, returning its raw bytes and
|
||||||
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
// advancing past it — bounds-checked so a corrupt selection can't overrun.
|
||||||
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
let read_selection = |heap_data: &[u8], pos: &mut usize| -> Result<Vec<u8>, FormatError> {
|
||||||
@@ -127,17 +146,57 @@ pub fn parse_vds_mappings(
|
|||||||
Ok(bytes)
|
Ok(bytes)
|
||||||
};
|
};
|
||||||
|
|
||||||
for _ in 0..nused {
|
if version > 1 {
|
||||||
// Source file name (with the version-1 same-file marker handled).
|
return Err(FormatError::ChunkedReadError(
|
||||||
let source_file = if version >= 1 && heap_data.get(pos) == Some(&0x04) {
|
"unsupported VDS mapping block version".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for i in 0..nused {
|
||||||
|
// Version 1 prefixes each entry with a flags byte; a name may then be
|
||||||
|
// omitted (same file) or replaced by the index of an earlier entry
|
||||||
|
// holding the same name (`H5D__virtual_load_layout`).
|
||||||
|
let flags = if version >= 1 {
|
||||||
|
let f = *heap_data.get(pos).ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: pos + 1,
|
||||||
|
available: heap_data.len(),
|
||||||
|
})?;
|
||||||
pos += 1;
|
pos += 1;
|
||||||
|
if f & !VDS_ALL_FLAGS != 0 {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"unknown VDS mapping flags".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
f
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
};
|
||||||
|
// Index of an earlier entry, for a shared name.
|
||||||
|
let earlier = |pos: &mut usize| -> Result<usize, FormatError> {
|
||||||
|
let idx = read_length(heap_data, *pos, length_size)?;
|
||||||
|
*pos += ls;
|
||||||
|
if idx >= i {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"VDS mapping shares a name with a later entry".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
Ok(idx as usize)
|
||||||
|
};
|
||||||
|
|
||||||
|
let source_file = if flags & VDS_SOURCE_SAME_FILE != 0 {
|
||||||
String::from(".")
|
String::from(".")
|
||||||
|
} else if flags & VDS_SOURCE_FILE_SHARED != 0 {
|
||||||
|
let idx = earlier(&mut pos)?;
|
||||||
|
mappings[idx].source_file.clone()
|
||||||
} else {
|
} else {
|
||||||
read_null_terminated_string(heap_data, &mut pos)?
|
read_null_terminated_string(heap_data, &mut pos)?
|
||||||
};
|
};
|
||||||
|
|
||||||
// Source dataset name.
|
let source_dataset = if flags & VDS_SOURCE_DSET_SHARED != 0 {
|
||||||
let source_dataset = read_null_terminated_string(heap_data, &mut pos)?;
|
let idx = earlier(&mut pos)?;
|
||||||
|
mappings[idx].source_dataset.clone()
|
||||||
|
} else {
|
||||||
|
read_null_terminated_string(heap_data, &mut pos)?
|
||||||
|
};
|
||||||
|
|
||||||
// Source selection, then virtual selection (both self-describing length).
|
// Source selection, then virtual selection (both self-describing length).
|
||||||
let source_selection = read_selection(heap_data, &mut pos)?;
|
let source_selection = read_selection(heap_data, &mut pos)?;
|
||||||
@@ -256,6 +315,7 @@ impl DataLayout {
|
|||||||
let layout_class = data[1];
|
let layout_class = data[1];
|
||||||
|
|
||||||
match version {
|
match version {
|
||||||
|
1 | 2 => Self::parse_v1_v2(data, offset_size),
|
||||||
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
3 => Self::parse_v3(data, layout_class, offset_size, length_size),
|
||||||
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
// v5 (emitted by HDF5 1.14+/2.0 with `libver=latest`) uses the same
|
||||||
// message structure as v4 — only the version number was bumped.
|
// message structure as v4 — only the version number was bumped.
|
||||||
@@ -264,6 +324,87 @@ impl DataLayout {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Layout message versions 1 and 2 (HDF5 before 1.6.3):
|
||||||
|
///
|
||||||
|
/// ```text
|
||||||
|
/// version(1) · dimensionality(1) · layout class(1) · reserved(5)
|
||||||
|
/// · address(offset_size) — contiguous and chunked only
|
||||||
|
/// · dimension sizes(4 × dimensionality)
|
||||||
|
/// · compact data size(4) · compact raw data — compact only
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// The dimension sizes are the dataset's (contiguous/compact) or the
|
||||||
|
/// chunk's (chunked) extent plus a trailing element-size dimension, as in
|
||||||
|
/// version 3's chunked form. libhdf5 ignores them for contiguous storage
|
||||||
|
/// and sizes the data from the dataspace; the product of the stored
|
||||||
|
/// dimensions is that same size, and a disagreement (a dimension that was
|
||||||
|
/// truncated to 32 bits) is caught by the reader's size check rather than
|
||||||
|
/// returning wrong data.
|
||||||
|
fn parse_v1_v2(data: &[u8], offset_size: u8) -> Result<DataLayout, FormatError> {
|
||||||
|
ensure_len(data, 0, 8)?;
|
||||||
|
let dimensionality = data[1] as usize;
|
||||||
|
let layout_class = data[2];
|
||||||
|
// H5O_LAYOUT_NDIMS: 32 dataspace dimensions + the element-size one.
|
||||||
|
if dimensionality > 33 {
|
||||||
|
return Err(FormatError::Overflow(format!(
|
||||||
|
"data layout dimensionality {dimensionality} exceeds 33"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
let mut p = 8;
|
||||||
|
let os = offset_size as usize;
|
||||||
|
let address = match layout_class {
|
||||||
|
1 | 2 => {
|
||||||
|
ensure_len(data, p, os)?;
|
||||||
|
let a = if is_undefined(data, p, offset_size) {
|
||||||
|
None
|
||||||
|
} else {
|
||||||
|
Some(read_offset(data, p, offset_size)?)
|
||||||
|
};
|
||||||
|
p += os;
|
||||||
|
a
|
||||||
|
}
|
||||||
|
0 => None,
|
||||||
|
_ => return Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||||
|
};
|
||||||
|
ensure_len(data, p, dimensionality * 4)?;
|
||||||
|
let dims: Vec<u32> = data[p..p + dimensionality * 4]
|
||||||
|
.as_chunks::<4>()
|
||||||
|
.0
|
||||||
|
.iter()
|
||||||
|
.map(|c| u32::from_le_bytes(*c))
|
||||||
|
.collect();
|
||||||
|
p += dimensionality * 4;
|
||||||
|
match layout_class {
|
||||||
|
0 => {
|
||||||
|
ensure_len(data, p, 4)?;
|
||||||
|
let size =
|
||||||
|
u32::from_le_bytes([data[p], data[p + 1], data[p + 2], data[p + 3]]) as usize;
|
||||||
|
ensure_len(data, p + 4, size)?;
|
||||||
|
Ok(DataLayout::Compact {
|
||||||
|
data: data[p + 4..p + 4 + size].to_vec(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
1 => {
|
||||||
|
let size = dims
|
||||||
|
.iter()
|
||||||
|
.try_fold(1u64, |acc, &d| acc.checked_mul(d as u64))
|
||||||
|
.ok_or_else(|| {
|
||||||
|
FormatError::Overflow(format!("contiguous layout size {dims:?}"))
|
||||||
|
})?;
|
||||||
|
Ok(DataLayout::Contiguous { address, size })
|
||||||
|
}
|
||||||
|
_ => Ok(DataLayout::Chunked {
|
||||||
|
chunk_dimensions: dims,
|
||||||
|
btree_address: address,
|
||||||
|
version: 3,
|
||||||
|
chunk_index_type: None,
|
||||||
|
single_chunk_filtered_size: None,
|
||||||
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn parse_v3(
|
fn parse_v3(
|
||||||
data: &[u8],
|
data: &[u8],
|
||||||
layout_class: u8,
|
layout_class: u8,
|
||||||
@@ -322,6 +463,7 @@ impl DataLayout {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
_ => Err(FormatError::InvalidLayoutClass(layout_class)),
|
||||||
@@ -505,6 +647,7 @@ impl DataLayout {
|
|||||||
chunk_index_type: Some(chunk_index_type),
|
chunk_index_type: Some(chunk_index_type),
|
||||||
single_chunk_filtered_size,
|
single_chunk_filtered_size,
|
||||||
single_chunk_filter_mask,
|
single_chunk_filter_mask,
|
||||||
|
dont_filter_partial_edge_chunks: flags & 0x01 != 0,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
3 => {
|
3 => {
|
||||||
@@ -539,6 +682,108 @@ impl DataLayout {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Version 1/2 header: version, dimensionality, class, reserved(5).
|
||||||
|
fn v1v2_header(version: u8, ndims: u8, class: u8) -> Vec<u8> {
|
||||||
|
vec![version, ndims, class, 0, 0, 0, 0, 0]
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v2_compact() {
|
||||||
|
let mut buf = v1v2_header(2, 2, 0);
|
||||||
|
// dims (3 elements of 2 bytes) — no address for compact
|
||||||
|
buf.extend_from_slice(&3u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&2u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&6u32.to_le_bytes()); // compact size (u32 in v1/v2)
|
||||||
|
buf.extend_from_slice(&[1, 0, 2, 0, 3, 0]);
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Compact {
|
||||||
|
data: vec![1, 0, 2, 0, 3, 0]
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_contiguous_size_from_dimensions() {
|
||||||
|
let mut buf = v1v2_header(1, 3, 1);
|
||||||
|
buf.extend_from_slice(&0x800u32.to_le_bytes()); // 4-byte address
|
||||||
|
for d in [10u32, 20, 4] {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 4, 4).unwrap(),
|
||||||
|
DataLayout::Contiguous {
|
||||||
|
address: Some(0x800),
|
||||||
|
size: 800,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_contiguous_undefined_address() {
|
||||||
|
let mut buf = v1v2_header(1, 2, 1);
|
||||||
|
buf.extend_from_slice(&[0xFF; 8]);
|
||||||
|
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&8u32.to_le_bytes());
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Contiguous {
|
||||||
|
address: None,
|
||||||
|
size: 40,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1_chunked_maps_to_btree_v1_index() {
|
||||||
|
let mut buf = v1v2_header(1, 3, 2);
|
||||||
|
buf.extend_from_slice(&0x1234u64.to_le_bytes());
|
||||||
|
for d in [50u32, 50, 4] {
|
||||||
|
buf.extend_from_slice(&d.to_le_bytes());
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap(),
|
||||||
|
DataLayout::Chunked {
|
||||||
|
chunk_dimensions: vec![50, 50, 4],
|
||||||
|
btree_address: Some(0x1234),
|
||||||
|
version: 3,
|
||||||
|
chunk_index_type: None,
|
||||||
|
single_chunk_filtered_size: None,
|
||||||
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
|
}
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v1v2_rejects_bad_class_dimensionality_and_truncation() {
|
||||||
|
assert_eq!(
|
||||||
|
DataLayout::parse(&v1v2_header(1, 1, 3), 8, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidLayoutClass(3)
|
||||||
|
);
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&v1v2_header(2, 34, 1), 8, 8).unwrap_err(),
|
||||||
|
FormatError::Overflow(_)
|
||||||
|
));
|
||||||
|
// Chunked, dims cut short.
|
||||||
|
let mut buf = v1v2_header(1, 2, 2);
|
||||||
|
buf.extend_from_slice(&0x10u64.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&7u32.to_le_bytes());
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnexpectedEof { .. }
|
||||||
|
));
|
||||||
|
// Compact, raw data shorter than its declared size.
|
||||||
|
let mut buf = v1v2_header(2, 1, 0);
|
||||||
|
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&100u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&[0; 4]);
|
||||||
|
assert!(matches!(
|
||||||
|
DataLayout::parse(&buf, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnexpectedEof { .. }
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v3_compact() {
|
fn v3_compact() {
|
||||||
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
let mut buf = vec![3u8, 0]; // version=3, class=0 (compact)
|
||||||
@@ -602,6 +847,7 @@ mod tests {
|
|||||||
chunk_index_type: None,
|
chunk_index_type: None,
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -679,10 +925,35 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: None,
|
single_chunk_filtered_size: None,
|
||||||
single_chunk_filter_mask: None,
|
single_chunk_filter_mask: None,
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_chunked_dont_filter_partial_edge_chunks_flag() {
|
||||||
|
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||||
|
buf.push(0x01); // flags bit 0 = don't filter partial edge chunks
|
||||||
|
buf.push(2); // dimensionality=2
|
||||||
|
buf.push(4); // dim_size_encoded_length=4
|
||||||
|
buf.extend_from_slice(&5u32.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&4u32.to_le_bytes());
|
||||||
|
buf.push(3); // Fixed Array
|
||||||
|
buf.push(10); // max_dblk_page_nelmts_bits
|
||||||
|
buf.extend_from_slice(&0x3000u64.to_le_bytes());
|
||||||
|
match DataLayout::parse(&buf, 8, 8).unwrap() {
|
||||||
|
DataLayout::Chunked {
|
||||||
|
dont_filter_partial_edge_chunks,
|
||||||
|
btree_address,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
assert!(dont_filter_partial_edge_chunks);
|
||||||
|
assert_eq!(btree_address, Some(0x3000));
|
||||||
|
}
|
||||||
|
other => panic!("expected Chunked, got {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v4_chunked_single_chunk_with_filters() {
|
fn v4_chunked_single_chunk_with_filters() {
|
||||||
let mut buf = vec![4u8, 2]; // version=4, class=2
|
let mut buf = vec![4u8, 2]; // version=4, class=2
|
||||||
@@ -705,6 +976,7 @@ mod tests {
|
|||||||
chunk_index_type: Some(1),
|
chunk_index_type: Some(1),
|
||||||
single_chunk_filtered_size: Some(1024),
|
single_chunk_filtered_size: Some(1024),
|
||||||
single_chunk_filter_mask: Some(0),
|
single_chunk_filter_mask: Some(0),
|
||||||
|
dont_filter_partial_edge_chunks: false,
|
||||||
}
|
}
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
@@ -815,6 +1087,62 @@ mod tests {
|
|||||||
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
assert_eq!(v1.iter_linear_1d(8).unwrap(), vec![4, 5, 6, 7]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_vds_mappings_v1_shared_names() {
|
||||||
|
// Written by HDF5 2.0 (h5py, libver=("v200", "v200")) for three
|
||||||
|
// mappings from `a_rather_long_source_file.h5:a_rather_long_dataset_name`
|
||||||
|
// and one from the same file: the entries carry flags 0x00, 0x03, 0x03
|
||||||
|
// and 0x06, so names after the first are stored as entry indices.
|
||||||
|
let blob: &[u8] = &[
|
||||||
|
0x01, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x61, 0x5f, 0x72, 0x61,
|
||||||
|
0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x73, 0x6f, 0x75, 0x72,
|
||||||
|
0x63, 0x65, 0x5f, 0x66, 0x69, 0x6c, 0x65, 0x2e, 0x68, 0x35, 0x00, 0x61, 0x5f, 0x72,
|
||||||
|
0x61, 0x74, 0x68, 0x65, 0x72, 0x5f, 0x6c, 0x6f, 0x6e, 0x67, 0x5f, 0x64, 0x61, 0x74,
|
||||||
|
0x61, 0x73, 0x65, 0x74, 0x5f, 0x6e, 0x61, 0x6d, 0x65, 0x00, 0x02, 0x00, 0x00, 0x00,
|
||||||
|
0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02,
|
||||||
|
0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00,
|
||||||
|
0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x04, 0x00, 0x01, 0x00, 0x01,
|
||||||
|
0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01,
|
||||||
|
0x00, 0x01, 0x00, 0x04, 0x00, 0x03, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00,
|
||||||
|
0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x08, 0x00, 0x01, 0x00, 0x01, 0x00,
|
||||||
|
0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x02, 0x00,
|
||||||
|
0x00, 0x00, 0x02, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x01, 0x00, 0x04, 0x00, 0x06, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x02,
|
||||||
|
0x00, 0x00, 0x00, 0x03, 0x00, 0x00, 0x00, 0x01, 0x02, 0x01, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x00,
|
||||||
|
0x00, 0x01, 0x02, 0x02, 0x00, 0x00, 0x00, 0x03, 0x00, 0x01, 0x00, 0x01, 0x00, 0x01,
|
||||||
|
0x00, 0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0x04, 0x00, 0x8e, 0xa7, 0xea, 0x7a,
|
||||||
|
];
|
||||||
|
let mappings = parse_vds_mappings(blob, 8).unwrap();
|
||||||
|
let names: Vec<(&str, &str)> = mappings
|
||||||
|
.iter()
|
||||||
|
.map(|m| (m.source_file.as_str(), m.source_dataset.as_str()))
|
||||||
|
.collect();
|
||||||
|
let (file, dset) = ("a_rather_long_source_file.h5", "a_rather_long_dataset_name");
|
||||||
|
assert_eq!(
|
||||||
|
names,
|
||||||
|
vec![(file, dset), (file, dset), (file, dset), (".", dset)]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_vds_mappings_v1_forward_reference_is_error() {
|
||||||
|
// Entry 0 claiming to share entry 0's file name must not index past
|
||||||
|
// the entries decoded so far.
|
||||||
|
let mut blob = vec![0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x01];
|
||||||
|
blob.extend_from_slice(&[0u8; 8]);
|
||||||
|
blob.extend_from_slice(b"d\0");
|
||||||
|
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||||
|
// Unknown flag bits are refused.
|
||||||
|
let blob = [0x01u8, 1, 0, 0, 0, 0, 0, 0, 0, 0x08, b'd', 0];
|
||||||
|
assert!(parse_vds_mappings(&blob, 8).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_vds_mappings_external_v0() {
|
fn parse_vds_mappings_external_v0() {
|
||||||
// Block version 0 with an explicit (external) source file name.
|
// Block version 0 with an explicit (external) source file name.
|
||||||
|
|||||||
@@ -191,14 +191,9 @@ fn read_raw_data_full_impl(
|
|||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
),
|
),
|
||||||
DataLayout::Virtual {
|
DataLayout::Virtual { .. } => read_virtual_data(
|
||||||
global_heap_address,
|
|
||||||
global_heap_index,
|
|
||||||
..
|
|
||||||
} => read_virtual_data(
|
|
||||||
file_data,
|
file_data,
|
||||||
*global_heap_address,
|
layout,
|
||||||
*global_heap_index,
|
|
||||||
dataspace,
|
dataspace,
|
||||||
datatype,
|
datatype,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -465,158 +460,54 @@ pub fn read_raw_data_selection(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Assemble a **Virtual Dataset (VDS)** from its source mappings.
|
/// Assemble a **Virtual Dataset (VDS)** through the raw-read API, which has no
|
||||||
|
/// access to the dataset's fill value message.
|
||||||
///
|
///
|
||||||
/// Supports virtual datasets of any rank. Same-file sources are read directly;
|
/// Delegates to [`crate::vds::read_virtual_dataset`]. Because the fill value
|
||||||
/// **external-file** sources are read through the caller-supplied `resolver`,
|
/// is unknown here, a virtual dataset with any element no mapping supplies
|
||||||
/// which maps a stored source file name to that file's bytes. Each mapping's
|
/// (an unmapped region, or a missing source file or dataset) is an error
|
||||||
/// selected source elements are scattered into the virtual buffer at the
|
/// rather than a guess at the fill value; so is one whose extent libhdf5
|
||||||
/// positions given by the virtual selection (both enumerated in row-major
|
/// would report differently from the stored dataspace (unlimited mappings).
|
||||||
/// order, as HDF5 pairs them). Unmapped regions are left at the zero fill value.
|
/// Use [`crate::vds::read_virtual_dataset`] to read those.
|
||||||
///
|
|
||||||
/// A mapping whose external source file the resolver cannot supply (`None`) is
|
|
||||||
/// skipped, leaving its region at fill — matching HDF5's tolerance of missing
|
|
||||||
/// sources. An external source with no resolver at all is a hard error.
|
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn read_virtual_data(
|
fn read_virtual_data(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
global_heap_address: Option<u64>,
|
layout: &DataLayout,
|
||||||
global_heap_index: u32,
|
|
||||||
dataspace: &Dataspace,
|
dataspace: &Dataspace,
|
||||||
datatype: &Datatype,
|
datatype: &Datatype,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
resolver: Option<&VdsSourceResolver>,
|
resolver: Option<&VdsSourceResolver>,
|
||||||
) -> Result<Vec<u8>, FormatError> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
use crate::data_layout::parse_vds_mappings;
|
let wrapped =
|
||||||
use crate::global_heap::GlobalHeapCollection;
|
resolver.map(|r| move |name: &str| -> Result<Option<Vec<u8>>, FormatError> { Ok(r(name)) });
|
||||||
use crate::selection::Selection;
|
let wrapped_ref = wrapped.as_ref().map(|w| w as &crate::vds::VdsFileResolver);
|
||||||
|
let v = crate::vds::read_virtual_dataset(
|
||||||
let elem_size = datatype.type_size() as usize;
|
file_data,
|
||||||
let mut out = crate::chunked_read::alloc_output(crate::chunked_read::checked_byte_len(
|
layout,
|
||||||
dataspace.checked_num_elements()?,
|
dataspace,
|
||||||
elem_size,
|
datatype,
|
||||||
)?)?;
|
None,
|
||||||
|
offset_size,
|
||||||
let virtual_dims = &dataspace.dimensions;
|
length_size,
|
||||||
|
wrapped_ref,
|
||||||
let addr = global_heap_address.ok_or_else(|| {
|
)?;
|
||||||
FormatError::ChunkedReadError("virtual dataset has no mapping global heap".into())
|
if v.dims != dataspace.dimensions {
|
||||||
})?;
|
|
||||||
let coll = GlobalHeapCollection::parse(file_data, addr as usize, length_size)?;
|
|
||||||
let obj =
|
|
||||||
coll.get_object(global_heap_index as u16)
|
|
||||||
.ok_or(FormatError::GlobalHeapObjectNotFound {
|
|
||||||
collection_address: addr,
|
|
||||||
index: global_heap_index as u16,
|
|
||||||
})?;
|
|
||||||
let mappings = parse_vds_mappings(&obj.data, length_size)?;
|
|
||||||
|
|
||||||
for m in &mappings {
|
|
||||||
let same_file = m.source_file.is_empty() || m.source_file == ".";
|
|
||||||
|
|
||||||
// Resolve the bytes of the file holding this source dataset.
|
|
||||||
let external;
|
|
||||||
let src_file_data: &[u8] = if same_file {
|
|
||||||
file_data
|
|
||||||
} else {
|
|
||||||
let r = resolver.ok_or_else(|| {
|
|
||||||
FormatError::ChunkedReadError(
|
|
||||||
"external-file virtual dataset sources require a file resolver".into(),
|
|
||||||
)
|
|
||||||
})?;
|
|
||||||
match r(&m.source_file) {
|
|
||||||
Some(bytes) => {
|
|
||||||
external = bytes;
|
|
||||||
&external
|
|
||||||
}
|
|
||||||
// Source file unavailable: leave this region at fill value.
|
|
||||||
None => continue,
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
let (vsel, _) = Selection::decode_serialized(&m.virtual_selection)?;
|
|
||||||
let (ssel, _) = Selection::decode_serialized(&m.source_selection)?;
|
|
||||||
|
|
||||||
let (src_raw, src_dims) =
|
|
||||||
read_named_dataset_raw(src_file_data, &m.source_dataset, offset_size, length_size)?;
|
|
||||||
|
|
||||||
let vidx = vsel.iter_linear(virtual_dims)?;
|
|
||||||
let sidx = ssel.iter_linear(&src_dims)?;
|
|
||||||
if vidx.len() != sidx.len() {
|
|
||||||
return Err(FormatError::ChunkedReadError(
|
|
||||||
"virtual/source selection element counts differ".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
|
|
||||||
for (&v, &s) in vidx.iter().zip(sidx.iter()) {
|
|
||||||
let (vo, so) = (v as usize * elem_size, s as usize * elem_size);
|
|
||||||
if vo + elem_size > out.len() || so + elem_size > src_raw.len() {
|
|
||||||
return Err(FormatError::ChunkedReadError(
|
|
||||||
"virtual dataset selection out of bounds".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
out[vo..vo + elem_size].copy_from_slice(&src_raw[so..so + elem_size]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(out)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read a named dataset's raw (decoded) bytes and its dimensions, navigating
|
|
||||||
/// from the superblock. Used to pull VDS source datasets out of the same file.
|
|
||||||
fn read_named_dataset_raw(
|
|
||||||
file_data: &[u8],
|
|
||||||
path: &str,
|
|
||||||
_offset_size: u8,
|
|
||||||
_length_size: u8,
|
|
||||||
) -> Result<(Vec<u8>, Vec<u64>), FormatError> {
|
|
||||||
use crate::filter_pipeline::FilterPipeline;
|
|
||||||
use crate::group_v2::resolve_path_any;
|
|
||||||
use crate::message_type::MessageType;
|
|
||||||
use crate::object_header::ObjectHeader;
|
|
||||||
use crate::signature::find_signature;
|
|
||||||
use crate::superblock::Superblock;
|
|
||||||
|
|
||||||
let sig = find_signature(file_data)?;
|
|
||||||
let sb = Superblock::parse(file_data, sig)?;
|
|
||||||
let addr = resolve_path_any(file_data, &sb, path)?;
|
|
||||||
let hdr = ObjectHeader::parse(file_data, addr as usize, sb.offset_size, sb.length_size)?;
|
|
||||||
|
|
||||||
let find = |t: MessageType| hdr.messages.iter().find(|m| m.msg_type == t);
|
|
||||||
let ds_msg = find(MessageType::Dataspace)
|
|
||||||
.ok_or_else(|| FormatError::ChunkedReadError("VDS source has no dataspace".into()))?;
|
|
||||||
let dataspace = Dataspace::parse(&ds_msg.data, sb.length_size)?;
|
|
||||||
let dt_msg = find(MessageType::Datatype)
|
|
||||||
.ok_or_else(|| FormatError::ChunkedReadError("VDS source has no datatype".into()))?;
|
|
||||||
let (datatype, _) = Datatype::parse(&dt_msg.data)?;
|
|
||||||
let dl_msg = find(MessageType::DataLayout)
|
|
||||||
.ok_or_else(|| FormatError::ChunkedReadError("VDS source has no data layout".into()))?;
|
|
||||||
let layout = DataLayout::parse(&dl_msg.data, sb.offset_size, sb.length_size)?;
|
|
||||||
// A virtual dataset whose source is itself another virtual dataset could
|
|
||||||
// form a cycle (A -> B -> A) and recurse into a stack overflow. Nested
|
|
||||||
// virtual sources are exotic and unsupported, so stop here cleanly.
|
|
||||||
if matches!(layout, DataLayout::Virtual { .. }) {
|
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"virtual dataset source is itself virtual (unsupported)".into(),
|
"virtual dataset extent differs from its stored dataspace; \
|
||||||
|
read it with vds::read_virtual_dataset"
|
||||||
|
.into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
let pipeline = find(MessageType::FilterPipeline)
|
if v.unmapped > 0 {
|
||||||
.map(|m| FilterPipeline::parse(&m.data))
|
return Err(FormatError::ChunkedReadError(
|
||||||
.transpose()?;
|
"virtual dataset has elements no source supplies, which read as its \
|
||||||
|
fill value; read it with vds::read_virtual_dataset and the fill value"
|
||||||
let raw = read_raw_data_full(
|
.into(),
|
||||||
file_data,
|
));
|
||||||
&layout,
|
}
|
||||||
&dataspace,
|
Ok(v.data)
|
||||||
&datatype,
|
|
||||||
pipeline.as_ref(),
|
|
||||||
sb.offset_size,
|
|
||||||
sb.length_size,
|
|
||||||
)?;
|
|
||||||
Ok((raw, dataspace.dimensions.clone()))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract selected elements from a full dataset buffer.
|
/// Extract selected elements from a full dataset buffer.
|
||||||
pub fn extract_selection_from_buffer(
|
pub fn extract_selection_from_buffer(
|
||||||
full_data: &[u8],
|
full_data: &[u8],
|
||||||
@@ -773,14 +664,7 @@ pub fn read_as_f64_zerocopy<'a>(raw: &'a [u8], datatype: &Datatype) -> Option<&'
|
|||||||
// Only native LE f64 is eligible
|
// Only native LE f64 is eligible
|
||||||
#[cfg(target_endian = "little")]
|
#[cfg(target_endian = "little")]
|
||||||
{
|
{
|
||||||
if !matches!(
|
if !is_native_le_float(datatype, FloatFormat::Double) {
|
||||||
datatype,
|
|
||||||
Datatype::FloatingPoint {
|
|
||||||
size: 8,
|
|
||||||
byte_order: DatatypeByteOrder::LittleEndian,
|
|
||||||
..
|
|
||||||
}
|
|
||||||
) {
|
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
if !raw.len().is_multiple_of(8) {
|
if !raw.len().is_multiple_of(8) {
|
||||||
@@ -809,14 +693,7 @@ pub fn read_as_f64_zerocopy<'a>(raw: &'a [u8], datatype: &Datatype) -> Option<&'
|
|||||||
pub fn read_as_f32_zerocopy<'a>(raw: &'a [u8], datatype: &Datatype) -> Option<&'a [f32]> {
|
pub fn read_as_f32_zerocopy<'a>(raw: &'a [u8], datatype: &Datatype) -> Option<&'a [f32]> {
|
||||||
#[cfg(target_endian = "little")]
|
#[cfg(target_endian = "little")]
|
||||||
{
|
{
|
||||||
if !matches!(
|
if !is_native_le_float(datatype, FloatFormat::Single) {
|
||||||
datatype,
|
|
||||||
Datatype::FloatingPoint {
|
|
||||||
size: 4,
|
|
||||||
byte_order: DatatypeByteOrder::LittleEndian,
|
|
||||||
..
|
|
||||||
}
|
|
||||||
) {
|
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
if !raw.len().is_multiple_of(4) {
|
if !raw.len().is_multiple_of(4) {
|
||||||
@@ -902,9 +779,9 @@ fn native_le_to_vec<T: Copy>(raw: &[u8], count: usize) -> Vec<T> {
|
|||||||
|
|
||||||
/// Convert raw bytes to `f64` values.
|
/// Convert raw bytes to `f64` values.
|
||||||
pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatError> {
|
pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatError> {
|
||||||
// Array datatypes (e.g. an array-typed compound member) are read as a flat
|
// Array datatypes read as a flat sequence of their base elements, and
|
||||||
// sequence of their base elements.
|
// enumerations (h5py's bool among them) as their integer values.
|
||||||
if let Datatype::Array { base_type, .. } = datatype {
|
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||||
return read_as_f64(raw, base_type);
|
return read_as_f64(raw, base_type);
|
||||||
}
|
}
|
||||||
ensure_numeric(datatype, "FloatingPoint or FixedPoint")?;
|
ensure_numeric(datatype, "FloatingPoint or FixedPoint")?;
|
||||||
@@ -919,20 +796,19 @@ pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatEr
|
|||||||
|
|
||||||
// Fast path: native-endian f64 — single bulk memcpy
|
// Fast path: native-endian f64 — single bulk memcpy
|
||||||
#[cfg(target_endian = "little")]
|
#[cfg(target_endian = "little")]
|
||||||
if matches!(
|
if is_native_le_float(datatype, FloatFormat::Double) {
|
||||||
datatype,
|
|
||||||
Datatype::FloatingPoint {
|
|
||||||
size: 8,
|
|
||||||
byte_order: DatatypeByteOrder::LittleEndian,
|
|
||||||
..
|
|
||||||
}
|
|
||||||
) {
|
|
||||||
return Ok(native_le_to_vec::<f64>(raw, count));
|
return Ok(native_le_to_vec::<f64>(raw, count));
|
||||||
}
|
}
|
||||||
|
|
||||||
let order = get_byte_order(datatype);
|
let order = get_byte_order(datatype);
|
||||||
let mut result = Vec::with_capacity(count);
|
let mut result = Vec::with_capacity(count);
|
||||||
|
if let Datatype::FloatingPoint { .. } = datatype {
|
||||||
|
let format = FloatFormat::of(datatype)?;
|
||||||
|
for chunk in raw.chunks_exact(elem_size) {
|
||||||
|
result.push(format.decode(chunk, &order));
|
||||||
|
}
|
||||||
|
return Ok(result);
|
||||||
|
}
|
||||||
for i in 0..count {
|
for i in 0..count {
|
||||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||||
let val = convert_to_f64(chunk, datatype, &order)?;
|
let val = convert_to_f64(chunk, datatype, &order)?;
|
||||||
@@ -947,18 +823,7 @@ fn convert_to_f64(
|
|||||||
order: &DatatypeByteOrder,
|
order: &DatatypeByteOrder,
|
||||||
) -> Result<f64, FormatError> {
|
) -> Result<f64, FormatError> {
|
||||||
match dt {
|
match dt {
|
||||||
Datatype::FloatingPoint { size, .. } => match size {
|
Datatype::FloatingPoint { .. } => Ok(FloatFormat::of(dt)?.decode(bytes, order)),
|
||||||
4 => {
|
|
||||||
let v = read_f32_bytes(bytes, order);
|
|
||||||
Ok(v as f64)
|
|
||||||
}
|
|
||||||
8 => Ok(read_f64_bytes(bytes, order)),
|
|
||||||
2 => Ok(read_f16_bytes(bytes, order) as f64),
|
|
||||||
_ => Err(FormatError::DataSizeMismatch {
|
|
||||||
expected: 8,
|
|
||||||
actual: *size as usize,
|
|
||||||
}),
|
|
||||||
},
|
|
||||||
Datatype::FixedPoint {
|
Datatype::FixedPoint {
|
||||||
size,
|
size,
|
||||||
signed,
|
signed,
|
||||||
@@ -982,9 +847,83 @@ fn convert_to_f64(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// One numeric element as stored, before conversion to the caller's type.
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||||
|
enum Scalar {
|
||||||
|
Signed(i64),
|
||||||
|
Unsigned(u64),
|
||||||
|
Float(f64),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Scalar {
|
||||||
|
// Every conversion follows libhdf5's default (hard) conversions: a value
|
||||||
|
// outside the target type's range saturates to its minimum or maximum —
|
||||||
|
// including a negative value read as unsigned, which reads as 0 — rather
|
||||||
|
// than being truncated to its low bits. Floats truncate toward zero; NaN
|
||||||
|
// converts to 0 (libhdf5 leaves that case to the C cast, whose result is
|
||||||
|
// platform-dependent).
|
||||||
|
|
||||||
|
fn to_i64(self) -> i64 {
|
||||||
|
match self {
|
||||||
|
Scalar::Signed(v) => v,
|
||||||
|
Scalar::Unsigned(v) => i64::try_from(v).unwrap_or(i64::MAX),
|
||||||
|
Scalar::Float(v) => v as i64,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn to_u64(self) -> u64 {
|
||||||
|
match self {
|
||||||
|
Scalar::Signed(v) => u64::try_from(v).unwrap_or(0),
|
||||||
|
Scalar::Unsigned(v) => v,
|
||||||
|
Scalar::Float(v) => v as u64,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn to_i32(self) -> i32 {
|
||||||
|
match self {
|
||||||
|
Scalar::Signed(v) => v.clamp(i32::MIN.into(), i32::MAX.into()) as i32,
|
||||||
|
Scalar::Unsigned(v) => i32::try_from(v).unwrap_or(i32::MAX),
|
||||||
|
Scalar::Float(v) => v as i32,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decode one element of a numeric datatype.
|
||||||
|
fn decode_scalar(
|
||||||
|
bytes: &[u8],
|
||||||
|
dt: &Datatype,
|
||||||
|
order: &DatatypeByteOrder,
|
||||||
|
) -> Result<Scalar, FormatError> {
|
||||||
|
match dt {
|
||||||
|
Datatype::FixedPoint {
|
||||||
|
size,
|
||||||
|
signed,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
..
|
||||||
|
} => {
|
||||||
|
let full = read_unsigned_int(bytes, *size as usize, order);
|
||||||
|
let (off, prec) = effective_bits(*size as usize, *bit_offset, *bit_precision);
|
||||||
|
Ok(if *signed {
|
||||||
|
Scalar::Signed(extract_signed(full, off, prec))
|
||||||
|
} else {
|
||||||
|
Scalar::Unsigned(extract_unsigned(full, off, prec))
|
||||||
|
})
|
||||||
|
}
|
||||||
|
_ => convert_to_f64(bytes, dt, order).map(Scalar::Float),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Convert raw bytes to `i64` values.
|
/// Convert raw bytes to `i64` values.
|
||||||
|
///
|
||||||
|
/// Values are converted the way libhdf5 converts them: integers outside the
|
||||||
|
/// target range saturate at its minimum or maximum (a negative value read as
|
||||||
|
/// unsigned is 0), and floating-point data is truncated toward zero and
|
||||||
|
/// saturated, with NaN read as 0.
|
||||||
pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatError> {
|
pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatError> {
|
||||||
if let Datatype::Array { base_type, .. } = datatype {
|
// Array datatypes read as a flat sequence of their base elements, and
|
||||||
|
// enumerations (h5py's bool among them) as their integer values.
|
||||||
|
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||||
return read_as_i64(raw, base_type);
|
return read_as_i64(raw, base_type);
|
||||||
}
|
}
|
||||||
ensure_numeric(datatype, "FixedPoint (signed)")?;
|
ensure_numeric(datatype, "FixedPoint (signed)")?;
|
||||||
@@ -1014,19 +953,24 @@ pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatEr
|
|||||||
}
|
}
|
||||||
|
|
||||||
let order = get_byte_order(datatype);
|
let order = get_byte_order(datatype);
|
||||||
let (off, prec) = fixed_bits(datatype);
|
|
||||||
let mut result = Vec::with_capacity(count);
|
let mut result = Vec::with_capacity(count);
|
||||||
for i in 0..count {
|
for i in 0..count {
|
||||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||||
let full = read_unsigned_int(chunk, elem_size, &order);
|
result.push(decode_scalar(chunk, datatype, &order)?.to_i64());
|
||||||
result.push(extract_signed(full, off, prec));
|
|
||||||
}
|
}
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Convert raw bytes to `u64` values.
|
/// Convert raw bytes to `u64` values.
|
||||||
|
///
|
||||||
|
/// Values are converted the way libhdf5 converts them: integers outside the
|
||||||
|
/// target range saturate at its minimum or maximum (a negative value read as
|
||||||
|
/// unsigned is 0), and floating-point data is truncated toward zero and
|
||||||
|
/// saturated, with NaN read as 0.
|
||||||
pub fn read_as_u64(raw: &[u8], datatype: &Datatype) -> Result<Vec<u64>, FormatError> {
|
pub fn read_as_u64(raw: &[u8], datatype: &Datatype) -> Result<Vec<u64>, FormatError> {
|
||||||
if let Datatype::Array { base_type, .. } = datatype {
|
// Array datatypes read as a flat sequence of their base elements, and
|
||||||
|
// enumerations (h5py's bool among them) as their integer values.
|
||||||
|
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||||
return read_as_u64(raw, base_type);
|
return read_as_u64(raw, base_type);
|
||||||
}
|
}
|
||||||
ensure_numeric(datatype, "FixedPoint (unsigned)")?;
|
ensure_numeric(datatype, "FixedPoint (unsigned)")?;
|
||||||
@@ -1039,19 +983,19 @@ pub fn read_as_u64(raw: &[u8], datatype: &Datatype) -> Result<Vec<u64>, FormatEr
|
|||||||
}
|
}
|
||||||
let count = raw.len() / elem_size;
|
let count = raw.len() / elem_size;
|
||||||
let order = get_byte_order(datatype);
|
let order = get_byte_order(datatype);
|
||||||
let (off, prec) = fixed_bits(datatype);
|
|
||||||
let mut result = Vec::with_capacity(count);
|
let mut result = Vec::with_capacity(count);
|
||||||
for i in 0..count {
|
for i in 0..count {
|
||||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||||
let full = read_unsigned_int(chunk, elem_size, &order);
|
result.push(decode_scalar(chunk, datatype, &order)?.to_u64());
|
||||||
result.push(extract_unsigned(full, off, prec));
|
|
||||||
}
|
}
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Convert raw bytes to `f32` values.
|
/// Convert raw bytes to `f32` values.
|
||||||
pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatError> {
|
pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatError> {
|
||||||
if let Datatype::Array { base_type, .. } = datatype {
|
// Array datatypes read as a flat sequence of their base elements, and
|
||||||
|
// enumerations (h5py's bool among them) as their integer values.
|
||||||
|
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||||
return read_as_f32(raw, base_type);
|
return read_as_f32(raw, base_type);
|
||||||
}
|
}
|
||||||
ensure_numeric(datatype, "FloatingPoint")?;
|
ensure_numeric(datatype, "FloatingPoint")?;
|
||||||
@@ -1066,25 +1010,11 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
|||||||
|
|
||||||
// Fast path: native-endian f32 — single bulk memcpy
|
// Fast path: native-endian f32 — single bulk memcpy
|
||||||
#[cfg(target_endian = "little")]
|
#[cfg(target_endian = "little")]
|
||||||
if matches!(
|
if is_native_le_float(datatype, FloatFormat::Single) {
|
||||||
datatype,
|
|
||||||
Datatype::FloatingPoint {
|
|
||||||
size: 4,
|
|
||||||
byte_order: DatatypeByteOrder::LittleEndian,
|
|
||||||
..
|
|
||||||
}
|
|
||||||
) {
|
|
||||||
return Ok(native_le_to_vec::<f32>(raw, count));
|
return Ok(native_le_to_vec::<f32>(raw, count));
|
||||||
}
|
}
|
||||||
// Little-endian half precision (numpy float16): widen directly.
|
// Little-endian IEEE half precision (numpy float16): widen directly.
|
||||||
if matches!(
|
if is_native_le_float(datatype, FloatFormat::Half) {
|
||||||
datatype,
|
|
||||||
Datatype::FloatingPoint {
|
|
||||||
size: 2,
|
|
||||||
byte_order: DatatypeByteOrder::LittleEndian,
|
|
||||||
..
|
|
||||||
}
|
|
||||||
) {
|
|
||||||
let (halves, _) = raw[..count * 2].as_chunks::<2>();
|
let (halves, _) = raw[..count * 2].as_chunks::<2>();
|
||||||
return Ok(halves
|
return Ok(halves
|
||||||
.iter()
|
.iter()
|
||||||
@@ -1094,18 +1024,22 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
|||||||
|
|
||||||
let order = get_byte_order(datatype);
|
let order = get_byte_order(datatype);
|
||||||
let mut result = Vec::with_capacity(count);
|
let mut result = Vec::with_capacity(count);
|
||||||
|
if let Datatype::FloatingPoint { .. } = datatype {
|
||||||
|
let format = FloatFormat::of(datatype)?;
|
||||||
|
for chunk in raw.chunks_exact(elem_size) {
|
||||||
|
result.push(match format {
|
||||||
|
FloatFormat::Single => read_f32_bytes(chunk, &order),
|
||||||
|
FloatFormat::Half => read_f16_bytes(chunk, &order),
|
||||||
|
// Double rounds; every other supported layout (bfloat16, FP8)
|
||||||
|
// is exact in f32.
|
||||||
|
_ => format.decode(chunk, &order) as f32,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
return Ok(result);
|
||||||
|
}
|
||||||
for i in 0..count {
|
for i in 0..count {
|
||||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||||
match datatype {
|
match datatype {
|
||||||
Datatype::FloatingPoint { size: 4, .. } => {
|
|
||||||
result.push(read_f32_bytes(chunk, &order));
|
|
||||||
}
|
|
||||||
Datatype::FloatingPoint { size: 8, .. } => {
|
|
||||||
result.push(read_f64_bytes(chunk, &order) as f32);
|
|
||||||
}
|
|
||||||
Datatype::FloatingPoint { size: 2, .. } => {
|
|
||||||
result.push(read_f16_bytes(chunk, &order));
|
|
||||||
}
|
|
||||||
Datatype::FixedPoint {
|
Datatype::FixedPoint {
|
||||||
signed: true,
|
signed: true,
|
||||||
size,
|
size,
|
||||||
@@ -1140,8 +1074,15 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Convert raw bytes to `i32` values.
|
/// Convert raw bytes to `i32` values.
|
||||||
|
///
|
||||||
|
/// Values are converted the way libhdf5 converts them: integers outside the
|
||||||
|
/// target range saturate at its minimum or maximum (a negative value read as
|
||||||
|
/// unsigned is 0), and floating-point data is truncated toward zero and
|
||||||
|
/// saturated, with NaN read as 0.
|
||||||
pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatError> {
|
pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatError> {
|
||||||
if let Datatype::Array { base_type, .. } = datatype {
|
// Array datatypes read as a flat sequence of their base elements, and
|
||||||
|
// enumerations (h5py's bool among them) as their integer values.
|
||||||
|
if let Datatype::Array { base_type, .. } | Datatype::Enumeration { base_type, .. } = datatype {
|
||||||
return read_as_i32(raw, base_type);
|
return read_as_i32(raw, base_type);
|
||||||
}
|
}
|
||||||
ensure_numeric(datatype, "FixedPoint")?;
|
ensure_numeric(datatype, "FixedPoint")?;
|
||||||
@@ -1162,6 +1103,7 @@ pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatEr
|
|||||||
datatype,
|
datatype,
|
||||||
Datatype::FixedPoint {
|
Datatype::FixedPoint {
|
||||||
byte_order: DatatypeByteOrder::LittleEndian,
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
signed: true,
|
||||||
..
|
..
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
@@ -1170,12 +1112,10 @@ pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatEr
|
|||||||
}
|
}
|
||||||
|
|
||||||
let order = get_byte_order(datatype);
|
let order = get_byte_order(datatype);
|
||||||
let (off, prec) = fixed_bits(datatype);
|
|
||||||
let mut result = Vec::with_capacity(count);
|
let mut result = Vec::with_capacity(count);
|
||||||
for i in 0..count {
|
for i in 0..count {
|
||||||
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
let chunk = &raw[i * elem_size..(i + 1) * elem_size];
|
||||||
let full = read_unsigned_int(chunk, elem_size, &order);
|
result.push(decode_scalar(chunk, datatype, &order)?.to_i32());
|
||||||
result.push(extract_signed(full, off, prec) as i32);
|
|
||||||
}
|
}
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
@@ -1616,6 +1556,174 @@ fn reorder_bytes(bytes: &[u8], order: &DatatypeByteOrder) -> [u8; 8] {
|
|||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// How the bits of a floating-point datatype are laid out, read from the
|
||||||
|
/// datatype message's fields rather than assumed from its size (a 2-byte
|
||||||
|
/// float may be IEEE half or bfloat16).
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||||
|
enum FloatFormat {
|
||||||
|
/// IEEE-754 binary16.
|
||||||
|
Half,
|
||||||
|
/// IEEE-754 binary32.
|
||||||
|
Single,
|
||||||
|
/// IEEE-754 binary64.
|
||||||
|
Double,
|
||||||
|
/// Any other IEEE-style layout (implied leading mantissa bit, all-ones
|
||||||
|
/// exponent for infinity/NaN) whose values are all exact in `f64`:
|
||||||
|
/// bfloat16, the FP8 formats, and similar.
|
||||||
|
Other(FloatLayout),
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq)]
|
||||||
|
struct FloatLayout {
|
||||||
|
exponent_location: u32,
|
||||||
|
exponent_size: u32,
|
||||||
|
mantissa_location: u32,
|
||||||
|
mantissa_size: u32,
|
||||||
|
exponent_bias: u32,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl FloatFormat {
|
||||||
|
fn of(dt: &Datatype) -> Result<FloatFormat, FormatError> {
|
||||||
|
let Datatype::FloatingPoint {
|
||||||
|
size,
|
||||||
|
exponent_location,
|
||||||
|
exponent_size,
|
||||||
|
mantissa_location,
|
||||||
|
mantissa_size,
|
||||||
|
exponent_bias,
|
||||||
|
..
|
||||||
|
} = dt
|
||||||
|
else {
|
||||||
|
return Err(FormatError::TypeMismatch {
|
||||||
|
expected: "FloatingPoint",
|
||||||
|
actual: datatype_name(dt),
|
||||||
|
});
|
||||||
|
};
|
||||||
|
let layout = FloatLayout {
|
||||||
|
exponent_location: u32::from(*exponent_location),
|
||||||
|
exponent_size: u32::from(*exponent_size),
|
||||||
|
mantissa_location: u32::from(*mantissa_location),
|
||||||
|
mantissa_size: u32::from(*mantissa_size),
|
||||||
|
exponent_bias: *exponent_bias,
|
||||||
|
};
|
||||||
|
let fields = (
|
||||||
|
layout.exponent_location,
|
||||||
|
layout.exponent_size,
|
||||||
|
layout.mantissa_location,
|
||||||
|
layout.mantissa_size,
|
||||||
|
layout.exponent_bias,
|
||||||
|
);
|
||||||
|
let bits = size.saturating_mul(8);
|
||||||
|
// The sign bit is not kept in `Datatype`; every standard layout has it
|
||||||
|
// directly above the exponent, with the mantissa below.
|
||||||
|
let well_formed = layout.exponent_size > 0
|
||||||
|
&& layout.mantissa_size > 0
|
||||||
|
&& layout.mantissa_location + layout.mantissa_size <= layout.exponent_location
|
||||||
|
&& layout.exponent_location + layout.exponent_size < bits;
|
||||||
|
match (size, fields) {
|
||||||
|
(2, (10, 5, 0, 10, 15)) => Ok(FloatFormat::Half),
|
||||||
|
(4, (23, 8, 0, 23, 127)) => Ok(FloatFormat::Single),
|
||||||
|
(8, (52, 11, 0, 52, 1023)) => Ok(FloatFormat::Double),
|
||||||
|
_ if well_formed
|
||||||
|
&& *size <= 8
|
||||||
|
&& layout.exponent_size <= 11
|
||||||
|
&& layout.mantissa_size <= 52 =>
|
||||||
|
{
|
||||||
|
Ok(FloatFormat::Other(layout))
|
||||||
|
}
|
||||||
|
// Fields that cannot describe any float (e.g. left zeroed by a
|
||||||
|
// hand-built datatype): fall back to the IEEE type of that size.
|
||||||
|
(2, _) if !well_formed => Ok(FloatFormat::Half),
|
||||||
|
(4, _) if !well_formed => Ok(FloatFormat::Single),
|
||||||
|
(8, _) if !well_formed => Ok(FloatFormat::Double),
|
||||||
|
// x87 80-bit extended, binary128, ...: not representable in f64.
|
||||||
|
_ => Err(FormatError::TypeMismatch {
|
||||||
|
expected: "floating point of at most 64 bits (IEEE-style layout)",
|
||||||
|
actual: "FloatingPoint",
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn decode(self, bytes: &[u8], order: &DatatypeByteOrder) -> f64 {
|
||||||
|
match self {
|
||||||
|
FloatFormat::Half => f64::from(read_f16_bytes(bytes, order)),
|
||||||
|
FloatFormat::Single => f64::from(read_f32_bytes(bytes, order)),
|
||||||
|
FloatFormat::Double => read_f64_bytes(bytes, order),
|
||||||
|
FloatFormat::Other(layout) => {
|
||||||
|
layout.decode(read_unsigned_int(bytes, bytes.len(), order))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl FloatLayout {
|
||||||
|
/// Decode the value held in the low `size * 8` bits of `bits`.
|
||||||
|
fn decode(self, bits: u64) -> f64 {
|
||||||
|
let field = |location: u32, size: u32| (bits >> location) & ((1u64 << size) - 1);
|
||||||
|
let exponent = field(self.exponent_location, self.exponent_size);
|
||||||
|
let mantissa = field(self.mantissa_location, self.mantissa_size);
|
||||||
|
let negative = field(self.exponent_location + self.exponent_size, 1) == 1;
|
||||||
|
let max_exponent = (1u64 << self.exponent_size) - 1;
|
||||||
|
let magnitude = if exponent == max_exponent {
|
||||||
|
if mantissa == 0 {
|
||||||
|
f64::INFINITY
|
||||||
|
} else {
|
||||||
|
f64::NAN
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
let bias = i64::from(self.exponent_bias);
|
||||||
|
let msize = i64::from(self.mantissa_size);
|
||||||
|
// value = significand * 2^power, with an implied leading 1 unless
|
||||||
|
// the number is subnormal (exponent field 0).
|
||||||
|
let (significand, power) = if exponent == 0 {
|
||||||
|
(mantissa, 1 - bias - msize)
|
||||||
|
} else {
|
||||||
|
(
|
||||||
|
mantissa | (1u64 << self.mantissa_size),
|
||||||
|
exponent as i64 - bias - msize,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
scale_by_pow2(significand as f64, power)
|
||||||
|
};
|
||||||
|
if negative { -magnitude } else { magnitude }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `x * 2^power` without `std` (no `powi`/`libm`). `x` is a non-negative
|
||||||
|
/// integer below 2^53, so it is exact.
|
||||||
|
fn scale_by_pow2(x: f64, power: i64) -> f64 {
|
||||||
|
if x == 0.0 || power < -1200 {
|
||||||
|
return 0.0;
|
||||||
|
}
|
||||||
|
if power > 1100 {
|
||||||
|
return f64::INFINITY;
|
||||||
|
}
|
||||||
|
let pow2 = |p: i64| f64::from_bits(((p + 1023) as u64) << 52);
|
||||||
|
let mut x = x;
|
||||||
|
let mut power = power;
|
||||||
|
while power > 1023 {
|
||||||
|
x *= pow2(1023);
|
||||||
|
power -= 1023;
|
||||||
|
}
|
||||||
|
while power < -1022 {
|
||||||
|
x *= pow2(-1022);
|
||||||
|
power += 1022;
|
||||||
|
}
|
||||||
|
x * pow2(power)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `datatype` is the little-endian IEEE float `format`, whose bytes
|
||||||
|
/// can be copied straight into native values on a little-endian target.
|
||||||
|
fn is_native_le_float(datatype: &Datatype, format: FloatFormat) -> bool {
|
||||||
|
matches!(
|
||||||
|
datatype,
|
||||||
|
Datatype::FloatingPoint {
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
..
|
||||||
|
}
|
||||||
|
) && FloatFormat::of(datatype).is_ok_and(|f| f == format)
|
||||||
|
}
|
||||||
|
|
||||||
fn read_f64_bytes(bytes: &[u8], order: &DatatypeByteOrder) -> f64 {
|
fn read_f64_bytes(bytes: &[u8], order: &DatatypeByteOrder) -> f64 {
|
||||||
let buf = reorder_bytes(bytes, order);
|
let buf = reorder_bytes(bytes, order);
|
||||||
f64::from_le_bytes(buf)
|
f64::from_le_bytes(buf)
|
||||||
@@ -1666,20 +1774,6 @@ fn effective_bits(size: usize, bit_offset: u16, bit_precision: u16) -> (u32, u32
|
|||||||
(bit_offset as u32, prec)
|
(bit_offset as u32, prec)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// `(bit_offset, bit_precision)` for a fixed-point datatype, full width for
|
|
||||||
/// other types.
|
|
||||||
fn fixed_bits(datatype: &Datatype) -> (u32, u32) {
|
|
||||||
match datatype {
|
|
||||||
Datatype::FixedPoint {
|
|
||||||
size,
|
|
||||||
bit_offset,
|
|
||||||
bit_precision,
|
|
||||||
..
|
|
||||||
} => effective_bits(*size as usize, *bit_offset, *bit_precision),
|
|
||||||
_ => (0, 0),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Whether a datatype occupies its full storage width (bit offset 0, precision
|
/// Whether a datatype occupies its full storage width (bit offset 0, precision
|
||||||
/// == size·8), in which case the bulk-copy fast read paths apply. Non
|
/// == size·8), in which case the bulk-copy fast read paths apply. Non
|
||||||
/// fixed-point types are treated as full width.
|
/// fixed-point types are treated as full width.
|
||||||
@@ -1892,6 +1986,67 @@ mod tests {
|
|||||||
assert_eq!(read_as_u64(&raw, &dt).unwrap(), vec![4095, 1, 2048]);
|
assert_eq!(read_as_u64(&raw, &dt).unwrap(), vec![4095, 1, 2048]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn float_to_int_truncates_and_saturates() {
|
||||||
|
// Values libhdf5 hands to an undefined C cast: NaN reads as 0 and
|
||||||
|
// exactly 2^63 saturates instead of wrapping to i64::MIN.
|
||||||
|
let dt = make_f64_le_type();
|
||||||
|
let vals = [f64::NAN, 2f64.powi(63), -2.5, 2.0f64.powi(64)];
|
||||||
|
let raw: Vec<u8> = vals.iter().flat_map(|v| v.to_le_bytes()).collect();
|
||||||
|
assert_eq!(
|
||||||
|
read_as_i64(&raw, &dt).unwrap(),
|
||||||
|
vec![0, i64::MAX, -2, i64::MAX]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
read_as_u64(&raw, &dt).unwrap(),
|
||||||
|
vec![0, 1 << 63, 0, u64::MAX]
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
read_as_i32(&raw, &dt).unwrap(),
|
||||||
|
vec![0, i32::MAX, -2, i32::MAX]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bfloat16_and_fp8_decode_by_fields() {
|
||||||
|
// bfloat16 is a 2-byte float that is not IEEE half.
|
||||||
|
let bf16 = Datatype::FloatingPoint {
|
||||||
|
size: 2,
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 16,
|
||||||
|
exponent_location: 7,
|
||||||
|
exponent_size: 8,
|
||||||
|
mantissa_location: 0,
|
||||||
|
mantissa_size: 7,
|
||||||
|
exponent_bias: 127,
|
||||||
|
};
|
||||||
|
let raw: Vec<u8> = [0x3FC0u16, 0xC010, 0x7F80, 0x0001]
|
||||||
|
.iter()
|
||||||
|
.flat_map(|v| v.to_le_bytes())
|
||||||
|
.collect();
|
||||||
|
let got = read_as_f64(&raw, &bf16).unwrap();
|
||||||
|
assert_eq!(&got[..3], &[1.5, -2.25, f64::INFINITY]);
|
||||||
|
assert_eq!(got[3], 2f64.powi(-133)); // smallest subnormal
|
||||||
|
assert_eq!(read_as_f32(&raw, &bf16).unwrap()[..2], [1.5, -2.25]);
|
||||||
|
|
||||||
|
// FP8 E4M3: 1, -1, 2, 0, NaN (IEEE-style, as libhdf5 treats it).
|
||||||
|
let e4m3 = Datatype::FloatingPoint {
|
||||||
|
size: 1,
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 8,
|
||||||
|
exponent_location: 3,
|
||||||
|
exponent_size: 4,
|
||||||
|
mantissa_location: 0,
|
||||||
|
mantissa_size: 3,
|
||||||
|
exponent_bias: 7,
|
||||||
|
};
|
||||||
|
let got = read_as_f64(&[0x38, 0xB8, 0x40, 0x00, 0x7E], &e4m3).unwrap();
|
||||||
|
assert_eq!(&got[..4], &[1.0, -1.0, 2.0, 0.0]);
|
||||||
|
assert!(got[4].is_nan());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn full_width_signed_unchanged() {
|
fn full_width_signed_unchanged() {
|
||||||
// Regression: full-width 32-bit signed must be unaffected.
|
// Regression: full-width 32-bit signed must be unaffected.
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
//! for compound, enumeration, variable-length, and array types.
|
//! for compound, enumeration, variable-length, and array types.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{boxed::Box, string::String, vec, vec::Vec};
|
use alloc::{boxed::Box, format, string::String, vec, vec::Vec};
|
||||||
|
|
||||||
use byteorder::{ByteOrder, LittleEndian};
|
use byteorder::{ByteOrder, LittleEndian};
|
||||||
|
|
||||||
@@ -137,6 +137,17 @@ pub enum Datatype {
|
|||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Longest opaque tag that can be stored: its NUL-padded length must fit
|
||||||
|
/// the 8-bit length in the datatype's class bits.
|
||||||
|
pub const MAX_OPAQUE_TAG_LEN: usize = 248;
|
||||||
|
|
||||||
|
/// An opaque tag up to (not including) its first NUL.
|
||||||
|
fn opaque_tag_text(tag: &[u8]) -> &[u8] {
|
||||||
|
tag.iter()
|
||||||
|
.position(|&b| b == 0)
|
||||||
|
.map_or(tag, |end| &tag[..end])
|
||||||
|
}
|
||||||
|
|
||||||
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
|
||||||
match offset.checked_add(needed) {
|
match offset.checked_add(needed) {
|
||||||
Some(end) if end <= data.len() => Ok(()),
|
Some(end) if end <= data.len() => Ok(()),
|
||||||
@@ -361,7 +372,10 @@ impl Datatype {
|
|||||||
// Opaque
|
// Opaque
|
||||||
let tag_len = bf0 as usize;
|
let tag_len = bf0 as usize;
|
||||||
ensure_len(data, pos, tag_len)?;
|
ensure_len(data, pos, tag_len)?;
|
||||||
let tag = data[pos..pos + tag_len].to_vec();
|
// The stored tag is NUL-padded to a multiple of 8 bytes; the
|
||||||
|
// tag itself ends at the first NUL (libhdf5 reads it with
|
||||||
|
// `strndup`).
|
||||||
|
let tag = opaque_tag_text(&data[pos..pos + tag_len]).to_vec();
|
||||||
// Tags are padded to multiple of 8 bytes
|
// Tags are padded to multiple of 8 bytes
|
||||||
let padded = (tag_len + 7) & !7;
|
let padded = (tag_len + 7) & !7;
|
||||||
let pos = 8 + padded; // from start of properties
|
let pos = 8 + padded; // from start of properties
|
||||||
@@ -409,13 +423,44 @@ impl Datatype {
|
|||||||
ensure_len(data, pos, 4)?;
|
ensure_len(data, pos, 4)?;
|
||||||
let byte_offset = LittleEndian::read_u32(&data[pos..pos + 4]) as u64;
|
let byte_offset = LittleEndian::read_u32(&data[pos..pos + 4]) as u64;
|
||||||
pos += 4;
|
pos += 4;
|
||||||
|
// v1 members can be fixed-size arrays of the member
|
||||||
|
// type (libhdf5 builds an array type from these
|
||||||
|
// fields; the permutation is ignored, as libhdf5
|
||||||
|
// does). Skipping them read a `[4] i32` member as
|
||||||
|
// one `i32`.
|
||||||
|
let mut array_dims = Vec::new();
|
||||||
if version == 1 {
|
if version == 1 {
|
||||||
ensure_len(data, pos, 28)?;
|
ensure_len(data, pos, 28)?;
|
||||||
|
let ndims = data[pos] as usize;
|
||||||
|
// libhdf5 refuses more than four dimensions and,
|
||||||
|
// when building the array type, a zero-sized one.
|
||||||
|
let zero_dim = (0..ndims.min(4)).any(|j| {
|
||||||
|
let at = pos + 12 + 4 * j;
|
||||||
|
LittleEndian::read_u32(&data[at..at + 4]) == 0
|
||||||
|
});
|
||||||
|
if ndims > 4 || zero_dim {
|
||||||
|
return Err(FormatError::InvalidDatatypeVersion {
|
||||||
|
class: class_id,
|
||||||
|
version,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
array_dims = (0..ndims)
|
||||||
|
.map(|j| {
|
||||||
|
let at = pos + 12 + 4 * j;
|
||||||
|
LittleEndian::read_u32(&data[at..at + 4])
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
pos += 28;
|
pos += 28;
|
||||||
}
|
}
|
||||||
let (member_dt, consumed) =
|
let (mut member_dt, consumed) =
|
||||||
Self::parse_with_depth(&data[pos..], depth + 1)?;
|
Self::parse_with_depth(&data[pos..], depth + 1)?;
|
||||||
pos += consumed;
|
pos += consumed;
|
||||||
|
if !array_dims.is_empty() {
|
||||||
|
member_dt = Datatype::Array {
|
||||||
|
base_type: Box::new(member_dt),
|
||||||
|
dimensions: array_dims,
|
||||||
|
};
|
||||||
|
}
|
||||||
members.push(CompoundMember {
|
members.push(CompoundMember {
|
||||||
name,
|
name,
|
||||||
byte_offset,
|
byte_offset,
|
||||||
@@ -767,7 +812,77 @@ impl Datatype {
|
|||||||
buf.extend_from_slice(&base_type.serialize());
|
buf.extend_from_slice(&base_type.serialize());
|
||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
_ => Vec::new(),
|
Datatype::Time {
|
||||||
|
size,
|
||||||
|
bit_precision,
|
||||||
|
} => {
|
||||||
|
// Byte order is not modelled for time types; write little-endian.
|
||||||
|
let mut buf = Self::build_header(2, 1, [0, 0, 0], *size);
|
||||||
|
buf.extend_from_slice(&bit_precision.to_le_bytes());
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
Datatype::BitField {
|
||||||
|
size,
|
||||||
|
byte_order,
|
||||||
|
bit_offset,
|
||||||
|
bit_precision,
|
||||||
|
} => {
|
||||||
|
let bf0 = u8::from(matches!(byte_order, DatatypeByteOrder::BigEndian));
|
||||||
|
let mut buf = Self::build_header(4, 1, [bf0, 0, 0], *size);
|
||||||
|
buf.extend_from_slice(&bit_offset.to_le_bytes());
|
||||||
|
buf.extend_from_slice(&bit_precision.to_le_bytes());
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
Datatype::Opaque { size, tag } => {
|
||||||
|
// The tag is stored NUL-padded to a multiple of 8 bytes and the
|
||||||
|
// padded length goes in the class bits, as libhdf5 writes it.
|
||||||
|
// A tag longer than MAX_OPAQUE_TAG_LEN cannot be encoded;
|
||||||
|
// `check_encodable` rejects it before a file is written.
|
||||||
|
let tag = opaque_tag_text(tag);
|
||||||
|
let tag = &tag[..tag.len().min(MAX_OPAQUE_TAG_LEN)];
|
||||||
|
let padded = tag.len().div_ceil(8) * 8;
|
||||||
|
let mut buf = Self::build_header(5, 1, [padded as u8, 0, 0], *size);
|
||||||
|
buf.extend_from_slice(tag);
|
||||||
|
buf.resize(8 + padded, 0);
|
||||||
|
buf
|
||||||
|
}
|
||||||
|
Datatype::Reference { size, ref_type } => {
|
||||||
|
// Legacy references are datatype version 1; the H5T_STD_REF
|
||||||
|
// kinds only exist from version 4, which also carries their
|
||||||
|
// encoding version (1) in the high nibble.
|
||||||
|
let (version, bf0) = match ref_type {
|
||||||
|
ReferenceType::Object => (1, 0),
|
||||||
|
ReferenceType::DatasetRegion => (1, 1),
|
||||||
|
ReferenceType::Object2 => (4, 0x12),
|
||||||
|
ReferenceType::DatasetRegion2 => (4, 0x13),
|
||||||
|
ReferenceType::Attribute => (4, 0x14),
|
||||||
|
};
|
||||||
|
Self::build_header(7, version, [bf0, 0, 0], *size)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check that this datatype can be written: every part of it has an
|
||||||
|
/// on-disk encoding. [`Self::serialize`] cannot report errors, so the
|
||||||
|
/// writer calls this first.
|
||||||
|
pub fn check_encodable(&self) -> Result<(), FormatError> {
|
||||||
|
match self {
|
||||||
|
Datatype::Opaque { tag, .. } if opaque_tag_text(tag).len() > MAX_OPAQUE_TAG_LEN => {
|
||||||
|
Err(FormatError::SerializationError(format!(
|
||||||
|
"opaque tag is {} bytes; at most {MAX_OPAQUE_TAG_LEN} can be stored",
|
||||||
|
opaque_tag_text(tag).len()
|
||||||
|
)))
|
||||||
|
}
|
||||||
|
Datatype::String { size: 0, .. } => Err(FormatError::SerializationError(
|
||||||
|
"fixed-length string datatype of size 0 (libhdf5 requires at least 1 byte)".into(),
|
||||||
|
)),
|
||||||
|
Datatype::Compound { members, .. } => members
|
||||||
|
.iter()
|
||||||
|
.try_for_each(|m| m.datatype.check_encodable()),
|
||||||
|
Datatype::Enumeration { base_type, .. }
|
||||||
|
| Datatype::VariableLength { base_type, .. }
|
||||||
|
| Datatype::Array { base_type, .. } => base_type.check_encodable(),
|
||||||
|
_ => Ok(()),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1252,6 +1367,64 @@ mod tests {
|
|||||||
assert_xyid_compound(dt);
|
assert_xyid_compound(dt);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_compound_v1_member_array_fields() {
|
||||||
|
// HDF5 1.6 wrote array members of a v1 compound through the legacy
|
||||||
|
// per-member fields (as in libhdf5's tools/test/testfiles/
|
||||||
|
// tcompound.h5 `type2`: `int_array` [4] i32, `float_array` [5][6]
|
||||||
|
// f32). They used to be skipped, reading each member as a scalar.
|
||||||
|
let i32le: [u8; 12] = [
|
||||||
|
0x10, 0x08, 0x00, 0x00, 0x04, 0x00, 0x00, 0x00, 0x00, 0x00, 0x20, 0x00,
|
||||||
|
];
|
||||||
|
let mut b = vec![0x16, 0x02, 0x00, 0x00, 0x88, 0x00, 0x00, 0x00];
|
||||||
|
for (name, offset, dims) in [
|
||||||
|
(&b"int_array"[..], 0u32, &[4u32][..]),
|
||||||
|
(&b"xy"[..], 16, &[5u32, 6][..]),
|
||||||
|
] {
|
||||||
|
let mut padded = name.to_vec();
|
||||||
|
padded.resize((name.len() + 1 + 7) & !7, 0);
|
||||||
|
b.extend_from_slice(&padded);
|
||||||
|
b.extend_from_slice(&offset.to_le_bytes());
|
||||||
|
b.push(dims.len() as u8);
|
||||||
|
b.extend_from_slice(&[0u8; 3 + 4 + 4]); // reserved, permutation, reserved
|
||||||
|
for j in 0..4 {
|
||||||
|
b.extend_from_slice(&dims.get(j).copied().unwrap_or(0).to_le_bytes());
|
||||||
|
}
|
||||||
|
b.extend_from_slice(&i32le);
|
||||||
|
}
|
||||||
|
let (dt, consumed) = Datatype::parse(&b).unwrap();
|
||||||
|
assert_eq!(consumed, b.len());
|
||||||
|
let Datatype::Compound { members, .. } = dt else {
|
||||||
|
panic!("expected Compound, got {dt:?}");
|
||||||
|
};
|
||||||
|
let got: Vec<(&str, u64, u32, Option<Vec<u32>>)> = members
|
||||||
|
.iter()
|
||||||
|
.map(|m| {
|
||||||
|
let dims = match &m.datatype {
|
||||||
|
Datatype::Array { dimensions, .. } => Some(dimensions.clone()),
|
||||||
|
_ => None,
|
||||||
|
};
|
||||||
|
(m.name.as_str(), m.byte_offset, m.datatype.type_size(), dims)
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
assert_eq!(
|
||||||
|
got,
|
||||||
|
vec![
|
||||||
|
("int_array", 0, 16, Some(vec![4])),
|
||||||
|
("xy", 16, 120, Some(vec![5, 6])),
|
||||||
|
]
|
||||||
|
);
|
||||||
|
|
||||||
|
// More than four dimensions cannot be encoded, and libhdf5 refuses a
|
||||||
|
// zero-sized dimension (a fuzzed tcompound.h5, cve-2024-32616.h5).
|
||||||
|
let mut bad = b.clone();
|
||||||
|
bad[8 + 16 + 4] = 5;
|
||||||
|
assert!(Datatype::parse(&bad).is_err());
|
||||||
|
let mut bad = b.clone();
|
||||||
|
bad[8 + 16 + 4] = 2; // [4, 0]
|
||||||
|
assert!(Datatype::parse(&bad).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_compound_v1_truncated_is_error_not_panic() {
|
fn test_compound_v1_truncated_is_error_not_panic() {
|
||||||
let bytes = compound_v1_bytes();
|
let bytes = compound_v1_bytes();
|
||||||
@@ -1625,6 +1798,122 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn hex(s: &str) -> Vec<u8> {
|
||||||
|
(0..s.len())
|
||||||
|
.step_by(2)
|
||||||
|
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `serialize` used to return an empty message for these four classes,
|
||||||
|
/// which libhdf5 rejects ("ran off end of input buffer while decoding").
|
||||||
|
/// Expected bytes are libhdf5's own encoding (HDF5 2.0 `H5Tencode`, or the
|
||||||
|
/// datatype message of an HDF5 2.0 file for `H5T_STD_REF`).
|
||||||
|
#[test]
|
||||||
|
fn serialize_matches_libhdf5_for_time_bitfield_opaque_reference() {
|
||||||
|
let cases = [
|
||||||
|
(
|
||||||
|
Datatype::Reference {
|
||||||
|
size: 8,
|
||||||
|
ref_type: ReferenceType::Object,
|
||||||
|
},
|
||||||
|
"1700000008000000",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Datatype::Reference {
|
||||||
|
size: 12,
|
||||||
|
ref_type: ReferenceType::DatasetRegion,
|
||||||
|
},
|
||||||
|
"170100000c000000",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Datatype::Reference {
|
||||||
|
size: 18,
|
||||||
|
ref_type: ReferenceType::Object2,
|
||||||
|
},
|
||||||
|
"4712000012000000",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Datatype::BitField {
|
||||||
|
size: 1,
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 8,
|
||||||
|
},
|
||||||
|
"140000000100000000000800",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Datatype::BitField {
|
||||||
|
size: 2,
|
||||||
|
byte_order: DatatypeByteOrder::BigEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 16,
|
||||||
|
},
|
||||||
|
"140100000200000000001000",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Datatype::Opaque {
|
||||||
|
size: 4,
|
||||||
|
tag: b"mytag".to_vec(),
|
||||||
|
},
|
||||||
|
"15080000040000006d79746167000000",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Datatype::Opaque {
|
||||||
|
size: 4,
|
||||||
|
tag: b"12345678".to_vec(),
|
||||||
|
},
|
||||||
|
"15080000040000003132333435363738",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Datatype::Opaque {
|
||||||
|
size: 4,
|
||||||
|
tag: vec![],
|
||||||
|
},
|
||||||
|
"1500000004000000",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
Datatype::Time {
|
||||||
|
size: 4,
|
||||||
|
bit_precision: 32,
|
||||||
|
},
|
||||||
|
"12000000040000002000",
|
||||||
|
),
|
||||||
|
];
|
||||||
|
for (dt, expected) in cases {
|
||||||
|
let bytes = dt.serialize();
|
||||||
|
assert_eq!(bytes, hex(expected), "{dt:?}");
|
||||||
|
let (parsed, consumed) = Datatype::parse(&bytes).unwrap();
|
||||||
|
assert_eq!(parsed, dt);
|
||||||
|
assert_eq!(consumed, bytes.len());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn opaque_tag_padding_is_not_part_of_the_tag() {
|
||||||
|
// libhdf5 pads "mytag" to 8 bytes; parsing must not return the NULs,
|
||||||
|
// or copying the type would grow the tag.
|
||||||
|
let (dt, _) = Datatype::parse(&hex("15080000040000006d79746167000000")).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
dt,
|
||||||
|
Datatype::Opaque {
|
||||||
|
size: 4,
|
||||||
|
tag: b"mytag".to_vec()
|
||||||
|
}
|
||||||
|
);
|
||||||
|
let long = Datatype::Opaque {
|
||||||
|
size: 1,
|
||||||
|
tag: vec![b'x'; MAX_OPAQUE_TAG_LEN + 1],
|
||||||
|
};
|
||||||
|
assert!(long.check_encodable().is_err());
|
||||||
|
let ok = Datatype::Opaque {
|
||||||
|
size: 1,
|
||||||
|
tag: vec![b'x'; MAX_OPAQUE_TAG_LEN],
|
||||||
|
};
|
||||||
|
assert!(ok.check_encodable().is_ok());
|
||||||
|
assert_eq!(Datatype::parse(&ok.serialize()).unwrap().0, ok);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_error_invalid_reference_type() {
|
fn test_error_invalid_reference_type() {
|
||||||
let buf = build_dt_header(7, 1, [5, 0, 0], 8);
|
let buf = build_dt_header(7, 1, [5, 0, 0], 8);
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ extern crate alloc;
|
|||||||
use alloc::{vec, vec::Vec};
|
use alloc::{vec, vec::Vec};
|
||||||
|
|
||||||
use crate::checksum::jenkins_lookup3;
|
use crate::checksum::jenkins_lookup3;
|
||||||
use crate::chunked_write::WrittenChunk;
|
use crate::chunked_write::{WrittenChunk, filtered_chunk_size_len, push_addr, push_index_element};
|
||||||
|
|
||||||
/// Serialize a v4 Extensible Array layout message.
|
/// Serialize a v4 Extensible Array layout message.
|
||||||
pub(crate) fn serialize_v4_extensible_array(
|
pub(crate) fn serialize_v4_extensible_array(
|
||||||
@@ -58,11 +58,11 @@ pub(crate) fn serialize_v4_extensible_array(
|
|||||||
buf.push(4);
|
buf.push(4);
|
||||||
|
|
||||||
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
// EA creation parameters (must match AEHD and HDF5 C library defaults)
|
||||||
buf.push(32); // max_nelmts_bits
|
buf.push(MAX_NELMTS_BITS);
|
||||||
buf.push(4); // idx_blk_elmts
|
buf.push(IDX_BLK_ELMTS);
|
||||||
buf.push(4); // super_blk_min_data_ptrs
|
buf.push(SUP_BLK_MIN_DATA_PTRS);
|
||||||
buf.push(16); // data_blk_min_elmts
|
buf.push(DATA_BLK_MIN_ELMTS);
|
||||||
buf.push(10); // max_dblk_page_nelmts_bits
|
buf.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||||
|
|
||||||
// EA header address
|
// EA header address
|
||||||
match offset_size {
|
match offset_size {
|
||||||
@@ -74,304 +74,281 @@ pub(crate) fn serialize_v4_extensible_array(
|
|||||||
buf
|
buf
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// EA creation parameters — the HDF5 library's defaults for chunk indexes
|
||||||
|
// (`H5D_EARRAY_*`); the layout message above and the header must agree.
|
||||||
|
const MAX_NELMTS_BITS: u8 = 32;
|
||||||
|
const IDX_BLK_ELMTS: u8 = 4;
|
||||||
|
const SUP_BLK_MIN_DATA_PTRS: u8 = 4;
|
||||||
|
const DATA_BLK_MIN_ELMTS: u8 = 16;
|
||||||
|
const MAX_DBLK_PAGE_NELMTS_BITS: u8 = 10;
|
||||||
|
|
||||||
|
/// One data block of the array: its first element (relative to the end of
|
||||||
|
/// the index block's own elements), element count, and address when it is
|
||||||
|
/// allocated.
|
||||||
|
struct DataBlock {
|
||||||
|
start: usize,
|
||||||
|
nelmts: usize,
|
||||||
|
addr: Option<u64>,
|
||||||
|
}
|
||||||
|
|
||||||
/// Build a complete Extensible Array at a known absolute address.
|
/// Build a complete Extensible Array at a known absolute address.
|
||||||
///
|
///
|
||||||
/// For simplicity, we put all elements inline in the index block when the
|
/// `slots[i]` is the element at linear index `i` (see `chunk_grid`); `None`
|
||||||
/// number of chunks is small (up to idx_blk_elmts), otherwise use inline +
|
/// marks an unallocated chunk. The first `IDX_BLK_ELMTS` elements live in
|
||||||
/// direct data blocks.
|
/// the index block, the rest in data blocks grouped by super block level
|
||||||
|
/// exactly as `H5EA__hdr_init` sizes them: level `u` has `2^(u/2)` data
|
||||||
|
/// blocks of `DATA_BLK_MIN_ELMTS * 2^ceil(u/2)` elements. The data blocks of
|
||||||
|
/// the first levels are addressed straight from the index block; later
|
||||||
|
/// levels go through a super block (EASB). Data blocks larger than a page
|
||||||
|
/// (`2^MAX_DBLK_PAGE_NELMTS_BITS` elements) are paged, with their page-init
|
||||||
|
/// bits kept in the owning super block. Only blocks holding a defined element
|
||||||
|
/// are allocated; the rest keep the undefined address, as in a file the
|
||||||
|
/// library wrote.
|
||||||
pub fn build_extensible_array_at(
|
pub fn build_extensible_array_at(
|
||||||
chunks: &[WrittenChunk],
|
slots: &[Option<WrittenChunk>],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
has_filters: bool,
|
has_filters: bool,
|
||||||
ea_base_address: u64,
|
ea_base_address: u64,
|
||||||
) -> Vec<u8> {
|
) -> Vec<u8> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let num_elements = chunks.len();
|
let chunk_size_bytes = has_filters.then(|| filtered_chunk_size_len(slots));
|
||||||
|
let elem_size = os + chunk_size_bytes.map_or(0, |n| n + 4);
|
||||||
// Compute element encoding size (same logic as Fixed Array)
|
|
||||||
let chunk_size_bytes: usize = if has_filters {
|
|
||||||
let max_raw = chunks.iter().map(|c| c.raw_size).max().unwrap_or(1);
|
|
||||||
let log2_val = if max_raw <= 1 {
|
|
||||||
0
|
|
||||||
} else {
|
|
||||||
63 - max_raw.leading_zeros()
|
|
||||||
};
|
|
||||||
let len = 1 + ((log2_val + 8) / 8) as usize;
|
|
||||||
len.min(8)
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
|
|
||||||
let elem_size = if has_filters {
|
|
||||||
os + chunk_size_bytes + 4
|
|
||||||
} else {
|
|
||||||
os
|
|
||||||
};
|
|
||||||
|
|
||||||
let client_id: u8 = if has_filters { 1 } else { 0 };
|
let client_id: u8 = if has_filters { 1 } else { 0 };
|
||||||
|
let arr_off_size = (MAX_NELMTS_BITS as usize).div_ceil(8);
|
||||||
|
let page_nelmts = 1usize << MAX_DBLK_PAGE_NELMTS_BITS;
|
||||||
|
let idx_blk = IDX_BLK_ELMTS as usize;
|
||||||
|
|
||||||
// EA creation parameters — must match HDF5 C library defaults exactly
|
// Elements past the last defined one are never realised
|
||||||
let max_nelmts_bits: u8 = 32;
|
// (`max_idx_set` is one past the highest index ever set).
|
||||||
let idx_blk_elmts: u8 = 4;
|
let max_idx_set = slots.iter().rposition(Option::is_some).map_or(0, |i| i + 1);
|
||||||
let min_dblk_nelmts: u8 = 16;
|
let slots = &slots[..max_idx_set];
|
||||||
let super_blk_min_nelmts: u8 = 4;
|
let defined_in = |start: usize, n: usize| -> bool {
|
||||||
let max_dblk_nelmts_bits: u8 = 10;
|
let lo = idx_blk.saturating_add(start).min(slots.len());
|
||||||
|
let hi = idx_blk
|
||||||
|
.saturating_add(start)
|
||||||
|
.saturating_add(n)
|
||||||
|
.min(slots.len());
|
||||||
|
slots[lo..hi].iter().any(Option::is_some)
|
||||||
|
};
|
||||||
|
|
||||||
// EAHD size: fixed(12) + 6 stats(6*length_size) + addr(offset_size) + checksum(4)
|
// Super block levels: (ndblks, dblk_nelmts, first element).
|
||||||
|
let log2_dmin = (DATA_BLK_MIN_ELMTS as u32).trailing_zeros() as usize;
|
||||||
|
let nsblks = 1 + MAX_NELMTS_BITS as usize - log2_dmin;
|
||||||
|
let ndblk_addrs = 2 * (SUP_BLK_MIN_DATA_PTRS as usize - 1);
|
||||||
|
let mut levels: Vec<(usize, usize, usize)> = Vec::with_capacity(nsblks);
|
||||||
|
let mut start = 0usize;
|
||||||
|
for u in 0..nsblks {
|
||||||
|
let ndblks = 1usize << (u / 2);
|
||||||
|
let nelmts = (DATA_BLK_MIN_ELMTS as usize) << u.div_ceil(2);
|
||||||
|
levels.push((ndblks, nelmts, start));
|
||||||
|
// Saturate: on 32-bit targets the last levels only need to compare
|
||||||
|
// as "beyond the end".
|
||||||
|
start = start.saturating_add(ndblks.saturating_mul(nelmts));
|
||||||
|
}
|
||||||
|
// Levels whose data blocks the index block addresses directly.
|
||||||
|
let mut direct_levels = 0;
|
||||||
|
let mut n = 0;
|
||||||
|
while n < ndblk_addrs {
|
||||||
|
n += levels[direct_levels].0;
|
||||||
|
direct_levels += 1;
|
||||||
|
}
|
||||||
|
let nsblk_addrs = nsblks - direct_levels;
|
||||||
|
|
||||||
|
let dblk_size = |nelmts: usize| -> usize {
|
||||||
|
let prefix = 4 + 1 + 1 + os + arr_off_size + 4;
|
||||||
|
if nelmts > page_nelmts {
|
||||||
|
prefix + (nelmts / page_nelmts) * (page_nelmts * elem_size + 4)
|
||||||
|
} else {
|
||||||
|
prefix + nelmts * elem_size
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let sblk_bitmap_len = |ndblks: usize, nelmts: usize| -> usize {
|
||||||
|
if nelmts > page_nelmts {
|
||||||
|
ndblks * (nelmts / page_nelmts).div_ceil(8)
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Plan addresses: header, index block, the direct data blocks, then each
|
||||||
|
// allocated super block followed by its allocated data blocks.
|
||||||
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
let aehd_size = 4 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 1 + 6 * length_size as usize + os + 4;
|
||||||
let aeib_address = ea_base_address + aehd_size as u64;
|
let aeib_address = ea_base_address + aehd_size as u64;
|
||||||
|
let aeib_size = 4 + 1 + 1 + os + idx_blk * elem_size + ndblk_addrs * os + nsblk_addrs * os + 4;
|
||||||
|
let mut cursor = aeib_address + aeib_size as u64;
|
||||||
|
|
||||||
// Determine how many elements go inline vs data blocks
|
let mut ndata_blks = 0u64;
|
||||||
let n_inline = (idx_blk_elmts as usize).min(num_elements);
|
let mut data_blk_size = 0u64;
|
||||||
let remaining_after_inline = num_elements.saturating_sub(n_inline);
|
let mut nsuper_blks = 0u64;
|
||||||
|
let mut super_blk_size = 0u64;
|
||||||
|
let mut realized = idx_blk as u64;
|
||||||
|
|
||||||
// Compute super block layout per HDF5 spec
|
let mut plan_dblk = |cursor: &mut u64, start: usize, nelmts: usize| -> DataBlock {
|
||||||
let sblk_min = super_blk_min_nelmts as usize;
|
let addr = defined_in(start, nelmts).then(|| {
|
||||||
let log2_dblk_min = if min_dblk_nelmts <= 1 {
|
let a = *cursor;
|
||||||
0
|
let size = dblk_size(nelmts) as u64;
|
||||||
} else {
|
*cursor += size;
|
||||||
(min_dblk_nelmts as u32).trailing_zeros() as usize
|
ndata_blks += 1;
|
||||||
|
data_blk_size += size;
|
||||||
|
realized += nelmts as u64;
|
||||||
|
a
|
||||||
|
});
|
||||||
|
DataBlock {
|
||||||
|
start,
|
||||||
|
nelmts,
|
||||||
|
addr,
|
||||||
|
}
|
||||||
};
|
};
|
||||||
let nsblks = (max_nelmts_bits as usize).saturating_sub(log2_dblk_min) + 1;
|
|
||||||
|
|
||||||
// Direct data block addresses (from super blocks 0..sblk_min-1)
|
let mut direct: Vec<DataBlock> = Vec::with_capacity(ndblk_addrs);
|
||||||
let mut dblk_sizes: Vec<usize> = Vec::new();
|
for &(ndblks, nelmts, first) in &levels[..direct_levels] {
|
||||||
for sblk_idx in 0..sblk_min.min(nsblks) {
|
for k in 0..ndblks {
|
||||||
let ndblks = 1usize << (sblk_idx / 2);
|
direct.push(plan_dblk(&mut cursor, first + k * nelmts, nelmts));
|
||||||
let dblk_nelmts = (min_dblk_nelmts as usize) * (1 << sblk_idx.div_ceil(2));
|
|
||||||
for _ in 0..ndblks {
|
|
||||||
dblk_sizes.push(dblk_nelmts);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
let n_direct_dblks = dblk_sizes.len();
|
// (super block address, level, its data blocks)
|
||||||
|
let mut supers: Vec<(Option<u64>, usize, Vec<DataBlock>)> = Vec::with_capacity(nsblk_addrs);
|
||||||
// Super block addresses (for super blocks sblk_min..nsblks-1)
|
for (u, &(ndblks, nelmts, first)) in levels.iter().enumerate().skip(direct_levels) {
|
||||||
let n_sblk_addrs = nsblks.saturating_sub(sblk_min);
|
if !defined_in(first, ndblks.saturating_mul(nelmts)) {
|
||||||
|
supers.push((None, u, Vec::new()));
|
||||||
// EAIB size
|
continue;
|
||||||
let aeib_size = 4
|
|
||||||
+ 1
|
|
||||||
+ 1
|
|
||||||
+ os
|
|
||||||
+ idx_blk_elmts as usize * elem_size
|
|
||||||
+ n_direct_dblks * os
|
|
||||||
+ n_sblk_addrs * os
|
|
||||||
+ 4;
|
|
||||||
|
|
||||||
// Build AEHD
|
|
||||||
let mut aehd = Vec::with_capacity(aehd_size);
|
|
||||||
aehd.extend_from_slice(b"EAHD");
|
|
||||||
aehd.push(0); // version
|
|
||||||
aehd.push(client_id);
|
|
||||||
aehd.push(elem_size as u8);
|
|
||||||
aehd.push(max_nelmts_bits);
|
|
||||||
aehd.push(idx_blk_elmts);
|
|
||||||
aehd.push(min_dblk_nelmts);
|
|
||||||
aehd.push(super_blk_min_nelmts);
|
|
||||||
aehd.push(max_dblk_nelmts_bits);
|
|
||||||
|
|
||||||
// Count data blocks that will have chunks
|
|
||||||
let n_active_dblks: u64 = if remaining_after_inline > 0 {
|
|
||||||
let mut count = 0u64;
|
|
||||||
let mut ci = n_inline;
|
|
||||||
for &sz in &dblk_sizes {
|
|
||||||
if ci < num_elements {
|
|
||||||
count += 1;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
count
|
let sb_size =
|
||||||
} else {
|
4 + 1 + 1 + os + arr_off_size + sblk_bitmap_len(ndblks, nelmts) + ndblks * os + 4;
|
||||||
0
|
let sb_addr = cursor;
|
||||||
};
|
cursor += sb_size as u64;
|
||||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
nsuper_blks += 1;
|
||||||
let aedb_header_overhead = 4 + 1 + 1 + os + blk_off_size + 4;
|
super_blk_size += sb_size as u64;
|
||||||
let data_blk_total_size: u64 = if remaining_after_inline > 0 {
|
let dblks = (0..ndblks)
|
||||||
let mut total = 0u64;
|
.map(|k| plan_dblk(&mut cursor, first + k * nelmts, nelmts))
|
||||||
let mut ci = n_inline;
|
.collect();
|
||||||
for &sz in &dblk_sizes {
|
supers.push((Some(sb_addr), u, dblks));
|
||||||
if ci < num_elements {
|
}
|
||||||
total += (aedb_header_overhead + sz * elem_size) as u64;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
total
|
|
||||||
} else {
|
|
||||||
0
|
|
||||||
};
|
|
||||||
let max_idx_set: u64 = if remaining_after_inline > 0 {
|
|
||||||
let mut max_set = idx_blk_elmts as u64;
|
|
||||||
let mut ci = n_inline;
|
|
||||||
for &sz in &dblk_sizes {
|
|
||||||
if ci < num_elements {
|
|
||||||
max_set += sz as u64;
|
|
||||||
ci += sz;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
max_set
|
|
||||||
} else {
|
|
||||||
idx_blk_elmts as u64
|
|
||||||
};
|
|
||||||
|
|
||||||
|
let slot = |i: usize| slots.get(i).and_then(Option::as_ref);
|
||||||
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
let write_length = |buf: &mut Vec<u8>, val: u64| match length_size {
|
||||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
||||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
||||||
};
|
};
|
||||||
let write_addr = |buf: &mut Vec<u8>, val: u64| match offset_size {
|
let write_addr_opt = |buf: &mut Vec<u8>, addr: Option<u64>| match addr {
|
||||||
4 => buf.extend_from_slice(&(val as u32).to_le_bytes()),
|
Some(a) => push_addr(buf, a, offset_size),
|
||||||
_ => buf.extend_from_slice(&val.to_le_bytes()),
|
None => buf.extend(core::iter::repeat_n(0xFF, os)),
|
||||||
|
};
|
||||||
|
let block_prefix = |buf: &mut Vec<u8>, sig: &[u8; 4], block_off: usize| {
|
||||||
|
buf.extend_from_slice(sig);
|
||||||
|
buf.push(0); // version
|
||||||
|
buf.push(client_id);
|
||||||
|
push_addr(buf, ea_base_address, offset_size);
|
||||||
|
buf.extend_from_slice(&(block_off as u64).to_le_bytes()[..arr_off_size]);
|
||||||
|
};
|
||||||
|
// Serialise one data block (paged or not) onto `out`.
|
||||||
|
let write_dblk = |out: &mut Vec<u8>, db: &DataBlock| {
|
||||||
|
let at = out.len();
|
||||||
|
block_prefix(out, b"EADB", db.start);
|
||||||
|
let first = idx_blk + db.start;
|
||||||
|
if db.nelmts > page_nelmts {
|
||||||
|
// Paged: the prefix carries only its own checksum; each page
|
||||||
|
// follows with one of its own.
|
||||||
|
let sum = jenkins_lookup3(&out[at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
for p in 0..db.nelmts / page_nelmts {
|
||||||
|
let page_at = out.len();
|
||||||
|
for e in 0..page_nelmts {
|
||||||
|
let i = first + p * page_nelmts + e;
|
||||||
|
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[page_at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
for i in first..first + db.nelmts {
|
||||||
|
push_index_element(out, slot(i), offset_size, chunk_size_bytes);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[at..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
}
|
||||||
|
debug_assert_eq!(out.len() - at, dblk_size(db.nelmts));
|
||||||
};
|
};
|
||||||
|
|
||||||
write_length(&mut aehd, 0);
|
// Header (EAHD). The six statistics are, in order: super blocks, their
|
||||||
write_length(&mut aehd, 0);
|
// bytes, data blocks, their bytes, max index set, elements realised.
|
||||||
write_length(&mut aehd, n_active_dblks);
|
let mut out = Vec::with_capacity((cursor - ea_base_address) as usize);
|
||||||
write_length(&mut aehd, data_blk_total_size);
|
out.extend_from_slice(b"EAHD");
|
||||||
write_length(&mut aehd, num_elements as u64);
|
out.push(0); // version
|
||||||
write_length(&mut aehd, max_idx_set);
|
out.push(client_id);
|
||||||
|
out.push(elem_size as u8);
|
||||||
|
out.push(MAX_NELMTS_BITS);
|
||||||
|
out.push(IDX_BLK_ELMTS);
|
||||||
|
out.push(DATA_BLK_MIN_ELMTS);
|
||||||
|
out.push(SUP_BLK_MIN_DATA_PTRS);
|
||||||
|
out.push(MAX_DBLK_PAGE_NELMTS_BITS);
|
||||||
|
write_length(&mut out, nsuper_blks);
|
||||||
|
write_length(&mut out, super_blk_size);
|
||||||
|
write_length(&mut out, ndata_blks);
|
||||||
|
write_length(&mut out, data_blk_size);
|
||||||
|
write_length(&mut out, max_idx_set as u64);
|
||||||
|
write_length(&mut out, realized);
|
||||||
|
push_addr(&mut out, aeib_address, offset_size);
|
||||||
|
let sum = jenkins_lookup3(&out);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len(), aehd_size);
|
||||||
|
|
||||||
write_addr(&mut aehd, aeib_address);
|
// Index block (EAIB): inline elements, data block and super block
|
||||||
|
// addresses.
|
||||||
let aehd_checksum = jenkins_lookup3(&aehd);
|
let ib_start = out.len();
|
||||||
aehd.extend_from_slice(&aehd_checksum.to_le_bytes());
|
out.extend_from_slice(b"EAIB");
|
||||||
debug_assert_eq!(aehd.len(), aehd_size);
|
out.push(0);
|
||||||
|
out.push(client_id);
|
||||||
// Build AEIB
|
push_addr(&mut out, ea_base_address, offset_size);
|
||||||
let mut aeib = Vec::with_capacity(aeib_size);
|
for i in 0..idx_blk {
|
||||||
aeib.extend_from_slice(b"EAIB");
|
push_index_element(&mut out, slot(i), offset_size, chunk_size_bytes);
|
||||||
aeib.push(0);
|
|
||||||
aeib.push(client_id);
|
|
||||||
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&ea_base_address.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
|
for db in &direct {
|
||||||
// Inline elements
|
write_addr_opt(&mut out, db.addr);
|
||||||
#[allow(clippy::needless_range_loop)]
|
|
||||||
for i in 0..idx_blk_elmts as usize {
|
|
||||||
if i < n_inline {
|
|
||||||
write_chunk_element(
|
|
||||||
&mut aeib,
|
|
||||||
&chunks[i],
|
|
||||||
offset_size,
|
|
||||||
has_filters,
|
|
||||||
chunk_size_bytes,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
write_undefined_element(&mut aeib, offset_size, has_filters, chunk_size_bytes);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
for (sb_addr, _, _) in &supers {
|
||||||
|
write_addr_opt(&mut out, *sb_addr);
|
||||||
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[ib_start..]);
|
||||||
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
|
debug_assert_eq!(out.len() - ib_start, aeib_size);
|
||||||
|
|
||||||
// Data block addresses + build data blocks
|
for db in direct.iter().filter(|d| d.addr.is_some()) {
|
||||||
let mut data_blocks_buf = Vec::new();
|
write_dblk(&mut out, db);
|
||||||
let dblks_base = aeib_address + aeib_size as u64;
|
}
|
||||||
let mut dblk_cursor = dblks_base;
|
for (sb_addr, u, dblks) in &supers {
|
||||||
let mut chunk_idx = n_inline;
|
if sb_addr.is_none() {
|
||||||
|
|
||||||
for &nelmts in &dblk_sizes {
|
|
||||||
if chunk_idx >= num_elements {
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
}
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
let (ndblks, nelmts, first) = levels[*u];
|
||||||
match offset_size {
|
let sb_start = out.len();
|
||||||
4 => aeib.extend_from_slice(&(dblk_cursor as u32).to_le_bytes()),
|
block_prefix(&mut out, b"EASB", first);
|
||||||
8 => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
if nelmts > page_nelmts {
|
||||||
_ => aeib.extend_from_slice(&dblk_cursor.to_le_bytes()),
|
// Page-init bits, `npages` per data block, packed MSB-first
|
||||||
}
|
// (`H5VM_bit_set`): every page of an allocated data block is
|
||||||
|
// written.
|
||||||
// Build EADB
|
let npages = nelmts / page_nelmts;
|
||||||
let mut aedb = Vec::new();
|
let mut bitmap = vec![0u8; sblk_bitmap_len(ndblks, nelmts)];
|
||||||
aedb.extend_from_slice(b"EADB");
|
for (k, db) in dblks.iter().enumerate() {
|
||||||
aedb.push(0);
|
if db.addr.is_some() {
|
||||||
aedb.push(client_id);
|
for p in 0..npages {
|
||||||
match offset_size {
|
let bit = k * npages + p;
|
||||||
4 => aedb.extend_from_slice(&(ea_base_address as u32).to_le_bytes()),
|
bitmap[bit / 8] |= 0x80 >> (bit % 8);
|
||||||
8 => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
}
|
||||||
_ => aedb.extend_from_slice(&ea_base_address.to_le_bytes()),
|
}
|
||||||
}
|
|
||||||
|
|
||||||
let blk_off_size = (max_nelmts_bits as usize).div_ceil(8);
|
|
||||||
let blk_off_val = (chunk_idx - n_inline) as u64;
|
|
||||||
aedb.extend_from_slice(&blk_off_val.to_le_bytes()[..blk_off_size]);
|
|
||||||
|
|
||||||
for slot in 0..nelmts {
|
|
||||||
if chunk_idx + slot < num_elements {
|
|
||||||
write_chunk_element(
|
|
||||||
&mut aedb,
|
|
||||||
&chunks[chunk_idx + slot],
|
|
||||||
offset_size,
|
|
||||||
has_filters,
|
|
||||||
chunk_size_bytes,
|
|
||||||
);
|
|
||||||
} else {
|
|
||||||
write_undefined_element(&mut aedb, offset_size, has_filters, chunk_size_bytes);
|
|
||||||
}
|
}
|
||||||
|
out.extend_from_slice(&bitmap);
|
||||||
}
|
}
|
||||||
|
for db in dblks {
|
||||||
let aedb_checksum = jenkins_lookup3(&aedb);
|
write_addr_opt(&mut out, db.addr);
|
||||||
aedb.extend_from_slice(&aedb_checksum.to_le_bytes());
|
}
|
||||||
|
let sum = jenkins_lookup3(&out[sb_start..]);
|
||||||
dblk_cursor += aedb.len() as u64;
|
out.extend_from_slice(&sum.to_le_bytes());
|
||||||
data_blocks_buf.extend_from_slice(&aedb);
|
for db in dblks.iter().filter(|d| d.addr.is_some()) {
|
||||||
chunk_idx += nelmts;
|
write_dblk(&mut out, db);
|
||||||
}
|
|
||||||
|
|
||||||
// Super block addresses (all undefined)
|
|
||||||
for _ in 0..n_sblk_addrs {
|
|
||||||
match offset_size {
|
|
||||||
4 => aeib.extend_from_slice(&u32::MAX.to_le_bytes()),
|
|
||||||
8 => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
_ => aeib.extend_from_slice(&u64::MAX.to_le_bytes()),
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
debug_assert_eq!(out.len() as u64, cursor - ea_base_address);
|
||||||
let aeib_checksum = jenkins_lookup3(&aeib);
|
out
|
||||||
aeib.extend_from_slice(&aeib_checksum.to_le_bytes());
|
|
||||||
debug_assert_eq!(aeib.len(), aeib_size);
|
|
||||||
|
|
||||||
let mut combined = aehd;
|
|
||||||
combined.extend_from_slice(&aeib);
|
|
||||||
combined.extend_from_slice(&data_blocks_buf);
|
|
||||||
combined
|
|
||||||
}
|
|
||||||
|
|
||||||
fn write_chunk_element(
|
|
||||||
buf: &mut Vec<u8>,
|
|
||||||
chunk: &WrittenChunk,
|
|
||||||
offset_size: u8,
|
|
||||||
has_filters: bool,
|
|
||||||
chunk_size_bytes: usize,
|
|
||||||
) {
|
|
||||||
match offset_size {
|
|
||||||
4 => buf.extend_from_slice(&(chunk.address as u32).to_le_bytes()),
|
|
||||||
8 => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
_ => buf.extend_from_slice(&chunk.address.to_le_bytes()),
|
|
||||||
}
|
|
||||||
if has_filters {
|
|
||||||
let cs_bytes = chunk.compressed_size.to_le_bytes();
|
|
||||||
buf.extend_from_slice(&cs_bytes[..chunk_size_bytes]);
|
|
||||||
buf.extend_from_slice(&chunk.filter_mask.to_le_bytes());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn write_undefined_element(
|
|
||||||
buf: &mut Vec<u8>,
|
|
||||||
offset_size: u8,
|
|
||||||
has_filters: bool,
|
|
||||||
chunk_size_bytes: usize,
|
|
||||||
) {
|
|
||||||
let os = offset_size as usize;
|
|
||||||
// Use extend with repeat to avoid heap-allocating a temporary Vec on each call.
|
|
||||||
buf.extend(core::iter::repeat_n(0xFF, os));
|
|
||||||
if has_filters {
|
|
||||||
buf.extend(core::iter::repeat_n(0x00, chunk_size_bytes));
|
|
||||||
buf.extend_from_slice(&0u32.to_le_bytes());
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -80,6 +80,9 @@ pub enum FormatError {
|
|||||||
InvalidLocalHeapSignature,
|
InvalidLocalHeapSignature,
|
||||||
/// Invalid local heap version.
|
/// Invalid local heap version.
|
||||||
InvalidLocalHeapVersion(u8),
|
InvalidLocalHeapVersion(u8),
|
||||||
|
/// A local heap's free list points outside its data segment (libhdf5:
|
||||||
|
/// "bad heap free list").
|
||||||
|
InvalidLocalHeapFreeList,
|
||||||
/// Invalid B-tree v1 signature.
|
/// Invalid B-tree v1 signature.
|
||||||
InvalidBTreeSignature,
|
InvalidBTreeSignature,
|
||||||
/// Invalid B-tree node type.
|
/// Invalid B-tree node type.
|
||||||
@@ -117,6 +120,14 @@ pub enum FormatError {
|
|||||||
/// A message is marked shared but was parsed without access to the file,
|
/// A message is marked shared but was parsed without access to the file,
|
||||||
/// so the reference to the real message could not be followed.
|
/// so the reference to the real message could not be followed.
|
||||||
UnresolvedSharedMessage,
|
UnresolvedSharedMessage,
|
||||||
|
/// A shared-message reference points at an object header that holds no
|
||||||
|
/// (unshared) message of the referenced type (raw message type id).
|
||||||
|
SharedMessageTargetMissing(u16),
|
||||||
|
/// A superblock was parsed at a non-zero offset of the buffer (the file
|
||||||
|
/// has a user block of this many bytes). HDF5 addresses are relative to
|
||||||
|
/// the superblock, so the buffer must start there: see
|
||||||
|
/// `signature::split_user_block`.
|
||||||
|
UserBlockNotStripped(u64),
|
||||||
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
/// A selection does not fit the dataset it was applied to (wrong rank, or
|
||||||
/// it reaches past a dimension's extent).
|
/// it reaches past a dimension's extent).
|
||||||
SelectionOutOfBounds(String),
|
SelectionOutOfBounds(String),
|
||||||
@@ -270,6 +281,9 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::InvalidLocalHeapSignature => {
|
FormatError::InvalidLocalHeapSignature => {
|
||||||
write!(f, "invalid local heap signature")
|
write!(f, "invalid local heap signature")
|
||||||
}
|
}
|
||||||
|
FormatError::InvalidLocalHeapFreeList => {
|
||||||
|
write!(f, "bad local heap free list")
|
||||||
|
}
|
||||||
FormatError::InvalidLocalHeapVersion(v) => {
|
FormatError::InvalidLocalHeapVersion(v) => {
|
||||||
write!(f, "invalid local heap version: {v}")
|
write!(f, "invalid local heap version: {v}")
|
||||||
}
|
}
|
||||||
@@ -339,6 +353,16 @@ impl fmt::Display for FormatError {
|
|||||||
FormatError::SelectionOutOfBounds(msg) => {
|
FormatError::SelectionOutOfBounds(msg) => {
|
||||||
write!(f, "selection out of bounds: {msg}")
|
write!(f, "selection out of bounds: {msg}")
|
||||||
}
|
}
|
||||||
|
FormatError::UserBlockNotStripped(n) => write!(
|
||||||
|
f,
|
||||||
|
"file has a {n}-byte user block: parse the bytes from the superblock on \
|
||||||
|
(signature::split_user_block)"
|
||||||
|
),
|
||||||
|
FormatError::SharedMessageTargetMissing(t) => write!(
|
||||||
|
f,
|
||||||
|
"shared message reference points at an object header with no message of type \
|
||||||
|
{t:#06x}"
|
||||||
|
),
|
||||||
FormatError::UnresolvedSharedMessage => write!(
|
FormatError::UnresolvedSharedMessage => write!(
|
||||||
f,
|
f,
|
||||||
"message is shared but no file data was available to resolve it"
|
"message is shared but no file data was available to resolve it"
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ extern crate alloc;
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{format, vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::chunk_grid::ChunkGrid;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
@@ -203,8 +204,7 @@ fn read_element(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
linear_index: usize,
|
linear_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Result<(Option<ChunkInfo>, usize), FormatError> {
|
) -> Result<(Option<ChunkInfo>, usize), FormatError> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
|
|
||||||
@@ -220,7 +220,10 @@ fn read_element(
|
|||||||
return Ok((None, os));
|
return Ok((None, os));
|
||||||
}
|
}
|
||||||
let address = read_offset(data, pos, offset_size)?;
|
let address = read_offset(data, pos, offset_size)?;
|
||||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
// A slot beyond the current extent is ignored, as the library does.
|
||||||
|
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||||
|
return Ok((None, os));
|
||||||
|
};
|
||||||
Ok((
|
Ok((
|
||||||
Some(ChunkInfo {
|
Some(ChunkInfo {
|
||||||
chunk_size: chunk_byte_size as u32,
|
chunk_size: chunk_byte_size as u32,
|
||||||
@@ -261,7 +264,9 @@ fn read_element(
|
|||||||
data[fm_off + 2],
|
data[fm_off + 2],
|
||||||
data[fm_off + 3],
|
data[fm_off + 3],
|
||||||
]);
|
]);
|
||||||
let offsets = index_to_chunk_offsets(linear_index, num_chunks_per_dim, chunk_dimensions);
|
let Some(offsets) = grid.offsets(linear_index as u64) else {
|
||||||
|
return Ok((None, elem_total));
|
||||||
|
};
|
||||||
Ok((
|
Ok((
|
||||||
Some(ChunkInfo {
|
Some(ChunkInfo {
|
||||||
chunk_size: chunk_size as u32,
|
chunk_size: chunk_size as u32,
|
||||||
@@ -274,27 +279,6 @@ fn read_element(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
|
||||||
fn index_to_chunk_offsets(
|
|
||||||
index: usize,
|
|
||||||
num_chunks_per_dim: &[u64],
|
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Vec<u64> {
|
|
||||||
let rank = num_chunks_per_dim.len();
|
|
||||||
let mut offsets = vec![0u64; rank];
|
|
||||||
let mut remaining = index as u64;
|
|
||||||
for d in (0..rank).rev() {
|
|
||||||
let nchunks = num_chunks_per_dim[d];
|
|
||||||
if nchunks == 0 {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
let chunk_idx = remaining % nchunks;
|
|
||||||
remaining /= nchunks;
|
|
||||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
|
||||||
}
|
|
||||||
offsets
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Collect elements from a data block at the given offset.
|
/// Collect elements from a data block at the given offset.
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
/// Layout of super block `u`, per the HDF5 spec: the number of data blocks it
|
/// Layout of super block `u`, per the HDF5 spec: the number of data blocks it
|
||||||
@@ -339,8 +323,7 @@ fn read_data_block_elements(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
start_index: usize,
|
start_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
page_init: &[u8],
|
page_init: &[u8],
|
||||||
first_page: usize,
|
first_page: usize,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
@@ -376,8 +359,7 @@ fn read_data_block_elements(
|
|||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
first_index + i,
|
first_index + i,
|
||||||
num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?;
|
)?;
|
||||||
if let Some(ci) = info {
|
if let Some(ci) = info {
|
||||||
chunks.push(ci);
|
chunks.push(ci);
|
||||||
@@ -449,25 +431,19 @@ pub fn read_extensible_array_chunks(
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &ExtensibleArrayHeader,
|
header: &ExtensibleArrayHeader,
|
||||||
dataset_dims: &[u64],
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
chunk_dimensions: &[u32],
|
chunk_dimensions: &[u32],
|
||||||
element_size: u32,
|
element_size: u32,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let rank = chunk_dimensions.len();
|
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
|
|
||||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
// Linear indexes follow the maximum dimensions, with the unlimited
|
||||||
for d in 0..rank {
|
// dimension swizzled to the slowest position (see `chunk_grid`).
|
||||||
let ch_dim = chunk_dimensions[d] as u64;
|
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||||
if ch_dim == 0 {
|
let grid = ChunkGrid::extensible_array(dataset_dims, max_dims, &dims_u64)?;
|
||||||
return Err(FormatError::ChunkedReadError(
|
let grid = &grid;
|
||||||
"chunk dimension is zero".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let ds_dim = dataset_dims[d];
|
|
||||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
|
||||||
}
|
|
||||||
|
|
||||||
let chunk_byte_size: u64 =
|
let chunk_byte_size: u64 =
|
||||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||||
@@ -557,8 +533,7 @@ pub fn read_extensible_array_chunks(
|
|||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
i,
|
i,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?;
|
)?;
|
||||||
if let Some(ci) = info {
|
if let Some(ci) = info {
|
||||||
chunks.push(ci);
|
chunks.push(ci);
|
||||||
@@ -594,8 +569,7 @@ pub fn read_extensible_array_chunks(
|
|||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_index,
|
global_index,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
&[],
|
&[],
|
||||||
0,
|
0,
|
||||||
)?);
|
)?);
|
||||||
@@ -625,8 +599,7 @@ pub fn read_extensible_array_chunks(
|
|||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_index,
|
global_index,
|
||||||
&num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
)?);
|
)?);
|
||||||
}
|
}
|
||||||
global_index =
|
global_index =
|
||||||
@@ -653,8 +626,7 @@ fn read_super_block(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
chunk_byte_size: u64,
|
chunk_byte_size: u64,
|
||||||
start_index: usize,
|
start_index: usize,
|
||||||
num_chunks_per_dim: &[u64],
|
grid: &ChunkGrid,
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let sb_header_size = 4 + 1 + 1 + os + arr_off_size(header);
|
let sb_header_size = 4 + 1 + 1 + os + arr_off_size(header);
|
||||||
@@ -710,8 +682,7 @@ fn read_super_block(
|
|||||||
offset_size,
|
offset_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
global_idx,
|
global_idx,
|
||||||
num_chunks_per_dim,
|
grid,
|
||||||
chunk_dimensions,
|
|
||||||
bitmap,
|
bitmap,
|
||||||
i * npages,
|
i * npages,
|
||||||
)?);
|
)?);
|
||||||
@@ -735,35 +706,18 @@ mod tests {
|
|||||||
}
|
}
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_1d() {
|
fn index_to_offsets_1d() {
|
||||||
let num_chunks = vec![5u64];
|
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||||
let chunk_dims = vec![20u32];
|
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![20]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
|
||||||
vec![80]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_2d() {
|
fn index_to_offsets_2d() {
|
||||||
let num_chunks = vec![3u64, 2];
|
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||||
let chunk_dims = vec![4u32, 3];
|
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||||
vec![0, 0]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![0, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 0]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -830,7 +784,7 @@ mod tests {
|
|||||||
index_block_address: (usize::MAX - 4) as u64,
|
index_block_address: (usize::MAX - 4) as u64,
|
||||||
};
|
};
|
||||||
let buf = vec![0u8; 64];
|
let buf = vec![0u8; 64];
|
||||||
let r = read_extensible_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_extensible_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -913,9 +867,17 @@ mod tests {
|
|||||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
||||||
let ds_dims = vec![40u64]; // 2 chunks × 20 elements
|
let ds_dims = vec![40u64]; // 2 chunks × 20 elements
|
||||||
let chunk_dims = vec![20u32];
|
let chunk_dims = vec![20u32];
|
||||||
let chunks =
|
let chunks = read_extensible_array_chunks(
|
||||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
&file_data,
|
||||||
.unwrap();
|
&header,
|
||||||
|
&ds_dims,
|
||||||
|
None,
|
||||||
|
&chunk_dims,
|
||||||
|
8,
|
||||||
|
os,
|
||||||
|
ls,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
assert_eq!(chunks.len(), 2);
|
assert_eq!(chunks.len(), 2);
|
||||||
assert_eq!(chunks[0].address, base_addr);
|
assert_eq!(chunks[0].address, base_addr);
|
||||||
@@ -1023,9 +985,17 @@ mod tests {
|
|||||||
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
let header = ExtensibleArrayHeader::parse(&file_data, aehd_offset, os, ls).unwrap();
|
||||||
let ds_dims = vec![40u64];
|
let ds_dims = vec![40u64];
|
||||||
let chunk_dims = vec![10u32];
|
let chunk_dims = vec![10u32];
|
||||||
let chunks =
|
let chunks = read_extensible_array_chunks(
|
||||||
read_extensible_array_chunks(&file_data, &header, &ds_dims, &chunk_dims, 8, os, ls)
|
&file_data,
|
||||||
.unwrap();
|
&header,
|
||||||
|
&ds_dims,
|
||||||
|
None,
|
||||||
|
&chunk_dims,
|
||||||
|
8,
|
||||||
|
os,
|
||||||
|
ls,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
assert_eq!(chunks.len(), 4);
|
assert_eq!(chunks.len(), 4);
|
||||||
for (i, c) in chunks.iter().enumerate() {
|
for (i, c) in chunks.iter().enumerate() {
|
||||||
@@ -1047,10 +1017,8 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn read_element_unallocated() {
|
fn read_element_unallocated() {
|
||||||
let data = vec![0xFFu8; 16];
|
let data = vec![0xFFu8; 16];
|
||||||
let num_chunks = vec![5u64];
|
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||||
let chunk_dims = vec![10u32];
|
let (info, consumed) = read_element(&data, 0, 0, 8, 8, 80, 0, &grid).unwrap();
|
||||||
let (info, consumed) =
|
|
||||||
read_element(&data, 0, 0, 8, 8, 80, 0, &num_chunks, &chunk_dims).unwrap();
|
|
||||||
assert!(info.is_none());
|
assert!(info.is_none());
|
||||||
assert_eq!(consumed, 8);
|
assert_eq!(consumed, 8);
|
||||||
}
|
}
|
||||||
@@ -1069,20 +1037,9 @@ mod tests {
|
|||||||
// Filter mask
|
// Filter mask
|
||||||
data[12..16].copy_from_slice(&0u32.to_le_bytes());
|
data[12..16].copy_from_slice(&0u32.to_le_bytes());
|
||||||
|
|
||||||
let num_chunks = vec![5u64];
|
let grid = ChunkGrid::fixed_array(&[50], None, &[10]).unwrap();
|
||||||
let chunk_dims = vec![10u32];
|
let (info, consumed) =
|
||||||
let (info, consumed) = read_element(
|
read_element(&data, 0, 1, elem_size as u8, os, 80, 2, &grid).unwrap();
|
||||||
&data,
|
|
||||||
0,
|
|
||||||
1,
|
|
||||||
elem_size as u8,
|
|
||||||
os,
|
|
||||||
80,
|
|
||||||
2,
|
|
||||||
&num_chunks,
|
|
||||||
&chunk_dims,
|
|
||||||
)
|
|
||||||
.unwrap();
|
|
||||||
let ci = info.unwrap();
|
let ci = info.unwrap();
|
||||||
assert_eq!(ci.address, 0x2000);
|
assert_eq!(ci.address, 0x2000);
|
||||||
assert_eq!(ci.chunk_size, 120);
|
assert_eq!(ci.chunk_size, 120);
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
//! link messages, contiguous datasets, inline and dense attributes.
|
//! link messages, contiguous datasets, inline and dense attributes.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{string::String, string::ToString, vec, vec::Vec};
|
use alloc::{format, string::String, string::ToString, vec, vec::Vec};
|
||||||
|
|
||||||
use crate::attribute::AttributeMessage;
|
use crate::attribute::AttributeMessage;
|
||||||
use crate::chunked_write::{
|
use crate::chunked_write::{
|
||||||
@@ -19,7 +19,7 @@ use crate::metadata_index::{DatasetMetadata, MetadataBlock, MetadataIndex};
|
|||||||
use crate::object_header_writer::ObjectHeaderWriter;
|
use crate::object_header_writer::ObjectHeaderWriter;
|
||||||
use crate::superblock::Superblock;
|
use crate::superblock::Superblock;
|
||||||
use crate::type_builders::{
|
use crate::type_builders::{
|
||||||
DatasetBuilder, FillTime, FinishedGroup, GroupBuilder, build_attr_message,
|
DatasetBuilder, FinishedGroup, GroupBuilder, build_attr_message, fill_value_message,
|
||||||
};
|
};
|
||||||
|
|
||||||
// Re-export public types that moved to type_builders for API compatibility.
|
// Re-export public types that moved to type_builders for API compatibility.
|
||||||
@@ -33,6 +33,49 @@ pub(crate) const OFFSET_SIZE: u8 = 8;
|
|||||||
pub(crate) const LENGTH_SIZE: u8 = 8;
|
pub(crate) const LENGTH_SIZE: u8 = 8;
|
||||||
const SUPERBLOCK_SIZE: usize = 48;
|
const SUPERBLOCK_SIZE: usize = 48;
|
||||||
|
|
||||||
|
/// Largest raw data a compact dataset can hold: the layout message (version,
|
||||||
|
/// class, 2-byte size, data) must fit an object header message, whose size
|
||||||
|
/// field is 2 bytes. Bigger "compact" requests fall back to contiguous storage.
|
||||||
|
const MAX_COMPACT_DATA_SIZE: usize = crate::object_header_writer::MAX_MESSAGE_SIZE - 4;
|
||||||
|
|
||||||
|
/// libhdf5's bounds on a file space page size (`H5F_FILE_SPACE_PAGE_SIZE_MIN`
|
||||||
|
/// and `_MAX`).
|
||||||
|
const MIN_FILE_SPACE_PAGE_SIZE: u32 = 512;
|
||||||
|
const MAX_FILE_SPACE_PAGE_SIZE: u32 = 1024 * 1024 * 1024;
|
||||||
|
|
||||||
|
/// Superblock extension object header for a file using the paged file-space
|
||||||
|
/// strategy: a single File Space Info message (0x0017), as libhdf5 writes it
|
||||||
|
/// for `fs_strategy="page"` without persisted free space.
|
||||||
|
fn build_paged_superblock_extension(page_size: u32) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let mut fsinfo = Vec::new();
|
||||||
|
fsinfo.push(1); // version
|
||||||
|
fsinfo.push(1); // strategy: H5F_FSPACE_STRATEGY_PAGE
|
||||||
|
fsinfo.push(0); // persisting free space: no
|
||||||
|
write_length(&mut fsinfo, 1, LENGTH_SIZE); // free-space section threshold
|
||||||
|
write_length(&mut fsinfo, u64::from(page_size), LENGTH_SIZE);
|
||||||
|
fsinfo.extend_from_slice(&0u16.to_le_bytes()); // page end metadata threshold
|
||||||
|
write_undef_offset(&mut fsinfo, OFFSET_SIZE); // EOA before free-space info
|
||||||
|
let mut w = ObjectHeaderWriter::new();
|
||||||
|
// Flags as libhdf5 sets them: bit 2 (never share) and bit 4 (mark if
|
||||||
|
// unknown). Not constant: libhdf5 rewrites the message when it closes a
|
||||||
|
// file it opened for writing.
|
||||||
|
w.add_message_with_flags(MessageType::Unknown(0x0017), fsinfo, 0x14);
|
||||||
|
w.serialize()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A group or dataset name must be one path component: not empty, not ".",
|
||||||
|
/// and without '/'. `FileWriter` writes a root group plus one level of
|
||||||
|
/// groups, and cannot create intermediate groups for a path.
|
||||||
|
fn check_link_name(name: &str) -> Result<(), FormatError> {
|
||||||
|
if name.is_empty() || name == "." || name.contains('/') {
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"invalid object name {name:?}: names must be a single path component \
|
||||||
|
(FileWriter does not create nested groups)"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Threshold for switching from compact (inline) to dense attribute storage.
|
/// Threshold for switching from compact (inline) to dense attribute storage.
|
||||||
const DENSE_ATTR_THRESHOLD: usize = 8;
|
const DENSE_ATTR_THRESHOLD: usize = 8;
|
||||||
|
|
||||||
@@ -50,12 +93,12 @@ pub(crate) fn build_chunked_dataset_oh(
|
|||||||
pipeline_message: Option<&[u8]>,
|
pipeline_message: Option<&[u8]>,
|
||||||
attrs: &[AttributeMessage],
|
attrs: &[AttributeMessage],
|
||||||
dense_blob: Option<&DenseAttrBlob>,
|
dense_blob: Option<&DenseAttrBlob>,
|
||||||
fill_time: FillTime,
|
fill_message: &[u8],
|
||||||
) -> Vec<u8> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
let mut w = ObjectHeaderWriter::new();
|
let mut w = ObjectHeaderWriter::new();
|
||||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||||
w.add_message_with_flags(MessageType::FillValue, vec![3, fill_time.to_byte()], 0x01);
|
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||||
w.add_message(MessageType::DataLayout, layout_message.to_vec());
|
w.add_message(MessageType::DataLayout, layout_message.to_vec());
|
||||||
if let Some(pm) = pipeline_message {
|
if let Some(pm) = pipeline_message {
|
||||||
w.add_message(MessageType::FilterPipeline, pm.to_vec());
|
w.add_message(MessageType::FilterPipeline, pm.to_vec());
|
||||||
@@ -77,12 +120,12 @@ pub(crate) fn build_dataset_oh(
|
|||||||
data_size: u64,
|
data_size: u64,
|
||||||
attrs: &[AttributeMessage],
|
attrs: &[AttributeMessage],
|
||||||
dense_blob: Option<&DenseAttrBlob>,
|
dense_blob: Option<&DenseAttrBlob>,
|
||||||
fill_time: FillTime,
|
fill_message: &[u8],
|
||||||
) -> Vec<u8> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
let mut w = ObjectHeaderWriter::new();
|
let mut w = ObjectHeaderWriter::new();
|
||||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||||
w.add_message_with_flags(MessageType::FillValue, vec![3, fill_time.to_byte()], 0x01);
|
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||||
let mut dl = Vec::new();
|
let mut dl = Vec::new();
|
||||||
dl.push(4); // version
|
dl.push(4); // version
|
||||||
dl.push(1); // class = contiguous
|
dl.push(1); // class = contiguous
|
||||||
@@ -112,12 +155,12 @@ pub(crate) fn build_compact_dataset_oh(
|
|||||||
data: &[u8],
|
data: &[u8],
|
||||||
attrs: &[AttributeMessage],
|
attrs: &[AttributeMessage],
|
||||||
dense_blob: Option<&DenseAttrBlob>,
|
dense_blob: Option<&DenseAttrBlob>,
|
||||||
fill_time: FillTime,
|
fill_message: &[u8],
|
||||||
) -> Vec<u8> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
let mut w = ObjectHeaderWriter::new();
|
let mut w = ObjectHeaderWriter::new();
|
||||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||||
w.add_message_with_flags(MessageType::FillValue, vec![3, fill_time.to_byte()], 0x01);
|
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||||
// Compact layout message: version=4, class=0, u16 size, inline data
|
// Compact layout message: version=4, class=0, u16 size, inline data
|
||||||
let mut dl = Vec::new();
|
let mut dl = Vec::new();
|
||||||
dl.push(4); // version
|
dl.push(4); // version
|
||||||
@@ -140,7 +183,7 @@ pub(crate) fn build_group_oh(
|
|||||||
dense_link_info: Option<&[u8]>,
|
dense_link_info: Option<&[u8]>,
|
||||||
attrs: &[AttributeMessage],
|
attrs: &[AttributeMessage],
|
||||||
dense_blob: Option<&DenseAttrBlob>,
|
dense_blob: Option<&DenseAttrBlob>,
|
||||||
) -> Vec<u8> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
let mut w = ObjectHeaderWriter::new();
|
let mut w = ObjectHeaderWriter::new();
|
||||||
if let Some(li) = dense_link_info {
|
if let Some(li) = dense_link_info {
|
||||||
// Dense link storage: a LinkInfo pointing at the fractal heap + name
|
// Dense link storage: a LinkInfo pointing at the fractal heap + name
|
||||||
@@ -902,12 +945,12 @@ pub(crate) fn build_vds_dataset_oh(
|
|||||||
global_heap_addr: u64,
|
global_heap_addr: u64,
|
||||||
attrs: &[AttributeMessage],
|
attrs: &[AttributeMessage],
|
||||||
dense_blob: Option<&DenseAttrBlob>,
|
dense_blob: Option<&DenseAttrBlob>,
|
||||||
fill_time: FillTime,
|
fill_message: &[u8],
|
||||||
) -> Vec<u8> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
let mut w = ObjectHeaderWriter::new();
|
let mut w = ObjectHeaderWriter::new();
|
||||||
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
w.add_message_with_flags(MessageType::Datatype, dt.serialize(), 0x01);
|
||||||
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
w.add_message(MessageType::Dataspace, ds.serialize(LENGTH_SIZE));
|
||||||
w.add_message_with_flags(MessageType::FillValue, vec![3, fill_time.to_byte()], 0x01);
|
w.add_message_with_flags(MessageType::FillValue, fill_message.to_vec(), 0x01);
|
||||||
// VDS layout message: version=4, class=3, global_heap_address(8), global_heap_index=1(4)
|
// VDS layout message: version=4, class=3, global_heap_address(8), global_heap_index=1(4)
|
||||||
let mut dl = Vec::new();
|
let mut dl = Vec::new();
|
||||||
dl.push(4u8); // version
|
dl.push(4u8); // version
|
||||||
@@ -956,7 +999,9 @@ pub struct FileWriter {
|
|||||||
alignment_threshold: usize,
|
alignment_threshold: usize,
|
||||||
/// Global alignment boundary in bytes (0 = disabled).
|
/// Global alignment boundary in bytes (0 = disabled).
|
||||||
alignment_bytes: usize,
|
alignment_bytes: usize,
|
||||||
/// Page size for page-buffer mode. When set, a v4 superblock is written.
|
/// File space page size. When set, the file uses libhdf5's paged
|
||||||
|
/// file-space strategy (a File Space Info message in the superblock
|
||||||
|
/// extension).
|
||||||
page_size: Option<u32>,
|
page_size: Option<u32>,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -988,9 +1033,16 @@ impl FileWriter {
|
|||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Enable page-buffer mode with the given page size. Writing this causes
|
/// Write the file with libhdf5's *paged* file-space strategy and the given
|
||||||
/// the file to be written with a v4 superblock (page_size field) instead
|
/// page size, as `H5Pset_file_space_strategy(H5F_FSPACE_STRATEGY_PAGE)` +
|
||||||
/// of the default v3.
|
/// `H5Pset_file_space_page_size` (h5py: `fs_strategy="page"`,
|
||||||
|
/// `fs_page_size=...`) do: a v3 superblock with an extension holding a
|
||||||
|
/// File Space Info message, and the file padded to a whole number of
|
||||||
|
/// pages. Readers with a page buffer can then fetch metadata page by page.
|
||||||
|
///
|
||||||
|
/// `page_size` must be between 512 bytes and 1 GiB (libhdf5's limits);
|
||||||
|
/// [`Self::finish`] fails otherwise. This used to write a "version 4"
|
||||||
|
/// superblock, which does not exist and no HDF5 library can open.
|
||||||
pub fn with_page_size(&mut self, page_size: u32) -> &mut Self {
|
pub fn with_page_size(&mut self, page_size: u32) -> &mut Self {
|
||||||
self.page_size = Some(page_size);
|
self.page_size = Some(page_size);
|
||||||
self
|
self
|
||||||
@@ -1015,6 +1067,14 @@ impl FileWriter {
|
|||||||
|
|
||||||
pub fn finish(self) -> Result<Vec<u8>, FormatError> {
|
pub fn finish(self) -> Result<Vec<u8>, FormatError> {
|
||||||
let page_size = self.page_size;
|
let page_size = self.page_size;
|
||||||
|
if let Some(ps) = page_size
|
||||||
|
&& !(MIN_FILE_SPACE_PAGE_SIZE..=MAX_FILE_SPACE_PAGE_SIZE).contains(&ps)
|
||||||
|
{
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"file space page size {ps} is outside libhdf5's \
|
||||||
|
{MIN_FILE_SPACE_PAGE_SIZE}..={MAX_FILE_SPACE_PAGE_SIZE} bytes"
|
||||||
|
)));
|
||||||
|
}
|
||||||
struct DsFlat {
|
struct DsFlat {
|
||||||
name: String,
|
name: String,
|
||||||
dt: Datatype,
|
dt: Datatype,
|
||||||
@@ -1023,7 +1083,8 @@ impl FileWriter {
|
|||||||
attrs: Vec<AttributeMessage>,
|
attrs: Vec<AttributeMessage>,
|
||||||
chunk_options: ChunkOptions,
|
chunk_options: ChunkOptions,
|
||||||
maxshape: Option<Vec<u64>>,
|
maxshape: Option<Vec<u64>>,
|
||||||
fill_time: FillTime,
|
/// Serialized Fill Value message.
|
||||||
|
fill_message: Vec<u8>,
|
||||||
compact: bool,
|
compact: bool,
|
||||||
alignment: usize,
|
alignment: usize,
|
||||||
/// VDS source mappings (set for Virtual datasets).
|
/// VDS source mappings (set for Virtual datasets).
|
||||||
@@ -1073,6 +1134,7 @@ impl FileWriter {
|
|||||||
};
|
};
|
||||||
attrs.extend(p.build_attrs(&raw));
|
attrs.extend(p.build_attrs(&raw));
|
||||||
}
|
}
|
||||||
|
let fill_message = fill_value_message(db.fill_time, db.fill_value.as_deref(), &dt)?;
|
||||||
Ok(DsFlat {
|
Ok(DsFlat {
|
||||||
name: db.name,
|
name: db.name,
|
||||||
dt,
|
dt,
|
||||||
@@ -1081,13 +1143,26 @@ impl FileWriter {
|
|||||||
attrs,
|
attrs,
|
||||||
chunk_options: db.chunk_options,
|
chunk_options: db.chunk_options,
|
||||||
maxshape: db.maxshape,
|
maxshape: db.maxshape,
|
||||||
fill_time: db.fill_time,
|
fill_message,
|
||||||
compact: db.compact,
|
compact: db.compact,
|
||||||
alignment: db.alignment,
|
alignment: db.alignment,
|
||||||
virtual_sources: db.virtual_sources,
|
virtual_sources: db.virtual_sources,
|
||||||
})
|
})
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Every name becomes a single link in its parent group. The writer
|
||||||
|
// has no nested groups, so a path like "a/b" would be stored as one
|
||||||
|
// link literally named "a/b" — which no HDF5 reader can resolve.
|
||||||
|
let root_names = self.root_datasets.iter().map(|d| d.name.as_str());
|
||||||
|
let group_names = self.groups.iter().flat_map(|g| {
|
||||||
|
core::iter::once(g.name.as_str())
|
||||||
|
.chain(g.datasets.iter().map(|d| d.name.as_str()))
|
||||||
|
.chain(g.external_links.iter().map(|l| l.0.as_str()))
|
||||||
|
});
|
||||||
|
for name in root_names.chain(group_names) {
|
||||||
|
check_link_name(name)?;
|
||||||
|
}
|
||||||
|
|
||||||
let mut all_ds: Vec<DsFlat> = Vec::new();
|
let mut all_ds: Vec<DsFlat> = Vec::new();
|
||||||
let mut groups: Vec<GrpFlat> = Vec::new();
|
let mut groups: Vec<GrpFlat> = Vec::new();
|
||||||
let mut root_ds_indices: Vec<usize> = Vec::new();
|
let mut root_ds_indices: Vec<usize> = Vec::new();
|
||||||
@@ -1120,17 +1195,35 @@ impl FileWriter {
|
|||||||
root_attrs.push(build_attr_message(n, v));
|
root_attrs.push(build_attr_message(n, v));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Every datatype must have an on-disk encoding before anything is laid
|
||||||
|
// out: `Datatype::serialize` itself cannot report a failure.
|
||||||
|
let group_attrs = groups.iter().flat_map(|g| &g.attrs);
|
||||||
|
let ds_attrs = all_ds.iter().flat_map(|d| &d.attrs);
|
||||||
|
for a in root_attrs.iter().chain(group_attrs).chain(ds_attrs) {
|
||||||
|
a.datatype.check_encodable()?;
|
||||||
|
}
|
||||||
|
for d in &all_ds {
|
||||||
|
d.dt.check_encodable()?;
|
||||||
|
}
|
||||||
|
|
||||||
let is_vds: Vec<bool> = all_ds.iter().map(|d| d.virtual_sources.is_some()).collect();
|
let is_vds: Vec<bool> = all_ds.iter().map(|d| d.virtual_sources.is_some()).collect();
|
||||||
let is_chunked: Vec<bool> = all_ds
|
let is_chunked: Vec<bool> = all_ds
|
||||||
.iter()
|
.iter()
|
||||||
.enumerate()
|
.enumerate()
|
||||||
.map(|(i, d)| !is_vds[i] && (d.chunk_options.is_chunked() || d.maxshape.is_some()))
|
.map(|(i, d)| {
|
||||||
|
// Only a dataset that can grow needs chunks; a maxshape equal
|
||||||
|
// to the shape is as fixed as no maxshape at all.
|
||||||
|
let resizable = d.maxshape.as_ref().is_some_and(|m| *m != d.ds.dimensions);
|
||||||
|
!is_vds[i] && (d.chunk_options.is_chunked() || resizable)
|
||||||
|
})
|
||||||
.collect();
|
.collect();
|
||||||
// Determine which datasets use compact storage
|
// Determine which datasets use compact storage
|
||||||
let is_compact: Vec<bool> = all_ds
|
let is_compact: Vec<bool> = all_ds
|
||||||
.iter()
|
.iter()
|
||||||
.enumerate()
|
.enumerate()
|
||||||
.map(|(i, d)| !is_vds[i] && !is_chunked[i] && d.compact && d.raw.len() <= 65535)
|
.map(|(i, d)| {
|
||||||
|
!is_vds[i] && !is_chunked[i] && d.compact && d.raw.len() <= MAX_COMPACT_DATA_SIZE
|
||||||
|
})
|
||||||
.collect();
|
.collect();
|
||||||
let root_dense = root_attrs.len() > DENSE_ATTR_THRESHOLD;
|
let root_dense = root_attrs.len() > DENSE_ATTR_THRESHOLD;
|
||||||
let group_dense: Vec<bool> = groups
|
let group_dense: Vec<bool> = groups
|
||||||
@@ -1169,9 +1262,9 @@ impl FileWriter {
|
|||||||
}
|
}
|
||||||
let attr_blob = group_dense[gi].then(|| build_dense_attrs(&g.attrs, 0));
|
let attr_blob = group_dense[gi].then(|| build_dense_attrs(&g.attrs, 0));
|
||||||
let dl = group_links_dense[gi].then_some(dummy_link_info.as_slice());
|
let dl = group_links_dense[gi].then_some(dummy_link_info.as_slice());
|
||||||
build_group_oh(&dummy_links, dl, &g.attrs, attr_blob.as_ref()).len()
|
build_group_oh(&dummy_links, dl, &g.attrs, attr_blob.as_ref()).map(|oh| oh.len())
|
||||||
})
|
})
|
||||||
.collect();
|
.collect::<Result<_, _>>()?;
|
||||||
|
|
||||||
let root_dummy_links: Vec<LinkMessage> = {
|
let root_dummy_links: Vec<LinkMessage> = {
|
||||||
let mut links = Vec::new();
|
let mut links = Vec::new();
|
||||||
@@ -1186,7 +1279,7 @@ impl FileWriter {
|
|||||||
let root_oh_size = {
|
let root_oh_size = {
|
||||||
let attr_blob = root_dense.then(|| build_dense_attrs(&root_attrs, 0));
|
let attr_blob = root_dense.then(|| build_dense_attrs(&root_attrs, 0));
|
||||||
let dl = root_links_dense.then_some(dummy_link_info.as_slice());
|
let dl = root_links_dense.then_some(dummy_link_info.as_slice());
|
||||||
build_group_oh(&root_dummy_links, dl, &root_attrs, attr_blob.as_ref()).len()
|
build_group_oh(&root_dummy_links, dl, &root_attrs, attr_blob.as_ref())?.len()
|
||||||
};
|
};
|
||||||
|
|
||||||
struct DataBlob {
|
struct DataBlob {
|
||||||
@@ -1214,8 +1307,8 @@ impl FileWriter {
|
|||||||
0, // dummy address
|
0, // dummy address
|
||||||
&d.attrs,
|
&d.attrs,
|
||||||
dense_blob.as_ref(),
|
dense_blob.as_ref(),
|
||||||
d.fill_time,
|
&d.fill_message,
|
||||||
);
|
)?;
|
||||||
// Global heap blob size is address-independent; compute it now
|
// Global heap blob size is address-independent; compute it now
|
||||||
// so pass 2 can place it correctly.
|
// so pass 2 can place it correctly.
|
||||||
let vds_mappings = d.virtual_sources.as_deref().unwrap_or(&[]);
|
let vds_mappings = d.virtual_sources.as_deref().unwrap_or(&[]);
|
||||||
@@ -1244,7 +1337,7 @@ impl FileWriter {
|
|||||||
&pre,
|
&pre,
|
||||||
dummy_cursor,
|
dummy_cursor,
|
||||||
d.maxshape.as_deref(),
|
d.maxshape.as_deref(),
|
||||||
);
|
)?;
|
||||||
dummy_cursor += result.data_bytes.len() as u64;
|
dummy_cursor += result.data_bytes.len() as u64;
|
||||||
let dense_blob = if ds_dense[i] {
|
let dense_blob = if ds_dense[i] {
|
||||||
Some(build_dense_attrs(&d.attrs, 0))
|
Some(build_dense_attrs(&d.attrs, 0))
|
||||||
@@ -1258,8 +1351,8 @@ impl FileWriter {
|
|||||||
result.pipeline_message.as_deref(),
|
result.pipeline_message.as_deref(),
|
||||||
&d.attrs,
|
&d.attrs,
|
||||||
dense_blob.as_ref(),
|
dense_blob.as_ref(),
|
||||||
d.fill_time,
|
&d.fill_message,
|
||||||
);
|
)?;
|
||||||
dummy_blobs.push(DataBlob {
|
dummy_blobs.push(DataBlob {
|
||||||
data: result.data_bytes,
|
data: result.data_bytes,
|
||||||
oh_bytes: oh,
|
oh_bytes: oh,
|
||||||
@@ -1277,8 +1370,8 @@ impl FileWriter {
|
|||||||
&d.raw,
|
&d.raw,
|
||||||
&d.attrs,
|
&d.attrs,
|
||||||
dense_blob.as_ref(),
|
dense_blob.as_ref(),
|
||||||
d.fill_time,
|
&d.fill_message,
|
||||||
);
|
)?;
|
||||||
dummy_blobs.push(DataBlob {
|
dummy_blobs.push(DataBlob {
|
||||||
data: vec![],
|
data: vec![],
|
||||||
oh_bytes: oh,
|
oh_bytes: oh,
|
||||||
@@ -1297,8 +1390,8 @@ impl FileWriter {
|
|||||||
d.raw.len() as u64,
|
d.raw.len() as u64,
|
||||||
&d.attrs,
|
&d.attrs,
|
||||||
dense_blob.as_ref(),
|
dense_blob.as_ref(),
|
||||||
d.fill_time,
|
&d.fill_message,
|
||||||
);
|
)?;
|
||||||
dummy_blobs.push(DataBlob {
|
dummy_blobs.push(DataBlob {
|
||||||
data: d.raw.clone(),
|
data: d.raw.clone(),
|
||||||
oh_bytes: oh,
|
oh_bytes: oh,
|
||||||
@@ -1310,12 +1403,12 @@ impl FileWriter {
|
|||||||
let actual_ds_oh_sizes: Vec<usize> = dummy_blobs.iter().map(|b| b.oh_bytes.len()).collect();
|
let actual_ds_oh_sizes: Vec<usize> = dummy_blobs.iter().map(|b| b.oh_bytes.len()).collect();
|
||||||
|
|
||||||
// Pass 2: compute real addresses
|
// Pass 2: compute real addresses
|
||||||
// v4 superblocks add a 4-byte page_size field before the checksum.
|
// A paged file carries its File Space Info in a superblock extension
|
||||||
let superblock_size = if page_size.is_some() {
|
// object header, placed right after the superblock.
|
||||||
SUPERBLOCK_SIZE + 4
|
let sb_ext = page_size
|
||||||
} else {
|
.map(build_paged_superblock_extension)
|
||||||
SUPERBLOCK_SIZE
|
.transpose()?;
|
||||||
};
|
let superblock_size = SUPERBLOCK_SIZE + sb_ext.as_ref().map_or(0, Vec::len);
|
||||||
let root_group_addr = superblock_size as u64;
|
let root_group_addr = superblock_size as u64;
|
||||||
let mut cursor2 = superblock_size + root_oh_size;
|
let mut cursor2 = superblock_size + root_oh_size;
|
||||||
|
|
||||||
@@ -1406,8 +1499,8 @@ impl FileWriter {
|
|||||||
heap_addr,
|
heap_addr,
|
||||||
&d.attrs,
|
&d.attrs,
|
||||||
ds_dense_blobs[i].as_ref(),
|
ds_dense_blobs[i].as_ref(),
|
||||||
d.fill_time,
|
&d.fill_message,
|
||||||
);
|
)?;
|
||||||
ds_blobs2.push(DataBlob {
|
ds_blobs2.push(DataBlob {
|
||||||
data: gcol_bytes.clone(),
|
data: gcol_bytes.clone(),
|
||||||
oh_bytes: oh,
|
oh_bytes: oh,
|
||||||
@@ -1424,7 +1517,7 @@ impl FileWriter {
|
|||||||
.expect("chunked dataset missing precompressed cache"),
|
.expect("chunked dataset missing precompressed cache"),
|
||||||
base_address,
|
base_address,
|
||||||
d.maxshape.as_deref(),
|
d.maxshape.as_deref(),
|
||||||
);
|
)?;
|
||||||
cursor2 += result.data_bytes.len();
|
cursor2 += result.data_bytes.len();
|
||||||
let oh = build_chunked_dataset_oh(
|
let oh = build_chunked_dataset_oh(
|
||||||
&d.dt,
|
&d.dt,
|
||||||
@@ -1433,8 +1526,8 @@ impl FileWriter {
|
|||||||
result.pipeline_message.as_deref(),
|
result.pipeline_message.as_deref(),
|
||||||
&d.attrs,
|
&d.attrs,
|
||||||
ds_dense_blobs[i].as_ref(),
|
ds_dense_blobs[i].as_ref(),
|
||||||
d.fill_time,
|
&d.fill_message,
|
||||||
);
|
)?;
|
||||||
ds_blobs2.push(DataBlob {
|
ds_blobs2.push(DataBlob {
|
||||||
data: result.data_bytes,
|
data: result.data_bytes,
|
||||||
oh_bytes: oh,
|
oh_bytes: oh,
|
||||||
@@ -1448,8 +1541,8 @@ impl FileWriter {
|
|||||||
&d.raw,
|
&d.raw,
|
||||||
&d.attrs,
|
&d.attrs,
|
||||||
ds_dense_blobs[i].as_ref(),
|
ds_dense_blobs[i].as_ref(),
|
||||||
d.fill_time,
|
&d.fill_message,
|
||||||
);
|
)?;
|
||||||
ds_blobs2.push(DataBlob {
|
ds_blobs2.push(DataBlob {
|
||||||
data: vec![],
|
data: vec![],
|
||||||
oh_bytes: oh,
|
oh_bytes: oh,
|
||||||
@@ -1473,8 +1566,8 @@ impl FileWriter {
|
|||||||
d.raw.len() as u64,
|
d.raw.len() as u64,
|
||||||
&d.attrs,
|
&d.attrs,
|
||||||
ds_dense_blobs[i].as_ref(),
|
ds_dense_blobs[i].as_ref(),
|
||||||
d.fill_time,
|
&d.fill_message,
|
||||||
);
|
)?;
|
||||||
let mut data = vec![0u8; padding];
|
let mut data = vec![0u8; padding];
|
||||||
data.extend_from_slice(&d.raw);
|
data.extend_from_slice(&d.raw);
|
||||||
cursor2 += d.raw.len();
|
cursor2 += d.raw.len();
|
||||||
@@ -1489,11 +1582,16 @@ impl FileWriter {
|
|||||||
let actual_ds_oh_sizes2: Vec<usize> = ds_blobs2.iter().map(|b| b.oh_bytes.len()).collect();
|
let actual_ds_oh_sizes2: Vec<usize> = ds_blobs2.iter().map(|b| b.oh_bytes.len()).collect();
|
||||||
debug_assert_eq!(actual_ds_oh_sizes, actual_ds_oh_sizes2);
|
debug_assert_eq!(actual_ds_oh_sizes, actual_ds_oh_sizes2);
|
||||||
|
|
||||||
|
// libhdf5 ends a paged file on a page boundary.
|
||||||
|
let data_end = cursor2;
|
||||||
|
if let Some(ps) = page_size {
|
||||||
|
cursor2 = cursor2.next_multiple_of(ps as usize);
|
||||||
|
}
|
||||||
let eof_addr2 = cursor2 as u64;
|
let eof_addr2 = cursor2 as u64;
|
||||||
let mut buf = Vec::with_capacity(cursor2);
|
let mut buf = Vec::with_capacity(cursor2);
|
||||||
|
|
||||||
let sb = Superblock {
|
let sb = Superblock {
|
||||||
version: if page_size.is_some() { 4 } else { 3 },
|
version: 3,
|
||||||
offset_size: OFFSET_SIZE,
|
offset_size: OFFSET_SIZE,
|
||||||
length_size: LENGTH_SIZE,
|
length_size: LENGTH_SIZE,
|
||||||
base_address: 0,
|
base_address: 0,
|
||||||
@@ -1505,11 +1603,18 @@ impl FileWriter {
|
|||||||
free_space_address: None,
|
free_space_address: None,
|
||||||
driver_info_address: None,
|
driver_info_address: None,
|
||||||
consistency_flags: 0,
|
consistency_flags: 0,
|
||||||
superblock_extension_address: Some(u64::MAX),
|
superblock_extension_address: Some(if sb_ext.is_some() {
|
||||||
|
SUPERBLOCK_SIZE as u64
|
||||||
|
} else {
|
||||||
|
u64::MAX
|
||||||
|
}),
|
||||||
checksum: None,
|
checksum: None,
|
||||||
page_size,
|
page_size: None,
|
||||||
};
|
};
|
||||||
buf.extend_from_slice(&sb.serialize());
|
buf.extend_from_slice(&sb.serialize());
|
||||||
|
if let Some(ref ext) = sb_ext {
|
||||||
|
buf.extend_from_slice(ext);
|
||||||
|
}
|
||||||
|
|
||||||
// Root group OH
|
// Root group OH
|
||||||
let mut root_links: Vec<LinkMessage> = Vec::new();
|
let mut root_links: Vec<LinkMessage> = Vec::new();
|
||||||
@@ -1530,7 +1635,7 @@ impl FileWriter {
|
|||||||
root_dl,
|
root_dl,
|
||||||
&root_attrs,
|
&root_attrs,
|
||||||
root_dense_blob.as_ref(),
|
root_dense_blob.as_ref(),
|
||||||
));
|
)?);
|
||||||
if let Some(ref b) = root_link_blob {
|
if let Some(ref b) = root_link_blob {
|
||||||
buf.extend_from_slice(&b.blob);
|
buf.extend_from_slice(&b.blob);
|
||||||
}
|
}
|
||||||
@@ -1555,7 +1660,7 @@ impl FileWriter {
|
|||||||
dl,
|
dl,
|
||||||
&g.attrs,
|
&g.attrs,
|
||||||
group_dense_blobs[gi].as_ref(),
|
group_dense_blobs[gi].as_ref(),
|
||||||
));
|
)?);
|
||||||
if let Some(ref b) = link_blob {
|
if let Some(ref b) = link_blob {
|
||||||
buf.extend_from_slice(&b.blob);
|
buf.extend_from_slice(&b.blob);
|
||||||
}
|
}
|
||||||
@@ -1577,7 +1682,8 @@ impl FileWriter {
|
|||||||
buf.extend_from_slice(&blob.data);
|
buf.extend_from_slice(&blob.data);
|
||||||
}
|
}
|
||||||
|
|
||||||
debug_assert_eq!(buf.len(), cursor2);
|
debug_assert_eq!(buf.len(), data_end);
|
||||||
|
buf.resize(cursor2, 0);
|
||||||
Ok(buf)
|
Ok(buf)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1841,6 +1947,12 @@ mod tests {
|
|||||||
root_block_address: 0,
|
root_block_address: 0,
|
||||||
current_rows_in_root_indirect_block: 0,
|
current_rows_in_root_indirect_block: 0,
|
||||||
managed_objects_count: 0,
|
managed_objects_count: 0,
|
||||||
|
huge_btree_address: u64::MAX,
|
||||||
|
filter_pipeline: None,
|
||||||
|
root_direct_block_filtered_size: 0,
|
||||||
|
root_direct_block_filter_mask: 0,
|
||||||
|
offset_size: 8,
|
||||||
|
length_size: 8,
|
||||||
};
|
};
|
||||||
let (off, len) = fh.decode_managed_id(&id).unwrap();
|
let (off, len) = fh.decode_managed_id(&id).unwrap();
|
||||||
assert_eq!(off, 100);
|
assert_eq!(off, 100);
|
||||||
@@ -2151,7 +2263,8 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn file_writer_v4_superblock() {
|
fn file_writer_paged_file_uses_v3_superblock_and_fsinfo_extension() {
|
||||||
|
// This used to write superblock "version 4", which does not exist.
|
||||||
let mut fw = FileWriter::new();
|
let mut fw = FileWriter::new();
|
||||||
fw.with_page_size(4096);
|
fw.with_page_size(4096);
|
||||||
fw.create_dataset("data").with_f64_data(&[1.0, 2.0]);
|
fw.create_dataset("data").with_f64_data(&[1.0, 2.0]);
|
||||||
@@ -2159,8 +2272,31 @@ mod tests {
|
|||||||
|
|
||||||
let sig = signature::find_signature(&bytes).unwrap();
|
let sig = signature::find_signature(&bytes).unwrap();
|
||||||
let sb = Superblock::parse(&bytes, sig).unwrap();
|
let sb = Superblock::parse(&bytes, sig).unwrap();
|
||||||
assert_eq!(sb.version, 4, "expected superblock v4");
|
assert_eq!(sb.version, 3);
|
||||||
assert_eq!(sb.page_size, Some(4096));
|
assert_eq!(sb.superblock_extension_address, Some(48));
|
||||||
|
assert_eq!(bytes.len() % 4096, 0);
|
||||||
|
assert_eq!(sb.eof_address, bytes.len() as u64);
|
||||||
|
let ext = ObjectHeader::parse(&bytes, 48, 8, 8).unwrap();
|
||||||
|
let fsinfo = &ext.messages[0];
|
||||||
|
assert_eq!(fsinfo.msg_type, MessageType::Unknown(0x0017));
|
||||||
|
// Byte-for-byte what HDF5 2.0 writes for fs_strategy="page",
|
||||||
|
// fs_page_size=4096.
|
||||||
|
let mut expected = vec![1u8, 1, 0, 1, 0, 0, 0, 0, 0, 0, 0];
|
||||||
|
expected.extend_from_slice(&4096u64.to_le_bytes());
|
||||||
|
expected.extend_from_slice(&[0, 0]);
|
||||||
|
expected.extend_from_slice(&[0xff; 8]);
|
||||||
|
assert_eq!(fsinfo.data, expected);
|
||||||
|
assert_eq!(fsinfo.flags, 0x14);
|
||||||
|
assert_eq!(read_dataset_f64(&bytes, "data"), vec![1.0, 2.0]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn file_writer_rejects_page_sizes_libhdf5_would() {
|
||||||
|
for ps in [0u32, 511, MAX_FILE_SPACE_PAGE_SIZE + 1] {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.with_page_size(ps);
|
||||||
|
assert!(fw.finish().is_err(), "page size {ps}");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -98,15 +98,50 @@ pub fn parse_fill_value(msg: &HeaderMessage) -> Result<Option<Vec<u8>>, FormatEr
|
|||||||
|
|
||||||
/// The fill value that applies to a dataset given its header messages. The new
|
/// The fill value that applies to a dataset given its header messages. The new
|
||||||
/// message wins over the old one when both are present.
|
/// message wins over the old one when both are present.
|
||||||
|
///
|
||||||
|
/// A *shared* fill value message holds only a reference to the real message,
|
||||||
|
/// which cannot be followed without the file: this returns
|
||||||
|
/// [`FormatError::UnresolvedSharedMessage`] for one (it used to answer "zeros").
|
||||||
|
/// Use [`dataset_fill_value_in`] when the file bytes are at hand.
|
||||||
pub fn dataset_fill_value(messages: &[HeaderMessage]) -> Result<Option<Vec<u8>>, FormatError> {
|
pub fn dataset_fill_value(messages: &[HeaderMessage]) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
|
fill_value_from(messages, |_| Err(FormatError::UnresolvedSharedMessage))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`dataset_fill_value`] for a dataset in `file_data`, following a shared
|
||||||
|
/// fill value message to where it lives: another object header, or the
|
||||||
|
/// file's shared-message (SOHM) heap, as libhdf5 writes it when the file has
|
||||||
|
/// a SOHM index for fill values.
|
||||||
|
pub fn dataset_fill_value_in(
|
||||||
|
file_data: &[u8],
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
|
fill_value_from(messages, |msg| {
|
||||||
|
crate::shared_message::message_data_with_sohm(file_data, msg, offset_size, length_size)
|
||||||
|
.map(|data| data.into_owned())
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fill_value_from(
|
||||||
|
messages: &[HeaderMessage],
|
||||||
|
resolve_shared: impl Fn(&HeaderMessage) -> Result<Vec<u8>, FormatError>,
|
||||||
|
) -> Result<Option<Vec<u8>>, FormatError> {
|
||||||
for wanted in [MessageType::FillValue, MessageType::FillValueOld] {
|
for wanted in [MessageType::FillValue, MessageType::FillValueOld] {
|
||||||
if let Some(msg) = messages.iter().find(|m| m.msg_type == wanted) {
|
if let Some(msg) = messages.iter().find(|m| m.msg_type == wanted) {
|
||||||
if crate::shared_message::is_shared(msg.flags) {
|
let value = if crate::shared_message::is_shared(msg.flags) {
|
||||||
// A shared fill value is legal but vanishingly rare; treat it
|
let data = resolve_shared(msg)?;
|
||||||
// as the default rather than misparsing the reference.
|
parse_fill_value(&HeaderMessage {
|
||||||
return Ok(None);
|
msg_type: msg.msg_type,
|
||||||
}
|
size: data.len(),
|
||||||
if let Some(value) = parse_fill_value(msg)? {
|
flags: msg.flags & !0x02,
|
||||||
|
creation_order: msg.creation_order,
|
||||||
|
data,
|
||||||
|
})?
|
||||||
|
} else {
|
||||||
|
parse_fill_value(msg)?
|
||||||
|
};
|
||||||
|
if let Some(value) = value {
|
||||||
return Ok(Some(value));
|
return Ok(Some(value));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -174,7 +209,7 @@ pub fn read_full_with_fill<E: From<FormatError>>(
|
|||||||
{
|
{
|
||||||
return Err(FormatError::ExternalDataFilesUnsupported.into());
|
return Err(FormatError::ExternalDataFilesUnsupported.into());
|
||||||
}
|
}
|
||||||
let fill = dataset_fill_value(messages)?;
|
let fill = dataset_fill_value_in(file_data, messages, offset_size, length_size)?;
|
||||||
if !has_storage(layout) {
|
if !has_storage(layout) {
|
||||||
return Ok(filled_dataset(dataspace, elem_size, fill.as_deref())?);
|
return Ok(filled_dataset(dataspace, elem_size, fill.as_deref())?);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -19,8 +19,23 @@ pub const FILTER_SCALEOFFSET: u16 = 6;
|
|||||||
pub const FILTER_LZ4: u16 = 32004;
|
pub const FILTER_LZ4: u16 = 32004;
|
||||||
/// Zstandard compression.
|
/// Zstandard compression.
|
||||||
pub const FILTER_ZSTD: u16 = 32015;
|
pub const FILTER_ZSTD: u16 = 32015;
|
||||||
/// Pcodec lossless numerical codec (clawhdf5 internal; not yet HDF5-registered).
|
/// Pcodec lossless numerical codec — a **private, unregistered** clawhdf5
|
||||||
pub const FILTER_PCODEC: u16 = 32023;
|
/// filter. Pcodec has no ID in the HDF Group's filter registry (checked
|
||||||
|
/// 2026-09-25, `hdf5_plugins/docs/RegisteredFilterPlugins.md`), so it uses an
|
||||||
|
/// ID from the registry's testing/private range (256–511). No libhdf5 plugin
|
||||||
|
/// decodes it: h5py/libhdf5 report the filter as unavailable. Only clawhdf5
|
||||||
|
/// (with the `pcodec` feature) reads these datasets.
|
||||||
|
pub const FILTER_PCODEC: u16 = 480;
|
||||||
|
/// Filter name written with [`FILTER_PCODEC`].
|
||||||
|
pub const FILTER_PCODEC_NAME: &str = "pcodec (clawhdf5 private)";
|
||||||
|
/// The ID clawhdf5 up to 2.7.0 wrote pcodec under. It is registered to
|
||||||
|
/// Granular BitRound (GBR), whose decode is a pass-through, so libhdf5 with
|
||||||
|
/// that plugin would have returned the compressed bytes as data. Read as
|
||||||
|
/// pcodec only when the filter is named exactly [`FILTER_PCODEC_LEGACY_NAME`],
|
||||||
|
/// the name those versions wrote; never written.
|
||||||
|
pub const FILTER_PCODEC_LEGACY: u16 = 32023;
|
||||||
|
/// The filter name clawhdf5 up to 2.7.0 wrote with [`FILTER_PCODEC_LEGACY`].
|
||||||
|
pub const FILTER_PCODEC_LEGACY_NAME: &str = "pcodec";
|
||||||
|
|
||||||
/// Description of a single filter in a pipeline.
|
/// Description of a single filter in a pipeline.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
|||||||
@@ -8,8 +8,9 @@ use alloc::{boxed::Box, vec, vec::Vec};
|
|||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::filter_pipeline::{
|
use crate::filter_pipeline::{
|
||||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_NBIT, FILTER_PCODEC, FILTER_SCALEOFFSET,
|
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_NBIT, FILTER_PCODEC,
|
||||||
FILTER_SHUFFLE, FILTER_SZIP, FILTER_ZSTD, FilterPipeline,
|
FILTER_PCODEC_LEGACY, FILTER_PCODEC_LEGACY_NAME, FILTER_SCALEOFFSET, FILTER_SHUFFLE,
|
||||||
|
FILTER_SZIP, FILTER_ZSTD, FilterPipeline,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Absolute ceiling on a single decompressed chunk's output size, used only
|
/// Absolute ceiling on a single decompressed chunk's output size, used only
|
||||||
@@ -19,33 +20,104 @@ pub(crate) const MAX_DECOMPRESS_SIZE: usize = 256 * 1024 * 1024;
|
|||||||
|
|
||||||
/// Apply a filter pipeline to decompress a chunk.
|
/// Apply a filter pipeline to decompress a chunk.
|
||||||
/// Filters are applied in REVERSE order for decompression.
|
/// Filters are applied in REVERSE order for decompression.
|
||||||
|
///
|
||||||
|
/// Equivalent to [`decompress_chunk_masked`] with a filter mask of 0 (every
|
||||||
|
/// filter was applied when the chunk was written).
|
||||||
pub fn decompress_chunk(
|
pub fn decompress_chunk(
|
||||||
compressed: &[u8],
|
compressed: &[u8],
|
||||||
pipeline: &FilterPipeline,
|
pipeline: &FilterPipeline,
|
||||||
chunk_size: usize,
|
chunk_size: usize,
|
||||||
element_size: u32,
|
element_size: u32,
|
||||||
) -> Result<Vec<u8>, FormatError> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
let mut data = compressed.to_vec();
|
decompress_chunk_masked(compressed, pipeline, chunk_size, element_size, 0)
|
||||||
|
}
|
||||||
|
|
||||||
for filter in pipeline.filters.iter().rev() {
|
/// Upper bound on the output of filter `filter_id` applied (in the write
|
||||||
|
/// direction) to `input` bytes. 0 means "unknown" and stays unknown.
|
||||||
|
///
|
||||||
|
/// Shuffle preserves the size and Fletcher32 appends a 4-byte checksum. Any
|
||||||
|
/// other filter is a codec whose output can exceed its input on
|
||||||
|
/// incompressible data (deflate's stored blocks, LZ4's and zstd's literal
|
||||||
|
/// runs, codec headers); `n + n/8 + 64` covers every supported codec's worst
|
||||||
|
/// case while still bounding a decompression bomb to a small multiple of the
|
||||||
|
/// chunk.
|
||||||
|
fn filter_output_bound(filter_id: u16, input: usize) -> usize {
|
||||||
|
if input == 0 {
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
match filter_id {
|
||||||
|
FILTER_SHUFFLE => input,
|
||||||
|
FILTER_FLETCHER32 => input.saturating_add(4),
|
||||||
|
_ => input.saturating_add(input / 8).saturating_add(64),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether bit `index` of a chunk's filter mask says filter `index` was
|
||||||
|
/// skipped when the chunk was written.
|
||||||
|
fn filter_skipped(filter_mask: u32, index: usize) -> bool {
|
||||||
|
index < 32 && filter_mask & (1u32 << index) != 0
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether `filter_mask` says none of `pipeline`'s filters were applied, so
|
||||||
|
/// the stored bytes are the chunk itself.
|
||||||
|
pub fn all_filters_skipped(pipeline: &FilterPipeline, filter_mask: u32) -> bool {
|
||||||
|
(0..pipeline.filters.len()).all(|i| filter_skipped(filter_mask, i))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decompress a chunk whose filter mask is `filter_mask`: bit *i* set means
|
||||||
|
/// filter *i* of the pipeline was not applied when the chunk was written (an
|
||||||
|
/// optional filter that declined, or a direct chunk write), so only that
|
||||||
|
/// filter is skipped here; the others are still undone, in reverse order.
|
||||||
|
///
|
||||||
|
/// `chunk_size` is the chunk's decoded size (0 if unknown). Each stage's
|
||||||
|
/// output is capped at what the filters before it (in write order) can have
|
||||||
|
/// produced from `chunk_size` bytes — e.g. a Fletcher32 checksum placed
|
||||||
|
/// before deflate (NetCDF-4's ordering) makes deflate's output 4 bytes
|
||||||
|
/// larger than the chunk — so the decompression-bomb limit stays tight
|
||||||
|
/// without rejecting valid pipelines.
|
||||||
|
pub fn decompress_chunk_masked(
|
||||||
|
compressed: &[u8],
|
||||||
|
pipeline: &FilterPipeline,
|
||||||
|
chunk_size: usize,
|
||||||
|
element_size: u32,
|
||||||
|
filter_mask: u32,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
// bounds[i]: the most bytes that entered filter i on the write side, and
|
||||||
|
// so the most that undoing filter i may produce.
|
||||||
|
let mut bounds = Vec::with_capacity(pipeline.filters.len());
|
||||||
|
let mut size = chunk_size;
|
||||||
|
for (i, filter) in pipeline.filters.iter().enumerate() {
|
||||||
|
bounds.push(size);
|
||||||
|
if !filter_skipped(filter_mask, i) {
|
||||||
|
size = filter_output_bound(filter.filter_id, size);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut data = compressed.to_vec();
|
||||||
|
for (i, filter) in pipeline.filters.iter().enumerate().rev() {
|
||||||
|
if filter_skipped(filter_mask, i) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let bound = bounds[i];
|
||||||
data = match filter.filter_id {
|
data = match filter.filter_id {
|
||||||
FILTER_SHUFFLE => shuffle_decompress(&data, element_size as usize)?,
|
FILTER_SHUFFLE => shuffle_decompress(&data, element_size as usize)?,
|
||||||
// `chunk_size` is the expected decompressed size (shuffle/fletcher32
|
// `bound` caps the decoded size so these decoders can't be forced
|
||||||
// are size-preserving, so it bounds these too); pass it so these
|
// into unbounded allocation by a hostile or corrupted payload.
|
||||||
// decoders can't be forced into unbounded allocation by a hostile
|
FILTER_DEFLATE => deflate_decompress(&data, bound)?,
|
||||||
// or corrupted compressed payload.
|
FILTER_LZ4 => lz4_decompress(&data, bound)?,
|
||||||
FILTER_DEFLATE => deflate_decompress(&data, chunk_size)?,
|
FILTER_ZSTD => zstd_decompress(&data, bound)?,
|
||||||
FILTER_LZ4 => lz4_decompress(&data, chunk_size)?,
|
|
||||||
FILTER_ZSTD => zstd_decompress(&data, chunk_size)?,
|
|
||||||
FILTER_FLETCHER32 => fletcher32_verify(&data)?,
|
FILTER_FLETCHER32 => fletcher32_verify(&data)?,
|
||||||
FILTER_PCODEC => pcodec_decompress(&data, element_size as usize, chunk_size)?,
|
FILTER_PCODEC => pcodec_decompress(&data, element_size as usize, bound)?,
|
||||||
// `chunk_size` is the expected decompressed size; pass it so these
|
// Pcodec chunks written by clawhdf5 <= 2.7.0 under the ID registered
|
||||||
// decoders can reject an element count that would over-allocate.
|
// to Granular BitRound; recognised by the name those versions wrote.
|
||||||
FILTER_SCALEOFFSET => scaleoffset_decompress(&data, &filter.client_data, chunk_size)?,
|
FILTER_PCODEC_LEGACY if filter.name.as_deref() == Some(FILTER_PCODEC_LEGACY_NAME) => {
|
||||||
FILTER_NBIT => nbit_decompress(&data, &filter.client_data, chunk_size)?,
|
pcodec_decompress(&data, element_size as usize, bound)?
|
||||||
FILTER_SZIP => {
|
|
||||||
crate::filters_szip::szip_decompress(&data, &filter.client_data, chunk_size)?
|
|
||||||
}
|
}
|
||||||
|
// These decoders also reject an element count that would
|
||||||
|
// over-allocate past `bound`.
|
||||||
|
FILTER_SCALEOFFSET => scaleoffset_decompress(&data, &filter.client_data, bound)?,
|
||||||
|
FILTER_NBIT => nbit_decompress(&data, &filter.client_data, bound)?,
|
||||||
|
FILTER_SZIP => crate::filters_szip::szip_decompress(&data, &filter.client_data, bound)?,
|
||||||
other => return Err(FormatError::UnsupportedFilter(other)),
|
other => return Err(FormatError::UnsupportedFilter(other)),
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -69,7 +141,7 @@ pub fn compress_chunk(
|
|||||||
let level = filter.client_data.first().copied().unwrap_or(6);
|
let level = filter.client_data.first().copied().unwrap_or(6);
|
||||||
deflate_compress(&result, level)?
|
deflate_compress(&result, level)?
|
||||||
}
|
}
|
||||||
FILTER_LZ4 => lz4_compress(&result)?,
|
FILTER_LZ4 => lz4_compress(&result, &filter.client_data)?,
|
||||||
FILTER_ZSTD => {
|
FILTER_ZSTD => {
|
||||||
let level = filter.client_data.first().copied().unwrap_or(3);
|
let level = filter.client_data.first().copied().unwrap_or(3);
|
||||||
zstd_compress(&result, level)?
|
zstd_compress(&result, level)?
|
||||||
@@ -240,8 +312,20 @@ fn scaleoffset_decompress(
|
|||||||
fill_value
|
fill_value
|
||||||
} else if is_escale {
|
} else if is_escale {
|
||||||
minval + code as f64 * powi_f64(2.0, scale_factor)
|
minval + code as f64 * powi_f64(2.0, scale_factor)
|
||||||
|
} else if elem_size == 4 {
|
||||||
|
// H5Z_scaleoffset_modify_3/4 for `float`: the code is
|
||||||
|
// read as an `int` and everything is single precision,
|
||||||
|
// `(float)code / powf(10, D) + min`. Doing it in f64 and
|
||||||
|
// rounding once at the end is off by 1 ULP at times.
|
||||||
|
let d = if scale_factor >= 0 {
|
||||||
|
powi_f64(10.0, scale_factor) as f32
|
||||||
|
} else {
|
||||||
|
1.0 / powi_f64(10.0, -scale_factor) as f32
|
||||||
|
};
|
||||||
|
((code as u32 as i32) as f32 / d + minval as f32) as f64
|
||||||
} else {
|
} else {
|
||||||
minval + code as f64 / powi_f64(10.0, scale_factor)
|
// ... and for `double`: `(double)(long)code / pow(10, D) + min`.
|
||||||
|
(code as i64) as f64 / powi_f64(10.0, scale_factor) + minval
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
@@ -379,6 +463,9 @@ enum NbitNode {
|
|||||||
count: usize,
|
count: usize,
|
||||||
base_size: usize,
|
base_size: usize,
|
||||||
},
|
},
|
||||||
|
/// `H5Z_NBIT_NOOPTYPE`: a field N-Bit does not reduce (enum, string,
|
||||||
|
/// opaque, ...), stored as all `size` bytes, 8 bits each.
|
||||||
|
Noop { size: usize },
|
||||||
}
|
}
|
||||||
|
|
||||||
impl NbitNode {
|
impl NbitNode {
|
||||||
@@ -389,6 +476,7 @@ impl NbitNode {
|
|||||||
NbitNode::Array {
|
NbitNode::Array {
|
||||||
count, base_size, ..
|
count, base_size, ..
|
||||||
} => count * base_size,
|
} => count * base_size,
|
||||||
|
NbitNode::Noop { size } => *size,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -408,6 +496,7 @@ fn parse_nbit_node(cd: &[u32], idx: &mut usize, depth: u32) -> Result<NbitNode,
|
|||||||
const ATOMIC: u32 = 1;
|
const ATOMIC: u32 = 1;
|
||||||
const ARRAY: u32 = 2;
|
const ARRAY: u32 = 2;
|
||||||
const COMPOUND: u32 = 3;
|
const COMPOUND: u32 = 3;
|
||||||
|
const NOOPTYPE: u32 = 4;
|
||||||
if depth > NBIT_MAX_DEPTH {
|
if depth > NBIT_MAX_DEPTH {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"nbit: type tree nested too deeply".into(),
|
"nbit: type tree nested too deeply".into(),
|
||||||
@@ -483,8 +572,17 @@ fn parse_nbit_node(cd: &[u32], idx: &mut usize, depth: u32) -> Result<NbitNode,
|
|||||||
members,
|
members,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
// Class 4 is H5Z_NBIT_NOOPTYPE (members copied verbatim) — not seen in
|
NOOPTYPE => {
|
||||||
// practice for the supported leaf types and left unsupported.
|
// class, size
|
||||||
|
let size = nbit_cd(cd, *idx + 1)? as usize;
|
||||||
|
*idx += 2;
|
||||||
|
if size == 0 {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"nbit: invalid no-op type size".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
Ok(NbitNode::Noop { size })
|
||||||
|
}
|
||||||
_ => Err(FormatError::UnsupportedFilter(FILTER_NBIT)),
|
_ => Err(FormatError::UnsupportedFilter(FILTER_NBIT)),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -549,6 +647,11 @@ fn decode_nbit_node(
|
|||||||
decode_nbit_node(bnode, br, elem, base + i * base_size)?;
|
decode_nbit_node(bnode, br, elem, base + i * base_size)?;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
NbitNode::Noop { size } => {
|
||||||
|
for slot in &mut elem[base..base + size] {
|
||||||
|
*slot = br.read(8)? as u8;
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
@@ -561,17 +664,28 @@ fn decode_nbit_node(
|
|||||||
/// type tree — atomic (`[1, size, order, precision, offset]`), array
|
/// type tree — atomic (`[1, size, order, precision, offset]`), array
|
||||||
/// (`[2, total_size, <base>]`) and compound
|
/// (`[2, total_size, <base>]`) and compound
|
||||||
/// (`[3, total_size, nmembers, (offset, <node>)*]`) — preceded by
|
/// (`[3, total_size, nmembers, (offset, <node>)*]`) — preceded by
|
||||||
/// `[nparms, flag, nelmts]`. Decompression walks the tree once per element,
|
/// `[nparms, need_not_compress, nelmts]`; when `need_not_compress` is set
|
||||||
|
/// (every field already uses its full width, e.g. a 32-bit int of precision
|
||||||
|
/// 32) libhdf5 stores the data unchanged and so do we. Fields N-Bit cannot
|
||||||
|
/// reduce (enums, strings, ...) are no-op nodes (`[4, size]`) copied whole.
|
||||||
|
/// Decompression walks the tree once per element,
|
||||||
/// placing each field's bits at its byte/bit offset in a zero-filled element
|
/// placing each field's bits at its byte/bit offset in a zero-filled element
|
||||||
/// (HDF5's canonical reduced-precision layout). Sign-extension of reduced
|
/// (HDF5's canonical reduced-precision layout). Sign-extension of reduced
|
||||||
/// precision signed integers is the datatype reader's job. Atomic floats are
|
/// precision signed integers is the datatype reader's job, and so is
|
||||||
/// encoded as full-precision atomics and handled transparently.
|
/// converting a reduced-precision float (its own sign/exponent/mantissa
|
||||||
|
/// layout, e.g. `le_data.h5`'s 20-bit `Nbit_float_data_*`) to IEEE: the
|
||||||
|
/// filter's output is the file type's bytes, as libhdf5's is before type
|
||||||
|
/// conversion.
|
||||||
fn nbit_decompress(data: &[u8], cd: &[u32], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
fn nbit_decompress(data: &[u8], cd: &[u32], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
if cd.len() < 4 {
|
if cd.len() < 3 {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"nbit: missing filter client data".into(),
|
"nbit: missing filter client data".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
// H5Z__filter_nbit: `if (cd_values[1]) HGOTO_DONE(*buf_size)`.
|
||||||
|
if cd[1] != 0 {
|
||||||
|
return Ok(data.to_vec());
|
||||||
|
}
|
||||||
let nelmts = cd[2] as usize;
|
let nelmts = cd[2] as usize;
|
||||||
let mut idx = 3;
|
let mut idx = 3;
|
||||||
let root = parse_nbit_node(cd, &mut idx, 0)?;
|
let root = parse_nbit_node(cd, &mut idx, 0)?;
|
||||||
@@ -813,12 +927,32 @@ fn deflate_compress(_data: &[u8], _level: u32) -> Result<Vec<u8>, FormatError> {
|
|||||||
Err(FormatError::UnsupportedFilter(FILTER_DEFLATE))
|
Err(FormatError::UnsupportedFilter(FILTER_DEFLATE))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Decompress LZ4 data. Format: 4 bytes LE original size + LZ4 block data.
|
/// Default LZ4 block size of the registered HDF5 LZ4 filter (`H5Zlz4.c`,
|
||||||
|
/// `DEFAULT_BLOCK_SIZE`): 1 GiB, so an HDF5 chunk is normally one block.
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
const LZ4_DEFAULT_BLOCK_SIZE: usize = 1 << 30;
|
||||||
|
|
||||||
|
/// Decompress an LZ4 (filter 32004) chunk.
|
||||||
///
|
///
|
||||||
/// The 4-byte "original size" header is part of the attacker-controlled
|
/// Two framings are read:
|
||||||
/// compressed payload itself, so it is bounded against `expected_bytes` (the
|
///
|
||||||
/// pipeline's declared chunk size) before being used to size the output
|
/// * The registered HDF5 LZ4 filter format (`H5Zlz4.c`, what libhdf5 +
|
||||||
/// allocation — otherwise a crafted 4-byte value can request up to ~4 GiB.
|
/// hdf5plugin write, and what clawhdf5 writes after 2.7.0): an 8-byte
|
||||||
|
/// big-endian total decompressed size, a 4-byte big-endian block size, then
|
||||||
|
/// per block a 4-byte big-endian compressed length followed by the block. A
|
||||||
|
/// block whose compressed length equals its decompressed length is stored
|
||||||
|
/// raw.
|
||||||
|
/// * The legacy clawhdf5 framing (up to 2.7.0): a 4-byte little-endian size
|
||||||
|
/// followed by one raw LZ4 block. libhdf5 cannot read it.
|
||||||
|
///
|
||||||
|
/// They are told apart unambiguously: an HDF5 chunk is smaller than 4 GiB, so
|
||||||
|
/// the registered format's big-endian `u64` size always starts with four zero
|
||||||
|
/// bytes and the whole chunk is at least 12 bytes; a legacy chunk starts with
|
||||||
|
/// four zero bytes only when it is empty, and is then 5 bytes long.
|
||||||
|
///
|
||||||
|
/// Every size read from the payload is bounded against `expected_bytes` (the
|
||||||
|
/// pipeline's declared chunk size) before it sizes an allocation, so a crafted
|
||||||
|
/// header cannot request gigabytes.
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
if data.len() < 4 {
|
if data.len() < 4 {
|
||||||
@@ -826,38 +960,112 @@ fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result<Vec<u8>, FormatE
|
|||||||
"lz4: data too short".into(),
|
"lz4: data too short".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
let check_size = |size: usize| -> Result<(), FormatError> {
|
||||||
|
if expected_bytes != 0 && size > expected_bytes {
|
||||||
|
return Err(FormatError::DecompressionError(
|
||||||
|
"lz4: declared size exceeds chunk size".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if size > MAX_DECOMPRESS_SIZE {
|
||||||
|
return Err(FormatError::DecompressionError(
|
||||||
|
"lz4: declared size exceeds limit".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
};
|
||||||
|
if data.len() >= 12 && data[..4] == [0, 0, 0, 0] {
|
||||||
|
return lz4_decompress_hdf5(data, check_size);
|
||||||
|
}
|
||||||
|
// Legacy clawhdf5 framing: 4-byte LE size + one LZ4 block.
|
||||||
let orig_size = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
let orig_size = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||||
if expected_bytes != 0 && orig_size > expected_bytes {
|
check_size(orig_size)?;
|
||||||
return Err(FormatError::DecompressionError(
|
|
||||||
"lz4: declared size exceeds chunk size".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
if orig_size > MAX_DECOMPRESS_SIZE {
|
|
||||||
return Err(FormatError::DecompressionError(
|
|
||||||
"lz4: declared size exceeds limit".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
lz4_flex::block::decompress(&data[4..], orig_size)
|
lz4_flex::block::decompress(&data[4..], orig_size)
|
||||||
.map_err(|e| FormatError::DecompressionError(format!("lz4: {e}")))
|
.map_err(|e| FormatError::DecompressionError(format!("lz4: {e}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Decode the registered HDF5 LZ4 framing (see [`lz4_decompress`]).
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_decompress_hdf5(
|
||||||
|
data: &[u8],
|
||||||
|
check_size: impl Fn(usize) -> Result<(), FormatError>,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let err = |m: &str| FormatError::DecompressionError(format!("lz4: {m}"));
|
||||||
|
let be32 = |b: &[u8]| u32::from_be_bytes([b[0], b[1], b[2], b[3]]) as usize;
|
||||||
|
// The first four bytes are zero (checked by the caller), so the size is
|
||||||
|
// the low 32 bits of the big-endian u64.
|
||||||
|
let orig_size = be32(&data[4..8]);
|
||||||
|
check_size(orig_size)?;
|
||||||
|
let block_size = be32(&data[8..12]).min(orig_size);
|
||||||
|
if block_size == 0 && orig_size != 0 {
|
||||||
|
return Err(err("zero block size"));
|
||||||
|
}
|
||||||
|
let mut out = vec![0u8; orig_size];
|
||||||
|
let mut pos = 12usize;
|
||||||
|
let mut done = 0usize;
|
||||||
|
while done < orig_size {
|
||||||
|
let this_block = block_size.min(orig_size - done);
|
||||||
|
let comp_len = be32(
|
||||||
|
data.get(pos..pos + 4)
|
||||||
|
.ok_or_else(|| err("truncated block header"))?,
|
||||||
|
);
|
||||||
|
pos += 4;
|
||||||
|
let block = data
|
||||||
|
.get(pos..pos.saturating_add(comp_len))
|
||||||
|
.ok_or_else(|| err("truncated block"))?;
|
||||||
|
let dst = &mut out[done..done + this_block];
|
||||||
|
if comp_len == this_block {
|
||||||
|
dst.copy_from_slice(block);
|
||||||
|
} else {
|
||||||
|
let n = lz4_flex::block::decompress_into(block, dst)
|
||||||
|
.map_err(|e| FormatError::DecompressionError(format!("lz4: {e}")))?;
|
||||||
|
if n != this_block {
|
||||||
|
return Err(err("block decompressed to the wrong size"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pos += comp_len;
|
||||||
|
done += this_block;
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "lz4"))]
|
#[cfg(not(feature = "lz4"))]
|
||||||
fn lz4_decompress(_data: &[u8], _expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
fn lz4_decompress(_data: &[u8], _expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compress data with LZ4 block format. Format: 4 bytes LE original size + LZ4 block data.
|
/// Compress data in the registered HDF5 LZ4 filter format (see
|
||||||
|
/// [`lz4_decompress`]), so libhdf5 with the LZ4 plugin (e.g. hdf5plugin) can
|
||||||
|
/// read it. `cd[0]`, when present and non-zero, is the block size in bytes,
|
||||||
|
/// as in `H5Zlz4.c`; otherwise the 1 GiB default applies.
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn lz4_compress(data: &[u8]) -> Result<Vec<u8>, FormatError> {
|
fn lz4_compress(data: &[u8], cd: &[u32]) -> Result<Vec<u8>, FormatError> {
|
||||||
let compressed = lz4_flex::block::compress(data);
|
let block_size = match cd.first() {
|
||||||
let mut result = Vec::with_capacity(4 + compressed.len());
|
Some(&b) if b != 0 => b as usize,
|
||||||
result.extend_from_slice(&(data.len() as u32).to_le_bytes());
|
_ => LZ4_DEFAULT_BLOCK_SIZE,
|
||||||
result.extend_from_slice(&compressed);
|
}
|
||||||
|
.min(data.len());
|
||||||
|
let mut result = Vec::with_capacity(16 + data.len() / 2);
|
||||||
|
result.extend_from_slice(&(data.len() as u64).to_be_bytes());
|
||||||
|
result.extend_from_slice(&(block_size as u32).to_be_bytes());
|
||||||
|
if block_size == 0 {
|
||||||
|
return Ok(result);
|
||||||
|
}
|
||||||
|
for block in data.chunks(block_size) {
|
||||||
|
let compressed = lz4_flex::block::compress(block);
|
||||||
|
if compressed.len() >= block.len() {
|
||||||
|
// Incompressible: stored raw, marked by length == block length.
|
||||||
|
result.extend_from_slice(&(block.len() as u32).to_be_bytes());
|
||||||
|
result.extend_from_slice(block);
|
||||||
|
} else {
|
||||||
|
result.extend_from_slice(&(compressed.len() as u32).to_be_bytes());
|
||||||
|
result.extend_from_slice(&compressed);
|
||||||
|
}
|
||||||
|
}
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "lz4"))]
|
#[cfg(not(feature = "lz4"))]
|
||||||
fn lz4_compress(_data: &[u8]) -> Result<Vec<u8>, FormatError> {
|
fn lz4_compress(_data: &[u8], _cd: &[u32]) -> Result<Vec<u8>, FormatError> {
|
||||||
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -894,10 +1102,14 @@ fn zstd_decompress(_data: &[u8], _expected_bytes: usize) -> Result<Vec<u8>, Form
|
|||||||
Err(FormatError::UnsupportedFilter(FILTER_ZSTD))
|
Err(FormatError::UnsupportedFilter(FILTER_ZSTD))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compress data with zstd.
|
/// Compress data with zstd as one frame whose header records the content
|
||||||
|
/// size. The registered HDF5 Zstandard filter (`H5Zzstd.c`, used by
|
||||||
|
/// libhdf5 + hdf5plugin) sizes its output buffer from
|
||||||
|
/// `ZSTD_getFrameContentSize` and fails on a frame without it, which is what
|
||||||
|
/// the streaming encoder (`zstd::encode_all`) produced.
|
||||||
#[cfg(feature = "zstd")]
|
#[cfg(feature = "zstd")]
|
||||||
fn zstd_compress(data: &[u8], level: u32) -> Result<Vec<u8>, FormatError> {
|
fn zstd_compress(data: &[u8], level: u32) -> Result<Vec<u8>, FormatError> {
|
||||||
zstd::encode_all(data, level as i32)
|
zstd::bulk::compress(data, level as i32)
|
||||||
.map_err(|e| FormatError::CompressionError(format!("zstd: {e}")))
|
.map_err(|e| FormatError::CompressionError(format!("zstd: {e}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -913,13 +1125,13 @@ fn shuffle_decompress(data: &[u8], element_size: usize) -> Result<Vec<u8>, Forma
|
|||||||
if element_size <= 1 {
|
if element_size <= 1 {
|
||||||
return Ok(data.to_vec());
|
return Ok(data.to_vec());
|
||||||
}
|
}
|
||||||
if !data.len().is_multiple_of(element_size) {
|
// Like libhdf5, only whole elements are shuffled; trailing bytes (e.g. a
|
||||||
return Err(FormatError::FilterError(
|
// Fletcher32 checksum appended before the shuffle) are stored as-is.
|
||||||
"shuffle: data length not a multiple of element size".into(),
|
let whole = data.len() - data.len() % element_size;
|
||||||
));
|
let (data, tail) = data.split_at(whole);
|
||||||
}
|
|
||||||
let num_elements = data.len() / element_size;
|
let num_elements = data.len() / element_size;
|
||||||
let mut result = vec![0u8; data.len()];
|
let mut result = vec![0u8; whole];
|
||||||
|
result.reserve_exact(tail.len());
|
||||||
|
|
||||||
// The shuffled stream is `element_size` byte planes of `num_elements`
|
// The shuffled stream is `element_size` byte planes of `num_elements`
|
||||||
// bytes each; un-shuffling interleaves them. This is on the read path of
|
// bytes each; un-shuffling interleaves them. This is on the read path of
|
||||||
@@ -950,6 +1162,7 @@ fn shuffle_decompress(data: &[u8], element_size: usize) -> Result<Vec<u8>, Forma
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
result.extend_from_slice(tail);
|
||||||
|
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
@@ -965,19 +1178,19 @@ fn shuffle_compress(data: &[u8], element_size: usize) -> Result<Vec<u8>, FormatE
|
|||||||
if element_size <= 1 {
|
if element_size <= 1 {
|
||||||
return Ok(data.to_vec());
|
return Ok(data.to_vec());
|
||||||
}
|
}
|
||||||
if !data.len().is_multiple_of(element_size) {
|
// Trailing bytes that don't make a whole element are left in place, as
|
||||||
return Err(FormatError::FilterError(
|
// libhdf5 does.
|
||||||
"shuffle: data length not a multiple of element size".into(),
|
let whole = data.len() - data.len() % element_size;
|
||||||
));
|
let (data, tail) = data.split_at(whole);
|
||||||
}
|
|
||||||
let num_elements = data.len() / element_size;
|
let num_elements = data.len() / element_size;
|
||||||
let mut result = vec![0u8; data.len()];
|
let mut result = vec![0u8; whole];
|
||||||
|
|
||||||
match element_size {
|
match element_size {
|
||||||
4 => shuffle_compress_4(data, num_elements, &mut result),
|
4 => shuffle_compress_4(data, num_elements, &mut result),
|
||||||
8 => shuffle_compress_general(data, num_elements, element_size, &mut result),
|
8 => shuffle_compress_general(data, num_elements, element_size, &mut result),
|
||||||
_ => shuffle_compress_general(data, num_elements, element_size, &mut result),
|
_ => shuffle_compress_general(data, num_elements, element_size, &mut result),
|
||||||
}
|
}
|
||||||
|
result.extend_from_slice(tail);
|
||||||
|
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
@@ -1413,6 +1626,110 @@ mod tests {
|
|||||||
assert_eq!(decompressed, data);
|
assert_eq!(decompressed, data);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn filter(filter_id: u16) -> FilterDescription {
|
||||||
|
FilterDescription {
|
||||||
|
filter_id,
|
||||||
|
name: None,
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![],
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "deflate")]
|
||||||
|
fn filter_mask_skips_only_the_masked_filters() {
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![filter(FILTER_SHUFFLE), filter(FILTER_DEFLATE)],
|
||||||
|
};
|
||||||
|
let only_shuffle = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![filter(FILTER_SHUFFLE)],
|
||||||
|
};
|
||||||
|
let only_deflate = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![filter(FILTER_DEFLATE)],
|
||||||
|
};
|
||||||
|
let data: Vec<u8> = (0..200).map(|i| (i * 7 % 256) as u8).collect();
|
||||||
|
let n = data.len();
|
||||||
|
|
||||||
|
let shuffled = compress_chunk(&data, &only_shuffle, 8).unwrap();
|
||||||
|
assert_ne!(shuffled, data);
|
||||||
|
let deflated = compress_chunk(&data, &only_deflate, 8).unwrap();
|
||||||
|
|
||||||
|
// Bit 1: deflate skipped, shuffle still undone.
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk_masked(&shuffled, &pipeline, n, 8, 0b10).unwrap(),
|
||||||
|
data
|
||||||
|
);
|
||||||
|
// Bit 0: shuffle skipped, deflate still undone.
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk_masked(&deflated, &pipeline, n, 8, 0b01).unwrap(),
|
||||||
|
data
|
||||||
|
);
|
||||||
|
// Both bits (and bits past the pipeline): stored as-is.
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk_masked(&data, &pipeline, n, 8, u32::MAX).unwrap(),
|
||||||
|
data
|
||||||
|
);
|
||||||
|
assert!(all_filters_skipped(&pipeline, 0b11));
|
||||||
|
assert!(!all_filters_skipped(&pipeline, 0b10));
|
||||||
|
// An unsupported filter is fine when the chunk skipped it.
|
||||||
|
let unknown = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![filter(32000), filter(FILTER_DEFLATE)],
|
||||||
|
};
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk_masked(&deflated, &unknown, n, 8, 0b01).unwrap(),
|
||||||
|
data
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "deflate")]
|
||||||
|
fn fletcher32_ahead_of_deflate_stays_bounded() {
|
||||||
|
// NetCDF-4 order: the checksum is appended before shuffle and deflate,
|
||||||
|
// so deflate decodes chunk + 4 bytes.
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
filter(FILTER_FLETCHER32),
|
||||||
|
filter(FILTER_SHUFFLE),
|
||||||
|
filter(FILTER_DEFLATE),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
let data: Vec<u8> = (0..400).map(|i| (i * 13 % 251) as u8).collect();
|
||||||
|
let n = data.len();
|
||||||
|
let stored = compress_chunk(&data, &pipeline, 8).unwrap();
|
||||||
|
assert_eq!(decompress_chunk(&stored, &pipeline, n, 8).unwrap(), data);
|
||||||
|
|
||||||
|
// The cap still bites: a stream that inflates past chunk + 4 bytes
|
||||||
|
// is rejected rather than allocated.
|
||||||
|
let only_deflate = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![filter(FILTER_DEFLATE)],
|
||||||
|
};
|
||||||
|
let oversized = compress_chunk(&vec![0u8; n + 5], &only_deflate, 8).unwrap();
|
||||||
|
let err = decompress_chunk(&oversized, &pipeline, n, 8).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
matches!(err, FormatError::DecompressionError(_)),
|
||||||
|
"expected a size-limit error, got {err:?}"
|
||||||
|
);
|
||||||
|
// A bomb is still stopped near the chunk size.
|
||||||
|
let bomb = compress_chunk(&vec![0u8; 64 * n], &only_deflate, 8).unwrap();
|
||||||
|
assert!(decompress_chunk(&bomb, &pipeline, n, 8).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn shuffle_leaves_a_partial_trailing_element_in_place() {
|
||||||
|
// libhdf5 shuffles whole elements and copies the remainder as-is.
|
||||||
|
let data: Vec<u8> = (0..20).collect();
|
||||||
|
let shuffled = shuffle_compress(&data, 8).unwrap();
|
||||||
|
assert_eq!(&shuffled[16..], &data[16..]);
|
||||||
|
assert_eq!(&shuffled[..4], &[0, 8, 1, 9]);
|
||||||
|
assert_eq!(shuffle_decompress(&shuffled, 8).unwrap(), data);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "deflate")]
|
#[cfg(feature = "deflate")]
|
||||||
fn pipeline_compress_decompress_roundtrip() {
|
fn pipeline_compress_decompress_roundtrip() {
|
||||||
@@ -1480,11 +1797,317 @@ mod tests {
|
|||||||
|
|
||||||
// --- LZ4 tests ---
|
// --- LZ4 tests ---
|
||||||
|
|
||||||
|
fn unhex(s: &str) -> Vec<u8> {
|
||||||
|
(0..s.len())
|
||||||
|
.step_by(2)
|
||||||
|
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn one_filter(filter_id: u16, client_data: Vec<u32>) -> FilterDescription {
|
||||||
|
FilterDescription {
|
||||||
|
filter_id,
|
||||||
|
name: None,
|
||||||
|
flags: 1,
|
||||||
|
client_data,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Fletcher32 before a compressor (libhdf5 applies filters in pipeline
|
||||||
|
/// order, so the compressor sees chunk + checksum): deflate's output is 4
|
||||||
|
/// bytes over the chunk size, which we rejected as "deflate: output
|
||||||
|
/// exceeds size limit". Chunks from h5py/libhdf5, values from h5py.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "deflate")]
|
||||||
|
fn fletcher32_before_deflate_decodes() {
|
||||||
|
// h5py: set_fletcher32(); set_deflate(4); i32 0..100, chunks of 10.
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
one_filter(FILTER_FLETCHER32, vec![]),
|
||||||
|
one_filter(FILTER_DEFLATE, vec![4]),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
let raw =
|
||||||
|
unhex("785e936360609007620520560462252056066215205605623520560762c6483e3d00234501f0");
|
||||||
|
let want: Vec<u8> = (30..40i32).flat_map(i32::to_le_bytes).collect();
|
||||||
|
assert_eq!(decompress_chunk(&raw, &pipeline, 40, 4).unwrap(), want);
|
||||||
|
|
||||||
|
// And our own writer's round trip through the same pipeline order.
|
||||||
|
let data: Vec<u8> = (0..400u32).map(|i| (i % 13) as u8).collect();
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
one_filter(FILTER_SHUFFLE, vec![4]),
|
||||||
|
one_filter(FILTER_FLETCHER32, vec![]),
|
||||||
|
one_filter(FILTER_DEFLATE, vec![9]),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
let c = compress_chunk(&data, &pipeline, 4).unwrap();
|
||||||
|
assert_eq!(decompress_chunk(&c, &pipeline, 400, 4).unwrap(), data);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `le_data.h5` scale-offset (D-scale, D = 3, fill -2.2) chunks, decoded
|
||||||
|
/// bit for bit as libhdf5 does: single-precision arithmetic for `float`
|
||||||
|
/// (we computed in f64 and rounded once, which was 1 ULP off for e.g.
|
||||||
|
/// 1.6663333: `694ad53f` instead of `6a4ad53f`), double for `double`.
|
||||||
|
#[test]
|
||||||
|
fn scaleoffset_float_dscale_matches_libhdf5_bits() {
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||||
|
let cd = |size: u32, order: u32, fill_lo: u32, fill_hi: u32| {
|
||||||
|
let mut cd = vec![0, 3, 12, 1, size, 0, order, 1, fill_lo, fill_hi];
|
||||||
|
cd.resize(20, 0);
|
||||||
|
cd
|
||||||
|
};
|
||||||
|
let f32_le = cd(4, 0, 0xC00C_CCCD, 0);
|
||||||
|
let f32_be = cd(4, 1, 0xC00C_CCCD, 0);
|
||||||
|
let f64_le = cd(8, 0, 2576980378, 3221330329);
|
||||||
|
#[rustfmt::skip]
|
||||||
|
let cases: [(usize, usize, &[u32], &str); 6] = [
|
||||||
|
(2816, 38, &f32_le, "abaaaa3ed2942a3fec0a803fd2942a3fec0a803fabaaaa3fec0a803fabaaaa3f694ad53fabaaaa3f694ad53f76050040"),
|
||||||
|
(2854, 38, &f32_le, "abaaaa3f6a4ad53f760500406a4ad53f7605004056551540760500405655154034a52a405655154034a52a4076054040"),
|
||||||
|
(712, 38, &f32_be, "3eaaaaab3f2a94d23f800aec3f2a94d23f800aec3faaaaab3f800aec3faaaaab3fd54a693faaaaab3fd54a6940000576"),
|
||||||
|
(750, 38, &f32_be, "3faaaaab3fd54a6a400005763fd54a6a40000576401555564000057640155556402aa53440155556402aa53440400576"),
|
||||||
|
(2050, 38, &f64_le, concat!(
|
||||||
|
"555555555555d53fb9d75c489a52e53fce3e7c865d01f03fb9d75c489a52e53fce3e7c865d01f03f555555555555f53f",
|
||||||
|
"ce3e7c865d01f03f555555555555f53fdc6b2e244da9fa3f555555555555f53fdc6b2e244da9fa3f671f3ec3ae000040")),
|
||||||
|
(2088, 38, &f64_le, concat!(
|
||||||
|
"555555555555f53fdc6b2e244da9fa3f671f3ec3ae000040dc6b2e244da9fa3f671f3ec3ae000040aaaaaaaaaaaa0240",
|
||||||
|
"671f3ec3ae000040aaaaaaaaaaaa0240ee351792a6540540aaaaaaaaaaaa0240ee351792a6540540671f3ec3ae000840")),
|
||||||
|
];
|
||||||
|
for (off, len, cd, want) in cases {
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![one_filter(FILTER_SCALEOFFSET, cd.to_vec())],
|
||||||
|
};
|
||||||
|
let want = unhex(want);
|
||||||
|
let got =
|
||||||
|
decompress_chunk(&file[off..off + len], &pipeline, want.len(), cd[4]).unwrap();
|
||||||
|
assert_eq!(got, want, "chunk at {off}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `le_data.h5` `/Nbit_float_data_{le,be}` chunk (0,0): a 20-bit float
|
||||||
|
/// (offset 7) packed by N-Bit. The filter must reproduce libhdf5's
|
||||||
|
/// decoded bytes in the *file* datatype (h5py `DatasetID.read` with the
|
||||||
|
/// file type as memory type, so no conversion); converting that custom
|
||||||
|
/// float layout to IEEE is the datatype reader's job, not the filter's.
|
||||||
|
#[test]
|
||||||
|
fn nbit_float_matches_libhdf5_file_type_bytes() {
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||||
|
let cases = [
|
||||||
|
(
|
||||||
|
55952,
|
||||||
|
0,
|
||||||
|
"8055d5018055e5010000f0018055e5010000f0018055f5010000f0018055f50180aafa018055f50180aafa0100000002",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
56076,
|
||||||
|
1,
|
||||||
|
"01d5558001e5558001f0000001e5558001f0000001f5558001f0000001f5558001faaa8001f5558001faaa8002000000",
|
||||||
|
),
|
||||||
|
];
|
||||||
|
for (off, order, want) in cases {
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![one_filter(FILTER_NBIT, vec![8, 0, 12, 1, 4, order, 20, 7])],
|
||||||
|
};
|
||||||
|
let got = decompress_chunk(&file[off..off + 31], &pipeline, 48, 4).unwrap();
|
||||||
|
assert_eq!(got, unhex(want), "byte order {order}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// libhdf5 sets `cd_values[1]` ("need not compress") when every field is
|
||||||
|
/// already full width and then stores the data unchanged; we unpacked it
|
||||||
|
/// anyway and failed with "nbit: packed data too short".
|
||||||
|
#[test]
|
||||||
|
fn nbit_need_not_compress_is_passthrough() {
|
||||||
|
let data: Vec<u8> = (0..200u32).map(|i| (i * 7) as u8).collect();
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![one_filter(FILTER_NBIT, vec![8, 1, 50, 1, 4, 0, 32, 0])],
|
||||||
|
};
|
||||||
|
assert_eq!(decompress_chunk(&data, &pipeline, 200, 4).unwrap(), data);
|
||||||
|
// A top-level type N-Bit has no parameters for (e.g. an enum) carries only
|
||||||
|
// [nparms, need_not_compress, nelmts].
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![one_filter(FILTER_NBIT, vec![3, 1, 50])],
|
||||||
|
};
|
||||||
|
assert_eq!(decompress_chunk(&data, &pipeline, 200, 4).unwrap(), data);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `tfilters.h5` `/all` chunk (0,0): shuffle, szip, deflate, fletcher32
|
||||||
|
/// and a pass-through N-Bit in one pipeline; values from h5py.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
fn nbit_in_multi_filter_pipeline_matches_libhdf5() {
|
||||||
|
let raw = unhex(concat!(
|
||||||
|
"785e3bc1c0c030cb6517ff039e556de1576c0f300b152c771070641170640b99ba879f8167d5cce82f760ccc4285d71b",
|
||||||
|
"20c2afc226329f60d60ab5636a3fc1905499c3c0a1d0c4a1d05c6a13d7f88271aaf6725fe7170c86363f1858c0ca0104",
|
||||||
|
"bf1e95c75f3eeb",
|
||||||
|
));
|
||||||
|
let want = unhex(concat!(
|
||||||
|
"00000000010000000200000003000000040000000a0000000b0000000c0000000d0000000e0000001400000015000000",
|
||||||
|
"1600000017000000180000001e0000001f00000020000000210000002200000028000000290000002a0000002b000000",
|
||||||
|
"2c00000032000000330000003400000035000000360000003c0000003d0000003e0000003f0000004000000046000000",
|
||||||
|
"4700000048000000490000004a00000050000000510000005200000053000000540000005a0000005b0000005c000000",
|
||||||
|
"5d0000005e000000",
|
||||||
|
));
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
one_filter(FILTER_SHUFFLE, vec![4]),
|
||||||
|
one_filter(FILTER_SZIP, vec![141, 4, 32, 5]),
|
||||||
|
one_filter(FILTER_DEFLATE, vec![5]),
|
||||||
|
one_filter(FILTER_FLETCHER32, vec![]),
|
||||||
|
one_filter(FILTER_NBIT, vec![8, 1, 50, 1, 4, 0, 32, 0]),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
assert_eq!(decompress_chunk(&raw, &pipeline, 200, 4).unwrap(), want);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `h5repack_nested_8bit_enum_deflated.h5` `/tracks/1/trace` chunk 0: a
|
||||||
|
/// 376-byte compound whose `u1` enum member N-Bit stores whole as a
|
||||||
|
/// no-op type (class 4), then deflate. Was `UnsupportedFilter(5)`.
|
||||||
|
/// Expected bytes: libhdf5's decode in the file datatype.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "deflate")]
|
||||||
|
fn nbit_compound_with_enum_member_matches_libhdf5() {
|
||||||
|
#[rustfmt::skip]
|
||||||
|
let cd: Vec<u32> = vec![
|
||||||
|
251, 0, 1, 3, 376, 38,
|
||||||
|
0, 1, 4, 0, 32, 0,
|
||||||
|
8, 2, 96, 1, 8, 0, 64, 0,
|
||||||
|
104, 1, 8, 0, 64, 0, 112, 1, 8, 0, 64, 0, 120, 1, 8, 0, 64, 0,
|
||||||
|
128, 1, 8, 0, 64, 0, 136, 1, 8, 0, 64, 0, 144, 1, 8, 0, 64, 0,
|
||||||
|
152, 1, 8, 0, 64, 0,
|
||||||
|
160, 2, 24, 1, 8, 0, 64, 0, 184, 2, 24, 1, 8, 0, 64, 0,
|
||||||
|
208, 2, 16, 1, 8, 0, 64, 0, 224, 2, 32, 1, 8, 0, 64, 0,
|
||||||
|
256, 2, 16, 1, 4, 0, 32, 0, 272, 2, 16, 1, 4, 0, 32, 0,
|
||||||
|
288, 2, 32, 1, 8, 0, 64, 0, 320, 2, 16, 1, 8, 0, 64, 0,
|
||||||
|
336, 2, 8, 1, 4, 0, 32, 0,
|
||||||
|
344, 4, 1,
|
||||||
|
346, 1, 2, 0, 16, 0, 348, 1, 4, 0, 32, 0, 352, 1, 1, 0, 4, 0,
|
||||||
|
354, 1, 2, 0, 16, 0, 356, 1, 1, 0, 4, 0, 357, 1, 1, 0, 4, 0,
|
||||||
|
358, 1, 1, 0, 4, 0, 359, 1, 1, 0, 4, 0, 360, 1, 1, 0, 4, 0,
|
||||||
|
361, 1, 1, 0, 4, 0, 362, 1, 1, 0, 4, 0, 364, 1, 1, 0, 4, 0,
|
||||||
|
363, 1, 1, 0, 4, 0, 365, 1, 1, 0, 4, 0, 366, 1, 1, 0, 4, 0,
|
||||||
|
367, 1, 1, 0, 4, 0, 368, 1, 1, 0, 4, 0, 369, 1, 1, 0, 4, 0,
|
||||||
|
370, 1, 1, 0, 4, 0,
|
||||||
|
];
|
||||||
|
assert_eq!(cd.len(), 251);
|
||||||
|
let raw = unhex(concat!(
|
||||||
|
"780163606078c930c880c3879573a60b2d7883eeac06a880835c47bda16cda7987a0c59ec9f74c4af6ff677e5df16445",
|
||||||
|
"adfd8493ce3ba592ddedab2a3bee87dd0faaff00d1808b66606006fa9d999b818125014837fd4703fba1fa71d150e720",
|
||||||
|
"512caa4073dc59212206500946060600604c37fc",
|
||||||
|
));
|
||||||
|
let want = unhex(concat!(
|
||||||
|
"e90000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||||
|
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||||
|
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||||
|
"0000000000000000eca012979ca9f040000000000000000000000000000000000000000000000080cf661d317f881e40",
|
||||||
|
"7434de6349a352407da8e478eb03ffbf47631ab943c9903f52df56df88797a3f000000000000f07f000000000000f07f",
|
||||||
|
"000000000000f07f000000000000f07fe90300000b0300006004000082030000ffffffffffffffffffffffffffffffff",
|
||||||
|
"000000000000f0bf000000000000f0bf000000000000f0bf000000000000f0bf00000000000000000000000000000000",
|
||||||
|
"25040000470300000500000000000000030000000000000000000000000000000100000000000000",
|
||||||
|
));
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
one_filter(FILTER_NBIT, cd),
|
||||||
|
one_filter(FILTER_DEFLATE, vec![1]),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
assert_eq!(decompress_chunk(&raw, &pipeline, 376, 376).unwrap(), want);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Chunk (0,0) of `/DS1` in the HDF Group's `h5ex_d_lz4.h5` example,
|
||||||
|
/// written by libhdf5's registered LZ4 plugin with a 3-byte block size
|
||||||
|
/// (so it has many blocks, some stored raw). Byte range from h5py's
|
||||||
|
/// `get_chunk_info`; values are `i*j - j` (i32 LE), as h5py reads them.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_reads_registered_hdf5_format() {
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/filters/h5ex_d_lz4.h5");
|
||||||
|
let chunk = &file[4016..4016 + 312];
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![FilterDescription {
|
||||||
|
filter_id: FILTER_LZ4,
|
||||||
|
name: None,
|
||||||
|
flags: 1,
|
||||||
|
client_data: vec![3],
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
let out = decompress_chunk(chunk, &pipeline, 4 * 8 * 4, 4).unwrap();
|
||||||
|
let expected: Vec<u8> = (0..4i32)
|
||||||
|
.flat_map(|i| (0..8i32).map(move |j| i * j - j))
|
||||||
|
.flat_map(i32::to_le_bytes)
|
||||||
|
.collect();
|
||||||
|
assert_eq!(out, expected);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_writes_registered_hdf5_format() {
|
||||||
|
// Compressible data, one block: 8-byte BE size, 4-byte BE block size,
|
||||||
|
// 4-byte BE compressed length, block.
|
||||||
|
let data = vec![7u8; 1000];
|
||||||
|
let c = lz4_compress(&data, &[]).unwrap();
|
||||||
|
assert_eq!(&c[0..8], &1000u64.to_be_bytes());
|
||||||
|
assert_eq!(&c[8..12], &1000u32.to_be_bytes());
|
||||||
|
let len = u32::from_be_bytes(c[12..16].try_into().unwrap()) as usize;
|
||||||
|
assert_eq!(c.len(), 16 + len);
|
||||||
|
assert!(len < 1000);
|
||||||
|
assert_eq!(lz4_decompress(&c, 1000).unwrap(), data);
|
||||||
|
|
||||||
|
// Several blocks, incompressible ones stored raw (length == block).
|
||||||
|
let data: Vec<u8> = (0..10u8).collect();
|
||||||
|
let c = lz4_compress(&data, &[3]).unwrap();
|
||||||
|
assert_eq!(&c[8..12], &3u32.to_be_bytes());
|
||||||
|
assert_eq!(&c[12..16], &3u32.to_be_bytes());
|
||||||
|
assert_eq!(&c[16..19], &[0, 1, 2]);
|
||||||
|
assert_eq!(c.len(), 12 + 3 * (4 + 3) + (4 + 1));
|
||||||
|
assert_eq!(lz4_decompress(&c, 10).unwrap(), data);
|
||||||
|
|
||||||
|
let c = lz4_compress(&[], &[]).unwrap();
|
||||||
|
assert_eq!(lz4_decompress(&c, 0).unwrap(), Vec::<u8>::new());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Chunks written by clawhdf5 up to 2.7.0 (4-byte LE size + one LZ4
|
||||||
|
/// block) must stay readable.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_reads_legacy_clawhdf5_format() {
|
||||||
|
for data in [vec![], vec![5u8; 300], (0..=255u8).collect::<Vec<u8>>()] {
|
||||||
|
let mut legacy = (data.len() as u32).to_le_bytes().to_vec();
|
||||||
|
legacy.extend_from_slice(&lz4_flex::block::compress(&data));
|
||||||
|
assert_eq!(lz4_decompress(&legacy, data.len()).unwrap(), data);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_registered_format_rejects_hostile_sizes() {
|
||||||
|
// Declared total larger than the chunk.
|
||||||
|
let mut c = 1000u64.to_be_bytes().to_vec();
|
||||||
|
c.extend_from_slice(&1000u32.to_be_bytes());
|
||||||
|
c.extend_from_slice(&[0u8; 8]);
|
||||||
|
assert!(lz4_decompress(&c, 64).is_err());
|
||||||
|
// Truncated block.
|
||||||
|
let mut c = 16u64.to_be_bytes().to_vec();
|
||||||
|
c.extend_from_slice(&16u32.to_be_bytes());
|
||||||
|
c.extend_from_slice(&16u32.to_be_bytes());
|
||||||
|
c.extend_from_slice(&[1u8; 4]);
|
||||||
|
assert!(lz4_decompress(&c, 16).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn lz4_compress_decompress_roundtrip() {
|
fn lz4_compress_decompress_roundtrip() {
|
||||||
let data: Vec<u8> = (0..256).map(|i| (i % 256) as u8).collect();
|
let data: Vec<u8> = (0..256).map(|i| (i % 256) as u8).collect();
|
||||||
let compressed = lz4_compress(&data).unwrap();
|
let compressed = lz4_compress(&data, &[]).unwrap();
|
||||||
let decompressed = lz4_decompress(&compressed, data.len()).unwrap();
|
let decompressed = lz4_decompress(&compressed, data.len()).unwrap();
|
||||||
assert_eq!(decompressed, data);
|
assert_eq!(decompressed, data);
|
||||||
}
|
}
|
||||||
@@ -1535,6 +2158,23 @@ mod tests {
|
|||||||
|
|
||||||
// --- Zstd tests ---
|
// --- Zstd tests ---
|
||||||
|
|
||||||
|
/// libhdf5's zstd plugin needs the frame content size to size its
|
||||||
|
/// output; frames without it fail to decode there.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "zstd")]
|
||||||
|
fn zstd_frames_record_content_size() {
|
||||||
|
for n in [0usize, 1, 200, 100_000] {
|
||||||
|
let data: Vec<u8> = (0..n).map(|i| (i % 7) as u8).collect();
|
||||||
|
let c = zstd_compress(&data, 3).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
zstd::zstd_safe::get_frame_content_size(&c).unwrap(),
|
||||||
|
Some(n as u64),
|
||||||
|
"{n} bytes"
|
||||||
|
);
|
||||||
|
assert_eq!(zstd_decompress(&c, n).unwrap(), data);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "zstd")]
|
#[cfg(feature = "zstd")]
|
||||||
fn zstd_compress_decompress_roundtrip() {
|
fn zstd_compress_decompress_roundtrip() {
|
||||||
@@ -1996,6 +2636,57 @@ mod tests {
|
|||||||
assert!(pcodec_decompress(&compressed, 4, 16).is_err());
|
assert!(pcodec_decompress(&compressed, 4, 16).is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Pcodec is written under the private ID 480, not 32023 (registered to
|
||||||
|
/// Granular BitRound, whose pass-through decode would hand libhdf5 users
|
||||||
|
/// the compressed bytes as data). Chunks under 32023 are read as pcodec
|
||||||
|
/// only with the name clawhdf5 <= 2.7.0 wrote.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "pcodec")]
|
||||||
|
fn pcodec_uses_private_id_and_reads_legacy_32023() {
|
||||||
|
use crate::chunked_write::ChunkOptions;
|
||||||
|
let opts = ChunkOptions {
|
||||||
|
pcodec: true,
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let pl = opts.build_pipeline(8).unwrap();
|
||||||
|
let f = pl.filters.iter().find(|f| f.filter_id == 480).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
f.name.as_deref(),
|
||||||
|
Some(crate::filter_pipeline::FILTER_PCODEC_NAME)
|
||||||
|
);
|
||||||
|
assert!(pl.filters.iter().all(|f| f.filter_id != 32023));
|
||||||
|
|
||||||
|
let data: Vec<f64> = (0..100).map(|i| i as f64 * 0.25).collect();
|
||||||
|
let raw: Vec<u8> = data.iter().flat_map(|x| x.to_le_bytes()).collect();
|
||||||
|
let compressed = pcodec_compress(&raw, 8).unwrap();
|
||||||
|
let pipeline = |id: u16, name: Option<&str>| FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![FilterDescription {
|
||||||
|
filter_id: id,
|
||||||
|
name: name.map(Into::into),
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![8],
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
let legacy = pipeline(32023, Some("pcodec"));
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(&compressed, &legacy, raw.len(), 8).unwrap(),
|
||||||
|
raw
|
||||||
|
);
|
||||||
|
let current = pipeline(480, None);
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(&compressed, ¤t, raw.len(), 8).unwrap(),
|
||||||
|
raw
|
||||||
|
);
|
||||||
|
// A real Granular BitRound filter is not pcodec.
|
||||||
|
for name in [None, Some("Granular BitRound")] {
|
||||||
|
assert!(matches!(
|
||||||
|
decompress_chunk(&compressed, &pipeline(32023, name), raw.len(), 8),
|
||||||
|
Err(FormatError::UnsupportedFilter(32023))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn decompress_chunk_rejects_hostile_lz4_size_via_public_entrypoint() {
|
fn decompress_chunk_rejects_hostile_lz4_size_via_public_entrypoint() {
|
||||||
|
|||||||
@@ -1,19 +1,39 @@
|
|||||||
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
||||||
//!
|
//!
|
||||||
//! Gated by the `szip` feature which links against the system libaec library.
|
//! Gated by the `szip` feature which links against the system libaec library.
|
||||||
|
//!
|
||||||
|
//! libhdf5's SZIP filter (`H5Zszip.c`) prefixes each chunk with its
|
||||||
|
//! uncompressed size and hands the rest to szlib's `SZ_BufftoBuffDecompress`.
|
||||||
|
//! libaec implements that call (`sz_compat.c`) on top of `aec_buffer_decode`
|
||||||
|
//! with some reshaping — 32/64-bit samples are coded as byte planes of 8-bit
|
||||||
|
//! samples, and scanlines that are not a whole number of blocks are padded —
|
||||||
|
//! which [`szip_decompress`] reproduces so its output matches libhdf5's.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
/// Decompress SZIP-compressed data using libaec.
|
/// `SZ_MSB_OPTION_MASK`: samples are big-endian.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
const SZ_MSB_OPTION_MASK: u32 = 16;
|
||||||
|
/// `SZ_NN_OPTION_MASK`: nearest-neighbour preprocessing.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
const SZ_NN_OPTION_MASK: u32 = 32;
|
||||||
|
|
||||||
|
/// Decompress one SZIP-filtered chunk.
|
||||||
///
|
///
|
||||||
/// `cd` is the HDF5 SZIP filter client data (matches `H5Z_SZIP_PARM_*` indices):
|
/// `cd` is the HDF5 SZIP filter client data (`H5Z_SZIP_PARM_*` indices):
|
||||||
/// cd[0] = options mask (`H5_SZIP_NN_OPTION_MASK = 0x20` enables NN preprocessing)
|
/// cd[0] = options mask (`SZ_*_OPTION_MASK`: 16 = MSB byte order,
|
||||||
/// cd[1] = pixels per block (H5Z_SZIP_PARM_PPB; 8, 10, 16, or 32)
|
/// 32 = nearest-neighbour preprocessing; K13/EC/LSB/RAW bits carry
|
||||||
/// cd[2] = bits per sample (H5Z_SZIP_PARM_BPP; element bit width)
|
/// no decoding information for libaec)
|
||||||
/// cd[3] = pixels per scan line (H5Z_SZIP_PARM_PPS; informational only)
|
/// cd[1] = pixels per block
|
||||||
|
/// cd[2] = bits per pixel (sample precision, rounded up to 32 or 64 above
|
||||||
|
/// 24 by libhdf5)
|
||||||
|
/// cd[3] = pixels per scanline
|
||||||
|
///
|
||||||
|
/// The chunk is a 4-byte little-endian uncompressed size followed by the
|
||||||
|
/// szlib stream.
|
||||||
pub(crate) fn szip_decompress(
|
pub(crate) fn szip_decompress(
|
||||||
_data: &[u8],
|
_data: &[u8],
|
||||||
_cd: &[u32],
|
_cd: &[u32],
|
||||||
@@ -33,62 +53,174 @@ pub(crate) fn szip_decompress(
|
|||||||
|
|
||||||
#[cfg(feature = "szip")]
|
#[cfg(feature = "szip")]
|
||||||
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
if cd.len() < 3 {
|
let err = |m: &str| FormatError::ChunkedReadError(format!("szip: {m}"));
|
||||||
return Err(FormatError::ChunkedReadError(
|
if cd.len() < 4 {
|
||||||
"szip: missing client data".into(),
|
return Err(err("missing client data"));
|
||||||
));
|
|
||||||
}
|
}
|
||||||
let options = cd[0];
|
let options = cd[0];
|
||||||
let pixels_per_block = cd[1];
|
let pixels_per_block = cd[1] as usize;
|
||||||
let bits_per_sample = cd[2]; // H5Z_SZIP_PARM_BPP
|
let bits_per_pixel = cd[2];
|
||||||
if bits_per_sample == 0 || bits_per_sample > 32 {
|
let pixels_per_scanline = cd[3] as usize;
|
||||||
return Err(FormatError::ChunkedReadError(
|
if !(1..=32).contains(&bits_per_pixel) && bits_per_pixel != 64 {
|
||||||
"szip: invalid bits per sample".into(),
|
return Err(err("invalid bits per sample"));
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if chunk_size == 0 {
|
if pixels_per_block == 0 || pixels_per_scanline == 0 {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(err("invalid block or scanline size"));
|
||||||
"szip: unknown output size".into(),
|
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if data.is_empty() {
|
if data.len() < 4 {
|
||||||
return Err(FormatError::ChunkedReadError("szip: empty input".into()));
|
return Err(err("chunk too short"));
|
||||||
}
|
}
|
||||||
|
// H5Zszip.c: UINT32DECODE of the uncompressed size, then the stream.
|
||||||
|
let dest_len = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||||
|
let limit = if chunk_size != 0 {
|
||||||
|
chunk_size
|
||||||
|
} else {
|
||||||
|
crate::filters::MAX_DECOMPRESS_SIZE
|
||||||
|
};
|
||||||
|
if dest_len > limit {
|
||||||
|
return Err(err("declared size exceeds chunk size"));
|
||||||
|
}
|
||||||
|
let stream = &data[4..];
|
||||||
|
|
||||||
// Map HDF5 option mask to libaec flags.
|
// --- libaec sz_compat.c: SZ_BufftoBuffDecompress ---
|
||||||
// HDF5 always stores SZIP data in MSB order, so AEC_DATA_MSB is unconditional.
|
let rsi = pixels_per_scanline.div_ceil(pixels_per_block);
|
||||||
// H5_SZIP_NN_OPTION_MASK (0x20): NN differential preprocessing.
|
let mut flags = 0;
|
||||||
let mut flags: u32 = libaec_sys::AEC_DATA_MSB;
|
if options & SZ_MSB_OPTION_MASK != 0 {
|
||||||
if options & 0x20 != 0 {
|
flags |= libaec_sys::AEC_DATA_MSB;
|
||||||
|
}
|
||||||
|
if options & SZ_NN_OPTION_MASK != 0 {
|
||||||
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
||||||
}
|
}
|
||||||
|
let pad_scanline = !pixels_per_scanline.is_multiple_of(pixels_per_block);
|
||||||
|
let deinterleave = bits_per_pixel == 32 || bits_per_pixel == 64;
|
||||||
|
let bits_per_sample = if deinterleave { 8 } else { bits_per_pixel };
|
||||||
|
let pixel_size = match bits_per_sample {
|
||||||
|
17.. => 4,
|
||||||
|
9.. => 2,
|
||||||
|
_ => 1,
|
||||||
|
};
|
||||||
|
let scanlines = (dest_len / pixel_size).div_ceil(pixels_per_scanline);
|
||||||
|
let buf_size = if pad_scanline {
|
||||||
|
rsi.checked_mul(pixels_per_block)
|
||||||
|
.and_then(|n| n.checked_mul(pixel_size))
|
||||||
|
.and_then(|n| n.checked_mul(scanlines))
|
||||||
|
.filter(|&n| n <= crate::filters::MAX_DECOMPRESS_SIZE.max(limit))
|
||||||
|
.ok_or_else(|| err("scanline padding too large"))?
|
||||||
|
} else {
|
||||||
|
dest_len
|
||||||
|
};
|
||||||
|
|
||||||
let mut out = vec![0u8; chunk_size];
|
let mut buf = vec![0u8; buf_size];
|
||||||
let mut strm = libaec_sys::AecStream::zeroed();
|
let mut strm = libaec_sys::AecStream::zeroed();
|
||||||
strm.next_in = data.as_ptr();
|
strm.next_in = stream.as_ptr();
|
||||||
strm.avail_in = data.len();
|
strm.avail_in = stream.len();
|
||||||
strm.next_out = out.as_mut_ptr();
|
strm.next_out = buf.as_mut_ptr();
|
||||||
strm.avail_out = chunk_size;
|
strm.avail_out = buf_size;
|
||||||
strm.bits_per_sample = bits_per_sample;
|
strm.bits_per_sample = bits_per_sample;
|
||||||
strm.block_size = pixels_per_block;
|
strm.block_size = pixels_per_block as u32;
|
||||||
strm.rsi = 128; // HDF5 default: 128 blocks per reference sample interval
|
strm.rsi = rsi as u32;
|
||||||
strm.flags = flags;
|
strm.flags = flags;
|
||||||
|
// SAFETY: next_in/avail_in and next_out/avail_out describe live buffers
|
||||||
|
// (`stream` and `buf`) that outlive the call.
|
||||||
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
||||||
if result != 0 {
|
if result != 0 {
|
||||||
return Err(FormatError::DecompressionError(format!(
|
return Err(FormatError::DecompressionError(format!(
|
||||||
"szip: libaec error {result}"
|
"szip: libaec error {result}"
|
||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
let decoded_len = chunk_size - strm.avail_out;
|
let mut total_out = strm.total_out;
|
||||||
out.truncate(decoded_len);
|
if pad_scanline {
|
||||||
Ok(out)
|
let line = pixels_per_scanline * pixel_size;
|
||||||
|
let padded_line = rsi * pixels_per_block * pixel_size;
|
||||||
|
// remove_padding: compact each padded line down to `line` bytes.
|
||||||
|
let mut i = line;
|
||||||
|
let mut j = padded_line;
|
||||||
|
while j < total_out {
|
||||||
|
let end = (j + line).min(buf.len());
|
||||||
|
buf.copy_within(j..end, i);
|
||||||
|
i += line;
|
||||||
|
j += padded_line;
|
||||||
|
}
|
||||||
|
total_out = scanlines * line;
|
||||||
|
}
|
||||||
|
if total_out < dest_len {
|
||||||
|
return Err(err("stream decoded to fewer bytes than declared"));
|
||||||
|
}
|
||||||
|
buf.truncate(dest_len);
|
||||||
|
if deinterleave {
|
||||||
|
// deinterleave_buffer: byte planes back into words.
|
||||||
|
let w = (bits_per_pixel / 8) as usize;
|
||||||
|
let n = dest_len / w;
|
||||||
|
let mut out = vec![0u8; dest_len];
|
||||||
|
for i in 0..n {
|
||||||
|
for j in 0..w {
|
||||||
|
out[i * w + j] = buf[j * n + i];
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
} else {
|
||||||
|
Ok(buf)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
fn unhex(s: &str) -> Vec<u8> {
|
||||||
|
(0..s.len())
|
||||||
|
.step_by(2)
|
||||||
|
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// SZIP chunks written by libhdf5, decoded exactly as libhdf5 decodes
|
||||||
|
/// them. Each case: fixture, chunk byte offset and size (from h5py's
|
||||||
|
/// `get_chunk_info`), the filter's cd_values, and the chunk's values as
|
||||||
|
/// h5py reads them (file byte order, hex). Before the fix every one of
|
||||||
|
/// these came back as garbage or zeros (or "invalid bits per sample" for
|
||||||
|
/// 64-bit): the 4-byte size prefix was fed to libaec, 32/64-bit samples
|
||||||
|
/// were not de-interleaved from byte planes, the reference sample
|
||||||
|
/// interval was fixed at 128 instead of derived from the scanline, padded
|
||||||
|
/// scanlines were not unpadded, and LE data was decoded as MSB.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
#[test]
|
||||||
|
fn szip_decodes_libhdf5_chunks_exactly() {
|
||||||
|
/// (name, file, chunk offset, chunk size, cd_values, decoded hex)
|
||||||
|
type Case<'a> = (&'a str, &'a [u8], usize, usize, [u32; 4], &'a str);
|
||||||
|
let noencoder: &[u8] = include_bytes!("../tests/fixtures/filters/noencoder.h5");
|
||||||
|
let le_data: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||||
|
let h5py: &[u8] = include_bytes!("../tests/fixtures/filters/szip_h5py.h5");
|
||||||
|
#[rustfmt::skip]
|
||||||
|
let cases: &[Case] = &[
|
||||||
|
// <i4, 10 px/scanline over 4 px/block: padded scanlines + byte planes.
|
||||||
|
("noencoder /noencoder_szip_dset.h5", noencoder, 6040, 16, [168, 4, 32, 10],
|
||||||
|
"00000000010000000200000003000000040000000500000006000000070000000800000009000000"),
|
||||||
|
// <f4, LSB + NN.
|
||||||
|
("le_data /Szip_float_data_le", le_data, 55224, 48, [169, 4, 32, 12],
|
||||||
|
"abaaaa3eabaa2a3f0000803fabaa2a3f0000803fabaaaa3f0000803fabaaaa3f5555d53fabaaaa3f5555d53f00000040"),
|
||||||
|
// >f4, MSB + NN.
|
||||||
|
("le_data /Szip_float_data_be", le_data, 55396, 48, [177, 4, 32, 12],
|
||||||
|
"3eaaaaab3f2aaaab3f8000003f2aaaab3f8000003faaaaab3f8000003faaaaab3fd555553faaaaab3fd5555540000000"),
|
||||||
|
// <f8 (64-bit), NN.
|
||||||
|
("szip_h5py /f8", h5py, 4016, 100, [169, 8, 64, 10],
|
||||||
|
"00000000000008c000000000000008c000000000000008c000000000000008c000000000000004c000000000000004c000000000000004c000000000000004c000000000000000c000000000000000c000000000000000c000000000000000c0000000000000f8bf000000000000f8bf000000000000f8bf000000000000f8bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000e0bf000000000000e0bf000000000000e0bf000000000000e0bf0000000000000000000000000000000000000000000000000000000000000000000000000000e03f000000000000e03f000000000000e03f000000000000e03f000000000000f03f000000000000f03f000000000000f03f000000000000f03f000000000000f83f000000000000f83f000000000000f83f000000000000f83f"),
|
||||||
|
// <i8 (64-bit), entropy coding without NN.
|
||||||
|
("szip_h5py /i8", h5py, 4188, 53, [141, 4, 64, 10],
|
||||||
|
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c00000000000000"),
|
||||||
|
// <u2, 35 px/scanline over 8 px/block: padded scanlines, 16-bit samples.
|
||||||
|
("szip_h5py /u2", h5py, 4308, 43, [169, 8, 16, 35],
|
||||||
|
"00000000000000006100610061006100c200c200c200c20023012301230123018401840184018401e501e501e501e5014602460246024602a702a702a702a702080308030803"),
|
||||||
|
];
|
||||||
|
for (name, file, off, len, cd, want) in cases {
|
||||||
|
let want = unhex(want);
|
||||||
|
let got = szip_decompress(&file[*off..off + len], cd, want.len())
|
||||||
|
.unwrap_or_else(|e| panic!("{name}: {e:?}"));
|
||||||
|
assert_eq!(got, want, "{name}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn szip_disabled_returns_unsupported() {
|
fn szip_disabled_returns_unsupported() {
|
||||||
#[cfg(not(feature = "szip"))]
|
#[cfg(not(feature = "szip"))]
|
||||||
@@ -132,6 +264,8 @@ mod tests {
|
|||||||
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
||||||
let enc_len = encoded.len() - enc.avail_out;
|
let enc_len = encoded.len() - enc.avail_out;
|
||||||
encoded.truncate(enc_len);
|
encoded.truncate(enc_len);
|
||||||
|
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||||
|
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||||
|
|
||||||
// Decode through our public interface.
|
// Decode through our public interface.
|
||||||
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
||||||
@@ -163,6 +297,8 @@ mod tests {
|
|||||||
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
||||||
let enc_len = encoded.len() - enc.avail_out;
|
let enc_len = encoded.len() - enc.avail_out;
|
||||||
encoded.truncate(enc_len);
|
encoded.truncate(enc_len);
|
||||||
|
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||||
|
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||||
|
|
||||||
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
||||||
let cd = [0x20u32, 8, 8, 1024];
|
let cd = [0x20u32, 8, 8, 1024];
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ extern crate alloc;
|
|||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::{format, vec, vec::Vec};
|
use alloc::{format, vec, vec::Vec};
|
||||||
|
|
||||||
|
use crate::chunk_grid::ChunkGrid;
|
||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
@@ -151,13 +152,13 @@ pub fn read_fixed_array_chunks(
|
|||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
header: &FixedArrayHeader,
|
header: &FixedArrayHeader,
|
||||||
dataset_dims: &[u64],
|
dataset_dims: &[u64],
|
||||||
|
max_dims: Option<&[u64]>,
|
||||||
chunk_dimensions: &[u32],
|
chunk_dimensions: &[u32],
|
||||||
element_size: u32,
|
element_size: u32,
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
_length_size: u8,
|
_length_size: u8,
|
||||||
) -> Result<Vec<ChunkInfo>, FormatError> {
|
) -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let db_offset = header.data_block_address as usize;
|
let db_offset = header.data_block_address as usize;
|
||||||
let rank = chunk_dimensions.len();
|
|
||||||
|
|
||||||
// Parse data block header: FADB(4) + version(1) + client_id(1) + header_address(offset_size)
|
// Parse data block header: FADB(4) + version(1) + client_id(1) + header_address(offset_size)
|
||||||
let db_header_size = 4 + 1 + 1 + offset_size as usize;
|
let db_header_size = 4 + 1 + 1 + offset_size as usize;
|
||||||
@@ -198,19 +199,10 @@ pub fn read_fixed_array_chunks(
|
|||||||
))
|
))
|
||||||
};
|
};
|
||||||
|
|
||||||
// Compute chunk offsets based on index.
|
// The index is laid out over the chunk grid of the *maximum* dimensions
|
||||||
// Chunks are stored in row-major order within the dataset space.
|
// (row-major), so a dataset smaller than its maxshape has gaps.
|
||||||
let mut num_chunks_per_dim = Vec::with_capacity(rank);
|
let dims_u64: Vec<u64> = chunk_dimensions.iter().map(|&d| d as u64).collect();
|
||||||
for d_idx in 0..rank {
|
let grid = ChunkGrid::fixed_array(dataset_dims, max_dims, &dims_u64)?;
|
||||||
let ch_dim = chunk_dimensions[d_idx] as u64;
|
|
||||||
if ch_dim == 0 {
|
|
||||||
return Err(FormatError::ChunkedReadError(
|
|
||||||
"chunk dimension is zero".into(),
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let ds_dim = dataset_dims[d_idx];
|
|
||||||
num_chunks_per_dim.push(ds_dim.div_ceil(ch_dim));
|
|
||||||
}
|
|
||||||
|
|
||||||
let chunk_byte_size: u64 =
|
let chunk_byte_size: u64 =
|
||||||
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
chunk_dimensions.iter().map(|&d| d as u64).product::<u64>() * element_size as u64;
|
||||||
@@ -226,7 +218,11 @@ pub fn read_fixed_array_chunks(
|
|||||||
header.element_size,
|
header.element_size,
|
||||||
chunk_byte_size,
|
chunk_byte_size,
|
||||||
)? {
|
)? {
|
||||||
let offsets = index_to_chunk_offsets(i, &num_chunks_per_dim, chunk_dimensions);
|
// A slot beyond the current extent is ignored, as the
|
||||||
|
// library does.
|
||||||
|
let Some(offsets) = grid.offsets(i as u64) else {
|
||||||
|
return Ok(());
|
||||||
|
};
|
||||||
chunks.push(ChunkInfo {
|
chunks.push(ChunkInfo {
|
||||||
chunk_size,
|
chunk_size,
|
||||||
filter_mask,
|
filter_mask,
|
||||||
@@ -367,27 +363,6 @@ fn parse_fa_element(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Convert a linear chunk index to N-dimensional chunk offsets in dataset space.
|
|
||||||
fn index_to_chunk_offsets(
|
|
||||||
index: usize,
|
|
||||||
num_chunks_per_dim: &[u64],
|
|
||||||
chunk_dimensions: &[u32],
|
|
||||||
) -> Vec<u64> {
|
|
||||||
let rank = num_chunks_per_dim.len();
|
|
||||||
let mut offsets = vec![0u64; rank];
|
|
||||||
let mut remaining = index as u64;
|
|
||||||
for d in (0..rank).rev() {
|
|
||||||
let nchunks = num_chunks_per_dim[d];
|
|
||||||
if nchunks == 0 {
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
let chunk_idx = remaining % nchunks;
|
|
||||||
remaining /= nchunks;
|
|
||||||
offsets[d] = chunk_idx * chunk_dimensions[d] as u64;
|
|
||||||
}
|
|
||||||
offsets
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Read a variable-length little-endian unsigned integer.
|
/// Read a variable-length little-endian unsigned integer.
|
||||||
fn read_variable_length(data: &[u8], size: usize) -> Result<u64, FormatError> {
|
fn read_variable_length(data: &[u8], size: usize) -> Result<u64, FormatError> {
|
||||||
if size > 8 || data.len() < size {
|
if size > 8 || data.len() < size {
|
||||||
@@ -416,44 +391,21 @@ mod tests {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_1d() {
|
fn index_to_offsets_1d() {
|
||||||
let num_chunks = vec![5u64];
|
let g = ChunkGrid::fixed_array(&[100], None, &[20]).unwrap();
|
||||||
let chunk_dims = vec![20u32];
|
assert_eq!(g.offsets(0).unwrap(), vec![0]);
|
||||||
assert_eq!(index_to_chunk_offsets(0, &num_chunks, &chunk_dims), vec![0]);
|
assert_eq!(g.offsets(1).unwrap(), vec![20]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(4).unwrap(), vec![80]);
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![20]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(4, &num_chunks, &chunk_dims),
|
|
||||||
vec![80]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn index_to_offsets_2d() {
|
fn index_to_offsets_2d() {
|
||||||
// 10x6 dataset with 4x3 chunks => ceil(10/4)=3, ceil(6/3)=2 => 6 chunks
|
// 10x6 dataset with 4x3 chunks => ceil(10/4)=3, ceil(6/3)=2 => 6 chunks
|
||||||
let num_chunks = vec![3u64, 2];
|
let g = ChunkGrid::fixed_array(&[10, 6], None, &[4, 3]).unwrap();
|
||||||
let chunk_dims = vec![4u32, 3];
|
assert_eq!(g.offsets(0).unwrap(), vec![0, 0]);
|
||||||
assert_eq!(
|
assert_eq!(g.offsets(1).unwrap(), vec![0, 3]);
|
||||||
index_to_chunk_offsets(0, &num_chunks, &chunk_dims),
|
assert_eq!(g.offsets(2).unwrap(), vec![4, 0]);
|
||||||
vec![0, 0]
|
assert_eq!(g.offsets(3).unwrap(), vec![4, 3]);
|
||||||
);
|
assert_eq!(g.offsets(5).unwrap(), vec![8, 3]);
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(1, &num_chunks, &chunk_dims),
|
|
||||||
vec![0, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(2, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 0]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(3, &num_chunks, &chunk_dims),
|
|
||||||
vec![4, 3]
|
|
||||||
);
|
|
||||||
assert_eq!(
|
|
||||||
index_to_chunk_offsets(5, &num_chunks, &chunk_dims),
|
|
||||||
vec![8, 3]
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -517,7 +469,7 @@ mod tests {
|
|||||||
|
|
||||||
let read = |f: &[u8], fahd: usize| -> Result<Vec<ChunkInfo>, FormatError> {
|
let read = |f: &[u8], fahd: usize| -> Result<Vec<ChunkInfo>, FormatError> {
|
||||||
let h = FixedArrayHeader::parse(f, fahd, 8, 8)?;
|
let h = FixedArrayHeader::parse(f, fahd, 8, 8)?;
|
||||||
read_fixed_array_chunks(f, &h, &[60], &[20], 8, 8, 8)
|
read_fixed_array_chunks(f, &h, &[60], None, &[20], 8, 8, 8)
|
||||||
};
|
};
|
||||||
|
|
||||||
let (clean, fahd) = build();
|
let (clean, fahd) = build();
|
||||||
@@ -562,7 +514,7 @@ mod tests {
|
|||||||
let db = 0x100usize;
|
let db = 0x100usize;
|
||||||
buf[db..db + 4].copy_from_slice(b"FADB");
|
buf[db..db + 4].copy_from_slice(b"FADB");
|
||||||
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -579,7 +531,7 @@ mod tests {
|
|||||||
stamp_checksum(&mut buf, fahd, fahd + 24);
|
stamp_checksum(&mut buf, fahd, fahd + 24);
|
||||||
buf[0x80..0x84].copy_from_slice(b"FADB");
|
buf[0x80..0x84].copy_from_slice(b"FADB");
|
||||||
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
let header = FixedArrayHeader::parse(&buf, fahd, 8, 8).unwrap();
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -602,7 +554,7 @@ mod tests {
|
|||||||
data_block_address: (usize::MAX - 4) as u64,
|
data_block_address: (usize::MAX - 4) as u64,
|
||||||
};
|
};
|
||||||
let buf = vec![0u8; 64];
|
let buf = vec![0u8; 64];
|
||||||
let r = read_fixed_array_chunks(&buf, &header, &[100], &[20], 8, 8, 8);
|
let r = read_fixed_array_chunks(&buf, &header, &[100], None, &[20], 8, 8, 8);
|
||||||
assert!(r.is_err());
|
assert!(r.is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -664,6 +616,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -740,6 +693,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
@@ -840,6 +794,7 @@ mod tests {
|
|||||||
&file_data,
|
&file_data,
|
||||||
&header,
|
&header,
|
||||||
&ds_dims,
|
&ds_dims,
|
||||||
|
None,
|
||||||
&chunk_dims,
|
&chunk_dims,
|
||||||
8,
|
8,
|
||||||
offset_size,
|
offset_size,
|
||||||
|
|||||||
@@ -1,12 +1,14 @@
|
|||||||
//! HDF5 Fractal Heap parsing for v2 group link storage.
|
//! HDF5 Fractal Heap parsing for v2 group link storage.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::{format, vec::Vec};
|
||||||
|
|
||||||
#[cfg(feature = "checksum")]
|
#[cfg(feature = "checksum")]
|
||||||
use byteorder::{ByteOrder, LittleEndian};
|
use byteorder::{ByteOrder, LittleEndian};
|
||||||
|
|
||||||
|
use crate::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
use crate::filter_pipeline::FilterPipeline;
|
||||||
|
|
||||||
/// Parsed fractal heap header (signature "FRHP").
|
/// Parsed fractal heap header (signature "FRHP").
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -33,6 +35,23 @@ pub struct FractalHeapHeader {
|
|||||||
pub current_rows_in_root_indirect_block: u16,
|
pub current_rows_in_root_indirect_block: u16,
|
||||||
/// Total number of managed objects.
|
/// Total number of managed objects.
|
||||||
pub managed_objects_count: u64,
|
pub managed_objects_count: u64,
|
||||||
|
/// Address of the v2 B-tree indexing "huge" objects (undefined address
|
||||||
|
/// when the heap has none). Huge objects are those larger than
|
||||||
|
/// `max_managed_object_size`; they live outside the heap's blocks.
|
||||||
|
pub huge_btree_address: u64,
|
||||||
|
/// The heap's I/O filter pipeline, if it has one. It applies to managed
|
||||||
|
/// direct blocks and to huge objects.
|
||||||
|
pub filter_pipeline: Option<FilterPipeline>,
|
||||||
|
/// Stored (filtered) size of the root direct block; meaningful only when
|
||||||
|
/// the heap is filtered and its root is a direct block.
|
||||||
|
pub root_direct_block_filtered_size: u64,
|
||||||
|
/// Filter mask of the root direct block (bit *i* set = filter *i*
|
||||||
|
/// skipped); meaningful only when the heap is filtered.
|
||||||
|
pub root_direct_block_filter_mask: u32,
|
||||||
|
/// Size of addresses in the file ("Size of Offsets").
|
||||||
|
pub offset_size: u8,
|
||||||
|
/// Size of lengths in the file ("Size of Lengths").
|
||||||
|
pub length_size: u8,
|
||||||
}
|
}
|
||||||
|
|
||||||
fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
|
||||||
@@ -79,6 +98,38 @@ fn is_undefined(val: u64, offset_size: u8) -> bool {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Little-endian unsigned integer of up to 8 bytes.
|
||||||
|
fn le_uint(bytes: &[u8]) -> u64 {
|
||||||
|
bytes
|
||||||
|
.iter()
|
||||||
|
.take(8)
|
||||||
|
.enumerate()
|
||||||
|
.fold(0u64, |acc, (i, &b)| acc | (u64::from(b) << (i * 8)))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn heap_error(msg: &str) -> FormatError {
|
||||||
|
FormatError::ChunkedReadError(format!("fractal heap: {msg}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Heap ID type, from bits 4-5 of an ID's first byte (libhdf5's
|
||||||
|
/// `H5HF_ID_TYPE_MASK`, 0x30); bits 6-7 are the ID version, which must be 0.
|
||||||
|
const HEAP_ID_MANAGED: u8 = 0;
|
||||||
|
const HEAP_ID_HUGE: u8 = 1;
|
||||||
|
const HEAP_ID_TINY: u8 = 2;
|
||||||
|
|
||||||
|
/// The type (0 managed, 1 huge, 2 tiny) of a heap ID from its first byte,
|
||||||
|
/// refusing an ID version other than 0.
|
||||||
|
fn heap_id_type(first: u8) -> Result<u8, FormatError> {
|
||||||
|
if first >> 6 != 0 {
|
||||||
|
return Err(heap_error("unsupported heap ID version"));
|
||||||
|
}
|
||||||
|
Ok((first >> 4) & 0x03)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// v2 B-tree record types indexing a heap's huge objects.
|
||||||
|
const BTREE_HUGE_INDIRECT: u8 = 1;
|
||||||
|
const BTREE_HUGE_INDIRECT_FILTERED: u8 = 2;
|
||||||
|
|
||||||
impl FractalHeapHeader {
|
impl FractalHeapHeader {
|
||||||
/// Parse a fractal heap header at the given offset.
|
/// Parse a fractal heap header at the given offset.
|
||||||
pub fn parse(
|
pub fn parse(
|
||||||
@@ -122,11 +173,17 @@ impl FractalHeapHeader {
|
|||||||
]);
|
]);
|
||||||
pos += 4;
|
pos += 4;
|
||||||
|
|
||||||
// Skip several fixed fields: next_huge_object_id(ls), btree_huge_objects_address(os),
|
// next_huge_object_id (length_size)
|
||||||
// free_space_managed_blocks(ls), managed_block_free_space_manager_address(os),
|
ensure_len(file_data, pos, ls)?;
|
||||||
|
pos += ls;
|
||||||
|
// btree_huge_objects_address (offset_size)
|
||||||
|
let huge_btree_address = read_offset(file_data, pos, offset_size)?;
|
||||||
|
pos += os;
|
||||||
|
|
||||||
|
// Skip: free_space_managed_blocks(ls), managed_block_free_space_manager_address(os),
|
||||||
// managed_space_in_heap(ls), allocated_managed_space_in_heap(ls),
|
// managed_space_in_heap(ls), allocated_managed_space_in_heap(ls),
|
||||||
// direct_block_allocation_iterator_offset(ls)
|
// direct_block_allocation_iterator_offset(ls)
|
||||||
let skip_size = 5 * ls + 2 * os;
|
let skip_size = 4 * ls + os;
|
||||||
ensure_len(file_data, pos, skip_size)?;
|
ensure_len(file_data, pos, skip_size)?;
|
||||||
pos += skip_size;
|
pos += skip_size;
|
||||||
|
|
||||||
@@ -134,14 +191,9 @@ impl FractalHeapHeader {
|
|||||||
let managed_objects_count = read_offset(file_data, pos, length_size)?;
|
let managed_objects_count = read_offset(file_data, pos, length_size)?;
|
||||||
pos += ls;
|
pos += ls;
|
||||||
|
|
||||||
// huge_objects_size (length_size)
|
// huge_objects_size, huge_objects_count, tiny_objects_size,
|
||||||
pos += ls;
|
// tiny_objects_count (length_size each)
|
||||||
// huge_objects_count (length_size)
|
pos += 4 * ls;
|
||||||
pos += ls;
|
|
||||||
// tiny_objects_size (length_size)
|
|
||||||
pos += ls;
|
|
||||||
// tiny_objects_count (length_size)
|
|
||||||
pos += ls;
|
|
||||||
|
|
||||||
// table_width (2)
|
// table_width (2)
|
||||||
ensure_len(file_data, pos, 2)?;
|
ensure_len(file_data, pos, 2)?;
|
||||||
@@ -175,16 +227,28 @@ impl FractalHeapHeader {
|
|||||||
ensure_len(file_data, pos, 2)?;
|
ensure_len(file_data, pos, 2)?;
|
||||||
let current_rows_in_root_indirect_block =
|
let current_rows_in_root_indirect_block =
|
||||||
u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
||||||
#[allow(unused_variables, unused_mut, unused_assignments)]
|
pos += 2;
|
||||||
let mut pos = pos + 2;
|
|
||||||
|
|
||||||
// Skip IO filter encoded info if present
|
// With I/O filters: root direct block's filtered size (length_size),
|
||||||
|
// its filter mask (4), then the encoded filter pipeline message.
|
||||||
|
let mut filter_pipeline = None;
|
||||||
|
let mut root_direct_block_filtered_size = 0;
|
||||||
|
let mut root_direct_block_filter_mask = 0;
|
||||||
if io_filter_encoded_length > 0 {
|
if io_filter_encoded_length > 0 {
|
||||||
// root_block_filter_info_size (length_size) + filter_mask (4)
|
root_direct_block_filtered_size = read_offset(file_data, pos, length_size)?;
|
||||||
#[allow(unused_assignments)]
|
pos += ls;
|
||||||
{
|
ensure_len(file_data, pos, 4)?;
|
||||||
pos += ls + 4;
|
root_direct_block_filter_mask = u32::from_le_bytes([
|
||||||
}
|
file_data[pos],
|
||||||
|
file_data[pos + 1],
|
||||||
|
file_data[pos + 2],
|
||||||
|
file_data[pos + 3],
|
||||||
|
]);
|
||||||
|
pos += 4;
|
||||||
|
let n = io_filter_encoded_length as usize;
|
||||||
|
ensure_len(file_data, pos, n)?;
|
||||||
|
filter_pipeline = Some(FilterPipeline::parse(&file_data[pos..pos + n])?);
|
||||||
|
pos += n;
|
||||||
}
|
}
|
||||||
|
|
||||||
// Validate header checksum
|
// Validate header checksum
|
||||||
@@ -200,6 +264,8 @@ impl FractalHeapHeader {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
#[cfg(not(feature = "checksum"))]
|
||||||
|
let _ = pos;
|
||||||
|
|
||||||
Ok(FractalHeapHeader {
|
Ok(FractalHeapHeader {
|
||||||
heap_id_length,
|
heap_id_length,
|
||||||
@@ -213,13 +279,19 @@ impl FractalHeapHeader {
|
|||||||
root_block_address,
|
root_block_address,
|
||||||
current_rows_in_root_indirect_block,
|
current_rows_in_root_indirect_block,
|
||||||
managed_objects_count,
|
managed_objects_count,
|
||||||
|
huge_btree_address,
|
||||||
|
filter_pipeline,
|
||||||
|
root_direct_block_filtered_size,
|
||||||
|
root_direct_block_filter_mask,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Decode a managed heap ID into (offset_in_heap, object_length).
|
/// Decode a managed heap ID into (offset_in_heap, object_length).
|
||||||
///
|
///
|
||||||
/// The heap ID layout for managed objects (type 0):
|
/// The heap ID layout for managed objects (type 0):
|
||||||
/// - Byte 0: bits 6-7 = type (0), bits 4-5 = version (0), bits 0-3 = reserved
|
/// - Byte 0: bits 6-7 = version (0), bits 4-5 = type (0), bits 0-3 = reserved
|
||||||
/// - Bytes 1+: offset (max_heap_size bits, LE) then length (remaining bits, LE)
|
/// - Bytes 1+: offset (max_heap_size bits, LE) then length (remaining bits, LE)
|
||||||
pub fn decode_managed_id(&self, id_bytes: &[u8]) -> Result<(u64, u64), FormatError> {
|
pub fn decode_managed_id(&self, id_bytes: &[u8]) -> Result<(u64, u64), FormatError> {
|
||||||
if id_bytes.is_empty() {
|
if id_bytes.is_empty() {
|
||||||
@@ -229,8 +301,8 @@ impl FractalHeapHeader {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
let id_type = (id_bytes[0] >> 6) & 0x03;
|
let id_type = heap_id_type(id_bytes[0])?;
|
||||||
if id_type != 0 {
|
if id_type != HEAP_ID_MANAGED {
|
||||||
return Err(FormatError::InvalidHeapIdType(id_type));
|
return Err(FormatError::InvalidHeapIdType(id_type));
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -269,12 +341,183 @@ impl FractalHeapHeader {
|
|||||||
Ok((heap_offset, length_val))
|
Ok((heap_offset, length_val))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Read a managed object from the heap given its raw heap ID bytes.
|
/// Read any object from the heap given its raw heap ID bytes: managed
|
||||||
|
/// (stored in the heap's blocks), huge (stored outside them, found
|
||||||
|
/// directly from the ID or through the huge-object v2 B-tree, optionally
|
||||||
|
/// filtered) or tiny (stored in the ID itself).
|
||||||
|
///
|
||||||
|
/// Despite its name this accepts every ID type; `offset_size` must match
|
||||||
|
/// the one the header was parsed with.
|
||||||
pub fn read_managed_object(
|
pub fn read_managed_object(
|
||||||
&self,
|
&self,
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
id_bytes: &[u8],
|
id_bytes: &[u8],
|
||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let Some(&first) = id_bytes.first() else {
|
||||||
|
return Err(FormatError::UnexpectedEof {
|
||||||
|
expected: 1,
|
||||||
|
available: 0,
|
||||||
|
});
|
||||||
|
};
|
||||||
|
match heap_id_type(first)? {
|
||||||
|
HEAP_ID_MANAGED => self.read_heap_managed(file_data, id_bytes, offset_size),
|
||||||
|
HEAP_ID_HUGE => self.read_huge_object(file_data, id_bytes),
|
||||||
|
HEAP_ID_TINY => self.read_tiny_object(id_bytes),
|
||||||
|
other => Err(FormatError::InvalidHeapIdType(other)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a huge object's ID holds its address and length directly
|
||||||
|
/// (libhdf5 does this when they fit in the ID), rather than a key into
|
||||||
|
/// the huge-object B-tree.
|
||||||
|
fn huge_ids_direct(&self) -> bool {
|
||||||
|
let room = usize::from(self.heap_id_length).saturating_sub(1);
|
||||||
|
let os = usize::from(self.offset_size);
|
||||||
|
let ls = usize::from(self.length_size);
|
||||||
|
if self.filter_pipeline.is_some() {
|
||||||
|
room >= os + ls + 4 + ls
|
||||||
|
} else {
|
||||||
|
room >= os + ls
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a huge object (heap ID type 1).
|
||||||
|
fn read_huge_object(&self, file_data: &[u8], id: &[u8]) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let os = usize::from(self.offset_size);
|
||||||
|
let ls = usize::from(self.length_size);
|
||||||
|
// (address, stored length, filter mask, decoded length); the last two
|
||||||
|
// only matter for a filtered heap.
|
||||||
|
let (addr, stored_len, mask, mem_len) = if self.huge_ids_direct() {
|
||||||
|
let body = &id[1..];
|
||||||
|
let need = if self.filter_pipeline.is_some() {
|
||||||
|
os + ls + 4 + ls
|
||||||
|
} else {
|
||||||
|
os + ls
|
||||||
|
};
|
||||||
|
ensure_len(body, 0, need)?;
|
||||||
|
let addr = le_uint(&body[..os]);
|
||||||
|
let len = le_uint(&body[os..os + ls]);
|
||||||
|
if self.filter_pipeline.is_some() {
|
||||||
|
let mask = u32::from_le_bytes([
|
||||||
|
body[os + ls],
|
||||||
|
body[os + ls + 1],
|
||||||
|
body[os + ls + 2],
|
||||||
|
body[os + ls + 3],
|
||||||
|
]);
|
||||||
|
let mem = le_uint(&body[os + ls + 4..os + ls + 4 + ls]);
|
||||||
|
(addr, len, mask, mem)
|
||||||
|
} else {
|
||||||
|
(addr, len, 0, len)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
let key_len = (usize::from(self.heap_id_length).saturating_sub(1)).min(8);
|
||||||
|
ensure_len(id, 1, key_len)?;
|
||||||
|
let key = le_uint(&id[1..1 + key_len]);
|
||||||
|
self.find_huge_record(file_data, key)?
|
||||||
|
};
|
||||||
|
|
||||||
|
let start = usize::try_from(addr).map_err(|_| heap_error("huge object address"))?;
|
||||||
|
let len = usize::try_from(stored_len).map_err(|_| heap_error("huge object length"))?;
|
||||||
|
ensure_len(file_data, start, len)?;
|
||||||
|
let stored = &file_data[start..start + len];
|
||||||
|
match &self.filter_pipeline {
|
||||||
|
None => Ok(stored.to_vec()),
|
||||||
|
Some(pipeline) => {
|
||||||
|
let mem = usize::try_from(mem_len).map_err(|_| heap_error("huge object size"))?;
|
||||||
|
let out = crate::filters::decompress_chunk_masked(stored, pipeline, mem, 1, mask)?;
|
||||||
|
if out.len() != mem {
|
||||||
|
return Err(heap_error("filtered huge object decoded to the wrong size"));
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Look up huge object `key` in the huge-object v2 B-tree, returning
|
||||||
|
/// (address, stored length, filter mask, decoded length).
|
||||||
|
fn find_huge_record(
|
||||||
|
&self,
|
||||||
|
file_data: &[u8],
|
||||||
|
key: u64,
|
||||||
|
) -> Result<(u64, u64, u32, u64), FormatError> {
|
||||||
|
if is_undefined(self.huge_btree_address, self.offset_size) {
|
||||||
|
return Err(heap_error(
|
||||||
|
"huge object ID but the heap has no huge-object index",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let hdr = BTreeV2Header::parse(
|
||||||
|
file_data,
|
||||||
|
self.huge_btree_address as usize,
|
||||||
|
self.offset_size,
|
||||||
|
self.length_size,
|
||||||
|
)?;
|
||||||
|
let os = usize::from(self.offset_size);
|
||||||
|
let ls = usize::from(self.length_size);
|
||||||
|
let filtered = self.filter_pipeline.is_some();
|
||||||
|
let (expected_type, rec_len) = if filtered {
|
||||||
|
(BTREE_HUGE_INDIRECT_FILTERED, os + ls + 4 + ls + ls)
|
||||||
|
} else {
|
||||||
|
(BTREE_HUGE_INDIRECT, os + ls + ls)
|
||||||
|
};
|
||||||
|
if hdr.tree_type != expected_type || usize::from(hdr.record_size) < rec_len {
|
||||||
|
return Err(heap_error("unexpected huge-object B-tree record type"));
|
||||||
|
}
|
||||||
|
let records =
|
||||||
|
collect_btree_v2_records(file_data, &hdr, self.offset_size, self.length_size)?;
|
||||||
|
for rec in &records {
|
||||||
|
let d = &rec.data;
|
||||||
|
if d.len() < rec_len {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let addr = le_uint(&d[..os]);
|
||||||
|
let len = le_uint(&d[os..os + ls]);
|
||||||
|
if filtered {
|
||||||
|
let mask = u32::from_le_bytes([
|
||||||
|
d[os + ls],
|
||||||
|
d[os + ls + 1],
|
||||||
|
d[os + ls + 2],
|
||||||
|
d[os + ls + 3],
|
||||||
|
]);
|
||||||
|
let mem = le_uint(&d[os + ls + 4..os + 2 * ls + 4]);
|
||||||
|
let id = le_uint(&d[os + 2 * ls + 4..os + 3 * ls + 4]);
|
||||||
|
if id == key {
|
||||||
|
return Ok((addr, len, mask, mem));
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
let id = le_uint(&d[os + ls..os + 2 * ls]);
|
||||||
|
if id == key {
|
||||||
|
return Ok((addr, len, 0, len));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(heap_error("huge object not found in its B-tree"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a tiny object (heap ID type 2), stored in the ID itself.
|
||||||
|
fn read_tiny_object(&self, id: &[u8]) -> Result<Vec<u8>, FormatError> {
|
||||||
|
// libhdf5 uses a one-byte length (low 4 bits of byte 0) unless the ID
|
||||||
|
// is long enough to need 12 bits, which then borrow byte 1.
|
||||||
|
let extended = usize::from(self.heap_id_length).saturating_sub(1) > 17;
|
||||||
|
let (len, start) = if extended {
|
||||||
|
ensure_len(id, 0, 2)?;
|
||||||
|
(
|
||||||
|
((usize::from(id[0] & 0x0F)) << 8 | usize::from(id[1])) + 1,
|
||||||
|
2,
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
(usize::from(id[0] & 0x0F) + 1, 1)
|
||||||
|
};
|
||||||
|
ensure_len(id, start, len)?;
|
||||||
|
Ok(id[start..start + len].to_vec())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a managed object (heap ID type 0).
|
||||||
|
fn read_heap_managed(
|
||||||
|
&self,
|
||||||
|
file_data: &[u8],
|
||||||
|
id_bytes: &[u8],
|
||||||
|
offset_size: u8,
|
||||||
) -> Result<Vec<u8>, FormatError> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
let (heap_offset, obj_len) = self.decode_managed_id(id_bytes)?;
|
let (heap_offset, obj_len) = self.decode_managed_id(id_bytes)?;
|
||||||
|
|
||||||
@@ -289,12 +532,15 @@ impl FractalHeapHeader {
|
|||||||
// Root is a direct block
|
// Root is a direct block
|
||||||
self.read_from_direct_block(
|
self.read_from_direct_block(
|
||||||
file_data,
|
file_data,
|
||||||
self.root_block_address as usize,
|
DirectBlock {
|
||||||
self.starting_block_size,
|
addr: self.root_block_address as usize,
|
||||||
0, // block offset in heap = 0 for root
|
size: self.starting_block_size,
|
||||||
|
heap_offset: 0,
|
||||||
|
filtered_size: self.root_direct_block_filtered_size,
|
||||||
|
filter_mask: self.root_direct_block_filter_mask,
|
||||||
|
},
|
||||||
heap_offset,
|
heap_offset,
|
||||||
obj_len as usize,
|
obj_len as usize,
|
||||||
offset_size,
|
|
||||||
)
|
)
|
||||||
} else {
|
} else {
|
||||||
// Root is an indirect block — limit recursion to 64 levels
|
// Root is an indirect block — limit recursion to 64 levels
|
||||||
@@ -313,27 +559,41 @@ impl FractalHeapHeader {
|
|||||||
|
|
||||||
/// Read an object from a direct block.
|
/// Read an object from a direct block.
|
||||||
///
|
///
|
||||||
/// The heap offset is relative to the start of the block (including its header),
|
/// The heap offset is relative to the start of the block (including its
|
||||||
/// so we just add it to the block address minus the block's heap offset.
|
/// header), so we just add it to the block address minus the block's heap
|
||||||
#[allow(clippy::too_many_arguments)]
|
/// offset. A filtered heap stores each direct block (header included)
|
||||||
|
/// through its filter pipeline, so the block is decoded first.
|
||||||
fn read_from_direct_block(
|
fn read_from_direct_block(
|
||||||
&self,
|
&self,
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
block_addr: usize,
|
block: DirectBlock,
|
||||||
_block_size: u64,
|
|
||||||
block_heap_offset: u64,
|
|
||||||
target_offset: u64,
|
target_offset: u64,
|
||||||
length: usize,
|
length: usize,
|
||||||
_offset_size: u8,
|
|
||||||
) -> Result<Vec<u8>, FormatError> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
if target_offset < block_heap_offset {
|
if target_offset < block.heap_offset {
|
||||||
return Err(FormatError::UnexpectedEof {
|
return Err(FormatError::UnexpectedEof {
|
||||||
expected: block_heap_offset as usize,
|
expected: block.heap_offset as usize,
|
||||||
available: target_offset as usize,
|
available: target_offset as usize,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
let local_offset = (target_offset - block_heap_offset) as usize;
|
let local_offset = (target_offset - block.heap_offset) as usize;
|
||||||
let pos = block_addr
|
if let Some(pipeline) = &self.filter_pipeline {
|
||||||
|
let stored_len = usize::try_from(block.filtered_size)
|
||||||
|
.map_err(|_| heap_error("direct block size"))?;
|
||||||
|
let size = usize::try_from(block.size).map_err(|_| heap_error("direct block size"))?;
|
||||||
|
ensure_len(file_data, block.addr, stored_len)?;
|
||||||
|
let decoded = crate::filters::decompress_chunk_masked(
|
||||||
|
&file_data[block.addr..block.addr + stored_len],
|
||||||
|
pipeline,
|
||||||
|
size,
|
||||||
|
1,
|
||||||
|
block.filter_mask,
|
||||||
|
)?;
|
||||||
|
ensure_len(&decoded, local_offset, length)?;
|
||||||
|
return Ok(decoded[local_offset..local_offset + length].to_vec());
|
||||||
|
}
|
||||||
|
let pos = block
|
||||||
|
.addr
|
||||||
.checked_add(local_offset)
|
.checked_add(local_offset)
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
.ok_or(FormatError::UnexpectedEof {
|
||||||
expected: usize::MAX,
|
expected: usize::MAX,
|
||||||
@@ -371,19 +631,13 @@ impl FractalHeapHeader {
|
|||||||
let iblock_header = 5 + offset_size as usize + block_offset_bytes;
|
let iblock_header = 5 + offset_size as usize + block_offset_bytes;
|
||||||
let mut pos = iblock_addr + iblock_header;
|
let mut pos = iblock_addr + iblock_header;
|
||||||
|
|
||||||
// Compute block sizes for each row using the doubling table
|
|
||||||
let tw = self.table_width as u64;
|
let tw = self.table_width as u64;
|
||||||
|
|
||||||
let nrows_usize = nrows as usize;
|
let nrows_usize = nrows as usize;
|
||||||
|
|
||||||
// Build table of (block_size, heap_offset) for each child entry
|
|
||||||
let mut current_heap_offset = iblock_heap_offset;
|
let mut current_heap_offset = iblock_heap_offset;
|
||||||
|
|
||||||
// Rows below max_direct_rows hold direct blocks; rows at/above hold
|
// Rows below max_direct_rows hold direct blocks; rows at/above hold
|
||||||
// child indirect blocks. (NOT the FRHP "starting rows" field.)
|
// child indirect blocks. (NOT the FRHP "starting rows" field.)
|
||||||
let start_indirect = self.max_direct_rows();
|
let start_indirect = self.max_direct_rows();
|
||||||
|
|
||||||
// Read child addresses for direct block rows
|
|
||||||
let max_direct_rows = nrows_usize.min(start_indirect);
|
let max_direct_rows = nrows_usize.min(start_indirect);
|
||||||
|
|
||||||
for row in 0..max_direct_rows {
|
for row in 0..max_direct_rows {
|
||||||
@@ -393,60 +647,74 @@ impl FractalHeapHeader {
|
|||||||
let child_addr = read_offset(file_data, pos, offset_size)?;
|
let child_addr = read_offset(file_data, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
|
|
||||||
if self.io_filter_encoded_length > 0 {
|
// A filtered heap stores each direct block's filtered size
|
||||||
// filtered_size(length_size) + filter_mask(4)
|
// (length_size) and filter mask (4) after its address.
|
||||||
// Skip for now - we don't handle filtered direct blocks in fractal heaps
|
let (filtered_size, filter_mask) = if self.filter_pipeline.is_some() {
|
||||||
pos += 4; // filter_mask - simplified
|
let size = read_offset(file_data, pos, self.length_size)?;
|
||||||
}
|
pos += usize::from(self.length_size);
|
||||||
|
ensure_len(file_data, pos, 4)?;
|
||||||
|
let mask = u32::from_le_bytes([
|
||||||
|
file_data[pos],
|
||||||
|
file_data[pos + 1],
|
||||||
|
file_data[pos + 2],
|
||||||
|
file_data[pos + 3],
|
||||||
|
]);
|
||||||
|
pos += 4;
|
||||||
|
(size, mask)
|
||||||
|
} else {
|
||||||
|
(0, 0)
|
||||||
|
};
|
||||||
|
|
||||||
if !is_undefined(child_addr, offset_size) {
|
let block_end = current_heap_offset.saturating_add(block_size);
|
||||||
let block_end = current_heap_offset + block_size;
|
if !is_undefined(child_addr, offset_size)
|
||||||
if target_offset >= current_heap_offset && target_offset < block_end {
|
&& target_offset >= current_heap_offset
|
||||||
return self.read_from_direct_block(
|
&& target_offset < block_end
|
||||||
file_data,
|
{
|
||||||
child_addr as usize,
|
return self.read_from_direct_block(
|
||||||
block_size,
|
file_data,
|
||||||
current_heap_offset,
|
DirectBlock {
|
||||||
target_offset,
|
addr: child_addr as usize,
|
||||||
length,
|
size: block_size,
|
||||||
offset_size,
|
heap_offset: current_heap_offset,
|
||||||
);
|
filtered_size,
|
||||||
}
|
filter_mask,
|
||||||
|
},
|
||||||
|
target_offset,
|
||||||
|
length,
|
||||||
|
);
|
||||||
}
|
}
|
||||||
current_heap_offset += block_size;
|
current_heap_offset = block_end;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// If we have indirect block rows
|
// Rows at and above `start_indirect` hold child indirect blocks. A
|
||||||
|
// child in row r spans exactly that row's block size of heap space,
|
||||||
|
// so it has as many rows as a table of that total size needs.
|
||||||
for row in start_indirect..nrows_usize {
|
for row in start_indirect..nrows_usize {
|
||||||
let _block_size = self.block_size_for_row(row);
|
let child_space = self.block_size_for_row(row);
|
||||||
let child_nrows = row - start_indirect + 1;
|
let child_nrows = self.rows_for_size(child_space);
|
||||||
|
|
||||||
for _col in 0..tw {
|
for _col in 0..tw {
|
||||||
let child_addr = read_offset(file_data, pos, offset_size)?;
|
let child_addr = read_offset(file_data, pos, offset_size)?;
|
||||||
pos += offset_size as usize;
|
pos += offset_size as usize;
|
||||||
|
|
||||||
if !is_undefined(child_addr, offset_size) {
|
let block_end = current_heap_offset.saturating_add(child_space);
|
||||||
// Calculate total heap space covered by this indirect block child
|
if !is_undefined(child_addr, offset_size)
|
||||||
let total_child_space = self.indirect_block_heap_size(child_nrows);
|
&& target_offset >= current_heap_offset
|
||||||
let block_end = current_heap_offset + total_child_space;
|
&& target_offset < block_end
|
||||||
if target_offset >= current_heap_offset && target_offset < block_end {
|
{
|
||||||
return self.read_from_indirect_block(
|
return self.read_from_indirect_block(
|
||||||
file_data,
|
file_data,
|
||||||
child_addr as usize,
|
child_addr as usize,
|
||||||
child_nrows as u16,
|
child_nrows,
|
||||||
current_heap_offset,
|
current_heap_offset,
|
||||||
target_offset,
|
target_offset,
|
||||||
length,
|
length,
|
||||||
offset_size,
|
offset_size,
|
||||||
depth_remaining - 1,
|
depth_remaining - 1,
|
||||||
);
|
);
|
||||||
}
|
|
||||||
current_heap_offset += total_child_space;
|
|
||||||
} else {
|
|
||||||
let total_child_space = self.indirect_block_heap_size(child_nrows);
|
|
||||||
current_heap_offset += total_child_space;
|
|
||||||
}
|
}
|
||||||
|
current_heap_offset = block_end;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -475,25 +743,34 @@ impl FractalHeapHeader {
|
|||||||
log2 + 2
|
log2 + 2
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Rows an indirect block needs to span `size` bytes of heap space:
|
||||||
|
/// `log2(size) - log2(starting_block_size * table_width) + 1`, as
|
||||||
|
/// libhdf5's `H5HF__dtable_size_to_rows`.
|
||||||
|
fn rows_for_size(&self, size: u64) -> u16 {
|
||||||
|
let log2 = |v: u64| 63u32.saturating_sub(v.max(1).leading_zeros());
|
||||||
|
let first_row_bits = log2(self.starting_block_size) + log2(u64::from(self.table_width));
|
||||||
|
(log2(size).saturating_sub(first_row_bits) + 1) as u16
|
||||||
|
}
|
||||||
|
|
||||||
/// Get block size for a given row in the doubling table.
|
/// Get block size for a given row in the doubling table.
|
||||||
fn block_size_for_row(&self, row: usize) -> u64 {
|
fn block_size_for_row(&self, row: usize) -> u64 {
|
||||||
let sbs = self.starting_block_size;
|
let sbs = self.starting_block_size;
|
||||||
if row <= 1 {
|
if row <= 1 {
|
||||||
sbs
|
sbs
|
||||||
} else {
|
} else {
|
||||||
sbs * (1u64 << (row - 1))
|
sbs.saturating_mul(1u64.checked_shl((row - 1) as u32).unwrap_or(u64::MAX))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Total heap space covered by an indirect block with the given number of rows.
|
/// A managed direct block's location, extent and (for a filtered heap) its
|
||||||
fn indirect_block_heap_size(&self, nrows: usize) -> u64 {
|
/// stored size and filter mask.
|
||||||
let tw = self.table_width as u64;
|
struct DirectBlock {
|
||||||
let mut total = 0u64;
|
addr: usize,
|
||||||
for row in 0..nrows {
|
size: u64,
|
||||||
total += self.block_size_for_row(row) * tw;
|
heap_offset: u64,
|
||||||
}
|
filtered_size: u64,
|
||||||
total
|
filter_mask: u32,
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -641,7 +918,7 @@ mod tests {
|
|||||||
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||||
|
|
||||||
// Build a managed heap ID:
|
// Build a managed heap ID:
|
||||||
// byte 0: type=0 (bits 6-7 = 00), version=0 (bits 4-5), reserved (bits 0-3)
|
// byte 0: version=0 (bits 6-7), type=0 (bits 4-5), reserved (bits 0-3)
|
||||||
// bytes 1-6: offset (max_heap_size=16 bits) then length (remaining bits)
|
// bytes 1-6: offset (max_heap_size=16 bits) then length (remaining bits)
|
||||||
// For offset=0, length=13:
|
// For offset=0, length=13:
|
||||||
// payload = offset | (length << 16) = 0 | (13 << 16) = 0x000D0000
|
// payload = offset | (length << 16) = 0 | (13 << 16) = 0x000D0000
|
||||||
@@ -705,9 +982,46 @@ mod tests {
|
|||||||
fn invalid_heap_id_type() {
|
fn invalid_heap_id_type() {
|
||||||
let (file_data, _) = build_simple_heap(8, 8);
|
let (file_data, _) = build_simple_heap(8, 8);
|
||||||
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||||
// Type = 1 (tiny) in bits 6-7
|
// Type = 1 (huge) in bits 4-5 is not a managed ID
|
||||||
let id = vec![0x40u8, 0, 0, 0, 0, 0, 0]; // bit 6 set = type 1
|
let id = vec![0x10u8, 0, 0, 0, 0, 0, 0];
|
||||||
let err = hdr.decode_managed_id(&id).unwrap_err();
|
let err = hdr.decode_managed_id(&id).unwrap_err();
|
||||||
assert_eq!(err, FormatError::InvalidHeapIdType(1));
|
assert_eq!(err, FormatError::InvalidHeapIdType(1));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn tiny_object_is_read_from_the_id() {
|
||||||
|
let (file_data, _) = build_simple_heap(8, 8);
|
||||||
|
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||||
|
// Type 2 (0x20), length - 1 in the low 4 bits, data after.
|
||||||
|
let id = [0x20 | 2, b'a', b'b', b'c', 0, 0, 0];
|
||||||
|
assert_eq!(hdr.read_managed_object(&file_data, &id, 8).unwrap(), b"abc");
|
||||||
|
// A length running past the ID is an error, not a short read.
|
||||||
|
let id = [0x20 | 9, b'a', b'b', b'c', 0, 0, 0];
|
||||||
|
assert!(hdr.read_managed_object(&file_data, &id, 8).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn huge_object_with_a_direct_id() {
|
||||||
|
// With IDs long enough for an address and a length, libhdf5 stores
|
||||||
|
// huge objects' location in the ID instead of the huge-object B-tree.
|
||||||
|
let (mut file_data, _) = build_simple_heap(8, 8);
|
||||||
|
let mut hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||||
|
hdr.heap_id_length = 17;
|
||||||
|
file_data[900..905].copy_from_slice(b"huge!");
|
||||||
|
let mut id = vec![0x10u8];
|
||||||
|
id.extend_from_slice(&900u64.to_le_bytes());
|
||||||
|
id.extend_from_slice(&5u64.to_le_bytes());
|
||||||
|
assert_eq!(
|
||||||
|
hdr.read_managed_object(&file_data, &id, 8).unwrap(),
|
||||||
|
b"huge!"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unknown_heap_id_version_is_refused() {
|
||||||
|
let (file_data, _) = build_simple_heap(8, 8);
|
||||||
|
let hdr = FractalHeapHeader::parse(&file_data, 0, 8, 8).unwrap();
|
||||||
|
let id = [0x40u8, 0, 0, 0, 0, 0, 0];
|
||||||
|
assert!(hdr.read_managed_object(&file_data, &id, 8).is_err());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -45,9 +45,16 @@ pub fn resolve_v1_group_entries(
|
|||||||
)?;
|
)?;
|
||||||
|
|
||||||
let mut entries = Vec::new();
|
let mut entries = Vec::new();
|
||||||
|
let mut heap_checked = false;
|
||||||
for snod_addr in snod_addrs {
|
for snod_addr in snod_addrs {
|
||||||
let snod = SymbolTableNode::parse(file_data, snod_addr as usize, offset_size)?;
|
let snod = SymbolTableNode::parse(file_data, snod_addr as usize, offset_size)?;
|
||||||
for entry in &snod.entries {
|
for entry in &snod.entries {
|
||||||
|
// Like libhdf5, look at the heap's free list only once a name is
|
||||||
|
// needed: an empty group with a damaged heap still lists.
|
||||||
|
if !heap_checked {
|
||||||
|
heap.validate_free_list(file_data, length_size)?;
|
||||||
|
heap_checked = true;
|
||||||
|
}
|
||||||
let name = heap.read_string(file_data, entry.link_name_offset)?;
|
let name = heap.read_string(file_data, entry.link_name_offset)?;
|
||||||
entries.push(GroupEntry {
|
entries.push(GroupEntry {
|
||||||
name,
|
name,
|
||||||
@@ -73,6 +80,53 @@ pub fn find_v1_soft_link(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Option<String>, FormatError> {
|
) -> Result<Option<String>, FormatError> {
|
||||||
|
let mut found = None;
|
||||||
|
for_each_v1_soft_link(
|
||||||
|
file_data,
|
||||||
|
sym_table_msg,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
|link_name| link_name == name,
|
||||||
|
|_, target| {
|
||||||
|
found = Some(target);
|
||||||
|
false
|
||||||
|
},
|
||||||
|
)?;
|
||||||
|
Ok(found)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every soft link in a v1 group, as `(name, target path)`.
|
||||||
|
pub fn v1_soft_links(
|
||||||
|
file_data: &[u8],
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Vec<(String, String)>, FormatError> {
|
||||||
|
let mut links = Vec::new();
|
||||||
|
for_each_v1_soft_link(
|
||||||
|
file_data,
|
||||||
|
sym_table_msg,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
|_| true,
|
||||||
|
|name, target| {
|
||||||
|
links.push((String::from(name), target));
|
||||||
|
true
|
||||||
|
},
|
||||||
|
)?;
|
||||||
|
Ok(links)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Visit the soft links of a v1 group whose name passes `wanted`, with their
|
||||||
|
/// target paths, until `visit` returns false.
|
||||||
|
fn for_each_v1_soft_link(
|
||||||
|
file_data: &[u8],
|
||||||
|
sym_table_msg: &SymbolTableMessage,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
wanted: impl Fn(&str) -> bool,
|
||||||
|
mut visit: impl FnMut(&str, String) -> bool,
|
||||||
|
) -> Result<(), FormatError> {
|
||||||
let heap = LocalHeap::parse(
|
let heap = LocalHeap::parse(
|
||||||
file_data,
|
file_data,
|
||||||
sym_table_msg.local_heap_address as usize,
|
sym_table_msg.local_heap_address as usize,
|
||||||
@@ -85,13 +139,19 @@ pub fn find_v1_soft_link(
|
|||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
)?;
|
)?;
|
||||||
|
let mut heap_checked = false;
|
||||||
for snod_addr in snod_addrs {
|
for snod_addr in snod_addrs {
|
||||||
let snod = SymbolTableNode::parse(file_data, snod_addr as usize, offset_size)?;
|
let snod = SymbolTableNode::parse(file_data, snod_addr as usize, offset_size)?;
|
||||||
for entry in &snod.entries {
|
for entry in &snod.entries {
|
||||||
if entry.cache_type != CACHE_TYPE_SOFT_LINK {
|
if entry.cache_type != CACHE_TYPE_SOFT_LINK {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
if heap.read_string(file_data, entry.link_name_offset)? != name {
|
if !heap_checked {
|
||||||
|
heap.validate_free_list(file_data, length_size)?;
|
||||||
|
heap_checked = true;
|
||||||
|
}
|
||||||
|
let name = heap.read_string(file_data, entry.link_name_offset)?;
|
||||||
|
if !wanted(&name) {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let value_offset = u32::from_le_bytes([
|
let value_offset = u32::from_le_bytes([
|
||||||
@@ -100,12 +160,19 @@ pub fn find_v1_soft_link(
|
|||||||
entry.scratch_pad[2],
|
entry.scratch_pad[2],
|
||||||
entry.scratch_pad[3],
|
entry.scratch_pad[3],
|
||||||
]);
|
]);
|
||||||
return heap
|
let target = heap.read_string(file_data, u64::from(value_offset))?;
|
||||||
.read_string(file_data, u64::from(value_offset))
|
if !visit(&name, target) {
|
||||||
.map(Some);
|
return Ok(());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Ok(None)
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a v1 symbol-table entry is a soft link (no object header of its
|
||||||
|
/// own; its target path is in the local heap).
|
||||||
|
pub fn is_v1_soft_link(entry: &GroupEntry) -> bool {
|
||||||
|
entry.cache_type == CACHE_TYPE_SOFT_LINK
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Extract the SymbolTableMessage from an object header's messages.
|
/// Extract the SymbolTableMessage from an object header's messages.
|
||||||
|
|||||||
@@ -38,6 +38,24 @@ pub fn resolve_v2_group_entries(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// First user-defined link type (HDF5 reserves 2-63; 64 is external).
|
||||||
|
const FIRST_USER_DEFINED_LINK_TYPE: u8 = 65;
|
||||||
|
|
||||||
|
/// Parse a Link message, or `None` for a user-defined link (type 65-255).
|
||||||
|
///
|
||||||
|
/// A user-defined link's target is only meaningful to the application that
|
||||||
|
/// registered its class, so, like libhdf5 without that class, we cannot
|
||||||
|
/// follow it. Leaving it out lets the rest of the group be listed and
|
||||||
|
/// resolved instead of one such link failing the whole group; reserved
|
||||||
|
/// types (2-63) are still an error.
|
||||||
|
fn parse_link(data: &[u8], offset_size: u8) -> Result<Option<LinkMessage>, FormatError> {
|
||||||
|
match LinkMessage::parse(data, offset_size) {
|
||||||
|
Ok(link) => Ok(Some(link)),
|
||||||
|
Err(FormatError::InvalidLinkType(t)) if t >= FIRST_USER_DEFINED_LINK_TYPE => Ok(None),
|
||||||
|
Err(e) => Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Extract link entries from Link messages directly in the object header (compact storage).
|
/// Extract link entries from Link messages directly in the object header (compact storage).
|
||||||
fn resolve_compact_entries(
|
fn resolve_compact_entries(
|
||||||
object_header: &ObjectHeader,
|
object_header: &ObjectHeader,
|
||||||
@@ -46,7 +64,9 @@ fn resolve_compact_entries(
|
|||||||
let mut entries = Vec::new();
|
let mut entries = Vec::new();
|
||||||
for msg in &object_header.messages {
|
for msg in &object_header.messages {
|
||||||
if msg.msg_type == MessageType::Link {
|
if msg.msg_type == MessageType::Link {
|
||||||
let link = LinkMessage::parse(&msg.data, offset_size)?;
|
let Some(link) = parse_link(&msg.data, offset_size)? else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
if let LinkTarget::Hard {
|
if let LinkTarget::Hard {
|
||||||
object_header_address,
|
object_header_address,
|
||||||
} = link.link_target
|
} = link.link_target
|
||||||
@@ -98,7 +118,9 @@ fn for_each_dense_link(
|
|||||||
|
|
||||||
// Read managed object from fractal heap
|
// Read managed object from fractal heap
|
||||||
let link_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
let link_data = fh.read_managed_object(file_data, id_bytes, offset_size)?;
|
||||||
visit(LinkMessage::parse(&link_data, offset_size)?);
|
if let Some(link) = parse_link(&link_data, offset_size)? {
|
||||||
|
visit(link);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
@@ -178,7 +200,9 @@ fn find_symbolic_link(
|
|||||||
} else {
|
} else {
|
||||||
for msg in &object_header.messages {
|
for msg in &object_header.messages {
|
||||||
if msg.msg_type == MessageType::Link {
|
if msg.msg_type == MessageType::Link {
|
||||||
let link = LinkMessage::parse(&msg.data, offset_size)?;
|
let Some(link) = parse_link(&msg.data, offset_size)? else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
if link.name == name && is_symbolic(&link.link_target) {
|
if link.name == name && is_symbolic(&link.link_target) {
|
||||||
found = Some(link.link_target);
|
found = Some(link.link_target);
|
||||||
}
|
}
|
||||||
@@ -231,32 +255,135 @@ pub fn resolve_path_any(
|
|||||||
superblock: &Superblock,
|
superblock: &Superblock,
|
||||||
path: &str,
|
path: &str,
|
||||||
) -> Result<u64, FormatError> {
|
) -> Result<u64, FormatError> {
|
||||||
resolve_path_following_links(file_data, superblock, path, 0)
|
resolve_path_following_links(
|
||||||
|
file_data,
|
||||||
|
superblock,
|
||||||
|
superblock.root_group_address,
|
||||||
|
path,
|
||||||
|
0,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve `path` relative to the group at `group_address` (an absolute path
|
||||||
|
/// starts at the root group instead), following soft links. This is how a
|
||||||
|
/// relative soft link's target is resolved: from the group holding the link.
|
||||||
|
pub fn resolve_path_from(
|
||||||
|
file_data: &[u8],
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
path: &str,
|
||||||
|
) -> Result<u64, FormatError> {
|
||||||
|
let start = if path.starts_with('/') {
|
||||||
|
superblock.root_group_address
|
||||||
|
} else {
|
||||||
|
group_address
|
||||||
|
};
|
||||||
|
resolve_path_following_links(file_data, superblock, start, path, 0)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The children of the group at `group_address` that can be opened, as h5py
|
||||||
|
/// lists them: hard links, and soft links resolved to the object they point
|
||||||
|
/// at (under the soft link's own name). Links that cannot be followed are
|
||||||
|
/// left out rather than failing the listing — a dangling or cyclic soft link
|
||||||
|
/// (h5py lists its name but cannot open it), an external link (another
|
||||||
|
/// file), and a user-defined link. An object header that is not a group has
|
||||||
|
/// no children.
|
||||||
|
///
|
||||||
|
/// Any other error, such as a corrupt structure met while resolving a soft
|
||||||
|
/// link, is returned.
|
||||||
|
pub fn resolve_group_children(
|
||||||
|
file_data: &[u8],
|
||||||
|
superblock: &Superblock,
|
||||||
|
group_address: u64,
|
||||||
|
) -> Result<Vec<GroupEntry>, FormatError> {
|
||||||
|
let os = superblock.offset_size;
|
||||||
|
let ls = superblock.length_size;
|
||||||
|
let header = ObjectHeader::parse(file_data, group_address as usize, os, ls)?;
|
||||||
|
|
||||||
|
let mut entries = Vec::new();
|
||||||
|
let mut soft = Vec::new();
|
||||||
|
if is_v1_group(&header) {
|
||||||
|
let sym_msg = header
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::SymbolTable)
|
||||||
|
.ok_or_else(|| FormatError::PathNotFound(String::from("no symbol table message")))?;
|
||||||
|
let stm = SymbolTableMessage::parse(&sym_msg.data, os)?;
|
||||||
|
let all = group_v1::resolve_v1_group_entries(file_data, &stm, os, ls)?;
|
||||||
|
if all.iter().any(group_v1::is_v1_soft_link) {
|
||||||
|
soft = group_v1::v1_soft_links(file_data, &stm, os, ls)?;
|
||||||
|
}
|
||||||
|
entries.extend(all.into_iter().filter(|e| !group_v1::is_v1_soft_link(e)));
|
||||||
|
} else if is_v2_group(&header) {
|
||||||
|
let mut visit = |link: LinkMessage| match link.link_target {
|
||||||
|
LinkTarget::Hard {
|
||||||
|
object_header_address,
|
||||||
|
} => entries.push(GroupEntry {
|
||||||
|
name: link.name,
|
||||||
|
object_header_address,
|
||||||
|
cache_type: 0,
|
||||||
|
}),
|
||||||
|
LinkTarget::Soft { target_path } => soft.push((link.name, target_path)),
|
||||||
|
LinkTarget::External { .. } => {}
|
||||||
|
};
|
||||||
|
let link_info = find_link_info(&header, os)?;
|
||||||
|
if let Some(fh_addr) = link_info.fractal_heap_address {
|
||||||
|
for_each_dense_link(file_data, &link_info, fh_addr, os, ls, visit)?;
|
||||||
|
} else {
|
||||||
|
for msg in &header.messages {
|
||||||
|
if msg.msg_type == MessageType::Link
|
||||||
|
&& let Some(link) = parse_link(&msg.data, os)?
|
||||||
|
{
|
||||||
|
visit(link);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for (name, target) in soft {
|
||||||
|
match resolve_path_from(file_data, superblock, group_address, &target) {
|
||||||
|
Ok(object_header_address) => entries.push(GroupEntry {
|
||||||
|
name,
|
||||||
|
object_header_address,
|
||||||
|
cache_type: 0,
|
||||||
|
}),
|
||||||
|
// Dangling, cyclic, or ending in another file: not openable here.
|
||||||
|
Err(
|
||||||
|
FormatError::PathNotFound(_)
|
||||||
|
| FormatError::NestingDepthExceeded
|
||||||
|
| FormatError::ExternalLinkUnsupported { .. },
|
||||||
|
) => {}
|
||||||
|
Err(e) => return Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(entries)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Soft links followed while resolving one path. Guards against link cycles
|
/// Soft links followed while resolving one path. Guards against link cycles
|
||||||
/// (`a -> b -> a`), which are legal to create.
|
/// (`a -> b -> a`), which are legal to create.
|
||||||
const MAX_SOFT_LINK_DEPTH: u8 = 16;
|
const MAX_SOFT_LINK_DEPTH: u8 = 16;
|
||||||
|
|
||||||
|
/// Walk `path` from the group at `start`, following soft links.
|
||||||
fn resolve_path_following_links(
|
fn resolve_path_following_links(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
superblock: &Superblock,
|
superblock: &Superblock,
|
||||||
|
start: u64,
|
||||||
path: &str,
|
path: &str,
|
||||||
depth: u8,
|
depth: u8,
|
||||||
) -> Result<u64, FormatError> {
|
) -> Result<u64, FormatError> {
|
||||||
let components: Vec<&str> = path.split('/').filter(|s| !s.is_empty()).collect();
|
let components: Vec<&str> = path
|
||||||
|
.split('/')
|
||||||
|
.filter(|s| !s.is_empty() && *s != ".")
|
||||||
|
.collect();
|
||||||
if components.is_empty() {
|
if components.is_empty() {
|
||||||
return Ok(superblock.root_group_address);
|
return Ok(start);
|
||||||
}
|
}
|
||||||
|
|
||||||
let os = superblock.offset_size;
|
let os = superblock.offset_size;
|
||||||
let ls = superblock.length_size;
|
let ls = superblock.length_size;
|
||||||
|
|
||||||
let root_header =
|
let mut current_addr = start;
|
||||||
ObjectHeader::parse(file_data, superblock.root_group_address as usize, os, ls)?;
|
let mut current_header = ObjectHeader::parse(file_data, start as usize, os, ls)?;
|
||||||
|
|
||||||
let mut current_addr = superblock.root_group_address;
|
|
||||||
let mut current_header = root_header;
|
|
||||||
|
|
||||||
for (i, component) in components.iter().enumerate() {
|
for (i, component) in components.iter().enumerate() {
|
||||||
let entries = resolve_group_entries(file_data, ¤t_header, os, ls)?;
|
let entries = resolve_group_entries(file_data, ¤t_header, os, ls)?;
|
||||||
@@ -280,20 +407,17 @@ fn resolve_path_following_links(
|
|||||||
}
|
}
|
||||||
// A relative target is relative to the group holding
|
// A relative target is relative to the group holding
|
||||||
// the link; then the rest of the original path.
|
// the link; then the rest of the original path.
|
||||||
let mut full = String::new();
|
let from = if target_path.starts_with('/') {
|
||||||
if !target_path.starts_with('/') {
|
superblock.root_group_address
|
||||||
for parent in &components[..i] {
|
} else {
|
||||||
full.push('/');
|
current_addr
|
||||||
full.push_str(parent);
|
};
|
||||||
}
|
let mut full = target_path;
|
||||||
}
|
|
||||||
full.push('/');
|
|
||||||
full.push_str(&target_path);
|
|
||||||
for rest in &components[i + 1..] {
|
for rest in &components[i + 1..] {
|
||||||
full.push('/');
|
full.push('/');
|
||||||
full.push_str(rest);
|
full.push_str(rest);
|
||||||
}
|
}
|
||||||
resolve_path_following_links(file_data, superblock, &full, depth + 1)
|
resolve_path_following_links(file_data, superblock, from, &full, depth + 1)
|
||||||
}
|
}
|
||||||
Some(LinkTarget::External {
|
Some(LinkTarget::External {
|
||||||
filename,
|
filename,
|
||||||
|
|||||||
@@ -26,12 +26,13 @@
|
|||||||
//! use clawhdf5_format::{signature, superblock, object_header, group_v2,
|
//! use clawhdf5_format::{signature, superblock, object_header, group_v2,
|
||||||
//! datatype, dataspace, data_layout, data_read, message_type::MessageType};
|
//! datatype, dataspace, data_layout, data_read, message_type::MessageType};
|
||||||
//!
|
//!
|
||||||
//! let file_data = std::fs::read("output.h5").unwrap();
|
//! let bytes = std::fs::read("output.h5").unwrap();
|
||||||
//! let sig = signature::find_signature(&file_data).unwrap();
|
//! // Addresses are relative to the superblock: skip any user block.
|
||||||
//! let sb = superblock::Superblock::parse(&file_data, sig).unwrap();
|
//! let (_user_block, file_data) = signature::split_user_block(&bytes).unwrap();
|
||||||
//! let addr = group_v2::resolve_path_any(&file_data, &sb, "data").unwrap();
|
//! let sb = superblock::Superblock::parse(file_data, 0).unwrap();
|
||||||
|
//! let addr = group_v2::resolve_path_any(file_data, &sb, "data").unwrap();
|
||||||
//! let hdr = object_header::ObjectHeader::parse(
|
//! let hdr = object_header::ObjectHeader::parse(
|
||||||
//! &file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
//! file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||||
//! ```
|
//! ```
|
||||||
//!
|
//!
|
||||||
//! # Features
|
//! # Features
|
||||||
@@ -54,6 +55,7 @@ pub mod btree_v1;
|
|||||||
pub mod btree_v2;
|
pub mod btree_v2;
|
||||||
pub mod checksum;
|
pub mod checksum;
|
||||||
pub mod chunk_cache;
|
pub mod chunk_cache;
|
||||||
|
mod chunk_grid;
|
||||||
pub mod chunk_index;
|
pub mod chunk_index;
|
||||||
pub mod chunked_read;
|
pub mod chunked_read;
|
||||||
pub mod chunked_write;
|
pub mod chunked_write;
|
||||||
@@ -99,6 +101,7 @@ pub mod signature;
|
|||||||
pub mod superblock;
|
pub mod superblock;
|
||||||
pub mod symbol_table;
|
pub mod symbol_table;
|
||||||
pub mod type_builders;
|
pub mod type_builders;
|
||||||
|
pub mod vds;
|
||||||
pub mod vl_data;
|
pub mod vl_data;
|
||||||
|
|
||||||
#[cfg(feature = "provenance")]
|
#[cfg(feature = "provenance")]
|
||||||
|
|||||||
@@ -87,6 +87,57 @@ impl LocalHeap {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Walk the free list the way libhdf5 does when it loads a heap's data
|
||||||
|
/// (`H5HL__fl_deserialize`), rejecting a heap whose free list points
|
||||||
|
/// outside the data segment. libhdf5 refuses such a heap ("bad heap free
|
||||||
|
/// list"), and names read from it would be garbage.
|
||||||
|
///
|
||||||
|
/// libhdf5 only loads a heap when it needs a name from it (an empty
|
||||||
|
/// group's broken heap goes unnoticed), so call this before the first
|
||||||
|
/// [`Self::read_string`], not on parse.
|
||||||
|
///
|
||||||
|
/// The end of the list is `H5HL_FREE_NULL` (1); an all-ones value (the
|
||||||
|
/// undefined address) is accepted as "no free list" too.
|
||||||
|
pub fn validate_free_list(&self, file_data: &[u8], length_size: u8) -> Result<(), FormatError> {
|
||||||
|
const FREE_NULL: u64 = 1;
|
||||||
|
let ls = length_size as usize;
|
||||||
|
let undefined = if ls >= 8 {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
(1u64 << (8 * ls)) - 1
|
||||||
|
};
|
||||||
|
let size = self.data_segment_size;
|
||||||
|
let seg = self.data_segment_address;
|
||||||
|
let mut next = self.free_list_head_offset;
|
||||||
|
// Each free block holds two lengths, so a list longer than this
|
||||||
|
// revisits a block: a cycle.
|
||||||
|
let max_blocks = size / (2 * ls as u64) + 1;
|
||||||
|
let mut walked = 0u64;
|
||||||
|
while next != FREE_NULL && next != undefined {
|
||||||
|
if next >= size || walked >= max_blocks {
|
||||||
|
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||||
|
}
|
||||||
|
walked += 1;
|
||||||
|
let at = seg
|
||||||
|
.checked_add(next)
|
||||||
|
.and_then(|a| usize::try_from(a).ok())
|
||||||
|
.ok_or(FormatError::InvalidLocalHeapFreeList)?;
|
||||||
|
let block_offset = next;
|
||||||
|
next = read_offset(file_data, at, length_size)?;
|
||||||
|
if next == 0 {
|
||||||
|
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||||
|
}
|
||||||
|
let block_size = read_offset(file_data, at + ls, length_size)?;
|
||||||
|
if block_offset
|
||||||
|
.checked_add(block_size)
|
||||||
|
.is_none_or(|end| end > size)
|
||||||
|
{
|
||||||
|
return Err(FormatError::InvalidLocalHeapFreeList);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Read a null-terminated string from the heap's data segment at the given byte offset.
|
/// Read a null-terminated string from the heap's data segment at the given byte offset.
|
||||||
pub fn read_string(&self, file_data: &[u8], string_offset: u64) -> Result<String, FormatError> {
|
pub fn read_string(&self, file_data: &[u8], string_offset: u64) -> Result<String, FormatError> {
|
||||||
let seg_addr = self.data_segment_address as usize;
|
let seg_addr = self.data_segment_address as usize;
|
||||||
@@ -162,8 +213,8 @@ mod tests {
|
|||||||
// data_segment_size
|
// data_segment_size
|
||||||
write_val(&mut file, pos, data_seg_size as u64, length_size);
|
write_val(&mut file, pos, data_seg_size as u64, length_size);
|
||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
// free_list_head_offset
|
// free_list_head_offset: H5HL_FREE_NULL (no free space)
|
||||||
write_val(&mut file, pos, 0xFFFFFFFF, length_size);
|
write_val(&mut file, pos, 1, length_size);
|
||||||
pos += length_size as usize;
|
pos += length_size as usize;
|
||||||
// data_segment_address
|
// data_segment_address
|
||||||
write_val(&mut file, pos, data_seg_offset as u64, offset_size);
|
write_val(&mut file, pos, data_seg_offset as u64, offset_size);
|
||||||
@@ -243,6 +294,50 @@ mod tests {
|
|||||||
assert_eq!(s, "test");
|
assert_eq!(s, "test");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Heap with data segment `[a, b, c, 0-padding]` whose free list starts
|
||||||
|
/// at `head` and has one block `(next, size)` at offset 8.
|
||||||
|
fn heap_with_free_block(head: u64, next: u64, size: u64) -> Vec<u8> {
|
||||||
|
let mut file = build_heap_file(0, 100, &["abcdefg"], 8, 8);
|
||||||
|
file.resize(200, 0);
|
||||||
|
write_val(&mut file, 8, 32, 8); // data segment size
|
||||||
|
write_val(&mut file, 16, head, 8);
|
||||||
|
write_val(&mut file, 108, next, 8);
|
||||||
|
write_val(&mut file, 116, size, 8);
|
||||||
|
file
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn free_list_inside_the_segment_is_accepted() {
|
||||||
|
let file = heap_with_free_block(8, 1, 24);
|
||||||
|
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||||
|
heap.validate_free_list(&file, 8).unwrap();
|
||||||
|
assert_eq!(heap.read_string(&file, 0).unwrap(), "abcdefg");
|
||||||
|
// An all-ones head is "no free list" too.
|
||||||
|
let file = heap_with_free_block(u64::MAX, 0, 0);
|
||||||
|
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||||
|
assert!(heap.validate_free_list(&file, 8).is_ok());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn bad_free_list_is_rejected_like_libhdf5() {
|
||||||
|
for (head, next, size, why) in [
|
||||||
|
(40, 1, 8, "head past the segment"),
|
||||||
|
(8, 1, 25, "block runs past the segment"),
|
||||||
|
(8, 0, 8, "next offset of zero"),
|
||||||
|
(8, 8, 8, "cycle"),
|
||||||
|
(8, 999, 8, "next past the segment"),
|
||||||
|
] {
|
||||||
|
let file = heap_with_free_block(head, next, size);
|
||||||
|
// The header itself parses; the free list is checked on use.
|
||||||
|
let heap = LocalHeap::parse(&file, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
heap.validate_free_list(&file, 8).unwrap_err(),
|
||||||
|
FormatError::InvalidLocalHeapFreeList,
|
||||||
|
"{why}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn invalid_version() {
|
fn invalid_version() {
|
||||||
let mut file = build_heap_file(0, 100, &["x"], 8, 8);
|
let mut file = build_heap_file(0, 100, &["x"], 8, 8);
|
||||||
|
|||||||
@@ -146,12 +146,7 @@ impl ObjectHeader {
|
|||||||
ensure_len(data, pos, msg_data_size)?;
|
ensure_len(data, pos, msg_data_size)?;
|
||||||
let msg_type = MessageType::from_u16(msg_type_raw);
|
let msg_type = MessageType::from_u16(msg_type_raw);
|
||||||
|
|
||||||
// Check if unknown + must-understand (bit 3 of msg_flags)
|
check_unknown_message(msg_type, msg_flags)?;
|
||||||
if let MessageType::Unknown(id) = msg_type
|
|
||||||
&& msg_flags & 0x08 != 0
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnsupportedMessage(id));
|
|
||||||
}
|
|
||||||
|
|
||||||
if msg_type != MessageType::Nil {
|
if msg_type != MessageType::Nil {
|
||||||
messages.push(HeaderMessage {
|
messages.push(HeaderMessage {
|
||||||
@@ -229,11 +224,7 @@ impl ObjectHeader {
|
|||||||
|
|
||||||
let msg_type = MessageType::from_u16(msg_type_raw);
|
let msg_type = MessageType::from_u16(msg_type_raw);
|
||||||
|
|
||||||
if let MessageType::Unknown(id) = msg_type
|
check_unknown_message(msg_type, msg_flags)?;
|
||||||
&& msg_flags & 0x08 != 0
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnsupportedMessage(id));
|
|
||||||
}
|
|
||||||
|
|
||||||
if msg_type != MessageType::Nil {
|
if msg_type != MessageType::Nil {
|
||||||
messages.push(HeaderMessage {
|
messages.push(HeaderMessage {
|
||||||
@@ -424,11 +415,7 @@ impl ObjectHeader {
|
|||||||
|
|
||||||
let msg_type = MessageType::from_u16(msg_type_raw);
|
let msg_type = MessageType::from_u16(msg_type_raw);
|
||||||
|
|
||||||
if let MessageType::Unknown(id) = msg_type
|
check_unknown_message(msg_type, msg_flags)?;
|
||||||
&& msg_flags & 0x08 != 0
|
|
||||||
{
|
|
||||||
return Err(FormatError::UnsupportedMessage(id));
|
|
||||||
}
|
|
||||||
|
|
||||||
let msg_data = data[pos..pos + msg_data_size].to_vec();
|
let msg_data = data[pos..pos + msg_data_size].to_vec();
|
||||||
|
|
||||||
@@ -509,6 +496,24 @@ impl ObjectHeader {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Header message flag bit 7: fail if the message is unknown, always.
|
||||||
|
const MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS: u8 = 0x80;
|
||||||
|
|
||||||
|
/// Refuse an unknown message the file says no reader may skip.
|
||||||
|
///
|
||||||
|
/// The parser only ever reads, so bit 3 (fail only when opened for writing)
|
||||||
|
/// is ignored, as libhdf5 ignores it for a read-only open; bit 7 fails
|
||||||
|
/// regardless of access mode. This had the two the wrong way round, failing
|
||||||
|
/// objects libhdf5 reads and reading ones it refuses (`tbogus.h5`).
|
||||||
|
fn check_unknown_message(msg_type: MessageType, msg_flags: u8) -> Result<(), FormatError> {
|
||||||
|
match msg_type {
|
||||||
|
MessageType::Unknown(id) if msg_flags & MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS != 0 => {
|
||||||
|
Err(FormatError::UnsupportedMessage(id))
|
||||||
|
}
|
||||||
|
_ => Ok(()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
@@ -632,14 +637,38 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_v1_unknown_must_understand_errors() {
|
fn parse_v1_unknown_fail_always_errors() {
|
||||||
// Bit 3 of msg_flags = must understand
|
// Bit 7 of msg_flags = fail if unknown, whatever the access mode.
|
||||||
let messages = [(0x00FFu16, &[0xAA][..], 0x08u8)];
|
let messages = [(0x00FFu16, &[0xAA][..], 0x80u8)];
|
||||||
let data = build_v1_header(&messages, 8, 8);
|
let data = build_v1_header(&messages, 8, 8);
|
||||||
let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err();
|
let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err();
|
||||||
assert_eq!(err, FormatError::UnsupportedMessage(0x00FF));
|
assert_eq!(err, FormatError::UnsupportedMessage(0x00FF));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_v1_unknown_fail_on_write_is_ignored_when_reading() {
|
||||||
|
// Bit 3 = fail if unknown *and the file is opened for writing*. This
|
||||||
|
// parser only reads, so libhdf5 (read-only) opens such an object and
|
||||||
|
// so must we. Bits 4/5 (mark if unknown / was unknown) never fail.
|
||||||
|
for flags in [0x08u8, 0x10, 0x20, 0x38] {
|
||||||
|
let messages = [(0x00FFu16, &[0xAA][..], flags)];
|
||||||
|
let data = build_v1_header(&messages, 8, 8);
|
||||||
|
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(hdr.messages[0].msg_type, MessageType::Unknown(0x00FF));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn parse_v2_unknown_message_flags() {
|
||||||
|
let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x08)], None);
|
||||||
|
assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok());
|
||||||
|
let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x80)], None);
|
||||||
|
assert_eq!(
|
||||||
|
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
|
||||||
|
FormatError::UnsupportedMessage(0xF0)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_v2_no_timestamps_one_message() {
|
fn parse_v2_no_timestamps_one_message() {
|
||||||
let data = build_v2_header(0x00, &[(0x01, &[10, 20], 0)], None);
|
let data = build_v2_header(0x00, &[(0x01, &[10, 20], 0)], None);
|
||||||
|
|||||||
@@ -1,11 +1,17 @@
|
|||||||
//! Object header writer for v2 format.
|
//! Object header writer for v2 format.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::{format, vec::Vec};
|
||||||
|
|
||||||
use crate::checksum::jenkins_lookup3;
|
use crate::checksum::jenkins_lookup3;
|
||||||
|
use crate::error::FormatError;
|
||||||
use crate::message_type::MessageType;
|
use crate::message_type::MessageType;
|
||||||
|
|
||||||
|
/// Largest message payload a v2 object header can describe: the per-message
|
||||||
|
/// size field is 2 bytes. A bigger message cannot be encoded at all — writing
|
||||||
|
/// its size truncated to 16 bits produced files libhdf5 refuses.
|
||||||
|
pub const MAX_MESSAGE_SIZE: usize = u16::MAX as usize;
|
||||||
|
|
||||||
/// Writer for v2 object headers with proper checksums.
|
/// Writer for v2 object headers with proper checksums.
|
||||||
pub struct ObjectHeaderWriter {
|
pub struct ObjectHeaderWriter {
|
||||||
messages: Vec<(MessageType, Vec<u8>, u8)>, // (type, data, msg_flags)
|
messages: Vec<(MessageType, Vec<u8>, u8)>, // (type, data, msg_flags)
|
||||||
@@ -30,7 +36,22 @@ impl ObjectHeaderWriter {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Serialize the complete v2 object header (OHDR + messages + checksum).
|
/// Serialize the complete v2 object header (OHDR + messages + checksum).
|
||||||
pub fn serialize(&self) -> Vec<u8> {
|
///
|
||||||
|
/// Fails with [`FormatError::SerializationError`] when a message is larger
|
||||||
|
/// than [`MAX_MESSAGE_SIZE`] (e.g. an attribute over ~64 KiB, which would
|
||||||
|
/// need dense attribute storage), rather than writing a corrupt header.
|
||||||
|
pub fn serialize(&self) -> Result<Vec<u8>, FormatError> {
|
||||||
|
if let Some((msg_type, data, _)) = self
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|(_, data, _)| data.len() > MAX_MESSAGE_SIZE)
|
||||||
|
{
|
||||||
|
return Err(FormatError::SerializationError(format!(
|
||||||
|
"{msg_type:?} message is {} bytes; an object header message holds at most \
|
||||||
|
{MAX_MESSAGE_SIZE} bytes",
|
||||||
|
data.len()
|
||||||
|
)));
|
||||||
|
}
|
||||||
// Calculate total message bytes: each message has type(1) + size(2) + flags(1) + data
|
// Calculate total message bytes: each message has type(1) + size(2) + flags(1) + data
|
||||||
let msg_bytes_total: usize = self
|
let msg_bytes_total: usize = self
|
||||||
.messages
|
.messages
|
||||||
@@ -80,7 +101,7 @@ impl ObjectHeaderWriter {
|
|||||||
let checksum = jenkins_lookup3(&buf);
|
let checksum = jenkins_lookup3(&buf);
|
||||||
buf.extend_from_slice(&checksum.to_le_bytes());
|
buf.extend_from_slice(&checksum.to_le_bytes());
|
||||||
|
|
||||||
buf
|
Ok(buf)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -125,15 +146,22 @@ impl BatchObjectHeaderWriter {
|
|||||||
|
|
||||||
/// Compute the serialized size of each header without actually serializing.
|
/// Compute the serialized size of each header without actually serializing.
|
||||||
/// Returns sizes in the same order as headers were added.
|
/// Returns sizes in the same order as headers were added.
|
||||||
pub fn compute_sizes(&self) -> Vec<usize> {
|
pub fn compute_sizes(&self) -> Result<Vec<usize>, FormatError> {
|
||||||
self.headers.iter().map(|h| h.serialize().len()).collect()
|
self.headers
|
||||||
|
.iter()
|
||||||
|
.map(|h| h.serialize().map(|b| b.len()))
|
||||||
|
.collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Serialize all headers into a single contiguous buffer.
|
/// Serialize all headers into a single contiguous buffer.
|
||||||
/// Returns `(combined_bytes, offsets)` where `offsets[i]` is the byte
|
/// Returns `(combined_bytes, offsets)` where `offsets[i]` is the byte
|
||||||
/// offset of header `i` within the combined buffer.
|
/// offset of header `i` within the combined buffer.
|
||||||
pub fn serialize_all(&self) -> (Vec<u8>, Vec<usize>) {
|
pub fn serialize_all(&self) -> Result<(Vec<u8>, Vec<usize>), FormatError> {
|
||||||
let serialized: Vec<Vec<u8>> = self.headers.iter().map(|h| h.serialize()).collect();
|
let serialized: Vec<Vec<u8>> = self
|
||||||
|
.headers
|
||||||
|
.iter()
|
||||||
|
.map(|h| h.serialize())
|
||||||
|
.collect::<Result<_, _>>()?;
|
||||||
let total: usize = serialized.iter().map(|s| s.len()).sum();
|
let total: usize = serialized.iter().map(|s| s.len()).sum();
|
||||||
let mut buf = Vec::with_capacity(total);
|
let mut buf = Vec::with_capacity(total);
|
||||||
let mut offsets = Vec::with_capacity(serialized.len());
|
let mut offsets = Vec::with_capacity(serialized.len());
|
||||||
@@ -141,7 +169,7 @@ impl BatchObjectHeaderWriter {
|
|||||||
offsets.push(buf.len());
|
offsets.push(buf.len());
|
||||||
buf.extend_from_slice(s);
|
buf.extend_from_slice(s);
|
||||||
}
|
}
|
||||||
(buf, offsets)
|
Ok((buf, offsets))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -159,7 +187,7 @@ mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn empty_header_roundtrip() {
|
fn empty_header_roundtrip() {
|
||||||
let writer = ObjectHeaderWriter::new();
|
let writer = ObjectHeaderWriter::new();
|
||||||
let bytes = writer.serialize();
|
let bytes = writer.serialize().unwrap();
|
||||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
assert_eq!(hdr.version, 2);
|
assert_eq!(hdr.version, 2);
|
||||||
assert_eq!(hdr.messages.len(), 0);
|
assert_eq!(hdr.messages.len(), 0);
|
||||||
@@ -170,7 +198,7 @@ mod tests {
|
|||||||
let mut writer = ObjectHeaderWriter::new();
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
writer.add_message(MessageType::Dataspace, vec![1, 2, 3, 4]);
|
writer.add_message(MessageType::Dataspace, vec![1, 2, 3, 4]);
|
||||||
writer.add_message(MessageType::Datatype, vec![5, 6]);
|
writer.add_message(MessageType::Datatype, vec![5, 6]);
|
||||||
let bytes = writer.serialize();
|
let bytes = writer.serialize().unwrap();
|
||||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
assert_eq!(hdr.messages.len(), 2);
|
assert_eq!(hdr.messages.len(), 2);
|
||||||
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
|
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
|
||||||
@@ -184,12 +212,30 @@ mod tests {
|
|||||||
let mut writer = ObjectHeaderWriter::new();
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
// Add a message with >255 bytes of payload
|
// Add a message with >255 bytes of payload
|
||||||
writer.add_message(MessageType::Datatype, vec![0xAA; 300]);
|
writer.add_message(MessageType::Datatype, vec![0xAA; 300]);
|
||||||
let bytes = writer.serialize();
|
let bytes = writer.serialize().unwrap();
|
||||||
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
assert_eq!(hdr.messages.len(), 1);
|
assert_eq!(hdr.messages.len(), 1);
|
||||||
assert_eq!(hdr.messages[0].data.len(), 300);
|
assert_eq!(hdr.messages[0].data.len(), 300);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn oversized_message_is_an_error_not_a_truncated_size() {
|
||||||
|
// 65535 bytes is the largest encodable payload.
|
||||||
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
|
writer.add_message(MessageType::Attribute, vec![0; MAX_MESSAGE_SIZE]);
|
||||||
|
let bytes = writer.serialize().unwrap();
|
||||||
|
let hdr = ObjectHeader::parse(&bytes, 0, 8, 8).unwrap();
|
||||||
|
assert_eq!(hdr.messages[0].data.len(), MAX_MESSAGE_SIZE);
|
||||||
|
|
||||||
|
// One byte more used to be written with its size wrapped to 0.
|
||||||
|
let mut writer = ObjectHeaderWriter::new();
|
||||||
|
writer.add_message(MessageType::Attribute, vec![0; MAX_MESSAGE_SIZE + 1]);
|
||||||
|
assert!(matches!(
|
||||||
|
writer.serialize(),
|
||||||
|
Err(FormatError::SerializationError(_))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn batch_writer_serialize_all() {
|
fn batch_writer_serialize_all() {
|
||||||
let mut batch = BatchObjectHeaderWriter::new();
|
let mut batch = BatchObjectHeaderWriter::new();
|
||||||
@@ -204,7 +250,7 @@ mod tests {
|
|||||||
batch.add(w2);
|
batch.add(w2);
|
||||||
assert_eq!(batch.len(), 2);
|
assert_eq!(batch.len(), 2);
|
||||||
|
|
||||||
let (buf, offsets) = batch.serialize_all();
|
let (buf, offsets) = batch.serialize_all().unwrap();
|
||||||
assert_eq!(offsets.len(), 2);
|
assert_eq!(offsets.len(), 2);
|
||||||
assert_eq!(offsets[0], 0);
|
assert_eq!(offsets[0], 0);
|
||||||
|
|
||||||
@@ -222,7 +268,7 @@ mod tests {
|
|||||||
fn batch_writer_empty() {
|
fn batch_writer_empty() {
|
||||||
let batch = BatchObjectHeaderWriter::new();
|
let batch = BatchObjectHeaderWriter::new();
|
||||||
assert!(batch.is_empty());
|
assert!(batch.is_empty());
|
||||||
let (buf, offsets) = batch.serialize_all();
|
let (buf, offsets) = batch.serialize_all().unwrap();
|
||||||
assert!(buf.is_empty());
|
assert!(buf.is_empty());
|
||||||
assert!(offsets.is_empty());
|
assert!(offsets.is_empty());
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -10,7 +10,7 @@
|
|||||||
use crate::chunked_read::ChunkInfo;
|
use crate::chunked_read::ChunkInfo;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::filter_pipeline::FilterPipeline;
|
use crate::filter_pipeline::FilterPipeline;
|
||||||
use crate::filters::decompress_chunk;
|
use crate::filters::decompress_chunk_masked;
|
||||||
use crate::lane_partition::{self, LaneStats, PartitionStats};
|
use crate::lane_partition::{self, LaneStats, PartitionStats};
|
||||||
|
|
||||||
/// Threshold: only use parallel decompression when chunk count exceeds this.
|
/// Threshold: only use parallel decompression when chunk count exceeds this.
|
||||||
@@ -84,11 +84,13 @@ pub fn decompress_chunks_lane_partitioned(
|
|||||||
}
|
}
|
||||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||||
|
|
||||||
let decompressed = if chunk_info.filter_mask == 0 {
|
let decompressed = decompress_chunk_masked(
|
||||||
decompress_chunk(raw_chunk, pipeline, chunk_total_bytes, element_size)?
|
raw_chunk,
|
||||||
} else {
|
pipeline,
|
||||||
raw_chunk.to_vec()
|
chunk_total_bytes,
|
||||||
};
|
element_size,
|
||||||
|
chunk_info.filter_mask,
|
||||||
|
)?;
|
||||||
|
|
||||||
stats.chunks_processed += 1;
|
stats.chunks_processed += 1;
|
||||||
stats.compressed_bytes += size as u64;
|
stats.compressed_bytes += size as u64;
|
||||||
@@ -158,11 +160,13 @@ pub fn decompress_chunks_parallel(
|
|||||||
}
|
}
|
||||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||||
|
|
||||||
let decompressed = if chunk_info.filter_mask == 0 {
|
let decompressed = decompress_chunk_masked(
|
||||||
decompress_chunk(raw_chunk, pipeline, chunk_total_bytes, element_size)?
|
raw_chunk,
|
||||||
} else {
|
pipeline,
|
||||||
raw_chunk.to_vec()
|
chunk_total_bytes,
|
||||||
};
|
element_size,
|
||||||
|
chunk_info.filter_mask,
|
||||||
|
)?;
|
||||||
|
|
||||||
Ok(DecompressedChunk {
|
Ok(DecompressedChunk {
|
||||||
index,
|
index,
|
||||||
@@ -200,11 +204,13 @@ pub fn decompress_chunks_sequential(
|
|||||||
let raw_chunk = &file_data[c_addr..c_addr + size];
|
let raw_chunk = &file_data[c_addr..c_addr + size];
|
||||||
|
|
||||||
let decompressed = if let Some(pl) = pipeline {
|
let decompressed = if let Some(pl) = pipeline {
|
||||||
if chunk_info.filter_mask == 0 {
|
decompress_chunk_masked(
|
||||||
decompress_chunk(raw_chunk, pl, chunk_total_bytes, element_size)?
|
raw_chunk,
|
||||||
} else {
|
pl,
|
||||||
raw_chunk.to_vec()
|
chunk_total_bytes,
|
||||||
}
|
element_size,
|
||||||
|
chunk_info.filter_mask,
|
||||||
|
)?
|
||||||
} else {
|
} else {
|
||||||
raw_chunk.to_vec()
|
raw_chunk.to_vec()
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ use crate::data_read::extract_selection_from_buffer;
|
|||||||
use crate::dataspace::Dataspace;
|
use crate::dataspace::Dataspace;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::filter_pipeline::FilterPipeline;
|
use crate::filter_pipeline::FilterPipeline;
|
||||||
use crate::filters::decompress_chunk;
|
use crate::filters::{all_filters_skipped, decompress_chunk_masked};
|
||||||
use crate::selection::Selection;
|
use crate::selection::Selection;
|
||||||
|
|
||||||
/// The smallest axis-aligned box containing every selected element, as
|
/// The smallest axis-aligned box containing every selected element, as
|
||||||
@@ -325,12 +325,18 @@ pub fn read_selection(
|
|||||||
expected: at.saturating_add(chunk.chunk_size as usize),
|
expected: at.saturating_add(chunk.chunk_size as usize),
|
||||||
available: file_data.len(),
|
available: file_data.len(),
|
||||||
})?;
|
})?;
|
||||||
// Mirrors the full-read path: a non-zero filter mask means the
|
// Mirrors the full-read path: filter-mask bit i set means
|
||||||
// chunk was stored unfiltered.
|
// filter i was not applied to this chunk.
|
||||||
let decoded;
|
let decoded;
|
||||||
let data: &[u8] = match pipeline {
|
let data: &[u8] = match pipeline {
|
||||||
Some(pl) if chunk.filter_mask == 0 => {
|
Some(pl) if !all_filters_skipped(pl, chunk.filter_mask) => {
|
||||||
decoded = decompress_chunk(raw, pl, chunk_bytes, elem_size as u32)?;
|
decoded = decompress_chunk_masked(
|
||||||
|
raw,
|
||||||
|
pl,
|
||||||
|
chunk_bytes,
|
||||||
|
elem_size as u32,
|
||||||
|
chunk.filter_mask,
|
||||||
|
)?;
|
||||||
&decoded
|
&decoded
|
||||||
}
|
}
|
||||||
_ => raw,
|
_ => raw,
|
||||||
|
|||||||
@@ -43,7 +43,7 @@ impl Default for DatasetCreateProps {
|
|||||||
fletcher32: false,
|
fletcher32: false,
|
||||||
lz4: false,
|
lz4: false,
|
||||||
zstd_level: None,
|
zstd_level: None,
|
||||||
fill_time: FillTime::Alloc,
|
fill_time: FillTime::IfSet,
|
||||||
compact: false,
|
compact: false,
|
||||||
alignment: 0,
|
alignment: 0,
|
||||||
}
|
}
|
||||||
@@ -335,7 +335,7 @@ mod tests {
|
|||||||
fn dcpl_defaults() {
|
fn dcpl_defaults() {
|
||||||
let dcpl = DatasetCreateProps::new();
|
let dcpl = DatasetCreateProps::new();
|
||||||
assert!(dcpl.chunk_dims.is_none());
|
assert!(dcpl.chunk_dims.is_none());
|
||||||
assert_eq!(dcpl.fill_time, FillTime::Alloc);
|
assert_eq!(dcpl.fill_time, FillTime::IfSet);
|
||||||
assert!(!dcpl.compact);
|
assert!(!dcpl.compact);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -229,44 +229,47 @@ impl Selection {
|
|||||||
/// self-describing in length, so the count lets a caller walk a packed list
|
/// self-describing in length, so the count lets a caller walk a packed list
|
||||||
/// of selections — as the Virtual Dataset global-heap block does).
|
/// of selections — as the Virtual Dataset global-heap block does).
|
||||||
///
|
///
|
||||||
/// Only the forms needed for VDS assembly are decoded: `ALL`, `NONE`, and
|
/// Decodes `ALL`, `NONE`, and hyperslabs at every version libhdf5 writes
|
||||||
/// **regular** hyperslabs serialized at **version 3** (the encoding HDF5
|
/// (1: irregular, 4-byte coordinates — the default-format encoding; 2:
|
||||||
/// 1.10+/2.0 emit). Point selections, irregular hyperslabs, and older
|
/// regular, 8-byte; 3: either, variable width). A regular hyperslab maps
|
||||||
/// hyperslab versions return an error rather than mis-decoding.
|
/// to [`Selection::Hyperslab`]; an *irregular* one (a union of blocks)
|
||||||
|
/// maps to a single-block hyperslab when it has one block, and otherwise to
|
||||||
|
/// [`Selection::Points`] listing the union in row-major order (the order
|
||||||
|
/// libhdf5 iterates it in). Unlimited counts/blocks decode as `u64::MAX`
|
||||||
|
/// (see [`SerializedSelection::decode`] for the raw form). Point
|
||||||
|
/// selections are refused: libhdf5 does not allow them in virtual datasets
|
||||||
|
/// either.
|
||||||
pub fn decode_serialized(data: &[u8]) -> Result<(Selection, usize), FormatError> {
|
pub fn decode_serialized(data: &[u8]) -> Result<(Selection, usize), FormatError> {
|
||||||
if data.len() < 8 {
|
let (raw, len) = SerializedSelection::decode(data)?;
|
||||||
return Err(FormatError::UnexpectedEof {
|
let sel = match raw {
|
||||||
expected: 8,
|
SerializedSelection::All => Selection::All,
|
||||||
available: data.len(),
|
SerializedSelection::None => Selection::None,
|
||||||
});
|
SerializedSelection::Regular {
|
||||||
}
|
start,
|
||||||
let sel_type = u32::from_le_bytes([data[0], data[1], data[2], data[3]]);
|
stride,
|
||||||
let version = u32::from_le_bytes([data[4], data[5], data[6], data[7]]);
|
count,
|
||||||
|
block,
|
||||||
match sel_type {
|
} => Selection::Hyperslab {
|
||||||
// ALL / NONE: type(4) + version(4) + reserved(4) + length(4) = 16 bytes.
|
start,
|
||||||
3 | 0 => {
|
stride,
|
||||||
if data.len() < 16 {
|
count,
|
||||||
return Err(FormatError::UnexpectedEof {
|
block,
|
||||||
expected: 16,
|
},
|
||||||
available: data.len(),
|
SerializedSelection::Blocks { rank, starts, ends } => {
|
||||||
});
|
if starts.len() == rank {
|
||||||
}
|
let block = starts.iter().zip(&ends).map(|(&s, &e)| e - s + 1).collect();
|
||||||
let sel = if sel_type == 3 {
|
Selection::Hyperslab {
|
||||||
Selection::All
|
start: starts,
|
||||||
|
stride: vec![1; rank],
|
||||||
|
count: vec![1; rank],
|
||||||
|
block,
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
Selection::None
|
Selection::Points(blocks_union_coords(rank, &starts, &ends)?)
|
||||||
};
|
}
|
||||||
Ok((sel, 16))
|
|
||||||
}
|
}
|
||||||
2 => decode_hyperslab_serialized(data, version),
|
};
|
||||||
1 => Err(FormatError::ChunkedReadError(
|
Ok((sel, len))
|
||||||
"VDS point selections are not supported".into(),
|
|
||||||
)),
|
|
||||||
_ => Err(FormatError::ChunkedReadError(
|
|
||||||
"unknown dataspace selection type".into(),
|
|
||||||
)),
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Enumerate the selected element indices of a **1-D** dataspace of the
|
/// Enumerate the selected element indices of a **1-D** dataspace of the
|
||||||
@@ -314,6 +317,11 @@ impl Selection {
|
|||||||
"VDS selection rank does not match dataspace rank".into(),
|
"VDS selection rank does not match dataspace rank".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
if count.iter().chain(block.iter()).any(|&v| v == UNLIMITED) {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"unlimited selection must be clipped before it is enumerated".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
// Selected coordinates along each dimension, in order.
|
// Selected coordinates along each dimension, in order.
|
||||||
let mut per_dim: Vec<Vec<u64>> = Vec::with_capacity(rank);
|
let mut per_dim: Vec<Vec<u64>> = Vec::with_capacity(rank);
|
||||||
for d in 0..rank {
|
for d in 0..rank {
|
||||||
@@ -400,84 +408,279 @@ impl Selection {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Decode an `H5S_SEL_HYPER` selection in its serialized form. Only version-3
|
/// Hyperslab count/block value meaning "unlimited" (`H5S_UNLIMITED`).
|
||||||
/// **regular** hyperslabs are supported.
|
pub const UNLIMITED: u64 = u64::MAX;
|
||||||
fn decode_hyperslab_serialized(
|
|
||||||
data: &[u8],
|
/// Largest number of elements an irregular selection is expanded to when it
|
||||||
version: u32,
|
/// is converted to a point list by [`Selection::decode_serialized`].
|
||||||
) -> Result<(Selection, usize), FormatError> {
|
const MAX_EXPANDED_POINTS: u64 = 1 << 26;
|
||||||
if version != 3 {
|
|
||||||
return Err(FormatError::ChunkedReadError(
|
/// A selection exactly as `H5S_select_serialize` stores it, before it is
|
||||||
"only version-3 hyperslab selections are supported".into(),
|
/// applied to any dataspace.
|
||||||
));
|
///
|
||||||
|
/// Unlike [`Selection`] this keeps an irregular hyperslab as its list of
|
||||||
|
/// blocks, and a regular hyperslab's count/block may be [`UNLIMITED`] (the
|
||||||
|
/// unlimited selections used by unlimited and "printf" virtual dataset
|
||||||
|
/// mappings).
|
||||||
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
pub enum SerializedSelection {
|
||||||
|
/// `H5S_SEL_ALL`.
|
||||||
|
All,
|
||||||
|
/// `H5S_SEL_NONE`.
|
||||||
|
None,
|
||||||
|
/// A regular hyperslab. `count[d]` or `block[d]` may be [`UNLIMITED`].
|
||||||
|
Regular {
|
||||||
|
start: Vec<u64>,
|
||||||
|
stride: Vec<u64>,
|
||||||
|
count: Vec<u64>,
|
||||||
|
block: Vec<u64>,
|
||||||
|
},
|
||||||
|
/// An irregular hyperslab: the union of `starts.len() / rank` blocks, each
|
||||||
|
/// given by its first (`starts`) and last (`ends`, inclusive) coordinate,
|
||||||
|
/// flattened block-major.
|
||||||
|
Blocks {
|
||||||
|
rank: usize,
|
||||||
|
starts: Vec<u64>,
|
||||||
|
ends: Vec<u64>,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
fn sel_err(msg: &str) -> FormatError {
|
||||||
|
FormatError::ChunkedReadError(msg.into())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bounds-checked little-endian reader over a serialized selection.
|
||||||
|
struct SelReader<'a> {
|
||||||
|
data: &'a [u8],
|
||||||
|
pos: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl SelReader<'_> {
|
||||||
|
fn take(&mut self, n: usize) -> Result<&[u8], FormatError> {
|
||||||
|
let end = self.pos.checked_add(n).filter(|&e| e <= self.data.len());
|
||||||
|
let end = end.ok_or(FormatError::UnexpectedEof {
|
||||||
|
expected: self.pos.saturating_add(n),
|
||||||
|
available: self.data.len(),
|
||||||
|
})?;
|
||||||
|
let s = &self.data[self.pos..end];
|
||||||
|
self.pos = end;
|
||||||
|
Ok(s)
|
||||||
}
|
}
|
||||||
// type(4) ver(4) flags(1) enc_size(1) rank(4) [start,stride,count,block]*rank
|
|
||||||
if data.len() < 14 {
|
fn uint(&mut self, size: usize) -> Result<u64, FormatError> {
|
||||||
return Err(FormatError::UnexpectedEof {
|
let bytes = self.take(size)?;
|
||||||
expected: 14,
|
Ok(bytes
|
||||||
available: data.len(),
|
.iter()
|
||||||
});
|
.enumerate()
|
||||||
|
.fold(0u64, |v, (i, &b)| v | (b as u64) << (i * 8)))
|
||||||
}
|
}
|
||||||
let flags = data[8];
|
|
||||||
let enc_size = data[9] as usize;
|
fn remaining(&self) -> usize {
|
||||||
// Bit 0 set => regular hyperslab. Irregular hyperslabs list explicit blocks.
|
self.data.len() - self.pos
|
||||||
if flags & 0x01 == 0 {
|
|
||||||
return Err(FormatError::ChunkedReadError(
|
|
||||||
"irregular VDS hyperslab selections are not supported".into(),
|
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if enc_size != 2 && enc_size != 4 && enc_size != 8 {
|
}
|
||||||
return Err(FormatError::ChunkedReadError(
|
|
||||||
"unsupported hyperslab coordinate encoding size".into(),
|
impl SerializedSelection {
|
||||||
));
|
/// Decode a serialized selection, returning it and the number of bytes it
|
||||||
}
|
/// occupies. Mirrors libhdf5's `H5S_select_deserialize`: `ALL`/`NONE` and
|
||||||
let rank = u32::from_le_bytes([data[10], data[11], data[12], data[13]]) as usize;
|
/// hyperslab versions 1-3 are decoded; point selections (which libhdf5
|
||||||
// HDF5 caps dataspace rank at 32 (H5S_MAX_RANK). Reject anything larger so a
|
/// refuses in virtual datasets) and malformed input are errors.
|
||||||
// corrupt rank can't drive a huge allocation or read loop.
|
pub fn decode(data: &[u8]) -> Result<(SerializedSelection, usize), FormatError> {
|
||||||
if rank > 32 {
|
let mut r = SelReader { data, pos: 0 };
|
||||||
return Err(FormatError::ChunkedReadError(
|
let sel_type = r.uint(4)?;
|
||||||
"hyperslab selection rank exceeds maximum (32)".into(),
|
let version = r.uint(4)?;
|
||||||
));
|
match sel_type {
|
||||||
}
|
// ALL / NONE: type(4) + version(4) + reserved(4) + length(4).
|
||||||
let mut pos = 14;
|
0 | 3 => {
|
||||||
let read_coord = |data: &[u8], pos: usize| -> Result<u64, FormatError> {
|
r.take(8)?;
|
||||||
if pos + enc_size > data.len() {
|
let sel = if sel_type == 3 {
|
||||||
return Err(FormatError::UnexpectedEof {
|
SerializedSelection::All
|
||||||
expected: pos + enc_size,
|
} else {
|
||||||
available: data.len(),
|
SerializedSelection::None
|
||||||
});
|
};
|
||||||
|
Ok((sel, r.pos))
|
||||||
|
}
|
||||||
|
2 => {
|
||||||
|
let sel = decode_hyperslab(&mut r, version)?;
|
||||||
|
Ok((sel, r.pos))
|
||||||
|
}
|
||||||
|
1 => Err(sel_err(
|
||||||
|
"VDS point selections are not supported (libhdf5 rejects them too)",
|
||||||
|
)),
|
||||||
|
_ => Err(sel_err("unknown dataspace selection type")),
|
||||||
}
|
}
|
||||||
let mut v = 0u64;
|
}
|
||||||
for (i, &b) in data[pos..pos + enc_size].iter().enumerate() {
|
|
||||||
v |= (b as u64) << (i * 8);
|
/// The single dimension in which this selection is unlimited, if any.
|
||||||
|
pub fn unlimited_dim(&self) -> Option<usize> {
|
||||||
|
match self {
|
||||||
|
SerializedSelection::Regular { count, block, .. } => count
|
||||||
|
.iter()
|
||||||
|
.zip(block)
|
||||||
|
.position(|(&c, &b)| c == UNLIMITED || b == UNLIMITED),
|
||||||
|
_ => None,
|
||||||
}
|
}
|
||||||
Ok(v)
|
}
|
||||||
|
|
||||||
|
/// The rank the selection was serialized with (`None` for ALL/NONE, which
|
||||||
|
/// carry no rank).
|
||||||
|
pub fn rank(&self) -> Option<usize> {
|
||||||
|
match self {
|
||||||
|
SerializedSelection::Regular { start, .. } => Some(start.len()),
|
||||||
|
SerializedSelection::Blocks { rank, .. } => Some(*rank),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `H5S__hyper_deserialize`: after the type and version words.
|
||||||
|
fn decode_hyperslab(r: &mut SelReader, version: u64) -> Result<SerializedSelection, FormatError> {
|
||||||
|
const REGULAR: u8 = 0x01;
|
||||||
|
let (flags, enc_size) = match version {
|
||||||
|
// v1: reserved(4) + length(4), always irregular, 4-byte coordinates.
|
||||||
|
1 => {
|
||||||
|
r.take(8)?;
|
||||||
|
(0u8, 4usize)
|
||||||
|
}
|
||||||
|
// v2: flags(1) + length(4), 8-byte coordinates.
|
||||||
|
2 => {
|
||||||
|
let flags = r.take(1)?[0];
|
||||||
|
r.take(4)?;
|
||||||
|
(flags, 8)
|
||||||
|
}
|
||||||
|
// v3: flags(1) + encoding size(1).
|
||||||
|
3 => {
|
||||||
|
let flags = r.take(1)?[0];
|
||||||
|
let enc = r.take(1)?[0] as usize;
|
||||||
|
(flags, enc)
|
||||||
|
}
|
||||||
|
_ => return Err(sel_err("unsupported hyperslab selection version")),
|
||||||
};
|
};
|
||||||
let (mut start, mut stride, mut count, mut block) = (
|
if flags & !REGULAR != 0 {
|
||||||
Vec::with_capacity(rank),
|
return Err(sel_err("unknown hyperslab selection flags"));
|
||||||
Vec::with_capacity(rank),
|
|
||||||
Vec::with_capacity(rank),
|
|
||||||
Vec::with_capacity(rank),
|
|
||||||
);
|
|
||||||
for _ in 0..rank {
|
|
||||||
start.push(read_coord(data, pos)?);
|
|
||||||
pos += enc_size;
|
|
||||||
stride.push(read_coord(data, pos)?);
|
|
||||||
pos += enc_size;
|
|
||||||
count.push(read_coord(data, pos)?);
|
|
||||||
pos += enc_size;
|
|
||||||
block.push(read_coord(data, pos)?);
|
|
||||||
pos += enc_size;
|
|
||||||
}
|
}
|
||||||
Ok((
|
if !matches!(enc_size, 2 | 4 | 8) {
|
||||||
Selection::Hyperslab {
|
return Err(sel_err("unsupported hyperslab coordinate encoding size"));
|
||||||
|
}
|
||||||
|
let rank = r.uint(4)? as usize;
|
||||||
|
// HDF5 caps dataspace rank at 32 (H5S_MAX_RANK). Reject anything else so a
|
||||||
|
// corrupt rank can't drive a huge allocation or read loop.
|
||||||
|
if rank == 0 || rank > 32 {
|
||||||
|
return Err(sel_err("hyperslab selection rank must be 1..=32"));
|
||||||
|
}
|
||||||
|
// The all-ones value of the encoding width means "unlimited".
|
||||||
|
let unlim_raw = if enc_size == 8 {
|
||||||
|
u64::MAX
|
||||||
|
} else {
|
||||||
|
(1u64 << (enc_size * 8)) - 1
|
||||||
|
};
|
||||||
|
|
||||||
|
if flags & REGULAR != 0 {
|
||||||
|
let (mut start, mut stride, mut count, mut block) = (
|
||||||
|
Vec::with_capacity(rank),
|
||||||
|
Vec::with_capacity(rank),
|
||||||
|
Vec::with_capacity(rank),
|
||||||
|
Vec::with_capacity(rank),
|
||||||
|
);
|
||||||
|
for _ in 0..rank {
|
||||||
|
start.push(r.uint(enc_size)?);
|
||||||
|
stride.push(r.uint(enc_size)?);
|
||||||
|
let c = r.uint(enc_size)?;
|
||||||
|
count.push(if c == unlim_raw { UNLIMITED } else { c });
|
||||||
|
let b = r.uint(enc_size)?;
|
||||||
|
block.push(if b == unlim_raw { UNLIMITED } else { b });
|
||||||
|
}
|
||||||
|
let unlimited = count
|
||||||
|
.iter()
|
||||||
|
.zip(&block)
|
||||||
|
.filter(|&(&c, &b)| c == UNLIMITED || b == UNLIMITED)
|
||||||
|
.count();
|
||||||
|
if unlimited > 1 {
|
||||||
|
return Err(sel_err(
|
||||||
|
"hyperslab selection is unlimited in more than one dimension",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for d in 0..rank {
|
||||||
|
// Overlapping blocks are not a valid regular hyperslab.
|
||||||
|
if count[d] > 1 && block[d] != UNLIMITED && block[d] > stride[d] {
|
||||||
|
return Err(sel_err("regular hyperslab blocks overlap"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return Ok(SerializedSelection::Regular {
|
||||||
start,
|
start,
|
||||||
stride,
|
stride,
|
||||||
count,
|
count,
|
||||||
block,
|
block,
|
||||||
},
|
});
|
||||||
pos,
|
}
|
||||||
))
|
|
||||||
|
// Irregular: number of blocks, then each block's start and end corners.
|
||||||
|
let nblocks = r.uint(enc_size)?;
|
||||||
|
let per_block = (rank * 2 * enc_size) as u64;
|
||||||
|
// Untrusted count: it must fit in what is left of the buffer.
|
||||||
|
if nblocks
|
||||||
|
.checked_mul(per_block)
|
||||||
|
.is_none_or(|need| need > r.remaining() as u64)
|
||||||
|
{
|
||||||
|
return Err(FormatError::UnexpectedEof {
|
||||||
|
expected: r
|
||||||
|
.pos
|
||||||
|
.saturating_add(nblocks.saturating_mul(per_block) as usize),
|
||||||
|
available: r.data.len(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
let n = nblocks as usize * rank;
|
||||||
|
let (mut starts, mut ends) = (Vec::with_capacity(n), Vec::with_capacity(n));
|
||||||
|
for _ in 0..nblocks {
|
||||||
|
for _ in 0..rank {
|
||||||
|
starts.push(r.uint(enc_size)?);
|
||||||
|
}
|
||||||
|
for _ in 0..rank {
|
||||||
|
ends.push(r.uint(enc_size)?);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if starts.iter().zip(&ends).any(|(s, e)| e < s) {
|
||||||
|
return Err(sel_err("hyperslab block ends before it starts"));
|
||||||
|
}
|
||||||
|
Ok(SerializedSelection::Blocks { rank, starts, ends })
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The coordinates of the union of the given blocks, in row-major order.
|
||||||
|
fn blocks_union_coords(
|
||||||
|
rank: usize,
|
||||||
|
starts: &[u64],
|
||||||
|
ends: &[u64],
|
||||||
|
) -> Result<Vec<Vec<u64>>, FormatError> {
|
||||||
|
let mut total = 0u64;
|
||||||
|
for (s, e) in starts.chunks_exact(rank).zip(ends.chunks_exact(rank)) {
|
||||||
|
let vol = s
|
||||||
|
.iter()
|
||||||
|
.zip(e)
|
||||||
|
.try_fold(1u64, |acc, (&s, &e)| acc.checked_mul(e - s + 1));
|
||||||
|
total = vol
|
||||||
|
.and_then(|v| total.checked_add(v))
|
||||||
|
.filter(|&t| t <= MAX_EXPANDED_POINTS)
|
||||||
|
.ok_or_else(|| sel_err("irregular hyperslab selection is too large to expand"))?;
|
||||||
|
}
|
||||||
|
let mut out = Vec::with_capacity(total as usize);
|
||||||
|
for (s, e) in starts.chunks_exact(rank).zip(ends.chunks_exact(rank)) {
|
||||||
|
let mut cur = s.to_vec();
|
||||||
|
'block: loop {
|
||||||
|
out.push(cur.clone());
|
||||||
|
for d in (0..rank).rev() {
|
||||||
|
if cur[d] < e[d] {
|
||||||
|
cur[d] += 1;
|
||||||
|
continue 'block;
|
||||||
|
}
|
||||||
|
cur[d] = s[d];
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Lexicographic order of coordinates is row-major order.
|
||||||
|
out.sort_unstable();
|
||||||
|
out.dedup();
|
||||||
|
Ok(out)
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -642,11 +845,100 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn decode_irregular_hyperslab_rejected() {
|
fn decode_truncated_irregular_hyperslab_is_error() {
|
||||||
|
// Irregular, rank 1, but the block count is missing.
|
||||||
let bytes = [0x02u8, 0, 0, 0, 0x03, 0, 0, 0, 0x00, 0x02, 0x01, 0, 0, 0];
|
let bytes = [0x02u8, 0, 0, 0, 0x03, 0, 0, 0, 0x00, 0x02, 0x01, 0, 0, 0];
|
||||||
assert!(Selection::decode_serialized(&bytes).is_err());
|
assert!(Selection::decode_serialized(&bytes).is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Version 1 as libhdf5 writes it for the default (earliest) format bounds:
|
||||||
|
/// type, version, reserved(4), length(4), rank(4), nblocks(4), then each
|
||||||
|
/// block's start and inclusive end corner as 4-byte values.
|
||||||
|
fn v1_blocks(rank: u32, blocks: &[(&[u32], &[u32])]) -> Vec<u8> {
|
||||||
|
let mut b = Vec::new();
|
||||||
|
for w in [2u32, 1, 0, 0, rank, blocks.len() as u32] {
|
||||||
|
b.extend_from_slice(&w.to_le_bytes());
|
||||||
|
}
|
||||||
|
for (s, e) in blocks {
|
||||||
|
for v in s.iter().chain(e.iter()) {
|
||||||
|
b.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
b
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn decode_v1_irregular_single_block() {
|
||||||
|
// Exactly what h5py/HDF5 2.0 writes for `[0:4]` with default libver.
|
||||||
|
let bytes = v1_blocks(1, &[(&[0], &[3])]);
|
||||||
|
let (sel, used) = Selection::decode_serialized(&bytes).unwrap();
|
||||||
|
assert_eq!(used, bytes.len());
|
||||||
|
assert_eq!(sel.iter_linear_1d(8).unwrap(), vec![0, 1, 2, 3]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn decode_v1_irregular_union_is_row_major() {
|
||||||
|
// Blocks given out of order and overlapping still enumerate once each,
|
||||||
|
// in row-major order (libhdf5 iterates the union, not the list).
|
||||||
|
let bytes = v1_blocks(2, &[(&[1, 0], &[1, 1]), (&[0, 2], &[1, 2])]);
|
||||||
|
let (sel, used) = Selection::decode_serialized(&bytes).unwrap();
|
||||||
|
assert_eq!(used, bytes.len());
|
||||||
|
// (0,2) (1,0) (1,1) (1,2) in a 2x3 space.
|
||||||
|
assert_eq!(sel.iter_linear(&[2, 3]).unwrap(), vec![2, 3, 4, 5]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn decode_v2_regular_with_unlimited_count() {
|
||||||
|
// v2: flags(1) + length(4), then 8-byte start/stride/count/block.
|
||||||
|
let mut b = Vec::new();
|
||||||
|
b.extend_from_slice(&2u32.to_le_bytes());
|
||||||
|
b.extend_from_slice(&2u32.to_le_bytes());
|
||||||
|
b.push(0x01);
|
||||||
|
b.extend_from_slice(&36u32.to_le_bytes());
|
||||||
|
b.extend_from_slice(&1u32.to_le_bytes());
|
||||||
|
for v in [0u64, 10, u64::MAX, 10] {
|
||||||
|
b.extend_from_slice(&v.to_le_bytes());
|
||||||
|
}
|
||||||
|
let (raw, used) = SerializedSelection::decode(&b).unwrap();
|
||||||
|
assert_eq!(used, b.len());
|
||||||
|
assert_eq!(raw.unlimited_dim(), Some(0));
|
||||||
|
assert_eq!(
|
||||||
|
raw,
|
||||||
|
SerializedSelection::Regular {
|
||||||
|
start: vec![0],
|
||||||
|
stride: vec![10],
|
||||||
|
count: vec![UNLIMITED],
|
||||||
|
block: vec![10],
|
||||||
|
}
|
||||||
|
);
|
||||||
|
// An unclipped unlimited selection cannot be enumerated.
|
||||||
|
let (sel, _) = Selection::decode_serialized(&b).unwrap();
|
||||||
|
assert!(sel.iter_linear_1d(100).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn decode_v3_two_byte_all_ones_is_unlimited() {
|
||||||
|
let bytes = [
|
||||||
|
0x02, 0, 0, 0, 0x03, 0, 0, 0, 0x01, 0x02, 0x01, 0, 0, 0, //
|
||||||
|
0x00, 0x00, 0x01, 0x00, 0x01, 0x00, 0xFF, 0xFF,
|
||||||
|
];
|
||||||
|
let (raw, _) = SerializedSelection::decode(&bytes).unwrap();
|
||||||
|
assert_eq!(raw.unlimited_dim(), Some(0));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn decode_irregular_block_count_beyond_buffer_is_error() {
|
||||||
|
let mut b = v1_blocks(1, &[(&[0], &[3])]);
|
||||||
|
b[20..24].copy_from_slice(&u32::MAX.to_le_bytes());
|
||||||
|
assert!(Selection::decode_serialized(&b).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn decode_point_selection_is_refused() {
|
||||||
|
let bytes = [1u8, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0];
|
||||||
|
assert!(Selection::decode_serialized(&bytes).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn iter_linear_2d_block_row_major() {
|
fn iter_linear_2d_block_row_major() {
|
||||||
// A 2x2 block at the top-left of a 4x4 space => linear 0,1,4,5.
|
// A 2x2 block at the top-left of a 4x4 space => linear 0,1,4,5.
|
||||||
|
|||||||
@@ -154,13 +154,29 @@ pub fn is_shared(msg_flags: u8) -> bool {
|
|||||||
///
|
///
|
||||||
/// When the shared flag is set on a message, the data contains a reference
|
/// When the shared flag is set on a message, the data contains a reference
|
||||||
/// instead of the actual message content.
|
/// instead of the actual message content.
|
||||||
|
///
|
||||||
|
/// Assumes the file's length size equals its offset size, which only matters
|
||||||
|
/// for version-1 references; use [`parse_shared_ref_sized`] when the
|
||||||
|
/// superblock's length size is known.
|
||||||
pub fn parse_shared_ref(data: &[u8], offset_size: u8) -> Result<SharedMessageRef, FormatError> {
|
pub fn parse_shared_ref(data: &[u8], offset_size: u8) -> Result<SharedMessageRef, FormatError> {
|
||||||
|
parse_shared_ref_sized(data, offset_size, offset_size)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`parse_shared_ref`] with the superblock's length size, which locates the
|
||||||
|
/// object header address in a version-1 reference.
|
||||||
|
pub fn parse_shared_ref_sized(
|
||||||
|
data: &[u8],
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<SharedMessageRef, FormatError> {
|
||||||
ensure_len(data, 0, 2)?;
|
ensure_len(data, 0, 2)?;
|
||||||
let version = data[0];
|
let version = data[0];
|
||||||
let ref_type = data[1];
|
let ref_type = data[1];
|
||||||
|
|
||||||
// Layouts (HDF5 spec IV.A.2 "Shared Message", and libhdf5's decoder):
|
// Layouts (HDF5 spec IV.A.2 "Shared Message", and libhdf5's decoder):
|
||||||
// v1: version, type, reserved(6), address — always "committed"
|
// v1: version, type, reserved(6), then an old-style symbol table
|
||||||
|
// entry: link-name offset(length_size), object header address,
|
||||||
|
// cache type(4), reserved(4), scratch(16) — always "committed"
|
||||||
// v2: version, type, address — always "committed"
|
// v2: version, type, address — always "committed"
|
||||||
// v3: version, type, then a fractal-heap ID if type == SOHM, otherwise
|
// v3: version, type, then a fractal-heap ID if type == SOHM, otherwise
|
||||||
// an address
|
// an address
|
||||||
@@ -177,7 +193,7 @@ pub fn parse_shared_ref(data: &[u8], offset_size: u8) -> Result<SharedMessageRef
|
|||||||
})
|
})
|
||||||
};
|
};
|
||||||
match version {
|
match version {
|
||||||
1 => address_at(2 + 6),
|
1 => address_at(2 + 6 + length_size as usize),
|
||||||
2 => address_at(2),
|
2 => address_at(2),
|
||||||
3 if ref_type == SHARE_TYPE_SOHM => {
|
3 if ref_type == SHARE_TYPE_SOHM => {
|
||||||
ensure_len(data, 2, FHEAP_ID_LEN)?;
|
ensure_len(data, 2, FHEAP_ID_LEN)?;
|
||||||
@@ -225,9 +241,12 @@ pub fn parse_sohm_table_message(
|
|||||||
|
|
||||||
/// Parse the SOHM table structure (signature "SMTB") from the file.
|
/// Parse the SOHM table structure (signature "SMTB") from the file.
|
||||||
///
|
///
|
||||||
/// Each index entry: index_type(1) + mesg_types(2) + min_mesg_size(4) +
|
/// Each index entry: version(1) + index_type(1) + mesg_types(2) +
|
||||||
/// list_max(2) + btree_min(2) + num_messages(2) + index_addr(offset_size) +
|
/// min_mesg_size(4) + list_max(2) + btree_min(2) + num_messages(2) +
|
||||||
/// heap_addr(offset_size)
|
/// index_addr(offset_size) + heap_addr(offset_size)
|
||||||
|
///
|
||||||
|
/// The leading per-index version byte (0) was missing here, so every field
|
||||||
|
/// after it was read one byte off — verified against an HDF5 2.0 file.
|
||||||
pub fn parse_sohm_table(
|
pub fn parse_sohm_table(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
table_addr: usize,
|
table_addr: usize,
|
||||||
@@ -240,11 +259,16 @@ pub fn parse_sohm_table(
|
|||||||
}
|
}
|
||||||
let mut pos = table_addr + 4;
|
let mut pos = table_addr + 4;
|
||||||
let os = offset_size as usize;
|
let os = offset_size as usize;
|
||||||
let entry_size = 1 + 2 + 4 + 2 + 2 + 2 + os + os; // 13 + 2*offset_size
|
let entry_size = 1 + 1 + 2 + 4 + 2 + 2 + 2 + os + os; // 14 + 2*offset_size
|
||||||
|
|
||||||
let mut indexes = Vec::with_capacity(nindexes as usize);
|
let mut indexes = Vec::with_capacity(nindexes as usize);
|
||||||
for _ in 0..nindexes {
|
for _ in 0..nindexes {
|
||||||
ensure_len(file_data, pos, entry_size)?;
|
ensure_len(file_data, pos, entry_size)?;
|
||||||
|
let version = file_data[pos];
|
||||||
|
if version != 0 {
|
||||||
|
return Err(FormatError::InvalidSohmTableVersion(version));
|
||||||
|
}
|
||||||
|
pos += 1;
|
||||||
let index_type = file_data[pos];
|
let index_type = file_data[pos];
|
||||||
pos += 1;
|
pos += 1;
|
||||||
let mesg_types = u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
let mesg_types = u16::from_le_bytes([file_data[pos], file_data[pos + 1]]);
|
||||||
@@ -381,6 +405,68 @@ pub fn parse_sohm_btree_entries(
|
|||||||
// ---- SOHM resolution ----
|
// ---- SOHM resolution ----
|
||||||
|
|
||||||
/// Find the SOHM index that handles the given message type.
|
/// Find the SOHM index that handles the given message type.
|
||||||
|
/// Load a file's SOHM table: superblock → superblock extension → Shared
|
||||||
|
/// Message Table message → SMTB. `Ok(None)` when the file has no superblock
|
||||||
|
/// extension or no shared-message table.
|
||||||
|
pub fn load_sohm_table(
|
||||||
|
file_data: &[u8],
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Option<SohmTable>, FormatError> {
|
||||||
|
let sig = crate::signature::find_signature(file_data)?;
|
||||||
|
let sb = crate::superblock::Superblock::parse(file_data, sig)?;
|
||||||
|
let Some(ext_addr) = sb
|
||||||
|
.superblock_extension_address
|
||||||
|
.filter(|&a| !is_undefined(a, offset_size))
|
||||||
|
else {
|
||||||
|
return Ok(None);
|
||||||
|
};
|
||||||
|
let ext = ObjectHeader::parse(file_data, ext_addr as usize, offset_size, length_size)?;
|
||||||
|
let Some(msg) = ext
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::SharedMessageTable)
|
||||||
|
else {
|
||||||
|
return Ok(None);
|
||||||
|
};
|
||||||
|
let table_msg = parse_sohm_table_message(&msg.data, offset_size)?;
|
||||||
|
parse_sohm_table(
|
||||||
|
file_data,
|
||||||
|
table_msg.table_address as usize,
|
||||||
|
table_msg.nindexes,
|
||||||
|
offset_size,
|
||||||
|
)
|
||||||
|
.map(Some)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like [`message_data`], but also follows references into the file's SOHM
|
||||||
|
/// heap (shared object header messages), loading the SOHM table on demand.
|
||||||
|
pub fn message_data_with_sohm<'a>(
|
||||||
|
file_data: &[u8],
|
||||||
|
msg: &'a crate::object_header::HeaderMessage,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<Cow<'a, [u8]>, FormatError> {
|
||||||
|
if !is_shared(msg.flags) {
|
||||||
|
return Ok(Cow::Borrowed(&msg.data));
|
||||||
|
}
|
||||||
|
let shared_ref = parse_shared_ref_sized(&msg.data, offset_size, length_size)?;
|
||||||
|
let table = if shared_ref.heap_id.is_some() {
|
||||||
|
load_sohm_table(file_data, offset_size, length_size)?
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
resolve_shared_message_with_sohm(
|
||||||
|
file_data,
|
||||||
|
&shared_ref,
|
||||||
|
msg.msg_type,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
table.as_ref(),
|
||||||
|
)
|
||||||
|
.map(Cow::Owned)
|
||||||
|
}
|
||||||
|
|
||||||
fn find_index_for_msg_type(table: &SohmTable, msg_type: MessageType) -> Option<&SohmIndex> {
|
fn find_index_for_msg_type(table: &SohmTable, msg_type: MessageType) -> Option<&SohmIndex> {
|
||||||
let type_bit = 1u16 << msg_type.to_u16();
|
let type_bit = 1u16 << msg_type.to_u16();
|
||||||
table
|
table
|
||||||
@@ -444,7 +530,7 @@ pub fn message_data<'a>(
|
|||||||
if !is_shared(msg.flags) {
|
if !is_shared(msg.flags) {
|
||||||
return Ok(Cow::Borrowed(&msg.data));
|
return Ok(Cow::Borrowed(&msg.data));
|
||||||
}
|
}
|
||||||
let shared_ref = parse_shared_ref(&msg.data, offset_size)?;
|
let shared_ref = parse_shared_ref_sized(&msg.data, offset_size, length_size)?;
|
||||||
resolve_shared_message(
|
resolve_shared_message(
|
||||||
file_data,
|
file_data,
|
||||||
&shared_ref,
|
&shared_ref,
|
||||||
@@ -459,7 +545,8 @@ pub fn message_data<'a>(
|
|||||||
///
|
///
|
||||||
/// For type 1/3 (shared in another object header), reads the target object header
|
/// For type 1/3 (shared in another object header), reads the target object header
|
||||||
/// and finds the message of the specified type.
|
/// and finds the message of the specified type.
|
||||||
/// For type 2 (SOHM), uses the fractal heap from the SOHM table.
|
/// For type 2 (SOHM), uses the fractal heap from the file's SOHM table,
|
||||||
|
/// loaded from the superblock extension on demand.
|
||||||
pub fn resolve_shared_message(
|
pub fn resolve_shared_message(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
shared_ref: &SharedMessageRef,
|
shared_ref: &SharedMessageRef,
|
||||||
@@ -467,13 +554,18 @@ pub fn resolve_shared_message(
|
|||||||
offset_size: u8,
|
offset_size: u8,
|
||||||
length_size: u8,
|
length_size: u8,
|
||||||
) -> Result<Vec<u8>, FormatError> {
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let table = if shared_ref.heap_id.is_some() {
|
||||||
|
load_sohm_table(file_data, offset_size, length_size)?
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
resolve_shared_message_with_sohm(
|
resolve_shared_message_with_sohm(
|
||||||
file_data,
|
file_data,
|
||||||
shared_ref,
|
shared_ref,
|
||||||
target_msg_type,
|
target_msg_type,
|
||||||
offset_size,
|
offset_size,
|
||||||
length_size,
|
length_size,
|
||||||
None,
|
table.as_ref(),
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -579,15 +671,26 @@ mod tests {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn parse_v1_ref() {
|
fn parse_v1_ref() {
|
||||||
let mut data = Vec::new();
|
// Datatype message of `/group1/dset2` in HDF5's `tcompound.h5`
|
||||||
data.push(1); // version
|
// (written in 2000): version 1, six reserved bytes, then an old-style
|
||||||
data.push(0); // type
|
// symbol table entry — link-name offset 0x10, object header address
|
||||||
data.extend_from_slice(&[0u8; 6]); // reserved
|
// 0x590 (the committed datatype `/type1`), cache type, reserved and
|
||||||
data.extend_from_slice(&0x5678u64.to_le_bytes());
|
// scratch.
|
||||||
|
let mut data = vec![1, 0, 0, 0, 0, 0, 0, 0];
|
||||||
|
data.extend_from_slice(&0x10u64.to_le_bytes());
|
||||||
|
data.extend_from_slice(&0x590u64.to_le_bytes());
|
||||||
|
data.extend_from_slice(&[0; 24]);
|
||||||
|
|
||||||
let shared = parse_shared_ref(&data, 8).unwrap();
|
let shared = parse_shared_ref_sized(&data, 8, 8).unwrap();
|
||||||
assert_eq!(shared.version, 1);
|
assert_eq!(shared.version, 1);
|
||||||
assert_eq!(shared.object_header_address, Some(0x5678));
|
assert_eq!(shared.object_header_address, Some(0x590));
|
||||||
|
|
||||||
|
// The name offset is a length: 4 bytes here, then an 8-byte address.
|
||||||
|
let mut data = vec![1, 0, 0, 0, 0, 0, 0, 0];
|
||||||
|
data.extend_from_slice(&0x10u32.to_le_bytes());
|
||||||
|
data.extend_from_slice(&0x590u64.to_le_bytes());
|
||||||
|
let shared = parse_shared_ref_sized(&data, 8, 4).unwrap();
|
||||||
|
assert_eq!(shared.object_header_address, Some(0x590));
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
@@ -707,6 +810,7 @@ mod tests {
|
|||||||
let mut buf = Vec::new();
|
let mut buf = Vec::new();
|
||||||
buf.extend_from_slice(b"SMTB");
|
buf.extend_from_slice(b"SMTB");
|
||||||
for idx in indexes {
|
for idx in indexes {
|
||||||
|
buf.push(0); // version
|
||||||
buf.push(idx.index_type);
|
buf.push(idx.index_type);
|
||||||
buf.extend_from_slice(&idx.mesg_types.to_le_bytes());
|
buf.extend_from_slice(&idx.mesg_types.to_le_bytes());
|
||||||
buf.extend_from_slice(&idx.min_mesg_size.to_le_bytes());
|
buf.extend_from_slice(&idx.min_mesg_size.to_le_bytes());
|
||||||
|
|||||||
@@ -11,6 +11,16 @@ pub const HDF5_SIGNATURE: [u8; 8] = [0x89, b'H', b'D', b'F', b'\r', b'\n', 0x1A,
|
|||||||
/// (powers of two starting at 512, plus offset 0).
|
/// (powers of two starting at 512, plus offset 0).
|
||||||
///
|
///
|
||||||
/// Returns the byte offset where the signature was found.
|
/// Returns the byte offset where the signature was found.
|
||||||
|
///
|
||||||
|
/// A non-zero offset means the file starts with a *user block*, and every
|
||||||
|
/// address inside the file is relative to the superblock's position, not to
|
||||||
|
/// byte 0 (libhdf5 uses the signature's position as the base address even
|
||||||
|
/// when the stored base-address field disagrees). The parsers in this crate
|
||||||
|
/// take addresses as indices into `file_data`, so they must be handed the
|
||||||
|
/// bytes from the signature on — use [`split_user_block`]. [`Superblock::parse`]
|
||||||
|
/// refuses a non-zero offset for this reason.
|
||||||
|
///
|
||||||
|
/// [`Superblock::parse`]: crate::superblock::Superblock::parse
|
||||||
pub fn find_signature(data: &[u8]) -> Result<usize, FormatError> {
|
pub fn find_signature(data: &[u8]) -> Result<usize, FormatError> {
|
||||||
// Check offset 0
|
// Check offset 0
|
||||||
if data.len() >= 8 && data[..8] == HDF5_SIGNATURE {
|
if data.len() >= 8 && data[..8] == HDF5_SIGNATURE {
|
||||||
@@ -29,6 +39,17 @@ pub fn find_signature(data: &[u8]) -> Result<usize, FormatError> {
|
|||||||
Err(FormatError::SignatureNotFound)
|
Err(FormatError::SignatureNotFound)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Split a file into its user block and its HDF5 bytes.
|
||||||
|
///
|
||||||
|
/// Returns `(user_block, hdf5)`: `user_block` is everything before the
|
||||||
|
/// superblock signature (empty for most files) and `hdf5` is the rest, in
|
||||||
|
/// which every HDF5 address is a plain index. Pass `hdf5` as `file_data` to
|
||||||
|
/// every parser in this crate, and parse the superblock at offset 0 of it.
|
||||||
|
pub fn split_user_block(data: &[u8]) -> Result<(&[u8], &[u8]), FormatError> {
|
||||||
|
let offset = find_signature(data)?;
|
||||||
|
Ok(data.split_at(offset))
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
@@ -88,6 +109,21 @@ mod tests {
|
|||||||
assert_eq!(find_signature(&data), Err(FormatError::SignatureNotFound));
|
assert_eq!(find_signature(&data), Err(FormatError::SignatureNotFound));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn split_user_block_rebases_at_the_signature() {
|
||||||
|
let mut data = vec![7u8; 1024];
|
||||||
|
data[512..520].copy_from_slice(&HDF5_SIGNATURE);
|
||||||
|
let (ub, hdf5) = split_user_block(&data).unwrap();
|
||||||
|
assert_eq!(ub.len(), 512);
|
||||||
|
assert_eq!(hdf5.len(), 512);
|
||||||
|
assert_eq!(&hdf5[..8], &HDF5_SIGNATURE);
|
||||||
|
|
||||||
|
data[..8].copy_from_slice(&HDF5_SIGNATURE);
|
||||||
|
let (ub, hdf5) = split_user_block(&data).unwrap();
|
||||||
|
assert!(ub.is_empty());
|
||||||
|
assert_eq!(hdf5.len(), 1024);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn signature_prefers_earliest() {
|
fn signature_prefers_earliest() {
|
||||||
// Signature at both 0 and 512, should return 0
|
// Signature at both 0 and 512, should return 0
|
||||||
|
|||||||
@@ -39,7 +39,13 @@ pub struct Superblock {
|
|||||||
pub superblock_extension_address: Option<u64>,
|
pub superblock_extension_address: Option<u64>,
|
||||||
/// CRC32C checksum (v2/v3 only).
|
/// CRC32C checksum (v2/v3 only).
|
||||||
pub checksum: Option<u32>,
|
pub checksum: Option<u32>,
|
||||||
/// Page size for page-buffer mode (v4 only). `None` for v0–v3.
|
/// Page size of the non-standard "version 4" superblock layout (v4 only).
|
||||||
|
/// `None` for v0–v3.
|
||||||
|
///
|
||||||
|
/// HDF5 has no superblock version 4 — libhdf5 refuses it. A real paged
|
||||||
|
/// file is a v2/v3 superblock whose extension holds a File Space Info
|
||||||
|
/// message (what `FileWriter::with_page_size` writes). This field is kept
|
||||||
|
/// only so such files written by older clawhdf5 versions still parse.
|
||||||
pub page_size: Option<u32>,
|
pub page_size: Option<u32>,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -127,8 +133,9 @@ impl Superblock {
|
|||||||
|
|
||||||
/// Serialize this superblock to bytes.
|
/// Serialize this superblock to bytes.
|
||||||
///
|
///
|
||||||
/// Writes v2/v3 format, or v4 (with `page_size`) when `self.version == 4`.
|
/// Writes v2/v3 format, or the non-standard v4 (with `page_size`) when
|
||||||
/// Computes and appends Jenkins lookup3 checksum.
|
/// `self.version == 4` — which no HDF5 library opens; see
|
||||||
|
/// [`Self::page_size`]. Computes and appends Jenkins lookup3 checksum.
|
||||||
pub fn serialize(&self) -> Vec<u8> {
|
pub fn serialize(&self) -> Vec<u8> {
|
||||||
let mut buf = Vec::with_capacity(48);
|
let mut buf = Vec::with_capacity(48);
|
||||||
buf.extend_from_slice(&HDF5_SIGNATURE);
|
buf.extend_from_slice(&HDF5_SIGNATURE);
|
||||||
@@ -167,8 +174,18 @@ impl Superblock {
|
|||||||
|
|
||||||
/// Parse a superblock from `data` starting at `signature_offset`.
|
/// Parse a superblock from `data` starting at `signature_offset`.
|
||||||
///
|
///
|
||||||
/// The signature must be present at the given offset.
|
/// The signature must be present at the given offset, and that offset
|
||||||
|
/// must be 0: every address in an HDF5 file is relative to the
|
||||||
|
/// superblock, so when a file has a user block (signature at 512, 1024,
|
||||||
|
/// …) the caller must pass the bytes from the signature on — see
|
||||||
|
/// [`crate::signature::split_user_block`] — and use that slice as
|
||||||
|
/// `file_data` everywhere. A non-zero offset is refused with
|
||||||
|
/// [`FormatError::UserBlockNotStripped`] because the addresses in the
|
||||||
|
/// returned superblock would otherwise be applied to the wrong bytes.
|
||||||
pub fn parse(data: &[u8], signature_offset: usize) -> Result<Superblock, FormatError> {
|
pub fn parse(data: &[u8], signature_offset: usize) -> Result<Superblock, FormatError> {
|
||||||
|
if signature_offset != 0 {
|
||||||
|
return Err(FormatError::UserBlockNotStripped(signature_offset as u64));
|
||||||
|
}
|
||||||
let d = data
|
let d = data
|
||||||
.get(signature_offset..)
|
.get(signature_offset..)
|
||||||
.ok_or(FormatError::UnexpectedEof {
|
.ok_or(FormatError::UnexpectedEof {
|
||||||
@@ -669,7 +686,16 @@ mod tests {
|
|||||||
let mut data = vec![0u8; 1024];
|
let mut data = vec![0u8; 1024];
|
||||||
let v0 = build_v0_bytes(8);
|
let v0 = build_v0_bytes(8);
|
||||||
data[512..512 + v0.len()].copy_from_slice(&v0);
|
data[512..512 + v0.len()].copy_from_slice(&v0);
|
||||||
let sb = Superblock::parse(&data, 512).unwrap();
|
// Addresses are relative to the superblock, so parsing in place
|
||||||
|
// (where they would be applied to the whole buffer) is refused...
|
||||||
|
assert_eq!(
|
||||||
|
Superblock::parse(&data, 512),
|
||||||
|
Err(FormatError::UserBlockNotStripped(512))
|
||||||
|
);
|
||||||
|
// ...and the caller parses the bytes from the signature on.
|
||||||
|
let (ub, hdf5) = crate::signature::split_user_block(&data).unwrap();
|
||||||
|
assert_eq!(ub.len(), 512);
|
||||||
|
let sb = Superblock::parse(hdf5, 0).unwrap();
|
||||||
assert_eq!(sb.version, 0);
|
assert_eq!(sb.version, 0);
|
||||||
assert_eq!(sb.root_group_address, 96);
|
assert_eq!(sb.root_group_address, 96);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -15,29 +15,81 @@ use crate::datatype::{
|
|||||||
|
|
||||||
/// Controls when fill values are written to dataset storage.
|
/// Controls when fill values are written to dataset storage.
|
||||||
///
|
///
|
||||||
/// Corresponds to the HDF5 fill value message's "fill time" field.
|
/// Corresponds to the HDF5 fill value message's "fill time" field
|
||||||
|
/// (`H5D_fill_time_t`).
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
||||||
pub enum FillTime {
|
pub enum FillTime {
|
||||||
/// Never write fill values (0x02). Avoids initialization overhead
|
/// Never write fill values (`H5D_FILL_TIME_NEVER`). Avoids
|
||||||
/// for datasets that will be fully written before any read.
|
/// initialization overhead for datasets that will be fully written
|
||||||
|
/// before any read.
|
||||||
Never,
|
Never,
|
||||||
/// Write fill values at allocation time (0x0a). This is the default
|
/// Write fill values when storage is allocated (`H5D_FILL_TIME_ALLOC`).
|
||||||
/// and matches the HDF5 C library's behavior.
|
|
||||||
#[default]
|
|
||||||
Alloc,
|
Alloc,
|
||||||
/// Write fill values only when the fill value has been explicitly set (0x06).
|
/// Write fill values at allocation only if one was set explicitly
|
||||||
|
/// (`H5D_FILL_TIME_IFSET`). The default, as in the HDF5 C library.
|
||||||
|
#[default]
|
||||||
IfSet,
|
IfSet,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Space allocation time written with every fill value message: late
|
||||||
|
/// (`H5D_ALLOC_TIME_LATE`), bits 0-1 of the flags byte.
|
||||||
|
const ALLOC_TIME_LATE: u8 = 2;
|
||||||
|
|
||||||
impl FillTime {
|
impl FillTime {
|
||||||
/// Serialize to the byte used in the fill value message (version 3).
|
/// Serialize to the flags byte of a version 3 fill value message: the
|
||||||
|
/// space allocation time (late) in bits 0-1 and the fill time in bits
|
||||||
|
/// 2-3 (`H5D_FILL_TIME_ALLOC` = 0, `NEVER` = 1, `IFSET` = 2).
|
||||||
|
///
|
||||||
|
/// This used to put `Never` in the ALLOC slot, `Alloc` in IFSET and
|
||||||
|
/// `IfSet` in NEVER, so libhdf5 saw every choice as a different one.
|
||||||
pub fn to_byte(self) -> u8 {
|
pub fn to_byte(self) -> u8 {
|
||||||
match self {
|
ALLOC_TIME_LATE | (self.code() << 2)
|
||||||
FillTime::Never => 0x02,
|
}
|
||||||
FillTime::Alloc => 0x0a,
|
|
||||||
FillTime::IfSet => 0x06,
|
/// Decode the fill time from a version 3 fill value message's flags.
|
||||||
|
pub fn from_byte(flags: u8) -> Option<FillTime> {
|
||||||
|
match (flags >> 2) & 0x03 {
|
||||||
|
0 => Some(FillTime::Alloc),
|
||||||
|
1 => Some(FillTime::Never),
|
||||||
|
2 => Some(FillTime::IfSet),
|
||||||
|
_ => None,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn code(self) -> u8 {
|
||||||
|
match self {
|
||||||
|
FillTime::Alloc => 0,
|
||||||
|
FillTime::Never => 1,
|
||||||
|
FillTime::IfSet => 2,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Serialize a version 3 Fill Value message for a dataset of `dt`: the fill
|
||||||
|
/// time, and the user-defined fill value if there is one (bit 5).
|
||||||
|
pub(crate) fn fill_value_message(
|
||||||
|
fill_time: FillTime,
|
||||||
|
value: Option<&[u8]>,
|
||||||
|
dt: &Datatype,
|
||||||
|
) -> Result<Vec<u8>, crate::error::FormatError> {
|
||||||
|
let mut msg = vec![3, fill_time.to_byte()];
|
||||||
|
if let Some(value) = value {
|
||||||
|
if matches!(dt, Datatype::VariableLength { .. }) {
|
||||||
|
return Err(crate::error::FormatError::SerializationError(
|
||||||
|
"a fill value for a variable-length datatype is not supported".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
if value.len() != dt.type_size() as usize {
|
||||||
|
return Err(crate::error::FormatError::DataSizeMismatch {
|
||||||
|
expected: dt.type_size() as usize,
|
||||||
|
actual: value.len(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
msg[1] |= 0x20; // fill value defined
|
||||||
|
msg.extend_from_slice(&(value.len() as u32).to_le_bytes());
|
||||||
|
msg.extend_from_slice(value);
|
||||||
|
}
|
||||||
|
Ok(msg)
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---- Datatype constructors ----
|
// ---- Datatype constructors ----
|
||||||
@@ -332,7 +384,11 @@ pub(crate) fn build_attr_message(name: &str, value: &AttrValue) -> AttributeMess
|
|||||||
raw_data: data.clone(),
|
raw_data: data.clone(),
|
||||||
},
|
},
|
||||||
AttrValue::String(s) => {
|
AttrValue::String(s) => {
|
||||||
let bytes = s.as_bytes();
|
// A fixed-length string type must be at least 1 byte: libhdf5
|
||||||
|
// rejects size 0 ("invalid datatype size") and with it every
|
||||||
|
// attribute on the object. h5py stores "" as one NUL byte.
|
||||||
|
let mut bytes = s.as_bytes().to_vec();
|
||||||
|
bytes.resize(bytes.len().max(1), 0);
|
||||||
AttributeMessage {
|
AttributeMessage {
|
||||||
name: name.to_string(),
|
name: name.to_string(),
|
||||||
datatype: Datatype::String {
|
datatype: Datatype::String {
|
||||||
@@ -341,11 +397,12 @@ pub(crate) fn build_attr_message(name: &str, value: &AttrValue) -> AttributeMess
|
|||||||
charset: CharacterSet::Utf8,
|
charset: CharacterSet::Utf8,
|
||||||
},
|
},
|
||||||
dataspace: scalar_ds(),
|
dataspace: scalar_ds(),
|
||||||
raw_data: bytes.to_vec(),
|
raw_data: bytes,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
AttrValue::StringArray(arr) => {
|
AttrValue::StringArray(arr) => {
|
||||||
let max_len = arr.iter().map(|s| s.len()).max().unwrap_or(0);
|
// At least 1 byte per element, as for a single string.
|
||||||
|
let max_len = arr.iter().map(|s| s.len()).max().unwrap_or(0).max(1);
|
||||||
let mut raw = Vec::new();
|
let mut raw = Vec::new();
|
||||||
for s in arr {
|
for s in arr {
|
||||||
let mut b = s.as_bytes().to_vec();
|
let mut b = s.as_bytes().to_vec();
|
||||||
@@ -431,8 +488,10 @@ pub struct DatasetBuilder {
|
|||||||
pub(crate) data: Option<Vec<u8>>,
|
pub(crate) data: Option<Vec<u8>>,
|
||||||
pub(crate) attrs: Vec<(String, AttrValue)>,
|
pub(crate) attrs: Vec<(String, AttrValue)>,
|
||||||
pub(crate) chunk_options: ChunkOptions,
|
pub(crate) chunk_options: ChunkOptions,
|
||||||
/// Controls when fill values are written. Default is `FillTime::Alloc`.
|
/// Controls when fill values are written. Default is `FillTime::IfSet`.
|
||||||
pub(crate) fill_time: FillTime,
|
pub(crate) fill_time: FillTime,
|
||||||
|
/// User-defined fill value: one element's bytes, as stored.
|
||||||
|
pub(crate) fill_value: Option<Vec<u8>>,
|
||||||
/// Use compact (inline) storage: data is stored in the object header.
|
/// Use compact (inline) storage: data is stored in the object header.
|
||||||
/// Only valid when raw data is <= 65536 bytes and dataset is not chunked.
|
/// Only valid when raw data is <= 65536 bytes and dataset is not chunked.
|
||||||
pub(crate) compact: bool,
|
pub(crate) compact: bool,
|
||||||
@@ -459,6 +518,7 @@ impl DatasetBuilder {
|
|||||||
attrs: Vec::new(),
|
attrs: Vec::new(),
|
||||||
chunk_options: ChunkOptions::default(),
|
chunk_options: ChunkOptions::default(),
|
||||||
fill_time: FillTime::default(),
|
fill_time: FillTime::default(),
|
||||||
|
fill_value: None,
|
||||||
compact: false,
|
compact: false,
|
||||||
alignment: 0,
|
alignment: 0,
|
||||||
virtual_sources: None,
|
virtual_sources: None,
|
||||||
@@ -671,7 +731,12 @@ impl DatasetBuilder {
|
|||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Enable Pcodec lossless numerical compression (clawhdf5 filter ID 32023).
|
/// Enable Pcodec lossless numerical compression (private clawhdf5 filter
|
||||||
|
/// ID 480).
|
||||||
|
///
|
||||||
|
/// **Not interoperable:** pcodec has no registered HDF5 filter ID and no
|
||||||
|
/// libhdf5 plugin, so h5py and other HDF5 readers cannot read the
|
||||||
|
/// dataset — only clawhdf5 built with the `pcodec` feature can.
|
||||||
///
|
///
|
||||||
/// Pcodec achieves 30–94% better compression ratio than Zstd for f32/f64
|
/// Pcodec achieves 30–94% better compression ratio than Zstd for f32/f64
|
||||||
/// columns at 1–5 GiB/s decompression speed (arXiv:2502.06112). Requires
|
/// columns at 1–5 GiB/s decompression speed (arXiv:2502.06112). Requires
|
||||||
@@ -715,10 +780,20 @@ impl DatasetBuilder {
|
|||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Set the dataset's fill value: what readers return for storage that
|
||||||
|
/// was never written (e.g. after the dataset is extended). `value` is one
|
||||||
|
/// element's bytes as stored — the dataset datatype's size and byte order
|
||||||
|
/// (`(-1i32).to_le_bytes()` for an `i32` dataset). A size mismatch, or a
|
||||||
|
/// variable-length datatype, makes `finish` fail.
|
||||||
|
pub fn with_fill_value(&mut self, value: &[u8]) -> &mut Self {
|
||||||
|
self.fill_value = Some(value.to_vec());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
/// Use compact (inline) storage for this dataset.
|
/// Use compact (inline) storage for this dataset.
|
||||||
///
|
///
|
||||||
/// The raw data is stored directly in the dataset's object header rather
|
/// The raw data is stored directly in the dataset's object header rather
|
||||||
/// than as a separate data blob. Only effective when raw data <= 65536 bytes
|
/// than as a separate data blob. Only effective when raw data <= 65531 bytes
|
||||||
/// and the dataset is not chunked.
|
/// and the dataset is not chunked.
|
||||||
pub fn compact(&mut self) -> &mut Self {
|
pub fn compact(&mut self) -> &mut Self {
|
||||||
self.compact = true;
|
self.compact = true;
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -148,7 +148,12 @@ pub fn read_vl_strings(
|
|||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve VL byte sequences from raw data.
|
/// Resolve VL sequences from raw data, returning each element's bytes.
|
||||||
|
///
|
||||||
|
/// Each element is the sequence's full encoding — element count × base type
|
||||||
|
/// size bytes, in the base type's byte order — so a sequence of `i32` yields
|
||||||
|
/// four bytes per value. Decode it with the base type (e.g.
|
||||||
|
/// [`crate::data_read::read_as_i64`]).
|
||||||
pub fn read_vl_bytes(
|
pub fn read_vl_bytes(
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
raw_data: &[u8],
|
raw_data: &[u8],
|
||||||
@@ -177,8 +182,10 @@ pub fn read_vl_bytes(
|
|||||||
},
|
},
|
||||||
)?;
|
)?;
|
||||||
|
|
||||||
let len = (vl.length as usize).min(obj.data.len());
|
// The heap object holds the whole sequence. `vl.length` counts
|
||||||
result.push(obj.data[..len].to_vec());
|
// elements, not bytes, so it is only the byte length when the base
|
||||||
|
// type is one byte wide.
|
||||||
|
result.push(obj.data.clone());
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(result)
|
Ok(result)
|
||||||
|
|||||||
@@ -0,0 +1,13 @@
|
|||||||
|
# Filter conformance fixtures
|
||||||
|
|
||||||
|
Files written by libhdf5 (and its registered filter plugins), used by the
|
||||||
|
filter regression tests in `src/filters.rs` to compare our decoders against
|
||||||
|
the values h5py/libhdf5 read from the same bytes. Chunk byte ranges quoted in
|
||||||
|
the tests come from h5py's `DatasetID.get_chunk_info`.
|
||||||
|
|
||||||
|
| File | Origin | Licence |
|
||||||
|
|------|--------|---------|
|
||||||
|
| `h5ex_d_lz4.h5` | HDF Group `HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||||
|
| `noencoder.h5` | HDF Group `test/testfiles/noencoder.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||||
|
| `le_data.h5` | HDF Group `test/testfiles/le_data.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||||
|
| `szip_h5py.h5` | Written for these tests with h5py 3 / libhdf5 2.0.0 (libaec szip): `f8` (8x10, chunks 4x10, `('nn', 8)`), `i8` (8x10, chunks 4x10, `('ec', 4)`), `u2` (70, chunks 35, `('nn', 8)`) | Same as this repository |
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,49 @@
|
|||||||
|
"""Generate shared_fill_value.h5: datasets whose Fill Value message is
|
||||||
|
*shared*, in the two ways libhdf5 can share one.
|
||||||
|
|
||||||
|
- /sohm_a, /sohm_b: the file has a shared-object-header-message (SOHM) index
|
||||||
|
for fill values, so libhdf5 stores the fill value (-7, int32) in the SOHM
|
||||||
|
heap and /sohm_b's header holds only a reference to it. Chunked, with only
|
||||||
|
the first chunk written, so the rest reads as the fill value.
|
||||||
|
- /unwritten_a, /unwritten_b: the same, never written: no storage at all,
|
||||||
|
read entirely as the fill value.
|
||||||
|
|
||||||
|
h5py has no API for SOHM indexes, so the file creation property list is
|
||||||
|
configured by calling the libhdf5 bundled in the h5py wheel through ctypes.
|
||||||
|
Written with h5py 3.16.0 / HDF5 2.0.0. Re-run only to regenerate:
|
||||||
|
|
||||||
|
python gen_shared_fill.py shared_fill_value.h5
|
||||||
|
"""
|
||||||
|
import ctypes
|
||||||
|
import glob
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import h5py
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
libdir = os.path.join(os.path.dirname(os.path.dirname(h5py.__file__)), "h5py.libs")
|
||||||
|
libs = [p for p in glob.glob(os.path.join(libdir, "libhdf5*.so*")) if "_hl" not in os.path.basename(p)]
|
||||||
|
lib = ctypes.CDLL(libs[0])
|
||||||
|
lib.H5open()
|
||||||
|
|
||||||
|
H5O_SHMESG_FILL_FLAG = 1 << 0x0005
|
||||||
|
|
||||||
|
fcpl = h5py.h5p.create(h5py.h5p.FILE_CREATE)
|
||||||
|
lib.H5Pset_shared_mesg_nindexes.argtypes = [ctypes.c_int64, ctypes.c_uint]
|
||||||
|
lib.H5Pset_shared_mesg_index.argtypes = [ctypes.c_int64, ctypes.c_uint, ctypes.c_uint, ctypes.c_uint]
|
||||||
|
assert lib.H5Pset_shared_mesg_nindexes(fcpl.id, 1) >= 0
|
||||||
|
assert lib.H5Pset_shared_mesg_index(fcpl.id, 0, H5O_SHMESG_FILL_FLAG, 0) >= 0
|
||||||
|
|
||||||
|
fapl = h5py.h5p.create(h5py.h5p.FILE_ACCESS)
|
||||||
|
fapl.set_libver_bounds(h5py.h5f.LIBVER_LATEST, h5py.h5f.LIBVER_LATEST)
|
||||||
|
fid = h5py.h5f.create(sys.argv[1].encode(), h5py.h5f.ACC_TRUNC, fcpl=fcpl, fapl=fapl)
|
||||||
|
with h5py.File(fid) as f:
|
||||||
|
# Chunked, with only the first chunk written: the rest reads as fill.
|
||||||
|
# libhdf5 keeps the first copy of a message in its own header; the second
|
||||||
|
# identical one (the `_b` datasets) is the SOHM reference.
|
||||||
|
for name in ("sohm_a", "sohm_b"):
|
||||||
|
d = f.create_dataset(name, shape=(8,), chunks=(4,), dtype="<i4", fillvalue=-7)
|
||||||
|
d[:4] = np.arange(4)
|
||||||
|
for name in ("unwritten_a", "unwritten_b"):
|
||||||
|
f.create_dataset(name, shape=(3,), dtype="<i4", fillvalue=-7)
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
# Legacy (HDF5 1.4/1.6-era) fixtures
|
||||||
|
|
||||||
|
Unmodified copies of the HDF Group's own test files from
|
||||||
|
https://github.com/HDFGroup/hdf5 at a3cf1ea82cc7a66e50029a688121e1b105a7ce88
|
||||||
|
(BSD-style license, see that repository's `LICENSE`). Current libraries cannot
|
||||||
|
write these structures, so they are kept as files.
|
||||||
|
|
||||||
|
| File | Upstream path | Exercises |
|
||||||
|
|---|---|---|
|
||||||
|
| `deflate.h5` | `test/testfiles/deflate.h5` | Data Layout message v1, chunked + deflate (v1 B-tree index) |
|
||||||
|
| `h5ex_g_iterate.h5` | `HDF5Examples/C/H5G/h5ex_g_iterate.h5` | Data Layout message v2, contiguous; an unallocated dataset |
|
||||||
|
| `tarrold.h5` | `test/testfiles/tarrold.h5` | Compound datatype v1 members with legacy array dimensions |
|
||||||
|
| `tcompound.h5` | `tools/test/testfiles/tcompound.h5` | Version-1 shared messages (committed datatypes); compound v1 array members with data |
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
@@ -83,6 +83,45 @@ fn read_chunked_dataset(file_data: &[u8], dataset_path: &str) -> (Vec<u8>, Datat
|
|||||||
(raw, datatype, dataspace)
|
(raw, datatype, dataspace)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Helper: read a virtual dataset with `vds::read_virtual_dataset`, giving it
|
||||||
|
/// the dataset's own fill value (same-file sources only).
|
||||||
|
fn read_virtual_fixture(file_data: &[u8], path: &str) -> (Vec<u8>, Datatype) {
|
||||||
|
let sig = find_signature(file_data).unwrap();
|
||||||
|
let sb = Superblock::parse(file_data, sig).unwrap();
|
||||||
|
let addr = resolve_path_any(file_data, &sb, path).unwrap();
|
||||||
|
let hdr =
|
||||||
|
ObjectHeader::parse(file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||||
|
let msg = |t: MessageType| hdr.messages.iter().find(|m| m.msg_type == t).unwrap();
|
||||||
|
let ds = Dataspace::parse(&msg(MessageType::Dataspace).data, sb.length_size).unwrap();
|
||||||
|
let (dt, _) = Datatype::parse(&msg(MessageType::Datatype).data).unwrap();
|
||||||
|
let layout = DataLayout::parse(
|
||||||
|
&msg(MessageType::DataLayout).data,
|
||||||
|
sb.offset_size,
|
||||||
|
sb.length_size,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||||
|
file_data,
|
||||||
|
&hdr.messages,
|
||||||
|
sb.offset_size,
|
||||||
|
sb.length_size,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let v = clawhdf5_format::vds::read_virtual_dataset(
|
||||||
|
file_data,
|
||||||
|
&layout,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
fill.as_deref(),
|
||||||
|
sb.offset_size,
|
||||||
|
sb.length_size,
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(v.dims, ds.dimensions);
|
||||||
|
(v.data, dt)
|
||||||
|
}
|
||||||
|
|
||||||
/// Helper: read any dataset (contiguous or chunked) as f64.
|
/// Helper: read any dataset (contiguous or chunked) as f64.
|
||||||
fn read_dataset_f64_any(bytes: &[u8], path: &str) -> Vec<f64> {
|
fn read_dataset_f64_any(bytes: &[u8], path: &str) -> Vec<f64> {
|
||||||
let sig = find_signature(bytes).unwrap();
|
let sig = find_signature(bytes).unwrap();
|
||||||
@@ -672,7 +711,7 @@ fn v4_virtual_dataset_same_file_read() {
|
|||||||
// virt[4:8] <- (unmapped) => fill 0
|
// virt[4:8] <- (unmapped) => fill 0
|
||||||
// virt[8:12] <- src_b[0:4] (ALL) => 20,21,22,23
|
// virt[8:12] <- src_b[0:4] (ALL) => 20,21,22,23
|
||||||
let file_data = include_bytes!("fixtures/vds_same_file.h5");
|
let file_data = include_bytes!("fixtures/vds_same_file.h5");
|
||||||
let (raw, datatype, _) = read_chunked_dataset(file_data, "virt");
|
let (raw, datatype) = read_virtual_fixture(file_data, "virt");
|
||||||
let values = read_as_i32(&raw, &datatype).unwrap();
|
let values = read_as_i32(&raw, &datatype).unwrap();
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
values,
|
values,
|
||||||
@@ -681,6 +720,38 @@ fn v4_virtual_dataset_same_file_read() {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn v4_virtual_dataset_raw_api_refuses_to_guess_the_fill_value() {
|
||||||
|
// The raw read API has no fill value message, so a virtual dataset with an
|
||||||
|
// unmapped region is an error there instead of zeros that may be wrong.
|
||||||
|
let file_data = include_bytes!("fixtures/vds_same_file.h5");
|
||||||
|
let sig = find_signature(file_data).unwrap();
|
||||||
|
let sb = Superblock::parse(file_data, sig).unwrap();
|
||||||
|
let addr = resolve_path_any(file_data, &sb, "virt").unwrap();
|
||||||
|
let hdr =
|
||||||
|
ObjectHeader::parse(file_data, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||||
|
let msg = |t: MessageType| hdr.messages.iter().find(|m| m.msg_type == t).unwrap();
|
||||||
|
let ds = Dataspace::parse(&msg(MessageType::Dataspace).data, sb.length_size).unwrap();
|
||||||
|
let (dt, _) = Datatype::parse(&msg(MessageType::Datatype).data).unwrap();
|
||||||
|
let layout = DataLayout::parse(
|
||||||
|
&msg(MessageType::DataLayout).data,
|
||||||
|
sb.offset_size,
|
||||||
|
sb.length_size,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
let err = read_raw_data_full(
|
||||||
|
file_data,
|
||||||
|
&layout,
|
||||||
|
&ds,
|
||||||
|
&dt,
|
||||||
|
None,
|
||||||
|
sb.offset_size,
|
||||||
|
sb.length_size,
|
||||||
|
)
|
||||||
|
.unwrap_err();
|
||||||
|
assert!(err.to_string().contains("fill value"), "{err}");
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn v4_virtual_dataset_2d_same_file_read() {
|
fn v4_virtual_dataset_2d_same_file_read() {
|
||||||
// A 4x4 virtual dataset assembled from two 2x2 same-file sources placed as
|
// A 4x4 virtual dataset assembled from two 2x2 same-file sources placed as
|
||||||
@@ -689,7 +760,7 @@ fn v4_virtual_dataset_2d_same_file_read() {
|
|||||||
// virt[2:4,2:4] <- src_b = [[5,6],[7,8]]
|
// virt[2:4,2:4] <- src_b = [[5,6],[7,8]]
|
||||||
// everything else -> fill 0
|
// everything else -> fill 0
|
||||||
let file_data = include_bytes!("fixtures/vds_2d_same_file.h5");
|
let file_data = include_bytes!("fixtures/vds_2d_same_file.h5");
|
||||||
let (raw, datatype, _) = read_chunked_dataset(file_data, "virt");
|
let (raw, datatype) = read_virtual_fixture(file_data, "virt");
|
||||||
let values = read_as_i32(&raw, &datatype).unwrap();
|
let values = read_as_i32(&raw, &datatype).unwrap();
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
values,
|
values,
|
||||||
|
|||||||
@@ -312,3 +312,49 @@ fn provenance_mismatch_on_corruption() {
|
|||||||
"corrupted data should produce hash mismatch"
|
"corrupted data should produce hash mismatch"
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
// Fuzzer finds, kept as regression tests
|
||||||
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
/// `fuzz_btree_v2` crash input from 2026-09-20 (82 bytes): a B-tree v2 header
|
||||||
|
/// followed by internal nodes that point back into themselves. It predates the
|
||||||
|
/// depth cap and record budget added to B-tree v2 traversal that day and no
|
||||||
|
/// longer crashes; this replays the fuzz target's exact code path on it so a
|
||||||
|
/// regression fails CI rather than waiting for a fuzz run.
|
||||||
|
#[test]
|
||||||
|
fn fuzz_btree_v2_crash_f98c19dc_is_a_clean_result() {
|
||||||
|
use clawhdf5_format::btree_v2::{BTreeV2Header, collect_btree_v2_records};
|
||||||
|
let data: &[u8] = &[
|
||||||
|
0x42, 0x54, 0x48, 0x44, 0x00, 0x06, 0x00, 0xed, 0xef, 0x00, 0x00, 0x00, 0x00, 0x01, 0x00,
|
||||||
|
0x00, 0x03, 0x40, 0x14, 0x93, 0x42, 0x54, 0x49, 0x4e, 0x42, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x42, 0x54, 0x48, 0x44, 0x00, 0x00, 0x00, 0x13, 0x05, 0x00, 0x00,
|
||||||
|
0x00, 0x00, 0x00, 0x00, 0x80, 0x00, 0x00, 0x00, 0x40, 0x14, 0x93, 0x42, 0x54, 0x00, 0x49,
|
||||||
|
0x00, 0x01, 0x4e, 0x42, 0x42, 0x54, 0xbe,
|
||||||
|
];
|
||||||
|
assert_eq!(data.len(), 82);
|
||||||
|
for offset_size in [4u8, 8] {
|
||||||
|
for length_size in [4u8, 8] {
|
||||||
|
if let Ok(header) = BTreeV2Header::parse(data, 0, offset_size, length_size) {
|
||||||
|
let _ = collect_btree_v2_records(data, &header, offset_size, length_size);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let (fields, file) = data.split_first_chunk::<20>().unwrap();
|
||||||
|
let header = BTreeV2Header {
|
||||||
|
tree_type: fields[0],
|
||||||
|
node_size: u32::from_le_bytes([fields[1], fields[2], fields[3], fields[4]]),
|
||||||
|
record_size: u16::from_le_bytes([fields[5], fields[6]]),
|
||||||
|
depth: u16::from_le_bytes([fields[7], fields[8]]),
|
||||||
|
root_node_address: u64::from(u32::from_le_bytes([
|
||||||
|
fields[9], fields[10], fields[11], fields[12],
|
||||||
|
])),
|
||||||
|
num_records_in_root: u16::from_le_bytes([fields[13], fields[14]]),
|
||||||
|
total_records: u64::from(u32::from_le_bytes([
|
||||||
|
fields[15], fields[16], fields[17], fields[18],
|
||||||
|
])),
|
||||||
|
};
|
||||||
|
let offset_size = if fields[19] & 1 == 0 { 4 } else { 8 };
|
||||||
|
let _ = collect_btree_v2_records(file, &header, offset_size, 8);
|
||||||
|
}
|
||||||
|
|||||||
@@ -940,3 +940,61 @@ fn provenance_verify_written_file() {
|
|||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(result, clawhdf5_format::provenance::VerifyResult::Ok);
|
assert_eq!(result, clawhdf5_format::provenance::VerifyResult::Ok);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- hdf5plugin interop: registered third-party compression filters ----
|
||||||
|
|
||||||
|
/// Write `data` (f64, 1-D, chunked) with `configure` applied, then read it
|
||||||
|
/// back with h5py + hdf5plugin (libhdf5's registered filter plugins) and
|
||||||
|
/// return the values it decodes.
|
||||||
|
#[cfg(any(feature = "lz4", feature = "zstd"))]
|
||||||
|
fn hdf5plugin_roundtrip(
|
||||||
|
tag: &str,
|
||||||
|
data: &[f64],
|
||||||
|
configure: impl FnOnce(&mut clawhdf5_format::type_builders::DatasetBuilder),
|
||||||
|
) -> Vec<f64> {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
let ds = fw.create_dataset("data");
|
||||||
|
ds.with_f64_data(data)
|
||||||
|
.with_shape(&[data.len() as u64])
|
||||||
|
.with_chunks(&[250]);
|
||||||
|
configure(ds);
|
||||||
|
let bytes = fw.finish().unwrap();
|
||||||
|
let path = std::env::temp_dir().join(format!("clawhdf5_hdf5plugin_{tag}.h5"));
|
||||||
|
std::fs::write(&path, &bytes).unwrap();
|
||||||
|
let script = format!(
|
||||||
|
"import h5py,hdf5plugin,json; f=h5py.File('{}','r'); print(json.dumps(f['data'][:].tolist()))",
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
let stdout = h5py_read(&path, &script);
|
||||||
|
serde_json::from_str(&stdout).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// libhdf5's LZ4 plugin must decode what we write (it could not while we
|
||||||
|
/// wrote a private 4-byte-LE-size framing).
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py + hdf5plugin"]
|
||||||
|
fn hdf5plugin_reads_our_lz4() {
|
||||||
|
let data: Vec<f64> = (0..1000).map(|i| (i % 37) as f64 * 0.5).collect();
|
||||||
|
let got = hdf5plugin_roundtrip("lz4", &data, |ds| {
|
||||||
|
ds.with_lz4();
|
||||||
|
});
|
||||||
|
assert_eq!(got, data);
|
||||||
|
let got = hdf5plugin_roundtrip("lz4_noshuffle", &data, |ds| {
|
||||||
|
ds.with_lz4().without_shuffle();
|
||||||
|
});
|
||||||
|
assert_eq!(got, data);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// libhdf5's Zstandard plugin must decode what we write (it could not while
|
||||||
|
/// our frames lacked the content size).
|
||||||
|
#[cfg(feature = "zstd")]
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py + hdf5plugin"]
|
||||||
|
fn hdf5plugin_reads_our_zstd() {
|
||||||
|
let data: Vec<f64> = (0..1000).map(|i| (i % 37) as f64 * 0.5).collect();
|
||||||
|
let got = hdf5plugin_roundtrip("zstd", &data, |ds| {
|
||||||
|
ds.with_zstd(3);
|
||||||
|
});
|
||||||
|
assert_eq!(got, data);
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,646 @@
|
|||||||
|
//! Regression tests for writer metadata bugs that produced files libhdf5
|
||||||
|
//! refuses (or reads differently from us), plus the reader-side counterparts.
|
||||||
|
//!
|
||||||
|
//! The plain tests check the bytes we write with our own parser. The
|
||||||
|
//! `#[ignore]`d ones are the interop half: they open what we write in h5py
|
||||||
|
//! (`CLAWHDF5_PYTHON`, as in `writer_h5py_tests.rs`) and run `h5dump` over it.
|
||||||
|
|
||||||
|
use clawhdf5_format::data_layout::DataLayout;
|
||||||
|
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder, ReferenceType};
|
||||||
|
use clawhdf5_format::file_writer::{AttrValue, FileWriter};
|
||||||
|
use clawhdf5_format::group_v2::resolve_path_any;
|
||||||
|
use clawhdf5_format::message_type::MessageType;
|
||||||
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
|
use clawhdf5_format::signature;
|
||||||
|
use clawhdf5_format::superblock::Superblock;
|
||||||
|
use clawhdf5_format::type_builders::{FillTime, make_u8_type};
|
||||||
|
|
||||||
|
// ---- helpers ----
|
||||||
|
|
||||||
|
fn header_at(bytes: &[u8], path: &str) -> (Superblock, ObjectHeader) {
|
||||||
|
let sig = signature::find_signature(bytes).unwrap();
|
||||||
|
let sb = Superblock::parse(bytes, sig).unwrap();
|
||||||
|
let addr = if path == "/" {
|
||||||
|
sb.root_group_address
|
||||||
|
} else {
|
||||||
|
resolve_path_any(bytes, &sb, path).unwrap()
|
||||||
|
};
|
||||||
|
let oh = ObjectHeader::parse(bytes, addr as usize, sb.offset_size, sb.length_size).unwrap();
|
||||||
|
(sb, oh)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn layout_of(bytes: &[u8], path: &str) -> DataLayout {
|
||||||
|
let (sb, oh) = header_at(bytes, path);
|
||||||
|
let msg = oh
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::DataLayout)
|
||||||
|
.unwrap();
|
||||||
|
DataLayout::parse(&msg.data, sb.offset_size, sb.length_size).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn python() -> String {
|
||||||
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn write_tmp(name: &str, bytes: &[u8]) -> std::path::PathBuf {
|
||||||
|
let path = std::env::temp_dir().join(format!("clawhdf5_writer_meta_{name}.h5"));
|
||||||
|
std::fs::write(&path, bytes).unwrap();
|
||||||
|
path
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run `script` (with `path` bound to the file) under h5py; return stdout.
|
||||||
|
fn h5py(path: &std::path::Path, script: &str) -> String {
|
||||||
|
let full = format!(
|
||||||
|
"import h5py, numpy as np, json\npath = {:?}\n{script}",
|
||||||
|
path.display().to_string()
|
||||||
|
);
|
||||||
|
let o = std::process::Command::new(python())
|
||||||
|
.args(["-c", &full])
|
||||||
|
.output()
|
||||||
|
.expect("python interpreter");
|
||||||
|
assert!(
|
||||||
|
o.status.success(),
|
||||||
|
"h5py failed: {}",
|
||||||
|
String::from_utf8_lossy(&o.stderr)
|
||||||
|
);
|
||||||
|
String::from_utf8(o.stdout).unwrap().trim().to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `h5dump` must read the whole file without error.
|
||||||
|
fn h5dump_ok(path: &std::path::Path) {
|
||||||
|
let o = std::process::Command::new("h5dump")
|
||||||
|
.arg(path)
|
||||||
|
.output()
|
||||||
|
.expect("h5dump");
|
||||||
|
assert!(
|
||||||
|
o.status.success(),
|
||||||
|
"h5dump failed: {}{}",
|
||||||
|
String::from_utf8_lossy(&o.stdout),
|
||||||
|
String::from_utf8_lossy(&o.stderr)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn u8_ramp(n: usize) -> Vec<u8> {
|
||||||
|
(0..n).map(|i| (i % 251) as u8).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- 1. object header message size limit ----
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn attribute_too_big_for_a_header_message_is_an_error() {
|
||||||
|
// Measured: a 70000-byte attribute was written with its message size
|
||||||
|
// wrapped to 16 bits, and libhdf5 refused the whole root group.
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.set_root_attr(
|
||||||
|
"a",
|
||||||
|
AttrValue::Raw {
|
||||||
|
datatype: make_u8_type(),
|
||||||
|
shape: vec![70_000],
|
||||||
|
data: u8_ramp(70_000),
|
||||||
|
},
|
||||||
|
);
|
||||||
|
assert!(fw.finish().is_err());
|
||||||
|
|
||||||
|
// 65500 bytes still fits and still works.
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.set_root_attr(
|
||||||
|
"a",
|
||||||
|
AttrValue::Raw {
|
||||||
|
datatype: make_u8_type(),
|
||||||
|
shape: vec![65_500],
|
||||||
|
data: u8_ramp(65_500),
|
||||||
|
},
|
||||||
|
);
|
||||||
|
let bytes = fw.finish().unwrap();
|
||||||
|
let (sb, oh) = header_at(&bytes, "/");
|
||||||
|
let attrs = clawhdf5_format::attribute::extract_attributes(&oh, sb.length_size).unwrap();
|
||||||
|
assert_eq!(attrs[0].raw_data, u8_ramp(65_500));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn compact_layout_falls_back_to_contiguous_past_the_message_limit() {
|
||||||
|
// Layout message = 4 bytes + data; data may be at most 65531 bytes.
|
||||||
|
for (n, compact) in [(65_531, true), (65_532, false), (65_534, false)] {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.create_dataset("d").with_u8_data(&u8_ramp(n)).compact();
|
||||||
|
let bytes = fw.finish().unwrap();
|
||||||
|
match layout_of(&bytes, "d") {
|
||||||
|
DataLayout::Compact { data } => {
|
||||||
|
assert!(compact, "{n} bytes must not be compact");
|
||||||
|
assert_eq!(data, u8_ramp(n));
|
||||||
|
}
|
||||||
|
DataLayout::Contiguous { .. } => assert!(!compact, "{n} bytes should be compact"),
|
||||||
|
other => panic!("unexpected layout {other:?}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py module and h5dump"]
|
||||||
|
fn h5py_reads_compact_datasets_at_the_limit() {
|
||||||
|
for n in [65_531usize, 65_534] {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.create_dataset("d").with_u8_data(&u8_ramp(n)).compact();
|
||||||
|
let path = write_tmp(&format!("compact_{n}"), &fw.finish().unwrap());
|
||||||
|
let out = h5py(
|
||||||
|
&path,
|
||||||
|
"f = h5py.File(path, 'r'); v = f['d'][()]\n\
|
||||||
|
print(bool((v == (np.arange(v.size) % 251).astype(np.uint8)).all()), v.size)",
|
||||||
|
);
|
||||||
|
assert_eq!(out, format!("True {n}"));
|
||||||
|
h5dump_ok(&path);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- 2. Time / BitField / Opaque / Reference datatypes ----
|
||||||
|
|
||||||
|
fn exotic_types() -> Vec<(&'static str, Datatype, Vec<u8>)> {
|
||||||
|
// Four elements each. The object references point at the root group,
|
||||||
|
// which a v3-superblock file without an extension puts at address 48.
|
||||||
|
let refs: Vec<u8> = (0..4).flat_map(|_| 48u64.to_le_bytes()).collect();
|
||||||
|
vec![
|
||||||
|
(
|
||||||
|
"bits",
|
||||||
|
Datatype::BitField {
|
||||||
|
size: 1,
|
||||||
|
byte_order: DatatypeByteOrder::LittleEndian,
|
||||||
|
bit_offset: 0,
|
||||||
|
bit_precision: 8,
|
||||||
|
},
|
||||||
|
vec![1, 2, 4, 8],
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"opaque",
|
||||||
|
Datatype::Opaque {
|
||||||
|
size: 4,
|
||||||
|
tag: b"mytag".to_vec(),
|
||||||
|
},
|
||||||
|
(0..16).collect(),
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"ref",
|
||||||
|
Datatype::Reference {
|
||||||
|
size: 8,
|
||||||
|
ref_type: ReferenceType::Object,
|
||||||
|
},
|
||||||
|
refs,
|
||||||
|
),
|
||||||
|
(
|
||||||
|
"time",
|
||||||
|
Datatype::Time {
|
||||||
|
size: 4,
|
||||||
|
bit_precision: 32,
|
||||||
|
},
|
||||||
|
(0..16).collect(),
|
||||||
|
),
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn exotic_file() -> Vec<u8> {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
for (name, dt, raw) in exotic_types() {
|
||||||
|
fw.create_dataset(name)
|
||||||
|
.with_compound_data(dt.clone(), raw.clone(), 4);
|
||||||
|
fw.set_root_attr(
|
||||||
|
name,
|
||||||
|
AttrValue::Raw {
|
||||||
|
datatype: dt,
|
||||||
|
shape: vec![4],
|
||||||
|
data: raw,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
}
|
||||||
|
fw.finish().unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn exotic_datatypes_are_written_not_emptied() {
|
||||||
|
let bytes = exotic_file();
|
||||||
|
let (sb, root) = header_at(&bytes, "/");
|
||||||
|
assert_eq!(sb.root_group_address, 48);
|
||||||
|
let attrs = clawhdf5_format::attribute::extract_attributes(&root, sb.length_size).unwrap();
|
||||||
|
for (name, dt, raw) in exotic_types() {
|
||||||
|
let (_, oh) = header_at(&bytes, name);
|
||||||
|
let msg = oh
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::Datatype)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(msg.data, dt.serialize(), "{name}");
|
||||||
|
assert_eq!(Datatype::parse(&msg.data).unwrap().0, dt, "{name}");
|
||||||
|
let attr = attrs.iter().find(|a| a.name == name).unwrap();
|
||||||
|
assert_eq!(attr.datatype, dt, "{name}");
|
||||||
|
assert_eq!(attr.raw_data, raw, "{name}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py module and h5dump"]
|
||||||
|
fn h5py_reads_exotic_datatypes() {
|
||||||
|
let path = write_tmp("exotic", &exotic_file());
|
||||||
|
let out = h5py(
|
||||||
|
&path,
|
||||||
|
"from h5py import h5t, h5s\n\
|
||||||
|
f = h5py.File(path, 'r')\n\
|
||||||
|
r = {}\n\
|
||||||
|
buf = np.zeros(4, dtype='V4')\n\
|
||||||
|
f['opaque'].id.read(h5s.ALL, h5s.ALL, buf, mtype=f['opaque'].id.get_type())\n\
|
||||||
|
r['bits'] = f['bits'][()].tolist(), f.attrs['bits'].tolist()\n\
|
||||||
|
r['opaque'] = (f['opaque'].id.get_type().get_tag().decode(),\n\
|
||||||
|
\x20 f.attrs.get_id('opaque').get_type().get_tag().decode(),\n\
|
||||||
|
\x20 buf.tobytes().hex())\n\
|
||||||
|
r['ref'] = [f[x].name for x in f['ref'][()]] + [f[x].name for x in f.attrs['ref']]\n\
|
||||||
|
r['time'] = (f['time'].id.get_type().get_class() == h5t.TIME,\n\
|
||||||
|
\x20 f.attrs.get_id('time').get_type().get_class() == h5t.TIME)\n\
|
||||||
|
print(json.dumps(r))",
|
||||||
|
);
|
||||||
|
let v: serde_json::Value = serde_json::from_str(&out).unwrap();
|
||||||
|
assert_eq!(v["bits"], serde_json::json!([[1, 2, 4, 8], [1, 2, 4, 8]]));
|
||||||
|
assert_eq!(
|
||||||
|
v["opaque"],
|
||||||
|
serde_json::json!(["mytag", "mytag", "000102030405060708090a0b0c0d0e0f"])
|
||||||
|
);
|
||||||
|
assert_eq!(v["ref"], serde_json::json!(vec!["/"; 8]));
|
||||||
|
assert_eq!(v["time"], serde_json::json!([true, true]));
|
||||||
|
h5dump_ok(&path);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py module and h5dump"]
|
||||||
|
fn raw_attributes_copied_from_h5py_survive_a_rewrite() {
|
||||||
|
// Read Raw attributes of the exotic classes out of an h5py file and write
|
||||||
|
// them back: this used to emit empty datatype messages.
|
||||||
|
let src = std::env::temp_dir().join("clawhdf5_writer_meta_exotic_src.h5");
|
||||||
|
h5py(
|
||||||
|
&src,
|
||||||
|
"from h5py import h5t, h5s, h5a\n\
|
||||||
|
f = h5py.File(path, 'w')\n\
|
||||||
|
f.attrs['ref'] = np.array([f.ref, f.ref], dtype=h5py.ref_dtype)\n\
|
||||||
|
f.attrs.create('opaque', np.frombuffer(b'abcdefgh', dtype='V4'))\n\
|
||||||
|
t = h5t.STD_B16BE.copy()\n\
|
||||||
|
a = h5a.create(f.id, b'bits', t, h5s.create_simple((2,)))\n\
|
||||||
|
a.write(np.array([0x0102, 0x0304], dtype='>u2'), mtype=t)\n\
|
||||||
|
a.close()\n\
|
||||||
|
f.close()",
|
||||||
|
);
|
||||||
|
let src_bytes = std::fs::read(&src).unwrap();
|
||||||
|
let (sb, root) = header_at(&src_bytes, "/");
|
||||||
|
let attrs = clawhdf5_format::attribute::extract_attributes(&root, sb.length_size).unwrap();
|
||||||
|
assert_eq!(attrs.len(), 3);
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
for a in &attrs {
|
||||||
|
let data = if a.name == "ref" {
|
||||||
|
// Re-target the references at our root group.
|
||||||
|
48u64.to_le_bytes().repeat(2)
|
||||||
|
} else {
|
||||||
|
a.raw_data.clone()
|
||||||
|
};
|
||||||
|
fw.set_root_attr(
|
||||||
|
&a.name,
|
||||||
|
AttrValue::Raw {
|
||||||
|
datatype: a.datatype.clone(),
|
||||||
|
shape: a.dataspace.dimensions.clone(),
|
||||||
|
data,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let path = write_tmp("exotic_copy", &fw.finish().unwrap());
|
||||||
|
let out = h5py(
|
||||||
|
&path,
|
||||||
|
"f = h5py.File(path, 'r')\n\
|
||||||
|
print(json.dumps([[f[x].name for x in f.attrs['ref']],\n\
|
||||||
|
\x20 f.attrs['opaque'].tobytes().decode(),\n\
|
||||||
|
\x20 f.attrs.get_id('bits').get_type().get_order(),\n\
|
||||||
|
\x20 f.attrs['bits'].tolist()]))",
|
||||||
|
);
|
||||||
|
assert_eq!(out, r#"[["/", "/"], "abcdefgh", 1, [258, 772]]"#);
|
||||||
|
h5dump_ok(&path);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- 3. paged file-space strategy ----
|
||||||
|
|
||||||
|
fn paged_file(page_size: u32) -> Vec<u8> {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.with_page_size(page_size);
|
||||||
|
fw.create_dataset("d").with_f64_data(&[1.0, 2.0, 3.0]);
|
||||||
|
fw.create_dataset("c")
|
||||||
|
.with_i32_data(&(0..100).collect::<Vec<_>>())
|
||||||
|
.with_chunks(&[10]);
|
||||||
|
fw.set_root_attr("a", AttrValue::I64(7));
|
||||||
|
let mut g = fw.create_group("g");
|
||||||
|
g.create_dataset("e").with_u8_data(&[9; 5000]);
|
||||||
|
fw.add_group(g.finish());
|
||||||
|
fw.finish().unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn paged_file_has_a_real_superblock() {
|
||||||
|
// Measured: `with_page_size` wrote superblock version 4, which does not
|
||||||
|
// exist ("bad superblock version number" in libhdf5).
|
||||||
|
for ps in [512u32, 4096, 65536] {
|
||||||
|
let bytes = paged_file(ps);
|
||||||
|
let (sb, _) = header_at(&bytes, "/");
|
||||||
|
assert_eq!(sb.version, 3);
|
||||||
|
assert_eq!(bytes.len() % ps as usize, 0);
|
||||||
|
let (_, e) = header_at(&bytes, "g/e");
|
||||||
|
assert!(
|
||||||
|
e.messages
|
||||||
|
.iter()
|
||||||
|
.any(|m| m.msg_type == MessageType::Dataspace)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py module and h5dump"]
|
||||||
|
fn h5py_opens_paged_files() {
|
||||||
|
for ps in [512u32, 4096, 65536] {
|
||||||
|
let path = write_tmp(&format!("paged_{ps}"), &paged_file(ps));
|
||||||
|
let out = h5py(
|
||||||
|
&path,
|
||||||
|
"f = h5py.File(path, 'r')\n\
|
||||||
|
p = f.id.get_create_plist()\n\
|
||||||
|
print(json.dumps([p.get_file_space_strategy()[0], p.get_file_space_page_size(),\n\
|
||||||
|
\x20 f['d'][()].tolist(), int(f['c'][()].sum()), int(f.attrs['a']),\n\
|
||||||
|
\x20 int(f['g/e'][()].sum())]))",
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
out,
|
||||||
|
format!("[1, {ps}, [1.0, 2.0, 3.0], 4950, 7, 45000]"),
|
||||||
|
"page size {ps}"
|
||||||
|
);
|
||||||
|
h5dump_ok(&path);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- 4. fill time and fill value ----
|
||||||
|
|
||||||
|
fn fill_message(bytes: &[u8], path: &str) -> clawhdf5_format::object_header::HeaderMessage {
|
||||||
|
let (_, oh) = header_at(bytes, path);
|
||||||
|
oh.messages
|
||||||
|
.into_iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::FillValue)
|
||||||
|
.unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fill_file() -> Vec<u8> {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.create_dataset("never")
|
||||||
|
.with_f64_data(&[1.0, 2.0])
|
||||||
|
.fill_time(FillTime::Never);
|
||||||
|
fw.create_dataset("alloc")
|
||||||
|
.with_f64_data(&[1.0, 2.0])
|
||||||
|
.fill_time(FillTime::Alloc);
|
||||||
|
fw.create_dataset("ifset")
|
||||||
|
.with_f64_data(&[1.0, 2.0])
|
||||||
|
.fill_time(FillTime::IfSet);
|
||||||
|
fw.create_dataset("default").with_f64_data(&[1.0, 2.0]);
|
||||||
|
fw.create_dataset("filled")
|
||||||
|
.with_i32_data(&[1, 2, 3, 4])
|
||||||
|
.with_chunks(&[2])
|
||||||
|
.with_maxshape(&[u64::MAX])
|
||||||
|
.with_fill_value(&(-1i32).to_le_bytes());
|
||||||
|
fw.finish().unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn fill_time_uses_libhdf5_codes() {
|
||||||
|
// H5D_FILL_TIME_ALLOC = 0, NEVER = 1, IFSET = 2, in bits 2-3. Measured:
|
||||||
|
// h5py saw our Never as ALLOC, Alloc as IFSET and IfSet as NEVER.
|
||||||
|
let bytes = fill_file();
|
||||||
|
for (path, code) in [("never", 1), ("alloc", 0), ("ifset", 2), ("default", 2)] {
|
||||||
|
let msg = fill_message(&bytes, path);
|
||||||
|
assert_eq!((msg.data[1] >> 2) & 3, code, "{path}");
|
||||||
|
assert_eq!(msg.data[1] & 3, 2, "{path}: allocation time stays late");
|
||||||
|
}
|
||||||
|
for ft in [FillTime::Never, FillTime::Alloc, FillTime::IfSet] {
|
||||||
|
assert_eq!(FillTime::from_byte(ft.to_byte()), Some(ft));
|
||||||
|
}
|
||||||
|
assert_eq!(FillTime::default(), FillTime::IfSet);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn fill_value_is_written_and_read_back() {
|
||||||
|
let bytes = fill_file();
|
||||||
|
let msg = fill_message(&bytes, "filled");
|
||||||
|
assert_eq!(
|
||||||
|
clawhdf5_format::fill_value::parse_fill_value(&msg).unwrap(),
|
||||||
|
Some((-1i32).to_le_bytes().to_vec())
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
clawhdf5_format::fill_value::parse_fill_value(&fill_message(&bytes, "ifset")).unwrap(),
|
||||||
|
None
|
||||||
|
);
|
||||||
|
|
||||||
|
// One element's bytes, no more, no less.
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.create_dataset("d")
|
||||||
|
.with_f64_data(&[1.0])
|
||||||
|
.with_fill_value(&[0; 4]);
|
||||||
|
assert!(fw.finish().is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py module and h5dump"]
|
||||||
|
fn h5py_sees_our_fill_time_and_fill_value() {
|
||||||
|
let path = write_tmp("fill", &fill_file());
|
||||||
|
let out = h5py(
|
||||||
|
&path,
|
||||||
|
"from h5py import h5d\n\
|
||||||
|
f = h5py.File(path, 'r')\n\
|
||||||
|
names = {h5d.FILL_TIME_NEVER: 'never', h5d.FILL_TIME_ALLOC: 'alloc', h5d.FILL_TIME_IFSET: 'ifset'}\n\
|
||||||
|
t = [names[f[n].id.get_create_plist().get_fill_time()] for n in ('never', 'alloc', 'ifset', 'default')]\n\
|
||||||
|
print(json.dumps([t, int(f['filled'].fillvalue), f['filled'][()].tolist()]))\n\
|
||||||
|
f.close()\n\
|
||||||
|
f = h5py.File(path, 'r+')\n\
|
||||||
|
f['filled'].resize((7,))\n\
|
||||||
|
f.close()\n\
|
||||||
|
print(json.dumps(h5py.File(path, 'r')['filled'][()].tolist()))",
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
out,
|
||||||
|
"[[\"never\", \"alloc\", \"ifset\", \"ifset\"], -1, [1, 2, 3, 4]]\n[1, 2, 3, 4, -1, -1, -1]"
|
||||||
|
);
|
||||||
|
h5dump_ok(&path);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- 5. empty string attributes ----
|
||||||
|
|
||||||
|
fn empty_string_file() -> Vec<u8> {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.set_root_attr("empty", AttrValue::String(String::new()));
|
||||||
|
fw.set_root_attr("x", AttrValue::String("héllo".into()));
|
||||||
|
fw.set_root_attr(
|
||||||
|
"empties",
|
||||||
|
AttrValue::StringArray(vec![String::new(), String::new()]),
|
||||||
|
);
|
||||||
|
fw.set_root_attr("n", AttrValue::I64(3));
|
||||||
|
fw.finish().unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn empty_string_attribute_has_a_one_byte_type() {
|
||||||
|
// Measured: "" got a size-0 string type, and libhdf5 then refused every
|
||||||
|
// attribute on the object ("invalid datatype size").
|
||||||
|
let bytes = empty_string_file();
|
||||||
|
let (sb, root) = header_at(&bytes, "/");
|
||||||
|
let attrs = clawhdf5_format::attribute::extract_attributes(&root, sb.length_size).unwrap();
|
||||||
|
for name in ["empty", "empties"] {
|
||||||
|
let a = attrs.iter().find(|a| a.name == name).unwrap();
|
||||||
|
assert_eq!(a.datatype.type_size(), 1, "{name}");
|
||||||
|
let strings = a.read_as_strings().unwrap();
|
||||||
|
assert!(strings.iter().all(String::is_empty), "{name}: {strings:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
// A size-0 string type handed in directly is refused, not written.
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.set_root_attr(
|
||||||
|
"raw",
|
||||||
|
AttrValue::Raw {
|
||||||
|
datatype: Datatype::String {
|
||||||
|
size: 0,
|
||||||
|
padding: clawhdf5_format::datatype::StringPadding::NullPad,
|
||||||
|
charset: clawhdf5_format::datatype::CharacterSet::Ascii,
|
||||||
|
},
|
||||||
|
shape: vec![],
|
||||||
|
data: vec![],
|
||||||
|
},
|
||||||
|
);
|
||||||
|
assert!(fw.finish().is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py module and h5dump"]
|
||||||
|
fn h5py_reads_all_attributes_next_to_an_empty_string() {
|
||||||
|
let path = write_tmp("empty_str", &empty_string_file());
|
||||||
|
let out = h5py(
|
||||||
|
&path,
|
||||||
|
"f = h5py.File(path, 'r')\n\
|
||||||
|
d = lambda v: v.decode() if isinstance(v, bytes) else v\n\
|
||||||
|
print(json.dumps([d(f.attrs['empty']), d(f.attrs['x']),\n\
|
||||||
|
\x20 [d(s) for s in f.attrs['empties']], int(f.attrs['n'])], ensure_ascii=False))",
|
||||||
|
);
|
||||||
|
assert_eq!(out, r#"["", "héllo", ["", ""], 3]"#);
|
||||||
|
h5dump_ok(&path);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- 6. path-like names ----
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn slash_in_a_group_or_dataset_name_is_an_error() {
|
||||||
|
// Measured: create_group("a/b") wrote one link literally named "a/b",
|
||||||
|
// which h5py cannot reach ("component not found"). The writer has no
|
||||||
|
// nested groups, so such names are refused.
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
let mut g = fw.create_group("a/b");
|
||||||
|
g.create_dataset("c").with_f64_data(&[1.0]);
|
||||||
|
fw.add_group(g.finish());
|
||||||
|
assert!(fw.finish().is_err());
|
||||||
|
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.create_dataset("x/y").with_f64_data(&[1.0]);
|
||||||
|
assert!(fw.finish().is_err());
|
||||||
|
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
let mut g = fw.create_group("g");
|
||||||
|
g.create_dataset("x/y").with_f64_data(&[1.0]);
|
||||||
|
fw.add_group(g.finish());
|
||||||
|
assert!(fw.finish().is_err());
|
||||||
|
|
||||||
|
for bad in ["", "."] {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
fw.create_dataset(bad).with_f64_data(&[1.0]);
|
||||||
|
assert!(fw.finish().is_err(), "{bad:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
// One level of groups still works, and '/' stays legal in attribute names.
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
let mut g = fw.create_group("g");
|
||||||
|
g.create_dataset("c").with_f64_data(&[1.0]);
|
||||||
|
g.set_attr("m/s", AttrValue::I64(1));
|
||||||
|
fw.add_group(g.finish());
|
||||||
|
let bytes = fw.finish().unwrap();
|
||||||
|
header_at(&bytes, "g/c");
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- 7. unknown-message flags on read ----
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn unknown_message_flags_follow_libhdf5_on_tbogus() {
|
||||||
|
// libhdf5's own test file (test/testfiles/tbogus.h5): datasets carrying
|
||||||
|
// an unknown message with various flags. libhdf5 (read-only) opens
|
||||||
|
// Dataset1, 2, 4 and 5 and refuses Dataset3 ("unknown message with 'fail
|
||||||
|
// if unknown' flag found"). We used to refuse Dataset2 (bit 3, which only
|
||||||
|
// applies when writing) and open Dataset3 (bit 7, fail always).
|
||||||
|
let bytes = include_bytes!("fixtures/tbogus.h5");
|
||||||
|
let sig = signature::find_signature(bytes).unwrap();
|
||||||
|
let sb = Superblock::parse(bytes, sig).unwrap();
|
||||||
|
for (name, readable) in [
|
||||||
|
("Dataset1", true),
|
||||||
|
("Dataset2", true),
|
||||||
|
("Dataset3", false),
|
||||||
|
("Dataset4", true),
|
||||||
|
("Dataset5", true),
|
||||||
|
] {
|
||||||
|
let addr = resolve_path_any(bytes, &sb, name).unwrap();
|
||||||
|
let parsed = ObjectHeader::parse(bytes, addr as usize, sb.offset_size, sb.length_size);
|
||||||
|
match parsed {
|
||||||
|
Ok(_) => assert!(readable, "{name} must be refused"),
|
||||||
|
Err(e) => {
|
||||||
|
assert!(!readable, "{name} must be readable, got {e:?}");
|
||||||
|
assert!(matches!(
|
||||||
|
e,
|
||||||
|
clawhdf5_format::error::FormatError::UnsupportedMessage(_)
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ---- 8. shared fill value messages ----
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn shared_fill_value_is_resolved_not_zero() {
|
||||||
|
// gen_shared_fill.py: HDF5 2.0 with a SOHM index for fill values, so each
|
||||||
|
// dataset's fill value message is a reference into the SOHM heap. It
|
||||||
|
// used to be read as "no fill value" (zeros) instead of -7.
|
||||||
|
let bytes = include_bytes!("fixtures/shared_fill_value.h5");
|
||||||
|
for (name, shared) in [
|
||||||
|
("sohm_a", false),
|
||||||
|
("sohm_b", true),
|
||||||
|
("unwritten_a", false),
|
||||||
|
("unwritten_b", true),
|
||||||
|
] {
|
||||||
|
let (sb, oh) = header_at(bytes, name);
|
||||||
|
let msg = oh
|
||||||
|
.messages
|
||||||
|
.iter()
|
||||||
|
.find(|m| m.msg_type == MessageType::FillValue)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
clawhdf5_format::shared_message::is_shared(msg.flags),
|
||||||
|
shared,
|
||||||
|
"{name}: fixture layout"
|
||||||
|
);
|
||||||
|
if shared {
|
||||||
|
// Without the file the reference cannot be followed: an error,
|
||||||
|
// never a silent default.
|
||||||
|
assert_eq!(
|
||||||
|
clawhdf5_format::fill_value::dataset_fill_value(&oh.messages),
|
||||||
|
Err(clawhdf5_format::error::FormatError::UnresolvedSharedMessage)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||||
|
bytes,
|
||||||
|
&oh.messages,
|
||||||
|
sb.offset_size,
|
||||||
|
sb.length_size
|
||||||
|
)
|
||||||
|
.unwrap(),
|
||||||
|
Some((-7i32).to_le_bytes().to_vec()),
|
||||||
|
"{name}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -268,28 +268,26 @@ impl AsyncHDF5File {
|
|||||||
///
|
///
|
||||||
/// Reads the entire file into memory, then parses the superblock.
|
/// Reads the entire file into memory, then parses the superblock.
|
||||||
pub async fn open<R: AsyncHDF5Read>(reader: &R) -> Result<Self, AsyncHDF5Error> {
|
pub async fn open<R: AsyncHDF5Read>(reader: &R) -> Result<Self, AsyncHDF5Error> {
|
||||||
let data = reader.read_all().await?;
|
Self::from_bytes(reader.read_all().await?)
|
||||||
let sig_offset = find_signature(&data)?;
|
|
||||||
let superblock = Superblock::parse(&data, sig_offset)?;
|
|
||||||
Ok(Self { data, superblock })
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Open an HDF5 file asynchronously from a file path.
|
/// Open an HDF5 file asynchronously from a file path.
|
||||||
pub async fn open_path<P: AsRef<Path>>(path: P) -> Result<Self, AsyncHDF5Error> {
|
pub async fn open_path<P: AsRef<Path>>(path: P) -> Result<Self, AsyncHDF5Error> {
|
||||||
let data = tokio::fs::read(path).await?;
|
Self::from_bytes(tokio::fs::read(path).await?)
|
||||||
let sig_offset = find_signature(&data)?;
|
|
||||||
let superblock = Superblock::parse(&data, sig_offset)?;
|
|
||||||
Ok(Self { data, superblock })
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Open an HDF5 file from bytes already in memory.
|
/// Open an HDF5 file from bytes already in memory.
|
||||||
pub fn from_bytes(data: Vec<u8>) -> Result<Self, AsyncHDF5Error> {
|
pub fn from_bytes(mut data: Vec<u8>) -> Result<Self, AsyncHDF5Error> {
|
||||||
let sig_offset = find_signature(&data)?;
|
// HDF5 addresses are relative to the superblock: drop any user block
|
||||||
let superblock = Superblock::parse(&data, sig_offset)?;
|
// so they index `data` directly.
|
||||||
|
let user_block = find_signature(&data)?;
|
||||||
|
data.drain(..user_block);
|
||||||
|
let superblock = Superblock::parse(&data, 0)?;
|
||||||
Ok(Self { data, superblock })
|
Ok(Self { data, superblock })
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Access the raw file bytes.
|
/// Access the file bytes from the superblock on (any user block is
|
||||||
|
/// dropped on open).
|
||||||
pub fn as_bytes(&self) -> &[u8] {
|
pub fn as_bytes(&self) -> &[u8] {
|
||||||
&self.data
|
&self.data
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -180,7 +180,7 @@ fn mpi_collective_read(vol: &MpiVol, location: &str, path: &str) -> Result<Vec<u
|
|||||||
use clawhdf5_format::{
|
use clawhdf5_format::{
|
||||||
data_layout::DataLayout, data_read::read_raw_data_full, dataspace::Dataspace,
|
data_layout::DataLayout, data_read::read_raw_data_full, dataspace::Dataspace,
|
||||||
datatype::Datatype, filter_pipeline::FilterPipeline, group_v2::resolve_path_any,
|
datatype::Datatype, filter_pipeline::FilterPipeline, group_v2::resolve_path_any,
|
||||||
message_type::MessageType, object_header::ObjectHeader, signature::find_signature,
|
message_type::MessageType, object_header::ObjectHeader, signature::split_user_block,
|
||||||
superblock::Superblock,
|
superblock::Superblock,
|
||||||
};
|
};
|
||||||
use mpi::traits::*;
|
use mpi::traits::*;
|
||||||
@@ -192,9 +192,10 @@ fn mpi_collective_read(vol: &MpiVol, location: &str, path: &str) -> Result<Vec<u
|
|||||||
let mut len_buf = [0usize; 1];
|
let mut len_buf = [0usize; 1];
|
||||||
|
|
||||||
if rank == 0 {
|
if rank == 0 {
|
||||||
let bytes = std::fs::read(location).map_err(VolError::Io)?;
|
let file = std::fs::read(location).map_err(VolError::Io)?;
|
||||||
let sig = find_signature(&bytes).map_err(|e| VolError::DataError(e.to_string()))?;
|
// Addresses are relative to the superblock: skip any user block.
|
||||||
let sb = Superblock::parse(&bytes, sig).map_err(|e| VolError::DataError(e.to_string()))?;
|
let (_, bytes) = split_user_block(&file).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||||
|
let sb = Superblock::parse(bytes, 0).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||||
let addr = resolve_path_any(&bytes, &sb, path)
|
let addr = resolve_path_any(&bytes, &sb, path)
|
||||||
.map_err(|e| VolError::NotFound(format!("{path}: {e}")))?;
|
.map_err(|e| VolError::NotFound(format!("{path}: {e}")))?;
|
||||||
let oh = ObjectHeader::parse(&bytes, addr as usize, sb.offset_size, sb.length_size)
|
let oh = ObjectHeader::parse(&bytes, addr as usize, sb.offset_size, sb.length_size)
|
||||||
|
|||||||
@@ -283,12 +283,13 @@ impl VirtualObjectLayer for NativeVol {
|
|||||||
use clawhdf5_format::{
|
use clawhdf5_format::{
|
||||||
data_layout::DataLayout, data_read::read_raw_data_full, dataspace::Dataspace,
|
data_layout::DataLayout, data_read::read_raw_data_full, dataspace::Dataspace,
|
||||||
datatype::Datatype, filter_pipeline::FilterPipeline, group_v2::resolve_path_any,
|
datatype::Datatype, filter_pipeline::FilterPipeline, group_v2::resolve_path_any,
|
||||||
message_type::MessageType, object_header::ObjectHeader, signature::find_signature,
|
message_type::MessageType, object_header::ObjectHeader, signature::split_user_block,
|
||||||
superblock::Superblock,
|
superblock::Superblock,
|
||||||
};
|
};
|
||||||
|
|
||||||
let sig = find_signature(data).map_err(|e| VolError::DataError(e.to_string()))?;
|
// Addresses are relative to the superblock: skip any user block.
|
||||||
let sb = Superblock::parse(data, sig).map_err(|e| VolError::DataError(e.to_string()))?;
|
let (_, data) = split_user_block(data).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||||
|
let sb = Superblock::parse(data, 0).map_err(|e| VolError::DataError(e.to_string()))?;
|
||||||
let addr = resolve_path_any(data, &sb, path)
|
let addr = resolve_path_any(data, &sb, path)
|
||||||
.map_err(|e| VolError::NotFound(format!("{path}: {e}")))?;
|
.map_err(|e| VolError::NotFound(format!("{path}: {e}")))?;
|
||||||
|
|
||||||
|
|||||||
+71
-68
@@ -12,25 +12,23 @@
|
|||||||
use std::cell::RefCell;
|
use std::cell::RefCell;
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
use clawhdf5_format::attribute::extract_attributes_full;
|
|
||||||
use clawhdf5_format::data_layout::DataLayout;
|
use clawhdf5_format::data_layout::DataLayout;
|
||||||
use clawhdf5_format::data_read;
|
use clawhdf5_format::data_read;
|
||||||
use clawhdf5_format::dataspace::Dataspace;
|
use clawhdf5_format::dataspace::Dataspace;
|
||||||
use clawhdf5_format::datatype::Datatype;
|
use clawhdf5_format::datatype::Datatype;
|
||||||
use clawhdf5_format::error::FormatError;
|
use clawhdf5_format::error::FormatError;
|
||||||
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||||
use clawhdf5_format::group_v1::{self, GroupEntry};
|
use clawhdf5_format::group_v1::GroupEntry;
|
||||||
use clawhdf5_format::group_v2;
|
use clawhdf5_format::group_v2;
|
||||||
use clawhdf5_format::message_type::MessageType;
|
use clawhdf5_format::message_type::MessageType;
|
||||||
use clawhdf5_format::object_header::ObjectHeader;
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
use clawhdf5_format::signature;
|
use clawhdf5_format::signature;
|
||||||
use clawhdf5_format::superblock::Superblock;
|
use clawhdf5_format::superblock::Superblock;
|
||||||
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
|
||||||
|
|
||||||
use clawhdf5_io::HDF5Read;
|
use clawhdf5_io::HDF5Read;
|
||||||
|
|
||||||
use crate::error::Error;
|
use crate::error::Error;
|
||||||
use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
use crate::types::{AttrValue, DType, classify_datatype, read_attrs};
|
||||||
|
|
||||||
/// A lazy HDF5 file handle that parses metadata on demand.
|
/// A lazy HDF5 file handle that parses metadata on demand.
|
||||||
///
|
///
|
||||||
@@ -42,6 +40,9 @@ use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
|||||||
/// `MemoryReader`, etc.
|
/// `MemoryReader`, etc.
|
||||||
pub struct LazyFile<R: HDF5Read> {
|
pub struct LazyFile<R: HDF5Read> {
|
||||||
reader: R,
|
reader: R,
|
||||||
|
/// Offset of the superblock in the file (the user-block size); every
|
||||||
|
/// HDF5 address is relative to it.
|
||||||
|
base: usize,
|
||||||
superblock: Superblock,
|
superblock: Superblock,
|
||||||
root_header: ObjectHeader,
|
root_header: ObjectHeader,
|
||||||
/// Cache of parsed object headers, keyed by address.
|
/// Cache of parsed object headers, keyed by address.
|
||||||
@@ -73,9 +74,9 @@ impl<R: HDF5Read> LazyFile<R> {
|
|||||||
///
|
///
|
||||||
/// Parses only the superblock and root group object header.
|
/// Parses only the superblock and root group object header.
|
||||||
pub fn open(reader: R) -> Result<Self, Error> {
|
pub fn open(reader: R) -> Result<Self, Error> {
|
||||||
let data = reader.as_bytes();
|
let (user_block, data) = signature::split_user_block(reader.as_bytes())?;
|
||||||
let sig_offset = signature::find_signature(data)?;
|
let base = user_block.len();
|
||||||
let superblock = Superblock::parse(data, sig_offset)?;
|
let superblock = Superblock::parse(data, 0)?;
|
||||||
let root_header = ObjectHeader::parse(
|
let root_header = ObjectHeader::parse(
|
||||||
data,
|
data,
|
||||||
superblock.root_group_address as usize,
|
superblock.root_group_address as usize,
|
||||||
@@ -84,15 +85,26 @@ impl<R: HDF5Read> LazyFile<R> {
|
|||||||
)?;
|
)?;
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
reader,
|
reader,
|
||||||
|
base,
|
||||||
superblock,
|
superblock,
|
||||||
root_header,
|
root_header,
|
||||||
header_cache: RefCell::new(HashMap::new()),
|
header_cache: RefCell::new(HashMap::new()),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns the raw file bytes.
|
/// Returns the file's bytes from the superblock on (after any user
|
||||||
|
/// block), which is the space every HDF5 address in the file indexes.
|
||||||
pub fn as_bytes(&self) -> &[u8] {
|
pub fn as_bytes(&self) -> &[u8] {
|
||||||
self.reader.as_bytes()
|
self.hdf5_bytes()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Size of the user block before the superblock (0 for most files).
|
||||||
|
pub fn user_block_size(&self) -> u64 {
|
||||||
|
self.base as u64
|
||||||
|
}
|
||||||
|
|
||||||
|
fn hdf5_bytes(&self) -> &[u8] {
|
||||||
|
&self.reader.as_bytes()[self.base..]
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns a reference to the parsed superblock.
|
/// Returns a reference to the parsed superblock.
|
||||||
@@ -110,7 +122,7 @@ impl<R: HDF5Read> LazyFile<R> {
|
|||||||
|
|
||||||
/// Resolve a path and return a `LazyDataset` handle.
|
/// Resolve a path and return a `LazyDataset` handle.
|
||||||
pub fn dataset(&self, path: &str) -> Result<LazyDataset<'_, R>, Error> {
|
pub fn dataset(&self, path: &str) -> Result<LazyDataset<'_, R>, Error> {
|
||||||
let data = self.reader.as_bytes();
|
let data = self.hdf5_bytes();
|
||||||
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
||||||
let hdr = self.get_or_parse_header(addr)?;
|
let hdr = self.get_or_parse_header(addr)?;
|
||||||
if !has_message(&hdr, MessageType::DataLayout) {
|
if !has_message(&hdr, MessageType::DataLayout) {
|
||||||
@@ -124,7 +136,7 @@ impl<R: HDF5Read> LazyFile<R> {
|
|||||||
|
|
||||||
/// Resolve a path and return a `LazyGroup` handle.
|
/// Resolve a path and return a `LazyGroup` handle.
|
||||||
pub fn group(&self, path: &str) -> Result<LazyGroup<'_, R>, Error> {
|
pub fn group(&self, path: &str) -> Result<LazyGroup<'_, R>, Error> {
|
||||||
let data = self.reader.as_bytes();
|
let data = self.hdf5_bytes();
|
||||||
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
||||||
Ok(LazyGroup {
|
Ok(LazyGroup {
|
||||||
file: self,
|
file: self,
|
||||||
@@ -163,7 +175,7 @@ impl<R: HDF5Read> LazyFile<R> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Parse and cache
|
// Parse and cache
|
||||||
let data = self.reader.as_bytes();
|
let data = self.hdf5_bytes();
|
||||||
let hdr = ObjectHeader::parse(
|
let hdr = ObjectHeader::parse(
|
||||||
data,
|
data,
|
||||||
address as usize,
|
address as usize,
|
||||||
@@ -187,7 +199,7 @@ impl<R: HDF5Read> LazyFile<R> {
|
|||||||
impl<R: HDF5Read> std::fmt::Debug for LazyFile<R> {
|
impl<R: HDF5Read> std::fmt::Debug for LazyFile<R> {
|
||||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
f.debug_struct("LazyFile")
|
f.debug_struct("LazyFile")
|
||||||
.field("size", &self.reader.as_bytes().len())
|
.field("size", &self.hdf5_bytes().len())
|
||||||
.field("superblock_version", &self.superblock.version)
|
.field("superblock_version", &self.superblock.version)
|
||||||
.field("cached_headers", &self.header_cache.borrow().len())
|
.field("cached_headers", &self.header_cache.borrow().len())
|
||||||
.finish()
|
.finish()
|
||||||
@@ -232,17 +244,26 @@ impl<'f, R: HDF5Read> LazyGroup<'f, R> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read all attributes of this group.
|
/// Read all attributes of this group.
|
||||||
|
///
|
||||||
|
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||||
|
/// message, or a dense-storage heap object that cannot be located — is
|
||||||
|
/// left out of the map instead of failing every attribute on the object;
|
||||||
|
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||||
|
/// Values that are returned are complete (never partially decoded). An
|
||||||
|
/// error in the index of the attributes itself (the attribute info
|
||||||
|
/// message, the dense heap header or B-tree) still fails the call.
|
||||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||||
|
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||||
|
/// attribute that could not be read and was left out.
|
||||||
|
pub fn attrs_with_errors(
|
||||||
|
&self,
|
||||||
|
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||||
let hdr = self.file.get_or_parse_header(self.address)?;
|
let hdr = self.file.get_or_parse_header(self.address)?;
|
||||||
let data = self.file.reader.as_bytes();
|
let data = self.file.hdf5_bytes();
|
||||||
let attr_msgs =
|
read_attrs(data, &hdr, self.file.offset_size(), self.file.length_size())
|
||||||
extract_attributes_full(data, &hdr, self.file.offset_size(), self.file.length_size())?;
|
|
||||||
Ok(attrs_to_map(
|
|
||||||
&attr_msgs,
|
|
||||||
data,
|
|
||||||
self.file.offset_size(),
|
|
||||||
self.file.length_size(),
|
|
||||||
))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Get a dataset within this group by name.
|
/// Get a dataset within this group by name.
|
||||||
@@ -275,12 +296,14 @@ impl<'f, R: HDF5Read> LazyGroup<'f, R> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// This group's links that can be opened: hard links, and soft links
|
||||||
|
/// resolved to their targets (see
|
||||||
|
/// [`group_v2::resolve_group_children`]); dangling, external and
|
||||||
|
/// user-defined links are left out.
|
||||||
fn children(&self) -> Result<Vec<GroupEntry>, Error> {
|
fn children(&self) -> Result<Vec<GroupEntry>, Error> {
|
||||||
let hdr = self.file.get_or_parse_header(self.address)?;
|
let data = self.file.hdf5_bytes();
|
||||||
let data = self.file.reader.as_bytes();
|
group_v2::resolve_group_children(data, &self.file.superblock, self.address)
|
||||||
let os = self.file.offset_size();
|
.map_err(Error::Format)
|
||||||
let ls = self.file.length_size();
|
|
||||||
resolve_group_entries(data, &hdr, os, ls).map_err(Error::Format)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -360,7 +383,7 @@ impl<'f, R: HDF5Read> LazyDataset<'f, R> {
|
|||||||
let dl = self.data_layout()?;
|
let dl = self.data_layout()?;
|
||||||
let ds = self.dataspace()?;
|
let ds = self.dataspace()?;
|
||||||
let dt = self.datatype()?;
|
let dt = self.datatype()?;
|
||||||
let slice = data_read::read_raw_data_zerocopy(self.file.reader.as_bytes(), &dl, &ds, &dt)?;
|
let slice = data_read::read_raw_data_zerocopy(self.file.hdf5_bytes(), &dl, &ds, &dt)?;
|
||||||
Ok(slice)
|
Ok(slice)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -400,20 +423,30 @@ impl<'f, R: HDF5Read> LazyDataset<'f, R> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read all attributes of this dataset.
|
/// Read all attributes of this dataset.
|
||||||
|
///
|
||||||
|
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||||
|
/// message, or a dense-storage heap object that cannot be located — is
|
||||||
|
/// left out of the map instead of failing every attribute on the object;
|
||||||
|
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||||
|
/// Values that are returned are complete (never partially decoded). An
|
||||||
|
/// error in the index of the attributes itself (the attribute info
|
||||||
|
/// message, the dense heap header or B-tree) still fails the call.
|
||||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||||
let data = self.file.reader.as_bytes();
|
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||||
let attr_msgs = extract_attributes_full(
|
}
|
||||||
|
|
||||||
|
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||||
|
/// attribute that could not be read and was left out.
|
||||||
|
pub fn attrs_with_errors(
|
||||||
|
&self,
|
||||||
|
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||||
|
let data = self.file.hdf5_bytes();
|
||||||
|
read_attrs(
|
||||||
data,
|
data,
|
||||||
&self.header,
|
&self.header,
|
||||||
self.file.offset_size(),
|
self.file.offset_size(),
|
||||||
self.file.length_size(),
|
self.file.length_size(),
|
||||||
)?;
|
)
|
||||||
Ok(attrs_to_map(
|
|
||||||
&attr_msgs,
|
|
||||||
data,
|
|
||||||
self.file.offset_size(),
|
|
||||||
self.file.length_size(),
|
|
||||||
))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A header message's payload, resolved through the shared-message
|
/// A header message's payload, resolved through the shared-message
|
||||||
@@ -479,7 +512,7 @@ impl<'f, R: HDF5Read> LazyDataset<'f, R> {
|
|||||||
let ds = self.dataspace()?;
|
let ds = self.dataspace()?;
|
||||||
let dl = self.data_layout()?;
|
let dl = self.data_layout()?;
|
||||||
let pipeline = self.filter_pipeline()?;
|
let pipeline = self.filter_pipeline()?;
|
||||||
let data = self.file.reader.as_bytes();
|
let data = self.file.hdf5_bytes();
|
||||||
// Unallocated storage reads as the dataset's fill value.
|
// Unallocated storage reads as the dataset's fill value.
|
||||||
clawhdf5_format::fill_value::read_full_with_fill(
|
clawhdf5_format::fill_value::read_full_with_fill(
|
||||||
&self.header.messages,
|
&self.header.messages,
|
||||||
@@ -530,33 +563,3 @@ fn is_group(header: &ObjectHeader) -> bool {
|
|||||||
|| m.msg_type == MessageType::SymbolTable
|
|| m.msg_type == MessageType::SymbolTable
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn resolve_group_entries(
|
|
||||||
file_data: &[u8],
|
|
||||||
object_header: &ObjectHeader,
|
|
||||||
offset_size: u8,
|
|
||||||
length_size: u8,
|
|
||||||
) -> Result<Vec<GroupEntry>, FormatError> {
|
|
||||||
let is_v1 = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.any(|m| m.msg_type == MessageType::SymbolTable);
|
|
||||||
let is_v2 = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link);
|
|
||||||
|
|
||||||
if is_v1 {
|
|
||||||
let sym_msg = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.find(|m| m.msg_type == MessageType::SymbolTable)
|
|
||||||
.ok_or_else(|| FormatError::PathNotFound("no symbol table message".into()))?;
|
|
||||||
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
|
||||||
group_v1::resolve_v1_group_entries(file_data, &stm, offset_size, length_size)
|
|
||||||
} else if is_v2 {
|
|
||||||
group_v2::resolve_v2_group_entries(file_data, object_header, offset_size, length_size)
|
|
||||||
} else {
|
|
||||||
Ok(Vec::new())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -7,25 +7,23 @@
|
|||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
use clawhdf5_format::attribute::extract_attributes_full;
|
|
||||||
use clawhdf5_format::data_layout::DataLayout;
|
use clawhdf5_format::data_layout::DataLayout;
|
||||||
use clawhdf5_format::data_read;
|
use clawhdf5_format::data_read;
|
||||||
use clawhdf5_format::dataspace::Dataspace;
|
use clawhdf5_format::dataspace::Dataspace;
|
||||||
use clawhdf5_format::datatype::Datatype;
|
use clawhdf5_format::datatype::Datatype;
|
||||||
use clawhdf5_format::error::FormatError;
|
use clawhdf5_format::error::FormatError;
|
||||||
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||||
use clawhdf5_format::group_v1::{self, GroupEntry};
|
use clawhdf5_format::group_v1::GroupEntry;
|
||||||
use clawhdf5_format::group_v2;
|
use clawhdf5_format::group_v2;
|
||||||
use clawhdf5_format::message_type::MessageType;
|
use clawhdf5_format::message_type::MessageType;
|
||||||
use clawhdf5_format::object_header::ObjectHeader;
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
use clawhdf5_format::signature;
|
use clawhdf5_format::signature;
|
||||||
use clawhdf5_format::superblock::Superblock;
|
use clawhdf5_format::superblock::Superblock;
|
||||||
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
|
||||||
|
|
||||||
use clawhdf5_io::MmapReader;
|
use clawhdf5_io::MmapReader;
|
||||||
|
|
||||||
use crate::error::Error;
|
use crate::error::Error;
|
||||||
use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
use crate::types::{AttrValue, DType, classify_datatype, read_attrs};
|
||||||
|
|
||||||
/// An HDF5 file opened via memory mapping.
|
/// An HDF5 file opened via memory mapping.
|
||||||
///
|
///
|
||||||
@@ -34,6 +32,9 @@ use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
|||||||
/// `&[u8]` slice via [`MmapDataset::read_raw_slice`].
|
/// `&[u8]` slice via [`MmapDataset::read_raw_slice`].
|
||||||
pub struct MmapFile {
|
pub struct MmapFile {
|
||||||
reader: MmapReader,
|
reader: MmapReader,
|
||||||
|
/// Offset of the superblock in the mapped file (the user-block size);
|
||||||
|
/// every HDF5 address is relative to it.
|
||||||
|
base: usize,
|
||||||
superblock: Superblock,
|
superblock: Superblock,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -41,10 +42,25 @@ impl MmapFile {
|
|||||||
/// Open an HDF5 file using memory-mapped I/O.
|
/// Open an HDF5 file using memory-mapped I/O.
|
||||||
pub fn open<P: AsRef<std::path::Path>>(path: P) -> Result<Self, Error> {
|
pub fn open<P: AsRef<std::path::Path>>(path: P) -> Result<Self, Error> {
|
||||||
let reader = MmapReader::open(path).map_err(Error::Io)?;
|
let reader = MmapReader::open(path).map_err(Error::Io)?;
|
||||||
let data = reader.as_bytes();
|
let (user_block, data) = signature::split_user_block(reader.as_bytes())?;
|
||||||
let sig_offset = signature::find_signature(data)?;
|
let base = user_block.len();
|
||||||
let superblock = Superblock::parse(data, sig_offset)?;
|
let superblock = Superblock::parse(data, 0)?;
|
||||||
Ok(Self { reader, superblock })
|
Ok(Self {
|
||||||
|
reader,
|
||||||
|
base,
|
||||||
|
superblock,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The file's bytes from the superblock on — the space HDF5 addresses
|
||||||
|
/// index into.
|
||||||
|
fn hdf5_bytes(&self) -> &[u8] {
|
||||||
|
&self.reader.as_bytes()[self.base..]
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Size of the user block before the superblock (0 for most files).
|
||||||
|
pub fn user_block_size(&self) -> u64 {
|
||||||
|
self.base as u64
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns a handle to the root group.
|
/// Returns a handle to the root group.
|
||||||
@@ -57,7 +73,7 @@ impl MmapFile {
|
|||||||
|
|
||||||
/// Resolve a path and return a `MmapDataset` handle.
|
/// Resolve a path and return a `MmapDataset` handle.
|
||||||
pub fn dataset(&self, path: &str) -> Result<MmapDataset<'_>, Error> {
|
pub fn dataset(&self, path: &str) -> Result<MmapDataset<'_>, Error> {
|
||||||
let data = self.reader.as_bytes();
|
let data = self.hdf5_bytes();
|
||||||
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
||||||
let hdr = self.parse_header(addr)?;
|
let hdr = self.parse_header(addr)?;
|
||||||
if !has_message(&hdr, MessageType::DataLayout) {
|
if !has_message(&hdr, MessageType::DataLayout) {
|
||||||
@@ -71,7 +87,7 @@ impl MmapFile {
|
|||||||
|
|
||||||
/// Resolve a path and return a `MmapGroup` handle.
|
/// Resolve a path and return a `MmapGroup` handle.
|
||||||
pub fn group(&self, path: &str) -> Result<MmapGroup<'_>, Error> {
|
pub fn group(&self, path: &str) -> Result<MmapGroup<'_>, Error> {
|
||||||
let data = self.reader.as_bytes();
|
let data = self.hdf5_bytes();
|
||||||
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
let addr = group_v2::resolve_path_any(data, &self.superblock, path)?;
|
||||||
Ok(MmapGroup {
|
Ok(MmapGroup {
|
||||||
file: self,
|
file: self,
|
||||||
@@ -79,9 +95,11 @@ impl MmapFile {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns the raw file bytes (zero-copy from mmap).
|
/// Returns the file's bytes from the superblock on (after any user
|
||||||
|
/// block), zero-copy from the mmap. Every HDF5 address in the file
|
||||||
|
/// indexes this slice.
|
||||||
pub fn as_bytes(&self) -> &[u8] {
|
pub fn as_bytes(&self) -> &[u8] {
|
||||||
self.reader.as_bytes()
|
self.hdf5_bytes()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns a reference to the parsed superblock.
|
/// Returns a reference to the parsed superblock.
|
||||||
@@ -91,7 +109,7 @@ impl MmapFile {
|
|||||||
|
|
||||||
fn parse_header(&self, address: u64) -> Result<ObjectHeader, FormatError> {
|
fn parse_header(&self, address: u64) -> Result<ObjectHeader, FormatError> {
|
||||||
ObjectHeader::parse(
|
ObjectHeader::parse(
|
||||||
self.reader.as_bytes(),
|
self.hdf5_bytes(),
|
||||||
address as usize,
|
address as usize,
|
||||||
self.superblock.offset_size,
|
self.superblock.offset_size,
|
||||||
self.superblock.length_size,
|
self.superblock.length_size,
|
||||||
@@ -154,17 +172,26 @@ impl<'f> MmapGroup<'f> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read all attributes of this group.
|
/// Read all attributes of this group.
|
||||||
|
///
|
||||||
|
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||||
|
/// message, or a dense-storage heap object that cannot be located — is
|
||||||
|
/// left out of the map instead of failing every attribute on the object;
|
||||||
|
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||||
|
/// Values that are returned are complete (never partially decoded). An
|
||||||
|
/// error in the index of the attributes itself (the attribute info
|
||||||
|
/// message, the dense heap header or B-tree) still fails the call.
|
||||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||||
let data = self.file.reader.as_bytes();
|
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||||
|
/// attribute that could not be read and was left out.
|
||||||
|
pub fn attrs_with_errors(
|
||||||
|
&self,
|
||||||
|
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||||
let hdr = self.file.parse_header(self.address)?;
|
let hdr = self.file.parse_header(self.address)?;
|
||||||
let attr_msgs =
|
let data = self.file.hdf5_bytes();
|
||||||
extract_attributes_full(data, &hdr, self.file.offset_size(), self.file.length_size())?;
|
read_attrs(data, &hdr, self.file.offset_size(), self.file.length_size())
|
||||||
Ok(attrs_to_map(
|
|
||||||
&attr_msgs,
|
|
||||||
data,
|
|
||||||
self.file.offset_size(),
|
|
||||||
self.file.length_size(),
|
|
||||||
))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Get a dataset within this group by name.
|
/// Get a dataset within this group by name.
|
||||||
@@ -197,12 +224,14 @@ impl<'f> MmapGroup<'f> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// This group's links that can be opened: hard links, and soft links
|
||||||
|
/// resolved to their targets (see
|
||||||
|
/// [`group_v2::resolve_group_children`]); dangling, external and
|
||||||
|
/// user-defined links are left out.
|
||||||
fn children(&self) -> Result<Vec<GroupEntry>, Error> {
|
fn children(&self) -> Result<Vec<GroupEntry>, Error> {
|
||||||
let data = self.file.reader.as_bytes();
|
let data = self.file.hdf5_bytes();
|
||||||
let hdr = self.file.parse_header(self.address)?;
|
group_v2::resolve_group_children(data, &self.file.superblock, self.address)
|
||||||
let os = self.file.offset_size();
|
.map_err(Error::Format)
|
||||||
let ls = self.file.length_size();
|
|
||||||
resolve_group_entries(data, &hdr, os, ls).map_err(Error::Format)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -326,7 +355,7 @@ impl<'f> MmapDataset<'f> {
|
|||||||
actual: sz,
|
actual: sz,
|
||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
let data = self.file.reader.as_bytes();
|
let data = self.file.hdf5_bytes();
|
||||||
let a = addr as usize;
|
let a = addr as usize;
|
||||||
if a + sz > data.len() {
|
if a + sz > data.len() {
|
||||||
return Err(Error::Format(FormatError::UnexpectedEof {
|
return Err(Error::Format(FormatError::UnexpectedEof {
|
||||||
@@ -341,20 +370,30 @@ impl<'f> MmapDataset<'f> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read all attributes of this dataset.
|
/// Read all attributes of this dataset.
|
||||||
|
///
|
||||||
|
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||||
|
/// message, or a dense-storage heap object that cannot be located — is
|
||||||
|
/// left out of the map instead of failing every attribute on the object;
|
||||||
|
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||||
|
/// Values that are returned are complete (never partially decoded). An
|
||||||
|
/// error in the index of the attributes itself (the attribute info
|
||||||
|
/// message, the dense heap header or B-tree) still fails the call.
|
||||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||||
let data = self.file.reader.as_bytes();
|
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||||
let attr_msgs = extract_attributes_full(
|
}
|
||||||
|
|
||||||
|
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||||
|
/// attribute that could not be read and was left out.
|
||||||
|
pub fn attrs_with_errors(
|
||||||
|
&self,
|
||||||
|
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||||
|
let data = self.file.hdf5_bytes();
|
||||||
|
read_attrs(
|
||||||
data,
|
data,
|
||||||
&self.header,
|
&self.header,
|
||||||
self.file.offset_size(),
|
self.file.offset_size(),
|
||||||
self.file.length_size(),
|
self.file.length_size(),
|
||||||
)?;
|
)
|
||||||
Ok(attrs_to_map(
|
|
||||||
&attr_msgs,
|
|
||||||
data,
|
|
||||||
self.file.offset_size(),
|
|
||||||
self.file.length_size(),
|
|
||||||
))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A header message's payload, resolved through the shared-message
|
/// A header message's payload, resolved through the shared-message
|
||||||
@@ -423,7 +462,7 @@ impl<'f> MmapDataset<'f> {
|
|||||||
// Unallocated storage reads as the dataset's fill value.
|
// Unallocated storage reads as the dataset's fill value.
|
||||||
clawhdf5_format::fill_value::read_full_with_fill(
|
clawhdf5_format::fill_value::read_full_with_fill(
|
||||||
&self.header.messages,
|
&self.header.messages,
|
||||||
self.file.reader.as_bytes(),
|
self.file.hdf5_bytes(),
|
||||||
&dl,
|
&dl,
|
||||||
&ds,
|
&ds,
|
||||||
dt.type_size() as usize,
|
dt.type_size() as usize,
|
||||||
@@ -431,7 +470,7 @@ impl<'f> MmapDataset<'f> {
|
|||||||
self.file.length_size(),
|
self.file.length_size(),
|
||||||
|| {
|
|| {
|
||||||
Ok(data_read::read_raw_data_full(
|
Ok(data_read::read_raw_data_full(
|
||||||
self.file.reader.as_bytes(),
|
self.file.hdf5_bytes(),
|
||||||
&dl,
|
&dl,
|
||||||
&ds,
|
&ds,
|
||||||
&dt,
|
&dt,
|
||||||
@@ -470,33 +509,3 @@ fn is_group(header: &ObjectHeader) -> bool {
|
|||||||
|| m.msg_type == MessageType::SymbolTable
|
|| m.msg_type == MessageType::SymbolTable
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn resolve_group_entries(
|
|
||||||
file_data: &[u8],
|
|
||||||
object_header: &ObjectHeader,
|
|
||||||
offset_size: u8,
|
|
||||||
length_size: u8,
|
|
||||||
) -> Result<Vec<GroupEntry>, FormatError> {
|
|
||||||
let is_v1 = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.any(|m| m.msg_type == MessageType::SymbolTable);
|
|
||||||
let is_v2 = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link);
|
|
||||||
|
|
||||||
if is_v1 {
|
|
||||||
let sym_msg = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.find(|m| m.msg_type == MessageType::SymbolTable)
|
|
||||||
.ok_or_else(|| FormatError::PathNotFound("no symbol table message".into()))?;
|
|
||||||
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
|
||||||
group_v1::resolve_v1_group_entries(file_data, &stm, offset_size, length_size)
|
|
||||||
} else if is_v2 {
|
|
||||||
group_v2::resolve_v2_group_entries(file_data, object_header, offset_size, length_size)
|
|
||||||
} else {
|
|
||||||
Ok(Vec::new())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
+169
-91
@@ -7,7 +7,6 @@
|
|||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
|
|
||||||
use clawhdf5_format::attribute::extract_attributes_full;
|
|
||||||
use clawhdf5_format::chunk_cache::ChunkCache;
|
use clawhdf5_format::chunk_cache::ChunkCache;
|
||||||
use clawhdf5_format::data_layout::DataLayout;
|
use clawhdf5_format::data_layout::DataLayout;
|
||||||
use clawhdf5_format::data_read;
|
use clawhdf5_format::data_read;
|
||||||
@@ -15,36 +14,58 @@ use clawhdf5_format::dataspace::Dataspace;
|
|||||||
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||||
use clawhdf5_format::error::FormatError;
|
use clawhdf5_format::error::FormatError;
|
||||||
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
use clawhdf5_format::filter_pipeline::FilterPipeline;
|
||||||
use clawhdf5_format::group_v1::{self, GroupEntry};
|
use clawhdf5_format::group_v1::GroupEntry;
|
||||||
use clawhdf5_format::group_v2;
|
use clawhdf5_format::group_v2;
|
||||||
use clawhdf5_format::message_type::MessageType;
|
use clawhdf5_format::message_type::MessageType;
|
||||||
use clawhdf5_format::object_header::ObjectHeader;
|
use clawhdf5_format::object_header::ObjectHeader;
|
||||||
use clawhdf5_format::signature;
|
use clawhdf5_format::signature;
|
||||||
use clawhdf5_format::superblock::Superblock;
|
use clawhdf5_format::superblock::Superblock;
|
||||||
use clawhdf5_format::symbol_table::SymbolTableMessage;
|
|
||||||
|
|
||||||
use crate::error::Error;
|
use crate::error::Error;
|
||||||
use crate::types::{AttrValue, DType, attrs_to_map, classify_datatype};
|
use crate::types::{AttrValue, DType, classify_datatype, read_attrs};
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
// FileData — internal storage for either owned bytes or an mmap
|
// FileData — internal storage for either owned bytes or an mmap
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
|
|
||||||
/// Internal storage: either an owned `Vec<u8>` or a memory-mapped region.
|
/// Internal storage: either an owned `Vec<u8>` or a memory-mapped region.
|
||||||
enum FileData {
|
enum Backing {
|
||||||
Owned(Vec<u8>),
|
Owned(Vec<u8>),
|
||||||
#[cfg(feature = "mmap")]
|
#[cfg(feature = "mmap")]
|
||||||
Mmap(clawhdf5_io::MmapReader),
|
Mmap(clawhdf5_io::MmapReader),
|
||||||
}
|
}
|
||||||
|
|
||||||
impl FileData {
|
impl Backing {
|
||||||
fn as_bytes(&self) -> &[u8] {
|
fn whole_file(&self) -> &[u8] {
|
||||||
match self {
|
match self {
|
||||||
FileData::Owned(v) => v,
|
Backing::Owned(v) => v,
|
||||||
#[cfg(feature = "mmap")]
|
#[cfg(feature = "mmap")]
|
||||||
FileData::Mmap(r) => r.as_bytes(),
|
Backing::Mmap(r) => r.as_bytes(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The file's bytes, viewed from the superblock on. A file may start with a
|
||||||
|
/// user block (the superblock at 512, 1024, …); every HDF5 address is
|
||||||
|
/// relative to the superblock, so all parsing goes through [`Self::as_bytes`].
|
||||||
|
struct FileData {
|
||||||
|
backing: Backing,
|
||||||
|
/// Offset of the superblock in the file (the user-block size).
|
||||||
|
base: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl FileData {
|
||||||
|
/// Locate the superblock and parse it.
|
||||||
|
fn new(backing: Backing) -> Result<(Self, Superblock), Error> {
|
||||||
|
let (user_block, hdf5) = signature::split_user_block(backing.whole_file())?;
|
||||||
|
let base = user_block.len();
|
||||||
|
let superblock = Superblock::parse(hdf5, 0)?;
|
||||||
|
Ok((Self { backing, base }, superblock))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn as_bytes(&self) -> &[u8] {
|
||||||
|
&self.backing.whole_file()[self.base..]
|
||||||
|
}
|
||||||
|
|
||||||
fn len(&self) -> usize {
|
fn len(&self) -> usize {
|
||||||
self.as_bytes().len()
|
self.as_bytes().len()
|
||||||
@@ -81,11 +102,9 @@ impl File {
|
|||||||
#[cfg(feature = "mmap")]
|
#[cfg(feature = "mmap")]
|
||||||
{
|
{
|
||||||
let reader = clawhdf5_io::MmapReader::open(path).map_err(Error::Io)?;
|
let reader = clawhdf5_io::MmapReader::open(path).map_err(Error::Io)?;
|
||||||
let data_ref = reader.as_bytes();
|
let (data, superblock) = FileData::new(Backing::Mmap(reader))?;
|
||||||
let sig_offset = signature::find_signature(data_ref)?;
|
|
||||||
let superblock = Superblock::parse(data_ref, sig_offset)?;
|
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
data: FileData::Mmap(reader),
|
data,
|
||||||
superblock,
|
superblock,
|
||||||
chunk_cache: ChunkCache::new(),
|
chunk_cache: ChunkCache::new(),
|
||||||
base_dir,
|
base_dir,
|
||||||
@@ -116,10 +135,9 @@ impl File {
|
|||||||
/// In-memory files have no directory, so external Virtual Dataset sources
|
/// In-memory files have no directory, so external Virtual Dataset sources
|
||||||
/// cannot be resolved automatically (same-file VDS still works).
|
/// cannot be resolved automatically (same-file VDS still works).
|
||||||
pub fn from_bytes(data: Vec<u8>) -> Result<Self, Error> {
|
pub fn from_bytes(data: Vec<u8>) -> Result<Self, Error> {
|
||||||
let sig_offset = signature::find_signature(&data)?;
|
let (data, superblock) = FileData::new(Backing::Owned(data))?;
|
||||||
let superblock = Superblock::parse(&data, sig_offset)?;
|
|
||||||
Ok(Self {
|
Ok(Self {
|
||||||
data: FileData::Owned(data),
|
data,
|
||||||
superblock,
|
superblock,
|
||||||
chunk_cache: ChunkCache::new(),
|
chunk_cache: ChunkCache::new(),
|
||||||
base_dir: None,
|
base_dir: None,
|
||||||
@@ -209,11 +227,19 @@ impl File {
|
|||||||
Ok(results.into_iter().map(|(_, data)| data).collect())
|
Ok(results.into_iter().map(|(_, data)| data).collect())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Returns the raw file bytes.
|
/// Returns the file's bytes from the superblock on (after any user
|
||||||
|
/// block). Every HDF5 address in the file indexes this slice, so it is
|
||||||
|
/// what the `clawhdf5_format` parsers expect as `file_data`.
|
||||||
pub fn as_bytes(&self) -> &[u8] {
|
pub fn as_bytes(&self) -> &[u8] {
|
||||||
self.data.as_bytes()
|
self.data.as_bytes()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Size of the user block before the superblock (0 for most files).
|
||||||
|
/// Matches h5py's `File.userblock_size`.
|
||||||
|
pub fn user_block_size(&self) -> u64 {
|
||||||
|
self.data.base as u64
|
||||||
|
}
|
||||||
|
|
||||||
/// Returns a reference to the parsed superblock.
|
/// Returns a reference to the parsed superblock.
|
||||||
pub fn superblock(&self) -> &Superblock {
|
pub fn superblock(&self) -> &Superblock {
|
||||||
&self.superblock
|
&self.superblock
|
||||||
@@ -221,10 +247,10 @@ impl File {
|
|||||||
|
|
||||||
/// Returns `true` when the file is backed by memory-mapped I/O.
|
/// Returns `true` when the file is backed by memory-mapped I/O.
|
||||||
pub fn is_mmap(&self) -> bool {
|
pub fn is_mmap(&self) -> bool {
|
||||||
match &self.data {
|
match &self.data.backing {
|
||||||
FileData::Owned(_) => false,
|
Backing::Owned(_) => false,
|
||||||
#[cfg(feature = "mmap")]
|
#[cfg(feature = "mmap")]
|
||||||
FileData::Mmap(_) => true,
|
Backing::Mmap(_) => true,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -294,17 +320,26 @@ impl<'f> Group<'f> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read all attributes of this group.
|
/// Read all attributes of this group.
|
||||||
|
///
|
||||||
|
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||||
|
/// message, or a dense-storage heap object that cannot be located — is
|
||||||
|
/// left out of the map instead of failing every attribute on the object;
|
||||||
|
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||||
|
/// Values that are returned are complete (never partially decoded). An
|
||||||
|
/// error in the index of the attributes itself (the attribute info
|
||||||
|
/// message, the dense heap header or B-tree) still fails the call.
|
||||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||||
let data = self.file.data.as_bytes();
|
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||||
|
/// attribute that could not be read and was left out.
|
||||||
|
pub fn attrs_with_errors(
|
||||||
|
&self,
|
||||||
|
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||||
let hdr = self.file.parse_header(self.address)?;
|
let hdr = self.file.parse_header(self.address)?;
|
||||||
let attr_msgs =
|
let data = self.file.data.as_bytes();
|
||||||
extract_attributes_full(data, &hdr, self.file.offset_size(), self.file.length_size())?;
|
read_attrs(data, &hdr, self.file.offset_size(), self.file.length_size())
|
||||||
Ok(attrs_to_map(
|
|
||||||
&attr_msgs,
|
|
||||||
data,
|
|
||||||
self.file.offset_size(),
|
|
||||||
self.file.length_size(),
|
|
||||||
))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Get a dataset within this group by name.
|
/// Get a dataset within this group by name.
|
||||||
@@ -337,12 +372,14 @@ impl<'f> Group<'f> {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// This group's links that can be opened: hard links, and soft links
|
||||||
|
/// resolved to their targets (see
|
||||||
|
/// [`group_v2::resolve_group_children`]); dangling, external and
|
||||||
|
/// user-defined links are left out.
|
||||||
fn children(&self) -> Result<Vec<GroupEntry>, Error> {
|
fn children(&self) -> Result<Vec<GroupEntry>, Error> {
|
||||||
let data = self.file.data.as_bytes();
|
let data = self.file.data.as_bytes();
|
||||||
let hdr = self.file.parse_header(self.address)?;
|
group_v2::resolve_group_children(data, &self.file.superblock, self.address)
|
||||||
let os = self.file.offset_size();
|
.map_err(Error::Format)
|
||||||
let ls = self.file.length_size();
|
|
||||||
resolve_group_entries(data, &hdr, os, ls).map_err(Error::Format)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -478,8 +515,14 @@ impl<'f> Dataset<'f> {
|
|||||||
// sparse) dataset — select from a fill-aware full read instead. (The
|
// sparse) dataset — select from a fill-aware full read instead. (The
|
||||||
// selection reader currently decodes the full dataset too, so this
|
// selection reader currently decodes the full dataset too, so this
|
||||||
// costs nothing extra.)
|
// costs nothing extra.)
|
||||||
let fill = clawhdf5_format::fill_value::dataset_fill_value(&self.header.messages)?;
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||||
|
self.file.data.as_bytes(),
|
||||||
|
&self.header.messages,
|
||||||
|
self.file.offset_size(),
|
||||||
|
self.file.length_size(),
|
||||||
|
)?;
|
||||||
let fill_matters = !clawhdf5_format::fill_value::has_storage(&dl)
|
let fill_matters = !clawhdf5_format::fill_value::has_storage(&dl)
|
||||||
|
|| matches!(dl, DataLayout::Virtual { .. })
|
||||||
|| (matches!(dl, DataLayout::Chunked { .. })
|
|| (matches!(dl, DataLayout::Chunked { .. })
|
||||||
&& !clawhdf5_format::fill_value::is_default(fill.as_deref()));
|
&& !clawhdf5_format::fill_value::is_default(fill.as_deref()));
|
||||||
if fill_matters {
|
if fill_matters {
|
||||||
@@ -726,20 +769,30 @@ impl<'f> Dataset<'f> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read all attributes of this dataset.
|
/// Read all attributes of this dataset.
|
||||||
|
///
|
||||||
|
/// An attribute that cannot be read — a corrupt or unsupported attribute
|
||||||
|
/// message, or a dense-storage heap object that cannot be located — is
|
||||||
|
/// left out of the map instead of failing every attribute on the object;
|
||||||
|
/// [`attrs_with_errors`](Self::attrs_with_errors) reports which failed.
|
||||||
|
/// Values that are returned are complete (never partially decoded). An
|
||||||
|
/// error in the index of the attributes itself (the attribute info
|
||||||
|
/// message, the dense heap header or B-tree) still fails the call.
|
||||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||||
|
self.attrs_with_errors().map(|(attrs, _)| attrs)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Like [`attrs`](Self::attrs), also returning one error for each
|
||||||
|
/// attribute that could not be read and was left out.
|
||||||
|
pub fn attrs_with_errors(
|
||||||
|
&self,
|
||||||
|
) -> Result<(HashMap<String, AttrValue>, Vec<FormatError>), Error> {
|
||||||
let data = self.file.data.as_bytes();
|
let data = self.file.data.as_bytes();
|
||||||
let attr_msgs = extract_attributes_full(
|
read_attrs(
|
||||||
data,
|
data,
|
||||||
&self.header,
|
&self.header,
|
||||||
self.file.offset_size(),
|
self.file.offset_size(),
|
||||||
self.file.length_size(),
|
self.file.length_size(),
|
||||||
)?;
|
)
|
||||||
Ok(attrs_to_map(
|
|
||||||
&attr_msgs,
|
|
||||||
data,
|
|
||||||
self.file.offset_size(),
|
|
||||||
self.file.length_size(),
|
|
||||||
))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Verify this dataset's content against its stored provenance hash
|
/// Verify this dataset's content against its stored provenance hash
|
||||||
@@ -803,7 +856,22 @@ impl<'f> Dataset<'f> {
|
|||||||
|
|
||||||
fn dataspace(&self) -> Result<Dataspace, Error> {
|
fn dataspace(&self) -> Result<Dataspace, Error> {
|
||||||
let data = self.required_payload(MessageType::Dataspace)?;
|
let data = self.required_payload(MessageType::Dataspace)?;
|
||||||
Ok(Dataspace::parse(&data, self.file.length_size())?)
|
let mut ds = Dataspace::parse(&data, self.file.length_size())?;
|
||||||
|
// libhdf5 reports a virtual dataset with unlimited or printf-style
|
||||||
|
// mappings at the extent its sources currently fill, not the stored
|
||||||
|
// one (`H5Dget_space`).
|
||||||
|
if let Ok(dl @ DataLayout::Virtual { .. }) = self.data_layout() {
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
ds.dimensions = clawhdf5_format::vds::virtual_dataset_extent(
|
||||||
|
self.file.data.as_bytes(),
|
||||||
|
&dl,
|
||||||
|
&ds,
|
||||||
|
self.file.offset_size(),
|
||||||
|
self.file.length_size(),
|
||||||
|
Some(&resolver),
|
||||||
|
)?;
|
||||||
|
}
|
||||||
|
Ok(ds)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn data_layout(&self) -> Result<DataLayout, Error> {
|
fn data_layout(&self) -> Result<DataLayout, Error> {
|
||||||
@@ -832,24 +900,9 @@ impl<'f> Dataset<'f> {
|
|||||||
let pipeline = self.filter_pipeline()?;
|
let pipeline = self.filter_pipeline()?;
|
||||||
|
|
||||||
// Virtual datasets are assembled from source datasets; the per-file
|
// Virtual datasets are assembled from source datasets; the per-file
|
||||||
// chunk cache does not apply. Route them through the resolver path so
|
// chunk cache does not apply.
|
||||||
// external sibling files resolve relative to this file's directory.
|
|
||||||
if matches!(dl, DataLayout::Virtual { .. }) {
|
if matches!(dl, DataLayout::Virtual { .. }) {
|
||||||
let base_dir = self.file.base_dir.clone();
|
return self.read_virtual(&dl, &ds, &dt);
|
||||||
let resolver = move |name: &str| -> Option<Vec<u8>> {
|
|
||||||
let dir = base_dir.as_ref()?;
|
|
||||||
std::fs::read(dir.join(sibling_file_name(name)?)).ok()
|
|
||||||
};
|
|
||||||
return Ok(data_read::read_raw_data_full_with_resolver(
|
|
||||||
self.file.data.as_bytes(),
|
|
||||||
&dl,
|
|
||||||
&ds,
|
|
||||||
&dt,
|
|
||||||
pipeline.as_ref(),
|
|
||||||
self.file.offset_size(),
|
|
||||||
self.file.length_size(),
|
|
||||||
Some(&resolver),
|
|
||||||
)?);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Unallocated storage reads as the dataset's fill value.
|
// Unallocated storage reads as the dataset's fill value.
|
||||||
@@ -875,6 +928,62 @@ impl<'f> Dataset<'f> {
|
|||||||
},
|
},
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Resolver for external Virtual Dataset source files: names are
|
||||||
|
/// resolved against the directory of the file that holds the virtual
|
||||||
|
/// dataset, as libhdf5 does. A missing file is `Ok(None)` (its mappings
|
||||||
|
/// read as the fill value); a name that would leave that directory is
|
||||||
|
/// refused with an error rather than read as fill.
|
||||||
|
fn vds_resolver(&self) -> impl Fn(&str) -> Result<Option<Vec<u8>>, FormatError> + use<> {
|
||||||
|
let base_dir = self.file.base_dir.clone();
|
||||||
|
move |name: &str| {
|
||||||
|
let Some(dir) = base_dir.as_ref() else {
|
||||||
|
return Err(FormatError::ChunkedReadError(format!(
|
||||||
|
"virtual dataset source file {name:?} cannot be resolved for an in-memory file"
|
||||||
|
)));
|
||||||
|
};
|
||||||
|
let rel = sibling_file_name(name).ok_or_else(|| {
|
||||||
|
FormatError::ChunkedReadError(format!(
|
||||||
|
"virtual dataset source file {name:?} is outside the virtual file's \
|
||||||
|
directory and is not followed"
|
||||||
|
))
|
||||||
|
})?;
|
||||||
|
match std::fs::read(dir.join(rel)) {
|
||||||
|
Ok(bytes) => Ok(Some(bytes)),
|
||||||
|
Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(None),
|
||||||
|
Err(e) => Err(FormatError::ChunkedReadError(format!(
|
||||||
|
"cannot read virtual dataset source file {name:?}: {e}"
|
||||||
|
))),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read a whole virtual dataset; unmapped elements hold its fill value.
|
||||||
|
fn read_virtual(
|
||||||
|
&self,
|
||||||
|
dl: &DataLayout,
|
||||||
|
ds: &Dataspace,
|
||||||
|
dt: &Datatype,
|
||||||
|
) -> Result<Vec<u8>, Error> {
|
||||||
|
let fill = clawhdf5_format::fill_value::dataset_fill_value_in(
|
||||||
|
self.file.data.as_bytes(),
|
||||||
|
&self.header.messages,
|
||||||
|
self.file.offset_size(),
|
||||||
|
self.file.length_size(),
|
||||||
|
)?;
|
||||||
|
let resolver = self.vds_resolver();
|
||||||
|
let v = clawhdf5_format::vds::read_virtual_dataset(
|
||||||
|
self.file.data.as_bytes(),
|
||||||
|
dl,
|
||||||
|
ds,
|
||||||
|
dt,
|
||||||
|
fill.as_deref(),
|
||||||
|
self.file.offset_size(),
|
||||||
|
self.file.length_size(),
|
||||||
|
Some(&resolver),
|
||||||
|
)?;
|
||||||
|
Ok(v.data)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
// ---------------------------------------------------------------------------
|
||||||
@@ -954,37 +1063,6 @@ fn is_group(header: &ObjectHeader) -> bool {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
fn resolve_group_entries(
|
|
||||||
file_data: &[u8],
|
|
||||||
object_header: &ObjectHeader,
|
|
||||||
offset_size: u8,
|
|
||||||
length_size: u8,
|
|
||||||
) -> Result<Vec<GroupEntry>, FormatError> {
|
|
||||||
let is_v1 = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.any(|m| m.msg_type == MessageType::SymbolTable);
|
|
||||||
let is_v2 = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.any(|m| m.msg_type == MessageType::LinkInfo || m.msg_type == MessageType::Link);
|
|
||||||
|
|
||||||
if is_v1 {
|
|
||||||
let sym_msg = object_header
|
|
||||||
.messages
|
|
||||||
.iter()
|
|
||||||
.find(|m| m.msg_type == MessageType::SymbolTable)
|
|
||||||
.ok_or_else(|| FormatError::PathNotFound("no symbol table message".into()))?;
|
|
||||||
let stm = SymbolTableMessage::parse(&sym_msg.data, offset_size)?;
|
|
||||||
group_v1::resolve_v1_group_entries(file_data, &stm, offset_size, length_size)
|
|
||||||
} else if is_v2 {
|
|
||||||
group_v2::resolve_v2_group_entries(file_data, object_header, offset_size, length_size)
|
|
||||||
} else {
|
|
||||||
// Empty group or unrecognized — return empty
|
|
||||||
Ok(Vec::new())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod sibling_file_name_tests {
|
mod sibling_file_name_tests {
|
||||||
use super::sibling_file_name;
|
use super::sibling_file_name;
|
||||||
|
|||||||
@@ -155,6 +155,33 @@ pub(crate) fn classify_datatype(dt: &clawhdf5_format::datatype::Datatype) -> DTy
|
|||||||
/// Read attribute messages into a `HashMap<String, AttrValue>`.
|
/// Read attribute messages into a `HashMap<String, AttrValue>`.
|
||||||
///
|
///
|
||||||
/// Best-effort: attributes that can't be decoded are silently skipped.
|
/// Best-effort: attributes that can't be decoded are silently skipped.
|
||||||
|
/// The attributes of the object with header `header` that could be read,
|
||||||
|
/// and one error for each that could not (see
|
||||||
|
/// [`extract_attributes_tolerant`](clawhdf5_format::attribute::extract_attributes_tolerant)).
|
||||||
|
pub(crate) fn read_attrs(
|
||||||
|
file_data: &[u8],
|
||||||
|
header: &clawhdf5_format::object_header::ObjectHeader,
|
||||||
|
offset_size: u8,
|
||||||
|
length_size: u8,
|
||||||
|
) -> Result<
|
||||||
|
(
|
||||||
|
HashMap<String, AttrValue>,
|
||||||
|
Vec<clawhdf5_format::error::FormatError>,
|
||||||
|
),
|
||||||
|
crate::Error,
|
||||||
|
> {
|
||||||
|
let (msgs, errors) = clawhdf5_format::attribute::extract_attributes_tolerant(
|
||||||
|
file_data,
|
||||||
|
header,
|
||||||
|
offset_size,
|
||||||
|
length_size,
|
||||||
|
)?;
|
||||||
|
Ok((
|
||||||
|
attrs_to_map(&msgs, file_data, offset_size, length_size),
|
||||||
|
errors,
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) fn attrs_to_map(
|
pub(crate) fn attrs_to_map(
|
||||||
attrs: &[clawhdf5_format::attribute::AttributeMessage],
|
attrs: &[clawhdf5_format::attribute::AttributeMessage],
|
||||||
file_data: &[u8],
|
file_data: &[u8],
|
||||||
|
|||||||
@@ -0,0 +1,593 @@
|
|||||||
|
//! Fixed Array / Extensible Array chunk-index interop with libhdf5 (via h5py).
|
||||||
|
//!
|
||||||
|
//! Both indexes place each chunk at a linear index computed from the
|
||||||
|
//! dataset's *maximum* dimensions, and the Extensible Array additionally
|
||||||
|
//! moves its unlimited dimension to the slowest-varying position. Getting
|
||||||
|
//! either wrong reads (or writes) every chunk after the first row in the
|
||||||
|
//! wrong place, silently, so these tests compare every value.
|
||||||
|
//!
|
||||||
|
//! Skipped when python3 with h5py is unavailable, unless
|
||||||
|
//! `CLAWHDF5_REQUIRE_INTEROP=1`.
|
||||||
|
|
||||||
|
use std::process::Command;
|
||||||
|
|
||||||
|
use clawhdf5::{File, FileBuilder};
|
||||||
|
|
||||||
|
fn python() -> String {
|
||||||
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn interop_required() -> bool {
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn python_available() -> bool {
|
||||||
|
Command::new(python())
|
||||||
|
.args(["-c", "import h5py"])
|
||||||
|
.output()
|
||||||
|
.map(|o| o.status.success())
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
macro_rules! skip_if_no_python {
|
||||||
|
() => {
|
||||||
|
if !python_available() {
|
||||||
|
assert!(
|
||||||
|
!interop_required(),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||||
|
);
|
||||||
|
eprintln!("SKIP: python3 with h5py not available");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run_python(script: &str) -> String {
|
||||||
|
let output = Command::new(python())
|
||||||
|
.args(["-c", script])
|
||||||
|
.output()
|
||||||
|
.expect("failed to run python");
|
||||||
|
if !output.status.success() {
|
||||||
|
panic!(
|
||||||
|
"Python script failed:\nSTDOUT: {}\nSTDERR: {}",
|
||||||
|
String::from_utf8_lossy(&output.stdout),
|
||||||
|
String::from_utf8_lossy(&output.stderr)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
String::from_utf8_lossy(&output.stdout).trim().to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Row-major `arange` of `shape`, cropped to `crop` (the current extent).
|
||||||
|
fn arange_cropped(full: &[usize], crop: &[usize]) -> Vec<i32> {
|
||||||
|
let n: usize = crop.iter().product();
|
||||||
|
let mut out = Vec::with_capacity(n);
|
||||||
|
for flat in 0..n {
|
||||||
|
let mut rem = flat;
|
||||||
|
let mut src = 0usize;
|
||||||
|
let mut stride = 1usize;
|
||||||
|
let mut coords = vec![0usize; crop.len()];
|
||||||
|
for d in (0..crop.len()).rev() {
|
||||||
|
coords[d] = rem % crop[d];
|
||||||
|
rem /= crop[d];
|
||||||
|
}
|
||||||
|
for d in (0..full.len()).rev() {
|
||||||
|
src += coords[d] * stride;
|
||||||
|
stride *= full[d];
|
||||||
|
}
|
||||||
|
out.push(src as i32);
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One `i4` dataset, filled with `arange` over `full` and then resized to
|
||||||
|
/// `shape` (equal to `full` unless the case shrinks it).
|
||||||
|
struct Case {
|
||||||
|
name: &'static str,
|
||||||
|
full: Vec<usize>,
|
||||||
|
shape: Vec<usize>,
|
||||||
|
chunks: Vec<usize>,
|
||||||
|
maxshape: &'static str,
|
||||||
|
extra: &'static str,
|
||||||
|
index: &'static str,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn py_tuple(v: &[usize]) -> String {
|
||||||
|
let parts: Vec<String> = v.iter().map(|x| x.to_string()).collect();
|
||||||
|
format!("({},)", parts.join(","))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Have h5py (`libver="latest"`, so Fixed/Extensible Array indexes) write
|
||||||
|
/// every case to one file, then read each back and compare every value.
|
||||||
|
fn check_h5py_written(cases: &[Case]) {
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("h5py_chunk_index.h5");
|
||||||
|
let path_str = path.display().to_string();
|
||||||
|
|
||||||
|
let mut script =
|
||||||
|
format!("import h5py, numpy as np\nf = h5py.File(r'{path_str}', 'w', libver='latest')\n");
|
||||||
|
for c in cases {
|
||||||
|
script += &format!(
|
||||||
|
"d = f.create_dataset('{name}', data=np.arange({n}, dtype='i4').reshape({full}), \
|
||||||
|
chunks={chunks}, maxshape={maxshape}{extra})\n\
|
||||||
|
d.resize({shape})\n",
|
||||||
|
name = c.name,
|
||||||
|
n = c.full.iter().product::<usize>(),
|
||||||
|
full = py_tuple(&c.full),
|
||||||
|
chunks = py_tuple(&c.chunks),
|
||||||
|
maxshape = c.maxshape,
|
||||||
|
extra = c.extra,
|
||||||
|
shape = py_tuple(&c.shape),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
script += "f.close()\n";
|
||||||
|
run_python(&script);
|
||||||
|
|
||||||
|
let file = File::open(&path).unwrap();
|
||||||
|
for c in cases {
|
||||||
|
let ds = file.dataset(c.name).unwrap();
|
||||||
|
let shape: Vec<usize> = ds.shape().unwrap().iter().map(|&d| d as usize).collect();
|
||||||
|
assert_eq!(shape, c.shape, "{}: shape", c.name);
|
||||||
|
let got = ds.read_i32().unwrap();
|
||||||
|
let want = arange_cropped(&c.full, &c.shape);
|
||||||
|
let bad = got.iter().zip(&want).filter(|(a, b)| a != b).count();
|
||||||
|
assert_eq!(
|
||||||
|
got,
|
||||||
|
want,
|
||||||
|
"{}: {bad} of {} values differ (index {})",
|
||||||
|
c.name,
|
||||||
|
want.len(),
|
||||||
|
c.index
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// h5py-written Extensible Array whose unlimited dimension is not the first,
|
||||||
|
/// with the current shape smaller than the finite maximum: the library
|
||||||
|
/// swizzles the unlimited dimension to the slowest position and strides the
|
||||||
|
/// rest by their maximum chunk counts.
|
||||||
|
#[test]
|
||||||
|
fn h5py_extensible_array_partial_extent_reads_correctly() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
check_h5py_written(&[
|
||||||
|
// The `ea_fa_partial.h5` repro from the conformance sweep.
|
||||||
|
Case {
|
||||||
|
name: "ea_10_none",
|
||||||
|
full: vec![4, 6],
|
||||||
|
shape: vec![4, 6],
|
||||||
|
chunks: vec![2, 3],
|
||||||
|
maxshape: "(10, None)",
|
||||||
|
extra: "",
|
||||||
|
index: "EA, unlimited dim 1",
|
||||||
|
},
|
||||||
|
Case {
|
||||||
|
name: "ea_none_10",
|
||||||
|
full: vec![4, 6],
|
||||||
|
shape: vec![4, 6],
|
||||||
|
chunks: vec![2, 3],
|
||||||
|
maxshape: "(None, 10)",
|
||||||
|
extra: "",
|
||||||
|
index: "EA, unlimited dim 0",
|
||||||
|
},
|
||||||
|
Case {
|
||||||
|
name: "ea_3d_mid",
|
||||||
|
full: vec![3, 4, 5],
|
||||||
|
shape: vec![3, 4, 5],
|
||||||
|
chunks: vec![2, 3, 2],
|
||||||
|
maxshape: "(5, None, 7)",
|
||||||
|
extra: "",
|
||||||
|
index: "EA, unlimited dim 1 of 3",
|
||||||
|
},
|
||||||
|
Case {
|
||||||
|
name: "ea_3d_last_gzip",
|
||||||
|
full: vec![3, 4, 5],
|
||||||
|
shape: vec![3, 4, 5],
|
||||||
|
chunks: vec![2, 3, 2],
|
||||||
|
maxshape: "(5, 9, None)",
|
||||||
|
extra: ", compression='gzip'",
|
||||||
|
index: "EA, unlimited dim 2 of 3, filtered",
|
||||||
|
},
|
||||||
|
// Many chunks: crosses data blocks, super blocks and paging.
|
||||||
|
Case {
|
||||||
|
name: "ea_many",
|
||||||
|
full: vec![3, 1500],
|
||||||
|
shape: vec![3, 1500],
|
||||||
|
chunks: vec![1, 1],
|
||||||
|
maxshape: "(4, None)",
|
||||||
|
extra: "",
|
||||||
|
index: "EA, 4500 slots",
|
||||||
|
},
|
||||||
|
// Shrunk after writing: chunks beyond the extent must be ignored.
|
||||||
|
Case {
|
||||||
|
name: "ea_shrunk",
|
||||||
|
full: vec![8, 9],
|
||||||
|
shape: vec![3, 4],
|
||||||
|
chunks: vec![2, 3],
|
||||||
|
maxshape: "(10, None)",
|
||||||
|
extra: "",
|
||||||
|
index: "EA, shrunk",
|
||||||
|
},
|
||||||
|
]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// h5py-written Fixed Array with the current shape smaller than a finite
|
||||||
|
/// maxshape: the index has one slot per chunk of the *maximum* extent.
|
||||||
|
#[test]
|
||||||
|
fn h5py_fixed_array_partial_extent_reads_correctly() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
check_h5py_written(&[
|
||||||
|
Case {
|
||||||
|
name: "fa_20_10",
|
||||||
|
full: vec![4, 6],
|
||||||
|
shape: vec![4, 6],
|
||||||
|
chunks: vec![2, 3],
|
||||||
|
maxshape: "(20, 10)",
|
||||||
|
extra: "",
|
||||||
|
index: "FA",
|
||||||
|
},
|
||||||
|
Case {
|
||||||
|
name: "fa_3d_gzip",
|
||||||
|
full: vec![3, 4, 5],
|
||||||
|
shape: vec![3, 4, 5],
|
||||||
|
chunks: vec![2, 3, 2],
|
||||||
|
maxshape: "(6, 8, 10)",
|
||||||
|
extra: ", compression='gzip'",
|
||||||
|
index: "FA, filtered",
|
||||||
|
},
|
||||||
|
// Paged (> 1024 slots) with most of them beyond the extent.
|
||||||
|
Case {
|
||||||
|
name: "fa_paged",
|
||||||
|
full: vec![30, 50],
|
||||||
|
shape: vec![30, 50],
|
||||||
|
chunks: vec![1, 1],
|
||||||
|
maxshape: "(40, 60)",
|
||||||
|
extra: "",
|
||||||
|
index: "FA, 2400 slots, paged",
|
||||||
|
},
|
||||||
|
Case {
|
||||||
|
name: "fa_shrunk",
|
||||||
|
full: vec![8, 9],
|
||||||
|
shape: vec![5, 2],
|
||||||
|
chunks: vec![2, 3],
|
||||||
|
maxshape: "(20, 10)",
|
||||||
|
extra: "",
|
||||||
|
index: "FA, shrunk",
|
||||||
|
},
|
||||||
|
]);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ===========================================================================
|
||||||
|
// Files we write, read back by libhdf5 (h5py and h5dump) and by us
|
||||||
|
// ===========================================================================
|
||||||
|
|
||||||
|
/// One `i4` dataset we write, filled with `arange` over `shape`.
|
||||||
|
struct WriteCase {
|
||||||
|
name: String,
|
||||||
|
shape: Vec<u64>,
|
||||||
|
chunks: Vec<u64>,
|
||||||
|
maxshape: Option<Vec<u64>>,
|
||||||
|
deflate: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
fn wcase(name: &str, shape: &[u64], chunks: &[u64], maxshape: Option<&[u64]>) -> WriteCase {
|
||||||
|
WriteCase {
|
||||||
|
name: name.to_string(),
|
||||||
|
shape: shape.to_vec(),
|
||||||
|
chunks: chunks.to_vec(),
|
||||||
|
maxshape: maxshape.map(<[u64]>::to_vec),
|
||||||
|
deflate: false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn h5dump_available() -> bool {
|
||||||
|
Command::new("h5dump")
|
||||||
|
.arg("--version")
|
||||||
|
.output()
|
||||||
|
.map(|o| o.status.success())
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write every case into one file with our writer, then check that our own
|
||||||
|
/// reader, h5py and h5dump (when installed) all return every value. Only the
|
||||||
|
/// libhdf5 half is skipped without h5py.
|
||||||
|
fn check_we_write(cases: &[WriteCase]) {
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("ours_chunk_index.h5");
|
||||||
|
let path_str = path.display().to_string();
|
||||||
|
|
||||||
|
let mut b = FileBuilder::new();
|
||||||
|
for c in cases {
|
||||||
|
let n: u64 = c.shape.iter().product();
|
||||||
|
let data: Vec<i32> = (0..n as i32).collect();
|
||||||
|
let ds = b.create_dataset(&c.name);
|
||||||
|
ds.with_i32_data(&data)
|
||||||
|
.with_shape(&c.shape)
|
||||||
|
.with_chunks(&c.chunks);
|
||||||
|
if let Some(ms) = &c.maxshape {
|
||||||
|
ds.with_maxshape(ms);
|
||||||
|
}
|
||||||
|
if c.deflate {
|
||||||
|
ds.with_deflate(4);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
b.write(&path).unwrap();
|
||||||
|
|
||||||
|
// Our reader.
|
||||||
|
let file = File::open(&path).unwrap();
|
||||||
|
for c in cases {
|
||||||
|
let got = file.dataset(&c.name).unwrap().read_i32().unwrap();
|
||||||
|
let n: u64 = c.shape.iter().product();
|
||||||
|
let bad = got
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.filter(|&(i, &v)| v != i as i32)
|
||||||
|
.count();
|
||||||
|
assert!(
|
||||||
|
got.len() == n as usize && bad == 0,
|
||||||
|
"{}: our reader: {bad} of {n} values wrong",
|
||||||
|
c.name
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// libhdf5 via h5py.
|
||||||
|
skip_if_no_python!();
|
||||||
|
let mut script =
|
||||||
|
format!("import h5py, numpy as np\nbad = []\nf = h5py.File(r'{path_str}', 'r')\n");
|
||||||
|
for c in cases {
|
||||||
|
let shape: Vec<String> = c.shape.iter().map(u64::to_string).collect();
|
||||||
|
let maxshape: Vec<String> = c
|
||||||
|
.maxshape
|
||||||
|
.as_ref()
|
||||||
|
.unwrap_or(&c.shape)
|
||||||
|
.iter()
|
||||||
|
.map(|&d| {
|
||||||
|
if d == u64::MAX {
|
||||||
|
"None".to_string()
|
||||||
|
} else {
|
||||||
|
d.to_string()
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
script += &format!(
|
||||||
|
"d = f['{name}']\n\
|
||||||
|
want = np.arange({n}, dtype='i4').reshape(({shape},))\n\
|
||||||
|
got = d[()]\n\
|
||||||
|
if d.maxshape != ({maxshape},): bad.append(('{name}', 'maxshape', d.maxshape))\n\
|
||||||
|
elif not np.array_equal(got, want): \
|
||||||
|
bad.append(('{name}', int((got != want).sum()), 'of', got.size))\n",
|
||||||
|
name = c.name,
|
||||||
|
n = c.shape.iter().product::<u64>(),
|
||||||
|
shape = shape.join(","),
|
||||||
|
maxshape = maxshape.join(","),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
script += "print(bad if bad else 'OK')\n";
|
||||||
|
let out = run_python(&script);
|
||||||
|
assert_eq!(out, "OK", "h5py disagrees");
|
||||||
|
|
||||||
|
// libhdf5's own tool, when installed.
|
||||||
|
if h5dump_available() {
|
||||||
|
let o = Command::new("h5dump").arg(&path).output().unwrap();
|
||||||
|
let stderr = String::from_utf8_lossy(&o.stderr);
|
||||||
|
assert!(
|
||||||
|
o.status.success() && !stderr.to_lowercase().contains("error"),
|
||||||
|
"h5dump failed: {stderr}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Let libhdf5 grow every resizable dataset by two chunks per dimension
|
||||||
|
// (capped at the maxshape) and rewrite it, which updates our index in
|
||||||
|
// place and inserts new chunks into it. Then both readers must agree.
|
||||||
|
let script = format!(
|
||||||
|
r#"
|
||||||
|
import h5py, numpy as np
|
||||||
|
grown = {{}}
|
||||||
|
with h5py.File(r'{path_str}', 'r+') as f:
|
||||||
|
for name in f:
|
||||||
|
d = f[name]
|
||||||
|
if d.chunks is None:
|
||||||
|
continue
|
||||||
|
new = tuple(s + 2 * c if m is None else min(m, s + 2 * c)
|
||||||
|
for s, m, c in zip(d.shape, d.maxshape, d.chunks))
|
||||||
|
if new == d.shape:
|
||||||
|
continue
|
||||||
|
old = d[()]
|
||||||
|
full = np.full(new, -7, 'i4')
|
||||||
|
full[tuple(slice(0, s) for s in old.shape)] = old
|
||||||
|
d.resize(new)
|
||||||
|
d[...] = full
|
||||||
|
grown[name] = (list(old.shape), list(new))
|
||||||
|
with h5py.File(r'{path_str}', 'r') as f:
|
||||||
|
for name, (old, new) in grown.items():
|
||||||
|
want = np.full(new, -7, 'i4')
|
||||||
|
want[tuple(slice(0, s) for s in old)] = np.arange(int(np.prod(old)), dtype='i4').reshape(old)
|
||||||
|
assert np.array_equal(f[name][()], want), name
|
||||||
|
for name, (old, new) in grown.items():
|
||||||
|
print(name, ','.join(map(str, old)), ','.join(map(str, new)))
|
||||||
|
"#
|
||||||
|
);
|
||||||
|
let out = run_python(&script);
|
||||||
|
let growable = cases
|
||||||
|
.iter()
|
||||||
|
.filter(|c| c.maxshape.as_ref().is_some_and(|m| *m != c.shape))
|
||||||
|
.count();
|
||||||
|
assert_eq!(out.lines().count(), growable, "libhdf5 grew: {out}");
|
||||||
|
let dims = |s: &str| -> Vec<usize> { s.split(',').map(|x| x.parse().unwrap()).collect() };
|
||||||
|
let file = File::open(&path).unwrap();
|
||||||
|
for line in out.lines() {
|
||||||
|
let mut parts = line.split(' ');
|
||||||
|
let (name, old, new) = (
|
||||||
|
parts.next().unwrap(),
|
||||||
|
dims(parts.next().unwrap()),
|
||||||
|
dims(parts.next().unwrap()),
|
||||||
|
);
|
||||||
|
let got = file.dataset(name).unwrap().read_i32().unwrap();
|
||||||
|
let n: usize = new.iter().product();
|
||||||
|
let mut want = vec![-7i32; n];
|
||||||
|
for (flat, w) in want.iter_mut().enumerate() {
|
||||||
|
let mut rem = flat;
|
||||||
|
let mut coords = vec![0usize; new.len()];
|
||||||
|
for d in (0..new.len()).rev() {
|
||||||
|
coords[d] = rem % new[d];
|
||||||
|
rem /= new[d];
|
||||||
|
}
|
||||||
|
if coords.iter().zip(&old).all(|(c, o)| c < o) {
|
||||||
|
*w = coords.iter().zip(&old).fold(0, |acc, (c, o)| acc * o + c) as i32;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let bad = got.iter().zip(&want).filter(|(a, b)| a != b).count();
|
||||||
|
assert!(
|
||||||
|
got.len() == n && bad == 0,
|
||||||
|
"{name}: after libhdf5 grew it, our reader got {bad} of {n} values wrong"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A Fixed Array with more than 1024 elements must be paged, or libhdf5
|
||||||
|
/// rejects the data block's checksum.
|
||||||
|
#[test]
|
||||||
|
fn we_write_paged_fixed_array() {
|
||||||
|
let mut cases: Vec<WriteCase> = [1023u64, 1024, 1025, 2048, 5000]
|
||||||
|
.iter()
|
||||||
|
.map(|&n| wcase(&format!("fa_{n}"), &[n * 4], &[4], None))
|
||||||
|
.collect();
|
||||||
|
// Filtered elements are wider; a 2-D grid pages the same way.
|
||||||
|
let mut filtered = wcase("fa_1500_deflate", &[1500 * 4], &[4], None);
|
||||||
|
filtered.deflate = true;
|
||||||
|
cases.push(filtered);
|
||||||
|
cases.push(wcase("fa_2d_1100", &[110, 40], &[1, 4], None));
|
||||||
|
check_we_write(&cases);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An Extensible Array holds 4 elements in its index block and 240 in the
|
||||||
|
/// data blocks the index block addresses; everything after that lives under
|
||||||
|
/// super blocks, and from ~131K elements on in paged data blocks. Chunks past
|
||||||
|
/// index 243 used to be written but never indexed (read back as fill by us
|
||||||
|
/// and by libhdf5).
|
||||||
|
#[test]
|
||||||
|
fn we_write_extensible_array_past_index_block() {
|
||||||
|
let unl: &[u64] = &[u64::MAX];
|
||||||
|
let mut cases: Vec<WriteCase> = [1u64, 4, 5, 243, 244, 245, 300, 1000, 5000]
|
||||||
|
.iter()
|
||||||
|
.map(|&n| wcase(&format!("ea_{n}"), &[n * 4], &[4], Some(unl)))
|
||||||
|
.collect();
|
||||||
|
let mut filtered = wcase("ea_300_deflate", &[300 * 4], &[4], Some(unl));
|
||||||
|
filtered.deflate = true;
|
||||||
|
cases.push(filtered);
|
||||||
|
// Several super blocks and paged data blocks (level 13, the first with
|
||||||
|
// data blocks over 1024 elements, starts at element 4 + 131056).
|
||||||
|
cases.push(wcase("ea_140000", &[140_000], &[1], Some(unl)));
|
||||||
|
check_we_write(&cases);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A maxshape larger than the shape: the index must be laid out over the
|
||||||
|
/// chunks of the maximum extent (libhdf5 read our Fixed Array past its end:
|
||||||
|
/// "addr overflow"), and an Extensible Array whose unlimited dimension is not
|
||||||
|
/// the first must swizzle it to the slowest position (libhdf5 read our
|
||||||
|
/// `(20, None)` dataset scrambled).
|
||||||
|
#[test]
|
||||||
|
fn we_write_maxshape_larger_than_shape() {
|
||||||
|
const U: u64 = u64::MAX;
|
||||||
|
let mut cases = vec![
|
||||||
|
// Fixed Array over the maximum extent.
|
||||||
|
wcase("fa2d_finite_max", &[20, 30], &[5, 5], Some(&[40, 60])),
|
||||||
|
wcase("fa1d_finite_max", &[40], &[4], Some(&[100])),
|
||||||
|
wcase("fa3d_edges", &[6, 7, 8], &[4, 3, 5], Some(&[10, 9, 20])),
|
||||||
|
wcase("fa_paged_max", &[30, 50], &[1, 1], Some(&[40, 60])),
|
||||||
|
wcase("fa_one_chunk_now", &[5], &[5], Some(&[50])),
|
||||||
|
// Extensible Array, unlimited dimension first (no swizzle) ...
|
||||||
|
wcase("ea2d_unl_fin", &[20, 30], &[5, 5], Some(&[U, 30])),
|
||||||
|
wcase("ea2d_unl_fin_max", &[20, 30], &[5, 5], Some(&[U, 60])),
|
||||||
|
// ... and not first (swizzled).
|
||||||
|
wcase("ea2d_fin_unl", &[20, 30], &[5, 5], Some(&[20, U])),
|
||||||
|
wcase("ea2d_fin_max_unl", &[20, 30], &[5, 5], Some(&[40, U])),
|
||||||
|
wcase("ea3d_mid", &[6, 7, 8], &[4, 3, 5], Some(&[10, U, 20])),
|
||||||
|
// Past the index block and into super blocks, swizzled.
|
||||||
|
wcase("ea2d_many", &[3, 2000], &[1, 1], Some(&[4, U])),
|
||||||
|
];
|
||||||
|
let mut filtered = wcase(
|
||||||
|
"ea3d_last_deflate",
|
||||||
|
&[6, 7, 8],
|
||||||
|
&[4, 3, 5],
|
||||||
|
Some(&[6, 8, U]),
|
||||||
|
);
|
||||||
|
filtered.deflate = true;
|
||||||
|
cases.push(filtered);
|
||||||
|
check_we_write(&cases);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// More than one unlimited dimension needs a version-2 B-tree chunk index,
|
||||||
|
/// as the library uses; an Extensible Array for `(None, None)` made libhdf5
|
||||||
|
/// refuse the whole file ("already found unlimited dimension").
|
||||||
|
#[test]
|
||||||
|
fn we_write_btree_v2_for_several_unlimited_dims() {
|
||||||
|
const U: u64 = u64::MAX;
|
||||||
|
let mut cases = vec![
|
||||||
|
wcase("unl_unl", &[20, 30], &[5, 5], Some(&[U, U])),
|
||||||
|
wcase("unl_fin_unl", &[6, 7, 8], &[4, 3, 5], Some(&[U, 9, U])),
|
||||||
|
// More records than the library's 2048-byte node holds (84 here).
|
||||||
|
wcase("unl_unl_2400", &[40, 60], &[1, 1], Some(&[U, U])),
|
||||||
|
wcase("unl_unl_empty", &[0, 0], &[4, 4], Some(&[U, U])),
|
||||||
|
];
|
||||||
|
let mut filtered = wcase("unl_unl_deflate", &[6, 7, 8], &[4, 3, 5], Some(&[U, U, U]));
|
||||||
|
filtered.deflate = true;
|
||||||
|
cases.push(filtered);
|
||||||
|
check_we_write(&cases);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A single-leaf B-tree has a 16-bit record count; beyond it the writer
|
||||||
|
/// refuses rather than writing a tree libhdf5 would misread.
|
||||||
|
#[test]
|
||||||
|
fn btree_v2_index_past_one_leaf_is_refused() {
|
||||||
|
let mut b = FileBuilder::new();
|
||||||
|
b.create_dataset("d")
|
||||||
|
.with_i32_data(&vec![0i32; 70_000])
|
||||||
|
.with_shape(&[70_000, 1])
|
||||||
|
.with_chunks(&[1, 1])
|
||||||
|
.with_maxshape(&[u64::MAX, u64::MAX]);
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
assert!(b.write(dir.path().join("too_many.h5")).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A maxshape equal to the shape cannot grow, so it needs no chunks: the
|
||||||
|
/// dataset stays contiguous (as h5py makes it) unless chunks are requested.
|
||||||
|
#[test]
|
||||||
|
fn maxshape_equal_to_shape_stays_contiguous() {
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("ms_eq.h5");
|
||||||
|
let data: Vec<i32> = (0..40).collect();
|
||||||
|
let mut b = FileBuilder::new();
|
||||||
|
b.create_dataset("plain")
|
||||||
|
.with_i32_data(&data)
|
||||||
|
.with_shape(&[40])
|
||||||
|
.with_maxshape(&[40]);
|
||||||
|
b.create_dataset("chunked")
|
||||||
|
.with_i32_data(&data)
|
||||||
|
.with_shape(&[40])
|
||||||
|
.with_maxshape(&[40])
|
||||||
|
.with_chunks(&[8]);
|
||||||
|
b.write(&path).unwrap();
|
||||||
|
|
||||||
|
let file = File::open(&path).unwrap();
|
||||||
|
let plain = file.dataset("plain").unwrap();
|
||||||
|
assert_eq!(plain.read_i32().unwrap(), data);
|
||||||
|
assert_eq!(plain.max_dimensions().unwrap(), Some(vec![40]));
|
||||||
|
assert!(
|
||||||
|
plain.read_raw_ref().unwrap().is_some(),
|
||||||
|
"maxshape == shape should be contiguous"
|
||||||
|
);
|
||||||
|
let chunked = file.dataset("chunked").unwrap();
|
||||||
|
assert_eq!(chunked.read_i32().unwrap(), data);
|
||||||
|
assert!(chunked.read_raw_ref().unwrap().is_none());
|
||||||
|
|
||||||
|
skip_if_no_python!();
|
||||||
|
let out = run_python(&format!(
|
||||||
|
"import h5py, numpy as np\n\
|
||||||
|
f = h5py.File(r'{}', 'r')\n\
|
||||||
|
for n in ('plain', 'chunked'):\n\
|
||||||
|
\x20 d = f[n]\n\
|
||||||
|
\x20 assert np.array_equal(d[()], np.arange(40, dtype='i4')), n\n\
|
||||||
|
\x20 print(n, d.chunks, d.maxshape)\n",
|
||||||
|
path.display()
|
||||||
|
));
|
||||||
|
assert_eq!(out, "plain None (40,)\nchunked (8,) (40,)");
|
||||||
|
}
|
||||||
@@ -0,0 +1,87 @@
|
|||||||
|
//! A `File` is `Send + Sync` and keeps one chunk cache for all its datasets.
|
||||||
|
//! Threads reading different chunked datasets through the same `File` must
|
||||||
|
//! each get their own dataset's data.
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use clawhdf5::{File, FileBuilder};
|
||||||
|
|
||||||
|
const DATASETS: usize = 24;
|
||||||
|
const THREADS: usize = 16;
|
||||||
|
const ROUNDS: usize = 40;
|
||||||
|
|
||||||
|
/// Contents of dataset `k`: distinct from every other dataset's, element for
|
||||||
|
/// element, so any chunk served from the wrong dataset shows.
|
||||||
|
fn values(k: usize, n: usize) -> Vec<f64> {
|
||||||
|
(0..n).map(|i| (k * 100_000 + i) as f64).collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn build() -> File {
|
||||||
|
let mut b = FileBuilder::new();
|
||||||
|
for k in 0..DATASETS {
|
||||||
|
let ds = b.create_dataset(&format!("d{k:02}"));
|
||||||
|
match k % 3 {
|
||||||
|
// 1-D, compressed: chunk offsets 0, 8, 16, ... in every dataset.
|
||||||
|
0 => {
|
||||||
|
ds.with_f64_data(&values(k, 64)).with_shape(&[64]);
|
||||||
|
ds.with_chunks(&[8]).with_deflate(1);
|
||||||
|
}
|
||||||
|
// 1-D, shuffle + compressed, a different length.
|
||||||
|
1 => {
|
||||||
|
ds.with_f64_data(&values(k, 40)).with_shape(&[40]);
|
||||||
|
ds.with_chunks(&[8]).with_shuffle().with_deflate(1);
|
||||||
|
}
|
||||||
|
// 2-D, compressed: coordinates (0,0), (0,4), (4,0), ... overlap
|
||||||
|
// the other datasets' in the first dimension.
|
||||||
|
_ => {
|
||||||
|
ds.with_f64_data(&values(k, 64)).with_shape(&[8, 8]);
|
||||||
|
ds.with_chunks(&[4, 4]).with_deflate(1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
File::from_bytes(b.finish().unwrap()).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn expected(k: usize) -> Vec<f64> {
|
||||||
|
values(k, if k % 3 == 1 { 40 } else { 64 })
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn threads_reading_different_datasets_get_their_own_chunks() {
|
||||||
|
let file = Arc::new(build());
|
||||||
|
// Sequential sanity check first.
|
||||||
|
for k in 0..DATASETS {
|
||||||
|
let got = file.dataset(&format!("d{k:02}")).unwrap().read_f64();
|
||||||
|
assert_eq!(got.unwrap(), expected(k), "sequential d{k:02}");
|
||||||
|
}
|
||||||
|
|
||||||
|
let handles: Vec<_> = (0..THREADS)
|
||||||
|
.map(|t| {
|
||||||
|
let file = Arc::clone(&file);
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
let mut wrong = Vec::new();
|
||||||
|
for round in 0..ROUNDS {
|
||||||
|
let k = (t * 7 + round * 5) % DATASETS;
|
||||||
|
let name = format!("d{k:02}");
|
||||||
|
match file.dataset(&name).unwrap().read_f64() {
|
||||||
|
Ok(v) if v == expected(k) => {}
|
||||||
|
Ok(v) => wrong.push(format!("{name}: wrong data, first {:?}", &v[..4])),
|
||||||
|
Err(e) => wrong.push(format!("{name}: {e}")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
wrong
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let failures: Vec<String> = handles
|
||||||
|
.into_iter()
|
||||||
|
.flat_map(|h| h.join().unwrap())
|
||||||
|
.collect();
|
||||||
|
assert!(
|
||||||
|
failures.is_empty(),
|
||||||
|
"{} of {} concurrent reads were wrong, e.g. {:?}",
|
||||||
|
failures.len(),
|
||||||
|
THREADS * ROUNDS,
|
||||||
|
&failures[..failures.len().min(5)]
|
||||||
|
);
|
||||||
|
}
|
||||||
@@ -0,0 +1,418 @@
|
|||||||
|
//! Dense ("new-style") link and attribute storage written by libhdf5 (via
|
||||||
|
//! h5py): groups whose links live in a fractal heap indexed by a v2 B-tree,
|
||||||
|
//! and objects whose attributes do. Every listing and value is compared with
|
||||||
|
//! what h5py itself reports for the same file.
|
||||||
|
//!
|
||||||
|
//! Skipped when python3 with h5py is unavailable, unless
|
||||||
|
//! `CLAWHDF5_REQUIRE_INTEROP=1`.
|
||||||
|
|
||||||
|
use std::process::Command;
|
||||||
|
|
||||||
|
use clawhdf5::{AttrValue, File};
|
||||||
|
|
||||||
|
fn python() -> String {
|
||||||
|
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn interop_required() -> bool {
|
||||||
|
std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn python_available() -> bool {
|
||||||
|
Command::new(python())
|
||||||
|
.args(["-c", "import h5py"])
|
||||||
|
.output()
|
||||||
|
.map(|o| o.status.success())
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
macro_rules! skip_if_no_python {
|
||||||
|
() => {
|
||||||
|
if !python_available() {
|
||||||
|
assert!(
|
||||||
|
!interop_required(),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available"
|
||||||
|
);
|
||||||
|
eprintln!("SKIP: python3 with h5py not available");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
fn run_python(script: &str) -> String {
|
||||||
|
let output = Command::new(python())
|
||||||
|
.args(["-c", script])
|
||||||
|
.output()
|
||||||
|
.expect("failed to run python");
|
||||||
|
if !output.status.success() {
|
||||||
|
panic!(
|
||||||
|
"Python script failed:\nSTDOUT: {}\nSTDERR: {}",
|
||||||
|
String::from_utf8_lossy(&output.stdout),
|
||||||
|
String::from_utf8_lossy(&output.stderr)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
String::from_utf8_lossy(&output.stdout).trim().to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Have h5py write a file with `body` (which sees `f`, `h5py` and `np`),
|
||||||
|
/// returning the temp dir holding it and its path.
|
||||||
|
fn h5py_file(body: &str) -> (tempfile::TempDir, String) {
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("dense.h5").display().to_string();
|
||||||
|
let script = format!(
|
||||||
|
"import h5py, numpy as np\n\
|
||||||
|
with h5py.File(r'{path}', 'w', libver='latest') as f:\n{}",
|
||||||
|
indent(body)
|
||||||
|
);
|
||||||
|
run_python(&script);
|
||||||
|
(dir, path)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn indent(body: &str) -> String {
|
||||||
|
body.lines()
|
||||||
|
.map(|l| format!(" {l}\n"))
|
||||||
|
.collect::<String>()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The datasets and groups h5py lists in `group`, sorted: links h5py can
|
||||||
|
/// resolve (hard and soft), without dangling soft links or external links.
|
||||||
|
fn h5py_listing(path: &str, group: &str) -> (Vec<String>, Vec<String>) {
|
||||||
|
let out = run_python(&format!(
|
||||||
|
"import h5py\n\
|
||||||
|
ds, gs = [], []\n\
|
||||||
|
with h5py.File(r'{path}', 'r') as f:\n\
|
||||||
|
\x20 g = f[{group:?}]\n\
|
||||||
|
\x20 for k in g.keys():\n\
|
||||||
|
\x20 if isinstance(g.get(k, getlink=True), h5py.ExternalLink):\n\
|
||||||
|
\x20 continue\n\
|
||||||
|
\x20 try:\n\
|
||||||
|
\x20 o = g[k]\n\
|
||||||
|
\x20 except Exception:\n\
|
||||||
|
\x20 continue\n\
|
||||||
|
\x20 (ds if isinstance(o, h5py.Dataset) else gs).append(k)\n\
|
||||||
|
print('\\x1f'.join(sorted(ds)))\n\
|
||||||
|
print('\\x1f'.join(sorted(gs)))\n"
|
||||||
|
));
|
||||||
|
let mut lines = out.lines();
|
||||||
|
let split = |l: Option<&str>| -> Vec<String> {
|
||||||
|
l.unwrap_or("")
|
||||||
|
.split('\x1f')
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
.map(str::to_string)
|
||||||
|
.collect()
|
||||||
|
};
|
||||||
|
let ds = split(lines.next());
|
||||||
|
let gs = split(lines.next());
|
||||||
|
(ds, gs)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn our_listing(path: &str, group: &str) -> (Vec<String>, Vec<String>) {
|
||||||
|
let f = File::open(path).unwrap();
|
||||||
|
let g = f.group(group).unwrap();
|
||||||
|
let mut ds = g.datasets().unwrap();
|
||||||
|
let mut gs = g.groups().unwrap();
|
||||||
|
ds.sort();
|
||||||
|
gs.sort();
|
||||||
|
(ds, gs)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn assert_same_listing(path: &str, group: &str) {
|
||||||
|
let ours = our_listing(path, group);
|
||||||
|
let theirs = h5py_listing(path, group);
|
||||||
|
assert_eq!(ours.0.len(), theirs.0.len(), "dataset count in {group}");
|
||||||
|
assert_eq!(ours, theirs, "listing of {group}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dense_group_whose_heap_outgrows_the_root_direct_rows() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
// Long link names make the link heap larger than the root indirect
|
||||||
|
// block's direct rows can hold (512 KiB with h5py's defaults), so links
|
||||||
|
// live in child indirect blocks. Those were sized from the wrong row
|
||||||
|
// count, and every link past the direct rows was unreachable.
|
||||||
|
let (_dir, path) = h5py_file(
|
||||||
|
"t = f.create_dataset('t', data=[1.0])\n\
|
||||||
|
g = f.create_group('g')\n\
|
||||||
|
for i in range(2500):\n\
|
||||||
|
\x20 g['n%05d_' % i + 'x' * 240] = t\n",
|
||||||
|
);
|
||||||
|
assert_same_listing(&path, "g");
|
||||||
|
let f = File::open(&path).unwrap();
|
||||||
|
let last = format!("g/n02499_{}", "x".repeat(240));
|
||||||
|
assert_eq!(f.dataset(&last).unwrap().read_f64().unwrap(), vec![1.0]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dense_group_with_a_three_level_name_index() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
// 24 000 links give the link-name v2 B-tree a depth of 3. Internal-node
|
||||||
|
// child pointers carry the subtree's total record count in a width that
|
||||||
|
// depends on the most records a subtree can hold; the reader estimated
|
||||||
|
// that as leaf_max^depth, read the root's pointers 3 bytes wide instead
|
||||||
|
// of 2, and decoded garbage heap IDs.
|
||||||
|
let (_dir, path) = h5py_file(
|
||||||
|
"t = f.create_dataset('t', data=[1.0])\n\
|
||||||
|
g = f.create_group('g')\n\
|
||||||
|
for i in range(24000):\n\
|
||||||
|
\x20 g['l%06d' % i] = t\n",
|
||||||
|
);
|
||||||
|
assert_same_listing(&path, "g");
|
||||||
|
let f = File::open(&path).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
f.dataset("g/l023999").unwrap().read_f64().unwrap(),
|
||||||
|
vec![1.0]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The attribute names h5py reports for `obj`, sorted.
|
||||||
|
fn h5py_attr_names(path: &str, obj: &str) -> Vec<String> {
|
||||||
|
let out = run_python(&format!(
|
||||||
|
"import h5py\n\
|
||||||
|
with h5py.File(r'{path}', 'r') as f:\n\
|
||||||
|
\x20 print('\\x1f'.join(sorted(f[{obj:?}].attrs.keys())))\n"
|
||||||
|
));
|
||||||
|
out.split('\x1f')
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
.map(str::to_string)
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dense_attribute_stored_as_a_huge_heap_object() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
// More than 8 attributes puts them in dense storage; one larger than the
|
||||||
|
// heap's 4 KiB managed-object limit is stored as a "huge" object, outside
|
||||||
|
// the heap blocks and found through the huge-object v2 B-tree. Its heap ID
|
||||||
|
// (type bits 4-5 = 1) was misread as a managed ID, and the error made
|
||||||
|
// every attribute on the object unreadable. NetCDF-4 files hit this
|
||||||
|
// (netcdf-c's issue671.nc / issue672.nc).
|
||||||
|
let (_dir, path) = h5py_file(
|
||||||
|
"d = f.create_dataset('d', data=[1.0])\n\
|
||||||
|
for i in range(10):\n\
|
||||||
|
\x20 d.attrs['a%d' % i] = i\n\
|
||||||
|
d.attrs['big'] = np.arange(1024, dtype='f8')\n\
|
||||||
|
d.attrs['bigger'] = np.arange(20000, dtype='i8') * 3\n",
|
||||||
|
);
|
||||||
|
let f = File::open(&path).unwrap();
|
||||||
|
let attrs = f.dataset("d").unwrap().attrs().unwrap();
|
||||||
|
let mut names: Vec<String> = attrs.keys().cloned().collect();
|
||||||
|
names.sort();
|
||||||
|
assert_eq!(names, h5py_attr_names(&path, "d"));
|
||||||
|
for i in 0..10 {
|
||||||
|
assert!(
|
||||||
|
matches!(attrs[&format!("a{i}")], AttrValue::I64(v) if v == i),
|
||||||
|
"a{i}: {:?}",
|
||||||
|
attrs[&format!("a{i}")]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let big: Vec<f64> = (0..1024).map(f64::from).collect();
|
||||||
|
assert!(matches!(&attrs["big"], AttrValue::F64Array(v) if *v == big));
|
||||||
|
let bigger: Vec<i64> = (0..20000).map(|v| v * 3).collect();
|
||||||
|
assert!(matches!(&attrs["bigger"], AttrValue::I64Array(v) if *v == bigger));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A group whose link heap has a deflate I/O filter (set on the group
|
||||||
|
/// creation property list), with 3 000 links and one link whose message is
|
||||||
|
/// larger than the heap's managed-object limit, so it is a huge object.
|
||||||
|
fn huge_link_group(filtered: bool) -> (tempfile::TempDir, String) {
|
||||||
|
let filter = if filtered {
|
||||||
|
"import ctypes, glob, os\n\
|
||||||
|
lib = ctypes.CDLL(glob.glob(os.path.join(os.path.dirname(h5py.__file__), '..', 'h5py.libs', 'libhdf5-*.so*'))[0])\n\
|
||||||
|
lib.H5Pset_deflate.argtypes = [ctypes.c_int64, ctypes.c_uint]\n\
|
||||||
|
assert lib.H5Pset_deflate(gcpl.id, 6) >= 0\n"
|
||||||
|
} else {
|
||||||
|
""
|
||||||
|
};
|
||||||
|
h5py_file(&format!(
|
||||||
|
"t = f.create_dataset('t', data=[1.0])\n\
|
||||||
|
gcpl = h5py.h5p.create(h5py.h5p.GROUP_CREATE)\n\
|
||||||
|
{filter}\
|
||||||
|
h5py.h5g.create(f.id, b'g', gcpl=gcpl)\n\
|
||||||
|
g = f['g']\n\
|
||||||
|
for i in range(3000):\n\
|
||||||
|
\x20 g['l%05d' % i] = t\n\
|
||||||
|
g['L' * 5000] = t\n"
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn check_huge_link_group(filtered: bool) {
|
||||||
|
let (_dir, path) = huge_link_group(filtered);
|
||||||
|
assert_same_listing(&path, "g");
|
||||||
|
let f = File::open(&path).unwrap();
|
||||||
|
let huge = format!("g/{}", "L".repeat(5000));
|
||||||
|
assert_eq!(f.dataset(&huge).unwrap().read_f64().unwrap(), vec![1.0]);
|
||||||
|
assert_eq!(
|
||||||
|
f.dataset("g/l02999").unwrap().read_f64().unwrap(),
|
||||||
|
vec![1.0]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dense_group_with_a_huge_link() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
check_huge_link_group(false);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn dense_group_with_a_filtered_link_heap() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
// libhdf5 applies a group's filter pipeline to its link heap: direct
|
||||||
|
// blocks and huge objects are stored deflated, and the heap header
|
||||||
|
// carries the pipeline. The header's checksum was looked for in the
|
||||||
|
// wrong place, and filtered blocks were read raw.
|
||||||
|
if run_python(
|
||||||
|
"import h5py, glob, os\nprint(len(glob.glob(os.path.join(os.path.dirname(h5py.__file__), '..', 'h5py.libs', 'libhdf5-*.so*'))))",
|
||||||
|
) == "0"
|
||||||
|
{
|
||||||
|
assert!(
|
||||||
|
!interop_required(),
|
||||||
|
"CLAWHDF5_REQUIRE_INTEROP=1 but h5py's bundled libhdf5 was not found"
|
||||||
|
);
|
||||||
|
eprintln!("SKIP: h5py's bundled libhdf5 not found (needed to set the filter)");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
check_huge_link_group(true);
|
||||||
|
}
|
||||||
|
|
||||||
|
fn fixture(name: &str) -> String {
|
||||||
|
format!("{}/tests/fixtures/{name}", env!("CARGO_MANIFEST_DIR"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `tall.h5` and `tudlink.h5` are libhdf5's own tool test files
|
||||||
|
/// (`tools/test/testfiles`, BSD-style HDF5 licence). Each has user-defined
|
||||||
|
/// links of class 187, which h5py lists by name but cannot open, and h5dump
|
||||||
|
/// prints as `USERDEFINED_LINK`. One such link made the whole group
|
||||||
|
/// unlistable (`InvalidLinkType(187)`); it is now left out of the listing
|
||||||
|
/// like any other link that cannot be followed.
|
||||||
|
#[test]
|
||||||
|
fn user_defined_links_do_not_break_the_listing() {
|
||||||
|
let f = File::open(fixture("tall.h5")).unwrap();
|
||||||
|
let g2 = f.group("g2").unwrap();
|
||||||
|
let mut ds = g2.datasets().unwrap();
|
||||||
|
ds.sort();
|
||||||
|
assert_eq!(ds, ["dset2.1", "dset2.2"]);
|
||||||
|
assert!(g2.groups().unwrap().is_empty());
|
||||||
|
assert!(f.dataset("g2/udlink").is_err());
|
||||||
|
|
||||||
|
let f = File::open(fixture("tudlink.h5")).unwrap();
|
||||||
|
assert!(f.root().datasets().unwrap().is_empty());
|
||||||
|
assert!(f.root().groups().unwrap().is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Soft links (absolute, relative, to a dataset and to a group), a dangling
|
||||||
|
/// one, a cycle and an external link, in an old-style (symbol table) or
|
||||||
|
/// new-style (link message) group.
|
||||||
|
fn soft_link_file(libver: &str) -> (tempfile::TempDir, String) {
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("soft.h5").display().to_string();
|
||||||
|
run_python(&format!(
|
||||||
|
"import h5py, numpy as np\n\
|
||||||
|
with h5py.File(r'{path}', 'w', libver='{libver}') as f:\n\
|
||||||
|
\x20 f.create_dataset('a/b/deep', data=np.arange(3.0))\n\
|
||||||
|
\x20 f.create_dataset('plain', data=[1.0])\n\
|
||||||
|
\x20 f['soft_ds'] = h5py.SoftLink('/a/b/deep')\n\
|
||||||
|
\x20 f['soft_grp'] = h5py.SoftLink('/a')\n\
|
||||||
|
\x20 f['a/rel'] = h5py.SoftLink('b/deep')\n\
|
||||||
|
\x20 f['a/rel_grp'] = h5py.SoftLink('b')\n\
|
||||||
|
\x20 f['dangling'] = h5py.SoftLink('/nope')\n\
|
||||||
|
\x20 f['loop1'] = h5py.SoftLink('/loop2')\n\
|
||||||
|
\x20 f['loop2'] = h5py.SoftLink('/loop1')\n\
|
||||||
|
\x20 f['ext'] = h5py.ExternalLink('elsewhere.h5', '/x')\n"
|
||||||
|
));
|
||||||
|
(dir, path)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn check_soft_links(libver: &str) {
|
||||||
|
let (_dir, path) = soft_link_file(libver);
|
||||||
|
assert_same_listing(&path, "/");
|
||||||
|
assert_same_listing(&path, "a");
|
||||||
|
|
||||||
|
let f = File::open(&path).unwrap();
|
||||||
|
let root = f.root();
|
||||||
|
let deep = vec![0.0, 1.0, 2.0];
|
||||||
|
assert_eq!(root.dataset("soft_ds").unwrap().read_f64().unwrap(), deep);
|
||||||
|
let a = root.group("soft_grp").unwrap();
|
||||||
|
assert_eq!(a.dataset("rel").unwrap().read_f64().unwrap(), deep);
|
||||||
|
assert_eq!(a.group("rel_grp").unwrap().datasets().unwrap(), ["deep"]);
|
||||||
|
// A dangling link is not listed and cannot be opened.
|
||||||
|
assert!(root.dataset("dangling").is_err());
|
||||||
|
assert!(root.dataset("loop1").is_err());
|
||||||
|
|
||||||
|
// The memory-mapped and lazy handles list the same way.
|
||||||
|
let (ds, gs) = h5py_listing(&path, "/");
|
||||||
|
let m = clawhdf5::MmapFile::open(&path).unwrap();
|
||||||
|
let mut mds = m.root().datasets().unwrap();
|
||||||
|
let mut mgs = m.root().groups().unwrap();
|
||||||
|
mds.sort();
|
||||||
|
mgs.sort();
|
||||||
|
assert_eq!((mds, mgs), (ds.clone(), gs.clone()));
|
||||||
|
let l = clawhdf5::LazyFile::from_bytes(std::fs::read(&path).unwrap()).unwrap();
|
||||||
|
let mut lds = l.root().datasets().unwrap();
|
||||||
|
let mut lgs = l.root().groups().unwrap();
|
||||||
|
lds.sort();
|
||||||
|
lgs.sort();
|
||||||
|
assert_eq!((lds, lgs), (ds, gs));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Soft links were left out of `datasets()`/`groups()` (and could not be
|
||||||
|
/// opened by name from a group handle); in old-style groups, where a soft
|
||||||
|
/// link has no object header address, they made the listing fail. h5py
|
||||||
|
/// lists a soft link under its own name as whatever it points at.
|
||||||
|
#[test]
|
||||||
|
fn soft_links_are_listed_as_their_targets() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
check_soft_links("latest");
|
||||||
|
check_soft_links("earliest");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One unreadable attribute used to fail `attrs()` for every attribute on
|
||||||
|
/// the object. Now it is left out (and reported by `attrs_with_errors`),
|
||||||
|
/// and the others are returned with their full values.
|
||||||
|
#[test]
|
||||||
|
fn one_unreadable_attribute_does_not_hide_the_others() {
|
||||||
|
skip_if_no_python!();
|
||||||
|
let (dir, path) = h5py_file(
|
||||||
|
"d = f.create_dataset('d', data=[1.0])\n\
|
||||||
|
for i in range(10):\n\
|
||||||
|
\x20 d.attrs['a%d' % i] = float(i)\n\
|
||||||
|
d.attrs['zz_broken_attribute'] = 42.0\n",
|
||||||
|
);
|
||||||
|
// Dense attributes live in a fractal heap whose blocks carry no checksum
|
||||||
|
// by default: give the one named `zz_broken_attribute` a nonexistent
|
||||||
|
// attribute message version (the byte 9 before its name in a v3 message).
|
||||||
|
let mut bytes = std::fs::read(&path).unwrap();
|
||||||
|
let needle = b"zz_broken_attribute";
|
||||||
|
let at = bytes
|
||||||
|
.windows(needle.len())
|
||||||
|
.position(|w| w == needle)
|
||||||
|
.expect("attribute name in the file");
|
||||||
|
assert_eq!(bytes[at - 9], 3, "expected a version-3 attribute message");
|
||||||
|
bytes[at - 9] = 0x7f;
|
||||||
|
let broken = dir.path().join("broken.h5");
|
||||||
|
std::fs::write(&broken, &bytes).unwrap();
|
||||||
|
|
||||||
|
for file in [
|
||||||
|
File::open(&broken).unwrap(),
|
||||||
|
File::from_bytes(bytes.clone()).unwrap(),
|
||||||
|
] {
|
||||||
|
let ds = file.dataset("d").unwrap();
|
||||||
|
let (attrs, errors) = ds.attrs_with_errors().unwrap();
|
||||||
|
assert_eq!(errors.len(), 1, "{errors:?}");
|
||||||
|
assert_eq!(attrs.len(), 10);
|
||||||
|
for i in 0..10 {
|
||||||
|
assert!(
|
||||||
|
matches!(attrs[&format!("a{i}")], AttrValue::F64(v) if v == f64::from(i)),
|
||||||
|
"a{i}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
assert!(!attrs.contains_key("zz_broken_attribute"));
|
||||||
|
assert_eq!(ds.attrs().unwrap().len(), 10);
|
||||||
|
}
|
||||||
|
let m = clawhdf5::MmapFile::open(&broken).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
m.dataset("d").unwrap().attrs_with_errors().unwrap().1.len(),
|
||||||
|
1
|
||||||
|
);
|
||||||
|
let l = clawhdf5::LazyFile::from_bytes(bytes).unwrap();
|
||||||
|
assert_eq!(l.dataset("d").unwrap().attrs().unwrap().len(), 10);
|
||||||
|
}
|
||||||
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
+14
@@ -0,0 +1,14 @@
|
|||||||
|
# VDS test files from libhdf5
|
||||||
|
|
||||||
|
Copied unchanged from the HDF Group's HDF5 repository,
|
||||||
|
`tools/test/testfiles/vds/` (the h5dump/h5ls VDS test data). HDF5 is
|
||||||
|
distributed under a BSD-style license (see `COPYING` in the HDF5 source).
|
||||||
|
|
||||||
|
| File | What it exercises |
|
||||||
|
|---|---|
|
||||||
|
| `vds-eiger.h5` + `f-0.h5`, `f-3.h5` | printf-style source name `f-%b.h5`; `f-3.h5` lies past the first missing source and must be ignored (extent 5, not 20) |
|
||||||
|
| `4_vds.h5` + `4_0.h5`..`4_2.h5` | printf-style `4_%b.h5` with version-2 (1.10 format) hyperslab selections |
|
||||||
|
| `vds-percival-unlim-maxmin.h5` + `a.h5`..`d.h5` | four interleaved unlimited mappings whose sources have different lengths |
|
||||||
|
|
||||||
|
Used by `crates/clawhdf5/tests/vds_interop.rs::vds_libhdf5_test_files`,
|
||||||
|
which compares our reads with h5py's.
|
||||||
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user