read_raw_data_selection computed which chunks a selection intersects, threw the answer away, decoded the entire dataset and picked elements out of it — for contiguous layouts too. A 64x64 window of a 64 MB deflate dataset cost 105 ms, about half a full read; every selection cost the same whatever its size. New partial_read module: materialise only the selection's bounding box — the overlapping rows of a contiguous dataset (straight from the file bytes) or the overlapping chunks (only those are decompressed) — then run the existing extractor over that buffer with the selection translated to the box origin, so extraction semantics are exactly the full-read ones. It declines (falling back to the old path) for All/None, compact/virtual/storage-less layouts, and boxes covering more than half the dataset. That window now takes 0.39 ms, one row 2.7 ms, one column 5.2 ms. Selections are validated against the dataset shape first. They were not: a hyperslab past an edge came back padded with zeros and a point with an out-of-range column wrapped into the next row, returning the wrong element with no error. Now FormatError::SelectionOutOfBounds (also rank mismatch and overlapping blocks); the facade's fill-aware path validates too. Tests: equivalence against a reference extraction from a full read over 60 random hyperslabs/point lists per layout (contiguous, chunked, deflate) for ranks 1-3. New read_harness bench binary with before/after in BENCHMARKS.md. Co-Authored-By: Claude Fable 5.1 <[email protected]>
92 lines
2.8 KiB
TOML
92 lines
2.8 KiB
TOML
[package]
|
|
name = "clawhdf5-bench"
|
|
version = "2.4.0"
|
|
edition = "2024"
|
|
description = "Benchmark harnesses for clawhdf5-agent (Track 8)"
|
|
license = "MIT"
|
|
|
|
[[bin]]
|
|
name = "longmemeval_bench"
|
|
path = "src/bin/longmemeval_bench.rs"
|
|
|
|
[[bin]]
|
|
name = "memory_arena"
|
|
path = "src/bin/memory_arena.rs"
|
|
|
|
[[bin]]
|
|
name = "read_harness"
|
|
path = "src/bin/read_harness.rs"
|
|
|
|
[[bin]]
|
|
name = "search_harness"
|
|
path = "src/bin/search_harness.rs"
|
|
|
|
[[bin]]
|
|
name = "footprint_bench"
|
|
path = "src/bin/footprint_bench.rs"
|
|
|
|
[[bin]]
|
|
name = "consolidation_efficiency"
|
|
path = "src/bin/consolidation_efficiency.rs"
|
|
|
|
[[bin]]
|
|
name = "ephemeral_perf"
|
|
path = "src/bin/ephemeral_perf.rs"
|
|
|
|
[[bin]]
|
|
name = "mpi_io_bench"
|
|
path = "src/bin/mpi_io_bench.rs"
|
|
required-features = ["mpi-io"]
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# h5bench-equivalent Criterion benchmarks
|
|
# ---------------------------------------------------------------------------
|
|
|
|
[[bench]]
|
|
name = "h5bench_write"
|
|
harness = false
|
|
|
|
[[bench]]
|
|
name = "h5bench_read"
|
|
harness = false
|
|
|
|
[[bench]]
|
|
name = "h5bench_meta"
|
|
harness = false
|
|
|
|
[dependencies]
|
|
clawhdf5-agent = { path = "../clawhdf5-agent" }
|
|
clawhdf5-ann = { path = "../clawhdf5-ann" }
|
|
clawhdf5 = { path = "../clawhdf5" }
|
|
clawhdf5-format = { path = "../clawhdf5-format" }
|
|
clawhdf5-io = { path = "../clawhdf5-io" }
|
|
mpi = { version = "0.8", optional = true }
|
|
serde = { workspace = true }
|
|
serde_json = "1"
|
|
tempfile = { workspace = true }
|
|
# Optional: libhdf5 C wrapper for side-by-side comparison (requires system libhdf5).
|
|
# Enable with: cargo bench -p clawhdf5-bench --features libhdf5-compare
|
|
# Uses hdf5-metno (fork of hdf5 crate) which supports HDF5 1.14.x.
|
|
hdf5 = { version = "0.12", optional = true, package = "hdf5-metno" }
|
|
# Optional: real sentence embeddings for the LongMemEval bench's vector stage.
|
|
# Enable with: cargo run --release --bin longmemeval_bench --features embeddings
|
|
# Off by default — nothing in the shipped crates depends on these.
|
|
candle-core = { version = "0.9", optional = true }
|
|
candle-nn = { version = "0.9", optional = true }
|
|
candle-transformers = { version = "0.9", optional = true }
|
|
tokenizers = { version = "0.21", optional = true }
|
|
|
|
[dev-dependencies]
|
|
clawhdf5 = { path = "../clawhdf5", features = ["zstd", "pcodec"] }
|
|
criterion = { workspace = true }
|
|
|
|
[features]
|
|
# When enabled, benchmarks add matching libhdf5 variants for side-by-side comparison.
|
|
libhdf5-compare = ["hdf5"]
|
|
mpi-io = ["clawhdf5-io/mpi-io", "mpi"]
|
|
# Real MiniLM embeddings for longmemeval_bench, so the vector stage is not inert.
|
|
embeddings = ["candle-core", "candle-nn", "candle-transformers", "tokenizers"]
|
|
# CUDA-accelerated embedding. MiniLM on a CPU takes hours over the full
|
|
# longmemeval_s haystack; on a GPU it is minutes.
|
|
embeddings-cuda = ["embeddings", "candle-core/cuda", "candle-nn/cuda", "candle-transformers/cuda"]
|