perf: eliminate double compression and improve shuffle filter throughput
Four independent write-path improvements: 1. Cache compressed chunks between Pass 1 and Pass 2 (chunked_write.rs, file_writer.rs): the two-pass layout writer previously called build_chunked_data_at_ext() twice per chunked dataset — once in Pass 1 to get blob sizes and once in Pass 2 with real addresses. Add PrecompressedChunks / precompress_chunks() / build_chunked_data_from_ precompressed() to compress once in Pass 1, cache the result, and only rebuild the address-dependent index structures in Pass 2. Expected ~2× speedup for chunked+deflate writes (512×512 deflate: 3.33ms → ~1.7ms). 2. SIMD-vectorisable shuffle filter (filters.rs): replace the naïve O(N·S) nested loop with an unrolled u32-load path for 4-byte elements (f32) and a cache-blocked tile loop for all other sizes. LLVM auto-vectorises the 4-byte path into SSE2/AVX2/NEON byte-deinterleave sequences. 3. Zstd benchmark variant (h5bench_write.rs): add write_2d_chunked_zstd group measuring Zstd level 3 vs deflate level 6 side-by-side. Also fix the existing write_2d_chunked benchmark — the clawhdf5 path was missing .with_deflate(6), making the comparison apples-to-oranges. Add arXiv- backed doc recommendation on DatasetBuilder::with_zstd(). 4. Zero-copy HNSW save (hnsw.rs, clawhdf5-io/lib.rs): add FileWriter::write_bytes_owned(Vec<u8>) that takes ownership to avoid the full-file clone in write_all_bytes(&[u8]). HNSW::save_to_hdf5 uses it. Co-Authored-By: Claude Sonnet 4.6 <[email protected]>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
3a1fcc5cb3
commit
2ddb22897c
@@ -83,7 +83,8 @@ fn bench_write_2d_chunked(c: &mut Criterion) {
|
||||
fb.create_dataset("matrix")
|
||||
.with_f32_data(d)
|
||||
.with_shape(&[rows as u64, cols as u64])
|
||||
.with_chunks(&[cr, cc]);
|
||||
.with_chunks(&[cr, cc])
|
||||
.with_deflate(6);
|
||||
fb.write(&path).unwrap();
|
||||
});
|
||||
});
|
||||
@@ -109,6 +110,64 @@ fn bench_write_2d_chunked(c: &mut Criterion) {
|
||||
group.finish();
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Workload: write_2d_chunked_zstd
|
||||
// Same matrix sizes as write_2d_chunked but uses Zstd level 3.
|
||||
// Zstd level 3 typically encodes 500+ MiB/s vs deflate's ~300 MiB/s at the
|
||||
// same or better compression ratio (arXiv 2604.06221, ROOT I/O 2019).
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
fn bench_write_2d_chunked_zstd(c: &mut Criterion) {
|
||||
let mut group = c.benchmark_group("write_2d_chunked_zstd");
|
||||
|
||||
let configs: &[(usize, usize, u64, u64)] = &[
|
||||
(32, 32, 8, 32),
|
||||
(128, 128, 32, 128),
|
||||
(512, 512, 64, 512),
|
||||
];
|
||||
|
||||
for &(rows, cols, cr, cc) in configs {
|
||||
let n = rows * cols;
|
||||
let data: Vec<f32> = (0..n).map(|i| i as f32).collect();
|
||||
let label = format!("{rows}x{cols}");
|
||||
group.throughput(Throughput::Bytes((n * size_of::<f32>()) as u64));
|
||||
|
||||
group.bench_with_input(BenchmarkId::new("clawhdf5/zstd-3", &label), &data, |b, d| {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let path = tmp.path().join("write_2d_chunked_zstd.h5");
|
||||
b.iter(|| {
|
||||
let mut fb = FileBuilder::new();
|
||||
fb.create_dataset("matrix")
|
||||
.with_f32_data(d)
|
||||
.with_shape(&[rows as u64, cols as u64])
|
||||
.with_chunks(&[cr, cc])
|
||||
.with_zstd(3);
|
||||
fb.write(&path).unwrap();
|
||||
});
|
||||
});
|
||||
|
||||
group.bench_with_input(
|
||||
BenchmarkId::new("clawhdf5/deflate-6", &label),
|
||||
&data,
|
||||
|b, d| {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let path = tmp.path().join("write_2d_chunked_deflate.h5");
|
||||
b.iter(|| {
|
||||
let mut fb = FileBuilder::new();
|
||||
fb.create_dataset("matrix")
|
||||
.with_f32_data(d)
|
||||
.with_shape(&[rows as u64, cols as u64])
|
||||
.with_chunks(&[cr, cc])
|
||||
.with_deflate(6);
|
||||
fb.write(&path).unwrap();
|
||||
});
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
group.finish();
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Workload: write_f64_batch
|
||||
// Write batches of f64 elements — simulates the clawhdf5-agent embedding
|
||||
@@ -205,6 +264,7 @@ criterion_group!(
|
||||
write_benches,
|
||||
bench_write_1d_contiguous,
|
||||
bench_write_2d_chunked,
|
||||
bench_write_2d_chunked_zstd,
|
||||
bench_write_f64_batch,
|
||||
bench_write_multi_dataset,
|
||||
bench_write_with_attrs,
|
||||
|
||||
Reference in New Issue
Block a user