parse_frame sized the offsets chunk from the frame header's own nbytes and chunksize, so a 173-byte frame declaring 32 Mi chunks, with a 40-byte repeated-value offsets chunk, built 256 MiB (up to 2 GiB) of offsets for a 1 MiB HDF5 chunk and then returned 4 bytes. The offsets chunk is now capped at the output limit (at least 128 bytes); a frame whose nbytes/chunksize imply more chunks than that is refused before anything is allocated. tests/blosc2_alloc_bounds.rs measures peak allocation with a counting global allocator; the reviewer's frame failed it (decoded Ok(4)) before. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
192 lines
6.5 KiB
Rust
192 lines
6.5 KiB
Rust
//! Crafted Blosc2 frames and chunks cannot make the decoder allocate out of
|
|
//! proportion to the HDF5 chunk it decodes.
|
|
//!
|
|
//! A frame's header, its offsets chunk and its chunk headers all declare
|
|
//! sizes, and the decoder used to allocate what they declared: a 173-byte
|
|
//! frame whose offsets chunk claimed 2 GiB was decoded in full before any
|
|
//! check failed. Every allocation is now bounded by the output limit (the
|
|
//! HDF5 chunk's size) and the input's length.
|
|
//!
|
|
//! Peak heap use is measured with a counting global allocator; the tests
|
|
//! share it, so each holds `SERIAL` for its whole run.
|
|
#![cfg(feature = "blosc2")]
|
|
|
|
use std::alloc::{GlobalAlloc, Layout, System};
|
|
use std::sync::Mutex;
|
|
use std::sync::atomic::{AtomicUsize, Ordering};
|
|
|
|
use clawhdf5_format::filters_blosc2::{blosc2_decompress, blosc2_decompress_chunk};
|
|
|
|
struct Counting;
|
|
|
|
static CURRENT: AtomicUsize = AtomicUsize::new(0);
|
|
static PEAK: AtomicUsize = AtomicUsize::new(0);
|
|
static SERIAL: Mutex<()> = Mutex::new(());
|
|
|
|
unsafe impl GlobalAlloc for Counting {
|
|
unsafe fn alloc(&self, layout: Layout) -> *mut u8 {
|
|
let p = unsafe { System.alloc(layout) };
|
|
if !p.is_null() {
|
|
let now = CURRENT.fetch_add(layout.size(), Ordering::Relaxed) + layout.size();
|
|
PEAK.fetch_max(now, Ordering::Relaxed);
|
|
}
|
|
p
|
|
}
|
|
|
|
unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 {
|
|
let p = unsafe { System.alloc_zeroed(layout) };
|
|
if !p.is_null() {
|
|
let now = CURRENT.fetch_add(layout.size(), Ordering::Relaxed) + layout.size();
|
|
PEAK.fetch_max(now, Ordering::Relaxed);
|
|
}
|
|
p
|
|
}
|
|
|
|
unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) {
|
|
unsafe { System.dealloc(ptr, layout) };
|
|
CURRENT.fetch_sub(layout.size(), Ordering::Relaxed);
|
|
}
|
|
}
|
|
|
|
#[global_allocator]
|
|
static ALLOC: Counting = Counting;
|
|
|
|
/// Bytes allocated at the peak of `f`, above what was live when it started.
|
|
fn peak_during<T>(f: impl FnOnce() -> T) -> (T, usize) {
|
|
let base = CURRENT.load(Ordering::Relaxed);
|
|
PEAK.store(base, Ordering::Relaxed);
|
|
let out = f();
|
|
(out, PEAK.load(Ordering::Relaxed).saturating_sub(base))
|
|
}
|
|
|
|
/// What decoding one HDF5 chunk of `limit` bytes from `input` may hold at
|
|
/// once: the output, a few blocks of scratch (each no larger than the
|
|
/// output), the offsets table, and the Zstandard decoder's state.
|
|
fn bound(limit: usize, input: &[u8]) -> usize {
|
|
6 * limit + 2 * input.len() + (1 << 20)
|
|
}
|
|
|
|
fn lock() -> std::sync::MutexGuard<'static, ()> {
|
|
SERIAL.lock().unwrap_or_else(|e| e.into_inner())
|
|
}
|
|
|
|
/// A 32-byte (extended) Blosc2 chunk header.
|
|
fn chunk_header(ts: u8, nbytes: i32, blocksize: i32, cbytes: i32, special: u8) -> Vec<u8> {
|
|
let mut c = vec![5u8, 1, 0x05, ts];
|
|
for v in [nbytes, blocksize, cbytes] {
|
|
c.extend_from_slice(&v.to_le_bytes());
|
|
}
|
|
c.resize(32, 0);
|
|
c[31] = special << 4;
|
|
c
|
|
}
|
|
|
|
/// A chunk of `nbytes` bytes that repeats one value (special type 3).
|
|
fn repeated(value: &[u8], nbytes: i32, blocksize: i32) -> Vec<u8> {
|
|
let mut c = chunk_header(value.len() as u8, nbytes, blocksize, 32 + value.len() as i32, 3);
|
|
c.extend_from_slice(value);
|
|
c
|
|
}
|
|
|
|
/// A frame offset recording a special chunk of `kind` (1 zeros, 2 NaN).
|
|
fn special_offset(kind: u8) -> [u8; 8] {
|
|
(((0x80 | kind) as i64) << 56).to_le_bytes()
|
|
}
|
|
|
|
/// A B2ND metalayer.
|
|
fn nd_meta(shape: &[i64], chunks: &[i32], blocks: &[i32]) -> Vec<u8> {
|
|
let n = shape.len() as u8;
|
|
let mut m = vec![0x95, 0, n, 0x90 | n];
|
|
for s in shape {
|
|
m.push(0xd3);
|
|
m.extend_from_slice(&s.to_be_bytes());
|
|
}
|
|
for dims in [chunks, blocks] {
|
|
m.push(0x90 | n);
|
|
for d in dims {
|
|
m.push(0xd2);
|
|
m.extend_from_slice(&d.to_be_bytes());
|
|
}
|
|
}
|
|
m
|
|
}
|
|
|
|
/// A contiguous frame: header (with a `b2nd` metalayer if given), the data
|
|
/// chunks, then the offsets chunk.
|
|
fn frame(
|
|
meta: Option<&[u8]>,
|
|
nbytes: i64,
|
|
typesize: i32,
|
|
chunksize: i32,
|
|
data: &[u8],
|
|
offsets: &[u8],
|
|
) -> Vec<u8> {
|
|
let mut h = vec![0u8; 91];
|
|
h[0] = 0x9e;
|
|
h[1] = 0xa8;
|
|
h[2..10].copy_from_slice(b"b2frame\0");
|
|
h[25] = 2;
|
|
match meta {
|
|
Some(m) => {
|
|
h.extend_from_slice(&[0xde, 0, 1, 0xa4]);
|
|
h.extend_from_slice(b"b2nd");
|
|
let at = h.len() as i32 + 5;
|
|
h.push(0xd2);
|
|
h.extend_from_slice(&at.to_be_bytes());
|
|
h.push(0xc6);
|
|
h.extend_from_slice(&(m.len() as u32).to_be_bytes());
|
|
h.extend_from_slice(m);
|
|
}
|
|
None => h.extend_from_slice(&[0xde, 0, 0]),
|
|
}
|
|
let header_len = h.len() as i32;
|
|
h[11..15].copy_from_slice(&header_len.to_be_bytes());
|
|
h[30..38].copy_from_slice(&nbytes.to_be_bytes());
|
|
h[39..47].copy_from_slice(&(data.len() as i64).to_be_bytes());
|
|
h[48..52].copy_from_slice(&typesize.to_be_bytes());
|
|
h[58..62].copy_from_slice(&chunksize.to_be_bytes());
|
|
h.extend_from_slice(data);
|
|
h.extend_from_slice(offsets);
|
|
let len = h.len() as u64;
|
|
h[16..24].copy_from_slice(&len.to_be_bytes());
|
|
h
|
|
}
|
|
|
|
/// The frame header's own sizes must not size the offsets chunk: a frame
|
|
/// declaring 32 Mi chunks of 4 bytes, whose offsets chunk (40 bytes) says
|
|
/// "one repeated offset, 256 MiB of them", made the decoder build all
|
|
/// 256 MiB of offsets for a 1 MiB HDF5 chunk and then return 4 bytes.
|
|
#[test]
|
|
fn offsets_chunk_is_bounded_by_the_output_limit() {
|
|
let _g = lock();
|
|
let limit = 1 << 20;
|
|
let offsets_len: i32 = 256 << 20;
|
|
let nchunks = offsets_len as i64 / 8;
|
|
let offsets = repeated(&special_offset(1), offsets_len, 64 << 20);
|
|
let f = frame(None, nchunks * 4, 4, 4, &[], &offsets);
|
|
let (r, peak) = peak_during(|| blosc2_decompress(&f, limit));
|
|
assert!(r.is_err(), "decoded {:?} bytes", r.map(|v| v.len()));
|
|
assert!(
|
|
peak <= bound(limit, &f),
|
|
"peak {peak} bytes for a {}-byte frame",
|
|
f.len()
|
|
);
|
|
// The same frame with a variable chunk size (0): the offsets chunk
|
|
// alone says how many chunks there are.
|
|
let f = frame(None, nchunks * 4, 4, 0, &[], &offsets);
|
|
let (r, peak) = peak_during(|| blosc2_decompress(&f, limit));
|
|
assert!(r.is_err());
|
|
assert!(peak <= bound(limit, &f), "chunksize 0: peak {peak} bytes");
|
|
}
|
|
|
|
/// A legitimate frame of this shape (one chunk, its offset special) still
|
|
/// decodes.
|
|
#[test]
|
|
fn small_frames_still_decode() {
|
|
let _g = lock();
|
|
let offsets = repeated(&special_offset(1), 8, 8);
|
|
let f = frame(None, 64, 4, 64, &[], &offsets);
|
|
assert_eq!(blosc2_decompress(&f, 64).unwrap(), vec![0; 64]);
|
|
let _ = blosc2_decompress_chunk;
|
|
}
|