Files
clawhdf5/crates/clawhdf5-format/src/object_header.rs
T
2026-09-27 23:16:27 -05:00

1738 lines
67 KiB
Rust

//! HDF5 Object Header parsing (v1 and v2).
#[cfg(not(feature = "std"))]
use alloc::{boxed::Box, collections::BTreeSet, vec::Vec};
#[cfg(feature = "std")]
use std::collections::BTreeSet;
use byteorder::{ByteOrder, LittleEndian};
use crate::addr::to_usize;
use crate::error::FormatError;
use crate::message_type::MessageType;
use crate::storage::{Storage, Window, len_usize, read_exact_at};
/// OHDR signature for v2 object headers.
const OHDR_SIGNATURE: [u8; 4] = *b"OHDR";
/// OCHK signature for v2 continuation chunks.
const OCHK_SIGNATURE: [u8; 4] = *b"OCHK";
/// A single parsed header message.
#[derive(Debug, Clone)]
pub struct HeaderMessage {
/// The message type.
pub msg_type: MessageType,
/// Size of the message data in bytes.
pub size: usize,
/// Message flags byte.
pub flags: u8,
/// Creation order (v2 only, when tracking is enabled).
pub creation_order: Option<u16>,
/// Raw message data bytes.
pub data: Vec<u8>,
}
/// Parsed HDF5 object header.
#[derive(Debug, Clone)]
pub struct ObjectHeader {
/// Header version (1 or 2).
pub version: u8,
/// All non-NIL messages collected from all chunks.
pub messages: Vec<HeaderMessage>,
/// Object reference count (v1 only).
pub reference_count: Option<u32>,
/// Object header flags (v2 only; 0 for v1).
pub flags: u8,
/// Access time (v2, when flags bit 2 set).
pub access_time: Option<u32>,
/// Modification time (v2, when flags bit 2 set).
pub modification_time: Option<u32>,
/// Change time (v2, when flags bit 2 set).
pub change_time: Option<u32>,
/// Birth time (v2, when flags bit 2 set).
pub birth_time: Option<u32>,
}
fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> {
match offset.checked_add(needed) {
Some(end) if end <= data.len() => Ok(()),
_ => Err(FormatError::UnexpectedEof {
expected: offset.saturating_add(needed),
available: data.len(),
}),
}
}
fn read_offset(data: &[u8], pos: usize, size: u8) -> Result<u64, FormatError> {
let s = size as usize;
ensure_len(data, pos, s)?;
let slice = &data[pos..pos + s];
Ok(match size {
2 => LittleEndian::read_u16(slice) as u64,
4 => LittleEndian::read_u32(slice) as u64,
8 => LittleEndian::read_u64(slice),
1 => slice[0] as u64,
_ => {
return Err(FormatError::InvalidOffsetSize(size));
}
})
}
/// The kind of object an object header describes.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum ObjectClass {
/// A group: the header has a Symbol Table or a Link Info message.
Group,
/// A dataset: the header has a Datatype and a Dataspace message.
Dataset,
/// A committed (named) datatype: a Datatype message, no Dataspace.
NamedDatatype,
}
impl ObjectHeader {
/// The kind of object this header describes, decided as libhdf5 decides
/// it (`H5O__obj_class_real`): group first (a Symbol Table or Link Info
/// message), then dataset (a Datatype *and* a Dataspace message — not a
/// Data Layout message), then named datatype (a Datatype message).
/// `None` when none applies; libhdf5 then cannot open the object
/// ("unable to determine object type").
///
/// A header with a Datatype and a Data Layout message but no Dataspace
/// is a named datatype to libhdf5, not a dataset.
pub fn object_class(&self) -> Option<ObjectClass> {
let has = |t: MessageType| self.messages.iter().any(|m| m.msg_type == t);
if has(MessageType::SymbolTable) || has(MessageType::LinkInfo) {
Some(ObjectClass::Group)
} else if has(MessageType::Datatype) && has(MessageType::Dataspace) {
Some(ObjectClass::Dataset)
} else if has(MessageType::Datatype) {
Some(ObjectClass::NamedDatatype)
} else {
None
}
}
/// Parse an object header at the given offset in the data buffer.
///
/// `offset_size` and `length_size` come from the superblock.
#[inline]
pub fn parse(
data: &[u8],
offset: usize,
offset_size: u8,
length_size: u8,
) -> Result<ObjectHeader, FormatError> {
Self::parse_slice(data, offset as u64, offset_size, length_size)
}
/// [`Self::parse`] over any [`Storage`].
///
/// Reads the prefix (at most [`V2_PREFIX_MAX`] bytes, signature
/// included), then each chunk as one bounded read, continuation chunks
/// included. A storage with the whole file in memory is parsed as its
/// slice, by code compiled in this crate (see
/// [`crate::storage`], "Slice entry points").
#[inline]
pub fn parse_in<S: Storage + ?Sized>(
file: &S,
offset: u64,
offset_size: u8,
length_size: u8,
) -> Result<ObjectHeader, FormatError> {
match file.as_contiguous() {
Some(all) => Self::parse_slice(all, offset, offset_size, length_size),
None => Self::parse_storage(file, offset, offset_size, length_size),
}
}
/// [`Self::parse_storage`] for the slice, compiled in this crate: the
/// one copy [`Self::parse`] and [`Self::parse_in`] (in memory) call.
fn parse_slice(
data: &[u8],
offset: u64,
offset_size: u8,
length_size: u8,
) -> Result<ObjectHeader, FormatError> {
Self::parse_storage(data, offset, offset_size, length_size)
}
fn parse_storage<S: Storage + ?Sized>(
file: &S,
offset: u64,
offset_size: u8,
length_size: u8,
) -> Result<ObjectHeader, FormatError> {
// The first chunk is read once the prefix says how long it is: say
// so (see `Storage::hint`), for a storage that fetches between
// attempts.
if file.as_contiguous().is_none() {
file.hint(offset, OBJECT_HEADER_HINT_LEN);
}
// The longest prefix of either version, in one read. It holds the
// whole prefix or ends at the end of the file, so its bounds checks
// are the whole-file ones.
let prefix = Window::read(file, offset, V2_PREFIX_MAX)?;
prefix.ensure(0, 4)?;
if prefix.bytes[..4] == OHDR_SIGNATURE {
Self::parse_v2(file, offset, &prefix, offset_size, length_size)
} else {
Self::parse_v1(file, offset, &prefix, offset_size, length_size)
}
}
fn parse_v1<S: Storage + ?Sized>(
file: &S,
offset: u64,
prefix: &Window<'_>,
offset_size: u8,
length_size: u8,
) -> Result<ObjectHeader, FormatError> {
// version(1) + reserved(1) + num_messages(2) + ref_count(4) + header_size(4) = 12
// then pad to 8-byte alignment from start of header
prefix.ensure(0, 12)?;
let prefix = &prefix.bytes[..12];
let version = prefix[0];
if version != 1 {
return Err(FormatError::InvalidObjectHeaderVersion(version));
}
let num_messages = LittleEndian::read_u16(&prefix[2..4]) as usize;
let reference_count = LittleEndian::read_u32(&prefix[4..8]);
let header_data_size = LittleEndian::read_u32(&prefix[8..12]) as usize;
// libhdf5 (H5O__prefix_deserialize): a header with messages needs room
// for at least one message header, and one without has an empty chunk.
if (num_messages > 0 && header_data_size < V1_MSG_HEADER_SIZE)
|| (num_messages == 0 && header_data_size > 0)
{
return Err(FormatError::InvalidObjectHeader(
"bad object header chunk size",
));
}
// Pad to 8-byte alignment: header prefix is 12 bytes, pad to 16
let padding = 4; // pad 12-byte prefix to 16-byte alignment
let msg_start = offset
.checked_add(12 + padding)
.ok_or(FormatError::UnexpectedEof {
expected: usize::MAX,
available: len_usize(file),
})?;
// parse_v1_chunk reads the chunk, with the bounds check that was here.
// The prefix's count (NIL messages included, capped: it is untrusted)
// sizes the list once instead of growing it message by message.
let mut messages = Vec::with_capacity(num_messages.min(64));
let chunk0_count = Self::parse_v1_chunk(
file,
msg_start,
header_data_size,
offset_size,
length_size,
&mut messages,
)?;
// libhdf5 reads every message in the first chunk and refuses a header
// whose prefix claims fewer than that (continuation chunks are read
// later and not held to the count). Stopping after the claimed number
// silently dropped the rest.
if chunk0_count > num_messages {
return Err(FormatError::InvalidObjectHeader(
"bad object header message count",
));
}
Ok(ObjectHeader {
version: 1,
messages,
reference_count: Some(reference_count),
flags: 0,
access_time: None,
modification_time: None,
change_time: None,
birth_time: None,
})
}
/// Parse the messages of one version-1 chunk (`length` bytes at
/// `offset`, no signature), following continuation messages as they are
/// met. Returns how many messages (NIL ones included) this chunk itself
/// holds.
///
/// A version-1 chunk is filled with messages whose sizes are multiples of
/// 8; libhdf5 refuses a message that is not aligned, that runs past the
/// end of the chunk, or leftover bytes too few for a message header (a
/// "gap", which only version 2 allows).
///
/// Continuation chunks are read in the order their messages are found,
/// as `H5O_protect` loads them (so the messages keep libhdf5's order):
/// a queue of (address, length) pairs, each chunk read, parsed and
/// released before the next, so only one chunk buffer is alive at a
/// time whatever the storage. Every chunk must start at a new address
/// (else a cycle), and the chunks together may be no larger than the
/// file, so the bytes read stay within the file's size; a header of
/// more than [`MAX_V1_CHUNKS`] chunks is refused.
fn parse_v1_chunk<S: Storage + ?Sized>(
file: &S,
offset: u64,
length: usize,
offset_size: u8,
length_size: u8,
messages: &mut Vec<HeaderMessage>,
) -> Result<usize, FormatError> {
// The chunks found so far are also the queue of chunks to read.
let mut spans = ChunkSpans::new(file.len(), offset, length)?;
let mut chunk0_count = 0usize;
let mut next = 0usize;
let hints = file.as_contiguous().is_none();
while let Some((chunk_offset, chunk_length)) = spans.get(next) {
let chunk = read_exact_at(file, chunk_offset, chunk_length)?;
let known = spans.len;
let count =
Self::parse_v1_messages(&chunk, offset_size, length_size, messages, &mut spans)?;
// The continuation chunks this one names are read next.
if hints {
for i in known..spans.len {
if let Some((o, l)) = spans.get(i) {
file.hint(o, l);
}
}
}
// Only the first chunk's messages are held to the prefix count.
if next == 0 {
chunk0_count = count;
}
next += 1;
}
Ok(chunk0_count)
}
/// The messages of one version-1 chunk: each checked and appended to
/// `messages` (NIL ones dropped), each continuation added to `spans`.
/// Returns how many messages (NIL ones included) the chunk holds.
///
/// Inlined into the chunk loop: kept out of line (`#[inline(never)]`,
/// 4313917), the call cost `ObjectHeader::parse` about 2.5 ns per
/// header, 4% on small version-1 headers (A/B builds, 2026-09-27; see
/// `BENCHMARKS.md`). Without an attribute the compiler keeps it out of
/// line too.
#[inline]
fn parse_v1_messages(
data: &[u8],
offset_size: u8,
length_size: u8,
messages: &mut Vec<HeaderMessage>,
spans: &mut ChunkSpans,
) -> Result<usize, FormatError> {
let end = data.len();
let mut pos = 0usize;
let mut count = 0usize;
while pos < end {
if end - pos < V1_MSG_HEADER_SIZE {
return Err(FormatError::InvalidObjectHeader(
"gap found in early version of file format",
));
}
let msg_type_raw = LittleEndian::read_u16(&data[pos..pos + 2]);
let msg_data_size = LittleEndian::read_u16(&data[pos + 2..pos + 4]) as usize;
let msg_flags = data[pos + 4];
// reserved(3) at pos+5..pos+8
pos += V1_MSG_HEADER_SIZE;
if !msg_data_size.is_multiple_of(8) {
return Err(FormatError::InvalidObjectHeader("message not aligned"));
}
if msg_data_size > end - pos {
return Err(FormatError::InvalidObjectHeader(
"message size exceeds buffer end",
));
}
let body = &data[pos..pos + msg_data_size];
check_message(1, msg_type_raw, msg_flags, body, offset_size, length_size)?;
count += 1;
let msg_type = MessageType::from_u16(msg_type_raw);
if msg_type != MessageType::Nil {
messages.push(HeaderMessage {
msg_type,
size: msg_data_size,
flags: msg_flags,
creation_order: None,
data: body.to_vec(),
});
}
// Queue continuations (v1 continuation chunks are just raw
// messages, no signature); check_message has checked the body.
if msg_type == MessageType::ObjectHeaderContinuation {
let cont_offset = read_offset(body, 0, offset_size)?;
let cont_length = to_usize(read_offset(body, offset_size as usize, length_size)?)?;
spans.add(cont_offset, cont_length)?;
}
pos += msg_data_size;
}
Ok(count)
}
fn parse_v2<S: Storage + ?Sized>(
file: &S,
offset: u64,
prefix: &Window<'_>,
offset_size: u8,
length_size: u8,
) -> Result<ObjectHeader, FormatError> {
// `ensure_len` checks positions relative to the header against the
// prefix window and reports them as the whole-file check did, with
// absolute positions and the file's length.
let data: &[u8] = &prefix.bytes;
let file_len = len_usize(file);
let base = usize::try_from(offset).unwrap_or(usize::MAX);
let abs = |rel: usize| base.saturating_add(rel);
let ensure_len = |_: &[u8], rel: usize, needed: usize| prefix.ensure(rel, needed);
let offset = 0usize;
// signature(4) + version(1) + flags(1) = 6
ensure_len(data, offset, 6)?;
let version = data[offset + 4];
if version != 2 {
return Err(FormatError::InvalidObjectHeaderVersion(version));
}
let flags = data[offset + 5];
if flags & !V2_HDR_ALL_FLAGS != 0 {
return Err(FormatError::InvalidObjectHeader(
"unknown object header status flag(s)",
));
}
let mut pos = offset + 6;
// Optional timestamps (flags bit 5)
let (access_time, modification_time, change_time, birth_time) = if flags & 0x20 != 0 {
ensure_len(data, pos, 16)?;
let at = LittleEndian::read_u32(&data[pos..pos + 4]);
let mt = LittleEndian::read_u32(&data[pos + 4..pos + 8]);
let ct = LittleEndian::read_u32(&data[pos + 8..pos + 12]);
let bt = LittleEndian::read_u32(&data[pos + 12..pos + 16]);
pos += 16;
(Some(at), Some(mt), Some(ct), Some(bt))
} else {
(None, None, None, None)
};
// Optional attribute storage thresholds (flags bit 4)
if flags & 0x10 != 0 {
ensure_len(data, pos, 4)?;
// max_compact_attrs(2) + min_dense_attrs(2) — checked, not stored
let max_compact = LittleEndian::read_u16(&data[pos..pos + 2]);
let min_dense = LittleEndian::read_u16(&data[pos + 2..pos + 4]);
if max_compact < min_dense {
return Err(FormatError::InvalidObjectHeader(
"bad object header attribute phase change values",
));
}
pos += 4;
}
// chunk0 size: width depends on flags bits 0-1
let chunk_size_width = match flags & 0x03 {
0 => 1u8,
1 => 2,
2 => 4,
3 => 8,
_ => unreachable!(),
};
ensure_len(data, pos, chunk_size_width as usize)?;
let chunk0_size = to_usize(read_offset(data, pos, chunk_size_width)?)?;
pos += chunk_size_width as usize;
// Bit 2: attribute creation order tracked → messages include creation order field
let has_creation_order = flags & 0x04 != 0;
let msg_header_size = if has_creation_order { 6 } else { 4 };
if chunk0_size > 0 && chunk0_size < msg_header_size {
return Err(FormatError::InvalidObjectHeader(
"bad object header chunk size",
));
}
let chunk0_msg_start = pos;
let Some(chunk0_abs_end) = abs(pos).checked_add(chunk0_size) else {
return Err(FormatError::UnexpectedEof {
expected: usize::MAX,
available: file_len,
});
};
let chunk0_msg_end = chunk0_abs_end - base;
// The whole first chunk, prefix to checksum, in one read (its
// bounds check is the one on the checksum's 4 bytes).
let chunk0 = read_exact_at(file, base as u64, chunk0_msg_end.saturating_add(4))?;
let data: &[u8] = &chunk0;
// Validate checksum: from OHDR signature through all messages (before checksum)
#[cfg(feature = "checksum")]
{
let stored = LittleEndian::read_u32(&data[chunk0_msg_end..chunk0_msg_end + 4]);
let computed = crate::checksum::jenkins_lookup3(&data[offset..chunk0_msg_end]);
if computed != stored {
return Err(FormatError::ChecksumMismatch {
expected: stored,
computed,
});
}
}
// Parse messages from chunk0
let mut messages = Vec::new();
let mut continuations = Vec::new();
Self::parse_v2_messages(
data,
chunk0_msg_start,
chunk0_msg_end,
has_creation_order,
offset_size,
length_size,
&mut messages,
&mut continuations,
)?;
// Follow continuations, one chunk buffer at a time. A chunk address
// seen twice is a cycle in malformed data, and the chunks may add up
// to no more than the file; a valid header can have many chunks (libhdf5 adds one
// whenever a message no longer fits), up to the same bound as a
// version-1 header.
let mut spans = ChunkSpans::new(file.len(), base as u64, chunk0_msg_end.saturating_add(4))?;
// The continuation chunks a chunk names are read next (see
// `Storage::hint`).
let hints = file.as_contiguous().is_none();
if hints {
for &(o, l) in &continuations {
file.hint(o as u64, l);
}
}
while let Some((cont_offset, cont_length)) = continuations.pop() {
spans.add(cont_offset as u64, cont_length)?;
let known = continuations.len();
Self::parse_v2_continuation(
file,
cont_offset as u64,
cont_length,
has_creation_order,
offset_size,
length_size,
&mut messages,
&mut continuations,
)?;
if hints {
for &(o, l) in &continuations[known..] {
file.hint(o as u64, l);
}
}
}
Ok(ObjectHeader {
version: 2,
messages,
reference_count: None,
flags,
access_time,
modification_time,
change_time,
birth_time,
})
}
#[allow(clippy::too_many_arguments)]
fn parse_v2_messages(
data: &[u8],
start: usize,
end: usize,
has_creation_order: bool,
offset_size: u8,
length_size: u8,
messages: &mut Vec<HeaderMessage>,
continuations: &mut Vec<(usize, usize)>,
) -> Result<(), FormatError> {
let msg_header_size = if has_creation_order { 6 } else { 4 };
let mut pos = start;
let mut null_count = 0usize;
while pos < end {
// Leftover bytes too few for a message header are a gap, which
// libhdf5 allows only in a chunk without NIL messages (a writer
// that leaves a gap had no NIL message to put the space in).
if end - pos < msg_header_size {
if null_count != 0 {
return Err(FormatError::InvalidObjectHeader(
"gap in chunk with no null messages",
));
}
break;
}
let msg_type_raw = data[pos] as u16;
let msg_data_size = LittleEndian::read_u16(&data[pos + 1..pos + 3]) as usize;
let msg_flags = data[pos + 3];
let creation_order = if has_creation_order {
Some(LittleEndian::read_u16(&data[pos + 4..pos + 6]))
} else {
None
};
pos += msg_header_size;
// `end` is where the messages stop and the checksum starts.
// libhdf5 bounds a message by the chunk including its checksum,
// but a message that runs into the checksum still fails there:
// its loop stops at the checksum, and reading the checksum from
// past its start overruns the chunk ("ran off end of input
// buffer while decoding"). Both refuse it; only the text
// differs.
if msg_data_size > end - pos {
return Err(FormatError::InvalidObjectHeader(
"message size exceeds buffer end",
));
}
let body = &data[pos..pos + msg_data_size];
check_message(2, msg_type_raw, msg_flags, body, offset_size, length_size)?;
let msg_type = MessageType::from_u16(msg_type_raw);
if msg_type == MessageType::ObjectHeaderContinuation {
// check_message has checked the body holds both fields.
let cont_off = to_usize(read_offset(body, 0, offset_size)?)?;
let cont_len = to_usize(read_offset(body, offset_size as usize, length_size)?)?;
continuations.push((cont_off, cont_len));
} else if msg_type == MessageType::Nil {
null_count += 1;
} else {
messages.push(HeaderMessage {
msg_type,
size: msg_data_size,
flags: msg_flags,
creation_order,
data: body.to_vec(),
});
}
pos += msg_data_size;
}
Ok(())
}
#[allow(clippy::too_many_arguments)]
fn parse_v2_continuation<S: Storage + ?Sized>(
file: &S,
offset: u64,
length: usize,
has_creation_order: bool,
offset_size: u8,
length_size: u8,
messages: &mut Vec<HeaderMessage>,
continuations: &mut Vec<(usize, usize)>,
) -> Result<(), FormatError> {
// OCHK signature(4) + messages + checksum(4)
let chunk = read_exact_at(file, offset, length)?;
let data: &[u8] = &chunk;
let offset = 0usize;
if length < 8 {
return Err(FormatError::UnexpectedEof {
expected: 8,
available: length,
});
}
ensure_len(data, offset, 4)?;
if data[offset..offset + 4] != OCHK_SIGNATURE {
return Err(FormatError::InvalidObjectHeaderSignature);
}
let msg_start = offset + 4;
let checksum_pos = offset + length - 4;
#[cfg(feature = "checksum")]
{
let stored = LittleEndian::read_u32(&data[checksum_pos..checksum_pos + 4]);
let computed = crate::checksum::jenkins_lookup3(&data[offset..checksum_pos]);
if computed != stored {
return Err(FormatError::ChecksumMismatch {
expected: stored,
computed,
});
}
}
Self::parse_v2_messages(
data,
msg_start,
checksum_pos,
has_creation_order,
offset_size,
length_size,
messages,
continuations,
)
}
}
/// What an object header is hinted to take before its prefix is read (see
/// [`Storage::hint`]): the first chunk of a typical dataset's header. A
/// longer header is read all the same.
pub(crate) const OBJECT_HEADER_HINT_LEN: usize = 512;
/// Longest version-2 object header prefix: signature(4) + version(1) +
/// flags(1) + times(16) + attribute phase change(4) + chunk-0 size(8).
const V2_PREFIX_MAX: usize = 34;
/// Size of a version-1 message header: type(2) + size(2) + flags(1) + reserved(3).
const V1_MSG_HEADER_SIZE: usize = 8;
/// The chunks of one object header read so far, in the order they were
/// found (which is the order version-1 chunks are read in). A chunk starting
/// where another did is a cycle. Chunks of a valid header do not overlap, so
/// together they are no larger than the file; a header whose chunks add up
/// to more is refused, which bounds what its chunks can make a reader read
/// (a crafted chain of chunks each nested in the last would otherwise read
/// the file over and over). Overlap itself is not refused: libhdf5 reads
/// such headers (`cve-2025-7067.h5` has one).
///
/// Almost every header has at most a few chunks, and this runs once per
/// header, so the first [`INLINE_CHUNKS`] live in an inline array and are
/// checked for cycles by a scan; only a longer header allocates (the rest
/// of the list, and a set of starts). Allocating a queue and a set for
/// every header made parsing 401 small headers 1.8x slower.
struct ChunkSpans {
inline: [(u64, usize); INLINE_CHUNKS],
/// Chunks after the first [`INLINE_CHUNKS`], and every chunk start.
spill: Option<Box<SpilledSpans>>,
/// How many chunks there are.
len: usize,
/// Bytes of the chunks so far, and the most they may add up to.
total: u64,
budget: u64,
}
/// The chunks of a [`ChunkSpans`] beyond its inline ones.
struct SpilledSpans {
chunks: Vec<(u64, usize)>,
starts: BTreeSet<u64>,
}
/// How many chunks [`ChunkSpans`] holds without allocating.
const INLINE_CHUNKS: usize = 8;
impl ChunkSpans {
#[inline]
fn new(file_len: u64, start: u64, len: usize) -> Result<Self, FormatError> {
let mut s = Self {
inline: [(0, 0); INLINE_CHUNKS],
spill: None,
len: 0,
total: 0,
budget: file_len,
};
s.add(start, len)?;
Ok(s)
}
/// Record the chunk `len` bytes at `start`.
#[inline]
fn add(&mut self, start: u64, len: usize) -> Result<(), FormatError> {
self.total = self.total.saturating_add(len as u64);
if self.len < INLINE_CHUNKS {
if self.inline[..self.len].iter().any(|&(s, _)| s == start) {
return Err(FormatError::NestingDepthExceeded);
}
self.inline[self.len] = (start, len);
} else {
self.add_spilled(start, len)?;
}
self.len += 1;
if self.total > self.budget {
return Err(FormatError::InvalidObjectHeader(
"object header chunks larger than the file",
));
}
Ok(())
}
#[cold]
#[inline(never)]
fn add_spilled(&mut self, start: u64, len: usize) -> Result<(), FormatError> {
let inline = &self.inline;
let spill = self.spill.get_or_insert_with(|| {
Box::new(SpilledSpans {
chunks: Vec::new(),
starts: inline.iter().map(|&(s, _)| s).collect(),
})
});
if !spill.starts.insert(start) || self.len >= MAX_V1_CHUNKS {
return Err(FormatError::NestingDepthExceeded);
}
spill.chunks.push((start, len));
Ok(())
}
/// The `i`th chunk recorded.
#[inline]
fn get(&self, i: usize) -> Option<(u64, usize)> {
if i < INLINE_CHUNKS {
(i < self.len).then(|| self.inline[i])
} else {
self.spill.as_ref()?.chunks.get(i - INLINE_CHUNKS).copied()
}
}
}
/// Most chunks a version-1 object header may have (malformed-data guard;
/// libhdf5 has no limit, and a header that gains one continuation chunk per
/// attribute added can have many).
const MAX_V1_CHUNKS: usize = 1 << 16;
/// Every defined version-2 object header status flag (libhdf5
/// `H5O_HDR_ALL_FLAGS`): chunk-0 size width (bits 0-1), attribute creation
/// order tracked/indexed, attribute phase-change values, times stored.
const V2_HDR_ALL_FLAGS: u8 = 0x3F;
// Header message flag bits (libhdf5 `H5O_MSG_FLAG_*`). Bit 0 (constant) needs
// no check. Bit 3 (fail if unknown and the file is opened for writing) never
// fails a read: the parser only ever reads, as libhdf5 ignores it for a
// read-only open.
const MSG_FLAG_SHARED: u8 = 0x02;
const MSG_FLAG_DONTSHARE: u8 = 0x04;
const MSG_FLAG_FAIL_IF_UNKNOWN_AND_OPEN_FOR_WRITE: u8 = 0x08;
const MSG_FLAG_MARK_IF_UNKNOWN: u8 = 0x10;
const MSG_FLAG_WAS_UNKNOWN: u8 = 0x20;
const MSG_FLAG_SHAREABLE: u8 = 0x40;
/// Fail if the message is unknown, whatever the access mode.
const MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS: u8 = 0x80;
/// Message type ids libhdf5 has a class for (`H5O_msg_class_g`): 0x00-0x18
/// except 0x09 (a test-only "bogus" message). Anything else is an unknown
/// message.
fn is_known_message(id: u16) -> bool {
id <= 0x18 && id != 0x09
}
/// Message classes that may be shared (`H5O_SHARE_IS_SHARABLE`): dataspace,
/// datatype, the two fill-value messages, filter pipeline and attribute.
fn is_shareable_message(id: u16) -> bool {
matches!(id, 0x01 | 0x03 | 0x04 | 0x05 | 0x0B | 0x0C)
}
/// Check one header message the way libhdf5 does while it loads an object
/// header (`H5O__chunk_deserialize`), so an object libhdf5 refuses to open is
/// refused here too instead of being read from a corrupt header:
///
/// - contradictory flag combinations;
/// - an unknown message the file says no reader may skip (bit 7). This had
/// bits 3 and 7 the wrong way round once, failing objects libhdf5 reads
/// and reading ones it refuses (`tbogus.h5`);
/// - a known message whose class cannot be shared, flagged shared or
/// shareable (`cve-2016-4332`);
/// - the messages libhdf5 decodes while loading the header, whose decode
/// errors fail the load: continuation, reference count (which a version-1
/// header cannot hold), and both modification-time messages.
fn check_message(
header_version: u8,
id: u16,
flags: u8,
body: &[u8],
offset_size: u8,
length_size: u8,
) -> Result<(), FormatError> {
let bad_flags = FormatError::InvalidObjectHeader("bad flag combination for message");
if flags & MSG_FLAG_SHARED != 0 && flags & MSG_FLAG_DONTSHARE != 0 {
return Err(bad_flags);
}
if flags & MSG_FLAG_WAS_UNKNOWN != 0
&& (flags & MSG_FLAG_FAIL_IF_UNKNOWN_AND_OPEN_FOR_WRITE != 0
|| flags & MSG_FLAG_MARK_IF_UNKNOWN == 0)
{
return Err(bad_flags);
}
if !is_known_message(id) {
if flags & MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS != 0 {
return Err(FormatError::UnsupportedMessage(id));
}
return Ok(());
}
if flags & (MSG_FLAG_SHARED | MSG_FLAG_SHAREABLE) != 0 && !is_shareable_message(id) {
return Err(FormatError::InvalidObjectHeader(
"message of unshareable class flagged as shareable",
));
}
let overrun = FormatError::InvalidObjectHeader("ran off end of input buffer while decoding");
match id {
// Continuation: address + length, and the chunk cannot be empty.
0x10 => {
if body.len() < offset_size as usize + length_size as usize {
return Err(overrun);
}
if read_offset(body, offset_size as usize, length_size)? == 0 {
return Err(FormatError::InvalidObjectHeader(
"invalid continuation chunk size (0)",
));
}
}
// Reference count: version-2 headers only; version 0 then a u32.
0x16 => {
if header_version == 1 {
return Err(FormatError::InvalidObjectHeader(
"object header version does not support reference count message",
));
}
match body.first() {
None => return Err(overrun),
Some(0) => {}
Some(_) => {
return Err(FormatError::InvalidObjectHeader(
"bad version number for reference count message",
));
}
}
if body.len() < 5 {
return Err(overrun);
}
}
// Old modification time: "YYYYMMDDhhmmss" and 2 reserved bytes.
0x0E => {
if body.len() < 16 {
return Err(overrun);
}
if !body[..14].iter().all(u8::is_ascii_digit) {
return Err(FormatError::InvalidObjectHeader(
"badly formatted modification time message",
));
}
}
// New modification time: version 1, 3 reserved bytes, u32 seconds.
0x12 => {
match body.first() {
None => return Err(overrun),
Some(1) => {}
Some(_) => {
return Err(FormatError::InvalidObjectHeader(
"bad version number for mtime message",
));
}
}
if body.len() < 8 {
return Err(overrun);
}
}
_ => {}
}
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
fn header_with(types: &[MessageType]) -> ObjectHeader {
ObjectHeader {
version: 2,
messages: types
.iter()
.map(|&msg_type| HeaderMessage {
msg_type,
size: 0,
flags: 0,
creation_order: None,
data: Vec::new(),
})
.collect(),
reference_count: None,
flags: 0,
access_time: None,
modification_time: None,
change_time: None,
birth_time: None,
}
}
#[test]
fn object_class_follows_libhdf5() {
use MessageType::*;
let class = |t: &[MessageType]| header_with(t).object_class();
assert_eq!(
class(&[Datatype, Dataspace, DataLayout]),
Some(ObjectClass::Dataset)
);
// A Data Layout message does not make a dataset without a dataspace
// (cve-2024-33874 `/Dset1`: h5py opens it as a named datatype).
assert_eq!(
class(&[Datatype, DataLayout]),
Some(ObjectClass::NamedDatatype)
);
assert_eq!(class(&[Datatype]), Some(ObjectClass::NamedDatatype));
// Group messages win over dataset messages.
assert_eq!(
class(&[Datatype, Dataspace, SymbolTable]),
Some(ObjectClass::Group)
);
assert_eq!(class(&[LinkInfo]), Some(ObjectClass::Group));
// Link messages alone are not a group; nothing is not an object.
assert_eq!(class(&[Link]), None);
assert_eq!(class(&[]), None);
}
// Helper: build a v1 object header with given messages
fn build_v1_header(
messages: &[(u16, &[u8], u8)], // (type, data, flags)
offset_size: u8,
length_size: u8,
) -> Vec<u8> {
let _ = (offset_size, length_size);
// Calculate total header message data size
let mut msg_bytes = Vec::new();
for (mtype, mdata, mflags) in messages {
// v1 message sizes are multiples of 8 (the data is zero-padded).
let padded = <[u8]>::len(mdata).div_ceil(8) * 8;
msg_bytes.extend_from_slice(&mtype.to_le_bytes()); // type(2)
msg_bytes.extend_from_slice(&(padded as u16).to_le_bytes()); // size(2)
msg_bytes.push(*mflags); // flags(1)
msg_bytes.extend_from_slice(&[0u8; 3]); // reserved(3)
msg_bytes.extend_from_slice(mdata); // data
msg_bytes.resize(msg_bytes.len() + padded - <[u8]>::len(mdata), 0);
}
let mut buf = Vec::new();
buf.push(1); // version
buf.push(0); // reserved
buf.extend_from_slice(&(messages.len() as u16).to_le_bytes()); // num_messages
buf.extend_from_slice(&1u32.to_le_bytes()); // reference_count
buf.extend_from_slice(&(msg_bytes.len() as u32).to_le_bytes()); // header_data_size
// Pad to 8-byte alignment (12 bytes so far, pad 4)
buf.extend_from_slice(&[0u8; 4]);
buf.extend_from_slice(&msg_bytes);
buf
}
// Helper: build a v2 object header chunk0 with given messages
fn build_v2_header(
flags: u8,
messages: &[(u8, &[u8], u8)], // (type, data, msg_flags)
timestamps: Option<(u32, u32, u32, u32)>,
) -> Vec<u8> {
let has_creation_order = flags & 0x04 != 0;
let has_timestamps = flags & 0x20 != 0;
let mut buf = Vec::new();
buf.extend_from_slice(&OHDR_SIGNATURE); // 4
buf.push(2); // version
buf.push(flags);
if has_timestamps && let Some((at, mt, ct, bt)) = timestamps {
buf.extend_from_slice(&at.to_le_bytes());
buf.extend_from_slice(&mt.to_le_bytes());
buf.extend_from_slice(&ct.to_le_bytes());
buf.extend_from_slice(&bt.to_le_bytes());
}
if flags & 0x10 != 0 {
buf.extend_from_slice(&8u16.to_le_bytes()); // max_compact
buf.extend_from_slice(&6u16.to_le_bytes()); // min_dense
}
// Build message bytes to get chunk size
let mut msg_bytes = Vec::new();
for (mtype, mdata, mflags) in messages {
msg_bytes.push(*mtype); // type(1)
msg_bytes.extend_from_slice(&(mdata.len() as u16).to_le_bytes()); // size(2)
msg_bytes.push(*mflags); // flags(1)
if has_creation_order {
msg_bytes.extend_from_slice(&0u16.to_le_bytes()); // creation_order(2)
}
msg_bytes.extend_from_slice(mdata);
}
let chunk_size = msg_bytes.len();
// Write chunk size based on flags bits 0-1
match flags & 0x03 {
0 => buf.push(chunk_size as u8),
1 => buf.extend_from_slice(&(chunk_size as u16).to_le_bytes()),
2 => buf.extend_from_slice(&(chunk_size as u32).to_le_bytes()),
3 => buf.extend_from_slice(&(chunk_size as u64).to_le_bytes()),
_ => unreachable!(),
}
buf.extend_from_slice(&msg_bytes);
// Checksum (CRC32C of everything from OHDR to here)
let checksum = crate::checksum::jenkins_lookup3(&buf);
buf.extend_from_slice(&checksum.to_le_bytes());
buf
}
#[test]
fn parse_v1_zero_messages() {
let data = build_v1_header(&[], 8, 8);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.version, 1);
assert_eq!(hdr.messages.len(), 0);
assert_eq!(hdr.reference_count, Some(1));
assert_eq!(hdr.flags, 0);
}
#[test]
fn parse_v1_two_messages() {
let messages = [
(0x0001u16, &[1u8, 2, 3, 4][..], 0u8), // Dataspace
(0x0008, &[5u8, 6][..], 0), // DataLayout
];
let data = build_v1_header(&messages, 8, 8);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 2);
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
// v1 message data is padded to a multiple of 8 bytes.
assert_eq!(hdr.messages[0].data, vec![1, 2, 3, 4, 0, 0, 0, 0]);
assert_eq!(hdr.messages[1].msg_type, MessageType::DataLayout);
assert_eq!(hdr.messages[1].data[..2], [5, 6]);
}
/// A version-1 header whose continuation chunks form a chain: chunk k
/// holds a Dataspace message `[k]` and the continuation to chunk k + 1.
/// With `cycle`, the last chunk points back at the first continuation
/// chunk.
fn v1_chain(n: usize, cycle: bool) -> Vec<u8> {
// Each continuation chunk: dataspace (8 + 8) + continuation (8 + 16).
let chunk_len = 40u64;
let first = 64u64;
let cont = |addr: u64| {
let mut b = addr.to_le_bytes().to_vec();
b.extend_from_slice(&chunk_len.to_le_bytes());
b
};
let mut data = build_v1_header(&[(0x0010, &cont(first)[..], 0)], 8, 8);
data.resize(first as usize, 0);
for k in 0..n {
let mut c = Vec::new();
c.extend_from_slice(&1u16.to_le_bytes());
c.extend_from_slice(&8u16.to_le_bytes());
c.extend_from_slice(&[0; 4]);
c.extend_from_slice(&(k as u64).to_le_bytes());
let next = if k + 1 < n {
first + (k as u64 + 1) * chunk_len
} else if cycle {
first
} else {
// The last chunk ends in a NIL message instead.
c.extend_from_slice(&[0, 0, 16, 0, 0, 0, 0, 0]);
c.extend_from_slice(&[0; 16]);
data.extend_from_slice(&c);
continue;
};
c.extend_from_slice(&0x10u16.to_le_bytes());
c.extend_from_slice(&16u16.to_le_bytes());
c.extend_from_slice(&[0; 4]);
c.extend_from_slice(&cont(next));
data.extend_from_slice(&c);
}
data
}
/// libhdf5 reads any chain of continuation chunks (a header grows one
/// per attribute added when full); the reader used to stop at 32.
#[test]
fn long_v1_continuation_chains_are_read() {
let data = v1_chain(200, false);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
let spaces: Vec<u8> = hdr
.messages
.iter()
.filter(|m| m.msg_type == MessageType::Dataspace)
.map(|m| m.data[0])
.collect();
assert_eq!(spaces, (0..200).map(|k| k as u8).collect::<Vec<_>>());
}
/// A crafted version-1 header whose continuation chunks nest: each
/// chunk's continuation message points at the rest of that chunk. Read
/// depth-first with every enclosing chunk kept alive, from storage that
/// hands out owned buffers, it read n^2 bytes and held them all at once
/// (a 192 KB file read 768 MB). Chunks adding up to more than the file
/// are refused, and the bytes read stay within the file's size.
#[test]
fn nested_v1_continuation_chunks_are_bounded() {
use crate::storage::CountingStorage;
let n = 2000u64;
let a = 64u64;
let cont = |addr: u64, len: u64| {
let mut m = vec![0x10, 0, 16, 0, 0, 0, 0, 0];
m.extend_from_slice(&addr.to_le_bytes());
m.extend_from_slice(&len.to_le_bytes());
m
};
// Prefix: version 1, one message, reference count 1, 24 bytes.
let mut buf = vec![1, 0, 1, 0, 1, 0, 0, 0, 24, 0, 0, 0, 0, 0, 0, 0];
buf.extend_from_slice(&cont(a, 24 * n));
buf.resize(a as usize, 0);
for k in 0..n {
if k + 1 < n {
buf.extend_from_slice(&cont(a + 24 * (k + 1), 24 * (n - k - 1)));
} else {
buf.extend_from_slice(&[0, 0, 16, 0, 0, 0, 0, 0]);
buf.extend_from_slice(&[0; 16]);
}
}
let len = buf.len() as u64;
let s = CountingStorage::new(buf);
assert!(matches!(
ObjectHeader::parse_in(&s, 0, 8, 8),
Err(FormatError::InvalidObjectHeader(
"object header chunks larger than the file"
))
));
assert!(
s.bytes_read() <= 2 * len,
"read {} of {len}",
s.bytes_read()
);
}
/// libhdf5 reads a continuation chunk that overlaps the chunk holding
/// its message (`cve-2025-7067.h5` has one), and so does this reader.
#[test]
fn overlapping_v1_continuation_chunk_is_read() {
// Chunk 0 (at 16): continuation (24 bytes), then a NIL message at
// 40; the continuation chunk is that NIL message's 8-byte header.
let mut cont = 40u64.to_le_bytes().to_vec();
cont.extend_from_slice(&8u64.to_le_bytes());
let data = build_v1_header(&[(0x0010, &cont[..], 0), (0x0000, &[][..], 0)], 8, 8);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 1);
}
/// A valid chain over owned-buffer storage reads each chunk once.
#[test]
fn long_v1_chain_reads_each_chunk_once() {
use crate::storage::CountingStorage;
let data = v1_chain(3000, false);
let len = data.len() as u64;
let s = CountingStorage::new(data);
let hdr = ObjectHeader::parse_in(&s, 0, 8, 8).unwrap();
assert_eq!(
hdr.messages
.iter()
.filter(|m| m.msg_type == MessageType::Dataspace)
.count(),
3000
);
assert!(s.bytes_read() <= len, "read {} of {len}", s.bytes_read());
}
/// Continuation chunks are read in the order their messages are found
/// (libhdf5's `H5O_protect`), so a chunk's messages follow every
/// message of the chunk before, not the continuation message.
#[test]
fn v1_continuation_messages_keep_libhdf5_order() {
// Chunk 0: continuation to A, dataspace [1]; A: dataspace [2].
let a = 64u64;
let mut cont = a.to_le_bytes().to_vec();
cont.extend_from_slice(&16u64.to_le_bytes());
let mut data = build_v1_header(&[(0x0010, &cont[..], 0), (0x0001, &[1; 8][..], 0)], 8, 8);
data.resize(a as usize, 0);
data.extend_from_slice(&[1, 0, 8, 0, 0, 0, 0, 0]);
data.extend_from_slice(&[2; 8]);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
let spaces: Vec<u8> = hdr
.messages
.iter()
.filter(|m| m.msg_type == MessageType::Dataspace)
.map(|m| m.data[0])
.collect();
assert_eq!(spaces, [1, 2]);
}
#[test]
fn v1_continuation_cycles_are_refused() {
// Within the inline chunk list, and past it (the cycle returns to
// an inline chunk once the list has spilled).
for n in [5, 7, 8, 9, 40] {
let data = v1_chain(n, true);
assert!(
matches!(
ObjectHeader::parse(&data, 0, 8, 8),
Err(FormatError::NestingDepthExceeded)
),
"{n} chunks"
);
}
}
#[test]
fn parse_v1_unknown_message_ok() {
let messages = [(0x00FFu16, &[0xAA, 0xBB][..], 0u8)];
let data = build_v1_header(&messages, 8, 8);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 1);
assert_eq!(hdr.messages[0].msg_type, MessageType::Unknown(0x00FF));
}
#[test]
fn parse_v1_unknown_fail_always_errors() {
// Bit 7 of msg_flags = fail if unknown, whatever the access mode.
let messages = [(0x00FFu16, &[0xAA][..], 0x80u8)];
let data = build_v1_header(&messages, 8, 8);
let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err();
assert_eq!(err, FormatError::UnsupportedMessage(0x00FF));
}
#[test]
fn parse_v1_unknown_fail_on_write_is_ignored_when_reading() {
// Bit 3 = fail if unknown *and the file is opened for writing*. This
// parser only reads, so libhdf5 (read-only) opens such an object and
// so must we. Bits 4/5 (mark if unknown / was unknown) never fail.
for flags in [0x08u8, 0x10, 0x30] {
let messages = [(0x00FFu16, &[0xAA][..], flags)];
let data = build_v1_header(&messages, 8, 8);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages[0].msg_type, MessageType::Unknown(0x00FF));
}
}
#[test]
fn contradictory_message_flags_are_refused() {
// libhdf5: "bad flag combination for message" for shared + don't
// share, was-unknown without mark-if-unknown, and was-unknown with
// fail-if-unknown-on-write.
for flags in [0x06u8, 0x20, 0x38] {
let data = build_v1_header(&[(0x00FFu16, &[0xAA][..], flags)], 8, 8);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("bad flag combination for message"),
"flags {flags:#x}"
);
let data = build_v2_header(0x00, &[(0xF0, &[1, 2], flags)], None);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("bad flag combination for message"),
"flags {flags:#x}"
);
}
}
#[test]
fn unshareable_message_flagged_shareable_is_refused() {
// A layout (0x08) or modification time (0x12) message cannot be
// shared; bit 1 (shared) or bit 6 (shareable) on one is corruption
// (cve-2016-4332). A datatype (0x03) may be shareable.
let mtime = [1u8, 0, 0, 0, 0x10, 0x20, 0x30, 0x40];
for (id, flags) in [(0x08u16, 0x40u8), (0x08, 0x02), (0x12, 0x40)] {
let data = build_v1_header(&[(id, &mtime[..], flags)], 8, 8);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader(
"message of unshareable class flagged as shareable"
),
"id {id:#x} flags {flags:#x}"
);
}
let data = build_v1_header(&[(0x03, &[0u8; 8][..], 0x40)], 8, 8);
assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok());
// An unknown message is never checked for shareability.
let data = build_v1_header(&[(0x00FF, &[0u8; 8][..], 0x40)], 8, 8);
assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok());
}
#[test]
fn v1_message_must_be_aligned() {
// cve-2018-13873: a v1 message whose size is not a multiple of 8.
let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8);
data[16 + 2] = 7; // size field of the only message
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("message not aligned")
);
}
#[test]
fn message_overrunning_its_chunk_is_refused() {
// It used to end the chunk quietly, dropping this message and any
// after it.
let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8);
data[16 + 2] = 16;
data.resize(data.len() + 64, 0);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("message size exceeds buffer end")
);
let mut data = build_v2_header(0x00, &[(0x01, &[1, 2], 0)], None);
data[7 + 1] = 9; // size of the only message (after OHDR, ver, flags, chunk size)
let chk = crate::checksum::jenkins_lookup3(&data[..data.len() - 4]);
let n = data.len();
data[n - 4..].copy_from_slice(&chk.to_le_bytes());
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("message size exceeds buffer end")
);
}
#[test]
fn v1_gap_after_last_message_is_refused() {
// Fewer than 8 bytes left over: a gap, which only version 2 allows.
let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8);
data[8] += 4; // header_data_size
data.extend_from_slice(&[0u8; 4]);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("gap found in early version of file format")
);
}
#[test]
fn v1_chunk_holding_more_messages_than_the_prefix_says_is_refused() {
// cve-2024-32619: the prefix says 1 message, the chunk holds 2. The
// second used to be dropped silently.
let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0), (0x03, &[0u8; 8][..], 0)], 8, 8);
data[2] = 1;
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("bad object header message count")
);
// Fewer in the chunk than the prefix says is fine (the rest may be in
// continuation chunks; libhdf5 only enforces that with strict checks).
data[2] = 3;
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap().messages.len(),
2
);
}
#[test]
fn v1_prefix_chunk_size_must_fit_the_message_count() {
let mut data = build_v1_header(&[], 8, 8);
data[8] = 8; // no messages but a non-empty chunk
data.extend_from_slice(&[0u8; 8]);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("bad object header chunk size")
);
}
#[test]
fn v1_header_cannot_hold_a_reference_count_message() {
// cve-2018-11204.
let data = build_v1_header(&[(0x16, &[0, 2, 0, 0, 0][..], 0)], 8, 8);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader(
"object header version does not support reference count message"
)
);
let data = build_v2_header(0x00, &[(0x16, &[0, 2, 0, 0, 0], 0)], None);
assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok());
}
#[test]
fn modification_time_messages_are_decoded_with_the_header() {
// cve-2024-33873 (version 0) and cve-2024-33874 (empty message).
for (body, why) in [
(
&[0u8, 0, 0, 0, 1, 2, 3, 4][..],
"bad version number for mtime message",
),
(&[][..], "ran off end of input buffer while decoding"),
] {
let data = build_v2_header(0x00, &[(0x12, body, 0)], None);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader(why)
);
}
let data = build_v2_header(0x00, &[(0x12, &[1, 0, 0, 0, 1, 2, 3, 4], 0)], None);
assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok());
// The old (0x0E) message is 14 ASCII digits and 2 reserved bytes.
let data = build_v1_header(&[(0x0E, &b"20110414214255\0\0"[..], 0)], 8, 8);
assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok());
let data = build_v1_header(&[(0x0E, &b"2011041421425x\0\0"[..], 0)], 8, 8);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("badly formatted modification time message")
);
}
#[test]
fn continuation_message_must_hold_a_nonempty_chunk() {
let mut cont = [0u8; 16];
cont[..8].copy_from_slice(&64u64.to_le_bytes());
let data = build_v2_header(0x00, &[(0x10, &cont, 0)], None);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("invalid continuation chunk size (0)")
);
let data = build_v2_header(0x00, &[(0x10, &cont[..8], 0)], None);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("ran off end of input buffer while decoding")
);
}
#[test]
fn v2_prefix_is_checked() {
let data = build_v2_header(0x40, &[(0x01, &[1], 0)], None);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("unknown object header status flag(s)")
);
// build_v2_header writes max_compact 8, min_dense 6; swap them.
let mut data = build_v2_header(0x10, &[(0x01, &[1], 0)], None);
data[6] = 6;
data[8] = 8;
let chk = crate::checksum::jenkins_lookup3(&data[..data.len() - 4]);
let n = data.len();
data[n - 4..].copy_from_slice(&chk.to_le_bytes());
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("bad object header attribute phase change values")
);
}
#[test]
fn v2_gap_is_allowed_only_without_nil_messages() {
// Three bytes after the last message: a gap (a message header is 4).
let mut data = build_v2_header(0x00, &[(0x01, &[1, 2, 3], 0), (0x03, &[], 0)], None);
// Turn the empty datatype message (4 header bytes) into a 3-byte gap
// by shrinking the chunk.
let n = data.len();
data.truncate(n - 5);
data[6] -= 1;
let chk = crate::checksum::jenkins_lookup3(&data);
data.extend_from_slice(&chk.to_le_bytes());
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap().messages.len(),
1
);
let mut data = build_v2_header(0x00, &[(0x00, &[1, 2, 3], 0), (0x03, &[], 0)], None);
let n = data.len();
data.truncate(n - 5);
data[6] -= 1;
let chk = crate::checksum::jenkins_lookup3(&data);
data.extend_from_slice(&chk.to_le_bytes());
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::InvalidObjectHeader("gap in chunk with no null messages")
);
}
#[test]
fn parse_v2_unknown_message_flags() {
let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x08)], None);
assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok());
let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x80)], None);
assert_eq!(
ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(),
FormatError::UnsupportedMessage(0xF0)
);
}
#[test]
fn parse_v2_no_timestamps_one_message() {
let data = build_v2_header(0x00, &[(0x01, &[10, 20], 0)], None);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.version, 2);
assert_eq!(hdr.flags, 0);
assert_eq!(hdr.messages.len(), 1);
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
assert_eq!(hdr.messages[0].data, vec![10, 20]);
assert!(hdr.access_time.is_none());
}
#[test]
fn parse_v2_with_timestamps() {
let data = build_v2_header(0x20, &[(0x01, &[1], 0)], Some((100, 200, 300, 400)));
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.access_time, Some(100));
assert_eq!(hdr.modification_time, Some(200));
assert_eq!(hdr.change_time, Some(300));
assert_eq!(hdr.birth_time, Some(400));
assert_eq!(hdr.messages.len(), 1);
// flags bit 5 = timestamps, but bit 2 not set → no creation order in messages
assert!(hdr.messages[0].creation_order.is_none());
}
#[test]
fn parse_v2_creation_order() {
// flags bit 2 enables attribute/message creation order tracking
// flags bit 5 enables timestamps
// Use 0x24 = bit 2 + bit 5
let data = build_v2_header(
0x24,
&[(0x03, &[9], 0), (0x05, &[8], 0)],
Some((0, 0, 0, 0)),
);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 2);
assert!(hdr.messages[0].creation_order.is_some());
assert!(hdr.messages[1].creation_order.is_some());
assert_eq!(hdr.access_time, Some(0));
}
#[test]
fn parse_v2_checksum_valid() {
let data = build_v2_header(0x00, &[(0x01, &[1, 2, 3], 0)], None);
// Should succeed — checksum is valid
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 1);
}
#[test]
fn parse_v2_checksum_invalid() {
let mut data = build_v2_header(0x00, &[(0x01, &[1, 2, 3], 0)], None);
// Corrupt checksum
let len = data.len();
data[len - 1] ^= 0xFF;
let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err();
assert!(matches!(err, FormatError::ChecksumMismatch { .. }));
}
#[test]
fn parse_v2_nil_padding_skipped() {
let data = build_v2_header(
0x00,
&[
(0x00, &[0, 0, 0, 0], 0), // NIL
(0x01, &[42], 0), // Dataspace
],
None,
);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 1);
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
}
#[test]
fn parse_v2_chunk_size_1byte() {
// flags bits 0-1 = 0 → 1-byte chunk size
let data = build_v2_header(0x00, &[(0x01, &[1], 0)], None);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 1);
}
#[test]
fn parse_v2_chunk_size_2byte() {
let data = build_v2_header(0x01, &[(0x01, &[1], 0)], None);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 1);
}
#[test]
fn parse_v2_chunk_size_4byte() {
let data = build_v2_header(0x02, &[(0x01, &[1], 0)], None);
let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 1);
}
#[test]
fn parse_v2_continuation() {
// Build a continuation chunk (OCHK) at a known offset
let ochk_offset = 256usize;
let ochk_msg_type = 0x03u8; // Datatype
let ochk_msg_data = [0xDE, 0xAD];
// Build the OCHK chunk
let mut ochk_buf = Vec::new();
ochk_buf.extend_from_slice(&OCHK_SIGNATURE);
ochk_buf.push(ochk_msg_type);
ochk_buf.extend_from_slice(&(ochk_msg_data.len() as u16).to_le_bytes());
ochk_buf.push(0); // msg flags
ochk_buf.extend_from_slice(&ochk_msg_data);
let checksum = crate::checksum::jenkins_lookup3(&ochk_buf);
ochk_buf.extend_from_slice(&checksum.to_le_bytes());
let ochk_length = ochk_buf.len();
// Build continuation message data: offset(8 LE) + length(8 LE)
let mut cont_data = Vec::new();
cont_data.extend_from_slice(&(ochk_offset as u64).to_le_bytes());
cont_data.extend_from_slice(&(ochk_length as u64).to_le_bytes());
// Build main header with continuation message + a regular message
let header = build_v2_header(
0x00,
&[
(0x01, &[42], 0), // Dataspace
(0x10, &cont_data, 0), // Continuation
],
None,
);
// Assemble full "file"
let total_size = ochk_offset + ochk_buf.len();
let mut file_data = vec![0u8; total_size];
file_data[..header.len()].copy_from_slice(&header);
file_data[ochk_offset..ochk_offset + ochk_buf.len()].copy_from_slice(&ochk_buf);
let hdr = ObjectHeader::parse(&file_data, 0, 8, 8).unwrap();
assert_eq!(hdr.messages.len(), 2);
assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace);
assert_eq!(hdr.messages[1].msg_type, MessageType::Datatype);
assert_eq!(hdr.messages[1].data, vec![0xDE, 0xAD]);
}
#[test]
fn truncated_v1_header() {
let data = vec![1u8, 0]; // version 1, but too short
let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err();
assert!(matches!(err, FormatError::UnexpectedEof { .. }));
}
#[test]
fn truncated_v2_header() {
let data = [b'O', b'H', b'D', b'R', 2]; // signature + version, but no flags
let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err();
assert!(matches!(err, FormatError::UnexpectedEof { .. }));
}
/// Every header, and every truncation of it, parses to the same result
/// (or the same error) through a `read_at`-only storage as from a slice;
/// a header in one chunk takes two reads (prefix, chunk).
#[test]
fn parse_in_matches_slice_parse() {
use crate::storage::CountingStorage;
let mut headers = vec![
build_v1_header(&[], 8, 8),
build_v1_header(&[(0x0001, &[1, 2, 3], 0), (0x0003, &[9; 8], 0)], 8, 8),
build_v2_header(0x00, &[(0x01, &[42], 0)], None),
build_v2_header(0x03, &[(0x01, &[1, 2], 0), (0x03, &[3], 0)], None),
build_v2_header(0x24, &[(0x01, &[1], 0)], Some((1, 2, 3, 4))),
build_v2_header(0x35, &[(0x01, &[1], 0)], Some((5, 6, 7, 8))),
];
// A v2 header with a continuation chunk at 256.
let mut ochk = OCHK_SIGNATURE.to_vec();
ochk.extend_from_slice(&[0x03, 2, 0, 0, 0xDE, 0xAD]);
let sum = crate::checksum::jenkins_lookup3(&ochk);
ochk.extend_from_slice(&sum.to_le_bytes());
let mut cont = 256u64.to_le_bytes().to_vec();
cont.extend_from_slice(&(ochk.len() as u64).to_le_bytes());
let main = build_v2_header(0x00, &[(0x01, &[42], 0), (0x10, &cont, 0)], None);
let mut with_cont = vec![0u8; 256 + ochk.len()];
with_cont[..main.len()].copy_from_slice(&main);
with_cont[256..].copy_from_slice(&ochk);
headers.push(with_cont);
for h in headers {
for at in [0usize, 3] {
for cut in 0..=h.len() {
let mut f = vec![0u8; at];
f.extend_from_slice(&h[..cut]);
if at == 0 && cut == h.len() {
f.resize(f.len() + 64, 0);
}
let want = ObjectHeader::parse(&f, at, 8, 8);
let storage = CountingStorage::new(f.clone());
let got = ObjectHeader::parse_in(&storage, at as u64, 8, 8);
assert_eq!(
format!("{got:?}"),
format!("{want:?}"),
"at {at}, cut {cut}"
);
}
}
}
let one_chunk = build_v2_header(0x00, &[(0x01, &[42], 0)], None);
let storage = CountingStorage::new(one_chunk);
ObjectHeader::parse_in(&storage, 0, 8, 8).unwrap();
assert_eq!(storage.reads(), 2);
}
}