//! HDF5 Object Header parsing (v1 and v2). #[cfg(not(feature = "std"))] use alloc::{boxed::Box, collections::BTreeSet, vec::Vec}; #[cfg(feature = "std")] use std::collections::BTreeSet; use byteorder::{ByteOrder, LittleEndian}; use crate::addr::to_usize; use crate::error::FormatError; use crate::message_type::MessageType; use crate::storage::{Storage, Window, len_usize, read_exact_at}; /// OHDR signature for v2 object headers. const OHDR_SIGNATURE: [u8; 4] = *b"OHDR"; /// OCHK signature for v2 continuation chunks. const OCHK_SIGNATURE: [u8; 4] = *b"OCHK"; /// A single parsed header message. #[derive(Debug, Clone)] pub struct HeaderMessage { /// The message type. pub msg_type: MessageType, /// Size of the message data in bytes. pub size: usize, /// Message flags byte. pub flags: u8, /// Creation order (v2 only, when tracking is enabled). pub creation_order: Option, /// Raw message data bytes. pub data: Vec, } /// Parsed HDF5 object header. #[derive(Debug, Clone)] pub struct ObjectHeader { /// Header version (1 or 2). pub version: u8, /// All non-NIL messages collected from all chunks. pub messages: Vec, /// Object reference count (v1 only). pub reference_count: Option, /// Object header flags (v2 only; 0 for v1). pub flags: u8, /// Access time (v2, when flags bit 2 set). pub access_time: Option, /// Modification time (v2, when flags bit 2 set). pub modification_time: Option, /// Change time (v2, when flags bit 2 set). pub change_time: Option, /// Birth time (v2, when flags bit 2 set). pub birth_time: Option, } fn ensure_len(data: &[u8], offset: usize, needed: usize) -> Result<(), FormatError> { match offset.checked_add(needed) { Some(end) if end <= data.len() => Ok(()), _ => Err(FormatError::UnexpectedEof { expected: offset.saturating_add(needed), available: data.len(), }), } } fn read_offset(data: &[u8], pos: usize, size: u8) -> Result { let s = size as usize; ensure_len(data, pos, s)?; let slice = &data[pos..pos + s]; Ok(match size { 2 => LittleEndian::read_u16(slice) as u64, 4 => LittleEndian::read_u32(slice) as u64, 8 => LittleEndian::read_u64(slice), 1 => slice[0] as u64, _ => { return Err(FormatError::InvalidOffsetSize(size)); } }) } /// The kind of object an object header describes. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ObjectClass { /// A group: the header has a Symbol Table or a Link Info message. Group, /// A dataset: the header has a Datatype and a Dataspace message. Dataset, /// A committed (named) datatype: a Datatype message, no Dataspace. NamedDatatype, } impl ObjectHeader { /// The kind of object this header describes, decided as libhdf5 decides /// it (`H5O__obj_class_real`): group first (a Symbol Table or Link Info /// message), then dataset (a Datatype *and* a Dataspace message — not a /// Data Layout message), then named datatype (a Datatype message). /// `None` when none applies; libhdf5 then cannot open the object /// ("unable to determine object type"). /// /// A header with a Datatype and a Data Layout message but no Dataspace /// is a named datatype to libhdf5, not a dataset. pub fn object_class(&self) -> Option { let has = |t: MessageType| self.messages.iter().any(|m| m.msg_type == t); if has(MessageType::SymbolTable) || has(MessageType::LinkInfo) { Some(ObjectClass::Group) } else if has(MessageType::Datatype) && has(MessageType::Dataspace) { Some(ObjectClass::Dataset) } else if has(MessageType::Datatype) { Some(ObjectClass::NamedDatatype) } else { None } } /// Parse an object header at the given offset in the data buffer. /// /// `offset_size` and `length_size` come from the superblock. #[inline] pub fn parse( data: &[u8], offset: usize, offset_size: u8, length_size: u8, ) -> Result { Self::parse_slice(data, offset as u64, offset_size, length_size) } /// [`Self::parse`] over any [`Storage`]. /// /// Reads the prefix (at most [`V2_PREFIX_MAX`] bytes, signature /// included), then each chunk as one bounded read, continuation chunks /// included. A storage with the whole file in memory is parsed as its /// slice, by code compiled in this crate (see /// [`crate::storage`], "Slice entry points"). #[inline] pub fn parse_in( file: &S, offset: u64, offset_size: u8, length_size: u8, ) -> Result { match file.as_contiguous() { Some(all) => Self::parse_slice(all, offset, offset_size, length_size), None => Self::parse_storage(file, offset, offset_size, length_size), } } /// [`Self::parse_storage`] for the slice, compiled in this crate: the /// one copy [`Self::parse`] and [`Self::parse_in`] (in memory) call. fn parse_slice( data: &[u8], offset: u64, offset_size: u8, length_size: u8, ) -> Result { Self::parse_storage(data, offset, offset_size, length_size) } fn parse_storage( file: &S, offset: u64, offset_size: u8, length_size: u8, ) -> Result { // The first chunk is read once the prefix says how long it is: say // so (see `Storage::hint`), for a storage that fetches between // attempts. if file.as_contiguous().is_none() { file.hint(offset, OBJECT_HEADER_HINT_LEN); } // The longest prefix of either version, in one read. It holds the // whole prefix or ends at the end of the file, so its bounds checks // are the whole-file ones. let prefix = Window::read(file, offset, V2_PREFIX_MAX)?; prefix.ensure(0, 4)?; if prefix.bytes[..4] == OHDR_SIGNATURE { Self::parse_v2(file, offset, &prefix, offset_size, length_size) } else { Self::parse_v1(file, offset, &prefix, offset_size, length_size) } } fn parse_v1( file: &S, offset: u64, prefix: &Window<'_>, offset_size: u8, length_size: u8, ) -> Result { // version(1) + reserved(1) + num_messages(2) + ref_count(4) + header_size(4) = 12 // then pad to 8-byte alignment from start of header prefix.ensure(0, 12)?; let prefix = &prefix.bytes[..12]; let version = prefix[0]; if version != 1 { return Err(FormatError::InvalidObjectHeaderVersion(version)); } let num_messages = LittleEndian::read_u16(&prefix[2..4]) as usize; let reference_count = LittleEndian::read_u32(&prefix[4..8]); let header_data_size = LittleEndian::read_u32(&prefix[8..12]) as usize; // libhdf5 (H5O__prefix_deserialize): a header with messages needs room // for at least one message header, and one without has an empty chunk. if (num_messages > 0 && header_data_size < V1_MSG_HEADER_SIZE) || (num_messages == 0 && header_data_size > 0) { return Err(FormatError::InvalidObjectHeader( "bad object header chunk size", )); } // Pad to 8-byte alignment: header prefix is 12 bytes, pad to 16 let padding = 4; // pad 12-byte prefix to 16-byte alignment let msg_start = offset .checked_add(12 + padding) .ok_or(FormatError::UnexpectedEof { expected: usize::MAX, available: len_usize(file), })?; // parse_v1_chunk reads the chunk, with the bounds check that was here. // The prefix's count (NIL messages included, capped: it is untrusted) // sizes the list once instead of growing it message by message. let mut messages = Vec::with_capacity(num_messages.min(64)); let chunk0_count = Self::parse_v1_chunk( file, msg_start, header_data_size, offset_size, length_size, &mut messages, )?; // libhdf5 reads every message in the first chunk and refuses a header // whose prefix claims fewer than that (continuation chunks are read // later and not held to the count). Stopping after the claimed number // silently dropped the rest. if chunk0_count > num_messages { return Err(FormatError::InvalidObjectHeader( "bad object header message count", )); } Ok(ObjectHeader { version: 1, messages, reference_count: Some(reference_count), flags: 0, access_time: None, modification_time: None, change_time: None, birth_time: None, }) } /// Parse the messages of one version-1 chunk (`length` bytes at /// `offset`, no signature), following continuation messages as they are /// met. Returns how many messages (NIL ones included) this chunk itself /// holds. /// /// A version-1 chunk is filled with messages whose sizes are multiples of /// 8; libhdf5 refuses a message that is not aligned, that runs past the /// end of the chunk, or leftover bytes too few for a message header (a /// "gap", which only version 2 allows). /// /// Continuation chunks are read in the order their messages are found, /// as `H5O_protect` loads them (so the messages keep libhdf5's order): /// a queue of (address, length) pairs, each chunk read, parsed and /// released before the next, so only one chunk buffer is alive at a /// time whatever the storage. Every chunk must start at a new address /// (else a cycle), and the chunks together may be no larger than the /// file, so the bytes read stay within the file's size; a header of /// more than [`MAX_V1_CHUNKS`] chunks is refused. fn parse_v1_chunk( file: &S, offset: u64, length: usize, offset_size: u8, length_size: u8, messages: &mut Vec, ) -> Result { // The chunks found so far are also the queue of chunks to read. let mut spans = ChunkSpans::new(file.len(), offset, length)?; let mut chunk0_count = 0usize; let mut next = 0usize; let hints = file.as_contiguous().is_none(); while let Some((chunk_offset, chunk_length)) = spans.get(next) { let chunk = read_exact_at(file, chunk_offset, chunk_length)?; let known = spans.len; let count = Self::parse_v1_messages(&chunk, offset_size, length_size, messages, &mut spans)?; // The continuation chunks this one names are read next. if hints { for i in known..spans.len { if let Some((o, l)) = spans.get(i) { file.hint(o, l); } } } // Only the first chunk's messages are held to the prefix count. if next == 0 { chunk0_count = count; } next += 1; } Ok(chunk0_count) } /// The messages of one version-1 chunk: each checked and appended to /// `messages` (NIL ones dropped), each continuation added to `spans`. /// Returns how many messages (NIL ones included) the chunk holds. /// /// Inlined into the chunk loop: kept out of line (`#[inline(never)]`, /// 4313917), the call cost `ObjectHeader::parse` about 2.5 ns per /// header, 4% on small version-1 headers (A/B builds, 2026-09-27; see /// `BENCHMARKS.md`). Without an attribute the compiler keeps it out of /// line too. #[inline] fn parse_v1_messages( data: &[u8], offset_size: u8, length_size: u8, messages: &mut Vec, spans: &mut ChunkSpans, ) -> Result { let end = data.len(); let mut pos = 0usize; let mut count = 0usize; while pos < end { if end - pos < V1_MSG_HEADER_SIZE { return Err(FormatError::InvalidObjectHeader( "gap found in early version of file format", )); } let msg_type_raw = LittleEndian::read_u16(&data[pos..pos + 2]); let msg_data_size = LittleEndian::read_u16(&data[pos + 2..pos + 4]) as usize; let msg_flags = data[pos + 4]; // reserved(3) at pos+5..pos+8 pos += V1_MSG_HEADER_SIZE; if !msg_data_size.is_multiple_of(8) { return Err(FormatError::InvalidObjectHeader("message not aligned")); } if msg_data_size > end - pos { return Err(FormatError::InvalidObjectHeader( "message size exceeds buffer end", )); } let body = &data[pos..pos + msg_data_size]; check_message(1, msg_type_raw, msg_flags, body, offset_size, length_size)?; count += 1; let msg_type = MessageType::from_u16(msg_type_raw); if msg_type != MessageType::Nil { messages.push(HeaderMessage { msg_type, size: msg_data_size, flags: msg_flags, creation_order: None, data: body.to_vec(), }); } // Queue continuations (v1 continuation chunks are just raw // messages, no signature); check_message has checked the body. if msg_type == MessageType::ObjectHeaderContinuation { let cont_offset = read_offset(body, 0, offset_size)?; let cont_length = to_usize(read_offset(body, offset_size as usize, length_size)?)?; spans.add(cont_offset, cont_length)?; } pos += msg_data_size; } Ok(count) } fn parse_v2( file: &S, offset: u64, prefix: &Window<'_>, offset_size: u8, length_size: u8, ) -> Result { // `ensure_len` checks positions relative to the header against the // prefix window and reports them as the whole-file check did, with // absolute positions and the file's length. let data: &[u8] = &prefix.bytes; let file_len = len_usize(file); let base = usize::try_from(offset).unwrap_or(usize::MAX); let abs = |rel: usize| base.saturating_add(rel); let ensure_len = |_: &[u8], rel: usize, needed: usize| prefix.ensure(rel, needed); let offset = 0usize; // signature(4) + version(1) + flags(1) = 6 ensure_len(data, offset, 6)?; let version = data[offset + 4]; if version != 2 { return Err(FormatError::InvalidObjectHeaderVersion(version)); } let flags = data[offset + 5]; if flags & !V2_HDR_ALL_FLAGS != 0 { return Err(FormatError::InvalidObjectHeader( "unknown object header status flag(s)", )); } let mut pos = offset + 6; // Optional timestamps (flags bit 5) let (access_time, modification_time, change_time, birth_time) = if flags & 0x20 != 0 { ensure_len(data, pos, 16)?; let at = LittleEndian::read_u32(&data[pos..pos + 4]); let mt = LittleEndian::read_u32(&data[pos + 4..pos + 8]); let ct = LittleEndian::read_u32(&data[pos + 8..pos + 12]); let bt = LittleEndian::read_u32(&data[pos + 12..pos + 16]); pos += 16; (Some(at), Some(mt), Some(ct), Some(bt)) } else { (None, None, None, None) }; // Optional attribute storage thresholds (flags bit 4) if flags & 0x10 != 0 { ensure_len(data, pos, 4)?; // max_compact_attrs(2) + min_dense_attrs(2) — checked, not stored let max_compact = LittleEndian::read_u16(&data[pos..pos + 2]); let min_dense = LittleEndian::read_u16(&data[pos + 2..pos + 4]); if max_compact < min_dense { return Err(FormatError::InvalidObjectHeader( "bad object header attribute phase change values", )); } pos += 4; } // chunk0 size: width depends on flags bits 0-1 let chunk_size_width = match flags & 0x03 { 0 => 1u8, 1 => 2, 2 => 4, 3 => 8, _ => unreachable!(), }; ensure_len(data, pos, chunk_size_width as usize)?; let chunk0_size = to_usize(read_offset(data, pos, chunk_size_width)?)?; pos += chunk_size_width as usize; // Bit 2: attribute creation order tracked → messages include creation order field let has_creation_order = flags & 0x04 != 0; let msg_header_size = if has_creation_order { 6 } else { 4 }; if chunk0_size > 0 && chunk0_size < msg_header_size { return Err(FormatError::InvalidObjectHeader( "bad object header chunk size", )); } let chunk0_msg_start = pos; let Some(chunk0_abs_end) = abs(pos).checked_add(chunk0_size) else { return Err(FormatError::UnexpectedEof { expected: usize::MAX, available: file_len, }); }; let chunk0_msg_end = chunk0_abs_end - base; // The whole first chunk, prefix to checksum, in one read (its // bounds check is the one on the checksum's 4 bytes). let chunk0 = read_exact_at(file, base as u64, chunk0_msg_end.saturating_add(4))?; let data: &[u8] = &chunk0; // Validate checksum: from OHDR signature through all messages (before checksum) #[cfg(feature = "checksum")] { let stored = LittleEndian::read_u32(&data[chunk0_msg_end..chunk0_msg_end + 4]); let computed = crate::checksum::jenkins_lookup3(&data[offset..chunk0_msg_end]); if computed != stored { return Err(FormatError::ChecksumMismatch { expected: stored, computed, }); } } // Parse messages from chunk0 let mut messages = Vec::new(); let mut continuations = Vec::new(); Self::parse_v2_messages( data, chunk0_msg_start, chunk0_msg_end, has_creation_order, offset_size, length_size, &mut messages, &mut continuations, )?; // Follow continuations, one chunk buffer at a time. A chunk address // seen twice is a cycle in malformed data, and the chunks may add up // to no more than the file; a valid header can have many chunks (libhdf5 adds one // whenever a message no longer fits), up to the same bound as a // version-1 header. let mut spans = ChunkSpans::new(file.len(), base as u64, chunk0_msg_end.saturating_add(4))?; // The continuation chunks a chunk names are read next (see // `Storage::hint`). let hints = file.as_contiguous().is_none(); if hints { for &(o, l) in &continuations { file.hint(o as u64, l); } } while let Some((cont_offset, cont_length)) = continuations.pop() { spans.add(cont_offset as u64, cont_length)?; let known = continuations.len(); Self::parse_v2_continuation( file, cont_offset as u64, cont_length, has_creation_order, offset_size, length_size, &mut messages, &mut continuations, )?; if hints { for &(o, l) in &continuations[known..] { file.hint(o as u64, l); } } } Ok(ObjectHeader { version: 2, messages, reference_count: None, flags, access_time, modification_time, change_time, birth_time, }) } #[allow(clippy::too_many_arguments)] fn parse_v2_messages( data: &[u8], start: usize, end: usize, has_creation_order: bool, offset_size: u8, length_size: u8, messages: &mut Vec, continuations: &mut Vec<(usize, usize)>, ) -> Result<(), FormatError> { let msg_header_size = if has_creation_order { 6 } else { 4 }; let mut pos = start; let mut null_count = 0usize; while pos < end { // Leftover bytes too few for a message header are a gap, which // libhdf5 allows only in a chunk without NIL messages (a writer // that leaves a gap had no NIL message to put the space in). if end - pos < msg_header_size { if null_count != 0 { return Err(FormatError::InvalidObjectHeader( "gap in chunk with no null messages", )); } break; } let msg_type_raw = data[pos] as u16; let msg_data_size = LittleEndian::read_u16(&data[pos + 1..pos + 3]) as usize; let msg_flags = data[pos + 3]; let creation_order = if has_creation_order { Some(LittleEndian::read_u16(&data[pos + 4..pos + 6])) } else { None }; pos += msg_header_size; // `end` is where the messages stop and the checksum starts. // libhdf5 bounds a message by the chunk including its checksum, // but a message that runs into the checksum still fails there: // its loop stops at the checksum, and reading the checksum from // past its start overruns the chunk ("ran off end of input // buffer while decoding"). Both refuse it; only the text // differs. if msg_data_size > end - pos { return Err(FormatError::InvalidObjectHeader( "message size exceeds buffer end", )); } let body = &data[pos..pos + msg_data_size]; check_message(2, msg_type_raw, msg_flags, body, offset_size, length_size)?; let msg_type = MessageType::from_u16(msg_type_raw); if msg_type == MessageType::ObjectHeaderContinuation { // check_message has checked the body holds both fields. let cont_off = to_usize(read_offset(body, 0, offset_size)?)?; let cont_len = to_usize(read_offset(body, offset_size as usize, length_size)?)?; continuations.push((cont_off, cont_len)); } else if msg_type == MessageType::Nil { null_count += 1; } else { messages.push(HeaderMessage { msg_type, size: msg_data_size, flags: msg_flags, creation_order, data: body.to_vec(), }); } pos += msg_data_size; } Ok(()) } #[allow(clippy::too_many_arguments)] fn parse_v2_continuation( file: &S, offset: u64, length: usize, has_creation_order: bool, offset_size: u8, length_size: u8, messages: &mut Vec, continuations: &mut Vec<(usize, usize)>, ) -> Result<(), FormatError> { // OCHK signature(4) + messages + checksum(4) let chunk = read_exact_at(file, offset, length)?; let data: &[u8] = &chunk; let offset = 0usize; if length < 8 { return Err(FormatError::UnexpectedEof { expected: 8, available: length, }); } ensure_len(data, offset, 4)?; if data[offset..offset + 4] != OCHK_SIGNATURE { return Err(FormatError::InvalidObjectHeaderSignature); } let msg_start = offset + 4; let checksum_pos = offset + length - 4; #[cfg(feature = "checksum")] { let stored = LittleEndian::read_u32(&data[checksum_pos..checksum_pos + 4]); let computed = crate::checksum::jenkins_lookup3(&data[offset..checksum_pos]); if computed != stored { return Err(FormatError::ChecksumMismatch { expected: stored, computed, }); } } Self::parse_v2_messages( data, msg_start, checksum_pos, has_creation_order, offset_size, length_size, messages, continuations, ) } } /// What an object header is hinted to take before its prefix is read (see /// [`Storage::hint`]): the first chunk of a typical dataset's header. A /// longer header is read all the same. pub(crate) const OBJECT_HEADER_HINT_LEN: usize = 512; /// Longest version-2 object header prefix: signature(4) + version(1) + /// flags(1) + times(16) + attribute phase change(4) + chunk-0 size(8). const V2_PREFIX_MAX: usize = 34; /// Size of a version-1 message header: type(2) + size(2) + flags(1) + reserved(3). const V1_MSG_HEADER_SIZE: usize = 8; /// The chunks of one object header read so far, in the order they were /// found (which is the order version-1 chunks are read in). A chunk starting /// where another did is a cycle. Chunks of a valid header do not overlap, so /// together they are no larger than the file; a header whose chunks add up /// to more is refused, which bounds what its chunks can make a reader read /// (a crafted chain of chunks each nested in the last would otherwise read /// the file over and over). Overlap itself is not refused: libhdf5 reads /// such headers (`cve-2025-7067.h5` has one). /// /// Almost every header has at most a few chunks, and this runs once per /// header, so the first [`INLINE_CHUNKS`] live in an inline array and are /// checked for cycles by a scan; only a longer header allocates (the rest /// of the list, and a set of starts). Allocating a queue and a set for /// every header made parsing 401 small headers 1.8x slower. struct ChunkSpans { inline: [(u64, usize); INLINE_CHUNKS], /// Chunks after the first [`INLINE_CHUNKS`], and every chunk start. spill: Option>, /// How many chunks there are. len: usize, /// Bytes of the chunks so far, and the most they may add up to. total: u64, budget: u64, } /// The chunks of a [`ChunkSpans`] beyond its inline ones. struct SpilledSpans { chunks: Vec<(u64, usize)>, starts: BTreeSet, } /// How many chunks [`ChunkSpans`] holds without allocating. const INLINE_CHUNKS: usize = 8; impl ChunkSpans { #[inline] fn new(file_len: u64, start: u64, len: usize) -> Result { let mut s = Self { inline: [(0, 0); INLINE_CHUNKS], spill: None, len: 0, total: 0, budget: file_len, }; s.add(start, len)?; Ok(s) } /// Record the chunk `len` bytes at `start`. #[inline] fn add(&mut self, start: u64, len: usize) -> Result<(), FormatError> { self.total = self.total.saturating_add(len as u64); if self.len < INLINE_CHUNKS { if self.inline[..self.len].iter().any(|&(s, _)| s == start) { return Err(FormatError::NestingDepthExceeded); } self.inline[self.len] = (start, len); } else { self.add_spilled(start, len)?; } self.len += 1; if self.total > self.budget { return Err(FormatError::InvalidObjectHeader( "object header chunks larger than the file", )); } Ok(()) } #[cold] #[inline(never)] fn add_spilled(&mut self, start: u64, len: usize) -> Result<(), FormatError> { let inline = &self.inline; let spill = self.spill.get_or_insert_with(|| { Box::new(SpilledSpans { chunks: Vec::new(), starts: inline.iter().map(|&(s, _)| s).collect(), }) }); if !spill.starts.insert(start) || self.len >= MAX_V1_CHUNKS { return Err(FormatError::NestingDepthExceeded); } spill.chunks.push((start, len)); Ok(()) } /// The `i`th chunk recorded. #[inline] fn get(&self, i: usize) -> Option<(u64, usize)> { if i < INLINE_CHUNKS { (i < self.len).then(|| self.inline[i]) } else { self.spill.as_ref()?.chunks.get(i - INLINE_CHUNKS).copied() } } } /// Most chunks a version-1 object header may have (malformed-data guard; /// libhdf5 has no limit, and a header that gains one continuation chunk per /// attribute added can have many). const MAX_V1_CHUNKS: usize = 1 << 16; /// Every defined version-2 object header status flag (libhdf5 /// `H5O_HDR_ALL_FLAGS`): chunk-0 size width (bits 0-1), attribute creation /// order tracked/indexed, attribute phase-change values, times stored. const V2_HDR_ALL_FLAGS: u8 = 0x3F; // Header message flag bits (libhdf5 `H5O_MSG_FLAG_*`). Bit 0 (constant) needs // no check. Bit 3 (fail if unknown and the file is opened for writing) never // fails a read: the parser only ever reads, as libhdf5 ignores it for a // read-only open. const MSG_FLAG_SHARED: u8 = 0x02; const MSG_FLAG_DONTSHARE: u8 = 0x04; const MSG_FLAG_FAIL_IF_UNKNOWN_AND_OPEN_FOR_WRITE: u8 = 0x08; const MSG_FLAG_MARK_IF_UNKNOWN: u8 = 0x10; const MSG_FLAG_WAS_UNKNOWN: u8 = 0x20; const MSG_FLAG_SHAREABLE: u8 = 0x40; /// Fail if the message is unknown, whatever the access mode. const MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS: u8 = 0x80; /// Message type ids libhdf5 has a class for (`H5O_msg_class_g`): 0x00-0x18 /// except 0x09 (a test-only "bogus" message). Anything else is an unknown /// message. fn is_known_message(id: u16) -> bool { id <= 0x18 && id != 0x09 } /// Message classes that may be shared (`H5O_SHARE_IS_SHARABLE`): dataspace, /// datatype, the two fill-value messages, filter pipeline and attribute. fn is_shareable_message(id: u16) -> bool { matches!(id, 0x01 | 0x03 | 0x04 | 0x05 | 0x0B | 0x0C) } /// Check one header message the way libhdf5 does while it loads an object /// header (`H5O__chunk_deserialize`), so an object libhdf5 refuses to open is /// refused here too instead of being read from a corrupt header: /// /// - contradictory flag combinations; /// - an unknown message the file says no reader may skip (bit 7). This had /// bits 3 and 7 the wrong way round once, failing objects libhdf5 reads /// and reading ones it refuses (`tbogus.h5`); /// - a known message whose class cannot be shared, flagged shared or /// shareable (`cve-2016-4332`); /// - the messages libhdf5 decodes while loading the header, whose decode /// errors fail the load: continuation, reference count (which a version-1 /// header cannot hold), and both modification-time messages. fn check_message( header_version: u8, id: u16, flags: u8, body: &[u8], offset_size: u8, length_size: u8, ) -> Result<(), FormatError> { let bad_flags = FormatError::InvalidObjectHeader("bad flag combination for message"); if flags & MSG_FLAG_SHARED != 0 && flags & MSG_FLAG_DONTSHARE != 0 { return Err(bad_flags); } if flags & MSG_FLAG_WAS_UNKNOWN != 0 && (flags & MSG_FLAG_FAIL_IF_UNKNOWN_AND_OPEN_FOR_WRITE != 0 || flags & MSG_FLAG_MARK_IF_UNKNOWN == 0) { return Err(bad_flags); } if !is_known_message(id) { if flags & MSG_FLAG_FAIL_IF_UNKNOWN_ALWAYS != 0 { return Err(FormatError::UnsupportedMessage(id)); } return Ok(()); } if flags & (MSG_FLAG_SHARED | MSG_FLAG_SHAREABLE) != 0 && !is_shareable_message(id) { return Err(FormatError::InvalidObjectHeader( "message of unshareable class flagged as shareable", )); } let overrun = FormatError::InvalidObjectHeader("ran off end of input buffer while decoding"); match id { // Continuation: address + length, and the chunk cannot be empty. 0x10 => { if body.len() < offset_size as usize + length_size as usize { return Err(overrun); } if read_offset(body, offset_size as usize, length_size)? == 0 { return Err(FormatError::InvalidObjectHeader( "invalid continuation chunk size (0)", )); } } // Reference count: version-2 headers only; version 0 then a u32. 0x16 => { if header_version == 1 { return Err(FormatError::InvalidObjectHeader( "object header version does not support reference count message", )); } match body.first() { None => return Err(overrun), Some(0) => {} Some(_) => { return Err(FormatError::InvalidObjectHeader( "bad version number for reference count message", )); } } if body.len() < 5 { return Err(overrun); } } // Old modification time: "YYYYMMDDhhmmss" and 2 reserved bytes. 0x0E => { if body.len() < 16 { return Err(overrun); } if !body[..14].iter().all(u8::is_ascii_digit) { return Err(FormatError::InvalidObjectHeader( "badly formatted modification time message", )); } } // New modification time: version 1, 3 reserved bytes, u32 seconds. 0x12 => { match body.first() { None => return Err(overrun), Some(1) => {} Some(_) => { return Err(FormatError::InvalidObjectHeader( "bad version number for mtime message", )); } } if body.len() < 8 { return Err(overrun); } } _ => {} } Ok(()) } #[cfg(test)] mod tests { use super::*; fn header_with(types: &[MessageType]) -> ObjectHeader { ObjectHeader { version: 2, messages: types .iter() .map(|&msg_type| HeaderMessage { msg_type, size: 0, flags: 0, creation_order: None, data: Vec::new(), }) .collect(), reference_count: None, flags: 0, access_time: None, modification_time: None, change_time: None, birth_time: None, } } #[test] fn object_class_follows_libhdf5() { use MessageType::*; let class = |t: &[MessageType]| header_with(t).object_class(); assert_eq!( class(&[Datatype, Dataspace, DataLayout]), Some(ObjectClass::Dataset) ); // A Data Layout message does not make a dataset without a dataspace // (cve-2024-33874 `/Dset1`: h5py opens it as a named datatype). assert_eq!( class(&[Datatype, DataLayout]), Some(ObjectClass::NamedDatatype) ); assert_eq!(class(&[Datatype]), Some(ObjectClass::NamedDatatype)); // Group messages win over dataset messages. assert_eq!( class(&[Datatype, Dataspace, SymbolTable]), Some(ObjectClass::Group) ); assert_eq!(class(&[LinkInfo]), Some(ObjectClass::Group)); // Link messages alone are not a group; nothing is not an object. assert_eq!(class(&[Link]), None); assert_eq!(class(&[]), None); } // Helper: build a v1 object header with given messages fn build_v1_header( messages: &[(u16, &[u8], u8)], // (type, data, flags) offset_size: u8, length_size: u8, ) -> Vec { let _ = (offset_size, length_size); // Calculate total header message data size let mut msg_bytes = Vec::new(); for (mtype, mdata, mflags) in messages { // v1 message sizes are multiples of 8 (the data is zero-padded). let padded = <[u8]>::len(mdata).div_ceil(8) * 8; msg_bytes.extend_from_slice(&mtype.to_le_bytes()); // type(2) msg_bytes.extend_from_slice(&(padded as u16).to_le_bytes()); // size(2) msg_bytes.push(*mflags); // flags(1) msg_bytes.extend_from_slice(&[0u8; 3]); // reserved(3) msg_bytes.extend_from_slice(mdata); // data msg_bytes.resize(msg_bytes.len() + padded - <[u8]>::len(mdata), 0); } let mut buf = Vec::new(); buf.push(1); // version buf.push(0); // reserved buf.extend_from_slice(&(messages.len() as u16).to_le_bytes()); // num_messages buf.extend_from_slice(&1u32.to_le_bytes()); // reference_count buf.extend_from_slice(&(msg_bytes.len() as u32).to_le_bytes()); // header_data_size // Pad to 8-byte alignment (12 bytes so far, pad 4) buf.extend_from_slice(&[0u8; 4]); buf.extend_from_slice(&msg_bytes); buf } // Helper: build a v2 object header chunk0 with given messages fn build_v2_header( flags: u8, messages: &[(u8, &[u8], u8)], // (type, data, msg_flags) timestamps: Option<(u32, u32, u32, u32)>, ) -> Vec { let has_creation_order = flags & 0x04 != 0; let has_timestamps = flags & 0x20 != 0; let mut buf = Vec::new(); buf.extend_from_slice(&OHDR_SIGNATURE); // 4 buf.push(2); // version buf.push(flags); if has_timestamps && let Some((at, mt, ct, bt)) = timestamps { buf.extend_from_slice(&at.to_le_bytes()); buf.extend_from_slice(&mt.to_le_bytes()); buf.extend_from_slice(&ct.to_le_bytes()); buf.extend_from_slice(&bt.to_le_bytes()); } if flags & 0x10 != 0 { buf.extend_from_slice(&8u16.to_le_bytes()); // max_compact buf.extend_from_slice(&6u16.to_le_bytes()); // min_dense } // Build message bytes to get chunk size let mut msg_bytes = Vec::new(); for (mtype, mdata, mflags) in messages { msg_bytes.push(*mtype); // type(1) msg_bytes.extend_from_slice(&(mdata.len() as u16).to_le_bytes()); // size(2) msg_bytes.push(*mflags); // flags(1) if has_creation_order { msg_bytes.extend_from_slice(&0u16.to_le_bytes()); // creation_order(2) } msg_bytes.extend_from_slice(mdata); } let chunk_size = msg_bytes.len(); // Write chunk size based on flags bits 0-1 match flags & 0x03 { 0 => buf.push(chunk_size as u8), 1 => buf.extend_from_slice(&(chunk_size as u16).to_le_bytes()), 2 => buf.extend_from_slice(&(chunk_size as u32).to_le_bytes()), 3 => buf.extend_from_slice(&(chunk_size as u64).to_le_bytes()), _ => unreachable!(), } buf.extend_from_slice(&msg_bytes); // Checksum (CRC32C of everything from OHDR to here) let checksum = crate::checksum::jenkins_lookup3(&buf); buf.extend_from_slice(&checksum.to_le_bytes()); buf } #[test] fn parse_v1_zero_messages() { let data = build_v1_header(&[], 8, 8); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.version, 1); assert_eq!(hdr.messages.len(), 0); assert_eq!(hdr.reference_count, Some(1)); assert_eq!(hdr.flags, 0); } #[test] fn parse_v1_two_messages() { let messages = [ (0x0001u16, &[1u8, 2, 3, 4][..], 0u8), // Dataspace (0x0008, &[5u8, 6][..], 0), // DataLayout ]; let data = build_v1_header(&messages, 8, 8); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 2); assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace); // v1 message data is padded to a multiple of 8 bytes. assert_eq!(hdr.messages[0].data, vec![1, 2, 3, 4, 0, 0, 0, 0]); assert_eq!(hdr.messages[1].msg_type, MessageType::DataLayout); assert_eq!(hdr.messages[1].data[..2], [5, 6]); } /// A version-1 header whose continuation chunks form a chain: chunk k /// holds a Dataspace message `[k]` and the continuation to chunk k + 1. /// With `cycle`, the last chunk points back at the first continuation /// chunk. fn v1_chain(n: usize, cycle: bool) -> Vec { // Each continuation chunk: dataspace (8 + 8) + continuation (8 + 16). let chunk_len = 40u64; let first = 64u64; let cont = |addr: u64| { let mut b = addr.to_le_bytes().to_vec(); b.extend_from_slice(&chunk_len.to_le_bytes()); b }; let mut data = build_v1_header(&[(0x0010, &cont(first)[..], 0)], 8, 8); data.resize(first as usize, 0); for k in 0..n { let mut c = Vec::new(); c.extend_from_slice(&1u16.to_le_bytes()); c.extend_from_slice(&8u16.to_le_bytes()); c.extend_from_slice(&[0; 4]); c.extend_from_slice(&(k as u64).to_le_bytes()); let next = if k + 1 < n { first + (k as u64 + 1) * chunk_len } else if cycle { first } else { // The last chunk ends in a NIL message instead. c.extend_from_slice(&[0, 0, 16, 0, 0, 0, 0, 0]); c.extend_from_slice(&[0; 16]); data.extend_from_slice(&c); continue; }; c.extend_from_slice(&0x10u16.to_le_bytes()); c.extend_from_slice(&16u16.to_le_bytes()); c.extend_from_slice(&[0; 4]); c.extend_from_slice(&cont(next)); data.extend_from_slice(&c); } data } /// libhdf5 reads any chain of continuation chunks (a header grows one /// per attribute added when full); the reader used to stop at 32. #[test] fn long_v1_continuation_chains_are_read() { let data = v1_chain(200, false); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); let spaces: Vec = hdr .messages .iter() .filter(|m| m.msg_type == MessageType::Dataspace) .map(|m| m.data[0]) .collect(); assert_eq!(spaces, (0..200).map(|k| k as u8).collect::>()); } /// A crafted version-1 header whose continuation chunks nest: each /// chunk's continuation message points at the rest of that chunk. Read /// depth-first with every enclosing chunk kept alive, from storage that /// hands out owned buffers, it read n^2 bytes and held them all at once /// (a 192 KB file read 768 MB). Chunks adding up to more than the file /// are refused, and the bytes read stay within the file's size. #[test] fn nested_v1_continuation_chunks_are_bounded() { use crate::storage::CountingStorage; let n = 2000u64; let a = 64u64; let cont = |addr: u64, len: u64| { let mut m = vec![0x10, 0, 16, 0, 0, 0, 0, 0]; m.extend_from_slice(&addr.to_le_bytes()); m.extend_from_slice(&len.to_le_bytes()); m }; // Prefix: version 1, one message, reference count 1, 24 bytes. let mut buf = vec![1, 0, 1, 0, 1, 0, 0, 0, 24, 0, 0, 0, 0, 0, 0, 0]; buf.extend_from_slice(&cont(a, 24 * n)); buf.resize(a as usize, 0); for k in 0..n { if k + 1 < n { buf.extend_from_slice(&cont(a + 24 * (k + 1), 24 * (n - k - 1))); } else { buf.extend_from_slice(&[0, 0, 16, 0, 0, 0, 0, 0]); buf.extend_from_slice(&[0; 16]); } } let len = buf.len() as u64; let s = CountingStorage::new(buf); assert!(matches!( ObjectHeader::parse_in(&s, 0, 8, 8), Err(FormatError::InvalidObjectHeader( "object header chunks larger than the file" )) )); assert!( s.bytes_read() <= 2 * len, "read {} of {len}", s.bytes_read() ); } /// libhdf5 reads a continuation chunk that overlaps the chunk holding /// its message (`cve-2025-7067.h5` has one), and so does this reader. #[test] fn overlapping_v1_continuation_chunk_is_read() { // Chunk 0 (at 16): continuation (24 bytes), then a NIL message at // 40; the continuation chunk is that NIL message's 8-byte header. let mut cont = 40u64.to_le_bytes().to_vec(); cont.extend_from_slice(&8u64.to_le_bytes()); let data = build_v1_header(&[(0x0010, &cont[..], 0), (0x0000, &[][..], 0)], 8, 8); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 1); } /// A valid chain over owned-buffer storage reads each chunk once. #[test] fn long_v1_chain_reads_each_chunk_once() { use crate::storage::CountingStorage; let data = v1_chain(3000, false); let len = data.len() as u64; let s = CountingStorage::new(data); let hdr = ObjectHeader::parse_in(&s, 0, 8, 8).unwrap(); assert_eq!( hdr.messages .iter() .filter(|m| m.msg_type == MessageType::Dataspace) .count(), 3000 ); assert!(s.bytes_read() <= len, "read {} of {len}", s.bytes_read()); } /// Continuation chunks are read in the order their messages are found /// (libhdf5's `H5O_protect`), so a chunk's messages follow every /// message of the chunk before, not the continuation message. #[test] fn v1_continuation_messages_keep_libhdf5_order() { // Chunk 0: continuation to A, dataspace [1]; A: dataspace [2]. let a = 64u64; let mut cont = a.to_le_bytes().to_vec(); cont.extend_from_slice(&16u64.to_le_bytes()); let mut data = build_v1_header(&[(0x0010, &cont[..], 0), (0x0001, &[1; 8][..], 0)], 8, 8); data.resize(a as usize, 0); data.extend_from_slice(&[1, 0, 8, 0, 0, 0, 0, 0]); data.extend_from_slice(&[2; 8]); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); let spaces: Vec = hdr .messages .iter() .filter(|m| m.msg_type == MessageType::Dataspace) .map(|m| m.data[0]) .collect(); assert_eq!(spaces, [1, 2]); } #[test] fn v1_continuation_cycles_are_refused() { // Within the inline chunk list, and past it (the cycle returns to // an inline chunk once the list has spilled). for n in [5, 7, 8, 9, 40] { let data = v1_chain(n, true); assert!( matches!( ObjectHeader::parse(&data, 0, 8, 8), Err(FormatError::NestingDepthExceeded) ), "{n} chunks" ); } } #[test] fn parse_v1_unknown_message_ok() { let messages = [(0x00FFu16, &[0xAA, 0xBB][..], 0u8)]; let data = build_v1_header(&messages, 8, 8); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 1); assert_eq!(hdr.messages[0].msg_type, MessageType::Unknown(0x00FF)); } #[test] fn parse_v1_unknown_fail_always_errors() { // Bit 7 of msg_flags = fail if unknown, whatever the access mode. let messages = [(0x00FFu16, &[0xAA][..], 0x80u8)]; let data = build_v1_header(&messages, 8, 8); let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(); assert_eq!(err, FormatError::UnsupportedMessage(0x00FF)); } #[test] fn parse_v1_unknown_fail_on_write_is_ignored_when_reading() { // Bit 3 = fail if unknown *and the file is opened for writing*. This // parser only reads, so libhdf5 (read-only) opens such an object and // so must we. Bits 4/5 (mark if unknown / was unknown) never fail. for flags in [0x08u8, 0x10, 0x30] { let messages = [(0x00FFu16, &[0xAA][..], flags)]; let data = build_v1_header(&messages, 8, 8); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages[0].msg_type, MessageType::Unknown(0x00FF)); } } #[test] fn contradictory_message_flags_are_refused() { // libhdf5: "bad flag combination for message" for shared + don't // share, was-unknown without mark-if-unknown, and was-unknown with // fail-if-unknown-on-write. for flags in [0x06u8, 0x20, 0x38] { let data = build_v1_header(&[(0x00FFu16, &[0xAA][..], flags)], 8, 8); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("bad flag combination for message"), "flags {flags:#x}" ); let data = build_v2_header(0x00, &[(0xF0, &[1, 2], flags)], None); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("bad flag combination for message"), "flags {flags:#x}" ); } } #[test] fn unshareable_message_flagged_shareable_is_refused() { // A layout (0x08) or modification time (0x12) message cannot be // shared; bit 1 (shared) or bit 6 (shareable) on one is corruption // (cve-2016-4332). A datatype (0x03) may be shareable. let mtime = [1u8, 0, 0, 0, 0x10, 0x20, 0x30, 0x40]; for (id, flags) in [(0x08u16, 0x40u8), (0x08, 0x02), (0x12, 0x40)] { let data = build_v1_header(&[(id, &mtime[..], flags)], 8, 8); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader( "message of unshareable class flagged as shareable" ), "id {id:#x} flags {flags:#x}" ); } let data = build_v1_header(&[(0x03, &[0u8; 8][..], 0x40)], 8, 8); assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); // An unknown message is never checked for shareability. let data = build_v1_header(&[(0x00FF, &[0u8; 8][..], 0x40)], 8, 8); assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); } #[test] fn v1_message_must_be_aligned() { // cve-2018-13873: a v1 message whose size is not a multiple of 8. let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8); data[16 + 2] = 7; // size field of the only message assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("message not aligned") ); } #[test] fn message_overrunning_its_chunk_is_refused() { // It used to end the chunk quietly, dropping this message and any // after it. let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8); data[16 + 2] = 16; data.resize(data.len() + 64, 0); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("message size exceeds buffer end") ); let mut data = build_v2_header(0x00, &[(0x01, &[1, 2], 0)], None); data[7 + 1] = 9; // size of the only message (after OHDR, ver, flags, chunk size) let chk = crate::checksum::jenkins_lookup3(&data[..data.len() - 4]); let n = data.len(); data[n - 4..].copy_from_slice(&chk.to_le_bytes()); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("message size exceeds buffer end") ); } #[test] fn v1_gap_after_last_message_is_refused() { // Fewer than 8 bytes left over: a gap, which only version 2 allows. let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0)], 8, 8); data[8] += 4; // header_data_size data.extend_from_slice(&[0u8; 4]); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("gap found in early version of file format") ); } #[test] fn v1_chunk_holding_more_messages_than_the_prefix_says_is_refused() { // cve-2024-32619: the prefix says 1 message, the chunk holds 2. The // second used to be dropped silently. let mut data = build_v1_header(&[(0x01, &[0u8; 8][..], 0), (0x03, &[0u8; 8][..], 0)], 8, 8); data[2] = 1; assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("bad object header message count") ); // Fewer in the chunk than the prefix says is fine (the rest may be in // continuation chunks; libhdf5 only enforces that with strict checks). data[2] = 3; assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap().messages.len(), 2 ); } #[test] fn v1_prefix_chunk_size_must_fit_the_message_count() { let mut data = build_v1_header(&[], 8, 8); data[8] = 8; // no messages but a non-empty chunk data.extend_from_slice(&[0u8; 8]); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("bad object header chunk size") ); } #[test] fn v1_header_cannot_hold_a_reference_count_message() { // cve-2018-11204. let data = build_v1_header(&[(0x16, &[0, 2, 0, 0, 0][..], 0)], 8, 8); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader( "object header version does not support reference count message" ) ); let data = build_v2_header(0x00, &[(0x16, &[0, 2, 0, 0, 0], 0)], None); assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); } #[test] fn modification_time_messages_are_decoded_with_the_header() { // cve-2024-33873 (version 0) and cve-2024-33874 (empty message). for (body, why) in [ ( &[0u8, 0, 0, 0, 1, 2, 3, 4][..], "bad version number for mtime message", ), (&[][..], "ran off end of input buffer while decoding"), ] { let data = build_v2_header(0x00, &[(0x12, body, 0)], None); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader(why) ); } let data = build_v2_header(0x00, &[(0x12, &[1, 0, 0, 0, 1, 2, 3, 4], 0)], None); assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); // The old (0x0E) message is 14 ASCII digits and 2 reserved bytes. let data = build_v1_header(&[(0x0E, &b"20110414214255\0\0"[..], 0)], 8, 8); assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); let data = build_v1_header(&[(0x0E, &b"2011041421425x\0\0"[..], 0)], 8, 8); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("badly formatted modification time message") ); } #[test] fn continuation_message_must_hold_a_nonempty_chunk() { let mut cont = [0u8; 16]; cont[..8].copy_from_slice(&64u64.to_le_bytes()); let data = build_v2_header(0x00, &[(0x10, &cont, 0)], None); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("invalid continuation chunk size (0)") ); let data = build_v2_header(0x00, &[(0x10, &cont[..8], 0)], None); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("ran off end of input buffer while decoding") ); } #[test] fn v2_prefix_is_checked() { let data = build_v2_header(0x40, &[(0x01, &[1], 0)], None); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("unknown object header status flag(s)") ); // build_v2_header writes max_compact 8, min_dense 6; swap them. let mut data = build_v2_header(0x10, &[(0x01, &[1], 0)], None); data[6] = 6; data[8] = 8; let chk = crate::checksum::jenkins_lookup3(&data[..data.len() - 4]); let n = data.len(); data[n - 4..].copy_from_slice(&chk.to_le_bytes()); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("bad object header attribute phase change values") ); } #[test] fn v2_gap_is_allowed_only_without_nil_messages() { // Three bytes after the last message: a gap (a message header is 4). let mut data = build_v2_header(0x00, &[(0x01, &[1, 2, 3], 0), (0x03, &[], 0)], None); // Turn the empty datatype message (4 header bytes) into a 3-byte gap // by shrinking the chunk. let n = data.len(); data.truncate(n - 5); data[6] -= 1; let chk = crate::checksum::jenkins_lookup3(&data); data.extend_from_slice(&chk.to_le_bytes()); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap().messages.len(), 1 ); let mut data = build_v2_header(0x00, &[(0x00, &[1, 2, 3], 0), (0x03, &[], 0)], None); let n = data.len(); data.truncate(n - 5); data[6] -= 1; let chk = crate::checksum::jenkins_lookup3(&data); data.extend_from_slice(&chk.to_le_bytes()); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::InvalidObjectHeader("gap in chunk with no null messages") ); } #[test] fn parse_v2_unknown_message_flags() { let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x08)], None); assert!(ObjectHeader::parse(&data, 0, 8, 8).is_ok()); let data = build_v2_header(0x00, &[(0xF0, &[1, 2], 0x80)], None); assert_eq!( ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(), FormatError::UnsupportedMessage(0xF0) ); } #[test] fn parse_v2_no_timestamps_one_message() { let data = build_v2_header(0x00, &[(0x01, &[10, 20], 0)], None); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.version, 2); assert_eq!(hdr.flags, 0); assert_eq!(hdr.messages.len(), 1); assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace); assert_eq!(hdr.messages[0].data, vec![10, 20]); assert!(hdr.access_time.is_none()); } #[test] fn parse_v2_with_timestamps() { let data = build_v2_header(0x20, &[(0x01, &[1], 0)], Some((100, 200, 300, 400))); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.access_time, Some(100)); assert_eq!(hdr.modification_time, Some(200)); assert_eq!(hdr.change_time, Some(300)); assert_eq!(hdr.birth_time, Some(400)); assert_eq!(hdr.messages.len(), 1); // flags bit 5 = timestamps, but bit 2 not set → no creation order in messages assert!(hdr.messages[0].creation_order.is_none()); } #[test] fn parse_v2_creation_order() { // flags bit 2 enables attribute/message creation order tracking // flags bit 5 enables timestamps // Use 0x24 = bit 2 + bit 5 let data = build_v2_header( 0x24, &[(0x03, &[9], 0), (0x05, &[8], 0)], Some((0, 0, 0, 0)), ); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 2); assert!(hdr.messages[0].creation_order.is_some()); assert!(hdr.messages[1].creation_order.is_some()); assert_eq!(hdr.access_time, Some(0)); } #[test] fn parse_v2_checksum_valid() { let data = build_v2_header(0x00, &[(0x01, &[1, 2, 3], 0)], None); // Should succeed — checksum is valid let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 1); } #[test] fn parse_v2_checksum_invalid() { let mut data = build_v2_header(0x00, &[(0x01, &[1, 2, 3], 0)], None); // Corrupt checksum let len = data.len(); data[len - 1] ^= 0xFF; let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(); assert!(matches!(err, FormatError::ChecksumMismatch { .. })); } #[test] fn parse_v2_nil_padding_skipped() { let data = build_v2_header( 0x00, &[ (0x00, &[0, 0, 0, 0], 0), // NIL (0x01, &[42], 0), // Dataspace ], None, ); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 1); assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace); } #[test] fn parse_v2_chunk_size_1byte() { // flags bits 0-1 = 0 → 1-byte chunk size let data = build_v2_header(0x00, &[(0x01, &[1], 0)], None); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 1); } #[test] fn parse_v2_chunk_size_2byte() { let data = build_v2_header(0x01, &[(0x01, &[1], 0)], None); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 1); } #[test] fn parse_v2_chunk_size_4byte() { let data = build_v2_header(0x02, &[(0x01, &[1], 0)], None); let hdr = ObjectHeader::parse(&data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 1); } #[test] fn parse_v2_continuation() { // Build a continuation chunk (OCHK) at a known offset let ochk_offset = 256usize; let ochk_msg_type = 0x03u8; // Datatype let ochk_msg_data = [0xDE, 0xAD]; // Build the OCHK chunk let mut ochk_buf = Vec::new(); ochk_buf.extend_from_slice(&OCHK_SIGNATURE); ochk_buf.push(ochk_msg_type); ochk_buf.extend_from_slice(&(ochk_msg_data.len() as u16).to_le_bytes()); ochk_buf.push(0); // msg flags ochk_buf.extend_from_slice(&ochk_msg_data); let checksum = crate::checksum::jenkins_lookup3(&ochk_buf); ochk_buf.extend_from_slice(&checksum.to_le_bytes()); let ochk_length = ochk_buf.len(); // Build continuation message data: offset(8 LE) + length(8 LE) let mut cont_data = Vec::new(); cont_data.extend_from_slice(&(ochk_offset as u64).to_le_bytes()); cont_data.extend_from_slice(&(ochk_length as u64).to_le_bytes()); // Build main header with continuation message + a regular message let header = build_v2_header( 0x00, &[ (0x01, &[42], 0), // Dataspace (0x10, &cont_data, 0), // Continuation ], None, ); // Assemble full "file" let total_size = ochk_offset + ochk_buf.len(); let mut file_data = vec![0u8; total_size]; file_data[..header.len()].copy_from_slice(&header); file_data[ochk_offset..ochk_offset + ochk_buf.len()].copy_from_slice(&ochk_buf); let hdr = ObjectHeader::parse(&file_data, 0, 8, 8).unwrap(); assert_eq!(hdr.messages.len(), 2); assert_eq!(hdr.messages[0].msg_type, MessageType::Dataspace); assert_eq!(hdr.messages[1].msg_type, MessageType::Datatype); assert_eq!(hdr.messages[1].data, vec![0xDE, 0xAD]); } #[test] fn truncated_v1_header() { let data = vec![1u8, 0]; // version 1, but too short let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(); assert!(matches!(err, FormatError::UnexpectedEof { .. })); } #[test] fn truncated_v2_header() { let data = [b'O', b'H', b'D', b'R', 2]; // signature + version, but no flags let err = ObjectHeader::parse(&data, 0, 8, 8).unwrap_err(); assert!(matches!(err, FormatError::UnexpectedEof { .. })); } /// Every header, and every truncation of it, parses to the same result /// (or the same error) through a `read_at`-only storage as from a slice; /// a header in one chunk takes two reads (prefix, chunk). #[test] fn parse_in_matches_slice_parse() { use crate::storage::CountingStorage; let mut headers = vec![ build_v1_header(&[], 8, 8), build_v1_header(&[(0x0001, &[1, 2, 3], 0), (0x0003, &[9; 8], 0)], 8, 8), build_v2_header(0x00, &[(0x01, &[42], 0)], None), build_v2_header(0x03, &[(0x01, &[1, 2], 0), (0x03, &[3], 0)], None), build_v2_header(0x24, &[(0x01, &[1], 0)], Some((1, 2, 3, 4))), build_v2_header(0x35, &[(0x01, &[1], 0)], Some((5, 6, 7, 8))), ]; // A v2 header with a continuation chunk at 256. let mut ochk = OCHK_SIGNATURE.to_vec(); ochk.extend_from_slice(&[0x03, 2, 0, 0, 0xDE, 0xAD]); let sum = crate::checksum::jenkins_lookup3(&ochk); ochk.extend_from_slice(&sum.to_le_bytes()); let mut cont = 256u64.to_le_bytes().to_vec(); cont.extend_from_slice(&(ochk.len() as u64).to_le_bytes()); let main = build_v2_header(0x00, &[(0x01, &[42], 0), (0x10, &cont, 0)], None); let mut with_cont = vec![0u8; 256 + ochk.len()]; with_cont[..main.len()].copy_from_slice(&main); with_cont[256..].copy_from_slice(&ochk); headers.push(with_cont); for h in headers { for at in [0usize, 3] { for cut in 0..=h.len() { let mut f = vec![0u8; at]; f.extend_from_slice(&h[..cut]); if at == 0 && cut == h.len() { f.resize(f.len() + 64, 0); } let want = ObjectHeader::parse(&f, at, 8, 8); let storage = CountingStorage::new(f.clone()); let got = ObjectHeader::parse_in(&storage, at as u64, 8, 8); assert_eq!( format!("{got:?}"), format!("{want:?}"), "at {at}, cut {cut}" ); } } } let one_chunk = build_v2_header(0x00, &[(0x01, &[42], 0)], None); let storage = CountingStorage::new(one_chunk); ObjectHeader::parse_in(&storage, 0, 8, 8).unwrap(); assert_eq!(storage.reads(), 2); } }