read: look names up through the dense name indexes
Finding one link or attribute by name read every entry: Group::dataset and Group::group (File, MmapFile, LazyFile) listed the whole group per call, and path resolution scanned each group's links. Opening every child of a 35 001-link group by name decoded ~1.2e9 links. Now a dense group's v2 B-tree name index (type 5, lookup3 hash of the name) is descended to the records with the name's hash (btree_v2::find_btree_v2_records reads only the nodes whose key interval overlaps), and only those links are read and compared; all hash-equal records are compared, so libhdf5's tie order does not matter. Dense attributes the same through their type 8 index (attribute::find_attribute_in_file, facade attr(name)); huge heap objects through their ID-ordered index. group_v2::resolve_child returns what the listing has under a name (soft links followed, dangling/external ones not found). Group::entries and File::group_at hand out a listing's addresses. The lookup-stats feature counts heap objects read. Tests: one lookup in an h5py-written 35 001-link group with colliding hashes reads at most two links (before: 35 001, failing), attribute lookups likewise (before: 3 000, failing), every child opens through all three readers and matches h5py, every link kind resolves as h5py resolves it in dense and compact groups, 300 huge attributes are found, and a range search matches a full scan at every tree depth. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -2,10 +2,12 @@
|
||||
|
||||
#[cfg(not(feature = "std"))]
|
||||
use alloc::vec::Vec;
|
||||
use core::cmp::Ordering;
|
||||
|
||||
#[cfg(feature = "checksum")]
|
||||
use byteorder::{ByteOrder, LittleEndian};
|
||||
|
||||
use crate::addr::to_usize;
|
||||
use crate::error::FormatError;
|
||||
|
||||
/// Parsed B-tree v2 header (signature "BTHD").
|
||||
@@ -216,7 +218,7 @@ pub fn collect_btree_v2_records(
|
||||
// Root is a leaf
|
||||
parse_leaf_records(
|
||||
file_data,
|
||||
header.root_node_address as usize,
|
||||
to_usize(header.root_node_address)?,
|
||||
header.num_records_in_root,
|
||||
header.record_size,
|
||||
)
|
||||
@@ -225,7 +227,7 @@ pub fn collect_btree_v2_records(
|
||||
let mut records = Vec::new();
|
||||
collect_internal_records(
|
||||
file_data,
|
||||
header.root_node_address as usize,
|
||||
to_usize(header.root_node_address)?,
|
||||
header.num_records_in_root,
|
||||
header.depth,
|
||||
header.record_size,
|
||||
@@ -289,9 +291,10 @@ fn parse_leaf_records(
|
||||
Ok(records)
|
||||
}
|
||||
|
||||
/// Recursively collect records from an internal node.
|
||||
#[allow(clippy::too_many_arguments, clippy::only_used_in_recursion)]
|
||||
fn collect_internal_records(
|
||||
/// An internal node's layout: where its records start, and its children as
|
||||
/// `(address, record count)`.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn read_internal_node(
|
||||
file_data: &[u8],
|
||||
offset: usize,
|
||||
num_records: u16,
|
||||
@@ -299,11 +302,8 @@ fn collect_internal_records(
|
||||
record_size: u16,
|
||||
node_size: u32,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
max_leaf_nrec: u64,
|
||||
budget: &mut usize,
|
||||
out: &mut Vec<BTreeV2Record>,
|
||||
) -> Result<(), FormatError> {
|
||||
) -> Result<(usize, Vec<(u64, u16)>), FormatError> {
|
||||
// signature(4) + version(1) + type(1) = 6
|
||||
ensure_len(file_data, offset, 6)?;
|
||||
if &file_data[offset..offset + 4] != b"BTIN" {
|
||||
@@ -314,7 +314,7 @@ fn collect_internal_records(
|
||||
let rs = record_size as usize;
|
||||
let mut pos = offset + 6;
|
||||
|
||||
// Read all records first
|
||||
// Records first
|
||||
let records_total = nr.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
||||
expected: usize::MAX,
|
||||
available: file_data.len(),
|
||||
@@ -346,7 +346,6 @@ fn collect_internal_records(
|
||||
let child_ptr_size = offset_size as usize + nrec_width + total_nrec_width;
|
||||
ensure_len(file_data, pos, num_children * child_ptr_size)?;
|
||||
|
||||
// Read child pointers
|
||||
let mut children = Vec::with_capacity(num_children);
|
||||
for _ in 0..num_children {
|
||||
let addr = read_offset(file_data, pos, offset_size)?;
|
||||
@@ -356,6 +355,61 @@ fn collect_internal_records(
|
||||
pos += total_nrec_width; // skip total records in subtree
|
||||
children.push((addr, child_nrec));
|
||||
}
|
||||
Ok((records_start, children))
|
||||
}
|
||||
|
||||
/// Record `i` of an internal node whose records start at `records_start`.
|
||||
fn internal_record(
|
||||
file_data: &[u8],
|
||||
records_start: usize,
|
||||
i: usize,
|
||||
rs: usize,
|
||||
) -> Result<&[u8], FormatError> {
|
||||
let overflow = || FormatError::UnexpectedEof {
|
||||
expected: usize::MAX,
|
||||
available: file_data.len(),
|
||||
};
|
||||
let rec_start = i
|
||||
.checked_mul(rs)
|
||||
.and_then(|o| records_start.checked_add(o))
|
||||
.ok_or_else(overflow)?;
|
||||
let rec_end = rec_start.checked_add(rs).ok_or_else(overflow)?;
|
||||
file_data
|
||||
.get(rec_start..rec_end)
|
||||
.ok_or(FormatError::UnexpectedEof {
|
||||
expected: rec_end,
|
||||
available: file_data.len(),
|
||||
})
|
||||
}
|
||||
|
||||
/// Recursively collect records from an internal node.
|
||||
#[allow(clippy::too_many_arguments, clippy::only_used_in_recursion)]
|
||||
fn collect_internal_records(
|
||||
file_data: &[u8],
|
||||
offset: usize,
|
||||
num_records: u16,
|
||||
depth: u16,
|
||||
record_size: u16,
|
||||
node_size: u32,
|
||||
offset_size: u8,
|
||||
length_size: u8,
|
||||
max_leaf_nrec: u64,
|
||||
budget: &mut usize,
|
||||
out: &mut Vec<BTreeV2Record>,
|
||||
) -> Result<(), FormatError> {
|
||||
let nr = num_records as usize;
|
||||
let rs = record_size as usize;
|
||||
let (records_start, children) = read_internal_node(
|
||||
file_data,
|
||||
offset,
|
||||
num_records,
|
||||
depth,
|
||||
record_size,
|
||||
node_size,
|
||||
offset_size,
|
||||
max_leaf_nrec,
|
||||
)?;
|
||||
let child_depth = depth - 1;
|
||||
|
||||
// Interleave: child[0], record[0], child[1], record[1], ..., child[nr]
|
||||
// We collect child[0] records, then record[0], then child[1], etc.
|
||||
@@ -364,12 +418,12 @@ fn collect_internal_records(
|
||||
// Before parsing, so a refused tree is not also a large allocation.
|
||||
spend(budget, usize::from(child_nrec))?;
|
||||
let leaf_recs =
|
||||
parse_leaf_records(file_data, child_addr as usize, child_nrec, record_size)?;
|
||||
parse_leaf_records(file_data, to_usize(child_addr)?, child_nrec, record_size)?;
|
||||
out.extend(leaf_recs);
|
||||
} else {
|
||||
collect_internal_records(
|
||||
file_data,
|
||||
child_addr as usize,
|
||||
to_usize(child_addr)?,
|
||||
child_nrec,
|
||||
child_depth,
|
||||
record_size,
|
||||
@@ -384,32 +438,10 @@ fn collect_internal_records(
|
||||
|
||||
// Add record[i] (except after the last child)
|
||||
if i < nr {
|
||||
let rec_offset = i.checked_mul(rs).ok_or(FormatError::UnexpectedEof {
|
||||
expected: usize::MAX,
|
||||
available: file_data.len(),
|
||||
})?;
|
||||
let rec_start =
|
||||
records_start
|
||||
.checked_add(rec_offset)
|
||||
.ok_or(FormatError::UnexpectedEof {
|
||||
expected: usize::MAX,
|
||||
available: file_data.len(),
|
||||
})?;
|
||||
let rec_end = rec_start
|
||||
.checked_add(rs)
|
||||
.ok_or(FormatError::UnexpectedEof {
|
||||
expected: usize::MAX,
|
||||
available: file_data.len(),
|
||||
})?;
|
||||
if rec_end > file_data.len() {
|
||||
return Err(FormatError::UnexpectedEof {
|
||||
expected: rec_end,
|
||||
available: file_data.len(),
|
||||
});
|
||||
}
|
||||
let data = internal_record(file_data, records_start, i, rs)?;
|
||||
spend(budget, 1)?;
|
||||
out.push(BTreeV2Record {
|
||||
data: file_data[rec_start..rec_end].to_vec(),
|
||||
data: data.to_vec(),
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -417,6 +449,116 @@ fn collect_internal_records(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The records of a B-tree v2 that fall in one key range, found by
|
||||
/// descending the tree instead of reading all of it.
|
||||
///
|
||||
/// `cmp` places a record relative to the range: `Less` if the record sorts
|
||||
/// before it, `Greater` if after, `Equal` if the record is in it. The tree
|
||||
/// must be ordered consistently with `cmp`, as libhdf5 orders it (a link or
|
||||
/// attribute name index by name hash, so all records with one hash form a
|
||||
/// range whatever order their names are in). Only the nodes whose key
|
||||
/// interval overlaps the range are read: O(depth) nodes plus those holding
|
||||
/// the matches. Matches come in tree order.
|
||||
pub fn find_btree_v2_records(
|
||||
file_data: &[u8],
|
||||
header: &BTreeV2Header,
|
||||
offset_size: u8,
|
||||
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||
) -> Result<Vec<BTreeV2Record>, FormatError> {
|
||||
if header.total_records == 0 || header.num_records_in_root == 0 {
|
||||
return Ok(Vec::new());
|
||||
}
|
||||
if header.depth > MAX_DEPTH {
|
||||
return Err(FormatError::NestingDepthExceeded);
|
||||
}
|
||||
// As in `collect_btree_v2_records`: a valid tree cannot hold more
|
||||
// records than the file has room for, however its children are shared.
|
||||
let mut budget = file_data.len() / usize::from(header.record_size.max(1));
|
||||
let max_leaf_nrec = max_records_leaf(header.node_size, header.record_size);
|
||||
let mut out = Vec::new();
|
||||
find_in_node(
|
||||
file_data,
|
||||
header,
|
||||
to_usize(header.root_node_address)?,
|
||||
header.num_records_in_root,
|
||||
header.depth,
|
||||
offset_size,
|
||||
max_leaf_nrec,
|
||||
cmp,
|
||||
&mut budget,
|
||||
&mut out,
|
||||
)?;
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn find_in_node(
|
||||
file_data: &[u8],
|
||||
header: &BTreeV2Header,
|
||||
offset: usize,
|
||||
num_records: u16,
|
||||
depth: u16,
|
||||
offset_size: u8,
|
||||
max_leaf_nrec: u64,
|
||||
cmp: &mut dyn FnMut(&[u8]) -> Ordering,
|
||||
budget: &mut usize,
|
||||
out: &mut Vec<BTreeV2Record>,
|
||||
) -> Result<(), FormatError> {
|
||||
spend(budget, usize::from(num_records))?;
|
||||
if depth == 0 {
|
||||
let records = parse_leaf_records(file_data, offset, num_records, header.record_size)?;
|
||||
out.extend(
|
||||
records
|
||||
.into_iter()
|
||||
.filter(|r| cmp(&r.data) == Ordering::Equal),
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
let rs = usize::from(header.record_size);
|
||||
let (records_start, children) = read_internal_node(
|
||||
file_data,
|
||||
offset,
|
||||
num_records,
|
||||
depth,
|
||||
header.record_size,
|
||||
header.node_size,
|
||||
offset_size,
|
||||
max_leaf_nrec,
|
||||
)?;
|
||||
let nr = usize::from(num_records);
|
||||
let mut order = Vec::with_capacity(nr);
|
||||
for i in 0..nr {
|
||||
order.push(cmp(internal_record(file_data, records_start, i, rs)?));
|
||||
}
|
||||
// Child `i` holds the keys between record `i - 1` and record `i`: it can
|
||||
// hold a match unless the record before it is already past the range or
|
||||
// the record after it is still before it.
|
||||
for (i, &(child_addr, child_nrec)) in children.iter().enumerate() {
|
||||
let after_left = i == 0 || order[i - 1] != Ordering::Greater;
|
||||
let before_right = i == nr || order[i] != Ordering::Less;
|
||||
if after_left && before_right {
|
||||
find_in_node(
|
||||
file_data,
|
||||
header,
|
||||
to_usize(child_addr)?,
|
||||
child_nrec,
|
||||
depth - 1,
|
||||
offset_size,
|
||||
max_leaf_nrec,
|
||||
cmp,
|
||||
budget,
|
||||
out,
|
||||
)?;
|
||||
}
|
||||
if i < nr && order[i] == Ordering::Equal {
|
||||
out.push(BTreeV2Record {
|
||||
data: internal_record(file_data, records_start, i, rs)?.to_vec(),
|
||||
});
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Most records a subtree whose root is at `depth` can hold (libhdf5's
|
||||
/// `cum_max_nrec`). See [`node_info`].
|
||||
fn cum_max_records(
|
||||
|
||||
Reference in New Issue
Block a user