format: group walks go on past a failed node and hint what they read next
Listing a large group over openUrl still took 6-11 passes (network round
trips) for the reviewer's 3000-dataset h5py file: each pass only found
the structures the walk reached before its first miss.
- The v1 and v2 B-tree collectors descend into every child of a node
after one fails (they only read the siblings before, so a sibling's
subtree came a pass later), then return the first error: results and
errors unchanged. The v2 walk stops once its record budget is spent,
so a shared-subtree tree still cannot multiply the work.
- Hints (`Storage::hint`, a no-op for every backend but the lazy one):
a group B-tree node's and a symbol table node's body (read once their
header gives a length, a round trip later when the body is in the
next block), an object header's first chunk and its continuation
chunks, the symbol table nodes a B-tree leaf names, a dense group's
name index header and the heap's root block (both read right after
the heap header). A listing also hints every child's object header as
its entry is read, even after a failure, and every direct block of a
dense group's heap (reading the indirect blocks, at most 4096 entries
and 4 levels deep); a lookup does not.
- The fractal heap's indirect-block layout (entry sizes, where the first
n entries end) is one helper used by the object reads and the hints.
Measured with tests/lazy.rs listing_cost_of_a_given_file on an h5py file
like the reviewer's (3000 datasets of 64 KiB, 198 MB), list('/'),
passes/requests/bytes, before -> after:
earliest, 1 MiB: 6/73/192.5 MB -> 4/68/192.5 MB
earliest, 64 KiB: 8/531/35.2 MB -> 5/530/35.3 MB
latest, 1 MiB: 9/98/196.5 MB -> 5/86/196.5 MB
latest, 64 KiB: 11/452/29.6 MB -> 6/454/30.5 MB
listing_a_large_group_takes_a_few_passes (512-byte blocks), budgets
tightened to the new counts: FileBuilder 600 children 5 -> 4 passes,
h5py 2000 children earliest 8 -> 5, latest 11 -> 6.
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -146,8 +146,8 @@ fn transcript(api: &impl Api) -> Vec<String> {
|
||||
/// takes), and agrees with the in-memory one: the same values, and an error
|
||||
/// wherever it has one (a malformed file can fail at a different check,
|
||||
/// with a different message, when read by ranges). Returns what the lazy
|
||||
/// reader fetched and its transcript.
|
||||
fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, Vec<String>) {
|
||||
/// reader fetched (requests, bytes, passes) and its transcript.
|
||||
fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, u64, Vec<String>) {
|
||||
let ctx = format!("{name} (blocks of {block} B)");
|
||||
let ranged = Reader::open_storage(Arc::new(CountingStorage::new(data.to_vec())));
|
||||
let local = Reader::open(data.to_vec());
|
||||
@@ -156,7 +156,7 @@ fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, Vec<String>) {
|
||||
(Ok(r), Ok(l), Ok(z)) => (r, l, z),
|
||||
(Err(r), Err(_), Err(z)) => {
|
||||
assert_eq!(z, r, "{ctx}: open error");
|
||||
return (0, 0, Vec::new());
|
||||
return (0, 0, 0, Vec::new());
|
||||
}
|
||||
(r, l, z) => panic!(
|
||||
"{ctx}: opens differently: ranged {:?}, in memory {:?}, lazily {:?}",
|
||||
@@ -184,7 +184,7 @@ fn check_equal(name: &str, data: &[u8], block: u64) -> (u64, u64, Vec<String>) {
|
||||
}
|
||||
assert_eq!(got.len(), local.len(), "{ctx}: transcript length");
|
||||
let st = lazy.storage.stats();
|
||||
(st.requests, st.bytes_fetched, got)
|
||||
(st.requests, st.bytes_fetched, st.passes, got)
|
||||
}
|
||||
|
||||
fn config(block: u64) -> LazyConfig {
|
||||
@@ -223,7 +223,7 @@ fn builder_file() -> Vec<u8> {
|
||||
fn builder_files_read_the_same_at_every_block_size() {
|
||||
let data = builder_file();
|
||||
for block in [512, 4096, 1 << 20] {
|
||||
let (requests, _, lines) = check_equal("builder", &data, block);
|
||||
let (requests, _, _, lines) = check_equal("builder", &data, block);
|
||||
assert!(requests > 0);
|
||||
// The transcript covers every object, values included.
|
||||
assert!(lines.iter().any(|l| l.starts_with("/grid read: Ok")));
|
||||
@@ -315,7 +315,7 @@ fn h5py_and_netcdf4_files_read_the_same_lazily() {
|
||||
for name in ["fixture.h5", "fixture.nc"] {
|
||||
let data = std::fs::read(dir.path().join(name)).unwrap();
|
||||
for block in [512, 64 * 1024] {
|
||||
let (_, _, lines) = check_equal(name, &data, block);
|
||||
let (_, _, _, lines) = check_equal(name, &data, block);
|
||||
assert!(lines.iter().filter(|l| l.contains(" read: Ok")).count() >= 2);
|
||||
}
|
||||
}
|
||||
@@ -460,16 +460,17 @@ fn corpus_files_read_the_same_lazily() {
|
||||
}
|
||||
files.sort();
|
||||
assert!(!files.is_empty(), "no HDF5 files under {dirs}");
|
||||
let (mut requests, mut bytes, mut total) = (0u64, 0u64, 0u64);
|
||||
let (mut requests, mut bytes, mut passes, mut total) = (0u64, 0u64, 0u64, 0u64);
|
||||
for f in &files {
|
||||
let data = std::fs::read(f).unwrap();
|
||||
total += data.len() as u64;
|
||||
let (r, b, _) = check_equal(&f.display().to_string(), &data, 64 * 1024);
|
||||
let (r, b, p, _) = check_equal(&f.display().to_string(), &data, 64 * 1024);
|
||||
requests += r;
|
||||
bytes += b;
|
||||
passes += p;
|
||||
}
|
||||
eprintln!(
|
||||
"{} files ({total} bytes): {requests} requests, {bytes} bytes fetched",
|
||||
"{} files ({total} bytes): {passes} passes, {requests} requests, {bytes} bytes fetched",
|
||||
files.len()
|
||||
);
|
||||
}
|
||||
@@ -496,13 +497,38 @@ fn listing_cost(data: &[u8], path: &str, block: u64) -> (u64, u64) {
|
||||
)
|
||||
}
|
||||
|
||||
/// An h5py file of `n` datasets of 256 `f32` each (`d0` ... ) in the root
|
||||
/// group, written with `libver`.
|
||||
fn h5py_many(dir: &Path, libver: &str, n: usize) -> Vec<u8> {
|
||||
let path = dir.join(format!("{libver}_{n}.h5"));
|
||||
let script = format!(
|
||||
"import h5py, numpy as np\n\
|
||||
with h5py.File({:?}, 'w', libver='{libver}') as f:\n\
|
||||
\x20 for i in range({n}):\n\
|
||||
\x20 f.create_dataset('d%d' % i, data=np.full(256, i, np.float32))\n",
|
||||
path.display().to_string()
|
||||
);
|
||||
let out = Command::new(python())
|
||||
.args(["-c", &script])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"{}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
std::fs::read(&path).unwrap()
|
||||
}
|
||||
|
||||
/// Listing a group reads every child's object header, and its index (B-tree
|
||||
/// and symbol table nodes, or B-tree v2 and heap blocks) before that. Each
|
||||
/// pass asks for every node of a level it is missing, not the first one
|
||||
/// only, so the passes (network round trips) grow with the depth of the
|
||||
/// index, not with the number of children: 2000 children with headers
|
||||
/// scattered over 512-byte blocks list in a handful of passes, where each
|
||||
/// header block used to cost its own.
|
||||
/// pass asks for every node of the index it can reach (a failed node does
|
||||
/// not stop the walk), the blocks it has been told it reads next (a node's
|
||||
/// body, a symbol table node's entries, the heap's blocks, each child's
|
||||
/// header: `Storage::hint`), so the passes (network round trips) follow the
|
||||
/// depth of the index, not the number of children: 2000 children with
|
||||
/// headers scattered over 512-byte blocks list in a handful of passes,
|
||||
/// where each header block used to cost its own.
|
||||
#[test]
|
||||
fn listing_a_large_group_takes_a_few_passes() {
|
||||
let mut b = FileBuilder::new();
|
||||
@@ -514,39 +540,31 @@ fn listing_a_large_group_takes_a_few_passes() {
|
||||
let data = b.finish().unwrap();
|
||||
let (passes, requests) = listing_cost(&data, "/many", 512);
|
||||
eprintln!("FileBuilder, 600 children: {passes} passes, {requests} requests");
|
||||
assert!(passes <= 6, "{passes} passes");
|
||||
// 5 before hints (2026-09-27), 102 before the walks went on past a miss.
|
||||
assert!(passes <= 4, "{passes} passes");
|
||||
|
||||
if !python_available() {
|
||||
assert!(
|
||||
!std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1"),
|
||||
"CLAWHDF5_REQUIRE_INTEROP=1 but {} lacks h5py/netCDF4/numpy",
|
||||
python()
|
||||
);
|
||||
return;
|
||||
}
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
for libver in ["earliest", "latest"] {
|
||||
let path = dir.path().join(format!("{libver}.h5"));
|
||||
let script = format!(
|
||||
"import h5py, numpy as np\n\
|
||||
with h5py.File({:?}, 'w', libver='{libver}') as f:\n\
|
||||
\x20 for i in range(2000):\n\
|
||||
\x20 f.create_dataset('d%d' % i, data=np.full(256, i, np.float32))\n",
|
||||
path.display().to_string()
|
||||
);
|
||||
let out = Command::new(python())
|
||||
.args(["-c", &script])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"{}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
let data = std::fs::read(&path).unwrap();
|
||||
// Most passes each may take: 8 and 11 before hints (2026-09-27).
|
||||
for (libver, most) in [("earliest", 5), ("latest", 6)] {
|
||||
let data = h5py_many(dir.path(), libver, 2000);
|
||||
let (passes, requests) = listing_cost(&data, "/", 512);
|
||||
eprintln!("h5py libver={libver}, 2000 children: {passes} passes, {requests} requests");
|
||||
assert!(passes <= 12, "{libver}: {passes} passes");
|
||||
assert!(passes <= most, "{libver}: {passes} passes");
|
||||
}
|
||||
}
|
||||
|
||||
/// `CLAWHDF5_WASM_LIST_FILE=file.h5`: what listing the root group of that
|
||||
/// file costs lazily, at 1 MiB and 64 KiB blocks (a measurement, printed).
|
||||
/// file costs lazily, and opening it and reading one dataset whole
|
||||
/// (`CLAWHDF5_WASM_READ`, by default the middle dataset of the listing),
|
||||
/// at 1 MiB and 64 KiB blocks (a measurement, printed).
|
||||
#[test]
|
||||
fn listing_cost_of_a_given_file() {
|
||||
let Ok(path) = std::env::var("CLAWHDF5_WASM_LIST_FILE") else {
|
||||
@@ -563,16 +581,44 @@ fn listing_cost_of_a_given_file() {
|
||||
)
|
||||
.unwrap();
|
||||
let open = lazy.storage.stats();
|
||||
let n = lazy.call(|r| r.list("/")).unwrap().len();
|
||||
let list = lazy.call(|r| r.list("/")).unwrap();
|
||||
let st = lazy.storage.stats();
|
||||
eprintln!(
|
||||
"{path} ({} bytes), {block}-byte blocks: open {} requests / {} passes; list('/') of {n}: {} passes, {} requests, {} bytes",
|
||||
"{path} ({} bytes), {block}-byte blocks: open {} requests / {} passes; list('/') of {}: {} passes, {} requests, {} bytes ({} blocks hinted)",
|
||||
data.len(),
|
||||
open.requests,
|
||||
open.passes,
|
||||
list.len(),
|
||||
st.passes - open.passes,
|
||||
st.requests - open.requests,
|
||||
st.bytes_fetched - open.bytes_fetched
|
||||
st.bytes_fetched - open.bytes_fetched,
|
||||
st.hinted_blocks - open.hinted_blocks,
|
||||
);
|
||||
let name = std::env::var("CLAWHDF5_WASM_READ").unwrap_or_else(|_| {
|
||||
let datasets: Vec<_> = list.iter().filter(|c| c.kind == Kind::Dataset).collect();
|
||||
format!("/{}", datasets[datasets.len() / 2].name)
|
||||
});
|
||||
// Open and read on a fresh cache: the open's own cost (the probe
|
||||
// block and its passes) and then the read's.
|
||||
let fresh = Lazy::open(
|
||||
data.clone(),
|
||||
LazyConfig {
|
||||
block_size: block,
|
||||
..LazyConfig::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
let open = fresh.storage.stats();
|
||||
let values = fresh.call(|r| r.read(&name, None)).unwrap().data.len();
|
||||
let st = fresh.storage.stats();
|
||||
eprintln!(
|
||||
" open + read('{name}') ({values} values): {} passes, {} requests, {} bytes (the open: {} passes, {} requests, {} bytes)",
|
||||
st.passes,
|
||||
st.requests,
|
||||
st.bytes_fetched,
|
||||
open.passes,
|
||||
open.requests,
|
||||
open.bytes_fetched,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user