fix(format): read Extensible Array chunk indexes correctly
A dataset with exactly one unlimited dimension — the ordinary append-only case — is indexed by an Extensible Array. Only its first few chunk entries (4 by default) sit inline in the index block, and everything past them was read with the wrong layout. In the default shape the 37th chunk onward came back from the wrong place: a 400-chunk dataset returned 364 wrong values while reporting success, and beyond about a thousand chunks the read failed outright. Silently wrong data is the worse half of that. It survived because the only Extensible Array fixture in the suite had three chunks — inside the inline limit — so no test ever reached a data block. Four layout errors, each confirmed against files written by HDF5 2.0 and against the library source rather than inferred: - super block `u` owns 2^(u/2) data blocks, not 2^u; - each holds 2^((u+1)/2) * data_blk_min_elmts elements — the two quantities double every *other* level, a half step apart; - a super block carries a block-offset field before its data block addresses, which was not skipped; - the page-init bitmap belongs to the super block, one bit per page packed across all of its data blocks and read MSB-first, rather than living inside the data block; a paged data block also ends its prefix with a checksum before the first page. Where the spec left room for doubt the file settled it: decoding a paged block's elements and reading the chunk values they address identifies the mapping exactly, and the bitmap's 68 set bits matched the 34 data blocks x 2 pages that 200 000 elements need, which only holds MSB-first. New interop tests cross every boundary — 4, 37, 400, 5 000 and 200 000 chunks, the last with paged data blocks — plus sparse (uninitialised pages taking fill values), gzip-filtered elements and a 2-D dataset. All three fail against the old traversal. Writing is untouched; this was a read-path bug. Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
This commit is contained in:
@@ -1091,3 +1091,153 @@ with h5py.File("{path_str}", "w", libver="latest") as f:
|
||||
assert_eq!(v, i as i32, "element {i}");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn h5py_extensible_array_chunk_index_clawhdf5_reads() {
|
||||
// One unlimited dimension means an Extensible Array chunk index. Only its
|
||||
// first few elements live inline in the index block (4 by default), and
|
||||
// every other fixture here is small enough to stop there — which is how
|
||||
// the data block and super block layouts came to be wrong without a test
|
||||
// noticing. The counts below step over each boundary in turn:
|
||||
// 4 inline elements only
|
||||
// 37 past the first direct data block
|
||||
// 400 into the first super block
|
||||
// 5000 several super block levels
|
||||
// 200000 data blocks large enough to be paged
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
|
||||
for n in [4usize, 37, 400, 5_000, 200_000] {
|
||||
let path = dir.path().join(format!("ea_{n}.h5"));
|
||||
let path_str = path.display().to_string();
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py, numpy as np
|
||||
with h5py.File("{path_str}", "w", libver="latest") as f:
|
||||
d = f.create_dataset("x", shape=({n},), maxshape=(None,), chunks=(1,), dtype="i4")
|
||||
d[...] = np.arange({n}, dtype="i4")
|
||||
"#
|
||||
));
|
||||
|
||||
let bytes = std::fs::read(&path).unwrap();
|
||||
assert!(
|
||||
bytes.windows(4).any(|w| w == b"EAHD"),
|
||||
"n={n}: fixture is not indexed by an Extensible Array"
|
||||
);
|
||||
|
||||
let file = File::open(&path).unwrap();
|
||||
let values = file.dataset("x").unwrap().read_i32().unwrap();
|
||||
assert_eq!(values.len(), n, "n={n}");
|
||||
let wrong = values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|&(i, &v)| v != i as i32)
|
||||
.count();
|
||||
assert_eq!(wrong, 0, "n={n}: {wrong} of {n} elements read back wrong");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn h5py_sparse_extensible_array_leaves_pages_uninitialised() {
|
||||
// Writing a scattered subset leaves whole pages of a paged data block
|
||||
// never initialised. Those pages still occupy their slot on disk, so the
|
||||
// reader has to skip them by stride and take the fill value instead —
|
||||
// driven by the page-init bitmap, which is packed one bit per page across
|
||||
// the whole super block, MSB first.
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("ea_sparse.h5");
|
||||
let path_str = path.display().to_string();
|
||||
let n = 200_000usize;
|
||||
let step = 997usize;
|
||||
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py
|
||||
with h5py.File("{path_str}", "w", libver="latest") as f:
|
||||
d = f.create_dataset("x", shape=({n},), maxshape=(None,), chunks=(1,),
|
||||
dtype="i4", fillvalue=-1)
|
||||
for i in list(range(0, {n}, {step})) + list(range(0, 40)):
|
||||
d[i] = i
|
||||
"#
|
||||
));
|
||||
|
||||
let file = File::open(&path).unwrap();
|
||||
let values = file.dataset("x").unwrap().read_i32().unwrap();
|
||||
assert_eq!(values.len(), n);
|
||||
let wrong = values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|&(i, &v)| {
|
||||
let expected = if i % step == 0 || i < 40 {
|
||||
i as i32
|
||||
} else {
|
||||
-1
|
||||
};
|
||||
v != expected
|
||||
})
|
||||
.count();
|
||||
assert_eq!(wrong, 0, "{wrong} of {n} elements read back wrong");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn h5py_filtered_and_2d_extensible_array_clawhdf5_reads() {
|
||||
// Filtered elements carry a size and filter mask beside the address, and
|
||||
// a second (fixed) dimension changes how a linear index maps back to
|
||||
// chunk offsets. Both run through the same traversal.
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
|
||||
let gz = dir.path().join("ea_gzip.h5");
|
||||
let gz_str = gz.display().to_string();
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py, numpy as np
|
||||
with h5py.File("{gz_str}", "w", libver="latest") as f:
|
||||
d = f.create_dataset("x", shape=(5000,), maxshape=(None,), chunks=(1,),
|
||||
dtype="i4", compression="gzip", compression_opts=4)
|
||||
d[...] = np.arange(5000, dtype="i4")
|
||||
"#
|
||||
));
|
||||
let values = File::open(&gz)
|
||||
.unwrap()
|
||||
.dataset("x")
|
||||
.unwrap()
|
||||
.read_i32()
|
||||
.unwrap();
|
||||
assert_eq!(values.len(), 5000);
|
||||
assert_eq!(
|
||||
values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|&(i, &v)| v != i as i32)
|
||||
.count(),
|
||||
0
|
||||
);
|
||||
|
||||
let two_d = dir.path().join("ea_2d.h5");
|
||||
let two_d_str = two_d.display().to_string();
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py, numpy as np
|
||||
with h5py.File("{two_d_str}", "w", libver="latest") as f:
|
||||
d = f.create_dataset("x", shape=(3000, 4), maxshape=(None, 4), chunks=(1, 4), dtype="i4")
|
||||
d[...] = np.arange(12000, dtype="i4").reshape(3000, 4)
|
||||
"#
|
||||
));
|
||||
let values = File::open(&two_d)
|
||||
.unwrap()
|
||||
.dataset("x")
|
||||
.unwrap()
|
||||
.read_i32()
|
||||
.unwrap();
|
||||
assert_eq!(values.len(), 12_000);
|
||||
assert_eq!(
|
||||
values
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|&(i, &v)| v != i as i32)
|
||||
.count(),
|
||||
0
|
||||
);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user