Chunks of 4 GiB or more: read in every index, write as libhdf5 2.x does #29

Open
osobh wants to merge 10 commits from feat/huge-chunks into main
2 changed files with 60 additions and 3 deletions
Showing only changes of commit b55768ff76 - Show all commits
+55 -2
View File
@@ -984,7 +984,13 @@ pub fn build_chunked_data_from_precompressed(
base_address: u64,
maxshape: Option<&[u64]>,
) -> Result<ChunkedDataResult, FormatError> {
build_chunked_data_from_precompressed_libver(pre, base_address, maxshape, LibVer::Latest)
build_chunked_data_from_precompressed_libver(
pre,
base_address,
maxshape,
LibVer::Latest,
LibVer::Latest,
)
}
/// [`build_chunked_data_from_precompressed`] for a file whose low library
@@ -992,13 +998,28 @@ pub fn build_chunked_data_from_precompressed(
/// every chunked dataset gets a version-3 layout message and a version-1
/// B-tree chunk index, whatever its maximum shape, as libhdf5 writes it;
/// otherwise the version-4 layout and the index libhdf5 picks for it.
///
/// A chunk of 4 GiB or more (over [`MAX_V4_CHUNK_BYTES`]) takes layout
/// message version 5 whatever `low` is, as in libhdf5 (a version-1 B-tree
/// key holds a 32-bit size), and needs a `high` bound of at least
/// [`LibVer::V200`] ([`FormatError::LibverBound`] otherwise).
pub fn build_chunked_data_from_precompressed_libver(
pre: &PrecompressedChunks,
base_address: u64,
maxshape: Option<&[u64]>,
low: LibVer,
high: LibVer,
) -> Result<ChunkedDataResult, FormatError> {
if low < LibVer::V110 {
let (_, chunk_bytes) = checked_chunk_dims(&pre.chunk_dims, pre.element_size)?;
if chunk_bytes > MAX_V4_CHUNK_BYTES {
if high < LibVer::V200 {
return Err(FormatError::LibverBound {
what: format!("a chunk of {chunk_bytes} bytes (4 GiB or more)"),
needs: LibVer::V200,
high,
});
}
} else if low < LibVer::V110 {
return build_btree_v1_chunked_data(pre, base_address, maxshape);
}
let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?;
@@ -2102,6 +2123,38 @@ mod tests {
assert_eq!(layout_version_for(HUGE_BYTES), 5);
}
/// A chunk of 4 GiB or more never goes into a version-1 B-tree (its key
/// holds a 32-bit size): under a 1.8 low bound it still takes layout
/// version 5, and a high bound below 2.0 refuses it, as in libhdf5.
#[test]
fn huge_chunks_ignore_the_v18_low_bound_and_need_v200() {
// One filtered chunk; the stored bytes stand in for its compression.
let pre = PrecompressedChunks {
chunks: vec![(HUGE_BYTES, vec![0u8; 16], 0)],
has_filters: true,
element_size: 8,
shape: vec![HUGE_DIM],
chunk_dims: vec![HUGE_DIM],
pipeline_message: None,
};
let r = build_chunked_data_from_precompressed_libver(
&pre,
4096,
None,
LibVer::V18,
LibVer::Latest,
)
.unwrap();
assert_eq!(r.layout_message[0], 5, "layout message version");
for high in [LibVer::V18, LibVer::V114] {
match build_chunked_data_from_precompressed_libver(&pre, 4096, None, LibVer::V18, high)
{
Err(FormatError::LibverBound { needs, .. }) => assert_eq!(needs, LibVer::V200),
other => panic!("high bound {high}: {:?}", other.map(|r| r.layout_message)),
}
}
}
/// `H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` (and its EA and v2 B-tree
/// twins) in libhdf5 2.2.0: one byte more than the chunk size needs up to
/// layout version 4, the size of lengths from version 5.
+5 -1
View File
@@ -1517,7 +1517,9 @@ impl FileWriter {
/// (virtual datasets, a paged file, the 1.12 reference types, native
/// complex numbers). With a low bound of 1.8 and a later high bound
/// such objects are written in the newer format, as libhdf5 writes them;
/// the rest of the file stays readable by 1.8.
/// the rest of the file stays readable by 1.8. A chunk of 4 GiB or more
/// always takes HDF5 2.0's version-5 layout (never a version-1 B-tree)
/// and so needs a high bound of at least [`LibVer::V200`].
///
/// A low bound above the high bound makes [`Self::finish`] fail.
pub fn libver_bounds(&mut self, low: LibVer, high: LibVer) -> &mut Self {
@@ -1830,6 +1832,7 @@ impl FileWriter {
dummy_cursor,
d.maxshape.as_deref(),
low,
high,
)?;
dummy_cursor += result.data_bytes.len() as u64;
let oh = build_chunked_dataset_oh(
@@ -1991,6 +1994,7 @@ impl FileWriter {
base_address,
d.maxshape.as_deref(),
low,
high,
)?;
cursor2 += result.data_bytes.len();
let oh = build_chunked_dataset_oh(