diff --git a/crates/clawhdf5-format/src/chunked_write.rs b/crates/clawhdf5-format/src/chunked_write.rs index 89b0f7f..d750fd8 100644 --- a/crates/clawhdf5-format/src/chunked_write.rs +++ b/crates/clawhdf5-format/src/chunked_write.rs @@ -984,7 +984,13 @@ pub fn build_chunked_data_from_precompressed( base_address: u64, maxshape: Option<&[u64]>, ) -> Result { - build_chunked_data_from_precompressed_libver(pre, base_address, maxshape, LibVer::Latest) + build_chunked_data_from_precompressed_libver( + pre, + base_address, + maxshape, + LibVer::Latest, + LibVer::Latest, + ) } /// [`build_chunked_data_from_precompressed`] for a file whose low library @@ -992,13 +998,28 @@ pub fn build_chunked_data_from_precompressed( /// every chunked dataset gets a version-3 layout message and a version-1 /// B-tree chunk index, whatever its maximum shape, as libhdf5 writes it; /// otherwise the version-4 layout and the index libhdf5 picks for it. +/// +/// A chunk of 4 GiB or more (over [`MAX_V4_CHUNK_BYTES`]) takes layout +/// message version 5 whatever `low` is, as in libhdf5 (a version-1 B-tree +/// key holds a 32-bit size), and needs a `high` bound of at least +/// [`LibVer::V200`] ([`FormatError::LibverBound`] otherwise). pub fn build_chunked_data_from_precompressed_libver( pre: &PrecompressedChunks, base_address: u64, maxshape: Option<&[u64]>, low: LibVer, + high: LibVer, ) -> Result { - if low < LibVer::V110 { + let (_, chunk_bytes) = checked_chunk_dims(&pre.chunk_dims, pre.element_size)?; + if chunk_bytes > MAX_V4_CHUNK_BYTES { + if high < LibVer::V200 { + return Err(FormatError::LibverBound { + what: format!("a chunk of {chunk_bytes} bytes (4 GiB or more)"), + needs: LibVer::V200, + high, + }); + } + } else if low < LibVer::V110 { return build_btree_v1_chunked_data(pre, base_address, maxshape); } let index = ChunkIndexPlan::new(&pre.shape, maxshape, &pre.chunk_dims)?; @@ -2102,6 +2123,38 @@ mod tests { assert_eq!(layout_version_for(HUGE_BYTES), 5); } + /// A chunk of 4 GiB or more never goes into a version-1 B-tree (its key + /// holds a 32-bit size): under a 1.8 low bound it still takes layout + /// version 5, and a high bound below 2.0 refuses it, as in libhdf5. + #[test] + fn huge_chunks_ignore_the_v18_low_bound_and_need_v200() { + // One filtered chunk; the stored bytes stand in for its compression. + let pre = PrecompressedChunks { + chunks: vec![(HUGE_BYTES, vec![0u8; 16], 0)], + has_filters: true, + element_size: 8, + shape: vec![HUGE_DIM], + chunk_dims: vec![HUGE_DIM], + pipeline_message: None, + }; + let r = build_chunked_data_from_precompressed_libver( + &pre, + 4096, + None, + LibVer::V18, + LibVer::Latest, + ) + .unwrap(); + assert_eq!(r.layout_message[0], 5, "layout message version"); + for high in [LibVer::V18, LibVer::V114] { + match build_chunked_data_from_precompressed_libver(&pre, 4096, None, LibVer::V18, high) + { + Err(FormatError::LibverBound { needs, .. }) => assert_eq!(needs, LibVer::V200), + other => panic!("high bound {high}: {:?}", other.map(|r| r.layout_message)), + } + } + } + /// `H5D_FARRAY_FILT_COMPUTE_CHUNK_SIZE_LEN` (and its EA and v2 B-tree /// twins) in libhdf5 2.2.0: one byte more than the chunk size needs up to /// layout version 4, the size of lengths from version 5. diff --git a/crates/clawhdf5-format/src/file_writer.rs b/crates/clawhdf5-format/src/file_writer.rs index 6dcfe47..4eb59ee 100644 --- a/crates/clawhdf5-format/src/file_writer.rs +++ b/crates/clawhdf5-format/src/file_writer.rs @@ -1517,7 +1517,9 @@ impl FileWriter { /// (virtual datasets, a paged file, the 1.12 reference types, native /// complex numbers). With a low bound of 1.8 and a later high bound /// such objects are written in the newer format, as libhdf5 writes them; - /// the rest of the file stays readable by 1.8. + /// the rest of the file stays readable by 1.8. A chunk of 4 GiB or more + /// always takes HDF5 2.0's version-5 layout (never a version-1 B-tree) + /// and so needs a high bound of at least [`LibVer::V200`]. /// /// A low bound above the high bound makes [`Self::finish`] fail. pub fn libver_bounds(&mut self, low: LibVer, high: LibVer) -> &mut Self { @@ -1830,6 +1832,7 @@ impl FileWriter { dummy_cursor, d.maxshape.as_deref(), low, + high, )?; dummy_cursor += result.data_bytes.len() as u64; let oh = build_chunked_dataset_oh( @@ -1991,6 +1994,7 @@ impl FileWriter { base_address, d.maxshape.as_deref(), low, + high, )?; cursor2 += result.data_bytes.len(); let oh = build_chunked_dataset_oh(