feat(format): choose chunk dimensions automatically for large datasets
Requesting a filter without chunk dimensions made the whole dataset a single chunk. Any read, even one row, then decompresses everything, and a large dataset cannot be decoded in parallel — which also made the new partial reads pointless for such files. auto_chunk_dims keeps datasets up to 1 MiB as one chunk (unchanged behaviour) and splits larger ones by halving the dimensions in turn, so chunks keep roughly the dataset's proportions, until a chunk is at most 1 MiB — h5py's approach. An empty (unlimited, unwritten) dimension is treated as 1024. The writer passes the element size through resolve_chunk_dims_for; the old resolve_chunk_dims assumes 8-byte elements. Explicit with_chunks always wins. Interop test: h5py reads an auto-chunked 13 MB deflate dataset, sees chunks between 128 KiB and 1 MiB, and a small dataset still has one chunk. Co-Authored-By: Claude Fable 5.1 <[email protected]>
This commit is contained in:
co-authored by
Claude Fable 5.1
parent
b36c6ec2af
commit
05c665a898
@@ -979,3 +979,62 @@ with h5py.File("{path_str}", "r") as f:
|
||||
expected["slab"]
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// clawhdf5 auto-chunks a large compressed dataset -> h5py reads
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// Compression without explicit chunk dimensions used to store the whole
|
||||
/// dataset as a single chunk. Large datasets are now split automatically;
|
||||
/// h5py must read the result and see sensibly sized chunks.
|
||||
#[test]
|
||||
fn clawhdf5_auto_chunked_dataset_h5py_reads() {
|
||||
skip_if_no_python!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("auto_chunk.h5");
|
||||
let path_str = path.display().to_string();
|
||||
|
||||
let (rows, cols) = (1500u64, 1100u64); // 13.2 MB of f64
|
||||
let data: Vec<f64> = (0..rows * cols).map(|i| (i % 9973) as f64 * 0.25).collect();
|
||||
let mut builder = FileBuilder::new();
|
||||
builder
|
||||
.create_dataset("big")
|
||||
.with_f64_data(&data)
|
||||
.with_shape(&[rows, cols])
|
||||
.with_deflate(4);
|
||||
builder
|
||||
.create_dataset("small")
|
||||
.with_f64_data(&data[..600])
|
||||
.with_shape(&[20, 30])
|
||||
.with_deflate(4);
|
||||
builder.write(&path).unwrap();
|
||||
|
||||
let out = run_python_output(&format!(
|
||||
r#"
|
||||
import h5py, numpy as np
|
||||
with h5py.File("{path_str}", "r") as f:
|
||||
big, small = f["big"], f["small"]
|
||||
expect = (np.arange(1500 * 1100) % 9973) * 0.25
|
||||
ok = bool(np.array_equal(big[...].ravel(), expect)) and bool(np.array_equal(small[...].ravel(), expect[:600]))
|
||||
chunk_bytes = int(np.prod(big.chunks)) * 8
|
||||
print(ok, chunk_bytes <= 1 << 20, chunk_bytes >= 1 << 17, small.chunks == (20, 30), big.compression)
|
||||
"#
|
||||
));
|
||||
assert_eq!(out.trim(), "True True True True gzip");
|
||||
|
||||
// And it reads back here, in full and partially.
|
||||
let file = File::open(&path).unwrap();
|
||||
let ds = file.dataset("big").unwrap();
|
||||
assert_eq!(ds.read_f64().unwrap(), data);
|
||||
let row = clawhdf5_format::selection::Selection::Hyperslab {
|
||||
start: vec![777, 0],
|
||||
stride: vec![1, 1],
|
||||
count: vec![1, cols],
|
||||
block: vec![1, 1],
|
||||
};
|
||||
let start = (777 * cols) as usize;
|
||||
assert_eq!(
|
||||
ds.read_f64_selection(&row).unwrap(),
|
||||
data[start..start + cols as usize]
|
||||
);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user