feat(format): read Blosc2 (filter 32026) in pure Rust
hdf5plugin's Blosc2 was a clear "not implemented" error. It stores each HDF5 chunk as a Blosc2 contiguous frame; for chunks of 2+ dimensions the frame holds a B2ND array whose data are cut into (padded) blocks stored one after another. filters_blosc2 (feature `blosc2`, in `plugin-filters`; the facade forwards both) decodes, following c-blosc2's decoder and hdf5-blosc2's blosc2_filter.c: - the frame: the msgpack header's fixed fields, the metalayer index, the offsets chunk (with the special zero/NaN/uninitialised offsets) and chunk lookup; - Blosc2 chunks: 16- and 32-byte headers, special chunks (zeros, NaN, uninitialised, one repeated value), split and unsplit streams, zero and run-length streams, and the filter pipeline run backwards (shuffle, shuffle with a byte-group size, bit shuffle including the version-2 and later handling of a partial group of 8, delta against the first block, truncated precision); - the codecs, shared with Blosc 1: BloscLZ, LZ4/LZ4HC, Zlib, Zstandard; - B2ND arrays: blocks gathered into C order, padding dropped, several chunks per array, and the array shape checked against the chunk shape in cd_values as the HDF5 filter does. Dictionaries, lazy chunks, variable-length blocks, user-defined codecs and registered filters (e.g. bytedelta) are errors. Uninitialised chunks read as zeros. No encoder. Tests: h5py + hdf5plugin write every codec x filter (none, shuffle, bitshuffle, delta) and levels 0-9 over the plugin-filter cases, then i1..u8/f4/f8 in 1-D to 5-D chunks with partial edge chunks, datasets of zeros, one value and NaN, and Fletcher32 before Blosc2 (plain frames for n-D chunks); clawhdf5 reads each exactly as its unfiltered twin, and truncated precision exactly as h5py reads it. Fixture frames from python-blosc2 (tests/fixtures/blosc2/generate.py) cover what hdf5plugin never writes: special chunks, delta over many blocks and odd type sizes, odd bit-shuffle blocks, forced splitting, multi-chunk B2ND arrays with a zero chunk, and the refused features. The decoder is fuzzed (random and mutated frames and chunks: no panic, output within the limit). Conformance: h5ex_d_blosc2.h5 now reads (576 of 697 ok, baseline 575). Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -76,8 +76,10 @@ bitshuffle = ["lz4_flex", "ruzstd"]
|
|||||||
bzip2 = ["dep:bzip2", "std"]
|
bzip2 = ["dep:bzip2", "std"]
|
||||||
# Blosc 1 (32001) with its BloscLZ, LZ4, Snappy, Zlib and Zstandard codecs.
|
# Blosc 1 (32001) with its BloscLZ, LZ4, Snappy, Zlib and Zstandard codecs.
|
||||||
blosc = ["lz4_flex", "ruzstd", "snap", "deflate", "std"]
|
blosc = ["lz4_flex", "ruzstd", "snap", "deflate", "std"]
|
||||||
|
# Blosc2 (32026), read-only: frames, B2ND arrays, and the Blosc codecs above.
|
||||||
|
blosc2 = ["blosc"]
|
||||||
# Every plugin filter above.
|
# Every plugin filter above.
|
||||||
plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc"]
|
plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc", "blosc2"]
|
||||||
|
|
||||||
[[bench]]
|
[[bench]]
|
||||||
name = "parallel_decompress_bench"
|
name = "parallel_decompress_bench"
|
||||||
|
|||||||
@@ -5,7 +5,8 @@
|
|||||||
//! * **Built-in filters** — a static table of the filters compiled into this
|
//! * **Built-in filters** — a static table of the filters compiled into this
|
||||||
//! build: the HDF5 standard filters (deflate, shuffle, Fletcher32, szip,
|
//! build: the HDF5 standard filters (deflate, shuffle, Fletcher32, szip,
|
||||||
//! N-Bit, scale-offset) and the plugin filters whose cargo features are
|
//! N-Bit, scale-offset) and the plugin filters whose cargo features are
|
||||||
//! enabled (LZ4, Zstandard, pcodec, LZF, bitshuffle, bzip2, blosc).
|
//! enabled (LZ4, Zstandard, pcodec, LZF, bitshuffle, bzip2, blosc,
|
||||||
|
//! blosc2).
|
||||||
//! [`builtin_filters`] lists them.
|
//! [`builtin_filters`] lists them.
|
||||||
//! * **Registered filters** (`std` only) — codecs the application supplies
|
//! * **Registered filters** (`std` only) — codecs the application supplies
|
||||||
//! for any other ID with [`register_filter`] (a [`FilterCodec`], or just a
|
//! for any other ID with [`register_filter`] (a [`FilterCodec`], or just a
|
||||||
@@ -166,7 +167,7 @@ pub fn known_filter(id: u16) -> Option<(&'static str, Option<&'static str>)> {
|
|||||||
32019 => ("JPEG", None),
|
32019 => ("JPEG", None),
|
||||||
32022 => ("BitGroom", None),
|
32022 => ("BitGroom", None),
|
||||||
32023 => ("Granular BitRound", None),
|
32023 => ("Granular BitRound", None),
|
||||||
32026 => ("Blosc2", None),
|
32026 => ("Blosc2", Some("blosc2")),
|
||||||
_ => return None,
|
_ => return None,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
@@ -452,12 +453,12 @@ pub(crate) mod tests {
|
|||||||
#[test]
|
#[test]
|
||||||
fn unsupported_filter_error_names_the_filter() {
|
fn unsupported_filter_error_names_the_filter() {
|
||||||
let msg = FormatError::UnsupportedFilter(32026).to_string();
|
let msg = FormatError::UnsupportedFilter(32026).to_string();
|
||||||
|
assert!(msg.contains("Blosc2") && msg.contains("`blosc2`"), "{msg}");
|
||||||
|
let msg = FormatError::UnsupportedFilter(32013).to_string();
|
||||||
assert!(
|
assert!(
|
||||||
msg.contains("Blosc2") && msg.contains("not implemented"),
|
msg.contains("ZFP") && msg.contains("not implemented"),
|
||||||
"{msg}"
|
"{msg}"
|
||||||
);
|
);
|
||||||
let msg = FormatError::UnsupportedFilter(32013).to_string();
|
|
||||||
assert!(msg.contains("ZFP"), "{msg}");
|
|
||||||
let msg = FormatError::UnsupportedFilter(32000).to_string();
|
let msg = FormatError::UnsupportedFilter(32000).to_string();
|
||||||
assert!(msg.contains("LZF") && msg.contains("`lzf`"), "{msg}");
|
assert!(msg.contains("LZF") && msg.contains("`lzf`"), "{msg}");
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
|
|||||||
@@ -278,6 +278,13 @@ pub(crate) static BUILTIN_FILTERS: &[BuiltinFilter] = &[
|
|||||||
},
|
},
|
||||||
encode: None,
|
encode: None,
|
||||||
},
|
},
|
||||||
|
#[cfg(feature = "blosc2")]
|
||||||
|
BuiltinFilter {
|
||||||
|
id: crate::filter_pipeline::FILTER_BLOSC2,
|
||||||
|
name: "blosc2",
|
||||||
|
decode: crate::filters_blosc2::blosc2_decode,
|
||||||
|
encode: None,
|
||||||
|
},
|
||||||
];
|
];
|
||||||
|
|
||||||
/// Decode the HDF5 scale-offset filter (id 6).
|
/// Decode the HDF5 scale-offset filter (id 6).
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ fn le32(b: &[u8], at: usize) -> Result<usize, FormatError> {
|
|||||||
|
|
||||||
/// The codec inside a Blosc frame (flags bits 5-7).
|
/// The codec inside a Blosc frame (flags bits 5-7).
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
enum Codec {
|
pub(crate) enum Codec {
|
||||||
BloscLz,
|
BloscLz,
|
||||||
Lz4,
|
Lz4,
|
||||||
Snappy,
|
Snappy,
|
||||||
@@ -58,7 +58,7 @@ enum Codec {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl Codec {
|
impl Codec {
|
||||||
fn from_flags(flags: u8) -> Result<Codec, FormatError> {
|
pub(crate) fn from_flags(flags: u8) -> Result<Codec, FormatError> {
|
||||||
match flags >> 5 {
|
match flags >> 5 {
|
||||||
0 => Ok(Codec::BloscLz),
|
0 => Ok(Codec::BloscLz),
|
||||||
1 => Ok(Codec::Lz4),
|
1 => Ok(Codec::Lz4),
|
||||||
@@ -71,7 +71,7 @@ impl Codec {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Decode one codec stream into exactly `dst`.
|
/// Decode one codec stream into exactly `dst`.
|
||||||
fn decode_stream(
|
pub(crate) fn decode_stream(
|
||||||
codec: Codec,
|
codec: Codec,
|
||||||
src: &[u8],
|
src: &[u8],
|
||||||
dst: &mut [u8],
|
dst: &mut [u8],
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -86,6 +86,8 @@ pub mod filters;
|
|||||||
mod filters_bitshuffle;
|
mod filters_bitshuffle;
|
||||||
#[cfg(feature = "blosc")]
|
#[cfg(feature = "blosc")]
|
||||||
pub mod filters_blosc;
|
pub mod filters_blosc;
|
||||||
|
#[cfg(feature = "blosc2")]
|
||||||
|
pub mod filters_blosc2;
|
||||||
#[cfg(feature = "bzip2")]
|
#[cfg(feature = "bzip2")]
|
||||||
mod filters_bzip2;
|
mod filters_bzip2;
|
||||||
#[cfg(feature = "lzf")]
|
#[cfg(feature = "lzf")]
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1 @@
|
|||||||
|
filter 35
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,129 @@
|
|||||||
|
"""Generate the Blosc2 frames the `filters_blosc2` unit tests decode.
|
||||||
|
|
||||||
|
Each case is `<name>.b2f` (a Blosc2 contiguous frame, what the HDF5 Blosc2
|
||||||
|
filter stores per chunk) and `<name>.out` (what decoding it must give: the
|
||||||
|
first chunk of a plain frame, or the whole array in C order for a B2ND
|
||||||
|
frame), or `<name>.err` (a frame clawhdf5 must refuse; the file holds a word
|
||||||
|
the error message must contain).
|
||||||
|
|
||||||
|
These cover what files written by h5py + hdf5plugin never contain, but a
|
||||||
|
Blosc2 frame may: special chunks (repeated value, NaN, uninitialised), the
|
||||||
|
delta filter over many blocks and odd type sizes, bit shuffle of blocks that
|
||||||
|
are not a multiple of 8 elements, shuffle with a byte-group size, forced
|
||||||
|
stream splitting, multi-chunk B2ND arrays with padded edge chunks and a
|
||||||
|
chunk of zeros, and features clawhdf5 refuses (dictionaries, registered
|
||||||
|
filters).
|
||||||
|
|
||||||
|
Written with python-blosc2 4.13.1 (c-blosc2 3.3.4) in a scratch venv
|
||||||
|
(`pip install blosc2`). Re-run only to regenerate:
|
||||||
|
|
||||||
|
python generate.py <this directory>
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import blosc2
|
||||||
|
import numpy as np
|
||||||
|
|
||||||
|
out = sys.argv[1]
|
||||||
|
|
||||||
|
|
||||||
|
def save(name, frame, expected):
|
||||||
|
with open(os.path.join(out, name + ".b2f"), "wb") as f:
|
||||||
|
f.write(frame)
|
||||||
|
with open(os.path.join(out, name + ".out"), "wb") as f:
|
||||||
|
f.write(expected)
|
||||||
|
|
||||||
|
|
||||||
|
def save_err(name, frame, word):
|
||||||
|
with open(os.path.join(out, name + ".b2f"), "wb") as f:
|
||||||
|
f.write(frame)
|
||||||
|
with open(os.path.join(out, name + ".err"), "w") as f:
|
||||||
|
f.write(word)
|
||||||
|
|
||||||
|
|
||||||
|
def plain(data, **cparams):
|
||||||
|
"""A one-chunk super-chunk frame of `data`, as hdf5-blosc2 writes."""
|
||||||
|
data = np.ascontiguousarray(data)
|
||||||
|
cp = blosc2.CParams(typesize=data.dtype.itemsize, **cparams)
|
||||||
|
sc = blosc2.SChunk(chunksize=data.nbytes, cparams=cp)
|
||||||
|
sc.append_data(data)
|
||||||
|
return sc.to_cframe(), data.tobytes()
|
||||||
|
|
||||||
|
|
||||||
|
def special(nitems, dtype, kind, value=None):
|
||||||
|
dt = np.dtype(dtype)
|
||||||
|
sc = blosc2.SChunk(chunksize=nitems * dt.itemsize,
|
||||||
|
cparams=blosc2.CParams(typesize=dt.itemsize))
|
||||||
|
sc.fill_special(nitems, kind, value)
|
||||||
|
return sc.to_cframe()
|
||||||
|
|
||||||
|
|
||||||
|
# Special chunks. A repeated value stays in the frame as a 33+ byte chunk;
|
||||||
|
# NaN and uninitialised chunks become special offsets.
|
||||||
|
save("value_i4", special(300, "<i4", blosc2.SpecialValue.VALUE, 123456),
|
||||||
|
np.full(300, 123456, "<i4").tobytes())
|
||||||
|
save("value_f8", special(250, "<f8", blosc2.SpecialValue.VALUE, -2.5),
|
||||||
|
np.full(250, -2.5, "<f8").tobytes())
|
||||||
|
save("nan_f4", special(500, "<f4", blosc2.SpecialValue.NAN),
|
||||||
|
np.full(500, np.nan, "<f4").tobytes())
|
||||||
|
save("nan_f8", special(300, "<f8", blosc2.SpecialValue.NAN),
|
||||||
|
np.full(300, np.nan, "<f8").tobytes())
|
||||||
|
save("zero_u2", special(2000, "<u2", blosc2.SpecialValue.ZERO), bytes(4000))
|
||||||
|
# Uninitialised values: libhdf5 would hand back whatever memory it had;
|
||||||
|
# clawhdf5 returns zeros.
|
||||||
|
save("uninit_i8", special(64, "<i8", blosc2.SpecialValue.UNINIT), bytes(512))
|
||||||
|
|
||||||
|
rng = np.random.default_rng(11)
|
||||||
|
ramp = lambda n, dt: ((np.arange(n) * 7) % 1000 + rng.integers(0, 3, n)).astype(dt)
|
||||||
|
# Slowly varying: what the delta filter is for (noise would be stored raw).
|
||||||
|
smooth = lambda n, dt: (np.arange(n) // 3 + 1000).astype(dt)
|
||||||
|
|
||||||
|
# Delta over many blocks, for type sizes 1, 2, 4, 8, 3 (bytes) and 16 (u64
|
||||||
|
# pairs).
|
||||||
|
for dt, n, codec in [("<u1", 2000, blosc2.Codec.LZ4), ("<i2", 1000, blosc2.Codec.BLOSCLZ),
|
||||||
|
("<i4", 700, blosc2.Codec.LZ4), ("<u8", 400, blosc2.Codec.BLOSCLZ)]:
|
||||||
|
save(f"delta_{np.dtype(dt).name}_{codec.name.lower()}",
|
||||||
|
*plain(smooth(n, dt), codec=codec, blocksize=256,
|
||||||
|
filters=[blosc2.Filter.DELTA], filters_meta=[0]))
|
||||||
|
rec3 = np.frombuffer(smooth(3 * 300, "<u1").tobytes(), dtype="V3")
|
||||||
|
save("delta_v3", *plain(rec3, blocksize=300, filters=[blosc2.Filter.DELTA], filters_meta=[0]))
|
||||||
|
rec16 = np.frombuffer(smooth(2 * 200, "<u8").tobytes(), dtype="V16")
|
||||||
|
save("delta_shuffle_v16", *plain(rec16, blocksize=512,
|
||||||
|
filters=[blosc2.Filter.DELTA, blosc2.Filter.SHUFFLE],
|
||||||
|
filters_meta=[0, 0]))
|
||||||
|
|
||||||
|
# Bit shuffle of blocks whose element count is not a multiple of 8 (44-byte
|
||||||
|
# blocks of 4-byte elements: 8 transposed, 3 copied).
|
||||||
|
save("bitshuffle_odd_blocks", *plain(ramp(500, "<i4"), codec=blosc2.Codec.ZSTD, blocksize=44,
|
||||||
|
filters=[blosc2.Filter.BITSHUFFLE], filters_meta=[0]))
|
||||||
|
# Shuffle in groups of 2 bytes of an 8-byte type (filters_meta).
|
||||||
|
save("shuffle_meta2", *plain(ramp(500, "<i8"), codec=blosc2.Codec.ZLIB,
|
||||||
|
filters=[blosc2.Filter.SHUFFLE], filters_meta=[2]))
|
||||||
|
# Streams split per byte, and never split.
|
||||||
|
save("always_split", *plain(ramp(1000, "<f4"), codec=blosc2.Codec.LZ4HC,
|
||||||
|
splitmode=blosc2.SplitMode.ALWAYS_SPLIT))
|
||||||
|
save("never_split", *plain(ramp(1500, "<u2"), codec=blosc2.Codec.ZSTD,
|
||||||
|
splitmode=blosc2.SplitMode.NEVER_SPLIT))
|
||||||
|
|
||||||
|
# B2ND arrays of several chunks whose edge chunks and blocks are padded, and
|
||||||
|
# one whose middle chunk is all zeros (a special offset).
|
||||||
|
for name, shape, chunks, blocks, dt in [
|
||||||
|
("b2nd_2d", (37, 29), (10, 16), (4, 6), "<i4"),
|
||||||
|
("b2nd_3d", (9, 10, 7), (4, 5, 3), (3, 2, 2), "<f4"),
|
||||||
|
("b2nd_4d", (5, 7, 5, 6), (3, 2, 5, 4), (2, 2, 3, 3), "<u2"),
|
||||||
|
]:
|
||||||
|
a = ramp(int(np.prod(shape)), dt).reshape(shape)
|
||||||
|
arr = blosc2.asarray(a, chunks=chunks, blocks=blocks)
|
||||||
|
save(name, arr.to_cframe(), a.tobytes())
|
||||||
|
a = ramp(60 * 20, "<i2").reshape(60, 20)
|
||||||
|
a[20:40, :] = 0
|
||||||
|
arr = blosc2.asarray(a, chunks=(20, 20), blocks=(8, 16))
|
||||||
|
save("b2nd_zero_chunk", arr.to_cframe(), a.tobytes())
|
||||||
|
|
||||||
|
# Refused: a dictionary, and a registered filter (bytedelta).
|
||||||
|
frame, _ = plain(ramp(4000, "<i4"), codec=blosc2.Codec.ZSTD, use_dict=True, blocksize=2048)
|
||||||
|
save_err("zstd_dict", frame, "dictionar")
|
||||||
|
frame, _ = plain(ramp(500, "<i4"), filters=[blosc2.Filter.SHUFFLE, blosc2.Filter.BYTEDELTA],
|
||||||
|
filters_meta=[0, 4])
|
||||||
|
save_err("bytedelta", frame, "filter 35")
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1 @@
|
|||||||
|
dictionar
|
||||||
@@ -47,8 +47,9 @@ lzf = ["clawhdf5-format/lzf"]
|
|||||||
bitshuffle = ["clawhdf5-format/bitshuffle"]
|
bitshuffle = ["clawhdf5-format/bitshuffle"]
|
||||||
bzip2 = ["clawhdf5-format/bzip2"]
|
bzip2 = ["clawhdf5-format/bzip2"]
|
||||||
blosc = ["clawhdf5-format/blosc"]
|
blosc = ["clawhdf5-format/blosc"]
|
||||||
|
blosc2 = ["clawhdf5-format/blosc2"]
|
||||||
# Every plugin filter.
|
# Every plugin filter.
|
||||||
plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc"]
|
plugin-filters = ["lzf", "bitshuffle", "bzip2", "blosc", "blosc2"]
|
||||||
# Dataset::verify_provenance() — recompute a dataset's SHA-256 and compare
|
# Dataset::verify_provenance() — recompute a dataset's SHA-256 and compare
|
||||||
# against its stored _provenance_sha256 attribute. On by default, matching
|
# against its stored _provenance_sha256 attribute. On by default, matching
|
||||||
# clawhdf5-format's own default-on `provenance` feature.
|
# clawhdf5-format's own default-on `provenance` feature.
|
||||||
|
|||||||
@@ -77,7 +77,7 @@ except ImportError:
|
|||||||
hdf5plugin = None
|
hdf5plugin = None
|
||||||
path = sys.argv[1]
|
path = sys.argv[1]
|
||||||
FILTERS = eval('(' + sys.argv[2] + ')')
|
FILTERS = eval('(' + sys.argv[2] + ')')
|
||||||
cases = [
|
cases = eval('(' + sys.argv[3] + ')') if len(sys.argv) > 3 else [
|
||||||
('<u1', (1000,), (128,), 'ramp'),
|
('<u1', (1000,), (128,), 'ramp'),
|
||||||
('<i2', (37, 53), (10, 16), 'ramp'),
|
('<i2', (37, 53), (10, 16), 'ramp'),
|
||||||
('<i4', (2000,), (300,), 'ramp'),
|
('<i4', (2000,), (300,), 'ramp'),
|
||||||
@@ -97,7 +97,10 @@ with h5py.File(path, 'w') as f:
|
|||||||
for label, kw in FILTERS:
|
for label, kw in FILTERS:
|
||||||
for dt, shape, chunks, kind in cases:
|
for dt, shape, chunks, kind in cases:
|
||||||
n = int(np.prod(shape))
|
n = int(np.prod(shape))
|
||||||
if kind == 'noise':
|
if kind in ('zeros', 'const', 'nan'):
|
||||||
|
fill = {'zeros': 0, 'const': 7, 'nan': np.nan}[kind]
|
||||||
|
data = np.full(n, fill, dtype=dt)
|
||||||
|
elif kind == 'noise':
|
||||||
raw = rng.integers(0, 256, n * np.dtype(dt).itemsize, dtype=np.uint8)
|
raw = rng.integers(0, 256, n * np.dtype(dt).itemsize, dtype=np.uint8)
|
||||||
data = raw.view(dt)
|
data = raw.view(dt)
|
||||||
if np.dtype(dt).kind == 'f':
|
if np.dtype(dt).kind == 'f':
|
||||||
@@ -117,11 +120,18 @@ print(i)
|
|||||||
/// `(label, create_dataset kwargs)`), then check that clawhdf5 reads each
|
/// `(label, create_dataset kwargs)`), then check that clawhdf5 reads each
|
||||||
/// filtered dataset exactly as its unfiltered twin.
|
/// filtered dataset exactly as its unfiltered twin.
|
||||||
fn check_h5py_written(tag: &str, filters: &str) {
|
fn check_h5py_written(tag: &str, filters: &str) {
|
||||||
|
check_h5py_written_cases(tag, filters, None);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`check_h5py_written`] over `cases` (a Python list of `(dtype, shape,
|
||||||
|
/// chunks, kind)`, kind one of ramp, noise, zeros, const, nan) instead of
|
||||||
|
/// the default ones.
|
||||||
|
fn check_h5py_written_cases(tag: &str, filters: &str, cases: Option<&str>) {
|
||||||
let dir = tempfile::tempdir().unwrap();
|
let dir = tempfile::tempdir().unwrap();
|
||||||
let path = dir.path().join(format!("{tag}.h5"));
|
let path = dir.path().join(format!("{tag}.h5"));
|
||||||
let n: usize = run_python(GENERATE, &[path.to_str().unwrap(), filters])
|
let mut args = vec![path.to_str().unwrap(), filters];
|
||||||
.parse()
|
args.extend(cases);
|
||||||
.unwrap();
|
let n: usize = run_python(GENERATE, &args).parse().unwrap();
|
||||||
assert!(n > 0);
|
assert!(n > 0);
|
||||||
let file = File::open(&path).unwrap();
|
let file = File::open(&path).unwrap();
|
||||||
for i in 0..n {
|
for i in 0..n {
|
||||||
@@ -353,6 +363,110 @@ fn blosc_written_by_clawhdf5_reads_in_hdf5plugin() {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(feature = "blosc2")]
|
||||||
|
#[test]
|
||||||
|
fn blosc2_written_by_hdf5plugin_reads_exactly() {
|
||||||
|
if !have_python("h5py, hdf5plugin") {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
// Every codec hdf5plugin's Blosc2 offers, each lossless filter, and
|
||||||
|
// levels from "store" to maximum. Truncating precision is lossy, so it
|
||||||
|
// is checked separately against h5py's own reading.
|
||||||
|
check_h5py_written(
|
||||||
|
"blosc2",
|
||||||
|
r#"[(f'{c} {f} {l}', hdf5plugin.Blosc2(cname=c, clevel=l, filters=f))
|
||||||
|
for c in ['blosclz', 'lz4', 'lz4hc', 'zlib', 'zstd']
|
||||||
|
for f, l in [(hdf5plugin.Blosc2.NOFILTER, 5),
|
||||||
|
(hdf5plugin.Blosc2.SHUFFLE, 9),
|
||||||
|
(hdf5plugin.Blosc2.BITSHUFFLE, 1),
|
||||||
|
(hdf5plugin.Blosc2.DELTA, 5)]]
|
||||||
|
+ [('blosclz level 0', hdf5plugin.Blosc2(cname='blosclz', clevel=0)),
|
||||||
|
('zstd level 0 bitshuffle',
|
||||||
|
hdf5plugin.Blosc2(cname='zstd', clevel=0, filters=hdf5plugin.Blosc2.BITSHUFFLE))]"#,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Blosc2 over more shapes and dtypes: every integer and float width,
|
||||||
|
/// 1-D to 5-D chunks (B2ND arrays from 2-D on) with partial edge chunks and
|
||||||
|
/// block shapes that pad the chunk, datasets of zeros, of one repeated value
|
||||||
|
/// and of NaN (Blosc2's "special" chunks), and Fletcher32 before Blosc2
|
||||||
|
/// (which makes hdf5-blosc2 fall back from B2ND to a plain frame).
|
||||||
|
#[cfg(feature = "blosc2")]
|
||||||
|
#[test]
|
||||||
|
fn blosc2_shapes_and_special_chunks_read_exactly() {
|
||||||
|
if !have_python("h5py, hdf5plugin") {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let cases = r#"[(dt, shape, chunks, kind)
|
||||||
|
for dt in ['<i1', '<u1', '<i2', '>u2', '<i4', '<u4', '<i8', '<u8', '<f4', '>f8']
|
||||||
|
for shape, chunks in [((777,), (100,)),
|
||||||
|
((37, 53), (10, 16)),
|
||||||
|
((9, 10, 11), (4, 5, 3)),
|
||||||
|
((6, 7, 5, 9), (3, 2, 5, 4))]
|
||||||
|
for kind in ['ramp', 'noise']]
|
||||||
|
+ [('<f8', (300, 30), (64, 7), 'zeros'), ('<i4', (50, 40, 3), (7, 9, 3), 'zeros'),
|
||||||
|
('<u2', (5000,), (1024,), 'const'), ('<i8', (40, 40), (16, 16), 'const'),
|
||||||
|
('<f4', (100, 20), (32, 8), 'nan'), ('<f8', (3000,), (1000,), 'nan'),
|
||||||
|
('<f4', (1000, 1000), (256, 512), 'ramp'),
|
||||||
|
('<i2', (3, 3, 3, 3, 3), (2, 2, 2, 2, 2), 'ramp')]"#;
|
||||||
|
check_h5py_written_cases(
|
||||||
|
"blosc2_shapes",
|
||||||
|
r#"[('lz4 shuffle', hdf5plugin.Blosc2(cname='lz4')),
|
||||||
|
('zstd bitshuffle',
|
||||||
|
hdf5plugin.Blosc2(cname='zstd', clevel=7, filters=hdf5plugin.Blosc2.BITSHUFFLE)),
|
||||||
|
('blosclz delta', hdf5plugin.Blosc2(cname='blosclz', filters=hdf5plugin.Blosc2.DELTA)),
|
||||||
|
('zlib + fletcher32', dict(**hdf5plugin.Blosc2(cname='zlib'), fletcher32=True))]"#,
|
||||||
|
Some(cases),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Truncating precision is lossy: clawhdf5 must read exactly what h5py
|
||||||
|
/// (libhdf5 with hdf5plugin) reads.
|
||||||
|
#[cfg(feature = "blosc2")]
|
||||||
|
#[test]
|
||||||
|
fn blosc2_truncated_precision_reads_as_h5py() {
|
||||||
|
if !have_python("h5py, hdf5plugin") {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let path = dir.path().join("blosc2_trunc.h5");
|
||||||
|
let n: usize = run_python(
|
||||||
|
r#"
|
||||||
|
import sys
|
||||||
|
import numpy as np, h5py, hdf5plugin
|
||||||
|
rng = np.random.default_rng(3)
|
||||||
|
i = 0
|
||||||
|
with h5py.File(sys.argv[1], 'w') as f:
|
||||||
|
for dt in ['<f4', '<f8']:
|
||||||
|
for shape, chunks in [((1000,), (300,)), ((37, 53), (10, 16))]:
|
||||||
|
d = (rng.standard_normal(shape) * 1000).astype(dt)
|
||||||
|
ds = f.create_dataset(f'f{i}', data=d, chunks=chunks,
|
||||||
|
**hdf5plugin.Blosc2(cname='lz4',
|
||||||
|
filters=hdf5plugin.Blosc2.TRUNC_PREC))
|
||||||
|
f.create_dataset(f'r{i}', data=ds[()])
|
||||||
|
i += 1
|
||||||
|
print(i)
|
||||||
|
"#,
|
||||||
|
&[path.to_str().unwrap()],
|
||||||
|
)
|
||||||
|
.parse()
|
||||||
|
.unwrap();
|
||||||
|
let file = File::open(&path).unwrap();
|
||||||
|
for i in 0..n {
|
||||||
|
let got = file
|
||||||
|
.dataset(&format!("f{i}"))
|
||||||
|
.unwrap()
|
||||||
|
.read_selection(&Selection::All)
|
||||||
|
.unwrap();
|
||||||
|
let want = file
|
||||||
|
.dataset(&format!("r{i}"))
|
||||||
|
.unwrap()
|
||||||
|
.read_selection(&Selection::All)
|
||||||
|
.unwrap();
|
||||||
|
assert!(got == want, "trunc_prec f{i}: data differs from h5py");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(feature = "lzf")]
|
#[cfg(feature = "lzf")]
|
||||||
#[test]
|
#[test]
|
||||||
fn lzf_written_by_h5py_reads_exactly() {
|
fn lzf_written_by_h5py_reads_exactly() {
|
||||||
@@ -381,8 +495,9 @@ fn lzf_written_by_clawhdf5_reads_in_h5py() {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Blosc2 and ZFP are not implemented: reading them must be a clear error
|
/// ZFP is not implemented, and Blosc2 is not in a build without the
|
||||||
/// naming the filter, never data.
|
/// `blosc2` feature: reading them must be a clear error naming the filter,
|
||||||
|
/// never data.
|
||||||
#[test]
|
#[test]
|
||||||
fn unimplemented_filters_are_a_clear_error() {
|
fn unimplemented_filters_are_a_clear_error() {
|
||||||
if !have_python("h5py, hdf5plugin") {
|
if !have_python("h5py, hdf5plugin") {
|
||||||
@@ -402,7 +517,11 @@ with h5py.File(sys.argv[1], 'w') as f:
|
|||||||
&[path.to_str().unwrap()],
|
&[path.to_str().unwrap()],
|
||||||
);
|
);
|
||||||
let file = File::open(&path).unwrap();
|
let file = File::open(&path).unwrap();
|
||||||
for (name, id, label) in [("blosc2", 32026u16, "Blosc2"), ("zfp", 32013, "ZFP")] {
|
let mut missing = vec![("zfp", 32013u16, "ZFP")];
|
||||||
|
if !cfg!(feature = "blosc2") {
|
||||||
|
missing.push(("blosc2", 32026, "Blosc2"));
|
||||||
|
}
|
||||||
|
for (name, id, label) in missing {
|
||||||
let err = file
|
let err = file
|
||||||
.dataset(name)
|
.dataset(name)
|
||||||
.unwrap()
|
.unwrap()
|
||||||
|
|||||||
+2
-2
@@ -67,7 +67,7 @@ run_step "cargo clippy --all-targets" cargo clippy \
|
|||||||
|
|
||||||
# 3. Clippy over clawhdf5-format's optional features, which the default
|
# 3. Clippy over clawhdf5-format's optional features, which the default
|
||||||
# workspace build never compiles (szip is left out: it needs libaec).
|
# workspace build never compiles (szip is left out: it needs libaec).
|
||||||
# plugin-filters = bitshuffle, bzip2, blosc (and the default-on lzf).
|
# plugin-filters = bitshuffle, bzip2, blosc, blosc2 (and the default-on lzf).
|
||||||
run_step "cargo clippy (format feature matrix)" cargo clippy \
|
run_step "cargo clippy (format feature matrix)" cargo clippy \
|
||||||
-p clawhdf5-format \
|
-p clawhdf5-format \
|
||||||
--all-targets \
|
--all-targets \
|
||||||
@@ -78,7 +78,7 @@ run_step "cargo clippy (format feature matrix)" cargo clippy \
|
|||||||
# dependencies (bitshuffle and blosc share code).
|
# dependencies (bitshuffle and blosc share code).
|
||||||
plugin_filters_alone() {
|
plugin_filters_alone() {
|
||||||
local f
|
local f
|
||||||
for f in bitshuffle bzip2 blosc; do
|
for f in bitshuffle bzip2 blosc blosc2; do
|
||||||
echo "--- $f"
|
echo "--- $f"
|
||||||
cargo clippy -p clawhdf5-format --all-targets --features "$f" -- -D warnings || return 1
|
cargo clippy -p clawhdf5-format --all-targets --features "$f" -- -D warnings || return 1
|
||||||
done
|
done
|
||||||
|
|||||||
Reference in New Issue
Block a user