Merge branch 'fix/p0-filters' into fix/phase0-correctness
# Conflicts: # crates/clawhdf5-format/src/filters.rs
This commit is contained in:
@@ -179,6 +179,31 @@
|
|||||||
`float16` rounding matches numpy's bit for bit on 4 020 probe values,
|
`float16` rounding matches numpy's bit for bit on 4 020 probe values,
|
||||||
including ties, subnormals and the overflow boundary), and an agent store —
|
including ties, subnormals and the overflow boundary), and an agent store —
|
||||||
`f32` and `float16` — opened by h5py with every dataset decoded.
|
`f32` and `float16` — opened by h5py with every dataset decoded.
|
||||||
|
- `clawhdf5-format` filters, checked against libhdf5 + hdf5plugin:
|
||||||
|
- **LZ4 (32004) now uses the registered HDF5 LZ4 format** (8-byte BE size,
|
||||||
|
4-byte BE block size, BE-length-prefixed blocks). Our old framing (4-byte
|
||||||
|
LE size + one block) was readable only by clawhdf5, and we could not read
|
||||||
|
libhdf5's (`h5ex_d_lz4.h5`). Old clawhdf5 LZ4 chunks still read; they are
|
||||||
|
told apart unambiguously (a registered chunk starts with four zero bytes).
|
||||||
|
- **Zstd (32015) frames now record the content size**, which libhdf5's zstd
|
||||||
|
plugin needs; h5py could not read our zstd datasets.
|
||||||
|
- **Pcodec moved from filter ID 32023 to 480.** 32023 is registered to
|
||||||
|
Granular BitRound, whose decode is a pass-through — libhdf5 with that
|
||||||
|
plugin would have returned compressed bytes as data. Pcodec has no
|
||||||
|
registered ID; 480 is in the registry's private range (256–511) and only
|
||||||
|
clawhdf5 can read it. Chunks written under 32023 with the filter name
|
||||||
|
`pcodec` (clawhdf5 ≤ 2.7.0) still read.
|
||||||
|
- **SZIP decode matches libhdf5.** It returned garbage or zeros with no
|
||||||
|
error for libhdf5-written files (the 4-byte size prefix, 32/64-bit
|
||||||
|
byte-plane interleaving, reference interval, scanline padding and byte
|
||||||
|
order were all handled wrongly) and rejected 64-bit data.
|
||||||
|
- N-Bit honours libhdf5's "need not compress" flag (multi-filter pipelines
|
||||||
|
such as `tfilters.h5` failed) and reads enum/no-op members.
|
||||||
|
- Scale-offset `float` decode uses libhdf5's single-precision arithmetic
|
||||||
|
(was 1 ULP off for some values).
|
||||||
|
- A pipeline with Fletcher32 ahead of the compressor (h5py
|
||||||
|
`set_fletcher32()` then `set_deflate()`) no longer fails with "deflate:
|
||||||
|
output exceeds size limit".
|
||||||
|
|
||||||
### Storage
|
### Storage
|
||||||
- `clawhdf5-format`: **half-precision datasets.**
|
- `clawhdf5-format`: **half-precision datasets.**
|
||||||
|
|||||||
@@ -696,7 +696,7 @@ stores keep their setting. Opt out with `float16 = false` or
|
|||||||
| `fast-checksum` | no | crc32fast-accelerated checksums |
|
| `fast-checksum` | no | crc32fast-accelerated checksums |
|
||||||
| `lz4` | no | LZ4 block compression filter (id 32004) |
|
| `lz4` | no | LZ4 block compression filter (id 32004) |
|
||||||
| `zstd` | no | Zstandard compression filter (id 32015) |
|
| `zstd` | no | Zstandard compression filter (id 32015) |
|
||||||
| `pcodec` | no | Pcodec lossless numerical codec (id 32023, via `pco` crate) |
|
| `pcodec` | no | Pcodec lossless numerical codec (via `pco` crate). Private, unregistered filter id 480: **only clawhdf5 can read these datasets** (h5py/libhdf5 cannot). Files from clawhdf5 <= 2.7.0 used id 32023, which is registered to Granular BitRound; they still read. |
|
||||||
| `system-zlib` | no | System zlib backend for deflate (C) |
|
| `system-zlib` | no | System zlib backend for deflate (C) |
|
||||||
| `blake3_hash` | no | BLAKE3 content hashing for provenance |
|
| `blake3_hash` | no | BLAKE3 content hashing for provenance |
|
||||||
| `szip` | no | SZIP filter (id 4) via libaec (C, through the internal `libaec-sys` crate) |
|
| `szip` | no | SZIP filter (id 4) via libaec (C, through the internal `libaec-sys` crate) |
|
||||||
|
|||||||
@@ -12,8 +12,8 @@ use crate::chunk_grid::ChunkGrid;
|
|||||||
use crate::ea_writer;
|
use crate::ea_writer;
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::filter_pipeline::{
|
use crate::filter_pipeline::{
|
||||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_PCODEC, FILTER_SHUFFLE, FILTER_ZSTD,
|
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_PCODEC, FILTER_PCODEC_NAME,
|
||||||
FilterDescription, FilterPipeline,
|
FILTER_SHUFFLE, FILTER_ZSTD, FilterDescription, FilterPipeline,
|
||||||
};
|
};
|
||||||
use crate::filters::compress_chunk;
|
use crate::filters::compress_chunk;
|
||||||
/// Round a file offset up to the next cache-line boundary.
|
/// Round a file offset up to the next cache-line boundary.
|
||||||
@@ -45,7 +45,8 @@ pub struct ChunkOptions {
|
|||||||
pub lz4: bool,
|
pub lz4: bool,
|
||||||
/// Zstandard compression level (1-22), None = no zstd. Filter ID 32015.
|
/// Zstandard compression level (1-22), None = no zstd. Filter ID 32015.
|
||||||
pub zstd_level: Option<u32>,
|
pub zstd_level: Option<u32>,
|
||||||
/// Pcodec lossless numerical compression. Filter ID 32023.
|
/// Pcodec lossless numerical compression. Private, unregistered filter
|
||||||
|
/// ID [`FILTER_PCODEC`] (480): only clawhdf5 can read it.
|
||||||
pub pcodec: bool,
|
pub pcodec: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -116,7 +117,7 @@ impl ChunkOptions {
|
|||||||
if self.pcodec {
|
if self.pcodec {
|
||||||
filters.push(FilterDescription {
|
filters.push(FilterDescription {
|
||||||
filter_id: FILTER_PCODEC,
|
filter_id: FILTER_PCODEC,
|
||||||
name: Some("pcodec".into()),
|
name: Some(FILTER_PCODEC_NAME.into()),
|
||||||
flags: 0,
|
flags: 0,
|
||||||
client_data: vec![element_size],
|
client_data: vec![element_size],
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -19,8 +19,23 @@ pub const FILTER_SCALEOFFSET: u16 = 6;
|
|||||||
pub const FILTER_LZ4: u16 = 32004;
|
pub const FILTER_LZ4: u16 = 32004;
|
||||||
/// Zstandard compression.
|
/// Zstandard compression.
|
||||||
pub const FILTER_ZSTD: u16 = 32015;
|
pub const FILTER_ZSTD: u16 = 32015;
|
||||||
/// Pcodec lossless numerical codec (clawhdf5 internal; not yet HDF5-registered).
|
/// Pcodec lossless numerical codec — a **private, unregistered** clawhdf5
|
||||||
pub const FILTER_PCODEC: u16 = 32023;
|
/// filter. Pcodec has no ID in the HDF Group's filter registry (checked
|
||||||
|
/// 2026-09-25, `hdf5_plugins/docs/RegisteredFilterPlugins.md`), so it uses an
|
||||||
|
/// ID from the registry's testing/private range (256–511). No libhdf5 plugin
|
||||||
|
/// decodes it: h5py/libhdf5 report the filter as unavailable. Only clawhdf5
|
||||||
|
/// (with the `pcodec` feature) reads these datasets.
|
||||||
|
pub const FILTER_PCODEC: u16 = 480;
|
||||||
|
/// Filter name written with [`FILTER_PCODEC`].
|
||||||
|
pub const FILTER_PCODEC_NAME: &str = "pcodec (clawhdf5 private)";
|
||||||
|
/// The ID clawhdf5 up to 2.7.0 wrote pcodec under. It is registered to
|
||||||
|
/// Granular BitRound (GBR), whose decode is a pass-through, so libhdf5 with
|
||||||
|
/// that plugin would have returned the compressed bytes as data. Read as
|
||||||
|
/// pcodec only when the filter is named exactly [`FILTER_PCODEC_LEGACY_NAME`],
|
||||||
|
/// the name those versions wrote; never written.
|
||||||
|
pub const FILTER_PCODEC_LEGACY: u16 = 32023;
|
||||||
|
/// The filter name clawhdf5 up to 2.7.0 wrote with [`FILTER_PCODEC_LEGACY`].
|
||||||
|
pub const FILTER_PCODEC_LEGACY_NAME: &str = "pcodec";
|
||||||
|
|
||||||
/// Description of a single filter in a pipeline.
|
/// Description of a single filter in a pipeline.
|
||||||
#[derive(Debug, Clone, PartialEq)]
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
|||||||
@@ -8,8 +8,9 @@ use alloc::{boxed::Box, vec, vec::Vec};
|
|||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
use crate::filter_pipeline::{
|
use crate::filter_pipeline::{
|
||||||
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_NBIT, FILTER_PCODEC, FILTER_SCALEOFFSET,
|
FILTER_DEFLATE, FILTER_FLETCHER32, FILTER_LZ4, FILTER_NBIT, FILTER_PCODEC,
|
||||||
FILTER_SHUFFLE, FILTER_SZIP, FILTER_ZSTD, FilterPipeline,
|
FILTER_PCODEC_LEGACY, FILTER_PCODEC_LEGACY_NAME, FILTER_SCALEOFFSET, FILTER_SHUFFLE,
|
||||||
|
FILTER_SZIP, FILTER_ZSTD, FilterPipeline,
|
||||||
};
|
};
|
||||||
|
|
||||||
/// Absolute ceiling on a single decompressed chunk's output size, used only
|
/// Absolute ceiling on a single decompressed chunk's output size, used only
|
||||||
@@ -107,6 +108,11 @@ pub fn decompress_chunk_masked(
|
|||||||
FILTER_ZSTD => zstd_decompress(&data, bound)?,
|
FILTER_ZSTD => zstd_decompress(&data, bound)?,
|
||||||
FILTER_FLETCHER32 => fletcher32_verify(&data)?,
|
FILTER_FLETCHER32 => fletcher32_verify(&data)?,
|
||||||
FILTER_PCODEC => pcodec_decompress(&data, element_size as usize, bound)?,
|
FILTER_PCODEC => pcodec_decompress(&data, element_size as usize, bound)?,
|
||||||
|
// Pcodec chunks written by clawhdf5 <= 2.7.0 under the ID registered
|
||||||
|
// to Granular BitRound; recognised by the name those versions wrote.
|
||||||
|
FILTER_PCODEC_LEGACY if filter.name.as_deref() == Some(FILTER_PCODEC_LEGACY_NAME) => {
|
||||||
|
pcodec_decompress(&data, element_size as usize, bound)?
|
||||||
|
}
|
||||||
// These decoders also reject an element count that would
|
// These decoders also reject an element count that would
|
||||||
// over-allocate past `bound`.
|
// over-allocate past `bound`.
|
||||||
FILTER_SCALEOFFSET => scaleoffset_decompress(&data, &filter.client_data, bound)?,
|
FILTER_SCALEOFFSET => scaleoffset_decompress(&data, &filter.client_data, bound)?,
|
||||||
@@ -135,7 +141,7 @@ pub fn compress_chunk(
|
|||||||
let level = filter.client_data.first().copied().unwrap_or(6);
|
let level = filter.client_data.first().copied().unwrap_or(6);
|
||||||
deflate_compress(&result, level)?
|
deflate_compress(&result, level)?
|
||||||
}
|
}
|
||||||
FILTER_LZ4 => lz4_compress(&result)?,
|
FILTER_LZ4 => lz4_compress(&result, &filter.client_data)?,
|
||||||
FILTER_ZSTD => {
|
FILTER_ZSTD => {
|
||||||
let level = filter.client_data.first().copied().unwrap_or(3);
|
let level = filter.client_data.first().copied().unwrap_or(3);
|
||||||
zstd_compress(&result, level)?
|
zstd_compress(&result, level)?
|
||||||
@@ -306,8 +312,20 @@ fn scaleoffset_decompress(
|
|||||||
fill_value
|
fill_value
|
||||||
} else if is_escale {
|
} else if is_escale {
|
||||||
minval + code as f64 * powi_f64(2.0, scale_factor)
|
minval + code as f64 * powi_f64(2.0, scale_factor)
|
||||||
|
} else if elem_size == 4 {
|
||||||
|
// H5Z_scaleoffset_modify_3/4 for `float`: the code is
|
||||||
|
// read as an `int` and everything is single precision,
|
||||||
|
// `(float)code / powf(10, D) + min`. Doing it in f64 and
|
||||||
|
// rounding once at the end is off by 1 ULP at times.
|
||||||
|
let d = if scale_factor >= 0 {
|
||||||
|
powi_f64(10.0, scale_factor) as f32
|
||||||
} else {
|
} else {
|
||||||
minval + code as f64 / powi_f64(10.0, scale_factor)
|
1.0 / powi_f64(10.0, -scale_factor) as f32
|
||||||
|
};
|
||||||
|
((code as u32 as i32) as f32 / d + minval as f32) as f64
|
||||||
|
} else {
|
||||||
|
// ... and for `double`: `(double)(long)code / pow(10, D) + min`.
|
||||||
|
(code as i64) as f64 / powi_f64(10.0, scale_factor) + minval
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
@@ -445,6 +463,9 @@ enum NbitNode {
|
|||||||
count: usize,
|
count: usize,
|
||||||
base_size: usize,
|
base_size: usize,
|
||||||
},
|
},
|
||||||
|
/// `H5Z_NBIT_NOOPTYPE`: a field N-Bit does not reduce (enum, string,
|
||||||
|
/// opaque, ...), stored as all `size` bytes, 8 bits each.
|
||||||
|
Noop { size: usize },
|
||||||
}
|
}
|
||||||
|
|
||||||
impl NbitNode {
|
impl NbitNode {
|
||||||
@@ -455,6 +476,7 @@ impl NbitNode {
|
|||||||
NbitNode::Array {
|
NbitNode::Array {
|
||||||
count, base_size, ..
|
count, base_size, ..
|
||||||
} => count * base_size,
|
} => count * base_size,
|
||||||
|
NbitNode::Noop { size } => *size,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -474,6 +496,7 @@ fn parse_nbit_node(cd: &[u32], idx: &mut usize, depth: u32) -> Result<NbitNode,
|
|||||||
const ATOMIC: u32 = 1;
|
const ATOMIC: u32 = 1;
|
||||||
const ARRAY: u32 = 2;
|
const ARRAY: u32 = 2;
|
||||||
const COMPOUND: u32 = 3;
|
const COMPOUND: u32 = 3;
|
||||||
|
const NOOPTYPE: u32 = 4;
|
||||||
if depth > NBIT_MAX_DEPTH {
|
if depth > NBIT_MAX_DEPTH {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"nbit: type tree nested too deeply".into(),
|
"nbit: type tree nested too deeply".into(),
|
||||||
@@ -549,8 +572,17 @@ fn parse_nbit_node(cd: &[u32], idx: &mut usize, depth: u32) -> Result<NbitNode,
|
|||||||
members,
|
members,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
// Class 4 is H5Z_NBIT_NOOPTYPE (members copied verbatim) — not seen in
|
NOOPTYPE => {
|
||||||
// practice for the supported leaf types and left unsupported.
|
// class, size
|
||||||
|
let size = nbit_cd(cd, *idx + 1)? as usize;
|
||||||
|
*idx += 2;
|
||||||
|
if size == 0 {
|
||||||
|
return Err(FormatError::ChunkedReadError(
|
||||||
|
"nbit: invalid no-op type size".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
Ok(NbitNode::Noop { size })
|
||||||
|
}
|
||||||
_ => Err(FormatError::UnsupportedFilter(FILTER_NBIT)),
|
_ => Err(FormatError::UnsupportedFilter(FILTER_NBIT)),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -615,6 +647,11 @@ fn decode_nbit_node(
|
|||||||
decode_nbit_node(bnode, br, elem, base + i * base_size)?;
|
decode_nbit_node(bnode, br, elem, base + i * base_size)?;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
NbitNode::Noop { size } => {
|
||||||
|
for slot in &mut elem[base..base + size] {
|
||||||
|
*slot = br.read(8)? as u8;
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
@@ -627,17 +664,28 @@ fn decode_nbit_node(
|
|||||||
/// type tree — atomic (`[1, size, order, precision, offset]`), array
|
/// type tree — atomic (`[1, size, order, precision, offset]`), array
|
||||||
/// (`[2, total_size, <base>]`) and compound
|
/// (`[2, total_size, <base>]`) and compound
|
||||||
/// (`[3, total_size, nmembers, (offset, <node>)*]`) — preceded by
|
/// (`[3, total_size, nmembers, (offset, <node>)*]`) — preceded by
|
||||||
/// `[nparms, flag, nelmts]`. Decompression walks the tree once per element,
|
/// `[nparms, need_not_compress, nelmts]`; when `need_not_compress` is set
|
||||||
|
/// (every field already uses its full width, e.g. a 32-bit int of precision
|
||||||
|
/// 32) libhdf5 stores the data unchanged and so do we. Fields N-Bit cannot
|
||||||
|
/// reduce (enums, strings, ...) are no-op nodes (`[4, size]`) copied whole.
|
||||||
|
/// Decompression walks the tree once per element,
|
||||||
/// placing each field's bits at its byte/bit offset in a zero-filled element
|
/// placing each field's bits at its byte/bit offset in a zero-filled element
|
||||||
/// (HDF5's canonical reduced-precision layout). Sign-extension of reduced
|
/// (HDF5's canonical reduced-precision layout). Sign-extension of reduced
|
||||||
/// precision signed integers is the datatype reader's job. Atomic floats are
|
/// precision signed integers is the datatype reader's job, and so is
|
||||||
/// encoded as full-precision atomics and handled transparently.
|
/// converting a reduced-precision float (its own sign/exponent/mantissa
|
||||||
|
/// layout, e.g. `le_data.h5`'s 20-bit `Nbit_float_data_*`) to IEEE: the
|
||||||
|
/// filter's output is the file type's bytes, as libhdf5's is before type
|
||||||
|
/// conversion.
|
||||||
fn nbit_decompress(data: &[u8], cd: &[u32], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
fn nbit_decompress(data: &[u8], cd: &[u32], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
if cd.len() < 4 {
|
if cd.len() < 3 {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(FormatError::ChunkedReadError(
|
||||||
"nbit: missing filter client data".into(),
|
"nbit: missing filter client data".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
// H5Z__filter_nbit: `if (cd_values[1]) HGOTO_DONE(*buf_size)`.
|
||||||
|
if cd[1] != 0 {
|
||||||
|
return Ok(data.to_vec());
|
||||||
|
}
|
||||||
let nelmts = cd[2] as usize;
|
let nelmts = cd[2] as usize;
|
||||||
let mut idx = 3;
|
let mut idx = 3;
|
||||||
let root = parse_nbit_node(cd, &mut idx, 0)?;
|
let root = parse_nbit_node(cd, &mut idx, 0)?;
|
||||||
@@ -879,12 +927,32 @@ fn deflate_compress(_data: &[u8], _level: u32) -> Result<Vec<u8>, FormatError> {
|
|||||||
Err(FormatError::UnsupportedFilter(FILTER_DEFLATE))
|
Err(FormatError::UnsupportedFilter(FILTER_DEFLATE))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Decompress LZ4 data. Format: 4 bytes LE original size + LZ4 block data.
|
/// Default LZ4 block size of the registered HDF5 LZ4 filter (`H5Zlz4.c`,
|
||||||
|
/// `DEFAULT_BLOCK_SIZE`): 1 GiB, so an HDF5 chunk is normally one block.
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
const LZ4_DEFAULT_BLOCK_SIZE: usize = 1 << 30;
|
||||||
|
|
||||||
|
/// Decompress an LZ4 (filter 32004) chunk.
|
||||||
///
|
///
|
||||||
/// The 4-byte "original size" header is part of the attacker-controlled
|
/// Two framings are read:
|
||||||
/// compressed payload itself, so it is bounded against `expected_bytes` (the
|
///
|
||||||
/// pipeline's declared chunk size) before being used to size the output
|
/// * The registered HDF5 LZ4 filter format (`H5Zlz4.c`, what libhdf5 +
|
||||||
/// allocation — otherwise a crafted 4-byte value can request up to ~4 GiB.
|
/// hdf5plugin write, and what clawhdf5 writes after 2.7.0): an 8-byte
|
||||||
|
/// big-endian total decompressed size, a 4-byte big-endian block size, then
|
||||||
|
/// per block a 4-byte big-endian compressed length followed by the block. A
|
||||||
|
/// block whose compressed length equals its decompressed length is stored
|
||||||
|
/// raw.
|
||||||
|
/// * The legacy clawhdf5 framing (up to 2.7.0): a 4-byte little-endian size
|
||||||
|
/// followed by one raw LZ4 block. libhdf5 cannot read it.
|
||||||
|
///
|
||||||
|
/// They are told apart unambiguously: an HDF5 chunk is smaller than 4 GiB, so
|
||||||
|
/// the registered format's big-endian `u64` size always starts with four zero
|
||||||
|
/// bytes and the whole chunk is at least 12 bytes; a legacy chunk starts with
|
||||||
|
/// four zero bytes only when it is empty, and is then 5 bytes long.
|
||||||
|
///
|
||||||
|
/// Every size read from the payload is bounded against `expected_bytes` (the
|
||||||
|
/// pipeline's declared chunk size) before it sizes an allocation, so a crafted
|
||||||
|
/// header cannot request gigabytes.
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
if data.len() < 4 {
|
if data.len() < 4 {
|
||||||
@@ -892,38 +960,112 @@ fn lz4_decompress(data: &[u8], expected_bytes: usize) -> Result<Vec<u8>, FormatE
|
|||||||
"lz4: data too short".into(),
|
"lz4: data too short".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
let orig_size = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
let check_size = |size: usize| -> Result<(), FormatError> {
|
||||||
if expected_bytes != 0 && orig_size > expected_bytes {
|
if expected_bytes != 0 && size > expected_bytes {
|
||||||
return Err(FormatError::DecompressionError(
|
return Err(FormatError::DecompressionError(
|
||||||
"lz4: declared size exceeds chunk size".into(),
|
"lz4: declared size exceeds chunk size".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
if orig_size > MAX_DECOMPRESS_SIZE {
|
if size > MAX_DECOMPRESS_SIZE {
|
||||||
return Err(FormatError::DecompressionError(
|
return Err(FormatError::DecompressionError(
|
||||||
"lz4: declared size exceeds limit".into(),
|
"lz4: declared size exceeds limit".into(),
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
Ok(())
|
||||||
|
};
|
||||||
|
if data.len() >= 12 && data[..4] == [0, 0, 0, 0] {
|
||||||
|
return lz4_decompress_hdf5(data, check_size);
|
||||||
|
}
|
||||||
|
// Legacy clawhdf5 framing: 4-byte LE size + one LZ4 block.
|
||||||
|
let orig_size = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||||
|
check_size(orig_size)?;
|
||||||
lz4_flex::block::decompress(&data[4..], orig_size)
|
lz4_flex::block::decompress(&data[4..], orig_size)
|
||||||
.map_err(|e| FormatError::DecompressionError(format!("lz4: {e}")))
|
.map_err(|e| FormatError::DecompressionError(format!("lz4: {e}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Decode the registered HDF5 LZ4 framing (see [`lz4_decompress`]).
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_decompress_hdf5(
|
||||||
|
data: &[u8],
|
||||||
|
check_size: impl Fn(usize) -> Result<(), FormatError>,
|
||||||
|
) -> Result<Vec<u8>, FormatError> {
|
||||||
|
let err = |m: &str| FormatError::DecompressionError(format!("lz4: {m}"));
|
||||||
|
let be32 = |b: &[u8]| u32::from_be_bytes([b[0], b[1], b[2], b[3]]) as usize;
|
||||||
|
// The first four bytes are zero (checked by the caller), so the size is
|
||||||
|
// the low 32 bits of the big-endian u64.
|
||||||
|
let orig_size = be32(&data[4..8]);
|
||||||
|
check_size(orig_size)?;
|
||||||
|
let block_size = be32(&data[8..12]).min(orig_size);
|
||||||
|
if block_size == 0 && orig_size != 0 {
|
||||||
|
return Err(err("zero block size"));
|
||||||
|
}
|
||||||
|
let mut out = vec![0u8; orig_size];
|
||||||
|
let mut pos = 12usize;
|
||||||
|
let mut done = 0usize;
|
||||||
|
while done < orig_size {
|
||||||
|
let this_block = block_size.min(orig_size - done);
|
||||||
|
let comp_len = be32(
|
||||||
|
data.get(pos..pos + 4)
|
||||||
|
.ok_or_else(|| err("truncated block header"))?,
|
||||||
|
);
|
||||||
|
pos += 4;
|
||||||
|
let block = data
|
||||||
|
.get(pos..pos.saturating_add(comp_len))
|
||||||
|
.ok_or_else(|| err("truncated block"))?;
|
||||||
|
let dst = &mut out[done..done + this_block];
|
||||||
|
if comp_len == this_block {
|
||||||
|
dst.copy_from_slice(block);
|
||||||
|
} else {
|
||||||
|
let n = lz4_flex::block::decompress_into(block, dst)
|
||||||
|
.map_err(|e| FormatError::DecompressionError(format!("lz4: {e}")))?;
|
||||||
|
if n != this_block {
|
||||||
|
return Err(err("block decompressed to the wrong size"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pos += comp_len;
|
||||||
|
done += this_block;
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "lz4"))]
|
#[cfg(not(feature = "lz4"))]
|
||||||
fn lz4_decompress(_data: &[u8], _expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
fn lz4_decompress(_data: &[u8], _expected_bytes: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compress data with LZ4 block format. Format: 4 bytes LE original size + LZ4 block data.
|
/// Compress data in the registered HDF5 LZ4 filter format (see
|
||||||
|
/// [`lz4_decompress`]), so libhdf5 with the LZ4 plugin (e.g. hdf5plugin) can
|
||||||
|
/// read it. `cd[0]`, when present and non-zero, is the block size in bytes,
|
||||||
|
/// as in `H5Zlz4.c`; otherwise the 1 GiB default applies.
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn lz4_compress(data: &[u8]) -> Result<Vec<u8>, FormatError> {
|
fn lz4_compress(data: &[u8], cd: &[u32]) -> Result<Vec<u8>, FormatError> {
|
||||||
let compressed = lz4_flex::block::compress(data);
|
let block_size = match cd.first() {
|
||||||
let mut result = Vec::with_capacity(4 + compressed.len());
|
Some(&b) if b != 0 => b as usize,
|
||||||
result.extend_from_slice(&(data.len() as u32).to_le_bytes());
|
_ => LZ4_DEFAULT_BLOCK_SIZE,
|
||||||
|
}
|
||||||
|
.min(data.len());
|
||||||
|
let mut result = Vec::with_capacity(16 + data.len() / 2);
|
||||||
|
result.extend_from_slice(&(data.len() as u64).to_be_bytes());
|
||||||
|
result.extend_from_slice(&(block_size as u32).to_be_bytes());
|
||||||
|
if block_size == 0 {
|
||||||
|
return Ok(result);
|
||||||
|
}
|
||||||
|
for block in data.chunks(block_size) {
|
||||||
|
let compressed = lz4_flex::block::compress(block);
|
||||||
|
if compressed.len() >= block.len() {
|
||||||
|
// Incompressible: stored raw, marked by length == block length.
|
||||||
|
result.extend_from_slice(&(block.len() as u32).to_be_bytes());
|
||||||
|
result.extend_from_slice(block);
|
||||||
|
} else {
|
||||||
|
result.extend_from_slice(&(compressed.len() as u32).to_be_bytes());
|
||||||
result.extend_from_slice(&compressed);
|
result.extend_from_slice(&compressed);
|
||||||
|
}
|
||||||
|
}
|
||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(feature = "lz4"))]
|
#[cfg(not(feature = "lz4"))]
|
||||||
fn lz4_compress(_data: &[u8]) -> Result<Vec<u8>, FormatError> {
|
fn lz4_compress(_data: &[u8], _cd: &[u32]) -> Result<Vec<u8>, FormatError> {
|
||||||
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
Err(FormatError::UnsupportedFilter(FILTER_LZ4))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -960,10 +1102,14 @@ fn zstd_decompress(_data: &[u8], _expected_bytes: usize) -> Result<Vec<u8>, Form
|
|||||||
Err(FormatError::UnsupportedFilter(FILTER_ZSTD))
|
Err(FormatError::UnsupportedFilter(FILTER_ZSTD))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Compress data with zstd.
|
/// Compress data with zstd as one frame whose header records the content
|
||||||
|
/// size. The registered HDF5 Zstandard filter (`H5Zzstd.c`, used by
|
||||||
|
/// libhdf5 + hdf5plugin) sizes its output buffer from
|
||||||
|
/// `ZSTD_getFrameContentSize` and fails on a frame without it, which is what
|
||||||
|
/// the streaming encoder (`zstd::encode_all`) produced.
|
||||||
#[cfg(feature = "zstd")]
|
#[cfg(feature = "zstd")]
|
||||||
fn zstd_compress(data: &[u8], level: u32) -> Result<Vec<u8>, FormatError> {
|
fn zstd_compress(data: &[u8], level: u32) -> Result<Vec<u8>, FormatError> {
|
||||||
zstd::encode_all(data, level as i32)
|
zstd::bulk::compress(data, level as i32)
|
||||||
.map_err(|e| FormatError::CompressionError(format!("zstd: {e}")))
|
.map_err(|e| FormatError::CompressionError(format!("zstd: {e}")))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1651,11 +1797,317 @@ mod tests {
|
|||||||
|
|
||||||
// --- LZ4 tests ---
|
// --- LZ4 tests ---
|
||||||
|
|
||||||
|
fn unhex(s: &str) -> Vec<u8> {
|
||||||
|
(0..s.len())
|
||||||
|
.step_by(2)
|
||||||
|
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn one_filter(filter_id: u16, client_data: Vec<u32>) -> FilterDescription {
|
||||||
|
FilterDescription {
|
||||||
|
filter_id,
|
||||||
|
name: None,
|
||||||
|
flags: 1,
|
||||||
|
client_data,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Fletcher32 before a compressor (libhdf5 applies filters in pipeline
|
||||||
|
/// order, so the compressor sees chunk + checksum): deflate's output is 4
|
||||||
|
/// bytes over the chunk size, which we rejected as "deflate: output
|
||||||
|
/// exceeds size limit". Chunks from h5py/libhdf5, values from h5py.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "deflate")]
|
||||||
|
fn fletcher32_before_deflate_decodes() {
|
||||||
|
// h5py: set_fletcher32(); set_deflate(4); i32 0..100, chunks of 10.
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
one_filter(FILTER_FLETCHER32, vec![]),
|
||||||
|
one_filter(FILTER_DEFLATE, vec![4]),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
let raw =
|
||||||
|
unhex("785e936360609007620520560462252056066215205605623520560762c6483e3d00234501f0");
|
||||||
|
let want: Vec<u8> = (30..40i32).flat_map(i32::to_le_bytes).collect();
|
||||||
|
assert_eq!(decompress_chunk(&raw, &pipeline, 40, 4).unwrap(), want);
|
||||||
|
|
||||||
|
// And our own writer's round trip through the same pipeline order.
|
||||||
|
let data: Vec<u8> = (0..400u32).map(|i| (i % 13) as u8).collect();
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
one_filter(FILTER_SHUFFLE, vec![4]),
|
||||||
|
one_filter(FILTER_FLETCHER32, vec![]),
|
||||||
|
one_filter(FILTER_DEFLATE, vec![9]),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
let c = compress_chunk(&data, &pipeline, 4).unwrap();
|
||||||
|
assert_eq!(decompress_chunk(&c, &pipeline, 400, 4).unwrap(), data);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `le_data.h5` scale-offset (D-scale, D = 3, fill -2.2) chunks, decoded
|
||||||
|
/// bit for bit as libhdf5 does: single-precision arithmetic for `float`
|
||||||
|
/// (we computed in f64 and rounded once, which was 1 ULP off for e.g.
|
||||||
|
/// 1.6663333: `694ad53f` instead of `6a4ad53f`), double for `double`.
|
||||||
|
#[test]
|
||||||
|
fn scaleoffset_float_dscale_matches_libhdf5_bits() {
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||||
|
let cd = |size: u32, order: u32, fill_lo: u32, fill_hi: u32| {
|
||||||
|
let mut cd = vec![0, 3, 12, 1, size, 0, order, 1, fill_lo, fill_hi];
|
||||||
|
cd.resize(20, 0);
|
||||||
|
cd
|
||||||
|
};
|
||||||
|
let f32_le = cd(4, 0, 0xC00C_CCCD, 0);
|
||||||
|
let f32_be = cd(4, 1, 0xC00C_CCCD, 0);
|
||||||
|
let f64_le = cd(8, 0, 2576980378, 3221330329);
|
||||||
|
#[rustfmt::skip]
|
||||||
|
let cases: [(usize, usize, &[u32], &str); 6] = [
|
||||||
|
(2816, 38, &f32_le, "abaaaa3ed2942a3fec0a803fd2942a3fec0a803fabaaaa3fec0a803fabaaaa3f694ad53fabaaaa3f694ad53f76050040"),
|
||||||
|
(2854, 38, &f32_le, "abaaaa3f6a4ad53f760500406a4ad53f7605004056551540760500405655154034a52a405655154034a52a4076054040"),
|
||||||
|
(712, 38, &f32_be, "3eaaaaab3f2a94d23f800aec3f2a94d23f800aec3faaaaab3f800aec3faaaaab3fd54a693faaaaab3fd54a6940000576"),
|
||||||
|
(750, 38, &f32_be, "3faaaaab3fd54a6a400005763fd54a6a40000576401555564000057640155556402aa53440155556402aa53440400576"),
|
||||||
|
(2050, 38, &f64_le, concat!(
|
||||||
|
"555555555555d53fb9d75c489a52e53fce3e7c865d01f03fb9d75c489a52e53fce3e7c865d01f03f555555555555f53f",
|
||||||
|
"ce3e7c865d01f03f555555555555f53fdc6b2e244da9fa3f555555555555f53fdc6b2e244da9fa3f671f3ec3ae000040")),
|
||||||
|
(2088, 38, &f64_le, concat!(
|
||||||
|
"555555555555f53fdc6b2e244da9fa3f671f3ec3ae000040dc6b2e244da9fa3f671f3ec3ae000040aaaaaaaaaaaa0240",
|
||||||
|
"671f3ec3ae000040aaaaaaaaaaaa0240ee351792a6540540aaaaaaaaaaaa0240ee351792a6540540671f3ec3ae000840")),
|
||||||
|
];
|
||||||
|
for (off, len, cd, want) in cases {
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![one_filter(FILTER_SCALEOFFSET, cd.to_vec())],
|
||||||
|
};
|
||||||
|
let want = unhex(want);
|
||||||
|
let got =
|
||||||
|
decompress_chunk(&file[off..off + len], &pipeline, want.len(), cd[4]).unwrap();
|
||||||
|
assert_eq!(got, want, "chunk at {off}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `le_data.h5` `/Nbit_float_data_{le,be}` chunk (0,0): a 20-bit float
|
||||||
|
/// (offset 7) packed by N-Bit. The filter must reproduce libhdf5's
|
||||||
|
/// decoded bytes in the *file* datatype (h5py `DatasetID.read` with the
|
||||||
|
/// file type as memory type, so no conversion); converting that custom
|
||||||
|
/// float layout to IEEE is the datatype reader's job, not the filter's.
|
||||||
|
#[test]
|
||||||
|
fn nbit_float_matches_libhdf5_file_type_bytes() {
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||||
|
let cases = [
|
||||||
|
(
|
||||||
|
55952,
|
||||||
|
0,
|
||||||
|
"8055d5018055e5010000f0018055e5010000f0018055f5010000f0018055f50180aafa018055f50180aafa0100000002",
|
||||||
|
),
|
||||||
|
(
|
||||||
|
56076,
|
||||||
|
1,
|
||||||
|
"01d5558001e5558001f0000001e5558001f0000001f5558001f0000001f5558001faaa8001f5558001faaa8002000000",
|
||||||
|
),
|
||||||
|
];
|
||||||
|
for (off, order, want) in cases {
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![one_filter(FILTER_NBIT, vec![8, 0, 12, 1, 4, order, 20, 7])],
|
||||||
|
};
|
||||||
|
let got = decompress_chunk(&file[off..off + 31], &pipeline, 48, 4).unwrap();
|
||||||
|
assert_eq!(got, unhex(want), "byte order {order}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// libhdf5 sets `cd_values[1]` ("need not compress") when every field is
|
||||||
|
/// already full width and then stores the data unchanged; we unpacked it
|
||||||
|
/// anyway and failed with "nbit: packed data too short".
|
||||||
|
#[test]
|
||||||
|
fn nbit_need_not_compress_is_passthrough() {
|
||||||
|
let data: Vec<u8> = (0..200u32).map(|i| (i * 7) as u8).collect();
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![one_filter(FILTER_NBIT, vec![8, 1, 50, 1, 4, 0, 32, 0])],
|
||||||
|
};
|
||||||
|
assert_eq!(decompress_chunk(&data, &pipeline, 200, 4).unwrap(), data);
|
||||||
|
// A top-level type N-Bit has no parameters for (e.g. an enum) carries only
|
||||||
|
// [nparms, need_not_compress, nelmts].
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![one_filter(FILTER_NBIT, vec![3, 1, 50])],
|
||||||
|
};
|
||||||
|
assert_eq!(decompress_chunk(&data, &pipeline, 200, 4).unwrap(), data);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `tfilters.h5` `/all` chunk (0,0): shuffle, szip, deflate, fletcher32
|
||||||
|
/// and a pass-through N-Bit in one pipeline; values from h5py.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
fn nbit_in_multi_filter_pipeline_matches_libhdf5() {
|
||||||
|
let raw = unhex(concat!(
|
||||||
|
"785e3bc1c0c030cb6517ff039e556de1576c0f300b152c771070641170640b99ba879f8167d5cce82f760ccc4285d71b",
|
||||||
|
"20c2afc226329f60d60ab5636a3fc1905499c3c0a1d0c4a1d05c6a13d7f88271aaf6725fe7170c86363f1858c0ca0104",
|
||||||
|
"bf1e95c75f3eeb",
|
||||||
|
));
|
||||||
|
let want = unhex(concat!(
|
||||||
|
"00000000010000000200000003000000040000000a0000000b0000000c0000000d0000000e0000001400000015000000",
|
||||||
|
"1600000017000000180000001e0000001f00000020000000210000002200000028000000290000002a0000002b000000",
|
||||||
|
"2c00000032000000330000003400000035000000360000003c0000003d0000003e0000003f0000004000000046000000",
|
||||||
|
"4700000048000000490000004a00000050000000510000005200000053000000540000005a0000005b0000005c000000",
|
||||||
|
"5d0000005e000000",
|
||||||
|
));
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
one_filter(FILTER_SHUFFLE, vec![4]),
|
||||||
|
one_filter(FILTER_SZIP, vec![141, 4, 32, 5]),
|
||||||
|
one_filter(FILTER_DEFLATE, vec![5]),
|
||||||
|
one_filter(FILTER_FLETCHER32, vec![]),
|
||||||
|
one_filter(FILTER_NBIT, vec![8, 1, 50, 1, 4, 0, 32, 0]),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
assert_eq!(decompress_chunk(&raw, &pipeline, 200, 4).unwrap(), want);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `h5repack_nested_8bit_enum_deflated.h5` `/tracks/1/trace` chunk 0: a
|
||||||
|
/// 376-byte compound whose `u1` enum member N-Bit stores whole as a
|
||||||
|
/// no-op type (class 4), then deflate. Was `UnsupportedFilter(5)`.
|
||||||
|
/// Expected bytes: libhdf5's decode in the file datatype.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "deflate")]
|
||||||
|
fn nbit_compound_with_enum_member_matches_libhdf5() {
|
||||||
|
#[rustfmt::skip]
|
||||||
|
let cd: Vec<u32> = vec![
|
||||||
|
251, 0, 1, 3, 376, 38,
|
||||||
|
0, 1, 4, 0, 32, 0,
|
||||||
|
8, 2, 96, 1, 8, 0, 64, 0,
|
||||||
|
104, 1, 8, 0, 64, 0, 112, 1, 8, 0, 64, 0, 120, 1, 8, 0, 64, 0,
|
||||||
|
128, 1, 8, 0, 64, 0, 136, 1, 8, 0, 64, 0, 144, 1, 8, 0, 64, 0,
|
||||||
|
152, 1, 8, 0, 64, 0,
|
||||||
|
160, 2, 24, 1, 8, 0, 64, 0, 184, 2, 24, 1, 8, 0, 64, 0,
|
||||||
|
208, 2, 16, 1, 8, 0, 64, 0, 224, 2, 32, 1, 8, 0, 64, 0,
|
||||||
|
256, 2, 16, 1, 4, 0, 32, 0, 272, 2, 16, 1, 4, 0, 32, 0,
|
||||||
|
288, 2, 32, 1, 8, 0, 64, 0, 320, 2, 16, 1, 8, 0, 64, 0,
|
||||||
|
336, 2, 8, 1, 4, 0, 32, 0,
|
||||||
|
344, 4, 1,
|
||||||
|
346, 1, 2, 0, 16, 0, 348, 1, 4, 0, 32, 0, 352, 1, 1, 0, 4, 0,
|
||||||
|
354, 1, 2, 0, 16, 0, 356, 1, 1, 0, 4, 0, 357, 1, 1, 0, 4, 0,
|
||||||
|
358, 1, 1, 0, 4, 0, 359, 1, 1, 0, 4, 0, 360, 1, 1, 0, 4, 0,
|
||||||
|
361, 1, 1, 0, 4, 0, 362, 1, 1, 0, 4, 0, 364, 1, 1, 0, 4, 0,
|
||||||
|
363, 1, 1, 0, 4, 0, 365, 1, 1, 0, 4, 0, 366, 1, 1, 0, 4, 0,
|
||||||
|
367, 1, 1, 0, 4, 0, 368, 1, 1, 0, 4, 0, 369, 1, 1, 0, 4, 0,
|
||||||
|
370, 1, 1, 0, 4, 0,
|
||||||
|
];
|
||||||
|
assert_eq!(cd.len(), 251);
|
||||||
|
let raw = unhex(concat!(
|
||||||
|
"780163606078c930c880c3879573a60b2d7883eeac06a880835c47bda16cda7987a0c59ec9f74c4af6ff677e5df16445",
|
||||||
|
"adfd8493ce3ba592ddedab2a3bee87dd0faaff00d1808b66606006fa9d999b818125014837fd4703fba1fa71d150e720",
|
||||||
|
"512caa4073dc59212206500946060600604c37fc",
|
||||||
|
));
|
||||||
|
let want = unhex(concat!(
|
||||||
|
"e90000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||||
|
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||||
|
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000",
|
||||||
|
"0000000000000000eca012979ca9f040000000000000000000000000000000000000000000000080cf661d317f881e40",
|
||||||
|
"7434de6349a352407da8e478eb03ffbf47631ab943c9903f52df56df88797a3f000000000000f07f000000000000f07f",
|
||||||
|
"000000000000f07f000000000000f07fe90300000b0300006004000082030000ffffffffffffffffffffffffffffffff",
|
||||||
|
"000000000000f0bf000000000000f0bf000000000000f0bf000000000000f0bf00000000000000000000000000000000",
|
||||||
|
"25040000470300000500000000000000030000000000000000000000000000000100000000000000",
|
||||||
|
));
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![
|
||||||
|
one_filter(FILTER_NBIT, cd),
|
||||||
|
one_filter(FILTER_DEFLATE, vec![1]),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
assert_eq!(decompress_chunk(&raw, &pipeline, 376, 376).unwrap(), want);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Chunk (0,0) of `/DS1` in the HDF Group's `h5ex_d_lz4.h5` example,
|
||||||
|
/// written by libhdf5's registered LZ4 plugin with a 3-byte block size
|
||||||
|
/// (so it has many blocks, some stored raw). Byte range from h5py's
|
||||||
|
/// `get_chunk_info`; values are `i*j - j` (i32 LE), as h5py reads them.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_reads_registered_hdf5_format() {
|
||||||
|
let file: &[u8] = include_bytes!("../tests/fixtures/filters/h5ex_d_lz4.h5");
|
||||||
|
let chunk = &file[4016..4016 + 312];
|
||||||
|
let pipeline = FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![FilterDescription {
|
||||||
|
filter_id: FILTER_LZ4,
|
||||||
|
name: None,
|
||||||
|
flags: 1,
|
||||||
|
client_data: vec![3],
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
let out = decompress_chunk(chunk, &pipeline, 4 * 8 * 4, 4).unwrap();
|
||||||
|
let expected: Vec<u8> = (0..4i32)
|
||||||
|
.flat_map(|i| (0..8i32).map(move |j| i * j - j))
|
||||||
|
.flat_map(i32::to_le_bytes)
|
||||||
|
.collect();
|
||||||
|
assert_eq!(out, expected);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_writes_registered_hdf5_format() {
|
||||||
|
// Compressible data, one block: 8-byte BE size, 4-byte BE block size,
|
||||||
|
// 4-byte BE compressed length, block.
|
||||||
|
let data = vec![7u8; 1000];
|
||||||
|
let c = lz4_compress(&data, &[]).unwrap();
|
||||||
|
assert_eq!(&c[0..8], &1000u64.to_be_bytes());
|
||||||
|
assert_eq!(&c[8..12], &1000u32.to_be_bytes());
|
||||||
|
let len = u32::from_be_bytes(c[12..16].try_into().unwrap()) as usize;
|
||||||
|
assert_eq!(c.len(), 16 + len);
|
||||||
|
assert!(len < 1000);
|
||||||
|
assert_eq!(lz4_decompress(&c, 1000).unwrap(), data);
|
||||||
|
|
||||||
|
// Several blocks, incompressible ones stored raw (length == block).
|
||||||
|
let data: Vec<u8> = (0..10u8).collect();
|
||||||
|
let c = lz4_compress(&data, &[3]).unwrap();
|
||||||
|
assert_eq!(&c[8..12], &3u32.to_be_bytes());
|
||||||
|
assert_eq!(&c[12..16], &3u32.to_be_bytes());
|
||||||
|
assert_eq!(&c[16..19], &[0, 1, 2]);
|
||||||
|
assert_eq!(c.len(), 12 + 3 * (4 + 3) + (4 + 1));
|
||||||
|
assert_eq!(lz4_decompress(&c, 10).unwrap(), data);
|
||||||
|
|
||||||
|
let c = lz4_compress(&[], &[]).unwrap();
|
||||||
|
assert_eq!(lz4_decompress(&c, 0).unwrap(), Vec::<u8>::new());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Chunks written by clawhdf5 up to 2.7.0 (4-byte LE size + one LZ4
|
||||||
|
/// block) must stay readable.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_reads_legacy_clawhdf5_format() {
|
||||||
|
for data in [vec![], vec![5u8; 300], (0..=255u8).collect::<Vec<u8>>()] {
|
||||||
|
let mut legacy = (data.len() as u32).to_le_bytes().to_vec();
|
||||||
|
legacy.extend_from_slice(&lz4_flex::block::compress(&data));
|
||||||
|
assert_eq!(lz4_decompress(&legacy, data.len()).unwrap(), data);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
fn lz4_registered_format_rejects_hostile_sizes() {
|
||||||
|
// Declared total larger than the chunk.
|
||||||
|
let mut c = 1000u64.to_be_bytes().to_vec();
|
||||||
|
c.extend_from_slice(&1000u32.to_be_bytes());
|
||||||
|
c.extend_from_slice(&[0u8; 8]);
|
||||||
|
assert!(lz4_decompress(&c, 64).is_err());
|
||||||
|
// Truncated block.
|
||||||
|
let mut c = 16u64.to_be_bytes().to_vec();
|
||||||
|
c.extend_from_slice(&16u32.to_be_bytes());
|
||||||
|
c.extend_from_slice(&16u32.to_be_bytes());
|
||||||
|
c.extend_from_slice(&[1u8; 4]);
|
||||||
|
assert!(lz4_decompress(&c, 16).is_err());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn lz4_compress_decompress_roundtrip() {
|
fn lz4_compress_decompress_roundtrip() {
|
||||||
let data: Vec<u8> = (0..256).map(|i| (i % 256) as u8).collect();
|
let data: Vec<u8> = (0..256).map(|i| (i % 256) as u8).collect();
|
||||||
let compressed = lz4_compress(&data).unwrap();
|
let compressed = lz4_compress(&data, &[]).unwrap();
|
||||||
let decompressed = lz4_decompress(&compressed, data.len()).unwrap();
|
let decompressed = lz4_decompress(&compressed, data.len()).unwrap();
|
||||||
assert_eq!(decompressed, data);
|
assert_eq!(decompressed, data);
|
||||||
}
|
}
|
||||||
@@ -1706,6 +2158,23 @@ mod tests {
|
|||||||
|
|
||||||
// --- Zstd tests ---
|
// --- Zstd tests ---
|
||||||
|
|
||||||
|
/// libhdf5's zstd plugin needs the frame content size to size its
|
||||||
|
/// output; frames without it fail to decode there.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "zstd")]
|
||||||
|
fn zstd_frames_record_content_size() {
|
||||||
|
for n in [0usize, 1, 200, 100_000] {
|
||||||
|
let data: Vec<u8> = (0..n).map(|i| (i % 7) as u8).collect();
|
||||||
|
let c = zstd_compress(&data, 3).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
zstd::zstd_safe::get_frame_content_size(&c).unwrap(),
|
||||||
|
Some(n as u64),
|
||||||
|
"{n} bytes"
|
||||||
|
);
|
||||||
|
assert_eq!(zstd_decompress(&c, n).unwrap(), data);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "zstd")]
|
#[cfg(feature = "zstd")]
|
||||||
fn zstd_compress_decompress_roundtrip() {
|
fn zstd_compress_decompress_roundtrip() {
|
||||||
@@ -2167,6 +2636,57 @@ mod tests {
|
|||||||
assert!(pcodec_decompress(&compressed, 4, 16).is_err());
|
assert!(pcodec_decompress(&compressed, 4, 16).is_err());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Pcodec is written under the private ID 480, not 32023 (registered to
|
||||||
|
/// Granular BitRound, whose pass-through decode would hand libhdf5 users
|
||||||
|
/// the compressed bytes as data). Chunks under 32023 are read as pcodec
|
||||||
|
/// only with the name clawhdf5 <= 2.7.0 wrote.
|
||||||
|
#[test]
|
||||||
|
#[cfg(feature = "pcodec")]
|
||||||
|
fn pcodec_uses_private_id_and_reads_legacy_32023() {
|
||||||
|
use crate::chunked_write::ChunkOptions;
|
||||||
|
let opts = ChunkOptions {
|
||||||
|
pcodec: true,
|
||||||
|
..Default::default()
|
||||||
|
};
|
||||||
|
let pl = opts.build_pipeline(8).unwrap();
|
||||||
|
let f = pl.filters.iter().find(|f| f.filter_id == 480).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
f.name.as_deref(),
|
||||||
|
Some(crate::filter_pipeline::FILTER_PCODEC_NAME)
|
||||||
|
);
|
||||||
|
assert!(pl.filters.iter().all(|f| f.filter_id != 32023));
|
||||||
|
|
||||||
|
let data: Vec<f64> = (0..100).map(|i| i as f64 * 0.25).collect();
|
||||||
|
let raw: Vec<u8> = data.iter().flat_map(|x| x.to_le_bytes()).collect();
|
||||||
|
let compressed = pcodec_compress(&raw, 8).unwrap();
|
||||||
|
let pipeline = |id: u16, name: Option<&str>| FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters: vec![FilterDescription {
|
||||||
|
filter_id: id,
|
||||||
|
name: name.map(Into::into),
|
||||||
|
flags: 0,
|
||||||
|
client_data: vec![8],
|
||||||
|
}],
|
||||||
|
};
|
||||||
|
let legacy = pipeline(32023, Some("pcodec"));
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(&compressed, &legacy, raw.len(), 8).unwrap(),
|
||||||
|
raw
|
||||||
|
);
|
||||||
|
let current = pipeline(480, None);
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk(&compressed, ¤t, raw.len(), 8).unwrap(),
|
||||||
|
raw
|
||||||
|
);
|
||||||
|
// A real Granular BitRound filter is not pcodec.
|
||||||
|
for name in [None, Some("Granular BitRound")] {
|
||||||
|
assert!(matches!(
|
||||||
|
decompress_chunk(&compressed, &pipeline(32023, name), raw.len(), 8),
|
||||||
|
Err(FormatError::UnsupportedFilter(32023))
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "lz4")]
|
#[cfg(feature = "lz4")]
|
||||||
fn decompress_chunk_rejects_hostile_lz4_size_via_public_entrypoint() {
|
fn decompress_chunk_rejects_hostile_lz4_size_via_public_entrypoint() {
|
||||||
|
|||||||
@@ -1,19 +1,39 @@
|
|||||||
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
//! SZIP (libaec Adaptive Entropy Coding) decompression.
|
||||||
//!
|
//!
|
||||||
//! Gated by the `szip` feature which links against the system libaec library.
|
//! Gated by the `szip` feature which links against the system libaec library.
|
||||||
|
//!
|
||||||
|
//! libhdf5's SZIP filter (`H5Zszip.c`) prefixes each chunk with its
|
||||||
|
//! uncompressed size and hands the rest to szlib's `SZ_BufftoBuffDecompress`.
|
||||||
|
//! libaec implements that call (`sz_compat.c`) on top of `aec_buffer_decode`
|
||||||
|
//! with some reshaping — 32/64-bit samples are coded as byte planes of 8-bit
|
||||||
|
//! samples, and scanlines that are not a whole number of blocks are padded —
|
||||||
|
//! which [`szip_decompress`] reproduces so its output matches libhdf5's.
|
||||||
|
|
||||||
#[cfg(not(feature = "std"))]
|
#[cfg(not(feature = "std"))]
|
||||||
use alloc::vec::Vec;
|
use alloc::vec::Vec;
|
||||||
|
|
||||||
use crate::error::FormatError;
|
use crate::error::FormatError;
|
||||||
|
|
||||||
/// Decompress SZIP-compressed data using libaec.
|
/// `SZ_MSB_OPTION_MASK`: samples are big-endian.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
const SZ_MSB_OPTION_MASK: u32 = 16;
|
||||||
|
/// `SZ_NN_OPTION_MASK`: nearest-neighbour preprocessing.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
const SZ_NN_OPTION_MASK: u32 = 32;
|
||||||
|
|
||||||
|
/// Decompress one SZIP-filtered chunk.
|
||||||
///
|
///
|
||||||
/// `cd` is the HDF5 SZIP filter client data (matches `H5Z_SZIP_PARM_*` indices):
|
/// `cd` is the HDF5 SZIP filter client data (`H5Z_SZIP_PARM_*` indices):
|
||||||
/// cd[0] = options mask (`H5_SZIP_NN_OPTION_MASK = 0x20` enables NN preprocessing)
|
/// cd[0] = options mask (`SZ_*_OPTION_MASK`: 16 = MSB byte order,
|
||||||
/// cd[1] = pixels per block (H5Z_SZIP_PARM_PPB; 8, 10, 16, or 32)
|
/// 32 = nearest-neighbour preprocessing; K13/EC/LSB/RAW bits carry
|
||||||
/// cd[2] = bits per sample (H5Z_SZIP_PARM_BPP; element bit width)
|
/// no decoding information for libaec)
|
||||||
/// cd[3] = pixels per scan line (H5Z_SZIP_PARM_PPS; informational only)
|
/// cd[1] = pixels per block
|
||||||
|
/// cd[2] = bits per pixel (sample precision, rounded up to 32 or 64 above
|
||||||
|
/// 24 by libhdf5)
|
||||||
|
/// cd[3] = pixels per scanline
|
||||||
|
///
|
||||||
|
/// The chunk is a 4-byte little-endian uncompressed size followed by the
|
||||||
|
/// szlib stream.
|
||||||
pub(crate) fn szip_decompress(
|
pub(crate) fn szip_decompress(
|
||||||
_data: &[u8],
|
_data: &[u8],
|
||||||
_cd: &[u32],
|
_cd: &[u32],
|
||||||
@@ -33,62 +53,174 @@ pub(crate) fn szip_decompress(
|
|||||||
|
|
||||||
#[cfg(feature = "szip")]
|
#[cfg(feature = "szip")]
|
||||||
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
fn szip_decode_impl(data: &[u8], cd: &[u32], chunk_size: usize) -> Result<Vec<u8>, FormatError> {
|
||||||
if cd.len() < 3 {
|
let err = |m: &str| FormatError::ChunkedReadError(format!("szip: {m}"));
|
||||||
return Err(FormatError::ChunkedReadError(
|
if cd.len() < 4 {
|
||||||
"szip: missing client data".into(),
|
return Err(err("missing client data"));
|
||||||
));
|
|
||||||
}
|
}
|
||||||
let options = cd[0];
|
let options = cd[0];
|
||||||
let pixels_per_block = cd[1];
|
let pixels_per_block = cd[1] as usize;
|
||||||
let bits_per_sample = cd[2]; // H5Z_SZIP_PARM_BPP
|
let bits_per_pixel = cd[2];
|
||||||
if bits_per_sample == 0 || bits_per_sample > 32 {
|
let pixels_per_scanline = cd[3] as usize;
|
||||||
return Err(FormatError::ChunkedReadError(
|
if !(1..=32).contains(&bits_per_pixel) && bits_per_pixel != 64 {
|
||||||
"szip: invalid bits per sample".into(),
|
return Err(err("invalid bits per sample"));
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if chunk_size == 0 {
|
if pixels_per_block == 0 || pixels_per_scanline == 0 {
|
||||||
return Err(FormatError::ChunkedReadError(
|
return Err(err("invalid block or scanline size"));
|
||||||
"szip: unknown output size".into(),
|
|
||||||
));
|
|
||||||
}
|
}
|
||||||
if data.is_empty() {
|
if data.len() < 4 {
|
||||||
return Err(FormatError::ChunkedReadError("szip: empty input".into()));
|
return Err(err("chunk too short"));
|
||||||
}
|
}
|
||||||
|
// H5Zszip.c: UINT32DECODE of the uncompressed size, then the stream.
|
||||||
|
let dest_len = u32::from_le_bytes([data[0], data[1], data[2], data[3]]) as usize;
|
||||||
|
let limit = if chunk_size != 0 {
|
||||||
|
chunk_size
|
||||||
|
} else {
|
||||||
|
crate::filters::MAX_DECOMPRESS_SIZE
|
||||||
|
};
|
||||||
|
if dest_len > limit {
|
||||||
|
return Err(err("declared size exceeds chunk size"));
|
||||||
|
}
|
||||||
|
let stream = &data[4..];
|
||||||
|
|
||||||
// Map HDF5 option mask to libaec flags.
|
// --- libaec sz_compat.c: SZ_BufftoBuffDecompress ---
|
||||||
// HDF5 always stores SZIP data in MSB order, so AEC_DATA_MSB is unconditional.
|
let rsi = pixels_per_scanline.div_ceil(pixels_per_block);
|
||||||
// H5_SZIP_NN_OPTION_MASK (0x20): NN differential preprocessing.
|
let mut flags = 0;
|
||||||
let mut flags: u32 = libaec_sys::AEC_DATA_MSB;
|
if options & SZ_MSB_OPTION_MASK != 0 {
|
||||||
if options & 0x20 != 0 {
|
flags |= libaec_sys::AEC_DATA_MSB;
|
||||||
|
}
|
||||||
|
if options & SZ_NN_OPTION_MASK != 0 {
|
||||||
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
flags |= libaec_sys::AEC_DATA_PREPROCESS;
|
||||||
}
|
}
|
||||||
|
let pad_scanline = !pixels_per_scanline.is_multiple_of(pixels_per_block);
|
||||||
|
let deinterleave = bits_per_pixel == 32 || bits_per_pixel == 64;
|
||||||
|
let bits_per_sample = if deinterleave { 8 } else { bits_per_pixel };
|
||||||
|
let pixel_size = match bits_per_sample {
|
||||||
|
17.. => 4,
|
||||||
|
9.. => 2,
|
||||||
|
_ => 1,
|
||||||
|
};
|
||||||
|
let scanlines = (dest_len / pixel_size).div_ceil(pixels_per_scanline);
|
||||||
|
let buf_size = if pad_scanline {
|
||||||
|
rsi.checked_mul(pixels_per_block)
|
||||||
|
.and_then(|n| n.checked_mul(pixel_size))
|
||||||
|
.and_then(|n| n.checked_mul(scanlines))
|
||||||
|
.filter(|&n| n <= crate::filters::MAX_DECOMPRESS_SIZE.max(limit))
|
||||||
|
.ok_or_else(|| err("scanline padding too large"))?
|
||||||
|
} else {
|
||||||
|
dest_len
|
||||||
|
};
|
||||||
|
|
||||||
let mut out = vec![0u8; chunk_size];
|
let mut buf = vec![0u8; buf_size];
|
||||||
let mut strm = libaec_sys::AecStream::zeroed();
|
let mut strm = libaec_sys::AecStream::zeroed();
|
||||||
strm.next_in = data.as_ptr();
|
strm.next_in = stream.as_ptr();
|
||||||
strm.avail_in = data.len();
|
strm.avail_in = stream.len();
|
||||||
strm.next_out = out.as_mut_ptr();
|
strm.next_out = buf.as_mut_ptr();
|
||||||
strm.avail_out = chunk_size;
|
strm.avail_out = buf_size;
|
||||||
strm.bits_per_sample = bits_per_sample;
|
strm.bits_per_sample = bits_per_sample;
|
||||||
strm.block_size = pixels_per_block;
|
strm.block_size = pixels_per_block as u32;
|
||||||
strm.rsi = 128; // HDF5 default: 128 blocks per reference sample interval
|
strm.rsi = rsi as u32;
|
||||||
strm.flags = flags;
|
strm.flags = flags;
|
||||||
|
// SAFETY: next_in/avail_in and next_out/avail_out describe live buffers
|
||||||
|
// (`stream` and `buf`) that outlive the call.
|
||||||
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
let result = unsafe { libaec_sys::aec_buffer_decode(&mut strm) };
|
||||||
if result != 0 {
|
if result != 0 {
|
||||||
return Err(FormatError::DecompressionError(format!(
|
return Err(FormatError::DecompressionError(format!(
|
||||||
"szip: libaec error {result}"
|
"szip: libaec error {result}"
|
||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
let decoded_len = chunk_size - strm.avail_out;
|
let mut total_out = strm.total_out;
|
||||||
out.truncate(decoded_len);
|
if pad_scanline {
|
||||||
|
let line = pixels_per_scanline * pixel_size;
|
||||||
|
let padded_line = rsi * pixels_per_block * pixel_size;
|
||||||
|
// remove_padding: compact each padded line down to `line` bytes.
|
||||||
|
let mut i = line;
|
||||||
|
let mut j = padded_line;
|
||||||
|
while j < total_out {
|
||||||
|
let end = (j + line).min(buf.len());
|
||||||
|
buf.copy_within(j..end, i);
|
||||||
|
i += line;
|
||||||
|
j += padded_line;
|
||||||
|
}
|
||||||
|
total_out = scanlines * line;
|
||||||
|
}
|
||||||
|
if total_out < dest_len {
|
||||||
|
return Err(err("stream decoded to fewer bytes than declared"));
|
||||||
|
}
|
||||||
|
buf.truncate(dest_len);
|
||||||
|
if deinterleave {
|
||||||
|
// deinterleave_buffer: byte planes back into words.
|
||||||
|
let w = (bits_per_pixel / 8) as usize;
|
||||||
|
let n = dest_len / w;
|
||||||
|
let mut out = vec![0u8; dest_len];
|
||||||
|
for i in 0..n {
|
||||||
|
for j in 0..w {
|
||||||
|
out[i * w + j] = buf[j * n + i];
|
||||||
|
}
|
||||||
|
}
|
||||||
Ok(out)
|
Ok(out)
|
||||||
|
} else {
|
||||||
|
Ok(buf)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
fn unhex(s: &str) -> Vec<u8> {
|
||||||
|
(0..s.len())
|
||||||
|
.step_by(2)
|
||||||
|
.map(|i| u8::from_str_radix(&s[i..i + 2], 16).unwrap())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// SZIP chunks written by libhdf5, decoded exactly as libhdf5 decodes
|
||||||
|
/// them. Each case: fixture, chunk byte offset and size (from h5py's
|
||||||
|
/// `get_chunk_info`), the filter's cd_values, and the chunk's values as
|
||||||
|
/// h5py reads them (file byte order, hex). Before the fix every one of
|
||||||
|
/// these came back as garbage or zeros (or "invalid bits per sample" for
|
||||||
|
/// 64-bit): the 4-byte size prefix was fed to libaec, 32/64-bit samples
|
||||||
|
/// were not de-interleaved from byte planes, the reference sample
|
||||||
|
/// interval was fixed at 128 instead of derived from the scanline, padded
|
||||||
|
/// scanlines were not unpadded, and LE data was decoded as MSB.
|
||||||
|
#[cfg(feature = "szip")]
|
||||||
|
#[test]
|
||||||
|
fn szip_decodes_libhdf5_chunks_exactly() {
|
||||||
|
/// (name, file, chunk offset, chunk size, cd_values, decoded hex)
|
||||||
|
type Case<'a> = (&'a str, &'a [u8], usize, usize, [u32; 4], &'a str);
|
||||||
|
let noencoder: &[u8] = include_bytes!("../tests/fixtures/filters/noencoder.h5");
|
||||||
|
let le_data: &[u8] = include_bytes!("../tests/fixtures/filters/le_data.h5");
|
||||||
|
let h5py: &[u8] = include_bytes!("../tests/fixtures/filters/szip_h5py.h5");
|
||||||
|
#[rustfmt::skip]
|
||||||
|
let cases: &[Case] = &[
|
||||||
|
// <i4, 10 px/scanline over 4 px/block: padded scanlines + byte planes.
|
||||||
|
("noencoder /noencoder_szip_dset.h5", noencoder, 6040, 16, [168, 4, 32, 10],
|
||||||
|
"00000000010000000200000003000000040000000500000006000000070000000800000009000000"),
|
||||||
|
// <f4, LSB + NN.
|
||||||
|
("le_data /Szip_float_data_le", le_data, 55224, 48, [169, 4, 32, 12],
|
||||||
|
"abaaaa3eabaa2a3f0000803fabaa2a3f0000803fabaaaa3f0000803fabaaaa3f5555d53fabaaaa3f5555d53f00000040"),
|
||||||
|
// >f4, MSB + NN.
|
||||||
|
("le_data /Szip_float_data_be", le_data, 55396, 48, [177, 4, 32, 12],
|
||||||
|
"3eaaaaab3f2aaaab3f8000003f2aaaab3f8000003faaaaab3f8000003faaaaab3fd555553faaaaab3fd5555540000000"),
|
||||||
|
// <f8 (64-bit), NN.
|
||||||
|
("szip_h5py /f8", h5py, 4016, 100, [169, 8, 64, 10],
|
||||||
|
"00000000000008c000000000000008c000000000000008c000000000000008c000000000000004c000000000000004c000000000000004c000000000000004c000000000000000c000000000000000c000000000000000c000000000000000c0000000000000f8bf000000000000f8bf000000000000f8bf000000000000f8bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000f0bf000000000000e0bf000000000000e0bf000000000000e0bf000000000000e0bf0000000000000000000000000000000000000000000000000000000000000000000000000000e03f000000000000e03f000000000000e03f000000000000e03f000000000000f03f000000000000f03f000000000000f03f000000000000f03f000000000000f83f000000000000f83f000000000000f83f000000000000f83f"),
|
||||||
|
// <i8 (64-bit), entropy coding without NN.
|
||||||
|
("szip_h5py /i8", h5py, 4188, 53, [141, 4, 64, 10],
|
||||||
|
"000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000003000000000000000300000000000000030000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000060000000000000006000000000000000600000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000900000000000000090000000000000009000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c000000000000000c00000000000000"),
|
||||||
|
// <u2, 35 px/scanline over 8 px/block: padded scanlines, 16-bit samples.
|
||||||
|
("szip_h5py /u2", h5py, 4308, 43, [169, 8, 16, 35],
|
||||||
|
"00000000000000006100610061006100c200c200c200c20023012301230123018401840184018401e501e501e501e5014602460246024602a702a702a702a702080308030803"),
|
||||||
|
];
|
||||||
|
for (name, file, off, len, cd, want) in cases {
|
||||||
|
let want = unhex(want);
|
||||||
|
let got = szip_decompress(&file[*off..off + len], cd, want.len())
|
||||||
|
.unwrap_or_else(|e| panic!("{name}: {e:?}"));
|
||||||
|
assert_eq!(got, want, "{name}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn szip_disabled_returns_unsupported() {
|
fn szip_disabled_returns_unsupported() {
|
||||||
#[cfg(not(feature = "szip"))]
|
#[cfg(not(feature = "szip"))]
|
||||||
@@ -132,6 +264,8 @@ mod tests {
|
|||||||
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
assert_eq!(rc, 0, "aec_buffer_encode failed: {rc}");
|
||||||
let enc_len = encoded.len() - enc.avail_out;
|
let enc_len = encoded.len() - enc.avail_out;
|
||||||
encoded.truncate(enc_len);
|
encoded.truncate(enc_len);
|
||||||
|
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||||
|
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||||
|
|
||||||
// Decode through our public interface.
|
// Decode through our public interface.
|
||||||
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
// cd[0]=0 (no NN bit 0x20), cd[1]=8 (ppb), cd[2]=8 (bpp), cd[3]=1024 (pps).
|
||||||
@@ -163,6 +297,8 @@ mod tests {
|
|||||||
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
assert_eq!(rc, 0, "aec_buffer_encode with NN failed: {rc}");
|
||||||
let enc_len = encoded.len() - enc.avail_out;
|
let enc_len = encoded.len() - enc.avail_out;
|
||||||
encoded.truncate(enc_len);
|
encoded.truncate(enc_len);
|
||||||
|
// H5Zszip.c prefixes the stream with the uncompressed size.
|
||||||
|
encoded.splice(0..0, (original.len() as u32).to_le_bytes());
|
||||||
|
|
||||||
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
// cd[0] = 0x20 (H5_SZIP_NN_OPTION_MASK) → decoder must set AEC_DATA_PREPROCESS.
|
||||||
let cd = [0x20u32, 8, 8, 1024];
|
let cd = [0x20u32, 8, 8, 1024];
|
||||||
|
|||||||
@@ -731,7 +731,12 @@ impl DatasetBuilder {
|
|||||||
self
|
self
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Enable Pcodec lossless numerical compression (clawhdf5 filter ID 32023).
|
/// Enable Pcodec lossless numerical compression (private clawhdf5 filter
|
||||||
|
/// ID 480).
|
||||||
|
///
|
||||||
|
/// **Not interoperable:** pcodec has no registered HDF5 filter ID and no
|
||||||
|
/// libhdf5 plugin, so h5py and other HDF5 readers cannot read the
|
||||||
|
/// dataset — only clawhdf5 built with the `pcodec` feature can.
|
||||||
///
|
///
|
||||||
/// Pcodec achieves 30–94% better compression ratio than Zstd for f32/f64
|
/// Pcodec achieves 30–94% better compression ratio than Zstd for f32/f64
|
||||||
/// columns at 1–5 GiB/s decompression speed (arXiv:2502.06112). Requires
|
/// columns at 1–5 GiB/s decompression speed (arXiv:2502.06112). Requires
|
||||||
|
|||||||
@@ -0,0 +1,13 @@
|
|||||||
|
# Filter conformance fixtures
|
||||||
|
|
||||||
|
Files written by libhdf5 (and its registered filter plugins), used by the
|
||||||
|
filter regression tests in `src/filters.rs` to compare our decoders against
|
||||||
|
the values h5py/libhdf5 read from the same bytes. Chunk byte ranges quoted in
|
||||||
|
the tests come from h5py's `DatasetID.get_chunk_info`.
|
||||||
|
|
||||||
|
| File | Origin | Licence |
|
||||||
|
|------|--------|---------|
|
||||||
|
| `h5ex_d_lz4.h5` | HDF Group `HDF5Examples/C/H5FLT/tfiles/h5ex_d_lz4.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||||
|
| `noencoder.h5` | HDF Group `test/testfiles/noencoder.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||||
|
| `le_data.h5` | HDF Group `test/testfiles/le_data.h5` (hdf5 repository) | HDF5 licence (BSD-3-Clause style) |
|
||||||
|
| `szip_h5py.h5` | Written for these tests with h5py 3 / libhdf5 2.0.0 (libaec szip): `f8` (8x10, chunks 4x10, `('nn', 8)`), `i8` (8x10, chunks 4x10, `('ec', 4)`), `u2` (70, chunks 35, `('nn', 8)`) | Same as this repository |
|
||||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -940,3 +940,61 @@ fn provenance_verify_written_file() {
|
|||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(result, clawhdf5_format::provenance::VerifyResult::Ok);
|
assert_eq!(result, clawhdf5_format::provenance::VerifyResult::Ok);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ---- hdf5plugin interop: registered third-party compression filters ----
|
||||||
|
|
||||||
|
/// Write `data` (f64, 1-D, chunked) with `configure` applied, then read it
|
||||||
|
/// back with h5py + hdf5plugin (libhdf5's registered filter plugins) and
|
||||||
|
/// return the values it decodes.
|
||||||
|
#[cfg(any(feature = "lz4", feature = "zstd"))]
|
||||||
|
fn hdf5plugin_roundtrip(
|
||||||
|
tag: &str,
|
||||||
|
data: &[f64],
|
||||||
|
configure: impl FnOnce(&mut clawhdf5_format::type_builders::DatasetBuilder),
|
||||||
|
) -> Vec<f64> {
|
||||||
|
let mut fw = FileWriter::new();
|
||||||
|
let ds = fw.create_dataset("data");
|
||||||
|
ds.with_f64_data(data)
|
||||||
|
.with_shape(&[data.len() as u64])
|
||||||
|
.with_chunks(&[250]);
|
||||||
|
configure(ds);
|
||||||
|
let bytes = fw.finish().unwrap();
|
||||||
|
let path = std::env::temp_dir().join(format!("clawhdf5_hdf5plugin_{tag}.h5"));
|
||||||
|
std::fs::write(&path, &bytes).unwrap();
|
||||||
|
let script = format!(
|
||||||
|
"import h5py,hdf5plugin,json; f=h5py.File('{}','r'); print(json.dumps(f['data'][:].tolist()))",
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
let stdout = h5py_read(&path, &script);
|
||||||
|
serde_json::from_str(&stdout).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// libhdf5's LZ4 plugin must decode what we write (it could not while we
|
||||||
|
/// wrote a private 4-byte-LE-size framing).
|
||||||
|
#[cfg(feature = "lz4")]
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py + hdf5plugin"]
|
||||||
|
fn hdf5plugin_reads_our_lz4() {
|
||||||
|
let data: Vec<f64> = (0..1000).map(|i| (i % 37) as f64 * 0.5).collect();
|
||||||
|
let got = hdf5plugin_roundtrip("lz4", &data, |ds| {
|
||||||
|
ds.with_lz4();
|
||||||
|
});
|
||||||
|
assert_eq!(got, data);
|
||||||
|
let got = hdf5plugin_roundtrip("lz4_noshuffle", &data, |ds| {
|
||||||
|
ds.with_lz4().without_shuffle();
|
||||||
|
});
|
||||||
|
assert_eq!(got, data);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// libhdf5's Zstandard plugin must decode what we write (it could not while
|
||||||
|
/// our frames lacked the content size).
|
||||||
|
#[cfg(feature = "zstd")]
|
||||||
|
#[test]
|
||||||
|
#[ignore = "requires Python h5py + hdf5plugin"]
|
||||||
|
fn hdf5plugin_reads_our_zstd() {
|
||||||
|
let data: Vec<f64> = (0..1000).map(|i| (i % 37) as f64 * 0.5).collect();
|
||||||
|
let got = hdf5plugin_roundtrip("zstd", &data, |ds| {
|
||||||
|
ds.with_zstd(3);
|
||||||
|
});
|
||||||
|
assert_eq!(got, data);
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user