clawhdf5-format: the writer skips optional filters that fail, as libhdf5 does
FileBuilder stored every chunk of an LZF or Blosc dataset through the
filter with filter mask 0. libhdf5 counts LZF and Blosc output no smaller
than the chunk as a failure of the optional filter and stores the chunk raw
with the filter's mask bit set. For an LZF chunk whose stream was exactly
the chunk's size, the first libhdf5 rewrite stored raw data at the same
size and kept our stale mask 0 in the index, and h5py could no longer read
the dataset.
precompress_chunks now runs chunks through compress_chunk_masked (as
FileEditor does since f7e2ab1), sequentially and on the parallel path, and
build_chunked_data_from_precompressed records each chunk's real mask in
every index the writer builds: single chunk (layout field), Fixed Array and
Extensible Array filtered elements, and version-2 B-tree type 11 records
(create_datasets_parallel goes through the same path). The writer builds
no version-1 B-tree or implicit index. PrecompressedChunks::chunks gains
the mask. Files whose chunks all compress are byte-identical.
Latent only in the unreleased LZF/Blosc writer (added 2026-09-26); no
tagged release writes either filter.
Tests:
- plugin_filters_interop skipped_optional_filters_are_masked_as_libhdf5_masks_them:
LZF, shuffle+LZF+fletcher32 and Blosc over random, compressible and
alternating chunks in every index; masks equal an h5py-written twin's;
h5py r+ rewrites and extends them; h5py, h5dump and our reader read
every value. Before: 20 of 24 datasets had masks other than h5py's, and
with that check disabled h5py failed to read the rewritten datasets
("filter returned failure during read").
- plugin_filters_interop files_whose_chunks_all_compress_are_unchanged:
pins the pre-fix bytes of five all-compressing files.
- chunked_write skipped_lzf_chunks_are_masked_in_every_index (fails before:
mask 0, want 2).
Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -586,3 +586,457 @@ with h5py.File(sys.argv[1], 'w') as f:
|
||||
let want: Vec<i32> = (0..16).collect();
|
||||
assert_eq!(file.dataset("lzf_ok").unwrap().read_i32().unwrap(), want);
|
||||
}
|
||||
|
||||
/// A family of datasets for the filter-mask tests: element type, chunk
|
||||
/// length along the last dimension, the h5py `create_dataset` keywords of
|
||||
/// the same filters, and how chunk `k` is filled.
|
||||
#[cfg(feature = "lzf")]
|
||||
struct MaskFamily {
|
||||
name: &'static str,
|
||||
/// 1 (`u1`) or 4 (`<i4`).
|
||||
elem: usize,
|
||||
chunk: u64,
|
||||
h5py_kw: &'static str,
|
||||
build: fn(&mut clawhdf5_format::type_builders::DatasetBuilder),
|
||||
fill: MaskFill,
|
||||
}
|
||||
|
||||
#[cfg(feature = "lzf")]
|
||||
#[derive(Clone, Copy)]
|
||||
enum MaskFill {
|
||||
/// `[x, 0, 0, 0, 0]`: LZF's output is exactly the chunk's size, so
|
||||
/// libhdf5's LZF filter fails and h5py stores the chunk raw.
|
||||
FiveBytes,
|
||||
/// Even chunks random bytes, odd chunks one repeated value.
|
||||
Alternating,
|
||||
/// Every chunk one repeated value.
|
||||
Compressible,
|
||||
}
|
||||
|
||||
/// The index shapes of the mask tests: (label, shape, chunk dims, maxshape
|
||||
/// with `u64::MAX` unlimited) for chunk length `c`. Single chunk, Fixed
|
||||
/// Array, Extensible Array and version-2 B-tree, in that order — every
|
||||
/// index the whole-file writer builds.
|
||||
#[cfg(feature = "lzf")]
|
||||
#[allow(clippy::type_complexity)]
|
||||
fn mask_layouts(c: u64) -> Vec<(&'static str, Vec<u64>, Vec<u64>, Option<Vec<u64>>)> {
|
||||
vec![
|
||||
("single", vec![c], vec![c], None),
|
||||
("fixed", vec![4 * c], vec![c], None),
|
||||
("ea", vec![4 * c], vec![c], Some(vec![u64::MAX])),
|
||||
(
|
||||
"bt2",
|
||||
vec![2, 2 * c],
|
||||
vec![1, c],
|
||||
Some(vec![u64::MAX, u64::MAX]),
|
||||
),
|
||||
]
|
||||
}
|
||||
|
||||
/// Raw little-endian bytes of the dataset `fam` fills over `shape`.
|
||||
#[cfg(feature = "lzf")]
|
||||
fn mask_data(fam: &MaskFamily, shape: &[u64], seed: u64) -> Vec<u8> {
|
||||
let c = fam.chunk as usize;
|
||||
let cols = *shape.last().unwrap() as usize;
|
||||
let n: usize = shape.iter().product::<u64>() as usize;
|
||||
let chunks_per_row = cols.div_ceil(c);
|
||||
let mut state = seed;
|
||||
let mut noise = move || {
|
||||
state = state.wrapping_add(0x9E37_79B9_7F4A_7C15);
|
||||
let mut z = state;
|
||||
z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9);
|
||||
z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB);
|
||||
z ^ (z >> 31)
|
||||
};
|
||||
let mut out = Vec::with_capacity(n * fam.elem);
|
||||
for i in 0..n {
|
||||
let (row, col) = (i / cols, i % cols);
|
||||
let k = row * chunks_per_row + col / c;
|
||||
let v: u64 = match fam.fill {
|
||||
MaskFill::FiveBytes if col % c == 0 => 182 - k as u64,
|
||||
MaskFill::FiveBytes => 0,
|
||||
MaskFill::Alternating if k.is_multiple_of(2) => noise(),
|
||||
MaskFill::Alternating | MaskFill::Compressible => 7,
|
||||
};
|
||||
out.extend_from_slice(&v.to_le_bytes()[..fam.elem]);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(feature = "lzf")]
|
||||
fn mask_families() -> Vec<MaskFamily> {
|
||||
#[cfg_attr(not(feature = "blosc"), allow(unused_mut))]
|
||||
let mut v = vec![
|
||||
MaskFamily {
|
||||
name: "lzf5",
|
||||
elem: 1,
|
||||
chunk: 5,
|
||||
h5py_kw: "dict(compression='lzf')",
|
||||
build: |d| {
|
||||
d.with_lzf().without_shuffle();
|
||||
},
|
||||
fill: MaskFill::FiveBytes,
|
||||
},
|
||||
MaskFamily {
|
||||
name: "lzf",
|
||||
elem: 4,
|
||||
chunk: 64,
|
||||
h5py_kw: "dict(compression='lzf')",
|
||||
build: |d| {
|
||||
d.with_lzf().without_shuffle();
|
||||
},
|
||||
fill: MaskFill::Alternating,
|
||||
},
|
||||
MaskFamily {
|
||||
name: "mix",
|
||||
elem: 4,
|
||||
chunk: 8,
|
||||
h5py_kw: "dict(compression='lzf', shuffle=True, fletcher32=True)",
|
||||
build: |d| {
|
||||
d.with_lzf().with_shuffle().with_fletcher32();
|
||||
},
|
||||
fill: MaskFill::Alternating,
|
||||
},
|
||||
MaskFamily {
|
||||
name: "lzfc",
|
||||
elem: 4,
|
||||
chunk: 64,
|
||||
h5py_kw: "dict(compression='lzf')",
|
||||
build: |d| {
|
||||
d.with_lzf().without_shuffle();
|
||||
},
|
||||
fill: MaskFill::Compressible,
|
||||
},
|
||||
];
|
||||
#[cfg(feature = "blosc")]
|
||||
{
|
||||
use clawhdf5_format::chunked_write::{BloscCodec, BloscShuffle};
|
||||
v.push(MaskFamily {
|
||||
name: "blosc",
|
||||
elem: 4,
|
||||
chunk: 64,
|
||||
h5py_kw: "hdf5plugin.Blosc(cname='lz4', clevel=5, shuffle=hdf5plugin.Blosc.SHUFFLE)",
|
||||
build: |d| {
|
||||
d.with_blosc(BloscCodec::Lz4, 5, BloscShuffle::Byte);
|
||||
},
|
||||
fill: MaskFill::Alternating,
|
||||
});
|
||||
v.push(MaskFamily {
|
||||
name: "blosc0",
|
||||
elem: 4,
|
||||
chunk: 64,
|
||||
h5py_kw: "hdf5plugin.Blosc(cname='lz4', clevel=0, shuffle=hdf5plugin.Blosc.SHUFFLE)",
|
||||
build: |d| {
|
||||
d.with_blosc(BloscCodec::Lz4, 0, BloscShuffle::Byte);
|
||||
},
|
||||
fill: MaskFill::Compressible,
|
||||
});
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
/// Builds h5py twins of our datasets and prints, per dataset, the filter
|
||||
/// masks by chunk offset in our file and in the twin.
|
||||
const MASK_TWIN: &str = r#"
|
||||
import sys, numpy as np, h5py
|
||||
try:
|
||||
import hdf5plugin
|
||||
except ImportError:
|
||||
hdf5plugin = None
|
||||
ours, twin, spec = sys.argv[1], sys.argv[2], eval(sys.argv[3])
|
||||
def masks(ds):
|
||||
return sorted((tuple(ds.id.get_chunk_info(k).chunk_offset), ds.id.get_chunk_info(k).filter_mask)
|
||||
for k in range(ds.id.get_num_chunks()))
|
||||
with h5py.File(ours, 'r') as o, h5py.File(twin, 'w', libver='v114') as t:
|
||||
for name, dt, shape, chunks, maxshape, kw, raw in spec:
|
||||
want = np.fromfile(raw, dtype=dt).reshape(shape)
|
||||
assert np.array_equal(o[name][()], want), name
|
||||
t.create_dataset(name, data=want, chunks=chunks, maxshape=maxshape, **eval(kw))
|
||||
print(name, masks(o[name]), '|', masks(t[name]))
|
||||
"#;
|
||||
|
||||
/// h5py (libhdf5) rewrites every chunk of our datasets — random chunks
|
||||
/// become compressible and the other way round; the `[x,0,0,0,0]` chunks
|
||||
/// change in place at the same size — then extends the resizable ones with
|
||||
/// random data, and saves what each dataset must now hold. Prints the
|
||||
/// datasets h5dump can decode (no chunk left LZF-encoded: h5dump has no
|
||||
/// LZF filter).
|
||||
const MASK_REWRITE: &str = r#"
|
||||
import sys, numpy as np, h5py
|
||||
try:
|
||||
import hdf5plugin
|
||||
except ImportError:
|
||||
hdf5plugin = None
|
||||
ours, spec = sys.argv[1], eval(sys.argv[2])
|
||||
rng = np.random.default_rng(3)
|
||||
def noise(shape, dt):
|
||||
return rng.integers(0, 256, int(np.prod(shape)) * np.dtype(dt).itemsize,
|
||||
dtype=np.uint8).view(dt).reshape(shape)
|
||||
dumpable = []
|
||||
with h5py.File(ours, 'r+') as f:
|
||||
for name, dt, shape, chunks, maxshape, kw, raw in spec:
|
||||
d = f[name]
|
||||
want = d[()]
|
||||
for s in d.iter_chunks():
|
||||
blk = want[s]
|
||||
if dt == 'u1':
|
||||
blk.flat[-1] = 1
|
||||
elif (blk == blk.flat[0]).all():
|
||||
blk[...] = noise(blk.shape, dt)
|
||||
else:
|
||||
blk[...] = 5
|
||||
d[...] = want
|
||||
if maxshape is not None:
|
||||
new = tuple(n + c for n, c in zip(shape, chunks))
|
||||
grown = noise(new, dt)
|
||||
grown[tuple(slice(0, n) for n in shape)] = want
|
||||
d.resize(new)
|
||||
d[...] = grown
|
||||
want = grown
|
||||
want.tofile(raw + '.want')
|
||||
pl = d.id.get_create_plist()
|
||||
ids = [pl.get_filter(i)[0] for i in range(pl.get_nfilters())]
|
||||
if 32000 in ids:
|
||||
bit = 1 << ids.index(32000)
|
||||
if not all(d.id.get_chunk_info(k).filter_mask & bit for k in range(d.id.get_num_chunks())):
|
||||
continue
|
||||
dumpable.append(name)
|
||||
with h5py.File(ours, 'r') as f:
|
||||
for name, dt, shape, chunks, maxshape, kw, raw in spec:
|
||||
want = np.fromfile(raw + '.want', dtype=dt).reshape(f[name].shape)
|
||||
assert np.array_equal(f[name][()], want), name
|
||||
print(' '.join(dumpable))
|
||||
"#;
|
||||
|
||||
/// Optional filters that fail are skipped in files `FileBuilder` writes,
|
||||
/// exactly as libhdf5 skips them: an LZF or Blosc output no smaller than the
|
||||
/// chunk leaves the chunk stored unfiltered with the filter's mask bit set.
|
||||
/// The writer used to store every chunk filtered with mask 0. For LZF, a
|
||||
/// chunk whose LZF stream is exactly the chunk's size (`[x,0,0,0,0]`) was
|
||||
/// then corrupted by the first libhdf5 rewrite of it: libhdf5 stores the
|
||||
/// new data raw at the same size and, the size being unchanged, leaves the
|
||||
/// stale mask in the index, so h5py could no longer read the dataset.
|
||||
///
|
||||
/// For every family × every chunk index the writer builds (single chunk,
|
||||
/// Fixed Array, Extensible Array, version-2 B-tree): our masks equal those
|
||||
/// of an h5py-written twin of the same data; then h5py r+ rewrites and
|
||||
/// extends the datasets, and h5py, h5dump (where it has the filter) and our
|
||||
/// reader read every value.
|
||||
#[cfg(feature = "lzf")]
|
||||
#[test]
|
||||
fn skipped_optional_filters_are_masked_as_libhdf5_masks_them() {
|
||||
let modules = if cfg!(feature = "blosc") {
|
||||
"h5py, numpy, hdf5plugin"
|
||||
} else {
|
||||
"h5py, numpy"
|
||||
};
|
||||
if !have_python(modules) {
|
||||
return;
|
||||
}
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let ours = dir.path().join("ours.h5");
|
||||
let twin = dir.path().join("twin.h5");
|
||||
let mut fb = clawhdf5::FileBuilder::new();
|
||||
let mut spec = Vec::new();
|
||||
let mut names = Vec::new();
|
||||
for (fi, fam) in mask_families().iter().enumerate() {
|
||||
for (label, shape, chunks, maxshape) in mask_layouts(fam.chunk) {
|
||||
let name = format!("{}_{label}", fam.name);
|
||||
let data = mask_data(fam, &shape, fi as u64 * 31 + shape.len() as u64);
|
||||
let raw = dir.path().join(format!("{name}.raw"));
|
||||
std::fs::write(&raw, &data).unwrap();
|
||||
let ds = fb.create_dataset(&name);
|
||||
if fam.elem == 1 {
|
||||
ds.with_u8_data(&data);
|
||||
} else {
|
||||
let v: Vec<i32> = data
|
||||
.as_chunks::<4>()
|
||||
.0
|
||||
.iter()
|
||||
.map(|&b| i32::from_le_bytes(b))
|
||||
.collect();
|
||||
ds.with_i32_data(&v);
|
||||
}
|
||||
ds.with_shape(&shape).with_chunks(&chunks);
|
||||
if let Some(ms) = &maxshape {
|
||||
ds.with_maxshape(ms);
|
||||
}
|
||||
(fam.build)(ds);
|
||||
let py_tuple = |v: &[u64]| {
|
||||
let items: Vec<String> = v
|
||||
.iter()
|
||||
.map(|&d| {
|
||||
if d == u64::MAX {
|
||||
"None".into()
|
||||
} else {
|
||||
d.to_string()
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
format!("({},)", items.join(","))
|
||||
};
|
||||
spec.push(format!(
|
||||
"({name:?}, {:?}, {}, {}, {}, {:?}, {:?})",
|
||||
if fam.elem == 1 { "u1" } else { "<i4" },
|
||||
py_tuple(&shape),
|
||||
py_tuple(&chunks),
|
||||
maxshape.as_deref().map_or("None".into(), py_tuple),
|
||||
fam.h5py_kw,
|
||||
raw.to_str().unwrap(),
|
||||
));
|
||||
names.push((name, fam.elem));
|
||||
}
|
||||
}
|
||||
fb.write(&ours).unwrap();
|
||||
let spec = format!("[{}]", spec.join(", "));
|
||||
|
||||
// Our masks are h5py's, chunk by chunk.
|
||||
let out = run_python(
|
||||
MASK_TWIN,
|
||||
&[ours.to_str().unwrap(), twin.to_str().unwrap(), &spec],
|
||||
);
|
||||
let mut skipped = 0;
|
||||
for line in out.lines() {
|
||||
let (name, rest) = line.split_once(' ').unwrap();
|
||||
let (got, want) = rest.split_once(" | ").unwrap();
|
||||
assert_eq!(got, want, "{name}: our filter masks (left) vs h5py's");
|
||||
skipped += usize::from(got.contains("), 1)") || got.contains("), 2)"));
|
||||
}
|
||||
assert_eq!(out.lines().count(), names.len());
|
||||
assert!(
|
||||
skipped >= 12,
|
||||
"too few datasets with skipped filters:\n{out}"
|
||||
);
|
||||
|
||||
// libhdf5 rewrites and extends them; everyone reads the new values.
|
||||
let dumpable = run_python(MASK_REWRITE, &[ours.to_str().unwrap(), &spec]);
|
||||
let plugin_path = run_python("import hdf5plugin; print(hdf5plugin.PLUGIN_PATH)", &[]);
|
||||
let file = File::open(&ours).unwrap();
|
||||
for (name, elem) in &names {
|
||||
let want = std::fs::read(dir.path().join(format!("{name}.raw.want"))).unwrap();
|
||||
let got = file
|
||||
.dataset(name)
|
||||
.unwrap()
|
||||
.read_selection(&Selection::All)
|
||||
.unwrap();
|
||||
assert!(got == want, "{name}: our reader after h5py r+");
|
||||
if !dumpable.split(' ').any(|d| d == name) || Command::new("h5dump").output().is_err() {
|
||||
continue;
|
||||
}
|
||||
let o = Command::new("h5dump")
|
||||
.env("HDF5_PLUGIN_PATH", &plugin_path)
|
||||
.args(["-d", name, "-y", "-w", "0", ours.to_str().unwrap()])
|
||||
.output()
|
||||
.unwrap();
|
||||
assert!(
|
||||
o.status.success(),
|
||||
"h5dump -d {name}: {}",
|
||||
String::from_utf8_lossy(&o.stderr)
|
||||
);
|
||||
let s = String::from_utf8_lossy(&o.stdout).into_owned();
|
||||
let vals: Vec<i64> = s
|
||||
.split_once("DATA {")
|
||||
.and_then(|(_, r)| r.split_once('}'))
|
||||
.map(|(d, _)| {
|
||||
d.split(|c: char| c == ',' || c.is_whitespace())
|
||||
.filter(|t| !t.is_empty())
|
||||
.map(|t| t.parse::<i64>().unwrap())
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default();
|
||||
let want_vals: Vec<i64> = want
|
||||
.chunks_exact(*elem)
|
||||
.map(|b| {
|
||||
if *elem == 1 {
|
||||
i64::from(b[0])
|
||||
} else {
|
||||
i64::from(i32::from_le_bytes(b.try_into().unwrap()))
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
assert_eq!(vals, want_vals, "h5dump -d {name}");
|
||||
}
|
||||
assert!(
|
||||
dumpable.split(' ').any(|d| d.starts_with("lzf5_")),
|
||||
"{dumpable}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Files whose chunks all compress are written exactly as before optional
|
||||
/// filters could be skipped: every mask is 0 and nothing else changed. The
|
||||
/// hashes are of the files the writer produced before that change.
|
||||
#[cfg(feature = "lzf")]
|
||||
#[test]
|
||||
fn files_whose_chunks_all_compress_are_unchanged() {
|
||||
use clawhdf5_format::checksum::jenkins_lookup3;
|
||||
#[allow(clippy::type_complexity)]
|
||||
#[cfg_attr(not(feature = "blosc"), allow(unused_mut))]
|
||||
let mut cases: Vec<(
|
||||
&str,
|
||||
fn(&mut clawhdf5_format::type_builders::DatasetBuilder),
|
||||
(usize, u32),
|
||||
)> = vec![
|
||||
(
|
||||
"lzf_fixed",
|
||||
|d| {
|
||||
d.with_i32_data(&ramp_i32(4000))
|
||||
.with_chunks(&[500])
|
||||
.with_lzf();
|
||||
},
|
||||
(3965, 449169442),
|
||||
),
|
||||
(
|
||||
"lzf_ea_noshuffle",
|
||||
|d| {
|
||||
d.with_i32_data(&ramp_i32(4000))
|
||||
.with_chunks(&[700])
|
||||
.with_maxshape(&[u64::MAX])
|
||||
.with_lzf()
|
||||
.without_shuffle();
|
||||
},
|
||||
(7213, 4277403206),
|
||||
),
|
||||
(
|
||||
"mix_bt2",
|
||||
|d| {
|
||||
d.with_f64_data(&ramp_f64(40 * 60))
|
||||
.with_shape(&[40, 60])
|
||||
.with_chunks(&[16, 16])
|
||||
.with_maxshape(&[u64::MAX, u64::MAX])
|
||||
.with_lzf()
|
||||
.with_fletcher32();
|
||||
},
|
||||
(11495, 3340532700),
|
||||
),
|
||||
(
|
||||
"lzf_single",
|
||||
|d| {
|
||||
d.with_u8_data(&ramp_u8(3000))
|
||||
.with_chunks(&[3000])
|
||||
.with_lzf();
|
||||
},
|
||||
(546, 690805477),
|
||||
),
|
||||
];
|
||||
#[cfg(feature = "blosc")]
|
||||
cases.push((
|
||||
"blosc_fixed",
|
||||
|d| {
|
||||
use clawhdf5_format::chunked_write::{BloscCodec, BloscShuffle};
|
||||
d.with_i32_data(&ramp_i32(5000))
|
||||
.with_chunks(&[1024])
|
||||
.with_blosc(BloscCodec::Lz4, 5, BloscShuffle::Byte);
|
||||
},
|
||||
(2776, 4278611376),
|
||||
));
|
||||
for (name, build, want) in &cases {
|
||||
let mut fb = clawhdf5::FileBuilder::new();
|
||||
build(fb.create_dataset("d"));
|
||||
let bytes = fb.finish().unwrap();
|
||||
assert_eq!(
|
||||
(bytes.len(), jenkins_lookup3(&bytes)),
|
||||
*want,
|
||||
"{name}: (length, lookup3 hash) of the file"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user