Range reads M0/M1 (indexed lookups, Storage trait), ZFP, in-place editing #17
+9
-1
@@ -19,7 +19,15 @@
|
|||||||
index, its data blocks, super blocks and pages, and splitting B-tree
|
index, its data blocks, super blocks and pages, and splitting B-tree
|
||||||
nodes, as libhdf5 does: after the same sequence of writes the B-tree has
|
nodes, as libhdf5 does: after the same sequence of writes the B-tree has
|
||||||
the same number of nodes per level and the Extensible Array header the
|
the same number of nodes per level and the Extensible Array header the
|
||||||
same block statistics as libhdf5's (tested).
|
same block statistics as libhdf5's (tested). Filters run as libhdf5's
|
||||||
|
`H5Z_pipeline` runs them (new
|
||||||
|
`clawhdf5_format::filters::compress_chunk_masked`): an optional filter
|
||||||
|
that fails — LZF or Blosc output no smaller than the chunk — is skipped
|
||||||
|
and its filter-mask bit set, so the chunk is stored exactly as h5py
|
||||||
|
stores it; a mandatory filter that fails fails the edit. (Storing such
|
||||||
|
a chunk LZF-encoded at the raw size with a clear mask let a later
|
||||||
|
libhdf5 rewrite of it keep the stale mask, and h5py could no longer
|
||||||
|
read the dataset.)
|
||||||
- `resize`: grow a chunked dataset up to its maximum dimensions (h5py's
|
- `resize`: grow a chunked dataset up to its maximum dimensions (h5py's
|
||||||
`Dataset.resize`).
|
`Dataset.resize`).
|
||||||
- `set_attr`: add or replace an attribute in an object header, in free
|
- `set_attr`: add or replace an attribute in an object header, in free
|
||||||
|
|||||||
@@ -346,6 +346,72 @@ pub fn compress_chunk(
|
|||||||
Ok(result)
|
Ok(result)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Filter flag bit 0: `H5Z_FLAG_OPTIONAL`.
|
||||||
|
const FILTER_FLAG_OPTIONAL: u16 = 0x0001;
|
||||||
|
|
||||||
|
/// Filters whose reference HDF5 filter (h5py's `lzf_filter.c`,
|
||||||
|
/// hdf5-blosc's `blosc_filter.c`) gives the encoder an output buffer only as
|
||||||
|
/// large as its input, so output that is not smaller than the input is a
|
||||||
|
/// failure there.
|
||||||
|
const FAIL_UNLESS_SMALLER: &[u16] = &[
|
||||||
|
crate::filter_pipeline::FILTER_LZF,
|
||||||
|
crate::filter_pipeline::FILTER_BLOSC,
|
||||||
|
];
|
||||||
|
|
||||||
|
/// Run a chunk through a filter pipeline for writing the way libhdf5's
|
||||||
|
/// `H5Z_pipeline` does, returning the bytes to store and the chunk's filter
|
||||||
|
/// mask (bit `i` set: filter `i` was skipped).
|
||||||
|
///
|
||||||
|
/// A filter that fails is skipped if the pipeline marks it optional
|
||||||
|
/// (`H5Z_FLAG_OPTIONAL`): its mask bit is set and the next filter gets the
|
||||||
|
/// same input. A mandatory filter that fails fails the write. Failure
|
||||||
|
/// includes what the reference filter counts as failure: LZF and Blosc
|
||||||
|
/// output that is not smaller than the input (h5py then stores the chunk
|
||||||
|
/// unfiltered with the bit set; storing it filtered with a clear mask can
|
||||||
|
/// leave a stale mask once libhdf5 rewrites the chunk at the same size).
|
||||||
|
///
|
||||||
|
/// A filter this build cannot encode is [`FormatError::UnsupportedFilter`]
|
||||||
|
/// even when optional: libhdf5 skips an optional filter only when its own
|
||||||
|
/// build lacks it, and every libhdf5 has the ones clawhdf5 cannot encode.
|
||||||
|
pub fn compress_chunk_masked(
|
||||||
|
data: &[u8],
|
||||||
|
pipeline: &FilterPipeline,
|
||||||
|
element_size: u32,
|
||||||
|
) -> Result<(Vec<u8>, u32), FormatError> {
|
||||||
|
if pipeline.filters.len() > 32 {
|
||||||
|
return Err(FormatError::CompressionError(
|
||||||
|
"more than 32 filters in a pipeline".into(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let mut result = data.to_vec();
|
||||||
|
let mut mask = 0u32;
|
||||||
|
for (i, filter) in pipeline.filters.iter().enumerate() {
|
||||||
|
let ctx = FilterContext {
|
||||||
|
filter,
|
||||||
|
element_size: element_size as usize,
|
||||||
|
max_output: 0,
|
||||||
|
};
|
||||||
|
let out = match filter_registry::encode(&result, &ctx) {
|
||||||
|
Ok(out)
|
||||||
|
if FAIL_UNLESS_SMALLER.contains(&filter.filter_id) && out.len() >= result.len() =>
|
||||||
|
{
|
||||||
|
Err(FormatError::CompressionError(format!(
|
||||||
|
"filter {} did not shrink the chunk",
|
||||||
|
filter.filter_id
|
||||||
|
)))
|
||||||
|
}
|
||||||
|
r => r,
|
||||||
|
};
|
||||||
|
match out {
|
||||||
|
Ok(out) => result = out,
|
||||||
|
Err(e @ FormatError::UnsupportedFilter(_)) => return Err(e),
|
||||||
|
Err(_) if filter.flags & FILTER_FLAG_OPTIONAL != 0 => mask |= 1 << i,
|
||||||
|
Err(e) => return Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok((result, mask))
|
||||||
|
}
|
||||||
|
|
||||||
/// The filters compiled into this build, sorted by ID (see
|
/// The filters compiled into this build, sorted by ID (see
|
||||||
/// [`crate::filter_registry`]). A filter whose cargo feature is off is left
|
/// [`crate::filter_registry`]). A filter whose cargo feature is off is left
|
||||||
/// out, so it fails as [`FormatError::UnsupportedFilter`] like any unknown ID.
|
/// out, so it fails as [`FormatError::UnsupportedFilter`] like any unknown ID.
|
||||||
@@ -2099,6 +2165,60 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `compress_chunk_masked` follows `H5Z_pipeline`: an optional LZF that
|
||||||
|
/// does not shrink the chunk is skipped with its mask bit set (h5py
|
||||||
|
/// stores `[182, 0, 0, 0, 0]` raw with mask 1), a mandatory one fails,
|
||||||
|
/// and filters that grow the data (deflate) are kept, as libhdf5 keeps
|
||||||
|
/// them.
|
||||||
|
#[test]
|
||||||
|
#[cfg(all(feature = "lzf", feature = "deflate"))]
|
||||||
|
fn masked_compression_skips_optional_filters_that_fail() {
|
||||||
|
use crate::filter_pipeline::FILTER_LZF;
|
||||||
|
let opt = |id: u16| FilterDescription {
|
||||||
|
flags: FILTER_FLAG_OPTIONAL,
|
||||||
|
..filter(id)
|
||||||
|
};
|
||||||
|
let pl = |filters: Vec<FilterDescription>| FilterPipeline {
|
||||||
|
version: 2,
|
||||||
|
filters,
|
||||||
|
};
|
||||||
|
let raw = [182u8, 0, 0, 0, 0];
|
||||||
|
let (out, mask) = compress_chunk_masked(&raw, &pl(vec![opt(FILTER_LZF)]), 1).unwrap();
|
||||||
|
assert_eq!((out.as_slice(), mask), (&raw[..], 1));
|
||||||
|
let (out, mask) = compress_chunk_masked(
|
||||||
|
&raw,
|
||||||
|
&pl(vec![
|
||||||
|
opt(FILTER_SHUFFLE),
|
||||||
|
opt(FILTER_LZF),
|
||||||
|
filter(FILTER_FLETCHER32),
|
||||||
|
]),
|
||||||
|
1,
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
assert_eq!((out.len(), mask), (raw.len() + 4, 2));
|
||||||
|
assert_eq!(
|
||||||
|
decompress_chunk_masked(
|
||||||
|
&out,
|
||||||
|
&pl(vec![
|
||||||
|
opt(FILTER_SHUFFLE),
|
||||||
|
opt(FILTER_LZF),
|
||||||
|
filter(FILTER_FLETCHER32)
|
||||||
|
]),
|
||||||
|
raw.len(),
|
||||||
|
1,
|
||||||
|
mask
|
||||||
|
)
|
||||||
|
.unwrap(),
|
||||||
|
raw
|
||||||
|
);
|
||||||
|
assert!(compress_chunk_masked(&raw, &pl(vec![filter(FILTER_LZF)]), 1).is_err());
|
||||||
|
let zeros = [0u8; 256];
|
||||||
|
let (out, mask) = compress_chunk_masked(&zeros, &pl(vec![opt(FILTER_LZF)]), 1).unwrap();
|
||||||
|
assert!(out.len() < zeros.len() && mask == 0);
|
||||||
|
let (out, mask) = compress_chunk_masked(&raw, &pl(vec![opt(FILTER_DEFLATE)]), 1).unwrap();
|
||||||
|
assert!(out.len() > raw.len() && mask == 0);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
#[cfg(feature = "deflate")]
|
#[cfg(feature = "deflate")]
|
||||||
fn filter_mask_skips_only_the_masked_filters() {
|
fn filter_mask_skips_only_the_masked_filters() {
|
||||||
|
|||||||
@@ -1187,3 +1187,131 @@ fn measure_append_waste() {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Optional filters that fail are skipped as libhdf5 skips them: an LZF
|
||||||
|
/// that does not shrink a chunk leaves it stored unfiltered with its mask
|
||||||
|
/// bit set, exactly as h5py stores it. Storing the LZF stream with a clear
|
||||||
|
/// mask at the raw chunk's size once let a later libhdf5 rewrite of the
|
||||||
|
/// chunk (raw, same size) keep the stale mask, and h5py then failed to read
|
||||||
|
/// the dataset. After the editor, h5py r+ rewrites and extends the datasets
|
||||||
|
/// and h5py, h5dump (on the chunks it can decode: LZF is not in h5dump) and
|
||||||
|
/// our reader read every value.
|
||||||
|
#[test]
|
||||||
|
fn optional_filters_that_fail_are_skipped() {
|
||||||
|
if !tools_ok() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
for (li, (lv, _)) in LIBVERS.iter().enumerate() {
|
||||||
|
let dir = tmpdir();
|
||||||
|
let path = dir.path().join(format!("optional_{li}.h5"));
|
||||||
|
let p = path.to_str().unwrap();
|
||||||
|
py(&format!(
|
||||||
|
"import h5py, numpy as np\n\
|
||||||
|
with h5py.File({p:?}, 'w', libver={lv}) as f:\n\
|
||||||
|
\x20 f.create_dataset('u8', shape=(5,), dtype='u1', chunks=(5,), maxshape=(None,), compression='lzf')\n\
|
||||||
|
\x20 f.create_dataset('mix', shape=(32,), dtype='<i4', chunks=(8,), maxshape=(None,), compression='lzf', shuffle=True, fletcher32=True)\n\
|
||||||
|
\x20 f.create_dataset('ref', shape=(5,), dtype='u1', chunks=(5,), maxshape=(None,), compression='lzf')\n\
|
||||||
|
\x20 f['ref'][...] = [182, 0, 0, 0, 0]\n"
|
||||||
|
));
|
||||||
|
let mut rng = Rng(li as u64 + 17);
|
||||||
|
// Chunks 0 and 2 incompressible, 1 and 3 compressible.
|
||||||
|
let mix: Vec<i32> = (0..32)
|
||||||
|
.map(|i| {
|
||||||
|
if (i / 8) % 2 == 0 {
|
||||||
|
rng.next() as i32
|
||||||
|
} else {
|
||||||
|
7
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
let mut ed = FileEditor::open(&path).unwrap();
|
||||||
|
ed.write_all("u8", &[182, 0, 0, 0, 0]).unwrap();
|
||||||
|
ed.write_values("mix", &Selection::All, &mix).unwrap();
|
||||||
|
drop(ed);
|
||||||
|
// Stored as h5py stores it: raw, LZF's bit (0; 1 behind shuffle) set.
|
||||||
|
let masks = py(&format!(
|
||||||
|
"import h5py\n\
|
||||||
|
f = h5py.File({p:?}, 'r')\n\
|
||||||
|
i = lambda d, k: f[d].id.get_chunk_info(k)\n\
|
||||||
|
print(i('u8', 0).filter_mask, i('u8', 0).size, i('ref', 0).filter_mask, i('ref', 0).size,\n\
|
||||||
|
\x20 *[(i('mix', k).filter_mask, i('mix', k).size < 36) for k in range(4)])\n"
|
||||||
|
));
|
||||||
|
assert_eq!(
|
||||||
|
masks, "1 5 1 5 (2, False) (0, True) (2, False) (0, True)",
|
||||||
|
"{lv}"
|
||||||
|
);
|
||||||
|
let dump = |ds: &str| {
|
||||||
|
let o = Command::new("h5dump")
|
||||||
|
.args(["-d", ds, "-y", "-w", "0", p])
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(o.status.success(), "h5dump -d {ds} {p}:\n{}", text(&o));
|
||||||
|
let s = String::from_utf8_lossy(&o.stdout).into_owned();
|
||||||
|
s.split_once("DATA {")
|
||||||
|
.and_then(|(_, r)| r.split_once('}'))
|
||||||
|
.map(|(d, _)| {
|
||||||
|
d.split(|c: char| c == ',' || c.is_whitespace())
|
||||||
|
.filter(|t| !t.is_empty())
|
||||||
|
.map(|t| t.parse::<i64>().unwrap())
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
})
|
||||||
|
.unwrap_or_default()
|
||||||
|
};
|
||||||
|
if *lv != "'latest'" {
|
||||||
|
assert_eq!(dump("/u8"), [182, 0, 0, 0, 0]);
|
||||||
|
}
|
||||||
|
// libhdf5 rewrites the chunk raw at the same size, then extends the
|
||||||
|
// datasets with more incompressible data.
|
||||||
|
py(&format!(
|
||||||
|
"import h5py, numpy as np\n\
|
||||||
|
with h5py.File({p:?}, 'r+') as f:\n\
|
||||||
|
\x20 f['u8'][...] = [182, 0, 0, 0, 1]\n\
|
||||||
|
\x20 f['u8'].resize((12,))\n\
|
||||||
|
\x20 f['u8'][5:] = [9, 200, 3, 77, 1, 250, 42]\n\
|
||||||
|
\x20 m = f['mix']\n\
|
||||||
|
\x20 m[8:16] = np.arange(8, dtype='<i4') * 104729 + 12345\n\
|
||||||
|
\x20 m[0:8] = 5\n\
|
||||||
|
\x20 m.resize((40,))\n\
|
||||||
|
\x20 m[32:] = 11\n"
|
||||||
|
));
|
||||||
|
let u8_want: Vec<u8> = vec![182, 0, 0, 0, 1, 9, 200, 3, 77, 1, 250, 42];
|
||||||
|
let mut mix_want = mix.clone();
|
||||||
|
mix_want[0..8].fill(5);
|
||||||
|
for (k, v) in mix_want[8..16].iter_mut().enumerate() {
|
||||||
|
*v = k as i32 * 104_729 + 12_345;
|
||||||
|
}
|
||||||
|
mix_want.extend([11; 8]);
|
||||||
|
let f = File::open(&path).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
f.dataset("u8")
|
||||||
|
.unwrap()
|
||||||
|
.read_selection(&Selection::All)
|
||||||
|
.unwrap(),
|
||||||
|
u8_want
|
||||||
|
);
|
||||||
|
assert_eq!(f.dataset("mix").unwrap().read_i32().unwrap(), mix_want);
|
||||||
|
drop(f);
|
||||||
|
verify(
|
||||||
|
&path,
|
||||||
|
"mix",
|
||||||
|
&Model {
|
||||||
|
shape: vec![40],
|
||||||
|
data: mix_want,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
py(&format!(
|
||||||
|
"import h5py\n\
|
||||||
|
f = h5py.File({p:?}, 'r')\n\
|
||||||
|
assert f['u8'][()].tolist() == {u8_want:?}, f['u8'][()]\n"
|
||||||
|
));
|
||||||
|
if *lv != "'latest'" {
|
||||||
|
let got: Vec<u8> = dump("/u8").into_iter().map(|v| v as u8).collect();
|
||||||
|
assert_eq!(got, u8_want);
|
||||||
|
}
|
||||||
|
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
|
||||||
|
.args(["check", "--data", "-q", p])
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -1116,11 +1116,10 @@ fn write_selection(
|
|||||||
Ok(())
|
Ok(())
|
||||||
})?;
|
})?;
|
||||||
for (scaled, buf) in bufs {
|
for (scaled, buf) in bufs {
|
||||||
|
// As libhdf5 does: an optional filter that fails (LZF that
|
||||||
|
// does not shrink the chunk) is skipped and its mask bit set.
|
||||||
let (bytes, mask) = match &t.pipeline {
|
let (bytes, mask) = match &t.pipeline {
|
||||||
Some(p) => (
|
Some(p) => clawhdf5_format::filters::compress_chunk_masked(&buf, p, es as u32)?,
|
||||||
clawhdf5_format::filters::compress_chunk(&buf, p, es as u32)?,
|
|
||||||
0u32,
|
|
||||||
),
|
|
||||||
None => (buf, 0u32),
|
None => (buf, 0u32),
|
||||||
};
|
};
|
||||||
let len = bytes.len() as u64;
|
let len = bytes.len() as u64;
|
||||||
|
|||||||
Reference in New Issue
Block a user