Range reads M0/M1 (indexed lookups, Storage trait), ZFP, in-place editing #17

Merged
osobh merged 52 commits from feat/p3-range-zfp-edit into main 2026-09-26 20:42:22 +00:00
4 changed files with 260 additions and 5 deletions
Showing only changes of commit f7e2ab12f2 - Show all commits
+9 -1
View File
@@ -19,7 +19,15 @@
index, its data blocks, super blocks and pages, and splitting B-tree index, its data blocks, super blocks and pages, and splitting B-tree
nodes, as libhdf5 does: after the same sequence of writes the B-tree has nodes, as libhdf5 does: after the same sequence of writes the B-tree has
the same number of nodes per level and the Extensible Array header the the same number of nodes per level and the Extensible Array header the
same block statistics as libhdf5's (tested). same block statistics as libhdf5's (tested). Filters run as libhdf5's
`H5Z_pipeline` runs them (new
`clawhdf5_format::filters::compress_chunk_masked`): an optional filter
that fails — LZF or Blosc output no smaller than the chunk — is skipped
and its filter-mask bit set, so the chunk is stored exactly as h5py
stores it; a mandatory filter that fails fails the edit. (Storing such
a chunk LZF-encoded at the raw size with a clear mask let a later
libhdf5 rewrite of it keep the stale mask, and h5py could no longer
read the dataset.)
- `resize`: grow a chunked dataset up to its maximum dimensions (h5py's - `resize`: grow a chunked dataset up to its maximum dimensions (h5py's
`Dataset.resize`). `Dataset.resize`).
- `set_attr`: add or replace an attribute in an object header, in free - `set_attr`: add or replace an attribute in an object header, in free
+120
View File
@@ -346,6 +346,72 @@ pub fn compress_chunk(
Ok(result) Ok(result)
} }
/// Filter flag bit 0: `H5Z_FLAG_OPTIONAL`.
const FILTER_FLAG_OPTIONAL: u16 = 0x0001;
/// Filters whose reference HDF5 filter (h5py's `lzf_filter.c`,
/// hdf5-blosc's `blosc_filter.c`) gives the encoder an output buffer only as
/// large as its input, so output that is not smaller than the input is a
/// failure there.
const FAIL_UNLESS_SMALLER: &[u16] = &[
crate::filter_pipeline::FILTER_LZF,
crate::filter_pipeline::FILTER_BLOSC,
];
/// Run a chunk through a filter pipeline for writing the way libhdf5's
/// `H5Z_pipeline` does, returning the bytes to store and the chunk's filter
/// mask (bit `i` set: filter `i` was skipped).
///
/// A filter that fails is skipped if the pipeline marks it optional
/// (`H5Z_FLAG_OPTIONAL`): its mask bit is set and the next filter gets the
/// same input. A mandatory filter that fails fails the write. Failure
/// includes what the reference filter counts as failure: LZF and Blosc
/// output that is not smaller than the input (h5py then stores the chunk
/// unfiltered with the bit set; storing it filtered with a clear mask can
/// leave a stale mask once libhdf5 rewrites the chunk at the same size).
///
/// A filter this build cannot encode is [`FormatError::UnsupportedFilter`]
/// even when optional: libhdf5 skips an optional filter only when its own
/// build lacks it, and every libhdf5 has the ones clawhdf5 cannot encode.
pub fn compress_chunk_masked(
data: &[u8],
pipeline: &FilterPipeline,
element_size: u32,
) -> Result<(Vec<u8>, u32), FormatError> {
if pipeline.filters.len() > 32 {
return Err(FormatError::CompressionError(
"more than 32 filters in a pipeline".into(),
));
}
let mut result = data.to_vec();
let mut mask = 0u32;
for (i, filter) in pipeline.filters.iter().enumerate() {
let ctx = FilterContext {
filter,
element_size: element_size as usize,
max_output: 0,
};
let out = match filter_registry::encode(&result, &ctx) {
Ok(out)
if FAIL_UNLESS_SMALLER.contains(&filter.filter_id) && out.len() >= result.len() =>
{
Err(FormatError::CompressionError(format!(
"filter {} did not shrink the chunk",
filter.filter_id
)))
}
r => r,
};
match out {
Ok(out) => result = out,
Err(e @ FormatError::UnsupportedFilter(_)) => return Err(e),
Err(_) if filter.flags & FILTER_FLAG_OPTIONAL != 0 => mask |= 1 << i,
Err(e) => return Err(e),
}
}
Ok((result, mask))
}
/// The filters compiled into this build, sorted by ID (see /// The filters compiled into this build, sorted by ID (see
/// [`crate::filter_registry`]). A filter whose cargo feature is off is left /// [`crate::filter_registry`]). A filter whose cargo feature is off is left
/// out, so it fails as [`FormatError::UnsupportedFilter`] like any unknown ID. /// out, so it fails as [`FormatError::UnsupportedFilter`] like any unknown ID.
@@ -2099,6 +2165,60 @@ mod tests {
} }
} }
/// `compress_chunk_masked` follows `H5Z_pipeline`: an optional LZF that
/// does not shrink the chunk is skipped with its mask bit set (h5py
/// stores `[182, 0, 0, 0, 0]` raw with mask 1), a mandatory one fails,
/// and filters that grow the data (deflate) are kept, as libhdf5 keeps
/// them.
#[test]
#[cfg(all(feature = "lzf", feature = "deflate"))]
fn masked_compression_skips_optional_filters_that_fail() {
use crate::filter_pipeline::FILTER_LZF;
let opt = |id: u16| FilterDescription {
flags: FILTER_FLAG_OPTIONAL,
..filter(id)
};
let pl = |filters: Vec<FilterDescription>| FilterPipeline {
version: 2,
filters,
};
let raw = [182u8, 0, 0, 0, 0];
let (out, mask) = compress_chunk_masked(&raw, &pl(vec![opt(FILTER_LZF)]), 1).unwrap();
assert_eq!((out.as_slice(), mask), (&raw[..], 1));
let (out, mask) = compress_chunk_masked(
&raw,
&pl(vec![
opt(FILTER_SHUFFLE),
opt(FILTER_LZF),
filter(FILTER_FLETCHER32),
]),
1,
)
.unwrap();
assert_eq!((out.len(), mask), (raw.len() + 4, 2));
assert_eq!(
decompress_chunk_masked(
&out,
&pl(vec![
opt(FILTER_SHUFFLE),
opt(FILTER_LZF),
filter(FILTER_FLETCHER32)
]),
raw.len(),
1,
mask
)
.unwrap(),
raw
);
assert!(compress_chunk_masked(&raw, &pl(vec![filter(FILTER_LZF)]), 1).is_err());
let zeros = [0u8; 256];
let (out, mask) = compress_chunk_masked(&zeros, &pl(vec![opt(FILTER_LZF)]), 1).unwrap();
assert!(out.len() < zeros.len() && mask == 0);
let (out, mask) = compress_chunk_masked(&raw, &pl(vec![opt(FILTER_DEFLATE)]), 1).unwrap();
assert!(out.len() > raw.len() && mask == 0);
}
#[test] #[test]
#[cfg(feature = "deflate")] #[cfg(feature = "deflate")]
fn filter_mask_skips_only_the_masked_filters() { fn filter_mask_skips_only_the_masked_filters() {
+128
View File
@@ -1187,3 +1187,131 @@ fn measure_append_waste() {
); );
} }
} }
/// Optional filters that fail are skipped as libhdf5 skips them: an LZF
/// that does not shrink a chunk leaves it stored unfiltered with its mask
/// bit set, exactly as h5py stores it. Storing the LZF stream with a clear
/// mask at the raw chunk's size once let a later libhdf5 rewrite of the
/// chunk (raw, same size) keep the stale mask, and h5py then failed to read
/// the dataset. After the editor, h5py r+ rewrites and extends the datasets
/// and h5py, h5dump (on the chunks it can decode: LZF is not in h5dump) and
/// our reader read every value.
#[test]
fn optional_filters_that_fail_are_skipped() {
if !tools_ok() {
return;
}
for (li, (lv, _)) in LIBVERS.iter().enumerate() {
let dir = tmpdir();
let path = dir.path().join(format!("optional_{li}.h5"));
let p = path.to_str().unwrap();
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'w', libver={lv}) as f:\n\
\x20 f.create_dataset('u8', shape=(5,), dtype='u1', chunks=(5,), maxshape=(None,), compression='lzf')\n\
\x20 f.create_dataset('mix', shape=(32,), dtype='<i4', chunks=(8,), maxshape=(None,), compression='lzf', shuffle=True, fletcher32=True)\n\
\x20 f.create_dataset('ref', shape=(5,), dtype='u1', chunks=(5,), maxshape=(None,), compression='lzf')\n\
\x20 f['ref'][...] = [182, 0, 0, 0, 0]\n"
));
let mut rng = Rng(li as u64 + 17);
// Chunks 0 and 2 incompressible, 1 and 3 compressible.
let mix: Vec<i32> = (0..32)
.map(|i| {
if (i / 8) % 2 == 0 {
rng.next() as i32
} else {
7
}
})
.collect();
let mut ed = FileEditor::open(&path).unwrap();
ed.write_all("u8", &[182, 0, 0, 0, 0]).unwrap();
ed.write_values("mix", &Selection::All, &mix).unwrap();
drop(ed);
// Stored as h5py stores it: raw, LZF's bit (0; 1 behind shuffle) set.
let masks = py(&format!(
"import h5py\n\
f = h5py.File({p:?}, 'r')\n\
i = lambda d, k: f[d].id.get_chunk_info(k)\n\
print(i('u8', 0).filter_mask, i('u8', 0).size, i('ref', 0).filter_mask, i('ref', 0).size,\n\
\x20 *[(i('mix', k).filter_mask, i('mix', k).size < 36) for k in range(4)])\n"
));
assert_eq!(
masks, "1 5 1 5 (2, False) (0, True) (2, False) (0, True)",
"{lv}"
);
let dump = |ds: &str| {
let o = Command::new("h5dump")
.args(["-d", ds, "-y", "-w", "0", p])
.output()
.unwrap();
assert!(o.status.success(), "h5dump -d {ds} {p}:\n{}", text(&o));
let s = String::from_utf8_lossy(&o.stdout).into_owned();
s.split_once("DATA {")
.and_then(|(_, r)| r.split_once('}'))
.map(|(d, _)| {
d.split(|c: char| c == ',' || c.is_whitespace())
.filter(|t| !t.is_empty())
.map(|t| t.parse::<i64>().unwrap())
.collect::<Vec<_>>()
})
.unwrap_or_default()
};
if *lv != "'latest'" {
assert_eq!(dump("/u8"), [182, 0, 0, 0, 0]);
}
// libhdf5 rewrites the chunk raw at the same size, then extends the
// datasets with more incompressible data.
py(&format!(
"import h5py, numpy as np\n\
with h5py.File({p:?}, 'r+') as f:\n\
\x20 f['u8'][...] = [182, 0, 0, 0, 1]\n\
\x20 f['u8'].resize((12,))\n\
\x20 f['u8'][5:] = [9, 200, 3, 77, 1, 250, 42]\n\
\x20 m = f['mix']\n\
\x20 m[8:16] = np.arange(8, dtype='<i4') * 104729 + 12345\n\
\x20 m[0:8] = 5\n\
\x20 m.resize((40,))\n\
\x20 m[32:] = 11\n"
));
let u8_want: Vec<u8> = vec![182, 0, 0, 0, 1, 9, 200, 3, 77, 1, 250, 42];
let mut mix_want = mix.clone();
mix_want[0..8].fill(5);
for (k, v) in mix_want[8..16].iter_mut().enumerate() {
*v = k as i32 * 104_729 + 12_345;
}
mix_want.extend([11; 8]);
let f = File::open(&path).unwrap();
assert_eq!(
f.dataset("u8")
.unwrap()
.read_selection(&Selection::All)
.unwrap(),
u8_want
);
assert_eq!(f.dataset("mix").unwrap().read_i32().unwrap(), mix_want);
drop(f);
verify(
&path,
"mix",
&Model {
shape: vec![40],
data: mix_want,
},
);
py(&format!(
"import h5py\n\
f = h5py.File({p:?}, 'r')\n\
assert f['u8'][()].tolist() == {u8_want:?}, f['u8'][()]\n"
));
if *lv != "'latest'" {
let got: Vec<u8> = dump("/u8").into_iter().map(|v| v as u8).collect();
assert_eq!(got, u8_want);
}
let o = Command::new(env!("CARGO_BIN_EXE_h5rs"))
.args(["check", "--data", "-q", p])
.output()
.unwrap();
assert!(o.status.success(), "h5rs check --data {p}:\n{}", text(&o));
}
}
+3 -4
View File
@@ -1116,11 +1116,10 @@ fn write_selection(
Ok(()) Ok(())
})?; })?;
for (scaled, buf) in bufs { for (scaled, buf) in bufs {
// As libhdf5 does: an optional filter that fails (LZF that
// does not shrink the chunk) is skipped and its mask bit set.
let (bytes, mask) = match &t.pipeline { let (bytes, mask) = match &t.pipeline {
Some(p) => ( Some(p) => clawhdf5_format::filters::compress_chunk_masked(&buf, p, es as u32)?,
clawhdf5_format::filters::compress_chunk(&buf, p, es as u32)?,
0u32,
),
None => (buf, 0u32), None => (buf, 0u32),
}; };
let len = bytes.len() as u64; let len = bytes.len() as u64;