edit: record the maximum before resizing a chunked dataset that has none

FileEditor::resize (on main since PR #18, a4c2ace) scrambled the values of
a chunked dataset whose dataspace records no maximum dimensions when it
shrank it: clawhdf5's writer stores such a dataspace for every chunked
dataset created without a maxshape, the maximum is then the current
dimensions, and the Fixed Array index linearises chunks by the maximum, so
patching only the current dimensions moved every chunk after the first
row. h5py, h5dump and our reader all read the wrong values; the dataset
could not grow back either.

libhdf5 never writes such a dataspace (H5S_set_extent_simple records the
maximum, equal to the dimensions when none is given); reading one,
H5S_extent_get_dims reports the current dimensions as the maximum and
H5S_set_extent checks against none, so its own H5Dset_extent scrambles
such a file the same way. The editor now records the maximum libhdf5 would
have written (the dimensions the index was built with) before changing the
current ones, moving the grown dataspace message in the header when it
must. The writer records the maximum of every chunked dataset too, so
h5py can resize what clawhdf5 writes (the pinned file hashes of three
no-maxshape cases in plugin_filters_interop change by 8 bytes a dimension).

Tests: edit_resize_interop.rs (a 2.7.0-written fixture, new FileBuilder
files and h5py files through shrinks, zero extents and growth, against a
model with our reader and h5py; h5py resizing a FileBuilder file), and in
test_edit.py resizes checked against a numpy model, independently of h5py.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-27 07:40:01 -05:00
co-authored by Claude Opus 5.5
parent 92fb0830e0
commit 1f7651644b
8 changed files with 482 additions and 7 deletions
+76
View File
@@ -427,13 +427,89 @@ def test_random_edits_match_h5py(h5py, tmp_path, source, seed):
key = _random_key(rng, shape)
sel_shape, fancy = _selection_shape(key, shape)
op = ("set", name, key, _random_value(rng, sel_shape, fancy, dtype))
before = None
if op[0] == "resize":
with h5py.File(ours_path, "r", locking=False) as o:
before, fill = o[op[1]][()], o[op[1]].fillvalue
if edit_both(h5py, theirs, ours_path, ours, op) is not None:
refused += 1
elif before is not None:
# Independently of h5py (which a wrong index layout fools
# the same way): the kept elements keep their values.
want = resized_model(before, op[2], fill)
np.testing.assert_array_equal(ours[op[1]][()], want, err_msg=repr(op))
with h5py.File(ours_path, "r", locking=False) as o:
np.testing.assert_array_equal(o[op[1]][()], want, err_msg=repr(op))
assert_ours_reads_like_h5py(h5py, ours, ours_path, "end")
assert refused < 30
h5dump_reads(ours_path, base)
def resized_model(before, shape, fill):
"""`before` resized to `shape` as HDF5 resizes: elements inside both
extents keep their values, the others read as the fill value."""
out = np.full(shape, fill, dtype=before.dtype)
common = tuple(slice(0, min(a, b)) for a, b in zip(before.shape, shape))
out[common] = before[common]
return out
RESIZES = [(15, 15), (3, 2), (20, 20), (1, 1), (1, 0), (0, 0), (7, 20), (20, 13), (20, 20)]
def _check_resizes(h5py, path, name, orig, maxshape=(20, 20), base=None):
"""Resize `name` through RESIZES in 'r+', each checked against a numpy
model with clawhdf5 and h5py and the file with `h5rs check`."""
model = orig
with clawhdf5.File(path, "r+") as f:
ds = f[name]
for shape in RESIZES:
ds.resize(shape)
model = resized_model(model, shape, 0)
np.testing.assert_array_equal(ds[()], model, err_msg=f"{name} {shape}")
with h5py.File(path, "r", locking=False) as t:
np.testing.assert_array_equal(t[name][()], model, err_msg=f"h5py: {name} {shape}")
assert t[name].maxshape == maxshape
h5rs = os.environ.get("CLAWHDF5_H5RS")
if h5rs: # h5dump would wait for the editor's lock
r = subprocess.run([h5rs, "check", "--data", path], capture_output=True, text=True)
assert r.returncode == 0, f"{name} {shape}: " + (r.stdout + r.stderr)[-2000:]
with pytest.raises(ValueError):
ds.resize((maxshape[0] + 1, 20))
# Written values survive a shrink.
ds[...] = orig
ds.resize((15, 15))
with h5py.File(path, "r") as t:
np.testing.assert_array_equal(t[name][()], orig[:15, :15])
h5dump_reads(path, base)
@pytest.mark.parametrize("source", SOURCES)
def test_resizes_keep_values(h5py, tmp_path, source):
"""Shrinking, zero extents and growing back keep the values a numpy
model keeps, on every source (on clawhdf5's own files a shrink once
moved every chunk: h5py read the same wrong values)."""
_, ours, base = _make(h5py, tmp_path, source)
with clawhdf5.File(ours, "r+") as f:
f["chunk_gzip"].resize((20, 20))
maxshape = (20, 20) if source == "clawhdf5" else (40, 40)
_check_resizes(h5py, ours, "chunk_gzip", np.arange(400, dtype="<f4").reshape(20, 20), maxshape, base)
FIXTURES = os.path.join(os.path.dirname(__file__), "..", "..", "clawhdf5", "tests", "fixtures")
@pytest.mark.parametrize("name", ["d", "z"])
def test_resizes_of_a_file_with_no_recorded_maxshape(h5py, tmp_path, name):
"""A file clawhdf5 2.7.0 wrote records no maximum dimensions for chunked
datasets; resizing it must keep the Fixed Array index's layout."""
path = str(tmp_path / "old.h5")
shutil.copy(os.path.join(FIXTURES, "chunked_no_maxshape_v2_7_0.h5"), path)
with h5py.File(path, "r") as t:
orig = t[name][()]
_check_resizes(h5py, path, name, orig)
NUMERIC = ["<i1", "<u1", "<i2", ">u2", "<i4", "<u4", ">i8", "<u8", "<f2", "<f4", ">f8"]