Lazy remote files in the browser (M4), SWMR reader (M5), Python remote reads and editing #19

Merged
osobh merged 33 commits from feat/p3-wasm-swmr-python into main 2026-09-27 14:47:26 +00:00
2 changed files with 94 additions and 7 deletions
Showing only changes of commit 4ad80736f4 - Show all commits
+27 -7
View File
@@ -262,6 +262,11 @@ struct CachedChunk {
struct DatasetEntry {
/// Chunk coordinate -> ChunkInfo (offset + size in file).
index: Option<Arc<HashMap<ChunkCoord, ChunkInfo>>>,
/// The same chunks in the order the chunk index lists them: what
/// [`ChunkCache::chunks_for`] returns, so a cached read walks (and, on a
/// damaged file, fails at) the chunks in the same order as an uncached
/// one, rather than in hash-map order.
ordered: Option<Arc<Vec<ChunkInfo>>>,
/// Pre-built chunk index for O(1) coordinate lookups.
chunk_index: Option<Arc<ChunkIndex>>,
/// Pre-computed chunk layout for fast assembly.
@@ -274,6 +279,7 @@ struct DatasetEntry {
impl DatasetEntry {
fn weight(&self) -> usize {
self.index.as_ref().map_or(0, |m| m.len())
+ self.ordered.as_ref().map_or(0, |o| o.len())
+ self.chunk_index.as_ref().map_or(0, |c| c.num_chunks())
}
}
@@ -562,11 +568,22 @@ impl ChunkCache {
rank: usize,
build: impl FnOnce() -> Result<Vec<ChunkInfo>, E>,
) -> Result<Vec<ChunkInfo>, E> {
Ok(self
.index_for(addr, rank, build)?
.values()
.cloned()
.collect())
if let Some(ordered) = self.lock().touch(addr).ordered.clone() {
return Ok(ordered.as_ref().clone());
}
let chunks = build()?;
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
.iter()
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci.clone()))
.collect();
let mut inner = self.lock();
let entry = inner.touch(addr);
// Another thread may have built this dataset's index meanwhile: keep
// the first one, so every reader sees the same order.
let ordered = Arc::clone(entry.ordered.get_or_insert_with(|| Arc::new(chunks)));
entry.index.get_or_insert_with(|| Arc::new(map));
inner.trim_datasets(addr);
Ok(ordered.as_ref().clone())
}
fn index_for<E>(
@@ -580,11 +597,14 @@ impl ChunkCache {
}
let chunks = build()?;
let map: HashMap<ChunkCoord, ChunkInfo> = chunks
.into_iter()
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci))
.iter()
.map(|ci| (ci.offsets.iter().take(rank).copied().collect(), ci.clone()))
.collect();
let mut inner = self.lock();
let entry = inner.touch(addr);
if entry.index.is_none() {
entry.ordered = Some(Arc::new(chunks));
}
let index = Arc::clone(entry.index.get_or_insert_with(|| Arc::new(map)));
inner.trim_datasets(addr);
Ok(index)
@@ -455,3 +455,70 @@ with h5py.File("{p}", "w") as f:
assert!(lazy.dataset("line").unwrap().read_i32().is_err());
assert!(lazy.dataset("grid").unwrap().read_f64().is_err());
}
// ---------------------------------------------------------------------------
// Several damaged chunks: which one is reported
// ---------------------------------------------------------------------------
/// With more than one damaged chunk, the error names the first damaged chunk
/// in the chunk index's order, every time and on every read path. The cached
/// reader used to walk the chunks in hash-map order, so two opens of the same
/// file could report different chunks (`cve-2025-2310.h5`).
#[test]
fn several_damaged_chunks_report_the_same_chunk_every_time() {
skip_if_no_python!();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("damaged.h5");
let p = path.display().to_string();
run_python(&format!(
r#"
import h5py, numpy as np, zlib
with h5py.File("{p}", "w") as f:
ds = f.create_dataset("d", shape=(512,), chunks=(8,), dtype="<f8", compression="gzip")
ds[...] = np.arange(512.0)
# Each damaged chunk inflates to a different short length, so each
# error names its own chunk.
for k, off in enumerate((40, 136, 320, 488)):
ds.id.write_direct_chunk((off,), zlib.compress(bytes(8 * (k + 1))))
"#
));
let first = File::open(&path)
.unwrap()
.dataset("d")
.unwrap()
.read_f64()
.expect_err("damaged chunks must fail")
.to_string();
// A whole-dataset selection goes through the file's chunk cache, whose index is built
// afresh (a new hash map) for every open.
let first_raw = File::open(&path)
.unwrap()
.dataset("d")
.unwrap()
.read_selection(&Selection::All)
.expect_err("damaged chunks must fail")
.to_string();
for _ in 0..40 {
let f = File::open(&path).unwrap();
let ds = f.dataset("d").unwrap();
assert_eq!(ds.read_f64().unwrap_err().to_string(), first);
assert_eq!(
ds.read_selection(&Selection::All).unwrap_err().to_string(),
first_raw
);
// A second read goes through the cached chunk index.
assert_eq!(
ds.read_selection(&Selection::All).unwrap_err().to_string(),
first_raw
);
}
let lazy = clawhdf5::LazyFile::open_mmap(&path).unwrap();
assert_eq!(
lazy.dataset("d")
.unwrap()
.read_f64()
.unwrap_err()
.to_string(),
first
);
}