//! Shared by the integration tests: the test server, a transcript of a //! file (tree, attributes, values) to compare two ways of reading it, and //! the test files. #![allow(dead_code)] pub mod server; use std::collections::{BTreeMap, HashSet, VecDeque}; use std::fmt::Write as _; use std::path::{Path, PathBuf}; use std::process::Command; use std::sync::Arc; use clawhdf5::{DType, File, Selection}; use clawhdf5_format::error::FormatError; /// Objects visited per file. const MAX_OBJECTS: usize = 2000; /// Datasets with more bytes than this are not read (their metadata is). pub const MAX_DATA_BYTES: u64 = 64 << 20; /// A short, stable digest of a value's `Debug` form. fn digest(v: &T) -> String { let s = format!("{v:?}"); if s.len() <= 200 { return s; } let mut h = 0xcbf2_9ce4_8422_2325u64; for b in s.bytes() { h = (h ^ u64::from(b)).wrapping_mul(0x100_0000_01b3); } format!("{}…[{} bytes, fnv {h:016x}]", &s[..80], s.len()) } /// A data read's value, or `Err` (which chunk a damaged dataset reports /// can vary between two `File`s: the chunk cache lists in hash order). fn value(r: &Result) -> String { match r { Ok(v) => digest(v), Err(_) => "Err".into(), } } fn sorted(m: std::collections::HashMap) -> BTreeMap { m.into_iter().collect() } /// Everything a reader sees in `file`: every group's entries, every /// object's attributes, and every dataset's shape, types and values. pub fn transcript(file: &File) -> String { let mut out = String::new(); let mut seen = HashSet::new(); let mut queue = VecDeque::from([(String::from("/"), file.superblock().root_group_address)]); while let Some((path, addr)) = queue.pop_front() { if seen.len() >= MAX_OBJECTS || !seen.insert(addr) { continue; } let group = file.group_at(addr); let entries = group.entries(); writeln!(out, "{path} @{addr} entries {}", digest(&entries)).unwrap(); if let Ok(ds) = file.dataset_at(addr) { dataset(&mut out, &path, &ds); } let attrs = group.attrs_with_errors().map(|(a, e)| (sorted(a), e)); writeln!(out, "{path} attrs {}", digest(&attrs)).unwrap(); if let Ok(entries) = entries { for (name, child) in entries { queue.push_back((format!("{}/{name}", path.trim_end_matches('/')), child)); } } } out } fn dataset(out: &mut String, path: &str, ds: &clawhdf5::Dataset<'_>) { let shape = ds.shape(); let dtype = ds.dtype(); writeln!( out, "{path} shape {} dtype {} raw {}", digest(&shape), digest(&dtype), digest(&ds.raw_datatype()) ) .unwrap(); let (Ok(shape), Ok(dtype), Ok(raw_dt)) = (shape, dtype, ds.raw_datatype()) else { return; }; let elements = shape.iter().try_fold(1u64, |a, &d| a.checked_mul(d)); let bytes = elements.and_then(|n| n.checked_mul(u64::from(raw_dt.type_size()))); if bytes.is_none_or(|b| b > MAX_DATA_BYTES) { writeln!(out, "{path} too large to read").unwrap(); return; } writeln!( out, "{path} all {}", value(&ds.read_selection(&Selection::All)) ) .unwrap(); if matches!( dtype, DType::F32 | DType::F64 | DType::I8 | DType::I16 | DType::I32 | DType::I64 | DType::U8 | DType::U16 | DType::U32 | DType::U64 ) { writeln!(out, "{path} f64 {}", value(&ds.read_f64())).unwrap(); if let Some(&d0) = shape.first() { let rank = shape.len(); let sel = Selection::Hyperslab { start: std::iter::once(d0 / 3) .chain(std::iter::repeat_n(0, rank - 1)) .collect(), stride: vec![1; rank], count: std::iter::once(d0.div_ceil(3)) .chain(shape[1..].iter().copied()) .collect(), block: vec![1; rank], }; writeln!( out, "{path} f64 third {}", value(&ds.read_f64_selection(&sel)) ) .unwrap(); } } match &raw_dt { clawhdf5_format::datatype::Datatype::String { .. } | clawhdf5_format::datatype::Datatype::VariableLength { is_string: true, .. } => { writeln!(out, "{path} strings {}", value(&ds.read_string_bytes())).unwrap(); } clawhdf5_format::datatype::Datatype::VariableLength { .. } => { writeln!(out, "{path} vlen {}", value(&ds.read_vlen::())).unwrap(); } _ => {} } } /// List the file as a tree view does — every group's entries, every /// dataset's shape and type — and return the largest dataset whose data /// is at most `MAX_DATA_BYTES` (address, bytes), the one a viewer would /// plot. pub fn list(file: &File) -> Option<(u64, u64)> { let mut seen = HashSet::new(); let mut largest: Option<(u64, u64)> = None; let mut queue = VecDeque::from([file.superblock().root_group_address]); while let Some(addr) = queue.pop_front() { if seen.len() >= MAX_OBJECTS || !seen.insert(addr) { continue; } let group = file.group_at(addr); if let Ok(ds) = file.dataset_at(addr) { let _ = (ds.shape(), ds.dtype()); let bytes = ds.shape().ok().and_then(|s| { let n = s.iter().try_fold(1u64, |a, &d| a.checked_mul(d))?; let size = u64::from(ds.raw_datatype().ok()?.type_size()); n.checked_mul(size).filter(|&b| b <= MAX_DATA_BYTES) }); if let Some(b) = bytes && largest.is_none_or(|(_, l)| b > l) { largest = Some((addr, b)); } } if let Ok(entries) = group.entries() { queue.extend(entries.into_iter().map(|(_, a)| a)); } } largest } /// Read the dataset at `addr` whole. pub fn read_one(file: &File, addr: u64) { if let Ok(ds) = file.dataset_at(addr) { let _ = ds.read_selection(&Selection::All); } } /// [`list`], then [`read_one`] of the dataset it picks. pub fn list_and_read_one(file: &File) { if let Some((addr, _)) = list(file) { read_one(file, addr); } } /// External virtual-dataset sources read from `dir`, as `File::open` finds /// them next to the file. pub fn sibling_resolver(dir: PathBuf) -> clawhdf5::VdsResolver { Arc::new(move |name: &str| { let p = Path::new(name); if name.is_empty() || !p .components() .all(|c| matches!(c, std::path::Component::Normal(_))) { return Err(FormatError::Storage(format!("{name:?} not followed"))); } match std::fs::read(dir.join(p)) { Ok(bytes) => Ok(Some(bytes)), Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(None), Err(e) => Err(FormatError::Storage(e.to_string())), } }) } /// HDF5 files under `dir`, recursively. pub fn hdf5_files(dir: &Path, out: &mut Vec) { let Ok(entries) = std::fs::read_dir(dir) else { return; }; for e in entries.flatten() { let p = e.path(); if p.is_dir() { hdf5_files(&p, out); } else if p .extension() .and_then(|x| x.to_str()) .is_some_and(|x| matches!(x, "h5" | "hdf5" | "he5" | "nc" | "h5ad" | "hdf")) { out.push(p); } } } /// The repository's HDF5 test fixtures. pub fn fixtures() -> Vec { let root = Path::new(env!("CARGO_MANIFEST_DIR")); let mut files = Vec::new(); hdf5_files(&root.join("../clawhdf5/tests/fixtures"), &mut files); hdf5_files(&root.join("../clawhdf5-format/tests/fixtures"), &mut files); files.sort(); files } /// Files of `CLAWHDF5_REMOTE_CORPUS` (directories separated like `PATH`). pub fn corpus() -> Option> { let dirs = std::env::var("CLAWHDF5_REMOTE_CORPUS").ok()?; let mut files = Vec::new(); for d in std::env::split_paths(&dirs) { hdf5_files(&d, &mut files); } files.sort(); Some(files) } pub fn python() -> String { std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string()) } pub fn interop_required() -> bool { std::env::var("CLAWHDF5_REQUIRE_INTEROP").is_ok_and(|v| v == "1") } /// Whether python3 with h5py and numpy runs; panics when interop is /// required and it does not. pub fn have_h5py() -> bool { let ok = Command::new(python()) .args(["-c", "import h5py, numpy"]) .output() .map(|o| o.status.success()) .unwrap_or(false); assert!( ok || !interop_required(), "CLAWHDF5_REQUIRE_INTEROP=1 but python3 with h5py is not available" ); if !ok { eprintln!("SKIP: python3 with h5py not available"); } ok } /// Run a Python script; its stdout. pub fn run_python(script: &str, args: &[&str]) -> String { let out = Command::new(python()) .arg("-c") .arg(script) .args(args) .output() .expect("failed to run python"); assert!( out.status.success(), "python failed:\n{}", String::from_utf8_lossy(&out.stderr) ); String::from_utf8(out.stdout).unwrap() } /// A clawhdf5-written file with a multi-block chunked dataset (`/big`, /// 1 000 000 f64 in chunks of 10 000, deflated), a contiguous one and a /// group — several blocks of 1 MiB, with no Python needed. pub fn multi_block_file() -> Vec { let mut b = clawhdf5::FileBuilder::new(); b.set_attr("title", clawhdf5::AttrValue::String("remote test".into())); let big: Vec = (0..1_000_000u64) .map(|i| ((i * 2_654_435_761) % 1_000_003) as f64 * 0.5) .collect(); b.create_dataset("big") .with_f64_data(&big) .with_chunks(&[10_000]) .with_deflate(1); let flat: Vec = (0..300_000u64).map(|i| i as f64).collect(); b.create_dataset("flat").with_f64_data(&flat); let mut g = b.create_group("grp"); g.create_dataset("small").with_f64_data(&[1.0, 2.0, 3.0]); b.add_group(g.finish()); b.finish().unwrap() }