clawhdf5-netcdf4: phony dimensions, skipped types and order as netCDF-C
CI / test-arm64 (pull_request) Successful in 1m43s
CI / test (pull_request) Successful in 20m28s

Read a file's metadata the way netCDF-C 4.9.3 does (libhdf5/hdf5open.c),
for the whole file on first use (src/model.rs, replacing src/scope.rs):

- links in creation order when the group tracks it, else name order;
  a group's datasets before its subgroups; dimension ids file-wide;
- variables' dimensions from _Netcdf4Coordinates (file-wide ids), else
  the scales DIMENSION_LIST attaches when the first axis has one, else
  netCDF-C's phony dimensions phony_dim_<id> (create_phony_dims: shared
  by length and unlimitedness within a group, not between two axes of
  one variable, numbered subgroups first, a zero length unlimited);
- datasets of types netCDF-C cannot represent are not variables
  (references, bit fields, time, arrays, compounds/enums/VLENs over
  them), replaying netCDF-C's file-wide type list, failed types
  included;
- unlimited lengths as nc4_find_dim_len (its group and below).

NcType gains Enum, Compound, VLen, Opaque and is #[non_exhaustive];
Variable::nc_type is netCDF-C's type (1-byte strings NC_CHAR). New
clawhdf5_format::group_v2::links_in_creation_order_in.

Tests compare with netCDF-C itself (tests/netcdf_c_view.py calls the
libnetcdf netCDF4-python bundles through ctypes): new interop cases for
h5py files without dimension scales, every type class, link order; and
the gated corpus_vs_netcdf_c (CLAWHDF5_NETCDF_CORPUS): 420 of the 429
conformance-corpus files netCDF-C opens match (main: 68); the other 9
are explained in tests/corpus_known_differences.txt and known-issues.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-29 20:38:51 -05:00
co-authored by Claude Opus 5.5
parent 4260af4f70
commit e5d6f59e12
18 changed files with 2263 additions and 524 deletions
+281
View File
@@ -0,0 +1,281 @@
//! clawhdf5-netcdf4 compared with netCDF-C itself: `tests/netcdf_c_view.py`
//! prints what netCDF-C (the libnetcdf netCDF4-python bundles) reports for
//! a file, and [`differences`] lists how clawhdf5-netcdf4's reading differs.
#![allow(dead_code)]
use std::collections::{BTreeMap, BTreeSet};
use std::path::{Path, PathBuf};
use std::process::Command;
use std::sync::atomic::{AtomicUsize, Ordering};
use clawhdf5_netcdf4::{NcType, NetCDF4File, NetCDF4Group, Variable};
/// The Python with netCDF4-python (`CLAWHDF5_PYTHON`, else `python3`).
pub fn python() -> String {
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
}
/// netCDF-C's view of one file.
#[derive(Default)]
pub struct View {
/// `G`, `D` and `V` lines, in order.
pub meta: Vec<String>,
/// `(group, variable)` → values, from the `X` lines.
pub values: BTreeMap<(String, String), Vec<f64>>,
}
/// netCDF-C's view of each file (`None`: netCDF-C cannot open it), from
/// `tests/netcdf_c_view.py`.
pub fn netcdf_c_views(files: &[PathBuf]) -> BTreeMap<PathBuf, Option<View>> {
let dir = tempfile::tempdir().unwrap();
let list = dir.path().join("files.txt");
let mut paths = String::new();
for f in files {
paths.push_str(&f.to_string_lossy());
paths.push('\n');
}
std::fs::write(&list, paths).unwrap();
let script = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/netcdf_c_view.py");
let out = Command::new(python())
.arg(&script)
.arg("--list")
.arg(&list)
.output()
.expect("failed to run python");
assert!(
out.status.success(),
"netcdf_c_view.py failed: {}",
String::from_utf8_lossy(&out.stderr)
);
parse_views(&String::from_utf8_lossy(&out.stdout))
}
/// The views in the output of `tests/netcdf_c_view.py`.
pub fn parse_views(text: &str) -> BTreeMap<PathBuf, Option<View>> {
let mut views = BTreeMap::new();
let mut current: Option<(PathBuf, Option<View>)> = None;
for line in text.lines() {
let fields: Vec<&str> = line.split('\t').collect();
match fields[0] {
"FILE" => current = Some((PathBuf::from(fields[1]), Some(View::default()))),
"END" => {
let (path, view) = current.take().expect("END without FILE");
views.insert(path, view);
}
"ERROR" => current.as_mut().expect("ERROR without FILE").1 = None,
"X" => {
let view = current.as_mut().and_then(|c| c.1.as_mut()).unwrap();
let vals = if fields[3] == "-" {
Vec::new()
} else {
fields[3]
.split(' ')
.map(|v| v.parse().expect("value"))
.collect()
};
view.values
.insert((fields[1].to_string(), fields[2].to_string()), vals);
}
_ => {
let view = current.as_mut().and_then(|c| c.1.as_mut()).unwrap();
view.meta.push(line.to_string());
}
}
}
views
}
/// Variables whose type is labelled as netCDF-C labels it rather than as
/// clawhdf5-netcdf4 does (see [`nc_type_label`]).
pub static RELABELLED: AtomicUsize = AtomicUsize::new(0);
/// Values not compared because this build of the HDF5 reader lacks a
/// filter (a cargo feature).
pub static FILTER_SKIPPED: AtomicUsize = AtomicUsize::new(0);
/// The same `G`/`D`/`V` lines from clawhdf5-netcdf4.
pub fn our_meta(file: &NetCDF4File) -> Result<Vec<String>, String> {
let e = |e: clawhdf5_netcdf4::Error| e.to_string();
let mut out = Vec::new();
out.push("G\t/".to_string());
push_group(
&mut out,
file.hdf5_file(),
"/",
file.dimensions().map_err(e)?,
file.variables().map_err(e)?,
)?;
for name in file.group_names().map_err(e)? {
walk(
&mut out,
file.hdf5_file(),
&format!("/{name}"),
&file.group(&name).map_err(e)?,
)?;
}
Ok(out)
}
fn walk(
out: &mut Vec<String>,
hdf5: &clawhdf5::File,
path: &str,
group: &NetCDF4Group<'_>,
) -> Result<(), String> {
let e = |e: clawhdf5_netcdf4::Error| e.to_string();
out.push(format!("G\t{path}"));
push_group(
out,
hdf5,
path,
group.dimensions().map_err(e)?,
group.variables().map_err(e)?,
)?;
for name in group.group_names().map_err(e)? {
walk(
out,
hdf5,
&format!("{path}/{name}"),
&group.group(&name).map_err(e)?,
)?;
}
Ok(())
}
/// The type of variable `name` of the group at `path` as the `V` line
/// shows it: its `NcType`, except for the one deliberate difference from
/// netCDF-C — a float of other than 4 or 8 bytes (half, bfloat16, the 4-,
/// 6- and 8-bit floats, `long double`), `NC_FLOAT`/`NC_DOUBLE` here, is
/// labelled `NC_STRING` by netCDF-C 4.9.3 on libhdf5 1.14.6; such a
/// variable is shown as netCDF-C shows it, and counted.
fn nc_type_label(hdf5: &clawhdf5::File, path: &str, var: &Variable<'_>) -> Result<String, String> {
let nc_type = var.nc_type().map_err(|e| e.to_string())?;
if matches!(nc_type, NcType::Float | NcType::Double) {
let dataset = format!("{}/{}", path.trim_end_matches('/'), var.name());
if let Ok(ds) = hdf5.dataset(&dataset)
&& let Ok(clawhdf5_format::datatype::Datatype::FloatingPoint { size, .. }) =
ds.raw_datatype()
&& size != 4
&& size != 8
{
RELABELLED.fetch_add(1, Ordering::Relaxed);
return Ok("NC_STRING".to_string());
}
}
Ok(nc_type.to_string())
}
fn push_group(
out: &mut Vec<String>,
hdf5: &clawhdf5::File,
path: &str,
dims: Vec<clawhdf5_netcdf4::Dimension>,
vars: Vec<Variable<'_>>,
) -> Result<(), String> {
for d in dims {
out.push(format!(
"D\t{path}\t{}\t{}\t{}",
d.name,
d.size,
u8::from(d.is_unlimited)
));
}
for v in vars {
let dims: Vec<&str> = v.dimensions().iter().map(|d| d.name.as_str()).collect();
let shape: Vec<String> = v
.shape()
.map_err(|e| e.to_string())?
.iter()
.map(u64::to_string)
.collect();
let or_dash = |s: String| if s.is_empty() { "-".to_string() } else { s };
out.push(format!(
"V\t{path}\t{}\t{}\t{}\t{}",
v.name(),
nc_type_label(hdf5, path, &v)?,
or_dash(dims.join(",")),
or_dash(shape.join(","))
));
}
Ok(())
}
/// The values of variable `name` of group `group`.
fn our_values(file: &NetCDF4File, group: &str, name: &str) -> Result<Vec<f64>, String> {
let var = if group == "/" {
file.variable(name)
} else {
file.variable(&format!("{}/{name}", group.trim_start_matches('/')))
}
.map_err(|e| e.to_string())?;
match var.nc_type().map_err(|e| e.to_string())? {
NcType::String | NcType::Char => Err("not numeric".into()),
_ => var.read_raw_f64().map_err(|e| e.to_string()),
}
}
/// How clawhdf5-netcdf4's reading of `path` differs from `want`; empty
/// when it does not.
pub fn differences(path: &Path, want: &View) -> Vec<String> {
let file = match NetCDF4File::open(path) {
Ok(f) => f,
Err(e) => return vec![format!("netCDF-C opens it, clawhdf5-netcdf4 does not: {e}")],
};
let meta = match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| our_meta(&file))) {
Ok(Ok(meta)) => meta,
Ok(Err(e)) => return vec![format!("error: {e}")],
Err(_) => return vec!["panic".to_string()],
};
let mut diffs = Vec::new();
if meta != want.meta {
let ours: BTreeSet<&String> = meta.iter().collect();
let theirs: BTreeSet<&String> = want.meta.iter().collect();
for line in want.meta.iter().filter(|l| !ours.contains(l)) {
diffs.push(format!("netCDF-C: {line}"));
}
for line in meta.iter().filter(|l| !theirs.contains(l)) {
diffs.push(format!("ours: {line}"));
}
if diffs.is_empty() {
diffs.push("same lines, different order".to_string());
}
}
for ((group, name), want) in &want.values {
let got = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
our_values(&file, group, name)
}))
.unwrap_or_else(|_| Err("panic".to_string()));
match got {
Ok(got) => {
let same = got.len() == want.len()
&& got
.iter()
.zip(want)
.all(|(a, b)| a.to_bits() == b.to_bits() || (a.is_nan() && b.is_nan()));
if !same {
diffs.push(format!("values of {group} {name} differ"));
}
}
Err(e) if e.contains("this build lacks the") => {
FILTER_SKIPPED.fetch_add(1, Ordering::Relaxed);
}
Err(e) => diffs.push(format!("values of {group} {name}: {e}")),
}
}
diffs
}
/// clawhdf5-netcdf4 reads the file at `path` as netCDF-C does: the same
/// groups, dimensions, variables and values (see [`differences`]).
pub fn assert_matches_netcdf_c(path: &Path) {
let views = netcdf_c_views(&[path.to_path_buf()]);
let Some(Some(want)) = views.get(path) else {
panic!("netCDF-C cannot open {}", path.display());
};
let diffs = differences(path, want);
assert!(
diffs.is_empty(),
"{} differs from netCDF-C:\n{}",
path.display(),
diffs.join("\n")
);
}
@@ -0,0 +1,26 @@
# Files of the conformance corpus (conformance/.cache/corpus) that netCDF-C
# opens and clawhdf5-netcdf4 reads differently, with the reason.
# <path relative to the corpus root><TAB><reason>
# Read by tests/corpus_vs_netcdf_c.rs; see docs/known-issues.md
# ("NetCDF-4: differences from netCDF-C").
#
# External links: libhdf5 follows them into the other file (present in the
# corpus); clawhdf5 does not follow external links (known-issues: "External
# links and external raw data are not followed"), so the linked groups and
# datasets are missing and, in files without dimension scales, the phony
# dimension numbers after them shift.
hdf5/test/testfiles/be_extlink1.h5 external link not followed
hdf5/test/testfiles/le_extlink1.h5 external link not followed
hdf5/tools/test/testfiles/h5diff_ext2softlink_src.h5 external link not followed
hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-1.h5 external link not followed
hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-2.h5 external link not followed
#
# Values the HDF5 reader refuses and libhdf5 1.14.6 (the netCDF4-python
# wheel's) returns; metadata matches. The first three are the scale-offset
# and N-Bit ref-bug / our-error files of CONFORMANCE.md (libhdf5 reads past
# the stored data); bad_nbit_decompress.h5 is not in the conformance run and
# is not investigated yet.
cve_hdf5/cvefiles/cve-2025-2308.h5 values: scale-offset chunk shorter than its values (libhdf5 over-read)
cve_hdf5/cvefiles/cve-2025-44904.h5 values: unfiltered chunks shorter than a chunk (libhdf5 over-read)
hdf5/test/testfiles/bad_nbit_parms_walk.h5 values: N-Bit parameters too short (conformance our-error/ref-bug)
hdf5/test/testfiles/bad_nbit_decompress.h5 values: N-Bit chunk refused ("element count exceeds chunk size"); not investigated
@@ -0,0 +1,173 @@
//! Every file of a corpus that netCDF-C opens, read by clawhdf5-netcdf4 and
//! compared with what netCDF-C reports: groups (order), dimensions (names,
//! lengths, unlimited, order), variables (names, order, types, dimensions,
//! shapes) and the values of numeric variables of at most 5000 elements.
//!
//! Gated: set `CLAWHDF5_NETCDF_CORPUS` to a directory (relative to the
//! workspace root, or absolute) — the conformance corpus
//! (`conformance/.cache/corpus`) or any tree of HDF5/netCDF-4 files — and
//! `CLAWHDF5_PYTHON` to a Python with netCDF4-python (whose bundled
//! libnetcdf `tests/netcdf_c_view.py` calls). Without the variable the test
//! does nothing. netCDF-C reads each file in its own process (4 at a time,
//! `CLAWHDF5_NETCDF_JOBS`), under a 60 s timeout and a 4 GiB address-space
//! limit: 40 s to 3 minutes for the conformance corpus on tank.
//! `CLAWHDF5_NETCDF_VIEW` may name the saved output of an earlier
//! `netcdf_c_view.py --list <file of paths>` run over the same (absolute)
//! paths, to skip that.
//!
//! A file whose differences are explained is listed, with the reason, in
//! `tests/corpus_known_differences.txt` (and in `docs/known-issues.md`); a
//! listed file that matches fails the test too, so the list stays true.
//!
//! ```sh
//! CLAWHDF5_NETCDF_CORPUS=conformance/.cache/corpus CLAWHDF5_PYTHON=$PWD/.venv/bin/python \
//! cargo test -p clawhdf5-netcdf4 --test corpus_vs_netcdf_c -- --nocapture
//! ```
mod common;
use std::collections::BTreeMap;
use std::path::{Path, PathBuf};
use std::sync::atomic::Ordering;
use common::{FILTER_SKIPPED, RELABELLED, differences, netcdf_c_views, parse_views};
/// The file extensions `conformance/list_files.py` sweeps.
const EXTS: [&str; 7] = ["h5", "hdf5", "he5", "nc", "nc4", "hdf", "h5f"];
/// The files of the corpus: those with an HDF5/netCDF-4 extension, except
/// netCDF classic files (magic `CDF`), following directory symlinks, in
/// byte order of their paths.
fn corpus_files(root: &Path) -> Vec<PathBuf> {
fn walk(dir: &Path, out: &mut Vec<PathBuf>, depth: usize) {
let Ok(entries) = std::fs::read_dir(dir) else {
return;
};
for entry in entries.flatten() {
let path = entry.path();
if path.file_name().is_some_and(|n| n == ".git") {
continue;
}
let Ok(meta) = std::fs::metadata(&path) else {
continue;
};
if meta.is_dir() && depth < 32 {
walk(&path, out, depth + 1);
} else if meta.is_file()
&& path
.extension()
.and_then(|e| e.to_str())
.is_some_and(|e| EXTS.contains(&e.to_ascii_lowercase().as_str()))
{
let mut magic = [0u8; 3];
let classic = std::fs::File::open(&path)
.and_then(|mut f| std::io::Read::read_exact(&mut f, &mut magic))
.is_ok()
&& &magic == b"CDF";
if !classic {
out.push(path);
}
}
}
}
let mut out = Vec::new();
walk(root, &mut out, 0);
out.sort_by(|a, b| {
a.as_os_str()
.as_encoded_bytes()
.cmp(b.as_os_str().as_encoded_bytes())
});
out
}
/// The files `tests/corpus_known_differences.txt` explains: path relative
/// to the corpus root → reason.
fn known_differences() -> BTreeMap<String, String> {
let list = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/corpus_known_differences.txt");
std::fs::read_to_string(list)
.expect("tests/corpus_known_differences.txt")
.lines()
.filter(|l| !l.trim().is_empty() && !l.starts_with('#'))
.map(|l| {
let (path, reason) = l.split_once('\t').unwrap_or((l, ""));
(path.trim().to_string(), reason.trim().to_string())
})
.collect()
}
#[test]
fn corpus_matches_netcdf_c() {
let Ok(root) = std::env::var("CLAWHDF5_NETCDF_CORPUS") else {
eprintln!("SKIP: set CLAWHDF5_NETCDF_CORPUS to a corpus directory");
return;
};
// A relative path is from the workspace root (cargo runs the test in
// the crate's directory).
let root = Path::new(env!("CARGO_MANIFEST_DIR"))
.join("../..")
.join(root);
let root = std::fs::canonicalize(&root).unwrap_or(root);
let files = corpus_files(&root);
assert!(!files.is_empty(), "no files under {}", root.display());
// `CLAWHDF5_NETCDF_VIEW`: the output of an earlier run of
// `tests/netcdf_c_view.py` over the same files, which takes a few
// minutes over the conformance corpus.
let views = match std::env::var("CLAWHDF5_NETCDF_VIEW") {
Ok(saved) => parse_views(&std::fs::read_to_string(saved).expect("CLAWHDF5_NETCDF_VIEW")),
Err(_) => netcdf_c_views(&files),
};
let known = known_differences();
let (mut opened, mut matched) = (0, 0);
let mut explained = Vec::new();
let mut unexplained = Vec::new();
let mut stale = Vec::new();
for path in &files {
let Some(Some(want)) = views.get(path) else {
continue;
};
opened += 1;
let rel = path
.strip_prefix(&root)
.unwrap_or(path)
.to_string_lossy()
.into_owned();
let diffs = differences(path, want);
match (diffs.is_empty(), known.get(&rel)) {
(true, None) => matched += 1,
(true, Some(_)) => stale.push(rel),
(false, Some(reason)) => explained.push((rel, reason.clone(), diffs)),
(false, None) => unexplained.push((rel, diffs)),
}
}
eprintln!(
"{} files; netCDF-C opens {opened}; {matched} match; {} differ as explained; \
{} differ unexplained; {} listed but match; {} variables relabelled \
(floats of other than 4 or 8 bytes); {} variables' values not compared \
(filter not in this build)",
files.len(),
explained.len(),
unexplained.len(),
stale.len(),
RELABELLED.load(Ordering::Relaxed),
FILTER_SKIPPED.load(Ordering::Relaxed),
);
for (rel, reason, diffs) in &explained {
eprintln!("explained: {rel} ({reason}): {} differences", diffs.len());
}
for (rel, diffs) in &unexplained {
eprintln!("DIFFERS: {rel}");
for d in diffs.iter().take(20) {
eprintln!(" {d}");
}
}
assert!(
unexplained.is_empty(),
"{} files differ from netCDF-C unexplained",
unexplained.len()
);
assert!(
stale.is_empty(),
"listed in corpus_known_differences.txt but match: {stale:?}"
);
}
@@ -2,6 +2,8 @@
//!
//! Tests are skipped if python3 or netCDF4/xarray Python packages are not available.
mod common;
use std::process::Command;
use clawhdf5_netcdf4::{AttrValue, NcType, NetCDF4File};
@@ -842,3 +844,267 @@ ds.to_netcdf({path:?}, engine={engine:?}, unlimited_dims=["time"])
assert_same_view(&path);
}
}
// ===========================================================================
// Compared with netCDF-C itself (tests/netcdf_c_view.py): groups,
// dimensions, variables, types and values, in netCDF-C's order
// ===========================================================================
/// The dimensions of the variable `name` (a path) of `file`.
fn dim_names(file: &NetCDF4File, name: &str) -> Vec<String> {
file.variable(name)
.unwrap()
.dimensions()
.iter()
.map(|d| d.name.clone())
.collect()
}
/// An h5py file without dimension scales: netCDF-C's phony dimensions,
/// numbered file-wide (subgroups before their parent's variables, each
/// group's datasets in name order, as h5py does not track creation order),
/// shared by length within a group — but not between two axes of one
/// variable, nor between a fixed and an unlimited axis — with a length of 0
/// always unlimited and never shared with a fixed axis, and the real
/// dimension of a scale taken by length too.
#[test]
fn phony_dimensions_match_netcdf_c() {
skip_if_no_netcdf4!();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("phony.h5");
run_python(&format!(
r#"
import h5py
import numpy as np
with h5py.File({path:?}, "w") as f:
f["zz"] = np.arange(12.0).reshape(2, 2, 3)
f["aa"] = np.arange(6, dtype="i4").reshape(3, 2)
f.create_dataset("un", data=np.ones((2, 3), "f4"), maxshape=(None, 3))
f.create_dataset("un2", data=np.ones(2, "f4"), maxshape=(None,))
f["zero"] = np.zeros((0,))
f.create_dataset("zero_un", (0,), "f4", maxshape=(None,))
f["zero2"] = np.zeros((0, 2))
f["s"] = 1.5
g = f.create_group("g")
g["x"] = np.arange(5, dtype="i2")
g["y"] = np.arange(2, dtype="u1")
g.create_group("h")["q"] = np.arange(7.0)
f.create_group("b")["w"] = np.arange(18, dtype="i8").reshape(2, 9)
f["sc"] = np.arange(6.0)
f["sc"].make_scale("sc")
f["second_only"] = np.zeros((2, 6))
f["second_only"].dims[1].attach_scale(f["sc"])
"#,
path = path.display().to_string()
));
common::assert_matches_netcdf_c(&path);
let file = NetCDF4File::open(&path).unwrap();
// `sc` is dimension 0; /b gets 1 and 2, /g/h 3, /g 4 and 5, / from 6.
assert_eq!(dim_names(&file, "b/w"), ["phony_dim_1", "phony_dim_2"]);
assert_eq!(dim_names(&file, "g/h/q"), ["phony_dim_3"]);
let h = file.group("g/h").unwrap();
assert_eq!(h.variable_names().unwrap(), ["q"]);
assert_eq!(h.dimensions().unwrap()[0].name, "phony_dim_3");
assert_eq!(dim_names(&file, "aa"), ["phony_dim_6", "phony_dim_7"]);
assert_eq!(
dim_names(&file, "zz"),
["phony_dim_7", "phony_dim_11", "phony_dim_6"]
);
assert_eq!(dim_names(&file, "second_only"), ["phony_dim_7", "sc"]);
assert_eq!(dim_names(&file, "un"), ["phony_dim_8", "phony_dim_6"]);
assert_eq!(dim_names(&file, "un2"), ["phony_dim_8"]);
assert_eq!(dim_names(&file, "zero"), ["phony_dim_9"]);
assert_eq!(dim_names(&file, "zero_un"), ["phony_dim_9"]);
assert_eq!(dim_names(&file, "zero2"), ["phony_dim_10", "phony_dim_7"]);
let root: Vec<(String, u64, bool)> = file
.dimensions()
.unwrap()
.into_iter()
.map(|d| (d.name, d.size, d.is_unlimited))
.collect();
assert_eq!(root[0], ("sc".to_string(), 6, false));
assert!(root.contains(&("phony_dim_8".to_string(), 2, true)));
assert!(root.contains(&("phony_dim_9".to_string(), 0, true)));
assert!(root.contains(&("phony_dim_10".to_string(), 0, true)));
assert_eq!(
file.variable_names().unwrap(),
[
"aa",
"s",
"sc",
"second_only",
"un",
"un2",
"zero",
"zero2",
"zero_un",
"zz"
]
);
}
/// Datasets of types netCDF-C cannot represent are not variables:
/// references, bit fields, array types, a compound with a reference member
/// or a half-float member, a compound nesting a compound not seen before;
/// enum, compound, variable-length and opaque types are variables of those
/// classes. netCDF-C also remembers a type it failed to read, so the second
/// dataset of a compound with a reference member is a variable, and a
/// nested compound is accepted once a dataset of the inner type was read.
#[test]
fn types_netcdf_c_skips_are_not_variables() {
skip_if_no_netcdf4!();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("types.h5");
run_python(&format!(
r#"
import h5py
import numpy as np
inner = np.dtype([("x", "i2"), ("y", "f4")])
with h5py.File({path:?}, "w") as f:
f["i4"] = np.arange(3, dtype="i4")
f["f2"] = np.arange(3, dtype="f2")
f["s1"] = np.array([b"a", b"b"], dtype="S1")
f["s5"] = np.array([b"abc", b"b"], dtype="S5")
f["vs"] = np.array(["x", "yy"], dtype=h5py.string_dtype())
f["bool"] = np.array([True, False])
f["cmp"] = np.array([(1, 2.0)], dtype=[("a", "i4"), ("b", "f8")])
f["en"] = np.array([0, 1], dtype=h5py.enum_dtype({{"A": 0, "B": 1}}, basetype="i1"))
d = f.create_dataset("vl", (2,), dtype=h5py.vlen_dtype("i4"))
d[0] = [1, 2]
d[1] = [3]
f["op"] = np.array([b"ab", b"cd"], dtype="V2")
f.create_dataset("ref", (1,), dtype=h5py.ref_dtype)[0] = f["i4"].ref
f["cref1"] = np.array([(1, f["i4"].ref)], dtype=[("a", "i4"), ("r", h5py.ref_dtype)])
f["cref2"] = np.array([(2, f["i4"].ref)], dtype=[("a", "i4"), ("r", h5py.ref_dtype)])
f["cmp_f2"] = np.zeros(2, dtype=[("a", "f2")])
f["in1"] = np.zeros(2, dtype=inner)
f["nested"] = np.zeros(2, dtype=[("a", "i4"), ("in", inner)])
f["a_nested"] = np.zeros(2, dtype=[("b", "i4"), ("in", np.dtype([("p", "i1")]))])
sid = h5py.h5s.create_simple((2,))
h5py.h5d.create(f.id, b"bitf", h5py.h5t.STD_B8LE.copy(), sid)
h5py.h5d.create(f.id, b"arr", h5py.h5t.array_create(h5py.h5t.NATIVE_INT32, (3,)), sid)
"#,
path = path.display().to_string()
));
common::assert_matches_netcdf_c(&path);
let file = NetCDF4File::open(&path).unwrap();
assert_eq!(
file.variable_names().unwrap(),
[
"bool", "cmp", "cref2", "en", "f2", "i4", "in1", "nested", "op", "s1", "s5", "vl", "vs"
]
);
let nc_type = |name: &str| file.variable(name).unwrap().nc_type().unwrap();
assert_eq!(nc_type("bool"), NcType::Enum);
assert_eq!(nc_type("cmp"), NcType::Compound);
assert_eq!(nc_type("vl"), NcType::VLen);
assert_eq!(nc_type("op"), NcType::Opaque);
assert_eq!(nc_type("s1"), NcType::Char);
assert_eq!(nc_type("s5"), NcType::String);
// netCDF-C 4.9.3 on libhdf5 1.14.6 says NC_STRING (deliberate
// difference, see the README).
assert_eq!(nc_type("f2"), NcType::Float);
assert_eq!(
file.variable("f2").unwrap().read_raw_f32().unwrap(),
[0.0, 1.0, 2.0]
);
assert!(matches!(
file.variable("ref"),
Err(clawhdf5_netcdf4::Error::VariableNotFound(_))
));
}
/// The order of groups and variables: creation order where the group
/// tracks it (every netCDF-4 file; also h5py with `track_order`), compact
/// or dense (more than 8 links), else name order (h5py by default).
#[test]
fn group_and_variable_order_match_netcdf_c() {
skip_if_no_netcdf4!();
let dir = tempfile::tempdir().unwrap();
let nc = dir.path().join("order.nc");
let h5 = dir.path().join("order.h5");
run_python(&format!(
r#"
import h5py
import netCDF4 as nc
import numpy as np
names = ["zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"]
with nc.Dataset({nc:?}, "w") as f:
f.createDimension("x", 2)
for i, n in enumerate(names):
f.createVariable(n, "i4", ("x",))[:] = [i, i + 1]
for n in ["gz", "ga", "gm"]:
f.createGroup(n).createVariable("v", "f8", ("x",))[:] = [1, 2]
small = f.createGroup("small")
for n in ["q", "b", "p"]:
small.createVariable(n, "i2", ("x",))[:] = [3, 4]
with h5py.File({h5:?}, "w") as f:
for i, n in enumerate(names):
f[n] = np.arange(i + 1)
t = f.create_group("tracked", track_order=True)
for i, n in enumerate(names):
t[n] = np.arange(3, dtype="i1")
for n in ["gz", "ga"]:
f.create_group(n)["v"] = np.arange(4.0)
"#,
nc = nc.display().to_string(),
h5 = h5.display().to_string()
));
common::assert_matches_netcdf_c(&nc);
common::assert_matches_netcdf_c(&h5);
let file = NetCDF4File::open(&nc).unwrap();
assert_eq!(
file.variable_names().unwrap(),
[
"zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"
]
);
assert_eq!(file.group_names().unwrap(), ["gz", "ga", "gm", "small"]);
let file = NetCDF4File::open(&h5).unwrap();
assert_eq!(file.group_names().unwrap(), ["ga", "gz", "tracked"]);
assert_eq!(
file.variable_names().unwrap(),
[
"a", "alpha", "beta", "c", "gamma", "k", "mid", "omega", "zeta", "zz"
]
);
assert_eq!(
file.group("tracked").unwrap().variable_names().unwrap(),
[
"zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"
]
);
}
/// The files of the earlier tests, compared with netCDF-C itself too:
/// dimension scales of h5py, netCDF4-python files with groups, unlimited
/// dimensions and non-coordinate variables named like a dimension.
#[test]
fn netcdf4_python_files_match_netcdf_c() {
skip_if_no_netcdf4!();
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("mixed.nc");
run_python(&format!(
r#"
import netCDF4 as nc
import numpy as np
with nc.Dataset({path:?}, "w") as f:
f.createDimension("time", None)
f.createDimension("p", 2)
f.createDimension("q", 2)
f.createVariable("a", "i4", ("time",))[0:2] = [1, 2]
f.createVariable("b", "f4", ("time", "q"))[0:5, :] = np.arange(10).reshape(5, 2)
f.createVariable("q", "f4", ("q",))[:] = [0, 1]
f.createVariable("p", "f4", ("q", "p"))[:] = np.array([[0, 1], [2, 3]])
g = f.createGroup("g")
g.createDimension("r", 3)
g.createVariable("w", "i4", ("r", "time"))[:, 0:1] = np.ones((3, 1))
g.createGroup("h").createVariable("z", "i8", ("p", "r"))[:] = np.arange(6).reshape(2, 3)
"#,
path = path.display().to_string()
));
common::assert_matches_netcdf_c(&path);
}
@@ -776,9 +776,12 @@ fn test_pure_dimensions_hidden_and_non_coord_names() {
.with_shape(&[2]);
let file = NetCDF4File::from_bytes(b.finish().unwrap()).unwrap();
// `x`, and the phony dimension netCDF-C gives the 3 values of
// `_nc4_non_coord_x` (no dimension scale attached).
let dims = file.dimensions().unwrap();
assert_eq!(dims.len(), 1);
assert_eq!(dims[0].name, "x");
let names: Vec<&str> = dims.iter().map(|d| d.name.as_str()).collect();
assert_eq!(names, ["x", "phony_dim_1"]);
assert_eq!(dims[1].size, 3);
let mut names = file.variable_names().unwrap();
names.sort();
assert_eq!(names, ["v", "x"]);
@@ -786,7 +789,8 @@ fn test_pure_dimensions_hidden_and_non_coord_names() {
assert_eq!(x.name(), "x");
assert_eq!(x.read_raw_f64().unwrap(), [1.0, 2.0, 3.0]);
assert!(!x.is_coordinate());
// No DIMENSION_LIST: `v` gets `x` by size, as before.
// No DIMENSION_LIST: `v` gets the group's first dimension of its length
// (netCDF-C's phony rule, which also takes real dimensions).
assert_eq!(file.variable("v").unwrap().dimensions()[0].name, "x");
}
@@ -0,0 +1,238 @@
#!/usr/bin/env python3
"""netcdf_c_view.py FILE...: print each file as netCDF-C sees it.
The metadata comes from netCDF-C itself (the libnetcdf that netCDF4-python
bundles, called through ctypes), not from netCDF4-python's objects, which
leave out variables of types netCDF-C supports but netCDF4-python does not
(opaque, compounds of vlen strings, ...). Values come from netCDF4-python.
Each file is read in its own process under a timeout, so a file that crashes
or hangs libnetcdf only costs that file.
Output, per file, fields separated by tabs:
FILE <path>
ERROR <message> netCDF-C cannot open it; nothing else
G <group> every group, pre-order, children in
netCDF-C's order ("/" is the root)
D <group> <name> <length> <0|1> its dimensions in dimension-id order
(1: unlimited)
V <group> <name> <type> <dims> <shape>
its variables in netCDF-C's order;
<dims> and <shape> comma-separated,
"-" when there are none; <type> as
clawhdf5_netcdf4::NcType prints it
X <group> <name> <values> the values of a numeric variable of
at most MAX_VALUES elements, as
space-separated Python float reprs
("-" when there are none)
END
Used by tests/corpus_vs_netcdf_c.rs (CLAWHDF5_NETCDF_CORPUS).
"""
import concurrent.futures
import ctypes
import glob
import os
import subprocess
import sys
MAX_VALUES = 5000
TIMEOUT = 60
MEMORY = 4 << 30 # address space of each file's process
ATOMIC = {
1: "NC_BYTE", 2: "NC_CHAR", 3: "NC_SHORT", 4: "NC_INT", 5: "NC_FLOAT",
6: "NC_DOUBLE", 7: "NC_UBYTE", 8: "NC_USHORT", 9: "NC_UINT",
10: "NC_INT64", 11: "NC_UINT64", 12: "NC_STRING",
}
USER_CLASS = {13: "NC_VLEN", 14: "NC_OPAQUE", 15: "NC_ENUM", 16: "NC_COMPOUND"}
NUMERIC = {"NC_BYTE", "NC_SHORT", "NC_INT", "NC_FLOAT", "NC_DOUBLE",
"NC_UBYTE", "NC_USHORT", "NC_UINT", "NC_INT64", "NC_UINT64"}
NAME = 257 # NC_MAX_NAME + 1
MAXDIMS = 1024
def libnetcdf():
"""The libnetcdf netCDF4-python is linked with (same process, same copy)."""
import netCDF4
site = os.path.dirname(os.path.dirname(netCDF4.__file__))
found = glob.glob(os.path.join(site, "netcdf4.libs", "libnetcdf*.so*")) + \
glob.glob(os.path.join(site, "netCDF4.libs", "libnetcdf*.so*"))
if found:
return ctypes.CDLL(found[0])
return ctypes.CDLL("libnetcdf.so")
def view(path, out):
nc = libnetcdf()
ncid = ctypes.c_int()
rc = nc.nc_open(path.encode(), 0, ctypes.byref(ncid)) # NC_NOWRITE
if rc != 0:
nc.nc_strerror.restype = ctypes.c_char_p
out.append("ERROR\t" + nc.nc_strerror(rc).decode(errors="replace"))
return
numeric = [] # (group path, variable name)
try:
walk(nc, ncid.value, "/", out, numeric)
finally:
nc.nc_close(ncid)
values(path, numeric, out)
def check(rc, what):
if rc != 0:
raise RuntimeError(f"{what} failed: {rc}")
def name_of(fn, *args):
buf = ctypes.create_string_buffer(NAME)
check(fn(*args, buf), fn.__name__)
return buf.value.decode(errors="replace")
def walk(nc, gid, path, out, numeric):
out.append(f"G\t{path}")
n = ctypes.c_int()
ids = (ctypes.c_int * MAXDIMS)()
check(nc.nc_inq_dimids(gid, ctypes.byref(n), ids, 0), "nc_inq_dimids")
for dimid in ids[: n.value]:
length = ctypes.c_size_t()
dname = ctypes.create_string_buffer(NAME)
check(nc.nc_inq_dim(gid, dimid, dname, ctypes.byref(length)), "nc_inq_dim")
out.append(f"D\t{path}\t{dname.value.decode(errors='replace')}\t{length.value}\t"
f"{int(is_unlimited(nc, gid, dimid))}")
nvars = ctypes.c_int()
varids = (ctypes.c_int * 65536)()
check(nc.nc_inq_varids(gid, ctypes.byref(nvars), varids), "nc_inq_varids")
for varid in varids[: nvars.value]:
vname = ctypes.create_string_buffer(NAME)
xtype = ctypes.c_int()
ndims = ctypes.c_int()
dimids = (ctypes.c_int * MAXDIMS)()
natts = ctypes.c_int()
check(nc.nc_inq_var(gid, varid, vname, ctypes.byref(xtype), ctypes.byref(ndims),
dimids, ctypes.byref(natts)), "nc_inq_var")
names, shape = [], []
for dimid in dimids[: ndims.value]:
dname = ctypes.create_string_buffer(NAME)
length = ctypes.c_size_t()
if nc.nc_inq_dim(gid, dimid, dname, ctypes.byref(length)) != 0:
names.append("?")
shape.append("?")
continue
names.append(dname.value.decode(errors="replace"))
shape.append(str(length.value))
tname = type_name(nc, gid, xtype.value)
vn = vname.value.decode(errors="replace")
out.append(f"V\t{path}\t{vn}\t{tname}\t{','.join(names) or '-'}\t{','.join(shape) or '-'}")
if tname in NUMERIC and "?" not in shape:
count = 1
for s in shape:
count *= int(s)
if count <= MAX_VALUES:
numeric.append((path, vn))
ngrps = ctypes.c_int()
grps = (ctypes.c_int * 65536)()
check(nc.nc_inq_grps(gid, ctypes.byref(ngrps), grps), "nc_inq_grps")
for child in grps[: ngrps.value]:
cname = ctypes.create_string_buffer(NAME)
check(nc.nc_inq_grpname(child, cname), "nc_inq_grpname")
cpath = path.rstrip("/") + "/" + cname.value.decode(errors="replace")
walk(nc, child, cpath, out, numeric)
def is_unlimited(nc, gid, dimid):
n = ctypes.c_int()
ids = (ctypes.c_int * MAXDIMS)()
# Unlimited dimensions visible from this group include its parents'.
if nc.nc_inq_unlimdims(gid, ctypes.byref(n), ids) != 0:
return False
return dimid in ids[: n.value]
def type_name(nc, gid, xtype):
if xtype in ATOMIC:
return ATOMIC[xtype]
size = ctypes.c_size_t()
base = ctypes.c_int()
nfields = ctypes.c_size_t()
klass = ctypes.c_int()
tname = ctypes.create_string_buffer(NAME)
if nc.nc_inq_user_type(gid, xtype, tname, ctypes.byref(size), ctypes.byref(base),
ctypes.byref(nfields), ctypes.byref(klass)) != 0:
return f"type{xtype}"
return USER_CLASS.get(klass.value, f"class{klass.value}")
def values(path, numeric, out):
if not numeric:
return
import numpy as np
import netCDF4
try:
ds = netCDF4.Dataset(path)
except Exception: # noqa: BLE001 - netCDF4-python refuses some files netCDF-C opens
return
with ds:
for gpath, name in numeric:
try:
group = ds if gpath == "/" else ds[gpath]
var = group.variables[name]
var.set_auto_maskandscale(False)
# A whole-variable read through netCDF-C 4.9.3 lays out a
# variable shorter than an unlimited dimension that is not its
# first wrongly (written values first); reads of one index of
# the leading axis are right.
if var.ndim >= 2:
data = np.stack([np.asarray(var[i]) for i in range(var.shape[0])]) \
if var.shape[0] else np.zeros(var.shape)
else:
data = np.asarray(var[...])
flat = np.asarray(data, dtype=np.float64).ravel()
except Exception: # noqa: BLE001
continue
vals = " ".join(repr(float(v)) for v in flat) or "-"
out.append(f"X\t{gpath}\t{name}\t{vals}")
def limit_memory():
import resource
resource.setrlimit(resource.RLIMIT_AS, (MEMORY, MEMORY))
def one(path):
"""Run view() for one file in a child process."""
try:
proc = subprocess.run([sys.executable, __file__, "--one", path],
capture_output=True, timeout=TIMEOUT, text=True,
preexec_fn=limit_memory)
except subprocess.TimeoutExpired:
return [f"FILE\t{path}", "ERROR\ttimeout", "END"]
lines = proc.stdout.splitlines()
if proc.returncode != 0 or not lines or lines[-1] != "END":
err = (proc.stderr.strip().splitlines() or [f"exit {proc.returncode}"])[-1]
return [f"FILE\t{path}", f"ERROR\tchild failed: {err}", "END"]
return lines
def main(argv):
if argv and argv[0] == "--one":
out = [f"FILE\t{argv[1]}"]
try:
view(argv[1], out)
except Exception as e: # noqa: BLE001
out = [f"FILE\t{argv[1]}", f"ERROR\t{e}"]
out.append("END")
sys.stdout.write("\n".join(out) + "\n")
return
if argv and argv[0] == "--list":
with open(argv[1]) as fh:
argv = [line.rstrip("\n") for line in fh if line.strip()]
jobs = int(os.environ.get("CLAWHDF5_NETCDF_JOBS", "4"))
with concurrent.futures.ThreadPoolExecutor(jobs) as pool:
for lines in pool.map(one, argv):
sys.stdout.write("\n".join(lines) + "\n")
if __name__ == "__main__":
main(sys.argv[1:])