clawhdf5-netcdf4: phony dimensions, skipped types and order as netCDF-C
Read a file's metadata the way netCDF-C 4.9.3 does (libhdf5/hdf5open.c), for the whole file on first use (src/model.rs, replacing src/scope.rs): - links in creation order when the group tracks it, else name order; a group's datasets before its subgroups; dimension ids file-wide; - variables' dimensions from _Netcdf4Coordinates (file-wide ids), else the scales DIMENSION_LIST attaches when the first axis has one, else netCDF-C's phony dimensions phony_dim_<id> (create_phony_dims: shared by length and unlimitedness within a group, not between two axes of one variable, numbered subgroups first, a zero length unlimited); - datasets of types netCDF-C cannot represent are not variables (references, bit fields, time, arrays, compounds/enums/VLENs over them), replaying netCDF-C's file-wide type list, failed types included; - unlimited lengths as nc4_find_dim_len (its group and below). NcType gains Enum, Compound, VLen, Opaque and is #[non_exhaustive]; Variable::nc_type is netCDF-C's type (1-byte strings NC_CHAR). New clawhdf5_format::group_v2::links_in_creation_order_in. Tests compare with netCDF-C itself (tests/netcdf_c_view.py calls the libnetcdf netCDF4-python bundles through ctypes): new interop cases for h5py files without dimension scales, every type class, link order; and the gated corpus_vs_netcdf_c (CLAWHDF5_NETCDF_CORPUS): 420 of the 429 conformance-corpus files netCDF-C opens match (main: 68); the other 9 are explained in tests/corpus_known_differences.txt and known-issues. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -0,0 +1,281 @@
|
||||
//! clawhdf5-netcdf4 compared with netCDF-C itself: `tests/netcdf_c_view.py`
|
||||
//! prints what netCDF-C (the libnetcdf netCDF4-python bundles) reports for
|
||||
//! a file, and [`differences`] lists how clawhdf5-netcdf4's reading differs.
|
||||
#![allow(dead_code)]
|
||||
|
||||
use std::collections::{BTreeMap, BTreeSet};
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::Command;
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
|
||||
use clawhdf5_netcdf4::{NcType, NetCDF4File, NetCDF4Group, Variable};
|
||||
|
||||
/// The Python with netCDF4-python (`CLAWHDF5_PYTHON`, else `python3`).
|
||||
pub fn python() -> String {
|
||||
std::env::var("CLAWHDF5_PYTHON").unwrap_or_else(|_| "python3".to_string())
|
||||
}
|
||||
|
||||
/// netCDF-C's view of one file.
|
||||
#[derive(Default)]
|
||||
pub struct View {
|
||||
/// `G`, `D` and `V` lines, in order.
|
||||
pub meta: Vec<String>,
|
||||
/// `(group, variable)` → values, from the `X` lines.
|
||||
pub values: BTreeMap<(String, String), Vec<f64>>,
|
||||
}
|
||||
|
||||
/// netCDF-C's view of each file (`None`: netCDF-C cannot open it), from
|
||||
/// `tests/netcdf_c_view.py`.
|
||||
pub fn netcdf_c_views(files: &[PathBuf]) -> BTreeMap<PathBuf, Option<View>> {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let list = dir.path().join("files.txt");
|
||||
let mut paths = String::new();
|
||||
for f in files {
|
||||
paths.push_str(&f.to_string_lossy());
|
||||
paths.push('\n');
|
||||
}
|
||||
std::fs::write(&list, paths).unwrap();
|
||||
let script = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/netcdf_c_view.py");
|
||||
let out = Command::new(python())
|
||||
.arg(&script)
|
||||
.arg("--list")
|
||||
.arg(&list)
|
||||
.output()
|
||||
.expect("failed to run python");
|
||||
assert!(
|
||||
out.status.success(),
|
||||
"netcdf_c_view.py failed: {}",
|
||||
String::from_utf8_lossy(&out.stderr)
|
||||
);
|
||||
parse_views(&String::from_utf8_lossy(&out.stdout))
|
||||
}
|
||||
|
||||
/// The views in the output of `tests/netcdf_c_view.py`.
|
||||
pub fn parse_views(text: &str) -> BTreeMap<PathBuf, Option<View>> {
|
||||
let mut views = BTreeMap::new();
|
||||
let mut current: Option<(PathBuf, Option<View>)> = None;
|
||||
for line in text.lines() {
|
||||
let fields: Vec<&str> = line.split('\t').collect();
|
||||
match fields[0] {
|
||||
"FILE" => current = Some((PathBuf::from(fields[1]), Some(View::default()))),
|
||||
"END" => {
|
||||
let (path, view) = current.take().expect("END without FILE");
|
||||
views.insert(path, view);
|
||||
}
|
||||
"ERROR" => current.as_mut().expect("ERROR without FILE").1 = None,
|
||||
"X" => {
|
||||
let view = current.as_mut().and_then(|c| c.1.as_mut()).unwrap();
|
||||
let vals = if fields[3] == "-" {
|
||||
Vec::new()
|
||||
} else {
|
||||
fields[3]
|
||||
.split(' ')
|
||||
.map(|v| v.parse().expect("value"))
|
||||
.collect()
|
||||
};
|
||||
view.values
|
||||
.insert((fields[1].to_string(), fields[2].to_string()), vals);
|
||||
}
|
||||
_ => {
|
||||
let view = current.as_mut().and_then(|c| c.1.as_mut()).unwrap();
|
||||
view.meta.push(line.to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
views
|
||||
}
|
||||
|
||||
/// Variables whose type is labelled as netCDF-C labels it rather than as
|
||||
/// clawhdf5-netcdf4 does (see [`nc_type_label`]).
|
||||
pub static RELABELLED: AtomicUsize = AtomicUsize::new(0);
|
||||
/// Values not compared because this build of the HDF5 reader lacks a
|
||||
/// filter (a cargo feature).
|
||||
pub static FILTER_SKIPPED: AtomicUsize = AtomicUsize::new(0);
|
||||
|
||||
/// The same `G`/`D`/`V` lines from clawhdf5-netcdf4.
|
||||
pub fn our_meta(file: &NetCDF4File) -> Result<Vec<String>, String> {
|
||||
let e = |e: clawhdf5_netcdf4::Error| e.to_string();
|
||||
let mut out = Vec::new();
|
||||
out.push("G\t/".to_string());
|
||||
push_group(
|
||||
&mut out,
|
||||
file.hdf5_file(),
|
||||
"/",
|
||||
file.dimensions().map_err(e)?,
|
||||
file.variables().map_err(e)?,
|
||||
)?;
|
||||
for name in file.group_names().map_err(e)? {
|
||||
walk(
|
||||
&mut out,
|
||||
file.hdf5_file(),
|
||||
&format!("/{name}"),
|
||||
&file.group(&name).map_err(e)?,
|
||||
)?;
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
fn walk(
|
||||
out: &mut Vec<String>,
|
||||
hdf5: &clawhdf5::File,
|
||||
path: &str,
|
||||
group: &NetCDF4Group<'_>,
|
||||
) -> Result<(), String> {
|
||||
let e = |e: clawhdf5_netcdf4::Error| e.to_string();
|
||||
out.push(format!("G\t{path}"));
|
||||
push_group(
|
||||
out,
|
||||
hdf5,
|
||||
path,
|
||||
group.dimensions().map_err(e)?,
|
||||
group.variables().map_err(e)?,
|
||||
)?;
|
||||
for name in group.group_names().map_err(e)? {
|
||||
walk(
|
||||
out,
|
||||
hdf5,
|
||||
&format!("{path}/{name}"),
|
||||
&group.group(&name).map_err(e)?,
|
||||
)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The type of variable `name` of the group at `path` as the `V` line
|
||||
/// shows it: its `NcType`, except for the one deliberate difference from
|
||||
/// netCDF-C — a float of other than 4 or 8 bytes (half, bfloat16, the 4-,
|
||||
/// 6- and 8-bit floats, `long double`), `NC_FLOAT`/`NC_DOUBLE` here, is
|
||||
/// labelled `NC_STRING` by netCDF-C 4.9.3 on libhdf5 1.14.6; such a
|
||||
/// variable is shown as netCDF-C shows it, and counted.
|
||||
fn nc_type_label(hdf5: &clawhdf5::File, path: &str, var: &Variable<'_>) -> Result<String, String> {
|
||||
let nc_type = var.nc_type().map_err(|e| e.to_string())?;
|
||||
if matches!(nc_type, NcType::Float | NcType::Double) {
|
||||
let dataset = format!("{}/{}", path.trim_end_matches('/'), var.name());
|
||||
if let Ok(ds) = hdf5.dataset(&dataset)
|
||||
&& let Ok(clawhdf5_format::datatype::Datatype::FloatingPoint { size, .. }) =
|
||||
ds.raw_datatype()
|
||||
&& size != 4
|
||||
&& size != 8
|
||||
{
|
||||
RELABELLED.fetch_add(1, Ordering::Relaxed);
|
||||
return Ok("NC_STRING".to_string());
|
||||
}
|
||||
}
|
||||
Ok(nc_type.to_string())
|
||||
}
|
||||
|
||||
fn push_group(
|
||||
out: &mut Vec<String>,
|
||||
hdf5: &clawhdf5::File,
|
||||
path: &str,
|
||||
dims: Vec<clawhdf5_netcdf4::Dimension>,
|
||||
vars: Vec<Variable<'_>>,
|
||||
) -> Result<(), String> {
|
||||
for d in dims {
|
||||
out.push(format!(
|
||||
"D\t{path}\t{}\t{}\t{}",
|
||||
d.name,
|
||||
d.size,
|
||||
u8::from(d.is_unlimited)
|
||||
));
|
||||
}
|
||||
for v in vars {
|
||||
let dims: Vec<&str> = v.dimensions().iter().map(|d| d.name.as_str()).collect();
|
||||
let shape: Vec<String> = v
|
||||
.shape()
|
||||
.map_err(|e| e.to_string())?
|
||||
.iter()
|
||||
.map(u64::to_string)
|
||||
.collect();
|
||||
let or_dash = |s: String| if s.is_empty() { "-".to_string() } else { s };
|
||||
out.push(format!(
|
||||
"V\t{path}\t{}\t{}\t{}\t{}",
|
||||
v.name(),
|
||||
nc_type_label(hdf5, path, &v)?,
|
||||
or_dash(dims.join(",")),
|
||||
or_dash(shape.join(","))
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// The values of variable `name` of group `group`.
|
||||
fn our_values(file: &NetCDF4File, group: &str, name: &str) -> Result<Vec<f64>, String> {
|
||||
let var = if group == "/" {
|
||||
file.variable(name)
|
||||
} else {
|
||||
file.variable(&format!("{}/{name}", group.trim_start_matches('/')))
|
||||
}
|
||||
.map_err(|e| e.to_string())?;
|
||||
match var.nc_type().map_err(|e| e.to_string())? {
|
||||
NcType::String | NcType::Char => Err("not numeric".into()),
|
||||
_ => var.read_raw_f64().map_err(|e| e.to_string()),
|
||||
}
|
||||
}
|
||||
|
||||
/// How clawhdf5-netcdf4's reading of `path` differs from `want`; empty
|
||||
/// when it does not.
|
||||
pub fn differences(path: &Path, want: &View) -> Vec<String> {
|
||||
let file = match NetCDF4File::open(path) {
|
||||
Ok(f) => f,
|
||||
Err(e) => return vec![format!("netCDF-C opens it, clawhdf5-netcdf4 does not: {e}")],
|
||||
};
|
||||
let meta = match std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| our_meta(&file))) {
|
||||
Ok(Ok(meta)) => meta,
|
||||
Ok(Err(e)) => return vec![format!("error: {e}")],
|
||||
Err(_) => return vec!["panic".to_string()],
|
||||
};
|
||||
let mut diffs = Vec::new();
|
||||
if meta != want.meta {
|
||||
let ours: BTreeSet<&String> = meta.iter().collect();
|
||||
let theirs: BTreeSet<&String> = want.meta.iter().collect();
|
||||
for line in want.meta.iter().filter(|l| !ours.contains(l)) {
|
||||
diffs.push(format!("netCDF-C: {line}"));
|
||||
}
|
||||
for line in meta.iter().filter(|l| !theirs.contains(l)) {
|
||||
diffs.push(format!("ours: {line}"));
|
||||
}
|
||||
if diffs.is_empty() {
|
||||
diffs.push("same lines, different order".to_string());
|
||||
}
|
||||
}
|
||||
for ((group, name), want) in &want.values {
|
||||
let got = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||
our_values(&file, group, name)
|
||||
}))
|
||||
.unwrap_or_else(|_| Err("panic".to_string()));
|
||||
match got {
|
||||
Ok(got) => {
|
||||
let same = got.len() == want.len()
|
||||
&& got
|
||||
.iter()
|
||||
.zip(want)
|
||||
.all(|(a, b)| a.to_bits() == b.to_bits() || (a.is_nan() && b.is_nan()));
|
||||
if !same {
|
||||
diffs.push(format!("values of {group} {name} differ"));
|
||||
}
|
||||
}
|
||||
Err(e) if e.contains("this build lacks the") => {
|
||||
FILTER_SKIPPED.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
Err(e) => diffs.push(format!("values of {group} {name}: {e}")),
|
||||
}
|
||||
}
|
||||
diffs
|
||||
}
|
||||
|
||||
/// clawhdf5-netcdf4 reads the file at `path` as netCDF-C does: the same
|
||||
/// groups, dimensions, variables and values (see [`differences`]).
|
||||
pub fn assert_matches_netcdf_c(path: &Path) {
|
||||
let views = netcdf_c_views(&[path.to_path_buf()]);
|
||||
let Some(Some(want)) = views.get(path) else {
|
||||
panic!("netCDF-C cannot open {}", path.display());
|
||||
};
|
||||
let diffs = differences(path, want);
|
||||
assert!(
|
||||
diffs.is_empty(),
|
||||
"{} differs from netCDF-C:\n{}",
|
||||
path.display(),
|
||||
diffs.join("\n")
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
# Files of the conformance corpus (conformance/.cache/corpus) that netCDF-C
|
||||
# opens and clawhdf5-netcdf4 reads differently, with the reason.
|
||||
# <path relative to the corpus root><TAB><reason>
|
||||
# Read by tests/corpus_vs_netcdf_c.rs; see docs/known-issues.md
|
||||
# ("NetCDF-4: differences from netCDF-C").
|
||||
#
|
||||
# External links: libhdf5 follows them into the other file (present in the
|
||||
# corpus); clawhdf5 does not follow external links (known-issues: "External
|
||||
# links and external raw data are not followed"), so the linked groups and
|
||||
# datasets are missing and, in files without dimension scales, the phony
|
||||
# dimension numbers after them shift.
|
||||
hdf5/test/testfiles/be_extlink1.h5 external link not followed
|
||||
hdf5/test/testfiles/le_extlink1.h5 external link not followed
|
||||
hdf5/tools/test/testfiles/h5diff_ext2softlink_src.h5 external link not followed
|
||||
hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-1.h5 external link not followed
|
||||
hdf5/tools/test/testfiles/h5diff_grp_recurse_ext2-2.h5 external link not followed
|
||||
#
|
||||
# Values the HDF5 reader refuses and libhdf5 1.14.6 (the netCDF4-python
|
||||
# wheel's) returns; metadata matches. The first three are the scale-offset
|
||||
# and N-Bit ref-bug / our-error files of CONFORMANCE.md (libhdf5 reads past
|
||||
# the stored data); bad_nbit_decompress.h5 is not in the conformance run and
|
||||
# is not investigated yet.
|
||||
cve_hdf5/cvefiles/cve-2025-2308.h5 values: scale-offset chunk shorter than its values (libhdf5 over-read)
|
||||
cve_hdf5/cvefiles/cve-2025-44904.h5 values: unfiltered chunks shorter than a chunk (libhdf5 over-read)
|
||||
hdf5/test/testfiles/bad_nbit_parms_walk.h5 values: N-Bit parameters too short (conformance our-error/ref-bug)
|
||||
hdf5/test/testfiles/bad_nbit_decompress.h5 values: N-Bit chunk refused ("element count exceeds chunk size"); not investigated
|
||||
@@ -0,0 +1,173 @@
|
||||
//! Every file of a corpus that netCDF-C opens, read by clawhdf5-netcdf4 and
|
||||
//! compared with what netCDF-C reports: groups (order), dimensions (names,
|
||||
//! lengths, unlimited, order), variables (names, order, types, dimensions,
|
||||
//! shapes) and the values of numeric variables of at most 5000 elements.
|
||||
//!
|
||||
//! Gated: set `CLAWHDF5_NETCDF_CORPUS` to a directory (relative to the
|
||||
//! workspace root, or absolute) — the conformance corpus
|
||||
//! (`conformance/.cache/corpus`) or any tree of HDF5/netCDF-4 files — and
|
||||
//! `CLAWHDF5_PYTHON` to a Python with netCDF4-python (whose bundled
|
||||
//! libnetcdf `tests/netcdf_c_view.py` calls). Without the variable the test
|
||||
//! does nothing. netCDF-C reads each file in its own process (4 at a time,
|
||||
//! `CLAWHDF5_NETCDF_JOBS`), under a 60 s timeout and a 4 GiB address-space
|
||||
//! limit: 40 s to 3 minutes for the conformance corpus on tank.
|
||||
//! `CLAWHDF5_NETCDF_VIEW` may name the saved output of an earlier
|
||||
//! `netcdf_c_view.py --list <file of paths>` run over the same (absolute)
|
||||
//! paths, to skip that.
|
||||
//!
|
||||
//! A file whose differences are explained is listed, with the reason, in
|
||||
//! `tests/corpus_known_differences.txt` (and in `docs/known-issues.md`); a
|
||||
//! listed file that matches fails the test too, so the list stays true.
|
||||
//!
|
||||
//! ```sh
|
||||
//! CLAWHDF5_NETCDF_CORPUS=conformance/.cache/corpus CLAWHDF5_PYTHON=$PWD/.venv/bin/python \
|
||||
//! cargo test -p clawhdf5-netcdf4 --test corpus_vs_netcdf_c -- --nocapture
|
||||
//! ```
|
||||
|
||||
mod common;
|
||||
|
||||
use std::collections::BTreeMap;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::atomic::Ordering;
|
||||
|
||||
use common::{FILTER_SKIPPED, RELABELLED, differences, netcdf_c_views, parse_views};
|
||||
|
||||
/// The file extensions `conformance/list_files.py` sweeps.
|
||||
const EXTS: [&str; 7] = ["h5", "hdf5", "he5", "nc", "nc4", "hdf", "h5f"];
|
||||
|
||||
/// The files of the corpus: those with an HDF5/netCDF-4 extension, except
|
||||
/// netCDF classic files (magic `CDF`), following directory symlinks, in
|
||||
/// byte order of their paths.
|
||||
fn corpus_files(root: &Path) -> Vec<PathBuf> {
|
||||
fn walk(dir: &Path, out: &mut Vec<PathBuf>, depth: usize) {
|
||||
let Ok(entries) = std::fs::read_dir(dir) else {
|
||||
return;
|
||||
};
|
||||
for entry in entries.flatten() {
|
||||
let path = entry.path();
|
||||
if path.file_name().is_some_and(|n| n == ".git") {
|
||||
continue;
|
||||
}
|
||||
let Ok(meta) = std::fs::metadata(&path) else {
|
||||
continue;
|
||||
};
|
||||
if meta.is_dir() && depth < 32 {
|
||||
walk(&path, out, depth + 1);
|
||||
} else if meta.is_file()
|
||||
&& path
|
||||
.extension()
|
||||
.and_then(|e| e.to_str())
|
||||
.is_some_and(|e| EXTS.contains(&e.to_ascii_lowercase().as_str()))
|
||||
{
|
||||
let mut magic = [0u8; 3];
|
||||
let classic = std::fs::File::open(&path)
|
||||
.and_then(|mut f| std::io::Read::read_exact(&mut f, &mut magic))
|
||||
.is_ok()
|
||||
&& &magic == b"CDF";
|
||||
if !classic {
|
||||
out.push(path);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
let mut out = Vec::new();
|
||||
walk(root, &mut out, 0);
|
||||
out.sort_by(|a, b| {
|
||||
a.as_os_str()
|
||||
.as_encoded_bytes()
|
||||
.cmp(b.as_os_str().as_encoded_bytes())
|
||||
});
|
||||
out
|
||||
}
|
||||
|
||||
/// The files `tests/corpus_known_differences.txt` explains: path relative
|
||||
/// to the corpus root → reason.
|
||||
fn known_differences() -> BTreeMap<String, String> {
|
||||
let list = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/corpus_known_differences.txt");
|
||||
std::fs::read_to_string(list)
|
||||
.expect("tests/corpus_known_differences.txt")
|
||||
.lines()
|
||||
.filter(|l| !l.trim().is_empty() && !l.starts_with('#'))
|
||||
.map(|l| {
|
||||
let (path, reason) = l.split_once('\t').unwrap_or((l, ""));
|
||||
(path.trim().to_string(), reason.trim().to_string())
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn corpus_matches_netcdf_c() {
|
||||
let Ok(root) = std::env::var("CLAWHDF5_NETCDF_CORPUS") else {
|
||||
eprintln!("SKIP: set CLAWHDF5_NETCDF_CORPUS to a corpus directory");
|
||||
return;
|
||||
};
|
||||
// A relative path is from the workspace root (cargo runs the test in
|
||||
// the crate's directory).
|
||||
let root = Path::new(env!("CARGO_MANIFEST_DIR"))
|
||||
.join("../..")
|
||||
.join(root);
|
||||
let root = std::fs::canonicalize(&root).unwrap_or(root);
|
||||
let files = corpus_files(&root);
|
||||
assert!(!files.is_empty(), "no files under {}", root.display());
|
||||
// `CLAWHDF5_NETCDF_VIEW`: the output of an earlier run of
|
||||
// `tests/netcdf_c_view.py` over the same files, which takes a few
|
||||
// minutes over the conformance corpus.
|
||||
let views = match std::env::var("CLAWHDF5_NETCDF_VIEW") {
|
||||
Ok(saved) => parse_views(&std::fs::read_to_string(saved).expect("CLAWHDF5_NETCDF_VIEW")),
|
||||
Err(_) => netcdf_c_views(&files),
|
||||
};
|
||||
let known = known_differences();
|
||||
|
||||
let (mut opened, mut matched) = (0, 0);
|
||||
let mut explained = Vec::new();
|
||||
let mut unexplained = Vec::new();
|
||||
let mut stale = Vec::new();
|
||||
for path in &files {
|
||||
let Some(Some(want)) = views.get(path) else {
|
||||
continue;
|
||||
};
|
||||
opened += 1;
|
||||
let rel = path
|
||||
.strip_prefix(&root)
|
||||
.unwrap_or(path)
|
||||
.to_string_lossy()
|
||||
.into_owned();
|
||||
let diffs = differences(path, want);
|
||||
match (diffs.is_empty(), known.get(&rel)) {
|
||||
(true, None) => matched += 1,
|
||||
(true, Some(_)) => stale.push(rel),
|
||||
(false, Some(reason)) => explained.push((rel, reason.clone(), diffs)),
|
||||
(false, None) => unexplained.push((rel, diffs)),
|
||||
}
|
||||
}
|
||||
eprintln!(
|
||||
"{} files; netCDF-C opens {opened}; {matched} match; {} differ as explained; \
|
||||
{} differ unexplained; {} listed but match; {} variables relabelled \
|
||||
(floats of other than 4 or 8 bytes); {} variables' values not compared \
|
||||
(filter not in this build)",
|
||||
files.len(),
|
||||
explained.len(),
|
||||
unexplained.len(),
|
||||
stale.len(),
|
||||
RELABELLED.load(Ordering::Relaxed),
|
||||
FILTER_SKIPPED.load(Ordering::Relaxed),
|
||||
);
|
||||
for (rel, reason, diffs) in &explained {
|
||||
eprintln!("explained: {rel} ({reason}): {} differences", diffs.len());
|
||||
}
|
||||
for (rel, diffs) in &unexplained {
|
||||
eprintln!("DIFFERS: {rel}");
|
||||
for d in diffs.iter().take(20) {
|
||||
eprintln!(" {d}");
|
||||
}
|
||||
}
|
||||
assert!(
|
||||
unexplained.is_empty(),
|
||||
"{} files differ from netCDF-C unexplained",
|
||||
unexplained.len()
|
||||
);
|
||||
assert!(
|
||||
stale.is_empty(),
|
||||
"listed in corpus_known_differences.txt but match: {stale:?}"
|
||||
);
|
||||
}
|
||||
@@ -2,6 +2,8 @@
|
||||
//!
|
||||
//! Tests are skipped if python3 or netCDF4/xarray Python packages are not available.
|
||||
|
||||
mod common;
|
||||
|
||||
use std::process::Command;
|
||||
|
||||
use clawhdf5_netcdf4::{AttrValue, NcType, NetCDF4File};
|
||||
@@ -842,3 +844,267 @@ ds.to_netcdf({path:?}, engine={engine:?}, unlimited_dims=["time"])
|
||||
assert_same_view(&path);
|
||||
}
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// Compared with netCDF-C itself (tests/netcdf_c_view.py): groups,
|
||||
// dimensions, variables, types and values, in netCDF-C's order
|
||||
// ===========================================================================
|
||||
|
||||
/// The dimensions of the variable `name` (a path) of `file`.
|
||||
fn dim_names(file: &NetCDF4File, name: &str) -> Vec<String> {
|
||||
file.variable(name)
|
||||
.unwrap()
|
||||
.dimensions()
|
||||
.iter()
|
||||
.map(|d| d.name.clone())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// An h5py file without dimension scales: netCDF-C's phony dimensions,
|
||||
/// numbered file-wide (subgroups before their parent's variables, each
|
||||
/// group's datasets in name order, as h5py does not track creation order),
|
||||
/// shared by length within a group — but not between two axes of one
|
||||
/// variable, nor between a fixed and an unlimited axis — with a length of 0
|
||||
/// always unlimited and never shared with a fixed axis, and the real
|
||||
/// dimension of a scale taken by length too.
|
||||
#[test]
|
||||
fn phony_dimensions_match_netcdf_c() {
|
||||
skip_if_no_netcdf4!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("phony.h5");
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py
|
||||
import numpy as np
|
||||
with h5py.File({path:?}, "w") as f:
|
||||
f["zz"] = np.arange(12.0).reshape(2, 2, 3)
|
||||
f["aa"] = np.arange(6, dtype="i4").reshape(3, 2)
|
||||
f.create_dataset("un", data=np.ones((2, 3), "f4"), maxshape=(None, 3))
|
||||
f.create_dataset("un2", data=np.ones(2, "f4"), maxshape=(None,))
|
||||
f["zero"] = np.zeros((0,))
|
||||
f.create_dataset("zero_un", (0,), "f4", maxshape=(None,))
|
||||
f["zero2"] = np.zeros((0, 2))
|
||||
f["s"] = 1.5
|
||||
g = f.create_group("g")
|
||||
g["x"] = np.arange(5, dtype="i2")
|
||||
g["y"] = np.arange(2, dtype="u1")
|
||||
g.create_group("h")["q"] = np.arange(7.0)
|
||||
f.create_group("b")["w"] = np.arange(18, dtype="i8").reshape(2, 9)
|
||||
f["sc"] = np.arange(6.0)
|
||||
f["sc"].make_scale("sc")
|
||||
f["second_only"] = np.zeros((2, 6))
|
||||
f["second_only"].dims[1].attach_scale(f["sc"])
|
||||
"#,
|
||||
path = path.display().to_string()
|
||||
));
|
||||
common::assert_matches_netcdf_c(&path);
|
||||
|
||||
let file = NetCDF4File::open(&path).unwrap();
|
||||
// `sc` is dimension 0; /b gets 1 and 2, /g/h 3, /g 4 and 5, / from 6.
|
||||
assert_eq!(dim_names(&file, "b/w"), ["phony_dim_1", "phony_dim_2"]);
|
||||
assert_eq!(dim_names(&file, "g/h/q"), ["phony_dim_3"]);
|
||||
let h = file.group("g/h").unwrap();
|
||||
assert_eq!(h.variable_names().unwrap(), ["q"]);
|
||||
assert_eq!(h.dimensions().unwrap()[0].name, "phony_dim_3");
|
||||
assert_eq!(dim_names(&file, "aa"), ["phony_dim_6", "phony_dim_7"]);
|
||||
assert_eq!(
|
||||
dim_names(&file, "zz"),
|
||||
["phony_dim_7", "phony_dim_11", "phony_dim_6"]
|
||||
);
|
||||
assert_eq!(dim_names(&file, "second_only"), ["phony_dim_7", "sc"]);
|
||||
assert_eq!(dim_names(&file, "un"), ["phony_dim_8", "phony_dim_6"]);
|
||||
assert_eq!(dim_names(&file, "un2"), ["phony_dim_8"]);
|
||||
assert_eq!(dim_names(&file, "zero"), ["phony_dim_9"]);
|
||||
assert_eq!(dim_names(&file, "zero_un"), ["phony_dim_9"]);
|
||||
assert_eq!(dim_names(&file, "zero2"), ["phony_dim_10", "phony_dim_7"]);
|
||||
let root: Vec<(String, u64, bool)> = file
|
||||
.dimensions()
|
||||
.unwrap()
|
||||
.into_iter()
|
||||
.map(|d| (d.name, d.size, d.is_unlimited))
|
||||
.collect();
|
||||
assert_eq!(root[0], ("sc".to_string(), 6, false));
|
||||
assert!(root.contains(&("phony_dim_8".to_string(), 2, true)));
|
||||
assert!(root.contains(&("phony_dim_9".to_string(), 0, true)));
|
||||
assert!(root.contains(&("phony_dim_10".to_string(), 0, true)));
|
||||
assert_eq!(
|
||||
file.variable_names().unwrap(),
|
||||
[
|
||||
"aa",
|
||||
"s",
|
||||
"sc",
|
||||
"second_only",
|
||||
"un",
|
||||
"un2",
|
||||
"zero",
|
||||
"zero2",
|
||||
"zero_un",
|
||||
"zz"
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
/// Datasets of types netCDF-C cannot represent are not variables:
|
||||
/// references, bit fields, array types, a compound with a reference member
|
||||
/// or a half-float member, a compound nesting a compound not seen before;
|
||||
/// enum, compound, variable-length and opaque types are variables of those
|
||||
/// classes. netCDF-C also remembers a type it failed to read, so the second
|
||||
/// dataset of a compound with a reference member is a variable, and a
|
||||
/// nested compound is accepted once a dataset of the inner type was read.
|
||||
#[test]
|
||||
fn types_netcdf_c_skips_are_not_variables() {
|
||||
skip_if_no_netcdf4!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("types.h5");
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py
|
||||
import numpy as np
|
||||
inner = np.dtype([("x", "i2"), ("y", "f4")])
|
||||
with h5py.File({path:?}, "w") as f:
|
||||
f["i4"] = np.arange(3, dtype="i4")
|
||||
f["f2"] = np.arange(3, dtype="f2")
|
||||
f["s1"] = np.array([b"a", b"b"], dtype="S1")
|
||||
f["s5"] = np.array([b"abc", b"b"], dtype="S5")
|
||||
f["vs"] = np.array(["x", "yy"], dtype=h5py.string_dtype())
|
||||
f["bool"] = np.array([True, False])
|
||||
f["cmp"] = np.array([(1, 2.0)], dtype=[("a", "i4"), ("b", "f8")])
|
||||
f["en"] = np.array([0, 1], dtype=h5py.enum_dtype({{"A": 0, "B": 1}}, basetype="i1"))
|
||||
d = f.create_dataset("vl", (2,), dtype=h5py.vlen_dtype("i4"))
|
||||
d[0] = [1, 2]
|
||||
d[1] = [3]
|
||||
f["op"] = np.array([b"ab", b"cd"], dtype="V2")
|
||||
f.create_dataset("ref", (1,), dtype=h5py.ref_dtype)[0] = f["i4"].ref
|
||||
f["cref1"] = np.array([(1, f["i4"].ref)], dtype=[("a", "i4"), ("r", h5py.ref_dtype)])
|
||||
f["cref2"] = np.array([(2, f["i4"].ref)], dtype=[("a", "i4"), ("r", h5py.ref_dtype)])
|
||||
f["cmp_f2"] = np.zeros(2, dtype=[("a", "f2")])
|
||||
f["in1"] = np.zeros(2, dtype=inner)
|
||||
f["nested"] = np.zeros(2, dtype=[("a", "i4"), ("in", inner)])
|
||||
f["a_nested"] = np.zeros(2, dtype=[("b", "i4"), ("in", np.dtype([("p", "i1")]))])
|
||||
sid = h5py.h5s.create_simple((2,))
|
||||
h5py.h5d.create(f.id, b"bitf", h5py.h5t.STD_B8LE.copy(), sid)
|
||||
h5py.h5d.create(f.id, b"arr", h5py.h5t.array_create(h5py.h5t.NATIVE_INT32, (3,)), sid)
|
||||
"#,
|
||||
path = path.display().to_string()
|
||||
));
|
||||
common::assert_matches_netcdf_c(&path);
|
||||
|
||||
let file = NetCDF4File::open(&path).unwrap();
|
||||
assert_eq!(
|
||||
file.variable_names().unwrap(),
|
||||
[
|
||||
"bool", "cmp", "cref2", "en", "f2", "i4", "in1", "nested", "op", "s1", "s5", "vl", "vs"
|
||||
]
|
||||
);
|
||||
let nc_type = |name: &str| file.variable(name).unwrap().nc_type().unwrap();
|
||||
assert_eq!(nc_type("bool"), NcType::Enum);
|
||||
assert_eq!(nc_type("cmp"), NcType::Compound);
|
||||
assert_eq!(nc_type("vl"), NcType::VLen);
|
||||
assert_eq!(nc_type("op"), NcType::Opaque);
|
||||
assert_eq!(nc_type("s1"), NcType::Char);
|
||||
assert_eq!(nc_type("s5"), NcType::String);
|
||||
// netCDF-C 4.9.3 on libhdf5 1.14.6 says NC_STRING (deliberate
|
||||
// difference, see the README).
|
||||
assert_eq!(nc_type("f2"), NcType::Float);
|
||||
assert_eq!(
|
||||
file.variable("f2").unwrap().read_raw_f32().unwrap(),
|
||||
[0.0, 1.0, 2.0]
|
||||
);
|
||||
assert!(matches!(
|
||||
file.variable("ref"),
|
||||
Err(clawhdf5_netcdf4::Error::VariableNotFound(_))
|
||||
));
|
||||
}
|
||||
|
||||
/// The order of groups and variables: creation order where the group
|
||||
/// tracks it (every netCDF-4 file; also h5py with `track_order`), compact
|
||||
/// or dense (more than 8 links), else name order (h5py by default).
|
||||
#[test]
|
||||
fn group_and_variable_order_match_netcdf_c() {
|
||||
skip_if_no_netcdf4!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let nc = dir.path().join("order.nc");
|
||||
let h5 = dir.path().join("order.h5");
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py
|
||||
import netCDF4 as nc
|
||||
import numpy as np
|
||||
names = ["zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"]
|
||||
with nc.Dataset({nc:?}, "w") as f:
|
||||
f.createDimension("x", 2)
|
||||
for i, n in enumerate(names):
|
||||
f.createVariable(n, "i4", ("x",))[:] = [i, i + 1]
|
||||
for n in ["gz", "ga", "gm"]:
|
||||
f.createGroup(n).createVariable("v", "f8", ("x",))[:] = [1, 2]
|
||||
small = f.createGroup("small")
|
||||
for n in ["q", "b", "p"]:
|
||||
small.createVariable(n, "i2", ("x",))[:] = [3, 4]
|
||||
with h5py.File({h5:?}, "w") as f:
|
||||
for i, n in enumerate(names):
|
||||
f[n] = np.arange(i + 1)
|
||||
t = f.create_group("tracked", track_order=True)
|
||||
for i, n in enumerate(names):
|
||||
t[n] = np.arange(3, dtype="i1")
|
||||
for n in ["gz", "ga"]:
|
||||
f.create_group(n)["v"] = np.arange(4.0)
|
||||
"#,
|
||||
nc = nc.display().to_string(),
|
||||
h5 = h5.display().to_string()
|
||||
));
|
||||
common::assert_matches_netcdf_c(&nc);
|
||||
common::assert_matches_netcdf_c(&h5);
|
||||
|
||||
let file = NetCDF4File::open(&nc).unwrap();
|
||||
assert_eq!(
|
||||
file.variable_names().unwrap(),
|
||||
[
|
||||
"zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"
|
||||
]
|
||||
);
|
||||
assert_eq!(file.group_names().unwrap(), ["gz", "ga", "gm", "small"]);
|
||||
let file = NetCDF4File::open(&h5).unwrap();
|
||||
assert_eq!(file.group_names().unwrap(), ["ga", "gz", "tracked"]);
|
||||
assert_eq!(
|
||||
file.variable_names().unwrap(),
|
||||
[
|
||||
"a", "alpha", "beta", "c", "gamma", "k", "mid", "omega", "zeta", "zz"
|
||||
]
|
||||
);
|
||||
assert_eq!(
|
||||
file.group("tracked").unwrap().variable_names().unwrap(),
|
||||
[
|
||||
"zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
/// The files of the earlier tests, compared with netCDF-C itself too:
|
||||
/// dimension scales of h5py, netCDF4-python files with groups, unlimited
|
||||
/// dimensions and non-coordinate variables named like a dimension.
|
||||
#[test]
|
||||
fn netcdf4_python_files_match_netcdf_c() {
|
||||
skip_if_no_netcdf4!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("mixed.nc");
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import netCDF4 as nc
|
||||
import numpy as np
|
||||
with nc.Dataset({path:?}, "w") as f:
|
||||
f.createDimension("time", None)
|
||||
f.createDimension("p", 2)
|
||||
f.createDimension("q", 2)
|
||||
f.createVariable("a", "i4", ("time",))[0:2] = [1, 2]
|
||||
f.createVariable("b", "f4", ("time", "q"))[0:5, :] = np.arange(10).reshape(5, 2)
|
||||
f.createVariable("q", "f4", ("q",))[:] = [0, 1]
|
||||
f.createVariable("p", "f4", ("q", "p"))[:] = np.array([[0, 1], [2, 3]])
|
||||
g = f.createGroup("g")
|
||||
g.createDimension("r", 3)
|
||||
g.createVariable("w", "i4", ("r", "time"))[:, 0:1] = np.ones((3, 1))
|
||||
g.createGroup("h").createVariable("z", "i8", ("p", "r"))[:] = np.arange(6).reshape(2, 3)
|
||||
"#,
|
||||
path = path.display().to_string()
|
||||
));
|
||||
common::assert_matches_netcdf_c(&path);
|
||||
}
|
||||
|
||||
@@ -776,9 +776,12 @@ fn test_pure_dimensions_hidden_and_non_coord_names() {
|
||||
.with_shape(&[2]);
|
||||
let file = NetCDF4File::from_bytes(b.finish().unwrap()).unwrap();
|
||||
|
||||
// `x`, and the phony dimension netCDF-C gives the 3 values of
|
||||
// `_nc4_non_coord_x` (no dimension scale attached).
|
||||
let dims = file.dimensions().unwrap();
|
||||
assert_eq!(dims.len(), 1);
|
||||
assert_eq!(dims[0].name, "x");
|
||||
let names: Vec<&str> = dims.iter().map(|d| d.name.as_str()).collect();
|
||||
assert_eq!(names, ["x", "phony_dim_1"]);
|
||||
assert_eq!(dims[1].size, 3);
|
||||
let mut names = file.variable_names().unwrap();
|
||||
names.sort();
|
||||
assert_eq!(names, ["v", "x"]);
|
||||
@@ -786,7 +789,8 @@ fn test_pure_dimensions_hidden_and_non_coord_names() {
|
||||
assert_eq!(x.name(), "x");
|
||||
assert_eq!(x.read_raw_f64().unwrap(), [1.0, 2.0, 3.0]);
|
||||
assert!(!x.is_coordinate());
|
||||
// No DIMENSION_LIST: `v` gets `x` by size, as before.
|
||||
// No DIMENSION_LIST: `v` gets the group's first dimension of its length
|
||||
// (netCDF-C's phony rule, which also takes real dimensions).
|
||||
assert_eq!(file.variable("v").unwrap().dimensions()[0].name, "x");
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,238 @@
|
||||
#!/usr/bin/env python3
|
||||
"""netcdf_c_view.py FILE...: print each file as netCDF-C sees it.
|
||||
|
||||
The metadata comes from netCDF-C itself (the libnetcdf that netCDF4-python
|
||||
bundles, called through ctypes), not from netCDF4-python's objects, which
|
||||
leave out variables of types netCDF-C supports but netCDF4-python does not
|
||||
(opaque, compounds of vlen strings, ...). Values come from netCDF4-python.
|
||||
Each file is read in its own process under a timeout, so a file that crashes
|
||||
or hangs libnetcdf only costs that file.
|
||||
|
||||
Output, per file, fields separated by tabs:
|
||||
|
||||
FILE <path>
|
||||
ERROR <message> netCDF-C cannot open it; nothing else
|
||||
G <group> every group, pre-order, children in
|
||||
netCDF-C's order ("/" is the root)
|
||||
D <group> <name> <length> <0|1> its dimensions in dimension-id order
|
||||
(1: unlimited)
|
||||
V <group> <name> <type> <dims> <shape>
|
||||
its variables in netCDF-C's order;
|
||||
<dims> and <shape> comma-separated,
|
||||
"-" when there are none; <type> as
|
||||
clawhdf5_netcdf4::NcType prints it
|
||||
X <group> <name> <values> the values of a numeric variable of
|
||||
at most MAX_VALUES elements, as
|
||||
space-separated Python float reprs
|
||||
("-" when there are none)
|
||||
END
|
||||
|
||||
Used by tests/corpus_vs_netcdf_c.rs (CLAWHDF5_NETCDF_CORPUS).
|
||||
"""
|
||||
import concurrent.futures
|
||||
import ctypes
|
||||
import glob
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
MAX_VALUES = 5000
|
||||
TIMEOUT = 60
|
||||
MEMORY = 4 << 30 # address space of each file's process
|
||||
|
||||
ATOMIC = {
|
||||
1: "NC_BYTE", 2: "NC_CHAR", 3: "NC_SHORT", 4: "NC_INT", 5: "NC_FLOAT",
|
||||
6: "NC_DOUBLE", 7: "NC_UBYTE", 8: "NC_USHORT", 9: "NC_UINT",
|
||||
10: "NC_INT64", 11: "NC_UINT64", 12: "NC_STRING",
|
||||
}
|
||||
USER_CLASS = {13: "NC_VLEN", 14: "NC_OPAQUE", 15: "NC_ENUM", 16: "NC_COMPOUND"}
|
||||
NUMERIC = {"NC_BYTE", "NC_SHORT", "NC_INT", "NC_FLOAT", "NC_DOUBLE",
|
||||
"NC_UBYTE", "NC_USHORT", "NC_UINT", "NC_INT64", "NC_UINT64"}
|
||||
NAME = 257 # NC_MAX_NAME + 1
|
||||
MAXDIMS = 1024
|
||||
|
||||
|
||||
def libnetcdf():
|
||||
"""The libnetcdf netCDF4-python is linked with (same process, same copy)."""
|
||||
import netCDF4
|
||||
site = os.path.dirname(os.path.dirname(netCDF4.__file__))
|
||||
found = glob.glob(os.path.join(site, "netcdf4.libs", "libnetcdf*.so*")) + \
|
||||
glob.glob(os.path.join(site, "netCDF4.libs", "libnetcdf*.so*"))
|
||||
if found:
|
||||
return ctypes.CDLL(found[0])
|
||||
return ctypes.CDLL("libnetcdf.so")
|
||||
|
||||
|
||||
def view(path, out):
|
||||
nc = libnetcdf()
|
||||
ncid = ctypes.c_int()
|
||||
rc = nc.nc_open(path.encode(), 0, ctypes.byref(ncid)) # NC_NOWRITE
|
||||
if rc != 0:
|
||||
nc.nc_strerror.restype = ctypes.c_char_p
|
||||
out.append("ERROR\t" + nc.nc_strerror(rc).decode(errors="replace"))
|
||||
return
|
||||
numeric = [] # (group path, variable name)
|
||||
try:
|
||||
walk(nc, ncid.value, "/", out, numeric)
|
||||
finally:
|
||||
nc.nc_close(ncid)
|
||||
values(path, numeric, out)
|
||||
|
||||
|
||||
def check(rc, what):
|
||||
if rc != 0:
|
||||
raise RuntimeError(f"{what} failed: {rc}")
|
||||
|
||||
|
||||
def name_of(fn, *args):
|
||||
buf = ctypes.create_string_buffer(NAME)
|
||||
check(fn(*args, buf), fn.__name__)
|
||||
return buf.value.decode(errors="replace")
|
||||
|
||||
|
||||
def walk(nc, gid, path, out, numeric):
|
||||
out.append(f"G\t{path}")
|
||||
n = ctypes.c_int()
|
||||
ids = (ctypes.c_int * MAXDIMS)()
|
||||
check(nc.nc_inq_dimids(gid, ctypes.byref(n), ids, 0), "nc_inq_dimids")
|
||||
for dimid in ids[: n.value]:
|
||||
length = ctypes.c_size_t()
|
||||
dname = ctypes.create_string_buffer(NAME)
|
||||
check(nc.nc_inq_dim(gid, dimid, dname, ctypes.byref(length)), "nc_inq_dim")
|
||||
out.append(f"D\t{path}\t{dname.value.decode(errors='replace')}\t{length.value}\t"
|
||||
f"{int(is_unlimited(nc, gid, dimid))}")
|
||||
nvars = ctypes.c_int()
|
||||
varids = (ctypes.c_int * 65536)()
|
||||
check(nc.nc_inq_varids(gid, ctypes.byref(nvars), varids), "nc_inq_varids")
|
||||
for varid in varids[: nvars.value]:
|
||||
vname = ctypes.create_string_buffer(NAME)
|
||||
xtype = ctypes.c_int()
|
||||
ndims = ctypes.c_int()
|
||||
dimids = (ctypes.c_int * MAXDIMS)()
|
||||
natts = ctypes.c_int()
|
||||
check(nc.nc_inq_var(gid, varid, vname, ctypes.byref(xtype), ctypes.byref(ndims),
|
||||
dimids, ctypes.byref(natts)), "nc_inq_var")
|
||||
names, shape = [], []
|
||||
for dimid in dimids[: ndims.value]:
|
||||
dname = ctypes.create_string_buffer(NAME)
|
||||
length = ctypes.c_size_t()
|
||||
if nc.nc_inq_dim(gid, dimid, dname, ctypes.byref(length)) != 0:
|
||||
names.append("?")
|
||||
shape.append("?")
|
||||
continue
|
||||
names.append(dname.value.decode(errors="replace"))
|
||||
shape.append(str(length.value))
|
||||
tname = type_name(nc, gid, xtype.value)
|
||||
vn = vname.value.decode(errors="replace")
|
||||
out.append(f"V\t{path}\t{vn}\t{tname}\t{','.join(names) or '-'}\t{','.join(shape) or '-'}")
|
||||
if tname in NUMERIC and "?" not in shape:
|
||||
count = 1
|
||||
for s in shape:
|
||||
count *= int(s)
|
||||
if count <= MAX_VALUES:
|
||||
numeric.append((path, vn))
|
||||
ngrps = ctypes.c_int()
|
||||
grps = (ctypes.c_int * 65536)()
|
||||
check(nc.nc_inq_grps(gid, ctypes.byref(ngrps), grps), "nc_inq_grps")
|
||||
for child in grps[: ngrps.value]:
|
||||
cname = ctypes.create_string_buffer(NAME)
|
||||
check(nc.nc_inq_grpname(child, cname), "nc_inq_grpname")
|
||||
cpath = path.rstrip("/") + "/" + cname.value.decode(errors="replace")
|
||||
walk(nc, child, cpath, out, numeric)
|
||||
|
||||
|
||||
def is_unlimited(nc, gid, dimid):
|
||||
n = ctypes.c_int()
|
||||
ids = (ctypes.c_int * MAXDIMS)()
|
||||
# Unlimited dimensions visible from this group include its parents'.
|
||||
if nc.nc_inq_unlimdims(gid, ctypes.byref(n), ids) != 0:
|
||||
return False
|
||||
return dimid in ids[: n.value]
|
||||
|
||||
|
||||
def type_name(nc, gid, xtype):
|
||||
if xtype in ATOMIC:
|
||||
return ATOMIC[xtype]
|
||||
size = ctypes.c_size_t()
|
||||
base = ctypes.c_int()
|
||||
nfields = ctypes.c_size_t()
|
||||
klass = ctypes.c_int()
|
||||
tname = ctypes.create_string_buffer(NAME)
|
||||
if nc.nc_inq_user_type(gid, xtype, tname, ctypes.byref(size), ctypes.byref(base),
|
||||
ctypes.byref(nfields), ctypes.byref(klass)) != 0:
|
||||
return f"type{xtype}"
|
||||
return USER_CLASS.get(klass.value, f"class{klass.value}")
|
||||
|
||||
|
||||
def values(path, numeric, out):
|
||||
if not numeric:
|
||||
return
|
||||
import numpy as np
|
||||
import netCDF4
|
||||
try:
|
||||
ds = netCDF4.Dataset(path)
|
||||
except Exception: # noqa: BLE001 - netCDF4-python refuses some files netCDF-C opens
|
||||
return
|
||||
with ds:
|
||||
for gpath, name in numeric:
|
||||
try:
|
||||
group = ds if gpath == "/" else ds[gpath]
|
||||
var = group.variables[name]
|
||||
var.set_auto_maskandscale(False)
|
||||
# A whole-variable read through netCDF-C 4.9.3 lays out a
|
||||
# variable shorter than an unlimited dimension that is not its
|
||||
# first wrongly (written values first); reads of one index of
|
||||
# the leading axis are right.
|
||||
if var.ndim >= 2:
|
||||
data = np.stack([np.asarray(var[i]) for i in range(var.shape[0])]) \
|
||||
if var.shape[0] else np.zeros(var.shape)
|
||||
else:
|
||||
data = np.asarray(var[...])
|
||||
flat = np.asarray(data, dtype=np.float64).ravel()
|
||||
except Exception: # noqa: BLE001
|
||||
continue
|
||||
vals = " ".join(repr(float(v)) for v in flat) or "-"
|
||||
out.append(f"X\t{gpath}\t{name}\t{vals}")
|
||||
|
||||
|
||||
def limit_memory():
|
||||
import resource
|
||||
resource.setrlimit(resource.RLIMIT_AS, (MEMORY, MEMORY))
|
||||
|
||||
|
||||
def one(path):
|
||||
"""Run view() for one file in a child process."""
|
||||
try:
|
||||
proc = subprocess.run([sys.executable, __file__, "--one", path],
|
||||
capture_output=True, timeout=TIMEOUT, text=True,
|
||||
preexec_fn=limit_memory)
|
||||
except subprocess.TimeoutExpired:
|
||||
return [f"FILE\t{path}", "ERROR\ttimeout", "END"]
|
||||
lines = proc.stdout.splitlines()
|
||||
if proc.returncode != 0 or not lines or lines[-1] != "END":
|
||||
err = (proc.stderr.strip().splitlines() or [f"exit {proc.returncode}"])[-1]
|
||||
return [f"FILE\t{path}", f"ERROR\tchild failed: {err}", "END"]
|
||||
return lines
|
||||
|
||||
|
||||
def main(argv):
|
||||
if argv and argv[0] == "--one":
|
||||
out = [f"FILE\t{argv[1]}"]
|
||||
try:
|
||||
view(argv[1], out)
|
||||
except Exception as e: # noqa: BLE001
|
||||
out = [f"FILE\t{argv[1]}", f"ERROR\t{e}"]
|
||||
out.append("END")
|
||||
sys.stdout.write("\n".join(out) + "\n")
|
||||
return
|
||||
if argv and argv[0] == "--list":
|
||||
with open(argv[1]) as fh:
|
||||
argv = [line.rstrip("\n") for line in fh if line.strip()]
|
||||
jobs = int(os.environ.get("CLAWHDF5_NETCDF_JOBS", "4"))
|
||||
with concurrent.futures.ThreadPoolExecutor(jobs) as pool:
|
||||
for lines in pool.map(one, argv):
|
||||
sys.stdout.write("\n".join(lines) + "\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main(sys.argv[1:])
|
||||
Reference in New Issue
Block a user