clawhdf5-netcdf4: phony dimensions, skipped types and order as netCDF-C
Read a file's metadata the way netCDF-C 4.9.3 does (libhdf5/hdf5open.c), for the whole file on first use (src/model.rs, replacing src/scope.rs): - links in creation order when the group tracks it, else name order; a group's datasets before its subgroups; dimension ids file-wide; - variables' dimensions from _Netcdf4Coordinates (file-wide ids), else the scales DIMENSION_LIST attaches when the first axis has one, else netCDF-C's phony dimensions phony_dim_<id> (create_phony_dims: shared by length and unlimitedness within a group, not between two axes of one variable, numbered subgroups first, a zero length unlimited); - datasets of types netCDF-C cannot represent are not variables (references, bit fields, time, arrays, compounds/enums/VLENs over them), replaying netCDF-C's file-wide type list, failed types included; - unlimited lengths as nc4_find_dim_len (its group and below). NcType gains Enum, Compound, VLen, Opaque and is #[non_exhaustive]; Variable::nc_type is netCDF-C's type (1-byte strings NC_CHAR). New clawhdf5_format::group_v2::links_in_creation_order_in. Tests compare with netCDF-C itself (tests/netcdf_c_view.py calls the libnetcdf netCDF4-python bundles through ctypes): new interop cases for h5py files without dimension scales, every type class, link order; and the gated corpus_vs_netcdf_c (CLAWHDF5_NETCDF_CORPUS): 420 of the 429 conformance-corpus files netCDF-C opens match (main: 68); the other 9 are explained in tests/corpus_known_differences.txt and known-issues. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -2,6 +2,8 @@
|
||||
//!
|
||||
//! Tests are skipped if python3 or netCDF4/xarray Python packages are not available.
|
||||
|
||||
mod common;
|
||||
|
||||
use std::process::Command;
|
||||
|
||||
use clawhdf5_netcdf4::{AttrValue, NcType, NetCDF4File};
|
||||
@@ -842,3 +844,267 @@ ds.to_netcdf({path:?}, engine={engine:?}, unlimited_dims=["time"])
|
||||
assert_same_view(&path);
|
||||
}
|
||||
}
|
||||
|
||||
// ===========================================================================
|
||||
// Compared with netCDF-C itself (tests/netcdf_c_view.py): groups,
|
||||
// dimensions, variables, types and values, in netCDF-C's order
|
||||
// ===========================================================================
|
||||
|
||||
/// The dimensions of the variable `name` (a path) of `file`.
|
||||
fn dim_names(file: &NetCDF4File, name: &str) -> Vec<String> {
|
||||
file.variable(name)
|
||||
.unwrap()
|
||||
.dimensions()
|
||||
.iter()
|
||||
.map(|d| d.name.clone())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// An h5py file without dimension scales: netCDF-C's phony dimensions,
|
||||
/// numbered file-wide (subgroups before their parent's variables, each
|
||||
/// group's datasets in name order, as h5py does not track creation order),
|
||||
/// shared by length within a group — but not between two axes of one
|
||||
/// variable, nor between a fixed and an unlimited axis — with a length of 0
|
||||
/// always unlimited and never shared with a fixed axis, and the real
|
||||
/// dimension of a scale taken by length too.
|
||||
#[test]
|
||||
fn phony_dimensions_match_netcdf_c() {
|
||||
skip_if_no_netcdf4!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("phony.h5");
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py
|
||||
import numpy as np
|
||||
with h5py.File({path:?}, "w") as f:
|
||||
f["zz"] = np.arange(12.0).reshape(2, 2, 3)
|
||||
f["aa"] = np.arange(6, dtype="i4").reshape(3, 2)
|
||||
f.create_dataset("un", data=np.ones((2, 3), "f4"), maxshape=(None, 3))
|
||||
f.create_dataset("un2", data=np.ones(2, "f4"), maxshape=(None,))
|
||||
f["zero"] = np.zeros((0,))
|
||||
f.create_dataset("zero_un", (0,), "f4", maxshape=(None,))
|
||||
f["zero2"] = np.zeros((0, 2))
|
||||
f["s"] = 1.5
|
||||
g = f.create_group("g")
|
||||
g["x"] = np.arange(5, dtype="i2")
|
||||
g["y"] = np.arange(2, dtype="u1")
|
||||
g.create_group("h")["q"] = np.arange(7.0)
|
||||
f.create_group("b")["w"] = np.arange(18, dtype="i8").reshape(2, 9)
|
||||
f["sc"] = np.arange(6.0)
|
||||
f["sc"].make_scale("sc")
|
||||
f["second_only"] = np.zeros((2, 6))
|
||||
f["second_only"].dims[1].attach_scale(f["sc"])
|
||||
"#,
|
||||
path = path.display().to_string()
|
||||
));
|
||||
common::assert_matches_netcdf_c(&path);
|
||||
|
||||
let file = NetCDF4File::open(&path).unwrap();
|
||||
// `sc` is dimension 0; /b gets 1 and 2, /g/h 3, /g 4 and 5, / from 6.
|
||||
assert_eq!(dim_names(&file, "b/w"), ["phony_dim_1", "phony_dim_2"]);
|
||||
assert_eq!(dim_names(&file, "g/h/q"), ["phony_dim_3"]);
|
||||
let h = file.group("g/h").unwrap();
|
||||
assert_eq!(h.variable_names().unwrap(), ["q"]);
|
||||
assert_eq!(h.dimensions().unwrap()[0].name, "phony_dim_3");
|
||||
assert_eq!(dim_names(&file, "aa"), ["phony_dim_6", "phony_dim_7"]);
|
||||
assert_eq!(
|
||||
dim_names(&file, "zz"),
|
||||
["phony_dim_7", "phony_dim_11", "phony_dim_6"]
|
||||
);
|
||||
assert_eq!(dim_names(&file, "second_only"), ["phony_dim_7", "sc"]);
|
||||
assert_eq!(dim_names(&file, "un"), ["phony_dim_8", "phony_dim_6"]);
|
||||
assert_eq!(dim_names(&file, "un2"), ["phony_dim_8"]);
|
||||
assert_eq!(dim_names(&file, "zero"), ["phony_dim_9"]);
|
||||
assert_eq!(dim_names(&file, "zero_un"), ["phony_dim_9"]);
|
||||
assert_eq!(dim_names(&file, "zero2"), ["phony_dim_10", "phony_dim_7"]);
|
||||
let root: Vec<(String, u64, bool)> = file
|
||||
.dimensions()
|
||||
.unwrap()
|
||||
.into_iter()
|
||||
.map(|d| (d.name, d.size, d.is_unlimited))
|
||||
.collect();
|
||||
assert_eq!(root[0], ("sc".to_string(), 6, false));
|
||||
assert!(root.contains(&("phony_dim_8".to_string(), 2, true)));
|
||||
assert!(root.contains(&("phony_dim_9".to_string(), 0, true)));
|
||||
assert!(root.contains(&("phony_dim_10".to_string(), 0, true)));
|
||||
assert_eq!(
|
||||
file.variable_names().unwrap(),
|
||||
[
|
||||
"aa",
|
||||
"s",
|
||||
"sc",
|
||||
"second_only",
|
||||
"un",
|
||||
"un2",
|
||||
"zero",
|
||||
"zero2",
|
||||
"zero_un",
|
||||
"zz"
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
/// Datasets of types netCDF-C cannot represent are not variables:
|
||||
/// references, bit fields, array types, a compound with a reference member
|
||||
/// or a half-float member, a compound nesting a compound not seen before;
|
||||
/// enum, compound, variable-length and opaque types are variables of those
|
||||
/// classes. netCDF-C also remembers a type it failed to read, so the second
|
||||
/// dataset of a compound with a reference member is a variable, and a
|
||||
/// nested compound is accepted once a dataset of the inner type was read.
|
||||
#[test]
|
||||
fn types_netcdf_c_skips_are_not_variables() {
|
||||
skip_if_no_netcdf4!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("types.h5");
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py
|
||||
import numpy as np
|
||||
inner = np.dtype([("x", "i2"), ("y", "f4")])
|
||||
with h5py.File({path:?}, "w") as f:
|
||||
f["i4"] = np.arange(3, dtype="i4")
|
||||
f["f2"] = np.arange(3, dtype="f2")
|
||||
f["s1"] = np.array([b"a", b"b"], dtype="S1")
|
||||
f["s5"] = np.array([b"abc", b"b"], dtype="S5")
|
||||
f["vs"] = np.array(["x", "yy"], dtype=h5py.string_dtype())
|
||||
f["bool"] = np.array([True, False])
|
||||
f["cmp"] = np.array([(1, 2.0)], dtype=[("a", "i4"), ("b", "f8")])
|
||||
f["en"] = np.array([0, 1], dtype=h5py.enum_dtype({{"A": 0, "B": 1}}, basetype="i1"))
|
||||
d = f.create_dataset("vl", (2,), dtype=h5py.vlen_dtype("i4"))
|
||||
d[0] = [1, 2]
|
||||
d[1] = [3]
|
||||
f["op"] = np.array([b"ab", b"cd"], dtype="V2")
|
||||
f.create_dataset("ref", (1,), dtype=h5py.ref_dtype)[0] = f["i4"].ref
|
||||
f["cref1"] = np.array([(1, f["i4"].ref)], dtype=[("a", "i4"), ("r", h5py.ref_dtype)])
|
||||
f["cref2"] = np.array([(2, f["i4"].ref)], dtype=[("a", "i4"), ("r", h5py.ref_dtype)])
|
||||
f["cmp_f2"] = np.zeros(2, dtype=[("a", "f2")])
|
||||
f["in1"] = np.zeros(2, dtype=inner)
|
||||
f["nested"] = np.zeros(2, dtype=[("a", "i4"), ("in", inner)])
|
||||
f["a_nested"] = np.zeros(2, dtype=[("b", "i4"), ("in", np.dtype([("p", "i1")]))])
|
||||
sid = h5py.h5s.create_simple((2,))
|
||||
h5py.h5d.create(f.id, b"bitf", h5py.h5t.STD_B8LE.copy(), sid)
|
||||
h5py.h5d.create(f.id, b"arr", h5py.h5t.array_create(h5py.h5t.NATIVE_INT32, (3,)), sid)
|
||||
"#,
|
||||
path = path.display().to_string()
|
||||
));
|
||||
common::assert_matches_netcdf_c(&path);
|
||||
|
||||
let file = NetCDF4File::open(&path).unwrap();
|
||||
assert_eq!(
|
||||
file.variable_names().unwrap(),
|
||||
[
|
||||
"bool", "cmp", "cref2", "en", "f2", "i4", "in1", "nested", "op", "s1", "s5", "vl", "vs"
|
||||
]
|
||||
);
|
||||
let nc_type = |name: &str| file.variable(name).unwrap().nc_type().unwrap();
|
||||
assert_eq!(nc_type("bool"), NcType::Enum);
|
||||
assert_eq!(nc_type("cmp"), NcType::Compound);
|
||||
assert_eq!(nc_type("vl"), NcType::VLen);
|
||||
assert_eq!(nc_type("op"), NcType::Opaque);
|
||||
assert_eq!(nc_type("s1"), NcType::Char);
|
||||
assert_eq!(nc_type("s5"), NcType::String);
|
||||
// netCDF-C 4.9.3 on libhdf5 1.14.6 says NC_STRING (deliberate
|
||||
// difference, see the README).
|
||||
assert_eq!(nc_type("f2"), NcType::Float);
|
||||
assert_eq!(
|
||||
file.variable("f2").unwrap().read_raw_f32().unwrap(),
|
||||
[0.0, 1.0, 2.0]
|
||||
);
|
||||
assert!(matches!(
|
||||
file.variable("ref"),
|
||||
Err(clawhdf5_netcdf4::Error::VariableNotFound(_))
|
||||
));
|
||||
}
|
||||
|
||||
/// The order of groups and variables: creation order where the group
|
||||
/// tracks it (every netCDF-4 file; also h5py with `track_order`), compact
|
||||
/// or dense (more than 8 links), else name order (h5py by default).
|
||||
#[test]
|
||||
fn group_and_variable_order_match_netcdf_c() {
|
||||
skip_if_no_netcdf4!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let nc = dir.path().join("order.nc");
|
||||
let h5 = dir.path().join("order.h5");
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import h5py
|
||||
import netCDF4 as nc
|
||||
import numpy as np
|
||||
names = ["zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"]
|
||||
with nc.Dataset({nc:?}, "w") as f:
|
||||
f.createDimension("x", 2)
|
||||
for i, n in enumerate(names):
|
||||
f.createVariable(n, "i4", ("x",))[:] = [i, i + 1]
|
||||
for n in ["gz", "ga", "gm"]:
|
||||
f.createGroup(n).createVariable("v", "f8", ("x",))[:] = [1, 2]
|
||||
small = f.createGroup("small")
|
||||
for n in ["q", "b", "p"]:
|
||||
small.createVariable(n, "i2", ("x",))[:] = [3, 4]
|
||||
with h5py.File({h5:?}, "w") as f:
|
||||
for i, n in enumerate(names):
|
||||
f[n] = np.arange(i + 1)
|
||||
t = f.create_group("tracked", track_order=True)
|
||||
for i, n in enumerate(names):
|
||||
t[n] = np.arange(3, dtype="i1")
|
||||
for n in ["gz", "ga"]:
|
||||
f.create_group(n)["v"] = np.arange(4.0)
|
||||
"#,
|
||||
nc = nc.display().to_string(),
|
||||
h5 = h5.display().to_string()
|
||||
));
|
||||
common::assert_matches_netcdf_c(&nc);
|
||||
common::assert_matches_netcdf_c(&h5);
|
||||
|
||||
let file = NetCDF4File::open(&nc).unwrap();
|
||||
assert_eq!(
|
||||
file.variable_names().unwrap(),
|
||||
[
|
||||
"zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"
|
||||
]
|
||||
);
|
||||
assert_eq!(file.group_names().unwrap(), ["gz", "ga", "gm", "small"]);
|
||||
let file = NetCDF4File::open(&h5).unwrap();
|
||||
assert_eq!(file.group_names().unwrap(), ["ga", "gz", "tracked"]);
|
||||
assert_eq!(
|
||||
file.variable_names().unwrap(),
|
||||
[
|
||||
"a", "alpha", "beta", "c", "gamma", "k", "mid", "omega", "zeta", "zz"
|
||||
]
|
||||
);
|
||||
assert_eq!(
|
||||
file.group("tracked").unwrap().variable_names().unwrap(),
|
||||
[
|
||||
"zeta", "alpha", "mid", "beta", "omega", "gamma", "k", "a", "zz", "c"
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
/// The files of the earlier tests, compared with netCDF-C itself too:
|
||||
/// dimension scales of h5py, netCDF4-python files with groups, unlimited
|
||||
/// dimensions and non-coordinate variables named like a dimension.
|
||||
#[test]
|
||||
fn netcdf4_python_files_match_netcdf_c() {
|
||||
skip_if_no_netcdf4!();
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = dir.path().join("mixed.nc");
|
||||
run_python(&format!(
|
||||
r#"
|
||||
import netCDF4 as nc
|
||||
import numpy as np
|
||||
with nc.Dataset({path:?}, "w") as f:
|
||||
f.createDimension("time", None)
|
||||
f.createDimension("p", 2)
|
||||
f.createDimension("q", 2)
|
||||
f.createVariable("a", "i4", ("time",))[0:2] = [1, 2]
|
||||
f.createVariable("b", "f4", ("time", "q"))[0:5, :] = np.arange(10).reshape(5, 2)
|
||||
f.createVariable("q", "f4", ("q",))[:] = [0, 1]
|
||||
f.createVariable("p", "f4", ("q", "p"))[:] = np.array([[0, 1], [2, 3]])
|
||||
g = f.createGroup("g")
|
||||
g.createDimension("r", 3)
|
||||
g.createVariable("w", "i4", ("r", "time"))[:, 0:1] = np.ones((3, 1))
|
||||
g.createGroup("h").createVariable("z", "i8", ("p", "r"))[:] = np.arange(6).reshape(2, 3)
|
||||
"#,
|
||||
path = path.display().to_string()
|
||||
));
|
||||
common::assert_matches_netcdf_c(&path);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user