Read a file's metadata the way netCDF-C 4.9.3 does (libhdf5/hdf5open.c), for the whole file on first use (src/model.rs, replacing src/scope.rs): - links in creation order when the group tracks it, else name order; a group's datasets before its subgroups; dimension ids file-wide; - variables' dimensions from _Netcdf4Coordinates (file-wide ids), else the scales DIMENSION_LIST attaches when the first axis has one, else netCDF-C's phony dimensions phony_dim_<id> (create_phony_dims: shared by length and unlimitedness within a group, not between two axes of one variable, numbered subgroups first, a zero length unlimited); - datasets of types netCDF-C cannot represent are not variables (references, bit fields, time, arrays, compounds/enums/VLENs over them), replaying netCDF-C's file-wide type list, failed types included; - unlimited lengths as nc4_find_dim_len (its group and below). NcType gains Enum, Compound, VLen, Opaque and is #[non_exhaustive]; Variable::nc_type is netCDF-C's type (1-byte strings NC_CHAR). New clawhdf5_format::group_v2::links_in_creation_order_in. Tests compare with netCDF-C itself (tests/netcdf_c_view.py calls the libnetcdf netCDF4-python bundles through ctypes): new interop cases for h5py files without dimension scales, every type class, link order; and the gated corpus_vs_netcdf_c (CLAWHDF5_NETCDF_CORPUS): 420 of the 429 conformance-corpus files netCDF-C opens match (main: 68); the other 9 are explained in tests/corpus_known_differences.txt and known-issues. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
239 lines
9.1 KiB
Python
239 lines
9.1 KiB
Python
#!/usr/bin/env python3
|
|
"""netcdf_c_view.py FILE...: print each file as netCDF-C sees it.
|
|
|
|
The metadata comes from netCDF-C itself (the libnetcdf that netCDF4-python
|
|
bundles, called through ctypes), not from netCDF4-python's objects, which
|
|
leave out variables of types netCDF-C supports but netCDF4-python does not
|
|
(opaque, compounds of vlen strings, ...). Values come from netCDF4-python.
|
|
Each file is read in its own process under a timeout, so a file that crashes
|
|
or hangs libnetcdf only costs that file.
|
|
|
|
Output, per file, fields separated by tabs:
|
|
|
|
FILE <path>
|
|
ERROR <message> netCDF-C cannot open it; nothing else
|
|
G <group> every group, pre-order, children in
|
|
netCDF-C's order ("/" is the root)
|
|
D <group> <name> <length> <0|1> its dimensions in dimension-id order
|
|
(1: unlimited)
|
|
V <group> <name> <type> <dims> <shape>
|
|
its variables in netCDF-C's order;
|
|
<dims> and <shape> comma-separated,
|
|
"-" when there are none; <type> as
|
|
clawhdf5_netcdf4::NcType prints it
|
|
X <group> <name> <values> the values of a numeric variable of
|
|
at most MAX_VALUES elements, as
|
|
space-separated Python float reprs
|
|
("-" when there are none)
|
|
END
|
|
|
|
Used by tests/corpus_vs_netcdf_c.rs (CLAWHDF5_NETCDF_CORPUS).
|
|
"""
|
|
import concurrent.futures
|
|
import ctypes
|
|
import glob
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
|
|
MAX_VALUES = 5000
|
|
TIMEOUT = 60
|
|
MEMORY = 4 << 30 # address space of each file's process
|
|
|
|
ATOMIC = {
|
|
1: "NC_BYTE", 2: "NC_CHAR", 3: "NC_SHORT", 4: "NC_INT", 5: "NC_FLOAT",
|
|
6: "NC_DOUBLE", 7: "NC_UBYTE", 8: "NC_USHORT", 9: "NC_UINT",
|
|
10: "NC_INT64", 11: "NC_UINT64", 12: "NC_STRING",
|
|
}
|
|
USER_CLASS = {13: "NC_VLEN", 14: "NC_OPAQUE", 15: "NC_ENUM", 16: "NC_COMPOUND"}
|
|
NUMERIC = {"NC_BYTE", "NC_SHORT", "NC_INT", "NC_FLOAT", "NC_DOUBLE",
|
|
"NC_UBYTE", "NC_USHORT", "NC_UINT", "NC_INT64", "NC_UINT64"}
|
|
NAME = 257 # NC_MAX_NAME + 1
|
|
MAXDIMS = 1024
|
|
|
|
|
|
def libnetcdf():
|
|
"""The libnetcdf netCDF4-python is linked with (same process, same copy)."""
|
|
import netCDF4
|
|
site = os.path.dirname(os.path.dirname(netCDF4.__file__))
|
|
found = glob.glob(os.path.join(site, "netcdf4.libs", "libnetcdf*.so*")) + \
|
|
glob.glob(os.path.join(site, "netCDF4.libs", "libnetcdf*.so*"))
|
|
if found:
|
|
return ctypes.CDLL(found[0])
|
|
return ctypes.CDLL("libnetcdf.so")
|
|
|
|
|
|
def view(path, out):
|
|
nc = libnetcdf()
|
|
ncid = ctypes.c_int()
|
|
rc = nc.nc_open(path.encode(), 0, ctypes.byref(ncid)) # NC_NOWRITE
|
|
if rc != 0:
|
|
nc.nc_strerror.restype = ctypes.c_char_p
|
|
out.append("ERROR\t" + nc.nc_strerror(rc).decode(errors="replace"))
|
|
return
|
|
numeric = [] # (group path, variable name)
|
|
try:
|
|
walk(nc, ncid.value, "/", out, numeric)
|
|
finally:
|
|
nc.nc_close(ncid)
|
|
values(path, numeric, out)
|
|
|
|
|
|
def check(rc, what):
|
|
if rc != 0:
|
|
raise RuntimeError(f"{what} failed: {rc}")
|
|
|
|
|
|
def name_of(fn, *args):
|
|
buf = ctypes.create_string_buffer(NAME)
|
|
check(fn(*args, buf), fn.__name__)
|
|
return buf.value.decode(errors="replace")
|
|
|
|
|
|
def walk(nc, gid, path, out, numeric):
|
|
out.append(f"G\t{path}")
|
|
n = ctypes.c_int()
|
|
ids = (ctypes.c_int * MAXDIMS)()
|
|
check(nc.nc_inq_dimids(gid, ctypes.byref(n), ids, 0), "nc_inq_dimids")
|
|
for dimid in ids[: n.value]:
|
|
length = ctypes.c_size_t()
|
|
dname = ctypes.create_string_buffer(NAME)
|
|
check(nc.nc_inq_dim(gid, dimid, dname, ctypes.byref(length)), "nc_inq_dim")
|
|
out.append(f"D\t{path}\t{dname.value.decode(errors='replace')}\t{length.value}\t"
|
|
f"{int(is_unlimited(nc, gid, dimid))}")
|
|
nvars = ctypes.c_int()
|
|
varids = (ctypes.c_int * 65536)()
|
|
check(nc.nc_inq_varids(gid, ctypes.byref(nvars), varids), "nc_inq_varids")
|
|
for varid in varids[: nvars.value]:
|
|
vname = ctypes.create_string_buffer(NAME)
|
|
xtype = ctypes.c_int()
|
|
ndims = ctypes.c_int()
|
|
dimids = (ctypes.c_int * MAXDIMS)()
|
|
natts = ctypes.c_int()
|
|
check(nc.nc_inq_var(gid, varid, vname, ctypes.byref(xtype), ctypes.byref(ndims),
|
|
dimids, ctypes.byref(natts)), "nc_inq_var")
|
|
names, shape = [], []
|
|
for dimid in dimids[: ndims.value]:
|
|
dname = ctypes.create_string_buffer(NAME)
|
|
length = ctypes.c_size_t()
|
|
if nc.nc_inq_dim(gid, dimid, dname, ctypes.byref(length)) != 0:
|
|
names.append("?")
|
|
shape.append("?")
|
|
continue
|
|
names.append(dname.value.decode(errors="replace"))
|
|
shape.append(str(length.value))
|
|
tname = type_name(nc, gid, xtype.value)
|
|
vn = vname.value.decode(errors="replace")
|
|
out.append(f"V\t{path}\t{vn}\t{tname}\t{','.join(names) or '-'}\t{','.join(shape) or '-'}")
|
|
if tname in NUMERIC and "?" not in shape:
|
|
count = 1
|
|
for s in shape:
|
|
count *= int(s)
|
|
if count <= MAX_VALUES:
|
|
numeric.append((path, vn))
|
|
ngrps = ctypes.c_int()
|
|
grps = (ctypes.c_int * 65536)()
|
|
check(nc.nc_inq_grps(gid, ctypes.byref(ngrps), grps), "nc_inq_grps")
|
|
for child in grps[: ngrps.value]:
|
|
cname = ctypes.create_string_buffer(NAME)
|
|
check(nc.nc_inq_grpname(child, cname), "nc_inq_grpname")
|
|
cpath = path.rstrip("/") + "/" + cname.value.decode(errors="replace")
|
|
walk(nc, child, cpath, out, numeric)
|
|
|
|
|
|
def is_unlimited(nc, gid, dimid):
|
|
n = ctypes.c_int()
|
|
ids = (ctypes.c_int * MAXDIMS)()
|
|
# Unlimited dimensions visible from this group include its parents'.
|
|
if nc.nc_inq_unlimdims(gid, ctypes.byref(n), ids) != 0:
|
|
return False
|
|
return dimid in ids[: n.value]
|
|
|
|
|
|
def type_name(nc, gid, xtype):
|
|
if xtype in ATOMIC:
|
|
return ATOMIC[xtype]
|
|
size = ctypes.c_size_t()
|
|
base = ctypes.c_int()
|
|
nfields = ctypes.c_size_t()
|
|
klass = ctypes.c_int()
|
|
tname = ctypes.create_string_buffer(NAME)
|
|
if nc.nc_inq_user_type(gid, xtype, tname, ctypes.byref(size), ctypes.byref(base),
|
|
ctypes.byref(nfields), ctypes.byref(klass)) != 0:
|
|
return f"type{xtype}"
|
|
return USER_CLASS.get(klass.value, f"class{klass.value}")
|
|
|
|
|
|
def values(path, numeric, out):
|
|
if not numeric:
|
|
return
|
|
import numpy as np
|
|
import netCDF4
|
|
try:
|
|
ds = netCDF4.Dataset(path)
|
|
except Exception: # noqa: BLE001 - netCDF4-python refuses some files netCDF-C opens
|
|
return
|
|
with ds:
|
|
for gpath, name in numeric:
|
|
try:
|
|
group = ds if gpath == "/" else ds[gpath]
|
|
var = group.variables[name]
|
|
var.set_auto_maskandscale(False)
|
|
# A whole-variable read through netCDF-C 4.9.3 lays out a
|
|
# variable shorter than an unlimited dimension that is not its
|
|
# first wrongly (written values first); reads of one index of
|
|
# the leading axis are right.
|
|
if var.ndim >= 2:
|
|
data = np.stack([np.asarray(var[i]) for i in range(var.shape[0])]) \
|
|
if var.shape[0] else np.zeros(var.shape)
|
|
else:
|
|
data = np.asarray(var[...])
|
|
flat = np.asarray(data, dtype=np.float64).ravel()
|
|
except Exception: # noqa: BLE001
|
|
continue
|
|
vals = " ".join(repr(float(v)) for v in flat) or "-"
|
|
out.append(f"X\t{gpath}\t{name}\t{vals}")
|
|
|
|
|
|
def limit_memory():
|
|
import resource
|
|
resource.setrlimit(resource.RLIMIT_AS, (MEMORY, MEMORY))
|
|
|
|
|
|
def one(path):
|
|
"""Run view() for one file in a child process."""
|
|
try:
|
|
proc = subprocess.run([sys.executable, __file__, "--one", path],
|
|
capture_output=True, timeout=TIMEOUT, text=True,
|
|
preexec_fn=limit_memory)
|
|
except subprocess.TimeoutExpired:
|
|
return [f"FILE\t{path}", "ERROR\ttimeout", "END"]
|
|
lines = proc.stdout.splitlines()
|
|
if proc.returncode != 0 or not lines or lines[-1] != "END":
|
|
err = (proc.stderr.strip().splitlines() or [f"exit {proc.returncode}"])[-1]
|
|
return [f"FILE\t{path}", f"ERROR\tchild failed: {err}", "END"]
|
|
return lines
|
|
|
|
|
|
def main(argv):
|
|
if argv and argv[0] == "--one":
|
|
out = [f"FILE\t{argv[1]}"]
|
|
try:
|
|
view(argv[1], out)
|
|
except Exception as e: # noqa: BLE001
|
|
out = [f"FILE\t{argv[1]}", f"ERROR\t{e}"]
|
|
out.append("END")
|
|
sys.stdout.write("\n".join(out) + "\n")
|
|
return
|
|
if argv and argv[0] == "--list":
|
|
with open(argv[1]) as fh:
|
|
argv = [line.rstrip("\n") for line in fh if line.strip()]
|
|
jobs = int(os.environ.get("CLAWHDF5_NETCDF_JOBS", "4"))
|
|
with concurrent.futures.ThreadPoolExecutor(jobs) as pool:
|
|
for lines in pool.map(one, argv):
|
|
sys.stdout.write("\n".join(lines) + "\n")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main(sys.argv[1:])
|