#!/usr/bin/env python3 """netcdf_c_view.py FILE...: print each file as netCDF-C sees it. The metadata comes from netCDF-C itself (the libnetcdf that netCDF4-python bundles, called through ctypes), not from netCDF4-python's objects, which leave out variables of types netCDF-C supports but netCDF4-python does not (opaque, compounds of vlen strings, ...). Values come from netCDF4-python. Each file is read in its own process under a timeout, so a file that crashes or hangs libnetcdf only costs that file. Output, per file, fields separated by tabs: FILE ERROR netCDF-C cannot open it; nothing else G every group, pre-order, children in netCDF-C's order ("/" is the root) D <0|1> its dimensions in dimension-id order (1: unlimited) V its variables in netCDF-C's order; and comma-separated, "-" when there are none; as clawhdf5_netcdf4::NcType prints it X the values of a numeric variable of at most MAX_VALUES elements, as space-separated Python float reprs ("-" when there are none) END Used by tests/corpus_vs_netcdf_c.rs (CLAWHDF5_NETCDF_CORPUS). """ import concurrent.futures import ctypes import glob import os import subprocess import sys MAX_VALUES = 5000 TIMEOUT = 60 MEMORY = 4 << 30 # address space of each file's process ATOMIC = { 1: "NC_BYTE", 2: "NC_CHAR", 3: "NC_SHORT", 4: "NC_INT", 5: "NC_FLOAT", 6: "NC_DOUBLE", 7: "NC_UBYTE", 8: "NC_USHORT", 9: "NC_UINT", 10: "NC_INT64", 11: "NC_UINT64", 12: "NC_STRING", } USER_CLASS = {13: "NC_VLEN", 14: "NC_OPAQUE", 15: "NC_ENUM", 16: "NC_COMPOUND"} NUMERIC = {"NC_BYTE", "NC_SHORT", "NC_INT", "NC_FLOAT", "NC_DOUBLE", "NC_UBYTE", "NC_USHORT", "NC_UINT", "NC_INT64", "NC_UINT64"} NAME = 257 # NC_MAX_NAME + 1 MAXDIMS = 1024 def libnetcdf(): """The libnetcdf netCDF4-python is linked with (same process, same copy).""" import netCDF4 site = os.path.dirname(os.path.dirname(netCDF4.__file__)) found = glob.glob(os.path.join(site, "netcdf4.libs", "libnetcdf*.so*")) + \ glob.glob(os.path.join(site, "netCDF4.libs", "libnetcdf*.so*")) if found: return ctypes.CDLL(found[0]) return ctypes.CDLL("libnetcdf.so") def view(path, out): nc = libnetcdf() ncid = ctypes.c_int() rc = nc.nc_open(path.encode(), 0, ctypes.byref(ncid)) # NC_NOWRITE if rc != 0: nc.nc_strerror.restype = ctypes.c_char_p out.append("ERROR\t" + nc.nc_strerror(rc).decode(errors="replace")) return numeric = [] # (group path, variable name) try: walk(nc, ncid.value, "/", out, numeric) finally: nc.nc_close(ncid) values(path, numeric, out) def check(rc, what): if rc != 0: raise RuntimeError(f"{what} failed: {rc}") def name_of(fn, *args): buf = ctypes.create_string_buffer(NAME) check(fn(*args, buf), fn.__name__) return buf.value.decode(errors="replace") def walk(nc, gid, path, out, numeric): out.append(f"G\t{path}") n = ctypes.c_int() ids = (ctypes.c_int * MAXDIMS)() check(nc.nc_inq_dimids(gid, ctypes.byref(n), ids, 0), "nc_inq_dimids") for dimid in ids[: n.value]: length = ctypes.c_size_t() dname = ctypes.create_string_buffer(NAME) check(nc.nc_inq_dim(gid, dimid, dname, ctypes.byref(length)), "nc_inq_dim") out.append(f"D\t{path}\t{dname.value.decode(errors='replace')}\t{length.value}\t" f"{int(is_unlimited(nc, gid, dimid))}") nvars = ctypes.c_int() varids = (ctypes.c_int * 65536)() check(nc.nc_inq_varids(gid, ctypes.byref(nvars), varids), "nc_inq_varids") for varid in varids[: nvars.value]: vname = ctypes.create_string_buffer(NAME) xtype = ctypes.c_int() ndims = ctypes.c_int() dimids = (ctypes.c_int * MAXDIMS)() natts = ctypes.c_int() check(nc.nc_inq_var(gid, varid, vname, ctypes.byref(xtype), ctypes.byref(ndims), dimids, ctypes.byref(natts)), "nc_inq_var") names, shape = [], [] for dimid in dimids[: ndims.value]: dname = ctypes.create_string_buffer(NAME) length = ctypes.c_size_t() if nc.nc_inq_dim(gid, dimid, dname, ctypes.byref(length)) != 0: names.append("?") shape.append("?") continue names.append(dname.value.decode(errors="replace")) shape.append(str(length.value)) tname = type_name(nc, gid, xtype.value) vn = vname.value.decode(errors="replace") out.append(f"V\t{path}\t{vn}\t{tname}\t{','.join(names) or '-'}\t{','.join(shape) or '-'}") if tname in NUMERIC and "?" not in shape: count = 1 for s in shape: count *= int(s) if count <= MAX_VALUES: numeric.append((path, vn)) ngrps = ctypes.c_int() grps = (ctypes.c_int * 65536)() check(nc.nc_inq_grps(gid, ctypes.byref(ngrps), grps), "nc_inq_grps") for child in grps[: ngrps.value]: cname = ctypes.create_string_buffer(NAME) check(nc.nc_inq_grpname(child, cname), "nc_inq_grpname") cpath = path.rstrip("/") + "/" + cname.value.decode(errors="replace") walk(nc, child, cpath, out, numeric) def is_unlimited(nc, gid, dimid): n = ctypes.c_int() ids = (ctypes.c_int * MAXDIMS)() # Unlimited dimensions visible from this group include its parents'. if nc.nc_inq_unlimdims(gid, ctypes.byref(n), ids) != 0: return False return dimid in ids[: n.value] def type_name(nc, gid, xtype): if xtype in ATOMIC: return ATOMIC[xtype] size = ctypes.c_size_t() base = ctypes.c_int() nfields = ctypes.c_size_t() klass = ctypes.c_int() tname = ctypes.create_string_buffer(NAME) if nc.nc_inq_user_type(gid, xtype, tname, ctypes.byref(size), ctypes.byref(base), ctypes.byref(nfields), ctypes.byref(klass)) != 0: return f"type{xtype}" return USER_CLASS.get(klass.value, f"class{klass.value}") def values(path, numeric, out): if not numeric: return import numpy as np import netCDF4 try: ds = netCDF4.Dataset(path) except Exception: # noqa: BLE001 - netCDF4-python refuses some files netCDF-C opens return with ds: for gpath, name in numeric: try: group = ds if gpath == "/" else ds[gpath] var = group.variables[name] var.set_auto_maskandscale(False) # A whole-variable read through netCDF-C 4.9.3 lays out a # variable shorter than an unlimited dimension that is not its # first wrongly (written values first); reads of one index of # the leading axis are right. if var.ndim >= 2: data = np.stack([np.asarray(var[i]) for i in range(var.shape[0])]) \ if var.shape[0] else np.zeros(var.shape) else: data = np.asarray(var[...]) flat = np.asarray(data, dtype=np.float64).ravel() except Exception: # noqa: BLE001 continue vals = " ".join(repr(float(v)) for v in flat) or "-" out.append(f"X\t{gpath}\t{name}\t{vals}") def limit_memory(): import resource resource.setrlimit(resource.RLIMIT_AS, (MEMORY, MEMORY)) def one(path): """Run view() for one file in a child process.""" try: proc = subprocess.run([sys.executable, __file__, "--one", path], capture_output=True, timeout=TIMEOUT, text=True, preexec_fn=limit_memory) except subprocess.TimeoutExpired: return [f"FILE\t{path}", "ERROR\ttimeout", "END"] lines = proc.stdout.splitlines() if proc.returncode != 0 or not lines or lines[-1] != "END": err = (proc.stderr.strip().splitlines() or [f"exit {proc.returncode}"])[-1] return [f"FILE\t{path}", f"ERROR\tchild failed: {err}", "END"] return lines def main(argv): if argv and argv[0] == "--one": out = [f"FILE\t{argv[1]}"] try: view(argv[1], out) except Exception as e: # noqa: BLE001 out = [f"FILE\t{argv[1]}", f"ERROR\t{e}"] out.append("END") sys.stdout.write("\n".join(out) + "\n") return if argv and argv[0] == "--list": with open(argv[1]) as fh: argv = [line.rstrip("\n") for line in fh if line.strip()] jobs = int(os.environ.get("CLAWHDF5_NETCDF_JOBS", "4")) with concurrent.futures.ThreadPoolExecutor(jobs) as pool: for lines in pool.map(one, argv): sys.stdout.write("\n".join(lines) + "\n") if __name__ == "__main__": main(sys.argv[1:])