clawhdf5-netcdf4: phony dimensions, skipped types and order as netCDF-C
Read a file's metadata the way netCDF-C 4.9.3 does (libhdf5/hdf5open.c), for the whole file on first use (src/model.rs, replacing src/scope.rs): - links in creation order when the group tracks it, else name order; a group's datasets before its subgroups; dimension ids file-wide; - variables' dimensions from _Netcdf4Coordinates (file-wide ids), else the scales DIMENSION_LIST attaches when the first axis has one, else netCDF-C's phony dimensions phony_dim_<id> (create_phony_dims: shared by length and unlimitedness within a group, not between two axes of one variable, numbered subgroups first, a zero length unlimited); - datasets of types netCDF-C cannot represent are not variables (references, bit fields, time, arrays, compounds/enums/VLENs over them), replaying netCDF-C's file-wide type list, failed types included; - unlimited lengths as nc4_find_dim_len (its group and below). NcType gains Enum, Compound, VLen, Opaque and is #[non_exhaustive]; Variable::nc_type is netCDF-C's type (1-byte strings NC_CHAR). New clawhdf5_format::group_v2::links_in_creation_order_in. Tests compare with netCDF-C itself (tests/netcdf_c_view.py calls the libnetcdf netCDF4-python bundles through ctypes): new interop cases for h5py files without dimension scales, every type class, link order; and the gated corpus_vs_netcdf_c (CLAWHDF5_NETCDF_CORPUS): 420 of the 429 conformance-corpus files netCDF-C opens match (main: 68); the other 9 are explained in tests/corpus_known_differences.txt and known-issues. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -1,16 +1,15 @@
|
||||
//! NetCDF-4 dimension representation.
|
||||
//!
|
||||
//! Dimensions in NetCDF-4 are stored as HDF5 datasets with the CLASS=DIMENSION_SCALE
|
||||
//! attribute and a `_Netcdf4Dimid` attribute. Unlimited dimensions are detected via
|
||||
//! the HDF5 dataspace max_dimensions (u64::MAX indicates unlimited); their length
|
||||
//! is the largest extent of the variables attached to them.
|
||||
//! attribute and a `_Netcdf4Dimid` attribute; variables name theirs in
|
||||
//! `_Netcdf4Coordinates` and `DIMENSION_LIST`. A file without them gets
|
||||
//! netCDF-C's phony dimensions. How they are put together is in
|
||||
//! `crate::model`.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use clawhdf5::AttrValue;
|
||||
|
||||
use crate::error::Error;
|
||||
|
||||
/// A NetCDF-4 dimension.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Dimension {
|
||||
@@ -22,202 +21,10 @@ pub struct Dimension {
|
||||
pub is_unlimited: bool,
|
||||
}
|
||||
|
||||
/// A dimension scale of one group: the dataset that defines a dimension.
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct Scale {
|
||||
/// Object header address of the scale's dataset (what a variable's
|
||||
/// `DIMENSION_LIST` references).
|
||||
pub address: u64,
|
||||
/// Its `_Netcdf4Dimid` (what a variable's `_Netcdf4Coordinates` lists).
|
||||
pub dimid: Option<i64>,
|
||||
/// Index of its dimension in [`GroupDims::dims`].
|
||||
pub dim: usize,
|
||||
}
|
||||
|
||||
/// The dimensions a group defines, with the scales that define them.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub(crate) struct GroupDims {
|
||||
/// The group's dimensions, in `_Netcdf4Dimid` order (then discovery order).
|
||||
pub dims: Vec<Dimension>,
|
||||
/// The dimension scales behind `dims`; empty when the group has no
|
||||
/// dimension scale and `dims` were inferred from 1-D datasets.
|
||||
pub scales: Vec<Scale>,
|
||||
}
|
||||
|
||||
impl GroupDims {
|
||||
/// The dimension defined by the scale at `address`.
|
||||
pub fn by_address(&self, address: u64) -> Option<&Dimension> {
|
||||
self.scales
|
||||
.iter()
|
||||
.find(|s| s.address == address)
|
||||
.map(|s| &self.dims[s.dim])
|
||||
}
|
||||
|
||||
/// The dimension whose scale has `_Netcdf4Dimid` `id`.
|
||||
pub fn by_dimid(&self, id: i64) -> Option<&Dimension> {
|
||||
self.scales
|
||||
.iter()
|
||||
.find(|s| s.dimid == Some(id))
|
||||
.map(|s| &self.dims[s.dim])
|
||||
}
|
||||
}
|
||||
|
||||
/// The dimensions of an HDF5 group (root or subgroup).
|
||||
///
|
||||
/// NetCDF-4 stores dimensions as datasets with `CLASS=DIMENSION_SCALE`. A fixed
|
||||
/// dimension's size is the dataset's first (and typically only) shape extent.
|
||||
/// Unlimited dimensions have `max_dimensions[0] == u64::MAX` in the HDF5 dataspace;
|
||||
/// their size is computed by `unlimited_len`. A group with no dimension
|
||||
/// scale at all (not written by a netCDF library) gets one dimension per
|
||||
/// 1-D dataset instead.
|
||||
pub(crate) fn group_dims(
|
||||
file: &clawhdf5::File,
|
||||
group: &clawhdf5::Group<'_>,
|
||||
) -> Result<GroupDims, Error> {
|
||||
let addresses: HashMap<String, u64> = group.entries()?.into_iter().collect();
|
||||
let dataset_names = group.datasets()?;
|
||||
// (dimid, dimension, scale address), in discovery order.
|
||||
let mut found: Vec<(Option<i64>, Dimension, u64)> = Vec::new();
|
||||
|
||||
for ds_name in &dataset_names {
|
||||
let ds = group.dataset(ds_name)?;
|
||||
let attrs = ds.attrs()?;
|
||||
if !is_dimension_scale(&attrs) {
|
||||
continue;
|
||||
}
|
||||
let Some(&address) = addresses.get(ds_name) else {
|
||||
continue;
|
||||
};
|
||||
let shape = ds.shape()?;
|
||||
let is_unlimited = is_unlimited(&ds);
|
||||
let size = if is_unlimited {
|
||||
unlimited_len(file, &attrs, &shape)
|
||||
} else {
|
||||
shape.first().copied().unwrap_or(0)
|
||||
};
|
||||
let dim = Dimension {
|
||||
name: ds_name.clone(),
|
||||
size,
|
||||
is_unlimited,
|
||||
};
|
||||
found.push((get_dimid(&attrs), dim, address));
|
||||
}
|
||||
|
||||
if found.is_empty() {
|
||||
// Fallback: infer dimensions from dataset shapes and names.
|
||||
// In NetCDF-4, coordinate variables are datasets whose name matches
|
||||
// a dimension name. If there are no explicit DIMENSION_SCALE attributes,
|
||||
// we look for 1-D datasets that might be coordinate variables.
|
||||
let mut dims = Vec::new();
|
||||
for ds_name in &dataset_names {
|
||||
let ds = group.dataset(ds_name)?;
|
||||
let shape = ds.shape()?;
|
||||
if shape.len() == 1 {
|
||||
dims.push(Dimension {
|
||||
name: ds_name.clone(),
|
||||
size: shape[0],
|
||||
is_unlimited: is_unlimited(&ds),
|
||||
});
|
||||
}
|
||||
}
|
||||
return Ok(GroupDims {
|
||||
dims,
|
||||
scales: Vec::new(),
|
||||
});
|
||||
}
|
||||
|
||||
// By dimid; scales without one keep their discovery order after those
|
||||
// with one (the sort is stable).
|
||||
found.sort_by_key(|(id, ..)| (id.is_none(), id.unwrap_or(0)));
|
||||
let mut out = GroupDims::default();
|
||||
for (i, (dimid, dim, address)) in found.into_iter().enumerate() {
|
||||
out.dims.push(dim);
|
||||
out.scales.push(Scale {
|
||||
address,
|
||||
dimid,
|
||||
dim: i,
|
||||
});
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// The start of the `NAME` attribute netCDF-C gives a dimension scale that
|
||||
/// is only a dimension, not also a (coordinate) variable.
|
||||
const PURE_DIMENSION_NAME: &str = "This is a netCDF dimension but not a netCDF variable";
|
||||
|
||||
/// The current length of an unlimited dimension, as netCDF-C reports it
|
||||
/// (`NC4_inq_dim` → `nc4_find_dim_len`): the largest current extent, along
|
||||
/// the dimension, of the variables that use it, in any group; 0 when none
|
||||
/// has been written. netCDF-C does not extend a dimension scale that is not
|
||||
/// also a variable, so such a scale's own extent (0) is not counted; a
|
||||
/// coordinate variable's is. The variables are the scale's attachments,
|
||||
/// listed with the axis they use in its `REFERENCE_LIST` attribute (the
|
||||
/// mirror of each variable's `DIMENSION_LIST`). Attachments that cannot be
|
||||
/// read are skipped; without a readable `REFERENCE_LIST` the length is the
|
||||
/// scale's own extent, as before.
|
||||
fn unlimited_len(file: &clawhdf5::File, attrs: &HashMap<String, AttrValue>, shape: &[u64]) -> u64 {
|
||||
let own = shape.first().copied().unwrap_or(0);
|
||||
let is_variable = !is_pure_dimension(attrs);
|
||||
let Some(refs) = reference_list(file, attrs) else {
|
||||
return own;
|
||||
};
|
||||
refs.into_iter()
|
||||
.filter_map(|(address, axis)| {
|
||||
let shape = file.dataset_at(address).ok()?.shape().ok()?;
|
||||
shape.get(usize::try_from(axis).ok()?).copied()
|
||||
})
|
||||
.chain(is_variable.then_some(own))
|
||||
.max()
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
/// The `(dataset address, axis)` pairs of a dimension scale's
|
||||
/// `REFERENCE_LIST` attribute (HDF5 dimension scales: a compound of an
|
||||
/// object reference `dataset` and an integer `dimension`), or `None` when it
|
||||
/// is missing or not in that form.
|
||||
fn reference_list(
|
||||
file: &clawhdf5::File,
|
||||
attrs: &HashMap<String, AttrValue>,
|
||||
) -> Option<Vec<(u64, u64)>> {
|
||||
use clawhdf5_format::data_read::{read_compound_field, read_object_references};
|
||||
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||
let Some(AttrValue::Raw { datatype, data, .. }) = attrs.get("REFERENCE_LIST") else {
|
||||
return None;
|
||||
};
|
||||
let dataset = read_compound_field(data, datatype, "dataset").ok()?;
|
||||
let addresses = read_object_references(
|
||||
&dataset.raw_data,
|
||||
&dataset.datatype,
|
||||
file.superblock().offset_size,
|
||||
)
|
||||
.ok()?;
|
||||
let dimension = read_compound_field(data, datatype, "dimension").ok()?;
|
||||
let Datatype::FixedPoint {
|
||||
size, byte_order, ..
|
||||
} = dimension.datatype
|
||||
else {
|
||||
return None;
|
||||
};
|
||||
let size = usize::try_from(size).ok().filter(|s| (1..=8).contains(s))?;
|
||||
let axes = dimension.raw_data.chunks_exact(size).map(|b| {
|
||||
let mut v = [0u8; 8];
|
||||
match byte_order {
|
||||
DatatypeByteOrder::BigEndian => {
|
||||
v[8 - size..].copy_from_slice(b);
|
||||
u64::from_be_bytes(v)
|
||||
}
|
||||
_ => {
|
||||
v[..size].copy_from_slice(b);
|
||||
u64::from_le_bytes(v)
|
||||
}
|
||||
}
|
||||
});
|
||||
if axes.len() != addresses.len() {
|
||||
return None;
|
||||
}
|
||||
Some(addresses.into_iter().map(|r| r.address).zip(axes).collect())
|
||||
}
|
||||
|
||||
/// The dimension scale attached to each axis of a variable, from its
|
||||
/// `DIMENSION_LIST` attribute (HDF5 dimension scales: one variable-length
|
||||
/// sequence of object references per axis) — the address of the scale
|
||||
@@ -281,13 +88,9 @@ pub(crate) fn is_dimension_scale(attrs: &HashMap<String, AttrValue>) -> bool {
|
||||
pub(crate) fn get_dimid(attrs: &HashMap<String, AttrValue>) -> Option<i64> {
|
||||
match attrs.get("_Netcdf4Dimid") {
|
||||
Some(AttrValue::I64(id)) => Some(*id),
|
||||
Some(AttrValue::U64(id)) => Some(*id as i64),
|
||||
Some(AttrValue::U64(id)) => i64::try_from(*id).ok(),
|
||||
Some(AttrValue::I64Array(ids)) if ids.len() == 1 => Some(ids[0]),
|
||||
Some(AttrValue::U64Array(ids)) if ids.len() == 1 => i64::try_from(ids[0]).ok(),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a dataset's first axis is unlimited (`max_dimensions[0] ==
|
||||
/// u64::MAX` in its dataspace).
|
||||
fn is_unlimited(ds: &clawhdf5::Dataset<'_>) -> bool {
|
||||
matches!(ds.max_dimensions(), Ok(Some(max_dims)) if max_dims.first() == Some(&u64::MAX))
|
||||
}
|
||||
|
||||
@@ -7,36 +7,36 @@ use std::collections::HashMap;
|
||||
|
||||
use clawhdf5::AttrValue;
|
||||
|
||||
use crate::dimension::{self, Dimension};
|
||||
use crate::dimension::Dimension;
|
||||
use crate::error::Error;
|
||||
use crate::scope::{self, Scope};
|
||||
use crate::model::Model;
|
||||
use crate::variable::Variable;
|
||||
|
||||
/// A NetCDF-4 group corresponding to an HDF5 group.
|
||||
pub struct NetCDF4Group<'f> {
|
||||
/// Group name.
|
||||
name: String,
|
||||
/// Path of the group from the root (`/`-separated).
|
||||
path: String,
|
||||
/// Underlying HDF5 file.
|
||||
file: &'f clawhdf5::File,
|
||||
/// Underlying HDF5 group.
|
||||
hdf5_group: clawhdf5::Group<'f>,
|
||||
/// The file's groups, dimensions and variables.
|
||||
model: &'f Model,
|
||||
/// This group's index in `model`.
|
||||
index: usize,
|
||||
}
|
||||
|
||||
impl<'f> NetCDF4Group<'f> {
|
||||
/// Create a new NetCDF4Group from an HDF5 group.
|
||||
/// The group `index` of `model`.
|
||||
pub(crate) fn new(
|
||||
name: String,
|
||||
path: String,
|
||||
file: &'f clawhdf5::File,
|
||||
hdf5_group: clawhdf5::Group<'f>,
|
||||
model: &'f Model,
|
||||
index: usize,
|
||||
) -> Self {
|
||||
Self {
|
||||
name,
|
||||
path,
|
||||
file,
|
||||
hdf5_group,
|
||||
model,
|
||||
index,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -45,52 +45,56 @@ impl<'f> NetCDF4Group<'f> {
|
||||
&self.name
|
||||
}
|
||||
|
||||
/// List dimensions defined in this group (not those of its parent
|
||||
/// groups, which its variables can also use).
|
||||
/// List dimensions defined in this group, in dimension-id order (not
|
||||
/// those of its parent groups, which its variables can also use).
|
||||
pub fn dimensions(&self) -> Result<Vec<Dimension>, Error> {
|
||||
Ok(dimension::group_dims(self.file, &self.hdf5_group)?.dims)
|
||||
Ok(self.model.dimensions(self.index))
|
||||
}
|
||||
|
||||
/// List variables in this group: its datasets, except the dimension
|
||||
/// scales that are only dimensions. Their dimensions can be defined in
|
||||
/// this group or a parent group.
|
||||
/// List variables in this group, in netCDF-C's order: its datasets,
|
||||
/// except the dimension scales that are only dimensions and datasets
|
||||
/// of types netCDF-C cannot represent. Their dimensions can be defined
|
||||
/// in this group or a parent group.
|
||||
pub fn variables(&self) -> Result<Vec<Variable<'f>>, Error> {
|
||||
Scope::new(self.file, &self.path)?.variables()
|
||||
self.model.variables(self.file, self.index)
|
||||
}
|
||||
|
||||
/// Get a specific variable by name.
|
||||
/// Get a specific variable by name (or by path, `"sub/var"`).
|
||||
pub fn variable(&self, name: &str) -> Result<Variable<'f>, Error> {
|
||||
scope::variable_at(self.file, &self.path, name)
|
||||
self.model.variable(self.file, self.index, name)
|
||||
}
|
||||
|
||||
/// Read all attributes of this group.
|
||||
pub fn attrs(&self) -> Result<HashMap<String, AttrValue>, Error> {
|
||||
Ok(self.hdf5_group.attrs()?)
|
||||
Ok(self
|
||||
.file
|
||||
.group_at(self.model.group_address(self.index))
|
||||
.attrs()?)
|
||||
}
|
||||
|
||||
/// List subgroup names.
|
||||
/// List subgroup names, in netCDF-C's order.
|
||||
pub fn group_names(&self) -> Result<Vec<String>, Error> {
|
||||
Ok(self.hdf5_group.groups()?)
|
||||
Ok(self.model.group_names(self.index))
|
||||
}
|
||||
|
||||
/// Get a subgroup by name.
|
||||
/// Get a subgroup by name (or by path, `"a/b"`).
|
||||
pub fn group(&self, name: &str) -> Result<NetCDF4Group<'f>, Error> {
|
||||
let hdf5_group = self
|
||||
.hdf5_group
|
||||
.group(name)
|
||||
.map_err(|_| Error::GroupNotFound(name.to_string()))?;
|
||||
let index = self
|
||||
.model
|
||||
.find_group(self.index, name)
|
||||
.ok_or_else(|| Error::GroupNotFound(name.to_string()))?;
|
||||
Ok(NetCDF4Group::new(
|
||||
name.to_string(),
|
||||
format!("{}/{name}", self.path),
|
||||
self.file,
|
||||
hdf5_group,
|
||||
self.model,
|
||||
index,
|
||||
))
|
||||
}
|
||||
|
||||
/// The names of this group's variables (see
|
||||
/// [`variables`](Self::variables)).
|
||||
pub fn variable_names(&self) -> Result<Vec<String>, Error> {
|
||||
Scope::new(self.file, &self.path)?.variable_names()
|
||||
Ok(self.model.variable_names(self.index))
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ pub mod cf;
|
||||
pub mod dimension;
|
||||
pub mod error;
|
||||
pub mod group;
|
||||
mod scope;
|
||||
mod model;
|
||||
pub mod types;
|
||||
pub mod variable;
|
||||
|
||||
@@ -40,26 +40,49 @@ pub use types::NcType;
|
||||
pub use variable::Variable;
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::sync::OnceLock;
|
||||
|
||||
use model::Model;
|
||||
|
||||
/// A NetCDF-4 file reader.
|
||||
///
|
||||
/// Wraps a clawhdf5 File and provides NetCDF-4 semantics: dimensions,
|
||||
/// variables with CF attributes, groups, and type mapping.
|
||||
///
|
||||
/// The groups, dimensions and variables are what netCDF-C reports for the
|
||||
/// file (see the crate README): the first call that needs them reads the
|
||||
/// metadata of the whole file, as `nc_open` does, and keeps it; the values
|
||||
/// are read when asked for.
|
||||
pub struct NetCDF4File {
|
||||
hdf5: clawhdf5::File,
|
||||
model: OnceLock<Model>,
|
||||
}
|
||||
|
||||
impl NetCDF4File {
|
||||
/// Open a NetCDF-4 file from a filesystem path.
|
||||
pub fn open<P: AsRef<std::path::Path>>(path: P) -> Result<Self, Error> {
|
||||
let hdf5 = clawhdf5::File::open(path)?;
|
||||
Ok(Self { hdf5 })
|
||||
Ok(Self::from_hdf5(clawhdf5::File::open(path)?))
|
||||
}
|
||||
|
||||
/// Open a NetCDF-4 file from in-memory bytes.
|
||||
pub fn from_bytes(data: Vec<u8>) -> Result<Self, Error> {
|
||||
let hdf5 = clawhdf5::File::from_bytes(data)?;
|
||||
Ok(Self { hdf5 })
|
||||
Ok(Self::from_hdf5(clawhdf5::File::from_bytes(data)?))
|
||||
}
|
||||
|
||||
fn from_hdf5(hdf5: clawhdf5::File) -> Self {
|
||||
Self {
|
||||
hdf5,
|
||||
model: OnceLock::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// The file's groups, dimensions and variables, read on first use.
|
||||
fn model(&self) -> Result<&Model, Error> {
|
||||
if let Some(model) = self.model.get() {
|
||||
return Ok(model);
|
||||
}
|
||||
let model = Model::build(&self.hdf5)?;
|
||||
Ok(self.model.get_or_init(|| model))
|
||||
}
|
||||
|
||||
/// Get the _NCProperties root attribute, if present.
|
||||
@@ -74,27 +97,29 @@ impl NetCDF4File {
|
||||
}
|
||||
}
|
||||
|
||||
/// List dimensions defined in the root group.
|
||||
/// List dimensions defined in the root group, in dimension-id order.
|
||||
pub fn dimensions(&self) -> Result<Vec<Dimension>, Error> {
|
||||
Ok(dimension::group_dims(&self.hdf5, &self.hdf5.root())?.dims)
|
||||
Ok(self.model()?.dimensions(0))
|
||||
}
|
||||
|
||||
/// List all variables in the root group: its datasets, except the
|
||||
/// dimension scales that are only dimensions (netCDF-C does not list
|
||||
/// them either).
|
||||
/// List the variables of the root group, in netCDF-C's order: its
|
||||
/// datasets, except the dimension scales that are only dimensions and
|
||||
/// datasets of types netCDF-C cannot represent (see
|
||||
/// [`NcType`]).
|
||||
pub fn variables(&self) -> Result<Vec<Variable<'_>>, Error> {
|
||||
scope::Scope::new(&self.hdf5, "/")?.variables()
|
||||
self.model()?.variables(&self.hdf5, 0)
|
||||
}
|
||||
|
||||
/// The names of the root group's variables (see
|
||||
/// [`variables`](Self::variables)).
|
||||
pub fn variable_names(&self) -> Result<Vec<String>, Error> {
|
||||
scope::Scope::new(&self.hdf5, "/")?.variable_names()
|
||||
Ok(self.model()?.variable_names(0))
|
||||
}
|
||||
|
||||
/// Get a specific variable by name from the root group.
|
||||
/// Get a specific variable by name from the root group; the name may be
|
||||
/// a path into a subgroup (`"sub/var"`).
|
||||
pub fn variable(&self, name: &str) -> Result<Variable<'_>, Error> {
|
||||
scope::variable_at(&self.hdf5, "", name)
|
||||
self.model()?.variable(&self.hdf5, 0, name)
|
||||
}
|
||||
|
||||
/// Read all global (root group) attributes.
|
||||
@@ -102,22 +127,22 @@ impl NetCDF4File {
|
||||
Ok(self.hdf5.root().attrs()?)
|
||||
}
|
||||
|
||||
/// List subgroup names in the root group.
|
||||
/// List subgroup names in the root group, in netCDF-C's order.
|
||||
pub fn group_names(&self) -> Result<Vec<String>, Error> {
|
||||
Ok(self.hdf5.root().groups()?)
|
||||
Ok(self.model()?.group_names(0))
|
||||
}
|
||||
|
||||
/// Get a subgroup by name.
|
||||
/// Get a subgroup by name (or by path, `"a/b"`).
|
||||
pub fn group(&self, name: &str) -> Result<NetCDF4Group<'_>, Error> {
|
||||
let hdf5_group = self
|
||||
.hdf5
|
||||
.group(name)
|
||||
.map_err(|_| Error::GroupNotFound(name.to_string()))?;
|
||||
let model = self.model()?;
|
||||
let index = model
|
||||
.find_group(0, name)
|
||||
.ok_or_else(|| Error::GroupNotFound(name.to_string()))?;
|
||||
Ok(NetCDF4Group::new(
|
||||
name.to_string(),
|
||||
name.to_string(),
|
||||
&self.hdf5,
|
||||
hdf5_group,
|
||||
model,
|
||||
index,
|
||||
))
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,853 @@
|
||||
//! A file's groups, dimensions and variables, as netCDF-C reads them.
|
||||
//!
|
||||
//! netCDF-C (`libhdf5/hdf5open.c`, 4.9.3) reads a file in two passes, and
|
||||
//! the names, order and sharing of the dimensions depend on both, so this
|
||||
//! module replays them for the whole file at once:
|
||||
//!
|
||||
//! 1. `rec_read_metadata`: each group's links in creation order when the
|
||||
//! group tracks it, else in name order; a group's datasets and named
|
||||
//! datatypes before its subgroups, which follow in the same order. A
|
||||
//! dimension scale (`CLASS` `DIMENSION_SCALE`) defines a dimension with
|
||||
//! the id in its `_Netcdf4Dimid`, else the next free id (ids are
|
||||
//! file-wide), its first extent as length, unlimited when that axis is
|
||||
//! (or the length is 0); it is a variable too unless its `NAME` says it
|
||||
//! is only a dimension. Every other dataset is a variable, unless its
|
||||
//! type is one netCDF-C cannot represent ([`VarTypes::nc_type`]); a
|
||||
//! dataset `_nc4_non_coord_<name>` is the variable `<name>`.
|
||||
//! 2. `rec_match_dimscales`, subgroups first, then the group's variables in
|
||||
//! order: a variable gets the dimensions whose ids its
|
||||
//! `_Netcdf4Coordinates` lists (looked up file-wide), else — when its
|
||||
//! `DIMENSION_LIST` attaches a scale to its first axis — the scales that
|
||||
//! list attaches, looked up in its group and then each parent, else
|
||||
//! "phony" dimensions (`create_phony_dims`): per axis, the first
|
||||
//! dimension of the variable's group of that length and unlimitedness
|
||||
//! not already used by an earlier axis of the variable, else a new one
|
||||
//! called `phony_dim_<id>`.
|
||||
//!
|
||||
//! An unlimited dimension's length is the largest extent, along it, of the
|
||||
//! variables that use it in its group and the groups below
|
||||
//! (`nc4_find_dim_len`).
|
||||
//!
|
||||
//! Where netCDF-C 4.9.3 leaves an axis without a dimension (an id or a scale
|
||||
//! it cannot find, or an axis without a scale of a variable whose first
|
||||
//! axis has one — there it reads uninitialised memory), netCDF4-python
|
||||
//! cannot open the file; this crate gives such an axis a dimension by the
|
||||
//! phony rule instead.
|
||||
|
||||
use std::collections::HashMap;
|
||||
|
||||
use clawhdf5::AttrValue;
|
||||
use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder};
|
||||
use clawhdf5_format::object_header::{ObjectClass, ObjectHeader};
|
||||
|
||||
use crate::dimension::{self, Dimension};
|
||||
use crate::error::Error;
|
||||
use crate::types::NcType;
|
||||
use crate::variable::Variable;
|
||||
|
||||
/// The prefix netCDF-C gives the dataset of a variable that has a
|
||||
/// dimension's name but is not that dimension's coordinate variable (the
|
||||
/// dimension's scale holds the name).
|
||||
const NON_COORD_PREFIX: &str = "_nc4_non_coord_";
|
||||
|
||||
/// Groups read at most, a guard against files whose groups are hard-linked
|
||||
/// into each other many times over (netCDF-C reads each link as its own
|
||||
/// group).
|
||||
const MAX_GROUPS: usize = 100_000;
|
||||
|
||||
/// A file as netCDF-C sees it.
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct Model {
|
||||
/// Every group; the root is the first.
|
||||
groups: Vec<Group>,
|
||||
/// Every dimension, in the order they were created.
|
||||
dims: Vec<Dim>,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Group {
|
||||
/// Object header address of the HDF5 group.
|
||||
address: u64,
|
||||
/// Its subgroups, `(name, index in Model::groups)`, in netCDF-C's order.
|
||||
children: Vec<(String, usize)>,
|
||||
parent: Option<usize>,
|
||||
/// The dimensions it defines (indexes in `Model::dims`), in creation
|
||||
/// order.
|
||||
dims: Vec<usize>,
|
||||
/// Its variables, in netCDF-C's order.
|
||||
vars: Vec<Var>,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Dim {
|
||||
id: i64,
|
||||
name: String,
|
||||
/// The length it was created with (for an unlimited dimension, replaced
|
||||
/// by its current length once every variable has its dimensions).
|
||||
len: u64,
|
||||
unlimited: bool,
|
||||
/// Object header address of its dimension scale; `None` for a phony
|
||||
/// dimension.
|
||||
scale: Option<u64>,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
struct Var {
|
||||
/// The netCDF name.
|
||||
name: String,
|
||||
/// Whether its dataset is `_nc4_non_coord_<name>`.
|
||||
non_coord: bool,
|
||||
address: u64,
|
||||
nc_type: NcType,
|
||||
extent: Vec<u64>,
|
||||
/// Whether each axis is unlimited in the dataspace.
|
||||
unlimited: Vec<bool>,
|
||||
/// How netCDF-C finds its dimensions.
|
||||
source: DimSource,
|
||||
/// Its dimensions (indexes in `Model::dims`), one per axis, once found.
|
||||
dims: Vec<usize>,
|
||||
}
|
||||
|
||||
#[derive(Debug)]
|
||||
enum DimSource {
|
||||
/// A one-dimensional coordinate variable: its own scale's dimension.
|
||||
OwnScale(usize),
|
||||
/// The ids in `_Netcdf4Coordinates`; for a multi-dimensional coordinate
|
||||
/// variable, also its own dimension (for the first axis, should an id
|
||||
/// not be found).
|
||||
Coordinates(Vec<i64>, Option<usize>),
|
||||
/// The scale `DIMENSION_LIST` attaches to each axis (the first has one).
|
||||
Scales(Vec<Option<u64>>, Option<usize>),
|
||||
/// None: phony dimensions.
|
||||
Phony(Option<usize>),
|
||||
}
|
||||
|
||||
impl Model {
|
||||
/// Read the metadata of `file` as netCDF-C does.
|
||||
pub fn build(file: &clawhdf5::File) -> Result<Self, Error> {
|
||||
let mut builder = Builder {
|
||||
file,
|
||||
model: Model {
|
||||
groups: Vec::new(),
|
||||
dims: Vec::new(),
|
||||
},
|
||||
next_id: 0,
|
||||
types: VarTypes::default(),
|
||||
};
|
||||
let root = file.superblock().root_group_address;
|
||||
builder.model.groups.push(Group {
|
||||
address: root,
|
||||
children: Vec::new(),
|
||||
parent: None,
|
||||
dims: Vec::new(),
|
||||
vars: Vec::new(),
|
||||
});
|
||||
builder.read_group(0, &mut vec![root])?;
|
||||
builder.match_dims(0);
|
||||
builder.unlimited_lengths();
|
||||
Ok(builder.model)
|
||||
}
|
||||
|
||||
/// The group at `path` (`/`-separated names), from the group `from`.
|
||||
pub fn find_group(&self, from: usize, path: &str) -> Option<usize> {
|
||||
path.split('/')
|
||||
.filter(|p| !p.is_empty())
|
||||
.try_fold(from, |g, name| {
|
||||
self.groups[g]
|
||||
.children
|
||||
.iter()
|
||||
.find(|(n, _)| n == name)
|
||||
.map(|&(_, i)| i)
|
||||
})
|
||||
}
|
||||
|
||||
/// The object header address of group `g`.
|
||||
pub fn group_address(&self, g: usize) -> u64 {
|
||||
self.groups[g].address
|
||||
}
|
||||
|
||||
/// The names of group `g`'s subgroups, in netCDF-C's order.
|
||||
pub fn group_names(&self, g: usize) -> Vec<String> {
|
||||
self.groups[g]
|
||||
.children
|
||||
.iter()
|
||||
.map(|(n, _)| n.clone())
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The dimensions group `g` defines, in id order (as `nc_inq_dimids`).
|
||||
pub fn dimensions(&self, g: usize) -> Vec<Dimension> {
|
||||
let mut dims: Vec<&Dim> = self.groups[g].dims.iter().map(|&d| &self.dims[d]).collect();
|
||||
dims.sort_by_key(|d| d.id);
|
||||
dims.into_iter().map(Dim::dimension).collect()
|
||||
}
|
||||
|
||||
/// The names of group `g`'s variables, in netCDF-C's order.
|
||||
pub fn variable_names(&self, g: usize) -> Vec<String> {
|
||||
self.groups[g].vars.iter().map(|v| v.name.clone()).collect()
|
||||
}
|
||||
|
||||
/// Group `g`'s variables, in netCDF-C's order.
|
||||
pub fn variables<'f>(
|
||||
&self,
|
||||
file: &'f clawhdf5::File,
|
||||
g: usize,
|
||||
) -> Result<Vec<Variable<'f>>, Error> {
|
||||
self.groups[g]
|
||||
.vars
|
||||
.iter()
|
||||
.map(|v| self.open(file, v))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The variable `name` of group `g`; `name` may be a path (`"sub/var"`)
|
||||
/// relative to it. Of two variables of one name (datasets `<name>` and
|
||||
/// `_nc4_non_coord_<name>`), the second.
|
||||
pub fn variable<'f>(
|
||||
&self,
|
||||
file: &'f clawhdf5::File,
|
||||
g: usize,
|
||||
name: &str,
|
||||
) -> Result<Variable<'f>, Error> {
|
||||
let not_found = || Error::VariableNotFound(name.to_string());
|
||||
let trimmed = name.trim_start_matches('/');
|
||||
let (g, leaf) = match trimmed.rsplit_once('/') {
|
||||
Some((dir, leaf)) => (self.find_group(g, dir).ok_or_else(not_found)?, leaf),
|
||||
None => (g, trimmed),
|
||||
};
|
||||
let vars = &self.groups[g].vars;
|
||||
let var = vars
|
||||
.iter()
|
||||
.find(|v| v.name == leaf && v.non_coord)
|
||||
.or_else(|| vars.iter().find(|v| v.name == leaf))
|
||||
.ok_or_else(not_found)?;
|
||||
self.open(file, var)
|
||||
}
|
||||
|
||||
fn open<'f>(&self, file: &'f clawhdf5::File, var: &Var) -> Result<Variable<'f>, Error> {
|
||||
let ds = file.dataset_at(var.address)?;
|
||||
let attrs = ds.attrs().unwrap_or_default();
|
||||
let dims = var.dims.iter().map(|&d| self.dims[d].dimension()).collect();
|
||||
Ok(Variable::new(
|
||||
var.name.clone(),
|
||||
ds,
|
||||
dims,
|
||||
attrs,
|
||||
var.nc_type,
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
impl Dim {
|
||||
fn dimension(&self) -> Dimension {
|
||||
Dimension {
|
||||
name: self.name.clone(),
|
||||
size: self.len,
|
||||
is_unlimited: self.unlimited,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
struct Builder<'f> {
|
||||
file: &'f clawhdf5::File,
|
||||
model: Model,
|
||||
/// netCDF-C's `next_dimid`.
|
||||
next_id: i64,
|
||||
types: VarTypes,
|
||||
}
|
||||
|
||||
impl Builder<'_> {
|
||||
/// Pass 1 for group `g` and, after its own links, its subgroups.
|
||||
/// `ancestors` holds the addresses of the groups from the root to `g`,
|
||||
/// so that a group linked into itself is not read forever.
|
||||
fn read_group(&mut self, g: usize, ancestors: &mut Vec<u64>) -> Result<(), Error> {
|
||||
let address = self.model.groups[g].address;
|
||||
let sb = self.file.superblock();
|
||||
let mut subgroups = Vec::new();
|
||||
for (name, child) in ordered_entries(self.file, address)? {
|
||||
let Ok(header) =
|
||||
ObjectHeader::parse_in(self.file.storage(), child, sb.offset_size, sb.length_size)
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
match header.object_class() {
|
||||
Some(ObjectClass::Dataset) => self.read_dataset(g, name, child),
|
||||
Some(ObjectClass::NamedDatatype) => {
|
||||
if let Some(dt) = header_datatype(&header) {
|
||||
self.types.named(&dt);
|
||||
}
|
||||
}
|
||||
_ if is_group(&header) => subgroups.push((name, child)),
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
for (name, child) in subgroups {
|
||||
if ancestors.contains(&child) || self.model.groups.len() >= MAX_GROUPS {
|
||||
continue;
|
||||
}
|
||||
let index = self.model.groups.len();
|
||||
self.model.groups.push(Group {
|
||||
address: child,
|
||||
children: Vec::new(),
|
||||
parent: Some(g),
|
||||
dims: Vec::new(),
|
||||
vars: Vec::new(),
|
||||
});
|
||||
self.model.groups[g].children.push((name, index));
|
||||
ancestors.push(child);
|
||||
let read = self.read_group(index, ancestors);
|
||||
ancestors.pop();
|
||||
read?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// `read_dataset`: a dimension for a dimension scale, a variable for
|
||||
/// the rest. A dataset that cannot be opened is skipped.
|
||||
fn read_dataset(&mut self, g: usize, name: String, address: u64) {
|
||||
let Ok(ds) = self.file.dataset_at(address) else {
|
||||
return;
|
||||
};
|
||||
let attrs = ds.attrs().unwrap_or_default();
|
||||
let Ok(extent) = ds.shape() else {
|
||||
return;
|
||||
};
|
||||
let max = ds.max_dimensions().ok().flatten();
|
||||
let unlimited: Vec<bool> = (0..extent.len())
|
||||
.map(|i| matches!(&max, Some(m) if m.get(i) == Some(&u64::MAX)))
|
||||
.collect();
|
||||
|
||||
let mut own = None;
|
||||
if dimension::is_dimension_scale(&attrs) && !extent.is_empty() {
|
||||
// read_scale
|
||||
let id = match dimension::get_dimid(&attrs) {
|
||||
Some(id) => {
|
||||
if id >= self.next_id {
|
||||
self.next_id = id.saturating_add(1);
|
||||
}
|
||||
id
|
||||
}
|
||||
None => self.take_id(),
|
||||
};
|
||||
let len = extent[0];
|
||||
own = Some(self.add_dim(
|
||||
g,
|
||||
Dim {
|
||||
id,
|
||||
name: name.clone(),
|
||||
len,
|
||||
unlimited: unlimited[0] || len == 0,
|
||||
scale: Some(address),
|
||||
},
|
||||
));
|
||||
if dimension::is_pure_dimension(&attrs) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
// read_var: a type netCDF-C cannot represent drops the variable
|
||||
// (the dimension of a scale stays).
|
||||
let Some(nc_type) = ds
|
||||
.raw_datatype()
|
||||
.ok()
|
||||
.and_then(|dt| self.types.nc_type(&dt))
|
||||
else {
|
||||
return;
|
||||
};
|
||||
let rank = extent.len();
|
||||
let coordinates = coordinates(&attrs).filter(|ids| ids.len() == rank && rank > 0);
|
||||
let source = match (own, coordinates) {
|
||||
(Some(dim), _) if rank == 1 => DimSource::OwnScale(dim),
|
||||
(_, Some(ids)) => DimSource::Coordinates(ids, own),
|
||||
_ => match dimension::dimension_list(self.file, &attrs) {
|
||||
Some(scales)
|
||||
if scales.len() == rank && scales.first().is_some_and(Option::is_some) =>
|
||||
{
|
||||
DimSource::Scales(scales, own)
|
||||
}
|
||||
_ => DimSource::Phony(own),
|
||||
},
|
||||
};
|
||||
let (name, non_coord) = match name.strip_prefix(NON_COORD_PREFIX) {
|
||||
Some(rest) if !rest.is_empty() => (rest.to_string(), true),
|
||||
_ => (name, false),
|
||||
};
|
||||
self.model.groups[g].vars.push(Var {
|
||||
name,
|
||||
non_coord,
|
||||
address,
|
||||
nc_type,
|
||||
extent,
|
||||
unlimited,
|
||||
source,
|
||||
dims: Vec::new(),
|
||||
});
|
||||
}
|
||||
|
||||
fn take_id(&mut self) -> i64 {
|
||||
let id = self.next_id;
|
||||
self.next_id = self.next_id.saturating_add(1);
|
||||
id
|
||||
}
|
||||
|
||||
fn add_dim(&mut self, g: usize, dim: Dim) -> usize {
|
||||
let index = self.model.dims.len();
|
||||
self.model.dims.push(dim);
|
||||
self.model.groups[g].dims.push(index);
|
||||
index
|
||||
}
|
||||
|
||||
/// Pass 2 (`rec_match_dimscales`): subgroups first, then the group's
|
||||
/// variables in order.
|
||||
fn match_dims(&mut self, g: usize) {
|
||||
let children: Vec<usize> = self.model.groups[g]
|
||||
.children
|
||||
.iter()
|
||||
.map(|&(_, c)| c)
|
||||
.collect();
|
||||
for child in children {
|
||||
self.match_dims(child);
|
||||
}
|
||||
for v in 0..self.model.groups[g].vars.len() {
|
||||
let var = &self.model.groups[g].vars[v];
|
||||
let rank = var.extent.len();
|
||||
let mut found: Vec<Option<usize>> = vec![None; rank];
|
||||
match &var.source {
|
||||
DimSource::OwnScale(dim) => found[0] = Some(*dim),
|
||||
DimSource::Coordinates(ids, own) => {
|
||||
for (slot, id) in found.iter_mut().zip(ids) {
|
||||
*slot = self.dim_by_id(*id);
|
||||
}
|
||||
if found[0].is_none() {
|
||||
found[0] = *own;
|
||||
}
|
||||
}
|
||||
DimSource::Scales(scales, own) => {
|
||||
for (slot, scale) in found.iter_mut().zip(scales) {
|
||||
*slot = scale.and_then(|s| self.dim_by_scale(g, s));
|
||||
}
|
||||
if found[0].is_none() {
|
||||
found[0] = *own;
|
||||
}
|
||||
}
|
||||
DimSource::Phony(own) => {
|
||||
if rank > 0 {
|
||||
found[0] = *own;
|
||||
}
|
||||
}
|
||||
}
|
||||
let mut dims: Vec<usize> = Vec::with_capacity(rank);
|
||||
for (axis, found) in found.into_iter().enumerate() {
|
||||
let dim = match found {
|
||||
Some(dim) => dim,
|
||||
None => {
|
||||
let var = &self.model.groups[g].vars[v];
|
||||
let (len, unlimited) = (var.extent[axis], var.unlimited[axis]);
|
||||
self.phony_dim(g, len, unlimited, &dims)
|
||||
}
|
||||
};
|
||||
dims.push(dim);
|
||||
}
|
||||
self.model.groups[g].vars[v].dims = dims;
|
||||
}
|
||||
}
|
||||
|
||||
/// The dimension with id `id`: the last one created with it
|
||||
/// (`nc4_find_dim` looks ids up in a file-wide table, where a later
|
||||
/// dimension of the same id replaces an earlier one).
|
||||
fn dim_by_id(&self, id: i64) -> Option<usize> {
|
||||
self.model.dims.iter().rposition(|d| d.id == id)
|
||||
}
|
||||
|
||||
/// The dimension of the scale at `address`, in group `g` or the nearest
|
||||
/// parent that has it.
|
||||
fn dim_by_scale(&self, g: usize, address: u64) -> Option<usize> {
|
||||
let mut group = Some(g);
|
||||
while let Some(i) = group {
|
||||
let found = self.model.groups[i]
|
||||
.dims
|
||||
.iter()
|
||||
.copied()
|
||||
.find(|&d| self.model.dims[d].scale == Some(address));
|
||||
if found.is_some() {
|
||||
return found;
|
||||
}
|
||||
group = self.model.groups[i].parent;
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// `create_phony_dims` for one axis: the first dimension of group `g`
|
||||
/// of length `len` and unlimitedness `unlimited` that no earlier axis
|
||||
/// of the variable uses (`taken`), else a new `phony_dim_<id>`.
|
||||
fn phony_dim(&mut self, g: usize, len: u64, unlimited: bool, taken: &[usize]) -> usize {
|
||||
let dims = &self.model.dims;
|
||||
let existing = self.model.groups[g].dims.iter().copied().find(|&d| {
|
||||
let dim = &dims[d];
|
||||
dim.len == len
|
||||
&& dim.unlimited == unlimited
|
||||
&& !taken.iter().any(|&t| dims[t].id == dim.id)
|
||||
});
|
||||
if let Some(dim) = existing {
|
||||
return dim;
|
||||
}
|
||||
let id = self.take_id();
|
||||
self.add_dim(
|
||||
g,
|
||||
Dim {
|
||||
id,
|
||||
name: format!("phony_dim_{id}"),
|
||||
len,
|
||||
// `nc4_dim_list_add`: a length of 0 is NC_UNLIMITED.
|
||||
unlimited: unlimited || len == 0,
|
||||
scale: None,
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
/// Each unlimited dimension's length: the largest extent along it of
|
||||
/// the variables using it in its group and the groups below.
|
||||
fn unlimited_lengths(&mut self) {
|
||||
let mut owner = vec![0usize; self.model.dims.len()];
|
||||
for (g, group) in self.model.groups.iter().enumerate() {
|
||||
for &d in &group.dims {
|
||||
owner[d] = g;
|
||||
}
|
||||
}
|
||||
let mut lens = vec![0u64; self.model.dims.len()];
|
||||
for (g, group) in self.model.groups.iter().enumerate() {
|
||||
for var in &group.vars {
|
||||
for (&d, &e) in var.dims.iter().zip(&var.extent) {
|
||||
if self.model.dims[d].unlimited && self.is_within(g, owner[d]) {
|
||||
lens[d] = lens[d].max(e);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
for (dim, len) in self.model.dims.iter_mut().zip(lens) {
|
||||
if dim.unlimited {
|
||||
dim.len = len;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether group `g` is `ancestor` or below it.
|
||||
fn is_within(&self, g: usize, ancestor: usize) -> bool {
|
||||
let mut group = Some(g);
|
||||
while let Some(i) = group {
|
||||
if i == ancestor {
|
||||
return true;
|
||||
}
|
||||
group = self.model.groups[i].parent;
|
||||
}
|
||||
false
|
||||
}
|
||||
}
|
||||
|
||||
/// A group's entries in netCDF-C's order: creation order when the group
|
||||
/// tracks it, else the byte order of the names.
|
||||
fn ordered_entries(file: &clawhdf5::File, address: u64) -> Result<Vec<(String, u64)>, Error> {
|
||||
let mut entries = file.group_at(address).entries()?;
|
||||
let order = clawhdf5_format::group_v2::links_in_creation_order_in(
|
||||
file.storage(),
|
||||
file.superblock(),
|
||||
address,
|
||||
)
|
||||
.ok()
|
||||
.flatten();
|
||||
match order {
|
||||
Some(names) => {
|
||||
let position: HashMap<&str, usize> = names
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, n)| (n.as_str(), i))
|
||||
.collect();
|
||||
entries.sort_by_key(|(n, _)| position.get(n.as_str()).copied().unwrap_or(usize::MAX));
|
||||
}
|
||||
None => entries.sort_by(|a, b| a.0.as_bytes().cmp(b.0.as_bytes())),
|
||||
}
|
||||
Ok(entries)
|
||||
}
|
||||
|
||||
fn is_group(header: &ObjectHeader) -> bool {
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
header.messages.iter().any(|m| {
|
||||
matches!(
|
||||
m.msg_type,
|
||||
MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable
|
||||
)
|
||||
})
|
||||
}
|
||||
|
||||
/// The datatype a named datatype's header holds.
|
||||
fn header_datatype(header: &ObjectHeader) -> Option<Datatype> {
|
||||
use clawhdf5_format::message_type::MessageType;
|
||||
let msg = header
|
||||
.messages
|
||||
.iter()
|
||||
.find(|m| m.msg_type == MessageType::Datatype)?;
|
||||
Datatype::parse_in_header(&msg.data, header.version)
|
||||
.ok()
|
||||
.map(|(dt, _)| dt)
|
||||
}
|
||||
|
||||
/// A variable's `_Netcdf4Coordinates`: the id of the dimension of each
|
||||
/// axis.
|
||||
fn coordinates(attrs: &HashMap<String, AttrValue>) -> Option<Vec<i64>> {
|
||||
match attrs.get("_Netcdf4Coordinates")? {
|
||||
AttrValue::I64Array(ids) => Some(ids.clone()),
|
||||
AttrValue::I64(id) => Some(vec![*id]),
|
||||
AttrValue::U64Array(ids) => ids.iter().map(|&id| i64::try_from(id).ok()).collect(),
|
||||
AttrValue::U64(id) => Some(vec![i64::try_from(*id).ok()?]),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// The user-defined types netCDF-C has read so far, and the netCDF type of
|
||||
/// a dataset.
|
||||
///
|
||||
/// netCDF-C keeps every type `read_type` meets in a file-wide list and
|
||||
/// finds one again by `H5Tequal` on the native types — also a type it
|
||||
/// failed to read: `read_type` adds the type (with its class, for a
|
||||
/// compound, enum, variable-length or opaque type) before it looks at the
|
||||
/// members or base type, and does not remove it when one of those fails.
|
||||
/// So a dataset of a type netCDF-C skipped once becomes a variable the
|
||||
/// second time (a compound with a reference member, say), and a compound
|
||||
/// member of a type it skipped (a bit field) is accepted. This replays
|
||||
/// that, except that a dataset whose skipped type has no class (a bit
|
||||
/// field, time, array or complex type) stays hidden: netCDF-C lists the
|
||||
/// second such dataset with an invalid type (class 0).
|
||||
#[derive(Debug, Default)]
|
||||
pub(crate) struct VarTypes {
|
||||
/// Every type read, with its class when it has one netCDF knows.
|
||||
known: Vec<(Datatype, Option<NcType>)>,
|
||||
}
|
||||
|
||||
impl VarTypes {
|
||||
/// The netCDF type netCDF-C gives a dataset of type `dt`
|
||||
/// (`get_type_info2`), or `None` when it skips the dataset
|
||||
/// (`NC_EBADTYPID`): a reference, bit field, time or array type, a
|
||||
/// compound with a member, or an enum or variable-length type with a
|
||||
/// base type, that is not a netCDF atomic type or a type read before
|
||||
/// (see the type docs).
|
||||
///
|
||||
/// Floats other than 4 and 8 bytes (half floats, bfloat16, the 4-, 6-
|
||||
/// and 8-bit floats, `long double`) are `Float` (up to 4 bytes) and
|
||||
/// `Double` here; netCDF-C 4.9.3 on libhdf5 1.14.6 labels them
|
||||
/// `NC_STRING` (their native type matches none of its own).
|
||||
pub fn nc_type(&mut self, dt: &Datatype) -> Option<NcType> {
|
||||
match dt {
|
||||
Datatype::FixedPoint { size, signed, .. } => int_type(*size, *signed),
|
||||
Datatype::FloatingPoint { size, .. } => Some(if *size <= 4 {
|
||||
NcType::Float
|
||||
} else {
|
||||
NcType::Double
|
||||
}),
|
||||
Datatype::String { size, .. } => Some(if *size > 1 {
|
||||
NcType::String
|
||||
} else {
|
||||
NcType::Char
|
||||
}),
|
||||
Datatype::VariableLength {
|
||||
is_string: true, ..
|
||||
} => Some(NcType::String),
|
||||
_ => match self.find(dt) {
|
||||
Some(class) => class,
|
||||
None => self.read_type(dt),
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
/// A named datatype of the file (`read_type` on a committed type).
|
||||
pub fn named(&mut self, dt: &Datatype) {
|
||||
if self.find(dt).is_none() {
|
||||
self.read_type(dt);
|
||||
}
|
||||
}
|
||||
|
||||
/// The type read before that is `dt`, by its class.
|
||||
fn find(&self, dt: &Datatype) -> Option<Option<NcType>> {
|
||||
self.known
|
||||
.iter()
|
||||
.find(|(k, _)| native_eq(k, dt))
|
||||
.map(|&(_, class)| class)
|
||||
}
|
||||
|
||||
/// `read_type`: remember `dt` (not a reference type), then its class if
|
||||
/// netCDF-C can represent its members or base type.
|
||||
fn read_type(&mut self, dt: &Datatype) -> Option<NcType> {
|
||||
let class = match dt {
|
||||
Datatype::Reference { .. } => return None,
|
||||
Datatype::Compound { .. } => Some(NcType::Compound),
|
||||
Datatype::VariableLength { .. } => Some(NcType::VLen),
|
||||
Datatype::Opaque { .. } => Some(NcType::Opaque),
|
||||
Datatype::Enumeration { .. } => Some(NcType::Enum),
|
||||
_ => None,
|
||||
};
|
||||
self.known.push((dt.clone(), class));
|
||||
let parts_ok = match dt {
|
||||
Datatype::Compound { members, .. } => members.iter().all(|m| match &m.datatype {
|
||||
Datatype::Array { base_type, .. } => self.is_member_type(base_type),
|
||||
other => self.is_member_type(other),
|
||||
}),
|
||||
Datatype::VariableLength { base_type, .. }
|
||||
| Datatype::Enumeration { base_type, .. } => self.is_member_type(base_type),
|
||||
_ => true,
|
||||
};
|
||||
class.filter(|_| parts_ok)
|
||||
}
|
||||
|
||||
/// `get_netcdf_type`: whether netCDF-C takes `dt` as the type of a
|
||||
/// compound member or the base of an enum or variable-length type — an
|
||||
/// atomic type (4- and 8-byte floats only) or a type read before.
|
||||
fn is_member_type(&self, dt: &Datatype) -> bool {
|
||||
match dt {
|
||||
Datatype::FixedPoint { size, signed, .. } => int_type(*size, *signed).is_some(),
|
||||
Datatype::FloatingPoint { size: 4 | 8, .. }
|
||||
| Datatype::String { .. }
|
||||
| Datatype::VariableLength {
|
||||
is_string: true, ..
|
||||
} => true,
|
||||
_ => self.find(dt).is_some(),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The netCDF integer type of an HDF5 integer: the native integer libhdf5
|
||||
/// converts it to (`H5Tget_native_type`: the smallest at least as wide).
|
||||
fn int_type(size: u32, signed: bool) -> Option<NcType> {
|
||||
Some(match (size, signed) {
|
||||
(1, true) => NcType::Byte,
|
||||
(1, false) => NcType::UByte,
|
||||
(2, true) => NcType::Short,
|
||||
(2, false) => NcType::UShort,
|
||||
(3..=4, true) => NcType::Int,
|
||||
(3..=4, false) => NcType::UInt,
|
||||
(5..=8, true) => NcType::Int64,
|
||||
(5..=8, false) => NcType::UInt64,
|
||||
_ => return None,
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether two datatypes have the same native type (`H5Tequal` after
|
||||
/// `H5Tget_native_type`): byte order, padding and compound member offsets
|
||||
/// do not count.
|
||||
fn native_eq(a: &Datatype, b: &Datatype) -> bool {
|
||||
use Datatype as D;
|
||||
match (a, b) {
|
||||
(
|
||||
D::FixedPoint {
|
||||
size: s1,
|
||||
signed: g1,
|
||||
..
|
||||
},
|
||||
D::FixedPoint {
|
||||
size: s2,
|
||||
signed: g2,
|
||||
..
|
||||
},
|
||||
) => g1 == g2 && int_type(*s1, *g1) == int_type(*s2, *g2),
|
||||
(D::FloatingPoint { size: s1, .. }, D::FloatingPoint { size: s2, .. }) => s1 == s2,
|
||||
(
|
||||
D::String {
|
||||
size: s1,
|
||||
padding: p1,
|
||||
charset: c1,
|
||||
},
|
||||
D::String {
|
||||
size: s2,
|
||||
padding: p2,
|
||||
charset: c2,
|
||||
},
|
||||
) => s1 == s2 && p1 == p2 && c1 == c2,
|
||||
(
|
||||
D::VariableLength {
|
||||
is_string: i1,
|
||||
base_type: b1,
|
||||
charset: c1,
|
||||
..
|
||||
},
|
||||
D::VariableLength {
|
||||
is_string: i2,
|
||||
base_type: b2,
|
||||
charset: c2,
|
||||
..
|
||||
},
|
||||
) => i1 == i2 && if *i1 { c1 == c2 } else { native_eq(b1, b2) },
|
||||
(D::Compound { members: m1, .. }, D::Compound { members: m2, .. }) => {
|
||||
m1.len() == m2.len()
|
||||
&& m1
|
||||
.iter()
|
||||
.zip(m2)
|
||||
.all(|(x, y)| x.name == y.name && native_eq(&x.datatype, &y.datatype))
|
||||
}
|
||||
(
|
||||
D::Enumeration {
|
||||
base_type: b1,
|
||||
members: m1,
|
||||
..
|
||||
},
|
||||
D::Enumeration {
|
||||
base_type: b2,
|
||||
members: m2,
|
||||
..
|
||||
},
|
||||
) => {
|
||||
native_eq(b1, b2)
|
||||
&& m1.len() == m2.len()
|
||||
&& m1.iter().zip(m2).all(|(x, y)| {
|
||||
x.name == y.name && enum_value(&x.value, b1) == enum_value(&y.value, b2)
|
||||
})
|
||||
}
|
||||
(D::Opaque { size: s1, tag: t1 }, D::Opaque { size: s2, tag: t2 }) => s1 == s2 && t1 == t2,
|
||||
(
|
||||
D::Array {
|
||||
base_type: b1,
|
||||
dimensions: d1,
|
||||
},
|
||||
D::Array {
|
||||
base_type: b2,
|
||||
dimensions: d2,
|
||||
},
|
||||
) => d1 == d2 && native_eq(b1, b2),
|
||||
(
|
||||
D::Reference {
|
||||
size: s1,
|
||||
ref_type: r1,
|
||||
},
|
||||
D::Reference {
|
||||
size: s2,
|
||||
ref_type: r2,
|
||||
},
|
||||
) => s1 == s2 && r1 == r2,
|
||||
(D::BitField { size: s1, .. }, D::BitField { size: s2, .. })
|
||||
| (D::Time { size: s1, .. }, D::Time { size: s2, .. }) => s1 == s2,
|
||||
(
|
||||
D::Complex {
|
||||
size: s1,
|
||||
base_type: b1,
|
||||
},
|
||||
D::Complex {
|
||||
size: s2,
|
||||
base_type: b2,
|
||||
},
|
||||
) => s1 == s2 && native_eq(b1, b2),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// An enum member's value, in the order of its base type's bytes.
|
||||
fn enum_value(value: &[u8], base: &Datatype) -> Vec<u8> {
|
||||
let big = matches!(
|
||||
base,
|
||||
Datatype::FixedPoint {
|
||||
byte_order: DatatypeByteOrder::BigEndian,
|
||||
..
|
||||
}
|
||||
);
|
||||
let mut v = value.to_vec();
|
||||
if big {
|
||||
v.reverse();
|
||||
}
|
||||
v
|
||||
}
|
||||
@@ -1,231 +0,0 @@
|
||||
//! A group's variables and the dimensions they are defined on.
|
||||
//!
|
||||
//! netCDF-C (`libhdf5/hdf5open.c`) gives a variable its dimensions from the
|
||||
//! file, never by size: the dimension ids in its `_Netcdf4Coordinates`
|
||||
//! attribute (each dimension scale's `_Netcdf4Dimid`), else the dimension
|
||||
//! scales its `DIMENSION_LIST` attribute references, looked up in the
|
||||
//! variable's group and then each parent group up to the root. Only an axis
|
||||
//! with neither (a file not written by a netCDF library) gets a dimension
|
||||
//! by size. Dimension scales that are only dimensions are not variables, and
|
||||
//! a variable stored as `_nc4_non_coord_<name>` (a variable sharing a
|
||||
//! dimension's name without being its coordinate variable) is `<name>`.
|
||||
|
||||
use std::collections::{HashMap, HashSet};
|
||||
|
||||
use clawhdf5::AttrValue;
|
||||
|
||||
use crate::dimension::{self, Dimension, GroupDims};
|
||||
use crate::error::Error;
|
||||
use crate::variable::Variable;
|
||||
|
||||
/// The prefix netCDF-C gives the dataset of a variable that has a
|
||||
/// dimension's name but is not that dimension's coordinate variable (the
|
||||
/// dimension's scale holds the name).
|
||||
const NON_COORD_PREFIX: &str = "_nc4_non_coord_";
|
||||
|
||||
/// A group, with the dimensions visible from it.
|
||||
pub(crate) struct Scope<'f> {
|
||||
file: &'f clawhdf5::File,
|
||||
group: clawhdf5::Group<'f>,
|
||||
/// This group's dimensions, then its parent's, and so on to the root's.
|
||||
levels: Vec<GroupDims>,
|
||||
}
|
||||
|
||||
impl<'f> Scope<'f> {
|
||||
/// The group at `path` (`/`-separated from the root; `""` or `"/"` is
|
||||
/// the root).
|
||||
pub fn new(file: &'f clawhdf5::File, path: &str) -> Result<Self, Error> {
|
||||
let parts: Vec<&str> = path.split('/').filter(|p| !p.is_empty()).collect();
|
||||
let mut levels = Vec::with_capacity(parts.len() + 1);
|
||||
for n in (0..=parts.len()).rev() {
|
||||
let group = file.group(&parts[..n].join("/"))?;
|
||||
levels.push(dimension::group_dims(file, &group)?);
|
||||
}
|
||||
let group = file.group(&parts.join("/"))?;
|
||||
Ok(Self {
|
||||
file,
|
||||
group,
|
||||
levels,
|
||||
})
|
||||
}
|
||||
|
||||
/// The group's datasets, as `(dataset name, object header address)` in
|
||||
/// listing order.
|
||||
fn datasets(&self) -> Result<Vec<(String, u64)>, Error> {
|
||||
let datasets: HashSet<String> = self.group.datasets()?.into_iter().collect();
|
||||
Ok(self
|
||||
.group
|
||||
.entries()?
|
||||
.into_iter()
|
||||
.filter(|(name, _)| datasets.contains(name))
|
||||
.collect())
|
||||
}
|
||||
|
||||
/// The group's variables: every dataset but the dimension scales that
|
||||
/// are only dimensions.
|
||||
pub fn variables(&self) -> Result<Vec<Variable<'f>>, Error> {
|
||||
let mut variables = Vec::new();
|
||||
for (ds_name, address) in self.datasets()? {
|
||||
let ds = self.file.dataset_at(address)?;
|
||||
let attrs = ds.attrs()?;
|
||||
if dimension::is_pure_dimension(&attrs) {
|
||||
continue;
|
||||
}
|
||||
variables.push(self.variable_from(nc_name(&ds_name), address, ds, attrs)?);
|
||||
}
|
||||
Ok(variables)
|
||||
}
|
||||
|
||||
/// The names of the group's variables.
|
||||
pub fn variable_names(&self) -> Result<Vec<String>, Error> {
|
||||
let mut names = Vec::new();
|
||||
for (ds_name, address) in self.datasets()? {
|
||||
let attrs = self.file.dataset_at(address)?.attrs()?;
|
||||
if !dimension::is_pure_dimension(&attrs) {
|
||||
names.push(nc_name(&ds_name));
|
||||
}
|
||||
}
|
||||
Ok(names)
|
||||
}
|
||||
|
||||
/// The variable called `name`: the dataset `_nc4_non_coord_<name>` if
|
||||
/// there is one, else the dataset `<name>` unless it is only a
|
||||
/// dimension.
|
||||
pub fn variable(&self, name: &str) -> Result<Variable<'f>, Error> {
|
||||
let not_found = || Error::VariableNotFound(name.to_string());
|
||||
let datasets = self.datasets()?;
|
||||
let prefixed = format!("{NON_COORD_PREFIX}{name}");
|
||||
let address = datasets
|
||||
.iter()
|
||||
.find(|(n, _)| *n == prefixed)
|
||||
.or_else(|| datasets.iter().find(|(n, _)| n == name))
|
||||
.map(|&(_, address)| address)
|
||||
.ok_or_else(not_found)?;
|
||||
let ds = self.file.dataset_at(address)?;
|
||||
let attrs = ds.attrs()?;
|
||||
if dimension::is_pure_dimension(&attrs) {
|
||||
return Err(not_found());
|
||||
}
|
||||
self.variable_from(nc_name(name), address, ds, attrs)
|
||||
}
|
||||
|
||||
fn variable_from(
|
||||
&self,
|
||||
name: String,
|
||||
address: u64,
|
||||
ds: clawhdf5::Dataset<'f>,
|
||||
attrs: HashMap<String, AttrValue>,
|
||||
) -> Result<Variable<'f>, Error> {
|
||||
let shape = ds.shape()?;
|
||||
let dims = self.variable_dims(address, &attrs, &shape);
|
||||
Ok(Variable::new(name, ds, dims, attrs))
|
||||
}
|
||||
|
||||
/// The first dimension, searching this group and then its ancestors,
|
||||
/// that `find` picks.
|
||||
fn find<'a>(
|
||||
&'a self,
|
||||
find: impl Fn(&'a GroupDims) -> Option<&'a Dimension>,
|
||||
) -> Option<Dimension> {
|
||||
self.levels.iter().find_map(find).cloned()
|
||||
}
|
||||
|
||||
/// The dimensions of the dataset at `address`, one per axis of `shape`,
|
||||
/// as netCDF-C resolves them (see the module docs).
|
||||
fn variable_dims(
|
||||
&self,
|
||||
address: u64,
|
||||
attrs: &HashMap<String, AttrValue>,
|
||||
shape: &[u64],
|
||||
) -> Vec<Dimension> {
|
||||
let rank = shape.len();
|
||||
let mut dims: Vec<Option<Dimension>> = vec![None; rank];
|
||||
if rank == 0 {
|
||||
return Vec::new();
|
||||
}
|
||||
// A coordinate variable is the scale of its (first) dimension.
|
||||
dims[0] = self.levels[0].by_address(address).cloned();
|
||||
|
||||
if let Some(ids) = coordinates(attrs).filter(|ids| ids.len() == rank) {
|
||||
for (slot, id) in dims.iter_mut().zip(ids) {
|
||||
if slot.is_none() {
|
||||
*slot = self.find(|level| level.by_dimid(id));
|
||||
}
|
||||
}
|
||||
}
|
||||
if dims.iter().any(Option::is_none)
|
||||
&& let Some(scales) =
|
||||
dimension::dimension_list(self.file, attrs).filter(|s| s.len() == rank)
|
||||
{
|
||||
for (slot, scale) in dims.iter_mut().zip(scales) {
|
||||
if slot.is_none()
|
||||
&& let Some(scale) = scale
|
||||
{
|
||||
*slot = self.find(|level| level.by_address(scale));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Neither: the first dimension of this group of the same size not
|
||||
// already taken by another such axis, else an anonymous one.
|
||||
let own = &self.levels[0].dims;
|
||||
let mut used = vec![false; own.len()];
|
||||
dims.into_iter()
|
||||
.zip(shape)
|
||||
.map(|(dim, &size)| {
|
||||
dim.unwrap_or_else(|| {
|
||||
match own
|
||||
.iter()
|
||||
.enumerate()
|
||||
.find(|&(i, d)| !used[i] && d.size == size)
|
||||
{
|
||||
Some((i, d)) => {
|
||||
used[i] = true;
|
||||
d.clone()
|
||||
}
|
||||
None => Dimension {
|
||||
name: format!("dim_{size}"),
|
||||
size,
|
||||
is_unlimited: false,
|
||||
},
|
||||
}
|
||||
})
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
}
|
||||
|
||||
/// The variable `name` of the group at `group_path`; `name` may itself be
|
||||
/// a path (`"sub/var"`), relative to that group.
|
||||
pub(crate) fn variable_at<'f>(
|
||||
file: &'f clawhdf5::File,
|
||||
group_path: &str,
|
||||
name: &str,
|
||||
) -> Result<Variable<'f>, Error> {
|
||||
match name.trim_start_matches('/').rsplit_once('/') {
|
||||
Some((dir, leaf)) => Scope::new(file, &format!("{group_path}/{dir}"))
|
||||
.map_err(|_| Error::VariableNotFound(name.to_string()))?
|
||||
.variable(leaf),
|
||||
None => Scope::new(file, group_path)?.variable(name.trim_start_matches('/')),
|
||||
}
|
||||
}
|
||||
|
||||
/// The netCDF name of the dataset `ds_name`.
|
||||
fn nc_name(ds_name: &str) -> String {
|
||||
ds_name
|
||||
.strip_prefix(NON_COORD_PREFIX)
|
||||
.unwrap_or(ds_name)
|
||||
.to_string()
|
||||
}
|
||||
|
||||
/// A variable's `_Netcdf4Coordinates`: the `_Netcdf4Dimid` of the dimension
|
||||
/// of each axis.
|
||||
fn coordinates(attrs: &HashMap<String, AttrValue>) -> Option<Vec<i64>> {
|
||||
match attrs.get("_Netcdf4Coordinates")? {
|
||||
AttrValue::I64Array(ids) => Some(ids.clone()),
|
||||
AttrValue::I64(id) => Some(vec![*id]),
|
||||
AttrValue::U64Array(ids) => ids.iter().map(|&id| i64::try_from(id).ok()).collect(),
|
||||
AttrValue::U64(id) => Some(vec![i64::try_from(*id).ok()?]),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
@@ -4,8 +4,16 @@
|
||||
|
||||
use clawhdf5::DType;
|
||||
|
||||
/// NetCDF-4 data types corresponding to the standard NetCDF type system.
|
||||
/// NetCDF-4 data types corresponding to the standard NetCDF type system:
|
||||
/// the atomic types, and the class of a user-defined type.
|
||||
///
|
||||
/// A dataset whose type netCDF-C cannot represent — a reference, bit
|
||||
/// field, time or array type; a compound with such a member, a
|
||||
/// half-precision float member, or a member of a user-defined type not
|
||||
/// read before it; an enum or variable-length type over such a base — is
|
||||
/// not a variable, as in netCDF-C.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||
#[non_exhaustive]
|
||||
pub enum NcType {
|
||||
/// NC_BYTE: signed 8-bit integer
|
||||
Byte,
|
||||
@@ -29,8 +37,18 @@ pub enum NcType {
|
||||
Double,
|
||||
/// NC_STRING: variable-length string
|
||||
String,
|
||||
/// NC_CHAR: fixed-length string / character data
|
||||
/// NC_CHAR: a fixed-length string of one byte (longer ones are
|
||||
/// `String`, as netCDF-C reads them)
|
||||
Char,
|
||||
/// NC_ENUM: an enumeration (user-defined type)
|
||||
Enum,
|
||||
/// NC_COMPOUND: a compound (user-defined type)
|
||||
Compound,
|
||||
/// NC_VLEN: a variable-length sequence (user-defined type)
|
||||
VLen,
|
||||
/// NC_OPAQUE: an opaque type (user-defined type; netCDF4-python skips
|
||||
/// such variables)
|
||||
Opaque,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for NcType {
|
||||
@@ -48,11 +66,16 @@ impl std::fmt::Display for NcType {
|
||||
NcType::Double => write!(f, "NC_DOUBLE"),
|
||||
NcType::String => write!(f, "NC_STRING"),
|
||||
NcType::Char => write!(f, "NC_CHAR"),
|
||||
NcType::Enum => write!(f, "NC_ENUM"),
|
||||
NcType::Compound => write!(f, "NC_COMPOUND"),
|
||||
NcType::VLen => write!(f, "NC_VLEN"),
|
||||
NcType::Opaque => write!(f, "NC_OPAQUE"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Map a clawhdf5 DType to a NetCDF type.
|
||||
/// Map a clawhdf5 DType to a NetCDF type. A variable's own type, as
|
||||
/// netCDF-C reads it, is [`Variable::nc_type`](crate::Variable::nc_type).
|
||||
pub fn dtype_to_nctype(dtype: &DType) -> NcType {
|
||||
match dtype {
|
||||
DType::I8 => NcType::Byte,
|
||||
@@ -66,6 +89,8 @@ pub fn dtype_to_nctype(dtype: &DType) -> NcType {
|
||||
DType::F32 => NcType::Float,
|
||||
DType::F64 => NcType::Double,
|
||||
DType::String | DType::VariableLengthString => NcType::String,
|
||||
DType::Compound(_) => NcType::Compound,
|
||||
DType::Enum(_) => NcType::Enum,
|
||||
_ => NcType::Char, // fallback for other types
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,7 +16,7 @@ use clawhdf5::AttrValue;
|
||||
use crate::cf::{self, CfAttributes, FillValue};
|
||||
use crate::dimension::Dimension;
|
||||
use crate::error::Error;
|
||||
use crate::types::{NcType, dtype_to_nctype};
|
||||
use crate::types::NcType;
|
||||
|
||||
/// A NetCDF-4 variable backed by an HDF5 dataset.
|
||||
pub struct Variable<'f> {
|
||||
@@ -28,6 +28,8 @@ pub struct Variable<'f> {
|
||||
dims: Vec<Dimension>,
|
||||
/// The dataset's attributes.
|
||||
attrs: HashMap<String, AttrValue>,
|
||||
/// Its netCDF type.
|
||||
nc_type: NcType,
|
||||
}
|
||||
|
||||
impl<'f> Variable<'f> {
|
||||
@@ -37,12 +39,14 @@ impl<'f> Variable<'f> {
|
||||
dataset: clawhdf5::Dataset<'f>,
|
||||
dims: Vec<Dimension>,
|
||||
attrs: HashMap<String, AttrValue>,
|
||||
nc_type: NcType,
|
||||
) -> Self {
|
||||
Self {
|
||||
name,
|
||||
dataset,
|
||||
dims,
|
||||
attrs,
|
||||
nc_type,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -51,11 +55,12 @@ impl<'f> Variable<'f> {
|
||||
&self.name
|
||||
}
|
||||
|
||||
/// The dimensions of this variable, one per axis: the ones the file
|
||||
/// gives it (`_Netcdf4Coordinates`, else `DIMENSION_LIST`), found in its
|
||||
/// group or a parent group. An axis the file gives no dimension (a file
|
||||
/// not written by a netCDF library) gets the first dimension of the
|
||||
/// variable's group of the same size, else an anonymous `dim_<size>`.
|
||||
/// The dimensions of this variable, one per axis, as netCDF-C gives
|
||||
/// them: the ones the file names (`_Netcdf4Coordinates`, else the
|
||||
/// scales `DIMENSION_LIST` attaches), found in its group or a parent
|
||||
/// group; for a dataset without them (a file not written by a netCDF
|
||||
/// library), netCDF-C's phony dimensions `phony_dim_<n>`, shared by
|
||||
/// the variables of a group by length.
|
||||
pub fn dimensions(&self) -> &[Dimension] {
|
||||
&self.dims
|
||||
}
|
||||
@@ -75,10 +80,11 @@ impl<'f> Variable<'f> {
|
||||
Ok(self.dataset.shape()?)
|
||||
}
|
||||
|
||||
/// The NetCDF data type of this variable.
|
||||
/// The NetCDF data type of this variable, as netCDF-C reads it (a
|
||||
/// fixed-length string longer than one byte is `String`, one byte
|
||||
/// long `Char`).
|
||||
pub fn nc_type(&self) -> Result<NcType, Error> {
|
||||
let dtype = self.dataset.dtype()?;
|
||||
Ok(dtype_to_nctype(&dtype))
|
||||
Ok(self.nc_type)
|
||||
}
|
||||
|
||||
/// Read all attributes as a HashMap.
|
||||
@@ -297,6 +303,9 @@ fn default_fill(nc_type: NcType) -> FillValue {
|
||||
NcType::Double => FillValue::Float(9.969_209_968_386_869e36),
|
||||
NcType::String => FillValue::String(String::new()),
|
||||
NcType::Char => FillValue::Int(0),
|
||||
// User-defined types have no default fill value in netCDF-C; their
|
||||
// values are not read through these methods.
|
||||
_ => FillValue::Int(0),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user