//! A file's groups, dimensions and variables, as netCDF-C reads them. //! //! netCDF-C (`libhdf5/hdf5open.c`, 4.9.3) reads a file in two passes, and //! the names, order and sharing of the dimensions depend on both, so this //! module replays them for the whole file at once: //! //! 1. `rec_read_metadata`: each group's links in creation order when the //! group tracks it, else in name order; a group's datasets and named //! datatypes before its subgroups, which follow in the same order. A //! dimension scale (`CLASS` `DIMENSION_SCALE`) defines a dimension with //! the id in its `_Netcdf4Dimid`, else the next free id (ids are //! file-wide), its first extent as length, unlimited when that axis is //! (or the length is 0); it is a variable too unless its `NAME` says it //! is only a dimension. Every other dataset is a variable, unless its //! type is one netCDF-C cannot represent ([`VarTypes::nc_type`]); a //! dataset `_nc4_non_coord_` is the variable ``. //! 2. `rec_match_dimscales`, subgroups first, then the group's variables in //! order: a variable gets the dimensions whose ids its //! `_Netcdf4Coordinates` lists (looked up file-wide), else — when its //! `DIMENSION_LIST` attaches a scale to its first axis — the scales that //! list attaches, looked up in its group and then each parent, else //! "phony" dimensions (`create_phony_dims`): per axis, the first //! dimension of the variable's group of that length and unlimitedness //! not already used by an earlier axis of the variable, else a new one //! called `phony_dim_`. //! //! An unlimited dimension's length is the largest extent, along it, of the //! variables that use it in its group and the groups below //! (`nc4_find_dim_len`). //! //! Where netCDF-C 4.9.3 leaves an axis without a dimension (an id or a scale //! it cannot find, or an axis without a scale of a variable whose first //! axis has one — there it reads uninitialised memory), netCDF4-python //! cannot open the file; this crate gives such an axis a dimension by the //! phony rule instead. use std::collections::HashMap; use clawhdf5::AttrValue; use clawhdf5_format::datatype::{Datatype, DatatypeByteOrder}; use clawhdf5_format::object_header::{ObjectClass, ObjectHeader}; use crate::dimension::{self, Dimension}; use crate::error::Error; use crate::types::NcType; use crate::variable::Variable; /// The prefix netCDF-C gives the dataset of a variable that has a /// dimension's name but is not that dimension's coordinate variable (the /// dimension's scale holds the name). const NON_COORD_PREFIX: &str = "_nc4_non_coord_"; /// Groups read at most, a guard against files whose groups are hard-linked /// into each other many times over (netCDF-C reads each link as its own /// group). const MAX_GROUPS: usize = 100_000; /// A file as netCDF-C sees it. #[derive(Debug)] pub(crate) struct Model { /// Every group; the root is the first. groups: Vec, /// Every dimension, in the order they were created. dims: Vec, } #[derive(Debug)] struct Group { /// Object header address of the HDF5 group. address: u64, /// Its subgroups, `(name, index in Model::groups)`, in netCDF-C's order. children: Vec<(String, usize)>, parent: Option, /// The dimensions it defines (indexes in `Model::dims`), in creation /// order. dims: Vec, /// Its variables, in netCDF-C's order. vars: Vec, } #[derive(Debug)] struct Dim { id: i64, name: String, /// The length it was created with (for an unlimited dimension, replaced /// by its current length once every variable has its dimensions). len: u64, unlimited: bool, /// Object header address of its dimension scale; `None` for a phony /// dimension. scale: Option, } #[derive(Debug)] struct Var { /// The netCDF name. name: String, /// Whether its dataset is `_nc4_non_coord_`. non_coord: bool, address: u64, nc_type: NcType, extent: Vec, /// Whether each axis is unlimited in the dataspace. unlimited: Vec, /// How netCDF-C finds its dimensions. source: DimSource, /// Its dimensions (indexes in `Model::dims`), one per axis, once found. dims: Vec, } #[derive(Debug)] enum DimSource { /// A one-dimensional coordinate variable: its own scale's dimension. OwnScale(usize), /// The ids in `_Netcdf4Coordinates`; for a multi-dimensional coordinate /// variable, also its own dimension (for the first axis, should an id /// not be found). Coordinates(Vec, Option), /// The scale `DIMENSION_LIST` attaches to each axis (the first has one). Scales(Vec>, Option), /// None: phony dimensions. Phony(Option), } impl Model { /// Read the metadata of `file` as netCDF-C does. pub fn build(file: &clawhdf5::File) -> Result { let mut builder = Builder { file, model: Model { groups: Vec::new(), dims: Vec::new(), }, next_id: 0, types: VarTypes::default(), }; let root = file.superblock().root_group_address; builder.model.groups.push(Group { address: root, children: Vec::new(), parent: None, dims: Vec::new(), vars: Vec::new(), }); builder.read_group(0, &mut vec![root])?; builder.match_dims(0); builder.unlimited_lengths(); Ok(builder.model) } /// The group at `path` (`/`-separated names), from the group `from`. pub fn find_group(&self, from: usize, path: &str) -> Option { path.split('/') .filter(|p| !p.is_empty()) .try_fold(from, |g, name| { self.groups[g] .children .iter() .find(|(n, _)| n == name) .map(|&(_, i)| i) }) } /// The object header address of group `g`. pub fn group_address(&self, g: usize) -> u64 { self.groups[g].address } /// The names of group `g`'s subgroups, in netCDF-C's order. pub fn group_names(&self, g: usize) -> Vec { self.groups[g] .children .iter() .map(|(n, _)| n.clone()) .collect() } /// The dimensions group `g` defines, in id order (as `nc_inq_dimids`). pub fn dimensions(&self, g: usize) -> Vec { let mut dims: Vec<&Dim> = self.groups[g].dims.iter().map(|&d| &self.dims[d]).collect(); dims.sort_by_key(|d| d.id); dims.into_iter().map(Dim::dimension).collect() } /// The names of group `g`'s variables, in netCDF-C's order. pub fn variable_names(&self, g: usize) -> Vec { self.groups[g].vars.iter().map(|v| v.name.clone()).collect() } /// Group `g`'s variables, in netCDF-C's order. pub fn variables<'f>( &self, file: &'f clawhdf5::File, g: usize, ) -> Result>, Error> { self.groups[g] .vars .iter() .map(|v| self.open(file, v)) .collect() } /// The variable `name` of group `g`; `name` may be a path (`"sub/var"`) /// relative to it. Of two variables of one name (datasets `` and /// `_nc4_non_coord_`), the second. pub fn variable<'f>( &self, file: &'f clawhdf5::File, g: usize, name: &str, ) -> Result, Error> { let not_found = || Error::VariableNotFound(name.to_string()); let trimmed = name.trim_start_matches('/'); let (g, leaf) = match trimmed.rsplit_once('/') { Some((dir, leaf)) => (self.find_group(g, dir).ok_or_else(not_found)?, leaf), None => (g, trimmed), }; let vars = &self.groups[g].vars; let var = vars .iter() .find(|v| v.name == leaf && v.non_coord) .or_else(|| vars.iter().find(|v| v.name == leaf)) .ok_or_else(not_found)?; self.open(file, var) } fn open<'f>(&self, file: &'f clawhdf5::File, var: &Var) -> Result, Error> { let ds = file.dataset_at(var.address)?; let attrs = ds.attrs().unwrap_or_default(); let dims = var.dims.iter().map(|&d| self.dims[d].dimension()).collect(); Ok(Variable::new( var.name.clone(), ds, dims, attrs, var.nc_type, )) } } impl Dim { fn dimension(&self) -> Dimension { Dimension { name: self.name.clone(), size: self.len, is_unlimited: self.unlimited, } } } struct Builder<'f> { file: &'f clawhdf5::File, model: Model, /// netCDF-C's `next_dimid`. next_id: i64, types: VarTypes, } impl Builder<'_> { /// Pass 1 for group `g` and, after its own links, its subgroups. /// `ancestors` holds the addresses of the groups from the root to `g`, /// so that a group linked into itself is not read forever. fn read_group(&mut self, g: usize, ancestors: &mut Vec) -> Result<(), Error> { let address = self.model.groups[g].address; let sb = self.file.superblock(); let mut subgroups = Vec::new(); for (name, child) in ordered_entries(self.file, address)? { let Ok(header) = ObjectHeader::parse_in(self.file.storage(), child, sb.offset_size, sb.length_size) else { continue; }; match header.object_class() { Some(ObjectClass::Dataset) => self.read_dataset(g, name, child), Some(ObjectClass::NamedDatatype) => { if let Some(dt) = header_datatype(&header) { self.types.named(&dt); } } _ if is_group(&header) => subgroups.push((name, child)), _ => {} } } for (name, child) in subgroups { if ancestors.contains(&child) || self.model.groups.len() >= MAX_GROUPS { continue; } let index = self.model.groups.len(); self.model.groups.push(Group { address: child, children: Vec::new(), parent: Some(g), dims: Vec::new(), vars: Vec::new(), }); self.model.groups[g].children.push((name, index)); ancestors.push(child); let read = self.read_group(index, ancestors); ancestors.pop(); read?; } Ok(()) } /// `read_dataset`: a dimension for a dimension scale, a variable for /// the rest. A dataset that cannot be opened is skipped. fn read_dataset(&mut self, g: usize, name: String, address: u64) { let Ok(ds) = self.file.dataset_at(address) else { return; }; let attrs = ds.attrs().unwrap_or_default(); let Ok(extent) = ds.shape() else { return; }; let max = ds.max_dimensions().ok().flatten(); let unlimited: Vec = (0..extent.len()) .map(|i| matches!(&max, Some(m) if m.get(i) == Some(&u64::MAX))) .collect(); let mut own = None; if dimension::is_dimension_scale(&attrs) && !extent.is_empty() { // read_scale let id = match dimension::get_dimid(&attrs) { Some(id) => { if id >= self.next_id { self.next_id = id.saturating_add(1); } id } None => self.take_id(), }; let len = extent[0]; own = Some(self.add_dim( g, Dim { id, name: name.clone(), len, unlimited: unlimited[0] || len == 0, scale: Some(address), }, )); if dimension::is_pure_dimension(&attrs) { return; } } // read_var: a type netCDF-C cannot represent drops the variable // (the dimension of a scale stays). let Some(nc_type) = ds .raw_datatype() .ok() .and_then(|dt| self.types.nc_type(&dt)) else { return; }; let rank = extent.len(); let coordinates = coordinates(&attrs).filter(|ids| ids.len() == rank && rank > 0); let source = match (own, coordinates) { (Some(dim), _) if rank == 1 => DimSource::OwnScale(dim), (_, Some(ids)) => DimSource::Coordinates(ids, own), _ => match dimension::dimension_list(self.file, &attrs) { Some(scales) if scales.len() == rank && scales.first().is_some_and(Option::is_some) => { DimSource::Scales(scales, own) } _ => DimSource::Phony(own), }, }; let (name, non_coord) = match name.strip_prefix(NON_COORD_PREFIX) { Some(rest) if !rest.is_empty() => (rest.to_string(), true), _ => (name, false), }; self.model.groups[g].vars.push(Var { name, non_coord, address, nc_type, extent, unlimited, source, dims: Vec::new(), }); } fn take_id(&mut self) -> i64 { let id = self.next_id; self.next_id = self.next_id.saturating_add(1); id } fn add_dim(&mut self, g: usize, dim: Dim) -> usize { let index = self.model.dims.len(); self.model.dims.push(dim); self.model.groups[g].dims.push(index); index } /// Pass 2 (`rec_match_dimscales`): subgroups first, then the group's /// variables in order. fn match_dims(&mut self, g: usize) { let children: Vec = self.model.groups[g] .children .iter() .map(|&(_, c)| c) .collect(); for child in children { self.match_dims(child); } for v in 0..self.model.groups[g].vars.len() { let var = &self.model.groups[g].vars[v]; let rank = var.extent.len(); let mut found: Vec> = vec![None; rank]; match &var.source { DimSource::OwnScale(dim) => found[0] = Some(*dim), DimSource::Coordinates(ids, own) => { for (slot, id) in found.iter_mut().zip(ids) { *slot = self.dim_by_id(*id); } if found[0].is_none() { found[0] = *own; } } DimSource::Scales(scales, own) => { for (slot, scale) in found.iter_mut().zip(scales) { *slot = scale.and_then(|s| self.dim_by_scale(g, s)); } if found[0].is_none() { found[0] = *own; } } DimSource::Phony(own) => { if rank > 0 { found[0] = *own; } } } let mut dims: Vec = Vec::with_capacity(rank); for (axis, found) in found.into_iter().enumerate() { let dim = match found { Some(dim) => dim, None => { let var = &self.model.groups[g].vars[v]; let (len, unlimited) = (var.extent[axis], var.unlimited[axis]); self.phony_dim(g, len, unlimited, &dims) } }; dims.push(dim); } self.model.groups[g].vars[v].dims = dims; } } /// The dimension with id `id`: the last one created with it /// (`nc4_find_dim` looks ids up in a file-wide table, where a later /// dimension of the same id replaces an earlier one). fn dim_by_id(&self, id: i64) -> Option { self.model.dims.iter().rposition(|d| d.id == id) } /// The dimension of the scale at `address`, in group `g` or the nearest /// parent that has it. fn dim_by_scale(&self, g: usize, address: u64) -> Option { let mut group = Some(g); while let Some(i) = group { let found = self.model.groups[i] .dims .iter() .copied() .find(|&d| self.model.dims[d].scale == Some(address)); if found.is_some() { return found; } group = self.model.groups[i].parent; } None } /// `create_phony_dims` for one axis: the first dimension of group `g` /// of length `len` and unlimitedness `unlimited` that no earlier axis /// of the variable uses (`taken`), else a new `phony_dim_`. fn phony_dim(&mut self, g: usize, len: u64, unlimited: bool, taken: &[usize]) -> usize { let dims = &self.model.dims; let existing = self.model.groups[g].dims.iter().copied().find(|&d| { let dim = &dims[d]; dim.len == len && dim.unlimited == unlimited && !taken.iter().any(|&t| dims[t].id == dim.id) }); if let Some(dim) = existing { return dim; } let id = self.take_id(); self.add_dim( g, Dim { id, name: format!("phony_dim_{id}"), len, // `nc4_dim_list_add`: a length of 0 is NC_UNLIMITED. unlimited: unlimited || len == 0, scale: None, }, ) } /// Each unlimited dimension's length: the largest extent along it of /// the variables using it in its group and the groups below. fn unlimited_lengths(&mut self) { let mut owner = vec![0usize; self.model.dims.len()]; for (g, group) in self.model.groups.iter().enumerate() { for &d in &group.dims { owner[d] = g; } } let mut lens = vec![0u64; self.model.dims.len()]; for (g, group) in self.model.groups.iter().enumerate() { for var in &group.vars { for (&d, &e) in var.dims.iter().zip(&var.extent) { if self.model.dims[d].unlimited && self.is_within(g, owner[d]) { lens[d] = lens[d].max(e); } } } } for (dim, len) in self.model.dims.iter_mut().zip(lens) { if dim.unlimited { dim.len = len; } } } /// Whether group `g` is `ancestor` or below it. fn is_within(&self, g: usize, ancestor: usize) -> bool { let mut group = Some(g); while let Some(i) = group { if i == ancestor { return true; } group = self.model.groups[i].parent; } false } } /// A group's entries in netCDF-C's order: creation order when the group /// tracks it, else the byte order of the names. fn ordered_entries(file: &clawhdf5::File, address: u64) -> Result, Error> { let mut entries = file.group_at(address).entries()?; let order = clawhdf5_format::group_v2::links_in_creation_order_in( file.storage(), file.superblock(), address, ) .ok() .flatten(); match order { Some(names) => { let position: HashMap<&str, usize> = names .iter() .enumerate() .map(|(i, n)| (n.as_str(), i)) .collect(); entries.sort_by_key(|(n, _)| position.get(n.as_str()).copied().unwrap_or(usize::MAX)); } None => entries.sort_by(|a, b| a.0.as_bytes().cmp(b.0.as_bytes())), } Ok(entries) } fn is_group(header: &ObjectHeader) -> bool { use clawhdf5_format::message_type::MessageType; header.messages.iter().any(|m| { matches!( m.msg_type, MessageType::LinkInfo | MessageType::Link | MessageType::SymbolTable ) }) } /// The datatype a named datatype's header holds. fn header_datatype(header: &ObjectHeader) -> Option { use clawhdf5_format::message_type::MessageType; let msg = header .messages .iter() .find(|m| m.msg_type == MessageType::Datatype)?; Datatype::parse_in_header(&msg.data, header.version) .ok() .map(|(dt, _)| dt) } /// A variable's `_Netcdf4Coordinates`: the id of the dimension of each /// axis. fn coordinates(attrs: &HashMap) -> Option> { match attrs.get("_Netcdf4Coordinates")? { AttrValue::I64Array(ids) => Some(ids.clone()), AttrValue::I64(id) => Some(vec![*id]), AttrValue::U64Array(ids) => ids.iter().map(|&id| i64::try_from(id).ok()).collect(), AttrValue::U64(id) => Some(vec![i64::try_from(*id).ok()?]), _ => None, } } /// The user-defined types netCDF-C has read so far, and the netCDF type of /// a dataset. /// /// netCDF-C keeps every type `read_type` meets in a file-wide list and /// finds one again by `H5Tequal` on the native types — also a type it /// failed to read: `read_type` adds the type (with its class, for a /// compound, enum, variable-length or opaque type) before it looks at the /// members or base type, and does not remove it when one of those fails. /// So a dataset of a type netCDF-C skipped once becomes a variable the /// second time (a compound with a reference member, say), and a compound /// member of a type it skipped (a bit field) is accepted. This replays /// that, except that a dataset whose skipped type has no class (a bit /// field, time, array or complex type) stays hidden: netCDF-C lists the /// second such dataset with an invalid type (class 0). #[derive(Debug, Default)] pub(crate) struct VarTypes { /// Every type read, with its class when it has one netCDF knows. known: Vec<(Datatype, Option)>, } impl VarTypes { /// The netCDF type netCDF-C gives a dataset of type `dt` /// (`get_type_info2`), or `None` when it skips the dataset /// (`NC_EBADTYPID`): a reference, bit field, time or array type, a /// compound with a member, or an enum or variable-length type with a /// base type, that is not a netCDF atomic type or a type read before /// (see the type docs). /// /// Floats other than 4 and 8 bytes (half floats, bfloat16, the 4-, 6- /// and 8-bit floats, `long double`) are `Float` (up to 4 bytes) and /// `Double` here; netCDF-C 4.9.3 on libhdf5 1.14.6 labels them /// `NC_STRING` (their native type matches none of its own). pub fn nc_type(&mut self, dt: &Datatype) -> Option { match dt { Datatype::FixedPoint { size, signed, .. } => int_type(*size, *signed), Datatype::FloatingPoint { size, .. } => Some(if *size <= 4 { NcType::Float } else { NcType::Double }), Datatype::String { size, .. } => Some(if *size > 1 { NcType::String } else { NcType::Char }), Datatype::VariableLength { is_string: true, .. } => Some(NcType::String), _ => match self.find(dt) { Some(class) => class, None => self.read_type(dt), }, } } /// A named datatype of the file (`read_type` on a committed type). pub fn named(&mut self, dt: &Datatype) { if self.find(dt).is_none() { self.read_type(dt); } } /// The type read before that is `dt`, by its class. fn find(&self, dt: &Datatype) -> Option> { self.known .iter() .find(|(k, _)| native_eq(k, dt)) .map(|&(_, class)| class) } /// `read_type`: remember `dt` (not a reference type), then its class if /// netCDF-C can represent its members or base type. fn read_type(&mut self, dt: &Datatype) -> Option { let class = match dt { Datatype::Reference { .. } => return None, Datatype::Compound { .. } => Some(NcType::Compound), Datatype::VariableLength { .. } => Some(NcType::VLen), Datatype::Opaque { .. } => Some(NcType::Opaque), Datatype::Enumeration { .. } => Some(NcType::Enum), _ => None, }; self.known.push((dt.clone(), class)); let parts_ok = match dt { Datatype::Compound { members, .. } => members.iter().all(|m| match &m.datatype { Datatype::Array { base_type, .. } => self.is_member_type(base_type), other => self.is_member_type(other), }), Datatype::VariableLength { base_type, .. } | Datatype::Enumeration { base_type, .. } => self.is_member_type(base_type), _ => true, }; class.filter(|_| parts_ok) } /// `get_netcdf_type`: whether netCDF-C takes `dt` as the type of a /// compound member or the base of an enum or variable-length type — an /// atomic type (4- and 8-byte floats only) or a type read before. fn is_member_type(&self, dt: &Datatype) -> bool { match dt { Datatype::FixedPoint { size, signed, .. } => int_type(*size, *signed).is_some(), Datatype::FloatingPoint { size: 4 | 8, .. } | Datatype::String { .. } | Datatype::VariableLength { is_string: true, .. } => true, _ => self.find(dt).is_some(), } } } /// The netCDF integer type of an HDF5 integer: the native integer libhdf5 /// converts it to (`H5Tget_native_type`: the smallest at least as wide). fn int_type(size: u32, signed: bool) -> Option { Some(match (size, signed) { (1, true) => NcType::Byte, (1, false) => NcType::UByte, (2, true) => NcType::Short, (2, false) => NcType::UShort, (3..=4, true) => NcType::Int, (3..=4, false) => NcType::UInt, (5..=8, true) => NcType::Int64, (5..=8, false) => NcType::UInt64, _ => return None, }) } /// Whether two datatypes have the same native type (`H5Tequal` after /// `H5Tget_native_type`): byte order, padding and compound member offsets /// do not count. fn native_eq(a: &Datatype, b: &Datatype) -> bool { use Datatype as D; match (a, b) { ( D::FixedPoint { size: s1, signed: g1, .. }, D::FixedPoint { size: s2, signed: g2, .. }, ) => g1 == g2 && int_type(*s1, *g1) == int_type(*s2, *g2), (D::FloatingPoint { size: s1, .. }, D::FloatingPoint { size: s2, .. }) => s1 == s2, ( D::String { size: s1, padding: p1, charset: c1, }, D::String { size: s2, padding: p2, charset: c2, }, ) => s1 == s2 && p1 == p2 && c1 == c2, ( D::VariableLength { is_string: i1, base_type: b1, charset: c1, .. }, D::VariableLength { is_string: i2, base_type: b2, charset: c2, .. }, ) => i1 == i2 && if *i1 { c1 == c2 } else { native_eq(b1, b2) }, (D::Compound { members: m1, .. }, D::Compound { members: m2, .. }) => { m1.len() == m2.len() && m1 .iter() .zip(m2) .all(|(x, y)| x.name == y.name && native_eq(&x.datatype, &y.datatype)) } ( D::Enumeration { base_type: b1, members: m1, .. }, D::Enumeration { base_type: b2, members: m2, .. }, ) => { native_eq(b1, b2) && m1.len() == m2.len() && m1.iter().zip(m2).all(|(x, y)| { x.name == y.name && enum_value(&x.value, b1) == enum_value(&y.value, b2) }) } (D::Opaque { size: s1, tag: t1 }, D::Opaque { size: s2, tag: t2 }) => s1 == s2 && t1 == t2, ( D::Array { base_type: b1, dimensions: d1, }, D::Array { base_type: b2, dimensions: d2, }, ) => d1 == d2 && native_eq(b1, b2), ( D::Reference { size: s1, ref_type: r1, }, D::Reference { size: s2, ref_type: r2, }, ) => s1 == s2 && r1 == r2, (D::BitField { size: s1, .. }, D::BitField { size: s2, .. }) | (D::Time { size: s1, .. }, D::Time { size: s2, .. }) => s1 == s2, ( D::Complex { size: s1, base_type: b1, }, D::Complex { size: s2, base_type: b2, }, ) => s1 == s2 && native_eq(b1, b2), _ => false, } } /// An enum member's value, in the order of its base type's bytes. fn enum_value(value: &[u8], base: &Datatype) -> Vec { let big = matches!( base, Datatype::FixedPoint { byte_order: DatatypeByteOrder::BigEndian, .. } ); let mut v = value.to_vec(); if big { v.reverse(); } v }