clawhdf5-netcdf4: variables' dimensions come from the file
Variables got the first unused dimension of equal size, so a variable on an unlimited dimension with fewer records got an anonymous dim_<n>, and dimensions of one size could be swapped. Resolve them as netCDF-C does (libhdf5/hdf5open.c): _Netcdf4Coordinates ids, else the scales DIMENSION_LIST references (the last one attached to an axis), searched in the variable's group and its parents; a coordinate variable is on its own scale. Size matching remains only for axes the file names nothing for. variables()/variable_names() leave out dimension scales that are only dimensions, and _nc4_non_coord_<name> is the variable <name>. Variable::shape is the netCDF shape (an unlimited dimension's length) and the reads pad unwritten records with the fill value (_FillValue, else NC_FILL_*; NaN from read_f64); Variable::stored_shape is the HDF5 extent. New NetCDF4File::variable_names. Tests compare with netCDF4-python variable by variable: the known-issues reproducer, equal sizes, (p, p), scalars, inherited dimensions, unwritten records, h5py dimension scales, h5netcdf and xarray files. CI installs h5netcdf. known-issues entry moved to Fixed (history); stale open-table row for the unlimited-size fix removed. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
@@ -22,73 +22,123 @@ pub struct Dimension {
|
||||
pub is_unlimited: bool,
|
||||
}
|
||||
|
||||
/// Extract dimensions from an HDF5 group (root or subgroup).
|
||||
/// A dimension scale of one group: the dataset that defines a dimension.
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct Scale {
|
||||
/// Object header address of the scale's dataset (what a variable's
|
||||
/// `DIMENSION_LIST` references).
|
||||
pub address: u64,
|
||||
/// Its `_Netcdf4Dimid` (what a variable's `_Netcdf4Coordinates` lists).
|
||||
pub dimid: Option<i64>,
|
||||
/// Index of its dimension in [`GroupDims::dims`].
|
||||
pub dim: usize,
|
||||
}
|
||||
|
||||
/// The dimensions a group defines, with the scales that define them.
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub(crate) struct GroupDims {
|
||||
/// The group's dimensions, in `_Netcdf4Dimid` order (then discovery order).
|
||||
pub dims: Vec<Dimension>,
|
||||
/// The dimension scales behind `dims`; empty when the group has no
|
||||
/// dimension scale and `dims` were inferred from 1-D datasets.
|
||||
pub scales: Vec<Scale>,
|
||||
}
|
||||
|
||||
impl GroupDims {
|
||||
/// The dimension defined by the scale at `address`.
|
||||
pub fn by_address(&self, address: u64) -> Option<&Dimension> {
|
||||
self.scales
|
||||
.iter()
|
||||
.find(|s| s.address == address)
|
||||
.map(|s| &self.dims[s.dim])
|
||||
}
|
||||
|
||||
/// The dimension whose scale has `_Netcdf4Dimid` `id`.
|
||||
pub fn by_dimid(&self, id: i64) -> Option<&Dimension> {
|
||||
self.scales
|
||||
.iter()
|
||||
.find(|s| s.dimid == Some(id))
|
||||
.map(|s| &self.dims[s.dim])
|
||||
}
|
||||
}
|
||||
|
||||
/// The dimensions of an HDF5 group (root or subgroup).
|
||||
///
|
||||
/// NetCDF-4 stores dimensions as datasets with `CLASS=DIMENSION_SCALE`. A fixed
|
||||
/// dimension's size is the dataset's first (and typically only) shape extent.
|
||||
/// Unlimited dimensions have `max_dimensions[0] == u64::MAX` in the HDF5 dataspace;
|
||||
/// their size is computed by `unlimited_len`.
|
||||
pub(crate) fn extract_dimensions(
|
||||
/// their size is computed by `unlimited_len`. A group with no dimension
|
||||
/// scale at all (not written by a netCDF library) gets one dimension per
|
||||
/// 1-D dataset instead.
|
||||
pub(crate) fn group_dims(
|
||||
file: &clawhdf5::File,
|
||||
group: &clawhdf5::Group<'_>,
|
||||
) -> Result<Vec<Dimension>, Error> {
|
||||
) -> Result<GroupDims, Error> {
|
||||
let addresses: HashMap<String, u64> = group.entries()?.into_iter().collect();
|
||||
let dataset_names = group.datasets()?;
|
||||
let mut dims = Vec::new();
|
||||
let mut seen_dimids: HashMap<i64, usize> = HashMap::new();
|
||||
// (dimid, dimension, scale address), in discovery order.
|
||||
let mut found: Vec<(Option<i64>, Dimension, u64)> = Vec::new();
|
||||
|
||||
for ds_name in &dataset_names {
|
||||
let ds = group.dataset(ds_name)?;
|
||||
let attrs = ds.attrs()?;
|
||||
|
||||
// Check if this is a dimension scale
|
||||
if !is_dimension_scale(&attrs) {
|
||||
continue;
|
||||
}
|
||||
|
||||
let Some(&address) = addresses.get(ds_name) else {
|
||||
continue;
|
||||
};
|
||||
let shape = ds.shape()?;
|
||||
let is_unlimited = check_unlimited(file, group, ds_name);
|
||||
let is_unlimited = is_unlimited(&ds);
|
||||
let size = if is_unlimited {
|
||||
unlimited_len(file, &attrs, &shape)
|
||||
} else {
|
||||
shape.first().copied().unwrap_or(0)
|
||||
};
|
||||
|
||||
let dimid = get_dimid(&attrs);
|
||||
|
||||
let dim = Dimension {
|
||||
name: ds_name.clone(),
|
||||
size,
|
||||
is_unlimited,
|
||||
};
|
||||
|
||||
if let Some(id) = dimid {
|
||||
seen_dimids.insert(id, dims.len());
|
||||
}
|
||||
dims.push(dim);
|
||||
found.push((get_dimid(&attrs), dim, address));
|
||||
}
|
||||
|
||||
// Sort by dimid if available, otherwise keep discovery order
|
||||
if !seen_dimids.is_empty() {
|
||||
let mut pairs: Vec<(i64, Dimension)> = Vec::new();
|
||||
let mut unordered = Vec::new();
|
||||
|
||||
for (i, dim) in dims.into_iter().enumerate() {
|
||||
let id = seen_dimids
|
||||
.iter()
|
||||
.find(|(_, idx)| **idx == i)
|
||||
.map(|(k, _)| *k);
|
||||
if let Some(id) = id {
|
||||
pairs.push((id, dim));
|
||||
} else {
|
||||
unordered.push(dim);
|
||||
if found.is_empty() {
|
||||
// Fallback: infer dimensions from dataset shapes and names.
|
||||
// In NetCDF-4, coordinate variables are datasets whose name matches
|
||||
// a dimension name. If there are no explicit DIMENSION_SCALE attributes,
|
||||
// we look for 1-D datasets that might be coordinate variables.
|
||||
let mut dims = Vec::new();
|
||||
for ds_name in &dataset_names {
|
||||
let ds = group.dataset(ds_name)?;
|
||||
let shape = ds.shape()?;
|
||||
if shape.len() == 1 {
|
||||
dims.push(Dimension {
|
||||
name: ds_name.clone(),
|
||||
size: shape[0],
|
||||
is_unlimited: is_unlimited(&ds),
|
||||
});
|
||||
}
|
||||
}
|
||||
pairs.sort_by_key(|(id, _)| *id);
|
||||
dims = pairs.into_iter().map(|(_, d)| d).collect();
|
||||
dims.extend(unordered);
|
||||
return Ok(GroupDims {
|
||||
dims,
|
||||
scales: Vec::new(),
|
||||
});
|
||||
}
|
||||
|
||||
Ok(dims)
|
||||
// By dimid; scales without one keep their discovery order after those
|
||||
// with one (the sort is stable).
|
||||
found.sort_by_key(|(id, ..)| (id.is_none(), id.unwrap_or(0)));
|
||||
let mut out = GroupDims::default();
|
||||
for (i, (dimid, dim, address)) in found.into_iter().enumerate() {
|
||||
out.dims.push(dim);
|
||||
out.scales.push(Scale {
|
||||
address,
|
||||
dimid,
|
||||
dim: i,
|
||||
});
|
||||
}
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
/// The start of the `NAME` attribute netCDF-C gives a dimension scale that
|
||||
@@ -107,10 +157,7 @@ const PURE_DIMENSION_NAME: &str = "This is a netCDF dimension but not a netCDF v
|
||||
/// scale's own extent, as before.
|
||||
fn unlimited_len(file: &clawhdf5::File, attrs: &HashMap<String, AttrValue>, shape: &[u64]) -> u64 {
|
||||
let own = shape.first().copied().unwrap_or(0);
|
||||
let is_variable = !matches!(
|
||||
attrs.get("NAME"),
|
||||
Some(AttrValue::String(n)) if n.starts_with(PURE_DIMENSION_NAME)
|
||||
);
|
||||
let is_variable = !is_pure_dimension(attrs);
|
||||
let Some(refs) = reference_list(file, attrs) else {
|
||||
return own;
|
||||
};
|
||||
@@ -171,8 +218,59 @@ fn reference_list(
|
||||
Some(addresses.into_iter().map(|r| r.address).zip(axes).collect())
|
||||
}
|
||||
|
||||
/// The dimension scale attached to each axis of a variable, from its
|
||||
/// `DIMENSION_LIST` attribute (HDF5 dimension scales: one variable-length
|
||||
/// sequence of object references per axis) — the address of the scale
|
||||
/// netCDF-C takes for the axis, or `None` for an axis with none. netCDF-C's
|
||||
/// `dimscale_visitor` lets `H5DSiterate_scales` visit every scale attached
|
||||
/// to the axis and keeps the last, so with several (h5py's `attach_scale`
|
||||
/// twice) the last one is the axis's dimension. `None` overall when the
|
||||
/// attribute is missing or not in that form.
|
||||
pub(crate) fn dimension_list(
|
||||
file: &clawhdf5::File,
|
||||
attrs: &HashMap<String, AttrValue>,
|
||||
) -> Option<Vec<Option<u64>>> {
|
||||
use clawhdf5_format::data_read::read_object_references;
|
||||
use clawhdf5_format::datatype::Datatype;
|
||||
use clawhdf5_format::vl_data::VlResolver;
|
||||
let Some(AttrValue::Raw { datatype, data, .. }) = attrs.get("DIMENSION_LIST") else {
|
||||
return None;
|
||||
};
|
||||
let Datatype::VariableLength {
|
||||
is_string: false,
|
||||
base_type,
|
||||
..
|
||||
} = datatype
|
||||
else {
|
||||
return None;
|
||||
};
|
||||
let sb = file.superblock();
|
||||
let base_size = usize::try_from(base_type.type_size()).ok()?;
|
||||
let sequences = VlResolver::new_in(file.storage(), sb.offset_size, sb.length_size)
|
||||
.sequences(data, base_size)
|
||||
.ok()?;
|
||||
sequences
|
||||
.iter()
|
||||
.map(|refs| {
|
||||
let refs = read_object_references(refs, base_type, sb.offset_size).ok()?;
|
||||
Some(refs.iter().rev().find(|r| !r.is_null()).map(|r| r.address))
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Whether a dataset is a dimension scale that is only a dimension, not a
|
||||
/// netCDF variable: netCDF-C and h5netcdf give it this `NAME`, and netCDF-C
|
||||
/// does not list it among the variables.
|
||||
pub(crate) fn is_pure_dimension(attrs: &HashMap<String, AttrValue>) -> bool {
|
||||
is_dimension_scale(attrs)
|
||||
&& matches!(
|
||||
attrs.get("NAME"),
|
||||
Some(AttrValue::String(n)) if n.starts_with(PURE_DIMENSION_NAME)
|
||||
)
|
||||
}
|
||||
|
||||
/// Check if a dataset's attributes mark it as a dimension scale.
|
||||
fn is_dimension_scale(attrs: &HashMap<String, AttrValue>) -> bool {
|
||||
pub(crate) fn is_dimension_scale(attrs: &HashMap<String, AttrValue>) -> bool {
|
||||
if let Some(AttrValue::String(class)) = attrs.get("CLASS") {
|
||||
return class == "DIMENSION_SCALE";
|
||||
}
|
||||
@@ -180,7 +278,7 @@ fn is_dimension_scale(attrs: &HashMap<String, AttrValue>) -> bool {
|
||||
}
|
||||
|
||||
/// Get the _Netcdf4Dimid attribute value if present.
|
||||
fn get_dimid(attrs: &HashMap<String, AttrValue>) -> Option<i64> {
|
||||
pub(crate) fn get_dimid(attrs: &HashMap<String, AttrValue>) -> Option<i64> {
|
||||
match attrs.get("_Netcdf4Dimid") {
|
||||
Some(AttrValue::I64(id)) => Some(*id),
|
||||
Some(AttrValue::U64(id)) => Some(*id as i64),
|
||||
@@ -188,56 +286,8 @@ fn get_dimid(attrs: &HashMap<String, AttrValue>) -> Option<i64> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Check if a dimension is unlimited by inspecting the HDF5 dataspace max_dimensions.
|
||||
///
|
||||
/// A dimension is unlimited when `max_dimensions[0] == u64::MAX` in the HDF5 dataspace.
|
||||
fn check_unlimited(_file: &clawhdf5::File, group: &clawhdf5::Group<'_>, ds_name: &str) -> bool {
|
||||
let ds = match group.dataset(ds_name) {
|
||||
Ok(ds) => ds,
|
||||
Err(_) => return false,
|
||||
};
|
||||
|
||||
match ds.max_dimensions() {
|
||||
Ok(Some(max_dims)) => max_dims.first().copied() == Some(u64::MAX),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract dimensions from an HDF5 group using both dimension scale attributes
|
||||
/// and variable DIMENSION_LIST references.
|
||||
///
|
||||
/// This is a more robust approach that also discovers dimensions from variables
|
||||
/// that reference them, even when dimension scales aren't explicitly set.
|
||||
pub(crate) fn extract_dimensions_from_datasets(
|
||||
group: &clawhdf5::Group<'_>,
|
||||
file: &clawhdf5::File,
|
||||
) -> Result<Vec<Dimension>, Error> {
|
||||
// First try the standard approach with DIMENSION_SCALE
|
||||
let mut dims = extract_dimensions(file, group)?;
|
||||
|
||||
// If we found dimensions, return them
|
||||
if !dims.is_empty() {
|
||||
return Ok(dims);
|
||||
}
|
||||
|
||||
// Fallback: infer dimensions from dataset shapes and names.
|
||||
// In NetCDF-4, coordinate variables are datasets whose name matches
|
||||
// a dimension name. If there are no explicit DIMENSION_SCALE attributes,
|
||||
// we look for 1-D datasets that might be coordinate variables.
|
||||
let dataset_names = group.datasets()?;
|
||||
for ds_name in &dataset_names {
|
||||
let ds = group.dataset(ds_name)?;
|
||||
let shape = ds.shape()?;
|
||||
if shape.len() == 1 {
|
||||
// This 1-D dataset could be a coordinate variable / dimension
|
||||
let is_unlimited = check_unlimited(file, group, ds_name);
|
||||
dims.push(Dimension {
|
||||
name: ds_name.clone(),
|
||||
size: shape[0],
|
||||
is_unlimited,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
Ok(dims)
|
||||
/// Whether a dataset's first axis is unlimited (`max_dimensions[0] ==
|
||||
/// u64::MAX` in its dataspace).
|
||||
fn is_unlimited(ds: &clawhdf5::Dataset<'_>) -> bool {
|
||||
matches!(ds.max_dimensions(), Ok(Some(max_dims)) if max_dims.first() == Some(&u64::MAX))
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user