feat: read array-typed datatypes (incl. array compound members)
The typed read paths (read_as_i32/i64/u64/f32/f64) rejected Array datatypes with a TypeMismatch, so an array-typed compound member (common with N-Bit / reduced-precision data) could not be read. They now unwrap an Array to its base type and read the flat sequence of base elements, recursing for nested arrays. Base-type precision rules (e.g. reduced-precision sign extension) apply to the elements. Validated end-to-end against an HDF5 2.0 compound with an array member: the array field reads [-1, 100, 1000, -32768] with correct 16-bit sign extension. Adds a regression test for flat and nested array reads. Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
This commit is contained in:
@@ -618,6 +618,11 @@ fn get_size(dt: &Datatype) -> usize {
|
|||||||
|
|
||||||
/// Convert raw bytes to `f64` values.
|
/// Convert raw bytes to `f64` values.
|
||||||
pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatError> {
|
pub fn read_as_f64(raw: &[u8], datatype: &Datatype) -> Result<Vec<f64>, FormatError> {
|
||||||
|
// Array datatypes (e.g. an array-typed compound member) are read as a flat
|
||||||
|
// sequence of their base elements.
|
||||||
|
if let Datatype::Array { base_type, .. } = datatype {
|
||||||
|
return read_as_f64(raw, base_type);
|
||||||
|
}
|
||||||
ensure_numeric(datatype, "FloatingPoint or FixedPoint")?;
|
ensure_numeric(datatype, "FloatingPoint or FixedPoint")?;
|
||||||
let elem_size = get_size(datatype);
|
let elem_size = get_size(datatype);
|
||||||
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
||||||
@@ -701,6 +706,9 @@ fn convert_to_f64(
|
|||||||
|
|
||||||
/// Convert raw bytes to `i64` values.
|
/// Convert raw bytes to `i64` values.
|
||||||
pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatError> {
|
pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatError> {
|
||||||
|
if let Datatype::Array { base_type, .. } = datatype {
|
||||||
|
return read_as_i64(raw, base_type);
|
||||||
|
}
|
||||||
ensure_numeric(datatype, "FixedPoint (signed)")?;
|
ensure_numeric(datatype, "FixedPoint (signed)")?;
|
||||||
let elem_size = get_size(datatype);
|
let elem_size = get_size(datatype);
|
||||||
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
||||||
@@ -745,6 +753,9 @@ pub fn read_as_i64(raw: &[u8], datatype: &Datatype) -> Result<Vec<i64>, FormatEr
|
|||||||
|
|
||||||
/// Convert raw bytes to `u64` values.
|
/// Convert raw bytes to `u64` values.
|
||||||
pub fn read_as_u64(raw: &[u8], datatype: &Datatype) -> Result<Vec<u64>, FormatError> {
|
pub fn read_as_u64(raw: &[u8], datatype: &Datatype) -> Result<Vec<u64>, FormatError> {
|
||||||
|
if let Datatype::Array { base_type, .. } = datatype {
|
||||||
|
return read_as_u64(raw, base_type);
|
||||||
|
}
|
||||||
ensure_numeric(datatype, "FixedPoint (unsigned)")?;
|
ensure_numeric(datatype, "FixedPoint (unsigned)")?;
|
||||||
let elem_size = get_size(datatype);
|
let elem_size = get_size(datatype);
|
||||||
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
||||||
@@ -767,6 +778,9 @@ pub fn read_as_u64(raw: &[u8], datatype: &Datatype) -> Result<Vec<u64>, FormatEr
|
|||||||
|
|
||||||
/// Convert raw bytes to `f32` values.
|
/// Convert raw bytes to `f32` values.
|
||||||
pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatError> {
|
pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatError> {
|
||||||
|
if let Datatype::Array { base_type, .. } = datatype {
|
||||||
|
return read_as_f32(raw, base_type);
|
||||||
|
}
|
||||||
ensure_numeric(datatype, "FloatingPoint")?;
|
ensure_numeric(datatype, "FloatingPoint")?;
|
||||||
let elem_size = get_size(datatype);
|
let elem_size = get_size(datatype);
|
||||||
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
||||||
@@ -841,6 +855,9 @@ pub fn read_as_f32(raw: &[u8], datatype: &Datatype) -> Result<Vec<f32>, FormatEr
|
|||||||
|
|
||||||
/// Convert raw bytes to `i32` values.
|
/// Convert raw bytes to `i32` values.
|
||||||
pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatError> {
|
pub fn read_as_i32(raw: &[u8], datatype: &Datatype) -> Result<Vec<i32>, FormatError> {
|
||||||
|
if let Datatype::Array { base_type, .. } = datatype {
|
||||||
|
return read_as_i32(raw, base_type);
|
||||||
|
}
|
||||||
ensure_numeric(datatype, "FixedPoint")?;
|
ensure_numeric(datatype, "FixedPoint")?;
|
||||||
let elem_size = get_size(datatype);
|
let elem_size = get_size(datatype);
|
||||||
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
if elem_size == 0 || !raw.len().is_multiple_of(elem_size) {
|
||||||
@@ -1473,6 +1490,28 @@ mod tests {
|
|||||||
assert_eq!(read_as_i32(&raw, &dt).unwrap(), vec![-1, 42]);
|
assert_eq!(read_as_i32(&raw, &dt).unwrap(), vec![-1, 42]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn array_datatype_reads_flat_base_elements() {
|
||||||
|
// An array-typed (e.g. compound member) datatype reads as a flat
|
||||||
|
// sequence of its base elements, applying base-type precision rules.
|
||||||
|
let arr = Datatype::Array {
|
||||||
|
base_type: Box::new(reduced_int(true, 16)),
|
||||||
|
dimensions: vec![2],
|
||||||
|
};
|
||||||
|
// [-1, 100, 1000, -32768] stored zero-filled at 16-bit precision.
|
||||||
|
let raw: Vec<u8> = vec![
|
||||||
|
0xff, 0xff, 0x00, 0x00, 0x64, 0x00, 0x00, 0x00, 0xe8, 0x03, 0x00, 0x00, 0x00, 0x80,
|
||||||
|
0x00, 0x00,
|
||||||
|
];
|
||||||
|
assert_eq!(read_as_i32(&raw, &arr).unwrap(), vec![-1, 100, 1000, -32768]);
|
||||||
|
// Nested array-of-array unwraps recursively.
|
||||||
|
let nested = Datatype::Array {
|
||||||
|
base_type: Box::new(arr),
|
||||||
|
dimensions: vec![2],
|
||||||
|
};
|
||||||
|
assert_eq!(read_as_i32(&raw, &nested).unwrap(), vec![-1, 100, 1000, -32768]);
|
||||||
|
}
|
||||||
|
|
||||||
fn make_f64_le_type() -> Datatype {
|
fn make_f64_le_type() -> Datatype {
|
||||||
Datatype::FloatingPoint {
|
Datatype::FloatingPoint {
|
||||||
size: 8,
|
size: 8,
|
||||||
|
|||||||
Reference in New Issue
Block a user