Every RemoteError message and HttpStorage's Debug output held the URL as given, with any user:password@ and the query string — for a presigned S3/GCS/Azure URL, its signature or token. An application logging the error leaked the credential. - New clawhdf5_remote::redact_url: no userinfo, no fragment, query values replaced by REDACTED (plain key names kept). - HttpStorage formats every message with the redacted URL, and scrubs the URL's secret parts from errors of the HTTP client (whose texts can echo the URI); Debug shows the redacted URL. storage_for_url's and the object store URL errors are redacted too. HttpStorage::url() still returns the URL as given, documented as not for logging. - h5rs prints FILE arguments that are URLs redacted: in errors and in dump/stat/check/diff output. - The test server can force a status and send a wrong Content-Range. Tests: 404, 403 (at open and on a read), wrong Content-Range (at open and on a read), no range support, encoded body, ETag change, timeout, connection closed and bad scheme errors, Display and Debug, contain none of the secrets; h5rs likewise for every subcommand. Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
1004 lines
35 KiB
Rust
1004 lines
35 KiB
Rust
//! `h5rs dump`: the whole file (or one dataset) as h5dump-style DDL text or
|
|
//! as hdf5-json.
|
|
|
|
use std::cell::OnceCell;
|
|
use std::collections::HashMap;
|
|
|
|
use clawhdf5_format::attribute::AttributeMessage;
|
|
use clawhdf5_format::data_layout::DataLayout;
|
|
use clawhdf5_format::dataspace::{Dataspace, DataspaceType};
|
|
use clawhdf5_format::datatype::{Datatype, StringPadding};
|
|
use clawhdf5_format::object_header::ObjectHeader;
|
|
use serde_json::{Map, Value as J, json};
|
|
|
|
use crate::cli::{Args, Out};
|
|
use crate::h5::{Error, H5, Kind, Link, LinkKind};
|
|
use crate::info::{self, DsInfo};
|
|
use crate::value::{self, Decoder, Value};
|
|
|
|
pub const USAGE: &str = "\
|
|
usage: h5rs dump [--json] [-A] [-p] [-d PATH] [--max-bytes N] FILE
|
|
|
|
Print FILE's groups, datasets, named datatypes, links and attributes, with
|
|
their values, as h5dump-style DDL text (default) or as JSON.
|
|
|
|
--json hdf5-json layout (see the crate README for the schema)
|
|
-A, --header no dataset values (attribute values are still printed,
|
|
as with h5dump -A)
|
|
-p, --properties also print each dataset's storage layout and filters
|
|
-d, --dataset P dump only the dataset at path P
|
|
--max-bytes N largest dataset or attribute decoded (default 1 GiB);
|
|
a larger one is reported instead of read
|
|
|
|
Exit status: 0 dumped, 1 something could not be read (reported on stderr
|
|
and marked in the output), 2 error.";
|
|
|
|
struct Opts {
|
|
json: bool,
|
|
header_only: bool,
|
|
props: bool,
|
|
}
|
|
|
|
struct Dump<'a> {
|
|
h5: &'a H5,
|
|
opts: Opts,
|
|
problems: usize,
|
|
paths: OnceCell<HashMap<u64, String>>,
|
|
}
|
|
|
|
pub fn run(args: &mut Args, out: &mut Out) -> std::io::Result<i32> {
|
|
let mut opts = Opts {
|
|
json: false,
|
|
header_only: false,
|
|
props: false,
|
|
};
|
|
let mut dataset = None;
|
|
let mut max_bytes = None;
|
|
let mut file = None;
|
|
while let Some(a) = args.next() {
|
|
match a.as_str() {
|
|
"--json" | "-j" => opts.json = true,
|
|
"-A" | "--header" => opts.header_only = true,
|
|
"-p" | "--properties" => opts.props = true,
|
|
"-d" | "--dataset" => match args.value() {
|
|
Some(p) => dataset = Some(p),
|
|
None => return args.usage_error(out, "-d needs a path", USAGE),
|
|
},
|
|
"--max-bytes" => match args.number() {
|
|
Some(n) => max_bytes = Some(n),
|
|
None => return args.usage_error(out, "--max-bytes needs a number", USAGE),
|
|
},
|
|
"-h" | "--help" => {
|
|
writeln!(out.o, "{USAGE}")?;
|
|
return Ok(0);
|
|
}
|
|
s if s.starts_with('-') && s.len() > 1 => {
|
|
return args.usage_error(out, &format!("unknown option {a}"), USAGE);
|
|
}
|
|
_ if file.is_none() => file = Some(a),
|
|
_ => return args.usage_error(out, &format!("unexpected argument {a}"), USAGE),
|
|
}
|
|
}
|
|
let Some(file) = file else {
|
|
return args.usage_error(out, "missing FILE", USAGE);
|
|
};
|
|
let mut h5 = match H5::open_arg(&file) {
|
|
Ok(h) => h,
|
|
Err(e) => {
|
|
writeln!(out.e, "h5rs dump: {e}")?;
|
|
return Ok(2);
|
|
}
|
|
};
|
|
if let Some(m) = max_bytes {
|
|
h5.max_bytes = m;
|
|
}
|
|
let mut d = Dump {
|
|
h5: &h5,
|
|
opts,
|
|
problems: 0,
|
|
paths: OnceCell::new(),
|
|
};
|
|
let file = crate::h5::shown(&file);
|
|
let fname = std::path::Path::new(&file)
|
|
.file_name()
|
|
.map(|s| s.to_string_lossy().into_owned())
|
|
.unwrap_or(file.clone());
|
|
let code = if d.opts.json {
|
|
d.json(out, dataset.as_deref())?
|
|
} else {
|
|
d.ddl(out, &fname, dataset.as_deref())?
|
|
};
|
|
if code != 0 {
|
|
return Ok(code);
|
|
}
|
|
Ok(if d.problems > 0 { 1 } else { 0 })
|
|
}
|
|
|
|
/// A full path for `name` inside the group at `base`.
|
|
fn join(base: &str, name: &str) -> String {
|
|
if base == "/" {
|
|
format!("/{name}")
|
|
} else {
|
|
format!("{base}/{name}")
|
|
}
|
|
}
|
|
|
|
fn quote(s: &str) -> String {
|
|
s.replace('\\', "\\\\").replace('"', "\\\"")
|
|
}
|
|
|
|
impl Dump<'_> {
|
|
/// Object address -> first path, for printing references.
|
|
fn paths(&self) -> &HashMap<u64, String> {
|
|
self.paths.get_or_init(|| {
|
|
let mut m = HashMap::new();
|
|
let _ = self.h5.walk(|it| {
|
|
if let (Some(a), None) = (it.addr, it.first_path) {
|
|
m.entry(a).or_insert_with(|| it.path.to_string());
|
|
}
|
|
});
|
|
m
|
|
})
|
|
}
|
|
|
|
fn problem(&mut self, out: &mut Out, what: &str, e: &Error) -> std::io::Result<()> {
|
|
self.problems += 1;
|
|
writeln!(out.e, "h5rs dump: {what}: {e}")
|
|
}
|
|
|
|
// ------------------------------------------------------------------
|
|
// DDL
|
|
// ------------------------------------------------------------------
|
|
|
|
fn ddl(&mut self, out: &mut Out, fname: &str, only: Option<&str>) -> std::io::Result<i32> {
|
|
writeln!(out.o, "HDF5 \"{}\" {{", quote(fname))?;
|
|
if let Some(p) = only {
|
|
let h = match self.h5.resolve(p).and_then(|a| self.h5.header(a)) {
|
|
Ok(h) => h,
|
|
Err(e) => {
|
|
writeln!(out.o, "}}")?;
|
|
writeln!(out.e, "h5rs dump: {p}: {e}")?;
|
|
return Ok(2);
|
|
}
|
|
};
|
|
if Kind::of(&h) != Kind::Dataset {
|
|
writeln!(out.o, "}}")?;
|
|
writeln!(out.e, "h5rs dump: {p}: not a dataset")?;
|
|
return Ok(2);
|
|
}
|
|
let full = if p.starts_with('/') {
|
|
p.to_string()
|
|
} else {
|
|
format!("/{p}")
|
|
};
|
|
self.ddl_dataset(out, &full, &full, &h, 0)?;
|
|
} else {
|
|
let root = self.h5.root();
|
|
let mut seen = HashMap::new();
|
|
match self.h5.header(root) {
|
|
Ok(h) => self.ddl_group(out, "/", "/", root, &h, 0, &mut seen)?,
|
|
Err(e) => {
|
|
writeln!(out.o, "GROUP \"/\" {{\n}}")?;
|
|
self.problem(out, "/", &e)?;
|
|
}
|
|
}
|
|
}
|
|
writeln!(out.o, "}}")?;
|
|
Ok(0)
|
|
}
|
|
|
|
#[allow(clippy::too_many_arguments)]
|
|
fn ddl_group(
|
|
&mut self,
|
|
out: &mut Out,
|
|
name: &str,
|
|
path: &str,
|
|
addr: u64,
|
|
h: &ObjectHeader,
|
|
ind: usize,
|
|
seen: &mut HashMap<u64, String>,
|
|
) -> std::io::Result<()> {
|
|
let pad = " ".repeat(ind);
|
|
seen.insert(addr, path.to_string());
|
|
writeln!(out.o, "{pad}GROUP \"{}\" {{", quote(name))?;
|
|
self.ddl_attributes(out, path, h, ind + 3)?;
|
|
let links = match self.h5.links(h) {
|
|
Ok(l) => l,
|
|
Err(e) => {
|
|
self.problem(out, path, &e)?;
|
|
Vec::new()
|
|
}
|
|
};
|
|
// The DDL recursion follows the group nesting; bound it (the walk used
|
|
// by the other commands is iterative).
|
|
if seen.len() > crate::h5::MAX_OBJECTS || ind > 3 * MAX_DDL_DEPTH {
|
|
let e = Error::new("too many objects, or groups nested too deeply; stopped");
|
|
self.problem(out, path, &e)?;
|
|
return writeln!(out.o, "{pad}}}");
|
|
}
|
|
for l in &links {
|
|
self.ddl_link(out, path, l, ind + 3, seen)?;
|
|
}
|
|
writeln!(out.o, "{pad}}}")
|
|
}
|
|
|
|
fn ddl_link(
|
|
&mut self,
|
|
out: &mut Out,
|
|
base: &str,
|
|
l: &Link,
|
|
ind: usize,
|
|
seen: &mut HashMap<u64, String>,
|
|
) -> std::io::Result<()> {
|
|
let pad = " ".repeat(ind);
|
|
let inner = " ".repeat(ind + 3);
|
|
let name = quote(&l.name);
|
|
let path = join(base, &l.name);
|
|
match &l.kind {
|
|
LinkKind::Soft(t) => writeln!(
|
|
out.o,
|
|
"{pad}SOFTLINK \"{name}\" {{\n{inner}LINKTARGET \"{}\"\n{pad}}}",
|
|
quote(t)
|
|
),
|
|
LinkKind::External { file, path: p } => writeln!(
|
|
out.o,
|
|
"{pad}EXTERNAL_LINK \"{name}\" {{\n{inner}TARGETFILE \"{}\"\n{inner}TARGETPATH \"{}\"\n{pad}}}",
|
|
quote(file),
|
|
quote(p)
|
|
),
|
|
LinkKind::UserDefined(t) => writeln!(
|
|
out.o,
|
|
"{pad}USERDEFINED_LINK \"{name}\" {{\n{inner}LINKCLASS {t}\n{pad}}}"
|
|
),
|
|
LinkKind::Hard(a) => {
|
|
let h = match self.h5.header(*a) {
|
|
Ok(h) => h,
|
|
Err(e) => {
|
|
self.problem(out, &path, &e)?;
|
|
return writeln!(out.o, "{pad}UNKNOWN_OBJECT \"{name}\" {{\n{pad}}}");
|
|
}
|
|
};
|
|
let kind = Kind::of(&h);
|
|
if let Some(first) = seen.get(a) {
|
|
let word = match kind {
|
|
Kind::Group => "GROUP",
|
|
Kind::Dataset => "DATASET",
|
|
Kind::Datatype => "DATATYPE",
|
|
Kind::Unknown => "OBJECT",
|
|
};
|
|
return writeln!(
|
|
out.o,
|
|
"{pad}{word} \"{name}\" {{\n{inner}HARDLINK \"{}\"\n{pad}}}",
|
|
quote(first)
|
|
);
|
|
}
|
|
match kind {
|
|
Kind::Group => self.ddl_group(out, &l.name, &path, *a, &h, ind, seen),
|
|
Kind::Dataset => {
|
|
seen.insert(*a, path.clone());
|
|
self.ddl_dataset(out, &l.name, &path, &h, ind)
|
|
}
|
|
Kind::Datatype => {
|
|
seen.insert(*a, path.clone());
|
|
match self.h5.datatype(&h) {
|
|
Ok(dt) => writeln!(
|
|
out.o,
|
|
"{pad}DATATYPE \"{name}\" {};",
|
|
crate::dtype::ddl(&dt, ind)
|
|
),
|
|
Err(e) => {
|
|
self.problem(out, &path, &e)?;
|
|
writeln!(out.o, "{pad}DATATYPE \"{name}\" ?;")
|
|
}
|
|
}
|
|
}
|
|
Kind::Unknown => {
|
|
seen.insert(*a, path.clone());
|
|
writeln!(out.o, "{pad}UNKNOWN_OBJECT \"{name}\" {{\n{pad}}}")
|
|
}
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
fn ddl_dataset(
|
|
&mut self,
|
|
out: &mut Out,
|
|
name: &str,
|
|
path: &str,
|
|
h: &ObjectHeader,
|
|
ind: usize,
|
|
) -> std::io::Result<()> {
|
|
let pad = " ".repeat(ind);
|
|
let inner = " ".repeat(ind + 3);
|
|
writeln!(out.o, "{pad}DATASET \"{}\" {{", quote(name))?;
|
|
let info = DsInfo::read(self.h5, path, h);
|
|
match &info.dt {
|
|
Ok(dt) => writeln!(out.o, "{inner}DATATYPE {}", crate::dtype::ddl(dt, ind + 3))?,
|
|
Err(e) => self.problem(out, path, e)?,
|
|
}
|
|
match &info.ds {
|
|
Ok(ds) => writeln!(out.o, "{inner}DATASPACE {}", info::dataspace_ddl(ds))?,
|
|
Err(e) => self.problem(out, path, e)?,
|
|
}
|
|
if self.opts.props {
|
|
self.ddl_properties(out, path, &info, ind + 3)?;
|
|
}
|
|
if !self.opts.header_only
|
|
&& let (Ok(dt), Ok(ds)) = (&info.dt, &info.ds)
|
|
{
|
|
match self.h5.read_dataset(path, dt, ds) {
|
|
Ok(raw) => self.ddl_data(out, dt, ds, &raw, ind + 3)?,
|
|
Err(e) => {
|
|
writeln!(out.o, "{inner}DATA {{\n{inner}<not read: {e}>\n{inner}}}")?;
|
|
self.problem(out, path, &e)?;
|
|
}
|
|
}
|
|
}
|
|
self.ddl_attributes(out, path, h, ind + 3)?;
|
|
writeln!(out.o, "{pad}}}")
|
|
}
|
|
|
|
fn ddl_properties(
|
|
&mut self,
|
|
out: &mut Out,
|
|
path: &str,
|
|
info: &DsInfo,
|
|
ind: usize,
|
|
) -> std::io::Result<()> {
|
|
let pad = " ".repeat(ind);
|
|
let inner = " ".repeat(ind + 3);
|
|
let Ok(layout) = &info.layout else {
|
|
if let Err(e) = &info.layout {
|
|
self.problem(out, path, e)?;
|
|
}
|
|
return Ok(());
|
|
};
|
|
writeln!(out.o, "{pad}STORAGE_LAYOUT {{")?;
|
|
match layout {
|
|
DataLayout::Chunked {
|
|
chunk_dimensions, ..
|
|
} => {
|
|
let rank = chunk_dimensions.len().saturating_sub(1);
|
|
writeln!(
|
|
out.o,
|
|
"{inner}CHUNKED ( {} )",
|
|
chunk_dimensions[..rank]
|
|
.iter()
|
|
.map(|d| d.to_string())
|
|
.collect::<Vec<_>>()
|
|
.join(", ")
|
|
)?;
|
|
writeln!(
|
|
out.o,
|
|
"{inner}INDEX {}",
|
|
info::chunk_index_name(layout).to_uppercase()
|
|
)?;
|
|
}
|
|
DataLayout::Contiguous { address, size } => {
|
|
writeln!(out.o, "{inner}CONTIGUOUS")?;
|
|
match address {
|
|
Some(a) => writeln!(out.o, "{inner}OFFSET {a}")?,
|
|
None => writeln!(out.o, "{inner}NOT ALLOCATED")?,
|
|
}
|
|
writeln!(out.o, "{inner}SIZE {size}")?;
|
|
}
|
|
DataLayout::Compact { data } => {
|
|
writeln!(out.o, "{inner}COMPACT")?;
|
|
writeln!(out.o, "{inner}SIZE {}", data.len())?;
|
|
}
|
|
DataLayout::Virtual { .. } => writeln!(out.o, "{inner}VIRTUAL")?,
|
|
}
|
|
if let Ok(a) = info::allocated_bytes(self.h5, info)
|
|
&& matches!(layout, DataLayout::Chunked { .. })
|
|
{
|
|
writeln!(out.o, "{inner}SIZE {a}")?;
|
|
}
|
|
writeln!(out.o, "{pad}}}")?;
|
|
writeln!(out.o, "{pad}FILTERS {{")?;
|
|
match &info.filters {
|
|
Ok(Some(p)) if !p.filters.is_empty() => {
|
|
for f in &p.filters {
|
|
writeln!(
|
|
out.o,
|
|
"{inner}{} {{ ID {}; PARAMS {{ {} }} }}",
|
|
info::filter_name(f).to_uppercase(),
|
|
f.filter_id,
|
|
f.client_data
|
|
.iter()
|
|
.map(|c| c.to_string())
|
|
.collect::<Vec<_>>()
|
|
.join(" ")
|
|
)?;
|
|
}
|
|
}
|
|
Ok(_) => writeln!(out.o, "{inner}NONE")?,
|
|
Err(e) => self.problem(out, path, e)?,
|
|
}
|
|
writeln!(out.o, "{pad}}}")
|
|
}
|
|
|
|
fn ddl_attributes(
|
|
&mut self,
|
|
out: &mut Out,
|
|
path: &str,
|
|
h: &ObjectHeader,
|
|
ind: usize,
|
|
) -> std::io::Result<()> {
|
|
let (attrs, errs) = match self.h5.attributes(h) {
|
|
Ok(x) => x,
|
|
Err(e) => return self.problem(out, path, &e),
|
|
};
|
|
for e in errs {
|
|
self.problem(out, path, &Error::new(format!("attribute: {e}")))?;
|
|
}
|
|
for a in &attrs {
|
|
self.ddl_attribute(out, path, a, ind)?;
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
fn ddl_attribute(
|
|
&mut self,
|
|
out: &mut Out,
|
|
path: &str,
|
|
a: &AttributeMessage,
|
|
ind: usize,
|
|
) -> std::io::Result<()> {
|
|
let pad = " ".repeat(ind);
|
|
let inner = " ".repeat(ind + 3);
|
|
writeln!(out.o, "{pad}ATTRIBUTE \"{}\" {{", quote(&a.name))?;
|
|
writeln!(
|
|
out.o,
|
|
"{inner}DATATYPE {}",
|
|
crate::dtype::ddl(&a.datatype, ind + 3)
|
|
)?;
|
|
writeln!(
|
|
out.o,
|
|
"{inner}DATASPACE {}",
|
|
info::dataspace_ddl(&a.dataspace)
|
|
)?;
|
|
{
|
|
match self.checked_attr(a) {
|
|
Ok(()) => self.ddl_data(out, &a.datatype, &a.dataspace, &a.raw_data, ind + 3)?,
|
|
Err(e) => {
|
|
writeln!(out.o, "{inner}DATA {{\n{inner}<not read: {e}>\n{inner}}}")?;
|
|
self.problem(out, &format!("{path} attribute {}", a.name), &e)?;
|
|
}
|
|
}
|
|
}
|
|
writeln!(out.o, "{pad}}}")
|
|
}
|
|
|
|
/// An attribute's value holds as many bytes as its dataspace says.
|
|
fn checked_attr(&self, a: &AttributeMessage) -> crate::h5::Result<()> {
|
|
let need = crate::h5::byte_len(&a.dataspace, &a.datatype)?;
|
|
if need > self.h5.max_bytes {
|
|
return Err(Error::new(format!(
|
|
"attribute is {need} bytes, over the {} byte limit (--max-bytes)",
|
|
self.h5.max_bytes
|
|
)));
|
|
}
|
|
if (a.raw_data.len() as u64) < need {
|
|
return Err(Error::new(format!(
|
|
"attribute holds {} bytes, its dataspace needs {need}",
|
|
a.raw_data.len()
|
|
)));
|
|
}
|
|
Ok(())
|
|
}
|
|
|
|
/// An h5dump DATA block: elements row by row, each line starting with
|
|
/// the index of its first element, wrapped where h5dump wraps (78 columns).
|
|
fn ddl_data(
|
|
&mut self,
|
|
out: &mut Out,
|
|
dt: &Datatype,
|
|
ds: &Dataspace,
|
|
raw: &[u8],
|
|
ind: usize,
|
|
) -> std::io::Result<()> {
|
|
let pad = " ".repeat(ind);
|
|
writeln!(out.o, "{pad}DATA {{")?;
|
|
let dims: Vec<u64> = match ds.space_type {
|
|
DataspaceType::Null => {
|
|
return writeln!(out.o, "{pad}}}");
|
|
}
|
|
DataspaceType::Scalar => vec![1],
|
|
DataspaceType::Simple => ds.dimensions.clone(),
|
|
};
|
|
let n = crate::h5::num_elements(ds).unwrap_or(0) as usize;
|
|
let dec = Decoder::new(self.h5);
|
|
let last = dims.last().copied().unwrap_or(1).max(1) as usize;
|
|
let paths = |a: u64| self.paths().get(&a).cloned();
|
|
let mut line = String::new();
|
|
let mut errors = 0usize;
|
|
for i in 0..n {
|
|
let v = dec.element(dt, raw, i);
|
|
if matches!(v, Value::Error(_)) {
|
|
errors += 1;
|
|
}
|
|
let comma = if i + 1 < n { "," } else { "" };
|
|
let esize = dt.type_size() as usize;
|
|
let eb = raw.get(i * esize..(i + 1) * esize).unwrap_or_default();
|
|
if let (Value::Compound(_), Datatype::Compound { members, .. }) = (&v, dt) {
|
|
// h5dump prints each compound element as a block, one
|
|
// member per line.
|
|
if !line.is_empty() {
|
|
writeln!(out.o, "{line}")?;
|
|
line.clear();
|
|
}
|
|
writeln!(out.o, "{pad}({}): {{", index_text(i as u64, &dims))?;
|
|
for (k, m) in members.iter().enumerate() {
|
|
let sep = if k + 1 < members.len() { "," } else { "" };
|
|
let mb = usize::try_from(m.byte_offset)
|
|
.ok()
|
|
.and_then(|o| eb.get(o..))
|
|
.unwrap_or_default();
|
|
let t = ddl_text(&dec, &m.datatype, mb, &paths, 1);
|
|
writeln!(out.o, "{pad} {t}{sep}")?;
|
|
}
|
|
writeln!(out.o, "{pad} }}{comma}")?;
|
|
continue;
|
|
}
|
|
let t = if matches!(v, Value::Error(_)) {
|
|
format!("{}{comma}", value::text(&v, &paths))
|
|
} else {
|
|
format!("{}{comma}", ddl_text(&dec, dt, eb, &paths, 0))
|
|
};
|
|
let row_start = i % last == 0;
|
|
if row_start || line.len() + 1 + t.len() > 77 {
|
|
if !line.is_empty() {
|
|
writeln!(out.o, "{line}")?;
|
|
}
|
|
line = format!("{pad}({}): {t}", index_text(i as u64, &dims));
|
|
} else {
|
|
line.push(' ');
|
|
line.push_str(&t);
|
|
}
|
|
}
|
|
if !line.is_empty() {
|
|
writeln!(out.o, "{line}")?;
|
|
}
|
|
if errors > 0 {
|
|
self.problems += 1;
|
|
writeln!(out.e, "h5rs dump: {errors} element(s) could not be decoded")?;
|
|
}
|
|
writeln!(out.o, "{pad}}}")
|
|
}
|
|
|
|
// ------------------------------------------------------------------
|
|
// JSON
|
|
// ------------------------------------------------------------------
|
|
|
|
fn json(&mut self, out: &mut Out, only: Option<&str>) -> std::io::Result<i32> {
|
|
let mut groups = Map::new();
|
|
let mut datasets = Map::new();
|
|
let mut datatypes = Map::new();
|
|
let mut doc = Map::new();
|
|
doc.insert("apiVersion".into(), json!("1.1.1"));
|
|
if let Some(p) = only {
|
|
let full = if p.starts_with('/') {
|
|
p.to_string()
|
|
} else {
|
|
format!("/{p}")
|
|
};
|
|
let r = self
|
|
.h5
|
|
.resolve(&full)
|
|
.and_then(|a| Ok((a, self.h5.header(a)?)));
|
|
match r {
|
|
Ok((a, h)) if Kind::of(&h) == Kind::Dataset => {
|
|
let obj = self.json_dataset(out, &full, &h)?;
|
|
datasets.insert(obj_id(Kind::Dataset, a), obj);
|
|
}
|
|
Ok(_) => {
|
|
writeln!(out.e, "h5rs dump: {p}: not a dataset")?;
|
|
return Ok(2);
|
|
}
|
|
Err(e) => {
|
|
writeln!(out.e, "h5rs dump: {p}: {e}")?;
|
|
return Ok(2);
|
|
}
|
|
}
|
|
} else {
|
|
doc.insert("root".into(), json!(obj_id(Kind::Group, self.h5.root())));
|
|
let (items, walk) = self.h5.walk_collect();
|
|
if let Err(e) = walk {
|
|
self.problem(out, "/", &e)?;
|
|
}
|
|
// Aliases: every path an object is reachable by.
|
|
let mut aliases: HashMap<u64, Vec<String>> = HashMap::new();
|
|
for it in &items {
|
|
if let Some(a) = it.addr {
|
|
aliases.entry(a).or_default().push(it.path.clone());
|
|
}
|
|
}
|
|
for it in items {
|
|
let path = it.path;
|
|
let (Some(addr), None, Some(header)) = (it.addr, it.first_path, it.header) else {
|
|
continue;
|
|
};
|
|
let h = match header {
|
|
Ok(h) => h,
|
|
Err(e) => {
|
|
self.problem(out, &path, &e)?;
|
|
continue;
|
|
}
|
|
};
|
|
let kind = Kind::of(&h);
|
|
let alias = aliases.remove(&addr).unwrap_or_default();
|
|
let mut obj = match kind {
|
|
Kind::Dataset => self.json_dataset(out, &path, &h)?,
|
|
Kind::Datatype => {
|
|
let mut m = Map::new();
|
|
match self.h5.datatype(&h) {
|
|
Ok(dt) => {
|
|
m.insert("type".into(), crate::dtype::json(&dt));
|
|
}
|
|
Err(e) => self.problem(out, &path, &e)?,
|
|
}
|
|
m.insert("attributes".into(), self.json_attributes(out, &path, &h)?);
|
|
J::Object(m)
|
|
}
|
|
_ => {
|
|
let mut m = Map::new();
|
|
m.insert("attributes".into(), self.json_attributes(out, &path, &h)?);
|
|
let links = match self.h5.links(&h) {
|
|
Ok(l) => l,
|
|
Err(e) => {
|
|
if kind == Kind::Group {
|
|
self.problem(out, &path, &e)?;
|
|
}
|
|
Vec::new()
|
|
}
|
|
};
|
|
let mut jl = Vec::new();
|
|
for l in &links {
|
|
jl.push(self.json_link(l));
|
|
}
|
|
m.insert("links".into(), J::Array(jl));
|
|
J::Object(m)
|
|
}
|
|
};
|
|
if let J::Object(m) = &mut obj {
|
|
m.insert("alias".into(), json!(alias));
|
|
}
|
|
let id = obj_id(kind, addr);
|
|
match kind {
|
|
Kind::Dataset => datasets.insert(id, obj),
|
|
Kind::Datatype => datatypes.insert(id, obj),
|
|
_ => groups.insert(id, obj),
|
|
};
|
|
}
|
|
doc.insert("groups".into(), J::Object(groups));
|
|
}
|
|
doc.insert("datasets".into(), J::Object(datasets));
|
|
if only.is_none() {
|
|
doc.insert("datatypes".into(), J::Object(datatypes));
|
|
}
|
|
let text = serde_json::to_string_pretty(&J::Object(doc)).map_err(std::io::Error::other)?;
|
|
writeln!(out.o, "{text}")?;
|
|
Ok(0)
|
|
}
|
|
|
|
fn json_link(&self, l: &Link) -> J {
|
|
match &l.kind {
|
|
LinkKind::Hard(a) => {
|
|
let kind = self
|
|
.h5
|
|
.header(*a)
|
|
.map(|h| Kind::of(&h))
|
|
.unwrap_or(Kind::Unknown);
|
|
let coll = match kind {
|
|
Kind::Dataset => "datasets",
|
|
Kind::Datatype => "datatypes",
|
|
_ => "groups",
|
|
};
|
|
json!({"class": "H5L_TYPE_HARD", "title": l.name, "collection": coll,
|
|
"id": obj_id(kind, *a)})
|
|
}
|
|
LinkKind::Soft(t) => json!({"class": "H5L_TYPE_SOFT", "title": l.name, "h5path": t}),
|
|
LinkKind::External { file, path } => json!({"class": "H5L_TYPE_EXTERNAL",
|
|
"title": l.name, "file": file, "h5path": path}),
|
|
LinkKind::UserDefined(t) => json!({"class": "H5L_TYPE_USER_DEFINED",
|
|
"title": l.name, "linkClass": t}),
|
|
}
|
|
}
|
|
|
|
fn json_dataset(&mut self, out: &mut Out, path: &str, h: &ObjectHeader) -> std::io::Result<J> {
|
|
let info = DsInfo::read(self.h5, path, h);
|
|
let mut m = Map::new();
|
|
match &info.dt {
|
|
Ok(dt) => {
|
|
m.insert("type".into(), crate::dtype::json(dt));
|
|
}
|
|
Err(e) => self.problem(out, path, e)?,
|
|
}
|
|
match &info.ds {
|
|
Ok(ds) => {
|
|
m.insert("shape".into(), shape_json(ds));
|
|
}
|
|
Err(e) => self.problem(out, path, e)?,
|
|
}
|
|
let mut cp = Map::new();
|
|
if let Ok(l) = &info.layout {
|
|
let mut lj = Map::new();
|
|
lj.insert(
|
|
"class".into(),
|
|
json!(match l {
|
|
DataLayout::Compact { .. } => "H5D_COMPACT",
|
|
DataLayout::Contiguous { .. } => "H5D_CONTIGUOUS",
|
|
DataLayout::Chunked { .. } => "H5D_CHUNKED",
|
|
DataLayout::Virtual { .. } => "H5D_VIRTUAL",
|
|
}),
|
|
);
|
|
if let DataLayout::Chunked {
|
|
chunk_dimensions, ..
|
|
} = l
|
|
{
|
|
let rank = chunk_dimensions.len().saturating_sub(1);
|
|
lj.insert("dims".into(), json!(chunk_dimensions[..rank]));
|
|
}
|
|
cp.insert("layout".into(), J::Object(lj));
|
|
} else if let Err(e) = &info.layout {
|
|
self.problem(out, path, e)?;
|
|
}
|
|
if let Ok(Some(p)) = &info.filters {
|
|
let fl: Vec<J> = p
|
|
.filters
|
|
.iter()
|
|
.map(|f| {
|
|
json!({"id": f.filter_id, "name": info::filter_name(f),
|
|
"class": filter_class(f.filter_id), "parameters": f.client_data})
|
|
})
|
|
.collect();
|
|
cp.insert("filters".into(), J::Array(fl));
|
|
}
|
|
m.insert("creationProperties".into(), J::Object(cp));
|
|
if !self.opts.header_only
|
|
&& let (Ok(dt), Ok(ds)) = (&info.dt, &info.ds)
|
|
{
|
|
// JSON values are built in memory, some 32+ bytes per element:
|
|
// hold them to the same budget as the raw data.
|
|
let n = crate::h5::num_elements(ds).unwrap_or(u64::MAX);
|
|
let read = if n.saturating_mul(JSON_BYTES_PER_ELEMENT) > self.h5.max_bytes {
|
|
Err(Error::new(format!(
|
|
"{n} elements are too many to hold as JSON within --max-bytes {}",
|
|
self.h5.max_bytes
|
|
))
|
|
.with_kind(crate::h5::ErrorKind::Limit))
|
|
} else {
|
|
self.h5.read_dataset(path, dt, ds)
|
|
};
|
|
match read {
|
|
Ok(raw) => {
|
|
let v = self.json_values(out, dt, ds, &raw)?;
|
|
m.insert("value".into(), v);
|
|
}
|
|
Err(e) => {
|
|
m.insert("value_error".into(), json!(e.to_string()));
|
|
self.problem(out, path, &e)?;
|
|
}
|
|
}
|
|
}
|
|
m.insert("attributes".into(), self.json_attributes(out, path, h)?);
|
|
Ok(J::Object(m))
|
|
}
|
|
|
|
fn json_attributes(
|
|
&mut self,
|
|
out: &mut Out,
|
|
path: &str,
|
|
h: &ObjectHeader,
|
|
) -> std::io::Result<J> {
|
|
let (attrs, errs) = match self.h5.attributes(h) {
|
|
Ok(x) => x,
|
|
Err(e) => {
|
|
self.problem(out, path, &e)?;
|
|
return Ok(J::Array(Vec::new()));
|
|
}
|
|
};
|
|
for e in errs {
|
|
self.problem(out, path, &Error::new(format!("attribute: {e}")))?;
|
|
}
|
|
let mut v = Vec::new();
|
|
for a in &attrs {
|
|
let mut m = Map::new();
|
|
m.insert("name".into(), json!(a.name));
|
|
m.insert("shape".into(), shape_json(&a.dataspace));
|
|
m.insert("type".into(), crate::dtype::json(&a.datatype));
|
|
{
|
|
match self.checked_attr(a) {
|
|
Ok(()) => {
|
|
let val = self.json_values(out, &a.datatype, &a.dataspace, &a.raw_data)?;
|
|
m.insert("value".into(), val);
|
|
}
|
|
Err(e) => {
|
|
m.insert("value_error".into(), json!(e.to_string()));
|
|
self.problem(out, &format!("{path} attribute {}", a.name), &e)?;
|
|
}
|
|
}
|
|
}
|
|
v.push(J::Object(m));
|
|
}
|
|
Ok(J::Array(v))
|
|
}
|
|
|
|
fn json_values(
|
|
&mut self,
|
|
out: &mut Out,
|
|
dt: &Datatype,
|
|
ds: &Dataspace,
|
|
raw: &[u8],
|
|
) -> std::io::Result<J> {
|
|
let n = crate::h5::num_elements(ds).unwrap_or(0) as usize;
|
|
let dec = Decoder::new(self.h5);
|
|
let paths = |a: u64| self.paths().get(&a).cloned();
|
|
let mut flat = Vec::with_capacity(n);
|
|
let mut errors = 0usize;
|
|
for i in 0..n {
|
|
let v = dec.element(dt, raw, i);
|
|
if matches!(v, Value::Error(_)) {
|
|
errors += 1;
|
|
}
|
|
flat.push(value::to_json(&v, &paths));
|
|
}
|
|
if errors > 0 {
|
|
self.problems += 1;
|
|
writeln!(out.e, "h5rs dump: {errors} element(s) could not be decoded")?;
|
|
}
|
|
Ok(match ds.space_type {
|
|
DataspaceType::Null => J::Null,
|
|
DataspaceType::Scalar => flat.into_iter().next().unwrap_or(J::Null),
|
|
DataspaceType::Simple => nest(flat, &ds.dimensions),
|
|
})
|
|
}
|
|
}
|
|
|
|
/// Deepest group nesting the DDL output follows.
|
|
const MAX_DDL_DEPTH: usize = 256;
|
|
|
|
/// Memory budgeted per element for a dataset's value held as JSON.
|
|
const JSON_BYTES_PER_ELEMENT: u64 = 64;
|
|
|
|
/// `(i,j,k)` index of flat element `i` of an array with `dims`.
|
|
/// One element as h5dump prints it in a DATA block: [`value::text`], except
|
|
/// that a null-padded fixed string shows its padding (every byte, NULs as
|
|
/// `\000`) at any depth, in compound members and array elements too.
|
|
fn ddl_text(
|
|
dec: &Decoder,
|
|
dt: &Datatype,
|
|
b: &[u8],
|
|
paths: &dyn Fn(u64) -> Option<String>,
|
|
depth: u32,
|
|
) -> String {
|
|
let err = || value::text(&Value::Error("short element".into()), paths);
|
|
if depth > 32 {
|
|
return value::text(&dec.decode(dt, b, depth), paths);
|
|
}
|
|
match dt {
|
|
Datatype::String {
|
|
padding: StringPadding::NullPad,
|
|
size,
|
|
..
|
|
} => match b.get(..*size as usize) {
|
|
Some(s) => value::quote_bytes(s),
|
|
None => err(),
|
|
},
|
|
Datatype::Compound { members, .. } => format!(
|
|
"{{ {} }}",
|
|
members
|
|
.iter()
|
|
.map(
|
|
|m| match usize::try_from(m.byte_offset).ok().and_then(|o| b.get(o..)) {
|
|
Some(mb) => ddl_text(dec, &m.datatype, mb, paths, depth + 1),
|
|
None => err(),
|
|
}
|
|
)
|
|
.collect::<Vec<_>>()
|
|
.join(", ")
|
|
),
|
|
Datatype::Array {
|
|
base_type,
|
|
dimensions,
|
|
} => {
|
|
let bs = base_type.type_size() as usize;
|
|
let n = dimensions
|
|
.iter()
|
|
.try_fold(1usize, |a, &d| a.checked_mul(d as usize));
|
|
match n.filter(|n| n.checked_mul(bs).is_some_and(|t| t <= b.len())) {
|
|
Some(n) => format!(
|
|
"[ {} ]",
|
|
(0..n)
|
|
.map(|k| ddl_text(dec, base_type, &b[k * bs..], paths, depth + 1))
|
|
.collect::<Vec<_>>()
|
|
.join(", ")
|
|
),
|
|
None => value::text(&dec.decode(dt, b, depth), paths),
|
|
}
|
|
}
|
|
_ => value::text(&dec.decode(dt, b, depth), paths),
|
|
}
|
|
}
|
|
|
|
fn index_text(mut i: u64, dims: &[u64]) -> String {
|
|
let mut idx = vec![0u64; dims.len()];
|
|
for (k, &d) in dims.iter().enumerate().rev() {
|
|
let d = d.max(1);
|
|
idx[k] = i % d;
|
|
i /= d;
|
|
}
|
|
idx.iter()
|
|
.map(|x| x.to_string())
|
|
.collect::<Vec<_>>()
|
|
.join(",")
|
|
}
|
|
|
|
/// Turn a flat row-major list into nested lists of shape `dims`.
|
|
fn nest(flat: Vec<J>, dims: &[u64]) -> J {
|
|
if dims.len() <= 1 {
|
|
return J::Array(flat);
|
|
}
|
|
let inner: usize = dims[1..].iter().map(|&d| d as usize).product();
|
|
if inner == 0 {
|
|
return J::Array((0..dims[0]).map(|_| nest(Vec::new(), &dims[1..])).collect());
|
|
}
|
|
let mut it = flat.into_iter();
|
|
let mut outer = Vec::with_capacity(dims[0] as usize);
|
|
for _ in 0..dims[0] {
|
|
let part: Vec<J> = it.by_ref().take(inner).collect();
|
|
outer.push(nest(part, &dims[1..]));
|
|
}
|
|
J::Array(outer)
|
|
}
|
|
|
|
fn shape_json(ds: &Dataspace) -> J {
|
|
match ds.space_type {
|
|
DataspaceType::Null => json!({"class": "H5S_NULL"}),
|
|
DataspaceType::Scalar => json!({"class": "H5S_SCALAR"}),
|
|
DataspaceType::Simple => {
|
|
let mut m = Map::new();
|
|
m.insert("class".into(), json!("H5S_SIMPLE"));
|
|
m.insert("dims".into(), json!(ds.dimensions));
|
|
if let Some(max) = &ds.max_dimensions {
|
|
let mx: Vec<J> = max
|
|
.iter()
|
|
.map(|&d| {
|
|
if d == u64::MAX {
|
|
json!("H5S_UNLIMITED")
|
|
} else {
|
|
json!(d)
|
|
}
|
|
})
|
|
.collect();
|
|
m.insert("maxdims".into(), J::Array(mx));
|
|
}
|
|
J::Object(m)
|
|
}
|
|
}
|
|
}
|
|
|
|
fn filter_class(id: u16) -> &'static str {
|
|
match id {
|
|
1 => "H5Z_FILTER_DEFLATE",
|
|
2 => "H5Z_FILTER_SHUFFLE",
|
|
3 => "H5Z_FILTER_FLETCHER32",
|
|
4 => "H5Z_FILTER_SZIP",
|
|
5 => "H5Z_FILTER_NBIT",
|
|
6 => "H5Z_FILTER_SCALEOFFSET",
|
|
_ => "H5Z_FILTER_USER",
|
|
}
|
|
}
|
|
|
|
/// Deterministic object id: kind prefix + header address (hdf5-json uses
|
|
/// UUIDs; these are stable for a given file instead).
|
|
pub fn obj_id(kind: Kind, addr: u64) -> String {
|
|
let p = match kind {
|
|
Kind::Dataset => 'd',
|
|
Kind::Datatype => 't',
|
|
_ => 'g',
|
|
};
|
|
format!("{p}-{addr:016x}")
|
|
}
|