"""Values for in-place writes (clawhdf5.File(path, 'r+')), prepared the way h5py prepares them, so `ds[key] = value` stores what h5py would store. Loaded by the extension module (src/edit.rs); not a public API. h5py converts in two ways, and so does this module: - a value that is not a numpy array (a list, a Python or numpy scalar) is converted by numpy straight to the dataset's dtype (`numpy.asarray(value, dtype=ds.dtype)`), with numpy's rules and errors; - a numpy array is converted by libhdf5, whose numeric conversions clip to the target's range instead of wrapping: integers saturate, floats are truncated toward zero and clipped, a double too large for a float becomes infinity. That is what `_convert_array` reproduces. Where libhdf5 has no meaningful answer — NaN into an integer, for which it writes a different arbitrary value per type — this raises ValueError instead of guessing. """ import numpy as np def _no_path(src, dst): return TypeError(f"No conversion path for dtype: {src!r} -> {dst!r}") def _to_int(arr, dtype): """Integer target: libhdf5's saturating conversion.""" info = np.iinfo(dtype) kind = arr.dtype.kind if kind == "b": return arr.astype(dtype) if kind in "iu": src = np.iinfo(arr.dtype) lo = max(info.min, src.min) hi = min(info.max, src.max) clipped = np.clip(arr, np.array(lo, arr.dtype), np.array(hi, arr.dtype)) return clipped.astype(dtype) if kind == "f": if np.isnan(arr).any(): raise ValueError( "cannot write NaN to an integer dataset (libhdf5 would store an arbitrary value)" ) t = np.trunc(arr.astype(np.float64)) # info.max + 1 and info.min are powers of two: exact as floats. over = t >= float(info.max + 1) under = t < float(info.min) out = np.where(over | under, 0.0, t).astype(dtype) out[over] = info.max out[under] = info.min return out raise _no_path(arr.dtype, dtype) def _convert_array(arr, dtype, category): kind = arr.dtype.kind if category in ("int", "enum"): if arr.dtype == dtype and kind in "iu": return arr if category == "enum" and kind not in "iu": raise _no_path(arr.dtype, dtype) return _to_int(arr, dtype) if category == "bool": if kind == "b": return arr.astype(dtype) if kind in "iu": # h5py's bool is an enum over int8; libhdf5 converts integers # into it by value (saturating), not to FALSE/TRUE, so 3 is # stored as 3. Keep those bytes: a view, not a cast. return _to_int(arr, np.dtype("i1")).view(dtype) raise _no_path(arr.dtype, dtype) if category == "float": if kind not in "biuf": raise _no_path(arr.dtype, dtype) with np.errstate(over="ignore", invalid="ignore"): return arr.astype(dtype) if category == "complex": if kind != "c": raise _no_path(arr.dtype, dtype) with np.errstate(over="ignore", invalid="ignore"): return arr.astype(dtype) if category == "string": if kind != "S": raise _no_path(arr.dtype, dtype) return arr.astype(dtype) # "exact": compound and opaque types, written only from the same dtype. if arr.dtype == dtype: return arr raise _no_path(arr.dtype, dtype) def _convert_other(value, dtype, category): if category == "string": items = np.asarray(value, dtype=object) if any(isinstance(x, str) for x in items.flat): meta = dtype.metadata or {} if meta.get("h5py_encoding") == "utf-8": enc = [x.encode("utf-8") if isinstance(x, str) else x for x in items.flat] return np.array(enc, dtype=dtype).reshape(items.shape) return np.asarray(value, dtype=dtype) def _broadcast(arr, shape, fancy, chunk_elems): """h5py's broadcasting: numpy's rules against the selection's shape (extra leading length-1 axes allowed) for slices and integers. For an index list, the exact shape; a scalar only where h5py expands it to the whole selection (a chunked dataset whose chunk holds at least as many elements as the selection).""" if arr.shape == shape: return arr if fancy: size = int(np.prod(shape)) if arr.ndim == 0 and ((chunk_elems > 0 and size <= chunk_elems) or len(shape) == 1): return np.broadcast_to(arr, shape) raise TypeError("Broadcasting is not supported for complex selections") if arr.ndim == 0: return np.broadcast_to(arr, shape) err = TypeError(f"Can't broadcast {arr.shape} -> {shape}") src = arr.shape while len(src) > len(shape) and src[0] == 1: src = src[1:] if len(src) > len(shape): raise err try: return np.broadcast_to(arr.reshape(src), shape) except ValueError: raise err from None def dataset_values(value, dtype, category, shape, fancy, chunk_elems): """The bytes to write for `value` under a selection of `shape`, as a C-ordered array of the dataset's dtype.""" if isinstance(value, np.ndarray): arr = _convert_array(value, dtype, category) else: arr = _convert_other(value, dtype, category) arr = _broadcast(arr, tuple(shape), fancy, chunk_elems) return np.ascontiguousarray(arr, dtype=dtype).tobytes() def attr_value(value, dtype=None, shape=None): """(kind, dtype string, shape, bytes) for an attribute value, h5py's `attrs[name] = value` / `attrs.create(name, data, shape, dtype)`: - "str": `str` data (h5py would store a variable-length string; this stores a fixed-length UTF-8 string, which clawhdf5 can write); the dtype string is the byte length of the longest element; - "raw": a numeric, bool or bytes array, as numpy lays it out. """ if dtype is not None: arr = np.asarray(value, dtype=dtype, order="C") else: arr = np.asarray(value, order="C") if shape is not None: arr = arr.reshape(shape) kind = arr.dtype.kind if kind == "O": if arr.size and all(isinstance(x, str) for x in arr.flat): kind = "U" elif arr.size and all(isinstance(x, bytes) for x in arr.flat): arr = arr.astype(bytes) kind = "S" else: raise TypeError( f"clawhdf5 cannot write an attribute of Python objects ({value!r:.60})" ) if kind == "U": enc = [str(x).encode("utf-8") for x in arr.flat] size = max([len(b) for b in enc] + [1]) data = np.array(enc, dtype=f"S{size}").reshape(arr.shape) return ("str", str(size), arr.shape, data.tobytes()) if kind in "biufcS": return ("raw", arr.dtype.str, arr.shape, np.ascontiguousarray(arr).tobytes()) raise NotImplementedError( f"clawhdf5 cannot write an attribute of dtype {arr.dtype} in place" )