py: remote files (clawhdf5.File(url), File.open_url) through File::storage()

The Python bindings could not open a remote file: they parsed through
File::as_bytes() in eight places (path lookups, object headers,
dataspaces, attributes, group listings, the global heap of
variable-length data), which a storage-backed file does not have.

- Every object of a File now shares one handle (src/handle.rs) that
  runs all file access, metadata included, with the GIL released and
  parses through File::storage() and the clawhdf5_format *_in functions.
  Local files take the same path (their storage is the mmap).
- clawhdf5.File(url) opens any scheme://... through
  clawhdf5_remote::storage_for_url (read-only; another mode is a
  ValueError). File.open_url(url, **options) takes the cache and HTTP
  options (block_size, cache_size, headers, retries, timeout,
  allow_full_download, max_full_download, require_validator,
  max_redirects, max_parallel); File.remote_stats gives the block
  cache's counters.
- Default build: plain HTTP only, no C. https (rustls/ring) and
  s3/gcs/azure (aws-lc-rs) are opt-in features of clawhdf5-py, and
  ci-test.sh's no-C check now covers the crate.
- A failed storage read (network error, file changed on the server) is an
  OSError, never KeyError/ValueError and never data; `key in group`
  raises it instead of answering False.

Tests: the read-vs-h5py suite runs locally and over HTTP (1 MiB and
1 KiB blocks) against a range-capable http.server in the test process
(conftest.RangeServer); test_remote.py covers request counts, cache
hits, a server without Range support, a changed file, a server that
hangs up, 16 threads, and a spinning thread that keeps running while a
read waits on 0.2 s requests.

Co-Authored-By: Claude Opus 5.5 (1M context) <[email protected]>
This commit is contained in:
osobh
2026-09-27 06:40:44 -05:00
co-authored by Claude Opus 5.5
parent a4c2aced55
commit 910d81904c
18 changed files with 1146 additions and 299 deletions
+177 -29
View File
@@ -1,13 +1,17 @@
//! PyFile — the main entry point for opening and creating HDF5 files.
use std::collections::HashMap;
use std::path::PathBuf;
use std::sync::{Arc, Mutex};
use std::time::Duration;
use pyo3::exceptions::PyValueError;
use pyo3::prelude::*;
use pyo3::types::PyList;
use pyo3::types::{PyDict, PyList};
use crate::attrs::PyAttrs;
use crate::group::{PyGroup, ReadGroup, WriteGroupState, finalize_write_group};
use crate::handle::Handle;
use crate::{DatasetSpec, OwnedAttrValue, apply_dataset_spec, extract_numpy_data, to_py_err};
/// Internal state for write mode.
@@ -23,8 +27,10 @@ struct WriteState {
/// Mirrors the h5py.File interface:
///
/// ```python
/// # Reading
/// # Reading, a local file or a URL (range requests, nothing downloaded
/// # up front)
/// f = clawhdf5.File('data.h5', 'r')
/// f = clawhdf5.File('https://example.org/data.h5')
/// ds = f['dataset']
/// f.close()
///
@@ -44,27 +50,51 @@ enum FileInner {
Write(WriteState),
}
/// Whether `s` is a URL (`scheme://…`) rather than a path: the scheme is a
/// letter followed by letters, digits, `+`, `-` or `.` (RFC 3986).
fn is_url(s: &str) -> bool {
let Some((scheme, _)) = s.split_once("://") else {
return false;
};
let mut chars = scheme.chars();
chars.next().is_some_and(|c| c.is_ascii_alphabetic())
&& chars.all(|c| c.is_ascii_alphanumeric() || matches!(c, '+' | '-' | '.'))
}
impl PyFile {
fn from_handle(handle: Arc<Handle>, filename: String) -> Self {
let root = handle.root;
Self {
inner: Some(FileInner::Read(ReadGroup::new(handle, String::new(), root))),
filename,
}
}
}
#[pymethods]
impl PyFile {
/// Open or create an HDF5 file.
///
/// Parameters:
/// path: file path
/// path: file path, or a URL (`http://`, `https://`, `s3://`, `gs://`,
/// `az://`; which schemes work depends on how the wheel was built)
/// to read the file remotely with default options (see `open_url`)
/// mode: 'r' for read (default), 'w' for write
#[new]
#[pyo3(signature = (path, mode="r"))]
fn new(py: Python<'_>, path: &str, mode: &str) -> PyResult<Self> {
let filename = path.to_string();
match mode {
"r" => {
let file = py.detach(|| {
crate::no_panic(|| clawhdf5_rs::File::open(path).map_err(to_py_err))
})?;
Ok(Self {
inner: Some(FileInner::Read(root_group(Arc::new(file)))),
filename,
})
if is_url(path) {
if mode != "r" {
return Err(PyValueError::new_err(format!(
"remote files are read-only: mode '{mode}' is not supported for a URL"
)));
}
let handle = Handle::open_url(py, path, &clawhdf5_remote::Options::default())?;
return Ok(Self::from_handle(handle, filename));
}
match mode {
"r" => Ok(Self::from_handle(Handle::open_local(py, path)?, filename)),
"w" => Ok(Self {
filename,
inner: Some(FileInner::Write(WriteState {
@@ -74,12 +104,122 @@ impl PyFile {
groups: Vec::new(),
})),
}),
other => Err(PyErr::new::<pyo3::exceptions::PyValueError, _>(format!(
other => Err(PyValueError::new_err(format!(
"unsupported mode '{other}'; expected 'r' or 'w'"
))),
}
}
/// Open a remote file for reading, with options.
///
/// The file is read through a block cache with range requests: opening
/// costs one request (it also fetches the first block), and a read
/// fetches only the blocks it needs. The GIL is released while waiting
/// on the network.
///
/// Parameters (all optional):
/// block_size: bytes per cached block (default 1 MiB)
/// cache_size: byte budget of the block cache (default 64 MiB)
/// headers: dict of extra HTTP headers (e.g. Authorization), sent only
/// to the URL's own origin
/// retries: retries of a request that failed transiently (default 3)
/// timeout: seconds to connect and receive response headers (default 30)
/// allow_full_download: when the server ignores Range requests,
/// download the whole file once instead of failing (default False)
/// max_full_download: largest file such a download may fetch
/// (default 1 GiB)
/// require_validator: refuse a server that sends neither ETag nor
/// Last-Modified (default False)
/// max_redirects: redirects followed per request (default 5)
/// max_parallel: requests of one read in flight at once (default 8)
#[staticmethod]
#[allow(clippy::too_many_arguments)]
#[pyo3(signature = (url, *, block_size=None, cache_size=None, headers=None, retries=None,
timeout=None, allow_full_download=None, max_full_download=None,
require_validator=None, max_redirects=None, max_parallel=None))]
fn open_url(
py: Python<'_>,
url: &str,
block_size: Option<u64>,
cache_size: Option<u64>,
headers: Option<HashMap<String, String>>,
retries: Option<u32>,
timeout: Option<f64>,
allow_full_download: Option<bool>,
max_full_download: Option<u64>,
require_validator: Option<bool>,
max_redirects: Option<u32>,
max_parallel: Option<usize>,
) -> PyResult<Self> {
let mut options = clawhdf5_remote::Options::default();
if let Some(b) = block_size {
if b == 0 {
return Err(PyValueError::new_err("block_size must be positive"));
}
options.cache.block_size = b;
options.cache.coalesce_gap = b;
// The opening request fetches the first block, not 1 MiB.
options.http.first_request = b;
}
if let Some(c) = cache_size {
options.cache.capacity = c;
}
let http = &mut options.http;
if let Some(h) = headers {
http.headers = h.into_iter().collect();
}
if let Some(r) = retries {
http.retries = r;
}
if let Some(t) = timeout {
if !(t.is_finite() && t > 0.0) {
return Err(PyValueError::new_err("timeout must be a positive number"));
}
http.timeout = Duration::from_secs_f64(t);
}
if let Some(a) = allow_full_download {
http.allow_full_download = a;
}
if let Some(m) = max_full_download {
http.max_full_download = m;
}
if let Some(v) = require_validator {
http.require_validator = v;
}
if let Some(r) = max_redirects {
http.max_redirects = r;
}
if let Some(p) = max_parallel {
if p == 0 {
return Err(PyValueError::new_err("max_parallel must be positive"));
}
http.max_parallel = p;
}
let handle = Handle::open_url(py, url, &options)?;
Ok(Self::from_handle(handle, url.to_string()))
}
/// For a remote file, what its block cache has done so far (reads,
/// hits, misses, requests, bytes fetched, ...); `None` for a local file.
#[getter]
fn remote_stats<'py>(&self, py: Python<'py>) -> PyResult<Option<Bound<'py, PyDict>>> {
let Some(storage) = self.read_file()?.handle.remote_storage() else {
return Ok(None);
};
let s = storage.stats();
let d = PyDict::new(py);
d.set_item("reads", s.reads)?;
d.set_item("hits", s.hits)?;
d.set_item("misses", s.misses)?;
d.set_item("waits", s.waits)?;
d.set_item("requests", s.requests)?;
d.set_item("fetch_calls", s.fetch_calls)?;
d.set_item("bytes_fetched", s.bytes_fetched)?;
d.set_item("evictions", s.evictions)?;
d.set_item("cached_bytes", s.cached_bytes)?;
Ok(Some(d))
}
/// Close the file. In write mode, this finalizes and writes the file.
fn close(&mut self) -> PyResult<()> {
let inner = self.inner.take().ok_or_else(|| {
@@ -121,7 +261,7 @@ impl PyFile {
/// List the names of all children in the root group.
fn keys(&self, py: Python<'_>) -> PyResult<Py<PyAny>> {
let names = self.read_file()?.member_names()?;
let names = self.read_file()?.member_names(py)?;
Ok(PyList::new(py, names)?.into_any().unbind())
}
@@ -139,8 +279,8 @@ impl PyFile {
self.keys(py)?.call_method0(py, "__iter__")
}
fn __len__(&self) -> PyResult<usize> {
Ok(self.read_file()?.member_names()?.len())
fn __len__(&self, py: Python<'_>) -> PyResult<usize> {
Ok(self.read_file()?.member_names(py)?.len())
}
/// The root group's name, `/`.
@@ -149,7 +289,7 @@ impl PyFile {
"/"
}
/// The path the file was opened with.
/// The path (or URL) the file was opened with.
#[getter]
fn filename(&self) -> &str {
&self.filename
@@ -204,9 +344,9 @@ impl PyFile {
/// Attribute access. In read mode, returns attributes of the root group.
/// In write mode, returns a writable attrs handle.
#[getter]
fn attrs(&self) -> PyResult<PyAttrs> {
fn attrs(&self, py: Python<'_>) -> PyResult<PyAttrs> {
match self.inner.as_ref() {
Some(FileInner::Read(root)) => root.attrs(),
Some(FileInner::Read(root)) => root.attrs(py),
Some(FileInner::Write(state)) => Ok(PyAttrs::from_write(Arc::clone(&state.root_attrs))),
None => Err(PyErr::new::<pyo3::exceptions::PyIOError, _>(
"file is closed",
@@ -216,9 +356,10 @@ impl PyFile {
fn __repr__(&self) -> String {
match &self.inner {
Some(FileInner::Read(root)) => {
format!("<HDF5 File (read, {} bytes)>", root.file.as_bytes().len())
}
Some(FileInner::Read(root)) => match root.handle.redacted_url() {
Some(url) => format!("<HDF5 File (read, \"{url}\")>"),
None => format!("<HDF5 File (read, \"{}\")>", self.filename),
},
Some(FileInner::Write(s)) => {
format!("<HDF5 File (write, \"{}\")>", s.path.display())
}
@@ -226,8 +367,8 @@ impl PyFile {
}
}
fn __contains__(&self, key: &str) -> PyResult<bool> {
Ok(self.read_file()?.contains(key))
fn __contains__(&self, py: Python<'_>, key: &str) -> PyResult<bool> {
self.read_file()?.contains(py, key)
}
}
@@ -271,11 +412,6 @@ fn parse_compression(
}
}
fn root_group(file: Arc<clawhdf5_rs::File>) -> ReadGroup {
let root = file.superblock().root_group_address;
ReadGroup::new(file, String::new(), root)
}
/// Build and write the HDF5 file from accumulated write state.
fn finalize_write(state: WriteState) -> PyResult<()> {
crate::no_panic(|| {
@@ -309,6 +445,18 @@ fn finalize_write(state: WriteState) -> PyResult<()> {
mod tests {
use super::*;
#[test]
fn urls_and_paths() {
assert!(is_url("http://h/f.h5"));
assert!(is_url("s3://bucket/key.h5"));
assert!(is_url("git+https://x"));
assert!(!is_url("data.h5"));
assert!(!is_url("/tmp/a://b.h5"));
assert!(!is_url("dir/x://y"));
assert!(!is_url("1http://x"));
assert!(!is_url("://x"));
}
#[test]
fn parse_gzip_compression() {
assert_eq!(parse_compression(Some("gzip"), Some(6)).unwrap(), Some(6));