Add shutdown-prep button to dashboard-v2 NodeDetail
Build with clawstor cache / Cargo build (clawstor-cached) (pull_request) Failing after 3s
Build with clawstor cache / Cargo build (clawstor-cached) (pull_request) Failing after 3s
Wires safe-shutdown-prep.sh into the dashboard so an operator can prep a node for hardware maintenance from a browser instead of SSH. New RPC methods (0x20/0x21): - ShutdownPrepCheck runs `--dry-run` to completion and returns the full report. Never stops anything, safe to call repeatedly. - ShutdownPrepExecute starts the real run detached (`systemd-run --user --scope --collect`), placing it in a cgroup outside claw-store.service's own -- the script's own step 6 stops that service, i.e. the process that would otherwise be running it, so it has to survive its own parent dying. Returns immediately with a "started" message; full output lands in /var/lib/claw-store/shutdown-prep.log for whoever's at the machine once it's gone dark, since there's no way to stream a live result past the point the daemon stops itself. - Execute double-checks confirm_node_name against the peer's own configured name server-side, on top of the aggregator's own path match -- defense in depth for a highly consequential action. Aggregator endpoints (admin-token gated, AuthedCaller::require_admin): POST /api/v2/node/:name/shutdown-prep/check POST /api/v2/node/:name/shutdown-prep/execute Frontend: ShutdownPrepPanel on NodeDetail. Check button always enabled; the real "stop services" button only unlocks after a ready check, and additionally requires typing the exact node name to confirm before it's clickable. Also fixes a script bug found while testing this against the live daemon process (not caught in manual interactive-shell testing): the zpool-detection line parsed raw `mount` output positionally, which returned the wrong field under the daemon's process context for reasons that didn't reproduce interactively. Switched to `df --output=source`, which is stable across both. Verified end-to-end against tank, architect, and morpheus, including cross-node targeting (tank's dashboard successfully triggered a check on morpheus over the fleet RPC layer). Co-Authored-By: Claude Sonnet 5 <[email protected]>
This commit is contained in:
@@ -217,6 +217,22 @@ pub enum Method {
|
||||
/// `payload`: JSON [`crate::cluster::repo_ensure::RepoReleaseRequest`].
|
||||
/// Reply: JSON [`crate::cluster::repo_ensure::RepoReleaseReply`].
|
||||
RepoRelease = 0x1f,
|
||||
/// Runs `safe-shutdown-prep.sh --dry-run` to completion on this
|
||||
/// node and returns the full report. Never stops anything —
|
||||
/// dry-run only, safe to call repeatedly.
|
||||
///
|
||||
/// `payload`: JSON [`crate::cluster::shutdown_prep::ShutdownPrepCheckRequest`].
|
||||
/// Reply: JSON [`crate::cluster::shutdown_prep::ShutdownPrepCheckReply`].
|
||||
ShutdownPrepCheck = 0x20,
|
||||
/// Starts the real `safe-shutdown-prep.sh` run in a detached
|
||||
/// systemd scope and returns immediately — the script's own step
|
||||
/// stops `claw-store.service`, so this RPC connection cannot
|
||||
/// outlive full completion. See
|
||||
/// [`crate::cluster::shutdown_prep`] module docs.
|
||||
///
|
||||
/// `payload`: JSON [`crate::cluster::shutdown_prep::ShutdownPrepExecuteRequest`].
|
||||
/// Reply: JSON [`crate::cluster::shutdown_prep::ShutdownPrepExecuteReply`].
|
||||
ShutdownPrepExecute = 0x21,
|
||||
}
|
||||
|
||||
impl Method {
|
||||
@@ -255,6 +271,8 @@ impl Method {
|
||||
0x1d => Some(Method::DashboardStorage),
|
||||
0x1e => Some(Method::RepoEnsure),
|
||||
0x1f => Some(Method::RepoRelease),
|
||||
0x20 => Some(Method::ShutdownPrepCheck),
|
||||
0x21 => Some(Method::ShutdownPrepExecute),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
@@ -1176,6 +1194,26 @@ impl RpcRouter {
|
||||
.context("encoding RepoReleaseReply as JSON")?;
|
||||
Ok(HandlerOutcome::Reply(json))
|
||||
}
|
||||
Method::ShutdownPrepCheck => {
|
||||
let reply = crate::cluster::shutdown_prep::check().await?;
|
||||
let json = serde_json::to_vec(&reply)
|
||||
.context("encoding ShutdownPrepCheckReply as JSON")?;
|
||||
Ok(HandlerOutcome::Reply(json))
|
||||
}
|
||||
Method::ShutdownPrepExecute => {
|
||||
let req: crate::cluster::shutdown_prep::ShutdownPrepExecuteRequest =
|
||||
match serde_json::from_slice(payload) {
|
||||
Ok(r) => r,
|
||||
Err(_) => {
|
||||
return Ok(HandlerOutcome::Error(ErrorCode::InvalidRequest))
|
||||
}
|
||||
};
|
||||
let reply =
|
||||
crate::cluster::shutdown_prep::execute(&self.local_name, &req).await?;
|
||||
let json = serde_json::to_vec(&reply)
|
||||
.context("encoding ShutdownPrepExecuteReply as JSON")?;
|
||||
Ok(HandlerOutcome::Reply(json))
|
||||
}
|
||||
Method::BlobStat => {
|
||||
let store = match &self.blob_store {
|
||||
Some(s) => s,
|
||||
|
||||
@@ -229,6 +229,45 @@ pub async fn call_repo_release(
|
||||
serde_json::from_slice(&reply).context("decoding RepoReleaseReply JSON")
|
||||
}
|
||||
|
||||
/// Convenience wrapper for [`Method::ShutdownPrepCheck`]. Runs
|
||||
/// `safe-shutdown-prep.sh --dry-run` on the connected peer and waits
|
||||
/// for the full report. Never stops anything on the peer.
|
||||
pub async fn call_shutdown_prep_check(
|
||||
conn: &Connection,
|
||||
) -> Result<crate::cluster::shutdown_prep::ShutdownPrepCheckReply> {
|
||||
let req = crate::cluster::shutdown_prep::ShutdownPrepCheckRequest {};
|
||||
let payload = serde_json::to_vec(&req).context("encoding ShutdownPrepCheckRequest")?;
|
||||
let reply = rpc_call(conn, Method::ShutdownPrepCheck, &payload).await?;
|
||||
if reply.len() == 1 {
|
||||
if let Some(code) = decode_error(reply[0]) {
|
||||
bail!("peer replied with error: {}", code.describe());
|
||||
}
|
||||
}
|
||||
serde_json::from_slice(&reply).context("decoding ShutdownPrepCheckReply JSON")
|
||||
}
|
||||
|
||||
/// Convenience wrapper for [`Method::ShutdownPrepExecute`]. Starts the
|
||||
/// real shutdown-prep run on the connected peer (detached — this call
|
||||
/// returns as soon as the peer confirms it started, not when it
|
||||
/// finishes, since the peer's own daemon stops itself partway
|
||||
/// through).
|
||||
pub async fn call_shutdown_prep_execute(
|
||||
conn: &Connection,
|
||||
confirm_node_name: &str,
|
||||
) -> Result<crate::cluster::shutdown_prep::ShutdownPrepExecuteReply> {
|
||||
let req = crate::cluster::shutdown_prep::ShutdownPrepExecuteRequest {
|
||||
confirm_node_name: confirm_node_name.to_string(),
|
||||
};
|
||||
let payload = serde_json::to_vec(&req).context("encoding ShutdownPrepExecuteRequest")?;
|
||||
let reply = rpc_call(conn, Method::ShutdownPrepExecute, &payload).await?;
|
||||
if reply.len() == 1 {
|
||||
if let Some(code) = decode_error(reply[0]) {
|
||||
bail!("peer replied with error: {}", code.describe());
|
||||
}
|
||||
}
|
||||
serde_json::from_slice(&reply).context("decoding ShutdownPrepExecuteReply JSON")
|
||||
}
|
||||
|
||||
/// Recognise a single-byte reply as one of our error codes. Returns
|
||||
/// `None` for any other single-byte value (which is a valid reply,
|
||||
/// just an unusually short one).
|
||||
|
||||
@@ -0,0 +1,166 @@
|
||||
//! Peer-side wiring for `deploy/scripts/safe-shutdown-prep.sh`,
|
||||
//! surfaced through the RPC layer so the dashboard-v2 aggregator can
|
||||
//! offer a "prepare this node for shutdown" action.
|
||||
//!
|
||||
//! Split into two RPCs deliberately:
|
||||
//!
|
||||
//! - [`Method::ShutdownPrepCheck`] runs the script's `--dry-run` mode
|
||||
//! and waits for it to finish. Dry-run never stops this node's own
|
||||
//! daemon, so the RPC connection survives to deliver the full
|
||||
//! report — this is the part a browser can meaningfully show.
|
||||
//! - [`Method::ShutdownPrepExecute`] runs the real script, which (by
|
||||
//! design) stops `claw-store.service` — i.e. the very process
|
||||
//! handling this RPC. There is no way to stream a live result past
|
||||
//! that point, so this RPC detaches the script into its own
|
||||
//! transient systemd scope (outside this daemon's service cgroup,
|
||||
//! so `systemctl stop claw-store.service` doesn't take the script
|
||||
//! down with it) and returns immediately. The full report lands in
|
||||
//! `SHUTDOWN_PREP_LOG` for whoever is physically at the machine (or
|
||||
//! over SSH) to read once the node has gone dark.
|
||||
//!
|
||||
//! [`Method::ShutdownPrepCheck`]: crate::cluster::rpc::Method::ShutdownPrepCheck
|
||||
//! [`Method::ShutdownPrepExecute`]: crate::cluster::rpc::Method::ShutdownPrepExecute
|
||||
|
||||
use anyhow::{bail, Context, Result};
|
||||
use serde::{Deserialize, Serialize};
|
||||
use std::path::PathBuf;
|
||||
use std::time::Duration;
|
||||
use tokio::process::Command;
|
||||
use tokio::time::timeout;
|
||||
|
||||
/// Dry-run does a real snapshot + replicate, which can legitimately
|
||||
/// take a while on a large delta. Generous but bounded so a stuck
|
||||
/// peer connection doesn't hang the RPC forever.
|
||||
const CHECK_TIMEOUT: Duration = Duration::from_secs(300);
|
||||
|
||||
/// Where the real run's output lands once this node's daemon (and
|
||||
/// therefore this RPC connection) is gone.
|
||||
pub const SHUTDOWN_PREP_LOG: &str = "/var/lib/claw-store/shutdown-prep.log";
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
pub struct ShutdownPrepCheckRequest {}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
pub struct ShutdownPrepCheckReply {
|
||||
/// True iff the script exited 0 (every guard passed, "SAFE TO
|
||||
/// POWER OFF" printed for the checked steps).
|
||||
pub ready: bool,
|
||||
/// Full combined stdout+stderr from `--dry-run`.
|
||||
pub output: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
pub struct ShutdownPrepExecuteRequest {
|
||||
/// Defense in depth beyond RPC targeting: the caller must name
|
||||
/// the exact node it thinks it's shutting down. Checked against
|
||||
/// this node's own configured name before anything runs.
|
||||
pub confirm_node_name: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
pub struct ShutdownPrepExecuteReply {
|
||||
pub started: bool,
|
||||
pub message: String,
|
||||
pub log_path: String,
|
||||
}
|
||||
|
||||
fn script_path() -> PathBuf {
|
||||
if let Ok(p) = std::env::var("CLAWSTOR_SHUTDOWN_SCRIPT") {
|
||||
if !p.is_empty() {
|
||||
return PathBuf::from(p);
|
||||
}
|
||||
}
|
||||
if let Ok(home) = std::env::var("HOME") {
|
||||
if !home.is_empty() {
|
||||
return PathBuf::from(home)
|
||||
.join("clawstor-deploy/scripts/safe-shutdown-prep.sh");
|
||||
}
|
||||
}
|
||||
PathBuf::from("/usr/local/share/claw-store/safe-shutdown-prep.sh")
|
||||
}
|
||||
|
||||
/// Run `safe-shutdown-prep.sh --dry-run` to completion and report the
|
||||
/// full output. Never stops anything on this node — safe to call any
|
||||
/// time, including repeatedly.
|
||||
pub async fn check() -> Result<ShutdownPrepCheckReply> {
|
||||
let script = script_path();
|
||||
if !script.exists() {
|
||||
bail!("shutdown-prep script not found at {}", script.display());
|
||||
}
|
||||
let run = Command::new("bash")
|
||||
.arg(&script)
|
||||
.arg("--dry-run")
|
||||
.output();
|
||||
let output = timeout(CHECK_TIMEOUT, run)
|
||||
.await
|
||||
.context("shutdown-prep --dry-run timed out")?
|
||||
.context("spawning shutdown-prep --dry-run")?;
|
||||
let mut combined = String::from_utf8_lossy(&output.stdout).into_owned();
|
||||
combined.push_str(&String::from_utf8_lossy(&output.stderr));
|
||||
Ok(ShutdownPrepCheckReply {
|
||||
ready: output.status.success(),
|
||||
output: combined,
|
||||
})
|
||||
}
|
||||
|
||||
/// Kick off the real (non-dry-run) script in a transient systemd
|
||||
/// scope detached from this daemon's own service cgroup, then return
|
||||
/// immediately without waiting for it. The script's own step 6 stops
|
||||
/// `claw-store.service` — waiting for it to exit here would mean
|
||||
/// waiting for our own process to be killed.
|
||||
pub async fn execute(local_node_name: &str, req: &ShutdownPrepExecuteRequest) -> Result<ShutdownPrepExecuteReply> {
|
||||
if req.confirm_node_name != local_node_name {
|
||||
bail!(
|
||||
"confirm_node_name '{}' does not match this node ('{}') — refusing",
|
||||
req.confirm_node_name,
|
||||
local_node_name
|
||||
);
|
||||
}
|
||||
let script = script_path();
|
||||
if !script.exists() {
|
||||
bail!("shutdown-prep script not found at {}", script.display());
|
||||
}
|
||||
let unit = format!(
|
||||
"clawstor-shutdown-prep-{}",
|
||||
std::time::SystemTime::now()
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.map(|d| d.as_secs())
|
||||
.unwrap_or(0)
|
||||
);
|
||||
// `--user --scope` places this under the user session's cgroup
|
||||
// tree (/user.slice/...), a sibling of — not a descendant of —
|
||||
// /system.slice/claw-store.service. `systemctl stop
|
||||
// claw-store.service` only tears down its own cgroup, so this
|
||||
// keeps running (and completes step 6, which stops that very
|
||||
// service) unaffected.
|
||||
// Deliberately no --force: the real run re-checks active builds
|
||||
// and the sync queue itself, even though `check()` may have run
|
||||
// moments ago — state can change between the two clicks, and
|
||||
// re-validating is cheap.
|
||||
let cmd = format!(
|
||||
"{} >> {} 2>&1",
|
||||
script.display(),
|
||||
SHUTDOWN_PREP_LOG
|
||||
);
|
||||
let spawn = Command::new("systemd-run")
|
||||
.arg("--user")
|
||||
.arg("--scope")
|
||||
.arg("--collect")
|
||||
.arg(format!("--unit={unit}"))
|
||||
.arg("bash")
|
||||
.arg("-c")
|
||||
.arg(&cmd)
|
||||
.spawn();
|
||||
match spawn {
|
||||
Ok(_child) => Ok(ShutdownPrepExecuteReply {
|
||||
started: true,
|
||||
message: format!(
|
||||
"shutdown-prep started on {local_node_name} as transient unit {unit}. \
|
||||
This node's daemon (and dashboard) will go offline as part of the \
|
||||
process — that is expected. Full output: {SHUTDOWN_PREP_LOG}."
|
||||
),
|
||||
log_path: SHUTDOWN_PREP_LOG.to_string(),
|
||||
}),
|
||||
Err(e) => Err(e).context("spawning systemd-run for shutdown-prep"),
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user