Files
clawmates/crates/cm-db/src/repo/node_metrics.rs
T
Omar SobhandClaude Opus 4.8 3554a3aaf2
ci / gates (push) Successful in 5s
ci / frontend (push) Successful in 23s
ci / rust (push) Failing after 27s
ci / e2e (push) Has been skipped
CI: remove k8s stages, fix the Docker-level pipeline green
Survey + fixes so the pipeline passes at the Docker level (no k8s).

- Remove k8s: drop the `sandbox-k8s` job (kind/Calico/--features k8s-tests) and the
  "Helm chart lints" gate step. release.yml was already k8s-clean.
- Rust job:
  - `cargo fmt --all` — fix pre-existing formatting drift (fmt --check was failing).
  - clippy -D warnings: fix 3 lib warnings (cm-brain sort_by_key→Reverse, cm-api
    fleet.rs doc list indentation, node_rules map_or→is_none_or).
  - Regenerate the .sqlx offline cache (was missing the cm-runtime run_loop test
    query → offline compile failed). DB-backed tests use testcontainers at runtime.
  - Set SQLX_OFFLINE=true on the rust + e2e jobs so query! macros compile against
    the committed cache deterministically (no DB needed at compile time).
- Frontend job:
  - Fix the 1 ESLint error (useAgentTelemetry: no setState-synchronously-in-effect;
    tag the slice with agentId + derive null on mismatch).
  - Fix 2 stale panel-params tests (`terminal` is a valid app id now; assert the
    current APP_IDS + use a genuinely-unknown id for the reject case).

Verified locally: fmt clean, clippy --all-targets -D warnings clean (offline),
frontend lint 0 errors, tsc clean, 86/86 frontend tests pass, build OK.

Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
2026-06-26 18:15:31 -07:00

146 lines
5.1 KiB
Rust

//! Rich per-node metrics (from the workspace's Beszel hub). One latest snapshot
//! per node: scalar columns for the fleet cards + rules engine, plus a JSONB blob
//! for the per-node monitor page.
use cm_domain::{NodeId, WorkspaceId};
use serde_json::Value;
use sqlx::{PgPool, Row};
use uuid::Uuid;
use crate::DbError;
/// The latest metric snapshot for a node (nullable scalars + the full blob).
#[derive(Debug, Clone, Default)]
pub struct NodeMetrics {
pub cpu_pct: Option<f64>,
pub mem_pct: Option<f64>,
pub disk_pct: Option<f64>,
pub gpu_pct: Option<f64>,
pub temp_max: Option<f64>,
pub net_sent_ps: Option<i64>,
pub net_recv_ps: Option<i64>,
pub disk_read_ps: Option<i64>,
pub disk_write_ps: Option<i64>,
pub load1: Option<f64>,
pub container_count: Option<i32>,
pub data: Value,
}
/// Upsert a node's latest metrics snapshot.
pub async fn upsert(pool: &PgPool, node_id: NodeId, m: &NodeMetrics) -> Result<(), DbError> {
sqlx::query(
"INSERT INTO node_metrics
(node_id, updated_at, cpu_pct, mem_pct, disk_pct, gpu_pct, temp_max,
net_sent_ps, net_recv_ps, disk_read_ps, disk_write_ps, load1, container_count, data)
VALUES ($1, now(), $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13)
ON CONFLICT (node_id) DO UPDATE SET
updated_at = now(), cpu_pct = excluded.cpu_pct, mem_pct = excluded.mem_pct,
disk_pct = excluded.disk_pct, gpu_pct = excluded.gpu_pct, temp_max = excluded.temp_max,
net_sent_ps = excluded.net_sent_ps, net_recv_ps = excluded.net_recv_ps,
disk_read_ps = excluded.disk_read_ps, disk_write_ps = excluded.disk_write_ps,
load1 = excluded.load1, container_count = excluded.container_count, data = excluded.data",
)
.bind(node_id.as_uuid())
.bind(m.cpu_pct)
.bind(m.mem_pct)
.bind(m.disk_pct)
.bind(m.gpu_pct)
.bind(m.temp_max)
.bind(m.net_sent_ps)
.bind(m.net_recv_ps)
.bind(m.disk_read_ps)
.bind(m.disk_write_ps)
.bind(m.load1)
.bind(m.container_count)
.bind(&m.data)
.execute(pool)
.await?;
Ok(())
}
/// A node's current evaluatable scalars (Beszel-tapped, falling back to the
/// heartbeat health), for the rules engine + metrics-aware placement.
#[derive(Debug, Clone)]
pub struct EvalRow {
pub node_id: NodeId,
pub workspace_id: WorkspaceId,
pub status: String,
pub cpu_pct: Option<f64>,
pub mem_pct: Option<f64>,
pub disk_pct: Option<f64>,
pub gpu_pct: Option<f64>,
pub temp_max: Option<f64>,
pub load1: Option<f64>,
}
impl EvalRow {
/// Look up a metric by rule name.
pub fn metric(&self, name: &str) -> Option<f64> {
match name {
"cpu_pct" => self.cpu_pct,
"mem_pct" => self.mem_pct,
"disk_pct" => self.disk_pct,
"gpu_pct" => self.gpu_pct,
"temp_max" => self.temp_max,
"load1" => self.load1,
_ => None,
}
}
/// Free headroom heuristic (higher = more capacity) for placement ranking.
pub fn headroom(&self) -> f64 {
let used = self.cpu_pct.unwrap_or(0.0).max(self.mem_pct.unwrap_or(0.0));
100.0 - used
}
}
/// Every node's current metric scalars (merged Beszel + heartbeat health).
pub async fn eval_all(pool: &PgPool) -> Result<Vec<EvalRow>, DbError> {
let rows = sqlx::query(
"SELECT n.id, n.workspace_id, n.status,
COALESCE(m.cpu_pct, h.cpu_pct) AS cpu_pct,
COALESCE(m.mem_pct, CASE WHEN h.mem_total > 0 THEN h.mem_used::float8 / h.mem_total * 100 END) AS mem_pct,
COALESCE(m.disk_pct, CASE WHEN h.disk_total > 0 THEN (h.disk_total - h.disk_free)::float8 / h.disk_total * 100 END) AS disk_pct,
m.gpu_pct, m.temp_max,
COALESCE(m.load1, h.load1) AS load1
FROM nodes n
LEFT JOIN node_health h ON h.node_id = n.id
LEFT JOIN node_metrics m ON m.node_id = n.id",
)
.fetch_all(pool)
.await?;
Ok(rows
.into_iter()
.map(|r| EvalRow {
node_id: NodeId::from(r.get::<Uuid, _>("id")),
workspace_id: WorkspaceId::from(r.get::<Uuid, _>("workspace_id")),
status: r.get("status"),
cpu_pct: r.get("cpu_pct"),
mem_pct: r.get("mem_pct"),
disk_pct: r.get("disk_pct"),
gpu_pct: r.get("gpu_pct"),
temp_max: r.get("temp_max"),
load1: r.get("load1"),
})
.collect())
}
/// The latest metrics blob for a node (the full snapshot for the monitor page).
pub async fn latest(pool: &PgPool, node_id: NodeId) -> Result<Option<Value>, DbError> {
let row = sqlx::query(
"SELECT data, extract(epoch from updated_at)::bigint AS updated FROM node_metrics WHERE node_id = $1",
)
.bind(node_id.as_uuid())
.fetch_optional(pool)
.await?;
Ok(row.map(|r| {
let mut data: Value = r.get("data");
if let Some(obj) = data.as_object_mut() {
obj.insert(
"updatedAt".into(),
serde_json::json!(r.get::<i64, _>("updated")),
);
}
data
}))
}