Compare commits
6
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2f7eabf034 | ||
|
|
7902c2e395 | ||
|
|
3d504ee6b3 | ||
|
|
8407c99ba4 | ||
|
|
088b88b225 | ||
|
|
c3f6b500fc |
@@ -32,7 +32,7 @@ install-systemd:
|
||||
install-dashboard:
|
||||
cd dashboard && npm ci && npm run build
|
||||
install -dm755 $(INSTALL_STATIC)
|
||||
cp -r dashboard/dist/. $(INSTALL_STATIC)/
|
||||
cp -r claw-store/static/. $(INSTALL_STATIC)/
|
||||
@echo "Dashboard installed to $(INSTALL_STATIC)"
|
||||
|
||||
## Install node-specific config (NODE=architect|tank)
|
||||
|
||||
@@ -21,8 +21,14 @@ pub fn write_cargo_config(warm_path: &Path, hot_target_path: &Path) -> Result<()
|
||||
String::new()
|
||||
};
|
||||
|
||||
// Drop any leading copies of our own marker comment before stripping the
|
||||
// [build] section — it sits above the [build] header, outside the range
|
||||
// strip_section tracks, so without this it would survive every
|
||||
// regenerate cycle and duplicate one more time.
|
||||
let existing = strip_leading_marker(&existing);
|
||||
|
||||
// Remove old [build] block (from its header to the next section or EOF).
|
||||
let stripped = strip_section(&existing, "build");
|
||||
let stripped = strip_section(existing, "build");
|
||||
|
||||
let new_content = format!(
|
||||
"# claw-store managed — do not edit manually\n\
|
||||
@@ -37,6 +43,21 @@ pub fn write_cargo_config(warm_path: &Path, hot_target_path: &Path) -> Result<()
|
||||
.with_context(|| format!("writing {}", config_path.display()))
|
||||
}
|
||||
|
||||
const MANAGED_MARKER: &str = "# claw-store managed — do not edit manually";
|
||||
|
||||
/// Strip leading copies of `MANAGED_MARKER`, one per line, from the start of `src`.
|
||||
fn strip_leading_marker(src: &str) -> &str {
|
||||
let mut rest = src;
|
||||
while let Some(line_end) = rest.find('\n') {
|
||||
if rest[..line_end].trim() == MANAGED_MARKER {
|
||||
rest = &rest[line_end + 1..];
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
rest
|
||||
}
|
||||
|
||||
/// Remove a TOML section `[name]` and all its key=value lines from `src`,
|
||||
/// stopping at the next `[section]` header or EOF.
|
||||
fn strip_section(src: &str, name: &str) -> String {
|
||||
@@ -102,6 +123,46 @@ mod tests {
|
||||
assert!(verify_cargo_config(&warm, &hot).unwrap());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_write_cargo_config_idempotent_no_duplicate_marker() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let warm = dir.path().join("proj");
|
||||
std::fs::create_dir_all(&warm).unwrap();
|
||||
let hot: std::path::PathBuf = "/hot/targets/proj".into();
|
||||
|
||||
write_cargo_config(&warm, &hot).unwrap();
|
||||
write_cargo_config(&warm, &hot).unwrap();
|
||||
write_cargo_config(&warm, &hot).unwrap();
|
||||
|
||||
let content = std::fs::read_to_string(warm.join(".cargo/config.toml")).unwrap();
|
||||
assert_eq!(content.matches("claw-store managed").count(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_write_cargo_config_heals_existing_duplicate_marker() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let warm = dir.path().join("proj");
|
||||
let cargo_dir = warm.join(".cargo");
|
||||
std::fs::create_dir_all(&cargo_dir).unwrap();
|
||||
std::fs::write(
|
||||
cargo_dir.join("config.toml"),
|
||||
"# claw-store managed — do not edit manually\n\
|
||||
[build]\n\
|
||||
target-dir = \"/hot/targets/proj\"\n\
|
||||
# claw-store managed — do not edit manually\n\
|
||||
[env]\n\
|
||||
FOO = \"bar\"\n",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let hot: std::path::PathBuf = "/hot/targets/proj".into();
|
||||
write_cargo_config(&warm, &hot).unwrap();
|
||||
|
||||
let content = std::fs::read_to_string(cargo_dir.join("config.toml")).unwrap();
|
||||
assert_eq!(content.matches("claw-store managed").count(), 1);
|
||||
assert!(content.contains("FOO = \"bar\""));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_verify_cargo_config_detects_missing() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
|
||||
+17
-17
@@ -7,7 +7,7 @@ use crate::manifest::Manifest;
|
||||
use crate::snapshot;
|
||||
use crate::sync::{SyncQueue, drain_sync_queue};
|
||||
use crate::zfs::SystemZfs;
|
||||
use anyhow::{Context, Result};
|
||||
use anyhow::Result;
|
||||
use chrono::Utc;
|
||||
use sysinfo::{ProcessRefreshKind, RefreshKind, System};
|
||||
use tokio::time::{interval, Duration};
|
||||
@@ -36,14 +36,7 @@ pub async fn run(cfg: Config, mut manifest: Manifest) -> Result<()> {
|
||||
let hot_dir = cfg.hot.path.clone();
|
||||
let hot_max_bytes = cfg.hot.max_gb.saturating_mul(1024 * 1024 * 1024);
|
||||
let blob_root = cluster_cfg.blob_store_root.clone();
|
||||
// Fail fast rather than degrade silently: a bind failure here is
|
||||
// almost always a boot-time race against DHCP/network-online
|
||||
// (the bind address isn't assigned to the interface yet). The
|
||||
// systemd unit has `Restart=on-failure`; exiting lets it retry
|
||||
// a few seconds later once the network is actually up, instead
|
||||
// of leaving the daemon running indefinitely with no gossip,
|
||||
// RPC, or Prometheus endpoint and no visible failure state.
|
||||
let svc = ClusterServices::start(
|
||||
match ClusterServices::start(
|
||||
cluster_cfg,
|
||||
cfg.node.name.clone(),
|
||||
hot_dir,
|
||||
@@ -51,14 +44,21 @@ pub async fn run(cfg: Config, mut manifest: Manifest) -> Result<()> {
|
||||
blob_root,
|
||||
)
|
||||
.await
|
||||
.context("starting cluster services")?;
|
||||
tracing::info!(
|
||||
rpc_enabled = svc.rpc_enabled(),
|
||||
blob_store_enabled = svc.blob_store_enabled(),
|
||||
zone = %cluster_cfg.zone,
|
||||
"cluster services online"
|
||||
);
|
||||
Some(svc)
|
||||
{
|
||||
Ok(svc) => {
|
||||
tracing::info!(
|
||||
rpc_enabled = svc.rpc_enabled(),
|
||||
blob_store_enabled = svc.blob_store_enabled(),
|
||||
zone = %cluster_cfg.zone,
|
||||
"cluster services online"
|
||||
);
|
||||
Some(svc)
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::error!(error = %e, "cluster services failed to start; continuing without cluster");
|
||||
None
|
||||
}
|
||||
}
|
||||
}
|
||||
None => {
|
||||
tracing::info!("no [cluster] section in config; running standalone");
|
||||
|
||||
@@ -91,30 +91,9 @@ impl V2State {
|
||||
.join("aggregator-sessions.json");
|
||||
let sessions = SessionStore::load(sessions_path)
|
||||
.map_err(|e| anyhow::anyhow!("loading session store: {e}"))?;
|
||||
// Bug fix 2026-07-31: the fleet view previously never included
|
||||
// the node actually serving the dashboard — `cluster.peers` is
|
||||
// by definition every *other* node, so hitting a given node's
|
||||
// `/api/v2/fleet` directly silently dropped that node from its
|
||||
// own view (looked like "node X is missing" from the UI, even
|
||||
// though X was perfectly healthy — it just never queried
|
||||
// itself). Fix: synthesize a self `PeerEntry` from our own
|
||||
// gossip bind address and include it in the fan-out list, same
|
||||
// as any other peer. `peer_rpc_addr` derives the RPC port from
|
||||
// `lan_addr`/`tailscale_addr` via the fleet's +1 convention, so
|
||||
// this resolves to the same `bind_rpc_lan`/`bind_rpc_tailscale`
|
||||
// the daemon actually listens on.
|
||||
let self_peer = PeerEntry {
|
||||
name: cfg.node.name.clone(),
|
||||
zone: cluster.zone.clone(),
|
||||
lan_addr: cluster.bind_lan,
|
||||
tailscale_addr: cluster.bind_tailscale,
|
||||
};
|
||||
let mut peers = cluster.peers.clone();
|
||||
peers.push(self_peer);
|
||||
|
||||
Ok(Self {
|
||||
aggregator_name: cfg.node.name.clone(),
|
||||
peers,
|
||||
peers: cluster.peers.clone(),
|
||||
client: std::sync::Arc::new(client),
|
||||
default_rpc_port_offset: 1,
|
||||
api_token: cfg.api_token.clone(),
|
||||
|
||||
@@ -19,6 +19,33 @@ archive_path = "/data/archive"
|
||||
zfs_dataset = "data/archive"
|
||||
retain_weeks = 12
|
||||
|
||||
[cluster]
|
||||
zone = "fabric-10g"
|
||||
bind_lan = "10.0.0.13:7701"
|
||||
prom_bind = "0.0.0.0:7703"
|
||||
bind_rpc_lan = "10.0.0.13:7702"
|
||||
bind_rpc_tailscale = "100.104.171.32:7702"
|
||||
blob_store_root = "/home/osobh/clawstor-deploy/data"
|
||||
|
||||
[[cluster.peers]]
|
||||
name = "tank"
|
||||
zone = "fabric-10g"
|
||||
lan_addr = "10.0.0.14:7701"
|
||||
rpc_lan_addr = "10.10.0.10:7702"
|
||||
tailscale_addr = "100.108.129.81:7702"
|
||||
|
||||
[[cluster.peers]]
|
||||
name = "morpheus"
|
||||
zone = "lan-1g"
|
||||
lan_addr = "10.0.0.5:7701"
|
||||
rpc_lan_addr = "10.0.0.5:7702"
|
||||
tailscale_addr = "100.123.224.84:7702"
|
||||
|
||||
[cluster.tls]
|
||||
ca_cert = "/home/osobh/clawstor-deploy/tls/ca.crt"
|
||||
node_cert = "/home/osobh/clawstor-deploy/tls/node.crt"
|
||||
node_key = "/home/osobh/clawstor-deploy/tls/node.key"
|
||||
|
||||
[replication]
|
||||
receive_from_peer = true
|
||||
peer_user = "osobh"
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
[node]
|
||||
name = "morpheus"
|
||||
role = "secondary"
|
||||
|
||||
[hot]
|
||||
path = "/hot/targets"
|
||||
max_gb = 80
|
||||
stale_hours = 48
|
||||
|
||||
[warm]
|
||||
projects_path = "/slab/projects"
|
||||
zfs_dataset = "none"
|
||||
snapshot_retain_hours = 24
|
||||
snapshot_retain_days = 7
|
||||
snapshot_retain_weeks = 4
|
||||
|
||||
[cluster]
|
||||
zone = "lan-1g"
|
||||
# Morpheus has no direct 10G to Architect/Tank — use main LAN for all traffic
|
||||
bind_lan = "10.0.0.5:7701"
|
||||
prom_bind = "0.0.0.0:7703"
|
||||
bind_rpc_lan = "10.0.0.5:7702"
|
||||
bind_rpc_tailscale = "100.123.224.84:7702"
|
||||
blob_store_root = "/home/osobh/clawstor-deploy/data"
|
||||
|
||||
[[cluster.peers]]
|
||||
name = "architect"
|
||||
zone = "fabric-10g"
|
||||
lan_addr = "10.0.0.13:7701"
|
||||
rpc_lan_addr = "10.0.0.13:7702"
|
||||
tailscale_addr = "100.104.171.32:7702"
|
||||
|
||||
[[cluster.peers]]
|
||||
name = "tank"
|
||||
zone = "fabric-10g"
|
||||
lan_addr = "10.0.0.14:7701"
|
||||
rpc_lan_addr = "10.0.0.14:7702"
|
||||
tailscale_addr = "100.108.129.81:7702"
|
||||
|
||||
[cluster.tls]
|
||||
ca_cert = "/home/osobh/clawstor-deploy/tls/ca.crt"
|
||||
node_cert = "/home/osobh/clawstor-deploy/tls/node.crt"
|
||||
node_key = "/home/osobh/clawstor-deploy/tls/node.key"
|
||||
+27
-5
@@ -14,13 +14,35 @@ snapshot_retain_hours = 24
|
||||
snapshot_retain_days = 7
|
||||
snapshot_retain_weeks = 4
|
||||
|
||||
[cluster]
|
||||
zone = "fabric-10g"
|
||||
bind_lan = "10.0.0.14:7701"
|
||||
prom_bind = "0.0.0.0:7703"
|
||||
bind_rpc_lan = "10.0.0.14:7702"
|
||||
bind_rpc_tailscale = "100.108.129.81:7702"
|
||||
blob_store_root = "/home/osobh/clawstor-deploy/data"
|
||||
|
||||
[[cluster.peers]]
|
||||
name = "architect"
|
||||
zone = "fabric-10g"
|
||||
lan_addr = "10.0.0.13:7701"
|
||||
rpc_lan_addr = "10.10.0.9:7702"
|
||||
tailscale_addr = "100.104.171.32:7702"
|
||||
|
||||
[[cluster.peers]]
|
||||
name = "morpheus"
|
||||
zone = "lan-1g"
|
||||
lan_addr = "10.0.0.5:7701"
|
||||
rpc_lan_addr = "10.0.0.5:7702"
|
||||
tailscale_addr = "100.123.224.84:7702"
|
||||
|
||||
[cluster.tls]
|
||||
ca_cert = "/home/osobh/clawstor-deploy/tls/ca.crt"
|
||||
node_cert = "/home/osobh/clawstor-deploy/tls/node.crt"
|
||||
node_key = "/home/osobh/clawstor-deploy/tls/node.key"
|
||||
|
||||
[replication]
|
||||
# 10.10.0.9 is architect-fab-tank — the dedicated 10G fabric link between
|
||||
# the two nodes. Intentionally used for replication to maximise bandwidth;
|
||||
# architect's primary LAN address is 10.0.0.13.
|
||||
send_to_host = "10.10.0.9"
|
||||
send_to_user = "osobh"
|
||||
cold_dataset_on_peer = "data/archive/tank-projects"
|
||||
# nightly_at is reserved for future use; replication schedule is currently
|
||||
# controlled by the claw-store-replicate.timer systemd unit.
|
||||
nightly_at = "03:30"
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
[Unit]
|
||||
Description=clawstor FUSE mount
|
||||
After=claw-store.service
|
||||
Requires=claw-store.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=osobh
|
||||
Environment=PATH=/home/osobh/.cargo/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
|
||||
ExecStartPre=/bin/mkdir -p /home/osobh/clawstor-mount
|
||||
ExecStart=/usr/local/bin/claw-fuse --data-dir /home/osobh/clawstor-deploy/data --mount /home/osobh/clawstor-mount
|
||||
ExecStop=/bin/fusermount -u /home/osobh/clawstor-mount
|
||||
Restart=on-failure
|
||||
RestartSec=30
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
@@ -11,7 +11,7 @@ Wants=network-online.target
|
||||
[Service]
|
||||
Type=simple
|
||||
User=osobh
|
||||
ExecStart=/usr/local/bin/claw-store serve --port 7700 --static-dir /usr/share/claw-store/static
|
||||
ExecStart=/usr/local/bin/claw-store serve --port 7700 --static-dir /usr/share/claw-store/static --v2-static-dir /usr/share/claw-store/v2
|
||||
Restart=on-failure
|
||||
RestartSec=15
|
||||
Environment=RUST_LOG=info
|
||||
|
||||
@@ -1,15 +1,19 @@
|
||||
[Unit]
|
||||
Description=claw-store fleet storage daemon
|
||||
After=zfs-mount.service network.target
|
||||
Wants=zfs-mount.service
|
||||
Description=clawstor cluster daemon (gossip + QUIC blob store + ZFS snapshots)
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=osobh
|
||||
Environment=PATH=/home/osobh/.cargo/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin
|
||||
Environment=RUST_LOG=info
|
||||
ExecStart=/usr/local/bin/claw-store daemon
|
||||
Restart=on-failure
|
||||
RestartSec=30
|
||||
Environment=RUST_LOG=info
|
||||
RestartSec=15
|
||||
TimeoutStopSec=60
|
||||
ProtectSystem=strict
|
||||
ReadWritePaths=/var/lib/claw-store /hot/targets /slab/projects /home/osobh/clawstor-deploy
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
|
||||
Reference in New Issue
Block a user