#!/usr/bin/env bash # Delete every mission and prove the disk came back with them. # # For getting to a clean slate before a UI session, and for the thing that keeps # being true here: deleting a row has never deleted a directory. The harness # leaves ~35 missions per full run and each carries a repo checkout plus a # runtime-data tree, on the gateway — the smallest disk in the fleet. # # Deletes through the API, never with SQL. `missions::delete` tears down the # per-mission runtime container, hard-purges the FK graph in order, and removes # the workspace directory (falling back to a root purge for the files the # per-mission daemon leaves as root). A `DELETE FROM missions` would skip all # three and orphan every one of those. # # Then it CHECKS, because the whole point is that the cleanup path has been # quietly incomplete before: rows gone is not the same as bytes back. # # Usage: # scripts/fleet-reset.sh # dry run — says what it would delete # scripts/fleet-reset.sh --yes # do it # KEEP='verify: a model sizes' scripts/fleet-reset.sh --yes # keep matches # KEEP_AGENTS=1 scripts/fleet-reset.sh --yes # missions only set -uo pipefail HOST="${CLAWMATES_HOST:-gw-04}" OWNER="${CLAWMATES_OWNER_EMAIL:-om.sobh@gmail.com}" MISSIONS_ROOT="${CLAWMATES_MISSIONS_ROOT:-/var/lib/clawmates-missions}" KEEP="${KEEP:-}" GO=0 [ "${1:-}" = "--yes" ] && GO=1 die() { printf 'ABORT %s\n' "$*" >&2; exit 2; } psql_() { ssh "$HOST" "docker exec clawmates_postgres_1 psql -U postgres -d clawmates -tAc \"$1\""; } mint_session() { local secret hash rows secret="reset-$(openssl rand -hex 16)" hash=$(printf '%s' "$secret" | openssl dgst -sha256 -binary | openssl base64 -A | tr '+/' '-_' | tr -d '=') rows=$(psql_ "insert into auth_sessions (user_id, token_hash, expires_at) \ select id, '$hash', now() + interval '30 minutes' from users where email='$OWNER' limit 1 returning 1;" \ 2>/dev/null | head -1 | tr -d '[:space:]') [ "$rows" = "1" ] || return 1 printf '%s' "$secret" } api() { # api ssh "$HOST" "docker run --rm --network clawmates_core curlimages/curl:latest -s -o /dev/null -w '%{http_code}' \ -X $2 -H 'Authorization: Bearer $1' http://clawmates_server_1:8080$3" } ssh -o BatchMode=yes -o ConnectTimeout=10 "$HOST" true 2>/dev/null || die "cannot ssh to $HOST" # ── Before ─────────────────────────────────────────────────────── before_rows=$(psql_ "select count(*) from missions;" | tr -d '[:space:]') before_dirs=$(ssh "$HOST" "ls $MISSIONS_ROOT 2>/dev/null | grep -c '^[0-9a-f-]\{36\}$'" | tr -d '[:space:]') before_kb=$(ssh "$HOST" "du -sk $MISSIONS_ROOT 2>/dev/null | cut -f1" | tr -d '[:space:]') printf 'before: %s mission row(s), %s director(y|ies), %s MiB\n' \ "$before_rows" "$before_dirs" "$((${before_kb:-0} / 1024))" # Never delete a mission that is still working. A reset that yanks a running # mission's checkout leaves a VM writing into a directory that no longer exists, # and the symptom is a phase that hangs rather than fails. busy=$(psql_ "select count(*) from missions where status = 'running';" | tr -d '[:space:]') if [ "${busy:-0}" != "0" ]; then die "$busy mission(s) are still RUNNING — wait for them or cancel them first" fi filter="" [ -n "$KEEP" ] && filter=" and title not like '%${KEEP}%'" ids=$(psql_ "select id from missions where true$filter order by created_at;" | tr -d '\r' | grep -v '^[[:space:]]*$') count=$(printf '%s\n' "$ids" | grep -c '[^[:space:]]') agents_n=$(psql_ "select count(*) from agents;" | tr -d '[:space:]') [ "${KEEP_AGENTS:-}" = "1" ] && agents_n=0 # Missions AND agents. Checking only missions here meant a fleet with zero # missions and four agents reported "nothing to delete" and exited — the reset # silently doing half its job, which is the shape this script exists to catch. if [ "${count:-0}" = "0" ] && [ "${agents_n:-0}" = "0" ]; then echo "nothing to delete" exit 0 fi if [ "$GO" != "1" ]; then printf '\nwould delete %s mission(s)%s and %s agent(s)%s. Re-run with --yes.\n' "$count" \ "${KEEP:+ (keeping titles matching '$KEEP')}" "$agents_n" \ "${KEEP_AGENTS:+ (KEEP_AGENTS=1)}" exit 0 fi token=$(mint_session) || die "could not mint a session for $OWNER" ok=0; bad=0 for id in ${ids:-}; do code=$(api "$token" DELETE "/api/missions/$id") case "$code" in 2*) ok=$((ok + 1)) ;; *) bad=$((bad + 1)); printf 'FAIL %s -> HTTP %s\n' "$id" "$code" ;; esac done printf 'deleted %s, failed %s\n' "$ok" "$bad" # ── After, and the part that matters ───────────────────────────── # # Rows gone is not the same as bytes back. Every incarnation of this cleanup has # been able to do the first while silently failing the second. sleep 5 after_rows=$(psql_ "select count(*) from missions;" | tr -d '[:space:]') after_dirs=$(ssh "$HOST" "ls $MISSIONS_ROOT 2>/dev/null | grep -c '^[0-9a-f-]\{36\}$'" | tr -d '[:space:]') after_kb=$(ssh "$HOST" "du -sk $MISSIONS_ROOT 2>/dev/null | cut -f1" | tr -d '[:space:]') printf 'after: %s mission row(s), %s director(y|ies), %s MiB\n' \ "$after_rows" "$after_dirs" "$((${after_kb:-0} / 1024))" # Any directory left without a row is an orphan the reaper missed. Named # individually — a count alone is something to shrug at. orphans=$(ssh "$HOST" "cd $MISSIONS_ROOT 2>/dev/null && ls -d */ 2>/dev/null | tr -d '/' \ | grep -E '^[0-9a-f-]{36}$'" | tr -d '\r' | grep -v '^[[:space:]]*$' || true) left=0 for d in $orphans; do row=$(psql_ "select count(*) from missions where id = '$d';" | tr -d '[:space:]') [ "${row:-0}" = "0" ] && { left=$((left + 1)); printf 'ORPHAN %s (no row, still on disk)\n' "$d"; } done # Root-owned residue is the specific way this fails: the server runs as 65532 # and cannot delete what the per-mission daemon wrote as root. rootfiles=$(ssh "$HOST" "find $MISSIONS_ROOT ! -uid 65532 2>/dev/null | wc -l" | tr -d '[:space:]') # ── Agents ─────────────────────────────────────────────────────── # # Deleted through the API for the same reason missions are: `purge_agent` is the # only path that deprovisions the ZeroClaw runtime agent, reaps the # sandbox/browser/terminal containers, unlinks the `.brain`/`.onion`, and THEN # purges the FK graph. Three call sites used to inline their own variant of that # sequence and two had silently drifted, leaving live containers behind — which # is exactly what a SQL delete would do here, every time. agents_deleted=0; agents_failed=0 if [ "${KEEP_AGENTS:-}" != "1" ]; then # `batch-delete`, not `DELETE /api/claws/{id}`. The single-agent route is a # SOFT delete, so it 404s on an agent already soft-deleted — which is every # agent that needs reaping. batch-delete hard-purges: runtime agent, # containers, `.brain`, then the FK graph. agent_ids=$(psql_ "select id from agents order by created_at;" | tr -d '\r' | grep -v '^[[:space:]]*$') before_agents=$(printf '%s\n' "$agent_ids" | grep -c '[^[:space:]]') if [ "${before_agents:-0}" -gt 0 ]; then body=$(printf '%s\n' "$agent_ids" | python3 -c 'import json,sys; print(json.dumps({"ids":[l.strip() for l in sys.stdin if l.strip()]}))') printf '%s' "$body" | ssh "$HOST" "docker run --rm -i --network clawmates_core curlimages/curl:latest -s -N \ -X POST -H 'Authorization: Bearer $token' -H 'Content-Type: application/json' -d @- \ http://clawmates_server_1:8080/api/claws/batch-delete" >/dev/null 2>&1 sleep 3 remaining=$(psql_ "select count(*) from agents;" | tr -d '[:space:]') agents_deleted=$((before_agents - ${remaining:-0})) agents_failed=${remaining:-0} fi [ "$agents_deleted$agents_failed" = "00" ] || \ printf 'agents: deleted %s, failed %s\n' "$agents_deleted" "$agents_failed" fi # Resources an agent owns beyond its row. A count of each, because "the row is # gone" has never meant "the container is gone" in this codebase. agent_rows=$(psql_ "select count(*) from agents;" | tr -d '[:space:]') stray_containers=0 stray_volumes=0 for h in $HOST ${CLAWMATES_NODES:-osobh@tank architect morpheus}; do c=$(ssh "$h" 'docker ps -a --format "{{.Names}}" 2>/dev/null | grep -cE "^(tc-agent|cm-sandbox|cm-browser)"' 2>/dev/null | tr -d '[:space:]') v=$(ssh "$h" 'docker volume ls -q 2>/dev/null | grep -c "^clawmates_agent_"' 2>/dev/null | tr -d '[:space:]') stray_containers=$((stray_containers + ${c:-0})) stray_volumes=$((stray_volumes + ${v:-0})) done echo if [ "$bad" = "0" ] && [ "$left" = "0" ] && [ "$agents_failed" = "0" ]; then printf 'clean: %s mission(s) and %s agent(s) deleted, no orphaned directories\n' \ "$ok" "$agents_deleted" [ "${rootfiles:-0}" = "0" ] || printf 'note: %s root-owned file(s) remain under %s\n' "$rootfiles" "$MISSIONS_ROOT" [ "${agent_rows:-0}" = "0" ] || printf 'note: %s agent row(s) remain (KEEP_AGENTS?)\n' "$agent_rows" # Reported, never auto-removed. An agent volume is `~/drives` — a user's # files — and a reset script is not the place to decide those are disposable. [ "$stray_containers" = "0" ] || printf 'note: %s agent container(s) still on the fleet\n' "$stray_containers" [ "$stray_volumes" = "0" ] || printf 'note: %s agent volume(s) (~/drives) remain — kept on purpose, they hold user files\n' "$stray_volumes" exit 0 fi printf 'NOT clean: %s mission delete(s) failed, %s orphaned director(y|ies), %s agent delete(s) failed\n' \ "$bad" "$left" "$agents_failed" exit 1