Files
clawmates/crates/cm-api/tests/orphan_sweep.rs
T
Omar SobhandClaude Opus 5 7525be3791
deploy / test (push) Successful in 5m44s
deploy / build (push) Successful in 5m59s
test(orphans): the destructive sweep test is opt-in
CI mounts /var/run/docker.sock into the test container and the runner is
gw04 — the host that runs production missions. So `cargo test --workspace`
there has full access to the production docker daemon, and this test
REMOVES containers.

`adopt_existing` protects everything already present, but it cannot
protect a mission container created in the seconds between that call and
the sweep. On a laptop that race is nothing; on gw04 it is somebody's
mission.

So the destructive case now requires `CM_TEST_ORPHAN_SWEEP=1` and CI
simply does not run it. The read-only probes still run everywhere — they
create fixtures and inspect them, and never sweep.

This is the second time this test's blast radius has bitten: it reaped
two real local mission containers on its first run, and this would have
been the same mistake with production's daemon. The sweep is not the
problem — a sweep is global by nature — the harness around it is.

Co-Authored-By: Claude Opus 5 <[email protected]>
Claude-Session: https://claude.ai/code/session_018i9Ten1LU4jUr5d7TAWda9
2026-08-21 13:23:58 -07:00

353 lines
13 KiB
Rust

//! The orphan sweep's two Docker-touching seams, against real containers.
//!
//! `sweep_orphans` force-removes containers. Its decision logic is pure and
//! unit-tested in `mission_runtime`, but the two calls that talk to Docker —
//! "which containers exist" and "does this checkout hold work no remote has" —
//! had never run against a daemon. Those are exactly the ones worth exercising
//! for real: the first decides what is considered at all, and the second is the
//! only thing standing between a reaper and ten unpushed commits.
//!
//! That is not hypothetical. The orphan that motivated this sweep held
//! +3451/-30 across 30 files on a branch that existed nowhere else.
//!
//! Skips cleanly when there is no Docker, so a machine or CI runner without one
//! reports "not run" rather than failing.
use cm_api::mission_runtime::{container_name, MissionRuntimeProvisioner, UnpushedWork};
use uuid::Uuid;
/// The image is already local on any machine that runs missions, and it has
/// `git`, which the probe needs.
const FIXTURE_IMAGE: &str = "clawmates-runtime:hooks";
/// These tests must not run at the same time as each other.
///
/// `sweep_orphans` is global: it reaps EVERY orphaned `cm-runtime-mission-*`
/// container on the daemon, which on a parallel test runner includes the
/// fixtures another test in this file just started. That is not a flaw in the
/// sweep — it is what a sweep is — but it means anything here that creates a
/// mission-shaped container has to hold this lock.
///
/// Found the honest way: the reap test deleted the listing test's fixture
/// mid-run and the listing test reported a container it could not see.
static FIXTURES: std::sync::LazyLock<tokio::sync::Mutex<()>> =
std::sync::LazyLock::new(|| tokio::sync::Mutex::new(()));
fn docker_available() -> bool {
std::process::Command::new("docker")
.args(["image", "inspect", FIXTURE_IMAGE])
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.status()
.map(|s| s.success())
.unwrap_or(false)
}
/// Start a fixture container named like a mission runtime, running a shell
/// script that leaves `/mission/repo` in a known state.
fn start_fixture(id: Uuid, setup: &str) -> String {
let name = container_name(id);
let _ = std::process::Command::new("docker")
.args(["rm", "-f", &name])
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.status();
let script = format!("{setup}\nsleep 3600");
let out = std::process::Command::new("docker")
.args([
"run", "-d", "--name", &name, "--entrypoint", "sh", FIXTURE_IMAGE, "-c", &script,
])
.output()
.expect("docker run");
assert!(
out.status.success(),
"could not start fixture {name}: {}",
String::from_utf8_lossy(&out.stderr)
);
// The script has to have finished its git work before the probe runs.
std::thread::sleep(std::time::Duration::from_secs(3));
name
}
fn remove(name: &str) {
let _ = std::process::Command::new("docker")
.args(["rm", "-f", name])
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.status();
}
const GIT_INIT: &str = "set -e
mkdir -p /mission/repo && cd /mission/repo
git init -q .
git config user.email t@t && git config user.name t
echo hello > a.txt && git add a.txt && git commit -qm 'work nobody else has'";
#[tokio::test]
async fn the_sweep_can_see_containers_and_refuses_the_ones_holding_work() {
if !docker_available() {
eprintln!("orphan_sweep: no docker or no {FIXTURE_IMAGE} — not run");
return;
}
let Some(prov) = MissionRuntimeProvisioner::from_env() else {
eprintln!("orphan_sweep: no docker connection — not run");
return;
};
let _serial = FIXTURES.lock().await;
let dirty_id = Uuid::now_v7();
let clean_id = Uuid::now_v7();
let empty_id = Uuid::now_v7();
// Commits, and no remote ref anywhere: this is the container that must
// survive. It is the one the real orphan looked like.
let dirty = start_fixture(dirty_id, GIT_INIT);
// The same repo, but every commit is reachable from a remote-tracking ref,
// which is what "already pushed" looks like to `git rev-list --not
// --remotes`.
let clean = start_fixture(
clean_id,
&format!("{GIT_INIT}\ngit update-ref refs/remotes/origin/main HEAD"),
);
// No checkout at all — nothing to lose.
let empty = start_fixture(empty_id, "set -e\nmkdir -p /root");
let result = async {
let names = prov.list_mission_containers().await?;
let found: Vec<&String> = names.iter().map(|(n, _)| n).collect();
for expected in [&dirty, &clean, &empty] {
assert!(
found.iter().any(|n| *n == expected),
"the sweep cannot see {expected}; a container it cannot list is \
one it can never reap, which is the whole defect this closes. \
saw: {found:?}"
);
}
// Docker's own creation timestamp must come back, or the sweep declines
// to reap for want of an age.
let (_, created) = names
.iter()
.find(|(n, _)| n == &dirty)
.expect("dirty in listing");
assert!(
MissionRuntimeProvisioner::container_age(*created).is_some(),
"a container docker will not date is never reaped, so an absent \
timestamp here would silently disable the sweep"
);
match prov.unpushed_commits(&dirty).await {
UnpushedWork::SomeOrUnknown(why) => {
assert!(why.contains("no remote"), "{why}");
}
UnpushedWork::None => panic!(
"the probe said a checkout with an unpushed commit holds nothing — \
this is the exact answer that destroys work"
),
}
assert_eq!(
prov.unpushed_commits(&clean).await,
UnpushedWork::None,
"every commit is reachable from a remote ref, so there is nothing to lose"
);
assert_eq!(
prov.unpushed_commits(&empty).await,
UnpushedWork::None,
"no /mission/repo at all means nothing to lose"
);
Ok::<(), String>(())
}
.await;
remove(&dirty);
remove(&clean);
remove(&empty);
result.expect("orphan sweep probes");
}
/// A container we cannot question is not a container we may delete.
#[tokio::test]
async fn a_container_that_is_gone_reads_as_holding_work() {
if !docker_available() {
eprintln!("orphan_sweep: no docker — not run");
return;
}
let Some(prov) = MissionRuntimeProvisioner::from_env() else {
return;
};
match prov
.unpushed_commits("cm-runtime-mission-does-not-exist-at-all")
.await
{
UnpushedWork::SomeOrUnknown(_) => {}
UnpushedWork::None => panic!(
"an unanswerable probe must never read as 'safe to delete' — every \
failure path in this check is one-sided for that reason"
),
}
}
/// Set to run the destructive sweep test.
///
/// The other tests in this file only create fixtures and read them. This one
/// calls `sweep_orphans`, which REMOVES containers — and CI runs
/// `cargo test --workspace` inside a container with `/var/run/docker.sock`
/// mounted, on gw04, which is the host that runs production missions.
///
/// `adopt_existing` protects everything already present, but it cannot protect
/// a mission container created in the seconds between that call and the sweep.
/// On a developer machine that race is nothing; on the production host it is a
/// mission. So the destructive test is opt-in, and CI simply does not run it.
const RUN_DESTRUCTIVE: &str = "CM_TEST_ORPHAN_SWEEP";
/// The reap decision itself, against real containers.
///
/// The probes above are the inputs; this is the act. A clean orphan past its
/// grace must go, an orphan holding unpushed work must stay, and a young one
/// must stay regardless — and all three have to be true of the same sweep, in
/// one pass, because that is how it runs.
#[tokio::test]
async fn the_sweep_reaps_the_clean_orphan_and_spares_the_others() {
if !docker_available() {
eprintln!("orphan_sweep: no docker — not run");
return;
}
if std::env::var(RUN_DESTRUCTIVE).is_err() {
eprintln!(
"orphan_sweep: not run — this test removes containers, and the CI \
runner shares a docker daemon with production. Set \
{RUN_DESTRUCTIVE}=1 to run it."
);
return;
}
if MissionRuntimeProvisioner::from_env().is_none() {
return;
}
let _serial = FIXTURES.lock().await;
let pool = cm_testkit::test_pool().await;
// Adopt every mission container that already exists on this daemon.
//
// The sweep asks the DATABASE whether a container is known, and a fresh
// test database knows nothing — so on a developer machine the sweep
// classifies the live local stack's mission containers as orphans and
// reaps them. It did exactly that on the first run of this test, deleting
// two real mission containers.
//
// Giving each a row makes the test safe AND covers the case the other
// assertions do not: a container the platform still knows about is never
// touched, whatever its checkout looks like.
let adopted = adopt_existing(&pool).await;
let clean_id = Uuid::now_v7();
let dirty_id = Uuid::now_v7();
let young_id = Uuid::now_v7();
let alive = |name: &str| {
std::process::Command::new("docker")
.args(["inspect", name])
.stdout(std::process::Stdio::null())
.stderr(std::process::Stdio::null())
.status()
.map(|s| s.success())
.unwrap_or(false)
};
let clean = start_fixture(
clean_id,
&format!("{GIT_INIT}\ngit update-ref refs/remotes/origin/main HEAD"),
);
let dirty = start_fixture(dirty_id, GIT_INIT);
// Pass one, no grace: everything present is past its window, so the
// decision is made purely on whether the checkout holds work.
let swept = cm_api::mission_runtime::sweep_orphans(&pool, std::time::Duration::ZERO).await;
let clean_gone = !alive(&clean);
let dirty_alive = alive(&dirty);
// Pass two, a real grace, on a container minted seconds ago. It is clean
// and orphaned — reapable on every axis except its age — so if the grace is
// decorative this is where that shows.
//
// Started AFTER the first pass on purpose: a grace applies to every
// container in the sweep, so a fixture created before a zero-grace pass is
// reaped by that pass and proves nothing about the window. The first
// version of this test made exactly that mistake and failed itself.
let young = start_fixture(
young_id,
&format!("{GIT_INIT}\ngit update-ref refs/remotes/origin/main HEAD"),
);
let swept2 =
cm_api::mission_runtime::sweep_orphans(&pool, std::time::Duration::from_secs(3600)).await;
let young_alive = alive(&young);
remove(&clean);
remove(&dirty);
remove(&young);
for name in &adopted {
assert!(
alive(name),
"the sweep reaped {name}, which HAS a mission row — a container the \
platform still knows about must never be touched"
);
}
swept.expect("first sweep");
swept2.expect("second sweep");
assert!(
clean_gone,
"a clean orphan past its grace is exactly what this sweep exists to \
reclaim; leaving it means the disk leak is still open"
);
assert!(
dirty_alive,
"an orphan holding commits no remote has MUST survive — the container \
that motivated this held ten of them"
);
assert!(
young_alive,
"a container inside the grace window must be left alone even when it is \
otherwise reapable, or the grace is decorative"
);
}
/// Give every mission container already on this daemon a row, so the sweep
/// treats it as known and leaves it alone.
///
/// Returns the names, which then double as an assertion: none of them may be
/// reaped.
async fn adopt_existing(pool: &sqlx::PgPool) -> Vec<String> {
let Some(prov) = MissionRuntimeProvisioner::from_env() else {
return Vec::new();
};
let Ok(existing) = prov.list_mission_containers().await else {
return Vec::new();
};
let ws = Uuid::now_v7();
sqlx::query("INSERT INTO workspaces (id, name, plan) VALUES ($1, 'orphan-test', 'free')")
.bind(ws)
.execute(pool)
.await
.expect("seed workspace");
let mut names = Vec::new();
for (name, _) in existing {
let Some(id) = cm_api::mission_runtime::mission_id_from_container(&name) else {
continue;
};
sqlx::query(
"INSERT INTO missions (id, workspace_id, title, template_kind)
VALUES ($1, $2, 'adopted by orphan_sweep test', 'research_only')
ON CONFLICT (id) DO NOTHING",
)
.bind(id)
.bind(ws)
.execute(pool)
.await
.expect("adopt container");
names.push(name);
}
names
}