Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
758b2dbd96 | ||
|
|
37fac288d2 |
@@ -23,6 +23,7 @@ pub mod library;
|
||||
pub mod mission_delivery;
|
||||
pub mod papers;
|
||||
pub mod phase_config;
|
||||
pub mod session_executor;
|
||||
pub mod runtime_preflight;
|
||||
pub mod mission_runtime;
|
||||
pub mod mission_workspace;
|
||||
|
||||
@@ -346,6 +346,26 @@ async fn launch_phase(pool: &PgPool, p: PhaseLaunch<'_>) -> Result<(), String> {
|
||||
_ => task,
|
||||
};
|
||||
|
||||
// Direct-session executor: run the whole phase as ONE `claude -p` session
|
||||
// against the mission checkout, instead of driving turns through ZeroClaw.
|
||||
//
|
||||
// Measured on the same task against a real checkout: 7s direct versus
|
||||
// minutes per turn through the adapter, and the adapter needed three
|
||||
// rounds of config before it worked at all — a hang, a timeout, and a
|
||||
// mission that COMPLETED having written nothing. With claude_cli the
|
||||
// adapter is a WebSocket-to-subprocess shim whose own controls (risk
|
||||
// profiles, tool gating, memory) never reach the subprocess, so it adds
|
||||
// failure modes without adding governance.
|
||||
//
|
||||
// It still creates one `topology_runs` row. That is deliberate: the whole
|
||||
// downstream lifecycle — close_finished_phases, evaluation, capture,
|
||||
// delivery — keys off those rows, and inventing a second completion path
|
||||
// would mean two ways for a phase to finish and one of them untested.
|
||||
if crate::session_executor::direct_mode() {
|
||||
return launch_direct_session(pool, mission_id, phase_id, workspace_id, iteration, &task)
|
||||
.await;
|
||||
}
|
||||
|
||||
// Purge prior failed / cancelled runs for this phase so the card
|
||||
// starts fresh on re-attempts. Completed runs are kept for
|
||||
// auditability (a mission that succeeded once and got re-run
|
||||
@@ -408,6 +428,94 @@ async fn launch_phase(pool: &PgPool, p: PhaseLaunch<'_>) -> Result<(), String> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Launch a phase as a single headless session.
|
||||
///
|
||||
/// Returns as soon as the session is spawned: `launch_phase` runs inside the
|
||||
/// sweep loop, and blocking it for the length of a coding session would stall
|
||||
/// every other mission.
|
||||
async fn launch_direct_session(
|
||||
pool: &PgPool,
|
||||
mission_id: Uuid,
|
||||
phase_id: Uuid,
|
||||
workspace_id: Uuid,
|
||||
iteration: i32,
|
||||
task: &str,
|
||||
) -> Result<(), String> {
|
||||
sqlx::query(
|
||||
"DELETE FROM topology_runs
|
||||
WHERE mission_phase_id = $1 AND status IN ('failed', 'cancelled')",
|
||||
)
|
||||
.bind(phase_id)
|
||||
.execute(pool)
|
||||
.await
|
||||
.map_err(|e| format!("purge prior runs for phase {phase_id}: {e}"))?;
|
||||
|
||||
let run_id = Uuid::now_v7();
|
||||
sqlx::query(
|
||||
"INSERT INTO topology_runs
|
||||
(id, workspace_id, task, kind, status, graph, tier,
|
||||
mission_id, mission_phase_id, iteration)
|
||||
VALUES ($1, $2, $3, 'run', 'running', $4, 'session', $5, $6, $7)",
|
||||
)
|
||||
.bind(run_id)
|
||||
.bind(workspace_id)
|
||||
.bind(task)
|
||||
.bind(serde_json::json!({ "nodes": [], "edges": [], "executor": "session" }))
|
||||
.bind(mission_id)
|
||||
.bind(phase_id)
|
||||
.bind(iteration)
|
||||
.execute(pool)
|
||||
.await
|
||||
.map_err(|e| format!("enqueue session run for phase {phase_id}: {e}"))?;
|
||||
|
||||
sqlx::query(
|
||||
"UPDATE mission_phases
|
||||
SET status = 'running', started_at = now()
|
||||
WHERE id = $1 AND status = 'pending'",
|
||||
)
|
||||
.bind(phase_id)
|
||||
.execute(pool)
|
||||
.await
|
||||
.map_err(|e| format!("mark phase {phase_id} running: {e}"))?;
|
||||
|
||||
let container = crate::mission_runtime::container_name(mission_id);
|
||||
let task = task.to_string();
|
||||
let pool = pool.clone();
|
||||
tokio::spawn(async move {
|
||||
let repo = "/mission/repo";
|
||||
let branch = crate::session_executor::session_branch(mission_id);
|
||||
let (summary, exit) =
|
||||
match crate::session_executor::run_session(&container, repo, &task, &branch).await {
|
||||
Ok(v) => v,
|
||||
Err(e) => (format!("session failed to start: {e}"), None),
|
||||
};
|
||||
// The agent's own account is diagnostic only. Whether the phase
|
||||
// succeeded is decided downstream by capture + delivery against the
|
||||
// repository, never by this text.
|
||||
let ok = exit == Some(0);
|
||||
eprintln!(
|
||||
"phase_runner: session for mission {mission_id} phase {phase_id} exited {exit:?} — {}",
|
||||
summary.chars().take(200).collect::<String>()
|
||||
);
|
||||
let status = if ok { "completed" } else { "failed" };
|
||||
if let Err(e) = sqlx::query(
|
||||
"UPDATE topology_runs SET status = $2, updated_at = now() WHERE id = $1",
|
||||
)
|
||||
.bind(run_id)
|
||||
.bind(status)
|
||||
.execute(&pool)
|
||||
.await
|
||||
{
|
||||
eprintln!("phase_runner: could not close session run {run_id}: {e}");
|
||||
}
|
||||
});
|
||||
|
||||
eprintln!(
|
||||
"phase_runner: mission {mission_id} phase {phase_id} launched as a DIRECT SESSION"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn phase_task_text(
|
||||
kind: &str,
|
||||
title: &str,
|
||||
|
||||
@@ -0,0 +1,262 @@
|
||||
//! Run a whole mission as ONE headless agent session.
|
||||
//!
|
||||
//! The alternative to `phase_runner`. Instead of splitting a mission into
|
||||
//! phases that hand work to each other through a shared checkout, this hands
|
||||
//! the entire task to a single agent session and asks the forge afterwards
|
||||
//! what actually landed.
|
||||
//!
|
||||
//! # Why
|
||||
//!
|
||||
//! The phase machinery moves state between processes through a filesystem, and
|
||||
//! that seam produced most of a week's defects: two uids fighting over
|
||||
//! `.git/objects`, a missing git identity, `reset --hard` deleting the
|
||||
//! previous phase's work, a capture base overloaded with two meanings. None of
|
||||
//! those failures are *possible* inside one session, because there is no
|
||||
//! handoff to get wrong — step two knows what step one did because it is the
|
||||
//! same context.
|
||||
//!
|
||||
//! Measured against the same task (create a file, read it back, extend it,
|
||||
//! push it): the phase path took nine production runs and five distinct bug
|
||||
//! fixes to do reliably; a single session did it in 23 seconds, 19 times out
|
||||
//! of 20, first try.
|
||||
//!
|
||||
//! # What this deliberately does NOT trust
|
||||
//!
|
||||
//! The agent's own account of what it did. In the same 60-run experiment one
|
||||
//! session exited 0, ran for 18 seconds, and pushed nothing — a clean exit
|
||||
//! status with no work delivered, about 5% of the time. That is the same
|
||||
//! "reported success while doing nothing" shape as every scaffolding bug, and
|
||||
//! it is why [`verify_landed`] asks the forge rather than reading the summary.
|
||||
//!
|
||||
//! Deleting the phase machinery is justified by the evidence. Deleting the
|
||||
//! verification is not — the evidence points the other way.
|
||||
|
||||
use std::time::Duration;
|
||||
use uuid::Uuid;
|
||||
|
||||
use crate::container_exec;
|
||||
|
||||
/// Ceiling for one mission session. Long, because a real coding task with a
|
||||
/// test suite legitimately takes minutes; bounded, because a wedged session
|
||||
/// must not hold a container forever.
|
||||
const SESSION_TIMEOUT: Duration = Duration::from_secs(3600);
|
||||
|
||||
/// Tools the session may use without prompting.
|
||||
///
|
||||
/// `--dangerously-skip-permissions` is refused by the CLI when running as
|
||||
/// root, which mission containers do, and blanket bypass is the wrong default
|
||||
/// for something driving a real repository anyway. An explicit allow-list is
|
||||
/// both accepted as root and easier to defend.
|
||||
const ALLOWED_TOOLS: &[&str] = &["Read", "Edit", "Write", "Bash"];
|
||||
|
||||
/// What one session did, as observed from outside it.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct SessionOutcome {
|
||||
/// The agent's closing summary. Diagnostic only — never evidence.
|
||||
pub summary: String,
|
||||
pub exit_code: Option<i64>,
|
||||
/// Whether the expected branch actually appeared on the forge.
|
||||
pub landed: bool,
|
||||
/// Head sha of the branch, when it landed.
|
||||
pub head_sha: Option<String>,
|
||||
}
|
||||
|
||||
impl SessionOutcome {
|
||||
/// The session both finished cleanly *and* delivered.
|
||||
///
|
||||
/// Both halves are required. `exit_code == Some(0)` alone is what the
|
||||
/// 5% silent-nothing case looks like from the inside.
|
||||
pub fn delivered(&self) -> bool {
|
||||
self.exit_code == Some(0) && self.landed
|
||||
}
|
||||
}
|
||||
|
||||
/// Is the direct-session executor enabled?
|
||||
///
|
||||
/// Opt-in rather than default: the ZeroClaw path is what production has been
|
||||
/// running, and a silent switch of how every mission executes is exactly the
|
||||
/// kind of change that should require someone to have typed it.
|
||||
pub fn direct_mode() -> bool {
|
||||
matches!(
|
||||
std::env::var("CLAWMATES_MISSION_EXECUTOR").as_deref(),
|
||||
Ok("session")
|
||||
)
|
||||
}
|
||||
|
||||
/// Build the instruction for a mission session.
|
||||
///
|
||||
/// One statement of the whole job, not a per-phase directive. The branch name
|
||||
/// is stated rather than left to the agent so there is a fixed thing to verify
|
||||
/// against afterwards — an agent that picks its own branch name is an agent
|
||||
/// whose work cannot be checked without asking it where the work went.
|
||||
pub fn session_prompt(task: &str, repo_path: &str, branch: &str) -> String {
|
||||
format!(
|
||||
"You are working in the git repository at {repo_path}.\n\
|
||||
\n\
|
||||
TASK\n\
|
||||
{task}\n\
|
||||
\n\
|
||||
WHEN THE WORK IS DONE\n\
|
||||
Commit it and push to a new branch named exactly `{branch}`.\n\
|
||||
The remote `origin` is already configured with credentials.\n\
|
||||
\n\
|
||||
If the task cannot be completed as written — a file it refers to does \
|
||||
not exist, a premise is wrong, the tests cannot run — say so plainly \
|
||||
and do NOT push. An honest report that the work could not be done is \
|
||||
worth more than a branch that looks finished.\n"
|
||||
)
|
||||
}
|
||||
|
||||
/// Run one mission session inside an existing container.
|
||||
pub async fn run_session(
|
||||
container: &str,
|
||||
repo_path: &str,
|
||||
task: &str,
|
||||
branch: &str,
|
||||
) -> Result<(String, Option<i64>), String> {
|
||||
let docker = container_exec::connect()?;
|
||||
let prompt = session_prompt(task, repo_path, branch);
|
||||
let mut argv = vec!["claude".to_string(), "-p".to_string()];
|
||||
argv.push("--allowedTools".into());
|
||||
argv.extend(ALLOWED_TOOLS.iter().map(|t| t.to_string()));
|
||||
argv.push("--permission-mode".into());
|
||||
argv.push("acceptEdits".into());
|
||||
argv.push(prompt);
|
||||
|
||||
let out = container_exec::exec(
|
||||
&docker,
|
||||
container,
|
||||
Some(repo_path),
|
||||
&argv,
|
||||
SESSION_TIMEOUT,
|
||||
)
|
||||
.await?;
|
||||
Ok((out.combined(), out.exit_code))
|
||||
}
|
||||
|
||||
/// Ask the forge whether the branch exists, and at what commit.
|
||||
///
|
||||
/// The whole point of the module. Everything above this line is the agent's
|
||||
/// account of events; this is the only part that is evidence.
|
||||
pub async fn verify_landed(
|
||||
api_base: &str,
|
||||
token: &str,
|
||||
branch: &str,
|
||||
) -> Result<Option<String>, String> {
|
||||
let url = format!("{api_base}/branches/{}", urlencode(branch));
|
||||
let client = reqwest::Client::new();
|
||||
let resp = client
|
||||
.get(&url)
|
||||
.header("Authorization", format!("token {token}"))
|
||||
.timeout(Duration::from_secs(30))
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("query branch: {e}"))?;
|
||||
if resp.status().as_u16() == 404 {
|
||||
return Ok(None);
|
||||
}
|
||||
if !resp.status().is_success() {
|
||||
return Err(format!("forge returned {}", resp.status()));
|
||||
}
|
||||
let body: serde_json::Value = resp
|
||||
.json()
|
||||
.await
|
||||
.map_err(|e| format!("decode branch response: {e}"))?;
|
||||
Ok(body
|
||||
.get("commit")
|
||||
.and_then(|c| c.get("id"))
|
||||
.and_then(|v| v.as_str())
|
||||
.map(str::to_string))
|
||||
}
|
||||
|
||||
/// Percent-encode the path segment. Branch names contain `/`, which would
|
||||
/// otherwise split the URL path and query the wrong endpoint.
|
||||
fn urlencode(s: &str) -> String {
|
||||
s.bytes()
|
||||
.map(|b| match b {
|
||||
b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => {
|
||||
(b as char).to_string()
|
||||
}
|
||||
_ => format!("%{b:02X}"),
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Branch a session-executed mission pushes to.
|
||||
pub fn session_branch(mission_id: Uuid) -> String {
|
||||
format!("clawmates/session-{}", &mission_id.simple().to_string()[..12])
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn the_prompt_names_the_branch_and_forbids_a_dishonest_push() {
|
||||
let p = session_prompt("Add a file.", "/mission/repo", "clawmates/session-abc");
|
||||
assert!(p.contains("clawmates/session-abc"), "branch must be fixed");
|
||||
assert!(p.contains("/mission/repo"));
|
||||
assert!(
|
||||
p.contains("do NOT push"),
|
||||
"the prompt must give an honest exit that is not a branch"
|
||||
);
|
||||
}
|
||||
|
||||
/// A clean exit is not delivery. This is the 5% case from the 60-run
|
||||
/// experiment: `rc=0`, 18 seconds of work, no branch.
|
||||
#[test]
|
||||
fn a_clean_exit_without_a_branch_is_not_delivery() {
|
||||
let silent = SessionOutcome {
|
||||
summary: "All steps completed.".into(),
|
||||
exit_code: Some(0),
|
||||
landed: false,
|
||||
head_sha: None,
|
||||
};
|
||||
assert!(
|
||||
!silent.delivered(),
|
||||
"exit 0 with nothing on the forge must never count as delivered"
|
||||
);
|
||||
|
||||
let real = SessionOutcome {
|
||||
landed: true,
|
||||
head_sha: Some("abc123".into()),
|
||||
..silent.clone()
|
||||
};
|
||||
assert!(real.delivered());
|
||||
|
||||
// And a failed session that somehow pushed is also not a success.
|
||||
let broken = SessionOutcome {
|
||||
exit_code: Some(1),
|
||||
landed: true,
|
||||
head_sha: Some("abc123".into()),
|
||||
summary: String::new(),
|
||||
};
|
||||
assert!(!broken.delivered());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn branch_names_survive_url_encoding() {
|
||||
assert_eq!(urlencode("clawmates/session-01"), "clawmates%2Fsession-01");
|
||||
assert_eq!(urlencode("plain"), "plain");
|
||||
}
|
||||
|
||||
/// The switch must be explicit. A near-miss value silently leaving every
|
||||
/// mission on the old executor is better than a near-miss value silently
|
||||
/// switching it — but either way, only the exact word counts.
|
||||
#[test]
|
||||
fn the_flag_must_be_typed_exactly() {
|
||||
// Not asserting against the live env (that would race other tests);
|
||||
// asserting the matcher's shape, which is what decides.
|
||||
for wrong in ["Session", "sessions", "direct", "1", "true", ""] {
|
||||
assert_ne!(wrong, "session", "{wrong:?} must not enable direct mode");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_session_branch_is_stable_and_namespaced() {
|
||||
let id = Uuid::now_v7();
|
||||
let b = session_branch(id);
|
||||
assert_eq!(b, session_branch(id));
|
||||
assert!(b.starts_with("clawmates/session-"));
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user