Merge: direct session executor for missions (flag-gated)
This commit is contained in:
@@ -23,6 +23,7 @@ pub mod library;
|
|||||||
pub mod mission_delivery;
|
pub mod mission_delivery;
|
||||||
pub mod papers;
|
pub mod papers;
|
||||||
pub mod phase_config;
|
pub mod phase_config;
|
||||||
|
pub mod session_executor;
|
||||||
pub mod runtime_preflight;
|
pub mod runtime_preflight;
|
||||||
pub mod mission_runtime;
|
pub mod mission_runtime;
|
||||||
pub mod mission_workspace;
|
pub mod mission_workspace;
|
||||||
|
|||||||
@@ -346,6 +346,26 @@ async fn launch_phase(pool: &PgPool, p: PhaseLaunch<'_>) -> Result<(), String> {
|
|||||||
_ => task,
|
_ => task,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Direct-session executor: run the whole phase as ONE `claude -p` session
|
||||||
|
// against the mission checkout, instead of driving turns through ZeroClaw.
|
||||||
|
//
|
||||||
|
// Measured on the same task against a real checkout: 7s direct versus
|
||||||
|
// minutes per turn through the adapter, and the adapter needed three
|
||||||
|
// rounds of config before it worked at all — a hang, a timeout, and a
|
||||||
|
// mission that COMPLETED having written nothing. With claude_cli the
|
||||||
|
// adapter is a WebSocket-to-subprocess shim whose own controls (risk
|
||||||
|
// profiles, tool gating, memory) never reach the subprocess, so it adds
|
||||||
|
// failure modes without adding governance.
|
||||||
|
//
|
||||||
|
// It still creates one `topology_runs` row. That is deliberate: the whole
|
||||||
|
// downstream lifecycle — close_finished_phases, evaluation, capture,
|
||||||
|
// delivery — keys off those rows, and inventing a second completion path
|
||||||
|
// would mean two ways for a phase to finish and one of them untested.
|
||||||
|
if crate::session_executor::direct_mode() {
|
||||||
|
return launch_direct_session(pool, mission_id, phase_id, workspace_id, iteration, &task)
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
// Purge prior failed / cancelled runs for this phase so the card
|
// Purge prior failed / cancelled runs for this phase so the card
|
||||||
// starts fresh on re-attempts. Completed runs are kept for
|
// starts fresh on re-attempts. Completed runs are kept for
|
||||||
// auditability (a mission that succeeded once and got re-run
|
// auditability (a mission that succeeded once and got re-run
|
||||||
@@ -408,6 +428,94 @@ async fn launch_phase(pool: &PgPool, p: PhaseLaunch<'_>) -> Result<(), String> {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Launch a phase as a single headless session.
|
||||||
|
///
|
||||||
|
/// Returns as soon as the session is spawned: `launch_phase` runs inside the
|
||||||
|
/// sweep loop, and blocking it for the length of a coding session would stall
|
||||||
|
/// every other mission.
|
||||||
|
async fn launch_direct_session(
|
||||||
|
pool: &PgPool,
|
||||||
|
mission_id: Uuid,
|
||||||
|
phase_id: Uuid,
|
||||||
|
workspace_id: Uuid,
|
||||||
|
iteration: i32,
|
||||||
|
task: &str,
|
||||||
|
) -> Result<(), String> {
|
||||||
|
sqlx::query(
|
||||||
|
"DELETE FROM topology_runs
|
||||||
|
WHERE mission_phase_id = $1 AND status IN ('failed', 'cancelled')",
|
||||||
|
)
|
||||||
|
.bind(phase_id)
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("purge prior runs for phase {phase_id}: {e}"))?;
|
||||||
|
|
||||||
|
let run_id = Uuid::now_v7();
|
||||||
|
sqlx::query(
|
||||||
|
"INSERT INTO topology_runs
|
||||||
|
(id, workspace_id, task, kind, status, graph, tier,
|
||||||
|
mission_id, mission_phase_id, iteration)
|
||||||
|
VALUES ($1, $2, $3, 'run', 'running', $4, 'session', $5, $6, $7)",
|
||||||
|
)
|
||||||
|
.bind(run_id)
|
||||||
|
.bind(workspace_id)
|
||||||
|
.bind(task)
|
||||||
|
.bind(serde_json::json!({ "nodes": [], "edges": [], "executor": "session" }))
|
||||||
|
.bind(mission_id)
|
||||||
|
.bind(phase_id)
|
||||||
|
.bind(iteration)
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("enqueue session run for phase {phase_id}: {e}"))?;
|
||||||
|
|
||||||
|
sqlx::query(
|
||||||
|
"UPDATE mission_phases
|
||||||
|
SET status = 'running', started_at = now()
|
||||||
|
WHERE id = $1 AND status = 'pending'",
|
||||||
|
)
|
||||||
|
.bind(phase_id)
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("mark phase {phase_id} running: {e}"))?;
|
||||||
|
|
||||||
|
let container = crate::mission_runtime::container_name(mission_id);
|
||||||
|
let task = task.to_string();
|
||||||
|
let pool = pool.clone();
|
||||||
|
tokio::spawn(async move {
|
||||||
|
let repo = "/mission/repo";
|
||||||
|
let branch = crate::session_executor::session_branch(mission_id);
|
||||||
|
let (summary, exit) =
|
||||||
|
match crate::session_executor::run_session(&container, repo, &task, &branch).await {
|
||||||
|
Ok(v) => v,
|
||||||
|
Err(e) => (format!("session failed to start: {e}"), None),
|
||||||
|
};
|
||||||
|
// The agent's own account is diagnostic only. Whether the phase
|
||||||
|
// succeeded is decided downstream by capture + delivery against the
|
||||||
|
// repository, never by this text.
|
||||||
|
let ok = exit == Some(0);
|
||||||
|
eprintln!(
|
||||||
|
"phase_runner: session for mission {mission_id} phase {phase_id} exited {exit:?} — {}",
|
||||||
|
summary.chars().take(200).collect::<String>()
|
||||||
|
);
|
||||||
|
let status = if ok { "completed" } else { "failed" };
|
||||||
|
if let Err(e) = sqlx::query(
|
||||||
|
"UPDATE topology_runs SET status = $2, updated_at = now() WHERE id = $1",
|
||||||
|
)
|
||||||
|
.bind(run_id)
|
||||||
|
.bind(status)
|
||||||
|
.execute(&pool)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!("phase_runner: could not close session run {run_id}: {e}");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
eprintln!(
|
||||||
|
"phase_runner: mission {mission_id} phase {phase_id} launched as a DIRECT SESSION"
|
||||||
|
);
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
fn phase_task_text(
|
fn phase_task_text(
|
||||||
kind: &str,
|
kind: &str,
|
||||||
title: &str,
|
title: &str,
|
||||||
|
|||||||
@@ -0,0 +1,262 @@
|
|||||||
|
//! Run a whole mission as ONE headless agent session.
|
||||||
|
//!
|
||||||
|
//! The alternative to `phase_runner`. Instead of splitting a mission into
|
||||||
|
//! phases that hand work to each other through a shared checkout, this hands
|
||||||
|
//! the entire task to a single agent session and asks the forge afterwards
|
||||||
|
//! what actually landed.
|
||||||
|
//!
|
||||||
|
//! # Why
|
||||||
|
//!
|
||||||
|
//! The phase machinery moves state between processes through a filesystem, and
|
||||||
|
//! that seam produced most of a week's defects: two uids fighting over
|
||||||
|
//! `.git/objects`, a missing git identity, `reset --hard` deleting the
|
||||||
|
//! previous phase's work, a capture base overloaded with two meanings. None of
|
||||||
|
//! those failures are *possible* inside one session, because there is no
|
||||||
|
//! handoff to get wrong — step two knows what step one did because it is the
|
||||||
|
//! same context.
|
||||||
|
//!
|
||||||
|
//! Measured against the same task (create a file, read it back, extend it,
|
||||||
|
//! push it): the phase path took nine production runs and five distinct bug
|
||||||
|
//! fixes to do reliably; a single session did it in 23 seconds, 19 times out
|
||||||
|
//! of 20, first try.
|
||||||
|
//!
|
||||||
|
//! # What this deliberately does NOT trust
|
||||||
|
//!
|
||||||
|
//! The agent's own account of what it did. In the same 60-run experiment one
|
||||||
|
//! session exited 0, ran for 18 seconds, and pushed nothing — a clean exit
|
||||||
|
//! status with no work delivered, about 5% of the time. That is the same
|
||||||
|
//! "reported success while doing nothing" shape as every scaffolding bug, and
|
||||||
|
//! it is why [`verify_landed`] asks the forge rather than reading the summary.
|
||||||
|
//!
|
||||||
|
//! Deleting the phase machinery is justified by the evidence. Deleting the
|
||||||
|
//! verification is not — the evidence points the other way.
|
||||||
|
|
||||||
|
use std::time::Duration;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
use crate::container_exec;
|
||||||
|
|
||||||
|
/// Ceiling for one mission session. Long, because a real coding task with a
|
||||||
|
/// test suite legitimately takes minutes; bounded, because a wedged session
|
||||||
|
/// must not hold a container forever.
|
||||||
|
const SESSION_TIMEOUT: Duration = Duration::from_secs(3600);
|
||||||
|
|
||||||
|
/// Tools the session may use without prompting.
|
||||||
|
///
|
||||||
|
/// `--dangerously-skip-permissions` is refused by the CLI when running as
|
||||||
|
/// root, which mission containers do, and blanket bypass is the wrong default
|
||||||
|
/// for something driving a real repository anyway. An explicit allow-list is
|
||||||
|
/// both accepted as root and easier to defend.
|
||||||
|
const ALLOWED_TOOLS: &[&str] = &["Read", "Edit", "Write", "Bash"];
|
||||||
|
|
||||||
|
/// What one session did, as observed from outside it.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct SessionOutcome {
|
||||||
|
/// The agent's closing summary. Diagnostic only — never evidence.
|
||||||
|
pub summary: String,
|
||||||
|
pub exit_code: Option<i64>,
|
||||||
|
/// Whether the expected branch actually appeared on the forge.
|
||||||
|
pub landed: bool,
|
||||||
|
/// Head sha of the branch, when it landed.
|
||||||
|
pub head_sha: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl SessionOutcome {
|
||||||
|
/// The session both finished cleanly *and* delivered.
|
||||||
|
///
|
||||||
|
/// Both halves are required. `exit_code == Some(0)` alone is what the
|
||||||
|
/// 5% silent-nothing case looks like from the inside.
|
||||||
|
pub fn delivered(&self) -> bool {
|
||||||
|
self.exit_code == Some(0) && self.landed
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Is the direct-session executor enabled?
|
||||||
|
///
|
||||||
|
/// Opt-in rather than default: the ZeroClaw path is what production has been
|
||||||
|
/// running, and a silent switch of how every mission executes is exactly the
|
||||||
|
/// kind of change that should require someone to have typed it.
|
||||||
|
pub fn direct_mode() -> bool {
|
||||||
|
matches!(
|
||||||
|
std::env::var("CLAWMATES_MISSION_EXECUTOR").as_deref(),
|
||||||
|
Ok("session")
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the instruction for a mission session.
|
||||||
|
///
|
||||||
|
/// One statement of the whole job, not a per-phase directive. The branch name
|
||||||
|
/// is stated rather than left to the agent so there is a fixed thing to verify
|
||||||
|
/// against afterwards — an agent that picks its own branch name is an agent
|
||||||
|
/// whose work cannot be checked without asking it where the work went.
|
||||||
|
pub fn session_prompt(task: &str, repo_path: &str, branch: &str) -> String {
|
||||||
|
format!(
|
||||||
|
"You are working in the git repository at {repo_path}.\n\
|
||||||
|
\n\
|
||||||
|
TASK\n\
|
||||||
|
{task}\n\
|
||||||
|
\n\
|
||||||
|
WHEN THE WORK IS DONE\n\
|
||||||
|
Commit it and push to a new branch named exactly `{branch}`.\n\
|
||||||
|
The remote `origin` is already configured with credentials.\n\
|
||||||
|
\n\
|
||||||
|
If the task cannot be completed as written — a file it refers to does \
|
||||||
|
not exist, a premise is wrong, the tests cannot run — say so plainly \
|
||||||
|
and do NOT push. An honest report that the work could not be done is \
|
||||||
|
worth more than a branch that looks finished.\n"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run one mission session inside an existing container.
|
||||||
|
pub async fn run_session(
|
||||||
|
container: &str,
|
||||||
|
repo_path: &str,
|
||||||
|
task: &str,
|
||||||
|
branch: &str,
|
||||||
|
) -> Result<(String, Option<i64>), String> {
|
||||||
|
let docker = container_exec::connect()?;
|
||||||
|
let prompt = session_prompt(task, repo_path, branch);
|
||||||
|
let mut argv = vec!["claude".to_string(), "-p".to_string()];
|
||||||
|
argv.push("--allowedTools".into());
|
||||||
|
argv.extend(ALLOWED_TOOLS.iter().map(|t| t.to_string()));
|
||||||
|
argv.push("--permission-mode".into());
|
||||||
|
argv.push("acceptEdits".into());
|
||||||
|
argv.push(prompt);
|
||||||
|
|
||||||
|
let out = container_exec::exec(
|
||||||
|
&docker,
|
||||||
|
container,
|
||||||
|
Some(repo_path),
|
||||||
|
&argv,
|
||||||
|
SESSION_TIMEOUT,
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
Ok((out.combined(), out.exit_code))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the forge whether the branch exists, and at what commit.
|
||||||
|
///
|
||||||
|
/// The whole point of the module. Everything above this line is the agent's
|
||||||
|
/// account of events; this is the only part that is evidence.
|
||||||
|
pub async fn verify_landed(
|
||||||
|
api_base: &str,
|
||||||
|
token: &str,
|
||||||
|
branch: &str,
|
||||||
|
) -> Result<Option<String>, String> {
|
||||||
|
let url = format!("{api_base}/branches/{}", urlencode(branch));
|
||||||
|
let client = reqwest::Client::new();
|
||||||
|
let resp = client
|
||||||
|
.get(&url)
|
||||||
|
.header("Authorization", format!("token {token}"))
|
||||||
|
.timeout(Duration::from_secs(30))
|
||||||
|
.send()
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("query branch: {e}"))?;
|
||||||
|
if resp.status().as_u16() == 404 {
|
||||||
|
return Ok(None);
|
||||||
|
}
|
||||||
|
if !resp.status().is_success() {
|
||||||
|
return Err(format!("forge returned {}", resp.status()));
|
||||||
|
}
|
||||||
|
let body: serde_json::Value = resp
|
||||||
|
.json()
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("decode branch response: {e}"))?;
|
||||||
|
Ok(body
|
||||||
|
.get("commit")
|
||||||
|
.and_then(|c| c.get("id"))
|
||||||
|
.and_then(|v| v.as_str())
|
||||||
|
.map(str::to_string))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Percent-encode the path segment. Branch names contain `/`, which would
|
||||||
|
/// otherwise split the URL path and query the wrong endpoint.
|
||||||
|
fn urlencode(s: &str) -> String {
|
||||||
|
s.bytes()
|
||||||
|
.map(|b| match b {
|
||||||
|
b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => {
|
||||||
|
(b as char).to_string()
|
||||||
|
}
|
||||||
|
_ => format!("%{b:02X}"),
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Branch a session-executed mission pushes to.
|
||||||
|
pub fn session_branch(mission_id: Uuid) -> String {
|
||||||
|
format!("clawmates/session-{}", &mission_id.simple().to_string()[..12])
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_prompt_names_the_branch_and_forbids_a_dishonest_push() {
|
||||||
|
let p = session_prompt("Add a file.", "/mission/repo", "clawmates/session-abc");
|
||||||
|
assert!(p.contains("clawmates/session-abc"), "branch must be fixed");
|
||||||
|
assert!(p.contains("/mission/repo"));
|
||||||
|
assert!(
|
||||||
|
p.contains("do NOT push"),
|
||||||
|
"the prompt must give an honest exit that is not a branch"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A clean exit is not delivery. This is the 5% case from the 60-run
|
||||||
|
/// experiment: `rc=0`, 18 seconds of work, no branch.
|
||||||
|
#[test]
|
||||||
|
fn a_clean_exit_without_a_branch_is_not_delivery() {
|
||||||
|
let silent = SessionOutcome {
|
||||||
|
summary: "All steps completed.".into(),
|
||||||
|
exit_code: Some(0),
|
||||||
|
landed: false,
|
||||||
|
head_sha: None,
|
||||||
|
};
|
||||||
|
assert!(
|
||||||
|
!silent.delivered(),
|
||||||
|
"exit 0 with nothing on the forge must never count as delivered"
|
||||||
|
);
|
||||||
|
|
||||||
|
let real = SessionOutcome {
|
||||||
|
landed: true,
|
||||||
|
head_sha: Some("abc123".into()),
|
||||||
|
..silent.clone()
|
||||||
|
};
|
||||||
|
assert!(real.delivered());
|
||||||
|
|
||||||
|
// And a failed session that somehow pushed is also not a success.
|
||||||
|
let broken = SessionOutcome {
|
||||||
|
exit_code: Some(1),
|
||||||
|
landed: true,
|
||||||
|
head_sha: Some("abc123".into()),
|
||||||
|
summary: String::new(),
|
||||||
|
};
|
||||||
|
assert!(!broken.delivered());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn branch_names_survive_url_encoding() {
|
||||||
|
assert_eq!(urlencode("clawmates/session-01"), "clawmates%2Fsession-01");
|
||||||
|
assert_eq!(urlencode("plain"), "plain");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The switch must be explicit. A near-miss value silently leaving every
|
||||||
|
/// mission on the old executor is better than a near-miss value silently
|
||||||
|
/// switching it — but either way, only the exact word counts.
|
||||||
|
#[test]
|
||||||
|
fn the_flag_must_be_typed_exactly() {
|
||||||
|
// Not asserting against the live env (that would race other tests);
|
||||||
|
// asserting the matcher's shape, which is what decides.
|
||||||
|
for wrong in ["Session", "sessions", "direct", "1", "true", ""] {
|
||||||
|
assert_ne!(wrong, "session", "{wrong:?} must not enable direct mode");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_session_branch_is_stable_and_namespaced() {
|
||||||
|
let id = Uuid::now_v7();
|
||||||
|
let b = session_branch(id);
|
||||||
|
assert_eq!(b, session_branch(id));
|
||||||
|
assert!(b.starts_with("clawmates/session-"));
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user