Seeding the name pick with the role index (0..n) started every crew at the top of the pool and took the next free names, so the first mission after the switch to per-mission crews hired Aarav, Abebe, Adaora, Adrian, Agnieszka. Unique and correct, and transparently generated. Seed from the claw's own uuid instead. UUIDv7 puts its random bytes LAST — the leading bytes are a timestamp, which would cluster the same way — so the tail is what spreads five picks across the whole pool. Co-Authored-By: Claude Opus 5 <[email protected]>
783 lines
35 KiB
Rust
783 lines
35 KiB
Rust
//! Mission-launch orchestrator. Slice 4.
|
||
//!
|
||
//! When a mission's status flips from `draft` to `running`, this
|
||
//! module fires:
|
||
//! 1. Materialize a team from `team_template_id` (if set + no
|
||
//! `team_id` yet) — inserts agents, ingests brain seeds,
|
||
//! records agent_template_link lineage, wires team_members.
|
||
//! 2. Bind the resulting `team_id` back onto the mission.
|
||
//!
|
||
//! Phase execution (running research/coding/benchmark/security_scan
|
||
//! against the world) is layered on top by Slices 5–8.
|
||
//!
|
||
//! Design notes:
|
||
//! - Runtime provisioning is opt-in. Claws are provisioned against the
|
||
//! mission's OWN daemon (`RuntimeProvisioner::for_gateway` with the
|
||
//! per-mission endpoint), falling back to the global gateway only when
|
||
//! there is no per-mission runtime. Provisioning into the global gateway
|
||
//! while the run executes on a per-mission daemon leaves that daemon
|
||
//! without the `claw_*` agents — it falls back to the default `scout`
|
||
//! agent, which cannot see `/mission/repo`.
|
||
//! Missing runtime = "insert DB rows only, no live claw" — the
|
||
//! mission still boots; live claws land the moment the runtime
|
||
//! env is configured + the mission re-launches.
|
||
//! - Every step is best-effort logged so a partial materialization
|
||
//! can be picked up by a subsequent launch rather than blocking.
|
||
|
||
use cm_db::repo::team_templates::TeamTemplateDetail;
|
||
use cm_domain::{AccessPolicy, Agent, AgentStatus, WorkspaceId};
|
||
use sqlx::PgPool;
|
||
use uuid::Uuid;
|
||
|
||
use crate::runtime_provision::RuntimeProvisioner;
|
||
|
||
/// Called from `routes::missions::set_status` when the transition is
|
||
/// draft→running. Materializes the team + returns the new team_id
|
||
/// (or an existing one — no-op when the mission already has a team).
|
||
///
|
||
/// Non-fatal on failure: logs + returns Ok(None) so the mission still
|
||
/// becomes `running` and the user can inspect the error via server
|
||
/// logs + re-attempt. Later slices surface this on the canvas.
|
||
pub async fn on_launch(
|
||
pool: &PgPool,
|
||
workspace_id: WorkspaceId,
|
||
user_id: cm_domain::UserId,
|
||
mission_id: Uuid,
|
||
node_hub: Option<std::sync::Arc<crate::fleet::NodeHub>>,
|
||
) -> Result<Option<Uuid>, String> {
|
||
eprintln!("mission_orchestrator::on_launch fired mission_id={mission_id}");
|
||
let Some(mission) = cm_db::repo::missions::get(pool, mission_id, workspace_id.as_uuid())
|
||
.await
|
||
.map_err(|e| format!("load mission: {e}"))?
|
||
else {
|
||
return Err("mission not found".into());
|
||
};
|
||
|
||
// ensure_checkout is idempotent (fetch+reset on existing clones,
|
||
// clone on missing dirs) so we run it BEFORE the team_id short-
|
||
// circuit: a re-launched or retried mission still needs a fresh
|
||
// repo checkout even though its team was minted on the first
|
||
// launch. Non-fatal — logs and continues on failure.
|
||
match crate::mission_workspace::ensure_checkout(pool, workspace_id, mission_id).await {
|
||
Ok(Some(path)) => eprintln!(
|
||
"mission_orchestrator: repo checked out at {} for mission {mission_id}",
|
||
path.display()
|
||
),
|
||
Ok(None) => eprintln!(
|
||
"mission_orchestrator: mission {mission_id} has no repo bound, skipping checkout"
|
||
),
|
||
Err(e) => eprintln!(
|
||
"mission_orchestrator: repo checkout for {mission_id} failed (continuing): {e}"
|
||
),
|
||
}
|
||
|
||
// Provision the per-mission ZeroClaw runtime container (C3).
|
||
// Idempotent: returns the endpoint if the container is already
|
||
// running. Falls back silently when docker is unreachable so
|
||
// dev-mode + tests still work — the topology_worker will use the
|
||
// shared runtime endpoint in that case.
|
||
// The mission's own runtime endpoint. Claws MUST be provisioned against
|
||
// THIS gateway, not the global one — see RuntimeProvisioner::for_gateway.
|
||
let mut mission_gateway: Option<String> = None;
|
||
// Not for a microVM mission: the ZeroClaw daemon it would start is never
|
||
// spoken to, and it would sit holding a pairing code and ~3 GB of image for
|
||
// the life of the mission. Observed doing exactly that on the first real run.
|
||
if let Some(prov) = crate::mission_runtime::MissionRuntimeProvisioner::from_env()
|
||
.filter(|_| mission.runtime_kind != "microvm")
|
||
{
|
||
match prov.ensure_container(mission_id).await {
|
||
Ok(ec) => {
|
||
mission_gateway = Some(ec.endpoint.clone());
|
||
let container_name = crate::mission_runtime::container_name(mission_id);
|
||
if let Err(e) = cm_db::repo::missions::set_runtime_binding(
|
||
pool,
|
||
mission_id,
|
||
workspace_id.as_uuid(),
|
||
Some(&container_name),
|
||
Some(&ec.endpoint),
|
||
ec.pairing_code.as_deref(),
|
||
)
|
||
.await
|
||
{
|
||
eprintln!(
|
||
"mission_orchestrator: bind runtime container for {mission_id} failed: {e}"
|
||
);
|
||
} else {
|
||
eprintln!(
|
||
"mission_orchestrator: runtime container {container_name} → {} (paired={}) for mission {mission_id}",
|
||
ec.endpoint,
|
||
ec.pairing_code.is_some()
|
||
);
|
||
}
|
||
}
|
||
Err(e) => eprintln!(
|
||
"mission_orchestrator: provision runtime container for {mission_id} failed (continuing with shared runtime): {e}"
|
||
),
|
||
}
|
||
} else {
|
||
eprintln!(
|
||
"mission_orchestrator: docker unreachable, mission {mission_id} will use shared runtime"
|
||
);
|
||
}
|
||
|
||
// microVM PLACEMENT MUST COME BEFORE the early return below. It did not, and
|
||
// the first real microvm mission failed with "mission has no target_node_id" —
|
||
// the executor's own guard firing correctly on a mission this function had
|
||
// returned from before ever choosing a node for it.
|
||
// microVM placement. KVM is a hard predicate, not a preference: gw-04 —
|
||
// where every mission runs today — is itself a VM without nested
|
||
// virtualisation and has no /dev/kvm, so a microvm mission landing there
|
||
// cannot start. Resolve a capable node now and fail the launch if there is
|
||
// none, because the alternative is a mission that sits in 'running' having
|
||
// never had anywhere to run.
|
||
if mission.runtime_kind == "microvm" {
|
||
// Capable means BOTH: it can host a microVM, and it holds the image this
|
||
// mission's backend names. Asking only for `microvm` sent the first real
|
||
// microVM mission to a node without `rootfs-claude.ext4`.
|
||
let backend = mission.backend.as_deref();
|
||
let capable =
|
||
cm_db::repo::nodes::online_for_backend(pool, mission.workspace_id, backend)
|
||
.await
|
||
.map_err(|e| format!("looking up nodes for backend {backend:?}: {e}"))?;
|
||
let how_to_fix = format!(
|
||
"needs /dev/kvm + firecracker (scripts/fc-node-setup.sh) AND the {} rootfs \
|
||
built on that node (scripts/fc-build-rootfs.sh <host> <image> {})",
|
||
backend.unwrap_or("default"),
|
||
backend.unwrap_or("<name>")
|
||
);
|
||
let how_to_fix = how_to_fix.as_str();
|
||
// CAPABILITY is checked here; CAPACITY is not, and no node is pinned.
|
||
//
|
||
// Placement moved to phase launch (`phase_runner`). A node chosen now
|
||
// would be chosen once, minutes before the first VM boots and hours
|
||
// before the last — and re-placing between phases is free, because
|
||
// mission state lives on the gateway checkout and every VM is
|
||
// inject → run → collect → destroy. Pinning early bought nothing and
|
||
// cost the ability to react to a node filling or draining mid-mission.
|
||
//
|
||
// Launching still FAILS here when no node could ever run this backend:
|
||
// that is not transient, waiting will not fix it, and the harness's
|
||
// `microvm-negctl` scenario asserts such a mission stays `draft`.
|
||
if capable.is_empty() {
|
||
return Err(format!(
|
||
"no online node can run backend {:?} — {how_to_fix}",
|
||
backend.unwrap_or("default")
|
||
));
|
||
}
|
||
eprintln!(
|
||
"mission_orchestrator: mission {mission_id} has {} node(s) able to run \
|
||
backend {:?}; placement happens per phase",
|
||
capable.len(),
|
||
backend.unwrap_or("default")
|
||
);
|
||
}
|
||
|
||
// A microVM mission materialises no team. Its phases run as one `claude -p`
|
||
// inside a VM (`microvm_executor`), so there is no claw graph to provision —
|
||
// and demanding one rejected the launch of a well-formed mission with "pick
|
||
// teams in the wizard". This is the third of three team gates on a path that
|
||
// uses no teams; the other two are in `routes::missions` (draft→running) and
|
||
// `phase_runner::launch_phase` (no matching teams → stay pending).
|
||
//
|
||
// Returning before the picks below, not filtering them, because provisioning
|
||
// claws that never run is not a cheaper version of the same thing — it is a
|
||
// runtime binding and a pairing code describing something nothing uses.
|
||
if mission.runtime_kind == "microvm" {
|
||
eprintln!(
|
||
"mission_orchestrator: mission {mission_id} is a microvm mission — no team to \
|
||
materialise; its phases execute in a VM"
|
||
);
|
||
return Ok(None);
|
||
}
|
||
|
||
// Skip team materialization if already bound.
|
||
if mission.team_id.is_some() {
|
||
eprintln!(
|
||
"mission_orchestrator::on_launch team_id already bound for mission_id={mission_id} — skipping team materialization"
|
||
);
|
||
return Ok(mission.team_id);
|
||
}
|
||
|
||
// New multi-team model: config.phase_teams = {
|
||
// "research": ["template-uuid", ...],
|
||
// "coding": ["template-uuid", ...]
|
||
// }
|
||
// Mints one team per (phase-purpose, template) pair. The FIRST
|
||
// minted team gets bound to mission.team_id for backward-compat
|
||
// with the single-team surfaces (Team tab, legacy code).
|
||
//
|
||
// Fallback: if config.phase_teams is absent, use the legacy
|
||
// single team_template_id path so existing missions still work.
|
||
let phase_teams = mission
|
||
.config
|
||
.get("phase_teams")
|
||
.and_then(|v| v.as_object());
|
||
|
||
let picks: Vec<(String, Uuid)> = if let Some(pt) = phase_teams {
|
||
let mut out = Vec::new();
|
||
for (purpose, list) in pt.iter() {
|
||
if let Some(arr) = list.as_array() {
|
||
for item in arr {
|
||
if let Some(id_str) = item.as_str() {
|
||
if let Ok(id) = Uuid::parse_str(id_str) {
|
||
out.push((purpose.clone(), id));
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
out
|
||
} else if let Some(id) = mission.team_template_id {
|
||
vec![("mission".to_string(), id)]
|
||
} else {
|
||
return Err(
|
||
"mission has no team_template_id and no config.phase_teams — pick teams in the wizard"
|
||
.to_string(),
|
||
);
|
||
};
|
||
|
||
if picks.is_empty() {
|
||
return Err(
|
||
"mission's config.phase_teams is empty — pick at least one team in the wizard".into(),
|
||
);
|
||
}
|
||
|
||
// Provision into the mission's own daemon when we have one (so the daemon
|
||
// that actually runs the turns knows these claws); fall back to the global
|
||
// gateway only for dev/no-docker setups where the run uses it too.
|
||
let provisioner = match mission_gateway.clone() {
|
||
Some(url) => RuntimeProvisioner::for_gateway(url),
|
||
None => RuntimeProvisioner::from_env(),
|
||
};
|
||
let mut first_team_id: Option<Uuid> = None;
|
||
let mut provisioned_claws: Vec<cm_domain::AgentId> = Vec::new();
|
||
for (purpose, template_id) in &picks {
|
||
let template = cm_db::repo::team_templates::get(pool, *template_id)
|
||
.await
|
||
.map_err(|e| format!("load template {template_id}: {e}"))?
|
||
.ok_or_else(|| format!("template {template_id} not found"))?;
|
||
let team_name = format!(
|
||
"{} · {} · {}",
|
||
mission.title, purpose, template.template.name
|
||
);
|
||
let team_id = mint_team_from_template(
|
||
TeamMint {
|
||
pool,
|
||
workspace_id,
|
||
user_id,
|
||
provisioner: provisioner.as_ref(),
|
||
template: &template,
|
||
team_name: &team_name,
|
||
default_model: MINTED_CLAW_MODEL,
|
||
},
|
||
&mut provisioned_claws,
|
||
)
|
||
.await?;
|
||
// Record (mission, team, purpose) in mission_teams so the Team
|
||
// tab can group by phase purpose without parsing team names.
|
||
sqlx::query("INSERT INTO mission_teams (mission_id, team_id, purpose) VALUES ($1, $2, $3)")
|
||
.bind(mission_id)
|
||
.bind(team_id)
|
||
.bind(purpose)
|
||
.execute(pool)
|
||
.await
|
||
.map_err(|e| format!("record mission_team {team_id}: {e}"))?;
|
||
if first_team_id.is_none() {
|
||
first_team_id = Some(team_id);
|
||
}
|
||
}
|
||
let team_id = first_team_id.expect("picks non-empty guaranteed above");
|
||
|
||
// Bind the first team onto the mission for legacy single-team paths.
|
||
sqlx::query("UPDATE missions SET team_id = $1, updated_at = now() WHERE id = $2")
|
||
.bind(team_id)
|
||
.bind(mission_id)
|
||
.execute(pool)
|
||
.await
|
||
.map_err(|e| format!("bind team on mission: {e}"))?;
|
||
|
||
// Pin every provisioned claw's workspace to /mission/repo so
|
||
// file_edit / content_search / glob_search / git_operations operate
|
||
// on the mission's checked-out repo instead of the empty per-agent
|
||
// sandbox. This CANNOT go through the config prop API (workspace.path
|
||
// is a PathBuf the prop-schema won't expose — see provision_claw), so
|
||
// we patch the shared config file directly on the per-mission runtime
|
||
// container. The daemon picks it up on the same reload that surfaces
|
||
// the freshly-provisioned claws for the run.
|
||
//
|
||
// FATAL, deliberately. This was "non-fatal: agents still write (to the
|
||
// sandbox) but the committer can't find the changes in /mission/repo" —
|
||
// which is to say, the mission runs to completion and delivers nothing.
|
||
// Mission `019fcf62` did exactly that: the pin failed with `argument list
|
||
// too long`, one line of stderr scrolled past, and phase 0 reported
|
||
// `completed` with zero files, no commit error and no push error. A launch
|
||
// that cannot bind its agents to the repo has no path to delivering work,
|
||
// so it must fail at launch where someone is still looking.
|
||
//
|
||
// Not for a microVM mission: its agent is a `claude -p` inside a VM on a
|
||
// fleet node, not a ZeroClaw claw in a container here, so there is no
|
||
// workspace to pin. Leaving it would make a microVM launch FAIL on a
|
||
// container it was never going to use.
|
||
if !provisioned_claws.is_empty() && mission_gateway.is_some() && mission.runtime_kind != "microvm"
|
||
{
|
||
if let Some(mp) = crate::mission_runtime::MissionRuntimeProvisioner::from_env() {
|
||
mp.pin_agent_workspaces(mission_id, &provisioned_claws, "/mission/repo")
|
||
.await
|
||
.map_err(|e| {
|
||
format!(
|
||
"could not pin agent workspaces to /mission/repo ({e}) — the mission \
|
||
would run with its agents writing to their sandboxes, delivering nothing"
|
||
)
|
||
})?;
|
||
// The daemon reads config ONCE at boot and never re-reads the
|
||
// file, so the pin is invisible until it restarts. Its agents were
|
||
// created through its own config API, so they are already
|
||
// persisted to the file and survive the restart; the pairing code
|
||
// is re-minted on every launch. Equally fatal: an unrestarted
|
||
// daemon is an unpinned daemon.
|
||
mp.restart_container(mission_id).await.map_err(|e| {
|
||
format!("could not restart the runtime to apply the workspace pin: {e}")
|
||
})?;
|
||
}
|
||
}
|
||
|
||
// Herdr second-runtime: if runtime_kind='local_herdr', spawn a
|
||
// pane on target_node running the first available local CLI.
|
||
// Non-fatal on failure — the operator sees the error in server
|
||
// logs and can manually retry via POST /herdr-dispatch.
|
||
if mission.runtime_kind == "local_herdr" {
|
||
if let (Some(hub), Some(node_id)) = (node_hub, mission.target_node_id) {
|
||
let prompt = mission.description.clone().unwrap_or_default();
|
||
// CLI selection: mission.config.cli overrides; else default.
|
||
// (Per-template default_cli fallback was in the single-team
|
||
// path; the multi-team path doesn't have one canonical
|
||
// template to consult, so we keep the mission-level knob.)
|
||
let cli = mission
|
||
.config
|
||
.get("cli")
|
||
.and_then(|v| v.as_str())
|
||
.map(str::to_string)
|
||
.unwrap_or_else(|| "claude".to_string());
|
||
match crate::fleet_herdr::dispatch(
|
||
hub,
|
||
cm_domain::NodeId::from(node_id),
|
||
mission_id,
|
||
&cli,
|
||
&prompt,
|
||
)
|
||
.await
|
||
{
|
||
Ok(handle) => eprintln!(
|
||
"mission_orchestrator: herdr pane {} spawned on node {}",
|
||
handle.pane_id, node_id
|
||
),
|
||
Err(e) => eprintln!(
|
||
"mission_orchestrator: herdr dispatch for {mission_id} failed (continuing): {e}"
|
||
),
|
||
}
|
||
} else {
|
||
eprintln!(
|
||
"mission_orchestrator: mission {mission_id} is local_herdr but node_hub or target_node missing"
|
||
);
|
||
}
|
||
}
|
||
|
||
Ok(Some(team_id))
|
||
}
|
||
|
||
/// The read-only inputs for minting a team. Grouped into a struct so the
|
||
/// signature stays readable as the orchestrator accumulates context — the
|
||
/// growing positional list was also easy to mis-order at the call site,
|
||
/// since `team_name` and `default_model` are both `&str`.
|
||
struct TeamMint<'a> {
|
||
pool: &'a PgPool,
|
||
workspace_id: WorkspaceId,
|
||
user_id: cm_domain::UserId,
|
||
provisioner: Option<&'a RuntimeProvisioner>,
|
||
template: &'a TeamTemplateDetail,
|
||
team_name: &'a str,
|
||
default_model: &'a str,
|
||
}
|
||
|
||
async fn mint_team_from_template(
|
||
mint: TeamMint<'_>,
|
||
provisioned_claws: &mut Vec<cm_domain::AgentId>,
|
||
) -> Result<Uuid, String> {
|
||
let TeamMint {
|
||
pool,
|
||
workspace_id,
|
||
user_id,
|
||
provisioner,
|
||
template,
|
||
team_name,
|
||
default_model,
|
||
} = mint;
|
||
// Build the topology graph from role slots so the team's `graph`
|
||
// NOT NULL column is satisfied + downstream topology executors
|
||
// have a valid shape to iterate over.
|
||
let roles: Vec<&str> = template.roles.iter().map(|r| r.slot.as_str()).collect();
|
||
let topology_kind = parse_topology_kind(&template.template.default_topology);
|
||
let graph = cm_topology::build(topology_kind, &roles)
|
||
.map_err(|e| format!("build topology graph: {e}"))?;
|
||
let graph_json =
|
||
serde_json::to_value(&graph).map_err(|e| format!("serialize topology graph: {e}"))?;
|
||
|
||
let team_id = Uuid::now_v7();
|
||
cm_db::repo::teams::insert_team_with_lifecycle(
|
||
pool,
|
||
team_id,
|
||
workspace_id,
|
||
team_name,
|
||
&template.template.default_topology,
|
||
&graph_json,
|
||
"permanent",
|
||
)
|
||
.await
|
||
.map_err(|e| format!("insert team: {e}"))?;
|
||
|
||
// Stamp template lineage on the team row (Slice 3 columns).
|
||
sqlx::query(
|
||
"UPDATE teams SET template_id = $1, template_version = $2,
|
||
risk_profile = $3, mcp_bundles = $4
|
||
WHERE id = $5",
|
||
)
|
||
.bind(template.template.id)
|
||
.bind(template.template.version)
|
||
.bind(&template.template.risk_profile)
|
||
.bind(serde_json::json!(template.template.mcp_bundles))
|
||
.bind(team_id)
|
||
.execute(pool)
|
||
.await
|
||
.map_err(|e| format!("stamp template lineage: {e}"))?;
|
||
|
||
// Names already on this workspace's roster, so a newly hired claw does not
|
||
// arrive sharing a name with someone already here. Read ONCE — a roster
|
||
// query per role would be N queries to answer one question — and extended
|
||
// locally as we mint, which also keeps names distinct WITHIN this team.
|
||
let mut taken_names: Vec<String> = cm_db::repo::agents::roster(pool, workspace_id)
|
||
.await
|
||
.map_err(|e| format!("read roster for naming: {e}"))?
|
||
.into_iter()
|
||
.map(|a| a.name)
|
||
.collect();
|
||
|
||
// For each role: create agent, provision runtime, ingest brain
|
||
// seed, record link, bind to topology node.
|
||
for (idx, role) in template.roles.iter().enumerate() {
|
||
// Node id in the graph follows cm_topology's `n0..` convention
|
||
// — read the actual node id from the graph so team_members
|
||
// stays consistent with the topology.
|
||
let Some(node) = graph.nodes.get(idx) else {
|
||
return Err(format!(
|
||
"topology graph produced {} nodes but template has {} roles",
|
||
graph.nodes.len(),
|
||
template.roles.len(),
|
||
));
|
||
};
|
||
// Every mission gets its OWN crew.
|
||
//
|
||
// This deliberately reverses the reuse added earlier. Reuse hired the
|
||
// existing claw for a (template, slot) so the roster stayed at one team
|
||
// and "My Workforce" was people you keep — but it also meant every
|
||
// mission was staffed by the same five names, and the workforce view
|
||
// showed one crew repeated down the page with nothing to tell the
|
||
// missions apart. Chosen by the operator: distinct crews read better
|
||
// than a bounded roster.
|
||
//
|
||
// The cost is real and is the cost that reuse existed to avoid: claws
|
||
// are `lifecycle = 'permanent'` and nothing reaps them until their
|
||
// MISSION is deleted, so the roster now grows by the team size on every
|
||
// mission. `agent_names::pick` keeps names unique workspace-wide and
|
||
// falls back to a numeric suffix once the pool is exhausted, so growth
|
||
// degrades the naming gracefully rather than colliding.
|
||
//
|
||
// `reusable_claw` in cm-db is kept, with its tests: this is a policy
|
||
// choice that has now flipped twice, and the query is the hard part.
|
||
let reused: Option<uuid::Uuid> = None;
|
||
|
||
// Seed the name choice from the claw's OWN id, not its position in the
|
||
// team.
|
||
//
|
||
// Seeding with the role index (0..n) started every crew near the top of
|
||
// the pool and took the next free names, so the first mission hired
|
||
// Aarav, Abebe, Adaora, Adrian, Agnieszka — correct, unique, and
|
||
// transparently alphabetical. A crew should look like a team, not like
|
||
// a listing. UUIDv7 puts its random bytes LAST (the leading bytes are a
|
||
// timestamp, which would cluster again), so the tail is what spreads
|
||
// the five picks across the whole pool.
|
||
let agent_id = cm_domain::AgentId::new();
|
||
let name_seed = {
|
||
let uuid = agent_id.as_uuid();
|
||
let b = uuid.as_bytes();
|
||
u64::from_le_bytes([b[8], b[9], b[10], b[11], b[12], b[13], b[14], b[15]])
|
||
};
|
||
|
||
let agent = Agent {
|
||
id: agent_id,
|
||
workspace_id,
|
||
// A PERSON's name, with the role in `job_title`.
|
||
//
|
||
// This was `"{mission title} · {purpose} · {template} · {slot}"` —
|
||
// names like "verify: a repo-less research mission keeps its output
|
||
// · mission · Rust SDLC · planner", unreadable in the roster, the
|
||
// API and every log line at once. Then it was the bare slot, which
|
||
// fixed the length but made the UI show the same word twice (name
|
||
// on top, role beneath) and made a roster of five read as five job
|
||
// tickets rather than a crew.
|
||
//
|
||
// The role still lives in `job_title`, which is what the mission
|
||
// machinery binds on — `team_members.role_slot` and the topology
|
||
// node carry the slot, so nothing downstream keys off the display
|
||
// name. Only the reused branch below ignores this, deliberately: a
|
||
// claw you already hired keeps the name it already had.
|
||
name: crate::agent_names::pick(&taken_names, name_seed),
|
||
job_title: role.slot.clone(),
|
||
// This is the ONLY consumer of the templates' `system_prompt` prose,
|
||
// and it feeds the *chat* path, not missions: it lands in
|
||
// `agents.system_prompt`, which `cm_runtime::brain::compose_system`
|
||
// uses as the base prompt for a claw's chat turns. A mission turn
|
||
// never sees it — `topology_exec::build_prompt` synthesizes its own
|
||
// one-line system text from the role slot alone. So deleting the
|
||
// template prose to save mission tokens would save exactly zero and
|
||
// would leave every mission-minted claw with no identity in chat.
|
||
system_prompt: role.system_prompt.clone(),
|
||
avatar: String::new(),
|
||
accent: default_accent_for(&role.slot).to_string(),
|
||
wallpaper: String::new(),
|
||
managed_by: user_id,
|
||
status: AgentStatus::Online,
|
||
};
|
||
let claw_id = match reused {
|
||
Some(existing) => {
|
||
eprintln!(
|
||
"mission_orchestrator: reusing claw {existing} for role {} \
|
||
(template {})",
|
||
role.slot, template.template.id
|
||
);
|
||
existing
|
||
}
|
||
None => {
|
||
cm_db::repo::agents::insert(pool, &agent, &AccessPolicy::default())
|
||
.await
|
||
.map_err(|e| format!("insert agent {}: {e}", role.slot))?;
|
||
// Claim the name for the rest of this loop. Without this the
|
||
// roster snapshot taken before the loop is stale from the
|
||
// second role onward and a five-person team can arrive with
|
||
// two Merediths.
|
||
taken_names.push(agent.name.clone());
|
||
agent.id.as_uuid()
|
||
}
|
||
};
|
||
let agent_id = cm_domain::AgentId::from(claw_id);
|
||
|
||
// The ROLE's model when the template names one, else the mint's default.
|
||
// Before migration 0071 there was no role model at all, so every claw of
|
||
// every mission team ran the same one — including a reviewer reviewing
|
||
// the coder it shares a model with.
|
||
let role_model = role
|
||
.model
|
||
.as_deref()
|
||
.map(str::trim)
|
||
.filter(|m| !m.is_empty())
|
||
.unwrap_or(default_model);
|
||
cm_db::repo::agents::set_model_binding(pool, agent_id, role_model)
|
||
.await
|
||
.map_err(|e| format!("set_model_binding {claw_id}: {e}"))?;
|
||
|
||
// Runtime provisioning is opt-in — no-op if unconfigured.
|
||
// Pass the team template's risk_profile so the claw actually
|
||
// gets the tools its role expects (research_readonly for
|
||
// scout/researcher, coding_readwrite for coder/tester/committer,
|
||
// etc.). Passing "toolfree" — the old default — left every
|
||
// agent with zero tools regardless of what its prompt asked for.
|
||
//
|
||
// Workspace pinning to /mission/repo is NOT done here (the
|
||
// config prop-schema can't set workspace.path — see
|
||
// provision_claw's doc); the caller pins the collected claws
|
||
// out-of-band via MissionRuntimeProvisioner::pin_agent_workspaces.
|
||
if let Some(p) = provisioner {
|
||
match p
|
||
.provision_claw(claw_id, role_model, &template.template.risk_profile)
|
||
.await
|
||
{
|
||
Ok(_) => provisioned_claws.push(agent_id),
|
||
Err(e) => eprintln!(
|
||
"mission_orchestrator: provision claw {claw_id} failed (continuing): {e}"
|
||
),
|
||
}
|
||
}
|
||
|
||
// Ingest brain seed (Slice 3.5d). Non-fatal on failure —
|
||
// agent still works from system_prompt alone.
|
||
// Seed only a NEW claw. A reused one carries what it learned on earlier
|
||
// missions, and re-seeding would overwrite that with the template's
|
||
// starting point — which is precisely the accumulation reuse exists for.
|
||
if reused.is_none() {
|
||
if let Some(seed) = role.brain_seed.as_deref().filter(|s| !s.trim().is_empty()) {
|
||
if let Err(e) =
|
||
crate::brain_seed::ingest(claw_id, seed.to_string(), role.system_prompt.clone())
|
||
.await
|
||
{
|
||
eprintln!(
|
||
"mission_orchestrator: brain_seed ingest for {claw_id} failed (continuing): {e}"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
// Record lineage (Slice 3.5d) so the MCP skills server can
|
||
// merge template default skills with per-agent overrides.
|
||
cm_db::repo::agent_template_link::upsert(
|
||
pool,
|
||
claw_id,
|
||
template.template.id,
|
||
template.template.version,
|
||
&role.slot,
|
||
)
|
||
.await
|
||
.map_err(|e| format!("agent_template_link upsert {claw_id}: {e}"))?;
|
||
cm_db::repo::agent_template_link::mark_seeded(pool, claw_id)
|
||
.await
|
||
.map_err(|e| format!("mark_seeded {claw_id}: {e}"))?;
|
||
|
||
// Wire team_members using the topology node id.
|
||
cm_db::repo::teams::add_member(pool, team_id, &node.id, claw_id, &role.slot)
|
||
.await
|
||
.map_err(|e| format!("team_members add {claw_id}: {e}"))?;
|
||
|
||
// Audit for parity with individual claw creation path.
|
||
let _ = cm_db::repo::audit::append(
|
||
pool,
|
||
workspace_id,
|
||
cm_db::repo::audit::Actor::User(user_id),
|
||
"agent.created",
|
||
"agent",
|
||
&agent_id.to_string(),
|
||
serde_json::json!({
|
||
"name": agent.name,
|
||
"reused": reused.is_some(),
|
||
"job_title": agent.job_title,
|
||
"source": "mission_orchestrator",
|
||
"template_id": template.template.id.to_string(),
|
||
"template_version": template.template.version,
|
||
"role_slot": role.slot,
|
||
}),
|
||
)
|
||
.await;
|
||
}
|
||
|
||
Ok(team_id)
|
||
}
|
||
|
||
/// The model a minted claw runs on when its template role does not name one.
|
||
///
|
||
/// A DEFAULT now, not a hardcode: `template_roles.model` (migration 0071) lets a
|
||
/// template put its reviewer on a different model from the coder it reviews,
|
||
/// which is the correlated failure the cross-provider judge exists to break,
|
||
/// one layer down. Roles that say nothing still land here, so every template
|
||
/// that existed before 0071 behaves exactly as it did.
|
||
const MINTED_CLAW_MODEL: &str = "claude-sonnet-5";
|
||
|
||
/// The graph a COMPOSED microVM mission runs, built from its team template
|
||
/// without minting a single claw.
|
||
///
|
||
/// A composed mission needs the template's *shape* — how many nodes, in what
|
||
/// pattern, playing what roles — and nothing else it carries. Its nodes are VMs,
|
||
/// so provisioning claws for them would create agents, containers and `.brain`
|
||
/// files that nothing ever dials; that is exactly why `on_launch` returns early
|
||
/// for a microVM mission, and this is how the composed path gets its graph
|
||
/// anyway rather than by undoing that.
|
||
///
|
||
/// `purposes` is the phase's purpose list, matched against `config.phase_teams`;
|
||
/// missions using the legacy single `team_template_id` fall back to it.
|
||
/// Returns `None` when the mission picked no template at all.
|
||
pub async fn composed_graph(
|
||
pool: &PgPool,
|
||
mission_id: Uuid,
|
||
purposes: &[&str],
|
||
) -> Result<Option<serde_json::Value>, String> {
|
||
let row: Option<(serde_json::Value, Option<Uuid>)> =
|
||
sqlx::query_as("SELECT config, team_template_id FROM missions WHERE id = $1")
|
||
.bind(mission_id)
|
||
.fetch_optional(pool)
|
||
.await
|
||
.map_err(|e| format!("load mission {mission_id}: {e}"))?;
|
||
let Some((config, legacy_template)) = row else {
|
||
return Err(format!("mission {mission_id} not found"));
|
||
};
|
||
|
||
// An APPROVED roster wins over the template. It is the more specific answer
|
||
// — a model sized it for this mission's actual task and a human accepted it
|
||
// — and it is the only path on which nodes carry per-node backends, which is
|
||
// how a mission runs more than one provider. Stored already built and
|
||
// validated (`routes::mission_roster::decide`), so nothing here can turn a
|
||
// refused roster into a running one.
|
||
if let Some(roster) = config.get("roster").filter(|v| v.is_object()) {
|
||
// Parsed rather than trusted: a graph the orchestrator cannot plan would
|
||
// otherwise be claimed and fail as "missing or invalid graph", which
|
||
// reads as a runtime fault instead of a bad roster.
|
||
serde_json::from_value::<cm_topology::TopologyGraph>(roster.clone())
|
||
.map_err(|e| format!("mission {mission_id}: the approved roster is not a runnable topology: {e}"))?;
|
||
return Ok(Some(roster.clone()));
|
||
}
|
||
|
||
let template_id = config
|
||
.get("phase_teams")
|
||
.and_then(|v| v.as_object())
|
||
.and_then(|pt| {
|
||
// First template named by any purpose this phase answers to, in the
|
||
// phase's own preference order — the same order `launch_phase` uses
|
||
// to pick teams, so a composed mission and a ZeroClaw one resolve the
|
||
// same template for the same phase.
|
||
purposes.iter().find_map(|p| {
|
||
pt.get(*p)
|
||
.and_then(|v| v.as_array())
|
||
.and_then(|a| a.first())
|
||
.and_then(|v| v.as_str())
|
||
.and_then(|s| Uuid::parse_str(s).ok())
|
||
})
|
||
})
|
||
.or(legacy_template);
|
||
let Some(template_id) = template_id else {
|
||
return Ok(None);
|
||
};
|
||
|
||
let template = cm_db::repo::team_templates::get(pool, template_id)
|
||
.await
|
||
.map_err(|e| format!("load template {template_id}: {e}"))?
|
||
.ok_or_else(|| format!("template {template_id} not found"))?;
|
||
let roles: Vec<&str> = template.roles.iter().map(|r| r.slot.as_str()).collect();
|
||
if roles.is_empty() {
|
||
return Err(format!("template {template_id} defines no roles"));
|
||
}
|
||
let graph = cm_topology::build(
|
||
parse_topology_kind(&template.template.default_topology),
|
||
&roles,
|
||
)
|
||
.map_err(|e| format!("build topology graph for template {template_id}: {e}"))?;
|
||
serde_json::to_value(&graph)
|
||
.map(Some)
|
||
.map_err(|e| format!("serialize topology graph: {e}"))
|
||
}
|
||
|
||
fn parse_topology_kind(s: &str) -> cm_topology::TopologyKind {
|
||
use cm_topology::TopologyKind;
|
||
match s {
|
||
"pipeline" => TopologyKind::Pipeline,
|
||
"hierarchical" => TopologyKind::Hierarchical,
|
||
"star_moe" => TopologyKind::StarMoe,
|
||
_ => TopologyKind::HubSpoke,
|
||
}
|
||
}
|
||
|
||
fn default_accent_for(slot: &str) -> &'static str {
|
||
match slot {
|
||
"planner" | "arch_analyst" | "designer" | "scene_designer" | "api_designer" => "#7cd6e0",
|
||
"coder" | "kernel_author" | "shader_author" => "#5fd08a",
|
||
"tester" | "bench_engineer" | "perf_engineer" => "#ffb44a",
|
||
"reviewer" | "db_engineer" => "#c98af0",
|
||
"committer" | "spec_integrator" => "#ff8a7a",
|
||
_ => "#8a8a92",
|
||
}
|
||
}
|