Survey + fixes so the pipeline passes at the Docker level (no k8s).
- Remove k8s: drop the `sandbox-k8s` job (kind/Calico/--features k8s-tests) and the
"Helm chart lints" gate step. release.yml was already k8s-clean.
- Rust job:
- `cargo fmt --all` — fix pre-existing formatting drift (fmt --check was failing).
- clippy -D warnings: fix 3 lib warnings (cm-brain sort_by_key→Reverse, cm-api
fleet.rs doc list indentation, node_rules map_or→is_none_or).
- Regenerate the .sqlx offline cache (was missing the cm-runtime run_loop test
query → offline compile failed). DB-backed tests use testcontainers at runtime.
- Set SQLX_OFFLINE=true on the rust + e2e jobs so query! macros compile against
the committed cache deterministically (no DB needed at compile time).
- Frontend job:
- Fix the 1 ESLint error (useAgentTelemetry: no setState-synchronously-in-effect;
tag the slice with agentId + derive null on mismatch).
- Fix 2 stale panel-params tests (`terminal` is a valid app id now; assert the
current APP_IDS + use a genuinely-unknown id for the reject case).
Verified locally: fmt clean, clippy --all-targets -D warnings clean (offline),
frontend lint 0 errors, tsc clean, 86/86 frontend tests pass, build OK.
Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
233 lines
7.1 KiB
Rust
233 lines
7.1 KiB
Rust
//! Agents execute code in their REAL hardened sandbox: shell.exec runs in
|
|
//! a per-agent container (Docker driver), output returns to the model as a
|
|
//! step result, and /home/agent state persists across calls in a run.
|
|
|
|
use std::process::Command;
|
|
use std::sync::Arc;
|
|
|
|
use cm_domain::{
|
|
AccessPolicy, Agent, AgentId, AgentStatus, Role, RunState, User, UserId, Workspace, WorkspaceId,
|
|
};
|
|
use cm_llm::ScriptedProvider;
|
|
use cm_runtime::{Runtime, RuntimeConfig, SandboxManager};
|
|
use cm_sandbox::DockerDriver;
|
|
|
|
const IMAGE: &str = "clawmates/agent-base:dev";
|
|
|
|
const SCENARIOS: &str = r#"
|
|
[[scenario]]
|
|
marker = "[[scenario:shell]]"
|
|
|
|
[[scenario.turns]]
|
|
events = [
|
|
{ type = "tool_use", name = "shell.exec", input = { command = "id -u && echo tick > /home/agent/state" } },
|
|
]
|
|
|
|
[[scenario.turns]]
|
|
events = [
|
|
{ type = "tool_use", name = "shell.exec", input = { command = "cat /home/agent/state" } },
|
|
]
|
|
|
|
[[scenario.turns]]
|
|
events = [
|
|
{ type = "text", text = "Ran both commands." },
|
|
]
|
|
"#;
|
|
|
|
fn ensure_image() {
|
|
let exists = Command::new("docker")
|
|
.args(["image", "inspect", IMAGE])
|
|
.output()
|
|
.expect("docker available")
|
|
.status
|
|
.success();
|
|
if !exists {
|
|
let root = env!("CARGO_MANIFEST_DIR");
|
|
let status = Command::new("docker")
|
|
.args([
|
|
"build",
|
|
"-t",
|
|
IMAGE,
|
|
"-f",
|
|
&format!("{root}/../../images/agent-base/Dockerfile"),
|
|
&format!("{root}/../../images/agent-base"),
|
|
])
|
|
.status()
|
|
.expect("docker build runs");
|
|
assert!(status.success(), "agent-base image build failed");
|
|
}
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn shell_exec_runs_in_the_agent_sandbox_with_persistent_home() {
|
|
ensure_image();
|
|
let pool = cm_testkit::test_pool().await;
|
|
|
|
let ws = Workspace {
|
|
id: WorkspaceId::new(),
|
|
name: "Acme".into(),
|
|
plan: "team".into(),
|
|
};
|
|
cm_db::repo::workspaces::insert(&pool, &ws).await.unwrap();
|
|
let owner = User {
|
|
id: UserId::new(),
|
|
workspace_id: ws.id,
|
|
email: format!("{}@acme.test", UserId::new()),
|
|
role: Role::Owner,
|
|
display_name: "Owner".into(),
|
|
created_at: time::OffsetDateTime::UNIX_EPOCH,
|
|
};
|
|
cm_db::repo::users::insert(&pool, &owner).await.unwrap();
|
|
let agent = Agent {
|
|
id: AgentId::new(),
|
|
workspace_id: ws.id,
|
|
name: "Scout".into(),
|
|
job_title: "Analyst".into(),
|
|
system_prompt: String::new(),
|
|
avatar: String::new(),
|
|
accent: String::new(),
|
|
wallpaper: String::new(),
|
|
managed_by: owner.id,
|
|
status: AgentStatus::Online,
|
|
};
|
|
cm_db::repo::agents::insert(&pool, &agent, &AccessPolicy::default())
|
|
.await
|
|
.unwrap();
|
|
|
|
let driver = DockerDriver::connect().expect("docker reachable");
|
|
let sandboxes = Arc::new(SandboxManager::new(
|
|
Arc::new(driver),
|
|
pool.clone(),
|
|
"local",
|
|
IMAGE,
|
|
));
|
|
let rt = Runtime::new(
|
|
pool.clone(),
|
|
Arc::new(ScriptedProvider::from_toml(SCENARIOS).unwrap()),
|
|
RuntimeConfig::basic("scripted", 1024).with_sandboxes(sandboxes.clone()),
|
|
);
|
|
|
|
let session = cm_db::repo::sessions::create(&pool, agent.id, ws.id, "Shell")
|
|
.await
|
|
.unwrap();
|
|
let started = rt
|
|
.send_message(session.id, "run it [[scenario:shell]]")
|
|
.await
|
|
.unwrap();
|
|
let mut rx = started.events;
|
|
while let Ok(envelope) = rx.recv().await {
|
|
if matches!(
|
|
envelope.event,
|
|
cm_runtime::RunEventBody::RunCompleted { .. } | cm_runtime::RunEventBody::Error { .. }
|
|
) {
|
|
break;
|
|
}
|
|
}
|
|
let run = cm_db::repo::runs::get(&pool, started.run_id).await.unwrap();
|
|
assert_eq!(run.state, RunState::Completed);
|
|
|
|
// Step outputs prove kernel-level facts: uid 10001 inside, and the
|
|
// second call saw the first call's file — same sandbox, same run.
|
|
let outputs: Vec<serde_json::Value> = sqlx::query_scalar(
|
|
"SELECT s.output FROM steps s
|
|
JOIN messages m ON m.id = s.message_id
|
|
WHERE m.session_id = $1 AND s.tool_name = 'shell.exec'
|
|
ORDER BY s.seq",
|
|
)
|
|
.bind(session.id.as_uuid())
|
|
.fetch_all(&pool)
|
|
.await
|
|
.unwrap();
|
|
assert_eq!(outputs.len(), 2);
|
|
assert_eq!(outputs[0]["exit_code"], 0, "first: {}", outputs[0]);
|
|
assert!(
|
|
outputs[0]["stdout"].as_str().unwrap().contains("10001"),
|
|
"sandbox runs as uid 10001: {}",
|
|
outputs[0]
|
|
);
|
|
assert_eq!(outputs[1]["exit_code"], 0, "second: {}", outputs[1]);
|
|
assert!(
|
|
outputs[1]["stdout"].as_str().unwrap().contains("tick"),
|
|
"home persists across execs: {}",
|
|
outputs[1]
|
|
);
|
|
|
|
sandboxes.shutdown().await;
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn shell_exec_without_a_sandbox_runtime_reports_a_tool_error() {
|
|
let pool = cm_testkit::test_pool().await;
|
|
let ws = Workspace {
|
|
id: WorkspaceId::new(),
|
|
name: "Acme".into(),
|
|
plan: "team".into(),
|
|
};
|
|
cm_db::repo::workspaces::insert(&pool, &ws).await.unwrap();
|
|
let owner = User {
|
|
id: UserId::new(),
|
|
workspace_id: ws.id,
|
|
email: format!("{}@acme.test", UserId::new()),
|
|
role: Role::Owner,
|
|
display_name: "Owner".into(),
|
|
created_at: time::OffsetDateTime::UNIX_EPOCH,
|
|
};
|
|
cm_db::repo::users::insert(&pool, &owner).await.unwrap();
|
|
let agent = Agent {
|
|
id: AgentId::new(),
|
|
workspace_id: ws.id,
|
|
name: "Scout".into(),
|
|
job_title: "Analyst".into(),
|
|
system_prompt: String::new(),
|
|
avatar: String::new(),
|
|
accent: String::new(),
|
|
wallpaper: String::new(),
|
|
managed_by: owner.id,
|
|
status: AgentStatus::Online,
|
|
};
|
|
cm_db::repo::agents::insert(&pool, &agent, &AccessPolicy::default())
|
|
.await
|
|
.unwrap();
|
|
|
|
let rt = Runtime::new(
|
|
pool.clone(),
|
|
Arc::new(ScriptedProvider::from_toml(SCENARIOS).unwrap()),
|
|
RuntimeConfig::basic("scripted", 1024),
|
|
);
|
|
let session = cm_db::repo::sessions::create(&pool, agent.id, ws.id, "NoBox")
|
|
.await
|
|
.unwrap();
|
|
let started = rt
|
|
.send_message(session.id, "run it [[scenario:shell]]")
|
|
.await
|
|
.unwrap();
|
|
let mut rx = started.events;
|
|
while let Ok(envelope) = rx.recv().await {
|
|
if matches!(
|
|
envelope.event,
|
|
cm_runtime::RunEventBody::RunCompleted { .. } | cm_runtime::RunEventBody::Error { .. }
|
|
) {
|
|
break;
|
|
}
|
|
}
|
|
// The run survives: the tool reports its error to the model, which
|
|
// finishes the scenario.
|
|
let run = cm_db::repo::runs::get(&pool, started.run_id).await.unwrap();
|
|
assert_eq!(run.state, RunState::Completed);
|
|
let statuses: Vec<String> = sqlx::query_scalar(
|
|
"SELECT s.status FROM steps s
|
|
JOIN messages m ON m.id = s.message_id
|
|
WHERE m.session_id = $1 AND s.tool_name = 'shell.exec'
|
|
ORDER BY s.seq",
|
|
)
|
|
.bind(session.id.as_uuid())
|
|
.fetch_all(&pool)
|
|
.await
|
|
.unwrap();
|
|
assert!(!statuses.is_empty());
|
|
assert!(
|
|
statuses.iter().all(|s| s == "error"),
|
|
"steps must record the failure: {statuses:?}"
|
|
);
|
|
}
|