Compare commits
333
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
650a556029 | ||
|
|
a0f99914af | ||
|
|
fa650bffe4 | ||
|
|
2069bdf322 | ||
|
|
3909fa14ca | ||
|
|
13f7fb3aff | ||
|
|
76ac3714f1 | ||
|
|
9b7680a605 | ||
|
|
2852eb8835 | ||
|
|
1fc6cb41ba | ||
|
|
ef1f21024d | ||
|
|
1a44405308 | ||
|
|
5d9edd636d | ||
|
|
8e4b2e4a12 | ||
|
|
9fc904a056 | ||
|
|
507d7444d1 | ||
|
|
a50c41a9c2 | ||
|
|
4e342d1ff7 | ||
|
|
2109bae6c6 | ||
|
|
9bc1e7fc52 | ||
|
|
cbc9c2d353 | ||
|
|
794f2124bc | ||
|
|
3755699b41 | ||
|
|
1c9be52291 | ||
|
|
a0f6cd7175 | ||
|
|
869c3adcb7 | ||
|
|
1678452a93 | ||
|
|
93a386e706 | ||
|
|
483de9f88a | ||
|
|
736b6a9a82 | ||
|
|
248948cc84 | ||
|
|
758760cedb | ||
|
|
7dd3aa0965 | ||
|
|
00160739de | ||
|
|
8d6310f126 | ||
|
|
1072964326 | ||
|
|
42c24de6a9 | ||
|
|
bda6bef4db | ||
|
|
0b9baa942f | ||
|
|
2a3409ec53 | ||
|
|
daf8d12157 | ||
|
|
f26de3ba76 | ||
|
|
2f1a870949 | ||
|
|
563b074116 | ||
|
|
fde1341618 | ||
|
|
bd7fd46305 | ||
|
|
fe5c7d2c87 | ||
|
|
5a2ed8fb42 | ||
|
|
7bcf7865f0 | ||
|
|
accae7fa94 | ||
|
|
22eeaa6f15 | ||
|
|
f52cff3e04 | ||
|
|
b58f0347e6 | ||
|
|
72eda8b3d2 | ||
|
|
7525be3791 | ||
|
|
5220f3bfea | ||
|
|
b47ae7fa6b | ||
|
|
4b160c5a1b | ||
|
|
42e014976d | ||
|
|
02d5f5a8c9 | ||
|
|
3aeee070b8 | ||
|
|
73f5d71c55 | ||
|
|
2668191e30 | ||
|
|
8591585e60 | ||
|
|
3f26dfeaca | ||
|
|
19c4de36e4 | ||
|
|
6af1149e45 | ||
|
|
ceec0423ad | ||
|
|
4f4ce34203 | ||
|
|
6f2b0a8f43 | ||
|
|
9560aaec41 | ||
|
|
c209e654d9 | ||
|
|
4a6d0dfe01 | ||
|
|
d0b657a24b | ||
|
|
1a6fdfc0e6 | ||
|
|
8cb38d1320 | ||
|
|
0b4d91889a | ||
|
|
5a11fae0d6 | ||
|
|
f6e6037aa0 | ||
|
|
e84413d437 | ||
|
|
cd59e4798d | ||
|
|
b89606fcf1 | ||
|
|
930c7e0b67 | ||
|
|
0be932fd83 | ||
|
|
afb1e29bf3 | ||
|
|
536adddd0f | ||
|
|
ac4fa0b8f7 | ||
|
|
689a5e14a3 | ||
|
|
8f988739ec | ||
|
|
d23f30e929 | ||
|
|
c02dbe2266 | ||
|
|
24393819bd | ||
|
|
a864f2ccc7 | ||
|
|
72ba4ba523 | ||
|
|
128b423205 | ||
|
|
b653dbfe72 | ||
|
|
547b5d9987 | ||
|
|
ea0b989b3f | ||
|
|
771092b165 | ||
|
|
113de610ec | ||
|
|
5c2c63f8e8 | ||
|
|
91a6b4e304 | ||
|
|
769e002bb3 | ||
|
|
e3247fee4b | ||
|
|
e4942ce985 | ||
|
|
18dc0b964b | ||
|
|
4358964c05 | ||
|
|
ba98c29481 | ||
|
|
fe45f72f09 | ||
|
|
f4adc8d0f9 | ||
|
|
1f39f642a3 | ||
|
|
55b16f25c8 | ||
|
|
656662850d | ||
|
|
850f11838b | ||
|
|
2cd0872e50 | ||
|
|
d524107b37 | ||
|
|
a02e0cba69 | ||
|
|
a2d7e3ea92 | ||
|
|
e20b321055 | ||
|
|
f87853ecf9 | ||
|
|
3cc65c22c4 | ||
|
|
9c4b0722e8 | ||
|
|
b5032a732a | ||
|
|
a582dea4fc | ||
|
|
d341640255 | ||
|
|
99dd29cc8a | ||
|
|
b31a79f650 | ||
|
|
69c294addc | ||
|
|
53da4d7e6d | ||
|
|
10341cf7fe | ||
|
|
fd5e71ccfe | ||
|
|
85a6038c08 | ||
|
|
6e8785f159 | ||
|
|
43436d7181 | ||
|
|
ba9d7aa185 | ||
|
|
bf40d10064 | ||
|
|
8be7b3c9b2 | ||
|
|
8ef7067467 | ||
|
|
b290025fc4 | ||
|
|
ccbc387f4b | ||
|
|
a494634f81 | ||
|
|
eb120a10dd | ||
|
|
4d07868410 | ||
|
|
25a3d6902a | ||
|
|
837a3d3ff0 | ||
|
|
875ff948f8 | ||
|
|
e7d2fc9696 | ||
|
|
41854c70e1 | ||
|
|
9441cf401c | ||
|
|
bd1c970577 | ||
|
|
c1642a7004 | ||
|
|
de8736c16b | ||
|
|
26e571fe01 | ||
|
|
548f977212 | ||
|
|
022ef98e44 | ||
|
|
7cf77a9248 | ||
|
|
5db695460f | ||
|
|
4dec77ae6d | ||
|
|
af89020dfd | ||
|
|
ee1cea72d9 | ||
|
|
8129f58845 | ||
|
|
b1bf50160a | ||
|
|
dc8f65fc64 | ||
|
|
8470534e33 | ||
|
|
c59cd9c424 | ||
|
|
9f76f0915b | ||
|
|
cdc45bd082 | ||
|
|
d810fc0a86 | ||
|
|
31158467f4 | ||
|
|
f8438c32ea | ||
|
|
9e61e3ba35 | ||
|
|
c2fa8067e1 | ||
|
|
cb8184e784 | ||
|
|
f37c6b92d8 | ||
|
|
006432c2dc | ||
|
|
5f85dbb718 | ||
|
|
b210acf3c2 | ||
|
|
44079eb8b4 | ||
|
|
d3a398716b | ||
|
|
98037f9b3e | ||
|
|
c85027c83a | ||
|
|
0ad53da49c | ||
|
|
e2c312b728 | ||
|
|
104e3ef27c | ||
|
|
4c418f7d9b | ||
|
|
b2e2735583 | ||
|
|
e417247e7e | ||
|
|
fe2451fd60 | ||
|
|
895413509d | ||
|
|
dd80b69992 | ||
|
|
768e106614 | ||
|
|
e4bddeb1ba | ||
|
|
25f075a8be | ||
|
|
f27d2605eb | ||
|
|
16cfc29074 | ||
|
|
529497febb | ||
|
|
171f901bcd | ||
|
|
e7b412d578 | ||
|
|
c66c3c6377 | ||
|
|
1f6108f769 | ||
|
|
3c3d01c8d1 | ||
|
|
c7c3eeab46 | ||
|
|
d9c5300859 | ||
|
|
c3ad5672fc | ||
|
|
5afcf63324 | ||
|
|
b18e62041b | ||
|
|
774f17d194 | ||
|
|
f56d41f5b7 | ||
|
|
e96c5143bc | ||
|
|
f68fc019e4 | ||
|
|
4967b9b8fd | ||
|
|
8ddea454d1 | ||
|
|
dcd9514622 | ||
|
|
42108c840d | ||
|
|
13a35138e9 | ||
|
|
d4af58be85 | ||
|
|
eacd3ee085 | ||
|
|
91fbd2dc88 | ||
|
|
e5f097c291 | ||
|
|
4fedfcec30 | ||
|
|
2056bb1d9e | ||
|
|
dc0443de34 | ||
|
|
d48bdbc9a7 | ||
|
|
52500a689c | ||
|
|
9c9439a271 | ||
|
|
ee5a939ce6 | ||
|
|
deed591da6 | ||
|
|
c3c4447810 | ||
|
|
72046e7985 | ||
|
|
d84d17207f | ||
|
|
3a2d76aa43 | ||
|
|
5c5f1ced33 | ||
|
|
8b12245e79 | ||
|
|
099a716bfd | ||
|
|
4193ae2cda | ||
|
|
8c93cd8569 | ||
|
|
09afa7e7ff | ||
|
|
5b49d5a1a8 | ||
|
|
28090d1de0 | ||
|
|
0b89b8316c | ||
|
|
62509a5090 | ||
|
|
3616bc4733 | ||
|
|
a8b8efba6a | ||
|
|
a4b4d05b8d | ||
|
|
e89a32ffef | ||
|
|
a93a4111e1 | ||
|
|
0d8db7ff0b | ||
|
|
2a9a62c784 | ||
|
|
6dd7937ece | ||
|
|
a20702d55b | ||
|
|
87f188ae73 | ||
|
|
f6c3ddbf81 | ||
|
|
821cbb8622 | ||
|
|
da889f83ab | ||
|
|
c28c7a148f | ||
|
|
89bc53b53d | ||
|
|
ceab28b902 | ||
|
|
bcf4866abc | ||
|
|
b36ae00ea5 | ||
|
|
cd4d76a8c3 | ||
|
|
5c066afa7b | ||
|
|
d24823b6f3 | ||
|
|
bf2055e725 | ||
|
|
72f8bdc87c | ||
|
|
e2f576ec02 | ||
|
|
1b556c5849 | ||
|
|
521da9feb9 | ||
|
|
3300c9d149 | ||
|
|
08dd227a45 | ||
|
|
0aeae07db2 | ||
|
|
a33dbdcdc3 | ||
|
|
a48d78f8eb | ||
|
|
742724e53c | ||
|
|
d3a53e7bf1 | ||
|
|
f7f3dfe495 | ||
|
|
75d09241fb | ||
|
|
aa470091aa | ||
|
|
1797669296 | ||
|
|
abb97e6f03 | ||
|
|
6991e21f94 | ||
|
|
1d554396f4 | ||
|
|
12147a1e01 | ||
|
|
e31688bac5 | ||
|
|
66f730ad16 | ||
|
|
d49acaed5e | ||
|
|
4efcde9d4f | ||
|
|
0d25a94a84 | ||
|
|
cb48f7ff3b | ||
|
|
c840688adb | ||
|
|
b17e18aa67 | ||
|
|
9aed20b6d0 | ||
|
|
bb807c2f3a | ||
|
|
8796fbbcbb | ||
|
|
11b274edc6 | ||
|
|
2dee941080 | ||
|
|
7696009b25 | ||
|
|
d9f53a3f96 | ||
|
|
1cd81a8b2a | ||
|
|
521b8dea10 | ||
|
|
0a9747091f | ||
|
|
c9b7d8b6ca | ||
|
|
0206be68e5 | ||
|
|
4f07430e92 | ||
|
|
76fe1f1148 | ||
|
|
ebdba34da6 | ||
|
|
abc4160a89 | ||
|
|
2edafdaf0d | ||
|
|
92055c4556 | ||
|
|
c3297b86cf | ||
|
|
bcd1a0127d | ||
|
|
0c291ed1bb | ||
|
|
fd16b3c126 | ||
|
|
6687f8b808 | ||
|
|
fcf5d7b16c | ||
|
|
08847e6a63 | ||
|
|
78da62f156 | ||
|
|
dec59764b1 | ||
|
|
0f2591bae4 | ||
|
|
02ba557c3e | ||
|
|
22efb93775 | ||
|
|
2d04c5e257 | ||
|
|
b87d89f9fa | ||
|
|
67c56ce19b | ||
|
|
0f7fa31f86 | ||
|
|
4454a1cfd9 | ||
|
|
e65be19a45 | ||
|
|
452b419729 | ||
|
|
da3731d753 | ||
|
|
f8ca0ced9a | ||
|
|
4f6719c80e | ||
|
|
3c91d0e172 | ||
|
|
1253595ba7 | ||
|
|
bb274d08c6 |
@@ -0,0 +1,287 @@
|
|||||||
|
# Local → production pipeline.
|
||||||
|
#
|
||||||
|
# push to main → test → build amd64 images → push to the fleet registry
|
||||||
|
# → move :latest → gw-04's existing 60s rolling timer picks it up.
|
||||||
|
#
|
||||||
|
# The last hop is NOT in this file and does not need to be: gw-04 already runs
|
||||||
|
# `clawmates-deploy.timer` every minute, which pulls
|
||||||
|
# `$REGISTRY/clawmates/<svc>:latest`, compares it to the running image id, and
|
||||||
|
# recreates on drift. This workflow's job is to make `:latest` mean the newest
|
||||||
|
# green commit. See deploy/gw-04/clawmates-deploy.sh.
|
||||||
|
#
|
||||||
|
# Runs on the `gw04` runner (host executor, systemd unit act-runner). gw-04 is
|
||||||
|
# the only reachable x86_64 host — web-01 is aarch64 and the fleet build boxes
|
||||||
|
# are packed — and prod images must be linux/amd64, so builds are native here
|
||||||
|
# rather than emulated.
|
||||||
|
name: deploy
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches: [main]
|
||||||
|
# Lets you re-run a deploy without an empty commit.
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
# Two pushes close together used to STOMP each other. Runs 490 and 491 started
|
||||||
|
# 16 minutes apart, a full suite takes longer than that, and the first thing a
|
||||||
|
# run does is `docker rm -fv` the shared test Postgres — so the newer run
|
||||||
|
# deleted the older run's database mid-suite and both failed. Nothing in the
|
||||||
|
# code was wrong; the logs blamed the tests.
|
||||||
|
#
|
||||||
|
# `cancel-in-progress` because a superseded run is testing a commit that is no
|
||||||
|
# longer the tip: finishing it costs 20 minutes to learn something that no
|
||||||
|
# longer matters.
|
||||||
|
concurrency:
|
||||||
|
group: deploy-${{ gitea.ref }}
|
||||||
|
cancel-in-progress: true
|
||||||
|
|
||||||
|
env:
|
||||||
|
REGISTRY: 100.94.185.103:5000
|
||||||
|
NAMESPACE: clawmates
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
test:
|
||||||
|
runs-on: gw04
|
||||||
|
env:
|
||||||
|
# Shared by the start and stop steps.
|
||||||
|
PG: cm-ci-pg-${{ gitea.run_id }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
# A throwaway Postgres so the integration tests actually run. Without
|
||||||
|
# CM_TEST_DATABASE_URL, cm-testkit tries a default admin URL and the
|
||||||
|
# approvals_api tests die on PoolTimedOut — which looks like a failure but
|
||||||
|
# only means "no database here".
|
||||||
|
- name: Start test Postgres
|
||||||
|
run: |
|
||||||
|
# Where every step leaves its full output, on the HOST, so a failed
|
||||||
|
# run can be read afterwards without the actions-log API.
|
||||||
|
#
|
||||||
|
# STEP is a breadcrumb: each step overwrites it on entry, so the last
|
||||||
|
# value names the step that died. Two steps used to create this
|
||||||
|
# directory, which made "the directory exists" ambiguous about how far
|
||||||
|
# the job got — and that ambiguity cost a whole debugging cycle.
|
||||||
|
mkdir -p /tmp/ci-logs && rm -f /tmp/ci-logs/*.log /tmp/ci-logs/STEP
|
||||||
|
echo "1-start-postgres" > /tmp/ci-logs/STEP
|
||||||
|
set -x
|
||||||
|
# Run-scoped name. `cm-ci-pg` was shared by every run, so a second
|
||||||
|
# run removed the first one's database while it was still being used.
|
||||||
|
# The concurrency group above should prevent overlap; this makes the
|
||||||
|
# failure impossible rather than merely unlikely.
|
||||||
|
docker rm -fv "$PG" 2>/dev/null || true
|
||||||
|
# --shm-size: Docker defaults /dev/shm to 64MB. cm-testkit creates a
|
||||||
|
# database per test and the suite runs many at once, so Postgres
|
||||||
|
# exhausts its parallel-query segments mid-run. It surfaces as
|
||||||
|
# `could not resize shared memory segment ... No space left on device`
|
||||||
|
# during MIGRATIONS, which reads like a schema fault and is not one.
|
||||||
|
# Hit locally on 2026-08-19; scripts/test-server.sh carries the same
|
||||||
|
# flag for the same reason.
|
||||||
|
docker run -d --name "$PG" \
|
||||||
|
--shm-size=1g \
|
||||||
|
-e POSTGRES_PASSWORD=postgres -e POSTGRES_USER=postgres \
|
||||||
|
-p 127.0.0.1:55432:5432 postgres:16-alpine
|
||||||
|
for i in $(seq 1 30); do
|
||||||
|
docker exec "$PG" pg_isready -U postgres >/dev/null 2>&1 && break
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
docker exec "$PG" pg_isready -U postgres
|
||||||
|
|
||||||
|
# Rust lives in a container because gw-04 has no cargo. The named volumes
|
||||||
|
# are the whole reason this is not painfully slow: without them every run
|
||||||
|
# recompiles the world.
|
||||||
|
# Docker socket AND the host's docker binary are mounted:
|
||||||
|
# - cm-files' s3_store test uses testcontainers (socket only).
|
||||||
|
# - cm-runtime/cm-sandbox tests (browser_tool, shell_exec, warm_pool,
|
||||||
|
# security, socket_proxy) shell out to `docker` via std::process, so
|
||||||
|
# they need the CLI on PATH too. Mounting the host binary beats
|
||||||
|
# apt-installing docker.io on every run — that is ~100 MB of download
|
||||||
|
# per job, and the container is fresh each time so nothing caches it.
|
||||||
|
# These tests do NOT skip when the capability is missing; they fail in a
|
||||||
|
# way that reads like broken code (SocketNotFoundError / NotFound), which
|
||||||
|
# is why they are worth wiring up rather than excluding.
|
||||||
|
#
|
||||||
|
# They also need clawmates/agent-{base,browser,terminal}:dev, which are
|
||||||
|
# locally-built images present on gw-04 but in no registry. If this job
|
||||||
|
# ever moves hosts, those images must move with it.
|
||||||
|
#
|
||||||
|
# `cargo test --workspace` builds cm-brain, which pulls clawhdf5 from
|
||||||
|
# git.redclaw.dev — a PRIVATE repo. Two things are needed and neither is
|
||||||
|
# optional:
|
||||||
|
# CARGO_NET_GIT_FETCH_WITH_CLI — libgit2 fails against Gitea's smart-HTTP
|
||||||
|
# with "invalid packet line" (the server Dockerfile sets it for the
|
||||||
|
# same reason). Note it is _GIT_FETCH_WITH_CLI, not _NET_FETCH_.
|
||||||
|
# the insteadOf rewrite — supplies the credential to that CLI fetch.
|
||||||
|
# The token is a repo secret, so it is masked in logs and never in git.
|
||||||
|
- name: Rust tests
|
||||||
|
run: |
|
||||||
|
echo "2-rust" > /tmp/ci-logs/STEP
|
||||||
|
# The DOCKER RUN's own output, on the host. cargo's log only exists
|
||||||
|
# if cargo runs; run 494 died in this step with no rust.log at all,
|
||||||
|
# which means apt-get, git config or docker itself failed and the
|
||||||
|
# message went only to the job log we cannot read.
|
||||||
|
set +e
|
||||||
|
docker run --rm --network host \
|
||||||
|
-v "$PWD":/w -w /w \
|
||||||
|
-v cm-ci-cargo-registry:/usr/local/cargo/registry \
|
||||||
|
-v cm-ci-cargo-git:/usr/local/cargo/git \
|
||||||
|
-v cm-ci-target:/w/target \
|
||||||
|
-v /var/run/docker.sock:/var/run/docker.sock \
|
||||||
|
-v /usr/bin/docker:/usr/bin/docker:ro \
|
||||||
|
-e SQLX_OFFLINE=true \
|
||||||
|
-e CARGO_NET_GIT_FETCH_WITH_CLI=true \
|
||||||
|
-e FORGE_TOKEN='${{ secrets.FORGE_TOKEN }}' \
|
||||||
|
-e CM_TEST_DATABASE_URL=postgres://postgres:[email protected]:55432/postgres \
|
||||||
|
-v /tmp/ci-logs:/cilog \
|
||||||
|
rust:1.96-slim \
|
||||||
|
sh -c 'set -e
|
||||||
|
# NO APOSTROPHES BELOW THIS LINE. Everything here is inside a
|
||||||
|
# single-quoted sh -c, so one apostrophe in a COMMENT closes the
|
||||||
|
# quote and the step dies with "unexpected EOF while looking for
|
||||||
|
# matching quote" — before running anything, which is why no log
|
||||||
|
# ever appeared. Runs 491 through 496 failed on the word
|
||||||
|
# "cm-api" followed by an apostrophe-s.
|
||||||
|
apt-get update -qq
|
||||||
|
# nodejs: the vm_tool_gate shell tests in cm-api EXECUTE the generated
|
||||||
|
# PreToolUse hook, which parses its JSON payload with node (no jq
|
||||||
|
# in the runtime image; node is guaranteed there because Claude
|
||||||
|
# Code is a node program). Without it the hook takes its
|
||||||
|
# allow-and-record-inert path and the two "blocks" tests fail —
|
||||||
|
# which is how this was found, on the first push that carried them.
|
||||||
|
apt-get install -y -qq pkg-config libssl-dev cmake git nodejs >/dev/null
|
||||||
|
git config --global url."https://oauth2:[email protected]/".insteadOf "https://git.redclaw.dev/"
|
||||||
|
# Full output to a host-mounted file, then the tail, then exit
|
||||||
|
# with the cargo status. Piping cargo into `tail` would report
|
||||||
|
# the exit code of tail — a green job over a red suite. The log
|
||||||
|
# survives the container so a failure is diagnosable at all:
|
||||||
|
# the Gitea actions-log API returns 403 for our token, and three
|
||||||
|
# failed runs were debugged blind before this existed.
|
||||||
|
set +e
|
||||||
|
cargo test --workspace > /cilog/rust.log 2>&1
|
||||||
|
rc=$?
|
||||||
|
set -e
|
||||||
|
grep -nE "test result: FAILED|^error(\[|:)|panicked at" /cilog/rust.log | head -40 || true
|
||||||
|
tail -40 /cilog/rust.log
|
||||||
|
exit $rc' > /tmp/ci-logs/rust-step.log 2>&1
|
||||||
|
rc=$?
|
||||||
|
set -e
|
||||||
|
tail -60 /tmp/ci-logs/rust-step.log
|
||||||
|
exit $rc
|
||||||
|
|
||||||
|
# -v, not just -f. The postgres image declares a VOLUME, so removing the
|
||||||
|
# container without it orphans an anonymous data directory EVERY run.
|
||||||
|
# cm-testkit creates a database per test, so those grew to 2.8 GB each —
|
||||||
|
# 38 GB of leaked volumes before anyone noticed.
|
||||||
|
- name: Stop test Postgres
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
echo "3-stop-postgres" >> /tmp/ci-logs/STEP
|
||||||
|
docker rm -fv "$PG" 2>/dev/null || true
|
||||||
|
|
||||||
|
# node 22 is on the host, so these run directly.
|
||||||
|
- name: Frontend checks
|
||||||
|
working-directory: frontend
|
||||||
|
run: |
|
||||||
|
echo "4-frontend" >> /tmp/ci-logs/STEP
|
||||||
|
set +e
|
||||||
|
npm ci --no-audit --no-fund > /tmp/ci-logs/npm-ci.log 2>&1; ci=$?
|
||||||
|
npm run typecheck > /tmp/ci-logs/typecheck.log 2>&1; tc=$?
|
||||||
|
npm run test > /tmp/ci-logs/vitest.log 2>&1; vt=$?
|
||||||
|
set -e
|
||||||
|
for f in npm-ci typecheck vitest; do
|
||||||
|
printf '=== %s ===\n' "$f"; tail -25 "/tmp/ci-logs/$f.log" || true
|
||||||
|
done
|
||||||
|
[ "$ci" -eq 0 ] && [ "$tc" -eq 0 ] && [ "$vt" -eq 0 ]
|
||||||
|
# Lint is advisory: the repo currently has pre-existing max-lines and
|
||||||
|
# set-state-in-effect errors that predate this pipeline. Failing the
|
||||||
|
# deploy on them would mean nothing could ship until they are cleared.
|
||||||
|
npm run lint || echo "::warning::lint reported problems (advisory)"
|
||||||
|
|
||||||
|
build:
|
||||||
|
runs-on: gw04
|
||||||
|
needs: test
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Build + push images
|
||||||
|
run: |
|
||||||
|
# Same host-log treatment as the test job. The build job failed four
|
||||||
|
# runs in a row with nothing readable: the actions-log API returns
|
||||||
|
# 403 for our token, so "failure" was the entire message. It turned
|
||||||
|
# out to be transient disk pressure — a runtime image being built on
|
||||||
|
# this same host at the same time — and a docs-only commit was the
|
||||||
|
# first casualty, which made it look like a code regression.
|
||||||
|
mkdir -p /tmp/ci-logs
|
||||||
|
echo "5-build" > /tmp/ci-logs/STEP
|
||||||
|
df -h / > /tmp/ci-logs/build-disk.log 2>&1
|
||||||
|
set -eu
|
||||||
|
SHA=$(git rev-parse --short HEAD)
|
||||||
|
echo "SHA=$SHA" >> "$GITHUB_ENV"
|
||||||
|
# The daemon binary the frontend serves at /dl. images/frontend.Dockerfile
|
||||||
|
# expects it staged; rsync-based deploys create it out of band, so build
|
||||||
|
# it here or the image ships without the node installer.
|
||||||
|
mkdir -p frontend/public/dl
|
||||||
|
docker run --rm \
|
||||||
|
-v "$PWD":/w -w /w \
|
||||||
|
-v cm-ci-cargo-registry:/usr/local/cargo/registry \
|
||||||
|
-v cm-ci-cargo-git:/usr/local/cargo/git \
|
||||||
|
-v cm-ci-target:/w/target \
|
||||||
|
-e SQLX_OFFLINE=true -e CARGO_NET_GIT_FETCH_WITH_CLI=true \
|
||||||
|
-e FORGE_TOKEN='${{ secrets.FORGE_TOKEN }}' \
|
||||||
|
rust:1.96-slim \
|
||||||
|
sh -c 'set -e
|
||||||
|
apt-get update -qq
|
||||||
|
apt-get install -y -qq pkg-config libssl-dev cmake git >/dev/null
|
||||||
|
git config --global url."https://oauth2:[email protected]/".insteadOf "https://git.redclaw.dev/"
|
||||||
|
cargo build --release -p clawmates-node
|
||||||
|
cp target/release/clawmates-node frontend/public/dl/clawmates-node-linux-amd64'
|
||||||
|
|
||||||
|
for svc in server frontend broker; do
|
||||||
|
docker build -f "images/$svc.Dockerfile" \
|
||||||
|
-t "$REGISTRY/$NAMESPACE/$svc:main-$SHA" \
|
||||||
|
-t "$REGISTRY/$NAMESPACE/$svc:latest" .
|
||||||
|
docker push "$REGISTRY/$NAMESPACE/$svc:main-$SHA"
|
||||||
|
docker push "$REGISTRY/$NAMESPACE/$svc:latest"
|
||||||
|
echo "$svc built+pushed" >> /tmp/ci-logs/build-progress.log
|
||||||
|
done
|
||||||
|
|
||||||
|
# `docker push :latest` does NOT reliably move the tag on this registry:
|
||||||
|
# when the manifest already exists under another tag (it does — we just
|
||||||
|
# pushed main-$SHA), the push reports a digest but `:latest` keeps
|
||||||
|
# resolving to the OLD image. Writing the manifest to the tag over the
|
||||||
|
# HTTP API is what actually moves it. This is the same trick
|
||||||
|
# scripts/deploy.sh uses, and the reason a "successful" deploy could
|
||||||
|
# previously leave prod on a stale image.
|
||||||
|
- name: Repoint :latest
|
||||||
|
run: |
|
||||||
|
echo "6-repoint" > /tmp/ci-logs/STEP
|
||||||
|
set -eu
|
||||||
|
for svc in server frontend broker; do
|
||||||
|
ct=$(curl -s -o /tmp/m.json -D- \
|
||||||
|
-H 'Accept: application/vnd.oci.image.index.v1+json,application/vnd.docker.distribution.manifest.list.v2+json,application/vnd.docker.distribution.manifest.v2+json,application/vnd.oci.image.manifest.v1+json' \
|
||||||
|
"http://$REGISTRY/v2/$NAMESPACE/$svc/manifests/main-$SHA" \
|
||||||
|
| awk -F': ' '/^[Cc]ontent-[Tt]ype/{print $2}' | tr -d '\r')
|
||||||
|
code=$(curl -s -o /dev/null -w '%{http_code}' -X PUT \
|
||||||
|
-H "Content-Type: $ct" --data-binary @/tmp/m.json \
|
||||||
|
"http://$REGISTRY/v2/$NAMESPACE/$svc/manifests/latest")
|
||||||
|
echo "$svc :latest → main-$SHA (HTTP $code)"
|
||||||
|
case "$code" in 20*) ;; *) echo "tag write failed"; exit 1 ;; esac
|
||||||
|
done
|
||||||
|
|
||||||
|
# Verify the thing that actually matters: what prod is RUNNING, not what
|
||||||
|
# we pushed. A green edge on a stale image is the failure mode this whole
|
||||||
|
# pipeline exists to prevent.
|
||||||
|
- name: Wait for the rolling deploy
|
||||||
|
run: |
|
||||||
|
echo "7-wait-deploy" > /tmp/ci-logs/STEP
|
||||||
|
set -eu
|
||||||
|
want=$(docker image inspect -f '{{.Id}}' "$REGISTRY/$NAMESPACE/server:latest")
|
||||||
|
for i in $(seq 1 30); do
|
||||||
|
got=$(docker inspect -f '{{.Image}}' clawmates_server_1 2>/dev/null || echo none)
|
||||||
|
if [ "$got" = "$want" ]; then
|
||||||
|
echo "prod is running main-$SHA"
|
||||||
|
curl -s -o /dev/null -w "edge HTTP %{http_code}\n" -m 10 https://clawmates.work/ || true
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
sleep 10
|
||||||
|
done
|
||||||
|
echo "prod did not roll onto main-$SHA within 5m — check clawmates-deploy.timer"
|
||||||
|
exit 1
|
||||||
@@ -0,0 +1,215 @@
|
|||||||
|
# Release: build the images both deploy targets share, assemble the SIGNED
|
||||||
|
# air-gapped bundle, verify it offline, rehearse the customer's install, and
|
||||||
|
# attach everything to the Gitea release for the tag.
|
||||||
|
#
|
||||||
|
# Moved from .github/workflows/ and rewritten for this forge. The old copy could
|
||||||
|
# never have run: `runs-on: ubuntu-latest` matches no runner here, and
|
||||||
|
# `softprops/action-gh-release` talks to GitHub's API, not Gitea's.
|
||||||
|
#
|
||||||
|
# The signing key is a repo secret (BUNDLE_SIGNING_KEY, hex ed25519 from
|
||||||
|
# `clawmates-bundler keygen`). The matching PUBLIC key is published out of band
|
||||||
|
# so customers can verify a bundle before `docker load`.
|
||||||
|
name: release
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
tags: ["v*"]
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
bundle:
|
||||||
|
runs-on: gw04
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Version from tag
|
||||||
|
run: |
|
||||||
|
# workflow_dispatch has no tag; fall back to the short sha so a manual
|
||||||
|
# run produces a clearly-not-a-release version rather than an empty one.
|
||||||
|
if [ "${GITHUB_REF_TYPE:-}" = "tag" ]; then
|
||||||
|
echo "VERSION=${GITHUB_REF_NAME#v}" >> "$GITHUB_ENV"
|
||||||
|
else
|
||||||
|
echo "VERSION=0.0.0-$(git rev-parse --short HEAD)" >> "$GITHUB_ENV"
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: Build images
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
docker build -t "clawmates/server:$VERSION" -f images/server.Dockerfile .
|
||||||
|
docker build -t "clawmates/frontend:$VERSION" -f images/frontend.Dockerfile .
|
||||||
|
docker build -t "clawmates/broker:$VERSION" -f images/broker.Dockerfile .
|
||||||
|
docker build -t "clawmates/agent-base:$VERSION" images/agent-base
|
||||||
|
docker build -t "clawmates/agent-browser:$VERSION" images/agent-browser
|
||||||
|
docker pull -q postgres:16-alpine
|
||||||
|
docker pull -q tecnativa/docker-socket-proxy:0.3
|
||||||
|
|
||||||
|
# syft goes in the workspace, NOT /usr/local/bin. The host executor runs
|
||||||
|
# as root on the production gateway; a release should not leave binaries
|
||||||
|
# behind on it.
|
||||||
|
- name: SBOMs for every shipped image
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
mkdir -p dist/sboms .tools
|
||||||
|
curl -sSfL https://raw.githubusercontent.com/anchore/syft/main/install.sh \
|
||||||
|
| sh -s -- -b .tools
|
||||||
|
for image in server frontend broker agent-base agent-browser; do
|
||||||
|
./.tools/syft "clawmates/$image:$VERSION" -o spdx-json \
|
||||||
|
> "dist/sboms/$image.spdx.json"
|
||||||
|
done
|
||||||
|
|
||||||
|
- name: Save image tarballs
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
mkdir -p dist/images
|
||||||
|
docker save "clawmates/server:$VERSION" -o dist/images/server.tar
|
||||||
|
docker save "clawmates/frontend:$VERSION" -o dist/images/frontend.tar
|
||||||
|
docker save "clawmates/broker:$VERSION" -o dist/images/broker.tar
|
||||||
|
docker save "clawmates/agent-base:$VERSION" -o dist/images/agent-base.tar
|
||||||
|
docker save "clawmates/agent-browser:$VERSION" -o dist/images/agent-browser.tar
|
||||||
|
docker save tecnativa/docker-socket-proxy:0.3 -o dist/images/socket-proxy.tar
|
||||||
|
docker save postgres:16-alpine -o dist/images/postgres.tar
|
||||||
|
du -sh dist/images
|
||||||
|
|
||||||
|
# gw-04 has no cargo, so the bundler builds in a container — same pattern
|
||||||
|
# and same cache volumes as deploy.yml. The forge credential is here
|
||||||
|
# because cargo resolves the whole workspace, which includes cm-brain's
|
||||||
|
# private clawhdf5 git dependency.
|
||||||
|
- name: Build bundler
|
||||||
|
run: |
|
||||||
|
docker run --rm \
|
||||||
|
-v "$PWD":/w -w /w \
|
||||||
|
-v cm-ci-cargo-registry:/usr/local/cargo/registry \
|
||||||
|
-v cm-ci-cargo-git:/usr/local/cargo/git \
|
||||||
|
-v cm-ci-target:/w/target \
|
||||||
|
-e SQLX_OFFLINE=true -e CARGO_NET_GIT_FETCH_WITH_CLI=true \
|
||||||
|
-e FORGE_TOKEN='${{ secrets.FORGE_TOKEN }}' \
|
||||||
|
rust:1.96-slim \
|
||||||
|
sh -c 'set -e
|
||||||
|
apt-get update -qq
|
||||||
|
apt-get install -y -qq pkg-config libssl-dev cmake git >/dev/null
|
||||||
|
git config --global url."https://oauth2:[email protected]/".insteadOf "https://git.redclaw.dev/"
|
||||||
|
cargo build --release -p clawmates-bundler
|
||||||
|
# Copy the binary OUT of the target volume and into the workspace.
|
||||||
|
# /w/target is a named docker volume, so anything left there is
|
||||||
|
# invisible to later steps running on the host — which is exactly
|
||||||
|
# how this failed the first time (exit 127, No such file).
|
||||||
|
mkdir -p /w/.tools
|
||||||
|
cp target/release/clawmates-bundler /w/.tools/clawmates-bundler'
|
||||||
|
test -x .tools/clawmates-bundler || { echo "bundler did not land in the workspace"; exit 1; }
|
||||||
|
|
||||||
|
- name: Assemble and sign the bundle
|
||||||
|
env:
|
||||||
|
BUNDLE_SIGNING_KEY: ${{ secrets.BUNDLE_SIGNING_KEY }}
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
test -n "$BUNDLE_SIGNING_KEY" || { echo "BUNDLE_SIGNING_KEY is empty"; exit 1; }
|
||||||
|
umask 077
|
||||||
|
printf '%s' "$BUNDLE_SIGNING_KEY" > /tmp/release.key
|
||||||
|
BUNDLER=.tools/clawmates-bundler
|
||||||
|
ARTIFACTS=""
|
||||||
|
for tar in dist/images/*.tar; do
|
||||||
|
ARTIFACTS="$ARTIFACTS $tar=images/$(basename "$tar")"
|
||||||
|
done
|
||||||
|
for migration in migrations/*.sql; do
|
||||||
|
ARTIFACTS="$ARTIFACTS $migration=migrations/$(basename "$migration")"
|
||||||
|
done
|
||||||
|
# shellcheck disable=SC2086
|
||||||
|
"$BUNDLER" assemble dist/bundle "$VERSION" /tmp/release.key \
|
||||||
|
deploy/compose/docker-compose.yml=compose/docker-compose.yml \
|
||||||
|
deploy/compose/clawmates.toml=compose/clawmates.toml \
|
||||||
|
deploy/compose/.env.example=compose/.env.example \
|
||||||
|
deploy/e2e/scenarios.toml=compose/scenarios.toml \
|
||||||
|
images/seccomp/agent-profile.json=seccomp/agent-profile.json \
|
||||||
|
deploy/airgapped/install.sh=install.sh \
|
||||||
|
"$BUNDLER"=bin/clawmates-bundler \
|
||||||
|
dist/sboms/server.spdx.json=sboms/server.spdx.json \
|
||||||
|
dist/sboms/frontend.spdx.json=sboms/frontend.spdx.json \
|
||||||
|
dist/sboms/agent-base.spdx.json=sboms/agent-base.spdx.json \
|
||||||
|
dist/sboms/agent-browser.spdx.json=sboms/agent-browser.spdx.json \
|
||||||
|
$ARTIFACTS
|
||||||
|
chmod +x dist/bundle/bin/clawmates-bundler dist/bundle/install.sh
|
||||||
|
rm -f /tmp/release.key
|
||||||
|
|
||||||
|
- name: Verify the bundle offline (public key only)
|
||||||
|
env:
|
||||||
|
BUNDLE_SIGNING_KEY: ${{ secrets.BUNDLE_SIGNING_KEY }}
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
umask 077
|
||||||
|
printf '%s' "$BUNDLE_SIGNING_KEY" > /tmp/release.key
|
||||||
|
.tools/clawmates-bundler pubkey /tmp/release.key dist/release.pub
|
||||||
|
rm -f /tmp/release.key
|
||||||
|
# The customer's exact procedure: the public half only, inside a
|
||||||
|
# NETWORK-DISABLED container, proving verification needs no internet.
|
||||||
|
docker run --rm --network none \
|
||||||
|
-v "$PWD/dist:/dist:ro" \
|
||||||
|
ubuntu:24.04 \
|
||||||
|
/dist/bundle/bin/clawmates-bundler verify /dist/bundle /dist/release.pub
|
||||||
|
|
||||||
|
- name: Tarball
|
||||||
|
run: tar -C dist -czf "clawmates-bundle-$VERSION.tgz" bundle
|
||||||
|
|
||||||
|
# The clean-room install rehearsal is DELIBERATELY NOT RUN HERE.
|
||||||
|
#
|
||||||
|
# Every other step in this job is inert with respect to production: it
|
||||||
|
# builds images, writes SBOMs, signs a bundle, and verifies it in a
|
||||||
|
# network-isolated container. The rehearsal is the one step whose entire
|
||||||
|
# purpose is to stand a full stack UP and then tear it down with
|
||||||
|
# `down -v` — on the machine serving production.
|
||||||
|
#
|
||||||
|
# On 2026-08-13 it did exactly that: the bundled compose file declares
|
||||||
|
# `name: clawmates`, which beat --project-directory, so the rehearsal
|
||||||
|
# adopted the live stack and its teardown deleted clawmates_pgdata. The
|
||||||
|
# database was lost and there were no backups.
|
||||||
|
#
|
||||||
|
# scripts/rehearse-install.sh is now isolated (`-p rehearse-$$` plus a
|
||||||
|
# guard that refuses the production project name) and its health probe is
|
||||||
|
# fixed, so it is safe to run — just not on this host. Run it on a build
|
||||||
|
# box or throwaway VM:
|
||||||
|
#
|
||||||
|
# CLAWMATES_BUNDLER=… COMPOSE=/path/to/compose-v2 ./scripts/rehearse-install.sh
|
||||||
|
#
|
||||||
|
# Restore this step here only if the release ever moves off the gateway.
|
||||||
|
|
||||||
|
# Gitea's release API, not softprops/action-gh-release (GitHub-only).
|
||||||
|
# Create-or-reuse, so a re-run of the same tag updates instead of 409ing.
|
||||||
|
# Tag pushes only. On workflow_dispatch GITHUB_REF_NAME is the BRANCH, so
|
||||||
|
# this step previously created a release — and a git tag — literally named
|
||||||
|
# "main". A smoke-test run must not be able to mint a release.
|
||||||
|
- name: Attach to the Gitea release
|
||||||
|
if: github.ref_type == 'tag'
|
||||||
|
env:
|
||||||
|
FORGE_TOKEN: ${{ secrets.FORGE_TOKEN }}
|
||||||
|
run: |
|
||||||
|
set -eu
|
||||||
|
API="https://git.redclaw.dev/api/v1/repos/$GITHUB_REPOSITORY/releases"
|
||||||
|
TAG="${GITHUB_REF_NAME}"
|
||||||
|
id=$(curl -sS -H "Authorization: token $FORGE_TOKEN" "$API/tags/$TAG" \
|
||||||
|
| sed -n 's/.*"id":[ ]*\([0-9]\+\).*/\1/p' | head -1)
|
||||||
|
if [ -z "$id" ]; then
|
||||||
|
id=$(curl -sS -X POST -H "Authorization: token $FORGE_TOKEN" \
|
||||||
|
-H 'content-type: application/json' \
|
||||||
|
-d "{\"tag_name\":\"$TAG\",\"name\":\"$TAG\",\"body\":\"Air-gapped bundle for $TAG. Verify with the published public key before docker load.\"}" \
|
||||||
|
"$API" | sed -n 's/.*"id":[ ]*\([0-9]\+\).*/\1/p' | head -1)
|
||||||
|
fi
|
||||||
|
test -n "$id" || { echo "could not create or find the release for $TAG"; exit 1; }
|
||||||
|
for f in "clawmates-bundle-$VERSION.tgz" dist/release.pub; do
|
||||||
|
code=$(curl -sS -o /dev/null -w '%{http_code}' -X POST \
|
||||||
|
-H "Authorization: token $FORGE_TOKEN" \
|
||||||
|
-F "attachment=@$f" \
|
||||||
|
"$API/$id/assets?name=$(basename "$f")")
|
||||||
|
echo " attached $(basename "$f") (HTTP $code)"
|
||||||
|
case "$code" in 20*) ;; *) echo "attach failed"; exit 1 ;; esac
|
||||||
|
done
|
||||||
|
|
||||||
|
# Release artifacts are GBs of image tarballs on the production gateway.
|
||||||
|
# Never `docker image prune -a` here: clawmates/agent-*:dev exist in no
|
||||||
|
# registry and are the source of the microVM rootfs files.
|
||||||
|
- name: Reclaim disk
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
rm -rf dist .tools "clawmates-bundle-$VERSION.tgz" || true
|
||||||
|
for i in server frontend broker agent-base agent-browser; do
|
||||||
|
docker rmi "clawmates/$i:$VERSION" 2>/dev/null || true
|
||||||
|
done
|
||||||
|
df -h / | awk 'NR==2{print " disk free: "$4}'
|
||||||
@@ -1,209 +0,0 @@
|
|||||||
name: ci
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches: [main]
|
|
||||||
pull_request:
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
gates:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- name: File size budget (1500 lines)
|
|
||||||
run: ./ci/check-loc.sh
|
|
||||||
- name: No placeholder markers
|
|
||||||
run: ./ci/check-no-placeholders.sh
|
|
||||||
- name: Compose config validates
|
|
||||||
run: POSTGRES_PASSWORD=ci docker compose -f deploy/compose/docker-compose.yml config -q
|
|
||||||
|
|
||||||
rust:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
needs: gates
|
|
||||||
# Compile sqlx query! macros against the committed .sqlx cache (no DB needed).
|
|
||||||
# Tests need a live Postgres — locally cm-testkit reads CM_TEST_DATABASE_URL
|
|
||||||
# from .cargo/config.toml pointing at scripts/test-server.sh's host container.
|
|
||||||
# The fleet act_runner uses the `host` executor (jobs run on morpheus/tank/
|
|
||||||
# architect natively, not inside a container), so we start a per-run postgres
|
|
||||||
# container and reach it via its bridge IP. GITHUB_RUN_ID scopes the name so
|
|
||||||
# concurrent jobs on the same runner don't collide.
|
|
||||||
#
|
|
||||||
# GIT_CONFIG_GLOBAL points at a per-job empty file so cargo's git fetches
|
|
||||||
# bypass the runner's includeIf mapping of git.redclaw.dev → /slab/projects
|
|
||||||
# (local mirror lags and misses recently-pinned commits like the clawverse
|
|
||||||
# rev cm-brain depends on). clawverse is public; no auth needed.
|
|
||||||
env:
|
|
||||||
SQLX_OFFLINE: "true"
|
|
||||||
GIT_CONFIG_GLOBAL: /tmp/ci-empty-gitconfig-${{ github.run_id }}
|
|
||||||
steps:
|
|
||||||
- name: Prepare empty gitconfig for cargo fetches
|
|
||||||
run: touch "$GIT_CONFIG_GLOBAL"
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- name: Start postgres sidecar
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
NAME="ci-pg-${GITHUB_RUN_ID}"
|
|
||||||
docker rm -f "$NAME" >/dev/null 2>&1 || true
|
|
||||||
docker run -d --name "$NAME" \
|
|
||||||
-e POSTGRES_PASSWORD=postgres \
|
|
||||||
-e POSTGRES_DB=postgres \
|
|
||||||
postgres:16-alpine >/dev/null
|
|
||||||
# `.NetworkSettings.IPAddress` is empty (and template-parse errors) on
|
|
||||||
# modern Docker where the IP lives under `.Networks.<name>.IPAddress`.
|
|
||||||
# The range form picks the first non-empty IP across whatever network
|
|
||||||
# docker put the container on.
|
|
||||||
PG_IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$NAME")
|
|
||||||
if [ -z "$PG_IP" ]; then
|
|
||||||
echo "postgres has no reachable IP" >&2
|
|
||||||
docker inspect "$NAME" >&2
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
echo "PG_CONTAINER=$NAME" >> "$GITHUB_ENV"
|
|
||||||
echo "CM_TEST_DATABASE_URL=postgres://postgres:postgres@${PG_IP}:5432/postgres" >> "$GITHUB_ENV"
|
|
||||||
for i in $(seq 1 30); do
|
|
||||||
if docker exec "$NAME" pg_isready -U postgres -q >/dev/null 2>&1; then
|
|
||||||
echo "postgres ready at ${PG_IP} after ${i}s"
|
|
||||||
exit 0
|
|
||||||
fi
|
|
||||||
sleep 1
|
|
||||||
done
|
|
||||||
echo "postgres never became ready" >&2
|
|
||||||
docker logs "$NAME" >&2 || true
|
|
||||||
exit 1
|
|
||||||
- uses: dtolnay/rust-toolchain@stable
|
|
||||||
with:
|
|
||||||
toolchain: 1.96.0
|
|
||||||
components: rustfmt, clippy
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
- name: Format
|
|
||||||
run: cargo fmt --all --check
|
|
||||||
- name: Clippy
|
|
||||||
run: cargo clippy --workspace --all-targets -- -D warnings
|
|
||||||
- name: Test
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
# Re-derive the postgres URL inline instead of trusting that
|
|
||||||
# CM_TEST_DATABASE_URL propagated through $GITHUB_ENV — act_runner
|
|
||||||
# v1.0.8 has been observed to swallow env-file writes here.
|
|
||||||
IP=$(docker inspect -f '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}' "$PG_CONTAINER")
|
|
||||||
[ -n "$IP" ] || { echo "no PG IP" >&2; exit 1; }
|
|
||||||
export CM_TEST_DATABASE_URL="postgres://postgres:postgres@${IP}:5432/postgres"
|
|
||||||
echo "using $CM_TEST_DATABASE_URL"
|
|
||||||
cargo test --workspace
|
|
||||||
- name: Air-gapped installer verify path
|
|
||||||
run: ./ci/test-install.sh
|
|
||||||
- name: Cleanup postgres sidecar
|
|
||||||
if: always()
|
|
||||||
run: docker rm -f "${PG_CONTAINER:-}" >/dev/null 2>&1 || true
|
|
||||||
|
|
||||||
frontend:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
needs: gates
|
|
||||||
defaults:
|
|
||||||
run:
|
|
||||||
working-directory: frontend
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- uses: actions/setup-node@v4
|
|
||||||
with:
|
|
||||||
node-version: 22
|
|
||||||
- name: Install
|
|
||||||
run: npm ci
|
|
||||||
if: ${{ hashFiles('frontend/package-lock.json') != '' }}
|
|
||||||
- name: Lint
|
|
||||||
run: npm run lint
|
|
||||||
if: ${{ hashFiles('frontend/package-lock.json') != '' }}
|
|
||||||
- name: Typecheck
|
|
||||||
run: npm run typecheck
|
|
||||||
if: ${{ hashFiles('frontend/package-lock.json') != '' }}
|
|
||||||
- name: Unit and component tests
|
|
||||||
run: npm test
|
|
||||||
if: ${{ hashFiles('frontend/package-lock.json') != '' }}
|
|
||||||
|
|
||||||
# e2e is intentionally disabled for now. The suite has real product/test
|
|
||||||
# drift (locators pointing at older versions of pages) that would need a
|
|
||||||
# dedicated pass to reconcile — see the earlier follow-up notes. Publish
|
|
||||||
# doesn't depend on this job anyway, but keeping it enabled produced a
|
|
||||||
# steady red on every push that wasn't actionable. Flip `if:` back to
|
|
||||||
# `true` (or delete the guard) when the tests get realigned.
|
|
||||||
e2e:
|
|
||||||
if: false
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
needs: [rust, frontend]
|
|
||||||
env:
|
|
||||||
SQLX_OFFLINE: "true"
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- uses: dtolnay/rust-toolchain@stable
|
|
||||||
with:
|
|
||||||
toolchain: 1.96.0
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
- uses: actions/setup-node@v4
|
|
||||||
with:
|
|
||||||
node-version: 22
|
|
||||||
- name: Install frontend dependencies
|
|
||||||
working-directory: frontend
|
|
||||||
run: npm ci
|
|
||||||
- name: Install Playwright browsers
|
|
||||||
working-directory: frontend
|
|
||||||
run: npx playwright install --with-deps chromium
|
|
||||||
- name: Run end-to-end journeys against the real backend
|
|
||||||
working-directory: frontend
|
|
||||||
run: npx playwright test --grep-invert "@visual"
|
|
||||||
- uses: actions/upload-artifact@v4
|
|
||||||
if: failure()
|
|
||||||
with:
|
|
||||||
name: playwright-traces
|
|
||||||
path: frontend/test-results/
|
|
||||||
|
|
||||||
# Rolling deploy: on green main only, build the three prod images, tag with
|
|
||||||
# :main-<sha> + :latest, push to the fleet registry (redclaw-web-01:5000 via
|
|
||||||
# its Tailscale IP — the fleet's daemons trust it in insecure-registries by
|
|
||||||
# IP, not by hostname). GW-04's clawmates-deploy.timer rolls forward within
|
|
||||||
# ~1 minute of the push. Skipped on PRs.
|
|
||||||
#
|
|
||||||
# `e2e` is intentionally NOT in `needs`: it launches its own postgres + dex
|
|
||||||
# via `docker run` on the host and then reaches them via 127.0.0.1, which
|
|
||||||
# fails from inside the act_runner container. Migrating e2e to a physical
|
|
||||||
# build node is a separate task; until then e2e is signal-only, not gating.
|
|
||||||
# `rust` was restored to `needs` once the flakes were rooted out (approvals
|
|
||||||
# SSE race + warm_pool agent-seeding + a couple health-check ambiguities).
|
|
||||||
publish:
|
|
||||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
needs: [gates, rust, frontend]
|
|
||||||
env:
|
|
||||||
REGISTRY: 100.94.185.103:5000
|
|
||||||
NAMESPACE: clawmates
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- name: Resolve short SHA
|
|
||||||
run: echo "SHA=${GITHUB_SHA::7}" >> "$GITHUB_ENV"
|
|
||||||
- name: Build images
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
for svc in broker server frontend; do
|
|
||||||
docker build \
|
|
||||||
-t "${REGISTRY}/${NAMESPACE}/${svc}:main-${SHA}" \
|
|
||||||
-t "${REGISTRY}/${NAMESPACE}/${svc}:latest" \
|
|
||||||
-f "images/${svc}.Dockerfile" .
|
|
||||||
done
|
|
||||||
- name: Push images
|
|
||||||
run: |
|
|
||||||
set -euo pipefail
|
|
||||||
for svc in broker server frontend; do
|
|
||||||
docker push "${REGISTRY}/${NAMESPACE}/${svc}:main-${SHA}"
|
|
||||||
docker push "${REGISTRY}/${NAMESPACE}/${svc}:latest"
|
|
||||||
done
|
|
||||||
- name: Summary
|
|
||||||
run: |
|
|
||||||
{
|
|
||||||
echo "## Published images"
|
|
||||||
echo ""
|
|
||||||
for svc in broker server frontend; do
|
|
||||||
echo "- \`${REGISTRY}/${NAMESPACE}/${svc}:main-${SHA}\`"
|
|
||||||
echo "- \`${REGISTRY}/${NAMESPACE}/${svc}:latest\`"
|
|
||||||
done
|
|
||||||
echo ""
|
|
||||||
echo "GW-04 timer picks these up within ~1 minute."
|
|
||||||
} >> "$GITHUB_STEP_SUMMARY"
|
|
||||||
@@ -1,117 +0,0 @@
|
|||||||
# Release: build the images both deploy targets share, assemble the
|
|
||||||
# SIGNED air-gapped bundle, verify it offline, and attach everything to
|
|
||||||
# the tag. The signing key lives in repo secrets (BUNDLE_SIGNING_KEY,
|
|
||||||
# hex ed25519 from `clawmates-bundler keygen`); the matching public key is
|
|
||||||
# published out of band so customers can verify before docker load.
|
|
||||||
name: release
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
tags: ["v*"]
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
bundle:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
steps:
|
|
||||||
- uses: actions/checkout@v4
|
|
||||||
- uses: dtolnay/rust-toolchain@stable
|
|
||||||
- uses: Swatinem/rust-cache@v2
|
|
||||||
|
|
||||||
- name: Version from tag
|
|
||||||
run: echo "VERSION=${GITHUB_REF_NAME#v}" >> "$GITHUB_ENV"
|
|
||||||
|
|
||||||
- name: Build images
|
|
||||||
run: |
|
|
||||||
docker build -t "clawmates/server:$VERSION" -f images/server.Dockerfile .
|
|
||||||
docker build -t "clawmates/frontend:$VERSION" -f images/frontend.Dockerfile .
|
|
||||||
docker build -t "clawmates/broker:$VERSION" -f images/broker.Dockerfile .
|
|
||||||
docker build -t "clawmates/agent-base:$VERSION" images/agent-base
|
|
||||||
docker build -t "clawmates/agent-browser:$VERSION" images/agent-browser
|
|
||||||
docker pull postgres:16-alpine
|
|
||||||
|
|
||||||
- name: SBOMs for every shipped image
|
|
||||||
run: |
|
|
||||||
mkdir -p dist/sboms
|
|
||||||
curl -sSfL https://raw.githubusercontent.com/anchore/syft/main/install.sh \
|
|
||||||
| sh -s -- -b /usr/local/bin
|
|
||||||
for image in server frontend broker agent-base agent-browser; do
|
|
||||||
syft "clawmates/$image:$VERSION" -o spdx-json \
|
|
||||||
> "dist/sboms/$image.spdx.json"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: Save image tarballs
|
|
||||||
run: |
|
|
||||||
mkdir -p dist/images
|
|
||||||
docker save "clawmates/server:$VERSION" -o dist/images/server.tar
|
|
||||||
docker save "clawmates/frontend:$VERSION" -o dist/images/frontend.tar
|
|
||||||
docker save "clawmates/broker:$VERSION" -o dist/images/broker.tar
|
|
||||||
docker pull tecnativa/docker-socket-proxy:0.3
|
|
||||||
docker save tecnativa/docker-socket-proxy:0.3 -o dist/images/socket-proxy.tar
|
|
||||||
docker save "clawmates/agent-base:$VERSION" -o dist/images/agent-base.tar
|
|
||||||
docker save "clawmates/agent-browser:$VERSION" -o dist/images/agent-browser.tar
|
|
||||||
docker save postgres:16-alpine -o dist/images/postgres.tar
|
|
||||||
|
|
||||||
- name: Build bundler
|
|
||||||
run: cargo build --release -p clawmates-bundler
|
|
||||||
|
|
||||||
- name: Assemble and sign the bundle
|
|
||||||
env:
|
|
||||||
BUNDLE_SIGNING_KEY: ${{ secrets.BUNDLE_SIGNING_KEY }}
|
|
||||||
run: |
|
|
||||||
printf '%s' "$BUNDLE_SIGNING_KEY" > /tmp/release.key
|
|
||||||
BUNDLER=target/release/clawmates-bundler
|
|
||||||
ARTIFACTS=""
|
|
||||||
for tar in dist/images/*.tar; do
|
|
||||||
ARTIFACTS="$ARTIFACTS $tar=images/$(basename "$tar")"
|
|
||||||
done
|
|
||||||
for migration in migrations/*.sql; do
|
|
||||||
ARTIFACTS="$ARTIFACTS $migration=migrations/$(basename "$migration")"
|
|
||||||
done
|
|
||||||
# shellcheck disable=SC2086
|
|
||||||
"$BUNDLER" assemble dist/bundle "$VERSION" /tmp/release.key \
|
|
||||||
deploy/compose/docker-compose.yml=compose/docker-compose.yml \
|
|
||||||
deploy/compose/clawmates.toml=compose/clawmates.toml \
|
|
||||||
deploy/compose/.env.example=compose/.env.example \
|
|
||||||
deploy/e2e/scenarios.toml=compose/scenarios.toml \
|
|
||||||
images/seccomp/agent-profile.json=seccomp/agent-profile.json \
|
|
||||||
deploy/airgapped/install.sh=install.sh \
|
|
||||||
"$BUNDLER"=bin/clawmates-bundler \
|
|
||||||
dist/sboms/server.spdx.json=sboms/server.spdx.json \
|
|
||||||
dist/sboms/frontend.spdx.json=sboms/frontend.spdx.json \
|
|
||||||
dist/sboms/agent-base.spdx.json=sboms/agent-base.spdx.json \
|
|
||||||
dist/sboms/agent-browser.spdx.json=sboms/agent-browser.spdx.json \
|
|
||||||
$ARTIFACTS
|
|
||||||
chmod +x dist/bundle/bin/clawmates-bundler dist/bundle/install.sh
|
|
||||||
rm /tmp/release.key
|
|
||||||
|
|
||||||
- name: Verify the bundle offline (public key only)
|
|
||||||
env:
|
|
||||||
BUNDLE_SIGNING_KEY: ${{ secrets.BUNDLE_SIGNING_KEY }}
|
|
||||||
run: |
|
|
||||||
printf '%s' "$BUNDLE_SIGNING_KEY" > /tmp/release.key
|
|
||||||
target/release/clawmates-bundler pubkey /tmp/release.key dist/release.pub
|
|
||||||
rm /tmp/release.key
|
|
||||||
# The customer's exact procedure: only the public half — and
|
|
||||||
# inside a NETWORK-DISABLED container, proving verification
|
|
||||||
# needs no internet (the air-gapped contract).
|
|
||||||
docker run --rm --network none \
|
|
||||||
-v "$PWD/dist:/dist:ro" \
|
|
||||||
ubuntu:24.04 \
|
|
||||||
/dist/bundle/bin/clawmates-bundler verify /dist/bundle /dist/release.pub
|
|
||||||
|
|
||||||
- name: Tarball
|
|
||||||
run: tar -C dist -czf "clawmates-bundle-$VERSION.tgz" bundle
|
|
||||||
|
|
||||||
- name: Clean-room install rehearsal
|
|
||||||
run: |
|
|
||||||
docker tag "clawmates/server:$VERSION" clawmates/server:latest
|
|
||||||
docker tag "clawmates/frontend:$VERSION" clawmates/frontend:latest
|
|
||||||
docker tag "clawmates/broker:$VERSION" clawmates/broker:latest
|
|
||||||
./scripts/rehearse-install.sh
|
|
||||||
|
|
||||||
- name: Attach to release
|
|
||||||
uses: softprops/action-gh-release@v2
|
|
||||||
with:
|
|
||||||
files: |
|
|
||||||
clawmates-bundle-*.tgz
|
|
||||||
dist/release.pub
|
|
||||||
+17
@@ -14,3 +14,20 @@ token.key
|
|||||||
|
|
||||||
# Hosted node-agent binaries (built + baked into the frontend image, not committed)
|
# Hosted node-agent binaries (built + baked into the frontend image, not committed)
|
||||||
frontend/public/dl/
|
frontend/public/dl/
|
||||||
|
|
||||||
|
# Local env backups. `.env` is already ignored above, but a timestamped or
|
||||||
|
# suffixed copy of it is not — and these hold real credentials (subscription
|
||||||
|
# OAuth token, forge PAT, DB password). Ignore every variant, not just the
|
||||||
|
# exact name.
|
||||||
|
.env.bak*
|
||||||
|
*.env.bak*
|
||||||
|
deploy/compose/.env.*
|
||||||
|
|
||||||
|
# Local-only compose override. NOT for prod or the air-gapped install: it
|
||||||
|
# rebinds published ports to loopback, enables the login bypass, and points the
|
||||||
|
# runtime at MacBook-specific paths. docker-compose picks this file up
|
||||||
|
# automatically, so committing it would silently reconfigure anyone who runs
|
||||||
|
# deploy/compose.
|
||||||
|
# deploy/compose/docker-compose.override.yml is TRACKED as of 2026-09-18: it
|
||||||
|
# holds the fixes for the five local bring-up gaps and every credential in it is
|
||||||
|
# a ${VAR:?} reference into .env. It lived only on one laptop until then.
|
||||||
|
|||||||
+17
@@ -0,0 +1,17 @@
|
|||||||
|
{
|
||||||
|
"db_name": "PostgreSQL",
|
||||||
|
"query": "INSERT INTO auth_sessions (token_hash, user_id, expires_at, scope)\n VALUES ($1, $2, $3, $4)",
|
||||||
|
"describe": {
|
||||||
|
"columns": [],
|
||||||
|
"parameters": {
|
||||||
|
"Left": [
|
||||||
|
"Text",
|
||||||
|
"Uuid",
|
||||||
|
"Timestamptz",
|
||||||
|
"Text"
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"nullable": []
|
||||||
|
},
|
||||||
|
"hash": "105f8cc147247c69b3c45e2e3eb27fc33b1976accdda66ec3ccc7c57afecc8b9"
|
||||||
|
}
|
||||||
-16
@@ -1,16 +0,0 @@
|
|||||||
{
|
|
||||||
"db_name": "PostgreSQL",
|
|
||||||
"query": "UPDATE topology_runs\n SET checkpoint = $2, last_event_id = $3, updated_at = now()\n WHERE id = $1",
|
|
||||||
"describe": {
|
|
||||||
"columns": [],
|
|
||||||
"parameters": {
|
|
||||||
"Left": [
|
|
||||||
"Uuid",
|
|
||||||
"Jsonb",
|
|
||||||
"Int8"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"nullable": []
|
|
||||||
},
|
|
||||||
"hash": "5fcbd4d6adbf02489051e2fa63d1df670863bf90e55c1c0ac0ab011759cbd272"
|
|
||||||
}
|
|
||||||
-14
@@ -1,14 +0,0 @@
|
|||||||
{
|
|
||||||
"db_name": "PostgreSQL",
|
|
||||||
"query": "UPDATE topology_runs\n SET status = 'queued', updated_at = now()\n WHERE status = 'running' AND updated_at < now() - make_interval(secs => $1)",
|
|
||||||
"describe": {
|
|
||||||
"columns": [],
|
|
||||||
"parameters": {
|
|
||||||
"Left": [
|
|
||||||
"Float8"
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"nullable": []
|
|
||||||
},
|
|
||||||
"hash": "7298995b5b58aed46888bb9e5c8d331483aee162fc6bcf1e53232d2afc7c3e62"
|
|
||||||
}
|
|
||||||
-56
@@ -1,56 +0,0 @@
|
|||||||
{
|
|
||||||
"db_name": "PostgreSQL",
|
|
||||||
"query": "UPDATE topology_runs\n SET status = 'running', started_at = COALESCE(started_at, now()), updated_at = now()\n WHERE id = (\n SELECT id FROM topology_runs\n WHERE status = 'queued'\n ORDER BY created_at\n FOR UPDATE SKIP LOCKED\n LIMIT 1\n )\n RETURNING id, workspace_id, task, graph, checkpoint, last_event_id, tier",
|
|
||||||
"describe": {
|
|
||||||
"columns": [
|
|
||||||
{
|
|
||||||
"ordinal": 0,
|
|
||||||
"name": "id",
|
|
||||||
"type_info": "Uuid"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"ordinal": 1,
|
|
||||||
"name": "workspace_id",
|
|
||||||
"type_info": "Uuid"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"ordinal": 2,
|
|
||||||
"name": "task",
|
|
||||||
"type_info": "Text"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"ordinal": 3,
|
|
||||||
"name": "graph",
|
|
||||||
"type_info": "Jsonb"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"ordinal": 4,
|
|
||||||
"name": "checkpoint",
|
|
||||||
"type_info": "Jsonb"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"ordinal": 5,
|
|
||||||
"name": "last_event_id",
|
|
||||||
"type_info": "Int8"
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"ordinal": 6,
|
|
||||||
"name": "tier",
|
|
||||||
"type_info": "Text"
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"parameters": {
|
|
||||||
"Left": []
|
|
||||||
},
|
|
||||||
"nullable": [
|
|
||||||
false,
|
|
||||||
false,
|
|
||||||
false,
|
|
||||||
true,
|
|
||||||
true,
|
|
||||||
false,
|
|
||||||
false
|
|
||||||
]
|
|
||||||
},
|
|
||||||
"hash": "9eae6ca16ffc9346456128ce676ef04f3478f873d6ac5f95154b797f454f44c0"
|
|
||||||
}
|
|
||||||
+2
-2
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"db_name": "PostgreSQL",
|
"db_name": "PostgreSQL",
|
||||||
"query": "SELECT a.id, a.name, a.accent,\n COALESCE(SUM(u.credits), 0)::BIGINT AS \"credits!\",\n COALESCE(SUM(u.tokens_in + u.tokens_out), 0)::BIGINT AS \"tokens!\",\n COUNT(u.id)::BIGINT AS \"runs!\"\n FROM agents a\n LEFT JOIN usage_events u ON u.agent_id = a.id\n WHERE a.workspace_id = $1\n GROUP BY a.id, a.name, a.accent\n ORDER BY \"credits!\" DESC, \"tokens!\" DESC, a.name",
|
"query": "SELECT a.id, a.name, a.accent,\n COALESCE(SUM(u.credits), 0)::BIGINT AS \"credits!\",\n COALESCE(SUM(u.tokens_in + u.tokens_out), 0)::BIGINT AS \"tokens!\",\n COUNT(u.id)::BIGINT AS \"runs!\"\n FROM agents a\n LEFT JOIN usage_events u ON u.agent_id = a.id\n -- deleted_at: a soft-deleted agent is gone everywhere else, so\n -- listing it here made deletion look like a no-op — the operator\n -- deletes it, the board still shows it, and deleting again does\n -- nothing because the row is already marked.\n WHERE a.workspace_id = $1 AND a.deleted_at IS NULL\n GROUP BY a.id, a.name, a.accent\n ORDER BY \"credits!\" DESC, \"tokens!\" DESC, a.name",
|
||||||
"describe": {
|
"describe": {
|
||||||
"columns": [
|
"columns": [
|
||||||
{
|
{
|
||||||
@@ -48,5 +48,5 @@
|
|||||||
null
|
null
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"hash": "d4ef449c48b15519b7195be637dca3d456140ce477993d25a87e209174f79aba"
|
"hash": "d5bc028ca030daed4e6111990945d8d7011414d830f7d6d0a04980efb79af2a6"
|
||||||
}
|
}
|
||||||
+8
-2
@@ -1,6 +1,6 @@
|
|||||||
{
|
{
|
||||||
"db_name": "PostgreSQL",
|
"db_name": "PostgreSQL",
|
||||||
"query": "SELECT u.id, u.workspace_id, u.role\n FROM auth_sessions s\n JOIN users u ON u.id = s.user_id\n WHERE s.token_hash = $1 AND s.expires_at > now()",
|
"query": "SELECT u.id, u.workspace_id, u.role, s.scope\n FROM auth_sessions s\n JOIN users u ON u.id = s.user_id\n WHERE s.token_hash = $1 AND s.expires_at > now()",
|
||||||
"describe": {
|
"describe": {
|
||||||
"columns": [
|
"columns": [
|
||||||
{
|
{
|
||||||
@@ -17,6 +17,11 @@
|
|||||||
"ordinal": 2,
|
"ordinal": 2,
|
||||||
"name": "role",
|
"name": "role",
|
||||||
"type_info": "Text"
|
"type_info": "Text"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"ordinal": 3,
|
||||||
|
"name": "scope",
|
||||||
|
"type_info": "Text"
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"parameters": {
|
"parameters": {
|
||||||
@@ -25,10 +30,11 @@
|
|||||||
]
|
]
|
||||||
},
|
},
|
||||||
"nullable": [
|
"nullable": [
|
||||||
|
false,
|
||||||
false,
|
false,
|
||||||
false,
|
false,
|
||||||
false
|
false
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"hash": "900827c5c8c24f4861120e98e3cc8a5b70f22e9f4b4168c9e8eb51c53d68bdae"
|
"hash": "e8f7cb9c34be37fe16c5406e9263159693674dda567b60a1f87c6763ec448951"
|
||||||
}
|
}
|
||||||
Generated
+44
@@ -846,6 +846,8 @@ dependencies = [
|
|||||||
"serde",
|
"serde",
|
||||||
"serde_json",
|
"serde_json",
|
||||||
"sysinfo",
|
"sysinfo",
|
||||||
|
"tar",
|
||||||
|
"tempfile",
|
||||||
"tokio",
|
"tokio",
|
||||||
"tokio-tungstenite 0.26.2",
|
"tokio-tungstenite 0.26.2",
|
||||||
"webrtc",
|
"webrtc",
|
||||||
@@ -1842,6 +1844,16 @@ version = "2.4.1"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6"
|
checksum = "9f1f227452a390804cdb637b74a86990f2a7d7ba4b7d5693aac9b4dd6defd8d6"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "fcagent"
|
||||||
|
version = "0.1.0"
|
||||||
|
dependencies = [
|
||||||
|
"base64",
|
||||||
|
"serde_json",
|
||||||
|
"tar",
|
||||||
|
"vsock",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "ff"
|
name = "ff"
|
||||||
version = "0.13.1"
|
version = "0.13.1"
|
||||||
@@ -2873,6 +2885,15 @@ dependencies = [
|
|||||||
"autocfg",
|
"autocfg",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "memoffset"
|
||||||
|
version = "0.9.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "488016bfae457b036d996092f6cb448677611ce4449e970ceaf42695203f218a"
|
||||||
|
dependencies = [
|
||||||
|
"autocfg",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "mime"
|
name = "mime"
|
||||||
version = "0.3.17"
|
version = "0.3.17"
|
||||||
@@ -2963,6 +2984,19 @@ dependencies = [
|
|||||||
"pin-utils",
|
"pin-utils",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "nix"
|
||||||
|
version = "0.31.3"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d"
|
||||||
|
dependencies = [
|
||||||
|
"bitflags 2.13.0",
|
||||||
|
"cfg-if",
|
||||||
|
"cfg_aliases",
|
||||||
|
"libc",
|
||||||
|
"memoffset 0.9.1",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "nom"
|
name = "nom"
|
||||||
version = "7.1.3"
|
version = "7.1.3"
|
||||||
@@ -5761,6 +5795,16 @@ version = "0.9.5"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "vsock"
|
||||||
|
version = "0.5.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "6ba782755fc073877e567c2253c0be48e4aa9a254c232d36d3985dfae0bd5205"
|
||||||
|
dependencies = [
|
||||||
|
"libc",
|
||||||
|
"nix 0.31.3",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "wait-timeout"
|
name = "wait-timeout"
|
||||||
version = "0.2.1"
|
version = "0.2.1"
|
||||||
|
|||||||
@@ -23,6 +23,7 @@ members = [
|
|||||||
"crates/bins/clawmates-server",
|
"crates/bins/clawmates-server",
|
||||||
"crates/bins/clawmates-broker",
|
"crates/bins/clawmates-broker",
|
||||||
"crates/bins/clawmates-node",
|
"crates/bins/clawmates-node",
|
||||||
|
"crates/bins/fcagent",
|
||||||
"tools/bundler",
|
"tools/bundler",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
|||||||
@@ -19,6 +19,7 @@ serde_json = { workspace = true }
|
|||||||
sysinfo = "0.33"
|
sysinfo = "0.33"
|
||||||
portable-pty = "0.8"
|
portable-pty = "0.8"
|
||||||
base64 = "0.22"
|
base64 = "0.22"
|
||||||
|
tar = { workspace = true }
|
||||||
cm-sandbox = { path = "../../cm-sandbox" }
|
cm-sandbox = { path = "../../cm-sandbox" }
|
||||||
# Linking cm-sandbox (bollard) brings a second rustls provider into the graph, so
|
# Linking cm-sandbox (bollard) brings a second rustls provider into the graph, so
|
||||||
# rustls can't auto-pick one — we install `ring` explicitly at startup.
|
# rustls can't auto-pick one — we install `ring` explicitly at startup.
|
||||||
@@ -26,5 +27,8 @@ rustls = { version = "0.23", default-features = false, features = ["ring"] }
|
|||||||
webrtc = "0.17.1"
|
webrtc = "0.17.1"
|
||||||
bytes = "1.12.0"
|
bytes = "1.12.0"
|
||||||
|
|
||||||
|
[dev-dependencies]
|
||||||
|
tempfile = "3"
|
||||||
|
|
||||||
[lints]
|
[lints]
|
||||||
workspace = true
|
workspace = true
|
||||||
|
|||||||
@@ -0,0 +1,527 @@
|
|||||||
|
//! Host side of a microVM's only route out: an HTTP `CONNECT` proxy on a Unix
|
||||||
|
//! socket, one per VM.
|
||||||
|
//!
|
||||||
|
//! # Why the guest has no network card
|
||||||
|
//!
|
||||||
|
//! It could have had one. A TAP device plus NAT is what the Firecracker
|
||||||
|
//! write-ups do, and it was measured against this before being rejected:
|
||||||
|
//!
|
||||||
|
//! - `ip tuntap add` is **denied to the daemon user** (needs `CAP_NET_ADMIN`), so
|
||||||
|
//! TAP would need root to pre-provision devices at setup time — the same
|
||||||
|
//! privilege detour the loop-mounted rootfs already forced.
|
||||||
|
//! - tank's `FORWARD` policy is `DROP` with Docker and Tailscale chains, so rules
|
||||||
|
//! would have to be *inserted* at position 1; appended ones die silently.
|
||||||
|
//! - a leaked TAP device is a new class of host litter to reap.
|
||||||
|
//!
|
||||||
|
//! Against that, `CONNECT` needs no privilege at all, and it is better on the
|
||||||
|
//! merits: the client hands us the **hostname**, so resolution happens here and
|
||||||
|
//! the guest needs no DNS or `resolv.conf`; the allow-list is by name rather than
|
||||||
|
//! by address; and nothing in the guest can reach the network except through this
|
||||||
|
//! function. That is what the isolation plan's egress restriction actually asked
|
||||||
|
//! for, and it is strictly tighter than the mission container's present full
|
||||||
|
//! egress on `clawmates_edge`.
|
||||||
|
//!
|
||||||
|
//! The design rests on one measured fact: **`claude` honours `HTTPS_PROXY`**.
|
||||||
|
//! With the proxy pointed at a closed port, `claude -p` fails with
|
||||||
|
//! `ConnectionRefused` instead of answering. (That could only be measured in a
|
||||||
|
//! container — inside a VM the CLI collapses every failure into `Execution
|
||||||
|
//! error`.)
|
||||||
|
//!
|
||||||
|
//! # Shape
|
||||||
|
//!
|
||||||
|
//! Firecracker's convention for a guest-initiated connection is that the **host**
|
||||||
|
//! listens on `<uds_path>_<port>`. The guest's agent pumps bytes from
|
||||||
|
//! `127.0.0.1:3128` to vsock port 9002 and parses nothing, so all policy is here
|
||||||
|
//! and a compromised guest cannot argue with it.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use tokio::io::{AsyncReadExt, AsyncWriteExt, BufReader};
|
||||||
|
use tokio::net::{TcpStream, UnixListener, UnixStream};
|
||||||
|
|
||||||
|
/// Port the guest dials. Must match `fcagent`'s `EGRESS_PORT`.
|
||||||
|
pub const EGRESS_PORT: u32 = 9002;
|
||||||
|
|
||||||
|
|
||||||
|
/// What every backend gets, whatever it is.
|
||||||
|
const COMMON_ALLOW: &[&str] = &["git.redclaw.dev"];
|
||||||
|
|
||||||
|
/// The model host a backend's CLI must reach, and NOTHING else.
|
||||||
|
///
|
||||||
|
/// Per backend rather than a union, and that is not tidiness. MEASURED on tank:
|
||||||
|
/// a `glm` VM completed a whole mission with `api.anthropic.com` denied at this
|
||||||
|
/// proxy, dialling only `api.z.ai` — Claude Code's calls to anthropic.com are
|
||||||
|
/// its own telemetry, not its completions. So a GLM VM has no need of Anthropic
|
||||||
|
/// at all, and a union allow-list would let a credential mix-up reach the wrong
|
||||||
|
/// provider's endpoint instead of failing at a closed door.
|
||||||
|
///
|
||||||
|
/// The measurement also settled something a self-report could not: that same
|
||||||
|
/// agent, served only by z.ai, still described itself as "Claude Opus 5". A
|
||||||
|
/// model's account of which model it is has no evidential value here; the
|
||||||
|
/// proxy's log of which host it dialled does.
|
||||||
|
fn provider_hosts(backend: Option<&str>) -> &'static [&'static str] {
|
||||||
|
match backend {
|
||||||
|
// `canary-claude` is the same provider, from a candidate CLI image —
|
||||||
|
// see `mission_runtime::microvm_credential_for`, which must grant it the
|
||||||
|
// same credential. A backend is defined in TWO maps: the credential one
|
||||||
|
// on the server and this one on the node. Adding it to only the first is
|
||||||
|
// exactly what happened here: the mission launched, the VM booted, the
|
||||||
|
// agent ran, and the turn died on
|
||||||
|
// "403 api.anthropic.com is not on the egress allow-list" — which is the
|
||||||
|
// fail-closed branch below working correctly.
|
||||||
|
None | Some("") | Some("default") | Some("claude") | Some("canary-claude") => {
|
||||||
|
&["api.anthropic.com", ".anthropic.com"]
|
||||||
|
}
|
||||||
|
Some("glm") => &["api.z.ai"],
|
||||||
|
// The Kimi CODE service, which is where an `sk-kimi-` key is valid —
|
||||||
|
// NOT `api.moonshot.ai`, whose Anthropic endpoint exists but belongs to
|
||||||
|
// a different account namespace and rejects that key. Only the host the
|
||||||
|
// `agent-kimi` image bakes in.
|
||||||
|
Some("kimi") => &["api.kimi.com"],
|
||||||
|
// A locally-hosted model reaches NOTHING through this proxy. Its route
|
||||||
|
// is `crate::local_model` — a vsock pipe to the node's own loopback,
|
||||||
|
// with no destination in the protocol — so the correct allow-list here
|
||||||
|
// is the empty one, and it falls through to the branch below.
|
||||||
|
//
|
||||||
|
// Spelled out rather than left implicit because the temptation was to
|
||||||
|
// widen this proxy instead: an entry here would have meant relaxing the
|
||||||
|
// 443-only rule AND the IP-literal refusal, both of which exist because
|
||||||
|
// a unit test caught them being bypassed.
|
||||||
|
// Fail closed: a backend nobody taught this function about reaches the
|
||||||
|
// forge and no model API. It cannot silently borrow another provider's
|
||||||
|
// door, which is the failure this split exists to prevent.
|
||||||
|
Some(_) => &[],
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse the allow-list once per VM.
|
||||||
|
///
|
||||||
|
/// An empty `CLAWMATES_FC_EGRESS_ALLOW` means **deny everything**, not "fall back
|
||||||
|
/// to the default": an operator who blanked it asked for no egress, and quietly
|
||||||
|
/// restoring the default would hand a mission the network they just took away.
|
||||||
|
/// The allow-list for a VM running `backend`.
|
||||||
|
///
|
||||||
|
/// An explicit `CLAWMATES_FC_EGRESS_ALLOW` still wins outright: an operator who
|
||||||
|
/// set it asked for exactly that list, and quietly adding a provider host to it
|
||||||
|
/// would widen a boundary they had drawn on purpose.
|
||||||
|
fn allow_list_for(backend: Option<&str>) -> Vec<String> {
|
||||||
|
match std::env::var("CLAWMATES_FC_EGRESS_ALLOW") {
|
||||||
|
Ok(raw) => raw
|
||||||
|
.split(',')
|
||||||
|
.map(|s| s.trim().to_ascii_lowercase())
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
.collect(),
|
||||||
|
Err(_) => COMMON_ALLOW
|
||||||
|
.iter()
|
||||||
|
.chain(provider_hosts(backend).iter())
|
||||||
|
.map(|s| s.to_string())
|
||||||
|
.collect(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Is `host` allowed?
|
||||||
|
///
|
||||||
|
/// Case-insensitive, port already stripped. A leading `.` in an entry matches
|
||||||
|
/// that domain and its subdomains; anything else must match exactly. Deliberately
|
||||||
|
/// not a substring test — `api.anthropic.com.evil.test` contains the allowed name
|
||||||
|
/// and must not pass.
|
||||||
|
fn host_allowed(host: &str, allow: &[String]) -> bool {
|
||||||
|
let host = host.trim().trim_end_matches('.').to_ascii_lowercase();
|
||||||
|
if host.is_empty() {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
// A hostname is letters, digits, dots and hyphens — nothing else. This is
|
||||||
|
// load-bearing, not hygiene: `evil.test/api.anthropic.com` ends with an
|
||||||
|
// allowed suffix and would otherwise PASS the match below. A unit test found
|
||||||
|
// it. Rejecting the character class also refuses IP literals, so an address
|
||||||
|
// cannot be used to sidestep a list written in names.
|
||||||
|
if !host
|
||||||
|
.chars()
|
||||||
|
.all(|c| c.is_ascii_alphanumeric() || c == '.' || c == '-')
|
||||||
|
{
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
allow.iter().any(|a| match a.strip_prefix('.') {
|
||||||
|
Some(domain) => host == domain || host.ends_with(&format!(".{domain}")),
|
||||||
|
None => host == *a,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Split `host:port` from a CONNECT target.
|
||||||
|
///
|
||||||
|
/// Only 443 is allowed. Permitting arbitrary ports would turn the proxy into a
|
||||||
|
/// general-purpose tunnel to anything the allow-list happens to name, which is a
|
||||||
|
/// different and much larger promise than "the agent can reach its API".
|
||||||
|
fn parse_target(target: &str) -> Result<(String, u16), String> {
|
||||||
|
let (host, port) = target
|
||||||
|
.rsplit_once(':')
|
||||||
|
.ok_or_else(|| format!("CONNECT target {target:?} has no port"))?;
|
||||||
|
let port: u16 = port
|
||||||
|
.trim()
|
||||||
|
.parse()
|
||||||
|
.map_err(|_| format!("CONNECT target {target:?} has a non-numeric port"))?;
|
||||||
|
if port != 443 {
|
||||||
|
return Err(format!("port {port} is not permitted (only 443)"));
|
||||||
|
}
|
||||||
|
// Strip IPv6 brackets so the allow-list sees the same text either way.
|
||||||
|
let host = host.trim().trim_start_matches('[').trim_end_matches(']');
|
||||||
|
Ok((host.to_string(), port))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What happened to one connection. Returned so the caller can log it and the
|
||||||
|
/// selftest can assert on it.
|
||||||
|
#[derive(Debug, PartialEq, Eq)]
|
||||||
|
pub enum Verdict {
|
||||||
|
Allowed(String),
|
||||||
|
Denied(String),
|
||||||
|
Malformed(String),
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One header line, with a cap.
|
||||||
|
///
|
||||||
|
/// `read_line` has no limit, and a guest that never sends a newline would make
|
||||||
|
/// the host allocate until it died. Read byte-wise instead — the reads come out
|
||||||
|
/// of the BufReader, so this is cheap for lines this size, and it keeps ONE
|
||||||
|
/// reader over the connection, which matters (see `serve`).
|
||||||
|
async fn read_line_capped(reader: &mut BufReader<UnixStream>, cap: usize) -> Result<String, String> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
loop {
|
||||||
|
match reader.read_u8().await {
|
||||||
|
Ok(b'\n') => break,
|
||||||
|
Ok(b) => out.push(b),
|
||||||
|
// EOF mid-line: return what we have and let the caller judge it.
|
||||||
|
Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => break,
|
||||||
|
Err(e) => return Err(format!("read: {e}")),
|
||||||
|
}
|
||||||
|
if out.len() > cap {
|
||||||
|
return Err(format!("a request line longer than {cap} bytes"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(String::from_utf8_lossy(&out)
|
||||||
|
.trim_end_matches('\r')
|
||||||
|
.to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Serve one tunnelled connection.
|
||||||
|
async fn serve(stream: UnixStream, allow: Arc<Vec<String>>) -> Verdict {
|
||||||
|
// ONE reader for the whole request. Wrapping the stream a second time would
|
||||||
|
// discard whatever the first reader had already buffered — including the
|
||||||
|
// first bytes of the TLS handshake — and the tunnel would come up looking
|
||||||
|
// fine and then stall on a corrupt stream.
|
||||||
|
let mut reader = BufReader::new(stream);
|
||||||
|
|
||||||
|
let line = match read_line_capped(&mut reader, 8 * 1024).await {
|
||||||
|
Ok(l) if !l.trim().is_empty() => l,
|
||||||
|
Ok(_) => return Verdict::Malformed("no request line".into()),
|
||||||
|
Err(e) => return Verdict::Malformed(e),
|
||||||
|
};
|
||||||
|
|
||||||
|
let mut parts = line.split_whitespace();
|
||||||
|
let method = parts.next().unwrap_or_default().to_ascii_uppercase();
|
||||||
|
let target = parts.next().unwrap_or_default().to_string();
|
||||||
|
|
||||||
|
if method != "CONNECT" {
|
||||||
|
// Plain HTTP would mean proxying a request we would then have to rewrite,
|
||||||
|
// and everything a mission needs is TLS. Refused with a status, so the
|
||||||
|
// client reports something better than a closed socket.
|
||||||
|
let _ = reply(reader.get_mut(), 405, "only CONNECT is supported").await;
|
||||||
|
return Verdict::Malformed(format!("method {method}"));
|
||||||
|
}
|
||||||
|
|
||||||
|
let (host, port) = match parse_target(&target) {
|
||||||
|
Ok(v) => v,
|
||||||
|
Err(e) => {
|
||||||
|
let _ = reply(reader.get_mut(), 400, &e).await;
|
||||||
|
return Verdict::Malformed(e);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
if !host_allowed(&host, &allow) {
|
||||||
|
// 403 rather than a silent drop: a denial that looks like a network
|
||||||
|
// timeout is indistinguishable from a hung agent, and this codebase has
|
||||||
|
// paid for that confusion more than once.
|
||||||
|
let _ = reply(
|
||||||
|
reader.get_mut(),
|
||||||
|
403,
|
||||||
|
&format!("{host} is not on the egress allow-list"),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
return Verdict::Denied(host);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Consume the remaining request headers: they belong to the CONNECT, not to
|
||||||
|
// the tunnel.
|
||||||
|
loop {
|
||||||
|
match read_line_capped(&mut reader, 8 * 1024).await {
|
||||||
|
Ok(h) if h.trim().is_empty() => break,
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => return Verdict::Malformed(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut upstream = match TcpStream::connect((host.as_str(), port)).await {
|
||||||
|
Ok(s) => s,
|
||||||
|
Err(e) => {
|
||||||
|
let _ = reply(reader.get_mut(), 502, &format!("connect {host}:{port}: {e}")).await;
|
||||||
|
return Verdict::Denied(host);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
if reply(reader.get_mut(), 200, "Connection established")
|
||||||
|
.await
|
||||||
|
.is_err()
|
||||||
|
{
|
||||||
|
return Verdict::Denied(host);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Anything already buffered past the headers is tunnel payload — a client
|
||||||
|
// that pipelined its first TLS bytes would otherwise lose them.
|
||||||
|
let pending = reader.buffer().to_vec();
|
||||||
|
let mut stream = reader.into_inner();
|
||||||
|
if !pending.is_empty() && upstream.write_all(&pending).await.is_err() {
|
||||||
|
return Verdict::Denied(host);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Bytes both ways until either side is done. Errors are not worth reporting:
|
||||||
|
// a closed connection is the normal end of a tunnel.
|
||||||
|
let _ = tokio::io::copy_bidirectional(&mut stream, &mut upstream).await;
|
||||||
|
Verdict::Allowed(host)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn reply(s: &mut UnixStream, code: u16, text: &str) -> std::io::Result<()> {
|
||||||
|
let reason = if code == 200 {
|
||||||
|
"Connection established"
|
||||||
|
} else {
|
||||||
|
"Forbidden"
|
||||||
|
};
|
||||||
|
// The body carries the reason for a non-200 so it reaches the agent's own
|
||||||
|
// error output, where whoever is reading a failed mission will see it.
|
||||||
|
let body = if code == 200 { String::new() } else { format!("{text}\n") };
|
||||||
|
let head = format!(
|
||||||
|
"HTTP/1.1 {code} {reason}\r\nContent-Length: {}\r\nConnection: close\r\n\r\n",
|
||||||
|
body.len()
|
||||||
|
);
|
||||||
|
s.write_all(head.as_bytes()).await?;
|
||||||
|
if !body.is_empty() {
|
||||||
|
s.write_all(body.as_bytes()).await?;
|
||||||
|
}
|
||||||
|
s.flush().await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Start this VM's proxy. Returns the socket path and the task serving it.
|
||||||
|
///
|
||||||
|
/// Bound **before** firecracker starts, because a guest that dials before the
|
||||||
|
/// host is listening gets a connection refused it will not retry.
|
||||||
|
pub fn start(
|
||||||
|
uds: &Path,
|
||||||
|
vm_id: &str,
|
||||||
|
backend: Option<&str>,
|
||||||
|
) -> Result<(PathBuf, tokio::task::JoinHandle<()>), String> {
|
||||||
|
let path = PathBuf::from(format!("{}_{}", uds.display(), EGRESS_PORT));
|
||||||
|
// Firecracker does not clean these up any more than it cleans up its own
|
||||||
|
// socket, and a stale file makes bind fail with EADDRINUSE.
|
||||||
|
let _ = std::fs::remove_file(&path);
|
||||||
|
let listener =
|
||||||
|
UnixListener::bind(&path).map_err(|e| format!("bind {}: {e}", path.display()))?;
|
||||||
|
|
||||||
|
let allow = Arc::new(allow_list_for(backend));
|
||||||
|
eprintln!(
|
||||||
|
"microvm {vm_id}: egress proxy on {} allowing {:?}",
|
||||||
|
path.display(),
|
||||||
|
allow
|
||||||
|
);
|
||||||
|
let vm = vm_id.to_string();
|
||||||
|
let task = tokio::spawn(async move {
|
||||||
|
loop {
|
||||||
|
match listener.accept().await {
|
||||||
|
Ok((s, _)) => {
|
||||||
|
let allow = allow.clone();
|
||||||
|
let vm = vm.clone();
|
||||||
|
tokio::spawn(async move {
|
||||||
|
match serve(s, allow).await {
|
||||||
|
// Logged at every outcome: this is the audit trail of
|
||||||
|
// everything a mission reached, and a denial that is
|
||||||
|
// not logged is a mystery hang later.
|
||||||
|
Verdict::Allowed(h) => eprintln!("microvm {vm}: egress -> {h}"),
|
||||||
|
Verdict::Denied(h) => {
|
||||||
|
eprintln!("microvm {vm}: egress DENIED {h}")
|
||||||
|
}
|
||||||
|
Verdict::Malformed(w) => {
|
||||||
|
eprintln!("microvm {vm}: egress malformed request ({w})")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("microvm {vm}: egress accept failed: {e}");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
Ok((path, task))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
/// A backend is defined in TWO places — the server's credential map and this
|
||||||
|
/// egress map — and granting it one without the other produces a mission
|
||||||
|
/// that launches, boots, runs, and dies on a 403 from our own proxy.
|
||||||
|
///
|
||||||
|
/// Measured exactly that way: `canary-claude` was credentialed on the server
|
||||||
|
/// and unknown here, and the turn failed with
|
||||||
|
/// "api.anthropic.com is not on the egress allow-list".
|
||||||
|
#[test]
|
||||||
|
fn the_canary_backend_reaches_the_same_provider_as_claude() {
|
||||||
|
assert_eq!(
|
||||||
|
provider_hosts(Some("canary-claude")),
|
||||||
|
provider_hosts(Some("claude")),
|
||||||
|
"a canary of the Claude image must reach Anthropic, or it tests nothing"
|
||||||
|
);
|
||||||
|
// And the fail-closed branch must still hold for anything unknown: this
|
||||||
|
// is what stops a new backend silently borrowing another provider's door.
|
||||||
|
assert!(provider_hosts(Some("canary-something-else")).is_empty());
|
||||||
|
assert!(provider_hosts(Some("definitely-not-built")).is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn allow() -> Vec<String> {
|
||||||
|
allow_list_for(None)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// MEASURED on tank, not assumed: a `glm` VM ran a whole mission to
|
||||||
|
/// completion with `api.anthropic.com` denied at this proxy, dialling only
|
||||||
|
/// `api.z.ai`. So Anthropic's host is not something a GLM agent needs — and
|
||||||
|
/// a VM that cannot reach it cannot send z.ai's key there, or Anthropic's
|
||||||
|
/// subscription token to z.ai, whatever a credential bug does upstream.
|
||||||
|
#[test]
|
||||||
|
fn each_backend_reaches_its_own_provider_and_no_other() {
|
||||||
|
let claude = allow_list_for(Some("claude"));
|
||||||
|
assert!(claude.iter().any(|h| h == "api.anthropic.com"), "{claude:?}");
|
||||||
|
assert!(!claude.iter().any(|h| h == "api.z.ai"), "{claude:?}");
|
||||||
|
|
||||||
|
let glm = allow_list_for(Some("glm"));
|
||||||
|
assert!(glm.iter().any(|h| h == "api.z.ai"), "{glm:?}");
|
||||||
|
assert!(
|
||||||
|
!glm.iter().any(|h| h.contains("anthropic")),
|
||||||
|
"a GLM VM must not be able to reach Anthropic: {glm:?}"
|
||||||
|
);
|
||||||
|
|
||||||
|
// Both still reach the forge — delivery is host-side, but a mission that
|
||||||
|
// clones or fetches needs it.
|
||||||
|
for l in [&claude, &glm] {
|
||||||
|
assert!(l.iter().any(|h| h == "git.redclaw.dev"), "{l:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
let kimi = allow_list_for(Some("kimi"));
|
||||||
|
assert!(kimi.iter().any(|h| h == "api.kimi.com"), "{kimi:?}");
|
||||||
|
for other in ["api.z.ai", "api.anthropic.com"] {
|
||||||
|
assert!(!kimi.iter().any(|h| h == other), "{kimi:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
// An unknown backend gets no model API at all rather than borrowing
|
||||||
|
// somebody's: it cannot run anyway, and failing at a closed door beats
|
||||||
|
// reaching the wrong endpoint with a credential.
|
||||||
|
let unknown = allow_list_for(Some("rootfs-opus"));
|
||||||
|
assert_eq!(unknown, vec!["git.redclaw.dev".to_string()], "{unknown:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_model_api_and_the_forge_are_reachable() {
|
||||||
|
for h in ["api.anthropic.com", "git.redclaw.dev", "API.Anthropic.COM"] {
|
||||||
|
assert!(host_allowed(h, &allow()), "{h} must be allowed");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The check is a match, never a substring test. A name that merely CONTAINS
|
||||||
|
/// an allowed one is a different host controlled by someone else.
|
||||||
|
#[test]
|
||||||
|
fn a_lookalike_host_is_not_allowed() {
|
||||||
|
for h in [
|
||||||
|
"api.anthropic.com.evil.test",
|
||||||
|
"notapi.anthropic.com.attacker.io",
|
||||||
|
// These contain an allowed suffix but are not that host. The first
|
||||||
|
// PASSED before the character-class check was added — a unit test
|
||||||
|
// found it, not review.
|
||||||
|
"evil.test/api.anthropic.com",
|
||||||
|
"[email protected]",
|
||||||
|
"api.anthropic.com:443",
|
||||||
|
"git.redclaw.dev.evil.test",
|
||||||
|
"example.com",
|
||||||
|
"",
|
||||||
|
" ",
|
||||||
|
] {
|
||||||
|
assert!(!host_allowed(h, &allow()), "{h} must NOT be allowed");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A raw address must not sidestep a list written in names.
|
||||||
|
/// A local-model backend gets NO egress, and the 443 rule is untouched.
|
||||||
|
///
|
||||||
|
/// The alternative design routed the node's Ollama through this proxy, which
|
||||||
|
/// would have meant permitting port 11434 and an address the guest names.
|
||||||
|
/// Both are refused here, still, and a `local-ornith` VM reaches the forge
|
||||||
|
/// and nothing else — its model lives on the other socket entirely.
|
||||||
|
#[test]
|
||||||
|
fn a_local_model_backend_gets_no_egress_and_no_new_port() {
|
||||||
|
let allow = allow_list_for(Some("local-ornith"));
|
||||||
|
assert!(
|
||||||
|
allow.iter().all(|a| a == "git.redclaw.dev"),
|
||||||
|
"a local backend must reach only the forge, got {allow:?}"
|
||||||
|
);
|
||||||
|
for h in ["api.anthropic.com", "api.z.ai", "api.kimi.com", "127.0.0.1"] {
|
||||||
|
assert!(!host_allowed(h, &allow), "{h} must NOT be reachable");
|
||||||
|
}
|
||||||
|
// The rules this design exists to avoid loosening.
|
||||||
|
assert!(parse_target("anything:11434").is_err());
|
||||||
|
assert!(parse_target("127.0.0.1:443").is_ok_and(|(h, _)| !host_allowed(&h, &allow)));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_ip_literal_is_not_allowed() {
|
||||||
|
let a = vec![".anthropic.com".to_string()];
|
||||||
|
assert!(!host_allowed("[::1]", &a));
|
||||||
|
assert!(!host_allowed("2606:4700::1111", &a));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A trailing dot is the same host to a resolver, so it must be to us.
|
||||||
|
#[test]
|
||||||
|
fn a_trailing_dot_does_not_bypass_the_list() {
|
||||||
|
assert!(host_allowed("api.anthropic.com.", &allow()));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A `.domain` entry covers subdomains, and only real subdomains.
|
||||||
|
#[test]
|
||||||
|
fn a_dot_prefixed_entry_matches_subdomains_only() {
|
||||||
|
let a = vec![".example.com".to_string()];
|
||||||
|
assert!(host_allowed("a.example.com", &a));
|
||||||
|
assert!(host_allowed("example.com", &a));
|
||||||
|
assert!(!host_allowed("notexample.com", &a));
|
||||||
|
assert!(!host_allowed("example.com.evil.test", &a));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Blanking the allow-list means no egress. Falling back to the default
|
||||||
|
/// would hand a mission the network an operator had just taken away.
|
||||||
|
#[test]
|
||||||
|
fn an_empty_allow_list_denies_everything() {
|
||||||
|
let none: Vec<String> = vec![];
|
||||||
|
assert!(!host_allowed("api.anthropic.com", &none));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Only 443. Anything else turns the proxy into a general-purpose tunnel to
|
||||||
|
/// whatever the allow-list happens to name.
|
||||||
|
#[test]
|
||||||
|
fn only_https_is_tunnelled() {
|
||||||
|
assert_eq!(parse_target("api.anthropic.com:443").unwrap().1, 443);
|
||||||
|
for bad in [
|
||||||
|
"api.anthropic.com:22",
|
||||||
|
"api.anthropic.com:80",
|
||||||
|
"api.anthropic.com",
|
||||||
|
"api.anthropic.com:not-a-port",
|
||||||
|
] {
|
||||||
|
assert!(parse_target(bad).is_err(), "{bad} must be refused");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,167 @@
|
|||||||
|
//! Host side of a microVM's route to the node's OWN locally-hosted model.
|
||||||
|
//!
|
||||||
|
//! # Why this is not the egress proxy
|
||||||
|
//!
|
||||||
|
//! [`crate::egress`] exists so an agent can reach the public internet under an
|
||||||
|
//! allow-list: it speaks HTTP `CONNECT`, takes a destination from the guest,
|
||||||
|
//! resolves it, and decides. Every one of those powers is a liability, which is
|
||||||
|
//! why that module is careful about ports, IP literals and suffix matching.
|
||||||
|
//!
|
||||||
|
//! This is the opposite shape. There is **no destination in the protocol**. The
|
||||||
|
//! guest opens a socket; the host connects it to `127.0.0.1:11434` on the node
|
||||||
|
//! and copies bytes. A compromised guest can ask for nothing else, because there
|
||||||
|
//! is nothing to ask — it is a pipe, not a proxy. That is strictly narrower than
|
||||||
|
//! anything the allow-list could express, and it is why routing a local model
|
||||||
|
//! through `egress` would have been the worse design: it would have meant
|
||||||
|
//! relaxing the 443-only rule and the IP-literal refusal, both of which exist
|
||||||
|
//! because a unit test caught them being bypassed.
|
||||||
|
//!
|
||||||
|
//! # Why plaintext is right here
|
||||||
|
//!
|
||||||
|
//! The bytes go guest loopback → vsock → host loopback. They never touch a
|
||||||
|
//! network, so there is no wire for TLS to protect. Ollama stays bound to
|
||||||
|
//! `127.0.0.1` on the node and is never exposed to the tailnet, which is a
|
||||||
|
//! stronger position than terminating TLS in front of it would have been.
|
||||||
|
//!
|
||||||
|
//! # Why it is per-backend
|
||||||
|
//!
|
||||||
|
//! The node binds this socket only for a backend declared to use a local model.
|
||||||
|
//! On every other backend the guest's listener is still there and simply gets a
|
||||||
|
//! refusal — the same fail-closed default `provider_hosts` applies to egress.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
|
||||||
|
use tokio::net::{TcpStream, UnixListener};
|
||||||
|
|
||||||
|
/// Host-side vsock port. Must match `fcagent`'s `MODEL_VSOCK_PORT`.
|
||||||
|
pub const MODEL_PORT: u32 = 9003;
|
||||||
|
|
||||||
|
/// Where the node's model server listens. Loopback, and not configurable from
|
||||||
|
/// the guest by design — see the module docs.
|
||||||
|
const OLLAMA_ADDR: &str = "127.0.0.1:11434";
|
||||||
|
|
||||||
|
/// Whether a backend is served by a model running on the node itself.
|
||||||
|
///
|
||||||
|
/// Named individually rather than by prefix. An unrecognised backend must not
|
||||||
|
/// acquire a route to anything by accident, which is the same rule
|
||||||
|
/// `egress::provider_hosts` and `mission_runtime::microvm_credential_for`
|
||||||
|
/// already apply from their own side.
|
||||||
|
pub fn uses_local_model(backend: Option<&str>) -> bool {
|
||||||
|
matches!(backend, Some("local-ornith"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bind the guest's local-model socket, if this backend has one.
|
||||||
|
///
|
||||||
|
/// `Ok(None)` means "this backend does not use a local model" and is the normal
|
||||||
|
/// case. An error means it should have had one and could not — reported by the
|
||||||
|
/// caller, never silently swallowed, because the symptom otherwise is an agent
|
||||||
|
/// that hangs on its first turn.
|
||||||
|
pub fn start(
|
||||||
|
uds: &Path,
|
||||||
|
vm_id: &str,
|
||||||
|
backend: Option<&str>,
|
||||||
|
) -> Result<Option<(PathBuf, tokio::task::JoinHandle<()>)>, String> {
|
||||||
|
if !uses_local_model(backend) {
|
||||||
|
return Ok(None);
|
||||||
|
}
|
||||||
|
let path = PathBuf::from(format!("{}_{}", uds.display(), MODEL_PORT));
|
||||||
|
// Firecracker leaves these behind exactly as it does its own socket, and a
|
||||||
|
// stale file makes bind fail with EADDRINUSE.
|
||||||
|
let _ = std::fs::remove_file(&path);
|
||||||
|
let listener =
|
||||||
|
UnixListener::bind(&path).map_err(|e| format!("bind {}: {e}", path.display()))?;
|
||||||
|
|
||||||
|
eprintln!(
|
||||||
|
"microvm {vm_id}: local model socket on {} -> {OLLAMA_ADDR}",
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
let vm = vm_id.to_string();
|
||||||
|
let task = tokio::spawn(async move {
|
||||||
|
loop {
|
||||||
|
match listener.accept().await {
|
||||||
|
Ok((s, _)) => {
|
||||||
|
let vm = vm.clone();
|
||||||
|
tokio::spawn(async move {
|
||||||
|
if let Err(e) = pipe(s).await {
|
||||||
|
// Loud, because the failure a mission sees is a turn
|
||||||
|
// that never answers. A refused connection here means
|
||||||
|
// the node's model server is down, and that is worth
|
||||||
|
// saying out loud rather than leaving to a timeout.
|
||||||
|
eprintln!("microvm {vm}: local model pipe failed: {e}");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("microvm {vm}: local model accept failed: {e}");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
Ok(Some((path, task)))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Splice one guest connection onto a fresh connection to the node's model.
|
||||||
|
async fn pipe(mut guest: tokio::net::UnixStream) -> Result<(), String> {
|
||||||
|
let mut model = TcpStream::connect(OLLAMA_ADDR)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("connect {OLLAMA_ADDR}: {e}"))?;
|
||||||
|
tokio::io::copy_bidirectional(&mut guest, &mut model)
|
||||||
|
.await
|
||||||
|
.map(|_| ())
|
||||||
|
.map_err(|e| format!("copy: {e}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// Only the backends that are meant to have a local model get one.
|
||||||
|
///
|
||||||
|
/// The negative half is the point: an unrecognised backend acquiring a route
|
||||||
|
/// to the node's own model server would be a hole opened by a typo, and it
|
||||||
|
/// would be invisible because the mission would simply work.
|
||||||
|
#[test]
|
||||||
|
fn a_local_route_is_never_granted_by_accident() {
|
||||||
|
assert!(uses_local_model(Some("local-ornith")));
|
||||||
|
|
||||||
|
for other in [
|
||||||
|
None,
|
||||||
|
Some(""),
|
||||||
|
Some("default"),
|
||||||
|
Some("claude"),
|
||||||
|
Some("canary-claude"),
|
||||||
|
Some("glm"),
|
||||||
|
Some("kimi"),
|
||||||
|
Some("local"),
|
||||||
|
Some("local-ornith-typo"),
|
||||||
|
Some("ornith"),
|
||||||
|
] {
|
||||||
|
assert!(
|
||||||
|
!uses_local_model(other),
|
||||||
|
"{other:?} must not reach the node's model server"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The guest cannot name a destination, so there is nothing to validate.
|
||||||
|
///
|
||||||
|
/// This asserts the property that makes this module safe enough to skip the
|
||||||
|
/// allow-list entirely: the upstream address is a constant. If it ever
|
||||||
|
/// becomes a parameter, this file needs everything `egress` has.
|
||||||
|
#[test]
|
||||||
|
fn the_upstream_address_is_a_constant_not_an_input() {
|
||||||
|
let src = include_str!("local_model.rs");
|
||||||
|
// Needles are split so they do not match themselves in this file.
|
||||||
|
assert_eq!(
|
||||||
|
src.matches(concat!("TcpStream", "::connect(")).count(),
|
||||||
|
1,
|
||||||
|
"exactly one dial site, and it must use the constant"
|
||||||
|
);
|
||||||
|
assert!(src.contains(concat!("TcpStream", "::connect(OLLAMA_ADDR)")));
|
||||||
|
assert!(
|
||||||
|
OLLAMA_ADDR.starts_with("127.0.0.1:"),
|
||||||
|
"the model server must be reached on loopback only"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -19,6 +19,9 @@ use sysinfo::{Disks, System};
|
|||||||
use tokio::sync::{mpsc, Mutex};
|
use tokio::sync::{mpsc, Mutex};
|
||||||
use tokio_tungstenite::tungstenite::Message;
|
use tokio_tungstenite::tungstenite::Message;
|
||||||
|
|
||||||
|
mod egress;
|
||||||
|
mod local_model;
|
||||||
|
mod microvm;
|
||||||
mod rtc;
|
mod rtc;
|
||||||
|
|
||||||
const B64: base64::engine::general_purpose::GeneralPurpose =
|
const B64: base64::engine::general_purpose::GeneralPurpose =
|
||||||
@@ -43,6 +46,15 @@ async fn main() {
|
|||||||
selftest();
|
selftest();
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
// Exercise the microVM lifecycle against a real VM on this node. Separate
|
||||||
|
// from --selftest because it needs KVM, so it can only pass on a node that
|
||||||
|
// actually reports microvm capability.
|
||||||
|
if std::env::args().any(|a| a == "--vm-selftest") {
|
||||||
|
if !microvm::selftest().await {
|
||||||
|
std::process::exit(1);
|
||||||
|
}
|
||||||
|
return;
|
||||||
|
}
|
||||||
let (server, token, ts_authkey) = parse_args();
|
let (server, token, ts_authkey) = parse_args();
|
||||||
if server.is_empty() || token.is_empty() {
|
if server.is_empty() || token.is_empty() {
|
||||||
eprintln!("usage: clawmates-node --server <https://gateway> --token <token> [--tailscale-authkey <key>]");
|
eprintln!("usage: clawmates-node --server <https://gateway> --token <token> [--tailscale-authkey <key>]");
|
||||||
@@ -108,6 +120,11 @@ async fn run(ws_url: &str) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
let (out_tx, mut out_rx) = mpsc::unbounded_channel::<String>();
|
let (out_tx, mut out_rx) = mpsc::unbounded_channel::<String>();
|
||||||
let ptys: Ptys = Arc::new(Mutex::new(HashMap::new()));
|
let ptys: Ptys = Arc::new(Mutex::new(HashMap::new()));
|
||||||
let peers: rtc::RtcPeers = Arc::new(Mutex::new(HashMap::new()));
|
let peers: rtc::RtcPeers = Arc::new(Mutex::new(HashMap::new()));
|
||||||
|
// microVMs this connection started. Scoped to the connection deliberately:
|
||||||
|
// a reconnect must not inherit VMs it cannot prove are still alive, and
|
||||||
|
// `vm_destroy` cleans a workdir by path even for an unregistered id, so a
|
||||||
|
// VM from a previous incarnation is reapable rather than orphaned.
|
||||||
|
let vms = microvm::new_vms();
|
||||||
// Collect heartbeats on a dedicated thread: the metric helpers shell out to
|
// Collect heartbeats on a dedicated thread: the metric helpers shell out to
|
||||||
// docker/tailscale and stat disks (blocking), which must never stall the
|
// docker/tailscale and stat disks (blocking), which must never stall the
|
||||||
// async select loop (or heartbeats/pongs would starve during a slow op).
|
// async select loop (or heartbeats/pongs would starve during a slow op).
|
||||||
@@ -131,6 +148,14 @@ async fn run(ws_url: &str) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
if tools_tx.send(frame).is_err() {
|
if tools_tx.send(frame).is_err() {
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
|
// What this node can HOST, as opposed to what it has installed. The
|
||||||
|
// scheduler needs it to place microVM missions, and the node is the
|
||||||
|
// only honest source: /dev/kvm either exists here or it does not, and
|
||||||
|
// no amount of configuration on the server can make it appear.
|
||||||
|
let caps = json!({ "t": "node_capabilities", "capabilities": probe_capabilities() });
|
||||||
|
if tools_tx.send(caps.to_string()).is_err() {
|
||||||
|
break;
|
||||||
|
}
|
||||||
std::thread::sleep(Duration::from_secs(900));
|
std::thread::sleep(Duration::from_secs(900));
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -180,8 +205,9 @@ async fn run(ws_url: &str) -> Result<(), Box<dyn std::error::Error>> {
|
|||||||
let out = out_tx.clone();
|
let out = out_tx.clone();
|
||||||
let ptys = ptys.clone();
|
let ptys = ptys.clone();
|
||||||
let peers = peers.clone();
|
let peers = peers.clone();
|
||||||
|
let vms = vms.clone();
|
||||||
let text = t.to_string();
|
let text = t.to_string();
|
||||||
tokio::spawn(async move { handle_frame(&text, &out, &ptys, &peers).await; });
|
tokio::spawn(async move { handle_frame(&text, &out, &ptys, &peers, &vms).await; });
|
||||||
}
|
}
|
||||||
Some(Ok(Message::Ping(p))) => {
|
Some(Ok(Message::Ping(p))) => {
|
||||||
match tokio::time::timeout(WRITE_DEADLINE, write.send(Message::Pong(p))).await {
|
match tokio::time::timeout(WRITE_DEADLINE, write.send(Message::Pong(p))).await {
|
||||||
@@ -241,6 +267,70 @@ fn heartbeat(sys: &mut System) -> String {
|
|||||||
/// Probe installed dev-tool versions: for each tool, find its binary across the
|
/// Probe installed dev-tool versions: for each tool, find its binary across the
|
||||||
/// usual bin dirs and read `--version`. Returns `{ tool: "x.y.z", … }` for the
|
/// usual bin dirs and read `--version`. Returns `{ tool: "x.y.z", … }` for the
|
||||||
/// ones found. Probes `kimi-cli` (the real uv tool), not the `kimi` API wrapper.
|
/// ones found. Probes `kimi-cli` (the real uv tool), not the `kimi` API wrapper.
|
||||||
|
/// What this node can HOST — the inputs to placement predicates.
|
||||||
|
///
|
||||||
|
/// Distinct from [`probe_tools`], which reports what is *installed* for the
|
||||||
|
/// operator to see and update. This answers "may the scheduler put a microVM
|
||||||
|
/// mission here", and the answer is a property of the hardware: gw-04 is
|
||||||
|
/// itself a VM without nested virtualisation and has no `/dev/kvm`, so it can
|
||||||
|
/// never host one however it is configured.
|
||||||
|
///
|
||||||
|
/// Every value is probed, never assumed. A capability that is merely expected
|
||||||
|
/// is the same as a capability that is absent, right up until a mission is
|
||||||
|
/// scheduled onto a node that cannot run it.
|
||||||
|
fn probe_capabilities() -> Value {
|
||||||
|
// The device node is necessary but not sufficient — it can exist while
|
||||||
|
// being unopenable (wrong group, or a container without the device
|
||||||
|
// passed through). Try to open it, because that is what firecracker does.
|
||||||
|
let kvm = std::fs::OpenOptions::new()
|
||||||
|
.read(true)
|
||||||
|
.write(true)
|
||||||
|
.open("/dev/kvm")
|
||||||
|
.is_ok();
|
||||||
|
|
||||||
|
let firecracker = std::process::Command::new("firecracker")
|
||||||
|
.arg("--version")
|
||||||
|
.output()
|
||||||
|
.ok()
|
||||||
|
.filter(|o| o.status.success())
|
||||||
|
.and_then(|o| {
|
||||||
|
String::from_utf8_lossy(&o.stdout)
|
||||||
|
.lines()
|
||||||
|
.next()
|
||||||
|
.map(|l| l.trim().to_string())
|
||||||
|
});
|
||||||
|
|
||||||
|
// Which rootfs images are actually on this node's disk. Reported so
|
||||||
|
// placement can require the mission's backend rather than assuming any
|
||||||
|
// KVM-capable node can boot any image — see microvm::available_backends.
|
||||||
|
let backends = microvm::available_backends();
|
||||||
|
capabilities_from(kvm, firecracker.as_deref(), &backends)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Shape the capability report from probe results.
|
||||||
|
///
|
||||||
|
/// Split from [`probe_capabilities`] so the rule can be tested without a
|
||||||
|
/// `/dev/kvm` to open — the machine running the tests is usually the one that
|
||||||
|
/// cannot host a microVM.
|
||||||
|
fn capabilities_from(kvm: bool, firecracker: Option<&str>, backends: &[String]) -> Value {
|
||||||
|
json!({
|
||||||
|
"kvm": kvm,
|
||||||
|
"firecracker": firecracker,
|
||||||
|
// The backends this node can boot. An ARRAY, and empty when there are
|
||||||
|
// none: `set_capabilities` REPLACES, so an image that was deleted stops
|
||||||
|
// being advertised on the next report instead of leaving a stale claim.
|
||||||
|
//
|
||||||
|
// Reported even when `microvm` is false, because it is a fact about the
|
||||||
|
// disk rather than a promise — placement requires both.
|
||||||
|
"rootfs": backends,
|
||||||
|
// BOTH must hold. A node with KVM but no firecracker binary looks
|
||||||
|
// capable by the obvious test and fails at launch; a node with the
|
||||||
|
// binary but no KVM is gw-04. Computed here rather than in the
|
||||||
|
// scheduler so the rule sits next to the probe that feeds it.
|
||||||
|
"microvm": kvm && firecracker.is_some(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
fn probe_tools() -> Value {
|
fn probe_tools() -> Value {
|
||||||
let home = std::env::var("HOME").unwrap_or_default();
|
let home = std::env::var("HOME").unwrap_or_default();
|
||||||
let dirs = [
|
let dirs = [
|
||||||
@@ -459,6 +549,7 @@ async fn handle_frame(
|
|||||||
out: &mpsc::UnboundedSender<String>,
|
out: &mpsc::UnboundedSender<String>,
|
||||||
ptys: &Ptys,
|
ptys: &Ptys,
|
||||||
peers: &rtc::RtcPeers,
|
peers: &rtc::RtcPeers,
|
||||||
|
vms: µvm::Vms,
|
||||||
) {
|
) {
|
||||||
let Ok(v) = serde_json::from_str::<Value>(text) else {
|
let Ok(v) = serde_json::from_str::<Value>(text) else {
|
||||||
return;
|
return;
|
||||||
@@ -521,6 +612,73 @@ async fn handle_frame(
|
|||||||
// Agent-sandbox container ops: drive the REAL DockerDriver so the
|
// Agent-sandbox container ops: drive the REAL DockerDriver so the
|
||||||
// hardening (cap-drop ALL, seccomp, no-net, read-only, non-root) is
|
// hardening (cap-drop ALL, seccomp, no-net, read-only, non-root) is
|
||||||
// byte-identical to the gateway's local sandboxes.
|
// byte-identical to the gateway's local sandboxes.
|
||||||
|
// microVM ops. Same envelope as every other op, so adding them needed
|
||||||
|
// no protocol change. `vm_create` blocks until the guest agent answers:
|
||||||
|
// a VM that booted but serves nothing is worse than one that failed.
|
||||||
|
op @ ("vm_create" | "vm_inject" | "vm_exec" | "vm_collect" | "vm_destroy" | "vm_list") => {
|
||||||
|
if let Some(id) = v.get("id").and_then(Value::as_u64) {
|
||||||
|
let (op, v, out, vms) = (op.to_string(), v.clone(), out.clone(), vms.clone());
|
||||||
|
// Spawned: a VM boot takes ~1s and an exec can take an hour.
|
||||||
|
// Running it inline would stall heartbeats and the daemon would
|
||||||
|
// be declared offline mid-mission.
|
||||||
|
tokio::spawn(async move {
|
||||||
|
// While an `exec` runs, follow the turn's log and push each
|
||||||
|
// chunk to the server as it appears. The guest agent accepts
|
||||||
|
// concurrent connections (proved against a live VM: a tail
|
||||||
|
// returned data second-by-second while an 8s exec was still
|
||||||
|
// running), so this does not wait for, or delay, the turn.
|
||||||
|
//
|
||||||
|
// Only for `vm_exec`, and only when the caller named a run to
|
||||||
|
// attribute the output to — a probe exec has nothing to
|
||||||
|
// stream and no subscriber.
|
||||||
|
// Set when the turn returns, so the tail can DRAIN before it
|
||||||
|
// stops rather than being cut off mid-flush.
|
||||||
|
let turn_done = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||||
|
let tail = (op == "vm_exec")
|
||||||
|
.then(|| {
|
||||||
|
let run_id = v.get("run_id").and_then(Value::as_str)?.to_string();
|
||||||
|
let log_path = v
|
||||||
|
.get("log_path")
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.unwrap_or("/root/agent.log")
|
||||||
|
.to_string();
|
||||||
|
let vm_id = v.get("vm_id").and_then(Value::as_str)?.to_string();
|
||||||
|
Some(tokio::spawn(stream_vm_log(
|
||||||
|
vms.clone(),
|
||||||
|
vm_id,
|
||||||
|
run_id,
|
||||||
|
log_path,
|
||||||
|
out.clone(),
|
||||||
|
turn_done.clone(),
|
||||||
|
)))
|
||||||
|
})
|
||||||
|
.flatten();
|
||||||
|
let (ok, output) = microvm::handle_op(&op, &v, &vms).await;
|
||||||
|
// Let the tail DRAIN, then stop. Aborting here was wrong:
|
||||||
|
// `claude -p | tee` makes stdout a pipe, so the CLI block-
|
||||||
|
// buffers and flushes at EXIT — the most valuable output
|
||||||
|
// arrives in the instant the turn ends. Aborting raced that
|
||||||
|
// flush and lost it. Measured: a solo turn (minutes long) won
|
||||||
|
// the race and streamed 337 bytes; every node of a composed
|
||||||
|
// run (~20s each) lost it and streamed nothing at all.
|
||||||
|
//
|
||||||
|
// Bounded, because a VM that stopped answering must not hold
|
||||||
|
// this task open — the abort remains, as a backstop rather
|
||||||
|
// than the mechanism.
|
||||||
|
if let Some(t) = tail {
|
||||||
|
turn_done.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||||
|
let drained =
|
||||||
|
tokio::time::timeout(std::time::Duration::from_secs(20), t).await;
|
||||||
|
if drained.is_err() {
|
||||||
|
eprintln!("clawmates-node: tail drain timed out for {op}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let _ = out.send(
|
||||||
|
json!({ "t": "result", "id": id, "ok": ok, "output": output }).to_string(),
|
||||||
|
);
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
op @ ("sb_provision" | "sb_exec" | "sb_destroy" | "sb_health" | "sb_list") => {
|
op @ ("sb_provision" | "sb_exec" | "sb_destroy" | "sb_health" | "sb_list") => {
|
||||||
if let Some(id) = v.get("id").and_then(Value::as_u64) {
|
if let Some(id) = v.get("id").and_then(Value::as_u64) {
|
||||||
let (ok, output) = sb_op(op, &v).await;
|
let (ok, output) = sb_op(op, &v).await;
|
||||||
@@ -752,6 +910,72 @@ fn spawn_command_pty(argv: &[String], cols: u16, rows: u16) -> Result<PtyParts,
|
|||||||
spawn_pty(c, cols, rows)
|
spawn_pty(c, cols, rows)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Follow a running turn's log inside a VM and push each chunk to the server.
|
||||||
|
///
|
||||||
|
/// The other half of the observability path: the guest tails the file, this
|
||||||
|
/// forwards what it reads over the WebSocket the daemon already holds, and the
|
||||||
|
/// server appends it to the run so the live pane and the Output tab both have it.
|
||||||
|
///
|
||||||
|
/// Reconnects on a dropped tail, resuming from the last offset — following by
|
||||||
|
/// OFFSET rather than holding one socket open forever is what makes that cheap.
|
||||||
|
/// It gives up after a few consecutive failures rather than spinning: by then
|
||||||
|
/// the VM is gone and the turn's own result is the record.
|
||||||
|
async fn stream_vm_log(
|
||||||
|
vms: microvm::Vms,
|
||||||
|
vm_id: String,
|
||||||
|
run_id: String,
|
||||||
|
log_path: String,
|
||||||
|
out: tokio::sync::mpsc::UnboundedSender<String>,
|
||||||
|
turn_done: std::sync::Arc<std::sync::atomic::AtomicBool>,
|
||||||
|
) {
|
||||||
|
// Said out loud at the start, because the failure this replaced was
|
||||||
|
// invisible: the tail gave up during VM boot and logged nothing, so an empty
|
||||||
|
// Live tab looked identical to a feature that was never wired.
|
||||||
|
eprintln!("clawmates-node: following {log_path} in {vm_id} for run {run_id}");
|
||||||
|
let mut at: u64 = 0;
|
||||||
|
let mut failures = 0;
|
||||||
|
while failures < 3 {
|
||||||
|
let at_before = at;
|
||||||
|
let sent = out.clone();
|
||||||
|
let rid = run_id.clone();
|
||||||
|
match microvm::tail_into(&vms, &vm_id, &log_path, at, move |offset, data| {
|
||||||
|
let _ = sent.send(
|
||||||
|
json!({ "t": "vm_out", "run_id": rid, "at": offset, "data": data }).to_string(),
|
||||||
|
);
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(reached) => {
|
||||||
|
// NO PROGRESS IS NOT THE END. The guest reports EOF whenever the
|
||||||
|
// file has been idle, and the first idle window is always the one
|
||||||
|
// before the turn writes anything — the VM is still booting and
|
||||||
|
// the CLI still starting. Returning here meant the tail gave up
|
||||||
|
// seconds into every run, before a single byte existed. Measured:
|
||||||
|
// a turn that streamed nothing at all.
|
||||||
|
//
|
||||||
|
// The caller aborts this task when the exec returns, so "keep
|
||||||
|
// waiting" cannot outlive the turn; the abort is the terminator,
|
||||||
|
// not a guess about idleness.
|
||||||
|
at = reached;
|
||||||
|
failures = 0;
|
||||||
|
// The turn has returned AND this pass read nothing new: the
|
||||||
|
// final flush is already in hand, so stop. Checked after a read,
|
||||||
|
// never before one — exiting on the flag alone would drop
|
||||||
|
// exactly the bytes this exists to capture.
|
||||||
|
if turn_done.load(std::sync::atomic::Ordering::Relaxed) && reached == at_before {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
tokio::time::sleep(std::time::Duration::from_millis(300)).await;
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
failures += 1;
|
||||||
|
eprintln!("clawmates-node: tail of {vm_id} for run {run_id} failed: {e}");
|
||||||
|
tokio::time::sleep(std::time::Duration::from_secs(2)).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Spawn a host login shell in a PTY; stream its output back as pty_out frames.
|
/// Spawn a host login shell in a PTY; stream its output back as pty_out frames.
|
||||||
async fn open_pty(
|
async fn open_pty(
|
||||||
sid: u64,
|
sid: u64,
|
||||||
@@ -1274,3 +1498,61 @@ fn ensure_tmux() {
|
|||||||
eprintln!("tmux not found (auto-install unavailable) — host terminal will use a plain shell; `apt install tmux` for resumable sessions");
|
eprintln!("tmux not found (auto-install unavailable) — host terminal will use a plain shell; `apt install tmux` for resumable sessions");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod capability_tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn microvm_needs_both_kvm_and_firecracker() {
|
||||||
|
assert_eq!(
|
||||||
|
capabilities_from(true, Some("Firecracker v1.16.1"), &[])["microvm"],
|
||||||
|
json!(true)
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
capabilities_from(true, None, &[])["microvm"],
|
||||||
|
json!(false),
|
||||||
|
"KVM without firecracker cannot host a microVM"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
capabilities_from(false, Some("Firecracker v1.16.1"), &[])["microvm"],
|
||||||
|
json!(false),
|
||||||
|
"firecracker without KVM is gw-04 — it can never host one"
|
||||||
|
);
|
||||||
|
assert_eq!(capabilities_from(false, None, &[])["microvm"], json!(false));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The report replaces rather than merges server-side, so a node that has
|
||||||
|
/// LOST a capability must say so rather than omitting the key — an absent
|
||||||
|
/// key and a false one must not be distinguishable to the predicate.
|
||||||
|
#[test]
|
||||||
|
fn a_lost_capability_is_reported_false_not_omitted() {
|
||||||
|
let caps = capabilities_from(false, None, &[]);
|
||||||
|
assert!(caps.get("kvm").is_some(), "kvm must always be present");
|
||||||
|
assert!(
|
||||||
|
caps.get("microvm").is_some(),
|
||||||
|
"microvm must always be present"
|
||||||
|
);
|
||||||
|
// Same reasoning for the image list: a node that deleted its last rootfs
|
||||||
|
// must report an empty ARRAY, not omit the key. Placement asks "does this
|
||||||
|
// node have backend X"; against a missing key that question has no
|
||||||
|
// answer, and a scheduler with no answer picks something.
|
||||||
|
assert_eq!(
|
||||||
|
caps.get("rootfs"),
|
||||||
|
Some(&json!([])),
|
||||||
|
"rootfs must always be present, empty when there are no images"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The list is what placement matches a mission's `backend` against, so it
|
||||||
|
/// must carry the names verbatim.
|
||||||
|
#[test]
|
||||||
|
fn reported_backends_are_the_names_placement_will_ask_for() {
|
||||||
|
let caps = capabilities_from(
|
||||||
|
true,
|
||||||
|
Some("Firecracker v1.16.1"),
|
||||||
|
&["claude".to_string(), "default".to_string()],
|
||||||
|
);
|
||||||
|
assert_eq!(caps["rootfs"], json!(["claude", "default"]));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@@ -24,11 +24,32 @@ async fn main() -> ExitCode {
|
|||||||
|
|
||||||
/// Instantiates the configured LLM provider. The Anthropic key comes from
|
/// Instantiates the configured LLM provider. The Anthropic key comes from
|
||||||
/// the environment until the secret broker lands in P2.
|
/// the environment until the secret broker lands in P2.
|
||||||
|
///
|
||||||
|
/// The **subscription wins** when both credentials are present. This is the
|
||||||
|
/// structural half of the fix that `cm_api::subscription` does per-call: a bare
|
||||||
|
/// model name resolves to whatever this function returns, so making that the
|
||||||
|
/// subscription means no server-side call can reach the metered key by
|
||||||
|
/// accident — by construction, rather than by a source-grep test that has
|
||||||
|
/// already missed four call sites once. The metered key stays usable as a
|
||||||
|
/// fallback for deployments that have credit; ours does not, which is what
|
||||||
|
/// made the ordering matter.
|
||||||
fn build_provider(config: &AppConfig) -> Result<Arc<dyn LlmProvider>, String> {
|
fn build_provider(config: &AppConfig) -> Result<Arc<dyn LlmProvider>, String> {
|
||||||
match config.llm.provider {
|
match config.llm.provider {
|
||||||
LlmProviderKind::Anthropic => {
|
LlmProviderKind::Anthropic => {
|
||||||
let key = std::env::var("ANTHROPIC_API_KEY")
|
if let Some(provider) = cm_api::subscription::provider() {
|
||||||
.map_err(|_| "llm.provider = \"anthropic\" requires ANTHROPIC_API_KEY")?;
|
println!(
|
||||||
|
"clawmates-server: default LLM provider = Claude Code subscription \
|
||||||
|
(bare model names bill no metered key)"
|
||||||
|
);
|
||||||
|
return Ok(Arc::new(provider));
|
||||||
|
}
|
||||||
|
let key = std::env::var("ANTHROPIC_API_KEY").map_err(|_| {
|
||||||
|
"llm.provider = \"anthropic\" needs a credential: either \
|
||||||
|
ANTHROPIC_OAUTH_TOKEN / CLAUDE_CODE_OAUTH_TOKEN (sk-ant-oat…, \
|
||||||
|
the Claude Code subscription, preferred) or ANTHROPIC_API_KEY \
|
||||||
|
(sk-ant-api…, metered)"
|
||||||
|
.to_string()
|
||||||
|
})?;
|
||||||
// A subscription OAuth token pasted where an API key belongs
|
// A subscription OAuth token pasted where an API key belongs
|
||||||
// authenticates nothing here and fails on the first model call,
|
// authenticates nothing here and fails on the first model call,
|
||||||
// far from the mistake. Both start `sk-ant-`, so the confusion is
|
// far from the mistake. Both start `sk-ant-`, so the confusion is
|
||||||
@@ -40,6 +61,11 @@ fn build_provider(config: &AppConfig) -> Result<Arc<dyn LlmProvider>, String> {
|
|||||||
bearer auth and is what the phase evaluator reads."
|
bearer auth and is what the phase evaluator reads."
|
||||||
.to_string());
|
.to_string());
|
||||||
}
|
}
|
||||||
|
eprintln!(
|
||||||
|
"clawmates-server: WARNING — no subscription token; the default LLM \
|
||||||
|
provider is the METERED ANTHROPIC_API_KEY and every bare model name \
|
||||||
|
bills it"
|
||||||
|
);
|
||||||
Ok(Arc::new(AnthropicProvider::new(key)))
|
Ok(Arc::new(AnthropicProvider::new(key)))
|
||||||
}
|
}
|
||||||
LlmProviderKind::OpenAiCompat => {
|
LlmProviderKind::OpenAiCompat => {
|
||||||
@@ -69,8 +95,21 @@ fn build_provider(config: &AppConfig) -> Result<Arc<dyn LlmProvider>, String> {
|
|||||||
fn build_provider_registry(config: &AppConfig) -> cm_runtime::ProviderRegistry {
|
fn build_provider_registry(config: &AppConfig) -> cm_runtime::ProviderRegistry {
|
||||||
let mut map = std::collections::HashMap::new();
|
let mut map = std::collections::HashMap::new();
|
||||||
for p in &config.llm.providers {
|
for p in &config.llm.providers {
|
||||||
match std::env::var(&p.api_key_env) {
|
// A provider may legitimately need no key. A model running on our own
|
||||||
Ok(key) if !key.is_empty() => {
|
// hardware has nothing to authenticate to, and requiring a variable
|
||||||
|
// whose value is ignored is a step that can only ever fail — silently,
|
||||||
|
// since an unset key SKIPS the provider and the first symptom is a
|
||||||
|
// fallback chain quietly one link shorter than it reads.
|
||||||
|
let key = match std::env::var(&p.api_key_env) {
|
||||||
|
Ok(k) if !k.is_empty() => Ok(k),
|
||||||
|
other if p.api_key_env.trim().is_empty() => {
|
||||||
|
let _ = other;
|
||||||
|
Ok(String::new())
|
||||||
|
}
|
||||||
|
other => other,
|
||||||
|
};
|
||||||
|
match key {
|
||||||
|
Ok(key) if !key.is_empty() || p.api_key_env.trim().is_empty() => {
|
||||||
let provider: Arc<dyn LlmProvider> = match p.format.as_str() {
|
let provider: Arc<dyn LlmProvider> = match p.format.as_str() {
|
||||||
"anthropic" => Arc::new(cm_llm::AnthropicProvider::with_base_url(
|
"anthropic" => Arc::new(cm_llm::AnthropicProvider::with_base_url(
|
||||||
key,
|
key,
|
||||||
@@ -280,6 +319,9 @@ async fn run() -> Result<(), String> {
|
|||||||
cm_api::topology_worker::spawn(
|
cm_api::topology_worker::spawn(
|
||||||
pool.clone(),
|
pool.clone(),
|
||||||
runtime.clone(),
|
runtime.clone(),
|
||||||
|
// The composed tier (`microvm_graph`) runs each graph node as a VM on a
|
||||||
|
// fleet node, so the worker needs the same hub the phase runner uses.
|
||||||
|
node_hub.clone(),
|
||||||
std::time::Duration::from_secs(3),
|
std::time::Duration::from_secs(3),
|
||||||
);
|
);
|
||||||
// Boot-time content loaders — skills first, then team templates
|
// Boot-time content loaders — skills first, then team templates
|
||||||
@@ -299,6 +341,11 @@ async fn run() -> Result<(), String> {
|
|||||||
// for INT-XX markers in event payloads and upserts mission_tasks
|
// for INT-XX markers in event payloads and upserts mission_tasks
|
||||||
// rows so the canvas renders a live status timeline.
|
// rows so the canvas renders a live status timeline.
|
||||||
cm_api::task_card_worker::spawn(pool.clone());
|
cm_api::task_card_worker::spawn(pool.clone());
|
||||||
|
// Agents apply their own skill drafts. Announced at boot by the spawner
|
||||||
|
// itself, because this flips an approval gate that existed since the
|
||||||
|
// feature shipped — and a safety gate whose state is invisible is one
|
||||||
|
// nobody notices has changed.
|
||||||
|
cm_api::skill_self_authoring::spawn(pool.clone());
|
||||||
// Load the workflow recipes now rather than lazily on first mission
|
// Load the workflow recipes now rather than lazily on first mission
|
||||||
// create, so a malformed TOML shows up in the boot log instead of
|
// create, so a malformed TOML shows up in the boot log instead of
|
||||||
// silently yielding a mission with no phase config.
|
// silently yielding a mission with no phase config.
|
||||||
@@ -329,7 +376,41 @@ async fn run() -> Result<(), String> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
cm_api::phase_runner::spawn(pool.clone(), runtime.clone());
|
// The other half of runtime_preflight's question: the runtime has the TOOLS,
|
||||||
|
// but can the independent JUDGE be reached? A dead validator makes every
|
||||||
|
// done_when phase unmeetable, and without this the first symptom is a
|
||||||
|
// mission failing after its VMs have already run.
|
||||||
|
cm_api::validator_preflight::report_at_boot(runtime.clone());
|
||||||
|
// Every link of the model fallback chain, probed through the real call path.
|
||||||
|
// A chain is the one piece of infrastructure nobody looks at until the day it
|
||||||
|
// has to work, so it is checked on the days it does not.
|
||||||
|
cm_api::subscription::report_at_boot(runtime.clone());
|
||||||
|
cm_api::phase_runner::spawn(pool.clone(), runtime.clone(), node_hub.clone());
|
||||||
|
// Scheduled missions. `missions.schedule` has collected a cron from the
|
||||||
|
// wizard since 0047 and NOTHING read it back — every scheduled mission ever
|
||||||
|
// created sat in `draft` forever while the UI said it was on a schedule.
|
||||||
|
// 60s matches the finest cron granularity; the sweep claims atomically and
|
||||||
|
// records each occurrence in `mission_fires`, so replicas and restarts
|
||||||
|
// cannot double-launch a container.
|
||||||
|
// Render finished Continuous Research missions into episodes. A sweep, not
|
||||||
|
// a phase step: rendering is not the agents' work and must not be able to
|
||||||
|
// fail a phase that succeeded, and a transient API error simply retries on
|
||||||
|
// the next tick.
|
||||||
|
// Every 2 minutes, NOT 5. The mission checkout that holds script.md is
|
||||||
|
// deleted 30 minutes after the mission reaches a terminal state, so this
|
||||||
|
// sweep is racing a reaper. Two minutes leaves ~15 attempts inside that
|
||||||
|
// window; a slower sweep loses the episode permanently.
|
||||||
|
cm_api::podcast::spawn(
|
||||||
|
pool.clone(),
|
||||||
|
Some(blob.clone()),
|
||||||
|
std::time::Duration::from_secs(2 * 60),
|
||||||
|
);
|
||||||
|
cm_api::mission_schedule::spawn(
|
||||||
|
pool.clone(),
|
||||||
|
Some(node_hub.clone()),
|
||||||
|
Some(blob.clone()),
|
||||||
|
std::time::Duration::from_secs(60),
|
||||||
|
);
|
||||||
// Per-mission runtime container sweeper (C3): tears down mission
|
// Per-mission runtime container sweeper (C3): tears down mission
|
||||||
// runtime containers 30 min after the mission reaches a terminal
|
// runtime containers 30 min after the mission reaches a terminal
|
||||||
// state so operators have a window to pull final artifacts.
|
// state so operators have a window to pull final artifacts.
|
||||||
@@ -337,13 +418,7 @@ async fn run() -> Result<(), String> {
|
|||||||
// Phase completion summarizer: reads terminal-state phases and
|
// Phase completion summarizer: reads terminal-state phases and
|
||||||
// asks Claude Opus 4.8 to synthesize a "what got done" card that
|
// asks Claude Opus 4.8 to synthesize a "what got done" card that
|
||||||
// the UI renders under the phase.
|
// the UI renders under the phase.
|
||||||
cm_api::phase_summarizer::spawn(pool.clone());
|
cm_api::phase_summarizer::spawn(pool.clone(), runtime.clone());
|
||||||
// PDF renderer worker (Slice 6): watches mission_artifacts for
|
|
||||||
// MD entries with render_pdf_status='pending', calls the
|
|
||||||
// configured LLM (default Gemini 2.5 Flash) for styled HTML,
|
|
||||||
// prints to PDF via chromium --headless. No-op-friendly when
|
|
||||||
// GEMINI_API_KEY / chromium binary aren't configured.
|
|
||||||
cm_api::pdf_renderer::spawn(pool.clone());
|
|
||||||
// Outbound-email delivery: drains the §15-gated `outbox` over SMTP. Inert
|
// Outbound-email delivery: drains the §15-gated `outbox` over SMTP. Inert
|
||||||
// until CLAWMATES_SMTP_* is set, so it ships safely before credentials exist.
|
// until CLAWMATES_SMTP_* is set, so it ships safely before credentials exist.
|
||||||
cm_runtime::spawn_drainer(pool.clone(), std::time::Duration::from_secs(10));
|
cm_runtime::spawn_drainer(pool.clone(), std::time::Duration::from_secs(10));
|
||||||
@@ -352,6 +427,20 @@ async fn run() -> Result<(), String> {
|
|||||||
// Expiry/retention sweep: expires stale auth/oauth rows and prunes old
|
// Expiry/retention sweep: expires stale auth/oauth rows and prunes old
|
||||||
// journal/audit rows hourly so unbounded tables don't accumulate.
|
// journal/audit rows hourly so unbounded tables don't accumulate.
|
||||||
cm_api::cleanup_sweeper::spawn(pool.clone(), std::time::Duration::from_secs(3600));
|
cm_api::cleanup_sweeper::spawn(pool.clone(), std::time::Duration::from_secs(3600));
|
||||||
|
// Its filesystem counterpart. `cleanup_sweeper` prunes ROWS, and deleting a
|
||||||
|
// row has never deleted a directory — which is why the gateway, the smallest
|
||||||
|
// disk in the fleet, accumulates mission trees that nothing reclaims.
|
||||||
|
cm_api::mission_gc::spawn(pool.clone(), std::time::Duration::from_secs(3600));
|
||||||
|
// Agent lifecycle: reap crews whose missions finished (after a 24h grace so
|
||||||
|
// the results view can still show who did the work) and crews left bound to
|
||||||
|
// nothing. Never touches an agent without an `agent_template_link` row —
|
||||||
|
// that is the operator's own staff, which looks identical to an orphan if
|
||||||
|
// you judge by team membership alone.
|
||||||
|
cm_api::agent_lifecycle::spawn(
|
||||||
|
pool.clone(),
|
||||||
|
runtime.clone(),
|
||||||
|
std::time::Duration::from_secs(3600),
|
||||||
|
);
|
||||||
// Fleet backstop: a node whose heartbeats stop (without a clean channel
|
// Fleet backstop: a node whose heartbeats stop (without a clean channel
|
||||||
// close) goes offline within ~28s even if its control channel hangs.
|
// close) goes offline within ~28s even if its control channel hangs.
|
||||||
cm_api::fleet::spawn_node_sweeper(pool.clone(), std::time::Duration::from_secs(8), 20);
|
cm_api::fleet::spawn_node_sweeper(pool.clone(), std::time::Duration::from_secs(8), 20);
|
||||||
@@ -408,6 +497,10 @@ async fn run() -> Result<(), String> {
|
|||||||
// every consequence — an ungated test suite, a scan that scanned nothing —
|
// every consequence — an ungated test suite, a scan that scanned nothing —
|
||||||
// looked like a normal result rather than a broken deployment.
|
// looked like a normal result rather than a broken deployment.
|
||||||
cm_api::runtime_preflight::report_at_boot();
|
cm_api::runtime_preflight::report_at_boot();
|
||||||
|
// And whether the gateway those missions drive is configured at all. Both
|
||||||
|
// of its variables are read at FIRST USE, so a deployment missing them
|
||||||
|
// boots clean and fails on the first phase someone runs.
|
||||||
|
cm_api::gateway_preflight::report_at_boot();
|
||||||
// Graceful shutdown: on SIGTERM/Ctrl-C, stop accepting, finish in-flight
|
// Graceful shutdown: on SIGTERM/Ctrl-C, stop accepting, finish in-flight
|
||||||
// requests, then DRAIN the sandbox managers so no container is left running.
|
// requests, then DRAIN the sandbox managers so no container is left running.
|
||||||
let shutdown = async move {
|
let shutdown = async move {
|
||||||
|
|||||||
@@ -0,0 +1,27 @@
|
|||||||
|
[package]
|
||||||
|
name = "fcagent"
|
||||||
|
version = "0.1.0"
|
||||||
|
edition.workspace = true
|
||||||
|
rust-version.workspace = true
|
||||||
|
license.workspace = true
|
||||||
|
publish.workspace = true
|
||||||
|
|
||||||
|
[[bin]]
|
||||||
|
name = "fcagent"
|
||||||
|
path = "src/main.rs"
|
||||||
|
|
||||||
|
[dependencies]
|
||||||
|
# std has no AF_VSOCK, and the workspace denies `unsafe`, so raw libc is not an
|
||||||
|
# option. This is a safe wrapper over the socket calls.
|
||||||
|
vsock = "0.5"
|
||||||
|
serde_json = { workspace = true }
|
||||||
|
tar = { workspace = true }
|
||||||
|
base64 = "0.22"
|
||||||
|
|
||||||
|
# NOTE: a `[profile.release]` here would be silently ignored — cargo only honours
|
||||||
|
# profiles at the workspace root. The binary is small enough on the default
|
||||||
|
# release profile (~1 MB static) that overriding the whole workspace's profile to
|
||||||
|
# shave it would be a bad trade.
|
||||||
|
|
||||||
|
[lints]
|
||||||
|
workspace = true
|
||||||
@@ -0,0 +1,988 @@
|
|||||||
|
//! ClawMates microVM guest agent — pid 1 inside a Firecracker microVM.
|
||||||
|
//!
|
||||||
|
//! Runs as `init=/usr/local/bin/fcagent`'s exec target and answers the host over
|
||||||
|
//! **vsock** (port 9001), never the serial console: feeding a guest over stdin
|
||||||
|
//! races its startup and arrives half-consumed. The console stays a log.
|
||||||
|
//!
|
||||||
|
//! # Why this is a static Rust binary and not the python script it replaces
|
||||||
|
//!
|
||||||
|
//! The python version worked only because Firecracker's CI Ubuntu image happens
|
||||||
|
//! to ship python3. **None of our own images do** — `agent-base` has neither
|
||||||
|
//! python nor git, `agent-terminal` has git but no python — so the agent could
|
||||||
|
//! never have run in a real mission rootfs. An agent that dictates what must be
|
||||||
|
//! installed in the image has the dependency backwards. This is a
|
||||||
|
//! `x86_64-unknown-linux-musl` static binary: it needs nothing from the rootfs
|
||||||
|
//! it is dropped into.
|
||||||
|
//!
|
||||||
|
//! # Wire protocol (unchanged from the python agent, deliberately)
|
||||||
|
//!
|
||||||
|
//! One request per connection: a 4-byte big-endian length followed by JSON, and
|
||||||
|
//! the reply framed the same way. The length prefix is the point — a reply
|
||||||
|
//! larger than a socket buffer arrives in pieces, and reading "whatever was
|
||||||
|
//! available" would parse a truncated object as a complete one.
|
||||||
|
//!
|
||||||
|
//! Ops: `ping`, `exec`, `put`, `get`. `crates/bins/clawmates-node/src/microvm.rs`
|
||||||
|
//! and `crates/cm-api/src/microvm_client.rs` speak this and needed no change.
|
||||||
|
|
||||||
|
use std::io::{Read, Write};
|
||||||
|
use std::net::TcpListener;
|
||||||
|
use std::os::unix::process::CommandExt;
|
||||||
|
use std::path::Path;
|
||||||
|
use std::process::{Command, Stdio};
|
||||||
|
use std::sync::atomic::{AtomicBool, Ordering};
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
|
|
||||||
|
use base64::Engine;
|
||||||
|
use serde_json::{json, Value};
|
||||||
|
|
||||||
|
const PORT: u32 = 9001;
|
||||||
|
/// Guest-side egress proxy. The VM has **no network interface at all** — see
|
||||||
|
/// `microvm.rs`, whose machine config declares no `network-interfaces` — so an
|
||||||
|
/// agent CLI cannot reach the model API on its own. It reaches it by honouring
|
||||||
|
/// `HTTPS_PROXY`, which is measured, not assumed: with the proxy pointed at a
|
||||||
|
/// closed port, `claude -p` fails with `ConnectionRefused` instead of answering.
|
||||||
|
///
|
||||||
|
/// This listener is a dumb byte pump. It parses nothing and enforces nothing:
|
||||||
|
/// the `CONNECT` request travels verbatim to the host, which speaks HTTP CONNECT
|
||||||
|
/// and owns the allow-list. Keeping policy on the host means nothing running in
|
||||||
|
/// the guest — including a compromised agent — can talk it into a different
|
||||||
|
/// answer.
|
||||||
|
const PROXY_PORT: u16 = 3128;
|
||||||
|
/// Host-side vsock port the tunnel lands on. Firecracker's convention for a
|
||||||
|
/// guest-initiated connection is that the HOST listens on `<uds_path>_<port>`.
|
||||||
|
const EGRESS_PORT: u32 = 9002;
|
||||||
|
/// Guest-side port for a LOCALLY HOSTED model, and the vsock port it lands on.
|
||||||
|
///
|
||||||
|
/// Separate from the egress proxy on purpose, and simpler than it. The egress
|
||||||
|
/// path exists to let an agent reach the public internet under an allow-list;
|
||||||
|
/// this one reaches exactly one thing — the Ollama the node itself is running,
|
||||||
|
/// on its own loopback — and can reach nothing else, because the host end is a
|
||||||
|
/// pipe to a fixed address rather than a proxy that takes a destination.
|
||||||
|
///
|
||||||
|
/// It therefore needs no `CONNECT`, no TLS and no allow-list. The bytes travel
|
||||||
|
/// guest loopback → vsock → host loopback and never touch a network, so there is
|
||||||
|
/// nothing on a wire for TLS to protect. `NO_PROXY` already contains
|
||||||
|
/// `127.0.0.1`, so an agent pointed at `http://127.0.0.1:11434` bypasses the
|
||||||
|
/// egress proxy entirely rather than trying to CONNECT through it.
|
||||||
|
///
|
||||||
|
/// The guest always listens. Whether anything answers is the HOST's decision:
|
||||||
|
/// the node only binds the vsock end for a backend that is meant to have a
|
||||||
|
/// local model, so on every other backend this port simply refuses.
|
||||||
|
const MODEL_PORT: u16 = 11434;
|
||||||
|
const MODEL_VSOCK_PORT: u32 = 9003;
|
||||||
|
/// `VMADDR_CID_HOST` — the hypervisor side of the vsock.
|
||||||
|
const HOST_CID: u32 = 2;
|
||||||
|
|
||||||
|
/// Whether the egress proxy is actually listening. Reported by `ping` so the
|
||||||
|
/// host can refuse to hand a mission to a VM with no way out, rather than
|
||||||
|
/// discovering it as an agent that hangs.
|
||||||
|
static PROXY_UP: AtomicBool = AtomicBool::new(false);
|
||||||
|
/// Cap on a single request. A hostile or broken host must not be able to make
|
||||||
|
/// pid 1 allocate without bound and get the VM OOM-killed.
|
||||||
|
const MAX_REQUEST: u32 = 512 * 1024 * 1024;
|
||||||
|
|
||||||
|
const B64: base64::engine::general_purpose::GeneralPurpose =
|
||||||
|
base64::engine::general_purpose::STANDARD;
|
||||||
|
|
||||||
|
fn main() {
|
||||||
|
// The mounts the init script would otherwise do. Done here so the agent
|
||||||
|
// works whether it is exec'd from a shell init or used as `init=` directly:
|
||||||
|
// /proc missing makes every process-inspecting tool in the guest lie.
|
||||||
|
for (fstype, target) in [
|
||||||
|
("proc", "/proc"),
|
||||||
|
("sysfs", "/sys"),
|
||||||
|
("devtmpfs", "/dev"),
|
||||||
|
("tmpfs", "/tmp"),
|
||||||
|
] {
|
||||||
|
if !Path::new(target).join(".").exists() {
|
||||||
|
let _ = std::fs::create_dir_all(target);
|
||||||
|
}
|
||||||
|
let _ = Command::new("mount")
|
||||||
|
.args(["-t", fstype, fstype, target])
|
||||||
|
.status();
|
||||||
|
}
|
||||||
|
|
||||||
|
start_egress_proxy();
|
||||||
|
|
||||||
|
let listener = match vsock::VsockListener::bind_with_cid_port(libc_vmaddr_cid_any(), PORT) {
|
||||||
|
Ok(l) => l,
|
||||||
|
Err(e) => {
|
||||||
|
// Printed to the console, which is where the host's boot check
|
||||||
|
// looks. Exiting pid 1 panics the kernel, which is the honest
|
||||||
|
// outcome: a VM whose agent cannot listen is unusable, and it must
|
||||||
|
// not sit there looking booted.
|
||||||
|
eprintln!("FC-AGENT-FATAL could not bind vsock port {PORT}: {e}");
|
||||||
|
std::process::exit(1);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// The host greps the console for this before it tries to connect.
|
||||||
|
println!("FC-AGENT-LISTENING port={PORT}");
|
||||||
|
let _ = std::io::stdout().flush();
|
||||||
|
|
||||||
|
for conn in listener.incoming() {
|
||||||
|
match conn {
|
||||||
|
Ok(mut s) => {
|
||||||
|
// One THREAD per connection, not one at a time.
|
||||||
|
//
|
||||||
|
// This loop used to call `serve_one` inline, which meant the
|
||||||
|
// agent accepted nothing while an op was running. A mission turn
|
||||||
|
// is an `exec` that can last an hour, so for that hour the guest
|
||||||
|
// was unreachable: the host could not tail its output, probe it,
|
||||||
|
// or ask it anything. Every existing probe runs AFTER the turn
|
||||||
|
// for exactly this reason.
|
||||||
|
//
|
||||||
|
// A thread rather than async: this is a static musl binary with
|
||||||
|
// no runtime, and the concurrency here is a handful of
|
||||||
|
// connections, not thousands.
|
||||||
|
//
|
||||||
|
// The panic discipline of the old inline call still applies, and
|
||||||
|
// matters MORE now — this process is pid 1, and a panic that
|
||||||
|
// unwound out of a worker used to take the accept loop with it.
|
||||||
|
// `catch_unwind` keeps a bad request from killing the VM.
|
||||||
|
std::thread::Builder::new()
|
||||||
|
.name("fcagent-conn".into())
|
||||||
|
.spawn(move || {
|
||||||
|
let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||||
|
serve_one(&mut s)
|
||||||
|
}));
|
||||||
|
match r {
|
||||||
|
Ok(Err(e)) => eprintln!("FC-AGENT-ERROR {e}"),
|
||||||
|
Err(_) => eprintln!("FC-AGENT-ERROR handler panicked"),
|
||||||
|
Ok(Ok(())) => {}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.map(|_| ())
|
||||||
|
.unwrap_or_else(|e| {
|
||||||
|
// Out of threads: answer nothing on this connection, but
|
||||||
|
// keep accepting. Dropping the listener would brick the VM.
|
||||||
|
eprintln!("FC-AGENT-ERROR spawn: {e}");
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Err(e) => eprintln!("FC-AGENT-ERROR accept: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `VMADDR_CID_ANY` — bind for any host CID.
|
||||||
|
fn libc_vmaddr_cid_any() -> u32 {
|
||||||
|
u32::MAX
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bring up loopback and start the egress tunnel.
|
||||||
|
///
|
||||||
|
/// Loopback is not optional and not free: the guest's `lo` exists but starts
|
||||||
|
/// **down**, and while it is down a listener on 127.0.0.1 *binds successfully*
|
||||||
|
/// and then refuses every connection with `ENETUNREACH`. A bind-only check would
|
||||||
|
/// have reported a working proxy. So `lo` goes up first, via `ip` — which is why
|
||||||
|
/// `iproute2` is in the agent images.
|
||||||
|
///
|
||||||
|
/// Failure here is recorded, not fatal: exec still works, so a VM is still
|
||||||
|
/// useful for work that needs no network. It is reported through `ping` so the
|
||||||
|
/// host can decide, instead of a mission discovering it as an agent that hangs.
|
||||||
|
fn start_egress_proxy() {
|
||||||
|
// Absolute paths, not `Command::new("ip")`. This process is pid 1, so its
|
||||||
|
// PATH is whatever the kernel handed it — and when PATH is unset, `execvp`
|
||||||
|
// falls back to a default that does NOT include `/usr/sbin`, which is exactly
|
||||||
|
// where Debian puts `ip`. Searching by name would fail on an image that has
|
||||||
|
// it, and the symptom would be a VM with no egress and no explanation.
|
||||||
|
const IP_CANDIDATES: &[&str] = &["/usr/sbin/ip", "/sbin/ip", "/usr/bin/ip", "/bin/ip"];
|
||||||
|
let Some(ip) = IP_CANDIDATES.iter().find(|p| Path::new(p).exists()) else {
|
||||||
|
eprintln!(
|
||||||
|
"FC-AGENT-NO-PROXY no `ip` binary in {IP_CANDIDATES:?} — no egress; \
|
||||||
|
add iproute2 to this image"
|
||||||
|
);
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
match Command::new(ip).args(["link", "set", "lo", "up"]).status() {
|
||||||
|
Ok(s) if s.success() => {}
|
||||||
|
other => {
|
||||||
|
eprintln!("FC-AGENT-NO-PROXY `{ip} link set lo up` failed ({other:?}) — no egress");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let listener = match TcpListener::bind(("127.0.0.1", PROXY_PORT)) {
|
||||||
|
Ok(l) => l,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("FC-AGENT-NO-PROXY could not listen on 127.0.0.1:{PROXY_PORT}: {e}");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
PROXY_UP.store(true, Ordering::Relaxed);
|
||||||
|
println!("FC-AGENT-PROXY listening on 127.0.0.1:{PROXY_PORT} -> vsock {EGRESS_PORT}");
|
||||||
|
let _ = std::io::stdout().flush();
|
||||||
|
pump(listener, EGRESS_PORT, "PROXY");
|
||||||
|
|
||||||
|
// The local-model port. Failure to bind is reported and non-fatal, exactly
|
||||||
|
// like the egress proxy: a VM whose backend does not use a local model is
|
||||||
|
// still perfectly useful, and a fatal error here would take out every
|
||||||
|
// backend to serve one.
|
||||||
|
match TcpListener::bind(("127.0.0.1", MODEL_PORT)) {
|
||||||
|
Ok(l) => {
|
||||||
|
println!("FC-AGENT-MODEL listening on 127.0.0.1:{MODEL_PORT} -> vsock {MODEL_VSOCK_PORT}");
|
||||||
|
let _ = std::io::stdout().flush();
|
||||||
|
pump(l, MODEL_VSOCK_PORT, "MODEL");
|
||||||
|
}
|
||||||
|
Err(e) => eprintln!("FC-AGENT-NO-MODEL could not listen on 127.0.0.1:{MODEL_PORT}: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Accept forever, splicing each connection onto its own vsock stream.
|
||||||
|
fn pump(listener: TcpListener, vsock_port: u32, tag: &'static str) {
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
for c in listener.incoming() {
|
||||||
|
match c {
|
||||||
|
// One thread per connection. An agent CLI opens several at once,
|
||||||
|
// and serving them in sequence would look like a hang.
|
||||||
|
Ok(tcp) => {
|
||||||
|
std::thread::spawn(move || {
|
||||||
|
if let Err(e) = tunnel(tcp, vsock_port) {
|
||||||
|
eprintln!("FC-AGENT-{tag}-ERROR {e}");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Err(e) => eprintln!("FC-AGENT-{tag}-ERROR accept: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Splice one TCP connection onto a fresh vsock connection to the host.
|
||||||
|
///
|
||||||
|
/// No parsing: whatever the client sent — `CONNECT host:443`, or an absolute-form
|
||||||
|
/// request — is the host's business. The host answers with real HTTP, so a
|
||||||
|
/// refusal reaches the client as a status code rather than a dropped socket.
|
||||||
|
fn tunnel(tcp: std::net::TcpStream, vsock_port: u32) -> Result<(), String> {
|
||||||
|
let vs = vsock::VsockStream::connect_with_cid_port(HOST_CID, vsock_port)
|
||||||
|
.map_err(|e| format!("vsock connect to host:{vsock_port}: {e}"))?;
|
||||||
|
|
||||||
|
let (mut tcp_r, mut tcp_w) = (
|
||||||
|
tcp.try_clone().map_err(|e| format!("clone tcp: {e}"))?,
|
||||||
|
tcp,
|
||||||
|
);
|
||||||
|
let (mut vs_r, mut vs_w) = (
|
||||||
|
vs.try_clone().map_err(|e| format!("clone vsock: {e}"))?,
|
||||||
|
vs,
|
||||||
|
);
|
||||||
|
|
||||||
|
// Each direction gets its own thread, and each shuts its peer's write side
|
||||||
|
// down when it ends. Without the shutdown the other half blocks forever on a
|
||||||
|
// half-closed connection and the CLI waits out its own timeout.
|
||||||
|
let up = std::thread::spawn(move || {
|
||||||
|
let _ = std::io::copy(&mut tcp_r, &mut vs_w);
|
||||||
|
let _ = vs_w.shutdown(std::net::Shutdown::Write);
|
||||||
|
});
|
||||||
|
let _ = std::io::copy(&mut vs_r, &mut tcp_w);
|
||||||
|
let _ = tcp_w.shutdown(std::net::Shutdown::Write);
|
||||||
|
let _ = up.join();
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn serve_one(s: &mut vsock::VsockStream) -> Result<(), String> {
|
||||||
|
let mut len = [0u8; 4];
|
||||||
|
s.read_exact(&mut len)
|
||||||
|
.map_err(|e| format!("read length: {e}"))?;
|
||||||
|
let len = u32::from_be_bytes(len);
|
||||||
|
if len > MAX_REQUEST {
|
||||||
|
// Answer rather than hang up: a caller that sent something absurd needs
|
||||||
|
// to be told, not left waiting for a reply that will never come.
|
||||||
|
return reply(s, &json!({ "ok": false, "error": format!("request of {len} bytes exceeds the {MAX_REQUEST} cap") }));
|
||||||
|
}
|
||||||
|
let mut buf = vec![0u8; len as usize];
|
||||||
|
s.read_exact(&mut buf)
|
||||||
|
.map_err(|e| format!("read body: {e}"))?;
|
||||||
|
|
||||||
|
let req = match serde_json::from_slice::<Value>(&buf) {
|
||||||
|
Ok(req) => req,
|
||||||
|
Err(e) => {
|
||||||
|
return reply(
|
||||||
|
s,
|
||||||
|
&json!({ "ok": false, "error": format!("undecodable request: {e}") }),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
};
|
||||||
|
// `tail` owns the connection for its lifetime, emitting a frame per chunk,
|
||||||
|
// so it cannot go through `handle`, which returns one Value.
|
||||||
|
if req.get("op").and_then(Value::as_str) == Some("tail") {
|
||||||
|
return op_tail(s, &req);
|
||||||
|
}
|
||||||
|
let resp = handle(&req);
|
||||||
|
reply(s, &resp)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stream a file to the host as it grows, one framed JSON chunk at a time.
|
||||||
|
///
|
||||||
|
/// This is how a mission turn's stdout/stderr reaches the platform while the
|
||||||
|
/// turn is still running. The turn writes to a log file (`… 2>&1 | tee`), and
|
||||||
|
/// the host opens a second connection to follow it — which only works because
|
||||||
|
/// the accept loop above is now threaded.
|
||||||
|
///
|
||||||
|
/// `from` lets the host resume without replaying: it reconnects with the offset
|
||||||
|
/// it last saw. Following by OFFSET rather than by holding one connection open
|
||||||
|
/// forever is what makes a dropped link cheap.
|
||||||
|
///
|
||||||
|
/// Ends when the file stops growing for `idle_ms`, or at `max_secs`. It must
|
||||||
|
/// end: a tail that never returns pins a thread for the life of the VM.
|
||||||
|
fn op_tail(s: &mut vsock::VsockStream, req: &Value) -> Result<(), String> {
|
||||||
|
use std::io::{Seek, SeekFrom};
|
||||||
|
|
||||||
|
let path = req.get("path").and_then(Value::as_str).unwrap_or_default();
|
||||||
|
let mut from = req.get("from").and_then(Value::as_u64).unwrap_or(0);
|
||||||
|
let idle_ms = req.get("idle_ms").and_then(Value::as_u64).unwrap_or(2_000);
|
||||||
|
let max_secs = req.get("max_secs").and_then(Value::as_u64).unwrap_or(3_600);
|
||||||
|
|
||||||
|
let started = std::time::Instant::now();
|
||||||
|
let mut last_data = std::time::Instant::now();
|
||||||
|
loop {
|
||||||
|
if started.elapsed().as_secs() >= max_secs {
|
||||||
|
return reply(s, &json!({ "ok": true, "eof": true, "at": from, "reason": "max_secs" }));
|
||||||
|
}
|
||||||
|
let mut f = match std::fs::File::open(path) {
|
||||||
|
Ok(f) => f,
|
||||||
|
// Not an error: the turn may not have created the log yet.
|
||||||
|
Err(_) => {
|
||||||
|
if last_data.elapsed().as_millis() as u64 >= idle_ms {
|
||||||
|
return reply(s, &json!({ "ok": true, "eof": true, "at": from, "reason": "absent" }));
|
||||||
|
}
|
||||||
|
std::thread::sleep(std::time::Duration::from_millis(200));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let len = f.metadata().map(|m| m.len()).unwrap_or(0);
|
||||||
|
if len < from {
|
||||||
|
// Truncated or rotated under us. Restart rather than read garbage.
|
||||||
|
from = 0;
|
||||||
|
}
|
||||||
|
if len > from {
|
||||||
|
f.seek(SeekFrom::Start(from))
|
||||||
|
.map_err(|e| format!("seek {path}: {e}"))?;
|
||||||
|
let mut buf = vec![0u8; (len - from).min(MAX_CHUNK) as usize];
|
||||||
|
let n = f.read(&mut buf).map_err(|e| format!("read {path}: {e}"))?;
|
||||||
|
buf.truncate(n);
|
||||||
|
from += n as u64;
|
||||||
|
last_data = std::time::Instant::now();
|
||||||
|
// Base64 so arbitrary bytes survive JSON — agent output is not
|
||||||
|
// guaranteed to be valid UTF-8 mid-chunk.
|
||||||
|
reply(
|
||||||
|
s,
|
||||||
|
&json!({ "ok": true, "eof": false, "at": from, "data": B64.encode(&buf) }),
|
||||||
|
)?;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if last_data.elapsed().as_millis() as u64 >= idle_ms {
|
||||||
|
return reply(s, &json!({ "ok": true, "eof": true, "at": from, "reason": "idle" }));
|
||||||
|
}
|
||||||
|
std::thread::sleep(std::time::Duration::from_millis(200));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Largest slice sent in one frame. Bounded so a burst of output cannot
|
||||||
|
/// allocate without limit inside a 2 GiB guest.
|
||||||
|
const MAX_CHUNK: u64 = 256 * 1024;
|
||||||
|
|
||||||
|
fn reply(s: &mut vsock::VsockStream, v: &Value) -> Result<(), String> {
|
||||||
|
let body = serde_json::to_vec(v).map_err(|e| format!("encode reply: {e}"))?;
|
||||||
|
s.write_all(&(body.len() as u32).to_be_bytes())
|
||||||
|
.map_err(|e| format!("write length: {e}"))?;
|
||||||
|
s.write_all(&body)
|
||||||
|
.map_err(|e| format!("write body: {e}"))?;
|
||||||
|
s.flush().map_err(|e| format!("flush: {e}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn handle(req: &Value) -> Value {
|
||||||
|
let op = req.get("op").and_then(Value::as_str).unwrap_or_default();
|
||||||
|
match op {
|
||||||
|
"ping" => json!({
|
||||||
|
"ok": true,
|
||||||
|
"pid": std::process::id(),
|
||||||
|
// The host refuses to run a mission in a VM with no way out; this is
|
||||||
|
// how it knows. Reported rather than assumed because the image, not
|
||||||
|
// this binary, decides whether loopback can come up.
|
||||||
|
"proxy": PROXY_UP.load(Ordering::Relaxed),
|
||||||
|
}),
|
||||||
|
"exec" => op_exec(req),
|
||||||
|
// `tail` is handled in `serve_one`, not here: it streams many frames
|
||||||
|
// over one connection and so cannot return a single Value.
|
||||||
|
"tail" => json!({ "ok": false, "error": "tail is streamed; handled by serve_one" }),
|
||||||
|
"put" => op_put(req),
|
||||||
|
"get" => op_get(req),
|
||||||
|
other => json!({ "ok": false, "error": format!("unknown op: {other}") }),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Extra environment for the command, on top of the image's own.
|
||||||
|
///
|
||||||
|
/// This is how credentials reach the agent CLI. An env var rather than a file
|
||||||
|
/// because the per-VM rootfs is destroyed with the VM but an env var never
|
||||||
|
/// touches the guest disk at all — it exists only in the process's environment
|
||||||
|
/// for the length of one exec.
|
||||||
|
///
|
||||||
|
/// **Every problem here fails the exec.** The tempting alternative — skip the
|
||||||
|
/// entry we could not use and run anyway — produces a `claude -p` with no
|
||||||
|
/// credential, and that does not error: it hangs. A phase stuck at `running`
|
||||||
|
/// for ten minutes with nothing in the logs is exactly what a missing token
|
||||||
|
/// looked like on the container path, so a request we cannot honour in full is
|
||||||
|
/// refused with a reason instead.
|
||||||
|
///
|
||||||
|
/// Errors name the key and never the value: the value is the secret, and an
|
||||||
|
/// error string travels back over the wire and into logs.
|
||||||
|
fn env_pairs(req: &Value) -> Result<Vec<(String, String)>, String> {
|
||||||
|
// Absent or `null` means the caller sent no variables of its own — which is
|
||||||
|
// NOT the same as "this command needs no environment". Both cases still get
|
||||||
|
// the proxy address below; returning early here meant every exec that passed
|
||||||
|
// no env ran with no HTTPS_PROXY, and the symptom was `curl` reporting
|
||||||
|
// "Could not resolve host" from a guest that had a working tunnel.
|
||||||
|
let empty = serde_json::Map::new();
|
||||||
|
let map = match req.get("env") {
|
||||||
|
None => &empty,
|
||||||
|
Some(v) if v.is_null() => &empty,
|
||||||
|
// Anything else that is not an object is a caller bug.
|
||||||
|
Some(v) => v
|
||||||
|
.as_object()
|
||||||
|
.ok_or("exec env must be an object of name → string")?,
|
||||||
|
};
|
||||||
|
let mut out = Vec::with_capacity(map.len() + 3);
|
||||||
|
for (k, v) in map {
|
||||||
|
let Some(val) = v.as_str() else {
|
||||||
|
return Err(format!("exec env {k}: value must be a string"));
|
||||||
|
};
|
||||||
|
// `putenv` semantics: a name containing '=' would be parsed as part of
|
||||||
|
// the value, silently defining a different variable than the one asked
|
||||||
|
// for. A NUL truncates at the C boundary, for the same class of reason.
|
||||||
|
if k.is_empty() {
|
||||||
|
return Err("exec env has an empty variable name".into());
|
||||||
|
}
|
||||||
|
if k.contains('=') || k.contains('\0') {
|
||||||
|
return Err(format!("exec env {k:?}: name may not contain '=' or NUL"));
|
||||||
|
}
|
||||||
|
if val.contains('\0') {
|
||||||
|
return Err(format!("exec env {k}: value may not contain NUL"));
|
||||||
|
}
|
||||||
|
out.push((k.clone(), val.to_string()));
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(with_proxy_env(out, PROXY_UP.load(Ordering::Relaxed)))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Add the proxy variables the guest's own listener serves.
|
||||||
|
///
|
||||||
|
/// The agent runs the proxy, so the agent declares where it is. Deriving this on
|
||||||
|
/// the host would mean two places agreeing on a port number, and the one that
|
||||||
|
/// drifts is the one nobody tests.
|
||||||
|
///
|
||||||
|
/// Explicit caller values win: a caller can still point a command elsewhere or
|
||||||
|
/// switch the proxy off for it. Matched case-insensitively because the lowercase
|
||||||
|
/// spellings are equally conventional and a duplicate would leave which one
|
||||||
|
/// applies up to the shell.
|
||||||
|
fn with_proxy_env(mut env: Vec<(String, String)>, proxy_up: bool) -> Vec<(String, String)> {
|
||||||
|
if !proxy_up {
|
||||||
|
return env;
|
||||||
|
}
|
||||||
|
let addr = format!("http://127.0.0.1:{PROXY_PORT}");
|
||||||
|
for (k, v) in [
|
||||||
|
("HTTPS_PROXY", addr.as_str()),
|
||||||
|
("HTTP_PROXY", addr.as_str()),
|
||||||
|
// Without this the client would ask the proxy to reach the proxy.
|
||||||
|
("NO_PROXY", "localhost,127.0.0.1"),
|
||||||
|
] {
|
||||||
|
// `eq_ignore_ascii_case` covers the lowercase spelling, which is equally
|
||||||
|
// conventional; setting both would leave which one applies to the client.
|
||||||
|
if !env.iter().any(|(have, _)| have.eq_ignore_ascii_case(k)) {
|
||||||
|
env.push((k.to_string(), v.to_string()));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
env
|
||||||
|
}
|
||||||
|
|
||||||
|
fn op_exec(req: &Value) -> Value {
|
||||||
|
let cmd = req.get("cmd").and_then(Value::as_str).unwrap_or_default();
|
||||||
|
if cmd.is_empty() {
|
||||||
|
return json!({ "ok": false, "error": "exec needs a cmd" });
|
||||||
|
}
|
||||||
|
let cwd = req.get("cwd").and_then(Value::as_str).unwrap_or("/");
|
||||||
|
let secs = req.get("timeout").and_then(Value::as_u64).unwrap_or(3600);
|
||||||
|
let env = match env_pairs(req) {
|
||||||
|
Ok(v) => v,
|
||||||
|
Err(e) => return json!({ "ok": false, "error": e }),
|
||||||
|
};
|
||||||
|
|
||||||
|
// The image's ENV was written to /etc/profile.d by the rootfs builder;
|
||||||
|
// `sh -c` does not read it, so source it here — otherwise a CLI that relies
|
||||||
|
// on `ENV PATH` behaves differently in the VM than in the container, which
|
||||||
|
// is exactly the drift the builder extracted that file to prevent.
|
||||||
|
//
|
||||||
|
// The `if [ -f ]` guard is load-bearing. `. missing-file` makes a
|
||||||
|
// NON-INTERACTIVE POSIX shell exit immediately with status 1, so the naive
|
||||||
|
// `. env.sh 2>/dev/null; cmd` returned rc=1 without running `cmd` at all on
|
||||||
|
// any rootfs lacking that file — every exec silently failing while looking
|
||||||
|
// like an ordinary non-zero exit. Caught by the exit-7 unit test.
|
||||||
|
const ENV_FILE: &str = "/etc/profile.d/00-image-env.sh";
|
||||||
|
let sourced = format!("if [ -f {ENV_FILE} ]; then . {ENV_FILE}; fi\n{cmd}");
|
||||||
|
let mut c = Command::new("/bin/sh");
|
||||||
|
c.arg("-c")
|
||||||
|
.arg(&sourced)
|
||||||
|
.envs(env)
|
||||||
|
.current_dir(if Path::new(cwd).is_dir() { cwd } else { "/" })
|
||||||
|
.stdin(Stdio::null())
|
||||||
|
.stdout(Stdio::piped())
|
||||||
|
.stderr(Stdio::piped())
|
||||||
|
// A new process group so a command that spawns background children can
|
||||||
|
// be killed wholesale. Without it a stray daemon keeps the run alive and
|
||||||
|
// the host's timeout is the only thing that ends it.
|
||||||
|
.process_group(0);
|
||||||
|
|
||||||
|
let mut child = match c.spawn() {
|
||||||
|
Ok(ch) => ch,
|
||||||
|
Err(e) => return json!({ "ok": false, "error": format!("spawn: {e}") }),
|
||||||
|
};
|
||||||
|
let pid = child.id() as i32;
|
||||||
|
|
||||||
|
// std has no wait-with-timeout, so poll. The output pipes are read after
|
||||||
|
// the wait, which is safe here because a command producing more than a pipe
|
||||||
|
// buffer of output while we are not draining it would deadlock — so the
|
||||||
|
// deadline is enforced by killing the group, and the pipes are drained by
|
||||||
|
// `wait_with_output` immediately after.
|
||||||
|
let deadline = Instant::now() + Duration::from_secs(secs);
|
||||||
|
let timed_out = loop {
|
||||||
|
match child.try_wait() {
|
||||||
|
Ok(Some(_)) => break false,
|
||||||
|
Ok(None) => {}
|
||||||
|
Err(e) => return json!({ "ok": false, "error": format!("wait: {e}") }),
|
||||||
|
}
|
||||||
|
if Instant::now() >= deadline {
|
||||||
|
kill_group(pid);
|
||||||
|
break true;
|
||||||
|
}
|
||||||
|
std::thread::sleep(Duration::from_millis(20));
|
||||||
|
};
|
||||||
|
|
||||||
|
let out = match child.wait_with_output() {
|
||||||
|
Ok(o) => o,
|
||||||
|
Err(e) => return json!({ "ok": false, "error": format!("collect output: {e}") }),
|
||||||
|
};
|
||||||
|
if timed_out {
|
||||||
|
// Reported as ok:false, not as rc=124: "we stopped it" is a different
|
||||||
|
// fact from "it exited non-zero", and the caller must be able to tell.
|
||||||
|
return json!({
|
||||||
|
"ok": false,
|
||||||
|
"error": format!("command exceeded its {secs}s budget and was killed"),
|
||||||
|
"stdout": String::from_utf8_lossy(&out.stdout),
|
||||||
|
"stderr": String::from_utf8_lossy(&out.stderr),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
json!({
|
||||||
|
"ok": true,
|
||||||
|
// A signalled process has no exit code; report the conventional
|
||||||
|
// 128+signal rather than silently claiming success.
|
||||||
|
"rc": exit_code(&out.status),
|
||||||
|
"stdout": String::from_utf8_lossy(&out.stdout),
|
||||||
|
"stderr": String::from_utf8_lossy(&out.stderr),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn exit_code(status: &std::process::ExitStatus) -> i32 {
|
||||||
|
use std::os::unix::process::ExitStatusExt;
|
||||||
|
status
|
||||||
|
.code()
|
||||||
|
.unwrap_or_else(|| 128 + status.signal().unwrap_or(0))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn kill_group(pid: i32) {
|
||||||
|
let _ = Command::new("kill")
|
||||||
|
.args(["-9", "--", &format!("-{pid}")])
|
||||||
|
.status();
|
||||||
|
}
|
||||||
|
|
||||||
|
fn op_put(req: &Value) -> Value {
|
||||||
|
let dest = req.get("dest").and_then(Value::as_str).unwrap_or_default();
|
||||||
|
if dest.is_empty() {
|
||||||
|
return json!({ "ok": false, "error": "put needs a dest" });
|
||||||
|
}
|
||||||
|
let b64 = req.get("tar_b64").and_then(Value::as_str).unwrap_or_default();
|
||||||
|
let raw = match B64.decode(b64) {
|
||||||
|
Ok(r) => r,
|
||||||
|
Err(e) => return json!({ "ok": false, "error": format!("undecodable archive: {e}") }),
|
||||||
|
};
|
||||||
|
if let Err(e) = std::fs::create_dir_all(dest) {
|
||||||
|
return json!({ "ok": false, "error": format!("mkdir {dest}: {e}") });
|
||||||
|
}
|
||||||
|
let mut ar = tar::Archive::new(&raw[..]);
|
||||||
|
ar.set_overwrite(true);
|
||||||
|
// Ownership from the host archive is meaningless in here and re-applying it
|
||||||
|
// is how the container path grew a uid split. The guest is root; let it own
|
||||||
|
// what it is given.
|
||||||
|
ar.set_preserve_permissions(false);
|
||||||
|
match ar.unpack(dest) {
|
||||||
|
Ok(()) => json!({ "ok": true, "dest": dest, "bytes": raw.len() }),
|
||||||
|
Err(e) => json!({ "ok": false, "error": format!("unpack into {dest}: {e}") }),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Recursive tar append that skips excluded directory NAMES at any depth.
|
||||||
|
///
|
||||||
|
/// Hand-rolled because `tar::Builder::append_dir_all` takes no filter. Matched on
|
||||||
|
/// the name rather than a path prefix: a workspace has a `target/` per crate, and
|
||||||
|
/// excluding only the root one still ships the rest.
|
||||||
|
fn append_filtered<W: Write>(
|
||||||
|
b: &mut tar::Builder<W>,
|
||||||
|
dir: &Path,
|
||||||
|
prefix: &Path,
|
||||||
|
exclude: &[String],
|
||||||
|
) -> std::io::Result<()> {
|
||||||
|
b.append_dir(prefix, dir)?;
|
||||||
|
let mut entries: Vec<_> = std::fs::read_dir(dir)?.collect::<Result<Vec<_>, _>>()?;
|
||||||
|
entries.sort_by_key(|e| e.file_name());
|
||||||
|
for entry in entries {
|
||||||
|
let name = entry.file_name();
|
||||||
|
let name_str = name.to_string_lossy().to_string();
|
||||||
|
let path = entry.path();
|
||||||
|
let dest = prefix.join(&name);
|
||||||
|
let meta = std::fs::symlink_metadata(&path)?;
|
||||||
|
if meta.is_dir() {
|
||||||
|
if exclude.contains(&name_str) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
append_filtered(b, &path, &dest, exclude)?;
|
||||||
|
} else if meta.is_symlink() {
|
||||||
|
let mut header = tar::Header::new_gnu();
|
||||||
|
header.set_metadata(&meta);
|
||||||
|
header.set_entry_type(tar::EntryType::Symlink);
|
||||||
|
header.set_size(0);
|
||||||
|
let target = std::fs::read_link(&path)?;
|
||||||
|
b.append_link(&mut header, &dest, &target)?;
|
||||||
|
} else {
|
||||||
|
let mut f = std::fs::File::open(&path)?;
|
||||||
|
b.append_file(&dest, &mut f)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn op_get(req: &Value) -> Value {
|
||||||
|
let path = req.get("path").and_then(Value::as_str).unwrap_or_default();
|
||||||
|
if path.is_empty() {
|
||||||
|
return json!({ "ok": false, "error": "get needs a path" });
|
||||||
|
}
|
||||||
|
let p = Path::new(path);
|
||||||
|
if !p.exists() {
|
||||||
|
// A missing path is an error, NOT an empty archive — an empty tar looks
|
||||||
|
// exactly like a run that produced nothing.
|
||||||
|
return json!({ "ok": false, "error": format!("no such path: {path}") });
|
||||||
|
}
|
||||||
|
let name = p
|
||||||
|
.file_name()
|
||||||
|
.map(|s| s.to_string_lossy().to_string())
|
||||||
|
.unwrap_or_else(|| "root".to_string());
|
||||||
|
|
||||||
|
// Directory names to leave out, sent by the host so the policy lives in one
|
||||||
|
// place (`mission_fs::transport_excludes`). Without it a phase that ran
|
||||||
|
// `cargo test` tars its whole `target/` directory: measured at 8.9 MB of 9.4 MB
|
||||||
|
// on our scratch repo, and enough to blow the 300s collect budget on a real
|
||||||
|
// build — which stranded a finished mission's work inside a VM twice.
|
||||||
|
let exclude: Vec<String> = req
|
||||||
|
.get("exclude")
|
||||||
|
.and_then(Value::as_array)
|
||||||
|
.map(|a| {
|
||||||
|
a.iter()
|
||||||
|
.filter_map(Value::as_str)
|
||||||
|
.map(str::to_string)
|
||||||
|
.collect()
|
||||||
|
})
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
let mut b = tar::Builder::new(Vec::new());
|
||||||
|
// Do not follow symlinks: a link pointing outside the collected tree would
|
||||||
|
// otherwise be dereferenced and its target smuggled back to the host.
|
||||||
|
b.follow_symlinks(false);
|
||||||
|
let added = if p.is_dir() {
|
||||||
|
append_filtered(&mut b, p, Path::new(&name), &exclude)
|
||||||
|
} else {
|
||||||
|
b.append_path_with_name(p, &name)
|
||||||
|
};
|
||||||
|
if let Err(e) = added {
|
||||||
|
return json!({ "ok": false, "error": format!("archive {path}: {e}") });
|
||||||
|
}
|
||||||
|
match b.into_inner() {
|
||||||
|
Ok(bytes) => json!({ "ok": true, "tar_b64": B64.encode(&bytes), "bytes": bytes.len() }),
|
||||||
|
Err(e) => json!({ "ok": false, "error": format!("finish archive for {path}: {e}") }),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
/// The tail loop must terminate. A tail that never returns pins a thread for
|
||||||
|
/// the life of the VM, and pid 1 running out of threads is an unbootable
|
||||||
|
/// machine, not a missing log.
|
||||||
|
#[test]
|
||||||
|
fn a_tail_of_a_file_that_never_appears_still_ends() {
|
||||||
|
// `absent` + idle_ms elapsed is the terminating branch; assert the
|
||||||
|
// constants that make it reachable rather than spinning a real socket.
|
||||||
|
assert!(MAX_CHUNK > 0, "a zero chunk cap would loop without progress");
|
||||||
|
assert!(
|
||||||
|
MAX_CHUNK <= 1024 * 1024,
|
||||||
|
"chunks must stay small enough for a 2 GiB guest"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The CLI reaches the API only by honouring HTTPS_PROXY (measured: with the
|
||||||
|
/// proxy at a closed port, `claude -p` fails ConnectionRefused instead of
|
||||||
|
/// answering), so a VM whose proxy is up must hand it the address.
|
||||||
|
#[test]
|
||||||
|
fn the_proxy_address_is_declared_when_the_proxy_is_up() {
|
||||||
|
let env = with_proxy_env(vec![], true);
|
||||||
|
let get = |k: &str| {
|
||||||
|
env.iter()
|
||||||
|
.find(|(a, _)| a == k)
|
||||||
|
.map(|(_, v)| v.as_str())
|
||||||
|
.unwrap_or("")
|
||||||
|
};
|
||||||
|
assert_eq!(get("HTTPS_PROXY"), "http://127.0.0.1:3128");
|
||||||
|
assert_eq!(get("HTTP_PROXY"), "http://127.0.0.1:3128");
|
||||||
|
// Otherwise the client asks the proxy to reach the proxy.
|
||||||
|
assert!(get("NO_PROXY").contains("127.0.0.1"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// And a VM with no proxy must not claim one: pointing a CLI at a listener
|
||||||
|
/// that is not there turns "no egress" into a connection error mid-run
|
||||||
|
/// instead of a fact the host can check before it starts.
|
||||||
|
#[test]
|
||||||
|
fn no_proxy_address_is_declared_when_the_proxy_is_down() {
|
||||||
|
assert!(with_proxy_env(vec![], false).is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An explicit value from the caller wins, in either spelling — otherwise
|
||||||
|
/// both would be set and which one applies would be up to the client.
|
||||||
|
#[test]
|
||||||
|
fn an_explicit_proxy_setting_is_not_overridden() {
|
||||||
|
let env = with_proxy_env(
|
||||||
|
vec![("https_proxy".into(), "http://elsewhere:8080".into())],
|
||||||
|
true,
|
||||||
|
);
|
||||||
|
let proxies: Vec<&str> = env
|
||||||
|
.iter()
|
||||||
|
.filter(|(k, _)| k.eq_ignore_ascii_case("https_proxy"))
|
||||||
|
.map(|(_, v)| v.as_str())
|
||||||
|
.collect();
|
||||||
|
assert_eq!(proxies, vec!["http://elsewhere:8080"]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The credential has to actually reach the command. This is the whole
|
||||||
|
/// point of the op, and the failure it prevents is silent: a `claude -p`
|
||||||
|
/// with no token hangs rather than erroring.
|
||||||
|
#[test]
|
||||||
|
fn injected_env_reaches_the_command() {
|
||||||
|
let r = op_exec(&json!({
|
||||||
|
"op": "exec",
|
||||||
|
"cmd": "printf %s \"$CLAUDE_CODE_OAUTH_TOKEN\"",
|
||||||
|
"env": { "CLAUDE_CODE_OAUTH_TOKEN": "sk-test-value" },
|
||||||
|
"timeout": 30,
|
||||||
|
}));
|
||||||
|
assert_eq!(r["rc"], json!(0));
|
||||||
|
assert_eq!(r["stdout"], json!("sk-test-value"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// And it must survive the profile.d sourcing that runs first — a
|
||||||
|
/// credential set on the process and then clobbered by the shell would
|
||||||
|
/// look identical to one that never arrived.
|
||||||
|
#[test]
|
||||||
|
fn injected_env_survives_the_image_env_file() {
|
||||||
|
let r = op_exec(&json!({
|
||||||
|
"op": "exec",
|
||||||
|
"cmd": "printf %s \"$INJECTED_PROBE\"",
|
||||||
|
"env": { "INJECTED_PROBE": "still-here" },
|
||||||
|
"timeout": 30,
|
||||||
|
}));
|
||||||
|
assert_eq!(r["stdout"], json!("still-here"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No env is the ordinary case and must not be an error.
|
||||||
|
#[test]
|
||||||
|
fn absent_or_null_env_is_not_an_error() {
|
||||||
|
for req in [
|
||||||
|
json!({ "op": "exec", "cmd": "true", "timeout": 30 }),
|
||||||
|
json!({ "op": "exec", "cmd": "true", "env": null, "timeout": 30 }),
|
||||||
|
json!({ "op": "exec", "cmd": "true", "env": {}, "timeout": 30 }),
|
||||||
|
] {
|
||||||
|
assert_eq!(op_exec(&req)["rc"], json!(0), "{req}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An env entry we cannot honour fails the whole exec rather than being
|
||||||
|
/// dropped. Running without the credential is the outcome this refuses:
|
||||||
|
/// it does not error, it hangs, which is far harder to diagnose than a
|
||||||
|
/// rejected request.
|
||||||
|
#[test]
|
||||||
|
fn an_unusable_env_entry_fails_the_exec_instead_of_being_skipped() {
|
||||||
|
let cases = [
|
||||||
|
json!({ "A=B": "x" }),
|
||||||
|
json!({ "": "x" }),
|
||||||
|
json!({ "TOKEN": 42 }),
|
||||||
|
json!({ "TOKEN": null }),
|
||||||
|
];
|
||||||
|
for env in cases {
|
||||||
|
let r = op_exec(&json!({
|
||||||
|
"op": "exec", "cmd": "true", "env": env.clone(), "timeout": 30,
|
||||||
|
}));
|
||||||
|
assert_eq!(r["ok"], json!(false), "env {env} should be refused");
|
||||||
|
assert!(r["rc"].is_null(), "nothing ran, so there is no rc: {r}");
|
||||||
|
}
|
||||||
|
// A non-object env is a caller bug, not an empty map.
|
||||||
|
let r = op_exec(&json!({ "op": "exec", "cmd": "true", "env": "TOKEN=x" }));
|
||||||
|
assert_eq!(r["ok"], json!(false));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An error about a credential must not quote the credential: it travels
|
||||||
|
/// back over the wire and into the server's logs.
|
||||||
|
#[test]
|
||||||
|
fn an_env_error_never_echoes_the_value() {
|
||||||
|
let r = op_exec(&json!({
|
||||||
|
"op": "exec", "cmd": "true", "timeout": 30,
|
||||||
|
"env": { "A=B": "super-secret-token" },
|
||||||
|
}));
|
||||||
|
let err = r["error"].as_str().unwrap_or_default();
|
||||||
|
assert!(!err.contains("super-secret-token"), "leaked the value: {err}");
|
||||||
|
assert!(err.contains("A=B"), "should name the key: {err}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_unknown_op_is_reported_not_ignored() {
|
||||||
|
let r = handle(&json!({ "op": "teleport" }));
|
||||||
|
assert_eq!(r["ok"], json!(false));
|
||||||
|
assert!(r["error"].as_str().unwrap().contains("teleport"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn ping_answers() {
|
||||||
|
assert_eq!(handle(&json!({ "op": "ping" }))["ok"], json!(true));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A missing path must be an error, not an empty archive: an empty tar is
|
||||||
|
/// indistinguishable from a run that produced nothing.
|
||||||
|
/// Build output is not work. It is regenerable, it dwarfs the source, and
|
||||||
|
/// tarring it over vsock stranded a finished mission inside a VM twice —
|
||||||
|
/// `vm_collect` timed out at 300s while the agent's three new modules sat in
|
||||||
|
/// the guest. Matched on the directory NAME at any depth, because a workspace
|
||||||
|
/// has a `target/` per crate.
|
||||||
|
#[test]
|
||||||
|
fn excluded_directories_stay_out_of_the_archive_at_any_depth() {
|
||||||
|
let dir = std::env::temp_dir().join(format!("fcagent-ex-{}", std::process::id()));
|
||||||
|
let _ = std::fs::remove_dir_all(&dir);
|
||||||
|
std::fs::create_dir_all(dir.join("src")).unwrap();
|
||||||
|
std::fs::create_dir_all(dir.join("target/debug")).unwrap();
|
||||||
|
std::fs::create_dir_all(dir.join("crates/inner/target")).unwrap();
|
||||||
|
std::fs::write(dir.join("src/lib.rs"), "fn a() {}").unwrap();
|
||||||
|
std::fs::write(dir.join("target/debug/blob"), vec![0u8; 4096]).unwrap();
|
||||||
|
std::fs::write(dir.join("crates/inner/target/blob"), vec![0u8; 4096]).unwrap();
|
||||||
|
std::fs::write(dir.join("crates/inner/keep.rs"), "fn b() {}").unwrap();
|
||||||
|
|
||||||
|
let r = op_get(&json!({
|
||||||
|
"op": "get",
|
||||||
|
"path": dir.to_string_lossy(),
|
||||||
|
"exclude": ["target"],
|
||||||
|
}));
|
||||||
|
assert_eq!(r["ok"], json!(true), "{r}");
|
||||||
|
let bytes = B64.decode(r["tar_b64"].as_str().unwrap()).unwrap();
|
||||||
|
let mut ar = tar::Archive::new(&bytes[..]);
|
||||||
|
let paths: Vec<String> = ar
|
||||||
|
.entries()
|
||||||
|
.unwrap()
|
||||||
|
.filter_map(Result::ok)
|
||||||
|
.map(|e| e.path().unwrap().to_string_lossy().to_string())
|
||||||
|
.collect();
|
||||||
|
let _ = std::fs::remove_dir_all(&dir);
|
||||||
|
|
||||||
|
assert!(paths.iter().any(|p| p.ends_with("src/lib.rs")), "{paths:?}");
|
||||||
|
assert!(paths.iter().any(|p| p.ends_with("inner/keep.rs")), "{paths:?}");
|
||||||
|
assert!(
|
||||||
|
!paths.iter().any(|p| p.contains("target")),
|
||||||
|
"a nested target/ came along: {paths:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No exclude list means everything, so an existing caller is unchanged.
|
||||||
|
#[test]
|
||||||
|
fn without_an_exclude_list_nothing_is_dropped() {
|
||||||
|
let dir = std::env::temp_dir().join(format!("fcagent-noex-{}", std::process::id()));
|
||||||
|
let _ = std::fs::remove_dir_all(&dir);
|
||||||
|
std::fs::create_dir_all(dir.join("target")).unwrap();
|
||||||
|
std::fs::write(dir.join("target/x"), "x").unwrap();
|
||||||
|
let r = op_get(&json!({ "op": "get", "path": dir.to_string_lossy() }));
|
||||||
|
let bytes = B64.decode(r["tar_b64"].as_str().unwrap()).unwrap();
|
||||||
|
let mut ar = tar::Archive::new(&bytes[..]);
|
||||||
|
let n = ar.entries().unwrap().filter_map(Result::ok).count();
|
||||||
|
let _ = std::fs::remove_dir_all(&dir);
|
||||||
|
assert!(n >= 2, "expected the target dir and its file, got {n}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn getting_a_missing_path_is_an_error() {
|
||||||
|
let r = op_get(&json!({ "op": "get", "path": "/definitely/not/here" }));
|
||||||
|
assert_eq!(r["ok"], json!(false));
|
||||||
|
assert!(r["tar_b64"].is_null(), "no archive may be returned");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A command that ran and failed reports `rc`; one we killed reports
|
||||||
|
/// `ok:false`. Collapsing the two would make a timeout look like a build
|
||||||
|
/// failure and vice versa.
|
||||||
|
#[test]
|
||||||
|
fn a_failing_command_reports_rc_and_a_killed_one_does_not() {
|
||||||
|
let r = op_exec(&json!({ "op": "exec", "cmd": "exit 7", "timeout": 30 }));
|
||||||
|
assert_eq!(r["ok"], json!(true), "it ran, so ok is true");
|
||||||
|
assert_eq!(r["rc"], json!(7));
|
||||||
|
|
||||||
|
let r = op_exec(&json!({ "op": "exec", "cmd": "sleep 30", "timeout": 1 }));
|
||||||
|
assert_eq!(r["ok"], json!(false), "we killed it, so ok is false");
|
||||||
|
assert!(r["rc"].is_null(), "a killed command has no exit code");
|
||||||
|
assert!(r["error"].as_str().unwrap().contains("budget"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn exec_needs_a_command() {
|
||||||
|
assert_eq!(op_exec(&json!({ "op": "exec" }))["ok"], json!(false));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A tar must round-trip through put and get.
|
||||||
|
#[test]
|
||||||
|
fn a_tar_round_trips_through_put_and_get() {
|
||||||
|
let tmp = std::env::temp_dir().join(format!("fcagent-test-{}", std::process::id()));
|
||||||
|
let _ = std::fs::remove_dir_all(&tmp);
|
||||||
|
|
||||||
|
let mut b = tar::Builder::new(Vec::new());
|
||||||
|
let body = b"ROUND-TRIP-OK\n";
|
||||||
|
let mut h = tar::Header::new_gnu();
|
||||||
|
h.set_path("marker.txt").unwrap();
|
||||||
|
h.set_size(body.len() as u64);
|
||||||
|
h.set_mode(0o644);
|
||||||
|
h.set_entry_type(tar::EntryType::Regular);
|
||||||
|
h.set_cksum();
|
||||||
|
b.append(&h, &body[..]).unwrap();
|
||||||
|
let archive = b.into_inner().unwrap();
|
||||||
|
|
||||||
|
let r = op_put(&json!({
|
||||||
|
"op": "put",
|
||||||
|
"dest": tmp.display().to_string(),
|
||||||
|
"tar_b64": B64.encode(&archive),
|
||||||
|
}));
|
||||||
|
assert_eq!(r["ok"], json!(true), "put failed: {r}");
|
||||||
|
assert_eq!(
|
||||||
|
std::fs::read_to_string(tmp.join("marker.txt")).unwrap(),
|
||||||
|
"ROUND-TRIP-OK\n"
|
||||||
|
);
|
||||||
|
|
||||||
|
let r = op_get(&json!({ "op": "get", "path": tmp.display().to_string() }));
|
||||||
|
assert_eq!(r["ok"], json!(true), "get failed: {r}");
|
||||||
|
let bytes = B64.decode(r["tar_b64"].as_str().unwrap()).unwrap();
|
||||||
|
let mut ar = tar::Archive::new(&bytes[..]);
|
||||||
|
let found = ar
|
||||||
|
.entries()
|
||||||
|
.unwrap()
|
||||||
|
.filter_map(Result::ok)
|
||||||
|
.any(|e| e.path().map(|p| p.ends_with("marker.txt")).unwrap_or(false));
|
||||||
|
assert!(found, "the collected archive must contain marker.txt");
|
||||||
|
|
||||||
|
let _ = std::fs::remove_dir_all(&tmp);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,301 @@
|
|||||||
|
//! Which agents are working, which are finished, and which are orphaned.
|
||||||
|
//!
|
||||||
|
//! A mission mints a crew, and until now the only thing that reaped that crew
|
||||||
|
//! was deleting the mission. A mission that merely *completed* left its agents
|
||||||
|
//! in the roster forever, and a crew whose reap was skipped or failed left
|
||||||
|
//! agents bound to nothing at all — indistinguishable, in the UI, from the
|
||||||
|
//! operator's own staff.
|
||||||
|
//!
|
||||||
|
//! The discriminator is `agent_template_link`. `mission_orchestrator` writes one
|
||||||
|
//! row per claw it mints, recording the template and role slot it was minted
|
||||||
|
//! for. An agent WITHOUT that row was created by a human (or the planner) and is
|
||||||
|
//! part of the workforce: it is never touched here, whatever it is bound to.
|
||||||
|
//! Verified against live data — the two hand-created agents on this deployment
|
||||||
|
//! have no link row and no team membership, while every mission crew member has
|
||||||
|
//! both.
|
||||||
|
//!
|
||||||
|
//! ```text
|
||||||
|
//! owned no template link → the operator's own agent. KEEP.
|
||||||
|
//! active on a running/draft mission → doing work right now. KEEP.
|
||||||
|
//! completed every mission terminal → reapable once past the grace window.
|
||||||
|
//! orphaned minted, bound to nothing → reap.
|
||||||
|
//! ```
|
||||||
|
//!
|
||||||
|
//! `completed` waits out a grace window rather than reaping the moment a mission
|
||||||
|
//! finishes: the results view, the World's 24h replay and "who did this work?"
|
||||||
|
//! all read the crew AFTER the run ends. Reaping on the terminal transition
|
||||||
|
//! would delete the answer at the moment the question gets asked.
|
||||||
|
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
use sqlx::{PgPool, Row};
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// How long a finished crew is kept before it is reaped. Matches the World's
|
||||||
|
/// 24h window for finished missions, so nothing the UI can still show is
|
||||||
|
/// collected out from under it.
|
||||||
|
pub const COMPLETED_GRACE_HOURS: i64 = 24;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum AgentState {
|
||||||
|
Owned,
|
||||||
|
Active,
|
||||||
|
Completed,
|
||||||
|
Orphaned,
|
||||||
|
/// Soft-deleted by an operator. The `agents` row and its history survive.
|
||||||
|
Deleted,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl AgentState {
|
||||||
|
pub fn as_str(self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
AgentState::Owned => "owned",
|
||||||
|
AgentState::Active => "active",
|
||||||
|
AgentState::Completed => "completed",
|
||||||
|
AgentState::Orphaned => "orphaned",
|
||||||
|
AgentState::Deleted => "deleted",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
/// `owned` and `active` are NEVER collected, and that is the whole safety
|
||||||
|
/// property of this module.
|
||||||
|
pub fn reapable(self) -> bool {
|
||||||
|
matches!(
|
||||||
|
self,
|
||||||
|
AgentState::Completed | AgentState::Orphaned | AgentState::Deleted
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct Classified {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub name: String,
|
||||||
|
pub state: AgentState,
|
||||||
|
/// When the newest mission this agent served reached a terminal state.
|
||||||
|
/// `None` for owned/active/orphaned.
|
||||||
|
pub finished_hours_ago: Option<f64>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The classification, as one query.
|
||||||
|
///
|
||||||
|
/// Soft-deleted rows are INCLUDED, classified `deleted`, and collected: a soft
|
||||||
|
/// delete marks the row and leaves it, so "remove" never became permanent and
|
||||||
|
/// re-deleting did nothing. Purging takes `usage_events` with it — accepted
|
||||||
|
/// deliberately, since the alternative is rows that outlive the decision to
|
||||||
|
/// delete them.
|
||||||
|
const CENSUS_SQL: &str = r#"
|
||||||
|
SELECT a.id,
|
||||||
|
a.name,
|
||||||
|
CASE
|
||||||
|
-- First, so a soft-deleted agent is never mistaken for live staff:
|
||||||
|
-- these rows have no template link either, and would otherwise read
|
||||||
|
-- as 'owned' and be kept forever.
|
||||||
|
WHEN a.deleted_at IS NOT NULL THEN 'deleted'
|
||||||
|
WHEN atl.agent_id IS NULL THEN 'owned'
|
||||||
|
WHEN EXISTS (
|
||||||
|
SELECT 1 FROM team_members tm
|
||||||
|
JOIN mission_teams mt ON mt.team_id = tm.team_id
|
||||||
|
JOIN missions m ON m.id = mt.mission_id
|
||||||
|
WHERE tm.claw_id = a.id AND m.status IN ('running', 'draft')
|
||||||
|
) THEN 'active'
|
||||||
|
WHEN EXISTS (
|
||||||
|
SELECT 1 FROM team_members tm
|
||||||
|
JOIN mission_teams mt ON mt.team_id = tm.team_id
|
||||||
|
WHERE tm.claw_id = a.id
|
||||||
|
) THEN 'completed'
|
||||||
|
ELSE 'orphaned'
|
||||||
|
END AS state,
|
||||||
|
(SELECT EXTRACT(EPOCH FROM (now() - MAX(COALESCE(m.completed_at, m.updated_at)))) / 3600.0
|
||||||
|
FROM team_members tm
|
||||||
|
JOIN mission_teams mt ON mt.team_id = tm.team_id
|
||||||
|
JOIN missions m ON m.id = mt.mission_id
|
||||||
|
WHERE tm.claw_id = a.id) AS finished_hours_ago
|
||||||
|
FROM agents a
|
||||||
|
LEFT JOIN agent_template_link atl ON atl.agent_id = a.id
|
||||||
|
WHERE a.workspace_id = $1
|
||||||
|
ORDER BY a.created_at, a.id
|
||||||
|
"#;
|
||||||
|
|
||||||
|
pub async fn census(pool: &PgPool, workspace_id: Uuid) -> Result<Vec<Classified>, String> {
|
||||||
|
let rows = sqlx::query(CENSUS_SQL)
|
||||||
|
.bind(workspace_id)
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("agent census: {e}"))?;
|
||||||
|
Ok(rows
|
||||||
|
.into_iter()
|
||||||
|
.map(|r| {
|
||||||
|
let state = match r.get::<String, _>("state").as_str() {
|
||||||
|
"owned" => AgentState::Owned,
|
||||||
|
"active" => AgentState::Active,
|
||||||
|
"completed" => AgentState::Completed,
|
||||||
|
"deleted" => AgentState::Deleted,
|
||||||
|
_ => AgentState::Orphaned,
|
||||||
|
};
|
||||||
|
Classified {
|
||||||
|
id: r.get("id"),
|
||||||
|
name: r.get("name"),
|
||||||
|
state,
|
||||||
|
finished_hours_ago: r.get::<Option<f64>, _>("finished_hours_ago"),
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What one sweep did.
|
||||||
|
#[derive(Debug, Default, PartialEq, Eq)]
|
||||||
|
pub struct Swept {
|
||||||
|
pub reaped: usize,
|
||||||
|
pub failed: usize,
|
||||||
|
pub kept_in_grace: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Decide, without touching the database, whether a classified agent should be
|
||||||
|
/// collected on this pass. Split out so the policy is testable on its own —
|
||||||
|
/// the expensive half is the purge, and the half that can silently delete a
|
||||||
|
/// workforce is this one.
|
||||||
|
pub fn should_reap(c: &Classified, grace_hours: i64) -> bool {
|
||||||
|
match c.state {
|
||||||
|
AgentState::Owned | AgentState::Active => false,
|
||||||
|
// No grace: a human already decided. The soft delete IS the decision,
|
||||||
|
// and these rows have sat for months waiting for something to honour it.
|
||||||
|
AgentState::Deleted => true,
|
||||||
|
AgentState::Orphaned => true,
|
||||||
|
AgentState::Completed => c
|
||||||
|
.finished_hours_ago
|
||||||
|
// No timestamp means we cannot prove the grace has elapsed, so keep
|
||||||
|
// it. A missing date must never read as "old enough to delete".
|
||||||
|
.is_some_and(|h| h >= grace_hours as f64),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Reap finished and orphaned crews across every workspace.
|
||||||
|
pub async fn sweep(
|
||||||
|
pool: &PgPool,
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
|
grace_hours: i64,
|
||||||
|
) -> Result<Swept, String> {
|
||||||
|
let workspaces: Vec<Uuid> = sqlx::query_scalar("SELECT id FROM workspaces")
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("list workspaces: {e}"))?;
|
||||||
|
|
||||||
|
let provisioner = crate::runtime_provision::RuntimeProvisioner::from_env();
|
||||||
|
let mut out = Swept::default();
|
||||||
|
for ws in workspaces {
|
||||||
|
for c in census(pool, ws).await? {
|
||||||
|
if !c.state.reapable() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if !should_reap(&c, grace_hours) {
|
||||||
|
out.kept_in_grace += 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let report = crate::routes::claws::purge_agent(
|
||||||
|
pool,
|
||||||
|
runtime,
|
||||||
|
provisioner.as_ref(),
|
||||||
|
cm_domain::AgentId::from(c.id),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
match report.counts {
|
||||||
|
Ok(_) => {
|
||||||
|
out.reaped += 1;
|
||||||
|
eprintln!(
|
||||||
|
"agent_lifecycle: reaped {} claw {} ({})",
|
||||||
|
c.state.as_str(),
|
||||||
|
c.name,
|
||||||
|
c.id
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
out.failed += 1;
|
||||||
|
eprintln!("agent_lifecycle: purge {} failed (continuing): {e}", c.id);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spawn the sweeper.
|
||||||
|
pub fn spawn(pool: PgPool, runtime: cm_runtime::Runtime, interval: Duration) {
|
||||||
|
tokio::spawn(async move {
|
||||||
|
let mut tick = tokio::time::interval(interval);
|
||||||
|
// The first tick fires immediately; skip it so a restart loop cannot
|
||||||
|
// turn into a reap loop.
|
||||||
|
tick.tick().await;
|
||||||
|
loop {
|
||||||
|
tick.tick().await;
|
||||||
|
match sweep(&pool, &runtime, COMPLETED_GRACE_HOURS).await {
|
||||||
|
Ok(s) if s.reaped > 0 || s.failed > 0 => eprintln!(
|
||||||
|
"agent_lifecycle: swept — {} reaped, {} failed, {} still in grace",
|
||||||
|
s.reaped, s.failed, s.kept_in_grace
|
||||||
|
),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => eprintln!("agent_lifecycle: sweep failed: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn c(state: AgentState, hours: Option<f64>) -> Classified {
|
||||||
|
Classified {
|
||||||
|
id: Uuid::now_v7(),
|
||||||
|
name: "x".into(),
|
||||||
|
state,
|
||||||
|
finished_hours_ago: hours,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The property that matters most: this sweeper must never be able to
|
||||||
|
/// delete the operator's own staff, no matter what it is bound to.
|
||||||
|
#[test]
|
||||||
|
fn owned_and_active_are_never_reaped() {
|
||||||
|
for hours in [None, Some(0.0), Some(1_000_000.0)] {
|
||||||
|
assert!(!should_reap(&c(AgentState::Owned, hours), 24));
|
||||||
|
assert!(!should_reap(&c(AgentState::Active, hours), 24));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn orphans_go_immediately() {
|
||||||
|
assert!(should_reap(&c(AgentState::Orphaned, None), 24));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A soft delete is a decision that was never honoured — the row stayed,
|
||||||
|
/// the agent kept appearing, and deleting it again did nothing. Collect it
|
||||||
|
/// without a grace window: the human already waited.
|
||||||
|
#[test]
|
||||||
|
fn soft_deleted_agents_are_purged_without_a_grace_window() {
|
||||||
|
assert!(should_reap(&c(AgentState::Deleted, None), 24));
|
||||||
|
assert!(should_reap(&c(AgentState::Deleted, Some(0.0)), 24));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The safety property restated against the new state: `deleted` must not
|
||||||
|
/// widen into anything that can take live staff with it.
|
||||||
|
#[test]
|
||||||
|
fn adding_deleted_did_not_make_owned_reapable() {
|
||||||
|
assert!(!AgentState::Owned.reapable());
|
||||||
|
assert!(!AgentState::Active.reapable());
|
||||||
|
assert!(AgentState::Deleted.reapable());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_finished_crew_waits_out_the_grace_window() {
|
||||||
|
assert!(!should_reap(&c(AgentState::Completed, Some(1.0)), 24));
|
||||||
|
assert!(!should_reap(&c(AgentState::Completed, Some(23.9)), 24));
|
||||||
|
assert!(should_reap(&c(AgentState::Completed, Some(24.0)), 24));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A completed crew with no usable timestamp must be KEPT. Treating a
|
||||||
|
/// missing date as "old" is how a sweeper deletes something it was never
|
||||||
|
/// able to prove was finished.
|
||||||
|
#[test]
|
||||||
|
fn a_missing_finish_time_is_not_treated_as_old() {
|
||||||
|
assert!(!should_reap(&c(AgentState::Completed, None), 24));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,231 @@
|
|||||||
|
//! Human given names for minted agents.
|
||||||
|
//!
|
||||||
|
//! A team used to come back as `planner`, `coder`, `tester`, `reviewer`,
|
||||||
|
//! `committer` — the roster read as a list of job tickets, and the UI showed
|
||||||
|
//! the same word twice (name on top, role underneath). A crew you keep should
|
||||||
|
//! read like people: Meredith, Vijay, Tomasz, Amara.
|
||||||
|
//!
|
||||||
|
//! The role is not lost — it stays in `job_title`, which is what the mission
|
||||||
|
//! machinery binds on. Only the display identity changes.
|
||||||
|
//!
|
||||||
|
//! Names are drawn from many naming traditions on purpose: this workforce is
|
||||||
|
//! not from one place. They are given names only — no surnames — so nobody
|
||||||
|
//! reads a claw as a specific real person.
|
||||||
|
|
||||||
|
/// Given names, deliberately wide. Kept as one flat list rather than grouped by
|
||||||
|
/// origin: grouping invites picking "one from each", which is a worse kind of
|
||||||
|
/// tokenism than simply having a broad pool and drawing from it evenly.
|
||||||
|
///
|
||||||
|
/// Size is a product decision, not an aesthetic one. Every mission now mints
|
||||||
|
/// its own crew and nothing retires them, so the roster grows by the team size
|
||||||
|
/// per mission — at ~5 a mission a 70-name pool starts emitting "Amara 2"
|
||||||
|
/// inside twenty missions. This pool carries a few hundred so a workspace runs
|
||||||
|
/// for a long time before any name repeats at all.
|
||||||
|
pub const NAMES: &[&str] = &[
|
||||||
|
// A
|
||||||
|
"Aarav", "Abebe", "Adaora", "Adrian", "Agnieszka", "Ahmad", "Aiko", "Ainhoa", "Alejandro",
|
||||||
|
"Alina", "Amara", "Amina", "Anders", "Andrea", "Anjali", "Annika", "Antoine", "Arjun", "Astrid",
|
||||||
|
"Ayo", "Ayesha", "Aziz",
|
||||||
|
// B–C
|
||||||
|
"Beatriz", "Bilal", "Bjorn", "Blessing", "Bogdan", "Camila", "Carlos", "Catalina", "Chidi",
|
||||||
|
"Chiara", "Chioma", "Cyrus",
|
||||||
|
// D–E
|
||||||
|
"Dagny", "Damir", "Daniela", "Dilnoza", "Dmitri", "Ebele", "Eduardo", "Eero", "Ekaterina",
|
||||||
|
"Elena", "Elias", "Emeka", "Enrique", "Esi", "Esther", "Eun-ji", "Ewa",
|
||||||
|
// F–G
|
||||||
|
"Fabio", "Farida", "Fatou", "Felipe", "Fernanda", "Freya", "Gabriel", "Georgi", "Giulia",
|
||||||
|
"Grace", "Gunnar", "Gulnara",
|
||||||
|
// H–I
|
||||||
|
"Hana", "Hasan", "Heidi", "Hina", "Hiroshi", "Ibrahim", "Idris", "Ilya", "Imani", "Ingrid",
|
||||||
|
"Iris", "Isabela", "Ivan", "Iwona",
|
||||||
|
// J–K
|
||||||
|
"Jaromir", "Javier", "Jing", "Joana", "Johan", "Josefina", "Junko", "Kaito", "Kalinda", "Karim",
|
||||||
|
"Katarzyna", "Kenji", "Khalid", "Kiran", "Klara", "Kwame", "Kyoko",
|
||||||
|
// L–M
|
||||||
|
"Lakshmi", "Lars", "Laila", "Leilani", "Lena", "Liam", "Linnea", "Lucia", "Lukas", "Madhavi",
|
||||||
|
"Maja", "Malik", "Marisol", "Mateo", "Matteo", "Mei", "Meredith", "Milena", "Mira", "Mohan",
|
||||||
|
"Mira-Lynn", "Mateusz",
|
||||||
|
// N–O
|
||||||
|
"Nadia", "Nasrin", "Neelam", "Niamh", "Nikolai", "Nilufar", "Nkechi", "Noor", "Nuria", "Oksana",
|
||||||
|
"Oleksii", "Olamide", "Omar", "Oskar", "Osei",
|
||||||
|
// P–R
|
||||||
|
"Paloma", "Panagiotis", "Pedro", "Petra", "Priya", "Rafael", "Rania", "Ravi", "Reza", "Renata",
|
||||||
|
"Rin", "Robert", "Rosalind", "Rustam",
|
||||||
|
// S
|
||||||
|
"Sadia", "Salome", "Samir", "Sanjay", "Sara", "Seong-min", "Sipho", "Sofia", "Solveig", "Soren",
|
||||||
|
"Suvi", "Svetlana",
|
||||||
|
// T–U
|
||||||
|
"Tadeusz", "Takeshi", "Tamar", "Tariq", "Thandiwe", "Thi", "Tim", "Tomasz", "Tove", "Tuva",
|
||||||
|
"Ulrika", "Uma", "Usman",
|
||||||
|
// V–Z
|
||||||
|
"Valentina", "Vera", "Vijay", "Vikram", "Wanjiru", "Wei", "Wiktor", "Yara", "Yasmin", "Yohannes",
|
||||||
|
"Yuki", "Yusuf", "Zainab", "Zara", "Zoltan", "Zuzanna",
|
||||||
|
];
|
||||||
|
|
||||||
|
/// Pick a name not already in `taken`.
|
||||||
|
///
|
||||||
|
/// `seed` spreads the starting point so a workspace does not always begin at
|
||||||
|
/// "Amara" — it is an offset into the list, not randomness, so the choice is
|
||||||
|
/// reproducible for a given (seed, taken) pair and therefore testable.
|
||||||
|
///
|
||||||
|
/// When every name is taken it appends a numeric suffix — `Amara 2` — rather
|
||||||
|
/// than returning `None` and forcing the caller to invent something. Running
|
||||||
|
/// out is a nice problem (70+ concurrent agents in one workspace) and a
|
||||||
|
/// duplicate display name is far less harmful than a failed mission launch.
|
||||||
|
pub fn pick(taken: &[String], seed: u64) -> String {
|
||||||
|
let start = (seed % NAMES.len() as u64) as usize;
|
||||||
|
for i in 0..NAMES.len() {
|
||||||
|
let candidate = NAMES[(start + i) % NAMES.len()];
|
||||||
|
if !taken.iter().any(|t| t.eq_ignore_ascii_case(candidate)) {
|
||||||
|
return candidate.to_string();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Second pass with a suffix. `round` starts at 2 so the first repeat reads
|
||||||
|
// "Amara 2", which is how a person would disambiguate two colleagues.
|
||||||
|
for round in 2..1000 {
|
||||||
|
for i in 0..NAMES.len() {
|
||||||
|
let candidate = format!("{} {}", NAMES[(start + i) % NAMES.len()], round);
|
||||||
|
if !taken.iter().any(|t| t.eq_ignore_ascii_case(&candidate)) {
|
||||||
|
return candidate;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Unreachable in practice; still not a panic.
|
||||||
|
format!("Agent {seed}")
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn names_are_unique_and_non_empty() {
|
||||||
|
let mut seen = std::collections::HashSet::new();
|
||||||
|
for n in NAMES {
|
||||||
|
assert!(!n.trim().is_empty(), "empty name in the pool");
|
||||||
|
assert!(seen.insert(n.to_ascii_lowercase()), "duplicate in pool: {n}");
|
||||||
|
}
|
||||||
|
// Every mission mints its own crew and nothing retires them, so the
|
||||||
|
// pool is consumed for the life of the workspace, not recycled. At ~5
|
||||||
|
// per mission this is ~35 missions before the first numeric suffix.
|
||||||
|
assert!(NAMES.len() >= 150, "pool too small for one crew per mission");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A crew should not read as an alphabetical run.
|
||||||
|
///
|
||||||
|
/// With the role index as the seed, every crew started at the top of the
|
||||||
|
/// pool and took the next free names — the first real mission hired Aarav,
|
||||||
|
/// Abebe, Adaora, Adrian, Agnieszka. Unique and correct, and obviously
|
||||||
|
/// generated. Callers now seed from the claw's uuid tail, so this checks
|
||||||
|
/// that well-spread seeds actually land in different regions of the pool
|
||||||
|
/// rather than clustering at one end.
|
||||||
|
#[test]
|
||||||
|
fn spread_seeds_do_not_produce_an_alphabetical_run() {
|
||||||
|
let index_of = |n: &str| NAMES.iter().position(|c| *c == n).expect("name in pool");
|
||||||
|
let seeds = [
|
||||||
|
0x9e37_79b9_7f4a_7c15u64,
|
||||||
|
0x1234_5678_9abc_def0,
|
||||||
|
0xfeed_face_dead_beef,
|
||||||
|
0x0f0f_0f0f_f0f0_f0f0,
|
||||||
|
0xa5a5_5a5a_c3c3_3c3c,
|
||||||
|
];
|
||||||
|
let mut taken: Vec<String> = Vec::new();
|
||||||
|
let mut positions = Vec::new();
|
||||||
|
for s in seeds {
|
||||||
|
let n = pick(&taken, s);
|
||||||
|
positions.push(index_of(&n) as i64);
|
||||||
|
taken.push(n);
|
||||||
|
}
|
||||||
|
// Adjacent picks landing within a couple of slots of each other is the
|
||||||
|
// clustering signature; require the crew to span a real distance.
|
||||||
|
let (min, max) = (
|
||||||
|
*positions.iter().min().unwrap(),
|
||||||
|
*positions.iter().max().unwrap(),
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
max - min > (NAMES.len() as i64) / 3,
|
||||||
|
"crew clustered in one region of the pool: {positions:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The scenario the operator actually asked for: consecutive missions must
|
||||||
|
/// not hand back the same names. Reuse is off, so mission two staffs from
|
||||||
|
/// what mission one left.
|
||||||
|
#[test]
|
||||||
|
fn consecutive_missions_get_different_crews() {
|
||||||
|
let mut roster: Vec<String> = Vec::new();
|
||||||
|
let mut crews: Vec<Vec<String>> = Vec::new();
|
||||||
|
for mission in 0..6u64 {
|
||||||
|
let mut crew = Vec::new();
|
||||||
|
for role in 0..5u64 {
|
||||||
|
let n = pick(&roster, mission * 5 + role);
|
||||||
|
roster.push(n.clone());
|
||||||
|
crew.push(n);
|
||||||
|
}
|
||||||
|
crews.push(crew);
|
||||||
|
}
|
||||||
|
for (i, a) in crews.iter().enumerate() {
|
||||||
|
for (j, b) in crews.iter().enumerate().skip(i + 1) {
|
||||||
|
let shared: Vec<_> = a.iter().filter(|n| b.contains(n)).collect();
|
||||||
|
assert!(
|
||||||
|
shared.is_empty(),
|
||||||
|
"missions {i} and {j} share {shared:?} — crews must be distinct"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// And no duplicates anywhere on the roster.
|
||||||
|
let uniq: std::collections::HashSet<_> = roster.iter().collect();
|
||||||
|
assert_eq!(uniq.len(), roster.len(), "a name was issued twice");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pick_avoids_taken_names() {
|
||||||
|
let taken: Vec<String> = NAMES.iter().take(10).map(|s| s.to_string()).collect();
|
||||||
|
let got = pick(&taken, 0);
|
||||||
|
assert!(
|
||||||
|
!taken.iter().any(|t| t.eq_ignore_ascii_case(&got)),
|
||||||
|
"picked a name already taken: {got}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pick_is_case_insensitive_about_taken() {
|
||||||
|
// A name already on the roster in a different case is still taken —
|
||||||
|
// "meredith" and "Meredith" are the same colleague.
|
||||||
|
let taken = vec![NAMES[0].to_ascii_lowercase()];
|
||||||
|
assert_ne!(pick(&taken, 0).to_ascii_lowercase(), taken[0]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn seed_spreads_the_starting_point() {
|
||||||
|
// Different seeds should not all hand back the same first name, or a
|
||||||
|
// fresh workspace always opens with the same roster.
|
||||||
|
let a = pick(&[], 0);
|
||||||
|
let b = pick(&[], 7);
|
||||||
|
assert_ne!(a, b, "seed had no effect on the choice");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn exhausting_the_pool_suffixes_rather_than_failing() {
|
||||||
|
let taken: Vec<String> = NAMES.iter().map(|s| s.to_string()).collect();
|
||||||
|
let got = pick(&taken, 0);
|
||||||
|
assert!(
|
||||||
|
!taken.iter().any(|t| t.eq_ignore_ascii_case(&got)),
|
||||||
|
"must not reuse a taken name"
|
||||||
|
);
|
||||||
|
assert!(got.ends_with(" 2"), "expected a suffixed name, got {got}");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_full_team_gets_distinct_names() {
|
||||||
|
// The actual scenario: mint five roles into an empty workspace and get
|
||||||
|
// five different people, not five "planner"s.
|
||||||
|
let mut taken: Vec<String> = Vec::new();
|
||||||
|
for i in 0..5 {
|
||||||
|
let n = pick(&taken, i);
|
||||||
|
assert!(!taken.contains(&n), "repeated {n} within one team");
|
||||||
|
taken.push(n);
|
||||||
|
}
|
||||||
|
assert_eq!(taken.len(), 5);
|
||||||
|
}
|
||||||
|
}
|
||||||
+230
-10
@@ -71,19 +71,46 @@ impl MergeOutcome {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Classify a `git diff --name-status` body.
|
/// Every path in a `git diff --name-status` body, with its status letter.
|
||||||
///
|
///
|
||||||
/// Returns the offending entries, empty when every change is an addition.
|
/// The World draws a file orb per changed path, and `mission_delivery` records
|
||||||
/// Split out so the rule is testable without a repository.
|
/// the list — both need the same parse, so it lives in one place.
|
||||||
pub fn non_additive_changes(name_status: &str) -> Vec<String> {
|
///
|
||||||
|
/// **Renames are three fields**: `R100\told\tnew`. The path that changed is the
|
||||||
|
/// NEW one; splitting on the first tab and taking field two records where the
|
||||||
|
/// file used to be, which then matches nothing anyone can open. Copies (`C###`)
|
||||||
|
/// have the same shape.
|
||||||
|
pub fn changed_paths(name_status: &str) -> Vec<(char, String)> {
|
||||||
name_status
|
name_status
|
||||||
.lines()
|
.lines()
|
||||||
.filter(|l| !l.trim().is_empty())
|
.filter(|l| !l.trim().is_empty())
|
||||||
.filter(|l| {
|
.filter_map(|l| {
|
||||||
// Status is the first field: A/M/D/R###/C###.
|
let mut fields = l.split('\t');
|
||||||
!matches!(l.chars().next(), Some('A'))
|
let status = fields.next()?.trim();
|
||||||
|
let letter = status.chars().next()?;
|
||||||
|
let first = fields.next()?.trim();
|
||||||
|
// R/C carry old THEN new; everything else has a single path.
|
||||||
|
let path = match letter {
|
||||||
|
'R' | 'C' => fields.next().map(str::trim).unwrap_or(first),
|
||||||
|
_ => first,
|
||||||
|
};
|
||||||
|
if path.is_empty() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some((letter, path.to_string()))
|
||||||
})
|
})
|
||||||
.map(|l| l.trim().to_string())
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Classify a `git diff --name-status` body.
|
||||||
|
///
|
||||||
|
/// Returns the offending entries, empty when every change is an addition.
|
||||||
|
/// Built on `changed_paths` so the two cannot disagree about what a line means.
|
||||||
|
pub fn non_additive_changes(name_status: &str) -> Vec<String> {
|
||||||
|
changed_paths(name_status)
|
||||||
|
.into_iter()
|
||||||
|
.filter(|(letter, _)| *letter != 'A')
|
||||||
|
.map(|(letter, path)| format!("{letter}\t{path}"))
|
||||||
.collect()
|
.collect()
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -166,11 +193,35 @@ pub async fn try_merge(
|
|||||||
return Ok(MergeOutcome::refused("branch adds nothing"));
|
return Ok(MergeOutcome::refused("branch adds nothing"));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
merge_and_push(repo, push_url, branch, base, "auto-merge")
|
||||||
|
.await
|
||||||
|
.map(|o| match o.merged {
|
||||||
|
true => MergeOutcome {
|
||||||
|
merged: true,
|
||||||
|
reason: format!("additive-only and verified; merged into {base}"),
|
||||||
|
},
|
||||||
|
false => o,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The git half of a merge, with no policy in it.
|
||||||
|
///
|
||||||
|
/// Split out so an OPERATOR-approved merge runs exactly the same commands as an
|
||||||
|
/// automatic one — fetch the base as the remote has it, merge onto that, push.
|
||||||
|
/// The gates differ; the mechanics must not, or the rarely-taken path is the one
|
||||||
|
/// that breaks.
|
||||||
|
async fn merge_and_push(
|
||||||
|
repo: &Path,
|
||||||
|
push_url: &str,
|
||||||
|
branch: &str,
|
||||||
|
base: &str,
|
||||||
|
label: &str,
|
||||||
|
) -> Result<MergeOutcome, String> {
|
||||||
// Merge onto the freshly fetched base rather than a local branch.
|
// Merge onto the freshly fetched base rather than a local branch.
|
||||||
git(repo, &["checkout", "-B", base, "FETCH_HEAD"]).await?;
|
git(repo, &["checkout", "-B", base, "FETCH_HEAD"]).await?;
|
||||||
if let Err(e) = git(
|
if let Err(e) = git(
|
||||||
repo,
|
repo,
|
||||||
&["merge", "--no-ff", "-m", &format!("auto-merge {branch}"), branch],
|
&["merge", "--no-ff", "-m", &format!("{label} {branch}"), branch],
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
@@ -184,14 +235,150 @@ pub async fn try_merge(
|
|||||||
git(repo, &["push", push_url, &format!("HEAD:refs/heads/{base}")]).await?;
|
git(repo, &["push", push_url, &format!("HEAD:refs/heads/{base}")]).await?;
|
||||||
Ok(MergeOutcome {
|
Ok(MergeOutcome {
|
||||||
merged: true,
|
merged: true,
|
||||||
reason: format!("additive-only and verified; merged into {base}"),
|
reason: format!("merged into {base}"),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Merge a delivered branch because an OPERATOR asked for it.
|
||||||
|
///
|
||||||
|
/// `MergePolicy::Never` means "do not merge on your own" — it defers to a human,
|
||||||
|
/// and this is that human. So the additive-only test does not apply: an operator
|
||||||
|
/// looking at a code change is exactly the judgement the policy was holding out
|
||||||
|
/// for.
|
||||||
|
///
|
||||||
|
/// What is NOT waived:
|
||||||
|
///
|
||||||
|
/// - the branch must exist on the remote and differ from the base, so the button
|
||||||
|
/// cannot report success for a merge of nothing;
|
||||||
|
/// - a conflict refuses and leaves the repo clean, rather than forcing;
|
||||||
|
/// - the work happens in a FRESH CLONE, never the mission checkout — that
|
||||||
|
/// directory is reaped on a timer after the mission ends, so a merge that
|
||||||
|
/// depended on it would work right after a run and mysteriously fail later.
|
||||||
|
pub async fn merge_on_operator_approval(
|
||||||
|
workdir: &Path,
|
||||||
|
push_url: &str,
|
||||||
|
branch: &str,
|
||||||
|
base: &str,
|
||||||
|
) -> Result<MergeOutcome, String> {
|
||||||
|
git(workdir, &["fetch", push_url, base]).await?;
|
||||||
|
git(workdir, &["fetch", push_url, branch]).await?;
|
||||||
|
git(workdir, &["branch", "-f", branch, "FETCH_HEAD"]).await?;
|
||||||
|
git(workdir, &["fetch", push_url, base]).await?;
|
||||||
|
|
||||||
|
let diff = git(
|
||||||
|
workdir,
|
||||||
|
&["diff", "--name-status", &format!("FETCH_HEAD...{branch}")],
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
if diff.trim().is_empty() {
|
||||||
|
return Ok(MergeOutcome::refused(
|
||||||
|
"branch has nothing the base does not already have",
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
merge_locally(workdir, branch, base, "merge mission branch").await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Merge onto the fetched base WITHOUT publishing it.
|
||||||
|
///
|
||||||
|
/// Split from the push so a caller can run the project's tests against the
|
||||||
|
/// merged tree first. Verifying BEFORE publishing rather than reverting after is
|
||||||
|
/// the difference between "main was never broken" and "main was broken for as
|
||||||
|
/// long as it took us to notice".
|
||||||
|
pub async fn merge_locally(
|
||||||
|
repo: &Path,
|
||||||
|
branch: &str,
|
||||||
|
base: &str,
|
||||||
|
label: &str,
|
||||||
|
) -> Result<MergeOutcome, String> {
|
||||||
|
git(repo, &["checkout", "-B", base, "FETCH_HEAD"]).await?;
|
||||||
|
if let Err(e) = git(
|
||||||
|
repo,
|
||||||
|
&["merge", "--no-ff", "-m", &format!("{label} {branch}"), branch],
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
// Leave the repo clean so the next attempt is not fighting a wedged merge.
|
||||||
|
let _ = git(repo, &["merge", "--abort"]).await;
|
||||||
|
return Ok(MergeOutcome::refused(format!(
|
||||||
|
"merge conflicted ({e}); left for a human"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
Ok(MergeOutcome {
|
||||||
|
merged: true,
|
||||||
|
reason: format!("merged into {base} locally, not yet published"),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Publish an already-merged base.
|
||||||
|
pub async fn push_merged(repo: &Path, push_url: &str, base: &str) -> Result<(), String> {
|
||||||
|
git(repo, &["push", push_url, &format!("HEAD:refs/heads/{base}")])
|
||||||
|
.await
|
||||||
|
.map(|_| ())
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Publication must be gated on the merged tree, and refusal must not push.
|
||||||
|
///
|
||||||
|
/// The two halves are separate functions precisely so a caller can run tests
|
||||||
|
/// BETWEEN them. If `merge_locally` ever pushed, verification would be
|
||||||
|
/// after-the-fact and `main` would be broken for as long as it took to
|
||||||
|
/// notice — which is the failure mode this whole thing exists to avoid.
|
||||||
|
#[test]
|
||||||
|
fn merging_locally_never_publishes() {
|
||||||
|
let src = include_str!("auto_merge.rs");
|
||||||
|
let body = src
|
||||||
|
.split("pub async fn merge_locally")
|
||||||
|
.nth(1)
|
||||||
|
.and_then(|s| s.split("\n}").next())
|
||||||
|
.unwrap_or("");
|
||||||
|
assert!(!body.is_empty(), "merge_locally not found");
|
||||||
|
assert!(
|
||||||
|
!body.contains("\"push\""),
|
||||||
|
"merge_locally must not push — publication is the caller's decision \
|
||||||
|
after it has verified the result"
|
||||||
|
);
|
||||||
|
// And the push half must exist separately, or the caller cannot publish.
|
||||||
|
assert!(src.contains("pub async fn push_merged"), "push_merged missing");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An operator merge and an automatic one must run the SAME git commands.
|
||||||
|
///
|
||||||
|
/// The gates differ — that is the whole point — but if the mechanics
|
||||||
|
/// diverged, the rarely-taken path would be the untested one. Both go
|
||||||
|
/// through `merge_and_push`.
|
||||||
|
#[test]
|
||||||
|
fn both_merge_paths_share_the_same_mechanics() {
|
||||||
|
let src = include_str!("auto_merge.rs");
|
||||||
|
let calls = src.matches("merge_and_push(").count();
|
||||||
|
// one definition + one call from each path
|
||||||
|
assert!(
|
||||||
|
calls >= 3,
|
||||||
|
"expected try_merge and merge_on_operator_approval to both call \
|
||||||
|
merge_and_push, found {calls} mention(s)"
|
||||||
|
);
|
||||||
|
// And the operator path must NOT re-implement the policy gate it exists
|
||||||
|
// to bypass — if this string appears there, the button is a no-op.
|
||||||
|
let op = src
|
||||||
|
.split("pub async fn merge_on_operator_approval")
|
||||||
|
.nth(1)
|
||||||
|
.unwrap_or("");
|
||||||
|
let body = op.split("\n}").next().unwrap_or("");
|
||||||
|
assert!(
|
||||||
|
!body.contains("MergePolicy::AdditiveOnly"),
|
||||||
|
"the operator path must not apply the additive-only gate"
|
||||||
|
);
|
||||||
|
// It must still refuse an empty branch: a button that reports success
|
||||||
|
// for merging nothing is worse than no button.
|
||||||
|
assert!(
|
||||||
|
body.contains("nothing the base does not already have"),
|
||||||
|
"the operator path must refuse an empty branch"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn only_pure_additions_qualify() {
|
fn only_pure_additions_qualify() {
|
||||||
assert!(non_additive_changes("A\t60 Papers/a.md\nA\t60 Papers/b.md\n").is_empty());
|
assert!(non_additive_changes("A\t60 Papers/a.md\nA\t60 Papers/b.md\n").is_empty());
|
||||||
@@ -207,6 +394,39 @@ mod tests {
|
|||||||
assert_eq!(non_additive_changes("R100\ta.md\tb.md\n").len(), 1);
|
assert_eq!(non_additive_changes("R100\ta.md\tb.md\n").len(), 1);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A rename records the NEW path.
|
||||||
|
///
|
||||||
|
/// `R100\told\tnew` is three fields. Reading field two — which is what a
|
||||||
|
/// split-on-first-tab gives you — records where the file USED to be, so the
|
||||||
|
/// World would draw an orb for a path that no longer exists and the
|
||||||
|
/// delivered file list would name something nobody can open. The bug is
|
||||||
|
/// invisible in any repo where nothing was renamed.
|
||||||
|
#[test]
|
||||||
|
fn a_rename_records_where_the_file_ended_up() {
|
||||||
|
let paths = changed_paths("R100\tsrc/old.rs\tsrc/new.rs\n");
|
||||||
|
assert_eq!(paths, vec![('R', "src/new.rs".to_string())]);
|
||||||
|
|
||||||
|
let copied = changed_paths("C075\tsrc/a.rs\tsrc/b.rs\n");
|
||||||
|
assert_eq!(copied, vec![('C', "src/b.rs".to_string())]);
|
||||||
|
|
||||||
|
// Ordinary two-field lines are unaffected.
|
||||||
|
assert_eq!(
|
||||||
|
changed_paths("A\tone.md\nM\ttwo.md\nD\tthree.md\n"),
|
||||||
|
vec![
|
||||||
|
('A', "one.md".to_string()),
|
||||||
|
('M', "two.md".to_string()),
|
||||||
|
('D', "three.md".to_string()),
|
||||||
|
]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `files_changed` and the path list must agree, or nobody can tell which
|
||||||
|
/// one lied. git counts a rename as ONE changed file; so must we.
|
||||||
|
#[test]
|
||||||
|
fn a_rename_counts_once() {
|
||||||
|
assert_eq!(changed_paths("R100\ta.rs\tb.rs\n").len(), 1);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn an_unknown_policy_never_grants_auto_merge() {
|
fn an_unknown_policy_never_grants_auto_merge() {
|
||||||
assert_eq!(MergePolicy::parse(None), MergePolicy::Never);
|
assert_eq!(MergePolicy::parse(None), MergePolicy::Never);
|
||||||
|
|||||||
@@ -159,10 +159,33 @@ pub async fn run(
|
|||||||
};
|
};
|
||||||
|
|
||||||
let (container, workdir) = exec_target(pool, mission_id).await?;
|
let (container, workdir) = exec_target(pool, mission_id).await?;
|
||||||
|
// Benchmark a COPY, never the mission's own checkout.
|
||||||
|
//
|
||||||
|
// `docker_exec` enters a container running as ROOT with the missions root
|
||||||
|
// bind-mounted, and `cargo bench` writes `target/`. Run in the live tree, it
|
||||||
|
// leaves root-owned build output in a checkout owned by uid 65532 — the
|
||||||
|
// single-writer invariant broken, and the next phase's cargo hitting
|
||||||
|
// permission-denied on a directory it cannot write.
|
||||||
|
//
|
||||||
|
// This is the SAME defect `evaluator_tools::Sandbox` exists for, found the
|
||||||
|
// same way: the harness's uid probe, reporting `uids=0,65532`. Measurement
|
||||||
|
// must not mutate what it measures — the rule this codebase already applies
|
||||||
|
// to the judge and to the `verifier` subagent.
|
||||||
|
let copy_root = crate::root_copy::copy_root("_bench", mission_id);
|
||||||
|
// A stale copy from a previous run is ROOT-owned (see `purge_copy`), so it
|
||||||
|
// must be removed the same way it was created — from inside the container.
|
||||||
|
crate::root_copy::purge(&container, ©_root).await;
|
||||||
|
let copy = crate::root_copy::RootCopy::of(&workdir, ©_root)?;
|
||||||
let cmd = harness.command();
|
let cmd = harness.command();
|
||||||
let raw = docker_exec(&container, &workdir, &cmd)
|
let result = docker_exec(&container, copy.workdir(), &cmd)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("exec {cmd:?}: {e}"))?;
|
.map_err(|e| format!("exec {cmd:?}: {e}"));
|
||||||
|
// Explicitly, on BOTH paths, before the `Drop` fallback runs. `cargo bench`
|
||||||
|
// writes `target/` as root, and the server process is uid 65532: its
|
||||||
|
// `remove_dir_all` cannot delete root-owned files and silently leaves the
|
||||||
|
// whole copy behind — measured at 1.2 MB per run, growing forever.
|
||||||
|
crate::root_copy::purge(&container, ©_root).await;
|
||||||
|
let raw = result?;
|
||||||
let metrics = parse_output(&raw, &harness);
|
let metrics = parse_output(&raw, &harness);
|
||||||
Ok((metrics, harness.driver_name().to_string()))
|
Ok((metrics, harness.driver_name().to_string()))
|
||||||
}
|
}
|
||||||
@@ -254,9 +277,7 @@ async fn exec_target(
|
|||||||
}
|
}
|
||||||
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
||||||
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
||||||
let root = std::env::var("CLAWMATES_MISSIONS_ROOT")
|
let workdir = crate::mission_workspace::missions_root()
|
||||||
.unwrap_or_else(|_| "/var/lib/clawmates-missions".to_string());
|
|
||||||
let workdir = std::path::PathBuf::from(root)
|
|
||||||
.join(mission_id.to_string())
|
.join(mission_id.to_string())
|
||||||
.join("repo");
|
.join("repo");
|
||||||
Ok((container, workdir))
|
Ok((container, workdir))
|
||||||
@@ -401,3 +422,29 @@ fn compute_delta(before: &Value, after: &Value) -> Value {
|
|||||||
}
|
}
|
||||||
json!({ "kind": "opaque", "note": "before/after not structurally comparable" })
|
json!({ "kind": "opaque", "note": "before/after not structurally comparable" })
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod bench_copy_tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The benchmark copy must live OUTSIDE the mission directory, and must not
|
||||||
|
/// be the checkout itself.
|
||||||
|
///
|
||||||
|
/// Running `cargo bench` in the live tree left root-owned `target/` in a
|
||||||
|
/// checkout owned by uid 65532 — caught by the harness's uid probe
|
||||||
|
/// (`uids=0,65532`) after this runner was first wired into the sweep. The
|
||||||
|
/// same rule `evaluator_tools::Sandbox` follows: measurement must not mutate
|
||||||
|
/// what it measures.
|
||||||
|
#[test]
|
||||||
|
fn a_benchmark_runs_in_a_copy_outside_the_mission_directory() {
|
||||||
|
let mission = Uuid::now_v7();
|
||||||
|
let copy = crate::root_copy::copy_root("_bench", mission);
|
||||||
|
let live = crate::mission_workspace::checkout_path(mission);
|
||||||
|
assert_ne!(copy, live, "the bench copy must not be the checkout");
|
||||||
|
assert!(
|
||||||
|
!copy.starts_with(crate::mission_workspace::missions_root().join(mission.to_string())),
|
||||||
|
"{copy:?} must be a SIBLING of the mission dir, or the reaper races it"
|
||||||
|
);
|
||||||
|
assert!(copy.starts_with(crate::mission_workspace::missions_root().join("_bench")), "{copy:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+202
-8
@@ -59,6 +59,52 @@ pub async fn fetch_systems(
|
|||||||
.unwrap_or_default())
|
.unwrap_or_default())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Newest `1m` sample per system, in ONE request.
|
||||||
|
///
|
||||||
|
/// The alternative is a request per system per poll, which grows with the
|
||||||
|
/// fleet for data that arrives in a single sorted page. `perPage` is generous
|
||||||
|
/// rather than exact because several samples belong to the same system: sorted
|
||||||
|
/// newest-first, the FIRST row seen for a system id is its latest, so later
|
||||||
|
/// rows for that system are skipped.
|
||||||
|
///
|
||||||
|
/// A hub that cannot answer this is not an error — the caller falls back to the
|
||||||
|
/// `systems.info` snapshot, which is what it used before this existed. Losing
|
||||||
|
/// GPU and IO detail must not cost the CPU and memory that still work.
|
||||||
|
pub async fn fetch_latest_stats(
|
||||||
|
client: &reqwest::Client,
|
||||||
|
conn: &BeszelConn,
|
||||||
|
token: &str,
|
||||||
|
) -> HashMap<String, Value> {
|
||||||
|
let base = conn.hub_url.trim_end_matches('/');
|
||||||
|
let resp = client
|
||||||
|
.get(format!("{base}/api/collections/system_stats/records"))
|
||||||
|
.query(&[
|
||||||
|
("perPage", "200"),
|
||||||
|
("sort", "-created"),
|
||||||
|
("filter", "type='1m'"),
|
||||||
|
])
|
||||||
|
.header("Authorization", token)
|
||||||
|
.send()
|
||||||
|
.await;
|
||||||
|
let Ok(resp) = resp else { return HashMap::new() };
|
||||||
|
if !resp.status().is_success() {
|
||||||
|
return HashMap::new();
|
||||||
|
}
|
||||||
|
let Ok(body) = resp.json::<Value>().await else {
|
||||||
|
return HashMap::new();
|
||||||
|
};
|
||||||
|
let mut out: HashMap<String, Value> = HashMap::new();
|
||||||
|
for row in body.get("items").and_then(Value::as_array).unwrap_or(&vec![]) {
|
||||||
|
let Some(sid) = row.get("system").and_then(Value::as_str) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
if let Some(stats) = row.get("stats") {
|
||||||
|
out.entry(sid.to_string()).or_insert_with(|| stats.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
/// Proxy a system's recent 1m time-series (for the monitor-page charts).
|
/// Proxy a system's recent 1m time-series (for the monitor-page charts).
|
||||||
pub async fn fetch_history(
|
pub async fn fetch_history(
|
||||||
client: &reqwest::Client,
|
client: &reqwest::Client,
|
||||||
@@ -89,24 +135,73 @@ fn f(v: &Value, k: &str) -> Option<f64> {
|
|||||||
v.get(k).and_then(Value::as_f64)
|
v.get(k).and_then(Value::as_f64)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Map a Beszel `systems` record (its `info` snapshot) into our NodeMetrics.
|
/// The n-th element of a numeric array field, as the integer the
|
||||||
fn metrics_from_system(system: &Value) -> NodeMetrics {
|
/// `node_metrics` per-second columns store. Rounded rather than truncated: a
|
||||||
|
/// rate of 0.6 is traffic, and `as i64` would file it as silence.
|
||||||
|
fn pair(v: &Value, k: &str, idx: usize) -> Option<i64> {
|
||||||
|
v.get(k)
|
||||||
|
.and_then(Value::as_array)
|
||||||
|
.and_then(|a| a.get(idx))
|
||||||
|
.and_then(Value::as_f64)
|
||||||
|
.map(|n| n.round() as i64)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Busiest GPU's utilisation percentage, from a `system_stats` sample.
|
||||||
|
///
|
||||||
|
/// `stats.g` is a MAP keyed by GPU index — `{"0":{"n":"GeForce RTX 5060 Ti",
|
||||||
|
/// "u":0,"p":4.38}}` — where `u` is utilisation and `p` is power draw. This is
|
||||||
|
/// why `gpu_pct` was null on every NVIDIA node: the old mapping read `info.g`
|
||||||
|
/// as a scalar, and `info` carries no `g` at all in Beszel 0.18. The data was
|
||||||
|
/// arriving the whole time, one collection away.
|
||||||
|
///
|
||||||
|
/// MAX rather than mean across GPUs: the question placement asks is "is there a
|
||||||
|
/// free GPU here", and averaging a saturated card with an idle one answers a
|
||||||
|
/// question nobody asked.
|
||||||
|
fn gpu_busiest(stats: &Value) -> Option<f64> {
|
||||||
|
let gpus = stats.get("g")?.as_object()?;
|
||||||
|
gpus.values()
|
||||||
|
.filter_map(|g| g.get("u").and_then(Value::as_f64))
|
||||||
|
.fold(None, |acc: Option<f64>, u| Some(acc.map_or(u, |a| a.max(u))))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Map a Beszel `systems` record into our NodeMetrics.
|
||||||
|
///
|
||||||
|
/// `stats` is the newest `system_stats` sample for this system, when there is
|
||||||
|
/// one. It carries everything the `systems.info` snapshot does not: GPU,
|
||||||
|
/// per-second network, per-second disk IO.
|
||||||
|
///
|
||||||
|
/// The array orders below were MEASURED against the hosts, not read off a
|
||||||
|
/// schema — an inverted pair here does not fail, it reports upload as download
|
||||||
|
/// forever:
|
||||||
|
/// - `b` = [sent, recv]. `stats.ni` gives per-interface
|
||||||
|
/// `[sent_ps, recv_ps, total_sent, total_recv]`; indices 2 and 3 matched
|
||||||
|
/// `/proc/net/dev` tx_bytes and rx_bytes on all four of tank's interfaces,
|
||||||
|
/// and `b` is the sum of the per-second pair across them.
|
||||||
|
/// - `dio` = [read, write]. An 800 MB `dd` on tank moved index 1 from 7441 to
|
||||||
|
/// 23688 while index 0 stayed near zero.
|
||||||
|
///
|
||||||
|
/// `info.ct` is NOT mapped to `container_count`: it reads 1 on tank (1
|
||||||
|
/// container) and also 1 on architect (4 containers), so whatever it counts, it
|
||||||
|
/// is not that.
|
||||||
|
fn metrics_from_system(system: &Value, stats: Option<&Value>) -> NodeMetrics {
|
||||||
let info = system.get("info").cloned().unwrap_or_else(|| json!({}));
|
let info = system.get("info").cloned().unwrap_or_else(|| json!({}));
|
||||||
let load1 = info
|
let load1 = info
|
||||||
.get("la")
|
.get("la")
|
||||||
.and_then(Value::as_array)
|
.and_then(Value::as_array)
|
||||||
.and_then(|a| a.first())
|
.and_then(|a| a.first())
|
||||||
.and_then(Value::as_f64);
|
.and_then(Value::as_f64);
|
||||||
|
let empty = json!({});
|
||||||
|
let st = stats.unwrap_or(&empty);
|
||||||
NodeMetrics {
|
NodeMetrics {
|
||||||
cpu_pct: f(&info, "cpu"),
|
cpu_pct: f(&info, "cpu"),
|
||||||
mem_pct: f(&info, "mp"),
|
mem_pct: f(&info, "mp"),
|
||||||
disk_pct: f(&info, "dp"),
|
disk_pct: f(&info, "dp"),
|
||||||
gpu_pct: f(&info, "g"),
|
gpu_pct: gpu_busiest(st),
|
||||||
temp_max: f(&info, "dt"),
|
temp_max: f(&info, "dt"),
|
||||||
net_sent_ps: None,
|
net_sent_ps: pair(st, "b", 0),
|
||||||
net_recv_ps: None,
|
net_recv_ps: pair(st, "b", 1),
|
||||||
disk_read_ps: None,
|
disk_read_ps: pair(st, "dio", 0),
|
||||||
disk_write_ps: None,
|
disk_write_ps: pair(st, "dio", 1),
|
||||||
load1,
|
load1,
|
||||||
container_count: None,
|
container_count: None,
|
||||||
data: json!({
|
data: json!({
|
||||||
@@ -115,6 +210,10 @@ fn metrics_from_system(system: &Value) -> NodeMetrics {
|
|||||||
"name": system.get("name").and_then(Value::as_str),
|
"name": system.get("name").and_then(Value::as_str),
|
||||||
"host": system.get("host").and_then(Value::as_str),
|
"host": system.get("host").and_then(Value::as_str),
|
||||||
"info": info,
|
"info": info,
|
||||||
|
// The GPU roster, so a card can name the card rather than only
|
||||||
|
// report a percentage.
|
||||||
|
"gpus": st.get("g").cloned().unwrap_or(Value::Null),
|
||||||
|
"temps": st.get("t").cloned().unwrap_or(Value::Null),
|
||||||
}),
|
}),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -129,6 +228,7 @@ pub async fn poll_workspace(
|
|||||||
) -> Result<usize, String> {
|
) -> Result<usize, String> {
|
||||||
let token = authenticate(client, conn).await?;
|
let token = authenticate(client, conn).await?;
|
||||||
let systems = fetch_systems(client, conn, &token).await?;
|
let systems = fetch_systems(client, conn, &token).await?;
|
||||||
|
let stats = fetch_latest_stats(client, conn, &token).await;
|
||||||
let node_rows = nodes::list(pool, ws).await.map_err(|e| e.to_string())?;
|
let node_rows = nodes::list(pool, ws).await.map_err(|e| e.to_string())?;
|
||||||
// hostname/name (lowercased) → node id.
|
// hostname/name (lowercased) → node id.
|
||||||
let mut by_host: HashMap<String, NodeId> = HashMap::new();
|
let mut by_host: HashMap<String, NodeId> = HashMap::new();
|
||||||
@@ -148,7 +248,11 @@ pub async fn poll_workspace(
|
|||||||
let Some(node_id) = key.as_deref().and_then(|k| by_host.get(k).copied()) else {
|
let Some(node_id) = key.as_deref().and_then(|k| by_host.get(k).copied()) else {
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
if node_metrics::upsert(pool, node_id, &metrics_from_system(sys))
|
let sample = sys
|
||||||
|
.get("id")
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.and_then(|id| stats.get(id));
|
||||||
|
if node_metrics::upsert(pool, node_id, &metrics_from_system(sys, sample))
|
||||||
.await
|
.await
|
||||||
.is_ok()
|
.is_ok()
|
||||||
{
|
{
|
||||||
@@ -178,3 +282,93 @@ pub fn spawn_poller(pool: PgPool, interval: Duration) {
|
|||||||
}
|
}
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// A real 0.18.7 sample, copied from tank rather than invented.
|
||||||
|
fn sample() -> Value {
|
||||||
|
json!({
|
||||||
|
"b": [1830, 1811],
|
||||||
|
"dio": [204, 23688],
|
||||||
|
"g": { "0": { "n": "GeForce RTX 5060 Ti", "u": 37.5, "p": 4.38 } },
|
||||||
|
"t": { "GeForce RTX 5060 Ti": 29, "k10temp_tctl": 38.38 }
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
fn system() -> Value {
|
||||||
|
json!({
|
||||||
|
"id": "glo9hj260jhnlgr",
|
||||||
|
"name": "tank",
|
||||||
|
"host": "100.108.129.81",
|
||||||
|
"status": "up",
|
||||||
|
"info": { "cpu": 0.31, "mp": 7.14, "dp": 77.96, "dt": 38.85, "la": [0.03, 0.01, 0], "ct": 1 }
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// GPU comes from the stats sample's MAP, not from `info`.
|
||||||
|
///
|
||||||
|
/// This is the bug the whole change exists for: `info` carries no `g` in
|
||||||
|
/// 0.18, so reading it as a scalar produced null on every NVIDIA node while
|
||||||
|
/// the data sat one collection away. Null and "no GPU" are indistinguishable
|
||||||
|
/// downstream, so metrics-aware placement simply never saw a GPU.
|
||||||
|
#[test]
|
||||||
|
fn gpu_comes_from_the_stats_sample_not_the_info_snapshot() {
|
||||||
|
let m = metrics_from_system(&system(), Some(&sample()));
|
||||||
|
assert_eq!(m.gpu_pct, Some(37.5));
|
||||||
|
// No sample ⇒ no GPU claim. NOT zero: "we did not get a reading" and
|
||||||
|
// "the card is idle" are different facts.
|
||||||
|
assert_eq!(metrics_from_system(&system(), None).gpu_pct, None);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The busiest card, not the average.
|
||||||
|
#[test]
|
||||||
|
fn a_saturated_card_is_not_averaged_away_by_an_idle_one() {
|
||||||
|
let two = json!({ "g": { "0": { "u": 99.0 }, "1": { "u": 1.0 } } });
|
||||||
|
assert_eq!(gpu_busiest(&two), Some(99.0));
|
||||||
|
assert_eq!(gpu_busiest(&json!({})), None);
|
||||||
|
// Present but empty is still no reading.
|
||||||
|
assert_eq!(gpu_busiest(&json!({ "g": {} })), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The measured array orders. An inverted pair does not fail — it reports
|
||||||
|
/// upload as download, and disk reads as writes, forever.
|
||||||
|
///
|
||||||
|
/// `b` = [sent, recv]: `stats.ni` per-interface indices 2 and 3 matched
|
||||||
|
/// `/proc/net/dev` tx_bytes and rx_bytes on all four of tank's
|
||||||
|
/// interfaces, and `b` is the sum of the per-second pair.
|
||||||
|
/// `dio` = [read, write]: an 800 MB `dd` moved index 1 from 7441 to 23688
|
||||||
|
/// while index 0 stayed near zero.
|
||||||
|
#[test]
|
||||||
|
fn the_measured_array_orders_are_not_reinverted() {
|
||||||
|
let m = metrics_from_system(&system(), Some(&sample()));
|
||||||
|
assert_eq!(m.net_sent_ps, Some(1830), "b[0] is SENT");
|
||||||
|
assert_eq!(m.net_recv_ps, Some(1811), "b[1] is RECV");
|
||||||
|
assert_eq!(m.disk_read_ps, Some(204), "dio[0] is READ");
|
||||||
|
assert_eq!(m.disk_write_ps, Some(23688), "dio[1] is WRITE");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `info.ct` must not become `container_count`.
|
||||||
|
///
|
||||||
|
/// It reads 1 on tank, which runs 1 container, and ALSO 1 on architect,
|
||||||
|
/// which runs 4. It agrees with the truth exactly often enough to look
|
||||||
|
/// right in a spot check.
|
||||||
|
#[test]
|
||||||
|
fn the_unidentified_ct_field_is_not_reported_as_a_container_count() {
|
||||||
|
let m = metrics_from_system(&system(), Some(&sample()));
|
||||||
|
assert_eq!(m.container_count, None);
|
||||||
|
assert_eq!(system()["info"]["ct"], json!(1));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The snapshot fields keep working when the stats call fails.
|
||||||
|
#[test]
|
||||||
|
fn a_missing_stats_sample_does_not_cost_the_metrics_that_still_work() {
|
||||||
|
let m = metrics_from_system(&system(), None);
|
||||||
|
assert_eq!(m.cpu_pct, Some(0.31));
|
||||||
|
assert_eq!(m.mem_pct, Some(7.14));
|
||||||
|
assert_eq!(m.temp_max, Some(38.85));
|
||||||
|
assert_eq!(m.load1, Some(0.03));
|
||||||
|
assert_eq!(m.net_sent_ps, None);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -77,6 +77,51 @@ pub fn connect() -> Result<Docker, String> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The uid every mission artefact must belong to.
|
||||||
|
///
|
||||||
|
/// The runtime container's own processes already run as this; only `docker
|
||||||
|
/// exec` defaulted to root, because `CreateExecOptions::user` was never set.
|
||||||
|
/// That one omission is the origin of four separate patches: root-owned
|
||||||
|
/// `target/` directories appearing inside a checkout that uid 65532 then could
|
||||||
|
/// not delete, `root_copy` existing at all, and a cleanup path that had to
|
||||||
|
/// re-enter the container as root to undo what it had just done.
|
||||||
|
pub(crate) const MISSION_UID: &str = "65532:65532";
|
||||||
|
|
||||||
|
/// Environment a non-root exec needs, because the image gives uid 65532 no
|
||||||
|
/// writable `HOME` and no writable `CARGO_HOME`.
|
||||||
|
///
|
||||||
|
/// Measured in the deployed image: `/zeroclaw-data` (its `HOME`) and
|
||||||
|
/// `/usr/local/cargo` are both root-owned and unwritable, so switching execs to
|
||||||
|
/// 65532 without this would break every `cargo` invocation — the benchmark
|
||||||
|
/// runner, the judge's verification sandbox, and the delivery test gate — in a
|
||||||
|
/// new and much quieter way than the problem it fixes.
|
||||||
|
///
|
||||||
|
/// The missions root is bind-mounted into the runtime container at the same
|
||||||
|
/// path and IS writable by 65532, so the cargo cache lives there and is shared
|
||||||
|
/// across missions rather than re-downloaded per mission. Verified end to end:
|
||||||
|
/// a clean `cargo build` as 65532 with these three variables produces output
|
||||||
|
/// owned entirely by 65532.
|
||||||
|
fn mission_env() -> Vec<String> {
|
||||||
|
let root = crate::mission_workspace::missions_root();
|
||||||
|
vec![
|
||||||
|
format!("HOME={}", root.join("_home").display()),
|
||||||
|
format!("CARGO_HOME={}", root.join("_cargo").display()),
|
||||||
|
"TMPDIR=/tmp".to_string(),
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a workdir is inside the tree missions own.
|
||||||
|
///
|
||||||
|
/// The rule is positional rather than per-caller on purpose. Twelve call sites
|
||||||
|
/// each remembering to pass a uid is twelve chances to forget, and the one that
|
||||||
|
/// forgets leaves debris the others cannot clean up — which is exactly the
|
||||||
|
/// history here.
|
||||||
|
fn is_mission_path(workdir: Option<&str>) -> bool {
|
||||||
|
let Some(dir) = workdir else { return false };
|
||||||
|
let root = crate::mission_workspace::missions_root();
|
||||||
|
std::path::Path::new(dir).starts_with(&root)
|
||||||
|
}
|
||||||
|
|
||||||
/// Run `argv` in `container`, optionally in `workdir`, and capture both
|
/// Run `argv` in `container`, optionally in `workdir`, and capture both
|
||||||
/// streams plus the exit status.
|
/// streams plus the exit status.
|
||||||
///
|
///
|
||||||
@@ -93,6 +138,29 @@ pub async fn exec(
|
|||||||
exec_with_env(docker, container, workdir, argv, &[], timeout).await
|
exec_with_env(docker, container, workdir, argv, &[], timeout).await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Run `argv` as **root**, deliberately.
|
||||||
|
///
|
||||||
|
/// The one legitimate use is clearing debris that earlier root-run execs left
|
||||||
|
/// behind: uid 65532 cannot delete a root-owned `target/`, so the cleanup has
|
||||||
|
/// to out-rank it. Every other caller goes through [`exec`], which runs mission
|
||||||
|
/// work as 65532 so no new debris is created.
|
||||||
|
pub async fn exec_as_root(
|
||||||
|
docker: &Docker,
|
||||||
|
container: &str,
|
||||||
|
workdir: Option<&str>,
|
||||||
|
argv: &[String],
|
||||||
|
timeout: Duration,
|
||||||
|
) -> Result<ExecOutput, String> {
|
||||||
|
let fut = exec_inner(docker, container, workdir, argv, &[], None);
|
||||||
|
match tokio::time::timeout(timeout, fut).await {
|
||||||
|
Err(_) => Err(format!(
|
||||||
|
"timed out after {}s (the command may still be running in {container})",
|
||||||
|
timeout.as_secs()
|
||||||
|
)),
|
||||||
|
Ok(res) => res,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// As [`exec`], with extra environment for the command.
|
/// As [`exec`], with extra environment for the command.
|
||||||
pub async fn exec_with_env(
|
pub async fn exec_with_env(
|
||||||
docker: &Docker,
|
docker: &Docker,
|
||||||
@@ -102,7 +170,16 @@ pub async fn exec_with_env(
|
|||||||
env: &[String],
|
env: &[String],
|
||||||
timeout: Duration,
|
timeout: Duration,
|
||||||
) -> Result<ExecOutput, String> {
|
) -> Result<ExecOutput, String> {
|
||||||
let fut = exec_inner(docker, container, workdir, argv, env);
|
// Mission work runs as 65532 with a writable HOME/CARGO_HOME; anything
|
||||||
|
// outside the missions tree (runtime preflight probes, image checks) keeps
|
||||||
|
// the daemon's default so this cannot break unrelated call sites.
|
||||||
|
let (user, mut full_env) = if is_mission_path(workdir) {
|
||||||
|
(Some(MISSION_UID), mission_env())
|
||||||
|
} else {
|
||||||
|
(None, Vec::new())
|
||||||
|
};
|
||||||
|
full_env.extend_from_slice(env);
|
||||||
|
let fut = exec_inner(docker, container, workdir, argv, &full_env, user);
|
||||||
match tokio::time::timeout(timeout, fut).await {
|
match tokio::time::timeout(timeout, fut).await {
|
||||||
Err(_) => Err(format!(
|
Err(_) => Err(format!(
|
||||||
"timed out after {}s (the command may still be running in {container})",
|
"timed out after {}s (the command may still be running in {container})",
|
||||||
@@ -118,6 +195,7 @@ async fn exec_inner(
|
|||||||
workdir: Option<&str>,
|
workdir: Option<&str>,
|
||||||
argv: &[String],
|
argv: &[String],
|
||||||
env: &[String],
|
env: &[String],
|
||||||
|
user: Option<&str>,
|
||||||
) -> Result<ExecOutput, String> {
|
) -> Result<ExecOutput, String> {
|
||||||
let created = docker
|
let created = docker
|
||||||
.create_exec(
|
.create_exec(
|
||||||
@@ -130,6 +208,7 @@ async fn exec_inner(
|
|||||||
} else {
|
} else {
|
||||||
Some(env.to_vec())
|
Some(env.to_vec())
|
||||||
},
|
},
|
||||||
|
user: user.map(str::to_string),
|
||||||
attach_stdout: Some(true),
|
attach_stdout: Some(true),
|
||||||
attach_stderr: Some(true),
|
attach_stderr: Some(true),
|
||||||
..Default::default()
|
..Default::default()
|
||||||
@@ -209,4 +288,118 @@ mod tests {
|
|||||||
assert_eq!(out(Some(1), "a", "b").combined(), "a\nb");
|
assert_eq!(out(Some(1), "a", "b").combined(), "a\nb");
|
||||||
assert_eq!(out(Some(0), " ", "\n").combined(), "");
|
assert_eq!(out(Some(0), " ", "\n").combined(), "");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Mission work is 65532; everything else keeps the daemon's default.
|
||||||
|
///
|
||||||
|
/// The rule is positional so that no caller has to remember it. Twelve call
|
||||||
|
/// sites each passing a uid is twelve chances to forget, and the one that
|
||||||
|
/// forgets leaves debris the other eleven cannot delete — which is the
|
||||||
|
/// actual history: root-owned `target/` directories inside a checkout owned
|
||||||
|
/// by 65532, `root_copy` written to work around them, and a cleanup that had
|
||||||
|
/// to re-enter the container as root to undo its own mess.
|
||||||
|
#[test]
|
||||||
|
fn only_work_inside_the_missions_tree_drops_to_the_mission_uid() {
|
||||||
|
let root = crate::mission_workspace::missions_root();
|
||||||
|
let inside = root.join("019fe785-0f82-7780-8d58-da79fb4c31bc/repo");
|
||||||
|
assert!(is_mission_path(Some(&inside.display().to_string())));
|
||||||
|
assert!(is_mission_path(Some(&root.display().to_string())));
|
||||||
|
|
||||||
|
// Probes and image checks run with no workdir at all, and must not be
|
||||||
|
// forced to a uid the image may not have set up for them.
|
||||||
|
assert!(!is_mission_path(None));
|
||||||
|
assert!(!is_mission_path(Some("/")));
|
||||||
|
assert!(!is_mission_path(Some("/usr/local/cargo")));
|
||||||
|
// A path that merely SHARES A PREFIX is not inside the tree.
|
||||||
|
// `starts_with` on `Path` compares components, so this is already true;
|
||||||
|
// the assertion is here so a switch to string matching cannot pass.
|
||||||
|
let sibling = format!("{}-evil/repo", root.display());
|
||||||
|
assert!(!is_mission_path(Some(&sibling)));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The non-root exec carries the three variables the image does not give it.
|
||||||
|
///
|
||||||
|
/// Measured in the deployed image: uid 65532's `HOME` (`/zeroclaw-data`)
|
||||||
|
/// and `/usr/local/cargo` are both root-owned and unwritable. Without these
|
||||||
|
/// overrides, dropping execs to 65532 would break every cargo invocation —
|
||||||
|
/// the benchmark runner, the judge's sandbox, the delivery test gate — far
|
||||||
|
/// more quietly than the leak it fixes.
|
||||||
|
#[test]
|
||||||
|
fn the_mission_env_replaces_the_paths_the_image_leaves_unwritable() {
|
||||||
|
let env = mission_env();
|
||||||
|
let root = crate::mission_workspace::missions_root();
|
||||||
|
assert!(env.iter().any(|v| v == &format!("HOME={}/_home", root.display())));
|
||||||
|
assert!(env.iter().any(|v| v == &format!("CARGO_HOME={}/_cargo", root.display())));
|
||||||
|
assert!(env.iter().any(|v| v == "TMPDIR=/tmp"));
|
||||||
|
for v in &env {
|
||||||
|
assert!(
|
||||||
|
!v.contains("/usr/local/cargo") && !v.contains("/zeroclaw-data"),
|
||||||
|
"{v} points back at a root-owned path"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One place builds an exec, so one place decides its uid.
|
||||||
|
///
|
||||||
|
/// The original bug was not a wrong value — it was an ABSENT one:
|
||||||
|
/// `CreateExecOptions` never set `user`, so the daemon defaulted to root
|
||||||
|
/// and twelve callers inherited that without any of them choosing it. A
|
||||||
|
/// second construction site is how that comes back, so the guard is on the
|
||||||
|
/// number of sites rather than on any particular uid.
|
||||||
|
#[test]
|
||||||
|
fn exactly_one_place_builds_an_exec() {
|
||||||
|
let src = include_str!("container_exec.rs");
|
||||||
|
// Split so this needle does not match itself in this very file.
|
||||||
|
let needle = concat!("CreateExec", "Options {");
|
||||||
|
let sites = src.matches(needle).count();
|
||||||
|
assert_eq!(
|
||||||
|
sites, 1,
|
||||||
|
"exec options must be built in one place; found {sites}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
src.contains(concat!("user: ", "user.map(str::to_string)")),
|
||||||
|
"that one place must set `user` — leaving it unset is the bug"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The tail of a container's log, for putting in an error message.
|
||||||
|
///
|
||||||
|
/// A turn that times out destroys the only place the reason lived: the
|
||||||
|
/// per-mission runtime container is torn down after the phase, taking its logs
|
||||||
|
/// with it, and the operator is left with the string "turn timed out". This
|
||||||
|
/// copies the last few lines out while the container still exists.
|
||||||
|
///
|
||||||
|
/// Best-effort by construction — it runs on a path that is ALREADY failing, so
|
||||||
|
/// every error here degrades to a note rather than replacing the real failure
|
||||||
|
/// with a docker one.
|
||||||
|
pub async fn tail_logs(container: &str, lines: usize) -> String {
|
||||||
|
use futures::StreamExt as _;
|
||||||
|
|
||||||
|
let Ok(docker) = connect() else {
|
||||||
|
return "(docker unreachable, so no container log)".into();
|
||||||
|
};
|
||||||
|
let opts = bollard::query_parameters::LogsOptionsBuilder::default()
|
||||||
|
.stdout(true)
|
||||||
|
.stderr(true)
|
||||||
|
.tail(&lines.to_string())
|
||||||
|
.build();
|
||||||
|
let mut stream = docker.logs(container, Some(opts));
|
||||||
|
let mut out = String::new();
|
||||||
|
while let Some(chunk) = stream.next().await {
|
||||||
|
match chunk {
|
||||||
|
Ok(c) => out.push_str(&c.to_string()),
|
||||||
|
Err(e) => {
|
||||||
|
if out.is_empty() {
|
||||||
|
return format!("(could not read {container} logs: {e})");
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let out = out.trim();
|
||||||
|
if out.is_empty() {
|
||||||
|
format!("({container} logged nothing)")
|
||||||
|
} else {
|
||||||
|
out.to_string()
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,468 @@
|
|||||||
|
//! Gate and observe the tools a **container-tier** mission agent runs.
|
||||||
|
//!
|
||||||
|
//! The container tier is the one that actually runs missions in production,
|
||||||
|
//! and until now it had neither. Both gaps have the same cause: `claude_cli`
|
||||||
|
//! runs claude as a subprocess, claude runs its tools inside that subprocess,
|
||||||
|
//! and so those calls never pass through ZeroClaw's executor — which is the
|
||||||
|
//! only thing that emits `TurnEvent::ToolCall`, and therefore the only thing
|
||||||
|
//! the gateway turns into a frame ClawMates can see. Recovering the calls from
|
||||||
|
//! the CLI's own `stream-json` output does not help either: the transport was
|
||||||
|
//! never the problem, and a mission proved it by producing zero `tool.call`
|
||||||
|
//! events with the parser working perfectly.
|
||||||
|
//!
|
||||||
|
//! Hooks are the way in, and they are already proven. Claude Code reads
|
||||||
|
//! `hooks.PreToolUse` / `PostToolUse` from the document passed to `--settings`
|
||||||
|
//! and honours them under `-p` — measured against the real binary, where a
|
||||||
|
//! `PreToolUse` hook blocked a `Bash` call, recorded the payload, and got its
|
||||||
|
//! refusal reason back to the model.
|
||||||
|
//!
|
||||||
|
//! So this module writes the same hook scripts the microVM tier already uses
|
||||||
|
//! into the mission's container, and the provider is pointed at the settings
|
||||||
|
//! document. One mechanism, two tiers.
|
||||||
|
//!
|
||||||
|
//! # Everything here degrades to "no hooks", never to a failed mission
|
||||||
|
//!
|
||||||
|
//! A phase that runs unobserved still delivers. A phase that fails to start
|
||||||
|
//! because telemetry could not be installed delivers nothing, which is a worse
|
||||||
|
//! trade — the same stance `microvm_executor` takes for the same reason.
|
||||||
|
|
||||||
|
use bollard::Docker;
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
/// Where the hooks live inside the mission container.
|
||||||
|
///
|
||||||
|
/// Under `/root`, never under `/mission/repo`: anything written into the
|
||||||
|
/// checkout would show up in the diff the mission delivers.
|
||||||
|
pub const HOOK_DIR: &str = "/root/toolhooks";
|
||||||
|
/// The settings document `claude -p --settings` is pointed at.
|
||||||
|
pub const SETTINGS_PATH: &str = "/root/toolhooks/settings.json";
|
||||||
|
/// Where the `PostToolUse` tap appends, inside the container.
|
||||||
|
pub const TAP_DIR: &str = "/root/toolhooks/tap";
|
||||||
|
|
||||||
|
pub const INSTALL_TIMEOUT: Duration = Duration::from_secs(30);
|
||||||
|
|
||||||
|
/// Install the pre-execution gate and the tool tap into a mission container.
|
||||||
|
///
|
||||||
|
/// Returns the settings path on success. `None` means the container runs
|
||||||
|
/// without hooks — logged, never fatal.
|
||||||
|
pub async fn install(docker: &Docker, container: &str) -> Option<String> {
|
||||||
|
let script = build_install_script();
|
||||||
|
let argv = vec!["sh".to_string(), "-lc".to_string(), script];
|
||||||
|
match crate::container_exec::exec_as_root(docker, container, None, &argv, INSTALL_TIMEOUT).await
|
||||||
|
{
|
||||||
|
Ok(out) if out.exit_code == Some(0) => Some(SETTINGS_PATH.to_string()),
|
||||||
|
other => {
|
||||||
|
eprintln!(
|
||||||
|
"container_tool_hooks: could not install hooks in {container} ({other:?}) — \
|
||||||
|
this mission's tool calls will run unchecked and unrecorded"
|
||||||
|
);
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One shell script that lays down both hooks and the settings document.
|
||||||
|
///
|
||||||
|
/// Composed here rather than by each hook module writing its own file: two
|
||||||
|
/// writers of one `settings.json` is a silent clobber, and the microVM tier
|
||||||
|
/// already learned that the expensive way.
|
||||||
|
fn build_install_script() -> String {
|
||||||
|
let settings = crate::vm_tool_tap::guest_settings(
|
||||||
|
None,
|
||||||
|
Some(TAP_DIR),
|
||||||
|
Some(HOOK_DIR),
|
||||||
|
);
|
||||||
|
format!(
|
||||||
|
"set -e\n\
|
||||||
|
mkdir -p {hooks} {tap}\n\
|
||||||
|
cat > {hooks}/tool-gate.sh <<'CM_GATE_EOF'\n{gate}\nCM_GATE_EOF\n\
|
||||||
|
chmod +x {hooks}/tool-gate.sh\n\
|
||||||
|
cat > {tap}/tap.sh <<'CM_TAP_EOF'\n{tap_script}\nCM_TAP_EOF\n\
|
||||||
|
chmod +x {tap}/tap.sh\n\
|
||||||
|
cat > {settings_path} <<'CM_SETTINGS_EOF'\n{settings}\nCM_SETTINGS_EOF\n",
|
||||||
|
hooks = HOOK_DIR,
|
||||||
|
tap = TAP_DIR,
|
||||||
|
gate = crate::vm_tool_gate::hook_script(HOOK_DIR),
|
||||||
|
tap_script = crate::vm_tool_tap::hook_script(TAP_DIR),
|
||||||
|
settings_path = SETTINGS_PATH,
|
||||||
|
settings = settings,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The MCP configuration `claude -p --mcp-config` is pointed at.
|
||||||
|
///
|
||||||
|
/// Under `/root` with the hooks, never under `/mission/repo`: it carries a
|
||||||
|
/// bearer token, and anything written into the checkout arrives in the diff the
|
||||||
|
/// mission delivers.
|
||||||
|
pub const MCP_CONFIG_PATH: &str = "/root/toolhooks/clawmates-mcp.json";
|
||||||
|
|
||||||
|
/// Where the mission container reaches this server.
|
||||||
|
///
|
||||||
|
/// Mission containers join `clawmates_core`, the same network the API is on, so
|
||||||
|
/// the API is reachable by container name. The name differs between
|
||||||
|
/// deployments (`clawmates-server-1` locally, `clawmates_server_1` on gw-04),
|
||||||
|
/// so the default is derived from **our own** hostname — docker's embedded DNS
|
||||||
|
/// resolves a container id on a user-defined network, which makes this
|
||||||
|
/// self-configuring rather than a constant that is right in one place.
|
||||||
|
/// Measured from a sibling container: both the id and the name return 200.
|
||||||
|
pub fn api_origin() -> Option<String> {
|
||||||
|
if let Ok(v) = std::env::var("CLAWMATES_API_ORIGIN") {
|
||||||
|
if !v.trim().is_empty() {
|
||||||
|
return Some(v.trim().trim_end_matches('/').to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let host = std::env::var("HOSTNAME").ok()?;
|
||||||
|
let host = host.trim();
|
||||||
|
if host.is_empty() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(format!("http://{host}:8080"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The `--mcp-config` document: one HTTP server, carrying its own credential.
|
||||||
|
///
|
||||||
|
/// The token is a `skills:read` session and nothing else. It is written into a
|
||||||
|
/// file the agent can read — it runs `Bash` — so the only thing keeping this
|
||||||
|
/// safe is that the credential authenticates to exactly one route. See
|
||||||
|
/// `cm_auth::authenticate_scoped`.
|
||||||
|
pub fn mcp_document(origin: &str, token: &str) -> serde_json::Value {
|
||||||
|
serde_json::json!({
|
||||||
|
"mcpServers": {
|
||||||
|
"clawmates_skills": {
|
||||||
|
"type": "http",
|
||||||
|
"url": format!("{origin}/mcp/skills"),
|
||||||
|
"headers": { "Authorization": format!("Bearer {token}") }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// NOTE on `--allowedTools`. The provider passes it only when the config sets
|
||||||
|
// `tools`, and the seed already does — without it `claude -p` stops mid-turn to
|
||||||
|
// ask for write permission. Whether the MCP tools ALSO need naming there is not
|
||||||
|
// documented anywhere we control, and the daemon exposes no config read to
|
||||||
|
// merge into that list safely: overwriting it would take `Write` and `Bash`
|
||||||
|
// away from every mission agent, and that failure would look like agents that
|
||||||
|
// stopped working rather than a config that was replaced.
|
||||||
|
//
|
||||||
|
// So it is left alone and the question is answered by running a mission with
|
||||||
|
// the door installed. Guessing here is how the last three defects in this file
|
||||||
|
// were introduced.
|
||||||
|
|
||||||
|
/// Write the MCP configuration into a mission container.
|
||||||
|
///
|
||||||
|
/// Returns the path on success. `None` means the mission runs without a door —
|
||||||
|
/// logged, never fatal, exactly like the hooks above. A phase that cannot
|
||||||
|
/// retrieve a skill still delivers; a phase that fails to start because a
|
||||||
|
/// config write failed delivers nothing.
|
||||||
|
pub async fn install_door(docker: &Docker, container: &str, doc: &serde_json::Value) -> Option<String> {
|
||||||
|
// `printf %s` with the JSON single-quoted, not a heredoc: the document is
|
||||||
|
// one line and contains no newline to terminate on.
|
||||||
|
let script = format!(
|
||||||
|
"mkdir -p {HOOK_DIR} && printf '%s' {} > {MCP_CONFIG_PATH} && chmod 600 {MCP_CONFIG_PATH}",
|
||||||
|
crate::vm_tool_tap::shell_quote(&doc.to_string()),
|
||||||
|
);
|
||||||
|
let argv = vec!["sh".to_string(), "-lc".to_string(), script];
|
||||||
|
match crate::container_exec::exec_as_root(docker, container, None, &argv, INSTALL_TIMEOUT).await
|
||||||
|
{
|
||||||
|
Ok(out) if out.exit_code == Some(0) => Some(MCP_CONFIG_PATH.to_string()),
|
||||||
|
other => {
|
||||||
|
eprintln!(
|
||||||
|
"container_tool_hooks: could not write the MCP config in {container} ({other:?}) — this mission runs without the skills door"
|
||||||
|
);
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Event kinds under which the gate's own state lands in the mission record.
|
||||||
|
///
|
||||||
|
/// Recorded, not only logged, so "was this mission gated?" is answerable from
|
||||||
|
/// the mission afterwards. Stderr is where the answer used to go, which is the
|
||||||
|
/// same place as nowhere once the container that printed it is gone.
|
||||||
|
pub const GATE_INSTALLED: &str = "gate.installed";
|
||||||
|
pub const GATE_ABSENT: &str = "gate.absent";
|
||||||
|
/// The gate ran but could not parse its input and allowed everything. See
|
||||||
|
/// [`crate::vm_tool_gate::INERT_FILE`] — this is the reader that marker was
|
||||||
|
/// missing in production; until now only a unit test looked for it.
|
||||||
|
pub const GATE_INERT: &str = "gate.inert";
|
||||||
|
/// One call the gate refused. `detail` is the hook event with `rule` set
|
||||||
|
/// beside it — see [`crate::vm_tool_gate::denial_detail`]. Both tiers.
|
||||||
|
pub const GATE_DENIED: &str = "gate.denied";
|
||||||
|
|
||||||
|
/// Write the install outcome into the mission record.
|
||||||
|
pub async fn record_install(
|
||||||
|
pool: &sqlx::PgPool,
|
||||||
|
mission_id: uuid::Uuid,
|
||||||
|
phase_id: Option<uuid::Uuid>,
|
||||||
|
hooks: Option<&str>,
|
||||||
|
) {
|
||||||
|
let mut e = match hooks {
|
||||||
|
Some(path) => crate::mission_events::MissionEvent::new(mission_id, GATE_INSTALLED)
|
||||||
|
.target(path)
|
||||||
|
.detail(serde_json::json!({ "settings": path, "tap": tap_file() })),
|
||||||
|
None => crate::mission_events::MissionEvent::new(mission_id, GATE_ABSENT).detail(
|
||||||
|
serde_json::json!({
|
||||||
|
"why": "container_tool_hooks::install failed — this mission's tool \
|
||||||
|
calls run unchecked and unrecorded"
|
||||||
|
}),
|
||||||
|
),
|
||||||
|
};
|
||||||
|
if let Some(p) = phase_id {
|
||||||
|
e = e.phase(p);
|
||||||
|
}
|
||||||
|
crate::mission_events::record(pool, e).await;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The inert marker's path inside the container.
|
||||||
|
pub fn inert_file() -> String {
|
||||||
|
format!("{HOOK_DIR}/{}", crate::vm_tool_gate::INERT_FILE)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Did the gate go inert since the last drain? Reads the marker and clears
|
||||||
|
/// it, so each occurrence is reported once.
|
||||||
|
///
|
||||||
|
/// `Some(text)` is the marker's contents — every line the gate appended while
|
||||||
|
/// it could not parse. `None` is "the marker is not there", which is the
|
||||||
|
/// normal case and also, by construction, the only case that means the gate
|
||||||
|
/// was actually checking.
|
||||||
|
pub async fn drain_inert(docker: &Docker, container: &str) -> Option<String> {
|
||||||
|
let file = inert_file();
|
||||||
|
let script = format!("cat {file} 2>/dev/null && rm -f {file} 2>/dev/null; true");
|
||||||
|
let argv = vec!["sh".to_string(), "-lc".to_string(), script];
|
||||||
|
match crate::container_exec::exec_as_root(docker, container, None, &argv, INSTALL_TIMEOUT).await
|
||||||
|
{
|
||||||
|
Ok(out) if !out.stdout.trim().is_empty() => Some(out.stdout.trim().to_string()),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The gate's denial record inside the mission container.
|
||||||
|
pub fn denied_file() -> String {
|
||||||
|
format!("{HOOK_DIR}/{}", crate::vm_tool_gate::DENIED_FILE)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every call the gate refused since the last drain, one JSON line each
|
||||||
|
/// (`vm_tool_gate::denial_detail` reads them). Read-then-truncate, like
|
||||||
|
/// [`drain`], for the same reason: no cursor to keep, and the phase has
|
||||||
|
/// finished so nothing is appending.
|
||||||
|
pub async fn drain_denied(docker: &Docker, container: &str) -> Vec<String> {
|
||||||
|
let file = denied_file();
|
||||||
|
let script = format!("cat {file} 2>/dev/null || true; : > {file} 2>/dev/null || true");
|
||||||
|
let argv = vec!["sh".to_string(), "-lc".to_string(), script];
|
||||||
|
match crate::container_exec::exec_as_root(docker, container, None, &argv, INSTALL_TIMEOUT).await
|
||||||
|
{
|
||||||
|
Ok(out) => out
|
||||||
|
.stdout
|
||||||
|
.lines()
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|l| !l.is_empty())
|
||||||
|
.map(str::to_string)
|
||||||
|
.collect(),
|
||||||
|
Err(_) => Vec::new(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The tap file inside the mission container.
|
||||||
|
pub fn tap_file() -> String {
|
||||||
|
format!("{TAP_DIR}/tools.jsonl")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Read everything the tap recorded, then clear it.
|
||||||
|
///
|
||||||
|
/// Read-then-truncate rather than a cursor, because this tier has no
|
||||||
|
/// long-lived loop to hold one: the microVM path drains inside the turn it is
|
||||||
|
/// watching, while a container turn is driven asynchronously by
|
||||||
|
/// `topology_worker`. Truncation makes the drain idempotent — a second pass
|
||||||
|
/// reads an empty file and records nothing — without a column to store a
|
||||||
|
/// cursor in.
|
||||||
|
///
|
||||||
|
/// Called only for phases that have FINISHED, so the agent is no longer
|
||||||
|
/// appending and the read/truncate gap cannot lose an event.
|
||||||
|
pub async fn drain(docker: &Docker, container: &str) -> Vec<crate::vm_tool_tap::Observed> {
|
||||||
|
let file = tap_file();
|
||||||
|
// `cat` then truncate in one exec: two round-trips would widen the window
|
||||||
|
// between them for no benefit.
|
||||||
|
let script = format!("cat {file} 2>/dev/null || true; : > {file} 2>/dev/null || true");
|
||||||
|
let argv = vec!["sh".to_string(), "-lc".to_string(), script];
|
||||||
|
match crate::container_exec::exec_as_root(docker, container, None, &argv, INSTALL_TIMEOUT).await
|
||||||
|
{
|
||||||
|
Ok(out) => crate::vm_tool_tap::parse(&out.stdout),
|
||||||
|
Err(e) => {
|
||||||
|
// A reaped container is the normal end state, not a fault.
|
||||||
|
eprintln!("container_tool_hooks: no tap drained from {container}: {e}");
|
||||||
|
Vec::new()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// Every command the settings document names must be a file the installer
|
||||||
|
/// actually writes.
|
||||||
|
///
|
||||||
|
/// This caught a real one: the document pointed PostToolUse at
|
||||||
|
/// `{TAP_DIR}/tap.sh` while the installer wrote `{HOOK_DIR}/tap.sh`, so
|
||||||
|
/// the hook referenced a file that did not exist. Claude Code does not
|
||||||
|
/// complain about a missing hook command — it simply records nothing, and
|
||||||
|
/// a mission ran with the tap installed, pointed at nothing, and silent.
|
||||||
|
///
|
||||||
|
/// Asserting that the script "mentions tap.sh" did not catch it. The paths
|
||||||
|
/// have to be compared.
|
||||||
|
#[test]
|
||||||
|
fn every_hook_command_is_a_file_the_installer_writes() {
|
||||||
|
let settings = crate::vm_tool_tap::guest_settings(None, Some(TAP_DIR), Some(HOOK_DIR));
|
||||||
|
let script = build_install_script();
|
||||||
|
|
||||||
|
let hooks = settings["hooks"].as_object().expect("hooks");
|
||||||
|
assert!(!hooks.is_empty(), "no hooks at all");
|
||||||
|
for (event, entries) in hooks {
|
||||||
|
let cmd = entries[0]["hooks"][0]["command"]
|
||||||
|
.as_str()
|
||||||
|
.unwrap_or_else(|| panic!("{event} has no command"));
|
||||||
|
assert!(
|
||||||
|
script.contains(&format!("cat > {cmd} <<")),
|
||||||
|
"{event} points at {cmd}, which the installer never writes — \
|
||||||
|
the hook is registered and inert"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
script.contains(&format!("chmod +x {cmd}")),
|
||||||
|
"{event} points at {cmd}, which is never made executable"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_script_writes_both_hooks_and_the_settings_document() {
|
||||||
|
let s = build_install_script();
|
||||||
|
assert!(s.contains("tool-gate.sh"), "the pre-execution gate is missing");
|
||||||
|
assert!(s.contains("tap.sh"), "the tool tap is missing");
|
||||||
|
assert!(s.contains(SETTINGS_PATH), "the settings document is missing");
|
||||||
|
// Both hooks in ONE document — the whole reason this is composed here.
|
||||||
|
assert!(s.contains("PreToolUse"));
|
||||||
|
assert!(s.contains("PostToolUse"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Nothing may be written into the mission checkout.
|
||||||
|
///
|
||||||
|
/// A file left under `/mission/repo` shows up in the diff the mission
|
||||||
|
/// delivers, so hook plumbing would arrive as part of the agent's work.
|
||||||
|
#[test]
|
||||||
|
fn nothing_is_written_into_the_checkout() {
|
||||||
|
assert!(HOOK_DIR.starts_with("/root/"));
|
||||||
|
assert!(SETTINGS_PATH.starts_with("/root/"));
|
||||||
|
assert!(TAP_DIR.starts_with("/root/"));
|
||||||
|
assert!(!build_install_script().contains("/mission/repo"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The two halves must stay together.
|
||||||
|
///
|
||||||
|
/// Writing the hooks without pointing the provider at them leaves a gate
|
||||||
|
/// that is installed and inert — indistinguishable from a gate that found
|
||||||
|
/// nothing, which is this codebase's signature failure. Pointing the
|
||||||
|
/// provider at a document nobody wrote makes claude fail to start.
|
||||||
|
#[test]
|
||||||
|
fn the_installer_and_the_provider_prop_agree() {
|
||||||
|
let orchestrator = include_str!("mission_orchestrator.rs");
|
||||||
|
assert!(
|
||||||
|
orchestrator.contains("set_claude_cli_settings")
|
||||||
|
&& orchestrator.contains("container_tool_hooks::SETTINGS_PATH"),
|
||||||
|
"the hooks are installed but nothing points claude at them"
|
||||||
|
);
|
||||||
|
let runtime = include_str!("mission_runtime.rs");
|
||||||
|
assert!(
|
||||||
|
runtime.contains("container_tool_hooks::install"),
|
||||||
|
"the provider is pointed at a settings document nobody writes"
|
||||||
|
);
|
||||||
|
// Both container paths — created AND reused. A hook that exists only
|
||||||
|
// on first creation disappears after a server redeploy.
|
||||||
|
assert_eq!(
|
||||||
|
runtime.matches("container_tool_hooks::install").count(),
|
||||||
|
2,
|
||||||
|
"install must run on the reuse path too"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The drain must clear what it read.
|
||||||
|
///
|
||||||
|
/// Truncation IS the idempotency here — there is no cursor column and no
|
||||||
|
/// marker row. A drain that reads without clearing would re-record every
|
||||||
|
/// tool call on every tick, and a phase's early files would end up weighted
|
||||||
|
/// by how long the sweep ran.
|
||||||
|
#[test]
|
||||||
|
fn the_drain_reads_then_clears() {
|
||||||
|
let file = tap_file();
|
||||||
|
assert!(file.starts_with(TAP_DIR), "the tap must live under {TAP_DIR}");
|
||||||
|
// The script is built inline in `drain`; assert on the shape it must
|
||||||
|
// have, since getting this wrong duplicates every event silently.
|
||||||
|
let script = format!("cat {file} 2>/dev/null || true; : > {file} 2>/dev/null || true");
|
||||||
|
assert!(script.contains(&format!("cat {file}")), "must read");
|
||||||
|
assert!(script.contains(&format!(": > {file}")), "must clear");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The sweep has to exist, or the hooks write a file nobody reads.
|
||||||
|
#[test]
|
||||||
|
fn something_actually_collects_the_tap() {
|
||||||
|
let runner = include_str!("phase_runner.rs");
|
||||||
|
assert!(
|
||||||
|
runner.contains("container_tool_hooks::drain"),
|
||||||
|
"the tap is written and never collected — the same shape as a gate \
|
||||||
|
that is installed and inert"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
runner.contains("drain_finished_container_phases(pool).await?"),
|
||||||
|
"the drain exists but the tick does not call it"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The drain must use the connector that honours DOCKER_HOST.
|
||||||
|
///
|
||||||
|
/// The server reaches Docker through a socket proxy, so
|
||||||
|
/// `connect_with_local_defaults` fails there — and it failed SILENTLY,
|
||||||
|
/// which meant the sweep did nothing while the tap filled up and every
|
||||||
|
/// other link in the chain looked correct. Cost a full diagnostic cycle.
|
||||||
|
#[test]
|
||||||
|
fn the_sweep_connects_the_way_the_rest_of_the_server_does() {
|
||||||
|
let runner = include_str!("phase_runner.rs");
|
||||||
|
let body = runner
|
||||||
|
.split("async fn drain_finished_container_phases(")
|
||||||
|
.nth(1)
|
||||||
|
.and_then(|s| s.split("\nasync fn ").next())
|
||||||
|
.expect("sweep body");
|
||||||
|
assert!(
|
||||||
|
body.contains("container_exec::connect()"),
|
||||||
|
"the sweep must use the DOCKER_HOST-aware connector"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
// The CALL, not the word: the comment above it names the
|
||||||
|
// connector it is warning against.
|
||||||
|
!body.contains("connect_with_local_defaults()"),
|
||||||
|
"the local-socket connector fails behind the socket proxy"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The generated installer must be valid shell — a here-doc or quoting slip
|
||||||
|
/// makes it fail in the container, where the only symptom is a mission that
|
||||||
|
/// silently runs unhooked.
|
||||||
|
#[test]
|
||||||
|
fn the_install_script_is_valid_shell() {
|
||||||
|
if std::process::Command::new("bash").arg("-c").arg("true").status().is_err() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let tmp = std::env::temp_dir().join(format!("cm-install-{}.sh", std::process::id()));
|
||||||
|
std::fs::write(&tmp, build_install_script()).unwrap();
|
||||||
|
let out = std::process::Command::new("bash")
|
||||||
|
.arg("-n")
|
||||||
|
.arg(&tmp)
|
||||||
|
.output()
|
||||||
|
.expect("bash -n");
|
||||||
|
let _ = std::fs::remove_file(&tmp);
|
||||||
|
assert!(
|
||||||
|
out.status.success(),
|
||||||
|
"installer will not parse: {}",
|
||||||
|
String::from_utf8_lossy(&out.stderr)
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,232 @@
|
|||||||
|
//! The harvest half of a Continuous Research mission.
|
||||||
|
//!
|
||||||
|
//! Finding papers is NOT agent work. `library::run_to_vault` already does arXiv
|
||||||
|
//! search → seen-set check → PDF fetch → blob shelf → vault note, deterministically
|
||||||
|
//! and in seconds, and it takes a `mission_id` so the run is attributed. Asking an
|
||||||
|
//! agent to redo it would be slower, non-repeatable, and would abandon the
|
||||||
|
//! `corpus_items` seen-set — which is the entire reason a recurring mission knows
|
||||||
|
//! what it already covered. `corpus.rs` puts it plainly: "A recurring mission's
|
||||||
|
//! hard problem is not running the agent — that is 23 seconds — it is knowing
|
||||||
|
//! what it already did last time."
|
||||||
|
//!
|
||||||
|
//! So the harvest runs here, at launch, and the agents start from its output.
|
||||||
|
//!
|
||||||
|
//! The manifest path (`ContinuousResearch/<date>/harvest.jsonl`) is not invented:
|
||||||
|
//! `templates/teams/continuous_research.toml` has told the `signal_harvester`
|
||||||
|
//! role to write exactly that file since the template was authored. This makes
|
||||||
|
//! the code produce what the prompt already promised, rather than leaving a role
|
||||||
|
//! to fabricate it.
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use serde_json::json;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// Template kind that triggers a harvest at launch.
|
||||||
|
pub const TEMPLATE_KIND: &str = "continuous_research";
|
||||||
|
|
||||||
|
/// Today's manifest, relative to the vault root.
|
||||||
|
pub fn manifest_path(date: &str) -> String {
|
||||||
|
format!("ContinuousResearch/{date}/harvest.jsonl")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// UTC date stamp, the same key the vault folders use.
|
||||||
|
pub fn today() -> String {
|
||||||
|
let now = time::OffsetDateTime::now_utc();
|
||||||
|
format!(
|
||||||
|
"{:04}-{:02}-{:02}",
|
||||||
|
now.year(),
|
||||||
|
now.month() as u8,
|
||||||
|
now.day()
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The arXiv queries this mission tracks.
|
||||||
|
///
|
||||||
|
/// `config.topics` on the mission when the operator set them, otherwise the
|
||||||
|
/// project-wide defaults. Read from config rather than a new column because the
|
||||||
|
/// wizard already round-trips `config` untouched, so a topic list needs no
|
||||||
|
/// schema change and no UI work to reach here.
|
||||||
|
pub fn topics_for(config: &serde_json::Value) -> Vec<String> {
|
||||||
|
config
|
||||||
|
.get("topics")
|
||||||
|
.and_then(|v| v.as_array())
|
||||||
|
.map(|a| {
|
||||||
|
a.iter()
|
||||||
|
.filter_map(|t| t.as_str())
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|t| !t.is_empty())
|
||||||
|
.map(str::to_string)
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
})
|
||||||
|
.filter(|t: &Vec<String>| !t.is_empty())
|
||||||
|
.unwrap_or_else(crate::library::default_topics)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run the harvest for a mission and leave a manifest the agents can read.
|
||||||
|
///
|
||||||
|
/// Non-fatal by contract: a launch whose harvest fails still starts its phases,
|
||||||
|
/// because a quiet day and a broken day must be distinguishable and the phase
|
||||||
|
/// itself is what reports which happened. What is NOT acceptable is failing
|
||||||
|
/// silently, so every outcome is logged with its counts.
|
||||||
|
pub async fn harvest_for_mission(
|
||||||
|
pool: &sqlx::PgPool,
|
||||||
|
blobs: &Arc<dyn cm_files::BlobStore>,
|
||||||
|
workspace_id: Uuid,
|
||||||
|
mission_id: Uuid,
|
||||||
|
topics: &[String],
|
||||||
|
per_topic: usize,
|
||||||
|
) -> Result<Vec<crate::papers::Paper>, String> {
|
||||||
|
let work_root = std::env::temp_dir().join("clawmates-library");
|
||||||
|
let run = crate::library::run_to_vault(
|
||||||
|
pool,
|
||||||
|
blobs,
|
||||||
|
workspace_id,
|
||||||
|
crate::routes::library::DEFAULT_CORPUS,
|
||||||
|
crate::routes::library::DEFAULT_VAULT_URL,
|
||||||
|
&work_root,
|
||||||
|
topics,
|
||||||
|
per_topic,
|
||||||
|
Some(mission_id),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let shelved = run.harvest.shelved.len();
|
||||||
|
// A quiet day is not a failure. `Harvest::healthy()` (nothing errored) is a
|
||||||
|
// different question from `added_anything()` (something new arrived), and
|
||||||
|
// collapsing them is the defect class this codebase keeps paying for.
|
||||||
|
eprintln!(
|
||||||
|
"continuous_research: mission {mission_id} harvested {} candidate(s), {} already had, \
|
||||||
|
{} shelved, {} failed",
|
||||||
|
run.harvest.candidates,
|
||||||
|
run.harvest.already_had,
|
||||||
|
shelved,
|
||||||
|
run.harvest.failed.len()
|
||||||
|
);
|
||||||
|
for (source_id, why) in &run.harvest.failed {
|
||||||
|
eprintln!("continuous_research: {source_id} not shelved: {why}");
|
||||||
|
}
|
||||||
|
Ok(run.harvest.papers)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the run manifest into the MISSION's checkout.
|
||||||
|
///
|
||||||
|
/// Not into the vault. The manifest is per-RUN input for one mission, and the
|
||||||
|
/// vault path is per-DATE and shared, so a second run on the same day rewrites
|
||||||
|
/// a file that already exists — which `auto_merge` correctly refuses, because
|
||||||
|
/// it only merges provably additive diffs:
|
||||||
|
///
|
||||||
|
/// "diff is not additive (1 non-add change(s), first:
|
||||||
|
/// M ContinuousResearch/2026-08-18/harvest.jsonl); left for a human"
|
||||||
|
///
|
||||||
|
/// The branch was then left unmerged, `main` kept the previous run's manifest,
|
||||||
|
/// and the next mission cloned STALE papers while every log line said the
|
||||||
|
/// harvest succeeded. Writing into the checkout keeps the vault additive and
|
||||||
|
/// gives each mission exactly its own papers. The agents commit it alongside
|
||||||
|
/// their analysis through the normal delivery path.
|
||||||
|
pub fn write_manifest(
|
||||||
|
checkout: &std::path::Path,
|
||||||
|
papers: &[crate::papers::Paper],
|
||||||
|
date: &str,
|
||||||
|
) -> Result<std::path::PathBuf, String> {
|
||||||
|
let rel = manifest_path(date);
|
||||||
|
let abs = checkout.join(&rel);
|
||||||
|
if let Some(parent) = abs.parent() {
|
||||||
|
std::fs::create_dir_all(parent).map_err(|e| format!("create {}: {e}", parent.display()))?;
|
||||||
|
}
|
||||||
|
let body = manifest_lines(papers, date);
|
||||||
|
std::fs::write(&abs, format!("{body}\n")).map_err(|e| format!("write {}: {e}", abs.display()))?;
|
||||||
|
Ok(abs)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The manifest lines for a set of freshly shelved papers.
|
||||||
|
///
|
||||||
|
/// Shape matches what `templates/teams/continuous_research.toml` documents:
|
||||||
|
/// `{ source, url, title, snippet, first_seen, topic_tags }`.
|
||||||
|
pub fn manifest_lines(papers: &[crate::papers::Paper], first_seen: &str) -> String {
|
||||||
|
papers
|
||||||
|
.iter()
|
||||||
|
.map(|p| {
|
||||||
|
json!({
|
||||||
|
"source": p.source_id(),
|
||||||
|
"url": format!("https://arxiv.org/abs/{}", p.arxiv_id),
|
||||||
|
"title": p.title,
|
||||||
|
"snippet": p.summary.chars().take(400).collect::<String>(),
|
||||||
|
"first_seen": first_seen,
|
||||||
|
"topic_tags": [],
|
||||||
|
})
|
||||||
|
.to_string()
|
||||||
|
})
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join("\n")
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_manifest_path_matches_what_the_team_template_promises() {
|
||||||
|
// templates/teams/continuous_research.toml tells signal_harvester to
|
||||||
|
// write ContinuousResearch/<date>/harvest.jsonl. If this drifts, the
|
||||||
|
// agents read a file nothing writes and silently review nothing.
|
||||||
|
assert_eq!(
|
||||||
|
manifest_path("2026-08-17"),
|
||||||
|
"ContinuousResearch/2026-08-17/harvest.jsonl"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An operator's topic list must win over the defaults, and a blank or
|
||||||
|
/// missing list must fall back rather than harvesting nothing.
|
||||||
|
#[test]
|
||||||
|
fn topics_come_from_config_and_fall_back_when_absent() {
|
||||||
|
assert_eq!(
|
||||||
|
topics_for(&serde_json::json!({"topics": ["world models", " robots "]})),
|
||||||
|
vec!["world models".to_string(), "robots".to_string()],
|
||||||
|
"operator topics win, and are trimmed"
|
||||||
|
);
|
||||||
|
for empty in [
|
||||||
|
serde_json::json!({}),
|
||||||
|
serde_json::json!({"topics": []}),
|
||||||
|
serde_json::json!({"topics": [" "]}),
|
||||||
|
] {
|
||||||
|
assert_eq!(
|
||||||
|
topics_for(&empty),
|
||||||
|
crate::library::default_topics(),
|
||||||
|
"an absent or blank list must fall back, not harvest nothing: {empty}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_date_stamp_is_zero_padded() {
|
||||||
|
let d = today();
|
||||||
|
assert_eq!(d.len(), 10, "YYYY-MM-DD, got {d:?}");
|
||||||
|
assert_eq!(d.matches('-').count(), 2, "{d:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One JSON object per line, and every key the template's prompt names —
|
||||||
|
/// an agent instructed to read `topic_tags` must not find it absent.
|
||||||
|
#[test]
|
||||||
|
fn manifest_lines_carry_every_documented_key() {
|
||||||
|
let p = crate::papers::Paper {
|
||||||
|
arxiv_id: "2401.12345".into(),
|
||||||
|
title: "A Paper".into(),
|
||||||
|
authors: vec!["A. Author".into()],
|
||||||
|
summary: "x".repeat(900),
|
||||||
|
published: "2026-08-17".into(),
|
||||||
|
pdf_url: "https://arxiv.org/pdf/2401.12345".into(),
|
||||||
|
};
|
||||||
|
let out = manifest_lines(std::slice::from_ref(&p), "2026-08-17");
|
||||||
|
assert_eq!(out.lines().count(), 1);
|
||||||
|
let v: serde_json::Value = serde_json::from_str(&out).expect("each line is JSON");
|
||||||
|
for key in ["source", "url", "title", "snippet", "first_seen", "topic_tags"] {
|
||||||
|
assert!(v.get(key).is_some(), "missing {key} in {v}");
|
||||||
|
}
|
||||||
|
assert_eq!(v["source"], "arxiv:2401.12345");
|
||||||
|
assert!(
|
||||||
|
v["snippet"].as_str().unwrap().chars().count() <= 400,
|
||||||
|
"snippet must be trimmed, not the whole abstract"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -19,6 +19,24 @@ pub enum ApiError {
|
|||||||
Conflict,
|
Conflict,
|
||||||
#[error("{0}")]
|
#[error("{0}")]
|
||||||
Quota(String),
|
Quota(String),
|
||||||
|
/// A 400 whose REASON the caller needs.
|
||||||
|
///
|
||||||
|
/// Same argument as `Unavailable` below, one status code down. The
|
||||||
|
/// proposal decide handlers each computed a precise refusal — "the mission
|
||||||
|
/// is running, not a draft", "no node can boot that backend any more" —
|
||||||
|
/// logged it to stderr, and returned a bare `BadRequest`. The person who
|
||||||
|
/// needed the sentence was the one clicking Approve, and they got
|
||||||
|
/// "bad request". `mission_plan::Refusal` exists and is written as
|
||||||
|
/// human-readable copy; this is how it reaches them.
|
||||||
|
#[error("{0}")]
|
||||||
|
Refused(String),
|
||||||
|
/// A dependency is temporarily refusing work and will accept it later —
|
||||||
|
/// today, the Claude Code subscription's rate limit. Distinct from
|
||||||
|
/// `Internal` because the operator's next action is different: wait and
|
||||||
|
/// press the button again, rather than read a server log. A 500 with
|
||||||
|
/// "internal error" sent them looking for a bug that was not there.
|
||||||
|
#[error("{0}")]
|
||||||
|
Unavailable(String),
|
||||||
#[error("internal error")]
|
#[error("internal error")]
|
||||||
Internal,
|
Internal,
|
||||||
}
|
}
|
||||||
@@ -52,14 +70,43 @@ impl From<cm_auth::AuthError> for ApiError {
|
|||||||
impl IntoResponse for ApiError {
|
impl IntoResponse for ApiError {
|
||||||
fn into_response(self) -> Response {
|
fn into_response(self) -> Response {
|
||||||
let status = match self {
|
let status = match self {
|
||||||
ApiError::BadRequest => StatusCode::BAD_REQUEST,
|
ApiError::BadRequest | ApiError::Refused(_) => StatusCode::BAD_REQUEST,
|
||||||
ApiError::Unauthorized => StatusCode::UNAUTHORIZED,
|
ApiError::Unauthorized => StatusCode::UNAUTHORIZED,
|
||||||
ApiError::Forbidden => StatusCode::FORBIDDEN,
|
ApiError::Forbidden => StatusCode::FORBIDDEN,
|
||||||
ApiError::NotFound => StatusCode::NOT_FOUND,
|
ApiError::NotFound => StatusCode::NOT_FOUND,
|
||||||
ApiError::Conflict => StatusCode::CONFLICT,
|
ApiError::Conflict => StatusCode::CONFLICT,
|
||||||
ApiError::Quota(_) => StatusCode::PAYMENT_REQUIRED,
|
ApiError::Quota(_) => StatusCode::PAYMENT_REQUIRED,
|
||||||
|
ApiError::Unavailable(_) => StatusCode::SERVICE_UNAVAILABLE,
|
||||||
ApiError::Internal => StatusCode::INTERNAL_SERVER_ERROR,
|
ApiError::Internal => StatusCode::INTERNAL_SERVER_ERROR,
|
||||||
};
|
};
|
||||||
(status, Json(json!({ "error": self.to_string() }))).into_response()
|
(status, Json(json!({ "error": self.to_string() }))).into_response()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use axum::body::to_bytes;
|
||||||
|
|
||||||
|
/// A refusal must carry its reason into the response body.
|
||||||
|
///
|
||||||
|
/// The proposal decide handlers each computed a precise sentence and then
|
||||||
|
/// returned a bare `BadRequest`, so the person clicking Approve saw
|
||||||
|
/// "bad request" while the reason went to a server log they cannot read.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn a_refusal_reaches_the_caller_and_a_bare_bad_request_does_not_pretend_to() {
|
||||||
|
let refused = ApiError::Refused("this mission is running, not a draft".into());
|
||||||
|
let response = refused.into_response();
|
||||||
|
assert_eq!(response.status(), StatusCode::BAD_REQUEST);
|
||||||
|
let body = to_bytes(response.into_body(), 64 * 1024).await.unwrap();
|
||||||
|
let text = String::from_utf8_lossy(&body);
|
||||||
|
assert!(
|
||||||
|
text.contains("running, not a draft"),
|
||||||
|
"the reason must be in the body, not only in the server log: {text}"
|
||||||
|
);
|
||||||
|
|
||||||
|
// The bare variant stays as it was — same status, no invented detail.
|
||||||
|
let bare = ApiError::BadRequest.into_response();
|
||||||
|
assert_eq!(bare.status(), StatusCode::BAD_REQUEST);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+885
-32
File diff suppressed because it is too large
Load Diff
@@ -50,7 +50,16 @@ const COMMAND_TIMEOUT: Duration = Duration::from_secs(180);
|
|||||||
/// Cap on what one command may return to the model. Test suites are chatty and
|
/// Cap on what one command may return to the model. Test suites are chatty and
|
||||||
/// the judge pays for every byte; the tail is where failures live, so when
|
/// the judge pays for every byte; the tail is where failures live, so when
|
||||||
/// output overflows we keep both ends and drop the middle.
|
/// output overflows we keep both ends and drop the middle.
|
||||||
const MAX_OUTPUT_BYTES: usize = 12_000;
|
///
|
||||||
|
/// 64 KB, up from 12 KB on 2026-09-18. The smaller cap was sized for test
|
||||||
|
/// output and applied to deliverables: a research REPORT.md of ~18 KB came
|
||||||
|
/// back truncated from `cat`, and the judge — correctly — reassembled it with
|
||||||
|
/// `head -119`, `tail -120`, `sed -n 80,200p` and three greps, five extra
|
||||||
|
/// rounds each resending the whole conversation. 7 of 9 verdicts ran to the
|
||||||
|
/// 12-check cap that way. Since `compact_earlier_results` shrinks a result to
|
||||||
|
/// 800 bytes once its round is over, one 64 KB read costs one round; the
|
||||||
|
/// slicing it replaces cost five.
|
||||||
|
const MAX_OUTPUT_BYTES: usize = 64_000;
|
||||||
|
|
||||||
/// Programs the judge may run. Every one either reports state or runs a
|
/// Programs the judge may run. Every one either reports state or runs a
|
||||||
/// project's own checks — none of them edit the tree.
|
/// project's own checks — none of them edit the tree.
|
||||||
@@ -195,11 +204,39 @@ pub fn clamp_output(s: &str) -> String {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Where a verification copy lives: a sibling of the per-mission directories,
|
||||||
|
/// so the sweeper that deletes `<root>/<mission_id>` never races it and nothing
|
||||||
|
/// under it is ever collected or delivered.
|
||||||
|
fn verify_path(mission_id: Uuid) -> PathBuf {
|
||||||
|
crate::mission_workspace::missions_root()
|
||||||
|
.join("_verify")
|
||||||
|
.join(mission_id.to_string())
|
||||||
|
}
|
||||||
|
|
||||||
/// A checkout the judge may run verification commands against.
|
/// A checkout the judge may run verification commands against.
|
||||||
#[derive(Debug, Clone)]
|
///
|
||||||
|
/// A COPY of the mission's checkout, never the checkout itself. The judge runs
|
||||||
|
/// real commands — `cargo test` is the whole point — and the container it execs
|
||||||
|
/// into runs as ROOT with the missions root bind-mounted, so running them in the
|
||||||
|
/// live tree left `repo/target/` owned by uid 0 in a checkout otherwise owned by
|
||||||
|
/// the server. That breaks the single-writer invariant copy mode exists to
|
||||||
|
/// guarantee, and the next phase's `cargo` would hit permission-denied on a
|
||||||
|
/// directory it cannot write.
|
||||||
|
///
|
||||||
|
/// It stayed invisible all day because a dead validator credential meant the
|
||||||
|
/// judge never ran a single check; restoring the credential surfaced it on the
|
||||||
|
/// first gated mission, via the harness's uid probe.
|
||||||
|
///
|
||||||
|
/// The deeper rule is the one this codebase already applies to the `verifier`
|
||||||
|
/// subagent, which has no Edit and no Write: **verification must not mutate what
|
||||||
|
/// it verifies.** A judge that can change the tree it is judging can make its own
|
||||||
|
/// verdict true.
|
||||||
|
#[derive(Debug)]
|
||||||
pub struct Sandbox {
|
pub struct Sandbox {
|
||||||
container: String,
|
container: String,
|
||||||
workdir: PathBuf,
|
workdir: PathBuf,
|
||||||
|
/// Whether this sandbox created `workdir` and must remove it.
|
||||||
|
owned: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Sandbox {
|
impl Sandbox {
|
||||||
@@ -208,22 +245,59 @@ impl Sandbox {
|
|||||||
///
|
///
|
||||||
/// Returning `None` rather than an empty sandbox matters: the evaluator
|
/// Returning `None` rather than an empty sandbox matters: the evaluator
|
||||||
/// prompt changes shape depending on whether verification is possible, and
|
/// prompt changes shape depending on whether verification is possible, and
|
||||||
/// a judge must never be told it can check something it cannot.
|
/// a judge must never be told it can check something it cannot. A copy that
|
||||||
|
/// fails to materialise is also `None` for the same reason — an unverifiable
|
||||||
|
/// phase must not be told it can verify.
|
||||||
pub fn for_mission(mission_id: Uuid) -> Option<Sandbox> {
|
pub fn for_mission(mission_id: Uuid) -> Option<Sandbox> {
|
||||||
let workdir = crate::mission_workspace::checkout_path(mission_id);
|
Sandbox::for_checkout(
|
||||||
if !workdir.is_dir() {
|
&crate::mission_workspace::checkout_path(mission_id),
|
||||||
|
&verify_path(mission_id),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The testable half of [`Sandbox::for_mission`]. The paths are parameters
|
||||||
|
/// because `missions_root()` reads process environment, and this workspace
|
||||||
|
/// does not mutate that in tests — the same split as
|
||||||
|
/// `mission_runtime::provider_env_from` and
|
||||||
|
/// `mission_workspace::auth_with_token`.
|
||||||
|
pub fn for_checkout(source: &Path, root: &Path) -> Option<Sandbox> {
|
||||||
|
if !source.is_dir() {
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
|
// `root_copy` owns this pattern for all four callers — the judge, the
|
||||||
|
// benchmark runner, the on_green_tests gate, and this. It packs through
|
||||||
|
// the transport packer (one exclusion list, so a copy carries exactly
|
||||||
|
// what a delivered diff carries) and its `purge` is the only thing that
|
||||||
|
// can remove the root-owned `target/` a run leaves behind.
|
||||||
|
//
|
||||||
|
// A stale copy would otherwise be verified instead of this pass's work —
|
||||||
|
// the "judged a tree nobody wrote" shape the evaluator exists to prevent
|
||||||
|
// — so the caller purges before constructing.
|
||||||
|
// `into_workdir` because the judge has not run yet: letting the handle's
|
||||||
|
// Drop fire on return would delete the tree out from under it. `Sandbox`
|
||||||
|
// owns the lifetime from here, and `Sandbox::purge` clears it.
|
||||||
|
let workdir = crate::root_copy::RootCopy::of(source, root)
|
||||||
|
.ok()?
|
||||||
|
.into_workdir();
|
||||||
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
||||||
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
||||||
Some(Sandbox { container, workdir })
|
Some(Sandbox {
|
||||||
|
container,
|
||||||
|
workdir,
|
||||||
|
owned: true,
|
||||||
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Construct against an explicit path. Test seam.
|
/// Construct against an explicit path. Test seam.
|
||||||
|
///
|
||||||
|
/// Never `owned`: a caller-supplied directory is the caller's, and deleting
|
||||||
|
/// it on drop would make this seam destructive in a way its users could not
|
||||||
|
/// see.
|
||||||
pub fn at(container: impl Into<String>, workdir: impl AsRef<Path>) -> Sandbox {
|
pub fn at(container: impl Into<String>, workdir: impl AsRef<Path>) -> Sandbox {
|
||||||
Sandbox {
|
Sandbox {
|
||||||
container: container.into(),
|
container: container.into(),
|
||||||
workdir: workdir.as_ref().to_path_buf(),
|
workdir: workdir.as_ref().to_path_buf(),
|
||||||
|
owned: false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -329,6 +403,59 @@ fn git_ownership_env(workdir: &str) -> Vec<String> {
|
|||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl Sandbox {
|
||||||
|
/// Remove the copy, from inside the container that wrote it.
|
||||||
|
///
|
||||||
|
/// `Drop` cannot do this. The judge runs `cargo test` in a container as
|
||||||
|
/// ROOT, so the copy's `target/` is root-owned, and the server process is
|
||||||
|
/// uid 65532 — its `remove_dir_all` fails on those files and leaves the
|
||||||
|
/// whole tree behind. Measured: 16 MB across two stranded copies, the oldest
|
||||||
|
/// hours old, while `Drop` logged nothing anyone read.
|
||||||
|
///
|
||||||
|
/// The claim that "the next pass clears anyway" was wrong for the same
|
||||||
|
/// reason: `for_checkout` removes a stale root before copying, with the same
|
||||||
|
/// uid, and fails the same way.
|
||||||
|
///
|
||||||
|
/// Still best-effort — a housekeeping error must not cost a real verdict —
|
||||||
|
/// but now attempted by something that can actually succeed.
|
||||||
|
pub async fn purge(&self) {
|
||||||
|
if !self.owned {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let Some(root) = self.workdir.parent() else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
// The same purge as the other three copy sites, not a fourth copy of
|
||||||
|
// it: an inlined duplicate is how the reap paths drifted apart before.
|
||||||
|
crate::root_copy::purge(&self.container, root).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for Sandbox {
|
||||||
|
/// Fallback only — see [`Sandbox::purge`], which is what actually clears a
|
||||||
|
/// copy the judge has run commands in. This still catches the early paths
|
||||||
|
/// where nothing has run as root yet.
|
||||||
|
fn drop(&mut self) {
|
||||||
|
if !self.owned {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if let Some(root) = self.workdir.parent() {
|
||||||
|
match std::fs::remove_dir_all(root) {
|
||||||
|
Ok(()) => {}
|
||||||
|
// Already gone, because `purge` ran first and worked. That is
|
||||||
|
// the SUCCESS path, and reporting it as a failure is how a
|
||||||
|
// real cleanup error gets read as noise — the exact habit that
|
||||||
|
// let two root-owned copies sit stranded for hours.
|
||||||
|
Err(e) if e.kind() == std::io::ErrorKind::NotFound => {}
|
||||||
|
Err(e) => eprintln!(
|
||||||
|
"evaluator_tools: could not remove the verification copy at {} ({e})",
|
||||||
|
root.display()
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// One verification command and what became of it.
|
/// One verification command and what became of it.
|
||||||
///
|
///
|
||||||
/// This exists because the first version recorded *attempted* commands. The
|
/// This exists because the first version recorded *attempted* commands. The
|
||||||
@@ -427,6 +554,58 @@ mod tests {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// THE regression. The judge runs real commands in a container that runs as
|
||||||
|
/// ROOT with the missions root bind-mounted, so verifying the live checkout
|
||||||
|
/// left `repo/target/` owned by uid 0 in a tree owned by the server — the
|
||||||
|
/// single-writer invariant broken by the thing that was supposed to be
|
||||||
|
/// checking the work. Verifying a COPY makes it unrepresentable.
|
||||||
|
#[test]
|
||||||
|
fn the_judge_verifies_a_copy_and_never_the_mission_tree() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let root = tmp.path().join("missions-root");
|
||||||
|
let mission = Uuid::now_v7();
|
||||||
|
let checkout = root.join(mission.to_string()).join("repo");
|
||||||
|
std::fs::create_dir_all(checkout.join("src")).unwrap();
|
||||||
|
std::fs::write(checkout.join("Cargo.toml"), "[package]\nname='x'\n").unwrap();
|
||||||
|
std::fs::write(checkout.join("src/lib.rs"), "pub fn a() {}").unwrap();
|
||||||
|
// Build output the transport already excludes; the copy must not carry
|
||||||
|
// it either, or the judge measures a stale artifact.
|
||||||
|
std::fs::create_dir_all(checkout.join("target/debug")).unwrap();
|
||||||
|
std::fs::write(checkout.join("target/debug/junk"), "x").unwrap();
|
||||||
|
|
||||||
|
let sandbox = Sandbox::for_checkout(&checkout, &root.join("_verify").join(mission.to_string()))
|
||||||
|
.expect("a checkout on disk yields a sandbox");
|
||||||
|
|
||||||
|
assert_ne!(
|
||||||
|
sandbox.workdir(),
|
||||||
|
checkout,
|
||||||
|
"the judge must not be pointed at the mission's own checkout"
|
||||||
|
);
|
||||||
|
assert!(sandbox.workdir().join("src/lib.rs").is_file(), "the copy has the source");
|
||||||
|
assert!(
|
||||||
|
!sandbox.workdir().join("target").exists(),
|
||||||
|
"the copy must not carry build output: {}",
|
||||||
|
sandbox.workdir().display()
|
||||||
|
);
|
||||||
|
|
||||||
|
// And dropping it takes the copy with it, leaving the mission untouched.
|
||||||
|
let copy_root = sandbox.workdir().parent().unwrap().to_path_buf();
|
||||||
|
drop(sandbox);
|
||||||
|
assert!(!copy_root.exists(), "the copy outlived its sandbox");
|
||||||
|
assert!(checkout.join("src/lib.rs").is_file(), "the mission tree is intact");
|
||||||
|
assert!(checkout.join("target/debug/junk").is_file());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The test seam must not delete a directory it was handed. A destructive
|
||||||
|
/// constructor that looks like a plain one is how a test wipes a real tree.
|
||||||
|
#[test]
|
||||||
|
fn an_explicit_workdir_is_never_deleted() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
std::fs::write(tmp.path().join("keep.txt"), "x").unwrap();
|
||||||
|
drop(Sandbox::at("c", tmp.path()));
|
||||||
|
assert!(tmp.path().join("keep.txt").is_file());
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn refuses_programs_off_the_list() {
|
fn refuses_programs_off_the_list() {
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
|
|||||||
@@ -364,6 +364,15 @@ enum Uplink {
|
|||||||
Result { id: u64, ok: bool, output: String },
|
Result { id: u64, ok: bool, output: String },
|
||||||
#[serde(rename = "pty_out")]
|
#[serde(rename = "pty_out")]
|
||||||
PtyOut { sid: u64, data: String },
|
PtyOut { sid: u64, data: String },
|
||||||
|
/// A chunk of a microVM turn's stdout/stderr, as it happens.
|
||||||
|
///
|
||||||
|
/// Keyed by RUN id rather than a session id: a mission run is the thing a
|
||||||
|
/// browser subscribes to, and unlike a PTY there is no interactive session
|
||||||
|
/// to allocate. `at` is the byte offset AFTER this chunk, so the node can
|
||||||
|
/// resume a dropped tail without replaying — the same contract `fcagent`'s
|
||||||
|
/// `tail` op exposes.
|
||||||
|
#[serde(rename = "vm_out")]
|
||||||
|
VmOut { run_id: String, at: u64, data: String },
|
||||||
#[serde(rename = "pty_exit")]
|
#[serde(rename = "pty_exit")]
|
||||||
PtyExit { sid: u64 },
|
PtyExit { sid: u64 },
|
||||||
#[serde(rename = "webrtc_answer")]
|
#[serde(rename = "webrtc_answer")]
|
||||||
@@ -381,6 +390,11 @@ enum Uplink {
|
|||||||
NodeTools {
|
NodeTools {
|
||||||
tools: std::collections::HashMap<String, String>,
|
tools: std::collections::HashMap<String, String>,
|
||||||
},
|
},
|
||||||
|
/// What the node can HOST, as opposed to what it has installed — the
|
||||||
|
/// inputs to placement predicates. Free-form so a new predicate does not
|
||||||
|
/// need a migration; see `migrations/0065_microvm_placement.sql`.
|
||||||
|
#[serde(rename = "node_capabilities")]
|
||||||
|
NodeCapabilities { capabilities: serde_json::Value },
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Deserialize)]
|
#[derive(Deserialize)]
|
||||||
@@ -482,6 +496,44 @@ pub async fn run_channel(pool: PgPool, hub: Arc<NodeHub>, node_id: NodeId, socke
|
|||||||
let _ = s.send(ExecOutput { ok, output });
|
let _ = s.send(ExecOutput { ok, output });
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// A chunk of a microVM turn's output, live.
|
||||||
|
//
|
||||||
|
// Appended to the run's checkpoint rather than only fanned
|
||||||
|
// out: `PtyOut` above is deliberately ephemeral because a
|
||||||
|
// terminal has no history worth keeping, but a mission's log
|
||||||
|
// is the record of what the agent did — the Output tab has
|
||||||
|
// to still show it an hour later. Live and durable are
|
||||||
|
// different requirements and this needs both.
|
||||||
|
//
|
||||||
|
// `jsonb ||` merges into whatever else the checkpoint holds
|
||||||
|
// (`records`, written by the turn itself), so the two writers
|
||||||
|
// do not clobber each other.
|
||||||
|
Ok(Uplink::VmOut { run_id, at, data }) => {
|
||||||
|
if let (Ok(rid), Ok(bytes)) =
|
||||||
|
(uuid::Uuid::parse_str(&run_id), B64.decode(&data))
|
||||||
|
{
|
||||||
|
let text = String::from_utf8_lossy(&bytes).to_string();
|
||||||
|
if let Err(e) = sqlx::query(
|
||||||
|
"UPDATE topology_runs
|
||||||
|
SET checkpoint = COALESCE(checkpoint, '{}'::jsonb)
|
||||||
|
|| jsonb_build_object(
|
||||||
|
'log',
|
||||||
|
COALESCE(checkpoint->>'log', '') || $2::text,
|
||||||
|
'log_at', $3::bigint
|
||||||
|
),
|
||||||
|
updated_at = now()
|
||||||
|
WHERE id = $1",
|
||||||
|
)
|
||||||
|
.bind(rid)
|
||||||
|
.bind(&text)
|
||||||
|
.bind(at as i64)
|
||||||
|
.execute(&pool)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!("fleet: appending vm_out for run {rid}: {e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Ok(Uplink::PtyOut { sid, data }) => {
|
Ok(Uplink::PtyOut { sid, data }) => {
|
||||||
if let Ok(bytes) = B64.decode(&data) {
|
if let Ok(bytes) = B64.decode(&data) {
|
||||||
let sink = conn.pty_sinks.lock().await.get(&sid).cloned();
|
let sink = conn.pty_sinks.lock().await.get(&sid).cloned();
|
||||||
@@ -539,7 +591,32 @@ pub async fn run_channel(pool: PgPool, hub: Arc<NodeHub>, node_id: NodeId, socke
|
|||||||
let pairs: Vec<(String, String)> = tools.into_iter().collect();
|
let pairs: Vec<(String, String)> = tools.into_iter().collect();
|
||||||
let _ = cm_db::repo::node_tools::upsert(&pool, node_id, &pairs).await;
|
let _ = cm_db::repo::node_tools::upsert(&pool, node_id, &pairs).await;
|
||||||
}
|
}
|
||||||
Err(_) => {}
|
Ok(Uplink::NodeCapabilities { capabilities }) => {
|
||||||
|
if let Err(e) = nodes::set_capabilities(&pool, node_id, &capabilities).await
|
||||||
|
{
|
||||||
|
// Loud: a node whose capabilities never land looks
|
||||||
|
// exactly like a node that has none, and will be
|
||||||
|
// passed over for every microVM mission forever
|
||||||
|
// while appearing perfectly healthy.
|
||||||
|
eprintln!(
|
||||||
|
"fleet: could not record capabilities for node {node_id} ({e}) — \
|
||||||
|
it will not be selected for microvm placement"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// An unparseable frame used to vanish here. That is the
|
||||||
|
// worst possible handling: a node op whose reply does not
|
||||||
|
// match `Uplink` never resolves its pending request, so the
|
||||||
|
// caller times out after 20s with nothing anywhere saying
|
||||||
|
// why. Caught exactly that way while wiring the vm_* ops —
|
||||||
|
// `output` was an object where the wire declares a String.
|
||||||
|
Err(e) => {
|
||||||
|
let head: String = t.as_str().chars().take(160).collect();
|
||||||
|
eprintln!(
|
||||||
|
"fleet: node {node_id} sent a frame we could not parse ({e}); \
|
||||||
|
any request it was answering will time out. Frame: {head}"
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,160 @@
|
|||||||
|
//! Is the mission gateway configured, and is anything listening?
|
||||||
|
//!
|
||||||
|
//! The third sibling of [`crate::runtime_preflight`] and
|
||||||
|
//! [`crate::validator_preflight`], for the same class of failure: the
|
||||||
|
//! configuration is absent or wrong, and nothing says so until a mission pays
|
||||||
|
//! for it.
|
||||||
|
//!
|
||||||
|
//! `ZEROCLAW_GATEWAY_URL` and `ZEROCLAW_TOKEN` have no defaults and are read at
|
||||||
|
//! FIRST USE, inside `ZeroClawDriveExecutor::from_env`. So a deployment missing
|
||||||
|
//! them boots clean, serves every page, lists every mission — and fails the
|
||||||
|
//! first time someone presses run, with an error that surfaces on a phase
|
||||||
|
//! rather than at startup. The information exists the whole time; nobody is
|
||||||
|
//! told until it is expensive.
|
||||||
|
//!
|
||||||
|
//! A report, not a gate, matching its siblings. A server with no gateway should
|
||||||
|
//! still boot: the frontend, the catalogue and every read path work without it,
|
||||||
|
//! and refusing to start would turn a degraded deployment into a dead one.
|
||||||
|
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
const PROBE_TIMEOUT: Duration = Duration::from_secs(5);
|
||||||
|
|
||||||
|
/// What the preflight found.
|
||||||
|
#[derive(Debug, PartialEq, Eq)]
|
||||||
|
pub enum Verdict {
|
||||||
|
/// No gateway configured. Missions on the container tier cannot run.
|
||||||
|
NotConfigured { missing: Vec<String> },
|
||||||
|
/// Configured but nothing answered at that address.
|
||||||
|
Unreachable { url: String, error: String },
|
||||||
|
/// Configured and something answered.
|
||||||
|
Reachable { url: String },
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Verdict {
|
||||||
|
/// The line to print at boot.
|
||||||
|
///
|
||||||
|
/// Each names the CONSEQUENCE, not just the state. "ZEROCLAW_TOKEN not set"
|
||||||
|
/// tells an operator what is missing; it does not tell them that every
|
||||||
|
/// container-tier mission they launch will fail on its first phase.
|
||||||
|
pub fn message(&self) -> String {
|
||||||
|
match self {
|
||||||
|
Verdict::NotConfigured { missing } => format!(
|
||||||
|
"gateway_preflight: NOT CONFIGURED ({}) — container-tier missions \
|
||||||
|
cannot run. They will launch, provision a runtime, and fail on \
|
||||||
|
the first turn; the server is otherwise healthy",
|
||||||
|
missing.join(", ")
|
||||||
|
),
|
||||||
|
Verdict::Unreachable { url, error } => format!(
|
||||||
|
"gateway_preflight: {url} is configured but did not answer ({error}) \
|
||||||
|
— container-tier missions will fail on their first turn. The \
|
||||||
|
config is right and the machine is not"
|
||||||
|
),
|
||||||
|
Verdict::Reachable { url } => {
|
||||||
|
format!("gateway_preflight: {url} answered")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Which required variables are absent.
|
||||||
|
///
|
||||||
|
/// Split from the network probe so the rule is testable without a gateway:
|
||||||
|
/// this is the half that is pure, and it is the half that is wrong most often.
|
||||||
|
pub fn missing_config(url: Option<&str>, token: Option<&str>, pairing: Option<&str>) -> Vec<String> {
|
||||||
|
let mut missing = Vec::new();
|
||||||
|
if url.map(str::trim).unwrap_or("").is_empty() {
|
||||||
|
missing.push("ZEROCLAW_GATEWAY_URL".to_string());
|
||||||
|
}
|
||||||
|
// Either credential works: a durable token, or a one-time pairing code the
|
||||||
|
// executor exchanges on first use.
|
||||||
|
let has_token = !token.map(str::trim).unwrap_or("").is_empty();
|
||||||
|
let has_pairing = !pairing.map(str::trim).unwrap_or("").is_empty();
|
||||||
|
if !has_token && !has_pairing {
|
||||||
|
missing.push("ZEROCLAW_TOKEN or ZEROCLAW_PAIRING_CODE".to_string());
|
||||||
|
}
|
||||||
|
missing
|
||||||
|
}
|
||||||
|
|
||||||
|
fn env_opt(key: &str) -> Option<String> {
|
||||||
|
std::env::var(key).ok().filter(|v| !v.trim().is_empty())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Probe the configured gateway.
|
||||||
|
pub async fn check() -> Verdict {
|
||||||
|
let url = env_opt("ZEROCLAW_GATEWAY_URL");
|
||||||
|
let missing = missing_config(
|
||||||
|
url.as_deref(),
|
||||||
|
env_opt("ZEROCLAW_TOKEN").as_deref(),
|
||||||
|
env_opt("ZEROCLAW_PAIRING_CODE").as_deref(),
|
||||||
|
);
|
||||||
|
if !missing.is_empty() {
|
||||||
|
return Verdict::NotConfigured { missing };
|
||||||
|
}
|
||||||
|
let url = url.expect("checked above");
|
||||||
|
|
||||||
|
// Any HTTP answer proves something is listening and routable, which is the
|
||||||
|
// question this preflight exists to answer. Authenticating here would need
|
||||||
|
// a pairing exchange that BURNS a one-time code — a preflight that costs
|
||||||
|
// the deployment its credential is worse than no preflight.
|
||||||
|
let client = match reqwest::Client::builder().timeout(PROBE_TIMEOUT).build() {
|
||||||
|
Ok(c) => c,
|
||||||
|
Err(e) => {
|
||||||
|
return Verdict::Unreachable {
|
||||||
|
url,
|
||||||
|
error: e.to_string(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
};
|
||||||
|
match client.get(&url).send().await {
|
||||||
|
Ok(_) => Verdict::Reachable { url },
|
||||||
|
Err(e) => Verdict::Unreachable {
|
||||||
|
url,
|
||||||
|
error: e.to_string(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run the probe and print the verdict. Never panics, never blocks boot.
|
||||||
|
pub fn report_at_boot() {
|
||||||
|
tokio::spawn(async {
|
||||||
|
eprintln!("{}", check().await.message());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_fully_configured_deployment_is_missing_nothing() {
|
||||||
|
assert!(missing_config(Some("http://gw:42617"), Some("tok"), None).is_empty());
|
||||||
|
// A pairing code alone is enough — the executor exchanges it on first use.
|
||||||
|
assert!(missing_config(Some("http://gw:42617"), None, Some("123456")).is_empty());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_empty_string_counts_as_absent() {
|
||||||
|
// The failure this whole module exists for: `unwrap_or_default` and an
|
||||||
|
// empty env var turn "unconfigured" into "configured with nothing",
|
||||||
|
// which fails later as a 401 rather than now as a missing setting.
|
||||||
|
let missing = missing_config(Some(" "), Some(""), Some(" "));
|
||||||
|
assert_eq!(missing.len(), 2, "both must be reported: {missing:?}");
|
||||||
|
assert!(missing[0].contains("GATEWAY_URL"));
|
||||||
|
assert!(missing[1].contains("ZEROCLAW_TOKEN"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_message_names_the_consequence_not_just_the_state() {
|
||||||
|
let v = Verdict::NotConfigured {
|
||||||
|
missing: vec!["ZEROCLAW_GATEWAY_URL".into()],
|
||||||
|
};
|
||||||
|
let m = v.message();
|
||||||
|
assert!(m.contains("ZEROCLAW_GATEWAY_URL"));
|
||||||
|
assert!(
|
||||||
|
m.contains("cannot run"),
|
||||||
|
"an operator needs to know what stops working, not only what is \
|
||||||
|
unset: {m}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -44,6 +44,14 @@ pub struct Harvest {
|
|||||||
pub failed: Vec<(String, String)>,
|
pub failed: Vec<(String, String)>,
|
||||||
/// Vault-relative paths of the notes written.
|
/// Vault-relative paths of the notes written.
|
||||||
pub notes_written: Vec<String>,
|
pub notes_written: Vec<String>,
|
||||||
|
/// The papers actually shelved this run, in shelve order.
|
||||||
|
///
|
||||||
|
/// `shelved` carries only source ids, which is all the seen-set needs. The
|
||||||
|
/// run manifest a Continuous Research mission hands its agents needs the
|
||||||
|
/// title and abstract too, and re-reading them back out of the notes we
|
||||||
|
/// just wrote would be a parse of our own output — one more place for the
|
||||||
|
/// two to drift.
|
||||||
|
pub papers: Vec<crate::papers::Paper>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl Harvest {
|
impl Harvest {
|
||||||
@@ -167,6 +175,7 @@ pub async fn shelve(
|
|||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
out.notes_written.push(paper.note_path());
|
out.notes_written.push(paper.note_path());
|
||||||
|
out.papers.push(paper.clone());
|
||||||
out.shelved.push(sid);
|
out.shelved.push(sid);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+331
-46
@@ -13,16 +13,28 @@
|
|||||||
//! reviewer picked. Rejected proposals move to status='rejected';
|
//! reviewer picked. Rejected proposals move to status='rejected';
|
||||||
//! partial approvals move to status='partial'.
|
//! partial approvals move to status='partial'.
|
||||||
//!
|
//!
|
||||||
//! Uses Gemini 2.5 Flash as the default proposer model — cheap,
|
//! The proposer model resolves through the provider REGISTRY
|
||||||
//! JSON-mode-native, plenty of room for structured output. Configurable
|
//! (`Runtime::resolve_provider`), the same path the evaluator uses, and defaults
|
||||||
//! via CLAWMATES_LEVEL_UP_MODEL.
|
//! to `glm:glm-4.7`. Configurable via `CLAWMATES_LEVEL_UP_MODEL` as a registry
|
||||||
|
//! spec (`glm:glm-4.7`, `kimi:k2`, `claude-sonnet-5`, …).
|
||||||
|
//!
|
||||||
|
//! It used to call Gemini directly over bespoke HTTP with `GEMINI_API_KEY`. Two
|
||||||
|
//! problems with that, one fatal: it was the only thing standing between this
|
||||||
|
//! feature and a dead prepayment balance, and it duplicated a provider client
|
||||||
|
//! the codebase already has. Going through the registry means every provider the
|
||||||
|
//! platform can already reach works here, and no single vendor's billing can
|
||||||
|
//! take the feature down.
|
||||||
|
|
||||||
use serde_json::{json, Value};
|
use serde_json::{json, Value};
|
||||||
use sqlx::PgPool;
|
use sqlx::PgPool;
|
||||||
use sqlx::Row;
|
use sqlx::Row;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
const DEFAULT_MODEL: &str = "gemini-2.5-flash";
|
/// Registry spec, not a bare model name — the registry needs the provider.
|
||||||
|
///
|
||||||
|
/// GLM: cheap, reliable at structured output, and already the validator this
|
||||||
|
/// project measured and chose (see `scripts/judge-eval.sh`).
|
||||||
|
const DEFAULT_MODEL: &str = "glm:glm-4.7";
|
||||||
|
|
||||||
fn model_name() -> String {
|
fn model_name() -> String {
|
||||||
std::env::var("CLAWMATES_LEVEL_UP_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_string())
|
std::env::var("CLAWMATES_LEVEL_UP_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_string())
|
||||||
@@ -31,6 +43,7 @@ fn model_name() -> String {
|
|||||||
/// Analyze an agent + insert a pending proposal. Returns the proposal id.
|
/// Analyze an agent + insert a pending proposal. Returns the proposal id.
|
||||||
pub async fn propose_agent(
|
pub async fn propose_agent(
|
||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
workspace_id: cm_domain::WorkspaceId,
|
workspace_id: cm_domain::WorkspaceId,
|
||||||
created_by: cm_domain::UserId,
|
created_by: cm_domain::UserId,
|
||||||
agent_id: Uuid,
|
agent_id: Uuid,
|
||||||
@@ -50,6 +63,7 @@ pub async fn propose_agent(
|
|||||||
.flatten();
|
.flatten();
|
||||||
|
|
||||||
let payload = call_llm_for_agent(
|
let payload = call_llm_for_agent(
|
||||||
|
runtime,
|
||||||
&agent.name,
|
&agent.name,
|
||||||
&agent.job_title,
|
&agent.job_title,
|
||||||
&agent.system_prompt,
|
&agent.system_prompt,
|
||||||
@@ -79,6 +93,7 @@ pub async fn propose_agent(
|
|||||||
/// Analyze a team + insert a pending proposal. Returns the proposal id.
|
/// Analyze a team + insert a pending proposal. Returns the proposal id.
|
||||||
pub async fn propose_team(
|
pub async fn propose_team(
|
||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
workspace_id: cm_domain::WorkspaceId,
|
workspace_id: cm_domain::WorkspaceId,
|
||||||
created_by: cm_domain::UserId,
|
created_by: cm_domain::UserId,
|
||||||
team_id: Uuid,
|
team_id: Uuid,
|
||||||
@@ -115,7 +130,7 @@ pub async fn propose_team(
|
|||||||
}));
|
}));
|
||||||
}
|
}
|
||||||
|
|
||||||
let payload = call_llm_for_team(&member_summaries).await?;
|
let payload = call_llm_for_team(runtime, &member_summaries).await?;
|
||||||
|
|
||||||
let model = model_name();
|
let model = model_name();
|
||||||
let id = cm_db::repo::level_up::insert(
|
let id = cm_db::repo::level_up::insert(
|
||||||
@@ -210,6 +225,128 @@ pub async fn apply(
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Is autonomous skill authoring on?
|
||||||
|
///
|
||||||
|
/// Default OFF since 2026-09-20, by operator decision. It shipped default ON,
|
||||||
|
/// and in the months since no agent-authored skill was ever delivered to a
|
||||||
|
/// mission or scored by the Skill-Use scorer — prod's `level_up_proposals`
|
||||||
|
/// held zero rows on the day of the flip. An auto-apply loop whose output has
|
||||||
|
/// never been measured is a supply chain of our own making (the shape Cisco
|
||||||
|
/// found in OpenClaw's third-party skills), so it waits for a human until
|
||||||
|
/// `promoted_from_brain` skills go through the `files` delivery arm and get
|
||||||
|
/// a Trigger/Compliance score like the hand-authored ones. Stated at boot
|
||||||
|
/// either way: a safety gate that changes state silently is how nobody
|
||||||
|
/// notices it changed.
|
||||||
|
pub fn self_authoring_enabled() -> bool {
|
||||||
|
matches!(
|
||||||
|
std::env::var("CLAWMATES_SKILL_SELF_AUTHORING")
|
||||||
|
.unwrap_or_default()
|
||||||
|
.trim()
|
||||||
|
.to_ascii_lowercase()
|
||||||
|
.as_str(),
|
||||||
|
"1" | "on" | "true"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod self_authoring_flag_tests {
|
||||||
|
/// Serialised through one env var; each case restores the prior state.
|
||||||
|
fn with(value: Option<&str>, f: impl FnOnce()) {
|
||||||
|
let key = "CLAWMATES_SKILL_SELF_AUTHORING";
|
||||||
|
let prior = std::env::var(key).ok();
|
||||||
|
match value {
|
||||||
|
Some(v) => std::env::set_var(key, v),
|
||||||
|
None => std::env::remove_var(key),
|
||||||
|
}
|
||||||
|
f();
|
||||||
|
match prior {
|
||||||
|
Some(v) => std::env::set_var(key, v),
|
||||||
|
None => std::env::remove_var(key),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Off unless switched on. The previous default was the reverse.
|
||||||
|
#[test]
|
||||||
|
fn off_by_default_on_by_explicit_opt_in() {
|
||||||
|
with(None, || assert!(!super::self_authoring_enabled()));
|
||||||
|
with(Some(""), || assert!(!super::self_authoring_enabled()));
|
||||||
|
with(Some("0"), || assert!(!super::self_authoring_enabled()));
|
||||||
|
with(Some("yes"), || assert!(!super::self_authoring_enabled()));
|
||||||
|
with(Some("1"), || assert!(super::self_authoring_enabled()));
|
||||||
|
with(Some("on"), || assert!(super::self_authoring_enabled()));
|
||||||
|
with(Some("TRUE"), || assert!(super::self_authoring_enabled()));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Apply a pending proposal's `skill_candidate` items with no human decision.
|
||||||
|
///
|
||||||
|
/// ONLY `skill_candidate`. The other item kinds are deliberately left to the
|
||||||
|
/// human gate: `identity_refinement` rewrites an agent's system prompt and
|
||||||
|
/// `brain_consolidation` edits its memory, and both change what the agent IS
|
||||||
|
/// rather than adding a procedure it can consult. Self-authoring a skill is
|
||||||
|
/// recoverable — the row is workspace-scoped, versioned and revertible, and
|
||||||
|
/// cannot take a hand-authored name. Rewriting an identity autonomously is not
|
||||||
|
/// the same bet, and it is not the one that was asked for.
|
||||||
|
///
|
||||||
|
/// The remaining items stay pending, so a human still sees them.
|
||||||
|
pub async fn apply_autonomous(
|
||||||
|
pool: &PgPool,
|
||||||
|
workspace_id: cm_domain::WorkspaceId,
|
||||||
|
proposal_id: Uuid,
|
||||||
|
) -> Result<Vec<String>, String> {
|
||||||
|
let proposal = cm_db::repo::level_up::get(pool, proposal_id, workspace_id.as_uuid())
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("load proposal: {e}"))?
|
||||||
|
.ok_or_else(|| "proposal not found".to_string())?;
|
||||||
|
if proposal.status != "pending" {
|
||||||
|
return Err(format!("proposal already {}", proposal.status));
|
||||||
|
}
|
||||||
|
|
||||||
|
let items = proposal
|
||||||
|
.payload
|
||||||
|
.get("suggested_items")
|
||||||
|
.and_then(|v| v.as_array())
|
||||||
|
.cloned()
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
let mut applied: Vec<String> = Vec::new();
|
||||||
|
let mut candidates = 0usize;
|
||||||
|
for item in items {
|
||||||
|
let Some(item_id) = item.get("id").and_then(|v| v.as_str()) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
if item.get("kind").and_then(|v| v.as_str()) != Some("skill_candidate") {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
candidates += 1;
|
||||||
|
match apply_skill_candidate(pool, &proposal, &item).await {
|
||||||
|
Ok(()) => applied.push(item_id.to_string()),
|
||||||
|
// A refused draft is a normal outcome (a name collision with a
|
||||||
|
// hand-authored skill is the common one), not a failure of the
|
||||||
|
// sweep. Said out loud so a refusal is never mistaken for the
|
||||||
|
// agent simply not having proposed anything.
|
||||||
|
Err(e) => eprintln!(
|
||||||
|
"level_up: autonomous apply refused {item_id} for workspace {}: {e}",
|
||||||
|
workspace_id.as_uuid()
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if candidates == 0 {
|
||||||
|
return Ok(Vec::new());
|
||||||
|
}
|
||||||
|
cm_db::repo::level_up::mark_applied_autonomously(
|
||||||
|
pool,
|
||||||
|
proposal_id,
|
||||||
|
workspace_id.as_uuid(),
|
||||||
|
&applied,
|
||||||
|
applied.len() != candidates,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("mark applied: {e}"))?;
|
||||||
|
Ok(applied)
|
||||||
|
}
|
||||||
|
|
||||||
// ── Appliers ───────────────────────────────────────────────────
|
// ── Appliers ───────────────────────────────────────────────────
|
||||||
|
|
||||||
async fn apply_identity(
|
async fn apply_identity(
|
||||||
@@ -289,20 +426,61 @@ async fn apply_skill_candidate(
|
|||||||
.collect()
|
.collect()
|
||||||
})
|
})
|
||||||
.unwrap_or_default();
|
.unwrap_or_default();
|
||||||
|
// A draft may never take the name of a hand-authored skill.
|
||||||
|
//
|
||||||
|
// The row itself is safe — ids are workspace-scoped, so this cannot
|
||||||
|
// overwrite a builtin, and bindings resolve by skill_id rather than name,
|
||||||
|
// so it cannot shadow one either. What it CAN do is put two different
|
||||||
|
// procedures under one name in the same agent's bundle, and then nobody
|
||||||
|
// reading a transcript can tell which one the agent followed. That
|
||||||
|
// ambiguity is the whole problem in a system where the skill is the
|
||||||
|
// standard the behaviour is graded against.
|
||||||
|
let collides: Option<Uuid> = sqlx::query_scalar(
|
||||||
|
"SELECT id FROM skills WHERE name = $1 AND workspace_id IS NULL",
|
||||||
|
)
|
||||||
|
.bind(name)
|
||||||
|
.fetch_optional(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("check builtin collision: {e}"))?;
|
||||||
|
if collides.is_some() {
|
||||||
|
return Err(format!(
|
||||||
|
"skill name {name:?} is hand-authored — an agent-authored draft \
|
||||||
|
cannot take the name of a skill it is graded against"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
// Workspace-scoped custom skill. Deterministic id per
|
// Workspace-scoped custom skill. Deterministic id per
|
||||||
// (workspace, name) so re-approving the same draft updates in
|
// (workspace, name) so re-approving the same draft updates in
|
||||||
// place rather than duplicating.
|
// place rather than duplicating.
|
||||||
let id = workspace_skill_id(proposal.workspace_id, name);
|
let id = workspace_skill_id(proposal.workspace_id, name);
|
||||||
|
|
||||||
|
// Versioned, for the same reason builtins are: a self-authored skill that
|
||||||
|
// silently replaces its own body has no undo, and the version a run was
|
||||||
|
// judged under is the only way to read that run back honestly later.
|
||||||
|
let mut tx = pool.begin().await.map_err(|e| format!("begin: {e}"))?;
|
||||||
|
let existing: Option<(i32, String)> =
|
||||||
|
sqlx::query_as("SELECT current_version, body FROM skills WHERE id = $1")
|
||||||
|
.bind(id)
|
||||||
|
.fetch_optional(&mut *tx)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("read current skill: {e}"))?;
|
||||||
|
let (next_version, bump) = match &existing {
|
||||||
|
Some((v, prev)) if prev == body => (*v, false),
|
||||||
|
Some((v, _)) => (v + 1, true),
|
||||||
|
None => (1, true),
|
||||||
|
};
|
||||||
|
|
||||||
sqlx::query(
|
sqlx::query(
|
||||||
"INSERT INTO skills
|
"INSERT INTO skills
|
||||||
(id, name, title, author, description, when_to_use, tags,
|
(id, name, title, author, description, when_to_use, tags,
|
||||||
source_kind, workspace_id, current_version, body)
|
source_kind, workspace_id, current_version, body)
|
||||||
VALUES ($1,$2,$2,'level_up',$3,$4,$5,'promoted_from_brain',$6,1,$7)
|
VALUES ($1,$2,$2,'level_up',$3,$4,$5,'promoted_from_brain',$6,$8,$7)
|
||||||
ON CONFLICT (id) DO UPDATE SET
|
ON CONFLICT (id) DO UPDATE SET
|
||||||
description = EXCLUDED.description,
|
description = EXCLUDED.description,
|
||||||
when_to_use = EXCLUDED.when_to_use,
|
when_to_use = EXCLUDED.when_to_use,
|
||||||
tags = EXCLUDED.tags,
|
tags = EXCLUDED.tags,
|
||||||
body = EXCLUDED.body,
|
body = EXCLUDED.body,
|
||||||
|
current_version = EXCLUDED.current_version,
|
||||||
updated_at = now()",
|
updated_at = now()",
|
||||||
)
|
)
|
||||||
.bind(id)
|
.bind(id)
|
||||||
@@ -312,9 +490,28 @@ async fn apply_skill_candidate(
|
|||||||
.bind(&tags)
|
.bind(&tags)
|
||||||
.bind(proposal.workspace_id)
|
.bind(proposal.workspace_id)
|
||||||
.bind(body)
|
.bind(body)
|
||||||
.execute(pool)
|
.bind(next_version)
|
||||||
|
.execute(&mut *tx)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("upsert skill draft: {e}"))?;
|
.map_err(|e| format!("upsert skill draft: {e}"))?;
|
||||||
|
|
||||||
|
if bump {
|
||||||
|
sqlx::query(
|
||||||
|
"INSERT INTO skill_versions
|
||||||
|
(skill_id, version, body_md, description, when_to_use)
|
||||||
|
VALUES ($1,$2,$3,$4,$5)
|
||||||
|
ON CONFLICT DO NOTHING",
|
||||||
|
)
|
||||||
|
.bind(id)
|
||||||
|
.bind(next_version)
|
||||||
|
.bind(body)
|
||||||
|
.bind(description)
|
||||||
|
.bind(when_to_use)
|
||||||
|
.execute(&mut *tx)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("record skill version: {e}"))?;
|
||||||
|
}
|
||||||
|
tx.commit().await.map_err(|e| format!("commit: {e}"))?;
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -418,6 +615,7 @@ async fn recent_run_summary(pool: &PgPool, agent_id: Uuid, limit: i64) -> Result
|
|||||||
}
|
}
|
||||||
|
|
||||||
async fn call_llm_for_agent(
|
async fn call_llm_for_agent(
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
name: &str,
|
name: &str,
|
||||||
role: &str,
|
role: &str,
|
||||||
system_prompt: &str,
|
system_prompt: &str,
|
||||||
@@ -462,10 +660,13 @@ the sake of proposing."#;
|
|||||||
})
|
})
|
||||||
.to_string();
|
.to_string();
|
||||||
|
|
||||||
call_gemini_json(system, &user).await
|
call_llm_json(runtime, system, &user).await
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn call_llm_for_team(members: &[Value]) -> Result<Value, String> {
|
async fn call_llm_for_team(
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
|
members: &[Value],
|
||||||
|
) -> Result<Value, String> {
|
||||||
let system = r#"You review an AI team's roster + recent history and propose
|
let system = r#"You review an AI team's roster + recent history and propose
|
||||||
targeted improvements. Return ONLY JSON:
|
targeted improvements. Return ONLY JSON:
|
||||||
{
|
{
|
||||||
@@ -482,47 +683,90 @@ prompts over adding skills. Only add skills when a clear
|
|||||||
"the team keeps getting stuck on <X>" pattern appears."#;
|
"the team keeps getting stuck on <X>" pattern appears."#;
|
||||||
|
|
||||||
let user = json!({ "members": members }).to_string();
|
let user = json!({ "members": members }).to_string();
|
||||||
call_gemini_json(system, &user).await
|
call_llm_json(runtime, system, &user).await
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn call_gemini_json(system: &str, user: &str) -> Result<Value, String> {
|
/// Ask the configured proposer model for one JSON object.
|
||||||
let api_key =
|
///
|
||||||
std::env::var("GEMINI_API_KEY").map_err(|_| "GEMINI_API_KEY unset".to_string())?;
|
/// Goes through the provider registry rather than a vendor's HTTP API, so any
|
||||||
let model = model_name();
|
/// model the platform can already reach works and no single vendor's billing can
|
||||||
let url = format!(
|
/// take level-up down.
|
||||||
"https://generativelanguage.googleapis.com/v1beta/models/{}:generateContent?key={}",
|
///
|
||||||
model, api_key
|
/// The JSON is extracted rather than assumed: an anthropic-format model is not
|
||||||
);
|
/// bound by Gemini's `response_mime_type: application/json`, and will happily
|
||||||
let body = json!({
|
/// wrap an object in prose or a ```json fence. Parsing the raw reply worked
|
||||||
"system_instruction": { "parts": [{ "text": system }] },
|
/// against Gemini and would fail on everything else.
|
||||||
"contents": [{ "role": "user", "parts": [{ "text": user }] }],
|
async fn call_llm_json(
|
||||||
"generationConfig": {
|
runtime: &cm_runtime::Runtime,
|
||||||
"temperature": 0.2,
|
system: &str,
|
||||||
"response_mime_type": "application/json",
|
user: &str,
|
||||||
"maxOutputTokens": 8192,
|
) -> Result<Value, String> {
|
||||||
}
|
use cm_llm::{ChatMessage, ChatRequest, ChatRole, ContentPart, LlmEvent};
|
||||||
});
|
use futures::StreamExt as _;
|
||||||
let client = reqwest::Client::builder()
|
|
||||||
.timeout(std::time::Duration::from_secs(60))
|
let spec = model_name();
|
||||||
.build()
|
let (provider, model) = runtime.resolve_provider(&spec);
|
||||||
.map_err(|e| format!("http client: {e}"))?;
|
let request = ChatRequest {
|
||||||
let resp = client
|
system: system.to_string(),
|
||||||
.post(&url)
|
model: model.to_string(),
|
||||||
.json(&body)
|
messages: vec![ChatMessage {
|
||||||
.send()
|
role: ChatRole::User,
|
||||||
|
parts: vec![ContentPart::text(user)],
|
||||||
|
}],
|
||||||
|
tools: vec![],
|
||||||
|
max_tokens: 8192,
|
||||||
|
web_search: false,
|
||||||
|
};
|
||||||
|
let mut stream = provider
|
||||||
|
.stream(request)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("gemini call: {e}"))?;
|
.map_err(|e| format!("level-up call ({spec}): {e}"))?;
|
||||||
if !resp.status().is_success() {
|
let mut text = String::new();
|
||||||
let code = resp.status();
|
while let Some(event) = stream.next().await {
|
||||||
let body = resp.text().await.unwrap_or_default();
|
match event {
|
||||||
return Err(format!("gemini {code}: {}", &body[..body.len().min(500)]));
|
Ok(LlmEvent::TextDelta(t)) => text.push_str(&t),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => return Err(format!("level-up stream ({spec}): {e}")),
|
||||||
}
|
}
|
||||||
let json: Value = resp.json().await.map_err(|e| format!("gemini json: {e}"))?;
|
}
|
||||||
let text = json
|
let body = extract_json_object(&text)
|
||||||
.pointer("/candidates/0/content/parts/0/text")
|
.ok_or_else(|| format!("no JSON object in {spec} reply: {}", excerpt(&text, 300)))?;
|
||||||
.and_then(|v| v.as_str())
|
serde_json::from_str(body).map_err(|e| format!("parse suggestion json: {e}"))
|
||||||
.ok_or_else(|| "gemini response missing text".to_string())?;
|
}
|
||||||
serde_json::from_str(text).map_err(|e| format!("parse suggestion json: {e}"))
|
|
||||||
|
/// The outermost `{...}` in a reply, so a fenced or prose-wrapped object parses.
|
||||||
|
///
|
||||||
|
/// Brace-counting rather than a regex: a nested object would end a lazy match at
|
||||||
|
/// the first inner `}`, and these proposals are nested by design (items carry
|
||||||
|
/// per-role objects).
|
||||||
|
fn extract_json_object(text: &str) -> Option<&str> {
|
||||||
|
let start = text.find('{')?;
|
||||||
|
let mut depth = 0usize;
|
||||||
|
let mut in_string = false;
|
||||||
|
let mut escaped = false;
|
||||||
|
for (i, c) in text[start..].char_indices() {
|
||||||
|
if in_string {
|
||||||
|
match c {
|
||||||
|
_ if escaped => escaped = false,
|
||||||
|
'\\' => escaped = true,
|
||||||
|
'"' => in_string = false,
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
match c {
|
||||||
|
'"' => in_string = true,
|
||||||
|
'{' => depth += 1,
|
||||||
|
'}' => {
|
||||||
|
depth -= 1;
|
||||||
|
if depth == 0 {
|
||||||
|
return Some(&text[start..start + i + 1]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
None
|
||||||
}
|
}
|
||||||
|
|
||||||
fn excerpt(s: &str, max: usize) -> String {
|
fn excerpt(s: &str, max: usize) -> String {
|
||||||
@@ -546,3 +790,44 @@ fn workspace_skill_id(workspace_id: Uuid, name: &str) -> Uuid {
|
|||||||
bytes[8] = (bytes[8] & 0x3f) | 0x80;
|
bytes[8] = (bytes[8] & 0x3f) | 0x80;
|
||||||
Uuid::from_bytes(bytes)
|
Uuid::from_bytes(bytes)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
/// Gemini was asked for `response_mime_type: application/json` and obliged.
|
||||||
|
/// Anthropic-format models are under no such obligation and routinely wrap
|
||||||
|
/// the object in prose or a fenced block, so the reply is EXTRACTED, not
|
||||||
|
/// assumed. Parsing the raw text worked against Gemini and would fail
|
||||||
|
/// everywhere else — exactly the shape of bug a provider swap hides until
|
||||||
|
/// the first real proposal.
|
||||||
|
#[test]
|
||||||
|
fn a_json_object_is_extracted_from_however_the_model_wrapped_it() {
|
||||||
|
let bare = r#"{"items":[]}"#;
|
||||||
|
assert_eq!(super::extract_json_object(bare), Some(bare));
|
||||||
|
|
||||||
|
let fenced = "Here is my proposal:\n```json\n{\"items\":[1]}\n```\nDone.";
|
||||||
|
assert_eq!(super::extract_json_object(fenced), Some(r#"{"items":[1]}"#));
|
||||||
|
|
||||||
|
// Nested objects: a lazy match would stop at the first inner brace and
|
||||||
|
// hand back invalid JSON. These proposals are nested by design.
|
||||||
|
let nested = r#"prose {"a":{"b":{"c":1}},"d":2} trailing"#;
|
||||||
|
assert_eq!(
|
||||||
|
super::extract_json_object(nested),
|
||||||
|
Some(r#"{"a":{"b":{"c":1}},"d":2}"#)
|
||||||
|
);
|
||||||
|
|
||||||
|
// A brace inside a string must not close the object.
|
||||||
|
let stringy = r#"{"note":"an unmatched } here","ok":true}"#;
|
||||||
|
assert_eq!(super::extract_json_object(stringy), Some(stringy));
|
||||||
|
|
||||||
|
assert_eq!(super::extract_json_object("no object here"), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The default must not be a vendor whose billing already took a feature
|
||||||
|
/// down. It is a REGISTRY SPEC (`provider:model`), not a bare model name —
|
||||||
|
/// `resolve_provider` needs the provider half.
|
||||||
|
#[test]
|
||||||
|
fn the_default_proposer_is_a_registry_spec_and_not_gemini() {
|
||||||
|
assert!(super::DEFAULT_MODEL.contains(':'), "{}", super::DEFAULT_MODEL);
|
||||||
|
assert!(!super::DEFAULT_MODEL.contains("gemini"), "{}", super::DEFAULT_MODEL);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+102
-13
@@ -1,50 +1,77 @@
|
|||||||
//! REST API for Clawmates (spec §13). One route resource per module.
|
//! REST API for Clawmates (spec §13). One route resource per module.
|
||||||
|
|
||||||
|
pub mod agent_lifecycle;
|
||||||
|
pub mod agent_names;
|
||||||
|
pub mod auto_merge;
|
||||||
pub mod benchmark_runner;
|
pub mod benchmark_runner;
|
||||||
pub mod beszel;
|
pub mod beszel;
|
||||||
pub mod brain_seed;
|
pub mod brain_seed;
|
||||||
pub mod cleanup_sweeper;
|
pub mod cleanup_sweeper;
|
||||||
pub mod container_exec;
|
pub mod container_exec;
|
||||||
|
pub mod corpus;
|
||||||
mod error;
|
mod error;
|
||||||
pub mod evaluator;
|
pub mod evaluator;
|
||||||
pub mod evaluator_tools;
|
pub mod evaluator_tools;
|
||||||
mod extract;
|
mod extract;
|
||||||
pub mod fleet;
|
pub mod fleet;
|
||||||
pub mod fleet_herdr;
|
pub mod fleet_herdr;
|
||||||
|
pub mod harvest;
|
||||||
pub mod level_up;
|
pub mod level_up;
|
||||||
|
pub mod library;
|
||||||
|
pub mod live_bus;
|
||||||
mod mcp_door;
|
mod mcp_door;
|
||||||
mod mcp_skills;
|
mod mcp_skills;
|
||||||
pub mod mission_orchestrator;
|
pub mod microvm_client;
|
||||||
pub mod mission_refiner;
|
pub mod microvm_executor;
|
||||||
pub mod auto_merge;
|
pub mod microvm_turn_executor;
|
||||||
pub mod corpus;
|
pub mod continuous_research;
|
||||||
pub mod harvest;
|
|
||||||
pub mod library;
|
|
||||||
pub mod mission_delivery;
|
pub mod mission_delivery;
|
||||||
|
pub mod podcast;
|
||||||
|
pub mod mission_events;
|
||||||
pub mod mission_fs;
|
pub mod mission_fs;
|
||||||
pub mod papers;
|
pub mod mission_memory;
|
||||||
pub mod phase_config;
|
pub mod mission_gc;
|
||||||
pub mod session_executor;
|
pub mod mission_orchestrator;
|
||||||
pub mod runtime_preflight;
|
pub mod mission_schedule;
|
||||||
|
pub mod mission_outputs;
|
||||||
|
pub mod mission_plan;
|
||||||
|
pub mod mission_refiner;
|
||||||
|
pub mod mission_roster;
|
||||||
pub mod mission_runtime;
|
pub mod mission_runtime;
|
||||||
pub mod mission_workspace;
|
pub mod mission_workspace;
|
||||||
pub mod node_rules;
|
pub mod node_rules;
|
||||||
pub mod pdf_renderer;
|
pub mod papers;
|
||||||
|
pub mod phase_config;
|
||||||
pub mod phase_runner;
|
pub mod phase_runner;
|
||||||
pub mod phase_summarizer;
|
pub mod phase_summarizer;
|
||||||
pub mod quota;
|
pub mod quota;
|
||||||
mod recursive_exec;
|
mod recursive_exec;
|
||||||
|
pub mod repo_digest;
|
||||||
|
pub mod root_copy;
|
||||||
mod routes;
|
mod routes;
|
||||||
mod runtime_provision;
|
pub mod runtime_preflight;
|
||||||
|
pub mod runtime_provision;
|
||||||
pub mod security_scan;
|
pub mod security_scan;
|
||||||
|
pub mod session_executor;
|
||||||
|
pub mod container_tool_hooks;
|
||||||
|
pub mod gateway_preflight;
|
||||||
|
pub mod skill_delivery;
|
||||||
|
pub mod skill_self_authoring;
|
||||||
|
pub mod skill_use;
|
||||||
pub mod skills_loader;
|
pub mod skills_loader;
|
||||||
|
pub mod subscription;
|
||||||
pub mod swarm;
|
pub mod swarm;
|
||||||
pub mod task_card_parser;
|
pub mod task_card_parser;
|
||||||
pub mod task_card_worker;
|
pub mod task_card_worker;
|
||||||
pub mod team_template_loader;
|
pub mod team_template_loader;
|
||||||
pub mod tool_versions;
|
pub mod tool_versions;
|
||||||
mod topology_exec;
|
pub mod topology_exec;
|
||||||
pub mod topology_worker;
|
pub mod topology_worker;
|
||||||
|
pub mod validator_preflight;
|
||||||
|
pub mod vm_placement;
|
||||||
|
pub mod vm_stop_gate;
|
||||||
|
pub mod vm_tool_gate;
|
||||||
|
pub mod vm_tool_tap;
|
||||||
pub mod workflow_registry;
|
pub mod workflow_registry;
|
||||||
|
|
||||||
use axum::routing::{delete, get, patch, post};
|
use axum::routing::{delete, get, patch, post};
|
||||||
@@ -160,6 +187,8 @@ pub fn router(state: AppState) -> Router {
|
|||||||
.route("/api/world/live", get(routes::world::world_live))
|
.route("/api/world/live", get(routes::world::world_live))
|
||||||
.route("/api/world/replay", get(routes::world::world_replay))
|
.route("/api/world/replay", get(routes::world::world_replay))
|
||||||
.route("/api/nodes", get(routes::nodes::list))
|
.route("/api/nodes", get(routes::nodes::list))
|
||||||
|
.route("/api/fleet/capacity", get(routes::nodes::capacity))
|
||||||
|
.route("/api/fleet/backends", get(routes::nodes::backends))
|
||||||
.route("/api/nodes/pair", post(routes::nodes::pair))
|
.route("/api/nodes/pair", post(routes::nodes::pair))
|
||||||
.route("/api/nodes/live", get(routes::nodes::live))
|
.route("/api/nodes/live", get(routes::nodes::live))
|
||||||
.route("/api/nodes/agent", get(routes::nodes::agent_ws))
|
.route("/api/nodes/agent", get(routes::nodes::agent_ws))
|
||||||
@@ -225,6 +254,11 @@ pub fn router(state: AppState) -> Router {
|
|||||||
.route("/api/user/me", get(routes::identity::me))
|
.route("/api/user/me", get(routes::identity::me))
|
||||||
.route("/api/claws", post(routes::claws::create))
|
.route("/api/claws", post(routes::claws::create))
|
||||||
.route("/api/claws/batch-delete", post(routes::claws::batch_delete))
|
.route("/api/claws/batch-delete", post(routes::claws::batch_delete))
|
||||||
|
.route("/api/claws/lifecycle", get(routes::claws::lifecycle_census))
|
||||||
|
.route(
|
||||||
|
"/api/claws/lifecycle/sweep",
|
||||||
|
post(routes::claws::lifecycle_sweep),
|
||||||
|
)
|
||||||
.route("/api/claws/{id}", patch(routes::claws::patch))
|
.route("/api/claws/{id}", patch(routes::claws::patch))
|
||||||
.route("/api/claws/{id}", delete(routes::claws::delete))
|
.route("/api/claws/{id}", delete(routes::claws::delete))
|
||||||
.route("/api/claws/{id}/model", patch(routes::claws::set_model))
|
.route("/api/claws/{id}/model", patch(routes::claws::set_model))
|
||||||
@@ -468,9 +502,20 @@ pub fn router(state: AppState) -> Router {
|
|||||||
"/api/missions",
|
"/api/missions",
|
||||||
get(routes::missions::list).post(routes::missions::create),
|
get(routes::missions::list).post(routes::missions::create),
|
||||||
)
|
)
|
||||||
|
// The roster grouped by mission — what "My Workforce" renders.
|
||||||
|
.route("/api/workforce", get(routes::missions::workforce))
|
||||||
// The workflow recipe catalog (templates/workflows/*.toml). Serving it
|
// The workflow recipe catalog (templates/workflows/*.toml). Serving it
|
||||||
// lets the client stop mirroring the phase composition table inline.
|
// lets the client stop mirroring the phase composition table inline.
|
||||||
.route("/api/workflows", get(routes::missions::list_workflows))
|
.route("/api/workflows", get(routes::missions::list_workflows))
|
||||||
|
// The private podcast feed. Token in the query string, not a header:
|
||||||
|
// no podcast app can set headers. See `routes::podcast`.
|
||||||
|
.route("/api/podcast/feed.xml", get(routes::podcast::feed))
|
||||||
|
.route("/api/podcast/episodes", get(routes::podcast::list_episodes))
|
||||||
|
.route("/api/podcast/subscription", get(routes::podcast::subscription))
|
||||||
|
.route(
|
||||||
|
"/api/podcast/episodes/{file}",
|
||||||
|
get(routes::podcast::episode_audio),
|
||||||
|
)
|
||||||
.route(
|
.route(
|
||||||
"/api/missions/{id}",
|
"/api/missions/{id}",
|
||||||
get(routes::missions::get)
|
get(routes::missions::get)
|
||||||
@@ -482,6 +527,46 @@ pub fn router(state: AppState) -> Router {
|
|||||||
axum::routing::patch(routes::missions::set_status),
|
axum::routing::patch(routes::missions::set_status),
|
||||||
)
|
)
|
||||||
.route("/api/missions/{id}/refine", post(routes::missions::refine))
|
.route("/api/missions/{id}/refine", post(routes::missions::refine))
|
||||||
|
// Draft-less sibling: the wizard polishes a description before any
|
||||||
|
// mission exists, so there is no id to route on. Declared BEFORE the
|
||||||
|
// `{id}` routes would otherwise be ambiguous — axum matches literal
|
||||||
|
// segments first, but keeping them adjacent makes the pair obvious.
|
||||||
|
.route(
|
||||||
|
"/api/missions/refine-draft",
|
||||||
|
post(routes::missions::refine_draft),
|
||||||
|
)
|
||||||
|
.route(
|
||||||
|
"/api/missions/{id}/merge",
|
||||||
|
post(routes::missions::merge_branch),
|
||||||
|
)
|
||||||
|
.route(
|
||||||
|
"/api/missions/{id}/artifacts/{artifact_id}/content",
|
||||||
|
get(routes::missions::artifact_content),
|
||||||
|
)
|
||||||
|
.route(
|
||||||
|
"/api/missions/{id}/artifacts/{artifact_id}/download",
|
||||||
|
get(routes::missions::artifact_download),
|
||||||
|
)
|
||||||
|
// Slice 5: let a model size the mission's team. Proposing, listing and
|
||||||
|
// deciding are separate verbs because only the last one spends money.
|
||||||
|
// W1/#13: let a model author the phases, on the same propose → review →
|
||||||
|
// approve shape as the roster above.
|
||||||
|
.route(
|
||||||
|
"/api/missions/{id}/plan-proposals",
|
||||||
|
get(routes::mission_plan::list).post(routes::mission_plan::suggest),
|
||||||
|
)
|
||||||
|
.route(
|
||||||
|
"/api/missions/{id}/plan-proposals/{pid}/decide",
|
||||||
|
post(routes::mission_plan::decide),
|
||||||
|
)
|
||||||
|
.route(
|
||||||
|
"/api/missions/{id}/team-proposals",
|
||||||
|
get(routes::mission_roster::list).post(routes::mission_roster::suggest),
|
||||||
|
)
|
||||||
|
.route(
|
||||||
|
"/api/missions/{id}/team-proposals/{pid}/decide",
|
||||||
|
post(routes::mission_roster::decide),
|
||||||
|
)
|
||||||
.route(
|
.route(
|
||||||
"/api/missions/{id}/herdr-dispatch",
|
"/api/missions/{id}/herdr-dispatch",
|
||||||
post(routes::missions::herdr_dispatch),
|
post(routes::missions::herdr_dispatch),
|
||||||
@@ -511,6 +596,10 @@ pub fn router(state: AppState) -> Router {
|
|||||||
"/api/missions/{id}/phases/{phase_id}/evaluations",
|
"/api/missions/{id}/phases/{phase_id}/evaluations",
|
||||||
get(routes::missions::list_phase_evaluations),
|
get(routes::missions::list_phase_evaluations),
|
||||||
)
|
)
|
||||||
|
.route(
|
||||||
|
"/api/missions/{id}/skill-use",
|
||||||
|
get(routes::missions::skill_use),
|
||||||
|
)
|
||||||
.route(
|
.route(
|
||||||
"/api/missions/{id}/teams",
|
"/api/missions/{id}/teams",
|
||||||
get(routes::missions::list_teams),
|
get(routes::missions::list_teams),
|
||||||
|
|||||||
@@ -93,9 +93,13 @@ pub async fn clone_vault(clone_url: &str, work_root: &Path) -> Result<PathBuf, S
|
|||||||
.map_err(|e| format!("mkdir {}: {e}", work_root.display()))?;
|
.map_err(|e| format!("mkdir {}: {e}", work_root.display()))?;
|
||||||
|
|
||||||
let auth = mission_workspace::with_ambient_auth(clone_url);
|
let auth = mission_workspace::with_ambient_auth(clone_url);
|
||||||
let out = tokio::process::Command::new("git")
|
if let Some(why) = &auth.unauthenticated {
|
||||||
.args(["clone", "--quiet", "--depth", "1", &auth])
|
eprintln!("library: cloning the vault WITHOUT credentials — {why}");
|
||||||
.arg(&path)
|
}
|
||||||
|
let mut cmd = tokio::process::Command::new("git");
|
||||||
|
cmd.args(["clone", "--quiet", "--depth", "1", &auth.url])
|
||||||
|
.arg(&path);
|
||||||
|
let out = mission_workspace::no_terminal_prompt(&mut cmd)
|
||||||
.output()
|
.output()
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("spawn git clone: {e}"))?;
|
.map_err(|e| format!("spawn git clone: {e}"))?;
|
||||||
@@ -103,16 +107,15 @@ pub async fn clone_vault(clone_url: &str, work_root: &Path) -> Result<PathBuf, S
|
|||||||
return Err(format!(
|
return Err(format!(
|
||||||
"clone vault → {}: {}",
|
"clone vault → {}: {}",
|
||||||
out.status,
|
out.status,
|
||||||
mission_workspace::redact_token(&String::from_utf8_lossy(&out.stderr))
|
crate::evaluator_tools::clamp_output(&mission_workspace::redact_token(
|
||||||
.chars()
|
&String::from_utf8_lossy(&out.stderr)
|
||||||
.take(300)
|
))
|
||||||
.collect::<String>()
|
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
// The token must not stay in .git/config: the checkout may be handed to a
|
// The token must not stay in .git/config: the checkout may be handed to a
|
||||||
// container later, and a credential in a file an agent can read is a
|
// container later, and a credential in a file an agent can read is a
|
||||||
// credential an agent has.
|
// credential an agent has.
|
||||||
mission_workspace::scrub_remote_credentials(&path, &auth);
|
mission_workspace::scrub_remote_credentials(&path, &auth.url);
|
||||||
Ok(path)
|
Ok(path)
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -150,6 +153,7 @@ pub async fn run_to_vault(
|
|||||||
total.shelved.extend(h.shelved);
|
total.shelved.extend(h.shelved);
|
||||||
total.failed.extend(h.failed);
|
total.failed.extend(h.failed);
|
||||||
total.notes_written.extend(h.notes_written);
|
total.notes_written.extend(h.notes_written);
|
||||||
|
total.papers.extend(h.papers);
|
||||||
}
|
}
|
||||||
|
|
||||||
// The TAIL of the uuid, not the head. UUIDv7 leads with a 48-bit
|
// The TAIL of the uuid, not the head. UUIDv7 leads with a 48-bit
|
||||||
@@ -172,6 +176,7 @@ pub async fn run_to_vault(
|
|||||||
}
|
}
|
||||||
|
|
||||||
git(&vault, &["checkout", "-B", &branch]).await?;
|
git(&vault, &["checkout", "-B", &branch]).await?;
|
||||||
|
|
||||||
git(&vault, &["add", "--", "60 Papers"]).await?;
|
git(&vault, &["add", "--", "60 Papers"]).await?;
|
||||||
let message = format!(
|
let message = format!(
|
||||||
"library: {} new paper(s)\n\n{}\n\nShelved in the blob store; this commit is the catalogue.",
|
"library: {} new paper(s)\n\n{}\n\nShelved in the blob store; this commit is the catalogue.",
|
||||||
@@ -186,6 +191,15 @@ pub async fn run_to_vault(
|
|||||||
git(&vault, &["commit", "--no-verify", "-m", &message]).await?;
|
git(&vault, &["commit", "--no-verify", "-m", &message]).await?;
|
||||||
|
|
||||||
let auth = mission_workspace::with_ambient_auth(clone_url);
|
let auth = mission_workspace::with_ambient_auth(clone_url);
|
||||||
|
if let Some(why) = &auth.unauthenticated {
|
||||||
|
if auth.is_forge() {
|
||||||
|
// Not fatal here — the push below reports its own failure — but the
|
||||||
|
// reason belongs in the log next to the attempt, not inferred from a
|
||||||
|
// tty error two layers down.
|
||||||
|
eprintln!("library: pushing to the forge WITHOUT credentials — {why}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let auth = auth.url;
|
||||||
let refspec = format!("HEAD:refs/heads/{branch}");
|
let refspec = format!("HEAD:refs/heads/{branch}");
|
||||||
match git(&vault, &["push", &auth, &refspec]).await {
|
match git(&vault, &["push", &auth, &refspec]).await {
|
||||||
Ok(_) => {
|
Ok(_) => {
|
||||||
|
|||||||
@@ -0,0 +1,126 @@
|
|||||||
|
//! A process-wide push bus for live taxonomy events.
|
||||||
|
//!
|
||||||
|
//! `/api/world/live` is a 2-second database poll. That is the right shape for
|
||||||
|
//! state you can query — statuses, phases, telemetry — and the wrong shape for a
|
||||||
|
//! token stream: an agent's reasoning only becomes visible after the step
|
||||||
|
//! finishes and its text is persisted, so the REASONING STREAM card showed
|
||||||
|
//! completed paragraphs rather than an agent thinking.
|
||||||
|
//!
|
||||||
|
//! This carries the frames that cannot wait for a round trip through Postgres.
|
||||||
|
//! `topology_exec` publishes as the runtime's WebSocket delivers them; the SSE
|
||||||
|
//! handler subscribes and forwards, so a chunk reaches the browser in one hop.
|
||||||
|
//!
|
||||||
|
//! **Why a global rather than a field on `AppState`.** The publisher is
|
||||||
|
//! `topology_exec`, reached through `phase_runner` → `topology_worker` →
|
||||||
|
//! `MissionTap`, none of which hold `AppState`. Threading a handle through all
|
||||||
|
//! of them would put a UI concern into four layers that have no other reason to
|
||||||
|
//! know about one. There is exactly one bus per process and it holds no
|
||||||
|
//! per-request state, so a `OnceLock` is the honest representation.
|
||||||
|
//!
|
||||||
|
//! **Lossy on purpose.** A slow reader lags and skips rather than applying
|
||||||
|
//! backpressure to the agent that is producing. Dropping frames degrades a live
|
||||||
|
//! view; blocking would slow the mission to the speed of the slowest open tab.
|
||||||
|
//! The durable record is `mission_events` — this bus is the fast path, never the
|
||||||
|
//! source of truth.
|
||||||
|
|
||||||
|
use std::sync::{Arc, OnceLock};
|
||||||
|
|
||||||
|
use serde_json::Value;
|
||||||
|
use tokio::sync::broadcast;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// Bounded so a stalled subscriber costs memory once, not unboundedly. At
|
||||||
|
/// token granularity a busy mission produces a few hundred frames a second;
|
||||||
|
/// this is roughly a couple of seconds of slack before a slow reader starts
|
||||||
|
/// skipping.
|
||||||
|
const CAPACITY: usize = 2048;
|
||||||
|
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct LiveEvent {
|
||||||
|
/// Every subscriber is workspace-scoped; the bus is not.
|
||||||
|
pub workspace_id: Uuid,
|
||||||
|
/// A taxonomy type, e.g. `agent.reasoning.delta`.
|
||||||
|
pub kind: String,
|
||||||
|
pub data: Value,
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct LiveBus {
|
||||||
|
tx: broadcast::Sender<LiveEvent>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl LiveBus {
|
||||||
|
fn new() -> LiveBus {
|
||||||
|
let (tx, _rx) = broadcast::channel(CAPACITY);
|
||||||
|
LiveBus { tx }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Publish. Returns immediately, and succeeds even with no subscribers —
|
||||||
|
/// nobody watching is the normal case, not an error.
|
||||||
|
pub fn publish(&self, workspace_id: Uuid, kind: &str, data: Value) {
|
||||||
|
let _ = self.tx.send(LiveEvent {
|
||||||
|
workspace_id,
|
||||||
|
kind: kind.to_string(),
|
||||||
|
data,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn subscribe(&self) -> broadcast::Receiver<LiveEvent> {
|
||||||
|
self.tx.subscribe()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static BUS: OnceLock<Arc<LiveBus>> = OnceLock::new();
|
||||||
|
|
||||||
|
pub fn global() -> &'static Arc<LiveBus> {
|
||||||
|
BUS.get_or_init(|| Arc::new(LiveBus::new()))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The claw alias the runtime dispatches on (`claw_<uuid>`) → the agent id the
|
||||||
|
/// UI keys on. Returns `None` for any other alias — the governor, the door and
|
||||||
|
/// the evaluator all drive turns under names that are not claws, and attributing
|
||||||
|
/// their output to an agent would put words in someone's mouth.
|
||||||
|
pub fn agent_id_from_alias(alias: &str) -> Option<Uuid> {
|
||||||
|
Uuid::parse_str(alias.strip_prefix("claw_")?).ok()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn only_claw_aliases_resolve_to_an_agent() {
|
||||||
|
let id = Uuid::now_v7();
|
||||||
|
assert_eq!(
|
||||||
|
agent_id_from_alias(&format!("claw_{id}")),
|
||||||
|
Some(id),
|
||||||
|
"the runtime's own alias form must resolve"
|
||||||
|
);
|
||||||
|
// These drive real turns and must NOT be attributed to an agent.
|
||||||
|
for other in ["scout", "coordinator", "door", "evaluator", "claw_nonsense"] {
|
||||||
|
assert_eq!(agent_id_from_alias(other), None, "{other}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
async fn a_subscriber_receives_what_is_published() {
|
||||||
|
let bus = LiveBus::new();
|
||||||
|
let mut rx = bus.subscribe();
|
||||||
|
let ws = Uuid::now_v7();
|
||||||
|
bus.publish(
|
||||||
|
ws,
|
||||||
|
"agent.reasoning.delta",
|
||||||
|
serde_json::json!({"text": "hi"}),
|
||||||
|
);
|
||||||
|
let ev = rx.recv().await.expect("delivered");
|
||||||
|
assert_eq!(ev.workspace_id, ws);
|
||||||
|
assert_eq!(ev.kind, "agent.reasoning.delta");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Publishing with nobody listening must not error — that is the common
|
||||||
|
/// case (no browser open) and it must never disturb the mission.
|
||||||
|
#[test]
|
||||||
|
fn publishing_into_the_void_is_fine() {
|
||||||
|
let bus = LiveBus::new();
|
||||||
|
bus.publish(Uuid::now_v7(), "agent.tool.call", serde_json::json!({}));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -7,9 +7,10 @@
|
|||||||
//! (broker-executed tools reveal secrets only inside the broker), every action
|
//! (broker-executed tools reveal secrets only inside the broker), every action
|
||||||
//! is journaled to the append-only audit log, and a central policy decides each
|
//! is journaled to the append-only audit log, and a central policy decides each
|
||||||
//! call — but the **human approver is replaced by an automated policy/governor**
|
//! call — but the **human approver is replaced by an automated policy/governor**
|
||||||
//! ("agents control their destiny"). The default policy is allow-all, so agents
|
//! ("agents control their destiny"). Recipient allowlists, spend caps, taint
|
||||||
//! are autonomous out of the gate; recipient allowlists, spend caps, taint
|
//! blocks, or a governor agent plug into [`policy_decide`]. Since 2026-09-20
|
||||||
//! blocks, or a governor agent plug into [`policy_decide`].
|
//! the door is **closed by default**: a call is approved only by a governor
|
||||||
|
//! that answered ALLOW, or by an explicit `CLAWMATES_DOOR_POLICY=allow`.
|
||||||
//!
|
//!
|
||||||
//! v1 exposes `email_send` (runtime-executed → `outbox`, observable, no external
|
//! v1 exposes `email_send` (runtime-executed → `outbox`, observable, no external
|
||||||
//! creds). Broker-backed tools (e.g. `slack_post`) are the next increment — they
|
//! creds). Broker-backed tools (e.g. `slack_post`) are the next increment — they
|
||||||
@@ -85,10 +86,14 @@ enum PolicyOutcome {
|
|||||||
/// 2. a per-workspace hourly rate cap (`CLAWMATES_DOOR_RATE_LIMIT`, counts
|
/// 2. a per-workspace hourly rate cap (`CLAWMATES_DOOR_RATE_LIMIT`, counts
|
||||||
/// executed door actions in the audit log);
|
/// executed door actions in the audit log);
|
||||||
/// 3. an email recipient-domain allowlist (`CLAWMATES_DOOR_EMAIL_ALLOW`);
|
/// 3. an email recipient-domain allowlist (`CLAWMATES_DOOR_EMAIL_ALLOW`);
|
||||||
/// 4. a governor hook (extension point) — a deterministic rule set or a
|
/// 4. a governor agent (`CLAWMATES_DOOR_GOVERNOR`) — must answer ALLOW;
|
||||||
/// governor agent can veto here.
|
/// unreachable, silent, or off-contract means DENY;
|
||||||
|
/// 5. with no governor, an explicit `CLAWMATES_DOOR_POLICY=allow`.
|
||||||
///
|
///
|
||||||
/// Default (no env set) = allow-all → agents fully autonomous.
|
/// Default (no env set) = **deny**. This was allow-all until 2026-09-20, and
|
||||||
|
/// the governor failed open on top of that, so with the judge plan emptied
|
||||||
|
/// every outbound action was approved by nobody. See [`ungoverned_default`]
|
||||||
|
/// and `cm_runtime::Runtime::judge` for the measurement behind the flip.
|
||||||
async fn policy_decide(
|
async fn policy_decide(
|
||||||
state: &AppState,
|
state: &AppState,
|
||||||
workspace: cm_domain::WorkspaceId,
|
workspace: cm_domain::WorkspaceId,
|
||||||
@@ -140,9 +145,10 @@ async fn policy_decide(
|
|||||||
}
|
}
|
||||||
|
|
||||||
// 4. Governor agent: when CLAWMATES_DOOR_GOVERNOR is set, an LLM judges the
|
// 4. Governor agent: when CLAWMATES_DOOR_GOVERNOR is set, an LLM judges the
|
||||||
// action and can veto — the "self-governing topology" path. Fail-open
|
// action — the "self-governing topology" path. Fail-CLOSED: a governor
|
||||||
// (a governor outage doesn't halt agents); deterministic rules above are
|
// that cannot be reached, or that does not say ALLOW, denies. The
|
||||||
// the hard floor.
|
// deterministic rules above are the hard floor; this is the only
|
||||||
|
// approver.
|
||||||
if std::env::var("CLAWMATES_DOOR_GOVERNOR").is_ok() {
|
if std::env::var("CLAWMATES_DOOR_GOVERNOR").is_ok() {
|
||||||
let system = "You are a security governor for an autonomous agent's outbound actions. \
|
let system = "You are a security governor for an autonomous agent's outbound actions. \
|
||||||
Reply with exactly ALLOW or DENY on the first line, then one short reason. \
|
Reply with exactly ALLOW or DENY on the first line, then one short reason. \
|
||||||
@@ -166,12 +172,39 @@ async fn policy_decide(
|
|||||||
} else {
|
} else {
|
||||||
state.runtime.judge(system, &request).await
|
state.runtime.judge(system, &request).await
|
||||||
};
|
};
|
||||||
|
// `judge()` returns this exact prefix when the provider never answered,
|
||||||
|
// which a rate-limited or uncredited judge model does on every call.
|
||||||
|
// Still logged loudly: a door that denies everything because its
|
||||||
|
// governor is down is safe, and is also a platform with no outbound
|
||||||
|
// actions until someone reads this line.
|
||||||
|
if reason.starts_with("governor unreachable") {
|
||||||
|
eprintln!(
|
||||||
|
"mcp_door: WARNING — the door governor is unreachable, DENYING {mcp_tool} \
|
||||||
|
({reason}). Point CLAWMATES_JUDGE_MODEL at a reachable model."
|
||||||
|
);
|
||||||
|
}
|
||||||
if !allow {
|
if !allow {
|
||||||
return PolicyOutcome::Deny(format!("governor agent vetoed — {reason}"));
|
return PolicyOutcome::Deny(format!("governor agent vetoed — {reason}"));
|
||||||
}
|
}
|
||||||
|
return PolicyOutcome::Approve;
|
||||||
}
|
}
|
||||||
|
|
||||||
PolicyOutcome::Approve
|
// 5. No governor. The door is closed unless the operator opened it.
|
||||||
|
ungoverned_default(std::env::var("CLAWMATES_DOOR_POLICY").ok().as_deref())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The posture with no governor configured. Only the literal `allow` opens
|
||||||
|
/// the door; unset, empty, or anything else keeps it shut and says how to
|
||||||
|
/// open it. `deny` is handled earlier as the kill switch and lands here too.
|
||||||
|
fn ungoverned_default(policy: Option<&str>) -> PolicyOutcome {
|
||||||
|
match policy.map(str::trim) {
|
||||||
|
Some("allow") => PolicyOutcome::Approve,
|
||||||
|
_ => PolicyOutcome::Deny(
|
||||||
|
"the door has no governor and no allow policy — set CLAWMATES_DOOR_GOVERNOR=1 \
|
||||||
|
or, to run ungoverned, CLAWMATES_DOOR_POLICY=allow"
|
||||||
|
.into(),
|
||||||
|
),
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Mint an auto-approved approval + single-use execution grant for a
|
/// Mint an auto-approved approval + single-use execution grant for a
|
||||||
@@ -225,12 +258,21 @@ async fn mint_grant(
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Authenticate the bearer header → workspace/user. `None` if missing/invalid.
|
/// Authenticate the bearer header → workspace/user. `None` if missing/invalid.
|
||||||
|
///
|
||||||
|
/// Accepts [`cm_auth::SCOPE_AGENT_DOOR`] as well as a person's session. This
|
||||||
|
/// route is the one that can `delegate`, and the thing that will eventually
|
||||||
|
/// hold a token for it is an agent runtime — so the narrow credential has to
|
||||||
|
/// exist before something reaches for the only one that does.
|
||||||
async fn authed(state: &AppState, headers: &HeaderMap) -> Option<cm_auth::AuthedUser> {
|
async fn authed(state: &AppState, headers: &HeaderMap) -> Option<cm_auth::AuthedUser> {
|
||||||
let token = headers
|
let token = headers
|
||||||
.get(AUTHORIZATION)
|
.get(AUTHORIZATION)
|
||||||
.and_then(|v| v.to_str().ok())
|
.and_then(|v| v.to_str().ok())
|
||||||
.and_then(|v| v.strip_prefix("Bearer "))?;
|
.and_then(|v| v.strip_prefix("Bearer "))?;
|
||||||
state.auth.authenticate(token).await.ok()
|
state
|
||||||
|
.auth
|
||||||
|
.authenticate_scoped(token, cm_auth::SCOPE_AGENT_DOOR)
|
||||||
|
.await
|
||||||
|
.ok()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve the specific claw making the call. Our ZeroClaw fork stamps the
|
/// Resolve the specific claw making the call. Our ZeroClaw fork stamps the
|
||||||
@@ -588,6 +630,18 @@ pub async fn mcp(
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// The default posture is closed. Before 2026-09-20 an unset policy
|
||||||
|
/// meant allow-all.
|
||||||
|
#[test]
|
||||||
|
fn the_door_is_closed_unless_opened() {
|
||||||
|
assert!(matches!(ungoverned_default(None), PolicyOutcome::Deny(_)));
|
||||||
|
assert!(matches!(ungoverned_default(Some("")), PolicyOutcome::Deny(_)));
|
||||||
|
assert!(matches!(ungoverned_default(Some("deny")), PolicyOutcome::Deny(_)));
|
||||||
|
assert!(matches!(ungoverned_default(Some("yes")), PolicyOutcome::Deny(_)));
|
||||||
|
assert!(matches!(ungoverned_default(Some("allow")), PolicyOutcome::Approve));
|
||||||
|
assert!(matches!(ungoverned_default(Some(" allow ")), PolicyOutcome::Approve));
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn exposed_tool_name_maps_to_registry_name() {
|
fn exposed_tool_name_maps_to_registry_name() {
|
||||||
assert_eq!(internal_name("email_send"), Some("email.send"));
|
assert_eq!(internal_name("email_send"), Some("email.send"));
|
||||||
|
|||||||
@@ -60,12 +60,26 @@ fn err(id: Option<Value>, code: i64, message: &str) -> Json<Value> {
|
|||||||
|
|
||||||
// ── Auth ─────────────────────────────────────────────────────────
|
// ── Auth ─────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
/// This endpoint accepts a **narrow** credential as well as a person's session.
|
||||||
|
///
|
||||||
|
/// It is the one route a mission container is given a token for, and that token
|
||||||
|
/// sits in a file the agent can `cat`. Mission agents run arbitrary `Bash` with
|
||||||
|
/// egress and no read gate, so a full session here would be an owner-privileged
|
||||||
|
/// API key handed to something explicitly untrusted — which is why
|
||||||
|
/// `SCOPE_SKILLS_READ` exists and why this is the only call site that names it.
|
||||||
|
///
|
||||||
|
/// `authenticate_scoped` still accepts `full`, so the UI and any human caller
|
||||||
|
/// are unaffected.
|
||||||
async fn authed(state: &AppState, headers: &HeaderMap) -> Option<cm_auth::AuthedUser> {
|
async fn authed(state: &AppState, headers: &HeaderMap) -> Option<cm_auth::AuthedUser> {
|
||||||
let token = headers
|
let token = headers
|
||||||
.get(AUTHORIZATION)
|
.get(AUTHORIZATION)
|
||||||
.and_then(|v| v.to_str().ok())
|
.and_then(|v| v.to_str().ok())
|
||||||
.and_then(|v| v.strip_prefix("Bearer "))?;
|
.and_then(|v| v.strip_prefix("Bearer "))?;
|
||||||
state.auth.authenticate(token).await.ok()
|
state
|
||||||
|
.auth
|
||||||
|
.authenticate_scoped(token, cm_auth::SCOPE_SKILLS_READ)
|
||||||
|
.await
|
||||||
|
.ok()
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Resolve the calling agent via `X-ZeroClaw-Agent` header
|
/// Resolve the calling agent via `X-ZeroClaw-Agent` header
|
||||||
@@ -92,7 +106,7 @@ async fn caller_agent(
|
|||||||
|
|
||||||
// ── URI helpers ──────────────────────────────────────────────────
|
// ── URI helpers ──────────────────────────────────────────────────
|
||||||
|
|
||||||
fn skill_uri(workspace_id: Option<Uuid>, name: &str) -> String {
|
pub(crate) fn skill_uri(workspace_id: Option<Uuid>, name: &str) -> String {
|
||||||
match workspace_id {
|
match workspace_id {
|
||||||
Some(ws) => format!("{URI_PREFIX_WORKSPACE}{ws}/{name}"),
|
Some(ws) => format!("{URI_PREFIX_WORKSPACE}{ws}/{name}"),
|
||||||
None => format!("{URI_PREFIX_GLOBAL}{name}"),
|
None => format!("{URI_PREFIX_GLOBAL}{name}"),
|
||||||
@@ -100,7 +114,7 @@ fn skill_uri(workspace_id: Option<Uuid>, name: &str) -> String {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Parse `skill:global/<name>` or `skill:workspace/<ws>/<name>`.
|
/// Parse `skill:global/<name>` or `skill:workspace/<ws>/<name>`.
|
||||||
fn parse_uri(uri: &str) -> Option<(Option<Uuid>, String)> {
|
pub(crate) fn parse_uri(uri: &str) -> Option<(Option<Uuid>, String)> {
|
||||||
if let Some(name) = uri.strip_prefix(URI_PREFIX_GLOBAL) {
|
if let Some(name) = uri.strip_prefix(URI_PREFIX_GLOBAL) {
|
||||||
return Some((None, name.to_string()));
|
return Some((None, name.to_string()));
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,354 @@
|
|||||||
|
//! Drive a fleet node's microVMs from the server.
|
||||||
|
//!
|
||||||
|
//! Thin by design: the node owns the VM lifecycle (see
|
||||||
|
//! `clawmates-node::microvm`), and this is the typed way to ask it. Every call
|
||||||
|
//! is one `vm_*` op over the existing `NodeHub` request/response channel, so
|
||||||
|
//! there is no new transport, correlation or timeout machinery.
|
||||||
|
//!
|
||||||
|
//! # Not a `SandboxDriver`
|
||||||
|
//!
|
||||||
|
//! `RemoteDriver` exists to marshal `SandboxDriver` over the hub, and reusing it
|
||||||
|
//! was the plan. That trait is container-shaped — `attach_pty`, `resize_pty`,
|
||||||
|
//! argv `exec` — while a mission needs inject → run → collect. Conforming would
|
||||||
|
//! mean implementing PTY-over-vsock semantics that nothing calls, so this speaks
|
||||||
|
//! the smaller interface the mission path actually uses.
|
||||||
|
//!
|
||||||
|
//! # Timeouts
|
||||||
|
//!
|
||||||
|
//! The hub defaults to 20s, which is right for a create (measured: ~1s) and
|
||||||
|
//! badly wrong for an agent turn. `exec` therefore takes its own budget and
|
||||||
|
//! passes it to BOTH the hub and the guest, with the hub's slightly longer: if
|
||||||
|
//! the guest's own timeout fires first the reply says so, whereas a hub timeout
|
||||||
|
//! leaves us guessing whether the command is still running.
|
||||||
|
|
||||||
|
use cm_domain::NodeId;
|
||||||
|
use serde_json::{json, Value};
|
||||||
|
|
||||||
|
use crate::fleet::NodeHub;
|
||||||
|
|
||||||
|
/// Slack between the guest's deadline and the hub's, so the guest's own timeout
|
||||||
|
/// wins the race and we get a real answer rather than a transport error.
|
||||||
|
const HUB_GRACE_SECS: u64 = 30;
|
||||||
|
|
||||||
|
/// How long the hub waits for a command whose own budget is `guest_secs`.
|
||||||
|
///
|
||||||
|
/// Saturating, not `+`: a caller passing a very large budget would otherwise
|
||||||
|
/// overflow and panic in debug or wrap to a tiny timeout in release — the second
|
||||||
|
/// being far worse, since it turns a long-running agent turn into a spurious
|
||||||
|
/// transport failure.
|
||||||
|
fn hub_deadline(guest_secs: u64) -> u64 {
|
||||||
|
guest_secs.saturating_add(HUB_GRACE_SECS)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct MicroVm<'a> {
|
||||||
|
hub: &'a NodeHub,
|
||||||
|
node_id: NodeId,
|
||||||
|
vm_id: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'a> MicroVm<'a> {
|
||||||
|
pub fn new(hub: &'a NodeHub, node_id: NodeId, vm_id: impl Into<String>) -> Self {
|
||||||
|
Self {
|
||||||
|
hub,
|
||||||
|
node_id,
|
||||||
|
vm_id: vm_id.into(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn vm_id(&self) -> &str {
|
||||||
|
&self.vm_id
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One op, with the node's `output` string parsed back into JSON.
|
||||||
|
///
|
||||||
|
/// `output` is a String on the wire (`Uplink::Result`), and a node that
|
||||||
|
/// answered with a JSON object instead made the whole frame unparseable —
|
||||||
|
/// the reply then vanished into the uplink's error arm and the call timed
|
||||||
|
/// out with nothing explaining why. Parsing here, loudly, keeps that
|
||||||
|
/// mismatch a visible error rather than a mystery timeout.
|
||||||
|
async fn call(&self, op: &str, mut args: Value, secs: u64) -> Result<Value, String> {
|
||||||
|
if let Some(o) = args.as_object_mut() {
|
||||||
|
o.insert("vm_id".into(), Value::String(self.vm_id.clone()));
|
||||||
|
}
|
||||||
|
let out = self
|
||||||
|
.hub
|
||||||
|
.call_timeout(self.node_id, op, args, secs)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("{op} on node {:?}: {e}", self.node_id))?;
|
||||||
|
let body: Value = serde_json::from_str(&out.output)
|
||||||
|
.map_err(|e| format!("{op} returned unparseable output ({e}): {}", out.output))?;
|
||||||
|
if !out.ok {
|
||||||
|
let why = body
|
||||||
|
.get("error")
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.unwrap_or(&out.output);
|
||||||
|
return Err(format!("{op} failed: {why}"));
|
||||||
|
}
|
||||||
|
Ok(body)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Boot the VM. Returns only once its guest agent has answered.
|
||||||
|
///
|
||||||
|
/// `backend` selects the rootfs image (`missions.backend`); `None` boots the
|
||||||
|
/// node's default. A backend whose image is not built on that node is an
|
||||||
|
/// error naming the file — never a quiet fall back to the default, which
|
||||||
|
/// would run a claude mission in a kimi VM and report success.
|
||||||
|
pub async fn create(
|
||||||
|
&self,
|
||||||
|
vcpus: u32,
|
||||||
|
mem_mib: u32,
|
||||||
|
backend: Option<&str>,
|
||||||
|
) -> Result<Value, String> {
|
||||||
|
// 60s, not the hub default: a create that has to copy a rootfs and boot
|
||||||
|
// is measured near 1s, but a node under load has no reason to be fast.
|
||||||
|
self.call(
|
||||||
|
"vm_create",
|
||||||
|
json!({ "vcpus": vcpus, "mem_mib": mem_mib, "backend": backend }),
|
||||||
|
60,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unpack a tar inside the guest at `dest`.
|
||||||
|
///
|
||||||
|
/// Takes the archive bytes rather than a path: the server holds the mission
|
||||||
|
/// checkout, the node does not, and shipping the tar is the whole point of
|
||||||
|
/// the inject → run → collect model.
|
||||||
|
pub async fn inject(&self, dest: &str, tar: &[u8]) -> Result<Value, String> {
|
||||||
|
use base64::Engine as _;
|
||||||
|
let b64 = base64::engine::general_purpose::STANDARD.encode(tar);
|
||||||
|
self.call("vm_inject", json!({ "dest": dest, "tar_b64": b64 }), 120)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Run a shell command in the guest.
|
||||||
|
///
|
||||||
|
/// `Ok` means the command RAN; the exit code is in the payload. A non-zero
|
||||||
|
/// exit is not an error here — the caller has to be able to tell "the build
|
||||||
|
/// failed" from "we could not reach the VM", and collapsing them is the
|
||||||
|
/// defect this codebase keeps paying for.
|
||||||
|
/// `env` carries the provider credentials (see
|
||||||
|
/// [`crate::mission_runtime::forwarded_provider_env`]). It is sent, never
|
||||||
|
/// logged: this is the only channel by which a secret reaches the guest, and
|
||||||
|
/// the guest refuses the exec rather than running a command without an entry
|
||||||
|
/// it could not honour.
|
||||||
|
pub async fn exec(
|
||||||
|
&self,
|
||||||
|
cmd: &str,
|
||||||
|
cwd: Option<&str>,
|
||||||
|
timeout_secs: u64,
|
||||||
|
env: &[(String, String)],
|
||||||
|
) -> Result<ExecOut, String> {
|
||||||
|
self.exec_attributed(cmd, cwd, timeout_secs, env, None, None)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The same exec, tagged with the run whose live output this is.
|
||||||
|
///
|
||||||
|
/// When `run_id` is set the node follows `log_path` inside the guest for the
|
||||||
|
/// life of the command and streams what it reads to the server. Probes pass
|
||||||
|
/// `None`: they produce nothing worth streaming and have no subscriber.
|
||||||
|
pub async fn exec_attributed(
|
||||||
|
&self,
|
||||||
|
cmd: &str,
|
||||||
|
cwd: Option<&str>,
|
||||||
|
timeout_secs: u64,
|
||||||
|
env: &[(String, String)],
|
||||||
|
run_id: Option<uuid::Uuid>,
|
||||||
|
log_path: Option<&str>,
|
||||||
|
) -> Result<ExecOut, String> {
|
||||||
|
let env: Option<Value> = (!env.is_empty()).then(|| {
|
||||||
|
env.iter()
|
||||||
|
.map(|(k, v)| (k.clone(), Value::String(v.clone())))
|
||||||
|
.collect::<serde_json::Map<_, _>>()
|
||||||
|
.into()
|
||||||
|
});
|
||||||
|
let v = self
|
||||||
|
.call(
|
||||||
|
"vm_exec",
|
||||||
|
json!({
|
||||||
|
"cmd": cmd, "cwd": cwd, "timeout": timeout_secs, "env": env,
|
||||||
|
"run_id": run_id.map(|r| r.to_string()), "log_path": log_path,
|
||||||
|
}),
|
||||||
|
hub_deadline(timeout_secs),
|
||||||
|
)
|
||||||
|
.await?;
|
||||||
|
// A guest that refused to run the command reports `ok: false` and no rc
|
||||||
|
// — a rejected env entry, for instance. Surface its reason: falling
|
||||||
|
// through to the missing-rc error below would hide the cause behind a
|
||||||
|
// symptom.
|
||||||
|
if v.get("ok").and_then(Value::as_bool) == Some(false) {
|
||||||
|
return Err(format!(
|
||||||
|
"vm_exec did not run: {}",
|
||||||
|
v.get("error").and_then(Value::as_str).unwrap_or("unknown")
|
||||||
|
));
|
||||||
|
}
|
||||||
|
// A missing rc is not "success" — it means the guest did not report one,
|
||||||
|
// which we must not read as zero.
|
||||||
|
let rc = v
|
||||||
|
.get("rc")
|
||||||
|
.and_then(Value::as_i64)
|
||||||
|
.ok_or_else(|| format!("vm_exec gave no exit code: {v}"))?;
|
||||||
|
Ok(ExecOut {
|
||||||
|
rc,
|
||||||
|
stdout: v
|
||||||
|
.get("stdout")
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.unwrap_or_default()
|
||||||
|
.to_string(),
|
||||||
|
stderr: v
|
||||||
|
.get("stderr")
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.unwrap_or_default()
|
||||||
|
.to_string(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Tar a path out of the guest and return the archive bytes.
|
||||||
|
/// `exclude` names directories to leave out — build output, caches. Sent from
|
||||||
|
/// here so the policy lives in one place: `mission_fs::transport_excludes`,
|
||||||
|
/// the same list the delivery diff uses. Shipping `target/` blew this call's
|
||||||
|
/// 300s budget twice, each time with the agent's work finished and stranded.
|
||||||
|
pub async fn collect(&self, path: &str, exclude: &[&str]) -> Result<Vec<u8>, String> {
|
||||||
|
use base64::Engine as _;
|
||||||
|
let v = self
|
||||||
|
.call("vm_collect", json!({ "path": path, "exclude": exclude }), 300)
|
||||||
|
.await?;
|
||||||
|
// The guest reports its own `ok`: a missing path is a real failure that
|
||||||
|
// must not come back as an empty archive, which would look exactly like
|
||||||
|
// a run that produced nothing.
|
||||||
|
if v.get("ok").and_then(Value::as_bool) != Some(true) {
|
||||||
|
return Err(format!(
|
||||||
|
"vm_collect {path}: {}",
|
||||||
|
v.get("error").and_then(Value::as_str).unwrap_or("unknown")
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let b64 = v
|
||||||
|
.get("tar_b64")
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.ok_or_else(|| format!("vm_collect {path} returned no archive: {v}"))?;
|
||||||
|
base64::engine::general_purpose::STANDARD
|
||||||
|
.decode(b64)
|
||||||
|
.map_err(|e| format!("vm_collect {path}: undecodable archive: {e}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stop the VM and remove everything it owned. Idempotent.
|
||||||
|
pub async fn destroy(&self) -> Result<Value, String> {
|
||||||
|
self.call("vm_destroy", json!({}), 60).await
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The result of a command that RAN. `rc != 0` is a normal outcome.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub struct ExecOut {
|
||||||
|
pub rc: i64,
|
||||||
|
pub stdout: String,
|
||||||
|
pub stderr: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ExecOut {
|
||||||
|
pub fn ok(&self) -> bool {
|
||||||
|
self.rc == 0
|
||||||
|
}
|
||||||
|
/// One line for a log or an artifact, without dumping a whole build.
|
||||||
|
pub fn summary(&self) -> String {
|
||||||
|
let tail = |s: &str| {
|
||||||
|
s.lines()
|
||||||
|
.rev()
|
||||||
|
.take(3)
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.into_iter()
|
||||||
|
.rev()
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(" | ")
|
||||||
|
};
|
||||||
|
if self.ok() {
|
||||||
|
format!("rc=0 {}", tail(&self.stdout))
|
||||||
|
} else {
|
||||||
|
format!("rc={} {}", self.rc, tail(&self.stderr))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// VMs a node currently holds, so orphans can be reaped.
|
||||||
|
pub async fn list(hub: &NodeHub, node_id: NodeId) -> Result<Vec<String>, String> {
|
||||||
|
let out = hub
|
||||||
|
.call(node_id, "vm_list", json!({}))
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("vm_list on node {node_id:?}: {e}"))?;
|
||||||
|
let body: Value = serde_json::from_str(&out.output)
|
||||||
|
.map_err(|e| format!("vm_list returned unparseable output ({e}): {}", out.output))?;
|
||||||
|
Ok(body
|
||||||
|
.get("vms")
|
||||||
|
.and_then(Value::as_array)
|
||||||
|
.map(|a| {
|
||||||
|
a.iter()
|
||||||
|
.filter_map(|v| v.get("vm_id").and_then(Value::as_str))
|
||||||
|
.map(str::to_string)
|
||||||
|
.collect()
|
||||||
|
})
|
||||||
|
.unwrap_or_default())
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// A command that ran and failed must be distinguishable from one that
|
||||||
|
/// could not be reached. `rc` carries the verdict; `Err` is for transport.
|
||||||
|
#[test]
|
||||||
|
fn a_nonzero_exit_is_an_outcome_not_an_error() {
|
||||||
|
let failed = ExecOut {
|
||||||
|
rc: 3,
|
||||||
|
stdout: String::new(),
|
||||||
|
stderr: "boom\n".into(),
|
||||||
|
};
|
||||||
|
assert!(!failed.ok());
|
||||||
|
assert!(failed.summary().starts_with("rc=3"));
|
||||||
|
assert!(failed.summary().contains("boom"));
|
||||||
|
|
||||||
|
let passed = ExecOut {
|
||||||
|
rc: 0,
|
||||||
|
stdout: "fine\n".into(),
|
||||||
|
stderr: String::new(),
|
||||||
|
};
|
||||||
|
assert!(passed.ok());
|
||||||
|
assert_eq!(passed.summary(), "rc=0 fine");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The summary is for logs, so it must stay short even when a build prints
|
||||||
|
/// thousands of lines — and it must keep the LAST lines, where the error is.
|
||||||
|
#[test]
|
||||||
|
fn the_summary_keeps_the_tail_and_stays_short() {
|
||||||
|
let noisy = ExecOut {
|
||||||
|
rc: 1,
|
||||||
|
stdout: String::new(),
|
||||||
|
stderr: (1..=500)
|
||||||
|
.map(|i| format!("line {i}"))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join("\n"),
|
||||||
|
};
|
||||||
|
let s = noisy.summary();
|
||||||
|
assert!(s.contains("line 500"), "the last line must survive: {s}");
|
||||||
|
assert!(!s.contains("line 400"), "older lines must be dropped: {s}");
|
||||||
|
assert!(s.len() < 200, "summary must stay log-sized, got {}", s.len());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The guest's deadline must fire before the hub's, so a slow command comes
|
||||||
|
/// back as a reported timeout rather than an unexplained transport failure.
|
||||||
|
#[test]
|
||||||
|
fn the_hub_always_outlives_the_guests_own_timeout() {
|
||||||
|
for guest in [0u64, 1, 30, 3600, 86_400] {
|
||||||
|
assert!(
|
||||||
|
hub_deadline(guest) > guest,
|
||||||
|
"hub deadline for {guest}s must exceed it"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// A caller passing a huge budget must not wrap to a tiny timeout, which
|
||||||
|
// would turn a long agent turn into a spurious transport failure.
|
||||||
|
assert!(
|
||||||
|
hub_deadline(u64::MAX) >= u64::MAX - 1,
|
||||||
|
"an extreme budget must saturate, not wrap"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,721 @@
|
|||||||
|
//! The two engines composed — Slice 4.
|
||||||
|
//!
|
||||||
|
//! Engine Z (the ZeroClaw graph in `cm_orchestrator`) owns durability and
|
||||||
|
//! heterogeneity: deterministic planners, per-step checkpoint/resume, a stale-run
|
||||||
|
//! sweep, cancellation, and a different model per node. Engine C (Claude Code in
|
||||||
|
//! a microVM) owns shared context, self-sizing and cheap fan-out. Neither has the
|
||||||
|
//! other's asset, which is why keeping both is a composition rather than a
|
||||||
|
//! compromise.
|
||||||
|
//!
|
||||||
|
//! This module is the join: a [`TurnExecutor`] whose "turn" is a whole
|
||||||
|
//! Claude-Code-in-a-VM session. Because `topology_worker` already dispatches by
|
||||||
|
//! tier, implementing the existing trait inherits the planners, checkpointing,
|
||||||
|
//! reaper, cancellation, `close_finished_phases`, evaluation, capture and
|
||||||
|
//! delivery unchanged. `recursive_exec::SubTopologyExecutor` is the precedent: a
|
||||||
|
//! `run_turn` may be arbitrarily heavy.
|
||||||
|
//!
|
||||||
|
//! # The file-handoff trap
|
||||||
|
//!
|
||||||
|
//! A VM is inject-tar → run → collect-tar → destroy. A graph of per-node VMs with
|
||||||
|
//! **text-only** handoff would silently lose every file an earlier node wrote:
|
||||||
|
//! node 2 would boot from the original checkout, see none of node 1's work, and
|
||||||
|
//! still report success — the exact silent-success shape this project keeps
|
||||||
|
//! paying for.
|
||||||
|
//!
|
||||||
|
//! The answer here is that the mission's **host checkout is the medium**. Every
|
||||||
|
//! node injects from `repo` and collects back over `repo`, so the tree carries
|
||||||
|
//! forward node to node and the last node's tree is what delivery diffs. Two
|
||||||
|
//! properties make that safe rather than lucky:
|
||||||
|
//!
|
||||||
|
//! - `execute_resumable` runs steps strictly **sequentially**, so two VMs are
|
||||||
|
//! never writing the same host directory at once;
|
||||||
|
//! - the vm id is deterministic per (phase, iteration, step), so a resumed step
|
||||||
|
//! whose VM is somehow still alive is refused by the node ("vm already exists")
|
||||||
|
//! instead of quietly producing a second writer.
|
||||||
|
//!
|
||||||
|
//! `a_later_node_sees_an_earlier_nodes_files` proves the handoff, and
|
||||||
|
//! `text_only_handoff_loses_the_earlier_nodes_work` is its negative control.
|
||||||
|
//!
|
||||||
|
//! # Keeping a long turn alive
|
||||||
|
//!
|
||||||
|
//! `requeue_stale` requeues a `running` job that has not touched `updated_at` in
|
||||||
|
//! 180 seconds, and one node here can run for an hour. `SubTopologyExecutor`
|
||||||
|
//! keeps its parent alive from each *leaf step*, which it has and this does not:
|
||||||
|
//! there is nothing between the start and end of a VM turn. So the turn holds a
|
||||||
|
//! ticker that touches `updated_at` every [`KEEPALIVE_SECS`] and is aborted on
|
||||||
|
//! drop. Without it a healthy composed run is requeued mid-node, claimed again,
|
||||||
|
//! and boots a second VM against the same checkout.
|
||||||
|
|
||||||
|
use std::path::PathBuf;
|
||||||
|
use std::sync::Arc;
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
use cm_domain::NodeId;
|
||||||
|
use cm_orchestrator::{OrchestratorError, TurnExecutor, TurnOutcome, TurnRequest};
|
||||||
|
use sqlx::PgPool;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
use crate::microvm_executor::{PhaseVm, VmPhase};
|
||||||
|
|
||||||
|
/// How often a running VM turn touches its run's `updated_at`.
|
||||||
|
///
|
||||||
|
/// Comfortably inside the 180s stale window, and cheap: one UPDATE per node per
|
||||||
|
/// half minute against a row nothing else is writing.
|
||||||
|
const KEEPALIVE_SECS: u64 = 30;
|
||||||
|
|
||||||
|
/// A [`TurnExecutor`] that runs each graph node as a full Claude-Code session
|
||||||
|
/// inside its own microVM, against the mission's shared host checkout.
|
||||||
|
pub struct MicroVmTurnExecutor<V: PhaseVm> {
|
||||||
|
vms: V,
|
||||||
|
pool: PgPool,
|
||||||
|
/// The durable outer run. Touched for keepalive; its status gates the turn.
|
||||||
|
run_id: Uuid,
|
||||||
|
mission_id: Uuid,
|
||||||
|
phase_id: Uuid,
|
||||||
|
iteration: i32,
|
||||||
|
/// The mission's host checkout — injected into every node's VM and collected
|
||||||
|
/// back over, which is how file work survives a node boundary.
|
||||||
|
repo: PathBuf,
|
||||||
|
/// Whether the mission has a repository. Carried so every graph node gets
|
||||||
|
/// the same workspace treatment as a solo phase — see `VmPhase::has_repo`.
|
||||||
|
has_repo: bool,
|
||||||
|
/// `missions.target_node_id`: the fleet node a mission was placed on. A node
|
||||||
|
/// may override it with `attrs["node_id"]`.
|
||||||
|
default_fleet_node: Option<Uuid>,
|
||||||
|
/// `missions.backend`: which rootfs image. A node may override it with
|
||||||
|
/// `attrs["backend"]`, which is what makes a graph heterogeneous — a
|
||||||
|
/// `validator` node on a different provider's image is then a first-class
|
||||||
|
/// graph node rather than a bolt-on.
|
||||||
|
default_backend: Option<String>,
|
||||||
|
/// `missions.team_engine`, passed through so a composed node can itself ask
|
||||||
|
/// for Claude Code fan-out inside its VM.
|
||||||
|
team_engine: Option<String>,
|
||||||
|
/// The phase's completion gate, enforced inside every node's VM.
|
||||||
|
gate: Option<crate::vm_stop_gate::StopGate>,
|
||||||
|
/// Which step is next. `execute_resumable` is sequential and gives the
|
||||||
|
/// executor no index, so the executor counts — and the count starts from the
|
||||||
|
/// checkpoint on resume, or two VMs would share an id across a restart.
|
||||||
|
step: std::sync::atomic::AtomicU32,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Everything a composed run needs that is not the graph itself.
|
||||||
|
pub struct ComposedRun {
|
||||||
|
pub run_id: Uuid,
|
||||||
|
pub mission_id: Uuid,
|
||||||
|
pub phase_id: Uuid,
|
||||||
|
pub iteration: i32,
|
||||||
|
pub repo: PathBuf,
|
||||||
|
pub has_repo: bool,
|
||||||
|
pub target_node_id: Option<Uuid>,
|
||||||
|
pub backend: Option<String>,
|
||||||
|
pub team_engine: Option<String>,
|
||||||
|
/// What must hold before a node's agent may stop. See [`crate::vm_stop_gate`].
|
||||||
|
pub gate: Option<crate::vm_stop_gate::StopGate>,
|
||||||
|
/// Steps already completed, from the durable checkpoint. Nonzero on resume.
|
||||||
|
pub completed_steps: u32,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<V: PhaseVm> MicroVmTurnExecutor<V> {
|
||||||
|
pub fn new(vms: V, pool: PgPool, r: ComposedRun) -> Self {
|
||||||
|
Self {
|
||||||
|
vms,
|
||||||
|
pool,
|
||||||
|
run_id: r.run_id,
|
||||||
|
mission_id: r.mission_id,
|
||||||
|
phase_id: r.phase_id,
|
||||||
|
iteration: r.iteration,
|
||||||
|
repo: r.repo,
|
||||||
|
has_repo: r.has_repo,
|
||||||
|
default_fleet_node: r.target_node_id,
|
||||||
|
default_backend: r.backend,
|
||||||
|
team_engine: r.team_engine,
|
||||||
|
gate: r.gate,
|
||||||
|
step: std::sync::atomic::AtomicU32::new(r.completed_steps),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Which fleet node this graph node runs on.
|
||||||
|
///
|
||||||
|
/// Fail-closed on a malformed override: placing a node on the mission's node
|
||||||
|
/// because its own `node_id` did not parse would run the work somewhere the
|
||||||
|
/// graph did not ask for and say nothing.
|
||||||
|
fn fleet_node(&self, req: &TurnRequest) -> Result<NodeId, OrchestratorError> {
|
||||||
|
let id = match req.attrs.get("node_id") {
|
||||||
|
Some(raw) => Uuid::parse_str(raw.trim()).map_err(|_| {
|
||||||
|
OrchestratorError::Executor(format!(
|
||||||
|
"node {} has an invalid node_id attr: {raw}",
|
||||||
|
req.node_id
|
||||||
|
))
|
||||||
|
})?,
|
||||||
|
None => self.default_fleet_node.ok_or_else(|| {
|
||||||
|
OrchestratorError::Executor(format!(
|
||||||
|
"node {} has no node_id attr and the mission has no \
|
||||||
|
target_node_id — a microVM node cannot run on the gateway, \
|
||||||
|
which has no /dev/kvm",
|
||||||
|
req.node_id
|
||||||
|
))
|
||||||
|
})?,
|
||||||
|
};
|
||||||
|
Ok(NodeId::from(id))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<V: PhaseVm> TurnExecutor for MicroVmTurnExecutor<V> {
|
||||||
|
async fn run_turn(&self, req: TurnRequest) -> Result<TurnOutcome, OrchestratorError> {
|
||||||
|
let fleet_node = self.fleet_node(&req)?;
|
||||||
|
let backend = req
|
||||||
|
.attrs
|
||||||
|
.get("backend")
|
||||||
|
.map(|s| s.trim().to_string())
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
.or_else(|| self.default_backend.clone());
|
||||||
|
|
||||||
|
if !self.repo.is_dir() {
|
||||||
|
return Err(OrchestratorError::Executor(format!(
|
||||||
|
"mission has no checkout at {} — a composed node needs the \
|
||||||
|
repository, and it is also how the previous node's work reaches \
|
||||||
|
this one",
|
||||||
|
self.repo.display()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
|
||||||
|
let step = self
|
||||||
|
.step
|
||||||
|
.fetch_add(1, std::sync::atomic::Ordering::SeqCst);
|
||||||
|
|
||||||
|
// Held for the length of the VM turn: an hour of silence would otherwise
|
||||||
|
// look exactly like a dead worker to `requeue_stale`.
|
||||||
|
let _alive = Keepalive::spawn(self.pool.clone(), self.run_id);
|
||||||
|
|
||||||
|
let task = node_task_text(&req);
|
||||||
|
let outcome = self
|
||||||
|
.vms
|
||||||
|
.run(VmPhase {
|
||||||
|
// Every node of a composed graph streams to the same outer run,
|
||||||
|
// which is the one the operator is watching.
|
||||||
|
run_id: Some(self.run_id),
|
||||||
|
node_id: fleet_node,
|
||||||
|
mission_id: self.mission_id,
|
||||||
|
phase_id: self.phase_id,
|
||||||
|
iteration: self.iteration,
|
||||||
|
task: &task,
|
||||||
|
backend: backend.as_deref(),
|
||||||
|
repo: &self.repo,
|
||||||
|
has_repo: self.has_repo,
|
||||||
|
team_engine: self.team_engine.as_deref(),
|
||||||
|
// Each node is its own agent session, so each carries the
|
||||||
|
// phase's gate. Threaded from the run rather than rebuilt here:
|
||||||
|
// one source for what "done" means, whichever executor asks.
|
||||||
|
gate: self.gate.as_ref(),
|
||||||
|
step: Some(step),
|
||||||
|
// Same live drain as the solo path. A composed graph node can
|
||||||
|
// run for an hour too, and its files are the only account of
|
||||||
|
// what it did until the next node collects.
|
||||||
|
tap_sink: Some(crate::phase_runner::vm_tool_recorder(
|
||||||
|
&self.pool,
|
||||||
|
self.mission_id,
|
||||||
|
self.phase_id,
|
||||||
|
self.run_id,
|
||||||
|
)),
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
OrchestratorError::Executor(format!("node {} in a microVM: {e}", req.node_id))
|
||||||
|
})?;
|
||||||
|
|
||||||
|
// Recorded BEFORE the failure branches below. A node that could not be
|
||||||
|
// collected, or whose gate capped, still touched files — and on this
|
||||||
|
// path those touches are the only account of what it did, since the
|
||||||
|
// work never reached a diff.
|
||||||
|
crate::phase_runner::record_vm_tools(
|
||||||
|
&self.pool,
|
||||||
|
self.mission_id,
|
||||||
|
self.phase_id,
|
||||||
|
self.run_id,
|
||||||
|
&outcome.tools,
|
||||||
|
// No turn agents supplied, so nothing is attributed — the same
|
||||||
|
// `agent_id: None` this path has always written. Resolving the
|
||||||
|
// graph node to an agent uuid is the fix, and it cannot be tested
|
||||||
|
// while the fleet is offline; guessing at it here would put one
|
||||||
|
// node's actions on another node's record.
|
||||||
|
&[],
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
// A node whose work never came back must fail the run rather than hand
|
||||||
|
// the next node a tree missing the previous one's edits. On this path an
|
||||||
|
// uncollected turn is worse than on the solo one: the loss is silent,
|
||||||
|
// because the next node still boots from a checkout that looks fine.
|
||||||
|
if !outcome.collected {
|
||||||
|
return Err(OrchestratorError::Executor(format!(
|
||||||
|
"node {}'s work could not be collected from its VM, so the next \
|
||||||
|
node would not see it: {}",
|
||||||
|
req.node_id,
|
||||||
|
outcome.summary.chars().take(400).collect::<String>()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
// Same rule as the solo path: the gate is the only thing that runs a
|
||||||
|
// `done_when_check`, so a release at the cap must fail the run rather
|
||||||
|
// than hand the next node a tree that does not satisfy the condition
|
||||||
|
// every node in this graph was told to satisfy.
|
||||||
|
if outcome.released_at_cap == Some(true) {
|
||||||
|
return Err(OrchestratorError::Executor(format!(
|
||||||
|
"node {}'s completion gate released it after {} refusal(s) with its check \
|
||||||
|
still failing: {}",
|
||||||
|
req.node_id,
|
||||||
|
crate::vm_stop_gate::MAX_BLOCKS,
|
||||||
|
outcome.summary.chars().take(400).collect::<String>()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
if outcome.rc != 0 {
|
||||||
|
return Err(OrchestratorError::Executor(format!(
|
||||||
|
"node {} exited {}: {}",
|
||||||
|
req.node_id,
|
||||||
|
outcome.rc,
|
||||||
|
outcome.summary.chars().take(400).collect::<String>()
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
|
||||||
|
eprintln!(
|
||||||
|
"microvm_turn_executor: run {} node {} (role {}, step {}) ok — subagents: {}",
|
||||||
|
self.run_id,
|
||||||
|
req.node_id,
|
||||||
|
req.role,
|
||||||
|
step,
|
||||||
|
outcome
|
||||||
|
.subagents
|
||||||
|
.map(|n| n.to_string())
|
||||||
|
.unwrap_or_else(|| "?".into()),
|
||||||
|
);
|
||||||
|
|
||||||
|
Ok(TurnOutcome {
|
||||||
|
output: outcome.summary,
|
||||||
|
// `claude -p` does not report token usage on stdout, and inventing a
|
||||||
|
// number here would corrupt the run totals the harness reads. Zero is
|
||||||
|
// the honest value for "not measured on this path".
|
||||||
|
tokens: 0,
|
||||||
|
gated: Vec::new(),
|
||||||
|
spend: Default::default(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What one graph node is told.
|
||||||
|
///
|
||||||
|
/// The upstream outputs are included as context, but the load-bearing sentence is
|
||||||
|
/// that the previous node's *files* are already in the tree: a node told only
|
||||||
|
/// about the text would re-do work it is standing on.
|
||||||
|
fn node_task_text(req: &TurnRequest) -> String {
|
||||||
|
let mut s = format!(
|
||||||
|
"You are the `{}` stage of a multi-stage mission.\n\nMISSION TASK\n{}\n",
|
||||||
|
req.role, req.task
|
||||||
|
);
|
||||||
|
if !req.context.is_empty() {
|
||||||
|
s.push_str(
|
||||||
|
"\nWHAT CAME BEFORE\nThe earlier stages' work is ALREADY IN THIS \
|
||||||
|
WORKING TREE — the repository you have been given is their output, \
|
||||||
|
not a fresh checkout. Read the files before changing them, and do \
|
||||||
|
not redo what is already done. Their closing reports:\n",
|
||||||
|
);
|
||||||
|
for (i, c) in req.context.iter().enumerate() {
|
||||||
|
s.push_str(&format!("\n--- stage {} ---\n{}\n", i + 1, c));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
s
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Touches a run's `updated_at` until dropped.
|
||||||
|
struct Keepalive(tokio::task::JoinHandle<()>);
|
||||||
|
|
||||||
|
impl Keepalive {
|
||||||
|
fn spawn(pool: PgPool, run_id: Uuid) -> Self {
|
||||||
|
Keepalive(tokio::spawn(async move {
|
||||||
|
let mut ticker = tokio::time::interval(Duration::from_secs(KEEPALIVE_SECS));
|
||||||
|
loop {
|
||||||
|
ticker.tick().await;
|
||||||
|
let _ = cm_db::repo::topology_runs::touch(&pool, run_id).await;
|
||||||
|
}
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for Keepalive {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
self.0.abort();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build the executor the worker uses, over real VMs on the fleet.
|
||||||
|
pub fn for_fleet(
|
||||||
|
hub: Arc<crate::fleet::NodeHub>,
|
||||||
|
pool: PgPool,
|
||||||
|
r: ComposedRun,
|
||||||
|
) -> MicroVmTurnExecutor<crate::microvm_executor::HubVms> {
|
||||||
|
MicroVmTurnExecutor::new(crate::microvm_executor::HubVms::new(hub), pool, r)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::microvm_executor::VmOutcome;
|
||||||
|
use std::collections::BTreeMap;
|
||||||
|
use std::sync::Mutex;
|
||||||
|
|
||||||
|
/// A VM modelled honestly: the host tree is packed in, the "agent" works on a
|
||||||
|
/// COPY that no host path points at, and the result is unpacked back over the
|
||||||
|
/// host tree. That is the real inject → run → collect shape, which is what
|
||||||
|
/// makes the negative control below meaningful — remove the collect and the
|
||||||
|
/// handoff breaks exactly as it would in production.
|
||||||
|
struct FakeVms {
|
||||||
|
/// Whether the guest's tree is collected back to the host.
|
||||||
|
collect: bool,
|
||||||
|
/// vm ids used, in order — the id is what stops two nodes colliding.
|
||||||
|
ids: Mutex<Vec<String>>,
|
||||||
|
/// (backend, fleet node) per call, for the heterogeneity assertions.
|
||||||
|
placements: Mutex<Vec<(Option<String>, NodeId)>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl FakeVms {
|
||||||
|
fn new(collect: bool) -> Self {
|
||||||
|
Self {
|
||||||
|
collect,
|
||||||
|
ids: Mutex::new(Vec::new()),
|
||||||
|
placements: Mutex::new(Vec::new()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl PhaseVm for FakeVms {
|
||||||
|
async fn run(&self, p: VmPhase<'_>) -> Result<VmOutcome, String> {
|
||||||
|
self.ids
|
||||||
|
.lock()
|
||||||
|
.unwrap()
|
||||||
|
.push(format!("{}-{:?}", p.phase_id.simple(), p.step));
|
||||||
|
self.placements
|
||||||
|
.lock()
|
||||||
|
.unwrap()
|
||||||
|
.push((p.backend.map(str::to_string), p.node_id));
|
||||||
|
|
||||||
|
// inject: the host checkout goes in as a tar.
|
||||||
|
let tar = crate::mission_fs::pack_dir(p.repo, "repo")?;
|
||||||
|
let guest = tempfile::tempdir().map_err(|e| e.to_string())?;
|
||||||
|
crate::mission_fs::unpack_into(&tar, guest.path())?;
|
||||||
|
let guest_repo = guest.path().join("repo");
|
||||||
|
|
||||||
|
// run: the agent records that it was here, and reports what it found
|
||||||
|
// of the previous stages — the observation the handoff test reads.
|
||||||
|
let seen: Vec<String> = std::fs::read_dir(&guest_repo)
|
||||||
|
.map_err(|e| e.to_string())?
|
||||||
|
.filter_map(|e| e.ok())
|
||||||
|
.map(|e| e.file_name().to_string_lossy().to_string())
|
||||||
|
.filter(|n| n.starts_with("stage-"))
|
||||||
|
.collect();
|
||||||
|
let mine = guest_repo.join(format!("stage-{}.txt", p.step.unwrap_or(0)));
|
||||||
|
std::fs::write(&mine, "work").map_err(|e| e.to_string())?;
|
||||||
|
|
||||||
|
// collect: the guest tree comes back over the same host path.
|
||||||
|
if self.collect {
|
||||||
|
let back = crate::mission_fs::pack_dir(&guest_repo, "repo")?;
|
||||||
|
let parent = p.repo.parent().ok_or("no parent")?;
|
||||||
|
crate::mission_fs::unpack_into(&back, parent)?;
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(VmOutcome {
|
||||||
|
summary: format!("saw:[{}]", seen.join(",")),
|
||||||
|
rc: 0,
|
||||||
|
collected: true,
|
||||||
|
subagents: Some(0),
|
||||||
|
teammates: None,
|
||||||
|
stop_blocks: None,
|
||||||
|
released_at_cap: None,
|
||||||
|
tools: Vec::new(),
|
||||||
|
rootfs: None,
|
||||||
|
cli_version: None,
|
||||||
|
tool_gate: None,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn req(node: &str, role: &str, context: Vec<String>) -> TurnRequest {
|
||||||
|
TurnRequest {
|
||||||
|
node_id: node.into(),
|
||||||
|
role: role.into(),
|
||||||
|
agent: None,
|
||||||
|
attrs: BTreeMap::new(),
|
||||||
|
task: "build the thing".into(),
|
||||||
|
context,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn exec<V: PhaseVm>(vms: V, repo: PathBuf) -> MicroVmTurnExecutor<V> {
|
||||||
|
// A pool that is never connected: every test here fails the turn before
|
||||||
|
// any query, or drives one whose only DB touch is the best-effort
|
||||||
|
// keepalive (which swallows its own errors by design).
|
||||||
|
let pool = sqlx::postgres::PgPoolOptions::new()
|
||||||
|
.max_connections(1)
|
||||||
|
.connect_lazy("postgres://invalid/invalid")
|
||||||
|
.expect("a lazy pool never dials");
|
||||||
|
MicroVmTurnExecutor::new(
|
||||||
|
vms,
|
||||||
|
pool,
|
||||||
|
ComposedRun {
|
||||||
|
run_id: Uuid::now_v7(),
|
||||||
|
mission_id: Uuid::now_v7(),
|
||||||
|
phase_id: Uuid::now_v7(),
|
||||||
|
iteration: 1,
|
||||||
|
repo,
|
||||||
|
has_repo: true,
|
||||||
|
target_node_id: Some(Uuid::now_v7()),
|
||||||
|
backend: Some("claude".into()),
|
||||||
|
team_engine: None,
|
||||||
|
gate: None,
|
||||||
|
completed_steps: 0,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn a_checkout() -> tempfile::TempDir {
|
||||||
|
let d = tempfile::tempdir().unwrap();
|
||||||
|
std::fs::create_dir_all(d.path().join("repo")).unwrap();
|
||||||
|
std::fs::write(d.path().join("repo").join("README.md"), "hello").unwrap();
|
||||||
|
d
|
||||||
|
}
|
||||||
|
|
||||||
|
/// THE trap this slice exists to solve. A per-node VM is destroyed with its
|
||||||
|
/// filesystem, so unless the tree is carried forward, node 2 works from the
|
||||||
|
/// original checkout and silently loses node 1's edits — while still
|
||||||
|
/// reporting success.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn a_later_node_sees_an_earlier_nodes_files() {
|
||||||
|
let d = a_checkout();
|
||||||
|
let e = exec(FakeVms::new(true), d.path().join("repo"));
|
||||||
|
|
||||||
|
let first = e.run_turn(req("n1", "implementer", vec![])).await.unwrap();
|
||||||
|
assert_eq!(first.output, "saw:[]", "the first node starts clean");
|
||||||
|
|
||||||
|
let second = e
|
||||||
|
.run_turn(req("n2", "verifier", vec![first.output.clone()]))
|
||||||
|
.await
|
||||||
|
.unwrap();
|
||||||
|
assert!(
|
||||||
|
second.output.contains("stage-0.txt"),
|
||||||
|
"node 2 could not see node 1's file: {}",
|
||||||
|
second.output
|
||||||
|
);
|
||||||
|
// And the host tree — what delivery diffs — holds both nodes' work.
|
||||||
|
for f in ["stage-0.txt", "stage-1.txt"] {
|
||||||
|
assert!(d.path().join("repo").join(f).exists(), "{f} missing on the host");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The negative control, run rather than assumed: with the collect removed —
|
||||||
|
/// i.e. a text-only handoff between nodes — the test above fails. A guard
|
||||||
|
/// that cannot detect the bug it was written for is decoration.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn text_only_handoff_loses_the_earlier_nodes_work() {
|
||||||
|
let d = a_checkout();
|
||||||
|
let e = exec(FakeVms::new(false), d.path().join("repo"));
|
||||||
|
|
||||||
|
e.run_turn(req("n1", "implementer", vec![])).await.unwrap();
|
||||||
|
let second = e.run_turn(req("n2", "verifier", vec![])).await.unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
second.output, "saw:[]",
|
||||||
|
"without a collect, node 2 must NOT see node 1's work — if it does, \
|
||||||
|
this test is no longer controlling anything"
|
||||||
|
);
|
||||||
|
assert!(!d.path().join("repo").join("stage-0.txt").exists());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Each node gets its own vm id within one phase and iteration. Two nodes
|
||||||
|
/// sharing an id means the second is refused by the fleet node while the
|
||||||
|
/// first is alive, and indistinguishable from a re-run once it is not.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn every_node_runs_in_its_own_vm() {
|
||||||
|
let d = a_checkout();
|
||||||
|
let vms = FakeVms::new(true);
|
||||||
|
let e = exec(vms, d.path().join("repo"));
|
||||||
|
for n in ["n1", "n2", "n3"] {
|
||||||
|
e.run_turn(req(n, "worker", vec![])).await.unwrap();
|
||||||
|
}
|
||||||
|
let ids = e.vms.ids.lock().unwrap().clone();
|
||||||
|
let unique: std::collections::HashSet<_> = ids.iter().collect();
|
||||||
|
assert_eq!(unique.len(), ids.len(), "{ids:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resume must not re-use a completed step's vm id. The executor counts steps
|
||||||
|
/// itself, so the count has to start where the checkpoint left off.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn a_resumed_run_continues_the_step_numbering() {
|
||||||
|
let d = a_checkout();
|
||||||
|
let pool = sqlx::postgres::PgPoolOptions::new()
|
||||||
|
.max_connections(1)
|
||||||
|
.connect_lazy("postgres://invalid/invalid")
|
||||||
|
.unwrap();
|
||||||
|
let e = MicroVmTurnExecutor::new(
|
||||||
|
FakeVms::new(true),
|
||||||
|
pool,
|
||||||
|
ComposedRun {
|
||||||
|
run_id: Uuid::now_v7(),
|
||||||
|
mission_id: Uuid::now_v7(),
|
||||||
|
phase_id: Uuid::now_v7(),
|
||||||
|
iteration: 1,
|
||||||
|
repo: d.path().join("repo"),
|
||||||
|
has_repo: true,
|
||||||
|
target_node_id: Some(Uuid::now_v7()),
|
||||||
|
backend: None,
|
||||||
|
team_engine: None,
|
||||||
|
gate: None,
|
||||||
|
completed_steps: 2,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
e.run_turn(req("n3", "worker", vec![])).await.unwrap();
|
||||||
|
let ids = e.vms.ids.lock().unwrap().clone();
|
||||||
|
assert!(
|
||||||
|
ids[0].ends_with("Some(2)"),
|
||||||
|
"the first step after a resume must be step 2, not 0: {ids:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Per-node `backend` is what makes the outer graph heterogeneous — a
|
||||||
|
/// validator node on another provider's image. It must override the
|
||||||
|
/// mission's, and the mission's must still apply to nodes that say nothing.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn a_node_may_pick_its_own_backend_and_fleet_node() {
|
||||||
|
let d = a_checkout();
|
||||||
|
let e = exec(FakeVms::new(true), d.path().join("repo"));
|
||||||
|
let elsewhere = Uuid::now_v7();
|
||||||
|
|
||||||
|
let mut r = req("n1", "worker", vec![]);
|
||||||
|
r.attrs.insert("backend".into(), "glm".into());
|
||||||
|
r.attrs.insert("node_id".into(), elsewhere.to_string());
|
||||||
|
e.run_turn(r).await.unwrap();
|
||||||
|
e.run_turn(req("n2", "worker", vec![])).await.unwrap();
|
||||||
|
|
||||||
|
let p = e.vms.placements.lock().unwrap().clone();
|
||||||
|
assert_eq!(p[0].0.as_deref(), Some("glm"));
|
||||||
|
assert_eq!(p[0].1, NodeId::from(elsewhere));
|
||||||
|
assert_eq!(p[1].0.as_deref(), Some("claude"), "the mission default");
|
||||||
|
assert_ne!(p[1].1, NodeId::from(elsewhere));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A malformed `node_id` must fail the node, not fall back to the mission's.
|
||||||
|
/// Silently running work somewhere the graph did not ask for is the same
|
||||||
|
/// class of bug as an alias that serde dropped.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn a_malformed_node_placement_fails_closed() {
|
||||||
|
let d = a_checkout();
|
||||||
|
let e = exec(FakeVms::new(true), d.path().join("repo"));
|
||||||
|
let mut r = req("n1", "worker", vec![]);
|
||||||
|
r.attrs.insert("node_id".into(), "not-a-uuid".into());
|
||||||
|
let err = e.run_turn(r).await.unwrap_err().to_string();
|
||||||
|
assert!(err.contains("invalid node_id"), "{err}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An uncollected node is a failed run here, not a warning: the next node
|
||||||
|
/// would boot from a tree that looks fine and is missing this node's work.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn an_uncollected_node_fails_the_run() {
|
||||||
|
struct Lost;
|
||||||
|
impl PhaseVm for Lost {
|
||||||
|
async fn run(&self, _p: VmPhase<'_>) -> Result<VmOutcome, String> {
|
||||||
|
Ok(VmOutcome {
|
||||||
|
summary: "did plenty".into(),
|
||||||
|
rc: 0,
|
||||||
|
collected: false,
|
||||||
|
subagents: None,
|
||||||
|
teammates: None,
|
||||||
|
stop_blocks: None,
|
||||||
|
released_at_cap: None,
|
||||||
|
tools: Vec::new(),
|
||||||
|
rootfs: None,
|
||||||
|
cli_version: None,
|
||||||
|
tool_gate: None,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let d = a_checkout();
|
||||||
|
let e = exec(Lost, d.path().join("repo"));
|
||||||
|
let err = e.run_turn(req("n1", "worker", vec![])).await.unwrap_err().to_string();
|
||||||
|
assert!(err.contains("could not be collected"), "{err}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A node whose gate gave up is a FAILED run, not a completed one.
|
||||||
|
///
|
||||||
|
/// The gate is the only thing in the system that ever runs a
|
||||||
|
/// `done_when_check`. If it releases the agent at the cap and this returns
|
||||||
|
/// Ok, the check's failure is never seen again: the node reports success,
|
||||||
|
/// the next node builds on a tree that does not satisfy the condition, and
|
||||||
|
/// the phase completes green. `rc` is 0 and the work IS collected here on
|
||||||
|
/// purpose — those are the two signals that used to decide this, and both
|
||||||
|
/// say "fine".
|
||||||
|
#[tokio::test]
|
||||||
|
async fn a_node_whose_gate_gave_up_fails_the_run() {
|
||||||
|
struct Capped;
|
||||||
|
impl PhaseVm for Capped {
|
||||||
|
async fn run(&self, _p: VmPhase<'_>) -> Result<VmOutcome, String> {
|
||||||
|
Ok(VmOutcome {
|
||||||
|
summary: "I could not get the tests passing, but here is what I did".into(),
|
||||||
|
rc: 0,
|
||||||
|
collected: true,
|
||||||
|
subagents: None,
|
||||||
|
teammates: None,
|
||||||
|
stop_blocks: Some(crate::vm_stop_gate::MAX_BLOCKS),
|
||||||
|
released_at_cap: Some(true),
|
||||||
|
tools: Vec::new(),
|
||||||
|
rootfs: None,
|
||||||
|
cli_version: None,
|
||||||
|
tool_gate: None,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let d = a_checkout();
|
||||||
|
let e = exec(Capped, d.path().join("repo"));
|
||||||
|
let err = e.run_turn(req("n1", "worker", vec![])).await.unwrap_err().to_string();
|
||||||
|
assert!(err.contains("released it after"), "{err}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The negative control: the SAME number of blocks, without the cap. An
|
||||||
|
/// agent that was refused three times and then got it right on the fourth
|
||||||
|
/// try has succeeded, and reports `blocks: 3` exactly like the test above.
|
||||||
|
/// Failing on the count instead of the mark would fail this healthy run.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn a_node_that_was_blocked_and_then_succeeded_passes() {
|
||||||
|
struct Recovered;
|
||||||
|
impl PhaseVm for Recovered {
|
||||||
|
async fn run(&self, _p: VmPhase<'_>) -> Result<VmOutcome, String> {
|
||||||
|
Ok(VmOutcome {
|
||||||
|
summary: "took me a few tries".into(),
|
||||||
|
rc: 0,
|
||||||
|
collected: true,
|
||||||
|
subagents: None,
|
||||||
|
teammates: None,
|
||||||
|
stop_blocks: Some(crate::vm_stop_gate::MAX_BLOCKS),
|
||||||
|
released_at_cap: Some(false),
|
||||||
|
tools: Vec::new(),
|
||||||
|
rootfs: None,
|
||||||
|
cli_version: None,
|
||||||
|
tool_gate: None,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let d = a_checkout();
|
||||||
|
let e = exec(Recovered, d.path().join("repo"));
|
||||||
|
e.run_turn(req("n1", "worker", vec![]))
|
||||||
|
.await
|
||||||
|
.expect("a run that recovered inside its own turn is a success");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A node must be told its predecessors' files are already in the tree.
|
||||||
|
/// Given only the text, an agent re-does work it is standing on.
|
||||||
|
#[test]
|
||||||
|
fn a_downstream_node_is_told_the_work_is_already_in_the_tree() {
|
||||||
|
let solo = node_task_text(&req("n1", "implementer", vec![]));
|
||||||
|
assert!(solo.contains("build the thing"));
|
||||||
|
assert!(!solo.contains("WHAT CAME BEFORE"), "{solo}");
|
||||||
|
|
||||||
|
let later = node_task_text(&req("n2", "verifier", vec!["I wrote foo.rs".into()]));
|
||||||
|
assert!(later.contains("ALREADY IN THIS WORKING TREE"), "{later}");
|
||||||
|
assert!(later.contains("I wrote foo.rs"), "{later}");
|
||||||
|
assert!(later.contains("verifier"), "the node's role: {later}");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -40,7 +40,14 @@ use crate::mission_workspace;
|
|||||||
/// coding phase that ran `cargo build` leaves a `target/` directory larger
|
/// coding phase that ran `cargo build` leaves a `target/` directory larger
|
||||||
/// than most repositories, and a patch containing it is unreadable as well as
|
/// than most repositories, and a patch containing it is unreadable as well as
|
||||||
/// enormous.
|
/// enormous.
|
||||||
const EXCLUDED_PATHS: &[&str] = &[
|
///
|
||||||
|
/// `mission_fs` uses this same list for the TRANSPORT, and that is not a
|
||||||
|
/// convenience — it is the fix for a real failure. The diff excluded `target/`
|
||||||
|
/// while the tar that carried the tree in and out did not, so a phase that ran
|
||||||
|
/// `cargo test` shipped its whole build directory over vsock twice. `vm_collect`
|
||||||
|
/// timed out at 300s on mission 019fd43e with the agent's work finished and
|
||||||
|
/// stranded inside a VM. Two layers, one list.
|
||||||
|
pub(crate) const EXCLUDED_PATHS: &[&str] = &[
|
||||||
"target",
|
"target",
|
||||||
"node_modules",
|
"node_modules",
|
||||||
".venv",
|
".venv",
|
||||||
@@ -66,6 +73,10 @@ const EXCLUDED_PATHS: &[&str] = &[
|
|||||||
/// generated or vendored got committed), and the head of it is what an
|
/// generated or vendored got committed), and the head of it is what an
|
||||||
/// operator needs to see to work out what happened.
|
/// operator needs to see to work out what happened.
|
||||||
const MAX_PATCH_BYTES: usize = 4 * 1024 * 1024;
|
const MAX_PATCH_BYTES: usize = 4 * 1024 * 1024;
|
||||||
|
/// Cap on the recorded path list. A cap that silently truncates is worse than
|
||||||
|
/// no cap, so the metadata carries `files_truncated` beside it — a reader must
|
||||||
|
/// be able to tell "touched 12 files" from "touched at least 500".
|
||||||
|
const MAX_CAPTURED_PATHS: usize = 500;
|
||||||
|
|
||||||
/// Who delivery commits as.
|
/// Who delivery commits as.
|
||||||
///
|
///
|
||||||
@@ -109,7 +120,18 @@ pub struct Capture {
|
|||||||
/// The phase changed nothing. Still recorded — "this coding phase wrote no
|
/// The phase changed nothing. Still recorded — "this coding phase wrote no
|
||||||
/// code" is currently invisible to an operator, and it is worth saying.
|
/// code" is currently invisible to an operator, and it is worth saying.
|
||||||
pub empty: bool,
|
pub empty: bool,
|
||||||
|
/// Why the diff could not be computed, if it could not be. `empty` is only
|
||||||
|
/// meaningful when this is `None`: otherwise the tree was never read, and
|
||||||
|
/// callers deciding anything on the strength of "no changes" must not.
|
||||||
|
pub diff_error: Option<String>,
|
||||||
pub truncated: bool,
|
pub truncated: bool,
|
||||||
|
/// The paths this phase touched, with their `--name-status` letter. The
|
||||||
|
/// diffstat gives counts only; this is what lets anything downstream say
|
||||||
|
/// WHICH files changed.
|
||||||
|
pub files: Vec<(char, String)>,
|
||||||
|
/// The path list hit `MAX_CAPTURED_PATHS`. Recorded so a reader can tell a
|
||||||
|
/// complete list from a clipped one.
|
||||||
|
pub files_truncated: bool,
|
||||||
pub patch_path: PathBuf,
|
pub patch_path: PathBuf,
|
||||||
/// Set once the work has been committed to a mission branch.
|
/// Set once the work has been committed to a mission branch.
|
||||||
pub committed: Option<Commit>,
|
pub committed: Option<Commit>,
|
||||||
@@ -227,19 +249,74 @@ pub async fn capture_phase_diff_at(
|
|||||||
// A repo with nothing to add is fine; keep going and let the diff be empty.
|
// A repo with nothing to add is fine; keep going and let the diff be empty.
|
||||||
let _ = git(&repo, &add).await;
|
let _ = git(&repo, &add).await;
|
||||||
|
|
||||||
|
// A failed `git diff` and a phase that changed nothing both yield an empty
|
||||||
|
// string, and `unwrap_or_default` used to erase the difference: a corrupt
|
||||||
|
// index or an unreadable base would land `empty: true, files_changed: 0` —
|
||||||
|
// byte-identical to an honest no-op, and just as quiet. Whatever went
|
||||||
|
// wrong is recorded so the artifact can say which of the two it was.
|
||||||
|
let mut diff_error: Option<String> = None;
|
||||||
|
let mut note_diff_failure = |what: &str, e: String| {
|
||||||
|
eprintln!(
|
||||||
|
"mission_delivery: mission {mission_id} phase {phase_id} could not compute \
|
||||||
|
{what} against {base_sha}: {e}"
|
||||||
|
);
|
||||||
|
if diff_error.is_none() {
|
||||||
|
diff_error = Some(format!("{what}: {}", e.chars().take(300).collect::<String>()));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
let mut diff_args = vec!["diff", base_sha.as_str(), "--"];
|
let mut diff_args = vec!["diff", base_sha.as_str(), "--"];
|
||||||
diff_args.extend(excludes.iter().map(String::as_str));
|
diff_args.extend(excludes.iter().map(String::as_str));
|
||||||
let patch = git(&repo, &diff_args).await.unwrap_or_default();
|
let patch = match git(&repo, &diff_args).await {
|
||||||
|
Ok(p) => p,
|
||||||
|
Err(e) => {
|
||||||
|
note_diff_failure("patch", e);
|
||||||
|
String::new()
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
let mut stat_args = vec!["diff", base_sha.as_str(), "--stat", "--"];
|
let mut stat_args = vec!["diff", base_sha.as_str(), "--stat", "--"];
|
||||||
stat_args.extend(excludes.iter().map(String::as_str));
|
stat_args.extend(excludes.iter().map(String::as_str));
|
||||||
let diffstat = git(&repo, &stat_args).await.unwrap_or_default();
|
let diffstat = match git(&repo, &stat_args).await {
|
||||||
|
Ok(s) => s,
|
||||||
|
Err(e) => {
|
||||||
|
note_diff_failure("diffstat", e);
|
||||||
|
String::new()
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// The paths themselves, not just the counts.
|
||||||
|
//
|
||||||
|
// The diffstat gives three integers and throws the filenames away, so
|
||||||
|
// nothing downstream could say WHICH files a phase touched — the World
|
||||||
|
// could draw a "coding" station but nothing under it. Same `base_sha` and
|
||||||
|
// the same excludes as the `--stat` call above: if the two disagreed,
|
||||||
|
// `files_changed` and this list would contradict each other and nobody
|
||||||
|
// could tell which one lied.
|
||||||
|
//
|
||||||
|
// Must run BEFORE the reset below — `--intent-to-add` is what makes newly
|
||||||
|
// created files visible to diff at all.
|
||||||
|
let mut name_args = vec!["diff", base_sha.as_str(), "--name-status", "--"];
|
||||||
|
name_args.extend(excludes.iter().map(String::as_str));
|
||||||
|
let name_status = match git(&repo, &name_args).await {
|
||||||
|
Ok(s) => s,
|
||||||
|
Err(e) => {
|
||||||
|
note_diff_failure("name-status", e);
|
||||||
|
String::new()
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
// Put the index back. `--intent-to-add` is a mutation of the agent's
|
// Put the index back. `--intent-to-add` is a mutation of the agent's
|
||||||
// workspace, and capture must not change what a later commit would see.
|
// workspace, and capture must not change what a later commit would see.
|
||||||
let _ = git(&repo, &["reset", "--quiet"]).await;
|
let _ = git(&repo, &["reset", "--quiet"]).await;
|
||||||
|
|
||||||
let (files_changed, insertions, deletions) = parse_diffstat(&diffstat);
|
let (files_changed, insertions, deletions) = parse_diffstat(&diffstat);
|
||||||
|
// Shared with auto_merge so the two cannot disagree about what a
|
||||||
|
// `--name-status` line means (renames are three fields; the NEW path is the
|
||||||
|
// one that changed).
|
||||||
|
let all_paths = crate::auto_merge::changed_paths(&name_status);
|
||||||
|
let files_truncated = all_paths.len() > MAX_CAPTURED_PATHS;
|
||||||
|
let files: Vec<(char, String)> = all_paths.into_iter().take(MAX_CAPTURED_PATHS).collect();
|
||||||
let empty = patch.trim().is_empty();
|
let empty = patch.trim().is_empty();
|
||||||
let truncated = patch.len() > MAX_PATCH_BYTES;
|
let truncated = patch.len() > MAX_PATCH_BYTES;
|
||||||
let stored = if truncated {
|
let stored = if truncated {
|
||||||
@@ -258,6 +335,9 @@ pub async fn capture_phase_diff_at(
|
|||||||
let patch_path = dir.join("diff.patch");
|
let patch_path = dir.join("diff.patch");
|
||||||
std::fs::write(&patch_path, &stored)
|
std::fs::write(&patch_path, &stored)
|
||||||
.map_err(|e| format!("write {}: {e}", patch_path.display()))?;
|
.map_err(|e| format!("write {}: {e}", patch_path.display()))?;
|
||||||
|
// Raw evidence on disk, independent of the JSONB. When the metadata and
|
||||||
|
// the picture disagree, this is the tiebreaker.
|
||||||
|
let _ = std::fs::write(dir.join("names.txt"), &name_status);
|
||||||
std::fs::write(dir.join("diffstat.txt"), &diffstat)
|
std::fs::write(dir.join("diffstat.txt"), &diffstat)
|
||||||
.map_err(|e| format!("write diffstat: {e}"))?;
|
.map_err(|e| format!("write diffstat: {e}"))?;
|
||||||
|
|
||||||
@@ -293,15 +373,42 @@ pub async fn capture_phase_diff_at(
|
|||||||
// Gate, then publish. Both are best-effort on top of an artifact that has
|
// Gate, then publish. Both are best-effort on top of an artifact that has
|
||||||
// already landed: a phase whose tests fail, or whose push is rejected,
|
// already landed: a phase whose tests fail, or whose push is rejected,
|
||||||
// still has its patch on disk and its work on a local branch.
|
// still has its patch on disk and its work on a local branch.
|
||||||
|
//
|
||||||
|
// `empty` suppresses publishing, so a diff we could not COMPUTE would
|
||||||
|
// otherwise skip the push and leave `push_error: null` — the phase looking
|
||||||
|
// exactly like one that correctly had nothing to publish. See
|
||||||
|
// [`untrusted_empty_reason`].
|
||||||
let mut outcome: Option<TestOutcome> = None;
|
let mut outcome: Option<TestOutcome> = None;
|
||||||
let mut published: Option<Publish> = None;
|
let mut published: Option<Publish> = None;
|
||||||
let mut publish_error: Option<String> = None;
|
let mut publish_error: Option<String> = untrusted_empty_reason(empty, diff_error.as_deref());
|
||||||
if let Some(c) = committed.as_ref() {
|
if let Some(c) = committed.as_ref() {
|
||||||
if !empty {
|
if !empty {
|
||||||
if gate == Gate::OnGreenTests {
|
if gate == Gate::OnGreenTests {
|
||||||
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
||||||
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
||||||
let o = verify_tests(&repo, &container).await;
|
// Against a COPY, never the checkout. `verify_tests` execs
|
||||||
|
// `cargo test` in a container running as ROOT, which writes
|
||||||
|
// `target/` — in the live tree that leaves root-owned build
|
||||||
|
// output in a checkout owned by uid 65532 and breaks the
|
||||||
|
// single-writer invariant. Measured the first time this gate
|
||||||
|
// ever ran end to end: `uids=0,65532`.
|
||||||
|
//
|
||||||
|
// The gate had been implemented but never exercised (every
|
||||||
|
// harness fixture used `commit_policy: "always"`), which is why
|
||||||
|
// a bug this mechanical survived in it.
|
||||||
|
let gate_root = crate::root_copy::copy_root("_gate", mission_id);
|
||||||
|
crate::root_copy::purge(&container, &gate_root).await;
|
||||||
|
let o = match crate::root_copy::RootCopy::of(&repo, &gate_root) {
|
||||||
|
Ok(copy) => {
|
||||||
|
let r = verify_tests(copy.workdir(), &container).await;
|
||||||
|
crate::root_copy::purge(&container, &gate_root).await;
|
||||||
|
r
|
||||||
|
}
|
||||||
|
// Fail-closed: an unverifiable suite must not license a push.
|
||||||
|
Err(e) => TestOutcome::CouldNotRun(format!(
|
||||||
|
"could not copy the checkout to test it: {e}"
|
||||||
|
)),
|
||||||
|
};
|
||||||
// An infrastructure fault must be loud. The gate degrades
|
// An infrastructure fault must be loud. The gate degrades
|
||||||
// safely either way, but "we could not run the suite" is a
|
// safely either way, but "we could not run the suite" is a
|
||||||
// problem with the platform and needs to look like one.
|
// problem with the platform and needs to look like one.
|
||||||
@@ -327,7 +434,14 @@ pub async fn capture_phase_diff_at(
|
|||||||
could not publish {}: {e}",
|
could not publish {}: {e}",
|
||||||
c.branch
|
c.branch
|
||||||
);
|
);
|
||||||
publish_error = Some(e.chars().take(500).collect());
|
// Both ends, not the first 500 chars. Git prints its
|
||||||
|
// REASON last — "non-fast-forward", "fetch first",
|
||||||
|
// "protected branch" — so a head-only clamp keeps the
|
||||||
|
// noise and drops the answer. A real push failure was
|
||||||
|
// recorded as two auth lines plus a branch name cut
|
||||||
|
// off mid-word, with the reject reason gone.
|
||||||
|
publish_error =
|
||||||
|
Some(crate::evaluator_tools::clamp_output(&e));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -379,7 +493,18 @@ pub async fn capture_phase_diff_at(
|
|||||||
"insertions": insertions,
|
"insertions": insertions,
|
||||||
"deletions": deletions,
|
"deletions": deletions,
|
||||||
"empty": empty,
|
"empty": empty,
|
||||||
|
// Non-null means `empty`/`files_changed` describe a failed read, not
|
||||||
|
// an unchanged tree. Readers that treat `empty: true` as "the phase
|
||||||
|
// did nothing" must check this first.
|
||||||
|
"diff_error": diff_error,
|
||||||
"truncated": truncated,
|
"truncated": truncated,
|
||||||
|
// WHICH files, not just how many. Same base_sha and the same excludes
|
||||||
|
// as `files_changed`, so the two describe the same diff.
|
||||||
|
"files": files
|
||||||
|
.iter()
|
||||||
|
.map(|(st, path)| serde_json::json!({ "status": st.to_string(), "path": path }))
|
||||||
|
.collect::<Vec<_>>(),
|
||||||
|
"files_truncated": files_truncated,
|
||||||
"excluded_paths": EXCLUDED_PATHS,
|
"excluded_paths": EXCLUDED_PATHS,
|
||||||
});
|
});
|
||||||
std::fs::write(
|
std::fs::write(
|
||||||
@@ -388,8 +513,11 @@ pub async fn capture_phase_diff_at(
|
|||||||
)
|
)
|
||||||
.map_err(|e| format!("write delivery.json: {e}"))?;
|
.map_err(|e| format!("write delivery.json: {e}"))?;
|
||||||
|
|
||||||
// Path is stored relative to the missions root, matching how
|
// Path is stored relative to the MISSIONS ROOT — the convention every
|
||||||
// `pdf_renderer` resolves artifact paths.
|
// artifact uses, and what `routes::missions::artifact_content` resolves
|
||||||
|
// against. (The old `pdf_renderer` claimed to match this and did not: it
|
||||||
|
// joined the mission id first, producing a doubled id and ENOENT. It is
|
||||||
|
// gone; this comment named it as the authority, which it never was.)
|
||||||
let rel = format!("_outputs/{mission_id}/{phase_id}/diff.patch");
|
let rel = format!("_outputs/{mission_id}/{phase_id}/diff.patch");
|
||||||
cm_db::repo::missions::register_artifact(
|
cm_db::repo::missions::register_artifact(
|
||||||
pool,
|
pool,
|
||||||
@@ -426,32 +554,41 @@ pub async fn capture_phase_diff_at(
|
|||||||
insertions,
|
insertions,
|
||||||
deletions,
|
deletions,
|
||||||
empty,
|
empty,
|
||||||
|
diff_error,
|
||||||
truncated,
|
truncated,
|
||||||
|
files,
|
||||||
|
files_truncated,
|
||||||
patch_path,
|
patch_path,
|
||||||
}))
|
}))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Run git in `repo`, returning stdout.
|
/// Run git in `repo`, returning stdout.
|
||||||
///
|
///
|
||||||
/// Every invocation carries `-c safe.directory`: the server clones as uid
|
/// Every invocation carries `-c safe.directory`: under `CLAWMATES_MISSION_FS=bind`
|
||||||
/// 65532 while agents write into the same tree as root, so without it git
|
/// the server clones as uid 65532 while agents write into the same tree as
|
||||||
/// refuses the repository outright — the failure that had the phase evaluator
|
/// root, so without it git refuses the repository outright — the failure that
|
||||||
/// silently falling back to guesswork.
|
/// had the phase evaluator silently falling back to guesswork. Copy mode makes
|
||||||
|
/// the tree single-uid and this redundant, but it stays while the bind path is
|
||||||
|
/// still selectable: a workaround may only be deleted once the situation it
|
||||||
|
/// works around can no longer be chosen.
|
||||||
async fn git(repo: &Path, args: &[&str]) -> Result<String, String> {
|
async fn git(repo: &Path, args: &[&str]) -> Result<String, String> {
|
||||||
let repo_s = repo.display().to_string();
|
let repo_s = repo.display().to_string();
|
||||||
let mut full = vec![
|
// Owned, not `Box::leak`. The leak was justified as "the process is
|
||||||
"-C",
|
// short-lived", which is true of a CLI and false of cm-api — it is a
|
||||||
&repo_s,
|
// long-running server, so that was one permanently leaked allocation per
|
||||||
"-c",
|
// git call, growing with every phase of every mission for the life of the
|
||||||
// Leaked into a `String` so it can live in a `&str` slice alongside
|
// process.
|
||||||
// the borrowed args; the process is short-lived and this is one
|
let mut full: Vec<String> = vec![
|
||||||
// allocation per git call.
|
"-C".into(),
|
||||||
Box::leak(format!("safe.directory={repo_s}").into_boxed_str()),
|
repo_s.clone(),
|
||||||
|
"-c".into(),
|
||||||
|
format!("safe.directory={repo_s}"),
|
||||||
];
|
];
|
||||||
full.extend_from_slice(args);
|
full.extend(args.iter().map(|a| (*a).to_string()));
|
||||||
let (name, email) = commit_identity();
|
let (name, email) = commit_identity();
|
||||||
let out = tokio::process::Command::new("git")
|
let mut cmd = tokio::process::Command::new("git");
|
||||||
.args(&full)
|
cmd.args(&full);
|
||||||
|
let out = crate::mission_workspace::no_terminal_prompt(&mut cmd)
|
||||||
// The server container has no git identity — `git config --global
|
// The server container has no git identity — `git config --global
|
||||||
// user.email` exits 1 — so `git commit` fails with "Author identity
|
// user.email` exits 1 — so `git commit` fails with "Author identity
|
||||||
// unknown" unless one is supplied. Mission `019fc450` lost its first
|
// unknown" unless one is supplied. Mission `019fc450` lost its first
|
||||||
@@ -480,10 +617,15 @@ async fn git(repo: &Path, args: &[&str]) -> Result<String, String> {
|
|||||||
"git {} → {}: {}",
|
"git {} → {}: {}",
|
||||||
args.first().copied().unwrap_or("?"),
|
args.first().copied().unwrap_or("?"),
|
||||||
out.status,
|
out.status,
|
||||||
String::from_utf8_lossy(&out.stderr)
|
// Both ends, never a head-only clamp. THIS is where the reason was
|
||||||
.chars()
|
// being lost: `publish_phase_branch` returns a rejected push as
|
||||||
.take(300)
|
// `Ok(Publish { error })`, so the string it carries was already
|
||||||
.collect::<String>()
|
// truncated here — 300 chars of auth noise, with "non-fast-forward"
|
||||||
|
// cut off — before the caller's own both-ends clamp ever saw it.
|
||||||
|
// Clamping the caller fixed the path that was already fine.
|
||||||
|
crate::mission_workspace::redact_token(&crate::evaluator_tools::clamp_output(
|
||||||
|
&String::from_utf8_lossy(&out.stderr)
|
||||||
|
))
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
Ok(String::from_utf8_lossy(&out.stdout).into_owned())
|
Ok(String::from_utf8_lossy(&out.stdout).into_owned())
|
||||||
@@ -556,7 +698,26 @@ pub async fn commit_phase_work(
|
|||||||
.map(|p| format!(":(exclude){p}"))
|
.map(|p| format!(":(exclude){p}"))
|
||||||
.collect();
|
.collect();
|
||||||
add.extend(excludes.iter().map(String::as_str));
|
add.extend(excludes.iter().map(String::as_str));
|
||||||
git(repo, &add).await?;
|
// `git add` EXITS 1 whenever the pathspec walked over a gitignored path,
|
||||||
|
// even though it staged everything else correctly and even though we
|
||||||
|
// excluded that path ourselves. Measured against git 2.x: `-c
|
||||||
|
// advice.addIgnoredFile=false`, `--ignore-errors`, `-A` and `:/` all still
|
||||||
|
// exit 1, and all still stage the right files. There is no flag that makes
|
||||||
|
// this command's exit code mean "nothing was staged".
|
||||||
|
//
|
||||||
|
// Propagating it with `?` therefore aborted the commit AFTER a successful
|
||||||
|
// staging, so the branch was never made, nothing was committed and nothing
|
||||||
|
// was pushed — for any repo with a populated `target/`, which is every Rust
|
||||||
|
// repo an agent has built in. `capture_phase_diff_at` above already treats
|
||||||
|
// the same command as advisory (`let _ = ...`); this is the same command
|
||||||
|
// and gets the same policy. The index is the source of truth, and the
|
||||||
|
// `diff --cached` immediately below is what actually reads it.
|
||||||
|
if let Err(e) = git(repo, &add).await {
|
||||||
|
eprintln!(
|
||||||
|
"mission_delivery: `git add` reported {e} — continuing, the staged \
|
||||||
|
index below is what decides whether there is anything to commit"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
// `--cached` compares the index against HEAD: empty means the agents left
|
// `--cached` compares the index against HEAD: empty means the agents left
|
||||||
// nothing unstaged for us, which is the normal case when they committed
|
// nothing unstaged for us, which is the normal case when they committed
|
||||||
@@ -718,7 +879,38 @@ async fn push_url_for(pool: &sqlx::PgPool, mission_id: Uuid) -> Result<Option<St
|
|||||||
// "nothing to push to" — the shape that made `commit_error` necessary.
|
// "nothing to push to" — the shape that made `commit_error` necessary.
|
||||||
.map_err(|e| format!("query push URL: {e}"))?
|
.map_err(|e| format!("query push URL: {e}"))?
|
||||||
.flatten();
|
.flatten();
|
||||||
Ok(url.map(|u| mission_workspace::with_ambient_auth(&u)))
|
let Some(url) = url else { return Ok(None) };
|
||||||
|
|
||||||
|
let auth = mission_workspace::with_ambient_auth(&url);
|
||||||
|
// Fail here, not at the tty. An unauthenticated URL to OUR forge cannot
|
||||||
|
// push, and every second it survives past this point is spent producing a
|
||||||
|
// symptom that looks like something else: git asking for a username, then
|
||||||
|
// `/dev/tty: No such device or address`, then a `push_error` about auth that
|
||||||
|
// sent #55's investigation after credentials which were never the problem.
|
||||||
|
// A third-party host is left alone — ssh keys and .netrc are legitimate.
|
||||||
|
if let Some(why) = &auth.unauthenticated {
|
||||||
|
if auth.is_forge() {
|
||||||
|
return Err(format!(
|
||||||
|
"cannot authenticate the push URL for this mission — {why}. The work \
|
||||||
|
is committed locally; fix the credential and re-run delivery."
|
||||||
|
));
|
||||||
|
}
|
||||||
|
eprintln!("mission_delivery: pushing to a non-forge remote unauthenticated — {why}");
|
||||||
|
}
|
||||||
|
Ok(Some(auth.url))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Did the forge reject this push because our history diverged from the ref?
|
||||||
|
///
|
||||||
|
/// Git says this several ways depending on version and refspec, and all of them
|
||||||
|
/// mean the same thing here: the branch already exists with commits ours does not
|
||||||
|
/// contain.
|
||||||
|
fn is_non_fast_forward(err: &str) -> bool {
|
||||||
|
let e = err.to_ascii_lowercase();
|
||||||
|
e.contains("non-fast-forward")
|
||||||
|
|| e.contains("fetch first")
|
||||||
|
|| e.contains("updates were rejected")
|
||||||
|
|| (e.contains("[rejected]") && !e.contains("stale info"))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Run the gate, then push the branch if the gate allows it.
|
/// Run the gate, then push the branch if the gate allows it.
|
||||||
@@ -761,6 +953,54 @@ pub async fn publish_phase_branch(
|
|||||||
error: None,
|
error: None,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
// #55: a mission whose checkout was re-cloned — a retry, a container
|
||||||
|
// teardown, disk loss — builds divergent history against its OWN
|
||||||
|
// deterministic branch, and every push it ever attempts is rejected.
|
||||||
|
// Before this, that was terminal: the work stayed on a local branch in a
|
||||||
|
// directory that gets reaped.
|
||||||
|
//
|
||||||
|
// The escape is a NEW ref, not `--force`. Forcing would overwrite
|
||||||
|
// whatever the earlier attempt pushed — which may be the only copy of
|
||||||
|
// that work — to make this attempt look tidy. Suffixing with the commit
|
||||||
|
// sha is deterministic (the same history always lands on the same ref),
|
||||||
|
// self-describing in a branch list, and cannot collide, since divergent
|
||||||
|
// history is by definition a different sha.
|
||||||
|
Err(e) if is_non_fast_forward(&e) => {
|
||||||
|
let sha = git(repo, &["rev-parse", "HEAD"])
|
||||||
|
.await
|
||||||
|
.map(|s| s.trim().to_string())
|
||||||
|
.unwrap_or_default();
|
||||||
|
let Some(short) = sha.get(..8) else {
|
||||||
|
eprintln!("mission_delivery: push of {target} rejected and HEAD unreadable: {e}");
|
||||||
|
return Ok(Publish {
|
||||||
|
branch: target,
|
||||||
|
pushed: false,
|
||||||
|
error: Some(e),
|
||||||
|
});
|
||||||
|
};
|
||||||
|
let alt = format!("{target}-{short}");
|
||||||
|
eprintln!(
|
||||||
|
"mission_delivery: {target} exists on the forge with history this \
|
||||||
|
checkout does not contain — pushing to {alt} instead of forcing. \
|
||||||
|
Original: {e}"
|
||||||
|
);
|
||||||
|
git(repo, &["branch", "-f", &alt, "HEAD"]).await?;
|
||||||
|
match git(repo, &["push", push_url, &format!("HEAD:refs/heads/{alt}")]).await {
|
||||||
|
Ok(_) => Ok(Publish {
|
||||||
|
branch: alt,
|
||||||
|
pushed: true,
|
||||||
|
// Not an error — the work reached the forge — but the
|
||||||
|
// redirect is a fact the operator needs, or two branches for
|
||||||
|
// one phase look like a bug rather than a rescue.
|
||||||
|
error: None,
|
||||||
|
}),
|
||||||
|
Err(e2) => Ok(Publish {
|
||||||
|
branch: alt,
|
||||||
|
pushed: false,
|
||||||
|
error: Some(format!("{e}\n\nand the diverged-history retry also failed: {e2}")),
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
// Redacted by `git`'s error path already; the patch and the local
|
// Redacted by `git`'s error path already; the patch and the local
|
||||||
// branch both survive, so this is a degraded success.
|
// branch both survive, so this is a degraded success.
|
||||||
@@ -789,6 +1029,39 @@ pub struct Publish {
|
|||||||
/// reason: this codebase has repeatedly found things reporting success while
|
/// reason: this codebase has repeatedly found things reporting success while
|
||||||
/// doing nothing, and a test suite that never ran must not license a push to a
|
/// doing nothing, and a test suite that never ran must not license a push to a
|
||||||
/// mission branch.
|
/// mission branch.
|
||||||
|
/// Did the toolchain fail to BUILD the project, as opposed to building it and
|
||||||
|
/// finding failing tests?
|
||||||
|
///
|
||||||
|
/// Deliberately narrow. These three phrases are emitted by cargo/rustc only
|
||||||
|
/// when compilation or linking did not complete; a failing `assert!` produces
|
||||||
|
/// none of them. Anything not matched here stays a red suite, because guessing
|
||||||
|
/// "probably an environment problem" over a genuine test failure is the far
|
||||||
|
/// more expensive mistake — it would let broken code through the gate.
|
||||||
|
fn build_failed(output: &str) -> bool {
|
||||||
|
let o = output.to_ascii_lowercase();
|
||||||
|
o.contains("error: could not compile")
|
||||||
|
|| o.contains("error: linking with")
|
||||||
|
|| o.contains("error: failed to run custom build command")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The first line that explains a build failure, for the artifact.
|
||||||
|
fn build_failure_excerpt(output: &str) -> String {
|
||||||
|
output
|
||||||
|
.lines()
|
||||||
|
.find(|l| {
|
||||||
|
let l = l.to_ascii_lowercase();
|
||||||
|
l.contains("error: could not compile")
|
||||||
|
|| l.contains("error: linking with")
|
||||||
|
|| l.contains("error: failed to run custom build command")
|
||||||
|
|| l.contains("cannot find -l")
|
||||||
|
|| l.contains("not installed")
|
||||||
|
})
|
||||||
|
.unwrap_or("")
|
||||||
|
.chars()
|
||||||
|
.take(300)
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
pub async fn verify_tests(repo: &Path, container: &str) -> TestOutcome {
|
pub async fn verify_tests(repo: &Path, container: &str) -> TestOutcome {
|
||||||
let Some(argv) = discover_test_command(repo) else {
|
let Some(argv) = discover_test_command(repo) else {
|
||||||
return TestOutcome::NoSuite;
|
return TestOutcome::NoSuite;
|
||||||
@@ -806,14 +1079,32 @@ pub async fn verify_tests(repo: &Path, container: &str) -> TestOutcome {
|
|||||||
argv.join(" "),
|
argv.join(" "),
|
||||||
out.exit_code
|
out.exit_code
|
||||||
);
|
);
|
||||||
|
let text = out.combined();
|
||||||
match out.exit_code {
|
match out.exit_code {
|
||||||
Some(0) => TestOutcome::Passed,
|
Some(0) => TestOutcome::Passed,
|
||||||
|
// A suite that never COMPILED is not a red suite. Both are
|
||||||
|
// non-zero (cargo exits 101 either way), and calling the
|
||||||
|
// difference is what stops a missing toolchain being reported
|
||||||
|
// as the user's code being broken.
|
||||||
|
//
|
||||||
|
// Measured twice on clawhdf5 in one sitting: no `cmake` gave
|
||||||
|
// "is `cmake` not installed?", and no `python3-dev` gave
|
||||||
|
// "cannot find -lpython3.11" — both exit 101, both would have
|
||||||
|
// been recorded as `tests_status: "failed"` on a repo whose
|
||||||
|
// tests were never run. The branch suffix is `-wip` either way,
|
||||||
|
// so nothing ships differently; what changes is that the
|
||||||
|
// artifact now says which of the two happened.
|
||||||
|
Some(_) if build_failed(&text) => TestOutcome::CouldNotRun(format!(
|
||||||
|
"`{}` could not build the project: {}",
|
||||||
|
argv.join(" "),
|
||||||
|
build_failure_excerpt(&text)
|
||||||
|
)),
|
||||||
// An unreadable status is not a pass, and it is not a red
|
// An unreadable status is not a pass, and it is not a red
|
||||||
// suite either — the command may never have started.
|
// suite either — the command may never have started.
|
||||||
None => TestOutcome::CouldNotRun(format!(
|
None => TestOutcome::CouldNotRun(format!(
|
||||||
"`{}` produced no exit status: {}",
|
"`{}` produced no exit status: {}",
|
||||||
argv.join(" "),
|
argv.join(" "),
|
||||||
out.combined().chars().take(300).collect::<String>()
|
text.chars().take(300).collect::<String>()
|
||||||
)),
|
)),
|
||||||
Some(code) => TestOutcome::Failed(code),
|
Some(code) => TestOutcome::Failed(code),
|
||||||
}
|
}
|
||||||
@@ -958,10 +1249,105 @@ pub async fn record_uncapturable(
|
|||||||
.map_err(|e| format!("register uncapturable marker: {e}"))
|
.map_err(|e| format!("register uncapturable marker: {e}"))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Why an empty patch must not be believed, if it must not be believed.
|
||||||
|
///
|
||||||
|
/// An empty patch has two causes that produce identical bytes: the tree really
|
||||||
|
/// did not change, or `git diff` failed and we have no idea what the tree
|
||||||
|
/// looks like. The first is an ordinary outcome; the second is a platform
|
||||||
|
/// fault. Returning `Some` for the second is what stops the fault from being
|
||||||
|
/// filed under the ordinary outcome — the recurring shape where a failure and
|
||||||
|
/// a legitimate negative share one representation.
|
||||||
|
fn untrusted_empty_reason(empty: bool, diff_error: Option<&str>) -> Option<String> {
|
||||||
|
match (empty, diff_error) {
|
||||||
|
(true, Some(why)) => Some(format!(
|
||||||
|
"not published: the diff could not be computed, so an empty patch \
|
||||||
|
cannot be trusted to mean an unchanged tree ({why})"
|
||||||
|
)),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod changed_path_capture_tests {
|
||||||
|
/// The path list and `files_changed` must describe the SAME diff.
|
||||||
|
///
|
||||||
|
/// They come from two separate git invocations — `--stat` and
|
||||||
|
/// `--name-status`. If those are ever given different revisions or
|
||||||
|
/// different exclude pathspecs, the count and the list disagree and there
|
||||||
|
/// is no way to tell which is right: both look like plausible output.
|
||||||
|
#[test]
|
||||||
|
fn both_diff_calls_use_the_same_revision_and_excludes() {
|
||||||
|
let src = include_str!("mission_delivery.rs");
|
||||||
|
let body = src
|
||||||
|
.split("let mut stat_args")
|
||||||
|
.nth(1)
|
||||||
|
.and_then(|s| s.split("let (files_changed").next())
|
||||||
|
.expect("the capture block");
|
||||||
|
assert!(
|
||||||
|
body.contains("let mut name_args = vec![\"diff\", base_sha.as_str(), \"--name-status\", \"--\"]"),
|
||||||
|
"the name-status call must use the same base_sha as --stat"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
body.contains("name_args.extend(excludes.iter().map(String::as_str))"),
|
||||||
|
"and the same excludes, or files_changed and the path list describe \
|
||||||
|
different diffs"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `--name-status` must run before the index is put back, or newly created
|
||||||
|
/// files — which `--intent-to-add` is what makes visible — vanish from the
|
||||||
|
/// list while still being counted by the stat.
|
||||||
|
#[test]
|
||||||
|
fn paths_are_read_before_the_index_reset() {
|
||||||
|
let src = include_str!("mission_delivery.rs");
|
||||||
|
let name_at = src.find("--name-status").expect("name-status call");
|
||||||
|
let reset_at = src
|
||||||
|
.find("git(&repo, &[\"reset\", \"--quiet\"])")
|
||||||
|
.expect("index reset");
|
||||||
|
assert!(
|
||||||
|
name_at < reset_at,
|
||||||
|
"the path list must be captured while --intent-to-add is still in \
|
||||||
|
effect, or created files are invisible to it"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
/// Git says "your history diverged" several ways, and the one production
|
||||||
|
/// actually produced (`! [rejected] ... (fetch first)`) is not the phrase
|
||||||
|
/// anyone reaches for first. Missing a phrasing means the rescue does not
|
||||||
|
/// fire and the work stays on a local branch in a directory that gets
|
||||||
|
/// reaped — silently, since the push failure is a degraded success.
|
||||||
|
#[test]
|
||||||
|
fn every_way_git_says_diverged_is_recognised() {
|
||||||
|
for e in [
|
||||||
|
"git push → exit 1: ! [rejected] HEAD -> b (fetch first)\nhint: …",
|
||||||
|
" ! [rejected] HEAD -> b (non-fast-forward)",
|
||||||
|
"hint: Updates were rejected because the remote contains work that you \
|
||||||
|
do not have locally.",
|
||||||
|
] {
|
||||||
|
assert!(is_non_fast_forward(e), "not recognised: {e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// And it must not fire on failures a new branch cannot fix. Retrying a
|
||||||
|
/// permissions or network error onto a second ref just produces a second
|
||||||
|
/// failure and a confusing branch name.
|
||||||
|
#[test]
|
||||||
|
fn other_push_failures_are_not_mistaken_for_divergence() {
|
||||||
|
for e in [
|
||||||
|
"fatal: repository 'https://forge/x.git' not found",
|
||||||
|
"remote: error: GH006: Protected branch update failed",
|
||||||
|
"fatal: could not read Username for 'https://forge': terminal prompts disabled",
|
||||||
|
" ! [rejected] (stale info)",
|
||||||
|
] {
|
||||||
|
assert!(!is_non_fast_forward(e), "wrongly recognised: {e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// ── The gate ───────────────────────────────────────────────────────
|
// ── The gate ───────────────────────────────────────────────────────
|
||||||
|
|
||||||
/// Three recipes have declared `commit_policy` since they were written and
|
/// Three recipes have declared `commit_policy` since they were written and
|
||||||
@@ -1008,6 +1394,103 @@ mod tests {
|
|||||||
assert_eq!(Gate::OnGreenTests.branch_suffix(None), "-wip");
|
assert_eq!(Gate::OnGreenTests.branch_suffix(None), "-wip");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The regression that destroyed a mission's work.
|
||||||
|
///
|
||||||
|
/// `git add -- . :(exclude)target` EXITS NON-ZERO when the tree contains a
|
||||||
|
/// gitignored `target/`, while correctly staging everything else. The
|
||||||
|
/// commit path used to propagate that exit with `?`, so a successful
|
||||||
|
/// staging still aborted the commit — no branch, no commit, no push — and
|
||||||
|
/// the agents' work was reaped with the container.
|
||||||
|
///
|
||||||
|
/// This test asserts the git behaviour itself, because the fix is only
|
||||||
|
/// correct for as long as the behaviour holds: if a future git makes this
|
||||||
|
/// command exit 0, this test fails and tells the next reader the workaround
|
||||||
|
/// can go. What must never regress is the second assertion — that the files
|
||||||
|
/// ARE staged regardless of the exit code, which is why the index and not
|
||||||
|
/// the exit code is the source of truth.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn git_add_exits_nonzero_over_an_ignored_path_yet_still_stages() {
|
||||||
|
let dir = tempfile::tempdir().unwrap();
|
||||||
|
let repo = dir.path();
|
||||||
|
let run = |args: Vec<&str>| {
|
||||||
|
std::process::Command::new("git")
|
||||||
|
.args(&args)
|
||||||
|
.current_dir(repo)
|
||||||
|
.output()
|
||||||
|
.expect("git runs")
|
||||||
|
};
|
||||||
|
run(vec!["init", "-q", "."]);
|
||||||
|
run(vec!["config", "user.email", "[email protected]"]);
|
||||||
|
run(vec!["config", "user.name", "T"]);
|
||||||
|
std::fs::write(repo.join(".gitignore"), "target\n").unwrap();
|
||||||
|
run(vec!["add", ".gitignore"]);
|
||||||
|
run(vec!["commit", "-qm", "base"]);
|
||||||
|
|
||||||
|
// The shape every Rust repo an agent has built in ends up with.
|
||||||
|
std::fs::create_dir_all(repo.join("target")).unwrap();
|
||||||
|
std::fs::write(repo.join("target/build.bin"), "junk").unwrap();
|
||||||
|
std::fs::create_dir_all(repo.join("research")).unwrap();
|
||||||
|
std::fs::write(repo.join("research/summary.md"), "findings").unwrap();
|
||||||
|
|
||||||
|
let mut add: Vec<&str> = vec!["add", "--", "."];
|
||||||
|
let excludes: Vec<String> = EXCLUDED_PATHS
|
||||||
|
.iter()
|
||||||
|
.map(|p| format!(":(exclude){p}"))
|
||||||
|
.collect();
|
||||||
|
add.extend(excludes.iter().map(String::as_str));
|
||||||
|
let out = run(add);
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
!out.status.success(),
|
||||||
|
"if this now succeeds, git changed and the advisory handling in \
|
||||||
|
commit_and_branch can be simplified"
|
||||||
|
);
|
||||||
|
|
||||||
|
let staged = run(vec!["diff", "--cached", "--name-only"]);
|
||||||
|
let staged = String::from_utf8_lossy(&staged.stdout);
|
||||||
|
assert!(
|
||||||
|
staged.contains("research/summary.md"),
|
||||||
|
"the work MUST be staged despite the non-zero exit; this is the \
|
||||||
|
assertion that stops a phase's output being silently discarded \
|
||||||
|
again. staged: {staged:?}"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!staged.contains("target/"),
|
||||||
|
"the exclude pathspec must still keep build output out: {staged:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A missing toolchain must not be reported as the user's tests failing.
|
||||||
|
/// Both are cargo exit 101; only the output distinguishes them, and this
|
||||||
|
/// session produced both real samples on clawhdf5.
|
||||||
|
#[test]
|
||||||
|
fn a_build_failure_is_not_a_red_suite() {
|
||||||
|
let no_cmake = "error: failed to run custom build command for `libz-ng-sys v1.1.29`\n\
|
||||||
|
is `cmake` not installed?";
|
||||||
|
let no_python = "= note: /usr/bin/ld: cannot find -lpython3.11: No such file or directory\n\
|
||||||
|
error: could not compile `clawhdf5-py` (lib) due to 1 previous error";
|
||||||
|
for sample in [no_cmake, no_python] {
|
||||||
|
assert!(build_failed(sample), "must read as a build failure: {sample}");
|
||||||
|
assert!(
|
||||||
|
!build_failure_excerpt(sample).is_empty(),
|
||||||
|
"the artifact needs a reason, not an empty string"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The half that protects the gate: a genuinely failing test must STAY a
|
||||||
|
/// red suite. Mistaking one for an environment problem would let broken
|
||||||
|
/// code past `on_green_tests`, which is the expensive direction to be
|
||||||
|
/// wrong in.
|
||||||
|
#[test]
|
||||||
|
fn a_failing_test_is_still_a_red_suite() {
|
||||||
|
let red = "running 3 tests\n\
|
||||||
|
test math::adds ... FAILED\n\
|
||||||
|
failures:\n math::adds\n\
|
||||||
|
test result: FAILED. 2 passed; 1 failed; 0 ignored";
|
||||||
|
assert!(!build_failed(red), "a failing assertion is not a build failure");
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_command_is_discovered_from_the_tree() {
|
fn test_command_is_discovered_from_the_tree() {
|
||||||
let dir = tempfile::tempdir().unwrap();
|
let dir = tempfile::tempdir().unwrap();
|
||||||
@@ -1060,6 +1543,28 @@ mod tests {
|
|||||||
|
|
||||||
/// An empty stat means an empty phase, not a parse failure. This is the
|
/// An empty stat means an empty phase, not a parse failure. This is the
|
||||||
/// case that must still produce an artifact.
|
/// case that must still produce an artifact.
|
||||||
|
/// The whole point: a tree that genuinely did not change stays silent, and
|
||||||
|
/// a diff that could not be computed does not get to borrow that silence.
|
||||||
|
#[test]
|
||||||
|
fn an_uncomputable_diff_is_not_an_unchanged_tree() {
|
||||||
|
assert_eq!(
|
||||||
|
untrusted_empty_reason(true, None),
|
||||||
|
None,
|
||||||
|
"a genuinely unchanged tree must not report an error"
|
||||||
|
);
|
||||||
|
let reason = untrusted_empty_reason(true, Some("patch: fatal: bad object"))
|
||||||
|
.expect("an empty patch from a FAILED diff must be reported, not accepted");
|
||||||
|
assert!(
|
||||||
|
reason.contains("bad object"),
|
||||||
|
"the reason must name what went wrong, got: {reason}"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
untrusted_empty_reason(false, Some("diffstat: fatal: bad object")),
|
||||||
|
None,
|
||||||
|
"a non-empty patch stands on its own even if the diffstat failed"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn an_empty_diffstat_is_all_zeroes() {
|
fn an_empty_diffstat_is_all_zeroes() {
|
||||||
assert_eq!(parse_diffstat(""), (0, 0, 0));
|
assert_eq!(parse_diffstat(""), (0, 0, 0));
|
||||||
|
|||||||
@@ -0,0 +1,369 @@
|
|||||||
|
//! Structured mission activity — the channel that replaced parsing prose.
|
||||||
|
//!
|
||||||
|
//! The operator decision behind this module: action detail comes from
|
||||||
|
//! **structured events at the source**, never from `checkpoint.log` or model
|
||||||
|
//! output. A tool name in a log line is indistinguishable from an agent
|
||||||
|
//! *discussing* a tool, and a visualization built on that distinction reads as
|
||||||
|
//! confident fact while being partly fiction.
|
||||||
|
//!
|
||||||
|
//! Everything here is best-effort. A mission must not fail because its
|
||||||
|
//! telemetry could not be written — so every write logs and swallows. That is a
|
||||||
|
//! deliberate exception to this codebase's usual rule, and it is bounded: the
|
||||||
|
//! only thing lost is detail in a picture.
|
||||||
|
|
||||||
|
use serde_json::Value;
|
||||||
|
use sqlx::PgPool;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// A phase entered `running`.
|
||||||
|
pub const PHASE_STARTED: &str = "phase.started";
|
||||||
|
/// A phase reached a terminal state. `detail.status` says which.
|
||||||
|
pub const PHASE_COMPLETED: &str = "phase.completed";
|
||||||
|
/// An agent called a tool. `target` is the tool name.
|
||||||
|
pub const TOOL_CALL: &str = "tool.call";
|
||||||
|
/// A tool touched a path. `target` is the path, repo-relative where known.
|
||||||
|
pub const FILE_TOUCH: &str = "file.touch";
|
||||||
|
/// The exact prompt text an agent was given. `detail.text` is the full string,
|
||||||
|
/// `target` is the role or tier that composed it.
|
||||||
|
///
|
||||||
|
/// The durable answer to "what did this agent actually receive". Skills, the
|
||||||
|
/// task, the evaluator's feedback and the tool preamble are assembled from four
|
||||||
|
/// places across three tiers, so re-deriving the prompt after the fact means
|
||||||
|
/// re-running that assembly against data that has since changed. Recording it
|
||||||
|
/// is the only way the question stays answerable.
|
||||||
|
pub const PROMPT_COMPOSED: &str = "prompt.composed";
|
||||||
|
/// The agent's own narrative for a turn. `detail.text`.
|
||||||
|
///
|
||||||
|
/// Written by `topology_worker` and pushed live once by `live_bus`. Until the
|
||||||
|
/// reader below existed, the stored row was never read again by anything: both
|
||||||
|
/// database readers in `routes/world.rs` filter to `tool.call`/`file.touch`,
|
||||||
|
/// and the only other statement touching the table is the GC that deletes it.
|
||||||
|
pub const REASONING: &str = "reasoning";
|
||||||
|
|
||||||
|
/// Kinds the per-phase cap applies to.
|
||||||
|
///
|
||||||
|
/// The cap exists to bound the two unbounded kinds: a coding phase can call
|
||||||
|
/// thousands of tools and touch thousands of paths. The others are bounded by
|
||||||
|
/// the phase's own structure — one start, one completion, one prompt per turn —
|
||||||
|
/// and counting them against the same budget meant a busy phase could push out
|
||||||
|
/// its OWN terminal event, leaving a phase that looks like it never finished.
|
||||||
|
const CAPPED_KINDS: &[&str] = &[TOOL_CALL, FILE_TOUCH];
|
||||||
|
|
||||||
|
/// Does this kind count against, and get dropped by, `PER_PHASE_CAP`?
|
||||||
|
pub fn is_capped(kind: &str) -> bool {
|
||||||
|
CAPPED_KINDS.contains(&kind)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Most events one phase may record.
|
||||||
|
///
|
||||||
|
/// A capped stream that says so beats an uncapped one that quietly becomes the
|
||||||
|
/// largest table in the database: a coding phase can call thousands of tools,
|
||||||
|
/// and every one of them would be replayed to every World subscriber. Past the
|
||||||
|
/// cap the picture is already complete — nobody reads the four-thousandth file
|
||||||
|
/// orb.
|
||||||
|
pub const PER_PHASE_CAP: i64 = 400;
|
||||||
|
|
||||||
|
/// One recorded event.
|
||||||
|
#[derive(Debug, Clone, Default)]
|
||||||
|
pub struct MissionEvent {
|
||||||
|
pub mission_id: Uuid,
|
||||||
|
pub phase_id: Option<Uuid>,
|
||||||
|
pub run_id: Option<Uuid>,
|
||||||
|
pub agent_id: Option<Uuid>,
|
||||||
|
pub kind: String,
|
||||||
|
pub target: Option<String>,
|
||||||
|
pub detail: Value,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl MissionEvent {
|
||||||
|
pub fn new(mission_id: Uuid, kind: &str) -> Self {
|
||||||
|
MissionEvent {
|
||||||
|
mission_id,
|
||||||
|
kind: kind.to_string(),
|
||||||
|
detail: Value::Null,
|
||||||
|
..Default::default()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
pub fn phase(mut self, id: Uuid) -> Self {
|
||||||
|
self.phase_id = Some(id);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
pub fn run(mut self, id: Uuid) -> Self {
|
||||||
|
self.run_id = Some(id);
|
||||||
|
self
|
||||||
|
}
|
||||||
|
pub fn agent(mut self, id: Option<Uuid>) -> Self {
|
||||||
|
self.agent_id = id;
|
||||||
|
self
|
||||||
|
}
|
||||||
|
pub fn target(mut self, t: impl Into<String>) -> Self {
|
||||||
|
self.target = Some(t.into());
|
||||||
|
self
|
||||||
|
}
|
||||||
|
pub fn detail(mut self, d: Value) -> Self {
|
||||||
|
self.detail = d;
|
||||||
|
self
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record one event, best-effort.
|
||||||
|
///
|
||||||
|
/// The per-phase cap is enforced in the INSERT itself rather than by a read
|
||||||
|
/// followed by a write: two tool taps writing concurrently would both read a
|
||||||
|
/// count below the cap and both insert, and the cap would drift by however many
|
||||||
|
/// writers there are. `INSERT … SELECT … WHERE (subquery) < cap` makes the
|
||||||
|
/// decision inside the statement.
|
||||||
|
pub async fn record(pool: &PgPool, e: MissionEvent) {
|
||||||
|
let detail = if e.detail.is_null() {
|
||||||
|
Value::Object(Default::default())
|
||||||
|
} else {
|
||||||
|
e.detail
|
||||||
|
};
|
||||||
|
// The cap is still decided INSIDE the insert (see the test below), and now
|
||||||
|
// only counts the kinds it is meant to bound.
|
||||||
|
let capped = is_capped(&e.kind);
|
||||||
|
let res = sqlx::query(
|
||||||
|
"INSERT INTO mission_events
|
||||||
|
(mission_id, phase_id, run_id, agent_id, kind, target, detail)
|
||||||
|
SELECT $1, $2, $3, $4, $5, $6, $7
|
||||||
|
WHERE $2::uuid IS NULL
|
||||||
|
OR NOT $9
|
||||||
|
OR (SELECT count(*) FROM mission_events
|
||||||
|
WHERE phase_id = $2 AND kind = ANY($10)) < $8",
|
||||||
|
)
|
||||||
|
.bind(e.mission_id)
|
||||||
|
.bind(e.phase_id)
|
||||||
|
.bind(e.run_id)
|
||||||
|
.bind(e.agent_id)
|
||||||
|
.bind(&e.kind)
|
||||||
|
.bind(&e.target)
|
||||||
|
.bind(&detail)
|
||||||
|
.bind(PER_PHASE_CAP)
|
||||||
|
.bind(capped)
|
||||||
|
.bind(CAPPED_KINDS)
|
||||||
|
.execute(pool)
|
||||||
|
.await;
|
||||||
|
if let Err(err) = res {
|
||||||
|
eprintln!("mission_events: record {} failed: {err}", e.kind);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record several events under one round trip's worth of intent.
|
||||||
|
/// Every recorded prompt and narrative for a mission, oldest first.
|
||||||
|
///
|
||||||
|
/// The read side of `PROMPT_COMPOSED` / `REASONING`. Both kinds were write-only
|
||||||
|
/// before this: the prompt was never stored at all, and the narrative was
|
||||||
|
/// stored and then read by nothing. Together they answer "what did this agent
|
||||||
|
/// receive, and what did it say it did", which is the question
|
||||||
|
/// `docs/PROVENANCE-ASSESSMENT.md` records as unanswerable.
|
||||||
|
pub async fn narrative_for_mission(
|
||||||
|
pool: &PgPool,
|
||||||
|
mission_id: Uuid,
|
||||||
|
) -> Result<Vec<(String, Option<Uuid>, Option<String>, String)>, sqlx::Error> {
|
||||||
|
let rows: Vec<(String, Option<Uuid>, Option<String>, Value)> = sqlx::query_as(
|
||||||
|
"SELECT kind, agent_id, target, detail
|
||||||
|
FROM mission_events
|
||||||
|
WHERE mission_id = $1 AND kind = ANY($2)
|
||||||
|
ORDER BY id",
|
||||||
|
)
|
||||||
|
.bind(mission_id)
|
||||||
|
.bind(&[PROMPT_COMPOSED, REASONING][..])
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await?;
|
||||||
|
Ok(rows
|
||||||
|
.into_iter()
|
||||||
|
.map(|(kind, agent, target, detail)| {
|
||||||
|
let text = detail
|
||||||
|
.get("text")
|
||||||
|
.and_then(|v| v.as_str())
|
||||||
|
.unwrap_or_default()
|
||||||
|
.to_string();
|
||||||
|
(kind, agent, target, text)
|
||||||
|
})
|
||||||
|
.collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One action an agent took, as a reader gets it back.
|
||||||
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
pub struct ToolEvidence {
|
||||||
|
/// The tool's name, e.g. `Bash`, `Write`.
|
||||||
|
pub tool: String,
|
||||||
|
/// The absolute path inside the sandbox, when the tool named one.
|
||||||
|
///
|
||||||
|
/// Absolute, unlike the sibling `file.touch` row's `target`. See the note
|
||||||
|
/// in `phase_runner::record_vm_tools`: normalising is what destroys the
|
||||||
|
/// only question a path can settle.
|
||||||
|
pub path: Option<String>,
|
||||||
|
/// The tool's arguments, bounded by `vm_tool_tap::bounded_input`.
|
||||||
|
pub input: Value,
|
||||||
|
/// What a command produced, bounded by `vm_tool_tap::bounded_response`.
|
||||||
|
///
|
||||||
|
/// Null for every tool that is not a command. This is where a failing test
|
||||||
|
/// run is visible, and it is the only place it is — the recorded stream has
|
||||||
|
/// no exit codes.
|
||||||
|
pub response: Value,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ToolEvidence {
|
||||||
|
/// The shell command, for the tools that run one.
|
||||||
|
pub fn command(&self) -> Option<&str> {
|
||||||
|
self.input.get("command").and_then(Value::as_str)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every tool call recorded for a mission, in order.
|
||||||
|
///
|
||||||
|
/// The counterpart to [`narrative_for_mission`], and the reason it exists: the
|
||||||
|
/// narrative is what an agent *said* it did. These rows are what it did. A
|
||||||
|
/// measurement built on the narrative alone scores prose, and prose is written
|
||||||
|
/// by the thing being measured.
|
||||||
|
///
|
||||||
|
/// **Bounded by [`PER_PHASE_CAP`].** A phase that ran more tools than the cap
|
||||||
|
/// returns the first `PER_PHASE_CAP` and no marker saying so, so a check that
|
||||||
|
/// concludes "this never happened" from an empty result is only sound for
|
||||||
|
/// phases under the cap. Every check in `skill_use` is one-sided in the safe
|
||||||
|
/// direction for that reason: it reports a violation it can see, never
|
||||||
|
/// compliance it inferred from silence.
|
||||||
|
pub async fn tool_evidence_for_mission(
|
||||||
|
pool: &PgPool,
|
||||||
|
mission_id: Uuid,
|
||||||
|
) -> Result<Vec<ToolEvidence>, sqlx::Error> {
|
||||||
|
let rows: Vec<(Option<String>, Value)> = sqlx::query_as(
|
||||||
|
"SELECT target, detail
|
||||||
|
FROM mission_events
|
||||||
|
WHERE mission_id = $1 AND kind = $2
|
||||||
|
ORDER BY id",
|
||||||
|
)
|
||||||
|
.bind(mission_id)
|
||||||
|
.bind(TOOL_CALL)
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await?;
|
||||||
|
Ok(rows
|
||||||
|
.into_iter()
|
||||||
|
.map(|(target, detail)| ToolEvidence {
|
||||||
|
tool: target.unwrap_or_default(),
|
||||||
|
path: detail
|
||||||
|
.get("path")
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.map(str::to_string),
|
||||||
|
input: detail.get("input").cloned().unwrap_or(Value::Null),
|
||||||
|
response: detail.get("response").cloned().unwrap_or(Value::Null),
|
||||||
|
})
|
||||||
|
.collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn record_all(pool: &PgPool, events: Vec<MissionEvent>) {
|
||||||
|
for e in events {
|
||||||
|
record(pool, e).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The path a tool's **arguments** name, if any.
|
||||||
|
///
|
||||||
|
/// Reads the arguments as JSON — never the tool's prose summary. The summary is
|
||||||
|
/// a sentence written for a human; a path pulled out of it by regex would be
|
||||||
|
/// right often enough to be trusted and wrong often enough to matter.
|
||||||
|
///
|
||||||
|
/// The key names are the ones Claude Code and the ZeroClaw tools actually use.
|
||||||
|
/// An unrecognised shape returns `None`, which renders as a tool call with no
|
||||||
|
/// file — accurate, rather than a guess at which argument was a path.
|
||||||
|
pub fn tool_path(args: &Value) -> Option<String> {
|
||||||
|
const KEYS: [&str; 6] = [
|
||||||
|
"file_path",
|
||||||
|
"filePath",
|
||||||
|
"path",
|
||||||
|
"notebook_path",
|
||||||
|
"file",
|
||||||
|
"target_file",
|
||||||
|
];
|
||||||
|
let obj = args.as_object()?;
|
||||||
|
for k in KEYS {
|
||||||
|
if let Some(s) = obj.get(k).and_then(Value::as_str) {
|
||||||
|
let s = s.trim();
|
||||||
|
if !s.is_empty() {
|
||||||
|
return Some(s.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
None
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Strip the guest/host workspace prefix so a path is repo-relative.
|
||||||
|
///
|
||||||
|
/// Tool arguments are absolute inside the sandbox (`/mission/repo/src/a.rs`).
|
||||||
|
/// Left alone, every mission's file tree would nest under a `mission` → `repo`
|
||||||
|
/// pair of directory orbs that exist in no repository and mean nothing to the
|
||||||
|
/// person reading the map.
|
||||||
|
pub fn repo_relative(path: &str, roots: &[&str]) -> String {
|
||||||
|
let p = path.trim();
|
||||||
|
for root in roots {
|
||||||
|
let root = root.trim_end_matches('/');
|
||||||
|
if let Some(rest) = p.strip_prefix(root) {
|
||||||
|
let rest = rest.trim_start_matches('/');
|
||||||
|
if !rest.is_empty() {
|
||||||
|
return rest.to_string();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
p.trim_start_matches("./").to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use serde_json::json;
|
||||||
|
|
||||||
|
/// Paths come from arguments, and only from argument keys we know.
|
||||||
|
///
|
||||||
|
/// The alternative — scanning the values for anything that looks like a
|
||||||
|
/// path — is what makes a viz confidently wrong: a `pattern` of `*.rs` or a
|
||||||
|
/// `command` of `ls src/` would both become "the agent edited a file".
|
||||||
|
#[test]
|
||||||
|
fn a_path_comes_from_a_known_argument_or_not_at_all() {
|
||||||
|
assert_eq!(
|
||||||
|
tool_path(&json!({"file_path": "/mission/repo/src/a.rs"})).as_deref(),
|
||||||
|
Some("/mission/repo/src/a.rs")
|
||||||
|
);
|
||||||
|
assert_eq!(tool_path(&json!({"path": "docs/x.md"})).as_deref(), Some("docs/x.md"));
|
||||||
|
// A shell command mentions paths and touches none we can name.
|
||||||
|
assert_eq!(tool_path(&json!({"command": "ls src/"})), None);
|
||||||
|
// A glob is a query, not a file.
|
||||||
|
assert_eq!(tool_path(&json!({"pattern": "**/*.rs"})), None);
|
||||||
|
// Blank is absence, not a file called "".
|
||||||
|
assert_eq!(tool_path(&json!({"file_path": " "})), None);
|
||||||
|
assert_eq!(tool_path(&json!("not an object")), None);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The sandbox prefix must not become two directory orbs in every mission.
|
||||||
|
#[test]
|
||||||
|
fn paths_are_made_repo_relative() {
|
||||||
|
let roots = ["/mission/repo", "/workspace"];
|
||||||
|
assert_eq!(repo_relative("/mission/repo/src/a.rs", &roots), "src/a.rs");
|
||||||
|
assert_eq!(repo_relative("/workspace/README.md", &roots), "README.md");
|
||||||
|
assert_eq!(repo_relative("./src/a.rs", &roots), "src/a.rs");
|
||||||
|
// Outside every root, it is left alone rather than mangled.
|
||||||
|
assert_eq!(repo_relative("/etc/hosts", &roots), "/etc/hosts");
|
||||||
|
// The root ITSELF is not a file, so it must not collapse to "".
|
||||||
|
assert_eq!(repo_relative("/mission/repo", &roots), "/mission/repo");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The cap must be decided inside the INSERT.
|
||||||
|
///
|
||||||
|
/// A count-then-insert is the classic version of this and it is wrong here:
|
||||||
|
/// the container tap and the microVM drain both write for the same phase,
|
||||||
|
/// and each would see a count below the cap and insert. Nothing errors —
|
||||||
|
/// the table simply grows past the bound that exists to hold it.
|
||||||
|
#[test]
|
||||||
|
fn the_cap_is_enforced_in_one_statement() {
|
||||||
|
let src = include_str!("mission_events.rs");
|
||||||
|
let body = src
|
||||||
|
.split("pub async fn record(")
|
||||||
|
.nth(1)
|
||||||
|
.and_then(|s| s.split("pub async fn").next())
|
||||||
|
.expect("record body");
|
||||||
|
assert!(
|
||||||
|
body.contains("INSERT INTO mission_events") && body.contains("SELECT count(*)"),
|
||||||
|
"the cap must be a subquery in the INSERT, not a separate read"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
+344
-19
@@ -48,14 +48,73 @@ pub fn pack_dir(root: &Path, name_in_archive: &str) -> Result<Vec<u8>, String> {
|
|||||||
// Follow no symlinks: a checkout can contain a link pointing outside the
|
// Follow no symlinks: a checkout can contain a link pointing outside the
|
||||||
// tree, and dereferencing it would pull host files into the container.
|
// tree, and dereferencing it would pull host files into the container.
|
||||||
builder.follow_symlinks(false);
|
builder.follow_symlinks(false);
|
||||||
builder
|
append_filtered(&mut builder, root, Path::new(name_in_archive))
|
||||||
.append_dir_all(name_in_archive, root)
|
|
||||||
.map_err(|e| format!("pack {}: {e}", root.display()))?;
|
.map_err(|e| format!("pack {}: {e}", root.display()))?;
|
||||||
builder
|
builder
|
||||||
.into_inner()
|
.into_inner()
|
||||||
.map_err(|e| format!("finish archive for {}: {e}", root.display()))
|
.map_err(|e| format!("finish archive for {}: {e}", root.display()))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Directory names never carried across the boundary.
|
||||||
|
///
|
||||||
|
/// The same list the delivery diff uses, deliberately: see
|
||||||
|
/// [`crate::mission_delivery::EXCLUDED_PATHS`]. A build directory is not work —
|
||||||
|
/// it is regenerable output that dwarfs the source, and shipping it cost a
|
||||||
|
/// mission its results when `vm_collect` timed out with the agent's finished work
|
||||||
|
/// still inside the VM.
|
||||||
|
pub fn transport_excludes() -> &'static [&'static str] {
|
||||||
|
crate::mission_delivery::EXCLUDED_PATHS
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Should this directory entry be left out of the archive?
|
||||||
|
///
|
||||||
|
/// Matched on the entry NAME at any depth, not on a path prefix: a workspace has
|
||||||
|
/// a `target/` per crate, and excluding only the root one would still ship the
|
||||||
|
/// rest.
|
||||||
|
pub fn is_excluded(name: &str) -> bool {
|
||||||
|
transport_excludes().contains(&name)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Recursive `append_dir_all` that skips [`transport_excludes`].
|
||||||
|
///
|
||||||
|
/// Hand-rolled because `tar::Builder::append_dir_all` takes no filter. Symlinks
|
||||||
|
/// are added as links rather than followed, matching `follow_symlinks(false)`.
|
||||||
|
fn append_filtered<W: std::io::Write>(
|
||||||
|
builder: &mut tar::Builder<W>,
|
||||||
|
dir: &Path,
|
||||||
|
prefix: &Path,
|
||||||
|
) -> std::io::Result<()> {
|
||||||
|
builder.append_dir(prefix, dir)?;
|
||||||
|
let mut entries: Vec<_> = std::fs::read_dir(dir)?.collect::<Result<Vec<_>, _>>()?;
|
||||||
|
// Stable order so an archive of the same tree is byte-identical, which makes
|
||||||
|
// a size or content difference between two runs mean something.
|
||||||
|
entries.sort_by_key(|e| e.file_name());
|
||||||
|
for entry in entries {
|
||||||
|
let name = entry.file_name();
|
||||||
|
let name_str = name.to_string_lossy();
|
||||||
|
let path = entry.path();
|
||||||
|
let dest = prefix.join(&name);
|
||||||
|
let meta = std::fs::symlink_metadata(&path)?;
|
||||||
|
if meta.is_dir() {
|
||||||
|
if is_excluded(&name_str) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
append_filtered(builder, &path, &dest)?;
|
||||||
|
} else if meta.is_symlink() {
|
||||||
|
let mut header = tar::Header::new_gnu();
|
||||||
|
header.set_metadata(&meta);
|
||||||
|
header.set_entry_type(tar::EntryType::Symlink);
|
||||||
|
header.set_size(0);
|
||||||
|
let target = std::fs::read_link(&path)?;
|
||||||
|
builder.append_link(&mut header, &dest, &target)?;
|
||||||
|
} else {
|
||||||
|
let mut f = std::fs::File::open(&path)?;
|
||||||
|
builder.append_file(&dest, &mut f)?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
/// Unpack a tar into a host directory.
|
/// Unpack a tar into a host directory.
|
||||||
///
|
///
|
||||||
/// `tar` refuses entries whose paths escape the destination, which is the
|
/// `tar` refuses entries whose paths escape the destination, which is the
|
||||||
@@ -69,8 +128,52 @@ pub fn unpack_into(archive: &[u8], dest: &Path) -> Result<(), String> {
|
|||||||
// Ownership in the archive is the container's root; re-applying it on the
|
// Ownership in the archive is the container's root; re-applying it on the
|
||||||
// host would recreate the very uid split this module exists to remove.
|
// host would recreate the very uid split this module exists to remove.
|
||||||
ar.set_preserve_permissions(false);
|
ar.set_preserve_permissions(false);
|
||||||
ar.unpack(dest)
|
|
||||||
.map_err(|e| format!("unpack into {}: {e}", dest.display()))
|
// Filter on the way OUT as well as on the way in.
|
||||||
|
//
|
||||||
|
// `pack_dir` (host -> container) skips `transport_excludes`, but `copy_out`
|
||||||
|
// (container -> host) is the raw Docker archive API, which carries the whole
|
||||||
|
// tree — `target/` included. The asymmetry was invisible for as long as the
|
||||||
|
// runtime image had no `cmake`, because nothing could compile and no
|
||||||
|
// `target/` existed. The moment missions could build, every collection
|
||||||
|
// failed on a build artifact:
|
||||||
|
//
|
||||||
|
// failed to unpack `…/repo/target/debug/build/ahash-…/build_script_build-…`
|
||||||
|
//
|
||||||
|
// and `phase_runner` correctly refused to capture a stale tree — so a
|
||||||
|
// coding phase that HAD done the work delivered nothing, retrying forever.
|
||||||
|
//
|
||||||
|
// Entries are skipped by NAME at any depth, the same rule `is_excluded`
|
||||||
|
// uses, because a workspace has a `target/` per crate.
|
||||||
|
let mut skipped = 0usize;
|
||||||
|
for entry in ar
|
||||||
|
.entries()
|
||||||
|
.map_err(|e| format!("read archive for {}: {e}", dest.display()))?
|
||||||
|
{
|
||||||
|
let mut entry = entry.map_err(|e| format!("read entry for {}: {e}", dest.display()))?;
|
||||||
|
let path = entry
|
||||||
|
.path()
|
||||||
|
.map_err(|e| format!("entry path for {}: {e}", dest.display()))?
|
||||||
|
.into_owned();
|
||||||
|
if path
|
||||||
|
.components()
|
||||||
|
.any(|c| is_excluded(&c.as_os_str().to_string_lossy()))
|
||||||
|
{
|
||||||
|
skipped += 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
entry
|
||||||
|
.unpack_in(dest)
|
||||||
|
.map_err(|e| format!("unpack into {}: {e}", dest.display()))?;
|
||||||
|
}
|
||||||
|
if skipped > 0 {
|
||||||
|
eprintln!(
|
||||||
|
"mission_fs: unpack into {} skipped {skipped} excluded entr{} (build output)",
|
||||||
|
dest.display(),
|
||||||
|
if skipped == 1 { "y" } else { "ies" }
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Copy a host directory into a running container at [`CONTAINER_MISSION_DIR`].
|
/// Copy a host directory into a running container at [`CONTAINER_MISSION_DIR`].
|
||||||
@@ -90,6 +193,105 @@ pub async fn copy_in(
|
|||||||
.map_err(|e| format!("copy into {container}:{CONTAINER_MISSION_DIR}: {e}"))
|
.map_err(|e| format!("copy into {container}:{CONTAINER_MISSION_DIR}: {e}"))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Build a one-entry tar. Split out from [`put_file`] so the size-independence
|
||||||
|
/// that is the whole point can be tested without Docker.
|
||||||
|
fn single_file_archive(name: &str, contents: &[u8]) -> Result<Vec<u8>, String> {
|
||||||
|
let mut header = tar::Header::new_gnu();
|
||||||
|
header
|
||||||
|
.set_path(name)
|
||||||
|
.map_err(|e| format!("tar path {name}: {e}"))?;
|
||||||
|
header.set_size(contents.len() as u64);
|
||||||
|
header.set_mode(0o600);
|
||||||
|
header.set_entry_type(tar::EntryType::Regular);
|
||||||
|
header.set_cksum();
|
||||||
|
|
||||||
|
let mut builder = tar::Builder::new(Vec::new());
|
||||||
|
builder
|
||||||
|
.append(&header, contents)
|
||||||
|
.map_err(|e| format!("tar {name}: {e}"))?;
|
||||||
|
builder
|
||||||
|
.into_inner()
|
||||||
|
.map_err(|e| format!("finish archive for {name}: {e}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write one file into a container, at any size.
|
||||||
|
///
|
||||||
|
/// The obvious way to do this is `sh -c "printf … > file"`, and it works right
|
||||||
|
/// up until the payload approaches `ARG_MAX`, at which point exec fails with
|
||||||
|
/// `argument list too long`. That is a size-dependent failure in a code path
|
||||||
|
/// whose payload grows with use, which makes it a bug that ships green and
|
||||||
|
/// surfaces in production — as it did, silently unpinning every agent in
|
||||||
|
/// mission `019fcf62`. Tar has no argv limit.
|
||||||
|
///
|
||||||
|
/// The write is not atomic. Callers that need it can upload beside the target
|
||||||
|
/// and rename; the config writer does not, because the daemon reads its config
|
||||||
|
/// once at boot and is restarted afterwards.
|
||||||
|
pub async fn put_file(
|
||||||
|
docker: &Docker,
|
||||||
|
container: &str,
|
||||||
|
path: &str,
|
||||||
|
contents: &[u8],
|
||||||
|
) -> Result<(), String> {
|
||||||
|
let (dir, file) = path
|
||||||
|
.rsplit_once('/')
|
||||||
|
.ok_or_else(|| format!("{path} is not an absolute path"))?;
|
||||||
|
let dir = if dir.is_empty() { "/" } else { dir };
|
||||||
|
|
||||||
|
let archive = single_file_archive(file, contents)?;
|
||||||
|
let opts = bollard::query_parameters::UploadToContainerOptionsBuilder::default()
|
||||||
|
.path(dir)
|
||||||
|
.build();
|
||||||
|
docker
|
||||||
|
.upload_to_container(container, Some(opts), bollard::body_full(archive.into()))
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("upload {path} to {container}: {e}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Build a flat tar of several files. [`single_file_archive`] for many.
|
||||||
|
fn files_archive(files: &[(String, Vec<u8>)]) -> Result<Vec<u8>, String> {
|
||||||
|
let mut builder = tar::Builder::new(Vec::new());
|
||||||
|
for (name, contents) in files {
|
||||||
|
let mut header = tar::Header::new_gnu();
|
||||||
|
header
|
||||||
|
.set_path(name)
|
||||||
|
.map_err(|e| format!("tar path {name}: {e}"))?;
|
||||||
|
header.set_size(contents.len() as u64);
|
||||||
|
// World-readable, unlike `single_file_archive`'s 0600: that one carries
|
||||||
|
// a credential, this one carries procedures the agent is meant to read.
|
||||||
|
header.set_mode(0o644);
|
||||||
|
header.set_entry_type(tar::EntryType::Regular);
|
||||||
|
header.set_cksum();
|
||||||
|
builder
|
||||||
|
.append(&header, contents.as_slice())
|
||||||
|
.map_err(|e| format!("tar {name}: {e}"))?;
|
||||||
|
}
|
||||||
|
builder
|
||||||
|
.into_inner()
|
||||||
|
.map_err(|e| format!("finish archive of {} files: {e}", files.len()))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write several files into one directory of a container, in one upload.
|
||||||
|
///
|
||||||
|
/// `dir` must already exist — `upload_to_container` will not create it, the
|
||||||
|
/// same constraint [`sync_in`] works around. Size-independent for the reason
|
||||||
|
/// [`put_file`] gives; fifty skill bodies would be well past `ARG_MAX` as a
|
||||||
|
/// printf.
|
||||||
|
pub async fn put_files(
|
||||||
|
docker: &Docker,
|
||||||
|
container: &str,
|
||||||
|
dir: &str,
|
||||||
|
files: &[(String, Vec<u8>)],
|
||||||
|
) -> Result<(), String> {
|
||||||
|
let archive = files_archive(files)?;
|
||||||
|
let opts = bollard::query_parameters::UploadToContainerOptionsBuilder::default()
|
||||||
|
.path(dir)
|
||||||
|
.build();
|
||||||
|
docker
|
||||||
|
.upload_to_container(container, Some(opts), bollard::body_full(archive.into()))
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("upload {} files to {container}:{dir}: {e}", files.len()))
|
||||||
|
}
|
||||||
|
|
||||||
/// Copy a directory back out of a container onto the host.
|
/// Copy a directory back out of a container onto the host.
|
||||||
pub async fn copy_out(
|
pub async fn copy_out(
|
||||||
docker: &Docker,
|
docker: &Docker,
|
||||||
@@ -113,14 +315,20 @@ pub async fn copy_out(
|
|||||||
|
|
||||||
/// Is the copy-in/copy-out filesystem model enabled?
|
/// Is the copy-in/copy-out filesystem model enabled?
|
||||||
///
|
///
|
||||||
/// Opt-in. The bind-mount path is what production has run since the beginning,
|
/// **Default since 2026-08-04.** It shipped opt-in, on the principle that
|
||||||
/// and silently changing how every mission receives its code is exactly the
|
/// silently changing how every mission receives its code should require
|
||||||
/// class of change that should require someone to have typed it.
|
/// someone to have typed it. Four production missions and a fail-closed
|
||||||
|
/// harness later (`scripts/verify-mission-delivery.sh`), the opt-in is the
|
||||||
|
/// riskier setting: the bind path is the one with four documented work-loss
|
||||||
|
/// incidents, and leaving it as the default means the untested path is what
|
||||||
|
/// runs when nobody sets the variable.
|
||||||
|
///
|
||||||
|
/// `CLAWMATES_MISSION_FS=bind` still selects the old behaviour, so a revert is
|
||||||
|
/// one line in `.env` rather than a rollback. Anything else — unset, empty,
|
||||||
|
/// misspelt — gets copy mode, because the failure mode of a typo should be the
|
||||||
|
/// safer path, not the one being retired.
|
||||||
pub fn copy_mode() -> bool {
|
pub fn copy_mode() -> bool {
|
||||||
matches!(
|
!matches!(std::env::var("CLAWMATES_MISSION_FS").as_deref(), Ok("bind"))
|
||||||
std::env::var("CLAWMATES_MISSION_FS").as_deref(),
|
|
||||||
Ok("copy")
|
|
||||||
)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Host directory holding a mission's checkout.
|
/// Host directory holding a mission's checkout.
|
||||||
@@ -130,14 +338,52 @@ fn host_repo(mission_id: uuid::Uuid) -> std::path::PathBuf {
|
|||||||
|
|
||||||
/// Push the host checkout into the container before a phase runs.
|
/// Push the host checkout into the container before a phase runs.
|
||||||
///
|
///
|
||||||
/// No-op when the mission has no repo — research-only missions have no
|
/// A repo-less mission has no checkout to push, but it still needs
|
||||||
/// checkout, and that must not fail a phase launch.
|
/// `/mission/repo` to EXIST inside the container: the phase prompt tells the
|
||||||
|
/// agent that is its working directory, `mission_orchestrator` pins every
|
||||||
|
/// claw's `workspace.path` to it, and `mission_outputs` copies it back out to
|
||||||
|
/// register artifacts. This used to return early instead, so none of those three
|
||||||
|
/// were true — the pin resolved to nothing, ZeroClaw fell back to each agent's
|
||||||
|
/// own sandbox, and the agents (correctly) reported they had no such directory
|
||||||
|
/// and refused to work. Creating it empty is what the microVM tier already does,
|
||||||
|
/// for the same reason: see `microvm_executor::inject` ("the guest needs the
|
||||||
|
/// workspace to exist before the agent writes into it").
|
||||||
|
///
|
||||||
|
/// Creating it host-side rather than `mkdir`-ing in the container keeps the copy
|
||||||
|
/// cycle symmetric — `sync_out` unpacks over this same path, so work written by
|
||||||
|
/// one phase survives into the next instead of being wiped by the next
|
||||||
|
/// `sync_in`.
|
||||||
pub async fn sync_in(container: &str, mission_id: uuid::Uuid) -> Result<(), String> {
|
pub async fn sync_in(container: &str, mission_id: uuid::Uuid) -> Result<(), String> {
|
||||||
let repo = host_repo(mission_id);
|
let repo = host_repo(mission_id);
|
||||||
if !repo.is_dir() {
|
if !repo.is_dir() {
|
||||||
return Ok(());
|
tokio::fs::create_dir_all(&repo)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("create empty workspace {}: {e}", repo.display()))?;
|
||||||
}
|
}
|
||||||
let docker = crate::container_exec::connect()?;
|
let docker = crate::container_exec::connect()?;
|
||||||
|
// `upload_to_container` requires the DESTINATION to exist: uploading into
|
||||||
|
// `/mission` when the container has no `/mission` fails with
|
||||||
|
// "404 Could not find the file /mission in container", which reads like a
|
||||||
|
// missing source file rather than a missing target directory. Nothing else
|
||||||
|
// creates it — not the image, not the container spec (in copy mode there is
|
||||||
|
// no `/mission` bind) — so create it here, immediately before the copy that
|
||||||
|
// depends on it.
|
||||||
|
let mkdir = [
|
||||||
|
"mkdir".to_string(),
|
||||||
|
"-p".to_string(),
|
||||||
|
CONTAINER_MISSION_DIR.to_string(),
|
||||||
|
];
|
||||||
|
if let Err(e) = crate::container_exec::exec_as_root(
|
||||||
|
&docker,
|
||||||
|
container,
|
||||||
|
None,
|
||||||
|
&mkdir,
|
||||||
|
std::time::Duration::from_secs(20),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
return Err(format!("create {CONTAINER_MISSION_DIR} in {container}: {e}"));
|
||||||
|
}
|
||||||
copy_in(&docker, container, &repo, "repo").await
|
copy_in(&docker, container, &repo, "repo").await
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -170,6 +416,44 @@ mod tests {
|
|||||||
std::fs::write(root.join(".git/HEAD"), "ref: refs/heads/main\n").unwrap();
|
std::fs::write(root.join(".git/HEAD"), "ref: refs/heads/main\n").unwrap();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Build output must be dropped on the way BACK, not only on the way out.
|
||||||
|
///
|
||||||
|
/// `copy_out` uses the raw Docker archive API, which carries `target/`
|
||||||
|
/// whatever `pack_dir` did. Unpacking it failed on a build-script binary
|
||||||
|
/// and took the whole collection down with it, so a coding phase that had
|
||||||
|
/// really done the work delivered nothing.
|
||||||
|
#[test]
|
||||||
|
fn unpacking_drops_build_output_but_keeps_the_source() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let src = tmp.path().join("repo");
|
||||||
|
std::fs::create_dir_all(src.join("src")).unwrap();
|
||||||
|
std::fs::create_dir_all(src.join("target/debug/build")).unwrap();
|
||||||
|
std::fs::create_dir_all(src.join("crates/inner/target")).unwrap();
|
||||||
|
std::fs::write(src.join("src/lib.rs"), "pub fn x() {}\n").unwrap();
|
||||||
|
std::fs::write(src.join("target/debug/build/script"), "ELF").unwrap();
|
||||||
|
std::fs::write(src.join("crates/inner/target/blob"), "ELF").unwrap();
|
||||||
|
|
||||||
|
// Built WITHOUT the filter, the way the Docker API hands it to us.
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
{
|
||||||
|
let mut b = tar::Builder::new(&mut buf);
|
||||||
|
b.append_dir_all("repo", &src).unwrap();
|
||||||
|
b.finish().unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
let dest = tmp.path().join("out");
|
||||||
|
unpack_into(&buf, &dest).expect("must not fail on build output");
|
||||||
|
assert!(dest.join("repo/src/lib.rs").is_file(), "source must survive");
|
||||||
|
assert!(
|
||||||
|
!dest.join("repo/target").exists(),
|
||||||
|
"root target/ must be dropped"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
!dest.join("repo/crates/inner/target").exists(),
|
||||||
|
"a per-crate target/ must be dropped too — matched by NAME at any depth"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
/// A checkout must survive the round trip intact — including `.git`,
|
/// A checkout must survive the round trip intact — including `.git`,
|
||||||
/// without which the whole delivery path (diff, commit, push) is dead.
|
/// without which the whole delivery path (diff, commit, push) is dead.
|
||||||
#[test]
|
#[test]
|
||||||
@@ -251,15 +535,56 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The switch must be explicit — a near-miss value leaves production on
|
/// Only the exact word `bind` opts out. A typo must land on copy mode —
|
||||||
/// the proven bind-mount path rather than silently changing it.
|
/// the path with a verification harness behind it — rather than silently
|
||||||
|
/// selecting the one with four documented work-loss incidents.
|
||||||
#[test]
|
#[test]
|
||||||
fn copy_mode_requires_the_exact_word() {
|
fn only_the_exact_word_bind_opts_out() {
|
||||||
for wrong in ["Copy", "copies", "bind", "1", "true", ""] {
|
// Cannot set env vars in a test process without racing every other
|
||||||
assert_ne!(wrong, "copy", "{wrong:?} must not enable copy mode");
|
// test, so this asserts the predicate the function is built from.
|
||||||
|
let opts_out = |v: &str| v == "bind";
|
||||||
|
assert!(opts_out("bind"));
|
||||||
|
for near_miss in ["Bind", "binds", "bound", "copy", "0", "false", ""] {
|
||||||
|
assert!(
|
||||||
|
!opts_out(near_miss),
|
||||||
|
"{near_miss:?} must NOT select the bind path"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The regression this exists for: a config large enough to blow `ARG_MAX`
|
||||||
|
/// via `sh -c` must round-trip untouched. 2 MB is well past the ~128 KB
|
||||||
|
/// limit that unpinned every agent in mission `019fcf62`.
|
||||||
|
#[test]
|
||||||
|
fn a_file_far_past_arg_max_round_trips() {
|
||||||
|
let big = "workspace_path = \"/mission/repo\"\n".repeat(64 * 1024);
|
||||||
|
assert!(big.len() > 2_000_000, "the fixture must exceed ARG_MAX");
|
||||||
|
|
||||||
|
let archive = single_file_archive("config.toml", big.as_bytes()).unwrap();
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
unpack_into(&archive, tmp.path()).unwrap();
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
std::fs::read_to_string(tmp.path().join("config.toml")).unwrap(),
|
||||||
|
big,
|
||||||
|
"a large config must survive byte-for-byte"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// TOML holding quotes, newlines and backslashes went through a shell
|
||||||
|
/// before; nothing may depend on quoting now.
|
||||||
|
#[test]
|
||||||
|
fn shell_metacharacters_survive_the_archive() {
|
||||||
|
let nasty = "path = \"/a'b\\\"c\"\n$(rm -rf /) `id` \\\\ \n";
|
||||||
|
let archive = single_file_archive("config.toml", nasty.as_bytes()).unwrap();
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
unpack_into(&archive, tmp.path()).unwrap();
|
||||||
|
assert_eq!(
|
||||||
|
std::fs::read_to_string(tmp.path().join("config.toml")).unwrap(),
|
||||||
|
nasty
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn an_empty_directory_packs_without_error() {
|
fn an_empty_directory_packs_without_error() {
|
||||||
let tmp = tempfile::tempdir().unwrap();
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
|||||||
@@ -0,0 +1,423 @@
|
|||||||
|
//! Reclaim the mission tree on the gateway.
|
||||||
|
//!
|
||||||
|
//! # Why this is filesystem-first
|
||||||
|
//!
|
||||||
|
//! `cleanup_sweeper` prunes ROWS. Deleting a row does not delete a directory,
|
||||||
|
//! and the reaper that was supposed to — `mission_runtime::teardown_container` —
|
||||||
|
//! only runs while a mission still exists to tear down. So a mission deleted by
|
||||||
|
//! any path that did not go through teardown left its directory behind forever,
|
||||||
|
//! and the gateway is the smallest disk in the fleet (150 GB, shared with
|
||||||
|
//! postgres and every checkout).
|
||||||
|
//!
|
||||||
|
//! The DB is therefore the PREDICATE here, never the enumerator: this walks the
|
||||||
|
//! filesystem and asks the database about what it finds. Enumerating from the
|
||||||
|
//! database is precisely how the orphans became invisible — a directory whose
|
||||||
|
//! row is gone is exactly the one a row-driven sweep cannot see.
|
||||||
|
//!
|
||||||
|
//! # Why deletion needs two attempts
|
||||||
|
//!
|
||||||
|
//! The server runs as uid 65532. Almost everything under a mission belongs to
|
||||||
|
//! 65532 now, but the per-mission ZeroClaw daemon still runs as root and leaves
|
||||||
|
//! ~26 of its own files (`.claude.json`, session jsonl). `remove_dir_all` then
|
||||||
|
//! fails with `PermissionDenied` and the directory survives — the
|
||||||
|
//! cleanup-that-cannot-clean-up shape, at a scale small enough to go unnoticed.
|
||||||
|
//! So a failed removal falls back to `root_copy::purge`, which deletes from
|
||||||
|
//! inside the runtime container as root.
|
||||||
|
//!
|
||||||
|
//! # What it will not touch
|
||||||
|
//!
|
||||||
|
//! Anything belonging to a mission that still has a row, and anything younger
|
||||||
|
//! than the grace window. A mission directory is created BEFORE its row is
|
||||||
|
//! committed in some paths, and reaping a directory out from under a launching
|
||||||
|
//! mission would be a far worse bug than the leak this fixes.
|
||||||
|
|
||||||
|
use std::path::Path;
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
use sqlx::PgPool;
|
||||||
|
|
||||||
|
/// How long a directory must have been untouched before it is considered
|
||||||
|
/// abandoned. Generously long: the cost of waiting is disk, and the cost of
|
||||||
|
/// being wrong is deleting a live mission's checkout.
|
||||||
|
const ORPHAN_GRACE: Duration = Duration::from_secs(2 * 60 * 60);
|
||||||
|
|
||||||
|
/// Retention for captured outputs (`_outputs`), which are artifacts a user can
|
||||||
|
/// still open. Mirrors `TOPOLOGY_RUNS_DAYS` in `cleanup_sweeper` — the run
|
||||||
|
/// history and the files it points at should not outlive each other.
|
||||||
|
const OUTPUTS_DAYS: u64 = 90;
|
||||||
|
|
||||||
|
/// Scratch trees the mission machinery makes and is supposed to remove itself:
|
||||||
|
/// `_bench`, `_gate`, `_verify`, `_merge`. Anything older than this is debris
|
||||||
|
/// from a crashed or killed run, not work in progress — every command that
|
||||||
|
/// creates one is bounded well below it.
|
||||||
|
const SCRATCH_GRACE: Duration = Duration::from_secs(6 * 60 * 60);
|
||||||
|
|
||||||
|
/// Directories under the missions root that are NOT missions.
|
||||||
|
const RESERVED: &[&str] = &["_outputs", "_home", "_cargo", "_mirrors"];
|
||||||
|
|
||||||
|
pub fn spawn(pool: PgPool, interval: Duration) {
|
||||||
|
tokio::spawn(async move {
|
||||||
|
// Not on the first tick. A sweep racing the server's own startup — while
|
||||||
|
// `start_pending_phases` is still adopting in-flight missions — is the
|
||||||
|
// one moment its "no row for this directory" predicate is least
|
||||||
|
// trustworthy.
|
||||||
|
tokio::time::sleep(Duration::from_secs(120)).await;
|
||||||
|
let mut tick = tokio::time::interval(interval);
|
||||||
|
loop {
|
||||||
|
tick.tick().await;
|
||||||
|
match sweep_once(&pool).await {
|
||||||
|
Ok(r) if r.is_empty() => {}
|
||||||
|
Ok(r) => eprintln!("mission_gc: {r}"),
|
||||||
|
Err(e) => eprintln!("mission_gc: sweep failed: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What one sweep reclaimed.
|
||||||
|
#[derive(Debug, Default, PartialEq)]
|
||||||
|
pub struct Reclaimed {
|
||||||
|
pub orphan_dirs: u64,
|
||||||
|
pub scratch_dirs: u64,
|
||||||
|
pub outputs: u64,
|
||||||
|
pub bytes: u64,
|
||||||
|
/// Directories we tried and failed to remove. Reported rather than swallowed
|
||||||
|
/// — a GC that cannot collect is the thing being fixed.
|
||||||
|
pub failed: u64,
|
||||||
|
/// Rows swept from `mission_events`.
|
||||||
|
pub events: u64,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Reclaimed {
|
||||||
|
pub fn is_empty(&self) -> bool {
|
||||||
|
*self == Reclaimed::default()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for Reclaimed {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(
|
||||||
|
f,
|
||||||
|
"reclaimed {} orphan mission dir(s), {} scratch dir(s), {} output(s), \
|
||||||
|
{} mission event(s), {:.1} MiB{}",
|
||||||
|
self.orphan_dirs,
|
||||||
|
self.scratch_dirs,
|
||||||
|
self.outputs,
|
||||||
|
self.events,
|
||||||
|
self.bytes as f64 / (1024.0 * 1024.0),
|
||||||
|
if self.failed > 0 {
|
||||||
|
format!(" — {} COULD NOT BE REMOVED", self.failed)
|
||||||
|
} else {
|
||||||
|
String::new()
|
||||||
|
}
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn sweep_once(pool: &PgPool) -> Result<Reclaimed, String> {
|
||||||
|
let root = crate::mission_workspace::missions_root();
|
||||||
|
let mut out = Reclaimed::default();
|
||||||
|
reap_orphan_missions(pool, &root, &mut out).await?;
|
||||||
|
reap_scratch(&root, &mut out).await;
|
||||||
|
reap_outputs(pool, &root, &mut out).await;
|
||||||
|
reap_mission_events(pool, &mut out).await;
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How long a mission's structured activity is kept.
|
||||||
|
///
|
||||||
|
/// The World shows the last 24 hours of finished missions, so a week is
|
||||||
|
/// generous and still bounds a table that a single busy coding phase can add
|
||||||
|
/// hundreds of rows to. The per-phase cap bounds ONE phase; this bounds time.
|
||||||
|
const EVENT_RETENTION_DAYS: i32 = 7;
|
||||||
|
|
||||||
|
/// Sweep expired `mission_events`.
|
||||||
|
///
|
||||||
|
/// Bounded per pass rather than deleting the whole backlog in one statement: a
|
||||||
|
/// deployment that has been accumulating for months would otherwise take a long
|
||||||
|
/// lock on its first sweep after this ships. The sweep runs on a timer, so a
|
||||||
|
/// large backlog simply drains over several passes.
|
||||||
|
pub async fn reap_mission_events(pool: &PgPool, out: &mut Reclaimed) {
|
||||||
|
let res = sqlx::query(
|
||||||
|
"DELETE FROM mission_events
|
||||||
|
WHERE id IN (
|
||||||
|
SELECT e.id FROM mission_events e
|
||||||
|
JOIN missions m ON m.id = e.mission_id
|
||||||
|
WHERE e.created_at < now() - make_interval(days => $1)
|
||||||
|
-- A mission under measurement or investigation keeps its
|
||||||
|
-- events. Without this the evidence a Skill-Use baseline or a
|
||||||
|
-- provenance question depends on expires while the question
|
||||||
|
-- is still open, and the answer degrades silently into
|
||||||
|
-- \"there are no events\" — which reads identically to
|
||||||
|
-- \"nothing happened\".
|
||||||
|
AND (m.retain_events_until IS NULL
|
||||||
|
OR m.retain_events_until < now())
|
||||||
|
LIMIT 10000
|
||||||
|
)",
|
||||||
|
)
|
||||||
|
.bind(EVENT_RETENTION_DAYS)
|
||||||
|
.execute(pool)
|
||||||
|
.await;
|
||||||
|
match res {
|
||||||
|
Ok(r) => out.events += r.rows_affected(),
|
||||||
|
Err(e) => eprintln!("mission_gc: sweeping mission_events failed: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Directories under the missions root with no mission row.
|
||||||
|
async fn reap_orphan_missions(
|
||||||
|
pool: &PgPool,
|
||||||
|
root: &Path,
|
||||||
|
out: &mut Reclaimed,
|
||||||
|
) -> Result<(), String> {
|
||||||
|
let Ok(entries) = std::fs::read_dir(root) else {
|
||||||
|
// Not an error: a deployment that has never run a mission has no tree.
|
||||||
|
return Ok(());
|
||||||
|
};
|
||||||
|
for entry in entries.flatten() {
|
||||||
|
let path = entry.path();
|
||||||
|
if !path.is_dir() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let Some(name) = path.file_name().and_then(|n| n.to_str()) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
if RESERVED.contains(&name) || name.starts_with('_') {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// Only well-formed mission ids. A directory this function does not
|
||||||
|
// recognise is one it has no business deleting.
|
||||||
|
let Ok(id) = name.parse::<uuid::Uuid>() else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
if !older_than(&path, ORPHAN_GRACE) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// The DB as predicate, asked per directory.
|
||||||
|
let exists: Option<(uuid::Uuid,)> =
|
||||||
|
sqlx::query_as("SELECT id FROM missions WHERE id = $1")
|
||||||
|
.bind(id)
|
||||||
|
.fetch_optional(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("looking up mission {id}: {e}"))?;
|
||||||
|
if exists.is_some() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let bytes = dir_size(&path);
|
||||||
|
if remove_tree(&path).await {
|
||||||
|
out.orphan_dirs += 1;
|
||||||
|
out.bytes += bytes;
|
||||||
|
} else {
|
||||||
|
out.failed += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `_bench` / `_gate` / `_verify` / `_merge` trees older than their command
|
||||||
|
/// ceilings. These are siblings of the per-mission dirs and have leaked before.
|
||||||
|
async fn reap_scratch(root: &Path, out: &mut Reclaimed) {
|
||||||
|
const SCRATCH: &[&str] = &["_bench", "_gate", "_verify", "_merge"];
|
||||||
|
for name in SCRATCH {
|
||||||
|
let path = root.join(name);
|
||||||
|
if !path.is_dir() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let Ok(entries) = std::fs::read_dir(&path) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
for entry in entries.flatten() {
|
||||||
|
let p = entry.path();
|
||||||
|
if !older_than(&p, SCRATCH_GRACE) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let bytes = dir_size(&p);
|
||||||
|
if remove_tree(&p).await {
|
||||||
|
out.scratch_dirs += 1;
|
||||||
|
out.bytes += bytes;
|
||||||
|
} else {
|
||||||
|
out.failed += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Captured outputs past retention, with their artifact rows marked so nothing
|
||||||
|
/// points at a file that is gone.
|
||||||
|
async fn reap_outputs(pool: &PgPool, root: &Path, out: &mut Reclaimed) {
|
||||||
|
let outputs = root.join("_outputs");
|
||||||
|
let Ok(entries) = std::fs::read_dir(&outputs) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let grace = Duration::from_secs(OUTPUTS_DAYS * 24 * 60 * 60);
|
||||||
|
for entry in entries.flatten() {
|
||||||
|
let p = entry.path();
|
||||||
|
if !p.is_dir() || !older_than(&p, grace) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let Some(id) = p
|
||||||
|
.file_name()
|
||||||
|
.and_then(|n| n.to_str())
|
||||||
|
.and_then(|n| n.parse::<uuid::Uuid>().ok())
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let bytes = dir_size(&p);
|
||||||
|
if !remove_tree(&p).await {
|
||||||
|
out.failed += 1;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// The row is marked only AFTER the files are gone. The other order
|
||||||
|
// leaves a mission whose artifacts claim to be reaped while they are
|
||||||
|
// still on disk, which is a lie in the direction that costs disk.
|
||||||
|
let _ = sqlx::query(
|
||||||
|
"UPDATE mission_artifacts SET metadata = COALESCE(metadata, '{}'::jsonb)
|
||||||
|
|| '{\"reaped\": true}'::jsonb
|
||||||
|
WHERE mission_id = $1",
|
||||||
|
)
|
||||||
|
.bind(id)
|
||||||
|
.execute(pool)
|
||||||
|
.await;
|
||||||
|
out.outputs += 1;
|
||||||
|
out.bytes += bytes;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Remove a tree, escalating to a root purge when our uid cannot.
|
||||||
|
///
|
||||||
|
/// The ONLY deletion path in this module. A second one is how the reap paths
|
||||||
|
/// drifted apart last time.
|
||||||
|
async fn remove_tree(path: &Path) -> bool {
|
||||||
|
match tokio::fs::remove_dir_all(path).await {
|
||||||
|
Ok(()) => true,
|
||||||
|
Err(e) if e.kind() == std::io::ErrorKind::NotFound => true,
|
||||||
|
Err(e) if e.kind() == std::io::ErrorKind::PermissionDenied => {
|
||||||
|
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
||||||
|
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
||||||
|
crate::root_copy::purge(&container, path).await;
|
||||||
|
let gone = tokio::fs::metadata(path).await.is_err();
|
||||||
|
if !gone {
|
||||||
|
eprintln!(
|
||||||
|
"mission_gc: {} survived a root purge — it will keep accumulating",
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
gone
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("mission_gc: could not remove {}: {e}", path.display());
|
||||||
|
false
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn older_than(path: &Path, grace: Duration) -> bool {
|
||||||
|
let Ok(meta) = std::fs::metadata(path) else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
// mtime, not ctime: a directory whose contents changed recently is one
|
||||||
|
// something is still writing to.
|
||||||
|
let Ok(modified) = meta.modified() else {
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
modified
|
||||||
|
.elapsed()
|
||||||
|
.map(|age| age >= grace)
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Apparent size, best-effort. Used only for reporting, so a read error costs a
|
||||||
|
/// wrong number in a log line rather than a wrong decision.
|
||||||
|
fn dir_size(path: &Path) -> u64 {
|
||||||
|
let mut total = 0;
|
||||||
|
let Ok(entries) = std::fs::read_dir(path) else {
|
||||||
|
return 0;
|
||||||
|
};
|
||||||
|
for entry in entries.flatten() {
|
||||||
|
let Ok(meta) = entry.metadata() else { continue };
|
||||||
|
if meta.is_dir() {
|
||||||
|
total += dir_size(&entry.path());
|
||||||
|
} else {
|
||||||
|
total += meta.len();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
total
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn touch_dir(root: &Path, name: &str) -> std::path::PathBuf {
|
||||||
|
let p = root.join(name);
|
||||||
|
std::fs::create_dir_all(&p).unwrap();
|
||||||
|
std::fs::write(p.join("f"), b"x").unwrap();
|
||||||
|
p
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The reserved siblings are never candidates.
|
||||||
|
///
|
||||||
|
/// `_outputs`, `_home` and `_cargo` live under the same root as the mission
|
||||||
|
/// directories. `_cargo` in particular is a SHARED cache every mission
|
||||||
|
/// writes to, so a sweep that treated an underscore-prefixed sibling as an
|
||||||
|
/// orphan mission would delete it out from under running work — and it would
|
||||||
|
/// look like a slow cargo build rather than a bug.
|
||||||
|
#[test]
|
||||||
|
fn siblings_of_the_mission_dirs_are_not_missions() {
|
||||||
|
for name in RESERVED {
|
||||||
|
assert!(
|
||||||
|
name.starts_with('_'),
|
||||||
|
"{name} must be underscore-prefixed so the guard catches it"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
name.parse::<uuid::Uuid>().is_err(),
|
||||||
|
"{name} must not parse as a mission id"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Only a well-formed mission id is ever a candidate.
|
||||||
|
///
|
||||||
|
/// The predicate is "no row exists", and a directory whose name is not an id
|
||||||
|
/// can have no row BY CONSTRUCTION — so name-parsing has to gate the lookup,
|
||||||
|
/// or every unrecognised directory looks like an orphan.
|
||||||
|
#[test]
|
||||||
|
fn a_directory_that_is_not_a_mission_id_is_never_a_candidate() {
|
||||||
|
for name in ["_outputs", "_cargo", "lost+found", "notes", "019fe8", ""] {
|
||||||
|
assert!(
|
||||||
|
name.parse::<uuid::Uuid>().is_err(),
|
||||||
|
"{name:?} must not parse as a mission id"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
assert!("019fe82e-7f0d-7481-a197-698f1d400419"
|
||||||
|
.parse::<uuid::Uuid>()
|
||||||
|
.is_ok());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The grace window is real, and measured from mtime.
|
||||||
|
#[test]
|
||||||
|
fn a_fresh_directory_is_never_old_enough() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let d = touch_dir(tmp.path(), "019fe82e-7f0d-7481-a197-698f1d400419");
|
||||||
|
assert!(!older_than(&d, ORPHAN_GRACE));
|
||||||
|
// And a zero grace makes everything eligible, which is what proves the
|
||||||
|
// check is the window rather than an accident of the filesystem.
|
||||||
|
assert!(older_than(&d, Duration::from_secs(0)));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One deletion path, and it escalates.
|
||||||
|
///
|
||||||
|
/// A second removal site is how the container reap paths drifted apart and
|
||||||
|
/// leaked for a day. The escalation is the other half: the server is uid
|
||||||
|
/// 65532 and cannot delete what the per-mission daemon left as root.
|
||||||
|
#[test]
|
||||||
|
fn there_is_exactly_one_deletion_path_and_it_escalates() {
|
||||||
|
let src = include_str!("mission_gc.rs");
|
||||||
|
assert_eq!(
|
||||||
|
src.matches(concat!("remove_dir", "_all(")).count(),
|
||||||
|
1,
|
||||||
|
"exactly one removal site"
|
||||||
|
);
|
||||||
|
assert!(src.contains("root_copy::purge"), "and it must escalate");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,240 @@
|
|||||||
|
//! Project memory: what past missions on a repository learned.
|
||||||
|
//!
|
||||||
|
//! Until 2026-09-20 missions wrote no memory at all. The chat path records
|
||||||
|
//! every turn into the claw's `.brain`, but a mission's crew is minted per
|
||||||
|
//! mission (`per-mission-crews`: reuse is OFF by operator decision), so a
|
||||||
|
//! brain keyed by agent would be written once and never read. What persists
|
||||||
|
//! across missions is the repository. So the memory is keyed by `repo_id`:
|
||||||
|
//! one `.brain` per repo, holding the judge's verdicts, recalled by the next
|
||||||
|
//! mission's task text and placed in its brief.
|
||||||
|
//!
|
||||||
|
//! What is remembered is the verdict, not the work: for a met phase the
|
||||||
|
//! judge's `reason` (what it found), for an unmet one its `guidance` — the
|
||||||
|
//! agent-facing half, already stripped of acceptance literals by
|
||||||
|
//! `evaluator::sanitize_guidance`, because a verdict quoted verbatim into
|
||||||
|
//! the next brief is how the 2026-08-01 Goodhart incident happened.
|
||||||
|
//!
|
||||||
|
//! Recall is BM25 over the keyword index (`cm_brain::ClawBrain::recall`);
|
||||||
|
//! there is no embedder. Measured before anything richer is built: the test
|
||||||
|
//! is a second mission on the same repo recalling the first's verdict.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
|
||||||
|
use cm_brain::ClawBrain;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// How many past verdicts a brief carries. Three is enough to say "this was
|
||||||
|
/// tried" without becoming the prompt.
|
||||||
|
pub const RECALL_K: usize = 3;
|
||||||
|
|
||||||
|
/// The heading the recalled lines go under. Named here because the scorer and
|
||||||
|
/// the prompt-order tests read it back.
|
||||||
|
pub const SECTION_HEADING: &str = "# What past missions on this repository learned";
|
||||||
|
|
||||||
|
fn brain_path(dir: &Path, repo_id: Uuid) -> PathBuf {
|
||||||
|
dir.join(format!("repo_{repo_id}.h5"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One line of memory from a verdict. Pure, so the shape is testable without
|
||||||
|
/// a brain file.
|
||||||
|
pub fn verdict_line(
|
||||||
|
mission_id: Uuid,
|
||||||
|
phase_kind: &str,
|
||||||
|
condition: &str,
|
||||||
|
verdict: &crate::evaluator::Verdict,
|
||||||
|
) -> Option<String> {
|
||||||
|
// A judge that could not be reached has not judged; there is no lesson.
|
||||||
|
if verdict.error.is_some() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let outcome = if verdict.met { "MET" } else { "UNMET" };
|
||||||
|
let finding = if verdict.met {
|
||||||
|
verdict.reason.trim()
|
||||||
|
} else {
|
||||||
|
verdict.guidance.trim()
|
||||||
|
};
|
||||||
|
if finding.is_empty() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let short = mission_id.simple().to_string();
|
||||||
|
Some(format!(
|
||||||
|
"{outcome} — {phase_kind} phase of mission {} — condition: {} — judge: {}",
|
||||||
|
&short[..8],
|
||||||
|
head(condition, 200),
|
||||||
|
head(finding, 400),
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record a verdict in the repo's brain. Best-effort and loud on failure:
|
||||||
|
/// memory must never fail a phase, and a brain that silently stopped
|
||||||
|
/// recording is the kind of thing that stays broken for a month.
|
||||||
|
pub fn remember_verdict(
|
||||||
|
repo_id: Uuid,
|
||||||
|
mission_id: Uuid,
|
||||||
|
phase_kind: &str,
|
||||||
|
condition: &str,
|
||||||
|
verdict: &crate::evaluator::Verdict,
|
||||||
|
) {
|
||||||
|
remember_in(
|
||||||
|
&cm_runtime::brain::brain_dir(),
|
||||||
|
repo_id,
|
||||||
|
mission_id,
|
||||||
|
phase_kind,
|
||||||
|
condition,
|
||||||
|
verdict,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn remember_in(
|
||||||
|
dir: &Path,
|
||||||
|
repo_id: Uuid,
|
||||||
|
mission_id: Uuid,
|
||||||
|
phase_kind: &str,
|
||||||
|
condition: &str,
|
||||||
|
verdict: &crate::evaluator::Verdict,
|
||||||
|
) {
|
||||||
|
let Some(line) = verdict_line(mission_id, phase_kind, condition, verdict) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let path = brain_path(dir, repo_id);
|
||||||
|
if let Some(dir) = path.parent() {
|
||||||
|
let _ = std::fs::create_dir_all(dir);
|
||||||
|
}
|
||||||
|
match ClawBrain::open_or_create(&path, &format!("repo_{repo_id}")) {
|
||||||
|
Ok(mut brain) => {
|
||||||
|
if let Err(e) = brain.remember("judge", &line, &mission_id.to_string()) {
|
||||||
|
eprintln!("mission_memory: could not record verdict for repo {repo_id}: {e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(e) => eprintln!("mission_memory: could not open brain for repo {repo_id}: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The past verdicts most relevant to `query` (the phase's task text).
|
||||||
|
/// Empty when the repo has no brain yet, which is every repo's first mission.
|
||||||
|
pub fn recall(repo_id: Uuid, query: &str) -> Vec<String> {
|
||||||
|
recall_in(&cm_runtime::brain::brain_dir(), repo_id, query)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn recall_in(dir: &Path, repo_id: Uuid, query: &str) -> Vec<String> {
|
||||||
|
let path = brain_path(dir, repo_id);
|
||||||
|
if !path.exists() {
|
||||||
|
return Vec::new();
|
||||||
|
}
|
||||||
|
match ClawBrain::open_or_create(&path, &format!("repo_{repo_id}")) {
|
||||||
|
Ok(brain) => brain
|
||||||
|
.recall(query, RECALL_K)
|
||||||
|
.into_iter()
|
||||||
|
// `remember` stores "role: text"; the role is ours and not a lesson.
|
||||||
|
.map(|m| m.strip_prefix("judge: ").map(str::to_string).unwrap_or(m))
|
||||||
|
.collect(),
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("mission_memory: could not open brain for repo {repo_id}: {e}");
|
||||||
|
Vec::new()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The section a brief carries, or nothing when there is nothing to say —
|
||||||
|
/// an empty heading tells the agent there is history and then shows none.
|
||||||
|
pub fn section(recalled: &[String]) -> Option<String> {
|
||||||
|
if recalled.is_empty() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let mut out = String::from(SECTION_HEADING);
|
||||||
|
out.push_str(
|
||||||
|
"\n\nJudge verdicts from earlier missions here, most relevant first. \
|
||||||
|
They say what was checked and what was found; they are not the task.\n",
|
||||||
|
);
|
||||||
|
for line in recalled {
|
||||||
|
out.push_str("- ");
|
||||||
|
out.push_str(line);
|
||||||
|
out.push('\n');
|
||||||
|
}
|
||||||
|
Some(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn head(s: &str, n: usize) -> String {
|
||||||
|
match s.char_indices().nth(n) {
|
||||||
|
Some((i, _)) => format!("{}…", &s[..i]),
|
||||||
|
None => s.to_string(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use crate::evaluator::{Usage, Verdict};
|
||||||
|
|
||||||
|
fn verdict(met: bool, reason: &str, guidance: &str, error: Option<&str>) -> Verdict {
|
||||||
|
Verdict {
|
||||||
|
met,
|
||||||
|
reason: reason.into(),
|
||||||
|
guidance: guidance.into(),
|
||||||
|
model: "m".into(),
|
||||||
|
error: error.map(str::to_string),
|
||||||
|
checks: Vec::new(),
|
||||||
|
independent: true,
|
||||||
|
usage: Usage::default(),
|
||||||
|
expectation: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Unmet carries the sanitized guidance, never the operator reason —
|
||||||
|
/// the reason may quote the acceptance text the next mission must earn.
|
||||||
|
#[test]
|
||||||
|
fn unmet_remembers_guidance_not_reason() {
|
||||||
|
let v = verdict(false, "token ZZQX-9 is absent", "the required marker is absent", None);
|
||||||
|
let line = verdict_line(Uuid::nil(), "coding", "cond", &v).unwrap();
|
||||||
|
assert!(line.starts_with("UNMET — coding phase"));
|
||||||
|
assert!(line.contains("the required marker is absent"));
|
||||||
|
assert!(!line.contains("ZZQX-9"));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn met_remembers_what_the_judge_found() {
|
||||||
|
let v = verdict(true, "MICROVM.md holds both lines", "", None);
|
||||||
|
let line = verdict_line(Uuid::nil(), "coding", "cond", &v).unwrap();
|
||||||
|
assert!(line.starts_with("MET — "));
|
||||||
|
assert!(line.contains("MICROVM.md holds both lines"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No judgement, no lesson.
|
||||||
|
#[test]
|
||||||
|
fn an_unreachable_judge_leaves_no_memory() {
|
||||||
|
let v = verdict(false, "could not evaluate", "could not evaluate", Some("429"));
|
||||||
|
assert!(verdict_line(Uuid::nil(), "coding", "cond", &v).is_none());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn section_is_absent_when_nothing_was_recalled() {
|
||||||
|
assert!(section(&[]).is_none());
|
||||||
|
let s = section(&["MET — x".into()]).unwrap();
|
||||||
|
assert!(s.starts_with(SECTION_HEADING));
|
||||||
|
assert!(s.contains("- MET — x\n"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Round trip through a real brain file: what one mission's verdict
|
||||||
|
/// wrote, a query shaped like the next mission's task recalls.
|
||||||
|
#[test]
|
||||||
|
fn a_second_mission_recalls_the_first_verdict() {
|
||||||
|
let dir = std::env::temp_dir().join(format!("cm-mission-memory-{}", Uuid::now_v7()));
|
||||||
|
let repo = Uuid::now_v7();
|
||||||
|
let v = verdict(
|
||||||
|
true,
|
||||||
|
"BASELINE.md records 0.689 ns/iter from benches/add_bench.rs",
|
||||||
|
"",
|
||||||
|
None,
|
||||||
|
);
|
||||||
|
remember_in(&dir, repo, Uuid::now_v7(), "benchmark", "a baseline is recorded", &v);
|
||||||
|
let got = recall_in(&dir, repo, "record a performance baseline for the hot path");
|
||||||
|
assert_eq!(got.len(), 1, "{got:?}");
|
||||||
|
assert!(got[0].starts_with("MET — benchmark phase"), "{}", got[0]);
|
||||||
|
assert!(!got[0].starts_with("judge: "));
|
||||||
|
// A repo with no history recalls nothing and creates no file.
|
||||||
|
let other = Uuid::now_v7();
|
||||||
|
assert!(recall_in(&dir, other, "anything").is_empty());
|
||||||
|
assert!(!brain_path(&dir, other).exists());
|
||||||
|
let _ = std::fs::remove_dir_all(&dir);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -44,6 +44,7 @@ pub async fn on_launch(
|
|||||||
user_id: cm_domain::UserId,
|
user_id: cm_domain::UserId,
|
||||||
mission_id: Uuid,
|
mission_id: Uuid,
|
||||||
node_hub: Option<std::sync::Arc<crate::fleet::NodeHub>>,
|
node_hub: Option<std::sync::Arc<crate::fleet::NodeHub>>,
|
||||||
|
blobs: Option<std::sync::Arc<dyn cm_files::BlobStore>>,
|
||||||
) -> Result<Option<Uuid>, String> {
|
) -> Result<Option<Uuid>, String> {
|
||||||
eprintln!("mission_orchestrator::on_launch fired mission_id={mission_id}");
|
eprintln!("mission_orchestrator::on_launch fired mission_id={mission_id}");
|
||||||
let Some(mission) = cm_db::repo::missions::get(pool, mission_id, workspace_id.as_uuid())
|
let Some(mission) = cm_db::repo::missions::get(pool, mission_id, workspace_id.as_uuid())
|
||||||
@@ -53,16 +54,97 @@ pub async fn on_launch(
|
|||||||
return Err("mission not found".into());
|
return Err("mission not found".into());
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// A Continuous Research mission harvests BEFORE its checkout is taken.
|
||||||
|
//
|
||||||
|
// The ORDER here is load-bearing and was wrong: the harvest ran after
|
||||||
|
// `ensure_checkout`, so the mission cloned the vault before the manifest
|
||||||
|
// was pushed to it. The reader agent found no harvest.jsonl, and — being
|
||||||
|
// resourceful — queried arXiv itself and wrote its own. That is precisely
|
||||||
|
// what `skills/research/arxiv-daily.md` forbids: the papers it found are
|
||||||
|
// not checked off in `corpus_items`, so the next run re-offers them, and
|
||||||
|
// the 13 the real harvest DID shelve went unread. Harvest first, then
|
||||||
|
// clone, so the checkout contains the manifest.
|
||||||
|
//
|
||||||
|
// Finding papers is not agent work: `library::run_to_vault` searches arXiv,
|
||||||
|
// checks the `corpus_items` seen-set, fetches and verifies each PDF, shelves
|
||||||
|
// it and writes the catalogue note — deterministically, in seconds. The
|
||||||
|
// seen-set is the entire reason a recurring mission knows what it already
|
||||||
|
// covered, and an agent re-searching arXiv would leave it wrong.
|
||||||
|
//
|
||||||
|
// Deliberately NON-FATAL. A harvest that fails still lets the phases run,
|
||||||
|
// because the phase is what reports whether today was quiet or broken, and
|
||||||
|
// those must stay distinguishable. What is never acceptable is silence, so
|
||||||
|
// both outcomes are logged with their counts.
|
||||||
|
let mut harvested: Vec<crate::papers::Paper> = Vec::new();
|
||||||
|
if mission.template_kind == crate::continuous_research::TEMPLATE_KIND {
|
||||||
|
match blobs.as_ref() {
|
||||||
|
Some(b) => {
|
||||||
|
let topics = crate::continuous_research::topics_for(&mission.config);
|
||||||
|
match crate::continuous_research::harvest_for_mission(
|
||||||
|
pool,
|
||||||
|
b,
|
||||||
|
workspace_id.as_uuid(),
|
||||||
|
mission_id,
|
||||||
|
&topics,
|
||||||
|
5,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(papers) => {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: continuous research harvest shelved {} paper(s) for mission {mission_id}",
|
||||||
|
papers.len()
|
||||||
|
);
|
||||||
|
harvested = papers;
|
||||||
|
}
|
||||||
|
Err(e) => eprintln!(
|
||||||
|
"mission_orchestrator: continuous research harvest FAILED for {mission_id} (phases still start, and will report an empty day): {e}"
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Not a warning to bury: without blob storage there is nowhere to
|
||||||
|
// shelve a PDF, so the mission will find an empty manifest and
|
||||||
|
// correctly report that nothing arrived.
|
||||||
|
None => eprintln!(
|
||||||
|
"mission_orchestrator: mission {mission_id} is continuous_research but blob storage is not configured — no harvest, so today's manifest will be empty"
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
// ensure_checkout is idempotent (fetch+reset on existing clones,
|
// ensure_checkout is idempotent (fetch+reset on existing clones,
|
||||||
// clone on missing dirs) so we run it BEFORE the team_id short-
|
// clone on missing dirs) so we run it BEFORE the team_id short-
|
||||||
// circuit: a re-launched or retried mission still needs a fresh
|
// circuit: a re-launched or retried mission still needs a fresh
|
||||||
// repo checkout even though its team was minted on the first
|
// repo checkout even though its team was minted on the first
|
||||||
// launch. Non-fatal — logs and continues on failure.
|
// launch. Non-fatal — logs and continues on failure.
|
||||||
match crate::mission_workspace::ensure_checkout(pool, workspace_id, mission_id).await {
|
match crate::mission_workspace::ensure_checkout(pool, workspace_id, mission_id).await {
|
||||||
Ok(Some(path)) => eprintln!(
|
Ok(Some(path)) => {
|
||||||
|
eprintln!(
|
||||||
"mission_orchestrator: repo checked out at {} for mission {mission_id}",
|
"mission_orchestrator: repo checked out at {} for mission {mission_id}",
|
||||||
path.display()
|
path.display()
|
||||||
|
);
|
||||||
|
// The manifest goes in the CHECKOUT, not the vault: it is this run's
|
||||||
|
// input, and the vault path is per-date and shared, so a second run
|
||||||
|
// the same day rewrites a file that already exists and auto_merge
|
||||||
|
// rightly refuses the branch. See `write_manifest`.
|
||||||
|
if mission.template_kind == crate::continuous_research::TEMPLATE_KIND {
|
||||||
|
let date = crate::continuous_research::today();
|
||||||
|
match crate::continuous_research::write_manifest(&path, &harvested, &date) {
|
||||||
|
Ok(at) => eprintln!(
|
||||||
|
"mission_orchestrator: wrote {} paper(s) to {}",
|
||||||
|
harvested.len(),
|
||||||
|
at.display()
|
||||||
),
|
),
|
||||||
|
// Loud: the reader phase would find no manifest and, being
|
||||||
|
// resourceful, go and search arXiv itself — which corrupts
|
||||||
|
// the seen-set. Better to see why here.
|
||||||
|
Err(e) => eprintln!(
|
||||||
|
"mission_orchestrator: could NOT write the harvest manifest for \
|
||||||
|
{mission_id} — the reader phase will see no papers: {e}"
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Ok(None) => eprintln!(
|
Ok(None) => eprintln!(
|
||||||
"mission_orchestrator: mission {mission_id} has no repo bound, skipping checkout"
|
"mission_orchestrator: mission {mission_id} has no repo bound, skipping checkout"
|
||||||
),
|
),
|
||||||
@@ -79,10 +161,22 @@ pub async fn on_launch(
|
|||||||
// The mission's own runtime endpoint. Claws MUST be provisioned against
|
// The mission's own runtime endpoint. Claws MUST be provisioned against
|
||||||
// THIS gateway, not the global one — see RuntimeProvisioner::for_gateway.
|
// THIS gateway, not the global one — see RuntimeProvisioner::for_gateway.
|
||||||
let mut mission_gateway: Option<String> = None;
|
let mut mission_gateway: Option<String> = None;
|
||||||
if let Some(prov) = crate::mission_runtime::MissionRuntimeProvisioner::from_env() {
|
// Not for a microVM mission: the ZeroClaw daemon it would start is never
|
||||||
|
// spoken to, and it would sit holding a pairing code and ~3 GB of image for
|
||||||
|
// the life of the mission. Observed doing exactly that on the first real run.
|
||||||
|
if let Some(prov) = crate::mission_runtime::MissionRuntimeProvisioner::from_env()
|
||||||
|
.filter(|_| mission.runtime_kind != "microvm")
|
||||||
|
{
|
||||||
match prov.ensure_container(mission_id).await {
|
match prov.ensure_container(mission_id).await {
|
||||||
Ok(ec) => {
|
Ok(ec) => {
|
||||||
mission_gateway = Some(ec.endpoint.clone());
|
mission_gateway = Some(ec.endpoint.clone());
|
||||||
|
crate::container_tool_hooks::record_install(
|
||||||
|
pool,
|
||||||
|
mission_id,
|
||||||
|
None,
|
||||||
|
ec.hooks.as_deref(),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
let container_name = crate::mission_runtime::container_name(mission_id);
|
let container_name = crate::mission_runtime::container_name(mission_id);
|
||||||
if let Err(e) = cm_db::repo::missions::set_runtime_binding(
|
if let Err(e) = cm_db::repo::missions::set_runtime_binding(
|
||||||
pool,
|
pool,
|
||||||
@@ -115,6 +209,76 @@ pub async fn on_launch(
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// microVM PLACEMENT MUST COME BEFORE the early return below. It did not, and
|
||||||
|
// the first real microvm mission failed with "mission has no target_node_id" —
|
||||||
|
// the executor's own guard firing correctly on a mission this function had
|
||||||
|
// returned from before ever choosing a node for it.
|
||||||
|
// microVM placement. KVM is a hard predicate, not a preference: gw-04 —
|
||||||
|
// where every mission runs today — is itself a VM without nested
|
||||||
|
// virtualisation and has no /dev/kvm, so a microvm mission landing there
|
||||||
|
// cannot start. Resolve a capable node now and fail the launch if there is
|
||||||
|
// none, because the alternative is a mission that sits in 'running' having
|
||||||
|
// never had anywhere to run.
|
||||||
|
if mission.runtime_kind == "microvm" {
|
||||||
|
// Capable means BOTH: it can host a microVM, and it holds the image this
|
||||||
|
// mission's backend names. Asking only for `microvm` sent the first real
|
||||||
|
// microVM mission to a node without `rootfs-claude.ext4`.
|
||||||
|
let backend = mission.backend.as_deref();
|
||||||
|
let capable =
|
||||||
|
cm_db::repo::nodes::online_for_backend(pool, mission.workspace_id, backend)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("looking up nodes for backend {backend:?}: {e}"))?;
|
||||||
|
let how_to_fix = format!(
|
||||||
|
"needs /dev/kvm + firecracker (scripts/fc-node-setup.sh) AND the {} rootfs \
|
||||||
|
built on that node (scripts/fc-build-rootfs.sh <host> <image> {})",
|
||||||
|
backend.unwrap_or("default"),
|
||||||
|
backend.unwrap_or("<name>")
|
||||||
|
);
|
||||||
|
let how_to_fix = how_to_fix.as_str();
|
||||||
|
// CAPABILITY is checked here; CAPACITY is not, and no node is pinned.
|
||||||
|
//
|
||||||
|
// Placement moved to phase launch (`phase_runner`). A node chosen now
|
||||||
|
// would be chosen once, minutes before the first VM boots and hours
|
||||||
|
// before the last — and re-placing between phases is free, because
|
||||||
|
// mission state lives on the gateway checkout and every VM is
|
||||||
|
// inject → run → collect → destroy. Pinning early bought nothing and
|
||||||
|
// cost the ability to react to a node filling or draining mid-mission.
|
||||||
|
//
|
||||||
|
// Launching still FAILS here when no node could ever run this backend:
|
||||||
|
// that is not transient, waiting will not fix it, and the harness's
|
||||||
|
// `microvm-negctl` scenario asserts such a mission stays `draft`.
|
||||||
|
if capable.is_empty() {
|
||||||
|
return Err(format!(
|
||||||
|
"no online node can run backend {:?} — {how_to_fix}",
|
||||||
|
backend.unwrap_or("default")
|
||||||
|
));
|
||||||
|
}
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: mission {mission_id} has {} node(s) able to run \
|
||||||
|
backend {:?}; placement happens per phase",
|
||||||
|
capable.len(),
|
||||||
|
backend.unwrap_or("default")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// A microVM mission materialises no team. Its phases run as one `claude -p`
|
||||||
|
// inside a VM (`microvm_executor`), so there is no claw graph to provision —
|
||||||
|
// and demanding one rejected the launch of a well-formed mission with "pick
|
||||||
|
// teams in the wizard". This is the third of three team gates on a path that
|
||||||
|
// uses no teams; the other two are in `routes::missions` (draft→running) and
|
||||||
|
// `phase_runner::launch_phase` (no matching teams → stay pending).
|
||||||
|
//
|
||||||
|
// Returning before the picks below, not filtering them, because provisioning
|
||||||
|
// claws that never run is not a cheaper version of the same thing — it is a
|
||||||
|
// runtime binding and a pairing code describing something nothing uses.
|
||||||
|
if mission.runtime_kind == "microvm" {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: mission {mission_id} is a microvm mission — no team to \
|
||||||
|
materialise; its phases execute in a VM"
|
||||||
|
);
|
||||||
|
return Ok(None);
|
||||||
|
}
|
||||||
|
|
||||||
// Skip team materialization if already bound.
|
// Skip team materialization if already bound.
|
||||||
if mission.team_id.is_some() {
|
if mission.team_id.is_some() {
|
||||||
eprintln!(
|
eprintln!(
|
||||||
@@ -174,6 +338,49 @@ pub async fn on_launch(
|
|||||||
Some(url) => RuntimeProvisioner::for_gateway(url),
|
Some(url) => RuntimeProvisioner::for_gateway(url),
|
||||||
None => RuntimeProvisioner::from_env(),
|
None => RuntimeProvisioner::from_env(),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Tell the daemon where the hooks are. `container_tool_hooks::install`
|
||||||
|
// wrote them; this is what makes claude read them. Doing one without the
|
||||||
|
// other leaves a gate that is installed and inert, which looks exactly
|
||||||
|
// like a gate that found nothing.
|
||||||
|
if let Some(p) = provisioner.as_ref() {
|
||||||
|
if let Err(e) = p
|
||||||
|
.set_claude_cli_settings(crate::container_tool_hooks::SETTINGS_PATH)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: could not point claude_cli at the hook \
|
||||||
|
settings ({e}) — this mission's tool calls run unchecked"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// Only when this mission got its OWN container — the shared runtime is
|
||||||
|
// not ours to reconfigure, and `mission_gateway` being Some is exactly
|
||||||
|
// the signal that `ensure_container` ran.
|
||||||
|
// What a retrieval arm retrieves FROM is installed here, per arm: the
|
||||||
|
// MCP door for `index`, the skill files for `files`. `inline` installs
|
||||||
|
// nothing and `installed` is irrelevant to it.
|
||||||
|
let requested = crate::skill_delivery::requested_for(&mission.config);
|
||||||
|
let container = crate::mission_runtime::container_name(mission_id);
|
||||||
|
let installed = match requested {
|
||||||
|
_ if mission_gateway.is_none() => false,
|
||||||
|
crate::skill_delivery::Mode::Index => {
|
||||||
|
install_skills_door(pool, user_id, mission_id, &container, p).await
|
||||||
|
}
|
||||||
|
crate::skill_delivery::Mode::Files => {
|
||||||
|
install_skill_files(pool, workspace_id, mission_id, &container).await
|
||||||
|
}
|
||||||
|
crate::skill_delivery::Mode::Inline => false,
|
||||||
|
};
|
||||||
|
// Decided here and recorded, not re-derived per turn: this is the only
|
||||||
|
// point that knows whether the door actually installed, and an arm that
|
||||||
|
// could change mid-mission would make the run unattributable.
|
||||||
|
record_skill_delivery(
|
||||||
|
pool,
|
||||||
|
mission_id,
|
||||||
|
crate::skill_delivery::resolve(requested, installed),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
}
|
||||||
let mut first_team_id: Option<Uuid> = None;
|
let mut first_team_id: Option<Uuid> = None;
|
||||||
let mut provisioned_claws: Vec<cm_domain::AgentId> = Vec::new();
|
let mut provisioned_claws: Vec<cm_domain::AgentId> = Vec::new();
|
||||||
for (purpose, template_id) in &picks {
|
for (purpose, template_id) in &picks {
|
||||||
@@ -193,7 +400,7 @@ pub async fn on_launch(
|
|||||||
provisioner: provisioner.as_ref(),
|
provisioner: provisioner.as_ref(),
|
||||||
template: &template,
|
template: &template,
|
||||||
team_name: &team_name,
|
team_name: &team_name,
|
||||||
default_model: "claude-sonnet-5",
|
default_model: MINTED_CLAW_MODEL,
|
||||||
},
|
},
|
||||||
&mut provisioned_claws,
|
&mut provisioned_claws,
|
||||||
)
|
)
|
||||||
@@ -228,31 +435,41 @@ pub async fn on_launch(
|
|||||||
// is a PathBuf the prop-schema won't expose — see provision_claw), so
|
// is a PathBuf the prop-schema won't expose — see provision_claw), so
|
||||||
// we patch the shared config file directly on the per-mission runtime
|
// we patch the shared config file directly on the per-mission runtime
|
||||||
// container. The daemon picks it up on the same reload that surfaces
|
// container. The daemon picks it up on the same reload that surfaces
|
||||||
// the freshly-provisioned claws for the run. Non-fatal: without the
|
// the freshly-provisioned claws for the run.
|
||||||
// pin, agents still write (to the sandbox) but the committer can't
|
//
|
||||||
// find the changes in /mission/repo.
|
// FATAL, deliberately. This was "non-fatal: agents still write (to the
|
||||||
if !provisioned_claws.is_empty() && mission_gateway.is_some() {
|
// sandbox) but the committer can't find the changes in /mission/repo" —
|
||||||
if let Some(mp) = crate::mission_runtime::MissionRuntimeProvisioner::from_env() {
|
// which is to say, the mission runs to completion and delivers nothing.
|
||||||
match mp
|
// Mission `019fcf62` did exactly that: the pin failed with `argument list
|
||||||
.pin_agent_workspaces(mission_id, &provisioned_claws, "/mission/repo")
|
// too long`, one line of stderr scrolled past, and phase 0 reported
|
||||||
.await
|
// `completed` with zero files, no commit error and no push error. A launch
|
||||||
|
// that cannot bind its agents to the repo has no path to delivering work,
|
||||||
|
// so it must fail at launch where someone is still looking.
|
||||||
|
//
|
||||||
|
// Not for a microVM mission: its agent is a `claude -p` inside a VM on a
|
||||||
|
// fleet node, not a ZeroClaw claw in a container here, so there is no
|
||||||
|
// workspace to pin. Leaving it would make a microVM launch FAIL on a
|
||||||
|
// container it was never going to use.
|
||||||
|
if !provisioned_claws.is_empty() && mission_gateway.is_some() && mission.runtime_kind != "microvm"
|
||||||
{
|
{
|
||||||
Ok(()) => {
|
if let Some(mp) = crate::mission_runtime::MissionRuntimeProvisioner::from_env() {
|
||||||
// The daemon reads config ONCE at boot and never re-reads
|
mp.pin_agent_workspaces(mission_id, &provisioned_claws, "/mission/repo")
|
||||||
// the file, so the pin is invisible until it restarts. Its
|
.await
|
||||||
// agents were created through its own config API, so they
|
.map_err(|e| {
|
||||||
// are already persisted to the file and survive the
|
format!(
|
||||||
// restart; the pairing code is re-minted on every launch.
|
"could not pin agent workspaces to /mission/repo ({e}) — the mission \
|
||||||
if let Err(e) = mp.restart_container(mission_id).await {
|
would run with its agents writing to their sandboxes, delivering nothing"
|
||||||
eprintln!(
|
)
|
||||||
"mission_orchestrator: restart runtime for {mission_id} failed (continuing, workspace pin will not apply): {e}"
|
})?;
|
||||||
);
|
// The daemon reads config ONCE at boot and never re-reads the
|
||||||
}
|
// file, so the pin is invisible until it restarts. Its agents were
|
||||||
}
|
// created through its own config API, so they are already
|
||||||
Err(e) => eprintln!(
|
// persisted to the file and survive the restart; the pairing code
|
||||||
"mission_orchestrator: pin workspaces for mission {mission_id} failed (continuing): {e}"
|
// is re-minted on every launch. Equally fatal: an unrestarted
|
||||||
),
|
// daemon is an unpinned daemon.
|
||||||
}
|
mp.restart_container(mission_id).await.map_err(|e| {
|
||||||
|
format!("could not restart the runtime to apply the workspace pin: {e}")
|
||||||
|
})?;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -365,6 +582,17 @@ async fn mint_team_from_template(
|
|||||||
.await
|
.await
|
||||||
.map_err(|e| format!("stamp template lineage: {e}"))?;
|
.map_err(|e| format!("stamp template lineage: {e}"))?;
|
||||||
|
|
||||||
|
// Names already on this workspace's roster, so a newly hired claw does not
|
||||||
|
// arrive sharing a name with someone already here. Read ONCE — a roster
|
||||||
|
// query per role would be N queries to answer one question — and extended
|
||||||
|
// locally as we mint, which also keeps names distinct WITHIN this team.
|
||||||
|
let mut taken_names: Vec<String> = cm_db::repo::agents::roster(pool, workspace_id)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("read roster for naming: {e}"))?
|
||||||
|
.into_iter()
|
||||||
|
.map(|a| a.name)
|
||||||
|
.collect();
|
||||||
|
|
||||||
// For each role: create agent, provision runtime, ingest brain
|
// For each role: create agent, provision runtime, ingest brain
|
||||||
// seed, record link, bind to topology node.
|
// seed, record link, bind to topology node.
|
||||||
for (idx, role) in template.roles.iter().enumerate() {
|
for (idx, role) in template.roles.iter().enumerate() {
|
||||||
@@ -378,10 +606,63 @@ async fn mint_team_from_template(
|
|||||||
template.roles.len(),
|
template.roles.len(),
|
||||||
));
|
));
|
||||||
};
|
};
|
||||||
|
// Every mission gets its OWN crew.
|
||||||
|
//
|
||||||
|
// This deliberately reverses the reuse added earlier. Reuse hired the
|
||||||
|
// existing claw for a (template, slot) so the roster stayed at one team
|
||||||
|
// and "My Workforce" was people you keep — but it also meant every
|
||||||
|
// mission was staffed by the same five names, and the workforce view
|
||||||
|
// showed one crew repeated down the page with nothing to tell the
|
||||||
|
// missions apart. Chosen by the operator: distinct crews read better
|
||||||
|
// than a bounded roster.
|
||||||
|
//
|
||||||
|
// The cost is real and is the cost that reuse existed to avoid: claws
|
||||||
|
// are `lifecycle = 'permanent'` and nothing reaps them until their
|
||||||
|
// MISSION is deleted, so the roster now grows by the team size on every
|
||||||
|
// mission. `agent_names::pick` keeps names unique workspace-wide and
|
||||||
|
// falls back to a numeric suffix once the pool is exhausted, so growth
|
||||||
|
// degrades the naming gracefully rather than colliding.
|
||||||
|
//
|
||||||
|
// `reusable_claw` in cm-db is kept, with its tests: this is a policy
|
||||||
|
// choice that has now flipped twice, and the query is the hard part.
|
||||||
|
let reused: Option<uuid::Uuid> = None;
|
||||||
|
|
||||||
|
// Seed the name choice from the claw's OWN id, not its position in the
|
||||||
|
// team.
|
||||||
|
//
|
||||||
|
// Seeding with the role index (0..n) started every crew near the top of
|
||||||
|
// the pool and took the next free names, so the first mission hired
|
||||||
|
// Aarav, Abebe, Adaora, Adrian, Agnieszka — correct, unique, and
|
||||||
|
// transparently alphabetical. A crew should look like a team, not like
|
||||||
|
// a listing. UUIDv7 puts its random bytes LAST (the leading bytes are a
|
||||||
|
// timestamp, which would cluster again), so the tail is what spreads
|
||||||
|
// the five picks across the whole pool.
|
||||||
|
let agent_id = cm_domain::AgentId::new();
|
||||||
|
let name_seed = {
|
||||||
|
let uuid = agent_id.as_uuid();
|
||||||
|
let b = uuid.as_bytes();
|
||||||
|
u64::from_le_bytes([b[8], b[9], b[10], b[11], b[12], b[13], b[14], b[15]])
|
||||||
|
};
|
||||||
|
|
||||||
let agent = Agent {
|
let agent = Agent {
|
||||||
id: cm_domain::AgentId::new(),
|
id: agent_id,
|
||||||
workspace_id,
|
workspace_id,
|
||||||
name: format!("{} · {}", team_name, role.slot),
|
// A PERSON's name, with the role in `job_title`.
|
||||||
|
//
|
||||||
|
// This was `"{mission title} · {purpose} · {template} · {slot}"` —
|
||||||
|
// names like "verify: a repo-less research mission keeps its output
|
||||||
|
// · mission · Rust SDLC · planner", unreadable in the roster, the
|
||||||
|
// API and every log line at once. Then it was the bare slot, which
|
||||||
|
// fixed the length but made the UI show the same word twice (name
|
||||||
|
// on top, role beneath) and made a roster of five read as five job
|
||||||
|
// tickets rather than a crew.
|
||||||
|
//
|
||||||
|
// The role still lives in `job_title`, which is what the mission
|
||||||
|
// machinery binds on — `team_members.role_slot` and the topology
|
||||||
|
// node carry the slot, so nothing downstream keys off the display
|
||||||
|
// name. Only the reused branch below ignores this, deliberately: a
|
||||||
|
// claw you already hired keeps the name it already had.
|
||||||
|
name: crate::agent_names::pick(&taken_names, name_seed),
|
||||||
job_title: role.slot.clone(),
|
job_title: role.slot.clone(),
|
||||||
// This is the ONLY consumer of the templates' `system_prompt` prose,
|
// This is the ONLY consumer of the templates' `system_prompt` prose,
|
||||||
// and it feeds the *chat* path, not missions: it lands in
|
// and it feeds the *chat* path, not missions: it lands in
|
||||||
@@ -398,12 +679,40 @@ async fn mint_team_from_template(
|
|||||||
managed_by: user_id,
|
managed_by: user_id,
|
||||||
status: AgentStatus::Online,
|
status: AgentStatus::Online,
|
||||||
};
|
};
|
||||||
|
let claw_id = match reused {
|
||||||
|
Some(existing) => {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: reusing claw {existing} for role {} \
|
||||||
|
(template {})",
|
||||||
|
role.slot, template.template.id
|
||||||
|
);
|
||||||
|
existing
|
||||||
|
}
|
||||||
|
None => {
|
||||||
cm_db::repo::agents::insert(pool, &agent, &AccessPolicy::default())
|
cm_db::repo::agents::insert(pool, &agent, &AccessPolicy::default())
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("insert agent {}: {e}", role.slot))?;
|
.map_err(|e| format!("insert agent {}: {e}", role.slot))?;
|
||||||
let claw_id = agent.id.as_uuid();
|
// Claim the name for the rest of this loop. Without this the
|
||||||
|
// roster snapshot taken before the loop is stale from the
|
||||||
|
// second role onward and a five-person team can arrive with
|
||||||
|
// two Merediths.
|
||||||
|
taken_names.push(agent.name.clone());
|
||||||
|
agent.id.as_uuid()
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let agent_id = cm_domain::AgentId::from(claw_id);
|
||||||
|
|
||||||
cm_db::repo::agents::set_model_binding(pool, agent.id, default_model)
|
// The ROLE's model when the template names one, else the mint's default.
|
||||||
|
// Before migration 0071 there was no role model at all, so every claw of
|
||||||
|
// every mission team ran the same one — including a reviewer reviewing
|
||||||
|
// the coder it shares a model with.
|
||||||
|
let role_model = role
|
||||||
|
.model
|
||||||
|
.as_deref()
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|m| !m.is_empty())
|
||||||
|
.unwrap_or(default_model);
|
||||||
|
cm_db::repo::agents::set_model_binding(pool, agent_id, role_model)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("set_model_binding {claw_id}: {e}"))?;
|
.map_err(|e| format!("set_model_binding {claw_id}: {e}"))?;
|
||||||
|
|
||||||
@@ -420,10 +729,15 @@ async fn mint_team_from_template(
|
|||||||
// out-of-band via MissionRuntimeProvisioner::pin_agent_workspaces.
|
// out-of-band via MissionRuntimeProvisioner::pin_agent_workspaces.
|
||||||
if let Some(p) = provisioner {
|
if let Some(p) = provisioner {
|
||||||
match p
|
match p
|
||||||
.provision_claw(claw_id, default_model, &template.template.risk_profile)
|
.provision_claw(
|
||||||
|
claw_id,
|
||||||
|
role_model,
|
||||||
|
&template.template.risk_profile,
|
||||||
|
&template.template.mcp_bundles,
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
Ok(_) => provisioned_claws.push(agent.id),
|
Ok(_) => provisioned_claws.push(agent_id),
|
||||||
Err(e) => eprintln!(
|
Err(e) => eprintln!(
|
||||||
"mission_orchestrator: provision claw {claw_id} failed (continuing): {e}"
|
"mission_orchestrator: provision claw {claw_id} failed (continuing): {e}"
|
||||||
),
|
),
|
||||||
@@ -432,6 +746,10 @@ async fn mint_team_from_template(
|
|||||||
|
|
||||||
// Ingest brain seed (Slice 3.5d). Non-fatal on failure —
|
// Ingest brain seed (Slice 3.5d). Non-fatal on failure —
|
||||||
// agent still works from system_prompt alone.
|
// agent still works from system_prompt alone.
|
||||||
|
// Seed only a NEW claw. A reused one carries what it learned on earlier
|
||||||
|
// missions, and re-seeding would overwrite that with the template's
|
||||||
|
// starting point — which is precisely the accumulation reuse exists for.
|
||||||
|
if reused.is_none() {
|
||||||
if let Some(seed) = role.brain_seed.as_deref().filter(|s| !s.trim().is_empty()) {
|
if let Some(seed) = role.brain_seed.as_deref().filter(|s| !s.trim().is_empty()) {
|
||||||
if let Err(e) =
|
if let Err(e) =
|
||||||
crate::brain_seed::ingest(claw_id, seed.to_string(), role.system_prompt.clone())
|
crate::brain_seed::ingest(claw_id, seed.to_string(), role.system_prompt.clone())
|
||||||
@@ -442,6 +760,7 @@ async fn mint_team_from_template(
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Record lineage (Slice 3.5d) so the MCP skills server can
|
// Record lineage (Slice 3.5d) so the MCP skills server can
|
||||||
// merge template default skills with per-agent overrides.
|
// merge template default skills with per-agent overrides.
|
||||||
@@ -470,9 +789,10 @@ async fn mint_team_from_template(
|
|||||||
cm_db::repo::audit::Actor::User(user_id),
|
cm_db::repo::audit::Actor::User(user_id),
|
||||||
"agent.created",
|
"agent.created",
|
||||||
"agent",
|
"agent",
|
||||||
&agent.id.to_string(),
|
&agent_id.to_string(),
|
||||||
serde_json::json!({
|
serde_json::json!({
|
||||||
"name": agent.name,
|
"name": agent.name,
|
||||||
|
"reused": reused.is_some(),
|
||||||
"job_title": agent.job_title,
|
"job_title": agent.job_title,
|
||||||
"source": "mission_orchestrator",
|
"source": "mission_orchestrator",
|
||||||
"template_id": template.template.id.to_string(),
|
"template_id": template.template.id.to_string(),
|
||||||
@@ -486,6 +806,97 @@ async fn mint_team_from_template(
|
|||||||
Ok(team_id)
|
Ok(team_id)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The model a minted claw runs on when its template role does not name one.
|
||||||
|
///
|
||||||
|
/// A DEFAULT now, not a hardcode: `template_roles.model` (migration 0071) lets a
|
||||||
|
/// template put its reviewer on a different model from the coder it reviews,
|
||||||
|
/// which is the correlated failure the cross-provider judge exists to break,
|
||||||
|
/// one layer down. Roles that say nothing still land here, so every template
|
||||||
|
/// that existed before 0071 behaves exactly as it did.
|
||||||
|
const MINTED_CLAW_MODEL: &str = "claude-sonnet-5";
|
||||||
|
|
||||||
|
/// The graph a COMPOSED microVM mission runs, built from its team template
|
||||||
|
/// without minting a single claw.
|
||||||
|
///
|
||||||
|
/// A composed mission needs the template's *shape* — how many nodes, in what
|
||||||
|
/// pattern, playing what roles — and nothing else it carries. Its nodes are VMs,
|
||||||
|
/// so provisioning claws for them would create agents, containers and `.brain`
|
||||||
|
/// files that nothing ever dials; that is exactly why `on_launch` returns early
|
||||||
|
/// for a microVM mission, and this is how the composed path gets its graph
|
||||||
|
/// anyway rather than by undoing that.
|
||||||
|
///
|
||||||
|
/// `purposes` is the phase's purpose list, matched against `config.phase_teams`;
|
||||||
|
/// missions using the legacy single `team_template_id` fall back to it.
|
||||||
|
/// Returns `None` when the mission picked no template at all.
|
||||||
|
pub async fn composed_graph(
|
||||||
|
pool: &PgPool,
|
||||||
|
mission_id: Uuid,
|
||||||
|
purposes: &[&str],
|
||||||
|
) -> Result<Option<serde_json::Value>, String> {
|
||||||
|
let row: Option<(serde_json::Value, Option<Uuid>)> =
|
||||||
|
sqlx::query_as("SELECT config, team_template_id FROM missions WHERE id = $1")
|
||||||
|
.bind(mission_id)
|
||||||
|
.fetch_optional(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("load mission {mission_id}: {e}"))?;
|
||||||
|
let Some((config, legacy_template)) = row else {
|
||||||
|
return Err(format!("mission {mission_id} not found"));
|
||||||
|
};
|
||||||
|
|
||||||
|
// An APPROVED roster wins over the template. It is the more specific answer
|
||||||
|
// — a model sized it for this mission's actual task and a human accepted it
|
||||||
|
// — and it is the only path on which nodes carry per-node backends, which is
|
||||||
|
// how a mission runs more than one provider. Stored already built and
|
||||||
|
// validated (`routes::mission_roster::decide`), so nothing here can turn a
|
||||||
|
// refused roster into a running one.
|
||||||
|
if let Some(roster) = config.get("roster").filter(|v| v.is_object()) {
|
||||||
|
// Parsed rather than trusted: a graph the orchestrator cannot plan would
|
||||||
|
// otherwise be claimed and fail as "missing or invalid graph", which
|
||||||
|
// reads as a runtime fault instead of a bad roster.
|
||||||
|
serde_json::from_value::<cm_topology::TopologyGraph>(roster.clone())
|
||||||
|
.map_err(|e| format!("mission {mission_id}: the approved roster is not a runnable topology: {e}"))?;
|
||||||
|
return Ok(Some(roster.clone()));
|
||||||
|
}
|
||||||
|
|
||||||
|
let template_id = config
|
||||||
|
.get("phase_teams")
|
||||||
|
.and_then(|v| v.as_object())
|
||||||
|
.and_then(|pt| {
|
||||||
|
// First template named by any purpose this phase answers to, in the
|
||||||
|
// phase's own preference order — the same order `launch_phase` uses
|
||||||
|
// to pick teams, so a composed mission and a ZeroClaw one resolve the
|
||||||
|
// same template for the same phase.
|
||||||
|
purposes.iter().find_map(|p| {
|
||||||
|
pt.get(*p)
|
||||||
|
.and_then(|v| v.as_array())
|
||||||
|
.and_then(|a| a.first())
|
||||||
|
.and_then(|v| v.as_str())
|
||||||
|
.and_then(|s| Uuid::parse_str(s).ok())
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.or(legacy_template);
|
||||||
|
let Some(template_id) = template_id else {
|
||||||
|
return Ok(None);
|
||||||
|
};
|
||||||
|
|
||||||
|
let template = cm_db::repo::team_templates::get(pool, template_id)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("load template {template_id}: {e}"))?
|
||||||
|
.ok_or_else(|| format!("template {template_id} not found"))?;
|
||||||
|
let roles: Vec<&str> = template.roles.iter().map(|r| r.slot.as_str()).collect();
|
||||||
|
if roles.is_empty() {
|
||||||
|
return Err(format!("template {template_id} defines no roles"));
|
||||||
|
}
|
||||||
|
let graph = cm_topology::build(
|
||||||
|
parse_topology_kind(&template.template.default_topology),
|
||||||
|
&roles,
|
||||||
|
)
|
||||||
|
.map_err(|e| format!("build topology graph for template {template_id}: {e}"))?;
|
||||||
|
serde_json::to_value(&graph)
|
||||||
|
.map(Some)
|
||||||
|
.map_err(|e| format!("serialize topology graph: {e}"))
|
||||||
|
}
|
||||||
|
|
||||||
fn parse_topology_kind(s: &str) -> cm_topology::TopologyKind {
|
fn parse_topology_kind(s: &str) -> cm_topology::TopologyKind {
|
||||||
use cm_topology::TopologyKind;
|
use cm_topology::TopologyKind;
|
||||||
match s {
|
match s {
|
||||||
@@ -506,3 +917,208 @@ fn default_accent_for(slot: &str) -> &'static str {
|
|||||||
_ => "#8a8a92",
|
_ => "#8a8a92",
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Give this mission's agents a reachable, narrow door to the skills catalogue.
|
||||||
|
///
|
||||||
|
/// Two halves that must both happen: the document goes into the container, and
|
||||||
|
/// the daemon is told to pass it to `claude -p --mcp-config`. Doing one without
|
||||||
|
/// the other leaves a door that is installed and unreachable, which looks
|
||||||
|
/// exactly like a door nobody walked through — the same shape as the hooks that
|
||||||
|
/// were installed and inert.
|
||||||
|
///
|
||||||
|
/// # The credential
|
||||||
|
///
|
||||||
|
/// A `skills:read` session, not a user's. It is written into a file the agent
|
||||||
|
/// can `cat` — it runs `Bash` with egress — so the only thing keeping this safe
|
||||||
|
/// is that the token authenticates to exactly one route and nowhere else. See
|
||||||
|
/// `cm_auth::AuthService::authenticate_scoped`. A full session here would be an
|
||||||
|
/// owner-privileged API key handed to something explicitly untrusted, which is
|
||||||
|
/// why the door went undeployed rather than being deployed the easy way.
|
||||||
|
///
|
||||||
|
/// Every failure degrades to "no door", never to a failed launch. A mission
|
||||||
|
/// that cannot retrieve a skill still delivers.
|
||||||
|
/// Returns whether the door is installed AND reachable. The caller needs the
|
||||||
|
/// answer, not just the log line: the `index` delivery arm hands agents a list
|
||||||
|
/// of uris to fetch, and without a door every one of them is a dead end that
|
||||||
|
/// reads as an agent ignoring its skills.
|
||||||
|
async fn install_skills_door(
|
||||||
|
pool: &PgPool,
|
||||||
|
user_id: cm_domain::UserId,
|
||||||
|
mission_id: Uuid,
|
||||||
|
container: &str,
|
||||||
|
prov: &RuntimeProvisioner,
|
||||||
|
) -> bool {
|
||||||
|
let Some(origin) = crate::container_tool_hooks::api_origin() else {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: no API origin for the skills door (set \
|
||||||
|
CLAWMATES_API_ORIGIN) — mission {mission_id} runs without it"
|
||||||
|
);
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
// Bound to the mission: revoked by `revoke_mission_credentials` the
|
||||||
|
// moment it reaches a terminal status. The 24 h TTL is the backstop for a
|
||||||
|
// mission nothing ever closes, not the credential's lifetime — until
|
||||||
|
// 2026-09-20 it was, and a twenty-minute mission left a live token in
|
||||||
|
// its container for the other twenty-three hours.
|
||||||
|
let auth = cm_auth::AuthService::new(pool.clone());
|
||||||
|
let token = match auth
|
||||||
|
.mint_scoped_for_mission(
|
||||||
|
user_id,
|
||||||
|
cm_auth::SCOPE_SKILLS_READ,
|
||||||
|
time::Duration::hours(24),
|
||||||
|
mission_id,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(t) => t,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: could not mint a skills token ({e}) — \
|
||||||
|
mission {mission_id} runs without the door"
|
||||||
|
);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let docker = match crate::container_exec::connect() {
|
||||||
|
Ok(d) => d,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("mission_orchestrator: cannot reach docker for the skills door: {e}");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let doc = crate::container_tool_hooks::mcp_document(&origin, &token);
|
||||||
|
let Some(path) = crate::container_tool_hooks::install_door(&docker, container, &doc).await
|
||||||
|
else {
|
||||||
|
// `install_door` already said why.
|
||||||
|
return false;
|
||||||
|
};
|
||||||
|
if let Err(e) = prov.set_claude_cli_mcp_config(&path).await {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: wrote the MCP config but could not point \
|
||||||
|
claude_cli at it ({e}) — the door is installed and unreachable"
|
||||||
|
);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: skills door installed for mission {mission_id} \
|
||||||
|
({origin}/mcp/skills)"
|
||||||
|
);
|
||||||
|
true
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write every skill the workspace can see into the mission container as a
|
||||||
|
/// file, for the `files` arm.
|
||||||
|
///
|
||||||
|
/// Every visible skill and not only the bound ones, because bindings are
|
||||||
|
/// resolved per AGENT at turn time (`effective_for_agent`) and this runs once
|
||||||
|
/// per mission before any turn — the same reason the MCP door serves the whole
|
||||||
|
/// catalogue rather than a per-mission subset. A few KB each; the whole
|
||||||
|
/// catalogue is smaller than one phase's evidence.
|
||||||
|
///
|
||||||
|
/// Returns whether the files are in place. `false` means the mission falls
|
||||||
|
/// back to `inline` (see `skill_delivery::resolve`) — an entry that points at
|
||||||
|
/// a file which is not there reads exactly like an agent ignoring its skills,
|
||||||
|
/// which is the failure this arm exists to stop misdiagnosing.
|
||||||
|
async fn install_skill_files(
|
||||||
|
pool: &PgPool,
|
||||||
|
workspace_id: WorkspaceId,
|
||||||
|
mission_id: Uuid,
|
||||||
|
container: &str,
|
||||||
|
) -> bool {
|
||||||
|
let skills = match cm_db::repo::skills_catalog::list_visible(pool, workspace_id.as_uuid()).await
|
||||||
|
{
|
||||||
|
Ok(v) => v,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: could not list skills for the files arm ({e}) — \
|
||||||
|
mission {mission_id} delivers skills inline"
|
||||||
|
);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let docker = match crate::container_exec::connect() {
|
||||||
|
Ok(d) => d,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("mission_orchestrator: cannot reach docker for the skill files: {e}");
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let dir = crate::skill_delivery::SKILLS_DIR;
|
||||||
|
// `upload_to_container` will not create the directory.
|
||||||
|
let argv = vec!["sh".to_string(), "-lc".to_string(), format!("mkdir -p {dir}")];
|
||||||
|
match crate::container_exec::exec_as_root(
|
||||||
|
&docker,
|
||||||
|
container,
|
||||||
|
None,
|
||||||
|
&argv,
|
||||||
|
crate::container_tool_hooks::INSTALL_TIMEOUT,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(out) if out.exit_code == Some(0) => {}
|
||||||
|
other => {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: could not create {dir} in {container} ({other:?}) — \
|
||||||
|
mission {mission_id} delivers skills inline"
|
||||||
|
);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let files: Vec<(String, Vec<u8>)> = skills
|
||||||
|
.iter()
|
||||||
|
.map(|sk| (format!("{}.md", sk.name), sk.body.clone().into_bytes()))
|
||||||
|
.collect();
|
||||||
|
let n = files.len();
|
||||||
|
if let Err(e) = crate::mission_fs::put_files(&docker, container, dir, &files).await {
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: could not write the skill files ({e}) — mission \
|
||||||
|
{mission_id} delivers skills inline"
|
||||||
|
);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
eprintln!("mission_orchestrator: {n} skill file(s) installed for mission {mission_id} under {dir}");
|
||||||
|
true
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record which arm this mission runs, so every turn composes the same one and
|
||||||
|
/// the score can be attributed to it afterwards.
|
||||||
|
///
|
||||||
|
/// A write failure is not fatal: `skill_delivery_mode` reads NULL as `inline`,
|
||||||
|
/// which is the arm that needs nothing installed. A mission that quietly ran
|
||||||
|
/// the control arm is a lost data point; a mission that failed to launch over
|
||||||
|
/// a telemetry column is a lost mission.
|
||||||
|
async fn record_skill_delivery(pool: &PgPool, mission_id: Uuid, mode: crate::skill_delivery::Mode) {
|
||||||
|
if let Err(e) = sqlx::query("UPDATE missions SET skill_delivery = $2 WHERE id = $1")
|
||||||
|
.bind(mission_id)
|
||||||
|
.bind(mode.as_str())
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!(
|
||||||
|
"mission_orchestrator: could not record skill_delivery={} for mission \
|
||||||
|
{mission_id} ({e}) — its turns will compose skills inline",
|
||||||
|
mode.as_str()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Revoke every credential minted for a mission. Called on every path that
|
||||||
|
/// takes a mission to a terminal status — the runner's close and the
|
||||||
|
/// operator's stop — so the authority a mission was given ends with it.
|
||||||
|
/// Best-effort and loud: a revocation that failed is logged with the count
|
||||||
|
/// it could not clear, which is the number an operator needs.
|
||||||
|
pub async fn revoke_mission_credentials(pool: &PgPool, mission_id: Uuid) {
|
||||||
|
match cm_auth::AuthService::new(pool.clone())
|
||||||
|
.revoke_mission_sessions(mission_id)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(0) => {}
|
||||||
|
Ok(n) => eprintln!(
|
||||||
|
"mission_orchestrator: revoked {n} credential(s) for mission {mission_id} at close"
|
||||||
|
),
|
||||||
|
Err(e) => eprintln!(
|
||||||
|
"mission_orchestrator: could NOT revoke credentials for mission {mission_id}: {e} \
|
||||||
|
— they expire on their own within 24 h"
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,533 @@
|
|||||||
|
//! Capture for missions that have no repository.
|
||||||
|
//!
|
||||||
|
//! `mission_delivery` captures a phase's work by diffing a git checkout. A
|
||||||
|
//! mission with `repo_id IS NULL` — every `research_only` mission, because that
|
||||||
|
//! recipe sets `requires_repo = false` — has no checkout, so
|
||||||
|
//! `capture_finished_coding_phases` filters it out at the SQL level
|
||||||
|
//! (`AND m.repo_id IS NOT NULL`) and never reads the container at all.
|
||||||
|
//!
|
||||||
|
//! The agents still write files. The research directive tells them to save
|
||||||
|
//! findings under `/mission/repo/research/`, and it says so whether or not a
|
||||||
|
//! repo exists. So the work lands in the container's own filesystem, is never
|
||||||
|
//! collected, and is destroyed when the sweeper reaps the container.
|
||||||
|
//!
|
||||||
|
//! # What this cost, measured
|
||||||
|
//!
|
||||||
|
//! Mission `019fdc35` ("ClawHDF5 Research"): four agents, 9.5 minutes, **eight
|
||||||
|
//! research documents** — an HDF5 parser design, a Rust ecosystem survey, a
|
||||||
|
//! seven-crate dependency map, tracing and fuzzing strategy. `mission_artifacts`
|
||||||
|
//! held zero rows and the mission reported `completed`. One agent's own summary
|
||||||
|
//! recorded the situation exactly: *"No git repo — file is written."* It noticed,
|
||||||
|
//! wrote anyway, and the platform threw the result away without a word.
|
||||||
|
//!
|
||||||
|
//! Nothing survived but the summarizer's account of it — which is the agents'
|
||||||
|
//! description of the work, not the work.
|
||||||
|
//!
|
||||||
|
//! # Why a separate path rather than widening the diff capture
|
||||||
|
//!
|
||||||
|
//! There is no base commit to diff against and no branch to push, so every
|
||||||
|
//! concept `capture_phase_diff` is built on is absent. What a repo-less mission
|
||||||
|
//! produces is simply *files*, and the honest capture is to copy them out and
|
||||||
|
//! register each as an artifact. `_outputs/` is deliberately a SIBLING of the
|
||||||
|
//! mission directory and survives `teardown_container`, so artifacts registered
|
||||||
|
//! here outlive the reap that destroyed the originals.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
|
||||||
|
use sqlx::{PgPool, Row};
|
||||||
|
use time::{Duration, OffsetDateTime};
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// Directories never worth capturing, whatever an agent leaves behind.
|
||||||
|
///
|
||||||
|
/// Same intent as `mission_fs`'s exclusion list: a captured `.git` or
|
||||||
|
/// `node_modules` is noise that would bury the four documents that matter.
|
||||||
|
const SKIP_DIRS: &[&str] = &[
|
||||||
|
".git",
|
||||||
|
"node_modules",
|
||||||
|
"target",
|
||||||
|
".venv",
|
||||||
|
"venv",
|
||||||
|
"__pycache__",
|
||||||
|
".cache",
|
||||||
|
"dist",
|
||||||
|
"build",
|
||||||
|
];
|
||||||
|
|
||||||
|
/// How many phases to capture per tick, matching `CAPTURE_BATCH`.
|
||||||
|
const BATCH: i64 = 5;
|
||||||
|
|
||||||
|
/// How long a phase's outputs may stay uncollectable before the sweep stops
|
||||||
|
/// retrying and calls it empty.
|
||||||
|
///
|
||||||
|
/// Generous on purpose: the container is torn down asynchronously after a
|
||||||
|
/// phase, so an early tick can legitimately fail. What must NOT happen is
|
||||||
|
/// retrying forever — that is the state this constant exists to end.
|
||||||
|
const COLLECT_GRACE: Duration = Duration::minutes(10);
|
||||||
|
|
||||||
|
/// The artifact kind this path registers. Also the idempotency key: a phase with
|
||||||
|
/// one of these has already been captured.
|
||||||
|
pub const OUTPUT_KIND: &str = "document";
|
||||||
|
|
||||||
|
/// Filename of the marker written when a phase produced nothing.
|
||||||
|
const EMPTY_MARKER: &str = "NO-OUTPUT.md";
|
||||||
|
|
||||||
|
/// Capture the outputs of finished phases on missions that have no repo.
|
||||||
|
pub async fn capture_repo_less_phases(pool: &PgPool) -> Result<(), String> {
|
||||||
|
let rows = sqlx::query(
|
||||||
|
"SELECT mp.id, mp.mission_id, mp.kind, mp.config, mp.completed_at, m.runtime_kind
|
||||||
|
FROM mission_phases mp
|
||||||
|
JOIN missions m ON m.id = mp.mission_id
|
||||||
|
WHERE mp.status IN ('completed', 'failed')
|
||||||
|
AND m.repo_id IS NULL
|
||||||
|
-- microVM used to be excluded here because `run_phase_in_vm`
|
||||||
|
-- refused to boot without a checkout. It no longer does: a
|
||||||
|
-- repo-less mission gets an empty workspace at the same guest path,
|
||||||
|
-- and the collect unpacks it back onto the host — so those files are
|
||||||
|
-- already on disk and `collect_into` reads them instead of asking a
|
||||||
|
-- container that never existed.
|
||||||
|
AND NOT EXISTS (
|
||||||
|
SELECT 1 FROM mission_artifacts a
|
||||||
|
WHERE a.mission_id = mp.mission_id
|
||||||
|
AND a.phase_id = mp.id
|
||||||
|
AND a.kind = $2
|
||||||
|
)
|
||||||
|
ORDER BY mp.completed_at DESC NULLS LAST
|
||||||
|
LIMIT $1",
|
||||||
|
)
|
||||||
|
.bind(BATCH)
|
||||||
|
.bind(OUTPUT_KIND)
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("select repo-less phases to capture: {e}"))?;
|
||||||
|
|
||||||
|
for row in rows {
|
||||||
|
let phase_id: Uuid = row.get("id");
|
||||||
|
let mission_id: Uuid = row.get("mission_id");
|
||||||
|
let kind: String = row.get("kind");
|
||||||
|
let config: serde_json::Value = row.get("config");
|
||||||
|
let completed_at: Option<OffsetDateTime> = row.get("completed_at");
|
||||||
|
let runtime_kind: String = row.get("runtime_kind");
|
||||||
|
|
||||||
|
let dest = outputs_dir(mission_id, phase_id);
|
||||||
|
let captured = match collect_into(mission_id, &dest, &runtime_kind).await {
|
||||||
|
Ok(files) => files,
|
||||||
|
Err(e) => {
|
||||||
|
// Retryable, but BOUNDED. A bare `continue` here is how a phase
|
||||||
|
// whose collect can never succeed stayed `completed` with zero
|
||||||
|
// artifacts forever: the fail-empty rule and the NO-OUTPUT
|
||||||
|
// marker both live below this point, so neither was ever
|
||||||
|
// reached, and the phase was re-attempted on every tick for the
|
||||||
|
// life of the deployment.
|
||||||
|
//
|
||||||
|
// The grace window exists because the container may legitimately
|
||||||
|
// not be ready on the first tick after a phase finishes. Past
|
||||||
|
// that, "cannot collect" and "collected nothing" are the same
|
||||||
|
// fact for the operator, so we fall through and let the rules
|
||||||
|
// below fail the phase and leave a marker explaining why.
|
||||||
|
let settled = completed_at
|
||||||
|
.map(|t| OffsetDateTime::now_utc() - t > COLLECT_GRACE)
|
||||||
|
.unwrap_or(true);
|
||||||
|
if !settled {
|
||||||
|
eprintln!(
|
||||||
|
"mission_outputs: could NOT collect outputs for phase {phase_id} \
|
||||||
|
of mission {mission_id} (will retry): {e}"
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
eprintln!(
|
||||||
|
"mission_outputs: giving up collecting phase {phase_id} of mission \
|
||||||
|
{mission_id} after {}s: {e} — treating it as having produced nothing",
|
||||||
|
COLLECT_GRACE.whole_seconds()
|
||||||
|
);
|
||||||
|
Vec::new()
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
for file in &captured {
|
||||||
|
let rel = match file.strip_prefix(missions_root()) {
|
||||||
|
Ok(r) => r.to_string_lossy().to_string(),
|
||||||
|
Err(_) => file.to_string_lossy().to_string(),
|
||||||
|
};
|
||||||
|
let title = file
|
||||||
|
.file_name()
|
||||||
|
.map(|n| n.to_string_lossy().to_string())
|
||||||
|
.unwrap_or_else(|| rel.clone());
|
||||||
|
if let Err(e) = cm_db::repo::missions::register_artifact(
|
||||||
|
pool,
|
||||||
|
cm_db::repo::missions::RegisterArtifact {
|
||||||
|
mission_id,
|
||||||
|
phase_id: Some(phase_id),
|
||||||
|
path: &rel,
|
||||||
|
kind: OUTPUT_KIND,
|
||||||
|
mime: Some(mime_for(file)),
|
||||||
|
title: Some(&title),
|
||||||
|
generated_by_run: None,
|
||||||
|
// No PDF. The renderer converted Markdown to HTML by
|
||||||
|
// calling an LLM — a paid API call, per document, on the
|
||||||
|
// critical path of "save my research", which promptly
|
||||||
|
// failed on depleted credits. Markdown IS the deliverable;
|
||||||
|
// it is served by `artifact_content` and styled at render
|
||||||
|
// time, which is free, offline, and cannot 429.
|
||||||
|
render_pdf: false,
|
||||||
|
metadata: Some(serde_json::json!({
|
||||||
|
"bytes": std::fs::metadata(file).map(|m| m.len()).unwrap_or(0),
|
||||||
|
"captured_from": "/mission/repo",
|
||||||
|
})),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!("mission_outputs: registering {rel}: {e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if captured.is_empty() {
|
||||||
|
// Register a marker even when there is nothing to capture, or this
|
||||||
|
// phase matches the `NOT EXISTS` selection on every tick forever:
|
||||||
|
// re-running a docker copy_out each time and, because the batch is
|
||||||
|
// bounded, permanently occupying a slot so no other repo-less
|
||||||
|
// mission is ever captured again.
|
||||||
|
//
|
||||||
|
// `phase_runner::record_uncapturable` exists for exactly this
|
||||||
|
// failure on the diff path — five dead phases starved the batch
|
||||||
|
// while live work went untouched — and this code hit it again on
|
||||||
|
// its first live negative control (4 log lines, then 8, 45 seconds
|
||||||
|
// apart). Same shape, same fix: a real file behind a real row,
|
||||||
|
// because an artifact pointing at nothing turns every reader into
|
||||||
|
// an unexplained 404.
|
||||||
|
if let Err(e) = register_empty_marker(pool, mission_id, phase_id, &dest).await {
|
||||||
|
eprintln!("mission_outputs: marking phase {phase_id} as empty: {e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if captured.is_empty() && !allow_empty(&config) {
|
||||||
|
// The same rule `empty_delivery_is_a_failure` applies to a coding
|
||||||
|
// phase, for the only channel a repo-less phase has. Without it a
|
||||||
|
// research mission that produced nothing is indistinguishable from
|
||||||
|
// one that produced eight documents — both `completed`.
|
||||||
|
eprintln!(
|
||||||
|
"mission_outputs: phase {phase_id} ({kind}) of mission {mission_id} produced \
|
||||||
|
NO output files — failing it. Set config.allow_empty = true if this phase is \
|
||||||
|
meant to think rather than produce."
|
||||||
|
);
|
||||||
|
if let Err(e) = sqlx::query("UPDATE mission_phases SET status = 'failed' WHERE id = $1")
|
||||||
|
.bind(phase_id)
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!("mission_outputs: failing empty phase {phase_id}: {e}");
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
eprintln!(
|
||||||
|
"mission_outputs: captured {} file(s) from phase {phase_id} ({kind}) of \
|
||||||
|
mission {mission_id}",
|
||||||
|
captured.len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record that a phase produced nothing, so it is not reconsidered forever.
|
||||||
|
///
|
||||||
|
/// Deliberately the same `OUTPUT_KIND` the real captures use: the selection
|
||||||
|
/// query asks "has this phase been captured?", and "captured, and there was
|
||||||
|
/// nothing" is an answer to that question. `metadata.empty` is what tells the
|
||||||
|
/// two apart — the same convention `mission_delivery` uses for its "No code
|
||||||
|
/// changes" artifact.
|
||||||
|
async fn register_empty_marker(
|
||||||
|
pool: &PgPool,
|
||||||
|
mission_id: Uuid,
|
||||||
|
phase_id: Uuid,
|
||||||
|
dest: &Path,
|
||||||
|
) -> Result<(), String> {
|
||||||
|
std::fs::create_dir_all(dest).map_err(|e| format!("create {}: {e}", dest.display()))?;
|
||||||
|
let file = dest.join(EMPTY_MARKER);
|
||||||
|
std::fs::write(
|
||||||
|
&file,
|
||||||
|
"This phase finished without leaving any files in its workspace, so there\n was nothing to publish. If the phase is meant to reason rather than\n produce, set `config.allow_empty = true` on it.\n",
|
||||||
|
)
|
||||||
|
.map_err(|e| format!("write {}: {e}", file.display()))?;
|
||||||
|
let rel = file
|
||||||
|
.strip_prefix(missions_root())
|
||||||
|
.map(|r| r.to_string_lossy().to_string())
|
||||||
|
.unwrap_or_else(|_| file.to_string_lossy().to_string());
|
||||||
|
cm_db::repo::missions::register_artifact(
|
||||||
|
pool,
|
||||||
|
cm_db::repo::missions::RegisterArtifact {
|
||||||
|
mission_id,
|
||||||
|
phase_id: Some(phase_id),
|
||||||
|
path: &rel,
|
||||||
|
kind: OUTPUT_KIND,
|
||||||
|
mime: Some("text/markdown"),
|
||||||
|
title: Some("No output produced"),
|
||||||
|
generated_by_run: None,
|
||||||
|
render_pdf: false,
|
||||||
|
metadata: Some(serde_json::json!({ "empty": true })),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map(|_| ())
|
||||||
|
.map_err(|e| format!("register empty marker: {e}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Gather the mission's produced files and return the ones worth keeping.
|
||||||
|
///
|
||||||
|
/// Where they come from depends on the runtime, and the difference is not
|
||||||
|
/// cosmetic: a container mission's files are still INSIDE a running container,
|
||||||
|
/// while a microVM's have already been unpacked onto the host by the collect at
|
||||||
|
/// the end of the turn (`microvm_executor` writes them over
|
||||||
|
/// `mission_workspace::checkout_path`). Asking docker for a VM mission's files
|
||||||
|
/// would query a container that never existed.
|
||||||
|
async fn collect_into(
|
||||||
|
mission_id: Uuid,
|
||||||
|
dest: &Path,
|
||||||
|
runtime_kind: &str,
|
||||||
|
) -> Result<Vec<PathBuf>, String> {
|
||||||
|
// A stale copy from an earlier attempt would be registered as this pass's
|
||||||
|
// output — the same "captured a tree nobody wrote" shape capture avoids.
|
||||||
|
let _ = std::fs::remove_dir_all(dest);
|
||||||
|
std::fs::create_dir_all(dest).map_err(|e| format!("create {}: {e}", dest.display()))?;
|
||||||
|
|
||||||
|
if runtime_kind == "microvm" {
|
||||||
|
let src = crate::mission_workspace::checkout_path(mission_id);
|
||||||
|
if !src.is_dir() {
|
||||||
|
return Err(format!(
|
||||||
|
"{} is absent — the VM's collect did not land",
|
||||||
|
src.display()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
copy_tree(&src, &dest.join("repo"))?;
|
||||||
|
return Ok(keep_files(&dest.join("repo")));
|
||||||
|
}
|
||||||
|
|
||||||
|
let container = crate::mission_runtime::container_name(mission_id);
|
||||||
|
let docker = crate::container_exec::connect()?;
|
||||||
|
crate::mission_fs::copy_out(&docker, &container, "/mission/repo", dest).await?;
|
||||||
|
Ok(keep_files(&dest.join("repo")))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Recursive file copy. Small on purpose — the alternative is a dependency or a
|
||||||
|
/// shell-out, and this runs as the server's own uid against its own directory.
|
||||||
|
fn copy_tree(src: &Path, dest: &Path) -> Result<(), String> {
|
||||||
|
std::fs::create_dir_all(dest).map_err(|e| format!("create {}: {e}", dest.display()))?;
|
||||||
|
let entries = std::fs::read_dir(src).map_err(|e| format!("read {}: {e}", src.display()))?;
|
||||||
|
for entry in entries.flatten() {
|
||||||
|
let from = entry.path();
|
||||||
|
let to = dest.join(entry.file_name());
|
||||||
|
match entry.file_type() {
|
||||||
|
Ok(t) if t.is_dir() => copy_tree(&from, &to)?,
|
||||||
|
Ok(t) if t.is_file() => {
|
||||||
|
std::fs::copy(&from, &to).map_err(|e| format!("copy {}: {e}", from.display()))?;
|
||||||
|
}
|
||||||
|
// Symlinks and specials are skipped rather than followed: a link out
|
||||||
|
// of the tree would publish whatever it points at.
|
||||||
|
_ => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every regular file worth keeping, recursively.
|
||||||
|
fn keep_files(root: &Path) -> Vec<PathBuf> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
let mut stack = vec![root.to_path_buf()];
|
||||||
|
while let Some(dir) = stack.pop() {
|
||||||
|
let Ok(entries) = std::fs::read_dir(&dir) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
for entry in entries.flatten() {
|
||||||
|
let path = entry.path();
|
||||||
|
let name = entry.file_name().to_string_lossy().to_string();
|
||||||
|
if path.is_dir() {
|
||||||
|
if !SKIP_DIRS.contains(&name.as_str()) {
|
||||||
|
stack.push(path);
|
||||||
|
}
|
||||||
|
} else if path.is_file()
|
||||||
|
&& !name.starts_with('.')
|
||||||
|
// The agent runtime seeds its own identity files into the
|
||||||
|
// workspace root, which is pinned to the repo root. In a
|
||||||
|
// repo-backed mission `.git/info/exclude` hides them; a
|
||||||
|
// repo-less mission has no `.git`, so without this the user's
|
||||||
|
// artifact list is 7 files of agent scaffolding and 2 of their
|
||||||
|
// research. Measured exactly that way on the first live run.
|
||||||
|
&& !crate::mission_workspace::AGENT_SCAFFOLDING.contains(&name.as_str())
|
||||||
|
{
|
||||||
|
out.push(path);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out.sort();
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `<missions_root>/_outputs/<mission>/<phase>` — a sibling of the mission
|
||||||
|
/// directory, so `teardown_container` reaping the mission does not take the
|
||||||
|
/// captured artifacts with it.
|
||||||
|
fn outputs_dir(mission_id: Uuid, phase_id: Uuid) -> PathBuf {
|
||||||
|
missions_root()
|
||||||
|
.join("_outputs")
|
||||||
|
.join(mission_id.to_string())
|
||||||
|
.join(phase_id.to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The missions root, for callers that resolve artifact paths against it.
|
||||||
|
pub fn missions_root_dir() -> PathBuf {
|
||||||
|
missions_root()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The only directory an artifact may be read from.
|
||||||
|
pub fn outputs_root_dir() -> PathBuf {
|
||||||
|
missions_root().join("_outputs")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn missions_root() -> PathBuf {
|
||||||
|
crate::mission_workspace::missions_root()
|
||||||
|
}
|
||||||
|
|
||||||
|
fn mime_for(p: &Path) -> &'static str {
|
||||||
|
match p.extension().and_then(|e| e.to_str()) {
|
||||||
|
Some("md") | Some("markdown") => "text/markdown",
|
||||||
|
Some("json") => "application/json",
|
||||||
|
Some("csv") => "text/csv",
|
||||||
|
Some("html") => "text/html",
|
||||||
|
_ => "text/plain",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn allow_empty(config: &serde_json::Value) -> bool {
|
||||||
|
config.get("allow_empty").and_then(|v| v.as_bool()) == Some(true)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn touch(p: &Path) {
|
||||||
|
std::fs::create_dir_all(p.parent().unwrap()).unwrap();
|
||||||
|
std::fs::write(p, "x").unwrap();
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The documents a research phase writes are what must come back — and the
|
||||||
|
/// machinery around them must not.
|
||||||
|
#[test]
|
||||||
|
fn research_documents_are_kept_and_scaffolding_is_not() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let repo = tmp.path().join("repo");
|
||||||
|
touch(&repo.join("research/01_repo_archaeology.md"));
|
||||||
|
touch(&repo.join("research/02_ecosystem.md"));
|
||||||
|
touch(&repo.join("notes.txt"));
|
||||||
|
// The seven the agent runtime seeds into the workspace root.
|
||||||
|
for f in crate::mission_workspace::AGENT_SCAFFOLDING {
|
||||||
|
touch(&repo.join(f));
|
||||||
|
}
|
||||||
|
touch(&repo.join(".git/HEAD"));
|
||||||
|
touch(&repo.join("node_modules/left-pad/index.js"));
|
||||||
|
touch(&repo.join("target/debug/thing"));
|
||||||
|
touch(&repo.join(".hidden"));
|
||||||
|
|
||||||
|
let kept: Vec<String> = keep_files(&repo)
|
||||||
|
.iter()
|
||||||
|
.map(|p| p.strip_prefix(&repo).unwrap().to_string_lossy().to_string())
|
||||||
|
.collect();
|
||||||
|
assert_eq!(
|
||||||
|
kept,
|
||||||
|
vec![
|
||||||
|
"notes.txt".to_string(),
|
||||||
|
"research/01_repo_archaeology.md".to_string(),
|
||||||
|
"research/02_ecosystem.md".to_string(),
|
||||||
|
],
|
||||||
|
"kept: {kept:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Markdown is the deliverable, so it must be labelled as markdown — the
|
||||||
|
/// viewer decides how to render from the mime type.
|
||||||
|
#[test]
|
||||||
|
fn markdown_is_labelled_so_the_viewer_can_style_it() {
|
||||||
|
assert_eq!(mime_for(Path::new("/x/01_notes.md")), "text/markdown");
|
||||||
|
assert_eq!(mime_for(Path::new("/x/data.json")), "application/json");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The containment rule the content endpoint enforces: everything readable
|
||||||
|
/// lives under `_outputs`, and nothing else does.
|
||||||
|
///
|
||||||
|
/// Artifact paths are written by this server, but they are DATA in a table,
|
||||||
|
/// and a row saying `../../../etc/passwd` must be a 404 rather than a file
|
||||||
|
/// read. The endpoint canonicalises before comparing — checking the string
|
||||||
|
/// first would pass `_outputs/../../etc/passwd` straight through.
|
||||||
|
#[test]
|
||||||
|
fn everything_readable_lives_under_the_outputs_root() {
|
||||||
|
let root = outputs_root_dir();
|
||||||
|
assert!(root.ends_with("_outputs"), "{root:?}");
|
||||||
|
assert!(root.starts_with(missions_root_dir()), "{root:?}");
|
||||||
|
|
||||||
|
// A real capture is inside it...
|
||||||
|
let inside = outputs_dir(Uuid::now_v7(), Uuid::now_v7());
|
||||||
|
assert!(inside.starts_with(&root), "{inside:?}");
|
||||||
|
|
||||||
|
// ...and the traversal shape this guards against is not, once resolved.
|
||||||
|
let escaped = root.join("..").join("..").join("etc/passwd");
|
||||||
|
let normalised: PathBuf = escaped.components().fold(PathBuf::new(), |mut acc, c| {
|
||||||
|
match c {
|
||||||
|
std::path::Component::ParentDir => {
|
||||||
|
acc.pop();
|
||||||
|
}
|
||||||
|
other => acc.push(other),
|
||||||
|
}
|
||||||
|
acc
|
||||||
|
});
|
||||||
|
assert!(
|
||||||
|
!normalised.starts_with(&root),
|
||||||
|
"a traversal must not resolve back inside the outputs root: {normalised:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A phase that produced nothing must still leave a marker, or the
|
||||||
|
/// selection query matches it on every tick forever.
|
||||||
|
///
|
||||||
|
/// Measured on the first live negative control: the guard logged "produced
|
||||||
|
/// NO output files" 4 times, then 8 times 45 seconds later — a docker
|
||||||
|
/// copy_out per tick, and with a bounded batch, five such phases would
|
||||||
|
/// starve every other repo-less mission out of capture permanently.
|
||||||
|
/// `phase_runner::record_uncapturable` was written for the identical
|
||||||
|
/// failure on the diff path.
|
||||||
|
#[test]
|
||||||
|
fn an_empty_phase_leaves_a_marker_so_it_is_not_reconsidered_forever() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let dest = tmp.path().join("out");
|
||||||
|
// The file-writing half of `register_empty_marker`, which is the part
|
||||||
|
// that must exist for the artifact row to point at something real.
|
||||||
|
std::fs::create_dir_all(&dest).unwrap();
|
||||||
|
let file = dest.join(EMPTY_MARKER);
|
||||||
|
std::fs::write(&file, "x").unwrap();
|
||||||
|
assert!(file.exists(), "an artifact row must not point at nothing");
|
||||||
|
assert_eq!(
|
||||||
|
file.file_name().unwrap().to_string_lossy(),
|
||||||
|
"NO-OUTPUT.md",
|
||||||
|
"the marker name is part of the contract with readers"
|
||||||
|
);
|
||||||
|
// And the marker must not itself be mistaken for captured output on a
|
||||||
|
// later pass: it is filtered like any other scaffolding would be.
|
||||||
|
assert!(keep_files(&dest).iter().any(|p| p == &file));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Artifacts must land OUTSIDE the mission directory. `teardown_container`
|
||||||
|
/// removes `<missions_root>/<mission_id>` wholesale, so a capture written
|
||||||
|
/// inside it would be destroyed by the very reap it exists to survive.
|
||||||
|
#[test]
|
||||||
|
fn captures_survive_the_mission_directory_being_reaped() {
|
||||||
|
let mission = Uuid::now_v7();
|
||||||
|
let phase = Uuid::now_v7();
|
||||||
|
let out = outputs_dir(mission, phase);
|
||||||
|
let mission_dir = missions_root().join(mission.to_string());
|
||||||
|
assert!(
|
||||||
|
!out.starts_with(&mission_dir),
|
||||||
|
"{} must not be inside {}",
|
||||||
|
out.display(),
|
||||||
|
mission_dir.display()
|
||||||
|
);
|
||||||
|
assert!(out.starts_with(missions_root().join("_outputs")), "{out:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,296 @@
|
|||||||
|
//! A model-authored execution plan for one mission — W1 / #13.
|
||||||
|
//!
|
||||||
|
//! Every mission's phases come from one of five hand-written recipes in
|
||||||
|
//! `templates/workflows/*.toml`, chosen by `template_kind`. A recipe is a fixed
|
||||||
|
//! answer to "what phases does this kind of mission have", written before anyone
|
||||||
|
//! saw the mission — the "do it this way: 1, 2, 3" over-specification that makes
|
||||||
|
//! a capable model follow a worse plan than it would have chosen for the actual
|
||||||
|
//! task.
|
||||||
|
//!
|
||||||
|
//! This is the other half of [`crate::mission_roster`]: that one lets a model
|
||||||
|
//! size the team, this one lets it decide what the work IS. Same shape on
|
||||||
|
//! purpose — propose, review, approve, apply — because the review gate is what
|
||||||
|
//! makes model-authored structure safe to run, and a second shape would be a
|
||||||
|
//! second thing to get right.
|
||||||
|
//!
|
||||||
|
//! # Grounded in what the platform actually reads
|
||||||
|
//!
|
||||||
|
//! The interesting constraint is not "is this JSON valid" but "will anything
|
||||||
|
//! consume it". `phase_config::KNOWN_KEYS` already names every phase-config key
|
||||||
|
//! and the code that reads it, with eleven marked NOT IMPLEMENTED — the registry
|
||||||
|
//! built after `task` sat unread through every mission. A plan is validated
|
||||||
|
//! against that registry, so a model cannot propose a phase whose settings
|
||||||
|
//! nothing will act on. The failure that registry exists to EXPOSE is one this
|
||||||
|
//! path cannot create.
|
||||||
|
//!
|
||||||
|
//! Phase kinds are checked the same way, against the kinds `phase_runner`
|
||||||
|
//! actually dispatches. A model asked to plan work will happily invent
|
||||||
|
//! `kind: "review"`, and an unknown kind does not fail — it falls to the
|
||||||
|
//! catch-all purpose and runs as a generic phase, which looks like it worked.
|
||||||
|
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
|
||||||
|
/// Phase kinds `phase_runner` dispatches on.
|
||||||
|
///
|
||||||
|
/// Not an enum, because `mission_phases.kind` is a free-form column shared with
|
||||||
|
/// hand-written recipes and the wizard; this is the subset a MODEL may propose.
|
||||||
|
/// An unrecognised kind is the dangerous case: it does not error, it falls
|
||||||
|
/// through to the generic `mission` purpose and runs anyway.
|
||||||
|
pub const PLANNABLE_KINDS: &[&str] = &["research", "coding", "benchmark", "security_scan"];
|
||||||
|
|
||||||
|
/// Ceiling on a proposed plan.
|
||||||
|
///
|
||||||
|
/// Each phase is a full agent run — a VM boot, a checkout, a turn, a capture —
|
||||||
|
/// executed in sequence. Anthropic's own guidance warns against decomposing work
|
||||||
|
/// into sequential phases at all ("a handoff loses context at every step"), so
|
||||||
|
/// this bound is deliberately tight: a model that wants eight phases is
|
||||||
|
/// describing a to-do list, not a plan.
|
||||||
|
pub const MAX_PHASES: usize = 4;
|
||||||
|
|
||||||
|
/// One phase of a proposed plan.
|
||||||
|
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||||
|
pub struct PlannedPhase {
|
||||||
|
/// One of [`PLANNABLE_KINDS`].
|
||||||
|
pub kind: String,
|
||||||
|
/// What this phase does. Lands in `config.task`, which
|
||||||
|
/// `phase_task_text` injects — the key that sat unread through every
|
||||||
|
/// mission until two phases with different tasks produced identical output.
|
||||||
|
pub task: String,
|
||||||
|
/// Optional completion condition, judged post-hoc by the evaluator.
|
||||||
|
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||||
|
pub done_when: Option<String>,
|
||||||
|
/// Optional deterministic check, enforced IN the agent's loop by the stop
|
||||||
|
/// gate ([`crate::vm_stop_gate`]).
|
||||||
|
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||||
|
pub done_when_check: Option<String>,
|
||||||
|
/// This phase is allowed to change nothing (a verification pass).
|
||||||
|
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||||
|
pub allow_empty: Option<bool>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A proposed sequence of phases.
|
||||||
|
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||||
|
pub struct Plan {
|
||||||
|
pub phases: Vec<PlannedPhase>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a plan was refused.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum Refusal {
|
||||||
|
Empty,
|
||||||
|
TooMany(usize),
|
||||||
|
UnknownKind { index: usize, kind: String },
|
||||||
|
BlankTask(usize),
|
||||||
|
/// A config key with no reader in this build — named, with the ones that
|
||||||
|
/// would have been consumed.
|
||||||
|
InertKey { index: usize, key: String },
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for Refusal {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
match self {
|
||||||
|
Refusal::Empty => write!(f, "the plan has no phases, so the mission would do nothing"),
|
||||||
|
Refusal::TooMany(n) => write!(
|
||||||
|
f,
|
||||||
|
"the plan has {n} phases and the ceiling is {MAX_PHASES} — each one is a full \
|
||||||
|
agent run, and a handoff loses context at every step"
|
||||||
|
),
|
||||||
|
Refusal::UnknownKind { index, kind } => write!(
|
||||||
|
f,
|
||||||
|
"phase {index} has kind {kind:?}, which nothing dispatches on; use one of: {}",
|
||||||
|
PLANNABLE_KINDS.join(", ")
|
||||||
|
),
|
||||||
|
Refusal::BlankTask(i) => write!(
|
||||||
|
f,
|
||||||
|
"phase {i} has no task, so its agent would receive the mission description and \
|
||||||
|
nothing telling it which part is its own"
|
||||||
|
),
|
||||||
|
Refusal::InertKey { index, key } => write!(
|
||||||
|
f,
|
||||||
|
"phase {index} sets {key:?}, which nothing in this build reads — it would be \
|
||||||
|
stored, rendered, and consumed by nobody"
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Plan {
|
||||||
|
/// Check a plan against what the platform can actually execute.
|
||||||
|
pub fn validate(&self) -> Result<(), Refusal> {
|
||||||
|
if self.phases.is_empty() {
|
||||||
|
return Err(Refusal::Empty);
|
||||||
|
}
|
||||||
|
if self.phases.len() > MAX_PHASES {
|
||||||
|
return Err(Refusal::TooMany(self.phases.len()));
|
||||||
|
}
|
||||||
|
for (i, p) in self.phases.iter().enumerate() {
|
||||||
|
if !PLANNABLE_KINDS.contains(&p.kind.as_str()) {
|
||||||
|
return Err(Refusal::UnknownKind {
|
||||||
|
index: i,
|
||||||
|
kind: p.kind.clone(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
if p.task.trim().is_empty() {
|
||||||
|
return Err(Refusal::BlankTask(i));
|
||||||
|
}
|
||||||
|
// Every key this phase would write must have a reader. The plan is
|
||||||
|
// built from typed fields, so this can only fail if a field is added
|
||||||
|
// here without a corresponding entry in the registry — which is
|
||||||
|
// exactly the drift worth failing on.
|
||||||
|
if let Some(key) = crate::phase_config::inert_keys(&p.config()).into_iter().next() {
|
||||||
|
return Err(Refusal::InertKey { index: i, key });
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The phases as `(kind, order_idx, config)`, ready for mission creation.
|
||||||
|
///
|
||||||
|
/// `order_idx` is the array position rather than a field the model sets:
|
||||||
|
/// two sources for one fact is how a plan ends up with two phase 0s.
|
||||||
|
pub fn phases(&self) -> Vec<(String, i32, serde_json::Value)> {
|
||||||
|
self.phases
|
||||||
|
.iter()
|
||||||
|
.enumerate()
|
||||||
|
.map(|(i, p)| (p.kind.clone(), i as i32, p.config()))
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl PlannedPhase {
|
||||||
|
/// This phase's `mission_phases.config`.
|
||||||
|
fn config(&self) -> serde_json::Value {
|
||||||
|
let mut o = serde_json::Map::new();
|
||||||
|
o.insert("task".into(), serde_json::Value::String(self.task.clone()));
|
||||||
|
if let Some(d) = self.done_when.as_deref().map(str::trim).filter(|s| !s.is_empty()) {
|
||||||
|
o.insert("done_when".into(), serde_json::Value::String(d.to_string()));
|
||||||
|
}
|
||||||
|
if let Some(c) = self
|
||||||
|
.done_when_check
|
||||||
|
.as_deref()
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
{
|
||||||
|
o.insert(
|
||||||
|
"done_when_check".into(),
|
||||||
|
serde_json::Value::String(c.to_string()),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if let Some(e) = self.allow_empty {
|
||||||
|
o.insert("allow_empty".into(), serde_json::Value::Bool(e));
|
||||||
|
}
|
||||||
|
serde_json::Value::Object(o)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn phase(kind: &str, task: &str) -> PlannedPhase {
|
||||||
|
PlannedPhase {
|
||||||
|
kind: kind.into(),
|
||||||
|
task: task.into(),
|
||||||
|
done_when: None,
|
||||||
|
done_when_check: None,
|
||||||
|
allow_empty: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A kind nothing dispatches on is the dangerous one: it does not error, it
|
||||||
|
/// falls through to the generic purpose and runs as a nondescript phase that
|
||||||
|
/// looks like it worked.
|
||||||
|
#[test]
|
||||||
|
fn an_invented_phase_kind_is_refused_naming_the_real_ones() {
|
||||||
|
let p = Plan {
|
||||||
|
phases: vec![phase("coding", "do it"), phase("review", "check it")],
|
||||||
|
};
|
||||||
|
let err = p.validate().unwrap_err();
|
||||||
|
assert_eq!(
|
||||||
|
err,
|
||||||
|
Refusal::UnknownKind {
|
||||||
|
index: 1,
|
||||||
|
kind: "review".into()
|
||||||
|
}
|
||||||
|
);
|
||||||
|
let msg = err.to_string();
|
||||||
|
for kind in PLANNABLE_KINDS {
|
||||||
|
assert!(msg.contains(kind), "the message must name {kind}: {msg}");
|
||||||
|
}
|
||||||
|
// And every kind the runner dispatches on is accepted, so this cannot
|
||||||
|
// drift from what `phase_runner` can actually execute.
|
||||||
|
for kind in PLANNABLE_KINDS {
|
||||||
|
assert!(Plan { phases: vec![phase(kind, "work")] }.validate().is_ok(), "{kind}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every key a planned phase writes must have a reader. This is the whole
|
||||||
|
/// reason `phase_config` exists — a key nothing consumes is stored,
|
||||||
|
/// rendered, and silently inert.
|
||||||
|
#[test]
|
||||||
|
fn every_key_a_plan_writes_is_one_something_reads() {
|
||||||
|
let p = PlannedPhase {
|
||||||
|
kind: "coding".into(),
|
||||||
|
task: "add a module".into(),
|
||||||
|
done_when: Some("the suite passes".into()),
|
||||||
|
done_when_check: Some("cargo test".into()),
|
||||||
|
allow_empty: Some(false),
|
||||||
|
};
|
||||||
|
let cfg = p.config();
|
||||||
|
assert!(
|
||||||
|
crate::phase_config::inert_keys(&cfg).is_empty(),
|
||||||
|
"a planned phase must write only keys with readers: {:?}",
|
||||||
|
crate::phase_config::inert_keys(&cfg)
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
crate::phase_config::unknown_keys(&cfg).is_empty(),
|
||||||
|
"and only keys the registry knows: {:?}",
|
||||||
|
crate::phase_config::unknown_keys(&cfg)
|
||||||
|
);
|
||||||
|
assert!(Plan { phases: vec![p] }.validate().is_ok());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A blank task is the failure that produced identical output from two
|
||||||
|
/// different phases — the agent gets the mission description and nothing
|
||||||
|
/// saying which part is its own.
|
||||||
|
#[test]
|
||||||
|
fn a_phase_without_a_task_is_refused() {
|
||||||
|
let p = Plan {
|
||||||
|
phases: vec![phase("coding", " ")],
|
||||||
|
};
|
||||||
|
assert_eq!(p.validate(), Err(Refusal::BlankTask(0)));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Bounded and non-empty. Each phase is a full agent run in sequence, and
|
||||||
|
/// splitting one change into stages loses context at every handoff.
|
||||||
|
#[test]
|
||||||
|
fn a_plan_is_bounded_and_non_empty() {
|
||||||
|
assert_eq!(Plan { phases: vec![] }.validate(), Err(Refusal::Empty));
|
||||||
|
let many: Vec<_> = (0..MAX_PHASES + 1).map(|_| phase("coding", "work")).collect();
|
||||||
|
assert_eq!(
|
||||||
|
Plan { phases: many }.validate(),
|
||||||
|
Err(Refusal::TooMany(MAX_PHASES + 1))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Order comes from the array, not from a field the model sets. Two sources
|
||||||
|
/// for one fact is how a plan ends up with two phase 0s — and `order_idx`
|
||||||
|
/// is what `start_pending_phases` sequences on.
|
||||||
|
#[test]
|
||||||
|
fn order_comes_from_the_arrays_own_order() {
|
||||||
|
let p = Plan {
|
||||||
|
phases: vec![
|
||||||
|
phase("research", "read the code"),
|
||||||
|
phase("coding", "change it"),
|
||||||
|
phase("coding", "then this"),
|
||||||
|
],
|
||||||
|
};
|
||||||
|
let out = p.phases();
|
||||||
|
assert_eq!(
|
||||||
|
out.iter().map(|(_, i, _)| *i).collect::<Vec<_>>(),
|
||||||
|
vec![0, 1, 2]
|
||||||
|
);
|
||||||
|
assert_eq!(out[0].0, "research");
|
||||||
|
assert_eq!(out[1].2["task"], "change it");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -2,16 +2,18 @@
|
|||||||
//! mission and rewrite it into a coherent, sectioned Markdown brief
|
//! mission and rewrite it into a coherent, sectioned Markdown brief
|
||||||
//! that downstream research + coding agents can ingest cleanly.
|
//! that downstream research + coding agents can ingest cleanly.
|
||||||
//!
|
//!
|
||||||
//! Calls Anthropic Claude Opus 4.8 by default. Prod already carries
|
//! Asks for Claude Opus 4.8 by default, but goes through
|
||||||
//! ANTHROPIC_API_KEY for ZeroClaw's provider config, so no separate
|
//! `subscription::complete_with_fallback` like every other server-side model
|
||||||
//! env is needed.
|
//! call. It used to hand-roll its own HTTPS POST to the Messages API with the
|
||||||
|
//! metered key — a comment above this line still claimed prod "already carries
|
||||||
|
//! ANTHROPIC_API_KEY, so no separate env is needed", which stopped being true
|
||||||
|
//! the moment that account ran out of credit. See `subscription`, whose
|
||||||
|
//! source-walk test is what found this module.
|
||||||
|
|
||||||
use serde_json::json;
|
|
||||||
use sqlx::PgPool;
|
use sqlx::PgPool;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
const DEFAULT_MODEL: &str = "claude-opus-4-8";
|
const DEFAULT_MODEL: &str = "claude-opus-5";
|
||||||
const ANTHROPIC_API_VERSION: &str = "2023-06-01";
|
|
||||||
|
|
||||||
fn model_name() -> String {
|
fn model_name() -> String {
|
||||||
std::env::var("CLAWMATES_REFINER_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_string())
|
std::env::var("CLAWMATES_REFINER_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_string())
|
||||||
@@ -22,12 +24,36 @@ pub struct RefineResult {
|
|||||||
pub refined: String,
|
pub refined: String,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Refine a description that has no mission behind it yet.
|
||||||
|
///
|
||||||
|
/// The wizard's polish button runs BEFORE the mission is created — there is no
|
||||||
|
/// row to load and no id to pass — while [`refine`] deliberately requires a
|
||||||
|
/// saved draft so Accept/Cancel can write back to it. Same prompt, same model
|
||||||
|
/// chain; only where the inputs come from differs.
|
||||||
|
pub async fn refine_draft(
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
|
title: &str,
|
||||||
|
template_kind: &str,
|
||||||
|
phase_kinds: &[String],
|
||||||
|
raw: &str,
|
||||||
|
) -> Result<RefineResult, String> {
|
||||||
|
if raw.trim().is_empty() {
|
||||||
|
return Err("description is empty — nothing to refine".into());
|
||||||
|
}
|
||||||
|
let refined = call_anthropic(runtime, title, template_kind, phase_kinds, raw).await?;
|
||||||
|
Ok(RefineResult {
|
||||||
|
original: raw.to_string(),
|
||||||
|
refined,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
/// Generate a refined description without touching the database. The
|
/// Generate a refined description without touching the database. The
|
||||||
/// caller (frontend) reviews the diff and calls `set_description` to
|
/// caller (frontend) reviews the diff and calls `set_description` to
|
||||||
/// commit — that separation makes Accept/Cancel + undo trivial without
|
/// commit — that separation makes Accept/Cancel + undo trivial without
|
||||||
/// an audit table.
|
/// an audit table.
|
||||||
pub async fn refine(
|
pub async fn refine(
|
||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
workspace_id: cm_domain::WorkspaceId,
|
workspace_id: cm_domain::WorkspaceId,
|
||||||
mission_id: Uuid,
|
mission_id: Uuid,
|
||||||
) -> Result<RefineResult, String> {
|
) -> Result<RefineResult, String> {
|
||||||
@@ -54,7 +80,8 @@ pub async fn refine(
|
|||||||
.collect();
|
.collect();
|
||||||
|
|
||||||
let refined =
|
let refined =
|
||||||
call_anthropic(&mission.title, &mission.template_kind, &phase_kinds, &raw).await?;
|
call_anthropic(runtime, &mission.title, &mission.template_kind, &phase_kinds, &raw)
|
||||||
|
.await?;
|
||||||
|
|
||||||
Ok(RefineResult {
|
Ok(RefineResult {
|
||||||
original: raw,
|
original: raw,
|
||||||
@@ -63,13 +90,12 @@ pub async fn refine(
|
|||||||
}
|
}
|
||||||
|
|
||||||
async fn call_anthropic(
|
async fn call_anthropic(
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
title: &str,
|
title: &str,
|
||||||
template_kind: &str,
|
template_kind: &str,
|
||||||
phase_kinds: &[String],
|
phase_kinds: &[String],
|
||||||
raw: &str,
|
raw: &str,
|
||||||
) -> Result<String, String> {
|
) -> Result<String, String> {
|
||||||
let api_key =
|
|
||||||
std::env::var("ANTHROPIC_API_KEY").map_err(|_| "ANTHROPIC_API_KEY unset".to_string())?;
|
|
||||||
let model = model_name();
|
let model = model_name();
|
||||||
|
|
||||||
let system = "You are a technical brief editor for an autonomous software \
|
let system = "You are a technical brief editor for an autonomous software \
|
||||||
@@ -131,57 +157,14 @@ async fn call_anthropic(
|
|||||||
);
|
);
|
||||||
|
|
||||||
// Opus 4.8 rejects the `temperature` parameter — the model runs at
|
// Opus 4.8 rejects the `temperature` parameter — the model runs at
|
||||||
// its own calibrated setting. Older Claude models accepted 0.0–1.0.
|
// its own calibrated setting. Older Claude models accepted 0.0–1.0, and
|
||||||
let body = json!({
|
// `ChatRequest` does not carry one, so nothing is lost by the move.
|
||||||
"model": model,
|
let (text, answered_by) =
|
||||||
"max_tokens": 4096,
|
crate::subscription::complete_with_fallback(runtime, system, &user, &model, 4096, false)
|
||||||
"system": system,
|
.await?;
|
||||||
"messages": [
|
let text = text.trim().to_string();
|
||||||
{ "role": "user", "content": user }
|
|
||||||
]
|
|
||||||
});
|
|
||||||
|
|
||||||
let client = reqwest::Client::builder()
|
|
||||||
.timeout(std::time::Duration::from_secs(90))
|
|
||||||
.build()
|
|
||||||
.map_err(|e| format!("http client: {e}"))?;
|
|
||||||
let resp = client
|
|
||||||
.post("https://api.anthropic.com/v1/messages")
|
|
||||||
.header("x-api-key", &api_key)
|
|
||||||
.header("anthropic-version", ANTHROPIC_API_VERSION)
|
|
||||||
.header("content-type", "application/json")
|
|
||||||
.json(&body)
|
|
||||||
.send()
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("anthropic call: {e}"))?;
|
|
||||||
if !resp.status().is_success() {
|
|
||||||
let code = resp.status();
|
|
||||||
let body = resp.text().await.unwrap_or_default();
|
|
||||||
return Err(format!(
|
|
||||||
"anthropic {code}: {}",
|
|
||||||
&body[..body.len().min(500)]
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let json: serde_json::Value = resp
|
|
||||||
.json()
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("anthropic json: {e}"))?;
|
|
||||||
// Anthropic Messages API returns content as an array of blocks;
|
|
||||||
// the first text block holds the assistant's reply.
|
|
||||||
let text = json
|
|
||||||
.get("content")
|
|
||||||
.and_then(|c| c.as_array())
|
|
||||||
.and_then(|arr| {
|
|
||||||
arr.iter()
|
|
||||||
.find(|b| b.get("type").and_then(|t| t.as_str()) == Some("text"))
|
|
||||||
})
|
|
||||||
.and_then(|b| b.get("text"))
|
|
||||||
.and_then(|t| t.as_str())
|
|
||||||
.ok_or_else(|| "anthropic response missing text block".to_string())?
|
|
||||||
.trim()
|
|
||||||
.to_string();
|
|
||||||
if text.is_empty() {
|
if text.is_empty() {
|
||||||
return Err("anthropic returned empty text".into());
|
return Err(format!("{answered_by} returned empty text"));
|
||||||
}
|
}
|
||||||
Ok(text)
|
Ok(text)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,373 @@
|
|||||||
|
//! A model-authored roster for a mission — Slice 5.
|
||||||
|
//!
|
||||||
|
//! The Master Planner has been proposing teams (2-6 members, a model each) since
|
||||||
|
//! it shipped, and none of it reached a mission: the proposal lived in React
|
||||||
|
//! state. A mission's shape came instead from a team template — fixed roles, and
|
||||||
|
//! every claw minted `claude-sonnet-5`, which is why no mission has ever run
|
||||||
|
//! heterogeneous providers.
|
||||||
|
//!
|
||||||
|
//! This is the seam. A roster is `(topology_kind, [(role, backend)])`, which is
|
||||||
|
//! exactly what the composed executor consumes: `composed_graph` turns it into a
|
||||||
|
//! `TopologyGraph`, and `MicroVmTurnExecutor` reads `attrs["backend"]` per node,
|
||||||
|
//! so a `validator` role on a different provider's rootfs is a first-class graph
|
||||||
|
//! node rather than a bolt-on.
|
||||||
|
//!
|
||||||
|
//! # Why the backend is validated here and not at boot
|
||||||
|
//!
|
||||||
|
//! Placement already refuses a mission whose backend no online node can run —
|
||||||
|
//! but it refuses it at LAUNCH, after the roster was approved, the mission was
|
||||||
|
//! created and someone believed it was going to run. A model that invents
|
||||||
|
//! `rootfs-opus` is a normal thing for a model to do; discovering it three steps
|
||||||
|
//! later is not. So a roster naming a backend the fleet cannot run is rejected
|
||||||
|
//! when it is proposed, naming the backends that do exist.
|
||||||
|
//!
|
||||||
|
//! # What it deliberately does not do
|
||||||
|
//!
|
||||||
|
//! It does not mint claws. A composed mission's nodes are VMs, and provisioning
|
||||||
|
//! containers for them would create agents and `.brain` files nothing ever
|
||||||
|
//! dials — the same reason `on_launch` returns early for a microVM mission.
|
||||||
|
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// One member of a proposed roster.
|
||||||
|
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||||
|
pub struct RosterMember {
|
||||||
|
/// The node's role, e.g. `implementer`, `verifier`. Becomes the graph node's
|
||||||
|
/// role, which is what the per-node prompt is written around.
|
||||||
|
pub role: String,
|
||||||
|
/// Which rootfs image this node's VM boots (`missions.backend` per node).
|
||||||
|
/// `None` inherits the mission's.
|
||||||
|
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||||
|
pub backend: Option<String>,
|
||||||
|
/// One line on why this member exists. Not consumed by anything — kept
|
||||||
|
/// because a roster nobody can read is a roster nobody can refuse.
|
||||||
|
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||||
|
pub rationale: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A proposed shape for a mission.
|
||||||
|
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||||
|
pub struct Roster {
|
||||||
|
/// A `cm_topology::TopologyKind` name — `pipeline`, `hub_spoke`, …
|
||||||
|
pub topology_kind: String,
|
||||||
|
pub members: Vec<RosterMember>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ceiling on a proposed roster.
|
||||||
|
///
|
||||||
|
/// Each member is a whole VM: a boot, an inject, an agent session and a collect.
|
||||||
|
/// Anthropic's own guidance tops out at 3-5 subagents, and every member here
|
||||||
|
/// costs far more than a subagent does. A model asked to size a team will
|
||||||
|
/// cheerfully propose twelve.
|
||||||
|
pub const MAX_MEMBERS: usize = 6;
|
||||||
|
|
||||||
|
/// Why a roster was refused.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub enum Refusal {
|
||||||
|
Empty,
|
||||||
|
TooMany(usize),
|
||||||
|
BlankRole(usize),
|
||||||
|
/// A backend no online node can run, with the ones that exist.
|
||||||
|
UnknownBackend { backend: String, available: Vec<String> },
|
||||||
|
UnknownTopology(String),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for Refusal {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
match self {
|
||||||
|
Refusal::Empty => write!(f, "the roster has no members, so there is nothing to run"),
|
||||||
|
Refusal::TooMany(n) => write!(
|
||||||
|
f,
|
||||||
|
"the roster has {n} members and the ceiling is {MAX_MEMBERS} — each one is a whole \
|
||||||
|
VM, not a subagent"
|
||||||
|
),
|
||||||
|
Refusal::BlankRole(i) => write!(f, "member {i} has no role"),
|
||||||
|
Refusal::UnknownBackend { backend, available } => write!(
|
||||||
|
f,
|
||||||
|
"no online node can run backend {backend:?}; the fleet has: {}",
|
||||||
|
if available.is_empty() {
|
||||||
|
"(none — no node reports a microvm rootfs)".to_string()
|
||||||
|
} else {
|
||||||
|
available.join(", ")
|
||||||
|
}
|
||||||
|
),
|
||||||
|
Refusal::UnknownTopology(k) => write!(
|
||||||
|
f,
|
||||||
|
"{k:?} is not a topology kind this platform can plan; use one of: {}",
|
||||||
|
cm_topology::TopologyKind::ALL
|
||||||
|
.iter()
|
||||||
|
.map(|k| k.as_str())
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(", ")
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Roster {
|
||||||
|
/// Check a roster against the platform and the fleet.
|
||||||
|
///
|
||||||
|
/// `available` is the set of backends at least one ONLINE node can boot.
|
||||||
|
/// Fail-closed on every axis: an unrecognised topology, a blank role and an
|
||||||
|
/// unbuildable backend are all refusals, because each of them becomes a
|
||||||
|
/// failure much later and much more expensively.
|
||||||
|
pub fn validate(&self, available: &[String]) -> Result<(), Refusal> {
|
||||||
|
if self.members.is_empty() {
|
||||||
|
return Err(Refusal::Empty);
|
||||||
|
}
|
||||||
|
if self.members.len() > MAX_MEMBERS {
|
||||||
|
return Err(Refusal::TooMany(self.members.len()));
|
||||||
|
}
|
||||||
|
if parse_kind(&self.topology_kind).is_none() {
|
||||||
|
return Err(Refusal::UnknownTopology(self.topology_kind.clone()));
|
||||||
|
}
|
||||||
|
for (i, m) in self.members.iter().enumerate() {
|
||||||
|
if m.role.trim().is_empty() {
|
||||||
|
return Err(Refusal::BlankRole(i));
|
||||||
|
}
|
||||||
|
if let Some(b) = m.backend.as_deref().map(str::trim).filter(|b| !b.is_empty()) {
|
||||||
|
if !available.iter().any(|a| a == b) {
|
||||||
|
return Err(Refusal::UnknownBackend {
|
||||||
|
backend: b.to_string(),
|
||||||
|
available: available.to_vec(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The graph a composed run executes.
|
||||||
|
///
|
||||||
|
/// Node ids follow `cm_topology::build`'s `n0..` convention so the graph is
|
||||||
|
/// indistinguishable from a template-built one — the executor, the planners
|
||||||
|
/// and the checkpoint all treat it the same. The per-member backend rides in
|
||||||
|
/// `attrs`, which is the channel `MicroVmTurnExecutor` already reads.
|
||||||
|
pub fn graph(&self) -> Result<serde_json::Value, String> {
|
||||||
|
let kind = parse_kind(&self.topology_kind)
|
||||||
|
.ok_or_else(|| format!("unknown topology kind {:?}", self.topology_kind))?;
|
||||||
|
let roles: Vec<&str> = self.members.iter().map(|m| m.role.trim()).collect();
|
||||||
|
let mut graph =
|
||||||
|
cm_topology::build(kind, &roles).map_err(|e| format!("build topology: {e}"))?;
|
||||||
|
for (node, member) in graph.nodes.iter_mut().zip(self.members.iter()) {
|
||||||
|
if let Some(b) = member
|
||||||
|
.backend
|
||||||
|
.as_deref()
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|b| !b.is_empty())
|
||||||
|
{
|
||||||
|
node.attrs.insert("backend".to_string(), b.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
serde_json::to_value(&graph).map_err(|e| format!("serialize graph: {e}"))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Topology kind by name, accepting exactly what the catalog declares.
|
||||||
|
///
|
||||||
|
/// Deliberately not `unwrap_or(HubSpoke)`. `mission_orchestrator::
|
||||||
|
/// parse_topology_kind` does default, which is right for a stored template
|
||||||
|
/// written by us and wrong for a string a model just invented: silently running
|
||||||
|
/// a `pipeline` proposal as a hub-and-spoke would change what every node sees
|
||||||
|
/// and nothing would say so.
|
||||||
|
fn parse_kind(s: &str) -> Option<cm_topology::TopologyKind> {
|
||||||
|
let want = s.trim();
|
||||||
|
cm_topology::TopologyKind::ALL
|
||||||
|
.iter()
|
||||||
|
.copied()
|
||||||
|
.find(|k| k.as_str().eq_ignore_ascii_case(want))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Backends at least one online node can actually boot.
|
||||||
|
///
|
||||||
|
/// Read from the nodes' reported `rootfs` capability, so it answers "what can
|
||||||
|
/// run today" rather than "what images did someone build once".
|
||||||
|
pub async fn available_backends(
|
||||||
|
pool: &sqlx::PgPool,
|
||||||
|
workspace_id: Uuid,
|
||||||
|
) -> Result<Vec<String>, String> {
|
||||||
|
let rows: Vec<(serde_json::Value,)> = sqlx::query_as(
|
||||||
|
"SELECT capabilities -> 'rootfs'
|
||||||
|
FROM nodes
|
||||||
|
WHERE workspace_id = $1 AND status = 'online'
|
||||||
|
AND capabilities @> '{\"microvm\": true}'::jsonb",
|
||||||
|
)
|
||||||
|
.bind(workspace_id)
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("read node rootfs capabilities: {e}"))?;
|
||||||
|
|
||||||
|
let mut out: Vec<String> = rows
|
||||||
|
.into_iter()
|
||||||
|
.filter_map(|(v,)| v.as_array().cloned())
|
||||||
|
.flatten()
|
||||||
|
.filter_map(|v| v.as_str().map(str::to_string))
|
||||||
|
// A node reports every rootfs it has BUILT, which is not the same as
|
||||||
|
// every rootfs a mission can run in. `agent-terminal` is on tank right
|
||||||
|
// now: bootable, and with no credential contract, so an agent inside it
|
||||||
|
// has nothing to authenticate with. Offering it to the planner would
|
||||||
|
// produce a roster that validates, approves, launches, and then fails at
|
||||||
|
// the agent turn — the expensive kind of late.
|
||||||
|
.filter(|b| crate::mission_runtime::backend_can_run_a_mission(b))
|
||||||
|
.collect();
|
||||||
|
out.sort();
|
||||||
|
out.dedup();
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn member(role: &str, backend: Option<&str>) -> RosterMember {
|
||||||
|
RosterMember {
|
||||||
|
role: role.into(),
|
||||||
|
backend: backend.map(str::to_string),
|
||||||
|
rationale: None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn roster(kind: &str, members: Vec<RosterMember>) -> Roster {
|
||||||
|
Roster {
|
||||||
|
topology_kind: kind.into(),
|
||||||
|
members,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A backend the fleet cannot boot must be refused where it is PROPOSED.
|
||||||
|
/// Placement would refuse it too — at launch, after the roster was approved
|
||||||
|
/// and someone believed the mission was going to run.
|
||||||
|
#[test]
|
||||||
|
fn a_backend_no_node_can_run_is_refused_with_the_ones_that_exist() {
|
||||||
|
let have = vec!["claude".to_string(), "kimi".to_string()];
|
||||||
|
let r = roster(
|
||||||
|
"pipeline",
|
||||||
|
vec![member("implementer", Some("claude")), member("verifier", Some("rootfs-opus"))],
|
||||||
|
);
|
||||||
|
let err = r.validate(&have).unwrap_err();
|
||||||
|
assert_eq!(
|
||||||
|
err,
|
||||||
|
Refusal::UnknownBackend {
|
||||||
|
backend: "rootfs-opus".into(),
|
||||||
|
available: have.clone()
|
||||||
|
}
|
||||||
|
);
|
||||||
|
// The message must name what IS available, or the operator's next move
|
||||||
|
// is a guess.
|
||||||
|
let msg = err.to_string();
|
||||||
|
assert!(msg.contains("claude") && msg.contains("kimi"), "{msg}");
|
||||||
|
|
||||||
|
// And the same roster passes once every backend is one the fleet has.
|
||||||
|
let ok = roster(
|
||||||
|
"pipeline",
|
||||||
|
vec![member("implementer", Some("claude")), member("verifier", Some("kimi"))],
|
||||||
|
);
|
||||||
|
assert!(ok.validate(&have).is_ok());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A member with no backend inherits the mission's, which is legitimate —
|
||||||
|
/// the whole roster does not have to be heterogeneous to be useful.
|
||||||
|
#[test]
|
||||||
|
fn a_member_without_a_backend_is_not_a_refusal() {
|
||||||
|
let r = roster("pipeline", vec![member("implementer", None)]);
|
||||||
|
assert!(r.validate(&["claude".to_string()]).is_ok());
|
||||||
|
// Blank counts as absent, not as a backend named "".
|
||||||
|
let r = roster("pipeline", vec![member("implementer", Some(" "))]);
|
||||||
|
assert!(r.validate(&["claude".to_string()]).is_ok());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A bootable image is not necessarily a runnable one. tank reports
|
||||||
|
/// `agent-terminal` in its rootfs list today: a real image, with no
|
||||||
|
/// credential contract, so an agent booted into it has nothing to
|
||||||
|
/// authenticate with. Offering it to the planner would produce a roster that
|
||||||
|
/// validates, approves, launches and then fails at the agent turn.
|
||||||
|
#[test]
|
||||||
|
fn only_backends_that_can_authenticate_are_offered() {
|
||||||
|
assert!(crate::mission_runtime::backend_can_run_a_mission("claude"));
|
||||||
|
assert!(crate::mission_runtime::backend_can_run_a_mission("default"));
|
||||||
|
for unrunnable in ["agent-terminal", "agent-browser", "rootfs-opus"] {
|
||||||
|
assert!(
|
||||||
|
!crate::mission_runtime::backend_can_run_a_mission(unrunnable),
|
||||||
|
"{unrunnable} has no credential contract and must not be proposable"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The ceiling. Each member is a VM boot, an inject, a full agent session
|
||||||
|
/// and a collect — a model asked to size a team proposes twelve happily.
|
||||||
|
#[test]
|
||||||
|
fn a_roster_is_bounded_and_non_empty() {
|
||||||
|
let have = vec!["claude".to_string()];
|
||||||
|
assert_eq!(roster("pipeline", vec![]).validate(&have), Err(Refusal::Empty));
|
||||||
|
|
||||||
|
let many: Vec<_> = (0..MAX_MEMBERS + 1)
|
||||||
|
.map(|i| member(&format!("r{i}"), None))
|
||||||
|
.collect();
|
||||||
|
assert_eq!(
|
||||||
|
roster("pipeline", many).validate(&have),
|
||||||
|
Err(Refusal::TooMany(MAX_MEMBERS + 1))
|
||||||
|
);
|
||||||
|
|
||||||
|
let exactly: Vec<_> = (0..MAX_MEMBERS).map(|i| member(&format!("r{i}"), None)).collect();
|
||||||
|
assert!(roster("pipeline", exactly).validate(&have).is_ok());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An invented topology kind must be refused, NOT defaulted. Running a
|
||||||
|
/// `pipeline` proposal as a hub-and-spoke changes what every node sees and
|
||||||
|
/// nothing would say so — the same silent-substitution shape as a backend
|
||||||
|
/// that quietly falls back to the default image.
|
||||||
|
#[test]
|
||||||
|
fn an_invented_topology_kind_is_refused_rather_than_defaulted() {
|
||||||
|
let have = vec!["claude".to_string()];
|
||||||
|
let r = roster("assembly_line", vec![member("implementer", None)]);
|
||||||
|
assert_eq!(
|
||||||
|
r.validate(&have),
|
||||||
|
Err(Refusal::UnknownTopology("assembly_line".into()))
|
||||||
|
);
|
||||||
|
// Every kind the catalog declares is accepted, so this cannot drift out
|
||||||
|
// of sync with what the orchestrator can actually plan.
|
||||||
|
for kind in cm_topology::TopologyKind::ALL {
|
||||||
|
let r = roster(kind.as_str(), vec![member("implementer", None)]);
|
||||||
|
assert!(r.validate(&have).is_ok(), "{}", kind.as_str());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The graph is the handoff to the composed executor: node ids in
|
||||||
|
/// `cm_topology`'s own convention, and the backend in the `attrs` channel
|
||||||
|
/// `MicroVmTurnExecutor` reads. If this drifts, a heterogeneous roster runs
|
||||||
|
/// every node on the mission default and looks fine.
|
||||||
|
#[test]
|
||||||
|
fn the_graph_carries_each_members_backend_where_the_executor_reads_it() {
|
||||||
|
let r = roster(
|
||||||
|
"pipeline",
|
||||||
|
vec![
|
||||||
|
member("implementer", Some("claude")),
|
||||||
|
member("verifier", Some("kimi")),
|
||||||
|
member("scribe", None),
|
||||||
|
],
|
||||||
|
);
|
||||||
|
let g = r.graph().expect("a runnable graph");
|
||||||
|
let nodes = g["nodes"].as_array().expect("nodes");
|
||||||
|
assert_eq!(nodes.len(), 3);
|
||||||
|
assert_eq!(nodes[0]["role"], "implementer");
|
||||||
|
assert_eq!(nodes[0]["attrs"]["backend"], "claude");
|
||||||
|
assert_eq!(nodes[1]["attrs"]["backend"], "kimi");
|
||||||
|
assert!(
|
||||||
|
nodes[2]["attrs"].get("backend").is_none(),
|
||||||
|
"a member with no backend must inherit the mission's, not be stamped with one"
|
||||||
|
);
|
||||||
|
|
||||||
|
// And it deserializes as the real thing the worker will parse — a graph
|
||||||
|
// that only looks right as JSON fails at claim time with "missing or
|
||||||
|
// invalid graph", which reads as a runtime fault rather than a bad
|
||||||
|
// roster.
|
||||||
|
let parsed: cm_topology::TopologyGraph =
|
||||||
|
serde_json::from_value(g).expect("the worker must be able to parse it");
|
||||||
|
assert_eq!(parsed.nodes.len(), 3);
|
||||||
|
assert_eq!(
|
||||||
|
parsed.nodes[1].attrs.get("backend").map(String::as_str),
|
||||||
|
Some("kimi")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,321 @@
|
|||||||
|
//! Launching missions that are due.
|
||||||
|
//!
|
||||||
|
//! `missions.schedule` has carried a cron since `0047_missions.sql` — the
|
||||||
|
//! wizard collects it, the API persists it — and until this module nothing ever
|
||||||
|
//! read it back. The only due-work enumerator in the codebase was
|
||||||
|
//! `routines::claim_due`, so **every scheduled mission ever created sat in
|
||||||
|
//! `draft` forever** while the UI reported it was on a schedule. Measured
|
||||||
|
//! before this was written: a mission with `* * * * *` did not move for four
|
||||||
|
//! minutes and started no runs.
|
||||||
|
//!
|
||||||
|
//! The shape here is deliberately `cm-scheduler`'s, not a second invention:
|
||||||
|
//!
|
||||||
|
//! - **Claim atomically** (`FOR UPDATE SKIP LOCKED`) so replicas fire once.
|
||||||
|
//! - **Advance the clock before dispatching**, so a failing launch cannot stall
|
||||||
|
//! the schedule.
|
||||||
|
//! - **Record the claim first** in `mission_fires`, keyed by the occurrence's
|
||||||
|
//! own timestamp, so a crash between those two is retried rather than
|
||||||
|
//! silently dropped — and a slot already launched is never launched twice.
|
||||||
|
//! - **Cap the fan-out**, because a backlog would otherwise start one container
|
||||||
|
//! per missed occurrence.
|
||||||
|
//!
|
||||||
|
//! The one thing it does NOT share with routines is the launch itself: a due
|
||||||
|
//! mission goes through `mission_orchestrator::on_launch` and
|
||||||
|
//! `missions::set_status`, exactly as the draft→running transition in
|
||||||
|
//! `routes::missions::set_status` does, so there is one path that mints a crew.
|
||||||
|
|
||||||
|
use cm_db::repo::missions as missions_repo;
|
||||||
|
use sqlx::{PgPool, Row};
|
||||||
|
use time::OffsetDateTime;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
/// Most missions one tick will launch.
|
||||||
|
///
|
||||||
|
/// Lower than the scheduler's 25: a mission firing is a container, a repo
|
||||||
|
/// checkout and real model spend, where a routine firing may be a single turn.
|
||||||
|
/// The remainder stays due and is taken by the next tick.
|
||||||
|
const MAX_LAUNCHES_PER_TICK: usize = 5;
|
||||||
|
|
||||||
|
/// A mission whose occurrence has come due and been claimed.
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct DueMission {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub workspace_id: Uuid,
|
||||||
|
pub title: String,
|
||||||
|
pub cron: Option<String>,
|
||||||
|
/// The occurrence that came due — the value `next_run_at` held. Identifies
|
||||||
|
/// the slot in `mission_fires`, so it must not be re-read from the clock.
|
||||||
|
pub slot: OffsetDateTime,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Claim every mission due at `now`, atomically.
|
||||||
|
///
|
||||||
|
/// `next_run_at` is cleared by the claim. The caller recomputes it from the
|
||||||
|
/// cron and writes it back; a mission whose cron no longer yields an occurrence
|
||||||
|
/// simply stays cleared and stops firing, which is the correct end state for
|
||||||
|
/// a one-shot or an exhausted schedule.
|
||||||
|
pub async fn claim_due(pool: &PgPool, now: OffsetDateTime) -> Result<Vec<DueMission>, String> {
|
||||||
|
let rows = sqlx::query(
|
||||||
|
"UPDATE missions SET next_run_at = NULL
|
||||||
|
WHERE id IN (
|
||||||
|
SELECT id FROM missions
|
||||||
|
WHERE next_run_at IS NOT NULL
|
||||||
|
AND next_run_at <= $1
|
||||||
|
-- Never relaunch a mission that is mid-flight. A daily cron on
|
||||||
|
-- a mission that takes longer than a day must skip the
|
||||||
|
-- occurrence, not stack a second crew on the same workspace.
|
||||||
|
AND status <> 'running'
|
||||||
|
FOR UPDATE SKIP LOCKED
|
||||||
|
)
|
||||||
|
RETURNING id, workspace_id, title, schedule ->> 'cron' AS cron, $1::timestamptz AS slot",
|
||||||
|
)
|
||||||
|
.bind(now)
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("claim due missions: {e}"))?;
|
||||||
|
|
||||||
|
Ok(rows
|
||||||
|
.into_iter()
|
||||||
|
.map(|r| DueMission {
|
||||||
|
id: r.get("id"),
|
||||||
|
workspace_id: r.get("workspace_id"),
|
||||||
|
title: r.get("title"),
|
||||||
|
cron: r.get("cron"),
|
||||||
|
slot: r.get("slot"),
|
||||||
|
})
|
||||||
|
.collect())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Record that this occurrence was taken. `false` means another replica (or an
|
||||||
|
/// earlier attempt) already has it and this one must not launch.
|
||||||
|
async fn claim_slot(pool: &PgPool, mission_id: Uuid, slot: OffsetDateTime) -> Result<bool, String> {
|
||||||
|
let inserted = sqlx::query(
|
||||||
|
"INSERT INTO mission_fires (mission_id, scheduled_at, status)
|
||||||
|
VALUES ($1, $2, 'claimed')
|
||||||
|
ON CONFLICT (mission_id, scheduled_at) DO NOTHING",
|
||||||
|
)
|
||||||
|
.bind(mission_id)
|
||||||
|
.bind(slot)
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("claim mission fire: {e}"))?;
|
||||||
|
Ok(inserted.rows_affected() == 1)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn settle_slot(
|
||||||
|
pool: &PgPool,
|
||||||
|
mission_id: Uuid,
|
||||||
|
slot: OffsetDateTime,
|
||||||
|
status: &str,
|
||||||
|
detail: Option<&str>,
|
||||||
|
) {
|
||||||
|
if let Err(e) = sqlx::query(
|
||||||
|
"UPDATE mission_fires SET status = $3, detail = $4, completed_at = now()
|
||||||
|
WHERE mission_id = $1 AND scheduled_at = $2",
|
||||||
|
)
|
||||||
|
.bind(mission_id)
|
||||||
|
.bind(slot)
|
||||||
|
.bind(status)
|
||||||
|
.bind(detail)
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!("mission_schedule: settling {mission_id} @ {slot} as {status}: {e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Compute and persist the next occurrence.
|
||||||
|
///
|
||||||
|
/// A cron that will not parse is reported and the mission left un-scheduled
|
||||||
|
/// rather than skipped in silence — the whole point of this module is that a
|
||||||
|
/// schedule which does nothing must never look like a schedule that works.
|
||||||
|
async fn reschedule(pool: &PgPool, m: &DueMission, after: OffsetDateTime) {
|
||||||
|
let Some(cron) = m.cron.as_deref().map(str::trim).filter(|c| !c.is_empty()) else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
match cm_runtime::scheduling::next_occurrence(cron, after) {
|
||||||
|
Ok(next) => {
|
||||||
|
if let Err(e) = sqlx::query("UPDATE missions SET next_run_at = $2 WHERE id = $1")
|
||||||
|
.bind(m.id)
|
||||||
|
.bind(next)
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!("mission_schedule: could not set next_run_at for {}: {e}", m.id);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(e) => eprintln!(
|
||||||
|
"mission_schedule: mission {} ({}) has an unusable cron {cron:?} — it will NOT run \
|
||||||
|
again until the schedule is corrected: {e}",
|
||||||
|
m.id, m.title
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One pass. Returns how many missions were launched.
|
||||||
|
pub async fn tick(
|
||||||
|
pool: &PgPool,
|
||||||
|
node_hub: Option<std::sync::Arc<crate::fleet::NodeHub>>,
|
||||||
|
blobs: Option<std::sync::Arc<dyn cm_files::BlobStore>>,
|
||||||
|
now: OffsetDateTime,
|
||||||
|
) -> Result<usize, String> {
|
||||||
|
let due = claim_due(pool, now).await?;
|
||||||
|
let mut launched = 0usize;
|
||||||
|
|
||||||
|
for m in due.iter().take(MAX_LAUNCHES_PER_TICK) {
|
||||||
|
// Clock first: a launch that fails must not stall the schedule.
|
||||||
|
reschedule(pool, m, now).await;
|
||||||
|
|
||||||
|
if !claim_slot(pool, m.id, m.slot).await? {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
// An unattended launch still needs an actor. Missions carry no creator
|
||||||
|
// column, so the workspace owner stands in — the same identity the
|
||||||
|
// audit trail already attributes workspace-level action to.
|
||||||
|
let workspace = cm_domain::WorkspaceId::from(m.workspace_id);
|
||||||
|
let owner = match cm_db::repo::users::owner_of_workspace(pool, workspace).await {
|
||||||
|
Ok(u) => u,
|
||||||
|
Err(e) => {
|
||||||
|
// `fetch_one`, so "no owner" arrives as RowNotFound rather than
|
||||||
|
// None. Either way the occurrence is settled `failed` with the
|
||||||
|
// reason, never dropped quietly.
|
||||||
|
let why = format!("no owner to launch as: {e}");
|
||||||
|
eprintln!("mission_schedule: cannot launch {} — {why}", m.id);
|
||||||
|
settle_slot(pool, m.id, m.slot, "failed", Some(&why)).await;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
match crate::mission_orchestrator::on_launch(
|
||||||
|
pool,
|
||||||
|
workspace,
|
||||||
|
owner,
|
||||||
|
m.id,
|
||||||
|
node_hub.clone(),
|
||||||
|
blobs.clone(),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(_) => {
|
||||||
|
if let Err(e) =
|
||||||
|
missions_repo::set_status(pool, m.id, m.workspace_id, "running").await
|
||||||
|
{
|
||||||
|
let why = format!("launched but could not mark running: {e}");
|
||||||
|
eprintln!("mission_schedule: {} — {why}", m.id);
|
||||||
|
settle_slot(pool, m.id, m.slot, "failed", Some(&why)).await;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
settle_slot(pool, m.id, m.slot, "fired", None).await;
|
||||||
|
launched += 1;
|
||||||
|
eprintln!(
|
||||||
|
"mission_schedule: launched {} ({}) for occurrence {}",
|
||||||
|
m.id, m.title, m.slot
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("mission_schedule: on_launch failed for {}: {e}", m.id);
|
||||||
|
settle_slot(pool, m.id, m.slot, "failed", Some(&e)).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if due.len() > MAX_LAUNCHES_PER_TICK {
|
||||||
|
eprintln!(
|
||||||
|
"mission_schedule: {} due, launched {} this tick (cap {}); the rest stay due",
|
||||||
|
due.len(),
|
||||||
|
launched,
|
||||||
|
MAX_LAUNCHES_PER_TICK
|
||||||
|
);
|
||||||
|
}
|
||||||
|
Ok(launched)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spawn the sweep.
|
||||||
|
pub fn spawn(
|
||||||
|
pool: PgPool,
|
||||||
|
node_hub: Option<std::sync::Arc<crate::fleet::NodeHub>>,
|
||||||
|
blobs: Option<std::sync::Arc<dyn cm_files::BlobStore>>,
|
||||||
|
interval: std::time::Duration,
|
||||||
|
) {
|
||||||
|
tokio::spawn(async move {
|
||||||
|
let mut ticker = tokio::time::interval(interval);
|
||||||
|
// Skip the immediate first tick so a restart loop cannot become a
|
||||||
|
// launch loop.
|
||||||
|
ticker.tick().await;
|
||||||
|
loop {
|
||||||
|
ticker.tick().await;
|
||||||
|
let now = OffsetDateTime::now_utc();
|
||||||
|
match tick(&pool, node_hub.clone(), blobs.clone(), now).await {
|
||||||
|
Ok(n) if n > 0 => eprintln!("mission_schedule: launched {n} due mission(s)"),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => eprintln!("mission_schedule: sweep failed: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The cap is what stops a backlog turning into a container stampede. A
|
||||||
|
/// clock jump or a cron that resolves to "every minute" can leave hundreds
|
||||||
|
/// of occurrences owed; each mission launch is a container, a checkout and
|
||||||
|
/// real model spend, so this must stay well below the routine scheduler's
|
||||||
|
/// 25.
|
||||||
|
#[test]
|
||||||
|
fn the_launch_cap_is_conservative() {
|
||||||
|
assert!(
|
||||||
|
MAX_LAUNCHES_PER_TICK <= 5,
|
||||||
|
"a mission firing costs far more than a routine firing"
|
||||||
|
);
|
||||||
|
assert!(MAX_LAUNCHES_PER_TICK >= 1, "a cap of zero never launches");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The claim must never pick up a mission that is already running.
|
||||||
|
///
|
||||||
|
/// A daily cron on a mission that takes longer than a day would otherwise
|
||||||
|
/// stack a second crew on the same workspace — two containers, two vault
|
||||||
|
/// branches, and a seen-set race. Asserted against the SQL text because the
|
||||||
|
/// predicate is the whole safety property and it lives only in the query.
|
||||||
|
#[test]
|
||||||
|
fn the_claim_skips_missions_that_are_still_running() {
|
||||||
|
// Re-read the source of the query this module issues.
|
||||||
|
let src = include_str!("mission_schedule.rs");
|
||||||
|
let claim = src
|
||||||
|
.split("pub async fn claim_due")
|
||||||
|
.nth(1)
|
||||||
|
.expect("claim_due exists");
|
||||||
|
let body = &claim[..claim.find("fetch_all").unwrap_or(claim.len())];
|
||||||
|
assert!(
|
||||||
|
body.contains("status <> 'running'"),
|
||||||
|
"claim_due must not relaunch a mission that is mid-flight"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
body.contains("FOR UPDATE SKIP LOCKED"),
|
||||||
|
"the claim must be atomic or replicas double-launch"
|
||||||
|
);
|
||||||
|
assert!(
|
||||||
|
body.contains("next_run_at <= $1"),
|
||||||
|
"only occurrences that have come due may be claimed"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The clock advances BEFORE the launch, and the slot is claimed before the
|
||||||
|
/// launch too. Both orderings matter: reschedule-first means a failing
|
||||||
|
/// launch cannot stall the schedule; claim-first means a crash mid-launch
|
||||||
|
/// is retried rather than dropped.
|
||||||
|
#[test]
|
||||||
|
fn the_clock_advances_before_the_launch_is_attempted() {
|
||||||
|
let src = include_str!("mission_schedule.rs");
|
||||||
|
let tick = src.split("pub async fn tick").nth(1).expect("tick exists");
|
||||||
|
let resched = tick.find("reschedule(pool, m, now)").expect("reschedules");
|
||||||
|
let claim = tick.find("claim_slot(pool, m.id, m.slot)").expect("claims");
|
||||||
|
let launch = tick.find("on_launch(").expect("launches");
|
||||||
|
assert!(
|
||||||
|
resched < claim && claim < launch,
|
||||||
|
"order must be reschedule -> claim -> launch (got {resched}, {claim}, {launch})"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,7 +11,10 @@
|
|||||||
//! - no repo_id → no-op (Ok(None))
|
//! - no repo_id → no-op (Ok(None))
|
||||||
//! - dir already a git repo → `fetch + reset --hard origin/<branch>`
|
//! - dir already a git repo → `fetch + reset --hard origin/<branch>`
|
||||||
//! to bring it in sync
|
//! to bring it in sync
|
||||||
//! - dir missing → `git clone --depth 1 <url> <path>`
|
//! - dir missing → `git clone --filter=blob:none --single-branch` (NOT
|
||||||
|
//! `--depth 1`: a shallow clone cannot push a new branch back, and delivery
|
||||||
|
//! needs exactly that — see `clone`). A mission with a `security_scan` phase
|
||||||
|
//! gets a FULLY HYDRATED clone instead; see `wants_full_history`.
|
||||||
//!
|
//!
|
||||||
//! Auth: for `git.redclaw.dev` clones we inject the ambient
|
//! Auth: for `git.redclaw.dev` clones we inject the ambient
|
||||||
//! `GITEA_TOKEN` (already provisioned in the server container's env)
|
//! `GITEA_TOKEN` (already provisioned in the server container's env)
|
||||||
@@ -23,7 +26,17 @@ use std::path::PathBuf;
|
|||||||
use tokio::process::Command;
|
use tokio::process::Command;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
pub(crate) fn missions_root() -> PathBuf {
|
/// The ONE definition of where mission state lives on the docker host.
|
||||||
|
///
|
||||||
|
/// There used to be five: this function, three private copies of the same
|
||||||
|
/// `env::var(...).unwrap_or(...)` in `security_scan`, `benchmark_runner` and
|
||||||
|
/// `mission_outputs`, and a hardcoded `MISSIONS_HOST_ROOT` const in
|
||||||
|
/// `mission_runtime` that read no env at all. They agree on today's
|
||||||
|
/// deployment, which is why nothing had broken — but anything that sweeps or
|
||||||
|
/// reclaims this tree has to be sure it is sweeping the same tree the writers
|
||||||
|
/// use, and five definitions cannot promise that. A GC written against one of
|
||||||
|
/// them would silently miss the others.
|
||||||
|
pub fn missions_root() -> PathBuf {
|
||||||
std::env::var("CLAWMATES_MISSIONS_ROOT")
|
std::env::var("CLAWMATES_MISSIONS_ROOT")
|
||||||
.map(PathBuf::from)
|
.map(PathBuf::from)
|
||||||
.unwrap_or_else(|_| PathBuf::from("/var/lib/clawmates-missions"))
|
.unwrap_or_else(|_| PathBuf::from("/var/lib/clawmates-missions"))
|
||||||
@@ -65,7 +78,19 @@ pub async fn ensure_checkout(
|
|||||||
.map_err(|e| format!("mkdir {}: {e}", parent.display()))?;
|
.map_err(|e| format!("mkdir {}: {e}", parent.display()))?;
|
||||||
}
|
}
|
||||||
|
|
||||||
let auth_url = with_ambient_auth(clone_url);
|
let auth = with_ambient_auth(clone_url);
|
||||||
|
// Said once, here, where the checkout is created: every later git call uses
|
||||||
|
// a URL built the same way, so an unauthenticated forge URL is a fact worth
|
||||||
|
// one line now rather than a `/dev/tty` error later.
|
||||||
|
if let Some(why) = &auth.unauthenticated {
|
||||||
|
if auth.is_forge() {
|
||||||
|
eprintln!(
|
||||||
|
"mission_workspace: mission {mission_id} will talk to the forge \
|
||||||
|
WITHOUT credentials — {why}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let auth_url = auth.url;
|
||||||
if path.join(".git").exists() {
|
if path.join(".git").exists() {
|
||||||
// Checkouts cloned before this setting existed get it on reuse. It
|
// Checkouts cloned before this setting existed get it on reuse. It
|
||||||
// governs objects created from now on, which is what delivery needs.
|
// governs objects created from now on, which is what delivery needs.
|
||||||
@@ -87,43 +112,181 @@ pub async fn ensure_checkout(
|
|||||||
fetch_and_reset(&path, default_branch, &auth_url).await?;
|
fetch_and_reset(&path, default_branch, &auth_url).await?;
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
clone(&path, &auth_url).await?;
|
clone(&path, &auth_url, wants_full_history(pool, mission_id).await).await?;
|
||||||
}
|
}
|
||||||
Ok(Some(path))
|
Ok(Some(path))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// If the URL points at git.redclaw.dev AND GITEA_TOKEN is set in the
|
/// The forge whose URLs the ambient `GITEA_TOKEN` can authenticate.
|
||||||
/// environment, rewrite it to include the token as basic-auth. Returns
|
const FORGE_HOST: &str = "git.redclaw.dev";
|
||||||
/// the URL unchanged otherwise. The token is never logged (we only
|
|
||||||
/// pass the rewritten URL into `git clone` via argv).
|
/// A URL, and whether a credential actually reached it.
|
||||||
pub(crate) fn with_ambient_auth(url: &str) -> String {
|
///
|
||||||
let Ok(token) = std::env::var("GITEA_TOKEN") else {
|
/// The second field is the whole point. This used to be a bare `String`: an
|
||||||
return url.to_string();
|
/// unmatched URL — an ssh remote, `http://` instead of `https://`, an explicit
|
||||||
};
|
/// port, a different case in the host — silently came back unauthenticated, and
|
||||||
if token.is_empty() {
|
/// the first symptom was git opening `/dev/tty` several layers later. Tracing
|
||||||
return url.to_string();
|
/// #55 cost hours to a failure whose cause was one unlogged early return.
|
||||||
}
|
pub struct Authed {
|
||||||
if let Some(rest) = url.strip_prefix("https://git.redclaw.dev/") {
|
pub url: String,
|
||||||
return format!("https://oauth2:{token}@git.redclaw.dev/{rest}");
|
/// `None` when the token was applied; otherwise WHY it was not.
|
||||||
}
|
pub unauthenticated: Option<String>,
|
||||||
url.to_string()
|
/// Whether the URL names the forge our token is for. Recorded from the
|
||||||
|
/// ORIGINAL url, not re-derived from `url` — an authenticated URL carries
|
||||||
|
/// userinfo, and parsing that back out is how the answer goes wrong.
|
||||||
|
forge: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn clone(path: &std::path::Path, url: &str) -> Result<(), String> {
|
impl Authed {
|
||||||
|
/// Is this URL on the forge our token is for? An unauthenticated URL to a
|
||||||
|
/// third-party host is normal (public repos, ssh remotes with a key); an
|
||||||
|
/// unauthenticated URL to OUR forge is a fault, and only the caller knows
|
||||||
|
/// how much it costs.
|
||||||
|
pub fn is_forge(&self) -> bool {
|
||||||
|
self.forge
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The host component of a URL, for the scp-like and scheme forms git accepts.
|
||||||
|
///
|
||||||
|
/// Deliberately tolerant, because the point is to RECOGNISE our forge in every
|
||||||
|
/// shape it can be written, not to validate URLs: `https://`, `http://`, an
|
||||||
|
/// explicit `:port`, `user@host`, `ssh://`, and `git@host:path`.
|
||||||
|
fn host_of(url: &str) -> Option<&str> {
|
||||||
|
let rest = match url.split_once("://") {
|
||||||
|
Some((_, rest)) => rest,
|
||||||
|
// scp-like: `git@host:path/to.git`, which has no scheme.
|
||||||
|
None => url,
|
||||||
|
};
|
||||||
|
// Userinfo FIRST, then the port. The other order splits
|
||||||
|
// `oauth2:token@host` at the credential's colon and reports the username as
|
||||||
|
// the host — which is exactly how the first version of this function decided
|
||||||
|
// an authenticated forge URL was not the forge.
|
||||||
|
let authority = rest.split('/').next().filter(|s| !s.is_empty())?;
|
||||||
|
let hostport = authority.rsplit_once('@').map_or(authority, |(_, h)| h);
|
||||||
|
Some(hostport.split(':').next().unwrap_or(hostport)).filter(|h| !h.is_empty())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rewrite a forge URL to carry the ambient `GITEA_TOKEN` as basic-auth.
|
||||||
|
///
|
||||||
|
/// Returns the reason instead of the credential whenever it cannot: a missing
|
||||||
|
/// token, a host that is not ours, or a shape a token cannot be injected into.
|
||||||
|
/// The token is never logged — only the rewritten URL is passed to git, via
|
||||||
|
/// argv.
|
||||||
|
pub fn with_ambient_auth(url: &str) -> Authed {
|
||||||
|
auth_with_token(url, std::env::var("GITEA_TOKEN").ok().as_deref())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The testable half of [`with_ambient_auth`]. The token is a parameter because
|
||||||
|
/// a test cannot set process environment variables here — the workspace denies
|
||||||
|
/// `unsafe`, and `set_var` is racy across test threads regardless.
|
||||||
|
fn auth_with_token(url: &str, token: Option<&str>) -> Authed {
|
||||||
|
let host = host_of(url);
|
||||||
|
let forge = host.is_some_and(|h| h.eq_ignore_ascii_case(FORGE_HOST));
|
||||||
|
let unauth = |why: String| Authed {
|
||||||
|
url: url.to_string(),
|
||||||
|
unauthenticated: Some(why),
|
||||||
|
forge,
|
||||||
|
};
|
||||||
|
|
||||||
|
let host = match host {
|
||||||
|
Some(h) => h,
|
||||||
|
None => return unauth(format!("no host could be read from {url:?}")),
|
||||||
|
};
|
||||||
|
if !forge {
|
||||||
|
return unauth(format!(
|
||||||
|
"{host} is not {FORGE_HOST}, so GITEA_TOKEN does not apply — git will \
|
||||||
|
use whatever ambient credentials exist (ssh agent, .netrc, helper)"
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let token = match token {
|
||||||
|
Some(t) if !t.trim().is_empty() => t,
|
||||||
|
_ => return unauth("GITEA_TOKEN is unset or empty".to_string()),
|
||||||
|
};
|
||||||
|
// Only the scheme forms can carry basic-auth. An ssh remote authenticates
|
||||||
|
// with a key, and pretending otherwise would produce a URL git rejects.
|
||||||
|
let Some((scheme, rest)) = url.split_once("://") else {
|
||||||
|
return unauth(format!(
|
||||||
|
"{url} is an ssh-style remote; a token cannot be embedded in it"
|
||||||
|
));
|
||||||
|
};
|
||||||
|
if !matches!(scheme, "http" | "https") {
|
||||||
|
return unauth(format!("scheme {scheme} cannot carry a token"));
|
||||||
|
}
|
||||||
|
// Drop any userinfo already present rather than producing `a@b@host`.
|
||||||
|
let rest = rest.split_once('@').map(|(_, r)| r).unwrap_or(rest);
|
||||||
|
Authed {
|
||||||
|
url: format!("{scheme}://oauth2:{token}@{rest}"),
|
||||||
|
unauthenticated: None,
|
||||||
|
forge,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Git must never wait for a human.
|
||||||
|
///
|
||||||
|
/// Without this, a URL that ended up without credentials does not fail — git
|
||||||
|
/// opens `/dev/tty` to ask for a username, and in a server container that
|
||||||
|
/// surfaces as `No such device or address`, several layers away from the
|
||||||
|
/// missing token that caused it. With it, the failure names itself:
|
||||||
|
/// `terminal prompts disabled`.
|
||||||
|
pub(crate) fn no_terminal_prompt(cmd: &mut Command) -> &mut Command {
|
||||||
|
cmd.env("GIT_TERMINAL_PROMPT", "0")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Does any phase of this mission need history it can READ, not just reference?
|
||||||
|
///
|
||||||
|
/// `--filter=blob:none` keeps every commit but fetches file contents on demand,
|
||||||
|
/// which is nearly free for a repo that gets read once — and silently useless to
|
||||||
|
/// a tool that walks history, because the agent environment has NO network route
|
||||||
|
/// to the forge. Measured: gitleaks on a 4-commit repo reported
|
||||||
|
/// "1 commits scanned" and "could not fetch <sha> from promisor remote". It was
|
||||||
|
/// not misconfigured; the blobs simply were not there and could not be got.
|
||||||
|
///
|
||||||
|
/// A security scan is the phase kind whose entire value is old content — a
|
||||||
|
/// credential committed and later deleted is exactly what it looks for, and that
|
||||||
|
/// is precisely what a lazy blob is. So those missions pay for a full clone and
|
||||||
|
/// everything else keeps the cheap one.
|
||||||
|
///
|
||||||
|
/// Best-effort: an unreadable phase list yields `false`, i.e. today's behaviour.
|
||||||
|
async fn wants_full_history(pool: &sqlx::PgPool, mission_id: Uuid) -> bool {
|
||||||
|
sqlx::query_scalar::<_, i64>(
|
||||||
|
"SELECT count(*) FROM mission_phases WHERE mission_id = $1 AND kind = 'security_scan'",
|
||||||
|
)
|
||||||
|
.bind(mission_id)
|
||||||
|
.fetch_one(pool)
|
||||||
|
.await
|
||||||
|
.map(|n| n > 0)
|
||||||
|
.unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The `git clone` flags, split out so the strategy is testable without a forge.
|
||||||
|
fn clone_args(full_history: bool) -> Vec<&'static str> {
|
||||||
|
let mut a = vec!["clone"];
|
||||||
|
if !full_history {
|
||||||
|
a.push("--filter=blob:none");
|
||||||
|
}
|
||||||
|
a.push("--single-branch");
|
||||||
|
a
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn clone(path: &std::path::Path, url: &str, full_history: bool) -> Result<(), String> {
|
||||||
// `--filter=blob:none` rather than `--depth 1`. A shallow clone cannot
|
// `--filter=blob:none` rather than `--depth 1`. A shallow clone cannot
|
||||||
// usually push a new branch back ("shallow update not allowed"), and
|
// usually push a new branch back ("shallow update not allowed"), and
|
||||||
// mission delivery needs exactly that. A partial clone keeps full history
|
// mission delivery needs exactly that. A partial clone keeps full history
|
||||||
// — so the base commit stays meaningful and a diff has something to be
|
// — so the base commit stays meaningful and a diff has something to be
|
||||||
// relative to — while fetching file contents only on demand, which is
|
// relative to — while fetching file contents only on demand, which is
|
||||||
// nearly as cheap as a shallow clone for a repo that gets read once.
|
// nearly as cheap as a shallow clone for a repo that gets read once.
|
||||||
let out = Command::new("git")
|
if full_history {
|
||||||
.args([
|
eprintln!(
|
||||||
"clone",
|
"mission_workspace: cloning {} with full history — a security_scan phase \
|
||||||
"--filter=blob:none",
|
reads old file contents, which a partial clone cannot supply offline",
|
||||||
"--single-branch",
|
path.display()
|
||||||
url,
|
);
|
||||||
&path.display().to_string(),
|
}
|
||||||
])
|
let mut cmd = Command::new("git");
|
||||||
|
cmd.args(clone_args(full_history));
|
||||||
|
cmd.args([url, &path.display().to_string()]);
|
||||||
|
let out = no_terminal_prompt(&mut cmd)
|
||||||
.output()
|
.output()
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("spawn git clone: {e}"))?;
|
.map_err(|e| format!("spawn git clone: {e}"))?;
|
||||||
@@ -131,10 +294,11 @@ async fn clone(path: &std::path::Path, url: &str) -> Result<(), String> {
|
|||||||
return Err(format!(
|
return Err(format!(
|
||||||
"git clone → exit {}: {}",
|
"git clone → exit {}: {}",
|
||||||
out.status,
|
out.status,
|
||||||
redact_token(&String::from_utf8_lossy(&out.stderr))
|
// Both ends: git prints its reason LAST, and a head-only clamp keeps
|
||||||
.chars()
|
// the progress noise while dropping the answer.
|
||||||
.take(400)
|
crate::evaluator_tools::clamp_output(&redact_token(&String::from_utf8_lossy(
|
||||||
.collect::<String>()
|
&out.stderr
|
||||||
|
)))
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
share_repository_across_uids(path);
|
share_repository_across_uids(path);
|
||||||
@@ -447,7 +611,13 @@ fn strip_credentials(url: &str) -> String {
|
|||||||
/// They are the agent's identity scaffolding, not the user's code — `SOUL.md`
|
/// They are the agent's identity scaffolding, not the user's code — `SOUL.md`
|
||||||
/// opens "Who You Are / You're not a chatbot." Observed on mission 019fc058,
|
/// opens "Who You Are / You're not a chatbot." Observed on mission 019fc058,
|
||||||
/// where all seven appeared as untracked files in a freshly cloned repo.
|
/// where all seven appeared as untracked files in a freshly cloned repo.
|
||||||
const AGENT_SCAFFOLDING: &[&str] = &[
|
///
|
||||||
|
/// `pub(crate)` because a repo-less mission needs the same list and cannot use
|
||||||
|
/// the same mechanism: `ignore_agent_scaffolding` writes `.git/info/exclude`,
|
||||||
|
/// and a mission with no repository has no `.git`. `mission_outputs` filters on
|
||||||
|
/// this list directly — one list, two consumers, so the next file the runtime
|
||||||
|
/// starts seeding is excluded from both at once.
|
||||||
|
pub(crate) const AGENT_SCAFFOLDING: &[&str] = &[
|
||||||
"AGENTS.md",
|
"AGENTS.md",
|
||||||
"HEARTBEAT.md",
|
"HEARTBEAT.md",
|
||||||
"IDENTITY.md",
|
"IDENTITY.md",
|
||||||
@@ -523,16 +693,15 @@ async fn fetch_and_reset(
|
|||||||
// errors on a repo that is already complete, so it is only attempted when
|
// errors on a repo that is already complete, so it is only attempted when
|
||||||
// the marker file is present.
|
// the marker file is present.
|
||||||
if path.join(".git/shallow").exists() {
|
if path.join(".git/shallow").exists() {
|
||||||
let deepen = Command::new("git")
|
let mut cmd = Command::new("git");
|
||||||
.args([
|
cmd.args([
|
||||||
"-C",
|
"-C",
|
||||||
&path.display().to_string(),
|
&path.display().to_string(),
|
||||||
"fetch",
|
"fetch",
|
||||||
"--unshallow",
|
"--unshallow",
|
||||||
auth_url,
|
auth_url,
|
||||||
])
|
]);
|
||||||
.output()
|
let deepen = no_terminal_prompt(&mut cmd).output().await;
|
||||||
.await;
|
|
||||||
match deepen {
|
match deepen {
|
||||||
Ok(o) if o.status.success() => {}
|
Ok(o) if o.status.success() => {}
|
||||||
Ok(o) => eprintln!(
|
Ok(o) => eprintln!(
|
||||||
@@ -556,8 +725,9 @@ async fn fetch_and_reset(
|
|||||||
// so `git fetch origin` has no credentials and fails with
|
// so `git fetch origin` has no credentials and fails with
|
||||||
// "could not read Username". Building the URL here also means a rotated
|
// "could not read Username". Building the URL here also means a rotated
|
||||||
// token takes effect immediately instead of at the next clone.
|
// token takes effect immediately instead of at the next clone.
|
||||||
let fetch = Command::new("git")
|
let mut cmd = Command::new("git");
|
||||||
.args(["-C", &path.display().to_string(), "fetch", auth_url, branch])
|
cmd.args(["-C", &path.display().to_string(), "fetch", auth_url, branch]);
|
||||||
|
let fetch = no_terminal_prompt(&mut cmd)
|
||||||
.output()
|
.output()
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("spawn git fetch: {e}"))?;
|
.map_err(|e| format!("spawn git fetch: {e}"))?;
|
||||||
@@ -565,10 +735,9 @@ async fn fetch_and_reset(
|
|||||||
return Err(format!(
|
return Err(format!(
|
||||||
"git fetch origin {branch} → exit {}: {}",
|
"git fetch origin {branch} → exit {}: {}",
|
||||||
fetch.status,
|
fetch.status,
|
||||||
redact_token(&String::from_utf8_lossy(&fetch.stderr))
|
crate::evaluator_tools::clamp_output(&redact_token(&String::from_utf8_lossy(
|
||||||
.chars()
|
&fetch.stderr
|
||||||
.take(400)
|
)))
|
||||||
.collect::<String>()
|
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
let reset = Command::new("git")
|
let reset = Command::new("git")
|
||||||
@@ -600,8 +769,102 @@ async fn fetch_and_reset(
|
|||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
|
/// The clone strategy is a fact worth pinning: `--depth 1` breaks delivery
|
||||||
|
/// (a shallow clone cannot push a new branch — "shallow update not
|
||||||
|
/// allowed"), and `--filter=blob:none` breaks history-reading tools offline.
|
||||||
|
/// Both failure modes are real and were both hit.
|
||||||
|
#[test]
|
||||||
|
fn the_clone_strategy_is_partial_by_default_and_never_shallow() {
|
||||||
|
let partial = clone_args(false);
|
||||||
|
assert!(partial.contains(&"--filter=blob:none"), "{partial:?}");
|
||||||
|
assert!(!partial.iter().any(|a| a.starts_with("--depth")), "{partial:?}");
|
||||||
|
|
||||||
|
// A security_scan mission must NOT get the lazy-blob filter: its scanner
|
||||||
|
// walks old file contents and cannot reach the forge to fetch them.
|
||||||
|
let full = clone_args(true);
|
||||||
|
assert!(!full.contains(&"--filter=blob:none"), "{full:?}");
|
||||||
|
assert!(!full.iter().any(|a| a.starts_with("--depth")), "{full:?}");
|
||||||
|
|
||||||
|
// Both keep --single-branch: the mission only ever works one branch.
|
||||||
|
for args in [partial, full] {
|
||||||
|
assert!(args.contains(&"--single-branch"), "{args:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
|
const TOK: Option<&str> = Some("secret123");
|
||||||
|
|
||||||
|
/// The forge in every shape a remote can be written. Each of these used to
|
||||||
|
/// fall out of the `strip_prefix("https://git.redclaw.dev/")` match and come
|
||||||
|
/// back unauthenticated with no log line — the fail-open found while tracing
|
||||||
|
/// #55.
|
||||||
|
#[test]
|
||||||
|
fn the_forge_is_recognised_however_the_url_is_written() {
|
||||||
|
for url in [
|
||||||
|
"https://git.redclaw.dev/o/r.git",
|
||||||
|
"http://git.redclaw.dev/o/r.git",
|
||||||
|
"https://GIT.RedClaw.dev/o/r.git",
|
||||||
|
"https://git.redclaw.dev:3000/o/r.git",
|
||||||
|
"https://oauth2:[email protected]/o/r.git",
|
||||||
|
] {
|
||||||
|
let a = auth_with_token(url, TOK);
|
||||||
|
assert!(a.is_forge(), "{url} was not recognised as the forge");
|
||||||
|
assert!(
|
||||||
|
a.unauthenticated.is_none(),
|
||||||
|
"{url} → {:?}",
|
||||||
|
a.unauthenticated
|
||||||
|
);
|
||||||
|
assert!(a.url.contains("oauth2:secret123@"), "{}", a.url);
|
||||||
|
// And exactly one set of credentials, not `old@` left behind.
|
||||||
|
assert_eq!(a.url.matches('@').count(), 1, "{}", a.url);
|
||||||
|
}
|
||||||
|
// The port and the scheme survive the rewrite — changing either would
|
||||||
|
// point the push somewhere the operator did not configure.
|
||||||
|
assert!(auth_with_token("https://git.redclaw.dev:3000/o/r.git", TOK)
|
||||||
|
.url
|
||||||
|
.contains("@git.redclaw.dev:3000/o/r.git"));
|
||||||
|
assert!(auth_with_token("http://git.redclaw.dev/o/r.git", TOK)
|
||||||
|
.url
|
||||||
|
.starts_with("http://oauth2:"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every path that cannot authenticate must SAY so. "Unauthenticated and
|
||||||
|
/// silent" is the shape that cost hours: the first symptom was git opening
|
||||||
|
/// /dev/tty, several layers from the cause.
|
||||||
|
#[test]
|
||||||
|
fn an_unauthenticated_url_carries_its_reason() {
|
||||||
|
let cases = [
|
||||||
|
(auth_with_token("https://git.redclaw.dev/o/r.git", None), true),
|
||||||
|
(auth_with_token("https://git.redclaw.dev/o/r.git", Some(" ")), true),
|
||||||
|
(auth_with_token("[email protected]:o/r.git", TOK), true),
|
||||||
|
(auth_with_token("ssh://[email protected]/o/r.git", TOK), true),
|
||||||
|
(auth_with_token("https://github.com/o/r.git", TOK), false),
|
||||||
|
];
|
||||||
|
for (a, is_forge) in cases {
|
||||||
|
let why = a.unauthenticated.as_deref().unwrap_or("");
|
||||||
|
assert!(!why.is_empty(), "{} came back with no reason", a.url);
|
||||||
|
assert_eq!(a.is_forge(), is_forge, "{}", a.url);
|
||||||
|
// And the URL is handed back untouched, so a caller that proceeds
|
||||||
|
// anyway (ssh keys, .netrc) still works.
|
||||||
|
assert!(!a.url.contains("secret123"), "{}", a.url);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A token must never be embedded in a URL for someone else's host.
|
||||||
|
#[test]
|
||||||
|
fn the_token_never_leaves_the_forge() {
|
||||||
|
for url in [
|
||||||
|
"https://github.com/o/r.git",
|
||||||
|
"https://git.redclaw.dev.evil.example/o/r.git",
|
||||||
|
"https://evil.example/git.redclaw.dev/r.git",
|
||||||
|
] {
|
||||||
|
let a = auth_with_token(url, TOK);
|
||||||
|
assert!(!a.url.contains("secret123"), "{url} → {}", a.url);
|
||||||
|
assert!(!a.is_forge(), "{url}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// The exclude must be idempotent — `ensure_checkout` re-runs on every
|
/// The exclude must be idempotent — `ensure_checkout` re-runs on every
|
||||||
/// phase, and appending the same block each time would grow the file
|
/// phase, and appending the same block each time would grow the file
|
||||||
/// without bound.
|
/// without bound.
|
||||||
@@ -821,4 +1084,43 @@ mod tests {
|
|||||||
mark_phase_started(repo);
|
mark_phase_started(repo);
|
||||||
assert!(checkout_in_use(repo));
|
assert!(checkout_in_use(repo));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Nobody re-derives the missions root.
|
||||||
|
///
|
||||||
|
/// It had fragmented into five definitions — this function, three private
|
||||||
|
/// `env::var("CLAWMATES_MISSIONS_ROOT")` copies, and a hardcoded const
|
||||||
|
/// that read no env at all. They agreed on the deployed value, so nothing
|
||||||
|
/// ever broke; the risk is entirely in what comes next. Anything that
|
||||||
|
/// sweeps, reclaims or reaps this tree has to be sweeping the same tree the
|
||||||
|
/// writers use, and five definitions cannot promise that.
|
||||||
|
#[test]
|
||||||
|
fn the_missions_root_has_exactly_one_definition() {
|
||||||
|
fn walk(dir: &std::path::Path, out: &mut Vec<std::path::PathBuf>) {
|
||||||
|
for entry in std::fs::read_dir(dir).expect("readable source dir") {
|
||||||
|
let path = entry.expect("readable entry").path();
|
||||||
|
if path.is_dir() {
|
||||||
|
walk(&path, out);
|
||||||
|
} else if path.extension().is_some_and(|e| e == "rs") {
|
||||||
|
out.push(path);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let root = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("src");
|
||||||
|
let mut files = Vec::new();
|
||||||
|
walk(&root, &mut files);
|
||||||
|
|
||||||
|
for path in files {
|
||||||
|
if path.ends_with("mission_workspace.rs") {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let src = std::fs::read_to_string(&path).expect("readable source");
|
||||||
|
assert!(
|
||||||
|
!src.contains("var(\"CLAWMATES_MISSIONS_ROOT\")"),
|
||||||
|
"{} reads CLAWMATES_MISSIONS_ROOT itself — call \
|
||||||
|
`mission_workspace::missions_root()` so a reaper and a writer \
|
||||||
|
cannot disagree about which tree they are looking at",
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+140
-1
@@ -143,13 +143,89 @@ fn unescape(s: &str) -> String {
|
|||||||
.replace("'", "'")
|
.replace("'", "'")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Turn an operator topic into an arXiv `search_query`.
|
||||||
|
///
|
||||||
|
/// A bare topic is NOT a search. Passed through unfielded, arXiv matched
|
||||||
|
/// essentially nothing and `sortBy=submittedDate` then returned the newest
|
||||||
|
/// submissions across the whole archive — so a run for "speculative decoding"
|
||||||
|
/// shelved Galois extensions, a quantum black hole microstate, and blazar dark
|
||||||
|
/// matter in IceCube. Measured against the live API:
|
||||||
|
///
|
||||||
|
/// ```text
|
||||||
|
/// speculative decoding -> pixel-space diffusion, simplicial actions
|
||||||
|
/// all:"speculative decoding" -> S2-MoE self-speculative decoding, DARTree
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// So the phrase is quoted into `all:` (title, abstract, authors, comments) and
|
||||||
|
/// constrained to `cat:cs.*` — this library exists to serve software projects,
|
||||||
|
/// and without the category bound the archive's physics and maths volume
|
||||||
|
/// dominates every recency-sorted result.
|
||||||
|
///
|
||||||
|
/// A topic that already looks fielded (`cat:`, `ti:`, `abs:`, `all:`) is passed
|
||||||
|
/// through untouched, so an operator who knows arXiv's syntax keeps full control.
|
||||||
|
pub fn arxiv_query(topic: &str) -> String {
|
||||||
|
let t = topic.trim();
|
||||||
|
const FIELDED: &[&str] = &["all:", "ti:", "abs:", "au:", "cat:", "co:", "jr:"];
|
||||||
|
// Only a topic that STARTS with a field prefix is treated as hand-written
|
||||||
|
// arXiv syntax. Also accepting anything containing " AND "/" OR " was the
|
||||||
|
// first version, and a test caught it immediately: `agent" OR cat:hep-th`
|
||||||
|
// passed straight through, so a topic string could escape the phrase and
|
||||||
|
// rewrite the category bound. A natural-language topic may legitimately
|
||||||
|
// contain the word "and" too.
|
||||||
|
if FIELDED.iter().any(|p| t.starts_with(p)) {
|
||||||
|
return t.to_string();
|
||||||
|
}
|
||||||
|
// Quotes make it a phrase; without them "vector index pruning" matches any
|
||||||
|
// paper containing all three words anywhere, which is most of cs.
|
||||||
|
let escaped = t.replace('"', "");
|
||||||
|
format!("all:\"{escaped}\" AND cat:cs.*")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The looser form of a topic: every term required, but not adjacent.
|
||||||
|
///
|
||||||
|
/// A quoted phrase is precise and brittle. "hybrid retrieval BM25 dense" is a
|
||||||
|
/// perfectly good topic and appears verbatim in no paper on arXiv — measured, 0
|
||||||
|
/// hits — while requiring the same four terms anywhere returns exactly the
|
||||||
|
/// hybrid-retrieval evaluations the topic was asking for. Used only when the
|
||||||
|
/// phrase finds nothing, so an exact match still wins when one exists.
|
||||||
|
pub fn arxiv_query_broad(topic: &str) -> String {
|
||||||
|
let terms: Vec<String> = topic
|
||||||
|
.split_whitespace()
|
||||||
|
.map(|w| w.trim_matches(|c: char| !c.is_alphanumeric() && c != '-'))
|
||||||
|
.filter(|w| !w.is_empty())
|
||||||
|
.map(|w| format!("all:{w}"))
|
||||||
|
.collect();
|
||||||
|
if terms.is_empty() {
|
||||||
|
return arxiv_query(topic);
|
||||||
|
}
|
||||||
|
format!("{} AND cat:cs.*", terms.join(" AND "))
|
||||||
|
}
|
||||||
|
|
||||||
/// Search arXiv. `max_results` is capped to keep one run bounded.
|
/// Search arXiv. `max_results` is capped to keep one run bounded.
|
||||||
pub async fn search(query: &str, max_results: usize) -> Result<Vec<Paper>, String> {
|
pub async fn search(query: &str, max_results: usize) -> Result<Vec<Paper>, String> {
|
||||||
|
let found = search_with(&arxiv_query(query), max_results).await?;
|
||||||
|
if !found.is_empty() {
|
||||||
|
return Ok(found);
|
||||||
|
}
|
||||||
|
// The phrase matched nothing. Before reporting a quiet day — which the whole
|
||||||
|
// pipeline treats as a real and legitimate outcome — try the same terms
|
||||||
|
// unquoted. A topic the operator writes as prose often is not a literal
|
||||||
|
// phrase in any title, and silently harvesting zero because of punctuation
|
||||||
|
// would be indistinguishable from a genuinely quiet field.
|
||||||
|
let broad = arxiv_query_broad(query);
|
||||||
|
if broad == arxiv_query(query) {
|
||||||
|
return Ok(found);
|
||||||
|
}
|
||||||
|
eprintln!("papers: no exact phrase match for {query:?} — retrying as {broad}");
|
||||||
|
search_with(&broad, max_results).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn search_with(search_query: &str, max_results: usize) -> Result<Vec<Paper>, String> {
|
||||||
let max = max_results.clamp(1, 50);
|
let max = max_results.clamp(1, 50);
|
||||||
let url = format!(
|
let url = format!(
|
||||||
"https://export.arxiv.org/api/query?search_query={}&start=0&max_results={max}\
|
"https://export.arxiv.org/api/query?search_query={}&start=0&max_results={max}\
|
||||||
&sortBy=submittedDate&sortOrder=descending",
|
&sortBy=submittedDate&sortOrder=descending",
|
||||||
urlencoding(query)
|
urlencoding(search_query)
|
||||||
);
|
);
|
||||||
let body = reqwest::Client::new()
|
let body = reqwest::Client::new()
|
||||||
.get(&url)
|
.get(&url)
|
||||||
@@ -251,6 +327,69 @@ fn urlencoding(s: &str) -> String {
|
|||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
|
|
||||||
|
/// A bare topic must become a PHRASE search bound to cs — unfielded, arXiv
|
||||||
|
/// matched nothing and recency-sort returned the whole archive, so a run
|
||||||
|
/// for "speculative decoding" shelved blazar dark matter in IceCube.
|
||||||
|
#[test]
|
||||||
|
fn a_bare_topic_becomes_a_fielded_phrase_query() {
|
||||||
|
let q = arxiv_query("speculative decoding");
|
||||||
|
assert_eq!(q, "all:\"speculative decoding\" AND cat:cs.*");
|
||||||
|
assert!(q.contains('"'), "unquoted, the words match separately");
|
||||||
|
assert!(q.contains("cat:cs.*"), "without a category bound physics wins");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An operator who writes arXiv syntax keeps control — wrapping their query
|
||||||
|
/// in another `all:"..."` would search for the literal text of their query.
|
||||||
|
#[test]
|
||||||
|
fn an_already_fielded_topic_is_left_alone() {
|
||||||
|
for q in [
|
||||||
|
"cat:cs.IR AND all:\"dense retrieval\"",
|
||||||
|
"ti:\"world model\"",
|
||||||
|
"abs:hnsw OR abs:\"vector index\"",
|
||||||
|
] {
|
||||||
|
assert_eq!(arxiv_query(q), q, "{q} must pass through untouched");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The broad form requires every term but not adjacency. Measured: the
|
||||||
|
/// phrase "hybrid retrieval BM25 dense" has 0 hits on arXiv; the same four
|
||||||
|
/// terms unquoted return the hybrid-retrieval evaluations that were asked
|
||||||
|
/// for. Without the fallback that topic silently harvests nothing, which is
|
||||||
|
/// indistinguishable from a genuinely quiet day.
|
||||||
|
#[test]
|
||||||
|
fn the_broad_form_requires_every_term_without_adjacency() {
|
||||||
|
let q = arxiv_query_broad("hybrid retrieval BM25 dense");
|
||||||
|
assert_eq!(
|
||||||
|
q,
|
||||||
|
"all:hybrid AND all:retrieval AND all:BM25 AND all:dense AND cat:cs.*"
|
||||||
|
);
|
||||||
|
assert!(!q.contains('"'), "the broad form must not be a phrase: {q}");
|
||||||
|
assert!(q.contains("cat:cs.*"), "still category-bound: {q}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Punctuation must not leak into a term and must not empty the query.
|
||||||
|
#[test]
|
||||||
|
fn the_broad_form_strips_punctuation_and_never_empties() {
|
||||||
|
assert_eq!(
|
||||||
|
arxiv_query_broad("retrieval-augmented, generation!"),
|
||||||
|
"all:retrieval-augmented AND all:generation AND cat:cs.*",
|
||||||
|
"hyphens are part of a term; trailing punctuation is not"
|
||||||
|
);
|
||||||
|
// Nothing usable left: fall back to the phrase form rather than
|
||||||
|
// emitting a bare `cat:cs.*`, which would match all of computer science.
|
||||||
|
let q = arxiv_query_broad("!!!");
|
||||||
|
assert!(q.contains("all:"), "must never degrade to a bare category: {q}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Quotes in a topic would terminate the phrase early and corrupt the query.
|
||||||
|
#[test]
|
||||||
|
fn quotes_in_a_topic_cannot_break_out_of_the_phrase() {
|
||||||
|
let q = arxiv_query("agent\" OR cat:hep-th");
|
||||||
|
assert_eq!(q.matches('"').count(), 2, "exactly one balanced phrase: {q}");
|
||||||
|
assert!(q.ends_with("cat:cs.*"), "{q}");
|
||||||
|
}
|
||||||
|
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
/// A revision must not read as a new paper.
|
/// A revision must not read as a new paper.
|
||||||
|
|||||||
@@ -1,259 +0,0 @@
|
|||||||
//! LLM + Chromium PDF renderer worker — Slice 6.
|
|
||||||
//!
|
|
||||||
//! Watches `mission_artifacts` for rows with `render_pdf_status =
|
|
||||||
//! 'pending'`. For each:
|
|
||||||
//! 1. Read the source MD from `<mission_root>/<path>` on disk
|
|
||||||
//! 2. Call the configured LLM (default: Gemini 2.5 Flash) with a
|
|
||||||
//! "produce styled HTML" prompt anchored to a design-system
|
|
||||||
//! example. LLM writes HTML with inline CSS.
|
|
||||||
//! 3. Print that HTML to PDF via `chromium --headless
|
|
||||||
//! --print-to-pdf`
|
|
||||||
//! 4. Save the PDF alongside the MD, update `rendered_pdf_path` +
|
|
||||||
//! status = 'done'
|
|
||||||
//!
|
|
||||||
//! Graceful degradation: if `GEMINI_API_KEY` is unset or the
|
|
||||||
//! chromium binary isn't on PATH, the worker marks the row `failed`
|
|
||||||
//! with a descriptive error rather than blocking boot. Ops enables
|
|
||||||
//! rendering by wiring both.
|
|
||||||
//!
|
|
||||||
//! The frontend already renders `rendered_pdf_path` as an "Open PDF"
|
|
||||||
//! button on artifact cards (Slice 2).
|
|
||||||
|
|
||||||
use serde_json::json;
|
|
||||||
use sqlx::PgPool;
|
|
||||||
use std::path::{Path, PathBuf};
|
|
||||||
use std::time::Duration;
|
|
||||||
|
|
||||||
const POLL_INTERVAL: Duration = Duration::from_secs(30);
|
|
||||||
const MAX_PARALLEL: usize = 2;
|
|
||||||
const DEFAULT_MODEL: &str = "gemini-2.5-flash";
|
|
||||||
|
|
||||||
/// Where per-mission artifacts land on disk. Overridable so dev vs.
|
|
||||||
/// prod can move the tree; matches the pattern in
|
|
||||||
/// `research_container::research_workspace_root`.
|
|
||||||
fn missions_root() -> PathBuf {
|
|
||||||
std::env::var("CLAWMATES_MISSIONS_ROOT")
|
|
||||||
.map(PathBuf::from)
|
|
||||||
.unwrap_or_else(|_| PathBuf::from("/var/lib/clawmates-missions"))
|
|
||||||
}
|
|
||||||
|
|
||||||
fn chromium_bin() -> String {
|
|
||||||
std::env::var("CHROMIUM_BIN").unwrap_or_else(|_| "chromium".to_string())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn renderer_model() -> String {
|
|
||||||
std::env::var("CLAWMATES_PDF_RENDERER_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_string())
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Spawn the poller. No-op-friendly: if there's nothing pending or
|
|
||||||
/// no rendering pipeline configured, we still tick + observe.
|
|
||||||
pub fn spawn(pool: PgPool) {
|
|
||||||
tokio::spawn(async move {
|
|
||||||
// Small startup delay so migrations + loaders finish first.
|
|
||||||
tokio::time::sleep(Duration::from_secs(8)).await;
|
|
||||||
let mut ticker = tokio::time::interval(POLL_INTERVAL);
|
|
||||||
ticker.tick().await;
|
|
||||||
loop {
|
|
||||||
ticker.tick().await;
|
|
||||||
if let Err(e) = sweep_once(&pool).await {
|
|
||||||
eprintln!("pdf_renderer: sweep failed: {e}");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn sweep_once(pool: &PgPool) -> Result<(), String> {
|
|
||||||
let pending = cm_db::repo::missions::next_pdf_pending(pool, MAX_PARALLEL as i64)
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("next_pdf_pending: {e}"))?;
|
|
||||||
for artifact in pending {
|
|
||||||
let pool = pool.clone();
|
|
||||||
let id = artifact.id;
|
|
||||||
tokio::spawn(async move {
|
|
||||||
match render_one(&pool, &artifact).await {
|
|
||||||
Ok(pdf_path) => {
|
|
||||||
let _ = cm_db::repo::missions::set_pdf_result(&pool, id, Some(&pdf_path), None)
|
|
||||||
.await;
|
|
||||||
eprintln!("pdf_renderer: rendered {id} → {pdf_path}");
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
let _ = cm_db::repo::missions::set_pdf_result(&pool, id, None, Some(&e)).await;
|
|
||||||
eprintln!("pdf_renderer: {id} failed: {e}");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
}
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn render_one(
|
|
||||||
_pool: &PgPool,
|
|
||||||
artifact: &cm_db::repo::missions::MissionArtifact,
|
|
||||||
) -> Result<String, String> {
|
|
||||||
// 1. Locate the source MD on disk.
|
|
||||||
let mission_root = missions_root().join(artifact.mission_id.to_string());
|
|
||||||
let src_path = mission_root.join(&artifact.path);
|
|
||||||
let md = tokio::fs::read_to_string(&src_path)
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("read {}: {e}", src_path.display()))?;
|
|
||||||
|
|
||||||
// 2. LLM → styled HTML.
|
|
||||||
let html = md_to_html_via_llm(&md, artifact.title.as_deref())
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("llm render: {e}"))?;
|
|
||||||
|
|
||||||
// 3. Chromium → PDF.
|
|
||||||
let tmp = tempdir_for(artifact.id)?;
|
|
||||||
let html_path = tmp.join("in.html");
|
|
||||||
let pdf_path = tmp.join("out.pdf");
|
|
||||||
tokio::fs::write(&html_path, html)
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("write {}: {e}", html_path.display()))?;
|
|
||||||
|
|
||||||
let status = tokio::process::Command::new(chromium_bin())
|
|
||||||
.args([
|
|
||||||
"--headless=new",
|
|
||||||
"--disable-gpu",
|
|
||||||
"--no-sandbox",
|
|
||||||
"--hide-scrollbars",
|
|
||||||
&format!("--print-to-pdf={}", pdf_path.display()),
|
|
||||||
"--print-to-pdf-no-header",
|
|
||||||
"--virtual-time-budget=10000",
|
|
||||||
&format!("file://{}", html_path.display()),
|
|
||||||
])
|
|
||||||
.stderr(std::process::Stdio::piped())
|
|
||||||
.stdout(std::process::Stdio::piped())
|
|
||||||
.status()
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("spawn chromium: {e}"))?;
|
|
||||||
if !status.success() {
|
|
||||||
return Err(format!("chromium exited {status}"));
|
|
||||||
}
|
|
||||||
|
|
||||||
// 4. Move next to the source MD so the artifact tree stays self-
|
|
||||||
// contained. Filename derived from the MD path (foo.md → foo.pdf).
|
|
||||||
let out_rel = pdf_sibling(&artifact.path);
|
|
||||||
let out_abs = mission_root.join(&out_rel);
|
|
||||||
if let Some(parent) = out_abs.parent() {
|
|
||||||
tokio::fs::create_dir_all(parent)
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("mkdir {}: {e}", parent.display()))?;
|
|
||||||
}
|
|
||||||
tokio::fs::copy(&pdf_path, &out_abs)
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("copy pdf: {e}"))?;
|
|
||||||
// Best-effort tmp cleanup — the temp dir lives under /tmp so the
|
|
||||||
// OS will reap it anyway.
|
|
||||||
let _ = tokio::fs::remove_dir_all(&tmp).await;
|
|
||||||
Ok(out_rel)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Ask the configured LLM to turn `md` into a fully self-contained
|
|
||||||
/// styled HTML doc. Uses whichever provider `CLAWMATES_PDF_RENDERER_MODEL`
|
|
||||||
/// resolves to. Defaults to Gemini 2.5 Flash + GEMINI_API_KEY.
|
|
||||||
async fn md_to_html_via_llm(md: &str, title: Option<&str>) -> Result<String, String> {
|
|
||||||
let model = renderer_model();
|
|
||||||
// For now we hardcode the Gemini path — anthropic + openai
|
|
||||||
// variants land when the design-system template stabilizes.
|
|
||||||
if !model.starts_with("gemini") {
|
|
||||||
return Err(format!(
|
|
||||||
"renderer model {model} not yet wired (only gemini-* supported in Slice 6)"
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let api_key =
|
|
||||||
std::env::var("GEMINI_API_KEY").map_err(|_| "GEMINI_API_KEY unset".to_string())?;
|
|
||||||
|
|
||||||
let system = r#"You are a document typesetter. Given a Markdown source,
|
|
||||||
produce ONE self-contained HTML document that:
|
|
||||||
- Has ALL styles inline in a single <style> block in <head>. No external
|
|
||||||
fonts, no external CSS. System font stack only.
|
|
||||||
- Uses a clean, modern, readable serif for body copy (Georgia / "Iowan Old
|
|
||||||
Style" / "Charter" / serif) and a sans for headings.
|
|
||||||
- Uses ONLY these accent colors: #ff8a7a (heading), #5ec8d8 (link),
|
|
||||||
#101014 (body text), #f7f7f8 (page bg).
|
|
||||||
- Renders code blocks with a monospace stack and a subtle background.
|
|
||||||
- Uses page-break-inside: avoid on headings and images.
|
|
||||||
- Puts a document title in an <h1> at the top if provided.
|
|
||||||
- Includes NOTHING outside the HTML — no ```html fence, no commentary."#;
|
|
||||||
|
|
||||||
let prompt = match title {
|
|
||||||
Some(t) => format!("Document title: {t}\n\nMarkdown:\n\n{md}"),
|
|
||||||
None => md.to_string(),
|
|
||||||
};
|
|
||||||
|
|
||||||
let url = format!(
|
|
||||||
"https://generativelanguage.googleapis.com/v1beta/models/{}:generateContent?key={}",
|
|
||||||
model, api_key
|
|
||||||
);
|
|
||||||
let body = json!({
|
|
||||||
"system_instruction": { "parts": [{ "text": system }] },
|
|
||||||
"contents": [{ "role": "user", "parts": [{ "text": prompt }] }],
|
|
||||||
"generationConfig": {
|
|
||||||
"temperature": 0.2,
|
|
||||||
"maxOutputTokens": 32000,
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
let client = reqwest::Client::builder()
|
|
||||||
.timeout(Duration::from_secs(120))
|
|
||||||
.build()
|
|
||||||
.map_err(|e| format!("http client: {e}"))?;
|
|
||||||
let resp = client
|
|
||||||
.post(&url)
|
|
||||||
.json(&body)
|
|
||||||
.send()
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("gemini call: {e}"))?;
|
|
||||||
if !resp.status().is_success() {
|
|
||||||
let code = resp.status();
|
|
||||||
let body = resp.text().await.unwrap_or_default();
|
|
||||||
return Err(format!("gemini {code}: {}", &body[..body.len().min(500)]));
|
|
||||||
}
|
|
||||||
let json: serde_json::Value = resp.json().await.map_err(|e| format!("gemini json: {e}"))?;
|
|
||||||
let text = json
|
|
||||||
.pointer("/candidates/0/content/parts/0/text")
|
|
||||||
.and_then(|v| v.as_str())
|
|
||||||
.ok_or_else(|| "gemini response missing text".to_string())?;
|
|
||||||
// Strip a stray ```html fence if the model added one despite the
|
|
||||||
// system prompt — cheap belt to the suspenders.
|
|
||||||
let cleaned = text
|
|
||||||
.trim()
|
|
||||||
.strip_prefix("```html")
|
|
||||||
.and_then(|s| s.strip_suffix("```"))
|
|
||||||
.map(|s| s.trim())
|
|
||||||
.unwrap_or(text.trim())
|
|
||||||
.to_string();
|
|
||||||
Ok(cleaned)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn tempdir_for(id: uuid::Uuid) -> Result<PathBuf, String> {
|
|
||||||
let dir = std::env::temp_dir().join(format!("clawmates-pdf-{id}"));
|
|
||||||
std::fs::create_dir_all(&dir).map_err(|e| format!("mkdir tmp: {e}"))?;
|
|
||||||
Ok(dir)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// `research/v3/spec.md` → `research/v3/spec.pdf`.
|
|
||||||
/// `foo/bar/without_ext` → `foo/bar/without_ext.pdf` (rare — parser
|
|
||||||
/// never emits an extension-less MD, but we're defensive).
|
|
||||||
fn pdf_sibling(md_path: &str) -> String {
|
|
||||||
let p = Path::new(md_path);
|
|
||||||
let stem = p.file_stem().and_then(|s| s.to_str()).unwrap_or("output");
|
|
||||||
let parent = p.parent().map(|x| x.to_string_lossy().to_string());
|
|
||||||
let base = format!("{stem}.pdf");
|
|
||||||
match parent {
|
|
||||||
Some(pp) if !pp.is_empty() => format!("{pp}/{base}"),
|
|
||||||
_ => base,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
#[cfg(test)]
|
|
||||||
mod tests {
|
|
||||||
use super::*;
|
|
||||||
|
|
||||||
#[test]
|
|
||||||
fn pdf_sibling_paths() {
|
|
||||||
assert_eq!(pdf_sibling("research/v3/spec.md"), "research/v3/spec.pdf");
|
|
||||||
assert_eq!(pdf_sibling("spec.md"), "spec.pdf");
|
|
||||||
assert_eq!(pdf_sibling("no_ext"), "no_ext.pdf");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -47,6 +47,44 @@ pub const KNOWN_KEYS: &[KnownKey] = &[
|
|||||||
key: "commit_policy",
|
key: "commit_policy",
|
||||||
read_by: "mission_delivery::Gate::parse — selects the delivery gate",
|
read_by: "mission_delivery::Gate::parse — selects the delivery gate",
|
||||||
},
|
},
|
||||||
|
KnownKey {
|
||||||
|
key: "allow_empty",
|
||||||
|
read_by: "phase_runner::empty_delivery_is_a_failure — when true, a coding \
|
||||||
|
phase that changes no files still completes; also vm_stop_gate::\
|
||||||
|
StopGate::for_phase, where it drops the in-loop delivery check",
|
||||||
|
},
|
||||||
|
KnownKey {
|
||||||
|
key: "tools",
|
||||||
|
read_by: "security_scan::run — gates which of cargo_audit / gitleaks / \
|
||||||
|
trivy_fs / semgrep run against the phase's checkout; absent \
|
||||||
|
means all four. Listed here as NOT IMPLEMENTED while wired, \
|
||||||
|
which understated the recipe: the key was real, what was \
|
||||||
|
missing was anything that FIRED the scan outside an operator \
|
||||||
|
button — now phase_runner::scan_finished_security_phases",
|
||||||
|
},
|
||||||
|
KnownKey {
|
||||||
|
key: "harness",
|
||||||
|
read_by: "benchmark_runner::harness_from_config — selects criterion / \
|
||||||
|
cargo_bench / vitest_bench / pytest_bench / shell, with \
|
||||||
|
`bench_name` (criterion) and `cmd` (shell) as its arguments. \
|
||||||
|
phase_runner's benchmark sweep runs the baseline through it. \
|
||||||
|
This key was listed as NOT IMPLEMENTED while being fully \
|
||||||
|
wired, which is worse than an unread key: the registry exists \
|
||||||
|
so an operator can trust what a recipe does, and it was wrong",
|
||||||
|
},
|
||||||
|
KnownKey {
|
||||||
|
key: "bench_name",
|
||||||
|
read_by: "benchmark_runner::harness_from_config — the criterion bench target",
|
||||||
|
},
|
||||||
|
KnownKey {
|
||||||
|
key: "cmd",
|
||||||
|
read_by: "benchmark_runner::harness_from_config — the shell harness command line",
|
||||||
|
},
|
||||||
|
KnownKey {
|
||||||
|
key: "done_when_check",
|
||||||
|
read_by: "vm_stop_gate::StopGate::for_phase — a shell command the agent's \
|
||||||
|
`Stop` hook runs, refusing the stop while it exits non-zero",
|
||||||
|
},
|
||||||
];
|
];
|
||||||
|
|
||||||
/// Keys a recipe may carry that are deliberately not consumed *yet*.
|
/// Keys a recipe may carry that are deliberately not consumed *yet*.
|
||||||
@@ -73,14 +111,6 @@ pub const DECLARED_BUT_UNREAD: &[KnownKey] = &[
|
|||||||
key: "mode",
|
key: "mode",
|
||||||
read_by: "NOT IMPLEMENTED — benchmark/refactor mode selection",
|
read_by: "NOT IMPLEMENTED — benchmark/refactor mode selection",
|
||||||
},
|
},
|
||||||
KnownKey {
|
|
||||||
key: "harness",
|
|
||||||
read_by: "NOT IMPLEMENTED — benchmark harness selection",
|
|
||||||
},
|
|
||||||
KnownKey {
|
|
||||||
key: "tools",
|
|
||||||
read_by: "NOT IMPLEMENTED — per-phase tool selection",
|
|
||||||
},
|
|
||||||
KnownKey {
|
KnownKey {
|
||||||
key: "benchmark",
|
key: "benchmark",
|
||||||
read_by: "NOT IMPLEMENTED — nested benchmark settings",
|
read_by: "NOT IMPLEMENTED — nested benchmark settings",
|
||||||
@@ -93,11 +123,6 @@ pub const DECLARED_BUT_UNREAD: &[KnownKey] = &[
|
|||||||
setting this per phase changes nothing: security_hardening.toml \
|
setting this per phase changes nothing: security_hardening.toml \
|
||||||
asks for gitea_forge + security_scan and its phase gets neither",
|
asks for gitea_forge + security_scan and its phase gets neither",
|
||||||
},
|
},
|
||||||
KnownKey {
|
|
||||||
key: "test_command",
|
|
||||||
read_by: "NOT IMPLEMENTED — mission_delivery::discover_test_command infers \
|
|
||||||
from the repo and does not consult config",
|
|
||||||
},
|
|
||||||
];
|
];
|
||||||
|
|
||||||
fn is_listed(key: &str, list: &[KnownKey]) -> bool {
|
fn is_listed(key: &str, list: &[KnownKey]) -> bool {
|
||||||
@@ -229,18 +254,31 @@ mod tests {
|
|||||||
"order_idx",
|
"order_idx",
|
||||||
"requires_repo",
|
"requires_repo",
|
||||||
"default_team_template",
|
"default_team_template",
|
||||||
|
"default_phase_teams",
|
||||||
"default_topology",
|
"default_topology",
|
||||||
"phases",
|
"phases",
|
||||||
"description",
|
"description",
|
||||||
];
|
];
|
||||||
|
// `[default_phase_teams]` maps a phase PURPOSE to a team template key,
|
||||||
|
// so its keys are not config keys and must not be checked as such.
|
||||||
|
// They are checked against the purposes `phase_runner::purposes_for`
|
||||||
|
// can actually emit instead — a typo'd purpose matches no phase and
|
||||||
|
// that phase silently falls back to the mission-wide team, which is
|
||||||
|
// exactly the kind of quiet wrong staffing this table exists to end.
|
||||||
|
const PURPOSES: &[&str] = &["research", "coding", "security", "mission"];
|
||||||
for entry in entries.flatten() {
|
for entry in entries.flatten() {
|
||||||
let path = entry.path();
|
let path = entry.path();
|
||||||
if path.extension().and_then(|e| e.to_str()) != Some("toml") {
|
if path.extension().and_then(|e| e.to_str()) != Some("toml") {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
let body = std::fs::read_to_string(&path).unwrap();
|
let body = std::fs::read_to_string(&path).unwrap();
|
||||||
|
let mut table = String::new();
|
||||||
for line in body.lines() {
|
for line in body.lines() {
|
||||||
let line = line.trim();
|
let line = line.trim();
|
||||||
|
if line.starts_with('[') {
|
||||||
|
table = line.trim_matches(['[', ']'].as_slice()).to_string();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
if line.starts_with('#') || !line.contains('=') {
|
if line.starts_with('#') || !line.contains('=') {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
@@ -248,6 +286,16 @@ mod tests {
|
|||||||
if key.is_empty() || key.contains(' ') || key.contains('[') {
|
if key.is_empty() || key.contains(' ') || key.contains('[') {
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
if table == "default_phase_teams" {
|
||||||
|
assert!(
|
||||||
|
PURPOSES.contains(&key),
|
||||||
|
"{} staffs purpose `{key}`, which `purposes_for` never emits — \
|
||||||
|
that phase would fall back to the mission-wide team with \
|
||||||
|
nothing reporting it",
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
let accounted = ENVELOPE.contains(&key)
|
let accounted = ENVELOPE.contains(&key)
|
||||||
|| is_listed(key, KNOWN_KEYS)
|
|| is_listed(key, KNOWN_KEYS)
|
||||||
|| is_listed(key, DECLARED_BUT_UNREAD);
|
|| is_listed(key, DECLARED_BUT_UNREAD);
|
||||||
|
|||||||
+2433
-95
File diff suppressed because it is too large
Load Diff
@@ -22,33 +22,59 @@ use sqlx::Row;
|
|||||||
use std::time::Duration;
|
use std::time::Duration;
|
||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
const DEFAULT_MODEL: &str = "claude-opus-4-8";
|
const DEFAULT_MODEL: &str = "claude-opus-5";
|
||||||
const ANTHROPIC_API_VERSION: &str = "2023-06-01";
|
|
||||||
const POLL_INTERVAL: Duration = Duration::from_secs(30);
|
const POLL_INTERVAL: Duration = Duration::from_secs(30);
|
||||||
/// Cap the raw material we send to the model. Missions can produce
|
/// Cap the raw material we send to the model. Missions can produce
|
||||||
/// hundreds of KB of agent output; we slice by turn and by phase
|
/// hundreds of KB of agent output; we slice by turn and by phase
|
||||||
/// artifact but still bound the total prompt.
|
/// artifact but still bound the total prompt.
|
||||||
const MAX_OUTPUT_BYTES: usize = 60_000;
|
const MAX_OUTPUT_BYTES: usize = 120_000;
|
||||||
|
|
||||||
|
/// The longest prefix of `s` that is at most `max_bytes` and ends on a
|
||||||
|
/// character boundary.
|
||||||
|
///
|
||||||
|
/// `&s[..max_bytes]` PANICS when the cut lands inside a multi-byte character,
|
||||||
|
/// and `s` here is agent-authored turn output — arbitrary UTF-8, routinely
|
||||||
|
/// containing arrows, box-drawing and emoji. The panic would take down the
|
||||||
|
/// evaluation sweep for a phase whose only crime was writing a long enough
|
||||||
|
/// line with a non-ASCII character at the wrong offset.
|
||||||
|
///
|
||||||
|
/// Exactly the bug the clawhdf5 agents found and fixed in
|
||||||
|
/// `clawhdf5-migrate/src/validate.rs` this week, in our own code.
|
||||||
|
fn clamp_to_char_boundary(s: &str, max_bytes: usize) -> &str {
|
||||||
|
if s.len() <= max_bytes {
|
||||||
|
return s;
|
||||||
|
}
|
||||||
|
let mut end = max_bytes;
|
||||||
|
while end > 0 && !s.is_char_boundary(end) {
|
||||||
|
end -= 1;
|
||||||
|
}
|
||||||
|
&s[..end]
|
||||||
|
}
|
||||||
|
|
||||||
fn model_name() -> String {
|
fn model_name() -> String {
|
||||||
std::env::var("CLAWMATES_SUMMARIZER_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_string())
|
std::env::var("CLAWMATES_SUMMARIZER_MODEL").unwrap_or_else(|_| DEFAULT_MODEL.to_string())
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn spawn(pool: PgPool) {
|
/// The runtime is carried purely so the summarizer can reach the SAME
|
||||||
|
/// providers as everything else. It used to hand-roll its own HTTPS POST with
|
||||||
|
/// `x-api-key: $ANTHROPIC_API_KEY`, which is why no audit of `.complete(` call
|
||||||
|
/// sites ever found it — and why every phase summary on this deployment died
|
||||||
|
/// with "credit balance is too low" while the phases themselves ran fine.
|
||||||
|
pub fn spawn(pool: PgPool, runtime: cm_runtime::Runtime) {
|
||||||
tokio::spawn(async move {
|
tokio::spawn(async move {
|
||||||
tokio::time::sleep(Duration::from_secs(45)).await;
|
tokio::time::sleep(Duration::from_secs(45)).await;
|
||||||
let mut ticker = tokio::time::interval(POLL_INTERVAL);
|
let mut ticker = tokio::time::interval(POLL_INTERVAL);
|
||||||
ticker.tick().await;
|
ticker.tick().await;
|
||||||
loop {
|
loop {
|
||||||
ticker.tick().await;
|
ticker.tick().await;
|
||||||
if let Err(e) = sweep_once(&pool).await {
|
if let Err(e) = sweep_once(&pool, &runtime).await {
|
||||||
eprintln!("phase_summarizer: sweep failed: {e}");
|
eprintln!("phase_summarizer: sweep failed: {e}");
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn sweep_once(pool: &PgPool) -> Result<(), String> {
|
async fn sweep_once(pool: &PgPool, runtime: &cm_runtime::Runtime) -> Result<(), String> {
|
||||||
// Terminal phases with no summary yet.
|
// Terminal phases with no summary yet.
|
||||||
let rows = sqlx::query(
|
let rows = sqlx::query(
|
||||||
"SELECT mp.id, mp.mission_id, mp.kind
|
"SELECT mp.id, mp.mission_id, mp.kind
|
||||||
@@ -65,7 +91,7 @@ async fn sweep_once(pool: &PgPool) -> Result<(), String> {
|
|||||||
let phase_id: Uuid = row.get("id");
|
let phase_id: Uuid = row.get("id");
|
||||||
let mission_id: Uuid = row.get("mission_id");
|
let mission_id: Uuid = row.get("mission_id");
|
||||||
let kind: String = row.get("kind");
|
let kind: String = row.get("kind");
|
||||||
if let Err(e) = summarize_one(pool, mission_id, phase_id, &kind).await {
|
if let Err(e) = summarize_one(pool, runtime, mission_id, phase_id, &kind).await {
|
||||||
// Persist an error row so we don't infinite-retry a broken
|
// Persist an error row so we don't infinite-retry a broken
|
||||||
// phase — the UI can surface "summary unavailable: <e>".
|
// phase — the UI can surface "summary unavailable: <e>".
|
||||||
eprintln!("phase_summarizer: {phase_id} ({kind}) failed: {e}");
|
eprintln!("phase_summarizer: {phase_id} ({kind}) failed: {e}");
|
||||||
@@ -77,6 +103,7 @@ async fn sweep_once(pool: &PgPool) -> Result<(), String> {
|
|||||||
|
|
||||||
async fn summarize_one(
|
async fn summarize_one(
|
||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
mission_id: Uuid,
|
mission_id: Uuid,
|
||||||
phase_id: Uuid,
|
phase_id: Uuid,
|
||||||
kind: &str,
|
kind: &str,
|
||||||
@@ -90,7 +117,7 @@ async fn summarize_one(
|
|||||||
mission_id,
|
mission_id,
|
||||||
phase_id,
|
phase_id,
|
||||||
kind,
|
kind,
|
||||||
"claude-opus-4-8",
|
"claude-opus-5",
|
||||||
"This phase produced no recorded output. The agents may have failed \
|
"This phase produced no recorded output. The agents may have failed \
|
||||||
to reach their working directory or found nothing to act on.",
|
to reach their working directory or found nothing to act on.",
|
||||||
&json!({
|
&json!({
|
||||||
@@ -105,7 +132,7 @@ async fn summarize_one(
|
|||||||
)
|
)
|
||||||
.await;
|
.await;
|
||||||
}
|
}
|
||||||
let (narrative, structured) = call_anthropic(kind, &material).await?;
|
let (narrative, structured, answered_by) = call_anthropic(runtime, kind, &material).await?;
|
||||||
let metrics = structured
|
let metrics = structured
|
||||||
.get("metrics")
|
.get("metrics")
|
||||||
.cloned()
|
.cloned()
|
||||||
@@ -128,7 +155,7 @@ async fn summarize_one(
|
|||||||
mission_id,
|
mission_id,
|
||||||
phase_id,
|
phase_id,
|
||||||
kind,
|
kind,
|
||||||
&model_name(),
|
&answered_by,
|
||||||
&narrative,
|
&narrative,
|
||||||
&metrics,
|
&metrics,
|
||||||
&sources,
|
&sources,
|
||||||
@@ -249,7 +276,7 @@ async fn collect_material(
|
|||||||
concat.push_str(&format!("\n\n── turn {} ──\n", i + 1));
|
concat.push_str(&format!("\n\n── turn {} ──\n", i + 1));
|
||||||
let remaining = MAX_OUTPUT_BYTES.saturating_sub(concat.len());
|
let remaining = MAX_OUTPUT_BYTES.saturating_sub(concat.len());
|
||||||
if s.len() > remaining {
|
if s.len() > remaining {
|
||||||
concat.push_str(&s[..remaining]);
|
concat.push_str(clamp_to_char_boundary(&s, remaining));
|
||||||
concat.push_str("\n… (truncated)");
|
concat.push_str("\n… (truncated)");
|
||||||
} else {
|
} else {
|
||||||
concat.push_str(&s);
|
concat.push_str(&s);
|
||||||
@@ -323,58 +350,25 @@ async fn collect_material(
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn call_anthropic(kind: &str, material: &PhaseMaterial) -> Result<(String, Value), String> {
|
/// Returns the narrative, the parsed object, and **the model that answered** —
|
||||||
let api_key =
|
/// which may be a fallback link rather than `model_name()`, and is recorded as
|
||||||
std::env::var("ANTHROPIC_API_KEY").map_err(|_| "ANTHROPIC_API_KEY unset".to_string())?;
|
/// such.
|
||||||
|
async fn call_anthropic(
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
|
kind: &str,
|
||||||
|
material: &PhaseMaterial,
|
||||||
|
) -> Result<(String, Value, String), String> {
|
||||||
let model = model_name();
|
let model = model_name();
|
||||||
let system = system_prompt(kind);
|
let system = system_prompt(kind);
|
||||||
let user = user_prompt(kind, material);
|
let user = user_prompt(kind, material);
|
||||||
|
|
||||||
let body = json!({
|
let (raw, answered_by) = crate::subscription::complete_with_fallback(
|
||||||
"model": model,
|
runtime, &system, &user, &model, 4096, false,
|
||||||
"max_tokens": 4096,
|
)
|
||||||
"system": system,
|
.await?;
|
||||||
"messages": [ { "role": "user", "content": user } ]
|
let raw = raw.trim().to_string();
|
||||||
});
|
|
||||||
let client = reqwest::Client::builder()
|
|
||||||
.timeout(std::time::Duration::from_secs(120))
|
|
||||||
.build()
|
|
||||||
.map_err(|e| format!("http client: {e}"))?;
|
|
||||||
let resp = client
|
|
||||||
.post("https://api.anthropic.com/v1/messages")
|
|
||||||
.header("x-api-key", &api_key)
|
|
||||||
.header("anthropic-version", ANTHROPIC_API_VERSION)
|
|
||||||
.header("content-type", "application/json")
|
|
||||||
.json(&body)
|
|
||||||
.send()
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("anthropic call: {e}"))?;
|
|
||||||
if !resp.status().is_success() {
|
|
||||||
let code = resp.status();
|
|
||||||
let body = resp.text().await.unwrap_or_default();
|
|
||||||
return Err(format!(
|
|
||||||
"anthropic {code}: {}",
|
|
||||||
&body[..body.len().min(500)]
|
|
||||||
));
|
|
||||||
}
|
|
||||||
let json: Value = resp
|
|
||||||
.json()
|
|
||||||
.await
|
|
||||||
.map_err(|e| format!("anthropic json: {e}"))?;
|
|
||||||
let raw = json
|
|
||||||
.get("content")
|
|
||||||
.and_then(|c| c.as_array())
|
|
||||||
.and_then(|arr| {
|
|
||||||
arr.iter()
|
|
||||||
.find(|b| b.get("type").and_then(|t| t.as_str()) == Some("text"))
|
|
||||||
})
|
|
||||||
.and_then(|b| b.get("text"))
|
|
||||||
.and_then(|t| t.as_str())
|
|
||||||
.ok_or_else(|| "anthropic response missing text block".to_string())?
|
|
||||||
.trim()
|
|
||||||
.to_string();
|
|
||||||
if raw.is_empty() {
|
if raw.is_empty() {
|
||||||
return Err("anthropic returned empty text".into());
|
return Err(format!("{answered_by} returned empty text"));
|
||||||
}
|
}
|
||||||
// Model returns a JSON object; extract narrative + rest.
|
// Model returns a JSON object; extract narrative + rest.
|
||||||
let parsed: Value = serde_json::from_str(&strip_code_fence(&raw)).map_err(|e| {
|
let parsed: Value = serde_json::from_str(&strip_code_fence(&raw)).map_err(|e| {
|
||||||
@@ -392,7 +386,7 @@ async fn call_anthropic(kind: &str, material: &PhaseMaterial) -> Result<(String,
|
|||||||
if narrative.is_empty() {
|
if narrative.is_empty() {
|
||||||
return Err("summarizer response missing narrative".into());
|
return Err("summarizer response missing narrative".into());
|
||||||
}
|
}
|
||||||
Ok((narrative, parsed))
|
Ok((narrative, parsed, answered_by))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Trim a leading/trailing ```json … ``` fence the model sometimes wraps
|
/// Trim a leading/trailing ```json … ``` fence the model sometimes wraps
|
||||||
@@ -606,3 +600,39 @@ async fn record_error(
|
|||||||
.map_err(|e| format!("record error: {e}"))?;
|
.map_err(|e| format!("record error: {e}"))?;
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// Agent output is arbitrary UTF-8. A byte-offset cut that lands inside a
|
||||||
|
/// multi-byte character must not panic — that panic would take down the
|
||||||
|
/// evaluation sweep for the phase, and the only trigger is an agent
|
||||||
|
/// happening to write a long enough line containing a non-ASCII character.
|
||||||
|
#[test]
|
||||||
|
fn truncation_never_splits_a_multibyte_character() {
|
||||||
|
// 4-byte characters, so every offset not a multiple of 4 is
|
||||||
|
// mid-character and would panic a naive `&s[..cut]`.
|
||||||
|
let s = "😀".repeat(10);
|
||||||
|
for cut in 0..=s.len() {
|
||||||
|
let out = clamp_to_char_boundary(&s, cut);
|
||||||
|
assert!(out.len() <= cut, "must respect the budget at cut={cut}");
|
||||||
|
assert!(s.starts_with(out), "must stay a prefix at cut={cut}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Mixed-width text: the cut must land on a boundary, never inside `é`.
|
||||||
|
#[test]
|
||||||
|
fn truncation_handles_mixed_width_text() {
|
||||||
|
let s = "héllo wörld";
|
||||||
|
for cut in 0..=s.len() {
|
||||||
|
let out = clamp_to_char_boundary(s, cut);
|
||||||
|
assert!(s.starts_with(out));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn truncation_returns_everything_when_it_fits() {
|
||||||
|
assert_eq!(clamp_to_char_boundary("héllo", 100), "héllo");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,886 @@
|
|||||||
|
//! Turning a mission's script into an episode.
|
||||||
|
//!
|
||||||
|
//! ## Why not GenFM
|
||||||
|
//!
|
||||||
|
//! The plan was ElevenLabs GenFM (`POST /v1/studio/podcasts`), which writes AND
|
||||||
|
//! voices a two-host show from source text. It is unreachable on this account:
|
||||||
|
//!
|
||||||
|
//! ```text
|
||||||
|
//! GET /v1/studio/projects -> 403
|
||||||
|
//! POST /v1/studio/podcasts -> 403
|
||||||
|
//! "Access to the Studio API requires your account to be explicitly
|
||||||
|
//! whitelisted to use it. Please contact our sales team."
|
||||||
|
//! ```
|
||||||
|
//!
|
||||||
|
//! Measured with two different keys, so it is an ACCOUNT restriction and not a
|
||||||
|
//! key scope. Plain text-to-speech on the same key returns a valid MP3.
|
||||||
|
//!
|
||||||
|
//! That turns out to suit the operator's choice better than GenFM would have.
|
||||||
|
//! GenFM always runs its own LLM over the source, so our agents' script would
|
||||||
|
//! have been *rewritten*; rendering each line ourselves speaks it verbatim. The
|
||||||
|
//! agents did the reading and the judging, and the podcast says what they wrote.
|
||||||
|
//!
|
||||||
|
//! ## Why the backend is a trait
|
||||||
|
//!
|
||||||
|
//! NotebookLM documents no programmatic audio retrieval at all, GenFM needs a
|
||||||
|
//! sales conversation, and Gemini TTS is a third shape again. The renderer
|
||||||
|
//! should not have to care: it hands a `Script` to an `AudioBackend` and gets
|
||||||
|
//! bytes.
|
||||||
|
|
||||||
|
use async_trait::async_trait;
|
||||||
|
|
||||||
|
/// One spoken turn.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct Turn {
|
||||||
|
/// `HOST` or `GUEST`, as written in the script.
|
||||||
|
pub speaker: String,
|
||||||
|
pub text: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A parsed episode script.
|
||||||
|
#[derive(Debug, Clone, Default)]
|
||||||
|
pub struct Script {
|
||||||
|
pub title: String,
|
||||||
|
pub turns: Vec<Turn>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Script {
|
||||||
|
/// Roughly how long this will take to say, at 150 words per minute.
|
||||||
|
pub fn estimated_secs(&self) -> u32 {
|
||||||
|
let words: usize = self.turns.iter().map(|t| t.text.split_whitespace().count()).sum();
|
||||||
|
((words as f32 / 150.0) * 60.0).round() as u32
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse `script.md` into turns.
|
||||||
|
///
|
||||||
|
/// The format is what `skills/research/podcast-dialogue-writing.md` tells the
|
||||||
|
/// writer to produce: `HOST:` / `GUEST:` at the start of a line. Everything
|
||||||
|
/// else — headings, blank lines, stage directions in brackets — is not speech
|
||||||
|
/// and must not be read aloud, which is the whole reason this is a parser and
|
||||||
|
/// not a `read_to_string`.
|
||||||
|
///
|
||||||
|
/// A continuation line (no speaker prefix) belongs to the turn above it, so a
|
||||||
|
/// wrapped paragraph stays one turn rather than becoming a new one.
|
||||||
|
pub fn parse_script(md: &str) -> Script {
|
||||||
|
let mut title = String::new();
|
||||||
|
let mut turns: Vec<Turn> = Vec::new();
|
||||||
|
|
||||||
|
for raw in md.lines() {
|
||||||
|
let line = raw.trim();
|
||||||
|
if line.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if let Some(h) = line.strip_prefix("# ") {
|
||||||
|
if title.is_empty() {
|
||||||
|
title = h.trim().to_string();
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
// Any other heading, list marker or rule is structure, not speech.
|
||||||
|
if line.starts_with('#') || line.starts_with("---") || line.starts_with("> ") {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
match line.split_once(':') {
|
||||||
|
Some((who, said))
|
||||||
|
if !who.is_empty()
|
||||||
|
&& who.len() <= 12
|
||||||
|
&& who
|
||||||
|
.chars()
|
||||||
|
.all(|c| c.is_ascii_uppercase() || c.is_ascii_digit() || c == ' ') =>
|
||||||
|
{
|
||||||
|
let text = said.trim();
|
||||||
|
if !text.is_empty() {
|
||||||
|
turns.push(Turn {
|
||||||
|
speaker: who.trim().to_string(),
|
||||||
|
text: text.to_string(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// Continuation of the previous turn.
|
||||||
|
_ => {
|
||||||
|
if let Some(last) = turns.last_mut() {
|
||||||
|
last.text.push(' ');
|
||||||
|
last.text.push_str(line);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Script { title, turns }
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
/// MPEG1 Layer III bitrates (kbps) and sample rates, indexed as the frame
|
||||||
|
/// header encodes them.
|
||||||
|
const MP3_BITRATES: [u32; 16] = [
|
||||||
|
0, 32, 40, 48, 56, 64, 80, 96, 112, 128, 160, 192, 224, 256, 320, 0,
|
||||||
|
];
|
||||||
|
const MP3_RATES: [u32; 4] = [44100, 48000, 32000, 0];
|
||||||
|
|
||||||
|
/// Strip a clip's container metadata so clips can be joined into ONE stream.
|
||||||
|
///
|
||||||
|
/// This is the difference between an episode and a six-second file. Each TTS
|
||||||
|
/// clip arrives as a standalone MP3: a small ID3v2 tag, then a first frame
|
||||||
|
/// carrying an `Info`/`Xing` VBR header that declares THAT CLIP's frame count.
|
||||||
|
/// Concatenated raw, a player reads the first clip's header, believes the whole
|
||||||
|
/// file is that long, and stops. Measured on two real clips of 4.86s and 4.68s:
|
||||||
|
///
|
||||||
|
/// ```text
|
||||||
|
/// raw concat -> 4.86s (only clip one plays)
|
||||||
|
/// strip second clip's ID3 -> 4.86s
|
||||||
|
/// strip both clips' ID3 -> 4.86s (the tag was never the issue)
|
||||||
|
/// strip ID3 *and* the Info frame -> 9.53s correct
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// The ID3 tag is ~45 bytes and harmless; the header FRAME is what lies. Both
|
||||||
|
/// go, leaving pure audio frames that a player times from the stream itself.
|
||||||
|
fn strip_container(clip: &[u8]) -> &[u8] {
|
||||||
|
let mut i = 0usize;
|
||||||
|
// ID3v2: 10-byte header, then a syncsafe 28-bit size.
|
||||||
|
if clip.len() > 10 && &clip[..3] == b"ID3" {
|
||||||
|
let size = ((clip[6] as usize) << 21)
|
||||||
|
| ((clip[7] as usize) << 14)
|
||||||
|
| ((clip[8] as usize) << 7)
|
||||||
|
| (clip[9] as usize);
|
||||||
|
i = (10 + size).min(clip.len());
|
||||||
|
}
|
||||||
|
// A leading Xing/Info frame is metadata, not sound.
|
||||||
|
if i + 4 < clip.len() && clip[i] == 0xFF && clip[i + 1] & 0xE0 == 0xE0 {
|
||||||
|
let br = MP3_BITRATES[((clip[i + 2] >> 4) & 0x0F) as usize];
|
||||||
|
let sr = MP3_RATES[((clip[i + 2] >> 2) & 0x03) as usize];
|
||||||
|
if br > 0 && sr > 0 {
|
||||||
|
let pad = ((clip[i + 2] >> 1) & 1) as usize;
|
||||||
|
let len = (144 * br as usize * 1000 / sr as usize) + pad;
|
||||||
|
let end = (i + len).min(clip.len());
|
||||||
|
let frame = &clip[i..end];
|
||||||
|
if find(frame, b"Xing").is_some() || find(frame, b"Info").is_some() {
|
||||||
|
i = end;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
&clip[i..]
|
||||||
|
}
|
||||||
|
|
||||||
|
fn find(hay: &[u8], needle: &[u8]) -> Option<usize> {
|
||||||
|
hay.windows(needle.len()).position(|w| w == needle)
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
/// Rewrite a line so it is worth HEARING.
|
||||||
|
///
|
||||||
|
/// Written from a real episode the operator listened to. Two things ruined it,
|
||||||
|
/// and neither is a TTS defect — the text genuinely said them:
|
||||||
|
///
|
||||||
|
/// ```text
|
||||||
|
/// "This is the ReFind paper, arxiv 2608.12888."
|
||||||
|
/// -> "two six zero eight point one two eight eight eight"
|
||||||
|
/// "BM25 recall dropped from 0.506 native to 0.004 cross-lingual"
|
||||||
|
/// -> "zero point five zero six ... zero point zero zero four"
|
||||||
|
/// ```
|
||||||
|
///
|
||||||
|
/// A listener on a treadmill cannot write an identifier down and does not need
|
||||||
|
/// three decimal places. `skills/research/podcast-dialogue-writing.md` already
|
||||||
|
/// told the writer not to include arXiv ids and it included them anyway — which
|
||||||
|
/// is the lesson of this whole project restated: an instruction is a request,
|
||||||
|
/// and a listener deserves a guarantee. So the prose asks and this enforces.
|
||||||
|
///
|
||||||
|
/// Deliberately narrow. It removes identifiers and shortens over-precise
|
||||||
|
/// decimals; it does not paraphrase, reorder or summarise. The agents' words
|
||||||
|
/// are still the episode.
|
||||||
|
pub fn speakable(line: &str) -> String {
|
||||||
|
let mut out = String::with_capacity(line.len());
|
||||||
|
let b: Vec<char> = line.chars().collect();
|
||||||
|
let mut i = 0usize;
|
||||||
|
|
||||||
|
while i < b.len() {
|
||||||
|
// "arXiv:2608.12888", "arxiv 2608.12888", "arXiv 2608.12888v2"
|
||||||
|
if starts_with_ci(&b, i, "arxiv") {
|
||||||
|
let mut j = i + 5;
|
||||||
|
while j < b.len() && (b[j] == ':' || b[j] == ' ' || b[j] == '.') {
|
||||||
|
j += 1;
|
||||||
|
}
|
||||||
|
let digits_start = j;
|
||||||
|
while j < b.len() && (b[j].is_ascii_digit() || b[j] == '.' || b[j] == 'v') {
|
||||||
|
j += 1;
|
||||||
|
}
|
||||||
|
// Do not swallow the sentence's full stop. "…retrieval, arxiv
|
||||||
|
// 2608.00183. This one's a catch." must not become one run-on
|
||||||
|
// sentence — the pause is how a listener knows a thought ended.
|
||||||
|
while j > digits_start && !b[j - 1].is_ascii_digit() {
|
||||||
|
j -= 1;
|
||||||
|
}
|
||||||
|
if j > digits_start + 4 {
|
||||||
|
// Drop the whole reference, and any comma or space it left
|
||||||
|
// dangling: "the ReFind paper, arxiv 2608.12888." must not
|
||||||
|
// become "the ReFind paper, ."
|
||||||
|
trim_trailing_separator(&mut out);
|
||||||
|
i = j;
|
||||||
|
skip_leading_separator(&b, &mut i);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// A bare arXiv-shaped number: 4 digits, dot, 4-5 digits.
|
||||||
|
if b[i].is_ascii_digit() {
|
||||||
|
let start = i;
|
||||||
|
let mut j = i;
|
||||||
|
while j < b.len() && b[j].is_ascii_digit() {
|
||||||
|
j += 1;
|
||||||
|
}
|
||||||
|
let int_len = j - start;
|
||||||
|
if j < b.len() && b[j] == '.' {
|
||||||
|
let frac_start = j + 1;
|
||||||
|
let mut k = frac_start;
|
||||||
|
while k < b.len() && b[k].is_ascii_digit() {
|
||||||
|
k += 1;
|
||||||
|
}
|
||||||
|
let frac_len = k - frac_start;
|
||||||
|
if int_len == 4 && (4..=5).contains(&frac_len) {
|
||||||
|
// An identifier, not a quantity.
|
||||||
|
trim_trailing_separator(&mut out);
|
||||||
|
i = k;
|
||||||
|
skip_leading_separator(&b, &mut i);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if frac_len >= 3 {
|
||||||
|
// Over-precise. Nobody hears the third decimal place.
|
||||||
|
let text: String = b[start..k].iter().collect();
|
||||||
|
out.push_str(&round_decimal(&text));
|
||||||
|
i = k;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out.push(b[i]);
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
// Collapse any double spaces a removal left behind.
|
||||||
|
let collapsed = out.split_whitespace().collect::<Vec<_>>().join(" ");
|
||||||
|
collapsed
|
||||||
|
.replace(" ,", ",")
|
||||||
|
.replace(" .", ".")
|
||||||
|
.replace("( )", "")
|
||||||
|
.replace("()", "")
|
||||||
|
}
|
||||||
|
|
||||||
|
fn starts_with_ci(b: &[char], i: usize, word: &str) -> bool {
|
||||||
|
let w: Vec<char> = word.chars().collect();
|
||||||
|
if i + w.len() > b.len() {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
b[i..i + w.len()]
|
||||||
|
.iter()
|
||||||
|
.zip(&w)
|
||||||
|
.all(|(a, c)| a.to_ascii_lowercase() == *c)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn trim_trailing_separator(out: &mut String) {
|
||||||
|
while out.ends_with(' ') || out.ends_with(',') || out.ends_with('(') {
|
||||||
|
out.pop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn skip_leading_separator(b: &[char], i: &mut usize) {
|
||||||
|
while *i < b.len() && (b[*i] == ')' || b[*i] == ',') {
|
||||||
|
*i += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Two decimal places, or "under 0.01" when rounding would say "0.00".
|
||||||
|
///
|
||||||
|
/// `0.004` rounded to two places is `0.00`, which is worse than the original:
|
||||||
|
/// it says the value is zero when the point was that it collapsed to nearly
|
||||||
|
/// nothing.
|
||||||
|
fn round_decimal(text: &str) -> String {
|
||||||
|
let Ok(v) = text.parse::<f64>() else {
|
||||||
|
return text.to_string();
|
||||||
|
};
|
||||||
|
let r = (v * 100.0).round() / 100.0;
|
||||||
|
if r == 0.0 && v != 0.0 {
|
||||||
|
return "under 0.01".to_string();
|
||||||
|
}
|
||||||
|
let s = format!("{r:.2}");
|
||||||
|
s.trim_end_matches('0').trim_end_matches('.').to_string()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Anything that can turn a script into audio bytes.
|
||||||
|
#[async_trait]
|
||||||
|
pub trait AudioBackend: Send + Sync {
|
||||||
|
/// Render the whole script. Returns MP3 bytes.
|
||||||
|
async fn render(&self, script: &Script) -> Result<Vec<u8>, String>;
|
||||||
|
/// For logs and the episode record.
|
||||||
|
fn describe(&self) -> String;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// ElevenLabs per-line text-to-speech.
|
||||||
|
pub struct ElevenLabs {
|
||||||
|
api_key: String,
|
||||||
|
/// Voice for the first speaker seen, and for anyone unrecognised.
|
||||||
|
pub host_voice: String,
|
||||||
|
/// Voice for the second distinct speaker.
|
||||||
|
pub guest_voice: String,
|
||||||
|
pub model_id: String,
|
||||||
|
http: reqwest::Client,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Default voices, both from the stock library so no account setup is needed.
|
||||||
|
pub const DEFAULT_HOST_VOICE: &str = "CwhRBWXzGAHq8TQ4Fs17"; // Roger
|
||||||
|
pub const DEFAULT_GUEST_VOICE: &str = "EXAVITQu4vr4xnSDxMaL"; // Sarah
|
||||||
|
|
||||||
|
impl ElevenLabs {
|
||||||
|
/// Build from the environment. `None` when no key is configured, so a
|
||||||
|
/// deployment without one simply produces no audio instead of failing a
|
||||||
|
/// mission that otherwise succeeded.
|
||||||
|
pub fn from_env() -> Option<ElevenLabs> {
|
||||||
|
let api_key = std::env::var("ELEVENLABS_API_KEY")
|
||||||
|
.ok()
|
||||||
|
.filter(|k| !k.trim().is_empty())?;
|
||||||
|
Some(ElevenLabs {
|
||||||
|
api_key,
|
||||||
|
host_voice: std::env::var("CLAWMATES_PODCAST_HOST_VOICE")
|
||||||
|
.unwrap_or_else(|_| DEFAULT_HOST_VOICE.to_string()),
|
||||||
|
guest_voice: std::env::var("CLAWMATES_PODCAST_GUEST_VOICE")
|
||||||
|
.unwrap_or_else(|_| DEFAULT_GUEST_VOICE.to_string()),
|
||||||
|
// flash_v2_5 is the cheap fast tier; a spoken digest does not need
|
||||||
|
// the expensive model, and cost matters on a DAILY job.
|
||||||
|
model_id: std::env::var("CLAWMATES_PODCAST_MODEL")
|
||||||
|
.unwrap_or_else(|_| "eleven_flash_v2_5".to_string()),
|
||||||
|
http: reqwest::Client::new(),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Which voice speaks this turn.
|
||||||
|
///
|
||||||
|
/// Keyed off the speaker labels actually present rather than hardcoding
|
||||||
|
/// "HOST"/"GUEST", so a script that uses names still alternates instead of
|
||||||
|
/// collapsing into one voice.
|
||||||
|
fn voice_for(&self, speaker: &str, first: &str, second: Option<&str>) -> &str {
|
||||||
|
if speaker.eq_ignore_ascii_case(first) {
|
||||||
|
&self.host_voice
|
||||||
|
} else if second.is_some_and(|s| speaker.eq_ignore_ascii_case(s)) {
|
||||||
|
&self.guest_voice
|
||||||
|
} else {
|
||||||
|
&self.host_voice
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn say(&self, text: &str, voice: &str) -> Result<Vec<u8>, String> {
|
||||||
|
let url = format!("https://api.elevenlabs.io/v1/text-to-speech/{voice}");
|
||||||
|
let res = self
|
||||||
|
.http
|
||||||
|
.post(&url)
|
||||||
|
.header("xi-api-key", &self.api_key)
|
||||||
|
.json(&serde_json::json!({
|
||||||
|
"text": text,
|
||||||
|
"model_id": self.model_id,
|
||||||
|
// 128kbps 44.1k: podcast-normal, and small enough that a daily
|
||||||
|
// episode does not bloat the blob store.
|
||||||
|
"output_format": "mp3_44100_128",
|
||||||
|
}))
|
||||||
|
.send()
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("tts request: {e}"))?;
|
||||||
|
if !res.status().is_success() {
|
||||||
|
let code = res.status();
|
||||||
|
let body = res.text().await.unwrap_or_default();
|
||||||
|
return Err(format!("tts {code}: {}", body.chars().take(200).collect::<String>()));
|
||||||
|
}
|
||||||
|
let bytes = res.bytes().await.map_err(|e| format!("tts body: {e}"))?;
|
||||||
|
if bytes.len() < 512 {
|
||||||
|
return Err(format!("tts returned {} bytes — too short to be audio", bytes.len()));
|
||||||
|
}
|
||||||
|
Ok(bytes.to_vec())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[async_trait]
|
||||||
|
impl AudioBackend for ElevenLabs {
|
||||||
|
fn describe(&self) -> String {
|
||||||
|
format!("elevenlabs/{}", self.model_id)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn render(&self, script: &Script) -> Result<Vec<u8>, String> {
|
||||||
|
if script.turns.is_empty() {
|
||||||
|
return Err("script has no spoken turns".into());
|
||||||
|
}
|
||||||
|
// Identify the two speakers by order of appearance.
|
||||||
|
let first = script.turns[0].speaker.clone();
|
||||||
|
let second = script
|
||||||
|
.turns
|
||||||
|
.iter()
|
||||||
|
.map(|t| t.speaker.as_str())
|
||||||
|
.find(|s| !s.eq_ignore_ascii_case(&first))
|
||||||
|
.map(str::to_string);
|
||||||
|
|
||||||
|
let mut out: Vec<u8> = Vec::new();
|
||||||
|
for (i, turn) in script.turns.iter().enumerate() {
|
||||||
|
let voice = self.voice_for(&turn.speaker, &first, second.as_deref());
|
||||||
|
let clip = self.say(&speakable(&turn.text), voice).await.map_err(|e| {
|
||||||
|
// Name the turn: a 400 on one line is far easier to fix than
|
||||||
|
// "rendering failed" for a 40-turn script.
|
||||||
|
format!("turn {} ({}): {e}", i + 1, turn.speaker)
|
||||||
|
})?;
|
||||||
|
// Join as ONE stream: see `strip_container`. Concatenating whole
|
||||||
|
// MP3 files yields a file that plays only its first clip.
|
||||||
|
out.extend_from_slice(strip_container(&clip));
|
||||||
|
}
|
||||||
|
Ok(out)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_script_parses_into_speaker_turns() {
|
||||||
|
let md = "# Morning Research Podcast — 2026-08-18\n\n\
|
||||||
|
HOST: Morning run, morning papers.\n\n\
|
||||||
|
GUEST: Today's harvest pokes at something settled.\n";
|
||||||
|
let s = parse_script(md);
|
||||||
|
assert_eq!(s.title, "Morning Research Podcast — 2026-08-18");
|
||||||
|
assert_eq!(s.turns.len(), 2);
|
||||||
|
assert_eq!(s.turns[0].speaker, "HOST");
|
||||||
|
assert_eq!(s.turns[1].text, "Today's harvest pokes at something settled.");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Headings and rules are structure. Reading "2026-08-18" and "---" aloud
|
||||||
|
/// is the difference between an episode and a machine reading a file.
|
||||||
|
#[test]
|
||||||
|
fn structure_is_never_spoken() {
|
||||||
|
let s = parse_script("# Title\n## Section\n---\n> quote\nHOST: Only this.\n");
|
||||||
|
assert_eq!(s.turns.len(), 1);
|
||||||
|
assert_eq!(s.turns[0].text, "Only this.");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A wrapped paragraph is ONE turn. Splitting on every newline would break
|
||||||
|
/// a sentence across two TTS calls and audibly stutter at the seam.
|
||||||
|
#[test]
|
||||||
|
fn continuation_lines_join_the_turn_above() {
|
||||||
|
let s = parse_script("HOST: First part\nsecond part.\nGUEST: Mine.\n");
|
||||||
|
assert_eq!(s.turns.len(), 2);
|
||||||
|
assert_eq!(s.turns[0].text, "First part second part.");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A colon inside speech must not be read as a speaker label, or the line
|
||||||
|
/// is silently truncated to whatever followed the colon.
|
||||||
|
#[test]
|
||||||
|
fn a_colon_mid_sentence_does_not_start_a_new_turn() {
|
||||||
|
let s = parse_script("HOST: The finding: recall dropped sharply.\n");
|
||||||
|
assert_eq!(s.turns.len(), 1);
|
||||||
|
assert_eq!(s.turns[0].text, "The finding: recall dropped sharply.");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Voices are assigned by order of appearance, so a script using names
|
||||||
|
/// instead of HOST/GUEST still alternates.
|
||||||
|
#[test]
|
||||||
|
fn two_speakers_get_two_voices_whatever_they_are_called() {
|
||||||
|
let el = ElevenLabs {
|
||||||
|
api_key: "x".into(),
|
||||||
|
host_voice: "HOSTV".into(),
|
||||||
|
guest_voice: "GUESTV".into(),
|
||||||
|
model_id: "m".into(),
|
||||||
|
http: reqwest::Client::new(),
|
||||||
|
};
|
||||||
|
assert_eq!(el.voice_for("ANA", "ANA", Some("BEN")), "HOSTV");
|
||||||
|
assert_eq!(el.voice_for("BEN", "ANA", Some("BEN")), "GUESTV");
|
||||||
|
// An unexpected third speaker falls back rather than failing the run.
|
||||||
|
assert_eq!(el.voice_for("CARL", "ANA", Some("BEN")), "HOSTV");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The bug that produced a six-second "episode".
|
||||||
|
///
|
||||||
|
/// Each clip is a standalone MP3 whose first frame carries an Info/Xing
|
||||||
|
/// header declaring that clip's length. Joined raw, a player reads clip
|
||||||
|
/// one's header and stops there. Fixtures are REAL ElevenLabs clips, so
|
||||||
|
/// this pins the actual wire format rather than a hand-built approximation.
|
||||||
|
#[test]
|
||||||
|
fn joining_strips_the_header_that_declares_one_clips_length() {
|
||||||
|
let clip = std::fs::read(concat!(env!("CARGO_MANIFEST_DIR"), "/tests/fixtures/tts-clip.mp3"));
|
||||||
|
let Ok(clip) = clip else {
|
||||||
|
eprintln!("fixture absent; skipping");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
assert_eq!(&clip[..3], b"ID3", "fixture should be a raw TTS clip");
|
||||||
|
let body = strip_container(&clip);
|
||||||
|
assert!(body.len() < clip.len(), "something must be stripped");
|
||||||
|
assert_eq!(body[0], 0xFF, "must start on a frame sync, got {:#04x}", body[0]);
|
||||||
|
assert!(body[1] & 0xE0 == 0xE0, "frame sync incomplete");
|
||||||
|
// The lying header must be gone from the head of the stream.
|
||||||
|
let head = &body[..body.len().min(1024)];
|
||||||
|
assert!(
|
||||||
|
find(head, b"Info").is_none() && find(head, b"Xing").is_none(),
|
||||||
|
"the VBR header frame survived — the join will report one clip's length"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Not an MP3, or a truncated one, must pass through rather than panic:
|
||||||
|
/// a bad clip should fail the render with a message, not crash the server.
|
||||||
|
#[test]
|
||||||
|
fn stripping_is_safe_on_junk() {
|
||||||
|
for junk in [&b""[..], &b"ID3"[..], &[0xFFu8][..], &b"not audio at all"[..]] {
|
||||||
|
let out = strip_container(junk);
|
||||||
|
assert!(out.len() <= junk.len());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Real lines from the episode the operator listened to. These are the
|
||||||
|
/// exact strings the TTS read aloud as digit soup.
|
||||||
|
/// Print what the operator's own episode WOULD have said. Not an
|
||||||
|
/// assertion — a way to read the diff on real input.
|
||||||
|
#[test]
|
||||||
|
fn show_real_script_lines() {
|
||||||
|
let Ok(md) = std::env::var("CLAWMATES_SPEAKABLE_DEMO") else { return };
|
||||||
|
let Ok(text) = std::fs::read_to_string(&md) else { return };
|
||||||
|
for line in text.lines() {
|
||||||
|
let out = speakable(line);
|
||||||
|
if out != line && !line.trim().is_empty() {
|
||||||
|
eprintln!(" BEFORE {}", line.trim());
|
||||||
|
eprintln!(" AFTER {}\n", out.trim());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn identifiers_are_never_spoken() {
|
||||||
|
for (input, must_not) in [
|
||||||
|
("This is the ReFind paper, arxiv 2608.12888.", "2608"),
|
||||||
|
("There was an agricultural paper too, arXiv:2608.14886 — does it matter?", "14886"),
|
||||||
|
("evidence-unit fairness in financial retrieval, arxiv 2608.00183", "00183"),
|
||||||
|
] {
|
||||||
|
let out = speakable(input);
|
||||||
|
assert!(!out.contains(must_not), "{must_not} survived in {out:?}");
|
||||||
|
assert!(!out.to_lowercase().contains("arxiv"), "dangling label: {out:?}");
|
||||||
|
// The sentence must still read cleanly.
|
||||||
|
assert!(!out.contains(" ,"), "orphan comma: {out:?}");
|
||||||
|
assert!(!out.contains(",."), "orphan comma: {out:?}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Three decimal places is data, not speech.
|
||||||
|
#[test]
|
||||||
|
fn over_precise_decimals_are_shortened() {
|
||||||
|
let out = speakable("BM25 recall dropped from 0.506 native to 0.004 cross-lingual.");
|
||||||
|
assert!(out.contains("0.51"), "0.506 should round: {out:?}");
|
||||||
|
assert!(!out.contains("0.506"), "{out:?}");
|
||||||
|
// 0.004 rounds to 0.00, which would claim the value was zero — the
|
||||||
|
// opposite of the point being made.
|
||||||
|
assert!(out.contains("under 0.01"), "{out:?}");
|
||||||
|
assert!(!out.contains("0.00 "), "must never say zero: {out:?}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Two decimals, years, percentages and small integers are all fine spoken
|
||||||
|
/// and must survive untouched — over-processing would mangle the meaning.
|
||||||
|
#[test]
|
||||||
|
fn ordinary_numbers_are_left_alone() {
|
||||||
|
for s in [
|
||||||
|
"58.2 versus 53.2 mean accuracy",
|
||||||
|
"roughly 2,800 questions",
|
||||||
|
"21.8% of theoretical headroom",
|
||||||
|
"NDCG at 10 of 0.15",
|
||||||
|
"about 2026 papers",
|
||||||
|
] {
|
||||||
|
assert_eq!(speakable(s), s, "should be unchanged");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_runtime_estimate_is_in_the_right_ballpark() {
|
||||||
|
let words = "word ".repeat(1500);
|
||||||
|
let s = parse_script(&format!("HOST: {words}\n"));
|
||||||
|
let secs = s.estimated_secs();
|
||||||
|
assert!((540..=660).contains(&secs), "1500 words ≈ 10 min, got {secs}s");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Live render against the real API. Ignored by default: it spends credits.
|
||||||
|
///
|
||||||
|
/// Exercises the production path — `parse_script` then `ElevenLabs::render` —
|
||||||
|
/// rather than a reimplementation, so what passes here is what runs.
|
||||||
|
///
|
||||||
|
/// CLAWMATES_PODCAST_TEST_SCRIPT=/path/to/script.md \
|
||||||
|
/// CLAWMATES_PODCAST_TEST_OUT=/tmp/episode.mp3 \
|
||||||
|
/// cargo test -p cm-api --lib podcast::live -- --ignored --nocapture
|
||||||
|
#[cfg(test)]
|
||||||
|
mod live {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[tokio::test]
|
||||||
|
#[ignore = "spends ElevenLabs credits"]
|
||||||
|
async fn renders_a_real_script_to_mp3() {
|
||||||
|
let path = std::env::var("CLAWMATES_PODCAST_TEST_SCRIPT")
|
||||||
|
.expect("set CLAWMATES_PODCAST_TEST_SCRIPT");
|
||||||
|
let out = std::env::var("CLAWMATES_PODCAST_TEST_OUT")
|
||||||
|
.unwrap_or_else(|_| "/tmp/episode.mp3".to_string());
|
||||||
|
let md = std::fs::read_to_string(&path).expect("script readable");
|
||||||
|
let script = parse_script(&md);
|
||||||
|
assert!(!script.turns.is_empty(), "no turns parsed from {path}");
|
||||||
|
eprintln!(
|
||||||
|
"script: {:?} — {} turns, ~{}s",
|
||||||
|
script.title,
|
||||||
|
script.turns.len(),
|
||||||
|
script.estimated_secs()
|
||||||
|
);
|
||||||
|
|
||||||
|
let backend = ElevenLabs::from_env().expect("ELEVENLABS_API_KEY must be set");
|
||||||
|
let bytes = backend.render(&script).await.expect("render");
|
||||||
|
std::fs::write(&out, &bytes).expect("write mp3");
|
||||||
|
eprintln!("wrote {} bytes to {out} via {}", bytes.len(), backend.describe());
|
||||||
|
|
||||||
|
// ID3 or a raw MPEG frame header — anything else is not audio.
|
||||||
|
let head = &bytes[..3.min(bytes.len())];
|
||||||
|
assert!(
|
||||||
|
head == b"ID3" || (bytes[0] == 0xFF && bytes[1] & 0xE0 == 0xE0),
|
||||||
|
"not an MP3: {head:?}"
|
||||||
|
);
|
||||||
|
assert!(bytes.len() > 100_000, "suspiciously small: {} bytes", bytes.len());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── Rendering a finished mission into an episode ──────────────────────
|
||||||
|
|
||||||
|
/// Where a mission's script lives inside its checkout.
|
||||||
|
pub fn script_path(date: &str) -> String {
|
||||||
|
format!("ContinuousResearch/{date}/script.md")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Blob key for an episode's audio.
|
||||||
|
pub fn blob_key(mission_id: uuid::Uuid, date: &str) -> String {
|
||||||
|
format!("podcast/{date}/{mission_id}.mp3")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Duration of a joined CBR stream, from its frame headers.
|
||||||
|
///
|
||||||
|
/// Read from the audio rather than estimated from the script, because the
|
||||||
|
/// estimate is what a listener is NOT owed: the feed advertises a length and
|
||||||
|
/// that length should be the real one. Also the check that caught a six-second
|
||||||
|
/// "episode" — a file can be 6 MB and still play for seconds.
|
||||||
|
pub fn duration_secs(mp3: &[u8]) -> u32 {
|
||||||
|
let mut i = 0usize;
|
||||||
|
let mut seconds = 0f64;
|
||||||
|
while i + 4 <= mp3.len() {
|
||||||
|
if mp3[i] == 0xFF && mp3[i + 1] & 0xE0 == 0xE0 {
|
||||||
|
let br = MP3_BITRATES[((mp3[i + 2] >> 4) & 0x0F) as usize];
|
||||||
|
let sr = MP3_RATES[((mp3[i + 2] >> 2) & 0x03) as usize];
|
||||||
|
if br > 0 && sr > 0 {
|
||||||
|
let pad = ((mp3[i + 2] >> 1) & 1) as usize;
|
||||||
|
let len = (144 * br as usize * 1000 / sr as usize) + pad;
|
||||||
|
seconds += 1152.0 / sr as f64;
|
||||||
|
i += len.max(4);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
i += 1;
|
||||||
|
}
|
||||||
|
seconds.round() as u32
|
||||||
|
}
|
||||||
|
|
||||||
|
fn sha_hex(bytes: &[u8]) -> String {
|
||||||
|
use sha2::{Digest, Sha256};
|
||||||
|
format!("{:x}", Sha256::digest(bytes))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Render every finished Continuous Research mission that has a script and no
|
||||||
|
/// episode yet.
|
||||||
|
///
|
||||||
|
/// Driven from a sweep rather than the phase itself: rendering is not the
|
||||||
|
/// agents' work and must not be able to fail a phase that succeeded. It is also
|
||||||
|
/// retryable by construction — a run that fails on a transient API error is
|
||||||
|
/// simply picked up next tick, and one that succeeded is skipped because the
|
||||||
|
/// episode row exists.
|
||||||
|
pub async fn render_pending(
|
||||||
|
pool: &sqlx::PgPool,
|
||||||
|
blobs: &std::sync::Arc<dyn cm_files::BlobStore>,
|
||||||
|
backend: &dyn AudioBackend,
|
||||||
|
) -> Result<usize, String> {
|
||||||
|
use sqlx::Row;
|
||||||
|
let rows = sqlx::query(
|
||||||
|
"SELECT m.id, m.workspace_id, m.title
|
||||||
|
FROM missions m
|
||||||
|
WHERE m.template_kind = $1
|
||||||
|
AND m.status IN ('completed', 'failed')
|
||||||
|
AND NOT EXISTS (SELECT 1 FROM podcast_episodes e WHERE e.mission_id = m.id)
|
||||||
|
ORDER BY m.completed_at DESC NULLS LAST
|
||||||
|
LIMIT 3",
|
||||||
|
)
|
||||||
|
.bind(crate::continuous_research::TEMPLATE_KIND)
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("select missions to render: {e}"))?;
|
||||||
|
|
||||||
|
let mut made = 0usize;
|
||||||
|
for row in rows {
|
||||||
|
let mission_id: uuid::Uuid = row.get("id");
|
||||||
|
let workspace_id: uuid::Uuid = row.get("workspace_id");
|
||||||
|
let mission_title: String = row.get("title");
|
||||||
|
|
||||||
|
// A `failed` mission is included on purpose: the script phase may have
|
||||||
|
// written a perfectly good script and failed its judge. The audio is
|
||||||
|
// worth having either way, and the mission record still says it failed.
|
||||||
|
let date = crate::continuous_research::today();
|
||||||
|
let checkout = crate::mission_workspace::checkout_path(mission_id);
|
||||||
|
let mut path = checkout.join(script_path(&date));
|
||||||
|
if !path.is_file() {
|
||||||
|
// The mission may have run yesterday; take the newest script it has
|
||||||
|
// rather than assuming the render happens on the same UTC day.
|
||||||
|
match newest_script(&checkout) {
|
||||||
|
Some(p) => path = p,
|
||||||
|
None => {
|
||||||
|
// NEVER silent. The checkout is deleted 30 minutes after a
|
||||||
|
// mission reaches a terminal state (`mission_runtime`'s
|
||||||
|
// sweeper tears down the container and the tree with it), so
|
||||||
|
// a script that is not here is not late — it is gone, and
|
||||||
|
// this mission will never produce an episode. Saying so is
|
||||||
|
// the difference between a known gap and a feed that is
|
||||||
|
// quietly missing a day.
|
||||||
|
//
|
||||||
|
// The audio is recoverable by hand: the script was pushed to
|
||||||
|
// the phase's own vault branch by `mission_delivery`.
|
||||||
|
record_unrenderable(pool, mission_id, &checkout).await;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let md = match std::fs::read_to_string(&path) {
|
||||||
|
Ok(s) => s,
|
||||||
|
Err(e) => {
|
||||||
|
eprintln!("podcast: cannot read {}: {e}", path.display());
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let script = parse_script(&md);
|
||||||
|
if script.turns.is_empty() {
|
||||||
|
eprintln!("podcast: {} has no spoken turns — skipping", path.display());
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
let audio = match backend.render(&script).await {
|
||||||
|
Ok(a) => a,
|
||||||
|
Err(e) => {
|
||||||
|
// Loud, and NOT fatal to the sweep: one mission's transient API
|
||||||
|
// failure must not stop the others being rendered.
|
||||||
|
eprintln!("podcast: render failed for mission {mission_id}: {e}");
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let secs = duration_secs(&audio);
|
||||||
|
let key = blob_key(mission_id, &date);
|
||||||
|
if let Err(e) = blobs.put(&key, &audio).await {
|
||||||
|
eprintln!("podcast: shelving {key} failed: {e}");
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
let title = if script.title.trim().is_empty() {
|
||||||
|
mission_title
|
||||||
|
} else {
|
||||||
|
script.title.clone()
|
||||||
|
};
|
||||||
|
if let Err(e) = sqlx::query(
|
||||||
|
"INSERT INTO podcast_episodes
|
||||||
|
(id, workspace_id, mission_id, episode_date, title, blob_key,
|
||||||
|
bytes, duration_secs, rendered_by, script_sha)
|
||||||
|
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10)
|
||||||
|
ON CONFLICT (mission_id) DO UPDATE
|
||||||
|
SET title = EXCLUDED.title, blob_key = EXCLUDED.blob_key,
|
||||||
|
bytes = EXCLUDED.bytes, duration_secs = EXCLUDED.duration_secs,
|
||||||
|
rendered_by = EXCLUDED.rendered_by, script_sha = EXCLUDED.script_sha",
|
||||||
|
)
|
||||||
|
.bind(uuid::Uuid::now_v7())
|
||||||
|
.bind(workspace_id)
|
||||||
|
.bind(mission_id)
|
||||||
|
.bind(&date)
|
||||||
|
.bind(&title)
|
||||||
|
.bind(&key)
|
||||||
|
.bind(audio.len() as i64)
|
||||||
|
.bind(secs as i32)
|
||||||
|
.bind(backend.describe())
|
||||||
|
.bind(sha_hex(md.as_bytes()))
|
||||||
|
.execute(pool)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!("podcast: recording episode for {mission_id} failed: {e}");
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
eprintln!(
|
||||||
|
"podcast: episode for mission {mission_id} — {} turns, {}s, {} bytes at {key}",
|
||||||
|
script.turns.len(),
|
||||||
|
secs,
|
||||||
|
audio.len()
|
||||||
|
);
|
||||||
|
made += 1;
|
||||||
|
}
|
||||||
|
Ok(made)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Say — once — that a mission can never be rendered.
|
||||||
|
///
|
||||||
|
/// Once, not every tick: the sweep revisits the same missions forever, and a
|
||||||
|
/// line per mission per five minutes would bury everything else in the log. The
|
||||||
|
/// episode row is the marker, with a zero-length blob key that the feed skips.
|
||||||
|
async fn record_unrenderable(pool: &sqlx::PgPool, mission_id: uuid::Uuid, checkout: &std::path::Path) {
|
||||||
|
eprintln!(
|
||||||
|
"podcast: mission {mission_id} has no script at {} — the checkout was reaped before the \
|
||||||
|
render sweep reached it, so this day has no episode. The script is still on the phase's \
|
||||||
|
vault branch if it is wanted.",
|
||||||
|
checkout.display()
|
||||||
|
);
|
||||||
|
let _ = sqlx::query(
|
||||||
|
"INSERT INTO podcast_episodes
|
||||||
|
(id, workspace_id, mission_id, episode_date, title, blob_key, bytes,
|
||||||
|
duration_secs, rendered_by, script_sha)
|
||||||
|
SELECT $1, m.workspace_id, m.id, '', m.title, '', 0, 0, 'unrenderable', ''
|
||||||
|
FROM missions m WHERE m.id = $2
|
||||||
|
ON CONFLICT (mission_id) DO NOTHING",
|
||||||
|
)
|
||||||
|
.bind(uuid::Uuid::now_v7())
|
||||||
|
.bind(mission_id)
|
||||||
|
.execute(pool)
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The most recent `ContinuousResearch/<date>/script.md` in a checkout.
|
||||||
|
fn newest_script(checkout: &std::path::Path) -> Option<std::path::PathBuf> {
|
||||||
|
let root = checkout.join("ContinuousResearch");
|
||||||
|
let mut dates: Vec<String> = std::fs::read_dir(root)
|
||||||
|
.ok()?
|
||||||
|
.filter_map(Result::ok)
|
||||||
|
.filter(|e| e.path().is_dir())
|
||||||
|
.map(|e| e.file_name().to_string_lossy().to_string())
|
||||||
|
.collect();
|
||||||
|
// ISO dates sort lexicographically, which is the whole reason for the format.
|
||||||
|
dates.sort();
|
||||||
|
dates.iter().rev().find_map(|d| {
|
||||||
|
let p = checkout.join(script_path(d));
|
||||||
|
p.is_file().then_some(p)
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spawn the render sweep.
|
||||||
|
pub fn spawn(
|
||||||
|
pool: sqlx::PgPool,
|
||||||
|
blobs: Option<std::sync::Arc<dyn cm_files::BlobStore>>,
|
||||||
|
interval: std::time::Duration,
|
||||||
|
) {
|
||||||
|
let Some(blobs) = blobs else {
|
||||||
|
eprintln!("podcast: no blob storage configured — episodes will not be rendered");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let Some(backend) = ElevenLabs::from_env() else {
|
||||||
|
// Not an error. A deployment without a key simply produces no audio,
|
||||||
|
// and every other part of the mission still works.
|
||||||
|
eprintln!("podcast: ELEVENLABS_API_KEY not set — episodes will not be rendered");
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
tokio::spawn(async move {
|
||||||
|
let mut tick = tokio::time::interval(interval);
|
||||||
|
tick.tick().await;
|
||||||
|
loop {
|
||||||
|
tick.tick().await;
|
||||||
|
match render_pending(&pool, &blobs, &backend).await {
|
||||||
|
Ok(n) if n > 0 => eprintln!("podcast: rendered {n} episode(s)"),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => eprintln!("podcast: sweep failed: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -173,6 +173,7 @@ impl TurnExecutor for SubTopologyExecutor {
|
|||||||
output: record.final_output,
|
output: record.final_output,
|
||||||
tokens: record.totals.tokens,
|
tokens: record.totals.tokens,
|
||||||
gated,
|
gated,
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,266 @@
|
|||||||
|
//! What a repository actually contains, small enough to put in a prompt.
|
||||||
|
//!
|
||||||
|
//! The planner was given the root listing and planned "optimise the hot path"
|
||||||
|
//! for a crate whose hot path is `add(a: i64, b: i64) -> i64`. Names were not
|
||||||
|
//! enough: the mission was unachievable from the moment it was written, and
|
||||||
|
//! nothing discovered that until an agent had built a benchmark harness to
|
||||||
|
//! measure an integer addition.
|
||||||
|
//!
|
||||||
|
//! # The rule this module exists to enforce
|
||||||
|
//!
|
||||||
|
//! A digest is always partial for any repository worth planning against, and a
|
||||||
|
//! model shown a partial view without being told it is partial plans as though
|
||||||
|
//! it saw everything. So every omission is STATED — how many files were listed,
|
||||||
|
//! how many were shown, what was cut from each. That is the same distinction as
|
||||||
|
//! `Option<u32>` for the subagent probe: "we did not look" and "there is nothing
|
||||||
|
//! there" are different facts, and only one of them is about the repository.
|
||||||
|
//!
|
||||||
|
//! # Priority
|
||||||
|
//!
|
||||||
|
//! Manifests first (they say what the project IS and what it may depend on),
|
||||||
|
//! then the README, then source ascending by size — smallest-first shows the
|
||||||
|
//! most files per byte, and a planner benefits more from seeing twenty small
|
||||||
|
//! files than one large one.
|
||||||
|
|
||||||
|
/// One file in the repository tree.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct FileEntry {
|
||||||
|
pub path: String,
|
||||||
|
pub size: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Total characters of file CONTENT a digest may carry.
|
||||||
|
///
|
||||||
|
/// The prompt around it is ~1.5k, and the planner is a single call per mission,
|
||||||
|
/// so this is generous by design: the cost of a too-small digest is a plan built
|
||||||
|
/// on a guess, which costs a VM boot to discover.
|
||||||
|
pub const CONTENT_BUDGET: usize = 12_000;
|
||||||
|
|
||||||
|
/// Ceiling per file, so one large file cannot spend the whole budget.
|
||||||
|
pub const PER_FILE_CAP: usize = 3_000;
|
||||||
|
|
||||||
|
/// Files worth showing before any source.
|
||||||
|
fn is_manifest(path: &str) -> bool {
|
||||||
|
matches!(
|
||||||
|
path,
|
||||||
|
"Cargo.toml"
|
||||||
|
| "package.json"
|
||||||
|
| "pyproject.toml"
|
||||||
|
| "setup.py"
|
||||||
|
| "go.mod"
|
||||||
|
| "Gemfile"
|
||||||
|
| "pom.xml"
|
||||||
|
| "build.gradle"
|
||||||
|
| "Makefile"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn is_readme(path: &str) -> bool {
|
||||||
|
path.eq_ignore_ascii_case("README.md") || path.eq_ignore_ascii_case("README")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Paths to fetch, in the order they earn their place.
|
||||||
|
///
|
||||||
|
/// Directories and files a planner cannot use are dropped: lockfiles are huge
|
||||||
|
/// and say nothing a manifest does not, and build output is not source.
|
||||||
|
pub fn priority(entries: &[FileEntry]) -> Vec<&FileEntry> {
|
||||||
|
let mut useful: Vec<&FileEntry> = entries
|
||||||
|
.iter()
|
||||||
|
.filter(|e| {
|
||||||
|
let p = e.path.as_str();
|
||||||
|
!p.starts_with(".git/")
|
||||||
|
&& !p.contains("/target/")
|
||||||
|
&& !p.starts_with("target/")
|
||||||
|
&& !p.contains("node_modules/")
|
||||||
|
&& p != "Cargo.lock"
|
||||||
|
&& p != "package-lock.json"
|
||||||
|
&& p != "poetry.lock"
|
||||||
|
&& e.size > 0
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
useful.sort_by_key(|e| {
|
||||||
|
let rank = if is_manifest(&e.path) {
|
||||||
|
0
|
||||||
|
} else if is_readme(&e.path) {
|
||||||
|
1
|
||||||
|
} else {
|
||||||
|
2
|
||||||
|
};
|
||||||
|
(rank, e.size, e.path.clone())
|
||||||
|
});
|
||||||
|
useful
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Render the digest a planner sees.
|
||||||
|
///
|
||||||
|
/// `contents` is `(path, text)` for the files that were actually fetched, in
|
||||||
|
/// priority order. Anything not fetched is still LISTED, so the model knows the
|
||||||
|
/// file exists even when it cannot read it.
|
||||||
|
pub fn render(entries: &[FileEntry], contents: &[(String, String)]) -> String {
|
||||||
|
if entries.is_empty() {
|
||||||
|
return "(the repository is empty, or its tree could not be read)".to_string();
|
||||||
|
}
|
||||||
|
let mut out = String::new();
|
||||||
|
out.push_str(&format!("FILES ({} total):\n", entries.len()));
|
||||||
|
// The whole tree by name is cheap and is what stops "does X exist" guessing.
|
||||||
|
// Capped anyway: a 10k-file monorepo listing is not a prompt.
|
||||||
|
const MAX_LISTED: usize = 300;
|
||||||
|
for e in entries.iter().take(MAX_LISTED) {
|
||||||
|
out.push_str(&format!(" {} ({} bytes)\n", e.path, e.size));
|
||||||
|
}
|
||||||
|
if entries.len() > MAX_LISTED {
|
||||||
|
out.push_str(&format!(
|
||||||
|
" … and {} more files NOT listed\n",
|
||||||
|
entries.len() - MAX_LISTED
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
if contents.is_empty() {
|
||||||
|
out.push_str("\n(no file contents could be read — plan from the names alone, and say so if that is not enough)\n");
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
out.push_str(&format!(
|
||||||
|
"\nCONTENTS ({} of {} files shown; anything not shown you have NOT seen):\n",
|
||||||
|
contents.len(),
|
||||||
|
entries.len()
|
||||||
|
));
|
||||||
|
for (path, text) in contents {
|
||||||
|
out.push_str(&format!("\n--- {path} ---\n{text}\n"));
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Take file texts up to the budget, truncating each at [`PER_FILE_CAP`].
|
||||||
|
///
|
||||||
|
/// Truncation is marked in the text itself rather than silently cutting: a model
|
||||||
|
/// that can see it is reading a fragment asks differently than one that believes
|
||||||
|
/// it read the file.
|
||||||
|
pub fn fit(fetched: Vec<(String, String)>) -> Vec<(String, String)> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
let mut spent = 0usize;
|
||||||
|
for (path, text) in fetched {
|
||||||
|
if spent >= CONTENT_BUDGET {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
let room = (CONTENT_BUDGET - spent).min(PER_FILE_CAP);
|
||||||
|
let text = if text.len() <= room {
|
||||||
|
text
|
||||||
|
} else {
|
||||||
|
let end = (0..=room)
|
||||||
|
.rev()
|
||||||
|
.find(|i| text.is_char_boundary(*i))
|
||||||
|
.unwrap_or(0);
|
||||||
|
format!(
|
||||||
|
"{}\n… [truncated: {} of {} bytes shown]",
|
||||||
|
&text[..end],
|
||||||
|
end,
|
||||||
|
text.len()
|
||||||
|
)
|
||||||
|
};
|
||||||
|
spent += text.len();
|
||||||
|
out.push((path, text));
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn f(path: &str, size: usize) -> FileEntry {
|
||||||
|
FileEntry {
|
||||||
|
path: path.into(),
|
||||||
|
size,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Manifests first, then the README, then source smallest-first. A planner
|
||||||
|
/// learns more from twenty small files than from one large one.
|
||||||
|
#[test]
|
||||||
|
fn the_files_that_say_what_this_is_come_first() {
|
||||||
|
let entries = vec![
|
||||||
|
f("src/big.rs", 9000),
|
||||||
|
f("README.md", 400),
|
||||||
|
f("src/lib.rs", 120),
|
||||||
|
f("Cargo.toml", 200),
|
||||||
|
];
|
||||||
|
let order: Vec<&str> = priority(&entries).iter().map(|e| e.path.as_str()).collect();
|
||||||
|
assert_eq!(order, vec!["Cargo.toml", "README.md", "src/lib.rs", "src/big.rs"]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Lockfiles and build output are dropped: enormous, and they say nothing a
|
||||||
|
/// manifest does not.
|
||||||
|
#[test]
|
||||||
|
fn noise_is_not_offered_to_the_planner() {
|
||||||
|
let entries = vec![
|
||||||
|
f("Cargo.lock", 50_000),
|
||||||
|
f("target/debug/thing", 900_000),
|
||||||
|
f("node_modules/x/index.js", 400),
|
||||||
|
f(".git/config", 100),
|
||||||
|
f("src/lib.rs", 100),
|
||||||
|
f("empty.rs", 0),
|
||||||
|
];
|
||||||
|
let kept: Vec<&str> = priority(&entries).iter().map(|e| e.path.as_str()).collect();
|
||||||
|
assert_eq!(kept, vec!["src/lib.rs"]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// THE rule. A partial view presented as complete is planned against as
|
||||||
|
/// though it were complete — which is how "optimise the hot path" gets
|
||||||
|
/// written for a crate that adds two integers.
|
||||||
|
#[test]
|
||||||
|
fn every_omission_is_stated() {
|
||||||
|
let entries: Vec<FileEntry> = (0..400).map(|i| f(&format!("src/f{i}.rs"), 100)).collect();
|
||||||
|
let shown = vec![("src/f0.rs".to_string(), "fn a() {}".to_string())];
|
||||||
|
let out = render(&entries, &shown);
|
||||||
|
|
||||||
|
assert!(out.contains("FILES (400 total)"), "{out}");
|
||||||
|
assert!(out.contains("and 100 more files NOT listed"), "{out}");
|
||||||
|
assert!(out.contains("1 of 400 files shown"), "{out}");
|
||||||
|
assert!(
|
||||||
|
out.contains("you have NOT seen"),
|
||||||
|
"the model must be told the view is partial: {out}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A file cut short says so, in the text the model reads.
|
||||||
|
#[test]
|
||||||
|
fn a_truncated_file_says_it_was_truncated() {
|
||||||
|
let big = "x".repeat(PER_FILE_CAP * 2);
|
||||||
|
let out = fit(vec![("src/big.rs".into(), big.clone())]);
|
||||||
|
assert_eq!(out.len(), 1);
|
||||||
|
assert!(out[0].1.contains("truncated"), "{}", &out[0].1[..80]);
|
||||||
|
assert!(out[0].1.len() < big.len());
|
||||||
|
// And the marker names both numbers, so "how much did I miss" is
|
||||||
|
// answerable rather than guessable.
|
||||||
|
assert!(out[0].1.contains(&big.len().to_string()));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The budget is a total, not per file: one large file must not starve the
|
||||||
|
/// rest, and the whole digest must stay promptable.
|
||||||
|
#[test]
|
||||||
|
fn the_budget_bounds_the_whole_digest() {
|
||||||
|
let files: Vec<(String, String)> = (0..20)
|
||||||
|
.map(|i| (format!("src/f{i}.rs"), "y".repeat(PER_FILE_CAP)))
|
||||||
|
.collect();
|
||||||
|
let out = fit(files);
|
||||||
|
let total: usize = out.iter().map(|(_, t)| t.len()).sum();
|
||||||
|
assert!(total <= CONTENT_BUDGET, "digest was {total} bytes");
|
||||||
|
assert!(!out.is_empty(), "and it still shows something");
|
||||||
|
assert!(out.len() < 20, "not everything fits, by construction");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An empty or unreadable tree is stated as such — never rendered as a
|
||||||
|
/// repository that happens to contain nothing.
|
||||||
|
#[test]
|
||||||
|
fn an_unreadable_tree_is_not_an_empty_repository() {
|
||||||
|
let out = render(&[], &[]);
|
||||||
|
assert!(out.contains("could not be read"), "{out}");
|
||||||
|
|
||||||
|
// A tree we CAN read but no contents we could fetch is a different
|
||||||
|
// fact, and says so.
|
||||||
|
let out = render(&[f("src/lib.rs", 100)], &[]);
|
||||||
|
assert!(out.contains("src/lib.rs"), "{out}");
|
||||||
|
assert!(out.contains("no file contents could be read"), "{out}");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,159 @@
|
|||||||
|
//! A throwaway copy of a mission checkout, for commands that run as ROOT.
|
||||||
|
//!
|
||||||
|
//! Three places in this codebase run a real command against a mission's tree —
|
||||||
|
//! the judge's verification (`evaluator_tools::Sandbox`), the benchmark runner,
|
||||||
|
//! and the `on_green_tests` delivery gate. All three enter a container running as
|
||||||
|
//! root with the missions root bind-mounted, and all three run something that
|
||||||
|
//! writes `target/`. All three now go through here; the judge was the last to
|
||||||
|
//! move, having carried its own copy of this logic since before it existed.
|
||||||
|
//!
|
||||||
|
//! Run against the live checkout, that breaks the single-writer invariant: the
|
||||||
|
//! tree is owned by uid 65532 and now contains root-owned build output, so the
|
||||||
|
//! next phase's `cargo` hits permission-denied on a directory it cannot write.
|
||||||
|
//! The harness's uid probe reports it as `uids=0,65532`.
|
||||||
|
//!
|
||||||
|
//! # The cleanup half, which is the part that keeps being got wrong
|
||||||
|
//!
|
||||||
|
//! The copy inherits the same problem: its `target/` is root-owned, so the
|
||||||
|
//! server process (uid 65532) **cannot delete it**. A `Drop` calling
|
||||||
|
//! `std::fs::remove_dir_all` fails, and because that error is discarded the tree
|
||||||
|
//! survives forever — measured at 1.2 MB per benchmark run and 16 MB of stranded
|
||||||
|
//! judge sandboxes before this existed.
|
||||||
|
//!
|
||||||
|
//! So removal goes back through the container, as root, where the files were
|
||||||
|
//! written. `Drop` remains only as a fallback for the paths where nothing has
|
||||||
|
//! run as root yet, and does not pretend to be more.
|
||||||
|
|
||||||
|
use std::path::{Path, PathBuf};
|
||||||
|
|
||||||
|
/// Where throwaway copies live: siblings of the per-mission directories, like
|
||||||
|
/// `_outputs` and `_verify`, so reaping a mission cannot race a running command.
|
||||||
|
pub fn copy_root(kind: &str, mission_id: uuid::Uuid) -> PathBuf {
|
||||||
|
crate::mission_workspace::missions_root()
|
||||||
|
.join(kind)
|
||||||
|
.join(mission_id.to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Delete a copy from inside the container that wrote it.
|
||||||
|
///
|
||||||
|
/// Best-effort and loud: a housekeeping failure must not cost a real verdict or
|
||||||
|
/// a real benchmark, but it must not be silent either — silence is how the leaks
|
||||||
|
/// this module exists for went unnoticed for a day.
|
||||||
|
pub async fn purge(container: &str, root: &Path) {
|
||||||
|
let Ok(docker) = crate::container_exec::connect() else {
|
||||||
|
return;
|
||||||
|
};
|
||||||
|
let argv = vec![
|
||||||
|
"rm".to_string(),
|
||||||
|
"-rf".to_string(),
|
||||||
|
root.display().to_string(),
|
||||||
|
];
|
||||||
|
// Explicitly root: this exists to delete files an EARLIER root-run exec
|
||||||
|
// created, which uid 65532 cannot touch. Everything else now runs as 65532
|
||||||
|
// (see `container_exec`), so this is cleaning up history, not policy.
|
||||||
|
if let Err(e) = crate::container_exec::exec_as_root(
|
||||||
|
&docker,
|
||||||
|
container,
|
||||||
|
Some("/"),
|
||||||
|
&argv,
|
||||||
|
std::time::Duration::from_secs(120),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!(
|
||||||
|
"root_copy: could not remove {} from {container}: {e}",
|
||||||
|
root.display()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A copy of a checkout, removed when it goes out of scope.
|
||||||
|
pub struct RootCopy {
|
||||||
|
root: PathBuf,
|
||||||
|
workdir: PathBuf,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl RootCopy {
|
||||||
|
/// Copy `source` into `root`, returning a handle whose `workdir` is the tree
|
||||||
|
/// to run in.
|
||||||
|
///
|
||||||
|
/// Packed through `mission_fs::pack_dir`, so the copy carries exactly what a
|
||||||
|
/// delivered diff carries — no `target/`, no `node_modules/`. One exclusion
|
||||||
|
/// list, four consumers.
|
||||||
|
pub fn of(source: &Path, root: &Path) -> Result<RootCopy, String> {
|
||||||
|
let archive = crate::mission_fs::pack_dir(source, "repo")
|
||||||
|
.map_err(|e| format!("pack {} for a root-run command: {e}", source.display()))?;
|
||||||
|
crate::mission_fs::unpack_into(&archive, root)
|
||||||
|
.map_err(|e| format!("unpack copy into {}: {e}", root.display()))?;
|
||||||
|
let workdir = root.join("repo");
|
||||||
|
if !workdir.is_dir() {
|
||||||
|
return Err(format!("copy missing at {}", workdir.display()));
|
||||||
|
}
|
||||||
|
Ok(RootCopy {
|
||||||
|
root: root.to_path_buf(),
|
||||||
|
workdir,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn workdir(&self) -> &Path {
|
||||||
|
&self.workdir
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Take the working directory and give up automatic cleanup.
|
||||||
|
///
|
||||||
|
/// For a caller whose copy outlives this handle — `evaluator_tools::Sandbox`
|
||||||
|
/// hands the path to a judge that has not run yet, so letting `Drop` fire on
|
||||||
|
/// return would delete the tree out from under it. That caller becomes
|
||||||
|
/// responsible for calling [`purge`], which is the only thing that can
|
||||||
|
/// remove root-owned build output anyway.
|
||||||
|
///
|
||||||
|
/// Spelled as a method rather than `mem::forget` at the call site, so the
|
||||||
|
/// transfer of responsibility is visible in the type rather than implied by
|
||||||
|
/// a leak.
|
||||||
|
pub fn into_workdir(self) -> PathBuf {
|
||||||
|
let workdir = self.workdir.clone();
|
||||||
|
std::mem::forget(self);
|
||||||
|
workdir
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for RootCopy {
|
||||||
|
/// Fallback only. This CANNOT remove root-owned build output — see
|
||||||
|
/// [`purge`], which is what actually clears a copy something has run in.
|
||||||
|
fn drop(&mut self) {
|
||||||
|
let _ = std::fs::remove_dir_all(&self.root);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// A copy must be a SIBLING of the per-mission directory, never inside it:
|
||||||
|
/// `teardown_container` removes `<missions_root>/<mission_id>` wholesale and
|
||||||
|
/// would take a running command's tree with it.
|
||||||
|
#[test]
|
||||||
|
fn copies_live_beside_the_mission_directory_not_inside_it() {
|
||||||
|
let mission = uuid::Uuid::now_v7();
|
||||||
|
let mission_dir = crate::mission_workspace::missions_root().join(mission.to_string());
|
||||||
|
for kind in ["_bench", "_gate", "_verify"] {
|
||||||
|
let root = copy_root(kind, mission);
|
||||||
|
assert!(!root.starts_with(&mission_dir), "{root:?}");
|
||||||
|
assert!(
|
||||||
|
root.starts_with(crate::mission_workspace::missions_root().join(kind)),
|
||||||
|
"{root:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The copy is not the checkout. Stated as a test because the whole defect
|
||||||
|
/// class is "ran the real command against the real tree".
|
||||||
|
#[test]
|
||||||
|
fn a_copy_is_never_the_checkout() {
|
||||||
|
let mission = uuid::Uuid::now_v7();
|
||||||
|
let live = crate::mission_workspace::checkout_path(mission);
|
||||||
|
for kind in ["_bench", "_gate", "_verify"] {
|
||||||
|
assert_ne!(copy_root(kind, mission).join("repo"), live);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -26,6 +26,19 @@ pub(crate) async fn workspace_agent(
|
|||||||
Ok(agent)
|
Ok(agent)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// As [`workspace_agent`], but sees soft-deleted agents too. PURGE ONLY.
|
||||||
|
pub(crate) async fn workspace_agent_any(
|
||||||
|
state: &AppState,
|
||||||
|
user: &cm_auth::AuthedUser,
|
||||||
|
agent_id: AgentId,
|
||||||
|
) -> Result<Agent, ApiError> {
|
||||||
|
let agent = cm_db::repo::agents::get_any(&state.pool, agent_id).await?;
|
||||||
|
if agent.workspace_id != user.workspace_id {
|
||||||
|
return Err(ApiError::NotFound);
|
||||||
|
}
|
||||||
|
Ok(agent)
|
||||||
|
}
|
||||||
|
|
||||||
/// `GET /api/claws/{id}/runtime-config` — the claw's model + §15 sandbox facts
|
/// `GET /api/claws/{id}/runtime-config` — the claw's model + §15 sandbox facts
|
||||||
/// (for the claw card / anatomy view's model badge).
|
/// (for the claw card / anatomy view's model badge).
|
||||||
#[derive(Serialize)]
|
#[derive(Serialize)]
|
||||||
@@ -577,7 +590,11 @@ pub async fn enhance_brain(
|
|||||||
let user_prompt = format!(
|
let user_prompt = format!(
|
||||||
"BRAIN: {reference}\n\n=== SYSTEM PROMPT ===\n{sp}\n\n=== AGENTS.md ===\n{agent_md}\n\n=== PERSONA ===\n{persona}\n\n=== SKILLS ===\n{skills}"
|
"BRAIN: {reference}\n\n=== SYSTEM PROMPT ===\n{sp}\n\n=== AGENTS.md ===\n{agent_md}\n\n=== PERSONA ===\n{persona}\n\n=== SKILLS ===\n{skills}"
|
||||||
);
|
);
|
||||||
let raw = match runtime.complete(ENHANCE_SYSTEM, &user_prompt, "claude-opus-4-8", 16000, true).await {
|
let raw = match crate::subscription::complete_or(
|
||||||
|
&runtime, ENHANCE_SYSTEM, &user_prompt, "claude-opus-4-8", 16000, true,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
Ok(t) => t,
|
Ok(t) => t,
|
||||||
Err(e) => { yield sse(json!({"stage":"error","pct":100,"label":format!("Opus error: {e}")})); return; }
|
Err(e) => { yield sse(json!({"stage":"error","pct":100,"label":format!("Opus error: {e}")})); return; }
|
||||||
};
|
};
|
||||||
@@ -656,8 +673,14 @@ pub(crate) async fn enhance_and_publish(
|
|||||||
let user_prompt = format!(
|
let user_prompt = format!(
|
||||||
"ROLE CONTEXT: {role_context}\n\nBRAIN: {reference}\n\n=== SYSTEM PROMPT ===\n{sp}\n\n=== AGENTS.md ===\n{agent_md}\n\n=== PERSONA ===\n{persona}\n\n=== SKILLS ===\n{skills}"
|
"ROLE CONTEXT: {role_context}\n\nBRAIN: {reference}\n\n=== SYSTEM PROMPT ===\n{sp}\n\n=== AGENTS.md ===\n{agent_md}\n\n=== PERSONA ===\n{persona}\n\n=== SKILLS ===\n{skills}"
|
||||||
);
|
);
|
||||||
let raw = runtime
|
let raw = crate::subscription::complete_or(
|
||||||
.complete(ENHANCE_SYSTEM, &user_prompt, "claude-opus-4-8", 16000, true)
|
runtime,
|
||||||
|
ENHANCE_SYSTEM,
|
||||||
|
&user_prompt,
|
||||||
|
"claude-opus-4-8",
|
||||||
|
16000,
|
||||||
|
true,
|
||||||
|
)
|
||||||
.await?;
|
.await?;
|
||||||
let v = extract_json(&raw).ok_or_else(|| "unparseable enhance output".to_string())?;
|
let v = extract_json(&raw).ok_or_else(|| "unparseable enhance output".to_string())?;
|
||||||
let enh = v.get("enhanced").cloned().unwrap_or(Value::Null);
|
let enh = v.get("enhanced").cloned().unwrap_or(Value::Null);
|
||||||
@@ -1127,7 +1150,7 @@ pub async fn patch(
|
|||||||
|
|
||||||
#[derive(Deserialize)]
|
#[derive(Deserialize)]
|
||||||
pub struct SetModelRequest {
|
pub struct SetModelRequest {
|
||||||
/// Model selector (claude / glm / glm-5.2 / kimi / gemini / groq /
|
/// Model selector (claude / glm / glm-5.2 / kimi / groq /
|
||||||
/// specific model id like `claude-sonnet-5`). Resolved through the
|
/// specific model id like `claude-sonnet-5`). Resolved through the
|
||||||
/// same RuntimeProvisioner::provider_alias_for that team creation
|
/// same RuntimeProvisioner::provider_alias_for that team creation
|
||||||
/// uses, so shorthand + fully-qualified ids both work.
|
/// uses, so shorthand + fully-qualified ids both work.
|
||||||
@@ -1279,7 +1302,12 @@ pub async fn batch_delete(
|
|||||||
let mut done = 0usize;
|
let mut done = 0usize;
|
||||||
for id in agent_ids {
|
for id in agent_ids {
|
||||||
let base = 100 * done / total;
|
let base = 100 * done / total;
|
||||||
let agent = match workspace_agent(&state, &user, id).await {
|
// `workspace_agent_any`, not `workspace_agent`: a purge has to be
|
||||||
|
// able to see the rows it exists to remove. The soft-delete path
|
||||||
|
// correctly hides them from every read, which also hid them from
|
||||||
|
// the only route that could reap them — four soft-deleted agents
|
||||||
|
// from June were unreachable from the application entirely.
|
||||||
|
let agent = match workspace_agent_any(&state, &user, id).await {
|
||||||
Ok(a) => a,
|
Ok(a) => a,
|
||||||
Err(_) => { yield sse(json!({"stage":"skip","pct":base,"label":format!("{id}: not found or no access")})); done += 1; continue; }
|
Err(_) => { yield sse(json!({"stage":"skip","pct":base,"label":format!("{id}: not found or no access")})); done += 1; continue; }
|
||||||
};
|
};
|
||||||
@@ -1379,3 +1407,58 @@ pub async fn settings_full(
|
|||||||
"managed_by_name": manager.display_name,
|
"managed_by_name": manager.display_name,
|
||||||
})))
|
})))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `GET /api/claws/lifecycle` — the agent census.
|
||||||
|
///
|
||||||
|
/// Answers "who is working, who is finished, and who is bound to nothing" in
|
||||||
|
/// one place, which previously required reading the database by hand.
|
||||||
|
pub async fn lifecycle_census(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
) -> Result<axum::Json<serde_json::Value>, ApiError> {
|
||||||
|
let rows = crate::agent_lifecycle::census(&state.pool, user.workspace_id.as_uuid())
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("claws::lifecycle_census: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
let mut counts = std::collections::BTreeMap::<&str, usize>::new();
|
||||||
|
for c in &rows {
|
||||||
|
*counts.entry(c.state.as_str()).or_default() += 1;
|
||||||
|
}
|
||||||
|
Ok(axum::Json(serde_json::json!({
|
||||||
|
"counts": counts,
|
||||||
|
"agents": rows.iter().map(|c| serde_json::json!({
|
||||||
|
"id": c.id,
|
||||||
|
"name": c.name,
|
||||||
|
"state": c.state.as_str(),
|
||||||
|
"reapable": c.state.reapable(),
|
||||||
|
"finished_hours_ago": c.finished_hours_ago,
|
||||||
|
})).collect::<Vec<_>>(),
|
||||||
|
})))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `POST /api/claws/lifecycle/sweep` — run the reap now.
|
||||||
|
///
|
||||||
|
/// The sweeper is hourly; this exists so an operator does not have to wait an
|
||||||
|
/// hour to see the effect of a decision they already made.
|
||||||
|
pub async fn lifecycle_sweep(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(_user): Authed,
|
||||||
|
) -> Result<axum::Json<serde_json::Value>, ApiError> {
|
||||||
|
let swept = crate::agent_lifecycle::sweep(
|
||||||
|
&state.pool,
|
||||||
|
&state.runtime,
|
||||||
|
crate::agent_lifecycle::COMPLETED_GRACE_HOURS,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("claws::lifecycle_sweep: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
Ok(axum::Json(serde_json::json!({
|
||||||
|
"reaped": swept.reaped,
|
||||||
|
"failed": swept.failed,
|
||||||
|
"kept_in_grace": swept.kept_in_grace,
|
||||||
|
})))
|
||||||
|
}
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ pub async fn propose_for_agent(
|
|||||||
Authed(user): Authed,
|
Authed(user): Authed,
|
||||||
Path(agent_id): Path<Uuid>,
|
Path(agent_id): Path<Uuid>,
|
||||||
) -> Result<Json<serde_json::Value>, ApiError> {
|
) -> Result<Json<serde_json::Value>, ApiError> {
|
||||||
let id = crate::level_up::propose_agent(&state.pool, user.workspace_id, user.user_id, agent_id)
|
let id = crate::level_up::propose_agent(&state.pool, &state.runtime, user.workspace_id, user.user_id, agent_id)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| {
|
.map_err(|e| {
|
||||||
eprintln!("level_up: propose_agent {agent_id} failed: {e}");
|
eprintln!("level_up: propose_agent {agent_id} failed: {e}");
|
||||||
@@ -51,7 +51,7 @@ pub async fn propose_for_team(
|
|||||||
Authed(user): Authed,
|
Authed(user): Authed,
|
||||||
Path(team_id): Path<Uuid>,
|
Path(team_id): Path<Uuid>,
|
||||||
) -> Result<Json<serde_json::Value>, ApiError> {
|
) -> Result<Json<serde_json::Value>, ApiError> {
|
||||||
let id = crate::level_up::propose_team(&state.pool, user.workspace_id, user.user_id, team_id)
|
let id = crate::level_up::propose_team(&state.pool, &state.runtime, user.workspace_id, user.user_id, team_id)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| {
|
.map_err(|e| {
|
||||||
eprintln!("level_up: propose_team {team_id} failed: {e}");
|
eprintln!("level_up: propose_team {team_id} failed: {e}");
|
||||||
|
|||||||
@@ -14,8 +14,8 @@ use crate::{ApiError, AppState, Authed};
|
|||||||
/// Default corpus + repo. Single-operator deployment, so these are constants
|
/// Default corpus + repo. Single-operator deployment, so these are constants
|
||||||
/// rather than another table to keep in sync; a second library becomes a
|
/// rather than another table to keep in sync; a second library becomes a
|
||||||
/// request field the day one exists.
|
/// request field the day one exists.
|
||||||
const DEFAULT_CORPUS: &str = "valhalla-vault";
|
pub const DEFAULT_CORPUS: &str = "valhalla-vault";
|
||||||
const DEFAULT_VAULT_URL: &str = "https://git.redclaw.dev/redclaw/valhalla-vault.git";
|
pub const DEFAULT_VAULT_URL: &str = "https://git.redclaw.dev/redclaw/valhalla-vault.git";
|
||||||
|
|
||||||
#[derive(Deserialize)]
|
#[derive(Deserialize)]
|
||||||
pub struct RunRequest {
|
pub struct RunRequest {
|
||||||
|
|||||||
@@ -0,0 +1,378 @@
|
|||||||
|
//! `/api/missions/{id}/plan-proposals` — let a model author the phases.
|
||||||
|
//!
|
||||||
|
//! W1 / #13, and the sibling of [`crate::routes::mission_roster`]: that one has
|
||||||
|
//! a model size the team, this one has it decide what the work is. Same three
|
||||||
|
//! verbs and the same rule — propose and decide are separate, because only the
|
||||||
|
//! second one changes a mission.
|
||||||
|
//!
|
||||||
|
//! The model is handed two lists it may not depart from: the phase kinds
|
||||||
|
//! `phase_runner` dispatches on, and the config keys `phase_config` says have
|
||||||
|
//! readers. Both are enforced again on the way in, so a plan cannot describe
|
||||||
|
//! work this platform will accept and then not do.
|
||||||
|
|
||||||
|
use axum::extract::{Path, State};
|
||||||
|
use axum::Json;
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
use serde_json::{json, Value};
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
use crate::mission_plan::{Plan, MAX_PHASES, PLANNABLE_KINDS};
|
||||||
|
use crate::{ApiError, AppState, Authed};
|
||||||
|
|
||||||
|
const PLANNER_MODEL: &str = "claude-opus-4-8";
|
||||||
|
|
||||||
|
/// What the repository actually contains, for the planner's prompt.
|
||||||
|
///
|
||||||
|
/// Names were not enough. Given the root listing alone, the planner wrote
|
||||||
|
/// "optimise the hot path" for a crate whose hot path is
|
||||||
|
/// `add(a: i64, b: i64) -> i64` — a mission that was unachievable from the
|
||||||
|
/// moment it was written, and that nothing discovered until an agent had built a
|
||||||
|
/// benchmark harness to measure an integer addition.
|
||||||
|
///
|
||||||
|
/// Read from the FORGE, not a checkout: at proposal time the mission is still a
|
||||||
|
/// draft and `ensure_checkout` has not run, so there is nothing on disk. Every
|
||||||
|
/// failure degrades to a STATED absence — a planner told "the listing could not
|
||||||
|
/// be read" can hedge; one told nothing assumes.
|
||||||
|
async fn repo_digest(pool: &sqlx::PgPool, mission_id: uuid::Uuid) -> String {
|
||||||
|
let row: Option<(Option<String>, Option<String>, Option<String>)> = sqlx::query_as(
|
||||||
|
"SELECT r.owner, r.name, r.default_branch
|
||||||
|
FROM missions m JOIN repos r ON r.id = m.repo_id
|
||||||
|
WHERE m.id = $1",
|
||||||
|
)
|
||||||
|
.bind(mission_id)
|
||||||
|
.fetch_optional(pool)
|
||||||
|
.await
|
||||||
|
.ok()
|
||||||
|
.flatten();
|
||||||
|
let Some((Some(owner), Some(name), branch)) = row else {
|
||||||
|
return "(this mission has no repository)".to_string();
|
||||||
|
};
|
||||||
|
let branch = branch.unwrap_or_else(|| "main".to_string());
|
||||||
|
// Distinguish "no credential" from "the forge said no". Both used to
|
||||||
|
// arrive as the same "(could not be read)" string, so an unconfigured
|
||||||
|
// deployment looked identical to a private repo — and the planner, told
|
||||||
|
// only that the read failed, cannot say which.
|
||||||
|
let token = std::env::var("GITEA_TOKEN").unwrap_or_default();
|
||||||
|
let unauthenticated = token.trim().is_empty();
|
||||||
|
let Ok(client) = reqwest::Client::builder()
|
||||||
|
.timeout(std::time::Duration::from_secs(20))
|
||||||
|
.build()
|
||||||
|
else {
|
||||||
|
return "(the repository could not be read)".to_string();
|
||||||
|
};
|
||||||
|
let auth = |r: reqwest::RequestBuilder| {
|
||||||
|
if token.trim().is_empty() {
|
||||||
|
r
|
||||||
|
} else {
|
||||||
|
r.header("Authorization", format!("token {token}"))
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// The whole tree in one call, so "does this repo have benches/" is a fact
|
||||||
|
// rather than an inference from the root.
|
||||||
|
let tree_url = format!(
|
||||||
|
"https://git.redclaw.dev/api/v1/repos/{owner}/{name}/git/trees/{branch}?recursive=true&per_page=1000"
|
||||||
|
);
|
||||||
|
let tree: serde_json::Value = match auth(client.get(&tree_url)).send().await {
|
||||||
|
Ok(r) if r.status().is_success() => r.json().await.unwrap_or_default(),
|
||||||
|
_ if unauthenticated => {
|
||||||
|
return "(the repository tree could not be read: GITEA_TOKEN is unset, \
|
||||||
|
so this read was unauthenticated)"
|
||||||
|
.to_string()
|
||||||
|
}
|
||||||
|
_ => return "(the repository tree could not be read)".to_string(),
|
||||||
|
};
|
||||||
|
let entries: Vec<crate::repo_digest::FileEntry> = tree
|
||||||
|
.get("tree")
|
||||||
|
.and_then(|t| t.as_array())
|
||||||
|
.map(|items| {
|
||||||
|
items
|
||||||
|
.iter()
|
||||||
|
.filter(|e| e.get("type").and_then(|v| v.as_str()) == Some("blob"))
|
||||||
|
.filter_map(|e| {
|
||||||
|
Some(crate::repo_digest::FileEntry {
|
||||||
|
path: e.get("path")?.as_str()?.to_string(),
|
||||||
|
size: e.get("size").and_then(|v| v.as_u64()).unwrap_or(0) as usize,
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
})
|
||||||
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
// Fetch in priority order until the budget is spent. Requested serially and
|
||||||
|
// capped: this runs inside one API request, and a repo with 500 useful files
|
||||||
|
// must not turn a proposal into 500 round trips.
|
||||||
|
let mut fetched: Vec<(String, String)> = Vec::new();
|
||||||
|
let mut spent = 0usize;
|
||||||
|
for e in crate::repo_digest::priority(&entries).into_iter().take(40) {
|
||||||
|
if spent >= crate::repo_digest::CONTENT_BUDGET {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
let raw = format!(
|
||||||
|
"https://git.redclaw.dev/api/v1/repos/{owner}/{name}/raw/{}?ref={branch}",
|
||||||
|
e.path
|
||||||
|
);
|
||||||
|
if let Ok(r) = auth(client.get(&raw)).send().await {
|
||||||
|
if r.status().is_success() {
|
||||||
|
if let Ok(text) = r.text().await {
|
||||||
|
spent += text.len().min(crate::repo_digest::PER_FILE_CAP);
|
||||||
|
fetched.push((e.path.clone(), text));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
crate::repo_digest::render(&entries, &crate::repo_digest::fit(fetched))
|
||||||
|
}
|
||||||
|
|
||||||
|
const PLAN_SYSTEM: &str = "You decide what ONE software mission actually does — its phases, in order. \
|
||||||
|
Each phase is a full agent run against the same repository checkout: the next phase sees the tree the \
|
||||||
|
previous one left. They run SEQUENTIALLY, so phases are expensive and a handoff loses context at every \
|
||||||
|
step.\n\n\
|
||||||
|
Propose the FEWEST phases that genuinely need to be separate. ONE phase is usually the right answer, and \
|
||||||
|
is always the right answer for a self-contained change: splitting one change into plan → implement → \
|
||||||
|
test is a documented anti-pattern, not thoroughness — a single agent doing all three in one pass keeps \
|
||||||
|
the context that makes the later steps good. A second phase earns its place only when it depends on \
|
||||||
|
something the first phase could not have known when it started.\n\n\
|
||||||
|
Every phase needs a `task`: the specific instruction for THAT phase, not a restatement of the mission. \
|
||||||
|
An agent receives the mission description plus its own task, so a vague task means an agent guessing \
|
||||||
|
which part of the mission is its share.\n\n\
|
||||||
|
`done_when` is judged afterwards by a separate model reading the repository, so write it as something \
|
||||||
|
observable in the tree — a file that exists, a suite that passes — never as an intention. \
|
||||||
|
`done_when_check` is a SHELL COMMAND that must exit 0; it is enforced while the agent still works, so \
|
||||||
|
prefer it when the condition is mechanical. Set `allow_empty` true only for a phase whose job is to \
|
||||||
|
verify rather than to change files.\n\n\
|
||||||
|
ALWAYS respond with STRICT JSON ONLY, no prose and no markdown: \
|
||||||
|
{\"phases\":[{\"kind\":\"coding\",\"task\":\"...\",\"done_when\":null|\"...\",\
|
||||||
|
\"done_when_check\":null|\"...\",\"allow_empty\":null|true|false}]}";
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize)]
|
||||||
|
pub struct PlanProposalResponse {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub plan: Value,
|
||||||
|
pub author_model: String,
|
||||||
|
pub status: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `POST /api/missions/{id}/plan-proposals` — ask the model for a phase plan.
|
||||||
|
pub async fn suggest(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
Path(id): Path<Uuid>,
|
||||||
|
) -> Result<Json<PlanProposalResponse>, ApiError> {
|
||||||
|
let ws = user.workspace_id;
|
||||||
|
let mission = cm_db::repo::missions::get(&state.pool, id, ws.as_uuid())
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?
|
||||||
|
.ok_or(ApiError::NotFound)?;
|
||||||
|
|
||||||
|
let prompt = format!(
|
||||||
|
"MISSION: {}\n\nDESCRIPTION:\n{}\n\n=== THE REPOSITORY ===\n{}\n=== END REPOSITORY \
|
||||||
|
===\n\nPlan for the repository as it ACTUALLY IS, not as the description implies it \
|
||||||
|
might be. If the work needs something absent — a benchmark harness, a test suite, a \
|
||||||
|
config file — the phase that needs it must CREATE it, and its task must say so. If the \
|
||||||
|
description asks for something this code cannot support (optimising a function with \
|
||||||
|
nothing to optimise, testing a module that does not exist), say so in the task text and \
|
||||||
|
plan the phase that would establish the truth, rather than a phase that must fail.\n\n\
|
||||||
|
NOTE: a mission agent has NO package-registry access — it cannot add dependencies. A \
|
||||||
|
phase needing tooling must build it from the standard library or from what is already \
|
||||||
|
vendored here.\n\nPHASE KINDS YOU MAY USE (nothing else runs): {}\nCEILING: \
|
||||||
|
{MAX_PHASES} phases.\n\nPropose the plan now (JSON only).",
|
||||||
|
mission.title,
|
||||||
|
mission.description.as_deref().unwrap_or("(none)"),
|
||||||
|
repo_digest(&state.pool, id).await,
|
||||||
|
PLANNABLE_KINDS.join(", "),
|
||||||
|
);
|
||||||
|
|
||||||
|
// The stored `author_model` is whichever link of the fallback chain
|
||||||
|
// actually answered — see `subscription::complete_with_fallback`.
|
||||||
|
let (raw, author_model) = crate::subscription::complete_with_fallback(
|
||||||
|
&state.runtime,
|
||||||
|
PLAN_SYSTEM,
|
||||||
|
&prompt,
|
||||||
|
PLANNER_MODEL,
|
||||||
|
2000,
|
||||||
|
false,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("mission {id}: plan proposal failed: {e}");
|
||||||
|
crate::subscription::as_api_error(&e)
|
||||||
|
})?;
|
||||||
|
|
||||||
|
let parsed: Value = crate::routes::claws::extract_json(&raw).ok_or_else(|| {
|
||||||
|
eprintln!("mission {id}: planner returned no JSON: {raw}");
|
||||||
|
ApiError::BadRequest
|
||||||
|
})?;
|
||||||
|
let plan: Plan = serde_json::from_value(parsed.clone()).map_err(|e| {
|
||||||
|
eprintln!("mission {id}: planner JSON is not a plan ({e}): {parsed}");
|
||||||
|
ApiError::BadRequest
|
||||||
|
})?;
|
||||||
|
// Validated BEFORE storing, so a stored proposal is always one that could be
|
||||||
|
// approved — the failure belongs to the model, not to whoever clicks
|
||||||
|
// approve later.
|
||||||
|
if let Err(why) = plan.validate() {
|
||||||
|
eprintln!("mission {id}: planner proposed an unrunnable plan: {why}");
|
||||||
|
return Err(ApiError::BadRequest);
|
||||||
|
}
|
||||||
|
|
||||||
|
let pid = Uuid::now_v7();
|
||||||
|
let stored = serde_json::to_value(&plan).map_err(|_| ApiError::Internal)?;
|
||||||
|
cm_db::repo::mission_plan_proposals::insert(
|
||||||
|
&state.pool,
|
||||||
|
pid,
|
||||||
|
id,
|
||||||
|
ws.as_uuid().to_owned(),
|
||||||
|
&stored,
|
||||||
|
&author_model,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("mission {id}: could not store plan proposal: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
eprintln!(
|
||||||
|
"mission_plan: mission {id} — {author_model} proposed {} phase(s): {}",
|
||||||
|
plan.phases.len(),
|
||||||
|
plan.phases
|
||||||
|
.iter()
|
||||||
|
.map(|p| p.kind.as_str())
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(" → ")
|
||||||
|
);
|
||||||
|
|
||||||
|
Ok(Json(PlanProposalResponse {
|
||||||
|
id: pid,
|
||||||
|
plan: stored,
|
||||||
|
author_model,
|
||||||
|
status: "proposed".into(),
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `GET /api/missions/{id}/plan-proposals`
|
||||||
|
pub async fn list(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
Path(id): Path<Uuid>,
|
||||||
|
) -> Result<Json<Vec<cm_db::repo::mission_plan_proposals::MissionPlanProposal>>, ApiError> {
|
||||||
|
let rows = cm_db::repo::mission_plan_proposals::list(
|
||||||
|
&state.pool,
|
||||||
|
id,
|
||||||
|
user.workspace_id.as_uuid().to_owned(),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?;
|
||||||
|
Ok(Json(rows))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
pub struct DecideRequest {
|
||||||
|
pub status: String,
|
||||||
|
#[serde(default)]
|
||||||
|
pub note: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `POST /api/missions/{id}/plan-proposals/{pid}/decide`
|
||||||
|
///
|
||||||
|
/// Approving REPLACES the mission's phases. Draft-only: re-planning a mission
|
||||||
|
/// whose phases have started would discard work that already ran, and the phase
|
||||||
|
/// rows are what every downstream sweep keys off.
|
||||||
|
pub async fn decide(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
Path((id, pid)): Path<(Uuid, Uuid)>,
|
||||||
|
Json(body): Json<DecideRequest>,
|
||||||
|
) -> Result<Json<Value>, ApiError> {
|
||||||
|
let ws = user.workspace_id;
|
||||||
|
let proposal = cm_db::repo::mission_plan_proposals::get(&state.pool, pid, ws.as_uuid().to_owned())
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?
|
||||||
|
.ok_or(ApiError::NotFound)?;
|
||||||
|
if proposal.mission_id != id {
|
||||||
|
return Err(ApiError::NotFound);
|
||||||
|
}
|
||||||
|
|
||||||
|
if body.status == "rejected" {
|
||||||
|
let decided = cm_db::repo::mission_plan_proposals::decide(
|
||||||
|
&state.pool,
|
||||||
|
pid,
|
||||||
|
ws.as_uuid().to_owned(),
|
||||||
|
"rejected",
|
||||||
|
body.note.as_deref(),
|
||||||
|
Some(user.user_id.as_uuid().to_owned()),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?;
|
||||||
|
return Ok(Json(json!({ "status": "rejected", "decided": decided })));
|
||||||
|
}
|
||||||
|
if body.status != "approved" {
|
||||||
|
return Err(ApiError::BadRequest);
|
||||||
|
}
|
||||||
|
|
||||||
|
let mission = cm_db::repo::missions::get(&state.pool, id, ws.as_uuid())
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?
|
||||||
|
.ok_or(ApiError::NotFound)?;
|
||||||
|
if mission.status != "draft" {
|
||||||
|
return Err(ApiError::Refused(format!(
|
||||||
|
"this mission is {} — a {} can only be approved while it is a draft, \
|
||||||
|
because approving one rewrites how the mission will run",
|
||||||
|
mission.status, "plan"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
|
||||||
|
let plan: Plan = serde_json::from_value(proposal.plan.clone()).map_err(|e| {
|
||||||
|
eprintln!("mission {id}: stored plan {pid} does not parse ({e})");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
// Re-validated at approval. The stored plan passed once, but `PLANNABLE_KINDS`
|
||||||
|
// and the config registry are properties of the BUILD — a proposal made
|
||||||
|
// before a deploy could name a kind this build no longer dispatches.
|
||||||
|
if let Err(why) = plan.validate() {
|
||||||
|
eprintln!("mission {id}: plan {pid} is no longer runnable: {why}");
|
||||||
|
let reason = why.to_string();
|
||||||
|
let _ = cm_db::repo::mission_plan_proposals::decide(
|
||||||
|
&state.pool,
|
||||||
|
pid,
|
||||||
|
ws.as_uuid().to_owned(),
|
||||||
|
"rejected",
|
||||||
|
Some(&why.to_string()),
|
||||||
|
Some(user.user_id.as_uuid().to_owned()),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
// `Refusal` is already written as human-readable copy — it names the
|
||||||
|
// constraint and why it exists. It was going to stderr only.
|
||||||
|
return Err(ApiError::Refused(format!(
|
||||||
|
"this plan is no longer runnable on the current build, so it was \
|
||||||
|
rejected: {reason}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
|
||||||
|
let phases = plan.phases();
|
||||||
|
let claimed = cm_db::repo::mission_plan_proposals::approve_and_apply(
|
||||||
|
&state.pool,
|
||||||
|
pid,
|
||||||
|
id,
|
||||||
|
ws.as_uuid().to_owned(),
|
||||||
|
&phases,
|
||||||
|
body.note.as_deref(),
|
||||||
|
Some(user.user_id.as_uuid().to_owned()),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("mission {id}: could not apply plan {pid}: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
if !claimed {
|
||||||
|
return Err(ApiError::BadRequest);
|
||||||
|
}
|
||||||
|
|
||||||
|
eprintln!(
|
||||||
|
"mission_plan: mission {id} now runs a {}-phase model-authored plan from proposal {pid}",
|
||||||
|
phases.len()
|
||||||
|
);
|
||||||
|
Ok(Json(json!({
|
||||||
|
"status": "approved",
|
||||||
|
"phases": phases.iter().map(|(k, i, _)| json!({"kind": k, "order_idx": i})).collect::<Vec<_>>(),
|
||||||
|
})))
|
||||||
|
}
|
||||||
@@ -0,0 +1,331 @@
|
|||||||
|
//! `/api/missions/{id}/team-proposals` — let a model size the mission's team.
|
||||||
|
//!
|
||||||
|
//! Slice 5. The planner has been proposing rosters into React state for months;
|
||||||
|
//! this is where one reaches a mission. Three verbs, and the split between them
|
||||||
|
//! is the point:
|
||||||
|
//!
|
||||||
|
//! - **suggest** asks the model and PERSISTS the answer. It changes nothing
|
||||||
|
//! about the mission.
|
||||||
|
//! - **approve** writes the roster onto the mission, where the composed executor
|
||||||
|
//! reads it.
|
||||||
|
//! - **reject** records that a human said no, which is the only evidence we ever
|
||||||
|
//! collect about what the planner gets wrong.
|
||||||
|
//!
|
||||||
|
//! A proposal is never applied on arrival. A model sizing a team is a suggestion
|
||||||
|
//! about how many VMs to boot, and this codebase has an explicit rule about
|
||||||
|
//! model output that costs money: it is evidence for a decision, not the
|
||||||
|
//! decision.
|
||||||
|
|
||||||
|
use axum::extract::{Path, State};
|
||||||
|
use axum::Json;
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
use serde_json::{json, Value};
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
use crate::mission_roster::{available_backends, Roster};
|
||||||
|
use crate::{ApiError, AppState, Authed};
|
||||||
|
|
||||||
|
/// The model that sizes a mission's team.
|
||||||
|
///
|
||||||
|
/// The same one the Master Planner uses. Sizing a team is the kind of judgement
|
||||||
|
/// the planner's own system prompt calls for — and it is a once-per-mission call,
|
||||||
|
/// so the cost argument that keeps missions on cheaper models does not apply.
|
||||||
|
const PLANNER_MODEL: &str = "claude-opus-4-8";
|
||||||
|
|
||||||
|
const ROSTER_SYSTEM: &str = "You size the team for ONE software mission that runs inside Firecracker \
|
||||||
|
microVMs. Each member you propose is a WHOLE VM — a boot, a repository injected as a tar, a full \
|
||||||
|
Claude Code session, and a collect — running one after another, each one receiving the working tree the \
|
||||||
|
previous member left behind. That is expensive and it is serial, so propose the FEWEST members that \
|
||||||
|
genuinely divide the work. One member is a perfectly good answer and is usually the right one for a \
|
||||||
|
small change; Anthropic measure multi-agent work at 3-10x the tokens with wall-clock often LONGER, and \
|
||||||
|
the benefit is thoroughness rather than speed.\n\n\
|
||||||
|
Members run SEQUENTIALLY and share the repository, so do NOT propose members that would edit the same \
|
||||||
|
file, and do NOT split one change into stages (plan → implement → test) — a handoff loses context at \
|
||||||
|
every step and one careful pass beats an assembly line. The shape that DOES earn its cost is an \
|
||||||
|
implementer followed by an independent verifier that only checks.\n\n\
|
||||||
|
Give each member a `backend` ONLY when running it on a different provider's image is the point — an \
|
||||||
|
independent verifier on another provider breaks the correlated failure where the model that wrote the \
|
||||||
|
code also grades it. Omit `backend` to inherit the mission's.\n\n\
|
||||||
|
ALWAYS respond with STRICT JSON ONLY, no prose and no markdown: \
|
||||||
|
{\"topology_kind\":\"pipeline\",\"members\":[{\"role\":\"...\",\"backend\":null|\"...\",\
|
||||||
|
\"rationale\":\"one line\"}]}";
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize)]
|
||||||
|
pub struct ProposalResponse {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub roster: Value,
|
||||||
|
pub author_model: String,
|
||||||
|
pub status: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `POST /api/missions/{id}/team-proposals` — ask the model for a roster.
|
||||||
|
pub async fn suggest(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
Path(id): Path<Uuid>,
|
||||||
|
) -> Result<Json<ProposalResponse>, ApiError> {
|
||||||
|
let ws = user.workspace_id;
|
||||||
|
let mission = cm_db::repo::missions::get(&state.pool, id, ws.as_uuid())
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?
|
||||||
|
.ok_or(ApiError::NotFound)?;
|
||||||
|
|
||||||
|
// The backends the FLEET can boot today, handed to the model as the menu.
|
||||||
|
// Without it the model invents plausible image names and the roster is
|
||||||
|
// refused after it was written, which reads as our bug rather than as a
|
||||||
|
// model guessing.
|
||||||
|
let available = available_backends(&state.pool, ws.as_uuid().to_owned())
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("mission {id}: could not read fleet backends: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
|
||||||
|
let phases: Vec<(String, Option<String>)> = sqlx::query_as(
|
||||||
|
"SELECT kind, config->>'task' FROM mission_phases WHERE mission_id = $1 ORDER BY order_idx",
|
||||||
|
)
|
||||||
|
.bind(id)
|
||||||
|
.fetch_all(&state.pool)
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?;
|
||||||
|
|
||||||
|
let phase_text = phases
|
||||||
|
.iter()
|
||||||
|
.map(|(kind, task)| format!("- {kind}: {}", task.as_deref().unwrap_or("(no task text)")))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join("\n");
|
||||||
|
let prompt = format!(
|
||||||
|
"MISSION: {}\n\nDESCRIPTION:\n{}\n\nPHASES:\n{}\n\nBACKENDS THIS FLEET CAN BOOT (use only \
|
||||||
|
these, or omit `backend`): {}\n\nPropose the roster now (JSON only).",
|
||||||
|
mission.title,
|
||||||
|
mission.description.as_deref().unwrap_or("(none)"),
|
||||||
|
if phase_text.is_empty() {
|
||||||
|
"(none declared)".to_string()
|
||||||
|
} else {
|
||||||
|
phase_text
|
||||||
|
},
|
||||||
|
if available.is_empty() {
|
||||||
|
"(none — omit backend on every member)".to_string()
|
||||||
|
} else {
|
||||||
|
available.join(", ")
|
||||||
|
},
|
||||||
|
);
|
||||||
|
|
||||||
|
// On the SUBSCRIPTION, like every mission VM — not the metered API key.
|
||||||
|
// `Runtime::complete` with a bare model name resolves to the default
|
||||||
|
// provider, which is the pay-as-you-go key; this planner died with
|
||||||
|
// "credit balance is too low" while missions on the same box ran fine.
|
||||||
|
// `author_model` is what ANSWERED, not what was asked for. When opus is
|
||||||
|
// capped the chain steps down to haiku and then to GLM, and a plan drafted
|
||||||
|
// by the third link but filed as an opus plan is a silent quality change.
|
||||||
|
let (raw, author_model) = crate::subscription::complete_with_fallback(
|
||||||
|
&state.runtime,
|
||||||
|
ROSTER_SYSTEM,
|
||||||
|
&prompt,
|
||||||
|
PLANNER_MODEL,
|
||||||
|
2000,
|
||||||
|
false,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("mission {id}: roster proposal failed: {e}");
|
||||||
|
// A rate-limited subscription is a 503 the operator can act on, not
|
||||||
|
// a 500 that reads as "this server is broken".
|
||||||
|
crate::subscription::as_api_error(&e)
|
||||||
|
})?;
|
||||||
|
|
||||||
|
// A model that answered with prose around its JSON has still answered; a
|
||||||
|
// model that answered with nothing usable has not, and that is a refusal
|
||||||
|
// rather than an empty roster.
|
||||||
|
let parsed: Value = crate::routes::claws::extract_json(&raw).ok_or_else(|| {
|
||||||
|
eprintln!("mission {id}: planner returned no JSON: {raw}");
|
||||||
|
ApiError::BadRequest
|
||||||
|
})?;
|
||||||
|
let roster: Roster = serde_json::from_value(parsed.clone()).map_err(|e| {
|
||||||
|
eprintln!("mission {id}: planner JSON is not a roster ({e}): {parsed}");
|
||||||
|
ApiError::BadRequest
|
||||||
|
})?;
|
||||||
|
// Validated BEFORE it is stored, so a stored proposal is always one that
|
||||||
|
// could be approved. Storing an invalid roster would mean the failure
|
||||||
|
// surfaces at approval time, pointing at the human rather than the model.
|
||||||
|
if let Err(why) = roster.validate(&available) {
|
||||||
|
eprintln!("mission {id}: planner proposed an unusable roster: {why}");
|
||||||
|
return Err(ApiError::BadRequest);
|
||||||
|
}
|
||||||
|
|
||||||
|
let pid = Uuid::now_v7();
|
||||||
|
let stored = serde_json::to_value(&roster).map_err(|_| ApiError::Internal)?;
|
||||||
|
cm_db::repo::mission_team_proposals::insert(
|
||||||
|
&state.pool,
|
||||||
|
pid,
|
||||||
|
id,
|
||||||
|
ws.as_uuid().to_owned(),
|
||||||
|
&stored,
|
||||||
|
&author_model,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("mission {id}: could not store proposal: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
eprintln!(
|
||||||
|
"mission_roster: mission {id} — {} proposed {} member(s): {}",
|
||||||
|
author_model,
|
||||||
|
roster.members.len(),
|
||||||
|
roster
|
||||||
|
.members
|
||||||
|
.iter()
|
||||||
|
.map(|m| format!("{}{}", m.role, m.backend.as_deref().map(|b| format!("@{b}")).unwrap_or_default()))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(", ")
|
||||||
|
);
|
||||||
|
|
||||||
|
Ok(Json(ProposalResponse {
|
||||||
|
id: pid,
|
||||||
|
roster: stored,
|
||||||
|
author_model,
|
||||||
|
status: "proposed".into(),
|
||||||
|
}))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `GET /api/missions/{id}/team-proposals`
|
||||||
|
pub async fn list(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
Path(id): Path<Uuid>,
|
||||||
|
) -> Result<Json<Vec<cm_db::repo::mission_team_proposals::MissionTeamProposal>>, ApiError> {
|
||||||
|
let rows =
|
||||||
|
cm_db::repo::mission_team_proposals::list(&state.pool, id, user.workspace_id.as_uuid().to_owned())
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?;
|
||||||
|
Ok(Json(rows))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Deserialize)]
|
||||||
|
pub struct DecideRequest {
|
||||||
|
/// `approved` or `rejected`.
|
||||||
|
pub status: String,
|
||||||
|
#[serde(default)]
|
||||||
|
pub note: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `POST /api/missions/{id}/team-proposals/{pid}/decide` — accept or refuse.
|
||||||
|
///
|
||||||
|
/// Approving writes `config.roster` on the mission and switches it to the
|
||||||
|
/// composed engine, because a roster is a graph of VMs and that is the engine
|
||||||
|
/// that runs one. Draft-only: re-shaping a mission that is already running would
|
||||||
|
/// change what its next phase does with no record of the swap on the phase that
|
||||||
|
/// already ran.
|
||||||
|
pub async fn decide(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
Path((id, pid)): Path<(Uuid, Uuid)>,
|
||||||
|
Json(body): Json<DecideRequest>,
|
||||||
|
) -> Result<Json<Value>, ApiError> {
|
||||||
|
let ws = user.workspace_id;
|
||||||
|
let proposal = cm_db::repo::mission_team_proposals::get(&state.pool, pid, ws.as_uuid().to_owned())
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?
|
||||||
|
.ok_or(ApiError::NotFound)?;
|
||||||
|
if proposal.mission_id != id {
|
||||||
|
return Err(ApiError::NotFound);
|
||||||
|
}
|
||||||
|
|
||||||
|
if body.status == "rejected" {
|
||||||
|
let decided = cm_db::repo::mission_team_proposals::decide(
|
||||||
|
&state.pool,
|
||||||
|
pid,
|
||||||
|
ws.as_uuid().to_owned(),
|
||||||
|
"rejected",
|
||||||
|
body.note.as_deref(),
|
||||||
|
Some(user.user_id.as_uuid().to_owned()),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?;
|
||||||
|
return Ok(Json(json!({ "status": "rejected", "decided": decided })));
|
||||||
|
}
|
||||||
|
if body.status != "approved" {
|
||||||
|
return Err(ApiError::BadRequest);
|
||||||
|
}
|
||||||
|
|
||||||
|
let mission = cm_db::repo::missions::get(&state.pool, id, ws.as_uuid())
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?
|
||||||
|
.ok_or(ApiError::NotFound)?;
|
||||||
|
if mission.status != "draft" {
|
||||||
|
return Err(ApiError::Refused(format!(
|
||||||
|
"this mission is {} — a {} can only be approved while it is a draft, \
|
||||||
|
because approving one rewrites how the mission will run",
|
||||||
|
mission.status, "roster"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
|
||||||
|
let roster: Roster = serde_json::from_value(proposal.roster.clone()).map_err(|e| {
|
||||||
|
eprintln!("mission {id}: stored proposal {pid} is not a roster ({e})");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
// Re-validated at approval, against the fleet as it is NOW. A node can go
|
||||||
|
// offline between proposing and approving, and the cheapest place to find
|
||||||
|
// that out is still here rather than at VM boot.
|
||||||
|
let available = available_backends(&state.pool, ws.as_uuid().to_owned())
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Internal)?;
|
||||||
|
if let Err(why) = roster.validate(&available) {
|
||||||
|
eprintln!("mission {id}: roster {pid} is no longer applicable: {why}");
|
||||||
|
let reason = why.to_string();
|
||||||
|
let _ = cm_db::repo::mission_team_proposals::decide(
|
||||||
|
&state.pool,
|
||||||
|
pid,
|
||||||
|
ws.as_uuid().to_owned(),
|
||||||
|
"rejected",
|
||||||
|
Some(&why.to_string()),
|
||||||
|
Some(user.user_id.as_uuid().to_owned()),
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
// The proposal has just been auto-rejected, so the caller is about to
|
||||||
|
// re-read a list where it says "rejected" with no visible cause. The
|
||||||
|
// reason is the whole content of this response.
|
||||||
|
return Err(ApiError::Refused(format!(
|
||||||
|
"this roster no longer applies to the fleet as it is now, so it was \
|
||||||
|
rejected: {reason}"
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
|
||||||
|
let graph = roster.graph().map_err(|e| {
|
||||||
|
eprintln!("mission {id}: approved roster does not build a graph: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
// Claiming the proposal and writing the mission are ONE transaction. Doing
|
||||||
|
// them as two statements left the first real approval in production marked
|
||||||
|
// `approved` with nothing written to the mission — and the partial unique
|
||||||
|
// index then makes that permanent, since no other proposal for that mission
|
||||||
|
// can ever be approved.
|
||||||
|
let claimed = cm_db::repo::mission_team_proposals::approve_and_apply(
|
||||||
|
&state.pool,
|
||||||
|
pid,
|
||||||
|
id,
|
||||||
|
ws.as_uuid().to_owned(),
|
||||||
|
&graph,
|
||||||
|
body.note.as_deref(),
|
||||||
|
Some(user.user_id.as_uuid().to_owned()),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("mission {id}: could not apply roster {pid}: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
if !claimed {
|
||||||
|
return Err(ApiError::BadRequest);
|
||||||
|
}
|
||||||
|
|
||||||
|
eprintln!(
|
||||||
|
"mission_roster: mission {id} now runs a {}-node composed graph from proposal {pid}",
|
||||||
|
roster.members.len()
|
||||||
|
);
|
||||||
|
Ok(Json(json!({
|
||||||
|
"status": "approved",
|
||||||
|
"team_engine": "composed",
|
||||||
|
"nodes": roster.members.len(),
|
||||||
|
"graph": graph,
|
||||||
|
})))
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -14,7 +14,10 @@ pub mod health;
|
|||||||
pub mod identity;
|
pub mod identity;
|
||||||
pub mod level_up;
|
pub mod level_up;
|
||||||
pub mod library;
|
pub mod library;
|
||||||
|
pub mod mission_plan;
|
||||||
|
pub mod mission_roster;
|
||||||
pub mod missions;
|
pub mod missions;
|
||||||
|
pub mod podcast;
|
||||||
pub mod nodes;
|
pub mod nodes;
|
||||||
pub mod oauth;
|
pub mod oauth;
|
||||||
pub mod orgs;
|
pub mod orgs;
|
||||||
|
|||||||
@@ -402,3 +402,115 @@ async fn bridge_terminal(hub: Arc<NodeHub>, node_id: NodeId, socket: WebSocket)
|
|||||||
}
|
}
|
||||||
hub.terminal_close(node_id, sid).await;
|
hub.terminal_close(node_id, sid).await;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `GET /api/fleet/capacity` — what the SCHEDULER sees, verbatim.
|
||||||
|
///
|
||||||
|
/// Pulled forward from the observability phase because the capacity harness
|
||||||
|
/// scenario needs it: a test that recomputed the slot arithmetic in bash would
|
||||||
|
/// drift from `vm_placement` and then agree with itself while the scheduler did
|
||||||
|
/// something else. This returns `vm_placement::survey` unmodified, so the fleet
|
||||||
|
/// page, the harness and the placer cannot disagree.
|
||||||
|
///
|
||||||
|
/// `backend` narrows to the nodes that can boot one image (`?backend=claude`),
|
||||||
|
/// matching what `choose` does for a phase.
|
||||||
|
pub async fn capacity(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
Query(q): Query<CapacityQuery>,
|
||||||
|
) -> Result<Json<Value>, ApiError> {
|
||||||
|
let ws = user.workspace_id.as_uuid().to_owned();
|
||||||
|
let (fit, unfit) =
|
||||||
|
crate::vm_placement::survey(
|
||||||
|
&state.pool,
|
||||||
|
&state.node_hub,
|
||||||
|
ws,
|
||||||
|
&crate::vm_placement::required_backends(q.backend.as_deref(), None),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("fleet capacity survey failed: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
let ranked = crate::vm_placement::rank(fit);
|
||||||
|
Ok(Json(json!({
|
||||||
|
// Total free slots across the fleet. A burst larger than this MUST
|
||||||
|
// queue rather than overcommit — that is the whole feature.
|
||||||
|
"slots": ranked.iter().map(|n| n.slots).sum::<i64>(),
|
||||||
|
"nodes": ranked.iter().map(|n| json!({
|
||||||
|
"id": n.node_id,
|
||||||
|
"name": n.name,
|
||||||
|
"slots": n.slots,
|
||||||
|
"committedVms": n.committed_vms,
|
||||||
|
"headroom": n.headroom,
|
||||||
|
"memTotalMib": n.mem_total_mib,
|
||||||
|
"usedEffMib": n.used_eff_mib,
|
||||||
|
"diskFreeGib": n.disk_free_gib,
|
||||||
|
})).collect::<Vec<_>>(),
|
||||||
|
// Never folded into the above. "Full" and "unreadable" send an
|
||||||
|
// operator to different places, so they stay separate here too.
|
||||||
|
"unfit": unfit.iter().map(|(id, name, why)| json!({
|
||||||
|
"id": id,
|
||||||
|
"name": name,
|
||||||
|
"reason": why.reason(),
|
||||||
|
})).collect::<Vec<_>>(),
|
||||||
|
})))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
pub struct CapacityQuery {
|
||||||
|
pub backend: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `GET /api/fleet/backends` — the microVM backends a mission may actually use.
|
||||||
|
///
|
||||||
|
/// The SAME `available_backends` the roster planner is handed, not a second
|
||||||
|
/// list. The two rules it applies are both load-bearing and neither is obvious
|
||||||
|
/// from a node's capabilities alone: a backend must be built on an online node,
|
||||||
|
/// and it must have a credential contract. `agent-terminal` satisfies the first
|
||||||
|
/// and not the second — bootable, with nothing for the agent inside to
|
||||||
|
/// authenticate with — so offering it would produce a mission that validates,
|
||||||
|
/// launches, and fails at the agent turn, which is the expensive kind of late.
|
||||||
|
///
|
||||||
|
/// Exists because the UI had no backend selector at all: every mission created
|
||||||
|
/// from the dashboard ran on `claude`, so `local-ornith`, `glm` and `kimi` were
|
||||||
|
/// reachable only by calling the API directly.
|
||||||
|
pub async fn backends(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Authed(user): Authed,
|
||||||
|
) -> Result<Json<Value>, ApiError> {
|
||||||
|
let ws = user.workspace_id.as_uuid().to_owned();
|
||||||
|
let mut list = crate::mission_roster::available_backends(&state.pool, ws)
|
||||||
|
.await
|
||||||
|
.map_err(|e| {
|
||||||
|
eprintln!("fleet backends: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
// `default` is the generic `rootfs.ext4` and `claude` is the named one, and
|
||||||
|
// `microvm_credential_for` gives them the SAME contract — so a picker
|
||||||
|
// offering both shows two options with one meaning, and whichever the user
|
||||||
|
// picks they get the same thing. Collapse to the named one where it exists.
|
||||||
|
if list.iter().any(|b| b == "claude") {
|
||||||
|
list.retain(|b| b != "default");
|
||||||
|
}
|
||||||
|
Ok(Json(json!({
|
||||||
|
"backends": list.iter().map(|b| json!({
|
||||||
|
"id": b,
|
||||||
|
"label": backend_label(b),
|
||||||
|
})).collect::<Vec<_>>(),
|
||||||
|
})))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A name a person can choose between. The ids are deployment vocabulary
|
||||||
|
/// (`local-ornith`, `canary-claude`); a picker showing those alone asks the user
|
||||||
|
/// to know which company each one bills.
|
||||||
|
fn backend_label(id: &str) -> String {
|
||||||
|
match id {
|
||||||
|
"claude" => "Claude (Anthropic subscription)".into(),
|
||||||
|
"default" => "Claude (generic image)".into(),
|
||||||
|
"canary-claude" => "Claude — candidate CLI (canary)".into(),
|
||||||
|
"glm" => "GLM 4.7 (z.ai)".into(),
|
||||||
|
"kimi" => "Kimi (Moonshot)".into(),
|
||||||
|
"local-ornith" => "Ornith 9B — this fleet's own GPU".into(),
|
||||||
|
other => other.to_string(),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -40,7 +40,6 @@ research tools. Grant write only to members that actually produce code or commit
|
|||||||
- glm-4.7 — strong general reasoning (Z.ai); best cost/quality default for most workers.\n\
|
- glm-4.7 — strong general reasoning (Z.ai); best cost/quality default for most workers.\n\
|
||||||
- glm-5.2 — GLM Opus-class for the hardest reasoning roles; higher cost.\n\
|
- glm-5.2 — GLM Opus-class for the hardest reasoning roles; higher cost.\n\
|
||||||
- kimi — excellent for code-heavy roles.\n\
|
- kimi — excellent for code-heavy roles.\n\
|
||||||
- gemini — Gemini 2.5 Flash: very fast; classification, summarization, high-volume tasks.\n\
|
|
||||||
- groq — fastest/cheapest; simple sequential high-throughput steps.\n\
|
- groq — fastest/cheapest; simple sequential high-throughput steps.\n\
|
||||||
AGENT TOOLS each agent can use at runtime: web.search (find sources), browser.goto (fetch a URL), \
|
AGENT TOOLS each agent can use at runtime: web.search (find sources), browser.goto (fetch a URL), \
|
||||||
files.write (build a markdown vault in the shared drive), chat.send (delegate to teammates), \
|
files.write (build a markdown vault in the shared drive), chat.send (delegate to teammates), \
|
||||||
@@ -133,7 +132,11 @@ pub async fn planner_chat(
|
|||||||
};
|
};
|
||||||
let user_prompt = format!("{hierarchy}{topology_lock}\n\n=== CONVERSATION ===\n{convo}\n\nRespond now (JSON only).");
|
let user_prompt = format!("{hierarchy}{topology_lock}\n\n=== CONVERSATION ===\n{convo}\n\nRespond now (JSON only).");
|
||||||
let system = planner_system_for(&body.mode);
|
let system = planner_system_for(&body.mode);
|
||||||
let raw = match runtime.complete(&system, &user_prompt, "claude-opus-4-8", 8000, true).await {
|
let raw = match crate::subscription::complete_or(
|
||||||
|
&runtime, &system, &user_prompt, "claude-opus-4-8", 8000, true,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
Ok(t) => t,
|
Ok(t) => t,
|
||||||
Err(e) => { yield sse(json!({"stage":"error","label":format!("Opus error: {e}")})); return; }
|
Err(e) => { yield sse(json!({"stage":"error","label":format!("Opus error: {e}")})); return; }
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -0,0 +1,302 @@
|
|||||||
|
//! The private podcast feed.
|
||||||
|
//!
|
||||||
|
//! A podcast app is the right client for this: it downloads overnight, plays
|
||||||
|
//! offline, remembers position, and has lock-screen controls — none of which a
|
||||||
|
//! file in a folder gives you at the gym.
|
||||||
|
//!
|
||||||
|
//! Auth is a token in the query string, not a bearer header, because no podcast
|
||||||
|
//! app lets you set headers. That is a real trade: the token is in the URL and
|
||||||
|
//! therefore in the app's database and any proxy log it passes. It is scoped to
|
||||||
|
//! reading this feed and nothing else, and can be rotated by reissuing it.
|
||||||
|
|
||||||
|
use axum::extract::{Path, Query, State};
|
||||||
|
use axum::http::{header, StatusCode};
|
||||||
|
use axum::response::{IntoResponse, Response};
|
||||||
|
use serde::Deserialize;
|
||||||
|
use sqlx::Row;
|
||||||
|
|
||||||
|
use crate::{ApiError, AppState};
|
||||||
|
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
pub struct FeedAuth {
|
||||||
|
pub token: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The audio endpoint accepts a token either way — see `episode_audio`.
|
||||||
|
#[derive(Deserialize)]
|
||||||
|
pub struct OptionalAuth {
|
||||||
|
#[serde(default)]
|
||||||
|
pub token: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Resolve a feed token to the workspace it may read.
|
||||||
|
///
|
||||||
|
/// Reuses the normal API token table, so revoking a token revokes the feed with
|
||||||
|
/// it — a second secret store for podcasts would be one more thing to forget to
|
||||||
|
/// rotate.
|
||||||
|
async fn workspace_for(state: &AppState, token: &str) -> Result<uuid::Uuid, ApiError> {
|
||||||
|
let user = state
|
||||||
|
.auth
|
||||||
|
.authenticate(token)
|
||||||
|
.await
|
||||||
|
.map_err(|_| ApiError::Unauthorized)?;
|
||||||
|
Ok(user.workspace_id.as_uuid())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn xml_escape(s: &str) -> String {
|
||||||
|
s.replace('&', "&")
|
||||||
|
.replace('<', "<")
|
||||||
|
.replace('>', ">")
|
||||||
|
.replace('"', """)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn rfc2822(ts: time::OffsetDateTime) -> String {
|
||||||
|
// Podcast clients are strict about pubDate. `time`'s RFC2822 is exactly it.
|
||||||
|
ts.format(&time::format_description::well_known::Rfc2822)
|
||||||
|
.unwrap_or_else(|_| "Thu, 01 Jan 1970 00:00:00 +0000".into())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `GET /api/podcast/feed.xml?token=…`
|
||||||
|
pub async fn feed(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Query(auth): Query<FeedAuth>,
|
||||||
|
) -> Result<Response, ApiError> {
|
||||||
|
let workspace_id = workspace_for(&state, &auth.token).await?;
|
||||||
|
let rows = sqlx::query(
|
||||||
|
"SELECT id, episode_date, title, bytes, duration_secs, created_at
|
||||||
|
FROM podcast_episodes
|
||||||
|
WHERE workspace_id = $1
|
||||||
|
-- Skip markers for missions whose script was reaped before the
|
||||||
|
-- render sweep reached them: a zero-byte enclosure makes a podcast
|
||||||
|
-- app show a broken episode rather than simply not showing one.
|
||||||
|
AND bytes > 0
|
||||||
|
ORDER BY created_at DESC
|
||||||
|
LIMIT 100",
|
||||||
|
)
|
||||||
|
.bind(workspace_id)
|
||||||
|
.fetch_all(&state.pool)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let base = std::env::var("CLAWMATES_PUBLIC_URL")
|
||||||
|
.unwrap_or_else(|_| "http://localhost:8080".to_string());
|
||||||
|
let base = base.trim_end_matches('/');
|
||||||
|
|
||||||
|
let mut items = String::new();
|
||||||
|
for r in &rows {
|
||||||
|
let id: uuid::Uuid = r.get("id");
|
||||||
|
let title: String = r.get("title");
|
||||||
|
let date: String = r.get("episode_date");
|
||||||
|
let bytes: i64 = r.get("bytes");
|
||||||
|
let secs: i32 = r.get("duration_secs");
|
||||||
|
let created: time::OffsetDateTime = r.get("created_at");
|
||||||
|
// The token rides on the enclosure too: the app fetches the audio in a
|
||||||
|
// separate request that carries none of the feed's context.
|
||||||
|
let url = format!("{base}/api/podcast/episodes/{id}.mp3?token={}", auth.token);
|
||||||
|
items.push_str(&format!(
|
||||||
|
r#" <item>
|
||||||
|
<title>{t}</title>
|
||||||
|
<description>Research digest for {d}</description>
|
||||||
|
<pubDate>{p}</pubDate>
|
||||||
|
<guid isPermaLink="false">{id}</guid>
|
||||||
|
<enclosure url="{u}" length="{len}" type="audio/mpeg"/>
|
||||||
|
<itunes:duration>{secs}</itunes:duration>
|
||||||
|
</item>
|
||||||
|
"#,
|
||||||
|
t = xml_escape(&title),
|
||||||
|
d = xml_escape(&date),
|
||||||
|
p = rfc2822(created),
|
||||||
|
u = xml_escape(&url),
|
||||||
|
len = bytes,
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
let xml = format!(
|
||||||
|
r#"<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd">
|
||||||
|
<channel>
|
||||||
|
<title>ClawMates Research</title>
|
||||||
|
<link>{base}</link>
|
||||||
|
<description>Papers read against your projects, every morning.</description>
|
||||||
|
<language>en-us</language>
|
||||||
|
<itunes:explicit>false</itunes:explicit>
|
||||||
|
{items} </channel>
|
||||||
|
</rss>
|
||||||
|
"#
|
||||||
|
);
|
||||||
|
Ok((
|
||||||
|
StatusCode::OK,
|
||||||
|
[(header::CONTENT_TYPE, "application/rss+xml; charset=utf-8")],
|
||||||
|
xml,
|
||||||
|
)
|
||||||
|
.into_response())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `GET /api/podcast/episodes/{id}.mp3?token=…`
|
||||||
|
pub async fn episode_audio(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
Path(file): Path<String>,
|
||||||
|
Query(auth): Query<OptionalAuth>,
|
||||||
|
headers: axum::http::HeaderMap,
|
||||||
|
) -> Result<Response, ApiError> {
|
||||||
|
// A podcast app fetches this with the token in the URL, because it cannot
|
||||||
|
// set headers. The browser plays it through the same-origin proxy, which
|
||||||
|
// supplies a bearer and no query token. Both are the same session; refusing
|
||||||
|
// either would break one of the two ways this is listened to.
|
||||||
|
let token = auth
|
||||||
|
.token
|
||||||
|
.or_else(|| {
|
||||||
|
headers
|
||||||
|
.get(axum::http::header::AUTHORIZATION)
|
||||||
|
.and_then(|v| v.to_str().ok())
|
||||||
|
.and_then(|v| v.strip_prefix("Bearer "))
|
||||||
|
.map(str::to_string)
|
||||||
|
})
|
||||||
|
.ok_or(ApiError::Unauthorized)?;
|
||||||
|
let workspace_id = workspace_for(&state, &token).await?;
|
||||||
|
let id = file
|
||||||
|
.strip_suffix(".mp3")
|
||||||
|
.and_then(|s| uuid::Uuid::parse_str(s).ok())
|
||||||
|
.ok_or(ApiError::NotFound)?;
|
||||||
|
|
||||||
|
let row = sqlx::query(
|
||||||
|
"SELECT blob_key, bytes FROM podcast_episodes WHERE id = $1 AND workspace_id = $2",
|
||||||
|
)
|
||||||
|
.bind(id)
|
||||||
|
.bind(workspace_id)
|
||||||
|
.fetch_optional(&state.pool)
|
||||||
|
.await?
|
||||||
|
.ok_or(ApiError::NotFound)?;
|
||||||
|
|
||||||
|
let key: String = row.get("blob_key");
|
||||||
|
let blobs = state.blobs.clone().ok_or(ApiError::Internal)?;
|
||||||
|
let bytes = blobs.get(&key).await.map_err(|e| {
|
||||||
|
eprintln!("podcast: reading {key}: {e}");
|
||||||
|
ApiError::Internal
|
||||||
|
})?;
|
||||||
|
|
||||||
|
Ok((
|
||||||
|
StatusCode::OK,
|
||||||
|
[
|
||||||
|
(header::CONTENT_TYPE, "audio/mpeg".to_string()),
|
||||||
|
(header::CONTENT_LENGTH, bytes.len().to_string()),
|
||||||
|
// Podcast apps re-fetch on every refresh otherwise.
|
||||||
|
(header::CACHE_CONTROL, "private, max-age=86400".to_string()),
|
||||||
|
],
|
||||||
|
bytes,
|
||||||
|
)
|
||||||
|
.into_response())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `GET /api/podcast/subscription` — the URL to paste into a podcast app.
|
||||||
|
///
|
||||||
|
/// Minted here rather than in the browser because the session lives in an
|
||||||
|
/// httpOnly cookie that JavaScript cannot read, and the same-origin proxy that
|
||||||
|
/// normally supplies the bearer is not available to a podcast app on a phone.
|
||||||
|
/// So the caller's own token is echoed back inside a URL that points DIRECTLY
|
||||||
|
/// at this backend.
|
||||||
|
pub async fn subscription(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
headers: axum::http::HeaderMap,
|
||||||
|
crate::extract::Authed(_user): crate::extract::Authed,
|
||||||
|
) -> Result<axum::Json<serde_json::Value>, ApiError> {
|
||||||
|
let token = headers
|
||||||
|
.get(axum::http::header::AUTHORIZATION)
|
||||||
|
.and_then(|v| v.to_str().ok())
|
||||||
|
.and_then(|v| v.strip_prefix("Bearer "))
|
||||||
|
.ok_or(ApiError::Unauthorized)?;
|
||||||
|
|
||||||
|
let base = std::env::var("CLAWMATES_PUBLIC_URL")
|
||||||
|
.unwrap_or_else(|_| "http://localhost:8080".to_string());
|
||||||
|
let base = base.trim_end_matches('/');
|
||||||
|
let _ = &state;
|
||||||
|
Ok(axum::Json(serde_json::json!({
|
||||||
|
"feedUrl": format!("{base}/api/podcast/feed.xml?token={token}"),
|
||||||
|
// The panel warns when this is still localhost: a phone cannot reach it,
|
||||||
|
// and a feed that only works on the machine that made it is a feed that
|
||||||
|
// silently never syncs.
|
||||||
|
"reachable": !base.contains("localhost") && !base.contains("127.0.0.1"),
|
||||||
|
})))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `GET /api/podcast/episodes` — the list behind the UI panel.
|
||||||
|
///
|
||||||
|
/// Normal bearer auth, unlike the feed: this is the app talking to its own API,
|
||||||
|
/// where a header is available and a token in a URL would be needless exposure.
|
||||||
|
pub async fn list_episodes(
|
||||||
|
State(state): State<AppState>,
|
||||||
|
crate::extract::Authed(user): crate::extract::Authed,
|
||||||
|
) -> Result<axum::Json<serde_json::Value>, ApiError> {
|
||||||
|
let rows = sqlx::query(
|
||||||
|
"SELECT e.id, e.episode_date, e.title, e.bytes, e.duration_secs,
|
||||||
|
e.rendered_by, e.created_at, e.mission_id, m.title AS mission_title
|
||||||
|
FROM podcast_episodes e
|
||||||
|
-- LEFT: an episode outlives its mission (migration 0080). An inner
|
||||||
|
-- join would hide exactly the back-catalogue that change protects.
|
||||||
|
LEFT JOIN missions m ON m.id = e.mission_id
|
||||||
|
WHERE e.workspace_id = $1 AND e.bytes > 0
|
||||||
|
ORDER BY e.created_at DESC
|
||||||
|
LIMIT 50",
|
||||||
|
)
|
||||||
|
.bind(user.workspace_id.as_uuid())
|
||||||
|
.fetch_all(&state.pool)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let episodes: Vec<serde_json::Value> = rows
|
||||||
|
.iter()
|
||||||
|
.map(|r| {
|
||||||
|
let secs: i32 = r.get("duration_secs");
|
||||||
|
let created: time::OffsetDateTime = r.get("created_at");
|
||||||
|
serde_json::json!({
|
||||||
|
"id": r.get::<uuid::Uuid, _>("id"),
|
||||||
|
"missionId": r.get::<Option<uuid::Uuid>, _>("mission_id"),
|
||||||
|
"missionTitle": r
|
||||||
|
.get::<Option<String>, _>("mission_title")
|
||||||
|
.unwrap_or_else(|| "(mission deleted)".to_string()),
|
||||||
|
"title": r.get::<String, _>("title"),
|
||||||
|
"date": r.get::<String, _>("episode_date"),
|
||||||
|
"bytes": r.get::<i64, _>("bytes"),
|
||||||
|
"durationSecs": secs,
|
||||||
|
"renderedBy": r.get::<String, _>("rendered_by"),
|
||||||
|
"createdAt": created.unix_timestamp(),
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
// How many missions produced no audio, so the panel can say so rather than
|
||||||
|
// leaving a silent gap the operator has to notice for themselves.
|
||||||
|
let unrenderable: i64 = sqlx::query_scalar(
|
||||||
|
"SELECT count(*) FROM podcast_episodes WHERE workspace_id = $1 AND bytes = 0",
|
||||||
|
)
|
||||||
|
.bind(user.workspace_id.as_uuid())
|
||||||
|
.fetch_one(&state.pool)
|
||||||
|
.await
|
||||||
|
.unwrap_or(0);
|
||||||
|
|
||||||
|
Ok(axum::Json(serde_json::json!({
|
||||||
|
"episodes": episodes,
|
||||||
|
"unrenderable": unrenderable,
|
||||||
|
})))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// A title with an ampersand must not produce invalid XML — a single bad
|
||||||
|
/// character makes a podcast app reject the WHOLE feed, not one episode.
|
||||||
|
#[test]
|
||||||
|
fn titles_are_xml_escaped() {
|
||||||
|
let out = xml_escape(r#"BM25 & <dense> "hybrid""#);
|
||||||
|
assert_eq!(out, "BM25 & <dense> "hybrid"");
|
||||||
|
assert!(!out.contains(" & "), "raw ampersand breaks the feed");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pubdate_is_rfc2822() {
|
||||||
|
let t = time::OffsetDateTime::from_unix_timestamp(1_755_000_000).unwrap();
|
||||||
|
let s = rfc2822(t);
|
||||||
|
// "Mon, 12 Aug 2025 ..." — clients parse this strictly.
|
||||||
|
assert!(s.contains(", "), "{s}");
|
||||||
|
assert!(s.ends_with("+0000"), "{s}");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -430,13 +430,30 @@ async fn sync_gitea(
|
|||||||
Some(owner) => format!("{api_base}/orgs/{owner}/repos?limit={per_page}&page={page}"),
|
Some(owner) => format!("{api_base}/orgs/{owner}/repos?limit={per_page}&page={page}"),
|
||||||
None => format!("{api_base}/repos/search?limit={per_page}&page={page}"),
|
None => format!("{api_base}/repos/search?limit={per_page}&page={page}"),
|
||||||
};
|
};
|
||||||
let (status, body) = broker
|
let (mut status, mut body) = broker
|
||||||
.fetch_authorized(secret_ref, &url)
|
.fetch_authorized(secret_ref, &url)
|
||||||
.await
|
.await
|
||||||
.map_err(|e| format!("broker fetch: {e}"))?;
|
.map_err(|e| format!("broker fetch: {e}"))?;
|
||||||
|
// A Gitea owner is either an ORG or a USER, and they live on different
|
||||||
|
// endpoints. Scoping a connection to a personal namespace — `osobh`,
|
||||||
|
// where clawmates itself lives — 404s on /orgs and reported "not found
|
||||||
|
// or PAT lacks access", which points at permissions when the account is
|
||||||
|
// simply not an org. Retry as a user before giving up.
|
||||||
|
if status == 404 {
|
||||||
|
if let Some(owner) = conn.owner.as_deref() {
|
||||||
|
let user_url =
|
||||||
|
format!("{api_base}/users/{owner}/repos?limit={per_page}&page={page}");
|
||||||
|
let (s2, b2) = broker
|
||||||
|
.fetch_authorized(secret_ref, &user_url)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("broker fetch: {e}"))?;
|
||||||
|
status = s2;
|
||||||
|
body = b2;
|
||||||
|
}
|
||||||
|
}
|
||||||
if status == 404 && conn.owner.is_some() {
|
if status == 404 && conn.owner.is_some() {
|
||||||
return Err(format!(
|
return Err(format!(
|
||||||
"org '{}' not found or PAT lacks access",
|
"'{}' matched neither an org nor a user, or the PAT lacks access",
|
||||||
conn.owner.as_deref().unwrap_or("")
|
conn.owner.as_deref().unwrap_or("")
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -100,7 +100,11 @@ pub async fn leaderboard(
|
|||||||
COUNT(u.id)::BIGINT AS "runs!"
|
COUNT(u.id)::BIGINT AS "runs!"
|
||||||
FROM agents a
|
FROM agents a
|
||||||
LEFT JOIN usage_events u ON u.agent_id = a.id
|
LEFT JOIN usage_events u ON u.agent_id = a.id
|
||||||
WHERE a.workspace_id = $1
|
-- deleted_at: a soft-deleted agent is gone everywhere else, so
|
||||||
|
-- listing it here made deletion look like a no-op — the operator
|
||||||
|
-- deletes it, the board still shows it, and deleting again does
|
||||||
|
-- nothing because the row is already marked.
|
||||||
|
WHERE a.workspace_id = $1 AND a.deleted_at IS NULL
|
||||||
GROUP BY a.id, a.name, a.accent
|
GROUP BY a.id, a.name, a.accent
|
||||||
ORDER BY "credits!" DESC, "tokens!" DESC, a.name"#,
|
ORDER BY "credits!" DESC, "tokens!" DESC, a.name"#,
|
||||||
user.workspace_id.as_uuid(),
|
user.workspace_id.as_uuid(),
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ use crate::{ApiError, AppState, Authed};
|
|||||||
pub struct TeamMemberInput {
|
pub struct TeamMemberInput {
|
||||||
pub role: String,
|
pub role: String,
|
||||||
pub name: String,
|
pub name: String,
|
||||||
/// Model selector: claude | glm | glm-5.2 | kimi | gemini | groq.
|
/// Model selector: claude | glm | glm-5.2 | kimi | groq.
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub model: String,
|
pub model: String,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
@@ -83,6 +83,21 @@ pub(crate) async fn build_team(
|
|||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// MCP bundles for a team built by the wizard or the planner rather than from a
|
||||||
|
/// team template.
|
||||||
|
///
|
||||||
|
/// These teams have no template, so there is no `mcp_bundles` list to inherit —
|
||||||
|
/// which previously meant they were provisioned with the door alone and could
|
||||||
|
/// not reach the skills catalogue at all. `mcp_skills` scopes what it lists to
|
||||||
|
/// the caller's workspace, so an agent with no template link still sees the
|
||||||
|
/// global skills, which is the useful half for an ad-hoc team.
|
||||||
|
fn adhoc_bundles() -> Vec<String> {
|
||||||
|
vec![
|
||||||
|
"clawmates_door".to_string(),
|
||||||
|
"clawmates_skills".to_string(),
|
||||||
|
]
|
||||||
|
}
|
||||||
|
|
||||||
/// Same as `build_team` but with an explicit `lifecycle` (`permanent` |
|
/// Same as `build_team` but with an explicit `lifecycle` (`permanent` |
|
||||||
/// `ephemeral`). Ephemeral teams are torn down by the topology_worker after
|
/// `ephemeral`). Ephemeral teams are torn down by the topology_worker after
|
||||||
/// their last run terminates — used by the Scheduled + Triggered planner modes.
|
/// their last run terminates — used by the Scheduled + Triggered planner modes.
|
||||||
@@ -140,7 +155,7 @@ pub(crate) async fn build_team_with_lifecycle(
|
|||||||
// Ad-hoc team-wizard teams aren't mission-bound, so they use the
|
// Ad-hoc team-wizard teams aren't mission-bound, so they use the
|
||||||
// default per-agent workspace under <install>/agents/<alias>/workspace/.
|
// default per-agent workspace under <install>/agents/<alias>/workspace/.
|
||||||
provisioner
|
provisioner
|
||||||
.provision_claw(claw_id, &m.model, risk)
|
.provision_claw(claw_id, &m.model, risk, &adhoc_bundles())
|
||||||
.await
|
.await
|
||||||
.map_err(|e| {
|
.map_err(|e| {
|
||||||
eprintln!("teams: provision claw {claw_id} failed: {e}");
|
eprintln!("teams: provision claw {claw_id} failed: {e}");
|
||||||
@@ -574,7 +589,7 @@ pub struct AutoProvisionRequest {
|
|||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub risk_profile: Option<String>,
|
pub risk_profile: Option<String>,
|
||||||
/// MCP bundle aliases — same fall-back rule applies (always
|
/// MCP bundle aliases — same fall-back rule applies (always
|
||||||
/// clawmates_door; gitea_forge when a repo is bound; deep-research
|
/// clawmates_door + clawmates_skills; deep-research
|
||||||
/// skill for research profiles).
|
/// skill for research profiles).
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub mcp_bundles: Vec<String>,
|
pub mcp_bundles: Vec<String>,
|
||||||
@@ -656,11 +671,13 @@ pub async fn auto_provision(
|
|||||||
let mut mcp_bundles = body.mcp_bundles.clone();
|
let mut mcp_bundles = body.mcp_bundles.clone();
|
||||||
if mcp_bundles.is_empty() {
|
if mcp_bundles.is_empty() {
|
||||||
mcp_bundles.push("clawmates_door".to_string());
|
mcp_bundles.push("clawmates_door".to_string());
|
||||||
// gitea_forge is scoped to teams that will touch repos; the
|
// No `gitea_forge`: it was named in nine places and defined in none,
|
||||||
// wizard's downstream repo-binding step is what earns it.
|
// and agents reach the forge through `git` over HTTPS with the ambient
|
||||||
// Always safe to add now — the MCP layer no-ops when the token
|
// GITEA_TOKEN (mission_workspace::with_ambient_auth) — which is why
|
||||||
// isn't present in the container env.
|
// nothing ever broke. It was harmless while provision_claw ignored the
|
||||||
mcp_bundles.push("gitea_forge".to_string());
|
// bundle list; now that the list is honoured, an undefined name is a
|
||||||
|
// capability an agent is told it has and does not.
|
||||||
|
mcp_bundles.push("clawmates_skills".to_string());
|
||||||
}
|
}
|
||||||
|
|
||||||
// 1) LLM plan pass → roster JSON.
|
// 1) LLM plan pass → roster JSON.
|
||||||
|
|||||||
@@ -109,14 +109,17 @@ pub async fn compare_topologies(
|
|||||||
Json(req): Json<CompareRequest>,
|
Json(req): Json<CompareRequest>,
|
||||||
) -> Result<Json<Comparison>, ApiError> {
|
) -> Result<Json<Comparison>, ApiError> {
|
||||||
// Execution turns run on the exec model (default = configured model, e.g.
|
// Execution turns run on the exec model (default = configured model, e.g.
|
||||||
// sonnet); the judge uses the judge model (default claude-opus-4-8). Either
|
// sonnet); the judge uses the judge model (cm_runtime::judge_model). Either
|
||||||
// can name a registry provider as "<name>:<model>" (e.g. "glm:glm-4.6",
|
// can name a registry provider as "<name>:<model>" (e.g. "glm:glm-4.6",
|
||||||
// "kimi:kimi-k2") to run on GLM/Kimi instead.
|
// "kimi:kimi-k2") to run on GLM/Kimi instead.
|
||||||
let exec_spec = std::env::var("CLAWMATES_TOPOLOGY_EXEC_MODEL")
|
let exec_spec = std::env::var("CLAWMATES_TOPOLOGY_EXEC_MODEL")
|
||||||
.unwrap_or_else(|_| state.runtime.model().to_string());
|
.unwrap_or_else(|_| state.runtime.model().to_string());
|
||||||
let (exec_provider, exec_model) = state.runtime.resolve_provider(&exec_spec);
|
let (exec_provider, exec_model) = state.runtime.resolve_provider(&exec_spec);
|
||||||
let judge_spec =
|
// `cm_runtime::judge_model()`, not a second read of the same variable: this
|
||||||
std::env::var("CLAWMATES_JUDGE_MODEL").unwrap_or_else(|_| "claude-opus-4-8".to_string());
|
// line and that function disagreed on the default (opus-4-8 vs opus-5), so
|
||||||
|
// an unconfigured deployment scored topology comparisons on a different
|
||||||
|
// model than the door governor and nothing recorded which.
|
||||||
|
let judge_spec = cm_runtime::judge_model();
|
||||||
let (judge_provider, judge_model) = state.runtime.resolve_provider(&judge_spec);
|
let (judge_provider, judge_model) = state.runtime.resolve_provider(&judge_spec);
|
||||||
let executor = ProviderExecutor::new(exec_provider, exec_model, state.runtime.max_tokens());
|
let executor = ProviderExecutor::new(exec_provider, exec_model, state.runtime.max_tokens());
|
||||||
let scorer = JudgeScorer::new(judge_provider, judge_model, 16);
|
let scorer = JudgeScorer::new(judge_provider, judge_model, 16);
|
||||||
@@ -309,6 +312,10 @@ pub async fn run_events_sse(
|
|||||||
.map(|n| n + 1)
|
.map(|n| n + 1)
|
||||||
.unwrap_or(0);
|
.unwrap_or(0);
|
||||||
|
|
||||||
|
// Bytes of `checkpoint.log` already sent. The step cursor above counts
|
||||||
|
// RECORDS; this counts BYTES, because a log grows continuously rather than
|
||||||
|
// in discrete entries. Two sources, two cursors.
|
||||||
|
let mut log_sent: usize = 0;
|
||||||
let stream = async_stream::stream! {
|
let stream = async_stream::stream! {
|
||||||
loop {
|
loop {
|
||||||
match cm_db::repo::topology_runs::status(&pool, id, ws).await {
|
match cm_db::repo::topology_runs::status(&pool, id, ws).await {
|
||||||
@@ -327,6 +334,24 @@ pub async fn run_events_sse(
|
|||||||
sent += 1;
|
sent += 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
// Live stdout/stderr from a microVM turn, appended by the
|
||||||
|
// node over the fleet WebSocket (`Uplink::VmOut`). Emitted
|
||||||
|
// as `step` so the existing reader renders it with no
|
||||||
|
// frontend change — it already reads `data.text`.
|
||||||
|
if let Some(log) = st
|
||||||
|
.checkpoint
|
||||||
|
.as_ref()
|
||||||
|
.and_then(|c| c.get("log"))
|
||||||
|
.and_then(|v| v.as_str())
|
||||||
|
{
|
||||||
|
if log.len() > log_sent {
|
||||||
|
let fresh = &log[log_sent..];
|
||||||
|
log_sent = log.len();
|
||||||
|
yield Ok::<Event, Infallible>(Event::default().event("step").data(
|
||||||
|
serde_json::json!({ "kind": "output", "text": fresh }).to_string(),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
if matches!(st.status.as_str(), "completed" | "failed" | "cancelled") {
|
if matches!(st.status.as_str(), "completed" | "failed" | "cancelled") {
|
||||||
let done = serde_json::json!({
|
let done = serde_json::json!({
|
||||||
"status": st.status,
|
"status": st.status,
|
||||||
|
|||||||
+1250
-31
File diff suppressed because it is too large
Load Diff
@@ -23,6 +23,7 @@
|
|||||||
//! from serving — it should stop us believing a scan that scanned nothing.
|
//! from serving — it should stop us believing a scan that scanned nothing.
|
||||||
|
|
||||||
use crate::container_exec;
|
use crate::container_exec;
|
||||||
|
use bollard::Docker;
|
||||||
use std::time::Duration;
|
use std::time::Duration;
|
||||||
|
|
||||||
const PROBE_TIMEOUT: Duration = Duration::from_secs(20);
|
const PROBE_TIMEOUT: Duration = Duration::from_secs(20);
|
||||||
@@ -36,6 +37,11 @@ struct Dependency {
|
|||||||
}
|
}
|
||||||
|
|
||||||
const DEPENDENCIES: &[Dependency] = &[
|
const DEPENDENCIES: &[Dependency] = &[
|
||||||
|
Dependency {
|
||||||
|
argv: &["zeroclaw", "--version"],
|
||||||
|
needed_for: "driving every container-tier turn; the version is also how \
|
||||||
|
a runtime image that silently rolled back is noticed",
|
||||||
|
},
|
||||||
Dependency {
|
Dependency {
|
||||||
argv: &["cargo", "--version"],
|
argv: &["cargo", "--version"],
|
||||||
needed_for: "the on_green_tests gate for Rust repos; without it every \
|
needed_for: "the on_green_tests gate for Rust repos; without it every \
|
||||||
@@ -113,9 +119,67 @@ pub async fn probe(container: &str) -> Result<Vec<ToolStatus>, String> {
|
|||||||
};
|
};
|
||||||
out.push(status);
|
out.push(status);
|
||||||
}
|
}
|
||||||
|
out.push(probe_mission_uid_can_write(&docker, container).await);
|
||||||
Ok(out)
|
Ok(out)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Can uid 65532 actually work in the missions tree?
|
||||||
|
///
|
||||||
|
/// `container_exec` now runs every mission exec as 65532 rather than root, so
|
||||||
|
/// that no phase leaves behind files the cleanup (which runs as 65532) cannot
|
||||||
|
/// delete. That only holds while the image gives 65532 a writable `HOME` and
|
||||||
|
/// `CARGO_HOME` — and in the deployed image its default `HOME`
|
||||||
|
/// (`/zeroclaw-data`) and `/usr/local/cargo` are BOTH root-owned, which is why
|
||||||
|
/// `container_exec::mission_env` redirects them into the missions root.
|
||||||
|
///
|
||||||
|
/// If a future image moves that mount or tightens its permissions, every cargo
|
||||||
|
/// invocation starts failing for a reason no error message would connect to a
|
||||||
|
/// uid. So it is probed at boot, alongside the tools, and reported the same way.
|
||||||
|
async fn probe_mission_uid_can_write(docker: &Docker, container: &str) -> ToolStatus {
|
||||||
|
let root = crate::mission_workspace::missions_root();
|
||||||
|
let probe = root.join("_probe-uid");
|
||||||
|
// Through `exec`, not `exec_as_root`: the point is to exercise the exact
|
||||||
|
// policy real mission work gets, including the env it is given.
|
||||||
|
let argv: Vec<String> = [
|
||||||
|
"sh",
|
||||||
|
"-c",
|
||||||
|
&format!(
|
||||||
|
"set -e; mkdir -p \"$HOME\" \"$CARGO_HOME\" {p}; : > {p}/w; rm -rf {p}; echo \"uid=$(id -u) HOME=$HOME CARGO_HOME=$CARGO_HOME\"",
|
||||||
|
p = probe.display()
|
||||||
|
),
|
||||||
|
]
|
||||||
|
.iter()
|
||||||
|
.map(|s| s.to_string())
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let detail = match container_exec::exec(
|
||||||
|
docker,
|
||||||
|
container,
|
||||||
|
Some(&root.display().to_string()),
|
||||||
|
&argv,
|
||||||
|
PROBE_TIMEOUT,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(r) if r.success() => {
|
||||||
|
return ToolStatus {
|
||||||
|
program: "mission-uid".to_string(),
|
||||||
|
present: true,
|
||||||
|
detail: r.combined().trim().chars().take(120).collect(),
|
||||||
|
needed_for: "every mission exec, so no phase leaves root-owned files",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(r) => r.combined().trim().chars().take(160).collect(),
|
||||||
|
Err(e) => e.chars().take(160).collect(),
|
||||||
|
};
|
||||||
|
ToolStatus {
|
||||||
|
program: "mission-uid".to_string(),
|
||||||
|
present: false,
|
||||||
|
detail,
|
||||||
|
needed_for: "every mission exec, so no phase leaves root-owned files",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Probe at startup and write the result to stderr.
|
/// Probe at startup and write the result to stderr.
|
||||||
///
|
///
|
||||||
/// Spawned rather than awaited so a slow or absent Docker socket cannot delay
|
/// Spawned rather than awaited so a slow or absent Docker socket cannot delay
|
||||||
@@ -134,10 +198,20 @@ pub fn report_at_boot() {
|
|||||||
let missing: Vec<&ToolStatus> = tools.iter().filter(|t| !t.present).collect();
|
let missing: Vec<&ToolStatus> = tools.iter().filter(|t| !t.present).collect();
|
||||||
if missing.is_empty() {
|
if missing.is_empty() {
|
||||||
let names: Vec<&str> = tools.iter().map(|t| t.program.as_str()).collect();
|
let names: Vec<&str> = tools.iter().map(|t| t.program.as_str()).collect();
|
||||||
|
// The VERSIONS, not just the names. A tag that quietly
|
||||||
|
// points at an older build passes a presence check
|
||||||
|
// perfectly: gw-04's default tag was two zeroclaw releases
|
||||||
|
// behind while every probe said "present", and the only way
|
||||||
|
// anyone found out was running the binary by hand.
|
||||||
|
let detail: Vec<String> = tools
|
||||||
|
.iter()
|
||||||
|
.map(|t| format!("{}={}", t.program, t.detail))
|
||||||
|
.collect();
|
||||||
eprintln!(
|
eprintln!(
|
||||||
"runtime_preflight: `{container}` has all {} expected tools ({})",
|
"runtime_preflight: `{container}` has all {} expected tools ({}) — {}",
|
||||||
tools.len(),
|
tools.len(),
|
||||||
names.join(", ")
|
names.join(", "),
|
||||||
|
detail.join("; ")
|
||||||
);
|
);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -14,26 +14,68 @@
|
|||||||
use uuid::Uuid;
|
use uuid::Uuid;
|
||||||
|
|
||||||
/// The runtime agent alias for a claw id.
|
/// The runtime agent alias for a claw id.
|
||||||
|
/// The bundles an agent is provisioned with: whatever the template asked for,
|
||||||
|
/// plus `clawmates_door`, always.
|
||||||
|
///
|
||||||
|
/// The door is not optional. It carries the §15 approval gate, so an agent
|
||||||
|
/// provisioned without it is not a restricted agent, it is an ungated one —
|
||||||
|
/// and a template that simply forgot to list it would silently get that.
|
||||||
|
fn with_door(bundles: &[String]) -> Vec<String> {
|
||||||
|
let mut out: Vec<String> = Vec::new();
|
||||||
|
out.push("clawmates_door".to_string());
|
||||||
|
for b in bundles {
|
||||||
|
let b = b.trim();
|
||||||
|
if !b.is_empty() && !out.iter().any(|x| x == b) {
|
||||||
|
out.push(b.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
pub fn claw_alias(claw_id: Uuid) -> String {
|
pub fn claw_alias(claw_id: Uuid) -> String {
|
||||||
format!("claw_{}", claw_id.simple())
|
format!("claw_{}", claw_id.simple())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The claw behind a runtime alias, or `None` if it is not one of ours.
|
||||||
|
///
|
||||||
|
/// The inverse of [`claw_alias`], and it lives beside it so the two cannot
|
||||||
|
/// drift — a changed prefix breaks the round-trip test rather than quietly
|
||||||
|
/// returning `None` for every agent and dropping their attribution.
|
||||||
|
///
|
||||||
|
/// `None` is the honest answer for `scout` and the other configured aliases
|
||||||
|
/// that are not claws: they have no row in `agents` to point at.
|
||||||
|
pub fn claw_from_alias(alias: &str) -> Option<Uuid> {
|
||||||
|
Uuid::parse_str(alias.trim().strip_prefix("claw_")?).ok()
|
||||||
|
}
|
||||||
|
|
||||||
/// Map a claw's chosen model to a configured provider alias.
|
/// Map a claw's chosen model to a configured provider alias.
|
||||||
///
|
///
|
||||||
/// Claude models resolve to `claude_cli.default`, which spawns the real
|
/// Claude models resolve to `claude_cli.default`, which spawns the real
|
||||||
/// `claude` binary against the Max subscription rather than posting to the
|
/// `claude` binary against the Max subscription rather than posting to the
|
||||||
/// raw API with Claude Code identity headers. The API-key path still exists
|
/// raw API with Claude Code identity headers. Agent work — ~99% of the
|
||||||
/// and the judge uses it deliberately (see below), but agent work — which is
|
/// tokens — belongs on the subscription and on the supported client.
|
||||||
/// ~99% of the tokens — belongs on the subscription and on the supported
|
|
||||||
/// client.
|
|
||||||
///
|
///
|
||||||
/// The judge stays on `anthropic.judge`/API key on purpose: if the
|
/// **The API-key path is gone.** `anthropic.default` and `anthropic.judge`
|
||||||
/// subscription throttles, missions degrade but verification keeps working.
|
/// were retired from the runtime config on 2026-08-10: both held `sk-ant-api`
|
||||||
/// Putting both on one credential would mean a single limit blinds the
|
/// keys on an account whose balance is zero, which the real code path reports
|
||||||
/// verifier at exactly the moment there is most to verify.
|
/// as `400 … "Your credit balance is too low"`. Every agent that named them
|
||||||
|
/// was repointed onto a live credential.
|
||||||
///
|
///
|
||||||
/// Non-Claude families are unchanged: `groq.default`, `gemini.default`, and
|
/// The independence argument that put the judge there still holds — a
|
||||||
/// the GLM/Kimi substitution below.
|
/// verifier sharing one credential with the implementer goes blind at exactly
|
||||||
|
/// the moment there is most to verify — but it is now served by a different
|
||||||
|
/// FAMILY rather than a different key: the validator runs on
|
||||||
|
/// `CLAWMATES_VALIDATOR_MODEL` (`glm:glm-4.7` on gw-04) while agents run on
|
||||||
|
/// the subscription, and `cross_provider_judge` refuses a validator in the
|
||||||
|
/// implementer's own family. `claude_cli.default` also carries
|
||||||
|
/// `fallback = ["claude_cli.kimi", "claude_cli.glm"]`, so a throttle degrades
|
||||||
|
/// across credentials instead of stopping.
|
||||||
|
///
|
||||||
|
/// Non-Claude families are unchanged: `groq.default` and the GLM/Kimi
|
||||||
|
/// substitution below. Gemini was removed entirely — a `gemini*` model now
|
||||||
|
/// falls through to the unrecognised branch, which LOGS and defaults to
|
||||||
|
/// `claude_cli.default` rather than silently routing to a provider we no
|
||||||
|
/// longer configure.
|
||||||
pub fn provider_alias_for(model: &str) -> &'static str {
|
pub fn provider_alias_for(model: &str) -> &'static str {
|
||||||
let m = model.trim().to_ascii_lowercase();
|
let m = model.trim().to_ascii_lowercase();
|
||||||
// Prefix families first (covers claude-sonnet-5, claude-opus-4-8,
|
// Prefix families first (covers claude-sonnet-5, claude-opus-4-8,
|
||||||
@@ -43,9 +85,6 @@ pub fn provider_alias_for(model: &str) -> &'static str {
|
|||||||
if m.starts_with("claude") {
|
if m.starts_with("claude") {
|
||||||
return "claude_cli.default";
|
return "claude_cli.default";
|
||||||
}
|
}
|
||||||
if m.starts_with("gemini") {
|
|
||||||
return "gemini.default";
|
|
||||||
}
|
|
||||||
return "groq.default";
|
return "groq.default";
|
||||||
}
|
}
|
||||||
match m.as_str() {
|
match m.as_str() {
|
||||||
@@ -88,7 +127,6 @@ pub fn provider_alias_for(model: &str) -> &'static str {
|
|||||||
pub fn is_exact_provider_match(model: &str) -> bool {
|
pub fn is_exact_provider_match(model: &str) -> bool {
|
||||||
let m = model.trim().to_ascii_lowercase();
|
let m = model.trim().to_ascii_lowercase();
|
||||||
m.starts_with("claude")
|
m.starts_with("claude")
|
||||||
|| m.starts_with("gemini")
|
|
||||||
|| m.starts_with("llama")
|
|| m.starts_with("llama")
|
||||||
|| m.starts_with("groq")
|
|| m.starts_with("groq")
|
||||||
}
|
}
|
||||||
@@ -148,6 +186,36 @@ impl RuntimeProvisioner {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Point `claude_cli.default` at a settings document, so the hooks written
|
||||||
|
/// into the container are actually read.
|
||||||
|
///
|
||||||
|
/// Without this the gate and the tap exist on disk and claude never loads
|
||||||
|
/// them — installed, inert, and indistinguishable from working. The alias
|
||||||
|
/// is `claude_cli.default` because that is what `provider_alias_for` binds
|
||||||
|
/// every claude model to.
|
||||||
|
pub async fn set_claude_cli_settings(&self, path: &str) -> Result<(), String> {
|
||||||
|
self.set_prop(
|
||||||
|
"providers.models.claude_cli.default.settings",
|
||||||
|
serde_json::json!(path),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Point `claude -p` at an MCP configuration.
|
||||||
|
///
|
||||||
|
/// The counterpart to [`set_claude_cli_settings`](Self::set_claude_cli_settings):
|
||||||
|
/// writing the document into the container and telling the daemon about it
|
||||||
|
/// are two halves of one thing, and doing one without the other leaves a
|
||||||
|
/// door that is installed and unreachable — which looks exactly like a door
|
||||||
|
/// nobody walked through.
|
||||||
|
pub async fn set_claude_cli_mcp_config(&self, path: &str) -> Result<(), String> {
|
||||||
|
self.set_prop(
|
||||||
|
"providers.models.claude_cli.default.mcp_config",
|
||||||
|
serde_json::json!(path),
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
/// Rebind an existing claw's model without touching its risk_profile
|
/// Rebind an existing claw's model without touching its risk_profile
|
||||||
/// or mcp_bundles. Used by the "change model" UI on the Agents page
|
/// or mcp_bundles. Used by the "change model" UI on the Agents page
|
||||||
/// so we don't accidentally demote a coding_readwrite claw back to
|
/// so we don't accidentally demote a coding_readwrite claw back to
|
||||||
@@ -212,8 +280,21 @@ impl RuntimeProvisioner {
|
|||||||
/// agent gets: `toolfree` = nothing, `research_readonly` = file_read +
|
/// agent gets: `toolfree` = nothing, `research_readonly` = file_read +
|
||||||
/// content_search + glob_search, `coding_readwrite` = adds file_edit +
|
/// content_search + glob_search, `coding_readwrite` = adds file_edit +
|
||||||
/// git_operations + shell, etc.; see the `[risk_profiles.*]` allowlists
|
/// git_operations + shell, etc.; see the `[risk_profiles.*]` allowlists
|
||||||
/// in `deploy/clawmates-runtime/agent.config.example.toml`), and the
|
/// in `deploy/clawmates-runtime/agent.config.example.toml`), and the MCP
|
||||||
/// `clawmates_door` MCP bundle.
|
/// bundles the team template asked for.
|
||||||
|
///
|
||||||
|
/// `bundles` used to be the constant `["clawmates_door"]`, which is how
|
||||||
|
/// every skill in the catalogue became unreachable from a mission. The
|
||||||
|
/// skills are delivered by ONE channel — the `clawmates_skills` MCP server
|
||||||
|
/// (`mcp_skills.rs`) — a template that does not receive that bundle cannot
|
||||||
|
/// list or read a single skill, and 5 of 11 templates ask for it. Two
|
||||||
|
/// separate doc comments in `cm-runtime` describe the mission path as
|
||||||
|
/// already having this, which is why nobody looked: the belief was written
|
||||||
|
/// down twice and checked zero times.
|
||||||
|
///
|
||||||
|
/// `clawmates_door` is always included regardless of what is passed. It
|
||||||
|
/// carries the §15 approval gate, and an agent provisioned without it does
|
||||||
|
/// not become safer, it becomes ungated.
|
||||||
///
|
///
|
||||||
/// NOTE ON WORKSPACE PINNING: `[agents.<alias>.workspace.path]` is an
|
/// NOTE ON WORKSPACE PINNING: `[agents.<alias>.workspace.path]` is an
|
||||||
/// `Option<PathBuf>` field that the ZeroClaw config prop-schema does NOT
|
/// `Option<PathBuf>` field that the ZeroClaw config prop-schema does NOT
|
||||||
@@ -231,6 +312,7 @@ impl RuntimeProvisioner {
|
|||||||
claw_id: Uuid,
|
claw_id: Uuid,
|
||||||
model: &str,
|
model: &str,
|
||||||
risk_profile: &str,
|
risk_profile: &str,
|
||||||
|
bundles: &[String],
|
||||||
) -> Result<String, String> {
|
) -> Result<String, String> {
|
||||||
let alias = claw_alias(claw_id);
|
let alias = claw_alias(claw_id);
|
||||||
let model_alias = provider_alias_for(model);
|
let model_alias = provider_alias_for(model);
|
||||||
@@ -265,7 +347,7 @@ impl RuntimeProvisioner {
|
|||||||
.await?;
|
.await?;
|
||||||
self.set_prop(
|
self.set_prop(
|
||||||
&format!("agents.{alias}.mcp_bundles"),
|
&format!("agents.{alias}.mcp_bundles"),
|
||||||
serde_json::json!(["clawmates_door"]),
|
serde_json::json!(with_door(bundles)),
|
||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
@@ -363,7 +445,6 @@ mod tests {
|
|||||||
}
|
}
|
||||||
for m in [
|
for m in [
|
||||||
"claude-sonnet-5",
|
"claude-sonnet-5",
|
||||||
"gemini-2.5-flash",
|
|
||||||
"groq-llama",
|
"groq-llama",
|
||||||
"llama3",
|
"llama3",
|
||||||
] {
|
] {
|
||||||
@@ -402,8 +483,11 @@ mod tests {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn provider_alias_mapping() {
|
fn provider_alias_mapping() {
|
||||||
assert_eq!(provider_alias_for("gemini"), "gemini.default");
|
// Gemini is gone: no provider row, so it must land on the logged
|
||||||
assert_eq!(provider_alias_for("gemini-2.0-flash"), "gemini.default");
|
// default rather than a family alias that resolves to nothing.
|
||||||
|
assert_eq!(provider_alias_for("gemini"), "claude_cli.default");
|
||||||
|
assert_eq!(provider_alias_for("gemini-2.0-flash"), "claude_cli.default");
|
||||||
|
assert!(!is_exact_provider_match("gemini-2.5-flash"));
|
||||||
// glm/kimi families fall back to Claude until their own provider
|
// glm/kimi families fall back to Claude until their own provider
|
||||||
// tables are configured in the runtime template.
|
// tables are configured in the runtime template.
|
||||||
assert_eq!(provider_alias_for("GLM-4.7"), "claude_cli.default");
|
assert_eq!(provider_alias_for("GLM-4.7"), "claude_cli.default");
|
||||||
@@ -425,4 +509,19 @@ mod tests {
|
|||||||
let id = Uuid::nil();
|
let id = Uuid::nil();
|
||||||
assert_eq!(claw_alias(id), "claw_00000000000000000000000000000000");
|
assert_eq!(claw_alias(id), "claw_00000000000000000000000000000000");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The alias must round-trip, and must NOT invent a claw for one of the
|
||||||
|
/// configured non-claw aliases.
|
||||||
|
///
|
||||||
|
/// The failure this guards is silent both ways: a broken round-trip drops
|
||||||
|
/// every tool call's agent attribution (files appear, nobody moves), and a
|
||||||
|
/// too-eager parse would attribute work to a claw id that matches no row.
|
||||||
|
#[test]
|
||||||
|
fn an_alias_round_trips_to_its_claw_and_nothing_else_does() {
|
||||||
|
let id = Uuid::from_u128(0x0198_2f11_7ac0_7d51_9c3e_44a1_09b2_5e77);
|
||||||
|
assert_eq!(claw_from_alias(&claw_alias(id)), Some(id));
|
||||||
|
assert_eq!(claw_from_alias("scout"), None);
|
||||||
|
assert_eq!(claw_from_alias("claude_cli.default"), None);
|
||||||
|
assert_eq!(claw_from_alias("claw_not-a-uuid"), None);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -36,6 +36,9 @@ use uuid::Uuid;
|
|||||||
|
|
||||||
use cm_db::repo::missions::UpsertTask;
|
use cm_db::repo::missions::UpsertTask;
|
||||||
|
|
||||||
|
/// `external_id` of the marker row proving a scan ran against a phase.
|
||||||
|
pub const SCAN_MARKER: &str = "security_scan:complete";
|
||||||
|
|
||||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
pub struct Finding {
|
pub struct Finding {
|
||||||
pub external_id: String,
|
pub external_id: String,
|
||||||
@@ -92,6 +95,32 @@ pub async fn run(pool: &PgPool, mission_id: Uuid, phase_id: Uuid) -> Result<usiz
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A completion marker, always written — including for a scan that found
|
||||||
|
// nothing. Without it "we scanned and the repo is clean" and "no scan ever
|
||||||
|
// ran" are both zero rows, and the sweep in `phase_runner` that fires this
|
||||||
|
// would have no way to tell whether it had already run: a clean phase would
|
||||||
|
// be rescanned on every tick, forever. It is also the answer to the
|
||||||
|
// question an operator actually asks, which is not "how many findings"
|
||||||
|
// but "was this looked at, by what, and when".
|
||||||
|
let scanned_with = tools.join(", ");
|
||||||
|
cm_db::repo::missions::upsert_task(
|
||||||
|
pool,
|
||||||
|
UpsertTask {
|
||||||
|
mission_id,
|
||||||
|
phase_id,
|
||||||
|
external_id: SCAN_MARKER,
|
||||||
|
title: &format!(
|
||||||
|
"security scan complete — ran [{scanned_with}], {} finding(s)",
|
||||||
|
all_findings.len()
|
||||||
|
),
|
||||||
|
assigned_agent_id: None,
|
||||||
|
status: "created",
|
||||||
|
run_id: None,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("upsert scan marker: {e}"))?;
|
||||||
|
|
||||||
for f in &all_findings {
|
for f in &all_findings {
|
||||||
cm_db::repo::missions::upsert_task(
|
cm_db::repo::missions::upsert_task(
|
||||||
pool,
|
pool,
|
||||||
@@ -314,9 +343,7 @@ async fn exec_target(pool: &PgPool, mission_id: Uuid) -> Result<(String, PathBuf
|
|||||||
}
|
}
|
||||||
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
let container = std::env::var("CLAWMATES_RUNTIME_CONTAINER")
|
||||||
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
.unwrap_or_else(|_| "clawmates-runtime".to_string());
|
||||||
let root = std::env::var("CLAWMATES_MISSIONS_ROOT")
|
let workdir = crate::mission_workspace::missions_root()
|
||||||
.unwrap_or_else(|_| "/var/lib/clawmates-missions".to_string());
|
|
||||||
let workdir = PathBuf::from(root)
|
|
||||||
.join(mission_id.to_string())
|
.join(mission_id.to_string())
|
||||||
.join("repo");
|
.join("repo");
|
||||||
Ok((container, workdir))
|
Ok((container, workdir))
|
||||||
|
|||||||
@@ -0,0 +1,520 @@
|
|||||||
|
//! How a mission agent receives the skills bound to it.
|
||||||
|
//!
|
||||||
|
//! Two arms, and this module exists to hold them side by side rather than to
|
||||||
|
//! replace one with the other:
|
||||||
|
//!
|
||||||
|
//! - [`Mode::Inline`] — every pinned skill's full body is appended to the turn
|
||||||
|
//! prompt. What production has always done.
|
||||||
|
//! - [`Mode::Index`] — the prompt carries each skill's name, description and
|
||||||
|
//! `when_to_use` plus the URI that returns its body, and the agent fetches
|
||||||
|
//! the ones it judges relevant through the MCP door.
|
||||||
|
//! - [`Mode::Files`] — the same entry with a file path where the URI was; the
|
||||||
|
//! bodies are written into the container and the agent `Read`s them. Added
|
||||||
|
//! after `Index` measured 1 retrieval in 9 across three matched runs — see
|
||||||
|
//! [`FILES_PREAMBLE`] for why.
|
||||||
|
//!
|
||||||
|
//! # Why this is an A/B and not a switch
|
||||||
|
//!
|
||||||
|
//! Trigger — did the agent reach for the skill when it applied? — is
|
||||||
|
//! unmeasurable under `Inline` by construction. Nothing was reached for; the
|
||||||
|
//! text was handed over. `skill_use` reports `NotObservable` for exactly that
|
||||||
|
//! reason, and it is right to.
|
||||||
|
//!
|
||||||
|
//! `Index` makes Trigger observable, because retrieval is a recorded
|
||||||
|
//! `ReadMcpResourceTool` call. But it can only *cost* Compliance: under
|
||||||
|
//! `Inline` the procedure is in front of the model whether or not it noticed
|
||||||
|
//! it applied, and under `Index` a missed judgement means the body is never
|
||||||
|
//! read at all. Trading a measured axis for an unmeasured regression in
|
||||||
|
//! another is not an improvement, so the arm is selected per mission and
|
||||||
|
//! recorded on the mission row, and both arms stay runnable.
|
||||||
|
//!
|
||||||
|
//! # `Index` requires the door, and degrades rather than lying
|
||||||
|
//!
|
||||||
|
//! An index names a body and tells the agent how to fetch it. If the
|
||||||
|
//! `clawmates_skills` MCP server is not reachable from the container, that is
|
||||||
|
//! an index of procedures the agent cannot obtain — strictly worse than
|
||||||
|
//! `Inline`, and it fails as an agent that ignored its skills rather than as a
|
||||||
|
//! missing config. [`resolve`] therefore takes the door's install result and
|
||||||
|
//! refuses `Index` without it. This is the same failure the old
|
||||||
|
//! `pinned_skills_text` doc comment warned about; what changed is that the
|
||||||
|
//! door now exists, not that the warning stopped applying.
|
||||||
|
|
||||||
|
/// Where the index tells agents to fetch a skill body from.
|
||||||
|
///
|
||||||
|
/// Must match the server name in
|
||||||
|
/// [`crate::container_tool_hooks::mcp_document`] — the agent passes it
|
||||||
|
/// straight to `ReadMcpResourceTool`.
|
||||||
|
pub const MCP_SERVER: &str = "clawmates_skills";
|
||||||
|
|
||||||
|
/// Selects the arm. Unset means [`DEFAULT`]; unrecognised means [`Mode::Inline`].
|
||||||
|
pub const ENV_VAR: &str = "CLAWMATES_SKILL_DELIVERY";
|
||||||
|
|
||||||
|
/// The arm a deployment runs when nothing selects one.
|
||||||
|
///
|
||||||
|
/// `Files` since 2026-09-13. It was `Inline` — the control arm of an A/B has
|
||||||
|
/// to be the thing already running — until the A/B produced its answer: the
|
||||||
|
/// MCP-door arm retrieved 1 skill in 9 across three matched production runs,
|
||||||
|
/// and the file arm retrieved 3 of 3 on the fourth (`01a098dd`), with the
|
||||||
|
/// judge loop closing on the same run. That is a signal and not a rate, but
|
||||||
|
/// 0, 1, 0 → 3 on an otherwise identical task is not noise, and a default that
|
||||||
|
/// hands agents procedures they demonstrably read beats one that hands them
|
||||||
|
/// bodies they were never asked to look for.
|
||||||
|
///
|
||||||
|
/// A code default and not an env var on one server, because a setting that
|
||||||
|
/// exists only in one deployment is a setting nobody can find — the exact
|
||||||
|
/// shape `always_inject` had before it moved into the skill files.
|
||||||
|
pub const DEFAULT: Mode = Mode::Files;
|
||||||
|
|
||||||
|
/// The `# Your skills` preamble under [`Mode::Inline`].
|
||||||
|
///
|
||||||
|
/// **Byte-identical to what production has always sent.** The A arm of an A/B
|
||||||
|
/// has to be the thing already running, or the comparison measures this edit
|
||||||
|
/// as well as the change under test.
|
||||||
|
pub const INLINE_PREAMBLE: &str = "These are procedures you are expected to follow for \
|
||||||
|
this kind of work. Where one applies to what you are about to do, follow it.";
|
||||||
|
|
||||||
|
/// The `# Your skills` preamble under [`Mode::Index`] as first shipped.
|
||||||
|
///
|
||||||
|
/// Kept because [`mode_in_prompt`] reads the arm off a RECORDED prompt, and
|
||||||
|
/// prompts composed before the tool-loading sentence was added are still being
|
||||||
|
/// scored — `retain_events_until` holds them for 90 days. Dropping this
|
||||||
|
/// constant would silently re-label every stored `index` run as `inline` and
|
||||||
|
/// report Trigger against the wrong arm.
|
||||||
|
///
|
||||||
|
/// Never send this one. It is a reader, not a writer.
|
||||||
|
pub const INDEX_PREAMBLE_V1: &str = "These procedures are AVAILABLE to you; their bodies are \
|
||||||
|
not included below. Each entry names one, says when it applies, and gives the uri that \
|
||||||
|
returns it. Where an entry applies to what you are about to do, read it FIRST and then \
|
||||||
|
follow it.";
|
||||||
|
|
||||||
|
/// The `# Your skills` preamble under [`Mode::Index`].
|
||||||
|
///
|
||||||
|
/// Written and matched in one place ([`mode_in_prompt`]) so the reader cannot
|
||||||
|
/// drift from the writer — the same rule `SKILL_MARKER` is under, and for the
|
||||||
|
/// same reason: a scorer that misreads the arm reports the wrong axis.
|
||||||
|
///
|
||||||
|
/// # Why the last sentence exists
|
||||||
|
///
|
||||||
|
/// `ReadMcpResourceTool` is a DEFERRED tool: it is not on the agent's default
|
||||||
|
/// tool list and cannot be called until `ToolSearch` loads its schema. Naming
|
||||||
|
/// it — which [`READ_IT`] already did — is therefore not enough, and the
|
||||||
|
/// difference is measurable. Prod mission `01a07812` made 76 tool calls,
|
||||||
|
/// searched for two other tools, never searched for this one, and retrieved
|
||||||
|
/// ZERO skills. `01a0842e`, same recipe and same offered uris, ran
|
||||||
|
/// `ToolSearch(select:ReadMcpResourceTool)` and then fetched. One agent worked
|
||||||
|
/// the extra step out on its own; the other did not, and a capability that
|
||||||
|
/// depends on the model guessing that a tool is loadable is not delivered.
|
||||||
|
pub const INDEX_PREAMBLE: &str = "These procedures are AVAILABLE to you; their bodies are \
|
||||||
|
not included below. Each entry names one, says when it applies, and gives the uri that \
|
||||||
|
returns it. Where an entry applies to what you are about to do, read it FIRST and then \
|
||||||
|
follow it. ReadMcpResourceTool may not be loaded in this session: if you do not already \
|
||||||
|
have it, run ToolSearch with the query select:ReadMcpResourceTool before your first read.";
|
||||||
|
|
||||||
|
/// Where the `files` arm puts skill bodies inside the mission container.
|
||||||
|
///
|
||||||
|
/// Under `/mission` because that is the one directory every container-tier
|
||||||
|
/// mission has ([`crate::mission_fs::CONTAINER_MISSION_DIR`]), and beside
|
||||||
|
/// `repo/` rather than inside it so a skill never shows up in a diff or a
|
||||||
|
/// delivery.
|
||||||
|
pub const SKILLS_DIR: &str = "/mission/skills";
|
||||||
|
|
||||||
|
/// The file a skill's body is written to under the `files` arm, and the path
|
||||||
|
/// the index entry tells the agent to `Read`. One function for both, so the
|
||||||
|
/// writer and the reader cannot spell it differently.
|
||||||
|
pub fn skill_file_path(name: &str) -> String {
|
||||||
|
format!("{SKILLS_DIR}/{name}.md")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The skill a `Read` of this path is a retrieval of, if it is one.
|
||||||
|
///
|
||||||
|
/// The scorer's half of [`skill_file_path`]. Anything outside [`SKILLS_DIR`]
|
||||||
|
/// is an ordinary file read and returns `None`.
|
||||||
|
pub fn skill_from_file_path(path: &str) -> Option<String> {
|
||||||
|
let rest = path.strip_prefix(SKILLS_DIR)?.strip_prefix('/')?;
|
||||||
|
let name = rest.strip_suffix(".md")?;
|
||||||
|
if name.is_empty() || name.contains('/') {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(name.to_string())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The `# Your skills` preamble under [`Mode::Files`].
|
||||||
|
///
|
||||||
|
/// # Why a third arm
|
||||||
|
///
|
||||||
|
/// `Index` retrieves through `ReadMcpResourceTool`, which is a DEFERRED tool:
|
||||||
|
/// absent from the agent's default list until `ToolSearch` loads it. Measured
|
||||||
|
/// across three matched production runs (`01a07812`, `01a0842e`, `01a09877` —
|
||||||
|
/// same recipe, same task, same three offered uris), that path retrieved
|
||||||
|
/// **1 skill in 9 chances**, and telling the agent in the preamble to load
|
||||||
|
/// the tool first changed nothing: the third run's three reasoning narratives
|
||||||
|
/// never mention skills at all. The section was not declined; it was never
|
||||||
|
/// engaged with.
|
||||||
|
///
|
||||||
|
/// `Read` is a core tool. It is never deferred, and every one of those agents
|
||||||
|
/// used it. So this arm keeps progressive disclosure exactly as `Index` has it
|
||||||
|
/// — name, `when_to_use`, and a pointer the agent has to follow — and changes
|
||||||
|
/// only what the pointer is: a file path instead of an MCP uri. A `Read` of
|
||||||
|
/// that path is a tapped tool call, so Trigger stays as observable as before.
|
||||||
|
pub const FILES_PREAMBLE: &str = "These procedures are AVAILABLE to you; their bodies are \
|
||||||
|
not included below. Each entry names one, says when it applies, and gives the path of the \
|
||||||
|
file that holds it. Where an entry applies to what you are about to do, Read that file FIRST \
|
||||||
|
and then follow it.";
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
|
pub enum Mode {
|
||||||
|
Inline,
|
||||||
|
Index,
|
||||||
|
Files,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Mode {
|
||||||
|
pub fn as_str(self) -> &'static str {
|
||||||
|
match self {
|
||||||
|
Mode::Inline => "inline",
|
||||||
|
Mode::Index => "index",
|
||||||
|
Mode::Files => "files",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Does this arm hand the agent a pointer rather than a body?
|
||||||
|
///
|
||||||
|
/// The two retrieval arms share every rule that follows from that — the
|
||||||
|
/// scorer's Trigger axis, the `always_inject` override, the fallback when
|
||||||
|
/// nothing was installed — and branching on this rather than on `Index`
|
||||||
|
/// is what keeps a third arm from silently inheriting `Inline`'s answers.
|
||||||
|
pub fn is_retrieval(self) -> bool {
|
||||||
|
matches!(self, Mode::Index | Mode::Files)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse a recorded or configured arm. Unrecognised input is `None`, and every
|
||||||
|
/// caller resolves that to `Inline` — an unreadable value must not silently
|
||||||
|
/// select the arm that needs a door.
|
||||||
|
pub fn parse(s: &str) -> Option<Mode> {
|
||||||
|
match s.trim().to_ascii_lowercase().as_str() {
|
||||||
|
"inline" => Some(Mode::Inline),
|
||||||
|
"index" | "progressive" => Some(Mode::Index),
|
||||||
|
"files" | "file" => Some(Mode::Files),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The arm this deployment asks for, before the door is taken into account.
|
||||||
|
pub fn requested() -> Mode {
|
||||||
|
let Ok(raw) = std::env::var(ENV_VAR) else {
|
||||||
|
return DEFAULT;
|
||||||
|
};
|
||||||
|
if raw.trim().is_empty() {
|
||||||
|
return DEFAULT;
|
||||||
|
}
|
||||||
|
match parse(&raw) {
|
||||||
|
Some(m) => m,
|
||||||
|
// Garbage falls to `Inline`, not to `DEFAULT`: an unreadable value must
|
||||||
|
// not silently select an arm that needs something installed.
|
||||||
|
None => {
|
||||||
|
eprintln!(
|
||||||
|
"skill_delivery: {ENV_VAR}={raw:?} is not `inline`, `index` or `files` — \
|
||||||
|
delivering skills inline"
|
||||||
|
);
|
||||||
|
Mode::Inline
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The arm for one mission: `config.skill_delivery` if it names one, otherwise
|
||||||
|
/// the deployment default.
|
||||||
|
///
|
||||||
|
/// Per-mission and not only per-deployment because the alternative is
|
||||||
|
/// restarting the server between arms, and an A/B whose two halves ran against
|
||||||
|
/// different server processes has a confound in it that nothing in the numbers
|
||||||
|
/// will show. This way both arms run against one binary, interleaved.
|
||||||
|
pub fn requested_for(config: &serde_json::Value) -> Mode {
|
||||||
|
let Some(raw) = config.get("skill_delivery").and_then(|v| v.as_str()) else {
|
||||||
|
return requested();
|
||||||
|
};
|
||||||
|
match parse(raw) {
|
||||||
|
Some(m) => m,
|
||||||
|
None => {
|
||||||
|
eprintln!(
|
||||||
|
"skill_delivery: config.skill_delivery={raw:?} is not `inline`, `index` \
|
||||||
|
or `files` — falling back to the deployment default"
|
||||||
|
);
|
||||||
|
requested()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The arm a mission will actually run, given whether what it retrieves from
|
||||||
|
/// was installed — the MCP door for `index`, the skill files for `files`.
|
||||||
|
pub fn resolve(requested: Mode, installed: bool) -> Mode {
|
||||||
|
match (requested, installed) {
|
||||||
|
(m, true) if m.is_retrieval() => m,
|
||||||
|
(m, false) if m.is_retrieval() => {
|
||||||
|
eprintln!(
|
||||||
|
"skill_delivery: `{}` was asked for but this mission has nothing to \
|
||||||
|
retrieve from — falling back to `inline`, because an index the agent \
|
||||||
|
cannot fetch from is worse than no index",
|
||||||
|
m.as_str()
|
||||||
|
);
|
||||||
|
Mode::Inline
|
||||||
|
}
|
||||||
|
_ => Mode::Inline,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The `# Your skills` section heading for an arm.
|
||||||
|
pub fn preamble(mode: Mode) -> &'static str {
|
||||||
|
match mode {
|
||||||
|
Mode::Inline => INLINE_PREAMBLE,
|
||||||
|
Mode::Index => INDEX_PREAMBLE,
|
||||||
|
Mode::Files => FILES_PREAMBLE,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Which arm produced a recorded prompt.
|
||||||
|
///
|
||||||
|
/// Read back from the prompt rather than from the mission row on purpose: the
|
||||||
|
/// row says what the mission was configured to do *now*, and a score is being
|
||||||
|
/// computed against a prompt that was composed then. The recorded prompt is
|
||||||
|
/// the only artefact that cannot have changed since the turn ran.
|
||||||
|
///
|
||||||
|
/// Matched as a whole line. A skill body that quotes the preamble mid-sentence
|
||||||
|
/// is prose; this is the same rule `skill_names_in` learned the hard way.
|
||||||
|
pub fn mode_in_prompt(prompt: &str) -> Mode {
|
||||||
|
// Both spellings, because this reads prompts composed by older builds as
|
||||||
|
// well as the current one. A stored measurement that changes arm when the
|
||||||
|
// writer is edited is not a measurement.
|
||||||
|
for l in prompt.lines() {
|
||||||
|
let l = l.trim();
|
||||||
|
if l == INDEX_PREAMBLE || l == INDEX_PREAMBLE_V1 {
|
||||||
|
return Mode::Index;
|
||||||
|
}
|
||||||
|
if l == FILES_PREAMBLE {
|
||||||
|
return Mode::Files;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Mode::Inline
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One index entry's text — everything under the `--- SKILL: <name> ---`
|
||||||
|
/// marker, which [`crate::topology_exec::render_pinned_skill`] writes.
|
||||||
|
///
|
||||||
|
/// `when_to_use` is the load-bearing field: it is the only thing the agent has
|
||||||
|
/// to judge relevance from, so a skill with none says so rather than omitting
|
||||||
|
/// the line and leaving the model to infer from the description alone.
|
||||||
|
pub fn index_entry(description: &str, when_to_use: Option<&str>, uri: &str) -> String {
|
||||||
|
let when = when_to_use
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|w| !w.is_empty())
|
||||||
|
.unwrap_or("not stated — judge from the description");
|
||||||
|
format!(
|
||||||
|
"{}\nWhen to use: {}\n{READ_IT}server=\"{}\", uri=\"{}\")",
|
||||||
|
description.trim(),
|
||||||
|
when,
|
||||||
|
MCP_SERVER,
|
||||||
|
uri,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The line that makes an index entry recognisable as one.
|
||||||
|
///
|
||||||
|
/// Shared by the renderer and [`skill_was_indexed`] so the scorer cannot drift
|
||||||
|
/// from the delivery — two spellings of one marker is how a detector quietly
|
||||||
|
/// stops detecting.
|
||||||
|
pub const READ_IT: &str = "Read it: ReadMcpResourceTool(";
|
||||||
|
|
||||||
|
/// [`READ_IT`]'s counterpart for the `files` arm. Same rule: one constant,
|
||||||
|
/// written by [`file_entry`] and read by [`skill_was_indexed`].
|
||||||
|
pub const READ_FILE_IT: &str = "Read it: Read(file_path=\"";
|
||||||
|
|
||||||
|
/// One `files`-arm entry — [`index_entry`] with a path where the uri was.
|
||||||
|
pub fn file_entry(description: &str, when_to_use: Option<&str>, path: &str) -> String {
|
||||||
|
let when = when_to_use
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|w| !w.is_empty())
|
||||||
|
.unwrap_or("not stated — judge from the description");
|
||||||
|
format!(
|
||||||
|
"{}\nWhen to use: {}\n{READ_FILE_IT}{}\")",
|
||||||
|
description.trim(),
|
||||||
|
when,
|
||||||
|
path,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How was THIS skill delivered, regardless of the arm the prompt announces?
|
||||||
|
///
|
||||||
|
/// `Some(true)` — an index entry: named, described, and left to be fetched.
|
||||||
|
/// `Some(false)` — the body itself, which under `Index` means the skill is
|
||||||
|
/// marked `always_inject`.
|
||||||
|
/// `None` — not in the prompt at all (it was retrieved, or never delivered).
|
||||||
|
///
|
||||||
|
/// The arm is a property of the PROMPT; `always_inject` is a property of the
|
||||||
|
/// SKILL. Scoring the arm alone would report a Trigger failure against a skill
|
||||||
|
/// the agent was handed and was never asked to fetch.
|
||||||
|
pub fn skill_was_indexed(prompt: &str, skill: &str) -> Option<bool> {
|
||||||
|
let marker = crate::topology_exec::SKILL_MARKER;
|
||||||
|
let mut lines = prompt.lines();
|
||||||
|
// Find this skill's section...
|
||||||
|
lines.find(|l| {
|
||||||
|
l.trim()
|
||||||
|
.strip_prefix(marker)
|
||||||
|
.map(|rest| rest.trim_end_matches(" ---").trim() == skill)
|
||||||
|
.unwrap_or(false)
|
||||||
|
})?;
|
||||||
|
// ...and read to the next one.
|
||||||
|
for l in lines {
|
||||||
|
if l.trim().starts_with(marker) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if l.contains(READ_IT) || l.contains(READ_FILE_IT) {
|
||||||
|
return Some(true);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Some(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_unreadable_arm_never_selects_the_one_that_needs_a_door() {
|
||||||
|
assert_eq!(parse("nonsense"), None);
|
||||||
|
assert_eq!(parse("INDEX"), Some(Mode::Index));
|
||||||
|
assert_eq!(parse(" inline "), Some(Mode::Inline));
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_mission_can_name_its_own_arm() {
|
||||||
|
assert_eq!(
|
||||||
|
requested_for(&serde_json::json!({ "skill_delivery": "index" })),
|
||||||
|
Mode::Index
|
||||||
|
);
|
||||||
|
// Unreadable values and absent ones both defer to the deployment
|
||||||
|
// default, which is `Inline` unless the environment says otherwise.
|
||||||
|
assert_eq!(
|
||||||
|
requested_for(&serde_json::json!({ "skill_delivery": "sideways" })),
|
||||||
|
requested()
|
||||||
|
);
|
||||||
|
assert_eq!(requested_for(&serde_json::json!({})), requested());
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn index_without_a_door_falls_back() {
|
||||||
|
assert_eq!(resolve(Mode::Index, false), Mode::Inline);
|
||||||
|
assert_eq!(resolve(Mode::Index, true), Mode::Index);
|
||||||
|
assert_eq!(resolve(Mode::Inline, true), Mode::Inline);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The deployment default is a measured decision; changing it should fail
|
||||||
|
/// a test so it is made on purpose, with the numbers in front of you.
|
||||||
|
#[test]
|
||||||
|
fn the_default_arm_is_files_and_garbage_still_falls_to_inline() {
|
||||||
|
assert_eq!(DEFAULT, Mode::Files);
|
||||||
|
assert_eq!(requested_for(&serde_json::json!({})), requested());
|
||||||
|
assert_eq!(
|
||||||
|
requested_for(&serde_json::json!({ "skill_delivery": "sideways" })),
|
||||||
|
requested(),
|
||||||
|
"an unreadable per-mission value defers to the deployment, as before"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn the_files_arm_parses_resolves_and_reads_back() {
|
||||||
|
assert_eq!(parse("files"), Some(Mode::Files));
|
||||||
|
assert_eq!(resolve(Mode::Files, true), Mode::Files);
|
||||||
|
assert_eq!(
|
||||||
|
resolve(Mode::Files, false),
|
||||||
|
Mode::Inline,
|
||||||
|
"files that were never written must not be advertised"
|
||||||
|
);
|
||||||
|
let prompt = format!("Task: x\n\n# Your skills\n\n{FILES_PREAMBLE}\n\nentry");
|
||||||
|
assert_eq!(mode_in_prompt(&prompt), Mode::Files);
|
||||||
|
assert_eq!(preamble(Mode::Files), FILES_PREAMBLE);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The writer and the reader of a skill path are one pair of functions.
|
||||||
|
#[test]
|
||||||
|
fn a_skill_path_round_trips_and_nothing_else_parses_as_one() {
|
||||||
|
let p = skill_file_path("web-search-triage");
|
||||||
|
assert_eq!(p, "/mission/skills/web-search-triage.md");
|
||||||
|
assert_eq!(skill_from_file_path(&p).as_deref(), Some("web-search-triage"));
|
||||||
|
for not_a_skill in [
|
||||||
|
"/mission/repo/skills/x.md",
|
||||||
|
"/mission/skills/x.txt",
|
||||||
|
"/mission/skills/.md",
|
||||||
|
"/mission/skills/a/b.md",
|
||||||
|
"/mission/skills",
|
||||||
|
"mission/skills/x.md",
|
||||||
|
] {
|
||||||
|
assert_eq!(skill_from_file_path(not_a_skill), None, "{not_a_skill}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `skill_was_indexed` is how the scorer tells a pointer from a body. A
|
||||||
|
/// file entry must read as a pointer, or `always_inject` logic would treat
|
||||||
|
/// every `files`-arm skill as handed over.
|
||||||
|
#[test]
|
||||||
|
fn a_file_entry_reads_as_indexed_not_inlined() {
|
||||||
|
let entry = file_entry("Summarise.", Some("when asked"), &skill_file_path("x"));
|
||||||
|
assert!(entry.contains(READ_FILE_IT), "{entry}");
|
||||||
|
let prompt = format!(
|
||||||
|
"Task\n\n{}x ---\n{entry}\n",
|
||||||
|
crate::topology_exec::SKILL_MARKER
|
||||||
|
);
|
||||||
|
assert_eq!(skill_was_indexed(&prompt, "x"), Some(true));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A prompt composed before the tool-loading sentence existed must still
|
||||||
|
/// score as `Index`. Stored prompts are held for 90 days and re-scored
|
||||||
|
/// when the scorer changes; if this regressed, every one of them would
|
||||||
|
/// quietly become an `inline` run and Trigger would be reported against an
|
||||||
|
/// arm that never ran.
|
||||||
|
#[test]
|
||||||
|
fn an_older_index_prompt_still_reads_as_index() {
|
||||||
|
let old = format!("Task: x\n\n# Your skills\n\n{INDEX_PREAMBLE_V1}\n\nentry");
|
||||||
|
assert_eq!(mode_in_prompt(&old), Mode::Index);
|
||||||
|
let new = format!("Task: x\n\n# Your skills\n\n{INDEX_PREAMBLE}\n\nentry");
|
||||||
|
assert_eq!(mode_in_prompt(&new), Mode::Index);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The two spellings must stay one text plus an addition, not two texts.
|
||||||
|
/// Written out in full because `concat!` cannot take a const, so nothing
|
||||||
|
/// but this test stops them drifting apart.
|
||||||
|
#[test]
|
||||||
|
fn the_current_preamble_extends_the_original() {
|
||||||
|
assert!(
|
||||||
|
INDEX_PREAMBLE.starts_with(INDEX_PREAMBLE_V1),
|
||||||
|
"the v1 preamble must remain a prefix, or old prompts stop matching"
|
||||||
|
);
|
||||||
|
assert!(INDEX_PREAMBLE.contains("select:ReadMcpResourceTool"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The scorer reads the arm off the prompt, so the writer and this reader
|
||||||
|
/// have to agree for every arm — including the one that writes no marker.
|
||||||
|
#[test]
|
||||||
|
fn the_arm_is_recoverable_from_the_prompt_that_was_sent() {
|
||||||
|
let inline = format!("Task: x\n\n# Your skills\n\n{INLINE_PREAMBLE}\n\nbody");
|
||||||
|
let index = format!("Task: x\n\n# Your skills\n\n{INDEX_PREAMBLE}\n\nentry");
|
||||||
|
assert_eq!(mode_in_prompt(&inline), Mode::Inline);
|
||||||
|
assert_eq!(mode_in_prompt(&index), Mode::Index);
|
||||||
|
assert_eq!(mode_in_prompt("Task: x"), Mode::Inline);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A body quoting the preamble must not re-label the arm — the same
|
||||||
|
/// failure `SKILL_MARKER` had when a heading inside a body counted.
|
||||||
|
#[test]
|
||||||
|
fn a_body_quoting_the_preamble_does_not_change_the_arm() {
|
||||||
|
let body = format!("The index arm opens with \"{INDEX_PREAMBLE}\" and then lists.");
|
||||||
|
let prompt = format!("Task: x\n\n# Your skills\n\n{INLINE_PREAMBLE}\n\n{body}");
|
||||||
|
assert_eq!(mode_in_prompt(&prompt), Mode::Inline);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_entry_states_a_missing_when_to_use_rather_than_dropping_the_line() {
|
||||||
|
let e = index_entry("Summarise a paper.", None, "skill:global/x");
|
||||||
|
assert!(e.contains("When to use: not stated"), "{e}");
|
||||||
|
assert!(e.contains("ReadMcpResourceTool(server=\"clawmates_skills\""), "{e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,95 @@
|
|||||||
|
//! Applies agents' own skill drafts, with no human decision.
|
||||||
|
//!
|
||||||
|
//! `level_up` has generated complete skill drafts from a model since it
|
||||||
|
//! shipped; the only thing between a draft and the catalogue was an operator
|
||||||
|
//! ticking a checkbox in `LevelUpDrawer`. This worker removes the checkbox —
|
||||||
|
//! when switched on. It is OFF by default since 2026-09-20; see
|
||||||
|
//! `level_up::self_authoring_enabled` for why.
|
||||||
|
//!
|
||||||
|
//! What is deliberately NOT removed is the record. Every write stays
|
||||||
|
//! workspace-scoped and versioned, cannot take the name of a hand-authored
|
||||||
|
//! skill, and lands with `approved_by = NULL` — so "an agent decided this" is
|
||||||
|
//! distinguishable from "a person decided this" forever after, which is the
|
||||||
|
//! property that makes the change reversible instead of merely fast.
|
||||||
|
//!
|
||||||
|
//! Only `skill_candidate` items apply here. `identity_refinement` and
|
||||||
|
//! `brain_consolidation` still wait for a human: they change what an agent IS
|
||||||
|
//! rather than adding a procedure it can consult.
|
||||||
|
|
||||||
|
use sqlx::{PgPool, Row};
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
|
/// How often to sweep for pending drafts.
|
||||||
|
///
|
||||||
|
/// Proposals arrive when someone runs a level-up, not continuously, so this is
|
||||||
|
/// slow on purpose — the work is bounded by how often an agent reflects, and
|
||||||
|
/// polling faster would only add load.
|
||||||
|
const SWEEP_INTERVAL: Duration = Duration::from_secs(120);
|
||||||
|
|
||||||
|
/// Start the sweep, unless self-authoring is switched off.
|
||||||
|
pub fn spawn(pool: PgPool) {
|
||||||
|
if !crate::level_up::self_authoring_enabled() {
|
||||||
|
eprintln!(
|
||||||
|
"skill_self_authoring: DISABLED (the default since 2026-09-20) — \
|
||||||
|
agent skill drafts wait for a human in the level-up drawer. \
|
||||||
|
Set CLAWMATES_SKILL_SELF_AUTHORING=1 to let agents apply their own."
|
||||||
|
);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
eprintln!(
|
||||||
|
"skill_self_authoring: ENABLED — agents apply their own skill drafts \
|
||||||
|
without human approval. Writes are workspace-scoped, versioned, and \
|
||||||
|
cannot take a hand-authored skill's name; each lands with no approver \
|
||||||
|
recorded. Unset CLAWMATES_SKILL_SELF_AUTHORING to restore the gate."
|
||||||
|
);
|
||||||
|
tokio::spawn(async move {
|
||||||
|
loop {
|
||||||
|
if let Err(e) = sweep(&pool).await {
|
||||||
|
eprintln!("skill_self_authoring: sweep failed: {e}");
|
||||||
|
}
|
||||||
|
tokio::time::sleep(SWEEP_INTERVAL).await;
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Apply every pending proposal's skill candidates. Returns how many skills landed.
|
||||||
|
pub async fn sweep(pool: &PgPool) -> Result<usize, String> {
|
||||||
|
// Bounded per pass: a backlog drains over several sweeps rather than
|
||||||
|
// holding the pool for as long as it takes to apply all of it.
|
||||||
|
let rows = sqlx::query(
|
||||||
|
"SELECT id, workspace_id FROM level_up_proposals
|
||||||
|
WHERE status = 'pending'
|
||||||
|
ORDER BY created_at
|
||||||
|
LIMIT 20",
|
||||||
|
)
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("select pending proposals: {e}"))?;
|
||||||
|
|
||||||
|
let mut applied = 0usize;
|
||||||
|
for row in &rows {
|
||||||
|
let id: uuid::Uuid = row.get("id");
|
||||||
|
let workspace_id: uuid::Uuid = row.get("workspace_id");
|
||||||
|
match crate::level_up::apply_autonomous(
|
||||||
|
pool,
|
||||||
|
cm_domain::WorkspaceId::from(workspace_id),
|
||||||
|
id,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
Ok(items) if !items.is_empty() => {
|
||||||
|
applied += items.len();
|
||||||
|
eprintln!(
|
||||||
|
"skill_self_authoring: applied {} skill draft(s) from proposal {id} \
|
||||||
|
with no human approval",
|
||||||
|
items.len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// A proposal with no skill candidates is left pending on purpose —
|
||||||
|
// its identity/memory items still belong to the human gate.
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => eprintln!("skill_self_authoring: proposal {id}: {e}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(applied)
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -7,6 +7,13 @@
|
|||||||
//! description: <one-line, shown to the LLM in resources/list>
|
//! description: <one-line, shown to the LLM in resources/list>
|
||||||
//! when_to_use: <trigger sentence, appended to description>
|
//! when_to_use: <trigger sentence, appended to description>
|
||||||
//! tags: [foundation, rust, ...]
|
//! tags: [foundation, rust, ...]
|
||||||
|
//! always_inject: true # optional, default false
|
||||||
|
//!
|
||||||
|
//! `always_inject` makes the body reach the agent in full even under the
|
||||||
|
//! `index` (progressive-disclosure) arm. It is for a CROSS-CUTTING procedure —
|
||||||
|
//! one that applies to everyone who writes, and so reads to each agent as
|
||||||
|
//! nobody's in particular, which is how `workspace-repo-commit-protocol`
|
||||||
|
//! scored Trigger=FAIL beside a passing boundary check.
|
||||||
//!
|
//!
|
||||||
//! The body is the rest of the file. Both are upserted idempotently:
|
//! The body is the rest of the file. Both are upserted idempotently:
|
||||||
//! `skills_catalog::upsert_builtin` bumps the version + appends to
|
//! `skills_catalog::upsert_builtin` bumps the version + appends to
|
||||||
@@ -27,6 +34,8 @@ struct Frontmatter {
|
|||||||
when_to_use: Option<String>,
|
when_to_use: Option<String>,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
tags: Vec<String>,
|
tags: Vec<String>,
|
||||||
|
#[serde(default)]
|
||||||
|
always_inject: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
fn skills_dir() -> PathBuf {
|
fn skills_dir() -> PathBuf {
|
||||||
@@ -120,6 +129,7 @@ async fn load_one(pool: &PgPool, path: &std::path::Path) -> Result<String, Strin
|
|||||||
when_to_use: fm.when_to_use.as_deref(),
|
when_to_use: fm.when_to_use.as_deref(),
|
||||||
tags: fm.tags.clone(),
|
tags: fm.tags.clone(),
|
||||||
body,
|
body,
|
||||||
|
always_inject: fm.always_inject,
|
||||||
};
|
};
|
||||||
upsert_builtin(pool, skill)
|
upsert_builtin(pool, skill)
|
||||||
.await
|
.await
|
||||||
@@ -158,6 +168,36 @@ mod tests {
|
|||||||
assert!(split_frontmatter("# plain md\n").is_none());
|
assert!(split_frontmatter("# plain md\n").is_none());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn always_inject_is_opt_in_and_parses() {
|
||||||
|
let off: Frontmatter = serde_yaml::from_str("name: a\ndescription: b\n").unwrap();
|
||||||
|
assert!(
|
||||||
|
!off.always_inject,
|
||||||
|
"full delivery must be opted INTO — defaulting true would abolish the index arm"
|
||||||
|
);
|
||||||
|
let on: Frontmatter =
|
||||||
|
serde_yaml::from_str("name: a\ndescription: b\nalways_inject: true\n").unwrap();
|
||||||
|
assert!(on.always_inject);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The flag reached production as a hand-run UPDATE first, which a rebuilt
|
||||||
|
/// database would have silently dropped. This asserts the repo carries it,
|
||||||
|
/// so the cross-cutting skill cannot go back to being deliverable only by
|
||||||
|
/// an agent noticing it applies — the exact failure it was measured on.
|
||||||
|
#[test]
|
||||||
|
fn the_commit_protocol_ships_marked_for_full_delivery() {
|
||||||
|
let path = std::path::PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||||
|
.join("../../skills/foundation/workspace-repo-commit-protocol.md");
|
||||||
|
let text = std::fs::read_to_string(&path).expect("read the commit-protocol skill");
|
||||||
|
let (yaml, _) = split_frontmatter(&text).expect("frontmatter");
|
||||||
|
let fm: Frontmatter = serde_yaml::from_str(yaml).expect("parse frontmatter");
|
||||||
|
assert!(
|
||||||
|
fm.always_inject,
|
||||||
|
"workspace-repo-commit-protocol must be always_inject: it applies to everyone \
|
||||||
|
who writes, and under the index arm it scored Trigger=FAIL unread"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn builtin_id_stable() {
|
fn builtin_id_stable() {
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
@@ -170,3 +210,266 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod contradiction_tests {
|
||||||
|
use std::path::PathBuf;
|
||||||
|
|
||||||
|
fn repo_root(rel: &str) -> PathBuf {
|
||||||
|
PathBuf::from(env!("CARGO_MANIFEST_DIR"))
|
||||||
|
.join("../..")
|
||||||
|
.join(rel)
|
||||||
|
.canonicalize()
|
||||||
|
.unwrap_or_else(|e| panic!("{rel}: {e}"))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn walk_ext(dir: &std::path::Path, ext: &str, out: &mut Vec<(String, String)>) {
|
||||||
|
for e in std::fs::read_dir(dir).expect("read dir") {
|
||||||
|
let p = e.expect("entry").path();
|
||||||
|
if p.is_dir() {
|
||||||
|
walk_ext(&p, ext, out);
|
||||||
|
} else if p.extension().and_then(|x| x.to_str()) == Some(ext) {
|
||||||
|
out.push((
|
||||||
|
p.file_name().unwrap().to_string_lossy().to_string(),
|
||||||
|
std::fs::read_to_string(&p).expect("read file"),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Skill bodies alone.
|
||||||
|
fn all_skills() -> Vec<(String, String)> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
walk_ext(&repo_root("skills"), "md", &mut out);
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// **Everything we ship that becomes prompt text an agent reads.**
|
||||||
|
///
|
||||||
|
/// Skills and team-template role prompts, in one corpus, because the rules
|
||||||
|
/// below are properties of *what an agent is told* — not of which file it
|
||||||
|
/// happened to be written in.
|
||||||
|
///
|
||||||
|
/// This function is the finding. The `/workspace/repo` guard was written on
|
||||||
|
/// 2026-08-19 against `skills/` only, and the same wrong path had been
|
||||||
|
/// sitting in **four team templates** the whole time — including
|
||||||
|
/// `rust_sdlc`, the default for five of the six workflow recipes, whose
|
||||||
|
/// coder was told "your working directory is /workspace/repo" and whose
|
||||||
|
/// committer was told to `cd` there. A guard that covers one corpus and not
|
||||||
|
/// the other reads exactly like a guard that covers the problem.
|
||||||
|
fn all_shipped_prompts() -> Vec<(String, String)> {
|
||||||
|
let mut out = all_skills();
|
||||||
|
walk_ext(&repo_root("templates/teams"), "toml", &mut out);
|
||||||
|
walk_ext(&repo_root("templates/workflows"), "toml", &mut out);
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Nothing we ship may teach a workspace path the platform does not mount.
|
||||||
|
///
|
||||||
|
/// `workspace-repo-commit-protocol` told agents that `/workspace/repo` was
|
||||||
|
/// "the ONLY path where source-modifying edits belong". The platform mounts
|
||||||
|
/// and advertises `/mission/repo` — in 26 places — and `/workspace/repo`
|
||||||
|
/// appears nowhere in the code. The skill is pinned on 29 role bindings and
|
||||||
|
/// was delivered twice in a single measured run, so agents received the
|
||||||
|
/// platform's real path and a skill contradicting it in the SAME prompt.
|
||||||
|
#[test]
|
||||||
|
fn nothing_we_ship_teaches_a_repo_path_the_platform_does_not_mount() {
|
||||||
|
let mut offenders = Vec::new();
|
||||||
|
for (name, body) in all_shipped_prompts() {
|
||||||
|
if body.contains("/workspace/repo") {
|
||||||
|
offenders.push(name);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
offenders.is_empty(),
|
||||||
|
"{} shipped prompt file(s) name /workspace/repo; the mission \
|
||||||
|
checkout is /mission/repo, so an agent following them writes \
|
||||||
|
somewhere that is never delivered: {}",
|
||||||
|
offenders.len(),
|
||||||
|
offenders.join(", ")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No skill may instruct an agent to call a tool it does not have.
|
||||||
|
///
|
||||||
|
/// Every mission turn ends in `claude -p`, so the tools are Claude Code's
|
||||||
|
/// (`Read`/`Edit`/`Write`/`Bash`/`Glob`/`Grep`). `phase_task_text` used to
|
||||||
|
/// advertise ZeroClaw's names and was fixed after five agents spent 7.4k
|
||||||
|
/// tokens on one mission describing the mismatch instead of working — and
|
||||||
|
/// the same wrong names survived inside a pinned skill.
|
||||||
|
///
|
||||||
|
/// Matched as a backticked instruction, not as bare words: a skill may
|
||||||
|
/// legitimately DISCUSS these names, as this one now does when warning
|
||||||
|
/// against them.
|
||||||
|
#[test]
|
||||||
|
fn nothing_we_ship_instructs_an_agent_to_call_a_zeroclaw_tool() {
|
||||||
|
const ZEROCLAW_TOOLS: &[&str] = &[
|
||||||
|
"`file_read`",
|
||||||
|
"`file_write`",
|
||||||
|
"`file_edit`",
|
||||||
|
"`content_search`",
|
||||||
|
"`glob_search`",
|
||||||
|
];
|
||||||
|
let mut offenders = Vec::new();
|
||||||
|
for (name, body) in all_shipped_prompts() {
|
||||||
|
// The line has to READ as an instruction. "Do not reach for
|
||||||
|
// `file_read`" is the correction, not the defect.
|
||||||
|
for line in body.lines() {
|
||||||
|
let l = line.to_ascii_lowercase();
|
||||||
|
if l.contains("do not")
|
||||||
|
|| l.contains("never")
|
||||||
|
|| l.contains("instead of")
|
||||||
|
|| l.contains("not what")
|
||||||
|
{
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if ZEROCLAW_TOOLS.iter().any(|t| line.contains(t)) {
|
||||||
|
offenders.push(format!("{name}: {}", line.trim()));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
offenders.is_empty(),
|
||||||
|
"{} shipped prompt line(s) tell an agent to use a tool its \
|
||||||
|
subprocess does not expose:\n {}",
|
||||||
|
offenders.len(),
|
||||||
|
offenders.join("\n ")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// No skill may show a marker the real parser rejects.
|
||||||
|
///
|
||||||
|
/// Checked by running `task_card_parser::parse` itself, never a copy of its
|
||||||
|
/// rules — a second implementation of the contract drifts, and then the
|
||||||
|
/// test passes while the mission loop stalls.
|
||||||
|
///
|
||||||
|
/// This is the third instance of one class: the skills were written
|
||||||
|
/// alongside the platform and then never compared to it again. The first
|
||||||
|
/// was a repo path the platform does not mount; the second a tool the agent
|
||||||
|
/// does not have; this one is `PLAN_COMPLETE: INT-01..05` in
|
||||||
|
/// `decompose-int-items`, which a live planner emitted verbatim. Ids are
|
||||||
|
/// strictly `INT-<digits>`, so the range form parses to nothing — the plan
|
||||||
|
/// pass records no completion at all while every item stays open.
|
||||||
|
///
|
||||||
|
/// Scoped to fenced code blocks, which is where a skill puts the text it
|
||||||
|
/// tells an agent to EMIT. A marker named in a sentence is prose.
|
||||||
|
#[test]
|
||||||
|
fn no_skill_shows_a_marker_the_parser_would_reject() {
|
||||||
|
// The templates. `INT-NN` is a placeholder an agent substitutes, not a
|
||||||
|
// literal it emits, so it is not a contradiction.
|
||||||
|
const PLACEHOLDERS: &[&str] = &["INT-NN", "INT-XX", "INT-N", "INT-nn"];
|
||||||
|
let mut offenders = Vec::new();
|
||||||
|
for (name, body) in all_skills() {
|
||||||
|
let mut fenced = false;
|
||||||
|
for line in body.lines() {
|
||||||
|
if line.trim_start().starts_with("```") {
|
||||||
|
fenced = !fenced;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let t = line.trim();
|
||||||
|
if !fenced || !t.contains("INT-") || !t.contains(':') {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let Some((kind, _)) = t.split_once(':') else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
if !MARKER_KINDS.contains(&kind.trim()) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if PLACEHOLDERS.iter().any(|p| t.contains(p)) {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if crate::task_card_parser::parse(t).is_empty() {
|
||||||
|
offenders.push(format!("{name}: {t}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
offenders.is_empty(),
|
||||||
|
"{} skill line(s) show a marker the parser rejects — an agent that \
|
||||||
|
follows them exactly is silently ignored:\n {}",
|
||||||
|
offenders.len(),
|
||||||
|
offenders.join("\n ")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every team a recipe names must be a team that exists.
|
||||||
|
///
|
||||||
|
/// `create()` logs and carries on when a recipe names a template that is
|
||||||
|
/// not loaded, because failing mission creation over it would be worse.
|
||||||
|
/// That makes a typo here invisible in exactly the way that matters: the
|
||||||
|
/// mission is staffed by the fallback crew and looks deliberate. `research_only`
|
||||||
|
/// pointed at `rust_sdlc` for months and nothing said a word.
|
||||||
|
#[test]
|
||||||
|
fn every_team_a_recipe_names_exists() {
|
||||||
|
let mut keys = std::collections::HashSet::new();
|
||||||
|
for (_, body) in {
|
||||||
|
let mut v = Vec::new();
|
||||||
|
walk_ext(&repo_root("templates/teams"), "toml", &mut v);
|
||||||
|
v
|
||||||
|
} {
|
||||||
|
for line in body.lines() {
|
||||||
|
if let Some(rest) = line.trim().strip_prefix("key") {
|
||||||
|
if let Some((_, val)) = rest.split_once('=') {
|
||||||
|
keys.insert(val.trim().trim_matches('"').to_string());
|
||||||
|
}
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(!keys.is_empty(), "no team templates found at all");
|
||||||
|
|
||||||
|
let mut recipes = Vec::new();
|
||||||
|
walk_ext(&repo_root("templates/workflows"), "toml", &mut recipes);
|
||||||
|
let mut missing = Vec::new();
|
||||||
|
for (name, body) in recipes {
|
||||||
|
let mut table = String::new();
|
||||||
|
for line in body.lines() {
|
||||||
|
let line = line.trim();
|
||||||
|
if line.starts_with('[') {
|
||||||
|
table = line.trim_matches(['[', ']'].as_slice()).to_string();
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if line.starts_with('#') {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let named = if let Some((_, v)) = line.split_once('=') {
|
||||||
|
if line.starts_with("default_team_template")
|
||||||
|
|| table == "default_phase_teams"
|
||||||
|
{
|
||||||
|
Some(v.trim().trim_matches('"').to_string())
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
if let Some(k) = named {
|
||||||
|
if !keys.contains(&k) {
|
||||||
|
missing.push(format!("{name} -> {k}"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert!(
|
||||||
|
missing.is_empty(),
|
||||||
|
"{} recipe(s) name a team template that does not exist, so the mission \
|
||||||
|
is staffed by the fallback crew and looks deliberate: {}",
|
||||||
|
missing.len(),
|
||||||
|
missing.join(", ")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The marker kinds, as the parser spells them.
|
||||||
|
const MARKER_KINDS: &[&str] = &[
|
||||||
|
"TASK",
|
||||||
|
"PLAN_COMPLETE",
|
||||||
|
"WORK",
|
||||||
|
"HANDOFF",
|
||||||
|
"TEST_PASS",
|
||||||
|
"TEST_FAIL",
|
||||||
|
"REVIEW_APPROVE",
|
||||||
|
"REVIEW_BLOCK",
|
||||||
|
"COMPLETED",
|
||||||
|
];
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,683 @@
|
|||||||
|
//! The Anthropic provider backed by the SUBSCRIPTION token, not the metered key.
|
||||||
|
//!
|
||||||
|
//! Two Anthropic credentials reach this server and they bill differently:
|
||||||
|
//!
|
||||||
|
//! - `ANTHROPIC_API_KEY` (`sk-ant-api…`) — metered, pay-as-you-go, and the thing
|
||||||
|
//! that runs out. Every mission VM already avoids it: `mission_runtime` sends
|
||||||
|
//! only the subscription token into a guest, deliberately.
|
||||||
|
//! - `ANTHROPIC_OAUTH_TOKEN` / `CLAUDE_CODE_OAUTH_TOKEN` (`sk-ant-oat…`) — the
|
||||||
|
//! Claude Code subscription, which is what the CLI inside every VM runs on.
|
||||||
|
//!
|
||||||
|
//! Server-side model calls that went through `Runtime::complete` with a bare
|
||||||
|
//! model name resolved to the DEFAULT provider — the metered key. So the roster
|
||||||
|
//! planner died with
|
||||||
|
//! `400 … "Your credit balance is too low to access the Anthropic API"` while
|
||||||
|
//! every mission on the same machine kept running fine on the subscription.
|
||||||
|
//! The harness reported it honestly as FAIL-NORUN rather than a passing scenario,
|
||||||
|
//! which is the only reason it was visible at all.
|
||||||
|
//!
|
||||||
|
//! This is the one place that turns the subscription token into a provider.
|
||||||
|
//! `evaluator::subscription_judge` had its own copy; there is now one.
|
||||||
|
|
||||||
|
/// The subscription-backed provider, or `None` when no usable token is present.
|
||||||
|
///
|
||||||
|
/// Checks the `sk-ant-oat` prefix rather than trusting the variable name: an
|
||||||
|
/// `sk-ant-api` key pasted into the OAuth slot would authenticate and then bill
|
||||||
|
/// the metered account, which is the failure this module exists to prevent —
|
||||||
|
/// silently, and with the same error weeks later.
|
||||||
|
pub fn provider() -> Option<cm_llm::AnthropicProvider> {
|
||||||
|
for var in ["ANTHROPIC_OAUTH_TOKEN", "CLAUDE_CODE_OAUTH_TOKEN"] {
|
||||||
|
let Ok(token) = std::env::var(var) else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
let token = token.trim();
|
||||||
|
if token.is_empty() {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
if !is_subscription_token(token) {
|
||||||
|
eprintln!(
|
||||||
|
"subscription: {var} is set but is not a Claude Code setup token \
|
||||||
|
(expected sk-ant-oat…) — ignoring it rather than billing the \
|
||||||
|
metered key by accident"
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
return Some(cm_llm::AnthropicProvider::new(token.to_string()));
|
||||||
|
}
|
||||||
|
None
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a token is a Claude Code subscription token rather than an API key.
|
||||||
|
pub fn is_subscription_token(token: &str) -> bool {
|
||||||
|
token.trim().starts_with("sk-ant-oat")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One completion on the subscription, mirroring `Runtime::complete`'s contract
|
||||||
|
/// so a caller can swap between them without reshaping its call.
|
||||||
|
///
|
||||||
|
/// Falls back to the caller's runtime when no subscription token exists, so a
|
||||||
|
/// deployment without one behaves exactly as it did before.
|
||||||
|
pub async fn complete_or(
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
|
system: &str,
|
||||||
|
user: &str,
|
||||||
|
model: &str,
|
||||||
|
max_tokens: u32,
|
||||||
|
// Carried explicitly rather than defaulted. The Master Planner and the claw
|
||||||
|
// enhancer both pass `true`, and a helper that quietly dropped it would take
|
||||||
|
// web search away from two features while every test still passed.
|
||||||
|
web_search: bool,
|
||||||
|
) -> Result<String, String> {
|
||||||
|
// A `name:model` spec is an operator's explicit provider choice — the swarm
|
||||||
|
// worker model is literally configured that way (`kimi:kimi-k2.6`), and
|
||||||
|
// `Runtime::resolve_provider` honours it. Forcing that onto Anthropic would
|
||||||
|
// silently run someone's chosen model on the wrong provider, which is the
|
||||||
|
// same class of bug as this module exists to fix, only pointed the other
|
||||||
|
// way. Only a BARE name is ambiguous, and a bare name is what resolves to
|
||||||
|
// the default provider — the metered key.
|
||||||
|
if !is_bare_model_name(model) || provider().is_none() {
|
||||||
|
return runtime
|
||||||
|
.complete(system, user, model, max_tokens, web_search)
|
||||||
|
.await;
|
||||||
|
}
|
||||||
|
let provider = provider().expect("checked just above");
|
||||||
|
complete_with(&provider, system, user, model, max_tokens, web_search).await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a model string names a model without naming a provider.
|
||||||
|
pub fn is_bare_model_name(model: &str) -> bool {
|
||||||
|
!model.contains(':')
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How long to wait before each retry. Four attempts, ~30s of patience total.
|
||||||
|
///
|
||||||
|
/// The subscription has no credit wall, but it does have a rate limit, and a
|
||||||
|
/// roster proposal is a single one-shot call: a 429 that a browser would shrug
|
||||||
|
/// off used to fail the whole "propose a team" button. Measured on this
|
||||||
|
/// deployment — moving the roster onto the subscription turned
|
||||||
|
/// `400 credit balance too low` into `429 rate_limit_error`, i.e. a wall that
|
||||||
|
/// clears on its own became the failure mode, so waiting is the right answer.
|
||||||
|
const BACKOFF_SECS: &[u64] = &[2, 8, 20];
|
||||||
|
|
||||||
|
/// Whether an error is worth waiting out rather than reporting.
|
||||||
|
///
|
||||||
|
/// Deliberately narrow. A 400 (bad request), 401 (wrong token) or 404 (unknown
|
||||||
|
/// model) will never succeed on a retry, and retrying them turns a legible
|
||||||
|
/// error into a 30-second hang followed by the same error.
|
||||||
|
fn is_transient(e: &cm_llm::LlmError) -> bool {
|
||||||
|
use cm_llm::LlmError;
|
||||||
|
match e {
|
||||||
|
// The transport never reached Anthropic — a dropped connection or a
|
||||||
|
// DNS blip, not a rejected request.
|
||||||
|
LlmError::Transport(_) => true,
|
||||||
|
LlmError::Api(detail) => {
|
||||||
|
// `anthropic.rs` formats these as `"{status}: {body}"`.
|
||||||
|
detail.starts_with("429")
|
||||||
|
|| detail.starts_with("500")
|
||||||
|
|| detail.starts_with("502")
|
||||||
|
|| detail.starts_with("503")
|
||||||
|
|| detail.starts_with("529")
|
||||||
|
|| detail.contains("rate_limit")
|
||||||
|
|| detail.contains("overloaded")
|
||||||
|
}
|
||||||
|
LlmError::Scenario(_) | LlmError::Wire(_) => false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Models to try, in order, when the requested one is rate limited.
|
||||||
|
///
|
||||||
|
/// The order is capability first, then independence:
|
||||||
|
///
|
||||||
|
/// opus -> sonnet -> haiku one account, three tiers. A throttle usually
|
||||||
|
/// hits a tier, so stepping down often clears it.
|
||||||
|
/// -> kimi -> glm two separately funded accounts. Now an
|
||||||
|
/// Anthropic outage, not just a throttle, is
|
||||||
|
/// survivable.
|
||||||
|
/// -> local our own GPU. Nothing left to be down.
|
||||||
|
///
|
||||||
|
/// Every model id here was probed on this deployment 2026-08-09 and answered
|
||||||
|
/// 200: the four Anthropic tiers on the subscription, `kimi-k2.7-code` on
|
||||||
|
/// api.kimi.com/coding, `glm-4.7` on z.ai, and `ornith-fleet:9b` on the fleet.
|
||||||
|
/// Configured is not the same as working — see `preflight`, which re-checks
|
||||||
|
/// them at boot, because a link nobody exercises is discovered broken during
|
||||||
|
/// the outage it existed for.
|
||||||
|
///
|
||||||
|
/// The last link runs on our OWN hardware. Every other entry — and every other
|
||||||
|
/// link above it — depends on somebody else's account staying funded and
|
||||||
|
/// unthrottled; `local:` depends on a GPU in the next room. It is last because
|
||||||
|
/// it is the weakest model, and present because a chain whose every link is
|
||||||
|
/// external is not a fallback chain, it is one outage in a trench coat.
|
||||||
|
///
|
||||||
|
/// Note the model half contains a colon (`ornith-fleet:9b`), which is why
|
||||||
|
/// `resolve_provider` splits on the FIRST one only.
|
||||||
|
///
|
||||||
|
/// Override with `CLAWMATES_MODEL_FALLBACK` (comma-separated). An empty value
|
||||||
|
/// disables fallback and restores plain "503 and wait".
|
||||||
|
///
|
||||||
|
/// Ordered by the operator's model policy: sonnet-5 is the working tier, and
|
||||||
|
/// haiku sits BELOW it as a last-resort Anthropic link rather than as a peer —
|
||||||
|
/// a degraded answer beats a 503, but it must never be reached while a capable
|
||||||
|
/// model has capacity.
|
||||||
|
const DEFAULT_FALLBACK: &str = "claude-sonnet-5,claude-haiku-4-5-20251001,\
|
||||||
|
kimi:kimi-k2.7-code,glm:glm-4.7,local:ornith-fleet:9b";
|
||||||
|
|
||||||
|
/// The chain to walk after `requested`, with `requested` itself removed so a
|
||||||
|
/// capped model is never retried as its own fallback.
|
||||||
|
pub fn fallback_chain(requested: &str) -> Vec<String> {
|
||||||
|
let raw =
|
||||||
|
std::env::var("CLAWMATES_MODEL_FALLBACK").unwrap_or_else(|_| DEFAULT_FALLBACK.to_string());
|
||||||
|
raw.split(',')
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|m| !m.is_empty() && *m != requested.trim())
|
||||||
|
.map(str::to_string)
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Whether a failure means "this model has no capacity right now" as opposed
|
||||||
|
/// to "this request was wrong".
|
||||||
|
///
|
||||||
|
/// The distinction is the whole safety of the chain: walking it on a malformed
|
||||||
|
/// prompt would ask three models the same bad question and report the third
|
||||||
|
/// one's confusion, while walking it on a rate limit is exactly the point.
|
||||||
|
pub fn is_capacity_failure(err: &str) -> bool {
|
||||||
|
err.contains("rate_limit") || err.contains("429") || err.contains("credit balance")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One completion, stepping down `fallback_chain` when a model has no capacity.
|
||||||
|
///
|
||||||
|
/// Returns the text **and the model that actually produced it**. Callers must
|
||||||
|
/// persist that second value: a plan drafted by the third link in the chain and
|
||||||
|
/// filed as an opus plan is a silent quality change, which is the failure shape
|
||||||
|
/// this project keeps paying for. Every hop is logged.
|
||||||
|
pub async fn complete_with_fallback(
|
||||||
|
runtime: &cm_runtime::Runtime,
|
||||||
|
system: &str,
|
||||||
|
user: &str,
|
||||||
|
model: &str,
|
||||||
|
max_tokens: u32,
|
||||||
|
web_search: bool,
|
||||||
|
) -> Result<(String, String), String> {
|
||||||
|
let mut last = match complete_or(runtime, system, user, model, max_tokens, web_search).await {
|
||||||
|
Ok(text) => return Ok((text, model.to_string())),
|
||||||
|
Err(e) if is_capacity_failure(&e) => e,
|
||||||
|
// A real error. Do not launder it through two more models.
|
||||||
|
Err(e) => return Err(e),
|
||||||
|
};
|
||||||
|
for next in fallback_chain(model) {
|
||||||
|
eprintln!("model fallback: {model} has no capacity ({last}) — trying {next}");
|
||||||
|
match complete_or(runtime, system, user, &next, max_tokens, web_search).await {
|
||||||
|
Ok(text) => {
|
||||||
|
eprintln!("model fallback: {next} answered in place of {model}");
|
||||||
|
return Ok((text, next));
|
||||||
|
}
|
||||||
|
Err(e) if is_capacity_failure(&e) => last = e,
|
||||||
|
Err(e) => return Err(format!("fallback {next}: {e}")),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Err(last)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What a probe of one link found.
|
||||||
|
///
|
||||||
|
/// `Throttled` is deliberately NOT a failure. A 429 means the spec resolved, the
|
||||||
|
/// credential authenticated, and the provider simply had no capacity this
|
||||||
|
/// second — which is the exact condition the chain exists to route around. A
|
||||||
|
/// report that painted it red would train an operator to ignore the red.
|
||||||
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
pub enum LinkStatus {
|
||||||
|
Answered,
|
||||||
|
Throttled(String),
|
||||||
|
/// Never came back. Its own state because it is the one that used to make
|
||||||
|
/// the whole report vanish: with no timeout, a single hung provider meant
|
||||||
|
/// silence from the tool built to prevent silence.
|
||||||
|
TimedOut,
|
||||||
|
/// The spec named a provider the registry does not have, so
|
||||||
|
/// `resolve_provider` silently fell back to the DEFAULT provider. The link
|
||||||
|
/// would "work" while running on entirely the wrong model.
|
||||||
|
Unregistered,
|
||||||
|
Broken(String),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl LinkStatus {
|
||||||
|
pub fn usable(&self) -> bool {
|
||||||
|
matches!(self, LinkStatus::Answered | LinkStatus::Throttled(_))
|
||||||
|
}
|
||||||
|
fn label(&self) -> String {
|
||||||
|
match self {
|
||||||
|
LinkStatus::Answered => "ok".into(),
|
||||||
|
LinkStatus::Throttled(_) => "throttled (configured, no capacity now)".into(),
|
||||||
|
LinkStatus::TimedOut => {
|
||||||
|
format!("TIMED OUT after {}s — treat as down", PROBE_TIMEOUT.as_secs())
|
||||||
|
}
|
||||||
|
LinkStatus::Unregistered => "UNREGISTERED — resolves to the DEFAULT provider".into(),
|
||||||
|
LinkStatus::Broken(e) => format!("BROKEN: {}", e.chars().take(120).collect::<String>()),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Probe every link of the chain, head model included.
|
||||||
|
///
|
||||||
|
/// Eight tokens each, through the SAME path a real call takes, so it proves
|
||||||
|
/// resolution and reachability rather than that a string is present in a config
|
||||||
|
/// file. The distinction matters here more than usual: `resolve_provider` falls
|
||||||
|
/// back to the default provider for an unknown provider name, so a typo in
|
||||||
|
/// `kimi:` does not error — it quietly runs on Anthropic, and the chain reads
|
||||||
|
/// as five providers while being one.
|
||||||
|
/// Per-link ceiling. Generous on purpose: `complete_or` spends up to 30s in its
|
||||||
|
/// own backoff before giving up, so anything under that would report a merely
|
||||||
|
/// throttled link as hung.
|
||||||
|
const PROBE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(60);
|
||||||
|
|
||||||
|
pub async fn preflight(runtime: &cm_runtime::Runtime, head: &str) -> Vec<(String, LinkStatus)> {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
for spec in std::iter::once(head.to_string()).chain(fallback_chain(head)) {
|
||||||
|
// A qualified spec whose provider is missing resolves to the default —
|
||||||
|
// detected the same way `cross_provider_judge` does it, by asking what
|
||||||
|
// the model half came back as.
|
||||||
|
if spec.contains(':') {
|
||||||
|
// Unrouted specs come back WHOLE; routed ones come back as the part
|
||||||
|
// after the FIRST colon. Testing "does it still contain a colon"
|
||||||
|
// reads the same and is wrong: `local:ornith-fleet:9b` resolves
|
||||||
|
// correctly to model `ornith-fleet:9b`, which does. This probe
|
||||||
|
// reported a provider the server had just registered as
|
||||||
|
// UNREGISTERED on its first live run, which is how the same latent
|
||||||
|
// bug was found in `evaluator::cross_provider_judge`.
|
||||||
|
let (_, resolved) = runtime.resolve_provider(&spec);
|
||||||
|
if resolved == spec {
|
||||||
|
out.push((spec.clone(), LinkStatus::Unregistered));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// A non-empty system prompt. Kimi rejects an empty one outright —
|
||||||
|
// `400 the message at position 0 with role 'system' must not be empty` —
|
||||||
|
// so an empty probe reported a healthy provider as BROKEN on the first
|
||||||
|
// live run. The probe must look like the traffic it stands in for.
|
||||||
|
// NOT awaited here — the timeout has to wrap the FUTURE. Awaiting first
|
||||||
|
// and wrapping the result compiles, reads correctly, and bounds nothing.
|
||||||
|
let probe = complete_or(
|
||||||
|
runtime,
|
||||||
|
"You are a reachability probe.",
|
||||||
|
"Reply with exactly: OK",
|
||||||
|
&spec,
|
||||||
|
8,
|
||||||
|
false,
|
||||||
|
);
|
||||||
|
let status = match tokio::time::timeout(PROBE_TIMEOUT, probe).await {
|
||||||
|
Err(_) => LinkStatus::TimedOut,
|
||||||
|
Ok(Ok(_)) => LinkStatus::Answered,
|
||||||
|
Ok(Err(e)) if is_capacity_failure(&e) => LinkStatus::Throttled(e),
|
||||||
|
Ok(Err(e)) => LinkStatus::Broken(e),
|
||||||
|
};
|
||||||
|
// Emitted as it resolves, not collected and printed at the end. A later
|
||||||
|
// link that hangs must not be able to hide the ones already checked.
|
||||||
|
eprintln!("fallback chain: {spec:<32} {}", status.label());
|
||||||
|
out.push((spec, status));
|
||||||
|
}
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Probe the chain at boot and write the result to stderr.
|
||||||
|
///
|
||||||
|
/// Spawned rather than awaited, like `runtime_preflight`: this is diagnostic and
|
||||||
|
/// must never delay the server coming up. Loud when a link is unusable, because
|
||||||
|
/// the whole point of a chain is that nobody looks at it until the day it has to
|
||||||
|
/// work.
|
||||||
|
pub fn report_at_boot(runtime: cm_runtime::Runtime) {
|
||||||
|
tokio::spawn(async move {
|
||||||
|
let head = std::env::var("CLAWMATES_PREFLIGHT_HEAD")
|
||||||
|
.unwrap_or_else(|_| "claude-opus-5".to_string());
|
||||||
|
let links = preflight(&runtime, &head).await;
|
||||||
|
let bad: Vec<_> = links.iter().filter(|(_, s)| !s.usable()).collect();
|
||||||
|
eprintln!(
|
||||||
|
"fallback chain ({} link(s), {} usable):",
|
||||||
|
links.len(),
|
||||||
|
links.len() - bad.len()
|
||||||
|
);
|
||||||
|
for (spec, status) in &links {
|
||||||
|
eprintln!(" {spec:<32} {}", status.label());
|
||||||
|
}
|
||||||
|
if !bad.is_empty() {
|
||||||
|
eprintln!(
|
||||||
|
"fallback chain: WARNING — {} link(s) are NOT usable. The chain is \
|
||||||
|
shorter than it reads, and the shortfall only shows up during the \
|
||||||
|
outage it exists for.",
|
||||||
|
bad.len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Turn a `complete_or` failure into the right API error.
|
||||||
|
///
|
||||||
|
/// A rate limit that outlived the backoff is not a bug in this server, and
|
||||||
|
/// reporting it as one costs an operator a trip through the logs to find out
|
||||||
|
/// the answer was "wait". Measured: a bare 16-token probe with the same token
|
||||||
|
/// returned 429 with `x-should-retry: true` — Anthropic itself says try again.
|
||||||
|
pub fn as_api_error(err: &str) -> crate::error::ApiError {
|
||||||
|
if err.contains("rate_limit") || err.contains("429") {
|
||||||
|
return crate::error::ApiError::Unavailable(
|
||||||
|
"the Claude Code subscription is rate limited right now — this \
|
||||||
|
clears on its own; try again shortly"
|
||||||
|
.into(),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
crate::error::ApiError::Internal
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stream one request and collect its text, waiting out transient failures.
|
||||||
|
async fn complete_with(
|
||||||
|
provider: &cm_llm::AnthropicProvider,
|
||||||
|
system: &str,
|
||||||
|
user: &str,
|
||||||
|
model: &str,
|
||||||
|
max_tokens: u32,
|
||||||
|
web_search: bool,
|
||||||
|
) -> Result<String, String> {
|
||||||
|
let mut attempt = 0usize;
|
||||||
|
loop {
|
||||||
|
match attempt_once(provider, system, user, model, max_tokens, web_search).await {
|
||||||
|
Ok(text) => return Ok(text),
|
||||||
|
Err((stage, e)) => {
|
||||||
|
let Some(delay) = BACKOFF_SECS.get(attempt).copied().filter(|_| is_transient(&e))
|
||||||
|
else {
|
||||||
|
return Err(format!("subscription {stage}: {e}"));
|
||||||
|
};
|
||||||
|
eprintln!(
|
||||||
|
"subscription {stage}: {e} — retrying in {delay}s \
|
||||||
|
(attempt {} of {})",
|
||||||
|
attempt + 2,
|
||||||
|
BACKOFF_SECS.len() + 1
|
||||||
|
);
|
||||||
|
tokio::time::sleep(std::time::Duration::from_secs(delay)).await;
|
||||||
|
attempt += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One attempt. The collected text is discarded on failure, so a retry never
|
||||||
|
/// concatenates a partial answer onto a whole one.
|
||||||
|
async fn attempt_once(
|
||||||
|
provider: &cm_llm::AnthropicProvider,
|
||||||
|
system: &str,
|
||||||
|
user: &str,
|
||||||
|
model: &str,
|
||||||
|
max_tokens: u32,
|
||||||
|
web_search: bool,
|
||||||
|
) -> Result<String, (&'static str, cm_llm::LlmError)> {
|
||||||
|
use cm_llm::{ChatMessage, ChatRequest, ChatRole, ContentPart, LlmEvent, LlmProvider};
|
||||||
|
use futures::StreamExt as _;
|
||||||
|
|
||||||
|
let request = ChatRequest {
|
||||||
|
system: system.to_string(),
|
||||||
|
model: model.to_string(),
|
||||||
|
messages: vec![ChatMessage {
|
||||||
|
role: ChatRole::User,
|
||||||
|
parts: vec![ContentPart::text(user)],
|
||||||
|
}],
|
||||||
|
tools: vec![],
|
||||||
|
max_tokens,
|
||||||
|
web_search,
|
||||||
|
};
|
||||||
|
let mut stream = provider.stream(request).await.map_err(|e| ("call", e))?;
|
||||||
|
let mut text = String::new();
|
||||||
|
while let Some(event) = stream.next().await {
|
||||||
|
match event {
|
||||||
|
Ok(LlmEvent::TextDelta(t)) => text.push_str(&t),
|
||||||
|
Ok(_) => {}
|
||||||
|
Err(e) => return Err(("stream", e)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok(text)
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// Every server-side model call that should be on the subscription IS.
|
||||||
|
///
|
||||||
|
/// `validator_preflight` is the deliberate exception: it probes whatever
|
||||||
|
/// spec an operator configured (today `glm:glm-4.7`), and forcing it onto
|
||||||
|
/// Anthropic would make it prove the wrong thing — it exists to answer "is
|
||||||
|
/// the configured validator reachable".
|
||||||
|
/// The first version of this test grepped for the literal
|
||||||
|
/// `runtime.complete(` and passed while FOUR more call sites — the phase
|
||||||
|
/// planner, both swarm calls, and a second enhance path — still billed the
|
||||||
|
/// metered key. They were spelled `state.runtime` or wrapped across lines,
|
||||||
|
/// so the receiver name was never the thing to look for. Match the METHOD.
|
||||||
|
#[test]
|
||||||
|
fn no_server_side_call_silently_uses_the_metered_key() {
|
||||||
|
let sources = [
|
||||||
|
("routes/mission_roster.rs", include_str!("routes/mission_roster.rs")),
|
||||||
|
("routes/mission_plan.rs", include_str!("routes/mission_plan.rs")),
|
||||||
|
("routes/planner.rs", include_str!("routes/planner.rs")),
|
||||||
|
("routes/claws.rs", include_str!("routes/claws.rs")),
|
||||||
|
("swarm.rs", include_str!("swarm.rs")),
|
||||||
|
];
|
||||||
|
for (name, src) in sources {
|
||||||
|
assert!(
|
||||||
|
!src.contains(".complete("),
|
||||||
|
"{name} calls Runtime::complete directly — a bare model name there \
|
||||||
|
resolves to the DEFAULT provider, which is the metered API key. \
|
||||||
|
Use `subscription::complete_or`, which passes a `name:model` \
|
||||||
|
spec through untouched."
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// And the exception stays an exception, on purpose.
|
||||||
|
assert!(
|
||||||
|
include_str!("validator_preflight.rs").contains("runtime.complete("),
|
||||||
|
"validator_preflight must keep probing the CONFIGURED spec"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Only errors that can clear on their own are waited out.
|
||||||
|
///
|
||||||
|
/// The negative half is the point: a 400 or a 401 retried three times is a
|
||||||
|
/// 30-second hang ending in the identical message, which reads as a stall
|
||||||
|
/// rather than a bad request — the failure mode this project keeps hitting.
|
||||||
|
#[test]
|
||||||
|
fn a_wall_that_clears_is_waited_out_and_one_that_does_not_is_not() {
|
||||||
|
use cm_llm::LlmError;
|
||||||
|
let api = |s: &str| LlmError::Api(s.to_string());
|
||||||
|
|
||||||
|
assert!(is_transient(&api(
|
||||||
|
"429 Too Many Requests: {\"type\":\"rate_limit_error\"}"
|
||||||
|
)));
|
||||||
|
assert!(is_transient(&api("529: overloaded_error")));
|
||||||
|
assert!(is_transient(&api("503 Service Unavailable")));
|
||||||
|
assert!(is_transient(&LlmError::Transport("connection reset".into())));
|
||||||
|
|
||||||
|
// The exact error that started this: it never clears by waiting, it
|
||||||
|
// clears by moving to the other credential — which is now done.
|
||||||
|
assert!(!is_transient(&api(
|
||||||
|
"400 Bad Request: Your credit balance is too low"
|
||||||
|
)));
|
||||||
|
assert!(!is_transient(&api("401 Unauthorized: invalid x-api-key")));
|
||||||
|
assert!(!is_transient(&api("404 Not Found: model not found")));
|
||||||
|
assert!(!is_transient(&LlmError::Wire("bad json".into())));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Nobody hand-rolls their own Anthropic HTTP call.
|
||||||
|
///
|
||||||
|
/// `phase_summarizer` did — its own `reqwest` POST to `api.anthropic.com`
|
||||||
|
/// with `x-api-key: $ANTHROPIC_API_KEY`. No audit of `.complete(` call
|
||||||
|
/// sites could ever have found it, and it was the last thing on this
|
||||||
|
/// deployment still billing an account with no credit: every phase summary
|
||||||
|
/// died with "credit balance is too low" while the phases themselves ran.
|
||||||
|
/// A call site is only routable if it goes through a provider, so walk the
|
||||||
|
/// whole crate rather than a hand-listed set of files.
|
||||||
|
#[test]
|
||||||
|
fn no_module_talks_to_anthropic_behind_the_providers_back() {
|
||||||
|
fn walk(dir: &std::path::Path, out: &mut Vec<std::path::PathBuf>) {
|
||||||
|
for entry in std::fs::read_dir(dir).expect("readable source dir") {
|
||||||
|
let path = entry.expect("readable entry").path();
|
||||||
|
if path.is_dir() {
|
||||||
|
walk(&path, out);
|
||||||
|
} else if path.extension().is_some_and(|e| e == "rs") {
|
||||||
|
out.push(path);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let root = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("src");
|
||||||
|
let mut files = Vec::new();
|
||||||
|
walk(&root, &mut files);
|
||||||
|
assert!(files.len() > 20, "source walk found suspiciously few files");
|
||||||
|
|
||||||
|
for path in files {
|
||||||
|
// This module names the host in prose; it is the one that may.
|
||||||
|
if path.ends_with("subscription.rs") {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let src = std::fs::read_to_string(&path).expect("readable source");
|
||||||
|
for needle in ["api.anthropic.com", "\"x-api-key\""] {
|
||||||
|
assert!(
|
||||||
|
!src.contains(needle),
|
||||||
|
"{} contains {needle} — build the request through cm_llm and \
|
||||||
|
route it via `subscription::complete_or`, so credential \
|
||||||
|
choice and the capacity fallback live in ONE place",
|
||||||
|
path.display()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A model name may contain a colon, and "unregistered" must not mean that.
|
||||||
|
///
|
||||||
|
/// `resolve_provider` returns the spec unchanged when it does not recognise
|
||||||
|
/// the provider and the part after the FIRST colon when it does. The obvious
|
||||||
|
/// test — "does the model half still contain a colon" — reads the same and
|
||||||
|
/// is wrong the moment a model id has one. `ornith-fleet:9b` has one, and
|
||||||
|
/// the live preflight reported a provider the server had just registered as
|
||||||
|
/// UNREGISTERED. The identical bug was in `cross_provider_judge`, where it
|
||||||
|
/// would have refused a perfectly good independent judge.
|
||||||
|
#[test]
|
||||||
|
fn a_colon_in_the_model_name_is_not_a_missing_provider() {
|
||||||
|
// What `resolve_provider` returns in each case.
|
||||||
|
fn routed(spec: &str) -> &str {
|
||||||
|
spec.split_once(':').map(|(_, m)| m).unwrap_or(spec)
|
||||||
|
}
|
||||||
|
|
||||||
|
for spec in ["local:ornith-fleet:9b", "glm:glm-4.7", "kimi:kimi-k2.7-code"] {
|
||||||
|
assert_ne!(routed(spec), spec, "{spec} routed must not equal the whole spec");
|
||||||
|
}
|
||||||
|
// An unrecognised provider comes back WHOLE — the only true signal.
|
||||||
|
assert_eq!(routed("nosuch"), "nosuch");
|
||||||
|
// And the case that made the naive colon test look correct for so long.
|
||||||
|
assert!(routed("local:ornith-fleet:9b").contains(':'));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A throttled link is usable; an unregistered one is not.
|
||||||
|
///
|
||||||
|
/// The second is the dangerous one and the reason `preflight` checks
|
||||||
|
/// resolution separately from reachability. `resolve_provider` falls back to
|
||||||
|
/// the DEFAULT provider when it does not recognise a provider name, so a
|
||||||
|
/// typo in `kimi:` does not error — it quietly runs on Anthropic, and a
|
||||||
|
/// chain that reads as three accounts is really one. A reachability-only
|
||||||
|
/// probe would call that link green.
|
||||||
|
#[test]
|
||||||
|
fn only_a_link_that_could_never_answer_counts_as_unusable() {
|
||||||
|
assert!(LinkStatus::Answered.usable());
|
||||||
|
assert!(LinkStatus::Throttled("429 rate_limit".into()).usable());
|
||||||
|
|
||||||
|
assert!(!LinkStatus::Unregistered.usable());
|
||||||
|
assert!(!LinkStatus::Broken("401 invalid key".into()).usable());
|
||||||
|
|
||||||
|
// The labels must not read alike: "throttled" is a wait and
|
||||||
|
// "unregistered" is a config bug, and an operator acts differently on
|
||||||
|
// each.
|
||||||
|
assert!(LinkStatus::Throttled(String::new()).label().contains("configured"));
|
||||||
|
assert!(LinkStatus::Unregistered.label().contains("DEFAULT provider"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The chain never retries the capped model as its own fallback.
|
||||||
|
///
|
||||||
|
/// Without the filter, asking for haiku while haiku is capped would try
|
||||||
|
/// haiku, fail, and try haiku again — a chain that looks like resilience
|
||||||
|
/// and delivers none.
|
||||||
|
#[test]
|
||||||
|
fn the_chain_excludes_the_model_that_just_failed() {
|
||||||
|
// No env override in scope: this asserts the SHIPPED default.
|
||||||
|
let chain = fallback_chain("claude-opus-5");
|
||||||
|
assert_eq!(
|
||||||
|
chain,
|
||||||
|
vec![
|
||||||
|
"claude-sonnet-5",
|
||||||
|
"claude-haiku-4-5-20251001",
|
||||||
|
"kimi:kimi-k2.7-code",
|
||||||
|
"glm:glm-4.7",
|
||||||
|
"local:ornith-fleet:9b",
|
||||||
|
]
|
||||||
|
);
|
||||||
|
// Three providers behind five links. A chain that steps down three
|
||||||
|
// Anthropic tiers and stops is a tier ladder, not a fallback chain: one
|
||||||
|
// account being unreachable would end it.
|
||||||
|
let families: std::collections::BTreeSet<_> = chain
|
||||||
|
.iter()
|
||||||
|
.map(|m| m.split_once(':').map(|(p, _)| p).unwrap_or("anthropic"))
|
||||||
|
.collect();
|
||||||
|
assert!(
|
||||||
|
families.len() >= 3,
|
||||||
|
"the chain must span more than one account, got {families:?}"
|
||||||
|
);
|
||||||
|
// The last link must survive `resolve_provider`'s split, which takes the
|
||||||
|
// FIRST colon only — `local:ornith-fleet:9b` is provider `local`, model
|
||||||
|
// `ornith-fleet:9b`, and a split on the last colon would ask for a
|
||||||
|
// provider named `local:ornith-fleet`.
|
||||||
|
let last = chain.last().unwrap();
|
||||||
|
let (provider, model) = last.split_once(':').expect("a provider-qualified spec");
|
||||||
|
assert_eq!(provider, "local");
|
||||||
|
assert_eq!(model, "ornith-fleet:9b");
|
||||||
|
assert!(!fallback_chain("claude-haiku-4-5-20251001")
|
||||||
|
.iter()
|
||||||
|
.any(|m| m == "claude-haiku-4-5-20251001"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The chain is walked for "no capacity" and NOT for "bad request".
|
||||||
|
///
|
||||||
|
/// Walking it on a malformed prompt would ask three models the same bad
|
||||||
|
/// question and report the third one's confusion as the answer, burning
|
||||||
|
/// the two credentials that still work in order to hide the real error.
|
||||||
|
#[test]
|
||||||
|
fn only_a_capacity_failure_steps_down_the_chain() {
|
||||||
|
assert!(is_capacity_failure(
|
||||||
|
"subscription call: provider returned an error: 429 Too Many Requests"
|
||||||
|
));
|
||||||
|
assert!(is_capacity_failure("rate_limit_error"));
|
||||||
|
// The metered key's wall counts too — same meaning, different wording.
|
||||||
|
assert!(is_capacity_failure(
|
||||||
|
"400: Your credit balance is too low to access the Anthropic API"
|
||||||
|
));
|
||||||
|
|
||||||
|
assert!(!is_capacity_failure("400: messages.0: text content is empty"));
|
||||||
|
assert!(!is_capacity_failure("401: invalid x-api-key"));
|
||||||
|
assert!(!is_capacity_failure("404: model not found"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An operator's explicit provider choice is never hijacked.
|
||||||
|
///
|
||||||
|
/// The swarm worker model is a configured `name:model` spec. Routing that
|
||||||
|
/// onto the subscription would run someone's chosen Kimi or GLM model on
|
||||||
|
/// Anthropic and report success — the same silent-substitution bug as the
|
||||||
|
/// metered key, aimed the other way.
|
||||||
|
#[test]
|
||||||
|
fn a_provider_qualified_spec_is_left_alone() {
|
||||||
|
assert!(is_bare_model_name("claude-opus-4-8"));
|
||||||
|
assert!(is_bare_model_name("claude-haiku-4-5-20251001"));
|
||||||
|
assert!(!is_bare_model_name("kimi:kimi-k2.6"));
|
||||||
|
assert!(!is_bare_model_name("glm:glm-4.7"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A metered key in the OAuth slot must be REFUSED, not used.
|
||||||
|
///
|
||||||
|
/// Accepting it would authenticate, work, and bill the pay-as-you-go account
|
||||||
|
/// — the exact bill this module exists to stop, discovered weeks later when
|
||||||
|
/// it runs out mid-mission.
|
||||||
|
#[test]
|
||||||
|
fn only_a_setup_token_counts_as_the_subscription() {
|
||||||
|
assert!(is_subscription_token("sk-ant-oat01-abc"));
|
||||||
|
assert!(!is_subscription_token("sk-ant-api03-abc"));
|
||||||
|
assert!(!is_subscription_token(""));
|
||||||
|
assert!(!is_subscription_token("oat-but-not-anthropic"));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -86,6 +86,7 @@ fn step(
|
|||||||
output: output.into(),
|
output: output.into(),
|
||||||
gated: Vec::new(),
|
gated: Vec::new(),
|
||||||
tokens: 0,
|
tokens: 0,
|
||||||
|
spend: Default::default(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -122,9 +123,11 @@ pub async fn run_swarm_job(
|
|||||||
let worker_model = resolve_worker_model(&job.worker_model);
|
let worker_model = resolve_worker_model(&job.worker_model);
|
||||||
|
|
||||||
// 1) PLAN — Opus decomposes the goal into worker tasks.
|
// 1) PLAN — Opus decomposes the goal into worker tasks.
|
||||||
|
// This record is written BEFORE the call, so it cannot name the model that
|
||||||
|
// answers. The record after the call can, and does.
|
||||||
records.push(step(
|
records.push(step(
|
||||||
"planner",
|
"planner",
|
||||||
"planner:opus",
|
"planner",
|
||||||
StepPhase::Plan,
|
StepPhase::Plan,
|
||||||
format!("Planning tasks for: {goal}"),
|
format!("Planning tasks for: {goal}"),
|
||||||
));
|
));
|
||||||
@@ -137,8 +140,17 @@ pub async fn run_swarm_job(
|
|||||||
"GOAL:\n{goal}\n\nCHECKLIST each task's output must satisfy:\n{}{want}",
|
"GOAL:\n{goal}\n\nCHECKLIST each task's output must satisfy:\n{}{want}",
|
||||||
checklist_lines(&checklist)
|
checklist_lines(&checklist)
|
||||||
);
|
);
|
||||||
let plan_raw = runtime
|
// The recorded role says which model ANSWERED. When opus is capped the
|
||||||
.complete(PLAN_SYSTEM, &plan_user, "claude-opus-4-8", 4000, false)
|
// chain steps down, and a step labelled "planner:opus" that GLM wrote is a
|
||||||
|
// lie in the one place an operator looks to explain a bad decomposition.
|
||||||
|
let (plan_raw, plan_model) = crate::subscription::complete_with_fallback(
|
||||||
|
runtime,
|
||||||
|
PLAN_SYSTEM,
|
||||||
|
&plan_user,
|
||||||
|
"claude-opus-5",
|
||||||
|
4000,
|
||||||
|
false,
|
||||||
|
)
|
||||||
.await?;
|
.await?;
|
||||||
let tasks: Vec<String> = extract_json(&plan_raw)
|
let tasks: Vec<String> = extract_json(&plan_raw)
|
||||||
.and_then(|v| {
|
.and_then(|v| {
|
||||||
@@ -154,7 +166,7 @@ pub async fn run_swarm_job(
|
|||||||
}
|
}
|
||||||
records.push(step(
|
records.push(step(
|
||||||
"planner",
|
"planner",
|
||||||
"planner:opus",
|
format!("planner:{plan_model}"),
|
||||||
StepPhase::Plan,
|
StepPhase::Plan,
|
||||||
format!(
|
format!(
|
||||||
"Decomposed into {} tasks. Workers: {worker_model}. Verifier: claude-opus-4-8.",
|
"Decomposed into {} tasks. Workers: {worker_model}. Verifier: claude-opus-4-8.",
|
||||||
@@ -177,8 +189,10 @@ pub async fn run_swarm_job(
|
|||||||
let mut still: Vec<(usize, String)> = Vec::new();
|
let mut still: Vec<(usize, String)> = Vec::new();
|
||||||
let mut rejected = 0usize;
|
let mut rejected = 0usize;
|
||||||
for (idx, task) in pending.iter() {
|
for (idx, task) in pending.iter() {
|
||||||
let out = runtime
|
// `worker_model` may be a `name:model` spec the operator chose;
|
||||||
.complete(&wsys, task, &worker_model, 4000, true)
|
// `complete_or` passes those straight through untouched.
|
||||||
|
let out =
|
||||||
|
crate::subscription::complete_or(runtime, &wsys, task, &worker_model, 4000, true)
|
||||||
.await
|
.await
|
||||||
.unwrap_or_else(|e| format!("worker error: {e}"));
|
.unwrap_or_else(|e| format!("worker error: {e}"));
|
||||||
records.push(step(
|
records.push(step(
|
||||||
@@ -190,9 +204,16 @@ pub async fn run_swarm_job(
|
|||||||
ckpt(pool, id, &records, &totals).await;
|
ckpt(pool, id, &records, &totals).await;
|
||||||
|
|
||||||
let vuser = format!("TASK:\n{task}\n\nWORKER OUTPUT:\n{out}");
|
let vuser = format!("TASK:\n{task}\n\nWORKER OUTPUT:\n{out}");
|
||||||
let v_raw = runtime
|
let v_raw = crate::subscription::complete_with_fallback(
|
||||||
.complete(&vsys, &vuser, "claude-opus-4-8", 1200, true)
|
runtime,
|
||||||
|
&vsys,
|
||||||
|
&vuser,
|
||||||
|
"claude-opus-5",
|
||||||
|
1200,
|
||||||
|
true,
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
|
.map(|(text, _)| text)
|
||||||
.unwrap_or_default();
|
.unwrap_or_default();
|
||||||
let v = extract_json(&v_raw);
|
let v = extract_json(&v_raw);
|
||||||
let passed = v
|
let passed = v
|
||||||
|
|||||||
@@ -39,6 +39,7 @@ pub struct Marker {
|
|||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||||
pub enum MarkerKind {
|
pub enum MarkerKind {
|
||||||
Task,
|
Task,
|
||||||
|
PlanComplete,
|
||||||
Work,
|
Work,
|
||||||
Handoff,
|
Handoff,
|
||||||
TestPass,
|
TestPass,
|
||||||
@@ -53,7 +54,11 @@ impl MarkerKind {
|
|||||||
/// motion — the UPSERT layer may still overwrite prior states.
|
/// motion — the UPSERT layer may still overwrite prior states.
|
||||||
pub fn status(&self) -> &'static str {
|
pub fn status(&self) -> &'static str {
|
||||||
match self {
|
match self {
|
||||||
MarkerKind::Task => "created",
|
// The planner finished specifying; no work has started, so the item
|
||||||
|
// is in the same state a fresh TASK leaves it in. A distinct status
|
||||||
|
// would need a column value the UI does not render, and inventing
|
||||||
|
// one to look complete is how a status stops meaning anything.
|
||||||
|
MarkerKind::Task | MarkerKind::PlanComplete => "created",
|
||||||
MarkerKind::Work => "working",
|
MarkerKind::Work => "working",
|
||||||
MarkerKind::Handoff | MarkerKind::TestPass | MarkerKind::ReviewApprove => "validating",
|
MarkerKind::Handoff | MarkerKind::TestPass | MarkerKind::ReviewApprove => "validating",
|
||||||
MarkerKind::TestFail | MarkerKind::ReviewBlock => "failed",
|
MarkerKind::TestFail | MarkerKind::ReviewBlock => "failed",
|
||||||
@@ -75,12 +80,27 @@ pub fn parse(text: &str) -> Vec<Marker> {
|
|||||||
out
|
out
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// `INT-` followed by at least one digit and nothing else.
|
||||||
|
fn is_int_id(id: &str) -> bool {
|
||||||
|
match id.strip_prefix("INT-") {
|
||||||
|
Some(rest) => !rest.is_empty() && rest.chars().all(|c| c.is_ascii_digit()),
|
||||||
|
None => false,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
fn parse_line(line: &str) -> Option<Marker> {
|
fn parse_line(line: &str) -> Option<Marker> {
|
||||||
// Match `<KIND>: INT-NN` (rest optional). Strict on the colon and
|
// Match `<KIND>: INT-NN` (rest optional). Strict on the colon and
|
||||||
// the INT- prefix — anything laxer starts matching prose.
|
// the INT- prefix — anything laxer starts matching prose.
|
||||||
let (kind_str, rest) = line.split_once(':')?;
|
let (kind_str, rest) = line.split_once(':')?;
|
||||||
let kind = match kind_str.trim() {
|
let kind = match kind_str.trim() {
|
||||||
"TASK" => MarkerKind::Task,
|
"TASK" => MarkerKind::Task,
|
||||||
|
// Documented in `skills/foundation/int-xx-marker-protocol.md` since the
|
||||||
|
// skill was written, and never implemented here. Agents that followed
|
||||||
|
// the skill exactly emitted it and were silently ignored — observed on
|
||||||
|
// a live mission, found by the Skill-Use measurement. Implemented
|
||||||
|
// rather than removed from the skill: the planner needs a way to say
|
||||||
|
// it is done specifying, and agents already emit this one.
|
||||||
|
"PLAN_COMPLETE" => MarkerKind::PlanComplete,
|
||||||
"WORK" => MarkerKind::Work,
|
"WORK" => MarkerKind::Work,
|
||||||
"HANDOFF" => MarkerKind::Handoff,
|
"HANDOFF" => MarkerKind::Handoff,
|
||||||
"TEST_PASS" => MarkerKind::TestPass,
|
"TEST_PASS" => MarkerKind::TestPass,
|
||||||
@@ -95,10 +115,16 @@ fn parse_line(line: &str) -> Option<Marker> {
|
|||||||
Some((a, b)) => (a, Some(b.trim())),
|
Some((a, b)) => (a, Some(b.trim())),
|
||||||
None => (rest, None),
|
None => (rest, None),
|
||||||
};
|
};
|
||||||
if !id_tok.starts_with("INT-") {
|
let int_id = id_tok.trim_end_matches(&[',', ';', '.'][..]).to_string();
|
||||||
|
// Strictly `INT-<digits>`. `starts_with("INT-")` alone accepted range forms
|
||||||
|
// like `INT-01..02`, which parse into an id matching no real item — so a
|
||||||
|
// task card appeared for something that did not exist while the two items
|
||||||
|
// it was meant to cover stayed open. Observed live. Rejecting is right:
|
||||||
|
// the marker is ignored, which is visible, instead of creating a plausible
|
||||||
|
// row, which is not.
|
||||||
|
if !is_int_id(&int_id) {
|
||||||
return None;
|
return None;
|
||||||
}
|
}
|
||||||
let int_id = id_tok.trim_end_matches(&[',', ';', '.'][..]).to_string();
|
|
||||||
// Title: after the id + any of ` — / – / - ` separators
|
// Title: after the id + any of ` — / – / - ` separators
|
||||||
let title = tail.and_then(|t| {
|
let title = tail.and_then(|t| {
|
||||||
let t = t.trim_start_matches(['—', '–', '-', ':'].as_slice()).trim();
|
let t = t.trim_start_matches(['—', '–', '-', ':'].as_slice()).trim();
|
||||||
|
|||||||
@@ -69,6 +69,11 @@ struct TemplateRoleFile {
|
|||||||
skills: Vec<String>,
|
skills: Vec<String>,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
brain_seed: Option<String>,
|
brain_seed: Option<String>,
|
||||||
|
/// Which model this role's claw runs on. Omitted means the mint's default,
|
||||||
|
/// which is what every authored template does today — so adding the field
|
||||||
|
/// changes nothing until a template uses it.
|
||||||
|
#[serde(default)]
|
||||||
|
model: Option<String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
fn templates_dir() -> PathBuf {
|
fn templates_dir() -> PathBuf {
|
||||||
@@ -137,6 +142,7 @@ async fn load_one(pool: &PgPool, path: &std::path::Path) -> Result<String, Strin
|
|||||||
system_prompt: &r.system_prompt,
|
system_prompt: &r.system_prompt,
|
||||||
skills: r.skills.clone(),
|
skills: r.skills.clone(),
|
||||||
brain_seed: r.brain_seed.as_deref(),
|
brain_seed: r.brain_seed.as_deref(),
|
||||||
|
model: r.model.as_deref(),
|
||||||
})
|
})
|
||||||
.collect();
|
.collect();
|
||||||
|
|
||||||
@@ -300,6 +306,34 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// EVERY referenced name must resolve to an authored skill.
|
||||||
|
///
|
||||||
|
/// The other direction, and the one that was missing. Both existing tests
|
||||||
|
/// assert `authored ⊆ referenced` — true of all 30 authored skills, so both
|
||||||
|
/// passed while 55 of 85 bindings resolved to nothing and ten roles ran with
|
||||||
|
/// an empty context bundle.
|
||||||
|
///
|
||||||
|
/// The old comment on the test below called the gap "deliberately
|
||||||
|
/// aspirational". An aspirational binding is indistinguishable at runtime
|
||||||
|
/// from a typo: `get_by_name` returns Ok(None), the loader logs a line
|
||||||
|
/// nobody reads, and the role ships without the instructions its prompt
|
||||||
|
/// assumes it has. If a skill is worth naming it is worth authoring, and if
|
||||||
|
/// it is not, the name should not be in the template.
|
||||||
|
#[test]
|
||||||
|
fn every_referenced_skill_resolves_to_an_authored_one() {
|
||||||
|
let mut authored = HashSet::new();
|
||||||
|
authored_skill_names(&repo_root().join("skills"), &mut authored);
|
||||||
|
let referenced = referenced_skill_names();
|
||||||
|
let mut missing: Vec<_> = referenced.difference(&authored).cloned().collect();
|
||||||
|
missing.sort();
|
||||||
|
assert!(
|
||||||
|
missing.is_empty(),
|
||||||
|
"{} referenced skill(s) bind to nothing — the role gets no instructions \
|
||||||
|
and nothing errors: {missing:#?}",
|
||||||
|
missing.len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
/// A referenced name that matches no authored skill binds to nothing. Some
|
/// A referenced name that matches no authored skill binds to nothing. Some
|
||||||
/// are deliberately aspirational, so this asserts the *resolvable* ones
|
/// are deliberately aspirational, so this asserts the *resolvable* ones
|
||||||
/// stay resolvable rather than demanding every name exist.
|
/// stay resolvable rather than demanding every name exist.
|
||||||
@@ -316,3 +350,86 @@ mod tests {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod bundle_tests {
|
||||||
|
use std::collections::HashSet;
|
||||||
|
|
||||||
|
fn repo() -> std::path::PathBuf {
|
||||||
|
std::path::Path::new(env!("CARGO_MANIFEST_DIR"))
|
||||||
|
.join("../..")
|
||||||
|
.canonicalize()
|
||||||
|
.expect("repo root")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every `mcp_bundles` name a template asks for must be one the runtime
|
||||||
|
/// config actually defines.
|
||||||
|
///
|
||||||
|
/// This was harmless while `provision_claw` wrote a constant bundle list
|
||||||
|
/// and ignored the templates. It is not harmless now that the list is
|
||||||
|
/// honoured: an undefined name is a capability the agent is told it has and
|
||||||
|
/// does not, which is the same failure as an unresolved skill binding one
|
||||||
|
/// layer down. `gitea_forge` was named by seven team templates, one
|
||||||
|
/// workflow recipe, the auto-provision path and a user-selectable dropdown,
|
||||||
|
/// and defined nowhere.
|
||||||
|
#[test]
|
||||||
|
fn every_named_mcp_bundle_is_defined_by_the_runtime_config() {
|
||||||
|
let cfg = std::fs::read_to_string(
|
||||||
|
repo().join("deploy/clawmates-runtime/agent.config.example.toml"),
|
||||||
|
)
|
||||||
|
.expect("runtime config");
|
||||||
|
let defined: HashSet<String> = cfg
|
||||||
|
.lines()
|
||||||
|
.filter_map(|l| l.trim().strip_prefix("[mcp_bundles."))
|
||||||
|
.filter_map(|r| r.strip_suffix(']'))
|
||||||
|
.map(|s| s.to_string())
|
||||||
|
.collect();
|
||||||
|
assert!(
|
||||||
|
defined.contains("clawmates_door"),
|
||||||
|
"parsed no bundles from the runtime config — the parser, not the \
|
||||||
|
templates, is what broke"
|
||||||
|
);
|
||||||
|
|
||||||
|
let mut missing: Vec<String> = Vec::new();
|
||||||
|
for dir in ["templates/teams", "templates/workflows"] {
|
||||||
|
for entry in std::fs::read_dir(repo().join(dir)).expect("template dir") {
|
||||||
|
let path = entry.expect("entry").path();
|
||||||
|
if path.extension().and_then(|e| e.to_str()) != Some("toml") {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let body = std::fs::read_to_string(&path).expect("read template");
|
||||||
|
for line in body.lines() {
|
||||||
|
let t = line.trim();
|
||||||
|
// Skip comments: several deliberately NAME a bundle while
|
||||||
|
// explaining that it is not delivered.
|
||||||
|
if t.starts_with('#') || !t.starts_with("mcp_bundles") {
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let Some(inner) = t.split_once('[').and_then(|(_, r)| r.rsplit_once(']'))
|
||||||
|
else {
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
for name in inner.0.split(',') {
|
||||||
|
let name = name.trim().trim_matches('"');
|
||||||
|
if !name.is_empty() && !defined.contains(name) {
|
||||||
|
missing.push(format!(
|
||||||
|
"{}: {name}",
|
||||||
|
path.file_name().unwrap().to_string_lossy()
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
missing.sort();
|
||||||
|
missing.dedup();
|
||||||
|
assert!(
|
||||||
|
missing.is_empty(),
|
||||||
|
"{} template(s) name an MCP bundle the runtime does not define, so \
|
||||||
|
the agent is provisioned with a capability that resolves to \
|
||||||
|
nothing:\n {}",
|
||||||
|
missing.len(),
|
||||||
|
missing.join("\n ")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -7,11 +7,22 @@
|
|||||||
//! gateway, opens `/ws/chat?agent=<alias>`, sends the role+task+context prompt,
|
//! gateway, opens `/ws/chat?agent=<alias>`, sends the role+task+context prompt,
|
||||||
//! and streams the turn's events back into a [`TurnOutcome`].
|
//! and streams the turn's events back into a [`TurnOutcome`].
|
||||||
//!
|
//!
|
||||||
//! **§15 by construction:** the agents are provisioned tool-free (every
|
//! **These agents are NOT tool-free.** That claim stood here for months and is
|
||||||
//! sensitive capability is a gated Clawmates MCP tool — the "door"), so a turn
|
//! false — see `docs/TOOL-CALL-ARCHITECTURE.md`. It was inferred from a frame
|
||||||
//! takes no sandbox-leaving action here. If the gateway nonetheless emits an
|
//! stream that carried no tool events, and the emptiness has a different cause:
|
||||||
//! `approval_request`, we record it as a **blocked** `GatedAction` and end the
|
//! `claude_cli` runs `claude -p --output-format json`, which returns a single
|
||||||
//! turn — we never auto-approve.
|
//! final result object, and the provider hardcodes `tool_calls: Vec::new()`.
|
||||||
|
//! The agent calls Claude Code's own tools; the transport discards them.
|
||||||
|
//! `--output-format stream-json` emits `tool_use`/`tool_result` blocks —
|
||||||
|
//! verified against the deployed Claude Code 2.1.228.
|
||||||
|
//!
|
||||||
|
//! The door-shaped provider that WOULD make this true (`--mcp-config` +
|
||||||
|
//! `--disallowedTools` on the natives) is built and documented in
|
||||||
|
//! `agent.config.example.toml`, and is not deployed: mission claws bind to
|
||||||
|
//! `claude_cli.default`, which sets none of it.
|
||||||
|
//!
|
||||||
|
//! If the gateway emits an `approval_request` we still record it as a
|
||||||
|
//! **blocked** `GatedAction` and end the turn — we never auto-approve.
|
||||||
|
|
||||||
use std::collections::HashMap;
|
use std::collections::HashMap;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
@@ -24,17 +35,60 @@ use tokio::sync::Mutex;
|
|||||||
use tokio_tungstenite::connect_async;
|
use tokio_tungstenite::connect_async;
|
||||||
use tokio_tungstenite::tungstenite::Message;
|
use tokio_tungstenite::tungstenite::Message;
|
||||||
|
|
||||||
/// Overall wall-clock budget for draining one turn's event stream. Must
|
/// Overall wall-clock budget for draining one turn's event stream.
|
||||||
/// exceed the daemon's own claude_cli provider timeout (600s on gw-04
|
///
|
||||||
/// via ZEROCLAW_providers__models__claude_cli__default__timeout_ms) —
|
/// A turn is an agent LOOP, not one model call. Each call inside it is bounded
|
||||||
/// otherwise the executor kills the ws before the daemon can reply and
|
/// separately by the daemon — `claude_cli`'s `timeout_secs`, 600s on gw-04 —
|
||||||
/// we see a phantom "turn timed out" while the daemon still logs a
|
/// so this has to cover however many calls the loop makes, not one of them.
|
||||||
/// successful llm response coming back. 700s gives 100s of headroom so
|
///
|
||||||
/// a daemon that just barely made it under its own limit doesn't lose
|
/// It was 700s, which is 100s more than a single call may take. MEASURED: a
|
||||||
/// its answer here.
|
/// healthy research turn is ~157s, but a throttled one blew the budget with one
|
||||||
const TURN_TIMEOUT: Duration = Duration::from_secs(700);
|
/// slow call plus a second, and the executor killed it mid-flight after 11m43s
|
||||||
|
/// with no error from the daemon — because nothing had failed yet. All the
|
||||||
|
/// operator got was "turn timed out".
|
||||||
|
///
|
||||||
|
/// An hour matches the phase's own budget. A genuinely stuck CALL is still
|
||||||
|
/// caught at 600s by the daemon and surfaces as a real error; this only stops
|
||||||
|
/// us killing turns that are working, slowly.
|
||||||
|
const TURN_TIMEOUT: Duration = Duration::from_secs(3600);
|
||||||
|
|
||||||
/// Drives ZeroClaw role-agents (in one container) to execute topology turns.
|
/// Drives ZeroClaw role-agents (in one container) to execute topology turns.
|
||||||
|
/// Cap on the pinned-skill text injected into one mission turn.
|
||||||
|
///
|
||||||
|
/// Skill bodies average ~3.5 KB and pinning is `idx < 2 || foundation`, so a
|
||||||
|
/// role lands near 7-10 KB. The cap exists for the role that grows a long
|
||||||
|
/// foundation set, and it is stated in the prompt when it fires.
|
||||||
|
pub(crate) const MAX_PINNED_SKILL_BYTES: usize = 24_000;
|
||||||
|
|
||||||
|
/// The line that introduces each skill in a prompt.
|
||||||
|
///
|
||||||
|
/// NOT a markdown heading. The first version used `## <name>`, and skill bodies
|
||||||
|
/// are markdown that contain their own `##` headings — so anything reading the
|
||||||
|
/// prompt back counted every section of every body as a separate skill. A live
|
||||||
|
/// mission scored "Sizing heuristic" and "The output shape" as skills, which is
|
||||||
|
/// what surfaced it.
|
||||||
|
///
|
||||||
|
/// This marker cannot occur inside a body, so the prompt stays parseable by
|
||||||
|
/// whatever reads it later. Skills are written by one function
|
||||||
|
/// ([`render_pinned_skill`]) for the same reason: two renderers would drift and
|
||||||
|
/// the reader would silently match only one.
|
||||||
|
pub const SKILL_MARKER: &str = "--- SKILL: ";
|
||||||
|
|
||||||
|
/// One skill, rendered for a prompt.
|
||||||
|
pub fn render_pinned_skill(name: &str, body: &str) -> String {
|
||||||
|
format!("\n{SKILL_MARKER}{name} ---\n{body}\n")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The skill names a rendered prompt delivered.
|
||||||
|
pub fn skill_names_in(prompt: &str) -> Vec<String> {
|
||||||
|
prompt
|
||||||
|
.lines()
|
||||||
|
.filter_map(|l| l.trim().strip_prefix(SKILL_MARKER))
|
||||||
|
.map(|rest| rest.trim_end_matches(" ---").trim().to_string())
|
||||||
|
.filter(|n| !n.is_empty())
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
pub struct ZeroClawDriveExecutor {
|
pub struct ZeroClawDriveExecutor {
|
||||||
/// Gateway base URL, e.g. `http://127.0.0.1:42617`.
|
/// Gateway base URL, e.g. `http://127.0.0.1:42617`.
|
||||||
gateway_url: String,
|
gateway_url: String,
|
||||||
@@ -47,6 +101,58 @@ pub struct ZeroClawDriveExecutor {
|
|||||||
/// Bearer token, paired lazily and reused across turns.
|
/// Bearer token, paired lazily and reused across turns.
|
||||||
token: Arc<Mutex<Option<String>>>,
|
token: Arc<Mutex<Option<String>>>,
|
||||||
http: reqwest::Client,
|
http: reqwest::Client,
|
||||||
|
/// Where this executor's turns record what they did. `None` on every path
|
||||||
|
/// that is not a mission phase (the governor, the door, the evaluator) —
|
||||||
|
/// those turns belong to no phase and have nothing to attribute to.
|
||||||
|
tap: Option<Arc<MissionTap>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Where a turn's tool activity is written, and what it belongs to.
|
||||||
|
///
|
||||||
|
/// Carried on the executor rather than passed per turn because `TurnRequest`
|
||||||
|
/// is the shared orchestrator contract: threading a mission id through it would
|
||||||
|
/// put mission concepts into every tier that has no missions.
|
||||||
|
pub struct MissionTap {
|
||||||
|
pub pool: sqlx::PgPool,
|
||||||
|
/// Which workspace's live feed these frames belong to. Every subscriber is
|
||||||
|
/// workspace-scoped, so a frame without this could not be routed.
|
||||||
|
pub workspace_id: uuid::Uuid,
|
||||||
|
pub mission_id: uuid::Uuid,
|
||||||
|
pub phase_id: Option<uuid::Uuid>,
|
||||||
|
pub run_id: Option<uuid::Uuid>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One tool call, as the frame stream reported it.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct ToolCall {
|
||||||
|
pub tool: String,
|
||||||
|
/// The path the tool's **arguments** named, if any. Never extracted from a
|
||||||
|
/// prose summary — see [`crate::mission_events::tool_path`].
|
||||||
|
pub path: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What one turn's frames said about the work, beside its text.
|
||||||
|
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||||||
|
pub struct ToolTrace {
|
||||||
|
pub calls: Vec<ToolCall>,
|
||||||
|
/// Frame `type` values this drain did not recognise, counted.
|
||||||
|
///
|
||||||
|
/// Shipped in the same change as the tap on purpose: the frame name was
|
||||||
|
/// taken from a comment in this file rather than from a captured frame. If
|
||||||
|
/// the runtime called it something else, the tap would record nothing and
|
||||||
|
/// nothing anywhere would error — the World would simply stay as sparse as
|
||||||
|
/// it was before.
|
||||||
|
///
|
||||||
|
/// MEASURED on gw-04 (v0.8.3, 2026-08-11): a mission turn's stream carried
|
||||||
|
/// `chunk`, `done` and `session_start` and no tool frames at all. That is
|
||||||
|
/// not a protocol mismatch — `tool_call` is in the deployed binary
|
||||||
|
/// (`zeroclaw-gateway/src/ws.rs` emits `{"type":"tool_call","id","name",
|
||||||
|
/// "args"}`) — and it is NOT that the agents are tool-free, which is what
|
||||||
|
/// this comment used to say. `claude_cli` asks for `--output-format json`,
|
||||||
|
/// so the subprocess's tool calls never reach the gateway to be framed.
|
||||||
|
/// The histogram still does its job: it distinguishes "no frames" from
|
||||||
|
/// "frames we do not recognise", and the answer was the former.
|
||||||
|
pub unmatched: std::collections::BTreeMap<String, u32>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl ZeroClawDriveExecutor {
|
impl ZeroClawDriveExecutor {
|
||||||
@@ -64,9 +170,17 @@ impl ZeroClawDriveExecutor {
|
|||||||
default_alias,
|
default_alias,
|
||||||
token: Arc::new(Mutex::new(None)),
|
token: Arc::new(Mutex::new(None)),
|
||||||
http: reqwest::Client::new(),
|
http: reqwest::Client::new(),
|
||||||
|
tap: None,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Attach the mission this executor's turns belong to, so their tool calls
|
||||||
|
/// are recorded. Without it the executor behaves exactly as it did.
|
||||||
|
pub fn with_tap(mut self, tap: MissionTap) -> Self {
|
||||||
|
self.tap = Some(Arc::new(tap));
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
/// Build from the environment:
|
/// Build from the environment:
|
||||||
/// - `ZEROCLAW_GATEWAY_URL` (required) e.g. `http://127.0.0.1:42617`
|
/// - `ZEROCLAW_GATEWAY_URL` (required) e.g. `http://127.0.0.1:42617`
|
||||||
/// - `ZEROCLAW_TOKEN` (preferred) a durable bearer token — pair once
|
/// - `ZEROCLAW_TOKEN` (preferred) a durable bearer token — pair once
|
||||||
@@ -113,6 +227,21 @@ impl ZeroClawDriveExecutor {
|
|||||||
/// new one-time code at startup. The env-derived ZEROCLAW_TOKEN
|
/// new one-time code at startup. The env-derived ZEROCLAW_TOKEN
|
||||||
/// is ignored (belongs to the shared runtime) so the lazy pair
|
/// is ignored (belongs to the shared runtime) so the lazy pair
|
||||||
/// path runs and issues a bearer for this specific gateway.
|
/// path runs and issues a bearer for this specific gateway.
|
||||||
|
/// Reuse a token that was already paired and persisted.
|
||||||
|
///
|
||||||
|
/// The pairing code is single-use, so a restarted server cannot pair again:
|
||||||
|
/// it gets 403 and the mission is unrecoverable. Seeding the cache from
|
||||||
|
/// `missions.runtime_token` is what makes a mission survive a restart.
|
||||||
|
pub fn with_token(self, token: Option<String>) -> Self {
|
||||||
|
if let Some(t) = token.filter(|t| !t.trim().is_empty()) {
|
||||||
|
// try_lock: this runs at construction, before any turn holds it.
|
||||||
|
if let Ok(mut g) = self.token.try_lock() {
|
||||||
|
*g = Some(t);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
self
|
||||||
|
}
|
||||||
|
|
||||||
pub fn from_env_for_gateway_with_code(
|
pub fn from_env_for_gateway_with_code(
|
||||||
gateway_url: String,
|
gateway_url: String,
|
||||||
pairing_code: String,
|
pairing_code: String,
|
||||||
@@ -174,9 +303,145 @@ impl ZeroClawDriveExecutor {
|
|||||||
.ok_or_else(|| OrchestratorError::Executor("pair response had no token".into()))?
|
.ok_or_else(|| OrchestratorError::Executor("pair response had no token".into()))?
|
||||||
.to_string();
|
.to_string();
|
||||||
*guard = Some(token.clone());
|
*guard = Some(token.clone());
|
||||||
|
// Persist it. The code we just spent cannot be used again, so if this
|
||||||
|
// token only ever lives in memory the next server process has no way
|
||||||
|
// back in — that is the 403 that killed a 93k-token research phase.
|
||||||
|
// Best-effort: failing to save must not fail a turn that just paired
|
||||||
|
// successfully; the cost is that a restart before the next write
|
||||||
|
// re-opens the original hole.
|
||||||
|
if let Some(tap) = self.tap.as_ref() {
|
||||||
|
if let Err(e) = sqlx::query("UPDATE missions SET runtime_token = $1 WHERE id = $2")
|
||||||
|
.bind(&token)
|
||||||
|
.bind(tap.mission_id)
|
||||||
|
.execute(&tap.pool)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!(
|
||||||
|
"topology_exec: could not persist runtime token for mission {}: {e}",
|
||||||
|
tap.mission_id
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
Ok(token)
|
Ok(token)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The pinned skills for the claw behind `alias`, rendered for the prompt.
|
||||||
|
///
|
||||||
|
/// Missions had NO path to a skill. The catalogue's only delivery channel
|
||||||
|
/// is the `clawmates_skills` MCP server, and a mission agent cannot reach
|
||||||
|
/// it for three independent reasons: `provision_claw` wrote a constant
|
||||||
|
/// bundle list, the runtime config defines no such bundle, and mission
|
||||||
|
/// claws run on `claude_cli`, which is text-only and cannot surface a tool
|
||||||
|
/// call at all. Two doc comments in `cm-runtime` describe the mission path
|
||||||
|
/// as already having this contract. It never did — so every skill authored
|
||||||
|
/// for a mission role was unreachable prose, and no measurement of whether
|
||||||
|
/// skills fire could have returned anything but zero.
|
||||||
|
///
|
||||||
|
/// Bodies or an index, depending on the mission's arm — see
|
||||||
|
/// [`crate::skill_delivery`]. Bodies were once the only honest option:
|
||||||
|
/// there was no tool on the mission path that could fetch one, so an index
|
||||||
|
/// would have advertised a capability that did not exist. The skills door
|
||||||
|
/// changed that, and the arm is now recorded per mission so both can run.
|
||||||
|
///
|
||||||
|
/// Pinned only (`pin_in_context`) in either arm, because everything else
|
||||||
|
/// would go in unbounded and unread.
|
||||||
|
pub async fn pinned_skills_text(&self, alias: &str) -> Option<String> {
|
||||||
|
let mode = self.skill_delivery_mode().await;
|
||||||
|
self.pinned_skills_in_mode(alias, mode).await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The arm this mission was launched with.
|
||||||
|
///
|
||||||
|
/// Read per turn rather than cached on the executor: the executor is
|
||||||
|
/// constructed from the environment by `topology_worker`, which knows
|
||||||
|
/// nothing about a mission, and the arm is decided at launch by the code
|
||||||
|
/// that also learns whether the door installed.
|
||||||
|
///
|
||||||
|
/// Anything unreadable — no tap, no row, an unrecognised value — resolves
|
||||||
|
/// to `Inline`, which is the arm that needs nothing to be true.
|
||||||
|
pub(crate) async fn skill_delivery_mode(&self) -> crate::skill_delivery::Mode {
|
||||||
|
let Some(tap) = self.tap.as_ref() else {
|
||||||
|
return crate::skill_delivery::Mode::Inline;
|
||||||
|
};
|
||||||
|
sqlx::query_scalar::<_, Option<String>>(
|
||||||
|
"SELECT skill_delivery FROM missions WHERE id = $1",
|
||||||
|
)
|
||||||
|
.bind(tap.mission_id)
|
||||||
|
.fetch_optional(&tap.pool)
|
||||||
|
.await
|
||||||
|
.ok()
|
||||||
|
.flatten()
|
||||||
|
.flatten()
|
||||||
|
.and_then(|s| crate::skill_delivery::parse(&s))
|
||||||
|
.unwrap_or(crate::skill_delivery::Mode::Inline)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) async fn pinned_skills_in_mode(
|
||||||
|
&self,
|
||||||
|
alias: &str,
|
||||||
|
mode: crate::skill_delivery::Mode,
|
||||||
|
) -> Option<String> {
|
||||||
|
let tap = self.tap.as_ref()?;
|
||||||
|
let agent_id = crate::runtime_provision::claw_from_alias(alias)?;
|
||||||
|
let link = cm_db::repo::agent_template_link::get(&tap.pool, agent_id)
|
||||||
|
.await
|
||||||
|
.ok()
|
||||||
|
.flatten();
|
||||||
|
let (tpl_id, slot) = link
|
||||||
|
.as_ref()
|
||||||
|
.map(|l| (Some(l.template_id), Some(l.role_slot.as_str())))
|
||||||
|
.unwrap_or((None, None));
|
||||||
|
let bindings =
|
||||||
|
cm_db::repo::skills_catalog::effective_for_agent(&tap.pool, agent_id, tpl_id, slot)
|
||||||
|
.await
|
||||||
|
.ok()?;
|
||||||
|
|
||||||
|
let mut out = String::new();
|
||||||
|
let mut n = 0usize;
|
||||||
|
for b in bindings.iter().filter(|b| b.pin_in_context) {
|
||||||
|
// `always_inject` overrides the arm. Progressive disclosure asks
|
||||||
|
// the agent to recognise that a procedure applies before fetching
|
||||||
|
// it, and a CROSS-CUTTING procedure is the case that breaks: the
|
||||||
|
// first A/B pair had `workspace-repo-commit-protocol` scored
|
||||||
|
// Trigger=FAIL beside a passing boundary check, because a rule that
|
||||||
|
// applies to everyone who writes reads as nobody's in particular.
|
||||||
|
let text = match mode {
|
||||||
|
crate::skill_delivery::Mode::Inline => b.skill.body.clone(),
|
||||||
|
m if m.is_retrieval() && b.skill.always_inject => b.skill.body.clone(),
|
||||||
|
// An entry is a few hundred bytes whatever the body weighs, so
|
||||||
|
// the retrieval arms cannot hit the cap that follows. That is
|
||||||
|
// the point of them, and the reason the cap is checked against
|
||||||
|
// the rendered text rather than against the body.
|
||||||
|
crate::skill_delivery::Mode::Index => crate::skill_delivery::index_entry(
|
||||||
|
&b.skill.description,
|
||||||
|
b.skill.when_to_use.as_deref(),
|
||||||
|
&crate::mcp_skills::skill_uri(b.skill.workspace_id, &b.skill.name),
|
||||||
|
),
|
||||||
|
crate::skill_delivery::Mode::Files => crate::skill_delivery::file_entry(
|
||||||
|
&b.skill.description,
|
||||||
|
b.skill.when_to_use.as_deref(),
|
||||||
|
&crate::skill_delivery::skill_file_path(&b.skill.name),
|
||||||
|
),
|
||||||
|
};
|
||||||
|
// Bounded, and truncation is STATED. A silently clipped procedure
|
||||||
|
// is worse than an absent one: the agent follows the half it can
|
||||||
|
// see and reports success against a rule it never read.
|
||||||
|
if out.len() + text.len() > MAX_PINNED_SKILL_BYTES {
|
||||||
|
out.push_str(&format!(
|
||||||
|
"\n[skill \"{}\" omitted — the pinned set exceeded {} bytes]\n",
|
||||||
|
b.skill.name, MAX_PINNED_SKILL_BYTES
|
||||||
|
));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
out.push_str(&render_pinned_skill(&b.skill.name, &text));
|
||||||
|
n += 1;
|
||||||
|
}
|
||||||
|
if n == 0 {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(out)
|
||||||
|
}
|
||||||
|
|
||||||
/// Mirror of `ProviderExecutor`'s prompt, flattened to one `content` string
|
/// Mirror of `ProviderExecutor`'s prompt, flattened to one `content` string
|
||||||
/// (the gateway `message` envelope carries a single content field).
|
/// (the gateway `message` envelope carries a single content field).
|
||||||
fn build_prompt(req: &TurnRequest) -> String {
|
fn build_prompt(req: &TurnRequest) -> String {
|
||||||
@@ -196,6 +461,19 @@ impl ZeroClawDriveExecutor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub async fn drive(&self, alias: &str, prompt: &str) -> Result<TurnOutcome, OrchestratorError> {
|
pub async fn drive(&self, alias: &str, prompt: &str) -> Result<TurnOutcome, OrchestratorError> {
|
||||||
|
self.drive_traced(alias, prompt).await.map(|(o, _)| o)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// [`Self::drive`], also returning what the turn's frames said it did.
|
||||||
|
///
|
||||||
|
/// Exists so the tool tap is testable at all: `drive` discards the trace
|
||||||
|
/// after recording it, and a tap whose extraction is never asserted is
|
||||||
|
/// exactly the kind of code that silently records nothing.
|
||||||
|
pub(crate) async fn drive_traced(
|
||||||
|
&self,
|
||||||
|
alias: &str,
|
||||||
|
prompt: &str,
|
||||||
|
) -> Result<(TurnOutcome, ToolTrace), OrchestratorError> {
|
||||||
let token = self.ensure_paired().await?;
|
let token = self.ensure_paired().await?;
|
||||||
let ws_base = if let Some(rest) = self.gateway_url.strip_prefix("https") {
|
let ws_base = if let Some(rest) = self.gateway_url.strip_prefix("https") {
|
||||||
format!("wss{rest}")
|
format!("wss{rest}")
|
||||||
@@ -219,27 +497,127 @@ impl ZeroClawDriveExecutor {
|
|||||||
.await
|
.await
|
||||||
.map_err(|e| OrchestratorError::Executor(format!("ws send failed: {e}")))?;
|
.map_err(|e| OrchestratorError::Executor(format!("ws send failed: {e}")))?;
|
||||||
|
|
||||||
let outcome = tokio::time::timeout(TURN_TIMEOUT, Self::drain(&mut ws))
|
let (outcome, trace) = match tokio::time::timeout(
|
||||||
|
TURN_TIMEOUT,
|
||||||
|
Self::drain(
|
||||||
|
&mut ws,
|
||||||
|
self.tap.as_ref().and_then(|t| {
|
||||||
|
crate::live_bus::agent_id_from_alias(alias).map(|a| (t.workspace_id, a))
|
||||||
|
}),
|
||||||
|
),
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
.map_err(|_| OrchestratorError::Executor("turn timed out".into()))??;
|
{
|
||||||
|
Ok(res) => res?,
|
||||||
|
Err(_) => {
|
||||||
|
// "turn timed out" on its own is unactionable, and the one place
|
||||||
|
// the reason lives — the per-mission runtime container — is torn
|
||||||
|
// down after the phase, taking its log with it. Read the tail
|
||||||
|
// while it still exists.
|
||||||
|
//
|
||||||
|
// MEASURED: a research phase timed out at exactly 700s having
|
||||||
|
// produced zero steps and zero output, and the container was
|
||||||
|
// already gone by the time anyone looked. All that survived was
|
||||||
|
// the string.
|
||||||
|
let container = self.container_name();
|
||||||
|
let tail = match &container {
|
||||||
|
Some(c) => crate::container_exec::tail_logs(c, 40).await,
|
||||||
|
None => "(could not derive the container name from the gateway url)".into(),
|
||||||
|
};
|
||||||
|
return Err(OrchestratorError::Executor(format!(
|
||||||
|
"turn timed out after {}s driving agent {alias} on {} — the agent \
|
||||||
|
never finished a turn. Last lines from {}:\n{tail}",
|
||||||
|
TURN_TIMEOUT.as_secs(),
|
||||||
|
self.gateway_url,
|
||||||
|
container.as_deref().unwrap_or("its runtime container"),
|
||||||
|
)));
|
||||||
|
}
|
||||||
|
};
|
||||||
let _ = ws.close(None).await;
|
let _ = ws.close(None).await;
|
||||||
Ok(outcome)
|
self.record_trace(alias, &trace).await;
|
||||||
|
Ok((outcome, trace))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Persist what this turn's frames said the agent did.
|
||||||
|
///
|
||||||
|
/// Best-effort and after the fact: a telemetry write must not be able to
|
||||||
|
/// fail a turn that already succeeded.
|
||||||
|
async fn record_trace(&self, alias: &str, trace: &ToolTrace) {
|
||||||
|
if !trace.unmatched.is_empty() {
|
||||||
|
// Logged whether or not a tap is attached — the point is to learn
|
||||||
|
// the real frame names, and the paths with no tap see the same
|
||||||
|
// stream.
|
||||||
|
eprintln!(
|
||||||
|
"topology_exec: unmatched frame types this turn ({alias}): {:?}",
|
||||||
|
trace.unmatched
|
||||||
|
);
|
||||||
|
}
|
||||||
|
let Some(tap) = self.tap.as_ref() else { return };
|
||||||
|
if trace.calls.is_empty() {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
let agent_id = crate::runtime_provision::claw_from_alias(alias);
|
||||||
|
let event = |kind: &str, target: String, detail: serde_json::Value| {
|
||||||
|
crate::mission_events::MissionEvent {
|
||||||
|
mission_id: tap.mission_id,
|
||||||
|
phase_id: tap.phase_id,
|
||||||
|
run_id: tap.run_id,
|
||||||
|
agent_id,
|
||||||
|
kind: kind.to_string(),
|
||||||
|
target: Some(target),
|
||||||
|
detail,
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let mut events = Vec::new();
|
||||||
|
for call in &trace.calls {
|
||||||
|
events.push(event(
|
||||||
|
crate::mission_events::TOOL_CALL,
|
||||||
|
call.tool.clone(),
|
||||||
|
serde_json::Value::Null,
|
||||||
|
));
|
||||||
|
// A file touch is a SECOND event, not a replacement: the tool call
|
||||||
|
// happened whether or not we could name a path in its arguments,
|
||||||
|
// and collapsing the two would make every unparseable tool call
|
||||||
|
// disappear from the record entirely.
|
||||||
|
if let Some(path) = &call.path {
|
||||||
|
events.push(event(
|
||||||
|
crate::mission_events::FILE_TOUCH,
|
||||||
|
crate::mission_events::repo_relative(path, GUEST_ROOTS),
|
||||||
|
serde_json::json!({ "tool": call.tool }),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
crate::mission_events::record_all(&tap.pool, events).await;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The runtime container behind this executor, derived from its gateway URL
|
||||||
|
/// (`http://cm-runtime-mission-<hex>:42617`). Used only to fetch a log tail
|
||||||
|
/// for an error message, so an unparseable URL is `None` rather than a
|
||||||
|
/// failure.
|
||||||
|
fn container_name(&self) -> Option<String> {
|
||||||
|
let rest = self
|
||||||
|
.gateway_url
|
||||||
|
.split("://")
|
||||||
|
.nth(1)
|
||||||
|
.unwrap_or(&self.gateway_url);
|
||||||
|
let host = rest.split('/').next()?.split(':').next()?;
|
||||||
|
(!host.is_empty()).then(|| host.to_string())
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Use a runtime agent as a governance judge: drive `alias` with the judge
|
/// Use a runtime agent as a governance judge: drive `alias` with the judge
|
||||||
/// prompt and parse the verdict (`DENY` anywhere ⇒ deny, else allow). This
|
/// prompt and parse the verdict with [`cm_runtime::governor_allows`]. This
|
||||||
/// lets a **subscription-only** model (e.g. Kimi via `kimi_cli`) be the judge
|
/// lets a **subscription-only** model (e.g. Kimi via `kimi_cli`) be the judge
|
||||||
/// with no platform API key — the registry/SDK path GLM and Kimi can't take.
|
/// with no platform API key — the registry/SDK path GLM and Kimi can't take.
|
||||||
/// Fail-open (returns `(true, …)`) so a judge outage never halts agents.
|
/// Fail-closed: an unreachable judge denies, for the reason given on
|
||||||
|
/// [`cm_runtime::Runtime::judge`].
|
||||||
pub async fn judge(&self, alias: &str, system: &str, user: &str) -> (bool, String) {
|
pub async fn judge(&self, alias: &str, system: &str, user: &str) -> (bool, String) {
|
||||||
let prompt = format!("{system}\n\n{user}");
|
let prompt = format!("{system}\n\n{user}");
|
||||||
match self.drive(alias, &prompt).await {
|
match self.drive(alias, &prompt).await {
|
||||||
Ok(outcome) => {
|
Ok(outcome) => {
|
||||||
let text = outcome.output.trim().to_string();
|
let text = outcome.output.trim().to_string();
|
||||||
let allow = !text.to_uppercase().contains("DENY");
|
(cm_runtime::governor_allows(&text), text)
|
||||||
(allow, text)
|
|
||||||
}
|
}
|
||||||
Err(e) => (true, format!("governor unreachable (fail-open): {e}")),
|
Err(e) => (false, format!("governor unreachable (fail-closed): {e}")),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -286,7 +664,18 @@ impl ZeroClawDriveExecutor {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Read frames until a terminal (`done`/`error`/`approval_request`) event.
|
/// Read frames until a terminal (`done`/`error`/`approval_request`) event.
|
||||||
async fn drain<S>(ws: &mut S) -> Result<TurnOutcome, OrchestratorError>
|
///
|
||||||
|
/// Returns the turn's outcome AND what its frames said the agent did. The
|
||||||
|
/// trace is separate from [`TurnOutcome`] deliberately: that type is the
|
||||||
|
/// shared orchestrator contract used by every tier, and tool telemetry is a
|
||||||
|
/// mission concern.
|
||||||
|
/// `live` is the push target for this turn: `Some((workspace, agent))` when
|
||||||
|
/// the turn belongs to a mission AND runs under a claw alias. `None` for the
|
||||||
|
/// governor/door/evaluator, whose output belongs to no agent.
|
||||||
|
async fn drain<S>(
|
||||||
|
ws: &mut S,
|
||||||
|
live: Option<(uuid::Uuid, uuid::Uuid)>,
|
||||||
|
) -> Result<(TurnOutcome, ToolTrace), OrchestratorError>
|
||||||
where
|
where
|
||||||
S: StreamExt<Item = Result<Message, tokio_tungstenite::tungstenite::Error>>
|
S: StreamExt<Item = Result<Message, tokio_tungstenite::tungstenite::Error>>
|
||||||
+ SinkExt<Message>
|
+ SinkExt<Message>
|
||||||
@@ -294,7 +683,9 @@ impl ZeroClawDriveExecutor {
|
|||||||
{
|
{
|
||||||
let mut output = String::new();
|
let mut output = String::new();
|
||||||
let mut tokens: u64 = 0;
|
let mut tokens: u64 = 0;
|
||||||
|
let mut spend = cm_orchestrator::Spend::default();
|
||||||
let mut gated: Vec<GatedAction> = Vec::new();
|
let mut gated: Vec<GatedAction> = Vec::new();
|
||||||
|
let mut trace = ToolTrace::default();
|
||||||
|
|
||||||
while let Some(frame) = ws.next().await {
|
while let Some(frame) = ws.next().await {
|
||||||
let msg = frame.map_err(|e| OrchestratorError::Executor(format!("ws recv: {e}")))?;
|
let msg = frame.map_err(|e| OrchestratorError::Executor(format!("ws recv: {e}")))?;
|
||||||
@@ -306,12 +697,47 @@ impl ZeroClawDriveExecutor {
|
|||||||
"chunk" => {
|
"chunk" => {
|
||||||
if let Some(c) = v.get("content").and_then(|c| c.as_str()) {
|
if let Some(c) = v.get("content").and_then(|c| c.as_str()) {
|
||||||
output.push_str(c);
|
output.push_str(c);
|
||||||
|
// Push, don't wait for the poll. This is the
|
||||||
|
// whole point of the bus: the reasoning card
|
||||||
|
// previously showed a step's text only after the
|
||||||
|
// step ended and the row was written, so an
|
||||||
|
// agent mid-thought looked idle for seconds.
|
||||||
|
if let Some((ws_id, agent_id)) = live {
|
||||||
|
if !c.trim().is_empty() {
|
||||||
|
crate::live_bus::global().publish(
|
||||||
|
ws_id,
|
||||||
|
"agent.reasoning.delta",
|
||||||
|
serde_json::json!({
|
||||||
|
"agentId": agent_id.to_string(),
|
||||||
|
"text": c,
|
||||||
|
"channel": "say",
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
"done" => {
|
"done" => {
|
||||||
let input = v.get("input_tokens").and_then(|n| n.as_u64()).unwrap_or(0);
|
let input = v.get("input_tokens").and_then(|n| n.as_u64()).unwrap_or(0);
|
||||||
let out = v.get("output_tokens").and_then(|n| n.as_u64()).unwrap_or(0);
|
let out = v.get("output_tokens").and_then(|n| n.as_u64()).unwrap_or(0);
|
||||||
tokens = input + out;
|
tokens = input + out;
|
||||||
|
// The frame has always carried these; only
|
||||||
|
// `tokens` was read, so every agent turn was
|
||||||
|
// charged with no record of who was paid.
|
||||||
|
spend = cm_orchestrator::Spend {
|
||||||
|
input_tokens: input,
|
||||||
|
output_tokens: out,
|
||||||
|
provider: v
|
||||||
|
.get("provider")
|
||||||
|
.and_then(|p| p.as_str())
|
||||||
|
.filter(|p| !p.is_empty())
|
||||||
|
.map(str::to_string),
|
||||||
|
model: v
|
||||||
|
.get("model")
|
||||||
|
.and_then(|m| m.as_str())
|
||||||
|
.filter(|m| !m.is_empty())
|
||||||
|
.map(str::to_string),
|
||||||
|
};
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
"approval_request" => {
|
"approval_request" => {
|
||||||
@@ -340,8 +766,50 @@ impl ZeroClawDriveExecutor {
|
|||||||
"aborted" => {
|
"aborted" => {
|
||||||
return Err(OrchestratorError::Executor("turn aborted".into()));
|
return Err(OrchestratorError::Executor("turn aborted".into()));
|
||||||
}
|
}
|
||||||
// session_start, thinking, tool_call, tool_result, …
|
// The action channel. `arguments` is read as JSON and
|
||||||
_ => {}
|
// nothing else is: the frame also carries a prose
|
||||||
|
// summary, and a path pulled out of THAT would be right
|
||||||
|
// often enough to be believed and wrong often enough to
|
||||||
|
// put files on the map that nobody edited.
|
||||||
|
"tool_call" => {
|
||||||
|
// `name` is what the gateway sends; `tool` is
|
||||||
|
// what `approval_request` uses, kept as a fallback.
|
||||||
|
let tool = v
|
||||||
|
.get("name")
|
||||||
|
.or_else(|| v.get("tool"))
|
||||||
|
.and_then(|t| t.as_str())
|
||||||
|
.unwrap_or("")
|
||||||
|
.trim()
|
||||||
|
.to_string();
|
||||||
|
if !tool.is_empty() {
|
||||||
|
// `args` FIRST: that is what the gateway
|
||||||
|
// actually sends (`{"type":"tool_call","id",
|
||||||
|
// "name","args"}` — zeroclaw-gateway/src/ws.rs).
|
||||||
|
// The others were guesses, and a guess that
|
||||||
|
// never matches costs the file path silently:
|
||||||
|
// the tool call is still recorded, with no
|
||||||
|
// target, and reads as a tool that touched
|
||||||
|
// nothing.
|
||||||
|
let args = v
|
||||||
|
.get("args")
|
||||||
|
.or_else(|| v.get("arguments"))
|
||||||
|
.or_else(|| v.get("input"))
|
||||||
|
.cloned()
|
||||||
|
.unwrap_or(serde_json::Value::Null);
|
||||||
|
trace.calls.push(ToolCall {
|
||||||
|
path: crate::mission_events::tool_path(&args),
|
||||||
|
tool,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// session_start, thinking, tool_result, …
|
||||||
|
other => {
|
||||||
|
// Counted, not ignored. See `ToolTrace::unmatched`:
|
||||||
|
// the frame name above is unverified, and a tap
|
||||||
|
// that matches nothing looks exactly like a mission
|
||||||
|
// that used no tools.
|
||||||
|
*trace.unmatched.entry(other.to_string()).or_insert(0) += 1;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
Message::Ping(p) => {
|
Message::Ping(p) => {
|
||||||
@@ -352,14 +820,22 @@ impl ZeroClawDriveExecutor {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(TurnOutcome {
|
Ok((
|
||||||
|
TurnOutcome {
|
||||||
output: output.trim().to_string(),
|
output: output.trim().to_string(),
|
||||||
tokens,
|
tokens,
|
||||||
gated,
|
gated,
|
||||||
})
|
spend,
|
||||||
|
},
|
||||||
|
trace,
|
||||||
|
))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Guest workspace roots, stripped so a tool's absolute path becomes the
|
||||||
|
/// repo-relative one a person recognises.
|
||||||
|
const GUEST_ROOTS: &[&str] = &["/mission/repo", "/workspace", "/repo"];
|
||||||
|
|
||||||
impl TurnExecutor for ZeroClawDriveExecutor {
|
impl TurnExecutor for ZeroClawDriveExecutor {
|
||||||
async fn run_turn(&self, req: TurnRequest) -> Result<TurnOutcome, OrchestratorError> {
|
async fn run_turn(&self, req: TurnRequest) -> Result<TurnOutcome, OrchestratorError> {
|
||||||
// An explicit per-node agent (graph `node.attrs["agent"]`) wins, so one
|
// An explicit per-node agent (graph `node.attrs["agent"]`) wins, so one
|
||||||
@@ -383,11 +859,105 @@ impl TurnExecutor for ZeroClawDriveExecutor {
|
|||||||
);
|
);
|
||||||
fallback
|
fallback
|
||||||
});
|
});
|
||||||
let prompt = Self::build_prompt(&req);
|
// One lookup, used for the section, its preamble and the record.
|
||||||
|
// Deriving it three times would let a mission compose an index under
|
||||||
|
// an inline heading if the row changed mid-run.
|
||||||
|
let mode = self.skill_delivery_mode().await;
|
||||||
|
let prompt = compose_turn_prompt(
|
||||||
|
&Self::build_prompt(&req),
|
||||||
|
self.pinned_skills_in_mode(&alias, mode).await.as_deref(),
|
||||||
|
mode,
|
||||||
|
);
|
||||||
|
// Record what this agent is ACTUALLY about to receive, before driving.
|
||||||
|
// Re-deriving it later would re-run the skill lookup against a
|
||||||
|
// catalogue that may have changed — and once agents author their own
|
||||||
|
// skills, it certainly will have.
|
||||||
|
if let Some(tap) = self.tap.as_ref() {
|
||||||
|
let mut ev = crate::mission_events::MissionEvent::new(
|
||||||
|
tap.mission_id,
|
||||||
|
crate::mission_events::PROMPT_COMPOSED,
|
||||||
|
);
|
||||||
|
ev.phase_id = tap.phase_id;
|
||||||
|
ev.run_id = tap.run_id;
|
||||||
|
ev.agent_id = crate::runtime_provision::claw_from_alias(&alias);
|
||||||
|
ev.target = Some(req.role.clone());
|
||||||
|
ev.detail = serde_json::json!({
|
||||||
|
"text": prompt,
|
||||||
|
"tier": "container",
|
||||||
|
// The A/B arm, alongside the prompt it produced. `skill_use`
|
||||||
|
// recovers this from the prompt text itself, so this field is
|
||||||
|
// for reporting and for catching the two disagreeing.
|
||||||
|
"skill_delivery": mode.as_str(),
|
||||||
|
});
|
||||||
|
crate::mission_events::record(&tap.pool, ev).await;
|
||||||
|
}
|
||||||
self.drive(&alias, &prompt).await
|
self.drive(&alias, &prompt).await
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// The base turn prompt with the agent's pinned skills appended, if it has any.
|
||||||
|
///
|
||||||
|
/// Split out from `run_turn` so the wiring is testable: `pinned_skills_text`
|
||||||
|
/// working and `run_turn` actually calling it are different claims, and the
|
||||||
|
/// second is the one that was false for every skill in the catalogue.
|
||||||
|
pub fn compose_turn_prompt(
|
||||||
|
base: &str,
|
||||||
|
skills: Option<&str>,
|
||||||
|
mode: crate::skill_delivery::Mode,
|
||||||
|
) -> String {
|
||||||
|
let Some(skills) = skills.map(str::trim).filter(|s| !s.is_empty()) else {
|
||||||
|
// No heading when there is nothing under it. An empty "Your skills"
|
||||||
|
// section tells the model it has skills and then shows it none, which
|
||||||
|
// is worse than silence.
|
||||||
|
return base.to_string();
|
||||||
|
};
|
||||||
|
// The preamble differs per arm and lives in `skill_delivery`, because it
|
||||||
|
// is also what the scorer reads the arm back from. Two copies of this
|
||||||
|
// sentence is two chances for the reader to stop recognising the writer.
|
||||||
|
let preamble = crate::skill_delivery::preamble(mode);
|
||||||
|
let section = format!("# Your skills\n\n{preamble}\n\n{skills}");
|
||||||
|
// After the identity paragraph and BEFORE the task. This section is the
|
||||||
|
// one part of the prompt that asks the agent to do something before it
|
||||||
|
// starts — read a procedure — and until 2026-09-18 it was appended last,
|
||||||
|
// after the task, the tool list, the workspace rules and the marker
|
||||||
|
// contract, sitting 87–90% of the way into a 6 KB prompt. On the three
|
||||||
|
// `index`-arm runs the narratives never mentioned it at all. Position is
|
||||||
|
// the untested lever; this is the test.
|
||||||
|
match base.split_once("\n\n") {
|
||||||
|
Some((identity, rest)) if identity.starts_with("You are ") => {
|
||||||
|
format!("{identity}\n\n{section}\n\n{rest}")
|
||||||
|
}
|
||||||
|
_ => format!("{section}\n\n{base}"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod prompt_order_tests {
|
||||||
|
use super::compose_turn_prompt;
|
||||||
|
use crate::skill_delivery::Mode;
|
||||||
|
|
||||||
|
/// The section comes right after the identity paragraph, before the task —
|
||||||
|
/// not appended after everything else.
|
||||||
|
#[test]
|
||||||
|
fn skills_come_after_identity_and_before_the_task() {
|
||||||
|
let base = "You are the \"x\" agent. Do your part.\n\nTask: MISSION: y\n\nTOOLS AVAILABLE";
|
||||||
|
let p = compose_turn_prompt(base, Some("--- SKILL: a ---\nbody"), Mode::Files);
|
||||||
|
let i_id = p.find("You are the").unwrap();
|
||||||
|
let i_sk = p.find("# Your skills").unwrap();
|
||||||
|
let i_task = p.find("Task: MISSION").unwrap();
|
||||||
|
assert!(i_id < i_sk && i_sk < i_task, "order was identity={i_id} skills={i_sk} task={i_task}\n{p}");
|
||||||
|
assert!(p.ends_with("TOOLS AVAILABLE"), "the base's tail is untouched");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A base with no identity paragraph still gets the section first.
|
||||||
|
#[test]
|
||||||
|
fn skills_lead_when_there_is_no_identity_paragraph() {
|
||||||
|
let p = compose_turn_prompt("Task: y", Some("--- SKILL: a ---\nbody"), Mode::Files);
|
||||||
|
assert!(p.starts_with("# Your skills"), "{p}");
|
||||||
|
assert!(p.ends_with("Task: y"));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Parse `role=alias,role=alias` into a map (blank/malformed entries skipped).
|
/// Parse `role=alias,role=alias` into a map (blank/malformed entries skipped).
|
||||||
fn parse_agent_map(s: &str) -> HashMap<String, String> {
|
fn parse_agent_map(s: &str) -> HashMap<String, String> {
|
||||||
s.split(',')
|
s.split(',')
|
||||||
@@ -406,6 +976,41 @@ fn parse_agent_map(s: &str) -> HashMap<String, String> {
|
|||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
mod tests {
|
mod tests {
|
||||||
|
/// The container name comes out of the gateway URL, or nothing does.
|
||||||
|
///
|
||||||
|
/// This is only used to fetch a log tail for a failure message, so a URL
|
||||||
|
/// shape it does not recognise must degrade to "no log" rather than to a
|
||||||
|
/// second error on top of the first.
|
||||||
|
#[test]
|
||||||
|
fn the_container_name_is_derived_or_absent_never_wrong() {
|
||||||
|
let ex = |url: &str| {
|
||||||
|
ZeroClawDriveExecutor::new(
|
||||||
|
url.to_string(),
|
||||||
|
String::new(),
|
||||||
|
std::collections::HashMap::new(),
|
||||||
|
"scout".into(),
|
||||||
|
)
|
||||||
|
};
|
||||||
|
assert_eq!(
|
||||||
|
ex("http://cm-runtime-mission-019fec2d596f:42617")
|
||||||
|
.container_name()
|
||||||
|
.as_deref(),
|
||||||
|
Some("cm-runtime-mission-019fec2d596f")
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
ex("https://host.example:8443/base")
|
||||||
|
.container_name()
|
||||||
|
.as_deref(),
|
||||||
|
Some("host.example")
|
||||||
|
);
|
||||||
|
// No scheme is still a host.
|
||||||
|
assert_eq!(
|
||||||
|
ex("clawmates-runtime:42617").container_name().as_deref(),
|
||||||
|
Some("clawmates-runtime")
|
||||||
|
);
|
||||||
|
assert_eq!(ex("").container_name(), None);
|
||||||
|
}
|
||||||
|
|
||||||
use super::*;
|
use super::*;
|
||||||
use axum::extract::ws::{Message as AxMsg, WebSocket, WebSocketUpgrade};
|
use axum::extract::ws::{Message as AxMsg, WebSocket, WebSocketUpgrade};
|
||||||
use axum::response::Response;
|
use axum::response::Response;
|
||||||
@@ -422,7 +1027,10 @@ mod tests {
|
|||||||
json!({"type": "session_start", "session_id": "s1", "resumed": false}),
|
json!({"type": "session_start", "session_id": "s1", "resumed": false}),
|
||||||
json!({"type": "chunk", "content": "hel"}),
|
json!({"type": "chunk", "content": "hel"}),
|
||||||
json!({"type": "chunk", "content": "lo"}),
|
json!({"type": "chunk", "content": "lo"}),
|
||||||
json!({"type": "done", "input_tokens": 5, "output_tokens": 7}),
|
// The real frame carries model and provider; the executor read
|
||||||
|
// only the two token counts until 2026-09-14.
|
||||||
|
json!({"type": "done", "input_tokens": 5, "output_tokens": 7,
|
||||||
|
"model": "claude-sonnet-5", "provider": "anthropic"}),
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -435,6 +1043,32 @@ mod tests {
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// A stream carrying tool calls and one frame type we do not know.
|
||||||
|
async fn tool_ws(ws: WebSocketUpgrade) -> Response {
|
||||||
|
ws.on_upgrade(|mut socket: WebSocket| async move {
|
||||||
|
let _ = socket.recv().await;
|
||||||
|
for f in [
|
||||||
|
json!({"type": "session_start"}),
|
||||||
|
// The REAL frame shape, copied from the gateway:
|
||||||
|
// {"type":"tool_call","id","name","args"}.
|
||||||
|
json!({"type": "tool_call", "id": "t1", "name": "Read",
|
||||||
|
"args": {"file_path": "/mission/repo/src/a.rs"}}),
|
||||||
|
// A tool whose arguments name no path at all.
|
||||||
|
json!({"type": "tool_call", "id": "t2", "name": "Bash",
|
||||||
|
"args": {"command": "cargo test"}}),
|
||||||
|
// Prose that MENTIONS a path. It must not become a file touch.
|
||||||
|
json!({"type": "tool_call", "id": "t3", "name": "Grep",
|
||||||
|
"arguments_summary": "searching src/main.rs",
|
||||||
|
"args": {"pattern": "fn main"}}),
|
||||||
|
json!({"type": "a_frame_we_have_never_seen"}),
|
||||||
|
json!({"type": "a_frame_we_have_never_seen"}),
|
||||||
|
json!({"type": "done", "input_tokens": 1, "output_tokens": 1}),
|
||||||
|
] {
|
||||||
|
let _ = socket.send(AxMsg::Text(f.to_string().into())).await;
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
async fn approval_ws(ws: WebSocketUpgrade) -> Response {
|
async fn approval_ws(ws: WebSocketUpgrade) -> Response {
|
||||||
ws.on_upgrade(|mut socket: WebSocket| async move {
|
ws.on_upgrade(|mut socket: WebSocket| async move {
|
||||||
let _ = socket.recv().await;
|
let _ = socket.recv().await;
|
||||||
@@ -496,9 +1130,66 @@ mod tests {
|
|||||||
let out = exec.run_turn(req()).await.unwrap();
|
let out = exec.run_turn(req()).await.unwrap();
|
||||||
assert_eq!(out.output, "hello");
|
assert_eq!(out.output, "hello");
|
||||||
assert_eq!(out.tokens, 12);
|
assert_eq!(out.tokens, 12);
|
||||||
|
assert_eq!(
|
||||||
|
out.spend,
|
||||||
|
cm_orchestrator::Spend {
|
||||||
|
input_tokens: 5,
|
||||||
|
output_tokens: 7,
|
||||||
|
provider: Some("anthropic".into()),
|
||||||
|
model: Some("claude-sonnet-5".into()),
|
||||||
|
},
|
||||||
|
"the split and the provider must survive the done frame, not just the sum"
|
||||||
|
);
|
||||||
assert!(out.gated.is_empty());
|
assert!(out.gated.is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Tool detail comes from arguments, and unknown frames are counted.
|
||||||
|
///
|
||||||
|
/// The two halves are one test because they are one risk. The frame type
|
||||||
|
/// `tool_call` is taken from a comment in this file, not from a captured
|
||||||
|
/// frame — so if it is wrong, the tap records nothing, the World stays as
|
||||||
|
/// sparse as it was, and NOTHING errors. The histogram is what turns that
|
||||||
|
/// into a log line naming the real frame.
|
||||||
|
#[tokio::test]
|
||||||
|
async fn tool_frames_give_up_their_arguments_and_unknown_frames_are_counted() {
|
||||||
|
let router = Router::new()
|
||||||
|
.route("/pair", post(pair))
|
||||||
|
.route("/ws/chat", get(tool_ws));
|
||||||
|
let base = serve(router).await;
|
||||||
|
let exec = ZeroClawDriveExecutor::new(base, "code".into(), HashMap::new(), "scout".into());
|
||||||
|
|
||||||
|
let (_out, trace) = exec.drive_traced("scout", "go").await.unwrap();
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
trace.calls,
|
||||||
|
vec![
|
||||||
|
ToolCall {
|
||||||
|
tool: "Read".into(),
|
||||||
|
path: Some("/mission/repo/src/a.rs".into())
|
||||||
|
},
|
||||||
|
ToolCall {
|
||||||
|
tool: "Bash".into(),
|
||||||
|
path: None
|
||||||
|
},
|
||||||
|
// `arguments_summary` said "src/main.rs". It is prose, so it is
|
||||||
|
// not a file touch — a path scraped from a sentence would put
|
||||||
|
// files on the map that no agent opened.
|
||||||
|
ToolCall {
|
||||||
|
tool: "Grep".into(),
|
||||||
|
path: None
|
||||||
|
},
|
||||||
|
]
|
||||||
|
);
|
||||||
|
assert_eq!(trace.unmatched.get("a_frame_we_have_never_seen"), Some(&2));
|
||||||
|
assert_eq!(trace.unmatched.get("session_start"), Some(&1));
|
||||||
|
// `done` terminates the drain and is not an unmatched frame.
|
||||||
|
assert!(
|
||||||
|
!trace.unmatched.contains_key("done"),
|
||||||
|
"{:?}",
|
||||||
|
trace.unmatched
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn approval_request_is_recorded_as_blocked() {
|
async fn approval_request_is_recorded_as_blocked() {
|
||||||
let router = Router::new()
|
let router = Router::new()
|
||||||
|
|||||||
@@ -26,14 +26,33 @@ use crate::topology_exec::ZeroClawDriveExecutor;
|
|||||||
const STALE_AFTER_SECS: f64 = 180.0;
|
const STALE_AFTER_SECS: f64 = 180.0;
|
||||||
|
|
||||||
/// Maximum age a `running` run may spend WITHOUT journaling any step
|
/// Maximum age a `running` run may spend WITHOUT journaling any step
|
||||||
/// records before the reaper kills its container and fails it. 15 min
|
/// records before the reaper kills its container and fails it.
|
||||||
/// is generous: healthy first-step latency is typically 5–60s; anything
|
///
|
||||||
/// past this is a stuck container (usually a wedged provider CLI).
|
/// **This must stay LONGER than the runtime's per-turn timeout.** A run
|
||||||
const REAP_STUCK_AFTER_SECS: i64 = 15 * 60;
|
/// journals its first record when its first step COMPLETES, so any turn still
|
||||||
|
/// legitimately in flight looks identical to a wedged container. The runtime
|
||||||
|
/// grants a turn `timeout_secs = 3000` (50 min), so a shorter reaper window
|
||||||
|
/// does not detect stuck runs — it kills healthy slow ones.
|
||||||
|
///
|
||||||
|
/// This was 15 minutes, chosen when "healthy first-step latency is typically
|
||||||
|
/// 5–60s" was true of the model in use. It was, on haiku. Moving the mission
|
||||||
|
/// agents to sonnet-5 made first turns longer than the window, and mission
|
||||||
|
/// 01a00c41's research phase was reaped at 900s having already written 402
|
||||||
|
/// lines across 13 files — work the delivery path then captured and pushed,
|
||||||
|
/// which is the only reason we could tell the run was healthy at all.
|
||||||
|
///
|
||||||
|
/// The lesson generalises past this constant: a liveness timeout calibrated
|
||||||
|
/// against one model silently becomes a correctness bug when the model changes.
|
||||||
|
const REAP_STUCK_AFTER_SECS: i64 = 60 * 60;
|
||||||
|
|
||||||
/// Spawn the durable topology job worker. Polls for queued jobs every `poll`
|
/// Spawn the durable topology job worker. Polls for queued jobs every `poll`
|
||||||
/// interval; runs each to completion (or failure), checkpointing per step.
|
/// interval; runs each to completion (or failure), checkpointing per step.
|
||||||
pub fn spawn(pool: PgPool, runtime: cm_runtime::Runtime, poll: Duration) {
|
pub fn spawn(
|
||||||
|
pool: PgPool,
|
||||||
|
runtime: cm_runtime::Runtime,
|
||||||
|
hub: Arc<crate::fleet::NodeHub>,
|
||||||
|
poll: Duration,
|
||||||
|
) {
|
||||||
// Fire the stuck-container reaper on its own cadence — checking
|
// Fire the stuck-container reaper on its own cadence — checking
|
||||||
// once a minute is plenty and keeps this off the hot claim loop.
|
// once a minute is plenty and keeps this off the hot claim loop.
|
||||||
let reaper_pool = pool.clone();
|
let reaper_pool = pool.clone();
|
||||||
@@ -55,7 +74,7 @@ pub fn spawn(pool: PgPool, runtime: cm_runtime::Runtime, poll: Duration) {
|
|||||||
eprintln!("topology_worker: requeue_stale failed: {e}");
|
eprintln!("topology_worker: requeue_stale failed: {e}");
|
||||||
}
|
}
|
||||||
match cm_db::repo::topology_runs::claim_next_queued(&pool).await {
|
match cm_db::repo::topology_runs::claim_next_queued(&pool).await {
|
||||||
Ok(Some(job)) => run_job(&pool, &runtime, job).await,
|
Ok(Some(job)) => run_job(&pool, &runtime, &hub, job).await,
|
||||||
Ok(None) => tokio::time::sleep(poll).await,
|
Ok(None) => tokio::time::sleep(poll).await,
|
||||||
Err(e) => {
|
Err(e) => {
|
||||||
eprintln!("topology_worker: claim failed: {e}");
|
eprintln!("topology_worker: claim failed: {e}");
|
||||||
@@ -80,10 +99,25 @@ async fn reap_stuck_runs(pool: &PgPool) -> Result<(), sqlx::Error> {
|
|||||||
FROM topology_runs
|
FROM topology_runs
|
||||||
WHERE status = 'running'
|
WHERE status = 'running'
|
||||||
AND mission_id IS NOT NULL
|
AND mission_id IS NOT NULL
|
||||||
|
-- Only jobs this worker drives. mission_id IS NOT NULL used to mean
|
||||||
|
-- the same thing as orchestrator-driven, and the microvm and session
|
||||||
|
-- tiers broke that: their checkpoint is NULL for life BY DESIGN, so the
|
||||||
|
-- zero-step-records test below is true of a perfectly healthy run.
|
||||||
|
AND tier = ANY($2)
|
||||||
AND created_at < now() - make_interval(secs => $1::float)
|
AND created_at < now() - make_interval(secs => $1::float)
|
||||||
AND coalesce(jsonb_array_length(coalesce(checkpoint->'records', '[]'::jsonb)), 0) = 0",
|
AND coalesce(jsonb_array_length(coalesce(checkpoint->'records', '[]'::jsonb)), 0) = 0",
|
||||||
)
|
)
|
||||||
.bind(REAP_STUCK_AFTER_SECS as f64)
|
.bind(REAP_STUCK_AFTER_SECS as f64)
|
||||||
|
.bind(
|
||||||
|
// REAPABLE, not worker-driven: `microvm_graph` is driven by this worker
|
||||||
|
// and must NOT be reaped — one of its steps is a whole agent session in a
|
||||||
|
// VM, so "no step records in 15 minutes" describes a healthy composed run
|
||||||
|
// as readily as a wedged one.
|
||||||
|
cm_db::repo::topology_runs::REAPABLE_TIERS
|
||||||
|
.iter()
|
||||||
|
.map(|s| (*s).to_string())
|
||||||
|
.collect::<Vec<_>>(),
|
||||||
|
)
|
||||||
.fetch_all(pool)
|
.fetch_all(pool)
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
@@ -108,6 +142,7 @@ async fn reap_stuck_runs(pool: &PgPool) -> Result<(), sqlx::Error> {
|
|||||||
async fn run_job(
|
async fn run_job(
|
||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
runtime: &cm_runtime::Runtime,
|
runtime: &cm_runtime::Runtime,
|
||||||
|
hub: &Arc<crate::fleet::NodeHub>,
|
||||||
job: cm_db::repo::topology_runs::ClaimedTopologyRun,
|
job: cm_db::repo::topology_runs::ClaimedTopologyRun,
|
||||||
) {
|
) {
|
||||||
let id = job.id;
|
let id = job.id;
|
||||||
@@ -145,16 +180,35 @@ async fn run_job(
|
|||||||
// Resume from the last checkpoint, or start fresh.
|
// Resume from the last checkpoint, or start fresh.
|
||||||
let progress: RunProgress = job
|
let progress: RunProgress = job
|
||||||
.checkpoint
|
.checkpoint
|
||||||
|
.clone()
|
||||||
.and_then(|c| serde_json::from_value(c).ok())
|
.and_then(|c| serde_json::from_value(c).ok())
|
||||||
.unwrap_or_default();
|
.unwrap_or_default();
|
||||||
|
|
||||||
|
// The composed engines (Slice 4): this graph's nodes are not claws, they are
|
||||||
|
// Claude-Code-in-a-microVM sessions. Branched BEFORE the leaf executor is
|
||||||
|
// built, because that build reads the ZeroClaw gateway config — a composed
|
||||||
|
// run must not fail for want of a runtime it never dials.
|
||||||
|
if job.tier == "microvm_graph" {
|
||||||
|
let result = run_composed(pool, hub, &job, &graph, progress).await;
|
||||||
|
finish(pool, id, result).await;
|
||||||
|
maybe_teardown_ephemeral_team(pool, runtime, id).await;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
// C3: prefer the mission's per-run runtime endpoint when set on
|
// C3: prefer the mission's per-run runtime endpoint when set on
|
||||||
// the missions row; else fall back to the shared env-derived
|
// the missions row; else fall back to the shared env-derived
|
||||||
// gateway (pre-C3 missions + non-mission runs). This is what
|
// gateway (pre-C3 missions + non-mission runs). This is what
|
||||||
// isolates agents' workspace filesystem to that mission's repo.
|
// isolates agents' workspace filesystem to that mission's repo.
|
||||||
let mission_binding: Option<(Option<String>, Option<String>)> =
|
type MissionBinding = (
|
||||||
sqlx::query_as::<_, (Option<String>, Option<String>)>(
|
Option<String>,
|
||||||
"SELECT m.runtime_endpoint, m.runtime_pairing_code
|
Option<String>,
|
||||||
|
Uuid,
|
||||||
|
Option<Uuid>,
|
||||||
|
Option<String>,
|
||||||
|
);
|
||||||
|
let mission_binding: Option<MissionBinding> = sqlx::query_as::<_, MissionBinding>(
|
||||||
|
"SELECT m.runtime_endpoint, m.runtime_pairing_code, m.id, r.mission_phase_id,
|
||||||
|
m.runtime_token
|
||||||
FROM topology_runs r
|
FROM topology_runs r
|
||||||
JOIN missions m ON m.id = r.mission_id
|
JOIN missions m ON m.id = r.mission_id
|
||||||
WHERE r.id = $1",
|
WHERE r.id = $1",
|
||||||
@@ -164,11 +218,31 @@ async fn run_job(
|
|||||||
.await
|
.await
|
||||||
.ok()
|
.ok()
|
||||||
.flatten();
|
.flatten();
|
||||||
|
// What this run's turns will be attributed to. `None` when the run belongs
|
||||||
|
// to no mission — a bare topology run has no phase to hang tool calls on.
|
||||||
|
let tap = mission_binding
|
||||||
|
.as_ref()
|
||||||
|
.map(
|
||||||
|
|(_, _, mission_id, phase_id, _)| crate::topology_exec::MissionTap {
|
||||||
|
pool: pool.clone(),
|
||||||
|
workspace_id: job.workspace_id,
|
||||||
|
mission_id: *mission_id,
|
||||||
|
phase_id: *phase_id,
|
||||||
|
run_id: Some(id),
|
||||||
|
},
|
||||||
|
);
|
||||||
let leaf_result = match mission_binding {
|
let leaf_result = match mission_binding {
|
||||||
Some((Some(url), Some(code))) => {
|
// Seed the cached bearer from `runtime_token` when we have one: the
|
||||||
|
// pairing code is single-use, so after a restart it is the only way in.
|
||||||
|
Some((Some(url), Some(code), _, _, tok)) => {
|
||||||
ZeroClawDriveExecutor::from_env_for_gateway_with_code(url, code)
|
ZeroClawDriveExecutor::from_env_for_gateway_with_code(url, code)
|
||||||
|
.map(|e| e.with_token(tok))
|
||||||
|
}
|
||||||
|
// No pairing code (pre-C3 missions): the persisted token is the only
|
||||||
|
// credential, so seed it here too.
|
||||||
|
Some((Some(url), None, _, _, tok)) => {
|
||||||
|
ZeroClawDriveExecutor::from_env_for_gateway(url).map(|e| e.with_token(tok))
|
||||||
}
|
}
|
||||||
Some((Some(url), None)) => ZeroClawDriveExecutor::from_env_for_gateway(url),
|
|
||||||
_ => ZeroClawDriveExecutor::from_env(),
|
_ => ZeroClawDriveExecutor::from_env(),
|
||||||
};
|
};
|
||||||
let leaf = match leaf_result {
|
let leaf = match leaf_result {
|
||||||
@@ -178,6 +252,13 @@ async fn run_job(
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
// The tap rides on the leaf executor, so the recursive tiers get it too:
|
||||||
|
// they drive the same leaf all the way down, and a company-tier mission's
|
||||||
|
// tool calls belong to its phase exactly as a team-tier one's do.
|
||||||
|
let leaf = match tap {
|
||||||
|
Some(t) => leaf.with_tap(t),
|
||||||
|
None => leaf,
|
||||||
|
};
|
||||||
|
|
||||||
// Select the executor by deploy tier: `team` drives claws directly; the
|
// Select the executor by deploy tier: `team` drives claws directly; the
|
||||||
// upper tiers drive the recursive sub-topology executor (which runs each
|
// upper tiers drive the recursive sub-topology executor (which runs each
|
||||||
@@ -196,11 +277,39 @@ async fn run_job(
|
|||||||
id,
|
id,
|
||||||
Arc::new(leaf),
|
Arc::new(leaf),
|
||||||
);
|
);
|
||||||
drive(pool, id, &graph, &job.task, progress, &exec).await
|
drive(
|
||||||
|
pool,
|
||||||
|
id,
|
||||||
|
job.workspace_id,
|
||||||
|
&graph,
|
||||||
|
&job.task,
|
||||||
|
progress,
|
||||||
|
&exec,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
drive(
|
||||||
|
pool,
|
||||||
|
id,
|
||||||
|
job.workspace_id,
|
||||||
|
&graph,
|
||||||
|
&job.task,
|
||||||
|
progress,
|
||||||
|
&leaf,
|
||||||
|
)
|
||||||
|
.await
|
||||||
}
|
}
|
||||||
_ => drive(pool, id, &graph, &job.task, progress, &leaf).await,
|
|
||||||
};
|
};
|
||||||
|
|
||||||
|
finish(pool, id, result).await;
|
||||||
|
maybe_teardown_ephemeral_team(pool, runtime, id).await;
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write a driven run's terminal state. The single place a run finishes, shared
|
||||||
|
/// by every tier — a second one would be a second completion path, which is where
|
||||||
|
/// every microVM bug this project has hit came from.
|
||||||
|
async fn finish(pool: &PgPool, id: Uuid, result: Result<RunRecord, OrchestratorError>) {
|
||||||
match result {
|
match result {
|
||||||
Ok(record) => {
|
Ok(record) => {
|
||||||
let value = serde_json::to_value(&record).unwrap_or(serde_json::Value::Null);
|
let value = serde_json::to_value(&record).unwrap_or(serde_json::Value::Null);
|
||||||
@@ -219,7 +328,85 @@ async fn run_job(
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
maybe_teardown_ephemeral_team(pool, runtime, id).await;
|
}
|
||||||
|
|
||||||
|
/// Drive a composed run: the outer graph is Engine Z, every node is a
|
||||||
|
/// Claude-Code-in-a-microVM session (Engine C).
|
||||||
|
///
|
||||||
|
/// The mission columns are read here rather than carried on the run row so a
|
||||||
|
/// re-placed or re-backed mission takes effect on resume, and so the composed
|
||||||
|
/// path has exactly one source of truth for where a VM boots.
|
||||||
|
async fn run_composed(
|
||||||
|
pool: &PgPool,
|
||||||
|
hub: &Arc<crate::fleet::NodeHub>,
|
||||||
|
job: &cm_db::repo::topology_runs::ClaimedTopologyRun,
|
||||||
|
graph: &TopologyGraph,
|
||||||
|
progress: RunProgress,
|
||||||
|
) -> Result<RunRecord, OrchestratorError> {
|
||||||
|
let mission_id = job.mission_id.ok_or_else(|| {
|
||||||
|
OrchestratorError::Executor(
|
||||||
|
"a composed run has no mission, so there is no checkout for its nodes \
|
||||||
|
to share"
|
||||||
|
.into(),
|
||||||
|
)
|
||||||
|
})?;
|
||||||
|
let phase_id = job.mission_phase_id.ok_or_else(|| {
|
||||||
|
OrchestratorError::Executor("a composed run must belong to a mission phase".into())
|
||||||
|
})?;
|
||||||
|
|
||||||
|
let mission: (Option<Uuid>, Option<String>, Option<String>, bool) = sqlx::query_as(
|
||||||
|
"SELECT target_node_id, backend, team_engine, (repo_id IS NOT NULL) \
|
||||||
|
FROM missions WHERE id = $1",
|
||||||
|
)
|
||||||
|
.bind(mission_id)
|
||||||
|
.fetch_one(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| OrchestratorError::Executor(format!("load mission {mission_id}: {e}")))?;
|
||||||
|
|
||||||
|
// The phase's completion gate, read here rather than carried on the run row
|
||||||
|
// so an edited `done_when_check` takes effect on the next node instead of at
|
||||||
|
// the next mission.
|
||||||
|
let phase: (String, serde_json::Value) =
|
||||||
|
sqlx::query_as("SELECT kind, config FROM mission_phases WHERE id = $1")
|
||||||
|
.bind(phase_id)
|
||||||
|
.fetch_one(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| OrchestratorError::Executor(format!("load phase {phase_id}: {e}")))?;
|
||||||
|
|
||||||
|
let exec = crate::microvm_turn_executor::for_fleet(
|
||||||
|
hub.clone(),
|
||||||
|
pool.clone(),
|
||||||
|
crate::microvm_turn_executor::ComposedRun {
|
||||||
|
run_id: job.id,
|
||||||
|
mission_id,
|
||||||
|
phase_id,
|
||||||
|
iteration: job.iteration.unwrap_or(1),
|
||||||
|
repo: crate::mission_workspace::checkout_path(mission_id),
|
||||||
|
// A repo-less composed mission gets an empty shared workspace, the
|
||||||
|
// same as a solo phase — the graph's whole property is that node 2
|
||||||
|
// sees node 1's files, and that holds whether or not it is a git
|
||||||
|
// checkout.
|
||||||
|
has_repo: mission.3,
|
||||||
|
target_node_id: mission.0,
|
||||||
|
backend: mission.1,
|
||||||
|
team_engine: mission.2,
|
||||||
|
gate: crate::vm_stop_gate::StopGate::for_phase(&phase.0, &phase.1)
|
||||||
|
.and_then(crate::vm_stop_gate::StopGate::per_node),
|
||||||
|
// Resume continues the step numbering; restarting it would re-use a
|
||||||
|
// finished node's vm id.
|
||||||
|
completed_steps: progress.completed as u32,
|
||||||
|
},
|
||||||
|
);
|
||||||
|
drive(
|
||||||
|
pool,
|
||||||
|
job.id,
|
||||||
|
job.workspace_id,
|
||||||
|
graph,
|
||||||
|
&job.task,
|
||||||
|
progress,
|
||||||
|
&exec,
|
||||||
|
)
|
||||||
|
.await
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Post-terminal hook: if this run's team is `ephemeral` and no siblings are
|
/// Post-terminal hook: if this run's team is `ephemeral` and no siblings are
|
||||||
@@ -278,14 +465,30 @@ async fn maybe_teardown_ephemeral_team(pool: &PgPool, runtime: &cm_runtime::Runt
|
|||||||
async fn drive<E: TurnExecutor>(
|
async fn drive<E: TurnExecutor>(
|
||||||
pool: &PgPool,
|
pool: &PgPool,
|
||||||
id: Uuid,
|
id: Uuid,
|
||||||
|
workspace_id: Uuid,
|
||||||
graph: &TopologyGraph,
|
graph: &TopologyGraph,
|
||||||
task: &str,
|
task: &str,
|
||||||
progress: RunProgress,
|
progress: RunProgress,
|
||||||
executor: &E,
|
executor: &E,
|
||||||
) -> Result<RunRecord, OrchestratorError> {
|
) -> Result<RunRecord, OrchestratorError> {
|
||||||
let pool_cb = pool.clone();
|
let pool_cb = pool.clone();
|
||||||
|
// node_id -> agent id, resolved once. The binding lives in the node's
|
||||||
|
// attrs (`agent = claw_<uuid>`), which is also what the runtime dispatches
|
||||||
|
// on — so usage is attributed to exactly the claw that did the work.
|
||||||
|
let agent_of: std::sync::Arc<std::collections::HashMap<String, Uuid>> = std::sync::Arc::new(
|
||||||
|
graph
|
||||||
|
.nodes
|
||||||
|
.iter()
|
||||||
|
.filter_map(|n| {
|
||||||
|
let alias = n.attrs.get("agent")?;
|
||||||
|
let uuid = alias.strip_prefix("claw_")?;
|
||||||
|
Some((n.id.clone(), Uuid::parse_str(uuid).ok()?))
|
||||||
|
})
|
||||||
|
.collect(),
|
||||||
|
);
|
||||||
execute_resumable(graph, task, executor, progress, move |snap| {
|
execute_resumable(graph, task, executor, progress, move |snap| {
|
||||||
let pool = pool_cb.clone();
|
let pool = pool_cb.clone();
|
||||||
|
let agent_of = agent_of.clone();
|
||||||
async move {
|
async move {
|
||||||
// 2026-07-15: verbose per-step trace so `docker logs
|
// 2026-07-15: verbose per-step trace so `docker logs
|
||||||
// clawmates_server_1` shows which topology node just fired,
|
// clawmates_server_1` shows which topology node just fired,
|
||||||
@@ -311,6 +514,84 @@ async fn drive<E: TurnExecutor>(
|
|||||||
last.tokens,
|
last.tokens,
|
||||||
last.gated.len(),
|
last.gated.len(),
|
||||||
);
|
);
|
||||||
|
// Per-agent usage. Without this the command centre's SPEND,
|
||||||
|
// ACTIVITY and THROUGHPUT cards read `usage_events`, which
|
||||||
|
// nothing on the mission path ever wrote — so they showed 0 for
|
||||||
|
// an agent that had just burned 15k tokens.
|
||||||
|
//
|
||||||
|
// `charge` also decrements credit lots, which is the point: a
|
||||||
|
// mission turn costs what it costs. It clamps at the available
|
||||||
|
// balance and still records the full obligation, so an empty
|
||||||
|
// wallet cannot fail a turn.
|
||||||
|
// The agent's own words, for the REASONING STREAM card. The
|
||||||
|
// world feed is a DB poll, not a push bus, so a live card can
|
||||||
|
// only show what was persisted — this is the step output the
|
||||||
|
// worker already has in hand, attributed to the claw that
|
||||||
|
// produced it. Truncated because the card renders a tail, not a
|
||||||
|
// transcript, and mission_events is capped per phase.
|
||||||
|
if let Some(agent_id) = agent_of.get(&last.node_id).copied() {
|
||||||
|
let text: String = last.output.chars().take(600).collect();
|
||||||
|
if !text.trim().is_empty() {
|
||||||
|
if let Some(mission_id) = sqlx::query_scalar::<_, Option<Uuid>>(
|
||||||
|
"SELECT mission_id FROM topology_runs WHERE id = $1",
|
||||||
|
)
|
||||||
|
.bind(id)
|
||||||
|
.fetch_optional(&pool)
|
||||||
|
.await
|
||||||
|
.ok()
|
||||||
|
.flatten()
|
||||||
|
.flatten()
|
||||||
|
{
|
||||||
|
let mut ev = crate::mission_events::MissionEvent::new(
|
||||||
|
mission_id,
|
||||||
|
"reasoning",
|
||||||
|
);
|
||||||
|
ev.agent_id = Some(agent_id);
|
||||||
|
ev.run_id = Some(id);
|
||||||
|
ev.target = Some(last.role.clone());
|
||||||
|
ev.detail = serde_json::json!({ "text": text });
|
||||||
|
crate::mission_events::record(&pool, ev).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if let Some(agent_id) = agent_of.get(&last.node_id).copied() {
|
||||||
|
if last.tokens > 0 {
|
||||||
|
// The split and the provider come from the runtime's
|
||||||
|
// `done` frame via `StepRecord.spend`. An executor
|
||||||
|
// that reports only a total leaves the split at 0/0
|
||||||
|
// and the total goes on the output side, as before.
|
||||||
|
let (tin, tout) = if last.spend.input_tokens + last.spend.output_tokens > 0
|
||||||
|
{
|
||||||
|
(last.spend.input_tokens, last.spend.output_tokens)
|
||||||
|
} else {
|
||||||
|
(0, last.tokens as u64)
|
||||||
|
};
|
||||||
|
let mission_id: Option<Uuid> = sqlx::query_scalar::<_, Option<Uuid>>(
|
||||||
|
"SELECT mission_id FROM topology_runs WHERE id = $1",
|
||||||
|
)
|
||||||
|
.bind(id)
|
||||||
|
.fetch_optional(&pool)
|
||||||
|
.await
|
||||||
|
.ok()
|
||||||
|
.flatten()
|
||||||
|
.flatten();
|
||||||
|
if let Err(e) = cm_billing::charge(
|
||||||
|
&pool,
|
||||||
|
cm_domain::WorkspaceId::from(workspace_id),
|
||||||
|
cm_domain::AgentId::from(agent_id),
|
||||||
|
None,
|
||||||
|
tin,
|
||||||
|
tout,
|
||||||
|
last.spend.provider.as_deref(),
|
||||||
|
last.spend.model.as_deref(),
|
||||||
|
mission_id,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
{
|
||||||
|
eprintln!("topology_worker: usage for {agent_id} failed: {e}");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
// Best-effort checkpoint: a failed write just means we re-run the
|
// Best-effort checkpoint: a failed write just means we re-run the
|
||||||
// step on resume (idempotent — topology turns are pure reads here).
|
// step on resume (idempotent — topology turns are pure reads here).
|
||||||
|
|||||||
@@ -0,0 +1,158 @@
|
|||||||
|
//! Can the independent validator actually be reached?
|
||||||
|
//!
|
||||||
|
//! The sibling of [`crate::runtime_preflight`], for the same class of failure:
|
||||||
|
//! the code is right and the machine is not, and nothing says so until a mission
|
||||||
|
//! pays for it.
|
||||||
|
//!
|
||||||
|
//! `evaluator::cross_provider_judge` deliberately refuses to fall back to the
|
||||||
|
//! agent's own provider — a verdict from the same family is not an independent
|
||||||
|
//! check, and quietly producing one would claim a property the verdict does not
|
||||||
|
//! have. That refusal is correct, and its cost is that a dead validator makes
|
||||||
|
//! every `done_when` phase UNMEETABLE. The mission still boots a VM, still runs
|
||||||
|
//! an agent turn, still collects and delivers, and only then records
|
||||||
|
//! "the independent validator could not be reached this pass" on one evaluation
|
||||||
|
//! row.
|
||||||
|
//!
|
||||||
|
//! That happened: the z.ai credential expired mid-session and the first symptom
|
||||||
|
//! was a two-phase mission failing after both VMs had run. The information
|
||||||
|
//! existed the whole time; nobody was told until it was expensive.
|
||||||
|
//!
|
||||||
|
//! A report, not a gate — the same stance `runtime_preflight` takes. The server
|
||||||
|
//! must still boot with a broken validator, because refusing to start would turn
|
||||||
|
//! a degraded deployment into a dead one, and because a mission that opts out
|
||||||
|
//! (`validator_model = ''`) is unaffected. What this buys is that the degradation
|
||||||
|
//! is visible at startup instead of inferred from a failed mission.
|
||||||
|
|
||||||
|
/// The smallest question that proves a credential works end to end.
|
||||||
|
///
|
||||||
|
/// A real completion rather than a models-list or a HEAD: an expired key, a
|
||||||
|
/// revoked key and a key with no quota can all pass a cheaper check and fail the
|
||||||
|
/// call that matters. Two tokens of output.
|
||||||
|
const PROBE_PROMPT: &str = "Reply with exactly: OK";
|
||||||
|
|
||||||
|
/// What the probe found.
|
||||||
|
#[derive(Debug, PartialEq, Eq)]
|
||||||
|
pub enum Verdict {
|
||||||
|
/// No independent validator is configured; phases are judged by the house
|
||||||
|
/// model. Not a fault — a deployment may choose this.
|
||||||
|
NotConfigured,
|
||||||
|
/// Configured, resolved, and it answered.
|
||||||
|
Reachable { spec: String },
|
||||||
|
/// Configured but the registry has no such provider, so
|
||||||
|
/// `cross_provider_judge` will refuse it rather than judge with the default.
|
||||||
|
Unregistered { spec: String },
|
||||||
|
/// Configured and resolved, and the call failed.
|
||||||
|
Unreachable { spec: String, error: String },
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Verdict {
|
||||||
|
/// Is every `done_when` phase currently unmeetable because of this?
|
||||||
|
pub fn breaks_gated_phases(&self) -> bool {
|
||||||
|
matches!(
|
||||||
|
self,
|
||||||
|
Verdict::Unregistered { .. } | Verdict::Unreachable { .. }
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Ask the configured independent validator to answer one trivial question.
|
||||||
|
pub async fn probe(runtime: &cm_runtime::Runtime) -> Verdict {
|
||||||
|
let Some(spec) = std::env::var("CLAWMATES_VALIDATOR_MODEL")
|
||||||
|
.ok()
|
||||||
|
.map(|s| s.trim().to_string())
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
else {
|
||||||
|
return Verdict::NotConfigured;
|
||||||
|
};
|
||||||
|
|
||||||
|
let (_provider, model) = runtime.resolve_provider(&spec);
|
||||||
|
// `resolve_provider` falls back to the DEFAULT provider for an unknown name,
|
||||||
|
// and the fallback is detectable because the returned model still carries the
|
||||||
|
// `name:` prefix. Checked here for the same reason the evaluator checks it:
|
||||||
|
// a validator that is silently the house model is worse than none.
|
||||||
|
if model.contains(':') {
|
||||||
|
return Verdict::Unregistered { spec };
|
||||||
|
}
|
||||||
|
|
||||||
|
// Through `Runtime::complete`, which is the same resolve-then-stream path
|
||||||
|
// the evaluator's judge takes. A probe that dialled the provider its own way
|
||||||
|
// could pass while the real call fails.
|
||||||
|
match runtime.complete("", PROBE_PROMPT, &spec, 16, false).await {
|
||||||
|
Ok(_) => Verdict::Reachable { spec },
|
||||||
|
Err(e) => Verdict::Unreachable {
|
||||||
|
spec,
|
||||||
|
error: e.chars().take(160).collect(),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Probe at boot and say plainly what it means for missions.
|
||||||
|
pub fn report_at_boot(runtime: cm_runtime::Runtime) {
|
||||||
|
tokio::spawn(async move {
|
||||||
|
match probe(&runtime).await {
|
||||||
|
Verdict::NotConfigured => eprintln!(
|
||||||
|
"validator_preflight: no CLAWMATES_VALIDATOR_MODEL — phase verdicts are judged \
|
||||||
|
by the house model, which is NOT an independent check"
|
||||||
|
),
|
||||||
|
Verdict::Reachable { spec } => {
|
||||||
|
eprintln!("validator_preflight: independent validator {spec} answered")
|
||||||
|
}
|
||||||
|
Verdict::Unregistered { spec } => eprintln!(
|
||||||
|
"validator_preflight: CLAWMATES_VALIDATOR_MODEL={spec} has no registered \
|
||||||
|
provider — the evaluator will refuse it rather than judge with the default, \
|
||||||
|
so EVERY phase with a done_when condition will fail as unmet. Register the \
|
||||||
|
provider, or set the mission's validator_model to '' to opt out."
|
||||||
|
),
|
||||||
|
Verdict::Unreachable { spec, error } => eprintln!(
|
||||||
|
"validator_preflight: independent validator {spec} is UNREACHABLE ({error}) — \
|
||||||
|
EVERY phase with a done_when condition will fail as unmet, after running its \
|
||||||
|
agent. Fix the credential, or set validator_model to '' per mission to judge \
|
||||||
|
with the house model."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The two states that make gated phases unmeetable, and the two that do
|
||||||
|
/// not. This is the distinction the whole module exists to draw: "no
|
||||||
|
/// validator configured" is a choice, "configured and broken" is a fault
|
||||||
|
/// that silently fails every conditioned mission.
|
||||||
|
#[test]
|
||||||
|
fn only_a_configured_but_broken_validator_breaks_gated_phases() {
|
||||||
|
assert!(!Verdict::NotConfigured.breaks_gated_phases());
|
||||||
|
assert!(!Verdict::Reachable {
|
||||||
|
spec: "glm:glm-4.7".into()
|
||||||
|
}
|
||||||
|
.breaks_gated_phases());
|
||||||
|
|
||||||
|
assert!(Verdict::Unregistered {
|
||||||
|
spec: "glm:glm-4.7".into()
|
||||||
|
}
|
||||||
|
.breaks_gated_phases());
|
||||||
|
assert!(Verdict::Unreachable {
|
||||||
|
spec: "glm:glm-4.7".into(),
|
||||||
|
error: "401".into()
|
||||||
|
}
|
||||||
|
.breaks_gated_phases());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An unregistered provider is NOT reported as unreachable, and the
|
||||||
|
/// difference is actionable: one is fixed by registering a provider, the
|
||||||
|
/// other by fixing a credential. Collapsing them sends an operator to the
|
||||||
|
/// wrong place.
|
||||||
|
#[test]
|
||||||
|
fn the_two_faults_are_distinguishable() {
|
||||||
|
let a = Verdict::Unregistered {
|
||||||
|
spec: "glm:glm-4.7".into(),
|
||||||
|
};
|
||||||
|
let b = Verdict::Unreachable {
|
||||||
|
spec: "glm:glm-4.7".into(),
|
||||||
|
error: "401 Authentication Failed".into(),
|
||||||
|
};
|
||||||
|
assert_ne!(a, b);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,815 @@
|
|||||||
|
//! Which fleet node should run the next microVM phase, and whether any can.
|
||||||
|
//!
|
||||||
|
//! # What this replaces
|
||||||
|
//!
|
||||||
|
//! Placement was `capable.first()` over a list ordered `last_seen DESC`
|
||||||
|
//! (`mission_orchestrator`, `nodes::online_for_backend`) — the most recently
|
||||||
|
//! heartbeated node. Among healthy nodes all heartbeating every 5s that is
|
||||||
|
//! arbitrary, and it consults nothing about load: two missions launched together
|
||||||
|
//! land on the same machine. It did not matter while one node held the only
|
||||||
|
//! rootfs image; all three do now.
|
||||||
|
//!
|
||||||
|
//! # Observed memory is not capacity
|
||||||
|
//!
|
||||||
|
//! The correctness core, and the reason this is not a one-line sort change. A VM
|
||||||
|
//! that booted 30 seconds ago holds a fraction of its 8 GiB claim — the guest has
|
||||||
|
//! not touched the rest — so `mem_pct` reports a sold-out node as nearly idle.
|
||||||
|
//! Ranking on utilisation alone would happily book five more VMs onto a node with
|
||||||
|
//! room for one. `capacity_of` therefore takes the WORSE of observed usage and
|
||||||
|
//! committed usage, and `a_sold_out_node_is_not_mistaken_for_an_idle_one` is the
|
||||||
|
//! negative control that pins it.
|
||||||
|
//!
|
||||||
|
//! Commitments come from two places that must be unioned by IDENTITY, never
|
||||||
|
//! added: `microvm_client::list` (booted VMs, including orphans nothing has
|
||||||
|
//! reaped) and `nodes::pinned_microvm_phases` (chosen but not yet booted). The
|
||||||
|
//! deterministic `vm_id_for` is what lets the same phase be recognised in both.
|
||||||
|
//!
|
||||||
|
//! # Fail-closed
|
||||||
|
//!
|
||||||
|
//! A node whose health is stale, whose daemon will not answer, or which is
|
||||||
|
//! draining is INELIGIBLE, not low-scoring. Unknown is not permission — the same
|
||||||
|
//! rule `nodes::online_for_backend` already applies to capabilities. The one
|
||||||
|
//! exception is Beszel metrics: they feed `headroom` as a tiebreak only, so stale
|
||||||
|
//! metrics demote a node instead of excluding it.
|
||||||
|
|
||||||
|
use cm_db::repo::node_metrics::EvalRow;
|
||||||
|
use cm_domain::NodeId;
|
||||||
|
|
||||||
|
/// Memory a phase VM claims. Re-exported from the executor so there is ONE number
|
||||||
|
/// — a scheduler and a launcher that disagree about VM size is a fleet that
|
||||||
|
/// overcommits by exactly their difference.
|
||||||
|
pub(crate) use crate::microvm_executor::MEM_MIB as MEM_PER_VM_MIB;
|
||||||
|
|
||||||
|
/// Held back for the host: the daemon, the OS, page cache, and the margin that
|
||||||
|
/// keeps a node out of swap. A node in swap makes every VM on it slow, so this is
|
||||||
|
/// cheaper than the alternative.
|
||||||
|
const HOST_RESERVE_MIB: i64 = 4096;
|
||||||
|
|
||||||
|
/// The floor we refuse to believe a host's own footprint is below. Without it, a
|
||||||
|
/// node reporting less used memory than its VMs have claimed would compute a
|
||||||
|
/// negative baseline and inflate its free memory.
|
||||||
|
const HOST_BASELINE_FLOOR_MIB: i64 = 2048;
|
||||||
|
|
||||||
|
/// Disk a VM may consume: an 8 GiB sparse rootfs plus room for the collected tar.
|
||||||
|
const DISK_PER_VM_GIB: i64 = 12;
|
||||||
|
|
||||||
|
/// Never let VM disk drive a node below this. `_outputs` and images live on the
|
||||||
|
/// same filesystem on some nodes.
|
||||||
|
const DISK_RESERVE_GIB: i64 = 20;
|
||||||
|
|
||||||
|
/// Health older than this and the node is ineligible. Deliberately close to the
|
||||||
|
/// 20s at which `fleet::spawn_node_sweeper` marks a node offline: the window in
|
||||||
|
/// which a node is "online with unreadable memory" should be narrow.
|
||||||
|
pub const MAX_HEALTH_AGE_SECS: f64 = 30.0;
|
||||||
|
|
||||||
|
/// Beszel metrics older than this rank as zero headroom. Only a tiebreak.
|
||||||
|
const MAX_METRICS_AGE_SECS: f64 = 60.0;
|
||||||
|
|
||||||
|
/// A node that can take at least one more phase VM.
|
||||||
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
pub struct NodeCapacity {
|
||||||
|
pub node_id: NodeId,
|
||||||
|
pub name: String,
|
||||||
|
/// How many MORE 8 GiB VMs fit.
|
||||||
|
pub slots: i64,
|
||||||
|
pub headroom: f64,
|
||||||
|
pub committed_vms: i64,
|
||||||
|
pub mem_total_mib: i64,
|
||||||
|
pub used_eff_mib: i64,
|
||||||
|
pub disk_free_gib: i64,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why a node cannot take this phase. Each renders a distinct, actionable line —
|
||||||
|
/// "at capacity" and "we could not read it" send an operator to different places.
|
||||||
|
#[derive(Debug, Clone, PartialEq)]
|
||||||
|
pub enum Unfit {
|
||||||
|
Draining,
|
||||||
|
NotConnected,
|
||||||
|
NoRecentHealth { age_secs: Option<f64> },
|
||||||
|
CapacityUnknown { err: String },
|
||||||
|
AtCapacity { committed: i64, used_eff_mib: i64, mem_total_mib: i64 },
|
||||||
|
NoDisk { free_gib: i64 },
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Unfit {
|
||||||
|
pub fn reason(&self) -> String {
|
||||||
|
match self {
|
||||||
|
Unfit::Draining => "draining".into(),
|
||||||
|
Unfit::NotConnected => "daemon not connected".into(),
|
||||||
|
Unfit::NoRecentHealth { age_secs } => match age_secs {
|
||||||
|
Some(a) => format!("health {a:.0}s stale (max {MAX_HEALTH_AGE_SECS:.0}s)"),
|
||||||
|
None => "never reported health".into(),
|
||||||
|
},
|
||||||
|
Unfit::CapacityUnknown { err } => format!("could not read running VMs: {err}"),
|
||||||
|
Unfit::AtCapacity { committed, used_eff_mib, mem_total_mib } => format!(
|
||||||
|
"at capacity: {committed} VM(s), {used_eff_mib}/{mem_total_mib} MiB committed"
|
||||||
|
),
|
||||||
|
Unfit::NoDisk { free_gib } => format!("only {free_gib} GiB free"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Why placement produced no node. Distinguished because the operator response
|
||||||
|
/// differs: wait, fix a daemon, or build an image.
|
||||||
|
#[derive(Debug, Clone)]
|
||||||
|
pub enum PlacementError {
|
||||||
|
/// No node has the image / KVM at all. Not a capacity problem.
|
||||||
|
NoCapableNode { backend: String, how_to_fix: String },
|
||||||
|
/// Every capable node is full. Transient — the caller should queue.
|
||||||
|
FleetAtCapacity { report: String },
|
||||||
|
/// We could not READ capacity. Must never be reported as "full".
|
||||||
|
FleetUnreadable { report: String },
|
||||||
|
}
|
||||||
|
|
||||||
|
impl PlacementError {
|
||||||
|
/// Whether the caller should wait and retry rather than fail the work.
|
||||||
|
pub fn is_transient(&self) -> bool {
|
||||||
|
matches!(
|
||||||
|
self,
|
||||||
|
PlacementError::FleetAtCapacity { .. } | PlacementError::FleetUnreadable { .. }
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn message(&self) -> String {
|
||||||
|
match self {
|
||||||
|
PlacementError::NoCapableNode { backend, how_to_fix } => {
|
||||||
|
format!("no online node can run backend {backend:?} — {how_to_fix}")
|
||||||
|
}
|
||||||
|
PlacementError::FleetAtCapacity { report } => format!(
|
||||||
|
"fleet at capacity — a phase VM runs up to 60 min; this phase waits for a slot.\n{report}"
|
||||||
|
),
|
||||||
|
PlacementError::FleetUnreadable { report } => format!(
|
||||||
|
"cannot read node capacity — this is NOT a full fleet; check the node daemons.\n{report}"
|
||||||
|
),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// What the host itself costs, excluding its phase VMs.
|
||||||
|
///
|
||||||
|
/// With nothing committed the answer is simply what the node reports. With VMs
|
||||||
|
/// committed it cannot be measured, only remembered or inferred — and inference
|
||||||
|
/// is where this went wrong: subtracting the VMs' FULL 8 GiB claim from
|
||||||
|
/// observed usage assumes they have already consumed it. A VM booted seconds
|
||||||
|
/// ago holds about an eighth of that, so the subtraction goes negative, hits
|
||||||
|
/// the floor, and hands back memory the host is really using.
|
||||||
|
///
|
||||||
|
/// Measured on morpheus (31757 MiB total, 4314 MiB idle, 2 slots) with 2 VMs
|
||||||
|
/// committed and young: the inferred baseline collapsed to the 2048 floor,
|
||||||
|
/// freeing 2266 MiB — exactly enough to admit a 3rd VM to a 2-slot node. The
|
||||||
|
/// `capacity` harness scenario caught it on its first full run.
|
||||||
|
///
|
||||||
|
/// So prefer the remembered idle reading, and take the LARGER of it and the
|
||||||
|
/// inference: a host that has genuinely started doing non-VM work must not be
|
||||||
|
/// under-charged just because it was once idle at a lower number.
|
||||||
|
fn host_baseline(mem_used_mib: i64, committed_vms: i64, baseline_mib: Option<i64>) -> i64 {
|
||||||
|
if committed_vms <= 0 {
|
||||||
|
// Directly observable, and the only moment it is.
|
||||||
|
return mem_used_mib.max(HOST_BASELINE_FLOOR_MIB);
|
||||||
|
}
|
||||||
|
let inferred = mem_used_mib - committed_vms * MEM_PER_VM_MIB as i64;
|
||||||
|
inferred
|
||||||
|
.max(baseline_mib.unwrap_or(0))
|
||||||
|
.max(HOST_BASELINE_FLOOR_MIB)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The whole capacity decision for one node, as pure arithmetic.
|
||||||
|
///
|
||||||
|
/// Separated from every I/O concern so the numbers can be tested against measured
|
||||||
|
/// fleet values without a database, a hub, or a VM.
|
||||||
|
pub fn capacity_of(
|
||||||
|
node_id: NodeId,
|
||||||
|
name: &str,
|
||||||
|
mem_total_mib: i64,
|
||||||
|
mem_used_mib: i64,
|
||||||
|
disk_free_gib: i64,
|
||||||
|
committed_vms: i64,
|
||||||
|
// What this node used the last time it was seen with nothing committed.
|
||||||
|
// `None` before it has ever been observed idle.
|
||||||
|
baseline_mib: Option<i64>,
|
||||||
|
headroom: f64,
|
||||||
|
) -> Result<NodeCapacity, Unfit> {
|
||||||
|
let host_baseline = host_baseline(mem_used_mib, committed_vms, baseline_mib);
|
||||||
|
let committed_use = committed_vms * MEM_PER_VM_MIB as i64 + host_baseline;
|
||||||
|
|
||||||
|
// The worse of the two views. Observed alone under-counts a freshly booted
|
||||||
|
// VM; committed alone under-counts a host doing real work outside its VMs.
|
||||||
|
let used_eff = mem_used_mib.max(committed_use);
|
||||||
|
let free = mem_total_mib - used_eff - HOST_RESERVE_MIB;
|
||||||
|
let slots = if free <= 0 { 0 } else { free / MEM_PER_VM_MIB as i64 };
|
||||||
|
|
||||||
|
if disk_free_gib - DISK_PER_VM_GIB < DISK_RESERVE_GIB {
|
||||||
|
return Err(Unfit::NoDisk { free_gib: disk_free_gib });
|
||||||
|
}
|
||||||
|
if slots < 1 {
|
||||||
|
return Err(Unfit::AtCapacity {
|
||||||
|
committed: committed_vms,
|
||||||
|
used_eff_mib: used_eff,
|
||||||
|
mem_total_mib,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Ok(NodeCapacity {
|
||||||
|
node_id,
|
||||||
|
name: name.to_string(),
|
||||||
|
slots,
|
||||||
|
headroom,
|
||||||
|
committed_vms,
|
||||||
|
mem_total_mib,
|
||||||
|
used_eff_mib: used_eff,
|
||||||
|
disk_free_gib,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Admission inputs drawn from an `EvalRow`, or why the node is ineligible.
|
||||||
|
///
|
||||||
|
/// Fail-closed on stale or absent health: a node whose memory we cannot read is
|
||||||
|
/// one whose capacity we would be guessing at.
|
||||||
|
pub fn from_eval(
|
||||||
|
row: &EvalRow,
|
||||||
|
name: &str,
|
||||||
|
committed_vms: i64,
|
||||||
|
) -> Result<NodeCapacity, Unfit> {
|
||||||
|
if row.status == "draining" {
|
||||||
|
return Err(Unfit::Draining);
|
||||||
|
}
|
||||||
|
let fresh = row
|
||||||
|
.health_age_secs
|
||||||
|
.is_some_and(|a| a <= MAX_HEALTH_AGE_SECS);
|
||||||
|
let (Some(total), Some(used)) = (row.mem_total_bytes, row.mem_used_bytes) else {
|
||||||
|
return Err(Unfit::NoRecentHealth { age_secs: row.health_age_secs });
|
||||||
|
};
|
||||||
|
if !fresh || total <= 0 {
|
||||||
|
return Err(Unfit::NoRecentHealth { age_secs: row.health_age_secs });
|
||||||
|
}
|
||||||
|
const MIB: i64 = 1024 * 1024;
|
||||||
|
const GIB: i64 = 1024 * 1024 * 1024;
|
||||||
|
capacity_of(
|
||||||
|
row.node_id,
|
||||||
|
name,
|
||||||
|
total / MIB,
|
||||||
|
used / MIB,
|
||||||
|
row.disk_free_bytes.unwrap_or(0) / GIB,
|
||||||
|
committed_vms,
|
||||||
|
row.mem_baseline_mib,
|
||||||
|
row.headroom_fresh(MAX_METRICS_AGE_SECS),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Rank admissible nodes: most free slots first, then live headroom, then id.
|
||||||
|
///
|
||||||
|
/// Slots before headroom SPREADS load rather than stacking it — two missions
|
||||||
|
/// launched together go to different machines. Headroom breaks ties with
|
||||||
|
/// real-time load, which is where a node mid-`cargo build` loses to an idle peer.
|
||||||
|
/// Node id last so the same fleet state always yields the same answer; the old
|
||||||
|
/// `last_seen DESC` made placement unreproducible between two identical runs.
|
||||||
|
pub fn rank(mut fit: Vec<NodeCapacity>) -> Vec<NodeCapacity> {
|
||||||
|
fit.sort_by(|a, b| {
|
||||||
|
b.slots
|
||||||
|
.cmp(&a.slots)
|
||||||
|
.then(
|
||||||
|
b.headroom
|
||||||
|
.partial_cmp(&a.headroom)
|
||||||
|
.unwrap_or(std::cmp::Ordering::Equal),
|
||||||
|
)
|
||||||
|
.then(a.node_id.as_uuid().cmp(&b.node_id.as_uuid()))
|
||||||
|
});
|
||||||
|
fit
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One line per node, for logs and for the message an operator reads.
|
||||||
|
pub fn report(fit: &[NodeCapacity], unfit: &[(NodeId, String, Unfit)]) -> String {
|
||||||
|
let mut out = Vec::new();
|
||||||
|
for f in fit {
|
||||||
|
out.push(format!(
|
||||||
|
" {}: {} slot(s) free, {} VM(s) committed, {}/{} MiB, headroom {:.0}",
|
||||||
|
f.name, f.slots, f.committed_vms, f.used_eff_mib, f.mem_total_mib, f.headroom
|
||||||
|
));
|
||||||
|
}
|
||||||
|
for (_, name, why) in unfit {
|
||||||
|
out.push(format!(" {name}: UNFIT — {}", why.reason()));
|
||||||
|
}
|
||||||
|
if out.is_empty() {
|
||||||
|
out.push(" (no capable nodes)".into());
|
||||||
|
}
|
||||||
|
out.join("\n")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Count a node's commitments, unioning booted VMs with pinned-not-yet-booted
|
||||||
|
/// phases BY IDENTITY.
|
||||||
|
///
|
||||||
|
/// A composed graph's step VMs (`...-s0`, `-s1`) each count: each is a real
|
||||||
|
/// Firecracker process holding 8 GiB. A pinned phase counts only while no live VM
|
||||||
|
/// carries its id — otherwise the same claim would be counted twice and the fleet
|
||||||
|
/// would shrink by the number of phases currently starting.
|
||||||
|
pub fn commitments(live_vm_ids: &[String], pinned_keys: &[String]) -> i64 {
|
||||||
|
let live = live_vm_ids.len() as i64;
|
||||||
|
let unbooted = pinned_keys
|
||||||
|
.iter()
|
||||||
|
.filter(|k| !live_vm_ids.iter().any(|v| v.starts_with(k.as_str())))
|
||||||
|
.count() as i64;
|
||||||
|
live + unbooted
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Every backend a phase needs on ONE node: the mission's, plus each backend
|
||||||
|
/// named by a node of its composed graph.
|
||||||
|
///
|
||||||
|
/// The roster stores them as `config.roster.nodes[].attrs.backend`, and they are
|
||||||
|
/// the reason this function exists. A 2-member roster with
|
||||||
|
/// `verifier@canary-claude` was placed on a node holding `claude` and not
|
||||||
|
/// `canary-claude`; the graph's first node ran, the second died with
|
||||||
|
/// `no rootfs for backend "canary-claude" on this node`, and the mission
|
||||||
|
/// delivered half its work and failed. Placement had asked only about the
|
||||||
|
/// mission's own backend, which was true and insufficient.
|
||||||
|
pub fn required_backends(mission_backend: Option<&str>, roster: Option<&serde_json::Value>) -> Vec<String> {
|
||||||
|
let mut out = vec![cm_db::repo::nodes::backend_key(mission_backend).to_string()];
|
||||||
|
if let Some(nodes) = roster.and_then(|r| r.get("nodes")).and_then(|n| n.as_array()) {
|
||||||
|
for n in nodes {
|
||||||
|
if let Some(b) = n
|
||||||
|
.get("attrs")
|
||||||
|
.and_then(|a| a.get("backend"))
|
||||||
|
.and_then(|b| b.as_str())
|
||||||
|
.filter(|b| !b.trim().is_empty())
|
||||||
|
{
|
||||||
|
out.push(b.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
out.sort();
|
||||||
|
out.dedup();
|
||||||
|
out
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Survey every capable node: which can take a phase VM, and why the rest cannot.
|
||||||
|
///
|
||||||
|
/// `vm_list` is asked of each candidate in parallel with a short deadline. A node
|
||||||
|
/// that will not answer is `CapacityUnknown` and therefore ineligible — we cannot
|
||||||
|
/// count what we cannot see, and guessing zero is how a node gets double-booked.
|
||||||
|
pub async fn survey(
|
||||||
|
pool: &sqlx::PgPool,
|
||||||
|
hub: &crate::fleet::NodeHub,
|
||||||
|
workspace_id: uuid::Uuid,
|
||||||
|
// EVERY backend the work needs, not just the mission's. A composed graph
|
||||||
|
// runs on ONE node and its nodes may each name their own — the roster's
|
||||||
|
// whole purpose is an independent verifier on another provider — so the
|
||||||
|
// node has to hold all of their rootfs images.
|
||||||
|
backends: &[String],
|
||||||
|
) -> Result<(Vec<NodeCapacity>, Vec<(NodeId, String, Unfit)>), String> {
|
||||||
|
let candidates = cm_db::repo::nodes::online_for_backends(pool, workspace_id, backends)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("looking up nodes for backends {backends:?}: {e}"))?;
|
||||||
|
if candidates.is_empty() {
|
||||||
|
return Ok((Vec::new(), Vec::new()));
|
||||||
|
}
|
||||||
|
|
||||||
|
let evals = cm_db::repo::node_metrics::eval_all(pool)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("reading node metrics: {e}"))?;
|
||||||
|
let pinned = cm_db::repo::nodes::pinned_microvm_phases(pool, workspace_id)
|
||||||
|
.await
|
||||||
|
.map_err(|e| format!("reading pinned phases: {e}"))?;
|
||||||
|
|
||||||
|
let names = node_names(pool, workspace_id).await;
|
||||||
|
let mut fit = Vec::new();
|
||||||
|
let mut unfit = Vec::new();
|
||||||
|
for node in candidates {
|
||||||
|
let row = evals.iter().find(|e| e.node_id == node);
|
||||||
|
let name = names
|
||||||
|
.get(&node.as_uuid())
|
||||||
|
.cloned()
|
||||||
|
.unwrap_or_else(|| node.as_uuid().to_string()[..8].to_string());
|
||||||
|
|
||||||
|
// Not connected: nothing can be asked of it, and nothing can run on it.
|
||||||
|
if !hub.is_connected(node) {
|
||||||
|
unfit.push((node, name, Unfit::NotConnected));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let Some(row) = row else {
|
||||||
|
unfit.push((node, name, Unfit::NoRecentHealth { age_secs: None }));
|
||||||
|
continue;
|
||||||
|
};
|
||||||
|
|
||||||
|
// Commitments: booted VMs unioned with phases pinned here but not yet
|
||||||
|
// booted, by the deterministic id both sides agree on.
|
||||||
|
let live = match crate::microvm_client::list(hub, node).await {
|
||||||
|
Ok(v) => v,
|
||||||
|
Err(e) => {
|
||||||
|
unfit.push((node, name, Unfit::CapacityUnknown { err: e }));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let keys: Vec<String> = pinned
|
||||||
|
.iter()
|
||||||
|
.filter(|(n, _, _)| *n == node)
|
||||||
|
.map(|(_, phase, iter)| crate::microvm_executor::vm_id_for(*phase, *iter, None))
|
||||||
|
.collect();
|
||||||
|
let committed = commitments(&live, &keys);
|
||||||
|
|
||||||
|
// An idle node is the ONLY time its own footprint is measurable rather
|
||||||
|
// than inferred, so take the reading whenever we get one. Cheap: an
|
||||||
|
// UPDATE per idle node per survey, and it is what stops a young VM's
|
||||||
|
// unconsumed memory from being handed out a second time.
|
||||||
|
if committed == 0 {
|
||||||
|
if let Some(used) = row.mem_used_bytes.filter(|_| {
|
||||||
|
row.health_age_secs
|
||||||
|
.is_some_and(|a| a <= MAX_HEALTH_AGE_SECS)
|
||||||
|
}) {
|
||||||
|
let mib = used / (1024 * 1024);
|
||||||
|
if row.mem_baseline_mib != Some(mib) {
|
||||||
|
let _ = cm_db::repo::nodes::set_mem_baseline(pool, node, mib).await;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
match from_eval(row, &name, committed) {
|
||||||
|
Ok(c) => fit.push(c),
|
||||||
|
Err(why) => unfit.push((node, name, why)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Ok((rank(fit), unfit))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Node names for readable reports. A capacity report naming two machines
|
||||||
|
/// "New node" is a report nobody can act on.
|
||||||
|
async fn node_names(
|
||||||
|
pool: &sqlx::PgPool,
|
||||||
|
workspace_id: uuid::Uuid,
|
||||||
|
) -> std::collections::HashMap<uuid::Uuid, String> {
|
||||||
|
sqlx::query_as::<_, (uuid::Uuid, String)>(
|
||||||
|
"SELECT id, name FROM nodes WHERE workspace_id = $1",
|
||||||
|
)
|
||||||
|
.bind(workspace_id)
|
||||||
|
.fetch_all(pool)
|
||||||
|
.await
|
||||||
|
.unwrap_or_default()
|
||||||
|
.into_iter()
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Choose a node for a phase, honouring an explicit target as a REQUEST.
|
||||||
|
///
|
||||||
|
/// `want` is honoured only if that node is genuinely admissible — the same
|
||||||
|
/// "a request, not a guarantee" rule the orchestrator already applied to
|
||||||
|
/// capability, now extended to capacity and draining.
|
||||||
|
pub async fn choose(
|
||||||
|
pool: &sqlx::PgPool,
|
||||||
|
hub: &crate::fleet::NodeHub,
|
||||||
|
workspace_id: uuid::Uuid,
|
||||||
|
backends: &[String],
|
||||||
|
want: Option<uuid::Uuid>,
|
||||||
|
) -> Result<NodeId, PlacementError> {
|
||||||
|
let named = backends.join(", ");
|
||||||
|
let how_to_fix = format!(
|
||||||
|
"needs /dev/kvm + firecracker (scripts/fc-node-setup.sh) AND the {named} rootfs \
|
||||||
|
built on ONE node (scripts/fc-build-rootfs.sh <host> <image> <name>) — a \
|
||||||
|
composed graph runs on a single node, so that node needs every image its \
|
||||||
|
nodes ask for"
|
||||||
|
);
|
||||||
|
let (fit, unfit) = survey(pool, hub, workspace_id, backends).await.map_err(|e| {
|
||||||
|
PlacementError::FleetUnreadable { report: format!(" survey failed: {e}") }
|
||||||
|
})?;
|
||||||
|
|
||||||
|
if fit.is_empty() && unfit.is_empty() {
|
||||||
|
return Err(PlacementError::NoCapableNode {
|
||||||
|
backend: named,
|
||||||
|
how_to_fix,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
let report = report(&fit, &unfit);
|
||||||
|
|
||||||
|
// `want` is ADVISORY, always. The only caller passes `missions.target_node_id`,
|
||||||
|
// which is simply where the PREVIOUS phase ran — not an operator's choice.
|
||||||
|
// Treating it as a requirement had two consequences, both wrong:
|
||||||
|
//
|
||||||
|
// - a previous node that had since filled up (or gone unreadable) failed
|
||||||
|
// the phase outright: `TargetUnfit` is not transient, so it never
|
||||||
|
// reached the queue. Note this was NOT the drain case — a draining node
|
||||||
|
// is already excluded by `online_for_backend`'s `status = 'online'`, so
|
||||||
|
// it never reaches `unfit` at all and the pin simply falls through.
|
||||||
|
// `drain-midmission` passes either way; the path it does not cover is
|
||||||
|
// "phase 1's node is now full", which is the one that used to fail.
|
||||||
|
// - and while the node stayed fit, every later phase went back to it
|
||||||
|
// regardless of ranking — accidental mission-to-node affinity, which
|
||||||
|
// this module's own header says must not exist.
|
||||||
|
//
|
||||||
|
// Mission state lives on the gateway (inject -> run -> collect -> destroy),
|
||||||
|
// so re-placing costs nothing. Prefer the pin when it still fits; say out
|
||||||
|
// loud why it did not when it does not, and rank as usual.
|
||||||
|
if let Some(want) = want {
|
||||||
|
if let Some(c) = fit.iter().find(|c| c.node_id.as_uuid() == want) {
|
||||||
|
return Ok(c.node_id);
|
||||||
|
}
|
||||||
|
if let Some((_, name, why)) = unfit.iter().find(|(n, _, _)| n.as_uuid() == want) {
|
||||||
|
eprintln!(
|
||||||
|
"vm_placement: the previous phase's node {name} is {} — re-placing this phase",
|
||||||
|
why.reason()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if let Some(best) = fit.into_iter().next() {
|
||||||
|
return Ok(best.node_id);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Nothing fit. Distinguish "full" from "blind": an operator sent to look for
|
||||||
|
// a load problem that is really a dead daemon wastes the outage.
|
||||||
|
let blind = unfit.iter().all(|(_, _, w)| {
|
||||||
|
matches!(w, Unfit::CapacityUnknown { .. } | Unfit::NotConnected | Unfit::NoRecentHealth { .. })
|
||||||
|
});
|
||||||
|
Err(if blind {
|
||||||
|
PlacementError::FleetUnreadable { report }
|
||||||
|
} else {
|
||||||
|
PlacementError::FleetAtCapacity { report }
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn nid(n: u128) -> NodeId {
|
||||||
|
NodeId::from(uuid::Uuid::from_u128(n))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// THE test. Measured on tank: 60 GiB total, and five VMs booted moments ago
|
||||||
|
/// showing only ~12 GiB used because the guests have not touched their claim.
|
||||||
|
///
|
||||||
|
/// Observed-usage-only arithmetic says (61440-12000-4096)/8192 = 5 more VMs.
|
||||||
|
/// The node has room for ONE. Booking those five is a node in swap, and every
|
||||||
|
/// VM on it slows down together.
|
||||||
|
/// A node whose VMs have not yet consumed their claim must not hand the
|
||||||
|
/// difference out again.
|
||||||
|
///
|
||||||
|
/// This is the bug the `capacity` harness scenario found on its first full
|
||||||
|
/// run — "morpheus peaked at 3 concurrent VM(s) with only 2 slot(s)" — and
|
||||||
|
/// the numbers here are that node's real ones. Idle it reports 4314 MiB of
|
||||||
|
/// 31757 and the survey correctly gives it 2 slots. Two VMs later, each
|
||||||
|
/// holding roughly 1 GiB of its 8 GiB, observed usage is ~6314 MiB;
|
||||||
|
/// inferring the baseline as 6314 - 16384 goes negative, clamps to the
|
||||||
|
/// 2048 floor, and invents 2266 MiB — exactly one more VM than exists.
|
||||||
|
/// A previous node that is no longer usable re-places the next phase; it
|
||||||
|
/// does not fail it.
|
||||||
|
///
|
||||||
|
/// `choose` treated `missions.target_node_id` — which is only ever "where
|
||||||
|
/// the last phase ran" — as a hard requirement, so a pinned node that had
|
||||||
|
/// since FILLED UP produced `TargetUnfit`, which is not transient, and the
|
||||||
|
/// phase failed instead of queueing or moving. It also gave every later
|
||||||
|
/// phase silent affinity back to the first node.
|
||||||
|
///
|
||||||
|
/// The drain case is not this one and never was: `online_for_backend`
|
||||||
|
/// filters on `status = 'online'`, so a draining node is not a candidate
|
||||||
|
/// and the pin falls through to ranking. `drain-midmission` passes on both
|
||||||
|
/// the old and new code, which is why the capacity half needs this test.
|
||||||
|
/// A composed graph's per-node backends are part of what placement needs.
|
||||||
|
///
|
||||||
|
/// The full harness found this: a 2-member roster with
|
||||||
|
/// `verifier@canary-claude` was placed on a node holding `claude` and not
|
||||||
|
/// `canary-claude`. The first graph node ran, the second died with
|
||||||
|
/// `no rootfs for backend "canary-claude" on this node`, and the mission
|
||||||
|
/// delivered half its work and failed. Placement had asked only about the
|
||||||
|
/// mission's own backend — true, and insufficient.
|
||||||
|
#[test]
|
||||||
|
fn a_composed_graph_needs_every_backend_its_nodes_name() {
|
||||||
|
let roster = serde_json::json!({
|
||||||
|
"kind": "pipeline",
|
||||||
|
"nodes": [
|
||||||
|
{"id": "n0", "role": "implementer"},
|
||||||
|
{"id": "n1", "role": "verifier", "attrs": {"backend": "canary-claude"}},
|
||||||
|
],
|
||||||
|
});
|
||||||
|
assert_eq!(
|
||||||
|
required_backends(Some("claude"), Some(&roster)),
|
||||||
|
vec!["canary-claude".to_string(), "claude".to_string()],
|
||||||
|
"both images have to be on the ONE node the graph runs on"
|
||||||
|
);
|
||||||
|
|
||||||
|
// A solo mission is unchanged — this must not make ordinary placement
|
||||||
|
// stricter than it was.
|
||||||
|
assert_eq!(required_backends(Some("claude"), None), vec!["claude"]);
|
||||||
|
assert_eq!(required_backends(None, None), vec!["default"]);
|
||||||
|
|
||||||
|
// A node with no explicit backend inherits the mission's, so it adds
|
||||||
|
// nothing. Deduped, or a 5-node graph would ask for `claude` five times
|
||||||
|
// and the containment query would still be right but the error message
|
||||||
|
// would be nonsense.
|
||||||
|
let inherit = serde_json::json!({"nodes": [
|
||||||
|
{"id": "n0", "role": "a"},
|
||||||
|
{"id": "n1", "role": "b", "attrs": {}},
|
||||||
|
{"id": "n2", "role": "c", "attrs": {"backend": ""}},
|
||||||
|
]});
|
||||||
|
assert_eq!(required_backends(Some("claude"), Some(&inherit)), vec!["claude"]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn an_unfit_previous_node_is_re_placed_not_refused() {
|
||||||
|
let drained = uuid::Uuid::from_u128(1);
|
||||||
|
let healthy = capacity_of(nid(2), "tank", 61440, 6144, 800, 0, None, 90.0).unwrap();
|
||||||
|
|
||||||
|
// Stand in for `choose`'s decision: the pin is consulted, then dropped.
|
||||||
|
let fit = vec![healthy.clone()];
|
||||||
|
let picked = fit
|
||||||
|
.iter()
|
||||||
|
.find(|c| c.node_id.as_uuid() == drained)
|
||||||
|
.or_else(|| fit.first())
|
||||||
|
.expect("a fit node exists");
|
||||||
|
assert_eq!(
|
||||||
|
picked.node_id,
|
||||||
|
nid(2),
|
||||||
|
"with the pinned node absent from `fit`, ranking must still yield a node"
|
||||||
|
);
|
||||||
|
|
||||||
|
// And the error that used to be produced here no longer exists, so it
|
||||||
|
// cannot be reintroduced as a non-transient failure by accident.
|
||||||
|
for e in [
|
||||||
|
PlacementError::FleetAtCapacity { report: String::new() },
|
||||||
|
PlacementError::FleetUnreadable { report: String::new() },
|
||||||
|
] {
|
||||||
|
assert!(e.is_transient(), "both no-node outcomes must QUEUE, not fail");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_young_vms_unconsumed_memory_is_not_handed_out_twice() {
|
||||||
|
// Idle: the reading that gets remembered, and the slot count it implies.
|
||||||
|
let idle = capacity_of(nid(3), "morpheus", 31757, 4314, 312, 0, None, 90.0)
|
||||||
|
.expect("an idle morpheus fits VMs");
|
||||||
|
assert_eq!(idle.slots, 2, "idle capacity is the number we are defending");
|
||||||
|
|
||||||
|
// Two committed, both young. WITHOUT the remembered baseline this
|
||||||
|
// returned 1 slot and admitted a third VM.
|
||||||
|
let inferred = capacity_of(nid(3), "morpheus", 31757, 6314, 312, 2, None, 90.0);
|
||||||
|
assert!(
|
||||||
|
inferred.is_ok(),
|
||||||
|
"the old inference is preserved as the no-baseline fallback"
|
||||||
|
);
|
||||||
|
|
||||||
|
// WITH it, the node is correctly full.
|
||||||
|
let remembered = capacity_of(nid(3), "morpheus", 31757, 6314, 312, 2, Some(4314), 90.0);
|
||||||
|
assert!(
|
||||||
|
matches!(remembered, Err(Unfit::AtCapacity { committed: 2, .. })),
|
||||||
|
"a 2-slot node with 2 VMs committed is FULL, got {remembered:?}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A host that starts doing real work outside its VMs is charged for it.
|
||||||
|
///
|
||||||
|
/// The remembered baseline is a floor, not a substitute. If it replaced the
|
||||||
|
/// inference outright, a node that was idle at 4 GiB and is now running a
|
||||||
|
/// 20 GiB build would still be scored as if it were idle — the same
|
||||||
|
/// over-commit, arrived at from the opposite direction.
|
||||||
|
#[test]
|
||||||
|
fn a_remembered_baseline_never_under_charges_a_busy_host() {
|
||||||
|
// 1 VM committed and consumed (8192), plus 20 GiB of non-VM work.
|
||||||
|
let used = 8192 + 20480;
|
||||||
|
let c = capacity_of(nid(3), "busy", 61440, used, 800, 1, Some(4096), 90.0)
|
||||||
|
.expect("still has room");
|
||||||
|
// Inference says 20480; the stale 4096 baseline must not win.
|
||||||
|
assert_eq!(c.used_eff_mib, used, "observed usage is charged in full");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn a_sold_out_node_is_not_mistaken_for_an_idle_one() {
|
||||||
|
let observed_only =
|
||||||
|
capacity_of(nid(1), "tank", 61440, 12000, 800, 0, None, 50.0).expect("fits");
|
||||||
|
assert_eq!(
|
||||||
|
observed_only.slots, 5,
|
||||||
|
"this is what utilisation alone claims — the bug being fixed"
|
||||||
|
);
|
||||||
|
|
||||||
|
let with_commitments =
|
||||||
|
capacity_of(nid(1), "tank", 61440, 12000, 800, 5, None, 50.0).expect("fits");
|
||||||
|
assert_eq!(
|
||||||
|
with_commitments.slots, 1,
|
||||||
|
"five 8 GiB claims are already spoken for, whatever the guests have touched"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The measured idle fleet. Numbers from `free`/`df` on the real machines, so
|
||||||
|
/// a future change to the constants has to face what it does to real nodes.
|
||||||
|
#[test]
|
||||||
|
fn the_measured_fleet_gets_the_slots_it_actually_has() {
|
||||||
|
// tank: 60 GiB, ~6 GiB used at idle.
|
||||||
|
let tank = capacity_of(nid(1), "tank", 61440, 6144, 869, 0, None, 90.0).unwrap();
|
||||||
|
assert_eq!(tank.slots, 6);
|
||||||
|
// architect: 60 GiB, ~7 GiB used.
|
||||||
|
let arch = capacity_of(nid(2), "architect", 61440, 7168, 388, 0, None, 90.0).unwrap();
|
||||||
|
assert_eq!(arch.slots, 6);
|
||||||
|
// morpheus: 31 GiB — deliberately the conservative 2, not 3. Three VMs
|
||||||
|
// would leave under 2 GiB for the host, which is where the OOM killer
|
||||||
|
// lives, and an OOM-killed VM looks like an agent that gave up.
|
||||||
|
let morph = capacity_of(nid(3), "morpheus", 31744, 5120, 312, 0, None, 90.0).unwrap();
|
||||||
|
assert_eq!(morph.slots, 2);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Spread, don't stack; then real load; then determinism.
|
||||||
|
#[test]
|
||||||
|
fn ranking_prefers_free_slots_then_headroom_then_a_stable_order() {
|
||||||
|
let a = capacity_of(nid(1), "a", 61440, 6144, 800, 0, None, 40.0).unwrap(); // 6 slots
|
||||||
|
let b = capacity_of(nid(2), "b", 61440, 6144, 800, 3, None, 90.0).unwrap(); // 3 slots
|
||||||
|
assert_eq!(rank(vec![b.clone(), a.clone()])[0].name, "a", "more slots wins");
|
||||||
|
|
||||||
|
// Equal slots → the node under less real load.
|
||||||
|
let busy = capacity_of(nid(3), "busy", 61440, 6144, 800, 0, None, 10.0).unwrap();
|
||||||
|
let idle = capacity_of(nid(4), "idle", 61440, 6144, 800, 0, None, 95.0).unwrap();
|
||||||
|
assert_eq!(rank(vec![busy.clone(), idle.clone()])[0].name, "idle");
|
||||||
|
|
||||||
|
// Equal on both → same answer twice. `last_seen DESC` could not promise this.
|
||||||
|
let x = capacity_of(nid(9), "x", 61440, 6144, 800, 0, None, 50.0).unwrap();
|
||||||
|
let y = capacity_of(nid(8), "y", 61440, 6144, 800, 0, None, 50.0).unwrap();
|
||||||
|
assert_eq!(rank(vec![x.clone(), y.clone()])[0].name, "y");
|
||||||
|
assert_eq!(rank(vec![y, x])[0].name, "y");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A booted VM and its pinned phase row are ONE claim, not two.
|
||||||
|
#[test]
|
||||||
|
fn commitments_union_by_identity_rather_than_adding() {
|
||||||
|
let live = vec!["m-abc123def456-0".to_string(), "m-abc123def456-0-s2".to_string()];
|
||||||
|
// Same phase as the live VMs: already counted.
|
||||||
|
assert_eq!(commitments(&live, &["m-abc123def456-0".to_string()]), 2);
|
||||||
|
// A different phase, pinned but not yet booted: a real additional claim.
|
||||||
|
assert_eq!(
|
||||||
|
commitments(&live, &["m-999888777666-0".to_string()]),
|
||||||
|
3,
|
||||||
|
"a phase chosen seconds ago holds 8 GiB no node can report yet"
|
||||||
|
);
|
||||||
|
assert_eq!(commitments(&[], &["m-1-0".into(), "m-2-0".into()]), 2);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Disk is a hard gate, and it is checked BEFORE capacity so the message
|
||||||
|
/// names the real problem.
|
||||||
|
#[test]
|
||||||
|
fn a_node_short_of_disk_is_refused_even_with_memory_to_spare() {
|
||||||
|
let e = capacity_of(nid(1), "tank", 61440, 6144, 25, 0, None, 90.0).unwrap_err();
|
||||||
|
assert!(matches!(e, Unfit::NoDisk { free_gib: 25 }), "{e:?}");
|
||||||
|
assert!(e.reason().contains("25 GiB"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Stale metrics may cost a tie; they may never win one, and they may never
|
||||||
|
/// exclude a node — that is health's job.
|
||||||
|
#[test]
|
||||||
|
fn stale_beszel_metrics_demote_but_do_not_exclude() {
|
||||||
|
let mut row = row_for(nid(1), 61440 * MIB_T, 6144 * MIB_T, 800 * GIB_T);
|
||||||
|
row.metrics_age_secs = Some(3600.0);
|
||||||
|
row.health_age_secs = Some(3.0);
|
||||||
|
row.cpu_pct = Some(5.0);
|
||||||
|
let fit = from_eval(&row, "tank", 0).expect("still eligible");
|
||||||
|
assert_eq!(fit.headroom, 95.0, "fresh health carries the headroom");
|
||||||
|
|
||||||
|
row.health_age_secs = Some(3600.0);
|
||||||
|
assert!(
|
||||||
|
matches!(from_eval(&row, "tank", 0), Err(Unfit::NoRecentHealth { .. })),
|
||||||
|
"stale HEALTH is exclusion, because memory is then a guess"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The two failures an operator must never confuse.
|
||||||
|
#[test]
|
||||||
|
fn unreadable_capacity_never_reads_as_a_full_fleet() {
|
||||||
|
let full = PlacementError::FleetAtCapacity { report: " tank: 0 slots".into() };
|
||||||
|
let blind = PlacementError::FleetUnreadable { report: " tank: UNFIT".into() };
|
||||||
|
assert!(full.message().contains("at capacity"));
|
||||||
|
assert!(blind.message().contains("cannot read"));
|
||||||
|
assert!(
|
||||||
|
!blind.message().contains("at capacity"),
|
||||||
|
"sends an operator hunting a load problem that does not exist"
|
||||||
|
);
|
||||||
|
assert!(full.is_transient() && blind.is_transient());
|
||||||
|
let missing = PlacementError::NoCapableNode {
|
||||||
|
backend: "claude".into(),
|
||||||
|
how_to_fix: "build the image".into(),
|
||||||
|
};
|
||||||
|
assert!(!missing.is_transient(), "a missing image will not fix itself by waiting");
|
||||||
|
}
|
||||||
|
|
||||||
|
const MIB_T: i64 = 1024 * 1024;
|
||||||
|
const GIB_T: i64 = 1024 * 1024 * 1024;
|
||||||
|
|
||||||
|
fn row_for(node_id: NodeId, total: i64, used: i64, disk_free: i64) -> EvalRow {
|
||||||
|
EvalRow {
|
||||||
|
node_id,
|
||||||
|
workspace_id: cm_domain::WorkspaceId::from(uuid::Uuid::from_u128(1)),
|
||||||
|
status: "online".into(),
|
||||||
|
cpu_pct: None,
|
||||||
|
mem_pct: None,
|
||||||
|
disk_pct: None,
|
||||||
|
gpu_pct: None,
|
||||||
|
temp_max: None,
|
||||||
|
load1: None,
|
||||||
|
mem_total_bytes: Some(total),
|
||||||
|
mem_used_bytes: Some(used),
|
||||||
|
disk_free_bytes: Some(disk_free),
|
||||||
|
mem_baseline_mib: None,
|
||||||
|
health_age_secs: Some(3.0),
|
||||||
|
metrics_age_secs: Some(3.0),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A draining node is ineligible, not merely unattractive. The microVM path
|
||||||
|
/// never checked this before: a mission pinned before a drain kept feeding
|
||||||
|
/// VMs to a node an operator had cordoned.
|
||||||
|
#[test]
|
||||||
|
fn a_draining_node_is_ineligible() {
|
||||||
|
let mut row = row_for(nid(1), 61440 * MIB_T, 6144 * MIB_T, 800 * GIB_T);
|
||||||
|
row.status = "draining".into();
|
||||||
|
assert_eq!(from_eval(&row, "tank", 0), Err(Unfit::Draining));
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,516 @@
|
|||||||
|
//! The completion gate, moved into the agent's own loop.
|
||||||
|
//!
|
||||||
|
//! Every check this platform has on a phase runs **after** the agent has
|
||||||
|
//! finished: the evaluator judges `done_when`, capture notices that a coding
|
||||||
|
//! phase delivered nothing, and either verdict costs a whole new VM — a fresh
|
||||||
|
//! boot, a fresh inject, and an agent starting again with none of the context
|
||||||
|
//! that got it that far. Meanwhile the documented failure of a long-running
|
||||||
|
//! agent is that it *stops too early*.
|
||||||
|
//!
|
||||||
|
//! Claude Code's `Stop` hook is the seam. **Exit code 2 blocks the stop and
|
||||||
|
//! feeds stderr back to the model as the reason.** Measured, not read off docs —
|
||||||
|
//! an agent told "say hello and do nothing else", whose `Stop` hook exited 2
|
||||||
|
//! saying `evidence.txt` was missing, created `evidence.txt` and then stopped.
|
||||||
|
//!
|
||||||
|
//! # What it may and may not check
|
||||||
|
//!
|
||||||
|
//! Deliberately mechanical: whether the repository changed, and whether a
|
||||||
|
//! command the phase author wrote exits 0. NOT the `done_when` verdict — that is
|
||||||
|
//! an LLM judgement made host-side by a *different provider* on purpose
|
||||||
|
//! ([[evaluator-verification]]), and re-implementing it inside the VM would put
|
||||||
|
//! the agent's own environment in charge of grading the agent, which is the
|
||||||
|
//! correlated failure the independent judge exists to break.
|
||||||
|
//!
|
||||||
|
//! # The cap is load-bearing
|
||||||
|
//!
|
||||||
|
//! A gate with no ceiling turns a stuck agent into a wedged one: it would be
|
||||||
|
//! blocked, retry, be blocked again, and burn the hour-long turn budget instead
|
||||||
|
//! of failing in a way the operator can see. After [`MAX_BLOCKS`] the gate lets
|
||||||
|
//! the agent stop, records that it did, and leaves the verdict to the existing
|
||||||
|
//! post-hoc path — which still runs, unchanged.
|
||||||
|
//!
|
||||||
|
//! # Which hooks exist here
|
||||||
|
//!
|
||||||
|
//! `TaskCompleted` / `TeammateIdle` were the plan's chosen seam. Measured under
|
||||||
|
//! `claude -p`: they never fire, because no team forms in print mode at all.
|
||||||
|
//! `Stop`, `SubagentStop`, `PreToolUse`, `PostToolUse`, `UserPromptSubmit` and
|
||||||
|
//! `SessionStart` do.
|
||||||
|
|
||||||
|
|
||||||
|
/// How many times the gate may refuse a stop before it gives up and lets the
|
||||||
|
/// agent finish. Three is enough for "you wrote nothing" → "you wrote something"
|
||||||
|
/// → "your check passes" without ever approaching the turn budget.
|
||||||
|
pub const MAX_BLOCKS: u32 = 3;
|
||||||
|
|
||||||
|
/// The file the gate writes when it gives up and lets the agent stop with its
|
||||||
|
/// condition still failing.
|
||||||
|
///
|
||||||
|
/// A separate file rather than a line in the log, because the log is not
|
||||||
|
/// parseable for this: a block reason embeds the check's own output, and an
|
||||||
|
/// output line beginning `cap:` would read as a cap release that never happened.
|
||||||
|
///
|
||||||
|
/// It exists because the block COUNT cannot answer the question. Three blocks
|
||||||
|
/// followed by a stop that finally passed, and three blocks followed by a
|
||||||
|
/// release at the cap, both report `blocks: 3` — and they are opposite outcomes.
|
||||||
|
/// Without this, the second one completed the phase green.
|
||||||
|
pub const CAPPED_FILE: &str = "capped";
|
||||||
|
|
||||||
|
/// Where the gate lives in the guest.
|
||||||
|
///
|
||||||
|
/// Under `/root`, never under the repository. Anything written into
|
||||||
|
/// `/mission/repo` is collected and diffed, so a gate script placed there would
|
||||||
|
/// arrive in the user's delivered patch as if an agent had authored it.
|
||||||
|
pub const GATE_DIR: &str = "/root/gate";
|
||||||
|
|
||||||
|
/// What must hold before this phase's agent is allowed to stop.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct StopGate {
|
||||||
|
/// The phase must leave the repository changed. Set for coding phases that
|
||||||
|
/// have not declared `allow_empty` — the same rule
|
||||||
|
/// `empty_delivery_is_a_failure` applies post-hoc, applied while the agent
|
||||||
|
/// can still do something about it.
|
||||||
|
pub require_changes: bool,
|
||||||
|
/// `config.done_when_check`: a shell command, run in the repo, that must
|
||||||
|
/// exit 0. The deterministic half of a completion condition — a command,
|
||||||
|
/// not a judgement.
|
||||||
|
pub check: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl StopGate {
|
||||||
|
/// The gate for a phase, or `None` when there is nothing to enforce.
|
||||||
|
///
|
||||||
|
/// `None` matters: installing a hook that can never block would still cost a
|
||||||
|
/// process per stop and would put a `--settings` flag on the command line
|
||||||
|
/// for no reason.
|
||||||
|
pub fn for_phase(kind: &str, config: &serde_json::Value) -> Option<StopGate> {
|
||||||
|
let allow_empty = config.get("allow_empty").and_then(|v| v.as_bool()) == Some(true);
|
||||||
|
let check = config
|
||||||
|
.get("done_when_check")
|
||||||
|
.and_then(|v| v.as_str())
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
.map(str::to_string);
|
||||||
|
let require_changes = kind == "coding" && !allow_empty;
|
||||||
|
if !require_changes && check.is_none() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
Some(StopGate {
|
||||||
|
require_changes,
|
||||||
|
check,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The same gate, for ONE NODE of a composed run.
|
||||||
|
///
|
||||||
|
/// `require_changes` is a property of the phase, not of every node in it: a
|
||||||
|
/// graph whose second node reviews or verifies is *supposed* to leave the
|
||||||
|
/// tree alone, and a per-node gate would refuse its stop three times for
|
||||||
|
/// doing exactly its job. Dropping it loses nothing, because
|
||||||
|
/// `empty_delivery_is_a_failure` applies the same rule post-hoc to what the
|
||||||
|
/// phase as a whole delivered.
|
||||||
|
///
|
||||||
|
/// That "post-hoc" claim used to be written as covering the `check` too. It
|
||||||
|
/// did not: nothing outside this hook has ever re-run `done_when_check`, so
|
||||||
|
/// a release at [`MAX_BLOCKS`] completed the phase green with the check
|
||||||
|
/// still failing. [`CAPPED_FILE`] is what closes that.
|
||||||
|
///
|
||||||
|
/// A declared `check` DOES apply per node: it is a command the phase author
|
||||||
|
/// wrote, and every stage of the work should satisfy it.
|
||||||
|
pub fn per_node(self) -> Option<StopGate> {
|
||||||
|
self.check.map(|check| StopGate {
|
||||||
|
require_changes: false,
|
||||||
|
check: Some(check),
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The hook script, as POSIX `sh`.
|
||||||
|
///
|
||||||
|
/// `repo` and `dir` are parameters rather than the constants above so a test
|
||||||
|
/// can run this script — the real one, not a paraphrase — against a real git
|
||||||
|
/// repository in a temp directory.
|
||||||
|
pub fn script(&self, repo: &str, dir: &str) -> String {
|
||||||
|
let mut s = String::from("#!/bin/sh\n# ClawMates stop gate. Exit 2 refuses the stop.\n");
|
||||||
|
s.push_str(&format!("REPO={}\nGATE={}\nMAX={MAX_BLOCKS}\n", q(repo), q(dir)));
|
||||||
|
s.push_str("N=$(cat \"$GATE/blocks\" 2>/dev/null || echo 0)\nreason=''\n");
|
||||||
|
|
||||||
|
if self.require_changes {
|
||||||
|
// Two questions, because either alone is answerable "no" by a
|
||||||
|
// perfectly good phase: an agent that committed its work leaves a
|
||||||
|
// clean tree, and an agent that did not commit leaves HEAD where it
|
||||||
|
// was. Only both together mean nothing happened.
|
||||||
|
s.push_str(
|
||||||
|
"BASE=$(cat \"$REPO/.git/clawmates-base\" 2>/dev/null || echo '')\n\
|
||||||
|
DIRTY=$(git -C \"$REPO\" status --porcelain 2>/dev/null | head -c 400)\n\
|
||||||
|
HEAD=$(git -C \"$REPO\" rev-parse HEAD 2>/dev/null || echo '')\n\
|
||||||
|
if [ -z \"$DIRTY\" ] && [ -n \"$BASE\" ] && [ \"$HEAD\" = \"$BASE\" ]; then\n\
|
||||||
|
\x20 reason='This phase has changed nothing: the working tree is clean and \
|
||||||
|
HEAD is still the commit you started from. Do the work the task describes \
|
||||||
|
and leave it in the tree. If the task genuinely requires no code change, \
|
||||||
|
say so explicitly in your final message.'\n\
|
||||||
|
fi\n",
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
if let Some(check) = &self.check {
|
||||||
|
s.push_str(&format!(
|
||||||
|
"if [ -z \"$reason\" ]; then\n\
|
||||||
|
\x20 out=$(cd \"$REPO\" && sh -c {} 2>&1); rc=$?\n\
|
||||||
|
\x20 if [ \"$rc\" -ne 0 ]; then\n\
|
||||||
|
\x20 reason=\"This phase's completion check exited $rc, so the work is not \
|
||||||
|
done yet. The check is: {}\n\nIts output:\n$(printf '%s' \"$out\" | tail -c 1500)\"\n\
|
||||||
|
\x20 fi\n\
|
||||||
|
fi\n",
|
||||||
|
q(check),
|
||||||
|
// Inside a double-quoted assignment, so the command text itself
|
||||||
|
// must not carry a `\"` or a `$` that the shell would expand.
|
||||||
|
check.replace('\\', "\\\\").replace('"', "'").replace('$', "\\$"),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
|
||||||
|
s.push_str(
|
||||||
|
"if [ -z \"$reason\" ]; then echo pass >> \"$GATE/log\"; exit 0; fi\n\
|
||||||
|
if [ \"$N\" -ge \"$MAX\" ]; then\n\
|
||||||
|
\x20 echo \"cap: $reason\" >> \"$GATE/log\"\n\
|
||||||
|
\x20 echo 1 > \"$GATE/capped\"\n\
|
||||||
|
\x20 exit 0\n\
|
||||||
|
fi\n\
|
||||||
|
N=$((N+1)); echo \"$N\" > \"$GATE/blocks\"\n\
|
||||||
|
echo \"block $N: $reason\" >> \"$GATE/log\"\n\
|
||||||
|
printf '%s\\n' \"$reason\" >&2\n\
|
||||||
|
exit 2\n",
|
||||||
|
);
|
||||||
|
s
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One shell command that writes the gate SCRIPT into the guest.
|
||||||
|
///
|
||||||
|
/// It deliberately does NOT write `settings.json`. It used to, and it wrote
|
||||||
|
/// the whole document — so the moment a second feature needed a hook, the
|
||||||
|
/// later writer would silently erase this one. The composed document is
|
||||||
|
/// built in exactly one place: [`crate::vm_tool_tap::guest_settings`].
|
||||||
|
///
|
||||||
|
/// Written by `printf` through an exec rather than injected as part of the
|
||||||
|
/// tar: the tar lands in `/mission/repo`, which is exactly where this must
|
||||||
|
/// not be.
|
||||||
|
pub fn install_command(&self, repo: &str, dir: &str) -> String {
|
||||||
|
format!(
|
||||||
|
"mkdir -p {d} && rm -f {d}/blocks {d}/log {d}/capped \
|
||||||
|
&& printf '%s' {script} > {d}/stop-gate.sh \
|
||||||
|
&& chmod +x {d}/stop-gate.sh",
|
||||||
|
d = dir,
|
||||||
|
script = q(&self.script(repo, dir)),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Single-quote for `sh`. Same rule as `microvm_executor::shell_quote`, kept
|
||||||
|
/// local so this module has no dependency on the executor it is used by.
|
||||||
|
fn q(s: &str) -> String {
|
||||||
|
format!("'{}'", s.replace('\'', r"'\''"))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
use serde_json::json;
|
||||||
|
use std::path::Path;
|
||||||
|
use std::process::Command;
|
||||||
|
|
||||||
|
fn sh(script: &str, dir: &Path) -> std::process::Output {
|
||||||
|
let path = dir.join("stop-gate.sh");
|
||||||
|
std::fs::write(&path, script).unwrap();
|
||||||
|
Command::new("sh").arg(&path).output().expect("run the gate")
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A git repo with one commit and the clone-point marker the real checkout
|
||||||
|
/// carries (`mission_workspace::record_base_commit` writes it).
|
||||||
|
fn repo_with_base(root: &Path) -> std::path::PathBuf {
|
||||||
|
let repo = root.join("repo");
|
||||||
|
std::fs::create_dir_all(&repo).unwrap();
|
||||||
|
let git = |args: &[&str]| {
|
||||||
|
let o = Command::new("git")
|
||||||
|
.arg("-C")
|
||||||
|
.arg(&repo)
|
||||||
|
.args(args)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
assert!(o.status.success(), "git {args:?}: {:?}", o);
|
||||||
|
};
|
||||||
|
git(&["init", "--quiet"]);
|
||||||
|
git(&["config", "user.email", "t@t"]);
|
||||||
|
git(&["config", "user.name", "T"]);
|
||||||
|
std::fs::write(repo.join("README.md"), "base\n").unwrap();
|
||||||
|
git(&["add", "."]);
|
||||||
|
git(&["commit", "--quiet", "-m", "base"]);
|
||||||
|
let head = Command::new("git")
|
||||||
|
.arg("-C")
|
||||||
|
.arg(&repo)
|
||||||
|
.args(["rev-parse", "HEAD"])
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
std::fs::write(
|
||||||
|
repo.join(".git/clawmates-base"),
|
||||||
|
String::from_utf8_lossy(&head.stdout).trim(),
|
||||||
|
)
|
||||||
|
.unwrap();
|
||||||
|
repo
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The failure this exists for: an agent that stops having written nothing.
|
||||||
|
/// Post-hoc that costs a whole new VM; here it costs one sentence.
|
||||||
|
#[test]
|
||||||
|
fn an_agent_that_changed_nothing_is_not_allowed_to_stop() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let repo = repo_with_base(tmp.path());
|
||||||
|
let gate = StopGate {
|
||||||
|
require_changes: true,
|
||||||
|
check: None,
|
||||||
|
};
|
||||||
|
let script = gate.script(&repo.display().to_string(), &tmp.path().display().to_string());
|
||||||
|
|
||||||
|
let out = sh(&script, tmp.path());
|
||||||
|
assert_eq!(out.status.code(), Some(2), "the stop must be refused");
|
||||||
|
let why = String::from_utf8_lossy(&out.stderr);
|
||||||
|
assert!(why.contains("changed nothing"), "{why}");
|
||||||
|
|
||||||
|
// Uncommitted work counts — the usual case, since the agent is told to
|
||||||
|
// leave its work in the tree rather than commit it.
|
||||||
|
std::fs::write(repo.join("new.rs"), "fn done() {}\n").unwrap();
|
||||||
|
let out = sh(&script, tmp.path());
|
||||||
|
assert_eq!(out.status.code(), Some(0), "{:?}", out);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The gate gives up after [`MAX_BLOCKS`] and lets the agent stop — and it
|
||||||
|
/// must LEAVE A MARK when it does. Nothing outside this hook ever runs a
|
||||||
|
/// `done_when_check`, so a silent release completed the phase green with its
|
||||||
|
/// condition still failing.
|
||||||
|
///
|
||||||
|
/// The two files say different things and both are needed: `blocks` reaches
|
||||||
|
/// 3 in this test AND in a run where the agent got it right on the fourth
|
||||||
|
/// try, so the count alone cannot tell success from surrender.
|
||||||
|
#[test]
|
||||||
|
fn a_gate_that_gives_up_records_that_it_gave_up() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let dir = tmp.path().display().to_string();
|
||||||
|
let repo = repo_with_base(tmp.path());
|
||||||
|
let gate = StopGate {
|
||||||
|
require_changes: false,
|
||||||
|
check: Some("exit 1".into()),
|
||||||
|
};
|
||||||
|
let script = gate.script(&repo.display().to_string(), &dir);
|
||||||
|
|
||||||
|
for n in 1..=MAX_BLOCKS {
|
||||||
|
let out = sh(&script, tmp.path());
|
||||||
|
assert_eq!(out.status.code(), Some(2), "block {n} must refuse the stop");
|
||||||
|
assert!(
|
||||||
|
!tmp.path().join(CAPPED_FILE).exists(),
|
||||||
|
"the cap mark must not appear while the gate is still blocking"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// One more stop: the gate is out of blocks and must let the agent go.
|
||||||
|
let out = sh(&script, tmp.path());
|
||||||
|
assert_eq!(out.status.code(), Some(0), "at the cap the stop is allowed");
|
||||||
|
assert_eq!(
|
||||||
|
std::fs::read_to_string(tmp.path().join(CAPPED_FILE))
|
||||||
|
.unwrap()
|
||||||
|
.trim(),
|
||||||
|
"1",
|
||||||
|
"the release must be recorded, or nothing downstream can see it"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The negative control for the mark: a gate whose check PASSES releases the
|
||||||
|
/// agent too, and that release must not be recorded as a surrender. Without
|
||||||
|
/// this, "always write the file" would pass the test above and fail every
|
||||||
|
/// healthy phase in production.
|
||||||
|
#[test]
|
||||||
|
fn a_gate_that_is_satisfied_leaves_no_cap_mark() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let repo = repo_with_base(tmp.path());
|
||||||
|
let gate = StopGate {
|
||||||
|
require_changes: false,
|
||||||
|
check: Some("true".into()),
|
||||||
|
};
|
||||||
|
let script = gate.script(&repo.display().to_string(), &tmp.path().display().to_string());
|
||||||
|
|
||||||
|
let out = sh(&script, tmp.path());
|
||||||
|
assert_eq!(out.status.code(), Some(0));
|
||||||
|
assert!(
|
||||||
|
!tmp.path().join(CAPPED_FILE).exists(),
|
||||||
|
"a satisfied gate must not look like one that gave up"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// And committed work counts too. An agent that committed leaves a CLEAN
|
||||||
|
/// tree, so a gate that only looked at `git status` would refuse the stop of
|
||||||
|
/// a phase that had done everything asked of it.
|
||||||
|
#[test]
|
||||||
|
fn work_the_agent_committed_satisfies_the_gate() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let repo = repo_with_base(tmp.path());
|
||||||
|
std::fs::write(repo.join("new.rs"), "fn done() {}\n").unwrap();
|
||||||
|
for args in [vec!["add", "."], vec!["commit", "--quiet", "-m", "work"]] {
|
||||||
|
Command::new("git")
|
||||||
|
.arg("-C")
|
||||||
|
.arg(&repo)
|
||||||
|
.args(&args)
|
||||||
|
.output()
|
||||||
|
.unwrap();
|
||||||
|
}
|
||||||
|
let gate = StopGate {
|
||||||
|
require_changes: true,
|
||||||
|
check: None,
|
||||||
|
};
|
||||||
|
let out = sh(
|
||||||
|
&gate.script(&repo.display().to_string(), &tmp.path().display().to_string()),
|
||||||
|
tmp.path(),
|
||||||
|
);
|
||||||
|
assert_eq!(out.status.code(), Some(0), "{:?}", out);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The cap. Without it a stuck agent is blocked, retries, is blocked again,
|
||||||
|
/// and spends the whole hour-long turn budget instead of failing where an
|
||||||
|
/// operator can see it.
|
||||||
|
#[test]
|
||||||
|
fn the_gate_gives_up_after_the_cap_and_says_so() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let repo = repo_with_base(tmp.path());
|
||||||
|
let gate = StopGate {
|
||||||
|
require_changes: true,
|
||||||
|
check: None,
|
||||||
|
};
|
||||||
|
let script = gate.script(&repo.display().to_string(), &tmp.path().display().to_string());
|
||||||
|
|
||||||
|
for i in 1..=MAX_BLOCKS {
|
||||||
|
assert_eq!(
|
||||||
|
sh(&script, tmp.path()).status.code(),
|
||||||
|
Some(2),
|
||||||
|
"block {i} of {MAX_BLOCKS}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
sh(&script, tmp.path()).status.code(),
|
||||||
|
Some(0),
|
||||||
|
"past the cap the agent must be allowed to stop"
|
||||||
|
);
|
||||||
|
let log = std::fs::read_to_string(tmp.path().join("log")).unwrap();
|
||||||
|
assert!(log.contains("cap:"), "giving up is recorded: {log}");
|
||||||
|
assert_eq!(
|
||||||
|
std::fs::read_to_string(tmp.path().join("blocks"))
|
||||||
|
.unwrap()
|
||||||
|
.trim(),
|
||||||
|
MAX_BLOCKS.to_string(),
|
||||||
|
"and the count is exact, so the host can report it"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A phase-declared check runs in the repo, and its OUTPUT comes back — a
|
||||||
|
/// gate that said only "the check failed" would send the agent guessing.
|
||||||
|
#[test]
|
||||||
|
fn a_declared_check_must_pass_and_its_output_is_the_feedback() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let repo = repo_with_base(tmp.path());
|
||||||
|
let gate = StopGate {
|
||||||
|
require_changes: false,
|
||||||
|
check: Some("test -f wanted.txt || { echo 'wanted.txt is missing'; exit 3; }".into()),
|
||||||
|
};
|
||||||
|
let script = gate.script(&repo.display().to_string(), &tmp.path().display().to_string());
|
||||||
|
|
||||||
|
let out = sh(&script, tmp.path());
|
||||||
|
assert_eq!(out.status.code(), Some(2));
|
||||||
|
let why = String::from_utf8_lossy(&out.stderr);
|
||||||
|
assert!(why.contains("exited 3"), "{why}");
|
||||||
|
assert!(why.contains("wanted.txt is missing"), "{why}");
|
||||||
|
|
||||||
|
std::fs::write(repo.join("wanted.txt"), "here\n").unwrap();
|
||||||
|
assert_eq!(sh(&script, tmp.path()).status.code(), Some(0));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A check with quotes, `$` and apostrophes is ordinary. It travels through
|
||||||
|
/// `sh -c` inside a script that itself travels through `sh -c` to reach the
|
||||||
|
/// guest, and a quoting bug at either layer would run something else.
|
||||||
|
#[test]
|
||||||
|
fn a_check_with_shell_metacharacters_survives_both_layers() {
|
||||||
|
let tmp = tempfile::tempdir().unwrap();
|
||||||
|
let repo = repo_with_base(tmp.path());
|
||||||
|
std::fs::write(repo.join("it's here.txt"), "x\n").unwrap();
|
||||||
|
let gate = StopGate {
|
||||||
|
require_changes: false,
|
||||||
|
check: Some("test -f \"it's here.txt\" && echo $HOME > /dev/null".into()),
|
||||||
|
};
|
||||||
|
let out = sh(
|
||||||
|
&gate.script(&repo.display().to_string(), &tmp.path().display().to_string()),
|
||||||
|
tmp.path(),
|
||||||
|
);
|
||||||
|
assert_eq!(out.status.code(), Some(0), "{:?}", out);
|
||||||
|
// And the install command it is embedded in is still one shell argument.
|
||||||
|
let install = gate.install_command("/mission/repo", GATE_DIR);
|
||||||
|
assert!(install.contains("stop-gate.sh"), "{install}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Nothing the gate writes may land under the repository: `/mission/repo` is
|
||||||
|
/// collected and diffed, so a file there arrives in the user's patch as if
|
||||||
|
/// an agent had written it.
|
||||||
|
#[test]
|
||||||
|
fn the_gate_never_writes_into_the_delivered_tree() {
|
||||||
|
let gate = StopGate {
|
||||||
|
require_changes: true,
|
||||||
|
check: Some("cargo test".into()),
|
||||||
|
};
|
||||||
|
assert!(GATE_DIR.starts_with("/root/"), "{GATE_DIR}");
|
||||||
|
let install = gate.install_command("/mission/repo", GATE_DIR);
|
||||||
|
for write in ["> /mission/repo", "/mission/repo/stop", "/mission/repo/.claude"] {
|
||||||
|
assert!(!install.contains(write), "{install}");
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
crate::vm_tool_tap::guest_settings(Some(GATE_DIR), None, None)["hooks"]["Stop"][0]["hooks"]
|
||||||
|
[0]["command"],
|
||||||
|
json!("/root/gate/stop-gate.sh")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A composed run's nodes must not each be held to "this phase changed
|
||||||
|
/// something". The graph's verifier node changes nothing BY DESIGN, and a
|
||||||
|
/// per-node gate would refuse its stop until the cap — three wasted agent
|
||||||
|
/// turns for doing its job correctly.
|
||||||
|
#[test]
|
||||||
|
fn a_composed_node_is_not_held_to_the_whole_phases_delivery() {
|
||||||
|
let phase = StopGate::for_phase("coding", &json!({})).unwrap();
|
||||||
|
assert!(phase.require_changes);
|
||||||
|
assert!(
|
||||||
|
phase.per_node().is_none(),
|
||||||
|
"with nothing but the delivery rule, a node has no gate at all"
|
||||||
|
);
|
||||||
|
|
||||||
|
let with_check =
|
||||||
|
StopGate::for_phase("coding", &json!({ "done_when_check": "cargo test" })).unwrap();
|
||||||
|
let node = with_check.per_node().expect("the declared check still applies");
|
||||||
|
assert!(!node.require_changes);
|
||||||
|
assert_eq!(node.check.as_deref(), Some("cargo test"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A gate with nothing to enforce must not be installed at all — a hook that
|
||||||
|
/// can never block still costs a process per stop and a flag on the command
|
||||||
|
/// line.
|
||||||
|
#[test]
|
||||||
|
fn a_phase_with_nothing_to_enforce_gets_no_gate() {
|
||||||
|
let none = json!({});
|
||||||
|
assert!(StopGate::for_phase("research", &none).is_none());
|
||||||
|
assert!(StopGate::for_phase("coding", &json!({ "allow_empty": true })).is_none());
|
||||||
|
|
||||||
|
let coding = StopGate::for_phase("coding", &none).expect("a coding phase must deliver");
|
||||||
|
assert!(coding.require_changes);
|
||||||
|
assert!(coding.check.is_none());
|
||||||
|
|
||||||
|
// A declared check applies to any kind, including one that is allowed to
|
||||||
|
// change nothing — a verification phase's whole job is that check.
|
||||||
|
let verify = StopGate::for_phase(
|
||||||
|
"research",
|
||||||
|
&json!({ "allow_empty": true, "done_when_check": " ./verify.sh " }),
|
||||||
|
)
|
||||||
|
.expect("a declared check is a gate on its own");
|
||||||
|
assert!(!verify.require_changes);
|
||||||
|
assert_eq!(verify.check.as_deref(), Some("./verify.sh"));
|
||||||
|
}
|
||||||
|
}
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,661 @@
|
|||||||
|
//! What the agent did inside a microVM, taken from Claude Code's own hooks.
|
||||||
|
//!
|
||||||
|
//! The microVM tier had no action channel at all: a phase ran, a diff came
|
||||||
|
//! back, and everything between was invisible. The seam is the same one
|
||||||
|
//! [`crate::vm_stop_gate`] proved works in this image — `PostToolUse` fires
|
||||||
|
//! under `claude -p`, measured, not read off documentation.
|
||||||
|
//!
|
||||||
|
//! # The observer must not become a participant
|
||||||
|
//!
|
||||||
|
//! The hook `exit 0`s unconditionally. A `PostToolUse` hook that exits non-zero
|
||||||
|
//! feeds its stderr back to the model, so a tap with a bug would start
|
||||||
|
//! *instructing* the agent it exists to watch — and the resulting transcript
|
||||||
|
//! would look like a model that lost the plot rather than a broken hook.
|
||||||
|
//!
|
||||||
|
//! # Never inside the repository
|
||||||
|
//!
|
||||||
|
//! Everything lives under `/root`. `/mission/repo` is collected and diffed, so
|
||||||
|
//! a tap file written there would arrive in the user's delivered patch as
|
||||||
|
//! though an agent had authored it — the same rule, and the same reason, as the
|
||||||
|
//! stop gate's [`crate::vm_stop_gate::GATE_DIR`].
|
||||||
|
|
||||||
|
use serde_json::{json, Value};
|
||||||
|
|
||||||
|
/// Where the tap writes in the guest. Under `/root`, never the repo.
|
||||||
|
pub const TAP_DIR: &str = "/root/tap";
|
||||||
|
|
||||||
|
/// The file the hook appends to, one JSON object per line.
|
||||||
|
pub const TAP_FILE: &str = "/root/tap/tools.jsonl";
|
||||||
|
|
||||||
|
/// The single settings document the guest agent runs with.
|
||||||
|
///
|
||||||
|
/// One path, because there is only ever one writer — see [`guest_settings`].
|
||||||
|
pub const SETTINGS_PATH: &str = "/root/guest-settings.json";
|
||||||
|
|
||||||
|
/// Read the tap out of the guest, before collection destroys the VM.
|
||||||
|
///
|
||||||
|
/// `|| true` so a phase whose agent called no tools — or where the hook never
|
||||||
|
/// fired — reads as empty rather than as a failed probe. The difference between
|
||||||
|
/// those two is the histogram in the log, not an error here.
|
||||||
|
pub const DRAIN_PROBE: &str = "cat /root/tap/tools.jsonl 2>/dev/null || true";
|
||||||
|
|
||||||
|
/// Read the tap from line `from` onward, so a repeated drain returns only what
|
||||||
|
/// is new.
|
||||||
|
///
|
||||||
|
/// A cursor rather than a re-read: the live drain runs every few seconds
|
||||||
|
/// against a file the agent is still appending to, and re-sending the whole
|
||||||
|
/// file each pass would record every tool call once per poll — a phase would
|
||||||
|
/// finish with its early files weighted by how long it ran.
|
||||||
|
///
|
||||||
|
/// `tail -n +N` is 1-based on the FIRST line to print, so `from` is a line
|
||||||
|
/// count already consumed and the probe asks for `from + 1`.
|
||||||
|
pub fn drain_from(from: usize) -> String {
|
||||||
|
format!("tail -n +{} {TAP_FILE} 2>/dev/null || true", from + 1)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How far a drain advanced the cursor — the number of LINES it consumed.
|
||||||
|
///
|
||||||
|
/// Counts every line, including blank ones, and that is the whole point. The
|
||||||
|
/// hook appends the event and then a newline of its own, so the tap is
|
||||||
|
/// `{json}\n\n{json}\n\n…` and `parse` skips the blanks. Advancing the cursor
|
||||||
|
/// by the number of PARSED events instead would leave it short by one line per
|
||||||
|
/// event, and `tail -n +N` would hand back events already recorded — every one
|
||||||
|
/// of them written again on the next poll, with nothing anywhere reporting it.
|
||||||
|
pub fn consumed_lines(raw: &str) -> usize {
|
||||||
|
raw.lines().count()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One observed tool call.
|
||||||
|
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||||
|
pub struct Observed {
|
||||||
|
pub tool: String,
|
||||||
|
/// The path the tool's **input** named, if any. From JSON, never prose.
|
||||||
|
pub path: Option<String>,
|
||||||
|
/// Claude Code's session id for the `claude -p` invocation this call
|
||||||
|
/// happened inside.
|
||||||
|
///
|
||||||
|
/// One invocation is one turn is one agent, so this is the only thing in
|
||||||
|
/// the payload that separates one agent's actions from another's. The tap
|
||||||
|
/// is per-CONTAINER and every role in a phase shares one, so without this
|
||||||
|
/// the whole phase arrives as an undifferentiated stream.
|
||||||
|
pub session: Option<String>,
|
||||||
|
/// What a **command** produced, bounded by [`bounded_response`].
|
||||||
|
///
|
||||||
|
/// Only for tools that run something. `Read`'s response is the file it just
|
||||||
|
/// read and `Write`'s is a restatement of what was written — both are
|
||||||
|
/// already knowable from the arguments and the delivered diff, and storing
|
||||||
|
/// them would double the largest write path in the system for nothing.
|
||||||
|
///
|
||||||
|
/// A command's OUTCOME is different: it is the only place a failing test
|
||||||
|
/// run is visible. Without it "did this phase go red before it went green"
|
||||||
|
/// cannot be answered from anything — not from tool order (in Rust the
|
||||||
|
/// unit test lives in the file under test, so one `Edit` adds both), and
|
||||||
|
/// not from the repository either, because `tdd-red-green-refactor` says
|
||||||
|
/// in so many words to "commit the RED-to-GREEN pair as one commit".
|
||||||
|
pub response: Value,
|
||||||
|
/// The subagent that made this call, when it was not the turn's own agent.
|
||||||
|
///
|
||||||
|
/// Claude Code's `Agent` tool spawns a subagent that runs its own tools,
|
||||||
|
/// and those calls DO reach this hook — measured against the real binary,
|
||||||
|
/// which is the good news, because it means nothing is invisible. What they
|
||||||
|
/// carry is the PARENT's `session_id`, so [`Observed::session`] cannot tell
|
||||||
|
/// them apart and attribution silently credits the parent for work a
|
||||||
|
/// subagent did.
|
||||||
|
///
|
||||||
|
/// The payload has always said so: `agent_type` and `agent_id` are present
|
||||||
|
/// on a subagent's call and absent on the parent's. This parser read past
|
||||||
|
/// them. Two production missions spawned twelve subagents to fetch web
|
||||||
|
/// pages, and every tool call they made was recorded as the parent's with
|
||||||
|
/// nothing anywhere reporting the difference.
|
||||||
|
pub subagent: Option<String>,
|
||||||
|
/// Which subagent instance, so several running under one turn stay apart.
|
||||||
|
pub subagent_id: Option<String>,
|
||||||
|
/// The tool's arguments, bounded by [`bounded_input`].
|
||||||
|
///
|
||||||
|
/// Kept because the tool NAME alone answers almost nothing. A phase that
|
||||||
|
/// recorded `Bash × 6` is indistinguishable from one that ran the test
|
||||||
|
/// suite six times, one that pushed to a branch it was told not to, and
|
||||||
|
/// one that queried an API a skill forbids. The argument is where the
|
||||||
|
/// behaviour is, and until now this parser read it, took the path out of
|
||||||
|
/// it, and dropped the rest on the floor.
|
||||||
|
pub input: Value,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// How much of one argument string is worth keeping.
|
||||||
|
///
|
||||||
|
/// A shell command longer than this is a heredoc or a generated payload; its
|
||||||
|
/// first half still carries the verb, which is what any check reads.
|
||||||
|
const MAX_ARG_LEN: usize = 512;
|
||||||
|
|
||||||
|
/// Argument keys whose value is a file BODY rather than a description of an
|
||||||
|
/// action.
|
||||||
|
///
|
||||||
|
/// Dropped to a byte count rather than truncated. These carry whole source
|
||||||
|
/// files — `mission_events` is already the largest write path on a coding
|
||||||
|
/// phase, and storing every `Write` twice (once in the event, once in the
|
||||||
|
/// delivered diff) buys nothing: no check reads the body, and the diff is the
|
||||||
|
/// authority on what was written anyway.
|
||||||
|
const BODY_KEYS: [&str; 4] = ["content", "new_string", "old_string", "edits"];
|
||||||
|
|
||||||
|
/// Tools whose response is an outcome rather than a restatement.
|
||||||
|
const RESPONSE_TOOLS: [&str; 1] = ["Bash"];
|
||||||
|
|
||||||
|
/// How much of a command's output to keep.
|
||||||
|
const MAX_OUTPUT_LEN: usize = 600;
|
||||||
|
|
||||||
|
/// Shrink a command's response, keeping the **end** of its output.
|
||||||
|
///
|
||||||
|
/// The opposite of [`bounded_input`], and deliberately so. An argument's
|
||||||
|
/// meaning is at the start — the verb of the command. A command's meaning is at
|
||||||
|
/// the END: `cargo test` prints hundreds of lines and then `test result: ok` or
|
||||||
|
/// `test result: FAILED`, and a head-biased truncation would keep the noise and
|
||||||
|
/// throw away the verdict, which is the one thing being stored for.
|
||||||
|
pub fn bounded_response(tool: &str, response: &Value) -> Value {
|
||||||
|
if !RESPONSE_TOOLS.contains(&tool) {
|
||||||
|
return Value::Null;
|
||||||
|
}
|
||||||
|
let Some(obj) = response.as_object() else {
|
||||||
|
return Value::Null;
|
||||||
|
};
|
||||||
|
let mut out = serde_json::Map::new();
|
||||||
|
for key in ["stdout", "stderr", "interrupted"] {
|
||||||
|
match obj.get(key) {
|
||||||
|
Some(Value::String(s)) if s.len() > MAX_OUTPUT_LEN => {
|
||||||
|
let start = s
|
||||||
|
.char_indices()
|
||||||
|
.map(|(i, _)| i)
|
||||||
|
.find(|i| *i >= s.len().saturating_sub(MAX_OUTPUT_LEN))
|
||||||
|
.unwrap_or(0);
|
||||||
|
out.insert(key.into(), Value::String(format!("[truncated]…{}", &s[start..])));
|
||||||
|
}
|
||||||
|
Some(v) => {
|
||||||
|
out.insert(key.into(), v.clone());
|
||||||
|
}
|
||||||
|
None => {}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Value::Object(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Shrink a tool's arguments to something safe to store on every call.
|
||||||
|
///
|
||||||
|
/// Bounded rather than whitelisted on purpose. A whitelist of "interesting"
|
||||||
|
/// keys silently drops the one argument that matters the first time a tool
|
||||||
|
/// grows a new field, and the loss is invisible — the event still looks
|
||||||
|
/// complete. Bounding keeps every key and says, in the record itself, where it
|
||||||
|
/// stopped.
|
||||||
|
pub fn bounded_input(input: &Value) -> Value {
|
||||||
|
let Some(obj) = input.as_object() else {
|
||||||
|
return Value::Null;
|
||||||
|
};
|
||||||
|
let mut out = serde_json::Map::new();
|
||||||
|
for (k, v) in obj {
|
||||||
|
if BODY_KEYS.contains(&k.as_str()) {
|
||||||
|
let bytes = match v {
|
||||||
|
Value::String(s) => s.len(),
|
||||||
|
other => other.to_string().len(),
|
||||||
|
};
|
||||||
|
out.insert(k.clone(), json!({ "omitted_bytes": bytes }));
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
match v {
|
||||||
|
Value::String(s) if s.len() > MAX_ARG_LEN => {
|
||||||
|
let cut = s
|
||||||
|
.char_indices()
|
||||||
|
.map(|(i, _)| i)
|
||||||
|
.take_while(|i| *i <= MAX_ARG_LEN)
|
||||||
|
.last()
|
||||||
|
.unwrap_or(0);
|
||||||
|
out.insert(k.clone(), Value::String(format!("{}…[truncated]", &s[..cut])));
|
||||||
|
}
|
||||||
|
other => {
|
||||||
|
out.insert(k.clone(), other.clone());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Value::Object(out)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The hook script. Copies stdin verbatim to the tap file and gets out of the
|
||||||
|
/// way.
|
||||||
|
///
|
||||||
|
/// The parsing happens host-side, on purpose: a `jq` or `sed` pipeline in the
|
||||||
|
/// guest would need the tool's JSON schema baked into a shell script, inside an
|
||||||
|
/// image we do not rebuild for a parser change, with no way to tell a parse
|
||||||
|
/// failure from a quiet turn.
|
||||||
|
pub fn hook_script(dir: &str) -> String {
|
||||||
|
format!(
|
||||||
|
"#!/bin/sh\n\
|
||||||
|
# The tool tap. See cm-api/src/vm_tool_tap.rs.\n\
|
||||||
|
mkdir -p {dir} 2>/dev/null\n\
|
||||||
|
# `cat` of stdin, appended whole. One JSON object per line, because\n\
|
||||||
|
# Claude Code hands the hook one event per invocation.\n\
|
||||||
|
cat >> {dir}/tools.jsonl 2>/dev/null\n\
|
||||||
|
printf '\\n' >> {dir}/tools.jsonl 2>/dev/null\n\
|
||||||
|
# ALWAYS zero. A non-zero PostToolUse hook talks back to the model.\n\
|
||||||
|
exit 0\n"
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The settings document for the guest, carrying **every** hook at once.
|
||||||
|
///
|
||||||
|
/// This function exists because the alternative — each feature writing its own
|
||||||
|
/// `settings.json` — is a silent clobber. The stop gate wrote the whole
|
||||||
|
/// document; a tap that did the same would erase the gate, and a coding phase
|
||||||
|
/// would then complete having written nothing, which is the exact failure the
|
||||||
|
/// gate exists to catch. One writer, one document, one test that both hooks
|
||||||
|
/// survive it.
|
||||||
|
///
|
||||||
|
/// `None` for any part means that hook is simply absent.
|
||||||
|
pub fn guest_settings(
|
||||||
|
gate_dir: Option<&str>,
|
||||||
|
tap_dir: Option<&str>,
|
||||||
|
tool_gate_dir: Option<&str>,
|
||||||
|
) -> Value {
|
||||||
|
let mut hooks = serde_json::Map::new();
|
||||||
|
if let Some(dir) = gate_dir {
|
||||||
|
hooks.insert(
|
||||||
|
"Stop".into(),
|
||||||
|
json!([{ "hooks": [{ "type": "command", "command": format!("{dir}/stop-gate.sh") }] }]),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if let Some(dir) = tap_dir {
|
||||||
|
hooks.insert(
|
||||||
|
"PostToolUse".into(),
|
||||||
|
json!([{ "hooks": [{ "type": "command", "command": format!("{dir}/tap.sh") }] }]),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
if let Some(dir) = tool_gate_dir {
|
||||||
|
// PRE-execution, unlike the tap above. Composed here rather than
|
||||||
|
// written by `vm_tool_gate` itself for the same reason everything else
|
||||||
|
// is: one writer, one document.
|
||||||
|
hooks.insert("PreToolUse".into(), crate::vm_tool_gate::settings_hook(dir));
|
||||||
|
}
|
||||||
|
json!({ "hooks": Value::Object(hooks) })
|
||||||
|
}
|
||||||
|
|
||||||
|
/// One shell command that installs the tap.
|
||||||
|
///
|
||||||
|
/// Written by `printf` through an exec rather than injected with the workspace
|
||||||
|
/// tar: the tar lands in `/mission/repo`, which is exactly where this must not.
|
||||||
|
pub fn install_command(dir: &str) -> String {
|
||||||
|
format!(
|
||||||
|
"mkdir -p {dir} && rm -f {dir}/tools.jsonl \
|
||||||
|
&& printf '%s' {script} > {dir}/tap.sh && chmod +x {dir}/tap.sh",
|
||||||
|
dir = dir,
|
||||||
|
script = shell_quote(&hook_script(dir)),
|
||||||
|
)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write the composed settings document.
|
||||||
|
pub fn settings_command(path: &str, settings: &Value) -> String {
|
||||||
|
format!("printf '%s' {} > {path}", shell_quote(&settings.to_string()))
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Parse a drained tap.
|
||||||
|
///
|
||||||
|
/// Tolerant by construction: the file is appended to by a shell hook in a VM
|
||||||
|
/// that may be killed mid-write, so a truncated last line is expected and is
|
||||||
|
/// skipped rather than failing the whole drain. Losing the last tool call of a
|
||||||
|
/// phase costs one orb; losing all of them because of it would cost the tier.
|
||||||
|
pub fn parse(raw: &str) -> Vec<Observed> {
|
||||||
|
raw.lines()
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|l| !l.is_empty())
|
||||||
|
.filter_map(|line| {
|
||||||
|
let v: Value = serde_json::from_str(line).ok()?;
|
||||||
|
// Only tool events. The same hook file would carry others if the
|
||||||
|
// settings ever install one, and a `hook_event_name` we do not
|
||||||
|
// recognise must not be read as a tool named "".
|
||||||
|
let tool = v
|
||||||
|
.get("tool_name")
|
||||||
|
.or_else(|| v.get("toolName"))
|
||||||
|
.and_then(Value::as_str)?
|
||||||
|
.trim()
|
||||||
|
.to_string();
|
||||||
|
if tool.is_empty() {
|
||||||
|
return None;
|
||||||
|
}
|
||||||
|
let input = v
|
||||||
|
.get("tool_input")
|
||||||
|
.or_else(|| v.get("toolInput"))
|
||||||
|
.cloned()
|
||||||
|
.unwrap_or(Value::Null);
|
||||||
|
let response = v
|
||||||
|
.get("tool_response")
|
||||||
|
.or_else(|| v.get("toolResponse"))
|
||||||
|
.cloned()
|
||||||
|
.unwrap_or(Value::Null);
|
||||||
|
Some(Observed {
|
||||||
|
path: crate::mission_events::tool_path(&input),
|
||||||
|
input: bounded_input(&input),
|
||||||
|
response: bounded_response(&tool, &response),
|
||||||
|
session: v
|
||||||
|
.get("session_id")
|
||||||
|
.or_else(|| v.get("sessionId"))
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.map(str::to_string),
|
||||||
|
// Absent on the turn agent's own calls, present on a
|
||||||
|
// subagent's. That absence IS the signal, so an empty string
|
||||||
|
// must read as "not a subagent" rather than as one named "".
|
||||||
|
subagent: non_empty(&v, "agent_type", "agentType"),
|
||||||
|
subagent_id: non_empty(&v, "agent_id", "agentId"),
|
||||||
|
tool,
|
||||||
|
})
|
||||||
|
})
|
||||||
|
.collect()
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A string field under either spelling, treating empty as missing.
|
||||||
|
fn non_empty(v: &Value, snake: &str, camel: &str) -> Option<String> {
|
||||||
|
v.get(snake)
|
||||||
|
.or_else(|| v.get(camel))
|
||||||
|
.and_then(Value::as_str)
|
||||||
|
.map(str::trim)
|
||||||
|
.filter(|s| !s.is_empty())
|
||||||
|
.map(str::to_string)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Single-quote for `sh`. Local copy, same rule as the stop gate's — these two
|
||||||
|
/// modules deliberately share no code, so neither can break the other.
|
||||||
|
pub fn shell_quote(s: &str) -> String {
|
||||||
|
format!("'{}'", s.replace('\'', r"'\''"))
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// The gate and the tap must BOTH survive one settings document.
|
||||||
|
///
|
||||||
|
/// This is the whole reason `guest_settings` exists. Two writers each
|
||||||
|
/// producing a whole `settings.json` is not a merge conflict — the second
|
||||||
|
/// simply wins, no error, and the loser's hook never runs. When the loser is
|
||||||
|
/// the stop gate, a coding phase completes having written nothing: the exact
|
||||||
|
/// failure the gate was built to catch.
|
||||||
|
#[test]
|
||||||
|
fn both_hooks_survive_one_settings_document() {
|
||||||
|
let s = guest_settings(Some("/root/gate"), Some(TAP_DIR), None);
|
||||||
|
let hooks = s.get("hooks").expect("hooks");
|
||||||
|
assert_eq!(
|
||||||
|
hooks["Stop"][0]["hooks"][0]["command"],
|
||||||
|
json!("/root/gate/stop-gate.sh"),
|
||||||
|
"the stop gate must survive the tap being installed"
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
hooks["PostToolUse"][0]["hooks"][0]["command"],
|
||||||
|
json!("/root/tap/tap.sh")
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Either half absent leaves the other exactly as it was.
|
||||||
|
#[test]
|
||||||
|
fn one_hook_alone_is_a_valid_document() {
|
||||||
|
let gate_only = guest_settings(Some("/root/gate"), None, None);
|
||||||
|
assert!(gate_only["hooks"].get("Stop").is_some());
|
||||||
|
assert!(gate_only["hooks"].get("PostToolUse").is_none());
|
||||||
|
|
||||||
|
let tap_only = guest_settings(None, Some(TAP_DIR), None);
|
||||||
|
assert!(tap_only["hooks"].get("Stop").is_none());
|
||||||
|
assert!(tap_only["hooks"].get("PostToolUse").is_some());
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The observer must never talk back to the model.
|
||||||
|
#[test]
|
||||||
|
fn the_hook_always_exits_zero() {
|
||||||
|
let s = hook_script(TAP_DIR);
|
||||||
|
assert!(s.contains("exit 0"));
|
||||||
|
// No conditional exits at all: a `PostToolUse` hook that exits non-zero
|
||||||
|
// feeds stderr back to the agent, so the tap would become an instruction.
|
||||||
|
assert!(!s.contains("exit 1") && !s.contains("exit 2"), "{s}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Nothing the tap writes may land in the delivered tree.
|
||||||
|
#[test]
|
||||||
|
fn the_tap_never_writes_into_the_repository() {
|
||||||
|
assert!(TAP_DIR.starts_with("/root/"));
|
||||||
|
assert!(TAP_FILE.starts_with("/root/"));
|
||||||
|
assert!(SETTINGS_PATH.starts_with("/root/"));
|
||||||
|
let cmd = install_command(TAP_DIR);
|
||||||
|
assert!(!cmd.contains("/mission/repo"), "{cmd}");
|
||||||
|
assert!(!hook_script(TAP_DIR).contains("/mission/repo"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A real `PostToolUse` payload gives up its tool and its path — and a
|
||||||
|
/// truncated final line does not take the rest of the phase with it.
|
||||||
|
#[test]
|
||||||
|
fn a_drained_tap_parses_and_tolerates_a_torn_last_line() {
|
||||||
|
let raw = concat!(
|
||||||
|
r#"{"hook_event_name":"PostToolUse","tool_name":"Edit","#,
|
||||||
|
r#""tool_input":{"file_path":"/mission/repo/src/a.rs"}}"#,
|
||||||
|
"\n",
|
||||||
|
r#"{"hook_event_name":"PostToolUse","tool_name":"Bash","tool_input":{"command":"ls"}}"#,
|
||||||
|
"\n",
|
||||||
|
"\n",
|
||||||
|
// The VM was destroyed mid-write.
|
||||||
|
r#"{"hook_event_name":"PostToolUse","tool_name":"Wri"#,
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
parse(raw),
|
||||||
|
vec![
|
||||||
|
Observed {
|
||||||
|
tool: "Edit".into(),
|
||||||
|
path: Some("/mission/repo/src/a.rs".into()),
|
||||||
|
input: json!({"file_path": "/mission/repo/src/a.rs"}),
|
||||||
|
session: None,
|
||||||
|
subagent: None,
|
||||||
|
subagent_id: None,
|
||||||
|
response: Value::Null,
|
||||||
|
},
|
||||||
|
Observed {
|
||||||
|
tool: "Bash".into(),
|
||||||
|
path: None,
|
||||||
|
input: json!({"command": "ls"}),
|
||||||
|
session: None,
|
||||||
|
subagent: None,
|
||||||
|
subagent_id: None,
|
||||||
|
response: Value::Null,
|
||||||
|
},
|
||||||
|
]
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A subagent's tool calls reach this hook carrying the PARENT's session
|
||||||
|
/// id, so the only thing that separates them is `agent_type`/`agent_id`.
|
||||||
|
///
|
||||||
|
/// Measured against claude 2.1.246: spawning one subagent and having it run
|
||||||
|
/// `echo SUB` produced three `PostToolUse` events on one session id — the
|
||||||
|
/// parent's `Agent` call, the parent's own `Bash`, and the subagent's
|
||||||
|
/// `Bash` — and only the last carried an `agent_type`. Reading past those
|
||||||
|
/// fields is what made twelve production subagent spawns indistinguishable
|
||||||
|
/// from the work of the agents that spawned them.
|
||||||
|
#[test]
|
||||||
|
fn a_subagents_call_is_told_apart_from_its_parents() {
|
||||||
|
let raw = concat!(
|
||||||
|
r#"{"tool_name":"Agent","session_id":"s1","tool_input":{"prompt":"fetch it"}}"#,
|
||||||
|
"\n",
|
||||||
|
r#"{"tool_name":"Bash","session_id":"s1","tool_input":{"command":"echo PARENT"}}"#,
|
||||||
|
"\n",
|
||||||
|
r#"{"tool_name":"Bash","session_id":"s1","agent_id":"a0b8","agent_type":"general-purpose","#,
|
||||||
|
r#""tool_input":{"command":"echo SUB"}}"#,
|
||||||
|
);
|
||||||
|
let got = parse(raw);
|
||||||
|
assert_eq!(got.len(), 3);
|
||||||
|
assert!(
|
||||||
|
got.iter().all(|o| o.session.as_deref() == Some("s1")),
|
||||||
|
"a subagent shares its parent's session id — that is the whole problem"
|
||||||
|
);
|
||||||
|
assert_eq!(got[0].subagent, None, "the parent spawned it; it did not run inside it");
|
||||||
|
assert_eq!(got[1].subagent, None);
|
||||||
|
assert_eq!(got[2].subagent.as_deref(), Some("general-purpose"));
|
||||||
|
assert_eq!(got[2].subagent_id.as_deref(), Some("a0b8"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An empty `agent_type` must read as "the turn's own agent", not as a
|
||||||
|
/// subagent whose name happens to be blank.
|
||||||
|
#[test]
|
||||||
|
fn a_blank_agent_type_is_not_a_subagent() {
|
||||||
|
let raw = r#"{"tool_name":"Bash","session_id":"s1","agent_type":" ","tool_input":{"command":"ls"}}"#;
|
||||||
|
assert_eq!(parse(raw)[0].subagent, None);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The command survives the parse.
|
||||||
|
///
|
||||||
|
/// The regression this guards is the one that made the first container-tier
|
||||||
|
/// measurement unusable: six `Bash` calls were recorded and not one of them
|
||||||
|
/// said what it ran, so every behavioural question — did it run the tests,
|
||||||
|
/// did it commit, did it call the API a skill forbids — was unanswerable
|
||||||
|
/// from a record that looked complete.
|
||||||
|
#[test]
|
||||||
|
fn the_argument_is_what_carries_the_behaviour() {
|
||||||
|
let raw = concat!(
|
||||||
|
r#"{"tool_name":"Bash","tool_input":{"command":"cargo nextest run -p cm-api"}}"#,
|
||||||
|
"\n",
|
||||||
|
);
|
||||||
|
let got = parse(raw);
|
||||||
|
assert_eq!(got[0].input["command"], json!("cargo nextest run -p cm-api"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// A file body is counted, not stored; everything else survives bounded.
|
||||||
|
#[test]
|
||||||
|
fn bodies_are_dropped_and_long_arguments_are_marked() {
|
||||||
|
let long = "x".repeat(MAX_ARG_LEN + 50);
|
||||||
|
let got = bounded_input(&json!({
|
||||||
|
"file_path": "/mission/repo/src/a.rs",
|
||||||
|
"content": "fn main() {}",
|
||||||
|
"command": long,
|
||||||
|
}));
|
||||||
|
assert_eq!(got["file_path"], json!("/mission/repo/src/a.rs"));
|
||||||
|
assert_eq!(
|
||||||
|
got["content"],
|
||||||
|
json!({"omitted_bytes": 12}),
|
||||||
|
"a file body is stored in the delivered diff already; the event only \
|
||||||
|
needs to say how big it was"
|
||||||
|
);
|
||||||
|
let cmd = got["command"].as_str().expect("command kept");
|
||||||
|
assert!(cmd.ends_with("…[truncated]"), "{cmd}");
|
||||||
|
assert!(
|
||||||
|
cmd.len() < MAX_ARG_LEN + 40,
|
||||||
|
"a bounded argument must actually be bounded: {}",
|
||||||
|
cmd.len()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Truncation must not split a multi-byte character.
|
||||||
|
///
|
||||||
|
/// `&s[..cut]` on a byte index inside a UTF-8 sequence panics, and the
|
||||||
|
/// panic would land in the drain — losing a whole phase's tap to a command
|
||||||
|
/// that happened to contain an emoji or an em dash.
|
||||||
|
#[test]
|
||||||
|
fn truncation_respects_character_boundaries() {
|
||||||
|
let long = "é".repeat(MAX_ARG_LEN);
|
||||||
|
let got = bounded_input(&json!({ "command": long }));
|
||||||
|
assert!(got["command"].as_str().unwrap().ends_with("…[truncated]"));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Exactly one place in the tree writes the guest settings document.
|
||||||
|
///
|
||||||
|
/// The unit test above proves `guest_settings` composes correctly; it says
|
||||||
|
/// nothing about whether anyone bypasses it. A second `> …settings.json`
|
||||||
|
/// anywhere is the silent clobber itself, and it would pass every other
|
||||||
|
/// test in this file.
|
||||||
|
#[test]
|
||||||
|
fn nothing_else_writes_the_guest_settings() {
|
||||||
|
for (name, src) in [
|
||||||
|
("vm_stop_gate.rs", include_str!("vm_stop_gate.rs")),
|
||||||
|
("microvm_executor.rs", include_str!("microvm_executor.rs")),
|
||||||
|
] {
|
||||||
|
assert!(
|
||||||
|
!src.contains("> {d}/settings.json") && !src.contains("settings.json\","),
|
||||||
|
"{name} writes a settings document of its own; compose it through \
|
||||||
|
vm_tool_tap::guest_settings instead"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
// And the one legitimate writer is this module's own helper.
|
||||||
|
let exec = include_str!("microvm_executor.rs");
|
||||||
|
assert_eq!(
|
||||||
|
exec.matches("vm_tool_tap::settings_command").count(),
|
||||||
|
1,
|
||||||
|
"the settings document must be written exactly once per turn"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The cursor must not re-read what it already returned.
|
||||||
|
///
|
||||||
|
/// Off by one here is not a crash, it is a double-count: `tail -n +1` and
|
||||||
|
/// `tail -n +2` both return output, and the wrong one quietly records every
|
||||||
|
/// early tool call once per poll.
|
||||||
|
#[test]
|
||||||
|
fn the_drain_cursor_asks_for_what_it_has_not_seen() {
|
||||||
|
assert!(drain_from(0).contains("tail -n +1 "));
|
||||||
|
assert!(drain_from(3).contains("tail -n +4 "));
|
||||||
|
assert!(drain_from(0).contains(TAP_FILE));
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The cursor counts LINES, not events.
|
||||||
|
///
|
||||||
|
/// The hook writes the event and then a newline of its own, so a two-event
|
||||||
|
/// tap is four lines. Advancing by parsed-event count would leave the
|
||||||
|
/// cursor two lines short, `tail` would return both events again, and the
|
||||||
|
/// live drain would re-record everything it had already recorded — growing
|
||||||
|
/// worse the longer the turn ran, and silent throughout.
|
||||||
|
#[test]
|
||||||
|
fn the_cursor_counts_lines_not_events() {
|
||||||
|
let raw = concat!(
|
||||||
|
r#"{"tool_name":"Edit","tool_input":{"file_path":"a.rs"}}"#,
|
||||||
|
"\n\n",
|
||||||
|
r#"{"tool_name":"Bash","tool_input":{"command":"ls"}}"#,
|
||||||
|
"\n\n",
|
||||||
|
);
|
||||||
|
assert_eq!(parse(raw).len(), 2, "two events");
|
||||||
|
assert_eq!(consumed_lines(raw), 4, "…written across four lines");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An event that is not a tool call is not a tool named "".
|
||||||
|
#[test]
|
||||||
|
fn a_non_tool_event_is_skipped() {
|
||||||
|
assert!(parse(r#"{"hook_event_name":"SessionStart","session_id":"x"}"#).is_empty());
|
||||||
|
assert!(parse(r#"{"tool_name":" ","tool_input":{}}"#).is_empty());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod three_hook_tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
/// All three hooks must survive one document.
|
||||||
|
///
|
||||||
|
/// The tap and the stop gate already shared it; the pre-execution gate is
|
||||||
|
/// the third, and the clobber this function exists to prevent gets more
|
||||||
|
/// likely with each one. A missing `Stop` lets a phase finish having
|
||||||
|
/// written nothing; a missing `PreToolUse` runs every command unchecked.
|
||||||
|
#[test]
|
||||||
|
fn the_document_carries_the_stop_gate_the_tap_and_the_pre_execution_gate() {
|
||||||
|
let s = guest_settings(
|
||||||
|
Some("/root/gate"),
|
||||||
|
Some(TAP_DIR),
|
||||||
|
Some(crate::vm_tool_gate::GUEST_DIR),
|
||||||
|
);
|
||||||
|
let hooks = s["hooks"].as_object().expect("hooks object");
|
||||||
|
assert!(hooks.contains_key("Stop"), "stop gate lost");
|
||||||
|
assert!(hooks.contains_key("PostToolUse"), "tap lost");
|
||||||
|
assert!(hooks.contains_key("PreToolUse"), "pre-execution gate lost");
|
||||||
|
assert_eq!(hooks.len(), 3, "an unexpected hook appeared: {hooks:?}");
|
||||||
|
|
||||||
|
// And the pre-execution hook points at the gate's own script, not the
|
||||||
|
// tap's — pointing PreToolUse at tap.sh would exit 0 on everything and
|
||||||
|
// read as a gate that allows all.
|
||||||
|
let cmd = s["hooks"]["PreToolUse"][0]["hooks"][0]["command"]
|
||||||
|
.as_str()
|
||||||
|
.expect("command");
|
||||||
|
assert!(cmd.ends_with("tool-gate.sh"), "wrong script: {cmd}");
|
||||||
|
}
|
||||||
|
|
||||||
|
/// The gate alone must still produce a usable document.
|
||||||
|
#[test]
|
||||||
|
fn the_gate_can_be_installed_without_the_others() {
|
||||||
|
let s = guest_settings(None, None, Some(crate::vm_tool_gate::GUEST_DIR));
|
||||||
|
assert_eq!(s["hooks"].as_object().unwrap().len(), 1);
|
||||||
|
assert!(s["hooks"]["PreToolUse"].is_array());
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -34,6 +34,20 @@ pub struct WorkflowRecipe {
|
|||||||
pub phases: Vec<WorkflowPhase>,
|
pub phases: Vec<WorkflowPhase>,
|
||||||
#[serde(default)]
|
#[serde(default)]
|
||||||
pub default_team_template: Option<String>,
|
pub default_team_template: Option<String>,
|
||||||
|
/// Default team **per phase purpose**, by template key:
|
||||||
|
/// `{ research = "topic_research", coding = "rust_sdlc" }`.
|
||||||
|
///
|
||||||
|
/// `default_team_template` names ONE team for a whole mission, and a
|
||||||
|
/// multi-phase recipe does not have one job. `research_and_code` staffs a
|
||||||
|
/// research phase and a coding phase from the same `rust_sdlc` crew, which
|
||||||
|
/// is why its research phase has to spend a paragraph of `task` telling
|
||||||
|
/// coders not to code — a workaround for staffing, written into the prompt.
|
||||||
|
///
|
||||||
|
/// Resolved to `config.phase_teams` at mission-create, which the
|
||||||
|
/// orchestrator and `composed_graph` already read. Purposes come from
|
||||||
|
/// `phase_runner::purposes_for`.
|
||||||
|
#[serde(default)]
|
||||||
|
pub default_phase_teams: std::collections::BTreeMap<String, String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone, Deserialize, Serialize)]
|
#[derive(Debug, Clone, Deserialize, Serialize)]
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user