From c1df958a1322d2d1d09a31da6e10b25fb37fa9a6 Mon Sep 17 00:00:00 2001 From: Omar Sobh Date: Tue, 22 Sep 2026 10:44:15 -0500 Subject: [PATCH] harness(continuous-research): a guard for the pipeline that had none MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The recipe with the most moving parts was the only one with no standing check, and it spent a month delivering its digest onto branches nobody merged. Six assertions, ordered by what they would have caught: 1. analysis.md is on the vault's DEFAULT branch — the defect itself 2. harvest.jsonl carries the triage fields on every line 3. evidence scores span more than patterns::SATURATED_BELOW — the saturation guard, live on a real harvest rather than a fixture 4. analysis.md names every harvested paper id 5. episode.json has a title and highlights within 10-70 chars — the recipe's own done_when, checked independently of the judge 6. audio rendered, distinguishing three outcomes a single boolean would have collapsed: no row at all, an `unrenderable` tombstone, and a real episode with a non-zero duration CLAWMATES_SKIP_AUDIO=1 skips 6, because it spends ElevenLabs credits and a run about the merge should not have to. Both blobs go to disk before comparison rather than through nested quoting — the rolepolicy assertion already reported an empty record for a correct one that way. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_01WZb5A2kfVfjpdwSochkuHz --- scripts/verify-mission-delivery.sh | 158 ++++++++++++++++++++++++++++- 1 file changed, 157 insertions(+), 1 deletion(-) diff --git a/scripts/verify-mission-delivery.sh b/scripts/verify-mission-delivery.sh index f0ebef2..8a3cc26 100755 --- a/scripts/verify-mission-delivery.sh +++ b/scripts/verify-mission-delivery.sh @@ -64,6 +64,10 @@ NODE_KERNEL=$(ssh "${FLEET_NODE:-osobh@tank}" 'uname -r' 2>/dev/null | tr -d '[: # with CLAWMATES_REPO_ID, or find it: select id from repos where name = # 'clawmates-delivery-scratch'. REPO_ID="${CLAWMATES_REPO_ID:-01a0052f-dc7f-7b73-8afd-2984a91338bb}" +# The vault the continuous-research recipe accrues into. A different repo +# from the delivery scratch above: this one is the operator's knowledge base +# and the scenario asserts against its DEFAULT branch. +VAULT_REPO_ID="${CLAWMATES_VAULT_REPO_ID:-01a0052f-ddae-7630-8f01-b8aa9ec89a4b}" TEAM_TEMPLATE="${CLAWMATES_TEAM_TEMPLATE:-7e453826-41c4-4425-bab5-8f11fd0a14d7}" MISSION_TIMEOUT="${MISSION_TIMEOUT:-1800}" @@ -1449,6 +1453,154 @@ scenario_rolepolicy() { esac } +# ── Scenario: continuous-research — the whole pipeline, to the vault ── +# +# The recipe with the most moving parts and, until now, no standing check: +# harvest → triage → manifest → analysis → ranking → script → episode.json → +# audio → vault. It ran fine for a month and delivered its digest onto +# branches nobody merged, which is exactly the kind of thing a scenario +# notices and operator attention does not. +# +# Assertions are ordered by what they would have caught. The first is the +# defect itself: the digest reaching the vault's default branch. +# +# CLAWMATES_SKIP_AUDIO=1 skips the episode assertion — it spends ElevenLabs +# credits, and a run that is about the merge should not have to. + +CR_BODY=$(cat < + local token repo enc + token=$(ssh "$HOST" 'docker exec clawmates_server_1 printenv GITEA_TOKEN' | tr -d '\r') + repo=$(ssh "$HOST" "docker exec clawmates_postgres_1 psql -U postgres -d clawmates -tAc \ + \"select owner || '/' || name from repos where id='$VAULT_REPO_ID';\"" | head -1 | tr -d '[:space:]') + [ -n "$token" ] && [ -n "$repo" ] || return 1 + enc=$(printf '%s' "$1" | sed 's|/|%2F|g') + ssh "$HOST" "curl -sf -H 'Authorization: token $token' \ + 'https://git.redclaw.dev/api/v1/repos/$repo/raw/$2?ref=$enc'" +} + +scenario_continuous_research() { + local token mission status today base manifest analysis episode n_papers + token=$(mint_session) || { norun "cr: could not mint a session"; return 1; } + mission=$(create_mission "$token" "$(echo "$CR_BODY" | tr -d '\n')") \ + || { norun "cr: mission create failed"; return 1; } + echo " cr: mission=$mission" + api "$token" PATCH "/api/missions/$mission/status" '{"status":"running"}' >/dev/null + status=$(await_mission "$token" "$mission") + case "$status" in + completed|failed) pass "cr: mission reached $status" ;; + *) norun "cr: mission did not reach a terminal status in ${MISSION_TIMEOUT}s"; return 1 ;; + esac + + today=$(ssh "$HOST" "date -u +%F" | tr -d '[:space:]') + base=$(ssh "$HOST" "docker exec clawmates_postgres_1 psql -U postgres -d clawmates -tAc \ + \"select coalesce(default_branch,'main') from repos where id='$VAULT_REPO_ID';\"" | head -1 | tr -d '[:space:]') + + # 1. THE DEFECT. The digest must be on the vault's default branch, not + # stranded on a delivery branch. + analysis=$(vault_file "$base" "ContinuousResearch/$today/analysis.md" 2>/dev/null || true) + if [ -n "$analysis" ]; then + pass "cr: analysis.md is on $base ($(printf '%s' "$analysis" | wc -c | tr -d ' ') bytes)" + else + fail "cr: ContinuousResearch/$today/analysis.md is NOT on $base — the digest was produced and stranded on a branch" + fi + + # 2. The manifest the agents read, with the triage fields on every line. + manifest=$(vault_file "$base" "ContinuousResearch/$today/harvest.jsonl" 2>/dev/null || true) + if [ -z "$manifest" ]; then + fail "cr: harvest.jsonl is not on $base" + else + printf '%s' "$manifest" | python3 -c ' +import json,sys +rows=[json.loads(l) for l in sys.stdin if l.strip()] +miss=[r["source"] for r in rows if not r.get("topic_tags") and not (r.get("kind") or {}).get("is")] +ev=[(r.get("evidence") or {}).get("score") for r in rows] +ev=[e for e in ev if e is not None] +spread=(max(ev)-min(ev)) if len(ev)>1 else 0.0 +print(f"{len(rows)} {len(miss)} {len(ev)} {spread:.2f}")' > /tmp/cr-manifest 2>/dev/null \ + || { fail "cr: harvest.jsonl did not parse as JSONL"; : > /tmp/cr-manifest; } + read -r n_papers n_untriaged n_scored spread < /tmp/cr-manifest 2>/dev/null || true + case "${n_papers:-0}" in + 0) fail "cr: the manifest has no papers" ;; + *) pass "cr: manifest carries $n_papers paper(s), $n_scored with an evidence score" ;; + esac + # 3. The saturation guard, live: a score that says the same thing about + # every paper ranks nothing. 0.5 is patterns::SATURATED_BELOW. + if [ "${n_scored:-0}" -gt 2 ]; then + awk -v s="${spread:-0}" 'BEGIN{exit !(s+0 > 0.5)}' \ + && pass "cr: evidence scores span $spread across the harvest" \ + || fail "cr: evidence spans only $spread — the question is not separating this harvest" + fi + fi + + # 4. The analysis covers what was harvested. Both files go to disk first: + # the ids come from one and the prose from the other, and passing two + # multi-KB blobs through nested quoting is how the last assertion in + # this script reported an empty record for a correct one. + if [ -n "$analysis" ] && [ -n "$manifest" ]; then + local covered + printf '%s' "$manifest" > /tmp/cr-man.jsonl + printf '%s' "$analysis" > /tmp/cr-analysis.md + covered=$(python3 -c ' +import json +ids=[json.loads(l)["source"].split(":")[1] + for l in open("/tmp/cr-man.jsonl").read().splitlines() if l.strip()] +txt=open("/tmp/cr-analysis.md").read() +print(sum(1 for i in ids if i in txt), len(ids))' 2>/dev/null || echo "0 0") + set -- $covered + if [ "${1:-0}" = "${2:-1}" ] && [ "${2:-0}" -gt 0 ]; then + pass "cr: analysis.md names all $2 harvested paper(s)" + else + fail "cr: analysis.md names ${1:-0} of ${2:-0} harvested papers" + fi + fi + + # 5. The recipe's own done_when for the script phase, checked independently + # of the judge that already ruled on it. + episode=$(vault_file "$base" "ContinuousResearch/$today/episode.json" 2>/dev/null || true) + if [ -z "$episode" ]; then + fail "cr: episode.json is not on $base" + else + printf '%s' "$episode" | python3 -c ' +import json,sys +d=json.load(sys.stdin) +hl=d.get("highlights") or [] +bad=[h for h in hl if not (10 <= len(h) <= 70)] +assert d.get("title"), "no title" +assert hl, "no highlights" +assert not bad, f"{len(bad)} highlight(s) outside 10-70 chars: {bad[:2]}" +print("ok")' >/dev/null 2>&1 \ + && pass "cr: episode.json has a title and highlights within 10-70 chars" \ + || fail "cr: episode.json missing a title, or a highlight outside 10-70 chars" + fi + + # 6. The audio leg. An `unrenderable` tombstone is a DIFFERENT outcome from + # no row at all, and both are different from a rendered episode. + if [ "${CLAWMATES_SKIP_AUDIO:-0}" = "1" ]; then + pass "cr: (audio assertion skipped by CLAWMATES_SKIP_AUDIO)" + else + local ep + ep=$(ssh "$HOST" "docker exec clawmates_postgres_1 psql -U postgres -d clawmates -tAc \ + \"select coalesce(rendered_by,'-') || ' ' || coalesce(duration_secs,0)::text \ + from podcast_episodes where mission_id='$mission';\"" | head -1 | tr -d '\r') + case "$ep" in + '') fail "cr: no podcast_episodes row — the render sweep never reached this mission" ;; + unrenderable*) fail "cr: the episode is a tombstone (unrenderable) — the script was not found in the checkout OR the vault" ;; + *' 0') fail "cr: an episode was recorded with zero duration: $ep" ;; + *) pass "cr: audio rendered by $ep" ;; + esac + fi +} + # ── Scenario: multi-role with a real test suite ────────────────── # # The workload that failed with `COMMIT_EDITMSG: Permission denied` under the @@ -2108,6 +2260,9 @@ case "${1:-all}" in rolepolicy) scenario_rolepolicy ;; + continuous-research) + scenario_continuous_research + ;; research-only) run_scenario research-only "$(echo "$RESEARCH_ONLY_BODY" | tr -d '\n')" assert_research_only no-checkout ;; @@ -2153,6 +2308,7 @@ case "${1:-all}" in run_scenario gatepolicy "$(echo "$GATEPOLICY_BODY" | tr -d '\n')" assert_gatepolicy scenario_door scenario_rolepolicy + scenario_continuous_research run_scenario research-only "$(echo "$RESEARCH_ONLY_BODY" | tr -d '\n')" assert_research_only no-checkout run_scenario research-vm "$(echo "$RESEARCH_VM_BODY" | tr -d '\n')" assert_research_only no-checkout run_scenario benchmark "$(echo "$BENCHMARK_BODY" | tr -d '\n')" assert_benchmark @@ -2165,7 +2321,7 @@ case "${1:-all}" in scenario_drain_midmission ;; *) - die "unknown scenario: $1 (selftest|uids|chain|multirole|noop|microvm|canary|glm|kimi|gatecap|goodhart|gatepolicy|door|rolepolicy|research-only|research-vm|benchmark|security|refactor|composed|roster|local-ornith|capacity|drain-midmission|all)" + die "unknown scenario: $1 (selftest|uids|chain|multirole|noop|microvm|canary|glm|kimi|gatecap|goodhart|gatepolicy|door|rolepolicy|continuous-research|research-only|research-vm|benchmark|security|refactor|composed|roster|local-ornith|capacity|drain-midmission|all)" ;; esac