feat(billing): agent-side spend records who was paid
Judge spend gained provider, model and mission on 2026-09-14; agent spend — the larger half — did not. The runtime's `done` frame has always carried `model` and `provider` beside the two token counts, and `topology_exec` read only the counts, summed them, and charged the sum as output with no record of which provider served the turn. `TurnOutcome` and `StepRecord` carry a `Spend` now (input/output split, provider, model), the worker passes it through `cm_billing::charge` along with the mission id, and the chat runtime records the model it requested — that loop drives one provider with no chain, so requested is answered. A bare model name is recorded without a guessed family. `StepRecord.spend` is `serde(default)` so journaled checkpoints from before this field still load, and `tokens` stays as the total every reader keys on. `charge` moved from `query!` to `query`: the macro pins the statement to offline metadata that a schema change then has to regenerate against a live database, for columns that are nullable text and uuid. The done-frame test now asserts the split and the provider survive, not just the sum. Co-Authored-By: Claude Opus 5 <[email protected]> Claude-Session: https://claude.ai/code/session_01WZb5A2kfVfjpdwSochkuHz
This commit is contained in:
co-authored by
Claude Opus 5
parent
736b6a9a82
commit
483de9f88a
@@ -295,6 +295,7 @@ impl<V: PhaseVm> TurnExecutor for MicroVmTurnExecutor<V> {
|
|||||||
// the honest value for "not measured on this path".
|
// the honest value for "not measured on this path".
|
||||||
tokens: 0,
|
tokens: 0,
|
||||||
gated: Vec::new(),
|
gated: Vec::new(),
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -173,6 +173,7 @@ impl TurnExecutor for SubTopologyExecutor {
|
|||||||
output: record.final_output,
|
output: record.final_output,
|
||||||
tokens: record.totals.tokens,
|
tokens: record.totals.tokens,
|
||||||
gated,
|
gated,
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -86,6 +86,7 @@ fn step(
|
|||||||
output: output.into(),
|
output: output.into(),
|
||||||
gated: Vec::new(),
|
gated: Vec::new(),
|
||||||
tokens: 0,
|
tokens: 0,
|
||||||
|
spend: Default::default(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -683,6 +683,7 @@ impl ZeroClawDriveExecutor {
|
|||||||
{
|
{
|
||||||
let mut output = String::new();
|
let mut output = String::new();
|
||||||
let mut tokens: u64 = 0;
|
let mut tokens: u64 = 0;
|
||||||
|
let mut spend = cm_orchestrator::Spend::default();
|
||||||
let mut gated: Vec<GatedAction> = Vec::new();
|
let mut gated: Vec<GatedAction> = Vec::new();
|
||||||
let mut trace = ToolTrace::default();
|
let mut trace = ToolTrace::default();
|
||||||
|
|
||||||
@@ -720,6 +721,23 @@ impl ZeroClawDriveExecutor {
|
|||||||
let input = v.get("input_tokens").and_then(|n| n.as_u64()).unwrap_or(0);
|
let input = v.get("input_tokens").and_then(|n| n.as_u64()).unwrap_or(0);
|
||||||
let out = v.get("output_tokens").and_then(|n| n.as_u64()).unwrap_or(0);
|
let out = v.get("output_tokens").and_then(|n| n.as_u64()).unwrap_or(0);
|
||||||
tokens = input + out;
|
tokens = input + out;
|
||||||
|
// The frame has always carried these; only
|
||||||
|
// `tokens` was read, so every agent turn was
|
||||||
|
// charged with no record of who was paid.
|
||||||
|
spend = cm_orchestrator::Spend {
|
||||||
|
input_tokens: input,
|
||||||
|
output_tokens: out,
|
||||||
|
provider: v
|
||||||
|
.get("provider")
|
||||||
|
.and_then(|p| p.as_str())
|
||||||
|
.filter(|p| !p.is_empty())
|
||||||
|
.map(str::to_string),
|
||||||
|
model: v
|
||||||
|
.get("model")
|
||||||
|
.and_then(|m| m.as_str())
|
||||||
|
.filter(|m| !m.is_empty())
|
||||||
|
.map(str::to_string),
|
||||||
|
};
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
"approval_request" => {
|
"approval_request" => {
|
||||||
@@ -807,6 +825,7 @@ impl ZeroClawDriveExecutor {
|
|||||||
output: output.trim().to_string(),
|
output: output.trim().to_string(),
|
||||||
tokens,
|
tokens,
|
||||||
gated,
|
gated,
|
||||||
|
spend,
|
||||||
},
|
},
|
||||||
trace,
|
trace,
|
||||||
))
|
))
|
||||||
@@ -968,7 +987,10 @@ mod tests {
|
|||||||
json!({"type": "session_start", "session_id": "s1", "resumed": false}),
|
json!({"type": "session_start", "session_id": "s1", "resumed": false}),
|
||||||
json!({"type": "chunk", "content": "hel"}),
|
json!({"type": "chunk", "content": "hel"}),
|
||||||
json!({"type": "chunk", "content": "lo"}),
|
json!({"type": "chunk", "content": "lo"}),
|
||||||
json!({"type": "done", "input_tokens": 5, "output_tokens": 7}),
|
// The real frame carries model and provider; the executor read
|
||||||
|
// only the two token counts until 2026-09-14.
|
||||||
|
json!({"type": "done", "input_tokens": 5, "output_tokens": 7,
|
||||||
|
"model": "claude-sonnet-5", "provider": "anthropic"}),
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1068,6 +1090,16 @@ mod tests {
|
|||||||
let out = exec.run_turn(req()).await.unwrap();
|
let out = exec.run_turn(req()).await.unwrap();
|
||||||
assert_eq!(out.output, "hello");
|
assert_eq!(out.output, "hello");
|
||||||
assert_eq!(out.tokens, 12);
|
assert_eq!(out.tokens, 12);
|
||||||
|
assert_eq!(
|
||||||
|
out.spend,
|
||||||
|
cm_orchestrator::Spend {
|
||||||
|
input_tokens: 5,
|
||||||
|
output_tokens: 7,
|
||||||
|
provider: Some("anthropic".into()),
|
||||||
|
model: Some("claude-sonnet-5".into()),
|
||||||
|
},
|
||||||
|
"the split and the provider must survive the done frame, not just the sum"
|
||||||
|
);
|
||||||
assert!(out.gated.is_empty());
|
assert!(out.gated.is_empty());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -556,16 +556,35 @@ async fn drive<E: TurnExecutor>(
|
|||||||
}
|
}
|
||||||
if let Some(agent_id) = agent_of.get(&last.node_id).copied() {
|
if let Some(agent_id) = agent_of.get(&last.node_id).copied() {
|
||||||
if last.tokens > 0 {
|
if last.tokens > 0 {
|
||||||
// The executor reports ONE total, not an in/out split.
|
// The split and the provider come from the runtime's
|
||||||
// Credits price the sum, so cost is right; the columns
|
// `done` frame via `StepRecord.spend`. An executor
|
||||||
// record it as output rather than inventing a split.
|
// that reports only a total leaves the split at 0/0
|
||||||
|
// and the total goes on the output side, as before.
|
||||||
|
let (tin, tout) = if last.spend.input_tokens + last.spend.output_tokens > 0
|
||||||
|
{
|
||||||
|
(last.spend.input_tokens, last.spend.output_tokens)
|
||||||
|
} else {
|
||||||
|
(0, last.tokens as u64)
|
||||||
|
};
|
||||||
|
let mission_id: Option<Uuid> = sqlx::query_scalar::<_, Option<Uuid>>(
|
||||||
|
"SELECT mission_id FROM topology_runs WHERE id = $1",
|
||||||
|
)
|
||||||
|
.bind(id)
|
||||||
|
.fetch_optional(&pool)
|
||||||
|
.await
|
||||||
|
.ok()
|
||||||
|
.flatten()
|
||||||
|
.flatten();
|
||||||
if let Err(e) = cm_billing::charge(
|
if let Err(e) = cm_billing::charge(
|
||||||
&pool,
|
&pool,
|
||||||
cm_domain::WorkspaceId::from(workspace_id),
|
cm_domain::WorkspaceId::from(workspace_id),
|
||||||
cm_domain::AgentId::from(agent_id),
|
cm_domain::AgentId::from(agent_id),
|
||||||
None,
|
None,
|
||||||
0,
|
tin,
|
||||||
last.tokens as u64,
|
tout,
|
||||||
|
last.spend.provider.as_deref(),
|
||||||
|
last.spend.model.as_deref(),
|
||||||
|
mission_id,
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
|
|||||||
@@ -36,21 +36,36 @@ pub async fn charge(
|
|||||||
run_id: Option<Uuid>,
|
run_id: Option<Uuid>,
|
||||||
input_tokens: u64,
|
input_tokens: u64,
|
||||||
output_tokens: u64,
|
output_tokens: u64,
|
||||||
|
// Who was paid, and for which mission. `None` where the executor did not
|
||||||
|
// say. Until 2026-09-14 no agent-side row carried these, so the larger
|
||||||
|
// half of the spend could not be asked per provider — the judge's half
|
||||||
|
// could, and that is how a plan emptying twice went unexplained.
|
||||||
|
provider: Option<&str>,
|
||||||
|
model: Option<&str>,
|
||||||
|
mission_id: Option<Uuid>,
|
||||||
) -> Result<i64, BillingError> {
|
) -> Result<i64, BillingError> {
|
||||||
let owed = credits_for_tokens(input_tokens + output_tokens);
|
let owed = credits_for_tokens(input_tokens + output_tokens);
|
||||||
let mut tx = pool.begin().await?;
|
let mut tx = pool.begin().await?;
|
||||||
|
|
||||||
sqlx::query!(
|
// `sqlx::query`, not `query!`: the macro pins this statement to offline
|
||||||
|
// metadata that a schema change then has to regenerate against a live
|
||||||
|
// database, and the columns added by migration 0085 are nullable text
|
||||||
|
// and uuid — nothing here that a compile-time check would catch.
|
||||||
|
sqlx::query(
|
||||||
"INSERT INTO usage_events
|
"INSERT INTO usage_events
|
||||||
(workspace_id, agent_id, run_id, kind, tokens_in, tokens_out, credits)
|
(workspace_id, agent_id, run_id, kind, tokens_in, tokens_out, credits,
|
||||||
VALUES ($1, $2, $3, 'llm_tokens', $4, $5, $6)",
|
provider, model, mission_id)
|
||||||
workspace_id.as_uuid(),
|
VALUES ($1, $2, $3, 'llm_tokens', $4, $5, $6, $7, $8, $9)",
|
||||||
agent_id.as_uuid(),
|
|
||||||
run_id,
|
|
||||||
input_tokens as i64,
|
|
||||||
output_tokens as i64,
|
|
||||||
sqlx::types::BigDecimal::from(owed),
|
|
||||||
)
|
)
|
||||||
|
.bind(workspace_id.as_uuid())
|
||||||
|
.bind(agent_id.as_uuid())
|
||||||
|
.bind(run_id)
|
||||||
|
.bind(input_tokens as i64)
|
||||||
|
.bind(output_tokens as i64)
|
||||||
|
.bind(sqlx::types::BigDecimal::from(owed))
|
||||||
|
.bind(provider)
|
||||||
|
.bind(model)
|
||||||
|
.bind(mission_id)
|
||||||
.execute(&mut *tx)
|
.execute(&mut *tx)
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
|
|||||||
@@ -62,7 +62,7 @@ async fn charges_span_lots_oldest_first_and_record_usage() {
|
|||||||
.unwrap();
|
.unwrap();
|
||||||
|
|
||||||
// 2500 tokens → 3 credits: drains the first lot (2) then one more.
|
// 2500 tokens → 3 credits: drains the first lot (2) then one more.
|
||||||
let deducted = charge(&pool, ws.id, agent.id, Some(run_id), 1500, 1000)
|
let deducted = charge(&pool, ws.id, agent.id, Some(run_id), 1500, 1000, None, None, None)
|
||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(deducted, 3);
|
assert_eq!(deducted, 3);
|
||||||
@@ -96,7 +96,7 @@ async fn an_empty_workspace_records_usage_but_clamps_at_zero() {
|
|||||||
.unwrap();
|
.unwrap();
|
||||||
|
|
||||||
// Owes 5, only 1 available: deducts 1, balance hits zero, never negative.
|
// Owes 5, only 1 available: deducts 1, balance hits zero, never negative.
|
||||||
let deducted = charge(&pool, ws.id, agent.id, Some(run_id), 4000, 500)
|
let deducted = charge(&pool, ws.id, agent.id, Some(run_id), 4000, 500, None, None, None)
|
||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
assert_eq!(deducted, 1);
|
assert_eq!(deducted, 1);
|
||||||
|
|||||||
@@ -145,6 +145,7 @@ mod tests {
|
|||||||
output: format!("{}<{}>", req.role, req.context.join("|")),
|
output: format!("{}<{}>", req.role, req.context.join("|")),
|
||||||
tokens: 10,
|
tokens: 10,
|
||||||
gated: vec![],
|
gated: vec![],
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -147,6 +147,7 @@ mod tests {
|
|||||||
output: format!("{}:{}", req.role, req.context.join(" ")),
|
output: format!("{}:{}", req.role, req.context.join(" ")),
|
||||||
tokens: 10,
|
tokens: 10,
|
||||||
gated,
|
gated,
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -95,15 +95,35 @@ pub struct TurnRequest {
|
|||||||
pub context: Vec<String>,
|
pub context: Vec<String>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// What a turn cost and who was paid — the part of the runtime's `done` frame
|
||||||
|
/// that `tokens` alone threw away.
|
||||||
|
///
|
||||||
|
/// `tokens` stayed as the one total every reader already keys on. This is
|
||||||
|
/// the split beside it, plus the provider and model that answered, so the
|
||||||
|
/// spend can be asked per provider BEFORE a plan limit asks it for you. Judge
|
||||||
|
/// spend gained this on 2026-09-14 and agent spend did not, which left the
|
||||||
|
/// larger of the two invisible.
|
||||||
|
#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
|
||||||
|
pub struct Spend {
|
||||||
|
pub input_tokens: u64,
|
||||||
|
pub output_tokens: u64,
|
||||||
|
/// Provider family that answered (`anthropic`, `glm`, …), when the
|
||||||
|
/// runtime said. `None` on executors that do not report one.
|
||||||
|
pub provider: Option<String>,
|
||||||
|
pub model: Option<String>,
|
||||||
|
}
|
||||||
|
|
||||||
/// The result of a single agent turn.
|
/// The result of a single agent turn.
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
pub struct TurnOutcome {
|
pub struct TurnOutcome {
|
||||||
/// The turn's textual output.
|
/// The turn's textual output.
|
||||||
pub output: String,
|
pub output: String,
|
||||||
/// Model tokens spent (cost proxy).
|
/// Model tokens spent (cost proxy). Input + output.
|
||||||
pub tokens: u64,
|
pub tokens: u64,
|
||||||
/// Any sandbox-leaving actions attempted during the turn.
|
/// Any sandbox-leaving actions attempted during the turn.
|
||||||
pub gated: Vec<GatedAction>,
|
pub gated: Vec<GatedAction>,
|
||||||
|
/// The split and the provider behind `tokens`.
|
||||||
|
pub spend: Spend,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Runs one safe agent turn. The real impl wraps `cm-runtime::Runtime`
|
/// Runs one safe agent turn. The real impl wraps `cm-runtime::Runtime`
|
||||||
@@ -143,6 +163,10 @@ pub struct StepRecord {
|
|||||||
pub gated: Vec<GatedAction>,
|
pub gated: Vec<GatedAction>,
|
||||||
/// Tokens it spent.
|
/// Tokens it spent.
|
||||||
pub tokens: u64,
|
pub tokens: u64,
|
||||||
|
/// The split and provider behind `tokens`. `default` so checkpoints
|
||||||
|
/// journaled before this field existed still load.
|
||||||
|
#[serde(default)]
|
||||||
|
pub spend: Spend,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// The full record of a topology run (journal + final output + totals).
|
/// The full record of a topology run (journal + final output + totals).
|
||||||
@@ -264,6 +288,7 @@ where
|
|||||||
output: outcome.output,
|
output: outcome.output,
|
||||||
gated: outcome.gated,
|
gated: outcome.gated,
|
||||||
tokens: outcome.tokens,
|
tokens: outcome.tokens,
|
||||||
|
spend: outcome.spend,
|
||||||
});
|
});
|
||||||
|
|
||||||
// Hand the caller a durable snapshot to persist before the next turn.
|
// Hand the caller a durable snapshot to persist before the next turn.
|
||||||
@@ -309,6 +334,7 @@ mod tests {
|
|||||||
output: format!("{}({})<{}>", req.role, req.node_id, req.context.join("|")),
|
output: format!("{}({})<{}>", req.role, req.node_id, req.context.join("|")),
|
||||||
tokens: 10,
|
tokens: 10,
|
||||||
gated,
|
gated,
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -383,6 +409,7 @@ mod tests {
|
|||||||
output: format!("{}({})<{}>", req.role, req.node_id, req.context.join("|")),
|
output: format!("{}({})<{}>", req.role, req.node_id, req.context.join("|")),
|
||||||
tokens: 10,
|
tokens: 10,
|
||||||
gated: vec![],
|
gated: vec![],
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -89,6 +89,7 @@ impl TurnExecutor for ProviderExecutor {
|
|||||||
tokens,
|
tokens,
|
||||||
// Tool-free reasoning turns leave the sandbox nowhere.
|
// Tool-free reasoning turns leave the sandbox nowhere.
|
||||||
gated: vec![],
|
gated: vec![],
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -74,6 +74,7 @@ mod tests {
|
|||||||
output: format!("{}<{}>", req.node_id, req.context.join("|")),
|
output: format!("{}<{}>", req.node_id, req.context.join("|")),
|
||||||
tokens: 5,
|
tokens: 5,
|
||||||
gated: vec![],
|
gated: vec![],
|
||||||
|
spend: Default::default(),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -875,6 +875,13 @@ impl Runtime {
|
|||||||
);
|
);
|
||||||
// Meter the run (§8.4). Billing failures never fail the run — the
|
// Meter the run (§8.4). Billing failures never fail the run — the
|
||||||
// usage ledger is the recovery path.
|
// usage ledger is the recovery path.
|
||||||
|
// This loop drives ONE provider with no fallback chain, so the model
|
||||||
|
// the request named is the model that answered. The family is taken
|
||||||
|
// only from an explicit `provider:` prefix — a bare model name is
|
||||||
|
// recorded as-is with no family rather than guessed at, and a chat
|
||||||
|
// run is not a mission.
|
||||||
|
let model = state.request.model.as_str();
|
||||||
|
let provider = model.split_once(':').map(|(p, _)| p);
|
||||||
if let Err(error) = cm_billing::charge(
|
if let Err(error) = cm_billing::charge(
|
||||||
&self.inner.pool,
|
&self.inner.pool,
|
||||||
state.workspace_id,
|
state.workspace_id,
|
||||||
@@ -882,6 +889,9 @@ impl Runtime {
|
|||||||
Some(run_id),
|
Some(run_id),
|
||||||
state.input_tokens,
|
state.input_tokens,
|
||||||
state.output_tokens,
|
state.output_tokens,
|
||||||
|
provider,
|
||||||
|
Some(model),
|
||||||
|
None,
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
|
|||||||
Reference in New Issue
Block a user