From 07c0833c934dd4467d7e77b4e76bcebd31b422ae Mon Sep 17 00:00:00 2001 From: Joseph Schorr Date: Wed, 30 Sep 2026 12:07:31 -0400 Subject: [PATCH 1/6] Add ByTool cost breakdown types and regenerate manifests --- ...tprimitives.authzed.com_agentsessions.yaml | 43 +++++++++++++++++-- pkg/apis/v1alpha1/agentsession_types.go | 43 ++++++++++++++++--- pkg/apis/v1alpha1/toolcostbucket_test.go | 24 +++++++++++ pkg/apis/v1alpha1/zz_generated.deepcopy.go | 20 +++++++++ pkg/platform/manifests/install.yaml | 43 +++++++++++++++++-- pkg/platform/pipeline/types.go | 21 +++++++++ site/content/docs/crd-agentsession.mdx | 8 +++- 7 files changed, 185 insertions(+), 17 deletions(-) create mode 100644 pkg/apis/v1alpha1/toolcostbucket_test.go diff --git a/config/crds/agentprimitives.authzed.com_agentsessions.yaml b/config/crds/agentprimitives.authzed.com_agentsessions.yaml index 00984ca9..3d8fd368 100644 --- a/config/crds/agentprimitives.authzed.com_agentsessions.yaml +++ b/config/crds/agentprimitives.authzed.com_agentsessions.yaml @@ -2012,8 +2012,11 @@ spec: properties: amountMicroUSD: description: |- - AmountMicroUSD is the session total in micro-USD (1e-6 USD); 0 and - meaningless when PricingKnown is false. + AmountMicroUSD is the session total in micro-USD (1e-6 USD): the sum of + every priced component (ByModel + ByTool). When PricingKnown is false, at + least one served MODEL had no rate, so this is a LOWER BOUND on spend (the + components that could be priced), not the full session cost — it is no + longer necessarily 0, because provider-reported tool cost is always priced. format: int64 type: integer asOf: @@ -2065,6 +2068,37 @@ spec: type: object type: array x-kubernetes-list-type: atomic + byTool: + description: |- + ByTool breaks out the cost of inner interactive toolkits (e.g. a `claude` + Claude Code sub-run's own provider-reported spend) a session drove. These + are ADDED to AmountMicroUSD alongside ByModel — a session's total is model + spend plus tool spend. Empty when no interactive toolkit reported a cost. + Tool cost is always provider-reported, never table-priced, so a bucket's + PricingKnown is always true. + items: + description: ToolCostBucket is one interactive toolkit's slice + of a session's cost. + properties: + amountMicroUSD: + description: |- + AmountMicroUSD is this tool's provider-reported cost, summed across its + invocations, in micro-USD (1e-6 USD). + format: int64 + type: integer + pricingKnown: + description: |- + PricingKnown is true when AmountMicroUSD is a real cost. Tool cost is + always provider-reported, so this is true whenever a cost was reported. + type: boolean + tool: + description: Tool is the outer tool name (e.g. "claude-oauth"). + type: string + required: + - tool + type: object + type: array + x-kubernetes-list-type: atomic currency: description: Currency is the ISO code the amount is denominated in ("USD"). @@ -2074,8 +2108,9 @@ spec: type: string pricingKnown: description: |- - PricingKnown is false when the model had no price; the amount is then 0 - and must not be shown as a real cost. + PricingKnown is false when a served model had no price. The amount is then + a lower bound (priced components only), not the full cost. Tool buckets are + always priced, so this tracks model-pricing coverage specifically. type: boolean type: object failureReason: diff --git a/pkg/apis/v1alpha1/agentsession_types.go b/pkg/apis/v1alpha1/agentsession_types.go index 0191a853..a323a3b5 100644 --- a/pkg/apis/v1alpha1/agentsession_types.go +++ b/pkg/apis/v1alpha1/agentsession_types.go @@ -964,12 +964,16 @@ type AgentSessionProgress struct { // EstimatedSessionCost is a best-effort USD estimate of a session's LLM spend, // stamped at SessionEnd from cumulative token usage × the provider's per-model -// pricing. Money is stored as integer micro-USD (1e-6 USD) to avoid float drift. -// PricingKnown=false means the model had no price; AmountMicroUSD is then 0 and -// must not be shown as a real cost. +// pricing PLUS any interactive toolkit's own provider-reported cost (ByTool). +// Money is stored as integer micro-USD (1e-6 USD) to avoid float drift. +// PricingKnown=false means a served model had no price, so AmountMicroUSD is a +// lower bound (the priced components only), not the full cost. type EstimatedSessionCost struct { - // AmountMicroUSD is the session total in micro-USD (1e-6 USD); 0 and - // meaningless when PricingKnown is false. + // AmountMicroUSD is the session total in micro-USD (1e-6 USD): the sum of + // every priced component (ByModel + ByTool). When PricingKnown is false, at + // least one served MODEL had no rate, so this is a LOWER BOUND on spend (the + // components that could be priced), not the full session cost — it is no + // longer necessarily 0, because provider-reported tool cost is always priced. // +optional AmountMicroUSD int64 `json:"amountMicroUSD,omitempty"` // Currency is the ISO code the amount is denominated in ("USD"). @@ -978,8 +982,9 @@ type EstimatedSessionCost struct { // Model is the configured model the session ran under. // +optional Model string `json:"model,omitempty"` - // PricingKnown is false when the model had no price; the amount is then 0 - // and must not be shown as a real cost. + // PricingKnown is false when a served model had no price. The amount is then + // a lower bound (priced components only), not the full cost. Tool buckets are + // always priced, so this tracks model-pricing coverage specifically. // +optional PricingKnown bool `json:"pricingKnown,omitempty"` // AsOf is when the estimate was computed. @@ -994,6 +999,16 @@ type EstimatedSessionCost struct { // +optional // +listType=atomic ByModel []ModelCostBucket `json:"byModel,omitempty"` + + // ByTool breaks out the cost of inner interactive toolkits (e.g. a `claude` + // Claude Code sub-run's own provider-reported spend) a session drove. These + // are ADDED to AmountMicroUSD alongside ByModel — a session's total is model + // spend plus tool spend. Empty when no interactive toolkit reported a cost. + // Tool cost is always provider-reported, never table-priced, so a bucket's + // PricingKnown is always true. + // +optional + // +listType=atomic + ByTool []ToolCostBucket `json:"byTool,omitempty"` } // ModelCostBucket is one served model's slice of a session's cost estimate. @@ -1020,6 +1035,20 @@ type ModelCostBucket struct { PricingKnown bool `json:"pricingKnown,omitempty"` } +// ToolCostBucket is one interactive toolkit's slice of a session's cost. +type ToolCostBucket struct { + // Tool is the outer tool name (e.g. "claude-oauth"). + Tool string `json:"tool"` + // AmountMicroUSD is this tool's provider-reported cost, summed across its + // invocations, in micro-USD (1e-6 USD). + // +optional + AmountMicroUSD int64 `json:"amountMicroUSD,omitempty"` + // PricingKnown is true when AmountMicroUSD is a real cost. Tool cost is + // always provider-reported, so this is true whenever a cost was reported. + // +optional + PricingKnown bool `json:"pricingKnown,omitempty"` +} + // ParentExchange is a delegated child's record of asking the agent that // delegated to it a question, and of whether that question is still // outstanding. diff --git a/pkg/apis/v1alpha1/toolcostbucket_test.go b/pkg/apis/v1alpha1/toolcostbucket_test.go new file mode 100644 index 00000000..b899caff --- /dev/null +++ b/pkg/apis/v1alpha1/toolcostbucket_test.go @@ -0,0 +1,24 @@ +package v1alpha1 + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestEstimatedSessionCost_DeepCopy_ByTool(t *testing.T) { + in := &EstimatedSessionCost{ + AmountMicroUSD: 19_860_000, + PricingKnown: true, + ByModel: []ModelCostBucket{{Model: "anthropic/claude-sonnet-5", AmountMicroUSD: 840_000, PricingKnown: true}}, + ByTool: []ToolCostBucket{{Tool: "claude-oauth", AmountMicroUSD: 19_020_000, PricingKnown: true}}, + } + out := in.DeepCopy() + require.Len(t, out.ByTool, 1) + assert.Equal(t, "claude-oauth", out.ByTool[0].Tool) + assert.Equal(t, int64(19_020_000), out.ByTool[0].AmountMicroUSD) + // Deep, not shallow: mutating the copy must not touch the original. + out.ByTool[0].AmountMicroUSD = 0 + assert.Equal(t, int64(19_020_000), in.ByTool[0].AmountMicroUSD) +} diff --git a/pkg/apis/v1alpha1/zz_generated.deepcopy.go b/pkg/apis/v1alpha1/zz_generated.deepcopy.go index dcb66ad0..030752be 100644 --- a/pkg/apis/v1alpha1/zz_generated.deepcopy.go +++ b/pkg/apis/v1alpha1/zz_generated.deepcopy.go @@ -3016,6 +3016,11 @@ func (in *EstimatedSessionCost) DeepCopyInto(out *EstimatedSessionCost) { *out = make([]ModelCostBucket, len(*in)) copy(*out, *in) } + if in.ByTool != nil { + in, out := &in.ByTool, &out.ByTool + *out = make([]ToolCostBucket, len(*in)) + copy(*out, *in) + } } // DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new EstimatedSessionCost. @@ -8026,6 +8031,21 @@ func (in *ToolCallsAuthz) DeepCopy() *ToolCallsAuthz { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *ToolCostBucket) DeepCopyInto(out *ToolCostBucket) { + *out = *in +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new ToolCostBucket. +func (in *ToolCostBucket) DeepCopy() *ToolCostBucket { + if in == nil { + return nil + } + out := new(ToolCostBucket) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *ToolGuardCeiling) DeepCopyInto(out *ToolGuardCeiling) { *out = *in diff --git a/pkg/platform/manifests/install.yaml b/pkg/platform/manifests/install.yaml index c0e362f8..b0a10980 100644 --- a/pkg/platform/manifests/install.yaml +++ b/pkg/platform/manifests/install.yaml @@ -6572,8 +6572,11 @@ spec: properties: amountMicroUSD: description: |- - AmountMicroUSD is the session total in micro-USD (1e-6 USD); 0 and - meaningless when PricingKnown is false. + AmountMicroUSD is the session total in micro-USD (1e-6 USD): the sum of + every priced component (ByModel + ByTool). When PricingKnown is false, at + least one served MODEL had no rate, so this is a LOWER BOUND on spend (the + components that could be priced), not the full session cost — it is no + longer necessarily 0, because provider-reported tool cost is always priced. format: int64 type: integer asOf: @@ -6625,6 +6628,37 @@ spec: type: object type: array x-kubernetes-list-type: atomic + byTool: + description: |- + ByTool breaks out the cost of inner interactive toolkits (e.g. a `claude` + Claude Code sub-run's own provider-reported spend) a session drove. These + are ADDED to AmountMicroUSD alongside ByModel — a session's total is model + spend plus tool spend. Empty when no interactive toolkit reported a cost. + Tool cost is always provider-reported, never table-priced, so a bucket's + PricingKnown is always true. + items: + description: ToolCostBucket is one interactive toolkit's slice + of a session's cost. + properties: + amountMicroUSD: + description: |- + AmountMicroUSD is this tool's provider-reported cost, summed across its + invocations, in micro-USD (1e-6 USD). + format: int64 + type: integer + pricingKnown: + description: |- + PricingKnown is true when AmountMicroUSD is a real cost. Tool cost is + always provider-reported, so this is true whenever a cost was reported. + type: boolean + tool: + description: Tool is the outer tool name (e.g. "claude-oauth"). + type: string + required: + - tool + type: object + type: array + x-kubernetes-list-type: atomic currency: description: Currency is the ISO code the amount is denominated in ("USD"). @@ -6634,8 +6668,9 @@ spec: type: string pricingKnown: description: |- - PricingKnown is false when the model had no price; the amount is then 0 - and must not be shown as a real cost. + PricingKnown is false when a served model had no price. The amount is then + a lower bound (priced components only), not the full cost. Tool buckets are + always priced, so this tracks model-pricing coverage specifically. type: boolean type: object failureReason: diff --git a/pkg/platform/pipeline/types.go b/pkg/platform/pipeline/types.go index 6a4125aa..8ef17201 100644 --- a/pkg/platform/pipeline/types.go +++ b/pkg/platform/pipeline/types.go @@ -106,6 +106,13 @@ type SessionEndInfo struct { // Model above. Empty means "no per-model breakdown available" — reporters // must NOT read that as zero usage. ByModel []ModelUsage + + // ByTool is the per-interactive-toolkit provider-reported cost accumulated + // across this session (e.g. an inner `claude` Claude Code sub-run's own + // billed cost), keyed by outer tool name in first-encountered order. Empty + // means no interactive toolkit reported a terminal cost — reporters must + // NOT read that as zero spend elsewhere. + ByTool []ToolUsage } // ModelUsage is one served model's cumulative usage for a session, keyed by its @@ -134,6 +141,20 @@ type ModelUsage struct { CostReported bool } +// ToolUsage is one interactive toolkit's cumulative provider-reported cost for +// a session, keyed by the outer tool name. Distinct from ModelUsage: a toolkit +// reports a single billed dollar figure (not token counts we re-price), so this +// carries only the summed cost and whether any invocation reported one. +type ToolUsage struct { + // Tool is the outer tool name (e.g. "claude-oauth"). + Tool string + // CostMicroUSD is the provider-reported cost summed across this tool's + // invocations, in micro-USD (1e-6 USD). + CostMicroUSD int64 + // CostReported is true when at least one invocation reported a cost. + CostReported bool +} + // MetaagentInfo is the payload for the four metaagent control-plane points // (MetaagentReceived/Extract/Decide/Apply). It carries ONLY generic primitives // — pkg/platform/pipeline imports no domain types, so the AgentClass envelope, the diff --git a/site/content/docs/crd-agentsession.mdx b/site/content/docs/crd-agentsession.mdx index a15a5b51..d7c7cafe 100644 --- a/site/content/docs/crd-agentsession.mdx +++ b/site/content/docs/crd-agentsession.mdx @@ -271,7 +271,7 @@ Status is controller-owned (observed state). | `status.effectiveSettings.toolGuard.namespace.rules[].rateLimit.maxCallsPerTurn` | `integer (int32)` | MaxCallsPerTurn caps calls within a single agent turn; 0 is unlimited. _(min 1)_ | | `status.effectiveSettings.toolGuard.namespace.rules[].rateLimit.window` | `string` | Window is the sliding-window span for MaxCalls; both must be set together. | | `status.estimatedCost` | `object` | EstimatedCost is the best-effort end-of-session USD cost estimate. Runner-owned; written at SessionEnd when reportSessionCost is on. | -| `status.estimatedCost.amountMicroUSD` | `integer (int64)` | AmountMicroUSD is the session total in micro-USD (1e-6 USD); 0 and meaningless when PricingKnown is false. | +| `status.estimatedCost.amountMicroUSD` | `integer (int64)` | AmountMicroUSD is the session total in micro-USD (1e-6 USD): the sum of every priced component (ByModel + ByTool). When PricingKnown is false, at least one served MODEL had no rate, so this is a LOWER BOUND on spend (the components that could be priced), not the full session cost — it is no longer necessarily 0, because provider-reported tool cost is always priced. | | `status.estimatedCost.asOf` | `string (date-time)` | AsOf is when the estimate was computed. | | `status.estimatedCost.byModel` | `[]object` | ByModel breaks the total down per actually-served model, so a session that fell back or auto-routed across models shows where the spend went instead of one blended total under the configured model. Empty when the runner reported no per-model usage; AmountMicroUSD and PricingKnown are then the sole source of truth. | | `status.estimatedCost.byModel[].amountMicroUSD` | `integer (int64)` | AmountMicroUSD is this bucket's share of the session cost, in micro-USD (1e-6 USD): the provider-reported cost when available, else tokens × the resolved per-model pricing table rate. | @@ -279,9 +279,13 @@ Status is controller-owned (observed state). | `status.estimatedCost.byModel[].model` \* | `string` | Model is the served-model display id verbatim (provider/model, or a provider-prefixed served model under routing) — never re-parsed here. | | `status.estimatedCost.byModel[].outputTokens` | `integer (int64)` | OutputTokens is this model's share of completion tokens. | | `status.estimatedCost.byModel[].pricingKnown` | `boolean` | PricingKnown is true when AmountMicroUSD is a real cost — either the provider reported it directly, or the pricing table had a rate for this model. False means AmountMicroUSD is 0 and must not be shown as real spend (same no-fabrication contract as EstimatedSessionCost). | +| `status.estimatedCost.byTool` | `[]object` | ByTool breaks out the cost of inner interactive toolkits (e.g. a `claude` Claude Code sub-run's own provider-reported spend) a session drove. These are ADDED to AmountMicroUSD alongside ByModel — a session's total is model spend plus tool spend. Empty when no interactive toolkit reported a cost. Tool cost is always provider-reported, never table-priced, so a bucket's PricingKnown is always true. | +| `status.estimatedCost.byTool[].amountMicroUSD` | `integer (int64)` | AmountMicroUSD is this tool's provider-reported cost, summed across its invocations, in micro-USD (1e-6 USD). | +| `status.estimatedCost.byTool[].pricingKnown` | `boolean` | PricingKnown is true when AmountMicroUSD is a real cost. Tool cost is always provider-reported, so this is true whenever a cost was reported. | +| `status.estimatedCost.byTool[].tool` \* | `string` | Tool is the outer tool name (e.g. "claude-oauth"). | | `status.estimatedCost.currency` | `string` | Currency is the ISO code the amount is denominated in ("USD"). | | `status.estimatedCost.model` | `string` | Model is the configured model the session ran under. | -| `status.estimatedCost.pricingKnown` | `boolean` | PricingKnown is false when the model had no price; the amount is then 0 and must not be shown as a real cost. | +| `status.estimatedCost.pricingKnown` | `boolean` | PricingKnown is false when a served model had no price. The amount is then a lower bound (priced components only), not the full cost. Tool buckets are always priced, so this tracks model-pricing coverage specifically. | | `status.failureReason` | `string` | FailureReason is the human-readable reason phase became Failed. | | `status.finishedAt` | `string (date-time)` | FinishedAt is when the session reached a terminal phase. | | `status.identityChoiceParkedAt` | `string (date-time)` | IdentityChoiceParkedAt is when the session entered AwaitingIdentityChoice. The operator measures IdentityChoiceTimeout from here. | From 4e13c8c37d9ee2d767ee8c8311deec3eabc01b05 Mon Sep 17 00:00:00 2001 From: Joseph Schorr Date: Wed, 30 Sep 2026 12:08:56 -0400 Subject: [PATCH 2/6] Accumulate interactive toolkit cost per tool on the Loop --- pkg/agent/runner/loop_deps.go | 6 ++++ pkg/agent/runner/loop_usage.go | 55 ++++++++++++++++++++++++++++++ pkg/agent/runner/toolusage_test.go | 53 ++++++++++++++++++++++++++++ 3 files changed, 114 insertions(+) create mode 100644 pkg/agent/runner/toolusage_test.go diff --git a/pkg/agent/runner/loop_deps.go b/pkg/agent/runner/loop_deps.go index 4702077a..d0bd358f 100644 --- a/pkg/agent/runner/loop_deps.go +++ b/pkg/agent/runner/loop_deps.go @@ -504,6 +504,12 @@ type Loop struct { // accumulate into the existing bucket instead of appending a duplicate. usageByModel []modelUsageBucket usageByModelIdx map[string]int + // usageByTool is the per-interactive-toolkit running cost accumulation, in + // first-encountered order, keyed by outer tool name in usageByToolIdx. Also + // guarded by usageMu (see above) — addToolCost fires from the same terminal + // bookkeeping path as addUsage/addModelUsage. + usageByTool []toolUsageBucket + usageByToolIdx map[string]int // Engine is the runner's single authz dependency. Production wiring comes from // internal/cmd/runner/main.go; tests inject a fake or leave nil (falling back to the diff --git a/pkg/agent/runner/loop_usage.go b/pkg/agent/runner/loop_usage.go index f6a913be..8b56b22b 100644 --- a/pkg/agent/runner/loop_usage.go +++ b/pkg/agent/runner/loop_usage.go @@ -120,6 +120,61 @@ func (l *Loop) usageByModelSnapshot() []pipeline.ModelUsage { return out } +// toolUsageBucket is one interactive toolkit's running cost accumulation across +// its invocations this session, keyed by outer tool name. +type toolUsageBucket struct { + tool string + costMicroUSD int64 + costReported bool +} + +// addToolCost folds one interactive toolkit result event's provider-reported +// cost into the bucket for outerTool. Shares usageMu with addUsage/addModelUsage +// (see usageMu's doc). ok is the toolkit's success flag; v1 accumulates cost +// regardless — a run the provider billed cost money whether or not it succeeded, +// and an unbilled run reports 0 anyway. +func (l *Loop) addToolCost(outerTool string, costUSD float64, ok bool) { + l.usageMu.Lock() + defer l.usageMu.Unlock() + if l.usageByToolIdx == nil { + l.usageByToolIdx = make(map[string]int) + } + i, seen := l.usageByToolIdx[outerTool] + if !seen { + i = len(l.usageByTool) + l.usageByToolIdx[outerTool] = i + l.usageByTool = append(l.usageByTool, toolUsageBucket{tool: outerTool}) + } + b := &l.usageByTool[i] + if costUSD < 0 { + // A negative provider-reported tool cost would reduce the session total. + // Treat it as unreported rather than netting it against real charges — + // same guard as addModelUsage. + slog.Default().Info("negative reported tool cost; ignoring", + "tool", outerTool, "costUSD", costUSD) + return + } + // Round once at the USD->micro-USD boundary so accumulation is pure int64. + b.costMicroUSD += int64(math.Round(costUSD * 1e6)) + b.costReported = true + _ = ok +} + +// usageByToolSnapshot returns the Loop's per-tool cost as pipeline.ToolUsage in +// first-encountered order (nil when unused). Safe for concurrent use. +func (l *Loop) usageByToolSnapshot() []pipeline.ToolUsage { + l.usageMu.Lock() + defer l.usageMu.Unlock() + if len(l.usageByTool) == 0 { + return nil + } + out := make([]pipeline.ToolUsage, len(l.usageByTool)) + for i, b := range l.usageByTool { + out[i] = pipeline.ToolUsage{Tool: b.tool, CostMicroUSD: b.costMicroUSD, CostReported: b.costReported} + } + return out +} + // StampEstimatedCost persists the cost estimate via the StatusPatcher. nil-safe: // a Loop without a Status (tests/kubectl-driven) silently no-ops. func (l *Loop) StampEstimatedCost(ctx context.Context, c spiceboxv1alpha1.EstimatedSessionCost) error { diff --git a/pkg/agent/runner/toolusage_test.go b/pkg/agent/runner/toolusage_test.go new file mode 100644 index 00000000..4bd7817a --- /dev/null +++ b/pkg/agent/runner/toolusage_test.go @@ -0,0 +1,53 @@ +package runner + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestLoopAddToolCost_AccumulatesPerTool(t *testing.T) { + var l Loop + // codebot's shape: one tool ("claude-oauth") invoked many times -> ONE + // summed bucket; a second tool -> a distinct bucket, first-seen order. + l.addToolCost("claude-oauth", 5.43, true) + l.addToolCost("claude-oauth", 1.31, true) + l.addToolCost("aider", 2.10, true) + + got := l.usageByToolSnapshot() + require.Len(t, got, 2, "two distinct tools -> two buckets in first-encountered order") + + assert.Equal(t, "claude-oauth", got[0].Tool) + assert.True(t, got[0].CostReported) + assert.Equal(t, int64(6_740_000), got[0].CostMicroUSD, "5.43+1.31 USD -> 6,740,000 micro-USD") + + assert.Equal(t, "aider", got[1].Tool) + assert.Equal(t, int64(2_100_000), got[1].CostMicroUSD) +} + +func TestLoopAddToolCost_NegativeCostIgnored(t *testing.T) { + var l Loop + l.addToolCost("claude-oauth", -1.0, true) + + got := l.usageByToolSnapshot() + require.Len(t, got, 1, "the tool is still recorded") + assert.False(t, got[0].CostReported, "negative cost -> treated as unreported") + assert.Equal(t, int64(0), got[0].CostMicroUSD, "no negative amount is ever added") +} + +func TestLoopAddToolCost_ZeroCostUnbilled(t *testing.T) { + // An unbilled / credential-halt run reports cost 0; it adds 0 and does not + // mark the bucket as having a reported (billed) cost. + var l Loop + l.addToolCost("claude-oauth", 0, false) + + got := l.usageByToolSnapshot() + require.Len(t, got, 1) + assert.Equal(t, int64(0), got[0].CostMicroUSD) +} + +func TestLoopUsageByToolSnapshot_EmptyWhenUnused(t *testing.T) { + var l Loop + assert.Nil(t, l.usageByToolSnapshot(), "no tool runs -> nil, not a slice of zero buckets") +} From 02007584f9bdd589c230c5695296cef60b1c3cad Mon Sep 17 00:00:00 2001 From: Joseph Schorr Date: Wed, 30 Sep 2026 12:12:10 -0400 Subject: [PATCH 3/6] Feed interactive toolkit cost into SessionEnd via ByTool --- .../runner/interactive_event_publish_test.go | 34 +++++++++++++++++-- internal/cmd/runner/main.go | 12 ++++++- internal/cmd/runner/main_test.go | 2 +- pkg/agent/runner/loop_failure.go | 1 + pkg/agent/runner/loop_usage.go | 8 +++-- pkg/agent/runner/modelusage_test.go | 15 ++++++++ pkg/agent/runner/toolusage_test.go | 10 +++--- 7 files changed, 70 insertions(+), 12 deletions(-) diff --git a/internal/cmd/runner/interactive_event_publish_test.go b/internal/cmd/runner/interactive_event_publish_test.go index 543928fc..8fe536f7 100644 --- a/internal/cmd/runner/interactive_event_publish_test.go +++ b/internal/cmd/runner/interactive_event_publish_test.go @@ -38,7 +38,7 @@ func TestBuildToolSessionEventPublisher_PublishesKindToolSessionEvent(t *testing mem := memory.NewLocal(inmem.NewBackend()) scope := memory.Scope{Kind: "session", ID: "default/agent-1"} onEvent := buildToolSessionEventPublisher( - context.Background(), pub, mem, scope, "off", "default", "agent-1", nil) + context.Background(), pub, mem, scope, "off", "default", "agent-1", nil, nil) onEvent("tc-abc", "", "", toolkitstream.Event{Type: toolkitstream.EventTextDelta, Text: "hi"}) mu.Lock() @@ -60,7 +60,7 @@ func TestBuildToolSessionEventPublisher_CarriesReasonAndOuterTool(t *testing.T) return nil } emit := buildToolSessionEventPublisher(context.Background(), pub, - nil /*mem*/, memory.Scope{} /*scope*/, spiceboxv1alpha1.ToolSessionLogOff /*no persist*/, "ns", "sess", nil) + nil /*mem*/, memory.Scope{} /*scope*/, spiceboxv1alpha1.ToolSessionLogOff /*no persist*/, "ns", "sess", nil, nil) emit("tc-1", "Have claude write the README", "claude", toolkitstream.Event{Type: toolkitstream.EventTextDelta, Text: "hi"}) @@ -91,7 +91,7 @@ func TestBuildToolSessionEventPublisher_AllFieldsRoundTrip(t *testing.T) { mem := memory.NewLocal(inmem.NewBackend()) scope := memory.Scope{Kind: "session", ID: "default/agent-1"} onEvent := buildToolSessionEventPublisher( - context.Background(), pub, mem, scope, "off", "default", "agent-1", nil) + context.Background(), pub, mem, scope, "off", "default", "agent-1", nil, nil) onEvent("tc-1", "", "", toolkitstream.Event{ Type: toolkitstream.EventToolUseStop, @@ -110,3 +110,31 @@ func TestBuildToolSessionEventPublisher_AllFieldsRoundTrip(t *testing.T) { assert.False(t, pl.OK) assert.Equal(t, "exit code 1", pl.Summary) } + +func TestBuildToolSessionEventPublisher_InvokesOnResultForResultEvent(t *testing.T) { + type call struct { + tool string + cost float64 + ok bool + } + var got []call + onResult := func(tool string, cost float64, ok bool) { + got = append(got, call{tool, cost, ok}) + } + pub := func(_ context.Context, _ string, _ []byte) error { return nil } + + emit := buildToolSessionEventPublisher(context.Background(), pub, + nil, memory.Scope{}, spiceboxv1alpha1.ToolSessionLogOff, "ns", "sess", nil, onResult) + + // A non-result event must NOT trigger onResult. + emit("tc-1", "reason", "claude-oauth", + toolkitstream.Event{Type: toolkitstream.EventTextDelta, Text: "hi"}) + // A result event MUST trigger it once, carrying tool/cost/ok. + emit("tc-1", "reason", "claude-oauth", + toolkitstream.Event{Type: toolkitstream.EventResult, OK: true, CostUSD: 5.43}) + + require.Len(t, got, 1, "onResult fires only on EventResult") + assert.Equal(t, "claude-oauth", got[0].tool) + assert.Equal(t, 5.43, got[0].cost) + assert.True(t, got[0].ok) +} diff --git a/internal/cmd/runner/main.go b/internal/cmd/runner/main.go index 0d1968e8..eb956d13 100644 --- a/internal/cmd/runner/main.go +++ b/internal/cmd/runner/main.go @@ -2984,7 +2984,8 @@ func run(cfg *config) error { } }, OnEvent: buildToolSessionEventPublisher( - rootCtx, pub, memSigned, scope, class.Spec.ToolSessionLog, hooksNS, hooksName, envSigner), + rootCtx, pub, memSigned, scope, class.Spec.ToolSessionLog, hooksNS, hooksName, envSigner, + loop.AddToolCost), Register: toolSessionReg.register, } } @@ -3241,6 +3242,7 @@ func buildToolSessionEventPublisher( logMode string, ns, name string, signer *channelevents.EnvelopeSigner, + onResult func(outerTool string, costUSD float64, ok bool), ) func(toolCallRef, reason, outerTool string, ev toolkitstream.Event) { return func(toolCallRef, reason, outerTool string, ev toolkitstream.Event) { // NATS -> channelsd -> Slack — unchanged, always runs. @@ -3287,6 +3289,14 @@ func buildToolSessionEventPublisher( "toolCallRef", toolCallRef, "eventType", string(ev.Type)) } + + // Fold this interactive toolkit's own provider-reported cost into the + // session total. Fires on the terminal result event only, and regardless + // of the ToolSessionLog persist gate above — accumulation must not depend + // on logging being on. + if ev.Type == toolkitstream.EventResult && onResult != nil { + onResult(outerTool, ev.CostUSD, ev.OK) + } } } diff --git a/internal/cmd/runner/main_test.go b/internal/cmd/runner/main_test.go index ab392ce4..1a03ab39 100644 --- a/internal/cmd/runner/main_test.go +++ b/internal/cmd/runner/main_test.go @@ -181,7 +181,7 @@ func TestBuildToolSessionEventPublisher_Persist(t *testing.T) { noopPub := func(context.Context, string, []byte) error { return nil } onEvent := buildToolSessionEventPublisher( - ctx, noopPub, mem, scope, tc.mode, "default", "sess1", nil) + ctx, noopPub, mem, scope, tc.mode, "default", "sess1", nil, nil) onEvent("ref-1", "", "", toolkitstream.Event{ Type: toolkitstream.EventToolUseStart, ToolName: "Edit", ToolID: "t1", }) diff --git a/pkg/agent/runner/loop_failure.go b/pkg/agent/runner/loop_failure.go index 5725f2ef..c2713bf7 100644 --- a/pkg/agent/runner/loop_failure.go +++ b/pkg/agent/runner/loop_failure.go @@ -111,6 +111,7 @@ func (l *Loop) fireSessionEnd(ctx context.Context, reason string) { CacheCreationTokens: snap.CacheCreationTokens, CacheReadTokens: snap.CacheReadTokens, ByModel: l.usageByModelSnapshot(), + ByTool: l.usageByToolSnapshot(), }, }, host); err != nil { slog.Default().Info("fireSessionEnd: SessionEnd executor errored (best-effort)", diff --git a/pkg/agent/runner/loop_usage.go b/pkg/agent/runner/loop_usage.go index 8b56b22b..f75438f2 100644 --- a/pkg/agent/runner/loop_usage.go +++ b/pkg/agent/runner/loop_usage.go @@ -128,12 +128,16 @@ type toolUsageBucket struct { costReported bool } -// addToolCost folds one interactive toolkit result event's provider-reported +// AddToolCost folds one interactive toolkit result event's provider-reported // cost into the bucket for outerTool. Shares usageMu with addUsage/addModelUsage // (see usageMu's doc). ok is the toolkit's success flag; v1 accumulates cost // regardless — a run the provider billed cost money whether or not it succeeded, // and an unbilled run reports 0 anyway. -func (l *Loop) addToolCost(outerTool string, costUSD float64, ok bool) { +// +// Exported (unlike addModelUsage) because it is wired from internal/cmd/runner's +// interactive-tool result callback, which lives in a different package; the +// runner's turn loop feeds addModelUsage internally, so that one stays private. +func (l *Loop) AddToolCost(outerTool string, costUSD float64, ok bool) { l.usageMu.Lock() defer l.usageMu.Unlock() if l.usageByToolIdx == nil { diff --git a/pkg/agent/runner/modelusage_test.go b/pkg/agent/runner/modelusage_test.go index 2cf3138d..85e36397 100644 --- a/pkg/agent/runner/modelusage_test.go +++ b/pkg/agent/runner/modelusage_test.go @@ -89,3 +89,18 @@ func TestFireSessionEnd_PopulatesByModel(t *testing.T) { assert.True(t, b.CostReported) assert.Equal(t, int64(50_000), b.ReportedCostMicroUSD) } + +func TestFireSessionEnd_PopulatesByTool(t *testing.T) { + cr := &endCapturingRunner{} + l := &Loop{Model: "claude-sonnet-5"} + l.pipelineExec = cr + l.pipelineOnce.Do(func() {}) + + l.AddToolCost("claude-oauth", 5.43, true) + l.fireSessionEnd(context.Background(), "completed") + + require.NotNil(t, cr.in.End) + require.Len(t, cr.in.End.ByTool, 1) + assert.Equal(t, "claude-oauth", cr.in.End.ByTool[0].Tool) + assert.Equal(t, int64(5_430_000), cr.in.End.ByTool[0].CostMicroUSD) +} diff --git a/pkg/agent/runner/toolusage_test.go b/pkg/agent/runner/toolusage_test.go index 4bd7817a..7673b515 100644 --- a/pkg/agent/runner/toolusage_test.go +++ b/pkg/agent/runner/toolusage_test.go @@ -11,9 +11,9 @@ func TestLoopAddToolCost_AccumulatesPerTool(t *testing.T) { var l Loop // codebot's shape: one tool ("claude-oauth") invoked many times -> ONE // summed bucket; a second tool -> a distinct bucket, first-seen order. - l.addToolCost("claude-oauth", 5.43, true) - l.addToolCost("claude-oauth", 1.31, true) - l.addToolCost("aider", 2.10, true) + l.AddToolCost("claude-oauth", 5.43, true) + l.AddToolCost("claude-oauth", 1.31, true) + l.AddToolCost("aider", 2.10, true) got := l.usageByToolSnapshot() require.Len(t, got, 2, "two distinct tools -> two buckets in first-encountered order") @@ -28,7 +28,7 @@ func TestLoopAddToolCost_AccumulatesPerTool(t *testing.T) { func TestLoopAddToolCost_NegativeCostIgnored(t *testing.T) { var l Loop - l.addToolCost("claude-oauth", -1.0, true) + l.AddToolCost("claude-oauth", -1.0, true) got := l.usageByToolSnapshot() require.Len(t, got, 1, "the tool is still recorded") @@ -40,7 +40,7 @@ func TestLoopAddToolCost_ZeroCostUnbilled(t *testing.T) { // An unbilled / credential-halt run reports cost 0; it adds 0 and does not // mark the bucket as having a reported (billed) cost. var l Loop - l.addToolCost("claude-oauth", 0, false) + l.AddToolCost("claude-oauth", 0, false) got := l.usageByToolSnapshot() require.Len(t, got, 1) From 240017a8dba0548b2c5f9d031964a01e5f0ea347 Mon Sep 17 00:00:00 2001 From: Joseph Schorr Date: Wed, 30 Sep 2026 12:13:30 -0400 Subject: [PATCH 4/6] Include interactive toolkit cost in the session total and notice --- pkg/agent/postsession/cost/cost.go | 54 ++++++++++++++++++++----- pkg/agent/postsession/cost/cost_test.go | 43 ++++++++++++++++++++ 2 files changed, 88 insertions(+), 9 deletions(-) diff --git a/pkg/agent/postsession/cost/cost.go b/pkg/agent/postsession/cost/cost.go index 1c6ff3f5..339f2068 100644 --- a/pkg/agent/postsession/cost/cost.go +++ b/pkg/agent/postsession/cost/cost.go @@ -75,13 +75,21 @@ func (r *Reporter) Eval(ctx context.Context, in pipeline.Input) pipeline.Decisio micro = microUSD(e, price) } + // Interactive toolkit cost (e.g. an inner `claude` sub-run's own billed + // spend) is provider-reported, so it is always "known" and is ADDED to the + // model total. When a served model is unpriced (known=false, micro=0), the + // grand total is still the priced tool component — a lower bound, per the + // EstimatedSessionCost.AmountMicroUSD contract. + toolBuckets, toolTotal := buildToolBuckets(e.ByTool) + cost := v1.EstimatedSessionCost{ - AmountMicroUSD: micro, + AmountMicroUSD: micro + toolTotal, Currency: currency, Model: e.Model, PricingKnown: known, AsOf: r.d.Now(), ByModel: buckets, + ByTool: toolBuckets, } if known && cost.Currency == "" { cost.Currency = "USD" @@ -165,6 +173,22 @@ func (r *Reporter) buildCostBuckets(usages []pipeline.ModelUsage) (buckets []v1. return buckets, total, allKnown, currency } +// buildToolBuckets maps each interactive-toolkit usage bucket to a priced +// ToolCostBucket. Tool cost is always provider-reported (already micro-USD and +// "known"), so there is no table lookup here — this is a pure projection plus a +// sum. Returns the buckets in caller order and their total. +func buildToolBuckets(usages []pipeline.ToolUsage) (buckets []v1.ToolCostBucket, total int64) { + if len(usages) == 0 { + return nil, 0 + } + buckets = make([]v1.ToolCostBucket, len(usages)) + for i, u := range usages { + buckets[i] = v1.ToolCostBucket{Tool: u.Tool, AmountMicroUSD: u.CostMicroUSD, PricingKnown: u.CostReported} + total += u.CostMicroUSD + } + return buckets, total +} + func costNotice(e *pipeline.SessionEndInfo, c v1.EstimatedSessionCost, known bool) *notice.Notice { if !known { return notice.New(categories.SessionCost, notice.Args{ @@ -177,15 +201,27 @@ func costNotice(e *pipeline.SessionEndInfo, c v1.EstimatedSessionCost, known boo Audience: channelevents.InteractionAudience{Scope: channelevents.AudienceParticipants}, }) } + fields := []channelevents.InteractionField{ + {Label: "Tokens", Value: fmt.Sprintf("%s in / %s out", + formatTokens(e.InputTokens), formatTokens(e.OutputTokens))}, + {Label: "Cache", Value: fmt.Sprintf("%s read / %s write", + formatTokens(e.CacheReadTokens), formatTokens(e.CacheCreationTokens))}, + {Label: "Model", Value: e.Model}, + } + // Break out inner interactive-toolkit spend (e.g. a passthrough `claude` + // sub-run) so the total's provenance is visible, not folded silently. + var toolTotal int64 + for _, b := range c.ByTool { + toolTotal += b.AmountMicroUSD + } + if toolTotal > 0 { + fields = append(fields, channelevents.InteractionField{ + Label: "Sub-agent tools", Value: formatUSD(toolTotal), + }) + } return notice.New(categories.SessionCost, notice.Args{ - Lead: fmt.Sprintf("This session cost ~%s", formatUSD(c.AmountMicroUSD)), - Fields: []channelevents.InteractionField{ - {Label: "Tokens", Value: fmt.Sprintf("%s in / %s out", - formatTokens(e.InputTokens), formatTokens(e.OutputTokens))}, - {Label: "Cache", Value: fmt.Sprintf("%s read / %s write", - formatTokens(e.CacheReadTokens), formatTokens(e.CacheCreationTokens))}, - {Label: "Model", Value: e.Model}, - }, + Lead: fmt.Sprintf("This session cost ~%s", formatUSD(c.AmountMicroUSD)), + Fields: fields, Audience: channelevents.InteractionAudience{Scope: channelevents.AudienceParticipants}, }) } diff --git a/pkg/agent/postsession/cost/cost_test.go b/pkg/agent/postsession/cost/cost_test.go index 81173827..35d62099 100644 --- a/pkg/agent/postsession/cost/cost_test.go +++ b/pkg/agent/postsession/cost/cost_test.go @@ -163,6 +163,49 @@ func TestEval_ByModel_ReportedPreferredOverEstimated(t *testing.T) { assert.True(t, stamped.PricingKnown, "every bucket priced -> session PricingKnown") } +func TestEval_ByTool_AddsToGrandTotalAndItemizes(t *testing.T) { + r, stamped := newWithCapturePricer(multiModelPricer) + in := endInput("completed", 0, 0, 0, 0) + in.End.ByModel = []pipeline.ModelUsage{ + {Display: "anthropic/claude-3.5-sonnet", Model: "anthropic/claude-3.5-sonnet", + InputTokens: 1_000_000, OutputTokens: 0}, // 1M in × $3 = $3.00 -> 3_000_000 micro + } + in.End.ByTool = []pipeline.ToolUsage{{Tool: "claude-oauth", CostMicroUSD: 19_020_000, CostReported: true}} + r.Eval(context.Background(), in) + + require.Len(t, stamped.ByTool, 1) + assert.Equal(t, "claude-oauth", stamped.ByTool[0].Tool) + assert.Equal(t, int64(19_020_000), stamped.ByTool[0].AmountMicroUSD) + assert.True(t, stamped.ByTool[0].PricingKnown) + assert.Equal(t, int64(3_000_000+19_020_000), stamped.AmountMicroUSD, "grand total = model + tool") + assert.True(t, stamped.PricingKnown, "model priced + tool always priced") +} + +func TestEval_NoTool_IdenticalToToday(t *testing.T) { + r, stamped := newWithCapturePricer(multiModelPricer) + in := endInput("completed", 0, 0, 0, 0) + in.End.ByModel = []pipeline.ModelUsage{ + {Display: "anthropic/claude-3.5-sonnet", Model: "anthropic/claude-3.5-sonnet", InputTokens: 1_000_000}, + } + r.Eval(context.Background(), in) + + assert.Nil(t, stamped.ByTool, "no tool -> no byTool") + assert.Equal(t, int64(3_000_000), stamped.AmountMicroUSD, "unchanged from today") +} + +func TestEval_UnknownModel_KnownTool_LowerBound(t *testing.T) { + r, stamped := newWithCapturePricer(multiModelPricer) // "mystery-model" unpriced + in := endInput("completed", 0, 0, 0, 0) + in.End.ByModel = []pipeline.ModelUsage{ + {Display: "openrouter/mystery/model", Model: "mystery-model", InputTokens: 1_000_000}, + } + in.End.ByTool = []pipeline.ToolUsage{{Tool: "claude-oauth", CostMicroUSD: 19_020_000, CostReported: true}} + r.Eval(context.Background(), in) + + assert.False(t, stamped.PricingKnown, "an unpriced model makes the total a lower bound") + assert.Equal(t, int64(19_020_000), stamped.AmountMicroUSD, "nonzero: the priced (tool) component") +} + func TestEval_ByModel_UnknownPriceBucket_ZeroAndUnknown(t *testing.T) { r, stamped := newWithCapturePricer(multiModelPricer) in := endInput("completed", 0, 0, 0, 0) From a13da7e450355321734397283758601539c504a7 Mon Sep 17 00:00:00 2001 From: Joseph Schorr Date: Wed, 30 Sep 2026 12:15:43 -0400 Subject: [PATCH 5/6] Surface interactive toolkit cost in admind budget and overview --- pkg/web/admind/aggregator.go | 6 +++++ pkg/web/admind/budget.go | 12 +++++++++- pkg/web/admind/budget_internal_test.go | 16 +++++++++++++ pkg/web/admind/budget_test.go | 32 ++++++++++++++++++++++++++ pkg/web/admind/handlers.go | 2 +- 5 files changed, 66 insertions(+), 2 deletions(-) create mode 100644 pkg/web/admind/budget_internal_test.go diff --git a/pkg/web/admind/aggregator.go b/pkg/web/admind/aggregator.go index 96de875c..2184b5d9 100644 --- a/pkg/web/admind/aggregator.go +++ b/pkg/web/admind/aggregator.go @@ -54,6 +54,10 @@ type SessionState struct { // yet, in which case the Overview/Budget aggregators attribute this // session's InputTokens/OutputTokens/Model as a single blended row. ByModel []spiceboxv1alpha1.ModelCostBucket `json:"byModel,omitempty"` + // ByTool is the session's per-interactive-toolkit cost breakdown + // (status.estimatedCost.byTool). Added to the token-repriced estimate in the + // budget/overview panels so admind totals include inner sub-agent spend. + ByTool []spiceboxv1alpha1.ToolCostBucket `json:"byTool,omitempty"` // BundleSessions mirrors AgentSession.status.bundleSessions — one entry per // tool bundle this session resolved to a SpiceboxSession, carried straight // through from the watched CRD (no new fetch). A bundle's sandbox backend @@ -362,8 +366,10 @@ func (a *Aggregator) UpsertSession(s *spiceboxv1alpha1.AgentSession) { // later Snapshot()/sortedModels-style sort-in-place elsewhere on the // SAME backing array would race the informer's own reads of it. st.ByModel = append([]spiceboxv1alpha1.ModelCostBucket(nil), ec.ByModel...) + st.ByTool = append([]spiceboxv1alpha1.ToolCostBucket(nil), ec.ByTool...) } else { st.ByModel = nil + st.ByTool = nil } // Defensive copy for the same reason as ByModel above: s.Status.BundleSessions // backs onto the informer cache's object. diff --git a/pkg/web/admind/budget.go b/pkg/web/admind/budget.go index f670229a..81352853 100644 --- a/pkg/web/admind/budget.go +++ b/pkg/web/admind/budget.go @@ -215,7 +215,7 @@ func (a *Admind) handleBudget(w http.ResponseWriter, r *http.Request) { // so it must be stripped of its provider prefix before pricing. bs := budgetSession{ State: s, - Est: prices.Estimate(bareModel(s.Model), s.InputTokens, s.OutputTokens), + Est: prices.Estimate(bareModel(s.Model), s.InputTokens, s.OutputTokens) + toolCostUSD(s.ByTool), Starters: starters, } for _, d := range budgetDimensions { @@ -259,6 +259,16 @@ func bucketCostUSD(b spiceboxv1alpha1.ModelCostBucket) cost.USD { return cost.USD(float64(b.AmountMicroUSD) / 1e6) } +// toolCostUSD sums a session's provider-reported interactive-toolkit spend. +// Nil/empty -> 0. Tool cost is always priced, so this never yields NaN. +func toolCostUSD(buckets []spiceboxv1alpha1.ToolCostBucket) cost.USD { + var micro int64 + for _, b := range buckets { + micro += b.AmountMicroUSD + } + return cost.USD(float64(micro) / 1e6) +} + // sortedBudgetRows flattens the accumulator into a non-nil slice ordered by // estimated cost desc, key asc as the tie-break — a stable dashboard order and // a JSON [] (never null) when empty. diff --git a/pkg/web/admind/budget_internal_test.go b/pkg/web/admind/budget_internal_test.go new file mode 100644 index 00000000..f231ada4 --- /dev/null +++ b/pkg/web/admind/budget_internal_test.go @@ -0,0 +1,16 @@ +package admind + +import ( + "testing" + + "github.com/stretchr/testify/assert" + + spiceboxv1alpha1 "github.com/authzed/openagentprimitives/pkg/apis/v1alpha1" +) + +func TestToolCostUSD_NilByToolContributesZero(t *testing.T) { + assert.Equal(t, 0.0, float64(toolCostUSD(nil)), "nil/empty -> 0") + assert.InDelta(t, 19.02, float64(toolCostUSD([]spiceboxv1alpha1.ToolCostBucket{ + {Tool: "claude-oauth", AmountMicroUSD: 19_020_000, PricingKnown: true}, + })), 1e-9) +} diff --git a/pkg/web/admind/budget_test.go b/pkg/web/admind/budget_test.go index abb4ce2a..9742b43b 100644 --- a/pkg/web/admind/budget_test.go +++ b/pkg/web/admind/budget_test.go @@ -422,3 +422,35 @@ func TestAdmindBudget_EmptyIsNonNilSlices(t *testing.T) { assert.Contains(t, body, "\""+axis+"\":[]", axis+" must be an empty array, not null") } } + +func TestAdmindBudget_ByToolAddsToEstimate(t *testing.T) { + s := &spiceboxv1alpha1.AgentSession{} + s.Namespace, s.Name = "default", "s-codebot" + s.Spec.Class = "codebot" + s.Status.Phase = "Completed" + s.Status.StartedAt = &metav1.Time{Time: time.Date(2026, 6, 1, 12, 0, 0, 0, time.UTC)} + s.Status.EffectiveSettings = &spiceboxv1alpha1.EffectiveSettings{ + Model: spiceboxv1alpha1.ModelConfig{Provider: "anthropic", Name: "claude-opus-4-8"}, + } + // Zero tokens -> $0 model estimate, isolating the tool-cost contribution. + s.Status.Progress = &spiceboxv1alpha1.AgentSessionProgress{InputTokens: 0, OutputTokens: 0} + s.Status.EstimatedCost = &spiceboxv1alpha1.EstimatedSessionCost{ + ByTool: []spiceboxv1alpha1.ToolCostBucket{ + {Tool: "claude-oauth", AmountMicroUSD: 19_020_000, PricingKnown: true}, + }, + } + + k8s := fake.NewClientBuilder().WithScheme(newScheme(t)).WithObjects(s).Build() + a := newTestAdmind(t, k8s) + a.Aggregator().UpsertSession(s) + + w := do(t, a.Handler(), http.MethodGet, "/admin/v1/budget", "test-token", "user:YWRtaW4", "") + require.Equal(t, http.StatusOK, w.Code) + var bd admind.BudgetBreakdown + require.NoError(t, json.Unmarshal(w.Body.Bytes(), &bd)) + + sess := rowsByKey(bd.BySession) + require.Contains(t, sess, "default/s-codebot") + assert.InDelta(t, 19.02, float64(sess["default/s-codebot"].EstimatedCostUSD), 1e-9, + "$0 model (0 tokens) + $19.02 tool cost") +} diff --git a/pkg/web/admind/handlers.go b/pkg/web/admind/handlers.go index c7c4cfaa..4f612ec8 100644 --- a/pkg/web/admind/handlers.go +++ b/pkg/web/admind/handlers.go @@ -670,7 +670,7 @@ func (a *Admind) enrichSessionRows(r *http.Request, rows []audit.EntityRow) { rows[i].OutputTokens = st.OutputTokens // st.Model is the uniform "/" display id; strip the // provider prefix before the bare-keyed price lookup. - rows[i].EstimatedCostUSD = prices.Estimate(bareModel(st.Model), st.InputTokens, st.OutputTokens) + rows[i].EstimatedCostUSD = prices.Estimate(bareModel(st.Model), st.InputTokens, st.OutputTokens) + toolCostUSD(st.ByTool) } if m, ok := meta[key]; ok { rows[i].StartedBy = m.StartedBy From e0f36f840b2d3e958106aefebb3b1ef0348ac9e7 Mon Sep 17 00:00:00 2001 From: Joseph Schorr Date: Wed, 30 Sep 2026 13:10:22 -0400 Subject: [PATCH 6/6] Include interactive toolkit cost in admind Overview spend panel --- pkg/web/admind/handler_test.go | 32 ++++++++++++++++++++++++++++++++ pkg/web/admind/handlers.go | 9 +++++++++ 2 files changed, 41 insertions(+) diff --git a/pkg/web/admind/handler_test.go b/pkg/web/admind/handler_test.go index 221248af..ee74761b 100644 --- a/pkg/web/admind/handler_test.go +++ b/pkg/web/admind/handler_test.go @@ -226,6 +226,38 @@ func TestAdmindAuditEndpoints(t *testing.T) { assert.Equal(t, http.StatusBadRequest, w.Code) } +func TestAdmindOverview_IncludesToolCost(t *testing.T) { + s := &spiceboxv1alpha1.AgentSession{} + s.Namespace, s.Name = "default", "s-codebot" + s.Spec.Class = "codebot" + s.Status.Phase = "Completed" + s.Status.EffectiveSettings = &spiceboxv1alpha1.EffectiveSettings{ + Model: spiceboxv1alpha1.ModelConfig{Provider: "anthropic", Name: "claude-opus-4-8"}, + } + // Zero tokens -> $0 model estimate, isolating the tool-cost contribution. + s.Status.Progress = &spiceboxv1alpha1.AgentSessionProgress{InputTokens: 0, OutputTokens: 0} + s.Status.EstimatedCost = &spiceboxv1alpha1.EstimatedSessionCost{ + ByTool: []spiceboxv1alpha1.ToolCostBucket{ + {Tool: "claude-oauth", AmountMicroUSD: 19_020_000, PricingKnown: true}, + }, + } + + k8s := fake.NewClientBuilder().WithScheme(newScheme(t)).WithObjects(s).Build() + a := newTestAdmind(t, k8s) + a.Aggregator().UpsertSession(s) + + w := do(t, a.Handler(), http.MethodGet, "/admin/v1/overview", "test-token", "user:YWRtaW4", "") + require.Equal(t, http.StatusOK, w.Code) + var ov struct { + Budget struct { + EstimatedCostUSD float64 `json:"estimatedCostUSD"` + } `json:"budget"` + } + require.NoError(t, json.Unmarshal(w.Body.Bytes(), &ov)) + assert.InDelta(t, 19.02, ov.Budget.EstimatedCostUSD, 1e-9, + "Overview headline spend must include inner tool cost ($0 model + $19.02 tool)") +} + func TestAdmindOverviewAndHealth(t *testing.T) { sess := sessionCR("default", "s1", "support-bot", "Running") k8s := fake.NewClientBuilder().WithScheme(newScheme(t)).WithObjects(sess).Build() diff --git a/pkg/web/admind/handlers.go b/pkg/web/admind/handlers.go index 4f612ec8..ecfcd3c3 100644 --- a/pkg/web/admind/handlers.go +++ b/pkg/web/admind/handlers.go @@ -537,6 +537,15 @@ func (a *Admind) handleOverview(w http.ResponseWriter, r *http.Request) { estCost += prices.Estimate(bareModel(m.Model), m.InputTokens, m.OutputTokens) tokensSpent += m.InputTokens + m.OutputTokens } + // Inner interactive-toolkit spend (e.g. a passthrough `claude` sub-run) is + // not token-based, so it never appears in ov.ByModel. Add it over the SAME + // session scope that produced ov.ByModel — the overview engine's own live + // input is liveSessionSnapshot(a.agg), so a.agg.Snapshot() is exactly that + // set (no double-count). Without this the headline spend understates a + // sub-agent session by the full tool cost (see the codebot case). + for _, s := range a.agg.Snapshot() { + estCost += toolCostUSD(s.ByTool) + } writeJSON(w, http.StatusOK, overviewResponse{ Overview: ov, Budget: budgetInfo{TokensSpent: tokensSpent, EstimatedCostUSD: estCost, Estimated: true},