Skip to content

Commit 22c9f80

Browse files
committed
fix(cli): count Claude advisor_message iteration tokens
`message.usage.iterations[]` breaks an assistant message into the calls that produced it. Entries of type `message` are the main model's own iterations and are already summed into the enclosing usage — verified on 201 real lines locally, where their counts match it exactly. Entries of type `advisor_message` are a different model's call that the enclosing usage does not carry, so their tokens were dropped entirely and the model billing at its own rate never showed up. Emit one model.usage per advisor iteration, attributed to the advisor's model. Each needs a refs.sourceId: the import key is built from (source, file, line, type, sourceId), so without one every usage event on the line would collapse into the main model's id. The main event deliberately keeps no sourceId so its id stays stable across re-imports. Claude-Session: https://claude.ai/code/session_019u3RKiNJY1sRknveHJy9iJ
1 parent b575203 commit 22c9f80

2 files changed

Lines changed: 191 additions & 1 deletion

File tree

‎packages/cli/src/adapters/claude-code.ts‎

Lines changed: 58 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -268,6 +268,26 @@ async function parseClaudeCodeSessionFile(
268268
}), lineNumber, topType, 'usage')
269269
}
270270

271+
// Advisor calls bill at their own model's rate and are not part of the usage
272+
// above, so each becomes its own model.usage event. They need a sourceId: the
273+
// import key is built from (source, file, line, type, sourceId), and without
274+
// one every usage event on this line would collapse into the main model's id.
275+
// The main event deliberately keeps no sourceId so its id stays stable.
276+
for (const [advisorIndex, advisor] of claudeAdvisorUsages(message).entries()) {
277+
push(baseClaudeEvent({
278+
ts,
279+
type: 'model.usage',
280+
sessionId,
281+
turnId: state.currentTurnId,
282+
cwd,
283+
project,
284+
model: advisor.model,
285+
confidence: 'partial',
286+
metrics: advisor.usage,
287+
refs: stringRefs({ sourceId: `advisor:${advisorIndex}` }),
288+
}), lineNumber, topType, 'usage')
289+
}
290+
271291
for (const toolUse of claudeToolUseItems(message)) {
272292
const tool = stringField(toolUse, 'name') || 'tool'
273293
const toolUseId = stringField(toolUse, 'id') || `tool_${createStableHash([filePath, lineNumber, tool]).slice(0, 24)}`
@@ -373,7 +393,44 @@ function baseClaudeEvent(
373393
}
374394

375395
export function claudeUsageFromMessage(message: Record<string, unknown>): Partial<MetricBag> | undefined {
376-
const usage = objectField(message, 'usage')
396+
return claudeUsageFromUsage(objectField(message, 'usage'))
397+
}
398+
399+
/**
400+
* Advisor-model calls folded into one assistant message.
401+
*
402+
* `message.usage.iterations[]` breaks the message down into the calls that
403+
* produced it. Entries of type `message` are the main model's own iterations and
404+
* are already summed into the enclosing `usage`, so counting them again would
405+
* double the message. `advisor_message` entries are a *different* model's call
406+
* whose tokens the enclosing usage does not carry — dropping them loses those
407+
* tokens entirely and hides a model that bills at its own rate. Mirrors ccusage
408+
* advisor_usages_from_line (adapter/claude/mod.rs, #1423).
409+
*/
410+
function claudeAdvisorUsages(
411+
message: Record<string, unknown>,
412+
): { model: string, usage: Partial<MetricBag> }[] {
413+
const iterations = objectField(message, 'usage').iterations
414+
if (!Array.isArray(iterations)) {
415+
return []
416+
}
417+
const advisors: { model: string, usage: Partial<MetricBag> }[] = []
418+
for (const iteration of iterations) {
419+
if (!isPlainObject(iteration) || stringField(iteration, 'type') !== 'advisor_message') {
420+
continue
421+
}
422+
const model = stringField(iteration, 'model')
423+
const usage = claudeUsageFromUsage(iteration)
424+
if (model && usage) {
425+
advisors.push({ model, usage })
426+
}
427+
}
428+
return advisors
429+
}
430+
431+
// Token counts live in the same field layout whether they come from
432+
// `message.usage` or from one of its `iterations[]` entries.
433+
function claudeUsageFromUsage(usage: Record<string, unknown>): Partial<MetricBag> | undefined {
377434
if (Object.keys(usage).length === 0) {
378435
return undefined
379436
}

‎packages/cli/test/claude-code.test.ts‎

Lines changed: 133 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -134,3 +134,136 @@ test('divergence: claude does NOT drop sidechain replay with a new requestId', a
134134
assert.equal(usages.length, 2)
135135
assert.equal(totalCacheRead, 20 + 50_000)
136136
})
137+
138+
// ── advisor-model iterations ──
139+
//
140+
// `message.usage.iterations[]` breaks an assistant message into the calls that
141+
// produced it. Main-model (`type: "message"`) entries are already summed into the
142+
// enclosing usage; `advisor_message` entries are a different model's call whose
143+
// tokens the enclosing usage does not carry. Mirrors ccusage #1423.
144+
145+
test('parity: ccusage claude calculates_advisor_cost_with_the_advisor_model', async () => {
146+
// Fixture from ccusage adapter/claude/mod.rs
147+
// calculates_advisor_cost_with_the_advisor_model.
148+
const usages = usageEvents(await parse([
149+
assistant('req-parent', 'msg-parent', {
150+
input_tokens: 1,
151+
output_tokens: 2,
152+
iterations: [{
153+
type: 'advisor_message',
154+
model: 'advisor-model',
155+
input_tokens: 10,
156+
output_tokens: 2,
157+
cache_creation_input_tokens: 0,
158+
cache_read_input_tokens: 0,
159+
}],
160+
} as unknown as Record<string, number>),
161+
]))
162+
163+
assert.equal(usages.length, 2)
164+
// The main model's totals are untouched.
165+
assert.equal(usages[0].model, 'claude-sonnet-4-20250514')
166+
assert.equal(usages[0].metrics?.tokensInput, 1)
167+
assert.equal(usages[0].metrics?.tokensOutput, 2)
168+
// The advisor's tokens are attributed to the advisor's own model.
169+
assert.equal(usages[1].model, 'advisor-model')
170+
assert.equal(usages[1].metrics?.tokensInput, 10)
171+
assert.equal(usages[1].metrics?.tokensOutput, 2)
172+
assert.equal(usages[1].metrics?.modelCalls, 1)
173+
// Distinct import identities, or the server would treat one as a repeat of the other.
174+
assert.notEqual(usages[0].id, usages[1].id)
175+
assert.notEqual(usages[0].refs?.importKey, usages[1].refs?.importKey)
176+
})
177+
178+
test('main-model iterations are never counted twice', async () => {
179+
// Real transcripts carry a `type: "message"` iteration whose counts equal the
180+
// enclosing usage exactly (verified across 201 such lines locally). Expanding
181+
// those would double every assistant message. Today those entries carry no
182+
// `model`, but the type check — not the missing field — has to be what stops
183+
// them, so this fixture supplies a model the filter must ignore.
184+
const usages = usageEvents(await parse([
185+
assistant('req-1', 'msg-1', {
186+
input_tokens: 2,
187+
output_tokens: 234,
188+
cache_creation_input_tokens: 29_586,
189+
cache_read_input_tokens: 0,
190+
iterations: [{
191+
type: 'message',
192+
model: 'claude-sonnet-4-20250514',
193+
input_tokens: 2,
194+
output_tokens: 234,
195+
cache_creation_input_tokens: 29_586,
196+
cache_read_input_tokens: 0,
197+
}],
198+
} as unknown as Record<string, number>),
199+
]))
200+
201+
assert.equal(usages.length, 1)
202+
assert.equal(usages[0].metrics?.tokensOutput, 234)
203+
})
204+
205+
test('advisor iterations inherit the TTL split and skip malformed entries', async () => {
206+
const usages = usageEvents(await parse([
207+
assistant('req-1', 'msg-1', {
208+
input_tokens: 1,
209+
output_tokens: 2,
210+
iterations: [
211+
// No model → cannot be priced, so it is not emitted.
212+
{ type: 'advisor_message', input_tokens: 5, output_tokens: 1 },
213+
// Not an object.
214+
'garbage',
215+
{
216+
type: 'advisor_message',
217+
model: 'advisor-model',
218+
input_tokens: 3,
219+
output_tokens: 1,
220+
cache_creation_input_tokens: 100,
221+
cache_read_input_tokens: 7,
222+
cache_creation: { ephemeral_5m_input_tokens: 40, ephemeral_1h_input_tokens: 60 },
223+
},
224+
],
225+
} as unknown as Record<string, number>),
226+
]))
227+
228+
assert.equal(usages.length, 2)
229+
const advisor = usages[1]
230+
assert.equal(advisor.model, 'advisor-model')
231+
assert.equal(advisor.metrics?.tokensInput, 3 + 100 + 7) // cache-inclusive
232+
assert.equal(advisor.metrics?.tokensCacheCreation5mInput, 40)
233+
assert.equal(advisor.metrics?.tokensCacheCreation1hInput, 60)
234+
})
235+
236+
test('two advisor iterations on one line get distinct import identities', async () => {
237+
const events = await parse([
238+
assistant('req-1', 'msg-1', {
239+
input_tokens: 1,
240+
output_tokens: 2,
241+
iterations: [
242+
{ type: 'advisor_message', model: 'advisor-a', input_tokens: 3, output_tokens: 1 },
243+
{ type: 'advisor_message', model: 'advisor-b', input_tokens: 4, output_tokens: 1 },
244+
],
245+
} as unknown as Record<string, number>),
246+
])
247+
248+
const usages = usageEvents(events)
249+
assert.equal(usages.length, 3)
250+
assert.equal(new Set(usages.map(usage => usage.id)).size, 3)
251+
})
252+
253+
test('a duplicate assistant entry does not re-emit its advisor usage', async () => {
254+
// The (messageId, requestId) dedup skips the whole entry, advisors included.
255+
const usages = usageEvents(await parse([
256+
assistant('req-1', 'msg-1', {
257+
input_tokens: 1,
258+
output_tokens: 2,
259+
iterations: [{ type: 'advisor_message', model: 'advisor-model', input_tokens: 10, output_tokens: 2 }],
260+
} as unknown as Record<string, number>),
261+
assistant('req-1', 'msg-1', {
262+
input_tokens: 1,
263+
output_tokens: 2,
264+
iterations: [{ type: 'advisor_message', model: 'advisor-model', input_tokens: 10, output_tokens: 2 }],
265+
} as unknown as Record<string, number>),
266+
]))
267+
268+
assert.equal(usages.length, 2)
269+
})

0 commit comments

Comments
 (0)