Skip to content

Commit 26937ff

Browse files
committed
feat(evals): compare agent tool-use across models
Run the same live scenarios across a list of models and write a scenario x model matrix. models.ts resolves provider:model specs (DeepSeek, OpenAI, Groq, OpenRouter) and reads each provider's key from <PROVIDER>_API_KEY. - agent-tool-use.compare.live.test.ts: EVAL_MODELS x scenarios x trials - report.ts: buildLiveComparisonReport + JSON/Markdown matrix - report.test.ts: key-free aggregation coverage - test:evals:compare script; README documents the spec format
1 parent c93426b commit 26937ff

6 files changed

Lines changed: 386 additions & 0 deletions

File tree

‎apps/sim/evals/README.md‎

Lines changed: 17 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -75,6 +75,23 @@ example, an exact retry count). The report is at
7575
`test-results/evals/agent-tool-use-live.{json,md}` with pass rates, average
7676
iterations, latency, and the failed check names.
7777

78+
### Compare models
79+
80+
Run the same scenarios across several models and get a scenario × model matrix:
81+
82+
```sh
83+
cd apps/sim
84+
EVAL_MODELS=deepseek:deepseek-chat,deepseek:deepseek-reasoner \
85+
DEEPSEEK_API_KEY=... bun run test:evals:compare
86+
```
87+
88+
A spec is `provider:model`; a bare model id defaults to DeepSeek. Providers are
89+
DeepSeek, OpenAI, Groq, and OpenRouter, each reading its key from
90+
`<PROVIDER>_API_KEY`. The report is
91+
`test-results/evals/agent-tool-use-compare.{json,md}`: per-model pass rate,
92+
iterations, latency, and tokens, plus a per-scenario pass-rate matrix.
93+
`EVAL_MIN_PASS_RATE` fails a model below a floor.
94+
7895
## Add a case
7996

8097
1. Open [`agent-tool-use/scenarios.ts`](./agent-tool-use/scenarios.ts) and add
Lines changed: 92 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,92 @@
1+
import { providersMock } from '@sim/testing/mocks/providers.mock'
2+
import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock'
3+
import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock'
4+
import { toolsMock } from '@sim/testing/mocks/tools.mock'
5+
import { afterAll, describe, expect, it, vi } from 'vitest'
6+
import { runScenario } from '@/evals/agent-tool-use/harness'
7+
import { createLiveModelCompletion, parseLiveModels } from '@/evals/agent-tool-use/models'
8+
import { type LiveModelRun, writeLiveComparisonReport } from '@/evals/agent-tool-use/report'
9+
import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios'
10+
11+
vi.mock('@/providers/conversation-history', () => providersConversationHistoryMock)
12+
vi.mock('@/tools', () => toolsMock)
13+
vi.mock('@/providers/utils', () => providersUtilsMock)
14+
vi.mock('@/providers', () => providersMock)
15+
16+
/**
17+
* Compare the same scenarios across several models. Opt-in, never in CI:
18+
*
19+
* EVAL_LIVE=1 EVAL_MODELS=deepseek:deepseek-chat,deepseek:deepseek-reasoner \
20+
* bun run --cwd apps/sim test --mode live evals/agent-tool-use/agent-tool-use.compare.live.test.ts
21+
*
22+
* Each provider reads its key from `<PROVIDER>_API_KEY`; a bare model id
23+
* defaults to DeepSeek. The report is a scenario × model matrix plus per-model
24+
* pass rate, iterations, latency, and tokens. Set `EVAL_MIN_PASS_RATE` (0–1) to
25+
* fail a model below a pass-rate floor.
26+
*/
27+
const LIVE = process.env.EVAL_LIVE === '1'
28+
const TRIALS = Number(process.env.EVAL_TRIALS ?? '3')
29+
const MIN_PASS_RATE = Number(process.env.EVAL_MIN_PASS_RATE ?? '0')
30+
const TIMEOUT_MS = Number(process.env.EVAL_TIMEOUT_MS ?? '180000')
31+
const models = parseLiveModels(process.env.EVAL_MODELS)
32+
const liveScenarios = AGENT_TOOL_USE_SCENARIOS.filter((scenario) => !scenario.scriptedOnly)
33+
34+
const runs: LiveModelRun[] = []
35+
36+
const cases = models.flatMap((spec) =>
37+
liveScenarios.map((scenario) => ({
38+
id: `${spec.provider.id}/${spec.model} · ${scenario.id}`,
39+
spec,
40+
scenario,
41+
}))
42+
)
43+
44+
afterAll(() => {
45+
if (!LIVE || runs.length === 0) return
46+
writeLiveComparisonReport(
47+
runs,
48+
process.env.EVAL_COMPARE_REPORT_PATH ?? 'test-results/evals/agent-tool-use-compare.json'
49+
)
50+
})
51+
52+
describe.skipIf(!LIVE)('agent tool-use model comparison', () => {
53+
it.each(cases)(
54+
'$id',
55+
async ({ spec, scenario }) => {
56+
const completion = createLiveModelCompletion(spec)
57+
for (let trial = 0; trial < TRIALS; trial++) {
58+
const result = await runScenario(scenario, {
59+
completion,
60+
mode: 'live',
61+
model: spec.model,
62+
providerName: spec.provider.label,
63+
})
64+
runs.push({
65+
provider: spec.provider.id,
66+
model: spec.model,
67+
scenarioId: scenario.id,
68+
result,
69+
})
70+
}
71+
},
72+
TIMEOUT_MS
73+
)
74+
75+
it('meets the per-model pass-rate floor', () => {
76+
if (MIN_PASS_RATE <= 0 || runs.length === 0) return
77+
const byModel = new Map<string, { passed: number; total: number }>()
78+
for (const run of runs) {
79+
const key = `${run.provider}/${run.model}`
80+
const entry = byModel.get(key) ?? { passed: 0, total: 0 }
81+
entry.total += 1
82+
if (run.result.passed) entry.passed += 1
83+
byModel.set(key, entry)
84+
}
85+
for (const [model, entry] of byModel) {
86+
expect(
87+
entry.passed / entry.total,
88+
`${model} passed ${entry.passed}/${entry.total}`
89+
).toBeGreaterThanOrEqual(MIN_PASS_RATE)
90+
}
91+
})
92+
})
Lines changed: 85 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,85 @@
1+
import type { OpenAICompatCreateCompletion } from '@/providers/openai-compat/streaming-tool-loop'
2+
import { createOpenAICompatLiveCompletion } from './live'
3+
4+
/**
5+
* Provider registry for the model-comparison live suite.
6+
*
7+
* A model spec is `provider:model` (or a bare model id, which defaults to
8+
* DeepSeek). Each provider reads its key from the matching environment
9+
* variable, so adding a provider is one entry plus its key.
10+
*/
11+
export interface LiveProviderSpec {
12+
id: string
13+
label: string
14+
baseURL?: string
15+
keyEnv: string
16+
baseURLEnv?: string
17+
}
18+
19+
const PROVIDERS: Record<string, LiveProviderSpec> = {
20+
deepseek: {
21+
id: 'deepseek',
22+
label: 'DeepSeek',
23+
baseURL: 'https://api.deepseek.com',
24+
baseURLEnv: 'DEEPSEEK_BASE_URL',
25+
keyEnv: 'DEEPSEEK_API_KEY',
26+
},
27+
openai: { id: 'openai', label: 'OpenAI', keyEnv: 'OPENAI_API_KEY' },
28+
groq: {
29+
id: 'groq',
30+
label: 'Groq',
31+
baseURL: 'https://api.groq.com/openai/v1',
32+
keyEnv: 'GROQ_API_KEY',
33+
},
34+
openrouter: {
35+
id: 'openrouter',
36+
label: 'OpenRouter',
37+
baseURL: 'https://openrouter.ai/api/v1',
38+
keyEnv: 'OPENROUTER_API_KEY',
39+
},
40+
}
41+
42+
export interface LiveModelSpec {
43+
provider: LiveProviderSpec
44+
model: string
45+
}
46+
47+
/** Resolves `provider:model`, or a bare model id against DeepSeek. */
48+
export function resolveLiveModelSpec(spec: string): LiveModelSpec {
49+
const separator = spec.indexOf(':')
50+
if (separator === -1) {
51+
return { provider: PROVIDERS.deepseek, model: spec.trim() }
52+
}
53+
const providerId = spec.slice(0, separator).trim().toLowerCase()
54+
const provider = PROVIDERS[providerId]
55+
if (!provider) {
56+
throw new Error(`Unknown eval provider "${providerId}" in spec "${spec}"`)
57+
}
58+
return { provider, model: spec.slice(separator + 1).trim() }
59+
}
60+
61+
export function parseLiveModels(value: string | undefined): LiveModelSpec[] {
62+
return (value ?? 'deepseek:deepseek-chat')
63+
.split(',')
64+
.map((spec) => spec.trim())
65+
.filter(Boolean)
66+
.map(resolveLiveModelSpec)
67+
}
68+
69+
/** Builds a real completion for one resolved `provider:model` spec. */
70+
export function createLiveModelCompletion(spec: LiveModelSpec): OpenAICompatCreateCompletion {
71+
const apiKey = process.env[spec.provider.keyEnv]
72+
if (!apiKey) {
73+
throw new Error(`${spec.provider.keyEnv} is required for ${spec.provider.label}/${spec.model}`)
74+
}
75+
const baseURL = spec.provider.baseURLEnv
76+
? (process.env[spec.provider.baseURLEnv] ?? spec.provider.baseURL)
77+
: spec.provider.baseURL
78+
79+
return createOpenAICompatLiveCompletion({
80+
apiKey,
81+
model: spec.model,
82+
...(baseURL ? { baseURL } : {}),
83+
...(process.env.EVAL_TIMEOUT_MS ? { timeoutMs: Number(process.env.EVAL_TIMEOUT_MS) } : {}),
84+
})
85+
}
Lines changed: 71 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,71 @@
1+
import { existsSync, mkdtempSync, rmSync } from 'node:fs'
2+
import { tmpdir } from 'node:os'
3+
import { join } from 'node:path'
4+
import { describe, expect, it } from 'vitest'
5+
import {
6+
buildLiveComparisonReport,
7+
type LiveModelRun,
8+
writeLiveComparisonReport,
9+
} from '@/evals/agent-tool-use/report'
10+
import type { AgentToolUseResult } from '@/evals/agent-tool-use/types'
11+
12+
function result(passed: boolean): AgentToolUseResult {
13+
return {
14+
id: 'scenario',
15+
name: 'scenario',
16+
category: 'recovery',
17+
passed,
18+
checks: [],
19+
finalContent: '',
20+
toolInvocations: [],
21+
metrics: {
22+
iterations: 1,
23+
toolCalls: 1,
24+
successfulToolCalls: 1,
25+
erroredToolCalls: 0,
26+
latencyMs: 10,
27+
modelTimeMs: 0,
28+
toolsTimeMs: 0,
29+
firstResponseTimeMs: 0,
30+
inputTokens: 1,
31+
outputTokens: 1,
32+
totalTokens: 2,
33+
},
34+
}
35+
}
36+
37+
function run(model: string, passed: boolean): LiveModelRun {
38+
return { provider: 'deepseek', model, scenarioId: 'scenario', result: result(passed) }
39+
}
40+
41+
describe('model comparison report', () => {
42+
it('groups runs by model, sorts by pass rate, and fills the scenario matrix', () => {
43+
const report = buildLiveComparisonReport([
44+
run('chat', true),
45+
run('chat', true),
46+
run('reasoner', false),
47+
run('reasoner', true),
48+
])
49+
50+
expect(report.models.map((model) => model.label)).toEqual([
51+
'deepseek/chat',
52+
'deepseek/reasoner',
53+
])
54+
expect(report.models[0]).toMatchObject({ passRate: 1, passed: 2, trials: 2 })
55+
expect(report.models[1]).toMatchObject({ passRate: 0.5, passed: 1, trials: 2 })
56+
expect(report.matrix.scenario['deepseek/chat']).toBe(1)
57+
expect(report.matrix.scenario['deepseek/reasoner']).toBe(0.5)
58+
})
59+
60+
it('writes JSON and Markdown reports', () => {
61+
const directory = mkdtempSync(join(tmpdir(), 'eval-compare-'))
62+
try {
63+
const path = join(directory, 'compare.json')
64+
writeLiveComparisonReport([run('chat', true)], path)
65+
expect(existsSync(path)).toBe(true)
66+
expect(existsSync(join(directory, 'compare.md'))).toBe(true)
67+
} finally {
68+
rmSync(directory, { recursive: true, force: true })
69+
}
70+
})
71+
})

‎apps/sim/evals/agent-tool-use/report.ts‎

Lines changed: 120 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -127,3 +127,123 @@ export function writeLiveEvalReport(summaries: LiveScenarioSummary[], reportPath
127127
writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`)
128128
writeFileSync(reportPath.replace(/\.json$/, '.md'), renderLiveMarkdown(report))
129129
}
130+
131+
/** One trial of one model on one scenario, for the comparison report. */
132+
export interface LiveModelRun {
133+
provider: string
134+
model: string
135+
scenarioId: string
136+
result: AgentToolUseResult
137+
}
138+
139+
/** Aggregate behavior of one model across the suite. */
140+
export interface LiveModelSummary {
141+
provider: string
142+
model: string
143+
label: string
144+
trials: number
145+
passed: number
146+
passRate: number
147+
avgIterations: number
148+
avgLatencyMs: number
149+
totalTokens: number
150+
}
151+
152+
/** Side-by-side comparison of several models on the same scenarios. */
153+
export interface LiveComparisonReport {
154+
suite: 'agent-tool-use-compare'
155+
generatedAt: string
156+
scenarios: string[]
157+
models: LiveModelSummary[]
158+
/** scenarioId -> model label -> pass rate. */
159+
matrix: Record<string, Record<string, number>>
160+
}
161+
162+
function modelLabel(run: LiveModelRun): string {
163+
return `${run.provider}/${run.model}`
164+
}
165+
166+
export function buildLiveComparisonReport(runs: LiveModelRun[]): LiveComparisonReport {
167+
const byModel = new Map<string, LiveModelRun[]>()
168+
for (const run of runs) {
169+
const key = modelLabel(run)
170+
const list = byModel.get(key) ?? []
171+
list.push(run)
172+
byModel.set(key, list)
173+
}
174+
175+
const scenarios = [...new Set(runs.map((run) => run.scenarioId))].sort()
176+
const models: LiveModelSummary[] = []
177+
const matrix: Record<string, Record<string, number>> = {}
178+
179+
for (const [label, modelRuns] of byModel) {
180+
const results = modelRuns.map((run) => run.result)
181+
const passed = results.filter((result) => result.passed).length
182+
const [provider, model] = label.split('/')
183+
models.push({
184+
provider,
185+
model,
186+
label,
187+
trials: results.length,
188+
passed,
189+
passRate: results.length === 0 ? 0 : passed / results.length,
190+
avgIterations: average(results.map((result) => result.metrics.iterations)),
191+
avgLatencyMs: average(results.map((result) => result.metrics.latencyMs)),
192+
totalTokens: results.reduce((sum, result) => sum + result.metrics.totalTokens, 0),
193+
})
194+
195+
for (const scenario of scenarios) {
196+
const scenarioRuns = modelRuns.filter((run) => run.scenarioId === scenario)
197+
const scenarioPassed = scenarioRuns.filter((run) => run.result.passed).length
198+
matrix[scenario] ??= {}
199+
matrix[scenario][label] = scenarioRuns.length === 0 ? 0 : scenarioPassed / scenarioRuns.length
200+
}
201+
}
202+
203+
models.sort((a, b) => b.passRate - a.passRate)
204+
return {
205+
suite: 'agent-tool-use-compare',
206+
generatedAt: new Date().toISOString(),
207+
scenarios,
208+
models,
209+
matrix,
210+
}
211+
}
212+
213+
function renderComparisonMarkdown(report: LiveComparisonReport): string {
214+
const labels = report.models.map((model) => model.label)
215+
const lines = [
216+
'# Agent tool-use model comparison',
217+
'',
218+
`Generated: ${report.generatedAt}`,
219+
'',
220+
'| Model | Pass rate | Trials | Avg iterations | Avg latency | Tokens |',
221+
'| --- | ---: | ---: | ---: | ---: | ---: |',
222+
...report.models.map(
223+
(model) =>
224+
`| ${model.label} | ${(model.passRate * 100).toFixed(0)}% (${model.passed}/${model.trials}) | ${model.trials} | ${model.avgIterations.toFixed(1)} | ${Math.round(model.avgLatencyMs)}ms | ${model.totalTokens} |`
225+
),
226+
'',
227+
`| Scenario | ${labels.join(' | ')} |`,
228+
`| --- | ${labels.map(() => '---:').join(' | ')} |`,
229+
...report.scenarios.map(
230+
(scenario) =>
231+
`| ${escapeCell(scenario)} | ${labels
232+
.map((label) => `${((report.matrix[scenario]?.[label] ?? 0) * 100).toFixed(0)}%`)
233+
.join(' | ')} |`
234+
),
235+
'',
236+
]
237+
return lines.join('\n')
238+
}
239+
240+
/**
241+
* Writes the model-comparison JSON report and a sibling Markdown table.
242+
* Unlike the single-model report, this is a matrix: scenario × model pass rates.
243+
*/
244+
export function writeLiveComparisonReport(runs: LiveModelRun[], reportPath: string): void {
245+
const report = buildLiveComparisonReport(runs)
246+
mkdirSync(dirname(reportPath), { recursive: true })
247+
writeFileSync(reportPath, `${JSON.stringify(report, null, 2)}\n`)
248+
writeFileSync(reportPath.replace(/\.json$/, '.md'), renderComparisonMarkdown(report))
249+
}

0 commit comments

Comments
 (0)