Skip to content

Commit 85525a4

Browse files
committed
feat(evals): run agent scenarios through the DAGExecutor
Add an executor-level harness: a real Start -> Agent workflow on DAGExecutor, with only executeProviderRequest mocked at the provider boundary. This covers agent-block input wiring, variable resolution from Start outputs, and executor run/error handling, which the direct loop harness cannot see. - executor-harness.ts: workflow builder + runExecutorScenario - shares the scorer (scoreExpectations) and report with the loop suite - two scenarios: Start->Agent output, and <start.message> resolution - README documents adding an executor-level scenario
1 parent 433fd0f commit 85525a4

4 files changed

Lines changed: 348 additions & 11 deletions

File tree

‎apps/sim/evals/README.md‎

Lines changed: 27 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -100,6 +100,27 @@ A scenario is data, not code — there is no harness change needed for a new cas
100100
string. The loop must not execute the call and must return the parse error to
101101
the model.
102102

103+
## Executor-level scenarios
104+
105+
[`agent-tool-use/executor-harness.ts`](./agent-tool-use/executor-harness.ts)
106+
runs a case through a real `DAGExecutor`: a Start block → Agent block workflow,
107+
with only the provider boundary (`executeProviderRequest`) mocked. This covers
108+
what the loop harness cannot — agent-block input wiring, variable resolution
109+
from Start outputs, and the executor's run/error handling. Tool dispatch stays
110+
covered by the loop suite.
111+
112+
Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`:
113+
114+
- `workflowInput` is exposed on the Start block; reference an output with
115+
`<start.field>` from the Agent prompt.
116+
- `agent` is the Agent block config (`model`, `systemPrompt`, `userPrompt`).
117+
- `providerResponse` is what the mocked provider returns (`content`,
118+
`toolCalls`, `tokens`).
119+
- `expect` uses the loop's checks plus `resolvedInput` (a substring that must
120+
reach the provider messages) and `succeeds` (expected `ExecutionResult.success`).
121+
122+
Both suites write one report, so executor rows appear alongside loop rows.
123+
103124
## Report shape
104125

105126
`report.json` is machine-readable for dashboards and trend tracking; `report.md`
@@ -110,8 +131,9 @@ model/tool time, first-response time, and token usage.
110131

111132
## Scope and next steps
112133

113-
This suite evaluates the tool loop directly. The next layer is a scenario that
114-
runs the same scripted model through the full `DAGExecutor` so agent block
115-
wiring, variable resolution, and the executor's retry/fallback policy are
116-
measured alongside the loop. The `AgentToolUseResult` shape is deliberately
117-
independent of the harness entry point so both can share scoring and reporting.
134+
Two harnesses share one result shape and report: the tool loop and the
135+
`DAGExecutor`. The executor harness mocks the provider boundary, so the
136+
executor's retry/fallback policy is not yet asserted; add a scenario with a
137+
first-call rejection and a block retry config to cover it. Further expansion
138+
(context/memory, model routing, subagent orchestration) is tracked as
139+
follow-up work.

‎apps/sim/evals/agent-tool-use/agent-tool-use.eval.test.ts‎

Lines changed: 49 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,8 +1,14 @@
1+
import {
2+
permissionCheckMock,
3+
permissionCheckMockFns,
4+
} from '@sim/testing/mocks/permission-check.mock'
15
import { providersMock } from '@sim/testing/mocks/providers.mock'
26
import { providersConversationHistoryMock } from '@sim/testing/mocks/providers-conversation-history.mock'
3-
import { providersUtilsMock } from '@sim/testing/mocks/providers-utils.mock'
7+
import { providersUtilsMock, providersUtilsMockFns } from '@sim/testing/mocks/providers-utils.mock'
48
import { toolsMock } from '@sim/testing/mocks/tools.mock'
5-
import { afterAll, describe, expect, it, vi } from 'vitest'
9+
import { workspaceFileSecretProvenanceMock } from '@sim/testing/mocks/workspace-file-secret-provenance.mock'
10+
import { afterAll, beforeEach, describe, expect, it, vi } from 'vitest'
11+
import { EXECUTOR_SCENARIOS, runExecutorScenario } from '@/evals/agent-tool-use/executor-harness'
612
import { runScenario } from '@/evals/agent-tool-use/harness'
713
import { writeEvalReport } from '@/evals/agent-tool-use/report'
814
import { AGENT_TOOL_USE_SCENARIOS } from '@/evals/agent-tool-use/scenarios'
@@ -12,9 +18,37 @@ vi.mock('@/providers/conversation-history', () => providersConversationHistoryMo
1218
vi.mock('@/tools', () => toolsMock)
1319
vi.mock('@/providers/utils', () => providersUtilsMock)
1420
vi.mock('@/providers', () => providersMock)
21+
vi.mock('@/ee/access-control/utils/permission-check', () => permissionCheckMock)
22+
vi.mock(
23+
'@/lib/uploads/contexts/workspace/workspace-file-secret-provenance',
24+
() => workspaceFileSecretProvenanceMock
25+
)
26+
vi.mock('@/lib/memory/agent-turn-session', () => ({
27+
openAgentTurnSession: vi.fn(async () => undefined),
28+
}))
29+
vi.mock('@/lib/internal/mcp/discover-tools', () => ({
30+
discoverMcpServerToolsAsExecutor: vi.fn(async () => []),
31+
}))
32+
vi.mock('@/lib/internal/custom-tools/read-available-by-id-or-title', () => ({
33+
readAvailableCustomToolByIdOrTitleAsExecutor: vi.fn(async () => undefined),
34+
}))
35+
vi.mock('@/executor/utils/http', () => ({
36+
buildAuthHeaders: vi.fn(async () => ({ 'Content-Type': 'application/json' })),
37+
buildAPIUrl: vi.fn((path: string) => path),
38+
extractAPIErrorMessage: vi.fn(async () => 'request failed'),
39+
}))
40+
vi.mock('@/lib/execution/cancellation', () => ({
41+
subscribeToExecutionCancellation: vi.fn(async () => () => {}),
42+
isExecutionCancelled: vi.fn(async () => false),
43+
}))
1544

1645
const results: AgentToolUseResult[] = []
1746

47+
beforeEach(() => {
48+
permissionCheckMockFns.mockValidateModelProvider.mockResolvedValue(undefined)
49+
providersUtilsMockFns.mockGetProviderFromModel.mockReturnValue('mock-provider')
50+
})
51+
1852
afterAll(() => {
1953
const reportPath = process.env.EVAL_REPORT_PATH
2054
if (reportPath) writeEvalReport(results, reportPath)
@@ -32,3 +66,16 @@ describe('agent tool-use eval suite', () => {
3266
).toEqual([])
3367
})
3468
})
69+
70+
describe('agent executor eval suite', () => {
71+
it.each(EXECUTOR_SCENARIOS)('$id: $name', async (scenario) => {
72+
const result = await runExecutorScenario(scenario)
73+
results.push(result)
74+
75+
const failed = result.checks.filter((entry) => !entry.passed)
76+
expect(
77+
failed,
78+
failed.map((entry) => `${entry.name}: ${entry.detail}`).join('; ') || undefined
79+
).toEqual([])
80+
})
81+
})
Lines changed: 262 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,262 @@
1+
import {
2+
createSerializedBlock,
3+
createSerializedWorkflow,
4+
} from '@sim/testing/factories/serialized-block.factory'
5+
import { providersMockFns } from '@sim/testing/mocks/providers.mock'
6+
import { DAGExecutor } from '@/executor/execution/executor'
7+
import type { SerializedWorkflow } from '@/serializer/types'
8+
import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness'
9+
import type {
10+
AgentToolUseExpectations,
11+
AgentToolUseResult,
12+
EvalCategory,
13+
EvalToolInvocation,
14+
} from './types'
15+
16+
/**
17+
* Executor-level harness.
18+
*
19+
* Drives a real `DAGExecutor` run: Start block → Agent block. The provider
20+
* boundary (`executeProviderRequest`) is the only thing mocked — the Agent
21+
* block handler, input/variable resolution, and the executor run/error handling
22+
* are real. Tool calls are what the mocked provider returns; tool *dispatch* is
23+
* covered by the loop harness.
24+
*/
25+
26+
/** One tool call the mocked provider reports in its response. */
27+
export interface ExecutorProviderToolCall {
28+
name: string
29+
arguments?: Record<string, unknown>
30+
result?: unknown
31+
}
32+
33+
/** The provider response `executeProviderRequest` returns for one model call. */
34+
export interface ExecutorProviderResponse {
35+
content: string
36+
model?: string
37+
tokens?: { input?: number; output?: number; total?: number }
38+
toolCalls?: ExecutorProviderToolCall[]
39+
cost?: unknown
40+
timing?: unknown
41+
}
42+
43+
export interface ExecutorScenario {
44+
id: string
45+
name: string
46+
category: EvalCategory
47+
description: string
48+
/** Exposed on the Start block and referenced from the Agent block. */
49+
workflowInput: Record<string, unknown>
50+
agent: {
51+
model: string
52+
systemPrompt?: string
53+
userPrompt?: string
54+
temperature?: number
55+
}
56+
/** One entry per model call; the last entry serves any extra fallback calls. */
57+
providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[]
58+
expect: AgentToolUseExpectations & {
59+
/** Substring that must appear in the messages sent to the provider. */
60+
resolvedInput?: string
61+
/** Expected `ExecutionResult.success`. */
62+
succeeds?: boolean
63+
}
64+
}
65+
66+
function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow {
67+
const start = createSerializedBlock({
68+
id: 'start',
69+
type: 'start_trigger',
70+
name: 'Start',
71+
})
72+
/** The trigger handler claims a block whose metadata says it is a trigger. */
73+
if (start.metadata) start.metadata.category = 'triggers'
74+
const agent = createSerializedBlock({
75+
id: 'agent',
76+
type: 'agent',
77+
name: 'Eval Agent',
78+
})
79+
agent.config.tool = 'agent'
80+
agent.config.params = {
81+
model: scenario.agent.model,
82+
systemPrompt: scenario.agent.systemPrompt,
83+
userPrompt: scenario.agent.userPrompt,
84+
...(scenario.agent.temperature !== undefined
85+
? { temperature: scenario.agent.temperature }
86+
: {}),
87+
}
88+
89+
return createSerializedWorkflow([start, agent], [{ source: 'start', target: 'agent' }])
90+
}
91+
92+
/**
93+
* Runs one executor scenario and scores it with the shared scorer, returning
94+
* the same result shape as the loop harness so both land in one report.
95+
*/
96+
export async function runExecutorScenario(
97+
scenario: ExecutorScenario,
98+
options: { mode?: EvalRunMode } = {}
99+
): Promise<AgentToolUseResult> {
100+
const mode = options.mode ?? 'scripted'
101+
const responses = Array.isArray(scenario.providerResponse)
102+
? [...scenario.providerResponse]
103+
: [scenario.providerResponse]
104+
const requests: Array<Record<string, unknown>> = []
105+
let callIndex = 0
106+
107+
providersMockFns.mockExecuteProviderRequest.mockImplementation(
108+
async (_providerId: string, request: Record<string, unknown>) => {
109+
requests.push(request)
110+
const response = responses[Math.min(callIndex, responses.length - 1)]
111+
callIndex += 1
112+
return {
113+
content: response.content,
114+
model: response.model ?? scenario.agent.model,
115+
tokens: response.tokens ?? { input: 0, output: 0, total: 0 },
116+
toolCalls: response.toolCalls ?? [],
117+
cost: response.cost ?? 0,
118+
timing: response.timing ?? { total: 0 },
119+
}
120+
}
121+
)
122+
123+
const executor = new DAGExecutor({
124+
workflow: buildWorkflow(scenario),
125+
workflowInput: scenario.workflowInput,
126+
contextExtensions: {
127+
workspaceId: 'eval-workspace',
128+
executionId: 'eval-execution',
129+
userId: 'eval-user',
130+
},
131+
})
132+
133+
let result: { success?: boolean; output?: Record<string, unknown> } | undefined
134+
let runError: unknown
135+
const startedAt = Date.now()
136+
try {
137+
result = (await executor.execute('eval-workflow')) as typeof result
138+
} catch (error) {
139+
runError = error
140+
}
141+
const latencyMs = Date.now() - startedAt
142+
143+
const output = (result?.output ?? {}) as Record<string, unknown>
144+
const finalContent = typeof output.content === 'string' ? output.content : ''
145+
const rawToolCalls = ((output.toolCalls as { list?: unknown[] } | undefined)?.list ??
146+
[]) as Array<Record<string, unknown>>
147+
148+
const toolCalls: ScoredToolCall[] = rawToolCalls.map((call) => ({
149+
name: typeof call.name === 'string' ? call.name : 'unknown',
150+
success: true,
151+
}))
152+
const toolInvocations: EvalToolInvocation[] = rawToolCalls.map((call) => ({
153+
name: typeof call.name === 'string' ? call.name : 'unknown',
154+
arguments: (call.arguments ?? {}) as Record<string, unknown>,
155+
success: true,
156+
durationMs: typeof call.duration === 'number' ? call.duration : 0,
157+
}))
158+
159+
const checks = scoreExpectations(scenario.expect, toolCalls, finalContent, 1, runError, mode)
160+
161+
if (scenario.expect.resolvedInput !== undefined) {
162+
const sent = JSON.stringify(requests)
163+
checks.push({
164+
name: 'resolved-input',
165+
passed: sent.includes(scenario.expect.resolvedInput),
166+
detail: `looking for ${JSON.stringify(scenario.expect.resolvedInput)} in provider messages`,
167+
})
168+
}
169+
170+
if (scenario.expect.succeeds !== undefined) {
171+
checks.push({
172+
name: 'workflow-success',
173+
passed: result?.success === scenario.expect.succeeds,
174+
detail: `success=${String(result?.success)}`,
175+
})
176+
}
177+
178+
const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number }
179+
180+
return {
181+
id: scenario.id,
182+
name: scenario.name,
183+
category: scenario.category,
184+
passed: checks.every((entry) => entry.passed),
185+
checks,
186+
finalContent,
187+
toolInvocations,
188+
metrics: {
189+
iterations: requests.length,
190+
toolCalls: toolCalls.length,
191+
successfulToolCalls: toolCalls.filter((call) => call.success).length,
192+
erroredToolCalls: 0,
193+
latencyMs,
194+
modelTimeMs: 0,
195+
toolsTimeMs: 0,
196+
firstResponseTimeMs: 0,
197+
inputTokens: tokens.input ?? 0,
198+
outputTokens: tokens.output ?? 0,
199+
totalTokens: tokens.total ?? 0,
200+
},
201+
...(runError ? { error: String(runError) } : {}),
202+
}
203+
}
204+
205+
/**
206+
* Executor-level scenarios. Two cover the wiring the loop suite cannot see:
207+
* Start → Agent execution, and variable resolution from a Start output into the
208+
* Agent's prompt.
209+
*/
210+
export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [
211+
{
212+
id: 'executor-agent-runs',
213+
name: 'runs a Start → Agent workflow and surfaces the Agent output',
214+
category: 'tool-selection',
215+
description:
216+
'The real Agent block handler runs inside the DAG. The mocked provider reports one tool call; the executor result must carry the content and the tool call through.',
217+
workflowInput: { message: 'What is the API rate limit?' },
218+
agent: {
219+
model: 'gpt-4o',
220+
systemPrompt: 'You are a documentation assistant.',
221+
userPrompt: 'What is the API rate limit?',
222+
},
223+
providerResponse: {
224+
content: 'The API rate limit is 100 requests per minute.',
225+
toolCalls: [
226+
{
227+
name: 'search_docs',
228+
arguments: { query: 'api rate limit' },
229+
result: { snippet: 'The API rate limit is 100 requests per minute.' },
230+
},
231+
],
232+
tokens: { input: 10, output: 20, total: 30 },
233+
},
234+
expect: {
235+
succeeds: true,
236+
finalContent: '100 requests per minute',
237+
toolCallSequence: ['search_docs'],
238+
successfulToolCalls: 1,
239+
},
240+
},
241+
{
242+
id: 'executor-resolves-start-input',
243+
name: 'resolves a Start output into the Agent prompt before the provider call',
244+
category: 'planning',
245+
description:
246+
'The Agent userPrompt references <start.message>. The value must be resolved by the executor and reach the provider request, not passed through verbatim.',
247+
workflowInput: { message: 'Summarize order A-1937' },
248+
agent: {
249+
model: 'gpt-4o',
250+
userPrompt: '<start.message>',
251+
},
252+
providerResponse: {
253+
content: 'Order A-1937 shipped via DHL.',
254+
tokens: { input: 8, output: 12, total: 20 },
255+
},
256+
expect: {
257+
succeeds: true,
258+
resolvedInput: 'Summarize order A-1937',
259+
finalContent: /A-1937/,
260+
},
261+
},
262+
]

0 commit comments

Comments
 (0)