|
| 1 | +import { |
| 2 | + createSerializedBlock, |
| 3 | + createSerializedWorkflow, |
| 4 | +} from '@sim/testing/factories/serialized-block.factory' |
| 5 | +import { providersMockFns } from '@sim/testing/mocks/providers.mock' |
| 6 | +import { DAGExecutor } from '@/executor/execution/executor' |
| 7 | +import type { SerializedWorkflow } from '@/serializer/types' |
| 8 | +import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness' |
| 9 | +import type { |
| 10 | + AgentToolUseExpectations, |
| 11 | + AgentToolUseResult, |
| 12 | + EvalCategory, |
| 13 | + EvalToolInvocation, |
| 14 | +} from './types' |
| 15 | + |
| 16 | +/** |
| 17 | + * Executor-level harness. |
| 18 | + * |
| 19 | + * Drives a real `DAGExecutor` run: Start block → Agent block. The provider |
| 20 | + * boundary (`executeProviderRequest`) is the only thing mocked — the Agent |
| 21 | + * block handler, input/variable resolution, and the executor run/error handling |
| 22 | + * are real. Tool calls are what the mocked provider returns; tool *dispatch* is |
| 23 | + * covered by the loop harness. |
| 24 | + */ |
| 25 | + |
| 26 | +/** One tool call the mocked provider reports in its response. */ |
| 27 | +export interface ExecutorProviderToolCall { |
| 28 | + name: string |
| 29 | + arguments?: Record<string, unknown> |
| 30 | + result?: unknown |
| 31 | +} |
| 32 | + |
| 33 | +/** The provider response `executeProviderRequest` returns for one model call. */ |
| 34 | +export interface ExecutorProviderResponse { |
| 35 | + content: string |
| 36 | + model?: string |
| 37 | + tokens?: { input?: number; output?: number; total?: number } |
| 38 | + toolCalls?: ExecutorProviderToolCall[] |
| 39 | + cost?: unknown |
| 40 | + timing?: unknown |
| 41 | +} |
| 42 | + |
| 43 | +export interface ExecutorScenario { |
| 44 | + id: string |
| 45 | + name: string |
| 46 | + category: EvalCategory |
| 47 | + description: string |
| 48 | + /** Exposed on the Start block and referenced from the Agent block. */ |
| 49 | + workflowInput: Record<string, unknown> |
| 50 | + agent: { |
| 51 | + model: string |
| 52 | + systemPrompt?: string |
| 53 | + userPrompt?: string |
| 54 | + temperature?: number |
| 55 | + } |
| 56 | + /** One entry per model call; the last entry serves any extra fallback calls. */ |
| 57 | + providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[] |
| 58 | + expect: AgentToolUseExpectations & { |
| 59 | + /** Substring that must appear in the messages sent to the provider. */ |
| 60 | + resolvedInput?: string |
| 61 | + /** Expected `ExecutionResult.success`. */ |
| 62 | + succeeds?: boolean |
| 63 | + } |
| 64 | +} |
| 65 | + |
| 66 | +function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow { |
| 67 | + const start = createSerializedBlock({ |
| 68 | + id: 'start', |
| 69 | + type: 'start_trigger', |
| 70 | + name: 'Start', |
| 71 | + }) |
| 72 | + /** The trigger handler claims a block whose metadata says it is a trigger. */ |
| 73 | + if (start.metadata) start.metadata.category = 'triggers' |
| 74 | + const agent = createSerializedBlock({ |
| 75 | + id: 'agent', |
| 76 | + type: 'agent', |
| 77 | + name: 'Eval Agent', |
| 78 | + }) |
| 79 | + agent.config.tool = 'agent' |
| 80 | + agent.config.params = { |
| 81 | + model: scenario.agent.model, |
| 82 | + systemPrompt: scenario.agent.systemPrompt, |
| 83 | + userPrompt: scenario.agent.userPrompt, |
| 84 | + ...(scenario.agent.temperature !== undefined |
| 85 | + ? { temperature: scenario.agent.temperature } |
| 86 | + : {}), |
| 87 | + } |
| 88 | + |
| 89 | + return createSerializedWorkflow([start, agent], [{ source: 'start', target: 'agent' }]) |
| 90 | +} |
| 91 | + |
| 92 | +/** |
| 93 | + * Runs one executor scenario and scores it with the shared scorer, returning |
| 94 | + * the same result shape as the loop harness so both land in one report. |
| 95 | + */ |
| 96 | +export async function runExecutorScenario( |
| 97 | + scenario: ExecutorScenario, |
| 98 | + options: { mode?: EvalRunMode } = {} |
| 99 | +): Promise<AgentToolUseResult> { |
| 100 | + const mode = options.mode ?? 'scripted' |
| 101 | + const responses = Array.isArray(scenario.providerResponse) |
| 102 | + ? [...scenario.providerResponse] |
| 103 | + : [scenario.providerResponse] |
| 104 | + const requests: Array<Record<string, unknown>> = [] |
| 105 | + let callIndex = 0 |
| 106 | + |
| 107 | + providersMockFns.mockExecuteProviderRequest.mockImplementation( |
| 108 | + async (_providerId: string, request: Record<string, unknown>) => { |
| 109 | + requests.push(request) |
| 110 | + const response = responses[Math.min(callIndex, responses.length - 1)] |
| 111 | + callIndex += 1 |
| 112 | + return { |
| 113 | + content: response.content, |
| 114 | + model: response.model ?? scenario.agent.model, |
| 115 | + tokens: response.tokens ?? { input: 0, output: 0, total: 0 }, |
| 116 | + toolCalls: response.toolCalls ?? [], |
| 117 | + cost: response.cost ?? 0, |
| 118 | + timing: response.timing ?? { total: 0 }, |
| 119 | + } |
| 120 | + } |
| 121 | + ) |
| 122 | + |
| 123 | + const executor = new DAGExecutor({ |
| 124 | + workflow: buildWorkflow(scenario), |
| 125 | + workflowInput: scenario.workflowInput, |
| 126 | + contextExtensions: { |
| 127 | + workspaceId: 'eval-workspace', |
| 128 | + executionId: 'eval-execution', |
| 129 | + userId: 'eval-user', |
| 130 | + }, |
| 131 | + }) |
| 132 | + |
| 133 | + let result: { success?: boolean; output?: Record<string, unknown> } | undefined |
| 134 | + let runError: unknown |
| 135 | + const startedAt = Date.now() |
| 136 | + try { |
| 137 | + result = (await executor.execute('eval-workflow')) as typeof result |
| 138 | + } catch (error) { |
| 139 | + runError = error |
| 140 | + } |
| 141 | + const latencyMs = Date.now() - startedAt |
| 142 | + |
| 143 | + const output = (result?.output ?? {}) as Record<string, unknown> |
| 144 | + const finalContent = typeof output.content === 'string' ? output.content : '' |
| 145 | + const rawToolCalls = ((output.toolCalls as { list?: unknown[] } | undefined)?.list ?? |
| 146 | + []) as Array<Record<string, unknown>> |
| 147 | + |
| 148 | + const toolCalls: ScoredToolCall[] = rawToolCalls.map((call) => ({ |
| 149 | + name: typeof call.name === 'string' ? call.name : 'unknown', |
| 150 | + success: true, |
| 151 | + })) |
| 152 | + const toolInvocations: EvalToolInvocation[] = rawToolCalls.map((call) => ({ |
| 153 | + name: typeof call.name === 'string' ? call.name : 'unknown', |
| 154 | + arguments: (call.arguments ?? {}) as Record<string, unknown>, |
| 155 | + success: true, |
| 156 | + durationMs: typeof call.duration === 'number' ? call.duration : 0, |
| 157 | + })) |
| 158 | + |
| 159 | + const checks = scoreExpectations(scenario.expect, toolCalls, finalContent, 1, runError, mode) |
| 160 | + |
| 161 | + if (scenario.expect.resolvedInput !== undefined) { |
| 162 | + const sent = JSON.stringify(requests) |
| 163 | + checks.push({ |
| 164 | + name: 'resolved-input', |
| 165 | + passed: sent.includes(scenario.expect.resolvedInput), |
| 166 | + detail: `looking for ${JSON.stringify(scenario.expect.resolvedInput)} in provider messages`, |
| 167 | + }) |
| 168 | + } |
| 169 | + |
| 170 | + if (scenario.expect.succeeds !== undefined) { |
| 171 | + checks.push({ |
| 172 | + name: 'workflow-success', |
| 173 | + passed: result?.success === scenario.expect.succeeds, |
| 174 | + detail: `success=${String(result?.success)}`, |
| 175 | + }) |
| 176 | + } |
| 177 | + |
| 178 | + const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number } |
| 179 | + |
| 180 | + return { |
| 181 | + id: scenario.id, |
| 182 | + name: scenario.name, |
| 183 | + category: scenario.category, |
| 184 | + passed: checks.every((entry) => entry.passed), |
| 185 | + checks, |
| 186 | + finalContent, |
| 187 | + toolInvocations, |
| 188 | + metrics: { |
| 189 | + iterations: requests.length, |
| 190 | + toolCalls: toolCalls.length, |
| 191 | + successfulToolCalls: toolCalls.filter((call) => call.success).length, |
| 192 | + erroredToolCalls: 0, |
| 193 | + latencyMs, |
| 194 | + modelTimeMs: 0, |
| 195 | + toolsTimeMs: 0, |
| 196 | + firstResponseTimeMs: 0, |
| 197 | + inputTokens: tokens.input ?? 0, |
| 198 | + outputTokens: tokens.output ?? 0, |
| 199 | + totalTokens: tokens.total ?? 0, |
| 200 | + }, |
| 201 | + ...(runError ? { error: String(runError) } : {}), |
| 202 | + } |
| 203 | +} |
| 204 | + |
| 205 | +/** |
| 206 | + * Executor-level scenarios. Two cover the wiring the loop suite cannot see: |
| 207 | + * Start → Agent execution, and variable resolution from a Start output into the |
| 208 | + * Agent's prompt. |
| 209 | + */ |
| 210 | +export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [ |
| 211 | + { |
| 212 | + id: 'executor-agent-runs', |
| 213 | + name: 'runs a Start → Agent workflow and surfaces the Agent output', |
| 214 | + category: 'tool-selection', |
| 215 | + description: |
| 216 | + 'The real Agent block handler runs inside the DAG. The mocked provider reports one tool call; the executor result must carry the content and the tool call through.', |
| 217 | + workflowInput: { message: 'What is the API rate limit?' }, |
| 218 | + agent: { |
| 219 | + model: 'gpt-4o', |
| 220 | + systemPrompt: 'You are a documentation assistant.', |
| 221 | + userPrompt: 'What is the API rate limit?', |
| 222 | + }, |
| 223 | + providerResponse: { |
| 224 | + content: 'The API rate limit is 100 requests per minute.', |
| 225 | + toolCalls: [ |
| 226 | + { |
| 227 | + name: 'search_docs', |
| 228 | + arguments: { query: 'api rate limit' }, |
| 229 | + result: { snippet: 'The API rate limit is 100 requests per minute.' }, |
| 230 | + }, |
| 231 | + ], |
| 232 | + tokens: { input: 10, output: 20, total: 30 }, |
| 233 | + }, |
| 234 | + expect: { |
| 235 | + succeeds: true, |
| 236 | + finalContent: '100 requests per minute', |
| 237 | + toolCallSequence: ['search_docs'], |
| 238 | + successfulToolCalls: 1, |
| 239 | + }, |
| 240 | + }, |
| 241 | + { |
| 242 | + id: 'executor-resolves-start-input', |
| 243 | + name: 'resolves a Start output into the Agent prompt before the provider call', |
| 244 | + category: 'planning', |
| 245 | + description: |
| 246 | + 'The Agent userPrompt references <start.message>. The value must be resolved by the executor and reach the provider request, not passed through verbatim.', |
| 247 | + workflowInput: { message: 'Summarize order A-1937' }, |
| 248 | + agent: { |
| 249 | + model: 'gpt-4o', |
| 250 | + userPrompt: '<start.message>', |
| 251 | + }, |
| 252 | + providerResponse: { |
| 253 | + content: 'Order A-1937 shipped via DHL.', |
| 254 | + tokens: { input: 8, output: 12, total: 20 }, |
| 255 | + }, |
| 256 | + expect: { |
| 257 | + succeeds: true, |
| 258 | + resolvedInput: 'Summarize order A-1937', |
| 259 | + finalContent: /A-1937/, |
| 260 | + }, |
| 261 | + }, |
| 262 | +] |
0 commit comments