Skip to content

Commit c93426b

Browse files
committed
test(evals): assert executor model fallback on primary failure
Add executor-falls-back-to-secondary-model: the primary call rejects, the Agent block has a fallback model, and the handler serves the answer from gpt-4o-mini. Asserts providerCalls === 2 and lastRequestModel, and fails without the fallback row (checked locally: got gpt-4o, run errored).
1 parent fb7f54c commit c93426b

2 files changed

Lines changed: 49 additions & 7 deletions

File tree

‎apps/sim/evals/README.md‎

Lines changed: 8 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -119,8 +119,10 @@ Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`:
119119
- `expect` uses the loop's checks plus `resolvedInput` (a substring that must
120120
reach the provider messages), `succeeds` (expected `ExecutionResult.success`),
121121
and `providerCalls` (exact provider call count).
122-
- Set `agent.retry` to exercise the executor's per-block retry policy; make the
123-
first `providerResponse` a `reject` and the retry lands on the next one.
122+
- Set `agent.retry` to exercise the executor's per-block retry policy, or
123+
`agent.fallbackModels` to exercise model fallback. Make the first
124+
`providerResponse` a `reject` and the next one succeeds; assert
125+
`providerCalls` and `lastRequestModel` to prove which path recovered.
124126

125127
Both suites write one report, so executor rows appear alongside loop rows.
126128

@@ -135,8 +137,7 @@ model/tool time, first-response time, and token usage.
135137
## Scope and next steps
136138

137139
Two harnesses share one result shape and report: the tool loop and the
138-
`DAGExecutor`. The executor suite covers block retry —
139-
`executor-retries-failed-block` makes the first provider call reject, the block
140-
is replayed, and the run completes. Model fallback (`fallbackModels`) is not
141-
asserted yet. Further expansion (context/memory, model routing, subagent
142-
orchestration) is tracked as follow-up work.
140+
`DAGExecutor`. The executor suite covers both recovery paths — block retry
141+
(`executor-retries-failed-block`) and model fallback
142+
(`executor-falls-back-to-secondary-model`). Further expansion (context/memory,
143+
model routing, subagent orchestration) is tracked as follow-up work.

‎apps/sim/evals/agent-tool-use/executor-harness.ts‎

Lines changed: 41 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -56,6 +56,8 @@ export interface ExecutorScenario {
5656
temperature?: number
5757
/** Enables the executor's per-block retry policy for the Agent block. */
5858
retry?: { enabled: boolean; maxTries: number; waitBetweenTriesMs: number }
59+
/** Ordered models the Agent handler tries after the primary fails. */
60+
fallbackModels?: Array<{ model: string }>
5961
}
6062
/** One entry per model call; the last entry serves any extra/retry calls. */
6163
providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[]
@@ -66,6 +68,8 @@ export interface ExecutorScenario {
6668
succeeds?: boolean
6769
/** Exact number of provider calls the executor made. */
6870
providerCalls?: number
71+
/** Model id sent on the final provider call (proves which candidate served). */
72+
lastRequestModel?: string
6973
}
7074
}
7175

@@ -90,6 +94,7 @@ function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow {
9094
...(scenario.agent.temperature !== undefined
9195
? { temperature: scenario.agent.temperature }
9296
: {}),
97+
...(scenario.agent.fallbackModels ? { fallbackModels: scenario.agent.fallbackModels } : {}),
9398
}
9499
if (scenario.agent.retry) agent.retry = scenario.agent.retry
95100

@@ -198,6 +203,15 @@ export async function runExecutorScenario(
198203
})
199204
}
200205

206+
if (scenario.expect.lastRequestModel !== undefined) {
207+
const lastModel = (requests.at(-1) as { model?: string } | undefined)?.model
208+
checks.push({
209+
name: 'last-request-model',
210+
passed: lastModel === scenario.expect.lastRequestModel,
211+
detail: `expected ${scenario.expect.lastRequestModel}, got ${String(lastModel)}`,
212+
})
213+
}
214+
201215
const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number }
202216

203217
return {
@@ -307,4 +321,31 @@ export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [
307321
providerCalls: 2,
308322
},
309323
},
324+
{
325+
id: 'executor-falls-back-to-secondary-model',
326+
name: 'falls back to the secondary model when the primary fails',
327+
category: 'recovery',
328+
description:
329+
'The primary model call rejects and the Agent block has a fallback model. The handler must serve the answer from the fallback and the run must complete.',
330+
workflowInput: { message: 'What is the API rate limit?' },
331+
agent: {
332+
model: 'gpt-4o',
333+
userPrompt: 'What is the API rate limit?',
334+
fallbackModels: [{ model: 'gpt-4o-mini' }],
335+
},
336+
providerResponse: [
337+
{ reject: '429 rate limited', content: '' },
338+
{
339+
content: 'The API rate limit is 100 requests per minute.',
340+
model: 'gpt-4o-mini',
341+
tokens: { input: 10, output: 20, total: 30 },
342+
},
343+
],
344+
expect: {
345+
succeeds: true,
346+
finalContent: '100 requests per minute',
347+
providerCalls: 2,
348+
lastRequestModel: 'gpt-4o-mini',
349+
},
350+
},
310351
]

0 commit comments

Comments
 (0)