Skip to content

Commit fb7f54c

Browse files
committed
test(evals): assert executor block retry on agent failure
Add executor-retries-failed-block: the first provider call rejects, the Agent block has retry enabled, and the executor replays it. The run must complete with the second response. Verifies providerCalls === 2, and fails without the retry policy (checked locally: expected 2, got 1).
1 parent 85525a4 commit fb7f54c

2 files changed

Lines changed: 63 additions & 12 deletions

File tree

‎apps/sim/evals/README.md‎

Lines changed: 9 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -117,7 +117,10 @@ Add a case to `EXECUTOR_SCENARIOS` in `executor-harness.ts`:
117117
- `providerResponse` is what the mocked provider returns (`content`,
118118
`toolCalls`, `tokens`).
119119
- `expect` uses the loop's checks plus `resolvedInput` (a substring that must
120-
reach the provider messages) and `succeeds` (expected `ExecutionResult.success`).
120+
reach the provider messages), `succeeds` (expected `ExecutionResult.success`),
121+
and `providerCalls` (exact provider call count).
122+
- Set `agent.retry` to exercise the executor's per-block retry policy; make the
123+
first `providerResponse` a `reject` and the retry lands on the next one.
121124

122125
Both suites write one report, so executor rows appear alongside loop rows.
123126

@@ -132,8 +135,8 @@ model/tool time, first-response time, and token usage.
132135
## Scope and next steps
133136

134137
Two harnesses share one result shape and report: the tool loop and the
135-
`DAGExecutor`. The executor harness mocks the provider boundary, so the
136-
executor's retry/fallback policy is not yet asserted; add a scenario with a
137-
first-call rejection and a block retry config to cover it. Further expansion
138-
(context/memory, model routing, subagent orchestration) is tracked as
139-
follow-up work.
138+
`DAGExecutor`. The executor suite covers block retry —
139+
`executor-retries-failed-block` makes the first provider call reject, the block
140+
is replayed, and the run completes. Model fallback (`fallbackModels`) is not
141+
asserted yet. Further expansion (context/memory, model routing, subagent
142+
orchestration) is tracked as follow-up work.

‎apps/sim/evals/agent-tool-use/executor-harness.ts‎

Lines changed: 54 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -4,7 +4,7 @@ import {
44
} from '@sim/testing/factories/serialized-block.factory'
55
import { providersMockFns } from '@sim/testing/mocks/providers.mock'
66
import { DAGExecutor } from '@/executor/execution/executor'
7-
import type { SerializedWorkflow } from '@/serializer/types'
7+
import type { SerializedBlock, SerializedWorkflow } from '@/serializer/types'
88
import { type EvalRunMode, type ScoredToolCall, scoreExpectations } from './harness'
99
import type {
1010
AgentToolUseExpectations,
@@ -30,8 +30,10 @@ export interface ExecutorProviderToolCall {
3030
result?: unknown
3131
}
3232

33-
/** The provider response `executeProviderRequest` returns for one model call. */
33+
/** One model call: either a response, or a rejection the block must recover from. */
3434
export interface ExecutorProviderResponse {
35+
/** When set, the call rejects with this message instead of resolving. */
36+
reject?: string
3537
content: string
3638
model?: string
3739
tokens?: { input?: number; output?: number; total?: number }
@@ -52,26 +54,30 @@ export interface ExecutorScenario {
5254
systemPrompt?: string
5355
userPrompt?: string
5456
temperature?: number
57+
/** Enables the executor's per-block retry policy for the Agent block. */
58+
retry?: { enabled: boolean; maxTries: number; waitBetweenTriesMs: number }
5559
}
56-
/** One entry per model call; the last entry serves any extra fallback calls. */
60+
/** One entry per model call; the last entry serves any extra/retry calls. */
5761
providerResponse: ExecutorProviderResponse | ExecutorProviderResponse[]
5862
expect: AgentToolUseExpectations & {
5963
/** Substring that must appear in the messages sent to the provider. */
6064
resolvedInput?: string
6165
/** Expected `ExecutionResult.success`. */
6266
succeeds?: boolean
67+
/** Exact number of provider calls the executor made. */
68+
providerCalls?: number
6369
}
6470
}
6571

6672
function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow {
67-
const start = createSerializedBlock({
73+
const start: SerializedBlock = createSerializedBlock({
6874
id: 'start',
6975
type: 'start_trigger',
7076
name: 'Start',
7177
})
7278
/** The trigger handler claims a block whose metadata says it is a trigger. */
7379
if (start.metadata) start.metadata.category = 'triggers'
74-
const agent = createSerializedBlock({
80+
const agent: SerializedBlock = createSerializedBlock({
7581
id: 'agent',
7682
type: 'agent',
7783
name: 'Eval Agent',
@@ -85,6 +91,7 @@ function buildWorkflow(scenario: ExecutorScenario): SerializedWorkflow {
8591
? { temperature: scenario.agent.temperature }
8692
: {}),
8793
}
94+
if (scenario.agent.retry) agent.retry = scenario.agent.retry
8895

8996
return createSerializedWorkflow([start, agent], [{ source: 'start', target: 'agent' }])
9097
}
@@ -109,6 +116,7 @@ export async function runExecutorScenario(
109116
requests.push(request)
110117
const response = responses[Math.min(callIndex, responses.length - 1)]
111118
callIndex += 1
119+
if (response.reject) throw new Error(response.reject)
112120
return {
113121
content: response.content,
114122
model: response.model ?? scenario.agent.model,
@@ -156,7 +164,14 @@ export async function runExecutorScenario(
156164
durationMs: typeof call.duration === 'number' ? call.duration : 0,
157165
}))
158166

159-
const checks = scoreExpectations(scenario.expect, toolCalls, finalContent, 1, runError, mode)
167+
const checks = scoreExpectations(
168+
scenario.expect,
169+
toolCalls,
170+
finalContent,
171+
requests.length,
172+
runError,
173+
mode
174+
)
160175

161176
if (scenario.expect.resolvedInput !== undefined) {
162177
const sent = JSON.stringify(requests)
@@ -175,6 +190,14 @@ export async function runExecutorScenario(
175190
})
176191
}
177192

193+
if (scenario.expect.providerCalls !== undefined) {
194+
checks.push({
195+
name: 'provider-calls',
196+
passed: requests.length === scenario.expect.providerCalls,
197+
detail: `expected ${scenario.expect.providerCalls}, got ${requests.length}`,
198+
})
199+
}
200+
178201
const tokens = (output.tokens ?? {}) as { input?: number; output?: number; total?: number }
179202

180203
return {
@@ -259,4 +282,29 @@ export const EXECUTOR_SCENARIOS: ExecutorScenario[] = [
259282
finalContent: /A-1937/,
260283
},
261284
},
285+
{
286+
id: 'executor-retries-failed-block',
287+
name: 'retries a failed Agent block and completes the run',
288+
category: 'recovery',
289+
description:
290+
'The first provider call rejects with a 503. The block has retry enabled, so the executor replays it and the second call succeeds — proving the executor retry policy, not the agent handler, recovered the turn.',
291+
workflowInput: { message: 'What is the API rate limit?' },
292+
agent: {
293+
model: 'gpt-4o',
294+
userPrompt: 'What is the API rate limit?',
295+
retry: { enabled: true, maxTries: 3, waitBetweenTriesMs: 0 },
296+
},
297+
providerResponse: [
298+
{ reject: '503 Service Unavailable', content: '' },
299+
{
300+
content: 'The API rate limit is 100 requests per minute.',
301+
tokens: { input: 10, output: 20, total: 30 },
302+
},
303+
],
304+
expect: {
305+
succeeds: true,
306+
finalContent: '100 requests per minute',
307+
providerCalls: 2,
308+
},
309+
},
262310
]

0 commit comments

Comments
 (0)