Skip to content

Commit 9d74b8d

Browse files
committed
Remove benchmark cutoffs and renew active stage leases
1 parent 077803a commit 9d74b8d

13 files changed

Lines changed: 199 additions & 74 deletions

File tree

‎apps/sim/app/api/organizations/[id]/benchmarks/[benchmarkId]/run/route.ts‎

Lines changed: 0 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -9,8 +9,6 @@ import { requireBenchmarkOperator } from '@/lib/benchmarks/application/access'
99
import { benchmarkOperations } from '@/lib/benchmarks/application/operations'
1010
import { runBenchmarkStage } from '@/lib/benchmarks/application/run-stage'
1111

12-
export const maxDuration = 660
13-
1412
export const POST = defineInternalJsonRoute({
1513
contract: runBenchmarkStageContract,
1614
auth: internalSessionAuth,

‎apps/sim/app/o/[organizationId]/benchmark/components/benchmark-json.tsx‎

Lines changed: 2 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -12,7 +12,7 @@ import { getErrorMessage } from '@sim/utils/errors'
1212
import { isRecordLike } from '@sim/utils/object'
1313
import type { BenchmarkCase } from '@/lib/api/contracts/benchmarks'
1414
import { validateBenchmarkRedaction } from '@/lib/benchmarks/artifacts'
15-
import { BENCHMARK_SPEC_MAX_LENGTH, benchmarkArtifactsSchema } from '@/lib/benchmarks/types'
15+
import { benchmarkArtifactsSchema } from '@/lib/benchmarks/types'
1616

1717
interface BenchmarkJsonProps {
1818
artifacts: BenchmarkCase['artifacts']
@@ -21,8 +21,6 @@ interface BenchmarkJsonProps {
2121
onClose: () => void
2222
}
2323

24-
const MAPPING_MAX_LENGTH = BENCHMARK_SPEC_MAX_LENGTH * 6 + 5_000
25-
2624
export function BenchmarkJson({ artifacts, disabled, onApply, onClose }: BenchmarkJsonProps) {
2725
const [redactedSpec, setRedactedSpec] = useState(
2826
artifacts.redactedSpec || artifacts.referenceSpec
@@ -53,7 +51,7 @@ export function BenchmarkJson({ artifacts, disabled, onApply, onClose }: Benchma
5351
)
5452
if (!parsed.success) {
5553
throw new Error(
56-
'Use up to 50 blank IDs (letters, numbers, underscores or hyphens, at most 64 characters), each mapped to a nonempty string of at most 10,000 characters.'
54+
'Use blank IDs with letters, numbers, underscores or hyphens (at most 64 characters), each mapped to a nonempty answer string.'
5755
)
5856
}
5957
const patch = { redactedSpec, blanks: parsed.data }
@@ -82,7 +80,6 @@ export function BenchmarkJson({ artifacts, disabled, onApply, onClose }: Benchma
8280
setRedactedSpec(value)
8381
setError(null)
8482
}}
85-
maxLength={BENCHMARK_SPEC_MAX_LENGTH}
8683
rows={6}
8784
resizable
8885
disabled={disabled}
@@ -99,7 +96,6 @@ export function BenchmarkJson({ artifacts, disabled, onApply, onClose }: Benchma
9996
placeholder={
10097
'{\n "queue": "Customer Escalations",\n "handoff": "Engineering explicitly accepts the case"\n}'
10198
}
102-
maxLength={MAPPING_MAX_LENGTH}
10399
rows={10}
104100
mono
105101
resizable

‎apps/sim/app/o/[organizationId]/benchmark/components/benchmark-reference.tsx‎

Lines changed: 0 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -5,7 +5,6 @@ import { Chip, ChipInput, ChipTextarea, toast } from '@sim/emcn'
55
import { Code, Plus, Trash, Upload } from '@sim/emcn/icons'
66
import { getErrorMessage } from '@sim/utils/errors'
77
import type { BenchmarkCase, RunBenchmarkStageBody } from '@/lib/api/contracts/benchmarks'
8-
import { BENCHMARK_MAX_BLANKS, BENCHMARK_SPEC_MAX_LENGTH } from '@/lib/benchmarks/types'
98
import { BenchmarkJson } from '@/app/o/[organizationId]/benchmark/components/benchmark-json'
109
import { BenchmarkStep } from '@/app/o/[organizationId]/benchmark/components/benchmark-step'
1110

@@ -40,13 +39,7 @@ export function BenchmarkReference({
4039
const importReference = async (file: File | undefined) => {
4140
if (!file) return
4241
try {
43-
if (file.size > BENCHMARK_SPEC_MAX_LENGTH * 4) {
44-
throw new Error('The reference is too large. Use a text file under 4 MB.')
45-
}
4642
const text = await file.text()
47-
if (text.length > BENCHMARK_SPEC_MAX_LENGTH) {
48-
throw new Error('The reference must be at most 1,000,000 characters.')
49-
}
5043
onChange({ referenceSpec: text })
5144
} catch (error) {
5245
toast.error(getErrorMessage(error, 'Could not import the reference'))
@@ -79,7 +72,6 @@ export function BenchmarkReference({
7972
value={taskBrief}
8073
onChange={(event) => onChange({ taskBrief: event.target.value })}
8174
placeholder='What should the new Mothership plan?'
82-
maxLength={20_000}
8375
rows={4}
8476
resizable
8577
/>
@@ -113,7 +105,6 @@ export function BenchmarkReference({
113105
value={referenceSpec}
114106
onChange={(event) => onChange({ referenceSpec: event.target.value })}
115107
placeholder='Generate a spec, paste one, or import Markdown, text, or JSON.'
116-
maxLength={BENCHMARK_SPEC_MAX_LENGTH}
117108
rows={12}
118109
resizable
119110
/>
@@ -158,7 +149,6 @@ export function BenchmarkReference({
158149
aria-label='Redacted reference spec'
159150
value={redactedSpec}
160151
onChange={(event) => onChange({ redactedSpec: event.target.value })}
161-
maxLength={BENCHMARK_SPEC_MAX_LENGTH}
162152
rows={10}
163153
resizable
164154
/>
@@ -188,7 +178,6 @@ export function BenchmarkReference({
188178
<ChipTextarea
189179
aria-label={`Expected answer for blank ${index + 1}`}
190180
value={blank.answer}
191-
maxLength={10_000}
192181
rows={2}
193182
resizable
194183
onChange={(event) =>
@@ -215,7 +204,6 @@ export function BenchmarkReference({
215204
<div>
216205
<Chip
217206
leftIcon={Plus}
218-
disabled={blanks.length >= BENCHMARK_MAX_BLANKS}
219207
onClick={() => {
220208
let suffix = blanks.length + 1
221209
while (blanks.some((blank) => blank.id === `detail_${suffix}`)) suffix += 1

‎apps/sim/app/o/[organizationId]/benchmark/components/create-benchmark.tsx‎

Lines changed: 0 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -82,7 +82,6 @@ export function CreateBenchmark({ organizationId, runAsUserId, onCreated }: Crea
8282
onChange={(event) => setTaskBrief(event.target.value)}
8383
placeholder='Paste the original request, or generate a brief from the workspace in step 1.'
8484
rows={4}
85-
maxLength={20_000}
8685
resizable
8786
disabled={createBenchmark.isPending}
8887
/>

‎apps/sim/lib/benchmarks/README.md‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -23,9 +23,9 @@ Grading opens the saved report for human review. On any detail, choose **Mark as
2323

2424
Reference specs can be prose or JSON text; **Import spec** accepts Markdown, text, and JSON files without reformatting their contents. **Generate blanks with AI** creates the redacted reference and expected answers together. **Edit JSON** also lets you paste or edit a mapping such as `{"queue":"Customer Escalations","handoff":"Engineering explicitly accepts the case"}` alongside the matching redacted reference. Keys are blank IDs and values are exact original passages as strings. Applying validates that the mapping restores the reference exactly and updates the draft; **Save changes** persists it.
2525

26-
Every stage is limited to ten minutes, with a twelve-minute persistence lease. Concurrent starts or stale completions cannot overwrite newer results. Interrupted steps are retryable; a hard server failure becomes retryable when its lease expires. This first version runs a stage within its HTTP request, so the deployment's request timeout must accommodate the run. Page reload may interrupt a pending request; saved completed stages remain available.
26+
Benchmark stages have no benchmark-specific duration cutoff. The active process renews its persistence lease every 30 seconds; a two-minute lease detects an abandoned process rather than limiting a healthy run. Concurrent starts or stale completions cannot overwrite newer results. Interrupted steps are retryable; a hard server failure becomes retryable when its lease expires. This version runs a stage within its HTTP request, so the deployment must support long-lived requests. Page reload may interrupt a pending request; saved completed stages remain available.
2727

28-
Reference generation uses the existing Mothership agent loop and workspace CLI to inspect active workflows and supporting resources incrementally. There is no benchmark-wide workflow-count or export-size gate. Reads remain authorized as the selected user; the benchmark transport refuses source mutations, workflow execution, services and scratch writes. Existing CLI pagination and searchable stored outputs keep individual model inputs bounded. Specs can contain up to 1,000,000 characters; structured output uses the worker’s 32,768-token per-response limit. The planner uses the dev worker's model configuration; keep worker/model configuration constant when comparing agent changes.
28+
Reference generation uses the existing Mothership agent loop and workspace CLI to inspect active workflows and supporting resources incrementally. There is no benchmark-wide workflow-count or export-size gate. Reads remain authorized as the selected user; the benchmark transport refuses source mutations, workflow execution, services and scratch writes. Existing CLI pagination and searchable stored outputs keep individual model inputs bounded. Benchmark-specific caps on spec length, blank count, answer length and model output are removed. The normal model/provider and transport constraints still apply; page sizes bound individual tool reads without limiting the complete workspace or document. The planner uses the dev worker's model configuration; keep worker/model configuration constant when comparing agent changes.
2929

3030
This version uses **live enterprise context**, not a frozen historical snapshot. The completed Sim workspace is excluded from planning, but historical solution documents in connected enterprise sources can still reveal answers. Treat these cases as retrospective evaluations and review source availability. The score measures recovery of the selected requirements, not execution correctness or every claim in the plan.
3131

‎apps/sim/lib/benchmarks/application/run-stage.ts‎

Lines changed: 16 additions & 17 deletions
Original file line numberDiff line numberDiff line change
@@ -6,6 +6,10 @@ import { defineAuthorizedBenchmarkUseCase } from '@/lib/benchmarks/application/a
66
import { requireBenchmarkCaseAccess } from '@/lib/benchmarks/application/cases'
77
import { benchmarkOperations } from '@/lib/benchmarks/application/operations'
88
import { prepareBenchmarkReference } from '@/lib/benchmarks/application/prepare-reference'
9+
import {
10+
BENCHMARK_LEASE_MS,
11+
withBenchmarkStageLease,
12+
} from '@/lib/benchmarks/application/stage-lease'
913
import { applyBenchmarkPatch, validateBenchmarkRedaction } from '@/lib/benchmarks/artifacts'
1014
import { getBenchmarkMothershipUrl } from '@/lib/benchmarks/config'
1115
import { gradeReconstruction, validateReconstruction } from '@/lib/benchmarks/evaluation'
@@ -21,7 +25,6 @@ import {
2125
failBenchmarkStage,
2226
} from '@/lib/benchmarks/repository'
2327
import {
24-
BENCHMARK_MAX_BLANKS,
2528
type BenchmarkArtifacts,
2629
type BenchmarkCase,
2730
type BenchmarkStage,
@@ -35,24 +38,20 @@ import { executeBenchmarkJson, executeBenchmarkPlan } from '@/lib/benchmarks/wor
3538
import { OrchestrationError } from '@/lib/core/orchestration/types'
3639

3740
const logger = createLogger('BenchmarkStage')
38-
const STAGE_TIMEOUT_MS = 10 * 60 * 1000
39-
const STAGE_LEASE_MS = STAGE_TIMEOUT_MS + 2 * 60 * 1000
4041

4142
const distillationSchema = z
4243
.object({ taskBrief: benchmarkBriefSchema.min(1), referenceSpec: benchmarkSpecSchema.min(1) })
4344
.strict()
4445
const redactionSchema = z
4546
.object({
4647
redactedSpec: benchmarkSpecSchema.min(1),
47-
blanks: z.array(benchmarkBlankSchema).min(1).max(BENCHMARK_MAX_BLANKS),
48+
blanks: z.array(benchmarkBlankSchema).min(1),
4849
})
4950
.strict()
5051
const reconstructionSchema = z
51-
.object({ answers: z.array(benchmarkReconstructionSchema).min(1).max(BENCHMARK_MAX_BLANKS) })
52-
.strict()
53-
const gradingSchema = z
54-
.object({ judgments: z.array(benchmarkGradeSchema).min(1).max(BENCHMARK_MAX_BLANKS) })
52+
.object({ answers: z.array(benchmarkReconstructionSchema).min(1) })
5553
.strict()
54+
const gradingSchema = z.object({ judgments: z.array(benchmarkGradeSchema).min(1) }).strict()
5655

5756
interface RunBenchmarkStageInput {
5857
organizationId: string
@@ -186,19 +185,19 @@ export const runBenchmarkStage = defineAuthorizedBenchmarkUseCase({
186185
expectedVersion: input.version,
187186
stage: input.stage,
188187
attemptId,
189-
leaseExpiresAt: new Date(Date.now() + STAGE_LEASE_MS),
188+
leaseExpiresAt: new Date(Date.now() + BENCHMARK_LEASE_MS),
190189
})
191190
const attempt = { ...scope, version: claimed.version, stage: input.stage, attemptId }
192191
logger.info('Benchmark step started', {
193192
...attempt,
194193
operatorUserId: current.userId,
195194
runAsUserId: current.runAsUserId ?? current.userId,
196195
})
197-
const timeout = AbortSignal.timeout(STAGE_TIMEOUT_MS)
198-
const signal = request?.signal ? AbortSignal.any([request.signal, timeout]) : timeout
199196
try {
200-
const output = await performStage(principal, claimed, input.stage, signal)
201-
signal.throwIfAborted()
197+
const output = await withBenchmarkStageLease(attempt, request?.signal, (signal) =>
198+
performStage(principal, claimed, input.stage, signal)
199+
)
200+
request?.signal?.throwIfAborted()
202201
await requireBenchmarkCaseAccess(principal, input)
203202
return {
204203
benchmark: await completeBenchmarkStage({
@@ -208,8 +207,8 @@ export const runBenchmarkStage = defineAuthorizedBenchmarkUseCase({
208207
}),
209208
}
210209
} catch (error) {
211-
const message = signal.aborted
212-
? 'This step was interrupted or exceeded ten minutes. Retry it.'
210+
const message = request?.signal?.aborted
211+
? 'This step was interrupted. Retry it.'
213212
: error instanceof OrchestrationError
214213
? error.message
215214
: 'This benchmark step failed. Retry it or inspect the server logs.'
@@ -229,8 +228,8 @@ export const runBenchmarkStage = defineAuthorizedBenchmarkUseCase({
229228
errorType: persistenceError instanceof Error ? persistenceError.name : 'UnknownError',
230229
})
231230
}
232-
if (!signal.aborted && error instanceof OrchestrationError) throw error
233-
throw new OrchestrationError(signal.aborted ? 'validation' : 'internal', message)
231+
if (!request?.signal?.aborted && error instanceof OrchestrationError) throw error
232+
throw new OrchestrationError(request?.signal?.aborted ? 'validation' : 'internal', message)
234233
}
235234
},
236235
})
Lines changed: 58 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,58 @@
1+
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
2+
3+
const { renew } = vi.hoisted(() => ({ renew: vi.fn() }))
4+
vi.mock('@/lib/benchmarks/repository', () => ({ renewBenchmarkStage: renew }))
5+
6+
import { withBenchmarkStageLease } from '@/lib/benchmarks/application/stage-lease'
7+
8+
/** Covers healthy runs beyond the old cutoff, loss of ownership, and caller cancellation. */
9+
describe('benchmark stage lifetime', () => {
10+
const attempt = {
11+
organizationId: 'org',
12+
userId: 'operator',
13+
benchmarkId: 'benchmark',
14+
version: 2,
15+
stage: 'distill' as const,
16+
attemptId: 'attempt',
17+
}
18+
19+
beforeEach(() => {
20+
vi.useFakeTimers()
21+
renew.mockResolvedValue(true)
22+
})
23+
afterEach(() => vi.useRealTimers())
24+
25+
it('keeps a healthy inspection alive beyond ten minutes and stops its heartbeat on completion', async () => {
26+
const result = withBenchmarkStageLease(attempt, undefined, async (signal) => {
27+
await vi.advanceTimersByTimeAsync(60 * 60_000)
28+
signal.throwIfAborted()
29+
return 'complete reference'
30+
})
31+
await expect(result).resolves.toBe('complete reference')
32+
expect(vi.getTimerCount()).toBe(0)
33+
})
34+
35+
it.each([false, new Error('database unavailable')])(
36+
'aborts the inspection when its ownership cannot be renewed (%s)',
37+
async (outcome) => {
38+
if (outcome instanceof Error) renew.mockRejectedValue(outcome)
39+
else renew.mockResolvedValue(outcome)
40+
const result = withBenchmarkStageLease(attempt, undefined, async (signal) => {
41+
await vi.advanceTimersByTimeAsync(60_000)
42+
signal.throwIfAborted()
43+
}).catch((error: unknown) => error)
44+
await expect(result).resolves.toMatchObject({ code: 'conflict' })
45+
expect(vi.getTimerCount()).toBe(0)
46+
}
47+
)
48+
49+
it('propagates caller cancellation and clears its heartbeat when work throws', async () => {
50+
const controller = new AbortController()
51+
const result = withBenchmarkStageLease(attempt, controller.signal, async (signal) => {
52+
controller.abort(new Error('cancelled by caller'))
53+
signal.throwIfAborted()
54+
})
55+
await expect(result).rejects.toThrow('cancelled by caller')
56+
expect(vi.getTimerCount()).toBe(0)
57+
})
58+
})
Lines changed: 48 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,48 @@
1+
import { type BenchmarkAttempt, renewBenchmarkStage } from '@/lib/benchmarks/repository'
2+
import { OrchestrationError } from '@/lib/core/orchestration/types'
3+
4+
/** This is crash detection, not a deadline: the current owner continually extends its lease. */
5+
export const BENCHMARK_LEASE_MS = 2 * 60_000
6+
const HEARTBEAT_MS = 30_000
7+
8+
export async function withBenchmarkStageLease<T>(
9+
attempt: BenchmarkAttempt,
10+
callerSignal: AbortSignal | undefined,
11+
execute: (signal: AbortSignal) => Promise<T>
12+
): Promise<T> {
13+
const ownership = new AbortController()
14+
const signal = callerSignal ? AbortSignal.any([callerSignal, ownership.signal]) : ownership.signal
15+
let pending: Promise<void> | undefined
16+
const heartbeat = setInterval(() => {
17+
if (pending || signal.aborted) return
18+
pending = renewBenchmarkStage({
19+
...attempt,
20+
leaseExpiresAt: new Date(Date.now() + BENCHMARK_LEASE_MS),
21+
})
22+
.then((renewed) => {
23+
if (!renewed) throw new Error('Benchmark lease lost')
24+
})
25+
.catch(() => {
26+
ownership.abort(
27+
new OrchestrationError(
28+
'conflict',
29+
'This step lost its run ownership. Refresh and retry it.'
30+
)
31+
)
32+
})
33+
.finally(() => {
34+
pending = undefined
35+
})
36+
}, HEARTBEAT_MS)
37+
heartbeat.unref?.()
38+
try {
39+
signal.throwIfAborted()
40+
const result = await execute(signal)
41+
await pending
42+
signal.throwIfAborted()
43+
return result
44+
} finally {
45+
clearInterval(heartbeat)
46+
await pending
47+
}
48+
}

‎apps/sim/lib/benchmarks/repository.integration.ts‎

Lines changed: 33 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -42,6 +42,7 @@ import {
4242
getBenchmarkRecord,
4343
getBenchmarkRunRecord,
4444
listBenchmarkRunRecords,
45+
renewBenchmarkStage,
4546
updateBenchmarkRecord,
4647
} from '@/lib/benchmarks/repository'
4748
import { type BenchmarkArtifacts, emptyBenchmarkArtifacts } from '@/lib/benchmarks/types'
@@ -359,6 +360,15 @@ describe('private benchmark persistence and attempt fencing', () => {
359360
attemptId: 'next',
360361
leaseExpiresAt: new Date(Date.now() + 60_000),
361362
})
363+
expect(
364+
await renewBenchmarkStage({
365+
...scope,
366+
version: old.version,
367+
stage: 'plan',
368+
attemptId: 'old',
369+
leaseExpiresAt: new Date(Date.now() + 120_000),
370+
})
371+
).toBe(false)
362372
await expect(
363373
completeBenchmarkStage({
364374
...scope,
@@ -388,6 +398,29 @@ describe('private benchmark persistence and attempt fencing', () => {
388398
expect(completed.runningStage).toBeNull()
389399
})
390400

401+
it('extends only the current live lease without invalidating the attempt version', async () => {
402+
const claimed = await claimBenchmarkStage({
403+
...scope,
404+
expectedVersion: 1,
405+
stage: 'distill',
406+
attemptId: 'long-inspection',
407+
leaseExpiresAt: new Date(Date.now() + 60_000),
408+
})
409+
const attempt = {
410+
...scope,
411+
version: claimed.version,
412+
stage: 'distill' as const,
413+
attemptId: 'long-inspection',
414+
}
415+
const leaseExpiresAt = new Date(Date.now() + 120_000)
416+
expect(await renewBenchmarkStage({ ...attempt, leaseExpiresAt })).toBe(true)
417+
const current = await getBenchmarkRecord(scope)
418+
expect(current.version).toBe(claimed.version)
419+
expect(current.leaseExpiresAt).toBe(leaseExpiresAt.toISOString())
420+
await completeBenchmarkStage({ ...attempt, artifacts })
421+
expect(await renewBenchmarkStage({ ...attempt, leaseExpiresAt })).toBe(false)
422+
})
423+
391424
it('keeps artifacts private even from another member who can read the same workspace', async () => {
392425
await expect(
393426
getBenchmark.execute({

0 commit comments

Comments
 (0)