Skip to content

Commit a7629e8

Browse files
1stvampTrigger.dev RepoOps
authored andcommitted
Replace the Go compute runtime with Rust, with run-CRD suspend and resume
Supervisor: with the Runner CRD backend on the microvm runtime, suspends are requested on the Runner and completed from its status, a completed suspend is still delivered after a supervisor restart, and a restored run is placed on the node that holds its snapshot. A restore whose Runner has already ended is reported as failed. Mono-RevId: 2707a91994f6ab5fb2f7f9d25e575eb21ff78a41
1 parent 9730004 commit a7629e8

10 files changed

Lines changed: 1938 additions & 309 deletions

File tree

‎apps/supervisor/src/env.test.ts‎

Lines changed: 13 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -62,3 +62,16 @@ describe("Env superRefine - backpressure source awareness", () => {
6262
).toThrow();
6363
});
6464
});
65+
66+
describe("Env superRefine - compute snapshots", () => {
67+
it("needs the metadata URL and workload API domain only for the gateway", () => {
68+
expect(() => Env.parse({ ...base, COMPUTE_SNAPSHOTS_ENABLED: "true" })).not.toThrow();
69+
expect(() =>
70+
Env.parse({
71+
...base,
72+
COMPUTE_SNAPSHOTS_ENABLED: "true",
73+
COMPUTE_GATEWAY_URL: "http://gateway:8080",
74+
})
75+
).toThrow(/TRIGGER_METADATA_URL[\s\S]*TRIGGER_WORKLOAD_API_DOMAIN/);
76+
});
77+
});

‎apps/supervisor/src/env.ts‎

Lines changed: 17 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -179,17 +179,14 @@ export const Env = z
179179

180180
// Kubernetes settings
181181
KUBERNETES_FORCE_ENABLED: BoolEnv.default(false),
182-
// Create a Runner and let the operator build the pod, instead of building
183-
// the pod here. The two are mutually exclusive by construction: whichever
184-
// one runs is the only thing that creates a workload for a cold start.
182+
// Create a Runner for the operator to build the pod from, instead of building it
183+
// here. Mutually exclusive with the pod backend, so one thing creates each cold start.
185184
KUBERNETES_RUN_CRD_ENABLED: BoolEnv.default(false),
186-
// Which isolation lane a Runner asks for. Cell-wide rather than per run,
187-
// because a cell's node pools decide what it can serve and nothing on a
188-
// dequeued message can express the choice. Ignored unless the run-crd
189-
// backend is the one running: the pod backends have no guest lane.
190-
//
191-
// Not to be confused with the task runtime (node-24, bun), which rides on
192-
// the same Runner as taskRuntime and says which interpreter the task needs.
185+
// Isolation lane for Runners, cell-wide because a cell's node pools decide what it
186+
// can serve. Only read by the run-crd backend; the pod backends have no guest lane.
187+
// Under microvm, checkpoints resume as restoring Runners, since only that lane's node
188+
// runtime can restore, and COMPUTE_SNAPSHOTS_ENABLED sends suspends to the operator.
189+
// Unrelated to the task runtime (node-24, bun), which is the Runner's taskRuntime.
193190
KUBERNETES_RUNNER_RUNTIME: z.enum(["container", "microvm"]).default("container"),
194191
KUBERNETES_NAMESPACE: z.string().default("default"),
195192
KUBERNETES_WORKER_NODETYPE_LABEL: NodeLabelValue.default("v4-worker"),
@@ -236,13 +233,13 @@ export const Env = z
236233
KUBERNETES_RUNNER_SECURITY_CONTEXT: z.enum(["off", "baseline", "restricted"]).default("off"),
237234
KUBERNETES_RUNNER_RUN_AS_USER: z.coerce.number().int().min(1).default(1000),
238235

239-
// Pod DNS config — override the cluster default ndots to `KUBERNETES_POD_DNS_NDOTS`.
236+
// Pod DNS config: override the cluster default ndots to `KUBERNETES_POD_DNS_NDOTS`.
240237
// Default k8s ndots is 5: any name with fewer than 5 dots (e.g. `api.example.com`, 2 dots) is first walked
241238
// through every entry in the cluster search list (`<ns>.svc.cluster.local`, `svc.cluster.local`, `cluster.local`)
242239
// before being tried as-is, turning one resolution into 4+ CoreDNS queries (×2 with A+AAAA).
243240
// Overriding the default can be useful to cut CoreDNS query amplification for external domains.
244241
// Note: before enabling, make sure no code path relies on search-list expansion for names with dots ≥ the value
245-
// set here — those names will now hit their as-is form first and could resolve externally before falling back.
242+
// set here, since those names will now hit their as-is form first and could resolve externally before falling back.
246243
KUBERNETES_POD_DNS_NDOTS_OVERRIDE_ENABLED: BoolEnv.default(false),
247244
KUBERNETES_POD_DNS_NDOTS: z.coerce.number().int().min(1).max(15).default(2),
248245
// Large machine affinity settings - large-* presets prefer a dedicated pool
@@ -357,14 +354,20 @@ export const Env = z
357354
}
358355
}
359356
}
360-
if (data.COMPUTE_SNAPSHOTS_ENABLED && !data.TRIGGER_METADATA_URL) {
357+
// Only the gateway reads these for a snapshot: it builds its callback URL
358+
// from the domain and hands the metadata URL to the instance.
359+
if (data.COMPUTE_SNAPSHOTS_ENABLED && data.COMPUTE_GATEWAY_URL && !data.TRIGGER_METADATA_URL) {
361360
ctx.addIssue({
362361
code: z.ZodIssueCode.custom,
363362
message: "TRIGGER_METADATA_URL is required when COMPUTE_SNAPSHOTS_ENABLED is true",
364363
path: ["TRIGGER_METADATA_URL"],
365364
});
366365
}
367-
if (data.COMPUTE_SNAPSHOTS_ENABLED && !data.TRIGGER_WORKLOAD_API_DOMAIN) {
366+
if (
367+
data.COMPUTE_SNAPSHOTS_ENABLED &&
368+
data.COMPUTE_GATEWAY_URL &&
369+
!data.TRIGGER_WORKLOAD_API_DOMAIN
370+
) {
368371
ctx.addIssue({
369372
code: z.ZodIssueCode.custom,
370373
message: "TRIGGER_WORKLOAD_API_DOMAIN is required when COMPUTE_SNAPSHOTS_ENABLED is true",

‎apps/supervisor/src/index.ts‎

Lines changed: 99 additions & 54 deletions
Original file line numberDiff line numberDiff line change
@@ -3,7 +3,11 @@ import { SimpleStructuredLogger } from "@trigger.dev/core/v3/utils/structuredLog
33
import { formatLogLine, startTelnetLogServer } from "@trigger.dev/core/v3/telnetLogServer";
44
import { env } from "./env.js";
55
import { WorkloadServer } from "./workloadServer/index.js";
6-
import type { WorkloadManagerOptions, WorkloadManager } from "./workloadManager/types.js";
6+
import type {
7+
WorkloadManagerCreateOptions,
8+
WorkloadManagerOptions,
9+
WorkloadManager,
10+
} from "./workloadManager/types.js";
711
import Docker from "dockerode";
812
import { z } from "zod";
913
import { type DequeuedMessage } from "@trigger.dev/core/v3";
@@ -84,6 +88,7 @@ class ManagedSupervisor {
8488
private readonly workloadManager: WorkloadManager;
8589
private readonly workloadManagerBackend: "compute" | "kubernetes" | "run-crd" | "docker";
8690
private readonly computeManager?: ComputeWorkloadManager;
91+
private readonly runCrdManager?: RunCrdWorkloadManager;
8792
private readonly logger = new SimpleStructuredLogger("managed-supervisor");
8893
private readonly resourceMonitor: ResourceMonitor;
8994
private readonly checkpointClient?: CheckpointClient;
@@ -192,11 +197,18 @@ class ManagedSupervisor {
192197
this.workloadManager = computeManager;
193198
this.workloadManagerBackend = "compute";
194199
} else if (this.isKubernetes && env.KUBERNETES_RUN_CRD_ENABLED) {
195-
this.workloadManager = new RunCrdWorkloadManager({
200+
const runCrdManager = new RunCrdWorkloadManager({
196201
...workloadManagerOptions,
197202
namespace: env.KUBERNETES_NAMESPACE,
198203
runtime: env.KUBERNETES_RUNNER_RUNTIME,
204+
snapshots: {
205+
enabled: env.COMPUTE_SNAPSHOTS_ENABLED,
206+
delayMs: env.COMPUTE_SNAPSHOT_DELAY_MS,
207+
dispatchLimit: env.COMPUTE_SNAPSHOT_DISPATCH_LIMIT,
208+
},
199209
});
210+
this.runCrdManager = runCrdManager;
211+
this.workloadManager = runCrdManager;
200212
this.workloadManagerBackend = "run-crd";
201213
} else if (this.isKubernetes) {
202214
this.workloadManager = new KubernetesWorkloadManager(workloadManagerOptions);
@@ -521,6 +533,11 @@ class ManagedSupervisor {
521533
return;
522534
}
523535

536+
if (this.runCrdManager?.restores(checkpoint)) {
537+
await this.restoreRunner(this.runCrdManager, message, checkpoint);
538+
return;
539+
}
540+
524541
if (!this.checkpointClient) {
525542
this.logger.error("No checkpoint client", { runId: message.run.id });
526543
return;
@@ -608,6 +625,7 @@ class ManagedSupervisor {
608625
workerClient: this.workerSession.httpClient,
609626
checkpointClient: this.checkpointClient,
610627
computeManager: this.computeManager,
628+
runnerSnapshotter: this.runCrdManager,
611629
tracing: this.tracing,
612630
snapshotCallbackSecret: workerToken,
613631
wideEventOpts: this.wideEventOpts,
@@ -630,57 +648,86 @@ class ManagedSupervisor {
630648
this.workerSession.unsubscribeFromRunNotifications([run.friendlyId]);
631649
}
632650

633-
private async createWorkload(message: DequeuedMessage, timings: WarmStartTimings) {
634-
const createStart = performance.now();
635-
try {
636-
if (!message.deployment.friendlyId) {
637-
// mostly a type guard, deployments always exists for deployed environments
638-
// a proper fix would be to use a discriminated union schema to differentiate between dequeued runs in dev and in deployed environments.
639-
throw new Error("Deployment is missing");
640-
}
651+
private async createOptionsFor(
652+
message: DequeuedMessage,
653+
timings?: WarmStartTimings
654+
): Promise<WorkloadManagerCreateOptions> {
655+
if (!message.deployment.friendlyId) {
656+
// mostly a type guard, deployments always exists for deployed environments
657+
// a proper fix would be to use a discriminated union schema to differentiate between dequeued runs in dev and in deployed environments.
658+
throw new Error("Deployment is missing");
659+
}
641660

642-
if (!message.image) {
643-
// same type-guard situation as deployment above
644-
throw new Error("Image is missing");
645-
}
661+
if (!message.image) {
662+
// same type-guard situation as deployment above
663+
throw new Error("Image is missing");
664+
}
646665

647-
const deploymentToken = await mintDeploymentToken({
648-
deployment: message.deployment.friendlyId,
649-
deployment_version: message.backgroundWorker.version,
650-
environment_id: message.environment.id,
651-
environment_type: message.environment.type,
652-
org_id: message.organization.id,
653-
project_id: message.project.id,
654-
});
666+
const deploymentToken = await mintDeploymentToken({
667+
deployment: message.deployment.friendlyId,
668+
deployment_version: message.backgroundWorker.version,
669+
environment_id: message.environment.id,
670+
environment_type: message.environment.type,
671+
org_id: message.organization.id,
672+
project_id: message.project.id,
673+
});
655674

656-
await this.workloadManager.create({
657-
dequeuedAt: message.dequeuedAt,
658-
dequeueResponseMs: timings.dequeueResponseMs,
659-
pollingIntervalMs: timings.pollingIntervalMs,
660-
warmStartCheckMs: timings.warmStartCheckMs,
661-
envId: message.environment.id,
662-
envType: message.environment.type,
663-
image: message.image,
664-
machine: message.run.machine,
665-
orgId: message.organization.id,
666-
projectId: message.project.id,
667-
deploymentFriendlyId: message.deployment.friendlyId,
668-
deploymentVersion: message.backgroundWorker.version,
669-
runtime: message.backgroundWorker.runtime,
670-
deploymentToken,
671-
runId: message.run.id,
672-
runFriendlyId: message.run.friendlyId,
673-
version: message.version,
674-
nextAttemptNumber: message.run.attemptNumber,
675-
snapshotId: message.snapshot.id,
676-
snapshotFriendlyId: message.snapshot.friendlyId,
677-
// Carry the run's storage route to the cold-start pod so its start request echoes it back.
678-
snapshotRoute: message.snapshotRoute,
679-
placementTags: message.placementTags,
680-
traceContext: message.run.traceContext,
681-
annotations: message.run.annotations,
682-
hasPrivateLink: message.organization.hasPrivateLink,
683-
});
675+
return {
676+
dequeuedAt: message.dequeuedAt,
677+
dequeueResponseMs: timings?.dequeueResponseMs,
678+
pollingIntervalMs: timings?.pollingIntervalMs,
679+
warmStartCheckMs: timings?.warmStartCheckMs,
680+
envId: message.environment.id,
681+
envType: message.environment.type,
682+
image: message.image,
683+
machine: message.run.machine,
684+
orgId: message.organization.id,
685+
projectId: message.project.id,
686+
deploymentFriendlyId: message.deployment.friendlyId,
687+
deploymentVersion: message.backgroundWorker.version,
688+
runtime: message.backgroundWorker.runtime,
689+
deploymentToken,
690+
runId: message.run.id,
691+
runFriendlyId: message.run.friendlyId,
692+
version: message.version,
693+
nextAttemptNumber: message.run.attemptNumber,
694+
snapshotId: message.snapshot.id,
695+
snapshotFriendlyId: message.snapshot.friendlyId,
696+
// Carry the run's storage route to the runner pod so its start request echoes it back.
697+
snapshotRoute: message.snapshotRoute,
698+
placementTags: message.placementTags,
699+
traceContext: message.run.traceContext,
700+
annotations: message.run.annotations,
701+
hasPrivateLink: message.organization.hasPrivateLink,
702+
};
703+
}
704+
705+
private async restoreRunner(
706+
manager: RunCrdWorkloadManager,
707+
message: DequeuedMessage,
708+
checkpoint: { id: string; location: string }
709+
) {
710+
const restoreStart = performance.now();
711+
try {
712+
await manager.restore(await this.createOptionsFor(message), checkpoint);
713+
recordPhaseSince("restore", restoreStart, undefined);
714+
setExtra(fromContext(), "did_restore", true);
715+
this.logger.debug("Runner restore created", { runId: message.run.id });
716+
} catch (error) {
717+
recordPhaseSince(
718+
"restore",
719+
restoreStart,
720+
error instanceof Error ? error : new Error(String(error))
721+
);
722+
setExtra(fromContext(), "did_restore", false);
723+
this.logger.error("Failed to restore run (run-crd)", { runId: message.run.id, error });
724+
}
725+
}
726+
727+
private async createWorkload(message: DequeuedMessage, timings: WarmStartTimings) {
728+
const createStart = performance.now();
729+
try {
730+
await this.workloadManager.create(await this.createOptionsFor(message, timings));
684731
recordPhaseSince("workload_create", createStart, undefined);
685732
workloadCreateDuration.observe(
686733
{ backend: this.workloadManagerBackend, outcome: "success" },
@@ -722,10 +769,8 @@ class ManagedSupervisor {
722769
const headers: Record<string, string> = {
723770
"Content-Type": "application/json",
724771
};
725-
// Propagate the inbound W3C traceparent so the upstream warm-start
726-
// receiver continues the same trace instead of minting a new one. Gated
727-
// by the same kill switch as the wide-event emission so the whole PR is
728-
// a no-op on the wire when disabled.
772+
// Lets the warm-start receiver continue this trace instead of minting a new one.
773+
// Gated by the wide-events kill switch, so the wire is unchanged when disabled.
729774
if (this.wideEventOpts.enabled && traceparent) {
730775
headers.traceparent = traceparent;
731776
}

0 commit comments

Comments
 (0)