@agent-compose/sdk 0.2.3 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/README.md +145 -33
  2. package/dist/agent/agent-loop.d.ts +83 -5
  3. package/dist/agent/run-agent.d.ts +34 -9
  4. package/dist/client.d.ts +247 -99
  5. package/dist/index.d.ts +26 -11
  6. package/dist/index.js +1967 -745
  7. package/dist/processors/builtins.d.ts +35 -0
  8. package/dist/processors/index.d.ts +4 -0
  9. package/dist/processors/processor.d.ts +91 -0
  10. package/dist/processors/processor.test.d.ts +1 -0
  11. package/dist/processors/runner.d.ts +19 -0
  12. package/dist/request-context/index.d.ts +2 -0
  13. package/dist/request-context/request-context.d.ts +159 -0
  14. package/dist/request-context/request-context.test.d.ts +1 -0
  15. package/dist/runtimes/claude.d.ts +27 -50
  16. package/dist/runtimes/openai-desktop.js +1918 -741
  17. package/dist/runtimes/vercel.d.ts +34 -0
  18. package/dist/runtimes/vercel.js +474 -0
  19. package/dist/sandbox.d.ts +29 -25
  20. package/dist/step-invocation/__tests__/invoker.test.d.ts +1 -0
  21. package/dist/step-invocation/__tests__/protocol.test.d.ts +1 -0
  22. package/dist/step-invocation/__tests__/server.test.d.ts +1 -0
  23. package/dist/step-invocation/index.d.ts +25 -0
  24. package/dist/step-invocation/invoker.d.ts +65 -0
  25. package/dist/step-invocation/protocol.d.ts +44 -0
  26. package/dist/step-invocation/server.d.ts +63 -0
  27. package/dist/step-invocation/types.d.ts +72 -0
  28. package/dist/tools/coding.d.ts +49 -0
  29. package/dist/tools/coding.test.d.ts +1 -0
  30. package/dist/tools/index.d.ts +2 -0
  31. package/dist/types/events.d.ts +36 -0
  32. package/dist/types/execution-context.d.ts +22 -0
  33. package/dist/types/runtime.d.ts +32 -0
  34. package/dist/types/sandbox-environment.d.ts +5 -2
  35. package/dist/types/sandbox.d.ts +14 -12
  36. package/dist/types/workflow-metadata.d.ts +51 -0
  37. package/dist/types/workflow-plan.d.ts +19 -0
  38. package/dist/types/workflow.d.ts +57 -17
  39. package/dist/utils/bundler.d.ts +62 -3
  40. package/dist/workflow-steps/__tests__/observability.test.d.ts +1 -0
  41. package/dist/workflow-steps/index.d.ts +10 -0
  42. package/dist/workflow-steps/observability.d.ts +58 -0
  43. package/dist/workflow-steps/runner.d.ts +96 -0
  44. package/dist/workflow-steps/step.d.ts +25 -0
  45. package/dist/workflow-steps/types.d.ts +135 -0
  46. package/dist/workflow-steps/workflow-steps.test.d.ts +1 -0
  47. package/dist/workflow-steps/workflow.d.ts +50 -0
  48. package/dist/workflows/engine.d.ts +27 -13
  49. package/dist/workflows/invoke-child.d.ts +10 -0
  50. package/package.json +25 -15
  51. package/src/agent/agent-loop.ts +197 -26
  52. package/src/agent/run-agent.ts +40 -15
  53. package/src/client.ts +326 -76
  54. package/src/index.ts +124 -10
  55. package/src/processors/builtins.ts +72 -0
  56. package/src/processors/index.ts +15 -0
  57. package/src/processors/processor.ts +103 -0
  58. package/src/processors/runner.ts +42 -0
  59. package/src/request-context/index.ts +17 -0
  60. package/src/request-context/request-context.ts +302 -0
  61. package/src/runtimes/claude.ts +123 -254
  62. package/src/runtimes/vercel.ts +180 -0
  63. package/src/sandbox.ts +53 -21
  64. package/src/step-invocation/index.ts +33 -0
  65. package/src/step-invocation/invoker.ts +204 -0
  66. package/src/step-invocation/protocol.ts +57 -0
  67. package/src/step-invocation/server.ts +184 -0
  68. package/src/step-invocation/types.ts +70 -0
  69. package/src/tools/coding.ts +126 -0
  70. package/src/tools/index.ts +8 -0
  71. package/src/types/events.ts +40 -0
  72. package/src/types/execution-context.ts +30 -0
  73. package/src/types/runtime.ts +24 -0
  74. package/src/types/sandbox-environment.ts +7 -5
  75. package/src/types/sandbox.ts +16 -12
  76. package/src/types/workflow-metadata.ts +84 -0
  77. package/src/types/workflow-plan.ts +24 -0
  78. package/src/types/workflow.ts +139 -25
  79. package/src/utils/bundler.ts +198 -18
  80. package/src/utils/source-loader.ts +2 -2
  81. package/src/workflow-steps/index.ts +30 -0
  82. package/src/workflow-steps/observability.ts +103 -0
  83. package/src/workflow-steps/runner.ts +244 -0
  84. package/src/workflow-steps/step.ts +38 -0
  85. package/src/workflow-steps/types.ts +134 -0
  86. package/src/workflow-steps/workflow.ts +95 -0
  87. package/src/workflows/engine.ts +69 -40
  88. package/src/workflows/invoke-child.ts +29 -0
package/src/sandbox.ts CHANGED
@@ -11,9 +11,11 @@ import { spawn } from "node:child_process";
11
11
  import { Sandbox } from "e2b";
12
12
  import { Sandbox as Desktop } from "@e2b/desktop";
13
13
  import pRetry from "p-retry";
14
- import type { SandboxProvider, DesktopSandboxProvider } from "./types/sandbox.js";
14
+ import type { SandboxProvider, DesktopSandboxProvider, SandboxCommandResult } from "./types/sandbox.js";
15
15
 
16
- export type { SandboxProvider, DesktopSandboxProvider } from "./types/sandbox.js";
16
+ export type { SandboxProvider, DesktopSandboxProvider, SandboxCommandRunOptions, SandboxCommandResult } from "./types/sandbox.js";
17
+
18
+ export type SandboxProviderName = "vercel" | "e2b" | "e2b-desktop";
17
19
 
18
20
  // Fleet-wide tag, NOT machine-scoped. Previously we embedded FLY_MACHINE_ID
19
21
  // so each replica scoped its own E2B sandboxes at boot, but that made
@@ -37,12 +39,25 @@ const VERCEL_VM_LIFETIME_WINDOW_MS = 6 * 60 * 60 * 1000;
37
39
  * firewall injects the specified headers before forwarding — credentials never
38
40
  * exist inside the VM. E2B ignores this field (future self-hosted mapping TBD).
39
41
  */
42
+ export interface SandboxNetworkHeaderTransform {
43
+ headers?: Record<string, string>;
44
+ }
45
+
46
+ export interface SandboxNetworkAllowRule {
47
+ transform?: SandboxNetworkHeaderTransform[];
48
+ }
49
+
50
+ export interface SandboxNetworkSubnetPolicy {
51
+ allow?: string[];
52
+ deny?: string[];
53
+ }
54
+
40
55
  export type SandboxNetworkPolicy =
41
56
  | "allow-all"
42
57
  | "deny-all"
43
58
  | {
44
- allow?: string[] | Record<string, Array<{ transform?: Array<{ headers?: Record<string, string> }> }>>;
45
- subnets?: { allow?: string[]; deny?: string[] };
59
+ allow?: string[] | Record<string, SandboxNetworkAllowRule[]>;
60
+ subnets?: SandboxNetworkSubnetPolicy;
46
61
  };
47
62
 
48
63
  export interface SandboxCreateOpts {
@@ -103,7 +118,14 @@ interface SandboxProviderDef {
103
118
  // ── E2B helpers ───────────────────────────────────────────────────────────────
104
119
 
105
120
  export function makeSandboxProvider(sb: Sandbox | Desktop): SandboxProvider {
106
- return { sandboxId: sb.sandboxId, commands: sb.commands, files: sb.files, kill: () => sb.kill() };
121
+ return {
122
+ sandboxId: sb.sandboxId,
123
+ commands: sb.commands,
124
+ files: {
125
+ async write(path, content) { await sb.files.write(path, content); },
126
+ },
127
+ kill: () => sb.kill(),
128
+ };
107
129
  }
108
130
 
109
131
  export function makeDesktopSandboxProvider(sb: Desktop): DesktopSandboxProvider {
@@ -126,14 +148,19 @@ export function makeDesktopSandboxProvider(sb: Desktop): DesktopSandboxProvider
126
148
 
127
149
  type SseEvent = { type: string; data?: string; exitCode?: number };
128
150
 
151
+ export interface ParseSseExecStreamOptions {
152
+ onStdout?: (data: string) => void;
153
+ onStderr?: (data: string) => void;
154
+ }
155
+
129
156
  /**
130
157
  * Parse an SSE exec stream from a ReadableStream.
131
158
  * Used by the agent sandbox broker in runner.ts.
132
159
  */
133
160
  export async function parseSseExecStream(
134
161
  body: ReadableStream<Uint8Array>,
135
- opts?: { onStdout?: (d: string) => void; onStderr?: (d: string) => void },
136
- ): Promise<{ exitCode: number; stdout: string }> {
162
+ opts?: ParseSseExecStreamOptions,
163
+ ): Promise<SandboxCommandResult> {
137
164
  let stdout = "", exitCode = 0, exited = false;
138
165
  const reader = body.getReader(), decoder = new TextDecoder();
139
166
  let buf = "";
@@ -175,7 +202,8 @@ function makeVercelSandboxProvider(sb: any, globalEnvs?: Record<string, string>)
175
202
  async run(cmd, opts) {
176
203
  if (opts?.background) {
177
204
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
178
- void (sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts.cwd, env: mergeEnvs(opts.envs), detached: true }) as Promise<any>).catch(() => {});
205
+ void (sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts.cwd, env: mergeEnvs(opts.envs), detached: true }) as Promise<any>)
206
+ .catch((err: unknown) => console.error(`[sandbox] background command failed: ${err instanceof Error ? err.message : String(err)}`));
179
207
  return { exitCode: 0, stdout: "" };
180
208
  }
181
209
  const signal = opts?.timeoutMs ? AbortSignal.timeout(opts.timeoutMs) : undefined;
@@ -334,7 +362,7 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
334
362
  const sandboxes = [];
335
363
  while (paginator.hasNext) sandboxes.push(...await paginator.nextItems());
336
364
  if (sandboxes.length === 0) return;
337
- await Promise.all(sandboxes.map(s => Sandbox.kill(s.sandboxId).catch(() => {})));
365
+ await Promise.all(sandboxes.map(s => Sandbox.kill(s.sandboxId)));
338
366
  console.info(`[sandbox] killed ${sandboxes.length} stale E2B sandboxes (tag: ${AGENT_COMPOSE_TAG})`);
339
367
  },
340
368
  getActiveCount: async () => (await SANDBOX_PROVIDERS.e2b.listOwned!({})).length,
@@ -364,7 +392,7 @@ const SANDBOX_PROVIDERS: Record<string, SandboxProviderDef> = {
364
392
  };
365
393
 
366
394
  /** Provision a sandbox for the named provider. */
367
- export async function createSandbox(provider: string, opts: SandboxCreateOpts): Promise<SandboxProvider> {
395
+ export async function createSandbox(provider: SandboxProviderName, opts: SandboxCreateOpts): Promise<SandboxProvider> {
368
396
  const def = SANDBOX_PROVIDERS[provider];
369
397
  if (!def) throw new Error(`Unknown sandbox provider: "${provider}". Known: ${Object.keys(SANDBOX_PROVIDERS).join(", ")}`);
370
398
  const missing = Object.entries(def.requiredEnv)
@@ -375,14 +403,14 @@ export async function createSandbox(provider: string, opts: SandboxCreateOpts):
375
403
  }
376
404
 
377
405
  /** Reconnect to an existing sandbox by provider + provider-native sandbox ID. */
378
- export async function reconnectSandbox(provider: string, sandboxId: string): Promise<SandboxProvider> {
406
+ export async function reconnectSandbox(provider: SandboxProviderName, sandboxId: string): Promise<SandboxProvider> {
379
407
  const def = SANDBOX_PROVIDERS[provider];
380
408
  if (!def?.reconnect) throw new Error(`Provider "${provider}" does not support reconnect`);
381
409
  return def.reconnect(sandboxId);
382
410
  }
383
411
 
384
412
  /** Delete a snapshot by id on the named provider. No live sandbox needed. */
385
- export async function deleteSandboxSnapshot(provider: string, snapshotId: string): Promise<void> {
413
+ export async function deleteSandboxSnapshot(provider: SandboxProviderName, snapshotId: string): Promise<void> {
386
414
  const def = SANDBOX_PROVIDERS[provider];
387
415
  if (!def?.deleteSnapshot) throw new Error(`Provider "${provider}" does not support deleteSnapshot`);
388
416
  const missing = Object.entries(def.requiredEnv)
@@ -398,10 +426,12 @@ export async function deleteSandboxSnapshot(provider: string, snapshotId: string
398
426
  * `getActiveCount`. Errors are surfaced per-provider so one flaky provider
399
427
  * doesn't silence the rest.
400
428
  */
401
- export async function getSandboxQuotas(): Promise<Record<string, number | Error>> {
402
- const out: Record<string, number | Error> = {};
429
+ export type SandboxQuotaResult = Partial<Record<SandboxProviderName, number | Error>>;
430
+
431
+ export async function getSandboxQuotas(): Promise<SandboxQuotaResult> {
432
+ const out: SandboxQuotaResult = {};
403
433
  await Promise.all(
404
- Object.entries(SANDBOX_PROVIDERS).map(async ([name, def]) => {
434
+ (Object.entries(SANDBOX_PROVIDERS) as Array<[SandboxProviderName, SandboxProviderDef]>).map(async ([name, def]) => {
405
435
  if (!def.getActiveCount) return;
406
436
  if (Object.keys(def.requiredEnv).some(k => !process.env[k])) return;
407
437
  const env = Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k]!]));
@@ -418,10 +448,12 @@ export async function getSandboxQuotas(): Promise<Record<string, number | Error>
418
448
  * or missing env are skipped. Errors propagate per-provider so one flaky
419
449
  * provider doesn't silence the rest.
420
450
  */
421
- export async function listOwnedSandboxes(): Promise<Record<string, OwnedSandbox[] | Error>> {
422
- const out: Record<string, OwnedSandbox[] | Error> = {};
451
+ export type OwnedSandboxResult = Partial<Record<SandboxProviderName, OwnedSandbox[] | Error>>;
452
+
453
+ export async function listOwnedSandboxes(): Promise<OwnedSandboxResult> {
454
+ const out: OwnedSandboxResult = {};
423
455
  await Promise.all(
424
- Object.entries(SANDBOX_PROVIDERS).map(async ([name, def]) => {
456
+ (Object.entries(SANDBOX_PROVIDERS) as Array<[SandboxProviderName, SandboxProviderDef]>).map(async ([name, def]) => {
425
457
  if (!def.listOwned) return;
426
458
  if (Object.keys(def.requiredEnv).some(k => !process.env[k])) return;
427
459
  const env = Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k]!]));
@@ -433,7 +465,7 @@ export async function listOwnedSandboxes(): Promise<Record<string, OwnedSandbox[
433
465
  }
434
466
 
435
467
  /** Kill a sandbox by provider + native ID. Used by the orphan reconciler. */
436
- export async function killSandboxById(provider: string, sandboxId: string): Promise<void> {
468
+ export async function killSandboxById(provider: SandboxProviderName, sandboxId: string): Promise<void> {
437
469
  const def = SANDBOX_PROVIDERS[provider];
438
470
  if (!def?.reconnect) throw new Error(`Provider "${provider}" does not support reconnect (required for kill-by-id)`);
439
471
  const sb = await def.reconnect(sandboxId);
@@ -442,10 +474,10 @@ export async function killSandboxById(provider: string, sandboxId: string): Prom
442
474
 
443
475
  /** Kill all sandboxes across all registered providers. Call at startup to clean up after crashes. */
444
476
  export async function killAllSandboxes(
445
- onError?: (provider: string, err: unknown) => void,
477
+ onError?: (provider: SandboxProviderName, err: unknown) => void,
446
478
  ): Promise<void> {
447
479
  await Promise.allSettled(
448
- Object.entries(SANDBOX_PROVIDERS)
480
+ (Object.entries(SANDBOX_PROVIDERS) as Array<[SandboxProviderName, SandboxProviderDef]>)
449
481
  .filter(([, def]) => def.killAll)
450
482
  .map(([name, def]) => {
451
483
  const env = Object.fromEntries(Object.keys(def.requiredEnv).map(k => [k, process.env[k] ?? ""]));
@@ -0,0 +1,33 @@
1
+ /**
2
+ * StepInvocation — cross-process protocol that lets a server-side
3
+ * activity (Temporal `executeStep`, future Inngest equivalent) drive
4
+ * one workflow step inside an existing runner sandbox.
5
+ *
6
+ * Two halves, single source of truth for the wire:
7
+ *
8
+ * - `invokeStep(sandbox, request)` — server side. Writes input +
9
+ * request context, mints a token, builds envs, spawns the runner,
10
+ * parses the tokenised stdout sentinel, returns a kinded `StepResult`.
11
+ *
12
+ * - `serveStep(handler)` — runner side. Reads input + request context
13
+ * from env-pointed files, runs the handler, emits the tokenised
14
+ * sentinel result line, exits with the right code.
15
+ *
16
+ * The protocol is intentionally narrow: one input file + one context
17
+ * file + one tokenised stdout result line. Easy to reason about, easy
18
+ * to test in isolation.
19
+ */
20
+
21
+ export {
22
+ STEP_RESULT_PREFIX,
23
+ STEP_ENV,
24
+ RUNNER_BUNDLE_PATH,
25
+ RUNNER_COMMAND,
26
+ stepInputPath,
27
+ requestContextPath,
28
+ } from "./protocol.js";
29
+ export { invokeStep, parseStepResult, buildStepEnvs } from "./invoker.js";
30
+ export { serveStep } from "./server.js";
31
+ export type { StepHandler, ServeStepRequest, StepHandlerResult } from "./server.js";
32
+ export { StepExecutionError } from "./types.js";
33
+ export type { StepRequest, StepResult, StepInvocationError } from "./types.js";
@@ -0,0 +1,204 @@
1
+ /**
2
+ * Server-side half of StepInvocation. The Temporal `executeStep` activity
3
+ * (and any future durable engine's equivalent) calls `invokeStep(...)` to
4
+ * run one step inside an existing runner sandbox.
5
+ *
6
+ * Owns:
7
+ * - per-invocation result token (protocol hygiene — keeps user-side
8
+ * stdout from accidentally contaminating the result channel)
9
+ * - input + request-context file writes
10
+ * - env stamping
11
+ * - sandbox.commands.run with the runner command
12
+ * - tokenised stdout sentinel parse
13
+ * - failure-mode classification (protocol / user-step / runner-exit)
14
+ *
15
+ * Caller owns:
16
+ * - reconnecting to the runner sandbox (per-team, per-run)
17
+ * - turning a kinded error into the activity's exception shape
18
+ *
19
+ * No timeout: Temporal's startToCloseTimeout already bounds the activity.
20
+ * `sandbox.commands.run({ timeoutMs: 0 })` lets the runner consume that
21
+ * full window before the activity times out and Temporal kills the call.
22
+ */
23
+
24
+ import { randomBytes } from "node:crypto";
25
+ import type { SandboxProvider } from "../types/sandbox.js";
26
+ import { RUNNER_COMMAND, STEP_ENV, stepResultLinePrefix, requestContextPath, stepInputPath } from "./protocol.js";
27
+ import type { StepRequest, StepResult } from "./types.js";
28
+
29
+ export { RUNNER_COMMAND } from "./protocol.js";
30
+
31
+ /** Build the env map for one step invocation. Pure function; tests use it
32
+ * to assert the env shape (and to assert that no server-held credentials
33
+ * leak in) without spawning a sandbox. */
34
+ export function buildStepEnvs(args: {
35
+ runId: string;
36
+ stepIndex: number;
37
+ resultToken: string;
38
+ }): Record<string, string> {
39
+ return {
40
+ [STEP_ENV.RUN_ID]: args.runId,
41
+ [STEP_ENV.STEP_MODE]: "1",
42
+ [STEP_ENV.STEP_INDEX]: String(args.stepIndex),
43
+ [STEP_ENV.STEP_INPUT_PATH]: stepInputPath(args.stepIndex),
44
+ [STEP_ENV.REQUEST_CONTEXT_PATH]: requestContextPath(args.stepIndex),
45
+ [STEP_ENV.STEP_RESULT_TOKEN]: args.resultToken,
46
+ };
47
+ }
48
+
49
+ /** Scan the runner's stdout for the tokenised sentinel line and return a
50
+ * classified result. Returns `null` when no sentinel is present — the
51
+ * caller distinguishes that case from a parsed `{ ok: false }` payload
52
+ * using exit code (see `invokeStep`). Pure function for testability.
53
+ *
54
+ * Failure classification is taken from the runner's `kind` field on the
55
+ * payload (the runner separates setup/protocol failures from handler
56
+ * throws). If `kind` is missing or unknown, treat as `protocol` —
57
+ * that's a malformed payload, which by definition is a wire violation. */
58
+ export function parseStepResult<TOutput = unknown>(
59
+ stdout: string,
60
+ resultToken: string,
61
+ ): StepResult<TOutput> | null {
62
+ const prefix = stepResultLinePrefix(resultToken);
63
+ const line = stdout.split(/\r?\n/).find((l) => l.startsWith(prefix));
64
+ if (!line) return null;
65
+ let parsed: {
66
+ ok: boolean;
67
+ output?: unknown;
68
+ kind?: string;
69
+ error?: string;
70
+ observability?: import("../workflow-steps/observability.js").StepObservability;
71
+ };
72
+ try {
73
+ parsed = JSON.parse(line.slice(prefix.length));
74
+ } catch {
75
+ return { ok: false, error: { kind: "protocol", message: "step result JSON parse failed" } };
76
+ }
77
+ if (parsed.ok) {
78
+ return parsed.observability
79
+ ? { ok: true, output: parsed.output as TOutput, observability: parsed.observability }
80
+ : { ok: true, output: parsed.output as TOutput };
81
+ }
82
+ const kind: "protocol" | "user-step" = parsed.kind === "user-step" ? "user-step" : "protocol";
83
+ const baseMessage = parsed.error ?? "step body failed without a message";
84
+ // An unknown kind value means the runner emitted something this invoker
85
+ // doesn't recognise — most likely a version-skew deploy. Preserve the
86
+ // original kind string in the message so operators see the actual value
87
+ // instead of silently re-classifying it as "protocol".
88
+ const message = parsed.kind && parsed.kind !== "user-step" && parsed.kind !== "protocol"
89
+ ? `(unknown kind "${parsed.kind}") ${baseMessage}`
90
+ : baseMessage;
91
+ return { ok: false, error: { kind, message } };
92
+ }
93
+
94
+ /**
95
+ * Run one step inside the given sandbox. The full ceremony is here so the
96
+ * caller body is "reconnect sandbox → invokeStep → narrow result". A
97
+ * stdout line that happens to start with the sentinel prefix but carries
98
+ * a different token is ignored — the per-invocation token guarantees the
99
+ * result channel can't be contaminated by user `console.log` output or
100
+ * by any dep that logs a JSON-shaped line.
101
+ */
102
+ export interface InvokeStepOptions {
103
+ envs?: Record<string, string>;
104
+ /** Live stdout / stderr from the runner subprocess, line-by-line. Called
105
+ * from inside `sandbox.commands.run` as chunks arrive. The sentinel line
106
+ * carrying the protocol result token is filtered out before delivery so
107
+ * callers only see user-visible runtime output.
108
+ *
109
+ * Callbacks are invoked synchronously from the sandbox stream loop — keep
110
+ * them cheap (e.g. push to an in-memory array). Persisting / shipping is
111
+ * the caller's job; the activity batches and inserts at step completion. */
112
+ onStdout?: (line: string) => void;
113
+ onStderr?: (line: string) => void;
114
+ }
115
+
116
+ export async function invokeStep<TOutput = unknown>(
117
+ sandbox: SandboxProvider,
118
+ request: StepRequest,
119
+ opts?: InvokeStepOptions,
120
+ ): Promise<StepResult<TOutput>> {
121
+ const resultToken = randomBytes(16).toString("hex");
122
+
123
+ await Promise.all([
124
+ sandbox.files.write(stepInputPath(request.stepIndex), JSON.stringify(request.input)),
125
+ sandbox.files.write(requestContextPath(request.stepIndex), JSON.stringify(request.requestContext)),
126
+ ]);
127
+
128
+ const envs = {
129
+ ...(opts?.envs ?? {}),
130
+ ...buildStepEnvs({
131
+ runId: request.runId,
132
+ stepIndex: request.stepIndex,
133
+ resultToken,
134
+ }),
135
+ };
136
+
137
+ // The sandbox emits stdout as raw chunks, not lines. Buffer between
138
+ // emissions so a `console.log` split across two chunks (or a partial
139
+ // trailing line) is delivered to onStdout/onStderr as one logical line.
140
+ // The sentinel line (carrying `resultToken`) is filtered out so callers
141
+ // never see protocol bytes in user-log capture. The prefix is built
142
+ // from `stepResultLinePrefix` — the same helper `parseStepResult` uses,
143
+ // so the filter and the parser can't drift.
144
+ const sentinelPrefix = stepResultLinePrefix(resultToken);
145
+ const makeLineSplitter = (sink: ((line: string) => void) | undefined, filterSentinel: boolean) => {
146
+ if (!sink) return { onChunk: undefined, flush: () => {} };
147
+ let buf = "";
148
+ return {
149
+ onChunk: (chunk: string) => {
150
+ buf += chunk;
151
+ let nl;
152
+ while ((nl = buf.indexOf("\n")) !== -1) {
153
+ const line = buf.slice(0, nl);
154
+ buf = buf.slice(nl + 1);
155
+ if (filterSentinel && line.startsWith(sentinelPrefix)) continue;
156
+ sink(line);
157
+ }
158
+ },
159
+ // Drain any remaining buffered output that ended without a newline.
160
+ // Called after `sandbox.commands.run` resolves so a runner that
161
+ // exits with `process.stdout.write("final")` (no trailing \n)
162
+ // doesn't silently drop its last line.
163
+ flush: () => {
164
+ if (buf.length === 0) return;
165
+ const line = buf;
166
+ buf = "";
167
+ if (filterSentinel && line.startsWith(sentinelPrefix)) return;
168
+ sink(line);
169
+ },
170
+ };
171
+ };
172
+ const stdoutSplitter = makeLineSplitter(opts?.onStdout, true);
173
+ const stderrSplitter = makeLineSplitter(opts?.onStderr, false);
174
+
175
+ const result = await sandbox.commands.run(RUNNER_COMMAND, {
176
+ envs,
177
+ timeoutMs: 0,
178
+ ...(stdoutSplitter.onChunk ? { onStdout: stdoutSplitter.onChunk } : {}),
179
+ ...(stderrSplitter.onChunk ? { onStderr: stderrSplitter.onChunk } : {}),
180
+ });
181
+ stdoutSplitter.flush();
182
+ stderrSplitter.flush();
183
+
184
+ const parsed = parseStepResult<TOutput>(result.stdout, resultToken);
185
+ if (parsed) return parsed;
186
+
187
+ // No tokenised sentinel on stdout. Distinguish "runner exited badly
188
+ // before emitting" (runner-exit) from "runner exited cleanly but didn't
189
+ // speak the protocol" (protocol) — the exit code is the evidence.
190
+ if (result.exitCode !== 0) {
191
+ return {
192
+ ok: false,
193
+ error: {
194
+ kind: "runner-exit",
195
+ message: `runner subprocess exited ${result.exitCode} before emitting a step result`,
196
+ exitCode: result.exitCode,
197
+ },
198
+ };
199
+ }
200
+ return {
201
+ ok: false,
202
+ error: { kind: "protocol", message: "no tokenised step result found on stdout" },
203
+ };
204
+ }
@@ -0,0 +1,57 @@
1
+ /**
2
+ * StepInvocation — wire protocol between server-side activity (the
3
+ * invoker) and runner-side step server. Both halves consume these
4
+ * constants so the format is single-sourced; changing it is one diff.
5
+ *
6
+ * The protocol carries one workflow step's input/output across an OS
7
+ * process boundary (server's Temporal worker process → runner sandbox
8
+ * subprocess). It deliberately does not depend on Temporal — a future
9
+ * Inngest provider can reuse the same wire.
10
+ */
11
+
12
+ /** Sentinel that prefixes the step result on the runner's stdout. The
13
+ * invoker scans for `<prefix><token>:<json>` so any user `console.log`
14
+ * with the same prefix but a wrong token is rejected. */
15
+ export const STEP_RESULT_PREFIX = "__AC_STEP_RESULT__";
16
+
17
+ /** Build the line prefix the runner emits and the invoker scans for. The
18
+ * runner-side serveStep writes `<prefix><token>:<json>\n`; the invoker's
19
+ * result parser AND the stdout line splitter both match on this exact
20
+ * prefix so a future change can't make them drift (review found a
21
+ * separator mismatch that leaked the sentinel into captured logs). */
22
+ export function stepResultLinePrefix(token: string): string {
23
+ return `${STEP_RESULT_PREFIX}${token}:`;
24
+ }
25
+
26
+ /** Sandbox-side path where dispatch writes the compiled runner bundle.
27
+ * Both modes (full-mode `dispatch.ts` and step-mode `invokeStep`) spawn
28
+ * the runner from this path; single source of truth. */
29
+ export const RUNNER_BUNDLE_PATH = "/tmp/runner.bundle.js";
30
+
31
+ /** Command the invoker (and dispatch) spawns inside the sandbox. */
32
+ export const RUNNER_COMMAND = `node ${RUNNER_BUNDLE_PATH}`;
33
+
34
+ /** Names of the env vars the invoker stamps onto the runner subprocess.
35
+ * The runner's serveStep reads from these. Centralising the names lets
36
+ * tests assert the env shape without duplicating string literals. */
37
+ export const STEP_ENV = {
38
+ RUN_ID: "RUN_ID",
39
+ STEP_MODE: "AC_STEP_MODE",
40
+ STEP_INDEX: "AC_STEP_INDEX",
41
+ STEP_INPUT_PATH: "AC_STEP_INPUT_PATH",
42
+ REQUEST_CONTEXT_PATH: "AC_REQUEST_CONTEXT_PATH",
43
+ STEP_RESULT_TOKEN: "AC_STEP_RESULT_TOKEN",
44
+ } as const;
45
+
46
+ /** Sandbox-side path where the invoker writes the JSON-encoded step input.
47
+ * The serveStep reads from this exact path; both halves use this helper. */
48
+ export function stepInputPath(stepIndex: number): string {
49
+ return `/tmp/wf/step-input-${stepIndex}.json`;
50
+ }
51
+
52
+ /** Sandbox-side path where the invoker writes the JSON-encoded request
53
+ * context. The serveStep reads from this exact path; both halves use
54
+ * this helper. */
55
+ export function requestContextPath(stepIndex: number): string {
56
+ return `/tmp/wf/request-context-${stepIndex}.json`;
57
+ }
@@ -0,0 +1,184 @@
1
+ /**
2
+ * Runner-side half of StepInvocation. The worker's runner.ts in step
3
+ * mode calls `serveStep(handler)` — read the input + request context
4
+ * from env-pointed files, run the user handler, emit the tokenised
5
+ * sentinel result line on stdout, exit with the right code.
6
+ *
7
+ * Owns:
8
+ * - reading inputs from the env-pointed JSON files
9
+ * - parse-error handling (protocol-violating env or malformed file)
10
+ * - tokenised sentinel emission (matched by the invoker's parser)
11
+ * - awaited stdout flush before process.exit (Linux pipes can drop
12
+ * unflushed bytes when the writer exits)
13
+ * - exit codes (0 on success, 1 on any failure)
14
+ *
15
+ * Handler owns:
16
+ * - what to do with the input (load the compiled workflow, run the
17
+ * step, return the output)
18
+ *
19
+ * `serveStep` never returns — it always exits the process. The return
20
+ * type is `Promise<never>` so callers can `await serveStep(...)` and
21
+ * trust nothing runs after.
22
+ */
23
+
24
+ import { readFileSync } from "node:fs";
25
+ import type { RequestContextWire } from "../request-context/request-context.js";
26
+ import type { StepObservability } from "../workflow-steps/observability.js";
27
+ import { STEP_ENV, STEP_RESULT_PREFIX } from "./protocol.js";
28
+
29
+ /** What the handler receives. The serveStep already parsed the input
30
+ * and request context from their JSON files. */
31
+ export interface ServeStepRequest<TInput = unknown> {
32
+ runId: string;
33
+ stepIndex: number;
34
+ input: TInput;
35
+ requestContext: RequestContextWire;
36
+ }
37
+
38
+ /** What the handler returns. `observability` is the buffered snapshot
39
+ * of `ctx.setMetadata` / `ctx.step` / `ctx.agentEvents` recorded during
40
+ * the step; serveStep forwards it on the wire and the activity persists
41
+ * it. Undefined for handlers that don't use the hooks. */
42
+ export interface StepHandlerResult<TOutput = unknown> {
43
+ output: TOutput;
44
+ observability?: StepObservability;
45
+ }
46
+
47
+ /** User handler signature: take parsed step request, return the step
48
+ * output (and optional observability bundle). Throwing turns into
49
+ * `{ ok: false, kind: "user-step", error: <message> }` on the wire. */
50
+ export type StepHandler<TInput = unknown, TOutput = unknown> =
51
+ (request: ServeStepRequest<TInput>) => Promise<StepHandlerResult<TOutput>>;
52
+
53
+ /** Wire payload shape. Failure carries the `kind` so the invoker can
54
+ * classify "setup/protocol problem on the runner side" (e.g. malformed
55
+ * request-context JSON) separately from "the user handler threw".
56
+ * `observability` rides along on success so the activity can persist
57
+ * it without a second round-trip. */
58
+ type SentinelPayload =
59
+ | { ok: true; output: unknown; observability?: StepObservability }
60
+ | { ok: false; kind: "user-step" | "protocol"; error: string };
61
+
62
+ /** Emit one tokenised sentinel line and resolve only after the kernel
63
+ * has accepted the bytes. Awaited flush is critical: process.exit()
64
+ * immediately after a non-awaited write can drop the data on Linux
65
+ * pipes, leaving the invoker with empty stdout and the runner-exit
66
+ * classification pointing at the wrong thing. */
67
+ function emitResult(token: string, payload: SentinelPayload): Promise<void> {
68
+ return new Promise<void>((resolve, reject) => {
69
+ const line = `${STEP_RESULT_PREFIX}${token}:${JSON.stringify(payload)}\n`;
70
+ process.stdout.write(line, (err) => {
71
+ if (err) reject(err);
72
+ else resolve();
73
+ });
74
+ });
75
+ }
76
+
77
+ /** Delete step-invocation transport envs so the user's workflow code
78
+ * (loaded inside the handler) can't read them. Matters most for
79
+ * AC_STEP_RESULT_TOKEN: if it stayed in env, a `console.log` from a dep
80
+ * that happens to include the token could emit a line indistinguishable
81
+ * from a real result and win `.find()` in the invoker. Scrubbing keeps
82
+ * the result channel uncorruptible by stdout traffic from anywhere in
83
+ * the workflow's dep tree. RUN_ID is intentionally preserved because it
84
+ * is part of the runner's public workflow environment in both modes. */
85
+ function scrubProtocolEnvs(): void {
86
+ for (const key of [
87
+ STEP_ENV.STEP_MODE,
88
+ STEP_ENV.STEP_INDEX,
89
+ STEP_ENV.STEP_INPUT_PATH,
90
+ STEP_ENV.REQUEST_CONTEXT_PATH,
91
+ STEP_ENV.STEP_RESULT_TOKEN,
92
+ ]) {
93
+ delete process.env[key];
94
+ }
95
+ }
96
+
97
+ /**
98
+ * Read step input + request context, invoke `handler`, emit the result,
99
+ * exit. `Promise<never>` because every code path ends in `process.exit`.
100
+ *
101
+ * Two distinct phases under separate try/catch:
102
+ * 1. Setup — read envs, parse the input + context files. Failures
103
+ * here mean the wire is broken (server bug); they emit
104
+ * `{ok:false, kind:"protocol"}`.
105
+ * 2. Handler — call the user-supplied step body. Failures here are
106
+ * user code throwing; they emit `{ok:false, kind:"user-step"}`.
107
+ *
108
+ * Between the two phases, step-invocation transport envs are scrubbed
109
+ * from process.env so user code (including any workflow module imported
110
+ * inside the handler) cannot read AC_STEP_RESULT_TOKEN — that keeps the
111
+ * result channel uncorruptible by stdout output from anywhere in the
112
+ * workflow's dep tree. RUN_ID remains available to match full-mode
113
+ * runner behaviour.
114
+ */
115
+ export async function serveStep<TInput, TOutput>(
116
+ handler: StepHandler<TInput, TOutput>,
117
+ ): Promise<never> {
118
+ const token = process.env[STEP_ENV.STEP_RESULT_TOKEN];
119
+ if (!token) {
120
+ // Without a token we cannot emit a result the invoker will accept.
121
+ // Fail loudly on stderr so the runner-exit path on the activity
122
+ // side has a useful breadcrumb.
123
+ process.stderr.write(`[step-server] missing ${STEP_ENV.STEP_RESULT_TOKEN}\n`);
124
+ process.exit(1);
125
+ }
126
+
127
+ // process.exit is intentionally outside the try/catch — a real failure
128
+ // (or a test-time mock that throws an exit sentinel) must not be caught
129
+ // and re-emitted as a step failure.
130
+ let exitCode = 0;
131
+ let payload!: SentinelPayload;
132
+
133
+ let setupResult: { runId: string; stepIndex: number; input: TInput; requestContext: RequestContextWire } | undefined;
134
+ try {
135
+ const runId = process.env[STEP_ENV.RUN_ID];
136
+ const stepIndexEnv = process.env[STEP_ENV.STEP_INDEX];
137
+ const inputPath = process.env[STEP_ENV.STEP_INPUT_PATH];
138
+ const ctxPath = process.env[STEP_ENV.REQUEST_CONTEXT_PATH];
139
+ const stepIndex = stepIndexEnv === undefined ? Number.NaN : Number(stepIndexEnv);
140
+
141
+ if (!runId || !inputPath || !ctxPath || !Number.isFinite(stepIndex)) {
142
+ throw new Error(
143
+ `step-server missing required env: ${[
144
+ !runId && STEP_ENV.RUN_ID,
145
+ !Number.isFinite(stepIndex) && STEP_ENV.STEP_INDEX,
146
+ !inputPath && STEP_ENV.STEP_INPUT_PATH,
147
+ !ctxPath && STEP_ENV.REQUEST_CONTEXT_PATH,
148
+ ].filter(Boolean).join(", ")}`,
149
+ );
150
+ }
151
+
152
+ const input = JSON.parse(readFileSync(inputPath, "utf8")) as TInput;
153
+ // RequestContextWire parsing is loose here on purpose — the wire is
154
+ // server-controlled, malformed shape is a server bug, and we want a
155
+ // useful failure reason in `failRun`. Strict validation belongs at
156
+ // the workflow-level RequestContext.parse boundary, not here.
157
+ const requestContext = JSON.parse(readFileSync(ctxPath, "utf8")) as RequestContextWire;
158
+
159
+ setupResult = { runId, stepIndex, input, requestContext };
160
+ } catch (err) {
161
+ exitCode = 1;
162
+ payload = { ok: false, kind: "protocol", error: err instanceof Error ? err.message : String(err) };
163
+ }
164
+
165
+ if (setupResult) {
166
+ // Scrub protocol envs BEFORE user code runs. After this point the
167
+ // transport-only process.env[STEP_ENV.*] reads return undefined, so a
168
+ // workflow dependency cannot read AC_STEP_RESULT_TOKEN and accidentally
169
+ // emit a stdout line that the invoker would accept as a real result.
170
+ scrubProtocolEnvs();
171
+ try {
172
+ const { output, observability } = await handler(setupResult);
173
+ payload = observability
174
+ ? { ok: true, output, observability }
175
+ : { ok: true, output };
176
+ } catch (err) {
177
+ exitCode = 1;
178
+ payload = { ok: false, kind: "user-step", error: err instanceof Error ? err.message : String(err) };
179
+ }
180
+ }
181
+
182
+ await emitResult(token, payload);
183
+ process.exit(exitCode);
184
+ }