@agent-compose/sdk 0.5.1 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/dist/active-step.d.ts +60 -0
  2. package/dist/agent/agent-loop-steer.test.d.ts +1 -0
  3. package/dist/agent/agent-loop.d.ts +46 -0
  4. package/dist/agent/async-queue.d.ts +29 -0
  5. package/dist/agent/protocol.d.ts +9 -1
  6. package/dist/agent/resolve-agent-id.test.d.ts +1 -0
  7. package/dist/agent/run-agent.d.ts +16 -3
  8. package/dist/agent/steer-control.d.ts +57 -0
  9. package/dist/agent/steer-control.test.d.ts +1 -0
  10. package/dist/client.d.ts +161 -0
  11. package/dist/index.d.ts +7 -4
  12. package/dist/index.js +1339 -157
  13. package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
  14. package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
  15. package/dist/pause/__tests__/errors.test.d.ts +1 -0
  16. package/dist/pause/__tests__/manager.test.d.ts +1 -0
  17. package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
  18. package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
  19. package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
  20. package/dist/pause/checkpoint.d.ts +28 -0
  21. package/dist/pause/errors.d.ts +52 -0
  22. package/dist/pause/manager.d.ts +63 -0
  23. package/dist/pause/pause-core.d.ts +101 -0
  24. package/dist/pause/state-dir.d.ts +80 -0
  25. package/dist/pause/wrappers.d.ts +41 -0
  26. package/dist/request-context/request-context.d.ts +12 -0
  27. package/dist/runtimes/claude.d.ts +6 -0
  28. package/dist/runtimes/openai-desktop.d.ts +2 -0
  29. package/dist/runtimes/openai-desktop.js +1336 -156
  30. package/dist/runtimes/vercel.d.ts +12 -0
  31. package/dist/runtimes/vercel.js +50 -7
  32. package/dist/runtimes/vercel.test.d.ts +1 -0
  33. package/dist/sse.d.ts +2 -3
  34. package/dist/step-invocation/index.d.ts +2 -2
  35. package/dist/step-invocation/invoker.d.ts +3 -0
  36. package/dist/step-invocation/protocol.d.ts +12 -0
  37. package/dist/step-invocation/server.d.ts +1 -0
  38. package/dist/step-invocation/types.d.ts +40 -5
  39. package/dist/types/events.d.ts +9 -0
  40. package/dist/types/execution-context.d.ts +25 -0
  41. package/dist/types/protocol.d.ts +8 -0
  42. package/dist/types/runtime.d.ts +55 -0
  43. package/dist/types/sandbox.d.ts +4 -4
  44. package/dist/utils/schemas.d.ts +2 -0
  45. package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
  46. package/dist/workflow-steps/index.d.ts +2 -0
  47. package/dist/workflow-steps/observability.d.ts +43 -11
  48. package/dist/workflow-steps/run-callback.d.ts +39 -0
  49. package/dist/workflow-steps/runner.d.ts +8 -0
  50. package/package.json +1 -1
  51. package/src/active-step.ts +124 -0
  52. package/src/agent/agent-loop.ts +253 -19
  53. package/src/agent/async-queue.ts +61 -0
  54. package/src/agent/protocol.ts +12 -2
  55. package/src/agent/run-agent.ts +184 -8
  56. package/src/agent/steer-control.ts +125 -0
  57. package/src/client.ts +277 -0
  58. package/src/index.ts +18 -2
  59. package/src/pause/checkpoint.ts +44 -0
  60. package/src/pause/errors.ts +70 -0
  61. package/src/pause/manager.ts +177 -0
  62. package/src/pause/pause-core.ts +267 -0
  63. package/src/pause/state-dir.ts +262 -0
  64. package/src/pause/wrappers.ts +79 -0
  65. package/src/request-context/request-context.ts +17 -2
  66. package/src/runtimes/claude.ts +101 -6
  67. package/src/runtimes/openai-desktop.ts +11 -0
  68. package/src/runtimes/vercel.ts +26 -0
  69. package/src/sandbox.ts +39 -20
  70. package/src/sse.ts +8 -6
  71. package/src/step-invocation/index.ts +2 -1
  72. package/src/step-invocation/invoker.ts +107 -29
  73. package/src/step-invocation/protocol.ts +16 -0
  74. package/src/step-invocation/server.ts +45 -12
  75. package/src/step-invocation/types.ts +43 -7
  76. package/src/tools/coding.ts +16 -5
  77. package/src/types/events.ts +9 -0
  78. package/src/types/execution-context.ts +25 -0
  79. package/src/types/protocol.ts +8 -0
  80. package/src/types/runtime.ts +52 -0
  81. package/src/types/sandbox.ts +8 -4
  82. package/src/types/workflow.ts +6 -1
  83. package/src/utils/bundler.ts +8 -3
  84. package/src/utils/schemas.ts +2 -0
  85. package/src/workflow-steps/index.ts +3 -0
  86. package/src/workflow-steps/observability.ts +84 -13
  87. package/src/workflow-steps/run-callback.ts +72 -0
  88. package/src/workflow-steps/runner.ts +70 -8
  89. package/dist/utils/discovery.d.ts +0 -2
  90. package/src/utils/discovery.ts +0 -4
@@ -15,6 +15,15 @@ export type RunEvent =
15
15
  seq?: number;
16
16
  agentId: string;
17
17
  label: string;
18
+ /** Tool whitelist passed to `agent({ tools: [...] })`. Drives the
19
+ * per-agent "tools" badge on the dashboard. */
20
+ allowedTools?: string[];
21
+ /** Resolved model id (e.g. `claude-sonnet-4-6`). */
22
+ model?: string;
23
+ /** Short runtime self-identifier (`claude`, `openai-desktop`, …)
24
+ * read off `ModelExecutionContract.kind`. The dashboard maps
25
+ * this to a small runtime icon on the agent card header. */
26
+ runtimeKind?: string;
18
27
  }
19
28
  | {
20
29
  event: "agent.message";
@@ -2,6 +2,8 @@
2
2
 
3
3
  import type { InvokeAndWaitOptions, RunStatus } from "../client.js";
4
4
  import type { RequestContext } from "../request-context/request-context.js";
5
+ import type { PauseRequest } from "../pause/pause-core.js";
6
+ import type { RequestDecisionRequest, WaitForEventRequest } from "../pause/wrappers.js";
5
7
  import type { SandboxProvider } from "./sandbox.js";
6
8
 
7
9
  /** The identity of this workflow run. */
@@ -27,4 +29,27 @@ export interface BaseExecutionContext {
27
29
  setMetadata?: (data: Record<string, unknown>) => Promise<void>;
28
30
  /** Invoke another registered workflow and wait for it to settle. */
29
31
  invokeChild: InvokeChild;
32
+ /**
33
+ * Disk-backed memoise across pause-resume. First call runs `fn` and
34
+ * atomically writes the result to the sandbox; on resume the recorded
35
+ * value is returned and `fn` is NOT re-executed. Use for expensive
36
+ * deterministic transforms; for side effects, use `invokeChild`.
37
+ * See ADR-0006 §"`ctx.checkpoint(name, fn)` — disk-backed memoisation".
38
+ */
39
+ checkpoint<T>(name: string, fn: () => Promise<T> | T): Promise<T>;
40
+ /**
41
+ * Pause for feedback. The step exits and the workflow waits durably until
42
+ * something resolves the pause (a resume call, a TTL expiry); on resume the
43
+ * step body re-runs from the top and this call returns the resume payload.
44
+ * Provide a `schema` to validate the payload, `ttlMs` + `onExpiry` to bound
45
+ * the wait, `correlationKey` for by-key resume. Throws PauseRequestError /
46
+ * PauseExpiredError / PauseSchemaError. See ADR-0006 / ADR-0011.
47
+ */
48
+ pause<T = unknown>(req: PauseRequest<T>): Promise<T>;
49
+ /** Pause for a typed decision (a `schema` is required). Wrapper over `pause`. */
50
+ requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
51
+ /** Lightweight timed pause — resolves after `durationMs`, no snapshot. */
52
+ sleep(durationMs: number): Promise<void>;
53
+ /** Pause until an event resumes by `correlationKey`. Wrapper over `pause`. */
54
+ waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
30
55
  }
@@ -71,4 +71,12 @@ export interface AgentStatus {
71
71
  completed: string[];
72
72
  blockers: string[];
73
73
  exit_signal: boolean;
74
+ /** PR 7 self-pause (honoured only for `mode: "hitl"` agents). The agent
75
+ * cannot proceed without a human decision: it sets this true, puts the
76
+ * question in `question`, and ends its turn. The loop pauses at the next
77
+ * boundary and injects the human's answer as the next turn. `auto` agents
78
+ * ignore it and keep going. Should be paired with `exit_signal: false`. */
79
+ needs_input?: boolean;
80
+ /** The question to put to the human when `needs_input` is true. */
81
+ question?: string;
74
82
  }
@@ -48,14 +48,66 @@ export type ToolCallGateResult =
48
48
  export interface ModelExecutionContract {
49
49
  /** True when this runtime can run `processToolCall` before tool execution. */
50
50
  supportsToolCallProcessor?: boolean;
51
+ /** Short self-identifier ("claude", "openai-desktop", "vercel", …). Read
52
+ * by the agent loop and surfaced on `agent.spawned` so the dashboard
53
+ * can show a per-agent runtime icon without re-fetching template
54
+ * metadata. Optional — runtimes that omit it stay anonymous. */
55
+ kind?: string;
56
+ /** Resolved model id used by this runtime instance (already merged with
57
+ * config + runtime defaults). Surfaced on `agent.spawned` so each
58
+ * agent card on the Agent tab can label which model it ran against. */
59
+ model?: string;
51
60
  /** Runtime-owned pre-tool gate. Adapters call the shared processor chain
52
61
  * through this seam; the agent loop stays SDK-agnostic. */
53
62
  gateToolCall?(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
63
+ /**
64
+ * Capture runtime-private in-memory state that will not survive the
65
+ * runner subprocess exit. Called by the agent loop at pause time,
66
+ * AFTER the loop has flushed its own state to disk.
67
+ *
68
+ * Return value is opaque to the loop — whatever the runtime needs to
69
+ * round-trip its conversation across pause-resume. Must be JSON-
70
+ * serialisable; the loop atomically writes it to
71
+ * `/tmp/wf/state/runtime-<agentInstanceId>.json` and reads it back
72
+ * on resume to hand to `restoreCheckpoint`.
73
+ *
74
+ * Default (method omitted): runtime holds no instance state that
75
+ * needs to round-trip across pause. The shipped example is the
76
+ * Claude runtime — the conversation lives server-side at
77
+ * Anthropic, addressed by `session_id`, and the loop already holds
78
+ * `lastSessionId` as part of its own state. On resume the loop
79
+ * restores the id, the next `sendMessage` passes it through, and
80
+ * Anthropic resumes the server-side conversation.
81
+ *
82
+ * Runtimes that hold the conversation in-process — Vercel's
83
+ * `VercelRunner.messages` is the canonical case — MUST implement
84
+ * both hooks: the messages array is reconstructed from
85
+ * `response.messages` on each `streamText` and would be lost the
86
+ * moment the subprocess exits. See ADR-0006 §"Concrete examples
87
+ * for shipped runtimes" for the audit + worked examples.
88
+ */
89
+ captureCheckpoint?(): unknown;
90
+ /**
91
+ * Restore runtime-private state previously returned by
92
+ * `captureCheckpoint`. Called by the agent loop on resume, AFTER
93
+ * the loop has restored its own state but BEFORE iterations resume.
94
+ *
95
+ * `blob` is whatever this same runtime returned at pause time. If
96
+ * `captureCheckpoint` is omitted, this is never called.
97
+ */
98
+ restoreCheckpoint?(blob: unknown): void;
54
99
  sendMessage(opts: {
55
100
  prompt: string;
56
101
  sessionId?: string;
57
102
  iteration?: number;
58
103
  signal?: AbortSignal;
104
+ /** Push-iterable of mid-turn user messages from outside the agent
105
+ * loop — e.g. dashboard chat injections. Runtimes that support
106
+ * streaming-input mode (Claude Agent SDK) read from this in
107
+ * parallel with the initial `prompt`; the SDK handles delivery
108
+ * at the next safe boundary. Runtimes without streaming-input
109
+ * support ignore this and fall back to per-iteration injection. */
110
+ inboxStream?: AsyncIterable<{ text: string; senderName?: string | null }>;
59
111
  }): AsyncGenerator<AgentMessage>;
60
112
  }
61
113
 
@@ -10,17 +10,17 @@ export interface SandboxCommandRunOptions {
10
10
  envs?: Record<string, string>;
11
11
  onStdout?: (data: string) => void;
12
12
  onStderr?: (data: string) => void;
13
- background?: boolean;
14
13
  /** Run the command with root privileges. Vercel maps this to its native
15
- * `sudo` flag; E2B runs the command as `user: "root"`; the local provider
16
- * prepends `sudo`. Defaults to false. Requires the sandbox image to grant
17
- * the command root (Vercel's runtimes do — passwordless). */
14
+ * `sudo` flag; the local provider prepends `sudo`. Defaults to false.
15
+ * Requires the sandbox image to grant the command root (Vercel's runtimes
16
+ * do — passwordless). */
18
17
  sudo?: boolean;
19
18
  }
20
19
 
21
20
  export interface SandboxCommandResult {
22
21
  exitCode: number;
23
22
  stdout: string;
23
+ stderr: string;
24
24
  }
25
25
 
26
26
  export interface SandboxProvider {
@@ -28,6 +28,10 @@ export interface SandboxProvider {
28
28
  /** Working directory for the agent process. Set by onStart after environment setup. */
29
29
  cwd?: string;
30
30
  commands: {
31
+ // NOTE: does NOT throw on non-zero exit. Callers must check
32
+ // `result.exitCode` themselves. The `agent-env` setup workflow's
33
+ // `run(sb, cmd)` helper is the canonical pattern — copy it into any
34
+ // setup workflow that needs to fail loudly on command errors.
31
35
  run(cmd: string, opts?: SandboxCommandRunOptions): Promise<SandboxCommandResult>;
32
36
  };
33
37
  files: {
@@ -210,7 +210,12 @@ function compileRunForm<TOutput, TInput extends Record<string, unknown>>(
210
210
  setMetadata: stepCtx.setMetadata,
211
211
  step: stepCtx.step,
212
212
  agentEvents: stepCtx.agentEvents,
213
- processors: metadata.processors ?? [],
213
+ checkpoint: stepCtx.checkpoint,
214
+ pause: stepCtx.pause,
215
+ requestDecision: stepCtx.requestDecision,
216
+ sleep: stepCtx.sleep,
217
+ waitForEvent: stepCtx.waitForEvent,
218
+ processors: metadata.processors ?? [],
214
219
  };
215
220
  const sandbox = stepCtx.sandbox;
216
221
  if (!sandbox) throw new Error("legacy run-form workflow requires a sandbox in StepContext");
@@ -111,8 +111,9 @@ async function bundle(path: string, label: string): Promise<string> {
111
111
  }
112
112
 
113
113
  /** Dynamically import a bundled source in-process and extract fields from its
114
- * default export. Safe for CLI/test contexts. Returns `undefined` on
115
- * resolver failure (missing deps) so callers fall through. */
114
+ * default export. Safe for CLI/test contexts. Evaluation failures are
115
+ * surfaced with their original cause so users fix the real import/runtime
116
+ * problem instead of being told the default export shape is wrong. */
116
117
  async function extractFromBundle<T>(
117
118
  source: string,
118
119
  label: string,
@@ -123,7 +124,11 @@ async function extractFromBundle<T>(
123
124
  const loaded = await importSourceModule<Record<string, unknown>>(source, tmpPath);
124
125
  try { return pick(loaded.mod.default); }
125
126
  finally { await loaded.cleanup(); }
126
- } catch { return undefined; }
127
+ } catch (err) {
128
+ throw new WorkflowSourceValidationError(
129
+ `Bundled ${label} could not be evaluated: ${err instanceof Error ? err.message : String(err)}`,
130
+ );
131
+ }
127
132
  }
128
133
 
129
134
  /**
@@ -7,4 +7,6 @@ export const AgentStatusSchema = z.object({
7
7
  completed: z.array(z.string()),
8
8
  blockers: z.array(z.string()),
9
9
  exit_signal: z.boolean(),
10
+ needs_input: z.boolean().optional(),
11
+ question: z.string().optional(),
10
12
  }) satisfies z.ZodType<AgentStatus>;
@@ -28,3 +28,6 @@ export { WORKFLOW_BRAND } from "./types.js";
28
28
 
29
29
  export { StepObservabilityCollector } from "./observability.js";
30
30
  export type { StepObservability, SubStepEvent } from "./observability.js";
31
+
32
+ export { makeRunCallbackEmitterFromEnv } from "./run-callback.js";
33
+ export type { LiveAgentEventEmitter } from "./run-callback.js";
@@ -5,19 +5,34 @@
5
5
  * tokenised stdout sentinel; the activity persists the snapshot to the
6
6
  * run's metadata + lifecycle event tables.
7
7
  *
8
- * STREAMING — known gap. `agentEvents` are currently batched at step
9
- * end. For long-running steps that emit many agent events, this hides
10
- * progress from the dashboard until the step completes. The follow-up
11
- * is to bring back a minimal `/internal/runs/:runId/steps/:stepIndex/events`
12
- * route gated by a per-step JIT-signed token (mint in the activity,
13
- * stamp into the runner env, verify on receive) so `agentEvents.emit`
14
- * POSTs in real time. Until then `metadata` and `subSteps` are
15
- * effectively boundary events (no streaming need) and batching them
16
- * matches their natural granularity.
8
+ * Live streaming + batch backstop — the no-double-write design:
9
+ *
10
+ * 1. `agentEvents.emit` immediately fires the optional `liveEmitter`
11
+ * (the runner's POST to `/internal/runs/.../events`).
12
+ * 2. The emitter returns `Promise<boolean>` — true means the server
13
+ * accepted and persisted this event, false means it failed (POST
14
+ * error, server 5xx, network blip).
15
+ * 3. The collector tracks which seqs were successfully ack'd.
16
+ * 4. At `snapshot()` time we await any in-flight emits (with a small
17
+ * grace window so the agent loop's final-burst posts can finish),
18
+ * then STRIP ack'd events from the returned `events` array.
19
+ *
20
+ * The result: the snapshot's `events` array only contains events
21
+ * that the live path didn't successfully deliver. The server's
22
+ * batch-flush in `persistStepObservability` becomes a true backstop
23
+ * for the FAILURE path — it never re-writes (and never re-notifies)
24
+ * the events the live route already handled. No double pg_notify,
25
+ * no dashboard duplicates.
26
+ *
27
+ * `metadata` and `subSteps` remain batch-only because they're
28
+ * naturally boundary events (no streaming benefit) and aren't
29
+ * written by the live route at all.
17
30
  */
18
31
 
19
32
  import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
20
33
  import type { AgentEventSink } from "../types/workflow.js";
34
+ import type { LiveAgentEventEmitter } from "./run-callback.js";
35
+ import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
21
36
 
22
37
  /** One named sub-step (from `ctx.step("name", async () => ...)`).
23
38
  * Becomes a `workflow_substep_*` lifecycle event on the run timeline. */
@@ -39,6 +54,13 @@ export interface StepObservability {
39
54
  subSteps?: SubStepEvent[];
40
55
  }
41
56
 
57
+ /** Maximum time we'll wait for in-flight live-emit POSTs to settle
58
+ * before taking the snapshot. Tuned to be longer than a healthy POST
59
+ * (~50ms) but short enough that a totally-broken live path doesn't
60
+ * block step return — the batch backstop will handle whatever doesn't
61
+ * resolve in time. */
62
+ const LIVE_EMIT_DRAIN_TIMEOUT_MS = 1_500;
63
+
42
64
  /**
43
65
  * Append-only collector bound to a single step's `StepContext`. The
44
66
  * step's `setMetadata` / `step` / `agentEvents` properties all point at
@@ -53,6 +75,15 @@ export class StepObservabilityCollector {
53
75
  private metadata: Record<string, unknown> = {};
54
76
  private events: AgentLifecycleEvent[] = [];
55
77
  private subSteps: SubStepEvent[] = [];
78
+ private readonly liveEmitter?: LiveAgentEventEmitter;
79
+ /** Seqs of events the server confirmed via the live route. */
80
+ private readonly ackedSeqs: Set<number> = new Set();
81
+ /** Promises for in-flight live emits — awaited at snapshot time. */
82
+ private readonly inFlight: Set<Promise<void>> = new Set();
83
+
84
+ constructor(opts: { liveEmitter?: LiveAgentEventEmitter } = {}) {
85
+ this.liveEmitter = opts.liveEmitter;
86
+ }
56
87
 
57
88
  readonly setMetadata = async (data: Record<string, unknown>): Promise<void> => {
58
89
  Object.assign(this.metadata, data);
@@ -70,6 +101,9 @@ export class StepObservabilityCollector {
70
101
  });
71
102
  return result;
72
103
  } catch (err) {
104
+ // A pause inside ctx.step is control flow, not a failed sub-step — let
105
+ // it propagate untouched so serveStep emits the pause sentinel.
106
+ if (isPauseSignal(err)) throw err;
73
107
  this.subSteps.push({
74
108
  name,
75
109
  startedAt,
@@ -83,20 +117,57 @@ export class StepObservabilityCollector {
83
117
 
84
118
  readonly agentEvents: AgentEventSink = {
85
119
  emit: (event: AgentLifecycleEvent) => {
120
+ // `seq` is the event's index in `events`, captured BEFORE push so
121
+ // it matches the index the server-side batch flush uses (its
122
+ // `.entries()` loop). Same index → same idempotency key on the
123
+ // server. The live route's `acceptedSeqs` response uses this seq;
124
+ // we strip those from the snapshot in `snapshot()`.
125
+ const seq = this.events.length;
86
126
  this.events.push(event);
127
+ if (this.liveEmitter) {
128
+ const tracked = this.liveEmitter(event, seq)
129
+ .then((ok) => { if (ok) this.ackedSeqs.add(seq); })
130
+ .catch(() => { /* leave unacked → batch backstop delivers */ });
131
+ this.inFlight.add(tracked);
132
+ // Self-clean so completed promises don't leak across long-
133
+ // running steps.
134
+ void tracked.finally(() => this.inFlight.delete(tracked));
135
+ }
87
136
  },
88
137
  };
89
138
 
139
+ /** Wait for in-flight live emits to settle (or timeout) so the
140
+ * ackedSeqs set is maximally up-to-date before we filter. Used by
141
+ * `snapshot()` — exposed separately for tests. */
142
+ private async drainInFlight(timeoutMs = LIVE_EMIT_DRAIN_TIMEOUT_MS): Promise<void> {
143
+ if (this.inFlight.size === 0) return;
144
+ await Promise.race([
145
+ Promise.allSettled([...this.inFlight]),
146
+ new Promise<void>((resolve) => setTimeout(resolve, timeoutMs)),
147
+ ]);
148
+ }
149
+
90
150
  /** Snapshot the accumulated state. Returns `undefined` when nothing
91
- * was recorded so the wire payload can drop the field entirely. */
92
- snapshot(): StepObservability | undefined {
151
+ * was recorded so the wire payload can drop the field entirely.
152
+ *
153
+ * Async because we drain in-flight live emits first. Any event the
154
+ * server acknowledged is REMOVED from the returned `events` array
155
+ * so the server-side batch flush doesn't re-write/re-notify it.
156
+ * Events that failed live delivery (POST error, timeout) stay in
157
+ * the array as the durable backstop. */
158
+ async snapshot(): Promise<StepObservability | undefined> {
159
+ await this.drainInFlight();
160
+
93
161
  const hasMetadata = Object.keys(this.metadata).length > 0;
94
- const hasEvents = this.events.length > 0;
95
162
  const hasSubSteps = this.subSteps.length > 0;
163
+ // Filter out ack'd events — the live route already wrote them.
164
+ const remainingEvents = this.events.filter((_, idx) => !this.ackedSeqs.has(idx));
165
+ const hasEvents = remainingEvents.length > 0;
166
+
96
167
  if (!hasMetadata && !hasEvents && !hasSubSteps) return undefined;
97
168
  const result: StepObservability = {};
98
169
  if (hasMetadata) result.metadata = { ...this.metadata };
99
- if (hasEvents) result.events = [...this.events];
170
+ if (hasEvents) result.events = remainingEvents;
100
171
  if (hasSubSteps) result.subSteps = [...this.subSteps];
101
172
  return result;
102
173
  }
@@ -0,0 +1,72 @@
1
+ /**
2
+ * Runner → server live event emitter.
3
+ *
4
+ * Pairs with the server-side `/api/v1/internal/runs/:runId/steps/:stepIndex/events`
5
+ * route. The runner reads `AGENT_COMPOSE_URL`, `AGENT_COMPOSE_RUN_TOKEN`,
6
+ * and `RUN_ID` from env; if all three are present it builds a POST
7
+ * function that sends each agent lifecycle event to the server as
8
+ * soon as the agent loop emits it.
9
+ *
10
+ * Non-blocking: the agent loop doesn't await the emitter — the runtime
11
+ * keeps emitting events at full speed. But the emitter still returns a
12
+ * `Promise<boolean>` so the collector can later decide whether to
13
+ * include each event in the batch-end durable backstop. `true` = server
14
+ * accepted the event (don't re-deliver in batch); `false` = POST
15
+ * failed (DO re-deliver). The collector awaits these promises with a
16
+ * short grace window at snapshot time.
17
+ *
18
+ * This is the seam that lets us avoid the double-write problem: every
19
+ * successfully-ack'd live event is stripped from `observability.events`
20
+ * before the batch flush runs, so the durable backstop only carries
21
+ * events that genuinely failed live delivery. No more two paths writing
22
+ * the same row + double pg_notify.
23
+ */
24
+
25
+ import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
26
+
27
+ export type LiveAgentEventEmitter =
28
+ (event: AgentLifecycleEvent, seq: number) => Promise<boolean>;
29
+
30
+ /**
31
+ * Build a live emitter from process env + the caller-supplied stepIndex.
32
+ * Returns `undefined` when any required env var is missing (local tests,
33
+ * non-sandbox callers) so the caller can wire `undefined` straight
34
+ * through to the collector and get batch-only delivery without
35
+ * conditional plumbing.
36
+ *
37
+ * `stepIndex` is a parameter rather than an env read because
38
+ * `serveStep` scrubs `AC_STEP_INDEX` from `process.env` before invoking
39
+ * the workflow handler (to keep the transport envelope unreachable
40
+ * from user code); the handler still holds the parsed integer and
41
+ * passes it here.
42
+ */
43
+ export function makeRunCallbackEmitterFromEnv(stepIndex: number): LiveAgentEventEmitter | undefined {
44
+ const baseUrl = process.env.AGENT_COMPOSE_URL;
45
+ const token = process.env.AGENT_COMPOSE_RUN_TOKEN;
46
+ const runId = process.env.RUN_ID;
47
+ if (!baseUrl || !token || !runId) return undefined;
48
+ if (!Number.isFinite(stepIndex) || stepIndex < 0) return undefined;
49
+ const url = `${baseUrl.replace(/\/+$/, "")}/api/v1/internal/runs/${runId}/steps/${stepIndex}/events`;
50
+ const headers = {
51
+ "content-type": "application/json",
52
+ "authorization": `Bearer ${token}`,
53
+ } as const;
54
+ return async (event, seq) => {
55
+ try {
56
+ const res = await fetch(url, {
57
+ method: "POST",
58
+ headers,
59
+ body: JSON.stringify({ events: [{ ...event, seq }] }),
60
+ });
61
+ if (!res.ok) return false;
62
+ // Parse the route's response to confirm THIS seq was accepted.
63
+ // The route returns `{ acceptedSeqs: number[] }`; absence means
64
+ // the server logged but didn't persist (rare — DB hiccup), and
65
+ // we should leave the event in the batch backstop.
66
+ const body = await res.json().catch(() => null) as { acceptedSeqs?: number[] } | null;
67
+ return Array.isArray(body?.acceptedSeqs) && body.acceptedSeqs.includes(seq);
68
+ } catch {
69
+ return false;
70
+ }
71
+ };
72
+ }
@@ -28,6 +28,12 @@ import type { RequestContext } from "../request-context/request-context.js";
28
28
  import type { SandboxProvider } from "../types/sandbox.js";
29
29
  import type { WorkflowRun, WorkflowCtx } from "../types/workflow.js";
30
30
  import { StepObservabilityCollector, type StepObservability } from "./observability.js";
31
+ import type { LiveAgentEventEmitter } from "./run-callback.js";
32
+ import { scopedCheckpoint } from "../pause/checkpoint.js";
33
+ import { corePause, PauseSignal, isPauseSignal, type PauseRequest } from "../pause/pause-core.js";
34
+ import type { StepPauseRequest } from "../step-invocation/types.js";
35
+ import { buildPauseWrappers, type PauseFn, type KindedPauseFn } from "../pause/wrappers.js";
36
+ import { runWithActiveStep, setActiveStepBridge, restoreActiveStepBridge } from "../active-step.js";
31
37
 
32
38
  export class StepValidationError extends Error {
33
39
  readonly kind = "step-validation" as const;
@@ -84,6 +90,11 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
84
90
  onStepStarted?(stepIndex: number, stepName: string): void | Promise<void>;
85
91
  /** Child workflow invocation implementation. Defaults to a clear unsupported error. */
86
92
  invokeChild?: WorkflowCtx["invokeChild"];
93
+ /** Optional live-stream emitter for agent lifecycle events. The runner
94
+ * passes a fetch-based emitter wired to the per-run callback token so
95
+ * the dashboard sees events as the agent loop produces them; tests
96
+ * leave it undefined and get batch-only delivery. */
97
+ liveAgentEventEmitter?: LiveAgentEventEmitter;
87
98
  }
88
99
 
89
100
  export interface RunWorkflowStepsResult<TOutput> {
@@ -101,6 +112,8 @@ export interface RunWorkflowSingleStepOpts {
101
112
  sandbox?: SandboxProvider;
102
113
  abortSignal?: AbortSignal;
103
114
  invokeChild?: WorkflowCtx["invokeChild"];
115
+ /** Optional live-stream emitter — see `RunWorkflowStepsOpts.liveAgentEventEmitter`. */
116
+ liveAgentEventEmitter?: LiveAgentEventEmitter;
104
117
  }
105
118
 
106
119
  /** Result of one step run — output plus whatever the step's observability
@@ -116,7 +129,14 @@ export async function runWorkflowSingleStep(opts: RunWorkflowSingleStepOpts): Pr
116
129
  if (!step) throw new Error(`Step index ${opts.stepIndex} not found in workflow "${opts.workflow.id}"`);
117
130
  const parsedInput = step.input.safeParse(opts.input);
118
131
  if (!parsedInput.success) throw new StepValidationError(step.name, "input", parsedInput.error);
119
- const collector = new StepObservabilityCollector();
132
+ const collector = new StepObservabilityCollector({ liveEmitter: opts.liveAgentEventEmitter });
133
+ const coord = { runId: opts.run.id, stepIndex: opts.stepIndex };
134
+ const pauseWithKind: KindedPauseFn = function <T>(req: PauseRequest<T>, kind: StepPauseRequest["kind"]): Promise<T> {
135
+ return corePause(req, coord, kind);
136
+ };
137
+ const pause: PauseFn = function <T>(req: PauseRequest<T>): Promise<T> {
138
+ return corePause(req, coord);
139
+ };
120
140
  const stepCtx: StepContext<unknown> = {
121
141
  input: parsedInput.data,
122
142
  requestContext: opts.requestContext,
@@ -128,11 +148,35 @@ export async function runWorkflowSingleStep(opts: RunWorkflowSingleStepOpts): Pr
128
148
  setMetadata: collector.setMetadata,
129
149
  step: collector.step,
130
150
  agentEvents: collector.agentEvents,
151
+ checkpoint: scopedCheckpoint(`step${opts.stepIndex}`),
152
+ pause,
153
+ ...buildPauseWrappers(pauseWithKind),
131
154
  };
132
- const output = await step.run(stepCtx);
155
+ let output: unknown;
156
+ try {
157
+ // Set the active step so `agent()` / `ctx.pause` derive deterministic
158
+ // ids from the un-scrubbed index (see active-step.ts). Async-local scope
159
+ // keeps concurrent in-process runs from clobbering each other. ALSO publish
160
+ // on the cross-instance bridge: a bundled workflow inlines its own SDK copy
161
+ // (a separate AsyncLocalStorage), so its `agent()` can't see the ALS we set
162
+ // here — the bridge carries the step across that boundary. One step per
163
+ // subprocess in the sandbox ⇒ no concurrency on the global slot.
164
+ const bridged = setActiveStepBridge({ stepIndex: opts.stepIndex });
165
+ try {
166
+ output = await runWithActiveStep({ stepIndex: opts.stepIndex }, () => Promise.resolve(step.run(stepCtx)));
167
+ } finally {
168
+ restoreActiveStepBridge(bridged);
169
+ }
170
+ } catch (err) {
171
+ if (isPauseSignal(err)) {
172
+ const observability = await collector.snapshot();
173
+ throw new PauseSignal(err.pauseId, err.pauseRequest, observability);
174
+ }
175
+ throw err;
176
+ }
133
177
  const parsedOutput = step.output.safeParse(output);
134
178
  if (!parsedOutput.success) throw new StepValidationError(step.name, "output", parsedOutput.error);
135
- const observability = collector.snapshot();
179
+ const observability = await collector.snapshot();
136
180
  return observability === undefined
137
181
  ? { output: parsedOutput.data }
138
182
  : { output: parsedOutput.data, observability };
@@ -179,7 +223,14 @@ export async function runWorkflowSteps<TInput, TOutput>(
179
223
  throw err;
180
224
  }
181
225
 
182
- const collector = new StepObservabilityCollector();
226
+ const collector = new StepObservabilityCollector({ liveEmitter: opts.liveAgentEventEmitter });
227
+ const coord = { runId: run.id, stepIndex: i };
228
+ const pauseWithKind: KindedPauseFn = function <T>(req: PauseRequest<T>, kind: StepPauseRequest["kind"]): Promise<T> {
229
+ return corePause(req, coord, kind);
230
+ };
231
+ const pause: PauseFn = function <T>(req: PauseRequest<T>): Promise<T> {
232
+ return corePause(req, coord);
233
+ };
183
234
  const stepCtx: StepContext<unknown> = {
184
235
  input: parsedStepInput.data,
185
236
  requestContext,
@@ -191,16 +242,27 @@ export async function runWorkflowSteps<TInput, TOutput>(
191
242
  setMetadata: collector.setMetadata,
192
243
  step: collector.step,
193
244
  agentEvents: collector.agentEvents,
245
+ checkpoint: scopedCheckpoint(`step${i}`),
246
+ pause,
247
+ ...buildPauseWrappers(pauseWithKind),
194
248
  };
195
249
 
196
250
  const startedAt = Date.now();
197
251
  let output: unknown;
198
252
  try {
199
- output = await step.run(stepCtx);
253
+ // Active step for deterministic agent/pause id derivation (active-step.ts).
254
+ // Async-local scope keeps concurrent in-process runs isolated.
255
+ output = await runWithActiveStep({ stepIndex: i }, () => Promise.resolve(step.run(stepCtx)));
200
256
  } catch (err) {
257
+ // A pause is control flow, not a failure — let it propagate so serveStep
258
+ // emits the pause sentinel instead of recording a failed step.
259
+ if (isPauseSignal(err)) {
260
+ const observability = await collector.snapshot();
261
+ throw new PauseSignal(err.pauseId, err.pauseRequest, observability);
262
+ }
201
263
  const wrapped = err instanceof Error ? err : new Error(String(err));
202
264
  const durationMs = Date.now() - startedAt;
203
- const observability = collector.snapshot();
265
+ const observability = await collector.snapshot();
204
266
  stepResults.push({
205
267
  name: step.name, status: "failed", error: wrapped.message, durationMs,
206
268
  ...(observability ? { observability } : {}),
@@ -213,7 +275,7 @@ export async function runWorkflowSteps<TInput, TOutput>(
213
275
  if (!parsedOutput.success) {
214
276
  const err = new StepValidationError(step.name, "output", parsedOutput.error);
215
277
  const durationMs = Date.now() - startedAt;
216
- const observability = collector.snapshot();
278
+ const observability = await collector.snapshot();
217
279
  stepResults.push({
218
280
  name: step.name, status: "failed", error: err.message, durationMs,
219
281
  ...(observability ? { observability } : {}),
@@ -224,7 +286,7 @@ export async function runWorkflowSteps<TInput, TOutput>(
224
286
 
225
287
  const durationMs = Date.now() - startedAt;
226
288
  current = parsedOutput.data;
227
- const observability = collector.snapshot();
289
+ const observability = await collector.snapshot();
228
290
  stepResults.push({
229
291
  name: step.name, status: "completed", output: parsedOutput.data, durationMs,
230
292
  ...(observability ? { observability } : {}),
@@ -1,2 +0,0 @@
1
- /** Find the runtime name referenced via runtime: "name" in workflow source. */
2
- export declare function discoverRuntimeName(source: string): string | null;
@@ -1,4 +0,0 @@
1
- /** Find the runtime name referenced via runtime: "name" in workflow source. */
2
- export function discoverRuntimeName(source: string): string | null {
3
- return source.match(/runtime:\s*["']([^"']+)["']/)?.[1] ?? null;
4
- }