@agent-compose/sdk 0.5.0 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/dist/active-step.d.ts +60 -0
  2. package/dist/agent/agent-loop-steer.test.d.ts +1 -0
  3. package/dist/agent/agent-loop.d.ts +46 -0
  4. package/dist/agent/async-queue.d.ts +29 -0
  5. package/dist/agent/protocol.d.ts +9 -1
  6. package/dist/agent/resolve-agent-id.test.d.ts +1 -0
  7. package/dist/agent/run-agent.d.ts +16 -3
  8. package/dist/agent/steer-control.d.ts +57 -0
  9. package/dist/agent/steer-control.test.d.ts +1 -0
  10. package/dist/client.d.ts +161 -0
  11. package/dist/index.d.ts +7 -4
  12. package/dist/index.js +1341 -157
  13. package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
  14. package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
  15. package/dist/pause/__tests__/errors.test.d.ts +1 -0
  16. package/dist/pause/__tests__/manager.test.d.ts +1 -0
  17. package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
  18. package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
  19. package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
  20. package/dist/pause/checkpoint.d.ts +28 -0
  21. package/dist/pause/errors.d.ts +52 -0
  22. package/dist/pause/manager.d.ts +63 -0
  23. package/dist/pause/pause-core.d.ts +101 -0
  24. package/dist/pause/state-dir.d.ts +80 -0
  25. package/dist/pause/wrappers.d.ts +41 -0
  26. package/dist/request-context/request-context.d.ts +12 -0
  27. package/dist/runtimes/claude.d.ts +6 -0
  28. package/dist/runtimes/openai-desktop.d.ts +2 -0
  29. package/dist/runtimes/openai-desktop.js +1338 -156
  30. package/dist/runtimes/vercel.d.ts +12 -0
  31. package/dist/runtimes/vercel.js +50 -7
  32. package/dist/runtimes/vercel.test.d.ts +1 -0
  33. package/dist/sse.d.ts +2 -3
  34. package/dist/step-invocation/index.d.ts +2 -2
  35. package/dist/step-invocation/invoker.d.ts +3 -0
  36. package/dist/step-invocation/protocol.d.ts +12 -0
  37. package/dist/step-invocation/server.d.ts +1 -0
  38. package/dist/step-invocation/types.d.ts +40 -5
  39. package/dist/types/events.d.ts +9 -0
  40. package/dist/types/execution-context.d.ts +25 -0
  41. package/dist/types/protocol.d.ts +8 -0
  42. package/dist/types/runtime.d.ts +55 -0
  43. package/dist/types/sandbox.d.ts +6 -1
  44. package/dist/utils/schemas.d.ts +2 -0
  45. package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
  46. package/dist/workflow-steps/index.d.ts +2 -0
  47. package/dist/workflow-steps/observability.d.ts +43 -11
  48. package/dist/workflow-steps/run-callback.d.ts +39 -0
  49. package/dist/workflow-steps/runner.d.ts +8 -0
  50. package/package.json +1 -1
  51. package/src/active-step.ts +124 -0
  52. package/src/agent/agent-loop.ts +253 -19
  53. package/src/agent/async-queue.ts +61 -0
  54. package/src/agent/protocol.ts +12 -2
  55. package/src/agent/run-agent.ts +184 -8
  56. package/src/agent/steer-control.ts +125 -0
  57. package/src/client.ts +277 -0
  58. package/src/index.ts +18 -2
  59. package/src/pause/checkpoint.ts +44 -0
  60. package/src/pause/errors.ts +70 -0
  61. package/src/pause/manager.ts +177 -0
  62. package/src/pause/pause-core.ts +267 -0
  63. package/src/pause/state-dir.ts +262 -0
  64. package/src/pause/wrappers.ts +79 -0
  65. package/src/request-context/request-context.ts +17 -2
  66. package/src/runtimes/claude.ts +101 -6
  67. package/src/runtimes/openai-desktop.ts +11 -0
  68. package/src/runtimes/vercel.ts +26 -0
  69. package/src/sandbox.ts +45 -17
  70. package/src/sse.ts +8 -6
  71. package/src/step-invocation/index.ts +2 -1
  72. package/src/step-invocation/invoker.ts +107 -29
  73. package/src/step-invocation/protocol.ts +16 -0
  74. package/src/step-invocation/server.ts +45 -12
  75. package/src/step-invocation/types.ts +43 -7
  76. package/src/tools/coding.ts +16 -5
  77. package/src/types/events.ts +9 -0
  78. package/src/types/execution-context.ts +25 -0
  79. package/src/types/protocol.ts +8 -0
  80. package/src/types/runtime.ts +52 -0
  81. package/src/types/sandbox.ts +10 -1
  82. package/src/types/workflow.ts +6 -1
  83. package/src/utils/bundler.ts +8 -3
  84. package/src/utils/schemas.ts +2 -0
  85. package/src/workflow-steps/index.ts +3 -0
  86. package/src/workflow-steps/observability.ts +84 -13
  87. package/src/workflow-steps/run-callback.ts +72 -0
  88. package/src/workflow-steps/runner.ts +70 -8
  89. package/dist/utils/discovery.d.ts +0 -2
  90. package/src/utils/discovery.ts +0 -4
@@ -0,0 +1,124 @@
1
+ /**
2
+ * The step currently executing in this runner subprocess.
3
+ *
4
+ * Step-mode derivations that must be byte-identical across a pause-resume
5
+ * re-entry — the agent loop's `agentId` (which names `agent-<id>.json`) and
6
+ * the pause primitive's `pauseId` — need the running step's index. They
7
+ * CANNOT read it from `process.env.AC_STEP_INDEX`: `serveStep` scrubs the
8
+ * transport envs (`AC_STEP_MODE`, `AC_STEP_INDEX`, the result token, …) from
9
+ * `process.env` *before* the user step body runs, so by the time `agent()` /
10
+ * `ctx.pause` execute those reads return `undefined`
11
+ * (`step-invocation/server.ts` `scrubProtocolEnvs`). Reading scrubbed env was
12
+ * a latent resume bug: `agentId` fell through to `randomUUID()` on every
13
+ * subprocess, so resume never matched the prior `agent-<id>.json`.
14
+ *
15
+ * Instead the step runner sets the index here from the un-scrubbed `stepIndex`
16
+ * it already holds, and the derivations read it from here.
17
+ *
18
+ * Async-local rather than process-global: public in-process execution can run
19
+ * multiple workflows concurrently, and ids must stay scoped to the async step
20
+ * body that is deriving them. A fresh subprocess on resume starts with a fresh
21
+ * async context, the runner re-sets the same `stepIndex`, and the per-step
22
+ * call-order counters re-derive identical ids — which is exactly what lets
23
+ * resume find its on-disk state.
24
+ */
25
+
26
+ import { AsyncLocalStorage } from "node:async_hooks";
27
+
28
+ export interface ActiveStep {
29
+ stepIndex: number;
30
+ }
31
+
32
+ interface ActiveStepState extends ActiveStep {
33
+ nextAgentCallIndex: number;
34
+ pauseOrdinalByScope: Map<string, number>;
35
+ }
36
+
37
+ const activeStepStorage = new AsyncLocalStorage<ActiveStepState>();
38
+
39
+ // Test/dev fallback for direct `setActiveStep(...)` users. The runner uses
40
+ // `runWithActiveStep`, so production step execution is async-local.
41
+ let fallbackActive: ActiveStepState | null = null;
42
+
43
+ // Cross-SDK-instance bridge. `bundleWorkflow` builds the workflow with
44
+ // `Bun.build` and NO `external`, so the bundle inlines its OWN copy of this
45
+ // module — a distinct `AsyncLocalStorage` instance from the runner's. The
46
+ // runner sets the active step via `runWithActiveStep` on ITS instance; the
47
+ // bundle's `agent()` reads `currentState()` on the BUNDLE's instance and would
48
+ // otherwise see null (→ a random agentId and an unwired steer-pause). So
49
+ // `runWithActiveStep` ALSO publishes the state on a realm-shared `globalThis`
50
+ // slot the bundle's copy can read. Consulted ONLY when the local ALS is empty,
51
+ // so in-process execution keeps full async-local isolation between concurrent
52
+ // runs (the ALS takes precedence below).
53
+ const BRIDGE_KEY = Symbol.for("agent-compose.activeStep.bridge");
54
+ function bridgeSlot(): { state: ActiveStepState | null } {
55
+ const g = globalThis as unknown as Record<symbol, { state: ActiveStepState | null } | undefined>;
56
+ return (g[BRIDGE_KEY] ??= { state: null });
57
+ }
58
+
59
+ function makeState(step: ActiveStep): ActiveStepState {
60
+ return {
61
+ stepIndex: step.stepIndex,
62
+ nextAgentCallIndex: 0,
63
+ pauseOrdinalByScope: new Map(),
64
+ };
65
+ }
66
+
67
+ function currentState(): ActiveStepState | null {
68
+ return activeStepStorage.getStore() ?? bridgeSlot().state ?? fallbackActive;
69
+ }
70
+
71
+ /** Set (or clear, with `null`) the step currently running. Called by the
72
+ * step runner around `step.run(...)`. Prefer `runWithActiveStep` for real
73
+ * async execution; this setter exists for tests and small synchronous helpers. */
74
+ export function setActiveStep(step: ActiveStep | null): void {
75
+ fallbackActive = step ? makeState(step) : null;
76
+ }
77
+
78
+ /** Run `fn` with a step context scoped to this async call tree. */
79
+ export async function runWithActiveStep<T>(step: ActiveStep, fn: () => Promise<T>): Promise<T> {
80
+ return activeStepStorage.run(makeState(step), fn);
81
+ }
82
+
83
+ /** Publish (or clear, with `null`) the active step on the cross-instance bridge.
84
+ * Called ONLY by the sandbox step runner (`runWorkflowSingleStep`), which runs
85
+ * exactly one step per subprocess — so there is no concurrency to race the
86
+ * single global slot. The bundle's `agent()` reads this when its own ALS is
87
+ * empty. Returns the prior value so the caller can restore it. */
88
+ export function setActiveStepBridge(step: ActiveStep | null): { state: ActiveStepState | null } {
89
+ const slot = bridgeSlot();
90
+ const prev = { state: slot.state };
91
+ slot.state = step ? makeState(step) : null;
92
+ return prev;
93
+ }
94
+
95
+ /** Restore a bridge value captured by `setActiveStepBridge`. */
96
+ export function restoreActiveStepBridge(prev: { state: ActiveStepState | null }): void {
97
+ bridgeSlot().state = prev.state;
98
+ }
99
+
100
+
101
+ /** The step currently running, or `null` outside step execution (local
102
+ * tests, dev, between steps). */
103
+ export function getActiveStep(): ActiveStep | null {
104
+ const state = currentState();
105
+ return state ? { stepIndex: state.stepIndex } : null;
106
+ }
107
+
108
+ /** Return the next implicit `agent()` call coordinate for the active step. */
109
+ export function nextAgentCallInActiveStep(): { stepIndex: number; callIndex: number } | null {
110
+ const state = currentState();
111
+ if (!state) return null;
112
+ const callIndex = state.nextAgentCallIndex;
113
+ state.nextAgentCallIndex += 1;
114
+ return { stepIndex: state.stepIndex, callIndex };
115
+ }
116
+
117
+ /** Return the next pause ordinal in `scope` for the active step invocation. */
118
+ export function nextPauseOrdinalInActiveStep(scope: string): number | null {
119
+ const state = currentState();
120
+ if (!state) return null;
121
+ const next = state.pauseOrdinalByScope.get(scope) ?? 0;
122
+ state.pauseOrdinalByScope.set(scope, next + 1);
123
+ return next;
124
+ }
@@ -11,6 +11,9 @@ import { randomUUID } from "node:crypto";
11
11
  import type { Processor, ProcessorContext } from "../processors/processor.js";
12
12
  import { runProcessorChain } from "../processors/runner.js";
13
13
  import { RequestContext } from "../request-context/request-context.js";
14
+ import { PauseManager } from "../pause/manager.js";
15
+ import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
16
+ import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
14
17
 
15
18
  export const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
16
19
 
@@ -42,7 +45,7 @@ export type AgentMessageSummary =
42
45
  | { type: "init"; sessionId: string }
43
46
  | { type: "text"; text: string }
44
47
  | { type: "thinking"; text: string }
45
- | { type: "tool_use"; toolName: string; toolUseId: string; toolInputPreview: string }
48
+ | { type: "tool_use"; toolName: string; toolUseId: string; toolInput: Record<string, unknown>; toolInputPreview: string }
46
49
  | { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
47
50
  | { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number }
48
51
  | { type: "done"; sessionId: string }
@@ -65,7 +68,7 @@ export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
65
68
  case "init": return { type: "init", sessionId: msg.sessionId };
66
69
  case "text": return { type: "text", text: msg.text };
67
70
  case "thinking": return { type: "thinking", text: msg.text };
68
- case "tool_use": return { type: "tool_use", toolName: msg.toolName, toolUseId: msg.toolUseId, toolInputPreview: preview(msg.toolInput) };
71
+ case "tool_use": return { type: "tool_use", toolName: msg.toolName, toolUseId: msg.toolUseId, toolInput: msg.toolInput, toolInputPreview: preview(msg.toolInput) };
69
72
  case "tool_result": return { type: "tool_result", toolUseId: msg.toolUseId, output: truncate(msg.output), isError: msg.isError };
70
73
  case "usage": return {
71
74
  type: "usage", inputTokens: msg.inputTokens, outputTokens: msg.outputTokens,
@@ -78,7 +81,16 @@ export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
78
81
  }
79
82
 
80
83
  export type AgentLifecycleEvent =
81
- | { event: "agent.spawned"; at: number; agentId: string; label: string; allowedTools?: string[] }
84
+ | {
85
+ event: "agent.spawned"; at: number; agentId: string; label: string;
86
+ allowedTools?: string[];
87
+ /** Resolved model id used for this agent (e.g. `claude-sonnet-4-6`).
88
+ * Read off `client.model` after runtime construction. */
89
+ model?: string;
90
+ /** Short runtime self-identifier (`claude`, `openai-desktop`, …).
91
+ * Drives the per-agent runtime icon on the dashboard. */
92
+ runtimeKind?: string;
93
+ }
82
94
  | { event: "agent.message"; at: number; agentId: string; label: string; iteration: number; message: AgentMessageSummary }
83
95
  | { event: "agent.iteration"; at: number; agentId: string; label: string; iteration: number; status: AgentStatus | null }
84
96
  | { event: "agent.settled"; at: number; agentId: string; label: string; outcome: "success" | "failed"; iterations: number; durationMs: number; failureReason?: string };
@@ -102,6 +114,37 @@ export interface AgentLoopOpts<TResponse = unknown> {
102
114
  processors?: readonly Processor[];
103
115
  /** Per-run typed bag — propagated to every processor as `ctx.requestContext`. */
104
116
  requestContext?: RequestContext;
117
+ /**
118
+ * Mid-iteration user-message injection. When set, the agent loop
119
+ * builds a per-iteration AsyncQueue and hands it to the runtime as
120
+ * `inboxStream`. The runner-side inbox poller calls `push()` on
121
+ * `inbox` when a user message arrives; the runtime (streaming-input
122
+ * Claude Agent SDK) picks it up at the next safe boundary.
123
+ *
124
+ * The loop drains `inbox` into the per-iteration queue while the
125
+ * iteration is active and stops draining at iteration end — any
126
+ * messages that arrive between iterations are buffered on `inbox`
127
+ * and drain into the NEXT iteration's queue. So no message is lost,
128
+ * but mid-iteration injection requires the runtime to support
129
+ * streaming input (ignored otherwise — handled at the runtime).
130
+ */
131
+ inbox?: import("./async-queue.js").AsyncQueue<{ text: string; senderName?: string | null }>;
132
+ /** PR 7 steer-pause: the boundary calls this to pause the workflow for a
133
+ * human steer. agent() builds it (a corePause closed over runId/stepIndex);
134
+ * the loop supplies the agentScope so the pauseId is stable across resume.
135
+ * Absent ⇒ no steer pause (local tests, non-sandbox callers). */
136
+ pause?: <T = unknown>(
137
+ req: { reason: string; correlationKey?: string; schema?: z.ZodType<T> },
138
+ agentScope: { agentId: string; iteration: number },
139
+ ) => Promise<T>;
140
+ /** PR 7: take-once read of this agent's pending steer (set by the control
141
+ * poller). Returns the steer's payload (reason / correlationKey) or null.
142
+ * The boundary consumes it once per check, and only when no steer is
143
+ * already staged. */
144
+ consumeSteerPending?: () => SteerPayload | null;
145
+ /** PR 7: `auto` (autonomous, default) or `hitl` (can be steer-paused). The
146
+ * boundary is inert unless `hitl`. */
147
+ mode?: "auto" | "hitl";
105
148
  }
106
149
 
107
150
  export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TResponse>): Promise<AgentLoopResult<TResponse>> {
@@ -154,13 +197,162 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
154
197
  let completedIterations = 0;
155
198
  let lastResponseText = "";
156
199
  let lastResponseValidationError = "";
200
+ // Iteration the loop should resume at on entry. 0 = fresh run; >0 =
201
+ // restored from disk after a pause-resume. The `for` loop reads
202
+ // `startIteration` instead of starting at 0 so already-completed
203
+ // iterations don't re-run.
204
+ let startIteration = 0;
205
+ // Was this invocation resumed from on-disk state? Drives an early
206
+ // call to `runtime.restoreCheckpoint(blob)` and suppresses the
207
+ // `agent.spawned` event (the agent already spawned in the prior
208
+ // subprocess; emitting again would double-count on the dashboard).
209
+ let resumed = false;
210
+ // PR 7: a pending human steer-pause intent. Persisted in loop state so it
211
+ // survives a resume and the boundary re-issues the pause on re-entry.
212
+ let pendingSteerPause: { reason: string; correlationKey: string | null; at: number } | null = null;
213
+
214
+ const pauseManager = new PauseManager(agentId);
215
+ const restore = await pauseManager.restoreAgentLoop<TResponse>(client);
216
+ if (restore.kind === "settled") {
217
+ process.stdout.write(`${logLabel} restored settled agent result from checkpoint after ${restore.result.iterations} iterations\n`);
218
+ return { agentId, label, ...restore.result };
219
+ }
220
+ if (restore.kind === "running") {
221
+ const s = restore.state;
222
+ startIteration = s.iteration;
223
+ completedIterations = s.completedIterations;
224
+ lastSessionId = s.lastSessionId;
225
+ lastStatus = s.lastStatus;
226
+ lastResponseText = s.lastResponseText;
227
+ lastResponseValidationError = s.lastResponseValidationError;
228
+ iterationsWithoutStatus = s.iterationsWithoutStatus;
229
+ blockerStreak = s.blockerStreak;
230
+ pendingSteerPause = s.pendingSteerPause ?? null;
231
+ resumed = true;
232
+ process.stdout.write(`${logLabel} resumed from agent-state checkpoint at iteration ${startIteration}/${maxIterations}\n`);
233
+ }
234
+ if (restore.kind === "invalid") {
235
+ // Forward-incompat or corrupt state file. Fall back to fresh start
236
+ // rather than partial-restore — a misread checkpoint would silently
237
+ // desync the loop from the model's actual conversation history.
238
+ process.stderr.write(
239
+ `${logLabel} agent-state checkpoint at agentId=${agentId} is unrecognised — starting fresh. ` +
240
+ `Schema version mismatch or corrupt file: ${restore.error}\n`,
241
+ );
242
+ }
243
+
244
+ // `agent.spawned` represents one logical agent instance, not one
245
+ // subprocess invocation. Skip on resume so the dashboard's per-agent
246
+ // card count doesn't increment every time a pause-resume re-enters
247
+ // the loop.
248
+ if (!resumed) {
249
+ opts.onAgentLifecycleEvent?.({
250
+ event: "agent.spawned",
251
+ at: startedAt,
252
+ agentId,
253
+ label,
254
+ allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
255
+ ...(client.model != null ? { model: client.model } : {}),
256
+ ...(client.kind != null ? { runtimeKind: client.kind } : {}),
257
+ });
258
+ }
259
+
260
+ /** Atomically flush loop state + runtime checkpoint to disk. Called
261
+ * at every iteration boundary so a `ctx.pause` firing inside the
262
+ * next iteration's body can rely on the most recent committed state.
263
+ * Two writes (loop state, runtime blob) — both atomic individually;
264
+ * the pair isn't transactional. If a stateful runtime later resumes
265
+ * without its blob, restore fails loudly rather than continuing with
266
+ * an empty conversation. */
267
+ const flushState = async (currentIteration: number): Promise<void> => {
268
+ await pauseManager.saveRunningAgentLoop({
269
+ iteration: currentIteration,
270
+ completedIterations,
271
+ lastSessionId,
272
+ lastStatus,
273
+ lastResponseText,
274
+ lastResponseValidationError,
275
+ iterationsWithoutStatus,
276
+ blockerStreak,
277
+ pendingSteerPause,
278
+ }, client);
279
+ };
157
280
 
158
- opts.onAgentLifecycleEvent?.({ event: "agent.spawned", at: startedAt, agentId, label, allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS });
281
+ const writeSettledState = async (result: AgentLoopResult<TResponse>): Promise<void> => {
282
+ await pauseManager.saveSettledAgentLoop({
283
+ sessionId: result.sessionId,
284
+ lastStatus: result.lastStatus,
285
+ iterations: result.iterations,
286
+ ...(result.response !== undefined ? { response: result.response } : {}),
287
+ });
288
+ };
159
289
 
160
290
  try {
161
- for (let iteration = 0; iteration < maxIterations; iteration++) {
291
+ // The `|| pendingSteerPause` clause lets a staged steer (a self-pause asked on
292
+ // the FINAL budgeted iteration, or a restored intent) run one boundary past
293
+ // the budget so its pause fires and the human's answer gets an iteration —
294
+ // otherwise an end-of-iteration self-pause on the last turn would fall through
295
+ // to the exhaustion path and settle "success" with the question silently lost.
296
+ for (let iteration = startIteration; iteration < maxIterations || (!!opts.pause && pendingSteerPause !== null); iteration++) {
297
+ // Boundary flush. Captures the state that, on resume, makes this
298
+ // iteration the one we re-enter. A pause firing inside this
299
+ // iteration's body resumes here; a pause firing AFTER this
300
+ // iteration completes gets re-captured at the next boundary.
301
+ await flushState(iteration);
302
+
303
+ // PR 7 — human steer-pause boundary. Take a pause when a human steer is
304
+ // pending: a fresh poller flag, OR a pendingSteerPause restored from a prior
305
+ // pass. Persist the intent BEFORE pausing so the snapshot carries it; on
306
+ // resume the same corePause (keyed on agentId:iteration) returns the
307
+ // decision — no re-prompt. UNGATED: a human can steer ANY running agent,
308
+ // regardless of `mode`. (`mode` governs whether the agent may pause ITSELF.)
309
+ let steerDecision: SteerDecision | null = null;
310
+ if (opts.pause) {
311
+ // Consume a fresh poller flag ONLY when nothing is already staged —
312
+ // evaluating consume() unconditionally would clear (and discard) a steer
313
+ // that lands while a self-pause or a restored intent is already pending.
314
+ // The steer's reason + correlationKey ride through to the pause row so
315
+ // resumePauseByKey can resolve it (a bare flag dropped both).
316
+ if (pendingSteerPause === null) {
317
+ const steer = opts.consumeSteerPending?.() ?? null;
318
+ if (steer) {
319
+ pendingSteerPause = {
320
+ reason: steer.reason ?? "human steer",
321
+ correlationKey: steer.correlationKey,
322
+ at: Date.now(),
323
+ };
324
+ }
325
+ }
326
+ if (pendingSteerPause !== null) {
327
+ await flushState(iteration); // intent durable before the throw
328
+ steerDecision = await opts.pause<SteerDecision>(
329
+ {
330
+ reason: pendingSteerPause.reason,
331
+ ...(pendingSteerPause.correlationKey !== null ? { correlationKey: pendingSteerPause.correlationKey } : {}),
332
+ schema: SteerDecisionSchema,
333
+ },
334
+ { agentId, iteration },
335
+ );
336
+ // Reached ONLY on resume — the fresh pass threw PauseSignal above and
337
+ // unwound to serveStep.
338
+ pendingSteerPause = null;
339
+ await flushState(iteration); // commit the cleared intent
340
+ }
341
+ }
342
+
162
343
  const procCtx = buildProcCtx(iteration + 1);
163
- const initialPrompt = opts.buildPrompt(lastStatus, iteration);
344
+ let initialPrompt = opts.buildPrompt(lastStatus, iteration);
345
+ if (steerDecision) {
346
+ // Deliver the human's answer as the agent's next user turn by appending
347
+ // it to the iteration prompt. `sendMessage({ prompt })` is the one input
348
+ // every runtime takes, so this is uniform — no per-runtime delivery path,
349
+ // no capability flag. On it>0 the prompt is PROTOCOL_SUFFIX and the
350
+ // runtime carries prior history (session resume / restored messages); on
351
+ // it==0 it's the original task, so the agent gets task-then-steer. No
352
+ // re-prompt — earlier turns are never resent.
353
+ const who = steerDecision.actor ? ` from ${steerDecision.actor}` : "";
354
+ initialPrompt = `${initialPrompt}\n\n[human steer${who}]: ${steerDecision.message}`;
355
+ }
164
356
 
165
357
  // processInput chain — deny ends the loop; abort ends the loop.
166
358
  const inputVerdict = await runProcessorChain(processors, (p) => p.processInput, initialPrompt, procCtx);
@@ -175,7 +367,19 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
175
367
  process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
176
368
 
177
369
  let responseText = "";
178
- for await (const rawMsg of client.sendMessage({ prompt, sessionId: iteration > 0 ? lastSessionId : undefined, iteration: iteration + 1 })) {
370
+ // The single `opts.inbox` is shared across iterations, but the
371
+ // runtime treats it as a per-call iterable: each iteration's
372
+ // sendMessage starts its own for-await over the same queue, so
373
+ // a message pushed mid-iteration goes into the current call's
374
+ // streaming input, and a message pushed between iterations gets
375
+ // picked up by the next iteration's for-await (AsyncQueue
376
+ // semantics — buffered until consumed).
377
+ for await (const rawMsg of client.sendMessage({
378
+ prompt,
379
+ sessionId: iteration > 0 ? lastSessionId : undefined,
380
+ iteration: iteration + 1,
381
+ ...(opts.inbox ? { inboxStream: opts.inbox } : {}),
382
+ })) {
179
383
  // processOutput chain — deny drops the message from accumulation;
180
384
  // abort ends the loop. Continue carries the (possibly mutated)
181
385
  // message forward.
@@ -202,23 +406,29 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
202
406
  lastResponseText = responseText;
203
407
 
204
408
  let status = parseAgentStatus(responseText);
205
- // Parse the response once per iteration; both the schema fast path
206
- // below and the exit-signal-driven validation a few lines down
207
- // need the same parsed payload. `null` when there's no <response>
208
- // block AND no top-level JSON object.
209
- const rawResponse = opts.responseSchema
409
+ // Inline safeParse (instead of letting parseAgentResponse validate)
410
+ // so a schema failure surfaces via lastResponseValidationError on
411
+ // the next iteration — the model needs that feedback to fix its
412
+ // output.
413
+ const rawResponse: unknown = opts.responseSchema
210
414
  ? parseAgentResponse(responseText) ?? parseRawJsonResponse(responseText)
211
415
  : null;
416
+ const validateResponse = (target: unknown) => {
417
+ const parsed = opts.responseSchema!.safeParse(target);
418
+ if (!parsed.success) lastResponseValidationError = parsed.error.message;
419
+ return parsed;
420
+ };
212
421
  if (opts.responseSchema && rawResponse !== null) {
213
- const parsed = opts.responseSchema.safeParse(rawResponse);
422
+ const parsed = validateResponse(rawResponse);
214
423
  if (parsed.success) {
215
424
  const successStatus = status ?? { summary: "structured response completed", completed: [], blockers: [], exit_signal: true };
216
425
  // Successful schema-validation is progress, not a warning.
217
426
  process.stdout.write(`${logLabel} structured response validated after ${iteration + 1}/${maxIterations} iterations\n`);
427
+ const result: AgentLoopResult<TResponse> = { agentId, label, sessionId: lastSessionId, lastStatus: successStatus, iterations: iteration + 1, response: parsed.data };
428
+ await writeSettledState(result);
218
429
  opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: iteration + 1, durationMs: Date.now() - startedAt });
219
- return { agentId, label, sessionId: lastSessionId, lastStatus: successStatus, iterations: iteration + 1, response: parsed.data };
430
+ return result;
220
431
  }
221
- lastResponseValidationError = parsed.error.message;
222
432
  }
223
433
  completedIterations = iteration + 1;
224
434
  // Per-iteration status is informational progress. The presence
@@ -243,9 +453,8 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
243
453
  // fields (`exit_signal`, `blockers`, …) the agent emitted in a
244
454
  // separate <status> block. Distinct from the fast path above,
245
455
  // which validates the raw response alone.
246
- const parsed = opts.responseSchema.safeParse({ ...status, ...(rawResponse as object) });
456
+ const parsed = validateResponse({ ...status, ...(rawResponse as object) });
247
457
  if (!parsed.success) {
248
- lastResponseValidationError = parsed.error.message;
249
458
  process.stderr.write(`${logLabel} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(rawResponse).slice(0, 400)}\n`);
250
459
  status = { ...status!, exit_signal: false, blockers: [`<response> schema validation failed: ${parsed.error.message}`] };
251
460
  opts.onIteration?.(iteration + 1, status);
@@ -255,8 +464,27 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
255
464
  }
256
465
  // Settled successfully — progress, not a warning.
257
466
  process.stdout.write(`${logLabel} done after ${iteration + 1}/${maxIterations} iterations\n`);
467
+ const result: AgentLoopResult<TResponse> = { agentId, label, sessionId: lastSessionId, lastStatus: status, iterations: iteration + 1, response: response as TResponse };
468
+ await writeSettledState(result);
258
469
  opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: iteration + 1, durationMs: Date.now() - startedAt });
259
- return { agentId, label, sessionId: lastSessionId, lastStatus: status, iterations: iteration + 1, response: response as TResponse };
470
+ return result;
471
+ }
472
+
473
+ // PR 7 — agent self-pause (mode: "hitl" only). The agent set needs_input in
474
+ // its <status> and ended its turn. Stage a pause the NEXT boundary takes —
475
+ // the same machinery as a human steer, triggered by the agent's own status.
476
+ // The boundary pause itself is UNGATED; only this TRIGGER is gated by mode,
477
+ // so an `auto` agent's needs_input is ignored — it just keeps iterating.
478
+ if (opts.pause && opts.mode === "hitl" && status?.needs_input && pendingSteerPause === null) {
479
+ pendingSteerPause = {
480
+ reason: status.question?.trim() || "the agent requested human input",
481
+ correlationKey: null,
482
+ at: Date.now(),
483
+ };
484
+ blockerStreak = null; // an explicit ask is not a stuck-loop
485
+ // Pause fires at the next boundary; the loop condition guarantees that
486
+ // boundary runs even when this was the final budgeted iteration.
487
+ continue;
260
488
  }
261
489
 
262
490
  if (!status) {
@@ -286,9 +514,15 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
286
514
  if (opts.responseSchema)
287
515
  throw new Error(`${logLabel} did not produce a valid <response> after ${maxIterations} iterations${lastResponseValidationError ? `: ${lastResponseValidationError}` : ""}. Response tail: ${lastResponseText.slice(-600)}`);
288
516
  process.stderr.write(`${logLabel} exhausted ${maxIterations} iterations, proceeding with available work\n`);
517
+ const result: AgentLoopResult<TResponse> = { agentId, label, sessionId: lastSessionId, lastStatus, iterations: maxIterations };
518
+ await writeSettledState(result);
289
519
  opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: maxIterations, durationMs: Date.now() - startedAt });
290
- return { agentId, label, sessionId: lastSessionId, lastStatus, iterations: maxIterations };
520
+ return result;
291
521
  } catch (err) {
522
+ // A boundary pause (PauseSignal) is control flow, not a failure — let it
523
+ // propagate so serveStep emits the pause sentinel; do not settle the agent
524
+ // as failed.
525
+ if (isPauseSignal(err)) throw err;
292
526
  opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "failed", iterations: completedIterations, durationMs: Date.now() - startedAt, failureReason: err instanceof Error ? err.message : String(err) });
293
527
  throw err;
294
528
  }
@@ -0,0 +1,61 @@
1
+ /**
2
+ * AsyncQueue — single-producer/single-consumer push-iterable.
3
+ *
4
+ * The Claude Agent SDK's streaming-input mode wants
5
+ * `prompt: AsyncIterable<SDKUserMessage>`. We need a primitive that:
6
+ * - lets the agent loop `push()` user turns from outside the iterator
7
+ * (initial prompt, plus async-arriving inbox messages)
8
+ * - lets the SDK's `for await` consume them in order, blocking until
9
+ * the next item arrives
10
+ * - terminates cleanly via `close()` so the SDK sees the iterable
11
+ * drain and finalises the assistant turn
12
+ *
13
+ * Why not an `EventEmitter` or RxJS — both are overkill for one shape.
14
+ * The implementation is ~30 lines of native Promise plumbing and stays
15
+ * inside the SDK package so it can be the canonical way agent-loop
16
+ * builds its prompt iterable.
17
+ */
18
+
19
+ export class AsyncQueue<T> implements AsyncIterable<T> {
20
+ private readonly buffer: T[] = [];
21
+ private readonly resolvers: Array<(v: IteratorResult<T>) => void> = [];
22
+ private done = false;
23
+
24
+ /** Push one item. If a consumer is awaiting, it wakes immediately;
25
+ * otherwise the item is buffered until the next `next()` call. */
26
+ push(value: T): void {
27
+ if (this.done) return;
28
+ const resolve = this.resolvers.shift();
29
+ if (resolve) {
30
+ resolve({ value, done: false });
31
+ } else {
32
+ this.buffer.push(value);
33
+ }
34
+ }
35
+
36
+ /** Signal end-of-stream. Any pending awaiters resolve as
37
+ * `{ done: true }`; subsequent `push()` calls are silent no-ops. */
38
+ close(): void {
39
+ if (this.done) return;
40
+ this.done = true;
41
+ while (this.resolvers.length > 0) {
42
+ const resolve = this.resolvers.shift()!;
43
+ resolve({ value: undefined as unknown as T, done: true });
44
+ }
45
+ }
46
+
47
+ [Symbol.asyncIterator](): AsyncIterator<T> {
48
+ return {
49
+ next: (): Promise<IteratorResult<T>> => {
50
+ const buffered = this.buffer.shift();
51
+ if (buffered !== undefined) return Promise.resolve({ value: buffered, done: false });
52
+ if (this.done) return Promise.resolve({ value: undefined as unknown as T, done: true });
53
+ return new Promise((resolve) => { this.resolvers.push(resolve); });
54
+ },
55
+ return: (): Promise<IteratorResult<T>> => {
56
+ this.close();
57
+ return Promise.resolve({ value: undefined as unknown as T, done: true });
58
+ },
59
+ };
60
+ }
61
+ }
@@ -15,8 +15,18 @@ export const AgentMessageSchema = z.object({
15
15
  timestamp: z.string(),
16
16
  }).passthrough() as unknown as z.ZodType<AgentMessage>;
17
17
 
18
- export function parseAgentResponse(text: string): unknown {
18
+ /** Parse the `<response>...</response>` block from agent text. Returns
19
+ * `null` if missing or malformed. Callers run schema validation on
20
+ * the returned `unknown` directly — the earlier generic overload that
21
+ * did the safeParse inline returned `null` on schema failure,
22
+ * swallowing the validation error without setting any breadcrumb on
23
+ * the loop's `lastResponseValidationError`. No caller used that
24
+ * overload; the dedicated `responseSchema` path in `agent-loop.ts`
25
+ * does the parse-with-error-capture itself. */
26
+ export function parseAgentResponse(text: string): unknown | null {
19
27
  const match = text.match(/<response>([\s\S]*?)<\/response>/);
20
28
  if (!match) return null;
21
- try { return JSON.parse(match[1].trim()); } catch { return null; }
29
+ try {
30
+ return JSON.parse(match[1].trim());
31
+ } catch { return null; }
22
32
  }