@agent-compose/sdk 0.5.1 → 0.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/active-step.d.ts +60 -0
- package/dist/agent/agent-loop-steer.test.d.ts +1 -0
- package/dist/agent/agent-loop.d.ts +46 -0
- package/dist/agent/async-queue.d.ts +29 -0
- package/dist/agent/protocol.d.ts +9 -1
- package/dist/agent/resolve-agent-id.test.d.ts +1 -0
- package/dist/agent/run-agent.d.ts +16 -3
- package/dist/agent/steer-control.d.ts +57 -0
- package/dist/agent/steer-control.test.d.ts +1 -0
- package/dist/client.d.ts +161 -0
- package/dist/index.d.ts +16 -6
- package/dist/index.js +1586 -158
- package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/errors.test.d.ts +1 -0
- package/dist/pause/__tests__/manager.test.d.ts +1 -0
- package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
- package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
- package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
- package/dist/pause/checkpoint.d.ts +28 -0
- package/dist/pause/errors.d.ts +52 -0
- package/dist/pause/manager.d.ts +63 -0
- package/dist/pause/pause-core.d.ts +101 -0
- package/dist/pause/state-dir.d.ts +80 -0
- package/dist/pause/wrappers.d.ts +41 -0
- package/dist/request-context/request-context.d.ts +12 -0
- package/dist/runtimes/_cli-agent.d.ts +72 -0
- package/dist/runtimes/amp.d.ts +22 -0
- package/dist/runtimes/claude.d.ts +6 -0
- package/dist/runtimes/codex.d.ts +20 -0
- package/dist/runtimes/openai-desktop.d.ts +2 -0
- package/dist/runtimes/openai-desktop.js +1578 -157
- package/dist/runtimes/vercel.d.ts +53 -1
- package/dist/runtimes/vercel.js +60 -8
- package/dist/runtimes/vercel.test.d.ts +1 -0
- package/dist/sse.d.ts +2 -3
- package/dist/step-invocation/index.d.ts +2 -2
- package/dist/step-invocation/invoker.d.ts +3 -0
- package/dist/step-invocation/protocol.d.ts +12 -0
- package/dist/step-invocation/server.d.ts +1 -0
- package/dist/step-invocation/types.d.ts +40 -5
- package/dist/types/events.d.ts +9 -0
- package/dist/types/execution-context.d.ts +25 -0
- package/dist/types/protocol.d.ts +8 -0
- package/dist/types/runtime.d.ts +55 -0
- package/dist/types/sandbox.d.ts +4 -4
- package/dist/utils/schemas.d.ts +2 -0
- package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
- package/dist/workflow-steps/index.d.ts +2 -0
- package/dist/workflow-steps/observability.d.ts +43 -11
- package/dist/workflow-steps/run-callback.d.ts +39 -0
- package/dist/workflow-steps/runner.d.ts +8 -0
- package/package.json +1 -1
- package/src/active-step.ts +124 -0
- package/src/agent/agent-loop.ts +253 -19
- package/src/agent/async-queue.ts +61 -0
- package/src/agent/protocol.ts +12 -2
- package/src/agent/run-agent.ts +184 -8
- package/src/agent/steer-control.ts +125 -0
- package/src/client.ts +277 -0
- package/src/index.ts +38 -4
- package/src/pause/checkpoint.ts +44 -0
- package/src/pause/errors.ts +70 -0
- package/src/pause/manager.ts +177 -0
- package/src/pause/pause-core.ts +267 -0
- package/src/pause/state-dir.ts +262 -0
- package/src/pause/wrappers.ts +79 -0
- package/src/request-context/request-context.ts +17 -2
- package/src/runtimes/_cli-agent.ts +161 -0
- package/src/runtimes/amp.ts +94 -0
- package/src/runtimes/claude.ts +101 -6
- package/src/runtimes/codex.ts +109 -0
- package/src/runtimes/openai-desktop.ts +11 -0
- package/src/runtimes/vercel.ts +78 -2
- package/src/sandbox.ts +39 -20
- package/src/sse.ts +8 -6
- package/src/step-invocation/index.ts +2 -1
- package/src/step-invocation/invoker.ts +107 -29
- package/src/step-invocation/protocol.ts +16 -0
- package/src/step-invocation/server.ts +45 -12
- package/src/step-invocation/types.ts +43 -7
- package/src/tools/coding.ts +16 -5
- package/src/types/events.ts +9 -0
- package/src/types/execution-context.ts +25 -0
- package/src/types/protocol.ts +8 -0
- package/src/types/runtime.ts +52 -0
- package/src/types/sandbox.ts +8 -4
- package/src/types/workflow.ts +6 -1
- package/src/utils/bundler.ts +8 -3
- package/src/utils/schemas.ts +2 -0
- package/src/workflow-steps/index.ts +3 -0
- package/src/workflow-steps/observability.ts +84 -13
- package/src/workflow-steps/run-callback.ts +72 -0
- package/src/workflow-steps/runner.ts +70 -8
- package/dist/utils/discovery.d.ts +0 -2
- package/src/utils/discovery.ts +0 -4
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Runner → server live event emitter.
|
|
3
|
+
*
|
|
4
|
+
* Pairs with the server-side `/api/v1/internal/runs/:runId/steps/:stepIndex/events`
|
|
5
|
+
* route. The runner reads `AGENT_COMPOSE_URL`, `AGENT_COMPOSE_RUN_TOKEN`,
|
|
6
|
+
* and `RUN_ID` from env; if all three are present it builds a POST
|
|
7
|
+
* function that sends each agent lifecycle event to the server as
|
|
8
|
+
* soon as the agent loop emits it.
|
|
9
|
+
*
|
|
10
|
+
* Non-blocking: the agent loop doesn't await the emitter — the runtime
|
|
11
|
+
* keeps emitting events at full speed. But the emitter still returns a
|
|
12
|
+
* `Promise<boolean>` so the collector can later decide whether to
|
|
13
|
+
* include each event in the batch-end durable backstop. `true` = server
|
|
14
|
+
* accepted the event (don't re-deliver in batch); `false` = POST
|
|
15
|
+
* failed (DO re-deliver). The collector awaits these promises with a
|
|
16
|
+
* short grace window at snapshot time.
|
|
17
|
+
*
|
|
18
|
+
* This is the seam that lets us avoid the double-write problem: every
|
|
19
|
+
* successfully-ack'd live event is stripped from `observability.events`
|
|
20
|
+
* before the batch flush runs, so the durable backstop only carries
|
|
21
|
+
* events that genuinely failed live delivery. No more two paths writing
|
|
22
|
+
* the same row + double pg_notify.
|
|
23
|
+
*/
|
|
24
|
+
import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
|
|
25
|
+
export type LiveAgentEventEmitter = (event: AgentLifecycleEvent, seq: number) => Promise<boolean>;
|
|
26
|
+
/**
|
|
27
|
+
* Build a live emitter from process env + the caller-supplied stepIndex.
|
|
28
|
+
* Returns `undefined` when any required env var is missing (local tests,
|
|
29
|
+
* non-sandbox callers) so the caller can wire `undefined` straight
|
|
30
|
+
* through to the collector and get batch-only delivery without
|
|
31
|
+
* conditional plumbing.
|
|
32
|
+
*
|
|
33
|
+
* `stepIndex` is a parameter rather than an env read because
|
|
34
|
+
* `serveStep` scrubs `AC_STEP_INDEX` from `process.env` before invoking
|
|
35
|
+
* the workflow handler (to keep the transport envelope unreachable
|
|
36
|
+
* from user code); the handler still holds the parsed integer and
|
|
37
|
+
* passes it here.
|
|
38
|
+
*/
|
|
39
|
+
export declare function makeRunCallbackEmitterFromEnv(stepIndex: number): LiveAgentEventEmitter | undefined;
|
|
@@ -27,6 +27,7 @@ import type { RequestContext } from "../request-context/request-context.js";
|
|
|
27
27
|
import type { SandboxProvider } from "../types/sandbox.js";
|
|
28
28
|
import type { WorkflowRun, WorkflowCtx } from "../types/workflow.js";
|
|
29
29
|
import { type StepObservability } from "./observability.js";
|
|
30
|
+
import type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
30
31
|
export declare class StepValidationError extends Error {
|
|
31
32
|
readonly stepName: string;
|
|
32
33
|
readonly side: "input" | "output";
|
|
@@ -67,6 +68,11 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
|
|
|
67
68
|
onStepStarted?(stepIndex: number, stepName: string): void | Promise<void>;
|
|
68
69
|
/** Child workflow invocation implementation. Defaults to a clear unsupported error. */
|
|
69
70
|
invokeChild?: WorkflowCtx["invokeChild"];
|
|
71
|
+
/** Optional live-stream emitter for agent lifecycle events. The runner
|
|
72
|
+
* passes a fetch-based emitter wired to the per-run callback token so
|
|
73
|
+
* the dashboard sees events as the agent loop produces them; tests
|
|
74
|
+
* leave it undefined and get batch-only delivery. */
|
|
75
|
+
liveAgentEventEmitter?: LiveAgentEventEmitter;
|
|
70
76
|
}
|
|
71
77
|
export interface RunWorkflowStepsResult<TOutput> {
|
|
72
78
|
output: TOutput;
|
|
@@ -84,6 +90,8 @@ export interface RunWorkflowSingleStepOpts {
|
|
|
84
90
|
sandbox?: SandboxProvider;
|
|
85
91
|
abortSignal?: AbortSignal;
|
|
86
92
|
invokeChild?: WorkflowCtx["invokeChild"];
|
|
93
|
+
/** Optional live-stream emitter — see `RunWorkflowStepsOpts.liveAgentEventEmitter`. */
|
|
94
|
+
liveAgentEventEmitter?: LiveAgentEventEmitter;
|
|
87
95
|
}
|
|
88
96
|
/** Result of one step run — output plus whatever the step's observability
|
|
89
97
|
* hooks recorded. `observability` is undefined when nothing was buffered,
|
package/package.json
CHANGED
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The step currently executing in this runner subprocess.
|
|
3
|
+
*
|
|
4
|
+
* Step-mode derivations that must be byte-identical across a pause-resume
|
|
5
|
+
* re-entry — the agent loop's `agentId` (which names `agent-<id>.json`) and
|
|
6
|
+
* the pause primitive's `pauseId` — need the running step's index. They
|
|
7
|
+
* CANNOT read it from `process.env.AC_STEP_INDEX`: `serveStep` scrubs the
|
|
8
|
+
* transport envs (`AC_STEP_MODE`, `AC_STEP_INDEX`, the result token, …) from
|
|
9
|
+
* `process.env` *before* the user step body runs, so by the time `agent()` /
|
|
10
|
+
* `ctx.pause` execute those reads return `undefined`
|
|
11
|
+
* (`step-invocation/server.ts` `scrubProtocolEnvs`). Reading scrubbed env was
|
|
12
|
+
* a latent resume bug: `agentId` fell through to `randomUUID()` on every
|
|
13
|
+
* subprocess, so resume never matched the prior `agent-<id>.json`.
|
|
14
|
+
*
|
|
15
|
+
* Instead the step runner sets the index here from the un-scrubbed `stepIndex`
|
|
16
|
+
* it already holds, and the derivations read it from here.
|
|
17
|
+
*
|
|
18
|
+
* Async-local rather than process-global: public in-process execution can run
|
|
19
|
+
* multiple workflows concurrently, and ids must stay scoped to the async step
|
|
20
|
+
* body that is deriving them. A fresh subprocess on resume starts with a fresh
|
|
21
|
+
* async context, the runner re-sets the same `stepIndex`, and the per-step
|
|
22
|
+
* call-order counters re-derive identical ids — which is exactly what lets
|
|
23
|
+
* resume find its on-disk state.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import { AsyncLocalStorage } from "node:async_hooks";
|
|
27
|
+
|
|
28
|
+
export interface ActiveStep {
|
|
29
|
+
stepIndex: number;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
interface ActiveStepState extends ActiveStep {
|
|
33
|
+
nextAgentCallIndex: number;
|
|
34
|
+
pauseOrdinalByScope: Map<string, number>;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
const activeStepStorage = new AsyncLocalStorage<ActiveStepState>();
|
|
38
|
+
|
|
39
|
+
// Test/dev fallback for direct `setActiveStep(...)` users. The runner uses
|
|
40
|
+
// `runWithActiveStep`, so production step execution is async-local.
|
|
41
|
+
let fallbackActive: ActiveStepState | null = null;
|
|
42
|
+
|
|
43
|
+
// Cross-SDK-instance bridge. `bundleWorkflow` builds the workflow with
|
|
44
|
+
// `Bun.build` and NO `external`, so the bundle inlines its OWN copy of this
|
|
45
|
+
// module — a distinct `AsyncLocalStorage` instance from the runner's. The
|
|
46
|
+
// runner sets the active step via `runWithActiveStep` on ITS instance; the
|
|
47
|
+
// bundle's `agent()` reads `currentState()` on the BUNDLE's instance and would
|
|
48
|
+
// otherwise see null (→ a random agentId and an unwired steer-pause). So
|
|
49
|
+
// `runWithActiveStep` ALSO publishes the state on a realm-shared `globalThis`
|
|
50
|
+
// slot the bundle's copy can read. Consulted ONLY when the local ALS is empty,
|
|
51
|
+
// so in-process execution keeps full async-local isolation between concurrent
|
|
52
|
+
// runs (the ALS takes precedence below).
|
|
53
|
+
const BRIDGE_KEY = Symbol.for("agent-compose.activeStep.bridge");
|
|
54
|
+
function bridgeSlot(): { state: ActiveStepState | null } {
|
|
55
|
+
const g = globalThis as unknown as Record<symbol, { state: ActiveStepState | null } | undefined>;
|
|
56
|
+
return (g[BRIDGE_KEY] ??= { state: null });
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function makeState(step: ActiveStep): ActiveStepState {
|
|
60
|
+
return {
|
|
61
|
+
stepIndex: step.stepIndex,
|
|
62
|
+
nextAgentCallIndex: 0,
|
|
63
|
+
pauseOrdinalByScope: new Map(),
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
function currentState(): ActiveStepState | null {
|
|
68
|
+
return activeStepStorage.getStore() ?? bridgeSlot().state ?? fallbackActive;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/** Set (or clear, with `null`) the step currently running. Called by the
|
|
72
|
+
* step runner around `step.run(...)`. Prefer `runWithActiveStep` for real
|
|
73
|
+
* async execution; this setter exists for tests and small synchronous helpers. */
|
|
74
|
+
export function setActiveStep(step: ActiveStep | null): void {
|
|
75
|
+
fallbackActive = step ? makeState(step) : null;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** Run `fn` with a step context scoped to this async call tree. */
|
|
79
|
+
export async function runWithActiveStep<T>(step: ActiveStep, fn: () => Promise<T>): Promise<T> {
|
|
80
|
+
return activeStepStorage.run(makeState(step), fn);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
/** Publish (or clear, with `null`) the active step on the cross-instance bridge.
|
|
84
|
+
* Called ONLY by the sandbox step runner (`runWorkflowSingleStep`), which runs
|
|
85
|
+
* exactly one step per subprocess — so there is no concurrency to race the
|
|
86
|
+
* single global slot. The bundle's `agent()` reads this when its own ALS is
|
|
87
|
+
* empty. Returns the prior value so the caller can restore it. */
|
|
88
|
+
export function setActiveStepBridge(step: ActiveStep | null): { state: ActiveStepState | null } {
|
|
89
|
+
const slot = bridgeSlot();
|
|
90
|
+
const prev = { state: slot.state };
|
|
91
|
+
slot.state = step ? makeState(step) : null;
|
|
92
|
+
return prev;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** Restore a bridge value captured by `setActiveStepBridge`. */
|
|
96
|
+
export function restoreActiveStepBridge(prev: { state: ActiveStepState | null }): void {
|
|
97
|
+
bridgeSlot().state = prev.state;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
/** The step currently running, or `null` outside step execution (local
|
|
102
|
+
* tests, dev, between steps). */
|
|
103
|
+
export function getActiveStep(): ActiveStep | null {
|
|
104
|
+
const state = currentState();
|
|
105
|
+
return state ? { stepIndex: state.stepIndex } : null;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/** Return the next implicit `agent()` call coordinate for the active step. */
|
|
109
|
+
export function nextAgentCallInActiveStep(): { stepIndex: number; callIndex: number } | null {
|
|
110
|
+
const state = currentState();
|
|
111
|
+
if (!state) return null;
|
|
112
|
+
const callIndex = state.nextAgentCallIndex;
|
|
113
|
+
state.nextAgentCallIndex += 1;
|
|
114
|
+
return { stepIndex: state.stepIndex, callIndex };
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** Return the next pause ordinal in `scope` for the active step invocation. */
|
|
118
|
+
export function nextPauseOrdinalInActiveStep(scope: string): number | null {
|
|
119
|
+
const state = currentState();
|
|
120
|
+
if (!state) return null;
|
|
121
|
+
const next = state.pauseOrdinalByScope.get(scope) ?? 0;
|
|
122
|
+
state.pauseOrdinalByScope.set(scope, next + 1);
|
|
123
|
+
return next;
|
|
124
|
+
}
|
package/src/agent/agent-loop.ts
CHANGED
|
@@ -11,6 +11,9 @@ import { randomUUID } from "node:crypto";
|
|
|
11
11
|
import type { Processor, ProcessorContext } from "../processors/processor.js";
|
|
12
12
|
import { runProcessorChain } from "../processors/runner.js";
|
|
13
13
|
import { RequestContext } from "../request-context/request-context.js";
|
|
14
|
+
import { PauseManager } from "../pause/manager.js";
|
|
15
|
+
import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
|
|
16
|
+
import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
|
|
14
17
|
|
|
15
18
|
export const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
|
|
16
19
|
|
|
@@ -42,7 +45,7 @@ export type AgentMessageSummary =
|
|
|
42
45
|
| { type: "init"; sessionId: string }
|
|
43
46
|
| { type: "text"; text: string }
|
|
44
47
|
| { type: "thinking"; text: string }
|
|
45
|
-
| { type: "tool_use"; toolName: string; toolUseId: string; toolInputPreview: string }
|
|
48
|
+
| { type: "tool_use"; toolName: string; toolUseId: string; toolInput: Record<string, unknown>; toolInputPreview: string }
|
|
46
49
|
| { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
|
|
47
50
|
| { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number }
|
|
48
51
|
| { type: "done"; sessionId: string }
|
|
@@ -65,7 +68,7 @@ export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
|
|
|
65
68
|
case "init": return { type: "init", sessionId: msg.sessionId };
|
|
66
69
|
case "text": return { type: "text", text: msg.text };
|
|
67
70
|
case "thinking": return { type: "thinking", text: msg.text };
|
|
68
|
-
case "tool_use": return { type: "tool_use", toolName: msg.toolName, toolUseId: msg.toolUseId, toolInputPreview: preview(msg.toolInput) };
|
|
71
|
+
case "tool_use": return { type: "tool_use", toolName: msg.toolName, toolUseId: msg.toolUseId, toolInput: msg.toolInput, toolInputPreview: preview(msg.toolInput) };
|
|
69
72
|
case "tool_result": return { type: "tool_result", toolUseId: msg.toolUseId, output: truncate(msg.output), isError: msg.isError };
|
|
70
73
|
case "usage": return {
|
|
71
74
|
type: "usage", inputTokens: msg.inputTokens, outputTokens: msg.outputTokens,
|
|
@@ -78,7 +81,16 @@ export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
|
|
|
78
81
|
}
|
|
79
82
|
|
|
80
83
|
export type AgentLifecycleEvent =
|
|
81
|
-
| {
|
|
84
|
+
| {
|
|
85
|
+
event: "agent.spawned"; at: number; agentId: string; label: string;
|
|
86
|
+
allowedTools?: string[];
|
|
87
|
+
/** Resolved model id used for this agent (e.g. `claude-sonnet-4-6`).
|
|
88
|
+
* Read off `client.model` after runtime construction. */
|
|
89
|
+
model?: string;
|
|
90
|
+
/** Short runtime self-identifier (`claude`, `openai-desktop`, …).
|
|
91
|
+
* Drives the per-agent runtime icon on the dashboard. */
|
|
92
|
+
runtimeKind?: string;
|
|
93
|
+
}
|
|
82
94
|
| { event: "agent.message"; at: number; agentId: string; label: string; iteration: number; message: AgentMessageSummary }
|
|
83
95
|
| { event: "agent.iteration"; at: number; agentId: string; label: string; iteration: number; status: AgentStatus | null }
|
|
84
96
|
| { event: "agent.settled"; at: number; agentId: string; label: string; outcome: "success" | "failed"; iterations: number; durationMs: number; failureReason?: string };
|
|
@@ -102,6 +114,37 @@ export interface AgentLoopOpts<TResponse = unknown> {
|
|
|
102
114
|
processors?: readonly Processor[];
|
|
103
115
|
/** Per-run typed bag — propagated to every processor as `ctx.requestContext`. */
|
|
104
116
|
requestContext?: RequestContext;
|
|
117
|
+
/**
|
|
118
|
+
* Mid-iteration user-message injection. When set, the agent loop
|
|
119
|
+
* builds a per-iteration AsyncQueue and hands it to the runtime as
|
|
120
|
+
* `inboxStream`. The runner-side inbox poller calls `push()` on
|
|
121
|
+
* `inbox` when a user message arrives; the runtime (streaming-input
|
|
122
|
+
* Claude Agent SDK) picks it up at the next safe boundary.
|
|
123
|
+
*
|
|
124
|
+
* The loop drains `inbox` into the per-iteration queue while the
|
|
125
|
+
* iteration is active and stops draining at iteration end — any
|
|
126
|
+
* messages that arrive between iterations are buffered on `inbox`
|
|
127
|
+
* and drain into the NEXT iteration's queue. So no message is lost,
|
|
128
|
+
* but mid-iteration injection requires the runtime to support
|
|
129
|
+
* streaming input (ignored otherwise — handled at the runtime).
|
|
130
|
+
*/
|
|
131
|
+
inbox?: import("./async-queue.js").AsyncQueue<{ text: string; senderName?: string | null }>;
|
|
132
|
+
/** PR 7 steer-pause: the boundary calls this to pause the workflow for a
|
|
133
|
+
* human steer. agent() builds it (a corePause closed over runId/stepIndex);
|
|
134
|
+
* the loop supplies the agentScope so the pauseId is stable across resume.
|
|
135
|
+
* Absent ⇒ no steer pause (local tests, non-sandbox callers). */
|
|
136
|
+
pause?: <T = unknown>(
|
|
137
|
+
req: { reason: string; correlationKey?: string; schema?: z.ZodType<T> },
|
|
138
|
+
agentScope: { agentId: string; iteration: number },
|
|
139
|
+
) => Promise<T>;
|
|
140
|
+
/** PR 7: take-once read of this agent's pending steer (set by the control
|
|
141
|
+
* poller). Returns the steer's payload (reason / correlationKey) or null.
|
|
142
|
+
* The boundary consumes it once per check, and only when no steer is
|
|
143
|
+
* already staged. */
|
|
144
|
+
consumeSteerPending?: () => SteerPayload | null;
|
|
145
|
+
/** PR 7: `auto` (autonomous, default) or `hitl` (can be steer-paused). The
|
|
146
|
+
* boundary is inert unless `hitl`. */
|
|
147
|
+
mode?: "auto" | "hitl";
|
|
105
148
|
}
|
|
106
149
|
|
|
107
150
|
export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TResponse>): Promise<AgentLoopResult<TResponse>> {
|
|
@@ -154,13 +197,162 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
154
197
|
let completedIterations = 0;
|
|
155
198
|
let lastResponseText = "";
|
|
156
199
|
let lastResponseValidationError = "";
|
|
200
|
+
// Iteration the loop should resume at on entry. 0 = fresh run; >0 =
|
|
201
|
+
// restored from disk after a pause-resume. The `for` loop reads
|
|
202
|
+
// `startIteration` instead of starting at 0 so already-completed
|
|
203
|
+
// iterations don't re-run.
|
|
204
|
+
let startIteration = 0;
|
|
205
|
+
// Was this invocation resumed from on-disk state? Drives an early
|
|
206
|
+
// call to `runtime.restoreCheckpoint(blob)` and suppresses the
|
|
207
|
+
// `agent.spawned` event (the agent already spawned in the prior
|
|
208
|
+
// subprocess; emitting again would double-count on the dashboard).
|
|
209
|
+
let resumed = false;
|
|
210
|
+
// PR 7: a pending human steer-pause intent. Persisted in loop state so it
|
|
211
|
+
// survives a resume and the boundary re-issues the pause on re-entry.
|
|
212
|
+
let pendingSteerPause: { reason: string; correlationKey: string | null; at: number } | null = null;
|
|
213
|
+
|
|
214
|
+
const pauseManager = new PauseManager(agentId);
|
|
215
|
+
const restore = await pauseManager.restoreAgentLoop<TResponse>(client);
|
|
216
|
+
if (restore.kind === "settled") {
|
|
217
|
+
process.stdout.write(`${logLabel} restored settled agent result from checkpoint after ${restore.result.iterations} iterations\n`);
|
|
218
|
+
return { agentId, label, ...restore.result };
|
|
219
|
+
}
|
|
220
|
+
if (restore.kind === "running") {
|
|
221
|
+
const s = restore.state;
|
|
222
|
+
startIteration = s.iteration;
|
|
223
|
+
completedIterations = s.completedIterations;
|
|
224
|
+
lastSessionId = s.lastSessionId;
|
|
225
|
+
lastStatus = s.lastStatus;
|
|
226
|
+
lastResponseText = s.lastResponseText;
|
|
227
|
+
lastResponseValidationError = s.lastResponseValidationError;
|
|
228
|
+
iterationsWithoutStatus = s.iterationsWithoutStatus;
|
|
229
|
+
blockerStreak = s.blockerStreak;
|
|
230
|
+
pendingSteerPause = s.pendingSteerPause ?? null;
|
|
231
|
+
resumed = true;
|
|
232
|
+
process.stdout.write(`${logLabel} resumed from agent-state checkpoint at iteration ${startIteration}/${maxIterations}\n`);
|
|
233
|
+
}
|
|
234
|
+
if (restore.kind === "invalid") {
|
|
235
|
+
// Forward-incompat or corrupt state file. Fall back to fresh start
|
|
236
|
+
// rather than partial-restore — a misread checkpoint would silently
|
|
237
|
+
// desync the loop from the model's actual conversation history.
|
|
238
|
+
process.stderr.write(
|
|
239
|
+
`${logLabel} agent-state checkpoint at agentId=${agentId} is unrecognised — starting fresh. ` +
|
|
240
|
+
`Schema version mismatch or corrupt file: ${restore.error}\n`,
|
|
241
|
+
);
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
// `agent.spawned` represents one logical agent instance, not one
|
|
245
|
+
// subprocess invocation. Skip on resume so the dashboard's per-agent
|
|
246
|
+
// card count doesn't increment every time a pause-resume re-enters
|
|
247
|
+
// the loop.
|
|
248
|
+
if (!resumed) {
|
|
249
|
+
opts.onAgentLifecycleEvent?.({
|
|
250
|
+
event: "agent.spawned",
|
|
251
|
+
at: startedAt,
|
|
252
|
+
agentId,
|
|
253
|
+
label,
|
|
254
|
+
allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
|
|
255
|
+
...(client.model != null ? { model: client.model } : {}),
|
|
256
|
+
...(client.kind != null ? { runtimeKind: client.kind } : {}),
|
|
257
|
+
});
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
/** Atomically flush loop state + runtime checkpoint to disk. Called
|
|
261
|
+
* at every iteration boundary so a `ctx.pause` firing inside the
|
|
262
|
+
* next iteration's body can rely on the most recent committed state.
|
|
263
|
+
* Two writes (loop state, runtime blob) — both atomic individually;
|
|
264
|
+
* the pair isn't transactional. If a stateful runtime later resumes
|
|
265
|
+
* without its blob, restore fails loudly rather than continuing with
|
|
266
|
+
* an empty conversation. */
|
|
267
|
+
const flushState = async (currentIteration: number): Promise<void> => {
|
|
268
|
+
await pauseManager.saveRunningAgentLoop({
|
|
269
|
+
iteration: currentIteration,
|
|
270
|
+
completedIterations,
|
|
271
|
+
lastSessionId,
|
|
272
|
+
lastStatus,
|
|
273
|
+
lastResponseText,
|
|
274
|
+
lastResponseValidationError,
|
|
275
|
+
iterationsWithoutStatus,
|
|
276
|
+
blockerStreak,
|
|
277
|
+
pendingSteerPause,
|
|
278
|
+
}, client);
|
|
279
|
+
};
|
|
157
280
|
|
|
158
|
-
|
|
281
|
+
const writeSettledState = async (result: AgentLoopResult<TResponse>): Promise<void> => {
|
|
282
|
+
await pauseManager.saveSettledAgentLoop({
|
|
283
|
+
sessionId: result.sessionId,
|
|
284
|
+
lastStatus: result.lastStatus,
|
|
285
|
+
iterations: result.iterations,
|
|
286
|
+
...(result.response !== undefined ? { response: result.response } : {}),
|
|
287
|
+
});
|
|
288
|
+
};
|
|
159
289
|
|
|
160
290
|
try {
|
|
161
|
-
|
|
291
|
+
// The `|| pendingSteerPause` clause lets a staged steer (a self-pause asked on
|
|
292
|
+
// the FINAL budgeted iteration, or a restored intent) run one boundary past
|
|
293
|
+
// the budget so its pause fires and the human's answer gets an iteration —
|
|
294
|
+
// otherwise an end-of-iteration self-pause on the last turn would fall through
|
|
295
|
+
// to the exhaustion path and settle "success" with the question silently lost.
|
|
296
|
+
for (let iteration = startIteration; iteration < maxIterations || (!!opts.pause && pendingSteerPause !== null); iteration++) {
|
|
297
|
+
// Boundary flush. Captures the state that, on resume, makes this
|
|
298
|
+
// iteration the one we re-enter. A pause firing inside this
|
|
299
|
+
// iteration's body resumes here; a pause firing AFTER this
|
|
300
|
+
// iteration completes gets re-captured at the next boundary.
|
|
301
|
+
await flushState(iteration);
|
|
302
|
+
|
|
303
|
+
// PR 7 — human steer-pause boundary. Take a pause when a human steer is
|
|
304
|
+
// pending: a fresh poller flag, OR a pendingSteerPause restored from a prior
|
|
305
|
+
// pass. Persist the intent BEFORE pausing so the snapshot carries it; on
|
|
306
|
+
// resume the same corePause (keyed on agentId:iteration) returns the
|
|
307
|
+
// decision — no re-prompt. UNGATED: a human can steer ANY running agent,
|
|
308
|
+
// regardless of `mode`. (`mode` governs whether the agent may pause ITSELF.)
|
|
309
|
+
let steerDecision: SteerDecision | null = null;
|
|
310
|
+
if (opts.pause) {
|
|
311
|
+
// Consume a fresh poller flag ONLY when nothing is already staged —
|
|
312
|
+
// evaluating consume() unconditionally would clear (and discard) a steer
|
|
313
|
+
// that lands while a self-pause or a restored intent is already pending.
|
|
314
|
+
// The steer's reason + correlationKey ride through to the pause row so
|
|
315
|
+
// resumePauseByKey can resolve it (a bare flag dropped both).
|
|
316
|
+
if (pendingSteerPause === null) {
|
|
317
|
+
const steer = opts.consumeSteerPending?.() ?? null;
|
|
318
|
+
if (steer) {
|
|
319
|
+
pendingSteerPause = {
|
|
320
|
+
reason: steer.reason ?? "human steer",
|
|
321
|
+
correlationKey: steer.correlationKey,
|
|
322
|
+
at: Date.now(),
|
|
323
|
+
};
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
if (pendingSteerPause !== null) {
|
|
327
|
+
await flushState(iteration); // intent durable before the throw
|
|
328
|
+
steerDecision = await opts.pause<SteerDecision>(
|
|
329
|
+
{
|
|
330
|
+
reason: pendingSteerPause.reason,
|
|
331
|
+
...(pendingSteerPause.correlationKey !== null ? { correlationKey: pendingSteerPause.correlationKey } : {}),
|
|
332
|
+
schema: SteerDecisionSchema,
|
|
333
|
+
},
|
|
334
|
+
{ agentId, iteration },
|
|
335
|
+
);
|
|
336
|
+
// Reached ONLY on resume — the fresh pass threw PauseSignal above and
|
|
337
|
+
// unwound to serveStep.
|
|
338
|
+
pendingSteerPause = null;
|
|
339
|
+
await flushState(iteration); // commit the cleared intent
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
|
|
162
343
|
const procCtx = buildProcCtx(iteration + 1);
|
|
163
|
-
|
|
344
|
+
let initialPrompt = opts.buildPrompt(lastStatus, iteration);
|
|
345
|
+
if (steerDecision) {
|
|
346
|
+
// Deliver the human's answer as the agent's next user turn by appending
|
|
347
|
+
// it to the iteration prompt. `sendMessage({ prompt })` is the one input
|
|
348
|
+
// every runtime takes, so this is uniform — no per-runtime delivery path,
|
|
349
|
+
// no capability flag. On it>0 the prompt is PROTOCOL_SUFFIX and the
|
|
350
|
+
// runtime carries prior history (session resume / restored messages); on
|
|
351
|
+
// it==0 it's the original task, so the agent gets task-then-steer. No
|
|
352
|
+
// re-prompt — earlier turns are never resent.
|
|
353
|
+
const who = steerDecision.actor ? ` from ${steerDecision.actor}` : "";
|
|
354
|
+
initialPrompt = `${initialPrompt}\n\n[human steer${who}]: ${steerDecision.message}`;
|
|
355
|
+
}
|
|
164
356
|
|
|
165
357
|
// processInput chain — deny ends the loop; abort ends the loop.
|
|
166
358
|
const inputVerdict = await runProcessorChain(processors, (p) => p.processInput, initialPrompt, procCtx);
|
|
@@ -175,7 +367,19 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
175
367
|
process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
|
|
176
368
|
|
|
177
369
|
let responseText = "";
|
|
178
|
-
|
|
370
|
+
// The single `opts.inbox` is shared across iterations, but the
|
|
371
|
+
// runtime treats it as a per-call iterable: each iteration's
|
|
372
|
+
// sendMessage starts its own for-await over the same queue, so
|
|
373
|
+
// a message pushed mid-iteration goes into the current call's
|
|
374
|
+
// streaming input, and a message pushed between iterations gets
|
|
375
|
+
// picked up by the next iteration's for-await (AsyncQueue
|
|
376
|
+
// semantics — buffered until consumed).
|
|
377
|
+
for await (const rawMsg of client.sendMessage({
|
|
378
|
+
prompt,
|
|
379
|
+
sessionId: iteration > 0 ? lastSessionId : undefined,
|
|
380
|
+
iteration: iteration + 1,
|
|
381
|
+
...(opts.inbox ? { inboxStream: opts.inbox } : {}),
|
|
382
|
+
})) {
|
|
179
383
|
// processOutput chain — deny drops the message from accumulation;
|
|
180
384
|
// abort ends the loop. Continue carries the (possibly mutated)
|
|
181
385
|
// message forward.
|
|
@@ -202,23 +406,29 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
202
406
|
lastResponseText = responseText;
|
|
203
407
|
|
|
204
408
|
let status = parseAgentStatus(responseText);
|
|
205
|
-
//
|
|
206
|
-
//
|
|
207
|
-
//
|
|
208
|
-
//
|
|
209
|
-
const rawResponse = opts.responseSchema
|
|
409
|
+
// Inline safeParse (instead of letting parseAgentResponse validate)
|
|
410
|
+
// so a schema failure surfaces via lastResponseValidationError on
|
|
411
|
+
// the next iteration — the model needs that feedback to fix its
|
|
412
|
+
// output.
|
|
413
|
+
const rawResponse: unknown = opts.responseSchema
|
|
210
414
|
? parseAgentResponse(responseText) ?? parseRawJsonResponse(responseText)
|
|
211
415
|
: null;
|
|
416
|
+
const validateResponse = (target: unknown) => {
|
|
417
|
+
const parsed = opts.responseSchema!.safeParse(target);
|
|
418
|
+
if (!parsed.success) lastResponseValidationError = parsed.error.message;
|
|
419
|
+
return parsed;
|
|
420
|
+
};
|
|
212
421
|
if (opts.responseSchema && rawResponse !== null) {
|
|
213
|
-
const parsed =
|
|
422
|
+
const parsed = validateResponse(rawResponse);
|
|
214
423
|
if (parsed.success) {
|
|
215
424
|
const successStatus = status ?? { summary: "structured response completed", completed: [], blockers: [], exit_signal: true };
|
|
216
425
|
// Successful schema-validation is progress, not a warning.
|
|
217
426
|
process.stdout.write(`${logLabel} structured response validated after ${iteration + 1}/${maxIterations} iterations\n`);
|
|
427
|
+
const result: AgentLoopResult<TResponse> = { agentId, label, sessionId: lastSessionId, lastStatus: successStatus, iterations: iteration + 1, response: parsed.data };
|
|
428
|
+
await writeSettledState(result);
|
|
218
429
|
opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: iteration + 1, durationMs: Date.now() - startedAt });
|
|
219
|
-
return
|
|
430
|
+
return result;
|
|
220
431
|
}
|
|
221
|
-
lastResponseValidationError = parsed.error.message;
|
|
222
432
|
}
|
|
223
433
|
completedIterations = iteration + 1;
|
|
224
434
|
// Per-iteration status is informational progress. The presence
|
|
@@ -243,9 +453,8 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
243
453
|
// fields (`exit_signal`, `blockers`, …) the agent emitted in a
|
|
244
454
|
// separate <status> block. Distinct from the fast path above,
|
|
245
455
|
// which validates the raw response alone.
|
|
246
|
-
const parsed =
|
|
456
|
+
const parsed = validateResponse({ ...status, ...(rawResponse as object) });
|
|
247
457
|
if (!parsed.success) {
|
|
248
|
-
lastResponseValidationError = parsed.error.message;
|
|
249
458
|
process.stderr.write(`${logLabel} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(rawResponse).slice(0, 400)}\n`);
|
|
250
459
|
status = { ...status!, exit_signal: false, blockers: [`<response> schema validation failed: ${parsed.error.message}`] };
|
|
251
460
|
opts.onIteration?.(iteration + 1, status);
|
|
@@ -255,8 +464,27 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
255
464
|
}
|
|
256
465
|
// Settled successfully — progress, not a warning.
|
|
257
466
|
process.stdout.write(`${logLabel} done after ${iteration + 1}/${maxIterations} iterations\n`);
|
|
467
|
+
const result: AgentLoopResult<TResponse> = { agentId, label, sessionId: lastSessionId, lastStatus: status, iterations: iteration + 1, response: response as TResponse };
|
|
468
|
+
await writeSettledState(result);
|
|
258
469
|
opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: iteration + 1, durationMs: Date.now() - startedAt });
|
|
259
|
-
return
|
|
470
|
+
return result;
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
// PR 7 — agent self-pause (mode: "hitl" only). The agent set needs_input in
|
|
474
|
+
// its <status> and ended its turn. Stage a pause the NEXT boundary takes —
|
|
475
|
+
// the same machinery as a human steer, triggered by the agent's own status.
|
|
476
|
+
// The boundary pause itself is UNGATED; only this TRIGGER is gated by mode,
|
|
477
|
+
// so an `auto` agent's needs_input is ignored — it just keeps iterating.
|
|
478
|
+
if (opts.pause && opts.mode === "hitl" && status?.needs_input && pendingSteerPause === null) {
|
|
479
|
+
pendingSteerPause = {
|
|
480
|
+
reason: status.question?.trim() || "the agent requested human input",
|
|
481
|
+
correlationKey: null,
|
|
482
|
+
at: Date.now(),
|
|
483
|
+
};
|
|
484
|
+
blockerStreak = null; // an explicit ask is not a stuck-loop
|
|
485
|
+
// Pause fires at the next boundary; the loop condition guarantees that
|
|
486
|
+
// boundary runs even when this was the final budgeted iteration.
|
|
487
|
+
continue;
|
|
260
488
|
}
|
|
261
489
|
|
|
262
490
|
if (!status) {
|
|
@@ -286,9 +514,15 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
286
514
|
if (opts.responseSchema)
|
|
287
515
|
throw new Error(`${logLabel} did not produce a valid <response> after ${maxIterations} iterations${lastResponseValidationError ? `: ${lastResponseValidationError}` : ""}. Response tail: ${lastResponseText.slice(-600)}`);
|
|
288
516
|
process.stderr.write(`${logLabel} exhausted ${maxIterations} iterations, proceeding with available work\n`);
|
|
517
|
+
const result: AgentLoopResult<TResponse> = { agentId, label, sessionId: lastSessionId, lastStatus, iterations: maxIterations };
|
|
518
|
+
await writeSettledState(result);
|
|
289
519
|
opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: maxIterations, durationMs: Date.now() - startedAt });
|
|
290
|
-
return
|
|
520
|
+
return result;
|
|
291
521
|
} catch (err) {
|
|
522
|
+
// A boundary pause (PauseSignal) is control flow, not a failure — let it
|
|
523
|
+
// propagate so serveStep emits the pause sentinel; do not settle the agent
|
|
524
|
+
// as failed.
|
|
525
|
+
if (isPauseSignal(err)) throw err;
|
|
292
526
|
opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "failed", iterations: completedIterations, durationMs: Date.now() - startedAt, failureReason: err instanceof Error ? err.message : String(err) });
|
|
293
527
|
throw err;
|
|
294
528
|
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* AsyncQueue — single-producer/single-consumer push-iterable.
|
|
3
|
+
*
|
|
4
|
+
* The Claude Agent SDK's streaming-input mode wants
|
|
5
|
+
* `prompt: AsyncIterable<SDKUserMessage>`. We need a primitive that:
|
|
6
|
+
* - lets the agent loop `push()` user turns from outside the iterator
|
|
7
|
+
* (initial prompt, plus async-arriving inbox messages)
|
|
8
|
+
* - lets the SDK's `for await` consume them in order, blocking until
|
|
9
|
+
* the next item arrives
|
|
10
|
+
* - terminates cleanly via `close()` so the SDK sees the iterable
|
|
11
|
+
* drain and finalises the assistant turn
|
|
12
|
+
*
|
|
13
|
+
* Why not an `EventEmitter` or RxJS — both are overkill for one shape.
|
|
14
|
+
* The implementation is ~30 lines of native Promise plumbing and stays
|
|
15
|
+
* inside the SDK package so it can be the canonical way agent-loop
|
|
16
|
+
* builds its prompt iterable.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
export class AsyncQueue<T> implements AsyncIterable<T> {
|
|
20
|
+
private readonly buffer: T[] = [];
|
|
21
|
+
private readonly resolvers: Array<(v: IteratorResult<T>) => void> = [];
|
|
22
|
+
private done = false;
|
|
23
|
+
|
|
24
|
+
/** Push one item. If a consumer is awaiting, it wakes immediately;
|
|
25
|
+
* otherwise the item is buffered until the next `next()` call. */
|
|
26
|
+
push(value: T): void {
|
|
27
|
+
if (this.done) return;
|
|
28
|
+
const resolve = this.resolvers.shift();
|
|
29
|
+
if (resolve) {
|
|
30
|
+
resolve({ value, done: false });
|
|
31
|
+
} else {
|
|
32
|
+
this.buffer.push(value);
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** Signal end-of-stream. Any pending awaiters resolve as
|
|
37
|
+
* `{ done: true }`; subsequent `push()` calls are silent no-ops. */
|
|
38
|
+
close(): void {
|
|
39
|
+
if (this.done) return;
|
|
40
|
+
this.done = true;
|
|
41
|
+
while (this.resolvers.length > 0) {
|
|
42
|
+
const resolve = this.resolvers.shift()!;
|
|
43
|
+
resolve({ value: undefined as unknown as T, done: true });
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
[Symbol.asyncIterator](): AsyncIterator<T> {
|
|
48
|
+
return {
|
|
49
|
+
next: (): Promise<IteratorResult<T>> => {
|
|
50
|
+
const buffered = this.buffer.shift();
|
|
51
|
+
if (buffered !== undefined) return Promise.resolve({ value: buffered, done: false });
|
|
52
|
+
if (this.done) return Promise.resolve({ value: undefined as unknown as T, done: true });
|
|
53
|
+
return new Promise((resolve) => { this.resolvers.push(resolve); });
|
|
54
|
+
},
|
|
55
|
+
return: (): Promise<IteratorResult<T>> => {
|
|
56
|
+
this.close();
|
|
57
|
+
return Promise.resolve({ value: undefined as unknown as T, done: true });
|
|
58
|
+
},
|
|
59
|
+
};
|
|
60
|
+
}
|
|
61
|
+
}
|
package/src/agent/protocol.ts
CHANGED
|
@@ -15,8 +15,18 @@ export const AgentMessageSchema = z.object({
|
|
|
15
15
|
timestamp: z.string(),
|
|
16
16
|
}).passthrough() as unknown as z.ZodType<AgentMessage>;
|
|
17
17
|
|
|
18
|
-
|
|
18
|
+
/** Parse the `<response>...</response>` block from agent text. Returns
|
|
19
|
+
* `null` if missing or malformed. Callers run schema validation on
|
|
20
|
+
* the returned `unknown` directly — the earlier generic overload that
|
|
21
|
+
* did the safeParse inline returned `null` on schema failure,
|
|
22
|
+
* swallowing the validation error without setting any breadcrumb on
|
|
23
|
+
* the loop's `lastResponseValidationError`. No caller used that
|
|
24
|
+
* overload; the dedicated `responseSchema` path in `agent-loop.ts`
|
|
25
|
+
* does the parse-with-error-capture itself. */
|
|
26
|
+
export function parseAgentResponse(text: string): unknown | null {
|
|
19
27
|
const match = text.match(/<response>([\s\S]*?)<\/response>/);
|
|
20
28
|
if (!match) return null;
|
|
21
|
-
try {
|
|
29
|
+
try {
|
|
30
|
+
return JSON.parse(match[1].trim());
|
|
31
|
+
} catch { return null; }
|
|
22
32
|
}
|