@agent-compose/sdk 0.2.2 → 0.2.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +145 -33
- package/dist/agent/agent-loop.d.ts +83 -5
- package/dist/agent/run-agent.d.ts +34 -9
- package/dist/client.d.ts +247 -99
- package/dist/index.d.ts +26 -11
- package/dist/index.js +1968 -746
- package/dist/processors/builtins.d.ts +35 -0
- package/dist/processors/index.d.ts +4 -0
- package/dist/processors/processor.d.ts +91 -0
- package/dist/processors/processor.test.d.ts +1 -0
- package/dist/processors/runner.d.ts +19 -0
- package/dist/request-context/index.d.ts +2 -0
- package/dist/request-context/request-context.d.ts +159 -0
- package/dist/request-context/request-context.test.d.ts +1 -0
- package/dist/runtimes/claude.d.ts +27 -50
- package/dist/runtimes/openai-desktop.js +1919 -742
- package/dist/runtimes/vercel.d.ts +34 -0
- package/dist/runtimes/vercel.js +474 -0
- package/dist/sandbox.d.ts +29 -25
- package/dist/step-invocation/__tests__/invoker.test.d.ts +1 -0
- package/dist/step-invocation/__tests__/protocol.test.d.ts +1 -0
- package/dist/step-invocation/__tests__/server.test.d.ts +1 -0
- package/dist/step-invocation/index.d.ts +25 -0
- package/dist/step-invocation/invoker.d.ts +65 -0
- package/dist/step-invocation/protocol.d.ts +44 -0
- package/dist/step-invocation/server.d.ts +63 -0
- package/dist/step-invocation/types.d.ts +72 -0
- package/dist/tools/coding.d.ts +49 -0
- package/dist/tools/coding.test.d.ts +1 -0
- package/dist/tools/index.d.ts +2 -0
- package/dist/types/events.d.ts +36 -0
- package/dist/types/execution-context.d.ts +22 -0
- package/dist/types/runtime.d.ts +32 -0
- package/dist/types/sandbox-environment.d.ts +5 -2
- package/dist/types/sandbox.d.ts +14 -12
- package/dist/types/workflow-metadata.d.ts +51 -0
- package/dist/types/workflow-plan.d.ts +19 -0
- package/dist/types/workflow.d.ts +57 -17
- package/dist/utils/bundler.d.ts +62 -3
- package/dist/workflow-steps/__tests__/observability.test.d.ts +1 -0
- package/dist/workflow-steps/index.d.ts +10 -0
- package/dist/workflow-steps/observability.d.ts +58 -0
- package/dist/workflow-steps/runner.d.ts +96 -0
- package/dist/workflow-steps/step.d.ts +25 -0
- package/dist/workflow-steps/types.d.ts +135 -0
- package/dist/workflow-steps/workflow-steps.test.d.ts +1 -0
- package/dist/workflow-steps/workflow.d.ts +50 -0
- package/dist/workflows/engine.d.ts +27 -13
- package/dist/workflows/invoke-child.d.ts +10 -0
- package/package.json +25 -15
- package/src/agent/agent-loop.ts +197 -26
- package/src/agent/run-agent.ts +40 -15
- package/src/client.ts +326 -76
- package/src/index.ts +124 -10
- package/src/processors/builtins.ts +72 -0
- package/src/processors/index.ts +15 -0
- package/src/processors/processor.ts +103 -0
- package/src/processors/runner.ts +42 -0
- package/src/request-context/index.ts +17 -0
- package/src/request-context/request-context.ts +302 -0
- package/src/runtimes/claude.ts +123 -254
- package/src/runtimes/vercel.ts +180 -0
- package/src/sandbox.ts +53 -21
- package/src/step-invocation/index.ts +33 -0
- package/src/step-invocation/invoker.ts +204 -0
- package/src/step-invocation/protocol.ts +57 -0
- package/src/step-invocation/server.ts +184 -0
- package/src/step-invocation/types.ts +70 -0
- package/src/tools/coding.ts +126 -0
- package/src/tools/index.ts +8 -0
- package/src/types/events.ts +40 -0
- package/src/types/execution-context.ts +30 -0
- package/src/types/runtime.ts +24 -0
- package/src/types/sandbox-environment.ts +7 -5
- package/src/types/sandbox.ts +16 -12
- package/src/types/workflow-metadata.ts +84 -0
- package/src/types/workflow-plan.ts +24 -0
- package/src/types/workflow.ts +139 -25
- package/src/utils/bundler.ts +213 -19
- package/src/utils/source-loader.ts +2 -2
- package/src/workflow-steps/index.ts +30 -0
- package/src/workflow-steps/observability.ts +103 -0
- package/src/workflow-steps/runner.ts +244 -0
- package/src/workflow-steps/step.ts +38 -0
- package/src/workflow-steps/types.ts +134 -0
- package/src/workflow-steps/workflow.ts +95 -0
- package/src/workflows/engine.ts +69 -40
- package/src/workflows/invoke-child.ts +29 -0
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
import type { WorkflowCtx } from "../types/workflow.js";
|
|
2
|
+
/**
|
|
3
|
+
* Build the public-API child workflow invoker used by legacy and sandboxed
|
|
4
|
+
* workflow execution. Provider-backed engines may inject a different
|
|
5
|
+
* implementation (Temporal child workflow, Inngest invoke, etc.).
|
|
6
|
+
*/
|
|
7
|
+
export declare function buildInvokeChild(runId: string, opts?: {
|
|
8
|
+
fallbackBaseUrl?: string;
|
|
9
|
+
defaultFactorySlug?: string;
|
|
10
|
+
}): WorkflowCtx["invokeChild"];
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@agent-compose/sdk",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.4",
|
|
4
4
|
"description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"repository": {
|
|
@@ -22,6 +22,12 @@
|
|
|
22
22
|
"types": "./dist/runtimes/openai-desktop.d.ts",
|
|
23
23
|
"import": "./dist/runtimes/openai-desktop.js",
|
|
24
24
|
"default": "./dist/runtimes/openai-desktop.js"
|
|
25
|
+
},
|
|
26
|
+
"./runtimes/vercel": {
|
|
27
|
+
"bun": "./src/runtimes/vercel.ts",
|
|
28
|
+
"types": "./dist/runtimes/vercel.d.ts",
|
|
29
|
+
"import": "./dist/runtimes/vercel.js",
|
|
30
|
+
"default": "./dist/runtimes/vercel.js"
|
|
25
31
|
}
|
|
26
32
|
},
|
|
27
33
|
"files": [
|
|
@@ -37,12 +43,12 @@
|
|
|
37
43
|
"node": ">=20"
|
|
38
44
|
},
|
|
39
45
|
"scripts": {
|
|
40
|
-
"typecheck":
|
|
41
|
-
"test":
|
|
42
|
-
"build":
|
|
43
|
-
"build:js":
|
|
44
|
-
"build:types":
|
|
45
|
-
"clean":
|
|
46
|
+
"typecheck": "tsc --noEmit",
|
|
47
|
+
"test": "vitest run src",
|
|
48
|
+
"build": "bun run build:js && bun run build:types",
|
|
49
|
+
"build:js": "bun build src/index.ts src/runtimes/openai-desktop.ts src/runtimes/vercel.ts --outdir dist --target node --format esm --packages external",
|
|
50
|
+
"build:types": "tsc -p tsconfig.build.json",
|
|
51
|
+
"clean": "rm -rf dist",
|
|
46
52
|
"prepublishOnly": "bun run clean && bun run build"
|
|
47
53
|
},
|
|
48
54
|
"peerDependencies": {
|
|
@@ -50,17 +56,21 @@
|
|
|
50
56
|
},
|
|
51
57
|
"devDependencies": {
|
|
52
58
|
"@types/node": "^22.19.15",
|
|
53
|
-
"typescript":
|
|
54
|
-
"vitest":
|
|
59
|
+
"typescript": "^5.8.0",
|
|
60
|
+
"vitest": "^3.0.0"
|
|
55
61
|
},
|
|
56
62
|
"dependencies": {
|
|
57
|
-
"@
|
|
63
|
+
"@anthropic-ai/claude-agent-sdk": "^0.2.129",
|
|
64
|
+
"@babel/parser": "^7.29.3",
|
|
65
|
+
"@babel/types": "^7.29.0",
|
|
66
|
+
"@e2b/desktop": "^1.1.2",
|
|
58
67
|
"@vercel/sandbox": "^1.10.0",
|
|
59
|
-
"
|
|
60
|
-
"
|
|
61
|
-
"
|
|
62
|
-
"
|
|
63
|
-
"
|
|
68
|
+
"ai": "^6.0.175",
|
|
69
|
+
"e2b": "^2.3.0",
|
|
70
|
+
"ofetch": "^1.5.1",
|
|
71
|
+
"openai": "^6.33.0",
|
|
72
|
+
"p-retry": "^6.2.0",
|
|
73
|
+
"sharp": "^0.34.5"
|
|
64
74
|
},
|
|
65
75
|
"publishConfig": {
|
|
66
76
|
"access": "public"
|
package/src/agent/agent-loop.ts
CHANGED
|
@@ -7,11 +7,16 @@ import type { RuntimeOptions, ModelExecutionContract } from "../index.js";
|
|
|
7
7
|
import { z } from "zod";
|
|
8
8
|
import { AgentStatusSchema, parseAgentResponse } from "./protocol.js";
|
|
9
9
|
import type { AgentStatus, AgentMessage } from "./protocol.js";
|
|
10
|
+
import { randomUUID } from "node:crypto";
|
|
11
|
+
import type { Processor, ProcessorContext } from "../processors/processor.js";
|
|
12
|
+
import { runProcessorChain } from "../processors/runner.js";
|
|
13
|
+
import { RequestContext } from "../request-context/request-context.js";
|
|
10
14
|
|
|
11
15
|
export const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
|
|
12
16
|
|
|
13
17
|
const SAME_BLOCKER_ITERATIONS = 3;
|
|
14
18
|
const STALL_ITERATIONS = 3;
|
|
19
|
+
const MESSAGE_PREVIEW_CHARS = 400;
|
|
15
20
|
|
|
16
21
|
export function parseAgentStatus(text: string): AgentStatus | null {
|
|
17
22
|
const match = text.match(/<status>([\s\S]*?)<\/status>/);
|
|
@@ -24,86 +29,239 @@ export function parseAgentStatus(text: string): AgentStatus | null {
|
|
|
24
29
|
|
|
25
30
|
const DEFAULT_ALLOWED_TOOLS = ["Read", "Write", "Edit", "Bash", "Glob", "Grep", "WebFetch"];
|
|
26
31
|
|
|
27
|
-
export interface AgentLoopResult {
|
|
32
|
+
export interface AgentLoopResult<TResponse = unknown> {
|
|
33
|
+
agentId: string;
|
|
34
|
+
label: string;
|
|
28
35
|
sessionId: string;
|
|
29
36
|
lastStatus: AgentStatus | null;
|
|
30
37
|
iterations: number;
|
|
31
|
-
response?:
|
|
38
|
+
response?: TResponse;
|
|
32
39
|
}
|
|
33
40
|
|
|
34
|
-
export
|
|
41
|
+
export type AgentMessageSummary =
|
|
42
|
+
| { type: "init"; sessionId: string }
|
|
43
|
+
| { type: "text"; text: string }
|
|
44
|
+
| { type: "thinking"; text: string }
|
|
45
|
+
| { type: "tool_use"; toolName: string; toolUseId: string; toolInputPreview: string }
|
|
46
|
+
| { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
|
|
47
|
+
| { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number }
|
|
48
|
+
| { type: "done"; sessionId: string }
|
|
49
|
+
| { type: "error"; text: string };
|
|
50
|
+
|
|
51
|
+
function truncate(value: string): string {
|
|
52
|
+
return value.length > MESSAGE_PREVIEW_CHARS ? `${value.slice(0, MESSAGE_PREVIEW_CHARS)}…` : value;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function preview(value: unknown): string {
|
|
56
|
+
try {
|
|
57
|
+
return truncate(JSON.stringify(value));
|
|
58
|
+
} catch {
|
|
59
|
+
return truncate(String(value));
|
|
60
|
+
}
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
|
|
64
|
+
switch (msg.type) {
|
|
65
|
+
case "init": return { type: "init", sessionId: msg.sessionId };
|
|
66
|
+
case "text": return { type: "text", text: msg.text };
|
|
67
|
+
case "thinking": return { type: "thinking", text: msg.text };
|
|
68
|
+
case "tool_use": return { type: "tool_use", toolName: msg.toolName, toolUseId: msg.toolUseId, toolInputPreview: preview(msg.toolInput) };
|
|
69
|
+
case "tool_result": return { type: "tool_result", toolUseId: msg.toolUseId, output: truncate(msg.output), isError: msg.isError };
|
|
70
|
+
case "usage": return {
|
|
71
|
+
type: "usage", inputTokens: msg.inputTokens, outputTokens: msg.outputTokens,
|
|
72
|
+
cacheReadTokens: msg.cacheReadTokens, cacheCreationTokens: msg.cacheCreationTokens,
|
|
73
|
+
durationMs: msg.durationMs, numTurns: msg.numTurns,
|
|
74
|
+
};
|
|
75
|
+
case "done": return { type: "done", sessionId: msg.sessionId };
|
|
76
|
+
case "error": return { type: "error", text: truncate(msg.text) };
|
|
77
|
+
}
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
export type AgentLifecycleEvent =
|
|
81
|
+
| { event: "agent.spawned"; at: number; agentId: string; label: string; allowedTools?: string[] }
|
|
82
|
+
| { event: "agent.message"; at: number; agentId: string; label: string; iteration: number; message: AgentMessageSummary }
|
|
83
|
+
| { event: "agent.iteration"; at: number; agentId: string; label: string; iteration: number; status: AgentStatus | null }
|
|
84
|
+
| { event: "agent.settled"; at: number; agentId: string; label: string; outcome: "success" | "failed"; iterations: number; durationMs: number; failureReason?: string };
|
|
85
|
+
|
|
86
|
+
export interface AgentLoopOpts<TResponse = unknown> {
|
|
87
|
+
agentId?: string;
|
|
35
88
|
label?: string;
|
|
89
|
+
onAgentLifecycleEvent?: (event: AgentLifecycleEvent) => void;
|
|
36
90
|
onIteration?: (iteration: number, status: AgentStatus | null) => void;
|
|
37
91
|
turnsPerIteration?: number;
|
|
38
92
|
maxIterations?: number;
|
|
39
93
|
buildPrompt: (lastStatus: AgentStatus | null, iteration: number) => string;
|
|
40
94
|
onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
|
|
41
95
|
allowedTools?: string[];
|
|
42
|
-
responseSchema?: z.ZodType<
|
|
96
|
+
responseSchema?: z.ZodType<TResponse>;
|
|
43
97
|
runtime?: (opts: RuntimeOptions) => ModelExecutionContract;
|
|
44
98
|
cwd?: string;
|
|
45
|
-
|
|
46
|
-
|
|
99
|
+
/** Processor chain — see sdk/src/processors. Runs sequentially around
|
|
100
|
+
* prompts and emitted messages. Tool-call gating is a no-op until a
|
|
101
|
+
* runtime that exposes a pre-tool-use seam is wired (candidate #1). */
|
|
102
|
+
processors?: readonly Processor[];
|
|
103
|
+
/** Per-run typed bag — propagated to every processor as `ctx.requestContext`. */
|
|
104
|
+
requestContext?: RequestContext;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TResponse>): Promise<AgentLoopResult<TResponse>> {
|
|
108
|
+
const agentId = opts.agentId ?? randomUUID();
|
|
109
|
+
const label = opts.label ?? "agent";
|
|
110
|
+
const logLabel = opts.label ?? "[Agent Loop]";
|
|
111
|
+
const startedAt = Date.now();
|
|
47
112
|
const turnsPerIteration = opts.turnsPerIteration ?? 40;
|
|
48
113
|
const maxIterations = opts.maxIterations ?? 8;
|
|
114
|
+
const processors = opts.processors ?? [];
|
|
115
|
+
const requestContext = opts.requestContext ?? RequestContext.fromReserved({
|
|
116
|
+
teamId: "", runId: "", workflowId: "",
|
|
117
|
+
factoryId: null, apiKeyScopes: [], parentRunId: null,
|
|
118
|
+
});
|
|
119
|
+
const loopAbort = new AbortController();
|
|
120
|
+
|
|
121
|
+
const buildProcCtx = (iteration: number): ProcessorContext => ({
|
|
122
|
+
requestContext,
|
|
123
|
+
abortSignal: loopAbort.signal,
|
|
124
|
+
retryCount: 0,
|
|
125
|
+
agentId,
|
|
126
|
+
iteration,
|
|
127
|
+
});
|
|
49
128
|
|
|
50
129
|
if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
|
|
51
130
|
const client = opts.runtime({
|
|
52
131
|
maxTurns: turnsPerIteration,
|
|
53
132
|
allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
|
|
54
|
-
label,
|
|
133
|
+
label: logLabel,
|
|
55
134
|
cwd: opts.cwd,
|
|
135
|
+
processors,
|
|
136
|
+
requestContext,
|
|
137
|
+
agentId,
|
|
138
|
+
...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: z.toJSONSchema(opts.responseSchema) as Record<string, unknown> } } : {}),
|
|
56
139
|
});
|
|
57
140
|
|
|
141
|
+
// Heads-up when a caller registers tool-call gating on a runtime that
|
|
142
|
+
// can't honour it. Better to surface this once at start than have users
|
|
143
|
+
// wonder why their `denyTools(...)` never fires.
|
|
144
|
+
if (processors.some((p) => p.processToolCall) && !client.supportsToolCallProcessor) {
|
|
145
|
+
process.stderr.write(
|
|
146
|
+
`${logLabel} processToolCall registered but runtime does not support pre-tool-use gating — hook is dormant\n`,
|
|
147
|
+
);
|
|
148
|
+
}
|
|
149
|
+
|
|
58
150
|
let lastSessionId = "";
|
|
59
151
|
let lastStatus: AgentStatus | null = null;
|
|
60
152
|
let iterationsWithoutStatus = 0;
|
|
61
153
|
let blockerStreak: { key: string; count: number } | null = null;
|
|
154
|
+
let completedIterations = 0;
|
|
155
|
+
let lastResponseText = "";
|
|
156
|
+
let lastResponseValidationError = "";
|
|
157
|
+
|
|
158
|
+
opts.onAgentLifecycleEvent?.({ event: "agent.spawned", at: startedAt, agentId, label, allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS });
|
|
62
159
|
|
|
160
|
+
try {
|
|
63
161
|
for (let iteration = 0; iteration < maxIterations; iteration++) {
|
|
64
|
-
const
|
|
65
|
-
|
|
162
|
+
const procCtx = buildProcCtx(iteration + 1);
|
|
163
|
+
const initialPrompt = opts.buildPrompt(lastStatus, iteration);
|
|
164
|
+
|
|
165
|
+
// processInput chain — deny ends the loop; abort ends the loop.
|
|
166
|
+
const inputVerdict = await runProcessorChain(processors, (p) => p.processInput, initialPrompt, procCtx);
|
|
167
|
+
if (inputVerdict.kind !== "continue") {
|
|
168
|
+
loopAbort.abort();
|
|
169
|
+
throw new Error(`${logLabel} input ${inputVerdict.kind === "deny" ? "rejected" : "aborted"}: ${inputVerdict.reason}`);
|
|
170
|
+
}
|
|
171
|
+
const prompt = inputVerdict.value;
|
|
172
|
+
|
|
173
|
+
// Progress, not an error — write to stdout so dashboards and
|
|
174
|
+
// log viewers don't visually flag it as a warning.
|
|
175
|
+
process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
|
|
66
176
|
|
|
67
177
|
let responseText = "";
|
|
68
|
-
for await (const
|
|
178
|
+
for await (const rawMsg of client.sendMessage({ prompt, sessionId: iteration > 0 ? lastSessionId : undefined, iteration: iteration + 1 })) {
|
|
179
|
+
// processOutput chain — deny drops the message from accumulation;
|
|
180
|
+
// abort ends the loop. Continue carries the (possibly mutated)
|
|
181
|
+
// message forward.
|
|
182
|
+
const outputVerdict = await runProcessorChain(processors, (p) => p.processOutput, rawMsg, procCtx);
|
|
183
|
+
if (outputVerdict.kind === "abort") {
|
|
184
|
+
loopAbort.abort();
|
|
185
|
+
throw new Error(`${logLabel} output aborted: ${outputVerdict.reason}`);
|
|
186
|
+
}
|
|
187
|
+
if (outputVerdict.kind === "deny") {
|
|
188
|
+
process.stderr.write(`${logLabel} message dropped by processor: ${outputVerdict.reason}\n`);
|
|
189
|
+
continue;
|
|
190
|
+
}
|
|
191
|
+
const msg = outputVerdict.value;
|
|
69
192
|
opts.onAgentEvent?.(iteration, msg);
|
|
193
|
+
opts.onAgentLifecycleEvent?.({ event: "agent.message", at: Date.now(), agentId, label, iteration: iteration + 1, message: summarizeAgentMessage(msg) });
|
|
70
194
|
if (msg.type === "init") lastSessionId = msg.sessionId;
|
|
71
195
|
if (msg.type === "text") responseText += msg.text;
|
|
72
196
|
if (msg.type === "error") throw new Error(`Agent error: ${msg.text}`);
|
|
73
197
|
}
|
|
74
|
-
|
|
198
|
+
// (No trailing newline here — the dashboard renders each stdout
|
|
199
|
+
// write as its own log row, so a bare "\n" produces an empty
|
|
200
|
+
// STDOUT row between agent iterations. The visual gap was meant
|
|
201
|
+
// for interactive TTYs and is just noise in structured logs.)
|
|
202
|
+
lastResponseText = responseText;
|
|
75
203
|
|
|
76
204
|
let status = parseAgentStatus(responseText);
|
|
77
|
-
|
|
205
|
+
// Parse the response once per iteration; both the schema fast path
|
|
206
|
+
// below and the exit-signal-driven validation a few lines down
|
|
207
|
+
// need the same parsed payload. `null` when there's no <response>
|
|
208
|
+
// block AND no top-level JSON object.
|
|
209
|
+
const rawResponse = opts.responseSchema
|
|
210
|
+
? parseAgentResponse(responseText) ?? parseRawJsonResponse(responseText)
|
|
211
|
+
: null;
|
|
212
|
+
if (opts.responseSchema && rawResponse !== null) {
|
|
213
|
+
const parsed = opts.responseSchema.safeParse(rawResponse);
|
|
214
|
+
if (parsed.success) {
|
|
215
|
+
const successStatus = status ?? { summary: "structured response completed", completed: [], blockers: [], exit_signal: true };
|
|
216
|
+
// Successful schema-validation is progress, not a warning.
|
|
217
|
+
process.stdout.write(`${logLabel} structured response validated after ${iteration + 1}/${maxIterations} iterations\n`);
|
|
218
|
+
opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: iteration + 1, durationMs: Date.now() - startedAt });
|
|
219
|
+
return { agentId, label, sessionId: lastSessionId, lastStatus: successStatus, iterations: iteration + 1, response: parsed.data };
|
|
220
|
+
}
|
|
221
|
+
lastResponseValidationError = parsed.error.message;
|
|
222
|
+
}
|
|
223
|
+
completedIterations = iteration + 1;
|
|
224
|
+
// Per-iteration status is informational progress. The presence
|
|
225
|
+
// of blockers in the JSON is just data — the operator decides
|
|
226
|
+
// whether to be alarmed by it. Writing to stderr made the line
|
|
227
|
+
// visually flag as a warning even on clean runs (blockers=[]).
|
|
228
|
+
process.stdout.write(`${logLabel} iteration ${iteration + 1} status: exit_signal=${status?.exit_signal ?? "(no status)"} blockers=${JSON.stringify(status?.blockers ?? [])}\n`);
|
|
78
229
|
lastStatus = status ?? lastStatus;
|
|
79
230
|
opts.onIteration?.(iteration + 1, status);
|
|
231
|
+
opts.onAgentLifecycleEvent?.({ event: "agent.iteration", at: Date.now(), agentId, label, iteration: iteration + 1, status });
|
|
80
232
|
|
|
81
233
|
if (status?.exit_signal && (status.blockers?.length ?? 0) === 0) {
|
|
82
|
-
let response:
|
|
234
|
+
let response: AgentStatus | TResponse = status;
|
|
83
235
|
if (opts.responseSchema) {
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
process.stderr.write(`\n${label} NO <response> BLOCK — response tail: ${responseText.slice(-400)}\n`);
|
|
236
|
+
if (rawResponse === null) {
|
|
237
|
+
process.stderr.write(`${logLabel} NO <response> BLOCK — response tail: ${responseText.slice(-400)}\n`);
|
|
87
238
|
status = { ...status!, exit_signal: false, blockers: ["No <response> block found — emit a <response> block with the required JSON fields before setting exit_signal: true"] };
|
|
88
239
|
opts.onIteration?.(iteration + 1, status);
|
|
89
240
|
continue;
|
|
90
241
|
}
|
|
91
|
-
|
|
242
|
+
// Status-merged validation — the schema may reference status
|
|
243
|
+
// fields (`exit_signal`, `blockers`, …) the agent emitted in a
|
|
244
|
+
// separate <status> block. Distinct from the fast path above,
|
|
245
|
+
// which validates the raw response alone.
|
|
246
|
+
const parsed = opts.responseSchema.safeParse({ ...status, ...(rawResponse as object) });
|
|
92
247
|
if (!parsed.success) {
|
|
93
|
-
|
|
248
|
+
lastResponseValidationError = parsed.error.message;
|
|
249
|
+
process.stderr.write(`${logLabel} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(rawResponse).slice(0, 400)}\n`);
|
|
94
250
|
status = { ...status!, exit_signal: false, blockers: [`<response> schema validation failed: ${parsed.error.message}`] };
|
|
95
251
|
opts.onIteration?.(iteration + 1, status);
|
|
96
252
|
continue;
|
|
97
253
|
}
|
|
98
254
|
response = parsed.data;
|
|
99
255
|
}
|
|
100
|
-
|
|
101
|
-
|
|
256
|
+
// Settled successfully — progress, not a warning.
|
|
257
|
+
process.stdout.write(`${logLabel} done after ${iteration + 1}/${maxIterations} iterations\n`);
|
|
258
|
+
opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: iteration + 1, durationMs: Date.now() - startedAt });
|
|
259
|
+
return { agentId, label, sessionId: lastSessionId, lastStatus: status, iterations: iteration + 1, response: response as TResponse };
|
|
102
260
|
}
|
|
103
261
|
|
|
104
262
|
if (!status) {
|
|
105
263
|
if (++iterationsWithoutStatus >= STALL_ITERATIONS)
|
|
106
|
-
throw new Error(`${
|
|
264
|
+
throw new Error(`${logLabel} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
|
|
107
265
|
} else {
|
|
108
266
|
iterationsWithoutStatus = 0;
|
|
109
267
|
}
|
|
@@ -112,7 +270,7 @@ export async function agentLoop(opts: {
|
|
|
112
270
|
const key = status.blockers.join("|");
|
|
113
271
|
if (blockerStreak !== null && blockerStreak.key === key) {
|
|
114
272
|
if (++blockerStreak.count >= SAME_BLOCKER_ITERATIONS)
|
|
115
|
-
throw new Error(`${
|
|
273
|
+
throw new Error(`${logLabel} circuit break: same blocker repeated ${blockerStreak.count}x — "${status.blockers[0]}"`);
|
|
116
274
|
} else {
|
|
117
275
|
blockerStreak = { key, count: 1 };
|
|
118
276
|
}
|
|
@@ -121,11 +279,24 @@ export async function agentLoop(opts: {
|
|
|
121
279
|
}
|
|
122
280
|
|
|
123
281
|
if (iteration + 1 < maxIterations)
|
|
124
|
-
|
|
282
|
+
// Loop continuation — progress.
|
|
283
|
+
process.stdout.write(`${logLabel} continuing to iteration ${iteration + 2}/${maxIterations}\n`);
|
|
125
284
|
}
|
|
126
285
|
|
|
127
286
|
if (opts.responseSchema)
|
|
128
|
-
throw new Error(`${
|
|
129
|
-
process.stderr.write(`${
|
|
130
|
-
|
|
287
|
+
throw new Error(`${logLabel} did not produce a valid <response> after ${maxIterations} iterations${lastResponseValidationError ? `: ${lastResponseValidationError}` : ""}. Response tail: ${lastResponseText.slice(-600)}`);
|
|
288
|
+
process.stderr.write(`${logLabel} exhausted ${maxIterations} iterations, proceeding with available work\n`);
|
|
289
|
+
opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: maxIterations, durationMs: Date.now() - startedAt });
|
|
290
|
+
return { agentId, label, sessionId: lastSessionId, lastStatus, iterations: maxIterations };
|
|
291
|
+
} catch (err) {
|
|
292
|
+
opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "failed", iterations: completedIterations, durationMs: Date.now() - startedAt, failureReason: err instanceof Error ? err.message : String(err) });
|
|
293
|
+
throw err;
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
function parseRawJsonResponse(text: string): unknown {
|
|
298
|
+
const trimmed = text.trim();
|
|
299
|
+
if (!trimmed.startsWith("{") || !trimmed.endsWith("}")) return null;
|
|
300
|
+
try { return JSON.parse(trimmed); }
|
|
301
|
+
catch { return null; }
|
|
131
302
|
}
|
package/src/agent/run-agent.ts
CHANGED
|
@@ -1,10 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* agent — canonical entry point for embedding an LLM agent inside a
|
|
3
3
|
* workflow. The workflow's `run()` body calls it; the loop executes
|
|
4
4
|
* against the runner's own VM.
|
|
5
5
|
*
|
|
6
6
|
* Glue packaged so workflows don't duplicate it:
|
|
7
|
-
* - Inject `{{VAR}}` placeholders into the prompt template.
|
|
8
7
|
* - Strip the `--- frontmatter ---` header authors use for IDE hints.
|
|
9
8
|
* - Append PROTOCOL_SUFFIX (status/response format instructions).
|
|
10
9
|
* - Append a response-format appendix when `responseSchema` is set.
|
|
@@ -13,11 +12,13 @@
|
|
|
13
12
|
|
|
14
13
|
import { z } from "zod";
|
|
15
14
|
import { agentLoop } from "./agent-loop.js";
|
|
16
|
-
import type { AgentLoopResult } from "./agent-loop.js";
|
|
15
|
+
import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
|
|
17
16
|
import type { AgentMessage, AgentStatus } from "../types/protocol.js";
|
|
18
17
|
import type { AgentRuntime, RuntimeOptions } from "../types/runtime.js";
|
|
19
18
|
import type { SandboxProvider } from "../types/sandbox.js";
|
|
20
19
|
import type { AgentBudget } from "../types/workflow.js";
|
|
20
|
+
import type { Processor } from "../processors/processor.js";
|
|
21
|
+
import { RequestContext } from "../request-context/request-context.js";
|
|
21
22
|
import PROTOCOL_SUFFIX_RAW from "./protocol-suffix.md" with { type: "text" };
|
|
22
23
|
|
|
23
24
|
const PROTOCOL_SUFFIX = stripFrontmatter(PROTOCOL_SUFFIX_RAW);
|
|
@@ -28,10 +29,6 @@ function stripFrontmatter(content: string): string {
|
|
|
28
29
|
return end === -1 ? content : content.slice(end + 4).trimStart();
|
|
29
30
|
}
|
|
30
31
|
|
|
31
|
-
function inject(template: string, vars: Record<string, string>): string {
|
|
32
|
-
return Object.entries(vars).reduce((t, [k, v]) => t.replaceAll(`{{${k}}}`, v), template);
|
|
33
|
-
}
|
|
34
|
-
|
|
35
32
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
36
33
|
function zodTypeName(schema: any): string {
|
|
37
34
|
if (schema instanceof z.ZodString) return "string";
|
|
@@ -74,7 +71,9 @@ function buildResponseFormatAppendix(schema: z.ZodType<any>): string {
|
|
|
74
71
|
return [...header, ...body].join("\n");
|
|
75
72
|
}
|
|
76
73
|
|
|
77
|
-
export interface
|
|
74
|
+
export interface AgentOpts<T = unknown> {
|
|
75
|
+
/** Stable system identity for this agent loop. Generated when omitted. */
|
|
76
|
+
agentId?: string;
|
|
78
77
|
/** Sandbox the runtime executes commands against. Inside a workflow,
|
|
79
78
|
* always pass `ctx.sandbox` — a pre-constructed local provider for
|
|
80
79
|
* the runner's own VM. Exposed as a parameter so tests and non-workflow
|
|
@@ -82,12 +81,9 @@ export interface RunAgentOpts<T = unknown> {
|
|
|
82
81
|
sandbox: SandboxProvider;
|
|
83
82
|
/** Runtime definition from `createClaudeRuntime({...})` (or custom). */
|
|
84
83
|
runtime: AgentRuntime;
|
|
85
|
-
/** Prompt
|
|
84
|
+
/** Prompt text. Authors can include YAML-style `--- frontmatter ---`
|
|
86
85
|
* at the top for IDE hints; it's stripped before the model sees it. */
|
|
87
86
|
prompt: string;
|
|
88
|
-
/** Substitution map for `{{VAR}}` placeholders in the prompt. `WORKING_DIR`
|
|
89
|
-
* and `DIFF_BASE` auto-populate from `opts.workingDir` unless overridden. */
|
|
90
|
-
promptVars?: Record<string, string>;
|
|
91
87
|
/** `cwd` forwarded to the runtime — every shell command runs here. */
|
|
92
88
|
workingDir?: string;
|
|
93
89
|
/** Tools the model may use. Defaults to a safe kitchen-sink set inside
|
|
@@ -101,12 +97,35 @@ export interface RunAgentOpts<T = unknown> {
|
|
|
101
97
|
responseSchema?: z.ZodType<T>;
|
|
102
98
|
/** Label prefix for runtime stderr ("[sbid][agent]" by default). */
|
|
103
99
|
label?: string;
|
|
100
|
+
/** Lifecycle event sink from workflow ctx. Emits agent.spawned / iteration / settled. */
|
|
101
|
+
events?: { emit: (event: AgentLifecycleEvent) => void | Promise<void> };
|
|
104
102
|
/** Per-message event callback — wire this to your workflow's event
|
|
105
103
|
* telemetry if you want per-tool-call observability. */
|
|
106
104
|
onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
|
|
107
105
|
/** Per-iteration status callback — fires after each model turn with the
|
|
108
106
|
* parsed `<status>` block (or null if the model didn't emit one). */
|
|
109
107
|
onIteration?: (iteration: number, status: AgentStatus | null) => void;
|
|
108
|
+
/**
|
|
109
|
+
* Processors run as a typed pre/post pipeline around the agent loop:
|
|
110
|
+
* - processInput before each iteration's prompt is sent
|
|
111
|
+
* - processOutput on each emitted message
|
|
112
|
+
* - processToolCall before each tool call (runtime-driven; dormant
|
|
113
|
+
* until a runtime that supports pre-tool gating is wired in)
|
|
114
|
+
*
|
|
115
|
+
* Common pattern with workflow-level defaults:
|
|
116
|
+
* agent({ processors: [...ctx.processors, mySpecific], ... })
|
|
117
|
+
*/
|
|
118
|
+
processors?: readonly Processor[];
|
|
119
|
+
/**
|
|
120
|
+
* Per-run request context. Passed through to processors as
|
|
121
|
+
* `ctx.requestContext`; carries tenant identity (teamId, factoryId, scopes)
|
|
122
|
+
* and the freeform user namespace.
|
|
123
|
+
*
|
|
124
|
+
* Inside a workflow, pass `ctx.requestContext`. Tests/non-workflow callers
|
|
125
|
+
* can omit; a degenerate context is synthesised so processors that don't
|
|
126
|
+
* read identity (e.g. redactPattern) still work.
|
|
127
|
+
*/
|
|
128
|
+
requestContext?: RequestContext;
|
|
110
129
|
}
|
|
111
130
|
|
|
112
131
|
/**
|
|
@@ -114,12 +133,12 @@ export interface RunAgentOpts<T = unknown> {
|
|
|
114
133
|
* `AgentLoopResult`, including `response` when a `responseSchema` was
|
|
115
134
|
* supplied and the model validated against it.
|
|
116
135
|
*/
|
|
117
|
-
export async function
|
|
136
|
+
export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopResult<T>> {
|
|
118
137
|
const workingDir = opts.workingDir ?? "";
|
|
119
|
-
const promptVars = { WORKING_DIR: workingDir, DIFF_BASE: "HEAD~1", ...opts.promptVars };
|
|
120
138
|
|
|
121
139
|
return agentLoop({
|
|
122
140
|
runtime: (runtimeOpts: RuntimeOptions) => opts.runtime.create(opts.sandbox, runtimeOpts),
|
|
141
|
+
...(opts.agentId !== undefined ? { agentId: opts.agentId } : {}),
|
|
123
142
|
...(opts.label !== undefined ? { label: opts.label } : {}),
|
|
124
143
|
...(opts.budget?.turnsPerIteration !== undefined ? { turnsPerIteration: opts.budget.turnsPerIteration } : {}),
|
|
125
144
|
...(opts.budget?.maxIterations !== undefined ? { maxIterations: opts.budget.maxIterations } : {}),
|
|
@@ -130,12 +149,18 @@ export async function runAgent<T = unknown>(opts: RunAgentOpts<T>): Promise<Agen
|
|
|
130
149
|
// Mid-loop: session resumes via --resume, model has history; only the
|
|
131
150
|
// protocol suffix is needed so tool-use + status block grammar stays fresh.
|
|
132
151
|
if (iteration > 0) return PROTOCOL_SUFFIX;
|
|
133
|
-
const base =
|
|
152
|
+
const base = stripFrontmatter(opts.prompt);
|
|
134
153
|
return opts.responseSchema
|
|
135
154
|
? `${base}\n\n${PROTOCOL_SUFFIX}\n\n${buildResponseFormatAppendix(opts.responseSchema)}`
|
|
136
155
|
: `${base}\n\n${PROTOCOL_SUFFIX}`;
|
|
137
156
|
},
|
|
157
|
+
...(opts.events ? { onAgentLifecycleEvent: (event: AgentLifecycleEvent) => { void opts.events?.emit(event); } } : {}),
|
|
138
158
|
...(opts.onAgentEvent ? { onAgentEvent: opts.onAgentEvent } : {}),
|
|
139
159
|
...(opts.onIteration ? { onIteration: opts.onIteration } : {}),
|
|
160
|
+
...(opts.processors?.length ? { processors: opts.processors } : {}),
|
|
161
|
+
requestContext: opts.requestContext ?? RequestContext.fromReserved({
|
|
162
|
+
teamId: "", runId: "", workflowId: "",
|
|
163
|
+
factoryId: null, apiKeyScopes: [], parentRunId: null,
|
|
164
|
+
}),
|
|
140
165
|
});
|
|
141
166
|
}
|