@agent-compose/sdk 0.2.2 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/README.md +145 -33
  2. package/dist/agent/agent-loop.d.ts +83 -5
  3. package/dist/agent/run-agent.d.ts +34 -9
  4. package/dist/client.d.ts +247 -99
  5. package/dist/index.d.ts +26 -11
  6. package/dist/index.js +1968 -746
  7. package/dist/processors/builtins.d.ts +35 -0
  8. package/dist/processors/index.d.ts +4 -0
  9. package/dist/processors/processor.d.ts +91 -0
  10. package/dist/processors/processor.test.d.ts +1 -0
  11. package/dist/processors/runner.d.ts +19 -0
  12. package/dist/request-context/index.d.ts +2 -0
  13. package/dist/request-context/request-context.d.ts +159 -0
  14. package/dist/request-context/request-context.test.d.ts +1 -0
  15. package/dist/runtimes/claude.d.ts +27 -50
  16. package/dist/runtimes/openai-desktop.js +1919 -742
  17. package/dist/runtimes/vercel.d.ts +34 -0
  18. package/dist/runtimes/vercel.js +474 -0
  19. package/dist/sandbox.d.ts +29 -25
  20. package/dist/step-invocation/__tests__/invoker.test.d.ts +1 -0
  21. package/dist/step-invocation/__tests__/protocol.test.d.ts +1 -0
  22. package/dist/step-invocation/__tests__/server.test.d.ts +1 -0
  23. package/dist/step-invocation/index.d.ts +25 -0
  24. package/dist/step-invocation/invoker.d.ts +65 -0
  25. package/dist/step-invocation/protocol.d.ts +44 -0
  26. package/dist/step-invocation/server.d.ts +63 -0
  27. package/dist/step-invocation/types.d.ts +72 -0
  28. package/dist/tools/coding.d.ts +49 -0
  29. package/dist/tools/coding.test.d.ts +1 -0
  30. package/dist/tools/index.d.ts +2 -0
  31. package/dist/types/events.d.ts +36 -0
  32. package/dist/types/execution-context.d.ts +22 -0
  33. package/dist/types/runtime.d.ts +32 -0
  34. package/dist/types/sandbox-environment.d.ts +5 -2
  35. package/dist/types/sandbox.d.ts +14 -12
  36. package/dist/types/workflow-metadata.d.ts +51 -0
  37. package/dist/types/workflow-plan.d.ts +19 -0
  38. package/dist/types/workflow.d.ts +57 -17
  39. package/dist/utils/bundler.d.ts +62 -3
  40. package/dist/workflow-steps/__tests__/observability.test.d.ts +1 -0
  41. package/dist/workflow-steps/index.d.ts +10 -0
  42. package/dist/workflow-steps/observability.d.ts +58 -0
  43. package/dist/workflow-steps/runner.d.ts +96 -0
  44. package/dist/workflow-steps/step.d.ts +25 -0
  45. package/dist/workflow-steps/types.d.ts +135 -0
  46. package/dist/workflow-steps/workflow-steps.test.d.ts +1 -0
  47. package/dist/workflow-steps/workflow.d.ts +50 -0
  48. package/dist/workflows/engine.d.ts +27 -13
  49. package/dist/workflows/invoke-child.d.ts +10 -0
  50. package/package.json +25 -15
  51. package/src/agent/agent-loop.ts +197 -26
  52. package/src/agent/run-agent.ts +40 -15
  53. package/src/client.ts +326 -76
  54. package/src/index.ts +124 -10
  55. package/src/processors/builtins.ts +72 -0
  56. package/src/processors/index.ts +15 -0
  57. package/src/processors/processor.ts +103 -0
  58. package/src/processors/runner.ts +42 -0
  59. package/src/request-context/index.ts +17 -0
  60. package/src/request-context/request-context.ts +302 -0
  61. package/src/runtimes/claude.ts +123 -254
  62. package/src/runtimes/vercel.ts +180 -0
  63. package/src/sandbox.ts +53 -21
  64. package/src/step-invocation/index.ts +33 -0
  65. package/src/step-invocation/invoker.ts +204 -0
  66. package/src/step-invocation/protocol.ts +57 -0
  67. package/src/step-invocation/server.ts +184 -0
  68. package/src/step-invocation/types.ts +70 -0
  69. package/src/tools/coding.ts +126 -0
  70. package/src/tools/index.ts +8 -0
  71. package/src/types/events.ts +40 -0
  72. package/src/types/execution-context.ts +30 -0
  73. package/src/types/runtime.ts +24 -0
  74. package/src/types/sandbox-environment.ts +7 -5
  75. package/src/types/sandbox.ts +16 -12
  76. package/src/types/workflow-metadata.ts +84 -0
  77. package/src/types/workflow-plan.ts +24 -0
  78. package/src/types/workflow.ts +139 -25
  79. package/src/utils/bundler.ts +213 -19
  80. package/src/utils/source-loader.ts +2 -2
  81. package/src/workflow-steps/index.ts +30 -0
  82. package/src/workflow-steps/observability.ts +103 -0
  83. package/src/workflow-steps/runner.ts +244 -0
  84. package/src/workflow-steps/step.ts +38 -0
  85. package/src/workflow-steps/types.ts +134 -0
  86. package/src/workflow-steps/workflow.ts +95 -0
  87. package/src/workflows/engine.ts +69 -40
  88. package/src/workflows/invoke-child.ts +29 -0
@@ -0,0 +1,10 @@
1
+ import type { WorkflowCtx } from "../types/workflow.js";
2
+ /**
3
+ * Build the public-API child workflow invoker used by legacy and sandboxed
4
+ * workflow execution. Provider-backed engines may inject a different
5
+ * implementation (Temporal child workflow, Inngest invoke, etc.).
6
+ */
7
+ export declare function buildInvokeChild(runId: string, opts?: {
8
+ fallbackBaseUrl?: string;
9
+ defaultFactorySlug?: string;
10
+ }): WorkflowCtx["invokeChild"];
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.2.2",
3
+ "version": "0.2.4",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -22,6 +22,12 @@
22
22
  "types": "./dist/runtimes/openai-desktop.d.ts",
23
23
  "import": "./dist/runtimes/openai-desktop.js",
24
24
  "default": "./dist/runtimes/openai-desktop.js"
25
+ },
26
+ "./runtimes/vercel": {
27
+ "bun": "./src/runtimes/vercel.ts",
28
+ "types": "./dist/runtimes/vercel.d.ts",
29
+ "import": "./dist/runtimes/vercel.js",
30
+ "default": "./dist/runtimes/vercel.js"
25
31
  }
26
32
  },
27
33
  "files": [
@@ -37,12 +43,12 @@
37
43
  "node": ">=20"
38
44
  },
39
45
  "scripts": {
40
- "typecheck": "tsc --noEmit",
41
- "test": "vitest run src/__tests__",
42
- "build": "bun run build:js && bun run build:types",
43
- "build:js": "bun build src/index.ts src/runtimes/openai-desktop.ts --outdir dist --target node --format esm --packages external",
44
- "build:types": "tsc -p tsconfig.build.json",
45
- "clean": "rm -rf dist",
46
+ "typecheck": "tsc --noEmit",
47
+ "test": "vitest run src",
48
+ "build": "bun run build:js && bun run build:types",
49
+ "build:js": "bun build src/index.ts src/runtimes/openai-desktop.ts src/runtimes/vercel.ts --outdir dist --target node --format esm --packages external",
50
+ "build:types": "tsc -p tsconfig.build.json",
51
+ "clean": "rm -rf dist",
46
52
  "prepublishOnly": "bun run clean && bun run build"
47
53
  },
48
54
  "peerDependencies": {
@@ -50,17 +56,21 @@
50
56
  },
51
57
  "devDependencies": {
52
58
  "@types/node": "^22.19.15",
53
- "typescript": "^5.8.0",
54
- "vitest": "^3.0.0"
59
+ "typescript": "^5.8.0",
60
+ "vitest": "^3.0.0"
55
61
  },
56
62
  "dependencies": {
57
- "@e2b/desktop": "^1.1.2",
63
+ "@anthropic-ai/claude-agent-sdk": "^0.2.129",
64
+ "@babel/parser": "^7.29.3",
65
+ "@babel/types": "^7.29.0",
66
+ "@e2b/desktop": "^1.1.2",
58
67
  "@vercel/sandbox": "^1.10.0",
59
- "e2b": "^2.3.0",
60
- "ofetch": "^1.5.1",
61
- "openai": "^6.33.0",
62
- "p-retry": "^6.2.0",
63
- "sharp": "^0.34.5"
68
+ "ai": "^6.0.175",
69
+ "e2b": "^2.3.0",
70
+ "ofetch": "^1.5.1",
71
+ "openai": "^6.33.0",
72
+ "p-retry": "^6.2.0",
73
+ "sharp": "^0.34.5"
64
74
  },
65
75
  "publishConfig": {
66
76
  "access": "public"
@@ -7,11 +7,16 @@ import type { RuntimeOptions, ModelExecutionContract } from "../index.js";
7
7
  import { z } from "zod";
8
8
  import { AgentStatusSchema, parseAgentResponse } from "./protocol.js";
9
9
  import type { AgentStatus, AgentMessage } from "./protocol.js";
10
+ import { randomUUID } from "node:crypto";
11
+ import type { Processor, ProcessorContext } from "../processors/processor.js";
12
+ import { runProcessorChain } from "../processors/runner.js";
13
+ import { RequestContext } from "../request-context/request-context.js";
10
14
 
11
15
  export const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
12
16
 
13
17
  const SAME_BLOCKER_ITERATIONS = 3;
14
18
  const STALL_ITERATIONS = 3;
19
+ const MESSAGE_PREVIEW_CHARS = 400;
15
20
 
16
21
  export function parseAgentStatus(text: string): AgentStatus | null {
17
22
  const match = text.match(/<status>([\s\S]*?)<\/status>/);
@@ -24,86 +29,239 @@ export function parseAgentStatus(text: string): AgentStatus | null {
24
29
 
25
30
  const DEFAULT_ALLOWED_TOOLS = ["Read", "Write", "Edit", "Bash", "Glob", "Grep", "WebFetch"];
26
31
 
27
- export interface AgentLoopResult {
32
+ export interface AgentLoopResult<TResponse = unknown> {
33
+ agentId: string;
34
+ label: string;
28
35
  sessionId: string;
29
36
  lastStatus: AgentStatus | null;
30
37
  iterations: number;
31
- response?: unknown;
38
+ response?: TResponse;
32
39
  }
33
40
 
34
- export async function agentLoop(opts: {
41
+ export type AgentMessageSummary =
42
+ | { type: "init"; sessionId: string }
43
+ | { type: "text"; text: string }
44
+ | { type: "thinking"; text: string }
45
+ | { type: "tool_use"; toolName: string; toolUseId: string; toolInputPreview: string }
46
+ | { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
47
+ | { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number }
48
+ | { type: "done"; sessionId: string }
49
+ | { type: "error"; text: string };
50
+
51
+ function truncate(value: string): string {
52
+ return value.length > MESSAGE_PREVIEW_CHARS ? `${value.slice(0, MESSAGE_PREVIEW_CHARS)}…` : value;
53
+ }
54
+
55
+ function preview(value: unknown): string {
56
+ try {
57
+ return truncate(JSON.stringify(value));
58
+ } catch {
59
+ return truncate(String(value));
60
+ }
61
+ }
62
+
63
+ export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
64
+ switch (msg.type) {
65
+ case "init": return { type: "init", sessionId: msg.sessionId };
66
+ case "text": return { type: "text", text: msg.text };
67
+ case "thinking": return { type: "thinking", text: msg.text };
68
+ case "tool_use": return { type: "tool_use", toolName: msg.toolName, toolUseId: msg.toolUseId, toolInputPreview: preview(msg.toolInput) };
69
+ case "tool_result": return { type: "tool_result", toolUseId: msg.toolUseId, output: truncate(msg.output), isError: msg.isError };
70
+ case "usage": return {
71
+ type: "usage", inputTokens: msg.inputTokens, outputTokens: msg.outputTokens,
72
+ cacheReadTokens: msg.cacheReadTokens, cacheCreationTokens: msg.cacheCreationTokens,
73
+ durationMs: msg.durationMs, numTurns: msg.numTurns,
74
+ };
75
+ case "done": return { type: "done", sessionId: msg.sessionId };
76
+ case "error": return { type: "error", text: truncate(msg.text) };
77
+ }
78
+ }
79
+
80
+ export type AgentLifecycleEvent =
81
+ | { event: "agent.spawned"; at: number; agentId: string; label: string; allowedTools?: string[] }
82
+ | { event: "agent.message"; at: number; agentId: string; label: string; iteration: number; message: AgentMessageSummary }
83
+ | { event: "agent.iteration"; at: number; agentId: string; label: string; iteration: number; status: AgentStatus | null }
84
+ | { event: "agent.settled"; at: number; agentId: string; label: string; outcome: "success" | "failed"; iterations: number; durationMs: number; failureReason?: string };
85
+
86
+ export interface AgentLoopOpts<TResponse = unknown> {
87
+ agentId?: string;
35
88
  label?: string;
89
+ onAgentLifecycleEvent?: (event: AgentLifecycleEvent) => void;
36
90
  onIteration?: (iteration: number, status: AgentStatus | null) => void;
37
91
  turnsPerIteration?: number;
38
92
  maxIterations?: number;
39
93
  buildPrompt: (lastStatus: AgentStatus | null, iteration: number) => string;
40
94
  onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
41
95
  allowedTools?: string[];
42
- responseSchema?: z.ZodType<unknown>;
96
+ responseSchema?: z.ZodType<TResponse>;
43
97
  runtime?: (opts: RuntimeOptions) => ModelExecutionContract;
44
98
  cwd?: string;
45
- }): Promise<AgentLoopResult> {
46
- const label = opts.label ?? "[Agent Loop]";
99
+ /** Processor chain — see sdk/src/processors. Runs sequentially around
100
+ * prompts and emitted messages. Tool-call gating is a no-op until a
101
+ * runtime that exposes a pre-tool-use seam is wired (candidate #1). */
102
+ processors?: readonly Processor[];
103
+ /** Per-run typed bag — propagated to every processor as `ctx.requestContext`. */
104
+ requestContext?: RequestContext;
105
+ }
106
+
107
+ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TResponse>): Promise<AgentLoopResult<TResponse>> {
108
+ const agentId = opts.agentId ?? randomUUID();
109
+ const label = opts.label ?? "agent";
110
+ const logLabel = opts.label ?? "[Agent Loop]";
111
+ const startedAt = Date.now();
47
112
  const turnsPerIteration = opts.turnsPerIteration ?? 40;
48
113
  const maxIterations = opts.maxIterations ?? 8;
114
+ const processors = opts.processors ?? [];
115
+ const requestContext = opts.requestContext ?? RequestContext.fromReserved({
116
+ teamId: "", runId: "", workflowId: "",
117
+ factoryId: null, apiKeyScopes: [], parentRunId: null,
118
+ });
119
+ const loopAbort = new AbortController();
120
+
121
+ const buildProcCtx = (iteration: number): ProcessorContext => ({
122
+ requestContext,
123
+ abortSignal: loopAbort.signal,
124
+ retryCount: 0,
125
+ agentId,
126
+ iteration,
127
+ });
49
128
 
50
129
  if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
51
130
  const client = opts.runtime({
52
131
  maxTurns: turnsPerIteration,
53
132
  allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
54
- label,
133
+ label: logLabel,
55
134
  cwd: opts.cwd,
135
+ processors,
136
+ requestContext,
137
+ agentId,
138
+ ...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: z.toJSONSchema(opts.responseSchema) as Record<string, unknown> } } : {}),
56
139
  });
57
140
 
141
+ // Heads-up when a caller registers tool-call gating on a runtime that
142
+ // can't honour it. Better to surface this once at start than have users
143
+ // wonder why their `denyTools(...)` never fires.
144
+ if (processors.some((p) => p.processToolCall) && !client.supportsToolCallProcessor) {
145
+ process.stderr.write(
146
+ `${logLabel} processToolCall registered but runtime does not support pre-tool-use gating — hook is dormant\n`,
147
+ );
148
+ }
149
+
58
150
  let lastSessionId = "";
59
151
  let lastStatus: AgentStatus | null = null;
60
152
  let iterationsWithoutStatus = 0;
61
153
  let blockerStreak: { key: string; count: number } | null = null;
154
+ let completedIterations = 0;
155
+ let lastResponseText = "";
156
+ let lastResponseValidationError = "";
157
+
158
+ opts.onAgentLifecycleEvent?.({ event: "agent.spawned", at: startedAt, agentId, label, allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS });
62
159
 
160
+ try {
63
161
  for (let iteration = 0; iteration < maxIterations; iteration++) {
64
- const prompt = opts.buildPrompt(lastStatus, iteration);
65
- process.stderr.write(`${label} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
162
+ const procCtx = buildProcCtx(iteration + 1);
163
+ const initialPrompt = opts.buildPrompt(lastStatus, iteration);
164
+
165
+ // processInput chain — deny ends the loop; abort ends the loop.
166
+ const inputVerdict = await runProcessorChain(processors, (p) => p.processInput, initialPrompt, procCtx);
167
+ if (inputVerdict.kind !== "continue") {
168
+ loopAbort.abort();
169
+ throw new Error(`${logLabel} input ${inputVerdict.kind === "deny" ? "rejected" : "aborted"}: ${inputVerdict.reason}`);
170
+ }
171
+ const prompt = inputVerdict.value;
172
+
173
+ // Progress, not an error — write to stdout so dashboards and
174
+ // log viewers don't visually flag it as a warning.
175
+ process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
66
176
 
67
177
  let responseText = "";
68
- for await (const msg of client.sendMessage({ prompt, sessionId: iteration > 0 ? lastSessionId : undefined })) {
178
+ for await (const rawMsg of client.sendMessage({ prompt, sessionId: iteration > 0 ? lastSessionId : undefined, iteration: iteration + 1 })) {
179
+ // processOutput chain — deny drops the message from accumulation;
180
+ // abort ends the loop. Continue carries the (possibly mutated)
181
+ // message forward.
182
+ const outputVerdict = await runProcessorChain(processors, (p) => p.processOutput, rawMsg, procCtx);
183
+ if (outputVerdict.kind === "abort") {
184
+ loopAbort.abort();
185
+ throw new Error(`${logLabel} output aborted: ${outputVerdict.reason}`);
186
+ }
187
+ if (outputVerdict.kind === "deny") {
188
+ process.stderr.write(`${logLabel} message dropped by processor: ${outputVerdict.reason}\n`);
189
+ continue;
190
+ }
191
+ const msg = outputVerdict.value;
69
192
  opts.onAgentEvent?.(iteration, msg);
193
+ opts.onAgentLifecycleEvent?.({ event: "agent.message", at: Date.now(), agentId, label, iteration: iteration + 1, message: summarizeAgentMessage(msg) });
70
194
  if (msg.type === "init") lastSessionId = msg.sessionId;
71
195
  if (msg.type === "text") responseText += msg.text;
72
196
  if (msg.type === "error") throw new Error(`Agent error: ${msg.text}`);
73
197
  }
74
- process.stdout.write("\n");
198
+ // (No trailing newline here — the dashboard renders each stdout
199
+ // write as its own log row, so a bare "\n" produces an empty
200
+ // STDOUT row between agent iterations. The visual gap was meant
201
+ // for interactive TTYs and is just noise in structured logs.)
202
+ lastResponseText = responseText;
75
203
 
76
204
  let status = parseAgentStatus(responseText);
77
- process.stderr.write(`\n${label} iteration ${iteration + 1} status: exit_signal=${status?.exit_signal ?? "(no status)"} blockers=${JSON.stringify(status?.blockers ?? [])}\n`);
205
+ // Parse the response once per iteration; both the schema fast path
206
+ // below and the exit-signal-driven validation a few lines down
207
+ // need the same parsed payload. `null` when there's no <response>
208
+ // block AND no top-level JSON object.
209
+ const rawResponse = opts.responseSchema
210
+ ? parseAgentResponse(responseText) ?? parseRawJsonResponse(responseText)
211
+ : null;
212
+ if (opts.responseSchema && rawResponse !== null) {
213
+ const parsed = opts.responseSchema.safeParse(rawResponse);
214
+ if (parsed.success) {
215
+ const successStatus = status ?? { summary: "structured response completed", completed: [], blockers: [], exit_signal: true };
216
+ // Successful schema-validation is progress, not a warning.
217
+ process.stdout.write(`${logLabel} structured response validated after ${iteration + 1}/${maxIterations} iterations\n`);
218
+ opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: iteration + 1, durationMs: Date.now() - startedAt });
219
+ return { agentId, label, sessionId: lastSessionId, lastStatus: successStatus, iterations: iteration + 1, response: parsed.data };
220
+ }
221
+ lastResponseValidationError = parsed.error.message;
222
+ }
223
+ completedIterations = iteration + 1;
224
+ // Per-iteration status is informational progress. The presence
225
+ // of blockers in the JSON is just data — the operator decides
226
+ // whether to be alarmed by it. Writing to stderr made the line
227
+ // visually flag as a warning even on clean runs (blockers=[]).
228
+ process.stdout.write(`${logLabel} iteration ${iteration + 1} status: exit_signal=${status?.exit_signal ?? "(no status)"} blockers=${JSON.stringify(status?.blockers ?? [])}\n`);
78
229
  lastStatus = status ?? lastStatus;
79
230
  opts.onIteration?.(iteration + 1, status);
231
+ opts.onAgentLifecycleEvent?.({ event: "agent.iteration", at: Date.now(), agentId, label, iteration: iteration + 1, status });
80
232
 
81
233
  if (status?.exit_signal && (status.blockers?.length ?? 0) === 0) {
82
- let response: unknown = status;
234
+ let response: AgentStatus | TResponse = status;
83
235
  if (opts.responseSchema) {
84
- const raw = parseAgentResponse(responseText);
85
- if (raw === null) {
86
- process.stderr.write(`\n${label} NO <response> BLOCK — response tail: ${responseText.slice(-400)}\n`);
236
+ if (rawResponse === null) {
237
+ process.stderr.write(`${logLabel} NO <response> BLOCK — response tail: ${responseText.slice(-400)}\n`);
87
238
  status = { ...status!, exit_signal: false, blockers: ["No <response> block found — emit a <response> block with the required JSON fields before setting exit_signal: true"] };
88
239
  opts.onIteration?.(iteration + 1, status);
89
240
  continue;
90
241
  }
91
- const parsed = opts.responseSchema.safeParse({ ...status, ...(raw as object) });
242
+ // Status-merged validation — the schema may reference status
243
+ // fields (`exit_signal`, `blockers`, …) the agent emitted in a
244
+ // separate <status> block. Distinct from the fast path above,
245
+ // which validates the raw response alone.
246
+ const parsed = opts.responseSchema.safeParse({ ...status, ...(rawResponse as object) });
92
247
  if (!parsed.success) {
93
- process.stderr.write(`\n${label} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(raw).slice(0, 400)}\n`);
248
+ lastResponseValidationError = parsed.error.message;
249
+ process.stderr.write(`${logLabel} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(rawResponse).slice(0, 400)}\n`);
94
250
  status = { ...status!, exit_signal: false, blockers: [`<response> schema validation failed: ${parsed.error.message}`] };
95
251
  opts.onIteration?.(iteration + 1, status);
96
252
  continue;
97
253
  }
98
254
  response = parsed.data;
99
255
  }
100
- process.stderr.write(`${label} done after ${iteration + 1}/${maxIterations} iterations\n`);
101
- return { sessionId: lastSessionId, lastStatus: status, iterations: iteration + 1, response };
256
+ // Settled successfully — progress, not a warning.
257
+ process.stdout.write(`${logLabel} done after ${iteration + 1}/${maxIterations} iterations\n`);
258
+ opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: iteration + 1, durationMs: Date.now() - startedAt });
259
+ return { agentId, label, sessionId: lastSessionId, lastStatus: status, iterations: iteration + 1, response: response as TResponse };
102
260
  }
103
261
 
104
262
  if (!status) {
105
263
  if (++iterationsWithoutStatus >= STALL_ITERATIONS)
106
- throw new Error(`${label} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
264
+ throw new Error(`${logLabel} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
107
265
  } else {
108
266
  iterationsWithoutStatus = 0;
109
267
  }
@@ -112,7 +270,7 @@ export async function agentLoop(opts: {
112
270
  const key = status.blockers.join("|");
113
271
  if (blockerStreak !== null && blockerStreak.key === key) {
114
272
  if (++blockerStreak.count >= SAME_BLOCKER_ITERATIONS)
115
- throw new Error(`${label} circuit break: same blocker repeated ${blockerStreak.count}x — "${status.blockers[0]}"`);
273
+ throw new Error(`${logLabel} circuit break: same blocker repeated ${blockerStreak.count}x — "${status.blockers[0]}"`);
116
274
  } else {
117
275
  blockerStreak = { key, count: 1 };
118
276
  }
@@ -121,11 +279,24 @@ export async function agentLoop(opts: {
121
279
  }
122
280
 
123
281
  if (iteration + 1 < maxIterations)
124
- process.stderr.write(`${label} continuing to iteration ${iteration + 2}/${maxIterations}\n`);
282
+ // Loop continuation — progress.
283
+ process.stdout.write(`${logLabel} continuing to iteration ${iteration + 2}/${maxIterations}\n`);
125
284
  }
126
285
 
127
286
  if (opts.responseSchema)
128
- throw new Error(`${label} did not produce a valid <response> after ${maxIterations} iterations`);
129
- process.stderr.write(`${label} exhausted ${maxIterations} iterations, proceeding with available work\n`);
130
- return { sessionId: lastSessionId, lastStatus, iterations: maxIterations };
287
+ throw new Error(`${logLabel} did not produce a valid <response> after ${maxIterations} iterations${lastResponseValidationError ? `: ${lastResponseValidationError}` : ""}. Response tail: ${lastResponseText.slice(-600)}`);
288
+ process.stderr.write(`${logLabel} exhausted ${maxIterations} iterations, proceeding with available work\n`);
289
+ opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "success", iterations: maxIterations, durationMs: Date.now() - startedAt });
290
+ return { agentId, label, sessionId: lastSessionId, lastStatus, iterations: maxIterations };
291
+ } catch (err) {
292
+ opts.onAgentLifecycleEvent?.({ event: "agent.settled", at: Date.now(), agentId, label, outcome: "failed", iterations: completedIterations, durationMs: Date.now() - startedAt, failureReason: err instanceof Error ? err.message : String(err) });
293
+ throw err;
294
+ }
295
+ }
296
+
297
+ function parseRawJsonResponse(text: string): unknown {
298
+ const trimmed = text.trim();
299
+ if (!trimmed.startsWith("{") || !trimmed.endsWith("}")) return null;
300
+ try { return JSON.parse(trimmed); }
301
+ catch { return null; }
131
302
  }
@@ -1,10 +1,9 @@
1
1
  /**
2
- * runAgent — canonical entry point for embedding an LLM agent inside a
2
+ * agent — canonical entry point for embedding an LLM agent inside a
3
3
  * workflow. The workflow's `run()` body calls it; the loop executes
4
4
  * against the runner's own VM.
5
5
  *
6
6
  * Glue packaged so workflows don't duplicate it:
7
- * - Inject `{{VAR}}` placeholders into the prompt template.
8
7
  * - Strip the `--- frontmatter ---` header authors use for IDE hints.
9
8
  * - Append PROTOCOL_SUFFIX (status/response format instructions).
10
9
  * - Append a response-format appendix when `responseSchema` is set.
@@ -13,11 +12,13 @@
13
12
 
14
13
  import { z } from "zod";
15
14
  import { agentLoop } from "./agent-loop.js";
16
- import type { AgentLoopResult } from "./agent-loop.js";
15
+ import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
17
16
  import type { AgentMessage, AgentStatus } from "../types/protocol.js";
18
17
  import type { AgentRuntime, RuntimeOptions } from "../types/runtime.js";
19
18
  import type { SandboxProvider } from "../types/sandbox.js";
20
19
  import type { AgentBudget } from "../types/workflow.js";
20
+ import type { Processor } from "../processors/processor.js";
21
+ import { RequestContext } from "../request-context/request-context.js";
21
22
  import PROTOCOL_SUFFIX_RAW from "./protocol-suffix.md" with { type: "text" };
22
23
 
23
24
  const PROTOCOL_SUFFIX = stripFrontmatter(PROTOCOL_SUFFIX_RAW);
@@ -28,10 +29,6 @@ function stripFrontmatter(content: string): string {
28
29
  return end === -1 ? content : content.slice(end + 4).trimStart();
29
30
  }
30
31
 
31
- function inject(template: string, vars: Record<string, string>): string {
32
- return Object.entries(vars).reduce((t, [k, v]) => t.replaceAll(`{{${k}}}`, v), template);
33
- }
34
-
35
32
  // eslint-disable-next-line @typescript-eslint/no-explicit-any
36
33
  function zodTypeName(schema: any): string {
37
34
  if (schema instanceof z.ZodString) return "string";
@@ -74,7 +71,9 @@ function buildResponseFormatAppendix(schema: z.ZodType<any>): string {
74
71
  return [...header, ...body].join("\n");
75
72
  }
76
73
 
77
- export interface RunAgentOpts<T = unknown> {
74
+ export interface AgentOpts<T = unknown> {
75
+ /** Stable system identity for this agent loop. Generated when omitted. */
76
+ agentId?: string;
78
77
  /** Sandbox the runtime executes commands against. Inside a workflow,
79
78
  * always pass `ctx.sandbox` — a pre-constructed local provider for
80
79
  * the runner's own VM. Exposed as a parameter so tests and non-workflow
@@ -82,12 +81,9 @@ export interface RunAgentOpts<T = unknown> {
82
81
  sandbox: SandboxProvider;
83
82
  /** Runtime definition from `createClaudeRuntime({...})` (or custom). */
84
83
  runtime: AgentRuntime;
85
- /** Prompt template. Authors can include YAML-style `--- frontmatter ---`
84
+ /** Prompt text. Authors can include YAML-style `--- frontmatter ---`
86
85
  * at the top for IDE hints; it's stripped before the model sees it. */
87
86
  prompt: string;
88
- /** Substitution map for `{{VAR}}` placeholders in the prompt. `WORKING_DIR`
89
- * and `DIFF_BASE` auto-populate from `opts.workingDir` unless overridden. */
90
- promptVars?: Record<string, string>;
91
87
  /** `cwd` forwarded to the runtime — every shell command runs here. */
92
88
  workingDir?: string;
93
89
  /** Tools the model may use. Defaults to a safe kitchen-sink set inside
@@ -101,12 +97,35 @@ export interface RunAgentOpts<T = unknown> {
101
97
  responseSchema?: z.ZodType<T>;
102
98
  /** Label prefix for runtime stderr ("[sbid][agent]" by default). */
103
99
  label?: string;
100
+ /** Lifecycle event sink from workflow ctx. Emits agent.spawned / iteration / settled. */
101
+ events?: { emit: (event: AgentLifecycleEvent) => void | Promise<void> };
104
102
  /** Per-message event callback — wire this to your workflow's event
105
103
  * telemetry if you want per-tool-call observability. */
106
104
  onAgentEvent?: (iteration: number, msg: AgentMessage) => void;
107
105
  /** Per-iteration status callback — fires after each model turn with the
108
106
  * parsed `<status>` block (or null if the model didn't emit one). */
109
107
  onIteration?: (iteration: number, status: AgentStatus | null) => void;
108
+ /**
109
+ * Processors run as a typed pre/post pipeline around the agent loop:
110
+ * - processInput before each iteration's prompt is sent
111
+ * - processOutput on each emitted message
112
+ * - processToolCall before each tool call (runtime-driven; dormant
113
+ * until a runtime that supports pre-tool gating is wired in)
114
+ *
115
+ * Common pattern with workflow-level defaults:
116
+ * agent({ processors: [...ctx.processors, mySpecific], ... })
117
+ */
118
+ processors?: readonly Processor[];
119
+ /**
120
+ * Per-run request context. Passed through to processors as
121
+ * `ctx.requestContext`; carries tenant identity (teamId, factoryId, scopes)
122
+ * and the freeform user namespace.
123
+ *
124
+ * Inside a workflow, pass `ctx.requestContext`. Tests/non-workflow callers
125
+ * can omit; a degenerate context is synthesised so processors that don't
126
+ * read identity (e.g. redactPattern) still work.
127
+ */
128
+ requestContext?: RequestContext;
110
129
  }
111
130
 
112
131
  /**
@@ -114,12 +133,12 @@ export interface RunAgentOpts<T = unknown> {
114
133
  * `AgentLoopResult`, including `response` when a `responseSchema` was
115
134
  * supplied and the model validated against it.
116
135
  */
117
- export async function runAgent<T = unknown>(opts: RunAgentOpts<T>): Promise<AgentLoopResult> {
136
+ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopResult<T>> {
118
137
  const workingDir = opts.workingDir ?? "";
119
- const promptVars = { WORKING_DIR: workingDir, DIFF_BASE: "HEAD~1", ...opts.promptVars };
120
138
 
121
139
  return agentLoop({
122
140
  runtime: (runtimeOpts: RuntimeOptions) => opts.runtime.create(opts.sandbox, runtimeOpts),
141
+ ...(opts.agentId !== undefined ? { agentId: opts.agentId } : {}),
123
142
  ...(opts.label !== undefined ? { label: opts.label } : {}),
124
143
  ...(opts.budget?.turnsPerIteration !== undefined ? { turnsPerIteration: opts.budget.turnsPerIteration } : {}),
125
144
  ...(opts.budget?.maxIterations !== undefined ? { maxIterations: opts.budget.maxIterations } : {}),
@@ -130,12 +149,18 @@ export async function runAgent<T = unknown>(opts: RunAgentOpts<T>): Promise<Agen
130
149
  // Mid-loop: session resumes via --resume, model has history; only the
131
150
  // protocol suffix is needed so tool-use + status block grammar stays fresh.
132
151
  if (iteration > 0) return PROTOCOL_SUFFIX;
133
- const base = inject(stripFrontmatter(opts.prompt), promptVars);
152
+ const base = stripFrontmatter(opts.prompt);
134
153
  return opts.responseSchema
135
154
  ? `${base}\n\n${PROTOCOL_SUFFIX}\n\n${buildResponseFormatAppendix(opts.responseSchema)}`
136
155
  : `${base}\n\n${PROTOCOL_SUFFIX}`;
137
156
  },
157
+ ...(opts.events ? { onAgentLifecycleEvent: (event: AgentLifecycleEvent) => { void opts.events?.emit(event); } } : {}),
138
158
  ...(opts.onAgentEvent ? { onAgentEvent: opts.onAgentEvent } : {}),
139
159
  ...(opts.onIteration ? { onIteration: opts.onIteration } : {}),
160
+ ...(opts.processors?.length ? { processors: opts.processors } : {}),
161
+ requestContext: opts.requestContext ?? RequestContext.fromReserved({
162
+ teamId: "", runId: "", workflowId: "",
163
+ factoryId: null, apiKeyScopes: [], parentRunId: null,
164
+ }),
140
165
  });
141
166
  }