@agent-compose/sdk 0.5.5 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/agent/agent-loop-contract.test.d.ts +1 -0
- package/dist/agent/agent-loop.d.ts +1 -1
- package/dist/client.d.ts +65 -23
- package/dist/index.d.ts +4 -2
- package/dist/index.js +389 -109
- package/dist/runtimes/_cli-agent.d.ts +25 -8
- package/dist/runtimes/amp.d.ts +7 -6
- package/dist/runtimes/claude.d.ts +9 -1
- package/dist/runtimes/cli-agent.test.d.ts +9 -0
- package/dist/runtimes/codex.d.ts +6 -5
- package/dist/runtimes/openai-desktop.js +387 -109
- package/dist/sandbox-errors.d.ts +49 -0
- package/dist/sandbox.d.ts +68 -13
- package/dist/step-invocation/protocol.d.ts +6 -0
- package/dist/types/sandbox-environment.d.ts +1 -10
- package/dist/types/sandbox.d.ts +27 -3
- package/dist/types/workflow-metadata.d.ts +88 -23
- package/dist/types/workflow.d.ts +38 -10
- package/dist/utils/bundler.d.ts +38 -9
- package/dist/workflow-steps/workflow.d.ts +1 -3
- package/package.json +2 -2
- package/src/agent/agent-loop.ts +56 -7
- package/src/client.ts +144 -25
- package/src/index.ts +4 -2
- package/src/runtimes/_cli-agent.ts +75 -42
- package/src/runtimes/amp.ts +11 -6
- package/src/runtimes/claude.ts +66 -8
- package/src/runtimes/codex.ts +10 -5
- package/src/sandbox-errors.ts +53 -0
- package/src/sandbox.ts +378 -56
- package/src/step-invocation/invoker.ts +66 -8
- package/src/step-invocation/protocol.ts +9 -0
- package/src/step-invocation/server.ts +27 -4
- package/src/types/sandbox-environment.ts +1 -11
- package/src/types/sandbox.ts +28 -3
- package/src/types/workflow-metadata.ts +97 -26
- package/src/types/workflow.ts +38 -11
- package/src/utils/bundler.ts +43 -13
- package/src/workflow-steps/workflow.ts +1 -3
package/src/runtimes/claude.ts
CHANGED
|
@@ -62,8 +62,20 @@ function translateMessage(message: Record<string, unknown>): AgentMessage[] {
|
|
|
62
62
|
numTurns: Number(message.num_turns ?? 0),
|
|
63
63
|
timestamp: ts,
|
|
64
64
|
});
|
|
65
|
-
if (message.
|
|
66
|
-
|
|
65
|
+
if (message.subtype === "error_max_turns") {
|
|
66
|
+
// Running out of turns is NOT a failure — the agent did real work and the
|
|
67
|
+
// files it wrote are on disk. End the iteration cleanly (don't throw) so the
|
|
68
|
+
// workflow keeps the partial result and moves on.
|
|
69
|
+
msgs.push({ type: "done", sessionId: String(message.session_id ?? ""), timestamp: ts });
|
|
70
|
+
} else if (message.is_error || message.subtype === "error_during_execution") {
|
|
71
|
+
// Error results sometimes carry NO error/result/message fields (e.g.
|
|
72
|
+
// the binary died early) — fall back to subtype + the raw envelope so
|
|
73
|
+
// the failure is diagnosable instead of "Agent error: undefined".
|
|
74
|
+
const detail = message.error ?? message.result ?? message.message;
|
|
75
|
+
const text = detail !== undefined
|
|
76
|
+
? formatError(detail)
|
|
77
|
+
: `${String(message.subtype ?? "unknown_error")} — raw result: ${JSON.stringify({ ...message, usage: undefined }).slice(0, 600)}`;
|
|
78
|
+
msgs.push({ type: "error", text, timestamp: ts });
|
|
67
79
|
} else {
|
|
68
80
|
msgs.push({ type: "done", sessionId: String(message.session_id ?? ""), timestamp: ts });
|
|
69
81
|
}
|
|
@@ -77,7 +89,11 @@ export interface ClaudeRuntimeConfig {
|
|
|
77
89
|
claudeMdContent?: string;
|
|
78
90
|
/** Env overrides for Agent SDK provider routing. */
|
|
79
91
|
env?: Record<string, string>;
|
|
80
|
-
/** Model to use.
|
|
92
|
+
/** Model to use. Accepts a caliber shorthand — "fable" (most capable),
|
|
93
|
+
* "opus", "sonnet", "haiku" (fastest/cheapest) — or an exact model id.
|
|
94
|
+
* Defaults to DEFAULT_CLAUDE_MODEL (Fable). Orchestrators pick a caliber
|
|
95
|
+
* per agent: fable/opus for planning + implementation, sonnet for focused
|
|
96
|
+
* single-responsibility work, haiku for mechanical tasks. */
|
|
81
97
|
model?: string;
|
|
82
98
|
/** MCP servers to configure for the Agent SDK. */
|
|
83
99
|
mcpServers?: Record<string, { command: string; args?: string[]; env?: Record<string, string> }>;
|
|
@@ -87,6 +103,31 @@ export interface ClaudeRuntimeConfig {
|
|
|
87
103
|
thinking?: ThinkingConfig;
|
|
88
104
|
/** Reasoning effort hint for models that support adaptive thinking. */
|
|
89
105
|
effort?: "low" | "medium" | "high" | "xhigh" | "max";
|
|
106
|
+
/** Skills to enable for the agent (the Agent SDK's `skills` option — also
|
|
107
|
+
* auto-adds the `Skill` tool). The agent-env bakes the `/ac:*` skills via
|
|
108
|
+
* `agentc init`; default `"all"` makes them usable. Pass `[]` to disable. */
|
|
109
|
+
skills?: string[] | "all";
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** Caliber shorthand → exact model id. Full ids pass through untouched. */
|
|
113
|
+
const MODEL_TIERS: Record<string, string> = {
|
|
114
|
+
fable: "claude-fable-5",
|
|
115
|
+
opus: "claude-opus-4-8",
|
|
116
|
+
sonnet: "claude-sonnet-4-6",
|
|
117
|
+
haiku: "claude-haiku-4-5",
|
|
118
|
+
};
|
|
119
|
+
function resolveClaudeModel(model: string | undefined): string {
|
|
120
|
+
if (!model) return DEFAULT_CLAUDE_MODEL;
|
|
121
|
+
return MODEL_TIERS[model.toLowerCase()] ?? model;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** Fable rejects an explicit `thinking: {type: "disabled"}` with a 400 (the only
|
|
125
|
+
* off-mode on Fable is omitting the param). Other models accept it. Resolve the
|
|
126
|
+
* thinking config against the chosen model so callers can keep passing
|
|
127
|
+
* `thinking: {type: "disabled"}` for cheap/fast agents regardless of tier. */
|
|
128
|
+
function resolveThinking(model: string, thinking: ThinkingConfig | undefined): ThinkingConfig | undefined {
|
|
129
|
+
if (model.startsWith("claude-fable") && thinking && (thinking as { type?: string }).type === "disabled") return undefined;
|
|
130
|
+
return thinking;
|
|
90
131
|
}
|
|
91
132
|
|
|
92
133
|
/** Canonical install path for Claude Code inside a Vercel sandbox.
|
|
@@ -130,7 +171,7 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
130
171
|
) {}
|
|
131
172
|
|
|
132
173
|
get model(): string {
|
|
133
|
-
return this.config.model ?? this.options.model
|
|
174
|
+
return resolveClaudeModel(this.config.model ?? this.options.model);
|
|
134
175
|
}
|
|
135
176
|
|
|
136
177
|
async gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult> {
|
|
@@ -179,7 +220,7 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
179
220
|
//
|
|
180
221
|
// Without an inboxStream we keep the simple string-prompt path —
|
|
181
222
|
// no queue, same behaviour as before. This keeps non-interactive
|
|
182
|
-
// workflows
|
|
223
|
+
// workflows on the original code path.
|
|
183
224
|
//
|
|
184
225
|
// The queue is closed when the SDK's `result` message indicates
|
|
185
226
|
// the iteration's assistant turns are done; without close() the
|
|
@@ -216,9 +257,9 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
216
257
|
tools: this.options.allowedTools,
|
|
217
258
|
allowedTools: this.options.allowedTools,
|
|
218
259
|
maxTurns: this.options.maxTurns,
|
|
219
|
-
model: this.
|
|
260
|
+
model: this.model,
|
|
220
261
|
outputFormat: this.options.outputFormat,
|
|
221
|
-
thinking: this.config.thinking,
|
|
262
|
+
thinking: resolveThinking(this.model, this.config.thinking),
|
|
222
263
|
effort: this.config.effort,
|
|
223
264
|
cwd: this.options.cwd,
|
|
224
265
|
env: { ...process.env, ...(this.config.env ?? {}) },
|
|
@@ -226,6 +267,10 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
226
267
|
?? process.env.CLAUDE_CODE_EXECUTABLE
|
|
227
268
|
?? DEFAULT_CLAUDE_PATH,
|
|
228
269
|
...(this.config.claudeMdContent ? { systemPrompt: { type: "preset" as const, preset: "claude_code" as const, append: this.config.claudeMdContent } } : {}),
|
|
270
|
+
// Turn skills ON (and auto-add the `Skill` tool). The agent-env
|
|
271
|
+
// bakes the `/ac:*` skills; without this the Agent SDK leaves
|
|
272
|
+
// them un-enabled and the agent can't invoke them.
|
|
273
|
+
skills: this.config.skills ?? "all",
|
|
229
274
|
resume: opts.sessionId,
|
|
230
275
|
mcpServers: this.config.mcpServers,
|
|
231
276
|
permissionMode: "acceptEdits",
|
|
@@ -251,7 +296,20 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
251
296
|
if (raw.type === "result" && inboxQueue) inboxQueue.close();
|
|
252
297
|
}
|
|
253
298
|
} catch (err) {
|
|
254
|
-
|
|
299
|
+
// The SDK ALSO signals "out of turns" by THROWING (separately from the
|
|
300
|
+
// result-message `error_max_turns` subtype handled in translateMessage — that
|
|
301
|
+
// typed path is the primary signal; this catch covers the SDK's separate
|
|
302
|
+
// throw, which is prose-only). Treat it the same: NOT a failure — end the
|
|
303
|
+
// iteration cleanly so the loop continues, keeping the work it did.
|
|
304
|
+
// Match ONLY the specific exhaustion phrase, NOT a loose "maxTurns" token: the
|
|
305
|
+
// latter would also swallow a genuine error like "maxTurns must be a positive
|
|
306
|
+
// integer" and silently report it as a clean finish.
|
|
307
|
+
const text = formatError(err);
|
|
308
|
+
if (/maximum number of turns/i.test(text)) {
|
|
309
|
+
yield { type: "done", sessionId: opts.sessionId ?? "", timestamp: now() };
|
|
310
|
+
} else {
|
|
311
|
+
yield { type: "error", text, timestamp: now() };
|
|
312
|
+
}
|
|
255
313
|
} finally {
|
|
256
314
|
// Belt-and-braces — close on error/abort too so the dangling
|
|
257
315
|
// iterable doesn't leak the inboxStream consumer.
|
package/src/runtimes/codex.ts
CHANGED
|
@@ -4,12 +4,13 @@
|
|
|
4
4
|
* Built on the shared CLI-agent base; Codex brings its own loop + tools, so we
|
|
5
5
|
* only stream-parse what it prints.
|
|
6
6
|
*
|
|
7
|
-
* Auth: set `
|
|
8
|
-
* workflow secret.
|
|
7
|
+
* Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
|
|
8
|
+
* workflow secret. The runtime installs the `codex` CLI (`@openai/codex`) on
|
|
9
|
+
* demand — no image baking needed; pair with `snapshots: { bootFrom: "reuse" }`
|
|
10
|
+
* to install once and boot from the captured snapshot on every run after.
|
|
9
11
|
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
* (developers.openai.com/codex/noninteractive). Verify before production use.
|
|
12
|
+
* Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
|
|
13
|
+
* with the command_execution / reasoning / agent_message item shapes below.
|
|
13
14
|
*/
|
|
14
15
|
|
|
15
16
|
import type { AgentMessage } from "../index.js";
|
|
@@ -21,6 +22,10 @@ function now(): string { return new Date().toISOString(); }
|
|
|
21
22
|
const codexSpec: CliAgentSpec = {
|
|
22
23
|
kind: "codex",
|
|
23
24
|
authEnv: "CODEX_API_KEY",
|
|
25
|
+
bin: "codex",
|
|
26
|
+
// Global npm install; symlink onto PATH only if the global bin dir isn't
|
|
27
|
+
// already there (so a non-login `sh -c` can find it).
|
|
28
|
+
install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
|
|
24
29
|
// Codex reads the prompt from stdin when invoked as `codex exec ... -`.
|
|
25
30
|
promptPayload: (prompt) => prompt,
|
|
26
31
|
buildCommand: ({ promptPath, sessionId, model, cwd }) => {
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sandbox-infrastructure error.
|
|
3
|
+
*
|
|
4
|
+
* Distinct from `StepExecutionError` (the user/runner step-failure wrapper):
|
|
5
|
+
* that's the workflow author's problem. A `SandboxUnavailableError` means the
|
|
6
|
+
* sandbox itself couldn't carry the step — the provider refused a command, the
|
|
7
|
+
* sandbox was reclaimed (idle/lifetime timeout, eviction), an API blip, or the
|
|
8
|
+
* connection dropped mid-stream. None of these are a fault in the customer's
|
|
9
|
+
* code, so they're surfaced as "infrastructure issue, retry" rather than
|
|
10
|
+
* blaming the user.
|
|
11
|
+
*
|
|
12
|
+
* ## Retryability is about *when*, not *which error*
|
|
13
|
+
*
|
|
14
|
+
* The Vercel provider (`sandbox.ts`) decides `retryable` purely from *where* it
|
|
15
|
+
* caught the error — not by inspecting status codes or error vocabularies:
|
|
16
|
+
* - `retryable: true` — caught before the runner launched the step (file
|
|
17
|
+
* write / command launch / reconnect refused). No user code ran, so
|
|
18
|
+
* re-provisioning a fresh sandbox and re-running the step is safe. ANY
|
|
19
|
+
* error here qualifies (sandbox reclaimed, 429, 5xx, network blip).
|
|
20
|
+
* - `retryable: false` — caught while/after the runner streamed, so user code
|
|
21
|
+
* may already have executed side effects. Surfaced honestly as an infra
|
|
22
|
+
* failure, but NOT auto-retried.
|
|
23
|
+
*
|
|
24
|
+
* ## Cross-process contract
|
|
25
|
+
*
|
|
26
|
+
* The thrown error crosses the Temporal activity→workflow serialisation
|
|
27
|
+
* boundary, which preserves the error **message** but discards custom instance
|
|
28
|
+
* fields. So the retryability signal is encoded IN the message via a stable
|
|
29
|
+
* prefix — `[sandbox-unavailable:retryable]` or `[sandbox-unavailable:terminal]`
|
|
30
|
+
* — following the same `[kind] …` convention `StepExecutionError` uses. The
|
|
31
|
+
* server (`utils/transient-errors.ts`) matches this prefix with a pure regex to
|
|
32
|
+
* classify the failure and decide whether the workflow re-provisions and
|
|
33
|
+
* retries. A unit test pins the SDK-produced string against the server matcher
|
|
34
|
+
* so the two can't drift.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
/** Stable message prefix. Both variants share this leading token so a single
|
|
38
|
+
* server-side regex recognises the class; the `:retryable` / `:terminal`
|
|
39
|
+
* suffix carries the recovery decision. */
|
|
40
|
+
export const SANDBOX_UNAVAILABLE_PREFIX = "[sandbox-unavailable";
|
|
41
|
+
|
|
42
|
+
export class SandboxUnavailableError extends Error {
|
|
43
|
+
/** True when caught before any user code ran (safe to re-provision + retry);
|
|
44
|
+
* false when the sandbox died mid/after execution. */
|
|
45
|
+
readonly retryable: boolean;
|
|
46
|
+
readonly sandboxId: string | undefined;
|
|
47
|
+
constructor(detail: string, opts: { retryable: boolean; sandboxId?: string }) {
|
|
48
|
+
super(`${SANDBOX_UNAVAILABLE_PREFIX}:${opts.retryable ? "retryable" : "terminal"}] ${detail}`);
|
|
49
|
+
this.name = "SandboxUnavailableError";
|
|
50
|
+
this.retryable = opts.retryable;
|
|
51
|
+
this.sandboxId = opts.sandboxId;
|
|
52
|
+
}
|
|
53
|
+
}
|