@agent-compose/sdk 0.5.2 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,194 @@
1
+ /**
2
+ * CLI-agent runtime base — drive an external agentic coding CLI inside the
3
+ * sandbox and map its JSON-Lines (JSONL) stream onto the `AgentMessage`
4
+ * contract.
5
+ *
6
+ * This is a different *mechanism* from the other runtimes: `claudeRuntime`
7
+ * drives the Anthropic Agent SDK and `vercelRuntime` drives the Vercel AI SDK,
8
+ * but a CLI-agent runtime spawns the provider's own CLI (`codex exec --json`,
9
+ * `amp -x --stream-json`) — the CLI brings its own agent loop + tools, and we
10
+ * only stream-parse the events it prints. Codex and Amp are the first two; this
11
+ * base is the shared machinery (spawn → line-buffer stdout → JSONL parse →
12
+ * AsyncQueue bridge → init/usage/done/error lifecycle), parameterised per-CLI
13
+ * by a `CliAgentSpec`.
14
+ *
15
+ * Provisioning (per spec): the runtime installs the provider CLI on demand —
16
+ * `command -v <bin>` before the first turn, falling back to the spec's
17
+ * `install` command when it's absent — so it works on a bare sandbox with no
18
+ * manual setup or hardcoded snapshot id. Pair it with
19
+ * `snapshots: { bootFrom: "reuse" }` on the workflow and the install happens
20
+ * exactly once: the first run installs the CLI and captures a snapshot, and
21
+ * every run after boots from that snapshot with the CLI already present (the
22
+ * probe short-circuits). The provider's API key must be in the sandbox env
23
+ * (see each spec's `authEnv`); `commands.run` inherits the sandbox env, so
24
+ * values set via `agentc secrets set` are visible to the CLI.
25
+ */
26
+
27
+ import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider } from "../index.js";
28
+ import { defineRuntime } from "../types/runtime.js";
29
+ import { AsyncQueue } from "../agent/async-queue.js";
30
+ import { formatError } from "../utils/errors.js";
31
+
32
+ function now(): string { return new Date().toISOString(); }
33
+
34
+ /** Single-quote a value for safe interpolation into a `sh -c` command line. */
35
+ export function shellQuote(value: string): string {
36
+ return `'${value.replace(/'/g, `'\\''`)}'`;
37
+ }
38
+
39
+ /** Per-CLI behaviour. The base owns the lifecycle (init/done/error) and the
40
+ * transport (spawn + JSONL parse); a spec owns the CLI-specific bits. */
41
+ export interface CliAgentSpec {
42
+ /** Runtime self-id surfaced on `agent.spawned` (dashboard runtime icon). */
43
+ kind: string;
44
+ /** Env var the CLI reads for auth — documentation only (must be set in the
45
+ * sandbox env via a workflow secret). */
46
+ authEnv: string;
47
+ /** Default model id when none is configured; omit to let the CLI choose. */
48
+ defaultModel?: string;
49
+ /** Binary the runtime spawns (`codex`, `amp`). Probed with `command -v`
50
+ * before the first turn; if absent, `install` provisions it. */
51
+ bin: string;
52
+ /** Shell command that installs `bin` when it's missing from the sandbox.
53
+ * Runs at most once per runner, and only when the probe fails — so booting
54
+ * from a snapshot that already has the CLI (the steady state under
55
+ * `snapshots: { bootFrom: "reuse" }`) skips it. Must leave `bin` resolvable
56
+ * on a non-login shell's PATH (the runtime spawns via `sh -c`). */
57
+ install: string;
58
+ /** Serialise the user prompt into the bytes written to the prompt file —
59
+ * plain text for a CLI that reads the prompt from stdin (codex `-`), or a
60
+ * JSONL user message for a `--stream-json-input` CLI (amp). */
61
+ promptPayload(prompt: string): string;
62
+ /** Build the one-shot shell command for a turn. `promptPath` is a file in the
63
+ * sandbox holding `promptPayload(prompt)`; `sessionId` continues a thread. */
64
+ buildCommand(args: { promptPath: string; sessionId?: string; model?: string; cwd?: string }): string;
65
+ /** Map one parsed JSONL stdout event to `AgentMessage`s. The base emits
66
+ * `init`/`done`/`error` lifecycle itself, so a spec maps only content +
67
+ * usage (text / thinking / tool_use / tool_result / usage). */
68
+ mapEvent(parsed: Record<string, unknown>): AgentMessage[];
69
+ /** Pull a session/thread id out of a parsed event so the next turn can
70
+ * resume it (codex `thread.started.thread_id`, amp `session_id`). */
71
+ extractSessionId(parsed: Record<string, unknown>): string | undefined;
72
+ }
73
+
74
+ export class CliAgentRunner implements ModelExecutionContract {
75
+ readonly kind: string;
76
+
77
+ constructor(
78
+ private readonly sandbox: SandboxProvider,
79
+ private readonly options: RuntimeOptions,
80
+ private readonly spec: CliAgentSpec,
81
+ private readonly configModel?: string,
82
+ ) {
83
+ this.kind = spec.kind;
84
+ }
85
+
86
+ get model(): string | undefined {
87
+ return this.configModel ?? this.options.model ?? this.spec.defaultModel;
88
+ }
89
+
90
+ private installed = false;
91
+
92
+ /** Provision the CLI on demand. A no-op once `bin` is on PATH — the steady
93
+ * state under `snapshots: { bootFrom: "reuse" }`, where the first run's
94
+ * install is baked into the snapshot every later run boots from. So the
95
+ * install command runs exactly once: on the first run of a content hash. */
96
+ private async ensureInstalled(): Promise<void> {
97
+ if (this.installed) return;
98
+ const probe = await this.sandbox.commands.run(`command -v ${this.spec.bin}`);
99
+ if (probe.exitCode === 0) { this.installed = true; return; }
100
+ const res = await this.sandbox.commands.run(this.spec.install, { timeoutMs: 300_000 });
101
+ if (res.exitCode !== 0) {
102
+ throw new Error(`failed to install ${this.spec.bin}: ${(res.stderr || res.stdout || "").slice(-500)}`);
103
+ }
104
+ this.installed = true;
105
+ }
106
+
107
+ // No captureCheckpoint/restoreCheckpoint: the CLI persists its thread/rollout
108
+ // on the sandbox filesystem (which round-trips through the pause snapshot),
109
+ // and the loop already carries the session id we emit on init/done and pass
110
+ // back as `sessionId` for resume — same model as claudeRuntime.
111
+
112
+ async *sendMessage(opts: {
113
+ prompt: string;
114
+ sessionId?: string;
115
+ iteration?: number;
116
+ signal?: AbortSignal;
117
+ }): AsyncGenerator<AgentMessage> {
118
+ yield { type: "init", sessionId: opts.sessionId ?? "", timestamp: now() };
119
+
120
+ let sessionId = opts.sessionId;
121
+ let sawError = false;
122
+ try {
123
+ // Provision the CLI before the first turn (no-op when it's already
124
+ // present, e.g. booting from a "reuse" snapshot). A failure here surfaces
125
+ // as an `error` AgentMessage via the catch below.
126
+ await this.ensureInstalled();
127
+
128
+ const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
129
+ await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
130
+ const cmd = this.spec.buildCommand({
131
+ promptPath,
132
+ sessionId: opts.sessionId,
133
+ model: this.model,
134
+ cwd: this.options.cwd,
135
+ });
136
+
137
+ // Bridge the streaming stdout callback into an async-iterable of complete
138
+ // JSONL lines. `onStdout` chunks aren't line-aligned, so buffer + split.
139
+ const lines = new AsyncQueue<string>();
140
+ let buf = "";
141
+ const onStdout = (data: string) => {
142
+ buf += data;
143
+ let nl: number;
144
+ while ((nl = buf.indexOf("\n")) >= 0) {
145
+ const line = buf.slice(0, nl).trim();
146
+ buf = buf.slice(nl + 1);
147
+ if (line) lines.push(line);
148
+ }
149
+ };
150
+
151
+ // `commands.run` resolves when the process exits. Kick it off (don't await
152
+ // yet); flush the trailing buffer + close the queue on completion so the
153
+ // for-await below drains and we can read the exit code.
154
+ const runPromise = this.sandbox.commands.run(cmd, {
155
+ ...(this.options.cwd ? { cwd: this.options.cwd } : {}),
156
+ onStdout,
157
+ }).then(
158
+ (res) => { const tail = buf.trim(); if (tail) lines.push(tail); lines.close(); return res; },
159
+ (err) => { lines.close(); throw err; },
160
+ );
161
+
162
+ for await (const line of lines) {
163
+ let parsed: Record<string, unknown>;
164
+ try {
165
+ parsed = JSON.parse(line) as Record<string, unknown>;
166
+ } catch {
167
+ continue; // skip any non-JSON noise that lands on stdout
168
+ }
169
+ const sid = this.spec.extractSessionId(parsed);
170
+ if (sid) sessionId = sid;
171
+ for (const msg of this.spec.mapEvent(parsed)) {
172
+ if (msg.type === "error") sawError = true;
173
+ yield msg;
174
+ }
175
+ }
176
+
177
+ const res = await runPromise;
178
+ if (res.exitCode !== 0 && !sawError) {
179
+ const tail = (res.stderr ?? "").slice(-2000);
180
+ yield { type: "error", text: `${this.spec.kind} exited with code ${res.exitCode}${tail ? `: ${tail}` : ""}`, timestamp: now() };
181
+ return;
182
+ }
183
+ if (!sawError) yield { type: "done", sessionId: sessionId ?? "", timestamp: now() };
184
+ } catch (err) {
185
+ yield { type: "error", text: formatError(err), timestamp: now() };
186
+ }
187
+ }
188
+ }
189
+
190
+ export function createCliAgentRuntime(spec: CliAgentSpec, configModel?: string) {
191
+ return defineRuntime({
192
+ create: (sandbox, opts) => new CliAgentRunner(sandbox, opts, spec, configModel),
193
+ });
194
+ }
@@ -0,0 +1,99 @@
1
+ /**
2
+ * Amp CLI runtime — drives Sourcegraph's `amp -x --stream-json` agentic CLI
3
+ * inside the sandbox and maps its (Claude-Code-compatible) JSONL stream onto
4
+ * the AgentMessage contract. Built on the shared CLI-agent base; Amp brings its
5
+ * own loop + tools, so we only stream-parse what it prints.
6
+ *
7
+ * Auth: set `AMP_API_KEY` (`sgamp_…`) in the sandbox env via a workflow secret.
8
+ * The runtime installs the `amp` CLI (`@ampcode/cli`) on demand — no image
9
+ * baking needed; pair with `snapshots: { bootFrom: "reuse" }` to install once
10
+ * and boot from the captured snapshot on every run after. The model is chosen
11
+ * by the AMP_API_KEY account (e.g. a GPT-only token runs GPT); the runtime
12
+ * doesn't pin a model.
13
+ *
14
+ * Verified against a live `amp -x --stream-json` run: the user-message stdin
15
+ * shape, assistant content blocks, and result usage below all round-trip.
16
+ */
17
+
18
+ import type { AgentMessage } from "../index.js";
19
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
20
+ import { formatError } from "../utils/errors.js";
21
+
22
+ function now(): string { return new Date().toISOString(); }
23
+
24
+ /** Block array off either `{ message: { content } }` (Claude shape) or a
25
+ * top-level `{ content }`, whichever the stream uses. */
26
+ function blocks(msg: Record<string, unknown>): Array<Record<string, unknown>> {
27
+ const inner = (msg.message as { content?: unknown[] } | undefined)?.content
28
+ ?? (msg.content as unknown[] | undefined)
29
+ ?? [];
30
+ return inner as Array<Record<string, unknown>>;
31
+ }
32
+
33
+ const ampSpec: CliAgentSpec = {
34
+ kind: "amp",
35
+ authEnv: "AMP_API_KEY",
36
+ bin: "amp",
37
+ // Global npm install; symlink onto PATH only if the global bin dir isn't
38
+ // already there (so a non-login `sh -c` can find it).
39
+ install: 'sudo npm install -g @ampcode/cli && (command -v amp >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/amp" /usr/local/bin/amp)',
40
+ // `--stream-json-input` reads JSON Lines user messages from stdin; write one.
41
+ // amp's --stream-json-input wants Claude-shaped content blocks, not a bare
42
+ // string (it rejects a string `content` with "expected array, received string").
43
+ promptPayload: (prompt) =>
44
+ JSON.stringify({ type: "user", message: { role: "user", content: [{ type: "text", text: prompt }] } }) + "\n",
45
+ buildCommand: ({ promptPath, sessionId }) => {
46
+ // Continue the prior thread by id when we have one; else start fresh.
47
+ const cont = sessionId ? `threads continue ${shellQuote(sessionId)} ` : "";
48
+ return `amp ${cont}-x --stream-json --stream-json-input < ${shellQuote(promptPath)}`;
49
+ },
50
+ // Amp stamps `session_id` (a "T-…" thread id) on every message.
51
+ extractSessionId: (p) => (typeof p.session_id === "string" ? p.session_id : undefined),
52
+ mapEvent: (p): AgentMessage[] => {
53
+ const ts = now();
54
+ if (p.type === "assistant") {
55
+ return blocks(p).flatMap((b): AgentMessage[] => {
56
+ if (b.type === "text") return [{ type: "text", text: String(b.text ?? ""), timestamp: ts }];
57
+ if (b.type === "thinking") return [{ type: "thinking", text: String(b.thinking ?? ""), timestamp: ts }];
58
+ if (b.type === "tool_use") return [{ type: "tool_use", toolName: String(b.name ?? ""), toolInput: (b.input ?? {}) as Record<string, unknown>, toolUseId: String(b.id ?? ""), timestamp: ts }];
59
+ return [];
60
+ });
61
+ }
62
+ if (p.type === "user") {
63
+ return blocks(p).flatMap((b): AgentMessage[] =>
64
+ b.type === "tool_result"
65
+ ? [{ type: "tool_result", toolUseId: String(b.tool_use_id ?? ""), output: typeof b.content === "string" ? b.content : JSON.stringify(b.content ?? ""), isError: Boolean(b.is_error), timestamp: ts }]
66
+ : []);
67
+ }
68
+ if (p.type === "result") {
69
+ const out: AgentMessage[] = [];
70
+ const u = p.usage as Record<string, number> | undefined;
71
+ if (u) out.push({
72
+ type: "usage",
73
+ inputTokens: u.input_tokens ?? 0,
74
+ outputTokens: u.output_tokens ?? 0,
75
+ cacheReadTokens: u.cache_read_input_tokens ?? 0,
76
+ cacheCreationTokens: u.cache_creation_input_tokens ?? 0,
77
+ durationMs: Number(p.duration_ms ?? 0),
78
+ numTurns: Number(p.num_turns ?? 0),
79
+ timestamp: ts,
80
+ });
81
+ if (p.is_error || p.subtype === "error") {
82
+ out.push({ type: "error", text: formatError(p.result ?? p.error), timestamp: ts });
83
+ }
84
+ return out;
85
+ }
86
+ return []; // "system" → session id captured by extractSessionId; init/done owned by the base
87
+ },
88
+ };
89
+
90
+ export interface AmpRuntimeConfig {
91
+ /** Amp uses its configured model; reserved for forward-compatibility. */
92
+ model?: string;
93
+ }
94
+
95
+ export function createAmpRuntime(config: AmpRuntimeConfig = {}) {
96
+ return createCliAgentRuntime(ampSpec, config.model);
97
+ }
98
+
99
+ export default createAmpRuntime();
@@ -0,0 +1,114 @@
1
+ /**
2
+ * Codex CLI runtime — drives OpenAI's `codex exec --json` agentic CLI inside
3
+ * the sandbox and maps its JSONL event stream onto the AgentMessage contract.
4
+ * Built on the shared CLI-agent base; Codex brings its own loop + tools, so we
5
+ * only stream-parse what it prints.
6
+ *
7
+ * Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
8
+ * workflow secret. The runtime installs the `codex` CLI (`@openai/codex`) on
9
+ * demand — no image baking needed; pair with `snapshots: { bootFrom: "reuse" }`
10
+ * to install once and boot from the captured snapshot on every run after.
11
+ *
12
+ * Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
13
+ * with the command_execution / reasoning / agent_message item shapes below.
14
+ */
15
+
16
+ import type { AgentMessage } from "../index.js";
17
+ import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
18
+ import { formatError } from "../utils/errors.js";
19
+
20
+ function now(): string { return new Date().toISOString(); }
21
+
22
+ const codexSpec: CliAgentSpec = {
23
+ kind: "codex",
24
+ authEnv: "CODEX_API_KEY",
25
+ bin: "codex",
26
+ // Global npm install; symlink onto PATH only if the global bin dir isn't
27
+ // already there (so a non-login `sh -c` can find it).
28
+ install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
29
+ // Codex reads the prompt from stdin when invoked as `codex exec ... -`.
30
+ promptPayload: (prompt) => prompt,
31
+ buildCommand: ({ promptPath, sessionId, model, cwd }) => {
32
+ const flags = [
33
+ "--json",
34
+ "--skip-git-repo-check",
35
+ // agent-compose already runs us inside an isolated sandbox VM, so codex
36
+ // must not try to nest its own seccomp/landlock sandbox or block on
37
+ // approvals (non-interactive). codex docs: this flag is "intended solely
38
+ // for running in environments that are externally sandboxed".
39
+ "--dangerously-bypass-approvals-and-sandbox",
40
+ ...(model ? ["-m", shellQuote(model)] : []),
41
+ ...(cwd ? ["-C", shellQuote(cwd)] : []),
42
+ ].join(" ");
43
+ // Fresh turn: `codex exec <flags> - < prompt`. Continue a thread:
44
+ // `codex exec resume <id> <flags> - < prompt`. (`-` = read prompt from stdin.)
45
+ const exec = sessionId
46
+ ? `codex exec resume ${shellQuote(sessionId)} ${flags}`
47
+ : `codex exec ${flags}`;
48
+ return `${exec} - < ${shellQuote(promptPath)}`;
49
+ },
50
+ extractSessionId: (p) =>
51
+ p.type === "thread.started" && typeof p.thread_id === "string" ? p.thread_id : undefined,
52
+ mapEvent: (p): AgentMessage[] => {
53
+ const ts = now();
54
+ switch (p.type) {
55
+ case "item.started":
56
+ case "item.completed": {
57
+ const item = p.item as Record<string, unknown> | undefined;
58
+ if (!item) return [];
59
+ const itype = String(item.type ?? "");
60
+ // Text + reasoning land on completion (started carries no final text).
61
+ if (itype === "agent_message") {
62
+ return p.type === "item.completed" ? [{ type: "text", text: String(item.text ?? ""), timestamp: ts }] : [];
63
+ }
64
+ if (itype === "reasoning") {
65
+ return p.type === "item.completed" ? [{ type: "thinking", text: String(item.text ?? ""), timestamp: ts }] : [];
66
+ }
67
+ // Command execution: started → tool_use, completed → tool_result.
68
+ if (itype === "command_execution") {
69
+ const id = String(item.id ?? "");
70
+ if (p.type === "item.started") {
71
+ return [{ type: "tool_use", toolName: "shell", toolInput: { command: String(item.command ?? "") }, toolUseId: id, timestamp: ts }];
72
+ }
73
+ const failed = item.status === "failed" || (typeof item.exit_code === "number" && item.exit_code !== 0);
74
+ return [{ type: "tool_result", toolUseId: id, output: String(item.aggregated_output ?? item.output ?? ""), isError: failed, timestamp: ts }];
75
+ }
76
+ // file_change / mcp_tool_call / web_search / todo: surface once, on completion.
77
+ if (p.type === "item.completed") {
78
+ return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
79
+ }
80
+ return [];
81
+ }
82
+ case "turn.completed": {
83
+ const u = p.usage as Record<string, number> | undefined;
84
+ if (!u) return [];
85
+ return [{
86
+ type: "usage",
87
+ inputTokens: u.input_tokens ?? 0,
88
+ outputTokens: u.output_tokens ?? 0,
89
+ cacheReadTokens: u.cached_input_tokens ?? 0,
90
+ cacheCreationTokens: 0,
91
+ durationMs: 0,
92
+ numTurns: 1,
93
+ timestamp: ts,
94
+ }];
95
+ }
96
+ case "turn.failed":
97
+ case "error":
98
+ return [{ type: "error", text: formatError(p.error ?? p.message ?? p), timestamp: ts }];
99
+ default:
100
+ return [];
101
+ }
102
+ },
103
+ };
104
+
105
+ export interface CodexRuntimeConfig {
106
+ /** Codex model id (`-m`). Omit to use the codex CLI's configured default. */
107
+ model?: string;
108
+ }
109
+
110
+ export function createCodexRuntime(config: CodexRuntimeConfig = {}) {
111
+ return createCliAgentRuntime(codexSpec, config.model);
112
+ }
113
+
114
+ export default createCodexRuntime();
@@ -3,7 +3,7 @@
3
3
  * owned coding tools over SandboxProvider.
4
4
  */
5
5
 
6
- import { streamText, stepCountIs, tool, type LanguageModel } from "ai";
6
+ import { streamText, stepCountIs, tool, gateway, type LanguageModel } from "ai";
7
7
  import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
8
8
  import { defineRuntime } from "../types/runtime.js";
9
9
  import { codingTools, type CodingTool } from "../tools/index.js";
@@ -15,12 +15,27 @@ import { formatError } from "../utils/errors.js";
15
15
  type AiToolSet = Record<string, ReturnType<typeof tool<Record<string, unknown>, string>>>;
16
16
 
17
17
  export interface VercelRuntimeConfig {
18
- /** Vercel AI SDK language model (e.g. openai("gpt-5"), anthropic("claude-sonnet-4-5")). */
18
+ /** The model to drive — this is the only thing that varies per provider;
19
+ * there is no per-provider runtime. `LanguageModel` accepts every provider:
20
+ * - a gateway model-id string routed via the Vercel AI Gateway (set
21
+ * `AI_GATEWAY_API_KEY`; no provider package needed), e.g. "openai/gpt-5",
22
+ * "google/gemini-2.5-pro", "xai/grok-4", "deepseek/deepseek-chat",
23
+ * "mistral/mistral-large-latest", "anthropic/claude-sonnet-4-5";
24
+ * - or a `LanguageModel` object from a provider package (add the dep + set
25
+ * its API-key env), e.g. `openai("gpt-5")`, `google("gemini-2.5-pro")`. */
19
26
  model: LanguageModel;
20
27
  /** Optional system prompt prepended to every model call. */
21
28
  system?: string;
22
29
  /** Override/extend the default coding tools. Defaults: Read, Write, Edit, Bash. */
23
30
  tools?: readonly CodingTool[];
31
+ /** Short runtime self-id surfaced on `agent.spawned` so the dashboard can
32
+ * show a per-agent runtime icon (e.g. "openai", "gemini"). Provider presets
33
+ * set this; bare `createVercelRuntime` callers can leave it unset. */
34
+ kind?: string;
35
+ /** Display model id surfaced on `agent.spawned` (the Agent tab labels which
36
+ * model each agent ran). `model` above is the AI SDK LanguageModel object;
37
+ * this is its human-readable id string. */
38
+ modelId?: string;
24
39
  }
25
40
 
26
41
  function now(): string { return new Date().toISOString(); }
@@ -71,6 +86,11 @@ function toAgentMessages(part: Record<string, unknown>): AgentMessage[] {
71
86
 
72
87
  export class VercelRunner implements ModelExecutionContract {
73
88
  supportsToolCallProcessor = true;
89
+ /** Surfaced on `agent.spawned` for the dashboard's per-agent runtime icon +
90
+ * model label. Set from the (provider preset's) config; undefined for a
91
+ * bare `createVercelRuntime` that didn't label itself. */
92
+ readonly kind?: string;
93
+ readonly model?: string;
74
94
  private readonly tools: readonly CodingTool[];
75
95
  private readonly messages: unknown[] = [];
76
96
 
@@ -80,6 +100,8 @@ export class VercelRunner implements ModelExecutionContract {
80
100
  private readonly config: VercelRuntimeConfig,
81
101
  ) {
82
102
  this.tools = config.tools ?? codingTools;
103
+ this.kind = config.kind;
104
+ this.model = config.modelId;
83
105
  }
84
106
 
85
107
  async gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult> {
@@ -204,3 +226,31 @@ export function createVercelRuntime(config: VercelRuntimeConfig) {
204
226
  create: (sandbox, opts) => new VercelRunner(sandbox, opts, config),
205
227
  });
206
228
  }
229
+
230
+ /** One model offered by the Vercel AI Gateway. Pass `id` straight to
231
+ * `createVercelRuntime({ model: id })`. */
232
+ export interface VercelRuntimeModel {
233
+ /** Gateway model id, e.g. "openai/gpt-5". Usable directly as the runtime model. */
234
+ id: string;
235
+ /** Human-readable display name. */
236
+ name: string;
237
+ }
238
+
239
+ /**
240
+ * List the models the Vercel runtime accepts as a gateway model-id string — the
241
+ * LIVE Vercel AI Gateway catalog, so it never goes stale. This is the canonical
242
+ * answer to "what models can I pass to `createVercelRuntime`?" for the string
243
+ * form (`createVercelRuntime({ model: "openai/gpt-5" })`).
244
+ *
245
+ * Requires `AI_GATEWAY_API_KEY`. The other form — a `LanguageModel` object from
246
+ * an `@ai-sdk/<provider>` package — supports whatever that provider package
247
+ * does (see its docs); there's no single cross-form list because the runtime is
248
+ * model-agnostic. Browse the catalog in a UI at https://vercel.com/ai-gateway/models.
249
+ */
250
+ export async function listVercelRuntimeModels(): Promise<VercelRuntimeModel[]> {
251
+ const { models } = await gateway.getAvailableModels();
252
+ return models
253
+ .filter((m) => m.modelType == null || m.modelType === "language")
254
+ .map((m) => ({ id: m.id, name: m.name }))
255
+ .sort((a, b) => a.id.localeCompare(b.id));
256
+ }
@@ -25,23 +25,32 @@ import type { Processor } from "../processors/processor.js";
25
25
  * in the separate `postRunHooks` array on `WorkflowMetadata`. */
26
26
  export type WorkflowMemoryConfig = boolean;
27
27
 
28
- /** Where a run boots from. The snapshot id is the unit of identity —
29
- * each captured snapshot already records the workflow + version it
30
- * came from on the snapshot row, so there's no separate "latest of
31
- * workflow X" resolution at dispatch time. Operators pick a snapshot
32
- * from the dashboard snapshot list (or `agentc snapshot list`) and
33
- * paste the id here.
34
- *
35
- * Omit `bootFrom` entirely to boot a fresh base sandbox. */
28
+ /** Boot from a specific captured snapshot, addressed by its id. Operators
29
+ * pick one from the dashboard snapshot list (or `agentc snapshot list`) and
30
+ * paste the id here. */
36
31
  export type BootSnapshot = { snapshotId: string };
37
32
 
33
+ /** Boot from this workflow's OWN most recent snapshot, scoped to its content
34
+ * hash. The first run — and the first after a re-register changes the source
35
+ * — finds none and boots a fresh base sandbox; `"reuse"` also implies
36
+ * `saveLatest`, so that run captures a snapshot and every run after it boots
37
+ * from it. This lets a runtime install its tooling once (e.g. a CLI-agent
38
+ * runtime `npm i -g`'ing its CLI) and skip the install on every later run,
39
+ * with no hardcoded snapshot id to manage. Re-registering with changed
40
+ * source rolls the content hash, which transparently invalidates the cache
41
+ * and re-installs on the next run. */
42
+ export type ReuseSnapshot = "reuse";
43
+
38
44
  /** Snapshot configuration — boot source plus capture knobs. One object
39
45
  * per workflow / per invocation; collapsing boot + capture under a
40
46
  * single key reads as "all snapshot config lives here." */
41
47
  export interface SnapshotConfig {
42
- /** Where the runner restores from at run start. Structured (workflow
43
- * ref or snapshot id) so the intent is explicit at the call site. */
44
- bootFrom?: BootSnapshot;
48
+ /** Where the runner restores from at run start:
49
+ * - `{ snapshotId }` — a specific captured snapshot.
50
+ * - `"reuse"` — this workflow's own latest snapshot (content-hash scoped);
51
+ * fresh on the first run / after a re-register. Implies `saveLatest`.
52
+ * - omitted — a fresh base sandbox. */
53
+ bootFrom?: BootSnapshot | ReuseSnapshot;
45
54
  /** Capture the sandbox state on terminal success. The latest pointer
46
55
  * on `workflow_runs.vercel_snapshot_id` always tracks the most
47
56
  * recent capture; without `retainSteps`, prior captures are deleted
@@ -37,8 +37,8 @@ export interface AgentEventSink {
37
37
  emit(event: AgentLifecycleEvent): void | Promise<void>;
38
38
  }
39
39
 
40
- import type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema } from "./workflow-metadata.js";
41
- export type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema };
40
+ import type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema } from "./workflow-metadata.js";
41
+ export type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema };
42
42
 
43
43
  /** Turn/iteration budget for `agent(opts)`. Re-exported here so authors
44
44
  * can type per-invoke budget overrides they pass as workflow input. */