@agent-compose/sdk 0.5.2 → 0.5.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +11 -4
- package/dist/index.js +267 -1
- package/dist/runtimes/_cli-agent.d.ts +89 -0
- package/dist/runtimes/amp.d.ts +23 -0
- package/dist/runtimes/cli-agent.test.d.ts +9 -0
- package/dist/runtimes/codex.d.ts +21 -0
- package/dist/runtimes/openai-desktop.js +262 -1
- package/dist/runtimes/vercel.d.ts +41 -1
- package/dist/runtimes/vercel.js +10 -1
- package/dist/types/workflow-metadata.d.ts +19 -11
- package/dist/types/workflow.d.ts +2 -2
- package/package.json +1 -1
- package/src/index.ts +21 -2
- package/src/runtimes/_cli-agent.ts +194 -0
- package/src/runtimes/amp.ts +99 -0
- package/src/runtimes/codex.ts +114 -0
- package/src/runtimes/vercel.ts +52 -2
- package/src/types/workflow-metadata.ts +20 -11
- package/src/types/workflow.ts +2 -2
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* CLI-agent runtime base — drive an external agentic coding CLI inside the
|
|
3
|
+
* sandbox and map its JSON-Lines (JSONL) stream onto the `AgentMessage`
|
|
4
|
+
* contract.
|
|
5
|
+
*
|
|
6
|
+
* This is a different *mechanism* from the other runtimes: `claudeRuntime`
|
|
7
|
+
* drives the Anthropic Agent SDK and `vercelRuntime` drives the Vercel AI SDK,
|
|
8
|
+
* but a CLI-agent runtime spawns the provider's own CLI (`codex exec --json`,
|
|
9
|
+
* `amp -x --stream-json`) — the CLI brings its own agent loop + tools, and we
|
|
10
|
+
* only stream-parse the events it prints. Codex and Amp are the first two; this
|
|
11
|
+
* base is the shared machinery (spawn → line-buffer stdout → JSONL parse →
|
|
12
|
+
* AsyncQueue bridge → init/usage/done/error lifecycle), parameterised per-CLI
|
|
13
|
+
* by a `CliAgentSpec`.
|
|
14
|
+
*
|
|
15
|
+
* Provisioning (per spec): the runtime installs the provider CLI on demand —
|
|
16
|
+
* `command -v <bin>` before the first turn, falling back to the spec's
|
|
17
|
+
* `install` command when it's absent — so it works on a bare sandbox with no
|
|
18
|
+
* manual setup or hardcoded snapshot id. Pair it with
|
|
19
|
+
* `snapshots: { bootFrom: "reuse" }` on the workflow and the install happens
|
|
20
|
+
* exactly once: the first run installs the CLI and captures a snapshot, and
|
|
21
|
+
* every run after boots from that snapshot with the CLI already present (the
|
|
22
|
+
* probe short-circuits). The provider's API key must be in the sandbox env
|
|
23
|
+
* (see each spec's `authEnv`); `commands.run` inherits the sandbox env, so
|
|
24
|
+
* values set via `agentc secrets set` are visible to the CLI.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider } from "../index.js";
|
|
28
|
+
import { defineRuntime } from "../types/runtime.js";
|
|
29
|
+
import { AsyncQueue } from "../agent/async-queue.js";
|
|
30
|
+
import { formatError } from "../utils/errors.js";
|
|
31
|
+
|
|
32
|
+
function now(): string { return new Date().toISOString(); }
|
|
33
|
+
|
|
34
|
+
/** Single-quote a value for safe interpolation into a `sh -c` command line. */
|
|
35
|
+
export function shellQuote(value: string): string {
|
|
36
|
+
return `'${value.replace(/'/g, `'\\''`)}'`;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** Per-CLI behaviour. The base owns the lifecycle (init/done/error) and the
|
|
40
|
+
* transport (spawn + JSONL parse); a spec owns the CLI-specific bits. */
|
|
41
|
+
export interface CliAgentSpec {
|
|
42
|
+
/** Runtime self-id surfaced on `agent.spawned` (dashboard runtime icon). */
|
|
43
|
+
kind: string;
|
|
44
|
+
/** Env var the CLI reads for auth — documentation only (must be set in the
|
|
45
|
+
* sandbox env via a workflow secret). */
|
|
46
|
+
authEnv: string;
|
|
47
|
+
/** Default model id when none is configured; omit to let the CLI choose. */
|
|
48
|
+
defaultModel?: string;
|
|
49
|
+
/** Binary the runtime spawns (`codex`, `amp`). Probed with `command -v`
|
|
50
|
+
* before the first turn; if absent, `install` provisions it. */
|
|
51
|
+
bin: string;
|
|
52
|
+
/** Shell command that installs `bin` when it's missing from the sandbox.
|
|
53
|
+
* Runs at most once per runner, and only when the probe fails — so booting
|
|
54
|
+
* from a snapshot that already has the CLI (the steady state under
|
|
55
|
+
* `snapshots: { bootFrom: "reuse" }`) skips it. Must leave `bin` resolvable
|
|
56
|
+
* on a non-login shell's PATH (the runtime spawns via `sh -c`). */
|
|
57
|
+
install: string;
|
|
58
|
+
/** Serialise the user prompt into the bytes written to the prompt file —
|
|
59
|
+
* plain text for a CLI that reads the prompt from stdin (codex `-`), or a
|
|
60
|
+
* JSONL user message for a `--stream-json-input` CLI (amp). */
|
|
61
|
+
promptPayload(prompt: string): string;
|
|
62
|
+
/** Build the one-shot shell command for a turn. `promptPath` is a file in the
|
|
63
|
+
* sandbox holding `promptPayload(prompt)`; `sessionId` continues a thread. */
|
|
64
|
+
buildCommand(args: { promptPath: string; sessionId?: string; model?: string; cwd?: string }): string;
|
|
65
|
+
/** Map one parsed JSONL stdout event to `AgentMessage`s. The base emits
|
|
66
|
+
* `init`/`done`/`error` lifecycle itself, so a spec maps only content +
|
|
67
|
+
* usage (text / thinking / tool_use / tool_result / usage). */
|
|
68
|
+
mapEvent(parsed: Record<string, unknown>): AgentMessage[];
|
|
69
|
+
/** Pull a session/thread id out of a parsed event so the next turn can
|
|
70
|
+
* resume it (codex `thread.started.thread_id`, amp `session_id`). */
|
|
71
|
+
extractSessionId(parsed: Record<string, unknown>): string | undefined;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export class CliAgentRunner implements ModelExecutionContract {
|
|
75
|
+
readonly kind: string;
|
|
76
|
+
|
|
77
|
+
constructor(
|
|
78
|
+
private readonly sandbox: SandboxProvider,
|
|
79
|
+
private readonly options: RuntimeOptions,
|
|
80
|
+
private readonly spec: CliAgentSpec,
|
|
81
|
+
private readonly configModel?: string,
|
|
82
|
+
) {
|
|
83
|
+
this.kind = spec.kind;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
get model(): string | undefined {
|
|
87
|
+
return this.configModel ?? this.options.model ?? this.spec.defaultModel;
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
private installed = false;
|
|
91
|
+
|
|
92
|
+
/** Provision the CLI on demand. A no-op once `bin` is on PATH — the steady
|
|
93
|
+
* state under `snapshots: { bootFrom: "reuse" }`, where the first run's
|
|
94
|
+
* install is baked into the snapshot every later run boots from. So the
|
|
95
|
+
* install command runs exactly once: on the first run of a content hash. */
|
|
96
|
+
private async ensureInstalled(): Promise<void> {
|
|
97
|
+
if (this.installed) return;
|
|
98
|
+
const probe = await this.sandbox.commands.run(`command -v ${this.spec.bin}`);
|
|
99
|
+
if (probe.exitCode === 0) { this.installed = true; return; }
|
|
100
|
+
const res = await this.sandbox.commands.run(this.spec.install, { timeoutMs: 300_000 });
|
|
101
|
+
if (res.exitCode !== 0) {
|
|
102
|
+
throw new Error(`failed to install ${this.spec.bin}: ${(res.stderr || res.stdout || "").slice(-500)}`);
|
|
103
|
+
}
|
|
104
|
+
this.installed = true;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// No captureCheckpoint/restoreCheckpoint: the CLI persists its thread/rollout
|
|
108
|
+
// on the sandbox filesystem (which round-trips through the pause snapshot),
|
|
109
|
+
// and the loop already carries the session id we emit on init/done and pass
|
|
110
|
+
// back as `sessionId` for resume — same model as claudeRuntime.
|
|
111
|
+
|
|
112
|
+
async *sendMessage(opts: {
|
|
113
|
+
prompt: string;
|
|
114
|
+
sessionId?: string;
|
|
115
|
+
iteration?: number;
|
|
116
|
+
signal?: AbortSignal;
|
|
117
|
+
}): AsyncGenerator<AgentMessage> {
|
|
118
|
+
yield { type: "init", sessionId: opts.sessionId ?? "", timestamp: now() };
|
|
119
|
+
|
|
120
|
+
let sessionId = opts.sessionId;
|
|
121
|
+
let sawError = false;
|
|
122
|
+
try {
|
|
123
|
+
// Provision the CLI before the first turn (no-op when it's already
|
|
124
|
+
// present, e.g. booting from a "reuse" snapshot). A failure here surfaces
|
|
125
|
+
// as an `error` AgentMessage via the catch below.
|
|
126
|
+
await this.ensureInstalled();
|
|
127
|
+
|
|
128
|
+
const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
|
|
129
|
+
await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
|
|
130
|
+
const cmd = this.spec.buildCommand({
|
|
131
|
+
promptPath,
|
|
132
|
+
sessionId: opts.sessionId,
|
|
133
|
+
model: this.model,
|
|
134
|
+
cwd: this.options.cwd,
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
// Bridge the streaming stdout callback into an async-iterable of complete
|
|
138
|
+
// JSONL lines. `onStdout` chunks aren't line-aligned, so buffer + split.
|
|
139
|
+
const lines = new AsyncQueue<string>();
|
|
140
|
+
let buf = "";
|
|
141
|
+
const onStdout = (data: string) => {
|
|
142
|
+
buf += data;
|
|
143
|
+
let nl: number;
|
|
144
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
145
|
+
const line = buf.slice(0, nl).trim();
|
|
146
|
+
buf = buf.slice(nl + 1);
|
|
147
|
+
if (line) lines.push(line);
|
|
148
|
+
}
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
// `commands.run` resolves when the process exits. Kick it off (don't await
|
|
152
|
+
// yet); flush the trailing buffer + close the queue on completion so the
|
|
153
|
+
// for-await below drains and we can read the exit code.
|
|
154
|
+
const runPromise = this.sandbox.commands.run(cmd, {
|
|
155
|
+
...(this.options.cwd ? { cwd: this.options.cwd } : {}),
|
|
156
|
+
onStdout,
|
|
157
|
+
}).then(
|
|
158
|
+
(res) => { const tail = buf.trim(); if (tail) lines.push(tail); lines.close(); return res; },
|
|
159
|
+
(err) => { lines.close(); throw err; },
|
|
160
|
+
);
|
|
161
|
+
|
|
162
|
+
for await (const line of lines) {
|
|
163
|
+
let parsed: Record<string, unknown>;
|
|
164
|
+
try {
|
|
165
|
+
parsed = JSON.parse(line) as Record<string, unknown>;
|
|
166
|
+
} catch {
|
|
167
|
+
continue; // skip any non-JSON noise that lands on stdout
|
|
168
|
+
}
|
|
169
|
+
const sid = this.spec.extractSessionId(parsed);
|
|
170
|
+
if (sid) sessionId = sid;
|
|
171
|
+
for (const msg of this.spec.mapEvent(parsed)) {
|
|
172
|
+
if (msg.type === "error") sawError = true;
|
|
173
|
+
yield msg;
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
const res = await runPromise;
|
|
178
|
+
if (res.exitCode !== 0 && !sawError) {
|
|
179
|
+
const tail = (res.stderr ?? "").slice(-2000);
|
|
180
|
+
yield { type: "error", text: `${this.spec.kind} exited with code ${res.exitCode}${tail ? `: ${tail}` : ""}`, timestamp: now() };
|
|
181
|
+
return;
|
|
182
|
+
}
|
|
183
|
+
if (!sawError) yield { type: "done", sessionId: sessionId ?? "", timestamp: now() };
|
|
184
|
+
} catch (err) {
|
|
185
|
+
yield { type: "error", text: formatError(err), timestamp: now() };
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
export function createCliAgentRuntime(spec: CliAgentSpec, configModel?: string) {
|
|
191
|
+
return defineRuntime({
|
|
192
|
+
create: (sandbox, opts) => new CliAgentRunner(sandbox, opts, spec, configModel),
|
|
193
|
+
});
|
|
194
|
+
}
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Amp CLI runtime — drives Sourcegraph's `amp -x --stream-json` agentic CLI
|
|
3
|
+
* inside the sandbox and maps its (Claude-Code-compatible) JSONL stream onto
|
|
4
|
+
* the AgentMessage contract. Built on the shared CLI-agent base; Amp brings its
|
|
5
|
+
* own loop + tools, so we only stream-parse what it prints.
|
|
6
|
+
*
|
|
7
|
+
* Auth: set `AMP_API_KEY` (`sgamp_…`) in the sandbox env via a workflow secret.
|
|
8
|
+
* The runtime installs the `amp` CLI (`@ampcode/cli`) on demand — no image
|
|
9
|
+
* baking needed; pair with `snapshots: { bootFrom: "reuse" }` to install once
|
|
10
|
+
* and boot from the captured snapshot on every run after. The model is chosen
|
|
11
|
+
* by the AMP_API_KEY account (e.g. a GPT-only token runs GPT); the runtime
|
|
12
|
+
* doesn't pin a model.
|
|
13
|
+
*
|
|
14
|
+
* Verified against a live `amp -x --stream-json` run: the user-message stdin
|
|
15
|
+
* shape, assistant content blocks, and result usage below all round-trip.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import type { AgentMessage } from "../index.js";
|
|
19
|
+
import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
|
|
20
|
+
import { formatError } from "../utils/errors.js";
|
|
21
|
+
|
|
22
|
+
function now(): string { return new Date().toISOString(); }
|
|
23
|
+
|
|
24
|
+
/** Block array off either `{ message: { content } }` (Claude shape) or a
|
|
25
|
+
* top-level `{ content }`, whichever the stream uses. */
|
|
26
|
+
function blocks(msg: Record<string, unknown>): Array<Record<string, unknown>> {
|
|
27
|
+
const inner = (msg.message as { content?: unknown[] } | undefined)?.content
|
|
28
|
+
?? (msg.content as unknown[] | undefined)
|
|
29
|
+
?? [];
|
|
30
|
+
return inner as Array<Record<string, unknown>>;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
const ampSpec: CliAgentSpec = {
|
|
34
|
+
kind: "amp",
|
|
35
|
+
authEnv: "AMP_API_KEY",
|
|
36
|
+
bin: "amp",
|
|
37
|
+
// Global npm install; symlink onto PATH only if the global bin dir isn't
|
|
38
|
+
// already there (so a non-login `sh -c` can find it).
|
|
39
|
+
install: 'sudo npm install -g @ampcode/cli && (command -v amp >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/amp" /usr/local/bin/amp)',
|
|
40
|
+
// `--stream-json-input` reads JSON Lines user messages from stdin; write one.
|
|
41
|
+
// amp's --stream-json-input wants Claude-shaped content blocks, not a bare
|
|
42
|
+
// string (it rejects a string `content` with "expected array, received string").
|
|
43
|
+
promptPayload: (prompt) =>
|
|
44
|
+
JSON.stringify({ type: "user", message: { role: "user", content: [{ type: "text", text: prompt }] } }) + "\n",
|
|
45
|
+
buildCommand: ({ promptPath, sessionId }) => {
|
|
46
|
+
// Continue the prior thread by id when we have one; else start fresh.
|
|
47
|
+
const cont = sessionId ? `threads continue ${shellQuote(sessionId)} ` : "";
|
|
48
|
+
return `amp ${cont}-x --stream-json --stream-json-input < ${shellQuote(promptPath)}`;
|
|
49
|
+
},
|
|
50
|
+
// Amp stamps `session_id` (a "T-…" thread id) on every message.
|
|
51
|
+
extractSessionId: (p) => (typeof p.session_id === "string" ? p.session_id : undefined),
|
|
52
|
+
mapEvent: (p): AgentMessage[] => {
|
|
53
|
+
const ts = now();
|
|
54
|
+
if (p.type === "assistant") {
|
|
55
|
+
return blocks(p).flatMap((b): AgentMessage[] => {
|
|
56
|
+
if (b.type === "text") return [{ type: "text", text: String(b.text ?? ""), timestamp: ts }];
|
|
57
|
+
if (b.type === "thinking") return [{ type: "thinking", text: String(b.thinking ?? ""), timestamp: ts }];
|
|
58
|
+
if (b.type === "tool_use") return [{ type: "tool_use", toolName: String(b.name ?? ""), toolInput: (b.input ?? {}) as Record<string, unknown>, toolUseId: String(b.id ?? ""), timestamp: ts }];
|
|
59
|
+
return [];
|
|
60
|
+
});
|
|
61
|
+
}
|
|
62
|
+
if (p.type === "user") {
|
|
63
|
+
return blocks(p).flatMap((b): AgentMessage[] =>
|
|
64
|
+
b.type === "tool_result"
|
|
65
|
+
? [{ type: "tool_result", toolUseId: String(b.tool_use_id ?? ""), output: typeof b.content === "string" ? b.content : JSON.stringify(b.content ?? ""), isError: Boolean(b.is_error), timestamp: ts }]
|
|
66
|
+
: []);
|
|
67
|
+
}
|
|
68
|
+
if (p.type === "result") {
|
|
69
|
+
const out: AgentMessage[] = [];
|
|
70
|
+
const u = p.usage as Record<string, number> | undefined;
|
|
71
|
+
if (u) out.push({
|
|
72
|
+
type: "usage",
|
|
73
|
+
inputTokens: u.input_tokens ?? 0,
|
|
74
|
+
outputTokens: u.output_tokens ?? 0,
|
|
75
|
+
cacheReadTokens: u.cache_read_input_tokens ?? 0,
|
|
76
|
+
cacheCreationTokens: u.cache_creation_input_tokens ?? 0,
|
|
77
|
+
durationMs: Number(p.duration_ms ?? 0),
|
|
78
|
+
numTurns: Number(p.num_turns ?? 0),
|
|
79
|
+
timestamp: ts,
|
|
80
|
+
});
|
|
81
|
+
if (p.is_error || p.subtype === "error") {
|
|
82
|
+
out.push({ type: "error", text: formatError(p.result ?? p.error), timestamp: ts });
|
|
83
|
+
}
|
|
84
|
+
return out;
|
|
85
|
+
}
|
|
86
|
+
return []; // "system" → session id captured by extractSessionId; init/done owned by the base
|
|
87
|
+
},
|
|
88
|
+
};
|
|
89
|
+
|
|
90
|
+
export interface AmpRuntimeConfig {
|
|
91
|
+
/** Amp uses its configured model; reserved for forward-compatibility. */
|
|
92
|
+
model?: string;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
export function createAmpRuntime(config: AmpRuntimeConfig = {}) {
|
|
96
|
+
return createCliAgentRuntime(ampSpec, config.model);
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
export default createAmpRuntime();
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Codex CLI runtime — drives OpenAI's `codex exec --json` agentic CLI inside
|
|
3
|
+
* the sandbox and maps its JSONL event stream onto the AgentMessage contract.
|
|
4
|
+
* Built on the shared CLI-agent base; Codex brings its own loop + tools, so we
|
|
5
|
+
* only stream-parse what it prints.
|
|
6
|
+
*
|
|
7
|
+
* Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
|
|
8
|
+
* workflow secret. The runtime installs the `codex` CLI (`@openai/codex`) on
|
|
9
|
+
* demand — no image baking needed; pair with `snapshots: { bootFrom: "reuse" }`
|
|
10
|
+
* to install once and boot from the captured snapshot on every run after.
|
|
11
|
+
*
|
|
12
|
+
* Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
|
|
13
|
+
* with the command_execution / reasoning / agent_message item shapes below.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
import type { AgentMessage } from "../index.js";
|
|
17
|
+
import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
|
|
18
|
+
import { formatError } from "../utils/errors.js";
|
|
19
|
+
|
|
20
|
+
function now(): string { return new Date().toISOString(); }
|
|
21
|
+
|
|
22
|
+
const codexSpec: CliAgentSpec = {
|
|
23
|
+
kind: "codex",
|
|
24
|
+
authEnv: "CODEX_API_KEY",
|
|
25
|
+
bin: "codex",
|
|
26
|
+
// Global npm install; symlink onto PATH only if the global bin dir isn't
|
|
27
|
+
// already there (so a non-login `sh -c` can find it).
|
|
28
|
+
install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
|
|
29
|
+
// Codex reads the prompt from stdin when invoked as `codex exec ... -`.
|
|
30
|
+
promptPayload: (prompt) => prompt,
|
|
31
|
+
buildCommand: ({ promptPath, sessionId, model, cwd }) => {
|
|
32
|
+
const flags = [
|
|
33
|
+
"--json",
|
|
34
|
+
"--skip-git-repo-check",
|
|
35
|
+
// agent-compose already runs us inside an isolated sandbox VM, so codex
|
|
36
|
+
// must not try to nest its own seccomp/landlock sandbox or block on
|
|
37
|
+
// approvals (non-interactive). codex docs: this flag is "intended solely
|
|
38
|
+
// for running in environments that are externally sandboxed".
|
|
39
|
+
"--dangerously-bypass-approvals-and-sandbox",
|
|
40
|
+
...(model ? ["-m", shellQuote(model)] : []),
|
|
41
|
+
...(cwd ? ["-C", shellQuote(cwd)] : []),
|
|
42
|
+
].join(" ");
|
|
43
|
+
// Fresh turn: `codex exec <flags> - < prompt`. Continue a thread:
|
|
44
|
+
// `codex exec resume <id> <flags> - < prompt`. (`-` = read prompt from stdin.)
|
|
45
|
+
const exec = sessionId
|
|
46
|
+
? `codex exec resume ${shellQuote(sessionId)} ${flags}`
|
|
47
|
+
: `codex exec ${flags}`;
|
|
48
|
+
return `${exec} - < ${shellQuote(promptPath)}`;
|
|
49
|
+
},
|
|
50
|
+
extractSessionId: (p) =>
|
|
51
|
+
p.type === "thread.started" && typeof p.thread_id === "string" ? p.thread_id : undefined,
|
|
52
|
+
mapEvent: (p): AgentMessage[] => {
|
|
53
|
+
const ts = now();
|
|
54
|
+
switch (p.type) {
|
|
55
|
+
case "item.started":
|
|
56
|
+
case "item.completed": {
|
|
57
|
+
const item = p.item as Record<string, unknown> | undefined;
|
|
58
|
+
if (!item) return [];
|
|
59
|
+
const itype = String(item.type ?? "");
|
|
60
|
+
// Text + reasoning land on completion (started carries no final text).
|
|
61
|
+
if (itype === "agent_message") {
|
|
62
|
+
return p.type === "item.completed" ? [{ type: "text", text: String(item.text ?? ""), timestamp: ts }] : [];
|
|
63
|
+
}
|
|
64
|
+
if (itype === "reasoning") {
|
|
65
|
+
return p.type === "item.completed" ? [{ type: "thinking", text: String(item.text ?? ""), timestamp: ts }] : [];
|
|
66
|
+
}
|
|
67
|
+
// Command execution: started → tool_use, completed → tool_result.
|
|
68
|
+
if (itype === "command_execution") {
|
|
69
|
+
const id = String(item.id ?? "");
|
|
70
|
+
if (p.type === "item.started") {
|
|
71
|
+
return [{ type: "tool_use", toolName: "shell", toolInput: { command: String(item.command ?? "") }, toolUseId: id, timestamp: ts }];
|
|
72
|
+
}
|
|
73
|
+
const failed = item.status === "failed" || (typeof item.exit_code === "number" && item.exit_code !== 0);
|
|
74
|
+
return [{ type: "tool_result", toolUseId: id, output: String(item.aggregated_output ?? item.output ?? ""), isError: failed, timestamp: ts }];
|
|
75
|
+
}
|
|
76
|
+
// file_change / mcp_tool_call / web_search / todo: surface once, on completion.
|
|
77
|
+
if (p.type === "item.completed") {
|
|
78
|
+
return [{ type: "tool_use", toolName: itype || "item", toolInput: item, toolUseId: String(item.id ?? ""), timestamp: ts }];
|
|
79
|
+
}
|
|
80
|
+
return [];
|
|
81
|
+
}
|
|
82
|
+
case "turn.completed": {
|
|
83
|
+
const u = p.usage as Record<string, number> | undefined;
|
|
84
|
+
if (!u) return [];
|
|
85
|
+
return [{
|
|
86
|
+
type: "usage",
|
|
87
|
+
inputTokens: u.input_tokens ?? 0,
|
|
88
|
+
outputTokens: u.output_tokens ?? 0,
|
|
89
|
+
cacheReadTokens: u.cached_input_tokens ?? 0,
|
|
90
|
+
cacheCreationTokens: 0,
|
|
91
|
+
durationMs: 0,
|
|
92
|
+
numTurns: 1,
|
|
93
|
+
timestamp: ts,
|
|
94
|
+
}];
|
|
95
|
+
}
|
|
96
|
+
case "turn.failed":
|
|
97
|
+
case "error":
|
|
98
|
+
return [{ type: "error", text: formatError(p.error ?? p.message ?? p), timestamp: ts }];
|
|
99
|
+
default:
|
|
100
|
+
return [];
|
|
101
|
+
}
|
|
102
|
+
},
|
|
103
|
+
};
|
|
104
|
+
|
|
105
|
+
export interface CodexRuntimeConfig {
|
|
106
|
+
/** Codex model id (`-m`). Omit to use the codex CLI's configured default. */
|
|
107
|
+
model?: string;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
export function createCodexRuntime(config: CodexRuntimeConfig = {}) {
|
|
111
|
+
return createCliAgentRuntime(codexSpec, config.model);
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
export default createCodexRuntime();
|
package/src/runtimes/vercel.ts
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
* owned coding tools over SandboxProvider.
|
|
4
4
|
*/
|
|
5
5
|
|
|
6
|
-
import { streamText, stepCountIs, tool, type LanguageModel } from "ai";
|
|
6
|
+
import { streamText, stepCountIs, tool, gateway, type LanguageModel } from "ai";
|
|
7
7
|
import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
|
|
8
8
|
import { defineRuntime } from "../types/runtime.js";
|
|
9
9
|
import { codingTools, type CodingTool } from "../tools/index.js";
|
|
@@ -15,12 +15,27 @@ import { formatError } from "../utils/errors.js";
|
|
|
15
15
|
type AiToolSet = Record<string, ReturnType<typeof tool<Record<string, unknown>, string>>>;
|
|
16
16
|
|
|
17
17
|
export interface VercelRuntimeConfig {
|
|
18
|
-
/**
|
|
18
|
+
/** The model to drive — this is the only thing that varies per provider;
|
|
19
|
+
* there is no per-provider runtime. `LanguageModel` accepts every provider:
|
|
20
|
+
* - a gateway model-id string routed via the Vercel AI Gateway (set
|
|
21
|
+
* `AI_GATEWAY_API_KEY`; no provider package needed), e.g. "openai/gpt-5",
|
|
22
|
+
* "google/gemini-2.5-pro", "xai/grok-4", "deepseek/deepseek-chat",
|
|
23
|
+
* "mistral/mistral-large-latest", "anthropic/claude-sonnet-4-5";
|
|
24
|
+
* - or a `LanguageModel` object from a provider package (add the dep + set
|
|
25
|
+
* its API-key env), e.g. `openai("gpt-5")`, `google("gemini-2.5-pro")`. */
|
|
19
26
|
model: LanguageModel;
|
|
20
27
|
/** Optional system prompt prepended to every model call. */
|
|
21
28
|
system?: string;
|
|
22
29
|
/** Override/extend the default coding tools. Defaults: Read, Write, Edit, Bash. */
|
|
23
30
|
tools?: readonly CodingTool[];
|
|
31
|
+
/** Short runtime self-id surfaced on `agent.spawned` so the dashboard can
|
|
32
|
+
* show a per-agent runtime icon (e.g. "openai", "gemini"). Provider presets
|
|
33
|
+
* set this; bare `createVercelRuntime` callers can leave it unset. */
|
|
34
|
+
kind?: string;
|
|
35
|
+
/** Display model id surfaced on `agent.spawned` (the Agent tab labels which
|
|
36
|
+
* model each agent ran). `model` above is the AI SDK LanguageModel object;
|
|
37
|
+
* this is its human-readable id string. */
|
|
38
|
+
modelId?: string;
|
|
24
39
|
}
|
|
25
40
|
|
|
26
41
|
function now(): string { return new Date().toISOString(); }
|
|
@@ -71,6 +86,11 @@ function toAgentMessages(part: Record<string, unknown>): AgentMessage[] {
|
|
|
71
86
|
|
|
72
87
|
export class VercelRunner implements ModelExecutionContract {
|
|
73
88
|
supportsToolCallProcessor = true;
|
|
89
|
+
/** Surfaced on `agent.spawned` for the dashboard's per-agent runtime icon +
|
|
90
|
+
* model label. Set from the (provider preset's) config; undefined for a
|
|
91
|
+
* bare `createVercelRuntime` that didn't label itself. */
|
|
92
|
+
readonly kind?: string;
|
|
93
|
+
readonly model?: string;
|
|
74
94
|
private readonly tools: readonly CodingTool[];
|
|
75
95
|
private readonly messages: unknown[] = [];
|
|
76
96
|
|
|
@@ -80,6 +100,8 @@ export class VercelRunner implements ModelExecutionContract {
|
|
|
80
100
|
private readonly config: VercelRuntimeConfig,
|
|
81
101
|
) {
|
|
82
102
|
this.tools = config.tools ?? codingTools;
|
|
103
|
+
this.kind = config.kind;
|
|
104
|
+
this.model = config.modelId;
|
|
83
105
|
}
|
|
84
106
|
|
|
85
107
|
async gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult> {
|
|
@@ -204,3 +226,31 @@ export function createVercelRuntime(config: VercelRuntimeConfig) {
|
|
|
204
226
|
create: (sandbox, opts) => new VercelRunner(sandbox, opts, config),
|
|
205
227
|
});
|
|
206
228
|
}
|
|
229
|
+
|
|
230
|
+
/** One model offered by the Vercel AI Gateway. Pass `id` straight to
|
|
231
|
+
* `createVercelRuntime({ model: id })`. */
|
|
232
|
+
export interface VercelRuntimeModel {
|
|
233
|
+
/** Gateway model id, e.g. "openai/gpt-5". Usable directly as the runtime model. */
|
|
234
|
+
id: string;
|
|
235
|
+
/** Human-readable display name. */
|
|
236
|
+
name: string;
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/**
|
|
240
|
+
* List the models the Vercel runtime accepts as a gateway model-id string — the
|
|
241
|
+
* LIVE Vercel AI Gateway catalog, so it never goes stale. This is the canonical
|
|
242
|
+
* answer to "what models can I pass to `createVercelRuntime`?" for the string
|
|
243
|
+
* form (`createVercelRuntime({ model: "openai/gpt-5" })`).
|
|
244
|
+
*
|
|
245
|
+
* Requires `AI_GATEWAY_API_KEY`. The other form — a `LanguageModel` object from
|
|
246
|
+
* an `@ai-sdk/<provider>` package — supports whatever that provider package
|
|
247
|
+
* does (see its docs); there's no single cross-form list because the runtime is
|
|
248
|
+
* model-agnostic. Browse the catalog in a UI at https://vercel.com/ai-gateway/models.
|
|
249
|
+
*/
|
|
250
|
+
export async function listVercelRuntimeModels(): Promise<VercelRuntimeModel[]> {
|
|
251
|
+
const { models } = await gateway.getAvailableModels();
|
|
252
|
+
return models
|
|
253
|
+
.filter((m) => m.modelType == null || m.modelType === "language")
|
|
254
|
+
.map((m) => ({ id: m.id, name: m.name }))
|
|
255
|
+
.sort((a, b) => a.id.localeCompare(b.id));
|
|
256
|
+
}
|
|
@@ -25,23 +25,32 @@ import type { Processor } from "../processors/processor.js";
|
|
|
25
25
|
* in the separate `postRunHooks` array on `WorkflowMetadata`. */
|
|
26
26
|
export type WorkflowMemoryConfig = boolean;
|
|
27
27
|
|
|
28
|
-
/**
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
* workflow X" resolution at dispatch time. Operators pick a snapshot
|
|
32
|
-
* from the dashboard snapshot list (or `agentc snapshot list`) and
|
|
33
|
-
* paste the id here.
|
|
34
|
-
*
|
|
35
|
-
* Omit `bootFrom` entirely to boot a fresh base sandbox. */
|
|
28
|
+
/** Boot from a specific captured snapshot, addressed by its id. Operators
|
|
29
|
+
* pick one from the dashboard snapshot list (or `agentc snapshot list`) and
|
|
30
|
+
* paste the id here. */
|
|
36
31
|
export type BootSnapshot = { snapshotId: string };
|
|
37
32
|
|
|
33
|
+
/** Boot from this workflow's OWN most recent snapshot, scoped to its content
|
|
34
|
+
* hash. The first run — and the first after a re-register changes the source
|
|
35
|
+
* — finds none and boots a fresh base sandbox; `"reuse"` also implies
|
|
36
|
+
* `saveLatest`, so that run captures a snapshot and every run after it boots
|
|
37
|
+
* from it. This lets a runtime install its tooling once (e.g. a CLI-agent
|
|
38
|
+
* runtime `npm i -g`'ing its CLI) and skip the install on every later run,
|
|
39
|
+
* with no hardcoded snapshot id to manage. Re-registering with changed
|
|
40
|
+
* source rolls the content hash, which transparently invalidates the cache
|
|
41
|
+
* and re-installs on the next run. */
|
|
42
|
+
export type ReuseSnapshot = "reuse";
|
|
43
|
+
|
|
38
44
|
/** Snapshot configuration — boot source plus capture knobs. One object
|
|
39
45
|
* per workflow / per invocation; collapsing boot + capture under a
|
|
40
46
|
* single key reads as "all snapshot config lives here." */
|
|
41
47
|
export interface SnapshotConfig {
|
|
42
|
-
/** Where the runner restores from at run start
|
|
43
|
-
*
|
|
44
|
-
|
|
48
|
+
/** Where the runner restores from at run start:
|
|
49
|
+
* - `{ snapshotId }` — a specific captured snapshot.
|
|
50
|
+
* - `"reuse"` — this workflow's own latest snapshot (content-hash scoped);
|
|
51
|
+
* fresh on the first run / after a re-register. Implies `saveLatest`.
|
|
52
|
+
* - omitted — a fresh base sandbox. */
|
|
53
|
+
bootFrom?: BootSnapshot | ReuseSnapshot;
|
|
45
54
|
/** Capture the sandbox state on terminal success. The latest pointer
|
|
46
55
|
* on `workflow_runs.vercel_snapshot_id` always tracks the most
|
|
47
56
|
* recent capture; without `retainSteps`, prior captures are deleted
|
package/src/types/workflow.ts
CHANGED
|
@@ -37,8 +37,8 @@ export interface AgentEventSink {
|
|
|
37
37
|
emit(event: AgentLifecycleEvent): void | Promise<void>;
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
-
import type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema } from "./workflow-metadata.js";
|
|
41
|
-
export type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema };
|
|
40
|
+
import type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema } from "./workflow-metadata.js";
|
|
41
|
+
export type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema };
|
|
42
42
|
|
|
43
43
|
/** Turn/iteration budget for `agent(opts)`. Re-exported here so authors
|
|
44
44
|
* can type per-invoke budget overrides they pass as workflow input. */
|