@agent-compose/sdk 0.5.5 → 0.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -16,7 +16,7 @@ export type { WorkflowDefinition } from "./types/workflow.js";
16
16
  export { defineSandboxEnvironment } from "./types/sandbox-environment.js";
17
17
  export type { SandboxEnvironmentDefinition } from "./types/sandbox-environment.js";
18
18
  export type { AgentRuntime, McpServerConfig, ModelExecutionContract, ToolCallGateResult, RuntimeOptions, } from "./types/runtime.js";
19
- export type { WorkflowFn, WorkflowCtx, WorkflowRun, AgentBudget, WorkflowHooks, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema, WorkflowMemoryConfig, } from "./types/workflow.js";
19
+ export type { WorkflowFn, WorkflowCtx, WorkflowRun, AgentBudget, WorkflowHooks, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema, WorkflowMemoryConfig, } from "./types/workflow.js";
20
20
  export type { RunSnapshotEntry } from "./client.js";
21
21
  export type { WorkflowPlan, WorkflowStepPlan } from "./types/workflow-plan.js";
22
22
  export type { BaseExecutionContext, InvokeChild } from "./types/execution-context.js";
package/dist/index.js CHANGED
@@ -2400,45 +2400,61 @@ class CliAgentRunner {
2400
2400
  get model() {
2401
2401
  return this.configModel ?? this.options.model ?? this.spec.defaultModel;
2402
2402
  }
2403
+ installed = false;
2404
+ async ensureInstalled() {
2405
+ if (this.installed)
2406
+ return;
2407
+ const probe = await this.sandbox.commands.run(`command -v ${this.spec.bin}`);
2408
+ if (probe.exitCode === 0) {
2409
+ this.installed = true;
2410
+ return;
2411
+ }
2412
+ const res = await this.sandbox.commands.run(this.spec.install, { timeoutMs: 300000 });
2413
+ if (res.exitCode !== 0) {
2414
+ throw new Error(`failed to install ${this.spec.bin}: ${(res.stderr || res.stdout || "").slice(-500)}`);
2415
+ }
2416
+ this.installed = true;
2417
+ }
2403
2418
  async* sendMessage(opts) {
2404
- const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
2405
- await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
2406
- const cmd = this.spec.buildCommand({
2407
- promptPath,
2408
- sessionId: opts.sessionId,
2409
- model: this.model,
2410
- cwd: this.options.cwd
2411
- });
2412
- const lines = new AsyncQueue;
2413
- let buf = "";
2414
- const onStdout = (data) => {
2415
- buf += data;
2416
- let nl;
2417
- while ((nl = buf.indexOf(`
2418
- `)) >= 0) {
2419
- const line = buf.slice(0, nl).trim();
2420
- buf = buf.slice(nl + 1);
2421
- if (line)
2422
- lines.push(line);
2423
- }
2424
- };
2425
- const runPromise = this.sandbox.commands.run(cmd, {
2426
- ...this.options.cwd ? { cwd: this.options.cwd } : {},
2427
- onStdout
2428
- }).then((res) => {
2429
- const tail = buf.trim();
2430
- if (tail)
2431
- lines.push(tail);
2432
- lines.close();
2433
- return res;
2434
- }, (err) => {
2435
- lines.close();
2436
- throw err;
2437
- });
2438
2419
  yield { type: "init", sessionId: opts.sessionId ?? "", timestamp: now3() };
2439
2420
  let sessionId = opts.sessionId;
2440
2421
  let sawError = false;
2441
2422
  try {
2423
+ await this.ensureInstalled();
2424
+ const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
2425
+ await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
2426
+ const cmd = this.spec.buildCommand({
2427
+ promptPath,
2428
+ sessionId: opts.sessionId,
2429
+ model: this.model,
2430
+ cwd: this.options.cwd
2431
+ });
2432
+ const lines = new AsyncQueue;
2433
+ let buf = "";
2434
+ const onStdout = (data) => {
2435
+ buf += data;
2436
+ let nl;
2437
+ while ((nl = buf.indexOf(`
2438
+ `)) >= 0) {
2439
+ const line = buf.slice(0, nl).trim();
2440
+ buf = buf.slice(nl + 1);
2441
+ if (line)
2442
+ lines.push(line);
2443
+ }
2444
+ };
2445
+ const runPromise = this.sandbox.commands.run(cmd, {
2446
+ ...this.options.cwd ? { cwd: this.options.cwd } : {},
2447
+ onStdout
2448
+ }).then((res2) => {
2449
+ const tail = buf.trim();
2450
+ if (tail)
2451
+ lines.push(tail);
2452
+ lines.close();
2453
+ return res2;
2454
+ }, (err) => {
2455
+ lines.close();
2456
+ throw err;
2457
+ });
2442
2458
  for await (const line of lines) {
2443
2459
  let parsed;
2444
2460
  try {
@@ -2481,6 +2497,8 @@ function now4() {
2481
2497
  var codexSpec = {
2482
2498
  kind: "codex",
2483
2499
  authEnv: "CODEX_API_KEY",
2500
+ bin: "codex",
2501
+ install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
2484
2502
  promptPayload: (prompt) => prompt,
2485
2503
  buildCommand: ({ promptPath, sessionId, model, cwd }) => {
2486
2504
  const flags = [
@@ -2560,6 +2578,8 @@ function blocks(msg) {
2560
2578
  var ampSpec = {
2561
2579
  kind: "amp",
2562
2580
  authEnv: "AMP_API_KEY",
2581
+ bin: "amp",
2582
+ install: 'sudo npm install -g @ampcode/cli && (command -v amp >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/amp" /usr/local/bin/amp)',
2563
2583
  promptPayload: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: [{ type: "text", text: prompt }] } }) + `
2564
2584
  `,
2565
2585
  buildCommand: ({ promptPath, sessionId }) => {
@@ -12,14 +12,16 @@
12
12
  * AsyncQueue bridge → init/usage/done/error lifecycle), parameterised per-CLI
13
13
  * by a `CliAgentSpec`.
14
14
  *
15
- * Requirements (per spec): the provider CLI is installed in the sandbox image,
16
- * and the provider's API key is present in the sandbox environment (see each
17
- * spec's `authEnv`). `commands.run` inherits the sandbox env, so secrets set via
18
- * `agentc secrets set` are visible to the CLI.
19
- *
20
- * NOTE: the per-CLI command construction + resume flags + exact event shapes in
21
- * the shipped specs are mapped from each tool's docs and have NOT been verified
22
- * against a live CLI run — verify before relying on them in production.
15
+ * Provisioning (per spec): the runtime installs the provider CLI on demand —
16
+ * `command -v <bin>` before the first turn, falling back to the spec's
17
+ * `install` command when it's absent — so it works on a bare sandbox with no
18
+ * manual setup or hardcoded snapshot id. Pair it with
19
+ * `snapshots: { bootFrom: "reuse" }` on the workflow and the install happens
20
+ * exactly once: the first run installs the CLI and captures a snapshot, and
21
+ * every run after boots from that snapshot with the CLI already present (the
22
+ * probe short-circuits). The provider's API key must be in the sandbox env
23
+ * (see each spec's `authEnv`); `commands.run` inherits the sandbox env, so
24
+ * values set via `agentc secrets set` are visible to the CLI.
23
25
  */
24
26
  import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider } from "../index.js";
25
27
  /** Single-quote a value for safe interpolation into a `sh -c` command line. */
@@ -34,6 +36,15 @@ export interface CliAgentSpec {
34
36
  authEnv: string;
35
37
  /** Default model id when none is configured; omit to let the CLI choose. */
36
38
  defaultModel?: string;
39
+ /** Binary the runtime spawns (`codex`, `amp`). Probed with `command -v`
40
+ * before the first turn; if absent, `install` provisions it. */
41
+ bin: string;
42
+ /** Shell command that installs `bin` when it's missing from the sandbox.
43
+ * Runs at most once per runner, and only when the probe fails — so booting
44
+ * from a snapshot that already has the CLI (the steady state under
45
+ * `snapshots: { bootFrom: "reuse" }`) skips it. Must leave `bin` resolvable
46
+ * on a non-login shell's PATH (the runtime spawns via `sh -c`). */
47
+ install: string;
37
48
  /** Serialise the user prompt into the bytes written to the prompt file —
38
49
  * plain text for a CLI that reads the prompt from stdin (codex `-`), or a
39
50
  * JSONL user message for a `--stream-json-input` CLI (amp). */
@@ -62,6 +73,12 @@ export declare class CliAgentRunner implements ModelExecutionContract {
62
73
  readonly kind: string;
63
74
  constructor(sandbox: SandboxProvider, options: RuntimeOptions, spec: CliAgentSpec, configModel?: string | undefined);
64
75
  get model(): string | undefined;
76
+ private installed;
77
+ /** Provision the CLI on demand. A no-op once `bin` is on PATH — the steady
78
+ * state under `snapshots: { bootFrom: "reuse" }`, where the first run's
79
+ * install is baked into the snapshot every later run boots from. So the
80
+ * install command runs exactly once: on the first run of a content hash. */
81
+ private ensureInstalled;
65
82
  sendMessage(opts: {
66
83
  prompt: string;
67
84
  sessionId?: string;
@@ -5,13 +5,14 @@
5
5
  * own loop + tools, so we only stream-parse what it prints.
6
6
  *
7
7
  * Auth: set `AMP_API_KEY` (`sgamp_…`) in the sandbox env via a workflow secret.
8
- * Requires the `amp` CLI (`@ampcode/cli`) installed in the sandbox image. The
9
- * model is chosen by the AMP_API_KEY account (e.g. a GPT-only token runs GPT);
10
- * the runtime doesn't pin a model.
8
+ * The runtime installs the `amp` CLI (`@ampcode/cli`) on demand — no image
9
+ * baking needed; pair with `snapshots: { bootFrom: "reuse" }` to install once
10
+ * and boot from the captured snapshot on every run after. The model is chosen
11
+ * by the AMP_API_KEY account (e.g. a GPT-only token runs GPT); the runtime
12
+ * doesn't pin a model.
11
13
  *
12
- * ⚠️ NOT verified against a live `amp` run — the thread-continue syntax and the
13
- * exact assistant/result shapes are mapped from the docs (ampcode.com). Verify
14
- * before production use.
14
+ * Verified against a live `amp -x --stream-json` run: the user-message stdin
15
+ * shape, assistant content blocks, and result usage below all round-trip.
15
16
  */
16
17
  export interface AmpRuntimeConfig {
17
18
  /** Amp uses its configured model; reserved for forward-compatibility. */
@@ -0,0 +1,9 @@
1
+ /**
2
+ * CliAgentRunner self-provisioning — the runtime installs its CLI on demand
3
+ * (`command -v <bin>` → install if missing), runs the install at most once per
4
+ * runner, and surfaces an install failure as an `error` message rather than
5
+ * blowing up. This is what lets `createCodexRuntime()` work on a bare sandbox
6
+ * with no image baking, and lets `snapshots: { bootFrom: "reuse" }` collapse
7
+ * the install to a one-time cost.
8
+ */
9
+ export {};
@@ -4,12 +4,13 @@
4
4
  * Built on the shared CLI-agent base; Codex brings its own loop + tools, so we
5
5
  * only stream-parse what it prints.
6
6
  *
7
- * Auth: set `CODEX_API_KEY` (or `OPENAI_API_KEY`) in the sandbox env via a
8
- * workflow secret. Requires the `codex` CLI installed in the sandbox image.
7
+ * Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
8
+ * workflow secret. The runtime installs the `codex` CLI (`@openai/codex`) on
9
+ * demand — no image baking needed; pair with `snapshots: { bootFrom: "reuse" }`
10
+ * to install once and boot from the captured snapshot on every run after.
9
11
  *
10
- * ⚠️ NOT verified against a live `codex` run — the resume flag and the exact
11
- * item shapes (command_execution / reasoning fields) are mapped from the docs
12
- * (developers.openai.com/codex/noninteractive). Verify before production use.
12
+ * Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
13
+ * with the command_execution / reasoning / agent_message item shapes below.
13
14
  */
14
15
  export interface CodexRuntimeConfig {
15
16
  /** Codex model id (`-m`). Omit to use the codex CLI's configured default. */
@@ -2400,45 +2400,61 @@ class CliAgentRunner {
2400
2400
  get model() {
2401
2401
  return this.configModel ?? this.options.model ?? this.spec.defaultModel;
2402
2402
  }
2403
+ installed = false;
2404
+ async ensureInstalled() {
2405
+ if (this.installed)
2406
+ return;
2407
+ const probe = await this.sandbox.commands.run(`command -v ${this.spec.bin}`);
2408
+ if (probe.exitCode === 0) {
2409
+ this.installed = true;
2410
+ return;
2411
+ }
2412
+ const res = await this.sandbox.commands.run(this.spec.install, { timeoutMs: 300000 });
2413
+ if (res.exitCode !== 0) {
2414
+ throw new Error(`failed to install ${this.spec.bin}: ${(res.stderr || res.stdout || "").slice(-500)}`);
2415
+ }
2416
+ this.installed = true;
2417
+ }
2403
2418
  async* sendMessage(opts) {
2404
- const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
2405
- await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
2406
- const cmd = this.spec.buildCommand({
2407
- promptPath,
2408
- sessionId: opts.sessionId,
2409
- model: this.model,
2410
- cwd: this.options.cwd
2411
- });
2412
- const lines = new AsyncQueue;
2413
- let buf = "";
2414
- const onStdout = (data) => {
2415
- buf += data;
2416
- let nl;
2417
- while ((nl = buf.indexOf(`
2418
- `)) >= 0) {
2419
- const line = buf.slice(0, nl).trim();
2420
- buf = buf.slice(nl + 1);
2421
- if (line)
2422
- lines.push(line);
2423
- }
2424
- };
2425
- const runPromise = this.sandbox.commands.run(cmd, {
2426
- ...this.options.cwd ? { cwd: this.options.cwd } : {},
2427
- onStdout
2428
- }).then((res) => {
2429
- const tail = buf.trim();
2430
- if (tail)
2431
- lines.push(tail);
2432
- lines.close();
2433
- return res;
2434
- }, (err) => {
2435
- lines.close();
2436
- throw err;
2437
- });
2438
2419
  yield { type: "init", sessionId: opts.sessionId ?? "", timestamp: now3() };
2439
2420
  let sessionId = opts.sessionId;
2440
2421
  let sawError = false;
2441
2422
  try {
2423
+ await this.ensureInstalled();
2424
+ const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
2425
+ await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
2426
+ const cmd = this.spec.buildCommand({
2427
+ promptPath,
2428
+ sessionId: opts.sessionId,
2429
+ model: this.model,
2430
+ cwd: this.options.cwd
2431
+ });
2432
+ const lines = new AsyncQueue;
2433
+ let buf = "";
2434
+ const onStdout = (data) => {
2435
+ buf += data;
2436
+ let nl;
2437
+ while ((nl = buf.indexOf(`
2438
+ `)) >= 0) {
2439
+ const line = buf.slice(0, nl).trim();
2440
+ buf = buf.slice(nl + 1);
2441
+ if (line)
2442
+ lines.push(line);
2443
+ }
2444
+ };
2445
+ const runPromise = this.sandbox.commands.run(cmd, {
2446
+ ...this.options.cwd ? { cwd: this.options.cwd } : {},
2447
+ onStdout
2448
+ }).then((res2) => {
2449
+ const tail = buf.trim();
2450
+ if (tail)
2451
+ lines.push(tail);
2452
+ lines.close();
2453
+ return res2;
2454
+ }, (err) => {
2455
+ lines.close();
2456
+ throw err;
2457
+ });
2442
2458
  for await (const line of lines) {
2443
2459
  let parsed;
2444
2460
  try {
@@ -2481,6 +2497,8 @@ function now4() {
2481
2497
  var codexSpec = {
2482
2498
  kind: "codex",
2483
2499
  authEnv: "CODEX_API_KEY",
2500
+ bin: "codex",
2501
+ install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
2484
2502
  promptPayload: (prompt) => prompt,
2485
2503
  buildCommand: ({ promptPath, sessionId, model, cwd }) => {
2486
2504
  const flags = [
@@ -2560,6 +2578,8 @@ function blocks(msg) {
2560
2578
  var ampSpec = {
2561
2579
  kind: "amp",
2562
2580
  authEnv: "AMP_API_KEY",
2581
+ bin: "amp",
2582
+ install: 'sudo npm install -g @ampcode/cli && (command -v amp >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/amp" /usr/local/bin/amp)',
2563
2583
  promptPayload: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: [{ type: "text", text: prompt }] } }) + `
2564
2584
  `,
2565
2585
  buildCommand: ({ promptPath, sessionId }) => {
@@ -22,24 +22,32 @@ import type { Processor } from "../processors/processor.js";
22
22
  * workflow completes. Boolean toggle — custom post-run workflows live
23
23
  * in the separate `postRunHooks` array on `WorkflowMetadata`. */
24
24
  export type WorkflowMemoryConfig = boolean;
25
- /** Where a run boots from. The snapshot id is the unit of identity —
26
- * each captured snapshot already records the workflow + version it
27
- * came from on the snapshot row, so there's no separate "latest of
28
- * workflow X" resolution at dispatch time. Operators pick a snapshot
29
- * from the dashboard snapshot list (or `agentc snapshot list`) and
30
- * paste the id here.
31
- *
32
- * Omit `bootFrom` entirely to boot a fresh base sandbox. */
25
+ /** Boot from a specific captured snapshot, addressed by its id. Operators
26
+ * pick one from the dashboard snapshot list (or `agentc snapshot list`) and
27
+ * paste the id here. */
33
28
  export type BootSnapshot = {
34
29
  snapshotId: string;
35
30
  };
31
+ /** Boot from this workflow's OWN most recent snapshot, scoped to its content
32
+ * hash. The first run — and the first after a re-register changes the source
33
+ * — finds none and boots a fresh base sandbox; `"reuse"` also implies
34
+ * `saveLatest`, so that run captures a snapshot and every run after it boots
35
+ * from it. This lets a runtime install its tooling once (e.g. a CLI-agent
36
+ * runtime `npm i -g`'ing its CLI) and skip the install on every later run,
37
+ * with no hardcoded snapshot id to manage. Re-registering with changed
38
+ * source rolls the content hash, which transparently invalidates the cache
39
+ * and re-installs on the next run. */
40
+ export type ReuseSnapshot = "reuse";
36
41
  /** Snapshot configuration — boot source plus capture knobs. One object
37
42
  * per workflow / per invocation; collapsing boot + capture under a
38
43
  * single key reads as "all snapshot config lives here." */
39
44
  export interface SnapshotConfig {
40
- /** Where the runner restores from at run start. Structured (workflow
41
- * ref or snapshot id) so the intent is explicit at the call site. */
42
- bootFrom?: BootSnapshot;
45
+ /** Where the runner restores from at run start:
46
+ * - `{ snapshotId }` — a specific captured snapshot.
47
+ * - `"reuse"` — this workflow's own latest snapshot (content-hash scoped);
48
+ * fresh on the first run / after a re-register. Implies `saveLatest`.
49
+ * - omitted — a fresh base sandbox. */
50
+ bootFrom?: BootSnapshot | ReuseSnapshot;
43
51
  /** Capture the sandbox state on terminal success. The latest pointer
44
52
  * on `workflow_runs.vercel_snapshot_id` always tracks the most
45
53
  * recent capture; without `retainSteps`, prior captures are deleted
@@ -26,8 +26,8 @@ export type { WorkflowMetadata } from "./workflow-metadata.js";
26
26
  export interface AgentEventSink {
27
27
  emit(event: AgentLifecycleEvent): void | Promise<void>;
28
28
  }
29
- import type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema } from "./workflow-metadata.js";
30
- export type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema };
29
+ import type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema } from "./workflow-metadata.js";
30
+ export type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema };
31
31
  /** Turn/iteration budget for `agent(opts)`. Re-exported here so authors
32
32
  * can type per-invoke budget overrides they pass as workflow input. */
33
33
  export interface AgentBudget {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.5.5",
3
+ "version": "0.5.6",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
package/src/index.ts CHANGED
@@ -38,6 +38,7 @@ export type {
38
38
  WorkflowHooks,
39
39
  SnapshotConfig,
40
40
  BootSnapshot,
41
+ ReuseSnapshot,
41
42
  IOSchema,
42
43
  OutputSchema,
43
44
  WorkflowMemoryConfig,
@@ -12,14 +12,16 @@
12
12
  * AsyncQueue bridge → init/usage/done/error lifecycle), parameterised per-CLI
13
13
  * by a `CliAgentSpec`.
14
14
  *
15
- * Requirements (per spec): the provider CLI is installed in the sandbox image,
16
- * and the provider's API key is present in the sandbox environment (see each
17
- * spec's `authEnv`). `commands.run` inherits the sandbox env, so secrets set via
18
- * `agentc secrets set` are visible to the CLI.
19
- *
20
- * NOTE: the per-CLI command construction + resume flags + exact event shapes in
21
- * the shipped specs are mapped from each tool's docs and have NOT been verified
22
- * against a live CLI run — verify before relying on them in production.
15
+ * Provisioning (per spec): the runtime installs the provider CLI on demand —
16
+ * `command -v <bin>` before the first turn, falling back to the spec's
17
+ * `install` command when it's absent — so it works on a bare sandbox with no
18
+ * manual setup or hardcoded snapshot id. Pair it with
19
+ * `snapshots: { bootFrom: "reuse" }` on the workflow and the install happens
20
+ * exactly once: the first run installs the CLI and captures a snapshot, and
21
+ * every run after boots from that snapshot with the CLI already present (the
22
+ * probe short-circuits). The provider's API key must be in the sandbox env
23
+ * (see each spec's `authEnv`); `commands.run` inherits the sandbox env, so
24
+ * values set via `agentc secrets set` are visible to the CLI.
23
25
  */
24
26
 
25
27
  import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider } from "../index.js";
@@ -44,6 +46,15 @@ export interface CliAgentSpec {
44
46
  authEnv: string;
45
47
  /** Default model id when none is configured; omit to let the CLI choose. */
46
48
  defaultModel?: string;
49
+ /** Binary the runtime spawns (`codex`, `amp`). Probed with `command -v`
50
+ * before the first turn; if absent, `install` provisions it. */
51
+ bin: string;
52
+ /** Shell command that installs `bin` when it's missing from the sandbox.
53
+ * Runs at most once per runner, and only when the probe fails — so booting
54
+ * from a snapshot that already has the CLI (the steady state under
55
+ * `snapshots: { bootFrom: "reuse" }`) skips it. Must leave `bin` resolvable
56
+ * on a non-login shell's PATH (the runtime spawns via `sh -c`). */
57
+ install: string;
47
58
  /** Serialise the user prompt into the bytes written to the prompt file —
48
59
  * plain text for a CLI that reads the prompt from stdin (codex `-`), or a
49
60
  * JSONL user message for a `--stream-json-input` CLI (amp). */
@@ -76,6 +87,23 @@ export class CliAgentRunner implements ModelExecutionContract {
76
87
  return this.configModel ?? this.options.model ?? this.spec.defaultModel;
77
88
  }
78
89
 
90
+ private installed = false;
91
+
92
+ /** Provision the CLI on demand. A no-op once `bin` is on PATH — the steady
93
+ * state under `snapshots: { bootFrom: "reuse" }`, where the first run's
94
+ * install is baked into the snapshot every later run boots from. So the
95
+ * install command runs exactly once: on the first run of a content hash. */
96
+ private async ensureInstalled(): Promise<void> {
97
+ if (this.installed) return;
98
+ const probe = await this.sandbox.commands.run(`command -v ${this.spec.bin}`);
99
+ if (probe.exitCode === 0) { this.installed = true; return; }
100
+ const res = await this.sandbox.commands.run(this.spec.install, { timeoutMs: 300_000 });
101
+ if (res.exitCode !== 0) {
102
+ throw new Error(`failed to install ${this.spec.bin}: ${(res.stderr || res.stdout || "").slice(-500)}`);
103
+ }
104
+ this.installed = true;
105
+ }
106
+
79
107
  // No captureCheckpoint/restoreCheckpoint: the CLI persists its thread/rollout
80
108
  // on the sandbox filesystem (which round-trips through the pause snapshot),
81
109
  // and the loop already carries the session id we emit on init/done and pass
@@ -87,45 +115,50 @@ export class CliAgentRunner implements ModelExecutionContract {
87
115
  iteration?: number;
88
116
  signal?: AbortSignal;
89
117
  }): AsyncGenerator<AgentMessage> {
90
- const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
91
- await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
92
- const cmd = this.spec.buildCommand({
93
- promptPath,
94
- sessionId: opts.sessionId,
95
- model: this.model,
96
- cwd: this.options.cwd,
97
- });
98
-
99
- // Bridge the streaming stdout callback into an async-iterable of complete
100
- // JSONL lines. `onStdout` chunks aren't line-aligned, so buffer + split.
101
- const lines = new AsyncQueue<string>();
102
- let buf = "";
103
- const onStdout = (data: string) => {
104
- buf += data;
105
- let nl: number;
106
- while ((nl = buf.indexOf("\n")) >= 0) {
107
- const line = buf.slice(0, nl).trim();
108
- buf = buf.slice(nl + 1);
109
- if (line) lines.push(line);
110
- }
111
- };
112
-
113
- // `commands.run` resolves when the process exits. Kick it off (don't await
114
- // yet); flush the trailing buffer + close the queue on completion so the
115
- // for-await below drains and we can read the exit code.
116
- const runPromise = this.sandbox.commands.run(cmd, {
117
- ...(this.options.cwd ? { cwd: this.options.cwd } : {}),
118
- onStdout,
119
- }).then(
120
- (res) => { const tail = buf.trim(); if (tail) lines.push(tail); lines.close(); return res; },
121
- (err) => { lines.close(); throw err; },
122
- );
123
-
124
118
  yield { type: "init", sessionId: opts.sessionId ?? "", timestamp: now() };
125
119
 
126
120
  let sessionId = opts.sessionId;
127
121
  let sawError = false;
128
122
  try {
123
+ // Provision the CLI before the first turn (no-op when it's already
124
+ // present, e.g. booting from a "reuse" snapshot). A failure here surfaces
125
+ // as an `error` AgentMessage via the catch below.
126
+ await this.ensureInstalled();
127
+
128
+ const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
129
+ await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
130
+ const cmd = this.spec.buildCommand({
131
+ promptPath,
132
+ sessionId: opts.sessionId,
133
+ model: this.model,
134
+ cwd: this.options.cwd,
135
+ });
136
+
137
+ // Bridge the streaming stdout callback into an async-iterable of complete
138
+ // JSONL lines. `onStdout` chunks aren't line-aligned, so buffer + split.
139
+ const lines = new AsyncQueue<string>();
140
+ let buf = "";
141
+ const onStdout = (data: string) => {
142
+ buf += data;
143
+ let nl: number;
144
+ while ((nl = buf.indexOf("\n")) >= 0) {
145
+ const line = buf.slice(0, nl).trim();
146
+ buf = buf.slice(nl + 1);
147
+ if (line) lines.push(line);
148
+ }
149
+ };
150
+
151
+ // `commands.run` resolves when the process exits. Kick it off (don't await
152
+ // yet); flush the trailing buffer + close the queue on completion so the
153
+ // for-await below drains and we can read the exit code.
154
+ const runPromise = this.sandbox.commands.run(cmd, {
155
+ ...(this.options.cwd ? { cwd: this.options.cwd } : {}),
156
+ onStdout,
157
+ }).then(
158
+ (res) => { const tail = buf.trim(); if (tail) lines.push(tail); lines.close(); return res; },
159
+ (err) => { lines.close(); throw err; },
160
+ );
161
+
129
162
  for await (const line of lines) {
130
163
  let parsed: Record<string, unknown>;
131
164
  try {
@@ -5,13 +5,14 @@
5
5
  * own loop + tools, so we only stream-parse what it prints.
6
6
  *
7
7
  * Auth: set `AMP_API_KEY` (`sgamp_…`) in the sandbox env via a workflow secret.
8
- * Requires the `amp` CLI (`@ampcode/cli`) installed in the sandbox image. The
9
- * model is chosen by the AMP_API_KEY account (e.g. a GPT-only token runs GPT);
10
- * the runtime doesn't pin a model.
8
+ * The runtime installs the `amp` CLI (`@ampcode/cli`) on demand — no image
9
+ * baking needed; pair with `snapshots: { bootFrom: "reuse" }` to install once
10
+ * and boot from the captured snapshot on every run after. The model is chosen
11
+ * by the AMP_API_KEY account (e.g. a GPT-only token runs GPT); the runtime
12
+ * doesn't pin a model.
11
13
  *
12
- * ⚠️ NOT verified against a live `amp` run — the thread-continue syntax and the
13
- * exact assistant/result shapes are mapped from the docs (ampcode.com). Verify
14
- * before production use.
14
+ * Verified against a live `amp -x --stream-json` run: the user-message stdin
15
+ * shape, assistant content blocks, and result usage below all round-trip.
15
16
  */
16
17
 
17
18
  import type { AgentMessage } from "../index.js";
@@ -32,6 +33,10 @@ function blocks(msg: Record<string, unknown>): Array<Record<string, unknown>> {
32
33
  const ampSpec: CliAgentSpec = {
33
34
  kind: "amp",
34
35
  authEnv: "AMP_API_KEY",
36
+ bin: "amp",
37
+ // Global npm install; symlink onto PATH only if the global bin dir isn't
38
+ // already there (so a non-login `sh -c` can find it).
39
+ install: 'sudo npm install -g @ampcode/cli && (command -v amp >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/amp" /usr/local/bin/amp)',
35
40
  // `--stream-json-input` reads JSON Lines user messages from stdin; write one.
36
41
  // amp's --stream-json-input wants Claude-shaped content blocks, not a bare
37
42
  // string (it rejects a string `content` with "expected array, received string").
@@ -4,12 +4,13 @@
4
4
  * Built on the shared CLI-agent base; Codex brings its own loop + tools, so we
5
5
  * only stream-parse what it prints.
6
6
  *
7
- * Auth: set `CODEX_API_KEY` (or `OPENAI_API_KEY`) in the sandbox env via a
8
- * workflow secret. Requires the `codex` CLI installed in the sandbox image.
7
+ * Auth: set `OPENAI_API_KEY` (or `CODEX_API_KEY`) in the sandbox env via a
8
+ * workflow secret. The runtime installs the `codex` CLI (`@openai/codex`) on
9
+ * demand — no image baking needed; pair with `snapshots: { bootFrom: "reuse" }`
10
+ * to install once and boot from the captured snapshot on every run after.
9
11
  *
10
- * ⚠️ NOT verified against a live `codex` run — the resume flag and the exact
11
- * item shapes (command_execution / reasoning fields) are mapped from the docs
12
- * (developers.openai.com/codex/noninteractive). Verify before production use.
12
+ * Verified against codex-cli 0.124.0: `codex exec --json` + resume-by-thread,
13
+ * with the command_execution / reasoning / agent_message item shapes below.
13
14
  */
14
15
 
15
16
  import type { AgentMessage } from "../index.js";
@@ -21,6 +22,10 @@ function now(): string { return new Date().toISOString(); }
21
22
  const codexSpec: CliAgentSpec = {
22
23
  kind: "codex",
23
24
  authEnv: "CODEX_API_KEY",
25
+ bin: "codex",
26
+ // Global npm install; symlink onto PATH only if the global bin dir isn't
27
+ // already there (so a non-login `sh -c` can find it).
28
+ install: 'sudo npm install -g @openai/codex && (command -v codex >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/codex" /usr/local/bin/codex)',
24
29
  // Codex reads the prompt from stdin when invoked as `codex exec ... -`.
25
30
  promptPayload: (prompt) => prompt,
26
31
  buildCommand: ({ promptPath, sessionId, model, cwd }) => {
@@ -25,23 +25,32 @@ import type { Processor } from "../processors/processor.js";
25
25
  * in the separate `postRunHooks` array on `WorkflowMetadata`. */
26
26
  export type WorkflowMemoryConfig = boolean;
27
27
 
28
- /** Where a run boots from. The snapshot id is the unit of identity —
29
- * each captured snapshot already records the workflow + version it
30
- * came from on the snapshot row, so there's no separate "latest of
31
- * workflow X" resolution at dispatch time. Operators pick a snapshot
32
- * from the dashboard snapshot list (or `agentc snapshot list`) and
33
- * paste the id here.
34
- *
35
- * Omit `bootFrom` entirely to boot a fresh base sandbox. */
28
+ /** Boot from a specific captured snapshot, addressed by its id. Operators
29
+ * pick one from the dashboard snapshot list (or `agentc snapshot list`) and
30
+ * paste the id here. */
36
31
  export type BootSnapshot = { snapshotId: string };
37
32
 
33
+ /** Boot from this workflow's OWN most recent snapshot, scoped to its content
34
+ * hash. The first run — and the first after a re-register changes the source
35
+ * — finds none and boots a fresh base sandbox; `"reuse"` also implies
36
+ * `saveLatest`, so that run captures a snapshot and every run after it boots
37
+ * from it. This lets a runtime install its tooling once (e.g. a CLI-agent
38
+ * runtime `npm i -g`'ing its CLI) and skip the install on every later run,
39
+ * with no hardcoded snapshot id to manage. Re-registering with changed
40
+ * source rolls the content hash, which transparently invalidates the cache
41
+ * and re-installs on the next run. */
42
+ export type ReuseSnapshot = "reuse";
43
+
38
44
  /** Snapshot configuration — boot source plus capture knobs. One object
39
45
  * per workflow / per invocation; collapsing boot + capture under a
40
46
  * single key reads as "all snapshot config lives here." */
41
47
  export interface SnapshotConfig {
42
- /** Where the runner restores from at run start. Structured (workflow
43
- * ref or snapshot id) so the intent is explicit at the call site. */
44
- bootFrom?: BootSnapshot;
48
+ /** Where the runner restores from at run start:
49
+ * - `{ snapshotId }` — a specific captured snapshot.
50
+ * - `"reuse"` — this workflow's own latest snapshot (content-hash scoped);
51
+ * fresh on the first run / after a re-register. Implies `saveLatest`.
52
+ * - omitted — a fresh base sandbox. */
53
+ bootFrom?: BootSnapshot | ReuseSnapshot;
45
54
  /** Capture the sandbox state on terminal success. The latest pointer
46
55
  * on `workflow_runs.vercel_snapshot_id` always tracks the most
47
56
  * recent capture; without `retainSteps`, prior captures are deleted
@@ -37,8 +37,8 @@ export interface AgentEventSink {
37
37
  emit(event: AgentLifecycleEvent): void | Promise<void>;
38
38
  }
39
39
 
40
- import type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema } from "./workflow-metadata.js";
41
- export type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, IOSchema, OutputSchema };
40
+ import type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema } from "./workflow-metadata.js";
41
+ export type { WorkflowMemoryConfig, SnapshotConfig, BootSnapshot, ReuseSnapshot, IOSchema, OutputSchema };
42
42
 
43
43
  /** Turn/iteration budget for `agent(opts)`. Re-exported here so authors
44
44
  * can type per-invoke budget overrides they pass as workflow input. */