@agent-compose/sdk 0.6.0 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/agent-context.d.ts +1 -1
- package/dist/index.d.ts +10 -1
- package/dist/index.js +288 -43
- package/dist/processors/ask-human.d.ts +30 -0
- package/dist/processors/ask-human.test.d.ts +1 -0
- package/dist/processors/index.d.ts +1 -0
- package/dist/runtimes/_cli-agent.d.ts +9 -0
- package/dist/runtimes/cursor.d.ts +9 -0
- package/dist/runtimes/droid.d.ts +9 -0
- package/dist/runtimes/openai-desktop.js +277 -43
- package/dist/runtimes/opencode.d.ts +25 -0
- package/dist/runtimes/vercel.js +11 -1
- package/dist/step-invocation/invoker.d.ts +14 -1
- package/dist/step-invocation/protocol.d.ts +8 -0
- package/dist/types/sandbox.d.ts +9 -0
- package/dist/types/workflow.d.ts +12 -0
- package/dist/utils/errors.d.ts +9 -1
- package/package.json +1 -1
- package/src/agent/agent-context.ts +23 -16
- package/src/agent/agent-loop.ts +9 -2
- package/src/agent/run-agent.ts +9 -1
- package/src/index.ts +13 -0
- package/src/processors/ask-human.ts +136 -0
- package/src/processors/index.ts +5 -0
- package/src/runtimes/_cli-agent.ts +12 -2
- package/src/runtimes/cursor.ts +59 -0
- package/src/runtimes/droid.ts +63 -0
- package/src/runtimes/opencode.ts +61 -0
- package/src/sandbox.ts +12 -0
- package/src/step-invocation/invoker.ts +185 -23
- package/src/step-invocation/protocol.ts +11 -0
- package/src/types/sandbox.ts +9 -0
- package/src/types/workflow.ts +12 -0
- package/src/utils/errors.ts +19 -2
|
@@ -71,29 +71,36 @@ Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
|
|
|
71
71
|
|
|
72
72
|
## Writing workflow / agent code — the SDK
|
|
73
73
|
|
|
74
|
-
\`@agent-compose/sdk\` is installed in \`/workspace
|
|
75
|
-
|
|
74
|
+
\`@agent-compose/sdk\` is installed in \`/workspace\`. **To author a workflow,
|
|
75
|
+
ALWAYS run \`/ac:generate-workflow\`** (and \`/ac:generate-agent\` for an agent
|
|
76
|
+
step) instead of writing source from memory — the skill scaffolds the correct,
|
|
77
|
+
current shape. Then \`agentc register <file.ts>\` (or \`/ac:register\`).
|
|
76
78
|
|
|
77
|
-
|
|
79
|
+
The skill writes **step-form** (a builder of discrete, durable \`.step()\`s).
|
|
80
|
+
Never write the legacy run-form (\`defineWorkflow({ run(ctx, sandbox) { … } })\`):
|
|
81
|
+
it is one opaque step, so any failure or resume re-runs the whole body — and
|
|
82
|
+
**pause does not work in run-form**.
|
|
78
83
|
|
|
79
|
-
|
|
80
|
-
\`agentc register <file.ts>\` (or \`/ac:register\`).
|
|
84
|
+
## Pausing to ask the human
|
|
81
85
|
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
When you can't or shouldn't proceed without a human, run \`agentc pause\`, then
|
|
85
|
-
**END YOUR TURN**:
|
|
86
|
+
To ask a human and get an answer back, use the **\`AskUserQuestion\`** tool if
|
|
87
|
+
you have it; otherwise run **\`agentc pause\`**:
|
|
86
88
|
|
|
87
89
|
agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
|
|
88
90
|
--option retry --option skip
|
|
89
91
|
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
92
|
+
**Both BLOCK and hand you the answer inline.** While you wait, the run is
|
|
93
|
+
suspended — your sandbox is frozen and compute stops, so a pause is free while
|
|
94
|
+
the human decides. When they answer, the call RETURNS with their decision: the
|
|
95
|
+
\`AskUserQuestion\` tool result, or \`agentc pause\`'s output
|
|
96
|
+
(\`▶ Resumed. The human answered: …\`), carries it.
|
|
97
|
+
|
|
98
|
+
**Then USE that answer to finish your work — do NOT end your turn.** This is NOT
|
|
99
|
+
fire-and-forget, and the answer does NOT arrive in a later message: it comes
|
|
100
|
+
back right where you called it, on the SAME turn. The shape is: ask → the call
|
|
101
|
+
blocks → it returns the human's answer → you act on it and produce your result.
|
|
102
|
+
Never end your turn before the call returns, never guess an answer, and never
|
|
103
|
+
proceed without one.
|
|
97
104
|
|
|
98
105
|
Reach for it the moment you hit — or foresee — any of these:
|
|
99
106
|
- **A wall only a human can clear:** a 401/403, a missing credential, an
|
package/src/agent/agent-loop.ts
CHANGED
|
@@ -164,9 +164,16 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
164
164
|
const logLabel = opts.label ?? "[Agent Loop]";
|
|
165
165
|
const startedAt = Date.now();
|
|
166
166
|
// No budget ⇒ no turn cap: the harness runtime (Claude Code) decides when it's
|
|
167
|
-
// done. A numeric budget is an explicit caller choice
|
|
167
|
+
// done in one unbounded session. A numeric budget is an explicit caller choice.
|
|
168
|
+
// Default to a SMALL allowance (not 1) so the loop can RE-PROMPT for the closing
|
|
169
|
+
// <status>/<response> when a session ends EARLY — e.g. a human pause (agentc
|
|
170
|
+
// pause / AskUserQuestion) freezes the VM mid-tool and the resumed session can
|
|
171
|
+
// drop its final turn after doing the work. The re-prompt (PROTOCOL_SUFFIX)
|
|
172
|
+
// only fires when an iteration produced no exit_signal, so a clean run still
|
|
173
|
+
// settles in ONE iteration — these extra iterations are a recovery path, not
|
|
174
|
+
// the norm.
|
|
168
175
|
const turnsPerIteration = opts.turnsPerIteration;
|
|
169
|
-
const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ?
|
|
176
|
+
const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 3 : 8);
|
|
170
177
|
// A responseSchema is a CONTRACT, not a hope: when the agent's <response> fails
|
|
171
178
|
// validation, the loop re-prompts with the exact errors until it conforms —
|
|
172
179
|
// without consuming the caller's iteration budget. The backstop below only
|
package/src/agent/run-agent.ts
CHANGED
|
@@ -14,6 +14,7 @@ import { z } from "zod";
|
|
|
14
14
|
import { randomUUID } from "node:crypto";
|
|
15
15
|
import { getActiveStep, nextAgentCallInActiveStep } from "../active-step.js";
|
|
16
16
|
import { corePause, type PauseRequest } from "../pause/pause-core.js";
|
|
17
|
+
import { createAskHumanProcessor } from "../processors/ask-human.js";
|
|
17
18
|
import { agentLoop } from "./agent-loop.js";
|
|
18
19
|
import { consumeSteerPending, runControlPoller } from "./steer-control.js";
|
|
19
20
|
import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
|
|
@@ -330,6 +331,13 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
|
|
|
330
331
|
// (the boundary self-pause trigger + the prompt instruction above). Human
|
|
331
332
|
// steering works regardless — see the ungated control poller.
|
|
332
333
|
const mode = opts.mode ?? "auto";
|
|
334
|
+
// Ask-a-human is TOOLS-driven, not a separate mode: the ask-human processor is
|
|
335
|
+
// a chain default that maps the `AskUserQuestion` tool to a server pause (the
|
|
336
|
+
// run freezes, the human answers, the answer returns as the tool result). It's
|
|
337
|
+
// a no-op for any agent that doesn't have / call `AskUserQuestion`, so granting
|
|
338
|
+
// that tool to an agent IS its "can ask a human" switch — withhold it and the
|
|
339
|
+
// agent simply can't (it reports blockers up instead).
|
|
340
|
+
const processors = [createAskHumanProcessor(), ...(opts.processors ?? [])];
|
|
333
341
|
|
|
334
342
|
try {
|
|
335
343
|
return await agentLoop({
|
|
@@ -361,7 +369,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
|
|
|
361
369
|
...(opts.events ? { onAgentLifecycleEvent: (event: AgentLifecycleEvent) => { void opts.events?.emit(event); } } : {}),
|
|
362
370
|
...(opts.onAgentEvent ? { onAgentEvent: opts.onAgentEvent } : {}),
|
|
363
371
|
...(opts.onIteration ? { onIteration: opts.onIteration } : {}),
|
|
364
|
-
...(
|
|
372
|
+
...(processors.length ? { processors } : {}),
|
|
365
373
|
requestContext: opts.requestContext ?? RequestContext.fromReserved({
|
|
366
374
|
teamId: "", runId: "", workflowId: "",
|
|
367
375
|
factoryId: null, apiKeyScopes: [], parentRunId: null,
|
package/src/index.ts
CHANGED
|
@@ -77,6 +77,8 @@ export {
|
|
|
77
77
|
requireScope,
|
|
78
78
|
redactPattern,
|
|
79
79
|
createGatePauseProcessor,
|
|
80
|
+
createAskHumanProcessor,
|
|
81
|
+
ASK_USER_QUESTION_TOOL,
|
|
80
82
|
} from "./processors/index.js";
|
|
81
83
|
export type {
|
|
82
84
|
Processor,
|
|
@@ -174,6 +176,17 @@ export { default as codexRuntime } from "./runtimes/codex.js";
|
|
|
174
176
|
export { createAmpRuntime } from "./runtimes/amp.js";
|
|
175
177
|
export type { AmpRuntimeConfig } from "./runtimes/amp.js";
|
|
176
178
|
export { default as ampRuntime } from "./runtimes/amp.js";
|
|
179
|
+
export { createOpencodeRuntime, opencodeSpec } from "./runtimes/opencode.js";
|
|
180
|
+
export type { OpencodeRuntimeConfig } from "./runtimes/opencode.js";
|
|
181
|
+
export { default as opencodeRuntime } from "./runtimes/opencode.js";
|
|
182
|
+
|
|
183
|
+
export { createCursorRuntime, cursorSpec } from "./runtimes/cursor.js";
|
|
184
|
+
export type { CursorRuntimeConfig } from "./runtimes/cursor.js";
|
|
185
|
+
export { default as cursorRuntime } from "./runtimes/cursor.js";
|
|
186
|
+
|
|
187
|
+
export { createDroidRuntime, droidSpec } from "./runtimes/droid.js";
|
|
188
|
+
export type { DroidRuntimeConfig } from "./runtimes/droid.js";
|
|
189
|
+
export { default as droidRuntime } from "./runtimes/droid.js";
|
|
177
190
|
|
|
178
191
|
// Built-in coding tools for Vercel AI SDK runtime.
|
|
179
192
|
export { bashTool, codingTools, editTool, readTool, writeTool } from "./tools/index.js";
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Ask-human processor (ADR-0028).
|
|
3
|
+
*
|
|
4
|
+
* Makes "ask a human" a FIRST-CLASS agent affordance: when the agent calls the
|
|
5
|
+
* built-in `AskUserQuestion` tool, this short-circuits it into a SERVER pause
|
|
6
|
+
* (`requestPauseAndAwait`) — the run freezes (compute stops), the question +
|
|
7
|
+
* options land on the human's pause feed, and the human's answer comes back as
|
|
8
|
+
* the tool result. No `agentc pause` CLI for the model to remember, and no
|
|
9
|
+
* dependency on prompt discipline: the moment the agent asks, the run pauses.
|
|
10
|
+
*
|
|
11
|
+
* Loud by construction — the danger this fixes is a pause that SILENTLY doesn't
|
|
12
|
+
* happen (a stale in-sandbox CLI, a non-E2B substrate, an auth error) letting
|
|
13
|
+
* the agent proceed as if it had an answer:
|
|
14
|
+
* - pause cannot be created (server reject) → `Verdict.abort` ENDS the agent
|
|
15
|
+
* loop with a WorkflowError. The run fails loud; it never guesses an answer.
|
|
16
|
+
* - pause expires / is cancelled → the tool result says NO answer came and to
|
|
17
|
+
* not assume one.
|
|
18
|
+
*
|
|
19
|
+
* Lives in the shared `gateToolCall` chain, so one implementation covers every
|
|
20
|
+
* runtime (the Claude Agent SDK `PreToolUse` hook and the ACP permission path).
|
|
21
|
+
* No run credential in the env (local / non-sandbox) → no-op: `AskUserQuestion`
|
|
22
|
+
* passes through untouched so a dev invocation isn't hard-failed.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
import type { Processor, ProcessorContext, ToolCall } from "./processor.js";
|
|
26
|
+
import { Verdict } from "./processor.js";
|
|
27
|
+
import { requestPauseAndAwait } from "../agent/pause-client.js";
|
|
28
|
+
import type { GatePauseConnection } from "./gate-pause.js";
|
|
29
|
+
|
|
30
|
+
/** The Claude built-in tool an agent uses to ask the user a question. */
|
|
31
|
+
export const ASK_USER_QUESTION_TOOL = "AskUserQuestion";
|
|
32
|
+
|
|
33
|
+
/** One question in an `AskUserQuestion` call (only the fields we read). */
|
|
34
|
+
interface AskQuestion { question?: unknown; header?: unknown; options?: unknown }
|
|
35
|
+
|
|
36
|
+
/** Pull the human-facing question + its option labels out of an
|
|
37
|
+
* `AskUserQuestion` tool input. We pause on the FIRST question (the common
|
|
38
|
+
* case); any others are folded into the reason so nothing is lost. */
|
|
39
|
+
function parseAsk(input: Record<string, unknown>): { reason: string; options: Array<{ id: string; label: string }> } {
|
|
40
|
+
const questions = Array.isArray(input.questions) ? (input.questions as AskQuestion[]) : [];
|
|
41
|
+
const first = questions[0] ?? {};
|
|
42
|
+
const head = typeof first.question === "string" && first.question.trim()
|
|
43
|
+
? first.question.trim()
|
|
44
|
+
: "The agent needs your input to continue.";
|
|
45
|
+
const extra = questions.length > 1
|
|
46
|
+
? ` (+${questions.length - 1} more question${questions.length > 2 ? "s" : ""})`
|
|
47
|
+
: "";
|
|
48
|
+
const options = Array.isArray(first.options)
|
|
49
|
+
? (first.options as Array<{ label?: unknown }>)
|
|
50
|
+
.map((o) => String(o?.label ?? "").trim())
|
|
51
|
+
.filter(Boolean)
|
|
52
|
+
.map((label) => ({ id: label, label }))
|
|
53
|
+
: [];
|
|
54
|
+
return { reason: head + extra, options };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/** Unwrap the dashboard's `{ decision }` resume payload to the raw answer text. */
|
|
58
|
+
function answerText(decision: unknown): string {
|
|
59
|
+
const raw = decision !== null && typeof decision === "object" && "decision" in decision
|
|
60
|
+
? (decision as { decision: unknown }).decision
|
|
61
|
+
: decision;
|
|
62
|
+
return typeof raw === "string" ? raw.trim() : raw == null ? "" : JSON.stringify(raw);
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** Recognise the agent shelling out to `agentc pause` and extract the same
|
|
66
|
+
* {reason, options} we'd get from AskUserQuestion. We INTERCEPT it here — in the
|
|
67
|
+
* tool gate, BEFORE the command runs — so the pause takes the DESIGNED path (the
|
|
68
|
+
* answer returns as the tool result and the agent loop continues) instead of the
|
|
69
|
+
* command actually blocking inside the sandbox shell, which froze the runner
|
|
70
|
+
* mid-tool-exec and never resumed the loop. */
|
|
71
|
+
function parseAgentcPause(command: string): { reason: string; options: Array<{ id: string; label: string }> } | null {
|
|
72
|
+
if (!/(^|\s|&&|;|\|)\s*agentc\s+pause(\s|$)/.test(command)) return null;
|
|
73
|
+
const r = command.match(/--reason(?:=|\s+)(?:"([^"]*)"|'([^']*)'|(\S+))/);
|
|
74
|
+
const reason = (r?.[1] ?? r?.[2] ?? r?.[3] ?? "The agent needs your input to continue.").trim();
|
|
75
|
+
const options = [...command.matchAll(/--option(?:=|\s+)(?:"([^"]*)"|'([^']*)'|(\S+))/g)]
|
|
76
|
+
.map((m) => (m[1] ?? m[2] ?? m[3] ?? "").trim())
|
|
77
|
+
.filter(Boolean)
|
|
78
|
+
.map((label) => ({ id: label, label }));
|
|
79
|
+
return { reason, options };
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** Pull the ask (reason + options) from either the `AskUserQuestion` tool OR a
|
|
83
|
+
* `Bash` call running `agentc pause`. Null for anything else. */
|
|
84
|
+
function extractAsk(call: ToolCall): { reason: string; options: Array<{ id: string; label: string }> } | null {
|
|
85
|
+
if (call.toolName === ASK_USER_QUESTION_TOOL) return parseAsk(call.toolInput);
|
|
86
|
+
if (call.toolName === "Bash") {
|
|
87
|
+
const cmd = (call.toolInput as { command?: unknown })?.command;
|
|
88
|
+
return typeof cmd === "string" ? parseAgentcPause(cmd) : null;
|
|
89
|
+
}
|
|
90
|
+
return null;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
export function createAskHumanProcessor(opts: { connection?: GatePauseConnection } = {}): Processor {
|
|
94
|
+
return {
|
|
95
|
+
name: "ask-human",
|
|
96
|
+
async processToolCall(call: ToolCall, ctx: ProcessorContext) {
|
|
97
|
+
// Fires for AskUserQuestion OR a `Bash` call running `agentc pause` — both
|
|
98
|
+
// ask a human and must take the SAME processor path so the answer returns
|
|
99
|
+
// as the tool result and the agent loop continues.
|
|
100
|
+
const ask = extractAsk(call);
|
|
101
|
+
if (!ask) return Verdict.continue(call);
|
|
102
|
+
|
|
103
|
+
const conn = opts.connection ?? {
|
|
104
|
+
baseUrl: process.env.AGENT_COMPOSE_URL ?? "",
|
|
105
|
+
token: process.env.AGENT_COMPOSE_RUN_TOKEN ?? "",
|
|
106
|
+
runId: process.env.RUN_ID ?? "",
|
|
107
|
+
};
|
|
108
|
+
// No run credential (local / non-sandbox) — can't pause; let the tool
|
|
109
|
+
// through rather than hard-failing a dev invocation.
|
|
110
|
+
if (!conn.baseUrl || !conn.token || !conn.runId) return Verdict.continue(call);
|
|
111
|
+
|
|
112
|
+
let decision;
|
|
113
|
+
try {
|
|
114
|
+
decision = await requestPauseAndAwait({
|
|
115
|
+
baseUrl: conn.baseUrl, token: conn.token, runId: conn.runId,
|
|
116
|
+
reason: ask.reason,
|
|
117
|
+
...(ask.options.length ? { options: ask.options } : {}),
|
|
118
|
+
action: { tool: call.toolName, input: call.toolInput },
|
|
119
|
+
signal: ctx.abortSignal,
|
|
120
|
+
});
|
|
121
|
+
} catch (err) {
|
|
122
|
+
// The pause could NOT be honored (server reject, non-E2B substrate, auth).
|
|
123
|
+
// Abort the loop — never let the agent proceed as if it had an answer.
|
|
124
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
125
|
+
return Verdict.abort(`Could not ask the human — the run could not be paused (${msg}). Stopping rather than guessing an answer.`);
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
if (decision.status === "resolved") {
|
|
129
|
+
const answer = answerText(decision.decision);
|
|
130
|
+
return Verdict.deny(`The human answered: ${answer || "(no text returned)"}. Continue using this answer.`);
|
|
131
|
+
}
|
|
132
|
+
// Expired / cancelled — no answer. Do NOT let the agent assume one.
|
|
133
|
+
return Verdict.deny(`No answer came back (${decision.status}). Do NOT assume an answer — ask again, or stop and report exactly what you need from a human.`);
|
|
134
|
+
},
|
|
135
|
+
};
|
|
136
|
+
}
|
package/src/processors/index.ts
CHANGED
|
@@ -20,3 +20,8 @@ export {
|
|
|
20
20
|
// through the shared gateToolCall chain.
|
|
21
21
|
export { createGatePauseProcessor } from "./gate-pause.js";
|
|
22
22
|
export type { GatePausePolicy, GatePauseApproval, GatePauseConnection } from "./gate-pause.js";
|
|
23
|
+
|
|
24
|
+
// ADR-0028 — first-class "ask a human": maps the agent's `AskUserQuestion` tool
|
|
25
|
+
// to a server pause and returns the human's answer as the tool result. Added by
|
|
26
|
+
// default to every agent (no-op unless the tool is granted + called).
|
|
27
|
+
export { createAskHumanProcessor, ASK_USER_QUESTION_TOOL } from "./ask-human.js";
|
|
@@ -56,6 +56,16 @@ const ACP_FALLBACK = Symbol("acp-fallback");
|
|
|
56
56
|
* ops tuning; defaults sane. */
|
|
57
57
|
export const ACP_HANDSHAKE_TIMEOUT_MS = Number(process.env.AC_ACP_HANDSHAKE_TIMEOUT_MS) || 10_000;
|
|
58
58
|
|
|
59
|
+
/** Idle deadline for a PROMPT TURN (distinct from the handshake gate above). A
|
|
60
|
+
* turn is killed only if it goes fully SILENT for this long — the deadline is
|
|
61
|
+
* re-armed on every streamed message, so a long, *streaming* turn never trips
|
|
62
|
+
* it. This must be generous: a reasoning model (GLM, gpt-5-codex) can think for
|
|
63
|
+
* tens of seconds between tool calls with no wire activity, which is NOT a hang.
|
|
64
|
+
* The 10s handshake timeout was far too tight here and killed live GLM turns
|
|
65
|
+
* mid-report. Only a genuinely wedged CLI (the Gemini-style hang) should trip
|
|
66
|
+
* this. Overridable via the env for ops tuning. */
|
|
67
|
+
export const ACP_TURN_IDLE_TIMEOUT_MS = Number(process.env.AC_ACP_TURN_IDLE_TIMEOUT_MS) || 120_000;
|
|
68
|
+
|
|
59
69
|
/** Readiness gate for the LIVE ACP attempt. The duplex-stdin transport in
|
|
60
70
|
* `spawnAcpProcess` is now real (`commands.spawnDuplex` on the local provider),
|
|
61
71
|
* so the agent's `initialize` request bytes are delivered and the handshake can
|
|
@@ -448,7 +458,7 @@ export class CliAgentRunner implements ModelExecutionContract {
|
|
|
448
458
|
let watchdog: ReturnType<typeof setTimeout> | undefined;
|
|
449
459
|
const armWatchdog = () => {
|
|
450
460
|
if (watchdog) clearTimeout(watchdog);
|
|
451
|
-
watchdog = setTimeout(() => { stalled = true; peer.cancel(); },
|
|
461
|
+
watchdog = setTimeout(() => { stalled = true; peer.cancel(); }, ACP_TURN_IDLE_TIMEOUT_MS);
|
|
452
462
|
};
|
|
453
463
|
try {
|
|
454
464
|
armWatchdog();
|
|
@@ -466,7 +476,7 @@ export class CliAgentRunner implements ModelExecutionContract {
|
|
|
466
476
|
// and surface the stall as an error.
|
|
467
477
|
await this.teardownAcp(proc, peer);
|
|
468
478
|
this.acpSession = undefined;
|
|
469
|
-
yield { type: "error", text: `${this.spec.kind} prompt turn stalled (no activity for ${
|
|
479
|
+
yield { type: "error", text: `${this.spec.kind} prompt turn stalled (no activity for ${ACP_TURN_IDLE_TIMEOUT_MS}ms)`, timestamp: now() };
|
|
470
480
|
return;
|
|
471
481
|
}
|
|
472
482
|
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Cursor CLI runtime — drives Cursor's `cursor-agent` inside the sandbox.
|
|
3
|
+
*
|
|
4
|
+
* ACP-native: `cursor-agent acp` is a protocolVersion-1 ACP server (verified
|
|
5
|
+
* live on E2B 2026-06-30), so the runner delegates the wire protocol to
|
|
6
|
+
* AcpClientPeer; the JSONL members below are the version-mismatch fallback.
|
|
7
|
+
*
|
|
8
|
+
* Auth: CURSOR_API_KEY — Cursor's OWN platform key, NOT OpenRouter. In ACP mode
|
|
9
|
+
* `cursor-agent acp` takes no model flag, so it runs the account's default model;
|
|
10
|
+
* the `--model` ids (auto, gpt-5.3-codex, composer-2.5,
|
|
11
|
+
* claude-opus-4-8-thinking-high; full list via `cursor-agent --list-models`)
|
|
12
|
+
* only apply to the JSONL `-p` fallback. cursor brings HARNESS diversity (a
|
|
13
|
+
* different agent scaffold) to a cross-functional / review panel.
|
|
14
|
+
*
|
|
15
|
+
* Verified live on E2B (2026-06-30): `curl https://cursor.com/install` →
|
|
16
|
+
* ~/.local/bin/cursor-agent (v2026.06.29); `cursor-agent acp` answered the ACP
|
|
17
|
+
* `initialize` with protocolVersion 1; CURSOR_API_KEY authenticated.
|
|
18
|
+
*/
|
|
19
|
+
import type { AgentMessage } from "../index.js";
|
|
20
|
+
import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
|
|
21
|
+
|
|
22
|
+
function now(): string { return new Date().toISOString(); }
|
|
23
|
+
|
|
24
|
+
export const cursorSpec: CliAgentSpec = {
|
|
25
|
+
kind: "cursor",
|
|
26
|
+
authEnv: "CURSOR_API_KEY",
|
|
27
|
+
bin: "cursor-agent",
|
|
28
|
+
// ACP runs the account default; "auto" is Cursor's own auto-routing label.
|
|
29
|
+
defaultModel: "auto",
|
|
30
|
+
// `cursor-agent acp` is a protocolVersion-1 ACP server — the runner delegates
|
|
31
|
+
// the whole wire protocol to AcpClientPeer. It takes no model flag, so the
|
|
32
|
+
// session runs Cursor's account-default model.
|
|
33
|
+
acp: { command: "cursor-agent", args: ["acp"] },
|
|
34
|
+
// Self-install on first use; symlink onto PATH for a non-login `sh -c`.
|
|
35
|
+
install: 'curl https://cursor.com/install -fsS | bash && (command -v cursor-agent >/dev/null 2>&1 || sudo ln -sf "$HOME/.local/bin/cursor-agent" /usr/local/bin/cursor-agent)',
|
|
36
|
+
// ── JSONL fallback (only if the ACP handshake negotiates a non-1 version;
|
|
37
|
+
// cursor is v1, so vestigial). `-p` needs `--force` to clear the
|
|
38
|
+
// workspace-trust gate non-interactively.
|
|
39
|
+
promptPayload: (prompt) => prompt,
|
|
40
|
+
buildCommand: ({ promptPath, model, cwd }) =>
|
|
41
|
+
`${cwd ? `cd ${shellQuote(cwd)} && ` : ""}cursor-agent -p --force ${model ? `--model ${shellQuote(model)} ` : ""}--output-format text "$(cat ${shellQuote(promptPath)})"`,
|
|
42
|
+
extractSessionId: () => undefined,
|
|
43
|
+
mapEvent: (p): AgentMessage[] => {
|
|
44
|
+
const ts = now();
|
|
45
|
+
const text = typeof p.text === "string" ? p.text : typeof p.content === "string" ? p.content : "";
|
|
46
|
+
return text ? [{ type: "text", text, timestamp: ts }] : [];
|
|
47
|
+
},
|
|
48
|
+
};
|
|
49
|
+
|
|
50
|
+
export interface CursorRuntimeConfig {
|
|
51
|
+
/** Cursor model id; ACP mode ignores it (account default) — applies to the `-p` fallback. */
|
|
52
|
+
model?: string;
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function createCursorRuntime(config: CursorRuntimeConfig = {}) {
|
|
56
|
+
return createCliAgentRuntime(cursorSpec, config.model ?? cursorSpec.defaultModel);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export default createCursorRuntime();
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Factory `droid` runtime — drives `droid exec` headless inside the sandbox,
|
|
3
|
+
* JSONL via `--output-format json`.
|
|
4
|
+
*
|
|
5
|
+
* Driven PURELY via OpenRouter BYOK — NO Factory login (verified live on E2B
|
|
6
|
+
* 2026-06-30): a `~/.factory/settings.json` `customModels` entry points at
|
|
7
|
+
* OpenRouter, and the model id is `custom:<displayName>-<index>`. The workflow
|
|
8
|
+
* provisions settings.json (see the dynamic-task route step); this spec just
|
|
9
|
+
* builds the exec command. Auth env is OPENROUTER_API_KEY (the BYOK inference
|
|
10
|
+
* key shared with codex + opencode).
|
|
11
|
+
*
|
|
12
|
+
* NOT ACP: `droid exec` is JSONL. (Its stream-jsonrpc mode ignores --model and
|
|
13
|
+
* sets model/autonomy via JSON-RPC; the plain `--output-format json` mode honours
|
|
14
|
+
* --model, which is what we use.) So it always drives the JSONL path. droid adds
|
|
15
|
+
* Factory's agent HARNESS to a cross-functional / review panel.
|
|
16
|
+
*
|
|
17
|
+
* Verified live on E2B (2026-06-30): install via app.factory.ai/cli;
|
|
18
|
+
* `droid exec --auto low --model custom:GLM-5.2-OR-0 --output-format json`
|
|
19
|
+
* returned `{"type":"result","result":"…","session_id":"…","usage":{…}}` driven
|
|
20
|
+
* through OpenRouter with no Factory account.
|
|
21
|
+
*/
|
|
22
|
+
import type { AgentMessage } from "../index.js";
|
|
23
|
+
import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
|
|
24
|
+
|
|
25
|
+
function now(): string { return new Date().toISOString(); }
|
|
26
|
+
|
|
27
|
+
export const droidSpec: CliAgentSpec = {
|
|
28
|
+
kind: "droid",
|
|
29
|
+
// OpenRouter is the inference gateway (BYOK custom model). No FACTORY_API_KEY.
|
|
30
|
+
authEnv: "OPENROUTER_API_KEY",
|
|
31
|
+
bin: "droid",
|
|
32
|
+
// Matches the first customModels entry the workflow writes to settings.json.
|
|
33
|
+
defaultModel: "custom:GLM-5.2-OR-0",
|
|
34
|
+
install: 'curl -fsSL https://app.factory.ai/cli | sh && (command -v droid >/dev/null 2>&1 || sudo ln -sf "$HOME/.local/bin/droid" /usr/local/bin/droid)',
|
|
35
|
+
// ── JSONL path (always — droid exec is not ACP). `--auto medium` lets the
|
|
36
|
+
// agent create/edit files + run commands; `-f` reads the prompt from a file.
|
|
37
|
+
promptPayload: (prompt) => prompt,
|
|
38
|
+
buildCommand: ({ promptPath, model, cwd }) =>
|
|
39
|
+
`${cwd ? `cd ${shellQuote(cwd)} && ` : ""}droid exec --auto medium ${model ? `--model ${shellQuote(model)} ` : ""}--output-format json -f ${shellQuote(promptPath)}`,
|
|
40
|
+
extractSessionId: (p) => (typeof p.session_id === "string" ? p.session_id : undefined),
|
|
41
|
+
mapEvent: (p): AgentMessage[] => {
|
|
42
|
+
const ts = now();
|
|
43
|
+
// `--output-format json` emits a single terminal result object.
|
|
44
|
+
const text =
|
|
45
|
+
p.type === "result" && typeof p.result === "string" ? p.result
|
|
46
|
+
: typeof p.text === "string" ? p.text
|
|
47
|
+
: typeof p.content === "string" ? p.content
|
|
48
|
+
: "";
|
|
49
|
+
return text ? [{ type: "text", text, timestamp: ts }] : [];
|
|
50
|
+
},
|
|
51
|
+
// No `acp` → JSONL-only spec.
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
export interface DroidRuntimeConfig {
|
|
55
|
+
/** `custom:<displayName>-<index>` matching the provisioned settings.json. */
|
|
56
|
+
model?: string;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
export function createDroidRuntime(config: DroidRuntimeConfig = {}) {
|
|
60
|
+
return createCliAgentRuntime(droidSpec, config.model ?? droidSpec.defaultModel);
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export default createDroidRuntime();
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenCode CLI runtime — drives sst's `opencode` agentic CLI inside the sandbox.
|
|
3
|
+
* OpenCode speaks ACP natively (`opencode acp`, protocolVersion 1 — verified
|
|
4
|
+
* live on E2B), so the runner drives it over ACP; the JSONL members below are
|
|
5
|
+
* only the version-mismatch fallback (vestigial for a v1 agent).
|
|
6
|
+
*
|
|
7
|
+
* Auth + model via OpenRouter: set OPENROUTER_API_KEY (a factory/workflow
|
|
8
|
+
* secret) and use a model id like `openrouter/z-ai/glm-5.2`. The runtime
|
|
9
|
+
* installs the `opencode-ai` CLI on demand; pair with
|
|
10
|
+
* `snapshots: { bootFrom: "reuse" }` to install once and boot from the capture.
|
|
11
|
+
*
|
|
12
|
+
* Verified live on E2B (2026-06-30): `npm i -g opencode-ai` (v1.17.12);
|
|
13
|
+
* `opencode run --model openrouter/z-ai/glm-5.2` drove a GLM-5.2 turn through
|
|
14
|
+
* OpenRouter; `opencode acp` answered the ACP `initialize` handshake with
|
|
15
|
+
* protocolVersion 1.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import type { AgentMessage } from "../index.js";
|
|
19
|
+
import { createCliAgentRuntime, shellQuote, type CliAgentSpec } from "./_cli-agent.js";
|
|
20
|
+
|
|
21
|
+
function now(): string { return new Date().toISOString(); }
|
|
22
|
+
|
|
23
|
+
export const opencodeSpec: CliAgentSpec = {
|
|
24
|
+
kind: "opencode",
|
|
25
|
+
// OpenRouter is the gateway: opencode reads OPENROUTER_API_KEY from the env
|
|
26
|
+
// and serves any `openrouter/<provider>/<model>` id (e.g. z-ai/glm-5.2).
|
|
27
|
+
authEnv: "OPENROUTER_API_KEY",
|
|
28
|
+
bin: "opencode",
|
|
29
|
+
defaultModel: "openrouter/z-ai/glm-5.2",
|
|
30
|
+
// ACP-mode invocation — `opencode acp` is a protocolVersion-1 ACP server, so
|
|
31
|
+
// the runner delegates the whole wire protocol to AcpClientPeer. The model is
|
|
32
|
+
// resolved from opencode's config / the `--model` it was started with; the
|
|
33
|
+
// run provisioning writes the OpenRouter default so ACP turns use GLM-5.2.
|
|
34
|
+
acp: { command: "opencode", args: ["acp"] },
|
|
35
|
+
// Global npm install; symlink onto PATH only if the global bin dir isn't
|
|
36
|
+
// already there (so a non-login `sh -c` can find it).
|
|
37
|
+
install: 'sudo npm install -g opencode-ai && (command -v opencode >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/opencode" /usr/local/bin/opencode)',
|
|
38
|
+
// ── JSONL fallback (only reached if the ACP handshake negotiates a non-1
|
|
39
|
+
// version; opencode is v1, so this is vestigial). `opencode run` prints
|
|
40
|
+
// human-formatted text, so we capture the prompt round-trip as one message.
|
|
41
|
+
promptPayload: (prompt) => prompt,
|
|
42
|
+
buildCommand: ({ promptPath, model, cwd }) =>
|
|
43
|
+
`${cwd ? `cd ${shellQuote(cwd)} && ` : ""}opencode run ${model ? `--model ${shellQuote(model)} ` : ""}"$(cat ${shellQuote(promptPath)})"`,
|
|
44
|
+
extractSessionId: () => undefined,
|
|
45
|
+
mapEvent: (p): AgentMessage[] => {
|
|
46
|
+
const ts = now();
|
|
47
|
+
const text = typeof p.text === "string" ? p.text : typeof p.content === "string" ? p.content : "";
|
|
48
|
+
return text ? [{ type: "text", text, timestamp: ts }] : [];
|
|
49
|
+
},
|
|
50
|
+
};
|
|
51
|
+
|
|
52
|
+
export interface OpencodeRuntimeConfig {
|
|
53
|
+
/** OpenRouter-prefixed model id (default `openrouter/z-ai/glm-5.2`). */
|
|
54
|
+
model?: string;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export function createOpencodeRuntime(config: OpencodeRuntimeConfig = {}) {
|
|
58
|
+
return createCliAgentRuntime(opencodeSpec, config.model ?? opencodeSpec.defaultModel);
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export default createOpencodeRuntime();
|
package/src/sandbox.ts
CHANGED
|
@@ -301,6 +301,15 @@ export function makeSandboxProvider(sb: Sandbox | Desktop): SandboxProvider {
|
|
|
301
301
|
async write(path, content) {
|
|
302
302
|
await (sb.files.write as (p: string, d: string) => Promise<unknown>)(path, content);
|
|
303
303
|
},
|
|
304
|
+
// Read over the envd HTTP API (`Sandbox.files.read` → `GET /files`), a
|
|
305
|
+
// DIFFERENT transport from `commands` (the connect-web gRPC stream). A large
|
|
306
|
+
// readback over HTTP is decoded by `fetch`'s native `Content-Encoding`
|
|
307
|
+
// handling, so it is immune to the connect-web "received unsupported
|
|
308
|
+
// compressed output" failure that can abort `commands.run` output on a big
|
|
309
|
+
// frame — the property `launchStep` relies on to recover full logs.
|
|
310
|
+
async read(path) {
|
|
311
|
+
return await (sb.files.read as (p: string) => Promise<string>)(path);
|
|
312
|
+
},
|
|
304
313
|
},
|
|
305
314
|
// e2b 2.30 `kill()` returns Promise<boolean>; our provider contract is
|
|
306
315
|
// Promise<void>, so discard the result.
|
|
@@ -877,6 +886,9 @@ export function makeLocalSandboxProvider(): SandboxProvider {
|
|
|
877
886
|
await fs.mkdir(dirname(path), { recursive: true });
|
|
878
887
|
await fs.writeFile(path, content);
|
|
879
888
|
},
|
|
889
|
+
async read(path) {
|
|
890
|
+
return await fs.readFile(path, "utf8");
|
|
891
|
+
},
|
|
880
892
|
},
|
|
881
893
|
async kill() { /* caller IS the sandbox — killing it is the server's job */ },
|
|
882
894
|
};
|