@agent-compose/sdk 0.5.7 → 0.5.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/__tests__/run-agent-liveness.test.d.ts +17 -0
- package/dist/agent/agent-context.d.ts +67 -0
- package/dist/agent/agent-loop.d.ts +1 -0
- package/dist/client.d.ts +65 -2
- package/dist/index.d.ts +6 -3
- package/dist/index.js +226 -22
- package/dist/pause/wrappers.d.ts +7 -11
- package/dist/runtimes/openai-desktop.js +223 -22
- package/dist/sandbox.d.ts +24 -0
- package/dist/step-invocation/types.d.ts +1 -1
- package/dist/types/execution-context.d.ts +1 -3
- package/dist/types/workflow-metadata.d.ts +12 -1
- package/dist/types/workflow.d.ts +9 -2
- package/dist/utils/bundler.d.ts +3 -1
- package/dist/workflow-steps/workflow.d.ts +4 -1
- package/package.json +1 -1
- package/src/agent/agent-context.ts +212 -0
- package/src/agent/agent-loop.ts +31 -12
- package/src/agent/run-agent.ts +37 -1
- package/src/client.ts +89 -2
- package/src/index.ts +6 -2
- package/src/pause/wrappers.ts +7 -21
- package/src/sandbox.ts +78 -23
- package/src/step-invocation/types.ts +1 -1
- package/src/types/execution-context.ts +1 -3
- package/src/types/workflow-metadata.ts +14 -1
- package/src/types/workflow.ts +9 -3
- package/src/utils/bundler.ts +4 -1
- package/src/workflow-steps/workflow.ts +4 -1
- package/src/workflows/invoke-child.ts +7 -1
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Harness-agnostic agent context delivery.
|
|
3
|
+
*
|
|
4
|
+
* Every coding-agent harness we drive (Claude Code, Codex, Amp, Gemini, …)
|
|
5
|
+
* looks for an instruction file in its working directory — but they disagree
|
|
6
|
+
* on the NAME (Codex/Amp read `AGENTS.md`; Claude reads `CLAUDE.md`; Gemini
|
|
7
|
+
* reads `GEMINI.md`). So `agent()` writes the SAME platform manual under all
|
|
8
|
+
* three names at the agent's working dir, and every harness finds the one it
|
|
9
|
+
* knows. The manual is the single source of truth here; `base-env` bakes a
|
|
10
|
+
* static copy at `/workspace/AGENTS.md` for plain shell sessions, but the
|
|
11
|
+
* per-run copy `agent()` writes is the authoritative one — it carries the
|
|
12
|
+
* live connector list and lands at the run's working dir.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import type { SandboxProvider } from "../types/sandbox.js";
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* The platform manual delivered to every agent, regardless of harness.
|
|
19
|
+
* Covers the three things an agent must know: where files go (the factory
|
|
20
|
+
* drive + the persist-by-default working dir), how to pause for a human, and
|
|
21
|
+
* that credentials are network-injected (never in the env). The live
|
|
22
|
+
* "Connectors & access" section is appended per-run by `buildAgentContextDoc`.
|
|
23
|
+
*/
|
|
24
|
+
export const AGENT_COMPOSE_MANUAL = `# Working inside an Agent Compose sandbox
|
|
25
|
+
|
|
26
|
+
You are an agent running in a per-run sandbox on the Agent Compose platform.
|
|
27
|
+
Use the **\`agentc\` CLI** and the **\`@agent-compose/sdk\`** for everything below —
|
|
28
|
+
do NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on
|
|
29
|
+
your PATH and already authenticated from the environment
|
|
30
|
+
(\`AGENT_COMPOSE_URL\` / \`AGENT_COMPOSE_API_KEY\` / \`AGENT_COMPOSE_FACTORY\` are
|
|
31
|
+
injected for this run), so commands just work — no login, no keys to manage.
|
|
32
|
+
|
|
33
|
+
The \`/ac:*\` skills are installed as Claude Code slash commands (\`/ac:invoke\`,
|
|
34
|
+
\`/ac:events\`, \`/ac:logs\`, \`/ac:register\`, …) — reach for them too.
|
|
35
|
+
|
|
36
|
+
## Files — your outputs persist by default
|
|
37
|
+
|
|
38
|
+
Your working directory defaults to **\`\$AGENT_COMPOSE_RUN_DIR\`** — a per-run
|
|
39
|
+
directory on the shared factory drive
|
|
40
|
+
(\`\$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/\`) the platform
|
|
41
|
+
creates and attributes to this run. **Files you write here persist by
|
|
42
|
+
default** — they show up in the dashboard's Files tab and the run's Artifacts
|
|
43
|
+
card, with no API calls to save them. The dir already exists and is writable.
|
|
44
|
+
|
|
45
|
+
Need throwaway scratch — heavy build output, package caches, temp files?
|
|
46
|
+
\`cd /tmp\` (or any path outside \`/factory\`): anything off the factory drive is
|
|
47
|
+
ephemeral and discarded when the sandbox ends. In short: **stay in your working
|
|
48
|
+
dir to keep something, \`cd\` out to throw it away.**
|
|
49
|
+
|
|
50
|
+
The whole shared drive is POSIX-mounted at \`/factory\`; the dashboard-visible
|
|
51
|
+
root is \`\$AGENT_COMPOSE_FACTORY_DIR\` (\`/factory/files\`). Earlier versions and
|
|
52
|
+
runs live in sibling dirs under
|
|
53
|
+
\`\$AGENT_COMPOSE_FACTORY_DIR/\$AGENT_COMPOSE_WORKFLOW/\` — read them for prior
|
|
54
|
+
context. Other workflows' dirs are present but not your concern.
|
|
55
|
+
|
|
56
|
+
## Events — the factory timeline
|
|
57
|
+
|
|
58
|
+
Record something on the run/factory timeline (the dashboard renders these)
|
|
59
|
+
with the CLI — your run id is \`$RUN_ID\`:
|
|
60
|
+
|
|
61
|
+
agentc events send "$RUN_ID" <name> --summary "<one line>" [--body '<json>']
|
|
62
|
+
|
|
63
|
+
Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
|
|
64
|
+
\`agentc events list\` reads them back. \`/ac:events\` is the skill equivalent.
|
|
65
|
+
|
|
66
|
+
## Runs
|
|
67
|
+
|
|
68
|
+
agentc list # registered workflows (/ac:list)
|
|
69
|
+
agentc logs "$RUN_ID" # a run's logs (/ac:logs)
|
|
70
|
+
agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)
|
|
71
|
+
|
|
72
|
+
## Writing workflow / agent code — the SDK
|
|
73
|
+
|
|
74
|
+
\`@agent-compose/sdk\` is installed in \`/workspace\` — import it from any script
|
|
75
|
+
you write there:
|
|
76
|
+
|
|
77
|
+
import { defineWorkflow, agent, AgentComposeClient } from "@agent-compose/sdk";
|
|
78
|
+
|
|
79
|
+
Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
|
|
80
|
+
\`agentc register <file.ts>\` (or \`/ac:register\`).
|
|
81
|
+
|
|
82
|
+
## Pausing to ask the human — \`agentc pause\`
|
|
83
|
+
|
|
84
|
+
When you can't or shouldn't proceed without a human, run \`agentc pause\`. It
|
|
85
|
+
blocks until they answer on the dashboard, then prints their answer to stdout:
|
|
86
|
+
|
|
87
|
+
ANSWER=$(agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
|
|
88
|
+
--option retry --option skip)
|
|
89
|
+
|
|
90
|
+
Reach for it the moment you hit — or foresee — any of these:
|
|
91
|
+
- **A wall only a human can clear:** a 401/403, a missing credential, an
|
|
92
|
+
unconnected provider, a host the network refuses. Do NOT retry blindly or try
|
|
93
|
+
to work around it — pause and say what needs enabling.
|
|
94
|
+
- **A durable or outward-facing action that needs sign-off:** registering a
|
|
95
|
+
workflow, deploying, sending email/messages, deleting or overwriting shared
|
|
96
|
+
data, spending money. Prepare everything, then pause for approval BEFORE you
|
|
97
|
+
commit it.
|
|
98
|
+
- **A judgment call only the human can settle:** an under-specified request,
|
|
99
|
+
several valid paths, a conflict with existing state, missing input only they have.
|
|
100
|
+
|
|
101
|
+
You compose the \`--reason\` (the ask) yourself; pass \`--option\` choices when
|
|
102
|
+
there are clear ones, omit them for a free-form answer. Read the printed answer
|
|
103
|
+
and act on it. Each agent pauses independently — pausing doesn't stop the others.
|
|
104
|
+
|
|
105
|
+
## Credentials
|
|
106
|
+
|
|
107
|
+
Connector credentials (Google, GitHub, …) are NEVER in your environment.
|
|
108
|
+
They're injected at the network layer when you call an allowed host — make the
|
|
109
|
+
request **without** an Authorization header and the platform adds it. Don't try
|
|
110
|
+
to read or exfiltrate tokens; they aren't here. The "Connectors & access"
|
|
111
|
+
section below (when present) lists exactly which providers this run can reach.
|
|
112
|
+
|
|
113
|
+
## Tools in this environment
|
|
114
|
+
|
|
115
|
+
- \`agentc\` — Agent Compose CLI (your primary interface; authed from env)
|
|
116
|
+
- \`@agent-compose/sdk\` — installed in /workspace for writing workflows
|
|
117
|
+
- \`/ac:*\` Claude Code skills — slash commands for the above
|
|
118
|
+
- \`archil\` (factory drive), \`rtk\`, \`bun\`
|
|
119
|
+
- A world-writable \`/workspace\` working directory`;
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* One connector this run can reach, as the agent should see it. Strictly
|
|
123
|
+
* NON-SECRET — hosts, methods, paths, identity only. The access token is
|
|
124
|
+
* injected at the network layer and never appears here. The server builds
|
|
125
|
+
* this list at dispatch from the run's connector grants × the provider
|
|
126
|
+
* catalogue and delivers it as the `AGENT_COMPOSE_CONNECTORS` env (JSON array).
|
|
127
|
+
*/
|
|
128
|
+
export interface AgentConnectorInfo {
|
|
129
|
+
/** Provider key (`github`, `notion`, …). */
|
|
130
|
+
provider: string;
|
|
131
|
+
/** Human label ("GitHub", "Notion"). */
|
|
132
|
+
name?: string;
|
|
133
|
+
/** API hosts the credential is injected for. */
|
|
134
|
+
hosts?: string[];
|
|
135
|
+
/** Allowed HTTP methods (Tier-2 narrowing). Empty/absent = any. */
|
|
136
|
+
methods?: string[];
|
|
137
|
+
/** Allowed path prefixes (Tier-2 narrowing). Empty/absent = any. */
|
|
138
|
+
pathPrefixes?: string[];
|
|
139
|
+
/** GitHub: the repository the minted token is scoped to. */
|
|
140
|
+
repository?: string;
|
|
141
|
+
/** Coarse capability the token was minted with. */
|
|
142
|
+
access?: string;
|
|
143
|
+
/** Human scope descriptions, when the provider declares them. */
|
|
144
|
+
scopes?: string[];
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/** Render the per-run "Connectors & access" markdown section, or "" when the
|
|
148
|
+
* run brokers no connectors. */
|
|
149
|
+
function renderConnectorsSection(connectors: AgentConnectorInfo[]): string {
|
|
150
|
+
if (connectors.length === 0) return "";
|
|
151
|
+
const rows = connectors.map((c) => {
|
|
152
|
+
const host = c.hosts?.length ? c.hosts.join(", ") : "(host set by the platform)";
|
|
153
|
+
const verbs = c.methods?.length ? c.methods.join("/") : "any method";
|
|
154
|
+
const paths = c.pathPrefixes?.length ? ` under ${c.pathPrefixes.join(", ")}` : "";
|
|
155
|
+
const repo = c.repository ? ` — repo \`${c.repository}\` (${c.access ?? "read"})` : "";
|
|
156
|
+
const why = c.scopes?.length ? ` \n _scopes: ${c.scopes.join(", ")}_` : "";
|
|
157
|
+
return `- **${c.name ?? c.provider}** → \`${host}\` — ${verbs}${paths}${repo}${why}`;
|
|
158
|
+
});
|
|
159
|
+
return `
|
|
160
|
+
|
|
161
|
+
## Connectors & access — what this run can reach
|
|
162
|
+
|
|
163
|
+
These providers are connected for this run. Call their APIs with plain
|
|
164
|
+
fetch/SDKs and **no Authorization header** — the platform injects the
|
|
165
|
+
credential at the network layer. Requests outside the listed method/path are
|
|
166
|
+
refused (403) and the token withheld. Anything NOT listed is unreachable; if
|
|
167
|
+
you need it, \`agentc pause\` and ask for it to be connected.
|
|
168
|
+
|
|
169
|
+
${rows.join("\n")}
|
|
170
|
+
`;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
/** Compose the full per-run agent doc: the static manual + the live
|
|
174
|
+
* connectors section read from `AGENT_COMPOSE_CONNECTORS` (a JSON array;
|
|
175
|
+
* malformed/absent → no section). */
|
|
176
|
+
export function buildAgentContextDoc(env: Record<string, string | undefined>): string {
|
|
177
|
+
let connectors: AgentConnectorInfo[] = [];
|
|
178
|
+
const raw = env.AGENT_COMPOSE_CONNECTORS;
|
|
179
|
+
if (raw) {
|
|
180
|
+
try {
|
|
181
|
+
const parsed: unknown = JSON.parse(raw);
|
|
182
|
+
if (Array.isArray(parsed)) connectors = parsed as AgentConnectorInfo[];
|
|
183
|
+
} catch { /* malformed manifest — render the manual without a connectors section */ }
|
|
184
|
+
}
|
|
185
|
+
return AGENT_COMPOSE_MANUAL + renderConnectorsSection(connectors);
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/**
|
|
189
|
+
* Write the platform context at the agent's working dir under every harness's
|
|
190
|
+
* instruction-file name, so whichever CLI runs finds the one it reads. Codex
|
|
191
|
+
* and Amp read `AGENTS.md` natively; Claude reads `CLAUDE.md`; Gemini reads
|
|
192
|
+
* `GEMINI.md` — we write identical content to all three rather than detect the
|
|
193
|
+
* harness (the runtime's `kind` isn't known until after spawn, and a few extra
|
|
194
|
+
* small files in our own run dir are harmless).
|
|
195
|
+
*
|
|
196
|
+
* Best-effort: a write failure logs and is swallowed — never fail an agent
|
|
197
|
+
* because its context file couldn't be written.
|
|
198
|
+
*/
|
|
199
|
+
export async function writeAgentContext(args: {
|
|
200
|
+
sandbox: Pick<SandboxProvider, "files">;
|
|
201
|
+
cwd: string;
|
|
202
|
+
env: Record<string, string | undefined>;
|
|
203
|
+
}): Promise<void> {
|
|
204
|
+
const doc = buildAgentContextDoc(args.env);
|
|
205
|
+
const dir = args.cwd.replace(/\/+$/, "") || "/workspace";
|
|
206
|
+
// AGENTS.md is the cross-harness standard; CLAUDE.md / GEMINI.md are the
|
|
207
|
+
// per-harness names. Same content under each — the harness that doesn't read
|
|
208
|
+
// a given name simply ignores it.
|
|
209
|
+
for (const name of ["AGENTS.md", "CLAUDE.md", "GEMINI.md"]) {
|
|
210
|
+
await args.sandbox.files.write(`${dir}/${name}`, doc);
|
|
211
|
+
}
|
|
212
|
+
}
|
package/src/agent/agent-loop.ts
CHANGED
|
@@ -18,7 +18,13 @@ import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./st
|
|
|
18
18
|
export const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
|
|
19
19
|
|
|
20
20
|
const SAME_BLOCKER_ITERATIONS = 3;
|
|
21
|
-
|
|
21
|
+
// Consecutive turns with NEITHER a <status> NOR a <response> before the loop
|
|
22
|
+
// declares the agent wedged. Kept generous because the most common benign
|
|
23
|
+
// cause is an agent that launched a useful BACKGROUND job and is waiting to be
|
|
24
|
+
// "notified" — the corrective re-prompt (below) tells it to poll synchronously,
|
|
25
|
+
// and these extra turns give that nudge (and the job) time to land before we
|
|
26
|
+
// give up.
|
|
27
|
+
const STALL_ITERATIONS = 6;
|
|
22
28
|
const MESSAGE_PREVIEW_CHARS = 400;
|
|
23
29
|
|
|
24
30
|
export function parseAgentStatus(text: string): AgentStatus | null {
|
|
@@ -47,7 +53,7 @@ export type AgentMessageSummary =
|
|
|
47
53
|
| { type: "thinking"; text: string }
|
|
48
54
|
| { type: "tool_use"; toolName: string; toolUseId: string; toolInput: Record<string, unknown>; toolInputPreview: string }
|
|
49
55
|
| { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
|
|
50
|
-
| { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number }
|
|
56
|
+
| { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number; model?: string }
|
|
51
57
|
| { type: "done"; sessionId: string }
|
|
52
58
|
| { type: "error"; text: string };
|
|
53
59
|
|
|
@@ -413,7 +419,13 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
413
419
|
}
|
|
414
420
|
const msg = outputVerdict.value;
|
|
415
421
|
opts.onAgentEvent?.(iteration, msg);
|
|
416
|
-
|
|
422
|
+
// Usage summaries carry the resolved model so the server can price
|
|
423
|
+
// token rows per model without correlating back to agent.spawned.
|
|
424
|
+
const summary = summarizeAgentMessage(msg);
|
|
425
|
+
opts.onAgentLifecycleEvent?.({
|
|
426
|
+
event: "agent.message", at: Date.now(), agentId, label, iteration: iteration + 1,
|
|
427
|
+
message: summary.type === "usage" && client.model != null ? { ...summary, model: client.model } : summary,
|
|
428
|
+
});
|
|
417
429
|
if (msg.type === "init") lastSessionId = msg.sessionId;
|
|
418
430
|
if (msg.type === "text") responseText += msg.text;
|
|
419
431
|
if (msg.type === "error") throw new Error(`Agent error: ${msg.text}`);
|
|
@@ -511,15 +523,22 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
511
523
|
if (!status && rawResponse === null) {
|
|
512
524
|
if (++iterationsWithoutStatus >= STALL_ITERATIONS)
|
|
513
525
|
throw new Error(`${logLabel} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
|
|
514
|
-
// EMPTY-OUTPUT RE-PROMPT:
|
|
515
|
-
//
|
|
516
|
-
//
|
|
517
|
-
//
|
|
518
|
-
//
|
|
519
|
-
//
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
526
|
+
// EMPTY-OUTPUT RE-PROMPT: a turn that emits neither <status> nor
|
|
527
|
+
// <response>. Refund the iteration and re-prompt with explicit feedback
|
|
528
|
+
// (bounded by the shared retry pool); the stall counter above still
|
|
529
|
+
// hard-bounds genuinely wedged agents. The directive call-out about
|
|
530
|
+
// BACKGROUND jobs is load-bearing: the dominant benign cause is an agent
|
|
531
|
+
// that ran `cmd &` and parked itself "waiting to be notified" — the loop
|
|
532
|
+
// delivers no such notification, so it must poll synchronously instead.
|
|
533
|
+
// Fires with or without a responseSchema (an agent with no schema still
|
|
534
|
+
// owes a <status>).
|
|
535
|
+
if (schemaRetriesLeft-- > 0) {
|
|
536
|
+
lastResponseValidationError =
|
|
537
|
+
"Your turn ended with no <status> block" + (opts.responseSchema ? " (and no <response> block)" : "") + ". " +
|
|
538
|
+
"If you launched a background job (`… &`) and are waiting to be notified when it finishes — STOP: nothing will notify you here. " +
|
|
539
|
+
"Poll it NOW (read its output / wait for it synchronously to completion), then emit your <status>" +
|
|
540
|
+
(opts.responseSchema ? " and the complete <response> JSON." : ".");
|
|
541
|
+
process.stdout.write(`${logLabel} empty turn (no <status>) — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
|
|
523
542
|
iteration--;
|
|
524
543
|
continue;
|
|
525
544
|
}
|
package/src/agent/run-agent.ts
CHANGED
|
@@ -21,6 +21,7 @@ import { AsyncQueue } from "./async-queue.js";
|
|
|
21
21
|
import type { AgentMessage, AgentStatus } from "../types/protocol.js";
|
|
22
22
|
import type { AgentRuntime, RuntimeOptions } from "../types/runtime.js";
|
|
23
23
|
import type { SandboxProvider } from "../types/sandbox.js";
|
|
24
|
+
import { writeAgentContext } from "./agent-context.js";
|
|
24
25
|
import type { AgentBudget } from "../types/workflow.js";
|
|
25
26
|
import type { Processor } from "../processors/processor.js";
|
|
26
27
|
import { RequestContext } from "../request-context/request-context.js";
|
|
@@ -232,7 +233,42 @@ export function resolveAgentId(explicitId?: string): string {
|
|
|
232
233
|
}
|
|
233
234
|
|
|
234
235
|
export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopResult<T>> {
|
|
235
|
-
|
|
236
|
+
// Persist-by-default: an agent's working dir is the run dir on the factory
|
|
237
|
+
// drive (the mount step creates it ahead of the run), so everything it
|
|
238
|
+
// writes is kept and attributed to this run; `cd /tmp` for throwaway
|
|
239
|
+
// scratch. Falls back to /workspace when this run has no factory drive.
|
|
240
|
+
// Never "" — an empty cwd is the one case the harnesses' instruction-file
|
|
241
|
+
// upward-walk can't resolve, and it left codex/amp without their AGENTS.md.
|
|
242
|
+
let workingDir = opts.workingDir || process.env.AGENT_COMPOSE_RUN_DIR || "/workspace";
|
|
243
|
+
|
|
244
|
+
// Deliver the platform context (file conventions, connectors & access, how to
|
|
245
|
+
// pause) as AGENTS.md / CLAUDE.md / GEMINI.md at the working dir — and use the
|
|
246
|
+
// write as a LIVENESS PROBE of the working dir. AGENT_COMPOSE_RUN_DIR points at
|
|
247
|
+
// the /factory FUSE drive, whose writes HANG (uninterruptible, no timeout) when
|
|
248
|
+
// the mount degraded — launching the agent there wedges it silently with zero
|
|
249
|
+
// output (no logs). Bound the write; on timeout/failure fall back to /workspace
|
|
250
|
+
// (always present + writable) so a degraded drive can never sink an agent.
|
|
251
|
+
const CONTEXT_WRITE_DEADLINE_MS = 15_000;
|
|
252
|
+
const tryWriteContext = (cwd: string): Promise<boolean> => {
|
|
253
|
+
// Capture the deadline timer so the fast path (write resolves first) can
|
|
254
|
+
// clear it in finally — otherwise each probe leaks a dangling 15s timer.
|
|
255
|
+
// Mirrors sandbox.ts's clear-in-finally pattern.
|
|
256
|
+
let timer: ReturnType<typeof setTimeout> | undefined;
|
|
257
|
+
return Promise.race([
|
|
258
|
+
writeAgentContext({ sandbox: opts.sandbox, cwd, env: process.env }).then(() => true),
|
|
259
|
+
new Promise<boolean>((resolve) => { timer = setTimeout(() => resolve(false), CONTEXT_WRITE_DEADLINE_MS); }),
|
|
260
|
+
]).catch((err: unknown) => {
|
|
261
|
+
console.error(`[agent] writeAgentContext(${cwd}) failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
262
|
+
return false;
|
|
263
|
+
}).finally(() => { if (timer) clearTimeout(timer); });
|
|
264
|
+
};
|
|
265
|
+
if (!(await tryWriteContext(workingDir)) && workingDir !== "/workspace") {
|
|
266
|
+
console.error(
|
|
267
|
+
`[agent] working dir ${workingDir} not writable within ${CONTEXT_WRITE_DEADLINE_MS}ms ` +
|
|
268
|
+
`(factory drive degraded?) — falling back to /workspace so the agent can run`);
|
|
269
|
+
workingDir = "/workspace";
|
|
270
|
+
await tryWriteContext(workingDir);
|
|
271
|
+
}
|
|
236
272
|
|
|
237
273
|
// Stable agentId for this invocation. Used by the inbox URL (server
|
|
238
274
|
// scopes pending messages by agentId), the agentLoop (which would
|
package/src/client.ts
CHANGED
|
@@ -14,10 +14,10 @@
|
|
|
14
14
|
import { ofetch } from "ofetch";
|
|
15
15
|
import { AgentComposeError } from "./errors.js";
|
|
16
16
|
import { parseSseStream } from "./sse.js";
|
|
17
|
-
import type { SandboxNetworkPolicy } from "./sandbox.js";
|
|
17
|
+
import type { SandboxNetworkPolicy, SandboxSize } from "./sandbox.js";
|
|
18
18
|
import type { RunEvent } from "./types/events.js";
|
|
19
19
|
import type { WorkflowPlan } from "./types/workflow-plan.js";
|
|
20
|
-
import type { SnapshotConfig, IOSchema, ConnectorRequirements, ConnectorOperationTag, InvokePolicy } from "./types/workflow-metadata.js";
|
|
20
|
+
import type { SnapshotConfig, IOSchema, ConnectorRequirements, ConnectorOperationTag, InvokePolicy, SandboxResources } from "./types/workflow-metadata.js";
|
|
21
21
|
import type { WorkflowManifest } from "./utils/bundler.js";
|
|
22
22
|
|
|
23
23
|
/** UUID-v4-ish — matches the server-side predicate. Used to auto-detect
|
|
@@ -103,6 +103,8 @@ export interface RegisterWorkflowInput {
|
|
|
103
103
|
/** All snapshot config — `bootFrom` (where to restore at run start),
|
|
104
104
|
* `save`, `retain`. See `WorkflowMetadata.snapshots`. */
|
|
105
105
|
snapshots?: SnapshotConfig;
|
|
106
|
+
/** Sandbox machine size (template default). See `WorkflowMetadata.resources`. */
|
|
107
|
+
resources?: SandboxResources;
|
|
106
108
|
/** Provider-neutral execution plan detected by the CLI bundler. */
|
|
107
109
|
workflowPlan?: WorkflowPlan;
|
|
108
110
|
/** Connector requirements declared via `defineWorkflow({ connectors })`
|
|
@@ -139,6 +141,10 @@ export interface InvokeWorkflowOptions {
|
|
|
139
141
|
* vars after brokering. Replaces the template-level placeholders for
|
|
140
142
|
* this run only — registered metadata is not mutated. */
|
|
141
143
|
placeholders?: Record<string, string>;
|
|
144
|
+
/** Per-invocation machine-size override of the template's `resources.size`.
|
|
145
|
+
* `small` (default) | `medium` | `large`; omit → the template default,
|
|
146
|
+
* else `small`. Honoured on Vercel (→ vCPUs); E2B ignores it. */
|
|
147
|
+
size?: SandboxSize;
|
|
142
148
|
/** Explicit parent run id. Pass `null` to suppress ambient RUN_ID auto-detection. */
|
|
143
149
|
parentRunId?: string | null;
|
|
144
150
|
/** Agent loop inside the parent run that caused this invoke, when applicable. */
|
|
@@ -179,6 +185,56 @@ export interface ListTemplatesOptions {
|
|
|
179
185
|
factorySlug?: string;
|
|
180
186
|
}
|
|
181
187
|
|
|
188
|
+
/** A human member of your team — the people an agent (or you) can @-flag. */
|
|
189
|
+
export interface TeamMember {
|
|
190
|
+
/** Membership row id. */
|
|
191
|
+
id: string;
|
|
192
|
+
/** The user id — what you pass to `createMentions({ mentionedUserIds })`. */
|
|
193
|
+
userId: string;
|
|
194
|
+
role: string;
|
|
195
|
+
email: string;
|
|
196
|
+
name: string;
|
|
197
|
+
joinedAt: string;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** A "you were flagged" ping, persisted server-side so it reaches the
|
|
201
|
+
* mentioned teammate in their Workbench. */
|
|
202
|
+
export interface Mention {
|
|
203
|
+
id: string;
|
|
204
|
+
factoryId: string;
|
|
205
|
+
mentionedUserId: string;
|
|
206
|
+
/** Who flagged: 'user' | 'api_key' | 'run' | 'system'. */
|
|
207
|
+
actorKind: string;
|
|
208
|
+
actorId: string | null;
|
|
209
|
+
actorLabel: string | null;
|
|
210
|
+
/** Where it lives: 'doc' | 'comment' | 'plan' | 'run'. */
|
|
211
|
+
contextKind: string;
|
|
212
|
+
contextPath: string | null;
|
|
213
|
+
/** Ready-made relative dashboard URL the Workbench card links to. */
|
|
214
|
+
contextUrl: string | null;
|
|
215
|
+
text: string;
|
|
216
|
+
runId: string | null;
|
|
217
|
+
seenAt: string | null;
|
|
218
|
+
resolvedAt: string | null;
|
|
219
|
+
createdAt: string;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
export interface CreateMentionsInput {
|
|
223
|
+
/** Team-member user ids to flag (1–20). Discover them via `listMembers()`.
|
|
224
|
+
* Non-members are dropped server-side. */
|
|
225
|
+
mentionedUserIds: string[];
|
|
226
|
+
/** The flag message shown in the teammate's Workbench. */
|
|
227
|
+
text: string;
|
|
228
|
+
contextKind: "doc" | "comment" | "plan" | "run";
|
|
229
|
+
/** Factory-relative file path or comment thread id, when applicable. */
|
|
230
|
+
contextPath?: string;
|
|
231
|
+
/** Ready-made relative dashboard URL the Workbench card links to (e.g.
|
|
232
|
+
* `/factories/<slug>/files/view?path=<plan>`). */
|
|
233
|
+
contextUrl?: string;
|
|
234
|
+
runId?: string;
|
|
235
|
+
factorySlug?: string;
|
|
236
|
+
}
|
|
237
|
+
|
|
182
238
|
export interface CreateFactoryInput {
|
|
183
239
|
slug: string;
|
|
184
240
|
name: string;
|
|
@@ -661,6 +717,7 @@ export class AgentComposeClient {
|
|
|
661
717
|
...(opts?.snapshots !== undefined ? { snapshots: opts.snapshots } : {}),
|
|
662
718
|
...(opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {}),
|
|
663
719
|
...(opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {}),
|
|
720
|
+
...(opts?.size !== undefined ? { size: opts.size } : {}),
|
|
664
721
|
...(parentRunId ? { parentRunId } : {}),
|
|
665
722
|
...(opts?.agentId ? { agentId: opts.agentId } : {}),
|
|
666
723
|
},
|
|
@@ -1014,6 +1071,36 @@ export class AgentComposeClient {
|
|
|
1014
1071
|
);
|
|
1015
1072
|
}
|
|
1016
1073
|
|
|
1074
|
+
/** List the human members of your team — the people you (or an agent) can
|
|
1075
|
+
* @-flag with `createMentions`. Each row's `userId` is what
|
|
1076
|
+
* `mentionedUserIds` expects. */
|
|
1077
|
+
async listMembers(): Promise<TeamMember[]> {
|
|
1078
|
+
const body = await this.fetch<{ members: TeamMember[] }>("/api/v1/team/members");
|
|
1079
|
+
return body.members;
|
|
1080
|
+
}
|
|
1081
|
+
|
|
1082
|
+
/** Flag one or more teammates — a durable ping that lands in their factory
|
|
1083
|
+
* Workbench. Use from an agent (e.g. a remediation plan that needs a human
|
|
1084
|
+
* to rotate a secret) or any team automation. Resolve `mentionedUserIds`
|
|
1085
|
+
* via `listMembers()`. When run inside a sandbox the run-callback token is
|
|
1086
|
+
* forwarded so the ping is attributed to the run ("flagged by <workflow>"). */
|
|
1087
|
+
async createMentions(input: CreateMentionsInput): Promise<Mention[]> {
|
|
1088
|
+
const factorySlug = input.factorySlug
|
|
1089
|
+
?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined)
|
|
1090
|
+
?? DEFAULT_FACTORY;
|
|
1091
|
+
const runToken = typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_RUN_TOKEN : undefined;
|
|
1092
|
+
const { factorySlug: _omit, ...payload } = input;
|
|
1093
|
+
const body = await this.fetch<{ mentions: Mention[] }>(
|
|
1094
|
+
`/api/v1/factories/${encodeURIComponent(factorySlug)}/mentions`,
|
|
1095
|
+
{
|
|
1096
|
+
method: "POST",
|
|
1097
|
+
body: payload,
|
|
1098
|
+
...(runToken ? { headers: { "x-run-token": runToken } } : {}),
|
|
1099
|
+
},
|
|
1100
|
+
);
|
|
1101
|
+
return body.mentions;
|
|
1102
|
+
}
|
|
1103
|
+
|
|
1017
1104
|
/** List events ingested into a factory, newest first. Supports
|
|
1018
1105
|
* case-insensitive substring filter (`name`) and timestamp-cursor
|
|
1019
1106
|
* pagination (`before`). Returns `{ events, has_more }` — the
|
package/src/index.ts
CHANGED
|
@@ -112,6 +112,7 @@ export type {
|
|
|
112
112
|
CreateFactoryInput, UpdateFactoryInput,
|
|
113
113
|
SecretOptions, SetSecretResult, SecretListEntry,
|
|
114
114
|
CreateApiKeyInput, StreamRunLogsOptions,
|
|
115
|
+
TeamMember, Mention, CreateMentionsInput,
|
|
115
116
|
EventSubjectType, EventRow, ReportEventInput, ListEventsOptions, ListEventsResult,
|
|
116
117
|
RunLogLine, ListRunLogsOptions,
|
|
117
118
|
RegisteredRuntime, RunState, RunStatus, FactoryRow, SnapshotListEntry, SnapshotListResponse,
|
|
@@ -185,9 +186,10 @@ export { SandboxUnavailableError, SANDBOX_UNAVAILABLE_PREFIX } from "./sandbox-e
|
|
|
185
186
|
export type {
|
|
186
187
|
SandboxCreateOpts, SandboxNetworkPolicy, SandboxNetworkHeaderTransform,
|
|
187
188
|
SandboxNetworkAllowRule, SandboxNetworkSubnetPolicy, SandboxProviderName,
|
|
188
|
-
SandboxQuotaResult, OwnedSandboxResult, OwnedSandbox,
|
|
189
|
+
SandboxQuotaResult, OwnedSandboxResult, OwnedSandbox, SandboxSize,
|
|
189
190
|
ParseSseExecStreamOptions, SandboxCommandRunOptions, SandboxCommandResult,
|
|
190
191
|
} from "./sandbox.js";
|
|
192
|
+
export type { SandboxResources } from "./types/workflow-metadata.js";
|
|
191
193
|
|
|
192
194
|
// Workflow engine
|
|
193
195
|
export { runWorkflow, WorkflowError, EngineError, classifyError, parseNameVersion } from "./workflows/engine.js";
|
|
@@ -249,7 +251,7 @@ export {
|
|
|
249
251
|
} from "./pause/errors.js";
|
|
250
252
|
export type { PauseErrorCode } from "./pause/errors.js";
|
|
251
253
|
export type { PauseRequest } from "./pause/pause-core.js";
|
|
252
|
-
export type {
|
|
254
|
+
export type { WaitForEventRequest } from "./pause/wrappers.js";
|
|
253
255
|
export type {
|
|
254
256
|
StepRequest,
|
|
255
257
|
StepResult,
|
|
@@ -265,5 +267,7 @@ export { agentLoop, parseAgentStatus, DEFAULT_CLAUDE_MODEL } from "./agent/agent
|
|
|
265
267
|
export type { AgentLifecycleEvent, AgentLoopOpts, AgentLoopResult } from "./agent/agent-loop.js";
|
|
266
268
|
export { agent } from "./agent/run-agent.js";
|
|
267
269
|
export type { AgentOpts } from "./agent/run-agent.js";
|
|
270
|
+
export { AGENT_COMPOSE_MANUAL, buildAgentContextDoc, writeAgentContext } from "./agent/agent-context.js";
|
|
271
|
+
export type { AgentConnectorInfo } from "./agent/agent-context.js";
|
|
268
272
|
export { AgentMessageSchema, parseAgentResponse } from "./agent/protocol.js";
|
|
269
273
|
export { importSourceModule, TMP_DIR, LATEST_VERSION } from "./utils/source-loader.js";
|
package/src/pause/wrappers.ts
CHANGED
|
@@ -1,12 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The
|
|
2
|
+
* The two opinionated pause wrappers (ADR-0006 §"SDK surface"), each a thin
|
|
3
3
|
* closure over `ctx.pause`:
|
|
4
4
|
*
|
|
5
|
-
* - `requestDecision` — pause for a typed human/agent decision (schema required).
|
|
6
5
|
* - `sleep` — a lightweight timed pause; resolves on its own TTL,
|
|
7
6
|
* skips the snapshot, returns void.
|
|
8
7
|
* - `waitForEvent` — pause until an event resumes by correlation key.
|
|
9
8
|
*
|
|
9
|
+
* A typed human/agent decision is NOT a wrapper — it's a plain `ctx.pause`
|
|
10
|
+
* with a `schema` (and `payload.options` for the dashboard's answer UI). The
|
|
11
|
+
* pause primitive carries the reason (the ask) and the resume value (the
|
|
12
|
+
* resolution); there is no separate `requestDecision`.
|
|
13
|
+
*
|
|
10
14
|
* Built from a `PauseFn` so the wrapper logic lives in one place and the step
|
|
11
15
|
* runner just spreads them onto the context next to `pause`.
|
|
12
16
|
*/
|
|
@@ -19,18 +23,10 @@ import type { PauseRequest } from "./pause-core.js";
|
|
|
19
23
|
export type PauseFn = <T = unknown>(req: PauseRequest<T>) => Promise<T>;
|
|
20
24
|
|
|
21
25
|
/** Internal: pause with an explicit wire `kind`. The wrappers stamp
|
|
22
|
-
* `
|
|
26
|
+
* `sleep`/`event` through this; the public `ctx.pause` is always
|
|
23
27
|
* `custom` and never exposes it. */
|
|
24
28
|
export type KindedPauseFn = <T = unknown>(req: PauseRequest<T>, kind: StepPauseRequest["kind"]) => Promise<T>;
|
|
25
29
|
|
|
26
|
-
export interface RequestDecisionRequest<T> {
|
|
27
|
-
reason: string;
|
|
28
|
-
payload?: Record<string, unknown>;
|
|
29
|
-
/** Required — a decision is always validated against a shape. */
|
|
30
|
-
schema: z.ZodType<T>;
|
|
31
|
-
ttlMs?: number;
|
|
32
|
-
}
|
|
33
|
-
|
|
34
30
|
export interface WaitForEventRequest<T> {
|
|
35
31
|
reason: string;
|
|
36
32
|
/** Required — the by-key resume route targets this. */
|
|
@@ -40,22 +36,12 @@ export interface WaitForEventRequest<T> {
|
|
|
40
36
|
}
|
|
41
37
|
|
|
42
38
|
export interface PauseWrappers {
|
|
43
|
-
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
|
|
44
39
|
sleep(durationMs: number): Promise<void>;
|
|
45
40
|
waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
|
|
46
41
|
}
|
|
47
42
|
|
|
48
43
|
export function buildPauseWrappers(pause: KindedPauseFn): PauseWrappers {
|
|
49
44
|
return {
|
|
50
|
-
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T> {
|
|
51
|
-
return pause<T>({
|
|
52
|
-
reason: req.reason,
|
|
53
|
-
schema: req.schema,
|
|
54
|
-
...(req.payload !== undefined ? { payload: req.payload } : {}),
|
|
55
|
-
...(req.ttlMs !== undefined ? { ttlMs: req.ttlMs } : {}),
|
|
56
|
-
}, "decision");
|
|
57
|
-
},
|
|
58
|
-
|
|
59
45
|
// Resolves on its OWN ttl — there is no external resumer for a sleep, so
|
|
60
46
|
// onExpiry resolves (not throws) with void. snapshot:false keeps it cheap.
|
|
61
47
|
sleep(durationMs: number): Promise<void> {
|