@agent-compose/sdk 0.5.6 → 0.5.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/agent/__tests__/run-agent-liveness.test.d.ts +17 -0
- package/dist/agent/agent-context.d.ts +67 -0
- package/dist/agent/agent-loop-contract.test.d.ts +1 -0
- package/dist/agent/agent-loop.d.ts +2 -1
- package/dist/client.d.ts +129 -24
- package/dist/index.d.ts +9 -4
- package/dist/index.js +553 -89
- package/dist/pause/wrappers.d.ts +7 -11
- package/dist/runtimes/claude.d.ts +9 -1
- package/dist/runtimes/openai-desktop.js +548 -89
- package/dist/sandbox-errors.d.ts +49 -0
- package/dist/sandbox.d.ts +92 -13
- package/dist/step-invocation/protocol.d.ts +6 -0
- package/dist/step-invocation/types.d.ts +1 -1
- package/dist/types/execution-context.d.ts +1 -3
- package/dist/types/sandbox-environment.d.ts +1 -10
- package/dist/types/sandbox.d.ts +27 -3
- package/dist/types/workflow-metadata.d.ts +81 -13
- package/dist/types/workflow.d.ts +45 -10
- package/dist/utils/bundler.d.ts +40 -9
- package/dist/workflow-steps/workflow.d.ts +4 -3
- package/package.json +2 -2
- package/src/agent/agent-context.ts +212 -0
- package/src/agent/agent-loop.ts +78 -10
- package/src/agent/run-agent.ts +37 -1
- package/src/client.ts +232 -26
- package/src/index.ts +9 -4
- package/src/pause/wrappers.ts +7 -21
- package/src/runtimes/claude.ts +66 -8
- package/src/sandbox-errors.ts +53 -0
- package/src/sandbox.ts +438 -61
- package/src/step-invocation/invoker.ts +66 -8
- package/src/step-invocation/protocol.ts +9 -0
- package/src/step-invocation/server.ts +27 -4
- package/src/step-invocation/types.ts +1 -1
- package/src/types/execution-context.ts +1 -3
- package/src/types/sandbox-environment.ts +1 -11
- package/src/types/sandbox.ts +28 -3
- package/src/types/workflow-metadata.ts +91 -16
- package/src/types/workflow.ts +45 -12
- package/src/utils/bundler.ts +46 -13
- package/src/workflow-steps/workflow.ts +4 -3
- package/src/workflows/invoke-child.ts +7 -1
package/src/client.ts
CHANGED
|
@@ -14,10 +14,10 @@
|
|
|
14
14
|
import { ofetch } from "ofetch";
|
|
15
15
|
import { AgentComposeError } from "./errors.js";
|
|
16
16
|
import { parseSseStream } from "./sse.js";
|
|
17
|
-
import type { SandboxNetworkPolicy } from "./sandbox.js";
|
|
17
|
+
import type { SandboxNetworkPolicy, SandboxSize } from "./sandbox.js";
|
|
18
18
|
import type { RunEvent } from "./types/events.js";
|
|
19
19
|
import type { WorkflowPlan } from "./types/workflow-plan.js";
|
|
20
|
-
import type { SnapshotConfig, IOSchema } from "./types/workflow-metadata.js";
|
|
20
|
+
import type { SnapshotConfig, IOSchema, ConnectorRequirements, ConnectorOperationTag, InvokePolicy, SandboxResources } from "./types/workflow-metadata.js";
|
|
21
21
|
import type { WorkflowManifest } from "./utils/bundler.js";
|
|
22
22
|
|
|
23
23
|
/** UUID-v4-ish — matches the server-side predicate. Used to auto-detect
|
|
@@ -52,13 +52,6 @@ export interface RegisterResult {
|
|
|
52
52
|
name: string;
|
|
53
53
|
version: string;
|
|
54
54
|
runtimes?: RegisteredRuntime[];
|
|
55
|
-
/** Non-fatal advisories from the server. Surfaced at register time so the
|
|
56
|
-
* operator sees them while still in front of the terminal — currently
|
|
57
|
-
* covers "memory extraction is configured but its workflow is not
|
|
58
|
-
* registered in this factory". Empty/undefined when registration was
|
|
59
|
-
* cleanly resolved against everything the workflow declares it
|
|
60
|
-
* depends on. */
|
|
61
|
-
warnings?: string[];
|
|
62
55
|
}
|
|
63
56
|
|
|
64
57
|
export interface RegisteredRuntime {
|
|
@@ -72,6 +65,19 @@ export interface RuntimeSourceInput {
|
|
|
72
65
|
source: string;
|
|
73
66
|
}
|
|
74
67
|
|
|
68
|
+
/** GitHub provenance for a registered template's source file — stored as
|
|
69
|
+
* `metadata.source` on the registration. `cloud-build` stamps the built
|
|
70
|
+
* commit's sha; the dashboard's manual link path writes `sha: "manual"`. */
|
|
71
|
+
export interface TemplateSourceRef {
|
|
72
|
+
owner: string;
|
|
73
|
+
repo: string;
|
|
74
|
+
branch: string;
|
|
75
|
+
/** Repo-relative file path, e.g. `.agentc/workflows/workflow-deploy.ts`. */
|
|
76
|
+
path: string;
|
|
77
|
+
/** Commit sha the version was built from, or `"manual"` for hand-links. */
|
|
78
|
+
sha: string;
|
|
79
|
+
}
|
|
80
|
+
|
|
75
81
|
export interface RegisterWorkflowInput {
|
|
76
82
|
name: string;
|
|
77
83
|
source: string;
|
|
@@ -83,6 +89,9 @@ export interface RegisterWorkflowInput {
|
|
|
83
89
|
* executed on the server. */
|
|
84
90
|
manifest: WorkflowManifest;
|
|
85
91
|
version?: string;
|
|
92
|
+
/** Where the source file lives on GitHub — stored as `metadata.source`.
|
|
93
|
+
* Named `sourceRef` because `source` is the bundled code itself. */
|
|
94
|
+
sourceRef?: TemplateSourceRef;
|
|
86
95
|
schedule?: string;
|
|
87
96
|
runtimes?: RuntimeSourceInput[];
|
|
88
97
|
/** Human-readable description declared via
|
|
@@ -94,15 +103,20 @@ export interface RegisterWorkflowInput {
|
|
|
94
103
|
/** All snapshot config — `bootFrom` (where to restore at run start),
|
|
95
104
|
* `save`, `retain`. See `WorkflowMetadata.snapshots`. */
|
|
96
105
|
snapshots?: SnapshotConfig;
|
|
106
|
+
/** Sandbox machine size (template default). See `WorkflowMetadata.resources`. */
|
|
107
|
+
resources?: SandboxResources;
|
|
97
108
|
/** Provider-neutral execution plan detected by the CLI bundler. */
|
|
98
109
|
workflowPlan?: WorkflowPlan;
|
|
99
|
-
/**
|
|
100
|
-
*
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
110
|
+
/** Connector requirements declared via `defineWorkflow({ connectors })`
|
|
111
|
+
* (ADR-0007). Validated against the server's provider registry at
|
|
112
|
+
* registration; tokens are injected at the network layer at dispatch. */
|
|
113
|
+
connectors?: ConnectorRequirements;
|
|
114
|
+
/** Connector-catalogue operation tag — see `ConnectorOperationTag`. */
|
|
115
|
+
connectorOperation?: ConnectorOperationTag;
|
|
116
|
+
/** Tier-1 invoke ACL declared via `defineWorkflow({ invokePolicy })`.
|
|
117
|
+
* Only meaningful when the workflow also declares `connectors` — the
|
|
118
|
+
* server gates dispatch on it before binding any grant. */
|
|
119
|
+
invokePolicy?: InvokePolicy;
|
|
106
120
|
/** Input schema extracted from the workflow's `input` zod schema. */
|
|
107
121
|
inputSchema?: IOSchema;
|
|
108
122
|
/** Output schema extracted from the workflow's `output` zod schema. */
|
|
@@ -127,19 +141,20 @@ export interface InvokeWorkflowOptions {
|
|
|
127
141
|
* vars after brokering. Replaces the template-level placeholders for
|
|
128
142
|
* this run only — registered metadata is not mutated. */
|
|
129
143
|
placeholders?: Record<string, string>;
|
|
130
|
-
/** Per-invocation
|
|
131
|
-
*
|
|
132
|
-
*
|
|
133
|
-
|
|
134
|
-
/** Per-invocation post-hook override — replaces the registered
|
|
135
|
-
* `postRunHooks` array for this run only. */
|
|
136
|
-
postRunHooks?: readonly string[];
|
|
144
|
+
/** Per-invocation machine-size override of the template's `resources.size`.
|
|
145
|
+
* `small` (default) | `medium` | `large`; omit → the template default,
|
|
146
|
+
* else `small`. Honoured on Vercel (→ vCPUs); E2B ignores it. */
|
|
147
|
+
size?: SandboxSize;
|
|
137
148
|
/** Explicit parent run id. Pass `null` to suppress ambient RUN_ID auto-detection. */
|
|
138
149
|
parentRunId?: string | null;
|
|
139
150
|
/** Agent loop inside the parent run that caused this invoke, when applicable. */
|
|
140
151
|
agentId?: string | null;
|
|
141
152
|
/** Factory slug. Defaults to `"default"`. */
|
|
142
153
|
factorySlug?: string;
|
|
154
|
+
/** Idempotency key — sent as the `Idempotency-Key` header. A repeat invoke
|
|
155
|
+
* with the same key inside the server's dedup window returns the original
|
|
156
|
+
* run instead of starting a new one (matches `resumePause`'s pattern). */
|
|
157
|
+
idempotencyKey?: string;
|
|
143
158
|
}
|
|
144
159
|
|
|
145
160
|
export interface InvokeAndWaitOptions extends InvokeWorkflowOptions {
|
|
@@ -170,6 +185,56 @@ export interface ListTemplatesOptions {
|
|
|
170
185
|
factorySlug?: string;
|
|
171
186
|
}
|
|
172
187
|
|
|
188
|
+
/** A human member of your team — the people an agent (or you) can @-flag. */
|
|
189
|
+
export interface TeamMember {
|
|
190
|
+
/** Membership row id. */
|
|
191
|
+
id: string;
|
|
192
|
+
/** The user id — what you pass to `createMentions({ mentionedUserIds })`. */
|
|
193
|
+
userId: string;
|
|
194
|
+
role: string;
|
|
195
|
+
email: string;
|
|
196
|
+
name: string;
|
|
197
|
+
joinedAt: string;
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** A "you were flagged" ping, persisted server-side so it reaches the
|
|
201
|
+
* mentioned teammate in their Workbench. */
|
|
202
|
+
export interface Mention {
|
|
203
|
+
id: string;
|
|
204
|
+
factoryId: string;
|
|
205
|
+
mentionedUserId: string;
|
|
206
|
+
/** Who flagged: 'user' | 'api_key' | 'run' | 'system'. */
|
|
207
|
+
actorKind: string;
|
|
208
|
+
actorId: string | null;
|
|
209
|
+
actorLabel: string | null;
|
|
210
|
+
/** Where it lives: 'doc' | 'comment' | 'plan' | 'run'. */
|
|
211
|
+
contextKind: string;
|
|
212
|
+
contextPath: string | null;
|
|
213
|
+
/** Ready-made relative dashboard URL the Workbench card links to. */
|
|
214
|
+
contextUrl: string | null;
|
|
215
|
+
text: string;
|
|
216
|
+
runId: string | null;
|
|
217
|
+
seenAt: string | null;
|
|
218
|
+
resolvedAt: string | null;
|
|
219
|
+
createdAt: string;
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
export interface CreateMentionsInput {
|
|
223
|
+
/** Team-member user ids to flag (1–20). Discover them via `listMembers()`.
|
|
224
|
+
* Non-members are dropped server-side. */
|
|
225
|
+
mentionedUserIds: string[];
|
|
226
|
+
/** The flag message shown in the teammate's Workbench. */
|
|
227
|
+
text: string;
|
|
228
|
+
contextKind: "doc" | "comment" | "plan" | "run";
|
|
229
|
+
/** Factory-relative file path or comment thread id, when applicable. */
|
|
230
|
+
contextPath?: string;
|
|
231
|
+
/** Ready-made relative dashboard URL the Workbench card links to (e.g.
|
|
232
|
+
* `/factories/<slug>/files/view?path=<plan>`). */
|
|
233
|
+
contextUrl?: string;
|
|
234
|
+
runId?: string;
|
|
235
|
+
factorySlug?: string;
|
|
236
|
+
}
|
|
237
|
+
|
|
173
238
|
export interface CreateFactoryInput {
|
|
174
239
|
slug: string;
|
|
175
240
|
name: string;
|
|
@@ -234,6 +299,13 @@ export interface RunStatus<TOutput = unknown> {
|
|
|
234
299
|
id: string;
|
|
235
300
|
status: RunState;
|
|
236
301
|
output?: TOutput;
|
|
302
|
+
/** The run's latest (`saveLatest`) snapshot id, populated once the run has
|
|
303
|
+
* succeeded — the boot source to fork this run's evolved filesystem from
|
|
304
|
+
* (pass as `snapshots.bootFrom` on a follow-up invoke). `null` while the run
|
|
305
|
+
* is still in flight or when it captured no snapshot. Lets an orchestrator
|
|
306
|
+
* fork a child straight off the `invokeChild` result without a separate
|
|
307
|
+
* `listRunSnapshots` call. */
|
|
308
|
+
latestSnapshotId?: string | null;
|
|
237
309
|
}
|
|
238
310
|
|
|
239
311
|
/** ADR-0006 step 10 — actor record returned on a successful resume.
|
|
@@ -380,6 +452,41 @@ export interface EventRow {
|
|
|
380
452
|
createdAt: string;
|
|
381
453
|
}
|
|
382
454
|
|
|
455
|
+
// The server emits these rows in snake_case; the client maps them to
|
|
456
|
+
// camelCase at the fetch boundary so the SDK surface stays uniform
|
|
457
|
+
// (`EventRow.createdAt`, `RunStatus.latestSnapshotId`, …).
|
|
458
|
+
export interface RunArtifactRow {
|
|
459
|
+
path: string;
|
|
460
|
+
factorySlug: string | null;
|
|
461
|
+
sizeBytes: number | null;
|
|
462
|
+
lastWriteAt: string;
|
|
463
|
+
/** Opening text of the file (≤320 chars) — null for binary/empty. */
|
|
464
|
+
preview: string | null;
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
export interface FactoryFileWriteResult {
|
|
468
|
+
path: string;
|
|
469
|
+
contentHash: string;
|
|
470
|
+
sizeBytes: number;
|
|
471
|
+
created: boolean;
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
/** Wire shapes — what the server actually emits (snake_case). */
|
|
475
|
+
interface RunArtifactWire {
|
|
476
|
+
path: string;
|
|
477
|
+
factory_slug: string | null;
|
|
478
|
+
size_bytes: number | null;
|
|
479
|
+
last_write_at: string;
|
|
480
|
+
preview: string | null;
|
|
481
|
+
}
|
|
482
|
+
|
|
483
|
+
interface FactoryFileWriteWire {
|
|
484
|
+
path: string;
|
|
485
|
+
content_hash: string;
|
|
486
|
+
size_bytes: number;
|
|
487
|
+
created: boolean;
|
|
488
|
+
}
|
|
489
|
+
|
|
383
490
|
export interface ReportEventInput {
|
|
384
491
|
name: string;
|
|
385
492
|
body: unknown;
|
|
@@ -395,7 +502,7 @@ export interface ListEventsOptions {
|
|
|
395
502
|
factorySlug?: string;
|
|
396
503
|
limit?: number;
|
|
397
504
|
/** Case-insensitive substring match. Server uses `ILIKE %name%`, so
|
|
398
|
-
* `"
|
|
505
|
+
* `"site"` matches `site.created`, `site.failed`, etc. Pass the
|
|
399
506
|
* full event name for an effectively-exact filter (any string is a
|
|
400
507
|
* substring of itself). */
|
|
401
508
|
name?: string;
|
|
@@ -600,15 +707,17 @@ export class AgentComposeClient {
|
|
|
600
707
|
? detectAmbientParentRunId()
|
|
601
708
|
: opts.parentRunId;
|
|
602
709
|
const factorySlug = opts?.factorySlug ?? DEFAULT_FACTORY;
|
|
710
|
+
const headers: Record<string, string> = {};
|
|
711
|
+
if (opts?.idempotencyKey) headers["Idempotency-Key"] = opts.idempotencyKey;
|
|
603
712
|
return this.fetch(templatePath(factorySlug, name, "invoke"), {
|
|
604
713
|
method: "POST",
|
|
714
|
+
...(opts?.idempotencyKey ? { headers } : {}),
|
|
605
715
|
body: {
|
|
606
716
|
input,
|
|
607
717
|
...(opts?.snapshots !== undefined ? { snapshots: opts.snapshots } : {}),
|
|
608
718
|
...(opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {}),
|
|
609
719
|
...(opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {}),
|
|
610
|
-
...(opts?.
|
|
611
|
-
...(opts?.postRunHooks !== undefined ? { postRunHooks: opts.postRunHooks } : {}),
|
|
720
|
+
...(opts?.size !== undefined ? { size: opts.size } : {}),
|
|
612
721
|
...(parentRunId ? { parentRunId } : {}),
|
|
613
722
|
...(opts?.agentId ? { agentId: opts.agentId } : {}),
|
|
614
723
|
},
|
|
@@ -895,6 +1004,103 @@ export class AgentComposeClient {
|
|
|
895
1004
|
return body.events;
|
|
896
1005
|
}
|
|
897
1006
|
|
|
1007
|
+
/** Files the run wrote on the factory drive — run-attributed revisions,
|
|
1008
|
+
* latest write per path, paths the run later deleted excluded. */
|
|
1009
|
+
async listRunArtifacts(runId: string): Promise<RunArtifactRow[]> {
|
|
1010
|
+
const body = await this.fetch<{ artifacts: RunArtifactWire[] }>(
|
|
1011
|
+
`/api/v1/workflows/${encodeURIComponent(runId)}/artifacts`,
|
|
1012
|
+
);
|
|
1013
|
+
return body.artifacts.map((a) => ({
|
|
1014
|
+
path: a.path,
|
|
1015
|
+
factorySlug: a.factory_slug,
|
|
1016
|
+
sizeBytes: a.size_bytes,
|
|
1017
|
+
lastWriteAt: a.last_write_at,
|
|
1018
|
+
preview: a.preview,
|
|
1019
|
+
}));
|
|
1020
|
+
}
|
|
1021
|
+
|
|
1022
|
+
// ── Factory files ──────────────────────────────────────────────────────────
|
|
1023
|
+
// The factory drive: documents surfaced in the dashboard's Files tab.
|
|
1024
|
+
// Writes from inside a sandbox automatically carry the run-callback token
|
|
1025
|
+
// (AGENT_COMPOSE_RUN_TOKEN), so the revision is attributed to the run —
|
|
1026
|
+
// that's what surfaces the doc in the Workbench docs section and the
|
|
1027
|
+
// editor-avatar run hover.
|
|
1028
|
+
|
|
1029
|
+
/** Write (create or overwrite) one file on a factory's drive. */
|
|
1030
|
+
async putFactoryFile(
|
|
1031
|
+
path: string,
|
|
1032
|
+
content: string | Uint8Array,
|
|
1033
|
+
opts?: { factorySlug?: string; contentType?: string },
|
|
1034
|
+
): Promise<FactoryFileWriteResult> {
|
|
1035
|
+
const factorySlug = opts?.factorySlug
|
|
1036
|
+
?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined)
|
|
1037
|
+
?? DEFAULT_FACTORY;
|
|
1038
|
+
const runToken = typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_RUN_TOKEN : undefined;
|
|
1039
|
+
const wire = await this.fetch<FactoryFileWriteWire>(
|
|
1040
|
+
`/api/v1/factories/${encodeURIComponent(factorySlug)}/files/content?path=${encodeURIComponent(path)}`,
|
|
1041
|
+
{
|
|
1042
|
+
method: "PUT",
|
|
1043
|
+
body: content,
|
|
1044
|
+
headers: {
|
|
1045
|
+
"content-type": opts?.contentType ?? "text/plain; charset=utf-8",
|
|
1046
|
+
...(runToken ? { "x-run-token": runToken } : {}),
|
|
1047
|
+
},
|
|
1048
|
+
},
|
|
1049
|
+
);
|
|
1050
|
+
return {
|
|
1051
|
+
path: wire.path,
|
|
1052
|
+
contentHash: wire.content_hash,
|
|
1053
|
+
sizeBytes: wire.size_bytes,
|
|
1054
|
+
created: wire.created,
|
|
1055
|
+
};
|
|
1056
|
+
}
|
|
1057
|
+
|
|
1058
|
+
/** Read one file's current content (or a specific revision) as text. */
|
|
1059
|
+
async getFactoryFile(
|
|
1060
|
+
path: string,
|
|
1061
|
+
opts?: { factorySlug?: string; revision?: number },
|
|
1062
|
+
): Promise<string> {
|
|
1063
|
+
const factorySlug = opts?.factorySlug
|
|
1064
|
+
?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined)
|
|
1065
|
+
?? DEFAULT_FACTORY;
|
|
1066
|
+
const q = new URLSearchParams({ path });
|
|
1067
|
+
if (opts?.revision !== undefined) q.set("revision", String(opts.revision));
|
|
1068
|
+
return this.fetch(
|
|
1069
|
+
`/api/v1/factories/${encodeURIComponent(factorySlug)}/files/content?${q}`,
|
|
1070
|
+
{ responseType: "text" },
|
|
1071
|
+
);
|
|
1072
|
+
}
|
|
1073
|
+
|
|
1074
|
+
/** List the human members of your team — the people you (or an agent) can
|
|
1075
|
+
* @-flag with `createMentions`. Each row's `userId` is what
|
|
1076
|
+
* `mentionedUserIds` expects. */
|
|
1077
|
+
async listMembers(): Promise<TeamMember[]> {
|
|
1078
|
+
const body = await this.fetch<{ members: TeamMember[] }>("/api/v1/team/members");
|
|
1079
|
+
return body.members;
|
|
1080
|
+
}
|
|
1081
|
+
|
|
1082
|
+
/** Flag one or more teammates — a durable ping that lands in their factory
|
|
1083
|
+
* Workbench. Use from an agent (e.g. a remediation plan that needs a human
|
|
1084
|
+
* to rotate a secret) or any team automation. Resolve `mentionedUserIds`
|
|
1085
|
+
* via `listMembers()`. When run inside a sandbox the run-callback token is
|
|
1086
|
+
* forwarded so the ping is attributed to the run ("flagged by <workflow>"). */
|
|
1087
|
+
async createMentions(input: CreateMentionsInput): Promise<Mention[]> {
|
|
1088
|
+
const factorySlug = input.factorySlug
|
|
1089
|
+
?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined)
|
|
1090
|
+
?? DEFAULT_FACTORY;
|
|
1091
|
+
const runToken = typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_RUN_TOKEN : undefined;
|
|
1092
|
+
const { factorySlug: _omit, ...payload } = input;
|
|
1093
|
+
const body = await this.fetch<{ mentions: Mention[] }>(
|
|
1094
|
+
`/api/v1/factories/${encodeURIComponent(factorySlug)}/mentions`,
|
|
1095
|
+
{
|
|
1096
|
+
method: "POST",
|
|
1097
|
+
body: payload,
|
|
1098
|
+
...(runToken ? { headers: { "x-run-token": runToken } } : {}),
|
|
1099
|
+
},
|
|
1100
|
+
);
|
|
1101
|
+
return body.mentions;
|
|
1102
|
+
}
|
|
1103
|
+
|
|
898
1104
|
/** List events ingested into a factory, newest first. Supports
|
|
899
1105
|
* case-insensitive substring filter (`name`) and timestamp-cursor
|
|
900
1106
|
* pagination (`before`). Returns `{ events, has_more }` — the
|
package/src/index.ts
CHANGED
|
@@ -41,8 +41,8 @@ export type {
|
|
|
41
41
|
ReuseSnapshot,
|
|
42
42
|
IOSchema,
|
|
43
43
|
OutputSchema,
|
|
44
|
-
WorkflowMemoryConfig,
|
|
45
44
|
} from "./types/workflow.js";
|
|
45
|
+
export type { ConnectorRequestRules } from "./types/workflow-metadata.js";
|
|
46
46
|
|
|
47
47
|
// Snapshot entry type re-exported for consumers (dashboard, CLI).
|
|
48
48
|
export type { RunSnapshotEntry } from "./client.js";
|
|
@@ -106,12 +106,13 @@ export type {
|
|
|
106
106
|
// HTTP client
|
|
107
107
|
export { AgentComposeClient } from "./client.js";
|
|
108
108
|
export type {
|
|
109
|
-
RegisterResult, RegisterWorkflowInput, RuntimeSourceInput,
|
|
109
|
+
RegisterResult, RegisterWorkflowInput, RuntimeSourceInput, TemplateSourceRef,
|
|
110
110
|
InvokeWorkflowOptions, InvokeAndWaitOptions, InvokeResult,
|
|
111
111
|
ListSnapshotsOptions, TemplateRow, ListTemplatesOptions,
|
|
112
112
|
CreateFactoryInput, UpdateFactoryInput,
|
|
113
113
|
SecretOptions, SetSecretResult, SecretListEntry,
|
|
114
114
|
CreateApiKeyInput, StreamRunLogsOptions,
|
|
115
|
+
TeamMember, Mention, CreateMentionsInput,
|
|
115
116
|
EventSubjectType, EventRow, ReportEventInput, ListEventsOptions, ListEventsResult,
|
|
116
117
|
RunLogLine, ListRunLogsOptions,
|
|
117
118
|
RegisteredRuntime, RunState, RunStatus, FactoryRow, SnapshotListEntry, SnapshotListResponse,
|
|
@@ -181,12 +182,14 @@ export { createSandbox, reconnectSandbox, killAllSandboxes, killSandboxById,
|
|
|
181
182
|
getSandboxQuotas, listOwnedSandboxes, deleteSandboxSnapshot,
|
|
182
183
|
makeSandboxProvider, makeDesktopSandboxProvider,
|
|
183
184
|
parseSseExecStream, AGENT_COMPOSE_TAG } from "./sandbox.js";
|
|
185
|
+
export { SandboxUnavailableError, SANDBOX_UNAVAILABLE_PREFIX } from "./sandbox-errors.js";
|
|
184
186
|
export type {
|
|
185
187
|
SandboxCreateOpts, SandboxNetworkPolicy, SandboxNetworkHeaderTransform,
|
|
186
188
|
SandboxNetworkAllowRule, SandboxNetworkSubnetPolicy, SandboxProviderName,
|
|
187
|
-
SandboxQuotaResult, OwnedSandboxResult, OwnedSandbox,
|
|
189
|
+
SandboxQuotaResult, OwnedSandboxResult, OwnedSandbox, SandboxSize,
|
|
188
190
|
ParseSseExecStreamOptions, SandboxCommandRunOptions, SandboxCommandResult,
|
|
189
191
|
} from "./sandbox.js";
|
|
192
|
+
export type { SandboxResources } from "./types/workflow-metadata.js";
|
|
190
193
|
|
|
191
194
|
// Workflow engine
|
|
192
195
|
export { runWorkflow, WorkflowError, EngineError, classifyError, parseNameVersion } from "./workflows/engine.js";
|
|
@@ -248,7 +251,7 @@ export {
|
|
|
248
251
|
} from "./pause/errors.js";
|
|
249
252
|
export type { PauseErrorCode } from "./pause/errors.js";
|
|
250
253
|
export type { PauseRequest } from "./pause/pause-core.js";
|
|
251
|
-
export type {
|
|
254
|
+
export type { WaitForEventRequest } from "./pause/wrappers.js";
|
|
252
255
|
export type {
|
|
253
256
|
StepRequest,
|
|
254
257
|
StepResult,
|
|
@@ -264,5 +267,7 @@ export { agentLoop, parseAgentStatus, DEFAULT_CLAUDE_MODEL } from "./agent/agent
|
|
|
264
267
|
export type { AgentLifecycleEvent, AgentLoopOpts, AgentLoopResult } from "./agent/agent-loop.js";
|
|
265
268
|
export { agent } from "./agent/run-agent.js";
|
|
266
269
|
export type { AgentOpts } from "./agent/run-agent.js";
|
|
270
|
+
export { AGENT_COMPOSE_MANUAL, buildAgentContextDoc, writeAgentContext } from "./agent/agent-context.js";
|
|
271
|
+
export type { AgentConnectorInfo } from "./agent/agent-context.js";
|
|
267
272
|
export { AgentMessageSchema, parseAgentResponse } from "./agent/protocol.js";
|
|
268
273
|
export { importSourceModule, TMP_DIR, LATEST_VERSION } from "./utils/source-loader.js";
|
package/src/pause/wrappers.ts
CHANGED
|
@@ -1,12 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The
|
|
2
|
+
* The two opinionated pause wrappers (ADR-0006 §"SDK surface"), each a thin
|
|
3
3
|
* closure over `ctx.pause`:
|
|
4
4
|
*
|
|
5
|
-
* - `requestDecision` — pause for a typed human/agent decision (schema required).
|
|
6
5
|
* - `sleep` — a lightweight timed pause; resolves on its own TTL,
|
|
7
6
|
* skips the snapshot, returns void.
|
|
8
7
|
* - `waitForEvent` — pause until an event resumes by correlation key.
|
|
9
8
|
*
|
|
9
|
+
* A typed human/agent decision is NOT a wrapper — it's a plain `ctx.pause`
|
|
10
|
+
* with a `schema` (and `payload.options` for the dashboard's answer UI). The
|
|
11
|
+
* pause primitive carries the reason (the ask) and the resume value (the
|
|
12
|
+
* resolution); there is no separate `requestDecision`.
|
|
13
|
+
*
|
|
10
14
|
* Built from a `PauseFn` so the wrapper logic lives in one place and the step
|
|
11
15
|
* runner just spreads them onto the context next to `pause`.
|
|
12
16
|
*/
|
|
@@ -19,18 +23,10 @@ import type { PauseRequest } from "./pause-core.js";
|
|
|
19
23
|
export type PauseFn = <T = unknown>(req: PauseRequest<T>) => Promise<T>;
|
|
20
24
|
|
|
21
25
|
/** Internal: pause with an explicit wire `kind`. The wrappers stamp
|
|
22
|
-
* `
|
|
26
|
+
* `sleep`/`event` through this; the public `ctx.pause` is always
|
|
23
27
|
* `custom` and never exposes it. */
|
|
24
28
|
export type KindedPauseFn = <T = unknown>(req: PauseRequest<T>, kind: StepPauseRequest["kind"]) => Promise<T>;
|
|
25
29
|
|
|
26
|
-
export interface RequestDecisionRequest<T> {
|
|
27
|
-
reason: string;
|
|
28
|
-
payload?: Record<string, unknown>;
|
|
29
|
-
/** Required — a decision is always validated against a shape. */
|
|
30
|
-
schema: z.ZodType<T>;
|
|
31
|
-
ttlMs?: number;
|
|
32
|
-
}
|
|
33
|
-
|
|
34
30
|
export interface WaitForEventRequest<T> {
|
|
35
31
|
reason: string;
|
|
36
32
|
/** Required — the by-key resume route targets this. */
|
|
@@ -40,22 +36,12 @@ export interface WaitForEventRequest<T> {
|
|
|
40
36
|
}
|
|
41
37
|
|
|
42
38
|
export interface PauseWrappers {
|
|
43
|
-
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
|
|
44
39
|
sleep(durationMs: number): Promise<void>;
|
|
45
40
|
waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
|
|
46
41
|
}
|
|
47
42
|
|
|
48
43
|
export function buildPauseWrappers(pause: KindedPauseFn): PauseWrappers {
|
|
49
44
|
return {
|
|
50
|
-
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T> {
|
|
51
|
-
return pause<T>({
|
|
52
|
-
reason: req.reason,
|
|
53
|
-
schema: req.schema,
|
|
54
|
-
...(req.payload !== undefined ? { payload: req.payload } : {}),
|
|
55
|
-
...(req.ttlMs !== undefined ? { ttlMs: req.ttlMs } : {}),
|
|
56
|
-
}, "decision");
|
|
57
|
-
},
|
|
58
|
-
|
|
59
45
|
// Resolves on its OWN ttl — there is no external resumer for a sleep, so
|
|
60
46
|
// onExpiry resolves (not throws) with void. snapshot:false keeps it cheap.
|
|
61
47
|
sleep(durationMs: number): Promise<void> {
|
package/src/runtimes/claude.ts
CHANGED
|
@@ -62,8 +62,20 @@ function translateMessage(message: Record<string, unknown>): AgentMessage[] {
|
|
|
62
62
|
numTurns: Number(message.num_turns ?? 0),
|
|
63
63
|
timestamp: ts,
|
|
64
64
|
});
|
|
65
|
-
if (message.
|
|
66
|
-
|
|
65
|
+
if (message.subtype === "error_max_turns") {
|
|
66
|
+
// Running out of turns is NOT a failure — the agent did real work and the
|
|
67
|
+
// files it wrote are on disk. End the iteration cleanly (don't throw) so the
|
|
68
|
+
// workflow keeps the partial result and moves on.
|
|
69
|
+
msgs.push({ type: "done", sessionId: String(message.session_id ?? ""), timestamp: ts });
|
|
70
|
+
} else if (message.is_error || message.subtype === "error_during_execution") {
|
|
71
|
+
// Error results sometimes carry NO error/result/message fields (e.g.
|
|
72
|
+
// the binary died early) — fall back to subtype + the raw envelope so
|
|
73
|
+
// the failure is diagnosable instead of "Agent error: undefined".
|
|
74
|
+
const detail = message.error ?? message.result ?? message.message;
|
|
75
|
+
const text = detail !== undefined
|
|
76
|
+
? formatError(detail)
|
|
77
|
+
: `${String(message.subtype ?? "unknown_error")} — raw result: ${JSON.stringify({ ...message, usage: undefined }).slice(0, 600)}`;
|
|
78
|
+
msgs.push({ type: "error", text, timestamp: ts });
|
|
67
79
|
} else {
|
|
68
80
|
msgs.push({ type: "done", sessionId: String(message.session_id ?? ""), timestamp: ts });
|
|
69
81
|
}
|
|
@@ -77,7 +89,11 @@ export interface ClaudeRuntimeConfig {
|
|
|
77
89
|
claudeMdContent?: string;
|
|
78
90
|
/** Env overrides for Agent SDK provider routing. */
|
|
79
91
|
env?: Record<string, string>;
|
|
80
|
-
/** Model to use.
|
|
92
|
+
/** Model to use. Accepts a caliber shorthand — "fable" (most capable),
|
|
93
|
+
* "opus", "sonnet", "haiku" (fastest/cheapest) — or an exact model id.
|
|
94
|
+
* Defaults to DEFAULT_CLAUDE_MODEL (Fable). Orchestrators pick a caliber
|
|
95
|
+
* per agent: fable/opus for planning + implementation, sonnet for focused
|
|
96
|
+
* single-responsibility work, haiku for mechanical tasks. */
|
|
81
97
|
model?: string;
|
|
82
98
|
/** MCP servers to configure for the Agent SDK. */
|
|
83
99
|
mcpServers?: Record<string, { command: string; args?: string[]; env?: Record<string, string> }>;
|
|
@@ -87,6 +103,31 @@ export interface ClaudeRuntimeConfig {
|
|
|
87
103
|
thinking?: ThinkingConfig;
|
|
88
104
|
/** Reasoning effort hint for models that support adaptive thinking. */
|
|
89
105
|
effort?: "low" | "medium" | "high" | "xhigh" | "max";
|
|
106
|
+
/** Skills to enable for the agent (the Agent SDK's `skills` option — also
|
|
107
|
+
* auto-adds the `Skill` tool). The agent-env bakes the `/ac:*` skills via
|
|
108
|
+
* `agentc init`; default `"all"` makes them usable. Pass `[]` to disable. */
|
|
109
|
+
skills?: string[] | "all";
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** Caliber shorthand → exact model id. Full ids pass through untouched. */
|
|
113
|
+
const MODEL_TIERS: Record<string, string> = {
|
|
114
|
+
fable: "claude-fable-5",
|
|
115
|
+
opus: "claude-opus-4-8",
|
|
116
|
+
sonnet: "claude-sonnet-4-6",
|
|
117
|
+
haiku: "claude-haiku-4-5",
|
|
118
|
+
};
|
|
119
|
+
function resolveClaudeModel(model: string | undefined): string {
|
|
120
|
+
if (!model) return DEFAULT_CLAUDE_MODEL;
|
|
121
|
+
return MODEL_TIERS[model.toLowerCase()] ?? model;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** Fable rejects an explicit `thinking: {type: "disabled"}` with a 400 (the only
|
|
125
|
+
* off-mode on Fable is omitting the param). Other models accept it. Resolve the
|
|
126
|
+
* thinking config against the chosen model so callers can keep passing
|
|
127
|
+
* `thinking: {type: "disabled"}` for cheap/fast agents regardless of tier. */
|
|
128
|
+
function resolveThinking(model: string, thinking: ThinkingConfig | undefined): ThinkingConfig | undefined {
|
|
129
|
+
if (model.startsWith("claude-fable") && thinking && (thinking as { type?: string }).type === "disabled") return undefined;
|
|
130
|
+
return thinking;
|
|
90
131
|
}
|
|
91
132
|
|
|
92
133
|
/** Canonical install path for Claude Code inside a Vercel sandbox.
|
|
@@ -130,7 +171,7 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
130
171
|
) {}
|
|
131
172
|
|
|
132
173
|
get model(): string {
|
|
133
|
-
return this.config.model ?? this.options.model
|
|
174
|
+
return resolveClaudeModel(this.config.model ?? this.options.model);
|
|
134
175
|
}
|
|
135
176
|
|
|
136
177
|
async gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult> {
|
|
@@ -179,7 +220,7 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
179
220
|
//
|
|
180
221
|
// Without an inboxStream we keep the simple string-prompt path —
|
|
181
222
|
// no queue, same behaviour as before. This keeps non-interactive
|
|
182
|
-
// workflows
|
|
223
|
+
// workflows on the original code path.
|
|
183
224
|
//
|
|
184
225
|
// The queue is closed when the SDK's `result` message indicates
|
|
185
226
|
// the iteration's assistant turns are done; without close() the
|
|
@@ -216,9 +257,9 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
216
257
|
tools: this.options.allowedTools,
|
|
217
258
|
allowedTools: this.options.allowedTools,
|
|
218
259
|
maxTurns: this.options.maxTurns,
|
|
219
|
-
model: this.
|
|
260
|
+
model: this.model,
|
|
220
261
|
outputFormat: this.options.outputFormat,
|
|
221
|
-
thinking: this.config.thinking,
|
|
262
|
+
thinking: resolveThinking(this.model, this.config.thinking),
|
|
222
263
|
effort: this.config.effort,
|
|
223
264
|
cwd: this.options.cwd,
|
|
224
265
|
env: { ...process.env, ...(this.config.env ?? {}) },
|
|
@@ -226,6 +267,10 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
226
267
|
?? process.env.CLAUDE_CODE_EXECUTABLE
|
|
227
268
|
?? DEFAULT_CLAUDE_PATH,
|
|
228
269
|
...(this.config.claudeMdContent ? { systemPrompt: { type: "preset" as const, preset: "claude_code" as const, append: this.config.claudeMdContent } } : {}),
|
|
270
|
+
// Turn skills ON (and auto-add the `Skill` tool). The agent-env
|
|
271
|
+
// bakes the `/ac:*` skills; without this the Agent SDK leaves
|
|
272
|
+
// them un-enabled and the agent can't invoke them.
|
|
273
|
+
skills: this.config.skills ?? "all",
|
|
229
274
|
resume: opts.sessionId,
|
|
230
275
|
mcpServers: this.config.mcpServers,
|
|
231
276
|
permissionMode: "acceptEdits",
|
|
@@ -251,7 +296,20 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
251
296
|
if (raw.type === "result" && inboxQueue) inboxQueue.close();
|
|
252
297
|
}
|
|
253
298
|
} catch (err) {
|
|
254
|
-
|
|
299
|
+
// The SDK ALSO signals "out of turns" by THROWING (separately from the
|
|
300
|
+
// result-message `error_max_turns` subtype handled in translateMessage — that
|
|
301
|
+
// typed path is the primary signal; this catch covers the SDK's separate
|
|
302
|
+
// throw, which is prose-only). Treat it the same: NOT a failure — end the
|
|
303
|
+
// iteration cleanly so the loop continues, keeping the work it did.
|
|
304
|
+
// Match ONLY the specific exhaustion phrase, NOT a loose "maxTurns" token: the
|
|
305
|
+
// latter would also swallow a genuine error like "maxTurns must be a positive
|
|
306
|
+
// integer" and silently report it as a clean finish.
|
|
307
|
+
const text = formatError(err);
|
|
308
|
+
if (/maximum number of turns/i.test(text)) {
|
|
309
|
+
yield { type: "done", sessionId: opts.sessionId ?? "", timestamp: now() };
|
|
310
|
+
} else {
|
|
311
|
+
yield { type: "error", text, timestamp: now() };
|
|
312
|
+
}
|
|
255
313
|
} finally {
|
|
256
314
|
// Belt-and-braces — close on error/abort too so the dangling
|
|
257
315
|
// iterable doesn't leak the inboxStream consumer.
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Sandbox-infrastructure error.
|
|
3
|
+
*
|
|
4
|
+
* Distinct from `StepExecutionError` (the user/runner step-failure wrapper):
|
|
5
|
+
* that's the workflow author's problem. A `SandboxUnavailableError` means the
|
|
6
|
+
* sandbox itself couldn't carry the step — the provider refused a command, the
|
|
7
|
+
* sandbox was reclaimed (idle/lifetime timeout, eviction), an API blip, or the
|
|
8
|
+
* connection dropped mid-stream. None of these are a fault in the customer's
|
|
9
|
+
* code, so they're surfaced as "infrastructure issue, retry" rather than
|
|
10
|
+
* blaming the user.
|
|
11
|
+
*
|
|
12
|
+
* ## Retryability is about *when*, not *which error*
|
|
13
|
+
*
|
|
14
|
+
* The Vercel provider (`sandbox.ts`) decides `retryable` purely from *where* it
|
|
15
|
+
* caught the error — not by inspecting status codes or error vocabularies:
|
|
16
|
+
* - `retryable: true` — caught before the runner launched the step (file
|
|
17
|
+
* write / command launch / reconnect refused). No user code ran, so
|
|
18
|
+
* re-provisioning a fresh sandbox and re-running the step is safe. ANY
|
|
19
|
+
* error here qualifies (sandbox reclaimed, 429, 5xx, network blip).
|
|
20
|
+
* - `retryable: false` — caught while/after the runner streamed, so user code
|
|
21
|
+
* may already have executed side effects. Surfaced honestly as an infra
|
|
22
|
+
* failure, but NOT auto-retried.
|
|
23
|
+
*
|
|
24
|
+
* ## Cross-process contract
|
|
25
|
+
*
|
|
26
|
+
* The thrown error crosses the Temporal activity→workflow serialisation
|
|
27
|
+
* boundary, which preserves the error **message** but discards custom instance
|
|
28
|
+
* fields. So the retryability signal is encoded IN the message via a stable
|
|
29
|
+
* prefix — `[sandbox-unavailable:retryable]` or `[sandbox-unavailable:terminal]`
|
|
30
|
+
* — following the same `[kind] …` convention `StepExecutionError` uses. The
|
|
31
|
+
* server (`utils/transient-errors.ts`) matches this prefix with a pure regex to
|
|
32
|
+
* classify the failure and decide whether the workflow re-provisions and
|
|
33
|
+
* retries. A unit test pins the SDK-produced string against the server matcher
|
|
34
|
+
* so the two can't drift.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
/** Stable message prefix. Both variants share this leading token so a single
|
|
38
|
+
* server-side regex recognises the class; the `:retryable` / `:terminal`
|
|
39
|
+
* suffix carries the recovery decision. */
|
|
40
|
+
export const SANDBOX_UNAVAILABLE_PREFIX = "[sandbox-unavailable";
|
|
41
|
+
|
|
42
|
+
export class SandboxUnavailableError extends Error {
|
|
43
|
+
/** True when caught before any user code ran (safe to re-provision + retry);
|
|
44
|
+
* false when the sandbox died mid/after execution. */
|
|
45
|
+
readonly retryable: boolean;
|
|
46
|
+
readonly sandboxId: string | undefined;
|
|
47
|
+
constructor(detail: string, opts: { retryable: boolean; sandboxId?: string }) {
|
|
48
|
+
super(`${SANDBOX_UNAVAILABLE_PREFIX}:${opts.retryable ? "retryable" : "terminal"}] ${detail}`);
|
|
49
|
+
this.name = "SandboxUnavailableError";
|
|
50
|
+
this.retryable = opts.retryable;
|
|
51
|
+
this.sandboxId = opts.sandboxId;
|
|
52
|
+
}
|
|
53
|
+
}
|