@agent-compose/sdk 0.5.5 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -4
- package/dist/agent/agent-loop-contract.test.d.ts +1 -0
- package/dist/agent/agent-loop.d.ts +1 -1
- package/dist/client.d.ts +65 -23
- package/dist/index.d.ts +4 -2
- package/dist/index.js +389 -109
- package/dist/runtimes/_cli-agent.d.ts +25 -8
- package/dist/runtimes/amp.d.ts +7 -6
- package/dist/runtimes/claude.d.ts +9 -1
- package/dist/runtimes/cli-agent.test.d.ts +9 -0
- package/dist/runtimes/codex.d.ts +6 -5
- package/dist/runtimes/openai-desktop.js +387 -109
- package/dist/sandbox-errors.d.ts +49 -0
- package/dist/sandbox.d.ts +68 -13
- package/dist/step-invocation/protocol.d.ts +6 -0
- package/dist/types/sandbox-environment.d.ts +1 -10
- package/dist/types/sandbox.d.ts +27 -3
- package/dist/types/workflow-metadata.d.ts +88 -23
- package/dist/types/workflow.d.ts +38 -10
- package/dist/utils/bundler.d.ts +38 -9
- package/dist/workflow-steps/workflow.d.ts +1 -3
- package/package.json +2 -2
- package/src/agent/agent-loop.ts +56 -7
- package/src/client.ts +144 -25
- package/src/index.ts +4 -2
- package/src/runtimes/_cli-agent.ts +75 -42
- package/src/runtimes/amp.ts +11 -6
- package/src/runtimes/claude.ts +66 -8
- package/src/runtimes/codex.ts +10 -5
- package/src/sandbox-errors.ts +53 -0
- package/src/sandbox.ts +378 -56
- package/src/step-invocation/invoker.ts +66 -8
- package/src/step-invocation/protocol.ts +9 -0
- package/src/step-invocation/server.ts +27 -4
- package/src/types/sandbox-environment.ts +1 -11
- package/src/types/sandbox.ts +28 -3
- package/src/types/workflow-metadata.ts +97 -26
- package/src/types/workflow.ts +38 -11
- package/src/utils/bundler.ts +43 -13
- package/src/workflow-steps/workflow.ts +1 -3
package/src/agent/agent-loop.ts
CHANGED
|
@@ -15,7 +15,7 @@ import { PauseManager } from "../pause/manager.js";
|
|
|
15
15
|
import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
|
|
16
16
|
import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
|
|
17
17
|
|
|
18
|
-
export const DEFAULT_CLAUDE_MODEL = "claude-
|
|
18
|
+
export const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
|
|
19
19
|
|
|
20
20
|
const SAME_BLOCKER_ITERATIONS = 3;
|
|
21
21
|
const STALL_ITERATIONS = 3;
|
|
@@ -152,8 +152,15 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
152
152
|
const label = opts.label ?? "agent";
|
|
153
153
|
const logLabel = opts.label ?? "[Agent Loop]";
|
|
154
154
|
const startedAt = Date.now();
|
|
155
|
-
|
|
156
|
-
|
|
155
|
+
// No budget ⇒ no turn cap: the harness runtime (Claude Code) decides when it's
|
|
156
|
+
// done. A numeric budget is an explicit caller choice, not a default we impose.
|
|
157
|
+
const turnsPerIteration = opts.turnsPerIteration;
|
|
158
|
+
const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 1 : 8);
|
|
159
|
+
// A responseSchema is a CONTRACT, not a hope: when the agent's <response> fails
|
|
160
|
+
// validation, the loop re-prompts with the exact errors until it conforms —
|
|
161
|
+
// without consuming the caller's iteration budget. The backstop below only
|
|
162
|
+
// guards against a truly wedged agent (never reached in normal operation).
|
|
163
|
+
let schemaRetriesLeft = 10;
|
|
157
164
|
const processors = opts.processors ?? [];
|
|
158
165
|
const requestContext = opts.requestContext ?? RequestContext.fromReserved({
|
|
159
166
|
teamId: "", runId: "", workflowId: "",
|
|
@@ -171,7 +178,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
171
178
|
|
|
172
179
|
if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
|
|
173
180
|
const client = opts.runtime({
|
|
174
|
-
maxTurns:
|
|
181
|
+
...(turnsPerIteration !== undefined ? { maxTurns: turnsPerIteration } : {}),
|
|
175
182
|
allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
|
|
176
183
|
label: logLabel,
|
|
177
184
|
cwd: opts.cwd,
|
|
@@ -342,6 +349,14 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
342
349
|
|
|
343
350
|
const procCtx = buildProcCtx(iteration + 1);
|
|
344
351
|
let initialPrompt = opts.buildPrompt(lastStatus, iteration);
|
|
352
|
+
// CONTRACT FEEDBACK: when the previous turn's <response> failed schema
|
|
353
|
+
// validation, the violation goes BACK TO THE MODEL as its next turn (the
|
|
354
|
+
// session carries the prior context). Without this, contract retries
|
|
355
|
+
// re-run the identical prompt and the model repeats the identical mistake.
|
|
356
|
+
if (lastResponseValidationError) {
|
|
357
|
+
initialPrompt = `${initialPrompt}\n\n[response contract violation — fix and re-emit]\nYour previous <response> failed schema validation with these errors:\n${lastResponseValidationError.slice(0, 2000)}\nRe-emit the COMPLETE corrected <response> JSON now: every required field present, correctly named and typed (no omissions, no renames).`;
|
|
358
|
+
lastResponseValidationError = "";
|
|
359
|
+
}
|
|
345
360
|
if (steerDecision) {
|
|
346
361
|
// Deliver the human's answer as the agent's next user turn by appending
|
|
347
362
|
// it to the iteration prompt. `sendMessage({ prompt })` is the one input
|
|
@@ -364,7 +379,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
364
379
|
|
|
365
380
|
// Progress, not an error — write to stdout so dashboards and
|
|
366
381
|
// log viewers don't visually flag it as a warning.
|
|
367
|
-
process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
|
|
382
|
+
process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration !== undefined ? `${turnsPerIteration} turns` : "harness-decided turns"}\n`);
|
|
368
383
|
|
|
369
384
|
let responseText = "";
|
|
370
385
|
// The single `opts.inbox` is shared across iterations, but the
|
|
@@ -376,7 +391,11 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
376
391
|
// semantics — buffered until consumed).
|
|
377
392
|
for await (const rawMsg of client.sendMessage({
|
|
378
393
|
prompt,
|
|
379
|
-
|
|
394
|
+
// Resume whenever a session exists — NOT keyed on `iteration > 0`, because a
|
|
395
|
+
// contract retry rolls `iteration` back to 0 while a session already exists;
|
|
396
|
+
// keying on the session id keeps the corrective re-prompt in the same session
|
|
397
|
+
// (otherwise it restarts the task in a fresh session and repeats side effects).
|
|
398
|
+
sessionId: lastSessionId ?? undefined,
|
|
380
399
|
iteration: iteration + 1,
|
|
381
400
|
...(opts.inbox ? { inboxStream: opts.inbox } : {}),
|
|
382
401
|
})) {
|
|
@@ -447,6 +466,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
447
466
|
process.stderr.write(`${logLabel} NO <response> BLOCK — response tail: ${responseText.slice(-400)}\n`);
|
|
448
467
|
status = { ...status!, exit_signal: false, blockers: ["No <response> block found — emit a <response> block with the required JSON fields before setting exit_signal: true"] };
|
|
449
468
|
opts.onIteration?.(iteration + 1, status);
|
|
469
|
+
if (schemaRetriesLeft-- > 0) iteration--; // contract enforcement — free, not billed to the iteration budget
|
|
450
470
|
continue;
|
|
451
471
|
}
|
|
452
472
|
// Status-merged validation — the schema may reference status
|
|
@@ -458,6 +478,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
458
478
|
process.stderr.write(`${logLabel} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(rawResponse).slice(0, 400)}\n`);
|
|
459
479
|
status = { ...status!, exit_signal: false, blockers: [`<response> schema validation failed: ${parsed.error.message}`] };
|
|
460
480
|
opts.onIteration?.(iteration + 1, status);
|
|
481
|
+
if (schemaRetriesLeft-- > 0) iteration--; // contract enforcement — free, not billed to the iteration budget
|
|
461
482
|
continue;
|
|
462
483
|
}
|
|
463
484
|
response = parsed.data;
|
|
@@ -487,10 +508,24 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
487
508
|
continue;
|
|
488
509
|
}
|
|
489
510
|
|
|
490
|
-
if (!status) {
|
|
511
|
+
if (!status && rawResponse === null) {
|
|
491
512
|
if (++iterationsWithoutStatus >= STALL_ITERATIONS)
|
|
492
513
|
throw new Error(`${logLabel} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
|
|
514
|
+
// EMPTY-OUTPUT RE-PROMPT: under the unbudgeted default (maxIterations 1)
|
|
515
|
+
// a single turn that emits neither <status> nor <response> would
|
|
516
|
+
// otherwise exhaust the budget with ZERO corrective feedback. Treat it
|
|
517
|
+
// like the contract-violation branches — refund the iteration and
|
|
518
|
+
// re-prompt with explicit feedback, bounded by the shared retry pool.
|
|
519
|
+
// The stall counter above still hard-bounds consecutive empty turns.
|
|
520
|
+
if (opts.responseSchema && schemaRetriesLeft-- > 0) {
|
|
521
|
+
lastResponseValidationError = "no <response> block found — the turn ended with neither a <status> nor a <response> block; emit the complete <response> JSON";
|
|
522
|
+
process.stdout.write(`${logLabel} no <status>/<response> emitted — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
|
|
523
|
+
iteration--;
|
|
524
|
+
continue;
|
|
525
|
+
}
|
|
493
526
|
} else {
|
|
527
|
+
// A parsed structured response (even one that failed schema validation)
|
|
528
|
+
// is real output, not a stall — the contract retry below handles it.
|
|
494
529
|
iterationsWithoutStatus = 0;
|
|
495
530
|
}
|
|
496
531
|
|
|
@@ -506,6 +541,20 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
|
|
|
506
541
|
blockerStreak = null;
|
|
507
542
|
}
|
|
508
543
|
|
|
544
|
+
// CONTRACT RETRY (catch-all): the agent EMITTED a <response> block this turn,
|
|
545
|
+
// it failed schema validation, and it didn't go through the <status> branches
|
|
546
|
+
// above (structured-only output, no <status>). Re-prompt with the validation
|
|
547
|
+
// errors as feedback, FREE of the iteration budget. Gated on an actual
|
|
548
|
+
// validation failure this turn AND on the agent not signalling it's still
|
|
549
|
+
// working (`exit_signal: false`) — an in-progress turn that happens to carry
|
|
550
|
+
// a draft <response> consumes its budget normally instead of draining the
|
|
551
|
+
// shared retry pool that genuine contract violations rely on.
|
|
552
|
+
if (opts.responseSchema && rawResponse !== null && lastResponseValidationError !== "" && status?.exit_signal !== false && schemaRetriesLeft-- > 0) {
|
|
553
|
+
process.stdout.write(`${logLabel} response contract not yet satisfied — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
|
|
554
|
+
iteration--;
|
|
555
|
+
continue;
|
|
556
|
+
}
|
|
557
|
+
|
|
509
558
|
if (iteration + 1 < maxIterations)
|
|
510
559
|
// Loop continuation — progress.
|
|
511
560
|
process.stdout.write(`${logLabel} continuing to iteration ${iteration + 2}/${maxIterations}\n`);
|
package/src/client.ts
CHANGED
|
@@ -17,7 +17,7 @@ import { parseSseStream } from "./sse.js";
|
|
|
17
17
|
import type { SandboxNetworkPolicy } from "./sandbox.js";
|
|
18
18
|
import type { RunEvent } from "./types/events.js";
|
|
19
19
|
import type { WorkflowPlan } from "./types/workflow-plan.js";
|
|
20
|
-
import type { SnapshotConfig, IOSchema } from "./types/workflow-metadata.js";
|
|
20
|
+
import type { SnapshotConfig, IOSchema, ConnectorRequirements, ConnectorOperationTag, InvokePolicy } from "./types/workflow-metadata.js";
|
|
21
21
|
import type { WorkflowManifest } from "./utils/bundler.js";
|
|
22
22
|
|
|
23
23
|
/** UUID-v4-ish — matches the server-side predicate. Used to auto-detect
|
|
@@ -52,13 +52,6 @@ export interface RegisterResult {
|
|
|
52
52
|
name: string;
|
|
53
53
|
version: string;
|
|
54
54
|
runtimes?: RegisteredRuntime[];
|
|
55
|
-
/** Non-fatal advisories from the server. Surfaced at register time so the
|
|
56
|
-
* operator sees them while still in front of the terminal — currently
|
|
57
|
-
* covers "memory extraction is configured but its workflow is not
|
|
58
|
-
* registered in this factory". Empty/undefined when registration was
|
|
59
|
-
* cleanly resolved against everything the workflow declares it
|
|
60
|
-
* depends on. */
|
|
61
|
-
warnings?: string[];
|
|
62
55
|
}
|
|
63
56
|
|
|
64
57
|
export interface RegisteredRuntime {
|
|
@@ -72,6 +65,19 @@ export interface RuntimeSourceInput {
|
|
|
72
65
|
source: string;
|
|
73
66
|
}
|
|
74
67
|
|
|
68
|
+
/** GitHub provenance for a registered template's source file — stored as
|
|
69
|
+
* `metadata.source` on the registration. `cloud-build` stamps the built
|
|
70
|
+
* commit's sha; the dashboard's manual link path writes `sha: "manual"`. */
|
|
71
|
+
export interface TemplateSourceRef {
|
|
72
|
+
owner: string;
|
|
73
|
+
repo: string;
|
|
74
|
+
branch: string;
|
|
75
|
+
/** Repo-relative file path, e.g. `.agentc/workflows/workflow-deploy.ts`. */
|
|
76
|
+
path: string;
|
|
77
|
+
/** Commit sha the version was built from, or `"manual"` for hand-links. */
|
|
78
|
+
sha: string;
|
|
79
|
+
}
|
|
80
|
+
|
|
75
81
|
export interface RegisterWorkflowInput {
|
|
76
82
|
name: string;
|
|
77
83
|
source: string;
|
|
@@ -83,6 +89,9 @@ export interface RegisterWorkflowInput {
|
|
|
83
89
|
* executed on the server. */
|
|
84
90
|
manifest: WorkflowManifest;
|
|
85
91
|
version?: string;
|
|
92
|
+
/** Where the source file lives on GitHub — stored as `metadata.source`.
|
|
93
|
+
* Named `sourceRef` because `source` is the bundled code itself. */
|
|
94
|
+
sourceRef?: TemplateSourceRef;
|
|
86
95
|
schedule?: string;
|
|
87
96
|
runtimes?: RuntimeSourceInput[];
|
|
88
97
|
/** Human-readable description declared via
|
|
@@ -96,13 +105,16 @@ export interface RegisterWorkflowInput {
|
|
|
96
105
|
snapshots?: SnapshotConfig;
|
|
97
106
|
/** Provider-neutral execution plan detected by the CLI bundler. */
|
|
98
107
|
workflowPlan?: WorkflowPlan;
|
|
99
|
-
/**
|
|
100
|
-
*
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
108
|
+
/** Connector requirements declared via `defineWorkflow({ connectors })`
|
|
109
|
+
* (ADR-0007). Validated against the server's provider registry at
|
|
110
|
+
* registration; tokens are injected at the network layer at dispatch. */
|
|
111
|
+
connectors?: ConnectorRequirements;
|
|
112
|
+
/** Connector-catalogue operation tag — see `ConnectorOperationTag`. */
|
|
113
|
+
connectorOperation?: ConnectorOperationTag;
|
|
114
|
+
/** Tier-1 invoke ACL declared via `defineWorkflow({ invokePolicy })`.
|
|
115
|
+
* Only meaningful when the workflow also declares `connectors` — the
|
|
116
|
+
* server gates dispatch on it before binding any grant. */
|
|
117
|
+
invokePolicy?: InvokePolicy;
|
|
106
118
|
/** Input schema extracted from the workflow's `input` zod schema. */
|
|
107
119
|
inputSchema?: IOSchema;
|
|
108
120
|
/** Output schema extracted from the workflow's `output` zod schema. */
|
|
@@ -127,19 +139,16 @@ export interface InvokeWorkflowOptions {
|
|
|
127
139
|
* vars after brokering. Replaces the template-level placeholders for
|
|
128
140
|
* this run only — registered metadata is not mutated. */
|
|
129
141
|
placeholders?: Record<string, string>;
|
|
130
|
-
/** Per-invocation memory-extractor override — `false` skips the
|
|
131
|
-
* built-in memory hook for this run; omitting leaves the registered
|
|
132
|
-
* default in place. */
|
|
133
|
-
memory?: boolean;
|
|
134
|
-
/** Per-invocation post-hook override — replaces the registered
|
|
135
|
-
* `postRunHooks` array for this run only. */
|
|
136
|
-
postRunHooks?: readonly string[];
|
|
137
142
|
/** Explicit parent run id. Pass `null` to suppress ambient RUN_ID auto-detection. */
|
|
138
143
|
parentRunId?: string | null;
|
|
139
144
|
/** Agent loop inside the parent run that caused this invoke, when applicable. */
|
|
140
145
|
agentId?: string | null;
|
|
141
146
|
/** Factory slug. Defaults to `"default"`. */
|
|
142
147
|
factorySlug?: string;
|
|
148
|
+
/** Idempotency key — sent as the `Idempotency-Key` header. A repeat invoke
|
|
149
|
+
* with the same key inside the server's dedup window returns the original
|
|
150
|
+
* run instead of starting a new one (matches `resumePause`'s pattern). */
|
|
151
|
+
idempotencyKey?: string;
|
|
143
152
|
}
|
|
144
153
|
|
|
145
154
|
export interface InvokeAndWaitOptions extends InvokeWorkflowOptions {
|
|
@@ -234,6 +243,13 @@ export interface RunStatus<TOutput = unknown> {
|
|
|
234
243
|
id: string;
|
|
235
244
|
status: RunState;
|
|
236
245
|
output?: TOutput;
|
|
246
|
+
/** The run's latest (`saveLatest`) snapshot id, populated once the run has
|
|
247
|
+
* succeeded — the boot source to fork this run's evolved filesystem from
|
|
248
|
+
* (pass as `snapshots.bootFrom` on a follow-up invoke). `null` while the run
|
|
249
|
+
* is still in flight or when it captured no snapshot. Lets an orchestrator
|
|
250
|
+
* fork a child straight off the `invokeChild` result without a separate
|
|
251
|
+
* `listRunSnapshots` call. */
|
|
252
|
+
latestSnapshotId?: string | null;
|
|
237
253
|
}
|
|
238
254
|
|
|
239
255
|
/** ADR-0006 step 10 — actor record returned on a successful resume.
|
|
@@ -380,6 +396,41 @@ export interface EventRow {
|
|
|
380
396
|
createdAt: string;
|
|
381
397
|
}
|
|
382
398
|
|
|
399
|
+
// The server emits these rows in snake_case; the client maps them to
|
|
400
|
+
// camelCase at the fetch boundary so the SDK surface stays uniform
|
|
401
|
+
// (`EventRow.createdAt`, `RunStatus.latestSnapshotId`, …).
|
|
402
|
+
export interface RunArtifactRow {
|
|
403
|
+
path: string;
|
|
404
|
+
factorySlug: string | null;
|
|
405
|
+
sizeBytes: number | null;
|
|
406
|
+
lastWriteAt: string;
|
|
407
|
+
/** Opening text of the file (≤320 chars) — null for binary/empty. */
|
|
408
|
+
preview: string | null;
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
export interface FactoryFileWriteResult {
|
|
412
|
+
path: string;
|
|
413
|
+
contentHash: string;
|
|
414
|
+
sizeBytes: number;
|
|
415
|
+
created: boolean;
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
/** Wire shapes — what the server actually emits (snake_case). */
|
|
419
|
+
interface RunArtifactWire {
|
|
420
|
+
path: string;
|
|
421
|
+
factory_slug: string | null;
|
|
422
|
+
size_bytes: number | null;
|
|
423
|
+
last_write_at: string;
|
|
424
|
+
preview: string | null;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
interface FactoryFileWriteWire {
|
|
428
|
+
path: string;
|
|
429
|
+
content_hash: string;
|
|
430
|
+
size_bytes: number;
|
|
431
|
+
created: boolean;
|
|
432
|
+
}
|
|
433
|
+
|
|
383
434
|
export interface ReportEventInput {
|
|
384
435
|
name: string;
|
|
385
436
|
body: unknown;
|
|
@@ -395,7 +446,7 @@ export interface ListEventsOptions {
|
|
|
395
446
|
factorySlug?: string;
|
|
396
447
|
limit?: number;
|
|
397
448
|
/** Case-insensitive substring match. Server uses `ILIKE %name%`, so
|
|
398
|
-
* `"
|
|
449
|
+
* `"site"` matches `site.created`, `site.failed`, etc. Pass the
|
|
399
450
|
* full event name for an effectively-exact filter (any string is a
|
|
400
451
|
* substring of itself). */
|
|
401
452
|
name?: string;
|
|
@@ -600,15 +651,16 @@ export class AgentComposeClient {
|
|
|
600
651
|
? detectAmbientParentRunId()
|
|
601
652
|
: opts.parentRunId;
|
|
602
653
|
const factorySlug = opts?.factorySlug ?? DEFAULT_FACTORY;
|
|
654
|
+
const headers: Record<string, string> = {};
|
|
655
|
+
if (opts?.idempotencyKey) headers["Idempotency-Key"] = opts.idempotencyKey;
|
|
603
656
|
return this.fetch(templatePath(factorySlug, name, "invoke"), {
|
|
604
657
|
method: "POST",
|
|
658
|
+
...(opts?.idempotencyKey ? { headers } : {}),
|
|
605
659
|
body: {
|
|
606
660
|
input,
|
|
607
661
|
...(opts?.snapshots !== undefined ? { snapshots: opts.snapshots } : {}),
|
|
608
662
|
...(opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {}),
|
|
609
663
|
...(opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {}),
|
|
610
|
-
...(opts?.memory !== undefined ? { memory: opts.memory } : {}),
|
|
611
|
-
...(opts?.postRunHooks !== undefined ? { postRunHooks: opts.postRunHooks } : {}),
|
|
612
664
|
...(parentRunId ? { parentRunId } : {}),
|
|
613
665
|
...(opts?.agentId ? { agentId: opts.agentId } : {}),
|
|
614
666
|
},
|
|
@@ -895,6 +947,73 @@ export class AgentComposeClient {
|
|
|
895
947
|
return body.events;
|
|
896
948
|
}
|
|
897
949
|
|
|
950
|
+
/** Files the run wrote on the factory drive — run-attributed revisions,
|
|
951
|
+
* latest write per path, paths the run later deleted excluded. */
|
|
952
|
+
async listRunArtifacts(runId: string): Promise<RunArtifactRow[]> {
|
|
953
|
+
const body = await this.fetch<{ artifacts: RunArtifactWire[] }>(
|
|
954
|
+
`/api/v1/workflows/${encodeURIComponent(runId)}/artifacts`,
|
|
955
|
+
);
|
|
956
|
+
return body.artifacts.map((a) => ({
|
|
957
|
+
path: a.path,
|
|
958
|
+
factorySlug: a.factory_slug,
|
|
959
|
+
sizeBytes: a.size_bytes,
|
|
960
|
+
lastWriteAt: a.last_write_at,
|
|
961
|
+
preview: a.preview,
|
|
962
|
+
}));
|
|
963
|
+
}
|
|
964
|
+
|
|
965
|
+
// ── Factory files ──────────────────────────────────────────────────────────
|
|
966
|
+
// The factory drive: documents surfaced in the dashboard's Files tab.
|
|
967
|
+
// Writes from inside a sandbox automatically carry the run-callback token
|
|
968
|
+
// (AGENT_COMPOSE_RUN_TOKEN), so the revision is attributed to the run —
|
|
969
|
+
// that's what surfaces the doc in the Workbench docs section and the
|
|
970
|
+
// editor-avatar run hover.
|
|
971
|
+
|
|
972
|
+
/** Write (create or overwrite) one file on a factory's drive. */
|
|
973
|
+
async putFactoryFile(
|
|
974
|
+
path: string,
|
|
975
|
+
content: string | Uint8Array,
|
|
976
|
+
opts?: { factorySlug?: string; contentType?: string },
|
|
977
|
+
): Promise<FactoryFileWriteResult> {
|
|
978
|
+
const factorySlug = opts?.factorySlug
|
|
979
|
+
?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined)
|
|
980
|
+
?? DEFAULT_FACTORY;
|
|
981
|
+
const runToken = typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_RUN_TOKEN : undefined;
|
|
982
|
+
const wire = await this.fetch<FactoryFileWriteWire>(
|
|
983
|
+
`/api/v1/factories/${encodeURIComponent(factorySlug)}/files/content?path=${encodeURIComponent(path)}`,
|
|
984
|
+
{
|
|
985
|
+
method: "PUT",
|
|
986
|
+
body: content,
|
|
987
|
+
headers: {
|
|
988
|
+
"content-type": opts?.contentType ?? "text/plain; charset=utf-8",
|
|
989
|
+
...(runToken ? { "x-run-token": runToken } : {}),
|
|
990
|
+
},
|
|
991
|
+
},
|
|
992
|
+
);
|
|
993
|
+
return {
|
|
994
|
+
path: wire.path,
|
|
995
|
+
contentHash: wire.content_hash,
|
|
996
|
+
sizeBytes: wire.size_bytes,
|
|
997
|
+
created: wire.created,
|
|
998
|
+
};
|
|
999
|
+
}
|
|
1000
|
+
|
|
1001
|
+
/** Read one file's current content (or a specific revision) as text. */
|
|
1002
|
+
async getFactoryFile(
|
|
1003
|
+
path: string,
|
|
1004
|
+
opts?: { factorySlug?: string; revision?: number },
|
|
1005
|
+
): Promise<string> {
|
|
1006
|
+
const factorySlug = opts?.factorySlug
|
|
1007
|
+
?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined)
|
|
1008
|
+
?? DEFAULT_FACTORY;
|
|
1009
|
+
const q = new URLSearchParams({ path });
|
|
1010
|
+
if (opts?.revision !== undefined) q.set("revision", String(opts.revision));
|
|
1011
|
+
return this.fetch(
|
|
1012
|
+
`/api/v1/factories/${encodeURIComponent(factorySlug)}/files/content?${q}`,
|
|
1013
|
+
{ responseType: "text" },
|
|
1014
|
+
);
|
|
1015
|
+
}
|
|
1016
|
+
|
|
898
1017
|
/** List events ingested into a factory, newest first. Supports
|
|
899
1018
|
* case-insensitive substring filter (`name`) and timestamp-cursor
|
|
900
1019
|
* pagination (`before`). Returns `{ events, has_more }` — the
|
package/src/index.ts
CHANGED
|
@@ -38,10 +38,11 @@ export type {
|
|
|
38
38
|
WorkflowHooks,
|
|
39
39
|
SnapshotConfig,
|
|
40
40
|
BootSnapshot,
|
|
41
|
+
ReuseSnapshot,
|
|
41
42
|
IOSchema,
|
|
42
43
|
OutputSchema,
|
|
43
|
-
WorkflowMemoryConfig,
|
|
44
44
|
} from "./types/workflow.js";
|
|
45
|
+
export type { ConnectorRequestRules } from "./types/workflow-metadata.js";
|
|
45
46
|
|
|
46
47
|
// Snapshot entry type re-exported for consumers (dashboard, CLI).
|
|
47
48
|
export type { RunSnapshotEntry } from "./client.js";
|
|
@@ -105,7 +106,7 @@ export type {
|
|
|
105
106
|
// HTTP client
|
|
106
107
|
export { AgentComposeClient } from "./client.js";
|
|
107
108
|
export type {
|
|
108
|
-
RegisterResult, RegisterWorkflowInput, RuntimeSourceInput,
|
|
109
|
+
RegisterResult, RegisterWorkflowInput, RuntimeSourceInput, TemplateSourceRef,
|
|
109
110
|
InvokeWorkflowOptions, InvokeAndWaitOptions, InvokeResult,
|
|
110
111
|
ListSnapshotsOptions, TemplateRow, ListTemplatesOptions,
|
|
111
112
|
CreateFactoryInput, UpdateFactoryInput,
|
|
@@ -180,6 +181,7 @@ export { createSandbox, reconnectSandbox, killAllSandboxes, killSandboxById,
|
|
|
180
181
|
getSandboxQuotas, listOwnedSandboxes, deleteSandboxSnapshot,
|
|
181
182
|
makeSandboxProvider, makeDesktopSandboxProvider,
|
|
182
183
|
parseSseExecStream, AGENT_COMPOSE_TAG } from "./sandbox.js";
|
|
184
|
+
export { SandboxUnavailableError, SANDBOX_UNAVAILABLE_PREFIX } from "./sandbox-errors.js";
|
|
183
185
|
export type {
|
|
184
186
|
SandboxCreateOpts, SandboxNetworkPolicy, SandboxNetworkHeaderTransform,
|
|
185
187
|
SandboxNetworkAllowRule, SandboxNetworkSubnetPolicy, SandboxProviderName,
|
|
@@ -12,14 +12,16 @@
|
|
|
12
12
|
* AsyncQueue bridge → init/usage/done/error lifecycle), parameterised per-CLI
|
|
13
13
|
* by a `CliAgentSpec`.
|
|
14
14
|
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
15
|
+
* Provisioning (per spec): the runtime installs the provider CLI on demand —
|
|
16
|
+
* `command -v <bin>` before the first turn, falling back to the spec's
|
|
17
|
+
* `install` command when it's absent — so it works on a bare sandbox with no
|
|
18
|
+
* manual setup or hardcoded snapshot id. Pair it with
|
|
19
|
+
* `snapshots: { bootFrom: "reuse" }` on the workflow and the install happens
|
|
20
|
+
* exactly once: the first run installs the CLI and captures a snapshot, and
|
|
21
|
+
* every run after boots from that snapshot with the CLI already present (the
|
|
22
|
+
* probe short-circuits). The provider's API key must be in the sandbox env
|
|
23
|
+
* (see each spec's `authEnv`); `commands.run` inherits the sandbox env, so
|
|
24
|
+
* values set via `agentc secrets set` are visible to the CLI.
|
|
23
25
|
*/
|
|
24
26
|
|
|
25
27
|
import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider } from "../index.js";
|
|
@@ -44,6 +46,15 @@ export interface CliAgentSpec {
|
|
|
44
46
|
authEnv: string;
|
|
45
47
|
/** Default model id when none is configured; omit to let the CLI choose. */
|
|
46
48
|
defaultModel?: string;
|
|
49
|
+
/** Binary the runtime spawns (`codex`, `amp`). Probed with `command -v`
|
|
50
|
+
* before the first turn; if absent, `install` provisions it. */
|
|
51
|
+
bin: string;
|
|
52
|
+
/** Shell command that installs `bin` when it's missing from the sandbox.
|
|
53
|
+
* Runs at most once per runner, and only when the probe fails — so booting
|
|
54
|
+
* from a snapshot that already has the CLI (the steady state under
|
|
55
|
+
* `snapshots: { bootFrom: "reuse" }`) skips it. Must leave `bin` resolvable
|
|
56
|
+
* on a non-login shell's PATH (the runtime spawns via `sh -c`). */
|
|
57
|
+
install: string;
|
|
47
58
|
/** Serialise the user prompt into the bytes written to the prompt file —
|
|
48
59
|
* plain text for a CLI that reads the prompt from stdin (codex `-`), or a
|
|
49
60
|
* JSONL user message for a `--stream-json-input` CLI (amp). */
|
|
@@ -76,6 +87,23 @@ export class CliAgentRunner implements ModelExecutionContract {
|
|
|
76
87
|
return this.configModel ?? this.options.model ?? this.spec.defaultModel;
|
|
77
88
|
}
|
|
78
89
|
|
|
90
|
+
private installed = false;
|
|
91
|
+
|
|
92
|
+
/** Provision the CLI on demand. A no-op once `bin` is on PATH — the steady
|
|
93
|
+
* state under `snapshots: { bootFrom: "reuse" }`, where the first run's
|
|
94
|
+
* install is baked into the snapshot every later run boots from. So the
|
|
95
|
+
* install command runs exactly once: on the first run of a content hash. */
|
|
96
|
+
private async ensureInstalled(): Promise<void> {
|
|
97
|
+
if (this.installed) return;
|
|
98
|
+
const probe = await this.sandbox.commands.run(`command -v ${this.spec.bin}`);
|
|
99
|
+
if (probe.exitCode === 0) { this.installed = true; return; }
|
|
100
|
+
const res = await this.sandbox.commands.run(this.spec.install, { timeoutMs: 300_000 });
|
|
101
|
+
if (res.exitCode !== 0) {
|
|
102
|
+
throw new Error(`failed to install ${this.spec.bin}: ${(res.stderr || res.stdout || "").slice(-500)}`);
|
|
103
|
+
}
|
|
104
|
+
this.installed = true;
|
|
105
|
+
}
|
|
106
|
+
|
|
79
107
|
// No captureCheckpoint/restoreCheckpoint: the CLI persists its thread/rollout
|
|
80
108
|
// on the sandbox filesystem (which round-trips through the pause snapshot),
|
|
81
109
|
// and the loop already carries the session id we emit on init/done and pass
|
|
@@ -87,45 +115,50 @@ export class CliAgentRunner implements ModelExecutionContract {
|
|
|
87
115
|
iteration?: number;
|
|
88
116
|
signal?: AbortSignal;
|
|
89
117
|
}): AsyncGenerator<AgentMessage> {
|
|
90
|
-
const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
|
|
91
|
-
await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
|
|
92
|
-
const cmd = this.spec.buildCommand({
|
|
93
|
-
promptPath,
|
|
94
|
-
sessionId: opts.sessionId,
|
|
95
|
-
model: this.model,
|
|
96
|
-
cwd: this.options.cwd,
|
|
97
|
-
});
|
|
98
|
-
|
|
99
|
-
// Bridge the streaming stdout callback into an async-iterable of complete
|
|
100
|
-
// JSONL lines. `onStdout` chunks aren't line-aligned, so buffer + split.
|
|
101
|
-
const lines = new AsyncQueue<string>();
|
|
102
|
-
let buf = "";
|
|
103
|
-
const onStdout = (data: string) => {
|
|
104
|
-
buf += data;
|
|
105
|
-
let nl: number;
|
|
106
|
-
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
107
|
-
const line = buf.slice(0, nl).trim();
|
|
108
|
-
buf = buf.slice(nl + 1);
|
|
109
|
-
if (line) lines.push(line);
|
|
110
|
-
}
|
|
111
|
-
};
|
|
112
|
-
|
|
113
|
-
// `commands.run` resolves when the process exits. Kick it off (don't await
|
|
114
|
-
// yet); flush the trailing buffer + close the queue on completion so the
|
|
115
|
-
// for-await below drains and we can read the exit code.
|
|
116
|
-
const runPromise = this.sandbox.commands.run(cmd, {
|
|
117
|
-
...(this.options.cwd ? { cwd: this.options.cwd } : {}),
|
|
118
|
-
onStdout,
|
|
119
|
-
}).then(
|
|
120
|
-
(res) => { const tail = buf.trim(); if (tail) lines.push(tail); lines.close(); return res; },
|
|
121
|
-
(err) => { lines.close(); throw err; },
|
|
122
|
-
);
|
|
123
|
-
|
|
124
118
|
yield { type: "init", sessionId: opts.sessionId ?? "", timestamp: now() };
|
|
125
119
|
|
|
126
120
|
let sessionId = opts.sessionId;
|
|
127
121
|
let sawError = false;
|
|
128
122
|
try {
|
|
123
|
+
// Provision the CLI before the first turn (no-op when it's already
|
|
124
|
+
// present, e.g. booting from a "reuse" snapshot). A failure here surfaces
|
|
125
|
+
// as an `error` AgentMessage via the catch below.
|
|
126
|
+
await this.ensureInstalled();
|
|
127
|
+
|
|
128
|
+
const promptPath = `/tmp/ac-${this.spec.kind}-${this.options.agentId ?? "agent"}-${opts.iteration ?? 0}.in`;
|
|
129
|
+
await this.sandbox.files.write(promptPath, this.spec.promptPayload(opts.prompt));
|
|
130
|
+
const cmd = this.spec.buildCommand({
|
|
131
|
+
promptPath,
|
|
132
|
+
sessionId: opts.sessionId,
|
|
133
|
+
model: this.model,
|
|
134
|
+
cwd: this.options.cwd,
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
// Bridge the streaming stdout callback into an async-iterable of complete
|
|
138
|
+
// JSONL lines. `onStdout` chunks aren't line-aligned, so buffer + split.
|
|
139
|
+
const lines = new AsyncQueue<string>();
|
|
140
|
+
let buf = "";
|
|
141
|
+
const onStdout = (data: string) => {
|
|
142
|
+
buf += data;
|
|
143
|
+
let nl: number;
|
|
144
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
145
|
+
const line = buf.slice(0, nl).trim();
|
|
146
|
+
buf = buf.slice(nl + 1);
|
|
147
|
+
if (line) lines.push(line);
|
|
148
|
+
}
|
|
149
|
+
};
|
|
150
|
+
|
|
151
|
+
// `commands.run` resolves when the process exits. Kick it off (don't await
|
|
152
|
+
// yet); flush the trailing buffer + close the queue on completion so the
|
|
153
|
+
// for-await below drains and we can read the exit code.
|
|
154
|
+
const runPromise = this.sandbox.commands.run(cmd, {
|
|
155
|
+
...(this.options.cwd ? { cwd: this.options.cwd } : {}),
|
|
156
|
+
onStdout,
|
|
157
|
+
}).then(
|
|
158
|
+
(res) => { const tail = buf.trim(); if (tail) lines.push(tail); lines.close(); return res; },
|
|
159
|
+
(err) => { lines.close(); throw err; },
|
|
160
|
+
);
|
|
161
|
+
|
|
129
162
|
for await (const line of lines) {
|
|
130
163
|
let parsed: Record<string, unknown>;
|
|
131
164
|
try {
|
package/src/runtimes/amp.ts
CHANGED
|
@@ -5,13 +5,14 @@
|
|
|
5
5
|
* own loop + tools, so we only stream-parse what it prints.
|
|
6
6
|
*
|
|
7
7
|
* Auth: set `AMP_API_KEY` (`sgamp_…`) in the sandbox env via a workflow secret.
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* the
|
|
8
|
+
* The runtime installs the `amp` CLI (`@ampcode/cli`) on demand — no image
|
|
9
|
+
* baking needed; pair with `snapshots: { bootFrom: "reuse" }` to install once
|
|
10
|
+
* and boot from the captured snapshot on every run after. The model is chosen
|
|
11
|
+
* by the AMP_API_KEY account (e.g. a GPT-only token runs GPT); the runtime
|
|
12
|
+
* doesn't pin a model.
|
|
11
13
|
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
* before production use.
|
|
14
|
+
* Verified against a live `amp -x --stream-json` run: the user-message stdin
|
|
15
|
+
* shape, assistant content blocks, and result usage below all round-trip.
|
|
15
16
|
*/
|
|
16
17
|
|
|
17
18
|
import type { AgentMessage } from "../index.js";
|
|
@@ -32,6 +33,10 @@ function blocks(msg: Record<string, unknown>): Array<Record<string, unknown>> {
|
|
|
32
33
|
const ampSpec: CliAgentSpec = {
|
|
33
34
|
kind: "amp",
|
|
34
35
|
authEnv: "AMP_API_KEY",
|
|
36
|
+
bin: "amp",
|
|
37
|
+
// Global npm install; symlink onto PATH only if the global bin dir isn't
|
|
38
|
+
// already there (so a non-login `sh -c` can find it).
|
|
39
|
+
install: 'sudo npm install -g @ampcode/cli && (command -v amp >/dev/null 2>&1 || sudo ln -sf "$(npm prefix -g)/bin/amp" /usr/local/bin/amp)',
|
|
35
40
|
// `--stream-json-input` reads JSON Lines user messages from stdin; write one.
|
|
36
41
|
// amp's --stream-json-input wants Claude-shaped content blocks, not a bare
|
|
37
42
|
// string (it rejects a string `content` with "expected array, received string").
|