@agent-compose/sdk 0.5.1 → 0.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/active-step.d.ts +60 -0
- package/dist/agent/agent-loop-steer.test.d.ts +1 -0
- package/dist/agent/agent-loop.d.ts +46 -0
- package/dist/agent/async-queue.d.ts +29 -0
- package/dist/agent/protocol.d.ts +9 -1
- package/dist/agent/resolve-agent-id.test.d.ts +1 -0
- package/dist/agent/run-agent.d.ts +16 -3
- package/dist/agent/steer-control.d.ts +57 -0
- package/dist/agent/steer-control.test.d.ts +1 -0
- package/dist/client.d.ts +161 -0
- package/dist/index.d.ts +16 -6
- package/dist/index.js +1586 -158
- package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/errors.test.d.ts +1 -0
- package/dist/pause/__tests__/manager.test.d.ts +1 -0
- package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
- package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
- package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
- package/dist/pause/checkpoint.d.ts +28 -0
- package/dist/pause/errors.d.ts +52 -0
- package/dist/pause/manager.d.ts +63 -0
- package/dist/pause/pause-core.d.ts +101 -0
- package/dist/pause/state-dir.d.ts +80 -0
- package/dist/pause/wrappers.d.ts +41 -0
- package/dist/request-context/request-context.d.ts +12 -0
- package/dist/runtimes/_cli-agent.d.ts +72 -0
- package/dist/runtimes/amp.d.ts +22 -0
- package/dist/runtimes/claude.d.ts +6 -0
- package/dist/runtimes/codex.d.ts +20 -0
- package/dist/runtimes/openai-desktop.d.ts +2 -0
- package/dist/runtimes/openai-desktop.js +1578 -157
- package/dist/runtimes/vercel.d.ts +53 -1
- package/dist/runtimes/vercel.js +60 -8
- package/dist/runtimes/vercel.test.d.ts +1 -0
- package/dist/sse.d.ts +2 -3
- package/dist/step-invocation/index.d.ts +2 -2
- package/dist/step-invocation/invoker.d.ts +3 -0
- package/dist/step-invocation/protocol.d.ts +12 -0
- package/dist/step-invocation/server.d.ts +1 -0
- package/dist/step-invocation/types.d.ts +40 -5
- package/dist/types/events.d.ts +9 -0
- package/dist/types/execution-context.d.ts +25 -0
- package/dist/types/protocol.d.ts +8 -0
- package/dist/types/runtime.d.ts +55 -0
- package/dist/types/sandbox.d.ts +4 -4
- package/dist/utils/schemas.d.ts +2 -0
- package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
- package/dist/workflow-steps/index.d.ts +2 -0
- package/dist/workflow-steps/observability.d.ts +43 -11
- package/dist/workflow-steps/run-callback.d.ts +39 -0
- package/dist/workflow-steps/runner.d.ts +8 -0
- package/package.json +1 -1
- package/src/active-step.ts +124 -0
- package/src/agent/agent-loop.ts +253 -19
- package/src/agent/async-queue.ts +61 -0
- package/src/agent/protocol.ts +12 -2
- package/src/agent/run-agent.ts +184 -8
- package/src/agent/steer-control.ts +125 -0
- package/src/client.ts +277 -0
- package/src/index.ts +38 -4
- package/src/pause/checkpoint.ts +44 -0
- package/src/pause/errors.ts +70 -0
- package/src/pause/manager.ts +177 -0
- package/src/pause/pause-core.ts +267 -0
- package/src/pause/state-dir.ts +262 -0
- package/src/pause/wrappers.ts +79 -0
- package/src/request-context/request-context.ts +17 -2
- package/src/runtimes/_cli-agent.ts +161 -0
- package/src/runtimes/amp.ts +94 -0
- package/src/runtimes/claude.ts +101 -6
- package/src/runtimes/codex.ts +109 -0
- package/src/runtimes/openai-desktop.ts +11 -0
- package/src/runtimes/vercel.ts +78 -2
- package/src/sandbox.ts +39 -20
- package/src/sse.ts +8 -6
- package/src/step-invocation/index.ts +2 -1
- package/src/step-invocation/invoker.ts +107 -29
- package/src/step-invocation/protocol.ts +16 -0
- package/src/step-invocation/server.ts +45 -12
- package/src/step-invocation/types.ts +43 -7
- package/src/tools/coding.ts +16 -5
- package/src/types/events.ts +9 -0
- package/src/types/execution-context.ts +25 -0
- package/src/types/protocol.ts +8 -0
- package/src/types/runtime.ts +52 -0
- package/src/types/sandbox.ts +8 -4
- package/src/types/workflow.ts +6 -1
- package/src/utils/bundler.ts +8 -3
- package/src/utils/schemas.ts +2 -0
- package/src/workflow-steps/index.ts +3 -0
- package/src/workflow-steps/observability.ts +84 -13
- package/src/workflow-steps/run-callback.ts +72 -0
- package/src/workflow-steps/runner.ts +70 -8
- package/dist/utils/discovery.d.ts +0 -2
- package/src/utils/discovery.ts +0 -4
|
@@ -7,22 +7,54 @@ import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvi
|
|
|
7
7
|
import { type CodingTool } from "../tools/index.js";
|
|
8
8
|
import type { ProcessorContext, ToolCall } from "../processors/processor.js";
|
|
9
9
|
export interface VercelRuntimeConfig {
|
|
10
|
-
/**
|
|
10
|
+
/** The model to drive — this is the only thing that varies per provider;
|
|
11
|
+
* there is no per-provider runtime. `LanguageModel` accepts every provider:
|
|
12
|
+
* - a gateway model-id string routed via the Vercel AI Gateway (set
|
|
13
|
+
* `AI_GATEWAY_API_KEY`; no provider package needed), e.g. "openai/gpt-5",
|
|
14
|
+
* "google/gemini-2.5-pro", "xai/grok-4", "deepseek/deepseek-chat",
|
|
15
|
+
* "mistral/mistral-large-latest", "anthropic/claude-sonnet-4-5";
|
|
16
|
+
* - or a `LanguageModel` object from a provider package (add the dep + set
|
|
17
|
+
* its API-key env), e.g. `openai("gpt-5")`, `google("gemini-2.5-pro")`. */
|
|
11
18
|
model: LanguageModel;
|
|
12
19
|
/** Optional system prompt prepended to every model call. */
|
|
13
20
|
system?: string;
|
|
14
21
|
/** Override/extend the default coding tools. Defaults: Read, Write, Edit, Bash. */
|
|
15
22
|
tools?: readonly CodingTool[];
|
|
23
|
+
/** Short runtime self-id surfaced on `agent.spawned` so the dashboard can
|
|
24
|
+
* show a per-agent runtime icon (e.g. "openai", "gemini"). Provider presets
|
|
25
|
+
* set this; bare `createVercelRuntime` callers can leave it unset. */
|
|
26
|
+
kind?: string;
|
|
27
|
+
/** Display model id surfaced on `agent.spawned` (the Agent tab labels which
|
|
28
|
+
* model each agent ran). `model` above is the AI SDK LanguageModel object;
|
|
29
|
+
* this is its human-readable id string. */
|
|
30
|
+
modelId?: string;
|
|
16
31
|
}
|
|
17
32
|
export declare class VercelRunner implements ModelExecutionContract {
|
|
18
33
|
private readonly sandbox;
|
|
19
34
|
private readonly options;
|
|
20
35
|
private readonly config;
|
|
21
36
|
supportsToolCallProcessor: boolean;
|
|
37
|
+
/** Surfaced on `agent.spawned` for the dashboard's per-agent runtime icon +
|
|
38
|
+
* model label. Set from the (provider preset's) config; undefined for a
|
|
39
|
+
* bare `createVercelRuntime` that didn't label itself. */
|
|
40
|
+
readonly kind?: string;
|
|
41
|
+
readonly model?: string;
|
|
22
42
|
private readonly tools;
|
|
23
43
|
private readonly messages;
|
|
24
44
|
constructor(sandbox: SandboxProvider, options: RuntimeOptions, config: VercelRuntimeConfig);
|
|
25
45
|
gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
|
|
46
|
+
/** ADR-0006 pause-resume hooks. The Vercel AI SDK holds the running
|
|
47
|
+
* conversation client-side in `this.messages` — every `streamText`
|
|
48
|
+
* call passes the full array as `messages` and reconstructs it from
|
|
49
|
+
* `response.messages` after completion. A pause-induced subprocess
|
|
50
|
+
* exit loses the array; the new subprocess constructs a fresh
|
|
51
|
+
* `VercelRunner` with `this.messages = []` and the next sendMessage
|
|
52
|
+
* would see only the current iteration's prompt — conversation
|
|
53
|
+
* history broken. The agent loop calls these at every iteration
|
|
54
|
+
* boundary so the messages array round-trips through the sandbox
|
|
55
|
+
* snapshot. */
|
|
56
|
+
captureCheckpoint(): unknown;
|
|
57
|
+
restoreCheckpoint(blob: unknown): void;
|
|
26
58
|
private buildTools;
|
|
27
59
|
sendMessage(opts: {
|
|
28
60
|
prompt: string;
|
|
@@ -32,3 +64,23 @@ export declare class VercelRunner implements ModelExecutionContract {
|
|
|
32
64
|
}): AsyncGenerator<AgentMessage>;
|
|
33
65
|
}
|
|
34
66
|
export declare function createVercelRuntime(config: VercelRuntimeConfig): import("../index.js").AgentRuntime<SandboxProvider>;
|
|
67
|
+
/** One model offered by the Vercel AI Gateway. Pass `id` straight to
|
|
68
|
+
* `createVercelRuntime({ model: id })`. */
|
|
69
|
+
export interface VercelRuntimeModel {
|
|
70
|
+
/** Gateway model id, e.g. "openai/gpt-5". Usable directly as the runtime model. */
|
|
71
|
+
id: string;
|
|
72
|
+
/** Human-readable display name. */
|
|
73
|
+
name: string;
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* List the models the Vercel runtime accepts as a gateway model-id string — the
|
|
77
|
+
* LIVE Vercel AI Gateway catalog, so it never goes stale. This is the canonical
|
|
78
|
+
* answer to "what models can I pass to `createVercelRuntime`?" for the string
|
|
79
|
+
* form (`createVercelRuntime({ model: "openai/gpt-5" })`).
|
|
80
|
+
*
|
|
81
|
+
* Requires `AI_GATEWAY_API_KEY`. The other form — a `LanguageModel` object from
|
|
82
|
+
* an `@ai-sdk/<provider>` package — supports whatever that provider package
|
|
83
|
+
* does (see its docs); there's no single cross-form list because the runtime is
|
|
84
|
+
* model-agnostic. Browse the catalog in a UI at https://vercel.com/ai-gateway/models.
|
|
85
|
+
*/
|
|
86
|
+
export declare function listVercelRuntimeModels(): Promise<VercelRuntimeModel[]>;
|
package/dist/runtimes/vercel.js
CHANGED
|
@@ -18,7 +18,7 @@ var __toESM = (mod, isNodeMode, target) => {
|
|
|
18
18
|
var __require = /* @__PURE__ */ createRequire(import.meta.url);
|
|
19
19
|
|
|
20
20
|
// src/runtimes/vercel.ts
|
|
21
|
-
import { streamText, stepCountIs, tool } from "ai";
|
|
21
|
+
import { streamText, stepCountIs, tool, gateway } from "ai";
|
|
22
22
|
|
|
23
23
|
// src/types/runtime.ts
|
|
24
24
|
function defineRuntime(pkg) {
|
|
@@ -36,6 +36,17 @@ function q(value) {
|
|
|
36
36
|
function resolvePath(path, cwd) {
|
|
37
37
|
return cwd && !isAbsolute(path) ? join(cwd, path) : path;
|
|
38
38
|
}
|
|
39
|
+
function formatCommandFailure(command, result) {
|
|
40
|
+
const parts = [`Command failed with exit code ${result.exitCode}: ${command}`];
|
|
41
|
+
if (result.stderr.trim())
|
|
42
|
+
parts.push(`stderr:
|
|
43
|
+
${result.stderr}`);
|
|
44
|
+
if (result.stdout.trim())
|
|
45
|
+
parts.push(`stdout:
|
|
46
|
+
${result.stdout}`);
|
|
47
|
+
return parts.join(`
|
|
48
|
+
`);
|
|
49
|
+
}
|
|
39
50
|
async function readFile(sandbox, path, opts) {
|
|
40
51
|
const script = `
|
|
41
52
|
const fs = require("fs");
|
|
@@ -49,13 +60,19 @@ const end = Math.min(lines.length, start + Math.max(0, limit) - 1);
|
|
|
49
60
|
for (let i = start; i <= end; i++) console.log(String(i) + ": " + lines[i - 1]);
|
|
50
61
|
`;
|
|
51
62
|
const args = [q(path), opts?.offset !== undefined ? String(opts.offset) : "", opts?.limit !== undefined ? String(opts.limit) : ""].filter(Boolean).map(q).join(" ");
|
|
52
|
-
const
|
|
53
|
-
|
|
63
|
+
const command = `node -e ${q(script)} ${args}`;
|
|
64
|
+
const result = await sandbox.commands.run(command, { cwd: opts?.cwd, timeoutMs: 30000 });
|
|
65
|
+
if (result.exitCode !== 0)
|
|
66
|
+
throw new Error(formatCommandFailure(command, result));
|
|
67
|
+
return result.stdout;
|
|
54
68
|
}
|
|
55
69
|
async function readRawFile(sandbox, path, cwd) {
|
|
56
70
|
const script = `const fs = require("fs"); process.stdout.write(fs.readFileSync(process.argv[1], "utf8"));`;
|
|
57
|
-
const
|
|
58
|
-
|
|
71
|
+
const command = `node -e ${q(script)} ${q(path)}`;
|
|
72
|
+
const result = await sandbox.commands.run(command, { cwd, timeoutMs: 30000 });
|
|
73
|
+
if (result.exitCode !== 0)
|
|
74
|
+
throw new Error(formatCommandFailure(command, result));
|
|
75
|
+
return result.stdout;
|
|
59
76
|
}
|
|
60
77
|
var readTool = {
|
|
61
78
|
name: "Read",
|
|
@@ -116,8 +133,7 @@ var bashTool = {
|
|
|
116
133
|
timeoutMs: timeoutMs ?? 120000
|
|
117
134
|
});
|
|
118
135
|
if (result.exitCode !== 0)
|
|
119
|
-
throw new Error(
|
|
120
|
-
${result.stdout}` : ""}`);
|
|
136
|
+
throw new Error(formatCommandFailure(command, result));
|
|
121
137
|
return result.stdout;
|
|
122
138
|
}
|
|
123
139
|
};
|
|
@@ -138,6 +154,7 @@ async function runProcessorChain(processors, selectHook, initial, ctx) {
|
|
|
138
154
|
}
|
|
139
155
|
|
|
140
156
|
// src/request-context/request-context.ts
|
|
157
|
+
import { z as z2 } from "zod";
|
|
141
158
|
var AC_RESERVED_PREFIX = "ac__";
|
|
142
159
|
var AC_TEAM_ID = "ac__teamId";
|
|
143
160
|
var AC_RUN_ID = "ac__runId";
|
|
@@ -163,6 +180,17 @@ var SERIALISABLE_RESERVED_KEYS = new Set([
|
|
|
163
180
|
AC_API_KEY_SCOPES,
|
|
164
181
|
AC_PARENT_RUN_ID
|
|
165
182
|
]);
|
|
183
|
+
var RequestContextWireSchema = z2.object({
|
|
184
|
+
reserved: z2.object({
|
|
185
|
+
teamId: z2.string(),
|
|
186
|
+
runId: z2.string(),
|
|
187
|
+
workflowId: z2.string(),
|
|
188
|
+
factoryId: z2.string().nullable(),
|
|
189
|
+
apiKeyScopes: z2.array(z2.string()).readonly(),
|
|
190
|
+
parentRunId: z2.string().nullable()
|
|
191
|
+
}),
|
|
192
|
+
user: z2.record(z2.string(), z2.unknown())
|
|
193
|
+
});
|
|
166
194
|
|
|
167
195
|
class ReservedKeyError extends Error {
|
|
168
196
|
key;
|
|
@@ -224,7 +252,8 @@ class RequestContext {
|
|
|
224
252
|
return new RequestContext(reserved, new Map);
|
|
225
253
|
}
|
|
226
254
|
static deserialise(wire) {
|
|
227
|
-
const
|
|
255
|
+
const parsed = RequestContextWireSchema.parse(wire);
|
|
256
|
+
const ctx = new RequestContext({ ...parsed.reserved, abortSignal: undefined }, new Map(Object.entries(parsed.user)));
|
|
228
257
|
return ctx;
|
|
229
258
|
}
|
|
230
259
|
withAbortSignal(signal) {
|
|
@@ -358,6 +387,8 @@ class VercelRunner {
|
|
|
358
387
|
options;
|
|
359
388
|
config;
|
|
360
389
|
supportsToolCallProcessor = true;
|
|
390
|
+
kind;
|
|
391
|
+
model;
|
|
361
392
|
tools;
|
|
362
393
|
messages = [];
|
|
363
394
|
constructor(sandbox, options, config) {
|
|
@@ -365,6 +396,8 @@ class VercelRunner {
|
|
|
365
396
|
this.options = options;
|
|
366
397
|
this.config = config;
|
|
367
398
|
this.tools = config.tools ?? codingTools;
|
|
399
|
+
this.kind = config.kind;
|
|
400
|
+
this.model = config.modelId;
|
|
368
401
|
}
|
|
369
402
|
async gateToolCall(call, ctx) {
|
|
370
403
|
const verdict = await runProcessorChain(this.options.processors ?? [], (p) => p.processToolCall, call, ctx);
|
|
@@ -372,6 +405,20 @@ class VercelRunner {
|
|
|
372
405
|
return { kind: "allow", call: verdict.value };
|
|
373
406
|
return verdict;
|
|
374
407
|
}
|
|
408
|
+
captureCheckpoint() {
|
|
409
|
+
return { messages: [...this.messages] };
|
|
410
|
+
}
|
|
411
|
+
restoreCheckpoint(blob) {
|
|
412
|
+
if (!blob || typeof blob !== "object") {
|
|
413
|
+
throw new Error("Vercel runtime checkpoint is invalid: expected an object with messages[]");
|
|
414
|
+
}
|
|
415
|
+
const incoming = blob.messages;
|
|
416
|
+
if (!Array.isArray(incoming)) {
|
|
417
|
+
throw new Error("Vercel runtime checkpoint is invalid: expected messages[]");
|
|
418
|
+
}
|
|
419
|
+
this.messages.length = 0;
|
|
420
|
+
this.messages.push(...incoming);
|
|
421
|
+
}
|
|
375
422
|
buildTools(iteration, signal) {
|
|
376
423
|
const set = {};
|
|
377
424
|
for (const t of this.tools) {
|
|
@@ -468,7 +515,12 @@ function createVercelRuntime(config) {
|
|
|
468
515
|
create: (sandbox, opts) => new VercelRunner(sandbox, opts, config)
|
|
469
516
|
});
|
|
470
517
|
}
|
|
518
|
+
async function listVercelRuntimeModels() {
|
|
519
|
+
const { models } = await gateway.getAvailableModels();
|
|
520
|
+
return models.filter((m) => m.modelType == null || m.modelType === "language").map((m) => ({ id: m.id, name: m.name })).sort((a, b) => a.id.localeCompare(b.id));
|
|
521
|
+
}
|
|
471
522
|
export {
|
|
523
|
+
listVercelRuntimeModels,
|
|
472
524
|
createVercelRuntime,
|
|
473
525
|
VercelRunner
|
|
474
526
|
};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/sse.d.ts
CHANGED
|
@@ -7,9 +7,8 @@
|
|
|
7
7
|
* - `event`: event name (from `event:` line; `""` if absent)
|
|
8
8
|
* - `data`: parsed JSON payload (the SDK's stream events are always JSON)
|
|
9
9
|
*
|
|
10
|
-
* Malformed payloads
|
|
11
|
-
*
|
|
12
|
-
* responsible for breaking on terminal events.
|
|
10
|
+
* Malformed payloads throw. A dropped terminal event is worse than a loud
|
|
11
|
+
* protocol error for log consumers.
|
|
13
12
|
*
|
|
14
13
|
* Output type stays loose (`Record<string, unknown>`) on purpose: typed
|
|
15
14
|
* unions like `RunEvent` aren't structurally narrowable from
|
|
@@ -17,9 +17,9 @@
|
|
|
17
17
|
* file + one tokenised stdout result line. Easy to reason about, easy
|
|
18
18
|
* to test in isolation.
|
|
19
19
|
*/
|
|
20
|
-
export { STEP_RESULT_PREFIX, STEP_ENV, RUNNER_BUNDLE_PATH, RUNNER_COMMAND, stepInputPath, requestContextPath, } from "./protocol.js";
|
|
20
|
+
export { STEP_RESULT_PREFIX, STEP_PAUSE_PREFIX, STEP_ENV, RUNNER_BUNDLE_PATH, RUNNER_COMMAND, stepInputPath, requestContextPath, } from "./protocol.js";
|
|
21
21
|
export { invokeStep, parseStepResult, buildStepEnvs } from "./invoker.js";
|
|
22
22
|
export { serveStep } from "./server.js";
|
|
23
23
|
export type { StepHandler, ServeStepRequest, StepHandlerResult } from "./server.js";
|
|
24
24
|
export { StepExecutionError } from "./types.js";
|
|
25
|
-
export type { StepRequest, StepResult, StepInvocationError } from "./types.js";
|
|
25
|
+
export type { StepRequest, StepResult, StepInvocationError, StepPauseRequest } from "./types.js";
|
|
@@ -30,6 +30,7 @@ export declare function buildStepEnvs(args: {
|
|
|
30
30
|
runId: string;
|
|
31
31
|
stepIndex: number;
|
|
32
32
|
resultToken: string;
|
|
33
|
+
isResume?: boolean;
|
|
33
34
|
}): Record<string, string>;
|
|
34
35
|
/** Scan the runner's stdout for the tokenised sentinel line and return a
|
|
35
36
|
* classified result. Returns `null` when no sentinel is present — the
|
|
@@ -51,6 +52,8 @@ export declare function parseStepResult<TOutput = unknown>(stdout: string, resul
|
|
|
51
52
|
*/
|
|
52
53
|
export interface InvokeStepOptions {
|
|
53
54
|
envs?: Record<string, string>;
|
|
55
|
+
/** True when re-entering a step after resolving or expiring a pause. */
|
|
56
|
+
isResume?: boolean;
|
|
54
57
|
/** Live stdout / stderr from the runner subprocess, line-by-line. Called
|
|
55
58
|
* from inside `sandbox.commands.run` as chunks arrive. The sentinel line
|
|
56
59
|
* carrying the protocol result token is filtered out before delivery so
|
|
@@ -12,12 +12,23 @@
|
|
|
12
12
|
* invoker scans for `<prefix><token>:<json>` so any user `console.log`
|
|
13
13
|
* with the same prefix but a wrong token is rejected. */
|
|
14
14
|
export declare const STEP_RESULT_PREFIX = "__AC_STEP_RESULT__";
|
|
15
|
+
/** Parallel sentinel for pause requests. The runner emits this — instead
|
|
16
|
+
* of (not in addition to) a step result — when user code calls
|
|
17
|
+
* `ctx.pause(...)`. Exits cleanly afterwards so the activity can snapshot
|
|
18
|
+
* the now-frozen sandbox and durably wait via Temporal `condition()`.
|
|
19
|
+
* See ADR-0006 §"How it actually pauses — Temporal-native, end-to-end". */
|
|
20
|
+
export declare const STEP_PAUSE_PREFIX = "__AC_STEP_PAUSE__";
|
|
15
21
|
/** Build the line prefix the runner emits and the invoker scans for. The
|
|
16
22
|
* runner-side serveStep writes `<prefix><token>:<json>\n`; the invoker's
|
|
17
23
|
* result parser AND the stdout line splitter both match on this exact
|
|
18
24
|
* prefix so a future change can't make them drift (review found a
|
|
19
25
|
* separator mismatch that leaked the sentinel into captured logs). */
|
|
20
26
|
export declare function stepResultLinePrefix(token: string): string;
|
|
27
|
+
/** Build the pause sentinel line prefix. Same per-invocation token as the
|
|
28
|
+
* result sentinel so the two channels share one secret — a user
|
|
29
|
+
* `console.log` can't forge either without knowing the token, and the
|
|
30
|
+
* invoker can filter both prefixes from captured stdout with one token. */
|
|
31
|
+
export declare function stepPauseLinePrefix(token: string): string;
|
|
21
32
|
/** Sandbox-side path where dispatch writes the compiled runner bundle.
|
|
22
33
|
* Both modes (full-mode `dispatch.ts` and step-mode `invokeStep`) spawn
|
|
23
34
|
* the runner from this path; single source of truth. */
|
|
@@ -34,6 +45,7 @@ export declare const STEP_ENV: {
|
|
|
34
45
|
readonly STEP_INPUT_PATH: "AC_STEP_INPUT_PATH";
|
|
35
46
|
readonly REQUEST_CONTEXT_PATH: "AC_REQUEST_CONTEXT_PATH";
|
|
36
47
|
readonly STEP_RESULT_TOKEN: "AC_STEP_RESULT_TOKEN";
|
|
48
|
+
readonly STEP_RESUME: "AC_STEP_RESUME";
|
|
37
49
|
};
|
|
38
50
|
/** Sandbox-side path where the invoker writes the JSON-encoded step input.
|
|
39
51
|
* The serveStep reads from this exact path; both halves use this helper. */
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
* StepInvocation public types — what callers (the activity) get back, and
|
|
3
3
|
* the discriminated error union that classifies failure modes.
|
|
4
4
|
*/
|
|
5
|
+
import { z } from "zod";
|
|
5
6
|
import type { RequestContextWire } from "../request-context/request-context.js";
|
|
6
7
|
import type { StepObservability } from "../workflow-steps/observability.js";
|
|
7
8
|
/** What the invoker needs to drive one step invocation. The step's
|
|
@@ -16,20 +17,54 @@ export interface StepRequest<TInput = unknown> {
|
|
|
16
17
|
requestContext: RequestContextWire;
|
|
17
18
|
}
|
|
18
19
|
/**
|
|
19
|
-
* Outcome of one step invocation.
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
20
|
+
* Outcome of one step invocation. Three terminal states:
|
|
21
|
+
*
|
|
22
|
+
* - `ok: true` — step body resolved with `output`.
|
|
23
|
+
* - `ok: false` — step body or runner errored; kind classifies why.
|
|
24
|
+
* - `ok: "paused"` — step body called `ctx.pause(...)` and exited cleanly.
|
|
25
|
+
* The activity captures a sandbox snapshot and the workflow waits on
|
|
26
|
+
* Temporal `condition()` until something resumes the pause (HTTP
|
|
27
|
+
* resume route, TTL expiry, cancellation). `pauseRequest` is what the
|
|
28
|
+
* caller passed to `ctx.pause` — the route surfaces it on the
|
|
29
|
+
* pending-pauses dashboard so an operator sees the question being
|
|
30
|
+
* asked. `observability` carries any ctx metadata/sub-step/agent events
|
|
31
|
+
* recorded before the pause unwound. See ADR-0006.
|
|
24
32
|
*/
|
|
25
33
|
export type StepResult<TOutput = unknown> = {
|
|
26
34
|
ok: true;
|
|
27
35
|
output: TOutput;
|
|
28
36
|
observability?: StepObservability;
|
|
37
|
+
} | {
|
|
38
|
+
ok: "paused";
|
|
39
|
+
pauseId: string;
|
|
40
|
+
pauseRequest: StepPauseRequest;
|
|
41
|
+
observability?: StepObservability;
|
|
29
42
|
} | {
|
|
30
43
|
ok: false;
|
|
31
44
|
error: StepInvocationError;
|
|
32
45
|
};
|
|
46
|
+
/** Wire schema for one pause request. Canonical here — `parseStepResult`
|
|
47
|
+
* imports it for validation, and `StepPauseRequest` is `z.infer`'d from
|
|
48
|
+
* it so the runtime type and the parsed shape can't drift.
|
|
49
|
+
*
|
|
50
|
+
* Loose by design (the inner zod shape stops at orchestration fields
|
|
51
|
+
* the engine needs): the SDK wrapper layer (`requestDecision` /
|
|
52
|
+
* `sleep` / `waitForEvent`) owns its own payload contract, and the
|
|
53
|
+
* engine treats `payload` as opaque. */
|
|
54
|
+
export declare const StepPauseRequestSchema: z.ZodObject<{
|
|
55
|
+
reason: z.ZodString;
|
|
56
|
+
kind: z.ZodEnum<{
|
|
57
|
+
custom: "custom";
|
|
58
|
+
decision: "decision";
|
|
59
|
+
sleep: "sleep";
|
|
60
|
+
event: "event";
|
|
61
|
+
}>;
|
|
62
|
+
payload: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
63
|
+
ttlMs: z.ZodOptional<z.ZodNumber>;
|
|
64
|
+
correlationKey: z.ZodOptional<z.ZodString>;
|
|
65
|
+
snapshot: z.ZodOptional<z.ZodBoolean>;
|
|
66
|
+
}, z.core.$strip>;
|
|
67
|
+
export type StepPauseRequest = z.infer<typeof StepPauseRequestSchema>;
|
|
33
68
|
/**
|
|
34
69
|
* Discriminated error union.
|
|
35
70
|
*
|
package/dist/types/events.d.ts
CHANGED
|
@@ -14,6 +14,15 @@ export type RunEvent = {
|
|
|
14
14
|
seq?: number;
|
|
15
15
|
agentId: string;
|
|
16
16
|
label: string;
|
|
17
|
+
/** Tool whitelist passed to `agent({ tools: [...] })`. Drives the
|
|
18
|
+
* per-agent "tools" badge on the dashboard. */
|
|
19
|
+
allowedTools?: string[];
|
|
20
|
+
/** Resolved model id (e.g. `claude-sonnet-4-6`). */
|
|
21
|
+
model?: string;
|
|
22
|
+
/** Short runtime self-identifier (`claude`, `openai-desktop`, …)
|
|
23
|
+
* read off `ModelExecutionContract.kind`. The dashboard maps
|
|
24
|
+
* this to a small runtime icon on the agent card header. */
|
|
25
|
+
runtimeKind?: string;
|
|
17
26
|
} | {
|
|
18
27
|
event: "agent.message";
|
|
19
28
|
runId: string;
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
/** Shared execution context capabilities for workflow functions and steps. */
|
|
2
2
|
import type { InvokeAndWaitOptions, RunStatus } from "../client.js";
|
|
3
3
|
import type { RequestContext } from "../request-context/request-context.js";
|
|
4
|
+
import type { PauseRequest } from "../pause/pause-core.js";
|
|
5
|
+
import type { RequestDecisionRequest, WaitForEventRequest } from "../pause/wrappers.js";
|
|
4
6
|
import type { SandboxProvider } from "./sandbox.js";
|
|
5
7
|
/** The identity of this workflow run. */
|
|
6
8
|
export interface WorkflowRun {
|
|
@@ -19,4 +21,27 @@ export interface BaseExecutionContext {
|
|
|
19
21
|
setMetadata?: (data: Record<string, unknown>) => Promise<void>;
|
|
20
22
|
/** Invoke another registered workflow and wait for it to settle. */
|
|
21
23
|
invokeChild: InvokeChild;
|
|
24
|
+
/**
|
|
25
|
+
* Disk-backed memoise across pause-resume. First call runs `fn` and
|
|
26
|
+
* atomically writes the result to the sandbox; on resume the recorded
|
|
27
|
+
* value is returned and `fn` is NOT re-executed. Use for expensive
|
|
28
|
+
* deterministic transforms; for side effects, use `invokeChild`.
|
|
29
|
+
* See ADR-0006 §"`ctx.checkpoint(name, fn)` — disk-backed memoisation".
|
|
30
|
+
*/
|
|
31
|
+
checkpoint<T>(name: string, fn: () => Promise<T> | T): Promise<T>;
|
|
32
|
+
/**
|
|
33
|
+
* Pause for feedback. The step exits and the workflow waits durably until
|
|
34
|
+
* something resolves the pause (a resume call, a TTL expiry); on resume the
|
|
35
|
+
* step body re-runs from the top and this call returns the resume payload.
|
|
36
|
+
* Provide a `schema` to validate the payload, `ttlMs` + `onExpiry` to bound
|
|
37
|
+
* the wait, `correlationKey` for by-key resume. Throws PauseRequestError /
|
|
38
|
+
* PauseExpiredError / PauseSchemaError. See ADR-0006 / ADR-0011.
|
|
39
|
+
*/
|
|
40
|
+
pause<T = unknown>(req: PauseRequest<T>): Promise<T>;
|
|
41
|
+
/** Pause for a typed decision (a `schema` is required). Wrapper over `pause`. */
|
|
42
|
+
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
|
|
43
|
+
/** Lightweight timed pause — resolves after `durationMs`, no snapshot. */
|
|
44
|
+
sleep(durationMs: number): Promise<void>;
|
|
45
|
+
/** Pause until an event resumes by `correlationKey`. Wrapper over `pause`. */
|
|
46
|
+
waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
|
|
22
47
|
}
|
package/dist/types/protocol.d.ts
CHANGED
|
@@ -52,5 +52,13 @@ export interface AgentStatus {
|
|
|
52
52
|
completed: string[];
|
|
53
53
|
blockers: string[];
|
|
54
54
|
exit_signal: boolean;
|
|
55
|
+
/** PR 7 self-pause (honoured only for `mode: "hitl"` agents). The agent
|
|
56
|
+
* cannot proceed without a human decision: it sets this true, puts the
|
|
57
|
+
* question in `question`, and ends its turn. The loop pauses at the next
|
|
58
|
+
* boundary and injects the human's answer as the next turn. `auto` agents
|
|
59
|
+
* ignore it and keep going. Should be paired with `exit_signal: false`. */
|
|
60
|
+
needs_input?: boolean;
|
|
61
|
+
/** The question to put to the human when `needs_input` is true. */
|
|
62
|
+
question?: string;
|
|
55
63
|
}
|
|
56
64
|
export {};
|
package/dist/types/runtime.d.ts
CHANGED
|
@@ -52,14 +52,69 @@ export type ToolCallGateResult = {
|
|
|
52
52
|
export interface ModelExecutionContract {
|
|
53
53
|
/** True when this runtime can run `processToolCall` before tool execution. */
|
|
54
54
|
supportsToolCallProcessor?: boolean;
|
|
55
|
+
/** Short self-identifier ("claude", "openai-desktop", "vercel", …). Read
|
|
56
|
+
* by the agent loop and surfaced on `agent.spawned` so the dashboard
|
|
57
|
+
* can show a per-agent runtime icon without re-fetching template
|
|
58
|
+
* metadata. Optional — runtimes that omit it stay anonymous. */
|
|
59
|
+
kind?: string;
|
|
60
|
+
/** Resolved model id used by this runtime instance (already merged with
|
|
61
|
+
* config + runtime defaults). Surfaced on `agent.spawned` so each
|
|
62
|
+
* agent card on the Agent tab can label which model it ran against. */
|
|
63
|
+
model?: string;
|
|
55
64
|
/** Runtime-owned pre-tool gate. Adapters call the shared processor chain
|
|
56
65
|
* through this seam; the agent loop stays SDK-agnostic. */
|
|
57
66
|
gateToolCall?(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
|
|
67
|
+
/**
|
|
68
|
+
* Capture runtime-private in-memory state that will not survive the
|
|
69
|
+
* runner subprocess exit. Called by the agent loop at pause time,
|
|
70
|
+
* AFTER the loop has flushed its own state to disk.
|
|
71
|
+
*
|
|
72
|
+
* Return value is opaque to the loop — whatever the runtime needs to
|
|
73
|
+
* round-trip its conversation across pause-resume. Must be JSON-
|
|
74
|
+
* serialisable; the loop atomically writes it to
|
|
75
|
+
* `/tmp/wf/state/runtime-<agentInstanceId>.json` and reads it back
|
|
76
|
+
* on resume to hand to `restoreCheckpoint`.
|
|
77
|
+
*
|
|
78
|
+
* Default (method omitted): runtime holds no instance state that
|
|
79
|
+
* needs to round-trip across pause. The shipped example is the
|
|
80
|
+
* Claude runtime — the conversation lives server-side at
|
|
81
|
+
* Anthropic, addressed by `session_id`, and the loop already holds
|
|
82
|
+
* `lastSessionId` as part of its own state. On resume the loop
|
|
83
|
+
* restores the id, the next `sendMessage` passes it through, and
|
|
84
|
+
* Anthropic resumes the server-side conversation.
|
|
85
|
+
*
|
|
86
|
+
* Runtimes that hold the conversation in-process — Vercel's
|
|
87
|
+
* `VercelRunner.messages` is the canonical case — MUST implement
|
|
88
|
+
* both hooks: the messages array is reconstructed from
|
|
89
|
+
* `response.messages` on each `streamText` and would be lost the
|
|
90
|
+
* moment the subprocess exits. See ADR-0006 §"Concrete examples
|
|
91
|
+
* for shipped runtimes" for the audit + worked examples.
|
|
92
|
+
*/
|
|
93
|
+
captureCheckpoint?(): unknown;
|
|
94
|
+
/**
|
|
95
|
+
* Restore runtime-private state previously returned by
|
|
96
|
+
* `captureCheckpoint`. Called by the agent loop on resume, AFTER
|
|
97
|
+
* the loop has restored its own state but BEFORE iterations resume.
|
|
98
|
+
*
|
|
99
|
+
* `blob` is whatever this same runtime returned at pause time. If
|
|
100
|
+
* `captureCheckpoint` is omitted, this is never called.
|
|
101
|
+
*/
|
|
102
|
+
restoreCheckpoint?(blob: unknown): void;
|
|
58
103
|
sendMessage(opts: {
|
|
59
104
|
prompt: string;
|
|
60
105
|
sessionId?: string;
|
|
61
106
|
iteration?: number;
|
|
62
107
|
signal?: AbortSignal;
|
|
108
|
+
/** Push-iterable of mid-turn user messages from outside the agent
|
|
109
|
+
* loop — e.g. dashboard chat injections. Runtimes that support
|
|
110
|
+
* streaming-input mode (Claude Agent SDK) read from this in
|
|
111
|
+
* parallel with the initial `prompt`; the SDK handles delivery
|
|
112
|
+
* at the next safe boundary. Runtimes without streaming-input
|
|
113
|
+
* support ignore this and fall back to per-iteration injection. */
|
|
114
|
+
inboxStream?: AsyncIterable<{
|
|
115
|
+
text: string;
|
|
116
|
+
senderName?: string | null;
|
|
117
|
+
}>;
|
|
63
118
|
}): AsyncGenerator<AgentMessage>;
|
|
64
119
|
}
|
|
65
120
|
/**
|
package/dist/types/sandbox.d.ts
CHANGED
|
@@ -9,16 +9,16 @@ export interface SandboxCommandRunOptions {
|
|
|
9
9
|
envs?: Record<string, string>;
|
|
10
10
|
onStdout?: (data: string) => void;
|
|
11
11
|
onStderr?: (data: string) => void;
|
|
12
|
-
background?: boolean;
|
|
13
12
|
/** Run the command with root privileges. Vercel maps this to its native
|
|
14
|
-
* `sudo` flag;
|
|
15
|
-
*
|
|
16
|
-
*
|
|
13
|
+
* `sudo` flag; the local provider prepends `sudo`. Defaults to false.
|
|
14
|
+
* Requires the sandbox image to grant the command root (Vercel's runtimes
|
|
15
|
+
* do — passwordless). */
|
|
17
16
|
sudo?: boolean;
|
|
18
17
|
}
|
|
19
18
|
export interface SandboxCommandResult {
|
|
20
19
|
exitCode: number;
|
|
21
20
|
stdout: string;
|
|
21
|
+
stderr: string;
|
|
22
22
|
}
|
|
23
23
|
export interface SandboxProvider {
|
|
24
24
|
sandboxId: string;
|
package/dist/utils/schemas.d.ts
CHANGED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -8,3 +8,5 @@ export type { Step, StepContext, StepRunResult, Workflow, } from "./types.js";
|
|
|
8
8
|
export { WORKFLOW_BRAND } from "./types.js";
|
|
9
9
|
export { StepObservabilityCollector } from "./observability.js";
|
|
10
10
|
export type { StepObservability, SubStepEvent } from "./observability.js";
|
|
11
|
+
export { makeRunCallbackEmitterFromEnv } from "./run-callback.js";
|
|
12
|
+
export type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
@@ -5,18 +5,32 @@
|
|
|
5
5
|
* tokenised stdout sentinel; the activity persists the snapshot to the
|
|
6
6
|
* run's metadata + lifecycle event tables.
|
|
7
7
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
8
|
+
* Live streaming + batch backstop — the no-double-write design:
|
|
9
|
+
*
|
|
10
|
+
* 1. `agentEvents.emit` immediately fires the optional `liveEmitter`
|
|
11
|
+
* (the runner's POST to `/internal/runs/.../events`).
|
|
12
|
+
* 2. The emitter returns `Promise<boolean>` — true means the server
|
|
13
|
+
* accepted and persisted this event, false means it failed (POST
|
|
14
|
+
* error, server 5xx, network blip).
|
|
15
|
+
* 3. The collector tracks which seqs were successfully ack'd.
|
|
16
|
+
* 4. At `snapshot()` time we await any in-flight emits (with a small
|
|
17
|
+
* grace window so the agent loop's final-burst posts can finish),
|
|
18
|
+
* then STRIP ack'd events from the returned `events` array.
|
|
19
|
+
*
|
|
20
|
+
* The result: the snapshot's `events` array only contains events
|
|
21
|
+
* that the live path didn't successfully deliver. The server's
|
|
22
|
+
* batch-flush in `persistStepObservability` becomes a true backstop
|
|
23
|
+
* for the FAILURE path — it never re-writes (and never re-notifies)
|
|
24
|
+
* the events the live route already handled. No double pg_notify,
|
|
25
|
+
* no dashboard duplicates.
|
|
26
|
+
*
|
|
27
|
+
* `metadata` and `subSteps` remain batch-only because they're
|
|
28
|
+
* naturally boundary events (no streaming benefit) and aren't
|
|
29
|
+
* written by the live route at all.
|
|
17
30
|
*/
|
|
18
31
|
import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
|
|
19
32
|
import type { AgentEventSink } from "../types/workflow.js";
|
|
33
|
+
import type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
20
34
|
/** One named sub-step (from `ctx.step("name", async () => ...)`).
|
|
21
35
|
* Becomes a `workflow_substep_*` lifecycle event on the run timeline. */
|
|
22
36
|
export interface SubStepEvent {
|
|
@@ -49,10 +63,28 @@ export declare class StepObservabilityCollector {
|
|
|
49
63
|
private metadata;
|
|
50
64
|
private events;
|
|
51
65
|
private subSteps;
|
|
66
|
+
private readonly liveEmitter?;
|
|
67
|
+
/** Seqs of events the server confirmed via the live route. */
|
|
68
|
+
private readonly ackedSeqs;
|
|
69
|
+
/** Promises for in-flight live emits — awaited at snapshot time. */
|
|
70
|
+
private readonly inFlight;
|
|
71
|
+
constructor(opts?: {
|
|
72
|
+
liveEmitter?: LiveAgentEventEmitter;
|
|
73
|
+
});
|
|
52
74
|
readonly setMetadata: (data: Record<string, unknown>) => Promise<void>;
|
|
53
75
|
readonly step: <T>(name: string, fn: () => Promise<T>) => Promise<T>;
|
|
54
76
|
readonly agentEvents: AgentEventSink;
|
|
77
|
+
/** Wait for in-flight live emits to settle (or timeout) so the
|
|
78
|
+
* ackedSeqs set is maximally up-to-date before we filter. Used by
|
|
79
|
+
* `snapshot()` — exposed separately for tests. */
|
|
80
|
+
private drainInFlight;
|
|
55
81
|
/** Snapshot the accumulated state. Returns `undefined` when nothing
|
|
56
|
-
* was recorded so the wire payload can drop the field entirely.
|
|
57
|
-
|
|
82
|
+
* was recorded so the wire payload can drop the field entirely.
|
|
83
|
+
*
|
|
84
|
+
* Async because we drain in-flight live emits first. Any event the
|
|
85
|
+
* server acknowledged is REMOVED from the returned `events` array
|
|
86
|
+
* so the server-side batch flush doesn't re-write/re-notify it.
|
|
87
|
+
* Events that failed live delivery (POST error, timeout) stay in
|
|
88
|
+
* the array as the durable backstop. */
|
|
89
|
+
snapshot(): Promise<StepObservability | undefined>;
|
|
58
90
|
}
|