@agent-compose/sdk 0.5.0 → 0.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/active-step.d.ts +60 -0
- package/dist/agent/agent-loop-steer.test.d.ts +1 -0
- package/dist/agent/agent-loop.d.ts +46 -0
- package/dist/agent/async-queue.d.ts +29 -0
- package/dist/agent/protocol.d.ts +9 -1
- package/dist/agent/resolve-agent-id.test.d.ts +1 -0
- package/dist/agent/run-agent.d.ts +16 -3
- package/dist/agent/steer-control.d.ts +57 -0
- package/dist/agent/steer-control.test.d.ts +1 -0
- package/dist/client.d.ts +161 -0
- package/dist/index.d.ts +7 -4
- package/dist/index.js +1341 -157
- package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/errors.test.d.ts +1 -0
- package/dist/pause/__tests__/manager.test.d.ts +1 -0
- package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
- package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
- package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
- package/dist/pause/checkpoint.d.ts +28 -0
- package/dist/pause/errors.d.ts +52 -0
- package/dist/pause/manager.d.ts +63 -0
- package/dist/pause/pause-core.d.ts +101 -0
- package/dist/pause/state-dir.d.ts +80 -0
- package/dist/pause/wrappers.d.ts +41 -0
- package/dist/request-context/request-context.d.ts +12 -0
- package/dist/runtimes/claude.d.ts +6 -0
- package/dist/runtimes/openai-desktop.d.ts +2 -0
- package/dist/runtimes/openai-desktop.js +1338 -156
- package/dist/runtimes/vercel.d.ts +12 -0
- package/dist/runtimes/vercel.js +50 -7
- package/dist/runtimes/vercel.test.d.ts +1 -0
- package/dist/sse.d.ts +2 -3
- package/dist/step-invocation/index.d.ts +2 -2
- package/dist/step-invocation/invoker.d.ts +3 -0
- package/dist/step-invocation/protocol.d.ts +12 -0
- package/dist/step-invocation/server.d.ts +1 -0
- package/dist/step-invocation/types.d.ts +40 -5
- package/dist/types/events.d.ts +9 -0
- package/dist/types/execution-context.d.ts +25 -0
- package/dist/types/protocol.d.ts +8 -0
- package/dist/types/runtime.d.ts +55 -0
- package/dist/types/sandbox.d.ts +6 -1
- package/dist/utils/schemas.d.ts +2 -0
- package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
- package/dist/workflow-steps/index.d.ts +2 -0
- package/dist/workflow-steps/observability.d.ts +43 -11
- package/dist/workflow-steps/run-callback.d.ts +39 -0
- package/dist/workflow-steps/runner.d.ts +8 -0
- package/package.json +1 -1
- package/src/active-step.ts +124 -0
- package/src/agent/agent-loop.ts +253 -19
- package/src/agent/async-queue.ts +61 -0
- package/src/agent/protocol.ts +12 -2
- package/src/agent/run-agent.ts +184 -8
- package/src/agent/steer-control.ts +125 -0
- package/src/client.ts +277 -0
- package/src/index.ts +18 -2
- package/src/pause/checkpoint.ts +44 -0
- package/src/pause/errors.ts +70 -0
- package/src/pause/manager.ts +177 -0
- package/src/pause/pause-core.ts +267 -0
- package/src/pause/state-dir.ts +262 -0
- package/src/pause/wrappers.ts +79 -0
- package/src/request-context/request-context.ts +17 -2
- package/src/runtimes/claude.ts +101 -6
- package/src/runtimes/openai-desktop.ts +11 -0
- package/src/runtimes/vercel.ts +26 -0
- package/src/sandbox.ts +45 -17
- package/src/sse.ts +8 -6
- package/src/step-invocation/index.ts +2 -1
- package/src/step-invocation/invoker.ts +107 -29
- package/src/step-invocation/protocol.ts +16 -0
- package/src/step-invocation/server.ts +45 -12
- package/src/step-invocation/types.ts +43 -7
- package/src/tools/coding.ts +16 -5
- package/src/types/events.ts +9 -0
- package/src/types/execution-context.ts +25 -0
- package/src/types/protocol.ts +8 -0
- package/src/types/runtime.ts +52 -0
- package/src/types/sandbox.ts +10 -1
- package/src/types/workflow.ts +6 -1
- package/src/utils/bundler.ts +8 -3
- package/src/utils/schemas.ts +2 -0
- package/src/workflow-steps/index.ts +3 -0
- package/src/workflow-steps/observability.ts +84 -13
- package/src/workflow-steps/run-callback.ts +72 -0
- package/src/workflow-steps/runner.ts +70 -8
- package/dist/utils/discovery.d.ts +0 -2
- package/src/utils/discovery.ts +0 -4
|
@@ -23,6 +23,18 @@ export declare class VercelRunner implements ModelExecutionContract {
|
|
|
23
23
|
private readonly messages;
|
|
24
24
|
constructor(sandbox: SandboxProvider, options: RuntimeOptions, config: VercelRuntimeConfig);
|
|
25
25
|
gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
|
|
26
|
+
/** ADR-0006 pause-resume hooks. The Vercel AI SDK holds the running
|
|
27
|
+
* conversation client-side in `this.messages` — every `streamText`
|
|
28
|
+
* call passes the full array as `messages` and reconstructs it from
|
|
29
|
+
* `response.messages` after completion. A pause-induced subprocess
|
|
30
|
+
* exit loses the array; the new subprocess constructs a fresh
|
|
31
|
+
* `VercelRunner` with `this.messages = []` and the next sendMessage
|
|
32
|
+
* would see only the current iteration's prompt — conversation
|
|
33
|
+
* history broken. The agent loop calls these at every iteration
|
|
34
|
+
* boundary so the messages array round-trips through the sandbox
|
|
35
|
+
* snapshot. */
|
|
36
|
+
captureCheckpoint(): unknown;
|
|
37
|
+
restoreCheckpoint(blob: unknown): void;
|
|
26
38
|
private buildTools;
|
|
27
39
|
sendMessage(opts: {
|
|
28
40
|
prompt: string;
|
package/dist/runtimes/vercel.js
CHANGED
|
@@ -36,6 +36,17 @@ function q(value) {
|
|
|
36
36
|
function resolvePath(path, cwd) {
|
|
37
37
|
return cwd && !isAbsolute(path) ? join(cwd, path) : path;
|
|
38
38
|
}
|
|
39
|
+
function formatCommandFailure(command, result) {
|
|
40
|
+
const parts = [`Command failed with exit code ${result.exitCode}: ${command}`];
|
|
41
|
+
if (result.stderr.trim())
|
|
42
|
+
parts.push(`stderr:
|
|
43
|
+
${result.stderr}`);
|
|
44
|
+
if (result.stdout.trim())
|
|
45
|
+
parts.push(`stdout:
|
|
46
|
+
${result.stdout}`);
|
|
47
|
+
return parts.join(`
|
|
48
|
+
`);
|
|
49
|
+
}
|
|
39
50
|
async function readFile(sandbox, path, opts) {
|
|
40
51
|
const script = `
|
|
41
52
|
const fs = require("fs");
|
|
@@ -49,13 +60,19 @@ const end = Math.min(lines.length, start + Math.max(0, limit) - 1);
|
|
|
49
60
|
for (let i = start; i <= end; i++) console.log(String(i) + ": " + lines[i - 1]);
|
|
50
61
|
`;
|
|
51
62
|
const args = [q(path), opts?.offset !== undefined ? String(opts.offset) : "", opts?.limit !== undefined ? String(opts.limit) : ""].filter(Boolean).map(q).join(" ");
|
|
52
|
-
const
|
|
53
|
-
|
|
63
|
+
const command = `node -e ${q(script)} ${args}`;
|
|
64
|
+
const result = await sandbox.commands.run(command, { cwd: opts?.cwd, timeoutMs: 30000 });
|
|
65
|
+
if (result.exitCode !== 0)
|
|
66
|
+
throw new Error(formatCommandFailure(command, result));
|
|
67
|
+
return result.stdout;
|
|
54
68
|
}
|
|
55
69
|
async function readRawFile(sandbox, path, cwd) {
|
|
56
70
|
const script = `const fs = require("fs"); process.stdout.write(fs.readFileSync(process.argv[1], "utf8"));`;
|
|
57
|
-
const
|
|
58
|
-
|
|
71
|
+
const command = `node -e ${q(script)} ${q(path)}`;
|
|
72
|
+
const result = await sandbox.commands.run(command, { cwd, timeoutMs: 30000 });
|
|
73
|
+
if (result.exitCode !== 0)
|
|
74
|
+
throw new Error(formatCommandFailure(command, result));
|
|
75
|
+
return result.stdout;
|
|
59
76
|
}
|
|
60
77
|
var readTool = {
|
|
61
78
|
name: "Read",
|
|
@@ -116,8 +133,7 @@ var bashTool = {
|
|
|
116
133
|
timeoutMs: timeoutMs ?? 120000
|
|
117
134
|
});
|
|
118
135
|
if (result.exitCode !== 0)
|
|
119
|
-
throw new Error(
|
|
120
|
-
${result.stdout}` : ""}`);
|
|
136
|
+
throw new Error(formatCommandFailure(command, result));
|
|
121
137
|
return result.stdout;
|
|
122
138
|
}
|
|
123
139
|
};
|
|
@@ -138,6 +154,7 @@ async function runProcessorChain(processors, selectHook, initial, ctx) {
|
|
|
138
154
|
}
|
|
139
155
|
|
|
140
156
|
// src/request-context/request-context.ts
|
|
157
|
+
import { z as z2 } from "zod";
|
|
141
158
|
var AC_RESERVED_PREFIX = "ac__";
|
|
142
159
|
var AC_TEAM_ID = "ac__teamId";
|
|
143
160
|
var AC_RUN_ID = "ac__runId";
|
|
@@ -163,6 +180,17 @@ var SERIALISABLE_RESERVED_KEYS = new Set([
|
|
|
163
180
|
AC_API_KEY_SCOPES,
|
|
164
181
|
AC_PARENT_RUN_ID
|
|
165
182
|
]);
|
|
183
|
+
var RequestContextWireSchema = z2.object({
|
|
184
|
+
reserved: z2.object({
|
|
185
|
+
teamId: z2.string(),
|
|
186
|
+
runId: z2.string(),
|
|
187
|
+
workflowId: z2.string(),
|
|
188
|
+
factoryId: z2.string().nullable(),
|
|
189
|
+
apiKeyScopes: z2.array(z2.string()).readonly(),
|
|
190
|
+
parentRunId: z2.string().nullable()
|
|
191
|
+
}),
|
|
192
|
+
user: z2.record(z2.string(), z2.unknown())
|
|
193
|
+
});
|
|
166
194
|
|
|
167
195
|
class ReservedKeyError extends Error {
|
|
168
196
|
key;
|
|
@@ -224,7 +252,8 @@ class RequestContext {
|
|
|
224
252
|
return new RequestContext(reserved, new Map);
|
|
225
253
|
}
|
|
226
254
|
static deserialise(wire) {
|
|
227
|
-
const
|
|
255
|
+
const parsed = RequestContextWireSchema.parse(wire);
|
|
256
|
+
const ctx = new RequestContext({ ...parsed.reserved, abortSignal: undefined }, new Map(Object.entries(parsed.user)));
|
|
228
257
|
return ctx;
|
|
229
258
|
}
|
|
230
259
|
withAbortSignal(signal) {
|
|
@@ -372,6 +401,20 @@ class VercelRunner {
|
|
|
372
401
|
return { kind: "allow", call: verdict.value };
|
|
373
402
|
return verdict;
|
|
374
403
|
}
|
|
404
|
+
captureCheckpoint() {
|
|
405
|
+
return { messages: [...this.messages] };
|
|
406
|
+
}
|
|
407
|
+
restoreCheckpoint(blob) {
|
|
408
|
+
if (!blob || typeof blob !== "object") {
|
|
409
|
+
throw new Error("Vercel runtime checkpoint is invalid: expected an object with messages[]");
|
|
410
|
+
}
|
|
411
|
+
const incoming = blob.messages;
|
|
412
|
+
if (!Array.isArray(incoming)) {
|
|
413
|
+
throw new Error("Vercel runtime checkpoint is invalid: expected messages[]");
|
|
414
|
+
}
|
|
415
|
+
this.messages.length = 0;
|
|
416
|
+
this.messages.push(...incoming);
|
|
417
|
+
}
|
|
375
418
|
buildTools(iteration, signal) {
|
|
376
419
|
const set = {};
|
|
377
420
|
for (const t of this.tools) {
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
package/dist/sse.d.ts
CHANGED
|
@@ -7,9 +7,8 @@
|
|
|
7
7
|
* - `event`: event name (from `event:` line; `""` if absent)
|
|
8
8
|
* - `data`: parsed JSON payload (the SDK's stream events are always JSON)
|
|
9
9
|
*
|
|
10
|
-
* Malformed payloads
|
|
11
|
-
*
|
|
12
|
-
* responsible for breaking on terminal events.
|
|
10
|
+
* Malformed payloads throw. A dropped terminal event is worse than a loud
|
|
11
|
+
* protocol error for log consumers.
|
|
13
12
|
*
|
|
14
13
|
* Output type stays loose (`Record<string, unknown>`) on purpose: typed
|
|
15
14
|
* unions like `RunEvent` aren't structurally narrowable from
|
|
@@ -17,9 +17,9 @@
|
|
|
17
17
|
* file + one tokenised stdout result line. Easy to reason about, easy
|
|
18
18
|
* to test in isolation.
|
|
19
19
|
*/
|
|
20
|
-
export { STEP_RESULT_PREFIX, STEP_ENV, RUNNER_BUNDLE_PATH, RUNNER_COMMAND, stepInputPath, requestContextPath, } from "./protocol.js";
|
|
20
|
+
export { STEP_RESULT_PREFIX, STEP_PAUSE_PREFIX, STEP_ENV, RUNNER_BUNDLE_PATH, RUNNER_COMMAND, stepInputPath, requestContextPath, } from "./protocol.js";
|
|
21
21
|
export { invokeStep, parseStepResult, buildStepEnvs } from "./invoker.js";
|
|
22
22
|
export { serveStep } from "./server.js";
|
|
23
23
|
export type { StepHandler, ServeStepRequest, StepHandlerResult } from "./server.js";
|
|
24
24
|
export { StepExecutionError } from "./types.js";
|
|
25
|
-
export type { StepRequest, StepResult, StepInvocationError } from "./types.js";
|
|
25
|
+
export type { StepRequest, StepResult, StepInvocationError, StepPauseRequest } from "./types.js";
|
|
@@ -30,6 +30,7 @@ export declare function buildStepEnvs(args: {
|
|
|
30
30
|
runId: string;
|
|
31
31
|
stepIndex: number;
|
|
32
32
|
resultToken: string;
|
|
33
|
+
isResume?: boolean;
|
|
33
34
|
}): Record<string, string>;
|
|
34
35
|
/** Scan the runner's stdout for the tokenised sentinel line and return a
|
|
35
36
|
* classified result. Returns `null` when no sentinel is present — the
|
|
@@ -51,6 +52,8 @@ export declare function parseStepResult<TOutput = unknown>(stdout: string, resul
|
|
|
51
52
|
*/
|
|
52
53
|
export interface InvokeStepOptions {
|
|
53
54
|
envs?: Record<string, string>;
|
|
55
|
+
/** True when re-entering a step after resolving or expiring a pause. */
|
|
56
|
+
isResume?: boolean;
|
|
54
57
|
/** Live stdout / stderr from the runner subprocess, line-by-line. Called
|
|
55
58
|
* from inside `sandbox.commands.run` as chunks arrive. The sentinel line
|
|
56
59
|
* carrying the protocol result token is filtered out before delivery so
|
|
@@ -12,12 +12,23 @@
|
|
|
12
12
|
* invoker scans for `<prefix><token>:<json>` so any user `console.log`
|
|
13
13
|
* with the same prefix but a wrong token is rejected. */
|
|
14
14
|
export declare const STEP_RESULT_PREFIX = "__AC_STEP_RESULT__";
|
|
15
|
+
/** Parallel sentinel for pause requests. The runner emits this — instead
|
|
16
|
+
* of (not in addition to) a step result — when user code calls
|
|
17
|
+
* `ctx.pause(...)`. Exits cleanly afterwards so the activity can snapshot
|
|
18
|
+
* the now-frozen sandbox and durably wait via Temporal `condition()`.
|
|
19
|
+
* See ADR-0006 §"How it actually pauses — Temporal-native, end-to-end". */
|
|
20
|
+
export declare const STEP_PAUSE_PREFIX = "__AC_STEP_PAUSE__";
|
|
15
21
|
/** Build the line prefix the runner emits and the invoker scans for. The
|
|
16
22
|
* runner-side serveStep writes `<prefix><token>:<json>\n`; the invoker's
|
|
17
23
|
* result parser AND the stdout line splitter both match on this exact
|
|
18
24
|
* prefix so a future change can't make them drift (review found a
|
|
19
25
|
* separator mismatch that leaked the sentinel into captured logs). */
|
|
20
26
|
export declare function stepResultLinePrefix(token: string): string;
|
|
27
|
+
/** Build the pause sentinel line prefix. Same per-invocation token as the
|
|
28
|
+
* result sentinel so the two channels share one secret — a user
|
|
29
|
+
* `console.log` can't forge either without knowing the token, and the
|
|
30
|
+
* invoker can filter both prefixes from captured stdout with one token. */
|
|
31
|
+
export declare function stepPauseLinePrefix(token: string): string;
|
|
21
32
|
/** Sandbox-side path where dispatch writes the compiled runner bundle.
|
|
22
33
|
* Both modes (full-mode `dispatch.ts` and step-mode `invokeStep`) spawn
|
|
23
34
|
* the runner from this path; single source of truth. */
|
|
@@ -34,6 +45,7 @@ export declare const STEP_ENV: {
|
|
|
34
45
|
readonly STEP_INPUT_PATH: "AC_STEP_INPUT_PATH";
|
|
35
46
|
readonly REQUEST_CONTEXT_PATH: "AC_REQUEST_CONTEXT_PATH";
|
|
36
47
|
readonly STEP_RESULT_TOKEN: "AC_STEP_RESULT_TOKEN";
|
|
48
|
+
readonly STEP_RESUME: "AC_STEP_RESUME";
|
|
37
49
|
};
|
|
38
50
|
/** Sandbox-side path where the invoker writes the JSON-encoded step input.
|
|
39
51
|
* The serveStep reads from this exact path; both halves use this helper. */
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
* StepInvocation public types — what callers (the activity) get back, and
|
|
3
3
|
* the discriminated error union that classifies failure modes.
|
|
4
4
|
*/
|
|
5
|
+
import { z } from "zod";
|
|
5
6
|
import type { RequestContextWire } from "../request-context/request-context.js";
|
|
6
7
|
import type { StepObservability } from "../workflow-steps/observability.js";
|
|
7
8
|
/** What the invoker needs to drive one step invocation. The step's
|
|
@@ -16,20 +17,54 @@ export interface StepRequest<TInput = unknown> {
|
|
|
16
17
|
requestContext: RequestContextWire;
|
|
17
18
|
}
|
|
18
19
|
/**
|
|
19
|
-
* Outcome of one step invocation.
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
20
|
+
* Outcome of one step invocation. Three terminal states:
|
|
21
|
+
*
|
|
22
|
+
* - `ok: true` — step body resolved with `output`.
|
|
23
|
+
* - `ok: false` — step body or runner errored; kind classifies why.
|
|
24
|
+
* - `ok: "paused"` — step body called `ctx.pause(...)` and exited cleanly.
|
|
25
|
+
* The activity captures a sandbox snapshot and the workflow waits on
|
|
26
|
+
* Temporal `condition()` until something resumes the pause (HTTP
|
|
27
|
+
* resume route, TTL expiry, cancellation). `pauseRequest` is what the
|
|
28
|
+
* caller passed to `ctx.pause` — the route surfaces it on the
|
|
29
|
+
* pending-pauses dashboard so an operator sees the question being
|
|
30
|
+
* asked. `observability` carries any ctx metadata/sub-step/agent events
|
|
31
|
+
* recorded before the pause unwound. See ADR-0006.
|
|
24
32
|
*/
|
|
25
33
|
export type StepResult<TOutput = unknown> = {
|
|
26
34
|
ok: true;
|
|
27
35
|
output: TOutput;
|
|
28
36
|
observability?: StepObservability;
|
|
37
|
+
} | {
|
|
38
|
+
ok: "paused";
|
|
39
|
+
pauseId: string;
|
|
40
|
+
pauseRequest: StepPauseRequest;
|
|
41
|
+
observability?: StepObservability;
|
|
29
42
|
} | {
|
|
30
43
|
ok: false;
|
|
31
44
|
error: StepInvocationError;
|
|
32
45
|
};
|
|
46
|
+
/** Wire schema for one pause request. Canonical here — `parseStepResult`
|
|
47
|
+
* imports it for validation, and `StepPauseRequest` is `z.infer`'d from
|
|
48
|
+
* it so the runtime type and the parsed shape can't drift.
|
|
49
|
+
*
|
|
50
|
+
* Loose by design (the inner zod shape stops at orchestration fields
|
|
51
|
+
* the engine needs): the SDK wrapper layer (`requestDecision` /
|
|
52
|
+
* `sleep` / `waitForEvent`) owns its own payload contract, and the
|
|
53
|
+
* engine treats `payload` as opaque. */
|
|
54
|
+
export declare const StepPauseRequestSchema: z.ZodObject<{
|
|
55
|
+
reason: z.ZodString;
|
|
56
|
+
kind: z.ZodEnum<{
|
|
57
|
+
custom: "custom";
|
|
58
|
+
decision: "decision";
|
|
59
|
+
sleep: "sleep";
|
|
60
|
+
event: "event";
|
|
61
|
+
}>;
|
|
62
|
+
payload: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
|
|
63
|
+
ttlMs: z.ZodOptional<z.ZodNumber>;
|
|
64
|
+
correlationKey: z.ZodOptional<z.ZodString>;
|
|
65
|
+
snapshot: z.ZodOptional<z.ZodBoolean>;
|
|
66
|
+
}, z.core.$strip>;
|
|
67
|
+
export type StepPauseRequest = z.infer<typeof StepPauseRequestSchema>;
|
|
33
68
|
/**
|
|
34
69
|
* Discriminated error union.
|
|
35
70
|
*
|
package/dist/types/events.d.ts
CHANGED
|
@@ -14,6 +14,15 @@ export type RunEvent = {
|
|
|
14
14
|
seq?: number;
|
|
15
15
|
agentId: string;
|
|
16
16
|
label: string;
|
|
17
|
+
/** Tool whitelist passed to `agent({ tools: [...] })`. Drives the
|
|
18
|
+
* per-agent "tools" badge on the dashboard. */
|
|
19
|
+
allowedTools?: string[];
|
|
20
|
+
/** Resolved model id (e.g. `claude-sonnet-4-6`). */
|
|
21
|
+
model?: string;
|
|
22
|
+
/** Short runtime self-identifier (`claude`, `openai-desktop`, …)
|
|
23
|
+
* read off `ModelExecutionContract.kind`. The dashboard maps
|
|
24
|
+
* this to a small runtime icon on the agent card header. */
|
|
25
|
+
runtimeKind?: string;
|
|
17
26
|
} | {
|
|
18
27
|
event: "agent.message";
|
|
19
28
|
runId: string;
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
/** Shared execution context capabilities for workflow functions and steps. */
|
|
2
2
|
import type { InvokeAndWaitOptions, RunStatus } from "../client.js";
|
|
3
3
|
import type { RequestContext } from "../request-context/request-context.js";
|
|
4
|
+
import type { PauseRequest } from "../pause/pause-core.js";
|
|
5
|
+
import type { RequestDecisionRequest, WaitForEventRequest } from "../pause/wrappers.js";
|
|
4
6
|
import type { SandboxProvider } from "./sandbox.js";
|
|
5
7
|
/** The identity of this workflow run. */
|
|
6
8
|
export interface WorkflowRun {
|
|
@@ -19,4 +21,27 @@ export interface BaseExecutionContext {
|
|
|
19
21
|
setMetadata?: (data: Record<string, unknown>) => Promise<void>;
|
|
20
22
|
/** Invoke another registered workflow and wait for it to settle. */
|
|
21
23
|
invokeChild: InvokeChild;
|
|
24
|
+
/**
|
|
25
|
+
* Disk-backed memoise across pause-resume. First call runs `fn` and
|
|
26
|
+
* atomically writes the result to the sandbox; on resume the recorded
|
|
27
|
+
* value is returned and `fn` is NOT re-executed. Use for expensive
|
|
28
|
+
* deterministic transforms; for side effects, use `invokeChild`.
|
|
29
|
+
* See ADR-0006 §"`ctx.checkpoint(name, fn)` — disk-backed memoisation".
|
|
30
|
+
*/
|
|
31
|
+
checkpoint<T>(name: string, fn: () => Promise<T> | T): Promise<T>;
|
|
32
|
+
/**
|
|
33
|
+
* Pause for feedback. The step exits and the workflow waits durably until
|
|
34
|
+
* something resolves the pause (a resume call, a TTL expiry); on resume the
|
|
35
|
+
* step body re-runs from the top and this call returns the resume payload.
|
|
36
|
+
* Provide a `schema` to validate the payload, `ttlMs` + `onExpiry` to bound
|
|
37
|
+
* the wait, `correlationKey` for by-key resume. Throws PauseRequestError /
|
|
38
|
+
* PauseExpiredError / PauseSchemaError. See ADR-0006 / ADR-0011.
|
|
39
|
+
*/
|
|
40
|
+
pause<T = unknown>(req: PauseRequest<T>): Promise<T>;
|
|
41
|
+
/** Pause for a typed decision (a `schema` is required). Wrapper over `pause`. */
|
|
42
|
+
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
|
|
43
|
+
/** Lightweight timed pause — resolves after `durationMs`, no snapshot. */
|
|
44
|
+
sleep(durationMs: number): Promise<void>;
|
|
45
|
+
/** Pause until an event resumes by `correlationKey`. Wrapper over `pause`. */
|
|
46
|
+
waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
|
|
22
47
|
}
|
package/dist/types/protocol.d.ts
CHANGED
|
@@ -52,5 +52,13 @@ export interface AgentStatus {
|
|
|
52
52
|
completed: string[];
|
|
53
53
|
blockers: string[];
|
|
54
54
|
exit_signal: boolean;
|
|
55
|
+
/** PR 7 self-pause (honoured only for `mode: "hitl"` agents). The agent
|
|
56
|
+
* cannot proceed without a human decision: it sets this true, puts the
|
|
57
|
+
* question in `question`, and ends its turn. The loop pauses at the next
|
|
58
|
+
* boundary and injects the human's answer as the next turn. `auto` agents
|
|
59
|
+
* ignore it and keep going. Should be paired with `exit_signal: false`. */
|
|
60
|
+
needs_input?: boolean;
|
|
61
|
+
/** The question to put to the human when `needs_input` is true. */
|
|
62
|
+
question?: string;
|
|
55
63
|
}
|
|
56
64
|
export {};
|
package/dist/types/runtime.d.ts
CHANGED
|
@@ -52,14 +52,69 @@ export type ToolCallGateResult = {
|
|
|
52
52
|
export interface ModelExecutionContract {
|
|
53
53
|
/** True when this runtime can run `processToolCall` before tool execution. */
|
|
54
54
|
supportsToolCallProcessor?: boolean;
|
|
55
|
+
/** Short self-identifier ("claude", "openai-desktop", "vercel", …). Read
|
|
56
|
+
* by the agent loop and surfaced on `agent.spawned` so the dashboard
|
|
57
|
+
* can show a per-agent runtime icon without re-fetching template
|
|
58
|
+
* metadata. Optional — runtimes that omit it stay anonymous. */
|
|
59
|
+
kind?: string;
|
|
60
|
+
/** Resolved model id used by this runtime instance (already merged with
|
|
61
|
+
* config + runtime defaults). Surfaced on `agent.spawned` so each
|
|
62
|
+
* agent card on the Agent tab can label which model it ran against. */
|
|
63
|
+
model?: string;
|
|
55
64
|
/** Runtime-owned pre-tool gate. Adapters call the shared processor chain
|
|
56
65
|
* through this seam; the agent loop stays SDK-agnostic. */
|
|
57
66
|
gateToolCall?(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
|
|
67
|
+
/**
|
|
68
|
+
* Capture runtime-private in-memory state that will not survive the
|
|
69
|
+
* runner subprocess exit. Called by the agent loop at pause time,
|
|
70
|
+
* AFTER the loop has flushed its own state to disk.
|
|
71
|
+
*
|
|
72
|
+
* Return value is opaque to the loop — whatever the runtime needs to
|
|
73
|
+
* round-trip its conversation across pause-resume. Must be JSON-
|
|
74
|
+
* serialisable; the loop atomically writes it to
|
|
75
|
+
* `/tmp/wf/state/runtime-<agentInstanceId>.json` and reads it back
|
|
76
|
+
* on resume to hand to `restoreCheckpoint`.
|
|
77
|
+
*
|
|
78
|
+
* Default (method omitted): runtime holds no instance state that
|
|
79
|
+
* needs to round-trip across pause. The shipped example is the
|
|
80
|
+
* Claude runtime — the conversation lives server-side at
|
|
81
|
+
* Anthropic, addressed by `session_id`, and the loop already holds
|
|
82
|
+
* `lastSessionId` as part of its own state. On resume the loop
|
|
83
|
+
* restores the id, the next `sendMessage` passes it through, and
|
|
84
|
+
* Anthropic resumes the server-side conversation.
|
|
85
|
+
*
|
|
86
|
+
* Runtimes that hold the conversation in-process — Vercel's
|
|
87
|
+
* `VercelRunner.messages` is the canonical case — MUST implement
|
|
88
|
+
* both hooks: the messages array is reconstructed from
|
|
89
|
+
* `response.messages` on each `streamText` and would be lost the
|
|
90
|
+
* moment the subprocess exits. See ADR-0006 §"Concrete examples
|
|
91
|
+
* for shipped runtimes" for the audit + worked examples.
|
|
92
|
+
*/
|
|
93
|
+
captureCheckpoint?(): unknown;
|
|
94
|
+
/**
|
|
95
|
+
* Restore runtime-private state previously returned by
|
|
96
|
+
* `captureCheckpoint`. Called by the agent loop on resume, AFTER
|
|
97
|
+
* the loop has restored its own state but BEFORE iterations resume.
|
|
98
|
+
*
|
|
99
|
+
* `blob` is whatever this same runtime returned at pause time. If
|
|
100
|
+
* `captureCheckpoint` is omitted, this is never called.
|
|
101
|
+
*/
|
|
102
|
+
restoreCheckpoint?(blob: unknown): void;
|
|
58
103
|
sendMessage(opts: {
|
|
59
104
|
prompt: string;
|
|
60
105
|
sessionId?: string;
|
|
61
106
|
iteration?: number;
|
|
62
107
|
signal?: AbortSignal;
|
|
108
|
+
/** Push-iterable of mid-turn user messages from outside the agent
|
|
109
|
+
* loop — e.g. dashboard chat injections. Runtimes that support
|
|
110
|
+
* streaming-input mode (Claude Agent SDK) read from this in
|
|
111
|
+
* parallel with the initial `prompt`; the SDK handles delivery
|
|
112
|
+
* at the next safe boundary. Runtimes without streaming-input
|
|
113
|
+
* support ignore this and fall back to per-iteration injection. */
|
|
114
|
+
inboxStream?: AsyncIterable<{
|
|
115
|
+
text: string;
|
|
116
|
+
senderName?: string | null;
|
|
117
|
+
}>;
|
|
63
118
|
}): AsyncGenerator<AgentMessage>;
|
|
64
119
|
}
|
|
65
120
|
/**
|
package/dist/types/sandbox.d.ts
CHANGED
|
@@ -9,11 +9,16 @@ export interface SandboxCommandRunOptions {
|
|
|
9
9
|
envs?: Record<string, string>;
|
|
10
10
|
onStdout?: (data: string) => void;
|
|
11
11
|
onStderr?: (data: string) => void;
|
|
12
|
-
|
|
12
|
+
/** Run the command with root privileges. Vercel maps this to its native
|
|
13
|
+
* `sudo` flag; the local provider prepends `sudo`. Defaults to false.
|
|
14
|
+
* Requires the sandbox image to grant the command root (Vercel's runtimes
|
|
15
|
+
* do — passwordless). */
|
|
16
|
+
sudo?: boolean;
|
|
13
17
|
}
|
|
14
18
|
export interface SandboxCommandResult {
|
|
15
19
|
exitCode: number;
|
|
16
20
|
stdout: string;
|
|
21
|
+
stderr: string;
|
|
17
22
|
}
|
|
18
23
|
export interface SandboxProvider {
|
|
19
24
|
sandboxId: string;
|
package/dist/utils/schemas.d.ts
CHANGED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -8,3 +8,5 @@ export type { Step, StepContext, StepRunResult, Workflow, } from "./types.js";
|
|
|
8
8
|
export { WORKFLOW_BRAND } from "./types.js";
|
|
9
9
|
export { StepObservabilityCollector } from "./observability.js";
|
|
10
10
|
export type { StepObservability, SubStepEvent } from "./observability.js";
|
|
11
|
+
export { makeRunCallbackEmitterFromEnv } from "./run-callback.js";
|
|
12
|
+
export type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
@@ -5,18 +5,32 @@
|
|
|
5
5
|
* tokenised stdout sentinel; the activity persists the snapshot to the
|
|
6
6
|
* run's metadata + lifecycle event tables.
|
|
7
7
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
8
|
+
* Live streaming + batch backstop — the no-double-write design:
|
|
9
|
+
*
|
|
10
|
+
* 1. `agentEvents.emit` immediately fires the optional `liveEmitter`
|
|
11
|
+
* (the runner's POST to `/internal/runs/.../events`).
|
|
12
|
+
* 2. The emitter returns `Promise<boolean>` — true means the server
|
|
13
|
+
* accepted and persisted this event, false means it failed (POST
|
|
14
|
+
* error, server 5xx, network blip).
|
|
15
|
+
* 3. The collector tracks which seqs were successfully ack'd.
|
|
16
|
+
* 4. At `snapshot()` time we await any in-flight emits (with a small
|
|
17
|
+
* grace window so the agent loop's final-burst posts can finish),
|
|
18
|
+
* then STRIP ack'd events from the returned `events` array.
|
|
19
|
+
*
|
|
20
|
+
* The result: the snapshot's `events` array only contains events
|
|
21
|
+
* that the live path didn't successfully deliver. The server's
|
|
22
|
+
* batch-flush in `persistStepObservability` becomes a true backstop
|
|
23
|
+
* for the FAILURE path — it never re-writes (and never re-notifies)
|
|
24
|
+
* the events the live route already handled. No double pg_notify,
|
|
25
|
+
* no dashboard duplicates.
|
|
26
|
+
*
|
|
27
|
+
* `metadata` and `subSteps` remain batch-only because they're
|
|
28
|
+
* naturally boundary events (no streaming benefit) and aren't
|
|
29
|
+
* written by the live route at all.
|
|
17
30
|
*/
|
|
18
31
|
import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
|
|
19
32
|
import type { AgentEventSink } from "../types/workflow.js";
|
|
33
|
+
import type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
20
34
|
/** One named sub-step (from `ctx.step("name", async () => ...)`).
|
|
21
35
|
* Becomes a `workflow_substep_*` lifecycle event on the run timeline. */
|
|
22
36
|
export interface SubStepEvent {
|
|
@@ -49,10 +63,28 @@ export declare class StepObservabilityCollector {
|
|
|
49
63
|
private metadata;
|
|
50
64
|
private events;
|
|
51
65
|
private subSteps;
|
|
66
|
+
private readonly liveEmitter?;
|
|
67
|
+
/** Seqs of events the server confirmed via the live route. */
|
|
68
|
+
private readonly ackedSeqs;
|
|
69
|
+
/** Promises for in-flight live emits — awaited at snapshot time. */
|
|
70
|
+
private readonly inFlight;
|
|
71
|
+
constructor(opts?: {
|
|
72
|
+
liveEmitter?: LiveAgentEventEmitter;
|
|
73
|
+
});
|
|
52
74
|
readonly setMetadata: (data: Record<string, unknown>) => Promise<void>;
|
|
53
75
|
readonly step: <T>(name: string, fn: () => Promise<T>) => Promise<T>;
|
|
54
76
|
readonly agentEvents: AgentEventSink;
|
|
77
|
+
/** Wait for in-flight live emits to settle (or timeout) so the
|
|
78
|
+
* ackedSeqs set is maximally up-to-date before we filter. Used by
|
|
79
|
+
* `snapshot()` — exposed separately for tests. */
|
|
80
|
+
private drainInFlight;
|
|
55
81
|
/** Snapshot the accumulated state. Returns `undefined` when nothing
|
|
56
|
-
* was recorded so the wire payload can drop the field entirely.
|
|
57
|
-
|
|
82
|
+
* was recorded so the wire payload can drop the field entirely.
|
|
83
|
+
*
|
|
84
|
+
* Async because we drain in-flight live emits first. Any event the
|
|
85
|
+
* server acknowledged is REMOVED from the returned `events` array
|
|
86
|
+
* so the server-side batch flush doesn't re-write/re-notify it.
|
|
87
|
+
* Events that failed live delivery (POST error, timeout) stay in
|
|
88
|
+
* the array as the durable backstop. */
|
|
89
|
+
snapshot(): Promise<StepObservability | undefined>;
|
|
58
90
|
}
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Runner → server live event emitter.
|
|
3
|
+
*
|
|
4
|
+
* Pairs with the server-side `/api/v1/internal/runs/:runId/steps/:stepIndex/events`
|
|
5
|
+
* route. The runner reads `AGENT_COMPOSE_URL`, `AGENT_COMPOSE_RUN_TOKEN`,
|
|
6
|
+
* and `RUN_ID` from env; if all three are present it builds a POST
|
|
7
|
+
* function that sends each agent lifecycle event to the server as
|
|
8
|
+
* soon as the agent loop emits it.
|
|
9
|
+
*
|
|
10
|
+
* Non-blocking: the agent loop doesn't await the emitter — the runtime
|
|
11
|
+
* keeps emitting events at full speed. But the emitter still returns a
|
|
12
|
+
* `Promise<boolean>` so the collector can later decide whether to
|
|
13
|
+
* include each event in the batch-end durable backstop. `true` = server
|
|
14
|
+
* accepted the event (don't re-deliver in batch); `false` = POST
|
|
15
|
+
* failed (DO re-deliver). The collector awaits these promises with a
|
|
16
|
+
* short grace window at snapshot time.
|
|
17
|
+
*
|
|
18
|
+
* This is the seam that lets us avoid the double-write problem: every
|
|
19
|
+
* successfully-ack'd live event is stripped from `observability.events`
|
|
20
|
+
* before the batch flush runs, so the durable backstop only carries
|
|
21
|
+
* events that genuinely failed live delivery. No more two paths writing
|
|
22
|
+
* the same row + double pg_notify.
|
|
23
|
+
*/
|
|
24
|
+
import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
|
|
25
|
+
export type LiveAgentEventEmitter = (event: AgentLifecycleEvent, seq: number) => Promise<boolean>;
|
|
26
|
+
/**
|
|
27
|
+
* Build a live emitter from process env + the caller-supplied stepIndex.
|
|
28
|
+
* Returns `undefined` when any required env var is missing (local tests,
|
|
29
|
+
* non-sandbox callers) so the caller can wire `undefined` straight
|
|
30
|
+
* through to the collector and get batch-only delivery without
|
|
31
|
+
* conditional plumbing.
|
|
32
|
+
*
|
|
33
|
+
* `stepIndex` is a parameter rather than an env read because
|
|
34
|
+
* `serveStep` scrubs `AC_STEP_INDEX` from `process.env` before invoking
|
|
35
|
+
* the workflow handler (to keep the transport envelope unreachable
|
|
36
|
+
* from user code); the handler still holds the parsed integer and
|
|
37
|
+
* passes it here.
|
|
38
|
+
*/
|
|
39
|
+
export declare function makeRunCallbackEmitterFromEnv(stepIndex: number): LiveAgentEventEmitter | undefined;
|
|
@@ -27,6 +27,7 @@ import type { RequestContext } from "../request-context/request-context.js";
|
|
|
27
27
|
import type { SandboxProvider } from "../types/sandbox.js";
|
|
28
28
|
import type { WorkflowRun, WorkflowCtx } from "../types/workflow.js";
|
|
29
29
|
import { type StepObservability } from "./observability.js";
|
|
30
|
+
import type { LiveAgentEventEmitter } from "./run-callback.js";
|
|
30
31
|
export declare class StepValidationError extends Error {
|
|
31
32
|
readonly stepName: string;
|
|
32
33
|
readonly side: "input" | "output";
|
|
@@ -67,6 +68,11 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
|
|
|
67
68
|
onStepStarted?(stepIndex: number, stepName: string): void | Promise<void>;
|
|
68
69
|
/** Child workflow invocation implementation. Defaults to a clear unsupported error. */
|
|
69
70
|
invokeChild?: WorkflowCtx["invokeChild"];
|
|
71
|
+
/** Optional live-stream emitter for agent lifecycle events. The runner
|
|
72
|
+
* passes a fetch-based emitter wired to the per-run callback token so
|
|
73
|
+
* the dashboard sees events as the agent loop produces them; tests
|
|
74
|
+
* leave it undefined and get batch-only delivery. */
|
|
75
|
+
liveAgentEventEmitter?: LiveAgentEventEmitter;
|
|
70
76
|
}
|
|
71
77
|
export interface RunWorkflowStepsResult<TOutput> {
|
|
72
78
|
output: TOutput;
|
|
@@ -84,6 +90,8 @@ export interface RunWorkflowSingleStepOpts {
|
|
|
84
90
|
sandbox?: SandboxProvider;
|
|
85
91
|
abortSignal?: AbortSignal;
|
|
86
92
|
invokeChild?: WorkflowCtx["invokeChild"];
|
|
93
|
+
/** Optional live-stream emitter — see `RunWorkflowStepsOpts.liveAgentEventEmitter`. */
|
|
94
|
+
liveAgentEventEmitter?: LiveAgentEventEmitter;
|
|
87
95
|
}
|
|
88
96
|
/** Result of one step run — output plus whatever the step's observability
|
|
89
97
|
* hooks recorded. `observability` is undefined when nothing was buffered,
|
package/package.json
CHANGED