@agent-compose/sdk 0.5.1 → 0.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/active-step.d.ts +60 -0
- package/dist/agent/agent-loop-steer.test.d.ts +1 -0
- package/dist/agent/agent-loop.d.ts +46 -0
- package/dist/agent/async-queue.d.ts +29 -0
- package/dist/agent/protocol.d.ts +9 -1
- package/dist/agent/resolve-agent-id.test.d.ts +1 -0
- package/dist/agent/run-agent.d.ts +16 -3
- package/dist/agent/steer-control.d.ts +57 -0
- package/dist/agent/steer-control.test.d.ts +1 -0
- package/dist/client.d.ts +161 -0
- package/dist/index.d.ts +7 -4
- package/dist/index.js +1339 -157
- package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
- package/dist/pause/__tests__/errors.test.d.ts +1 -0
- package/dist/pause/__tests__/manager.test.d.ts +1 -0
- package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
- package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
- package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
- package/dist/pause/checkpoint.d.ts +28 -0
- package/dist/pause/errors.d.ts +52 -0
- package/dist/pause/manager.d.ts +63 -0
- package/dist/pause/pause-core.d.ts +101 -0
- package/dist/pause/state-dir.d.ts +80 -0
- package/dist/pause/wrappers.d.ts +41 -0
- package/dist/request-context/request-context.d.ts +12 -0
- package/dist/runtimes/claude.d.ts +6 -0
- package/dist/runtimes/openai-desktop.d.ts +2 -0
- package/dist/runtimes/openai-desktop.js +1336 -156
- package/dist/runtimes/vercel.d.ts +12 -0
- package/dist/runtimes/vercel.js +50 -7
- package/dist/runtimes/vercel.test.d.ts +1 -0
- package/dist/sse.d.ts +2 -3
- package/dist/step-invocation/index.d.ts +2 -2
- package/dist/step-invocation/invoker.d.ts +3 -0
- package/dist/step-invocation/protocol.d.ts +12 -0
- package/dist/step-invocation/server.d.ts +1 -0
- package/dist/step-invocation/types.d.ts +40 -5
- package/dist/types/events.d.ts +9 -0
- package/dist/types/execution-context.d.ts +25 -0
- package/dist/types/protocol.d.ts +8 -0
- package/dist/types/runtime.d.ts +55 -0
- package/dist/types/sandbox.d.ts +4 -4
- package/dist/utils/schemas.d.ts +2 -0
- package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
- package/dist/workflow-steps/index.d.ts +2 -0
- package/dist/workflow-steps/observability.d.ts +43 -11
- package/dist/workflow-steps/run-callback.d.ts +39 -0
- package/dist/workflow-steps/runner.d.ts +8 -0
- package/package.json +1 -1
- package/src/active-step.ts +124 -0
- package/src/agent/agent-loop.ts +253 -19
- package/src/agent/async-queue.ts +61 -0
- package/src/agent/protocol.ts +12 -2
- package/src/agent/run-agent.ts +184 -8
- package/src/agent/steer-control.ts +125 -0
- package/src/client.ts +277 -0
- package/src/index.ts +18 -2
- package/src/pause/checkpoint.ts +44 -0
- package/src/pause/errors.ts +70 -0
- package/src/pause/manager.ts +177 -0
- package/src/pause/pause-core.ts +267 -0
- package/src/pause/state-dir.ts +262 -0
- package/src/pause/wrappers.ts +79 -0
- package/src/request-context/request-context.ts +17 -2
- package/src/runtimes/claude.ts +101 -6
- package/src/runtimes/openai-desktop.ts +11 -0
- package/src/runtimes/vercel.ts +26 -0
- package/src/sandbox.ts +39 -20
- package/src/sse.ts +8 -6
- package/src/step-invocation/index.ts +2 -1
- package/src/step-invocation/invoker.ts +107 -29
- package/src/step-invocation/protocol.ts +16 -0
- package/src/step-invocation/server.ts +45 -12
- package/src/step-invocation/types.ts +43 -7
- package/src/tools/coding.ts +16 -5
- package/src/types/events.ts +9 -0
- package/src/types/execution-context.ts +25 -0
- package/src/types/protocol.ts +8 -0
- package/src/types/runtime.ts +52 -0
- package/src/types/sandbox.ts +8 -4
- package/src/types/workflow.ts +6 -1
- package/src/utils/bundler.ts +8 -3
- package/src/utils/schemas.ts +2 -0
- package/src/workflow-steps/index.ts +3 -0
- package/src/workflow-steps/observability.ts +84 -13
- package/src/workflow-steps/run-callback.ts +72 -0
- package/src/workflow-steps/runner.ts +70 -8
- package/dist/utils/discovery.d.ts +0 -2
- package/src/utils/discovery.ts +0 -4
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pause state directory — durable per-agent / per-checkpoint / per-pause
|
|
3
|
+
* blobs the sandbox snapshot carries across pause-resume.
|
|
4
|
+
*
|
|
5
|
+
* Layout under `/tmp/wf/state/`:
|
|
6
|
+
*
|
|
7
|
+
* agent-<agentInstanceId>.json ← agent loop state (iteration, messages, processor cursor)
|
|
8
|
+
* runtime-<agentInstanceId>.json ← runtime-private blob (opaque to the loop)
|
|
9
|
+
* checkpoints/<name>.json ← user `ctx.checkpoint(name, fn)` memoised values
|
|
10
|
+
* pauses/<pauseId>.json ← pause records (request + resume payload once available)
|
|
11
|
+
*
|
|
12
|
+
* Atomic writes (write-tmp → fsync → rename) so a mid-write `sandbox.snapshot()`
|
|
13
|
+
* cannot capture partial state. Most read helpers return `null` on missing-file
|
|
14
|
+
* rather than throwing — callers branch on presence to distinguish "first
|
|
15
|
+
* run" from "post-resume restore". User checkpoints use an explicit
|
|
16
|
+
* found/value result because `null` is a valid memoised value.
|
|
17
|
+
*
|
|
18
|
+
* The state-dir layout is treated as wire protocol between the SDK
|
|
19
|
+
* runner (writer) and the SDK on the next subprocess invocation (reader).
|
|
20
|
+
* Changing a filename or shape is a breaking change to pause resumption
|
|
21
|
+
* for any in-flight paused workflow. See ADR-0006.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { promises as fs } from "node:fs";
|
|
25
|
+
import { join } from "node:path";
|
|
26
|
+
import { randomBytes } from "node:crypto";
|
|
27
|
+
|
|
28
|
+
/** Root of the pause state-dir. Defaults to `/tmp/wf/state` — the
|
|
29
|
+
* canonical location inside the sandbox. Tests and non-`/tmp`-writable
|
|
30
|
+
* environments can override via `AGENT_COMPOSE_STATE_DIR`.
|
|
31
|
+
*
|
|
32
|
+
* Read at CALL TIME, not module load. An earlier draft cached it on
|
|
33
|
+
* first import, but that turned out to be test-ordering-fragile: once
|
|
34
|
+
* any test imported state-dir (directly or transitively via
|
|
35
|
+
* agent-loop) the path got locked to whatever env was set at that
|
|
36
|
+
* instant, and a later test setting `AGENT_COMPOSE_STATE_DIR` would
|
|
37
|
+
* silently write to the cached path instead. Reading the env each
|
|
38
|
+
* call costs nothing measurable next to the fsync. */
|
|
39
|
+
export function getStateDir(): string {
|
|
40
|
+
return process.env.AGENT_COMPOSE_STATE_DIR ?? "/tmp/wf/state";
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
const checkpointsDir = () => join(getStateDir(), "checkpoints");
|
|
44
|
+
const pausesDir = () => join(getStateDir(), "pauses");
|
|
45
|
+
|
|
46
|
+
const agentStatePath = (agentInstanceId: string) => join(getStateDir(), `agent-${agentInstanceId}.json`);
|
|
47
|
+
const runtimeStatePath = (agentInstanceId: string) => join(getStateDir(), `runtime-${agentInstanceId}.json`);
|
|
48
|
+
const checkpointPath = (name: string) => join(checkpointsDir(), `${name}.json`);
|
|
49
|
+
const pausePath = (pauseId: string) => join(pausesDir(), `${pauseId}.json`);
|
|
50
|
+
|
|
51
|
+
/** `name` is interpolated into a filename; reject anything outside a
|
|
52
|
+
* conservative slug. Prevents a workflow author from writing
|
|
53
|
+
* `ctx.checkpoint("../../etc/passwd", fn)` and escaping the dir.
|
|
54
|
+
*
|
|
55
|
+
* First character must be alphanumeric or `_` so dotfile names (`.foo`)
|
|
56
|
+
* and parent-directory tokens (`.`, `..`) are rejected regardless of
|
|
57
|
+
* any future change to the `.json` suffix invariant. Empty strings are
|
|
58
|
+
* rejected because the first character is required.
|
|
59
|
+
*
|
|
60
|
+
* Allowed: `agent-1`, `expensive_load`, `p.42`, `decision-v2`.
|
|
61
|
+
* Rejected: ``, `.`, `..`, `.hidden`, `-leading`, `a/b`, `a\b`. */
|
|
62
|
+
const NAME_RE = /^[a-zA-Z0-9_][a-zA-Z0-9_.-]*$/;
|
|
63
|
+
|
|
64
|
+
function assertSafeName(label: string, name: string): void {
|
|
65
|
+
if (!NAME_RE.test(name)) {
|
|
66
|
+
throw new Error(`${label} name must match ${NAME_RE} (got ${JSON.stringify(name)})`);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
export function assertSafeCheckpointName(name: string): void {
|
|
71
|
+
assertSafeName("checkpoint", name);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Ensure the state-dir and its subdirectories exist. Idempotent —
|
|
75
|
+
* re-running is a no-op. Called lazily by every write entry-point so
|
|
76
|
+
* callers don't need to remember to call it. */
|
|
77
|
+
async function ensureDirs(): Promise<void> {
|
|
78
|
+
await fs.mkdir(checkpointsDir(), { recursive: true });
|
|
79
|
+
await fs.mkdir(pausesDir(), { recursive: true });
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/** Atomic write: temp file in the same directory, fsync to flush
|
|
83
|
+
* contents to disk, then rename onto the target. POSIX rename is
|
|
84
|
+
* atomic when source and target are on the same filesystem (always
|
|
85
|
+
* true here — same dir), so a snapshot in progress sees either the
|
|
86
|
+
* pre-write file or the fully-written file, never a partial. */
|
|
87
|
+
async function atomicWriteJSON(path: string, value: unknown): Promise<void> {
|
|
88
|
+
await ensureDirs();
|
|
89
|
+
const tmp = `${path}.tmp.${randomBytes(6).toString("hex")}`;
|
|
90
|
+
const payload = JSON.stringify(value);
|
|
91
|
+
if (payload === undefined) {
|
|
92
|
+
throw new Error("pause state values must be JSON-serialisable");
|
|
93
|
+
}
|
|
94
|
+
const handle = await fs.open(tmp, "w");
|
|
95
|
+
try {
|
|
96
|
+
await handle.writeFile(payload, { encoding: "utf8" });
|
|
97
|
+
await handle.sync();
|
|
98
|
+
} finally {
|
|
99
|
+
await handle.close();
|
|
100
|
+
}
|
|
101
|
+
await fs.rename(tmp, path);
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/** Read + JSON-parse. Returns `null` when the file doesn't exist —
|
|
105
|
+
* the absence IS the signal "no state yet".
|
|
106
|
+
* Any other error (permission, JSON syntax, partial write) bubbles
|
|
107
|
+
* so the runner fails loudly rather than silently re-running. */
|
|
108
|
+
async function readJSON<T = unknown>(path: string): Promise<T | null> {
|
|
109
|
+
try {
|
|
110
|
+
const raw = await fs.readFile(path, "utf8");
|
|
111
|
+
return JSON.parse(raw) as T;
|
|
112
|
+
} catch (err) {
|
|
113
|
+
if ((err as NodeJS.ErrnoException).code === "ENOENT") return null;
|
|
114
|
+
throw err;
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export type CheckpointRead<T> =
|
|
119
|
+
| { found: true; value: T }
|
|
120
|
+
| { found: false };
|
|
121
|
+
|
|
122
|
+
async function readJSONWithPresence<T = unknown>(path: string): Promise<CheckpointRead<T>> {
|
|
123
|
+
try {
|
|
124
|
+
const raw = await fs.readFile(path, "utf8");
|
|
125
|
+
return { found: true, value: JSON.parse(raw) as T };
|
|
126
|
+
} catch (err) {
|
|
127
|
+
if ((err as NodeJS.ErrnoException).code === "ENOENT") return { found: false };
|
|
128
|
+
throw err;
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
// ── Agent loop state ────────────────────────────────────────────────────────
|
|
133
|
+
//
|
|
134
|
+
// `agentInstanceId` flows from workflow-author code (`agent({ agentId })`)
|
|
135
|
+
// and from the step-mode derivation in `run-agent.ts`. Both are
|
|
136
|
+
// validated here so a malformed id can't escape the state-dir even
|
|
137
|
+
// through the agent / runtime entry points.
|
|
138
|
+
|
|
139
|
+
export async function readAgentState<T = unknown>(agentInstanceId: string): Promise<T | null> {
|
|
140
|
+
assertSafeName("agent", agentInstanceId);
|
|
141
|
+
return readJSON<T>(agentStatePath(agentInstanceId));
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
export async function writeAgentState(agentInstanceId: string, state: unknown): Promise<void> {
|
|
145
|
+
assertSafeName("agent", agentInstanceId);
|
|
146
|
+
await atomicWriteJSON(agentStatePath(agentInstanceId), state);
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
// ── Runtime-private blob ────────────────────────────────────────────────────
|
|
150
|
+
|
|
151
|
+
export async function readRuntimeCheckpoint<T = unknown>(agentInstanceId: string): Promise<T | null> {
|
|
152
|
+
assertSafeName("runtime", agentInstanceId);
|
|
153
|
+
return readJSON<T>(runtimeStatePath(agentInstanceId));
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
export async function writeRuntimeCheckpoint(agentInstanceId: string, blob: unknown): Promise<void> {
|
|
157
|
+
assertSafeName("runtime", agentInstanceId);
|
|
158
|
+
await atomicWriteJSON(runtimeStatePath(agentInstanceId), blob);
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
// ── User checkpoints ────────────────────────────────────────────────────────
|
|
162
|
+
|
|
163
|
+
export async function readCheckpoint<T = unknown>(name: string): Promise<CheckpointRead<T>> {
|
|
164
|
+
assertSafeCheckpointName(name);
|
|
165
|
+
return readJSONWithPresence<T>(checkpointPath(name));
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
export async function writeCheckpoint(name: string, value: unknown): Promise<void> {
|
|
169
|
+
assertSafeCheckpointName(name);
|
|
170
|
+
await atomicWriteJSON(checkpointPath(name), value);
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
// ── Pause records ───────────────────────────────────────────────────────────
|
|
174
|
+
|
|
175
|
+
/** The `pauses/<pauseId>.json` file — the only durable handoff between the
|
|
176
|
+
* paused subprocess and the resumed one.
|
|
177
|
+
*
|
|
178
|
+
* Written twice in two execution contexts:
|
|
179
|
+
* - At pause time the runner writes the placeholder `{ pauseId, request }`
|
|
180
|
+
* (no `resolved`). The snapshot captures it.
|
|
181
|
+
* - At resume time the engine activity writes the resolution — usually
|
|
182
|
+
* merged on top of the placeholder, but snapshotless pauses may only have
|
|
183
|
+
* `{ resolved: true, resumePayload, expired }` in a fresh sandbox. Those
|
|
184
|
+
* three resolution fields are always present together so the byte shape
|
|
185
|
+
* distinguishes a real resume (`expired:false`) from a TTL expiry
|
|
186
|
+
* (`expired:true`) from an unresolved placeholder (`resolved` absent).
|
|
187
|
+
*
|
|
188
|
+
* `corePause` reads it on re-entry: `resolved !== true` ⇒ still my own
|
|
189
|
+
* placeholder, pause again; `expired` ⇒ apply `onExpiry`; else return
|
|
190
|
+
* `resumePayload`. */
|
|
191
|
+
export interface PauseRecord {
|
|
192
|
+
pauseId: string;
|
|
193
|
+
request: unknown;
|
|
194
|
+
/** Set true once the engine has written the resolution (resume or expiry). */
|
|
195
|
+
resolved?: boolean;
|
|
196
|
+
/** Value to return from `ctx.pause` on a real resume; null for expiry. */
|
|
197
|
+
resumePayload?: unknown;
|
|
198
|
+
/** True when the resolution is a TTL expiry, not a caller-driven resume. */
|
|
199
|
+
expired?: boolean;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
function hasOwn(obj: object, key: string): boolean {
|
|
203
|
+
return Object.prototype.hasOwnProperty.call(obj, key);
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
function validatePauseRecord(value: unknown, expectedPauseId: string): PauseRecord {
|
|
207
|
+
if (value === null || typeof value !== "object" || Array.isArray(value)) {
|
|
208
|
+
throw new Error(`pause record ${expectedPauseId} is malformed: expected object`);
|
|
209
|
+
}
|
|
210
|
+
const record = value as Record<string, unknown>;
|
|
211
|
+
if (record.pauseId !== undefined && record.pauseId !== expectedPauseId) {
|
|
212
|
+
throw new Error(`pause record ${expectedPauseId} is malformed: pauseId mismatch`);
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
if (record.resolved === true) {
|
|
216
|
+
if (typeof record.expired !== "boolean") {
|
|
217
|
+
throw new Error(`pause record ${expectedPauseId} is malformed: resolved records require boolean expired`);
|
|
218
|
+
}
|
|
219
|
+
if (!hasOwn(record, "resumePayload") || record.resumePayload === undefined) {
|
|
220
|
+
throw new Error(`pause record ${expectedPauseId} is malformed: resolved records require resumePayload`);
|
|
221
|
+
}
|
|
222
|
+
return {
|
|
223
|
+
pauseId: expectedPauseId,
|
|
224
|
+
request: hasOwn(record, "request") ? record.request : null,
|
|
225
|
+
...record,
|
|
226
|
+
} as unknown as PauseRecord;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
if (record.pauseId !== expectedPauseId) {
|
|
230
|
+
throw new Error(`pause record ${expectedPauseId} is malformed: unresolved records require matching pauseId`);
|
|
231
|
+
}
|
|
232
|
+
if (!hasOwn(record, "request") || record.request === undefined) {
|
|
233
|
+
throw new Error(`pause record ${expectedPauseId} is malformed: unresolved records require request`);
|
|
234
|
+
}
|
|
235
|
+
if (record.resolved !== undefined && record.resolved !== false) {
|
|
236
|
+
throw new Error(`pause record ${expectedPauseId} is malformed: resolved must be true, false, or absent`);
|
|
237
|
+
}
|
|
238
|
+
if (hasOwn(record, "expired") || hasOwn(record, "resumePayload")) {
|
|
239
|
+
throw new Error(`pause record ${expectedPauseId} is malformed: unresolved records must not include resolution fields`);
|
|
240
|
+
}
|
|
241
|
+
return record as unknown as PauseRecord;
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
export async function readPauseRecord(pauseId: string): Promise<PauseRecord | null> {
|
|
245
|
+
assertSafeName("pause", pauseId);
|
|
246
|
+
const record = await readJSON<unknown>(pausePath(pauseId));
|
|
247
|
+
return record === null ? null : validatePauseRecord(record, pauseId);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
export async function writePauseRecord(record: PauseRecord): Promise<void> {
|
|
251
|
+
assertSafeName("pause", record.pauseId);
|
|
252
|
+
validatePauseRecord(record, record.pauseId);
|
|
253
|
+
await atomicWriteJSON(pausePath(record.pauseId), record);
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
/** Wipe the entire state directory. Used by the runner at first-boot
|
|
257
|
+
* (no pauseId in env) so a freshly-dispatched run never inherits state
|
|
258
|
+
* from a prior dispatch that shared the sandbox snapshot template.
|
|
259
|
+
* No-op when the directory doesn't exist. */
|
|
260
|
+
export async function resetStateDir(): Promise<void> {
|
|
261
|
+
await fs.rm(getStateDir(), { recursive: true, force: true });
|
|
262
|
+
}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The three opinionated pause wrappers (ADR-0006 §"SDK surface"), each a thin
|
|
3
|
+
* closure over `ctx.pause`:
|
|
4
|
+
*
|
|
5
|
+
* - `requestDecision` — pause for a typed human/agent decision (schema required).
|
|
6
|
+
* - `sleep` — a lightweight timed pause; resolves on its own TTL,
|
|
7
|
+
* skips the snapshot, returns void.
|
|
8
|
+
* - `waitForEvent` — pause until an event resumes by correlation key.
|
|
9
|
+
*
|
|
10
|
+
* Built from a `PauseFn` so the wrapper logic lives in one place and the step
|
|
11
|
+
* runner just spreads them onto the context next to `pause`.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import type { z } from "zod";
|
|
15
|
+
import type { StepPauseRequest } from "../step-invocation/types.js";
|
|
16
|
+
import type { PauseRequest } from "./pause-core.js";
|
|
17
|
+
|
|
18
|
+
/** The public `ctx.pause` callable (always `kind: "custom"`). */
|
|
19
|
+
export type PauseFn = <T = unknown>(req: PauseRequest<T>) => Promise<T>;
|
|
20
|
+
|
|
21
|
+
/** Internal: pause with an explicit wire `kind`. The wrappers stamp
|
|
22
|
+
* `decision`/`sleep`/`event` through this; the public `ctx.pause` is always
|
|
23
|
+
* `custom` and never exposes it. */
|
|
24
|
+
export type KindedPauseFn = <T = unknown>(req: PauseRequest<T>, kind: StepPauseRequest["kind"]) => Promise<T>;
|
|
25
|
+
|
|
26
|
+
export interface RequestDecisionRequest<T> {
|
|
27
|
+
reason: string;
|
|
28
|
+
payload?: Record<string, unknown>;
|
|
29
|
+
/** Required — a decision is always validated against a shape. */
|
|
30
|
+
schema: z.ZodType<T>;
|
|
31
|
+
ttlMs?: number;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export interface WaitForEventRequest<T> {
|
|
35
|
+
reason: string;
|
|
36
|
+
/** Required — the by-key resume route targets this. */
|
|
37
|
+
correlationKey: string;
|
|
38
|
+
schema?: z.ZodType<T>;
|
|
39
|
+
ttlMs?: number;
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export interface PauseWrappers {
|
|
43
|
+
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
|
|
44
|
+
sleep(durationMs: number): Promise<void>;
|
|
45
|
+
waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function buildPauseWrappers(pause: KindedPauseFn): PauseWrappers {
|
|
49
|
+
return {
|
|
50
|
+
requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T> {
|
|
51
|
+
return pause<T>({
|
|
52
|
+
reason: req.reason,
|
|
53
|
+
schema: req.schema,
|
|
54
|
+
...(req.payload !== undefined ? { payload: req.payload } : {}),
|
|
55
|
+
...(req.ttlMs !== undefined ? { ttlMs: req.ttlMs } : {}),
|
|
56
|
+
}, "decision");
|
|
57
|
+
},
|
|
58
|
+
|
|
59
|
+
// Resolves on its OWN ttl — there is no external resumer for a sleep, so
|
|
60
|
+
// onExpiry resolves (not throws) with void. snapshot:false keeps it cheap.
|
|
61
|
+
sleep(durationMs: number): Promise<void> {
|
|
62
|
+
return pause<void>({
|
|
63
|
+
reason: `sleep ${durationMs}ms`,
|
|
64
|
+
ttlMs: durationMs,
|
|
65
|
+
snapshot: false,
|
|
66
|
+
onExpiry: { mode: "resolve", value: undefined },
|
|
67
|
+
}, "sleep");
|
|
68
|
+
},
|
|
69
|
+
|
|
70
|
+
waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T> {
|
|
71
|
+
return pause<T>({
|
|
72
|
+
reason: req.reason,
|
|
73
|
+
correlationKey: req.correlationKey,
|
|
74
|
+
...(req.schema !== undefined ? { schema: req.schema } : {}),
|
|
75
|
+
...(req.ttlMs !== undefined ? { ttlMs: req.ttlMs } : {}),
|
|
76
|
+
}, "event");
|
|
77
|
+
},
|
|
78
|
+
};
|
|
79
|
+
}
|
|
@@ -39,6 +39,8 @@
|
|
|
39
39
|
* is per-run by design (cross-run state must be explicit via `input`).
|
|
40
40
|
*/
|
|
41
41
|
|
|
42
|
+
import { z } from "zod";
|
|
43
|
+
|
|
42
44
|
/** Reserved key namespace prefix — anything starting with this is platform-owned. */
|
|
43
45
|
export const AC_RESERVED_PREFIX = "ac__";
|
|
44
46
|
|
|
@@ -99,6 +101,18 @@ export interface RequestContextWire {
|
|
|
99
101
|
user: Record<string, unknown>;
|
|
100
102
|
}
|
|
101
103
|
|
|
104
|
+
export const RequestContextWireSchema = z.object({
|
|
105
|
+
reserved: z.object({
|
|
106
|
+
teamId: z.string(),
|
|
107
|
+
runId: z.string(),
|
|
108
|
+
workflowId: z.string(),
|
|
109
|
+
factoryId: z.string().nullable(),
|
|
110
|
+
apiKeyScopes: z.array(z.string()).readonly(),
|
|
111
|
+
parentRunId: z.string().nullable(),
|
|
112
|
+
}),
|
|
113
|
+
user: z.record(z.string(), z.unknown()),
|
|
114
|
+
}) satisfies z.ZodType<RequestContextWire>;
|
|
115
|
+
|
|
102
116
|
/** Thrown when a caller tries to `set` or `delete` a reserved key downstream. */
|
|
103
117
|
export class ReservedKeyError extends Error {
|
|
104
118
|
readonly key: string;
|
|
@@ -185,9 +199,10 @@ export class RequestContext<U extends Record<string, unknown> = Record<string, u
|
|
|
185
199
|
* its own `AbortSignal` separately via `withAbortSignal()`.
|
|
186
200
|
*/
|
|
187
201
|
static deserialise(wire: RequestContextWire): RequestContext {
|
|
202
|
+
const parsed = RequestContextWireSchema.parse(wire);
|
|
188
203
|
const ctx = new RequestContext(
|
|
189
|
-
{ ...
|
|
190
|
-
new Map(Object.entries(
|
|
204
|
+
{ ...parsed.reserved, abortSignal: undefined },
|
|
205
|
+
new Map(Object.entries(parsed.user)),
|
|
191
206
|
);
|
|
192
207
|
return ctx;
|
|
193
208
|
}
|
package/src/runtimes/claude.ts
CHANGED
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
/** Claude Agent SDK runtime — replaces the old Claude CLI subprocess runtime. */
|
|
2
2
|
|
|
3
|
-
import { query, type HookCallback, type PreToolUseHookInput, type ThinkingConfig } from "@anthropic-ai/claude-agent-sdk";
|
|
3
|
+
import { query, type HookCallback, type PreToolUseHookInput, type SDKUserMessage, type ThinkingConfig } from "@anthropic-ai/claude-agent-sdk";
|
|
4
4
|
import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";
|
|
5
5
|
import { defineRuntime } from "../types/runtime.js";
|
|
6
6
|
import { DEFAULT_CLAUDE_MODEL } from "../agent/agent-loop.js";
|
|
7
|
+
import { AsyncQueue } from "../agent/async-queue.js";
|
|
7
8
|
import { runProcessorChain } from "../processors/runner.js";
|
|
8
9
|
import type { ProcessorContext, ToolCall } from "../processors/processor.js";
|
|
9
10
|
import { RequestContext } from "../request-context/request-context.js";
|
|
@@ -26,6 +27,12 @@ function translateMessage(message: Record<string, unknown>): AgentMessage[] {
|
|
|
26
27
|
const b = block as Record<string, unknown>;
|
|
27
28
|
if (b.type === "text") msgs.push({ type: "text", text: String(b.text ?? ""), timestamp: ts });
|
|
28
29
|
if (b.type === "thinking") msgs.push({ type: "thinking", text: String(b.thinking ?? ""), timestamp: ts });
|
|
30
|
+
// `redacted_thinking` is the encrypted-thinking variant emitted by
|
|
31
|
+
// reasoning models that don't expose chain-of-thought (the content
|
|
32
|
+
// is an opaque encrypted blob, not human-readable text). Surface
|
|
33
|
+
// it as a placeholder so the Agent tab's thinking badge isn't
|
|
34
|
+
// visually empty for those models.
|
|
35
|
+
if (b.type === "redacted_thinking") msgs.push({ type: "thinking", text: "(redacted)", timestamp: ts });
|
|
29
36
|
if (b.type === "tool_use") msgs.push({ type: "tool_use", toolName: String(b.name ?? ""), toolInput: (b.input ?? {}) as Record<string, unknown>, toolUseId: String(b.id ?? ""), timestamp: ts });
|
|
30
37
|
}
|
|
31
38
|
return msgs;
|
|
@@ -82,24 +89,63 @@ export interface ClaudeRuntimeConfig {
|
|
|
82
89
|
effort?: "low" | "medium" | "high" | "xhigh" | "max";
|
|
83
90
|
}
|
|
84
91
|
|
|
92
|
+
/** Canonical install path for Claude Code inside a Vercel sandbox.
|
|
93
|
+
* Our `agent-env` setup copies the native binary here as a real file
|
|
94
|
+
* (not a symlink) so it survives snapshotting. The Anthropic installer
|
|
95
|
+
* drops a symlink at `~/.local/bin/claude` pointing into user-home,
|
|
96
|
+
* which Vercel sandbox snapshots don't preserve reliably — pinning
|
|
97
|
+
* `/usr/local/bin/claude` instead bypasses both PATH-priority issues
|
|
98
|
+
* (`~/.local/bin` comes earlier than `/usr/local/bin`) and the
|
|
99
|
+
* broken-symlink-after-restore problem. Callers can override via
|
|
100
|
+
* `pathToClaudeCodeExecutable` or `CLAUDE_CODE_EXECUTABLE`. */
|
|
101
|
+
const DEFAULT_CLAUDE_PATH = "/usr/local/bin/claude";
|
|
102
|
+
|
|
85
103
|
export class ClaudeRunner implements ModelExecutionContract {
|
|
86
104
|
supportsToolCallProcessor = true;
|
|
105
|
+
readonly kind = "claude";
|
|
106
|
+
|
|
107
|
+
// ADR-0006 pause-resume hooks intentionally omitted.
|
|
108
|
+
//
|
|
109
|
+
// The Claude Agent SDK keeps the conversation SERVER-SIDE at
|
|
110
|
+
// Anthropic, addressed by `session_id`. Each `sendMessage` call
|
|
111
|
+
// passes `resume: sessionId` and Anthropic rehydrates the prior
|
|
112
|
+
// transcript; `ClaudeRunner` holds no instance state across calls.
|
|
113
|
+
//
|
|
114
|
+
// `sessionId` itself lives in the agent loop (see
|
|
115
|
+
// agent-loop.ts :: `lastSessionId`) and is part of the loop's own
|
|
116
|
+
// state file — restored automatically on pause-resume. So the loop
|
|
117
|
+
// gets everything it needs without any per-runtime blob; defining
|
|
118
|
+
// `captureCheckpoint`/`restoreCheckpoint` here would just be empty.
|
|
119
|
+
//
|
|
120
|
+
// Caveat: Anthropic's server-side session retention is bounded by
|
|
121
|
+
// their TTL (not ours). For workflow pauses inside that window
|
|
122
|
+
// (default workflowRunTimeout is 7 days) the resume succeeds; a
|
|
123
|
+
// 30-day pause may get session-not-found from `resume:` and would
|
|
124
|
+
// need a fallback to client-managed messages. Documented limitation.
|
|
87
125
|
|
|
88
126
|
constructor(
|
|
89
|
-
// Claude Agent SDK executes in this process. In dispatched workflows this
|
|
90
|
-
// process is already the runner sandbox, so there is no remote provider hop.
|
|
91
127
|
_sandbox: SandboxProvider,
|
|
92
128
|
private readonly options: RuntimeOptions = {},
|
|
93
129
|
private readonly config: ClaudeRuntimeConfig = {},
|
|
94
130
|
) {}
|
|
95
131
|
|
|
132
|
+
get model(): string {
|
|
133
|
+
return this.config.model ?? this.options.model ?? DEFAULT_CLAUDE_MODEL;
|
|
134
|
+
}
|
|
135
|
+
|
|
96
136
|
async gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult> {
|
|
97
137
|
const verdict = await runProcessorChain(this.options.processors ?? [], p => p.processToolCall, call, ctx);
|
|
98
138
|
if (verdict.kind === "continue") return { kind: "allow", call: verdict.value };
|
|
99
139
|
return verdict;
|
|
100
140
|
}
|
|
101
141
|
|
|
102
|
-
async *sendMessage(opts: {
|
|
142
|
+
async *sendMessage(opts: {
|
|
143
|
+
prompt: string;
|
|
144
|
+
sessionId?: string;
|
|
145
|
+
iteration?: number;
|
|
146
|
+
signal?: AbortSignal;
|
|
147
|
+
inboxStream?: AsyncIterable<{ text: string; senderName?: string | null }>;
|
|
148
|
+
}): AsyncGenerator<AgentMessage> {
|
|
103
149
|
const requestContext = this.options.requestContext ?? RequestContext.fromReserved({
|
|
104
150
|
teamId: "", runId: "", workflowId: "",
|
|
105
151
|
factoryId: null, apiKeyScopes: [], parentRunId: null,
|
|
@@ -124,10 +170,48 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
124
170
|
return { continue: false, systemMessage: gate.reason };
|
|
125
171
|
};
|
|
126
172
|
|
|
173
|
+
// Streaming-input mode (Claude Agent SDK):
|
|
174
|
+
// When the caller passes `inboxStream`, we build a pushable
|
|
175
|
+
// AsyncIterable<SDKUserMessage> queue. The initial prompt is
|
|
176
|
+
// pushed as the first user turn; inbox messages arriving while
|
|
177
|
+
// the assistant is mid-turn are pushed as additional user turns
|
|
178
|
+
// and the SDK delivers them at the next safe boundary.
|
|
179
|
+
//
|
|
180
|
+
// Without an inboxStream we keep the simple string-prompt path —
|
|
181
|
+
// no queue, same behaviour as before. This keeps non-interactive
|
|
182
|
+
// workflows (memory extractor, etc.) on the original code path.
|
|
183
|
+
//
|
|
184
|
+
// The queue is closed when the SDK's `result` message indicates
|
|
185
|
+
// the iteration's assistant turns are done; without close() the
|
|
186
|
+
// iterable would block forever waiting for more user input.
|
|
187
|
+
const inboxQueue = opts.inboxStream ? new AsyncQueue<SDKUserMessage>() : null;
|
|
188
|
+
if (inboxQueue) {
|
|
189
|
+
inboxQueue.push({
|
|
190
|
+
type: "user",
|
|
191
|
+
message: { role: "user", content: opts.prompt },
|
|
192
|
+
parent_tool_use_id: null,
|
|
193
|
+
});
|
|
194
|
+
// Drain the inbox iterable into the queue. Runs in parallel
|
|
195
|
+
// with the assistant; pushes user turns as they arrive.
|
|
196
|
+
void (async () => {
|
|
197
|
+
try {
|
|
198
|
+
for await (const ev of opts.inboxStream!) {
|
|
199
|
+
const tag = ev.senderName ? `[message from ${ev.senderName}]\n` : "[user message]\n";
|
|
200
|
+
inboxQueue.push({
|
|
201
|
+
type: "user",
|
|
202
|
+
message: { role: "user", content: tag + ev.text },
|
|
203
|
+
parent_tool_use_id: null,
|
|
204
|
+
});
|
|
205
|
+
}
|
|
206
|
+
} catch { /* iterable ended or aborted */ }
|
|
207
|
+
})();
|
|
208
|
+
}
|
|
209
|
+
const promptForQuery: string | AsyncIterable<SDKUserMessage> = inboxQueue ?? opts.prompt;
|
|
210
|
+
|
|
127
211
|
try {
|
|
128
212
|
let emittedAssistantText = false;
|
|
129
213
|
for await (const message of query({
|
|
130
|
-
prompt:
|
|
214
|
+
prompt: promptForQuery,
|
|
131
215
|
options: {
|
|
132
216
|
tools: this.options.allowedTools,
|
|
133
217
|
allowedTools: this.options.allowedTools,
|
|
@@ -138,7 +222,9 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
138
222
|
effort: this.config.effort,
|
|
139
223
|
cwd: this.options.cwd,
|
|
140
224
|
env: { ...process.env, ...(this.config.env ?? {}) },
|
|
141
|
-
pathToClaudeCodeExecutable: this.config.pathToClaudeCodeExecutable
|
|
225
|
+
pathToClaudeCodeExecutable: this.config.pathToClaudeCodeExecutable
|
|
226
|
+
?? process.env.CLAUDE_CODE_EXECUTABLE
|
|
227
|
+
?? DEFAULT_CLAUDE_PATH,
|
|
142
228
|
...(this.config.claudeMdContent ? { systemPrompt: { type: "preset" as const, preset: "claude_code" as const, append: this.config.claudeMdContent } } : {}),
|
|
143
229
|
resume: opts.sessionId,
|
|
144
230
|
mcpServers: this.config.mcpServers,
|
|
@@ -158,9 +244,18 @@ export class ClaudeRunner implements ModelExecutionContract {
|
|
|
158
244
|
if (msg.type === "text" && raw.type === "assistant") emittedAssistantText = true;
|
|
159
245
|
yield msg;
|
|
160
246
|
}
|
|
247
|
+
// End-of-iteration signal: SDK emits a `result` message after the
|
|
248
|
+
// assistant's turn completes. Close the queue so the AsyncIterable
|
|
249
|
+
// drains and query() returns. Without this, the iterable would
|
|
250
|
+
// block forever waiting for the next user turn that never comes.
|
|
251
|
+
if (raw.type === "result" && inboxQueue) inboxQueue.close();
|
|
161
252
|
}
|
|
162
253
|
} catch (err) {
|
|
163
254
|
yield { type: "error", text: formatError(err), timestamp: now() };
|
|
255
|
+
} finally {
|
|
256
|
+
// Belt-and-braces — close on error/abort too so the dangling
|
|
257
|
+
// iterable doesn't leak the inboxStream consumer.
|
|
258
|
+
inboxQueue?.close();
|
|
164
259
|
}
|
|
165
260
|
}
|
|
166
261
|
}
|
|
@@ -21,9 +21,20 @@ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
|
|
|
21
21
|
}
|
|
22
22
|
|
|
23
23
|
export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
24
|
+
readonly kind = "openai-desktop";
|
|
25
|
+
readonly model = "computer-use-preview";
|
|
24
26
|
private openai: OpenAI;
|
|
25
27
|
private label: string;
|
|
26
28
|
|
|
29
|
+
// ADR-0006 pause-resume hooks intentionally omitted.
|
|
30
|
+
//
|
|
31
|
+
// The computer-use-preview model treats every `sendMessage` as a
|
|
32
|
+
// fresh conversation seeded with the current desktop screenshot —
|
|
33
|
+
// the `messages` array and `previousResponseId` are scoped to ONE
|
|
34
|
+
// call, never carried across. The desktop screenshot IS the state,
|
|
35
|
+
// and the sandbox snapshot already captures that on the filesystem.
|
|
36
|
+
// No runtime-private blob needed.
|
|
37
|
+
|
|
27
38
|
constructor(
|
|
28
39
|
private sandbox: DesktopSandboxProvider,
|
|
29
40
|
opts: OpenAIDesktopRunnerOptions,
|
package/src/runtimes/vercel.ts
CHANGED
|
@@ -88,6 +88,32 @@ export class VercelRunner implements ModelExecutionContract {
|
|
|
88
88
|
return verdict;
|
|
89
89
|
}
|
|
90
90
|
|
|
91
|
+
/** ADR-0006 pause-resume hooks. The Vercel AI SDK holds the running
|
|
92
|
+
* conversation client-side in `this.messages` — every `streamText`
|
|
93
|
+
* call passes the full array as `messages` and reconstructs it from
|
|
94
|
+
* `response.messages` after completion. A pause-induced subprocess
|
|
95
|
+
* exit loses the array; the new subprocess constructs a fresh
|
|
96
|
+
* `VercelRunner` with `this.messages = []` and the next sendMessage
|
|
97
|
+
* would see only the current iteration's prompt — conversation
|
|
98
|
+
* history broken. The agent loop calls these at every iteration
|
|
99
|
+
* boundary so the messages array round-trips through the sandbox
|
|
100
|
+
* snapshot. */
|
|
101
|
+
captureCheckpoint(): unknown {
|
|
102
|
+
return { messages: [...this.messages] };
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
restoreCheckpoint(blob: unknown): void {
|
|
106
|
+
if (!blob || typeof blob !== "object") {
|
|
107
|
+
throw new Error("Vercel runtime checkpoint is invalid: expected an object with messages[]");
|
|
108
|
+
}
|
|
109
|
+
const incoming = (blob as { messages?: unknown }).messages;
|
|
110
|
+
if (!Array.isArray(incoming)) {
|
|
111
|
+
throw new Error("Vercel runtime checkpoint is invalid: expected messages[]");
|
|
112
|
+
}
|
|
113
|
+
this.messages.length = 0;
|
|
114
|
+
this.messages.push(...incoming);
|
|
115
|
+
}
|
|
116
|
+
|
|
91
117
|
private buildTools(iteration: number, signal?: AbortSignal): AiToolSet {
|
|
92
118
|
const set: AiToolSet = {};
|
|
93
119
|
for (const t of this.tools) {
|