@agent-compose/sdk 0.5.8 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/agent-context.d.ts +1 -1
- package/dist/agent/agent-loop.d.ts +14 -12
- package/dist/agent/pause-client.d.ts +50 -0
- package/dist/agent/pause-client.test.d.ts +1 -0
- package/dist/agent/steer-control.d.ts +22 -6
- package/dist/client.d.ts +12 -1
- package/dist/index.d.ts +7 -5
- package/dist/index.js +2379 -1463
- package/dist/pause/checkpoint.d.ts +27 -10
- package/dist/pause/manager.d.ts +1 -0
- package/dist/pause/pause-core.d.ts +23 -0
- package/dist/pause/state-dir.d.ts +1 -1
- package/dist/processors/builtins.d.ts +20 -1
- package/dist/processors/gate-pause.d.ts +46 -0
- package/dist/processors/gate-pause.test.d.ts +1 -0
- package/dist/processors/index.d.ts +3 -1
- package/dist/processors/processor.d.ts +13 -0
- package/dist/runtimes/_acp-client.d.ts +140 -0
- package/dist/runtimes/_cli-agent.d.ts +155 -3
- package/dist/runtimes/amp.d.ts +2 -2
- package/dist/runtimes/cli-agent-acp-live.test.d.ts +30 -0
- package/dist/runtimes/cli-agent.test.d.ts +22 -6
- package/dist/runtimes/codex.d.ts +7 -2
- package/dist/runtimes/openai-desktop.js +2365 -1463
- package/dist/runtimes/vercel.js +389 -2
- package/dist/sandbox.d.ts +113 -19
- package/dist/step-invocation/__tests__/background-invoker.test.d.ts +1 -0
- package/dist/step-invocation/index.d.ts +2 -1
- package/dist/step-invocation/invoker.d.ts +36 -0
- package/dist/types/__tests__/environment-build-flag.test.d.ts +1 -0
- package/dist/types/__tests__/workflow-metadata-provider.test.d.ts +1 -0
- package/dist/types/execution-context.d.ts +0 -8
- package/dist/types/protocol.d.ts +32 -1
- package/dist/types/runtime.d.ts +14 -0
- package/dist/types/sandbox-environment.d.ts +6 -1
- package/dist/types/sandbox.d.ts +86 -6
- package/dist/types/workflow-metadata.d.ts +40 -10
- package/dist/types/workflow.d.ts +22 -6
- package/dist/utils/bundler.d.ts +5 -1
- package/dist/workflow-steps/observability.d.ts +28 -2
- package/dist/workflow-steps/types.d.ts +11 -7
- package/dist/workflow-steps/workflow.d.ts +3 -2
- package/package.json +3 -2
- package/src/agent/agent-context.ts +14 -6
- package/src/agent/agent-loop.ts +32 -10
- package/src/agent/pause-client.ts +108 -0
- package/src/agent/run-agent.ts +9 -4
- package/src/agent/steer-control.ts +21 -7
- package/src/client.ts +35 -1
- package/src/index.ts +20 -2
- package/src/pause/checkpoint.ts +33 -14
- package/src/pause/manager.ts +2 -2
- package/src/pause/pause-core.ts +35 -0
- package/src/pause/state-dir.ts +2 -2
- package/src/processors/builtins.ts +44 -1
- package/src/processors/gate-pause.ts +94 -0
- package/src/processors/index.ts +7 -0
- package/src/processors/processor.ts +13 -0
- package/src/runtimes/_acp-client.ts +516 -0
- package/src/runtimes/_cli-agent.ts +416 -3
- package/src/runtimes/claude.ts +31 -3
- package/src/runtimes/codex.ts +21 -1
- package/src/runtimes/vercel.ts +4 -1
- package/src/sandbox.ts +426 -56
- package/src/step-invocation/index.ts +2 -1
- package/src/step-invocation/invoker.ts +195 -84
- package/src/types/execution-context.ts +0 -8
- package/src/types/protocol.ts +27 -1
- package/src/types/runtime.ts +14 -0
- package/src/types/sandbox-environment.ts +12 -1
- package/src/types/sandbox.ts +84 -6
- package/src/types/workflow-metadata.ts +42 -10
- package/src/types/workflow.ts +22 -7
- package/src/utils/bundler.ts +6 -1
- package/src/workflow-steps/observability.ts +51 -5
- package/src/workflow-steps/runner.ts +9 -5
- package/src/workflow-steps/types.ts +11 -7
- package/src/workflow-steps/workflow.ts +3 -2
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* In-sandbox pause client (ADR-0028). The single way an agent pauses its OWN
|
|
3
|
+
* run: `agentc pause` and the gate-pause processor both call
|
|
4
|
+
* `requestPauseAndAwait`, which drives the server pause API and blocks until a
|
|
5
|
+
* human resolves it.
|
|
6
|
+
*
|
|
7
|
+
* Create-then-poll, NOT a held connection:
|
|
8
|
+
* 1. POST /pauses creates the pending row. The server wakes the step
|
|
9
|
+
* activity, which freezes the live VM in place (E2B native suspend). The
|
|
10
|
+
* poll loop below is part of that frozen process — between polls it costs
|
|
11
|
+
* ZERO compute while the run is parked.
|
|
12
|
+
* 2. GET /pauses/:id long-polls (each request a bounded ~5s server-side
|
|
13
|
+
* LISTEN race) until the row is terminal, then returns the human's answer.
|
|
14
|
+
*
|
|
15
|
+
* There is no marker file and no exception threaded through workflow code — the
|
|
16
|
+
* pause is a server operation, so nothing a `try/catch` can swallow.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
export interface PauseDecision {
|
|
20
|
+
status: "resolved" | "expired" | "cancelled";
|
|
21
|
+
/** The human's answer (present on `resolved`). Caller-defined shape; the
|
|
22
|
+
* dashboard sends `{ decision: string }`. Null on expiry/cancel. */
|
|
23
|
+
decision: unknown;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export interface RequestPauseOptions {
|
|
27
|
+
/** `AGENT_COMPOSE_URL` — the server base URL injected into the sandbox. */
|
|
28
|
+
baseUrl: string;
|
|
29
|
+
/** `AGENT_COMPOSE_RUN_TOKEN` — the run credential the agent already carries. */
|
|
30
|
+
token: string;
|
|
31
|
+
/** `RUN_ID`. */
|
|
32
|
+
runId: string;
|
|
33
|
+
/** Human-readable reason shown on the pause feed / approval UI. */
|
|
34
|
+
reason: string;
|
|
35
|
+
/** Optional pause TTL; the workflow auto-expires the pause after this. */
|
|
36
|
+
ttlMs?: number;
|
|
37
|
+
/** Optional decision options the human picks from (e.g. approve / deny). */
|
|
38
|
+
options?: Array<{ id: string; label: string }>;
|
|
39
|
+
/** The action under review (e.g. the tool call), surfaced to the human. */
|
|
40
|
+
action?: Record<string, unknown>;
|
|
41
|
+
/** Optional second-key resume route. */
|
|
42
|
+
correlationKey?: string;
|
|
43
|
+
/** Abort the wait (the agent loop's signal). */
|
|
44
|
+
signal?: AbortSignal;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** Inject a custom fetch (tests). Defaults to global fetch. */
|
|
48
|
+
type FetchFn = typeof fetch;
|
|
49
|
+
|
|
50
|
+
const isAbort = (signal: AbortSignal | undefined) => Boolean(signal?.aborted);
|
|
51
|
+
|
|
52
|
+
export async function requestPauseAndAwait(
|
|
53
|
+
opts: RequestPauseOptions,
|
|
54
|
+
fetchImpl: FetchFn = fetch,
|
|
55
|
+
): Promise<PauseDecision> {
|
|
56
|
+
const base = opts.baseUrl.replace(/\/+$/, "");
|
|
57
|
+
const headers = {
|
|
58
|
+
authorization: `Bearer ${opts.token}`,
|
|
59
|
+
"content-type": "application/json",
|
|
60
|
+
accept: "application/json",
|
|
61
|
+
} as const;
|
|
62
|
+
|
|
63
|
+
// 1. Create the pending pause. The server wakes the step activity → the live
|
|
64
|
+
// VM suspends in place. 202 → { pauseId }.
|
|
65
|
+
const createRes = await fetchImpl(`${base}/api/v1/runs/${opts.runId}/pauses`, {
|
|
66
|
+
method: "POST",
|
|
67
|
+
headers,
|
|
68
|
+
body: JSON.stringify({
|
|
69
|
+
reason: opts.reason,
|
|
70
|
+
...(opts.ttlMs !== undefined ? { ttlMs: opts.ttlMs } : {}),
|
|
71
|
+
...(opts.options ? { options: opts.options } : {}),
|
|
72
|
+
...(opts.action ? { action: opts.action } : {}),
|
|
73
|
+
...(opts.correlationKey ? { correlationKey: opts.correlationKey } : {}),
|
|
74
|
+
}),
|
|
75
|
+
...(opts.signal ? { signal: opts.signal } : {}),
|
|
76
|
+
});
|
|
77
|
+
if (!createRes.ok) {
|
|
78
|
+
const text = await createRes.text().catch(() => "");
|
|
79
|
+
throw new Error(`pause create failed (${createRes.status}): ${text.slice(0, 300)}`);
|
|
80
|
+
}
|
|
81
|
+
const { pauseId } = await createRes.json() as { pauseId: string };
|
|
82
|
+
|
|
83
|
+
// 2. Poll until terminal. Each GET is a bounded server long-poll; the loop
|
|
84
|
+
// reissues. A transient error backs off and retries — the durable row is
|
|
85
|
+
// the source of truth, so a dropped poll never loses the decision.
|
|
86
|
+
const pollUrl = `${base}/api/v1/runs/${opts.runId}/pauses/${pauseId}`;
|
|
87
|
+
let backoffMs = 0;
|
|
88
|
+
for (;;) {
|
|
89
|
+
if (isAbort(opts.signal)) throw new Error("pause wait aborted");
|
|
90
|
+
if (backoffMs > 0) await new Promise((r) => setTimeout(r, backoffMs));
|
|
91
|
+
let res: Awaited<ReturnType<FetchFn>>;
|
|
92
|
+
try {
|
|
93
|
+
res = await fetchImpl(pollUrl, { method: "GET", headers, ...(opts.signal ? { signal: opts.signal } : {}) });
|
|
94
|
+
} catch (err) {
|
|
95
|
+
if (isAbort(opts.signal)) throw err;
|
|
96
|
+
backoffMs = Math.min(8_000, (backoffMs || 500) * 2);
|
|
97
|
+
continue;
|
|
98
|
+
}
|
|
99
|
+
if (!res.ok) {
|
|
100
|
+
backoffMs = Math.min(8_000, (backoffMs || 500) * 2);
|
|
101
|
+
continue;
|
|
102
|
+
}
|
|
103
|
+
backoffMs = 0;
|
|
104
|
+
const body = await res.json() as { status: string; resumePayload?: unknown };
|
|
105
|
+
if (body.status === "pending") continue;
|
|
106
|
+
return { status: body.status as PauseDecision["status"], decision: body.resumePayload ?? null };
|
|
107
|
+
}
|
|
108
|
+
}
|
package/src/agent/run-agent.ts
CHANGED
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
import { z } from "zod";
|
|
14
14
|
import { randomUUID } from "node:crypto";
|
|
15
15
|
import { getActiveStep, nextAgentCallInActiveStep } from "../active-step.js";
|
|
16
|
-
import { corePause } from "../pause/pause-core.js";
|
|
16
|
+
import { corePause, type PauseRequest } from "../pause/pause-core.js";
|
|
17
17
|
import { agentLoop } from "./agent-loop.js";
|
|
18
18
|
import { consumeSteerPending, runControlPoller } from "./steer-control.js";
|
|
19
19
|
import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
|
|
@@ -21,7 +21,7 @@ import { AsyncQueue } from "./async-queue.js";
|
|
|
21
21
|
import type { AgentMessage, AgentStatus } from "../types/protocol.js";
|
|
22
22
|
import type { AgentRuntime, RuntimeOptions } from "../types/runtime.js";
|
|
23
23
|
import type { SandboxProvider } from "../types/sandbox.js";
|
|
24
|
-
import { writeAgentContext } from "./agent-context.js";
|
|
24
|
+
import { writeAgentContext, buildAgentContextDoc } from "./agent-context.js";
|
|
25
25
|
import type { AgentBudget } from "../types/workflow.js";
|
|
26
26
|
import type { Processor } from "../processors/processor.js";
|
|
27
27
|
import { RequestContext } from "../request-context/request-context.js";
|
|
@@ -315,7 +315,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
|
|
|
315
315
|
const activeStepForPause = getActiveStep();
|
|
316
316
|
const steerPause = activeStepForPause && runId
|
|
317
317
|
? function steerPauseFn<T>(
|
|
318
|
-
req:
|
|
318
|
+
req: PauseRequest<T>,
|
|
319
319
|
agentScope: { agentId: string; iteration: number },
|
|
320
320
|
): Promise<T> {
|
|
321
321
|
return corePause<T>(req, {
|
|
@@ -333,7 +333,12 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
|
|
|
333
333
|
|
|
334
334
|
try {
|
|
335
335
|
return await agentLoop({
|
|
336
|
-
|
|
336
|
+
// Inject the platform manual into every runtime create() so a runtime that
|
|
337
|
+
// supports a system-prompt append (claude) carries it IN CONTEXT — not
|
|
338
|
+
// dependent on the agent choosing to `cat` AGENTS.md (which the SDK doesn't
|
|
339
|
+
// auto-load) or on writeAgentContext succeeding (it EACCES's on a read-only
|
|
340
|
+
// /workspace). buildAgentContextDoc is the same content writeAgentContext writes.
|
|
341
|
+
runtime: (runtimeOpts: RuntimeOptions) => opts.runtime.create(opts.sandbox, { ...runtimeOpts, agentManual: buildAgentContextDoc(process.env) }),
|
|
337
342
|
agentId,
|
|
338
343
|
...(opts.label !== undefined ? { label: opts.label } : {}),
|
|
339
344
|
...(opts.budget?.turnsPerIteration !== undefined ? { turnsPerIteration: opts.budget.turnsPerIteration } : {}),
|
|
@@ -17,13 +17,27 @@
|
|
|
17
17
|
import { z } from "zod";
|
|
18
18
|
|
|
19
19
|
/** What a steer/resume answer carries. Used as the boundary pause's `schema`
|
|
20
|
-
* so it is validated client-side in the re-spawned runner —
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
|
|
24
|
-
message
|
|
25
|
-
|
|
26
|
-
}
|
|
20
|
+
* so it is validated client-side in the re-spawned runner — an empty answer
|
|
21
|
+
* is rejected (PauseSchemaError) so an agent never resumes on nothing. The
|
|
22
|
+
* normalised `message` becomes the agent's next user turn.
|
|
23
|
+
*
|
|
24
|
+
* Accepts BOTH resume-payload conventions and normalises to `{ message }`:
|
|
25
|
+
* - `{ message }` — the SDK / API steer convention (`answerSteer`).
|
|
26
|
+
* - `{ decision }` — what the dashboard's RunPausePanel universally sends
|
|
27
|
+
* for every pause (option click or free text). Without this, resuming an
|
|
28
|
+
* agent steer / `needs_input` / `agentc pause` pause from the dashboard
|
|
29
|
+
* failed schema validation in the re-spawned runner and the agent never
|
|
30
|
+
* got the answer — the resume looked like it did nothing. */
|
|
31
|
+
export const SteerDecisionSchema = z
|
|
32
|
+
.object({
|
|
33
|
+
message: z.string().min(1).optional(),
|
|
34
|
+
decision: z.string().min(1).optional(),
|
|
35
|
+
actor: z.string().nullish(),
|
|
36
|
+
})
|
|
37
|
+
.transform((d) => ({ message: (d.message ?? d.decision ?? "").trim(), actor: d.actor ?? null }))
|
|
38
|
+
.refine((d) => d.message.length > 0, {
|
|
39
|
+
message: "a steer/resume answer needs a non-empty `message` or `decision`",
|
|
40
|
+
});
|
|
27
41
|
export type SteerDecision = z.infer<typeof SteerDecisionSchema>;
|
|
28
42
|
|
|
29
43
|
/** Metadata a pending steer carries from the control message to the boundary
|
package/src/client.ts
CHANGED
|
@@ -103,7 +103,8 @@ export interface RegisterWorkflowInput {
|
|
|
103
103
|
/** All snapshot config — `bootFrom` (where to restore at run start),
|
|
104
104
|
* `save`, `retain`. See `WorkflowMetadata.snapshots`. */
|
|
105
105
|
snapshots?: SnapshotConfig;
|
|
106
|
-
/** Sandbox machine size (template
|
|
106
|
+
/** Sandbox machine resources — size + provider (template defaults).
|
|
107
|
+
* See `WorkflowMetadata.resources`. */
|
|
107
108
|
resources?: SandboxResources;
|
|
108
109
|
/** Provider-neutral execution plan detected by the CLI bundler. */
|
|
109
110
|
workflowPlan?: WorkflowPlan;
|
|
@@ -121,6 +122,10 @@ export interface RegisterWorkflowInput {
|
|
|
121
122
|
inputSchema?: IOSchema;
|
|
122
123
|
/** Output schema extracted from the workflow's `output` zod schema. */
|
|
123
124
|
outputSchema?: IOSchema;
|
|
125
|
+
/** Set by `defineSandboxEnvironment` — marks an environment build so the
|
|
126
|
+
* server skips the /factory mount for its runs (#13). See
|
|
127
|
+
* `WorkflowMetadata.environmentBuild`. */
|
|
128
|
+
environmentBuild?: boolean;
|
|
124
129
|
/** Factory slug. Defaults to `"default"`. */
|
|
125
130
|
factorySlug?: string;
|
|
126
131
|
}
|
|
@@ -1240,6 +1245,35 @@ export class AgentComposeClient {
|
|
|
1240
1245
|
return this.fetch(templatePath(factorySlug, workflowName, "secrets", key), { method: "DELETE" });
|
|
1241
1246
|
}
|
|
1242
1247
|
|
|
1248
|
+
// ── Factory-level secrets (ADR-0014) ────────────────────────────────────────
|
|
1249
|
+
// The inherited tier: a factory secret is visible to EVERY workflow in the
|
|
1250
|
+
// factory; a workflow secret of the same key overrides it. Values are stored
|
|
1251
|
+
// in GCP Secret Manager; never returned by reads.
|
|
1252
|
+
|
|
1253
|
+
/** Create or update a factory-level secret. */
|
|
1254
|
+
setFactorySecret(key: string, value: string, opts?: SecretOptions): Promise<SetSecretResult> {
|
|
1255
|
+
const factorySlug = opts?.factorySlug ?? DEFAULT_FACTORY;
|
|
1256
|
+
return this.fetch(`/api/v1/factories/${encodeURIComponent(factorySlug)}/secrets`, {
|
|
1257
|
+
method: "POST",
|
|
1258
|
+
body: { key, value },
|
|
1259
|
+
});
|
|
1260
|
+
}
|
|
1261
|
+
|
|
1262
|
+
/** List factory-level secret keys (metadata only — values are never returned). */
|
|
1263
|
+
async listFactorySecrets(opts?: SecretOptions): Promise<SecretListEntry[]> {
|
|
1264
|
+
const factorySlug = opts?.factorySlug ?? DEFAULT_FACTORY;
|
|
1265
|
+
const body = await this.fetch<{ secrets: Array<{ secretKey: string; createdAt: string; updatedAt: string }> }>(
|
|
1266
|
+
`/api/v1/factories/${encodeURIComponent(factorySlug)}/secrets`,
|
|
1267
|
+
);
|
|
1268
|
+
return body.secrets.map(s => ({ key: s.secretKey, createdAt: s.createdAt, updatedAt: s.updatedAt }));
|
|
1269
|
+
}
|
|
1270
|
+
|
|
1271
|
+
/** Delete a factory-level secret. */
|
|
1272
|
+
deleteFactorySecret(key: string, opts?: SecretOptions): Promise<void> {
|
|
1273
|
+
const factorySlug = opts?.factorySlug ?? DEFAULT_FACTORY;
|
|
1274
|
+
return this.fetch(`/api/v1/factories/${encodeURIComponent(factorySlug)}/secrets/${encodeURIComponent(key)}`, { method: "DELETE" });
|
|
1275
|
+
}
|
|
1276
|
+
|
|
1243
1277
|
// ── API keys ───────────────────────────────────────────────────────────────
|
|
1244
1278
|
// Both endpoints require an admin-scoped key as the bearer token.
|
|
1245
1279
|
|
package/src/index.ts
CHANGED
|
@@ -73,14 +73,19 @@ export {
|
|
|
73
73
|
Verdict,
|
|
74
74
|
runProcessorChain,
|
|
75
75
|
denyTools,
|
|
76
|
+
humanApproval,
|
|
76
77
|
requireScope,
|
|
77
78
|
redactPattern,
|
|
79
|
+
createGatePauseProcessor,
|
|
78
80
|
} from "./processors/index.js";
|
|
79
81
|
export type {
|
|
80
82
|
Processor,
|
|
81
83
|
ProcessorContext,
|
|
82
84
|
ProcessorVerdict,
|
|
83
85
|
ToolCall,
|
|
86
|
+
GatePausePolicy,
|
|
87
|
+
GatePauseApproval,
|
|
88
|
+
GatePauseConnection,
|
|
84
89
|
} from "./processors/index.js";
|
|
85
90
|
|
|
86
91
|
// Protocol types (agent-loop input/output shapes)
|
|
@@ -179,9 +184,12 @@ export type { RunEvent } from "./types/events.js";
|
|
|
179
184
|
|
|
180
185
|
// Sandbox providers
|
|
181
186
|
export { createSandbox, reconnectSandbox, killAllSandboxes, killSandboxById,
|
|
182
|
-
getSandboxQuotas, listOwnedSandboxes, deleteSandboxSnapshot,
|
|
187
|
+
getSandboxQuotas, listOwnedSandboxes, deleteSandboxSnapshot, snapshotResolves,
|
|
183
188
|
makeSandboxProvider, makeDesktopSandboxProvider,
|
|
184
|
-
parseSseExecStream, AGENT_COMPOSE_TAG
|
|
189
|
+
parseSseExecStream, AGENT_COMPOSE_TAG,
|
|
190
|
+
SANDBOX_VCPUS, DEFAULT_SANDBOX_SIZE, E2B_TEMPLATE_SIZES,
|
|
191
|
+
isE2bSupportedSize, e2bMachineSpec, e2bBaseTemplate, e2bAgentEnvTemplate,
|
|
192
|
+
isPlatformE2bTemplateAlias } from "./sandbox.js";
|
|
185
193
|
export { SandboxUnavailableError, SANDBOX_UNAVAILABLE_PREFIX } from "./sandbox-errors.js";
|
|
186
194
|
export type {
|
|
187
195
|
SandboxCreateOpts, SandboxNetworkPolicy, SandboxNetworkHeaderTransform,
|
|
@@ -228,6 +236,8 @@ export type {
|
|
|
228
236
|
// `invokeStep` (server) and `serveStep` (runner).
|
|
229
237
|
export {
|
|
230
238
|
invokeStep,
|
|
239
|
+
launchStep,
|
|
240
|
+
reconnectStep,
|
|
231
241
|
serveStep,
|
|
232
242
|
parseStepResult,
|
|
233
243
|
buildStepEnvs,
|
|
@@ -252,6 +262,12 @@ export {
|
|
|
252
262
|
export type { PauseErrorCode } from "./pause/errors.js";
|
|
253
263
|
export type { PauseRequest } from "./pause/pause-core.js";
|
|
254
264
|
export type { WaitForEventRequest } from "./pause/wrappers.js";
|
|
265
|
+
// ADR-0028 — the in-sandbox pause client. `agentc pause` and the gate-pause
|
|
266
|
+
// processor both pause their OWN run through the server pause API and block
|
|
267
|
+
// until a human answers (create-then-poll; the VM suspends in place while
|
|
268
|
+
// parked). No marker, no exception threaded through workflow code.
|
|
269
|
+
export { requestPauseAndAwait } from "./agent/pause-client.js";
|
|
270
|
+
export type { PauseDecision, RequestPauseOptions } from "./agent/pause-client.js";
|
|
255
271
|
export type {
|
|
256
272
|
StepRequest,
|
|
257
273
|
StepResult,
|
|
@@ -260,6 +276,8 @@ export type {
|
|
|
260
276
|
StepHandler,
|
|
261
277
|
StepHandlerResult,
|
|
262
278
|
ServeStepRequest,
|
|
279
|
+
RunningStep,
|
|
280
|
+
InvokeStepOptions,
|
|
263
281
|
} from "./step-invocation/index.js";
|
|
264
282
|
|
|
265
283
|
// Agent loop — for workflows that embed an LLM agent in their run() body.
|
package/src/pause/checkpoint.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* Disk-backed memoise across pause-resume — the engine behind durable
|
|
3
|
+
* `ctx.step(name, fn)` (ADR-0012).
|
|
3
4
|
*
|
|
4
5
|
* First call: `fn()` runs, the result is atomically written to
|
|
5
6
|
* `/tmp/wf/state/checkpoints/<name>.json`. On any subsequent invocation
|
|
@@ -16,29 +17,47 @@
|
|
|
16
17
|
* memoisation is in-sandbox only and would silently retry on a sandbox
|
|
17
18
|
* recreation that lost the checkpoint file.
|
|
18
19
|
*
|
|
19
|
-
*
|
|
20
|
+
* INTERNAL only — there is no public `ctx.checkpoint`. The durable
|
|
21
|
+
* `ctx.step` wires `scopedMemoize("step<idx>")` and reports a "restored"
|
|
22
|
+
* sub-step on a cache hit. See ADR-0012.
|
|
20
23
|
*/
|
|
21
24
|
|
|
22
25
|
import { assertSafeCheckpointName, readCheckpoint, writeCheckpoint } from "./state-dir.js";
|
|
23
26
|
|
|
24
|
-
/**
|
|
25
|
-
*
|
|
26
|
-
*
|
|
27
|
-
|
|
27
|
+
/** Result of a memoise call — `restored` distinguishes a disk hit (fn was
|
|
28
|
+
* NOT run, value read from a prior subprocess) from a fresh run (fn ran,
|
|
29
|
+
* value just written). The durable `ctx.step` maps `restored` onto the
|
|
30
|
+
* duration-0 "restored" sub-step status. */
|
|
31
|
+
export interface MemoizeResult<T> {
|
|
32
|
+
value: T;
|
|
33
|
+
restored: boolean;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/** Read-or-run-and-write helper reporting whether the value was restored
|
|
37
|
+
* from disk. `name` is validated by the state-dir layer (rejects
|
|
38
|
+
* path-escape attempts); `fn` is awaited if it returns a Promise so sync
|
|
39
|
+
* and async callers compose identically.
|
|
40
|
+
*
|
|
41
|
+
* A pause inside `fn` (PauseSignal) propagates BEFORE any write — a
|
|
42
|
+
* partially-completed body is never memoised, so the resume re-runs it. */
|
|
43
|
+
export async function memoize<T>(name: string, fn: () => Promise<T> | T): Promise<MemoizeResult<T>> {
|
|
28
44
|
const existing = await readCheckpoint<T>(name);
|
|
29
|
-
if (existing.found) return existing.value;
|
|
45
|
+
if (existing.found) return { value: existing.value, restored: true };
|
|
30
46
|
const value = await fn();
|
|
31
47
|
await writeCheckpoint(name, value);
|
|
32
|
-
return value;
|
|
48
|
+
return { value, restored: false };
|
|
33
49
|
}
|
|
34
50
|
|
|
35
|
-
/**
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
51
|
+
/** A memoise function bound to a runner-owned scope. */
|
|
52
|
+
export type ScopedMemoize = <T>(name: string, fn: () => Promise<T> | T) => Promise<MemoizeResult<T>>;
|
|
53
|
+
|
|
54
|
+
/** Bind memoise names to a runner-owned namespace (e.g. `step<idx>`). The
|
|
55
|
+
* caller's `name` is still validated independently so a scope prefix
|
|
56
|
+
* cannot turn an empty or path-escaping name into a valid filename. */
|
|
57
|
+
export function scopedMemoize(scope: string): ScopedMemoize {
|
|
39
58
|
assertSafeCheckpointName(scope);
|
|
40
|
-
return async <T>(name: string, fn: () => Promise<T> | T): Promise<T
|
|
59
|
+
return async <T>(name: string, fn: () => Promise<T> | T): Promise<MemoizeResult<T>> => {
|
|
41
60
|
assertSafeCheckpointName(name);
|
|
42
|
-
return
|
|
61
|
+
return memoize(`${scope}.${name}`, fn);
|
|
43
62
|
};
|
|
44
63
|
}
|
package/src/pause/manager.ts
CHANGED
|
@@ -44,7 +44,7 @@ const RunningAgentLoopStateSchema = z.object({
|
|
|
44
44
|
blockerStreak: z.object({ key: z.string(), count: z.number().int().nonnegative() }).nullable(),
|
|
45
45
|
// PR 7: a human-requested steer-pause intent, persisted so it survives a
|
|
46
46
|
// resume and the loop re-issues the boundary pause on re-entry.
|
|
47
|
-
pendingSteerPause: z.object({ reason: z.string(), correlationKey: z.string().nullable(), at: z.number().int() }).nullable().default(null),
|
|
47
|
+
pendingSteerPause: z.object({ reason: z.string(), correlationKey: z.string().nullable(), at: z.number().int(), payload: z.record(z.string(), z.unknown()).optional() }).nullable().default(null),
|
|
48
48
|
});
|
|
49
49
|
|
|
50
50
|
const SettledAgentLoopStateSchema = z.object({
|
|
@@ -76,7 +76,7 @@ export interface AgentLoopProgressState {
|
|
|
76
76
|
blockerStreak: { key: string; count: number } | null;
|
|
77
77
|
/** PR 7 steer-pause intent — see RunningAgentLoopStateSchema. Optional so the
|
|
78
78
|
* loop can omit it until it wires steer (the schema defaults it to null). */
|
|
79
|
-
pendingSteerPause?: { reason: string; correlationKey: string | null; at: number } | null;
|
|
79
|
+
pendingSteerPause?: { reason: string; correlationKey: string | null; at: number; payload?: Record<string, unknown> } | null;
|
|
80
80
|
}
|
|
81
81
|
|
|
82
82
|
export interface SettledAgentLoopResult<TResponse = unknown> {
|
package/src/pause/pause-core.ts
CHANGED
|
@@ -94,6 +94,41 @@ export function isPauseSignal(err: unknown): err is PauseSignal {
|
|
|
94
94
|
return typeof err === "object" && err !== null && (err as Record<symbol, unknown>)[PAUSE_SIGNAL_BRAND] === true;
|
|
95
95
|
}
|
|
96
96
|
|
|
97
|
+
/**
|
|
98
|
+
* The run's pause boundary. The runner (`run-agent` `steerPause`) closes a
|
|
99
|
+
* `corePause` over the ambient `runId`/`stepIndex`; the agent loop / runtime
|
|
100
|
+
* supplies the per-iteration `agentScope` so the pauseId stays stable across
|
|
101
|
+
* the loop's iteration-skipping on resume. Absent outside a workflow step
|
|
102
|
+
* (local / non-sandbox callers) — see `boundProcessorPause`.
|
|
103
|
+
*/
|
|
104
|
+
export type BoundaryPauseFn = <T = unknown>(
|
|
105
|
+
req: PauseRequest<T>,
|
|
106
|
+
agentScope: { agentId: string; iteration: number },
|
|
107
|
+
) => Promise<T>;
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Build the `ProcessorContext.pause` callable for a given hook scope: binds the
|
|
111
|
+
* boundary to this hook's `{ agentId, iteration }` so a processor calling
|
|
112
|
+
* `ctx.pause(req)` gets a deterministic, resume-stable pauseId. When no boundary
|
|
113
|
+
* is wired (local tests, non-sandbox callers), the returned fn rejects loudly
|
|
114
|
+
* rather than silently no-op'ing — pause genuinely cannot work without the
|
|
115
|
+
* runner's snapshot/Temporal machinery.
|
|
116
|
+
*/
|
|
117
|
+
export function boundProcessorPause(
|
|
118
|
+
boundary: BoundaryPauseFn | undefined,
|
|
119
|
+
scope: { agentId: string; iteration: number },
|
|
120
|
+
): <T = unknown>(req: PauseRequest<T>) => Promise<T> {
|
|
121
|
+
return <T = unknown>(req: PauseRequest<T>): Promise<T> => {
|
|
122
|
+
if (!boundary) {
|
|
123
|
+
return Promise.reject(new PauseRequestError(
|
|
124
|
+
"ctx.pause is unavailable here: no pause boundary is wired (a processor can " +
|
|
125
|
+
"only pause inside a sandboxed workflow step, not in a local/unit-test run)",
|
|
126
|
+
));
|
|
127
|
+
}
|
|
128
|
+
return boundary(req, scope);
|
|
129
|
+
};
|
|
130
|
+
}
|
|
131
|
+
|
|
97
132
|
// ── Deterministic pauseId ───────────────────────────────────────────────────
|
|
98
133
|
|
|
99
134
|
/** Fixed namespace for agent-compose pause ids (UUIDv5). Constant — must
|
package/src/pause/state-dir.ts
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
*
|
|
7
7
|
* agent-<agentInstanceId>.json ← agent loop state (iteration, messages, processor cursor)
|
|
8
8
|
* runtime-<agentInstanceId>.json ← runtime-private blob (opaque to the loop)
|
|
9
|
-
* checkpoints/<name>.json ←
|
|
9
|
+
* checkpoints/<name>.json ← durable `ctx.step(name, fn)` memoised values (keyed `step<idx>.<name>`)
|
|
10
10
|
* pauses/<pauseId>.json ← pause records (request + resume payload once available)
|
|
11
11
|
*
|
|
12
12
|
* Atomic writes (write-tmp → fsync → rename) so a mid-write `sandbox.snapshot()`
|
|
@@ -50,7 +50,7 @@ const pausePath = (pauseId: string) => join(pausesDir(), `${pause
|
|
|
50
50
|
|
|
51
51
|
/** `name` is interpolated into a filename; reject anything outside a
|
|
52
52
|
* conservative slug. Prevents a workflow author from writing
|
|
53
|
-
* `ctx.
|
|
53
|
+
* `ctx.step("../../etc/passwd", fn)` and escaping the dir.
|
|
54
54
|
*
|
|
55
55
|
* First character must be alphanumeric or `_` so dotfile names (`.foo`)
|
|
56
56
|
* and parent-directory tokens (`.`, `..`) are rejected regardless of
|
|
@@ -4,7 +4,8 @@
|
|
|
4
4
|
* exist as load-bearing examples and as defaults for common policies.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
|
-
import
|
|
7
|
+
import { z } from "zod";
|
|
8
|
+
import type { Processor, ToolCall } from "./processor.js";
|
|
8
9
|
import { Verdict } from "./processor.js";
|
|
9
10
|
|
|
10
11
|
/**
|
|
@@ -27,6 +28,48 @@ export function denyTools(names: readonly string[]): Processor {
|
|
|
27
28
|
};
|
|
28
29
|
}
|
|
29
30
|
|
|
31
|
+
/** Resume payload a reviewer sends back to a `humanApproval` pause. */
|
|
32
|
+
const ApprovalDecision = z.object({
|
|
33
|
+
approved: z.boolean(),
|
|
34
|
+
reason: z.string().optional(),
|
|
35
|
+
});
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Pause the run for HUMAN APPROVAL before a matching tool executes — the
|
|
39
|
+
* human-in-the-loop pre-tool gate (ADR-0006 `ctx.pause`). Over an ACP runtime
|
|
40
|
+
* this is exactly the `session/request_permission` path: the CLI asks to run a
|
|
41
|
+
* tool, the gate runs here, and `ctx.pause` snapshots the workflow and waits
|
|
42
|
+
* durably until a reviewer resolves the pause with `{ approved, reason? }`.
|
|
43
|
+
* Approve → the tool runs; deny → the reason is returned to the model as the
|
|
44
|
+
* tool result and the loop continues.
|
|
45
|
+
*
|
|
46
|
+
* tools? — only these tool names require approval (default: EVERY tool call).
|
|
47
|
+
* reason? — build the human-facing prompt from the call (default names the tool).
|
|
48
|
+
*
|
|
49
|
+
* Dormant on runtimes without pre-tool gating; active on those that wire
|
|
50
|
+
* `processToolCall` (the ACP CLI runtimes, the Claude pre-tool hook, Vercel).
|
|
51
|
+
*/
|
|
52
|
+
export function humanApproval(opts?: {
|
|
53
|
+
tools?: readonly string[];
|
|
54
|
+
reason?: (call: ToolCall) => string;
|
|
55
|
+
}): Processor {
|
|
56
|
+
const gated = opts?.tools ? new Set(opts.tools) : null;
|
|
57
|
+
return {
|
|
58
|
+
name: "humanApproval",
|
|
59
|
+
async processToolCall(call, ctx) {
|
|
60
|
+
if (gated && !gated.has(call.toolName)) return Verdict.continue(call);
|
|
61
|
+
const decision = await ctx.pause({
|
|
62
|
+
reason: opts?.reason?.(call) ?? `Approve tool call: ${call.toolName}`,
|
|
63
|
+
payload: { tool: call.toolName, input: call.toolInput, toolUseId: call.toolUseId },
|
|
64
|
+
schema: ApprovalDecision,
|
|
65
|
+
});
|
|
66
|
+
return decision.approved
|
|
67
|
+
? Verdict.continue(call)
|
|
68
|
+
: Verdict.deny(decision.reason ?? `tool "${call.toolName}" denied by human reviewer`);
|
|
69
|
+
},
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
|
|
30
73
|
/**
|
|
31
74
|
* Require the calling API key to carry every scope in `required`. Aborts the
|
|
32
75
|
* agent loop with a clear reason if any are missing — defence-in-depth on
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Gate-pause processor (ADR-0028).
|
|
3
|
+
*
|
|
4
|
+
* A pre-tool gate that, when its `policy` flags a tool call as needing human
|
|
5
|
+
* approval, pauses the run as a SERVER operation (`requestPauseAndAwait`) and
|
|
6
|
+
* blocks until a human answers — then allows or denies the tool.
|
|
7
|
+
*
|
|
8
|
+
* It pauses through the server pause API (`requestPauseAndAwait`), so the pause
|
|
9
|
+
* is a server operation no `try/catch` can swallow. It does NOT use today's
|
|
10
|
+
* `ctx.pause`, which still throws `PauseSignal` — exactly the swallowable path
|
|
11
|
+
* ADR-0028 supersedes for agents. (Routing `ctx.pause` itself through this same
|
|
12
|
+
* server pause is the natural unification — ADR-0028 open question — at which
|
|
13
|
+
* point "the pause API" and "ctx.pause" become one thing.) Because it lives in
|
|
14
|
+
* the shared `gateToolCall` chain, one implementation covers every runtime — the
|
|
15
|
+
* Claude Agent SDK `PreToolUse` hook and the ACP `session/request_permission`
|
|
16
|
+
* path both route through it.
|
|
17
|
+
*
|
|
18
|
+
* Opt-in: the workflow/agent supplies the `policy` (which may be a heuristic or
|
|
19
|
+
* an async classifier agent). The connection defaults to the run credential in
|
|
20
|
+
* the sandbox env; absent it (local / non-sandbox), the gate is a no-op and
|
|
21
|
+
* tool calls pass through untouched.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import type { Processor, ProcessorContext, ToolCall } from "./processor.js";
|
|
25
|
+
import { Verdict } from "./processor.js";
|
|
26
|
+
import { requestPauseAndAwait } from "../agent/pause-client.js";
|
|
27
|
+
|
|
28
|
+
export interface GatePauseApproval {
|
|
29
|
+
/** Human-readable question shown on the approval UI. */
|
|
30
|
+
reason: string;
|
|
31
|
+
/** Decision options the human picks from. Defaults to Approve / Deny. */
|
|
32
|
+
options?: Array<{ id: string; label: string }>;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/** Decide whether a tool call needs human approval. Return `false` to let it
|
|
36
|
+
* through untouched, or an approval request to pause the run until a human
|
|
37
|
+
* answers. May be async (e.g. a small classifier agent). */
|
|
38
|
+
export type GatePausePolicy = (
|
|
39
|
+
call: ToolCall,
|
|
40
|
+
ctx: ProcessorContext,
|
|
41
|
+
) => (false | GatePauseApproval) | Promise<false | GatePauseApproval>;
|
|
42
|
+
|
|
43
|
+
export interface GatePauseConnection { baseUrl: string; token: string; runId: string }
|
|
44
|
+
|
|
45
|
+
const DENY_ANSWERS = new Set(["deny", "no", "reject", "decline", "block"]);
|
|
46
|
+
|
|
47
|
+
/** Unwrap the dashboard's `{ decision }` resume payload to the raw answer text. */
|
|
48
|
+
function answerText(decision: unknown): string {
|
|
49
|
+
const raw = decision !== null && typeof decision === "object" && "decision" in decision
|
|
50
|
+
? (decision as { decision: unknown }).decision
|
|
51
|
+
: decision;
|
|
52
|
+
return typeof raw === "string" ? raw.trim() : raw == null ? "" : JSON.stringify(raw);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function createGatePauseProcessor(opts: {
|
|
56
|
+
policy: GatePausePolicy;
|
|
57
|
+
/** Override the server connection (defaults to the sandbox run-credential env). */
|
|
58
|
+
connection?: GatePauseConnection;
|
|
59
|
+
}): Processor {
|
|
60
|
+
return {
|
|
61
|
+
name: "gate-pause",
|
|
62
|
+
async processToolCall(call: ToolCall, ctx: ProcessorContext) {
|
|
63
|
+
const approval = await opts.policy(call, ctx);
|
|
64
|
+
if (!approval) return Verdict.continue(call);
|
|
65
|
+
|
|
66
|
+
const conn = opts.connection ?? {
|
|
67
|
+
baseUrl: process.env.AGENT_COMPOSE_URL ?? "",
|
|
68
|
+
token: process.env.AGENT_COMPOSE_RUN_TOKEN ?? "",
|
|
69
|
+
runId: process.env.RUN_ID ?? "",
|
|
70
|
+
};
|
|
71
|
+
// No run credential (local / non-sandbox) — can't pause; let it through
|
|
72
|
+
// rather than hard-failing a dev invocation.
|
|
73
|
+
if (!conn.baseUrl || !conn.token || !conn.runId) return Verdict.continue(call);
|
|
74
|
+
|
|
75
|
+
const decision = await requestPauseAndAwait({
|
|
76
|
+
baseUrl: conn.baseUrl, token: conn.token, runId: conn.runId,
|
|
77
|
+
reason: approval.reason,
|
|
78
|
+
action: { tool: call.toolName, input: call.toolInput },
|
|
79
|
+
options: approval.options ?? [{ id: "approve", label: "Approve" }, { id: "deny", label: "Deny" }],
|
|
80
|
+
signal: ctx.abortSignal,
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
const answer = answerText(decision.decision);
|
|
84
|
+
// Approve unless the human explicitly denied (or no answer came back on
|
|
85
|
+
// expiry/cancel). A free-form answer that isn't a deny word lets the tool
|
|
86
|
+
// run, with the human's guidance available to the model on the next turn.
|
|
87
|
+
if (decision.status === "resolved" && !DENY_ANSWERS.has(answer.toLowerCase())) {
|
|
88
|
+
return Verdict.continue(call);
|
|
89
|
+
}
|
|
90
|
+
const why = decision.status === "resolved" ? `denied: ${answer}` : `${decision.status} with no approval`;
|
|
91
|
+
return Verdict.deny(`Human ${why}. Tool "${call.toolName}" was not run — adjust course or ask again.`);
|
|
92
|
+
},
|
|
93
|
+
};
|
|
94
|
+
}
|
package/src/processors/index.ts
CHANGED
|
@@ -10,6 +10,13 @@ export { runProcessorChain } from "./runner.js";
|
|
|
10
10
|
|
|
11
11
|
export {
|
|
12
12
|
denyTools,
|
|
13
|
+
humanApproval,
|
|
13
14
|
requireScope,
|
|
14
15
|
redactPattern,
|
|
15
16
|
} from "./builtins.js";
|
|
17
|
+
|
|
18
|
+
// ADR-0028 — server-driven human-approval gate. Pauses the run through the
|
|
19
|
+
// server pause API and blocks until a human answers; covers every runtime
|
|
20
|
+
// through the shared gateToolCall chain.
|
|
21
|
+
export { createGatePauseProcessor } from "./gate-pause.js";
|
|
22
|
+
export type { GatePausePolicy, GatePauseApproval, GatePauseConnection } from "./gate-pause.js";
|
|
@@ -30,6 +30,7 @@
|
|
|
30
30
|
|
|
31
31
|
import type { AgentMessage } from "../types/protocol.js";
|
|
32
32
|
import type { RequestContext } from "../request-context/request-context.js";
|
|
33
|
+
import type { PauseRequest } from "../pause/pause-core.js";
|
|
33
34
|
|
|
34
35
|
/** Proposed tool call. Mirrors the relevant fields of AgentMessageToolUse but
|
|
35
36
|
* lives as its own type so runtime adapters (candidate #1) can populate it
|
|
@@ -53,6 +54,18 @@ export interface ProcessorContext {
|
|
|
53
54
|
agentId: string;
|
|
54
55
|
/** Iteration of the agent loop (1-based). */
|
|
55
56
|
iteration: number;
|
|
57
|
+
/**
|
|
58
|
+
* Pause the run from inside a processor — the human-approval primitive for a
|
|
59
|
+
* pre-tool-use gate (ADR-0006 §"reachable from every user code path";
|
|
60
|
+
* ADR-0020 the ACP `session/request_permission` path). On the fresh pass it
|
|
61
|
+
* throws `PauseSignal` (internal control flow) so the workflow snapshots and
|
|
62
|
+
* waits durably; on step re-entry after resolution it returns the resume
|
|
63
|
+
* payload (validated against `req.schema` if given). The agent loop / runtime
|
|
64
|
+
* wires it to the run's pause boundary with a deterministic, resume-stable
|
|
65
|
+
* pauseId (keyed on this hook's agentId + iteration). Calling it outside a
|
|
66
|
+
* sandboxed workflow step throws — pause requires the runner's pause boundary.
|
|
67
|
+
*/
|
|
68
|
+
pause<T = unknown>(req: PauseRequest<T>): Promise<T>;
|
|
56
69
|
}
|
|
57
70
|
|
|
58
71
|
/**
|