@agent-compose/sdk 0.5.7 → 0.5.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/dist/agent/__tests__/run-agent-liveness.test.d.ts +17 -0
  2. package/dist/agent/agent-context.d.ts +67 -0
  3. package/dist/agent/agent-loop.d.ts +23 -12
  4. package/dist/agent/local-pause-request.d.ts +49 -0
  5. package/dist/agent/local-pause-request.test.d.ts +1 -0
  6. package/dist/agent/steer-control.d.ts +22 -6
  7. package/dist/client.d.ts +76 -2
  8. package/dist/index.d.ts +10 -5
  9. package/dist/index.js +2409 -1457
  10. package/dist/pause/checkpoint.d.ts +27 -10
  11. package/dist/pause/manager.d.ts +1 -0
  12. package/dist/pause/pause-core.d.ts +23 -0
  13. package/dist/pause/state-dir.d.ts +1 -1
  14. package/dist/pause/wrappers.d.ts +7 -11
  15. package/dist/processors/builtins.d.ts +20 -1
  16. package/dist/processors/index.d.ts +1 -1
  17. package/dist/processors/processor.d.ts +13 -0
  18. package/dist/runtimes/_acp-client.d.ts +140 -0
  19. package/dist/runtimes/_cli-agent.d.ts +155 -3
  20. package/dist/runtimes/amp.d.ts +2 -2
  21. package/dist/runtimes/cli-agent-acp-live.test.d.ts +30 -0
  22. package/dist/runtimes/cli-agent.test.d.ts +22 -6
  23. package/dist/runtimes/codex.d.ts +7 -2
  24. package/dist/runtimes/openai-desktop.js +2394 -1457
  25. package/dist/runtimes/vercel.js +389 -2
  26. package/dist/sandbox.d.ts +132 -14
  27. package/dist/step-invocation/types.d.ts +1 -1
  28. package/dist/types/__tests__/environment-build-flag.test.d.ts +1 -0
  29. package/dist/types/__tests__/workflow-metadata-provider.test.d.ts +1 -0
  30. package/dist/types/execution-context.d.ts +1 -11
  31. package/dist/types/protocol.d.ts +32 -1
  32. package/dist/types/runtime.d.ts +7 -0
  33. package/dist/types/sandbox-environment.d.ts +6 -1
  34. package/dist/types/sandbox.d.ts +41 -6
  35. package/dist/types/workflow-metadata.d.ts +47 -6
  36. package/dist/types/workflow.d.ts +27 -4
  37. package/dist/utils/bundler.d.ts +7 -1
  38. package/dist/workflow-steps/observability.d.ts +28 -2
  39. package/dist/workflow-steps/types.d.ts +11 -7
  40. package/dist/workflow-steps/workflow.d.ts +5 -1
  41. package/package.json +3 -2
  42. package/src/agent/agent-context.ts +220 -0
  43. package/src/agent/agent-loop.ts +90 -22
  44. package/src/agent/local-pause-request.ts +90 -0
  45. package/src/agent/run-agent.ts +43 -3
  46. package/src/agent/steer-control.ts +21 -7
  47. package/src/client.ts +123 -2
  48. package/src/index.ts +16 -4
  49. package/src/pause/checkpoint.ts +33 -14
  50. package/src/pause/manager.ts +2 -2
  51. package/src/pause/pause-core.ts +35 -0
  52. package/src/pause/state-dir.ts +2 -2
  53. package/src/pause/wrappers.ts +7 -21
  54. package/src/processors/builtins.ts +44 -1
  55. package/src/processors/index.ts +1 -0
  56. package/src/processors/processor.ts +13 -0
  57. package/src/runtimes/_acp-client.ts +516 -0
  58. package/src/runtimes/_cli-agent.ts +418 -3
  59. package/src/runtimes/claude.ts +27 -3
  60. package/src/runtimes/codex.ts +21 -1
  61. package/src/runtimes/vercel.ts +4 -1
  62. package/src/sandbox.ts +429 -67
  63. package/src/step-invocation/types.ts +1 -1
  64. package/src/types/execution-context.ts +1 -11
  65. package/src/types/protocol.ts +27 -1
  66. package/src/types/runtime.ts +7 -0
  67. package/src/types/sandbox-environment.ts +12 -1
  68. package/src/types/sandbox.ts +40 -6
  69. package/src/types/workflow-metadata.ts +51 -6
  70. package/src/types/workflow.ts +27 -6
  71. package/src/utils/bundler.ts +9 -1
  72. package/src/workflow-steps/observability.ts +51 -5
  73. package/src/workflow-steps/runner.ts +9 -5
  74. package/src/workflow-steps/types.ts +11 -7
  75. package/src/workflow-steps/workflow.ts +5 -1
  76. package/src/workflows/invoke-child.ts +7 -1
@@ -0,0 +1,220 @@
1
+ /**
2
+ * Harness-agnostic agent context delivery.
3
+ *
4
+ * Every coding-agent harness we drive (Claude Code, Codex, Amp, Gemini, …)
5
+ * looks for an instruction file in its working directory — but they disagree
6
+ * on the NAME (Codex/Amp read `AGENTS.md`; Claude reads `CLAUDE.md`; Gemini
7
+ * reads `GEMINI.md`). So `agent()` writes the SAME platform manual under all
8
+ * three names at the agent's working dir, and every harness finds the one it
9
+ * knows. The manual is the single source of truth here; `base-env` bakes a
10
+ * static copy at `/workspace/AGENTS.md` for plain shell sessions, but the
11
+ * per-run copy `agent()` writes is the authoritative one — it carries the
12
+ * live connector list and lands at the run's working dir.
13
+ */
14
+
15
+ import type { SandboxProvider } from "../types/sandbox.js";
16
+
17
+ /**
18
+ * The platform manual delivered to every agent, regardless of harness.
19
+ * Covers the three things an agent must know: where files go (the factory
20
+ * drive + the persist-by-default working dir), how to pause for a human, and
21
+ * that credentials are network-injected (never in the env). The live
22
+ * "Connectors & access" section is appended per-run by `buildAgentContextDoc`.
23
+ */
24
+ export const AGENT_COMPOSE_MANUAL = `# Working inside an Agent Compose sandbox
25
+
26
+ You are an agent running in a per-run sandbox on the Agent Compose platform.
27
+ Use the **\`agentc\` CLI** and the **\`@agent-compose/sdk\`** for everything below —
28
+ do NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on
29
+ your PATH and already authenticated from the environment
30
+ (\`AGENT_COMPOSE_URL\` / \`AGENT_COMPOSE_API_KEY\` / \`AGENT_COMPOSE_FACTORY\` are
31
+ injected for this run), so commands just work — no login, no keys to manage.
32
+
33
+ The \`/ac:*\` skills are installed as Claude Code slash commands (\`/ac:invoke\`,
34
+ \`/ac:events\`, \`/ac:logs\`, \`/ac:register\`, …) — reach for them too.
35
+
36
+ ## Files — your outputs persist by default
37
+
38
+ Your working directory defaults to **\`\$AGENT_COMPOSE_RUN_DIR\`** — a per-run
39
+ directory on the shared factory drive
40
+ (\`\$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/\`) the platform
41
+ creates and attributes to this run. **Files you write here persist by
42
+ default** — they show up in the dashboard's Files tab and the run's Artifacts
43
+ card, with no API calls to save them. The dir already exists and is writable.
44
+
45
+ Need throwaway scratch — heavy build output, package caches, temp files?
46
+ \`cd /tmp\` (or any path outside \`/factory\`): anything off the factory drive is
47
+ ephemeral and discarded when the sandbox ends. In short: **stay in your working
48
+ dir to keep something, \`cd\` out to throw it away.**
49
+
50
+ The whole shared drive is POSIX-mounted at \`/factory\`; the dashboard-visible
51
+ root is \`\$AGENT_COMPOSE_FACTORY_DIR\` (\`/factory/files\`). Earlier versions and
52
+ runs live in sibling dirs under
53
+ \`\$AGENT_COMPOSE_FACTORY_DIR/\$AGENT_COMPOSE_WORKFLOW/\` — read them for prior
54
+ context. Other workflows' dirs are present but not your concern.
55
+
56
+ ## Events — the factory timeline
57
+
58
+ Record something on the run/factory timeline (the dashboard renders these)
59
+ with the CLI — your run id is \`$RUN_ID\`:
60
+
61
+ agentc events send "$RUN_ID" <name> --summary "<one line>" [--body '<json>']
62
+
63
+ Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
64
+ \`agentc events list\` reads them back. \`/ac:events\` is the skill equivalent.
65
+
66
+ ## Runs
67
+
68
+ agentc list # registered workflows (/ac:list)
69
+ agentc logs "$RUN_ID" # a run's logs (/ac:logs)
70
+ agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)
71
+
72
+ ## Writing workflow / agent code — the SDK
73
+
74
+ \`@agent-compose/sdk\` is installed in \`/workspace\` — import it from any script
75
+ you write there:
76
+
77
+ import { defineWorkflow, agent, AgentComposeClient } from "@agent-compose/sdk";
78
+
79
+ Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
80
+ \`agentc register <file.ts>\` (or \`/ac:register\`).
81
+
82
+ ## Pausing to ask the human — \`agentc pause\`
83
+
84
+ When you can't or shouldn't proceed without a human, run \`agentc pause\`, then
85
+ **END YOUR TURN**:
86
+
87
+ agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
+ --option retry --option skip
89
+
90
+ \`agentc pause\` does NOT block and does NOT print the answer. It records your
91
+ question and returns immediately. The moment you end your turn, the run pauses
92
+ (your sandbox is snapshotted and compute stops while the human decides) and the
93
+ human's answer is delivered to you as your **next message** — you pick up
94
+ exactly where you left off, with the answer in hand. So: ask, end your turn,
95
+ and wait. Do NOT keep working, do NOT call more tools, and do NOT mark the task
96
+ complete after pausing.
97
+
98
+ Reach for it the moment you hit — or foresee — any of these:
99
+ - **A wall only a human can clear:** a 401/403, a missing credential, an
100
+ unconnected provider, a host the network refuses. Do NOT retry blindly or try
101
+ to work around it — pause and say what needs enabling.
102
+ - **A durable or outward-facing action that needs sign-off:** registering a
103
+ workflow, deploying, sending email/messages, deleting or overwriting shared
104
+ data, spending money. Prepare everything, then pause for approval BEFORE you
105
+ commit it.
106
+ - **A judgment call only the human can settle:** an under-specified request,
107
+ several valid paths, a conflict with existing state, missing input only they have.
108
+
109
+ You compose the \`--reason\` (the ask) yourself; pass \`--option\` choices when
110
+ there are clear ones, omit them for a free-form answer. Each agent pauses
111
+ independently — pausing doesn't stop the others.
112
+
113
+ ## Credentials
114
+
115
+ Connector credentials (Google, GitHub, …) are NEVER in your environment.
116
+ They're injected at the network layer when you call an allowed host — make the
117
+ request **without** an Authorization header and the platform adds it. Don't try
118
+ to read or exfiltrate tokens; they aren't here. The "Connectors & access"
119
+ section below (when present) lists exactly which providers this run can reach.
120
+
121
+ ## Tools in this environment
122
+
123
+ - \`agentc\` — Agent Compose CLI (your primary interface; authed from env)
124
+ - \`@agent-compose/sdk\` — installed in /workspace for writing workflows
125
+ - \`/ac:*\` Claude Code skills — slash commands for the above
126
+ - \`archil\` (factory drive), \`rtk\`, \`bun\`
127
+ - A world-writable \`/workspace\` working directory`;
128
+
129
+ /**
130
+ * One connector this run can reach, as the agent should see it. Strictly
131
+ * NON-SECRET — hosts, methods, paths, identity only. The access token is
132
+ * injected at the network layer and never appears here. The server builds
133
+ * this list at dispatch from the run's connector grants × the provider
134
+ * catalogue and delivers it as the `AGENT_COMPOSE_CONNECTORS` env (JSON array).
135
+ */
136
+ export interface AgentConnectorInfo {
137
+ /** Provider key (`github`, `notion`, …). */
138
+ provider: string;
139
+ /** Human label ("GitHub", "Notion"). */
140
+ name?: string;
141
+ /** API hosts the credential is injected for. */
142
+ hosts?: string[];
143
+ /** Allowed HTTP methods (Tier-2 narrowing). Empty/absent = any. */
144
+ methods?: string[];
145
+ /** Allowed path prefixes (Tier-2 narrowing). Empty/absent = any. */
146
+ pathPrefixes?: string[];
147
+ /** GitHub: the repository the minted token is scoped to. */
148
+ repository?: string;
149
+ /** Coarse capability the token was minted with. */
150
+ access?: string;
151
+ /** Human scope descriptions, when the provider declares them. */
152
+ scopes?: string[];
153
+ }
154
+
155
+ /** Render the per-run "Connectors & access" markdown section, or "" when the
156
+ * run brokers no connectors. */
157
+ function renderConnectorsSection(connectors: AgentConnectorInfo[]): string {
158
+ if (connectors.length === 0) return "";
159
+ const rows = connectors.map((c) => {
160
+ const host = c.hosts?.length ? c.hosts.join(", ") : "(host set by the platform)";
161
+ const verbs = c.methods?.length ? c.methods.join("/") : "any method";
162
+ const paths = c.pathPrefixes?.length ? ` under ${c.pathPrefixes.join(", ")}` : "";
163
+ const repo = c.repository ? ` — repo \`${c.repository}\` (${c.access ?? "read"})` : "";
164
+ const why = c.scopes?.length ? ` \n _scopes: ${c.scopes.join(", ")}_` : "";
165
+ return `- **${c.name ?? c.provider}** → \`${host}\` — ${verbs}${paths}${repo}${why}`;
166
+ });
167
+ return `
168
+
169
+ ## Connectors & access — what this run can reach
170
+
171
+ These providers are connected for this run. Call their APIs with plain
172
+ fetch/SDKs and **no Authorization header** — the platform injects the
173
+ credential at the network layer. Requests outside the listed method/path are
174
+ refused (403) and the token withheld. Anything NOT listed is unreachable; if
175
+ you need it, \`agentc pause\` and ask for it to be connected.
176
+
177
+ ${rows.join("\n")}
178
+ `;
179
+ }
180
+
181
+ /** Compose the full per-run agent doc: the static manual + the live
182
+ * connectors section read from `AGENT_COMPOSE_CONNECTORS` (a JSON array;
183
+ * malformed/absent → no section). */
184
+ export function buildAgentContextDoc(env: Record<string, string | undefined>): string {
185
+ let connectors: AgentConnectorInfo[] = [];
186
+ const raw = env.AGENT_COMPOSE_CONNECTORS;
187
+ if (raw) {
188
+ try {
189
+ const parsed: unknown = JSON.parse(raw);
190
+ if (Array.isArray(parsed)) connectors = parsed as AgentConnectorInfo[];
191
+ } catch { /* malformed manifest — render the manual without a connectors section */ }
192
+ }
193
+ return AGENT_COMPOSE_MANUAL + renderConnectorsSection(connectors);
194
+ }
195
+
196
+ /**
197
+ * Write the platform context at the agent's working dir under every harness's
198
+ * instruction-file name, so whichever CLI runs finds the one it reads. Codex
199
+ * and Amp read `AGENTS.md` natively; Claude reads `CLAUDE.md`; Gemini reads
200
+ * `GEMINI.md` — we write identical content to all three rather than detect the
201
+ * harness (the runtime's `kind` isn't known until after spawn, and a few extra
202
+ * small files in our own run dir are harmless).
203
+ *
204
+ * Best-effort: a write failure logs and is swallowed — never fail an agent
205
+ * because its context file couldn't be written.
206
+ */
207
+ export async function writeAgentContext(args: {
208
+ sandbox: Pick<SandboxProvider, "files">;
209
+ cwd: string;
210
+ env: Record<string, string | undefined>;
211
+ }): Promise<void> {
212
+ const doc = buildAgentContextDoc(args.env);
213
+ const dir = args.cwd.replace(/\/+$/, "") || "/workspace";
214
+ // AGENTS.md is the cross-harness standard; CLAUDE.md / GEMINI.md are the
215
+ // per-harness names. Same content under each — the harness that doesn't read
216
+ // a given name simply ignores it.
217
+ for (const name of ["AGENTS.md", "CLAUDE.md", "GEMINI.md"]) {
218
+ await args.sandbox.files.write(`${dir}/${name}`, doc);
219
+ }
220
+ }
@@ -9,16 +9,24 @@ import { AgentStatusSchema, parseAgentResponse } from "./protocol.js";
9
9
  import type { AgentStatus, AgentMessage } from "./protocol.js";
10
10
  import { randomUUID } from "node:crypto";
11
11
  import type { Processor, ProcessorContext } from "../processors/processor.js";
12
+ import { boundProcessorPause, type BoundaryPauseFn } from "../pause/pause-core.js";
12
13
  import { runProcessorChain } from "../processors/runner.js";
13
14
  import { RequestContext } from "../request-context/request-context.js";
14
15
  import { PauseManager } from "../pause/manager.js";
15
16
  import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
16
17
  import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
18
+ import { type LocalPauseRequest } from "./local-pause-request.js";
17
19
 
18
20
  export const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
19
21
 
20
22
  const SAME_BLOCKER_ITERATIONS = 3;
21
- const STALL_ITERATIONS = 3;
23
+ // Consecutive turns with NEITHER a <status> NOR a <response> before the loop
24
+ // declares the agent wedged. Kept generous because the most common benign
25
+ // cause is an agent that launched a useful BACKGROUND job and is waiting to be
26
+ // "notified" — the corrective re-prompt (below) tells it to poll synchronously,
27
+ // and these extra turns give that nudge (and the job) time to land before we
28
+ // give up.
29
+ const STALL_ITERATIONS = 6;
22
30
  const MESSAGE_PREVIEW_CHARS = 400;
23
31
 
24
32
  export function parseAgentStatus(text: string): AgentStatus | null {
@@ -47,9 +55,10 @@ export type AgentMessageSummary =
47
55
  | { type: "thinking"; text: string }
48
56
  | { type: "tool_use"; toolName: string; toolUseId: string; toolInput: Record<string, unknown>; toolInputPreview: string }
49
57
  | { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
50
- | { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number }
58
+ | { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number; model?: string }
51
59
  | { type: "done"; sessionId: string }
52
- | { type: "error"; text: string };
60
+ | { type: "error"; text: string }
61
+ | { type: "plan"; entries: { content: string; priority: "high" | "medium" | "low"; status: "pending" | "in_progress" | "completed" }[] };
53
62
 
54
63
  function truncate(value: string): string {
55
64
  return value.length > MESSAGE_PREVIEW_CHARS ? `${value.slice(0, MESSAGE_PREVIEW_CHARS)}…` : value;
@@ -77,6 +86,11 @@ export function summarizeAgentMessage(msg: AgentMessage): AgentMessageSummary {
77
86
  };
78
87
  case "done": return { type: "done", sessionId: msg.sessionId };
79
88
  case "error": return { type: "error", text: truncate(msg.text) };
89
+ // ACP `plan` (WS-C / ADR-0020 Q2). Pure observability — forwarded to
90
+ // onAgentEvent / agent.message; it is NOT an AgentStatus and never feeds
91
+ // self-pause. Pass the entries straight through (the dashboard renders the
92
+ // structured plan; no preview truncation needed — entries are short).
93
+ case "plan": return { type: "plan", entries: msg.entries };
80
94
  }
81
95
  }
82
96
 
@@ -129,19 +143,24 @@ export interface AgentLoopOpts<TResponse = unknown> {
129
143
  * streaming input (ignored otherwise — handled at the runtime).
130
144
  */
131
145
  inbox?: import("./async-queue.js").AsyncQueue<{ text: string; senderName?: string | null }>;
132
- /** PR 7 steer-pause: the boundary calls this to pause the workflow for a
133
- * human steer. agent() builds it (a corePause closed over runId/stepIndex);
134
- * the loop supplies the agentScope so the pauseId is stable across resume.
135
- * Absent ⇒ no steer pause (local tests, non-sandbox callers). */
136
- pause?: <T = unknown>(
137
- req: { reason: string; correlationKey?: string; schema?: z.ZodType<T> },
138
- agentScope: { agentId: string; iteration: number },
139
- ) => Promise<T>;
146
+ /** The run's pause boundary. agent() builds it (a corePause closed over
147
+ * runId/stepIndex); the loop supplies the agentScope so the pauseId is
148
+ * stable across resume. Drives both PR-7 steer-pause AND a processor's
149
+ * `ctx.pause` (human-approval gate). Absent ⇒ no pause (local tests,
150
+ * non-sandbox callers). */
151
+ pause?: BoundaryPauseFn;
140
152
  /** PR 7: take-once read of this agent's pending steer (set by the control
141
153
  * poller). Returns the steer's payload (reason / correlationKey) or null.
142
154
  * The boundary consumes it once per check, and only when no steer is
143
155
  * already staged. */
144
156
  consumeSteerPending?: () => SteerPayload | null;
157
+ /** Take-once read of a local `agentc pause` request this agent dropped in the
158
+ * state dir during its turn (the CLI writes it; see local-pause-request.ts).
159
+ * Returns the request (reason + offered options) or null. The loop turns it
160
+ * into a staged self-pause the next boundary takes — the same snapshot-release
161
+ * path as `needs_input`, but triggered by an explicit `agentc pause` call so
162
+ * it is honored regardless of `mode`. */
163
+ consumeLocalPauseRequest?: () => LocalPauseRequest | null;
145
164
  /** PR 7: `auto` (autonomous, default) or `hitl` (can be steer-paused). The
146
165
  * boundary is inert unless `hitl`. */
147
166
  mode?: "auto" | "hitl";
@@ -174,6 +193,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
174
193
  retryCount: 0,
175
194
  agentId,
176
195
  iteration,
196
+ pause: boundProcessorPause(opts.pause, { agentId, iteration }),
177
197
  });
178
198
 
179
199
  if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
@@ -185,6 +205,10 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
185
205
  processors,
186
206
  requestContext,
187
207
  agentId,
208
+ // Thread the pause boundary so a runtime-driven pre-tool gate (e.g. the ACP
209
+ // `session/request_permission` path through CliAgentRunner.gateToolCall) can
210
+ // raise a human-approval `ctx.pause`, not just on the loop's own hooks.
211
+ ...(opts.pause ? { pause: opts.pause } : {}),
188
212
  ...(opts.responseSchema ? { outputFormat: { type: "json_schema" as const, schema: z.toJSONSchema(opts.responseSchema) as Record<string, unknown> } } : {}),
189
213
  });
190
214
 
@@ -216,7 +240,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
216
240
  let resumed = false;
217
241
  // PR 7: a pending human steer-pause intent. Persisted in loop state so it
218
242
  // survives a resume and the boundary re-issues the pause on re-entry.
219
- let pendingSteerPause: { reason: string; correlationKey: string | null; at: number } | null = null;
243
+ let pendingSteerPause: { reason: string; correlationKey: string | null; at: number; payload?: Record<string, unknown> } | null = null;
220
244
 
221
245
  const pauseManager = new PauseManager(agentId);
222
246
  const restore = await pauseManager.restoreAgentLoop<TResponse>(client);
@@ -336,6 +360,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
336
360
  {
337
361
  reason: pendingSteerPause.reason,
338
362
  ...(pendingSteerPause.correlationKey !== null ? { correlationKey: pendingSteerPause.correlationKey } : {}),
363
+ ...(pendingSteerPause.payload !== undefined ? { payload: pendingSteerPause.payload } : {}),
339
364
  schema: SteerDecisionSchema,
340
365
  },
341
366
  { agentId, iteration },
@@ -397,6 +422,11 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
397
422
  // (otherwise it restarts the task in a fresh session and repeats side effects).
398
423
  sessionId: lastSessionId ?? undefined,
399
424
  iteration: iteration + 1,
425
+ // Thread the loop's abort signal so a runtime that owns a cancellable
426
+ // transport (the ACP path's `session/cancel`, ADR-0020) actually unwinds
427
+ // when the loop aborts — input/output processor `abort`, or any external
428
+ // cancel. Without this the cancel/session-cancel wiring is dead.
429
+ signal: loopAbort.signal,
400
430
  ...(opts.inbox ? { inboxStream: opts.inbox } : {}),
401
431
  })) {
402
432
  // processOutput chain — deny drops the message from accumulation;
@@ -413,7 +443,13 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
413
443
  }
414
444
  const msg = outputVerdict.value;
415
445
  opts.onAgentEvent?.(iteration, msg);
416
- opts.onAgentLifecycleEvent?.({ event: "agent.message", at: Date.now(), agentId, label, iteration: iteration + 1, message: summarizeAgentMessage(msg) });
446
+ // Usage summaries carry the resolved model so the server can price
447
+ // token rows per model without correlating back to agent.spawned.
448
+ const summary = summarizeAgentMessage(msg);
449
+ opts.onAgentLifecycleEvent?.({
450
+ event: "agent.message", at: Date.now(), agentId, label, iteration: iteration + 1,
451
+ message: summary.type === "usage" && client.model != null ? { ...summary, model: client.model } : summary,
452
+ });
417
453
  if (msg.type === "init") lastSessionId = msg.sessionId;
418
454
  if (msg.type === "text") responseText += msg.text;
419
455
  if (msg.type === "error") throw new Error(`Agent error: ${msg.text}`);
@@ -425,6 +461,31 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
425
461
  lastResponseText = responseText;
426
462
 
427
463
  let status = parseAgentStatus(responseText);
464
+
465
+ // `agentc pause` — the agent shelled out to the CLI during this turn, which
466
+ // dropped a durable pause-request marker in the state dir. Honor it like a
467
+ // self-pause: stage the pause the NEXT boundary takes, carrying the agent's
468
+ // question as the reason and any offered choices as the pause payload (the
469
+ // dashboard renders them as buttons). Checked BEFORE the settle / needs_input
470
+ // paths and independent of the <status> block — the common case is an agent
471
+ // that called the tool and ended its turn with no status at all, which would
472
+ // otherwise fall through to the empty-output re-prompt below. UNGATED by
473
+ // `mode`: an explicit `agentc pause` is a deliberate ask, not the `needs_input`
474
+ // heuristic that only `hitl` agents may trigger.
475
+ if (opts.pause && pendingSteerPause === null) {
476
+ const localPause = opts.consumeLocalPauseRequest?.() ?? null;
477
+ if (localPause) {
478
+ pendingSteerPause = {
479
+ reason: localPause.reason,
480
+ correlationKey: null,
481
+ ...(localPause.options && localPause.options.length > 0 ? { payload: { options: localPause.options } } : {}),
482
+ at: Date.now(),
483
+ };
484
+ blockerStreak = null; // an explicit ask is not a stuck loop
485
+ continue; // pause fires at the next boundary
486
+ }
487
+ }
488
+
428
489
  // Inline safeParse (instead of letting parseAgentResponse validate)
429
490
  // so a schema failure surfaces via lastResponseValidationError on
430
491
  // the next iteration — the model needs that feedback to fix its
@@ -511,15 +572,22 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
511
572
  if (!status && rawResponse === null) {
512
573
  if (++iterationsWithoutStatus >= STALL_ITERATIONS)
513
574
  throw new Error(`${logLabel} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
514
- // EMPTY-OUTPUT RE-PROMPT: under the unbudgeted default (maxIterations 1)
515
- // a single turn that emits neither <status> nor <response> would
516
- // otherwise exhaust the budget with ZERO corrective feedback. Treat it
517
- // like the contract-violation branches — refund the iteration and
518
- // re-prompt with explicit feedback, bounded by the shared retry pool.
519
- // The stall counter above still hard-bounds consecutive empty turns.
520
- if (opts.responseSchema && schemaRetriesLeft-- > 0) {
521
- lastResponseValidationError = "no <response> block found — the turn ended with neither a <status> nor a <response> block; emit the complete <response> JSON";
522
- process.stdout.write(`${logLabel} no <status>/<response> emitted — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
575
+ // EMPTY-OUTPUT RE-PROMPT: a turn that emits neither <status> nor
576
+ // <response>. Refund the iteration and re-prompt with explicit feedback
577
+ // (bounded by the shared retry pool); the stall counter above still
578
+ // hard-bounds genuinely wedged agents. The directive call-out about
579
+ // BACKGROUND jobs is load-bearing: the dominant benign cause is an agent
580
+ // that ran `cmd &` and parked itself "waiting to be notified" — the loop
581
+ // delivers no such notification, so it must poll synchronously instead.
582
+ // Fires with or without a responseSchema (an agent with no schema still
583
+ // owes a <status>).
584
+ if (schemaRetriesLeft-- > 0) {
585
+ lastResponseValidationError =
586
+ "Your turn ended with no <status> block" + (opts.responseSchema ? " (and no <response> block)" : "") + ". " +
587
+ "If you launched a background job (`… &`) and are waiting to be notified when it finishes — STOP: nothing will notify you here. " +
588
+ "Poll it NOW (read its output / wait for it synchronously to completion), then emit your <status>" +
589
+ (opts.responseSchema ? " and the complete <response> JSON." : ".");
590
+ process.stdout.write(`${logLabel} empty turn (no <status>) — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
523
591
  iteration--;
524
592
  continue;
525
593
  }
@@ -0,0 +1,90 @@
1
+ /**
2
+ * Local pause-request marker — the in-sandbox bridge from `agentc pause` to
3
+ * the agent loop's snapshot-release pause boundary.
4
+ *
5
+ * `agentc pause` runs as a grandchild subprocess of the runner (the agent CLI
6
+ * shells out to it). It cannot throw a `PauseSignal` into the loop and the
7
+ * in-memory steer flag (`signalSteerPending`) lives in a different process, so
8
+ * the only reliable channel is the shared sandbox filesystem. The CLI writes a
9
+ * durable marker here; the agent loop consumes it at the end of the turn
10
+ * (alongside the `needs_input` self-pause) and stages a real `ctx.pause` the
11
+ * next boundary takes — which snapshots the sandbox, releases the activity
12
+ * (compute stops), and parks the workflow. On resume the human's answer is
13
+ * delivered as the agent's next user turn.
14
+ *
15
+ * This is the ONLY place the marker path + shape are defined — both the CLI
16
+ * (writer) and the SDK loop (reader) import it, so the two halves can never
17
+ * drift. Like the rest of the state-dir, the layout is a wire protocol between
18
+ * the runner and the next subprocess invocation (ADR-0006). The marker is
19
+ * consumed BEFORE the snapshot, so it never needs to survive a pause.
20
+ *
21
+ * Scoped by `agentId` so concurrent `agent()` calls sharing one sandbox each
22
+ * see only their own request. When the id is unavailable (older agent env that
23
+ * doesn't inject `AGENT_COMPOSE_AGENT_ID`) both sides fall back to a single
24
+ * unscoped slot — correct for the common single-agent step, and the only case
25
+ * where an unscoped marker can be ambiguous (two anonymous agents) is one the
26
+ * old block-poll CLI couldn't handle either.
27
+ */
28
+
29
+ import { existsSync, mkdirSync, readFileSync, renameSync, rmSync, writeFileSync } from "node:fs";
30
+ import { join } from "node:path";
31
+ import { randomBytes } from "node:crypto";
32
+
33
+ import { getStateDir } from "../pause/state-dir.js";
34
+
35
+ /** A choice offered to the human. A bare string is shorthand for
36
+ * `{ label, value }` with both equal — exactly what the dashboard's
37
+ * `readOptions` accepts. */
38
+ export type PauseOption = string | { label: string; value: string };
39
+
40
+ /** What `agentc pause` records for the loop to turn into a `ctx.pause`. */
41
+ export interface LocalPauseRequest {
42
+ /** The question shown to the human (becomes the pause `reason`). */
43
+ reason: string;
44
+ /** Optional offered choices, rendered as buttons in the dashboard. */
45
+ options?: PauseOption[];
46
+ }
47
+
48
+ const UNSCOPED = "_unscoped";
49
+
50
+ const requestsDir = () => join(getStateDir(), "pause-requests");
51
+ const markerPath = (agentId: string | undefined | null) =>
52
+ join(requestsDir(), `${slug(agentId) || UNSCOPED}.json`);
53
+
54
+ /** Keep the agentId filename-safe. Agent ids are `step<idx>-agent-<n>` shaped,
55
+ * but defend against anything exotic so the marker can never escape the dir. */
56
+ function slug(agentId: string | undefined | null): string {
57
+ return (agentId ?? "").replace(/[^a-zA-Z0-9_.-]/g, "_");
58
+ }
59
+
60
+ /** Write the marker atomically (tmp → rename) so the loop never reads a
61
+ * partial file mid-write. Called by `agentc pause`. */
62
+ export function writeLocalPauseRequest(agentId: string | undefined | null, req: LocalPauseRequest): void {
63
+ mkdirSync(requestsDir(), { recursive: true });
64
+ const final = markerPath(agentId);
65
+ const tmp = `${final}.${randomBytes(6).toString("hex")}.tmp`;
66
+ writeFileSync(tmp, JSON.stringify(req), "utf8");
67
+ renameSync(tmp, final);
68
+ }
69
+
70
+ /** Take-once read: returns + deletes this agent's pending pause request, else
71
+ * null. Checks the agent-scoped slot first, then the unscoped fallback. The
72
+ * loop calls this once per turn; a malformed marker is dropped (deleted +
73
+ * null) rather than wedging the loop. */
74
+ export function consumeLocalPauseRequest(agentId: string | undefined | null): LocalPauseRequest | null {
75
+ const paths = [...new Set([markerPath(agentId), markerPath(null)])]; // dedupe when agentId is absent
76
+ for (const path of paths) {
77
+ if (!existsSync(path)) continue;
78
+ try {
79
+ const raw = JSON.parse(readFileSync(path, "utf8")) as unknown;
80
+ rmSync(path, { force: true });
81
+ if (raw && typeof raw === "object" && typeof (raw as LocalPauseRequest).reason === "string" && (raw as LocalPauseRequest).reason.trim().length > 0) {
82
+ const r = raw as LocalPauseRequest;
83
+ return { reason: r.reason.trim(), ...(Array.isArray(r.options) && r.options.length > 0 ? { options: r.options } : {}) };
84
+ }
85
+ } catch {
86
+ rmSync(path, { force: true }); // unreadable / partial → drop it
87
+ }
88
+ }
89
+ return null;
90
+ }
@@ -13,14 +13,16 @@
13
13
  import { z } from "zod";
14
14
  import { randomUUID } from "node:crypto";
15
15
  import { getActiveStep, nextAgentCallInActiveStep } from "../active-step.js";
16
- import { corePause } from "../pause/pause-core.js";
16
+ import { corePause, type PauseRequest } from "../pause/pause-core.js";
17
17
  import { agentLoop } from "./agent-loop.js";
18
18
  import { consumeSteerPending, runControlPoller } from "./steer-control.js";
19
+ import { consumeLocalPauseRequest } from "./local-pause-request.js";
19
20
  import type { AgentLifecycleEvent, AgentLoopResult } from "./agent-loop.js";
20
21
  import { AsyncQueue } from "./async-queue.js";
21
22
  import type { AgentMessage, AgentStatus } from "../types/protocol.js";
22
23
  import type { AgentRuntime, RuntimeOptions } from "../types/runtime.js";
23
24
  import type { SandboxProvider } from "../types/sandbox.js";
25
+ import { writeAgentContext } from "./agent-context.js";
24
26
  import type { AgentBudget } from "../types/workflow.js";
25
27
  import type { Processor } from "../processors/processor.js";
26
28
  import { RequestContext } from "../request-context/request-context.js";
@@ -232,7 +234,42 @@ export function resolveAgentId(explicitId?: string): string {
232
234
  }
233
235
 
234
236
  export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopResult<T>> {
235
- const workingDir = opts.workingDir ?? "";
237
+ // Persist-by-default: an agent's working dir is the run dir on the factory
238
+ // drive (the mount step creates it ahead of the run), so everything it
239
+ // writes is kept and attributed to this run; `cd /tmp` for throwaway
240
+ // scratch. Falls back to /workspace when this run has no factory drive.
241
+ // Never "" — an empty cwd is the one case the harnesses' instruction-file
242
+ // upward-walk can't resolve, and it left codex/amp without their AGENTS.md.
243
+ let workingDir = opts.workingDir || process.env.AGENT_COMPOSE_RUN_DIR || "/workspace";
244
+
245
+ // Deliver the platform context (file conventions, connectors & access, how to
246
+ // pause) as AGENTS.md / CLAUDE.md / GEMINI.md at the working dir — and use the
247
+ // write as a LIVENESS PROBE of the working dir. AGENT_COMPOSE_RUN_DIR points at
248
+ // the /factory FUSE drive, whose writes HANG (uninterruptible, no timeout) when
249
+ // the mount degraded — launching the agent there wedges it silently with zero
250
+ // output (no logs). Bound the write; on timeout/failure fall back to /workspace
251
+ // (always present + writable) so a degraded drive can never sink an agent.
252
+ const CONTEXT_WRITE_DEADLINE_MS = 15_000;
253
+ const tryWriteContext = (cwd: string): Promise<boolean> => {
254
+ // Capture the deadline timer so the fast path (write resolves first) can
255
+ // clear it in finally — otherwise each probe leaks a dangling 15s timer.
256
+ // Mirrors sandbox.ts's clear-in-finally pattern.
257
+ let timer: ReturnType<typeof setTimeout> | undefined;
258
+ return Promise.race([
259
+ writeAgentContext({ sandbox: opts.sandbox, cwd, env: process.env }).then(() => true),
260
+ new Promise<boolean>((resolve) => { timer = setTimeout(() => resolve(false), CONTEXT_WRITE_DEADLINE_MS); }),
261
+ ]).catch((err: unknown) => {
262
+ console.error(`[agent] writeAgentContext(${cwd}) failed: ${err instanceof Error ? err.message : String(err)}`);
263
+ return false;
264
+ }).finally(() => { if (timer) clearTimeout(timer); });
265
+ };
266
+ if (!(await tryWriteContext(workingDir)) && workingDir !== "/workspace") {
267
+ console.error(
268
+ `[agent] working dir ${workingDir} not writable within ${CONTEXT_WRITE_DEADLINE_MS}ms ` +
269
+ `(factory drive degraded?) — falling back to /workspace so the agent can run`);
270
+ workingDir = "/workspace";
271
+ await tryWriteContext(workingDir);
272
+ }
236
273
 
237
274
  // Stable agentId for this invocation. Used by the inbox URL (server
238
275
  // scopes pending messages by agentId), the agentLoop (which would
@@ -279,7 +316,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
279
316
  const activeStepForPause = getActiveStep();
280
317
  const steerPause = activeStepForPause && runId
281
318
  ? function steerPauseFn<T>(
282
- req: { reason: string; correlationKey?: string; schema?: z.ZodType<T> },
319
+ req: PauseRequest<T>,
283
320
  agentScope: { agentId: string; iteration: number },
284
321
  ): Promise<T> {
285
322
  return corePause<T>(req, {
@@ -329,6 +366,9 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
329
366
  // PR 7 steer-pause wiring.
330
367
  mode,
331
368
  consumeSteerPending: () => consumeSteerPending(agentId),
369
+ // `agentc pause` self-pause: the loop reads the marker this agent's CLI
370
+ // dropped in the state dir and stages a snapshot-release pause from it.
371
+ consumeLocalPauseRequest: () => consumeLocalPauseRequest(agentId),
332
372
  ...(steerPause ? { pause: steerPause } : {}),
333
373
  });
334
374
  } finally {
@@ -17,13 +17,27 @@
17
17
  import { z } from "zod";
18
18
 
19
19
  /** What a steer/resume answer carries. Used as the boundary pause's `schema`
20
- * so it is validated client-side in the re-spawned runner — a null/empty
21
- * message is rejected (PauseSchemaError) so an agent never resumes on an
22
- * empty steer. `message` becomes the agent's next user turn. */
23
- export const SteerDecisionSchema = z.object({
24
- message: z.string().min(1),
25
- actor: z.string().nullish(),
26
- });
20
+ * so it is validated client-side in the re-spawned runner — an empty answer
21
+ * is rejected (PauseSchemaError) so an agent never resumes on nothing. The
22
+ * normalised `message` becomes the agent's next user turn.
23
+ *
24
+ * Accepts BOTH resume-payload conventions and normalises to `{ message }`:
25
+ * - `{ message }` — the SDK / API steer convention (`answerSteer`).
26
+ * - `{ decision }` — what the dashboard's RunPausePanel universally sends
27
+ * for every pause (option click or free text). Without this, resuming an
28
+ * agent steer / `needs_input` / `agentc pause` pause from the dashboard
29
+ * failed schema validation in the re-spawned runner and the agent never
30
+ * got the answer — the resume looked like it did nothing. */
31
+ export const SteerDecisionSchema = z
32
+ .object({
33
+ message: z.string().min(1).optional(),
34
+ decision: z.string().min(1).optional(),
35
+ actor: z.string().nullish(),
36
+ })
37
+ .transform((d) => ({ message: (d.message ?? d.decision ?? "").trim(), actor: d.actor ?? null }))
38
+ .refine((d) => d.message.length > 0, {
39
+ message: "a steer/resume answer needs a non-empty `message` or `decision`",
40
+ });
27
41
  export type SteerDecision = z.infer<typeof SteerDecisionSchema>;
28
42
 
29
43
  /** Metadata a pending steer carries from the control message to the boundary