@agent-compose/sdk 0.5.6 → 0.5.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +4 -4
  2. package/dist/agent/__tests__/run-agent-liveness.test.d.ts +17 -0
  3. package/dist/agent/agent-context.d.ts +67 -0
  4. package/dist/agent/agent-loop-contract.test.d.ts +1 -0
  5. package/dist/agent/agent-loop.d.ts +2 -1
  6. package/dist/client.d.ts +129 -24
  7. package/dist/index.d.ts +9 -4
  8. package/dist/index.js +553 -89
  9. package/dist/pause/wrappers.d.ts +7 -11
  10. package/dist/runtimes/claude.d.ts +9 -1
  11. package/dist/runtimes/openai-desktop.js +548 -89
  12. package/dist/sandbox-errors.d.ts +49 -0
  13. package/dist/sandbox.d.ts +92 -13
  14. package/dist/step-invocation/protocol.d.ts +6 -0
  15. package/dist/step-invocation/types.d.ts +1 -1
  16. package/dist/types/execution-context.d.ts +1 -3
  17. package/dist/types/sandbox-environment.d.ts +1 -10
  18. package/dist/types/sandbox.d.ts +27 -3
  19. package/dist/types/workflow-metadata.d.ts +81 -13
  20. package/dist/types/workflow.d.ts +45 -10
  21. package/dist/utils/bundler.d.ts +40 -9
  22. package/dist/workflow-steps/workflow.d.ts +4 -3
  23. package/package.json +2 -2
  24. package/src/agent/agent-context.ts +212 -0
  25. package/src/agent/agent-loop.ts +78 -10
  26. package/src/agent/run-agent.ts +37 -1
  27. package/src/client.ts +232 -26
  28. package/src/index.ts +9 -4
  29. package/src/pause/wrappers.ts +7 -21
  30. package/src/runtimes/claude.ts +66 -8
  31. package/src/sandbox-errors.ts +53 -0
  32. package/src/sandbox.ts +438 -61
  33. package/src/step-invocation/invoker.ts +66 -8
  34. package/src/step-invocation/protocol.ts +9 -0
  35. package/src/step-invocation/server.ts +27 -4
  36. package/src/step-invocation/types.ts +1 -1
  37. package/src/types/execution-context.ts +1 -3
  38. package/src/types/sandbox-environment.ts +1 -11
  39. package/src/types/sandbox.ts +28 -3
  40. package/src/types/workflow-metadata.ts +91 -16
  41. package/src/types/workflow.ts +45 -12
  42. package/src/utils/bundler.ts +46 -13
  43. package/src/workflow-steps/workflow.ts +4 -3
  44. package/src/workflows/invoke-child.ts +7 -1
@@ -0,0 +1,212 @@
1
+ /**
2
+ * Harness-agnostic agent context delivery.
3
+ *
4
+ * Every coding-agent harness we drive (Claude Code, Codex, Amp, Gemini, …)
5
+ * looks for an instruction file in its working directory — but they disagree
6
+ * on the NAME (Codex/Amp read `AGENTS.md`; Claude reads `CLAUDE.md`; Gemini
7
+ * reads `GEMINI.md`). So `agent()` writes the SAME platform manual under all
8
+ * three names at the agent's working dir, and every harness finds the one it
9
+ * knows. The manual is the single source of truth here; `base-env` bakes a
10
+ * static copy at `/workspace/AGENTS.md` for plain shell sessions, but the
11
+ * per-run copy `agent()` writes is the authoritative one — it carries the
12
+ * live connector list and lands at the run's working dir.
13
+ */
14
+
15
+ import type { SandboxProvider } from "../types/sandbox.js";
16
+
17
+ /**
18
+ * The platform manual delivered to every agent, regardless of harness.
19
+ * Covers the three things an agent must know: where files go (the factory
20
+ * drive + the persist-by-default working dir), how to pause for a human, and
21
+ * that credentials are network-injected (never in the env). The live
22
+ * "Connectors & access" section is appended per-run by `buildAgentContextDoc`.
23
+ */
24
+ export const AGENT_COMPOSE_MANUAL = `# Working inside an Agent Compose sandbox
25
+
26
+ You are an agent running in a per-run sandbox on the Agent Compose platform.
27
+ Use the **\`agentc\` CLI** and the **\`@agent-compose/sdk\`** for everything below —
28
+ do NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on
29
+ your PATH and already authenticated from the environment
30
+ (\`AGENT_COMPOSE_URL\` / \`AGENT_COMPOSE_API_KEY\` / \`AGENT_COMPOSE_FACTORY\` are
31
+ injected for this run), so commands just work — no login, no keys to manage.
32
+
33
+ The \`/ac:*\` skills are installed as Claude Code slash commands (\`/ac:invoke\`,
34
+ \`/ac:events\`, \`/ac:logs\`, \`/ac:register\`, …) — reach for them too.
35
+
36
+ ## Files — your outputs persist by default
37
+
38
+ Your working directory defaults to **\`\$AGENT_COMPOSE_RUN_DIR\`** — a per-run
39
+ directory on the shared factory drive
40
+ (\`\$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/\`) the platform
41
+ creates and attributes to this run. **Files you write here persist by
42
+ default** — they show up in the dashboard's Files tab and the run's Artifacts
43
+ card, with no API calls to save them. The dir already exists and is writable.
44
+
45
+ Need throwaway scratch — heavy build output, package caches, temp files?
46
+ \`cd /tmp\` (or any path outside \`/factory\`): anything off the factory drive is
47
+ ephemeral and discarded when the sandbox ends. In short: **stay in your working
48
+ dir to keep something, \`cd\` out to throw it away.**
49
+
50
+ The whole shared drive is POSIX-mounted at \`/factory\`; the dashboard-visible
51
+ root is \`\$AGENT_COMPOSE_FACTORY_DIR\` (\`/factory/files\`). Earlier versions and
52
+ runs live in sibling dirs under
53
+ \`\$AGENT_COMPOSE_FACTORY_DIR/\$AGENT_COMPOSE_WORKFLOW/\` — read them for prior
54
+ context. Other workflows' dirs are present but not your concern.
55
+
56
+ ## Events — the factory timeline
57
+
58
+ Record something on the run/factory timeline (the dashboard renders these)
59
+ with the CLI — your run id is \`$RUN_ID\`:
60
+
61
+ agentc events send "$RUN_ID" <name> --summary "<one line>" [--body '<json>']
62
+
63
+ Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
64
+ \`agentc events list\` reads them back. \`/ac:events\` is the skill equivalent.
65
+
66
+ ## Runs
67
+
68
+ agentc list # registered workflows (/ac:list)
69
+ agentc logs "$RUN_ID" # a run's logs (/ac:logs)
70
+ agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)
71
+
72
+ ## Writing workflow / agent code — the SDK
73
+
74
+ \`@agent-compose/sdk\` is installed in \`/workspace\` — import it from any script
75
+ you write there:
76
+
77
+ import { defineWorkflow, agent, AgentComposeClient } from "@agent-compose/sdk";
78
+
79
+ Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
80
+ \`agentc register <file.ts>\` (or \`/ac:register\`).
81
+
82
+ ## Pausing to ask the human — \`agentc pause\`
83
+
84
+ When you can't or shouldn't proceed without a human, run \`agentc pause\`. It
85
+ blocks until they answer on the dashboard, then prints their answer to stdout:
86
+
87
+ ANSWER=$(agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
88
+ --option retry --option skip)
89
+
90
+ Reach for it the moment you hit — or foresee — any of these:
91
+ - **A wall only a human can clear:** a 401/403, a missing credential, an
92
+ unconnected provider, a host the network refuses. Do NOT retry blindly or try
93
+ to work around it — pause and say what needs enabling.
94
+ - **A durable or outward-facing action that needs sign-off:** registering a
95
+ workflow, deploying, sending email/messages, deleting or overwriting shared
96
+ data, spending money. Prepare everything, then pause for approval BEFORE you
97
+ commit it.
98
+ - **A judgment call only the human can settle:** an under-specified request,
99
+ several valid paths, a conflict with existing state, missing input only they have.
100
+
101
+ You compose the \`--reason\` (the ask) yourself; pass \`--option\` choices when
102
+ there are clear ones, omit them for a free-form answer. Read the printed answer
103
+ and act on it. Each agent pauses independently — pausing doesn't stop the others.
104
+
105
+ ## Credentials
106
+
107
+ Connector credentials (Google, GitHub, …) are NEVER in your environment.
108
+ They're injected at the network layer when you call an allowed host — make the
109
+ request **without** an Authorization header and the platform adds it. Don't try
110
+ to read or exfiltrate tokens; they aren't here. The "Connectors & access"
111
+ section below (when present) lists exactly which providers this run can reach.
112
+
113
+ ## Tools in this environment
114
+
115
+ - \`agentc\` — Agent Compose CLI (your primary interface; authed from env)
116
+ - \`@agent-compose/sdk\` — installed in /workspace for writing workflows
117
+ - \`/ac:*\` Claude Code skills — slash commands for the above
118
+ - \`archil\` (factory drive), \`rtk\`, \`bun\`
119
+ - A world-writable \`/workspace\` working directory`;
120
+
121
+ /**
122
+ * One connector this run can reach, as the agent should see it. Strictly
123
+ * NON-SECRET — hosts, methods, paths, identity only. The access token is
124
+ * injected at the network layer and never appears here. The server builds
125
+ * this list at dispatch from the run's connector grants × the provider
126
+ * catalogue and delivers it as the `AGENT_COMPOSE_CONNECTORS` env (JSON array).
127
+ */
128
+ export interface AgentConnectorInfo {
129
+ /** Provider key (`github`, `notion`, …). */
130
+ provider: string;
131
+ /** Human label ("GitHub", "Notion"). */
132
+ name?: string;
133
+ /** API hosts the credential is injected for. */
134
+ hosts?: string[];
135
+ /** Allowed HTTP methods (Tier-2 narrowing). Empty/absent = any. */
136
+ methods?: string[];
137
+ /** Allowed path prefixes (Tier-2 narrowing). Empty/absent = any. */
138
+ pathPrefixes?: string[];
139
+ /** GitHub: the repository the minted token is scoped to. */
140
+ repository?: string;
141
+ /** Coarse capability the token was minted with. */
142
+ access?: string;
143
+ /** Human scope descriptions, when the provider declares them. */
144
+ scopes?: string[];
145
+ }
146
+
147
+ /** Render the per-run "Connectors & access" markdown section, or "" when the
148
+ * run brokers no connectors. */
149
+ function renderConnectorsSection(connectors: AgentConnectorInfo[]): string {
150
+ if (connectors.length === 0) return "";
151
+ const rows = connectors.map((c) => {
152
+ const host = c.hosts?.length ? c.hosts.join(", ") : "(host set by the platform)";
153
+ const verbs = c.methods?.length ? c.methods.join("/") : "any method";
154
+ const paths = c.pathPrefixes?.length ? ` under ${c.pathPrefixes.join(", ")}` : "";
155
+ const repo = c.repository ? ` — repo \`${c.repository}\` (${c.access ?? "read"})` : "";
156
+ const why = c.scopes?.length ? ` \n _scopes: ${c.scopes.join(", ")}_` : "";
157
+ return `- **${c.name ?? c.provider}** → \`${host}\` — ${verbs}${paths}${repo}${why}`;
158
+ });
159
+ return `
160
+
161
+ ## Connectors & access — what this run can reach
162
+
163
+ These providers are connected for this run. Call their APIs with plain
164
+ fetch/SDKs and **no Authorization header** — the platform injects the
165
+ credential at the network layer. Requests outside the listed method/path are
166
+ refused (403) and the token withheld. Anything NOT listed is unreachable; if
167
+ you need it, \`agentc pause\` and ask for it to be connected.
168
+
169
+ ${rows.join("\n")}
170
+ `;
171
+ }
172
+
173
+ /** Compose the full per-run agent doc: the static manual + the live
174
+ * connectors section read from `AGENT_COMPOSE_CONNECTORS` (a JSON array;
175
+ * malformed/absent → no section). */
176
+ export function buildAgentContextDoc(env: Record<string, string | undefined>): string {
177
+ let connectors: AgentConnectorInfo[] = [];
178
+ const raw = env.AGENT_COMPOSE_CONNECTORS;
179
+ if (raw) {
180
+ try {
181
+ const parsed: unknown = JSON.parse(raw);
182
+ if (Array.isArray(parsed)) connectors = parsed as AgentConnectorInfo[];
183
+ } catch { /* malformed manifest — render the manual without a connectors section */ }
184
+ }
185
+ return AGENT_COMPOSE_MANUAL + renderConnectorsSection(connectors);
186
+ }
187
+
188
+ /**
189
+ * Write the platform context at the agent's working dir under every harness's
190
+ * instruction-file name, so whichever CLI runs finds the one it reads. Codex
191
+ * and Amp read `AGENTS.md` natively; Claude reads `CLAUDE.md`; Gemini reads
192
+ * `GEMINI.md` — we write identical content to all three rather than detect the
193
+ * harness (the runtime's `kind` isn't known until after spawn, and a few extra
194
+ * small files in our own run dir are harmless).
195
+ *
196
+ * Best-effort: a write failure logs and is swallowed — never fail an agent
197
+ * because its context file couldn't be written.
198
+ */
199
+ export async function writeAgentContext(args: {
200
+ sandbox: Pick<SandboxProvider, "files">;
201
+ cwd: string;
202
+ env: Record<string, string | undefined>;
203
+ }): Promise<void> {
204
+ const doc = buildAgentContextDoc(args.env);
205
+ const dir = args.cwd.replace(/\/+$/, "") || "/workspace";
206
+ // AGENTS.md is the cross-harness standard; CLAUDE.md / GEMINI.md are the
207
+ // per-harness names. Same content under each — the harness that doesn't read
208
+ // a given name simply ignores it.
209
+ for (const name of ["AGENTS.md", "CLAUDE.md", "GEMINI.md"]) {
210
+ await args.sandbox.files.write(`${dir}/${name}`, doc);
211
+ }
212
+ }
@@ -15,10 +15,16 @@ import { PauseManager } from "../pause/manager.js";
15
15
  import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
16
16
  import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
17
17
 
18
- export const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
18
+ export const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
19
19
 
20
20
  const SAME_BLOCKER_ITERATIONS = 3;
21
- const STALL_ITERATIONS = 3;
21
+ // Consecutive turns with NEITHER a <status> NOR a <response> before the loop
22
+ // declares the agent wedged. Kept generous because the most common benign
23
+ // cause is an agent that launched a useful BACKGROUND job and is waiting to be
24
+ // "notified" — the corrective re-prompt (below) tells it to poll synchronously,
25
+ // and these extra turns give that nudge (and the job) time to land before we
26
+ // give up.
27
+ const STALL_ITERATIONS = 6;
22
28
  const MESSAGE_PREVIEW_CHARS = 400;
23
29
 
24
30
  export function parseAgentStatus(text: string): AgentStatus | null {
@@ -47,7 +53,7 @@ export type AgentMessageSummary =
47
53
  | { type: "thinking"; text: string }
48
54
  | { type: "tool_use"; toolName: string; toolUseId: string; toolInput: Record<string, unknown>; toolInputPreview: string }
49
55
  | { type: "tool_result"; toolUseId: string; output: string; isError: boolean }
50
- | { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number }
56
+ | { type: "usage"; inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheCreationTokens: number; durationMs: number; numTurns: number; model?: string }
51
57
  | { type: "done"; sessionId: string }
52
58
  | { type: "error"; text: string };
53
59
 
@@ -152,8 +158,15 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
152
158
  const label = opts.label ?? "agent";
153
159
  const logLabel = opts.label ?? "[Agent Loop]";
154
160
  const startedAt = Date.now();
155
- const turnsPerIteration = opts.turnsPerIteration ?? 40;
156
- const maxIterations = opts.maxIterations ?? 8;
161
+ // No budget ⇒ no turn cap: the harness runtime (Claude Code) decides when it's
162
+ // done. A numeric budget is an explicit caller choice, not a default we impose.
163
+ const turnsPerIteration = opts.turnsPerIteration;
164
+ const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 1 : 8);
165
+ // A responseSchema is a CONTRACT, not a hope: when the agent's <response> fails
166
+ // validation, the loop re-prompts with the exact errors until it conforms —
167
+ // without consuming the caller's iteration budget. The backstop below only
168
+ // guards against a truly wedged agent (never reached in normal operation).
169
+ let schemaRetriesLeft = 10;
157
170
  const processors = opts.processors ?? [];
158
171
  const requestContext = opts.requestContext ?? RequestContext.fromReserved({
159
172
  teamId: "", runId: "", workflowId: "",
@@ -171,7 +184,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
171
184
 
172
185
  if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
173
186
  const client = opts.runtime({
174
- maxTurns: turnsPerIteration,
187
+ ...(turnsPerIteration !== undefined ? { maxTurns: turnsPerIteration } : {}),
175
188
  allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
176
189
  label: logLabel,
177
190
  cwd: opts.cwd,
@@ -342,6 +355,14 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
342
355
 
343
356
  const procCtx = buildProcCtx(iteration + 1);
344
357
  let initialPrompt = opts.buildPrompt(lastStatus, iteration);
358
+ // CONTRACT FEEDBACK: when the previous turn's <response> failed schema
359
+ // validation, the violation goes BACK TO THE MODEL as its next turn (the
360
+ // session carries the prior context). Without this, contract retries
361
+ // re-run the identical prompt and the model repeats the identical mistake.
362
+ if (lastResponseValidationError) {
363
+ initialPrompt = `${initialPrompt}\n\n[response contract violation — fix and re-emit]\nYour previous <response> failed schema validation with these errors:\n${lastResponseValidationError.slice(0, 2000)}\nRe-emit the COMPLETE corrected <response> JSON now: every required field present, correctly named and typed (no omissions, no renames).`;
364
+ lastResponseValidationError = "";
365
+ }
345
366
  if (steerDecision) {
346
367
  // Deliver the human's answer as the agent's next user turn by appending
347
368
  // it to the iteration prompt. `sendMessage({ prompt })` is the one input
@@ -364,7 +385,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
364
385
 
365
386
  // Progress, not an error — write to stdout so dashboards and
366
387
  // log viewers don't visually flag it as a warning.
367
- process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
388
+ process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration !== undefined ? `${turnsPerIteration} turns` : "harness-decided turns"}\n`);
368
389
 
369
390
  let responseText = "";
370
391
  // The single `opts.inbox` is shared across iterations, but the
@@ -376,7 +397,11 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
376
397
  // semantics — buffered until consumed).
377
398
  for await (const rawMsg of client.sendMessage({
378
399
  prompt,
379
- sessionId: iteration > 0 ? lastSessionId : undefined,
400
+ // Resume whenever a session exists — NOT keyed on `iteration > 0`, because a
401
+ // contract retry rolls `iteration` back to 0 while a session already exists;
402
+ // keying on the session id keeps the corrective re-prompt in the same session
403
+ // (otherwise it restarts the task in a fresh session and repeats side effects).
404
+ sessionId: lastSessionId ?? undefined,
380
405
  iteration: iteration + 1,
381
406
  ...(opts.inbox ? { inboxStream: opts.inbox } : {}),
382
407
  })) {
@@ -394,7 +419,13 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
394
419
  }
395
420
  const msg = outputVerdict.value;
396
421
  opts.onAgentEvent?.(iteration, msg);
397
- opts.onAgentLifecycleEvent?.({ event: "agent.message", at: Date.now(), agentId, label, iteration: iteration + 1, message: summarizeAgentMessage(msg) });
422
+ // Usage summaries carry the resolved model so the server can price
423
+ // token rows per model without correlating back to agent.spawned.
424
+ const summary = summarizeAgentMessage(msg);
425
+ opts.onAgentLifecycleEvent?.({
426
+ event: "agent.message", at: Date.now(), agentId, label, iteration: iteration + 1,
427
+ message: summary.type === "usage" && client.model != null ? { ...summary, model: client.model } : summary,
428
+ });
398
429
  if (msg.type === "init") lastSessionId = msg.sessionId;
399
430
  if (msg.type === "text") responseText += msg.text;
400
431
  if (msg.type === "error") throw new Error(`Agent error: ${msg.text}`);
@@ -447,6 +478,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
447
478
  process.stderr.write(`${logLabel} NO <response> BLOCK — response tail: ${responseText.slice(-400)}\n`);
448
479
  status = { ...status!, exit_signal: false, blockers: ["No <response> block found — emit a <response> block with the required JSON fields before setting exit_signal: true"] };
449
480
  opts.onIteration?.(iteration + 1, status);
481
+ if (schemaRetriesLeft-- > 0) iteration--; // contract enforcement — free, not billed to the iteration budget
450
482
  continue;
451
483
  }
452
484
  // Status-merged validation — the schema may reference status
@@ -458,6 +490,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
458
490
  process.stderr.write(`${logLabel} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(rawResponse).slice(0, 400)}\n`);
459
491
  status = { ...status!, exit_signal: false, blockers: [`<response> schema validation failed: ${parsed.error.message}`] };
460
492
  opts.onIteration?.(iteration + 1, status);
493
+ if (schemaRetriesLeft-- > 0) iteration--; // contract enforcement — free, not billed to the iteration budget
461
494
  continue;
462
495
  }
463
496
  response = parsed.data;
@@ -487,10 +520,31 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
487
520
  continue;
488
521
  }
489
522
 
490
- if (!status) {
523
+ if (!status && rawResponse === null) {
491
524
  if (++iterationsWithoutStatus >= STALL_ITERATIONS)
492
525
  throw new Error(`${logLabel} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
526
+ // EMPTY-OUTPUT RE-PROMPT: a turn that emits neither <status> nor
527
+ // <response>. Refund the iteration and re-prompt with explicit feedback
528
+ // (bounded by the shared retry pool); the stall counter above still
529
+ // hard-bounds genuinely wedged agents. The directive call-out about
530
+ // BACKGROUND jobs is load-bearing: the dominant benign cause is an agent
531
+ // that ran `cmd &` and parked itself "waiting to be notified" — the loop
532
+ // delivers no such notification, so it must poll synchronously instead.
533
+ // Fires with or without a responseSchema (an agent with no schema still
534
+ // owes a <status>).
535
+ if (schemaRetriesLeft-- > 0) {
536
+ lastResponseValidationError =
537
+ "Your turn ended with no <status> block" + (opts.responseSchema ? " (and no <response> block)" : "") + ". " +
538
+ "If you launched a background job (`… &`) and are waiting to be notified when it finishes — STOP: nothing will notify you here. " +
539
+ "Poll it NOW (read its output / wait for it synchronously to completion), then emit your <status>" +
540
+ (opts.responseSchema ? " and the complete <response> JSON." : ".");
541
+ process.stdout.write(`${logLabel} empty turn (no <status>) — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
542
+ iteration--;
543
+ continue;
544
+ }
493
545
  } else {
546
+ // A parsed structured response (even one that failed schema validation)
547
+ // is real output, not a stall — the contract retry below handles it.
494
548
  iterationsWithoutStatus = 0;
495
549
  }
496
550
 
@@ -506,6 +560,20 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
506
560
  blockerStreak = null;
507
561
  }
508
562
 
563
+ // CONTRACT RETRY (catch-all): the agent EMITTED a <response> block this turn,
564
+ // it failed schema validation, and it didn't go through the <status> branches
565
+ // above (structured-only output, no <status>). Re-prompt with the validation
566
+ // errors as feedback, FREE of the iteration budget. Gated on an actual
567
+ // validation failure this turn AND on the agent not signalling it's still
568
+ // working (`exit_signal: false`) — an in-progress turn that happens to carry
569
+ // a draft <response> consumes its budget normally instead of draining the
570
+ // shared retry pool that genuine contract violations rely on.
571
+ if (opts.responseSchema && rawResponse !== null && lastResponseValidationError !== "" && status?.exit_signal !== false && schemaRetriesLeft-- > 0) {
572
+ process.stdout.write(`${logLabel} response contract not yet satisfied — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
573
+ iteration--;
574
+ continue;
575
+ }
576
+
509
577
  if (iteration + 1 < maxIterations)
510
578
  // Loop continuation — progress.
511
579
  process.stdout.write(`${logLabel} continuing to iteration ${iteration + 2}/${maxIterations}\n`);
@@ -21,6 +21,7 @@ import { AsyncQueue } from "./async-queue.js";
21
21
  import type { AgentMessage, AgentStatus } from "../types/protocol.js";
22
22
  import type { AgentRuntime, RuntimeOptions } from "../types/runtime.js";
23
23
  import type { SandboxProvider } from "../types/sandbox.js";
24
+ import { writeAgentContext } from "./agent-context.js";
24
25
  import type { AgentBudget } from "../types/workflow.js";
25
26
  import type { Processor } from "../processors/processor.js";
26
27
  import { RequestContext } from "../request-context/request-context.js";
@@ -232,7 +233,42 @@ export function resolveAgentId(explicitId?: string): string {
232
233
  }
233
234
 
234
235
  export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopResult<T>> {
235
- const workingDir = opts.workingDir ?? "";
236
+ // Persist-by-default: an agent's working dir is the run dir on the factory
237
+ // drive (the mount step creates it ahead of the run), so everything it
238
+ // writes is kept and attributed to this run; `cd /tmp` for throwaway
239
+ // scratch. Falls back to /workspace when this run has no factory drive.
240
+ // Never "" — an empty cwd is the one case the harnesses' instruction-file
241
+ // upward-walk can't resolve, and it left codex/amp without their AGENTS.md.
242
+ let workingDir = opts.workingDir || process.env.AGENT_COMPOSE_RUN_DIR || "/workspace";
243
+
244
+ // Deliver the platform context (file conventions, connectors & access, how to
245
+ // pause) as AGENTS.md / CLAUDE.md / GEMINI.md at the working dir — and use the
246
+ // write as a LIVENESS PROBE of the working dir. AGENT_COMPOSE_RUN_DIR points at
247
+ // the /factory FUSE drive, whose writes HANG (uninterruptible, no timeout) when
248
+ // the mount degraded — launching the agent there wedges it silently with zero
249
+ // output (no logs). Bound the write; on timeout/failure fall back to /workspace
250
+ // (always present + writable) so a degraded drive can never sink an agent.
251
+ const CONTEXT_WRITE_DEADLINE_MS = 15_000;
252
+ const tryWriteContext = (cwd: string): Promise<boolean> => {
253
+ // Capture the deadline timer so the fast path (write resolves first) can
254
+ // clear it in finally — otherwise each probe leaks a dangling 15s timer.
255
+ // Mirrors sandbox.ts's clear-in-finally pattern.
256
+ let timer: ReturnType<typeof setTimeout> | undefined;
257
+ return Promise.race([
258
+ writeAgentContext({ sandbox: opts.sandbox, cwd, env: process.env }).then(() => true),
259
+ new Promise<boolean>((resolve) => { timer = setTimeout(() => resolve(false), CONTEXT_WRITE_DEADLINE_MS); }),
260
+ ]).catch((err: unknown) => {
261
+ console.error(`[agent] writeAgentContext(${cwd}) failed: ${err instanceof Error ? err.message : String(err)}`);
262
+ return false;
263
+ }).finally(() => { if (timer) clearTimeout(timer); });
264
+ };
265
+ if (!(await tryWriteContext(workingDir)) && workingDir !== "/workspace") {
266
+ console.error(
267
+ `[agent] working dir ${workingDir} not writable within ${CONTEXT_WRITE_DEADLINE_MS}ms ` +
268
+ `(factory drive degraded?) — falling back to /workspace so the agent can run`);
269
+ workingDir = "/workspace";
270
+ await tryWriteContext(workingDir);
271
+ }
236
272
 
237
273
  // Stable agentId for this invocation. Used by the inbox URL (server
238
274
  // scopes pending messages by agentId), the agentLoop (which would