@agent-compose/sdk 0.5.6 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -15,7 +15,7 @@ import { PauseManager } from "../pause/manager.js";
15
15
  import { PauseSignal, isPauseSignal } from "../pause/pause-core.js";
16
16
  import { SteerDecisionSchema, type SteerDecision, type SteerPayload } from "./steer-control.js";
17
17
 
18
- export const DEFAULT_CLAUDE_MODEL = "claude-opus-4-7";
18
+ export const DEFAULT_CLAUDE_MODEL = "claude-fable-5";
19
19
 
20
20
  const SAME_BLOCKER_ITERATIONS = 3;
21
21
  const STALL_ITERATIONS = 3;
@@ -152,8 +152,15 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
152
152
  const label = opts.label ?? "agent";
153
153
  const logLabel = opts.label ?? "[Agent Loop]";
154
154
  const startedAt = Date.now();
155
- const turnsPerIteration = opts.turnsPerIteration ?? 40;
156
- const maxIterations = opts.maxIterations ?? 8;
155
+ // No budget ⇒ no turn cap: the harness runtime (Claude Code) decides when it's
156
+ // done. A numeric budget is an explicit caller choice, not a default we impose.
157
+ const turnsPerIteration = opts.turnsPerIteration;
158
+ const maxIterations = opts.maxIterations ?? (turnsPerIteration === undefined ? 1 : 8);
159
+ // A responseSchema is a CONTRACT, not a hope: when the agent's <response> fails
160
+ // validation, the loop re-prompts with the exact errors until it conforms —
161
+ // without consuming the caller's iteration budget. The backstop below only
162
+ // guards against a truly wedged agent (never reached in normal operation).
163
+ let schemaRetriesLeft = 10;
157
164
  const processors = opts.processors ?? [];
158
165
  const requestContext = opts.requestContext ?? RequestContext.fromReserved({
159
166
  teamId: "", runId: "", workflowId: "",
@@ -171,7 +178,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
171
178
 
172
179
  if (!opts.runtime) throw new Error("agentLoop: opts.runtime is required");
173
180
  const client = opts.runtime({
174
- maxTurns: turnsPerIteration,
181
+ ...(turnsPerIteration !== undefined ? { maxTurns: turnsPerIteration } : {}),
175
182
  allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
176
183
  label: logLabel,
177
184
  cwd: opts.cwd,
@@ -342,6 +349,14 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
342
349
 
343
350
  const procCtx = buildProcCtx(iteration + 1);
344
351
  let initialPrompt = opts.buildPrompt(lastStatus, iteration);
352
+ // CONTRACT FEEDBACK: when the previous turn's <response> failed schema
353
+ // validation, the violation goes BACK TO THE MODEL as its next turn (the
354
+ // session carries the prior context). Without this, contract retries
355
+ // re-run the identical prompt and the model repeats the identical mistake.
356
+ if (lastResponseValidationError) {
357
+ initialPrompt = `${initialPrompt}\n\n[response contract violation — fix and re-emit]\nYour previous <response> failed schema validation with these errors:\n${lastResponseValidationError.slice(0, 2000)}\nRe-emit the COMPLETE corrected <response> JSON now: every required field present, correctly named and typed (no omissions, no renames).`;
358
+ lastResponseValidationError = "";
359
+ }
345
360
  if (steerDecision) {
346
361
  // Deliver the human's answer as the agent's next user turn by appending
347
362
  // it to the iteration prompt. `sendMessage({ prompt })` is the one input
@@ -364,7 +379,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
364
379
 
365
380
  // Progress, not an error — write to stdout so dashboards and
366
381
  // log viewers don't visually flag it as a warning.
367
- process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration} turns\n`);
382
+ process.stdout.write(`${logLabel} iteration ${iteration + 1}/${maxIterations} · ${turnsPerIteration !== undefined ? `${turnsPerIteration} turns` : "harness-decided turns"}\n`);
368
383
 
369
384
  let responseText = "";
370
385
  // The single `opts.inbox` is shared across iterations, but the
@@ -376,7 +391,11 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
376
391
  // semantics — buffered until consumed).
377
392
  for await (const rawMsg of client.sendMessage({
378
393
  prompt,
379
- sessionId: iteration > 0 ? lastSessionId : undefined,
394
+ // Resume whenever a session exists — NOT keyed on `iteration > 0`, because a
395
+ // contract retry rolls `iteration` back to 0 while a session already exists;
396
+ // keying on the session id keeps the corrective re-prompt in the same session
397
+ // (otherwise it restarts the task in a fresh session and repeats side effects).
398
+ sessionId: lastSessionId ?? undefined,
380
399
  iteration: iteration + 1,
381
400
  ...(opts.inbox ? { inboxStream: opts.inbox } : {}),
382
401
  })) {
@@ -447,6 +466,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
447
466
  process.stderr.write(`${logLabel} NO <response> BLOCK — response tail: ${responseText.slice(-400)}\n`);
448
467
  status = { ...status!, exit_signal: false, blockers: ["No <response> block found — emit a <response> block with the required JSON fields before setting exit_signal: true"] };
449
468
  opts.onIteration?.(iteration + 1, status);
469
+ if (schemaRetriesLeft-- > 0) iteration--; // contract enforcement — free, not billed to the iteration budget
450
470
  continue;
451
471
  }
452
472
  // Status-merged validation — the schema may reference status
@@ -458,6 +478,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
458
478
  process.stderr.write(`${logLabel} <response> SCHEMA FAILED: ${parsed.error.message}\nraw: ${JSON.stringify(rawResponse).slice(0, 400)}\n`);
459
479
  status = { ...status!, exit_signal: false, blockers: [`<response> schema validation failed: ${parsed.error.message}`] };
460
480
  opts.onIteration?.(iteration + 1, status);
481
+ if (schemaRetriesLeft-- > 0) iteration--; // contract enforcement — free, not billed to the iteration budget
461
482
  continue;
462
483
  }
463
484
  response = parsed.data;
@@ -487,10 +508,24 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
487
508
  continue;
488
509
  }
489
510
 
490
- if (!status) {
511
+ if (!status && rawResponse === null) {
491
512
  if (++iterationsWithoutStatus >= STALL_ITERATIONS)
492
513
  throw new Error(`${logLabel} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
514
+ // EMPTY-OUTPUT RE-PROMPT: under the unbudgeted default (maxIterations 1)
515
+ // a single turn that emits neither <status> nor <response> would
516
+ // otherwise exhaust the budget with ZERO corrective feedback. Treat it
517
+ // like the contract-violation branches — refund the iteration and
518
+ // re-prompt with explicit feedback, bounded by the shared retry pool.
519
+ // The stall counter above still hard-bounds consecutive empty turns.
520
+ if (opts.responseSchema && schemaRetriesLeft-- > 0) {
521
+ lastResponseValidationError = "no <response> block found — the turn ended with neither a <status> nor a <response> block; emit the complete <response> JSON";
522
+ process.stdout.write(`${logLabel} no <status>/<response> emitted — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
523
+ iteration--;
524
+ continue;
525
+ }
493
526
  } else {
527
+ // A parsed structured response (even one that failed schema validation)
528
+ // is real output, not a stall — the contract retry below handles it.
494
529
  iterationsWithoutStatus = 0;
495
530
  }
496
531
 
@@ -506,6 +541,20 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
506
541
  blockerStreak = null;
507
542
  }
508
543
 
544
+ // CONTRACT RETRY (catch-all): the agent EMITTED a <response> block this turn,
545
+ // it failed schema validation, and it didn't go through the <status> branches
546
+ // above (structured-only output, no <status>). Re-prompt with the validation
547
+ // errors as feedback, FREE of the iteration budget. Gated on an actual
548
+ // validation failure this turn AND on the agent not signalling it's still
549
+ // working (`exit_signal: false`) — an in-progress turn that happens to carry
550
+ // a draft <response> consumes its budget normally instead of draining the
551
+ // shared retry pool that genuine contract violations rely on.
552
+ if (opts.responseSchema && rawResponse !== null && lastResponseValidationError !== "" && status?.exit_signal !== false && schemaRetriesLeft-- > 0) {
553
+ process.stdout.write(`${logLabel} response contract not yet satisfied — corrective re-prompt (${schemaRetriesLeft} retries left)\n`);
554
+ iteration--;
555
+ continue;
556
+ }
557
+
509
558
  if (iteration + 1 < maxIterations)
510
559
  // Loop continuation — progress.
511
560
  process.stdout.write(`${logLabel} continuing to iteration ${iteration + 2}/${maxIterations}\n`);
package/src/client.ts CHANGED
@@ -17,7 +17,7 @@ import { parseSseStream } from "./sse.js";
17
17
  import type { SandboxNetworkPolicy } from "./sandbox.js";
18
18
  import type { RunEvent } from "./types/events.js";
19
19
  import type { WorkflowPlan } from "./types/workflow-plan.js";
20
- import type { SnapshotConfig, IOSchema } from "./types/workflow-metadata.js";
20
+ import type { SnapshotConfig, IOSchema, ConnectorRequirements, ConnectorOperationTag, InvokePolicy } from "./types/workflow-metadata.js";
21
21
  import type { WorkflowManifest } from "./utils/bundler.js";
22
22
 
23
23
  /** UUID-v4-ish — matches the server-side predicate. Used to auto-detect
@@ -52,13 +52,6 @@ export interface RegisterResult {
52
52
  name: string;
53
53
  version: string;
54
54
  runtimes?: RegisteredRuntime[];
55
- /** Non-fatal advisories from the server. Surfaced at register time so the
56
- * operator sees them while still in front of the terminal — currently
57
- * covers "memory extraction is configured but its workflow is not
58
- * registered in this factory". Empty/undefined when registration was
59
- * cleanly resolved against everything the workflow declares it
60
- * depends on. */
61
- warnings?: string[];
62
55
  }
63
56
 
64
57
  export interface RegisteredRuntime {
@@ -72,6 +65,19 @@ export interface RuntimeSourceInput {
72
65
  source: string;
73
66
  }
74
67
 
68
+ /** GitHub provenance for a registered template's source file — stored as
69
+ * `metadata.source` on the registration. `cloud-build` stamps the built
70
+ * commit's sha; the dashboard's manual link path writes `sha: "manual"`. */
71
+ export interface TemplateSourceRef {
72
+ owner: string;
73
+ repo: string;
74
+ branch: string;
75
+ /** Repo-relative file path, e.g. `.agentc/workflows/workflow-deploy.ts`. */
76
+ path: string;
77
+ /** Commit sha the version was built from, or `"manual"` for hand-links. */
78
+ sha: string;
79
+ }
80
+
75
81
  export interface RegisterWorkflowInput {
76
82
  name: string;
77
83
  source: string;
@@ -83,6 +89,9 @@ export interface RegisterWorkflowInput {
83
89
  * executed on the server. */
84
90
  manifest: WorkflowManifest;
85
91
  version?: string;
92
+ /** Where the source file lives on GitHub — stored as `metadata.source`.
93
+ * Named `sourceRef` because `source` is the bundled code itself. */
94
+ sourceRef?: TemplateSourceRef;
86
95
  schedule?: string;
87
96
  runtimes?: RuntimeSourceInput[];
88
97
  /** Human-readable description declared via
@@ -96,13 +105,16 @@ export interface RegisterWorkflowInput {
96
105
  snapshots?: SnapshotConfig;
97
106
  /** Provider-neutral execution plan detected by the CLI bundler. */
98
107
  workflowPlan?: WorkflowPlan;
99
- /** Run the built-in memory extractor after this workflow completes.
100
- * Opt-in; defaults to false when omitted. */
101
- memory?: boolean;
102
- /** Ordered list of workflow names that run as post-hooks after this
103
- * workflow completes. The memory extractor (when `memory: true`)
104
- * runs as an additional hook alongside these. */
105
- postRunHooks?: readonly string[];
108
+ /** Connector requirements declared via `defineWorkflow({ connectors })`
109
+ * (ADR-0007). Validated against the server's provider registry at
110
+ * registration; tokens are injected at the network layer at dispatch. */
111
+ connectors?: ConnectorRequirements;
112
+ /** Connector-catalogue operation tag — see `ConnectorOperationTag`. */
113
+ connectorOperation?: ConnectorOperationTag;
114
+ /** Tier-1 invoke ACL declared via `defineWorkflow({ invokePolicy })`.
115
+ * Only meaningful when the workflow also declares `connectors` — the
116
+ * server gates dispatch on it before binding any grant. */
117
+ invokePolicy?: InvokePolicy;
106
118
  /** Input schema extracted from the workflow's `input` zod schema. */
107
119
  inputSchema?: IOSchema;
108
120
  /** Output schema extracted from the workflow's `output` zod schema. */
@@ -127,19 +139,16 @@ export interface InvokeWorkflowOptions {
127
139
  * vars after brokering. Replaces the template-level placeholders for
128
140
  * this run only — registered metadata is not mutated. */
129
141
  placeholders?: Record<string, string>;
130
- /** Per-invocation memory-extractor override — `false` skips the
131
- * built-in memory hook for this run; omitting leaves the registered
132
- * default in place. */
133
- memory?: boolean;
134
- /** Per-invocation post-hook override — replaces the registered
135
- * `postRunHooks` array for this run only. */
136
- postRunHooks?: readonly string[];
137
142
  /** Explicit parent run id. Pass `null` to suppress ambient RUN_ID auto-detection. */
138
143
  parentRunId?: string | null;
139
144
  /** Agent loop inside the parent run that caused this invoke, when applicable. */
140
145
  agentId?: string | null;
141
146
  /** Factory slug. Defaults to `"default"`. */
142
147
  factorySlug?: string;
148
+ /** Idempotency key — sent as the `Idempotency-Key` header. A repeat invoke
149
+ * with the same key inside the server's dedup window returns the original
150
+ * run instead of starting a new one (matches `resumePause`'s pattern). */
151
+ idempotencyKey?: string;
143
152
  }
144
153
 
145
154
  export interface InvokeAndWaitOptions extends InvokeWorkflowOptions {
@@ -234,6 +243,13 @@ export interface RunStatus<TOutput = unknown> {
234
243
  id: string;
235
244
  status: RunState;
236
245
  output?: TOutput;
246
+ /** The run's latest (`saveLatest`) snapshot id, populated once the run has
247
+ * succeeded — the boot source to fork this run's evolved filesystem from
248
+ * (pass as `snapshots.bootFrom` on a follow-up invoke). `null` while the run
249
+ * is still in flight or when it captured no snapshot. Lets an orchestrator
250
+ * fork a child straight off the `invokeChild` result without a separate
251
+ * `listRunSnapshots` call. */
252
+ latestSnapshotId?: string | null;
237
253
  }
238
254
 
239
255
  /** ADR-0006 step 10 — actor record returned on a successful resume.
@@ -380,6 +396,41 @@ export interface EventRow {
380
396
  createdAt: string;
381
397
  }
382
398
 
399
+ // The server emits these rows in snake_case; the client maps them to
400
+ // camelCase at the fetch boundary so the SDK surface stays uniform
401
+ // (`EventRow.createdAt`, `RunStatus.latestSnapshotId`, …).
402
+ export interface RunArtifactRow {
403
+ path: string;
404
+ factorySlug: string | null;
405
+ sizeBytes: number | null;
406
+ lastWriteAt: string;
407
+ /** Opening text of the file (≤320 chars) — null for binary/empty. */
408
+ preview: string | null;
409
+ }
410
+
411
+ export interface FactoryFileWriteResult {
412
+ path: string;
413
+ contentHash: string;
414
+ sizeBytes: number;
415
+ created: boolean;
416
+ }
417
+
418
+ /** Wire shapes — what the server actually emits (snake_case). */
419
+ interface RunArtifactWire {
420
+ path: string;
421
+ factory_slug: string | null;
422
+ size_bytes: number | null;
423
+ last_write_at: string;
424
+ preview: string | null;
425
+ }
426
+
427
+ interface FactoryFileWriteWire {
428
+ path: string;
429
+ content_hash: string;
430
+ size_bytes: number;
431
+ created: boolean;
432
+ }
433
+
383
434
  export interface ReportEventInput {
384
435
  name: string;
385
436
  body: unknown;
@@ -395,7 +446,7 @@ export interface ListEventsOptions {
395
446
  factorySlug?: string;
396
447
  limit?: number;
397
448
  /** Case-insensitive substring match. Server uses `ILIKE %name%`, so
398
- * `"mem"` matches `memory.fact`, `memory.usage`, etc. Pass the
449
+ * `"site"` matches `site.created`, `site.failed`, etc. Pass the
399
450
  * full event name for an effectively-exact filter (any string is a
400
451
  * substring of itself). */
401
452
  name?: string;
@@ -600,15 +651,16 @@ export class AgentComposeClient {
600
651
  ? detectAmbientParentRunId()
601
652
  : opts.parentRunId;
602
653
  const factorySlug = opts?.factorySlug ?? DEFAULT_FACTORY;
654
+ const headers: Record<string, string> = {};
655
+ if (opts?.idempotencyKey) headers["Idempotency-Key"] = opts.idempotencyKey;
603
656
  return this.fetch(templatePath(factorySlug, name, "invoke"), {
604
657
  method: "POST",
658
+ ...(opts?.idempotencyKey ? { headers } : {}),
605
659
  body: {
606
660
  input,
607
661
  ...(opts?.snapshots !== undefined ? { snapshots: opts.snapshots } : {}),
608
662
  ...(opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {}),
609
663
  ...(opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {}),
610
- ...(opts?.memory !== undefined ? { memory: opts.memory } : {}),
611
- ...(opts?.postRunHooks !== undefined ? { postRunHooks: opts.postRunHooks } : {}),
612
664
  ...(parentRunId ? { parentRunId } : {}),
613
665
  ...(opts?.agentId ? { agentId: opts.agentId } : {}),
614
666
  },
@@ -895,6 +947,73 @@ export class AgentComposeClient {
895
947
  return body.events;
896
948
  }
897
949
 
950
+ /** Files the run wrote on the factory drive — run-attributed revisions,
951
+ * latest write per path, paths the run later deleted excluded. */
952
+ async listRunArtifacts(runId: string): Promise<RunArtifactRow[]> {
953
+ const body = await this.fetch<{ artifacts: RunArtifactWire[] }>(
954
+ `/api/v1/workflows/${encodeURIComponent(runId)}/artifacts`,
955
+ );
956
+ return body.artifacts.map((a) => ({
957
+ path: a.path,
958
+ factorySlug: a.factory_slug,
959
+ sizeBytes: a.size_bytes,
960
+ lastWriteAt: a.last_write_at,
961
+ preview: a.preview,
962
+ }));
963
+ }
964
+
965
+ // ── Factory files ──────────────────────────────────────────────────────────
966
+ // The factory drive: documents surfaced in the dashboard's Files tab.
967
+ // Writes from inside a sandbox automatically carry the run-callback token
968
+ // (AGENT_COMPOSE_RUN_TOKEN), so the revision is attributed to the run —
969
+ // that's what surfaces the doc in the Workbench docs section and the
970
+ // editor-avatar run hover.
971
+
972
+ /** Write (create or overwrite) one file on a factory's drive. */
973
+ async putFactoryFile(
974
+ path: string,
975
+ content: string | Uint8Array,
976
+ opts?: { factorySlug?: string; contentType?: string },
977
+ ): Promise<FactoryFileWriteResult> {
978
+ const factorySlug = opts?.factorySlug
979
+ ?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined)
980
+ ?? DEFAULT_FACTORY;
981
+ const runToken = typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_RUN_TOKEN : undefined;
982
+ const wire = await this.fetch<FactoryFileWriteWire>(
983
+ `/api/v1/factories/${encodeURIComponent(factorySlug)}/files/content?path=${encodeURIComponent(path)}`,
984
+ {
985
+ method: "PUT",
986
+ body: content,
987
+ headers: {
988
+ "content-type": opts?.contentType ?? "text/plain; charset=utf-8",
989
+ ...(runToken ? { "x-run-token": runToken } : {}),
990
+ },
991
+ },
992
+ );
993
+ return {
994
+ path: wire.path,
995
+ contentHash: wire.content_hash,
996
+ sizeBytes: wire.size_bytes,
997
+ created: wire.created,
998
+ };
999
+ }
1000
+
1001
+ /** Read one file's current content (or a specific revision) as text. */
1002
+ async getFactoryFile(
1003
+ path: string,
1004
+ opts?: { factorySlug?: string; revision?: number },
1005
+ ): Promise<string> {
1006
+ const factorySlug = opts?.factorySlug
1007
+ ?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined)
1008
+ ?? DEFAULT_FACTORY;
1009
+ const q = new URLSearchParams({ path });
1010
+ if (opts?.revision !== undefined) q.set("revision", String(opts.revision));
1011
+ return this.fetch(
1012
+ `/api/v1/factories/${encodeURIComponent(factorySlug)}/files/content?${q}`,
1013
+ { responseType: "text" },
1014
+ );
1015
+ }
1016
+
898
1017
  /** List events ingested into a factory, newest first. Supports
899
1018
  * case-insensitive substring filter (`name`) and timestamp-cursor
900
1019
  * pagination (`before`). Returns `{ events, has_more }` — the
package/src/index.ts CHANGED
@@ -41,8 +41,8 @@ export type {
41
41
  ReuseSnapshot,
42
42
  IOSchema,
43
43
  OutputSchema,
44
- WorkflowMemoryConfig,
45
44
  } from "./types/workflow.js";
45
+ export type { ConnectorRequestRules } from "./types/workflow-metadata.js";
46
46
 
47
47
  // Snapshot entry type re-exported for consumers (dashboard, CLI).
48
48
  export type { RunSnapshotEntry } from "./client.js";
@@ -106,7 +106,7 @@ export type {
106
106
  // HTTP client
107
107
  export { AgentComposeClient } from "./client.js";
108
108
  export type {
109
- RegisterResult, RegisterWorkflowInput, RuntimeSourceInput,
109
+ RegisterResult, RegisterWorkflowInput, RuntimeSourceInput, TemplateSourceRef,
110
110
  InvokeWorkflowOptions, InvokeAndWaitOptions, InvokeResult,
111
111
  ListSnapshotsOptions, TemplateRow, ListTemplatesOptions,
112
112
  CreateFactoryInput, UpdateFactoryInput,
@@ -181,6 +181,7 @@ export { createSandbox, reconnectSandbox, killAllSandboxes, killSandboxById,
181
181
  getSandboxQuotas, listOwnedSandboxes, deleteSandboxSnapshot,
182
182
  makeSandboxProvider, makeDesktopSandboxProvider,
183
183
  parseSseExecStream, AGENT_COMPOSE_TAG } from "./sandbox.js";
184
+ export { SandboxUnavailableError, SANDBOX_UNAVAILABLE_PREFIX } from "./sandbox-errors.js";
184
185
  export type {
185
186
  SandboxCreateOpts, SandboxNetworkPolicy, SandboxNetworkHeaderTransform,
186
187
  SandboxNetworkAllowRule, SandboxNetworkSubnetPolicy, SandboxProviderName,
@@ -62,8 +62,20 @@ function translateMessage(message: Record<string, unknown>): AgentMessage[] {
62
62
  numTurns: Number(message.num_turns ?? 0),
63
63
  timestamp: ts,
64
64
  });
65
- if (message.is_error || message.subtype === "error_during_execution") {
66
- msgs.push({ type: "error", text: formatError(message.error ?? message.result ?? message.message), timestamp: ts });
65
+ if (message.subtype === "error_max_turns") {
66
+ // Running out of turns is NOT a failure — the agent did real work and the
67
+ // files it wrote are on disk. End the iteration cleanly (don't throw) so the
68
+ // workflow keeps the partial result and moves on.
69
+ msgs.push({ type: "done", sessionId: String(message.session_id ?? ""), timestamp: ts });
70
+ } else if (message.is_error || message.subtype === "error_during_execution") {
71
+ // Error results sometimes carry NO error/result/message fields (e.g.
72
+ // the binary died early) — fall back to subtype + the raw envelope so
73
+ // the failure is diagnosable instead of "Agent error: undefined".
74
+ const detail = message.error ?? message.result ?? message.message;
75
+ const text = detail !== undefined
76
+ ? formatError(detail)
77
+ : `${String(message.subtype ?? "unknown_error")} — raw result: ${JSON.stringify({ ...message, usage: undefined }).slice(0, 600)}`;
78
+ msgs.push({ type: "error", text, timestamp: ts });
67
79
  } else {
68
80
  msgs.push({ type: "done", sessionId: String(message.session_id ?? ""), timestamp: ts });
69
81
  }
@@ -77,7 +89,11 @@ export interface ClaudeRuntimeConfig {
77
89
  claudeMdContent?: string;
78
90
  /** Env overrides for Agent SDK provider routing. */
79
91
  env?: Record<string, string>;
80
- /** Model to use. Defaults to DEFAULT_CLAUDE_MODEL. */
92
+ /** Model to use. Accepts a caliber shorthand — "fable" (most capable),
93
+ * "opus", "sonnet", "haiku" (fastest/cheapest) — or an exact model id.
94
+ * Defaults to DEFAULT_CLAUDE_MODEL (Fable). Orchestrators pick a caliber
95
+ * per agent: fable/opus for planning + implementation, sonnet for focused
96
+ * single-responsibility work, haiku for mechanical tasks. */
81
97
  model?: string;
82
98
  /** MCP servers to configure for the Agent SDK. */
83
99
  mcpServers?: Record<string, { command: string; args?: string[]; env?: Record<string, string> }>;
@@ -87,6 +103,31 @@ export interface ClaudeRuntimeConfig {
87
103
  thinking?: ThinkingConfig;
88
104
  /** Reasoning effort hint for models that support adaptive thinking. */
89
105
  effort?: "low" | "medium" | "high" | "xhigh" | "max";
106
+ /** Skills to enable for the agent (the Agent SDK's `skills` option — also
107
+ * auto-adds the `Skill` tool). The agent-env bakes the `/ac:*` skills via
108
+ * `agentc init`; default `"all"` makes them usable. Pass `[]` to disable. */
109
+ skills?: string[] | "all";
110
+ }
111
+
112
+ /** Caliber shorthand → exact model id. Full ids pass through untouched. */
113
+ const MODEL_TIERS: Record<string, string> = {
114
+ fable: "claude-fable-5",
115
+ opus: "claude-opus-4-8",
116
+ sonnet: "claude-sonnet-4-6",
117
+ haiku: "claude-haiku-4-5",
118
+ };
119
+ function resolveClaudeModel(model: string | undefined): string {
120
+ if (!model) return DEFAULT_CLAUDE_MODEL;
121
+ return MODEL_TIERS[model.toLowerCase()] ?? model;
122
+ }
123
+
124
+ /** Fable rejects an explicit `thinking: {type: "disabled"}` with a 400 (the only
125
+ * off-mode on Fable is omitting the param). Other models accept it. Resolve the
126
+ * thinking config against the chosen model so callers can keep passing
127
+ * `thinking: {type: "disabled"}` for cheap/fast agents regardless of tier. */
128
+ function resolveThinking(model: string, thinking: ThinkingConfig | undefined): ThinkingConfig | undefined {
129
+ if (model.startsWith("claude-fable") && thinking && (thinking as { type?: string }).type === "disabled") return undefined;
130
+ return thinking;
90
131
  }
91
132
 
92
133
  /** Canonical install path for Claude Code inside a Vercel sandbox.
@@ -130,7 +171,7 @@ export class ClaudeRunner implements ModelExecutionContract {
130
171
  ) {}
131
172
 
132
173
  get model(): string {
133
- return this.config.model ?? this.options.model ?? DEFAULT_CLAUDE_MODEL;
174
+ return resolveClaudeModel(this.config.model ?? this.options.model);
134
175
  }
135
176
 
136
177
  async gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult> {
@@ -179,7 +220,7 @@ export class ClaudeRunner implements ModelExecutionContract {
179
220
  //
180
221
  // Without an inboxStream we keep the simple string-prompt path —
181
222
  // no queue, same behaviour as before. This keeps non-interactive
182
- // workflows (memory extractor, etc.) on the original code path.
223
+ // workflows on the original code path.
183
224
  //
184
225
  // The queue is closed when the SDK's `result` message indicates
185
226
  // the iteration's assistant turns are done; without close() the
@@ -216,9 +257,9 @@ export class ClaudeRunner implements ModelExecutionContract {
216
257
  tools: this.options.allowedTools,
217
258
  allowedTools: this.options.allowedTools,
218
259
  maxTurns: this.options.maxTurns,
219
- model: this.config.model ?? this.options.model ?? DEFAULT_CLAUDE_MODEL,
260
+ model: this.model,
220
261
  outputFormat: this.options.outputFormat,
221
- thinking: this.config.thinking,
262
+ thinking: resolveThinking(this.model, this.config.thinking),
222
263
  effort: this.config.effort,
223
264
  cwd: this.options.cwd,
224
265
  env: { ...process.env, ...(this.config.env ?? {}) },
@@ -226,6 +267,10 @@ export class ClaudeRunner implements ModelExecutionContract {
226
267
  ?? process.env.CLAUDE_CODE_EXECUTABLE
227
268
  ?? DEFAULT_CLAUDE_PATH,
228
269
  ...(this.config.claudeMdContent ? { systemPrompt: { type: "preset" as const, preset: "claude_code" as const, append: this.config.claudeMdContent } } : {}),
270
+ // Turn skills ON (and auto-add the `Skill` tool). The agent-env
271
+ // bakes the `/ac:*` skills; without this the Agent SDK leaves
272
+ // them un-enabled and the agent can't invoke them.
273
+ skills: this.config.skills ?? "all",
229
274
  resume: opts.sessionId,
230
275
  mcpServers: this.config.mcpServers,
231
276
  permissionMode: "acceptEdits",
@@ -251,7 +296,20 @@ export class ClaudeRunner implements ModelExecutionContract {
251
296
  if (raw.type === "result" && inboxQueue) inboxQueue.close();
252
297
  }
253
298
  } catch (err) {
254
- yield { type: "error", text: formatError(err), timestamp: now() };
299
+ // The SDK ALSO signals "out of turns" by THROWING (separately from the
300
+ // result-message `error_max_turns` subtype handled in translateMessage — that
301
+ // typed path is the primary signal; this catch covers the SDK's separate
302
+ // throw, which is prose-only). Treat it the same: NOT a failure — end the
303
+ // iteration cleanly so the loop continues, keeping the work it did.
304
+ // Match ONLY the specific exhaustion phrase, NOT a loose "maxTurns" token: the
305
+ // latter would also swallow a genuine error like "maxTurns must be a positive
306
+ // integer" and silently report it as a clean finish.
307
+ const text = formatError(err);
308
+ if (/maximum number of turns/i.test(text)) {
309
+ yield { type: "done", sessionId: opts.sessionId ?? "", timestamp: now() };
310
+ } else {
311
+ yield { type: "error", text, timestamp: now() };
312
+ }
255
313
  } finally {
256
314
  // Belt-and-braces — close on error/abort too so the dangling
257
315
  // iterable doesn't leak the inboxStream consumer.
@@ -0,0 +1,53 @@
1
+ /**
2
+ * Sandbox-infrastructure error.
3
+ *
4
+ * Distinct from `StepExecutionError` (the user/runner step-failure wrapper):
5
+ * that's the workflow author's problem. A `SandboxUnavailableError` means the
6
+ * sandbox itself couldn't carry the step — the provider refused a command, the
7
+ * sandbox was reclaimed (idle/lifetime timeout, eviction), an API blip, or the
8
+ * connection dropped mid-stream. None of these are a fault in the customer's
9
+ * code, so they're surfaced as "infrastructure issue, retry" rather than
10
+ * blaming the user.
11
+ *
12
+ * ## Retryability is about *when*, not *which error*
13
+ *
14
+ * The Vercel provider (`sandbox.ts`) decides `retryable` purely from *where* it
15
+ * caught the error — not by inspecting status codes or error vocabularies:
16
+ * - `retryable: true` — caught before the runner launched the step (file
17
+ * write / command launch / reconnect refused). No user code ran, so
18
+ * re-provisioning a fresh sandbox and re-running the step is safe. ANY
19
+ * error here qualifies (sandbox reclaimed, 429, 5xx, network blip).
20
+ * - `retryable: false` — caught while/after the runner streamed, so user code
21
+ * may already have executed side effects. Surfaced honestly as an infra
22
+ * failure, but NOT auto-retried.
23
+ *
24
+ * ## Cross-process contract
25
+ *
26
+ * The thrown error crosses the Temporal activity→workflow serialisation
27
+ * boundary, which preserves the error **message** but discards custom instance
28
+ * fields. So the retryability signal is encoded IN the message via a stable
29
+ * prefix — `[sandbox-unavailable:retryable]` or `[sandbox-unavailable:terminal]`
30
+ * — following the same `[kind] …` convention `StepExecutionError` uses. The
31
+ * server (`utils/transient-errors.ts`) matches this prefix with a pure regex to
32
+ * classify the failure and decide whether the workflow re-provisions and
33
+ * retries. A unit test pins the SDK-produced string against the server matcher
34
+ * so the two can't drift.
35
+ */
36
+
37
+ /** Stable message prefix. Both variants share this leading token so a single
38
+ * server-side regex recognises the class; the `:retryable` / `:terminal`
39
+ * suffix carries the recovery decision. */
40
+ export const SANDBOX_UNAVAILABLE_PREFIX = "[sandbox-unavailable";
41
+
42
+ export class SandboxUnavailableError extends Error {
43
+ /** True when caught before any user code ran (safe to re-provision + retry);
44
+ * false when the sandbox died mid/after execution. */
45
+ readonly retryable: boolean;
46
+ readonly sandboxId: string | undefined;
47
+ constructor(detail: string, opts: { retryable: boolean; sandboxId?: string }) {
48
+ super(`${SANDBOX_UNAVAILABLE_PREFIX}:${opts.retryable ? "retryable" : "terminal"}] ${detail}`);
49
+ this.name = "SandboxUnavailableError";
50
+ this.retryable = opts.retryable;
51
+ this.sandboxId = opts.sandboxId;
52
+ }
53
+ }