@agent-compose/sdk 0.5.7 → 0.5.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,17 @@
1
+ /**
2
+ * run-agent working-dir liveness probe (the degraded-FUSE-wedge fix).
3
+ *
4
+ * `agent()` delivers the platform context file (AGENTS.md / CLAUDE.md /
5
+ * GEMINI.md) to the working dir AND uses that write as a bounded liveness
6
+ * probe of the dir. AGENT_COMPOSE_RUN_DIR points at the /factory FUSE drive,
7
+ * whose writes HANG uninterruptibly (no timeout) when the mount degraded —
8
+ * launching the agent there wedges it silently with zero output. The probe
9
+ * bounds the write at 15s and falls back to /workspace (always present +
10
+ * writable) when the run dir is wedged.
11
+ *
12
+ * These tests pin: (1) the happy path keeps RUN_DIR; (2) a RUN_DIR whose write
13
+ * hangs forever still selects /workspace within the deadline. `agentLoop` is
14
+ * mocked so we observe only the `cwd` the loop is launched with; fake timers
15
+ * drive the 15s deadline without waiting in real time.
16
+ */
17
+ export {};
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Harness-agnostic agent context delivery.
3
+ *
4
+ * Every coding-agent harness we drive (Claude Code, Codex, Amp, Gemini, …)
5
+ * looks for an instruction file in its working directory — but they disagree
6
+ * on the NAME (Codex/Amp read `AGENTS.md`; Claude reads `CLAUDE.md`; Gemini
7
+ * reads `GEMINI.md`). So `agent()` writes the SAME platform manual under all
8
+ * three names at the agent's working dir, and every harness finds the one it
9
+ * knows. The manual is the single source of truth here; `base-env` bakes a
10
+ * static copy at `/workspace/AGENTS.md` for plain shell sessions, but the
11
+ * per-run copy `agent()` writes is the authoritative one — it carries the
12
+ * live connector list and lands at the run's working dir.
13
+ */
14
+ import type { SandboxProvider } from "../types/sandbox.js";
15
+ /**
16
+ * The platform manual delivered to every agent, regardless of harness.
17
+ * Covers the three things an agent must know: where files go (the factory
18
+ * drive + the persist-by-default working dir), how to pause for a human, and
19
+ * that credentials are network-injected (never in the env). The live
20
+ * "Connectors & access" section is appended per-run by `buildAgentContextDoc`.
21
+ */
22
+ export declare const AGENT_COMPOSE_MANUAL = "# Working inside an Agent Compose sandbox\n\nYou are an agent running in a per-run sandbox on the Agent Compose platform.\nUse the **`agentc` CLI** and the **`@agent-compose/sdk`** for everything below \u2014\ndo NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on\nyour PATH and already authenticated from the environment\n(`AGENT_COMPOSE_URL` / `AGENT_COMPOSE_API_KEY` / `AGENT_COMPOSE_FACTORY` are\ninjected for this run), so commands just work \u2014 no login, no keys to manage.\n\nThe `/ac:*` skills are installed as Claude Code slash commands (`/ac:invoke`,\n`/ac:events`, `/ac:logs`, `/ac:register`, \u2026) \u2014 reach for them too.\n\n## Files \u2014 your outputs persist by default\n\nYour working directory defaults to **`$AGENT_COMPOSE_RUN_DIR`** \u2014 a per-run\ndirectory on the shared factory drive\n(`$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/`) the platform\ncreates and attributes to this run. **Files you write here persist by\ndefault** \u2014 they show up in the dashboard's Files tab and the run's Artifacts\ncard, with no API calls to save them. The dir already exists and is writable.\n\nNeed throwaway scratch \u2014 heavy build output, package caches, temp files?\n`cd /tmp` (or any path outside `/factory`): anything off the factory drive is\nephemeral and discarded when the sandbox ends. In short: **stay in your working\ndir to keep something, `cd` out to throw it away.**\n\nThe whole shared drive is POSIX-mounted at `/factory`; the dashboard-visible\nroot is `$AGENT_COMPOSE_FACTORY_DIR` (`/factory/files`). Earlier versions and\nruns live in sibling dirs under\n`$AGENT_COMPOSE_FACTORY_DIR/$AGENT_COMPOSE_WORKFLOW/` \u2014 read them for prior\ncontext. Other workflows' dirs are present but not your concern.\n\n## Events \u2014 the factory timeline\n\nRecord something on the run/factory timeline (the dashboard renders these)\nwith the CLI \u2014 your run id is `$RUN_ID`:\n\n agentc events send \"$RUN_ID\" <name> --summary \"<one line>\" [--body '<json>']\n\nNames like `note.created` / `brief.posted` surface in the Workbench;\n`agentc events list` reads them back. `/ac:events` is the skill equivalent.\n\n## Runs\n\n agentc list # registered workflows (/ac:list)\n agentc logs \"$RUN_ID\" # a run's logs (/ac:logs)\n agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)\n\n## Writing workflow / agent code \u2014 the SDK\n\n`@agent-compose/sdk` is installed in `/workspace` \u2014 import it from any script\nyou write there:\n\n import { defineWorkflow, agent, AgentComposeClient } from \"@agent-compose/sdk\";\n\nUse `/ac:generate-workflow` / `/ac:generate-agent` to scaffold, then\n`agentc register <file.ts>` (or `/ac:register`).\n\n## Pausing to ask the human \u2014 `agentc pause`\n\nWhen you can't or shouldn't proceed without a human, run `agentc pause`. It\nblocks until they answer on the dashboard, then prints their answer to stdout:\n\n ANSWER=$(agentc pause --reason \"Notion returned 401 \u2014 connect Notion to continue\" \\\n --option retry --option skip)\n\nReach for it the moment you hit \u2014 or foresee \u2014 any of these:\n- **A wall only a human can clear:** a 401/403, a missing credential, an\n unconnected provider, a host the network refuses. Do NOT retry blindly or try\n to work around it \u2014 pause and say what needs enabling.\n- **A durable or outward-facing action that needs sign-off:** registering a\n workflow, deploying, sending email/messages, deleting or overwriting shared\n data, spending money. Prepare everything, then pause for approval BEFORE you\n commit it.\n- **A judgment call only the human can settle:** an under-specified request,\n several valid paths, a conflict with existing state, missing input only they have.\n\nYou compose the `--reason` (the ask) yourself; pass `--option` choices when\nthere are clear ones, omit them for a free-form answer. Read the printed answer\nand act on it. Each agent pauses independently \u2014 pausing doesn't stop the others.\n\n## Credentials\n\nConnector credentials (Google, GitHub, \u2026) are NEVER in your environment.\nThey're injected at the network layer when you call an allowed host \u2014 make the\nrequest **without** an Authorization header and the platform adds it. Don't try\nto read or exfiltrate tokens; they aren't here. The \"Connectors & access\"\nsection below (when present) lists exactly which providers this run can reach.\n\n## Tools in this environment\n\n- `agentc` \u2014 Agent Compose CLI (your primary interface; authed from env)\n- `@agent-compose/sdk` \u2014 installed in /workspace for writing workflows\n- `/ac:*` Claude Code skills \u2014 slash commands for the above\n- `archil` (factory drive), `rtk`, `bun`\n- A world-writable `/workspace` working directory";
23
+ /**
24
+ * One connector this run can reach, as the agent should see it. Strictly
25
+ * NON-SECRET — hosts, methods, paths, identity only. The access token is
26
+ * injected at the network layer and never appears here. The server builds
27
+ * this list at dispatch from the run's connector grants × the provider
28
+ * catalogue and delivers it as the `AGENT_COMPOSE_CONNECTORS` env (JSON array).
29
+ */
30
+ export interface AgentConnectorInfo {
31
+ /** Provider key (`github`, `notion`, …). */
32
+ provider: string;
33
+ /** Human label ("GitHub", "Notion"). */
34
+ name?: string;
35
+ /** API hosts the credential is injected for. */
36
+ hosts?: string[];
37
+ /** Allowed HTTP methods (Tier-2 narrowing). Empty/absent = any. */
38
+ methods?: string[];
39
+ /** Allowed path prefixes (Tier-2 narrowing). Empty/absent = any. */
40
+ pathPrefixes?: string[];
41
+ /** GitHub: the repository the minted token is scoped to. */
42
+ repository?: string;
43
+ /** Coarse capability the token was minted with. */
44
+ access?: string;
45
+ /** Human scope descriptions, when the provider declares them. */
46
+ scopes?: string[];
47
+ }
48
+ /** Compose the full per-run agent doc: the static manual + the live
49
+ * connectors section read from `AGENT_COMPOSE_CONNECTORS` (a JSON array;
50
+ * malformed/absent → no section). */
51
+ export declare function buildAgentContextDoc(env: Record<string, string | undefined>): string;
52
+ /**
53
+ * Write the platform context at the agent's working dir under every harness's
54
+ * instruction-file name, so whichever CLI runs finds the one it reads. Codex
55
+ * and Amp read `AGENTS.md` natively; Claude reads `CLAUDE.md`; Gemini reads
56
+ * `GEMINI.md` — we write identical content to all three rather than detect the
57
+ * harness (the runtime's `kind` isn't known until after spawn, and a few extra
58
+ * small files in our own run dir are harmless).
59
+ *
60
+ * Best-effort: a write failure logs and is swallowed — never fail an agent
61
+ * because its context file couldn't be written.
62
+ */
63
+ export declare function writeAgentContext(args: {
64
+ sandbox: Pick<SandboxProvider, "files">;
65
+ cwd: string;
66
+ env: Record<string, string | undefined>;
67
+ }): Promise<void>;
@@ -46,6 +46,7 @@ export type AgentMessageSummary = {
46
46
  cacheCreationTokens: number;
47
47
  durationMs: number;
48
48
  numTurns: number;
49
+ model?: string;
49
50
  } | {
50
51
  type: "done";
51
52
  sessionId: string;
package/dist/client.d.ts CHANGED
@@ -10,10 +10,10 @@
10
10
  * `register()` accepts pre-built sources — use the CLI (`agent-compose
11
11
  * register`) or build sources yourself and pass them directly.
12
12
  */
13
- import type { SandboxNetworkPolicy } from "./sandbox.js";
13
+ import type { SandboxNetworkPolicy, SandboxSize } from "./sandbox.js";
14
14
  import type { RunEvent } from "./types/events.js";
15
15
  import type { WorkflowPlan } from "./types/workflow-plan.js";
16
- import type { SnapshotConfig, IOSchema, ConnectorRequirements, ConnectorOperationTag, InvokePolicy } from "./types/workflow-metadata.js";
16
+ import type { SnapshotConfig, IOSchema, ConnectorRequirements, ConnectorOperationTag, InvokePolicy, SandboxResources } from "./types/workflow-metadata.js";
17
17
  import type { WorkflowManifest } from "./utils/bundler.js";
18
18
  export interface RegisterResult {
19
19
  id: string;
@@ -67,6 +67,8 @@ export interface RegisterWorkflowInput {
67
67
  /** All snapshot config — `bootFrom` (where to restore at run start),
68
68
  * `save`, `retain`. See `WorkflowMetadata.snapshots`. */
69
69
  snapshots?: SnapshotConfig;
70
+ /** Sandbox machine size (template default). See `WorkflowMetadata.resources`. */
71
+ resources?: SandboxResources;
70
72
  /** Provider-neutral execution plan detected by the CLI bundler. */
71
73
  workflowPlan?: WorkflowPlan;
72
74
  /** Connector requirements declared via `defineWorkflow({ connectors })`
@@ -101,6 +103,10 @@ export interface InvokeWorkflowOptions {
101
103
  * vars after brokering. Replaces the template-level placeholders for
102
104
  * this run only — registered metadata is not mutated. */
103
105
  placeholders?: Record<string, string>;
106
+ /** Per-invocation machine-size override of the template's `resources.size`.
107
+ * `small` (default) | `medium` | `large`; omit → the template default,
108
+ * else `small`. Honoured on Vercel (→ vCPUs); E2B ignores it. */
109
+ size?: SandboxSize;
104
110
  /** Explicit parent run id. Pass `null` to suppress ambient RUN_ID auto-detection. */
105
111
  parentRunId?: string | null;
106
112
  /** Agent loop inside the parent run that caused this invoke, when applicable. */
@@ -135,6 +141,53 @@ export interface TemplateRow {
135
141
  export interface ListTemplatesOptions {
136
142
  factorySlug?: string;
137
143
  }
144
+ /** A human member of your team — the people an agent (or you) can @-flag. */
145
+ export interface TeamMember {
146
+ /** Membership row id. */
147
+ id: string;
148
+ /** The user id — what you pass to `createMentions({ mentionedUserIds })`. */
149
+ userId: string;
150
+ role: string;
151
+ email: string;
152
+ name: string;
153
+ joinedAt: string;
154
+ }
155
+ /** A "you were flagged" ping, persisted server-side so it reaches the
156
+ * mentioned teammate in their Workbench. */
157
+ export interface Mention {
158
+ id: string;
159
+ factoryId: string;
160
+ mentionedUserId: string;
161
+ /** Who flagged: 'user' | 'api_key' | 'run' | 'system'. */
162
+ actorKind: string;
163
+ actorId: string | null;
164
+ actorLabel: string | null;
165
+ /** Where it lives: 'doc' | 'comment' | 'plan' | 'run'. */
166
+ contextKind: string;
167
+ contextPath: string | null;
168
+ /** Ready-made relative dashboard URL the Workbench card links to. */
169
+ contextUrl: string | null;
170
+ text: string;
171
+ runId: string | null;
172
+ seenAt: string | null;
173
+ resolvedAt: string | null;
174
+ createdAt: string;
175
+ }
176
+ export interface CreateMentionsInput {
177
+ /** Team-member user ids to flag (1–20). Discover them via `listMembers()`.
178
+ * Non-members are dropped server-side. */
179
+ mentionedUserIds: string[];
180
+ /** The flag message shown in the teammate's Workbench. */
181
+ text: string;
182
+ contextKind: "doc" | "comment" | "plan" | "run";
183
+ /** Factory-relative file path or comment thread id, when applicable. */
184
+ contextPath?: string;
185
+ /** Ready-made relative dashboard URL the Workbench card links to (e.g.
186
+ * `/factories/<slug>/files/view?path=<plan>`). */
187
+ contextUrl?: string;
188
+ runId?: string;
189
+ factorySlug?: string;
190
+ }
138
191
  export interface CreateFactoryInput {
139
192
  slug: string;
140
193
  name: string;
@@ -622,6 +675,16 @@ export declare class AgentComposeClient {
622
675
  factorySlug?: string;
623
676
  revision?: number;
624
677
  }): Promise<string>;
678
+ /** List the human members of your team — the people you (or an agent) can
679
+ * @-flag with `createMentions`. Each row's `userId` is what
680
+ * `mentionedUserIds` expects. */
681
+ listMembers(): Promise<TeamMember[]>;
682
+ /** Flag one or more teammates — a durable ping that lands in their factory
683
+ * Workbench. Use from an agent (e.g. a remediation plan that needs a human
684
+ * to rotate a secret) or any team automation. Resolve `mentionedUserIds`
685
+ * via `listMembers()`. When run inside a sandbox the run-callback token is
686
+ * forwarded so the ping is attributed to the run ("flagged by <workflow>"). */
687
+ createMentions(input: CreateMentionsInput): Promise<Mention[]>;
625
688
  /** List events ingested into a factory, newest first. Supports
626
689
  * case-insensitive substring filter (`name`) and timestamp-cursor
627
690
  * pagination (`before`). Returns `{ events, has_more }` — the
package/dist/index.d.ts CHANGED
@@ -28,7 +28,7 @@ export type { Processor, ProcessorContext, ProcessorVerdict, ToolCall, } from ".
28
28
  export type { AgentMessage, AgentMessageInit, AgentMessageText, AgentMessageThinking, AgentMessageToolUse, AgentMessageToolResult, AgentMessageDone, AgentMessageError, AgentMessageUsage, AgentStatus, } from "./types/protocol.js";
29
29
  export type { SandboxProvider, DesktopSandboxProvider, } from "./types/sandbox.js";
30
30
  export { AgentComposeClient } from "./client.js";
31
- export type { RegisterResult, RegisterWorkflowInput, RuntimeSourceInput, TemplateSourceRef, InvokeWorkflowOptions, InvokeAndWaitOptions, InvokeResult, ListSnapshotsOptions, TemplateRow, ListTemplatesOptions, CreateFactoryInput, UpdateFactoryInput, SecretOptions, SetSecretResult, SecretListEntry, CreateApiKeyInput, StreamRunLogsOptions, EventSubjectType, EventRow, ReportEventInput, ListEventsOptions, ListEventsResult, RunLogLine, ListRunLogsOptions, RegisteredRuntime, RunState, RunStatus, FactoryRow, SnapshotListEntry, SnapshotListResponse, ApiKey, ApiKeyCreated, UsageRollupRow, UsageResponse, CancelRunResponse, RequestAgentPauseOptions, RequestAgentPauseResponse, SendAgentMessageOptions, SendAgentMessageResponse, AnswerSteerOptions, ResumePauseOptions, ResumePauseResponse, ResumePauseSuccess, ResumePausePending, ResumePauseActor, } from "./client.js";
31
+ export type { RegisterResult, RegisterWorkflowInput, RuntimeSourceInput, TemplateSourceRef, InvokeWorkflowOptions, InvokeAndWaitOptions, InvokeResult, ListSnapshotsOptions, TemplateRow, ListTemplatesOptions, CreateFactoryInput, UpdateFactoryInput, SecretOptions, SetSecretResult, SecretListEntry, CreateApiKeyInput, StreamRunLogsOptions, TeamMember, Mention, CreateMentionsInput, EventSubjectType, EventRow, ReportEventInput, ListEventsOptions, ListEventsResult, RunLogLine, ListRunLogsOptions, RegisteredRuntime, RunState, RunStatus, FactoryRow, SnapshotListEntry, SnapshotListResponse, ApiKey, ApiKeyCreated, UsageRollupRow, UsageResponse, CancelRunResponse, RequestAgentPauseOptions, RequestAgentPauseResponse, SendAgentMessageOptions, SendAgentMessageResponse, AnswerSteerOptions, ResumePauseOptions, ResumePauseResponse, ResumePauseSuccess, ResumePausePending, ResumePauseActor, } from "./client.js";
32
32
  export { parseSseStream } from "./sse.js";
33
33
  export { AgentComposeError } from "./errors.js";
34
34
  export { formatError } from "./utils/errors.js";
@@ -52,7 +52,8 @@ export type { CodingTool } from "./tools/index.js";
52
52
  export type { RunEvent } from "./types/events.js";
53
53
  export { createSandbox, reconnectSandbox, killAllSandboxes, killSandboxById, getSandboxQuotas, listOwnedSandboxes, deleteSandboxSnapshot, makeSandboxProvider, makeDesktopSandboxProvider, parseSseExecStream, AGENT_COMPOSE_TAG } from "./sandbox.js";
54
54
  export { SandboxUnavailableError, SANDBOX_UNAVAILABLE_PREFIX } from "./sandbox-errors.js";
55
- export type { SandboxCreateOpts, SandboxNetworkPolicy, SandboxNetworkHeaderTransform, SandboxNetworkAllowRule, SandboxNetworkSubnetPolicy, SandboxProviderName, SandboxQuotaResult, OwnedSandboxResult, OwnedSandbox, ParseSseExecStreamOptions, SandboxCommandRunOptions, SandboxCommandResult, } from "./sandbox.js";
55
+ export type { SandboxCreateOpts, SandboxNetworkPolicy, SandboxNetworkHeaderTransform, SandboxNetworkAllowRule, SandboxNetworkSubnetPolicy, SandboxProviderName, SandboxQuotaResult, OwnedSandboxResult, OwnedSandbox, SandboxSize, ParseSseExecStreamOptions, SandboxCommandRunOptions, SandboxCommandResult, } from "./sandbox.js";
56
+ export type { SandboxResources } from "./types/workflow-metadata.js";
56
57
  export { runWorkflow, WorkflowError, EngineError, classifyError, parseNameVersion } from "./workflows/engine.js";
57
58
  export type { WorkflowResult, RunWorkflowOptions, EngineSubsystem } from "./workflows/engine.js";
58
59
  export { buildInvokeChild } from "./workflows/invoke-child.js";
@@ -62,11 +63,13 @@ export { invokeStep, serveStep, parseStepResult, buildStepEnvs, StepExecutionErr
62
63
  export { PauseError, PauseExpiredError, PauseSchemaError, PauseRequestError, } from "./pause/errors.js";
63
64
  export type { PauseErrorCode } from "./pause/errors.js";
64
65
  export type { PauseRequest } from "./pause/pause-core.js";
65
- export type { RequestDecisionRequest, WaitForEventRequest } from "./pause/wrappers.js";
66
+ export type { WaitForEventRequest } from "./pause/wrappers.js";
66
67
  export type { StepRequest, StepResult, StepInvocationError, StepPauseRequest, StepHandler, StepHandlerResult, ServeStepRequest, } from "./step-invocation/index.js";
67
68
  export { agentLoop, parseAgentStatus, DEFAULT_CLAUDE_MODEL } from "./agent/agent-loop.js";
68
69
  export type { AgentLifecycleEvent, AgentLoopOpts, AgentLoopResult } from "./agent/agent-loop.js";
69
70
  export { agent } from "./agent/run-agent.js";
70
71
  export type { AgentOpts } from "./agent/run-agent.js";
72
+ export { AGENT_COMPOSE_MANUAL, buildAgentContextDoc, writeAgentContext } from "./agent/agent-context.js";
73
+ export type { AgentConnectorInfo } from "./agent/agent-context.js";
71
74
  export { AgentMessageSchema, parseAgentResponse } from "./agent/protocol.js";
72
75
  export { importSourceModule, TMP_DIR, LATEST_VERSION } from "./utils/source-loader.js";
package/dist/index.js CHANGED
@@ -540,6 +540,8 @@ function extractMetadata(source) {
540
540
  out.outputSchema = freezeMetadataValue(source.outputSchema);
541
541
  if (source.snapshots !== undefined)
542
542
  out.snapshots = Object.freeze({ ...source.snapshots });
543
+ if (source.resources !== undefined)
544
+ out.resources = Object.freeze({ ...source.resources });
543
545
  if (source.processors !== undefined)
544
546
  out.processors = Object.freeze([...source.processors]);
545
547
  if (source.connectors !== undefined)
@@ -617,7 +619,6 @@ function compileRunForm(def, metadata) {
617
619
  agentEvents: stepCtx.agentEvents,
618
620
  checkpoint: stepCtx.checkpoint,
619
621
  pause: stepCtx.pause,
620
- requestDecision: stepCtx.requestDecision,
621
622
  sleep: stepCtx.sleep,
622
623
  waitForEvent: stepCtx.waitForEvent,
623
624
  processors: metadata.processors ?? []
@@ -812,6 +813,7 @@ class AgentComposeClient {
812
813
  ...opts?.snapshots !== undefined ? { snapshots: opts.snapshots } : {},
813
814
  ...opts?.networkPolicy !== undefined ? { networkPolicy: opts.networkPolicy } : {},
814
815
  ...opts?.placeholders !== undefined ? { placeholders: opts.placeholders } : {},
816
+ ...opts?.size !== undefined ? { size: opts.size } : {},
815
817
  ...parentRunId ? { parentRunId } : {},
816
818
  ...opts?.agentId ? { agentId: opts.agentId } : {}
817
819
  }
@@ -971,6 +973,21 @@ class AgentComposeClient {
971
973
  q2.set("revision", String(opts.revision));
972
974
  return this.fetch(`/api/v1/factories/${encodeURIComponent(factorySlug)}/files/content?${q2}`, { responseType: "text" });
973
975
  }
976
+ async listMembers() {
977
+ const body = await this.fetch("/api/v1/team/members");
978
+ return body.members;
979
+ }
980
+ async createMentions(input) {
981
+ const factorySlug = input.factorySlug ?? (typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_FACTORY : undefined) ?? DEFAULT_FACTORY;
982
+ const runToken = typeof process !== "undefined" ? process.env?.AGENT_COMPOSE_RUN_TOKEN : undefined;
983
+ const { factorySlug: _omit, ...payload } = input;
984
+ const body = await this.fetch(`/api/v1/factories/${encodeURIComponent(factorySlug)}/mentions`, {
985
+ method: "POST",
986
+ body: payload,
987
+ ...runToken ? { headers: { "x-run-token": runToken } } : {}
988
+ });
989
+ return body.mentions;
990
+ }
974
991
  async listFactoryEvents(opts) {
975
992
  const factorySlug = opts?.factorySlug ?? DEFAULT_FACTORY;
976
993
  const q2 = new URLSearchParams;
@@ -1222,6 +1239,7 @@ async function bundleWorkflow(workflowPath, overrides) {
1222
1239
  ...inputSchema !== undefined ? { inputSchema } : {},
1223
1240
  ...outputSchema !== undefined ? { outputSchema } : {},
1224
1241
  ...metadata.snapshots !== undefined ? { snapshots: metadata.snapshots } : {},
1242
+ ...metadata.resources !== undefined ? { resources: metadata.resources } : {},
1225
1243
  ...metadata.connectors !== undefined ? { connectors: metadata.connectors } : {},
1226
1244
  ...metadata.connectorOperation !== undefined ? { connectorOperation: metadata.connectorOperation } : {},
1227
1245
  ...metadata.invokePolicy !== undefined ? { invokePolicy: metadata.invokePolicy } : {}
@@ -1853,7 +1871,7 @@ async function runControlPoller(opts) {
1853
1871
  // src/agent/agent-loop.ts
1854
1872
  var DEFAULT_CLAUDE_MODEL = "claude-fable-5";
1855
1873
  var SAME_BLOCKER_ITERATIONS = 3;
1856
- var STALL_ITERATIONS = 3;
1874
+ var STALL_ITERATIONS = 6;
1857
1875
  var MESSAGE_PREVIEW_CHARS = 400;
1858
1876
  function parseAgentStatus(text) {
1859
1877
  const match = text.match(/<status>([\s\S]*?)<\/status>/);
@@ -2084,7 +2102,15 @@ Re-emit the COMPLETE corrected <response> JSON now: every required field present
2084
2102
  }
2085
2103
  const msg = outputVerdict.value;
2086
2104
  opts.onAgentEvent?.(iteration, msg);
2087
- opts.onAgentLifecycleEvent?.({ event: "agent.message", at: Date.now(), agentId, label, iteration: iteration + 1, message: summarizeAgentMessage(msg) });
2105
+ const summary = summarizeAgentMessage(msg);
2106
+ opts.onAgentLifecycleEvent?.({
2107
+ event: "agent.message",
2108
+ at: Date.now(),
2109
+ agentId,
2110
+ label,
2111
+ iteration: iteration + 1,
2112
+ message: summary.type === "usage" && client.model != null ? { ...summary, model: client.model } : summary
2113
+ });
2088
2114
  if (msg.type === "init")
2089
2115
  lastSessionId = msg.sessionId;
2090
2116
  if (msg.type === "text")
@@ -2163,9 +2189,9 @@ raw: ${JSON.stringify(rawResponse).slice(0, 400)}
2163
2189
  if (!status && rawResponse === null) {
2164
2190
  if (++iterationsWithoutStatus >= STALL_ITERATIONS)
2165
2191
  throw new Error(`${logLabel} stalled: no <status> block after ${iterationsWithoutStatus} iterations`);
2166
- if (opts.responseSchema && schemaRetriesLeft-- > 0) {
2167
- lastResponseValidationError = "no <response> block found — the turn ended with neither a <status> nor a <response> block; emit the complete <response> JSON";
2168
- process.stdout.write(`${logLabel} no <status>/<response> emitted — corrective re-prompt (${schemaRetriesLeft} retries left)
2192
+ if (schemaRetriesLeft-- > 0) {
2193
+ lastResponseValidationError = "Your turn ended with no <status> block" + (opts.responseSchema ? " (and no <response> block)" : "") + ". " + "If you launched a background job (`… &`) and are waiting to be notified when it finishes — STOP: nothing will notify you here. " + "Poll it NOW (read its output / wait for it synchronously to completion), then emit your <status>" + (opts.responseSchema ? " and the complete <response> JSON." : ".");
2194
+ process.stdout.write(`${logLabel} empty turn (no <status>) — corrective re-prompt (${schemaRetriesLeft} retries left)
2169
2195
  `);
2170
2196
  iteration--;
2171
2197
  continue;
@@ -2746,6 +2772,13 @@ class SandboxUnavailableError extends Error {
2746
2772
  // src/sandbox.ts
2747
2773
  var AGENT_COMPOSE_TAG = process.env.AGENT_COMPOSE_TAG ?? `agent-compose-${process.env.AGENT_COMPOSE_ENV ?? "dev"}`;
2748
2774
  var VERCEL_VM_LIFETIME_WINDOW_MS = 6 * 60 * 60 * 1000;
2775
+ var SANDBOX_VCPUS = {
2776
+ "2vcpu-4gb": 2,
2777
+ "4vcpu-8gb": 4,
2778
+ "8vcpu-16gb": 8,
2779
+ "32vcpu-64gb": 32
2780
+ };
2781
+ var DEFAULT_SANDBOX_SIZE = "2vcpu-4gb";
2749
2782
  function makeSandboxProvider(sb) {
2750
2783
  const commands = sb.commands;
2751
2784
  return {
@@ -2911,7 +2944,7 @@ function makeVercelSandboxProvider(sb, globalEnvs) {
2911
2944
  let stdout = "";
2912
2945
  let stderr = "";
2913
2946
  const handle = await sb.runCommand({ cmd: "sh", args: ["-c", cmd], cwd: opts?.cwd, env: mergeEnvs(opts?.envs), detached: true, signal, ...opts?.sudo ? { sudo: true } : {} }).catch((err) => asUnavailable(err, true));
2914
- try {
2947
+ const collect = async () => {
2915
2948
  await pRetry(async (attempt) => {
2916
2949
  const h = attempt === 1 ? handle : await sb.getCommand(handle.cmdId);
2917
2950
  if (attempt > 1) {
@@ -2930,11 +2963,22 @@ function makeVercelSandboxProvider(sb, globalEnvs) {
2930
2963
  }, { retries: 3, minTimeout: 1000, factor: 2 });
2931
2964
  const finished = await handle.wait();
2932
2965
  return { exitCode: finished.exitCode, stdout, stderr };
2966
+ };
2967
+ let timer;
2968
+ const deadline = opts?.timeoutMs ? new Promise((_, reject) => {
2969
+ timer = setTimeout(() => reject(new Error(`command timed out after ${opts.timeoutMs}ms: ${cmd.slice(0, 200)}`)), opts.timeoutMs);
2970
+ }) : undefined;
2971
+ try {
2972
+ return await (deadline ? Promise.race([collect(), deadline]) : collect());
2933
2973
  } catch (err) {
2934
- if (signal?.aborted) {
2974
+ if (err instanceof Error && err.message.startsWith("command timed out after"))
2975
+ throw err;
2976
+ if (signal?.aborted)
2935
2977
  throw new Error(`command timed out after ${opts?.timeoutMs}ms: ${cmd.slice(0, 200)}`);
2936
- }
2937
2978
  return asUnavailable(err, false);
2979
+ } finally {
2980
+ if (timer)
2981
+ clearTimeout(timer);
2938
2982
  }
2939
2983
  }
2940
2984
  },
@@ -3011,7 +3055,8 @@ var SANDBOX_PROVIDERS = {
3011
3055
  const np = opts.networkPolicy ? { networkPolicy: toVercelNetworkPolicy(opts.networkPolicy) } : {};
3012
3056
  const tags = buildVercelTags(opts.metadata);
3013
3057
  const tmpl = opts.template ?? process.env.VERCEL_DEFAULT_SNAPSHOT;
3014
- const sb = await VercelSandbox.create(tmpl ? { source: { type: "snapshot", snapshotId: tmpl }, timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, ...np, ...creds } : { runtime: "node24", timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, ...np, ...creds });
3058
+ const resources = { vcpus: SANDBOX_VCPUS[opts.size ?? DEFAULT_SANDBOX_SIZE] };
3059
+ const sb = await VercelSandbox.create(tmpl ? { source: { type: "snapshot", snapshotId: tmpl }, timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, resources, ...np, ...creds } : { runtime: "node24", timeout: opts.timeoutMs, env: opts.envs, tags, persistent: false, resources, ...np, ...creds });
3015
3060
  return makeVercelSandboxProvider(sb, opts.envs);
3016
3061
  },
3017
3062
  reconnect: async (sandboxId) => {
@@ -3057,7 +3102,7 @@ var SANDBOX_PROVIDERS = {
3057
3102
  },
3058
3103
  e2b: {
3059
3104
  requiredEnv: { E2B_API_KEY: "E2B API key — e2b.dev/dashboard" },
3060
- create: async ({ template, timeoutMs, networkPolicy: _np, ...rest }) => {
3105
+ create: async ({ template, timeoutMs, networkPolicy: _np, size: _size, ...rest }) => {
3061
3106
  const maxMs = e2bMaxSandboxMs();
3062
3107
  if (timeoutMs > maxMs) {
3063
3108
  console.warn(`[sandbox] e2b create timeout clamped: requested ${timeoutMs}ms exceeds the plan cap ${maxMs}ms — ` + `the sandbox dies at the cap, not the requested deadline. Raise E2B_MAX_SANDBOX_MS on plans that allow more.`);
@@ -3099,7 +3144,7 @@ var SANDBOX_PROVIDERS = {
3099
3144
  },
3100
3145
  "e2b-desktop": {
3101
3146
  requiredEnv: { E2B_API_KEY: "E2B API key — e2b.dev/dashboard" },
3102
- create: async ({ template, timeoutMs, networkPolicy: _np, ...rest }) => {
3147
+ create: async ({ template, timeoutMs, networkPolicy: _np, size: _size, ...rest }) => {
3103
3148
  if (!template)
3104
3149
  throw new Error("E2B Desktop provider requires an explicit `template` (Dockerfile-based — no default base image)");
3105
3150
  return makeDesktopSandboxProvider(await Desktop.create(template, { ...rest, timeoutMs: Math.min(timeoutMs, e2bMaxSandboxMs()) }));
@@ -3283,14 +3328,6 @@ function scopedCheckpoint(scope) {
3283
3328
  // src/pause/wrappers.ts
3284
3329
  function buildPauseWrappers(pause) {
3285
3330
  return {
3286
- requestDecision(req) {
3287
- return pause({
3288
- reason: req.reason,
3289
- schema: req.schema,
3290
- ...req.payload !== undefined ? { payload: req.payload } : {},
3291
- ...req.ttlMs !== undefined ? { ttlMs: req.ttlMs } : {}
3292
- }, "decision");
3293
- },
3294
3331
  sleep(durationMs) {
3295
3332
  return pause({
3296
3333
  reason: `sleep ${durationMs}ms`,
@@ -3600,7 +3637,7 @@ function buildInvokeChild(runId, opts = {}) {
3600
3637
  return (name, input, childOpts) => getChildClient().invokeAndWait(name, input, {
3601
3638
  ...childOpts,
3602
3639
  parentRunId: runId,
3603
- factorySlug: childOpts?.factorySlug ?? opts.defaultFactorySlug ?? process.env.AGENT_COMPOSE_FACTORY_SLUG ?? "default"
3640
+ factorySlug: childOpts?.factorySlug ?? opts.defaultFactorySlug ?? process.env.AGENT_COMPOSE_FACTORY ?? process.env.AGENT_COMPOSE_FACTORY_SLUG ?? "default"
3604
3641
  });
3605
3642
  }
3606
3643
  // src/workflow-steps/step.ts
@@ -3967,6 +4004,149 @@ async function serveStep(handler) {
3967
4004
  import { z as z11 } from "zod";
3968
4005
  import { randomUUID as randomUUID2 } from "node:crypto";
3969
4006
 
4007
+ // src/agent/agent-context.ts
4008
+ var AGENT_COMPOSE_MANUAL = `# Working inside an Agent Compose sandbox
4009
+
4010
+ You are an agent running in a per-run sandbox on the Agent Compose platform.
4011
+ Use the **\`agentc\` CLI** and the **\`@agent-compose/sdk\`** for everything below —
4012
+ do NOT hand-roll raw HTTP/curl calls against the platform API. The CLI is on
4013
+ your PATH and already authenticated from the environment
4014
+ (\`AGENT_COMPOSE_URL\` / \`AGENT_COMPOSE_API_KEY\` / \`AGENT_COMPOSE_FACTORY\` are
4015
+ injected for this run), so commands just work — no login, no keys to manage.
4016
+
4017
+ The \`/ac:*\` skills are installed as Claude Code slash commands (\`/ac:invoke\`,
4018
+ \`/ac:events\`, \`/ac:logs\`, \`/ac:register\`, …) — reach for them too.
4019
+
4020
+ ## Files — your outputs persist by default
4021
+
4022
+ Your working directory defaults to **\`$AGENT_COMPOSE_RUN_DIR\`** — a per-run
4023
+ directory on the shared factory drive
4024
+ (\`$AGENT_COMPOSE_FACTORY_DIR/<workflow>/<version>/<run-id>/\`) the platform
4025
+ creates and attributes to this run. **Files you write here persist by
4026
+ default** — they show up in the dashboard's Files tab and the run's Artifacts
4027
+ card, with no API calls to save them. The dir already exists and is writable.
4028
+
4029
+ Need throwaway scratch — heavy build output, package caches, temp files?
4030
+ \`cd /tmp\` (or any path outside \`/factory\`): anything off the factory drive is
4031
+ ephemeral and discarded when the sandbox ends. In short: **stay in your working
4032
+ dir to keep something, \`cd\` out to throw it away.**
4033
+
4034
+ The whole shared drive is POSIX-mounted at \`/factory\`; the dashboard-visible
4035
+ root is \`$AGENT_COMPOSE_FACTORY_DIR\` (\`/factory/files\`). Earlier versions and
4036
+ runs live in sibling dirs under
4037
+ \`$AGENT_COMPOSE_FACTORY_DIR/$AGENT_COMPOSE_WORKFLOW/\` — read them for prior
4038
+ context. Other workflows' dirs are present but not your concern.
4039
+
4040
+ ## Events — the factory timeline
4041
+
4042
+ Record something on the run/factory timeline (the dashboard renders these)
4043
+ with the CLI — your run id is \`$RUN_ID\`:
4044
+
4045
+ agentc events send "$RUN_ID" <name> --summary "<one line>" [--body '<json>']
4046
+
4047
+ Names like \`note.created\` / \`brief.posted\` surface in the Workbench;
4048
+ \`agentc events list\` reads them back. \`/ac:events\` is the skill equivalent.
4049
+
4050
+ ## Runs
4051
+
4052
+ agentc list # registered workflows (/ac:list)
4053
+ agentc logs "$RUN_ID" # a run's logs (/ac:logs)
4054
+ agentc invoke <workflow> -i '<json>' # dispatch a workflow (/ac:invoke)
4055
+
4056
+ ## Writing workflow / agent code — the SDK
4057
+
4058
+ \`@agent-compose/sdk\` is installed in \`/workspace\` — import it from any script
4059
+ you write there:
4060
+
4061
+ import { defineWorkflow, agent, AgentComposeClient } from "@agent-compose/sdk";
4062
+
4063
+ Use \`/ac:generate-workflow\` / \`/ac:generate-agent\` to scaffold, then
4064
+ \`agentc register <file.ts>\` (or \`/ac:register\`).
4065
+
4066
+ ## Pausing to ask the human — \`agentc pause\`
4067
+
4068
+ When you can't or shouldn't proceed without a human, run \`agentc pause\`. It
4069
+ blocks until they answer on the dashboard, then prints their answer to stdout:
4070
+
4071
+ ANSWER=$(agentc pause --reason "Notion returned 401 — connect Notion to continue" \\
4072
+ --option retry --option skip)
4073
+
4074
+ Reach for it the moment you hit — or foresee — any of these:
4075
+ - **A wall only a human can clear:** a 401/403, a missing credential, an
4076
+ unconnected provider, a host the network refuses. Do NOT retry blindly or try
4077
+ to work around it — pause and say what needs enabling.
4078
+ - **A durable or outward-facing action that needs sign-off:** registering a
4079
+ workflow, deploying, sending email/messages, deleting or overwriting shared
4080
+ data, spending money. Prepare everything, then pause for approval BEFORE you
4081
+ commit it.
4082
+ - **A judgment call only the human can settle:** an under-specified request,
4083
+ several valid paths, a conflict with existing state, missing input only they have.
4084
+
4085
+ You compose the \`--reason\` (the ask) yourself; pass \`--option\` choices when
4086
+ there are clear ones, omit them for a free-form answer. Read the printed answer
4087
+ and act on it. Each agent pauses independently — pausing doesn't stop the others.
4088
+
4089
+ ## Credentials
4090
+
4091
+ Connector credentials (Google, GitHub, …) are NEVER in your environment.
4092
+ They're injected at the network layer when you call an allowed host — make the
4093
+ request **without** an Authorization header and the platform adds it. Don't try
4094
+ to read or exfiltrate tokens; they aren't here. The "Connectors & access"
4095
+ section below (when present) lists exactly which providers this run can reach.
4096
+
4097
+ ## Tools in this environment
4098
+
4099
+ - \`agentc\` — Agent Compose CLI (your primary interface; authed from env)
4100
+ - \`@agent-compose/sdk\` — installed in /workspace for writing workflows
4101
+ - \`/ac:*\` Claude Code skills — slash commands for the above
4102
+ - \`archil\` (factory drive), \`rtk\`, \`bun\`
4103
+ - A world-writable \`/workspace\` working directory`;
4104
+ function renderConnectorsSection(connectors) {
4105
+ if (connectors.length === 0)
4106
+ return "";
4107
+ const rows = connectors.map((c) => {
4108
+ const host = c.hosts?.length ? c.hosts.join(", ") : "(host set by the platform)";
4109
+ const verbs = c.methods?.length ? c.methods.join("/") : "any method";
4110
+ const paths = c.pathPrefixes?.length ? ` under ${c.pathPrefixes.join(", ")}` : "";
4111
+ const repo = c.repository ? ` — repo \`${c.repository}\` (${c.access ?? "read"})` : "";
4112
+ const why = c.scopes?.length ? `
4113
+ _scopes: ${c.scopes.join(", ")}_` : "";
4114
+ return `- **${c.name ?? c.provider}** → \`${host}\` — ${verbs}${paths}${repo}${why}`;
4115
+ });
4116
+ return `
4117
+
4118
+ ## Connectors & access — what this run can reach
4119
+
4120
+ These providers are connected for this run. Call their APIs with plain
4121
+ fetch/SDKs and **no Authorization header** — the platform injects the
4122
+ credential at the network layer. Requests outside the listed method/path are
4123
+ refused (403) and the token withheld. Anything NOT listed is unreachable; if
4124
+ you need it, \`agentc pause\` and ask for it to be connected.
4125
+
4126
+ ${rows.join(`
4127
+ `)}
4128
+ `;
4129
+ }
4130
+ function buildAgentContextDoc(env) {
4131
+ let connectors = [];
4132
+ const raw = env.AGENT_COMPOSE_CONNECTORS;
4133
+ if (raw) {
4134
+ try {
4135
+ const parsed = JSON.parse(raw);
4136
+ if (Array.isArray(parsed))
4137
+ connectors = parsed;
4138
+ } catch {}
4139
+ }
4140
+ return AGENT_COMPOSE_MANUAL + renderConnectorsSection(connectors);
4141
+ }
4142
+ async function writeAgentContext(args) {
4143
+ const doc = buildAgentContextDoc(args.env);
4144
+ const dir = args.cwd.replace(/\/+$/, "") || "/workspace";
4145
+ for (const name of ["AGENTS.md", "CLAUDE.md", "GEMINI.md"]) {
4146
+ await args.sandbox.files.write(`${dir}/${name}`, doc);
4147
+ }
4148
+ }
4149
+
3970
4150
  // src/agent/protocol-suffix.md
3971
4151
  var protocol_suffix_default = '## Status Signal\n\nWhen you have finished your work or are blocked, emit a `<status>` block at the end of your response:\n\n```json\n<status>\n{\n "summary": "one sentence describing what was done or what is blocking",\n "completed": ["each acceptance criterion that is now fully met"],\n "blockers": [],\n "changed_files": ["relative/path/to/file"],\n "tests_run": true,\n "exit_signal": true\n}\n</status>\n```\n\n**Field semantics:**\n- `summary`: one sentence — what was accomplished or what is blocking\n- `completed`: acceptance criteria items that are fully done — be specific\n- `blockers`: non-empty when `exit_signal: false` — describe the exact obstacle\n- `changed_files`: relative paths of files you created or modified\n- `tests_run`: `true` if you ran any test suite (pass or fail); `false` if no tests exist or you skipped them\n- `exit_signal: true` — set when ALL acceptance criteria are met and no blockers remain\n- `exit_signal: false` — set when blocked or unfinished; `blockers` must be non-empty\n\n**If you are still actively working** and have not reached a natural stopping point, do NOT emit a `<status>` block — just keep working.\n\n**Example (done):**\n\n```json\n<status>\n{\n "summary": "Added input validation middleware to /api/tasks with tests",\n "completed": ["POST /api/tasks validates required fields", "Returns 400 with details on invalid input", "Unit tests passing"],\n "blockers": [],\n "changed_files": ["src/middleware/validate.ts", "src/routes/tasks.ts", "tests/validate.test.ts"],\n "tests_run": true,\n "exit_signal": true\n}\n</status>\n```\n\n**Example (blocked):**\n\n```json\n<status>\n{\n "summary": "Implemented middleware but tests are failing due to module resolution",\n "completed": ["Middleware created and wired into route"],\n "blockers": ["Tests fail: cannot resolve import \'./validate\' — module resolution config unclear"],\n "changed_files": ["src/middleware/validate.ts"],\n "tests_run": true,\n "exit_signal": false\n}\n</status>\n```\n';
3972
4152
 
@@ -4082,7 +4262,28 @@ function resolveAgentId(explicitId) {
4082
4262
  return `step${activeCall.stepIndex}-agent-${activeCall.callIndex}`;
4083
4263
  }
4084
4264
  async function agent(opts) {
4085
- const workingDir = opts.workingDir ?? "";
4265
+ let workingDir = opts.workingDir || process.env.AGENT_COMPOSE_RUN_DIR || "/workspace";
4266
+ const CONTEXT_WRITE_DEADLINE_MS = 15000;
4267
+ const tryWriteContext = (cwd) => {
4268
+ let timer;
4269
+ return Promise.race([
4270
+ writeAgentContext({ sandbox: opts.sandbox, cwd, env: process.env }).then(() => true),
4271
+ new Promise((resolve) => {
4272
+ timer = setTimeout(() => resolve(false), CONTEXT_WRITE_DEADLINE_MS);
4273
+ })
4274
+ ]).catch((err) => {
4275
+ console.error(`[agent] writeAgentContext(${cwd}) failed: ${err instanceof Error ? err.message : String(err)}`);
4276
+ return false;
4277
+ }).finally(() => {
4278
+ if (timer)
4279
+ clearTimeout(timer);
4280
+ });
4281
+ };
4282
+ if (!await tryWriteContext(workingDir) && workingDir !== "/workspace") {
4283
+ console.error(`[agent] working dir ${workingDir} not writable within ${CONTEXT_WRITE_DEADLINE_MS}ms ` + `(factory drive degraded?) — falling back to /workspace so the agent can run`);
4284
+ workingDir = "/workspace";
4285
+ await tryWriteContext(workingDir);
4286
+ }
4086
4287
  const agentId = resolveAgentId(opts.agentId);
4087
4288
  const callbackUrl = process.env.AGENT_COMPOSE_URL;
4088
4289
  const callbackToken = process.env.AGENT_COMPOSE_RUN_TOKEN;
@@ -4166,6 +4367,7 @@ async function agent(opts) {
4166
4367
  }
4167
4368
  export {
4168
4369
  writeTool,
4370
+ writeAgentContext,
4169
4371
  stepInputPath,
4170
4372
  serveStep,
4171
4373
  runWorkflowSteps,
@@ -4213,6 +4415,7 @@ export {
4213
4415
  bundleWorkflow,
4214
4416
  buildStepEnvs,
4215
4417
  buildInvokeChild,
4418
+ buildAgentContextDoc,
4216
4419
  bashTool,
4217
4420
  assertDefaultExportIsDefineWorkflow,
4218
4421
  amp_default as ampRuntime,
@@ -4251,6 +4454,7 @@ export {
4251
4454
  AgentComposeError,
4252
4455
  AgentComposeClient,
4253
4456
  AGENT_COMPOSE_TAG,
4457
+ AGENT_COMPOSE_MANUAL,
4254
4458
  AC_WORKFLOW_ID,
4255
4459
  AC_TEAM_ID,
4256
4460
  AC_RUN_ID,