@agent-compose/sdk 0.8.0 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/dist/agent/agent-context.d.ts +1 -1
  2. package/dist/agent/agent-loop.d.ts +8 -0
  3. package/dist/agent/run-agent.d.ts +4 -0
  4. package/dist/client.d.ts +77 -15
  5. package/dist/display.d.ts +16 -0
  6. package/dist/index.d.ts +6 -6
  7. package/dist/index.js +522 -123
  8. package/dist/runtimes/_cli-agent.d.ts +34 -7
  9. package/dist/runtimes/claude-code.d.ts +10 -8
  10. package/dist/runtimes/codex.buildcommand.test.d.ts +9 -0
  11. package/dist/runtimes/codex.d.ts +4 -1
  12. package/dist/runtimes/openai-desktop.js +507 -122
  13. package/dist/sandbox/sizes.d.ts +120 -30
  14. package/dist/sandbox.d.ts +1 -1
  15. package/dist/types/api-conversations.d.ts +198 -0
  16. package/dist/types/api-factory.d.ts +84 -7
  17. package/dist/types/api-runs.d.ts +48 -2
  18. package/dist/types/protocol.d.ts +8 -0
  19. package/dist/types/workflow-metadata.d.ts +14 -5
  20. package/dist/utils/bundler.d.ts +56 -0
  21. package/dist/workflow-steps/workflow.d.ts +7 -0
  22. package/dist/workflows/invoke-child.d.ts +18 -0
  23. package/dist/workflows/invoke-child.test.d.ts +9 -0
  24. package/package.json +2 -2
  25. package/src/agent/agent-context.ts +28 -17
  26. package/src/agent/agent-loop.ts +9 -0
  27. package/src/agent/run-agent.ts +5 -0
  28. package/src/client.ts +201 -30
  29. package/src/display.ts +61 -15
  30. package/src/index.ts +22 -9
  31. package/src/runtimes/_cli-agent.ts +302 -63
  32. package/src/runtimes/claude-code.ts +25 -15
  33. package/src/runtimes/codex.ts +19 -5
  34. package/src/sandbox/providers/e2b.ts +8 -4
  35. package/src/sandbox/sizes.ts +127 -44
  36. package/src/sandbox.ts +8 -0
  37. package/src/types/api-conversations.ts +180 -0
  38. package/src/types/api-factory.ts +89 -7
  39. package/src/types/api-runs.ts +50 -2
  40. package/src/types/protocol.ts +8 -0
  41. package/src/types/workflow-metadata.ts +15 -5
  42. package/src/utils/bundler.ts +213 -3
  43. package/src/workflow-steps/workflow.ts +7 -0
  44. package/src/workflows/invoke-child.ts +47 -11
@@ -31,6 +31,12 @@ export interface AgentMessageToolUse extends AgentMessageBase {
31
31
  toolName: string;
32
32
  toolInput: Record<string, unknown>;
33
33
  toolUseId: string;
34
+ /** The spawning subagent call's tool_use id when this call ran INSIDE a
35
+ * subagent (claude stream-json stamps `parent_tool_use_id` on every
36
+ * sidechain event) — lets renderers nest child activity under the
37
+ * Agent/Task call instead of flattening it into the parent transcript.
38
+ * Optional and additive: producers without sidechains omit it. */
39
+ parentToolUseId?: string;
34
40
  }
35
41
  export interface AgentMessageToolResult extends AgentMessageBase {
36
42
  type: "tool_result";
@@ -54,6 +60,8 @@ export interface AgentMessageToolResult extends AgentMessageBase {
54
60
  path: string;
55
61
  line?: number;
56
62
  }[];
63
+ /** Sidechain attribution, mirroring AgentMessageToolUse.parentToolUseId. */
64
+ parentToolUseId?: string;
57
65
  }
58
66
  export interface AgentMessageDone extends AgentMessageBase {
59
67
  type: "done";
@@ -146,11 +146,12 @@ export interface InvokePolicy {
146
146
  * today; kept as its own object so finer controls (disk, gpu, …) can be
147
147
  * added later without reshaping `WorkflowMetadata`. */
148
148
  export interface SandboxResources {
149
- /** Machine hardware SKU — one of the `SandboxSize` vCPU strings
150
- * (`2vcpu-4gb` | `4vcpu-8gb` | `8vcpu-16gb` | `32vcpu-64gb`). Maps to
151
- * provider specs at create time (Vercel: 2 / 4 / 8 / 32 vCPU, 2048 MB RAM
152
- * per vCPU). Omit → the smallest SKU. E2B sizing is template-defined and
153
- * ignores this. */
149
+ /** Machine hardware SKU. The vocabulary is `SANDBOX_SIZES` in
150
+ * `sandbox/sizes.ts` — the single source; do not restate it here or
151
+ * anywhere else. Maps to provider specs at create time: Vercel takes the
152
+ * vCPU count and allocates RAM at 2048 MB/vCPU; E2B has no create-time
153
+ * cpu/mem knob at all, so the size selects a PRE-BUILT per-size template
154
+ * (`E2B_TEMPLATE_SIZES`). Omit → `DEFAULT_SANDBOX_SIZE`. */
154
155
  size?: SandboxSize;
155
156
  /** Sandbox provider this workflow's runs execute on — `"vercel"` or
156
157
  * `"e2b"`. Optional and additive: omit and the run resolves to the
@@ -231,6 +232,14 @@ export interface WorkflowMetadata {
231
232
  * canonical metadata hash (frozen-metadata rule), so existing workflows are
232
233
  * not forced to re-register. */
233
234
  environmentBuild?: boolean;
235
+ /** Whether this workflow's runs need the factory drive. ABSENT ⇒
236
+ * `"required"`: on a drive-backed factory the server treats the /factory
237
+ * mount as load-bearing — a mount failure FAILS the run instead of
238
+ * silently proceeding drive-less. Declare `"none"` for a workflow that
239
+ * genuinely never touches /factory: the server skips the mount entirely
240
+ * for its runs (the explicit no-drive mode; there is no silent degrade).
241
+ * Optional + additive (frozen-metadata rule). */
242
+ factoryDrive?: "required" | "none";
234
243
  }
235
244
  /**
236
245
  * Pull the server-readable declarations off a `StepWorkflowDefinition`
@@ -18,6 +18,19 @@
18
18
  * parses or imports user source — it only validates the structured manifest
19
19
  * this function returns alongside the bundled bytes, and cross-checks the
20
20
  * manifest's `sourceHash` against the source it received.
21
+ *
22
+ * Module resolution carries two more layers, because most workflow source is
23
+ * now written by an AGENT and the bundler's error is the only feedback it
24
+ * gets (the Workflow Studio's agent authored `import … from "agentc/sdk"`
25
+ * and prod answered with bun's `Maybe you need to "bun install"?` — advice
26
+ * nobody could act on inside a sandbox with no package.json):
27
+ *
28
+ * - TOLERATE (`SDK_SPECIFIER_ALIASES` + `sdkAliasPlugin`) — near-miss
29
+ * spellings of `@agent-compose/sdk` resolve to the real package, so a
30
+ * draft already written with the wrong one builds unedited.
31
+ * - SELF-CORRECT (`explainBundleFailure`) — anything that still fails to
32
+ * resolve produces an error NAMING the real package, which an agent
33
+ * reading its own tool error can fix on the next turn.
21
34
  */
22
35
  import type { SnapshotConfig } from "../types/workflow.js";
23
36
  import type { IOSchema, ConnectorRequirements, ConnectorOperationTag, InvokePolicy, SandboxResources, DriveMergePolicy } from "../types/workflow-metadata.js";
@@ -108,7 +121,50 @@ export interface BundledWorkflow {
108
121
  /** Set by `defineSandboxEnvironment` — marks an environment build so the
109
122
  * server skips the /factory mount for its runs (#13). */
110
123
  environmentBuild?: boolean;
124
+ /** Drive requirement declared via `defineWorkflow({ factoryDrive })`.
125
+ * Absent ⇒ `"required"` (a failed /factory mount fails the run);
126
+ * `"none"` = explicit no-drive opt-out. */
127
+ factoryDrive?: "required" | "none";
111
128
  }
129
+ /** The one true import specifier for the platform SDK. */
130
+ export declare const SDK_PACKAGE = "@agent-compose/sdk";
131
+ /**
132
+ * Import specifiers that can only have MEANT `@agent-compose/sdk`, rewritten
133
+ * to it at bundle time so a draft written with the wrong spelling builds
134
+ * unedited.
135
+ *
136
+ * Every entry carries an `sdk` segment or suffix, so none of them can be a
137
+ * real third-party package a workflow might legitimately depend on: `a/b`
138
+ * forms are subpaths of packages that don't exist, and the `*-sdk` forms
139
+ * name this platform explicitly. Bare `agentc` / `agent-compose` are
140
+ * deliberately NOT aliased — those are plausible npm package names, and
141
+ * silently redirecting a real dependency is worse than a clear error.
142
+ *
143
+ * Exported for the unit test that pins the table.
144
+ */
145
+ export declare const SDK_SPECIFIER_ALIASES: readonly string[];
146
+ /** `@agent-compose/sdk` when `specifier` is a known near-miss for it, else
147
+ * null. The single authority: the plugin's regex filter is only a fast
148
+ * pre-filter, and this decides. */
149
+ export declare function resolveSdkAlias(specifier: string): string | null;
150
+ /**
151
+ * The SELF-CORRECTING layer. Turns a Bun bundle failure into a message that
152
+ * names the real fix instead of leaking Bun's internals.
153
+ *
154
+ * An unresolved import is almost always one thing: workflow source naming
155
+ * the platform SDK by a specifier that isn't its package name. Bun answers
156
+ * that with `Could not resolve: "agentc/sdk". Maybe you need to "bun
157
+ * install"?` — advice the author cannot act on (there is no package.json to
158
+ * install into; the drive holds one file). The rewritten message says the
159
+ * package name, so an agent reading its own tool error can fix the import on
160
+ * the next turn.
161
+ *
162
+ * Non-resolve failures (syntax errors, transform failures) keep their
163
+ * verbatim diagnostics — those are already actionable.
164
+ *
165
+ * Exported for the unit test; `bundleWorkflow` is the supported entrypoint.
166
+ */
167
+ export declare function explainBundleFailure(err: unknown, label: string): string;
112
168
  /**
113
169
  * Parse the bundled source and assert the default export is a CallExpression
114
170
  * to an identifier named `defineWorkflow` (or to a `.step(...).build()` chain
@@ -72,6 +72,13 @@ export interface StepWorkflowDefinition<TInput, TOutput> {
72
72
  * server skips mounting the shared factory drive for its runs (#13). See
73
73
  * `WorkflowMetadata.environmentBuild`. */
74
74
  environmentBuild?: boolean;
75
+ /** Whether this workflow's runs need the factory drive. Omit (⇒
76
+ * `"required"`) for any workflow that reads or writes /factory: a failed
77
+ * mount then FAILS the run instead of silently proceeding drive-less.
78
+ * Declare `"none"` for a drive-agnostic workflow — the server skips the
79
+ * /factory mount entirely for its runs. See
80
+ * `WorkflowMetadata.factoryDrive`. */
81
+ factoryDrive?: "required" | "none";
75
82
  }
76
83
  export declare function createStepWorkflow<TInput, TOutput>(opts: StepWorkflowDefinition<TInput, TOutput>): WorkflowBuilder<TInput, TInput>;
77
84
  /** Type guard — true when `value` is a `Workflow`. */
@@ -1,4 +1,22 @@
1
1
  import type { InvokeChild } from "../types/execution-context.js";
2
+ /**
3
+ * Derive the deterministic per-call `Idempotency-Key` for a `ctx.invokeChild`
4
+ * dispatch: `invoke-child:<parentRunId>:step<stepIndex>:<childName>:<ordinal>`.
5
+ *
6
+ * Replay safety: a replayed step re-runs its body from the top, so the k-th
7
+ * `invokeChild(name)` call in a step re-derives the SAME key (the per-scope
8
+ * ordinal counter lives on the active-step state, which resets identically on
9
+ * every (re-)entry — the `nextPauseOrdinalInActiveStep` pattern). The server's
10
+ * idempotency window then returns the original child run instead of
11
+ * double-dispatching. Outside step execution (local dev, tests) there is no
12
+ * stable coordinate to key on — returns null and the dispatch is unkeyed,
13
+ * exactly the old behavior.
14
+ *
15
+ * Key syntax matches the server's `Idempotency-Key` grammar
16
+ * (`[A-Za-z0-9_\-:.]{1,255}`): runId is a UUID, step index a number, and
17
+ * workflow names are kebab-case.
18
+ */
19
+ export declare function deriveInvokeChildIdempotencyKey(parentRunId: string, childName: string): string | null;
2
20
  /**
3
21
  * Build the public-API child workflow invoker used by legacy and sandboxed
4
22
  * workflow execution. Provider-backed engines may inject a different
@@ -0,0 +1,9 @@
1
+ /**
2
+ * Replay-safety pin for `ctx.invokeChild` (parity plan P3.3): the
3
+ * Idempotency-Key a child dispatch carries is DETERMINISTIC per
4
+ * (parentRunId, stepIndex, childName, call ordinal) — a step body re-run
5
+ * from the top (pause/resume re-entry, crash-replayed attempt) re-derives
6
+ * byte-identical keys, so the server's idempotency window collapses the
7
+ * replayed dispatch into the original child run instead of double-running.
8
+ */
9
+ export {};
package/package.json CHANGED
@@ -1,11 +1,11 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.8.0",
3
+ "version": "0.8.2",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
7
7
  "type": "git",
8
- "url": "git+https://github.com/Layr-Labs/agent-compose.git",
8
+ "url": "git+https://github.com/Chris-Moller/agent-compose.git",
9
9
  "directory": "sdk"
10
10
  },
11
11
  "type": "module",
@@ -173,38 +173,49 @@ there is nothing to type, paste, or screenshot a credential from.
173
173
 
174
174
  ## Recording a demo — the desktop, captured to a video the human can play
175
175
 
176
- "Record a demo of you using X" is a normal ask, and this machine does it:
177
- start a screen recording, drive the app with \`xdotool\` exactly as in Computer
178
- Use, stop the recording, and report the file. (For a LIVE view no recording is
179
- needed — the session header's **Desktop** button already streams this display
180
- to any teammate watching; a recording is the durable, replayable artifact.
181
- Both modes exist; say so when it matters.)
176
+ "Record a demo of you using X" is a normal ask, and this machine does it.
177
+ (For a LIVE view no recording is needed — the session header's **Desktop**
178
+ button already streams this display to any teammate watching; a recording is
179
+ the durable, replayable artifact. Both modes exist; say so when it matters.)
182
180
 
183
- **ffmpeg is NOT pre-installed** — install it first, once per machine:
181
+ **Use \`ac-record\` — the platform recorder is already on PATH** (cloud
182
+ sessions; \`command -v ac-record\` to confirm on older machines):
184
183
 
185
- sudo apt-get update -q && sudo apt-get install -y -q ffmpeg
184
+ ac-record start # begins capturing the desktop (display :0)
185
+ # ... drive the app with xdotool, screenshotting as you go ...
186
+ ac-record stop # finishes + saves to recordings/ in your workspace
187
+ ac-record status # one JSON line: {"recording":true,...}
188
+
189
+ It records the whole display (with desktop audio when the machine has a
190
+ PulseAudio monitor), enforces sane caps (5 min / 200 MB — start a fresh
191
+ recording per scene rather than one long take), keeps the file playable even
192
+ if the machine dies mid-take, and \`stop\` prints the saved path — the file
193
+ lands ON THE DRIVE in \`recordings/\`, visible in Files and playable in the
194
+ dashboard. A human watching the Desktop pane sees the recording indicator
195
+ while you record.
186
196
 
187
- (drop \`sudo\` if you are already root). Then the whole recipe:
197
+ If \`ac-record\` is missing (older machine), record by hand.
198
+ **ffmpeg IS pre-installed** on platform images (\`command -v ffmpeg\`; only
199
+ if absent: \`sudo apt-get update -q && sudo apt-get install -y -q ffmpeg\`):
188
200
 
189
201
  DISPLAY=:0 ffmpeg -f x11grab \\
190
202
  -video_size "$(DISPLAY=:0 xdotool getdisplaygeometry | tr ' ' x)" \\
191
203
  -framerate 10 -i :0 -c:v libvpx -b:v 1M -deadline realtime -cpu-used 8 \\
192
204
  demo.webm &
193
205
  FFMPEG_PID=$!
194
- # ... drive the app with xdotool, screenshotting as you go ...
206
+ # ... drive the app with xdotool ...
195
207
  kill -INT "$FFMPEG_PID" && wait "$FFMPEG_PID"
196
208
 
197
- The gotchas, each one earned:
209
+ The hand-rolled gotchas, each one earned:
198
210
  - **Stop with SIGINT (\`kill -INT\`), never SIGKILL** — ffmpeg finalizes the
199
211
  file on SIGINT; a hard kill truncates the encode mid-write.
200
- - **Record WebM (matroska-family), not MP4** — mp4 writes its moov atom at the
201
- END, so a killed or crashed encode leaves an UNPLAYABLE file; webm stays
202
- playable up to the last written frame and plays natively in the browser.
203
- MP4's only edge is compatibility with some external players — transcode
204
- afterwards if you truly need it, never record straight to it.
212
+ - **Record WebM (matroska-family), not plain MP4** — mp4 writes its moov atom
213
+ at the END, so a killed or crashed encode leaves an UNPLAYABLE file; webm
214
+ stays playable up to the last written frame and plays natively in the
215
+ browser. (\`ac-record\` sidesteps this with fragmented mp4.)
205
216
  - **\`-video_size\` must match the real screen** — x11grab does not default to
206
217
  it; read the geometry from \`xdotool getdisplaygeometry\` as above.
207
- - **10–12 fps is right for a screen demo** — small files, legible UI motion;
218
+ - **10–15 fps is right for a screen demo** — small files, legible UI motion;
208
219
  this is not video production.
209
220
  - **Write to the drive, not /tmp** — the recording must land in your working
210
221
  directory to persist and show up in Files; a file in /tmp dies with the
@@ -126,6 +126,11 @@ export type AgentLifecycleEvent =
126
126
  /** Short runtime self-identifier (`claude`, `openai-desktop`, …).
127
127
  * Drives the per-agent runtime icon on the dashboard. */
128
128
  runtimeKind?: string;
129
+ /** Authored plan-phase this agent belongs to (e.g. dynamic-task's
130
+ * `phase.name`). Purely observability: the dashboard groups agents
131
+ * under named phases without parsing the label prefix. Optional and
132
+ * additive — absent for agents outside a phased plan. */
133
+ phase?: string;
129
134
  }
130
135
  | { event: "agent.message"; at: number; agentId: string; label: string; iteration: number; message: AgentMessageSummary }
131
136
  | { event: "agent.iteration"; at: number; agentId: string; label: string; iteration: number; status: AgentStatus | null }
@@ -134,6 +139,9 @@ export type AgentLifecycleEvent =
134
139
  export interface AgentLoopOpts<TResponse = unknown> {
135
140
  agentId?: string;
136
141
  label?: string;
142
+ /** Authored plan-phase name carried onto `agent.spawned` (see
143
+ * AgentLifecycleEvent.phase). Optional, observability-only. */
144
+ phase?: string;
137
145
  onAgentLifecycleEvent?: (event: AgentLifecycleEvent) => void;
138
146
  onIteration?: (iteration: number, status: AgentStatus | null) => void;
139
147
  turnsPerIteration?: number;
@@ -307,6 +315,7 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
307
315
  allowedTools: opts.allowedTools ?? DEFAULT_ALLOWED_TOOLS,
308
316
  ...(client.model != null ? { model: client.model } : {}),
309
317
  ...(client.kind != null ? { runtimeKind: client.kind } : {}),
318
+ ...(opts.phase != null ? { phase: opts.phase } : {}),
310
319
  });
311
320
  }
312
321
 
@@ -183,6 +183,10 @@ export interface AgentOpts<T = unknown> {
183
183
  responseSchema?: z.ZodType<T>;
184
184
  /** Label prefix for runtime stderr ("[sbid][agent]" by default). */
185
185
  label?: string;
186
+ /** Authored plan-phase this agent belongs to. Rides `agent.spawned` so the
187
+ * dashboard groups agents under named phases. Optional, additive,
188
+ * observability-only — it never affects execution. */
189
+ phase?: string;
186
190
  /** Lifecycle event sink from workflow ctx. Emits agent.spawned / iteration / settled. */
187
191
  events?: { emit: (event: AgentLifecycleEvent) => void | Promise<void> };
188
192
  /** Per-message event callback — wire this to your workflow's event
@@ -349,6 +353,7 @@ export async function agent<T = unknown>(opts: AgentOpts<T>): Promise<AgentLoopR
349
353
  runtime: (runtimeOpts: RuntimeOptions) => opts.runtime.create(opts.sandbox, { ...runtimeOpts, agentManual: buildAgentContextDoc(process.env) }),
350
354
  agentId,
351
355
  ...(opts.label !== undefined ? { label: opts.label } : {}),
356
+ ...(opts.phase !== undefined ? { phase: opts.phase } : {}),
352
357
  ...(opts.budget?.turnsPerIteration !== undefined ? { turnsPerIteration: opts.budget.turnsPerIteration } : {}),
353
358
  ...(opts.budget?.maxIterations !== undefined ? { maxIterations: opts.budget.maxIterations } : {}),
354
359
  ...(opts.tools !== undefined ? { allowedTools: opts.tools } : {}),