@agent-compose/sdk 0.8.4 → 0.8.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +213 -189
- package/dist/agent/agent-context.d.ts +9 -1
- package/dist/agent/agent-loop.d.ts +14 -6
- package/dist/agent/perf-sampler.d.ts +27 -2
- package/dist/agent/run-agent.d.ts +1 -1
- package/dist/client.d.ts +250 -59
- package/dist/directives.d.ts +14 -0
- package/dist/display.d.ts +7 -0
- package/dist/errors.d.ts +1 -1
- package/dist/generated/agentc-commands.d.ts +34 -0
- package/dist/index.d.ts +13 -11
- package/dist/index.js +1692 -194
- package/dist/request-context/request-context.d.ts +1 -1
- package/dist/runtimes/_cli-agent.d.ts +278 -58
- package/dist/runtimes/claude-code.d.ts +90 -1
- package/dist/runtimes/claude.d.ts +1 -1
- package/dist/runtimes/codex.d.ts +94 -6
- package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
- package/dist/runtimes/openai-desktop.d.ts +50 -0
- package/dist/runtimes/openai-desktop.js +1689 -211
- package/dist/runtimes/openai-desktop.test.d.ts +20 -0
- package/dist/runtimes/opencode.d.ts +48 -11
- package/dist/runtimes/opencode.test.d.ts +14 -0
- package/dist/runtimes/tool-pulse.test.d.ts +17 -0
- package/dist/sandbox/baked-clis.d.ts +75 -0
- package/dist/sandbox/devbox.d.ts +5 -5
- package/dist/sandbox/exec-stream.d.ts +1 -2
- package/dist/sandbox/network-policy.d.ts +23 -5
- package/dist/sandbox/registry.d.ts +12 -0
- package/dist/sandbox/sizes.d.ts +11 -5
- package/dist/sandbox.d.ts +5 -3
- package/dist/step-invocation/protocol.d.ts +3 -4
- package/dist/step-invocation/server.d.ts +2 -2
- package/dist/step-invocation/types.d.ts +2 -2
- package/dist/types/api-conversations.d.ts +513 -27
- package/dist/types/api-factory.d.ts +183 -3
- package/dist/types/api-projects.d.ts +480 -0
- package/dist/types/api-runs.d.ts +8 -0
- package/dist/types/api-scopes.d.ts +32 -3
- package/dist/types/conversation-stream.d.ts +27 -1
- package/dist/types/execution-context.d.ts +1 -1
- package/dist/types/protocol.d.ts +182 -2
- package/dist/types/runtime.d.ts +80 -2
- package/dist/types/workflow-metadata.d.ts +2 -4
- package/dist/types/workflow-plan.d.ts +1 -3
- package/dist/utils/bundler.d.ts +23 -0
- package/dist/workflow-steps/observability.d.ts +2 -3
- package/dist/workflow-steps/runner.d.ts +5 -8
- package/dist/workflow-steps/types.d.ts +8 -10
- package/dist/workflow-steps/workflow.d.ts +2 -1
- package/dist/workflows/engine.d.ts +3 -5
- package/dist/workflows/invoke-child.d.ts +2 -2
- package/package.json +2 -2
- package/src/agent/agent-context.ts +193 -116
- package/src/agent/agent-loop.ts +16 -9
- package/src/agent/desktop-open.ts +13 -1
- package/src/agent/perf-sampler.ts +54 -3
- package/src/agent/run-agent.ts +1 -1
- package/src/client.ts +418 -80
- package/src/directives.ts +21 -1
- package/src/display.ts +12 -0
- package/src/errors.ts +1 -0
- package/src/generated/agentc-commands.ts +571 -0
- package/src/index.ts +65 -18
- package/src/pause/pause-core.ts +2 -1
- package/src/request-context/request-context.ts +1 -1
- package/src/runtimes/_cli-agent.ts +607 -132
- package/src/runtimes/claude-code.ts +427 -20
- package/src/runtimes/claude.ts +1 -1
- package/src/runtimes/codex.ts +188 -19
- package/src/runtimes/openai-desktop.ts +82 -19
- package/src/runtimes/opencode.ts +195 -26
- package/src/sandbox/baked-clis.ts +86 -0
- package/src/sandbox/devbox.ts +5 -5
- package/src/sandbox/exec-stream.ts +1 -2
- package/src/sandbox/network-policy.ts +51 -7
- package/src/sandbox/providers/e2b.ts +63 -19
- package/src/sandbox/providers/vercel.ts +6 -6
- package/src/sandbox/registry.ts +19 -1
- package/src/sandbox/sizes.ts +11 -5
- package/src/sandbox.ts +9 -2
- package/src/step-invocation/invoker.ts +2 -6
- package/src/step-invocation/protocol.ts +3 -4
- package/src/step-invocation/server.ts +2 -2
- package/src/types/api-conversations.ts +424 -29
- package/src/types/api-factory.ts +189 -3
- package/src/types/api-projects.ts +443 -0
- package/src/types/api-runs.ts +5 -0
- package/src/types/api-scopes.ts +32 -3
- package/src/types/conversation-stream.ts +29 -1
- package/src/types/execution-context.ts +1 -1
- package/src/types/protocol.ts +180 -2
- package/src/types/runtime.ts +71 -2
- package/src/types/sandbox-environment.ts +1 -2
- package/src/types/workflow-metadata.ts +2 -4
- package/src/types/workflow-plan.ts +1 -3
- package/src/utils/bundler.ts +88 -19
- package/src/workflow-steps/observability.ts +2 -3
- package/src/workflow-steps/runner.ts +5 -8
- package/src/workflow-steps/types.ts +8 -10
- package/src/workflow-steps/workflow.ts +2 -1
- package/src/workflows/engine.ts +3 -5
- package/src/workflows/invoke-child.ts +2 -2
- package/dist/pause/__tests__/errors.test.d.ts +0 -1
- package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
- package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
package/dist/types/protocol.d.ts
CHANGED
|
@@ -71,6 +71,21 @@ export interface AgentMessageError extends AgentMessageBase {
|
|
|
71
71
|
type: "error";
|
|
72
72
|
text: string;
|
|
73
73
|
}
|
|
74
|
+
/** One model's share of a harness's turn-end report (claude-code
|
|
75
|
+
* `result.modelUsage[<model>]`): the same four token classes as the turn
|
|
76
|
+
* totals, plus what the harness reports beside them. `costUsd` is the
|
|
77
|
+
* harness's OWN estimate at its price table — never a bill. Fields the
|
|
78
|
+
* harness did not report are absent, never zeroed. */
|
|
79
|
+
export interface AgentMessageModelUsage {
|
|
80
|
+
inputTokens: number;
|
|
81
|
+
outputTokens: number;
|
|
82
|
+
cacheReadTokens: number;
|
|
83
|
+
cacheCreationTokens: number;
|
|
84
|
+
/** Thinking tokens, already counted inside `outputTokens`. */
|
|
85
|
+
thinkingTokens?: number;
|
|
86
|
+
webSearchRequests?: number;
|
|
87
|
+
costUsd?: number;
|
|
88
|
+
}
|
|
74
89
|
export interface AgentMessageUsage extends AgentMessageBase {
|
|
75
90
|
type: "usage";
|
|
76
91
|
inputTokens: number;
|
|
@@ -79,6 +94,53 @@ export interface AgentMessageUsage extends AgentMessageBase {
|
|
|
79
94
|
cacheCreationTokens: number;
|
|
80
95
|
durationMs: number;
|
|
81
96
|
numTurns: number;
|
|
97
|
+
/** Reasoning tokens, already counted inside `outputTokens` (codex
|
|
98
|
+
* `turn.completed.usage.reasoning_output_tokens`). Absent when the
|
|
99
|
+
* harness reports no such class. */
|
|
100
|
+
reasoningOutputTokens?: number;
|
|
101
|
+
/** Per-model totals the harness reported beside the turn totals
|
|
102
|
+
* (claude-code `result.modelUsage`): every model the query pipeline
|
|
103
|
+
* called — main loop, subagents, compaction. As the CLI reports them:
|
|
104
|
+
* CUMULATIVE for the guest session (a streaming-input or resumed
|
|
105
|
+
* session carries its earlier turns), so a per-turn share is the
|
|
106
|
+
* difference from the previous report of the same session. Absent when
|
|
107
|
+
* the harness reports none. */
|
|
108
|
+
byModel?: Record<string, AgentMessageModelUsage>;
|
|
109
|
+
/** The harness's own cost estimate for the same scope as `byModel`
|
|
110
|
+
* (claude-code `result.total_cost_usd`): list-price arithmetic, an
|
|
111
|
+
* estimate and never a billing statement. */
|
|
112
|
+
costUsd?: number;
|
|
113
|
+
}
|
|
114
|
+
/** The harness's reading of the account's PLAN LIMITS (claude-code
|
|
115
|
+
* `rate_limit_event`, emitted whenever its rate-limit information changes
|
|
116
|
+
* — subscription-funded sessions only; the headers it reads exist for
|
|
117
|
+
* claude.ai plans). `status` is the verdict for the request just made:
|
|
118
|
+
* `rejected` means the plan's wall, with `resetsAt` the authoritative
|
|
119
|
+
* reset. `window` names which window the verdict speaks for (the CLI's
|
|
120
|
+
* `rateLimitType`: five_hour, seven_day, seven_day_opus, …). `utilization`
|
|
121
|
+
* is carried only once a window crosses a warning threshold (the CLI omits
|
|
122
|
+
* it while plainly allowed), as the CLI reports it — a 0..1 fraction of
|
|
123
|
+
* the window. Everything the event did not carry is absent; nothing is
|
|
124
|
+
* invented. Additive kind: existing producers never emit it. */
|
|
125
|
+
export interface AgentMessagePlanLimits extends AgentMessageBase {
|
|
126
|
+
type: "plan_limits";
|
|
127
|
+
status: "allowed" | "allowed_warning" | "rejected";
|
|
128
|
+
window?: string;
|
|
129
|
+
/** ISO time the named window resets. */
|
|
130
|
+
resetsAt?: string;
|
|
131
|
+
utilization?: number;
|
|
132
|
+
/** The warning threshold the window crossed (as the CLI reports it). */
|
|
133
|
+
surpassedThreshold?: number;
|
|
134
|
+
/** The plan's extra-usage (overage) lane, when the event spoke of it. */
|
|
135
|
+
overage?: {
|
|
136
|
+
status?: "allowed" | "allowed_warning" | "rejected";
|
|
137
|
+
resetsAt?: string;
|
|
138
|
+
disabledReason?: string;
|
|
139
|
+
inUse?: boolean;
|
|
140
|
+
};
|
|
141
|
+
/** Which spend limit blocked the request when not the member's own cap. */
|
|
142
|
+
limitScope?: string;
|
|
143
|
+
errorCode?: string;
|
|
82
144
|
}
|
|
83
145
|
/** LIVE-ONLY incremental usage off the harness's raw provider stream — the
|
|
84
146
|
* `text_delta` of token counts. claude-code's `--include-partial-messages`
|
|
@@ -109,7 +171,8 @@ export interface AgentMessagePlan extends AgentMessageBase {
|
|
|
109
171
|
type: "plan";
|
|
110
172
|
entries: {
|
|
111
173
|
content: string;
|
|
112
|
-
|
|
174
|
+
/** ACP names one; codex's `--json` plan names none. */
|
|
175
|
+
priority?: "high" | "medium" | "low";
|
|
113
176
|
status: "pending" | "in_progress" | "completed";
|
|
114
177
|
}[];
|
|
115
178
|
}
|
|
@@ -144,6 +207,71 @@ export interface AgentMessageTaskNotification extends AgentMessageBase {
|
|
|
144
207
|
durationMs?: number;
|
|
145
208
|
};
|
|
146
209
|
}
|
|
210
|
+
/** One entry of a harness workflow's live progress feed — the
|
|
211
|
+
* `workflow_progress` array claude-code's `system`/`task_progress` events
|
|
212
|
+
* carry for its in-harness Workflow tool (the dynamic-workflow
|
|
213
|
+
* orchestrator). `phase` entries are the script's declared phases (seeded
|
|
214
|
+
* up front, 1-based `index`); `agent` entries are the spawned workflow
|
|
215
|
+
* agents, updated in place as they queue → run → settle. Preview text the
|
|
216
|
+
* harness includes (prompt/result previews) is internal plumbing and is
|
|
217
|
+
* deliberately NOT forwarded — same rule as task notifications. */
|
|
218
|
+
export type WorkflowProgressEntry = {
|
|
219
|
+
kind: "phase";
|
|
220
|
+
index: number;
|
|
221
|
+
title: string;
|
|
222
|
+
} | {
|
|
223
|
+
kind: "agent";
|
|
224
|
+
index: number;
|
|
225
|
+
label: string;
|
|
226
|
+
/** Lifecycle: `start` (spawned) / `progress` (heartbeat) are live;
|
|
227
|
+
* `done` / `error` are settled. Verbatim from the harness. */
|
|
228
|
+
state: "start" | "progress" | "done" | "error";
|
|
229
|
+
phaseIndex?: number;
|
|
230
|
+
phaseTitle?: string;
|
|
231
|
+
model?: string;
|
|
232
|
+
agentId?: string;
|
|
233
|
+
/** Self-reported usage so far (cumulative for this agent). */
|
|
234
|
+
tokens?: number;
|
|
235
|
+
toolCalls?: number;
|
|
236
|
+
durationMs?: number;
|
|
237
|
+
/** Epoch ms the agent actually started (for live elapsed). */
|
|
238
|
+
startedAt?: number;
|
|
239
|
+
/** The failure message when `state: "error"`, clamped. */
|
|
240
|
+
error?: string;
|
|
241
|
+
/** Replayed from a resume cache — settled instantly, no fresh spend. */
|
|
242
|
+
cached?: true;
|
|
243
|
+
/** Skipped by the user (workflow dialog) — an error state that is
|
|
244
|
+
* not a failure. */
|
|
245
|
+
skipped?: true;
|
|
246
|
+
};
|
|
247
|
+
/** LIVE progress of a harness BACKGROUND task (claude-code
|
|
248
|
+
* `system`/`task_progress`). For the in-harness Workflow tool the message
|
|
249
|
+
* carries the CUMULATIVE `workflow_progress` entry array (the harness
|
|
250
|
+
* re-sends the whole picture: state changes immediately, heartbeats
|
|
251
|
+
* throttled ~10s), so a consumer treats the latest message as
|
|
252
|
+
* authoritative per entry `(kind, index)`. A plain background AGENT task
|
|
253
|
+
* (Task tool, run_in_background) heartbeats on the same event with NO
|
|
254
|
+
* workflow entries — forwarded with `workflow: []` as liveness evidence
|
|
255
|
+
* so the platform can declare the running task as session background
|
|
256
|
+
* work; its completion evidence rides `task_notification`. `toolUseId`
|
|
257
|
+
* names the spawning call — the correlation key to its card. Additive
|
|
258
|
+
* kind: existing producers never emit it. */
|
|
259
|
+
export interface AgentMessageTaskProgress extends AgentMessageBase {
|
|
260
|
+
type: "task_progress";
|
|
261
|
+
/** The harness's background task id. */
|
|
262
|
+
taskId: string;
|
|
263
|
+
/** The SPAWNING Workflow/Task call's tool_use id, when carried. */
|
|
264
|
+
toolUseId?: string;
|
|
265
|
+
/** The task's cumulative usage totals so far. */
|
|
266
|
+
usage?: {
|
|
267
|
+
tokens?: number;
|
|
268
|
+
toolUses?: number;
|
|
269
|
+
durationMs?: number;
|
|
270
|
+
};
|
|
271
|
+
/** The cumulative workflow progress entries — EMPTY for a plain
|
|
272
|
+
* background Agent-task heartbeat (only Workflow tasks carry entries). */
|
|
273
|
+
workflow: WorkflowProgressEntry[];
|
|
274
|
+
}
|
|
147
275
|
/** HARNESS-authored notice text — content the CLI composed itself rather
|
|
148
276
|
* than the model speaking: slash-command stdout, model/skills advisories,
|
|
149
277
|
* queued-input notes. claude-code marks these structurally (assistant
|
|
@@ -156,7 +284,59 @@ export interface AgentMessageHarnessNotice extends AgentMessageBase {
|
|
|
156
284
|
type: "harness_notice";
|
|
157
285
|
text: string;
|
|
158
286
|
}
|
|
159
|
-
|
|
287
|
+
/** Context-compaction lifecycle — the harness summarizing its own
|
|
288
|
+
* conversation to reclaim context. Mapped 1:1 from claude-code's wire
|
|
289
|
+
* (verified live on 2.1.212, the baked sandbox pin, and 2.1.241 — both
|
|
290
|
+
* emit the identical shapes, `/compact` and auto alike):
|
|
291
|
+
*
|
|
292
|
+
* `system`/`status` `{status:"compacting"}` → phase "start"
|
|
293
|
+
* `system`/`status` `{status:null, compact_result, → phase "settled"
|
|
294
|
+
* compact_error?}`
|
|
295
|
+
* `system`/`compact_boundary` `{compact_metadata: → phase "boundary"
|
|
296
|
+
* {trigger, pre_tokens, post_tokens,
|
|
297
|
+
* cumulative_dropped_tokens, duration_ms, …}}`
|
|
298
|
+
*
|
|
299
|
+
* Order on the wire: start → settled → (fresh init) → boundary → the
|
|
300
|
+
* continuation summary as a SYNTHETIC user message (never forwarded — it
|
|
301
|
+
* quotes conversation content verbatim). "boundary" only follows a
|
|
302
|
+
* successful settle and only on the turn the compaction ran (verified: it
|
|
303
|
+
* does NOT replay on later resumes). A compaction can span MINUTES of
|
|
304
|
+
* otherwise-silent stream — the whole point of forwarding it is that
|
|
305
|
+
* downstream can show the silence as work (the 2026-08-23 dead-air
|
|
306
|
+
* incident: 94% auto-compact read as a dead session). Additive kind:
|
|
307
|
+
* existing producers never emit it. */
|
|
308
|
+
export interface AgentMessageCompaction extends AgentMessageBase {
|
|
309
|
+
type: "compaction";
|
|
310
|
+
phase: "start" | "settled" | "boundary";
|
|
311
|
+
/** settled: how it ended. Absent on start/boundary (a boundary IS a
|
|
312
|
+
* success by construction — the harness only emits it after one). */
|
|
313
|
+
result?: "success" | "failed";
|
|
314
|
+
/** settled+failed: the harness's own reason, clamped. */
|
|
315
|
+
error?: string;
|
|
316
|
+
/** boundary: what initiated the compaction. */
|
|
317
|
+
trigger?: "auto" | "manual";
|
|
318
|
+
/** boundary: context tokens before / after, dropped total, wall time. */
|
|
319
|
+
preTokens?: number;
|
|
320
|
+
postTokens?: number;
|
|
321
|
+
droppedTokens?: number;
|
|
322
|
+
durationMs?: number;
|
|
323
|
+
}
|
|
324
|
+
/** A user-role message landing INSIDE a subagent's thread — the delivered
|
|
325
|
+
* form of a steer (the parent's `SendMessage` to a RUNNING child, queued
|
|
326
|
+
* "for delivery at its next tool round") or any other message the harness
|
|
327
|
+
* folds into a child's conversation mid-flight. Emitted ONLY with sidechain
|
|
328
|
+
* attribution: `parentToolUseId` (the spawning Agent/Task call's tool_use
|
|
329
|
+
* id) is REQUIRED — an unattributed user event is the parent's own prompt
|
|
330
|
+
* echo, which stays unmapped as before. Lets renderers show the steer as a
|
|
331
|
+
* user-role message inside the child's mini-session instead of leaving it
|
|
332
|
+
* an opaque SendMessage tool call on the parent only (task #97, owner
|
|
333
|
+
* directive 2026-08-27). Additive kind: existing producers never emit it. */
|
|
334
|
+
export interface AgentMessageSubagentUserMessage extends AgentMessageBase {
|
|
335
|
+
type: "subagent_user_message";
|
|
336
|
+
text: string;
|
|
337
|
+
parentToolUseId: string;
|
|
338
|
+
}
|
|
339
|
+
export type AgentMessage = AgentMessageInit | AgentMessageText | AgentMessageTextDelta | AgentMessageThinking | AgentMessageToolUse | AgentMessageToolResult | AgentMessageDone | AgentMessageError | AgentMessageUsage | AgentMessageUsageDelta | AgentMessagePlanLimits | AgentMessagePlan | AgentMessageTaskNotification | AgentMessageTaskProgress | AgentMessageHarnessNotice | AgentMessageCompaction | AgentMessageSubagentUserMessage;
|
|
160
340
|
/** Status block the agent emits to signal iteration completion or blockers. */
|
|
161
341
|
export interface AgentStatus {
|
|
162
342
|
summary: string;
|
package/dist/types/runtime.d.ts
CHANGED
|
@@ -93,6 +93,20 @@ export interface RuntimeOptions {
|
|
|
93
93
|
* Absent ⇒ nothing extra is sourced. Must be a plain absolute path — no
|
|
94
94
|
* quotes, no `..`; the runtime validates and drops anything else. */
|
|
95
95
|
credEnvFile?: string;
|
|
96
|
+
/** Durable launch report (boot-time turn adoption): called by the durable
|
|
97
|
+
* detached transport the moment its in-guest runner exists — with the
|
|
98
|
+
* guest prompt path (every durable file derives from it), the exit
|
|
99
|
+
* sentinel string, and the detached wrapper's pid. The caller stamps
|
|
100
|
+
* these on the turn row so a SUCCESSOR process (a deploy roll's new
|
|
101
|
+
* server) can re-attach to the runner's durable `.out` without this
|
|
102
|
+
* process's memory. Fired once per detached launch (a retry with a fresh
|
|
103
|
+
* runner fires again with the new paths); never on transports without
|
|
104
|
+
* durable files. Must not throw — the transport calls it inline. */
|
|
105
|
+
onDetachedLaunch?: (info: {
|
|
106
|
+
promptPath: string;
|
|
107
|
+
sentinel: string;
|
|
108
|
+
pid: number;
|
|
109
|
+
}) => void;
|
|
96
110
|
}
|
|
97
111
|
/** Three-valued liveness verdict for a runtime's CURRENT turn, read from
|
|
98
112
|
* DURABLE guest state (heartbeat file, stdout file, exit sentinel, pid) over
|
|
@@ -184,6 +198,25 @@ export interface ModelExecutionContract {
|
|
|
184
198
|
* fallback probe it already had; null is never a verdict.
|
|
185
199
|
*/
|
|
186
200
|
probeTurnLiveness?(): Promise<RunnerLivenessVerdict | null>;
|
|
201
|
+
/**
|
|
202
|
+
* TELEMETRY, NEVER A VERDICT (tool-run pulse, 2026-08-29): the newest
|
|
203
|
+
* tool-run pulse the durable liveness probe carried — the guest
|
|
204
|
+
* heartbeat's sample of the runner's own session (aggregate CPU jiffies,
|
|
205
|
+
* written bytes, live process count) plus the guest clock it was read
|
|
206
|
+
* against. The executor's evidence ticker peeks it AFTER its liveness
|
|
207
|
+
* check and compares successive samples: counters ADVANCING is proof a
|
|
208
|
+
* long silent foreground tool is working, fanned to clients as a live
|
|
209
|
+
* `tool_pulse` frame. Never probes on its own; null before any probe or
|
|
210
|
+
* on a transport without the pulse file. No liveness decision may ever
|
|
211
|
+
* read it.
|
|
212
|
+
*/
|
|
213
|
+
peekTurnPulse?(): {
|
|
214
|
+
atMs: number;
|
|
215
|
+
cpuJiffies: number;
|
|
216
|
+
ioBytes: number;
|
|
217
|
+
procs: number;
|
|
218
|
+
guestNowMs: number;
|
|
219
|
+
} | null;
|
|
187
220
|
/**
|
|
188
221
|
* DOORBELL, NEVER A VERDICT (exit-event push, v0.10.43): wake the current
|
|
189
222
|
* turn's durable watchdog NOW so it runs its normal verification pass —
|
|
@@ -233,7 +266,52 @@ export interface ModelExecutionContract {
|
|
|
233
266
|
* phase, a spec/transport without stream input, ACP path).
|
|
234
267
|
* Calls are serialized per turn; never throws.
|
|
235
268
|
*/
|
|
236
|
-
injectUserMessage?(text: string
|
|
269
|
+
injectUserMessage?(text: string,
|
|
270
|
+
/** Who the message speaks for when it is not the session user
|
|
271
|
+
* (MidTurnEnvelopeOptions): a relayed person, named, or the thread
|
|
272
|
+
* agent that owns this worker. */
|
|
273
|
+
opts?: {
|
|
274
|
+
relayedFrom?: string | null;
|
|
275
|
+
fromOwnerAgent?: boolean;
|
|
276
|
+
}): Promise<"delivered" | "pending" | "closed" | "unsupported">;
|
|
277
|
+
/**
|
|
278
|
+
* Request an in-band STEP INTERRUPT of the currently running turn — the
|
|
279
|
+
* ESC equivalent. Where `injectUserMessage` queues content for the turn
|
|
280
|
+
* loop's next boundary, this rides the same durable inbox but carries a
|
|
281
|
+
* control line the CLI handles immediately, mid-step included: the
|
|
282
|
+
* running tool call aborts, the run ends within ~100ms, and the guest
|
|
283
|
+
* session stays resumable with the whole turn context (verified live
|
|
284
|
+
* against claude 2.1.236). Verdicts mirror `injectUserMessage`; only
|
|
285
|
+
* "delivered" means the CLI got the control line — callers escalate
|
|
286
|
+
* anything else (and "unsupported": no stream-input turn, or a runtime
|
|
287
|
+
* with no in-band interrupt, e.g. codex) to kill semantics, which stay
|
|
288
|
+
* honest because thread stores are durable and the successor turn
|
|
289
|
+
* resumes them. Never throws.
|
|
290
|
+
*/
|
|
291
|
+
interruptTurn?(): Promise<"delivered" | "pending" | "closed" | "unsupported">;
|
|
292
|
+
/**
|
|
293
|
+
* Guest pid of the CURRENT turn's detached runner wrapper (the setsid
|
|
294
|
+
* process-group leader recorded at launch), or null when no detached
|
|
295
|
+
* durable-transport runner is live (boot phase, ACP path, single-exec
|
|
296
|
+
* transports). Advisory identity, NEVER a liveness verdict: the platform
|
|
297
|
+
* reads it to DECLARE harness-reported background work (an in-harness
|
|
298
|
+
* Workflow task) against the process tree that hosts it, so the park
|
|
299
|
+
* machinery can verify the tree from `/proc/<pid>` later. Runtimes
|
|
300
|
+
* without a detached guest simply omit the method.
|
|
301
|
+
*/
|
|
302
|
+
currentRunnerPid?(): number | null;
|
|
303
|
+
/**
|
|
304
|
+
* Durable byte offset of the CURRENT turn's `.out` file just past the
|
|
305
|
+
* last line whose messages have ALL been yielded to the consumer — the
|
|
306
|
+
* safe harvest watermark for boot-time turn adoption. Null when no
|
|
307
|
+
* durable-transport turn is live, or before the first line completes.
|
|
308
|
+
* The contract is deliberately one line BEHIND the parse cursor: a
|
|
309
|
+
* caller that persists parts after each yielded message may stamp this
|
|
310
|
+
* offset at any time and a successor re-parses AT MOST the line whose
|
|
311
|
+
* parts were mid-persist (the same crash window the workflow tailer's
|
|
312
|
+
* flush-before-advance ordering accepts). Advisory, never a verdict.
|
|
313
|
+
*/
|
|
314
|
+
currentTurnDurableOffset?(): number | null;
|
|
237
315
|
sendMessage(opts: {
|
|
238
316
|
prompt: string;
|
|
239
317
|
sessionId?: string;
|
|
@@ -257,7 +335,7 @@ export interface ModelExecutionContract {
|
|
|
257
335
|
* The sandbox provider (e.g. "vercel", "e2b") is an infrastructure concern
|
|
258
336
|
* configured via SANDBOX_PROVIDER — not part of the runtime definition.
|
|
259
337
|
* For non-sandbox agents (API calls, etc.) make the call directly in the workflow;
|
|
260
|
-
*
|
|
338
|
+
* `agent()` is a sandbox concept.
|
|
261
339
|
*/
|
|
262
340
|
export interface AgentRuntime<S extends SandboxProvider = SandboxProvider> {
|
|
263
341
|
create(sandbox: S, opts: RuntimeOptions): ModelExecutionContract;
|
|
@@ -224,10 +224,8 @@ export interface WorkflowMetadata {
|
|
|
224
224
|
* BUILD — a setup-only workflow whose job is to leave its VM configured and
|
|
225
225
|
* snapshot it (base-env / agent-env). Environment builds build a platform
|
|
226
226
|
* IMAGE and never use the shared factory drive, so the server SKIPS mounting
|
|
227
|
-
* /factory for them
|
|
228
|
-
*
|
|
229
|
-
* for this mountpoint"), degrading /factory for every workflow booting from
|
|
230
|
-
* that snapshot. Absent on ordinary workflows — which mount /factory exactly
|
|
227
|
+
* /factory for them and no live drive mount bakes into the captured
|
|
228
|
+
* snapshot. Absent on ordinary workflows — which mount /factory exactly
|
|
231
229
|
* as before. Optional + additive: an ABSENT flag contributes nothing to the
|
|
232
230
|
* canonical metadata hash (frozen-metadata rule), so existing workflows are
|
|
233
231
|
* not forced to re-register. */
|
|
@@ -5,9 +5,7 @@
|
|
|
5
5
|
* dispatch time. The CLI/bundler inspects the bundled module in the user's
|
|
6
6
|
* environment and sends this compact plan as metadata.
|
|
7
7
|
*
|
|
8
|
-
* Every workflow has a step plan.
|
|
9
|
-
* workflows are wrapped at the SDK boundary as a single-step compiled
|
|
10
|
-
* workflow (step name = "run"); the bundler sees the same shape regardless.
|
|
8
|
+
* Every workflow has a step plan.
|
|
11
9
|
*/
|
|
12
10
|
/** One artifact a step promises to produce — mirrors `StepDeliverable`,
|
|
13
11
|
* restated here so the plan stays a self-contained wire shape. */
|
package/dist/utils/bundler.d.ts
CHANGED
|
@@ -147,6 +147,29 @@ export declare const SDK_SPECIFIER_ALIASES: readonly string[];
|
|
|
147
147
|
* null. The single authority: the plugin's regex filter is only a fast
|
|
148
148
|
* pre-filter, and this decides. */
|
|
149
149
|
export declare function resolveSdkAlias(specifier: string): string | null;
|
|
150
|
+
/**
|
|
151
|
+
* The packages the PLATFORM provides to every workflow source — the SDK and
|
|
152
|
+
* its zod peer — which therefore resolve from the platform's own install
|
|
153
|
+
* roots when the source file's directory has no node_modules of its own.
|
|
154
|
+
* The live incident (2026-09-20, a claude-code session at the drive root):
|
|
155
|
+
* `agentc invoke --source /factory/files/wf.ts` died with `Could not
|
|
156
|
+
* resolve "@agent-compose/sdk", "zod"` because Bun walked up from
|
|
157
|
+
* /factory/files and found nothing, while the baked SDK sat in
|
|
158
|
+
* /workspace/node_modules the whole time (infra/e2b-template/template.ts —
|
|
159
|
+
* "SDK into /workspace/node_modules so any script written under /workspace
|
|
160
|
+
* can import it"). A workflow's OWN third-party deps still have to be
|
|
161
|
+
* installed beside the source; only these two are platform-resolved.
|
|
162
|
+
*/
|
|
163
|
+
export declare const PLATFORM_RESOLVED_PACKAGES: readonly string[];
|
|
164
|
+
/** Env override for the fallback roots (colon-separated, tried first) — the
|
|
165
|
+
* test seam, and an ops knob for a sandbox image that plants the SDK
|
|
166
|
+
* elsewhere. */
|
|
167
|
+
export declare const SDK_FALLBACK_ROOTS_ENV = "AGENT_COMPOSE_SDK_FALLBACK_ROOTS";
|
|
168
|
+
/** Where the platform SDK lives when the source's own walk-up finds nothing:
|
|
169
|
+
* the env override's roots, the sandbox's baked /workspace install, then
|
|
170
|
+
* the bundling process's cwd (the CLI's own resolution context). Exported
|
|
171
|
+
* for the unit test. */
|
|
172
|
+
export declare function sdkFallbackRoots(env?: Record<string, string | undefined>): string[];
|
|
150
173
|
/**
|
|
151
174
|
* The SELF-CORRECTING layer. Turns a Bun bundle failure into a message that
|
|
152
175
|
* names the real fix instead of leaking Bun's internals.
|
|
@@ -68,9 +68,8 @@ export interface StepObservability {
|
|
|
68
68
|
* this instance. After the step finishes, the engine calls `snapshot()`
|
|
69
69
|
* to extract the bundle for transport.
|
|
70
70
|
*
|
|
71
|
-
* Metadata writes merge (later keys win)
|
|
72
|
-
*
|
|
73
|
-
* across migrations.
|
|
71
|
+
* Metadata writes merge (later keys win), as the server's run-metadata
|
|
72
|
+
* merge does.
|
|
74
73
|
*/
|
|
75
74
|
export declare class StepObservabilityCollector {
|
|
76
75
|
private metadata;
|
|
@@ -6,7 +6,6 @@
|
|
|
6
6
|
* 2. For each step:
|
|
7
7
|
* a. Optionally check `getCachedOutput(stepIndex, step.name)` — if a
|
|
8
8
|
* previous run completed this step, skip and reuse its output.
|
|
9
|
-
* (Phase 1b uses this for crash recovery.)
|
|
10
9
|
* b. Validate current input against `step.input`.
|
|
11
10
|
* c. Call `step.run({ input, ... })`.
|
|
12
11
|
* d. Validate return value against `step.output`.
|
|
@@ -17,9 +16,10 @@
|
|
|
17
16
|
* Errors during any step bubble through `onStepFailed` and re-throw so the
|
|
18
17
|
* caller can decide whether to mark the run failed.
|
|
19
18
|
*
|
|
20
|
-
* The cache + completion hooks are injection points
|
|
21
|
-
*
|
|
22
|
-
*
|
|
19
|
+
* The cache + completion hooks are injection points; without a cache every
|
|
20
|
+
* step runs. The platform's sandbox path does not walk the chain here: the
|
|
21
|
+
* Temporal `executeStep` activity runs one step per runner subprocess via
|
|
22
|
+
* `runWorkflowSingleStep` below.
|
|
23
23
|
*/
|
|
24
24
|
import { z } from "zod";
|
|
25
25
|
import type { Workflow, StepRunResult } from "./types.js";
|
|
@@ -54,13 +54,10 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
|
|
|
54
54
|
/**
|
|
55
55
|
* Crash-recovery hook. Called before a step executes. Return the cached
|
|
56
56
|
* output to skip execution; return undefined to run the step.
|
|
57
|
-
*
|
|
58
|
-
* Phase 1b implementations will look up `workflow_step_runs` rows for
|
|
59
|
-
* (runId, stepIndex, stepName) and return completed step outputs here.
|
|
60
57
|
* Default: always undefined (no caching).
|
|
61
58
|
*/
|
|
62
59
|
getCachedOutput?(stepIndex: number, stepName: string): unknown | undefined | Promise<unknown | undefined>;
|
|
63
|
-
/** Fire after a step's `
|
|
60
|
+
/** Fire after a step's `run` and output validation succeed. */
|
|
64
61
|
onStepCompleted?(stepIndex: number, stepName: string, output: unknown, durationMs: number): void | Promise<void>;
|
|
65
62
|
/** Fire when a step throws or fails validation. The error is re-thrown after this returns. */
|
|
66
63
|
onStepFailed?(stepIndex: number, stepName: string, error: Error, durationMs: number): void | Promise<void>;
|
|
@@ -6,9 +6,8 @@
|
|
|
6
6
|
*
|
|
7
7
|
* Why: durable suspend/resume requires step boundaries to be addressable as
|
|
8
8
|
* data, not opaque async-function bodies. Each step's input + output is
|
|
9
|
-
* serialisable JSON so engine
|
|
10
|
-
*
|
|
11
|
-
* completed step.
|
|
9
|
+
* serialisable JSON so the durable engine (Temporal, one activity per step)
|
|
10
|
+
* can record completion and replay from the last completed step.
|
|
12
11
|
*/
|
|
13
12
|
import type { z } from "zod";
|
|
14
13
|
import type { BaseExecutionContext } from "../types/execution-context.js";
|
|
@@ -16,7 +15,7 @@ import type { AgentEventSink } from "../types/workflow.js";
|
|
|
16
15
|
import type { WorkflowMetadata } from "../types/workflow-metadata.js";
|
|
17
16
|
import type { StepObservability } from "./observability.js";
|
|
18
17
|
/**
|
|
19
|
-
* Per-step execution context. Threaded into every step's `
|
|
18
|
+
* Per-step execution context. Threaded into every step's `run(...)` so
|
|
20
19
|
* the step can read tenant identity and run identity, log progress, and
|
|
21
20
|
* invoke sandbox commands.
|
|
22
21
|
*
|
|
@@ -31,9 +30,9 @@ export interface StepContext<TInput = unknown> extends BaseExecutionContext {
|
|
|
31
30
|
abortSignal: AbortSignal;
|
|
32
31
|
/**
|
|
33
32
|
* Merge key-value metadata onto the run record. Buffered during the
|
|
34
|
-
* step and flushed when the step completes; the durable engine
|
|
35
|
-
* it
|
|
36
|
-
*
|
|
33
|
+
* step and flushed when the step completes; the durable engine merges
|
|
34
|
+
* it into the run's metadata (`persistStepObservability`). Later keys
|
|
35
|
+
* win.
|
|
37
36
|
*/
|
|
38
37
|
setMetadata(data: Record<string, unknown>): Promise<void>;
|
|
39
38
|
/**
|
|
@@ -93,9 +92,8 @@ export interface Step<TInput, TOutput> {
|
|
|
93
92
|
readonly deliverables?: readonly StepDeliverable[];
|
|
94
93
|
}
|
|
95
94
|
/**
|
|
96
|
-
* The result of running one step
|
|
97
|
-
* `
|
|
98
|
-
* completed steps. `observability` carries the snapshot of
|
|
95
|
+
* The result of running one step, as `runWorkflowSteps` reports it per
|
|
96
|
+
* step. `observability` carries the snapshot of
|
|
99
97
|
* `ctx.setMetadata` / `ctx.step` / `ctx.agentEvents` recorded during
|
|
100
98
|
* the step; undefined when no hooks were used.
|
|
101
99
|
*/
|
|
@@ -16,7 +16,8 @@
|
|
|
16
16
|
* matches the final step's `output` schema.
|
|
17
17
|
*
|
|
18
18
|
* Workflows-as-data — the result is consumable by any engine adapter
|
|
19
|
-
* (in-process
|
|
19
|
+
* (in-process `runWorkflow`; the sandbox runner one step at a time under
|
|
20
|
+
* Temporal).
|
|
20
21
|
*/
|
|
21
22
|
import type { z } from "zod";
|
|
22
23
|
import type { Step, Workflow } from "./types.js";
|
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
* runner.
|
|
7
7
|
*
|
|
8
8
|
* Errors classified into `WorkflowError` (user code threw) vs `EngineError`
|
|
9
|
-
* (platform problem)
|
|
9
|
+
* (platform problem).
|
|
10
10
|
*/
|
|
11
11
|
import type { WorkflowHooks } from "../types/workflow.js";
|
|
12
12
|
import type { InvokeChild } from "../types/execution-context.js";
|
|
@@ -35,7 +35,7 @@ export declare class EngineError extends Error {
|
|
|
35
35
|
readonly subsystem: EngineSubsystem;
|
|
36
36
|
constructor(message: string, subsystem?: EngineSubsystem, options?: ErrorOptions);
|
|
37
37
|
}
|
|
38
|
-
/** Classify any thrown value
|
|
38
|
+
/** Classify any thrown value as a `workflow` or an `engine` failure. */
|
|
39
39
|
export declare function classifyError(err: unknown): "workflow" | "engine";
|
|
40
40
|
export declare function parseNameVersion(ref: string): {
|
|
41
41
|
name: string;
|
|
@@ -52,9 +52,7 @@ export interface RunWorkflowOptions {
|
|
|
52
52
|
onStepStarted?: RunWorkflowStepsOpts<unknown, unknown>["onStepStarted"];
|
|
53
53
|
onStepCompleted?: RunWorkflowStepsOpts<unknown, unknown>["onStepCompleted"];
|
|
54
54
|
onStepFailed?: RunWorkflowStepsOpts<unknown, unknown>["onStepFailed"];
|
|
55
|
-
/**
|
|
56
|
-
* inject their native child-workflow primitive; the LocalProvider injects
|
|
57
|
-
* the public Agent Compose API client. */
|
|
55
|
+
/** Child workflow invocation. Unset, a step's `invokeChild` throws. */
|
|
58
56
|
invokeChild?: InvokeChild;
|
|
59
57
|
}
|
|
60
58
|
export declare function runWorkflow<TInput, TOutput>(wf: Workflow<TInput, TOutput>, ctx: {
|
|
@@ -18,8 +18,8 @@ import type { InvokeChild } from "../types/execution-context.js";
|
|
|
18
18
|
*/
|
|
19
19
|
export declare function deriveInvokeChildIdempotencyKey(parentRunId: string, childName: string): string | null;
|
|
20
20
|
/**
|
|
21
|
-
* Build the public-API child workflow invoker used by
|
|
22
|
-
*
|
|
21
|
+
* Build the public-API child workflow invoker used by sandboxed workflow
|
|
22
|
+
* execution (the step runner). Provider-backed engines may inject a different
|
|
23
23
|
* implementation (Temporal child workflow, Inngest invoke, etc.).
|
|
24
24
|
*/
|
|
25
25
|
export declare function buildInvokeChild(runId: string, opts?: {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@agent-compose/sdk",
|
|
3
|
-
"version": "0.8.
|
|
3
|
+
"version": "0.8.6",
|
|
4
4
|
"description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"repository": {
|
|
@@ -71,7 +71,7 @@
|
|
|
71
71
|
"ofetch": "^1.5.1",
|
|
72
72
|
"openai": "^6.33.0",
|
|
73
73
|
"p-retry": "^6.2.0",
|
|
74
|
-
"sharp": "
|
|
74
|
+
"sharp": "0.35.4"
|
|
75
75
|
},
|
|
76
76
|
"publishConfig": {
|
|
77
77
|
"access": "public"
|