@agent-compose/sdk 0.8.2 → 0.8.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/agent/__tests__/perf-sampler.test.d.ts +10 -0
  2. package/dist/agent/agent-context.d.ts +1 -1
  3. package/dist/agent/agent-loop.d.ts +5 -1
  4. package/dist/agent/desktop-open.d.ts +184 -0
  5. package/dist/agent/perf-sampler.d.ts +99 -0
  6. package/dist/agent/services-manifest.d.ts +88 -0
  7. package/dist/agent/services-restore.d.ts +58 -0
  8. package/dist/client.d.ts +164 -8
  9. package/dist/display.d.ts +17 -0
  10. package/dist/index.d.ts +13 -4
  11. package/dist/index.js +1374 -51
  12. package/dist/runtimes/_cli-agent.d.ts +347 -2
  13. package/dist/runtimes/claude-code.d.ts +12 -0
  14. package/dist/runtimes/codex.d.ts +8 -0
  15. package/dist/runtimes/openai-desktop.js +1312 -51
  16. package/dist/runtimes/session-env.test.d.ts +14 -0
  17. package/dist/types/api-conversations.d.ts +309 -1
  18. package/dist/types/api-factory.d.ts +115 -10
  19. package/dist/types/api-runs.d.ts +21 -0
  20. package/dist/types/protocol.d.ts +32 -1
  21. package/dist/types/runtime.d.ts +120 -0
  22. package/package.json +1 -1
  23. package/src/agent/agent-context.ts +100 -11
  24. package/src/agent/agent-loop.ts +10 -3
  25. package/src/agent/desktop-open.ts +418 -0
  26. package/src/agent/perf-sampler.ts +202 -0
  27. package/src/agent/services-manifest.ts +356 -0
  28. package/src/agent/services-restore.ts +195 -0
  29. package/src/client.ts +328 -12
  30. package/src/display.ts +44 -1
  31. package/src/index.ts +63 -1
  32. package/src/runtimes/_cli-agent.ts +891 -35
  33. package/src/runtimes/claude-code.ts +187 -12
  34. package/src/runtimes/codex.ts +58 -1
  35. package/src/sandbox/providers/local.ts +16 -4
  36. package/src/types/api-conversations.ts +307 -3
  37. package/src/types/api-factory.ts +118 -10
  38. package/src/types/api-runs.ts +23 -0
  39. package/src/types/protocol.ts +30 -1
  40. package/src/types/runtime.ts +122 -0
@@ -12,6 +12,33 @@ export interface McpServerConfig {
12
12
  args?: string[];
13
13
  env?: Record<string, string>;
14
14
  }
15
+ /**
16
+ * Exit-event push (the completion DOORBELL, v0.10.43). When set, the durable
17
+ * detached transport's in-guest wrapper fires ONE best-effort HTTP POST
18
+ * announcing `{ turnId, exitCode }` immediately AFTER the exit sentinel is
19
+ * durably written — so the server can verify-and-harvest at event latency
20
+ * instead of the watchdog's poll cadence.
21
+ *
22
+ * THE PUSH IS A DOORBELL, NEVER A VERDICT: the durable file stays the only
23
+ * truth; the push's arrival triggers verification against it, its absence
24
+ * means nothing (the poll ladder is the unchanged backstop), and a lost /
25
+ * duplicate / spoofed push must be harmless. Accordingly the wrapper never
26
+ * blocks sentinel-writing on the push (sentinel first, push after; failures
27
+ * are invisible to the runner lifecycle).
28
+ *
29
+ * Auth: `tokenEnv` NAMES a guest env var (e.g. the cloud session's
30
+ * `AGENT_COMPOSE_API_KEY`) — the wrapper reads it at push time, so the
31
+ * credential never appears in the generated script text or any log.
32
+ */
33
+ export interface TurnExitNotify {
34
+ /** Absolute URL of the server's turn-exit-event endpoint. */
35
+ url: string;
36
+ /** The bridge turn id this runner executes (rides the POST body). */
37
+ turnId: string;
38
+ /** Guest env var holding the bearer credential. A name that is not a
39
+ * plain env identifier disables the push (never risks shell injection). */
40
+ tokenEnv: string;
41
+ }
15
42
  /** Options passed to a runtime when creating a ModelExecutionContract. */
16
43
  export interface RuntimeOptions {
17
44
  allowedTools?: string[];
@@ -47,7 +74,33 @@ export interface RuntimeOptions {
47
74
  type: "json_schema";
48
75
  schema: Record<string, unknown>;
49
76
  };
77
+ /** Exit-event doorbell config (cloud sessions) — see `TurnExitNotify`.
78
+ * Absent ⇒ no push; the durable transport behaves exactly as before. */
79
+ turnExitNotify?: TurnExitNotify;
80
+ /** $HOME-relative path of a shell env file the CLI process sources at
81
+ * launch (cloud sessions: the platform-managed session-secrets file,
82
+ * `SESSION_ENV_FILE_RELPATH`). Sourced fresh at EVERY turn launch, so a
83
+ * rewrite between turns lands on the next turn without a VM recycle.
84
+ * Absent ⇒ nothing is sourced (local/BYOM runs never read a user's own
85
+ * dotfiles by surprise). Must be a plain relative path — no quotes, no
86
+ * `..`; the runtime validates and drops anything else. */
87
+ sessionEnvFile?: string;
88
+ /** ABSOLUTE guest path of a per-turn model-credential shell fragment
89
+ * (cloud subscription sessions: the server ships the TURN ACTOR's own
90
+ * subscription credential there before dispatching the turn). Sourced at
91
+ * launch AFTER `sessionEnvFile`, so the turn's credential always wins —
92
+ * each turn runs on its actor's plan, never a baked or session-wide one.
93
+ * Absent ⇒ nothing extra is sourced. Must be a plain absolute path — no
94
+ * quotes, no `..`; the runtime validates and drops anything else. */
95
+ credEnvFile?: string;
50
96
  }
97
+ /** Three-valued liveness verdict for a runtime's CURRENT turn, read from
98
+ * DURABLE guest state (heartbeat file, stdout file, exit sentinel, pid) over
99
+ * a fresh short exec — never from the health of any long-lived stream.
100
+ * `probe-failed` (exec timeout, transport fault, unparseable output) is
101
+ * NEVER evidence of death: the caller tracks it separately and only many
102
+ * consecutive failures escalate to a sandbox-unreachable verdict. */
103
+ export type RunnerLivenessVerdict = "alive" | "dead" | "probe-failed";
51
104
  /** Runtime-normalized result of running pre-tool processors. */
52
105
  export type ToolCallGateResult = {
53
106
  kind: "allow";
@@ -114,6 +167,73 @@ export interface ModelExecutionContract {
114
167
  * `captureCheckpoint` is omitted, this is never called.
115
168
  */
116
169
  restoreCheckpoint?(blob: unknown): void;
170
+ /**
171
+ * Durable liveness probe for the CURRENT turn (2026-08-15 incident, turn
172
+ * 18dc5261: a silently wedged tail stream blinded the executor's
173
+ * process-grep probe to a turn that had FINISHED into its durable file —
174
+ * ten minutes of "no evidence" over a completed answer). Runtimes that run
175
+ * the detached durable transport implement this by reading the guest's
176
+ * durable trio — heartbeat file, stdout size, exit sentinel, pid — over a
177
+ * fresh short exec. The executor's evidence ticker prefers this over its
178
+ * generic process-grep probe.
179
+ *
180
+ * Contract: resolves fast (the implementation carries its own explicit
181
+ * exec timeout) and never throws — faults map to "probe-failed". Null
182
+ * means NO durable probe exists right now (boot phase before the runner
183
+ * launched, a transport without durable files): the caller keeps whatever
184
+ * fallback probe it already had; null is never a verdict.
185
+ */
186
+ probeTurnLiveness?(): Promise<RunnerLivenessVerdict | null>;
187
+ /**
188
+ * DOORBELL, NEVER A VERDICT (exit-event push, v0.10.43): wake the current
189
+ * turn's durable watchdog NOW so it runs its normal verification pass —
190
+ * durable probe, then harvest-on-sentinel / honest no-sentinel death —
191
+ * immediately instead of at the next poll interval. Carries NO information
192
+ * of its own: a nudge for a live turn verifies alive and is a no-op; a
193
+ * spurious / duplicate / stale nudge is harmless; the poll ladder is the
194
+ * unchanged backstop when no nudge arrives. Never throws; a nudge while no
195
+ * watchdog is armed is remembered for the next arm (or dropped at turn
196
+ * start — a new turn owes nothing to the previous turn's doorbell).
197
+ */
198
+ nudgeTurnProbe?(): void;
199
+ /**
200
+ * RESUME HANDOFF, NEVER A KILL (redispatch-carries-session, 2026-08-15
201
+ * forensics): mark the CURRENT turn's in-guest runner as handed off — the
202
+ * caller intends a successor turn to RESUME the same guest session, so
203
+ * abort/early-exit must unwind the transport WITHOUT reaping the detached
204
+ * process tree or deleting its durable files (heartbeat included — the
205
+ * park path's busy signal keeps reading it). Without this, "stop
206
+ * consuming" and "kill the guest tree" are fused on the abort seam, and a
207
+ * supersede-then-resume would destroy the very session it resumes.
208
+ * One-way for the runner instance; a runtime without a detachable guest
209
+ * (single-exec transports, ACP) simply omits the method — its abort
210
+ * semantics are unchanged and the caller falls back to kill semantics.
211
+ */
212
+ detachGuest?(): void;
213
+ /**
214
+ * Deliver ONE user message INTO the currently running turn (the cloud
215
+ * analogue of `inboxStream` for runtimes whose agent loop lives in a
216
+ * detached in-guest CLI). The message rides a durable per-turn inbox
217
+ * file; the guest-side feeder forwards it to the CLI's stdin, where a
218
+ * steering-capable harness folds it into the live turn at the next safe
219
+ * boundary. Four-valued and honest:
220
+ * - "delivered" — the guest feeder forwarded the message to the CLI's
221
+ * stdin BEFORE the turn's terminal result: the running turn saw it.
222
+ * Only this verdict may advance any answered watermark.
223
+ * - "pending" — the append landed in the durable inbox but the ack
224
+ * window exhausted before the feeder forwarded it (guest exit 5: the
225
+ * CLI is not draining stdin mid-step — a long single tool call — or
226
+ * the guest is crawling). NOT seen yet; leave it owed. The line stays
227
+ * appended, so the CLI may still read it when the current step
228
+ * finishes — callers may narrate that bound but must never advance a
229
+ * watermark on it.
230
+ * - "closed" — the turn ended (guest exit 4 / exec fault) before
231
+ * the message was consumed: it was NOT seen; leave it owed.
232
+ * - "unsupported" — no live stream-input turn exists right now (boot
233
+ * phase, a spec/transport without stream input, ACP path).
234
+ * Calls are serialized per turn; never throws.
235
+ */
236
+ injectUserMessage?(text: string): Promise<"delivered" | "pending" | "closed" | "unsupported">;
117
237
  sendMessage(opts: {
118
238
  prompt: string;
119
239
  sessionId?: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.8.2",
3
+ "version": "0.8.3",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {
@@ -145,27 +145,81 @@ like?" is answered by opening it on this desktop and screenshotting it — not b
145
145
  reasoning about the code, and not by a headless render (which proves the process
146
146
  starts, not that the thing draws). Verify visually before you report visually.
147
147
 
148
+ **This is how you ACT on the web.** When the task is to DO something on a
149
+ website — book, order, reserve, sign up, fill a form, operate a dashboard —
150
+ and no connector or API covers it, the desktop browser IS the tool: \`ac-open\`
151
+ the site, do the errand there, and show the human the screen at decision
152
+ points (\`agentc display desktop\` in a cloud session). Research/search tools
153
+ answer QUESTIONS; an errand is an ACTION — "book me a table" means open the
154
+ booking site and book it, never a research report of options.
155
+
148
156
  - **Input** — \`xdotool\` against \`DISPLAY=:0\`: \`DISPLAY=:0 xdotool mousemove <x> <y>\`,
149
157
  \`DISPLAY=:0 xdotool click 1\` (1=left, 3=right), \`DISPLAY=:0 xdotool type 'text'\`,
150
158
  \`DISPLAY=:0 xdotool key Return\` (also \`ctrl+c\`, \`Tab\`, \`super\`, …).
151
159
  - **Screenshots** — \`scrot\` (or ImageMagick's \`import\`):
152
160
  \`DISPLAY=:0 scrot /tmp/screen.png\`, then READ the PNG to see the screen,
153
161
  before and after you act. A screenshot is your only eyes here.
154
- - **Apps + windows** — a plain X session. Launch in the background:
155
- \`DISPLAY=:0 <app> &\`. Two things that trip agents up, both normal:
162
+ - **The browser is chromium, preinstalled** — headful, on this display
163
+ (\`command -v chromium\` to confirm on an older machine). If an older machine
164
+ is missing it, the platform is already installing it in the background from
165
+ boot — \`ac-open <url>\` tells you when that is the case; retry it in ~30s.
166
+ Only if \`ac-open\` reports the background install FAILED do you relay that
167
+ one line to the human — never an apt-get expedition of your own.
168
+ - **Launching apps — use \`ac-open\`, never a plain \`&\`.** A GUI process
169
+ launched with \`<app> &\` DIES the moment your shell command returns — the
170
+ sandbox reaps each command's process group, so "the window vanished when
171
+ the shell finished" is that reaping, not a broken app. \`ac-open\` is the
172
+ platform launcher that survives it (\`command -v ac-open\` on older machines):
173
+
174
+ ac-open https://github.com # the browser — a running instance gets a tab
175
+ ac-open ./report.html # a local file, in the browser
176
+ ac-open . # a directory, in the file manager
177
+ ac-open gimp # any GUI app by command name
178
+
179
+ It detaches the app into its own session (setsid, stdio off your command's
180
+ pipes), records a pidfile + log under \`/tmp/.ac-desktop-open.<uid>/\`
181
+ (per-uid — yours is \`/tmp/.ac-desktop-open.$(id -u)\`), and
182
+ re-invoking it for a running app FOCUSES the existing window instead of
183
+ spawning a second copy. \`xdg-open\` and \`sensible-browser\` route through
184
+ it too. The whole recipe for looking at a page: \`ac-open <url>\`, then
185
+ \`sleep 5\`, then \`DISPLAY=:0 scrot /tmp/screen.png\` and read it. Without
186
+ \`ac-open\` (older machine), detach by hand:
187
+ \`setsid <app> </dev/null >/tmp/app.log 2>&1 &\` — and note **chromium as
188
+ root also needs \`--no-sandbox\`** (nested sandbox; \`ac-open\` and the baked
189
+ chromium defaults already handle it).
190
+ - **Two things that trip agents up, both normal:**
156
191
  - a GUI app needs a **beat to map its window** — screenshot, and if you see
157
192
  only wallpaper, wait a couple of seconds and screenshot again before
158
193
  concluding anything;
159
- - **Chromium needs \`--no-sandbox\`** in this environment (nested sandbox).
160
- The whole recipe for looking at a local page:
161
- \`DISPLAY=:0 chromium --no-sandbox --disable-gpu --start-maximized <url> &\`
162
- then \`sleep 5\`, then \`DISPLAY=:0 scrot /tmp/screen.png\` and read it.
163
- If a window still never appears, read the app's own log (\`/tmp/*.log\`) — the
164
- desktop is not the thing that failed. Do NOT abandon it for a headless
165
- screenshot: headless cannot tell you what the human will see.
194
+ - if a window still never appears, read the app's own log
195
+ (\`/tmp/.ac-desktop-open.$(id -u)/*.log\`, \`/tmp/*.log\`) — the desktop is not the
196
+ thing that failed. Do NOT abandon it for a headless
197
+ screenshot: headless cannot tell you what the human will see.
166
198
  - **A human can watch** — the session header carries a **Desktop** button in the
167
199
  dashboard, and what a teammate sees there is exactly this display. The desktop
168
200
  runs whether or not anyone is looking; never wait for a viewer.
201
+ - **Show the human the screen** — in a cloud session,
202
+ \`agentc display desktop --note "<caption>"\` captures this display and posts
203
+ it into the conversation as a snapshot card with an "Open desktop" door to
204
+ the live view. Use it to report visual results, and ALWAYS when you hit a
205
+ wall on the desktop that only a human can clear — a login form, a 2FA
206
+ prompt, a CAPTCHA, an unexpected dialog: snapshot it so they SEE the wall,
207
+ then ask (AskUserQuestion when you have it) and wait; never guess
208
+ credentials or click around a wall. The rule is SCREEN FOR ACTIONS,
209
+ VAULT FOR SECRETS. For non-sensitive interaction that needs the human's
210
+ own hands or judgment — pick an option, review a page, solve a CAPTCHA —
211
+ the display + ask pair is right: the platform merges them into ONE live
212
+ desktop card — the human clicks in, acts on the live screen, and answers
213
+ "I'm done" to hand it back; treat that answer as the wall being cleared,
214
+ re-check the screen, and continue. For SECRETS — a password, payment
215
+ details, any sensitive value —
216
+ \`agentc secrets session request <KEY...> --reason "<why>" --wait\` mints a
217
+ secure vault link (a one-tap approval when the user has these saved as a
218
+ personal set); the values land in the session env and YOU type them into
219
+ the site on the user's behalf. Never ask the human to type a password or
220
+ card number into this machine's browser, and never suggest they "log in
221
+ on the Desktop view" — the vault carries the secret, then you act with
222
+ it. A one-time 2FA code from their phone is the chat-OK exception.
169
223
 
170
224
  Nothing here changes the credentials rule above: tokens are injected at the
171
225
  network layer, never present on the desktop or in any file you can read — so
@@ -256,14 +310,49 @@ origin of its own — absolute asset paths and client-side routing work, the
256
310
  whole app is navigable — so serve normally and let the platform address it;
257
311
  never rewrite your app to a path prefix.
258
312
 
313
+ ## Durable services — the machine is cattle, the manifest is the pet (cloud sessions)
314
+
315
+ Parking preserves detached processes; a machine RECYCLE (resize, eviction,
316
+ failed reconnect) does not — every process and every byte off the drive is
317
+ discarded, and recycles are normal. When you start a long-running service the
318
+ human will rely on across turns (a dev server, a docker compose stack, a
319
+ database), record it in \`.ac/services.yml\` at the drive root so the platform
320
+ relaunches it automatically on the next fresh machine:
321
+
322
+ agentc services add <name> --command '<cmd>' # record a service
323
+ agentc services list # manifest + live status
324
+ agentc services restore # run the manifest now
325
+ agentc services remove <name>
326
+
327
+ Each entry can carry \`cwd\`, \`port\`, a bounded \`health\` probe (cmd or
328
+ http), one-time \`setup\` (e.g. \`docker compose pull\`), and \`data\` hooks.
329
+ After a recycle the platform posts "Machine restarted — restored N services"
330
+ into the conversation; on seeing it, VERIFY health rather than rebuilding —
331
+ logs live at \`/tmp/ac-services/<name>.log\`. Data honesty: sandbox-local
332
+ database state dies with the machine. Keep seeds/dumps ON THE DRIVE; declare
333
+ \`data.restore\` (reload on fresh boot) and \`data.dump\` (written before a
334
+ DELIBERATE recycle such as a resize — evictions give no warning, so treat the
335
+ drive copy as the truth).
336
+
259
337
  ## Tools in this environment
260
338
 
261
339
  - \`agentc\` — Agent Compose CLI (your primary interface; authed from env)
262
340
  - \`@agent-compose/sdk\` — installed in /workspace for writing workflows
263
341
  - \`/ac:*\` Claude Code skills — slash commands for the above
264
- - \`archil\` (factory drive), \`rtk\`, \`bun\`
342
+ - \`rtk\`, \`bun\`
265
343
  - \`xdotool\` / \`scrot\` — drive + screenshot the desktop (if this machine has one; see Computer Use)
266
- - A world-writable \`/workspace\` working directory`;
344
+ - \`chromium\` — the desktop browser; \`ac-open <url|file|app>\` — open it on the
345
+ desktop, detached (survives your command; see Computer Use)
346
+ - A world-writable \`/workspace\` working directory
347
+
348
+ If a system capability you need is genuinely missing — no browser, no display,
349
+ no \`ac-open\`, a daemon that isn't there — say so to the human in ONE honest
350
+ line (what is missing and what it blocks) instead of mounting a
351
+ package-manager expedition. An in-session \`apt-get install\` dies with the
352
+ sandbox, burns turns, and hides the real gap; missing platform capabilities
353
+ are the platform's to bake in, and \`agentc pause\` is the door to ask through.
354
+ (Your own project's dependencies are different — installing those is normal
355
+ work.)`;
267
356
 
268
357
  /** Parameters for the `agentc session add` education brief (ADR-0055 §8). */
269
358
  export interface AddedSessionBriefParams {
@@ -91,8 +91,13 @@ function preview(value: unknown): string {
91
91
 
92
92
  /** Everything but the live-only streaming chunk: `text_delta` never becomes
93
93
  * an agent.message event (the terminating `text` carries the whole block) —
94
- * the loop filters it before summarizing. */
95
- type DurableAgentMessage = Exclude<AgentMessage, { type: "text_delta" } | { type: "usage_delta" }>;
94
+ * the loop filters it before summarizing. `task_notification` is filtered
95
+ * too: it is session-transport metadata (a parent harness's background-task
96
+ * completion echo), not the agent's own output. */
97
+ type DurableAgentMessage = Exclude<
98
+ AgentMessage,
99
+ { type: "text_delta" } | { type: "usage_delta" } | { type: "task_notification" }
100
+ >;
96
101
 
97
102
  export function summarizeAgentMessage(msg: DurableAgentMessage): AgentMessageSummary {
98
103
  switch (msg.type) {
@@ -478,7 +483,9 @@ export async function agentLoop<TResponse = unknown>(opts: AgentLoopOpts<TRespon
478
483
  continue;
479
484
  }
480
485
  const msg = outputVerdict.value;
481
- if (msg.type === "text_delta" || msg.type === "usage_delta") continue; // a processor cannot re-introduce a live-only chunk
486
+ // A processor cannot re-introduce a live-only chunk; task
487
+ // notifications are transport metadata, never loop output.
488
+ if (msg.type === "text_delta" || msg.type === "usage_delta" || msg.type === "task_notification") continue;
482
489
  opts.onAgentEvent?.(iteration, msg);
483
490
  // Usage summaries carry the resolved model so the server can price
484
491
  // token rows per model without correlating back to agent.spawned.