@agent-compose/sdk 0.8.4 → 0.8.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/agent-context.d.ts +9 -1
- package/dist/agent/agent-loop.d.ts +10 -1
- package/dist/client.d.ts +171 -33
- package/dist/directives.d.ts +14 -0
- package/dist/generated/verb-synopsis.d.ts +34 -0
- package/dist/index.d.ts +6 -4
- package/dist/index.js +1024 -39
- package/dist/runtimes/_cli-agent.d.ts +106 -0
- package/dist/runtimes/claude-code.d.ts +31 -1
- package/dist/runtimes/openai-desktop.d.ts +50 -0
- package/dist/runtimes/openai-desktop.js +1048 -57
- package/dist/runtimes/openai-desktop.test.d.ts +20 -0
- package/dist/runtimes/tool-pulse.test.d.ts +17 -0
- package/dist/sandbox/devbox.d.ts +5 -5
- package/dist/sandbox/registry.d.ts +12 -0
- package/dist/sandbox/sizes.d.ts +11 -5
- package/dist/sandbox.d.ts +1 -1
- package/dist/step-invocation/types.d.ts +1 -1
- package/dist/types/api-conversations.d.ts +85 -12
- package/dist/types/api-factory.d.ts +111 -1
- package/dist/types/conversation-stream.d.ts +22 -1
- package/dist/types/protocol.d.ts +118 -1
- package/dist/types/runtime.d.ts +71 -0
- package/package.json +1 -1
- package/src/agent/agent-context.ts +43 -9
- package/src/agent/agent-loop.ts +11 -5
- package/src/agent/desktop-open.ts +13 -1
- package/src/client.ts +256 -38
- package/src/directives.ts +21 -1
- package/src/generated/verb-synopsis.ts +544 -0
- package/src/index.ts +17 -3
- package/src/runtimes/_cli-agent.ts +313 -22
- package/src/runtimes/claude-code.ts +249 -12
- package/src/runtimes/openai-desktop.ts +82 -19
- package/src/sandbox/devbox.ts +5 -5
- package/src/sandbox/providers/e2b.ts +60 -16
- package/src/sandbox/registry.ts +19 -1
- package/src/sandbox/sizes.ts +11 -5
- package/src/sandbox.ts +1 -0
- package/src/types/api-conversations.ts +65 -13
- package/src/types/api-factory.ts +121 -1
- package/src/types/conversation-stream.ts +24 -1
- package/src/types/protocol.ts +113 -1
- package/src/types/runtime.ts +63 -0
|
@@ -34,7 +34,10 @@
|
|
|
34
34
|
* token-metering gateway's Anthropic passthrough (ADR-0039).
|
|
35
35
|
*/
|
|
36
36
|
|
|
37
|
-
import type {
|
|
37
|
+
import type {
|
|
38
|
+
AgentMessage, AgentMessageCompaction, AgentMessageTaskNotification, AgentMessageTaskProgress,
|
|
39
|
+
WorkflowProgressEntry,
|
|
40
|
+
} from "../index.js";
|
|
38
41
|
import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
|
|
39
42
|
import { formatError } from "../utils/errors.js";
|
|
40
43
|
|
|
@@ -171,6 +174,37 @@ export function parseTaskNotifications(
|
|
|
171
174
|
return out;
|
|
172
175
|
}
|
|
173
176
|
|
|
177
|
+
/** Clamp for a steer delivered into a child's thread — same ceiling as a
|
|
178
|
+
* notification report: plenty for any real addendum, bounded against a
|
|
179
|
+
* runaway blob. */
|
|
180
|
+
const SUBAGENT_USER_MESSAGE_MAX = 20_000;
|
|
181
|
+
|
|
182
|
+
/** One user-role TEXT blob from the stream, mapped with sidechain awareness.
|
|
183
|
+
* Task-notification blocks parse into structure wherever they appear (their
|
|
184
|
+
* arrival side is resume-dependent — see the transport notes above). What
|
|
185
|
+
* remains is then split by attribution: TOP-LEVEL text (no parent id) stays
|
|
186
|
+
* unmapped exactly as before — prompt echoes and system reminders are not
|
|
187
|
+
* agent output. SIDECHAIN text (parent id present) is a message landing in
|
|
188
|
+
* a child subagent's thread — a delivered SendMessage steer — and forwards
|
|
189
|
+
* as `subagent_user_message`, except harness plumbing (`<system-reminder>`
|
|
190
|
+
* wrappers the CLI injects into child threads), which no renderer should
|
|
191
|
+
* see. Exported for tests. */
|
|
192
|
+
export function sidechainAwareUserText(
|
|
193
|
+
text: string, parentToolUseId: string | undefined, timestamp: string,
|
|
194
|
+
): AgentMessage[] {
|
|
195
|
+
const notifications = parseTaskNotifications(text, timestamp);
|
|
196
|
+
if (notifications.length > 0) return notifications;
|
|
197
|
+
if (!parentToolUseId) return [];
|
|
198
|
+
const trimmed = text.trim();
|
|
199
|
+
if (trimmed.length === 0 || trimmed.startsWith("<system-reminder>")) return [];
|
|
200
|
+
return [{
|
|
201
|
+
type: "subagent_user_message",
|
|
202
|
+
text: clip(trimmed, SUBAGENT_USER_MESSAGE_MAX),
|
|
203
|
+
parentToolUseId,
|
|
204
|
+
timestamp,
|
|
205
|
+
}];
|
|
206
|
+
}
|
|
207
|
+
|
|
174
208
|
/** One `system`/`task_notification` stream-json event mapped onto the same
|
|
175
209
|
* structured message the XML parse produces, or null when the event names
|
|
176
210
|
* no task id. Pure and tolerant over untrusted harness JSON: absent fields
|
|
@@ -206,6 +240,175 @@ export function parseSystemTaskNotification(
|
|
|
206
240
|
};
|
|
207
241
|
}
|
|
208
242
|
|
|
243
|
+
// ── Task progress (live background-workflow evidence) ───────────────────────
|
|
244
|
+
//
|
|
245
|
+
// While a harness Workflow (the CLI's in-harness dynamic-workflow tool)
|
|
246
|
+
// runs in the background, stream-json emits `system`/`task_progress`
|
|
247
|
+
// events carrying the workflow's CUMULATIVE `workflow_progress` array —
|
|
248
|
+
// verified live on 2.1.241 (`-p --output-format stream-json`): the array
|
|
249
|
+
// re-sends on every agent state change immediately and on a ~10s heartbeat
|
|
250
|
+
// otherwise, each `workflow_agent` entry updated in place (label,
|
|
251
|
+
// phase, model, state, tokens, toolCalls, durationMs). This is the ONLY
|
|
252
|
+
// per-agent evidence the wire ever carries — the completion notification
|
|
253
|
+
// reports aggregates only — so dropping it leaves workflow cards blind.
|
|
254
|
+
// Preview text (promptPreview / resultPreview) is internal plumbing and is
|
|
255
|
+
// not forwarded, same rule as task notifications.
|
|
256
|
+
|
|
257
|
+
const PROGRESS_PHASES_MAX = 32;
|
|
258
|
+
const PROGRESS_AGENTS_MAX = 128;
|
|
259
|
+
const PROGRESS_LABEL_MAX = 200;
|
|
260
|
+
const PROGRESS_ERROR_MAX = 500;
|
|
261
|
+
|
|
262
|
+
const AGENT_STATES: ReadonlySet<string> = new Set(["start", "progress", "done", "error"]);
|
|
263
|
+
|
|
264
|
+
/** One `system`/`task_progress` event mapped onto the structured progress
|
|
265
|
+
* message, or null only when it names no task. Workflow tasks carry the
|
|
266
|
+
* cumulative `workflow_progress` entries; a plain background Agent-task
|
|
267
|
+
* heartbeat (no workflow entries — verified live on 2.1.241, ~10s cadence
|
|
268
|
+
* while the task runs) forwards with `workflow: []` so the platform can
|
|
269
|
+
* DECLARE the running task as session background work (the delegation-
|
|
270
|
+
* survives-the-human contract: parks, user interrupts, and supersedes all
|
|
271
|
+
* read the declared-children registry). Completion evidence still rides
|
|
272
|
+
* `task_notification`. Pure and tolerant over untrusted harness JSON:
|
|
273
|
+
* malformed entries are skipped, everything is clamped (phases keep the
|
|
274
|
+
* FIRST 32 — they are seeded up front; agents keep the LAST 128 — late
|
|
275
|
+
* spawns matter more than a long-settled prefix). Exported for tests. */
|
|
276
|
+
export function parseSystemTaskProgress(
|
|
277
|
+
p: Record<string, unknown>, timestamp: string,
|
|
278
|
+
): AgentMessageTaskProgress | null {
|
|
279
|
+
const taskId = typeof p.task_id === "string" && p.task_id.trim().length > 0 ? p.task_id.trim() : null;
|
|
280
|
+
if (!taskId) return null;
|
|
281
|
+
const raw = Array.isArray(p.workflow_progress) ? p.workflow_progress : [];
|
|
282
|
+
const nonneg = (v: unknown): number | undefined =>
|
|
283
|
+
typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
|
|
284
|
+
const str = (v: unknown, max: number): string | undefined =>
|
|
285
|
+
typeof v === "string" && v.length > 0 ? clip(v, max) : undefined;
|
|
286
|
+
const phases: WorkflowProgressEntry[] = [];
|
|
287
|
+
const agents: WorkflowProgressEntry[] = [];
|
|
288
|
+
for (const e of raw) {
|
|
289
|
+
if (typeof e !== "object" || e === null) continue;
|
|
290
|
+
const o = e as Record<string, unknown>;
|
|
291
|
+
const index = nonneg(o.index);
|
|
292
|
+
if (index === undefined) continue;
|
|
293
|
+
if (o.type === "workflow_phase") {
|
|
294
|
+
const title = str(o.title, PROGRESS_LABEL_MAX);
|
|
295
|
+
if (title && phases.length < PROGRESS_PHASES_MAX) phases.push({ kind: "phase", index, title });
|
|
296
|
+
continue;
|
|
297
|
+
}
|
|
298
|
+
if (o.type !== "workflow_agent") continue;
|
|
299
|
+
const label = str(o.label, PROGRESS_LABEL_MAX);
|
|
300
|
+
const state = typeof o.state === "string" && AGENT_STATES.has(o.state) ? o.state as "start" | "progress" | "done" | "error" : null;
|
|
301
|
+
if (!label || !state) continue;
|
|
302
|
+
agents.push({
|
|
303
|
+
kind: "agent", index, label, state,
|
|
304
|
+
...(nonneg(o.phaseIndex) !== undefined ? { phaseIndex: nonneg(o.phaseIndex) } : {}),
|
|
305
|
+
...(str(o.phaseTitle, PROGRESS_LABEL_MAX) ? { phaseTitle: str(o.phaseTitle, PROGRESS_LABEL_MAX) } : {}),
|
|
306
|
+
...(str(o.model, 100) ? { model: str(o.model, 100) } : {}),
|
|
307
|
+
...(str(o.agentId, 128) ? { agentId: str(o.agentId, 128) } : {}),
|
|
308
|
+
...(nonneg(o.tokens) !== undefined ? { tokens: nonneg(o.tokens) } : {}),
|
|
309
|
+
...(nonneg(o.toolCalls) !== undefined ? { toolCalls: nonneg(o.toolCalls) } : {}),
|
|
310
|
+
...(nonneg(o.durationMs) !== undefined ? { durationMs: nonneg(o.durationMs) } : {}),
|
|
311
|
+
...(nonneg(o.startedAt) !== undefined ? { startedAt: nonneg(o.startedAt) } : {}),
|
|
312
|
+
...(state === "error" && str(o.error, PROGRESS_ERROR_MAX) ? { error: str(o.error, PROGRESS_ERROR_MAX) } : {}),
|
|
313
|
+
...(o.cached === true ? { cached: true as const } : {}),
|
|
314
|
+
...(o.skipped === true ? { skipped: true as const } : {}),
|
|
315
|
+
});
|
|
316
|
+
}
|
|
317
|
+
// Empty = a plain Agent-task heartbeat, forwarded as liveness evidence
|
|
318
|
+
// (see the header) — the workflow delta folder ignores it (no entries),
|
|
319
|
+
// and only the background-work declare consumes it.
|
|
320
|
+
const workflow = [...phases, ...agents.slice(-PROGRESS_AGENTS_MAX)];
|
|
321
|
+
const toolUseId = typeof p.tool_use_id === "string" && p.tool_use_id.length > 0 ? p.tool_use_id : null;
|
|
322
|
+
const u = (typeof p.usage === "object" && p.usage !== null ? p.usage : {}) as Record<string, unknown>;
|
|
323
|
+
const tokens = nonneg(u.total_tokens);
|
|
324
|
+
const toolUses = nonneg(u.tool_uses);
|
|
325
|
+
const durationMs = nonneg(u.duration_ms);
|
|
326
|
+
return {
|
|
327
|
+
type: "task_progress",
|
|
328
|
+
taskId: clip(taskId, 128),
|
|
329
|
+
...(toolUseId ? { toolUseId: clip(toolUseId, 128) } : {}),
|
|
330
|
+
...(tokens !== undefined || toolUses !== undefined || durationMs !== undefined
|
|
331
|
+
? { usage: {
|
|
332
|
+
...(tokens !== undefined ? { tokens } : {}),
|
|
333
|
+
...(toolUses !== undefined ? { toolUses } : {}),
|
|
334
|
+
...(durationMs !== undefined ? { durationMs } : {}),
|
|
335
|
+
} }
|
|
336
|
+
: {}),
|
|
337
|
+
workflow,
|
|
338
|
+
timestamp,
|
|
339
|
+
};
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
// ── Context compaction (the harness summarizing its own conversation) ───────
|
|
343
|
+
//
|
|
344
|
+
// Captured live against 2.1.212 (the baked E2B pin) and 2.1.241 — identical
|
|
345
|
+
// wire shapes on both, `/compact` and forced-auto alike:
|
|
346
|
+
//
|
|
347
|
+
// {"type":"system","subtype":"status","status":"compacting"}
|
|
348
|
+
// {"type":"system","subtype":"status","status":null,
|
|
349
|
+
// "compact_result":"success"|"failed"[,"compact_error":"…"]}
|
|
350
|
+
// {"type":"system","subtype":"init",…} (fresh init, success only)
|
|
351
|
+
// {"type":"system","subtype":"compact_boundary","compact_metadata":
|
|
352
|
+
// {"trigger":"auto"|"manual","pre_tokens":N,"post_tokens":N,
|
|
353
|
+
// "cumulative_dropped_tokens":N,"duration_ms":N,"preserved_segment":…}}
|
|
354
|
+
//
|
|
355
|
+
// Followed by the continuation summary as a SYNTHETIC user message (string
|
|
356
|
+
// content, `isSynthetic: true`) — deliberately NOT forwarded: it quotes
|
|
357
|
+
// conversation content verbatim (its "All user messages" section replays
|
|
358
|
+
// platform notices word for word), which is exactly how the 2026-08-23
|
|
359
|
+
// incident re-labeled a platform system notice as a "recalled memory".
|
|
360
|
+
// The boundary does not replay on later resumes (verified live).
|
|
361
|
+
//
|
|
362
|
+
// A compaction spans 13-21s in the small captures and MINUTES at real
|
|
363
|
+
// context sizes — forwarding the lifecycle is what lets downstream render
|
|
364
|
+
// the silence as work instead of death (the 94%-auto-compact dead-air
|
|
365
|
+
// incident).
|
|
366
|
+
|
|
367
|
+
const COMPACT_ERROR_MAX = 500;
|
|
368
|
+
|
|
369
|
+
/** One `system` event of the compaction family mapped onto the structured
|
|
370
|
+
* lifecycle message, or null when the event is not compaction-shaped (a
|
|
371
|
+
* plain `status` event with neither the compacting status nor a
|
|
372
|
+
* compact_result stays unmapped). Pure and tolerant over untrusted harness
|
|
373
|
+
* JSON. Exported for tests. */
|
|
374
|
+
export function parseSystemCompaction(
|
|
375
|
+
p: Record<string, unknown>, timestamp: string,
|
|
376
|
+
): AgentMessageCompaction | null {
|
|
377
|
+
const nonneg = (v: unknown): number | undefined =>
|
|
378
|
+
typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
|
|
379
|
+
if (p.subtype === "status") {
|
|
380
|
+
if (p.status === "compacting") return { type: "compaction", phase: "start", timestamp };
|
|
381
|
+
if (typeof p.compact_result === "string") {
|
|
382
|
+
const failed = p.compact_result !== "success";
|
|
383
|
+
const error = typeof p.compact_error === "string" && p.compact_error.length > 0
|
|
384
|
+
? clip(p.compact_error, COMPACT_ERROR_MAX) : undefined;
|
|
385
|
+
return {
|
|
386
|
+
type: "compaction", phase: "settled",
|
|
387
|
+
result: failed ? "failed" : "success",
|
|
388
|
+
...(failed && error ? { error } : {}),
|
|
389
|
+
timestamp,
|
|
390
|
+
};
|
|
391
|
+
}
|
|
392
|
+
return null;
|
|
393
|
+
}
|
|
394
|
+
if (p.subtype !== "compact_boundary") return null;
|
|
395
|
+
const meta = (typeof p.compact_metadata === "object" && p.compact_metadata !== null
|
|
396
|
+
? p.compact_metadata : {}) as Record<string, unknown>;
|
|
397
|
+
const preTokens = nonneg(meta.pre_tokens);
|
|
398
|
+
const postTokens = nonneg(meta.post_tokens);
|
|
399
|
+
const droppedTokens = nonneg(meta.cumulative_dropped_tokens);
|
|
400
|
+
const durationMs = nonneg(meta.duration_ms);
|
|
401
|
+
return {
|
|
402
|
+
type: "compaction", phase: "boundary",
|
|
403
|
+
trigger: meta.trigger === "manual" ? "manual" : "auto",
|
|
404
|
+
...(preTokens !== undefined ? { preTokens } : {}),
|
|
405
|
+
...(postTokens !== undefined ? { postTokens } : {}),
|
|
406
|
+
...(droppedTokens !== undefined ? { droppedTokens } : {}),
|
|
407
|
+
...(durationMs !== undefined ? { durationMs } : {}),
|
|
408
|
+
timestamp,
|
|
409
|
+
};
|
|
410
|
+
}
|
|
411
|
+
|
|
209
412
|
/** Claude Code's real reasoning knob is its own `--effort <level>` flag
|
|
210
413
|
* (low|medium|high|xhigh|max — verified against `claude -p --help`). The
|
|
211
414
|
* CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
|
|
@@ -247,6 +450,14 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
247
450
|
streamInput: {
|
|
248
451
|
promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
|
|
249
452
|
messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
|
|
453
|
+
// The ESC equivalent (verified live against claude 2.1.236): the CLI's
|
|
454
|
+
// control layer processes this OUT OF BAND — a 120s foreground Bash call
|
|
455
|
+
// aborted 6s in, the control_response answered instantly, the run ended
|
|
456
|
+
// ~100ms later with `result: error_during_execution`, and `--resume` on
|
|
457
|
+
// the same session id kept the full turn context.
|
|
458
|
+
interruptLine: (requestId) => JSON.stringify({
|
|
459
|
+
type: "control_request", request_id: requestId, request: { subtype: "interrupt" },
|
|
460
|
+
}),
|
|
250
461
|
},
|
|
251
462
|
buildCommand: ({ promptPath, sessionId, model, cwd, effort, streamInput }) => {
|
|
252
463
|
const flags = [
|
|
@@ -285,9 +496,10 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
285
496
|
// tool_use id; null at top level). Forwarded on tool_use/tool_result so
|
|
286
497
|
// renderers can nest child activity under the spawning call instead of
|
|
287
498
|
// flattening it into the parent transcript unattributed.
|
|
288
|
-
const
|
|
289
|
-
?
|
|
290
|
-
:
|
|
499
|
+
const parentId = typeof p.parent_tool_use_id === "string" && p.parent_tool_use_id.length > 0
|
|
500
|
+
? p.parent_tool_use_id
|
|
501
|
+
: undefined;
|
|
502
|
+
const parent = parentId ? { parentToolUseId: parentId } : {};
|
|
291
503
|
switch (p.type) {
|
|
292
504
|
// Assistant API message: content blocks → text / thinking / tool_use.
|
|
293
505
|
case "assistant": {
|
|
@@ -321,12 +533,25 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
321
533
|
// User API message: the CLI echoes tool results back as user content,
|
|
322
534
|
// and injects `<task-notification>` blocks (background-task completion
|
|
323
535
|
// evidence) as user TEXT — parsed into structure, never dropped and
|
|
324
|
-
// never forwarded raw.
|
|
325
|
-
// reminders) stays unmapped: it is not agent output.
|
|
536
|
+
// never forwarded raw. TOP-LEVEL user text (the echo of the prompt,
|
|
537
|
+
// system reminders) stays unmapped: it is not agent output. SIDECHAIN
|
|
538
|
+
// user text (parent_tool_use_id present) is a message landing in a
|
|
539
|
+
// CHILD's thread — the delivered form of a SendMessage steer to a
|
|
540
|
+
// running subagent ("queued for delivery at its next tool round") —
|
|
541
|
+
// and forwards as `subagent_user_message` so the steer renders inside
|
|
542
|
+
// the child's mini-session instead of vanishing (task #97; before
|
|
543
|
+
// this, a queued steer was visible only as the parent's opaque
|
|
544
|
+
// SendMessage tool call).
|
|
326
545
|
case "user": {
|
|
327
546
|
const message = p.message as { content?: ClaudeContentBlock[] | string } | undefined;
|
|
547
|
+
// HARNESS-synthesized user messages (`isSynthetic` — e.g. a
|
|
548
|
+
// post-compaction continuation summary, including a CHILD's own)
|
|
549
|
+
// are never a delivered steer: suppress the steer arm, keep the
|
|
550
|
+
// notification parse. Structural marker only, same doctrine as the
|
|
551
|
+
// `<synthetic>` model stamp above.
|
|
552
|
+
const steerParent = p.isSynthetic === true ? undefined : parentId;
|
|
328
553
|
if (typeof message?.content === "string") {
|
|
329
|
-
return
|
|
554
|
+
return sidechainAwareUserText(message.content, steerParent, ts);
|
|
330
555
|
}
|
|
331
556
|
const blocks = Array.isArray(message?.content) ? message.content : [];
|
|
332
557
|
return blocks.flatMap((b): AgentMessage[] => {
|
|
@@ -337,7 +562,7 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
337
562
|
}];
|
|
338
563
|
}
|
|
339
564
|
if (b.type === "text" && typeof b.text === "string") {
|
|
340
|
-
return
|
|
565
|
+
return sidechainAwareUserText(b.text, steerParent, ts);
|
|
341
566
|
}
|
|
342
567
|
return [];
|
|
343
568
|
});
|
|
@@ -410,11 +635,23 @@ export const claudeCodeSpec: CliAgentSpec = {
|
|
|
410
635
|
// System events: init carries the session id (extractSessionId), and
|
|
411
636
|
// the background-task lane rides here too — `task_notification` is
|
|
412
637
|
// the completion evidence stream-json actually emits (see the section
|
|
413
|
-
// header above),
|
|
414
|
-
//
|
|
415
|
-
//
|
|
416
|
-
//
|
|
638
|
+
// header above), and `task_progress` the LIVE background-task feed
|
|
639
|
+
// (parseSystemTaskProgress): per-agent workflow entries for Workflow
|
|
640
|
+
// tasks, `workflow: []` heartbeats for plain background Agent tasks —
|
|
641
|
+
// both forward so the platform can declare the running task as
|
|
642
|
+
// session background work. Compaction rides here as well (the status
|
|
643
|
+
// pair + compact_boundary — parseSystemCompaction). Everything else
|
|
644
|
+
// under system (task_started, task_updated, background_tasks_changed,
|
|
645
|
+
// thinking_tokens) is lifecycle noise here.
|
|
417
646
|
case "system": {
|
|
647
|
+
if (p.subtype === "task_progress") {
|
|
648
|
+
const progress = parseSystemTaskProgress(p, ts);
|
|
649
|
+
return progress ? [progress] : [];
|
|
650
|
+
}
|
|
651
|
+
if (p.subtype === "status" || p.subtype === "compact_boundary") {
|
|
652
|
+
const compaction = parseSystemCompaction(p, ts);
|
|
653
|
+
return compaction ? [compaction] : [];
|
|
654
|
+
}
|
|
418
655
|
if (p.subtype !== "task_notification") return [];
|
|
419
656
|
const notification = parseSystemTaskNotification(p, ts);
|
|
420
657
|
return notification ? [notification] : [];
|
|
@@ -3,6 +3,17 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Works against DesktopSandboxProvider — no E2B SDK dependency here.
|
|
5
5
|
* The action→screenshot loop runs until the model emits text, then yields it.
|
|
6
|
+
*
|
|
7
|
+
* COORDINATE SPACES (incident note). Every screenshot is downscaled to the
|
|
8
|
+
* model's DISPLAY geometry (1024×720) before it is sent, so the model emits
|
|
9
|
+
* coordinates in the SCALED screenshot's space — never the real desktop's.
|
|
10
|
+
* An earlier revision executed those coordinates against the real desktop
|
|
11
|
+
* using hardcoded real-geometry constants, so on any sandbox whose actual
|
|
12
|
+
* geometry differed, every click landed in the wrong place. The provider has
|
|
13
|
+
* no geometry call, so the real geometry is read from each screenshot buffer
|
|
14
|
+
* itself (sharp metadata), and the resulting per-capture `CaptureScale` maps
|
|
15
|
+
* the model's coordinates back onto the exact frame it saw. The old constants
|
|
16
|
+
* survive ONLY as a fallback for the metadata-missing case.
|
|
6
17
|
*/
|
|
7
18
|
|
|
8
19
|
import OpenAI from "openai";
|
|
@@ -12,8 +23,11 @@ import { defineRuntime, formatError } from "../index.js";
|
|
|
12
23
|
|
|
13
24
|
const DISPLAY_WIDTH = 1024;
|
|
14
25
|
const DISPLAY_HEIGHT = 720;
|
|
15
|
-
|
|
16
|
-
|
|
26
|
+
/** FALLBACK real-desktop geometry — consulted ONLY when sharp cannot read a
|
|
27
|
+
* screenshot's dimensions. The truth is derived per capture from the raw
|
|
28
|
+
* screenshot buffer itself (see `captureScaledScreenshot`). */
|
|
29
|
+
const FALLBACK_DESKTOP_WIDTH = 1280;
|
|
30
|
+
const FALLBACK_DESKTOP_HEIGHT = 800;
|
|
17
31
|
const MAX_TURNS = 40;
|
|
18
32
|
|
|
19
33
|
export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
|
|
@@ -27,8 +41,9 @@ export interface OpenAIDesktopRunnerOptions extends RuntimeOptions {
|
|
|
27
41
|
// so the exchange is described here instead of with SDK types.
|
|
28
42
|
|
|
29
43
|
/** One model-emitted desktop action. Coordinates arrive in DISPLAY (model)
|
|
30
|
-
* space and are rescaled
|
|
31
|
-
|
|
44
|
+
* space and are rescaled through the CURRENT capture's `CaptureScale`
|
|
45
|
+
* before dispatch. Exported so tests can type their fixtures. */
|
|
46
|
+
export interface ComputerAction {
|
|
32
47
|
type: string;
|
|
33
48
|
coordinate?: [number, number];
|
|
34
49
|
startCoordinate?: [number, number];
|
|
@@ -99,12 +114,16 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
|
99
114
|
|
|
100
115
|
yield { type: "init", sessionId: "e2b-desktop", timestamp: ts() };
|
|
101
116
|
|
|
102
|
-
const
|
|
117
|
+
const first = await captureScaledScreenshot(sandbox);
|
|
118
|
+
// The scale of the LATEST frame the model has seen — its next batch of
|
|
119
|
+
// coordinates is in that frame's space, so every fresh capture below
|
|
120
|
+
// replaces this before the frame is sent back.
|
|
121
|
+
let scale = first.scale;
|
|
103
122
|
|
|
104
123
|
const messages: InputMessage[] = [{
|
|
105
124
|
role: "user",
|
|
106
125
|
content: [
|
|
107
|
-
{ type: "input_image", image_url: `data:image/png;base64,${
|
|
126
|
+
{ type: "input_image", image_url: `data:image/png;base64,${first.b64}` },
|
|
108
127
|
{ type: "input_text", text: opts.prompt },
|
|
109
128
|
],
|
|
110
129
|
}];
|
|
@@ -137,7 +156,7 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
|
137
156
|
|
|
138
157
|
for (const call of computerCalls) {
|
|
139
158
|
console.log(`${label} → ${call.action.type}`);
|
|
140
|
-
await executeAction(sandbox, call.action).catch(
|
|
159
|
+
await executeAction(sandbox, call.action, scale).catch(
|
|
141
160
|
(err: unknown) => console.warn(`${label} Action failed (non-fatal): ${formatError(err)}`),
|
|
142
161
|
);
|
|
143
162
|
}
|
|
@@ -145,10 +164,11 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
|
145
164
|
if (computerCalls.length === 0) break;
|
|
146
165
|
|
|
147
166
|
const next = await captureScaledScreenshot(sandbox);
|
|
167
|
+
scale = next.scale;
|
|
148
168
|
messages.push({ role: "user", content: computerCalls.map((call): InputPart => ({
|
|
149
169
|
type: "computer_call_output",
|
|
150
170
|
call_id: call.call_id,
|
|
151
|
-
output: { type: "input_image", image_url: `data:image/png;base64,${next}` },
|
|
171
|
+
output: { type: "input_image", image_url: `data:image/png;base64,${next.b64}` },
|
|
152
172
|
})) });
|
|
153
173
|
}
|
|
154
174
|
|
|
@@ -160,21 +180,58 @@ export class OpenAIDesktopRunner implements ModelExecutionContract {
|
|
|
160
180
|
// Helpers — work against DesktopSandboxProvider interface
|
|
161
181
|
// ---------------------------------------------------------------------------
|
|
162
182
|
|
|
163
|
-
|
|
183
|
+
/** Real-desktop pixels per model-DISPLAY pixel, derived from ONE capture.
|
|
184
|
+
* Coordinates the model emits against that capture are multiplied by these
|
|
185
|
+
* factors (and rounded) before touching the provider. */
|
|
186
|
+
export interface CaptureScale {
|
|
187
|
+
scaleX: number;
|
|
188
|
+
scaleY: number;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** One captured frame: the model-facing image plus the scale that maps the
|
|
192
|
+
* model's coordinates back onto this frame's real desktop. */
|
|
193
|
+
export interface ScaledCapture {
|
|
194
|
+
/** Base64 PNG resized to DISPLAY_WIDTH×DISPLAY_HEIGHT (fit "fill"). */
|
|
195
|
+
b64: string;
|
|
196
|
+
scale: CaptureScale;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/** Derive a capture's scale from its real pixel dimensions. The constants
|
|
200
|
+
* are a FALLBACK only — used when sharp cannot read the frame's metadata;
|
|
201
|
+
* a real dimension always wins. */
|
|
202
|
+
export function scaleFromDimensions(width: number | undefined, height: number | undefined): CaptureScale {
|
|
203
|
+
return {
|
|
204
|
+
scaleX: (width ?? FALLBACK_DESKTOP_WIDTH) / DISPLAY_WIDTH,
|
|
205
|
+
scaleY: (height ?? FALLBACK_DESKTOP_HEIGHT) / DISPLAY_HEIGHT,
|
|
206
|
+
};
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
/** Capture one frame: read the REAL geometry from the raw screenshot's own
|
|
210
|
+
* metadata, downscale to the model's DISPLAY geometry, and return both the
|
|
211
|
+
* image and the scale that maps model coordinates back onto this frame. */
|
|
212
|
+
export async function captureScaledScreenshot(sandbox: DesktopSandboxProvider): Promise<ScaledCapture> {
|
|
164
213
|
const raw = await sandbox.screenshot();
|
|
165
|
-
const
|
|
214
|
+
const image = sharp(raw);
|
|
215
|
+
const { width, height } = await image.metadata();
|
|
216
|
+
const scaled = await image
|
|
166
217
|
.resize(DISPLAY_WIDTH, DISPLAY_HEIGHT, { kernel: "lanczos3", fit: "fill" })
|
|
167
218
|
.png()
|
|
168
219
|
.toBuffer();
|
|
169
|
-
return scaled.toString("base64");
|
|
220
|
+
return { b64: scaled.toString("base64"), scale: scaleFromDimensions(width, height) };
|
|
170
221
|
}
|
|
171
222
|
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
async function executeAction(
|
|
223
|
+
/** Dispatch one model action against the provider, mapping every coordinate
|
|
224
|
+
* through the CURRENT capture's scale. Non-coordinate payloads — scroll
|
|
225
|
+
* deltas (ticks), typed text, key chords — pass through untouched. */
|
|
226
|
+
export async function executeAction(
|
|
227
|
+
sandbox: DesktopSandboxProvider,
|
|
228
|
+
action: ComputerAction,
|
|
229
|
+
scale: CaptureScale,
|
|
230
|
+
): Promise<void> {
|
|
231
|
+
const mapX = (x: number) => Math.round(x * scale.scaleX);
|
|
232
|
+
const mapY = (y: number) => Math.round(y * scale.scaleY);
|
|
176
233
|
const [x, y] = action.coordinate
|
|
177
|
-
? [
|
|
234
|
+
? [mapX(action.coordinate[0]), mapY(action.coordinate[1])]
|
|
178
235
|
: [0, 0];
|
|
179
236
|
|
|
180
237
|
switch (action.type) {
|
|
@@ -185,12 +242,18 @@ async function executeAction(sandbox: DesktopSandboxProvider, action: ComputerAc
|
|
|
185
242
|
case "move": await sandbox.moveMouse(x, y); break;
|
|
186
243
|
case "type": await sandbox.write(action.text ?? ""); break;
|
|
187
244
|
case "key": await sandbox.press(action.key ?? ""); break;
|
|
188
|
-
case "scroll":
|
|
245
|
+
case "scroll":
|
|
246
|
+
// The scroll POSITION is a coordinate — mapped (the provider has no
|
|
247
|
+
// positional scroll, so position lands via moveMouse). The scroll
|
|
248
|
+
// DELTA (ticks) is not a coordinate — untouched.
|
|
249
|
+
if (action.coordinate) await sandbox.moveMouse(x, y);
|
|
250
|
+
await sandbox.scroll(action.direction === "up" ? "up" : "down", action.ticks ?? 3);
|
|
251
|
+
break;
|
|
189
252
|
case "drag":
|
|
190
253
|
if (action.startCoordinate && action.endCoordinate) {
|
|
191
254
|
await sandbox.drag(
|
|
192
|
-
[
|
|
193
|
-
[
|
|
255
|
+
[mapX(action.startCoordinate[0]), mapY(action.startCoordinate[1])],
|
|
256
|
+
[mapX(action.endCoordinate[0]), mapY(action.endCoordinate[1])],
|
|
194
257
|
);
|
|
195
258
|
}
|
|
196
259
|
break;
|
package/src/sandbox/devbox.ts
CHANGED
|
@@ -5,13 +5,13 @@
|
|
|
5
5
|
* This module is the SINGLE SOURCE for the machine's template identity: the
|
|
6
6
|
* recipe lives in `infra/e2b-template/devbox.ts`, `infra/e2b-template/build.ts`
|
|
7
7
|
* bakes it under the alias below, and the server imports the alias from here to
|
|
8
|
-
* provision a machine (exactly how `
|
|
9
|
-
* a stable ALIAS, never a snapshot id, so the same string resolves
|
|
10
|
-
* E2B account a deployment uses).
|
|
8
|
+
* provision a machine (exactly how `e2bAgentEnvTemplate` reaches the E2B
|
|
9
|
+
* provider — a stable ALIAS, never a snapshot id, so the same string resolves
|
|
10
|
+
* in whichever E2B account a deployment uses).
|
|
11
11
|
*
|
|
12
12
|
* Interactive only. Workflow RUNS never execute on a devbox (ADR-0038 §2.6):
|
|
13
|
-
* runs boot the clean per-size `agent-
|
|
14
|
-
*
|
|
13
|
+
* runs boot the clean per-size `agent-env-<size>` template (the same image
|
|
14
|
+
* sessions boot) so reproducibility never inherits hand-configured drift.
|
|
15
15
|
*/
|
|
16
16
|
|
|
17
17
|
/** Stable E2B template ALIAS for the per-member desktop machine. Baked by
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
* registry entry.
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
|
-
import { Sandbox } from "e2b";
|
|
8
|
+
import { Sandbox, Template } from "e2b";
|
|
9
9
|
import type { CommandHandle } from "e2b";
|
|
10
10
|
import type { Sandbox as Desktop } from "@e2b/desktop";
|
|
11
11
|
import pRetry from "p-retry";
|
|
@@ -13,7 +13,7 @@ import type {
|
|
|
13
13
|
SandboxProvider, SandboxCommandRunOptions, SandboxBackgroundProcess, SandboxPtyHandle,
|
|
14
14
|
} from "../../types/sandbox.js";
|
|
15
15
|
import { toE2bNetwork } from "../network-policy.js";
|
|
16
|
-
import { DEFAULT_SANDBOX_SIZE,
|
|
16
|
+
import { DEFAULT_SANDBOX_SIZE, e2bAgentEnvTemplate, isE2bSupportedSize } from "../sizes.js";
|
|
17
17
|
import { AGENT_COMPOSE_TAG } from "../provider-def.js";
|
|
18
18
|
import type { OwnedSandbox, SandboxProviderDef } from "../provider-def.js";
|
|
19
19
|
|
|
@@ -297,18 +297,24 @@ export const e2bProviderDef: SandboxProviderDef = {
|
|
|
297
297
|
// E2B has no create-time resource knob (specs are baked into the
|
|
298
298
|
// template/snapshot), so honouring `size` on E2B = picking a PRE-SIZED
|
|
299
299
|
// template, not passing the field through. We resolve the boot template
|
|
300
|
-
// from `size` below (`
|
|
300
|
+
// from `size` below (`e2bAgentEnvTemplate(size)`) when the caller gave no
|
|
301
301
|
// explicit template/bootFrom; the field itself is never forwarded to E2B.
|
|
302
302
|
create: async ({ template, timeoutMs, networkPolicy, size, ...rest }) => {
|
|
303
303
|
// `template` is an E2B template id/alias, a snapshot id (a valid create
|
|
304
304
|
// source that persists beyond its origin sandbox — bootFrom parity), or
|
|
305
|
-
// absent. When absent we pick the SIZE-MATCHED platform
|
|
306
|
-
// (`agent-
|
|
307
|
-
//
|
|
308
|
-
//
|
|
309
|
-
//
|
|
310
|
-
//
|
|
311
|
-
//
|
|
305
|
+
// absent. When absent we pick the SIZE-MATCHED platform agent-env alias
|
|
306
|
+
// (`agent-env-<size>`) — the SAME image every cloud session boots
|
|
307
|
+
// (server/src/sandbox/persistent.ts resolves the identical
|
|
308
|
+
// `e2bAgentEnvTemplate(size)`), so a template-less workflow run gets the
|
|
309
|
+
// full session toolchain (claude runtime, dev toolbelt, agentc, fsgw
|
|
310
|
+
// client) instead of a thinner image. One seam, both lanes: the
|
|
311
|
+
// session/run divergence that caused fleet-wide `exit 127`s (missing
|
|
312
|
+
// harness CLIs, fixed by on-demand install in v0.10.73) is structurally
|
|
313
|
+
// gone — there is no separate run-lane template to drift. The per-size
|
|
314
|
+
// aliasing also keeps `resources.size` giving the same machine spec on
|
|
315
|
+
// E2B as on Vercel (cross-provider parity). Only when no size is
|
|
316
|
+
// resolvable at all do we fall back to E2B_DEFAULT_TEMPLATE if set (the
|
|
317
|
+
// per-deployment analogue of Vercel's node24); else E2B's stock base.
|
|
312
318
|
// Self-provisioning runtimes (claude/codex/amp via bootFrom:"reuse") install
|
|
313
319
|
// their CLI on the base and cache it in the captured snapshot, so a
|
|
314
320
|
// template-less first run boots, installs, snapshots. Clamp the lifetime to
|
|
@@ -346,18 +352,19 @@ export const e2bProviderDef: SandboxProviderDef = {
|
|
|
346
352
|
};
|
|
347
353
|
// Boot template resolution, in priority order:
|
|
348
354
|
// 1. explicit `template`/bootFrom (a pinned snapshot or alias),
|
|
349
|
-
// 2. else the SIZE-MATCHED
|
|
350
|
-
// (`size` resolved to DEFAULT_SANDBOX_SIZE
|
|
351
|
-
//
|
|
355
|
+
// 2. else the SIZE-MATCHED agent-env alias `agent-env-<size>` — the
|
|
356
|
+
// session-identical image (`size` resolved to DEFAULT_SANDBOX_SIZE
|
|
357
|
+
// when unset). NEVER the legacy `agent-compose-base-<size>` — that
|
|
358
|
+
// family remains a valid EXPLICIT bootFrom target only,
|
|
352
359
|
// 3. else E2B_DEFAULT_TEMPLATE as an ultimate per-deployment fallback,
|
|
353
360
|
// 4. else E2B's stock base.
|
|
354
|
-
// `32vcpu-64gb` has no
|
|
361
|
+
// `32vcpu-64gb` has no per-size template (Pro caps ~8 vCPU); the
|
|
355
362
|
// register + invoke guards reject it before a run reaches here, so we
|
|
356
|
-
// never synthesize a non-existent `agent-
|
|
363
|
+
// never synthesize a non-existent `agent-env-32vcpu-64gb` alias.
|
|
357
364
|
const resolvedSize = size ?? DEFAULT_SANDBOX_SIZE;
|
|
358
365
|
const tmpl =
|
|
359
366
|
template ??
|
|
360
|
-
(isE2bSupportedSize(resolvedSize) ?
|
|
367
|
+
(isE2bSupportedSize(resolvedSize) ? e2bAgentEnvTemplate(resolvedSize) : undefined) ??
|
|
361
368
|
process.env.E2B_DEFAULT_TEMPLATE;
|
|
362
369
|
return makeE2bSandboxProvider(
|
|
363
370
|
await (tmpl ? Sandbox.create(tmpl, sandboxOpts) : Sandbox.create(sandboxOpts)),
|
|
@@ -383,4 +390,41 @@ export const e2bProviderDef: SandboxProviderDef = {
|
|
|
383
390
|
// E2B snapshots are team-scoped; delete by id. Best-effort like Vercel's.
|
|
384
391
|
await Sandbox.deleteSnapshot(snapshotId, { apiKey: env.E2B_API_KEY });
|
|
385
392
|
},
|
|
393
|
+
snapshotExists: async (snapshotId, env) => {
|
|
394
|
+
// Existence probe for a CREATE SOURCE (the `snapshotResolves` contract:
|
|
395
|
+
// `true` = resolves, `false` = the provider definitively says it does not
|
|
396
|
+
// exist, anything indeterminate PROPAGATES — never reported as missing).
|
|
397
|
+
//
|
|
398
|
+
// e2b 2.30.5 has no GET-snapshot-by-id, so this composes the two
|
|
399
|
+
// documented lookups, cheapest first:
|
|
400
|
+
// 1. `Template.exists` — ONE round-trip to the template-existence
|
|
401
|
+
// endpoint (`GET /templates/aliases/{alias}`), with DOCUMENTED
|
|
402
|
+
// not-found semantics: 404 → false, 403 → exists but owned by
|
|
403
|
+
// another team → true (the SDK's own mapping). Snapshots are
|
|
404
|
+
// templates provider-side, and this probe also resolves the
|
|
405
|
+
// `agent-env-*` template ALIASES that E2B-pinned default templates
|
|
406
|
+
// register as their bootFrom — boot-time platform validation
|
|
407
|
+
// (validatePlatformSnapshots) probes those through this same seam,
|
|
408
|
+
// so a snapshots-only lookup would falsely alert "unresolvable" on
|
|
409
|
+
// every alias.
|
|
410
|
+
// 2. A paged scan of the team's snapshot list (`Sandbox.listSnapshots`)
|
|
411
|
+
// — authoritative for `createSnapshot` artifacts whatever their id
|
|
412
|
+
// shape, reached only when the template probe answered not-found.
|
|
413
|
+
// Bounded by the team's snapshot count (session rings are GC'd to a
|
|
414
|
+
// fixed retain depth), and this is rare-path code: machine-loss
|
|
415
|
+
// recovery and boot validation, never a hot loop.
|
|
416
|
+
// `false` therefore means BOTH documented lookups answered not-found.
|
|
417
|
+
// (`Sandbox.listSnapshots({ sandboxId })` — source-filtered — was
|
|
418
|
+
// rejected as the primary: callers hold only the snapshot id, and a
|
|
419
|
+
// filtered miss would still need the full scan before "absent" is
|
|
420
|
+
// honest.) Transport/auth faults from either call throw — indeterminate.
|
|
421
|
+
const opts = { apiKey: env.E2B_API_KEY };
|
|
422
|
+
if (await Template.exists(snapshotId, opts)) return true;
|
|
423
|
+
const paginator = Sandbox.listSnapshots(opts);
|
|
424
|
+
while (paginator.hasNext) {
|
|
425
|
+
const items = await pRetry(() => paginator.nextItems(), { retries: 3, minTimeout: 500, factor: 2 });
|
|
426
|
+
if (items.some((s) => s.snapshotId === snapshotId || s.names.includes(snapshotId))) return true;
|
|
427
|
+
}
|
|
428
|
+
return false;
|
|
429
|
+
},
|
|
386
430
|
};
|