@agent-compose/sdk 0.8.4 → 0.8.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. package/README.md +213 -189
  2. package/dist/agent/agent-context.d.ts +9 -1
  3. package/dist/agent/agent-loop.d.ts +14 -6
  4. package/dist/agent/perf-sampler.d.ts +27 -2
  5. package/dist/agent/run-agent.d.ts +1 -1
  6. package/dist/client.d.ts +250 -59
  7. package/dist/directives.d.ts +14 -0
  8. package/dist/display.d.ts +7 -0
  9. package/dist/errors.d.ts +1 -1
  10. package/dist/generated/agentc-commands.d.ts +34 -0
  11. package/dist/index.d.ts +13 -11
  12. package/dist/index.js +1692 -194
  13. package/dist/request-context/request-context.d.ts +1 -1
  14. package/dist/runtimes/_cli-agent.d.ts +278 -58
  15. package/dist/runtimes/claude-code.d.ts +90 -1
  16. package/dist/runtimes/claude.d.ts +1 -1
  17. package/dist/runtimes/codex.d.ts +94 -6
  18. package/dist/runtimes/codex.mid-turn-hook.test.d.ts +10 -0
  19. package/dist/runtimes/openai-desktop.d.ts +50 -0
  20. package/dist/runtimes/openai-desktop.js +1689 -211
  21. package/dist/runtimes/openai-desktop.test.d.ts +20 -0
  22. package/dist/runtimes/opencode.d.ts +48 -11
  23. package/dist/runtimes/opencode.test.d.ts +14 -0
  24. package/dist/runtimes/tool-pulse.test.d.ts +17 -0
  25. package/dist/sandbox/baked-clis.d.ts +75 -0
  26. package/dist/sandbox/devbox.d.ts +5 -5
  27. package/dist/sandbox/exec-stream.d.ts +1 -2
  28. package/dist/sandbox/network-policy.d.ts +23 -5
  29. package/dist/sandbox/registry.d.ts +12 -0
  30. package/dist/sandbox/sizes.d.ts +11 -5
  31. package/dist/sandbox.d.ts +5 -3
  32. package/dist/step-invocation/protocol.d.ts +3 -4
  33. package/dist/step-invocation/server.d.ts +2 -2
  34. package/dist/step-invocation/types.d.ts +2 -2
  35. package/dist/types/api-conversations.d.ts +513 -27
  36. package/dist/types/api-factory.d.ts +183 -3
  37. package/dist/types/api-projects.d.ts +480 -0
  38. package/dist/types/api-runs.d.ts +8 -0
  39. package/dist/types/api-scopes.d.ts +32 -3
  40. package/dist/types/conversation-stream.d.ts +27 -1
  41. package/dist/types/execution-context.d.ts +1 -1
  42. package/dist/types/protocol.d.ts +182 -2
  43. package/dist/types/runtime.d.ts +80 -2
  44. package/dist/types/workflow-metadata.d.ts +2 -4
  45. package/dist/types/workflow-plan.d.ts +1 -3
  46. package/dist/utils/bundler.d.ts +23 -0
  47. package/dist/workflow-steps/observability.d.ts +2 -3
  48. package/dist/workflow-steps/runner.d.ts +5 -8
  49. package/dist/workflow-steps/types.d.ts +8 -10
  50. package/dist/workflow-steps/workflow.d.ts +2 -1
  51. package/dist/workflows/engine.d.ts +3 -5
  52. package/dist/workflows/invoke-child.d.ts +2 -2
  53. package/package.json +2 -2
  54. package/src/agent/agent-context.ts +193 -116
  55. package/src/agent/agent-loop.ts +16 -9
  56. package/src/agent/desktop-open.ts +13 -1
  57. package/src/agent/perf-sampler.ts +54 -3
  58. package/src/agent/run-agent.ts +1 -1
  59. package/src/client.ts +418 -80
  60. package/src/directives.ts +21 -1
  61. package/src/display.ts +12 -0
  62. package/src/errors.ts +1 -0
  63. package/src/generated/agentc-commands.ts +571 -0
  64. package/src/index.ts +65 -18
  65. package/src/pause/pause-core.ts +2 -1
  66. package/src/request-context/request-context.ts +1 -1
  67. package/src/runtimes/_cli-agent.ts +607 -132
  68. package/src/runtimes/claude-code.ts +427 -20
  69. package/src/runtimes/claude.ts +1 -1
  70. package/src/runtimes/codex.ts +188 -19
  71. package/src/runtimes/openai-desktop.ts +82 -19
  72. package/src/runtimes/opencode.ts +195 -26
  73. package/src/sandbox/baked-clis.ts +86 -0
  74. package/src/sandbox/devbox.ts +5 -5
  75. package/src/sandbox/exec-stream.ts +1 -2
  76. package/src/sandbox/network-policy.ts +51 -7
  77. package/src/sandbox/providers/e2b.ts +63 -19
  78. package/src/sandbox/providers/vercel.ts +6 -6
  79. package/src/sandbox/registry.ts +19 -1
  80. package/src/sandbox/sizes.ts +11 -5
  81. package/src/sandbox.ts +9 -2
  82. package/src/step-invocation/invoker.ts +2 -6
  83. package/src/step-invocation/protocol.ts +3 -4
  84. package/src/step-invocation/server.ts +2 -2
  85. package/src/types/api-conversations.ts +424 -29
  86. package/src/types/api-factory.ts +189 -3
  87. package/src/types/api-projects.ts +443 -0
  88. package/src/types/api-runs.ts +5 -0
  89. package/src/types/api-scopes.ts +32 -3
  90. package/src/types/conversation-stream.ts +29 -1
  91. package/src/types/execution-context.ts +1 -1
  92. package/src/types/protocol.ts +180 -2
  93. package/src/types/runtime.ts +71 -2
  94. package/src/types/sandbox-environment.ts +1 -2
  95. package/src/types/workflow-metadata.ts +2 -4
  96. package/src/types/workflow-plan.ts +1 -3
  97. package/src/utils/bundler.ts +88 -19
  98. package/src/workflow-steps/observability.ts +2 -3
  99. package/src/workflow-steps/runner.ts +5 -8
  100. package/src/workflow-steps/types.ts +8 -10
  101. package/src/workflow-steps/workflow.ts +2 -1
  102. package/src/workflows/engine.ts +3 -5
  103. package/src/workflows/invoke-child.ts +2 -2
  104. package/dist/pause/__tests__/errors.test.d.ts +0 -1
  105. package/dist/pause/__tests__/wrappers.test.d.ts +0 -1
  106. package/dist/step-invocation/__tests__/protocol.test.d.ts +0 -1
@@ -34,9 +34,14 @@
34
34
  * token-metering gateway's Anthropic passthrough (ADR-0039).
35
35
  */
36
36
 
37
- import type { AgentMessage, AgentMessageTaskNotification } from "../index.js";
37
+ import type {
38
+ AgentMessage, AgentMessageCompaction, AgentMessageModelUsage, AgentMessagePlanLimits,
39
+ AgentMessageTaskNotification, AgentMessageTaskProgress, AgentMessageUsage,
40
+ WorkflowProgressEntry,
41
+ } from "../index.js";
38
42
  import { createCliAgentRuntime, shellQuote, type CliAgentSpec, type CliReasoningEffort } from "./_cli-agent.js";
39
43
  import { formatError } from "../utils/errors.js";
44
+ import { CLAUDE_CODE_VERSION } from "../sandbox/baked-clis.js";
40
45
 
41
46
  function now(): string { return new Date().toISOString(); }
42
47
 
@@ -171,6 +176,37 @@ export function parseTaskNotifications(
171
176
  return out;
172
177
  }
173
178
 
179
+ /** Clamp for a steer delivered into a child's thread — same ceiling as a
180
+ * notification report: plenty for any real addendum, bounded against a
181
+ * runaway blob. */
182
+ const SUBAGENT_USER_MESSAGE_MAX = 20_000;
183
+
184
+ /** One user-role TEXT blob from the stream, mapped with sidechain awareness.
185
+ * Task-notification blocks parse into structure wherever they appear (their
186
+ * arrival side is resume-dependent — see the transport notes above). What
187
+ * remains is then split by attribution: TOP-LEVEL text (no parent id) stays
188
+ * unmapped exactly as before — prompt echoes and system reminders are not
189
+ * agent output. SIDECHAIN text (parent id present) is a message landing in
190
+ * a child subagent's thread — a delivered SendMessage steer — and forwards
191
+ * as `subagent_user_message`, except harness plumbing (`<system-reminder>`
192
+ * wrappers the CLI injects into child threads), which no renderer should
193
+ * see. Exported for tests. */
194
+ export function sidechainAwareUserText(
195
+ text: string, parentToolUseId: string | undefined, timestamp: string,
196
+ ): AgentMessage[] {
197
+ const notifications = parseTaskNotifications(text, timestamp);
198
+ if (notifications.length > 0) return notifications;
199
+ if (!parentToolUseId) return [];
200
+ const trimmed = text.trim();
201
+ if (trimmed.length === 0 || trimmed.startsWith("<system-reminder>")) return [];
202
+ return [{
203
+ type: "subagent_user_message",
204
+ text: clip(trimmed, SUBAGENT_USER_MESSAGE_MAX),
205
+ parentToolUseId,
206
+ timestamp,
207
+ }];
208
+ }
209
+
174
210
  /** One `system`/`task_notification` stream-json event mapped onto the same
175
211
  * structured message the XML parse produces, or null when the event names
176
212
  * no task id. Pure and tolerant over untrusted harness JSON: absent fields
@@ -206,6 +242,275 @@ export function parseSystemTaskNotification(
206
242
  };
207
243
  }
208
244
 
245
+ // ── Task progress (live background-workflow evidence) ───────────────────────
246
+ //
247
+ // While a harness Workflow (the CLI's in-harness dynamic-workflow tool)
248
+ // runs in the background, stream-json emits `system`/`task_progress`
249
+ // events carrying the workflow's CUMULATIVE `workflow_progress` array —
250
+ // verified live on 2.1.241 (`-p --output-format stream-json`): the array
251
+ // re-sends on every agent state change immediately and on a ~10s heartbeat
252
+ // otherwise, each `workflow_agent` entry updated in place (label,
253
+ // phase, model, state, tokens, toolCalls, durationMs). This is the ONLY
254
+ // per-agent evidence the wire ever carries — the completion notification
255
+ // reports aggregates only — so dropping it leaves workflow cards blind.
256
+ // Preview text (promptPreview / resultPreview) is internal plumbing and is
257
+ // not forwarded, same rule as task notifications.
258
+
259
+ const PROGRESS_PHASES_MAX = 32;
260
+ const PROGRESS_AGENTS_MAX = 128;
261
+ const PROGRESS_LABEL_MAX = 200;
262
+ const PROGRESS_ERROR_MAX = 500;
263
+
264
+ const AGENT_STATES: ReadonlySet<string> = new Set(["start", "progress", "done", "error"]);
265
+
266
+ /** One `system`/`task_progress` event mapped onto the structured progress
267
+ * message, or null only when it names no task. Workflow tasks carry the
268
+ * cumulative `workflow_progress` entries; a plain background Agent-task
269
+ * heartbeat (no workflow entries — verified live on 2.1.241, ~10s cadence
270
+ * while the task runs) forwards with `workflow: []` so the platform can
271
+ * DECLARE the running task as session background work (the delegation-
272
+ * survives-the-human contract: parks, user interrupts, and supersedes all
273
+ * read the declared-children registry). Completion evidence still rides
274
+ * `task_notification`. Pure and tolerant over untrusted harness JSON:
275
+ * malformed entries are skipped, everything is clamped (phases keep the
276
+ * FIRST 32 — they are seeded up front; agents keep the LAST 128 — late
277
+ * spawns matter more than a long-settled prefix). Exported for tests. */
278
+ export function parseSystemTaskProgress(
279
+ p: Record<string, unknown>, timestamp: string,
280
+ ): AgentMessageTaskProgress | null {
281
+ const taskId = typeof p.task_id === "string" && p.task_id.trim().length > 0 ? p.task_id.trim() : null;
282
+ if (!taskId) return null;
283
+ const raw = Array.isArray(p.workflow_progress) ? p.workflow_progress : [];
284
+ const nonneg = (v: unknown): number | undefined =>
285
+ typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
286
+ const str = (v: unknown, max: number): string | undefined =>
287
+ typeof v === "string" && v.length > 0 ? clip(v, max) : undefined;
288
+ const phases: WorkflowProgressEntry[] = [];
289
+ const agents: WorkflowProgressEntry[] = [];
290
+ for (const e of raw) {
291
+ if (typeof e !== "object" || e === null) continue;
292
+ const o = e as Record<string, unknown>;
293
+ const index = nonneg(o.index);
294
+ if (index === undefined) continue;
295
+ if (o.type === "workflow_phase") {
296
+ const title = str(o.title, PROGRESS_LABEL_MAX);
297
+ if (title && phases.length < PROGRESS_PHASES_MAX) phases.push({ kind: "phase", index, title });
298
+ continue;
299
+ }
300
+ if (o.type !== "workflow_agent") continue;
301
+ const label = str(o.label, PROGRESS_LABEL_MAX);
302
+ const state = typeof o.state === "string" && AGENT_STATES.has(o.state) ? o.state as "start" | "progress" | "done" | "error" : null;
303
+ if (!label || !state) continue;
304
+ agents.push({
305
+ kind: "agent", index, label, state,
306
+ ...(nonneg(o.phaseIndex) !== undefined ? { phaseIndex: nonneg(o.phaseIndex) } : {}),
307
+ ...(str(o.phaseTitle, PROGRESS_LABEL_MAX) ? { phaseTitle: str(o.phaseTitle, PROGRESS_LABEL_MAX) } : {}),
308
+ ...(str(o.model, 100) ? { model: str(o.model, 100) } : {}),
309
+ ...(str(o.agentId, 128) ? { agentId: str(o.agentId, 128) } : {}),
310
+ ...(nonneg(o.tokens) !== undefined ? { tokens: nonneg(o.tokens) } : {}),
311
+ ...(nonneg(o.toolCalls) !== undefined ? { toolCalls: nonneg(o.toolCalls) } : {}),
312
+ ...(nonneg(o.durationMs) !== undefined ? { durationMs: nonneg(o.durationMs) } : {}),
313
+ ...(nonneg(o.startedAt) !== undefined ? { startedAt: nonneg(o.startedAt) } : {}),
314
+ ...(state === "error" && str(o.error, PROGRESS_ERROR_MAX) ? { error: str(o.error, PROGRESS_ERROR_MAX) } : {}),
315
+ ...(o.cached === true ? { cached: true as const } : {}),
316
+ ...(o.skipped === true ? { skipped: true as const } : {}),
317
+ });
318
+ }
319
+ // Empty = a plain Agent-task heartbeat, forwarded as liveness evidence
320
+ // (see the header) — the workflow delta folder ignores it (no entries),
321
+ // and only the background-work declare consumes it.
322
+ const workflow = [...phases, ...agents.slice(-PROGRESS_AGENTS_MAX)];
323
+ const toolUseId = typeof p.tool_use_id === "string" && p.tool_use_id.length > 0 ? p.tool_use_id : null;
324
+ const u = (typeof p.usage === "object" && p.usage !== null ? p.usage : {}) as Record<string, unknown>;
325
+ const tokens = nonneg(u.total_tokens);
326
+ const toolUses = nonneg(u.tool_uses);
327
+ const durationMs = nonneg(u.duration_ms);
328
+ return {
329
+ type: "task_progress",
330
+ taskId: clip(taskId, 128),
331
+ ...(toolUseId ? { toolUseId: clip(toolUseId, 128) } : {}),
332
+ ...(tokens !== undefined || toolUses !== undefined || durationMs !== undefined
333
+ ? { usage: {
334
+ ...(tokens !== undefined ? { tokens } : {}),
335
+ ...(toolUses !== undefined ? { toolUses } : {}),
336
+ ...(durationMs !== undefined ? { durationMs } : {}),
337
+ } }
338
+ : {}),
339
+ workflow,
340
+ timestamp,
341
+ };
342
+ }
343
+
344
+ // ── Context compaction (the harness summarizing its own conversation) ───────
345
+ //
346
+ // Captured live against 2.1.212 (the baked E2B pin) and 2.1.241 — identical
347
+ // wire shapes on both, `/compact` and forced-auto alike:
348
+ //
349
+ // {"type":"system","subtype":"status","status":"compacting"}
350
+ // {"type":"system","subtype":"status","status":null,
351
+ // "compact_result":"success"|"failed"[,"compact_error":"…"]}
352
+ // {"type":"system","subtype":"init",…} (fresh init, success only)
353
+ // {"type":"system","subtype":"compact_boundary","compact_metadata":
354
+ // {"trigger":"auto"|"manual","pre_tokens":N,"post_tokens":N,
355
+ // "cumulative_dropped_tokens":N,"duration_ms":N,"preserved_segment":…}}
356
+ //
357
+ // Followed by the continuation summary as a SYNTHETIC user message (string
358
+ // content, `isSynthetic: true`) — deliberately NOT forwarded: it quotes
359
+ // conversation content verbatim (its "All user messages" section replays
360
+ // platform notices word for word), which is exactly how the 2026-08-23
361
+ // incident re-labeled a platform system notice as a "recalled memory".
362
+ // The boundary does not replay on later resumes (verified live).
363
+ //
364
+ // A compaction spans 13-21s in the small captures and MINUTES at real
365
+ // context sizes — forwarding the lifecycle is what lets downstream render
366
+ // the silence as work instead of death (the 94%-auto-compact dead-air
367
+ // incident).
368
+
369
+ const COMPACT_ERROR_MAX = 500;
370
+
371
+ /** One `system` event of the compaction family mapped onto the structured
372
+ * lifecycle message, or null when the event is not compaction-shaped (a
373
+ * plain `status` event with neither the compacting status nor a
374
+ * compact_result stays unmapped). Pure and tolerant over untrusted harness
375
+ * JSON. Exported for tests. */
376
+ export function parseSystemCompaction(
377
+ p: Record<string, unknown>, timestamp: string,
378
+ ): AgentMessageCompaction | null {
379
+ const nonneg = (v: unknown): number | undefined =>
380
+ typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.floor(v) : undefined;
381
+ if (p.subtype === "status") {
382
+ if (p.status === "compacting") return { type: "compaction", phase: "start", timestamp };
383
+ if (typeof p.compact_result === "string") {
384
+ const failed = p.compact_result !== "success";
385
+ const error = typeof p.compact_error === "string" && p.compact_error.length > 0
386
+ ? clip(p.compact_error, COMPACT_ERROR_MAX) : undefined;
387
+ return {
388
+ type: "compaction", phase: "settled",
389
+ result: failed ? "failed" : "success",
390
+ ...(failed && error ? { error } : {}),
391
+ timestamp,
392
+ };
393
+ }
394
+ return null;
395
+ }
396
+ if (p.subtype !== "compact_boundary") return null;
397
+ const meta = (typeof p.compact_metadata === "object" && p.compact_metadata !== null
398
+ ? p.compact_metadata : {}) as Record<string, unknown>;
399
+ const preTokens = nonneg(meta.pre_tokens);
400
+ const postTokens = nonneg(meta.post_tokens);
401
+ const droppedTokens = nonneg(meta.cumulative_dropped_tokens);
402
+ const durationMs = nonneg(meta.duration_ms);
403
+ return {
404
+ type: "compaction", phase: "boundary",
405
+ trigger: meta.trigger === "manual" ? "manual" : "auto",
406
+ ...(preTokens !== undefined ? { preTokens } : {}),
407
+ ...(postTokens !== undefined ? { postTokens } : {}),
408
+ ...(droppedTokens !== undefined ? { droppedTokens } : {}),
409
+ ...(durationMs !== undefined ? { durationMs } : {}),
410
+ timestamp,
411
+ };
412
+ }
413
+
414
+ // ── Plan limits (the account meter, off the stream) ─────────────────────────
415
+ //
416
+ // A subscription-funded `claude -p` reads the `anthropic-ratelimit-unified-*`
417
+ // headers off every response and re-emits them as `rate_limit_event` lines
418
+ // whenever its reading changes (the public Agent SDK's SDKRateLimitEvent;
419
+ // anthropics/claude-code#50518 shows the plainly-allowed shape:
420
+ // `{status: "allowed", resetsAt: 1729281600, rateLimitType: "five_hour"}`,
421
+ // with `utilization` carried only once a window crosses a warning
422
+ // threshold). `resetsAt` is epoch SECONDS. A `rejected` event is the plan's
423
+ // wall for the request just made, with the authoritative reset. The platform
424
+ // folds these into the funding account's meter (one reading per window) and
425
+ // learns a cooldown from a rejected one — never from the notice text.
426
+
427
+ const PLAN_LIMIT_STATUSES: ReadonlySet<string> = new Set(["allowed", "allowed_warning", "rejected"]);
428
+
429
+ /** Epoch seconds (the CLI's `resetsAt`) as ISO; undefined for anything else. */
430
+ function epochSecondsIso(v: unknown): string | undefined {
431
+ if (typeof v !== "number" || !Number.isFinite(v) || v <= 0) return undefined;
432
+ return new Date(v * 1000).toISOString();
433
+ }
434
+
435
+ /** One `rate_limit_event` mapped onto the structured plan-limits message,
436
+ * or null when the event names no recognisable status. Pure and tolerant
437
+ * over untrusted harness JSON: every field is forwarded only when present
438
+ * and well-typed; nothing is defaulted or invented. Exported for tests. */
439
+ export function parsePlanLimits(
440
+ p: Record<string, unknown>, timestamp: string,
441
+ ): AgentMessagePlanLimits | null {
442
+ const info = (typeof p.rate_limit_info === "object" && p.rate_limit_info !== null
443
+ ? p.rate_limit_info : null) as Record<string, unknown> | null;
444
+ if (!info || typeof info.status !== "string" || !PLAN_LIMIT_STATUSES.has(info.status)) return null;
445
+ const num = (v: unknown): number | undefined =>
446
+ typeof v === "number" && Number.isFinite(v) ? v : undefined;
447
+ const str = (v: unknown): string | undefined =>
448
+ typeof v === "string" && v.length > 0 ? clip(v, 100) : undefined;
449
+ const window = str(info.rateLimitType);
450
+ const resetsAt = epochSecondsIso(info.resetsAt);
451
+ const utilization = num(info.utilization);
452
+ const surpassedThreshold = num(info.surpassedThreshold);
453
+ const overageStatus = typeof info.overageStatus === "string" && PLAN_LIMIT_STATUSES.has(info.overageStatus)
454
+ ? info.overageStatus as AgentMessagePlanLimits["status"] : undefined;
455
+ const overageResetsAt = epochSecondsIso(info.overageResetsAt);
456
+ const overageDisabledReason = str(info.overageDisabledReason);
457
+ const overageInUse = typeof info.isUsingOverage === "boolean" ? info.isUsingOverage
458
+ : typeof info.overageInUse === "boolean" ? info.overageInUse : undefined;
459
+ const overage = overageStatus !== undefined || overageResetsAt !== undefined
460
+ || overageDisabledReason !== undefined || overageInUse !== undefined
461
+ ? {
462
+ ...(overageStatus !== undefined ? { status: overageStatus } : {}),
463
+ ...(overageResetsAt !== undefined ? { resetsAt: overageResetsAt } : {}),
464
+ ...(overageDisabledReason !== undefined ? { disabledReason: overageDisabledReason } : {}),
465
+ ...(overageInUse !== undefined ? { inUse: overageInUse } : {}),
466
+ }
467
+ : undefined;
468
+ const limitScope = str(info.limitScope);
469
+ const errorCode = str(info.errorCode);
470
+ return {
471
+ type: "plan_limits",
472
+ status: info.status as AgentMessagePlanLimits["status"],
473
+ ...(window !== undefined ? { window } : {}),
474
+ ...(resetsAt !== undefined ? { resetsAt } : {}),
475
+ ...(utilization !== undefined ? { utilization } : {}),
476
+ ...(surpassedThreshold !== undefined ? { surpassedThreshold } : {}),
477
+ ...(overage !== undefined ? { overage } : {}),
478
+ ...(limitScope !== undefined ? { limitScope } : {}),
479
+ ...(errorCode !== undefined ? { errorCode } : {}),
480
+ timestamp,
481
+ };
482
+ }
483
+
484
+ /** The terminal result's per-model report (`result.modelUsage`, keyed by
485
+ * the raw model string) mapped onto the protocol's per-model shape, or
486
+ * undefined when the result carries none. Counts the harness did not
487
+ * report stay absent (the four classes default to 0 only when the entry
488
+ * exists at all — an entry IS a report of that model). Pure; exported for
489
+ * tests. */
490
+ export function parseModelUsage(raw: unknown): Record<string, AgentMessageModelUsage> | undefined {
491
+ if (typeof raw !== "object" || raw === null) return undefined;
492
+ const num = (v: unknown): number => (typeof v === "number" && Number.isFinite(v) ? v : 0);
493
+ const opt = (v: unknown): number | undefined => (typeof v === "number" && Number.isFinite(v) ? v : undefined);
494
+ const out: Record<string, AgentMessageModelUsage> = {};
495
+ for (const [model, entry] of Object.entries(raw as Record<string, unknown>)) {
496
+ if (typeof entry !== "object" || entry === null || model.length === 0) continue;
497
+ const e = entry as Record<string, unknown>;
498
+ const thinkingTokens = opt(e.thinkingTokens);
499
+ const webSearchRequests = opt(e.webSearchRequests);
500
+ const costUsd = opt(e.costUSD);
501
+ out[clip(model, 200)] = {
502
+ inputTokens: num(e.inputTokens),
503
+ outputTokens: num(e.outputTokens),
504
+ cacheReadTokens: num(e.cacheReadInputTokens),
505
+ cacheCreationTokens: num(e.cacheCreationInputTokens),
506
+ ...(thinkingTokens !== undefined ? { thinkingTokens } : {}),
507
+ ...(webSearchRequests !== undefined ? { webSearchRequests } : {}),
508
+ ...(costUsd !== undefined ? { costUsd } : {}),
509
+ };
510
+ }
511
+ return Object.keys(out).length > 0 ? out : undefined;
512
+ }
513
+
209
514
  /** Claude Code's real reasoning knob is its own `--effort <level>` flag
210
515
  * (low|medium|high|xhigh|max — verified against `claude -p --help`). The
211
516
  * CliReasoningEffort union IS the CLI's vocabulary, so the level rides the
@@ -216,6 +521,50 @@ export function parseSystemTaskNotification(
216
521
  export const CLAUDE_CODE_EFFORT_LEVELS: readonly CliReasoningEffort[] =
217
522
  ["low", "medium", "high", "xhigh", "max"];
218
523
 
524
+ /** The Bash PreToolUse hook rtk installs for Claude Code (`rtk init -g
525
+ * --hook-only` writes exactly this entry — matcher `Bash`, command
526
+ * `rtk hook claude` — into ~/.claude/settings.json; the image bakes
527
+ * RTK_VERSION, sandbox/baked-clis.ts, which this payload was run against).
528
+ * `rtk hook claude` reads the PreToolUse JSON on stdin and, when it has a
529
+ * filter for the command, answers with `updatedInput` rewriting `git
530
+ * status` to `rtk git status` (ls, find, grep, test runners, bun, curl,
531
+ * docker, …), so the model reads rtk's compact output. A command it
532
+ * cannot compress — an unknown tool, a pipe into one, command or process
533
+ * substitution, a heredoc, a file redirect, anything already prefixed
534
+ * `rtk`, or a `RTK_DISABLED=1` prefix — gets no output and exit 0, which
535
+ * Claude Code treats as "no decision": the original command runs
536
+ * unchanged (all verified against the binary).
537
+ *
538
+ * The wrapper fails OPEN both ways. `command -v` covers images without
539
+ * rtk (the devbox, a bare Vercel VM): there a bare `rtk hook claude` would
540
+ * fail every Bash call's hook with exit 127 — non-blocking, but a stderr
541
+ * warning per call; rtk's own legacy shell hook degraded the same way
542
+ * ("binary not found: exit 0"). The trailing `exit 0` covers an rtk that
543
+ * cannot answer: Claude Code treats a PreToolUse hook's exit 2 as a BLOCK
544
+ * of the tool call with stderr fed to the model, and clap exits 2 with
545
+ * its usage for a subcommand it does not know — which is how the codex
546
+ * hook (RTK_CODEX_HOOK_COMMAND, codex.ts) blocked every codex shell
547
+ * command on 2026-10-03, when the image's rtk was a cache-served 0.45.0.
548
+ * rtk's hooks never exit non-zero on purpose, so a non-zero exit is a
549
+ * broken rtk, and the compressor must never cost the worker its shell:
550
+ * the exit code is dropped, and the smoke gate's rtk-hook-rewrite check
551
+ * is what proves the rewrite itself. */
552
+ export const RTK_BASH_HOOK_COMMAND =
553
+ "command -v rtk >/dev/null 2>&1 && rtk hook claude; exit 0";
554
+
555
+ /** Settings every platform `claude` launch passes as `--settings` (inline
556
+ * JSON). Claude Code MERGES hook entries across its settings sources
557
+ * instead of replacing them (user → project → local → flag → managed), so
558
+ * this rides beside whatever the machine's own ~/.claude/settings.json
559
+ * carries, and nothing is written to that file — the right outcome, since
560
+ * the server never owns it: home carry tar-restores it whole across
561
+ * machines and the image capture strips it as personal state. */
562
+ export const CLAUDE_CODE_PLATFORM_SETTINGS = {
563
+ hooks: {
564
+ PreToolUse: [{ matcher: "Bash", hooks: [{ type: "command", command: RTK_BASH_HOOK_COMMAND }] }],
565
+ },
566
+ } as const;
567
+
219
568
  export const claudeCodeSpec: CliAgentSpec = {
220
569
  kind: "claude-code",
221
570
  authEnv: "ANTHROPIC_API_KEY",
@@ -225,11 +574,14 @@ export const claudeCodeSpec: CliAgentSpec = {
225
574
  acp: { command: "npx", args: ["--yes", CLAUDE_CODE_ACP_ADAPTER] },
226
575
  // Bare-sandbox fallback only: the agent-env image bakes /usr/local/bin/claude
227
576
  // (same recipe — infra/e2b-template/build.ts), so the probe short-circuits
228
- // on the platform path. Anthropic's installer drops a versioned binary under
229
- // $HOME with a ~/.local/bin/claude launcher; resolve the symlink and copy
230
- // the self-contained binary to a world-executable system path.
577
+ // on the platform path. The SAME pinned version as the image
578
+ // (CLAUDE_CODE_VERSION), never latest, so a machine the image predates runs
579
+ // what the image's machines run. Anthropic's installer drops a versioned
580
+ // binary under $HOME with a ~/.local/bin/claude launcher; resolve the
581
+ // symlink and copy the self-contained binary to a world-executable system
582
+ // path.
231
583
  install:
232
- 'curl -fsSL https://claude.ai/install.sh | bash && ' +
584
+ `curl -fsSL https://claude.ai/install.sh | bash -s ${CLAUDE_CODE_VERSION} && ` +
233
585
  'REAL=$(readlink -f "$HOME/.local/bin/claude") && test -f "$REAL" && ' +
234
586
  'sudo cp "$REAL" /usr/local/bin/claude && sudo chmod 0755 /usr/local/bin/claude',
235
587
  // Claude reads the prompt from stdin in -p mode.
@@ -244,9 +596,18 @@ export const claudeCodeSpec: CliAgentSpec = {
244
596
  // the same build). A message that lands after the result would start a
245
597
  // NEW turn in-process, which is why the transport's feeder stops at the
246
598
  // result line instead of forwarding past it.
247
- streamInput: {
599
+ midTurnInput: {
600
+ transport: "stdin-stream",
248
601
  promptLine: (prompt) => JSON.stringify({ type: "user", message: { role: "user", content: prompt } }),
249
602
  messageLine: (text) => JSON.stringify({ type: "user", message: { role: "user", content: text } }),
603
+ // The ESC equivalent (verified live against claude 2.1.236): the CLI's
604
+ // control layer processes this OUT OF BAND — a 120s foreground Bash call
605
+ // aborted 6s in, the control_response answered instantly, the run ended
606
+ // ~100ms later with `result: error_during_execution`, and `--resume` on
607
+ // the same session id kept the full turn context.
608
+ interruptLine: (requestId) => JSON.stringify({
609
+ type: "control_request", request_id: requestId, request: { subtype: "interrupt" },
610
+ }),
250
611
  },
251
612
  buildCommand: ({ promptPath, sessionId, model, cwd, effort, streamInput }) => {
252
613
  const flags = [
@@ -268,6 +629,10 @@ export const claudeCodeSpec: CliAgentSpec = {
268
629
  // claude's own attestation env for exactly this: it lifts the flag's
269
630
  // root-user refusal (the E2B agent-env user is root).
270
631
  "--dangerously-skip-permissions",
632
+ // rtk's Bash PreToolUse hook (RTK_BASH_HOOK_COMMAND): shell output is
633
+ // compressed before it reaches the model. Flag settings merge with the
634
+ // machine's own settings; they never replace them.
635
+ `--settings ${shellQuote(JSON.stringify(CLAUDE_CODE_PLATFORM_SETTINGS))}`,
271
636
  ...(model ? [`--model ${shellQuote(model)}`] : []),
272
637
  // Reasoning effort is the CLI's own flag; the value comes from the
273
638
  // closed CliReasoningEffort set, so it is shell-safe unquoted.
@@ -285,9 +650,10 @@ export const claudeCodeSpec: CliAgentSpec = {
285
650
  // tool_use id; null at top level). Forwarded on tool_use/tool_result so
286
651
  // renderers can nest child activity under the spawning call instead of
287
652
  // flattening it into the parent transcript unattributed.
288
- const parent = typeof p.parent_tool_use_id === "string" && p.parent_tool_use_id.length > 0
289
- ? { parentToolUseId: p.parent_tool_use_id }
290
- : {};
653
+ const parentId = typeof p.parent_tool_use_id === "string" && p.parent_tool_use_id.length > 0
654
+ ? p.parent_tool_use_id
655
+ : undefined;
656
+ const parent = parentId ? { parentToolUseId: parentId } : {};
291
657
  switch (p.type) {
292
658
  // Assistant API message: content blocks → text / thinking / tool_use.
293
659
  case "assistant": {
@@ -321,12 +687,25 @@ export const claudeCodeSpec: CliAgentSpec = {
321
687
  // User API message: the CLI echoes tool results back as user content,
322
688
  // and injects `<task-notification>` blocks (background-task completion
323
689
  // evidence) as user TEXT — parsed into structure, never dropped and
324
- // never forwarded raw. Other user text (the echo of the prompt, system
325
- // reminders) stays unmapped: it is not agent output.
690
+ // never forwarded raw. TOP-LEVEL user text (the echo of the prompt,
691
+ // system reminders) stays unmapped: it is not agent output. SIDECHAIN
692
+ // user text (parent_tool_use_id present) is a message landing in a
693
+ // CHILD's thread — the delivered form of a SendMessage steer to a
694
+ // running subagent ("queued for delivery at its next tool round") —
695
+ // and forwards as `subagent_user_message` so the steer renders inside
696
+ // the child's mini-session instead of vanishing (task #97; before
697
+ // this, a queued steer was visible only as the parent's opaque
698
+ // SendMessage tool call).
326
699
  case "user": {
327
700
  const message = p.message as { content?: ClaudeContentBlock[] | string } | undefined;
701
+ // HARNESS-synthesized user messages (`isSynthetic` — e.g. a
702
+ // post-compaction continuation summary, including a CHILD's own)
703
+ // are never a delivered steer: suppress the steer arm, keep the
704
+ // notification parse. Structural marker only, same doctrine as the
705
+ // `<synthetic>` model stamp above.
706
+ const steerParent = p.isSynthetic === true ? undefined : parentId;
328
707
  if (typeof message?.content === "string") {
329
- return parseTaskNotifications(message.content, ts);
708
+ return sidechainAwareUserText(message.content, steerParent, ts);
330
709
  }
331
710
  const blocks = Array.isArray(message?.content) ? message.content : [];
332
711
  return blocks.flatMap((b): AgentMessage[] => {
@@ -337,7 +716,7 @@ export const claudeCodeSpec: CliAgentSpec = {
337
716
  }];
338
717
  }
339
718
  if (b.type === "text" && typeof b.text === "string") {
340
- return parseTaskNotifications(b.text, ts);
719
+ return sidechainAwareUserText(b.text, steerParent, ts);
341
720
  }
342
721
  return [];
343
722
  });
@@ -389,14 +768,21 @@ export const claudeCodeSpec: CliAgentSpec = {
389
768
  return [];
390
769
  }
391
770
  // Terminal result: usage on success (the text already streamed via the
392
- // assistant events); an honest string error on failure.
771
+ // assistant events); an honest string error on failure. The turn
772
+ // totals are the main loop's `usage`; `modelUsage` (every model the
773
+ // query pipeline called, with the CLI's own cost estimate) and
774
+ // `total_cost_usd` ride beside them as the harness reported them —
775
+ // cumulative for the guest session (AgentMessageUsage.byModel).
393
776
  case "result": {
394
777
  if (p.is_error === true) {
395
778
  return [{ type: "error", text: formatError(p.result ?? p.subtype ?? p), timestamp: ts }];
396
779
  }
397
780
  const u = p.usage as Record<string, number> | undefined;
398
781
  if (!u) return [];
399
- return [{
782
+ const byModel = parseModelUsage(p.modelUsage);
783
+ const costUsd = typeof p.total_cost_usd === "number" && Number.isFinite(p.total_cost_usd)
784
+ ? p.total_cost_usd : undefined;
785
+ const usage: AgentMessageUsage = {
400
786
  type: "usage",
401
787
  inputTokens: u.input_tokens ?? 0,
402
788
  outputTokens: u.output_tokens ?? 0,
@@ -404,17 +790,38 @@ export const claudeCodeSpec: CliAgentSpec = {
404
790
  cacheCreationTokens: u.cache_creation_input_tokens ?? 0,
405
791
  durationMs: typeof p.duration_ms === "number" ? p.duration_ms : 0,
406
792
  numTurns: typeof p.num_turns === "number" ? p.num_turns : 1,
793
+ ...(byModel !== undefined ? { byModel } : {}),
794
+ ...(costUsd !== undefined ? { costUsd } : {}),
407
795
  timestamp: ts,
408
- }];
796
+ };
797
+ return [usage];
798
+ }
799
+ // The account's plan limits as the CLI read them off the response
800
+ // headers (subscription-funded sessions; parsePlanLimits above).
801
+ case "rate_limit_event": {
802
+ const limits = parsePlanLimits(p, ts);
803
+ return limits ? [limits] : [];
409
804
  }
410
805
  // System events: init carries the session id (extractSessionId), and
411
806
  // the background-task lane rides here too — `task_notification` is
412
807
  // the completion evidence stream-json actually emits (see the section
413
- // header above), so it maps to the structured message BOTH turn
414
- // engines persist. Everything else under system (task_started,
415
- // task_updated, background_tasks_changed, thinking_tokens) is
416
- // lifecycle noise here.
808
+ // header above), and `task_progress` the LIVE background-task feed
809
+ // (parseSystemTaskProgress): per-agent workflow entries for Workflow
810
+ // tasks, `workflow: []` heartbeats for plain background Agent tasks —
811
+ // both forward so the platform can declare the running task as
812
+ // session background work. Compaction rides here as well (the status
813
+ // pair + compact_boundary — parseSystemCompaction). Everything else
814
+ // under system (task_started, task_updated, background_tasks_changed,
815
+ // thinking_tokens) is lifecycle noise here.
417
816
  case "system": {
817
+ if (p.subtype === "task_progress") {
818
+ const progress = parseSystemTaskProgress(p, ts);
819
+ return progress ? [progress] : [];
820
+ }
821
+ if (p.subtype === "status" || p.subtype === "compact_boundary") {
822
+ const compaction = parseSystemCompaction(p, ts);
823
+ return compaction ? [compaction] : [];
824
+ }
418
825
  if (p.subtype !== "task_notification") return [];
419
826
  const notification = parseSystemTaskNotification(p, ts);
420
827
  return notification ? [notification] : [];
@@ -1,4 +1,4 @@
1
- /** Claude Agent SDK runtime — replaces the old Claude CLI subprocess runtime. */
1
+ /** Claude Agent SDK runtime. */
2
2
 
3
3
  import { query, type HookCallback, type PreToolUseHookInput, type SDKUserMessage, type ThinkingConfig } from "@anthropic-ai/claude-agent-sdk";
4
4
  import type { AgentMessage, ModelExecutionContract, RuntimeOptions, SandboxProvider, ToolCallGateResult } from "../index.js";