@agent-compose/sdk 0.5.0 → 0.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. package/dist/active-step.d.ts +60 -0
  2. package/dist/agent/agent-loop-steer.test.d.ts +1 -0
  3. package/dist/agent/agent-loop.d.ts +46 -0
  4. package/dist/agent/async-queue.d.ts +29 -0
  5. package/dist/agent/protocol.d.ts +9 -1
  6. package/dist/agent/resolve-agent-id.test.d.ts +1 -0
  7. package/dist/agent/run-agent.d.ts +16 -3
  8. package/dist/agent/steer-control.d.ts +57 -0
  9. package/dist/agent/steer-control.test.d.ts +1 -0
  10. package/dist/client.d.ts +161 -0
  11. package/dist/index.d.ts +7 -4
  12. package/dist/index.js +1341 -157
  13. package/dist/pause/__tests__/agent-loop-checkpoint.test.d.ts +1 -0
  14. package/dist/pause/__tests__/checkpoint.test.d.ts +1 -0
  15. package/dist/pause/__tests__/errors.test.d.ts +1 -0
  16. package/dist/pause/__tests__/manager.test.d.ts +1 -0
  17. package/dist/pause/__tests__/pause-core.test.d.ts +1 -0
  18. package/dist/pause/__tests__/state-dir.test.d.ts +1 -0
  19. package/dist/pause/__tests__/wrappers.test.d.ts +1 -0
  20. package/dist/pause/checkpoint.d.ts +28 -0
  21. package/dist/pause/errors.d.ts +52 -0
  22. package/dist/pause/manager.d.ts +63 -0
  23. package/dist/pause/pause-core.d.ts +101 -0
  24. package/dist/pause/state-dir.d.ts +80 -0
  25. package/dist/pause/wrappers.d.ts +41 -0
  26. package/dist/request-context/request-context.d.ts +12 -0
  27. package/dist/runtimes/claude.d.ts +6 -0
  28. package/dist/runtimes/openai-desktop.d.ts +2 -0
  29. package/dist/runtimes/openai-desktop.js +1338 -156
  30. package/dist/runtimes/vercel.d.ts +12 -0
  31. package/dist/runtimes/vercel.js +50 -7
  32. package/dist/runtimes/vercel.test.d.ts +1 -0
  33. package/dist/sse.d.ts +2 -3
  34. package/dist/step-invocation/index.d.ts +2 -2
  35. package/dist/step-invocation/invoker.d.ts +3 -0
  36. package/dist/step-invocation/protocol.d.ts +12 -0
  37. package/dist/step-invocation/server.d.ts +1 -0
  38. package/dist/step-invocation/types.d.ts +40 -5
  39. package/dist/types/events.d.ts +9 -0
  40. package/dist/types/execution-context.d.ts +25 -0
  41. package/dist/types/protocol.d.ts +8 -0
  42. package/dist/types/runtime.d.ts +55 -0
  43. package/dist/types/sandbox.d.ts +6 -1
  44. package/dist/utils/schemas.d.ts +2 -0
  45. package/dist/workflow-steps/__tests__/pause-wiring.test.d.ts +1 -0
  46. package/dist/workflow-steps/index.d.ts +2 -0
  47. package/dist/workflow-steps/observability.d.ts +43 -11
  48. package/dist/workflow-steps/run-callback.d.ts +39 -0
  49. package/dist/workflow-steps/runner.d.ts +8 -0
  50. package/package.json +1 -1
  51. package/src/active-step.ts +124 -0
  52. package/src/agent/agent-loop.ts +253 -19
  53. package/src/agent/async-queue.ts +61 -0
  54. package/src/agent/protocol.ts +12 -2
  55. package/src/agent/run-agent.ts +184 -8
  56. package/src/agent/steer-control.ts +125 -0
  57. package/src/client.ts +277 -0
  58. package/src/index.ts +18 -2
  59. package/src/pause/checkpoint.ts +44 -0
  60. package/src/pause/errors.ts +70 -0
  61. package/src/pause/manager.ts +177 -0
  62. package/src/pause/pause-core.ts +267 -0
  63. package/src/pause/state-dir.ts +262 -0
  64. package/src/pause/wrappers.ts +79 -0
  65. package/src/request-context/request-context.ts +17 -2
  66. package/src/runtimes/claude.ts +101 -6
  67. package/src/runtimes/openai-desktop.ts +11 -0
  68. package/src/runtimes/vercel.ts +26 -0
  69. package/src/sandbox.ts +45 -17
  70. package/src/sse.ts +8 -6
  71. package/src/step-invocation/index.ts +2 -1
  72. package/src/step-invocation/invoker.ts +107 -29
  73. package/src/step-invocation/protocol.ts +16 -0
  74. package/src/step-invocation/server.ts +45 -12
  75. package/src/step-invocation/types.ts +43 -7
  76. package/src/tools/coding.ts +16 -5
  77. package/src/types/events.ts +9 -0
  78. package/src/types/execution-context.ts +25 -0
  79. package/src/types/protocol.ts +8 -0
  80. package/src/types/runtime.ts +52 -0
  81. package/src/types/sandbox.ts +10 -1
  82. package/src/types/workflow.ts +6 -1
  83. package/src/utils/bundler.ts +8 -3
  84. package/src/utils/schemas.ts +2 -0
  85. package/src/workflow-steps/index.ts +3 -0
  86. package/src/workflow-steps/observability.ts +84 -13
  87. package/src/workflow-steps/run-callback.ts +72 -0
  88. package/src/workflow-steps/runner.ts +70 -8
  89. package/dist/utils/discovery.d.ts +0 -2
  90. package/src/utils/discovery.ts +0 -4
@@ -23,6 +23,18 @@ export declare class VercelRunner implements ModelExecutionContract {
23
23
  private readonly messages;
24
24
  constructor(sandbox: SandboxProvider, options: RuntimeOptions, config: VercelRuntimeConfig);
25
25
  gateToolCall(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
26
+ /** ADR-0006 pause-resume hooks. The Vercel AI SDK holds the running
27
+ * conversation client-side in `this.messages` — every `streamText`
28
+ * call passes the full array as `messages` and reconstructs it from
29
+ * `response.messages` after completion. A pause-induced subprocess
30
+ * exit loses the array; the new subprocess constructs a fresh
31
+ * `VercelRunner` with `this.messages = []` and the next sendMessage
32
+ * would see only the current iteration's prompt — conversation
33
+ * history broken. The agent loop calls these at every iteration
34
+ * boundary so the messages array round-trips through the sandbox
35
+ * snapshot. */
36
+ captureCheckpoint(): unknown;
37
+ restoreCheckpoint(blob: unknown): void;
26
38
  private buildTools;
27
39
  sendMessage(opts: {
28
40
  prompt: string;
@@ -36,6 +36,17 @@ function q(value) {
36
36
  function resolvePath(path, cwd) {
37
37
  return cwd && !isAbsolute(path) ? join(cwd, path) : path;
38
38
  }
39
+ function formatCommandFailure(command, result) {
40
+ const parts = [`Command failed with exit code ${result.exitCode}: ${command}`];
41
+ if (result.stderr.trim())
42
+ parts.push(`stderr:
43
+ ${result.stderr}`);
44
+ if (result.stdout.trim())
45
+ parts.push(`stdout:
46
+ ${result.stdout}`);
47
+ return parts.join(`
48
+ `);
49
+ }
39
50
  async function readFile(sandbox, path, opts) {
40
51
  const script = `
41
52
  const fs = require("fs");
@@ -49,13 +60,19 @@ const end = Math.min(lines.length, start + Math.max(0, limit) - 1);
49
60
  for (let i = start; i <= end; i++) console.log(String(i) + ": " + lines[i - 1]);
50
61
  `;
51
62
  const args = [q(path), opts?.offset !== undefined ? String(opts.offset) : "", opts?.limit !== undefined ? String(opts.limit) : ""].filter(Boolean).map(q).join(" ");
52
- const { stdout } = await sandbox.commands.run(`node -e ${q(script)} ${args}`, { cwd: opts?.cwd, timeoutMs: 30000 });
53
- return stdout;
63
+ const command = `node -e ${q(script)} ${args}`;
64
+ const result = await sandbox.commands.run(command, { cwd: opts?.cwd, timeoutMs: 30000 });
65
+ if (result.exitCode !== 0)
66
+ throw new Error(formatCommandFailure(command, result));
67
+ return result.stdout;
54
68
  }
55
69
  async function readRawFile(sandbox, path, cwd) {
56
70
  const script = `const fs = require("fs"); process.stdout.write(fs.readFileSync(process.argv[1], "utf8"));`;
57
- const { stdout } = await sandbox.commands.run(`node -e ${q(script)} ${q(path)}`, { cwd, timeoutMs: 30000 });
58
- return stdout;
71
+ const command = `node -e ${q(script)} ${q(path)}`;
72
+ const result = await sandbox.commands.run(command, { cwd, timeoutMs: 30000 });
73
+ if (result.exitCode !== 0)
74
+ throw new Error(formatCommandFailure(command, result));
75
+ return result.stdout;
59
76
  }
60
77
  var readTool = {
61
78
  name: "Read",
@@ -116,8 +133,7 @@ var bashTool = {
116
133
  timeoutMs: timeoutMs ?? 120000
117
134
  });
118
135
  if (result.exitCode !== 0)
119
- throw new Error(`Command failed with exit code ${result.exitCode}${result.stdout ? `
120
- ${result.stdout}` : ""}`);
136
+ throw new Error(formatCommandFailure(command, result));
121
137
  return result.stdout;
122
138
  }
123
139
  };
@@ -138,6 +154,7 @@ async function runProcessorChain(processors, selectHook, initial, ctx) {
138
154
  }
139
155
 
140
156
  // src/request-context/request-context.ts
157
+ import { z as z2 } from "zod";
141
158
  var AC_RESERVED_PREFIX = "ac__";
142
159
  var AC_TEAM_ID = "ac__teamId";
143
160
  var AC_RUN_ID = "ac__runId";
@@ -163,6 +180,17 @@ var SERIALISABLE_RESERVED_KEYS = new Set([
163
180
  AC_API_KEY_SCOPES,
164
181
  AC_PARENT_RUN_ID
165
182
  ]);
183
+ var RequestContextWireSchema = z2.object({
184
+ reserved: z2.object({
185
+ teamId: z2.string(),
186
+ runId: z2.string(),
187
+ workflowId: z2.string(),
188
+ factoryId: z2.string().nullable(),
189
+ apiKeyScopes: z2.array(z2.string()).readonly(),
190
+ parentRunId: z2.string().nullable()
191
+ }),
192
+ user: z2.record(z2.string(), z2.unknown())
193
+ });
166
194
 
167
195
  class ReservedKeyError extends Error {
168
196
  key;
@@ -224,7 +252,8 @@ class RequestContext {
224
252
  return new RequestContext(reserved, new Map);
225
253
  }
226
254
  static deserialise(wire) {
227
- const ctx = new RequestContext({ ...wire.reserved, abortSignal: undefined }, new Map(Object.entries(wire.user)));
255
+ const parsed = RequestContextWireSchema.parse(wire);
256
+ const ctx = new RequestContext({ ...parsed.reserved, abortSignal: undefined }, new Map(Object.entries(parsed.user)));
228
257
  return ctx;
229
258
  }
230
259
  withAbortSignal(signal) {
@@ -372,6 +401,20 @@ class VercelRunner {
372
401
  return { kind: "allow", call: verdict.value };
373
402
  return verdict;
374
403
  }
404
+ captureCheckpoint() {
405
+ return { messages: [...this.messages] };
406
+ }
407
+ restoreCheckpoint(blob) {
408
+ if (!blob || typeof blob !== "object") {
409
+ throw new Error("Vercel runtime checkpoint is invalid: expected an object with messages[]");
410
+ }
411
+ const incoming = blob.messages;
412
+ if (!Array.isArray(incoming)) {
413
+ throw new Error("Vercel runtime checkpoint is invalid: expected messages[]");
414
+ }
415
+ this.messages.length = 0;
416
+ this.messages.push(...incoming);
417
+ }
375
418
  buildTools(iteration, signal) {
376
419
  const set = {};
377
420
  for (const t of this.tools) {
@@ -0,0 +1 @@
1
+ export {};
package/dist/sse.d.ts CHANGED
@@ -7,9 +7,8 @@
7
7
  * - `event`: event name (from `event:` line; `""` if absent)
8
8
  * - `data`: parsed JSON payload (the SDK's stream events are always JSON)
9
9
  *
10
- * Malformed payloads are silently skipped — same behaviour as the previous
11
- * CLI-side parser. Caller drives the loop via `for await (...)` and is
12
- * responsible for breaking on terminal events.
10
+ * Malformed payloads throw. A dropped terminal event is worse than a loud
11
+ * protocol error for log consumers.
13
12
  *
14
13
  * Output type stays loose (`Record<string, unknown>`) on purpose: typed
15
14
  * unions like `RunEvent` aren't structurally narrowable from
@@ -17,9 +17,9 @@
17
17
  * file + one tokenised stdout result line. Easy to reason about, easy
18
18
  * to test in isolation.
19
19
  */
20
- export { STEP_RESULT_PREFIX, STEP_ENV, RUNNER_BUNDLE_PATH, RUNNER_COMMAND, stepInputPath, requestContextPath, } from "./protocol.js";
20
+ export { STEP_RESULT_PREFIX, STEP_PAUSE_PREFIX, STEP_ENV, RUNNER_BUNDLE_PATH, RUNNER_COMMAND, stepInputPath, requestContextPath, } from "./protocol.js";
21
21
  export { invokeStep, parseStepResult, buildStepEnvs } from "./invoker.js";
22
22
  export { serveStep } from "./server.js";
23
23
  export type { StepHandler, ServeStepRequest, StepHandlerResult } from "./server.js";
24
24
  export { StepExecutionError } from "./types.js";
25
- export type { StepRequest, StepResult, StepInvocationError } from "./types.js";
25
+ export type { StepRequest, StepResult, StepInvocationError, StepPauseRequest } from "./types.js";
@@ -30,6 +30,7 @@ export declare function buildStepEnvs(args: {
30
30
  runId: string;
31
31
  stepIndex: number;
32
32
  resultToken: string;
33
+ isResume?: boolean;
33
34
  }): Record<string, string>;
34
35
  /** Scan the runner's stdout for the tokenised sentinel line and return a
35
36
  * classified result. Returns `null` when no sentinel is present — the
@@ -51,6 +52,8 @@ export declare function parseStepResult<TOutput = unknown>(stdout: string, resul
51
52
  */
52
53
  export interface InvokeStepOptions {
53
54
  envs?: Record<string, string>;
55
+ /** True when re-entering a step after resolving or expiring a pause. */
56
+ isResume?: boolean;
54
57
  /** Live stdout / stderr from the runner subprocess, line-by-line. Called
55
58
  * from inside `sandbox.commands.run` as chunks arrive. The sentinel line
56
59
  * carrying the protocol result token is filtered out before delivery so
@@ -12,12 +12,23 @@
12
12
  * invoker scans for `<prefix><token>:<json>` so any user `console.log`
13
13
  * with the same prefix but a wrong token is rejected. */
14
14
  export declare const STEP_RESULT_PREFIX = "__AC_STEP_RESULT__";
15
+ /** Parallel sentinel for pause requests. The runner emits this — instead
16
+ * of (not in addition to) a step result — when user code calls
17
+ * `ctx.pause(...)`. Exits cleanly afterwards so the activity can snapshot
18
+ * the now-frozen sandbox and durably wait via Temporal `condition()`.
19
+ * See ADR-0006 §"How it actually pauses — Temporal-native, end-to-end". */
20
+ export declare const STEP_PAUSE_PREFIX = "__AC_STEP_PAUSE__";
15
21
  /** Build the line prefix the runner emits and the invoker scans for. The
16
22
  * runner-side serveStep writes `<prefix><token>:<json>\n`; the invoker's
17
23
  * result parser AND the stdout line splitter both match on this exact
18
24
  * prefix so a future change can't make them drift (review found a
19
25
  * separator mismatch that leaked the sentinel into captured logs). */
20
26
  export declare function stepResultLinePrefix(token: string): string;
27
+ /** Build the pause sentinel line prefix. Same per-invocation token as the
28
+ * result sentinel so the two channels share one secret — a user
29
+ * `console.log` can't forge either without knowing the token, and the
30
+ * invoker can filter both prefixes from captured stdout with one token. */
31
+ export declare function stepPauseLinePrefix(token: string): string;
21
32
  /** Sandbox-side path where dispatch writes the compiled runner bundle.
22
33
  * Both modes (full-mode `dispatch.ts` and step-mode `invokeStep`) spawn
23
34
  * the runner from this path; single source of truth. */
@@ -34,6 +45,7 @@ export declare const STEP_ENV: {
34
45
  readonly STEP_INPUT_PATH: "AC_STEP_INPUT_PATH";
35
46
  readonly REQUEST_CONTEXT_PATH: "AC_REQUEST_CONTEXT_PATH";
36
47
  readonly STEP_RESULT_TOKEN: "AC_STEP_RESULT_TOKEN";
48
+ readonly STEP_RESUME: "AC_STEP_RESUME";
37
49
  };
38
50
  /** Sandbox-side path where the invoker writes the JSON-encoded step input.
39
51
  * The serveStep reads from this exact path; both halves use this helper. */
@@ -27,6 +27,7 @@ import type { StepObservability } from "../workflow-steps/observability.js";
27
27
  export interface ServeStepRequest<TInput = unknown> {
28
28
  runId: string;
29
29
  stepIndex: number;
30
+ isResume: boolean;
30
31
  input: TInput;
31
32
  requestContext: RequestContextWire;
32
33
  }
@@ -2,6 +2,7 @@
2
2
  * StepInvocation public types — what callers (the activity) get back, and
3
3
  * the discriminated error union that classifies failure modes.
4
4
  */
5
+ import { z } from "zod";
5
6
  import type { RequestContextWire } from "../request-context/request-context.js";
6
7
  import type { StepObservability } from "../workflow-steps/observability.js";
7
8
  /** What the invoker needs to drive one step invocation. The step's
@@ -16,20 +17,54 @@ export interface StepRequest<TInput = unknown> {
16
17
  requestContext: RequestContextWire;
17
18
  }
18
19
  /**
19
- * Outcome of one step invocation. Successful runs carry the step's output;
20
- * failed runs carry a kinded error so the caller can distinguish "user
21
- * code threw" from "runner crashed before emitting" from "wire protocol
22
- * violation". The dashboard surfaces the kind to operators; the activity
23
- * uses the kind to pick a useful failRun reason.
20
+ * Outcome of one step invocation. Three terminal states:
21
+ *
22
+ * - `ok: true` — step body resolved with `output`.
23
+ * - `ok: false` — step body or runner errored; kind classifies why.
24
+ * - `ok: "paused"` — step body called `ctx.pause(...)` and exited cleanly.
25
+ * The activity captures a sandbox snapshot and the workflow waits on
26
+ * Temporal `condition()` until something resumes the pause (HTTP
27
+ * resume route, TTL expiry, cancellation). `pauseRequest` is what the
28
+ * caller passed to `ctx.pause` — the route surfaces it on the
29
+ * pending-pauses dashboard so an operator sees the question being
30
+ * asked. `observability` carries any ctx metadata/sub-step/agent events
31
+ * recorded before the pause unwound. See ADR-0006.
24
32
  */
25
33
  export type StepResult<TOutput = unknown> = {
26
34
  ok: true;
27
35
  output: TOutput;
28
36
  observability?: StepObservability;
37
+ } | {
38
+ ok: "paused";
39
+ pauseId: string;
40
+ pauseRequest: StepPauseRequest;
41
+ observability?: StepObservability;
29
42
  } | {
30
43
  ok: false;
31
44
  error: StepInvocationError;
32
45
  };
46
+ /** Wire schema for one pause request. Canonical here — `parseStepResult`
47
+ * imports it for validation, and `StepPauseRequest` is `z.infer`'d from
48
+ * it so the runtime type and the parsed shape can't drift.
49
+ *
50
+ * Loose by design (the inner zod shape stops at orchestration fields
51
+ * the engine needs): the SDK wrapper layer (`requestDecision` /
52
+ * `sleep` / `waitForEvent`) owns its own payload contract, and the
53
+ * engine treats `payload` as opaque. */
54
+ export declare const StepPauseRequestSchema: z.ZodObject<{
55
+ reason: z.ZodString;
56
+ kind: z.ZodEnum<{
57
+ custom: "custom";
58
+ decision: "decision";
59
+ sleep: "sleep";
60
+ event: "event";
61
+ }>;
62
+ payload: z.ZodOptional<z.ZodRecord<z.ZodString, z.ZodUnknown>>;
63
+ ttlMs: z.ZodOptional<z.ZodNumber>;
64
+ correlationKey: z.ZodOptional<z.ZodString>;
65
+ snapshot: z.ZodOptional<z.ZodBoolean>;
66
+ }, z.core.$strip>;
67
+ export type StepPauseRequest = z.infer<typeof StepPauseRequestSchema>;
33
68
  /**
34
69
  * Discriminated error union.
35
70
  *
@@ -14,6 +14,15 @@ export type RunEvent = {
14
14
  seq?: number;
15
15
  agentId: string;
16
16
  label: string;
17
+ /** Tool whitelist passed to `agent({ tools: [...] })`. Drives the
18
+ * per-agent "tools" badge on the dashboard. */
19
+ allowedTools?: string[];
20
+ /** Resolved model id (e.g. `claude-sonnet-4-6`). */
21
+ model?: string;
22
+ /** Short runtime self-identifier (`claude`, `openai-desktop`, …)
23
+ * read off `ModelExecutionContract.kind`. The dashboard maps
24
+ * this to a small runtime icon on the agent card header. */
25
+ runtimeKind?: string;
17
26
  } | {
18
27
  event: "agent.message";
19
28
  runId: string;
@@ -1,6 +1,8 @@
1
1
  /** Shared execution context capabilities for workflow functions and steps. */
2
2
  import type { InvokeAndWaitOptions, RunStatus } from "../client.js";
3
3
  import type { RequestContext } from "../request-context/request-context.js";
4
+ import type { PauseRequest } from "../pause/pause-core.js";
5
+ import type { RequestDecisionRequest, WaitForEventRequest } from "../pause/wrappers.js";
4
6
  import type { SandboxProvider } from "./sandbox.js";
5
7
  /** The identity of this workflow run. */
6
8
  export interface WorkflowRun {
@@ -19,4 +21,27 @@ export interface BaseExecutionContext {
19
21
  setMetadata?: (data: Record<string, unknown>) => Promise<void>;
20
22
  /** Invoke another registered workflow and wait for it to settle. */
21
23
  invokeChild: InvokeChild;
24
+ /**
25
+ * Disk-backed memoise across pause-resume. First call runs `fn` and
26
+ * atomically writes the result to the sandbox; on resume the recorded
27
+ * value is returned and `fn` is NOT re-executed. Use for expensive
28
+ * deterministic transforms; for side effects, use `invokeChild`.
29
+ * See ADR-0006 §"`ctx.checkpoint(name, fn)` — disk-backed memoisation".
30
+ */
31
+ checkpoint<T>(name: string, fn: () => Promise<T> | T): Promise<T>;
32
+ /**
33
+ * Pause for feedback. The step exits and the workflow waits durably until
34
+ * something resolves the pause (a resume call, a TTL expiry); on resume the
35
+ * step body re-runs from the top and this call returns the resume payload.
36
+ * Provide a `schema` to validate the payload, `ttlMs` + `onExpiry` to bound
37
+ * the wait, `correlationKey` for by-key resume. Throws PauseRequestError /
38
+ * PauseExpiredError / PauseSchemaError. See ADR-0006 / ADR-0011.
39
+ */
40
+ pause<T = unknown>(req: PauseRequest<T>): Promise<T>;
41
+ /** Pause for a typed decision (a `schema` is required). Wrapper over `pause`. */
42
+ requestDecision<T>(req: RequestDecisionRequest<T>): Promise<T>;
43
+ /** Lightweight timed pause — resolves after `durationMs`, no snapshot. */
44
+ sleep(durationMs: number): Promise<void>;
45
+ /** Pause until an event resumes by `correlationKey`. Wrapper over `pause`. */
46
+ waitForEvent<T = unknown>(req: WaitForEventRequest<T>): Promise<T>;
22
47
  }
@@ -52,5 +52,13 @@ export interface AgentStatus {
52
52
  completed: string[];
53
53
  blockers: string[];
54
54
  exit_signal: boolean;
55
+ /** PR 7 self-pause (honoured only for `mode: "hitl"` agents). The agent
56
+ * cannot proceed without a human decision: it sets this true, puts the
57
+ * question in `question`, and ends its turn. The loop pauses at the next
58
+ * boundary and injects the human's answer as the next turn. `auto` agents
59
+ * ignore it and keep going. Should be paired with `exit_signal: false`. */
60
+ needs_input?: boolean;
61
+ /** The question to put to the human when `needs_input` is true. */
62
+ question?: string;
55
63
  }
56
64
  export {};
@@ -52,14 +52,69 @@ export type ToolCallGateResult = {
52
52
  export interface ModelExecutionContract {
53
53
  /** True when this runtime can run `processToolCall` before tool execution. */
54
54
  supportsToolCallProcessor?: boolean;
55
+ /** Short self-identifier ("claude", "openai-desktop", "vercel", …). Read
56
+ * by the agent loop and surfaced on `agent.spawned` so the dashboard
57
+ * can show a per-agent runtime icon without re-fetching template
58
+ * metadata. Optional — runtimes that omit it stay anonymous. */
59
+ kind?: string;
60
+ /** Resolved model id used by this runtime instance (already merged with
61
+ * config + runtime defaults). Surfaced on `agent.spawned` so each
62
+ * agent card on the Agent tab can label which model it ran against. */
63
+ model?: string;
55
64
  /** Runtime-owned pre-tool gate. Adapters call the shared processor chain
56
65
  * through this seam; the agent loop stays SDK-agnostic. */
57
66
  gateToolCall?(call: ToolCall, ctx: ProcessorContext): Promise<ToolCallGateResult>;
67
+ /**
68
+ * Capture runtime-private in-memory state that will not survive the
69
+ * runner subprocess exit. Called by the agent loop at pause time,
70
+ * AFTER the loop has flushed its own state to disk.
71
+ *
72
+ * Return value is opaque to the loop — whatever the runtime needs to
73
+ * round-trip its conversation across pause-resume. Must be JSON-
74
+ * serialisable; the loop atomically writes it to
75
+ * `/tmp/wf/state/runtime-<agentInstanceId>.json` and reads it back
76
+ * on resume to hand to `restoreCheckpoint`.
77
+ *
78
+ * Default (method omitted): runtime holds no instance state that
79
+ * needs to round-trip across pause. The shipped example is the
80
+ * Claude runtime — the conversation lives server-side at
81
+ * Anthropic, addressed by `session_id`, and the loop already holds
82
+ * `lastSessionId` as part of its own state. On resume the loop
83
+ * restores the id, the next `sendMessage` passes it through, and
84
+ * Anthropic resumes the server-side conversation.
85
+ *
86
+ * Runtimes that hold the conversation in-process — Vercel's
87
+ * `VercelRunner.messages` is the canonical case — MUST implement
88
+ * both hooks: the messages array is reconstructed from
89
+ * `response.messages` on each `streamText` and would be lost the
90
+ * moment the subprocess exits. See ADR-0006 §"Concrete examples
91
+ * for shipped runtimes" for the audit + worked examples.
92
+ */
93
+ captureCheckpoint?(): unknown;
94
+ /**
95
+ * Restore runtime-private state previously returned by
96
+ * `captureCheckpoint`. Called by the agent loop on resume, AFTER
97
+ * the loop has restored its own state but BEFORE iterations resume.
98
+ *
99
+ * `blob` is whatever this same runtime returned at pause time. If
100
+ * `captureCheckpoint` is omitted, this is never called.
101
+ */
102
+ restoreCheckpoint?(blob: unknown): void;
58
103
  sendMessage(opts: {
59
104
  prompt: string;
60
105
  sessionId?: string;
61
106
  iteration?: number;
62
107
  signal?: AbortSignal;
108
+ /** Push-iterable of mid-turn user messages from outside the agent
109
+ * loop — e.g. dashboard chat injections. Runtimes that support
110
+ * streaming-input mode (Claude Agent SDK) read from this in
111
+ * parallel with the initial `prompt`; the SDK handles delivery
112
+ * at the next safe boundary. Runtimes without streaming-input
113
+ * support ignore this and fall back to per-iteration injection. */
114
+ inboxStream?: AsyncIterable<{
115
+ text: string;
116
+ senderName?: string | null;
117
+ }>;
63
118
  }): AsyncGenerator<AgentMessage>;
64
119
  }
65
120
  /**
@@ -9,11 +9,16 @@ export interface SandboxCommandRunOptions {
9
9
  envs?: Record<string, string>;
10
10
  onStdout?: (data: string) => void;
11
11
  onStderr?: (data: string) => void;
12
- background?: boolean;
12
+ /** Run the command with root privileges. Vercel maps this to its native
13
+ * `sudo` flag; the local provider prepends `sudo`. Defaults to false.
14
+ * Requires the sandbox image to grant the command root (Vercel's runtimes
15
+ * do — passwordless). */
16
+ sudo?: boolean;
13
17
  }
14
18
  export interface SandboxCommandResult {
15
19
  exitCode: number;
16
20
  stdout: string;
21
+ stderr: string;
17
22
  }
18
23
  export interface SandboxProvider {
19
24
  sandboxId: string;
@@ -5,4 +5,6 @@ export declare const AgentStatusSchema: z.ZodObject<{
5
5
  completed: z.ZodArray<z.ZodString>;
6
6
  blockers: z.ZodArray<z.ZodString>;
7
7
  exit_signal: z.ZodBoolean;
8
+ needs_input: z.ZodOptional<z.ZodBoolean>;
9
+ question: z.ZodOptional<z.ZodString>;
8
10
  }, z.core.$strip>;
@@ -8,3 +8,5 @@ export type { Step, StepContext, StepRunResult, Workflow, } from "./types.js";
8
8
  export { WORKFLOW_BRAND } from "./types.js";
9
9
  export { StepObservabilityCollector } from "./observability.js";
10
10
  export type { StepObservability, SubStepEvent } from "./observability.js";
11
+ export { makeRunCallbackEmitterFromEnv } from "./run-callback.js";
12
+ export type { LiveAgentEventEmitter } from "./run-callback.js";
@@ -5,18 +5,32 @@
5
5
  * tokenised stdout sentinel; the activity persists the snapshot to the
6
6
  * run's metadata + lifecycle event tables.
7
7
  *
8
- * STREAMING — known gap. `agentEvents` are currently batched at step
9
- * end. For long-running steps that emit many agent events, this hides
10
- * progress from the dashboard until the step completes. The follow-up
11
- * is to bring back a minimal `/internal/runs/:runId/steps/:stepIndex/events`
12
- * route gated by a per-step JIT-signed token (mint in the activity,
13
- * stamp into the runner env, verify on receive) so `agentEvents.emit`
14
- * POSTs in real time. Until then `metadata` and `subSteps` are
15
- * effectively boundary events (no streaming need) and batching them
16
- * matches their natural granularity.
8
+ * Live streaming + batch backstop — the no-double-write design:
9
+ *
10
+ * 1. `agentEvents.emit` immediately fires the optional `liveEmitter`
11
+ * (the runner's POST to `/internal/runs/.../events`).
12
+ * 2. The emitter returns `Promise<boolean>` — true means the server
13
+ * accepted and persisted this event, false means it failed (POST
14
+ * error, server 5xx, network blip).
15
+ * 3. The collector tracks which seqs were successfully ack'd.
16
+ * 4. At `snapshot()` time we await any in-flight emits (with a small
17
+ * grace window so the agent loop's final-burst posts can finish),
18
+ * then STRIP ack'd events from the returned `events` array.
19
+ *
20
+ * The result: the snapshot's `events` array only contains events
21
+ * that the live path didn't successfully deliver. The server's
22
+ * batch-flush in `persistStepObservability` becomes a true backstop
23
+ * for the FAILURE path — it never re-writes (and never re-notifies)
24
+ * the events the live route already handled. No double pg_notify,
25
+ * no dashboard duplicates.
26
+ *
27
+ * `metadata` and `subSteps` remain batch-only because they're
28
+ * naturally boundary events (no streaming benefit) and aren't
29
+ * written by the live route at all.
17
30
  */
18
31
  import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
19
32
  import type { AgentEventSink } from "../types/workflow.js";
33
+ import type { LiveAgentEventEmitter } from "./run-callback.js";
20
34
  /** One named sub-step (from `ctx.step("name", async () => ...)`).
21
35
  * Becomes a `workflow_substep_*` lifecycle event on the run timeline. */
22
36
  export interface SubStepEvent {
@@ -49,10 +63,28 @@ export declare class StepObservabilityCollector {
49
63
  private metadata;
50
64
  private events;
51
65
  private subSteps;
66
+ private readonly liveEmitter?;
67
+ /** Seqs of events the server confirmed via the live route. */
68
+ private readonly ackedSeqs;
69
+ /** Promises for in-flight live emits — awaited at snapshot time. */
70
+ private readonly inFlight;
71
+ constructor(opts?: {
72
+ liveEmitter?: LiveAgentEventEmitter;
73
+ });
52
74
  readonly setMetadata: (data: Record<string, unknown>) => Promise<void>;
53
75
  readonly step: <T>(name: string, fn: () => Promise<T>) => Promise<T>;
54
76
  readonly agentEvents: AgentEventSink;
77
+ /** Wait for in-flight live emits to settle (or timeout) so the
78
+ * ackedSeqs set is maximally up-to-date before we filter. Used by
79
+ * `snapshot()` — exposed separately for tests. */
80
+ private drainInFlight;
55
81
  /** Snapshot the accumulated state. Returns `undefined` when nothing
56
- * was recorded so the wire payload can drop the field entirely. */
57
- snapshot(): StepObservability | undefined;
82
+ * was recorded so the wire payload can drop the field entirely.
83
+ *
84
+ * Async because we drain in-flight live emits first. Any event the
85
+ * server acknowledged is REMOVED from the returned `events` array
86
+ * so the server-side batch flush doesn't re-write/re-notify it.
87
+ * Events that failed live delivery (POST error, timeout) stay in
88
+ * the array as the durable backstop. */
89
+ snapshot(): Promise<StepObservability | undefined>;
58
90
  }
@@ -0,0 +1,39 @@
1
+ /**
2
+ * Runner → server live event emitter.
3
+ *
4
+ * Pairs with the server-side `/api/v1/internal/runs/:runId/steps/:stepIndex/events`
5
+ * route. The runner reads `AGENT_COMPOSE_URL`, `AGENT_COMPOSE_RUN_TOKEN`,
6
+ * and `RUN_ID` from env; if all three are present it builds a POST
7
+ * function that sends each agent lifecycle event to the server as
8
+ * soon as the agent loop emits it.
9
+ *
10
+ * Non-blocking: the agent loop doesn't await the emitter — the runtime
11
+ * keeps emitting events at full speed. But the emitter still returns a
12
+ * `Promise<boolean>` so the collector can later decide whether to
13
+ * include each event in the batch-end durable backstop. `true` = server
14
+ * accepted the event (don't re-deliver in batch); `false` = POST
15
+ * failed (DO re-deliver). The collector awaits these promises with a
16
+ * short grace window at snapshot time.
17
+ *
18
+ * This is the seam that lets us avoid the double-write problem: every
19
+ * successfully-ack'd live event is stripped from `observability.events`
20
+ * before the batch flush runs, so the durable backstop only carries
21
+ * events that genuinely failed live delivery. No more two paths writing
22
+ * the same row + double pg_notify.
23
+ */
24
+ import type { AgentLifecycleEvent } from "../agent/agent-loop.js";
25
+ export type LiveAgentEventEmitter = (event: AgentLifecycleEvent, seq: number) => Promise<boolean>;
26
+ /**
27
+ * Build a live emitter from process env + the caller-supplied stepIndex.
28
+ * Returns `undefined` when any required env var is missing (local tests,
29
+ * non-sandbox callers) so the caller can wire `undefined` straight
30
+ * through to the collector and get batch-only delivery without
31
+ * conditional plumbing.
32
+ *
33
+ * `stepIndex` is a parameter rather than an env read because
34
+ * `serveStep` scrubs `AC_STEP_INDEX` from `process.env` before invoking
35
+ * the workflow handler (to keep the transport envelope unreachable
36
+ * from user code); the handler still holds the parsed integer and
37
+ * passes it here.
38
+ */
39
+ export declare function makeRunCallbackEmitterFromEnv(stepIndex: number): LiveAgentEventEmitter | undefined;
@@ -27,6 +27,7 @@ import type { RequestContext } from "../request-context/request-context.js";
27
27
  import type { SandboxProvider } from "../types/sandbox.js";
28
28
  import type { WorkflowRun, WorkflowCtx } from "../types/workflow.js";
29
29
  import { type StepObservability } from "./observability.js";
30
+ import type { LiveAgentEventEmitter } from "./run-callback.js";
30
31
  export declare class StepValidationError extends Error {
31
32
  readonly stepName: string;
32
33
  readonly side: "input" | "output";
@@ -67,6 +68,11 @@ export interface RunWorkflowStepsOpts<TInput, TOutput> {
67
68
  onStepStarted?(stepIndex: number, stepName: string): void | Promise<void>;
68
69
  /** Child workflow invocation implementation. Defaults to a clear unsupported error. */
69
70
  invokeChild?: WorkflowCtx["invokeChild"];
71
+ /** Optional live-stream emitter for agent lifecycle events. The runner
72
+ * passes a fetch-based emitter wired to the per-run callback token so
73
+ * the dashboard sees events as the agent loop produces them; tests
74
+ * leave it undefined and get batch-only delivery. */
75
+ liveAgentEventEmitter?: LiveAgentEventEmitter;
70
76
  }
71
77
  export interface RunWorkflowStepsResult<TOutput> {
72
78
  output: TOutput;
@@ -84,6 +90,8 @@ export interface RunWorkflowSingleStepOpts {
84
90
  sandbox?: SandboxProvider;
85
91
  abortSignal?: AbortSignal;
86
92
  invokeChild?: WorkflowCtx["invokeChild"];
93
+ /** Optional live-stream emitter — see `RunWorkflowStepsOpts.liveAgentEventEmitter`. */
94
+ liveAgentEventEmitter?: LiveAgentEventEmitter;
87
95
  }
88
96
  /** Result of one step run — output plus whatever the step's observability
89
97
  * hooks recorded. `observability` is undefined when nothing was buffered,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@agent-compose/sdk",
3
- "version": "0.5.0",
3
+ "version": "0.5.2",
4
4
  "description": "Client library for agent-compose — define agents, runtimes, and workflows, and invoke them against an agent-compose server.",
5
5
  "license": "MIT",
6
6
  "repository": {