@github/copilot-sdk 1.0.8 → 1.0.9-preview.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,5 +1,5 @@
1
- import type { SessionFsHandler, SessionFsStatResult, SessionFsReaddirWithTypesEntry, SessionFsSqliteQueryResult as GeneratedSqliteQueryResult, SessionFsSqliteQueryType } from "./generated/rpc.js";
2
- export type { SessionFsSqliteQueryType };
1
+ import type { SessionFsHandler, SessionFsStatResult, SessionFsReaddirWithTypesEntry, SessionFsSqliteQueryResult as GeneratedSqliteQueryResult, SessionFsSqliteTransactionErrorClass, SessionFsSqliteQueryType } from "./generated/rpc.js";
2
+ export type { SessionFsSqliteQueryType, SessionFsSqliteTransactionErrorClass };
3
3
  /**
4
4
  * File metadata returned by {@link SessionFsProvider.stat}.
5
5
  * Same shape as the generated {@link SessionFsStatResult} but without the
@@ -12,6 +12,31 @@ export type SessionFsFileInfo = Omit<SessionFsStatResult, "error">;
12
12
  * `error` field, since providers signal errors by throwing.
13
13
  */
14
14
  export type SessionFsSqliteQueryResult = Omit<GeneratedSqliteQueryResult, "error">;
15
+ /**
16
+ * One statement in an atomic SQLite transaction passed to
17
+ * {@link SessionFsSqliteProvider.transaction}.
18
+ */
19
+ export interface SessionFsSqliteStatement {
20
+ /** How to execute: `"exec"` for DDL/multi-statement, `"query"` for SELECT, `"run"` for INSERT/UPDATE/DELETE. */
21
+ queryType: SessionFsSqliteQueryType;
22
+ /** SQL statement to execute. */
23
+ query: string;
24
+ /** Optional named bind parameters. */
25
+ params?: Record<string, string | number | null>;
26
+ }
27
+ /**
28
+ * Error thrown by {@link SessionFsSqliteProvider.transaction} to classify a
29
+ * transaction failure for the runtime.
30
+ *
31
+ * Any other thrown value is reported as `"fatal"`. Throw this with
32
+ * `"busyOrLocked"` when SQLite reported BUSY/LOCKED before commit and the
33
+ * transaction was rolled back, so the runtime knows the call is safe to retry.
34
+ */
35
+ export declare class SessionFsSqliteTransactionFailure extends Error {
36
+ /** Failure classification reported to the runtime. */
37
+ readonly errorClass: SessionFsSqliteTransactionErrorClass;
38
+ constructor(message: string, errorClass?: SessionFsSqliteTransactionErrorClass);
39
+ }
15
40
  /**
16
41
  * SQLite operations for the per-session database.
17
42
  * Implementers provide query execution and existence checking.
@@ -25,6 +50,17 @@ export interface SessionFsSqliteProvider {
25
50
  * @param params - Optional named bind parameters.
26
51
  */
27
52
  query(queryType: SessionFsSqliteQueryType, query: string, params?: Record<string, string | number | null>): Promise<SessionFsSqliteQueryResult | undefined>;
53
+ /**
54
+ * Execute `statements` atomically against the per-session database.
55
+ *
56
+ * Apply busy handling to every statement and roll back the whole batch if
57
+ * any statement fails. Throw {@link SessionFsSqliteTransactionFailure} to
58
+ * classify the failure; any other thrown value is reported as `"fatal"`.
59
+ *
60
+ * @param statements - Statements to execute in order inside a single transaction.
61
+ * @returns One result per statement, in the same order.
62
+ */
63
+ transaction?(statements: SessionFsSqliteStatement[]): Promise<SessionFsSqliteQueryResult[]>;
28
64
  /**
29
65
  * Check whether the per-session database already exists, without creating it.
30
66
  */
@@ -1,3 +1,12 @@
1
+ class SessionFsSqliteTransactionFailure extends Error {
2
+ /** Failure classification reported to the runtime. */
3
+ errorClass;
4
+ constructor(message, errorClass = "fatal") {
5
+ super(message);
6
+ this.name = "SessionFsSqliteTransactionFailure";
7
+ this.errorClass = errorClass;
8
+ }
9
+ }
1
10
  function normalizeSqliteParams(params) {
2
11
  if (!params) {
3
12
  return void 0;
@@ -113,6 +122,29 @@ function createSessionFsAdapter(provider) {
113
122
  );
114
123
  return result ?? { rows: [], columns: [], rowsAffected: 0 };
115
124
  },
125
+ sqliteTransaction: async ({ statements }) => {
126
+ if (!provider.sqlite?.transaction) {
127
+ return {
128
+ results: [],
129
+ error: {
130
+ errorClass: "fatal",
131
+ message: "SQLite transactions are not supported by this provider"
132
+ }
133
+ };
134
+ }
135
+ try {
136
+ const results = await provider.sqlite.transaction(
137
+ statements.map((statement) => ({
138
+ queryType: statement.queryType,
139
+ query: statement.query,
140
+ params: normalizeSqliteParams(statement.params)
141
+ }))
142
+ );
143
+ return { results: results.map((result) => ({ ...result })) };
144
+ } catch (err) {
145
+ return { results: [], error: toSqliteTransactionError(err) };
146
+ }
147
+ },
116
148
  sqliteExists: async () => {
117
149
  if (!provider.sqlite) {
118
150
  throw new Error("SQLite is not supported by this provider");
@@ -126,6 +158,16 @@ function toSessionFsError(err) {
126
158
  const code = e.code === "ENOENT" ? "ENOENT" : "UNKNOWN";
127
159
  return { code, message: e.message ?? String(err) };
128
160
  }
161
+ function toSqliteTransactionError(err) {
162
+ if (err instanceof SessionFsSqliteTransactionFailure) {
163
+ return { errorClass: err.errorClass, message: err.message };
164
+ }
165
+ return {
166
+ errorClass: "fatal",
167
+ message: err instanceof Error ? err.message : String(err)
168
+ };
169
+ }
129
170
  export {
171
+ SessionFsSqliteTransactionFailure,
130
172
  createSessionFsAdapter
131
173
  };
package/dist/types.d.ts CHANGED
@@ -20,6 +20,9 @@ export type { SessionFsFileInfo } from "./sessionFsProvider.js";
20
20
  export type { SessionFsSqliteQueryResult } from "./sessionFsProvider.js";
21
21
  export type { SessionFsSqliteQueryType } from "./sessionFsProvider.js";
22
22
  export type { SessionFsSqliteProvider } from "./sessionFsProvider.js";
23
+ export type { SessionFsSqliteStatement } from "./sessionFsProvider.js";
24
+ export type { SessionFsSqliteTransactionErrorClass } from "./sessionFsProvider.js";
25
+ export { SessionFsSqliteTransactionFailure } from "./sessionFsProvider.js";
23
26
  export type { LlmInferenceHeaders } from "./generated/rpc.js";
24
27
  export type { CopilotRequestContext } from "./copilotRequestHandler.js";
25
28
  export { CopilotRequestHandler, CopilotWebSocketHandler, CopilotWebSocketCloseStatus, CopilotWebSocketForwarder, } from "./copilotRequestHandler.js";
@@ -1154,6 +1157,45 @@ export interface ErrorOccurredHookOutput {
1154
1157
  export type ErrorOccurredHandler = (input: ErrorOccurredHookInput, invocation: {
1155
1158
  sessionId: string;
1156
1159
  }) => Promise<ErrorOccurredHookOutput | void> | ErrorOccurredHookOutput | void;
1160
+ /**
1161
+ * Input for the agent-stop hook.
1162
+ *
1163
+ * Fires for the top-level (main) agent when it reaches a natural terminal stop
1164
+ * — i.e. the agent has gone idle without a pending non-terminal tool call and
1165
+ * was not aborted or blocked by a rejected tool. (For sub-agents, the runtime
1166
+ * fires a separate sub-agent stop lifecycle.)
1167
+ */
1168
+ export interface AgentStopHookInput extends BaseHookInput {
1169
+ /** Why the agent stopped (for example, `"end_turn"`). */
1170
+ stopReason?: string;
1171
+ /** Path to the on-disk session transcript, when available. */
1172
+ transcriptPath?: string;
1173
+ /**
1174
+ * True when this stop is a re-entry triggered by a previous agent-stop
1175
+ * `block` decision (Claude-compatible `stop_hook_active` semantics). Lets a
1176
+ * handler avoid blocking indefinitely.
1177
+ */
1178
+ stopHookActive?: boolean;
1179
+ }
1180
+ /**
1181
+ * Output for the agent-stop hook.
1182
+ *
1183
+ * Return `{ decision: "block", reason }` to keep the agent running: the
1184
+ * `reason` is enqueued as a follow-up user message so the agent continues
1185
+ * working (for example, to remediate findings surfaced by the hook). The
1186
+ * runtime caps consecutive blocks to prevent runaway loops. Returning nothing
1187
+ * (or omitting `decision`) lets the agent stop normally.
1188
+ */
1189
+ export interface AgentStopHookOutput {
1190
+ decision?: "block";
1191
+ reason?: string;
1192
+ }
1193
+ /**
1194
+ * Handler for the agent-stop hook.
1195
+ */
1196
+ export type AgentStopHandler = (input: AgentStopHookInput, invocation: {
1197
+ sessionId: string;
1198
+ }) => Promise<AgentStopHookOutput | void> | AgentStopHookOutput | void;
1157
1199
  /**
1158
1200
  * Configuration for session hooks
1159
1201
  */
@@ -1197,6 +1239,15 @@ export interface SessionHooks {
1197
1239
  * Called when an error occurs
1198
1240
  */
1199
1241
  onErrorOccurred?: ErrorOccurredHandler;
1242
+ /**
1243
+ * Called when the top-level agent reaches a natural terminal stop (it went
1244
+ * idle without pending work and was not aborted). Return
1245
+ * `{ decision: "block", reason }` to keep the agent running with `reason`
1246
+ * enqueued as a follow-up message — for example, to have the agent
1247
+ * remediate findings the handler surfaced. Returning nothing lets the
1248
+ * agent stop.
1249
+ */
1250
+ onAgentStop?: AgentStopHandler;
1200
1251
  }
1201
1252
  /**
1202
1253
  * Base interface for MCP server configuration.
@@ -1302,8 +1353,8 @@ export interface CustomAgentConfig {
1302
1353
  model?: string;
1303
1354
  /**
1304
1355
  * Reasoning effort level for this agent's model.
1305
- * When omitted, no per-agent override is sent and the backend chooses its
1306
- * default. The parent session effort is not inherited.
1356
+ * When omitted, the runtime resolves the effort from model configuration,
1357
+ * then inherits the parent effort only if this agent uses the same model.
1307
1358
  */
1308
1359
  reasoningEffort?: ReasoningEffort;
1309
1360
  }
@@ -1474,6 +1525,45 @@ export interface CanvasProviderIdentity {
1474
1525
  /** Optional display name surfaced as the canvas extension name. */
1475
1526
  name?: string;
1476
1527
  }
1528
+ /**
1529
+ * Static resource ceilings declared by a factory before it runs.
1530
+ *
1531
+ * @experimental Part of the experimental Agent Factories surface and may
1532
+ * change or be removed in future SDK or CLI releases.
1533
+ */
1534
+ export interface FactoryLimits {
1535
+ /** Maximum number of factory subagents that may run concurrently. Must be positive when present. */
1536
+ maxConcurrentSubagents?: number;
1537
+ /** Maximum total number of factory subagents that may be spawned. Must be positive when present. */
1538
+ maxTotalSubagents?: number;
1539
+ /** Maximum AI credits consumed by factory subagents and descendants. This post-paid ceiling is soft. */
1540
+ maxAiCredits?: number;
1541
+ /**
1542
+ * Maximum accumulated active-execution time, in seconds. Active execution includes the entire extension body,
1543
+ * subprocess waits, queued-agent waits, and sleeps. The limit is armed from the remaining headroom when a run
1544
+ * resumes; time between attempts is not counted. Must be finite and positive when present.
1545
+ */
1546
+ timeoutSeconds?: number;
1547
+ }
1548
+ /**
1549
+ * Registration metadata for an extension-authored factory.
1550
+ *
1551
+ * @experimental Part of the experimental Agent Factories surface and may
1552
+ * change or be removed in future SDK or CLI releases.
1553
+ */
1554
+ export interface FactoryMeta {
1555
+ /** Stable factory name used for invocation. */
1556
+ name: string;
1557
+ /** Human-readable factory description. */
1558
+ description: string;
1559
+ /** Display metadata for the progress phases the factory may report. */
1560
+ phases: Array<{
1561
+ title: string;
1562
+ detail?: string;
1563
+ }>;
1564
+ /** Optional resource ceilings presented to the user before execution. */
1565
+ limits?: FactoryLimits;
1566
+ }
1477
1567
  /**
1478
1568
  * Provider-scoped options for the Copilot API (CAPI).
1479
1569
  *
@@ -1580,13 +1670,8 @@ export interface SessionConfigBase {
1580
1670
  */
1581
1671
  configDirectory?: string;
1582
1672
  /**
1583
- * When true, automatically discovers MCP server configurations (e.g. `.mcp.json`,
1584
- * `.vscode/mcp.json`) and skill directories from the working directory and merges
1585
- * them with any explicitly provided `mcpServers` and `skillDirectories`, with
1586
- * explicit values taking precedence on name collision.
1587
- *
1588
- * Note: custom instruction files (`.github/copilot-instructions.md`, `AGENTS.md`, etc.)
1589
- * are always loaded from the working directory regardless of this setting.
1673
+ * Enables runtime discovery of supported configuration. Explicitly supplied
1674
+ * configuration takes precedence over discovered values.
1590
1675
  *
1591
1676
  * @default false
1592
1677
  */
@@ -2157,7 +2242,7 @@ export interface ProviderConfig {
2157
2242
  */
2158
2243
  azure?: {
2159
2244
  /**
2160
- * API version. Defaults to "2024-10-21".
2245
+ * API version. When omitted, the runtime uses the GA versionless v1 route.
2161
2246
  */
2162
2247
  apiVersion?: string;
2163
2248
  };
package/dist/types.js CHANGED
@@ -1,4 +1,5 @@
1
1
  import { createSessionFsAdapter } from "./sessionFsProvider.js";
2
+ import { SessionFsSqliteTransactionFailure } from "./sessionFsProvider.js";
2
3
  import {
3
4
  CopilotRequestHandler,
4
5
  CopilotWebSocketHandler,
@@ -122,6 +123,7 @@ export {
122
123
  CopilotWebSocketHandler,
123
124
  RuntimeConnection,
124
125
  SYSTEM_MESSAGE_SECTIONS,
126
+ SessionFsSqliteTransactionFailure,
125
127
  approveAll,
126
128
  convertMcpCallToolResult,
127
129
  createSessionFsAdapter,
@@ -56,4 +56,5 @@ The `session` object provides methods for sending messages, logging to the timel
56
56
  ## Further Reading
57
57
 
58
58
  - `examples.md` — Practical code examples for tools, hooks, events, and complete extensions
59
+ - `factories.md`: Authoring, running, resuming, and observing Agent Factories
59
60
  - `agent-author.md` — Step-by-step workflow for agents authoring extensions programmatically
@@ -0,0 +1,240 @@
1
+ # Agent Factories
2
+
3
+ Agent Factories are extension-authored, session-scoped workflows that coordinate subagents and durable steps. The API is experimental.
4
+
5
+ ## Define and register a factory
6
+
7
+ Use `defineFactory` and pass the returned handle to `joinSession`:
8
+
9
+ ```js
10
+ import { defineFactory, joinSession } from "@github/copilot-sdk/extension";
11
+
12
+ const reviewChanged = defineFactory({
13
+ meta: {
14
+ name: "review-changed",
15
+ description:
16
+ "Review changed files and verify the findings. " +
17
+ "args: { files: string[] } — the paths to review.",
18
+ phases: [{ title: "Review" }, { title: "Verify" }],
19
+ limits: {
20
+ maxConcurrentSubagents: 3,
21
+ maxTotalSubagents: 10,
22
+ timeoutSeconds: 90.5,
23
+ maxAiCredits: 5,
24
+ },
25
+ },
26
+ run: async (ctx) => {
27
+ ctx.phase("Review");
28
+ const reviews = await ctx.parallel(
29
+ ctx.args.files.map(
30
+ (file) => () => ctx.agent(`Review ${file}`, { label: `Review ${file}` })
31
+ )
32
+ );
33
+
34
+ ctx.phase("Verify");
35
+ const report = await ctx.step("report", () => ({ reviews }));
36
+ ctx.log(`Completed factory run ${ctx.runId}`);
37
+ return report;
38
+ },
39
+ });
40
+
41
+ const session = await joinSession({ factories: [reviewChanged] });
42
+ ```
43
+
44
+ Factory metadata contains a stable `name`, a human-readable `description`, declared `phases`, and optional `limits`. Phase entries contain a `title` and optional `detail`.
45
+
46
+ There is no declared schema for `ctx.args`. The `run_factory` tool forwards `args` verbatim and its parameter is untyped, so **the `description` is the only thing telling an agent what arguments to supply** — state the expected shape there whenever a factory reads `ctx.args`, as the example above does. Arguments supplied by an extension calling `session.factory.run(...)` directly are typed through `defineFactory<TArgs>`, but that typing does not reach the model. A factory that reads `ctx.args` should validate it rather than assume a shape.
47
+
48
+ `defineFactory<TArgs, TResult>` accepts a `run(context)` function returning `Promise<TResult>`, where `TResult` is `JsonValue | void`. Objects, arrays, strings, numbers, booleans, and `null` are valid results. Returning `undefined` completes the factory with no result. Other non-JSON values are rejected.
49
+
50
+ ## Factory context
51
+
52
+ The `run()` context provides:
53
+
54
+ - `ctx.runId`: Stable ID reused across resumed attempts.
55
+ - `ctx.args`: Invocation arguments, forwarded verbatim. When the caller omits `args`, this is `{}` rather than `undefined`.
56
+ - `ctx.agent(prompt, options?)`: Runs one factory-owned subagent. Options are exactly `label`, `schema`, and `model`. See [Subagent calls](#subagent-calls).
57
+ - `ctx.parallel(thunks)`: Runs thunks concurrently and awaits all of them (a barrier). A thunk that throws becomes `null` in the result array, so one failed item does not lose the rest. Cancellation and hard runtime failures (`ResponseError`, `ConnectionError`) are the exception — those propagate and reject the whole call, because they mean the run itself is in trouble rather than one item having failed. Handle them at run level; do not assume every failure arrives as a `null`. Rejects above 4096 items.
58
+ - `ctx.pipeline(items, ...stages)`: Flows each item through every stage without a barrier between stages, so one item can be in a later stage while another is still in an earlier one. Each stage is called as `(previous, item, index)`, where `previous` is the prior stage's result and `item` is the original input. A stage that throws drops that item to `null` and skips its remaining stages, with the same exception for cancellation and hard runtime failures. Rejects above 4096 items.
59
+ - `ctx.phase(title)`: Starts a named progress phase. This sets a single run-global value, so calling it from inside concurrent `parallel`/`pipeline` stages races. Call it at run-level transitions and distinguish concurrent work by `label` instead.
60
+ - `ctx.log(message)`: Appends a progress line. When a factory bounds its own coverage (top-N, sampling), log what was dropped.
61
+ - `ctx.step(key, producer, options?)`: Journals the producer's JSON result under a stable key so a resume replays it without re-running the producer. A journaled (default) producer must return a JSON-serializable value; `undefined` or a non-JSON value is rejected. Pass `{ volatile: true }` to bypass the journal and run the producer every time.
62
+
63
+ The key is the *sole* identity: neither the producer body nor its inputs contribute to it. A resume replays the cached value for a matching key even if the producer has since changed, so version the key (`"scan-v2"`) whenever its inputs or meaning change. Journaled producers are best-effort at-least-once and may run again across crashes or concurrent same-key callers, so keep side effects idempotent.
64
+ - `ctx.session`: The full session returned by `joinSession`.
65
+ - `ctx.signal`: Cooperative cancellation signal for extension work and subprocesses.
66
+ - `ctx.factory(...)`: Always rejects because nested factories are not supported.
67
+
68
+ Factory-owned subagents are intentionally hidden from `read_agent` and `write_agent`. Use the factory observability APIs instead.
69
+
70
+ ### Subagent calls
71
+
72
+ `ctx.agent(prompt, options?)` spawns one factory-scoped subagent and awaits it. Without a schema it resolves to the subagent's final text. With `options.schema` it resolves to the parsed JSON value.
73
+
74
+ **Identical calls are memoized into one subagent.** Each call is journaled by its canonical prompt and options, including `label`. Two calls with the same prompt and the same options return one shared result — even when issued concurrently. To spawn N *independent* subagents, give each a unique `label` or vary the prompt:
75
+
76
+ ```js
77
+ // One subagent, awaited five times — almost certainly not what you want.
78
+ await ctx.parallel([1, 2, 3, 4, 5].map(() => () => ctx.agent("Find a bug")));
79
+
80
+ // Five independent subagents.
81
+ await ctx.parallel(
82
+ [1, 2, 3, 4, 5].map((i) => () => ctx.agent("Find a bug", { label: `finder:${i}` }))
83
+ );
84
+ ```
85
+
86
+ **An ordinary failure resolves to `null` — it does not throw.** A subagent that errors, returns nothing, or (with a schema) produces output that still fails to parse or match after its one retry resolves `null`. Always guard the result before using it, including a bare `await ctx.agent(...)`:
87
+
88
+ ```js
89
+ const finding = await ctx.agent(prompt, { label: "inspector" });
90
+ if (!finding) return { finding: null };
91
+ ```
92
+
93
+ Cancellation and hard runtime failures — a reached limit, a durable-state failure — reject instead, aborting the run. When filtering results, prefer `v => v !== null` over `Boolean`, which also discards a valid `false`, `0`, or `""`.
94
+
95
+ **`schema` is a structural subset of JSON Schema, not a validator.** Honored: `type`, `required`, `enum`, `const`, recursive `properties`/`items`, and `anyOf`/`oneOf`/`allOf` — where `oneOf` is treated as `anyOf`, meaning at least one branch matches rather than exactly one. Ignored and *not* enforced: `additionalProperties`, `pattern`, `minLength`/`maxLength`, `format`, numeric ranges, and boolean schemas. Do not rely on an ignored keyword to constrain a result. A schema call retries once on a parse or match failure, so it may spawn twice, and both spawns count toward `maxTotalSubagents`.
96
+
97
+ ### Choosing between pipeline and parallel
98
+
99
+ Prefer `pipeline` for multi-stage work. It has no barrier between stages, so each item advances as soon as its own prior stage finishes.
100
+
101
+ Reach for a barrier — `parallel` between stages — only when a stage genuinely needs every prior result at once: deduplicating or merging across the full set, an early exit based on the total, or a prompt that compares one result against the others. Needing to map, filter, or flatten is not a reason to use a barrier; do that inside a pipeline stage. Barrier latency is real: if the slowest of N subagents takes three times the fastest, a barrier wastes the rest of the pool's time.
102
+
103
+ See [factory-patterns.md](./factory-patterns.md) for composable orchestration patterns built on these primitives.
104
+
105
+ ## Resource limits
106
+
107
+ Limits may be declared in `meta.limits` and overridden per invocation. All limits must be positive when present.
108
+
109
+ - `maxConcurrentSubagents`: Positive integer concurrent-subagent cap. Additional subagents wait in a queue. Queueing applies backpressure and does not fail the run.
110
+ - `maxTotalSubagents`: Positive integer cumulative admission cap. An attempted subagent beyond the cap ends the attempt with failure kind `maxTotalSubagents`.
111
+ - `timeoutSeconds`: Positive finite number of seconds, including positive fractions, capped at `2_147_483.647`. It measures accumulated active-execution time across attempts, including the extension body, subprocess waits, queued-agent waits, and sleeps. Time between attempts is excluded. The timeout is soft because already-running work may take time to stop. Its failure kind is `timeoutSeconds`.
112
+ - `maxAiCredits`: Positive finite AI-credit budget for the whole run's factory subagent subtree, including descendants. AI credits are GitHub Copilot's universal usage metric. This is a soft, post-paid ceiling, so completed or parallel turns can settle above it before the run stops. Accounting is fail-closed: an accounting failure stops a budgeted run rather than allowing untracked use. Its failure kind is `maxAiCredits`.
113
+
114
+ `maxTotalSubagents`, `timeoutSeconds`, and `maxAiCredits` use reject-and-retry semantics. A rejected attempt ends with run status `error` and `failure.type` set to `factory_limit_reached`. The failed run keeps its ID, arguments, journal, and accounting. Resume the run with a raised limit when additional work is approved. Previously consumed resources still count.
115
+
116
+ ## Run and resume
117
+
118
+ Run by registered name or handle:
119
+
120
+ ```ts
121
+ const run = await session.factory.run("review-changed", {
122
+ args: { files: ["src/a.ts"] },
123
+ limits: { maxAiCredits: 3 },
124
+ });
125
+
126
+ if (run.status === "completed") {
127
+ console.log(run.result);
128
+ } else {
129
+ console.error(`run ${run.runId} ended as ${run.status}`, run.failure ?? run.error);
130
+ }
131
+ ```
132
+
133
+ The name overload is:
134
+
135
+ ```ts
136
+ session.factory.run(
137
+ name: string,
138
+ options?: { args?: JsonValue; limits?: FactoryLimits },
139
+ ): Promise<FactoryRunResult>;
140
+ ```
141
+
142
+ Resume by run ID without resending the name or arguments:
143
+
144
+ ```ts
145
+ const run = await session.factory.resume(runId, {
146
+ limits: { maxAiCredits: 6 },
147
+ });
148
+ ```
149
+
150
+ The signature is:
151
+
152
+ ```ts
153
+ session.factory.resume(
154
+ runId: string,
155
+ options?: { limits?: FactoryLimits },
156
+ ): Promise<FactoryRunResult>;
157
+ ```
158
+
159
+ Both resolve with the run envelope (`FactoryRunResult`) for **every** outcome — `completed`, `error`, `halted`, and `cancelled` alike. Inspect `status` and read `result` only when the run completed; a limit breach carries a typed `failure`. A declined fresh run is not a pre-execution failure: the run row already exists by the time the prompt is answered, so it resolves with a terminal `cancelled` envelope carrying the run ID. Only failures that occur *before* a run exists reject: an unknown factory name or an already-active session. Pre-execution resume failures, including a declined reapproval, throw `FactoryResumeError`, whose `code` is one of `not_found`, `non_resumable`, `already_active`, `reapproval_declined`, or `no_approval_provider`.
160
+
161
+ An agent that no longer has a prior run's ID in context can recover it with `factories_manage` and `operation: "runs"`, which lists the session's factory runs with their IDs and statuses. This matters for resume: a run that reached a limit keeps its journal, so resuming it replays completed work for free, while restarting it from scratch pays for that work twice.
162
+
163
+ The agent-facing `run_factory` tool has exactly two input branches:
164
+
165
+ ```ts
166
+ { name: string; args?: JsonValue; limits?: FactoryLimits }
167
+ { resumeFromRunId: string; limits?: FactoryLimits }
168
+ ```
169
+
170
+ ## Authoring a factory from inside a session
171
+
172
+ The agent-facing `factories_manage` tool writes a factory into a session-scoped extension at runtime with `operation: "author"`. The rules above all apply, plus one constraint that does not affect an extension author.
173
+
174
+ **The `run` body is self-contained.** It is emitted verbatim into a generated module as a single async function expression. It closes over nothing: not the conversation that authored it, and not any authoring-time binding. Only its own locals, its `ctx` parameter, and standard Node and JavaScript globals are in scope, so every schema, constant, and helper must be defined *inside* the function. The generated module imports the SDK itself; the expression cannot add static `import` statements or use `require`. Load anything else with a dynamic `await import("...")` in the body.
175
+
176
+ ```js
177
+ async ({ args, agent, phase }) => {
178
+ // Defined inside — there is no outer scope to close over.
179
+ const VERDICT = { type: "object", properties: { real: { type: "boolean" } }, required: ["real"] };
180
+
181
+ phase("Inspect");
182
+ const finding = await agent(`Name one likely bug in ${args.file ?? "the code"}.`, {
183
+ label: "inspector",
184
+ });
185
+ if (!finding) return { finding: null, real: false };
186
+
187
+ phase("Verify");
188
+ const verdict = await agent(`Is this a real bug? Claim: ${finding}`, {
189
+ label: "verifier",
190
+ schema: VERDICT,
191
+ });
192
+ return { finding, real: verdict?.real === true };
193
+ };
194
+ ```
195
+
196
+ Authoring registers the factory but does not run it. Invoke it afterwards with `run_factory`. Use `factories_manage` with `operation: "list"` to see the factories already registered in the session and `operation: "inspect"` to read one factory's description, phases, and limits before running it.
197
+
198
+ ## Observe a run
199
+
200
+ The calling session can inspect its own factory runs:
201
+
202
+ ```ts
203
+ const runs = await session.factory.listRuns();
204
+ const detail = await session.factory.getRunDetail(runId);
205
+ const page = await session.factory.getRunProgress(runId, {
206
+ phaseId,
207
+ afterSeq,
208
+ beforeSeq,
209
+ limit,
210
+ });
211
+ ```
212
+
213
+ - `listRuns()` returns summaries in durable creation order.
214
+ - `getRunDetail(runId)` returns phases, prompt-safe agent summaries, and the latest progress page.
215
+ - `getRunProgress(runId, options?)` pages progress forward, backward, by phase, or from the latest tail.
216
+
217
+ `getRun(runId)` reads the latest run envelope, and `cancel(runId)` cancels a run and returns its terminal envelope.
218
+
219
+ `waitForRun(runId, options?)` resolves with the terminal envelope once the run settles into `completed`, `error`, `halted`, or `cancelled`, and resolves immediately when it has already settled:
220
+
221
+ ```ts
222
+ const settled = await session.factory.waitForRun(runId);
223
+ if (settled.status === "completed") {
224
+ console.log(settled.result);
225
+ }
226
+ ```
227
+
228
+ It watches `factory.run_updated` and re-reads the durable envelope on each invalidation, collapsing a burst of events into a single in-flight read. A low-frequency periodic re-read runs alongside the subscription, so a dropped or missing invalidation degrades into a slightly late resolution rather than an unbounded wait. Pass a `signal` to stop waiting:
229
+
230
+ ```ts
231
+ const controller = new AbortController();
232
+ setTimeout(() => controller.abort(), 30_000);
233
+ const settled = await session.factory.waitForRun(runId, { signal: controller.signal });
234
+ ```
235
+
236
+ Aborting rejects the wait and has no effect on the run, which keeps executing — use `cancel(runId)` to actually stop it. Because a terminal envelope is final, the resolved value never changes afterwards. `isFactoryRunTerminal(status)` exposes the same terminal-status test for callers driving their own loop.
237
+
238
+ Listen for the ephemeral `factory.run_updated` event. Its `{ runId, revision }` payload is an invalidation signal. Re-read the desired API when a newer monotonic revision arrives.
239
+
240
+ Revisions cover durable lifecycle, accounting, phase, agent, and progress changes. Continuous read-time fields can change without a new revision. These include `observedAt`, active-time calculations, live counts, and a live agent's status or prompt-safe activity text. Factory prompts are never exposed by these APIs. A run is visible only through the session that owns it.