@tangle-network/agent-runtime 0.128.0 → 0.131.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/README.md +70 -20
  2. package/dist/{activation-DhWJ3p8N.js → activation-BCzMOTaV.js} +3 -3
  3. package/dist/{activation-DhWJ3p8N.js.map → activation-BCzMOTaV.js.map} +1 -1
  4. package/dist/agent.d.ts +2 -3
  5. package/dist/agent.js +4 -5
  6. package/dist/agent.js.map +1 -1
  7. package/dist/{analyst-loop-DvSciOfB.js → analyst-loop-BE8cDs5Q.js} +2 -2
  8. package/dist/{analyst-loop-DvSciOfB.js.map → analyst-loop-BE8cDs5Q.js.map} +1 -1
  9. package/dist/analyst-loop.d.ts +1 -1
  10. package/dist/analyst-loop.js +1 -1
  11. package/dist/authoring-Dv3t6SXe.js +163 -0
  12. package/dist/authoring-Dv3t6SXe.js.map +1 -0
  13. package/dist/candidate-execution/index.js +4 -4
  14. package/dist/{candidate-execution-BFpq-Xi6.js → candidate-execution-BNxKr-Bu.js} +5 -5
  15. package/dist/{candidate-execution-BFpq-Xi6.js.map → candidate-execution-BNxKr-Bu.js.map} +1 -1
  16. package/dist/{conversation-BpLQZGPH.js → conversation-DNtxaJ1Z.js} +113 -31
  17. package/dist/conversation-DNtxaJ1Z.js.map +1 -0
  18. package/dist/conversation.d.ts +2 -2
  19. package/dist/conversation.js +2 -2
  20. package/dist/environment-provider-CxvSd1W6.d.ts +86 -0
  21. package/dist/{environment-provider-DqFS6FSZ.js → environment-provider-Dyg8DtLK.js} +8 -4
  22. package/dist/environment-provider-Dyg8DtLK.js.map +1 -0
  23. package/dist/environment-provider.d.ts +1 -1
  24. package/dist/environment-provider.js +1 -1
  25. package/dist/graph-xWdv53Le.js +471 -0
  26. package/dist/graph-xWdv53Le.js.map +1 -0
  27. package/dist/{improvement-cycle-IJgCbKWQ.js → improvement-cycle-Csp38cWg.js} +133 -224
  28. package/dist/improvement-cycle-Csp38cWg.js.map +1 -0
  29. package/dist/{index-BhZhQw77.d.ts → index-CoO7atyo.d.ts} +556 -1278
  30. package/dist/{index-BhuzfG2r.d.ts → index-DwGtu9nc.d.ts} +7 -9
  31. package/dist/{index-Efjb3nrQ.d.ts → index-qYHpsmG2.d.ts} +36 -18
  32. package/dist/index.d.ts +353 -11
  33. package/dist/index.js +111 -354
  34. package/dist/index.js.map +1 -1
  35. package/dist/intelligence.d.ts +6 -6
  36. package/dist/intelligence.js +9 -8
  37. package/dist/intelligence.js.map +1 -1
  38. package/dist/kernel.d.ts +7 -5
  39. package/dist/kernel.js +13 -9
  40. package/dist/{knowledge-DF63xPr4.js → knowledge-DPEu4f-0.js} +19 -17
  41. package/dist/knowledge-DPEu4f-0.js.map +1 -0
  42. package/dist/knowledge.d.ts +1 -1
  43. package/dist/knowledge.js +1 -1
  44. package/dist/{loop-runner-bin-Ckp_9tmD.d.ts → loop-runner-bin-BwgZ8m_7.d.ts} +6 -3
  45. package/dist/{loop-runner-bin-CWqOpCEw.js → loop-runner-bin-dg6li2-b.js} +5 -27
  46. package/dist/loop-runner-bin-dg6li2-b.js.map +1 -0
  47. package/dist/loop-runner-bin.d.ts +1 -1
  48. package/dist/loop-runner-bin.js +1 -1
  49. package/dist/materialization-COJ1UYQ-.js +272 -0
  50. package/dist/materialization-COJ1UYQ-.js.map +1 -0
  51. package/dist/mcp/bin.js +39 -47
  52. package/dist/mcp/bin.js.map +1 -1
  53. package/dist/mcp/index.d.ts +24 -26
  54. package/dist/mcp/index.js +67 -84
  55. package/dist/mcp/index.js.map +1 -1
  56. package/dist/mcp/memory-bin.js +1 -1
  57. package/dist/{memory-server-DL6cE2Ag.js → memory-server-eD2baiRO.js} +3 -3
  58. package/dist/{memory-server-DL6cE2Ag.js.map → memory-server-eD2baiRO.js.map} +1 -1
  59. package/dist/model-policy-CqziaqS1.js +232 -0
  60. package/dist/model-policy-CqziaqS1.js.map +1 -0
  61. package/dist/{openai-tools-B68JaOCx.d.ts → openai-tools-D3oMyVGl.d.ts} +2 -2
  62. package/dist/{openai-tools-D3XfrrQ6.js → openai-tools-zRphjXS4.js} +2 -2
  63. package/dist/openai-tools-zRphjXS4.js.map +1 -0
  64. package/dist/{prepare--8EvLqCr.js → prepare-DYWjVcPx.js} +169 -153
  65. package/dist/prepare-DYWjVcPx.js.map +1 -0
  66. package/dist/primeintellect/index.d.ts +7 -6
  67. package/dist/primeintellect/index.js +9 -11
  68. package/dist/primeintellect/index.js.map +1 -1
  69. package/dist/profiles.d.ts +21 -174
  70. package/dist/profiles.js +67 -276
  71. package/dist/profiles.js.map +1 -1
  72. package/dist/{protected-model-port-CXVfOUu_.js → protected-model-port-48ALLGxT.js} +2 -2
  73. package/dist/{protected-model-port-CXVfOUu_.js.map → protected-model-port-48ALLGxT.js.map} +1 -1
  74. package/dist/{redact-PQzmE1Jn.d.ts → redact-9_Gf-8m3.d.ts} +58 -69
  75. package/dist/{researcher-CoVqNhfI.js → researcher-Skz5-Uc8.js} +50 -26
  76. package/dist/researcher-Skz5-Uc8.js.map +1 -0
  77. package/dist/{run-layout-C2FGmZ3v.js → run-layout-CeJsEAom.js} +2 -2
  78. package/dist/{run-layout-C2FGmZ3v.js.map → run-layout-CeJsEAom.js.map} +1 -1
  79. package/dist/runtime-D-QfLbSd.d.ts +893 -0
  80. package/dist/{runtime-5uDVVfER.js → runtime-cOzDOOHr.js} +315 -1191
  81. package/dist/runtime-cOzDOOHr.js.map +1 -0
  82. package/dist/{sandbox-events-Yhd1GYWl.js → sandbox-events-CRDwc5WN.js} +42 -5
  83. package/dist/sandbox-events-CRDwc5WN.js.map +1 -0
  84. package/dist/snapshot-CXiiuHhL.js +21 -0
  85. package/dist/snapshot-CXiiuHhL.js.map +1 -0
  86. package/dist/{spawn-journal-DsZKDqeh.js → spawn-journal-saHQzqYi.js} +15 -21
  87. package/dist/spawn-journal-saHQzqYi.js.map +1 -0
  88. package/dist/stream-agent-turn-C852AgMT.d.ts +128 -0
  89. package/dist/stream-agent-turn-rYgaOLO0.js +910 -0
  90. package/dist/stream-agent-turn-rYgaOLO0.js.map +1 -0
  91. package/dist/{structural-rollout-zY0oqQzO.js → structural-rollout-3uxVGcg2.js} +459 -235
  92. package/dist/structural-rollout-3uxVGcg2.js.map +1 -0
  93. package/dist/{supervise-CsTKbH9R.js → supervise-DHYX8gO2.js} +867 -4788
  94. package/dist/supervise-DHYX8gO2.js.map +1 -0
  95. package/dist/{supervisor-DpjO0Gmy.js → supervisor-CV6Jh28D.js} +7439 -2897
  96. package/dist/supervisor-CV6Jh28D.js.map +1 -0
  97. package/dist/testing.d.ts +3 -1
  98. package/dist/testing.js +271 -221
  99. package/dist/testing.js.map +1 -1
  100. package/dist/{tool-server-RcWgLIsL.js → tool-server-Gs3VvfSK.js} +22 -9
  101. package/dist/tool-server-Gs3VvfSK.js.map +1 -0
  102. package/dist/{top-app-5unxqovu.js → top-app-G4M18b6i.js} +3 -3
  103. package/dist/{top-app-5unxqovu.js.map → top-app-G4M18b6i.js.map} +1 -1
  104. package/dist/tui/bin.js +1 -1
  105. package/dist/tui/index.js +1 -1
  106. package/dist/{environment-provider-CUFsyymu.d.ts → types-C6Q-J0Dt.d.ts} +51 -114
  107. package/dist/{types-DnNGJ5Gz.d.ts → types-ebIY0dMG.d.ts} +556 -30
  108. package/dist/{util-MVgdwuIS.js → util-Bw6srryQ.js} +3 -2
  109. package/dist/{util-MVgdwuIS.js.map → util-Bw6srryQ.js.map} +1 -1
  110. package/dist/{workspace-archive-BQxvkypI.js → workspace-archive-aOfJ47ms.js} +3 -3
  111. package/dist/{workspace-archive-BQxvkypI.js.map → workspace-archive-aOfJ47ms.js.map} +1 -1
  112. package/package.json +13 -15
  113. package/skills/agent-graphs/IMPROVE.md +3 -3
  114. package/skills/agent-graphs/SKILL.md +4 -5
  115. package/skills/agent-graphs/cases/review-pipeline.json +1 -2
  116. package/skills/agent-graphs/cases/unmeasured-harness.json +2 -4
  117. package/dist/backends-CiOCyRHb.js +0 -743
  118. package/dist/backends-CiOCyRHb.js.map +0 -1
  119. package/dist/conversation-BpLQZGPH.js.map +0 -1
  120. package/dist/environment-provider-DqFS6FSZ.js.map +0 -1
  121. package/dist/improvement-cycle-IJgCbKWQ.js.map +0 -1
  122. package/dist/index-DLM0W1h1.d.ts +0 -545
  123. package/dist/knowledge-DF63xPr4.js.map +0 -1
  124. package/dist/local-harness-BIajef4A.d.ts +0 -465
  125. package/dist/loop-runner-bin-CWqOpCEw.js.map +0 -1
  126. package/dist/model-resolution-Btd9iIKV.js +0 -98
  127. package/dist/model-resolution-Btd9iIKV.js.map +0 -1
  128. package/dist/openai-tools-D3XfrrQ6.js.map +0 -1
  129. package/dist/prepare--8EvLqCr.js.map +0 -1
  130. package/dist/researcher-CoVqNhfI.js.map +0 -1
  131. package/dist/runtime-5uDVVfER.js.map +0 -1
  132. package/dist/sandbox-events-Yhd1GYWl.js.map +0 -1
  133. package/dist/spawn-journal-DsZKDqeh.js.map +0 -1
  134. package/dist/structural-rollout-zY0oqQzO.js.map +0 -1
  135. package/dist/supervise-CsTKbH9R.js.map +0 -1
  136. package/dist/supervisor-DpjO0Gmy.js.map +0 -1
  137. package/dist/tool-server-RcWgLIsL.js.map +0 -1
  138. package/dist/types-C9j4qg6l.d.ts +0 -500
  139. package/skills/agent-graphs/cases/floor-trap-pi.json +0 -11
@@ -0,0 +1,893 @@
1
+ import { E as SandboxClient, i as ExecCtx } from "./types-ebIY0dMG.js";
2
+ import { B as Spend, d as ExecutorFactory, m as ExecutorRegistry, q as UsageEvent, rn as TraceSource, un as ExecutorProgress } from "./types-C6Q-J0Dt.js";
3
+ import { i as AgentEnvironmentProvider, o as AgentEnvironmentProviderRegistry, w as ProviderExecutorOptions } from "./environment-provider-CxvSd1W6.js";
4
+ import { AgentProfile, HarnessType, ReasoningEffort } from "@tangle-network/agent-interface";
5
+ import { WorkspacePlanReceipt } from "@tangle-network/agent-profile-materialize";
6
+ import { ChildProcess } from "node:child_process";
7
+ import { BackendType } from "@tangle-network/sandbox";
8
+ //#region src/runtime/router-retry-policy.d.ts
9
+ /** Exact retry controls accepted at `AgentProfile.model.metadata.retry`. */
10
+ interface RouterRetryPolicy {
11
+ /** Total attempts, including the first request. */
12
+ readonly maxAttempts?: number;
13
+ /** Delay before the second attempt. Later delays grow exponentially. */
14
+ readonly initialBackoffMs?: number;
15
+ /** Maximum delay between attempts. */
16
+ readonly maxBackoffMs?: number;
17
+ /** Symmetric random variation around each delay, from 0 through 1. */
18
+ readonly jitter?: number;
19
+ /** HTTP statuses that may be retried. */
20
+ readonly retryStatuses?: ReadonlyArray<number>;
21
+ /** Deadline for receiving one attempt's response headers. Zero disables it. */
22
+ readonly requestTimeoutMs?: number;
23
+ }
24
+ //#endregion
25
+ //#region src/runtime/tool-loop.d.ts
26
+ /** Provider-neutral conversation record accepted by a tool-loop brain. */
27
+ type ToolLoopMessageRecord = Record<string, unknown>;
28
+ /** One provider-neutral tool request emitted by a tool-loop model. */
29
+ interface ToolLoopToolCall {
30
+ id: string;
31
+ name: string;
32
+ /** Raw JSON arguments emitted by the model. */
33
+ arguments: string;
34
+ }
35
+ /** Runtime-owned identity and cancellation for one logical inference call. The wrapper is frozen
36
+ * before dispatch; a transport may observe the signal but cannot replace the authority it names. */
37
+ interface ToolLoopCallContext {
38
+ readonly signal: AbortSignal;
39
+ readonly callId: string;
40
+ readonly correlationId: string;
41
+ }
42
+ /** One inference turn over the running conversation + the tool specs → the model's text, any
43
+ * tool calls, and token usage. The seam every brain satisfies. */
44
+ type ToolLoopChat = (messages: ReadonlyArray<ToolLoopMessageRecord>, tools: ReadonlyArray<ToolSpec>, context?: ToolLoopCallContext) => Promise<{
45
+ content?: string | null;
46
+ toolCalls: ToolLoopToolCall[];
47
+ usage?: {
48
+ input: number;
49
+ output: number;
50
+ reasoning?: number;
51
+ };
52
+ /** Dollar value reported for the turn. It is not billed spend unless provenance says so. */
53
+ costUsd?: number;
54
+ costProvenance?: 'provider-receipt' | 'billing-receipt' | 'catalog-estimate';
55
+ /** The turn ran but its usage was not reported when the transport EXPECTED one (the streamed
56
+ * router transport asks for usage and this says it never arrived). A metering caller records an
57
+ * unknown turn on it; `runBrainLoop` itself ignores it. */
58
+ usageUnknown?: true;
59
+ /** Provider-observed model identity. Profile-bound callers validate it before accepting output. */
60
+ model?: string;
61
+ /** Provider-reported prompt-cache evidence; missing fields remain missing. */
62
+ promptCache?: Readonly<Record<string, number | string>>;
63
+ /** Physical HTTP/injected-transport attempts spent by this one logical call. */
64
+ transportAttempts?: number;
65
+ }>;
66
+ /** Self-compaction — bound the loop's OWN context window the way a fresh-respawn (dumb-Ralph) loop
67
+ * does, but in place. A stateless chat API re-sends the WHOLE running conversation every turn, so an
68
+ * agent that accumulates dozens of turns of tool results re-bills its entire transcript on every
69
+ * inference — the context-overflow-one-level-up that the conserved budget pool cannot fix. With
70
+ * compaction set, once the conversation exceeds `thresholdTokens` the accumulated middle (every prior
71
+ * assistant turn + tool result) is distilled into ONE compact progress note and the conversation is
72
+ * reset to `[...head, digest]`: the preserved head (system + the original task) survives, the stale
73
+ * turn-by-turn history does not. The model keeps deciding; it stops re-billing the whole transcript.
74
+ * Fires at a CLEAN turn boundary (after a turn's tool results are folded in, before the next
75
+ * inference) so it never orphans an assistant `tool_calls` from its `tool` replies. */
76
+ interface ToolLoopCompaction {
77
+ /** Compact once the estimated token count of the conversation exceeds this. */
78
+ readonly thresholdTokens: number;
79
+ /** Distill the conversation into a compact progress note that REPLACES the middle. Receives the
80
+ * full conversation (so it can summarize everything done so far); returns the digest string. */
81
+ readonly distill: (messages: ReadonlyArray<ToolLoopMessageRecord>) => Promise<string> | string;
82
+ /** Leading messages preserved verbatim (system + the original task). Default 2. */
83
+ readonly preserveHead?: number;
84
+ /** Token estimator over the conversation. Default ≈ chars/4 (incl. tool-call arguments). */
85
+ readonly estimateTokens?: (messages: ReadonlyArray<ToolLoopMessageRecord>) => number;
86
+ /** Notified each time a compaction fires — for observability/metering. */
87
+ readonly onCompact?: (info: {
88
+ turn: number;
89
+ beforeTokens: number;
90
+ afterTokens: number;
91
+ }) => void;
92
+ }
93
+ /** Public supervisor-facing compaction config: same knobs as the primitive, but `distill` is optional
94
+ * because the supervisor has a default digest that combines a brain note with live worker state. */
95
+ type ToolLoopCompactionOptions = Omit<ToolLoopCompaction, 'distill'> & {
96
+ readonly distill?: ToolLoopCompaction['distill'];
97
+ };
98
+ //#endregion
99
+ //#region src/runtime/router-client.d.ts
100
+ /**
101
+ * Connection details for Runtime's Router-backed executors.
102
+ *
103
+ * This is deliberately transport-only: model, prompt, tools, generation settings, and retry
104
+ * policy belong to the exact executable `AgentProfile` consumed by `streamAgentTurn`.
105
+ */
106
+ interface RouterTransportConfig {
107
+ routerBaseUrl: string;
108
+ routerKey: string;
109
+ /** Injectable OpenAI-compatible transport for offline execution. */
110
+ complete?: (body: Record<string, unknown>, request?: {
111
+ readonly headers: Readonly<Record<string, string>>;
112
+ readonly signal?: AbortSignal;
113
+ }) => Promise<unknown>;
114
+ }
115
+ /**
116
+ * Private request configuration used by Runtime's Router adapter.
117
+ *
118
+ * Do not export this through a package entry point. Public callers execute a concrete
119
+ * `AgentProfile` through `createExecutor` + `streamAgentTurn`; only Runtime may lower that profile
120
+ * into these provider request fields.
121
+ */
122
+ interface RouterConfig extends RouterTransportConfig {
123
+ model: string;
124
+ /** Exact retry controls lowered from `AgentProfile.model.metadata.retry`. */
125
+ retry?: RouterRetryPolicy;
126
+ /**
127
+ * Optional ceiling for one completion, forwarded as `max_tokens`.
128
+ *
129
+ * A REASONING model spends this budget on hidden thinking BEFORE it emits a visible token, so
130
+ * the default can truncate one mid-thought and return no content at all — observed live with a
131
+ * model that spent 8,188 of the 8,192 on reasoning and answered with nothing. Raise it for a
132
+ * thinking model; the ceiling belongs to the router and model a caller chose, which is why it
133
+ * lives here rather than on one call site.
134
+ */
135
+ maxTokens?: number;
136
+ /**
137
+ * Take the tool-calling completion over SSE instead of one buffered POST. Off by default —
138
+ * `routerChatWithTools` never streams, and every existing caller keeps the buffered transport
139
+ * byte for byte.
140
+ *
141
+ * Why it exists: a buffered POST holds one connection idle for the WHOLE completion, and a
142
+ * supervisor turn is the longest completion in the system. An intermediary gateway with an
143
+ * idle-read timeout kills that connection mid-completion (the 524/503 family). A streamed
144
+ * response puts bytes on the wire from the first generated token on, so the connection is only
145
+ * idle through prefill. It does NOT shorten prefill, so a gateway whose deadline is
146
+ * time-to-FIRST-byte is unaffected; only an idle-timeout gateway is.
147
+ *
148
+ * Mutually exclusive with `complete`: the injected transport returns one parsed JSON body and has
149
+ * no stream to read, so setting both throws rather than silently taking the buffered path.
150
+ *
151
+ * WHICH PATHS CAN OPT IN. This flag is read in exactly one place (the private `chatWithTools` transport switch), so
152
+ * every entry point that takes a caller-supplied `RouterConfig` honors it: `routerBrain`,
153
+ * `routerToolLoop`, and `supervisorAgent` (which spreads `deps.router` into the brain's config —
154
+ * the supervisor turn this exists for). Two production call sites build a `RouterConfig` literal
155
+ * from their own options and therefore CANNOT express it today: the bench strategy's
156
+ * `routerToolLoop` config in `strategy.ts` and the local sandbox client's `routerBrain` config in
157
+ * `local-sandbox-client.ts`. Neither drives a supervisor-length turn; setting `stream` on a
158
+ * config handed to either has no path to reach them, and they stay buffered.
159
+ */
160
+ stream?: boolean;
161
+ }
162
+ interface ToolSpec {
163
+ type: 'function';
164
+ function: {
165
+ name: string;
166
+ description?: string;
167
+ parameters: unknown;
168
+ };
169
+ }
170
+ //#endregion
171
+ //#region src/mcp/worktree.d.ts
172
+ /**
173
+ *
174
+ * Git worktree helpers for the in-process delegation executor. Each
175
+ * delegation runs in its own worktree so multiple parallel harness
176
+ * subprocesses (claude / codex / opencode in a 3-way fanout) don't clobber
177
+ * each other's edits on the shared workspace.
178
+ *
179
+ * Worktrees live under `<repoRoot>/.agent-worktrees/<runId>/`. After the
180
+ * harness exits + the diff is captured, the worktree is removed.
181
+ *
182
+ * All operations spawn `git` via `child_process.spawn` synchronously
183
+ * (via a `runGit` helper). Stays narrow on purpose: no commits, no rebases.
184
+ * Diff capture stages all changes (`git add -A`) into the ephemeral worktree's
185
+ * index so created (untracked) files appear in the `--cached` diff.
186
+ *
187
+ * @experimental
188
+ */
189
+ /** @experimental */
190
+ interface WorktreeHandle {
191
+ /** Absolute path to the worktree directory. */
192
+ path: string;
193
+ /** SHA the worktree was created at. */
194
+ baseSha: string;
195
+ /** Branch name created for this worktree (typically `delegate/<runId>`). */
196
+ branch: string;
197
+ }
198
+ /** @experimental */
199
+ interface CreateWorktreeOptions {
200
+ /** Absolute path to the main git checkout. */
201
+ repoRoot: string;
202
+ /** Unique id for the worktree path + branch. Use the delegation run id. */
203
+ runId: string;
204
+ /** Parent directory the worktree lives under. Defaults to `.agent-worktrees`. */
205
+ variantsDir?: string;
206
+ /** Override the base ref (default `HEAD`). */
207
+ baseRef?: string;
208
+ /** Test seam — inject a custom git runner. */
209
+ runGit?: GitRunner;
210
+ }
211
+ /** @experimental */
212
+ interface DiffOptions {
213
+ /** Worktree to diff. */
214
+ worktree: WorktreeHandle;
215
+ /** What to compare against. Default `worktree.baseSha`. */
216
+ baseRef?: string;
217
+ /**
218
+ * Repository-relative input paths to omit from the captured worker patch.
219
+ * Paths are passed to Git with literal exclusion magic, so profile-provided
220
+ * `*`, `?`, `[` and `:` characters can never expand into broader pathspecs.
221
+ */
222
+ excludePaths?: ReadonlyArray<string>;
223
+ /** Test seam. */
224
+ runGit?: GitRunner;
225
+ }
226
+ /** @experimental */
227
+ interface DiffResult {
228
+ patch: string;
229
+ stats: {
230
+ filesChanged: number;
231
+ insertions: number;
232
+ deletions: number;
233
+ };
234
+ }
235
+ /** @experimental */
236
+ interface RemoveWorktreeOptions {
237
+ worktree: WorktreeHandle;
238
+ repoRoot: string;
239
+ /** Force removal even if dirty (default true; the loser of a fanout has uncommitted changes). */
240
+ force?: boolean;
241
+ /** Test seam. */
242
+ runGit?: GitRunner;
243
+ }
244
+ /** Pluggable git runner (sync) — replaceable in tests. */
245
+ type GitRunner = (args: ReadonlyArray<string>, opts: {
246
+ cwd: string;
247
+ }) => {
248
+ stdout: string;
249
+ stderr: string;
250
+ exitCode: number;
251
+ };
252
+ /** Checkout a fresh git worktree for a delegation run on a new branch under `variantsDir`. @experimental */
253
+ declare function createWorktree(options: CreateWorktreeOptions): Promise<WorktreeHandle>;
254
+ /** Stage worker changes and return the diff + shortstat, excluding declared input paths. @experimental */
255
+ declare function captureWorktreeDiff(options: DiffOptions): Promise<DiffResult>;
256
+ /**
257
+ * Remove a git worktree and delete its branch. Already-removed paths are harmless; every other
258
+ * Git failure rejects so callers cannot report a worktree as destroyed when cleanup failed.
259
+ * @experimental
260
+ */
261
+ declare function removeWorktree(options: RemoveWorktreeOptions): Promise<void>;
262
+ //#endregion
263
+ //#region src/mcp/local-harness.d.ts
264
+ /**
265
+ * Local coding harness available inside the sandbox — a narrowing of the shared `HarnessType`
266
+ * vocabulary, NOT a private spelling of it. The harness id is `claude-code`; `claude` is the
267
+ * EXECUTABLE name and lives only in the `command` field below. Keeping one vocabulary is what
268
+ * lets a `LocalHarness` be handed straight to the profile materializer and the capability table
269
+ * with no translation step.
270
+ */
271
+ type LocalHarness = Extract<HarnessType, 'claude-code' | 'codex' | 'opencode'>;
272
+ /** Every local harness, in table order — the one list `AGENT_RUNTIME_LOCAL_HARNESSES` and any
273
+ * other harness enumeration reads, so adding a row above is the only edit a new harness needs. */
274
+ declare const LOCAL_HARNESSES: ReadonlyArray<LocalHarness>;
275
+ /** The harness a caller gets when it expresses no preference. A composition-root default, not a
276
+ * capability claim: one constant so the several entry points cannot drift apart. */
277
+ declare const DEFAULT_LOCAL_HARNESS: LocalHarness;
278
+ /** The CLI binary a harness id runs. The two are NOT the same string (`claude-code` runs `claude`),
279
+ * so anything spawning a harness — a version probe, a login check — reads it from here rather than
280
+ * passing the harness id as a command. */
281
+ declare function localHarnessExecutable(harness: LocalHarness): string;
282
+ /**
283
+ * Whether the harness's native control can express this reasoning effort. Admission checks read
284
+ * this so a profile the invocation would later refuse is rejected BEFORE any workspace state is
285
+ * created, against the same table that emits the argv.
286
+ */
287
+ declare function harnessSupportsReasoningEffort(harness: LocalHarness, reasoningEffort: ReasoningEffort): boolean;
288
+ /** @experimental */
289
+ interface RunLocalHarnessOptions {
290
+ harness: LocalHarness;
291
+ /** Working directory for the subprocess (typically a worktree path). */
292
+ cwd: string;
293
+ /** Prompt forwarded as the harness CLI's task argument. */
294
+ taskPrompt: string;
295
+ /**
296
+ * Pre-built command + args (e.g. from `harnessInvocation` so the full authored
297
+ * `AgentProfile` — systemPrompt + model — reaches the harness). When set it OVERRIDES the
298
+ * default prompt-only `buildArgs(taskPrompt)` path; `command` defaults to the harness's
299
+ * default binary when only `args` is supplied. When absent the legacy prompt-only shape
300
+ * is used unchanged.
301
+ */
302
+ invocation?: {
303
+ command?: string;
304
+ args: ReadonlyArray<string>;
305
+ };
306
+ /** Allow autonomous edits without an interactive approval gate, using whichever bypass argv the
307
+ * harness declares. Use only when `cwd` is an isolated candidate worktree. */
308
+ dangerouslySkipPermissions?: boolean;
309
+ /** Isolate Codex from ambient configuration/instructions and require JSONL token usage.
310
+ * The invocation should come from `harnessInvocation(..., { codexReproducible: true })`. */
311
+ codexReproducible?: boolean;
312
+ /** Absolute host paths that reproducible Codex must not read. The normalized set is compiled
313
+ * into the controlled permission profile and its digest is returned in execution evidence. */
314
+ codexReadDeniedPaths?: ReadonlyArray<string>;
315
+ /** Wall-clock kill deadline (ms). Default 5 min. Subprocess SIGTERMed on expiry. */
316
+ timeoutMs?: number;
317
+ /** Newest stdout/stderr bytes retained per stream. Default 64 MiB. */
318
+ maxOutputBytes?: number;
319
+ /** Caller cancellation. SIGTERM is sent on abort. */
320
+ signal?: AbortSignal;
321
+ /** Override env (defaults to inheriting from the parent). */
322
+ env?: NodeJS.ProcessEnv;
323
+ /**
324
+ * Test seam — inject a custom spawner so unit tests can mock the
325
+ * subprocess without touching the OS. Defaults to node's `child_process.spawn`.
326
+ */
327
+ spawn?: (command: string, args: ReadonlyArray<string>, opts: {
328
+ cwd: string;
329
+ env: NodeJS.ProcessEnv;
330
+ stdio: 'pipe';
331
+ detached: boolean;
332
+ }) => ChildProcess;
333
+ /** Test seam for locating the native Codex executable before it is staged in the worktree. */
334
+ resolveCodexExecutable?: (command: string, env: NodeJS.ProcessEnv) => Promise<string>;
335
+ }
336
+ /** Exact aggregate usage emitted by Codex's terminal `turn.completed` JSONL event. */
337
+ interface CodexTokenUsage {
338
+ inputTokens: number;
339
+ cachedInputTokens: number;
340
+ outputTokens: number;
341
+ reasoningOutputTokens: number;
342
+ }
343
+ /** Isolation settings asserted before a reproducible Codex run is allowed to start. */
344
+ interface CodexExecutionPolicy {
345
+ sessionPersistence: 'ephemeral';
346
+ userConfig: false;
347
+ rules: false;
348
+ projectInstructions: false;
349
+ skillInstructions: false;
350
+ appInstructions: false;
351
+ toolSuggestions: false;
352
+ multiAgentInstructions: false;
353
+ sandbox: 'workspace-write';
354
+ permissionProfile: 'agent_runtime_reproducible';
355
+ approvalPolicy: 'never';
356
+ shellNetwork: false;
357
+ webSearch: false;
358
+ serviceTier: 'default';
359
+ shellEnvironment: 'core-filtered';
360
+ loginShell: false;
361
+ credentialsReadable: false;
362
+ hostHomeReadable: false;
363
+ procEnvironment: 'private-sanitized';
364
+ sensitiveEnvironmentNamesVisible: false;
365
+ parentRepoRead: false;
366
+ gitMetadata: false;
367
+ temporaryDirectory: 'workspace-private';
368
+ stagedExecutable: 'static-elf-read-only';
369
+ callerReadDeniedPaths: 'enforced';
370
+ containerSockets: false;
371
+ }
372
+ /** Zero-model-call evidence for the exact Codex process about to run. */
373
+ interface CodexExecutionEvidence {
374
+ cliVersion: string;
375
+ executableSha256: string;
376
+ /** SHA-256 of the exact composed prompt argument proved present in the rendered prompt. */
377
+ requestedPromptSha256: string;
378
+ effectivePromptSha256: string;
379
+ nonPromptArgsSha256: string;
380
+ controlledConfigSha256: string;
381
+ /** Sorted normalized paths compiled into the permission profile. */
382
+ readDeniedPaths: string[];
383
+ readDeniedPathsSha256: string;
384
+ readDeniedPathCount: number;
385
+ policy: CodexExecutionPolicy;
386
+ }
387
+ /** @experimental */
388
+ interface LocalHarnessResult {
389
+ /** OS exit code. `null` when killed before exit. */
390
+ exitCode: number | null;
391
+ /** Concatenated stdout. */
392
+ stdout: string;
393
+ /** Concatenated stderr. */
394
+ stderr: string;
395
+ /** Set when the process exited via signal (timeout / abort). */
396
+ killedBySignal: NodeJS.Signals | null;
397
+ /** Wall-clock duration ms (spawn → exit). */
398
+ durationMs: number;
399
+ /** Set when timeoutMs elapsed before exit. */
400
+ timedOut: boolean;
401
+ /**
402
+ * Set when the caller's AbortSignal fired before this result settled.
403
+ * Optional so injected runners and stored results from older releases remain valid.
404
+ */
405
+ aborted?: boolean;
406
+ /** Present for a reproducible Codex run; parsed from the real terminal JSONL event. */
407
+ usage?: CodexTokenUsage;
408
+ /** Present for reproducible Codex runs; generated and checked before model execution. */
409
+ evidence?: CodexExecutionEvidence;
410
+ }
411
+ /**
412
+ * Spawn a local coding harness CLI as a subprocess + collect its output.
413
+ *
414
+ * NOT responsible for parsing the harness's output or extracting a diff —
415
+ * the in-process executor's `streamPrompt` orchestrates `git diff` against
416
+ * the worktree after this resolves. This function is intentionally narrow:
417
+ * spawn, wait, capture, return.
418
+ *
419
+ * Fails loud — throws when:
420
+ * - `cwd` doesn't exist (subprocess emits ENOENT; surfaced as Error)
421
+ * - the harness binary is not on PATH (ENOENT)
422
+ * - the caller signal was already aborted before process launch
423
+ *
424
+ * Does NOT throw when:
425
+ * - the subprocess exits non-zero (`result.exitCode` carries the code)
426
+ * - a non-reproducible subprocess is aborted / timed out (`result.aborted` /
427
+ * `result.timedOut` carries the reason even when a TERM-aware child exits zero)
428
+ *
429
+ * Reproducible Codex additionally requires a terminal usage event. If cancellation
430
+ * prevents that event, this rejects with `CodexExecutionDiagnosticError` instead of
431
+ * returning an incomplete reproducibility receipt.
432
+ *
433
+ * @experimental
434
+ */
435
+ declare function runLocalHarness(options: RunLocalHarnessOptions): Promise<LocalHarnessResult>;
436
+ /** Parse and validate the one terminal usage event emitted by `codex exec --json`. */
437
+ declare function parseCodexTokenUsage(stdout: string): CodexTokenUsage;
438
+ //#endregion
439
+ //#region src/mcp/worktree-harness.d.ts
440
+ /** Outcome of one verification command run in the worktree (test or typecheck). */
441
+ interface WorktreeCommandResult {
442
+ /** The shell command line that was run. */
443
+ command: string;
444
+ /** Did the command exit 0? The PASS signal a deliverable gate / coder output reads. */
445
+ passed: boolean;
446
+ /** OS exit code, or `null` when killed before exit. */
447
+ exitCode: number | null;
448
+ /** Combined stdout+stderr (capped) — surfaced in traces for diagnosis. */
449
+ output: string;
450
+ }
451
+ /** Proof of the profile inputs delivered before the worker process started. */
452
+ interface WorktreeProfileMaterializationReceipt {
453
+ /** Digest of the exact materializer plan: files, modes, environment, flags, and unsupported rows. */
454
+ workspacePlanDigest: string;
455
+ /** Repository-relative profile input files written into the worker worktree. */
456
+ writtenPaths: string[];
457
+ /** Must be empty on a successful run because this path fails closed. */
458
+ unsupported: WorkspacePlanReceipt['unsupported'];
459
+ /** Environment variable names added to the worker process. Values remain out of telemetry. */
460
+ environmentNames: string[];
461
+ /** Exact additional CLI arguments emitted by the materializer. */
462
+ flags: string[];
463
+ /** `resources.instructions` bypasses native project files so reproducible Codex cannot drop it. */
464
+ resourceInstructions: {
465
+ delivery: 'none' | 'invocation-prompt';
466
+ sha256: string | null;
467
+ byteLength: number;
468
+ };
469
+ }
470
+ /** The canonical result of one worktree-harness run, projected by each port to its own shape. */
471
+ interface WorktreeHarnessResult {
472
+ /** The branch the worktree was cut on (`delegate/<runId>`). */
473
+ branch: string;
474
+ /** `git diff` of the worktree against its base — the unified patch the harness produced. */
475
+ patch: string;
476
+ /** Shortstat-derived change counts. */
477
+ stats: {
478
+ filesChanged: number;
479
+ insertions: number;
480
+ deletions: number;
481
+ };
482
+ /**
483
+ * Exact profile materialization applied before the harness launched.
484
+ * Absent on transports that cannot return a materializer receipt; never fabricated.
485
+ */
486
+ profileMaterialization?: WorktreeProfileMaterializationReceipt;
487
+ /** The harness subprocess outcome. */
488
+ harness: {
489
+ name: LocalHarness | 'bridge';
490
+ exitCode: number | null;
491
+ timedOut: boolean;
492
+ killedBySignal: NodeJS.Signals | null;
493
+ durationMs: number;
494
+ stdout: string;
495
+ stderr: string;
496
+ /** Exact Codex JSONL usage when reproducible mode is enabled. */
497
+ usage?: CodexTokenUsage;
498
+ /** Installed CLI version captured immediately before execution. */
499
+ cliVersion?: string;
500
+ /** SHA-256 of the native Codex executable staged read-only in the candidate worktree. */
501
+ executableSha256?: string;
502
+ /** SHA-256 of the exact composed prompt argument proved present in Codex's rendered prompt. */
503
+ requestedPromptSha256?: string;
504
+ /** SHA-256 of `codex debug prompt-input` output for the exact isolated prompt. */
505
+ effectivePromptSha256?: string;
506
+ /** SHA-256 of the exact executable + argv with prompt content replaced by `<PROMPT>`. */
507
+ nonPromptArgsSha256?: string;
508
+ /** SHA-256 of the isolated config that fixes permissions and shell environment. */
509
+ controlledConfigSha256?: string;
510
+ /** SHA-256 of the normalized caller-supplied host read-denial paths. */
511
+ readDeniedPathsSha256?: string;
512
+ /** Sorted normalized caller-supplied host read-denial paths. */
513
+ readDeniedPaths?: string[];
514
+ /** Number of normalized caller-supplied host read-denial paths. */
515
+ readDeniedPathCount?: number;
516
+ /** Explicit isolation claims checked before model execution. */
517
+ executionPolicy?: CodexExecutionPolicy;
518
+ };
519
+ /** Verification signals derived in the live worktree (present only when commands were given). */
520
+ checks?: {
521
+ tests?: WorktreeCommandResult;
522
+ typecheck?: WorktreeCommandResult;
523
+ };
524
+ }
525
+ /** The single shell-command-in-worktree runner seam (replaces the per-executor copies). */
526
+ type WorktreeCheckRunner = (opts: {
527
+ command: string;
528
+ cwd: string;
529
+ timeoutMs: number;
530
+ signal?: AbortSignal;
531
+ }) => Promise<{
532
+ exitCode: number | null;
533
+ output: string;
534
+ }>;
535
+ //#endregion
536
+ //#region src/runtime/supervise/inbox.d.ts
537
+ /**
538
+ *
539
+ * The worker-side receive end of the down-leg: a per-worker inbox an executor exposes as
540
+ * `Executor.deliver`. The driver's `steer_agent` / `answer_question` land here,
541
+ * and the worker's agent loop drains them at two points (Drew's two delivery modes):
542
+ *
543
+ * - QUEUED (default): the message accumulates and is FLUSHED at the next step boundary — folded
544
+ * into the conversation before the next think. A worker is also forced to flush BEFORE it may
545
+ * settle, so it can never finish while a steer/answer it never read is still pending.
546
+ * - FORCEFUL (`interrupt: true`): trips `freshInterrupt()`'s signal so the loop can abort its
547
+ * in-flight turn immediately, then re-plan with the message folded in — breaking the worker out
548
+ * of a wrong path mid-task instead of waiting for it to finish the step.
549
+ *
550
+ * `deliver` never throws — a malformed message is ignored and returns `false`, so no caller can
551
+ * report delivery for bytes this inbox discarded.
552
+ *
553
+ * @experimental
554
+ */
555
+ interface InboxMessage {
556
+ readonly kind: 'steer' | 'answer';
557
+ readonly text: string;
558
+ /** Forceful messages abort the in-flight turn; queued ones wait for the boundary flush. */
559
+ readonly interrupt: boolean;
560
+ /** Present for an `answer` — the question id it resolves. */
561
+ readonly questionId?: string;
562
+ }
563
+ interface Inbox {
564
+ /** The `Executor.deliver` implementation. Returns false when the raw message is malformed and
565
+ * therefore was not queued; callers must not acknowledge a message this inbox discarded. */
566
+ deliver(msg: unknown): boolean;
567
+ /** Remove and return all pending messages (the flush). */
568
+ drain(): InboxMessage[];
569
+ pending(): number;
570
+ /** Open a fresh per-turn interrupt signal; a later forceful `deliver` aborts it. The loop links
571
+ * this into the signal it passes to its inference call, then re-plans when it fires. */
572
+ freshInterrupt(): AbortSignal;
573
+ /** Render drained messages as ONE operator turn to fold into the worker's conversation. */
574
+ fold(messages: ReadonlyArray<InboxMessage>): string;
575
+ }
576
+ /** Create the worker-side inbox for the down-leg: the driver's `steer_agent` / `answer_question` messages queue here and the worker's loop drains them at step boundaries and before settle. */
577
+ declare function createInbox(): Inbox;
578
+ //#endregion
579
+ //#region src/runtime/supervise/sandbox-session.d.ts
580
+ /** Ceiling on continuation turns. Turn 0 is the task; every later turn is a folded steer, so
581
+ * this bounds how many times a supervisor may redirect ONE worker before it must respawn. */
582
+ declare const DEFAULT_SANDBOX_STEERING_MAX_TURNS = 24;
583
+ /** Opt-in configuration for the steerable sandbox worker (`SandboxSeam.steering`). Absent, the
584
+ * sandbox executor keeps its historical single-shot `runAgentRounds` composition verbatim. */
585
+ interface SandboxSteeringOptions {
586
+ /** Max turns for one worker (turn 0 + folded steers). Default {@link DEFAULT_SANDBOX_STEERING_MAX_TURNS}. */
587
+ readonly maxTurns?: number;
588
+ /** How many recent tool/turn notes `progress()` reports. Default 12. */
589
+ readonly activityWindow?: number;
590
+ /** Per-turn wall-clock ceiling; the turn's stream is aborted when it elapses. */
591
+ readonly turnTimeoutMs?: number;
592
+ }
593
+ /** What the steerable session exposes to its executor: the usage stream plus the live reads. */
594
+ interface SteerableSandboxSession {
595
+ /** Drive the worker to settlement. `signal` is the spawn-scoped abort handed to `execute`. */
596
+ stream(task: unknown, signal: AbortSignal): AsyncIterable<UsageEvent>;
597
+ progress(): ExecutorProgress;
598
+ traceSource(): TraceSource;
599
+ artifact(): {
600
+ outRef: string;
601
+ out: unknown;
602
+ spent: Spend;
603
+ } | undefined;
604
+ teardown(): Promise<void>;
605
+ }
606
+ interface SteerableSandboxArgs {
607
+ readonly controller: AbortController;
608
+ readonly profile: AgentProfile;
609
+ readonly harness: BackendType;
610
+ readonly sandboxClient: SandboxClient;
611
+ readonly inbox: Inbox;
612
+ readonly taskToPrompt: (task: unknown) => string;
613
+ readonly options?: SandboxSteeringOptions;
614
+ readonly loopCtx?: Partial<Omit<ExecCtx, 'sandboxClient' | 'signal'>>;
615
+ /**
616
+ * Inherited `TRACE_ID` / `PARENT_SPAN_ID` for the box, merged into `CreateSandboxOptions.env` so
617
+ * the remote worker's own spans join the supervisor's trace under the spawning node's span.
618
+ * Absent when the run records no spans — the create options are then untouched.
619
+ */
620
+ readonly traceEnv?: Record<string, string>;
621
+ readonly contentRef: (prefix: string, value: unknown) => string;
622
+ readonly now?: () => number;
623
+ }
624
+ /** One steerable sandbox worker. The returned session is inert until `stream()` is drained. */
625
+ declare function createSteerableSandboxSession(args: SteerableSandboxArgs): SteerableSandboxSession;
626
+ //#endregion
627
+ //#region src/runtime/supervise/runtime.d.ts
628
+ /**
629
+ * Router/inline transport seam. The profile owns model, prompt, and generation behavior.
630
+ */
631
+ interface RouterSeam {
632
+ routerBaseUrl: string;
633
+ routerKey: string;
634
+ /** Injectable transport for offline/local execution; still passes through Runtime metering. */
635
+ complete?: RouterConfig['complete'];
636
+ /** When present, return one turn's requested tool calls without executing them. */
637
+ tools?: ReadonlyArray<ToolSpec>;
638
+ }
639
+ /**
640
+ * Sandbox executor seam. The `sandboxClient` the composed `runAgentRounds` creates
641
+ * boxes through, plus the optional trace/run/lineage wiring forwarded into the
642
+ * loop. `lineage` is opaque here (PR #150's `RunAgentRoundsOptions.lineage`): forwarded
643
+ * forward-compatibly, never inspected — this executor does NOT reinvent
644
+ * checkpoint/fork.
645
+ */
646
+ interface SandboxSeam {
647
+ sandboxClient: SandboxClient;
648
+ /** Forwarded into the composed `runAgentRounds`'s `ctx` (trace emitter, run handle, etc.). */
649
+ loopCtx?: Partial<Omit<ExecCtx, 'sandboxClient' | 'signal'>>;
650
+ /** PR #150 `RunAgentRoundsOptions.lineage` passthrough — opaque; forwarded, not parsed. */
651
+ lineage?: unknown;
652
+ /** Hard cap on the composed loop's iterations. The budget pool reserves against
653
+ * the spawn `Budget.maxIterations`; this is the leaf's own ceiling. Default 1. */
654
+ maxIterations?: number;
655
+ /**
656
+ * OPT-IN: run this worker as a multi-turn, STEERABLE session instead of the historical
657
+ * single-shot `runAgentRounds` composition. Setting it gives the sandbox worker an `Executor.deliver`
658
+ * inbox (so `Scope.send` / `steer_agent` actually reach it), a live tool-activity trace, and a
659
+ * `progress()` read — turning the default cloud worker from something a supervisor can only
660
+ * wait on into something it can watch and correct.
661
+ *
662
+ * Absent, nothing changes: the same `runAgentRounds` leaf, no inbox, `steer_agent` still reports
663
+ * `delivered:false`. Opt-in because a steerable worker holds ONE box across several turns,
664
+ * which is a different resource profile from a fire-and-forget shot.
665
+ */
666
+ steering?: SandboxSteeringOptions;
667
+ }
668
+ /**
669
+ * UNMETERED CLI subprocess seam. `bin` + `args` describe the process to spawn.
670
+ *
671
+ * READ THIS BEFORE CHOOSING `backend: 'cli'`. This backend pipes a prompt to a subprocess's stdin
672
+ * and reads its stdout. It has no usage receipt of any kind, so it reports its spend with
673
+ * `Spend.tokensKnown: false`: the work is recorded, its `{0,0}` tokens and `$0` are a FLOOR rather
674
+ * than a measurement, and a ceiling priced from either is a ceiling that cannot fire. The executor
675
+ * is also `budgetExempt: true`, which is why `driveHarnessFromBackend` refuses it outright rather
676
+ * than pretending to budget it.
677
+ *
678
+ * If you need a metered harness worker, use `backend: 'bridge'` (a cli-bridge session, which
679
+ * reports the harness's real per-turn tokens and cost) or `backend: 'cli-worktree'` with
680
+ * `codexReproducible`. Reach for this seam only when the subprocess genuinely is not an inference
681
+ * agent, or when you have accepted that its cost is invisible.
682
+ *
683
+ * `args` is argv for a LOCAL, in-process spawn under this process's own privileges. It is not a
684
+ * remote channel and nothing forwards it over a wire.
685
+ */
686
+ interface CliSeam {
687
+ bin: string;
688
+ args?: string[];
689
+ /** Extra environment for the subprocess (merged over `process.env`). */
690
+ env?: Record<string, string>;
691
+ /** Working directory for the subprocess. */
692
+ cwd?: string;
693
+ }
694
+ /**
695
+ * cli-worktree seam. A supervisor-authored `AgentProfile` driving a local coding-harness CLI
696
+ * (claude / codex / opencode) on its own git worktree — the leaf `createWorktreeCliExecutor`
697
+ * named as data. `repoRoot` is transport data; `AgentProfile.harness` selects the CLI.
698
+ * `taskPrompt` remains an optional direct-call fallback for callers that execute with `undefined`.
699
+ * The authored
700
+ * `profile.prompt.systemPrompt` + `profile.model.default` reach the harness via the §1.5
701
+ * `harnessInvocation` mapper. Everything else mirrors `WorktreeCliExecutorOptions`.
702
+ */
703
+ interface CliWorktreeSeam {
704
+ repoRoot: string;
705
+ taskPrompt?: string;
706
+ runId?: string;
707
+ baseRef?: string;
708
+ harnessTimeoutMs?: number;
709
+ /** Isolated, network-off Codex execution with terminal JSONL usage capture. */
710
+ codexReproducible?: boolean;
711
+ /** Absolute host paths denied to reproducible Codex. */
712
+ codexReadDeniedPaths?: ReadonlyArray<string>;
713
+ testCmd?: string;
714
+ typecheckCmd?: string;
715
+ checkTimeoutMs?: number;
716
+ checkOutputCap?: number;
717
+ budgetExempt?: boolean;
718
+ /** Live cli-bridge transport inside the worktree. When set, the worktree leaf accepts
719
+ * `deliver()` messages and resumes the same bridge session in this worktree cwd. */
720
+ bridge?: CliWorktreeBridgeSeam;
721
+ /** Test seam — forwarded to worktree helpers. */
722
+ runGit?: GitRunner;
723
+ /** Test seam — forwarded to verification checks. */
724
+ runCommand?: WorktreeCheckRunner;
725
+ }
726
+ interface CliWorktreeBridgeSeam {
727
+ bridgeUrl: string;
728
+ bridgeBearer: string;
729
+ /** Caller-owned deadline for each bridge turn. Runtime enforces it locally and sends the
730
+ * same value in `execution.timeoutMs` so cli-bridge cannot substitute its own cutoff. */
731
+ timeoutMs?: number;
732
+ /** Stable cli-bridge session id. Defaults to `bridge-worktree-${runId}`. */
733
+ sessionId?: string;
734
+ /** Transport reconnects allowed after the first POST. Default 3; set 0 to disable. */
735
+ maxReconnects?: number;
736
+ }
737
+ /**
738
+ * cli-bridge seam. A local OpenAI-compatible bridge that fronts harness CLIs
739
+ * (claude-code / opencode / kimi / pi) behind one HTTP surface. The spawned
740
+ * `AgentProfile` is the sole harness/provider/model and behavioral authority and
741
+ * is forwarded verbatim per request; this seam carries transport data only.
742
+ *
743
+ * The executor opens a resumable cli-bridge session. `sessionId` identifies the
744
+ * harness conversation across turns; each turn also receives its own durable run id.
745
+ * A dropped HTTP reader reattaches to that exact run and explicit cancel is the only
746
+ * operation allowed to stop it. Omit `sessionId` and the executor mints one per spawn.
747
+ *
748
+ * ── HOW TO CONTROL WHAT THE HARNESS LOADS (there is no argv field, by design) ──
749
+ *
750
+ * A worker often needs the harness started in a KNOWN state — no ambient extensions, skills,
751
+ * context files, or prompt templates — because ambient state is how a paired experiment silently
752
+ * loses its pairing: an installed extension that persists memory across runs carries arm A's state
753
+ * into arm B, and nothing reports it.
754
+ *
755
+ * That is what the spawned `AgentProfile` is FOR. `agent_profile`
756
+ * rides every request verbatim, and cli-bridge maps it onto each harness's own native controls:
757
+ *
758
+ * - Materializing any profile at all already starts the harness isolated from ambient
759
+ * workspace state — for pi that is `--no-context-files --no-skills --no-prompt-templates`,
760
+ * applied to every request that carries an `agent_profile`.
761
+ * - `AgentProfile.extensions.<harness>` is the named, per-harness control channel. An explicit
762
+ * `extensions: { pi: { load: [] } }` disables ambient extension discovery outright
763
+ * (pi's `--no-extensions`); listing package names loads exactly those and nothing else.
764
+ * - `permissions` / `tools` / `mcp` map onto the harness's native tool and server controls.
765
+ *
766
+ * A caller therefore does NOT need to hand-roll an `Executor` to isolate a harness run, and the
767
+ * profile expressing it stays portable: the same declaration means the same thing on a different
768
+ * harness, whereas an argv string means nothing anywhere else.
769
+ *
770
+ * WHY NOT A GENERAL ARGV PASSTHROUGH. `bridgeUrl` addresses a process-spawning server. Forwarding
771
+ * an arbitrary argv array to it would let any caller holding a bearer token choose the flags of a
772
+ * process on the bridge host — which for real harness CLIs includes flags that load code from a
773
+ * path, read a file into the prompt, redirect the working directory, or turn off the isolation the
774
+ * bridge applies. cli-bridge deliberately confines workers (a filesystem jail and deny-by-default
775
+ * network egress), and every one of those confinements is expressed as spawn configuration, so an
776
+ * argv channel is a channel for unwinding them. It would also break this executor's own contract:
777
+ * the durable-run replay protocol, session pinning, and streaming mode are all argv the bridge
778
+ * owns, and a caller-supplied duplicate silently wins or corrupts the parse. The structured profile
779
+ * channel is validated, per-harness, portable, and refuses controls it does not understand — keep
780
+ * new harness capability there.
781
+ */
782
+ interface BridgeSeam {
783
+ bridgeUrl: string;
784
+ bridgeBearer: string;
785
+ /** Optional working directory forwarded to cli-bridge and persisted with the session. */
786
+ cwd?: string;
787
+ /** Caller-owned deadline for each bridge turn. Runtime enforces it locally and sends the
788
+ * same value in `execution.timeoutMs` so the bridge-owned process follows the same policy. */
789
+ timeoutMs?: number;
790
+ /** Stable, caller-owned cli-bridge session id for harness-side resume. Defaults
791
+ * to a freshly minted per-spawn id so each worker is its own resumable session. */
792
+ sessionId?: string;
793
+ /** Transport reconnects allowed after the first POST. Default 3; set 0 to disable. */
794
+ maxReconnects?: number;
795
+ /** Newest-last activity window `progress()` reports. Default 12. */
796
+ activityWindow?: number;
797
+ }
798
+ /** Generic environment provider executor config. External packages implement
799
+ * `AgentEnvironmentProvider`; this built-in wrapper lets `createExecutor`
800
+ * consume them as backend data while preserving the existing usage channel. */
801
+ interface ProviderSeam extends ProviderExecutorOptions {
802
+ provider: AgentEnvironmentProvider | string;
803
+ registry?: AgentEnvironmentProviderRegistry;
804
+ /**
805
+ * Compose the provider through the existing steerable sandbox session.
806
+ * The exact profile must name its harness, and the provider must expose live
807
+ * continuation plus session controls. The provider still owns environment
808
+ * creation and session semantics.
809
+ */
810
+ steering?: SandboxSteeringOptions;
811
+ }
812
+ /**
813
+ * Router seam WITH tool use — the tool-using router backend. Same direct
814
+ * OpenAI-compatible endpoint as `RouterSeam`, but each turn passes `tools`; when
815
+ * the model emits tool_calls they run via `executeToolCall` ON THIS HOST and the
816
+ * results fold back as `tool` messages, repeating until the model answers without
817
+ * a tool or `maxTurns` is hit. A real agentic loop, OFF-BOX — no sandbox, so it
818
+ * is unaffected by a box's egress allowlist. One turn = one completion = the
819
+ * equal-compute unit. `executeToolCall` receives the task so per-task tool
820
+ * surfaces (e.g. a gym keyed by task) can dispatch correctly.
821
+ */
822
+ interface RouterToolsSeam {
823
+ routerBaseUrl: string;
824
+ routerKey: string;
825
+ complete?: RouterConfig['complete'];
826
+ tools: ReadonlyArray<ToolSpec>;
827
+ executeToolCall: (name: string, args: Record<string, unknown>, task: unknown) => Promise<string>;
828
+ /** Exact conversation to continue. Runtime validates its system message against the profile. */
829
+ initialMessages?: ReadonlyArray<Readonly<Record<string, unknown>>>;
830
+ /** Observe the detached final conversation for session persistence. */
831
+ onMessages?: (messages: ReadonlyArray<Readonly<Record<string, unknown>>>) => void | Promise<void>;
832
+ /** Online observer of each tool step — the seam a `DetectorMonitor` taps to watch the live pipe
833
+ * (raise a `finding` when the worker loops/errors). Called after every tool call resolves, with
834
+ * real per-call wall-clock (`startedAt`/`endedAt`/`durationMs`) so a push `TraceSource` can carry
835
+ * non-zero span durations onto the unified timeline. */
836
+ onToolStep?: (step: {
837
+ toolName: string;
838
+ args: Record<string, unknown>;
839
+ status: 'ok' | 'error';
840
+ startedAt?: number;
841
+ endedAt?: number;
842
+ durationMs?: number;
843
+ }) => void;
844
+ }
845
+ /**
846
+ * The leaf `createWorktreeCliExecutor` as a backend-as-data factory: a supervisor-authored
847
+ * `AgentProfile` driving claude / codex / opencode on its own worktree. `budgetExempt` like
848
+ * the other CLI leaves; the authored systemPrompt + model reach the harness via §1.5.
849
+ */
850
+ declare const cliWorktreeExecutor: ExecutorFactory<unknown>;
851
+ /**
852
+ * Config for {@link createExecutor}: the backend is DATA — the cost dial a profile,
853
+ * an experiment config, or a replay journal can name — not an import choice. Each
854
+ * variant carries its backend's seam (router/router-tools/bridge/cli/cli-worktree/sandbox).
855
+ */
856
+ type ExecutorConfig = ({
857
+ backend: 'router';
858
+ } & RouterSeam) | ({
859
+ backend: 'router-tools';
860
+ } & RouterToolsSeam) | ({
861
+ backend: 'bridge';
862
+ } & BridgeSeam) | ({
863
+ backend: 'cli';
864
+ } & CliSeam) | ({
865
+ backend: 'cli-worktree';
866
+ } & CliWorktreeSeam) | ({
867
+ backend: 'provider';
868
+ } & ProviderSeam) | ({
869
+ backend: 'sandbox';
870
+ } & SandboxSeam);
871
+ /**
872
+ * The single built-in executor factory. Picks a leaf backend by data (`config.backend`),
873
+ * injects the matching seam, and delegates to that backend's built-in implementation.
874
+ * The `Executor` port stays OPEN: bring-your-own agents implement `Executor` directly, while Scope
875
+ * or `createExecutorRegistry` still parses and seals their exact profile before use. Use this instead of a
876
+ * per-vendor adapter or a closed `inline|sandbox|cli` switch — those bypass the
877
+ * `UsageEvent` reporting channel.
878
+ */
879
+ declare function createExecutor(config: ExecutorConfig): ExecutorFactory<unknown>;
880
+ /**
881
+ * The open resolver/registry. Pre-registers the three built-ins under their
882
+ * runtime tags (`'router'`, `'sandbox'`, `'cli'`) and accepts `register(name,
883
+ * factory)` for any additional runtime. A BYO `AgentSpec.executor` has highest routing precedence
884
+ * after the same exact-profile intake validation. Registration + BYO remain open extension points.
885
+ *
886
+ * `resolve` precedence (frozen in `ExecutorRegistry`): a BYO `spec.executorFactory` →
887
+ * `spec.executor` → `harness === null` → the `'router'` factory; else a registered factory for the
888
+ * harness-derived runtime (`'sandbox'` for any `BackendType`); else fail loud.
889
+ */
890
+ declare function createExecutorRegistry(): ExecutorRegistry;
891
+ //#endregion
892
+ export { ToolLoopToolCall as $, LocalHarness as A, GitRunner as B, WorktreeHarnessResult as C, CodexTokenUsage as D, CodexExecutionPolicy as E, parseCodexTokenUsage as F, removeWorktree as G, WorktreeHandle as H, runLocalHarness as I, ToolLoopCallContext as J, RouterTransportConfig as K, CreateWorktreeOptions as L, RunLocalHarnessOptions as M, harnessSupportsReasoningEffort as N, DEFAULT_LOCAL_HARNESS as O, localHarnessExecutable as P, ToolLoopMessageRecord as Q, DiffOptions as R, WorktreeCommandResult as S, CodexExecutionEvidence as T, captureWorktreeDiff as U, RemoveWorktreeOptions as V, createWorktree as W, ToolLoopCompaction as X, ToolLoopChat as Y, ToolLoopCompactionOptions as Z, createSteerableSandboxSession as _, ExecutorConfig as a, createInbox as b, RouterToolsSeam as c, createExecutor as d, createExecutorRegistry as f, SteerableSandboxSession as g, SteerableSandboxArgs as h, CliWorktreeSeam as i, LocalHarnessResult as j, LOCAL_HARNESSES as k, SandboxSeam as l, SandboxSteeringOptions as m, CliSeam as n, ProviderSeam as o, DEFAULT_SANDBOX_STEERING_MAX_TURNS as p, ToolSpec as q, CliWorktreeBridgeSeam as r, RouterSeam as s, BridgeSeam as t, cliWorktreeExecutor as u, Inbox as v, WorktreeProfileMaterializationReceipt as w, WorktreeCheckRunner as x, InboxMessage as y, DiffResult as z };
893
+ //# sourceMappingURL=runtime-D-QfLbSd.d.ts.map