@ordewell/core 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/LICENSE +211 -0
  2. package/README.md +37 -0
  3. package/dist/ITerminalRunner-Bbpe8uts.d.ts +535 -0
  4. package/dist/ITerminalRunner-D8Bf-HD4.d.mts +535 -0
  5. package/dist/Task-1NwmUImI.d.mts +264 -0
  6. package/dist/Task-1NwmUImI.d.ts +264 -0
  7. package/dist/chunk-4KIWPW4K.mjs +510 -0
  8. package/dist/chunk-4KIWPW4K.mjs.map +1 -0
  9. package/dist/chunk-SRGQBGI5.mjs +463 -0
  10. package/dist/chunk-SRGQBGI5.mjs.map +1 -0
  11. package/dist/chunk-YXAIQKVE.mjs +398 -0
  12. package/dist/chunk-YXAIQKVE.mjs.map +1 -0
  13. package/dist/index.d.mts +2990 -0
  14. package/dist/index.d.ts +2990 -0
  15. package/dist/index.js +12927 -0
  16. package/dist/index.js.map +1 -0
  17. package/dist/index.mjs +11449 -0
  18. package/dist/index.mjs.map +1 -0
  19. package/dist/parsing-RjN9JbyG.d.ts +161 -0
  20. package/dist/parsing-xpozY5ac.d.mts +161 -0
  21. package/dist/parsing.d.mts +2 -0
  22. package/dist/parsing.d.ts +2 -0
  23. package/dist/parsing.js +432 -0
  24. package/dist/parsing.js.map +1 -0
  25. package/dist/parsing.mjs +15 -0
  26. package/dist/parsing.mjs.map +1 -0
  27. package/dist/plan-utils-Cgc0cCH1.d.ts +111 -0
  28. package/dist/plan-utils-nfHKSbd3.d.mts +111 -0
  29. package/dist/plan-utils.d.mts +2 -0
  30. package/dist/plan-utils.d.ts +2 -0
  31. package/dist/plan-utils.js +181 -0
  32. package/dist/plan-utils.js.map +1 -0
  33. package/dist/plan-utils.mjs +18 -0
  34. package/dist/plan-utils.mjs.map +1 -0
  35. package/dist/testing.d.mts +28 -0
  36. package/dist/testing.d.ts +28 -0
  37. package/dist/testing.js +113 -0
  38. package/dist/testing.js.map +1 -0
  39. package/dist/testing.mjs +86 -0
  40. package/dist/testing.mjs.map +1 -0
  41. package/package.json +92 -0
@@ -0,0 +1,2990 @@
1
+ import { h as RunnerId, c as PlanStatus, L as LegacyPlanState, d as ResearchProgress, a as DiscoveredModel, T as Task, R as ResearchLogEntry, C as ConversationMessage, n as TaskSnapshot, Q as QueuedMessage, A as ActiveTaskSession, g as ResearchToolType, u as Verdict, e as ResearchStep, b as PlanState } from './Task-1NwmUImI.js';
2
+ export { D as DiscoveredMode, M as Message, P as PlanModificationWarnings, f as ResearchStepOutcome, S as StreamEvent, i as StreamStepEvent, j as StreamThinkingEvent, k as TaskMode, l as TaskModelAssignment, m as TaskOutputSummary, o as TaskStatus, p as TaskType, q as ThinkingBlock, U as UserPromptEntry, r as UserStep, V as ValidationCheck, s as ValidationContext, t as ValidationResult, v as VerificationCheck, w as addTaskToPlan, x as createEmptyPlan, y as createTask, z as emptyWarnings, B as flattenTasks, E as migrateLegacyPlan, F as migratePlanState, G as migrateTask, H as removeTaskFromPlan, I as renumberTasks, J as updateTaskInPlan, K as validateModifiedPlan, N as warningsText } from './Task-1NwmUImI.js';
3
+ import { n as IFileSystem, I as IApproval, R as ReadFileOpts, T as ToolOutcome, k as GrepOptions, j as GlobOptions, i as FindSymbolOptions, f as ApprovalRequest, m as IConfig, A as AiProvider, y as ProviderModelLists, c as ApprovalMode, q as ITerminalSession, p as ITerminalRunner, J as RunnerRegistry, s as OrchestratorOption, b as ApprovalKind, g as ApprovalSource, E as RunnerInvocation, o as IPluginStore, H as RunnerPluginManifest, B as ResolveContext } from './ITerminalRunner-Bbpe8uts.js';
4
+ export { a as AllProviderModels, d as ApprovalPolicy, e as ApprovalPolicyOptions, C as CatalogModel, D as DENY_ALL, h as DiscoveryCommand, F as FetchAllProviderModelsOptions, G as GREP_DEFAULT_HEAD_LIMIT, l as GrepOutputMode, M as ModelCatalog, r as ModelShortcut, O as ORCHESTRATOR_SHORTCUTS, P as PluginEntry, t as PluginFeatures, u as PluginMode, v as PluginModelDiscovery, w as PluginRunnerDef, x as ProviderCredentialSource, z as ProviderModelsResult, S as SEARCH_EXCLUSIONS, K as collectProviderCredentials, L as enabledRunners, N as fetchAllProviderModels, Q as knownModelId, U as resolveModelShortcut, V as resolveProvider, W as toOrchestratorOptions } from './ITerminalRunner-Bbpe8uts.js';
5
+ import { R as RunnerModeInfo, b as PlanParseError } from './parsing-RjN9JbyG.js';
6
+ export { M as ManifestLookup, P as PLAN_ENVELOPE_KEY, a as PartialPlanTask, T as TASK_OPS_ENVELOPE_KEY, c as buildModeGuide, d as checkDepsResolve, e as checkImmutableLog, f as checkInProgress, g as checkNoCycles, h as checkUniqueIds, i as escapeControlCharsInStrings, j as extractJsonObject, k as extractObjectWithBalance, l as extractObjectsWithKey, m as filteredBuildModes, n as looksLikePlanAttempt, p as parsePartialPlan, o as parsePlanJson, r as resolveDefaultMode, q as resolveTaskMode, s as runnerModesFrom, t as stripModelNoise, u as stripTrailingCommas, v as validatePlanModification } from './parsing-RjN9JbyG.js';
7
+ import { T as TaskOp } from './plan-utils-Cgc0cCH1.js';
8
+ export { A as ApplyTaskOpsResult, a as TaskRef, b as applyTaskOps, c as canMergeTasks, d as canSetDependencies, e as canSplitTask, f as classifyOutcome, g as dependencyCandidates, h as dependentsOf, p as parseTaskOpsJson, s as summarizeToolCall, t as textHasTaskOps } from './plan-utils-Cgc0cCH1.js';
9
+ import { ChildProcess } from 'child_process';
10
+ import { EventEmitter } from 'events';
11
+
12
+ interface SessionMeta {
13
+ id: string;
14
+ goal: string;
15
+ runners: RunnerId[];
16
+ taskCount: number;
17
+ status: PlanStatus;
18
+ createdAt: string;
19
+ updatedAt: string;
20
+ }
21
+ interface SessionData {
22
+ meta: SessionMeta;
23
+ /** The full plan state — including conversationHistory, prdMarkdown, and queuedMessages. */
24
+ plan: LegacyPlanState;
25
+ }
26
+
27
+ /**
28
+ * Which interpreter runs the planner's `bash` tool, and in which language.
29
+ *
30
+ * `AUTO_COMMANDS` — the read-only tier that runs with no prompt — is `ls`, `cat`,
31
+ * `wc`, `head`, `tail`, `du`, `df`, `file`, `sort`, `uniq`, `basename`, `find`,
32
+ * `grep`. That set is POSIX, and passing it to `shell: true` on Windows means
33
+ * cmd.exe, where most of those names do not exist and `find` is an unrelated
34
+ * program that returns plausible wrong output instead of an error. The planner's
35
+ * whole research surface degraded to guessing, quietly.
36
+ *
37
+ * Rather than write a Windows dialect of every research command — a second
38
+ * behavior to keep in step forever — this finds a POSIX shell to run the
39
+ * existing ones in. Git for Windows ships a full one, and git is already a
40
+ * prerequisite for everything Ordewell does, so in practice it is there. Failing
41
+ * that, cmd.exe is used and the classifier is told so, because the one thing
42
+ * that must never happen is classifying a command in one language and running it
43
+ * in another: `commandPolicy` reads {@link ResearchShell.dialect} for exactly
44
+ * that reason.
45
+ *
46
+ * POSIX resolves to `{ file: null }`, meaning "use `shell: true`" — byte for
47
+ * byte what the adapters did before this module existed.
48
+ *
49
+ * Windows paths are built with `path.win32` explicitly, not the host-flavoured
50
+ * `path`, so the probe is the same on a Windows host as it is under a Linux
51
+ * test runner.
52
+ */
53
+ type ShellDialect = 'posix' | 'cmd';
54
+ interface ResearchShell {
55
+ /**
56
+ * Executable that takes a command string, or null to use the host default
57
+ * via `shell: true`.
58
+ */
59
+ file: string | null;
60
+ /** Arguments preceding the command string. */
61
+ args: string[];
62
+ /** The language the command string will be interpreted in. */
63
+ dialect: ShellDialect;
64
+ /**
65
+ * Directory holding this shell's POSIX utilities, when it brings its own.
66
+ * Prepended to PATH for the search subprocesses (`grep`, `tree`) that
67
+ * `execFile` starts without a shell, so they resolve too.
68
+ */
69
+ utilsDir: string | null;
70
+ }
71
+ interface ResearchShellDeps {
72
+ platform?: NodeJS.Platform;
73
+ exists?: (candidate: string) => boolean;
74
+ env?: NodeJS.ProcessEnv;
75
+ }
76
+ /**
77
+ * The shell the planner's `bash` tool should use on this host. Resolved once per
78
+ * process — the answer cannot change while Ordewell runs, and the probe costs a
79
+ * handful of `stat` calls.
80
+ */
81
+ declare function resolveResearchShell(deps?: ResearchShellDeps): ResearchShell;
82
+ /** Test seam: drop the per-process cache so the next call re-probes. */
83
+ declare function clearResearchShellCache(): void;
84
+ /**
85
+ * PATH for the search subprocesses the adapters start without a shell, with the
86
+ * research shell's own utilities in front. Without this, `grep` — the fallback
87
+ * when ripgrep is absent — is unresolvable on a Windows box even when a
88
+ * perfectly good `grep.exe` sits in the Git tree beside the shell.
89
+ */
90
+ declare function researchToolsPath(shell: ResearchShell, basePath: string | undefined): string;
91
+ /**
92
+ * A message explaining a degraded research surface, or null when there is
93
+ * nothing to explain. Surfaced by the adapters on the first refused command so
94
+ * the limitation is visible rather than inferred from bad answers.
95
+ */
96
+ declare function researchShellWarning(shell: ResearchShell): string | null;
97
+
98
+ /**
99
+ * Tiered classification for the planner's `bash` tool.
100
+ *
101
+ * The planner is a read-only researcher: it never edits files, and the runners
102
+ * do the real work. But research legitimately means running things — querying a
103
+ * cloud control plane (`az`, `gh`), reproducing a failure (`npm test`), or
104
+ * shaping output (`jq`). The old allowlist refused all of that, so the model's
105
+ * only escape was to guess.
106
+ *
107
+ * Three tiers replace the flat allowlist:
108
+ *
109
+ * auto read-only inspection — runs with no prompt (the historical list)
110
+ * ask anything else that is not obviously destructive — one approval,
111
+ * remembered for the rest of the session at `scope` granularity
112
+ * refuse writes, privilege escalation, and anything that would smuggle
113
+ * arbitrary code past this classifier — never runs, never prompts
114
+ *
115
+ * `refuse` is deliberately not promptable. A planner that can `rm` is a planner
116
+ * that can silently break the workspace it was asked to reason about, and the
117
+ * architecture already says mutation belongs to the runners.
118
+ *
119
+ * Classification walks every segment of the command line (pipes, `&&`, `;`, and
120
+ * command substitution) rather than matching substrings against the raw string.
121
+ * The old substring denylist both over-matched (`ls docs/removed` tripped `rm`)
122
+ * and under-matched (`$(rm -rf /)` was invisible once chaining was allowed).
123
+ *
124
+ * The lexer is dialect-aware, because the interpreter that will actually run
125
+ * the command decides what the tokens are, and getting that wrong is not a
126
+ * cosmetic error here — it is the difference between classifying what runs and
127
+ * classifying something else. See {@link Dialect}.
128
+ */
129
+
130
+ declare const AUTO_COMMANDS: string[];
131
+ declare const GIT_READONLY_SUBCOMMANDS: string[];
132
+ /**
133
+ * Never runs, with or without approval.
134
+ *
135
+ * The `cmd.exe` builtins at the end matter as much as the POSIX names above
136
+ * them. `del`, `rd`, `move`, and friends mutate exactly what `rm` and `mv` do,
137
+ * and on Windows they are what a model reaches for — so without them the whole
138
+ * refusal tier was bypassable on that platform by writing the command the way
139
+ * the platform spells it. Listed unconditionally rather than per-platform: a
140
+ * POSIX box has no `del` to refuse, so the extra names cost nothing there.
141
+ */
142
+ declare const REFUSED_COMMANDS: string[];
143
+ /**
144
+ * Options threaded through classification.
145
+ *
146
+ * `dialect` is keyed to the interpreter rather than the OS on purpose: a Windows
147
+ * host that has Git Bash runs the planner's commands in a POSIX shell (see
148
+ * {@link resolveResearchShell}), and classifying those under cmd.exe rules
149
+ * would be the same mismatch in the other direction. Callers pass the dialect of
150
+ * the shell they are actually going to use; omitting it falls back to the host
151
+ * default.
152
+ */
153
+ interface CommandPolicyOptions {
154
+ dialect?: ShellDialect;
155
+ }
156
+ type CommandTier = 'auto' | 'ask' | 'refuse';
157
+ interface CommandClassification {
158
+ tier: CommandTier;
159
+ /** What a grant covers, for `ask`. Derived from the non-auto segments only. */
160
+ scope: string;
161
+ /** Populated for `refuse`: why, in a sentence the model can act on. */
162
+ reason?: string;
163
+ }
164
+ /**
165
+ * Classify one command line. Output redirection is refused outright: a planner
166
+ * that writes files has stopped being a planner.
167
+ */
168
+ declare function classifyCommand(command: string, opts?: CommandPolicyOptions): CommandClassification;
169
+
170
+ /**
171
+ * The policy half of the planner's filesystem: path confinement, the tiered
172
+ * `bash` gate, and the definition-first symbol lookup. Adapters supply only the
173
+ * mechanics (`*Impl`), so the rules live in one place across the web server and
174
+ * the VS Code extension rather than being re-derived per surface.
175
+ *
176
+ * Every public method resolves and authorizes before delegating, and every
177
+ * `*Impl` therefore receives an absolute, already-approved path. Previously
178
+ * each adapter did its own resolution, and `path.isAbsolute(p) ? p : …` meant
179
+ * an absolute path walked straight out of the workspace with no prompt and no
180
+ * record.
181
+ */
182
+ declare abstract class BaseFileSystem implements IFileSystem {
183
+ private approval;
184
+ /**
185
+ * The interpreter `execBashImpl` will hand the command to. Owned here rather
186
+ * than per-adapter because {@link classifyCommand} has to be told the same
187
+ * answer: a command lexed under POSIX rules and then run by cmd.exe is a
188
+ * command this class did not actually classify.
189
+ */
190
+ protected readonly researchShell: ResearchShell;
191
+ /** Surfaces inject the human channel here; without it, external access is denied. */
192
+ setApproval(approval: IApproval): void;
193
+ abstract getWorkspaceRoot(): string;
194
+ protected abstract readFileImpl(absPath: string, opts?: ReadFileOpts): Promise<ToolOutcome>;
195
+ protected abstract globImpl(pattern: string, absRoot: string, headLimit: number): Promise<ToolOutcome>;
196
+ protected abstract grepImpl(pattern: string, absRoot: string, opts: GrepOptions): Promise<ToolOutcome>;
197
+ protected abstract listDirImpl(absPath: string, depth: number): Promise<ToolOutcome>;
198
+ protected abstract execBashImpl(command: string): Promise<ToolOutcome>;
199
+ /**
200
+ * Resolve `p` and confirm the planner may touch it. In-workspace paths pass
201
+ * silently; anything else needs one approval, remembered per containing
202
+ * directory so a second file in the same place does not prompt again.
203
+ */
204
+ protected authorizePath(p: string, kind?: 'file' | 'directory'): Promise<{
205
+ ok: true;
206
+ abs: string;
207
+ } | {
208
+ ok: false;
209
+ outcome: ToolOutcome;
210
+ }>;
211
+ readFile(p: string, opts?: ReadFileOpts): Promise<ToolOutcome>;
212
+ readFiles(paths: string[]): Promise<ToolOutcome>;
213
+ glob(pattern: string, opts?: GlobOptions): Promise<ToolOutcome>;
214
+ grep(pattern: string, opts?: GrepOptions): Promise<ToolOutcome>;
215
+ listDir(p: string, depth?: number): Promise<ToolOutcome>;
216
+ /**
217
+ * Definitions first, then a reference tally. Two bounded searches beat one
218
+ * unbounded `grep` because the 100-row budget gets spent on the rows that
219
+ * answer the question.
220
+ */
221
+ findSymbol(symbol: string, opts?: FindSymbolOptions): Promise<ToolOutcome>;
222
+ /**
223
+ * Path confinement for `bash`: an `auto`-tier binary (`cat`, `find`, `rg`, …)
224
+ * is only auto because *reading* is read-only — its arguments can still
225
+ * name a path outside the workspace, which is the exact escape confinement
226
+ * closes for `readFile`/`glob`/`grep`. Each escaping path needs its own
227
+ * approval (scoped to its containing directory); approving one does not
228
+ * approve another, so a single command touching two external dirs prompts
229
+ * once per distinct scope rather than carrying the first grant to the rest.
230
+ */
231
+ private authorizeCommandPaths;
232
+ /**
233
+ * Three tiers (see `commandPolicy.ts`): read-only inspection runs silently,
234
+ * anything else asks once and is remembered, and writes are refused outright
235
+ * because a planner that mutates the workspace has stopped being a planner.
236
+ */
237
+ bash(command: string): Promise<ToolOutcome>;
238
+ }
239
+
240
+ /**
241
+ * Workspace containment, in one place. Both the filesystem policy layer
242
+ * (which may ask the user to approve an escape) and the research-subagent
243
+ * wrapper (which may not, because nothing can prompt on its behalf) need the
244
+ * same answer to "is this path inside the workspace", so neither re-derives it.
245
+ *
246
+ * Symlinks are resolved lexically, not on disk: a symlink inside the workspace
247
+ * pointing outward still reads as inside. Closing that would mean a `realpath`
248
+ * syscall on every path check, and the threat model here is an LLM wandering,
249
+ * not an adversary planting links in a repo the user already trusts enough to
250
+ * point Ordewell at.
251
+ */
252
+ interface ResolvedPath {
253
+ abs: string;
254
+ inside: boolean;
255
+ }
256
+ declare function resolveWithin(root: string, target: string): ResolvedPath;
257
+ /** The directory a grant covers for `target` — its parent for files, itself for directories. */
258
+ declare function grantScopeFor(abs: string, kind: 'file' | 'directory'): string;
259
+
260
+ /**
261
+ * The bridge between "core needs an answer" and "a human somewhere is looking
262
+ * at a UI". Core cannot prompt: the human may be at a browser, a TUI, a CLI
263
+ * stream, or a VS Code webview, and on the web server they are on the far end
264
+ * of a socket. So the Session parks a promise here, announces the request
265
+ * through the normal broadcast seam, and every surface answers through the
266
+ * same `resolve(id, granted)` call.
267
+ *
268
+ * Timeouts are load-bearing rather than defensive: a planner turn that blocks
269
+ * forever on an unanswered prompt would hang the whole research loop with no
270
+ * visible cause. On expiry the request resolves to denied and the model gets a
271
+ * normal, actionable tool result.
272
+ */
273
+ interface PendingApproval {
274
+ id: string;
275
+ request: ApprovalRequest;
276
+ createdAt: string;
277
+ }
278
+ interface PendingApprovalsOptions {
279
+ /** Denies and resolves after this long with no answer. Default 5 minutes. */
280
+ timeoutMs?: number;
281
+ /** Announce a new request to the surfaces. */
282
+ onRequest?: (pending: PendingApproval) => void;
283
+ /** Announce that a request is no longer actionable (answered or expired). */
284
+ onSettled?: (id: string, granted: boolean) => void;
285
+ }
286
+ declare class PendingApprovals {
287
+ private readonly opts;
288
+ private readonly entries;
289
+ constructor(opts?: PendingApprovalsOptions);
290
+ /** Park a request and return the promise the approval policy awaits. */
291
+ ask(request: ApprovalRequest): Promise<boolean>;
292
+ /** Answer one request. Returns false when the id is unknown or already settled. */
293
+ resolve(id: string, granted: boolean): boolean;
294
+ /** Everything still awaiting an answer — replayed to a surface that connects late. */
295
+ outstanding(): PendingApproval[];
296
+ /** Deny everything in flight. Called on abort and on session reset. */
297
+ clear(): void;
298
+ }
299
+
300
+ /**
301
+ * Definition-shaped search patterns for the `find_symbol` tool.
302
+ *
303
+ * Plain `grep` is bad at the question the planner actually asks. Searching for
304
+ * `VerdictEngine` returns every import, call site, comment and string alongside
305
+ * the one line that defines it, and with a hard result cap the definition is
306
+ * often not even in the returned page. `find_symbol` spends the budget on
307
+ * declarations first and reports references as a count plus a sample.
308
+ *
309
+ * This is deliberately regex over a real parser. An LSP-grade answer would mean
310
+ * shipping per-language servers and waiting for them to index — the wrong trade
311
+ * for the half of the architecture whose whole point is being cheap and fast.
312
+ * The planner needs to scope tasks ("defined here, used across ~14 files in 3
313
+ * packages"), not to prove rename-safety; that is the runner's job, and runners
314
+ * have their own tools.
315
+ *
316
+ * Patterns target the Rust regex syntax ripgrep uses: non-capturing groups and
317
+ * `\b` are available, lookaround and backreferences are not.
318
+ */
319
+ interface SymbolLanguage {
320
+ id: string;
321
+ extensions: string[];
322
+ /** Keywords that introduce a declaration in `kw Name` position. */
323
+ keywords: string[];
324
+ }
325
+ declare const SYMBOL_LANGUAGES: SymbolLanguage[];
326
+ declare function languageForId(id: string): SymbolLanguage | undefined;
327
+ /** The `--glob` filter that narrows a search to one language's files. */
328
+ declare function includeGlobFor(language: SymbolLanguage): string;
329
+ /**
330
+ * Build the definition pattern for `symbol`. Three shapes cover essentially
331
+ * every mainstream language:
332
+ *
333
+ * 1. `keyword Name` — declarations across all of them
334
+ * 2. `Name = function|(` — assigned function expressions and arrows
335
+ * 3. `… Name(args) {` — C/Java/Go-style bodies with a return type
336
+ *
337
+ * Passing a `language` narrows shape 1 to that language's keywords, which cuts
338
+ * cross-language noise in polyglot repos.
339
+ */
340
+ declare function definitionPattern(symbol: string, language?: SymbolLanguage): string;
341
+ /** Every mention of the symbol as a whole word — the reference side of the report. */
342
+ declare function referencePattern(symbol: string): string;
343
+
344
+ interface GrepInvocation {
345
+ args: string[];
346
+ /** Rows beyond this are dropped by the caller — see `applyHeadLimit`. */
347
+ headLimit: number;
348
+ }
349
+ /**
350
+ * Build the ripgrep invocation for one grep call.
351
+ *
352
+ * `--max-count` is deliberately absent. It caps matches *per file*, so the old
353
+ * `--max-count 100` on a 300-file hit returned ~30 000 lines, blew the 1 MB
354
+ * exec buffer, and surfaced as `{ success: false, output: '' }` — a silent
355
+ * empty result on exactly the broad searches where the model most needed a
356
+ * signal. The cap belongs at the row level, applied after the fact.
357
+ */
358
+ declare function buildGrepArgs(pattern: string, opts: GrepOptions, root: string): GrepInvocation;
359
+ /** Build the ripgrep invocation that lists files matching a glob. */
360
+ declare function buildGlobArgs(pattern: string, root: string): string[];
361
+ /**
362
+ * The POSIX-grep fallback for machines without ripgrep. Ordering and `--sort`
363
+ * are unavailable.
364
+ *
365
+ * `-P` (PCRE) is required, not optional: patterns built in core — most
366
+ * notably `find_symbol`'s `definitionPattern` — use non-capturing groups and
367
+ * `\b`, which BRE has no syntax for and ERE (`-E`) still cannot express
368
+ * ((?:...) is a PCRE construct). Without `-P`, GNU grep either errors
369
+ * ("Unmatched \{") or, worse, silently treats `(`/`)`/`|` as literal
370
+ * characters and reports a confident empty result. `-P` and `-F` are
371
+ * mutually exclusive, so literal mode skips it.
372
+ */
373
+ declare function buildFallbackGrepArgs(pattern: string, opts: GrepOptions, root: string): string[];
374
+ /**
375
+ * GNU grep's own `--include` glob only ever matches a basename, so it silently
376
+ * matches nothing against an anchored pattern like `subdir/*.txt` (there is no
377
+ * `/` in a basename to match against). The fallback path drops such patterns
378
+ * from the grep invocation and filters matches by relative path here instead.
379
+ */
380
+ declare function filterFallbackByAnchoredInclude(stdout: string, include: string, anchor: string, outputMode?: GrepOptions['outputMode']): string;
381
+ interface CappedRows {
382
+ rows: string[];
383
+ truncated: boolean;
384
+ total: number;
385
+ }
386
+ /** Apply the global row cap and report honestly whether anything was dropped. */
387
+ declare function applyHeadLimit(stdout: string, headLimit: number): CappedRows;
388
+ /**
389
+ * Render capped rows for the model, with paths made workspace-relative and an
390
+ * explicit note when rows were dropped — a silently truncated list reads as a
391
+ * complete answer and the model plans against it.
392
+ */
393
+ declare function formatSearchOutput(capped: CappedRows, root: string, opts: {
394
+ emptyMessage: string;
395
+ hint?: string;
396
+ sep?: string;
397
+ }): string;
398
+
399
+ declare function normalizeGeminiModel(id: string): string;
400
+ declare abstract class BaseConfig implements IConfig {
401
+ abstract aiProvider: AiProvider;
402
+ abstract apiKey: string;
403
+ abstract planningModel: string;
404
+ abstract enabledRunners: string[];
405
+ abstract setProviderModelLists(lists: ProviderModelLists): void;
406
+ get openAiBaseUrl(): string;
407
+ get openAiApiKey(): string;
408
+ get openrouterKey(): string;
409
+ get geminiKey(): string;
410
+ get geminiBaseUrl(): string | undefined;
411
+ get openaiCompatibleBaseUrl(): string;
412
+ get openaiCompatibleApiKey(): string;
413
+ get orchestratorModel(): string;
414
+ get researchSubagentModel(): string;
415
+ get geminiModel(): string;
416
+ get plannerThinkingEffort(): string | undefined;
417
+ get openaiBaseUrl(): string;
418
+ get openaiApiKey(): string;
419
+ get xaiBaseUrl(): string;
420
+ get xaiApiKey(): string;
421
+ get groqBaseUrl(): string;
422
+ get groqApiKey(): string;
423
+ get deepseekBaseUrl(): string;
424
+ get deepseekApiKey(): string;
425
+ get togetherBaseUrl(): string;
426
+ get togetherApiKey(): string;
427
+ get mistralBaseUrl(): string;
428
+ get mistralApiKey(): string;
429
+ get anthropicBaseUrl(): string;
430
+ get anthropicApiKey(): string;
431
+ get fireworksBaseUrl(): string;
432
+ get fireworksApiKey(): string;
433
+ get perplexityBaseUrl(): string;
434
+ get perplexityApiKey(): string;
435
+ get zhipuBaseUrl(): string;
436
+ get zhipuApiKey(): string;
437
+ get kimiBaseUrl(): string;
438
+ get kimiApiKey(): string;
439
+ get cerebrasBaseUrl(): string;
440
+ get cerebrasApiKey(): string;
441
+ get deepinfraBaseUrl(): string;
442
+ get deepinfraApiKey(): string;
443
+ get doubaoBaseUrl(): string;
444
+ get doubaoApiKey(): string;
445
+ get qwenBaseUrl(): string;
446
+ get qwenApiKey(): string;
447
+ get hunyuanBaseUrl(): string;
448
+ get hunyuanApiKey(): string;
449
+ get baichuanBaseUrl(): string;
450
+ get baichuanApiKey(): string;
451
+ get minimaxBaseUrl(): string;
452
+ get minimaxApiKey(): string;
453
+ get yiBaseUrl(): string;
454
+ get yiApiKey(): string;
455
+ get stepfunBaseUrl(): string;
456
+ get stepfunApiKey(): string;
457
+ get siliconflowBaseUrl(): string;
458
+ get siliconflowApiKey(): string;
459
+ get cohereBaseUrl(): string;
460
+ get cohereApiKey(): string;
461
+ get novitaBaseUrl(): string;
462
+ get novitaApiKey(): string;
463
+ getProviderBaseUrl(provider: AiProvider): string;
464
+ getProviderApiKey(provider: AiProvider): string;
465
+ get maxParallelSessions(): number;
466
+ get researchEnabled(): boolean;
467
+ get researchMaxSteps(): number;
468
+ get researchMaxFileSize(): number;
469
+ get planMapEnabled(): boolean;
470
+ get autonomousMode(): boolean;
471
+ get approvalMode(): ApprovalMode;
472
+ get approvalPreApproved(): string[];
473
+ protected static detectProvider(fallback: AiProvider): AiProvider;
474
+ }
475
+
476
+ /**
477
+ * A fully `process.env`-backed IConfig for headless callers (e.g. the CLI's
478
+ * `ordewell models` command) that need provider keys/base URLs but have no
479
+ * editor settings or secret store. All resolution lives in BaseConfig; this
480
+ * subclass only supplies the abstract members from the environment.
481
+ */
482
+ declare class EnvConfig extends BaseConfig {
483
+ get aiProvider(): AiProvider;
484
+ get apiKey(): string;
485
+ get planningModel(): string;
486
+ get enabledRunners(): string[];
487
+ setProviderModelLists(): void;
488
+ }
489
+
490
+ type NotificationAction = {
491
+ label: string;
492
+ value: string;
493
+ };
494
+ interface INotification {
495
+ info(msg: string): void;
496
+ warn(msg: string): void;
497
+ error(msg: string): void;
498
+ confirm(msg: string, options: string[]): Promise<string | undefined>;
499
+ }
500
+
501
+ interface ILogger {
502
+ warn(scope: string, message: string, err?: unknown): void;
503
+ }
504
+ declare class ConsoleLogger implements ILogger {
505
+ warn(scope: string, message: string, err?: unknown): void;
506
+ }
507
+ declare const defaultLogger: ILogger;
508
+
509
+ interface IWebFetcher {
510
+ confirm(url: string): Promise<boolean>;
511
+ fetch(url: string): Promise<ToolOutcome>;
512
+ /**
513
+ * Optional web search. A fetcher without it makes the `web_search` tool
514
+ * report itself unavailable rather than failing the turn — the same
515
+ * degradation `fetch` already uses when no fetcher is wired at all.
516
+ */
517
+ search?(query: string): Promise<ToolOutcome>;
518
+ }
519
+
520
+ interface UserSettings {
521
+ grillMe: {
522
+ enabled: boolean;
523
+ };
524
+ tdd: {
525
+ enabled: boolean;
526
+ };
527
+ prd: {
528
+ enabled: boolean;
529
+ };
530
+ review: {
531
+ enabled: boolean;
532
+ };
533
+ verification: {
534
+ enabled: boolean;
535
+ };
536
+ researchSubagents: {
537
+ enabled: boolean;
538
+ };
539
+ modelAllowlist?: Record<string, string[]>;
540
+ }
541
+ /**
542
+ * Where the user's toggles live. `ORDEWELL_SETTINGS_PATH` overrides it so several
543
+ * Ordewell processes on one machine can hold *different* settings at the same
544
+ * time. Without it the file is a single shared mutable global: the benchmark
545
+ * harness runs parallel lanes that each pin the mode toggles, and because
546
+ * `getAll()` re-reads whenever the mtime moves, a lane needing
547
+ * `researchSubagents: true` would silently plan with `false` the moment another
548
+ * lane pinned its own — turning an A/B of that toggle into a comparison of one
549
+ * condition against itself.
550
+ */
551
+ declare function getSettingsPath(): string;
552
+ declare class SettingsService {
553
+ private filePath;
554
+ private cache;
555
+ private cachedMtimeMs;
556
+ constructor(filePath?: string);
557
+ getAll(): UserSettings;
558
+ private fileMtimeMs;
559
+ getGrillMe(): boolean;
560
+ getTdd(): boolean;
561
+ getPrd(): boolean;
562
+ setGrillMe(enabled: boolean): void;
563
+ setTdd(enabled: boolean): void;
564
+ setPrd(enabled: boolean): void;
565
+ getReview(): boolean;
566
+ setReview(enabled: boolean): void;
567
+ getVerification(): boolean;
568
+ setVerification(enabled: boolean): void;
569
+ getResearchSubagents(): boolean;
570
+ setResearchSubagents(enabled: boolean): void;
571
+ getModelAllowlist(runner: string): string[] | undefined;
572
+ setModelAllowlist(runner: string, ids: string[] | undefined): void;
573
+ private load;
574
+ private persist;
575
+ }
576
+
577
+ /** How the same toggle is named once a host has read it off disk. */
578
+ interface PlannerRuntimeToggles {
579
+ grillMeEnabled: boolean;
580
+ tddEnabled: boolean;
581
+ prdEnabled: boolean;
582
+ reviewEnabled: boolean;
583
+ verificationEnabled: boolean;
584
+ researchSubagentsEnabled: boolean;
585
+ }
586
+ /**
587
+ * The planner-facing mode set for one operation. Replaces the boolean tail that
588
+ * every planner signature used to carry positionally — where a thirteenth
589
+ * parameter was the only place left to put a new toggle.
590
+ */
591
+ interface PlannerModes {
592
+ autonomousDefault: boolean;
593
+ grillMe: boolean;
594
+ prd: boolean;
595
+ review: boolean;
596
+ verification: boolean;
597
+ researchSubagents: boolean;
598
+ }
599
+
600
+ declare abstract class AbstractTerminalSession implements ITerminalSession {
601
+ id: string;
602
+ taskId: string;
603
+ protected exited: boolean;
604
+ protected outputEmitter: EventEmitter<any>;
605
+ protected exitEmitter: EventEmitter<any>;
606
+ constructor(id: string, taskId: string);
607
+ protected baseHandleExit(code: number): void;
608
+ onOutput(callback: (text: string) => void): void;
609
+ onExit(callback: (code: number) => void): void;
610
+ abstract kill(): void;
611
+ abstract getOutput(): string;
612
+ abstract write(text: string): void;
613
+ }
614
+ declare abstract class AbstractRunner<S extends ITerminalSession> implements ITerminalRunner {
615
+ protected sessions: Map<string, S>;
616
+ get activeCount(): number;
617
+ stop(sessionId: string): void;
618
+ stopAll(): void;
619
+ protected registerSession(id: string, session: S): void;
620
+ abstract spawn(opts: {
621
+ taskId: string;
622
+ runner: string;
623
+ prompt: string;
624
+ modelId?: string;
625
+ thinkingEffort?: string;
626
+ mode?: string;
627
+ headless?: boolean;
628
+ cwd: string;
629
+ registry?: RunnerRegistry;
630
+ }): Promise<ITerminalSession>;
631
+ }
632
+
633
+ /**
634
+ * cmd.exe's command-line buffer. A longer line is truncated rather than
635
+ * rejected, which would corrupt a planner's system prompt or a task's prompt
636
+ * mid-sentence and produce a confident answer to half a question — so the
637
+ * batch route refuses instead. See {@link CommandLineTooLongError}.
638
+ */
639
+ declare const CMD_EXE_MAX_COMMAND_LINE = 8191;
640
+ /**
641
+ * CreateProcess's own ceiling, which the native and PowerShell routes are
642
+ * bounded by instead. Windows truncates here too, so the same refusal applies —
643
+ * it is simply four times further away.
644
+ */
645
+ declare const WINDOWS_MAX_COMMAND_LINE = 32767;
646
+ interface LaunchPlan {
647
+ /** The executable handed to `spawn()` or `vscode.window.createTerminal`. */
648
+ file: string;
649
+ /** Arguments for `file`. Pass verbatim when {@link verbatim} is set. */
650
+ args: string[];
651
+ /**
652
+ * Windows batch route only: `args` is already a quoted command line and must
653
+ * not be re-quoted. Maps to `windowsVerbatimArguments` for `spawn`, and to
654
+ * the string form of `shellArgs` for a VS Code terminal.
655
+ */
656
+ verbatim?: boolean;
657
+ }
658
+ /** Test seam: every OS touchpoint is injectable, and production uses the defaults. */
659
+ interface LaunchDeps {
660
+ platform?: NodeJS.Platform;
661
+ /** The PATH executables are looked up on. Defaults to the augmented PATH. */
662
+ resolvePath?: () => Promise<string>;
663
+ /** True when `candidate` names an existing file. */
664
+ exists?: (candidate: string) => boolean;
665
+ /** Absolute path to the Windows command interpreter. */
666
+ comSpec?: () => string;
667
+ /** Absolute path to Windows PowerShell. */
668
+ powerShell?: () => string;
669
+ /** PATHEXT, as the environment reports it. */
670
+ pathExt?: () => string;
671
+ }
672
+ /**
673
+ * Thrown when a command's arguments do not fit the buffer of the only
674
+ * interpreter that can start it. Windows truncates rather than rejecting, and a
675
+ * system prompt cut off mid-sentence makes the planner answer half a question
676
+ * confidently — the silent success this repo refuses — so this is raised
677
+ * instead. `TaskOrchestrator.startTask` catches it and holds the task, so the
678
+ * message is what the user reads: it names the fix, because they cannot infer
679
+ * it from a truncated prompt.
680
+ */
681
+ declare class CommandLineTooLongError extends Error {
682
+ readonly command: string;
683
+ readonly length: number;
684
+ readonly limit: number;
685
+ constructor(command: string, length: number, limit?: number);
686
+ }
687
+ /**
688
+ * Thrown when only a batch shim resolved for a multi-line argument. cmd.exe
689
+ * reads up to the first CR/LF and discards the rest with no error and exit code
690
+ * 0 — quoting does not help — so the agent would get the first paragraph of its
691
+ * prompt without the completion marker instruction, then exit looking successful.
692
+ */
693
+ declare class EmbeddedNewlineError extends Error {
694
+ readonly command: string;
695
+ constructor(command: string);
696
+ }
697
+ /** The verbatim command line cmd.exe receives after `/d /s /c`, before wrapping. */
698
+ declare function windowsCommandLine(file: string, args: string[]): string;
699
+ /**
700
+ * How to start `command` with `args` through `spawn()`, with no shell.
701
+ *
702
+ * POSIX returns its input unchanged — execvp already searches PATH, and adding
703
+ * a resolution step there would be a new way for a working setup to break.
704
+ *
705
+ * Windows resolves the command against PATH × PATHEXT, preferring a native
706
+ * executable (spawned directly) over a batch shim (through cmd.exe) over a
707
+ * PowerShell script shim (through `powershell.exe -File`). A command that
708
+ * resolves to nothing is returned unchanged, so the caller's existing ENOENT —
709
+ * which names what the user typed — is what surfaces rather than a second,
710
+ * vaguer error from here.
711
+ *
712
+ * The tiers are tried in preference order and the first that *fits* wins, with
713
+ * one deliberate exception: an overflowing batch shim does not fall through to
714
+ * PowerShell. Overflow means a very large prompt, which is precisely where
715
+ * `-File` argument fidelity is least worth betting on, and where a clear held
716
+ * task beats a plausibly-mangled one. So capacity does not reorder the tiers —
717
+ * a `.ps1` beside a too-long `.cmd` still raises.
718
+ *
719
+ * A line break does reorder them: cmd.exe cannot carry one at any length, so a
720
+ * `.ps1` beside a `.cmd` wins, and a lone `.cmd` raises.
721
+ *
722
+ * @throws {CommandLineTooLongError} when the selected route's buffer cannot
723
+ * carry the arguments.
724
+ * @throws {EmbeddedNewlineError} when the arguments span lines and only the
725
+ * batch route resolved.
726
+ */
727
+ declare function planDirectLaunch(command: string, args: string[], deps?: LaunchDeps): Promise<LaunchPlan>;
728
+ /**
729
+ * How to start `command` for a surface that hands an executable and arguments
730
+ * to a terminal — the VS Code runner today, a Windows TUI later.
731
+ *
732
+ * On POSIX this is the login shell, unchanged: `bash -lc` runs the user's
733
+ * profile, which is how nvm/volta/asdf-managed runner binaries resolve at all.
734
+ * Windows has no login-shell equivalent (its PATH comes from the registry and
735
+ * is already inherited), so it takes the direct route instead. That is not just
736
+ * a simplification: it means the runner's own exit code is the terminal's exit
737
+ * code, rather than a `$LASTEXITCODE` that PowerShell propagates unreliably —
738
+ * and the exit code is half of what {@link VerdictEngine} judges a task on.
739
+ */
740
+ declare function planShellLaunch(command: string, args: string[], deps?: LaunchDeps): Promise<LaunchPlan>;
741
+
742
+ type SpawnFn = (command: string, args: string[], options: {
743
+ env: NodeJS.ProcessEnv;
744
+ stdio: ['pipe', 'pipe', 'pipe'];
745
+ cwd: string;
746
+ /** Set by the Windows batch route, where `args` is already a quoted command line. */
747
+ windowsVerbatimArguments?: boolean;
748
+ }) => ChildProcess;
749
+ /** Test seam: every OS touchpoint is injectable; production uses the defaults. */
750
+ interface HeadlessRunnerDeps {
751
+ spawnImpl?: SpawnFn;
752
+ hasScriptCmd?: () => boolean;
753
+ resolvePath?: () => Promise<string>;
754
+ /**
755
+ * Overrides for executable resolution ({@link planDirectLaunch}). Only the
756
+ * Windows branch consults them, so a POSIX test never needs to pass anything.
757
+ */
758
+ launchDeps?: LaunchDeps;
759
+ }
760
+ declare class HeadlessSession extends AbstractTerminalSession {
761
+ private spawnImpl;
762
+ private process;
763
+ private outputBuffer;
764
+ constructor(id: string, taskId: string, spawnImpl: SpawnFn);
765
+ start(launch: LaunchPlan, cwd: string, resolvedPath: string, env?: Record<string, string>): void;
766
+ kill(): void;
767
+ getOutput(): string;
768
+ write(text: string): void;
769
+ }
770
+ declare class HeadlessRunner extends AbstractRunner<HeadlessSession> {
771
+ private spawnImpl;
772
+ private hasScriptCmd;
773
+ private resolvePath;
774
+ private launchDeps;
775
+ constructor(deps?: HeadlessRunnerDeps);
776
+ spawn(opts: {
777
+ taskId: string;
778
+ runner: string;
779
+ prompt: string;
780
+ modelId?: string;
781
+ thinkingEffort?: string;
782
+ modelVariants?: string[];
783
+ mode?: string;
784
+ headless?: boolean;
785
+ cwd: string;
786
+ registry?: RunnerRegistry;
787
+ }): Promise<ITerminalSession>;
788
+ }
789
+
790
+ /**
791
+ * The harness-planner transport contract (ADR-0009).
792
+ *
793
+ * One adapter per coding agent, each speaking that agent's own programmatic
794
+ * protocol and normalizing it to the event union below. Everything above this
795
+ * line — reply classification, the repair loop, plan validation, the four
796
+ * surfaces — is already provider-agnostic, so an adapter is the entire cost of
797
+ * teaching Ordewell to plan with another agent.
798
+ */
799
+ /**
800
+ * One normalized event from a running agent turn. Deliberately smaller than
801
+ * any single agent's native protocol: this is the intersection Ordewell can act
802
+ * on, not a lossless re-encoding. Event fidelity differs by agent — `thinking`
803
+ * is rich on Claude Code and absent elsewhere — so consumers must tolerate a
804
+ * turn that emits nothing but `assistant_text` and `turn_end`.
805
+ */
806
+ type AgentEvent =
807
+ /** A chunk of the assistant's reply. Concatenated in order to form the turn's text. */
808
+ {
809
+ type: 'assistant_text';
810
+ text: string;
811
+ }
812
+ /** Reasoning the agent chose to expose. Never contributes to the reply text. */
813
+ | {
814
+ type: 'thinking';
815
+ text: string;
816
+ } | {
817
+ type: 'tool_call';
818
+ id: string;
819
+ name: string;
820
+ args: Record<string, unknown>;
821
+ } | {
822
+ type: 'tool_result';
823
+ id: string;
824
+ name: string;
825
+ output: string;
826
+ success: boolean;
827
+ }
828
+ /**
829
+ * The agent asked to do something its read-only mode does not cover. Always
830
+ * auto-denied (T1) — a planner that can mutate is not a planner. The adapter
831
+ * is responsible for answering the agent so the turn does not hang.
832
+ */
833
+ | {
834
+ type: 'permission_request';
835
+ id: string;
836
+ name: string;
837
+ detail: string;
838
+ }
839
+ /** The agent finished its turn and is waiting for the next user message. */
840
+ | {
841
+ type: 'turn_end';
842
+ }
843
+ /** The turn failed. Carries the agent's own words — never a Ordewell paraphrase. */
844
+ | {
845
+ type: 'error';
846
+ message: string;
847
+ };
848
+ interface AgentStartOptions {
849
+ /** Workspace root. The agent explores from here and, in read-only mode, cannot leave it. */
850
+ cwd: string;
851
+ /** The planner system prompt, in its harness variant. */
852
+ systemPrompt: string;
853
+ /** Model id from the runner's own discovery catalog. Omitted means the agent's default. */
854
+ model?: string;
855
+ /** Variant / reasoning effort id from that model's `variants` list. */
856
+ effort?: string;
857
+ /**
858
+ * The agent's own session id from a previous run. A hint only: Ordewell's
859
+ * transcript is the source of truth (T4), so a failed resume degrades to a
860
+ * fresh session seeded from the stored history rather than an error.
861
+ */
862
+ resumeSessionId?: string;
863
+ }
864
+ interface AgentAdapter {
865
+ /** The runner id this adapter drives — `claude-code`, `codex`, `opencode`. */
866
+ readonly agentId: string;
867
+ /** Spawn the agent in its read-only mode and get it ready to receive messages. */
868
+ start(opts: AgentStartOptions): Promise<void>;
869
+ /**
870
+ * Send one user message and stream the turn's events until it ends. Resolves
871
+ * when the agent yields the floor; rejects only when the transport itself
872
+ * failed in a way no `error` event could describe.
873
+ */
874
+ send(message: string, onEvent: (event: AgentEvent) => void, signal?: AbortSignal): Promise<void>;
875
+ /** The agent's native session id once it has announced one. Resumption hint only. */
876
+ nativeSessionId(): string | null;
877
+ /** Kill the process and release its resources. Idempotent. */
878
+ dispose(): void;
879
+ }
880
+ /**
881
+ * The single injected boundary between Ordewell and the operating system —
882
+ * the same pattern `HeadlessRunnerDeps` uses for task execution. Tests feed
883
+ * recorded agent output through `spawn` (and, for HTTP-transport agents,
884
+ * `fetch`) so one test exercises adapter parsing, event mapping, reply
885
+ * classification and the repair loop as a single observable behavior.
886
+ */
887
+ interface AgentProcessDeps {
888
+ spawn: SpawnFn;
889
+ fetch: typeof globalThis.fetch;
890
+ /** Resolves the PATH agents are spawned under. Defaults to the augmented PATH. */
891
+ resolvePath?: () => Promise<string>;
892
+ /** Host platform. Defaults to the real one; injected so OS-specific behavior is testable anywhere. */
893
+ platform?: NodeJS.Platform;
894
+ }
895
+ /** Builds the adapter for one runner id, or null when that runner cannot plan. */
896
+ type AgentAdapterFactory = (runner: string, deps: AgentProcessDeps) => AgentAdapter | null;
897
+ /**
898
+ * Split a stream of chunks into complete lines. Every agent transport here is
899
+ * newline-delimited JSON of some shape, and a chunk boundary lands mid-object
900
+ * often enough that parsing per-chunk silently drops events.
901
+ */
902
+ declare class LineBuffer {
903
+ private buffer;
904
+ push(chunk: string, onLine: (line: string) => void): void;
905
+ /** Anything left unterminated when the stream closed. */
906
+ flush(): string;
907
+ }
908
+
909
+ interface CliAgentAiServiceDeps extends Partial<AgentProcessDeps> {
910
+ /** Overrides adapter construction. Tests supply a fake agent; production picks by runner id. */
911
+ createAdapter?: (runner: string, deps: AgentProcessDeps) => AgentAdapter | null;
912
+ /** Workspace root the agent explores. Defaults to the host process's cwd. */
913
+ workspaceRoot?: () => string;
914
+ }
915
+ /**
916
+ * A coding agent driven as Ordewell's planner (ADR-0009).
917
+ *
918
+ * The second transport behind `IAiService`, sitting beside `OpenAiService` and
919
+ * `GeminiService`. It deliberately does **not** extend {@link BaseAiService}:
920
+ * that class's body is Ordewell executing research tools on a model's behalf,
921
+ * which is precisely the part a coding agent replaces. What it reuses instead
922
+ * is everything above the transport — `classifyPlannerReply`, the bounded
923
+ * corrective re-emit loop, `parsePlanJson`, the `ResearchProgress` events the
924
+ * four surfaces already render. That is why this backend reaches VS Code, the
925
+ * web UI, the CLI and the TUI without any of them learning a coding agent is
926
+ * on the other end.
927
+ */
928
+ declare class CliAgentAiService implements IAiService {
929
+ private config;
930
+ private readonly runner;
931
+ private readonly processDeps;
932
+ private readonly makeAdapter;
933
+ private readonly workspaceRoot;
934
+ private adapter;
935
+ /**
936
+ * The agent's own session id, kept across a process death so the next turn
937
+ * can resume warm context instead of re-reading the repository. Cleared at
938
+ * every session boundary — nothing from one goal may reach the next.
939
+ */
940
+ private lastNativeSessionId;
941
+ private conversation;
942
+ private activeAbort;
943
+ constructor(config: IConfig, deps?: CliAgentAiServiceDeps);
944
+ hasActiveConversation(): boolean;
945
+ /**
946
+ * False when the running agent process was spawned under a model/effort the
947
+ * user has since changed in the picker. Unlike a vendor backend, the model
948
+ * is a spawn-time argument to the agent CLI (`--model`), not a per-request
949
+ * field — `continueConversation` sends the next turn to whatever process is
950
+ * already running, so a plain config read here would silently keep planning
951
+ * on the old model. No conversation yet is vacuously "current".
952
+ */
953
+ conversationMatchesConfig(): boolean;
954
+ /** The agent's own session id, a resumption hint only — Ordewell's transcript is authoritative (T4). */
955
+ nativeSessionId(): string | null;
956
+ reset(): void;
957
+ startConversation(req: ConversationRequest): Promise<ConversationTurn>;
958
+ continueConversation(userMessage: string, onProgress: (progress: ResearchProgress) => void, signal?: AbortSignal): Promise<ConversationTurn>;
959
+ private openingMessage;
960
+ /**
961
+ * Drive one user message to a settled planner turn. Same shape as the API
962
+ * backend's conversation loop minus the tool rounds — those belong to the
963
+ * agent now — and with the same three policies layered on top: the PRD gate,
964
+ * the empty-reply nudge, and a bounded corrective re-emit for botched JSON.
965
+ */
966
+ private runConversation;
967
+ /**
968
+ * Run one agent turn: stream its events into `ResearchProgress`, collect the
969
+ * reply text, and return the research steps produced. Tool calls are matched
970
+ * to their results by the agent's own call id — matching by tool name puts
971
+ * one file's body on another file's row the moment an agent runs two reads
972
+ * at once, which all three of these do routinely.
973
+ */
974
+ private runTurn;
975
+ private startAdapter;
976
+ /**
977
+ * The live agent process, restarted from its own session id if it died
978
+ * between turns. Resume is a hint: when it fails, the caller's next
979
+ * `startConversation` reseeds from Ordewell's transcript, which is the same
980
+ * degradation `restoreChat` already performs on every surface.
981
+ */
982
+ private ensureAdapter;
983
+ private startAbortScope;
984
+ private plannerModel;
985
+ /**
986
+ * A single agent session that answers one prompt and exits. Used by every
987
+ * non-conversational entry point; the plan is parsed from the reply text by
988
+ * the same extractor the conversational path uses.
989
+ */
990
+ private oneShot;
991
+ researchAndPlan(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
992
+ tasks: Task[];
993
+ researchLog: ResearchLogEntry[];
994
+ researchResults: string;
995
+ }>;
996
+ generatePlanDirect(userDescription: string, runners?: RunnerId[], modelsByRunner?: Partial<Record<RunnerId, DiscoveredModel[]>>, onToken?: (token: string) => void, _fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<Task[]>;
997
+ modifyPlan(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onProgress?: (progress: ResearchProgress) => void, _fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
998
+ tasks: Task[];
999
+ }>;
1000
+ sendPlanningPrompt(prompt: string, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): Promise<Task[]>;
1001
+ }
1002
+
1003
+ /**
1004
+ * Everything the conversation loop needs to start planning (ADR-0002).
1005
+ * The AI service keeps the tool-use message history internally across turns;
1006
+ * the surfaces only exchange user/assistant messages with it.
1007
+ */
1008
+ interface ConversationRequest {
1009
+ goal: string;
1010
+ runners: RunnerId[];
1011
+ modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>;
1012
+ fs: IFileSystem;
1013
+ onProgress: (progress: ResearchProgress) => void;
1014
+ fetcher?: IWebFetcher;
1015
+ runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
1016
+ autonomousDefault?: boolean;
1017
+ grillMeEnabled?: boolean;
1018
+ prdEnabled?: boolean;
1019
+ reviewEnabled?: boolean;
1020
+ verificationEnabled?: boolean;
1021
+ /** Declare the spawn_research_agent tool to the planner (issue #34, default off). */
1022
+ researchSubagentsEnabled?: boolean;
1023
+ signal?: AbortSignal;
1024
+ /**
1025
+ * Persisted dialogue to seed a resumed conversation (session reload). The
1026
+ * turns are injected into the message history verbatim — no LLM call happens
1027
+ * for them; the first call is the turn opened by `initialMessage`.
1028
+ */
1029
+ priorHistory?: ConversationMessage[];
1030
+ /**
1031
+ * The user message that opens this turn. Defaults to `goal` — set it when
1032
+ * resuming so the system prompt keeps the original goal while the turn
1033
+ * carries the user's new message.
1034
+ */
1035
+ initialMessage?: string;
1036
+ }
1037
+ /**
1038
+ * One planner turn's outcome. The planner talks to the user (`message`),
1039
+ * commits a full plan (`plan`, a `{tasks:[...]}` JSON object), or emits
1040
+ * targeted task edits (`task_ops`, a `{taskOps:[...]}` JSON object). The
1041
+ * Session validates and applies task ops atomically; the AI service only
1042
+ * parses them.
1043
+ */
1044
+ type ConversationTurn = {
1045
+ kind: 'message';
1046
+ text: string;
1047
+ researchLog: ResearchLogEntry[];
1048
+ } | {
1049
+ kind: 'plan';
1050
+ tasks: Task[];
1051
+ text: string;
1052
+ researchLog: ResearchLogEntry[];
1053
+ } | {
1054
+ kind: 'task_ops';
1055
+ ops: TaskOp[];
1056
+ text: string;
1057
+ researchLog: ResearchLogEntry[];
1058
+ };
1059
+ interface IAiService {
1060
+ /**
1061
+ * Begin the planner conversation: collect workspace context, run the
1062
+ * research/tool loop, and return the first planner turn. The service
1063
+ * retains the full API message history for subsequent
1064
+ * {@link continueConversation} calls.
1065
+ */
1066
+ startConversation(req: ConversationRequest): Promise<ConversationTurn>;
1067
+ /**
1068
+ * Feed the user's reply into the active conversation and return the next
1069
+ * planner turn. Throws if no conversation is active.
1070
+ */
1071
+ continueConversation(userMessage: string, onProgress: (progress: ResearchProgress) => void, signal?: AbortSignal): Promise<ConversationTurn>;
1072
+ /**
1073
+ * Whether a planner conversation is currently held in memory. A branching
1074
+ * predicate only — never a test for whether there is anything to release.
1075
+ * The harness backend holds an OS process that outlives its conversation, so
1076
+ * guarding {@link reset} with this leaks an agent process; call `reset`
1077
+ * unconditionally instead.
1078
+ */
1079
+ hasActiveConversation(): boolean;
1080
+ /**
1081
+ * Whether the active conversation (if any) still matches the planner
1082
+ * model/effort currently configured. Optional, and true when absent: a
1083
+ * vendor backend re-reads its model from config on every turn, so there is
1084
+ * nothing to drift. A harness planner (ADR-0009) is the exception — its
1085
+ * model is baked into the agent process at spawn — so only
1086
+ * {@link CliAgentAiService} answers this for real. False tells the caller
1087
+ * the same thing an absent conversation would: don't call
1088
+ * `continueConversation`, restart instead so the new model takes effect.
1089
+ */
1090
+ conversationMatchesConfig?(): boolean;
1091
+ /**
1092
+ * One-shot research + plan for non-conversational surfaces (CLI `plan --goal`,
1093
+ * web REST). Never asks questions — it plans with what it can find.
1094
+ */
1095
+ researchAndPlan(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
1096
+ tasks: Task[];
1097
+ researchLog: ResearchLogEntry[];
1098
+ researchResults: string;
1099
+ }>;
1100
+ generatePlanDirect(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onToken?: (token: string) => void, fs?: IFileSystem, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<Task[]>;
1101
+ modifyPlan(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onProgress?: (progress: ResearchProgress) => void, fs?: IFileSystem, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
1102
+ tasks: Task[];
1103
+ }>;
1104
+ /**
1105
+ * Release everything this service holds: the conversation, any in-flight
1106
+ * turn, and — on the harness backend — the agent process itself. Idempotent
1107
+ * and cheap in every implementation, so callers never gate it.
1108
+ */
1109
+ reset(): void;
1110
+ sendPlanningPrompt(prompt: string, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): Promise<Task[]>;
1111
+ }
1112
+ /**
1113
+ * The single branch point between planner transports. Two of them talk HTTP to
1114
+ * an LLM vendor; the third (ADR-0009) drives a coding agent already installed
1115
+ * on the machine. Everything downstream — the plan contract, the research-step
1116
+ * stream, the four surfaces — is shared.
1117
+ */
1118
+ declare function createAiService(config: IConfig, deps?: CliAgentAiServiceDeps): IAiService;
1119
+
1120
+ interface ToolCall {
1121
+ name: string;
1122
+ args: Record<string, unknown>;
1123
+ id?: string;
1124
+ }
1125
+ interface ToolResult {
1126
+ name: string;
1127
+ output: string;
1128
+ truncated: boolean;
1129
+ totalChars: number;
1130
+ id?: string;
1131
+ }
1132
+ interface ResearchTurn {
1133
+ text: string;
1134
+ toolCalls: ToolCall[];
1135
+ hasToolCalls: boolean;
1136
+ /** Reasoning/chain-of-thought captured separately from `text` so it never pollutes plan JSON. */
1137
+ reasoning?: string;
1138
+ /** Provider finish reason, normalized: 'length' means the output-token limit cut the reply mid-stream. */
1139
+ finishReason?: string;
1140
+ /** Exact prompt tokens this turn consumed, when the provider reports usage — drives proactive compaction. */
1141
+ promptTokens?: number;
1142
+ }
1143
+ interface ResearchChat {
1144
+ sendMessage(text: string, signal?: AbortSignal): Promise<ResearchTurn>;
1145
+ sendToolResults(results: ToolResult[], signal?: AbortSignal): Promise<ResearchTurn>;
1146
+ /**
1147
+ * Optional: prune bulky raw tool outputs from the history in place (subagent
1148
+ * digests are kept) so a follow-up emission has context to spend on output.
1149
+ * Returns the number of characters removed. Providers whose SDK owns the
1150
+ * history (Gemini) may not support it.
1151
+ */
1152
+ compactHistory?(): number;
1153
+ }
1154
+ /** Everything one conversation turn needs beyond the user message itself. */
1155
+ interface ConversationTurnContext {
1156
+ chat: ResearchChat;
1157
+ fs: IFileSystem;
1158
+ runners: RunnerId[];
1159
+ runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
1160
+ autonomousDefault?: boolean;
1161
+ fetcher?: IWebFetcher;
1162
+ /** PRD toggle: a plan must not commit before a PRD block has appeared. */
1163
+ prdEnabled?: boolean;
1164
+ /** Set once any turn carries an ORDEWELL_PRD block; gates the missing-PRD nudge. */
1165
+ prdCaptured?: boolean;
1166
+ /** The missing-PRD corrective nudge is sent at most once per conversation. */
1167
+ prdNudgeSent?: boolean;
1168
+ }
1169
+ declare abstract class BaseAiService {
1170
+ protected config: IConfig;
1171
+ protected conversation: {
1172
+ ctx: ConversationTurnContext;
1173
+ setProgress: (fn: (p: ResearchProgress) => void) => void;
1174
+ } | null;
1175
+ protected activeAbort: AbortController | null;
1176
+ /** researchSubagents toggle (issue #34): set per operation from the live settings snapshot; off means bit-for-bit sequential behavior. */
1177
+ protected researchSubagentsEnabled: boolean;
1178
+ constructor(config: IConfig);
1179
+ abstract reset(): void;
1180
+ abstract ensureInit(): void;
1181
+ hasActiveConversation(): boolean;
1182
+ protected startAbortScope(callerSignal?: AbortSignal): AbortSignal | undefined;
1183
+ protected stopAbortScope(): void;
1184
+ sendPlanningPrompt(prompt: string, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): Promise<Task[]>;
1185
+ continueConversation(userMessage: string, onProgress: (progress: ResearchProgress) => void, signal?: AbortSignal): Promise<ConversationTurn>;
1186
+ protected abstract streamPlanText(prompt: string, repairHint: string | undefined, onToken: (token: string) => void, onReasoning?: (token: string) => void, signal?: AbortSignal): Promise<string>;
1187
+ /**
1188
+ * Build a fresh chat for one research subagent (own history, subagent system
1189
+ * prompt, cheap model). Null means the provider does not support subagents —
1190
+ * the spawn tool then degrades to a steering message. `onReasoning` streams
1191
+ * live reasoning deltas on models that expose them, same as the top-level loop.
1192
+ */
1193
+ protected createSubagentChat(_onReasoning?: (delta: string) => void): ResearchChat | null;
1194
+ /**
1195
+ * One spawn_research_agent tool call, executed at the service layer (not in
1196
+ * executeTool — that would cycle the imports). Every failure path returns a
1197
+ * steering/failed result and the loop continues sequentially: a subagent can
1198
+ * never fail a plan or a turn.
1199
+ */
1200
+ private executeSpawnAgent;
1201
+ /** Collect project context for the planning phase. Shared with the harness backend. */
1202
+ protected static collectResearchContext(fs: IFileSystem, runners: RunnerId[]): Promise<string>;
1203
+ /**
1204
+ * Execute one round of tool calls and report each through onProgress.
1205
+ * Returns the results to feed back plus the log entries produced.
1206
+ */
1207
+ private executeToolCalls;
1208
+ /**
1209
+ * Countdown appended to the last tool result of the final few rounds, so the
1210
+ * model lands the turn on its own terms instead of being cut off exactly at
1211
+ * the budget boundary mid-exploration.
1212
+ */
1213
+ private static appendBudgetCountdown;
1214
+ /**
1215
+ * Run one planner conversation turn (ADR-0002): send the message, satisfy
1216
+ * tool calls until the model answers in prose or JSON, then classify the
1217
+ * result. The model decides transitions — there are no sentinels, no
1218
+ * question tags, and no correction nags. A turn whose final text parses as
1219
+ * a `{tasks:[...]}` object commits the plan; anything else is a message to
1220
+ * the user.
1221
+ */
1222
+ protected runConversationTurn(ctx: ConversationTurnContext, message: string, onProgress: (progress: ResearchProgress) => void, signal?: AbortSignal): Promise<ConversationTurn>;
1223
+ /**
1224
+ * Run the LLM tool-calling research loop for one-shot planning. Returns
1225
+ * parsed tasks if a plan was emitted mid-loop, or null if the loop
1226
+ * exhausted without a valid plan.
1227
+ */
1228
+ protected runResearchLoop(chat: ResearchChat, firstMessage: string, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, runners: RunnerId[], initialLog?: ResearchLogEntry[], fetcher?: IWebFetcher, userGoal?: string, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean, signal?: AbortSignal): Promise<{
1229
+ tasks: Task[] | null;
1230
+ researchLog: ResearchLogEntry[];
1231
+ researchResults: string;
1232
+ }>;
1233
+ protected generatePlanFallback(userDescription: string, contextStr: string, researchResults: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], onProgress: (progress: ResearchProgress) => void, researchLog?: ResearchLogEntry[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean, signal?: AbortSignal, modes?: PlannerModes): Promise<{
1234
+ tasks: Task[];
1235
+ researchLog: ResearchLogEntry[];
1236
+ }>;
1237
+ }
1238
+
1239
+ declare class GeminiService extends BaseAiService implements IAiService {
1240
+ private genAI;
1241
+ private model;
1242
+ constructor(config: IConfig);
1243
+ private init;
1244
+ private getPlanningModel;
1245
+ ensureInit(): void;
1246
+ reset(): void;
1247
+ startConversation(req: ConversationRequest): Promise<ConversationTurn>;
1248
+ protected streamPlanText(prompt: string, repairHint: string | undefined, onToken: (token: string) => void, onReasoning?: (token: string) => void, _signal?: AbortSignal): Promise<string>;
1249
+ /**
1250
+ * Drain a Gemini content stream, routing "thought" parts to the reasoning channel
1251
+ * and answer text to the parser. `chunk.text()` throws when a chunk is reasoning-only,
1252
+ * so parts are inspected directly.
1253
+ */
1254
+ private consumePlanStream;
1255
+ /** Gemini-specific stream over raw Parts for generatePlanDirect / modifyPlan. */
1256
+ private streamPlanTextWithParts;
1257
+ researchAndPlan(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
1258
+ tasks: Task[];
1259
+ researchLog: ResearchLogEntry[];
1260
+ researchResults: string;
1261
+ }>;
1262
+ generatePlanDirect(userDescription: string, runners?: RunnerId[], modelsByRunner?: Partial<Record<RunnerId, DiscoveredModel[]>>, onToken?: (token: string) => void, fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, _signal?: AbortSignal): Promise<Task[]>;
1263
+ modifyPlan(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onProgress?: (progress: ResearchProgress) => void, fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, _signal?: AbortSignal): Promise<{
1264
+ tasks: Task[];
1265
+ }>;
1266
+ }
1267
+
1268
+ declare class OpenAiService extends BaseAiService implements IAiService {
1269
+ private client;
1270
+ constructor(config: IConfig);
1271
+ private getClient;
1272
+ ensureInit(): void;
1273
+ private requireModel;
1274
+ reset(): void;
1275
+ protected streamPlanText(prompt: string, repairHint: string | undefined, onToken: (token: string) => void, onReasoning?: (token: string) => void, signal?: AbortSignal): Promise<string>;
1276
+ /** A research subagent: fresh history, digest contract, cheap model, read-only tools. */
1277
+ protected createSubagentChat(onReasoning?: (delta: string) => void): ResearchChat | null;
1278
+ startConversation(req: ConversationRequest): Promise<ConversationTurn>;
1279
+ researchAndPlan(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
1280
+ tasks: Task[];
1281
+ researchLog: ResearchLogEntry[];
1282
+ researchResults: string;
1283
+ }>;
1284
+ generatePlanDirect(userDescription: string, runners?: RunnerId[], modelsByRunner?: Partial<Record<RunnerId, DiscoveredModel[]>>, onToken?: (token: string) => void, fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<Task[]>;
1285
+ modifyPlan(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onProgress?: (progress: ResearchProgress) => void, fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
1286
+ tasks: Task[];
1287
+ }>;
1288
+ }
1289
+
1290
+ /**
1291
+ * The deep module owning all plan-shaped state. Holds `planTasks` (the ordered
1292
+ * tree the user edits), the flattened `allTasks` view, the `taskMap` index, and
1293
+ * the `completedTasks`/`failedTasks` sets the scheduler reads. The orchestrator
1294
+ * calls `markCompleted`/`markFailed`/`markInProgress`/`retry` to update task
1295
+ * status — it never mutates task state directly.
1296
+ *
1297
+ * `rebuild` is the internal seam that keeps the flat views in sync with the
1298
+ * tree. Structural removals (remove/merge/split) additionally prune
1299
+ * `completedTasks`/`failedTasks` for ids that no longer exist; `removeFromActive`
1300
+ * deliberately does not, so a completed task that leaves the active list still
1301
+ * satisfies its dependents' dependency checks.
1302
+ *
1303
+ * `planRunners` lives here because it's part of the plan's identity (the runner
1304
+ * set, carried on the plan). `validateAssignedRunners` is pure store logic.
1305
+ */
1306
+ declare class PlanStore {
1307
+ private _planTasks;
1308
+ private _allTasks;
1309
+ private _taskMap;
1310
+ private _completedTasks;
1311
+ private _failedTasks;
1312
+ private _planRunners;
1313
+ private _onMutate;
1314
+ private _executionLog;
1315
+ /** Hook called after every structural mutation (add/remove/update/merge/split/load). */
1316
+ set onMutate(cb: (() => void) | null);
1317
+ get planTasks(): Task[];
1318
+ get allTasks(): Task[];
1319
+ get planRunners(): RunnerId[];
1320
+ get completedCount(): number;
1321
+ get failedCount(): number;
1322
+ isAllComplete(): boolean;
1323
+ isAnyFailed(): boolean;
1324
+ isCompleted(id: string): boolean;
1325
+ isFailed(id: string): boolean;
1326
+ get(taskId: string): Task | undefined;
1327
+ getExecutionLog(): TaskSnapshot[];
1328
+ appendToLog(snapshot: TaskSnapshot): void;
1329
+ /**
1330
+ * Drop a task's archived snapshot. Un-marking a completion has to erase the
1331
+ * "finished" record too — dependent tasks are prompted from the log, so a
1332
+ * left-behind snapshot would keep feeding them a result that no longer exists.
1333
+ */
1334
+ removeFromLog(taskId: string): void;
1335
+ removeFromActive(taskId: string): void;
1336
+ clearLog(): void;
1337
+ private notifyMutate;
1338
+ load(tasks: Task[], runners: RunnerId[]): void;
1339
+ add(partial: Partial<Task>): Task;
1340
+ remove(taskId: string): void;
1341
+ update(taskId: string, changes: Partial<Task>): Task | undefined;
1342
+ mergeMultiple(taskIds: string[]): Task;
1343
+ merge(taskIdA: string, taskIdB: string): Task;
1344
+ split(taskId: string, newTaskSpecs: Partial<Task>[]): Task[];
1345
+ /**
1346
+ * Run preparation as one named op: flip every AI task to 'approved'. By
1347
+ * default completed tasks keep their status so a reloaded half-finished plan
1348
+ * resumes the remainder instead of re-running work that already succeeded;
1349
+ * `preserveCompleted: false` is for a freshly committed plan, where
1350
+ * everything starts over.
1351
+ */
1352
+ resetForRun(opts?: {
1353
+ preserveCompleted?: boolean;
1354
+ }): void;
1355
+ markCompleted(id: string): void;
1356
+ markFailed(id: string): void;
1357
+ markInProgress(id: string): void;
1358
+ markAwaitingUser(id: string): void;
1359
+ markPending(id: string): void;
1360
+ retry(id: string): void;
1361
+ blockDependents(id: string): void;
1362
+ unblockDependents(id: string): void;
1363
+ setTaskVerdict(id: string, verdict: Task['verdict']): void;
1364
+ setTaskOutputSummary(id: string, summary: Task['outputSummary']): void;
1365
+ getPlanVisualization(): {
1366
+ tasks: {
1367
+ id: string;
1368
+ title: string;
1369
+ dependencies: string[];
1370
+ parallelGroups: number[][];
1371
+ }[];
1372
+ parallelGroups: number[][];
1373
+ };
1374
+ /**
1375
+ * Widen the plan's runner set. Retargeting a task onto a runner the plan has
1376
+ * not used before has to land here as well as on the plan state, or
1377
+ * {@link resolveTaskRunner} reads the new runner as foreign and spawns the
1378
+ * plan's first one instead.
1379
+ */
1380
+ admitRunner(runner: RunnerId): void;
1381
+ resolveTaskRunner(task: Task): RunnerId;
1382
+ private rebuild;
1383
+ /**
1384
+ * Drop completed/failed ids that no longer exist in the plan. Called by the
1385
+ * structural removals (remove/merge/split) — but NOT by removeFromActive,
1386
+ * where a completed task leaves the active list yet must still satisfy its
1387
+ * dependents' dependency checks.
1388
+ */
1389
+ private pruneTerminalSets;
1390
+ private validateAssignedRunners;
1391
+ }
1392
+
1393
+ /**
1394
+ * The one notification channel out of the orchestrator. Everything that used
1395
+ * to travel over separate callbacks (onRefresh, onQueueReady) is an observer
1396
+ * event; the Session subscribes once and turns these into SessionMessages.
1397
+ */
1398
+ interface OrchestratorObserver {
1399
+ /** Any task-shaped state changed (store mutation, checkpoint, retry, …). */
1400
+ onTaskChanged?(): void;
1401
+ onTick?(): void;
1402
+ onExecutionComplete?(): void;
1403
+ /** Queued user messages are ready to be processed by the planner. */
1404
+ onQueueReady?(): void;
1405
+ onReviewNeeded?(data: {
1406
+ tasks: Task[];
1407
+ planRunners: RunnerId[];
1408
+ }): void;
1409
+ onReviewApproved?(data: {
1410
+ tasks: Task[];
1411
+ }): void;
1412
+ onCheckpoint?(data: {
1413
+ taskId: string;
1414
+ taskTitle: string;
1415
+ summary: string;
1416
+ }): void;
1417
+ }
1418
+ /**
1419
+ * The pure scheduler. Owns execution state (`running`, `planStatus`,
1420
+ * `reviewApproved`, `activeTaskSessions`, `messageQueue`) and the verifier.
1421
+ * All task-shaped state — the plan tree, the flat index, the completed set
1422
+ * — lives in {@link PlanStore}, injected at construction. The orchestrator
1423
+ * calls `store.markCompleted(id)` / `store.markFailed(id)` instead of mutating
1424
+ * task state directly. A task completes only after the runner emits its
1425
+ * per-task completion marker; process exit without that evidence is a visible
1426
+ * failure and does not unblock dependent work.
1427
+ */
1428
+ declare class TaskOrchestrator {
1429
+ private config;
1430
+ private notifications;
1431
+ private terminalRunner;
1432
+ private store;
1433
+ private activeTaskSessions;
1434
+ private startingTaskIds;
1435
+ private verifier;
1436
+ private running;
1437
+ private planStatus;
1438
+ private messageQueue;
1439
+ private reviewApproved;
1440
+ private retryCounts;
1441
+ /**
1442
+ * Tasks pulled out of auto-scheduling (user-cancelled or failed to spawn).
1443
+ * They stay 'pending' — "not executed" — but the scheduler skips them until
1444
+ * the user retries or force-starts, which would otherwise loop forever on a
1445
+ * task whose spawn always throws.
1446
+ */
1447
+ private onHold;
1448
+ private registry;
1449
+ private workspaceRootFn;
1450
+ private observers;
1451
+ private tddEnabled;
1452
+ constructor(config: IConfig, notifications: INotification, terminalRunner: ITerminalRunner, store?: PlanStore);
1453
+ get storeInstance(): PlanStore;
1454
+ setWorkspaceRoot(fn: () => string): void;
1455
+ setRegistry(registry: RunnerRegistry): void;
1456
+ /**
1457
+ * A getter rather than a value where the caller has one: every task gets its
1458
+ * prompt composed at spawn time, but only a full-plan run passes through a
1459
+ * point where a snapshot could be refreshed — so "Run task", force-start and
1460
+ * retry would compose against whatever the last run happened to set.
1461
+ */
1462
+ setTddEnabled(enabled: boolean | (() => boolean)): void;
1463
+ approveCheckpoint(taskId: string): void;
1464
+ rejectCheckpoint(taskId: string, reason?: string): void;
1465
+ subscribe(observer: OrchestratorObserver): () => void;
1466
+ private emit;
1467
+ get isRunning(): boolean;
1468
+ get isReviewApproved(): boolean;
1469
+ get status(): 'approved' | 'running' | 'completed';
1470
+ get activeTaskIds(): string[];
1471
+ get activeSessionMap(): Map<string, string>;
1472
+ get queuedCount(): number;
1473
+ queueMessage(text: string): void;
1474
+ getQueuedMessages(): QueuedMessage[];
1475
+ setQueuedMessages(messages: QueuedMessage[]): void;
1476
+ clearQueuedMessages(): void;
1477
+ processNextQueuedMessage(): QueuedMessage | null;
1478
+ loadPlan(tasks: Task[], planRunners?: RunnerId[]): void;
1479
+ reconcilePlan(newTasks: Task[], planRunners?: RunnerId[]): void;
1480
+ start(): Promise<void>;
1481
+ stop(): void;
1482
+ onUserTaskComplete(taskId: string): Promise<void>;
1483
+ private onVerdict;
1484
+ getReadyTasks(): Task[];
1485
+ isBlocked(task: Task): boolean;
1486
+ /**
1487
+ * Cancel a running (or scheduled) task: kill its session and return it to
1488
+ * 'pending' — "not executed". The task is put on hold so the scheduler
1489
+ * doesn't immediately restart it; Retry / Force Start release the hold.
1490
+ */
1491
+ cancelTask(taskId: string): Promise<void>;
1492
+ markTaskComplete(taskId: string): Promise<void>;
1493
+ markAiTaskComplete(taskId: string): Promise<void>;
1494
+ /**
1495
+ * Undo a completion: return the task to "not executed" — pending, verdict and
1496
+ * summary dropped, archive entry removed. It is put on hold like a cancel, so
1497
+ * a running plan does not immediately re-spawn the work the user just
1498
+ * un-marked; Retry / Force Start / Run release the hold. Dependents fall back
1499
+ * to waiting on their own, because the scheduler gates on `isCompleted`.
1500
+ */
1501
+ markTaskIncomplete(taskId: string): Promise<void>;
1502
+ private logAndArchive;
1503
+ retryTask(taskId: string): Promise<void>;
1504
+ /**
1505
+ * Manually start a single AI task right now, bypassing dependency/readiness
1506
+ * gating (the "force start" affordance on a task card). Reuses the scheduler's
1507
+ * own startTask so a force-started task gets the same augmented prompt, session
1508
+ * tracking, and exit handling — callers must not re-spawn the runner themselves.
1509
+ * No-op if the task is unknown, not an AI task, or already running.
1510
+ */
1511
+ forceStartTask(taskId: string): Promise<void>;
1512
+ /**
1513
+ * Run exactly one task outside full-plan scheduling. The active/starting
1514
+ * session still contributes to isRunning so every surface exposes Stop and
1515
+ * disables Execute Plan, but onVerdict cannot auto-schedule other tasks
1516
+ * because the plan scheduler's `running` flag remains false.
1517
+ */
1518
+ runTask(taskId: string): Promise<void>;
1519
+ getCompletedCount(): number;
1520
+ getTotalCount(): number;
1521
+ isAllComplete(): boolean;
1522
+ isAnyFailed(): boolean;
1523
+ approveReview(): Promise<void>;
1524
+ getPlanVisualization(): {
1525
+ tasks: {
1526
+ id: string;
1527
+ title: string;
1528
+ dependencies: string[];
1529
+ parallelGroups: number[][];
1530
+ }[];
1531
+ parallelGroups: number[][];
1532
+ };
1533
+ tick(): Promise<void>;
1534
+ private startTask;
1535
+ }
1536
+
1537
+ /**
1538
+ * Everything needed to produce a plan from a goal. Carries the planning context
1539
+ * that previously threaded through the scheduler as positional args.
1540
+ */
1541
+ interface PlanRequest {
1542
+ goal: string;
1543
+ runners: RunnerId[];
1544
+ modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>;
1545
+ fs?: IFileSystem;
1546
+ fetcher?: IWebFetcher;
1547
+ onProgress?: (progress: ResearchProgress) => void;
1548
+ runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
1549
+ autonomousDefault?: boolean;
1550
+ signal?: AbortSignal;
1551
+ perRunnerAllowlist?: Partial<Record<RunnerId, string[]>>;
1552
+ /** The mode toggles this run honours. `modesFor('one-shot', …)` decides which apply. */
1553
+ modes?: PlannerModes;
1554
+ }
1555
+ interface ModifyPlanRequest {
1556
+ existingPlan: LegacyPlanState;
1557
+ userRequest: string;
1558
+ modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>;
1559
+ fs?: IFileSystem;
1560
+ fetcher?: IWebFetcher;
1561
+ onProgress?: (progress: ResearchProgress) => void;
1562
+ runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
1563
+ autonomousDefault?: boolean;
1564
+ perRunnerAllowlist?: Partial<Record<RunnerId, string[]>>;
1565
+ modes?: PlannerModes;
1566
+ }
1567
+ interface ModifyDuringExecutionRequest {
1568
+ executionLog: TaskSnapshot[];
1569
+ pendingTasks: Task[];
1570
+ activeSessions: Map<string, ActiveTaskSession>;
1571
+ userMessage: string;
1572
+ modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>;
1573
+ runners: RunnerId[];
1574
+ runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
1575
+ autonomousDefault?: boolean;
1576
+ perRunnerAllowlist?: Partial<Record<RunnerId, string[]>>;
1577
+ }
1578
+ interface ModifyDuringExecutionResult {
1579
+ pendingTasks: Task[];
1580
+ message: string;
1581
+ }
1582
+ /**
1583
+ * Owns one-shot plan generation and plan modification — the "planning is a
1584
+ * different workload from execution" thesis, as a module. The conversational
1585
+ * planning loop (ADR-0002) lives on {@link Session}, which talks to the AI
1586
+ * service directly because the conversation state is held there.
1587
+ */
1588
+ declare class Planner {
1589
+ private config;
1590
+ /**
1591
+ * Resolved per call, never captured: the planner backend can change while a
1592
+ * surface is open (`/planner`, the webview pills), and a Planner holding the
1593
+ * service it was built with would keep planning on the backend the host has
1594
+ * already switched away from.
1595
+ */
1596
+ private readonly resolveAiService;
1597
+ constructor(config: IConfig, aiService: IAiService | (() => IAiService));
1598
+ private get aiService();
1599
+ /** One-shot plan generation for non-conversational surfaces. Never asks questions. */
1600
+ generate(req: PlanRequest): Promise<LegacyPlanState>;
1601
+ modify(req: ModifyPlanRequest): Promise<{
1602
+ tasks: Task[];
1603
+ }>;
1604
+ modifyDuringExecution(req: ModifyDuringExecutionRequest): Promise<ModifyDuringExecutionResult>;
1605
+ }
1606
+
1607
+ interface CollectedContext {
1608
+ agentConfig: string | null;
1609
+ agentConfigPath: string | null;
1610
+ dirStructure: string | null;
1611
+ aiflowContext: string | null;
1612
+ }
1613
+ declare class ContextCollector {
1614
+ private fs;
1615
+ private registry?;
1616
+ constructor(fs: IFileSystem, registry?: RunnerRegistry | undefined);
1617
+ setRegistry(registry: RunnerRegistry): void;
1618
+ collect(runner: string): Promise<CollectedContext>;
1619
+ private readIfExists;
1620
+ }
1621
+
1622
+ /**
1623
+ * Injectable command runner. Defaults to the real `child_process` exec wrapper;
1624
+ * tests pass a fake so discovery never spawns a process.
1625
+ */
1626
+ type ExecImpl = (command: string, options?: {
1627
+ timeout?: number;
1628
+ }) => Promise<{
1629
+ stdout: string;
1630
+ }>;
1631
+ declare function discoverGeminiModels(apiKey: string, baseUrl?: string, fetchImpl?: typeof fetch): Promise<DiscoveredModel[]>;
1632
+
1633
+ interface ModelResolverDeps {
1634
+ fetchImpl?: typeof fetch;
1635
+ execImpl?: ExecImpl;
1636
+ }
1637
+ declare class ModelResolver {
1638
+ private registry;
1639
+ private config;
1640
+ private discovery;
1641
+ private fetchImpl?;
1642
+ private cachedPickerOptions;
1643
+ private lastDiscoveryErrors;
1644
+ constructor(registry: RunnerRegistry, config: IConfig, deps?: ModelResolverDeps);
1645
+ static builtinOptions(): OrchestratorOption[];
1646
+ modelsForRunners(runners: RunnerId[]): Promise<Partial<Record<RunnerId, DiscoveredModel[]>>>;
1647
+ pickerOptions(): Promise<OrchestratorOption[]>;
1648
+ /**
1649
+ * Per-provider failures from the most recent `pickerOptions()` fetch, keyed
1650
+ * by provider id. Empty when every configured provider's catalog loaded (or
1651
+ * before the first fetch). Surfaces consume this to flag a provider whose
1652
+ * key/endpoint is set but whose catalog could not be reached.
1653
+ */
1654
+ getDiscoveryErrors(): Record<string, string>;
1655
+ refresh(): Promise<ProviderModelLists>;
1656
+ refreshRunnerModels(): void;
1657
+ invalidate(): void;
1658
+ }
1659
+
1660
+ /**
1661
+ * Detects whether a runner's underlying CLI is actually installed on the host,
1662
+ * by invoking `<command> --version`. Mirrors ModelDiscovery: injectable exec,
1663
+ * augmented PATH, per-process cache. A runner that isn't installed should never
1664
+ * be offered for selection in any surface.
1665
+ */
1666
+ declare class RunnerInstallation {
1667
+ private cache;
1668
+ private inflight;
1669
+ private registry;
1670
+ private execImpl;
1671
+ constructor(registry: RunnerRegistry, execImpl?: ExecImpl);
1672
+ /** True if the runner's CLI responds to `--version`. Cached per process. */
1673
+ isInstalled(runner: string): Promise<boolean>;
1674
+ /**
1675
+ * Whether this runner can serve as a harness planner (ADR-0009, T9), and why
1676
+ * not when it cannot. Surfaces call this to grey out an unusable agent in the
1677
+ * planner picker with the reason attached — discovering a missing CLI after
1678
+ * typing a real goal is the failure this exists to prevent.
1679
+ *
1680
+ * "Usable" is deliberately shallow: the binary answers `--version`, and the
1681
+ * runner declares a planner transport. Whether the user's subscription is
1682
+ * live cannot be known without spending a turn on it, so an expired login
1683
+ * surfaces where it actually bites — as the agent's own error text on the
1684
+ * first turn.
1685
+ */
1686
+ plannerUsability(runner: string): Promise<{
1687
+ usable: boolean;
1688
+ reason?: string;
1689
+ }>;
1690
+ /**
1691
+ * Whether the CLI can be started the way the harness planner actually starts
1692
+ * it: `spawn` with no shell.
1693
+ *
1694
+ * `isInstalled` probes through `exec`, which goes through cmd.exe on Windows
1695
+ * and so happily resolves a `.cmd` shim — while the planner's own spawn does
1696
+ * not, because CreateProcess performs no PATHEXT lookup. That gap produced
1697
+ * the worst failure shape available: the picker reported the runner healthy,
1698
+ * then the session died on ENOENT with nothing to act on. POSIX has no such
1699
+ * split (`execvp` and `exec` search PATH identically), so this always agrees
1700
+ * with `isInstalled` there.
1701
+ */
1702
+ private isSpawnable;
1703
+ /** Subset of the given runner ids whose CLI is installed. */
1704
+ filterInstalled(runners: string[]): Promise<string[]>;
1705
+ clear(): void;
1706
+ }
1707
+
1708
+ declare const PROVIDER_LABEL: Record<AiProvider, string>;
1709
+ declare const PROVIDER_SHORT_LABEL: Record<AiProvider, string>;
1710
+ interface ProviderRegistration {
1711
+ id: AiProvider;
1712
+ label: string;
1713
+ shortLabel: string;
1714
+ /** `cli` is a harness planner (ADR-0009): a local coding agent, not an HTTP vendor. */
1715
+ serviceType: 'openai' | 'google' | 'cli';
1716
+ /** Harness planners only: the runner id whose manifest, models and binary this provider drives. */
1717
+ runnerId?: string;
1718
+ defaultBaseUrl: string;
1719
+ apiKeyEnvVar: string;
1720
+ detectEnvVars: string[];
1721
+ baseUrlEnvVar?: string;
1722
+ secretStoreKey?: string;
1723
+ vscodeBaseUrlKey?: string;
1724
+ vscodeApiKeyKey?: string;
1725
+ discoversModels: boolean;
1726
+ modelPrefix?: string;
1727
+ }
1728
+ declare const ALL_PROVIDERS: Record<AiProvider, ProviderRegistration>;
1729
+ declare function getProviderMeta(id: AiProvider): ProviderRegistration;
1730
+ declare function prefixModelId(provider: AiProvider, modelId: string): string;
1731
+ declare function stripModelPrefix(id: string, provider: AiProvider): string;
1732
+ declare function resolveProviderFromPrefix(id: string): AiProvider | null;
1733
+ /**
1734
+ * The harness planners, in display order. Deliberately NOT part of
1735
+ * `PROVIDER_PRIORITY`: that list means "vendors that take an API key", and
1736
+ * every consumer of it — key prompts, catalog fetches, base-URL cache clearing
1737
+ * — would be asking a local binary for a credential it does not have. Surfaces
1738
+ * that offer a planner picker concatenate the two lists and gate these on
1739
+ * `RunnerInstallation.plannerUsability`.
1740
+ */
1741
+ declare const CLI_PROVIDERS: AiProvider[];
1742
+ declare const PROVIDER_PRIORITY: AiProvider[];
1743
+ declare const PROVIDER_DETECT_PRIORITY: AiProvider[];
1744
+ declare function isOpenAiProvider(id: AiProvider): boolean;
1745
+ /**
1746
+ * The one guard that separates a harness planner from an LLM vendor (ADR-0009).
1747
+ * Three places consult it: API-key resolution (skipped — the subscription is
1748
+ * the credential), provider routing (skipped — there is no HTTP endpoint), and
1749
+ * the planner-model picker (fed by per-runner discovery, not the vendor
1750
+ * catalog).
1751
+ */
1752
+ declare function isCliProvider(id: AiProvider): boolean;
1753
+ /** The runner a harness planner drives, or null for a vendor provider. */
1754
+ declare function runnerForProvider(id: AiProvider): string | null;
1755
+ /** The harness planner that wraps a given runner, or null when none does. */
1756
+ declare function providerForRunner(runner: string): AiProvider | null;
1757
+ /**
1758
+ * The single definition of which providers count as "configured" — one API
1759
+ * key each, or (for `openai_compatible`) an explicit base URL. Every surface
1760
+ * (CLI listing, VS Code picker gating) derives its configured-provider set
1761
+ * from here so they can never diverge on what a user has set up. Order follows
1762
+ * PROVIDER_PRIORITY.
1763
+ */
1764
+ declare function configuredProviders(config: {
1765
+ getProviderApiKey(provider: AiProvider): string;
1766
+ openaiCompatibleBaseUrl: string;
1767
+ }): AiProvider[];
1768
+
1769
+ interface SpawnSpec {
1770
+ command: string;
1771
+ args: string[];
1772
+ env?: Record<string, string>;
1773
+ }
1774
+ /**
1775
+ * Shared plumbing for the agents that speak newline-delimited JSON over stdio.
1776
+ *
1777
+ * One long-lived process per planner session (ADR-0009, T3): spawned at
1778
+ * `start`, fed one message per turn, disposed at the session boundary. The
1779
+ * alternative — respawning per turn — pays a cold start on every message
1780
+ * *including each corrective re-emit*, and re-reads the repository from
1781
+ * scratch when it cannot resume.
1782
+ *
1783
+ * Subclasses own their agent's protocol: what to spawn, how to phrase a user
1784
+ * turn, and how to read one protocol line. Everything below — line framing,
1785
+ * turn lifecycle, abort, stderr capture, premature-exit detection — is the
1786
+ * same for all of them.
1787
+ */
1788
+ declare abstract class StdioAgentAdapter implements AgentAdapter {
1789
+ protected deps: AgentProcessDeps;
1790
+ abstract readonly agentId: string;
1791
+ protected process: ChildProcess | null;
1792
+ protected sessionId: string | null;
1793
+ private readonly stdout;
1794
+ private stderrTail;
1795
+ private exited;
1796
+ /** Set for the duration of a turn. Outside one, events are buffered rather than dropped. */
1797
+ private turnEmit;
1798
+ /**
1799
+ * Events the agent produced between turns — in practice the startup warnings
1800
+ * that arrive during the handshake. Dropping them hid the one diagnostic
1801
+ * that explains a planner which cannot read the workspace, so they are held
1802
+ * until a turn exists to show them in.
1803
+ */
1804
+ private betweenTurns;
1805
+ private disposed;
1806
+ /** Resolves when the process ends, so a handshake can lose the race instead of waiting out its timeout. */
1807
+ protected processEnded: Promise<void>;
1808
+ /** The environment the agent was spawned under, for any side process a handshake needs. */
1809
+ protected spawnEnv: NodeJS.ProcessEnv;
1810
+ private markEnded;
1811
+ constructor(deps: AgentProcessDeps);
1812
+ /** The command line that starts this agent in its read-only mode. */
1813
+ protected abstract spawnSpec(opts: AgentStartOptions): SpawnSpec;
1814
+ /** The bytes written to stdin to open one user turn. Must end with a newline. */
1815
+ protected abstract turnPayload(message: string): string;
1816
+ /**
1817
+ * Interpret one line of the agent's protocol, emitting normalized events.
1818
+ * Emitting `turn_end` ends the turn; emitting `error` ends it as a failure.
1819
+ */
1820
+ protected abstract handleLine(line: string, emit: (event: AgentEvent) => void): void;
1821
+ /** Protocol handshake, if the agent needs one before it accepts a turn. */
1822
+ protected handshake(_opts: AgentStartOptions): Promise<void>;
1823
+ start(opts: AgentStartOptions): Promise<void>;
1824
+ send(message: string, onEvent: (event: AgentEvent) => void, signal?: AbortSignal): Promise<void>;
1825
+ nativeSessionId(): string | null;
1826
+ dispose(): void;
1827
+ /** Write one raw protocol line to the agent. */
1828
+ protected writeLine(payload: unknown): void;
1829
+ /** Fail-safe contract: a dead agent reports its own last words, never an empty bubble. */
1830
+ protected exitMessage(): string;
1831
+ /** Parse a protocol line, ignoring the non-JSON banners some CLIs print. */
1832
+ protected static parse<T>(line: string): T | null;
1833
+ }
1834
+
1835
+ /**
1836
+ * Claude Code as a planner, over its bidirectional streaming-JSON transport
1837
+ * (ADR-0009).
1838
+ *
1839
+ * `-p --input-format stream-json --output-format stream-json` keeps one process
1840
+ * alive across turns: user messages go in as JSON lines, and the session's
1841
+ * assistant blocks, tool uses, tool results and turn boundaries come back the
1842
+ * same way. It is the richest of the three streams — partial messages and
1843
+ * separate thinking blocks — which is why this agent went first.
1844
+ */
1845
+ declare class ClaudeCodeAdapter extends StdioAgentAdapter {
1846
+ readonly agentId = "claude-code";
1847
+ protected spawnSpec(opts: AgentStartOptions): SpawnSpec;
1848
+ protected turnPayload(message: string): string;
1849
+ protected handleLine(line: string, emit: (event: AgentEvent) => void): void;
1850
+ }
1851
+
1852
+ /**
1853
+ * Codex as a planner, over its `app-server` stdio JSON-RPC transport
1854
+ * (ADR-0009).
1855
+ *
1856
+ * Ordewell already speaks a slice of this protocol — `ModelDiscovery` drives
1857
+ * `initialize` → `model/list` to build the Codex model catalog — so the
1858
+ * handshake here is that one continued into a thread.
1859
+ *
1860
+ * Codex's own CLI marks `app-server` experimental, and it is the transport
1861
+ * most likely to drift: the method names below (`thread/start`, `turn/start`,
1862
+ * `item/completed`) come from the schema the installed binary generates, and a
1863
+ * version that renames them will surface as a visible dead turn rather than a
1864
+ * hang, because the base class watches the process as well as the protocol.
1865
+ */
1866
+ declare class CodexAdapter extends StdioAgentAdapter {
1867
+ readonly agentId = "codex";
1868
+ private threadId;
1869
+ private nextRequestId;
1870
+ private settleHandshake;
1871
+ private handshakeError;
1872
+ private startOpts;
1873
+ /** Whether this turn has already emitted prose — see the `agentMessage` case. */
1874
+ private turnHasText;
1875
+ private resumeAttempted;
1876
+ private resumeFallbackSent;
1877
+ private sandbox;
1878
+ protected spawnSpec(opts: AgentStartOptions): SpawnSpec;
1879
+ /**
1880
+ * `initialize`, then `thread/start`, both before the first user message. The
1881
+ * thread is pinned to the read-only sandbox with approvals set to `never` —
1882
+ * with nobody watching a planner's prompts, "ask" would mean "hang", which is
1883
+ * ADR-0008's absent-is-denial invariant kept by construction.
1884
+ */
1885
+ protected handshake(opts: AgentStartOptions): Promise<void>;
1886
+ /**
1887
+ * Open the thread this session plans in. A resume id means the previous
1888
+ * process died mid-session: `thread/resume` puts the agent back in front of
1889
+ * the context it already paid to read. A failed resume is not an error — the
1890
+ * response handler falls back to a fresh thread, which is the same
1891
+ * degradation `restoreChat` performs on every surface (T4).
1892
+ */
1893
+ private startThread;
1894
+ protected turnPayload(message: string): string;
1895
+ protected handleLine(line: string, emit: (event: AgentEvent) => void): void;
1896
+ /**
1897
+ * Refuse one server→client request. Requests whose result schema can express
1898
+ * a refusal get that payload; everything else — a permission grant, a
1899
+ * question for a user who is not watching, a tool call the client is supposed
1900
+ * to run — gets a JSON-RPC error, which Codex surfaces to the model as a
1901
+ * failed request and plans around, rather than waiting on.
1902
+ */
1903
+ private answerServerRequest;
1904
+ /** A tool item entering `inProgress` — announce the call so the timeline moves. */
1905
+ private emitItemStart;
1906
+ private emitItemDone;
1907
+ }
1908
+
1909
+ /**
1910
+ * OpenCode as a planner, over its headless HTTP server (ADR-0009).
1911
+ *
1912
+ * The odd one out: `opencode serve` is a real server rather than a stdio
1913
+ * protocol, so this adapter owns both halves of the boundary — it spawns the
1914
+ * process through the same injected `spawn` every other adapter uses, then
1915
+ * talks to it through the injected `fetch`. Both are part of the one seam the
1916
+ * tests drive.
1917
+ *
1918
+ * The turn ends when the message POST resolves. Live events stream from the
1919
+ * server's `/event` channel, but the POST is what settles the turn: an event
1920
+ * name that changes between OpenCode versions then costs liveness, not
1921
+ * correctness.
1922
+ */
1923
+ declare class OpenCodeAdapter implements AgentAdapter {
1924
+ private deps;
1925
+ readonly agentId = "opencode";
1926
+ private process;
1927
+ private baseUrl;
1928
+ private sessionId;
1929
+ private stderrTail;
1930
+ private exited;
1931
+ private disposed;
1932
+ private opts;
1933
+ constructor(deps: AgentProcessDeps);
1934
+ start(opts: AgentStartOptions): Promise<void>;
1935
+ send(message: string, onEvent: (event: AgentEvent) => void, signal?: AbortSignal): Promise<void>;
1936
+ /**
1937
+ * Emit one message part, once. OpenCode reports a tool part repeatedly as it
1938
+ * moves through pending → running → completed, so parts are keyed by id and
1939
+ * only the terminal state produces a result.
1940
+ */
1941
+ private emitPart;
1942
+ /**
1943
+ * Deny one permission request (T1). OpenCode blocks the turn until the
1944
+ * request is answered, so this must answer — `reject` rather than a silent
1945
+ * drop, which is the same "absent answer is a denial" invariant ADR-0008
1946
+ * states for Ordewell's own tools. The refusal is announced so the timeline
1947
+ * shows the planner reaching for something it may not have.
1948
+ */
1949
+ private denyPermission;
1950
+ /**
1951
+ * Server-sent events from `/event`. Tool activity on it is liveness only —
1952
+ * the settled POST repeats it — but permission requests arrive nowhere else,
1953
+ * so the stream is load-bearing for {@link denyPermission}.
1954
+ */
1955
+ private streamEvents;
1956
+ private json;
1957
+ nativeSessionId(): string | null;
1958
+ dispose(): void;
1959
+ private exitMessage;
1960
+ }
1961
+
1962
+ interface MappedTool {
1963
+ tool: ResearchToolType;
1964
+ /** The agent's own name, kept whenever it differs from the member it mapped to. */
1965
+ toolLabel?: string;
1966
+ }
1967
+ /**
1968
+ * Classify one agent tool name. Case- and separator-insensitive, because the
1969
+ * three agents disagree on `Read` vs `read` vs `read_file` for the same thing.
1970
+ */
1971
+ declare function mapAgentTool(name: string): MappedTool;
1972
+ /**
1973
+ * Normalize an agent's tool arguments into the shapes `summarizeToolCall`
1974
+ * already knows how to read, so a harness `Read` gets a one-line summary
1975
+ * instead of a bare tool name. Unknown keys are preserved — the raw args stay
1976
+ * visible under the surfaces' output chevron.
1977
+ */
1978
+ declare function normalizeAgentArgs(tool: ResearchToolType, args: Record<string, unknown>): Record<string, unknown>;
1979
+
1980
+ type SerializedTaskStatus = {
1981
+ id: string;
1982
+ status: string;
1983
+ verdict: {
1984
+ outcome: 'pass' | 'fail';
1985
+ reason: string;
1986
+ checks: Verdict['checks'];
1987
+ } | null;
1988
+ };
1989
+ type SerializedTask = {
1990
+ id: string;
1991
+ order: number;
1992
+ title: string;
1993
+ type: string;
1994
+ description: string;
1995
+ dependencies: string[];
1996
+ assignedRunner: RunnerId;
1997
+ assignedModel: Task['assignedModel'] | null;
1998
+ taskMode: string;
1999
+ prompt: string | null;
2000
+ subtasks: SerializedTask[];
2001
+ userSteps: Task['userSteps'];
2002
+ thinkingEffort: Task['thinkingEffort'];
2003
+ autonomy: Task['autonomy'];
2004
+ sliceType: Task['sliceType'];
2005
+ userStoriesCovered: Task['userStoriesCovered'];
2006
+ };
2007
+ type SerializedPlan = {
2008
+ tasks: SerializedTask[];
2009
+ runners: RunnerId[];
2010
+ generatedAt: string;
2011
+ conversationHistory?: LegacyPlanState['conversationHistory'];
2012
+ prdMarkdown?: string;
2013
+ queuedMessages?: QueuedMessage[];
2014
+ };
2015
+ type SessionMessage = {
2016
+ type: 'plan_generated';
2017
+ plan: SerializedPlan;
2018
+ goal: string;
2019
+ runners: RunnerId[];
2020
+ } | {
2021
+ type: 'planner_message';
2022
+ content: string;
2023
+ timestamp: string;
2024
+ } | {
2025
+ type: 'status_update';
2026
+ tasks: SerializedTaskStatus[];
2027
+ } | {
2028
+ type: 'review_needed';
2029
+ tasks: SerializedTask[];
2030
+ } | {
2031
+ type: 'review_approved';
2032
+ } | {
2033
+ type: 'checkpoint';
2034
+ taskId: string;
2035
+ taskTitle: string;
2036
+ summary: string;
2037
+ } | {
2038
+ type: 'execution_complete';
2039
+ summary: {
2040
+ total: number;
2041
+ completed: number;
2042
+ failed: number;
2043
+ };
2044
+ } | {
2045
+ type: 'execution_stopped';
2046
+ } | {
2047
+ type: 'queue_ready';
2048
+ } | {
2049
+ type: 'task_updated';
2050
+ taskId: string;
2051
+ changes: Record<string, unknown>;
2052
+ } | {
2053
+ type: 'task_started';
2054
+ taskId: string;
2055
+ order: number;
2056
+ title: string;
2057
+ runner: RunnerId;
2058
+ modelId?: string;
2059
+ } | {
2060
+ type: 'task_output';
2061
+ taskId: string;
2062
+ text: string;
2063
+ } | {
2064
+ type: 'plan_thinking';
2065
+ text: string;
2066
+ } | {
2067
+ type: 'research_step';
2068
+ tool: string;
2069
+ toolLabel?: string;
2070
+ args: string;
2071
+ subagentId?: string;
2072
+ toolCallId?: string;
2073
+ } | {
2074
+ type: 'plan_token';
2075
+ token: string;
2076
+ } | {
2077
+ type: 'research_step_done';
2078
+ step: ResearchStep;
2079
+ subagentId?: string;
2080
+ } | {
2081
+ type: 'approval_request';
2082
+ id: string;
2083
+ kind: ApprovalKind;
2084
+ subject: string;
2085
+ scope: string;
2086
+ detail?: string;
2087
+ } | {
2088
+ type: 'approval_settled';
2089
+ id: string;
2090
+ granted: boolean;
2091
+ } | {
2092
+ type: 'approval_decided';
2093
+ kind: ApprovalKind;
2094
+ subject: string;
2095
+ scope: string;
2096
+ detail?: string;
2097
+ granted: boolean;
2098
+ source: Exclude<ApprovalSource, 'asked'>;
2099
+ };
2100
+ type SessionBroadcaster = (msg: SessionMessage) => void;
2101
+ declare function serializeTask(t: Task): SerializedTask;
2102
+ declare function serializeTaskStatus(t: Task): SerializedTaskStatus;
2103
+ declare function serializePlan(plan: LegacyPlanState): SerializedPlan;
2104
+ declare function executionSummary(tasks: Task[]): {
2105
+ total: number;
2106
+ completed: number;
2107
+ failed: number;
2108
+ };
2109
+
2110
+ /**
2111
+ * Options for plan generation. Progress is not overridable: every planner
2112
+ * progress event is translated to a SessionMessage inside the Session and
2113
+ * emitted through the broadcast seam, so all surfaces consume one union.
2114
+ */
2115
+ interface GeneratePlanOptions {
2116
+ signal?: AbortSignal;
2117
+ }
2118
+ /** The slice of Planner the Session drives — the injection seam for tests. */
2119
+ type SessionPlanner = Pick<Planner, 'generate' | 'modify' | 'modifyDuringExecution'>;
2120
+ /** Runtime prefs read live — may toggle between operations. */
2121
+ interface SessionRuntimeSettings {
2122
+ tddEnabled: boolean;
2123
+ grillMeEnabled: boolean;
2124
+ prdEnabled?: boolean;
2125
+ reviewEnabled?: boolean;
2126
+ verificationEnabled?: boolean;
2127
+ researchSubagentsEnabled?: boolean;
2128
+ modelAllowlist?: Record<string, string[]>;
2129
+ }
2130
+ /**
2131
+ * The whole of what a host reads off disk for a Session. Both hosts used to
2132
+ * assemble this by hand, mapping each toggle's settings key to its runtime key
2133
+ * in two blocks nothing kept in step — which is how one toggle came to be
2134
+ * dropped. `MODE_TOGGLES` holds the mapping now; this adds the one field that
2135
+ * is not a toggle.
2136
+ */
2137
+ declare function sessionRuntimeSettings(settings: UserSettings): SessionRuntimeSettings;
2138
+ /**
2139
+ * Everything a delivery surface constructs to host a session. Structural config
2140
+ * (enabledRunners, orchestratorModel, providerModelLists) is snapshotted inside
2141
+ * `config` at construction and never re-read from the environment. Runtime
2142
+ * settings (tdd, grillMe) are read live via the `settings` callback so a toggle
2143
+ * between generate and execute takes effect.
2144
+ */
2145
+ interface SessionDeps {
2146
+ config: IConfig;
2147
+ notifications: INotification;
2148
+ runner: ITerminalRunner;
2149
+ registry: RunnerRegistry;
2150
+ /** Resolves the workspace root for the orchestrator (lazy — VS Code can change it). */
2151
+ workspaceRoot: () => string;
2152
+ /** Filesystem adapter for planner research. */
2153
+ fsAdapter: IFileSystem;
2154
+ /** Emits plan-lifecycle events to the surface. Transport-agnostic. */
2155
+ broadcast: SessionBroadcaster;
2156
+ /** Shared across sessions — sole producer of model catalogs and routing lists. */
2157
+ modelResolver: ModelResolver;
2158
+ /** Live runtime settings (tdd, grillMe). Read at each operation that needs them. */
2159
+ settings: () => SessionRuntimeSettings;
2160
+ /**
2161
+ * Host-assigned session id. When set, every persist writes under this id so
2162
+ * the host's REST/UI ids match the saved-session store. When omitted, the
2163
+ * Session mints a fresh id per plan (generatePlan/startPlanning).
2164
+ */
2165
+ sessionId?: string;
2166
+ /**
2167
+ * Planner-conversation seam. Defaults to the provider service for
2168
+ * `config.aiProvider`; inject a fake to test the conversation half without
2169
+ * an LLM.
2170
+ */
2171
+ aiService?: IAiService;
2172
+ /** Plan-generation seam. Defaults to a Planner over the session's aiService. */
2173
+ planner?: SessionPlanner;
2174
+ }
2175
+ /**
2176
+ * The per-session execution stack — deepened from a wiring bag into the
2177
+ * lifecycle owner. Owns plan generation, execution, mutation, persistence, and
2178
+ * the orchestrator observer wiring. The orchestrator's observer is subscribed
2179
+ * once for the session's lifetime (not per-operation), which kills the
2180
+ * double-subscribe class of bug. Persistence is an internal seam: every plan
2181
+ * mutation routes through `persist()`, so the obligation has a home instead of
2182
+ * being scattered across 11 call sites.
2183
+ *
2184
+ * The broadcast seam carries {@link SessionMessage} — the 15 plan-lifecycle
2185
+ * events. Catalog/config messages (setModels, setRunnerList, …) stay on the
2186
+ * host; Session never emits them.
2187
+ */
2188
+ declare class Session {
2189
+ /** Injected by a test; when present it is the service, forever. */
2190
+ private readonly pinnedAiService?;
2191
+ private liveAiService;
2192
+ private liveAiProvider;
2193
+ private readonly workspaceRootFn;
2194
+ private planner;
2195
+ private orchestrator;
2196
+ private store;
2197
+ private config;
2198
+ private registry;
2199
+ private plan;
2200
+ private goal;
2201
+ private workspace;
2202
+ private broadcast;
2203
+ private modelResolver;
2204
+ private fsAdapter;
2205
+ private approvals;
2206
+ private approvalPolicy;
2207
+ private fetcher;
2208
+ private settingsFn;
2209
+ private currentAllowlist;
2210
+ /** Last discovered model catalog — lets sync plan commits clamp thinking efforts to real variants. */
2211
+ private modelsCache;
2212
+ private unsubObserver;
2213
+ private readonly hostSessionId?;
2214
+ private currentSessionId;
2215
+ constructor(deps: SessionDeps);
2216
+ /**
2217
+ * The planner transport for the provider configured *right now* (ADR-0009).
2218
+ *
2219
+ * Resolved on every read rather than once in the constructor, because a
2220
+ * Session outlives the choice: VS Code hosts exactly one for the whole
2221
+ * window, and the webview pills and `/planner` switch backends underneath it.
2222
+ * The model id was already read live, so a service captured at construction
2223
+ * meant a switched planner kept the old backend and got handed the new one's
2224
+ * model — an OpenCode model id spawned as `claude --model opencode/…`, which
2225
+ * the agent rejects as nonexistent.
2226
+ *
2227
+ * Switching releases the outgoing service: a harness planner holds an OS
2228
+ * process, so dropping the reference without `reset()` leaks an agent.
2229
+ */
2230
+ private get aiService();
2231
+ /**
2232
+ * Answer an outstanding approval. Every surface funnels here — the web
2233
+ * server's HTTP route, the VS Code webview, the TUI prompt — so the decision
2234
+ * path is identical regardless of who is looking.
2235
+ */
2236
+ resolveApproval(id: string, granted: boolean): boolean;
2237
+ /** Requests still waiting for an answer, replayed to a surface that connects mid-prompt. */
2238
+ outstandingApprovals(): PendingApproval[];
2239
+ /** Scopes the user granted this session — surfaced so a UI can show what is already allowed. */
2240
+ approvedScopes(): string[];
2241
+ /** The stable id this session persists under — matches the host's id when one was provided. */
2242
+ get sessionId(): string;
2243
+ get executionLog(): TaskSnapshot[];
2244
+ /** Tasks always read from PlanStore — the single source of truth. */
2245
+ get planTasks(): Task[];
2246
+ private attachObserver;
2247
+ private buildObserver;
2248
+ private translateProgress;
2249
+ /** Persists PlanStore state to disk. PlanStore is the single authority;
2250
+ * LegacyPlanState.tasks is populated only here, at persist time. */
2251
+ private persist;
2252
+ /** A new plan on a long-lived Session gets its own persisted identity (unless the host fixed one). */
2253
+ private remintSessionId;
2254
+ /**
2255
+ * A new plan starts from zero: drop the live planner conversation and every
2256
+ * task, log, and queued message left over from a previous plan on this
2257
+ * Session. Without this, a long-lived Session (VS Code hosts exactly one)
2258
+ * leaks the previous session's tasks into `planContextBlock()` — the model
2259
+ * is told they are the CURRENT plan and re-presents them as its draft.
2260
+ */
2261
+ private beginFreshPlan;
2262
+ /**
2263
+ * Return the Session to a blank slate — hosts call this on "new session".
2264
+ * Everything scoped to the old session goes: the live AI conversation, plan,
2265
+ * goal, tasks, execution log, queued messages, model cache, and (unless the
2266
+ * host fixed one) the persisted identity, so nothing can bleed into the next
2267
+ * session.
2268
+ */
2269
+ reset(): void;
2270
+ private runnerModesFor;
2271
+ get planState(): LegacyPlanState | null;
2272
+ /**
2273
+ * The live plan in the shape a saved session is read back as. The disk
2274
+ * boundary rewrites `in_progress` to `pending` — nothing is running when a
2275
+ * session comes off a file — so a surface that re-reads the plan mid-run must
2276
+ * come here instead, or every task it is watching reads as never started.
2277
+ */
2278
+ get currentPlanState(): PlanState | null;
2279
+ get currentGoal(): string;
2280
+ get isPlanning(): boolean;
2281
+ get isExecuting(): boolean;
2282
+ get status(): 'approved' | 'running' | 'completed';
2283
+ get sessionConfig(): IConfig;
2284
+ startExecution(): Promise<void>;
2285
+ generatePlan(goal: string, runners: RunnerId[], options?: GeneratePlanOptions): Promise<LegacyPlanState>;
2286
+ /**
2287
+ * Kick off the planner conversation (ADR-0002): research + the first planner
2288
+ * message. The AI service retains the tool-use history; the Session persists
2289
+ * the user/assistant dialogue on the plan state.
2290
+ */
2291
+ startPlanning(goal: string, runners: RunnerId[], options?: GeneratePlanOptions): Promise<LegacyPlanState>;
2292
+ /**
2293
+ * Every subsequent user reply in the planning conversation — grill-me
2294
+ * answers, PRD accept/adjust, outline confirm. One branch, no phase ladder.
2295
+ */
2296
+ continueConversation(userMessage: string, options?: GeneratePlanOptions): Promise<LegacyPlanState>;
2297
+ /**
2298
+ * Drive a planner turn to a persisted, broadcast outcome. Task edits apply
2299
+ * atomically; validation failures are fed back to the model for up to 2
2300
+ * silent retries, then surfaced as a message with the plan untouched. The
2301
+ * first turn (startPlanning) and every later turn route through here — one
2302
+ * path, not two.
2303
+ */
2304
+ private settleTurn;
2305
+ /**
2306
+ * The "you are here" block for post-plan chat: current tasks with stable
2307
+ * references, plus the task-ops protocol. Injected per turn (never
2308
+ * persisted) so the model always sees live statuses — including which tasks
2309
+ * are locked by a running execution.
2310
+ */
2311
+ private planContextBlock;
2312
+ /** Validate + commit a task_ops turn atomically. Returns the errors on rejection (plan untouched). */
2313
+ private applyTaskOpsTurn;
2314
+ /**
2315
+ * Resume a persisted dialogue onto a fresh AI-service conversation — either
2316
+ * because the in-memory one is gone (session reload, extension restart), or
2317
+ * because {@link continueConversation} found it stale against the planner
2318
+ * config now in effect (a harness planner's model/effort switched mid-chat).
2319
+ * The saved transcript seeds the new conversation; no LLM call happens
2320
+ * before the user's message is sent.
2321
+ */
2322
+ private resumeConversation;
2323
+ /** Whether the planner conversation is live (started and not yet committed to a plan). */
2324
+ get isConversationActive(): boolean;
2325
+ /** Append one dialogue entry to the persisted transcript. Callers persist via the mutatePlan ritual. */
2326
+ private appendTranscript;
2327
+ /** Commit a settled (non-task_ops) turn. Both branches run the mutatePlan ritual. */
2328
+ private applyConversationTurn;
2329
+ /**
2330
+ * PRD mode: when the planner writes the full markdown PRD, save it to
2331
+ * .scratch/<slug>/PRD.md (Matt Pocock to-prd convention) and keep it on the plan.
2332
+ */
2333
+ private capturePrd;
2334
+ executePlan(): Promise<void>;
2335
+ approveReview(): Promise<LegacyPlanState>;
2336
+ forceStartTask(taskId: string): Promise<void>;
2337
+ runTask(taskId: string): Promise<void>;
2338
+ retryTask(taskId: string): Promise<void>;
2339
+ cancelTask(taskId: string): Promise<void>;
2340
+ markAiTaskComplete(taskId: string): Promise<void>;
2341
+ markTaskComplete(taskId: string): Promise<void>;
2342
+ markTaskIncomplete(taskId: string): Promise<void>;
2343
+ tick(): Promise<void>;
2344
+ processQueuedMessages(): Promise<void>;
2345
+ approveCheckpoint(taskId: string): void;
2346
+ rejectCheckpoint(taskId: string, reason?: string): void;
2347
+ queueMessage(text: string): void;
2348
+ getQueuedMessages(): QueuedMessage[];
2349
+ setQueuedMessages(msgs: ReturnType<TaskOrchestrator['getQueuedMessages']>): void;
2350
+ processNextQueuedMessage(): QueuedMessage | null;
2351
+ get queuedCount(): number;
2352
+ getTask(taskId: string): Task | undefined;
2353
+ get isReviewApproved(): boolean;
2354
+ stopExecution(): void;
2355
+ /**
2356
+ * The one mutation seam: every structural plan mutation runs the same
2357
+ * ritual — store op → persist (which snapshots tasks from PlanStore) →
2358
+ * broadcast — in that order, once. A store op returning false aborts
2359
+ * before anything is persisted. PlanStore is the single authority for
2360
+ * task state; LegacyPlanState.tasks is populated only at persist time.
2361
+ */
2362
+ private mutatePlan;
2363
+ updateTask(taskId: string, changes: Partial<Task>): LegacyPlanState | null;
2364
+ /**
2365
+ * Move one task onto a different runner. Distinct from {@link updateTask}
2366
+ * because a runner change is never a single-field edit: the task's model,
2367
+ * thinking effort and mode are all scoped to its runner, so they are
2368
+ * re-derived from the new runner's catalog (see {@link retargetTaskRunner}).
2369
+ * That needs discovery, which is async — hence a method of its own rather
2370
+ * than a branch inside the sync `updateTask`.
2371
+ *
2372
+ * The runner is also admitted into `plan.runners` — see {@link admitRunner}.
2373
+ */
2374
+ setTaskRunner(taskId: string, runner: RunnerId): Promise<LegacyPlanState | null>;
2375
+ /** What a runner offers, as {@link runnerAssignment} needs it. Spawns the runner's CLI to list models. */
2376
+ private catalogFor;
2377
+ /**
2378
+ * What a *derived* assignment may draw from: the runner's catalog narrowed to
2379
+ * the user's allowlist. Deriving from the full catalog would hand a task the
2380
+ * runner's first model regardless of a restriction the user set — the next
2381
+ * planner turn's `coerceAssignments` would snap it back anyway, so the user
2382
+ * would see their pick silently change instead of never being offered.
2383
+ *
2384
+ * `catalogFor` stays unnarrowed because {@link admitRunner} caches it as what
2385
+ * the runner really offers, which is what effort clamping needs.
2386
+ */
2387
+ private allowedCatalog;
2388
+ /**
2389
+ * Make a runner a first-class member of this plan. Without this, the next
2390
+ * planner turn's `coerceAssignments` would treat it as disallowed and snap
2391
+ * every task on it back, silently undoing the user's choice; and that same
2392
+ * pass clamps efforts against `modelsCache`, so a catalog missing from there
2393
+ * makes the effort we just derived read as unverifiable.
2394
+ */
2395
+ private admitRunner;
2396
+ /**
2397
+ * Replace one task's dependency list. Separate from {@link updateTask}
2398
+ * because a hand-edited graph is the one task edit that can leave a plan
2399
+ * unschedulable: `canSetDependencies` is the guard, and it lives behind this
2400
+ * one method so the API and both surfaces' pickers reject the same edits
2401
+ * rather than each carrying a copy of the rule. Throws so a surface can say
2402
+ * why the edit was refused.
2403
+ */
2404
+ setTaskDependencies(taskId: string, dependencies: string[]): LegacyPlanState | null;
2405
+ completeTask(taskId: string): Promise<void>;
2406
+ removeTask(taskId: string): LegacyPlanState | null;
2407
+ /**
2408
+ * Add one task, filling in whatever the caller left unset. A task with no
2409
+ * runnable assignment is not a lighter task but an unspawnable one, so the
2410
+ * runner falls back to the plan's first and the model, effort and mode are
2411
+ * derived from that runner's catalog — the same derivation a runner change
2412
+ * uses ({@link runnerAssignment}), which is why this is async like
2413
+ * {@link setTaskRunner}. Anything the caller did choose survives when the
2414
+ * runner offers it.
2415
+ *
2416
+ * Dependencies naming tasks that don't exist are dropped rather than rejected:
2417
+ * the caller is a picker over the current plan, so a stale id means the plan
2418
+ * moved on, not that the whole task should be refused.
2419
+ */
2420
+ addTask(draft: Partial<Task>): Promise<LegacyPlanState | null>;
2421
+ mergeTasks(taskIdA: string, taskIdB: string): LegacyPlanState | null;
2422
+ mergeMultipleTasks(taskIds: string[]): LegacyPlanState | null;
2423
+ splitTask(taskId: string, newTasks: Partial<Task>[]): LegacyPlanState | null;
2424
+ /**
2425
+ * Planner-driven merge: validate compatibility up front, then ask the planner
2426
+ * LLM to produce a single "merge" taskOps op combining the selected tasks.
2427
+ * Goes through the same conversation loop + validated-atomic-edit + corrective
2428
+ * retry flow as every other task_ops edit (ADR-0002). Throws on a
2429
+ * pre-flight compatibility failure so the host surfaces an inline error
2430
+ * before any LLM call.
2431
+ */
2432
+ requestMerge(taskIds: string[], options?: GeneratePlanOptions): Promise<LegacyPlanState>;
2433
+ /**
2434
+ * Planner-driven split: ask the planner LLM to decompose one task into a
2435
+ * sequence of smaller tasks. The model generates the breakdown (no manual
2436
+ * per-task specs from the user). Same conversation-loop/repair path as merge.
2437
+ */
2438
+ requestSplit(taskId: string, options?: GeneratePlanOptions): Promise<LegacyPlanState>;
2439
+ loadPlan(plan: LegacyPlanState, goal: string, workspace: string, opts?: {
2440
+ sessionId?: string;
2441
+ persist?: boolean;
2442
+ }): void;
2443
+ modifyPlan(userRequest: string): Promise<Task[]>;
2444
+ destroy(): void;
2445
+ private broadcastPlan;
2446
+ get aiServiceInstance(): IAiService;
2447
+ }
2448
+
2449
+ type VerdictListener = (taskId: string, verdict: Verdict, output: string) => void;
2450
+ type CheckpointListener = (taskId: string, summary: string) => void;
2451
+ declare class VerdictEngine {
2452
+ private markerSeen;
2453
+ private buffers;
2454
+ private checkpointCounts;
2455
+ private pausedSessions;
2456
+ private listeners;
2457
+ private checkpointListeners;
2458
+ /**
2459
+ * Per-task generation counter. Incremented on every watch() and clear().
2460
+ * Stale callbacks (from a prior session whose generation doesn't match
2461
+ * the current one) bail out instead of delivering a verdict for the
2462
+ * wrong session.
2463
+ */
2464
+ private generations;
2465
+ onVerdict(listener: VerdictListener): void;
2466
+ onCheckpoint(listener: CheckpointListener): void;
2467
+ approveCheckpoint(taskId: string): void;
2468
+ rejectCheckpoint(taskId: string, reason: string): void;
2469
+ /**
2470
+ * Attach to a spawned session: buffer output, scan for the task's completion
2471
+ * marker (delivering a verdict immediately while leaving interactive sessions
2472
+ * open), scan for checkpoint markers, and on exit produce a failed verdict
2473
+ * when the marker was never observed.
2474
+ */
2475
+ watch(task: Task, session: ITerminalSession): void;
2476
+ /** Manual "Mark complete" override: a pass verdict that bypasses evidence. */
2477
+ markComplete(task: Task): Verdict;
2478
+ /** Clear verification state for a task (used on retry). */
2479
+ clear(task: Task): void;
2480
+ /** Drop all tracking state (used on stop / loadPlan). */
2481
+ reset(): void;
2482
+ private decide;
2483
+ }
2484
+
2485
+ /**
2486
+ * The ids a runner may actually run: the user's allowlist with the ids that
2487
+ * provably belong to a *different* runner dropped. `undefined` means no
2488
+ * restriction.
2489
+ *
2490
+ * Model ids are scoped to the agent that lists them, so an OpenRouter slug
2491
+ * allowlisted for Claude Code does not limit it — it points it at something it
2492
+ * cannot spawn, and the plan only dies once a task is already running. But
2493
+ * "this runner didn't list it" is not enough to call an id wrong: discovery can
2494
+ * be stale, and a plugin runner may have no list at all. What settles it is
2495
+ * whether *another* runner listed the id. So:
2496
+ *
2497
+ * - listed for this runner → keep;
2498
+ * - listed for no runner → unknown, and the user said it explicitly, so keep it
2499
+ * and let the runner validate last (the same call `coerceAssignments` and
2500
+ * `TaskRetarget` make);
2501
+ * - listed only for other runners → provably not this runner's, so drop it.
2502
+ *
2503
+ * When that leaves nothing, the whole allowlist was about some other runner
2504
+ * (a settings file written before a surface scoped its picker to one). No
2505
+ * restriction is the only safe reading — the alternative is handing the planner
2506
+ * a list of models that cannot run.
2507
+ */
2508
+ declare function effectiveAllowlist(allowlist: string[] | undefined, runner: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>> | undefined): string[] | undefined;
2509
+ declare function filterModelsForPrompt(modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, perRunnerAllowlist: Partial<Record<RunnerId, string[]>>): Partial<Record<RunnerId, DiscoveredModel[]>>;
2510
+ /**
2511
+ * Clamp a planner/user-supplied thinking effort to what the model actually
2512
+ * offers. Invalid values map to the nearest rung the model supports (so a
2513
+ * Claude-style "xhigh" on a low/medium/high model becomes "high") and fall
2514
+ * back to undefined — the runner's own default — when no mapping exists.
2515
+ */
2516
+ declare function clampThinkingEffort(effort: string | undefined, variants: {
2517
+ id: string;
2518
+ }[]): string | undefined;
2519
+ declare function coerceAssignments(tasks: Task[], perRunnerAllowlist: Partial<Record<RunnerId, string[]>>, allowedRunners?: RunnerId[], modelsByRunner?: Partial<Record<RunnerId, DiscoveredModel[]>>): Task[];
2520
+
2521
+ /** What a runner offers a task: the models discovered for it and the modes its manifest declares. */
2522
+ interface RunnerCatalog {
2523
+ models: DiscoveredModel[];
2524
+ modes: RunnerModeInfo[];
2525
+ }
2526
+ /**
2527
+ * The model, thinking effort and mode a runner offers for a task.
2528
+ *
2529
+ * A task's model, effort and mode are all scoped to its runner:
2530
+ * `claude-sonnet-4-5` on Codex, or `acceptEdits` on OpenCode, are not degraded
2531
+ * choices but unspawnable ones. So this is where any task acquires a runnable
2532
+ * assignment — a runner change (below) and a hand-added task derive it the same
2533
+ * way. Each field in `current` is preserved when the runner also offers it, and
2534
+ * otherwise snapped to that runner's preferred entry (discovery already sorts
2535
+ * models by the manifest's `preferredPatterns`; `modes[0]` is the manifest's own
2536
+ * first choice).
2537
+ *
2538
+ * An empty catalog means discovery failed or the runner is a plugin we have no
2539
+ * list for — not that the runner offers nothing. That field is left out of the
2540
+ * patch and the runner validates last, matching `coerceAssignments`.
2541
+ */
2542
+ declare function runnerAssignment(catalog: RunnerCatalog, current?: Pick<Task, 'assignedModel' | 'taskMode'>): Partial<Task>;
2543
+ /**
2544
+ * Move a task onto a different runner, carrying its model, thinking effort and
2545
+ * mode over to values that runner actually offers. A runner change is never a
2546
+ * single-field edit — it either brings the other three with it or leaves the
2547
+ * task unrunnable.
2548
+ *
2549
+ * Returns the patch to apply, or `{}` when there is nothing to change.
2550
+ */
2551
+ declare function retargetTaskRunner(task: Task, runner: RunnerId, catalog: RunnerCatalog): Partial<Task>;
2552
+
2553
+ declare function buildResearchToolsPrompt(subagentsEnabled?: boolean): string;
2554
+ /**
2555
+ * System prompt for one read-only research subagent (issue #34). The digest
2556
+ * contract matters: the reply goes back to the planner as a tool result, so it
2557
+ * must be dense, self-contained, and carry exact file paths — never questions,
2558
+ * never a task plan.
2559
+ */
2560
+ declare function buildSubagentSystemPrompt(): string;
2561
+ declare function buildConversationSystemPrompt(goal: string, context: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean, grillMeEnabled?: boolean, prdEnabled?: boolean, reviewEnabled?: boolean, verificationEnabled?: boolean,
2562
+ /** Harness planner (ADR-0009): the agent owns its own tools and research budget. */
2563
+ harnessMode?: boolean): string;
2564
+ declare const CORE_PLANNER_PROMPT: string;
2565
+ declare function buildResearchPrompt(userGoal: string, context: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes): string;
2566
+ declare function buildPlanWithResults(userGoal: string, context: string, researchResults: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes): string;
2567
+ declare function buildModifyPlanPrompt(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, aiflowContext?: string, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): string;
2568
+ declare function modelContextBlock(modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): string;
2569
+ declare function executionLogBlock(executionLog: TaskSnapshot[]): string;
2570
+ declare function pendingEditRulesBlock(): string;
2571
+ declare function buildModifyDuringExecutionPrompt(executionLog: TaskSnapshot[], pendingTasks: string, userMessage: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): string;
2572
+ /**
2573
+ * User-message prompt for a planner-driven merge: asks the model to combine the
2574
+ * selected tasks into one. Routed through the conversation loop, so the
2575
+ * task-ops protocol (with the "merge" op) is injected alongside it by
2576
+ * `planContextBlock`. The model emits a single taskOps merge op.
2577
+ */
2578
+ declare function buildMergePrompt(taskIds: string[], tasks: Task[]): string;
2579
+ /**
2580
+ * User-message prompt for a planner-driven split: asks the model to decompose
2581
+ * one task into a sequence of smaller tasks. The model decides the breakdown —
2582
+ * the user does not hand-type the parts.
2583
+ */
2584
+ declare function buildSplitPrompt(taskId: string, tasks: Task[]): string;
2585
+
2586
+ /**
2587
+ * The one owner of "the model emitted something unusable — correct it and
2588
+ * retry". Every repair path in the planner routes through this module:
2589
+ *
2590
+ * - {@link repairLoop} is the bounded driver (first reply → interpret →
2591
+ * corrective re-send), used by plan generation ({@link generatePlanWithRepair}),
2592
+ * the Session's task-ops settlement, and Planner.modifyDuringExecution.
2593
+ * - {@link classifyPlannerReply} decides what a planner reply *is* — a plan,
2594
+ * targeted task edits, a botched attempt at either (worth a corrective
2595
+ * retry), or prose. The conversation loop in BaseAiService keeps its own
2596
+ * driver (it also runs the tool rounds) but delegates classification here.
2597
+ * - The corrective prompt texts live here, once.
2598
+ *
2599
+ * Policies stay at the call sites (PRD nudge, ops validation, abort guards) —
2600
+ * they are inputs to the loop, not part of it.
2601
+ */
2602
+ type RepairVerdict<T> = {
2603
+ done: T;
2604
+ } | {
2605
+ retry: {
2606
+ errors: string[];
2607
+ corrective: string;
2608
+ cause?: unknown;
2609
+ };
2610
+ };
2611
+ interface RepairLoopOpts<R, T> {
2612
+ /** Produce the first reply (attempt 0) — often already in hand. */
2613
+ first: () => R | Promise<R>;
2614
+ /** Ask the model again with a corrective prompt after a retryable failure. */
2615
+ resend: (corrective: string) => Promise<R>;
2616
+ /**
2617
+ * Judge one reply: `done` with the settled value, or `retry` with the
2618
+ * errors and the corrective prompt to send. Throw for non-retryable
2619
+ * failures — those propagate immediately.
2620
+ */
2621
+ interpret: (reply: R) => RepairVerdict<T> | Promise<RepairVerdict<T>>;
2622
+ /** Corrective re-sends allowed after the first attempt (N repairs = N+1 attempts). */
2623
+ maxRepairs: number;
2624
+ /** Budget exhausted: receives the last reply and its errors; return a fallback or throw. */
2625
+ onExhausted: (last: {
2626
+ reply: R;
2627
+ errors: string[];
2628
+ cause?: unknown;
2629
+ }) => T;
2630
+ }
2631
+ declare function repairLoop<R, T>(opts: RepairLoopOpts<R, T>): Promise<T>;
2632
+ /** Instruction appended to a follow-up request asking the model to re-emit strict JSON. */
2633
+ declare const JSON_REPAIR_INSTRUCTION: string;
2634
+ /**
2635
+ * The emission hit the output-token limit: a plain "re-send the JSON" retry
2636
+ * would be cut off at the same point, so this one asks for terser output.
2637
+ */
2638
+ declare const TRUNCATED_PLAN_REPAIR_INSTRUCTION: string;
2639
+ /** A plan attempt was botched (invalid or truncated JSON): re-emit the whole plan. */
2640
+ declare function reEmitPlanPrompt(detail: string): string;
2641
+ /** A plan emission was truncated mid-JSON; the caller may also have compacted the history to free input context. */
2642
+ declare function truncatedPlanReEmitPrompt(historyCompacted: boolean): string;
2643
+ /** A task-ops attempt was botched (unparseable JSON): re-emit the ops object. */
2644
+ declare function reEmitTaskOpsPrompt(detail: string): string;
2645
+ /** Task ops parsed but failed semantic validation (cycles, unknown refs, locked tasks). */
2646
+ declare function taskOpsRejectedPrompt(errors: string[]): string;
2647
+ /** A modified plan failed validation during execution: full re-prompt feedback block. */
2648
+ declare function modifyValidationFeedback(errors: string[]): string;
2649
+ type PlannerReplyClassification = {
2650
+ kind: 'plan';
2651
+ tasks: Task[];
2652
+ } | {
2653
+ kind: 'task_ops';
2654
+ ops: TaskOp[];
2655
+ }
2656
+ /** Clearly attempted a plan (tasks-keyed object, or JSON cut off mid-stream) and botched it — worth a corrective retry. */
2657
+ | {
2658
+ kind: 'broken_plan';
2659
+ error: PlanParseError;
2660
+ }
2661
+ /** Clearly attempted an ops object and botched it — worth a corrective retry. */
2662
+ | {
2663
+ kind: 'broken_task_ops';
2664
+ error: PlanParseError;
2665
+ } | {
2666
+ kind: 'prose';
2667
+ };
2668
+ /**
2669
+ * Classify one planner reply. Task ops are checked before the plan key — an
2670
+ * ops object never carries a top-level tasks array, but a model may mention
2671
+ * the word in prose around it. "Broken" is deliberately narrower than "failed
2672
+ * to parse": prose that merely mentions the envelope key is left alone.
2673
+ */
2674
+ declare function classifyPlannerReply(text: string, opts: {
2675
+ runners: RunnerId[];
2676
+ runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
2677
+ autonomousDefault?: boolean;
2678
+ }): PlannerReplyClassification;
2679
+ /**
2680
+ * Generate a plan, retrying on unparseable JSON. `generate` is called with an
2681
+ * optional repair hint (undefined on the first attempt, {@link JSON_REPAIR_INSTRUCTION}
2682
+ * thereafter) and must return the model's raw text. Only {@link PlanParseError} is
2683
+ * retried; transport/other errors propagate immediately. The last parse error is
2684
+ * re-thrown if every attempt fails.
2685
+ */
2686
+ declare function generatePlanWithRepair(generate: (repairHint?: string) => Promise<string>, runners: RunnerId[], maxAttempts?: number, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): Promise<Task[]>;
2687
+
2688
+ type RunnerMode = string;
2689
+
2690
+ declare function buildRunnerInvocation(opts: {
2691
+ runner: string;
2692
+ prompt: string;
2693
+ modelId?: string;
2694
+ thinkingEffort?: string;
2695
+ modelVariants?: string[];
2696
+ mode?: RunnerMode;
2697
+ headless?: boolean;
2698
+ registry: RunnerRegistry;
2699
+ }): RunnerInvocation;
2700
+
2701
+ type ExecFileFn = (command: string, args: string[]) => Promise<{
2702
+ stdout: string;
2703
+ stderr: string;
2704
+ }>;
2705
+ /** Test seam: every OS touchpoint is injectable; production uses the defaults. */
2706
+ interface TmuxRunnerDeps {
2707
+ port: number;
2708
+ execFileImpl?: ExecFileFn;
2709
+ resolvePath?: () => Promise<string>;
2710
+ pollIntervalMs?: number;
2711
+ logDir?: string;
2712
+ }
2713
+ declare class TmuxSession extends AbstractTerminalSession {
2714
+ private tmuxSession;
2715
+ private windowName;
2716
+ private execFileImpl;
2717
+ private pollIntervalMs;
2718
+ private logPath;
2719
+ private socket;
2720
+ private outputBuffer;
2721
+ private offset;
2722
+ private timer;
2723
+ constructor(id: string, taskId: string, tmuxSession: string, windowName: string, execFileImpl: ExecFileFn, pollIntervalMs: number, logPath: string, socket: string);
2724
+ private get target();
2725
+ /** Every tmux call must name the daemon's own socket; an unprefixed one
2726
+ * silently targets the shared default server (see `tmuxSocketName`). */
2727
+ private tmux;
2728
+ /**
2729
+ * Runs the invocation in a fresh tmux window, tailed via `pipe-pane` into a
2730
+ * log file rather than polling `capture-pane` snapshots — a byte-exact
2731
+ * stream needs no diff/resync heuristics. The command is wrapped so the
2732
+ * window survives its own exit (an `exec`'d login shell), which is what
2733
+ * lets the user keep poking at a finished task's terminal.
2734
+ */
2735
+ start(command: string, args: string[], cwd: string, env?: Record<string, string>): Promise<void>;
2736
+ private poll;
2737
+ /**
2738
+ * A task finishing (or being killed) stops observation, but never the
2739
+ * window itself on the sentinel path — only `kill()` closes the window.
2740
+ * Piping is turned off and the log file removed either way, so a finished
2741
+ * task doesn't leave a `cat` process appending to it for the rest of the
2742
+ * daemon's life.
2743
+ */
2744
+ private finish;
2745
+ kill(): void;
2746
+ getOutput(): string;
2747
+ write(text: string): void;
2748
+ }
2749
+ /**
2750
+ * Runs tasks in a real tmux window instead of a piped subprocess, so a user
2751
+ * can open a genuine, interactive terminal on any task (running or
2752
+ * finished) from outside Ordewell's own process. `ITerminalSession` hides the
2753
+ * transport from `VerdictEngine`/`PoolAwareRunner`, so orchestration is
2754
+ * unchanged — this is a drop-in replacement for `HeadlessRunner`.
2755
+ */
2756
+ declare class TmuxRunner extends AbstractRunner<TmuxSession> {
2757
+ private sessionName;
2758
+ private socket;
2759
+ private execFileImpl;
2760
+ private resolvePath;
2761
+ private pollIntervalMs;
2762
+ private logDir;
2763
+ /** Retries get a freshly named window so a failed attempt's output stays inspectable. */
2764
+ private attempts;
2765
+ private ready;
2766
+ constructor(deps: TmuxRunnerDeps);
2767
+ /**
2768
+ * Reaps a session orphaned by a crashed prior daemon on the same port, then
2769
+ * creates a fresh one. Memoized: `spawn` awaits it too, so a task spawned
2770
+ * before startup's own call has settled never lands in a missing session —
2771
+ * and a settled failure clears the memo so the next spawn can retry.
2772
+ */
2773
+ ensureSession(): Promise<void>;
2774
+ /** Every tmux call must name this daemon's socket; see `tmuxSocketName`. */
2775
+ private tmux;
2776
+ private createFreshSession;
2777
+ /**
2778
+ * Makes an attached runner terminal scrollable with the mouse wheel and
2779
+ * Page Up/Down. A fresh tmux session ships with `mouse off`, a 2000-line
2780
+ * scrollback, and no PageUp binding — none of which lets a user page back
2781
+ * through a finished task's output, which is the whole point of opening its
2782
+ * window. Scoping is harmless: the daemon owns a private socket
2783
+ * (`tmuxSocketName`), so these server-wide options touch nothing but its
2784
+ * own session. Best-effort so an ancient or stripped tmux cannot stop tasks
2785
+ * from spawning — the session still works without the scroll comforts.
2786
+ */
2787
+ private configureScrolling;
2788
+ /**
2789
+ * Called on daemon shutdown — the one guarantee against leaked tmux
2790
+ * processes. Kills the whole server, not just the session: the socket
2791
+ * belongs to this daemon alone, so nothing else can be on it, and a runner
2792
+ * that somehow escaped its session would otherwise keep running (and keep
2793
+ * billing) unattached for as long as the server lived.
2794
+ */
2795
+ killSession(): Promise<void>;
2796
+ spawn(opts: {
2797
+ taskId: string;
2798
+ runner: string;
2799
+ prompt: string;
2800
+ modelId?: string;
2801
+ thinkingEffort?: string;
2802
+ modelVariants?: string[];
2803
+ mode?: string;
2804
+ headless?: boolean;
2805
+ cwd: string;
2806
+ registry?: RunnerRegistry;
2807
+ planSessionId?: string;
2808
+ }): Promise<ITerminalSession>;
2809
+ }
2810
+
2811
+ declare function stripAnsi(text: string): string;
2812
+ declare function posixShellQuote(s: string): string;
2813
+ /**
2814
+ * Wrap a command for execution inside a POSIX login shell, which is what
2815
+ * resolves runner binaries managed by nvm/volta/asdf.
2816
+ *
2817
+ * Deliberately POSIX-only. This used to take a `platform` and emit
2818
+ * `powershell.exe -Command "'claude' '-p' '…'"` for Windows, which PowerShell
2819
+ * cannot run at all: a quoted string in leading position is parsed in
2820
+ * expression mode, so the invocation died with a parse error before the runner
2821
+ * started — a task that failed instantly, every time, on that platform. Windows
2822
+ * has no login-shell equivalent to emulate (its PATH comes from the registry
2823
+ * and is already inherited), so {@link planShellLaunch} starts the runner
2824
+ * directly there instead of routing it through a shell. Platform choice belongs
2825
+ * to that function; this one only knows how to phrase the POSIX half.
2826
+ */
2827
+ declare function buildShellInvocation(command: string, args: string[]): {
2828
+ shellPath: string;
2829
+ shellArgs: string[];
2830
+ };
2831
+ /**
2832
+ * Wrap a command in `script` to allocate the PTY some runners require when
2833
+ * headless; `-e` propagates the child's exit code so verification still works.
2834
+ *
2835
+ * POSIX-only by nature — there is no `script` on Windows, which
2836
+ * `HeadlessRunner`'s `hasScriptCmd` probe already discovers, so this is never
2837
+ * reached there.
2838
+ */
2839
+ declare function wrapWithPty(command: string, args: string[]): {
2840
+ command: string;
2841
+ args: string[];
2842
+ };
2843
+
2844
+ interface KillTreeDeps {
2845
+ platform?: NodeJS.Platform;
2846
+ /** Runs `taskkill`. Injected so the Windows path is testable off Windows. */
2847
+ execFileImpl?: (file: string, args: string[], cb: (err: Error | null) => void) => void;
2848
+ /** Schedules the forced kill. Injected so tests need no timers. */
2849
+ setTimeoutImpl?: (fn: () => void, ms: number) => {
2850
+ unref?: () => void;
2851
+ };
2852
+ clearTimeoutImpl?: (handle: unknown) => void;
2853
+ }
2854
+ /**
2855
+ * Terminate `proc` and everything it started.
2856
+ *
2857
+ * Returns immediately; the forced follow-up (SIGKILL, or `taskkill /F`) is
2858
+ * scheduled and unref'd, so a disposed session is never the reason the host
2859
+ * process refuses to exit.
2860
+ */
2861
+ declare function killTree(proc: ChildProcess | null, deps?: KillTreeDeps): void;
2862
+
2863
+ type ExecFn = (command: string, options?: {
2864
+ timeout?: number;
2865
+ }) => Promise<{
2866
+ stdout: string;
2867
+ }>;
2868
+ /**
2869
+ * Exported for test: the Windows arm cannot be exercised from a Linux CI box
2870
+ * through {@link augmentedPath}, which reads `process.platform` and the real
2871
+ * environment. Production always calls it through the no-argument form.
2872
+ */
2873
+ declare function wellKnownBinDirs(deps?: {
2874
+ platform?: NodeJS.Platform;
2875
+ env?: NodeJS.ProcessEnv;
2876
+ home?: string;
2877
+ }): string[];
2878
+ /**
2879
+ * The augmented PATH for spawning user-installed CLI tools. Resolved once per
2880
+ * process and cached (the login shell query costs ~50-200ms).
2881
+ */
2882
+ declare function augmentedPath(execImpl?: ExecFn): Promise<string>;
2883
+ /** Test seam: drop the per-process cache so the next call re-resolves. */
2884
+ declare function clearAugmentedPathCache(): void;
2885
+ /**
2886
+ * Build a child environment whose PATH is `resolvedPath`, with exactly one
2887
+ * PATH-ish key in it.
2888
+ *
2889
+ * `{ ...process.env, PATH }` is the obvious spelling and it is wrong on
2890
+ * Windows. Node's `process.env` is a case-insensitive proxy, but spreading it
2891
+ * yields the OS's actual casing — `Path` — so adding `PATH` produces an
2892
+ * environment block carrying both. Which one the child sees is undefined, and
2893
+ * the loser is silently discarded. Every existing site passed the same value
2894
+ * under both keys, so the bug was latent rather than live; this makes it
2895
+ * impossible instead of unlikely.
2896
+ */
2897
+ declare function withPath(base: NodeJS.ProcessEnv, resolvedPath: string, overrides?: Record<string, string>): NodeJS.ProcessEnv;
2898
+
2899
+ /** One shared tmux session per daemon, scoped by port so multiple daemons never collide. */
2900
+ declare function tmuxSessionName(port: number): string;
2901
+ /**
2902
+ * Each daemon gets its own tmux *server*, not just its own session name.
2903
+ *
2904
+ * A plain `tmux new-session` attaches to whatever server already owns the
2905
+ * default socket, and a session created on an existing server inherits that
2906
+ * **server's** environment — not the environment of the process that asked for
2907
+ * it (tmux only refreshes `update-environment` vars, and only on attach). So a
2908
+ * daemon started with, say, a particular provider API key would silently hand
2909
+ * its runners a stale key left behind by whoever started the server first,
2910
+ * possibly hours earlier under a different configuration entirely.
2911
+ *
2912
+ * A private socket makes the server a child of this daemon, so the runner
2913
+ * inherits the daemon's environment the way any child process would, and
2914
+ * `kill-server` at shutdown is authoritative — a runner cannot outlive the
2915
+ * daemon by hiding on a shared server. Users attaching by hand need the same
2916
+ * flag: `tmux -L ordewell-<port> attach -t ordewell-<port>`.
2917
+ */
2918
+ declare function tmuxSocketName(port: number): string;
2919
+ /**
2920
+ * tmux window targeting breaks on `:` and other punctuation, so ids are
2921
+ * slugged. Task ids are only unique within one plan ("task-1" is every
2922
+ * planner's favourite), so the window is also scoped by the plan session id —
2923
+ * without it, the second plan run in a daemon's lifetime would collide with
2924
+ * the first plan's still-open windows and pipe its output into the wrong one.
2925
+ */
2926
+ declare function tmuxWindowName(taskId: string, planSessionId?: string): string;
2927
+ type ProbeFn = () => void;
2928
+ /** Feature-detects tmux the same way `HeadlessRunner` detects `script`. */
2929
+ declare function hasTmux(probe?: ProbeFn): boolean;
2930
+
2931
+ declare class FsPluginStore implements IPluginStore {
2932
+ getUserPluginsDir(): string;
2933
+ listUserPluginDirs(): string[];
2934
+ loadManifest(pluginDir: string): RunnerPluginManifest | null;
2935
+ copyDir(sourceDir: string, destDir: string): void;
2936
+ removeDir(dir: string): void;
2937
+ ensureDir(dir: string): void;
2938
+ writeFile(filePath: string, content: string): void;
2939
+ readFile(filePath: string): string | null;
2940
+ dirExists(path: string): boolean;
2941
+ exists(path: string): boolean;
2942
+ }
2943
+
2944
+ declare function resolveArgs(manifest: RunnerPluginManifest, ctx: ResolveContext): RunnerInvocation;
2945
+
2946
+ declare const CLAUDE_CODE_MANIFEST: RunnerPluginManifest;
2947
+
2948
+ declare const OPENCODE_MANIFEST: RunnerPluginManifest;
2949
+
2950
+ declare const STATE_DIR = ".ordewell";
2951
+ declare function ensureDir(dir: string): void;
2952
+ declare function getStateDir(baseDir?: string): string;
2953
+
2954
+ declare function saveState(plan: LegacyPlanState, baseDir?: string): void;
2955
+ declare function loadState(baseDir?: string, logger?: ILogger): LegacyPlanState | null;
2956
+ declare function clearState(baseDir?: string): void;
2957
+ declare function stateExists(baseDir?: string): boolean;
2958
+
2959
+ declare function saveSession(plan: LegacyPlanState, goal: string, baseDir?: string, id?: string): SessionMeta;
2960
+ declare function listSessions(baseDir?: string, logger?: ILogger): SessionMeta[];
2961
+ declare function loadSession(sessionId: string, baseDir?: string, logger?: ILogger): {
2962
+ meta: SessionMeta;
2963
+ plan: LegacyPlanState;
2964
+ } | null;
2965
+ declare function loadSessionPlanState(sessionId: string, baseDir?: string, logger?: ILogger): {
2966
+ meta: SessionMeta;
2967
+ plan: PlanState;
2968
+ } | null;
2969
+ declare function getLatestSession(baseDir?: string, logger?: ILogger): {
2970
+ meta: SessionMeta;
2971
+ plan: LegacyPlanState;
2972
+ } | null;
2973
+ declare function deleteSession(sessionId: string, baseDir?: string, logger?: ILogger): boolean;
2974
+
2975
+ interface PrdBlock {
2976
+ slug: string;
2977
+ markdown: string;
2978
+ }
2979
+ /**
2980
+ * Detect a full markdown PRD in a planner message. The markers are the only
2981
+ * structured artifact left in the conversation loop — they exist so the PRD
2982
+ * can be saved to disk, not to drive any state machine.
2983
+ */
2984
+ declare function extractPrdBlock(text: string): PrdBlock | null;
2985
+ /** Kebab-case, path-safe feature slug. */
2986
+ declare function sanitizeSlug(raw: string): string;
2987
+ /** Save the PRD to `<workspace>/.scratch/<slug>/PRD.md` (to-prd native convention). */
2988
+ declare function savePrdMarkdown(workspace: string, slug: string, markdown: string): string;
2989
+
2990
+ export { ALL_PROVIDERS, AUTO_COMMANDS, AbstractRunner, AbstractTerminalSession, ActiveTaskSession, type AgentAdapter, type AgentAdapterFactory, type AgentEvent, type AgentProcessDeps, type AgentStartOptions, AiProvider, ApprovalKind, ApprovalMode, ApprovalRequest, ApprovalSource, BaseAiService, BaseConfig, BaseFileSystem, CLAUDE_CODE_MANIFEST, CLI_PROVIDERS, CMD_EXE_MAX_COMMAND_LINE, CORE_PLANNER_PROMPT, type CappedRows, type CheckpointListener, ClaudeCodeAdapter, CliAgentAiService, type CliAgentAiServiceDeps, CodexAdapter, type CommandClassification, CommandLineTooLongError, type CommandTier, ConsoleLogger, ContextCollector, ConversationMessage, type ConversationRequest, type ConversationTurn, DiscoveredModel, EmbeddedNewlineError, EnvConfig, type ExecFileFn, type ExecImpl, FindSymbolOptions, FsPluginStore, GIT_READONLY_SUBCOMMANDS, GeminiService, GlobOptions, type GrepInvocation, GrepOptions, HeadlessRunner, type HeadlessRunnerDeps, type IAiService, IApproval, IConfig, IFileSystem, type ILogger, type INotification, IPluginStore, ITerminalRunner, ITerminalSession, JSON_REPAIR_INSTRUCTION, type KillTreeDeps, type LaunchDeps, type LaunchPlan, LegacyPlanState, LineBuffer, type MappedTool, ModelResolver, type ModelResolverDeps, type ModifyPlanRequest, type NotificationAction, OPENCODE_MANIFEST, OpenAiService, OpenCodeAdapter, type OrchestratorObserver, OrchestratorOption, PROVIDER_DETECT_PRIORITY, PROVIDER_LABEL, PROVIDER_PRIORITY, PROVIDER_SHORT_LABEL, type PendingApproval, PendingApprovals, type PendingApprovalsOptions, PlanParseError, type PlanRequest, PlanState, PlanStatus, PlanStore, Planner, type PlannerReplyClassification, type PlannerRuntimeToggles, type PrdBlock, type ProbeFn, ProviderModelLists, type ProviderRegistration, QueuedMessage, REFUSED_COMMANDS, ReadFileOpts, type RepairLoopOpts, type RepairVerdict, type ResearchChat, ResearchLogEntry, ResearchProgress, type ResearchShell, type ResearchShellDeps, ResearchStep, ResearchToolType, type ResearchTurn, ResolveContext, type RunnerCatalog, RunnerId, RunnerInstallation, RunnerInvocation, type RunnerMode, RunnerModeInfo, RunnerPluginManifest, RunnerRegistry, STATE_DIR, SYMBOL_LANGUAGES, type SerializedPlan, type SerializedTask, type SerializedTaskStatus, Session, type SessionBroadcaster, type SessionData, type SessionDeps, type SessionMessage, type SessionMeta, type SessionPlanner, type SessionRuntimeSettings, SettingsService, type ShellDialect, type SpawnSpec, StdioAgentAdapter, TRUNCATED_PLAN_REPAIR_INSTRUCTION, Task, TaskOp, TaskOrchestrator, TaskSnapshot, TmuxRunner, type TmuxRunnerDeps, type ToolCall, ToolOutcome, type ToolResult, type UserSettings, Verdict, VerdictEngine, type VerdictListener, WINDOWS_MAX_COMMAND_LINE, applyHeadLimit, augmentedPath, buildConversationSystemPrompt, buildFallbackGrepArgs, buildGlobArgs, buildGrepArgs, buildMergePrompt, buildModifyDuringExecutionPrompt, buildModifyPlanPrompt, buildPlanWithResults, buildResearchPrompt, buildResearchToolsPrompt, buildRunnerInvocation, buildShellInvocation, buildSplitPrompt, buildSubagentSystemPrompt, clampThinkingEffort, classifyCommand, classifyPlannerReply, clearAugmentedPathCache, clearResearchShellCache, clearState, coerceAssignments, configuredProviders, createAiService, defaultLogger, definitionPattern, deleteSession, discoverGeminiModels, effectiveAllowlist, ensureDir, executionLogBlock, executionSummary, extractPrdBlock, filterFallbackByAnchoredInclude, filterModelsForPrompt, formatSearchOutput, generatePlanWithRepair, getLatestSession, getProviderMeta, getSettingsPath, getStateDir, grantScopeFor, hasTmux, includeGlobFor, isCliProvider, isOpenAiProvider, killTree, languageForId, listSessions, loadSession, loadSessionPlanState, loadState, mapAgentTool, modelContextBlock, modifyValidationFeedback, normalizeAgentArgs, normalizeGeminiModel, pendingEditRulesBlock, planDirectLaunch, planShellLaunch, posixShellQuote, prefixModelId, providerForRunner, reEmitPlanPrompt, reEmitTaskOpsPrompt, referencePattern, repairLoop, researchShellWarning, researchToolsPath, resolveArgs, resolveProviderFromPrefix, resolveResearchShell, resolveWithin, retargetTaskRunner, runnerAssignment, runnerForProvider, sanitizeSlug, savePrdMarkdown, saveSession, saveState, serializePlan, serializeTask, serializeTaskStatus, sessionRuntimeSettings, stateExists, stripAnsi, stripModelPrefix, taskOpsRejectedPrompt, tmuxSessionName, tmuxSocketName, tmuxWindowName, truncatedPlanReEmitPrompt, wellKnownBinDirs, windowsCommandLine, withPath, wrapWithPty };