@ordewell/core 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +211 -0
- package/README.md +37 -0
- package/dist/ITerminalRunner-Bbpe8uts.d.ts +535 -0
- package/dist/ITerminalRunner-D8Bf-HD4.d.mts +535 -0
- package/dist/Task-1NwmUImI.d.mts +264 -0
- package/dist/Task-1NwmUImI.d.ts +264 -0
- package/dist/chunk-4KIWPW4K.mjs +510 -0
- package/dist/chunk-4KIWPW4K.mjs.map +1 -0
- package/dist/chunk-SRGQBGI5.mjs +463 -0
- package/dist/chunk-SRGQBGI5.mjs.map +1 -0
- package/dist/chunk-YXAIQKVE.mjs +398 -0
- package/dist/chunk-YXAIQKVE.mjs.map +1 -0
- package/dist/index.d.mts +2990 -0
- package/dist/index.d.ts +2990 -0
- package/dist/index.js +12927 -0
- package/dist/index.js.map +1 -0
- package/dist/index.mjs +11449 -0
- package/dist/index.mjs.map +1 -0
- package/dist/parsing-RjN9JbyG.d.ts +161 -0
- package/dist/parsing-xpozY5ac.d.mts +161 -0
- package/dist/parsing.d.mts +2 -0
- package/dist/parsing.d.ts +2 -0
- package/dist/parsing.js +432 -0
- package/dist/parsing.js.map +1 -0
- package/dist/parsing.mjs +15 -0
- package/dist/parsing.mjs.map +1 -0
- package/dist/plan-utils-Cgc0cCH1.d.ts +111 -0
- package/dist/plan-utils-nfHKSbd3.d.mts +111 -0
- package/dist/plan-utils.d.mts +2 -0
- package/dist/plan-utils.d.ts +2 -0
- package/dist/plan-utils.js +181 -0
- package/dist/plan-utils.js.map +1 -0
- package/dist/plan-utils.mjs +18 -0
- package/dist/plan-utils.mjs.map +1 -0
- package/dist/testing.d.mts +28 -0
- package/dist/testing.d.ts +28 -0
- package/dist/testing.js +113 -0
- package/dist/testing.js.map +1 -0
- package/dist/testing.mjs +86 -0
- package/dist/testing.mjs.map +1 -0
- package/package.json +92 -0
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,2990 @@
|
|
|
1
|
+
import { h as RunnerId, c as PlanStatus, L as LegacyPlanState, d as ResearchProgress, a as DiscoveredModel, T as Task, R as ResearchLogEntry, C as ConversationMessage, n as TaskSnapshot, Q as QueuedMessage, A as ActiveTaskSession, g as ResearchToolType, u as Verdict, e as ResearchStep, b as PlanState } from './Task-1NwmUImI.js';
|
|
2
|
+
export { D as DiscoveredMode, M as Message, P as PlanModificationWarnings, f as ResearchStepOutcome, S as StreamEvent, i as StreamStepEvent, j as StreamThinkingEvent, k as TaskMode, l as TaskModelAssignment, m as TaskOutputSummary, o as TaskStatus, p as TaskType, q as ThinkingBlock, U as UserPromptEntry, r as UserStep, V as ValidationCheck, s as ValidationContext, t as ValidationResult, v as VerificationCheck, w as addTaskToPlan, x as createEmptyPlan, y as createTask, z as emptyWarnings, B as flattenTasks, E as migrateLegacyPlan, F as migratePlanState, G as migrateTask, H as removeTaskFromPlan, I as renumberTasks, J as updateTaskInPlan, K as validateModifiedPlan, N as warningsText } from './Task-1NwmUImI.js';
|
|
3
|
+
import { n as IFileSystem, I as IApproval, R as ReadFileOpts, T as ToolOutcome, k as GrepOptions, j as GlobOptions, i as FindSymbolOptions, f as ApprovalRequest, m as IConfig, A as AiProvider, y as ProviderModelLists, c as ApprovalMode, q as ITerminalSession, p as ITerminalRunner, J as RunnerRegistry, s as OrchestratorOption, b as ApprovalKind, g as ApprovalSource, E as RunnerInvocation, o as IPluginStore, H as RunnerPluginManifest, B as ResolveContext } from './ITerminalRunner-Bbpe8uts.js';
|
|
4
|
+
export { a as AllProviderModels, d as ApprovalPolicy, e as ApprovalPolicyOptions, C as CatalogModel, D as DENY_ALL, h as DiscoveryCommand, F as FetchAllProviderModelsOptions, G as GREP_DEFAULT_HEAD_LIMIT, l as GrepOutputMode, M as ModelCatalog, r as ModelShortcut, O as ORCHESTRATOR_SHORTCUTS, P as PluginEntry, t as PluginFeatures, u as PluginMode, v as PluginModelDiscovery, w as PluginRunnerDef, x as ProviderCredentialSource, z as ProviderModelsResult, S as SEARCH_EXCLUSIONS, K as collectProviderCredentials, L as enabledRunners, N as fetchAllProviderModels, Q as knownModelId, U as resolveModelShortcut, V as resolveProvider, W as toOrchestratorOptions } from './ITerminalRunner-Bbpe8uts.js';
|
|
5
|
+
import { R as RunnerModeInfo, b as PlanParseError } from './parsing-RjN9JbyG.js';
|
|
6
|
+
export { M as ManifestLookup, P as PLAN_ENVELOPE_KEY, a as PartialPlanTask, T as TASK_OPS_ENVELOPE_KEY, c as buildModeGuide, d as checkDepsResolve, e as checkImmutableLog, f as checkInProgress, g as checkNoCycles, h as checkUniqueIds, i as escapeControlCharsInStrings, j as extractJsonObject, k as extractObjectWithBalance, l as extractObjectsWithKey, m as filteredBuildModes, n as looksLikePlanAttempt, p as parsePartialPlan, o as parsePlanJson, r as resolveDefaultMode, q as resolveTaskMode, s as runnerModesFrom, t as stripModelNoise, u as stripTrailingCommas, v as validatePlanModification } from './parsing-RjN9JbyG.js';
|
|
7
|
+
import { T as TaskOp } from './plan-utils-Cgc0cCH1.js';
|
|
8
|
+
export { A as ApplyTaskOpsResult, a as TaskRef, b as applyTaskOps, c as canMergeTasks, d as canSetDependencies, e as canSplitTask, f as classifyOutcome, g as dependencyCandidates, h as dependentsOf, p as parseTaskOpsJson, s as summarizeToolCall, t as textHasTaskOps } from './plan-utils-Cgc0cCH1.js';
|
|
9
|
+
import { ChildProcess } from 'child_process';
|
|
10
|
+
import { EventEmitter } from 'events';
|
|
11
|
+
|
|
12
|
+
interface SessionMeta {
|
|
13
|
+
id: string;
|
|
14
|
+
goal: string;
|
|
15
|
+
runners: RunnerId[];
|
|
16
|
+
taskCount: number;
|
|
17
|
+
status: PlanStatus;
|
|
18
|
+
createdAt: string;
|
|
19
|
+
updatedAt: string;
|
|
20
|
+
}
|
|
21
|
+
interface SessionData {
|
|
22
|
+
meta: SessionMeta;
|
|
23
|
+
/** The full plan state — including conversationHistory, prdMarkdown, and queuedMessages. */
|
|
24
|
+
plan: LegacyPlanState;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Which interpreter runs the planner's `bash` tool, and in which language.
|
|
29
|
+
*
|
|
30
|
+
* `AUTO_COMMANDS` — the read-only tier that runs with no prompt — is `ls`, `cat`,
|
|
31
|
+
* `wc`, `head`, `tail`, `du`, `df`, `file`, `sort`, `uniq`, `basename`, `find`,
|
|
32
|
+
* `grep`. That set is POSIX, and passing it to `shell: true` on Windows means
|
|
33
|
+
* cmd.exe, where most of those names do not exist and `find` is an unrelated
|
|
34
|
+
* program that returns plausible wrong output instead of an error. The planner's
|
|
35
|
+
* whole research surface degraded to guessing, quietly.
|
|
36
|
+
*
|
|
37
|
+
* Rather than write a Windows dialect of every research command — a second
|
|
38
|
+
* behavior to keep in step forever — this finds a POSIX shell to run the
|
|
39
|
+
* existing ones in. Git for Windows ships a full one, and git is already a
|
|
40
|
+
* prerequisite for everything Ordewell does, so in practice it is there. Failing
|
|
41
|
+
* that, cmd.exe is used and the classifier is told so, because the one thing
|
|
42
|
+
* that must never happen is classifying a command in one language and running it
|
|
43
|
+
* in another: `commandPolicy` reads {@link ResearchShell.dialect} for exactly
|
|
44
|
+
* that reason.
|
|
45
|
+
*
|
|
46
|
+
* POSIX resolves to `{ file: null }`, meaning "use `shell: true`" — byte for
|
|
47
|
+
* byte what the adapters did before this module existed.
|
|
48
|
+
*
|
|
49
|
+
* Windows paths are built with `path.win32` explicitly, not the host-flavoured
|
|
50
|
+
* `path`, so the probe is the same on a Windows host as it is under a Linux
|
|
51
|
+
* test runner.
|
|
52
|
+
*/
|
|
53
|
+
type ShellDialect = 'posix' | 'cmd';
|
|
54
|
+
interface ResearchShell {
|
|
55
|
+
/**
|
|
56
|
+
* Executable that takes a command string, or null to use the host default
|
|
57
|
+
* via `shell: true`.
|
|
58
|
+
*/
|
|
59
|
+
file: string | null;
|
|
60
|
+
/** Arguments preceding the command string. */
|
|
61
|
+
args: string[];
|
|
62
|
+
/** The language the command string will be interpreted in. */
|
|
63
|
+
dialect: ShellDialect;
|
|
64
|
+
/**
|
|
65
|
+
* Directory holding this shell's POSIX utilities, when it brings its own.
|
|
66
|
+
* Prepended to PATH for the search subprocesses (`grep`, `tree`) that
|
|
67
|
+
* `execFile` starts without a shell, so they resolve too.
|
|
68
|
+
*/
|
|
69
|
+
utilsDir: string | null;
|
|
70
|
+
}
|
|
71
|
+
interface ResearchShellDeps {
|
|
72
|
+
platform?: NodeJS.Platform;
|
|
73
|
+
exists?: (candidate: string) => boolean;
|
|
74
|
+
env?: NodeJS.ProcessEnv;
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* The shell the planner's `bash` tool should use on this host. Resolved once per
|
|
78
|
+
* process — the answer cannot change while Ordewell runs, and the probe costs a
|
|
79
|
+
* handful of `stat` calls.
|
|
80
|
+
*/
|
|
81
|
+
declare function resolveResearchShell(deps?: ResearchShellDeps): ResearchShell;
|
|
82
|
+
/** Test seam: drop the per-process cache so the next call re-probes. */
|
|
83
|
+
declare function clearResearchShellCache(): void;
|
|
84
|
+
/**
|
|
85
|
+
* PATH for the search subprocesses the adapters start without a shell, with the
|
|
86
|
+
* research shell's own utilities in front. Without this, `grep` — the fallback
|
|
87
|
+
* when ripgrep is absent — is unresolvable on a Windows box even when a
|
|
88
|
+
* perfectly good `grep.exe` sits in the Git tree beside the shell.
|
|
89
|
+
*/
|
|
90
|
+
declare function researchToolsPath(shell: ResearchShell, basePath: string | undefined): string;
|
|
91
|
+
/**
|
|
92
|
+
* A message explaining a degraded research surface, or null when there is
|
|
93
|
+
* nothing to explain. Surfaced by the adapters on the first refused command so
|
|
94
|
+
* the limitation is visible rather than inferred from bad answers.
|
|
95
|
+
*/
|
|
96
|
+
declare function researchShellWarning(shell: ResearchShell): string | null;
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Tiered classification for the planner's `bash` tool.
|
|
100
|
+
*
|
|
101
|
+
* The planner is a read-only researcher: it never edits files, and the runners
|
|
102
|
+
* do the real work. But research legitimately means running things — querying a
|
|
103
|
+
* cloud control plane (`az`, `gh`), reproducing a failure (`npm test`), or
|
|
104
|
+
* shaping output (`jq`). The old allowlist refused all of that, so the model's
|
|
105
|
+
* only escape was to guess.
|
|
106
|
+
*
|
|
107
|
+
* Three tiers replace the flat allowlist:
|
|
108
|
+
*
|
|
109
|
+
* auto read-only inspection — runs with no prompt (the historical list)
|
|
110
|
+
* ask anything else that is not obviously destructive — one approval,
|
|
111
|
+
* remembered for the rest of the session at `scope` granularity
|
|
112
|
+
* refuse writes, privilege escalation, and anything that would smuggle
|
|
113
|
+
* arbitrary code past this classifier — never runs, never prompts
|
|
114
|
+
*
|
|
115
|
+
* `refuse` is deliberately not promptable. A planner that can `rm` is a planner
|
|
116
|
+
* that can silently break the workspace it was asked to reason about, and the
|
|
117
|
+
* architecture already says mutation belongs to the runners.
|
|
118
|
+
*
|
|
119
|
+
* Classification walks every segment of the command line (pipes, `&&`, `;`, and
|
|
120
|
+
* command substitution) rather than matching substrings against the raw string.
|
|
121
|
+
* The old substring denylist both over-matched (`ls docs/removed` tripped `rm`)
|
|
122
|
+
* and under-matched (`$(rm -rf /)` was invisible once chaining was allowed).
|
|
123
|
+
*
|
|
124
|
+
* The lexer is dialect-aware, because the interpreter that will actually run
|
|
125
|
+
* the command decides what the tokens are, and getting that wrong is not a
|
|
126
|
+
* cosmetic error here — it is the difference between classifying what runs and
|
|
127
|
+
* classifying something else. See {@link Dialect}.
|
|
128
|
+
*/
|
|
129
|
+
|
|
130
|
+
declare const AUTO_COMMANDS: string[];
|
|
131
|
+
declare const GIT_READONLY_SUBCOMMANDS: string[];
|
|
132
|
+
/**
|
|
133
|
+
* Never runs, with or without approval.
|
|
134
|
+
*
|
|
135
|
+
* The `cmd.exe` builtins at the end matter as much as the POSIX names above
|
|
136
|
+
* them. `del`, `rd`, `move`, and friends mutate exactly what `rm` and `mv` do,
|
|
137
|
+
* and on Windows they are what a model reaches for — so without them the whole
|
|
138
|
+
* refusal tier was bypassable on that platform by writing the command the way
|
|
139
|
+
* the platform spells it. Listed unconditionally rather than per-platform: a
|
|
140
|
+
* POSIX box has no `del` to refuse, so the extra names cost nothing there.
|
|
141
|
+
*/
|
|
142
|
+
declare const REFUSED_COMMANDS: string[];
|
|
143
|
+
/**
|
|
144
|
+
* Options threaded through classification.
|
|
145
|
+
*
|
|
146
|
+
* `dialect` is keyed to the interpreter rather than the OS on purpose: a Windows
|
|
147
|
+
* host that has Git Bash runs the planner's commands in a POSIX shell (see
|
|
148
|
+
* {@link resolveResearchShell}), and classifying those under cmd.exe rules
|
|
149
|
+
* would be the same mismatch in the other direction. Callers pass the dialect of
|
|
150
|
+
* the shell they are actually going to use; omitting it falls back to the host
|
|
151
|
+
* default.
|
|
152
|
+
*/
|
|
153
|
+
interface CommandPolicyOptions {
|
|
154
|
+
dialect?: ShellDialect;
|
|
155
|
+
}
|
|
156
|
+
type CommandTier = 'auto' | 'ask' | 'refuse';
|
|
157
|
+
interface CommandClassification {
|
|
158
|
+
tier: CommandTier;
|
|
159
|
+
/** What a grant covers, for `ask`. Derived from the non-auto segments only. */
|
|
160
|
+
scope: string;
|
|
161
|
+
/** Populated for `refuse`: why, in a sentence the model can act on. */
|
|
162
|
+
reason?: string;
|
|
163
|
+
}
|
|
164
|
+
/**
|
|
165
|
+
* Classify one command line. Output redirection is refused outright: a planner
|
|
166
|
+
* that writes files has stopped being a planner.
|
|
167
|
+
*/
|
|
168
|
+
declare function classifyCommand(command: string, opts?: CommandPolicyOptions): CommandClassification;
|
|
169
|
+
|
|
170
|
+
/**
|
|
171
|
+
* The policy half of the planner's filesystem: path confinement, the tiered
|
|
172
|
+
* `bash` gate, and the definition-first symbol lookup. Adapters supply only the
|
|
173
|
+
* mechanics (`*Impl`), so the rules live in one place across the web server and
|
|
174
|
+
* the VS Code extension rather than being re-derived per surface.
|
|
175
|
+
*
|
|
176
|
+
* Every public method resolves and authorizes before delegating, and every
|
|
177
|
+
* `*Impl` therefore receives an absolute, already-approved path. Previously
|
|
178
|
+
* each adapter did its own resolution, and `path.isAbsolute(p) ? p : …` meant
|
|
179
|
+
* an absolute path walked straight out of the workspace with no prompt and no
|
|
180
|
+
* record.
|
|
181
|
+
*/
|
|
182
|
+
declare abstract class BaseFileSystem implements IFileSystem {
|
|
183
|
+
private approval;
|
|
184
|
+
/**
|
|
185
|
+
* The interpreter `execBashImpl` will hand the command to. Owned here rather
|
|
186
|
+
* than per-adapter because {@link classifyCommand} has to be told the same
|
|
187
|
+
* answer: a command lexed under POSIX rules and then run by cmd.exe is a
|
|
188
|
+
* command this class did not actually classify.
|
|
189
|
+
*/
|
|
190
|
+
protected readonly researchShell: ResearchShell;
|
|
191
|
+
/** Surfaces inject the human channel here; without it, external access is denied. */
|
|
192
|
+
setApproval(approval: IApproval): void;
|
|
193
|
+
abstract getWorkspaceRoot(): string;
|
|
194
|
+
protected abstract readFileImpl(absPath: string, opts?: ReadFileOpts): Promise<ToolOutcome>;
|
|
195
|
+
protected abstract globImpl(pattern: string, absRoot: string, headLimit: number): Promise<ToolOutcome>;
|
|
196
|
+
protected abstract grepImpl(pattern: string, absRoot: string, opts: GrepOptions): Promise<ToolOutcome>;
|
|
197
|
+
protected abstract listDirImpl(absPath: string, depth: number): Promise<ToolOutcome>;
|
|
198
|
+
protected abstract execBashImpl(command: string): Promise<ToolOutcome>;
|
|
199
|
+
/**
|
|
200
|
+
* Resolve `p` and confirm the planner may touch it. In-workspace paths pass
|
|
201
|
+
* silently; anything else needs one approval, remembered per containing
|
|
202
|
+
* directory so a second file in the same place does not prompt again.
|
|
203
|
+
*/
|
|
204
|
+
protected authorizePath(p: string, kind?: 'file' | 'directory'): Promise<{
|
|
205
|
+
ok: true;
|
|
206
|
+
abs: string;
|
|
207
|
+
} | {
|
|
208
|
+
ok: false;
|
|
209
|
+
outcome: ToolOutcome;
|
|
210
|
+
}>;
|
|
211
|
+
readFile(p: string, opts?: ReadFileOpts): Promise<ToolOutcome>;
|
|
212
|
+
readFiles(paths: string[]): Promise<ToolOutcome>;
|
|
213
|
+
glob(pattern: string, opts?: GlobOptions): Promise<ToolOutcome>;
|
|
214
|
+
grep(pattern: string, opts?: GrepOptions): Promise<ToolOutcome>;
|
|
215
|
+
listDir(p: string, depth?: number): Promise<ToolOutcome>;
|
|
216
|
+
/**
|
|
217
|
+
* Definitions first, then a reference tally. Two bounded searches beat one
|
|
218
|
+
* unbounded `grep` because the 100-row budget gets spent on the rows that
|
|
219
|
+
* answer the question.
|
|
220
|
+
*/
|
|
221
|
+
findSymbol(symbol: string, opts?: FindSymbolOptions): Promise<ToolOutcome>;
|
|
222
|
+
/**
|
|
223
|
+
* Path confinement for `bash`: an `auto`-tier binary (`cat`, `find`, `rg`, …)
|
|
224
|
+
* is only auto because *reading* is read-only — its arguments can still
|
|
225
|
+
* name a path outside the workspace, which is the exact escape confinement
|
|
226
|
+
* closes for `readFile`/`glob`/`grep`. Each escaping path needs its own
|
|
227
|
+
* approval (scoped to its containing directory); approving one does not
|
|
228
|
+
* approve another, so a single command touching two external dirs prompts
|
|
229
|
+
* once per distinct scope rather than carrying the first grant to the rest.
|
|
230
|
+
*/
|
|
231
|
+
private authorizeCommandPaths;
|
|
232
|
+
/**
|
|
233
|
+
* Three tiers (see `commandPolicy.ts`): read-only inspection runs silently,
|
|
234
|
+
* anything else asks once and is remembered, and writes are refused outright
|
|
235
|
+
* because a planner that mutates the workspace has stopped being a planner.
|
|
236
|
+
*/
|
|
237
|
+
bash(command: string): Promise<ToolOutcome>;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* Workspace containment, in one place. Both the filesystem policy layer
|
|
242
|
+
* (which may ask the user to approve an escape) and the research-subagent
|
|
243
|
+
* wrapper (which may not, because nothing can prompt on its behalf) need the
|
|
244
|
+
* same answer to "is this path inside the workspace", so neither re-derives it.
|
|
245
|
+
*
|
|
246
|
+
* Symlinks are resolved lexically, not on disk: a symlink inside the workspace
|
|
247
|
+
* pointing outward still reads as inside. Closing that would mean a `realpath`
|
|
248
|
+
* syscall on every path check, and the threat model here is an LLM wandering,
|
|
249
|
+
* not an adversary planting links in a repo the user already trusts enough to
|
|
250
|
+
* point Ordewell at.
|
|
251
|
+
*/
|
|
252
|
+
interface ResolvedPath {
|
|
253
|
+
abs: string;
|
|
254
|
+
inside: boolean;
|
|
255
|
+
}
|
|
256
|
+
declare function resolveWithin(root: string, target: string): ResolvedPath;
|
|
257
|
+
/** The directory a grant covers for `target` — its parent for files, itself for directories. */
|
|
258
|
+
declare function grantScopeFor(abs: string, kind: 'file' | 'directory'): string;
|
|
259
|
+
|
|
260
|
+
/**
|
|
261
|
+
* The bridge between "core needs an answer" and "a human somewhere is looking
|
|
262
|
+
* at a UI". Core cannot prompt: the human may be at a browser, a TUI, a CLI
|
|
263
|
+
* stream, or a VS Code webview, and on the web server they are on the far end
|
|
264
|
+
* of a socket. So the Session parks a promise here, announces the request
|
|
265
|
+
* through the normal broadcast seam, and every surface answers through the
|
|
266
|
+
* same `resolve(id, granted)` call.
|
|
267
|
+
*
|
|
268
|
+
* Timeouts are load-bearing rather than defensive: a planner turn that blocks
|
|
269
|
+
* forever on an unanswered prompt would hang the whole research loop with no
|
|
270
|
+
* visible cause. On expiry the request resolves to denied and the model gets a
|
|
271
|
+
* normal, actionable tool result.
|
|
272
|
+
*/
|
|
273
|
+
interface PendingApproval {
|
|
274
|
+
id: string;
|
|
275
|
+
request: ApprovalRequest;
|
|
276
|
+
createdAt: string;
|
|
277
|
+
}
|
|
278
|
+
interface PendingApprovalsOptions {
|
|
279
|
+
/** Denies and resolves after this long with no answer. Default 5 minutes. */
|
|
280
|
+
timeoutMs?: number;
|
|
281
|
+
/** Announce a new request to the surfaces. */
|
|
282
|
+
onRequest?: (pending: PendingApproval) => void;
|
|
283
|
+
/** Announce that a request is no longer actionable (answered or expired). */
|
|
284
|
+
onSettled?: (id: string, granted: boolean) => void;
|
|
285
|
+
}
|
|
286
|
+
declare class PendingApprovals {
|
|
287
|
+
private readonly opts;
|
|
288
|
+
private readonly entries;
|
|
289
|
+
constructor(opts?: PendingApprovalsOptions);
|
|
290
|
+
/** Park a request and return the promise the approval policy awaits. */
|
|
291
|
+
ask(request: ApprovalRequest): Promise<boolean>;
|
|
292
|
+
/** Answer one request. Returns false when the id is unknown or already settled. */
|
|
293
|
+
resolve(id: string, granted: boolean): boolean;
|
|
294
|
+
/** Everything still awaiting an answer — replayed to a surface that connects late. */
|
|
295
|
+
outstanding(): PendingApproval[];
|
|
296
|
+
/** Deny everything in flight. Called on abort and on session reset. */
|
|
297
|
+
clear(): void;
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
/**
|
|
301
|
+
* Definition-shaped search patterns for the `find_symbol` tool.
|
|
302
|
+
*
|
|
303
|
+
* Plain `grep` is bad at the question the planner actually asks. Searching for
|
|
304
|
+
* `VerdictEngine` returns every import, call site, comment and string alongside
|
|
305
|
+
* the one line that defines it, and with a hard result cap the definition is
|
|
306
|
+
* often not even in the returned page. `find_symbol` spends the budget on
|
|
307
|
+
* declarations first and reports references as a count plus a sample.
|
|
308
|
+
*
|
|
309
|
+
* This is deliberately regex over a real parser. An LSP-grade answer would mean
|
|
310
|
+
* shipping per-language servers and waiting for them to index — the wrong trade
|
|
311
|
+
* for the half of the architecture whose whole point is being cheap and fast.
|
|
312
|
+
* The planner needs to scope tasks ("defined here, used across ~14 files in 3
|
|
313
|
+
* packages"), not to prove rename-safety; that is the runner's job, and runners
|
|
314
|
+
* have their own tools.
|
|
315
|
+
*
|
|
316
|
+
* Patterns target the Rust regex syntax ripgrep uses: non-capturing groups and
|
|
317
|
+
* `\b` are available, lookaround and backreferences are not.
|
|
318
|
+
*/
|
|
319
|
+
interface SymbolLanguage {
|
|
320
|
+
id: string;
|
|
321
|
+
extensions: string[];
|
|
322
|
+
/** Keywords that introduce a declaration in `kw Name` position. */
|
|
323
|
+
keywords: string[];
|
|
324
|
+
}
|
|
325
|
+
declare const SYMBOL_LANGUAGES: SymbolLanguage[];
|
|
326
|
+
declare function languageForId(id: string): SymbolLanguage | undefined;
|
|
327
|
+
/** The `--glob` filter that narrows a search to one language's files. */
|
|
328
|
+
declare function includeGlobFor(language: SymbolLanguage): string;
|
|
329
|
+
/**
|
|
330
|
+
* Build the definition pattern for `symbol`. Three shapes cover essentially
|
|
331
|
+
* every mainstream language:
|
|
332
|
+
*
|
|
333
|
+
* 1. `keyword Name` — declarations across all of them
|
|
334
|
+
* 2. `Name = function|(` — assigned function expressions and arrows
|
|
335
|
+
* 3. `… Name(args) {` — C/Java/Go-style bodies with a return type
|
|
336
|
+
*
|
|
337
|
+
* Passing a `language` narrows shape 1 to that language's keywords, which cuts
|
|
338
|
+
* cross-language noise in polyglot repos.
|
|
339
|
+
*/
|
|
340
|
+
declare function definitionPattern(symbol: string, language?: SymbolLanguage): string;
|
|
341
|
+
/** Every mention of the symbol as a whole word — the reference side of the report. */
|
|
342
|
+
declare function referencePattern(symbol: string): string;
|
|
343
|
+
|
|
344
|
+
interface GrepInvocation {
|
|
345
|
+
args: string[];
|
|
346
|
+
/** Rows beyond this are dropped by the caller — see `applyHeadLimit`. */
|
|
347
|
+
headLimit: number;
|
|
348
|
+
}
|
|
349
|
+
/**
|
|
350
|
+
* Build the ripgrep invocation for one grep call.
|
|
351
|
+
*
|
|
352
|
+
* `--max-count` is deliberately absent. It caps matches *per file*, so the old
|
|
353
|
+
* `--max-count 100` on a 300-file hit returned ~30 000 lines, blew the 1 MB
|
|
354
|
+
* exec buffer, and surfaced as `{ success: false, output: '' }` — a silent
|
|
355
|
+
* empty result on exactly the broad searches where the model most needed a
|
|
356
|
+
* signal. The cap belongs at the row level, applied after the fact.
|
|
357
|
+
*/
|
|
358
|
+
declare function buildGrepArgs(pattern: string, opts: GrepOptions, root: string): GrepInvocation;
|
|
359
|
+
/** Build the ripgrep invocation that lists files matching a glob. */
|
|
360
|
+
declare function buildGlobArgs(pattern: string, root: string): string[];
|
|
361
|
+
/**
|
|
362
|
+
* The POSIX-grep fallback for machines without ripgrep. Ordering and `--sort`
|
|
363
|
+
* are unavailable.
|
|
364
|
+
*
|
|
365
|
+
* `-P` (PCRE) is required, not optional: patterns built in core — most
|
|
366
|
+
* notably `find_symbol`'s `definitionPattern` — use non-capturing groups and
|
|
367
|
+
* `\b`, which BRE has no syntax for and ERE (`-E`) still cannot express
|
|
368
|
+
* ((?:...) is a PCRE construct). Without `-P`, GNU grep either errors
|
|
369
|
+
* ("Unmatched \{") or, worse, silently treats `(`/`)`/`|` as literal
|
|
370
|
+
* characters and reports a confident empty result. `-P` and `-F` are
|
|
371
|
+
* mutually exclusive, so literal mode skips it.
|
|
372
|
+
*/
|
|
373
|
+
declare function buildFallbackGrepArgs(pattern: string, opts: GrepOptions, root: string): string[];
|
|
374
|
+
/**
|
|
375
|
+
* GNU grep's own `--include` glob only ever matches a basename, so it silently
|
|
376
|
+
* matches nothing against an anchored pattern like `subdir/*.txt` (there is no
|
|
377
|
+
* `/` in a basename to match against). The fallback path drops such patterns
|
|
378
|
+
* from the grep invocation and filters matches by relative path here instead.
|
|
379
|
+
*/
|
|
380
|
+
declare function filterFallbackByAnchoredInclude(stdout: string, include: string, anchor: string, outputMode?: GrepOptions['outputMode']): string;
|
|
381
|
+
interface CappedRows {
|
|
382
|
+
rows: string[];
|
|
383
|
+
truncated: boolean;
|
|
384
|
+
total: number;
|
|
385
|
+
}
|
|
386
|
+
/** Apply the global row cap and report honestly whether anything was dropped. */
|
|
387
|
+
declare function applyHeadLimit(stdout: string, headLimit: number): CappedRows;
|
|
388
|
+
/**
|
|
389
|
+
* Render capped rows for the model, with paths made workspace-relative and an
|
|
390
|
+
* explicit note when rows were dropped — a silently truncated list reads as a
|
|
391
|
+
* complete answer and the model plans against it.
|
|
392
|
+
*/
|
|
393
|
+
declare function formatSearchOutput(capped: CappedRows, root: string, opts: {
|
|
394
|
+
emptyMessage: string;
|
|
395
|
+
hint?: string;
|
|
396
|
+
sep?: string;
|
|
397
|
+
}): string;
|
|
398
|
+
|
|
399
|
+
declare function normalizeGeminiModel(id: string): string;
|
|
400
|
+
declare abstract class BaseConfig implements IConfig {
|
|
401
|
+
abstract aiProvider: AiProvider;
|
|
402
|
+
abstract apiKey: string;
|
|
403
|
+
abstract planningModel: string;
|
|
404
|
+
abstract enabledRunners: string[];
|
|
405
|
+
abstract setProviderModelLists(lists: ProviderModelLists): void;
|
|
406
|
+
get openAiBaseUrl(): string;
|
|
407
|
+
get openAiApiKey(): string;
|
|
408
|
+
get openrouterKey(): string;
|
|
409
|
+
get geminiKey(): string;
|
|
410
|
+
get geminiBaseUrl(): string | undefined;
|
|
411
|
+
get openaiCompatibleBaseUrl(): string;
|
|
412
|
+
get openaiCompatibleApiKey(): string;
|
|
413
|
+
get orchestratorModel(): string;
|
|
414
|
+
get researchSubagentModel(): string;
|
|
415
|
+
get geminiModel(): string;
|
|
416
|
+
get plannerThinkingEffort(): string | undefined;
|
|
417
|
+
get openaiBaseUrl(): string;
|
|
418
|
+
get openaiApiKey(): string;
|
|
419
|
+
get xaiBaseUrl(): string;
|
|
420
|
+
get xaiApiKey(): string;
|
|
421
|
+
get groqBaseUrl(): string;
|
|
422
|
+
get groqApiKey(): string;
|
|
423
|
+
get deepseekBaseUrl(): string;
|
|
424
|
+
get deepseekApiKey(): string;
|
|
425
|
+
get togetherBaseUrl(): string;
|
|
426
|
+
get togetherApiKey(): string;
|
|
427
|
+
get mistralBaseUrl(): string;
|
|
428
|
+
get mistralApiKey(): string;
|
|
429
|
+
get anthropicBaseUrl(): string;
|
|
430
|
+
get anthropicApiKey(): string;
|
|
431
|
+
get fireworksBaseUrl(): string;
|
|
432
|
+
get fireworksApiKey(): string;
|
|
433
|
+
get perplexityBaseUrl(): string;
|
|
434
|
+
get perplexityApiKey(): string;
|
|
435
|
+
get zhipuBaseUrl(): string;
|
|
436
|
+
get zhipuApiKey(): string;
|
|
437
|
+
get kimiBaseUrl(): string;
|
|
438
|
+
get kimiApiKey(): string;
|
|
439
|
+
get cerebrasBaseUrl(): string;
|
|
440
|
+
get cerebrasApiKey(): string;
|
|
441
|
+
get deepinfraBaseUrl(): string;
|
|
442
|
+
get deepinfraApiKey(): string;
|
|
443
|
+
get doubaoBaseUrl(): string;
|
|
444
|
+
get doubaoApiKey(): string;
|
|
445
|
+
get qwenBaseUrl(): string;
|
|
446
|
+
get qwenApiKey(): string;
|
|
447
|
+
get hunyuanBaseUrl(): string;
|
|
448
|
+
get hunyuanApiKey(): string;
|
|
449
|
+
get baichuanBaseUrl(): string;
|
|
450
|
+
get baichuanApiKey(): string;
|
|
451
|
+
get minimaxBaseUrl(): string;
|
|
452
|
+
get minimaxApiKey(): string;
|
|
453
|
+
get yiBaseUrl(): string;
|
|
454
|
+
get yiApiKey(): string;
|
|
455
|
+
get stepfunBaseUrl(): string;
|
|
456
|
+
get stepfunApiKey(): string;
|
|
457
|
+
get siliconflowBaseUrl(): string;
|
|
458
|
+
get siliconflowApiKey(): string;
|
|
459
|
+
get cohereBaseUrl(): string;
|
|
460
|
+
get cohereApiKey(): string;
|
|
461
|
+
get novitaBaseUrl(): string;
|
|
462
|
+
get novitaApiKey(): string;
|
|
463
|
+
getProviderBaseUrl(provider: AiProvider): string;
|
|
464
|
+
getProviderApiKey(provider: AiProvider): string;
|
|
465
|
+
get maxParallelSessions(): number;
|
|
466
|
+
get researchEnabled(): boolean;
|
|
467
|
+
get researchMaxSteps(): number;
|
|
468
|
+
get researchMaxFileSize(): number;
|
|
469
|
+
get planMapEnabled(): boolean;
|
|
470
|
+
get autonomousMode(): boolean;
|
|
471
|
+
get approvalMode(): ApprovalMode;
|
|
472
|
+
get approvalPreApproved(): string[];
|
|
473
|
+
protected static detectProvider(fallback: AiProvider): AiProvider;
|
|
474
|
+
}
|
|
475
|
+
|
|
476
|
+
/**
|
|
477
|
+
* A fully `process.env`-backed IConfig for headless callers (e.g. the CLI's
|
|
478
|
+
* `ordewell models` command) that need provider keys/base URLs but have no
|
|
479
|
+
* editor settings or secret store. All resolution lives in BaseConfig; this
|
|
480
|
+
* subclass only supplies the abstract members from the environment.
|
|
481
|
+
*/
|
|
482
|
+
declare class EnvConfig extends BaseConfig {
|
|
483
|
+
get aiProvider(): AiProvider;
|
|
484
|
+
get apiKey(): string;
|
|
485
|
+
get planningModel(): string;
|
|
486
|
+
get enabledRunners(): string[];
|
|
487
|
+
setProviderModelLists(): void;
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
type NotificationAction = {
|
|
491
|
+
label: string;
|
|
492
|
+
value: string;
|
|
493
|
+
};
|
|
494
|
+
interface INotification {
|
|
495
|
+
info(msg: string): void;
|
|
496
|
+
warn(msg: string): void;
|
|
497
|
+
error(msg: string): void;
|
|
498
|
+
confirm(msg: string, options: string[]): Promise<string | undefined>;
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
interface ILogger {
|
|
502
|
+
warn(scope: string, message: string, err?: unknown): void;
|
|
503
|
+
}
|
|
504
|
+
declare class ConsoleLogger implements ILogger {
|
|
505
|
+
warn(scope: string, message: string, err?: unknown): void;
|
|
506
|
+
}
|
|
507
|
+
declare const defaultLogger: ILogger;
|
|
508
|
+
|
|
509
|
+
interface IWebFetcher {
|
|
510
|
+
confirm(url: string): Promise<boolean>;
|
|
511
|
+
fetch(url: string): Promise<ToolOutcome>;
|
|
512
|
+
/**
|
|
513
|
+
* Optional web search. A fetcher without it makes the `web_search` tool
|
|
514
|
+
* report itself unavailable rather than failing the turn — the same
|
|
515
|
+
* degradation `fetch` already uses when no fetcher is wired at all.
|
|
516
|
+
*/
|
|
517
|
+
search?(query: string): Promise<ToolOutcome>;
|
|
518
|
+
}
|
|
519
|
+
|
|
520
|
+
interface UserSettings {
|
|
521
|
+
grillMe: {
|
|
522
|
+
enabled: boolean;
|
|
523
|
+
};
|
|
524
|
+
tdd: {
|
|
525
|
+
enabled: boolean;
|
|
526
|
+
};
|
|
527
|
+
prd: {
|
|
528
|
+
enabled: boolean;
|
|
529
|
+
};
|
|
530
|
+
review: {
|
|
531
|
+
enabled: boolean;
|
|
532
|
+
};
|
|
533
|
+
verification: {
|
|
534
|
+
enabled: boolean;
|
|
535
|
+
};
|
|
536
|
+
researchSubagents: {
|
|
537
|
+
enabled: boolean;
|
|
538
|
+
};
|
|
539
|
+
modelAllowlist?: Record<string, string[]>;
|
|
540
|
+
}
|
|
541
|
+
/**
|
|
542
|
+
* Where the user's toggles live. `ORDEWELL_SETTINGS_PATH` overrides it so several
|
|
543
|
+
* Ordewell processes on one machine can hold *different* settings at the same
|
|
544
|
+
* time. Without it the file is a single shared mutable global: the benchmark
|
|
545
|
+
* harness runs parallel lanes that each pin the mode toggles, and because
|
|
546
|
+
* `getAll()` re-reads whenever the mtime moves, a lane needing
|
|
547
|
+
* `researchSubagents: true` would silently plan with `false` the moment another
|
|
548
|
+
* lane pinned its own — turning an A/B of that toggle into a comparison of one
|
|
549
|
+
* condition against itself.
|
|
550
|
+
*/
|
|
551
|
+
declare function getSettingsPath(): string;
|
|
552
|
+
declare class SettingsService {
|
|
553
|
+
private filePath;
|
|
554
|
+
private cache;
|
|
555
|
+
private cachedMtimeMs;
|
|
556
|
+
constructor(filePath?: string);
|
|
557
|
+
getAll(): UserSettings;
|
|
558
|
+
private fileMtimeMs;
|
|
559
|
+
getGrillMe(): boolean;
|
|
560
|
+
getTdd(): boolean;
|
|
561
|
+
getPrd(): boolean;
|
|
562
|
+
setGrillMe(enabled: boolean): void;
|
|
563
|
+
setTdd(enabled: boolean): void;
|
|
564
|
+
setPrd(enabled: boolean): void;
|
|
565
|
+
getReview(): boolean;
|
|
566
|
+
setReview(enabled: boolean): void;
|
|
567
|
+
getVerification(): boolean;
|
|
568
|
+
setVerification(enabled: boolean): void;
|
|
569
|
+
getResearchSubagents(): boolean;
|
|
570
|
+
setResearchSubagents(enabled: boolean): void;
|
|
571
|
+
getModelAllowlist(runner: string): string[] | undefined;
|
|
572
|
+
setModelAllowlist(runner: string, ids: string[] | undefined): void;
|
|
573
|
+
private load;
|
|
574
|
+
private persist;
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
/** How the same toggle is named once a host has read it off disk. */
|
|
578
|
+
interface PlannerRuntimeToggles {
|
|
579
|
+
grillMeEnabled: boolean;
|
|
580
|
+
tddEnabled: boolean;
|
|
581
|
+
prdEnabled: boolean;
|
|
582
|
+
reviewEnabled: boolean;
|
|
583
|
+
verificationEnabled: boolean;
|
|
584
|
+
researchSubagentsEnabled: boolean;
|
|
585
|
+
}
|
|
586
|
+
/**
|
|
587
|
+
* The planner-facing mode set for one operation. Replaces the boolean tail that
|
|
588
|
+
* every planner signature used to carry positionally — where a thirteenth
|
|
589
|
+
* parameter was the only place left to put a new toggle.
|
|
590
|
+
*/
|
|
591
|
+
interface PlannerModes {
|
|
592
|
+
autonomousDefault: boolean;
|
|
593
|
+
grillMe: boolean;
|
|
594
|
+
prd: boolean;
|
|
595
|
+
review: boolean;
|
|
596
|
+
verification: boolean;
|
|
597
|
+
researchSubagents: boolean;
|
|
598
|
+
}
|
|
599
|
+
|
|
600
|
+
declare abstract class AbstractTerminalSession implements ITerminalSession {
|
|
601
|
+
id: string;
|
|
602
|
+
taskId: string;
|
|
603
|
+
protected exited: boolean;
|
|
604
|
+
protected outputEmitter: EventEmitter<any>;
|
|
605
|
+
protected exitEmitter: EventEmitter<any>;
|
|
606
|
+
constructor(id: string, taskId: string);
|
|
607
|
+
protected baseHandleExit(code: number): void;
|
|
608
|
+
onOutput(callback: (text: string) => void): void;
|
|
609
|
+
onExit(callback: (code: number) => void): void;
|
|
610
|
+
abstract kill(): void;
|
|
611
|
+
abstract getOutput(): string;
|
|
612
|
+
abstract write(text: string): void;
|
|
613
|
+
}
|
|
614
|
+
declare abstract class AbstractRunner<S extends ITerminalSession> implements ITerminalRunner {
|
|
615
|
+
protected sessions: Map<string, S>;
|
|
616
|
+
get activeCount(): number;
|
|
617
|
+
stop(sessionId: string): void;
|
|
618
|
+
stopAll(): void;
|
|
619
|
+
protected registerSession(id: string, session: S): void;
|
|
620
|
+
abstract spawn(opts: {
|
|
621
|
+
taskId: string;
|
|
622
|
+
runner: string;
|
|
623
|
+
prompt: string;
|
|
624
|
+
modelId?: string;
|
|
625
|
+
thinkingEffort?: string;
|
|
626
|
+
mode?: string;
|
|
627
|
+
headless?: boolean;
|
|
628
|
+
cwd: string;
|
|
629
|
+
registry?: RunnerRegistry;
|
|
630
|
+
}): Promise<ITerminalSession>;
|
|
631
|
+
}
|
|
632
|
+
|
|
633
|
+
/**
|
|
634
|
+
* cmd.exe's command-line buffer. A longer line is truncated rather than
|
|
635
|
+
* rejected, which would corrupt a planner's system prompt or a task's prompt
|
|
636
|
+
* mid-sentence and produce a confident answer to half a question — so the
|
|
637
|
+
* batch route refuses instead. See {@link CommandLineTooLongError}.
|
|
638
|
+
*/
|
|
639
|
+
declare const CMD_EXE_MAX_COMMAND_LINE = 8191;
|
|
640
|
+
/**
|
|
641
|
+
* CreateProcess's own ceiling, which the native and PowerShell routes are
|
|
642
|
+
* bounded by instead. Windows truncates here too, so the same refusal applies —
|
|
643
|
+
* it is simply four times further away.
|
|
644
|
+
*/
|
|
645
|
+
declare const WINDOWS_MAX_COMMAND_LINE = 32767;
|
|
646
|
+
interface LaunchPlan {
|
|
647
|
+
/** The executable handed to `spawn()` or `vscode.window.createTerminal`. */
|
|
648
|
+
file: string;
|
|
649
|
+
/** Arguments for `file`. Pass verbatim when {@link verbatim} is set. */
|
|
650
|
+
args: string[];
|
|
651
|
+
/**
|
|
652
|
+
* Windows batch route only: `args` is already a quoted command line and must
|
|
653
|
+
* not be re-quoted. Maps to `windowsVerbatimArguments` for `spawn`, and to
|
|
654
|
+
* the string form of `shellArgs` for a VS Code terminal.
|
|
655
|
+
*/
|
|
656
|
+
verbatim?: boolean;
|
|
657
|
+
}
|
|
658
|
+
/** Test seam: every OS touchpoint is injectable, and production uses the defaults. */
|
|
659
|
+
interface LaunchDeps {
|
|
660
|
+
platform?: NodeJS.Platform;
|
|
661
|
+
/** The PATH executables are looked up on. Defaults to the augmented PATH. */
|
|
662
|
+
resolvePath?: () => Promise<string>;
|
|
663
|
+
/** True when `candidate` names an existing file. */
|
|
664
|
+
exists?: (candidate: string) => boolean;
|
|
665
|
+
/** Absolute path to the Windows command interpreter. */
|
|
666
|
+
comSpec?: () => string;
|
|
667
|
+
/** Absolute path to Windows PowerShell. */
|
|
668
|
+
powerShell?: () => string;
|
|
669
|
+
/** PATHEXT, as the environment reports it. */
|
|
670
|
+
pathExt?: () => string;
|
|
671
|
+
}
|
|
672
|
+
/**
|
|
673
|
+
* Thrown when a command's arguments do not fit the buffer of the only
|
|
674
|
+
* interpreter that can start it. Windows truncates rather than rejecting, and a
|
|
675
|
+
* system prompt cut off mid-sentence makes the planner answer half a question
|
|
676
|
+
* confidently — the silent success this repo refuses — so this is raised
|
|
677
|
+
* instead. `TaskOrchestrator.startTask` catches it and holds the task, so the
|
|
678
|
+
* message is what the user reads: it names the fix, because they cannot infer
|
|
679
|
+
* it from a truncated prompt.
|
|
680
|
+
*/
|
|
681
|
+
declare class CommandLineTooLongError extends Error {
|
|
682
|
+
readonly command: string;
|
|
683
|
+
readonly length: number;
|
|
684
|
+
readonly limit: number;
|
|
685
|
+
constructor(command: string, length: number, limit?: number);
|
|
686
|
+
}
|
|
687
|
+
/**
|
|
688
|
+
* Thrown when only a batch shim resolved for a multi-line argument. cmd.exe
|
|
689
|
+
* reads up to the first CR/LF and discards the rest with no error and exit code
|
|
690
|
+
* 0 — quoting does not help — so the agent would get the first paragraph of its
|
|
691
|
+
* prompt without the completion marker instruction, then exit looking successful.
|
|
692
|
+
*/
|
|
693
|
+
declare class EmbeddedNewlineError extends Error {
|
|
694
|
+
readonly command: string;
|
|
695
|
+
constructor(command: string);
|
|
696
|
+
}
|
|
697
|
+
/** The verbatim command line cmd.exe receives after `/d /s /c`, before wrapping. */
|
|
698
|
+
declare function windowsCommandLine(file: string, args: string[]): string;
|
|
699
|
+
/**
|
|
700
|
+
* How to start `command` with `args` through `spawn()`, with no shell.
|
|
701
|
+
*
|
|
702
|
+
* POSIX returns its input unchanged — execvp already searches PATH, and adding
|
|
703
|
+
* a resolution step there would be a new way for a working setup to break.
|
|
704
|
+
*
|
|
705
|
+
* Windows resolves the command against PATH × PATHEXT, preferring a native
|
|
706
|
+
* executable (spawned directly) over a batch shim (through cmd.exe) over a
|
|
707
|
+
* PowerShell script shim (through `powershell.exe -File`). A command that
|
|
708
|
+
* resolves to nothing is returned unchanged, so the caller's existing ENOENT —
|
|
709
|
+
* which names what the user typed — is what surfaces rather than a second,
|
|
710
|
+
* vaguer error from here.
|
|
711
|
+
*
|
|
712
|
+
* The tiers are tried in preference order and the first that *fits* wins, with
|
|
713
|
+
* one deliberate exception: an overflowing batch shim does not fall through to
|
|
714
|
+
* PowerShell. Overflow means a very large prompt, which is precisely where
|
|
715
|
+
* `-File` argument fidelity is least worth betting on, and where a clear held
|
|
716
|
+
* task beats a plausibly-mangled one. So capacity does not reorder the tiers —
|
|
717
|
+
* a `.ps1` beside a too-long `.cmd` still raises.
|
|
718
|
+
*
|
|
719
|
+
* A line break does reorder them: cmd.exe cannot carry one at any length, so a
|
|
720
|
+
* `.ps1` beside a `.cmd` wins, and a lone `.cmd` raises.
|
|
721
|
+
*
|
|
722
|
+
* @throws {CommandLineTooLongError} when the selected route's buffer cannot
|
|
723
|
+
* carry the arguments.
|
|
724
|
+
* @throws {EmbeddedNewlineError} when the arguments span lines and only the
|
|
725
|
+
* batch route resolved.
|
|
726
|
+
*/
|
|
727
|
+
declare function planDirectLaunch(command: string, args: string[], deps?: LaunchDeps): Promise<LaunchPlan>;
|
|
728
|
+
/**
|
|
729
|
+
* How to start `command` for a surface that hands an executable and arguments
|
|
730
|
+
* to a terminal — the VS Code runner today, a Windows TUI later.
|
|
731
|
+
*
|
|
732
|
+
* On POSIX this is the login shell, unchanged: `bash -lc` runs the user's
|
|
733
|
+
* profile, which is how nvm/volta/asdf-managed runner binaries resolve at all.
|
|
734
|
+
* Windows has no login-shell equivalent (its PATH comes from the registry and
|
|
735
|
+
* is already inherited), so it takes the direct route instead. That is not just
|
|
736
|
+
* a simplification: it means the runner's own exit code is the terminal's exit
|
|
737
|
+
* code, rather than a `$LASTEXITCODE` that PowerShell propagates unreliably —
|
|
738
|
+
* and the exit code is half of what {@link VerdictEngine} judges a task on.
|
|
739
|
+
*/
|
|
740
|
+
declare function planShellLaunch(command: string, args: string[], deps?: LaunchDeps): Promise<LaunchPlan>;
|
|
741
|
+
|
|
742
|
+
type SpawnFn = (command: string, args: string[], options: {
|
|
743
|
+
env: NodeJS.ProcessEnv;
|
|
744
|
+
stdio: ['pipe', 'pipe', 'pipe'];
|
|
745
|
+
cwd: string;
|
|
746
|
+
/** Set by the Windows batch route, where `args` is already a quoted command line. */
|
|
747
|
+
windowsVerbatimArguments?: boolean;
|
|
748
|
+
}) => ChildProcess;
|
|
749
|
+
/** Test seam: every OS touchpoint is injectable; production uses the defaults. */
|
|
750
|
+
interface HeadlessRunnerDeps {
|
|
751
|
+
spawnImpl?: SpawnFn;
|
|
752
|
+
hasScriptCmd?: () => boolean;
|
|
753
|
+
resolvePath?: () => Promise<string>;
|
|
754
|
+
/**
|
|
755
|
+
* Overrides for executable resolution ({@link planDirectLaunch}). Only the
|
|
756
|
+
* Windows branch consults them, so a POSIX test never needs to pass anything.
|
|
757
|
+
*/
|
|
758
|
+
launchDeps?: LaunchDeps;
|
|
759
|
+
}
|
|
760
|
+
declare class HeadlessSession extends AbstractTerminalSession {
|
|
761
|
+
private spawnImpl;
|
|
762
|
+
private process;
|
|
763
|
+
private outputBuffer;
|
|
764
|
+
constructor(id: string, taskId: string, spawnImpl: SpawnFn);
|
|
765
|
+
start(launch: LaunchPlan, cwd: string, resolvedPath: string, env?: Record<string, string>): void;
|
|
766
|
+
kill(): void;
|
|
767
|
+
getOutput(): string;
|
|
768
|
+
write(text: string): void;
|
|
769
|
+
}
|
|
770
|
+
declare class HeadlessRunner extends AbstractRunner<HeadlessSession> {
|
|
771
|
+
private spawnImpl;
|
|
772
|
+
private hasScriptCmd;
|
|
773
|
+
private resolvePath;
|
|
774
|
+
private launchDeps;
|
|
775
|
+
constructor(deps?: HeadlessRunnerDeps);
|
|
776
|
+
spawn(opts: {
|
|
777
|
+
taskId: string;
|
|
778
|
+
runner: string;
|
|
779
|
+
prompt: string;
|
|
780
|
+
modelId?: string;
|
|
781
|
+
thinkingEffort?: string;
|
|
782
|
+
modelVariants?: string[];
|
|
783
|
+
mode?: string;
|
|
784
|
+
headless?: boolean;
|
|
785
|
+
cwd: string;
|
|
786
|
+
registry?: RunnerRegistry;
|
|
787
|
+
}): Promise<ITerminalSession>;
|
|
788
|
+
}
|
|
789
|
+
|
|
790
|
+
/**
|
|
791
|
+
* The harness-planner transport contract (ADR-0009).
|
|
792
|
+
*
|
|
793
|
+
* One adapter per coding agent, each speaking that agent's own programmatic
|
|
794
|
+
* protocol and normalizing it to the event union below. Everything above this
|
|
795
|
+
* line — reply classification, the repair loop, plan validation, the four
|
|
796
|
+
* surfaces — is already provider-agnostic, so an adapter is the entire cost of
|
|
797
|
+
* teaching Ordewell to plan with another agent.
|
|
798
|
+
*/
|
|
799
|
+
/**
|
|
800
|
+
* One normalized event from a running agent turn. Deliberately smaller than
|
|
801
|
+
* any single agent's native protocol: this is the intersection Ordewell can act
|
|
802
|
+
* on, not a lossless re-encoding. Event fidelity differs by agent — `thinking`
|
|
803
|
+
* is rich on Claude Code and absent elsewhere — so consumers must tolerate a
|
|
804
|
+
* turn that emits nothing but `assistant_text` and `turn_end`.
|
|
805
|
+
*/
|
|
806
|
+
type AgentEvent =
|
|
807
|
+
/** A chunk of the assistant's reply. Concatenated in order to form the turn's text. */
|
|
808
|
+
{
|
|
809
|
+
type: 'assistant_text';
|
|
810
|
+
text: string;
|
|
811
|
+
}
|
|
812
|
+
/** Reasoning the agent chose to expose. Never contributes to the reply text. */
|
|
813
|
+
| {
|
|
814
|
+
type: 'thinking';
|
|
815
|
+
text: string;
|
|
816
|
+
} | {
|
|
817
|
+
type: 'tool_call';
|
|
818
|
+
id: string;
|
|
819
|
+
name: string;
|
|
820
|
+
args: Record<string, unknown>;
|
|
821
|
+
} | {
|
|
822
|
+
type: 'tool_result';
|
|
823
|
+
id: string;
|
|
824
|
+
name: string;
|
|
825
|
+
output: string;
|
|
826
|
+
success: boolean;
|
|
827
|
+
}
|
|
828
|
+
/**
|
|
829
|
+
* The agent asked to do something its read-only mode does not cover. Always
|
|
830
|
+
* auto-denied (T1) — a planner that can mutate is not a planner. The adapter
|
|
831
|
+
* is responsible for answering the agent so the turn does not hang.
|
|
832
|
+
*/
|
|
833
|
+
| {
|
|
834
|
+
type: 'permission_request';
|
|
835
|
+
id: string;
|
|
836
|
+
name: string;
|
|
837
|
+
detail: string;
|
|
838
|
+
}
|
|
839
|
+
/** The agent finished its turn and is waiting for the next user message. */
|
|
840
|
+
| {
|
|
841
|
+
type: 'turn_end';
|
|
842
|
+
}
|
|
843
|
+
/** The turn failed. Carries the agent's own words — never a Ordewell paraphrase. */
|
|
844
|
+
| {
|
|
845
|
+
type: 'error';
|
|
846
|
+
message: string;
|
|
847
|
+
};
|
|
848
|
+
interface AgentStartOptions {
|
|
849
|
+
/** Workspace root. The agent explores from here and, in read-only mode, cannot leave it. */
|
|
850
|
+
cwd: string;
|
|
851
|
+
/** The planner system prompt, in its harness variant. */
|
|
852
|
+
systemPrompt: string;
|
|
853
|
+
/** Model id from the runner's own discovery catalog. Omitted means the agent's default. */
|
|
854
|
+
model?: string;
|
|
855
|
+
/** Variant / reasoning effort id from that model's `variants` list. */
|
|
856
|
+
effort?: string;
|
|
857
|
+
/**
|
|
858
|
+
* The agent's own session id from a previous run. A hint only: Ordewell's
|
|
859
|
+
* transcript is the source of truth (T4), so a failed resume degrades to a
|
|
860
|
+
* fresh session seeded from the stored history rather than an error.
|
|
861
|
+
*/
|
|
862
|
+
resumeSessionId?: string;
|
|
863
|
+
}
|
|
864
|
+
interface AgentAdapter {
|
|
865
|
+
/** The runner id this adapter drives — `claude-code`, `codex`, `opencode`. */
|
|
866
|
+
readonly agentId: string;
|
|
867
|
+
/** Spawn the agent in its read-only mode and get it ready to receive messages. */
|
|
868
|
+
start(opts: AgentStartOptions): Promise<void>;
|
|
869
|
+
/**
|
|
870
|
+
* Send one user message and stream the turn's events until it ends. Resolves
|
|
871
|
+
* when the agent yields the floor; rejects only when the transport itself
|
|
872
|
+
* failed in a way no `error` event could describe.
|
|
873
|
+
*/
|
|
874
|
+
send(message: string, onEvent: (event: AgentEvent) => void, signal?: AbortSignal): Promise<void>;
|
|
875
|
+
/** The agent's native session id once it has announced one. Resumption hint only. */
|
|
876
|
+
nativeSessionId(): string | null;
|
|
877
|
+
/** Kill the process and release its resources. Idempotent. */
|
|
878
|
+
dispose(): void;
|
|
879
|
+
}
|
|
880
|
+
/**
|
|
881
|
+
* The single injected boundary between Ordewell and the operating system —
|
|
882
|
+
* the same pattern `HeadlessRunnerDeps` uses for task execution. Tests feed
|
|
883
|
+
* recorded agent output through `spawn` (and, for HTTP-transport agents,
|
|
884
|
+
* `fetch`) so one test exercises adapter parsing, event mapping, reply
|
|
885
|
+
* classification and the repair loop as a single observable behavior.
|
|
886
|
+
*/
|
|
887
|
+
interface AgentProcessDeps {
|
|
888
|
+
spawn: SpawnFn;
|
|
889
|
+
fetch: typeof globalThis.fetch;
|
|
890
|
+
/** Resolves the PATH agents are spawned under. Defaults to the augmented PATH. */
|
|
891
|
+
resolvePath?: () => Promise<string>;
|
|
892
|
+
/** Host platform. Defaults to the real one; injected so OS-specific behavior is testable anywhere. */
|
|
893
|
+
platform?: NodeJS.Platform;
|
|
894
|
+
}
|
|
895
|
+
/** Builds the adapter for one runner id, or null when that runner cannot plan. */
|
|
896
|
+
type AgentAdapterFactory = (runner: string, deps: AgentProcessDeps) => AgentAdapter | null;
|
|
897
|
+
/**
|
|
898
|
+
* Split a stream of chunks into complete lines. Every agent transport here is
|
|
899
|
+
* newline-delimited JSON of some shape, and a chunk boundary lands mid-object
|
|
900
|
+
* often enough that parsing per-chunk silently drops events.
|
|
901
|
+
*/
|
|
902
|
+
declare class LineBuffer {
|
|
903
|
+
private buffer;
|
|
904
|
+
push(chunk: string, onLine: (line: string) => void): void;
|
|
905
|
+
/** Anything left unterminated when the stream closed. */
|
|
906
|
+
flush(): string;
|
|
907
|
+
}
|
|
908
|
+
|
|
909
|
+
interface CliAgentAiServiceDeps extends Partial<AgentProcessDeps> {
|
|
910
|
+
/** Overrides adapter construction. Tests supply a fake agent; production picks by runner id. */
|
|
911
|
+
createAdapter?: (runner: string, deps: AgentProcessDeps) => AgentAdapter | null;
|
|
912
|
+
/** Workspace root the agent explores. Defaults to the host process's cwd. */
|
|
913
|
+
workspaceRoot?: () => string;
|
|
914
|
+
}
|
|
915
|
+
/**
|
|
916
|
+
* A coding agent driven as Ordewell's planner (ADR-0009).
|
|
917
|
+
*
|
|
918
|
+
* The second transport behind `IAiService`, sitting beside `OpenAiService` and
|
|
919
|
+
* `GeminiService`. It deliberately does **not** extend {@link BaseAiService}:
|
|
920
|
+
* that class's body is Ordewell executing research tools on a model's behalf,
|
|
921
|
+
* which is precisely the part a coding agent replaces. What it reuses instead
|
|
922
|
+
* is everything above the transport — `classifyPlannerReply`, the bounded
|
|
923
|
+
* corrective re-emit loop, `parsePlanJson`, the `ResearchProgress` events the
|
|
924
|
+
* four surfaces already render. That is why this backend reaches VS Code, the
|
|
925
|
+
* web UI, the CLI and the TUI without any of them learning a coding agent is
|
|
926
|
+
* on the other end.
|
|
927
|
+
*/
|
|
928
|
+
declare class CliAgentAiService implements IAiService {
|
|
929
|
+
private config;
|
|
930
|
+
private readonly runner;
|
|
931
|
+
private readonly processDeps;
|
|
932
|
+
private readonly makeAdapter;
|
|
933
|
+
private readonly workspaceRoot;
|
|
934
|
+
private adapter;
|
|
935
|
+
/**
|
|
936
|
+
* The agent's own session id, kept across a process death so the next turn
|
|
937
|
+
* can resume warm context instead of re-reading the repository. Cleared at
|
|
938
|
+
* every session boundary — nothing from one goal may reach the next.
|
|
939
|
+
*/
|
|
940
|
+
private lastNativeSessionId;
|
|
941
|
+
private conversation;
|
|
942
|
+
private activeAbort;
|
|
943
|
+
constructor(config: IConfig, deps?: CliAgentAiServiceDeps);
|
|
944
|
+
hasActiveConversation(): boolean;
|
|
945
|
+
/**
|
|
946
|
+
* False when the running agent process was spawned under a model/effort the
|
|
947
|
+
* user has since changed in the picker. Unlike a vendor backend, the model
|
|
948
|
+
* is a spawn-time argument to the agent CLI (`--model`), not a per-request
|
|
949
|
+
* field — `continueConversation` sends the next turn to whatever process is
|
|
950
|
+
* already running, so a plain config read here would silently keep planning
|
|
951
|
+
* on the old model. No conversation yet is vacuously "current".
|
|
952
|
+
*/
|
|
953
|
+
conversationMatchesConfig(): boolean;
|
|
954
|
+
/** The agent's own session id, a resumption hint only — Ordewell's transcript is authoritative (T4). */
|
|
955
|
+
nativeSessionId(): string | null;
|
|
956
|
+
reset(): void;
|
|
957
|
+
startConversation(req: ConversationRequest): Promise<ConversationTurn>;
|
|
958
|
+
continueConversation(userMessage: string, onProgress: (progress: ResearchProgress) => void, signal?: AbortSignal): Promise<ConversationTurn>;
|
|
959
|
+
private openingMessage;
|
|
960
|
+
/**
|
|
961
|
+
* Drive one user message to a settled planner turn. Same shape as the API
|
|
962
|
+
* backend's conversation loop minus the tool rounds — those belong to the
|
|
963
|
+
* agent now — and with the same three policies layered on top: the PRD gate,
|
|
964
|
+
* the empty-reply nudge, and a bounded corrective re-emit for botched JSON.
|
|
965
|
+
*/
|
|
966
|
+
private runConversation;
|
|
967
|
+
/**
|
|
968
|
+
* Run one agent turn: stream its events into `ResearchProgress`, collect the
|
|
969
|
+
* reply text, and return the research steps produced. Tool calls are matched
|
|
970
|
+
* to their results by the agent's own call id — matching by tool name puts
|
|
971
|
+
* one file's body on another file's row the moment an agent runs two reads
|
|
972
|
+
* at once, which all three of these do routinely.
|
|
973
|
+
*/
|
|
974
|
+
private runTurn;
|
|
975
|
+
private startAdapter;
|
|
976
|
+
/**
|
|
977
|
+
* The live agent process, restarted from its own session id if it died
|
|
978
|
+
* between turns. Resume is a hint: when it fails, the caller's next
|
|
979
|
+
* `startConversation` reseeds from Ordewell's transcript, which is the same
|
|
980
|
+
* degradation `restoreChat` already performs on every surface.
|
|
981
|
+
*/
|
|
982
|
+
private ensureAdapter;
|
|
983
|
+
private startAbortScope;
|
|
984
|
+
private plannerModel;
|
|
985
|
+
/**
|
|
986
|
+
* A single agent session that answers one prompt and exits. Used by every
|
|
987
|
+
* non-conversational entry point; the plan is parsed from the reply text by
|
|
988
|
+
* the same extractor the conversational path uses.
|
|
989
|
+
*/
|
|
990
|
+
private oneShot;
|
|
991
|
+
researchAndPlan(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
|
|
992
|
+
tasks: Task[];
|
|
993
|
+
researchLog: ResearchLogEntry[];
|
|
994
|
+
researchResults: string;
|
|
995
|
+
}>;
|
|
996
|
+
generatePlanDirect(userDescription: string, runners?: RunnerId[], modelsByRunner?: Partial<Record<RunnerId, DiscoveredModel[]>>, onToken?: (token: string) => void, _fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<Task[]>;
|
|
997
|
+
modifyPlan(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onProgress?: (progress: ResearchProgress) => void, _fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
|
|
998
|
+
tasks: Task[];
|
|
999
|
+
}>;
|
|
1000
|
+
sendPlanningPrompt(prompt: string, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): Promise<Task[]>;
|
|
1001
|
+
}
|
|
1002
|
+
|
|
1003
|
+
/**
|
|
1004
|
+
* Everything the conversation loop needs to start planning (ADR-0002).
|
|
1005
|
+
* The AI service keeps the tool-use message history internally across turns;
|
|
1006
|
+
* the surfaces only exchange user/assistant messages with it.
|
|
1007
|
+
*/
|
|
1008
|
+
interface ConversationRequest {
|
|
1009
|
+
goal: string;
|
|
1010
|
+
runners: RunnerId[];
|
|
1011
|
+
modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>;
|
|
1012
|
+
fs: IFileSystem;
|
|
1013
|
+
onProgress: (progress: ResearchProgress) => void;
|
|
1014
|
+
fetcher?: IWebFetcher;
|
|
1015
|
+
runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
|
|
1016
|
+
autonomousDefault?: boolean;
|
|
1017
|
+
grillMeEnabled?: boolean;
|
|
1018
|
+
prdEnabled?: boolean;
|
|
1019
|
+
reviewEnabled?: boolean;
|
|
1020
|
+
verificationEnabled?: boolean;
|
|
1021
|
+
/** Declare the spawn_research_agent tool to the planner (issue #34, default off). */
|
|
1022
|
+
researchSubagentsEnabled?: boolean;
|
|
1023
|
+
signal?: AbortSignal;
|
|
1024
|
+
/**
|
|
1025
|
+
* Persisted dialogue to seed a resumed conversation (session reload). The
|
|
1026
|
+
* turns are injected into the message history verbatim — no LLM call happens
|
|
1027
|
+
* for them; the first call is the turn opened by `initialMessage`.
|
|
1028
|
+
*/
|
|
1029
|
+
priorHistory?: ConversationMessage[];
|
|
1030
|
+
/**
|
|
1031
|
+
* The user message that opens this turn. Defaults to `goal` — set it when
|
|
1032
|
+
* resuming so the system prompt keeps the original goal while the turn
|
|
1033
|
+
* carries the user's new message.
|
|
1034
|
+
*/
|
|
1035
|
+
initialMessage?: string;
|
|
1036
|
+
}
|
|
1037
|
+
/**
|
|
1038
|
+
* One planner turn's outcome. The planner talks to the user (`message`),
|
|
1039
|
+
* commits a full plan (`plan`, a `{tasks:[...]}` JSON object), or emits
|
|
1040
|
+
* targeted task edits (`task_ops`, a `{taskOps:[...]}` JSON object). The
|
|
1041
|
+
* Session validates and applies task ops atomically; the AI service only
|
|
1042
|
+
* parses them.
|
|
1043
|
+
*/
|
|
1044
|
+
type ConversationTurn = {
|
|
1045
|
+
kind: 'message';
|
|
1046
|
+
text: string;
|
|
1047
|
+
researchLog: ResearchLogEntry[];
|
|
1048
|
+
} | {
|
|
1049
|
+
kind: 'plan';
|
|
1050
|
+
tasks: Task[];
|
|
1051
|
+
text: string;
|
|
1052
|
+
researchLog: ResearchLogEntry[];
|
|
1053
|
+
} | {
|
|
1054
|
+
kind: 'task_ops';
|
|
1055
|
+
ops: TaskOp[];
|
|
1056
|
+
text: string;
|
|
1057
|
+
researchLog: ResearchLogEntry[];
|
|
1058
|
+
};
|
|
1059
|
+
interface IAiService {
|
|
1060
|
+
/**
|
|
1061
|
+
* Begin the planner conversation: collect workspace context, run the
|
|
1062
|
+
* research/tool loop, and return the first planner turn. The service
|
|
1063
|
+
* retains the full API message history for subsequent
|
|
1064
|
+
* {@link continueConversation} calls.
|
|
1065
|
+
*/
|
|
1066
|
+
startConversation(req: ConversationRequest): Promise<ConversationTurn>;
|
|
1067
|
+
/**
|
|
1068
|
+
* Feed the user's reply into the active conversation and return the next
|
|
1069
|
+
* planner turn. Throws if no conversation is active.
|
|
1070
|
+
*/
|
|
1071
|
+
continueConversation(userMessage: string, onProgress: (progress: ResearchProgress) => void, signal?: AbortSignal): Promise<ConversationTurn>;
|
|
1072
|
+
/**
|
|
1073
|
+
* Whether a planner conversation is currently held in memory. A branching
|
|
1074
|
+
* predicate only — never a test for whether there is anything to release.
|
|
1075
|
+
* The harness backend holds an OS process that outlives its conversation, so
|
|
1076
|
+
* guarding {@link reset} with this leaks an agent process; call `reset`
|
|
1077
|
+
* unconditionally instead.
|
|
1078
|
+
*/
|
|
1079
|
+
hasActiveConversation(): boolean;
|
|
1080
|
+
/**
|
|
1081
|
+
* Whether the active conversation (if any) still matches the planner
|
|
1082
|
+
* model/effort currently configured. Optional, and true when absent: a
|
|
1083
|
+
* vendor backend re-reads its model from config on every turn, so there is
|
|
1084
|
+
* nothing to drift. A harness planner (ADR-0009) is the exception — its
|
|
1085
|
+
* model is baked into the agent process at spawn — so only
|
|
1086
|
+
* {@link CliAgentAiService} answers this for real. False tells the caller
|
|
1087
|
+
* the same thing an absent conversation would: don't call
|
|
1088
|
+
* `continueConversation`, restart instead so the new model takes effect.
|
|
1089
|
+
*/
|
|
1090
|
+
conversationMatchesConfig?(): boolean;
|
|
1091
|
+
/**
|
|
1092
|
+
* One-shot research + plan for non-conversational surfaces (CLI `plan --goal`,
|
|
1093
|
+
* web REST). Never asks questions — it plans with what it can find.
|
|
1094
|
+
*/
|
|
1095
|
+
researchAndPlan(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
|
|
1096
|
+
tasks: Task[];
|
|
1097
|
+
researchLog: ResearchLogEntry[];
|
|
1098
|
+
researchResults: string;
|
|
1099
|
+
}>;
|
|
1100
|
+
generatePlanDirect(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onToken?: (token: string) => void, fs?: IFileSystem, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<Task[]>;
|
|
1101
|
+
modifyPlan(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onProgress?: (progress: ResearchProgress) => void, fs?: IFileSystem, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
|
|
1102
|
+
tasks: Task[];
|
|
1103
|
+
}>;
|
|
1104
|
+
/**
|
|
1105
|
+
* Release everything this service holds: the conversation, any in-flight
|
|
1106
|
+
* turn, and — on the harness backend — the agent process itself. Idempotent
|
|
1107
|
+
* and cheap in every implementation, so callers never gate it.
|
|
1108
|
+
*/
|
|
1109
|
+
reset(): void;
|
|
1110
|
+
sendPlanningPrompt(prompt: string, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): Promise<Task[]>;
|
|
1111
|
+
}
|
|
1112
|
+
/**
|
|
1113
|
+
* The single branch point between planner transports. Two of them talk HTTP to
|
|
1114
|
+
* an LLM vendor; the third (ADR-0009) drives a coding agent already installed
|
|
1115
|
+
* on the machine. Everything downstream — the plan contract, the research-step
|
|
1116
|
+
* stream, the four surfaces — is shared.
|
|
1117
|
+
*/
|
|
1118
|
+
declare function createAiService(config: IConfig, deps?: CliAgentAiServiceDeps): IAiService;
|
|
1119
|
+
|
|
1120
|
+
interface ToolCall {
|
|
1121
|
+
name: string;
|
|
1122
|
+
args: Record<string, unknown>;
|
|
1123
|
+
id?: string;
|
|
1124
|
+
}
|
|
1125
|
+
interface ToolResult {
|
|
1126
|
+
name: string;
|
|
1127
|
+
output: string;
|
|
1128
|
+
truncated: boolean;
|
|
1129
|
+
totalChars: number;
|
|
1130
|
+
id?: string;
|
|
1131
|
+
}
|
|
1132
|
+
interface ResearchTurn {
|
|
1133
|
+
text: string;
|
|
1134
|
+
toolCalls: ToolCall[];
|
|
1135
|
+
hasToolCalls: boolean;
|
|
1136
|
+
/** Reasoning/chain-of-thought captured separately from `text` so it never pollutes plan JSON. */
|
|
1137
|
+
reasoning?: string;
|
|
1138
|
+
/** Provider finish reason, normalized: 'length' means the output-token limit cut the reply mid-stream. */
|
|
1139
|
+
finishReason?: string;
|
|
1140
|
+
/** Exact prompt tokens this turn consumed, when the provider reports usage — drives proactive compaction. */
|
|
1141
|
+
promptTokens?: number;
|
|
1142
|
+
}
|
|
1143
|
+
interface ResearchChat {
|
|
1144
|
+
sendMessage(text: string, signal?: AbortSignal): Promise<ResearchTurn>;
|
|
1145
|
+
sendToolResults(results: ToolResult[], signal?: AbortSignal): Promise<ResearchTurn>;
|
|
1146
|
+
/**
|
|
1147
|
+
* Optional: prune bulky raw tool outputs from the history in place (subagent
|
|
1148
|
+
* digests are kept) so a follow-up emission has context to spend on output.
|
|
1149
|
+
* Returns the number of characters removed. Providers whose SDK owns the
|
|
1150
|
+
* history (Gemini) may not support it.
|
|
1151
|
+
*/
|
|
1152
|
+
compactHistory?(): number;
|
|
1153
|
+
}
|
|
1154
|
+
/** Everything one conversation turn needs beyond the user message itself. */
|
|
1155
|
+
interface ConversationTurnContext {
|
|
1156
|
+
chat: ResearchChat;
|
|
1157
|
+
fs: IFileSystem;
|
|
1158
|
+
runners: RunnerId[];
|
|
1159
|
+
runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
|
|
1160
|
+
autonomousDefault?: boolean;
|
|
1161
|
+
fetcher?: IWebFetcher;
|
|
1162
|
+
/** PRD toggle: a plan must not commit before a PRD block has appeared. */
|
|
1163
|
+
prdEnabled?: boolean;
|
|
1164
|
+
/** Set once any turn carries an ORDEWELL_PRD block; gates the missing-PRD nudge. */
|
|
1165
|
+
prdCaptured?: boolean;
|
|
1166
|
+
/** The missing-PRD corrective nudge is sent at most once per conversation. */
|
|
1167
|
+
prdNudgeSent?: boolean;
|
|
1168
|
+
}
|
|
1169
|
+
declare abstract class BaseAiService {
|
|
1170
|
+
protected config: IConfig;
|
|
1171
|
+
protected conversation: {
|
|
1172
|
+
ctx: ConversationTurnContext;
|
|
1173
|
+
setProgress: (fn: (p: ResearchProgress) => void) => void;
|
|
1174
|
+
} | null;
|
|
1175
|
+
protected activeAbort: AbortController | null;
|
|
1176
|
+
/** researchSubagents toggle (issue #34): set per operation from the live settings snapshot; off means bit-for-bit sequential behavior. */
|
|
1177
|
+
protected researchSubagentsEnabled: boolean;
|
|
1178
|
+
constructor(config: IConfig);
|
|
1179
|
+
abstract reset(): void;
|
|
1180
|
+
abstract ensureInit(): void;
|
|
1181
|
+
hasActiveConversation(): boolean;
|
|
1182
|
+
protected startAbortScope(callerSignal?: AbortSignal): AbortSignal | undefined;
|
|
1183
|
+
protected stopAbortScope(): void;
|
|
1184
|
+
sendPlanningPrompt(prompt: string, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): Promise<Task[]>;
|
|
1185
|
+
continueConversation(userMessage: string, onProgress: (progress: ResearchProgress) => void, signal?: AbortSignal): Promise<ConversationTurn>;
|
|
1186
|
+
protected abstract streamPlanText(prompt: string, repairHint: string | undefined, onToken: (token: string) => void, onReasoning?: (token: string) => void, signal?: AbortSignal): Promise<string>;
|
|
1187
|
+
/**
|
|
1188
|
+
* Build a fresh chat for one research subagent (own history, subagent system
|
|
1189
|
+
* prompt, cheap model). Null means the provider does not support subagents —
|
|
1190
|
+
* the spawn tool then degrades to a steering message. `onReasoning` streams
|
|
1191
|
+
* live reasoning deltas on models that expose them, same as the top-level loop.
|
|
1192
|
+
*/
|
|
1193
|
+
protected createSubagentChat(_onReasoning?: (delta: string) => void): ResearchChat | null;
|
|
1194
|
+
/**
|
|
1195
|
+
* One spawn_research_agent tool call, executed at the service layer (not in
|
|
1196
|
+
* executeTool — that would cycle the imports). Every failure path returns a
|
|
1197
|
+
* steering/failed result and the loop continues sequentially: a subagent can
|
|
1198
|
+
* never fail a plan or a turn.
|
|
1199
|
+
*/
|
|
1200
|
+
private executeSpawnAgent;
|
|
1201
|
+
/** Collect project context for the planning phase. Shared with the harness backend. */
|
|
1202
|
+
protected static collectResearchContext(fs: IFileSystem, runners: RunnerId[]): Promise<string>;
|
|
1203
|
+
/**
|
|
1204
|
+
* Execute one round of tool calls and report each through onProgress.
|
|
1205
|
+
* Returns the results to feed back plus the log entries produced.
|
|
1206
|
+
*/
|
|
1207
|
+
private executeToolCalls;
|
|
1208
|
+
/**
|
|
1209
|
+
* Countdown appended to the last tool result of the final few rounds, so the
|
|
1210
|
+
* model lands the turn on its own terms instead of being cut off exactly at
|
|
1211
|
+
* the budget boundary mid-exploration.
|
|
1212
|
+
*/
|
|
1213
|
+
private static appendBudgetCountdown;
|
|
1214
|
+
/**
|
|
1215
|
+
* Run one planner conversation turn (ADR-0002): send the message, satisfy
|
|
1216
|
+
* tool calls until the model answers in prose or JSON, then classify the
|
|
1217
|
+
* result. The model decides transitions — there are no sentinels, no
|
|
1218
|
+
* question tags, and no correction nags. A turn whose final text parses as
|
|
1219
|
+
* a `{tasks:[...]}` object commits the plan; anything else is a message to
|
|
1220
|
+
* the user.
|
|
1221
|
+
*/
|
|
1222
|
+
protected runConversationTurn(ctx: ConversationTurnContext, message: string, onProgress: (progress: ResearchProgress) => void, signal?: AbortSignal): Promise<ConversationTurn>;
|
|
1223
|
+
/**
|
|
1224
|
+
* Run the LLM tool-calling research loop for one-shot planning. Returns
|
|
1225
|
+
* parsed tasks if a plan was emitted mid-loop, or null if the loop
|
|
1226
|
+
* exhausted without a valid plan.
|
|
1227
|
+
*/
|
|
1228
|
+
protected runResearchLoop(chat: ResearchChat, firstMessage: string, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, runners: RunnerId[], initialLog?: ResearchLogEntry[], fetcher?: IWebFetcher, userGoal?: string, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean, signal?: AbortSignal): Promise<{
|
|
1229
|
+
tasks: Task[] | null;
|
|
1230
|
+
researchLog: ResearchLogEntry[];
|
|
1231
|
+
researchResults: string;
|
|
1232
|
+
}>;
|
|
1233
|
+
protected generatePlanFallback(userDescription: string, contextStr: string, researchResults: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], onProgress: (progress: ResearchProgress) => void, researchLog?: ResearchLogEntry[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean, signal?: AbortSignal, modes?: PlannerModes): Promise<{
|
|
1234
|
+
tasks: Task[];
|
|
1235
|
+
researchLog: ResearchLogEntry[];
|
|
1236
|
+
}>;
|
|
1237
|
+
}
|
|
1238
|
+
|
|
1239
|
+
declare class GeminiService extends BaseAiService implements IAiService {
|
|
1240
|
+
private genAI;
|
|
1241
|
+
private model;
|
|
1242
|
+
constructor(config: IConfig);
|
|
1243
|
+
private init;
|
|
1244
|
+
private getPlanningModel;
|
|
1245
|
+
ensureInit(): void;
|
|
1246
|
+
reset(): void;
|
|
1247
|
+
startConversation(req: ConversationRequest): Promise<ConversationTurn>;
|
|
1248
|
+
protected streamPlanText(prompt: string, repairHint: string | undefined, onToken: (token: string) => void, onReasoning?: (token: string) => void, _signal?: AbortSignal): Promise<string>;
|
|
1249
|
+
/**
|
|
1250
|
+
* Drain a Gemini content stream, routing "thought" parts to the reasoning channel
|
|
1251
|
+
* and answer text to the parser. `chunk.text()` throws when a chunk is reasoning-only,
|
|
1252
|
+
* so parts are inspected directly.
|
|
1253
|
+
*/
|
|
1254
|
+
private consumePlanStream;
|
|
1255
|
+
/** Gemini-specific stream over raw Parts for generatePlanDirect / modifyPlan. */
|
|
1256
|
+
private streamPlanTextWithParts;
|
|
1257
|
+
researchAndPlan(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
|
|
1258
|
+
tasks: Task[];
|
|
1259
|
+
researchLog: ResearchLogEntry[];
|
|
1260
|
+
researchResults: string;
|
|
1261
|
+
}>;
|
|
1262
|
+
generatePlanDirect(userDescription: string, runners?: RunnerId[], modelsByRunner?: Partial<Record<RunnerId, DiscoveredModel[]>>, onToken?: (token: string) => void, fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, _signal?: AbortSignal): Promise<Task[]>;
|
|
1263
|
+
modifyPlan(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onProgress?: (progress: ResearchProgress) => void, fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, _signal?: AbortSignal): Promise<{
|
|
1264
|
+
tasks: Task[];
|
|
1265
|
+
}>;
|
|
1266
|
+
}
|
|
1267
|
+
|
|
1268
|
+
declare class OpenAiService extends BaseAiService implements IAiService {
|
|
1269
|
+
private client;
|
|
1270
|
+
constructor(config: IConfig);
|
|
1271
|
+
private getClient;
|
|
1272
|
+
ensureInit(): void;
|
|
1273
|
+
private requireModel;
|
|
1274
|
+
reset(): void;
|
|
1275
|
+
protected streamPlanText(prompt: string, repairHint: string | undefined, onToken: (token: string) => void, onReasoning?: (token: string) => void, signal?: AbortSignal): Promise<string>;
|
|
1276
|
+
/** A research subagent: fresh history, digest contract, cheap model, read-only tools. */
|
|
1277
|
+
protected createSubagentChat(onReasoning?: (delta: string) => void): ResearchChat | null;
|
|
1278
|
+
startConversation(req: ConversationRequest): Promise<ConversationTurn>;
|
|
1279
|
+
researchAndPlan(userDescription: string, runners: RunnerId[], modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, fs: IFileSystem, onProgress: (progress: ResearchProgress) => void, fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
|
|
1280
|
+
tasks: Task[];
|
|
1281
|
+
researchLog: ResearchLogEntry[];
|
|
1282
|
+
researchResults: string;
|
|
1283
|
+
}>;
|
|
1284
|
+
generatePlanDirect(userDescription: string, runners?: RunnerId[], modelsByRunner?: Partial<Record<RunnerId, DiscoveredModel[]>>, onToken?: (token: string) => void, fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<Task[]>;
|
|
1285
|
+
modifyPlan(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, onProgress?: (progress: ResearchProgress) => void, fs?: IFileSystem, _fetcher?: IWebFetcher, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes, signal?: AbortSignal): Promise<{
|
|
1286
|
+
tasks: Task[];
|
|
1287
|
+
}>;
|
|
1288
|
+
}
|
|
1289
|
+
|
|
1290
|
+
/**
|
|
1291
|
+
* The deep module owning all plan-shaped state. Holds `planTasks` (the ordered
|
|
1292
|
+
* tree the user edits), the flattened `allTasks` view, the `taskMap` index, and
|
|
1293
|
+
* the `completedTasks`/`failedTasks` sets the scheduler reads. The orchestrator
|
|
1294
|
+
* calls `markCompleted`/`markFailed`/`markInProgress`/`retry` to update task
|
|
1295
|
+
* status — it never mutates task state directly.
|
|
1296
|
+
*
|
|
1297
|
+
* `rebuild` is the internal seam that keeps the flat views in sync with the
|
|
1298
|
+
* tree. Structural removals (remove/merge/split) additionally prune
|
|
1299
|
+
* `completedTasks`/`failedTasks` for ids that no longer exist; `removeFromActive`
|
|
1300
|
+
* deliberately does not, so a completed task that leaves the active list still
|
|
1301
|
+
* satisfies its dependents' dependency checks.
|
|
1302
|
+
*
|
|
1303
|
+
* `planRunners` lives here because it's part of the plan's identity (the runner
|
|
1304
|
+
* set, carried on the plan). `validateAssignedRunners` is pure store logic.
|
|
1305
|
+
*/
|
|
1306
|
+
declare class PlanStore {
|
|
1307
|
+
private _planTasks;
|
|
1308
|
+
private _allTasks;
|
|
1309
|
+
private _taskMap;
|
|
1310
|
+
private _completedTasks;
|
|
1311
|
+
private _failedTasks;
|
|
1312
|
+
private _planRunners;
|
|
1313
|
+
private _onMutate;
|
|
1314
|
+
private _executionLog;
|
|
1315
|
+
/** Hook called after every structural mutation (add/remove/update/merge/split/load). */
|
|
1316
|
+
set onMutate(cb: (() => void) | null);
|
|
1317
|
+
get planTasks(): Task[];
|
|
1318
|
+
get allTasks(): Task[];
|
|
1319
|
+
get planRunners(): RunnerId[];
|
|
1320
|
+
get completedCount(): number;
|
|
1321
|
+
get failedCount(): number;
|
|
1322
|
+
isAllComplete(): boolean;
|
|
1323
|
+
isAnyFailed(): boolean;
|
|
1324
|
+
isCompleted(id: string): boolean;
|
|
1325
|
+
isFailed(id: string): boolean;
|
|
1326
|
+
get(taskId: string): Task | undefined;
|
|
1327
|
+
getExecutionLog(): TaskSnapshot[];
|
|
1328
|
+
appendToLog(snapshot: TaskSnapshot): void;
|
|
1329
|
+
/**
|
|
1330
|
+
* Drop a task's archived snapshot. Un-marking a completion has to erase the
|
|
1331
|
+
* "finished" record too — dependent tasks are prompted from the log, so a
|
|
1332
|
+
* left-behind snapshot would keep feeding them a result that no longer exists.
|
|
1333
|
+
*/
|
|
1334
|
+
removeFromLog(taskId: string): void;
|
|
1335
|
+
removeFromActive(taskId: string): void;
|
|
1336
|
+
clearLog(): void;
|
|
1337
|
+
private notifyMutate;
|
|
1338
|
+
load(tasks: Task[], runners: RunnerId[]): void;
|
|
1339
|
+
add(partial: Partial<Task>): Task;
|
|
1340
|
+
remove(taskId: string): void;
|
|
1341
|
+
update(taskId: string, changes: Partial<Task>): Task | undefined;
|
|
1342
|
+
mergeMultiple(taskIds: string[]): Task;
|
|
1343
|
+
merge(taskIdA: string, taskIdB: string): Task;
|
|
1344
|
+
split(taskId: string, newTaskSpecs: Partial<Task>[]): Task[];
|
|
1345
|
+
/**
|
|
1346
|
+
* Run preparation as one named op: flip every AI task to 'approved'. By
|
|
1347
|
+
* default completed tasks keep their status so a reloaded half-finished plan
|
|
1348
|
+
* resumes the remainder instead of re-running work that already succeeded;
|
|
1349
|
+
* `preserveCompleted: false` is for a freshly committed plan, where
|
|
1350
|
+
* everything starts over.
|
|
1351
|
+
*/
|
|
1352
|
+
resetForRun(opts?: {
|
|
1353
|
+
preserveCompleted?: boolean;
|
|
1354
|
+
}): void;
|
|
1355
|
+
markCompleted(id: string): void;
|
|
1356
|
+
markFailed(id: string): void;
|
|
1357
|
+
markInProgress(id: string): void;
|
|
1358
|
+
markAwaitingUser(id: string): void;
|
|
1359
|
+
markPending(id: string): void;
|
|
1360
|
+
retry(id: string): void;
|
|
1361
|
+
blockDependents(id: string): void;
|
|
1362
|
+
unblockDependents(id: string): void;
|
|
1363
|
+
setTaskVerdict(id: string, verdict: Task['verdict']): void;
|
|
1364
|
+
setTaskOutputSummary(id: string, summary: Task['outputSummary']): void;
|
|
1365
|
+
getPlanVisualization(): {
|
|
1366
|
+
tasks: {
|
|
1367
|
+
id: string;
|
|
1368
|
+
title: string;
|
|
1369
|
+
dependencies: string[];
|
|
1370
|
+
parallelGroups: number[][];
|
|
1371
|
+
}[];
|
|
1372
|
+
parallelGroups: number[][];
|
|
1373
|
+
};
|
|
1374
|
+
/**
|
|
1375
|
+
* Widen the plan's runner set. Retargeting a task onto a runner the plan has
|
|
1376
|
+
* not used before has to land here as well as on the plan state, or
|
|
1377
|
+
* {@link resolveTaskRunner} reads the new runner as foreign and spawns the
|
|
1378
|
+
* plan's first one instead.
|
|
1379
|
+
*/
|
|
1380
|
+
admitRunner(runner: RunnerId): void;
|
|
1381
|
+
resolveTaskRunner(task: Task): RunnerId;
|
|
1382
|
+
private rebuild;
|
|
1383
|
+
/**
|
|
1384
|
+
* Drop completed/failed ids that no longer exist in the plan. Called by the
|
|
1385
|
+
* structural removals (remove/merge/split) — but NOT by removeFromActive,
|
|
1386
|
+
* where a completed task leaves the active list yet must still satisfy its
|
|
1387
|
+
* dependents' dependency checks.
|
|
1388
|
+
*/
|
|
1389
|
+
private pruneTerminalSets;
|
|
1390
|
+
private validateAssignedRunners;
|
|
1391
|
+
}
|
|
1392
|
+
|
|
1393
|
+
/**
|
|
1394
|
+
* The one notification channel out of the orchestrator. Everything that used
|
|
1395
|
+
* to travel over separate callbacks (onRefresh, onQueueReady) is an observer
|
|
1396
|
+
* event; the Session subscribes once and turns these into SessionMessages.
|
|
1397
|
+
*/
|
|
1398
|
+
interface OrchestratorObserver {
|
|
1399
|
+
/** Any task-shaped state changed (store mutation, checkpoint, retry, …). */
|
|
1400
|
+
onTaskChanged?(): void;
|
|
1401
|
+
onTick?(): void;
|
|
1402
|
+
onExecutionComplete?(): void;
|
|
1403
|
+
/** Queued user messages are ready to be processed by the planner. */
|
|
1404
|
+
onQueueReady?(): void;
|
|
1405
|
+
onReviewNeeded?(data: {
|
|
1406
|
+
tasks: Task[];
|
|
1407
|
+
planRunners: RunnerId[];
|
|
1408
|
+
}): void;
|
|
1409
|
+
onReviewApproved?(data: {
|
|
1410
|
+
tasks: Task[];
|
|
1411
|
+
}): void;
|
|
1412
|
+
onCheckpoint?(data: {
|
|
1413
|
+
taskId: string;
|
|
1414
|
+
taskTitle: string;
|
|
1415
|
+
summary: string;
|
|
1416
|
+
}): void;
|
|
1417
|
+
}
|
|
1418
|
+
/**
|
|
1419
|
+
* The pure scheduler. Owns execution state (`running`, `planStatus`,
|
|
1420
|
+
* `reviewApproved`, `activeTaskSessions`, `messageQueue`) and the verifier.
|
|
1421
|
+
* All task-shaped state — the plan tree, the flat index, the completed set
|
|
1422
|
+
* — lives in {@link PlanStore}, injected at construction. The orchestrator
|
|
1423
|
+
* calls `store.markCompleted(id)` / `store.markFailed(id)` instead of mutating
|
|
1424
|
+
* task state directly. A task completes only after the runner emits its
|
|
1425
|
+
* per-task completion marker; process exit without that evidence is a visible
|
|
1426
|
+
* failure and does not unblock dependent work.
|
|
1427
|
+
*/
|
|
1428
|
+
declare class TaskOrchestrator {
|
|
1429
|
+
private config;
|
|
1430
|
+
private notifications;
|
|
1431
|
+
private terminalRunner;
|
|
1432
|
+
private store;
|
|
1433
|
+
private activeTaskSessions;
|
|
1434
|
+
private startingTaskIds;
|
|
1435
|
+
private verifier;
|
|
1436
|
+
private running;
|
|
1437
|
+
private planStatus;
|
|
1438
|
+
private messageQueue;
|
|
1439
|
+
private reviewApproved;
|
|
1440
|
+
private retryCounts;
|
|
1441
|
+
/**
|
|
1442
|
+
* Tasks pulled out of auto-scheduling (user-cancelled or failed to spawn).
|
|
1443
|
+
* They stay 'pending' — "not executed" — but the scheduler skips them until
|
|
1444
|
+
* the user retries or force-starts, which would otherwise loop forever on a
|
|
1445
|
+
* task whose spawn always throws.
|
|
1446
|
+
*/
|
|
1447
|
+
private onHold;
|
|
1448
|
+
private registry;
|
|
1449
|
+
private workspaceRootFn;
|
|
1450
|
+
private observers;
|
|
1451
|
+
private tddEnabled;
|
|
1452
|
+
constructor(config: IConfig, notifications: INotification, terminalRunner: ITerminalRunner, store?: PlanStore);
|
|
1453
|
+
get storeInstance(): PlanStore;
|
|
1454
|
+
setWorkspaceRoot(fn: () => string): void;
|
|
1455
|
+
setRegistry(registry: RunnerRegistry): void;
|
|
1456
|
+
/**
|
|
1457
|
+
* A getter rather than a value where the caller has one: every task gets its
|
|
1458
|
+
* prompt composed at spawn time, but only a full-plan run passes through a
|
|
1459
|
+
* point where a snapshot could be refreshed — so "Run task", force-start and
|
|
1460
|
+
* retry would compose against whatever the last run happened to set.
|
|
1461
|
+
*/
|
|
1462
|
+
setTddEnabled(enabled: boolean | (() => boolean)): void;
|
|
1463
|
+
approveCheckpoint(taskId: string): void;
|
|
1464
|
+
rejectCheckpoint(taskId: string, reason?: string): void;
|
|
1465
|
+
subscribe(observer: OrchestratorObserver): () => void;
|
|
1466
|
+
private emit;
|
|
1467
|
+
get isRunning(): boolean;
|
|
1468
|
+
get isReviewApproved(): boolean;
|
|
1469
|
+
get status(): 'approved' | 'running' | 'completed';
|
|
1470
|
+
get activeTaskIds(): string[];
|
|
1471
|
+
get activeSessionMap(): Map<string, string>;
|
|
1472
|
+
get queuedCount(): number;
|
|
1473
|
+
queueMessage(text: string): void;
|
|
1474
|
+
getQueuedMessages(): QueuedMessage[];
|
|
1475
|
+
setQueuedMessages(messages: QueuedMessage[]): void;
|
|
1476
|
+
clearQueuedMessages(): void;
|
|
1477
|
+
processNextQueuedMessage(): QueuedMessage | null;
|
|
1478
|
+
loadPlan(tasks: Task[], planRunners?: RunnerId[]): void;
|
|
1479
|
+
reconcilePlan(newTasks: Task[], planRunners?: RunnerId[]): void;
|
|
1480
|
+
start(): Promise<void>;
|
|
1481
|
+
stop(): void;
|
|
1482
|
+
onUserTaskComplete(taskId: string): Promise<void>;
|
|
1483
|
+
private onVerdict;
|
|
1484
|
+
getReadyTasks(): Task[];
|
|
1485
|
+
isBlocked(task: Task): boolean;
|
|
1486
|
+
/**
|
|
1487
|
+
* Cancel a running (or scheduled) task: kill its session and return it to
|
|
1488
|
+
* 'pending' — "not executed". The task is put on hold so the scheduler
|
|
1489
|
+
* doesn't immediately restart it; Retry / Force Start release the hold.
|
|
1490
|
+
*/
|
|
1491
|
+
cancelTask(taskId: string): Promise<void>;
|
|
1492
|
+
markTaskComplete(taskId: string): Promise<void>;
|
|
1493
|
+
markAiTaskComplete(taskId: string): Promise<void>;
|
|
1494
|
+
/**
|
|
1495
|
+
* Undo a completion: return the task to "not executed" — pending, verdict and
|
|
1496
|
+
* summary dropped, archive entry removed. It is put on hold like a cancel, so
|
|
1497
|
+
* a running plan does not immediately re-spawn the work the user just
|
|
1498
|
+
* un-marked; Retry / Force Start / Run release the hold. Dependents fall back
|
|
1499
|
+
* to waiting on their own, because the scheduler gates on `isCompleted`.
|
|
1500
|
+
*/
|
|
1501
|
+
markTaskIncomplete(taskId: string): Promise<void>;
|
|
1502
|
+
private logAndArchive;
|
|
1503
|
+
retryTask(taskId: string): Promise<void>;
|
|
1504
|
+
/**
|
|
1505
|
+
* Manually start a single AI task right now, bypassing dependency/readiness
|
|
1506
|
+
* gating (the "force start" affordance on a task card). Reuses the scheduler's
|
|
1507
|
+
* own startTask so a force-started task gets the same augmented prompt, session
|
|
1508
|
+
* tracking, and exit handling — callers must not re-spawn the runner themselves.
|
|
1509
|
+
* No-op if the task is unknown, not an AI task, or already running.
|
|
1510
|
+
*/
|
|
1511
|
+
forceStartTask(taskId: string): Promise<void>;
|
|
1512
|
+
/**
|
|
1513
|
+
* Run exactly one task outside full-plan scheduling. The active/starting
|
|
1514
|
+
* session still contributes to isRunning so every surface exposes Stop and
|
|
1515
|
+
* disables Execute Plan, but onVerdict cannot auto-schedule other tasks
|
|
1516
|
+
* because the plan scheduler's `running` flag remains false.
|
|
1517
|
+
*/
|
|
1518
|
+
runTask(taskId: string): Promise<void>;
|
|
1519
|
+
getCompletedCount(): number;
|
|
1520
|
+
getTotalCount(): number;
|
|
1521
|
+
isAllComplete(): boolean;
|
|
1522
|
+
isAnyFailed(): boolean;
|
|
1523
|
+
approveReview(): Promise<void>;
|
|
1524
|
+
getPlanVisualization(): {
|
|
1525
|
+
tasks: {
|
|
1526
|
+
id: string;
|
|
1527
|
+
title: string;
|
|
1528
|
+
dependencies: string[];
|
|
1529
|
+
parallelGroups: number[][];
|
|
1530
|
+
}[];
|
|
1531
|
+
parallelGroups: number[][];
|
|
1532
|
+
};
|
|
1533
|
+
tick(): Promise<void>;
|
|
1534
|
+
private startTask;
|
|
1535
|
+
}
|
|
1536
|
+
|
|
1537
|
+
/**
|
|
1538
|
+
* Everything needed to produce a plan from a goal. Carries the planning context
|
|
1539
|
+
* that previously threaded through the scheduler as positional args.
|
|
1540
|
+
*/
|
|
1541
|
+
interface PlanRequest {
|
|
1542
|
+
goal: string;
|
|
1543
|
+
runners: RunnerId[];
|
|
1544
|
+
modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>;
|
|
1545
|
+
fs?: IFileSystem;
|
|
1546
|
+
fetcher?: IWebFetcher;
|
|
1547
|
+
onProgress?: (progress: ResearchProgress) => void;
|
|
1548
|
+
runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
|
|
1549
|
+
autonomousDefault?: boolean;
|
|
1550
|
+
signal?: AbortSignal;
|
|
1551
|
+
perRunnerAllowlist?: Partial<Record<RunnerId, string[]>>;
|
|
1552
|
+
/** The mode toggles this run honours. `modesFor('one-shot', …)` decides which apply. */
|
|
1553
|
+
modes?: PlannerModes;
|
|
1554
|
+
}
|
|
1555
|
+
interface ModifyPlanRequest {
|
|
1556
|
+
existingPlan: LegacyPlanState;
|
|
1557
|
+
userRequest: string;
|
|
1558
|
+
modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>;
|
|
1559
|
+
fs?: IFileSystem;
|
|
1560
|
+
fetcher?: IWebFetcher;
|
|
1561
|
+
onProgress?: (progress: ResearchProgress) => void;
|
|
1562
|
+
runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
|
|
1563
|
+
autonomousDefault?: boolean;
|
|
1564
|
+
perRunnerAllowlist?: Partial<Record<RunnerId, string[]>>;
|
|
1565
|
+
modes?: PlannerModes;
|
|
1566
|
+
}
|
|
1567
|
+
interface ModifyDuringExecutionRequest {
|
|
1568
|
+
executionLog: TaskSnapshot[];
|
|
1569
|
+
pendingTasks: Task[];
|
|
1570
|
+
activeSessions: Map<string, ActiveTaskSession>;
|
|
1571
|
+
userMessage: string;
|
|
1572
|
+
modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>;
|
|
1573
|
+
runners: RunnerId[];
|
|
1574
|
+
runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
|
|
1575
|
+
autonomousDefault?: boolean;
|
|
1576
|
+
perRunnerAllowlist?: Partial<Record<RunnerId, string[]>>;
|
|
1577
|
+
}
|
|
1578
|
+
interface ModifyDuringExecutionResult {
|
|
1579
|
+
pendingTasks: Task[];
|
|
1580
|
+
message: string;
|
|
1581
|
+
}
|
|
1582
|
+
/**
|
|
1583
|
+
* Owns one-shot plan generation and plan modification — the "planning is a
|
|
1584
|
+
* different workload from execution" thesis, as a module. The conversational
|
|
1585
|
+
* planning loop (ADR-0002) lives on {@link Session}, which talks to the AI
|
|
1586
|
+
* service directly because the conversation state is held there.
|
|
1587
|
+
*/
|
|
1588
|
+
declare class Planner {
|
|
1589
|
+
private config;
|
|
1590
|
+
/**
|
|
1591
|
+
* Resolved per call, never captured: the planner backend can change while a
|
|
1592
|
+
* surface is open (`/planner`, the webview pills), and a Planner holding the
|
|
1593
|
+
* service it was built with would keep planning on the backend the host has
|
|
1594
|
+
* already switched away from.
|
|
1595
|
+
*/
|
|
1596
|
+
private readonly resolveAiService;
|
|
1597
|
+
constructor(config: IConfig, aiService: IAiService | (() => IAiService));
|
|
1598
|
+
private get aiService();
|
|
1599
|
+
/** One-shot plan generation for non-conversational surfaces. Never asks questions. */
|
|
1600
|
+
generate(req: PlanRequest): Promise<LegacyPlanState>;
|
|
1601
|
+
modify(req: ModifyPlanRequest): Promise<{
|
|
1602
|
+
tasks: Task[];
|
|
1603
|
+
}>;
|
|
1604
|
+
modifyDuringExecution(req: ModifyDuringExecutionRequest): Promise<ModifyDuringExecutionResult>;
|
|
1605
|
+
}
|
|
1606
|
+
|
|
1607
|
+
interface CollectedContext {
|
|
1608
|
+
agentConfig: string | null;
|
|
1609
|
+
agentConfigPath: string | null;
|
|
1610
|
+
dirStructure: string | null;
|
|
1611
|
+
aiflowContext: string | null;
|
|
1612
|
+
}
|
|
1613
|
+
declare class ContextCollector {
|
|
1614
|
+
private fs;
|
|
1615
|
+
private registry?;
|
|
1616
|
+
constructor(fs: IFileSystem, registry?: RunnerRegistry | undefined);
|
|
1617
|
+
setRegistry(registry: RunnerRegistry): void;
|
|
1618
|
+
collect(runner: string): Promise<CollectedContext>;
|
|
1619
|
+
private readIfExists;
|
|
1620
|
+
}
|
|
1621
|
+
|
|
1622
|
+
/**
|
|
1623
|
+
* Injectable command runner. Defaults to the real `child_process` exec wrapper;
|
|
1624
|
+
* tests pass a fake so discovery never spawns a process.
|
|
1625
|
+
*/
|
|
1626
|
+
type ExecImpl = (command: string, options?: {
|
|
1627
|
+
timeout?: number;
|
|
1628
|
+
}) => Promise<{
|
|
1629
|
+
stdout: string;
|
|
1630
|
+
}>;
|
|
1631
|
+
declare function discoverGeminiModels(apiKey: string, baseUrl?: string, fetchImpl?: typeof fetch): Promise<DiscoveredModel[]>;
|
|
1632
|
+
|
|
1633
|
+
interface ModelResolverDeps {
|
|
1634
|
+
fetchImpl?: typeof fetch;
|
|
1635
|
+
execImpl?: ExecImpl;
|
|
1636
|
+
}
|
|
1637
|
+
declare class ModelResolver {
|
|
1638
|
+
private registry;
|
|
1639
|
+
private config;
|
|
1640
|
+
private discovery;
|
|
1641
|
+
private fetchImpl?;
|
|
1642
|
+
private cachedPickerOptions;
|
|
1643
|
+
private lastDiscoveryErrors;
|
|
1644
|
+
constructor(registry: RunnerRegistry, config: IConfig, deps?: ModelResolverDeps);
|
|
1645
|
+
static builtinOptions(): OrchestratorOption[];
|
|
1646
|
+
modelsForRunners(runners: RunnerId[]): Promise<Partial<Record<RunnerId, DiscoveredModel[]>>>;
|
|
1647
|
+
pickerOptions(): Promise<OrchestratorOption[]>;
|
|
1648
|
+
/**
|
|
1649
|
+
* Per-provider failures from the most recent `pickerOptions()` fetch, keyed
|
|
1650
|
+
* by provider id. Empty when every configured provider's catalog loaded (or
|
|
1651
|
+
* before the first fetch). Surfaces consume this to flag a provider whose
|
|
1652
|
+
* key/endpoint is set but whose catalog could not be reached.
|
|
1653
|
+
*/
|
|
1654
|
+
getDiscoveryErrors(): Record<string, string>;
|
|
1655
|
+
refresh(): Promise<ProviderModelLists>;
|
|
1656
|
+
refreshRunnerModels(): void;
|
|
1657
|
+
invalidate(): void;
|
|
1658
|
+
}
|
|
1659
|
+
|
|
1660
|
+
/**
|
|
1661
|
+
* Detects whether a runner's underlying CLI is actually installed on the host,
|
|
1662
|
+
* by invoking `<command> --version`. Mirrors ModelDiscovery: injectable exec,
|
|
1663
|
+
* augmented PATH, per-process cache. A runner that isn't installed should never
|
|
1664
|
+
* be offered for selection in any surface.
|
|
1665
|
+
*/
|
|
1666
|
+
declare class RunnerInstallation {
|
|
1667
|
+
private cache;
|
|
1668
|
+
private inflight;
|
|
1669
|
+
private registry;
|
|
1670
|
+
private execImpl;
|
|
1671
|
+
constructor(registry: RunnerRegistry, execImpl?: ExecImpl);
|
|
1672
|
+
/** True if the runner's CLI responds to `--version`. Cached per process. */
|
|
1673
|
+
isInstalled(runner: string): Promise<boolean>;
|
|
1674
|
+
/**
|
|
1675
|
+
* Whether this runner can serve as a harness planner (ADR-0009, T9), and why
|
|
1676
|
+
* not when it cannot. Surfaces call this to grey out an unusable agent in the
|
|
1677
|
+
* planner picker with the reason attached — discovering a missing CLI after
|
|
1678
|
+
* typing a real goal is the failure this exists to prevent.
|
|
1679
|
+
*
|
|
1680
|
+
* "Usable" is deliberately shallow: the binary answers `--version`, and the
|
|
1681
|
+
* runner declares a planner transport. Whether the user's subscription is
|
|
1682
|
+
* live cannot be known without spending a turn on it, so an expired login
|
|
1683
|
+
* surfaces where it actually bites — as the agent's own error text on the
|
|
1684
|
+
* first turn.
|
|
1685
|
+
*/
|
|
1686
|
+
plannerUsability(runner: string): Promise<{
|
|
1687
|
+
usable: boolean;
|
|
1688
|
+
reason?: string;
|
|
1689
|
+
}>;
|
|
1690
|
+
/**
|
|
1691
|
+
* Whether the CLI can be started the way the harness planner actually starts
|
|
1692
|
+
* it: `spawn` with no shell.
|
|
1693
|
+
*
|
|
1694
|
+
* `isInstalled` probes through `exec`, which goes through cmd.exe on Windows
|
|
1695
|
+
* and so happily resolves a `.cmd` shim — while the planner's own spawn does
|
|
1696
|
+
* not, because CreateProcess performs no PATHEXT lookup. That gap produced
|
|
1697
|
+
* the worst failure shape available: the picker reported the runner healthy,
|
|
1698
|
+
* then the session died on ENOENT with nothing to act on. POSIX has no such
|
|
1699
|
+
* split (`execvp` and `exec` search PATH identically), so this always agrees
|
|
1700
|
+
* with `isInstalled` there.
|
|
1701
|
+
*/
|
|
1702
|
+
private isSpawnable;
|
|
1703
|
+
/** Subset of the given runner ids whose CLI is installed. */
|
|
1704
|
+
filterInstalled(runners: string[]): Promise<string[]>;
|
|
1705
|
+
clear(): void;
|
|
1706
|
+
}
|
|
1707
|
+
|
|
1708
|
+
declare const PROVIDER_LABEL: Record<AiProvider, string>;
|
|
1709
|
+
declare const PROVIDER_SHORT_LABEL: Record<AiProvider, string>;
|
|
1710
|
+
interface ProviderRegistration {
|
|
1711
|
+
id: AiProvider;
|
|
1712
|
+
label: string;
|
|
1713
|
+
shortLabel: string;
|
|
1714
|
+
/** `cli` is a harness planner (ADR-0009): a local coding agent, not an HTTP vendor. */
|
|
1715
|
+
serviceType: 'openai' | 'google' | 'cli';
|
|
1716
|
+
/** Harness planners only: the runner id whose manifest, models and binary this provider drives. */
|
|
1717
|
+
runnerId?: string;
|
|
1718
|
+
defaultBaseUrl: string;
|
|
1719
|
+
apiKeyEnvVar: string;
|
|
1720
|
+
detectEnvVars: string[];
|
|
1721
|
+
baseUrlEnvVar?: string;
|
|
1722
|
+
secretStoreKey?: string;
|
|
1723
|
+
vscodeBaseUrlKey?: string;
|
|
1724
|
+
vscodeApiKeyKey?: string;
|
|
1725
|
+
discoversModels: boolean;
|
|
1726
|
+
modelPrefix?: string;
|
|
1727
|
+
}
|
|
1728
|
+
declare const ALL_PROVIDERS: Record<AiProvider, ProviderRegistration>;
|
|
1729
|
+
declare function getProviderMeta(id: AiProvider): ProviderRegistration;
|
|
1730
|
+
declare function prefixModelId(provider: AiProvider, modelId: string): string;
|
|
1731
|
+
declare function stripModelPrefix(id: string, provider: AiProvider): string;
|
|
1732
|
+
declare function resolveProviderFromPrefix(id: string): AiProvider | null;
|
|
1733
|
+
/**
|
|
1734
|
+
* The harness planners, in display order. Deliberately NOT part of
|
|
1735
|
+
* `PROVIDER_PRIORITY`: that list means "vendors that take an API key", and
|
|
1736
|
+
* every consumer of it — key prompts, catalog fetches, base-URL cache clearing
|
|
1737
|
+
* — would be asking a local binary for a credential it does not have. Surfaces
|
|
1738
|
+
* that offer a planner picker concatenate the two lists and gate these on
|
|
1739
|
+
* `RunnerInstallation.plannerUsability`.
|
|
1740
|
+
*/
|
|
1741
|
+
declare const CLI_PROVIDERS: AiProvider[];
|
|
1742
|
+
declare const PROVIDER_PRIORITY: AiProvider[];
|
|
1743
|
+
declare const PROVIDER_DETECT_PRIORITY: AiProvider[];
|
|
1744
|
+
declare function isOpenAiProvider(id: AiProvider): boolean;
|
|
1745
|
+
/**
|
|
1746
|
+
* The one guard that separates a harness planner from an LLM vendor (ADR-0009).
|
|
1747
|
+
* Three places consult it: API-key resolution (skipped — the subscription is
|
|
1748
|
+
* the credential), provider routing (skipped — there is no HTTP endpoint), and
|
|
1749
|
+
* the planner-model picker (fed by per-runner discovery, not the vendor
|
|
1750
|
+
* catalog).
|
|
1751
|
+
*/
|
|
1752
|
+
declare function isCliProvider(id: AiProvider): boolean;
|
|
1753
|
+
/** The runner a harness planner drives, or null for a vendor provider. */
|
|
1754
|
+
declare function runnerForProvider(id: AiProvider): string | null;
|
|
1755
|
+
/** The harness planner that wraps a given runner, or null when none does. */
|
|
1756
|
+
declare function providerForRunner(runner: string): AiProvider | null;
|
|
1757
|
+
/**
|
|
1758
|
+
* The single definition of which providers count as "configured" — one API
|
|
1759
|
+
* key each, or (for `openai_compatible`) an explicit base URL. Every surface
|
|
1760
|
+
* (CLI listing, VS Code picker gating) derives its configured-provider set
|
|
1761
|
+
* from here so they can never diverge on what a user has set up. Order follows
|
|
1762
|
+
* PROVIDER_PRIORITY.
|
|
1763
|
+
*/
|
|
1764
|
+
declare function configuredProviders(config: {
|
|
1765
|
+
getProviderApiKey(provider: AiProvider): string;
|
|
1766
|
+
openaiCompatibleBaseUrl: string;
|
|
1767
|
+
}): AiProvider[];
|
|
1768
|
+
|
|
1769
|
+
interface SpawnSpec {
|
|
1770
|
+
command: string;
|
|
1771
|
+
args: string[];
|
|
1772
|
+
env?: Record<string, string>;
|
|
1773
|
+
}
|
|
1774
|
+
/**
|
|
1775
|
+
* Shared plumbing for the agents that speak newline-delimited JSON over stdio.
|
|
1776
|
+
*
|
|
1777
|
+
* One long-lived process per planner session (ADR-0009, T3): spawned at
|
|
1778
|
+
* `start`, fed one message per turn, disposed at the session boundary. The
|
|
1779
|
+
* alternative — respawning per turn — pays a cold start on every message
|
|
1780
|
+
* *including each corrective re-emit*, and re-reads the repository from
|
|
1781
|
+
* scratch when it cannot resume.
|
|
1782
|
+
*
|
|
1783
|
+
* Subclasses own their agent's protocol: what to spawn, how to phrase a user
|
|
1784
|
+
* turn, and how to read one protocol line. Everything below — line framing,
|
|
1785
|
+
* turn lifecycle, abort, stderr capture, premature-exit detection — is the
|
|
1786
|
+
* same for all of them.
|
|
1787
|
+
*/
|
|
1788
|
+
declare abstract class StdioAgentAdapter implements AgentAdapter {
|
|
1789
|
+
protected deps: AgentProcessDeps;
|
|
1790
|
+
abstract readonly agentId: string;
|
|
1791
|
+
protected process: ChildProcess | null;
|
|
1792
|
+
protected sessionId: string | null;
|
|
1793
|
+
private readonly stdout;
|
|
1794
|
+
private stderrTail;
|
|
1795
|
+
private exited;
|
|
1796
|
+
/** Set for the duration of a turn. Outside one, events are buffered rather than dropped. */
|
|
1797
|
+
private turnEmit;
|
|
1798
|
+
/**
|
|
1799
|
+
* Events the agent produced between turns — in practice the startup warnings
|
|
1800
|
+
* that arrive during the handshake. Dropping them hid the one diagnostic
|
|
1801
|
+
* that explains a planner which cannot read the workspace, so they are held
|
|
1802
|
+
* until a turn exists to show them in.
|
|
1803
|
+
*/
|
|
1804
|
+
private betweenTurns;
|
|
1805
|
+
private disposed;
|
|
1806
|
+
/** Resolves when the process ends, so a handshake can lose the race instead of waiting out its timeout. */
|
|
1807
|
+
protected processEnded: Promise<void>;
|
|
1808
|
+
/** The environment the agent was spawned under, for any side process a handshake needs. */
|
|
1809
|
+
protected spawnEnv: NodeJS.ProcessEnv;
|
|
1810
|
+
private markEnded;
|
|
1811
|
+
constructor(deps: AgentProcessDeps);
|
|
1812
|
+
/** The command line that starts this agent in its read-only mode. */
|
|
1813
|
+
protected abstract spawnSpec(opts: AgentStartOptions): SpawnSpec;
|
|
1814
|
+
/** The bytes written to stdin to open one user turn. Must end with a newline. */
|
|
1815
|
+
protected abstract turnPayload(message: string): string;
|
|
1816
|
+
/**
|
|
1817
|
+
* Interpret one line of the agent's protocol, emitting normalized events.
|
|
1818
|
+
* Emitting `turn_end` ends the turn; emitting `error` ends it as a failure.
|
|
1819
|
+
*/
|
|
1820
|
+
protected abstract handleLine(line: string, emit: (event: AgentEvent) => void): void;
|
|
1821
|
+
/** Protocol handshake, if the agent needs one before it accepts a turn. */
|
|
1822
|
+
protected handshake(_opts: AgentStartOptions): Promise<void>;
|
|
1823
|
+
start(opts: AgentStartOptions): Promise<void>;
|
|
1824
|
+
send(message: string, onEvent: (event: AgentEvent) => void, signal?: AbortSignal): Promise<void>;
|
|
1825
|
+
nativeSessionId(): string | null;
|
|
1826
|
+
dispose(): void;
|
|
1827
|
+
/** Write one raw protocol line to the agent. */
|
|
1828
|
+
protected writeLine(payload: unknown): void;
|
|
1829
|
+
/** Fail-safe contract: a dead agent reports its own last words, never an empty bubble. */
|
|
1830
|
+
protected exitMessage(): string;
|
|
1831
|
+
/** Parse a protocol line, ignoring the non-JSON banners some CLIs print. */
|
|
1832
|
+
protected static parse<T>(line: string): T | null;
|
|
1833
|
+
}
|
|
1834
|
+
|
|
1835
|
+
/**
|
|
1836
|
+
* Claude Code as a planner, over its bidirectional streaming-JSON transport
|
|
1837
|
+
* (ADR-0009).
|
|
1838
|
+
*
|
|
1839
|
+
* `-p --input-format stream-json --output-format stream-json` keeps one process
|
|
1840
|
+
* alive across turns: user messages go in as JSON lines, and the session's
|
|
1841
|
+
* assistant blocks, tool uses, tool results and turn boundaries come back the
|
|
1842
|
+
* same way. It is the richest of the three streams — partial messages and
|
|
1843
|
+
* separate thinking blocks — which is why this agent went first.
|
|
1844
|
+
*/
|
|
1845
|
+
declare class ClaudeCodeAdapter extends StdioAgentAdapter {
|
|
1846
|
+
readonly agentId = "claude-code";
|
|
1847
|
+
protected spawnSpec(opts: AgentStartOptions): SpawnSpec;
|
|
1848
|
+
protected turnPayload(message: string): string;
|
|
1849
|
+
protected handleLine(line: string, emit: (event: AgentEvent) => void): void;
|
|
1850
|
+
}
|
|
1851
|
+
|
|
1852
|
+
/**
|
|
1853
|
+
* Codex as a planner, over its `app-server` stdio JSON-RPC transport
|
|
1854
|
+
* (ADR-0009).
|
|
1855
|
+
*
|
|
1856
|
+
* Ordewell already speaks a slice of this protocol — `ModelDiscovery` drives
|
|
1857
|
+
* `initialize` → `model/list` to build the Codex model catalog — so the
|
|
1858
|
+
* handshake here is that one continued into a thread.
|
|
1859
|
+
*
|
|
1860
|
+
* Codex's own CLI marks `app-server` experimental, and it is the transport
|
|
1861
|
+
* most likely to drift: the method names below (`thread/start`, `turn/start`,
|
|
1862
|
+
* `item/completed`) come from the schema the installed binary generates, and a
|
|
1863
|
+
* version that renames them will surface as a visible dead turn rather than a
|
|
1864
|
+
* hang, because the base class watches the process as well as the protocol.
|
|
1865
|
+
*/
|
|
1866
|
+
declare class CodexAdapter extends StdioAgentAdapter {
|
|
1867
|
+
readonly agentId = "codex";
|
|
1868
|
+
private threadId;
|
|
1869
|
+
private nextRequestId;
|
|
1870
|
+
private settleHandshake;
|
|
1871
|
+
private handshakeError;
|
|
1872
|
+
private startOpts;
|
|
1873
|
+
/** Whether this turn has already emitted prose — see the `agentMessage` case. */
|
|
1874
|
+
private turnHasText;
|
|
1875
|
+
private resumeAttempted;
|
|
1876
|
+
private resumeFallbackSent;
|
|
1877
|
+
private sandbox;
|
|
1878
|
+
protected spawnSpec(opts: AgentStartOptions): SpawnSpec;
|
|
1879
|
+
/**
|
|
1880
|
+
* `initialize`, then `thread/start`, both before the first user message. The
|
|
1881
|
+
* thread is pinned to the read-only sandbox with approvals set to `never` —
|
|
1882
|
+
* with nobody watching a planner's prompts, "ask" would mean "hang", which is
|
|
1883
|
+
* ADR-0008's absent-is-denial invariant kept by construction.
|
|
1884
|
+
*/
|
|
1885
|
+
protected handshake(opts: AgentStartOptions): Promise<void>;
|
|
1886
|
+
/**
|
|
1887
|
+
* Open the thread this session plans in. A resume id means the previous
|
|
1888
|
+
* process died mid-session: `thread/resume` puts the agent back in front of
|
|
1889
|
+
* the context it already paid to read. A failed resume is not an error — the
|
|
1890
|
+
* response handler falls back to a fresh thread, which is the same
|
|
1891
|
+
* degradation `restoreChat` performs on every surface (T4).
|
|
1892
|
+
*/
|
|
1893
|
+
private startThread;
|
|
1894
|
+
protected turnPayload(message: string): string;
|
|
1895
|
+
protected handleLine(line: string, emit: (event: AgentEvent) => void): void;
|
|
1896
|
+
/**
|
|
1897
|
+
* Refuse one server→client request. Requests whose result schema can express
|
|
1898
|
+
* a refusal get that payload; everything else — a permission grant, a
|
|
1899
|
+
* question for a user who is not watching, a tool call the client is supposed
|
|
1900
|
+
* to run — gets a JSON-RPC error, which Codex surfaces to the model as a
|
|
1901
|
+
* failed request and plans around, rather than waiting on.
|
|
1902
|
+
*/
|
|
1903
|
+
private answerServerRequest;
|
|
1904
|
+
/** A tool item entering `inProgress` — announce the call so the timeline moves. */
|
|
1905
|
+
private emitItemStart;
|
|
1906
|
+
private emitItemDone;
|
|
1907
|
+
}
|
|
1908
|
+
|
|
1909
|
+
/**
|
|
1910
|
+
* OpenCode as a planner, over its headless HTTP server (ADR-0009).
|
|
1911
|
+
*
|
|
1912
|
+
* The odd one out: `opencode serve` is a real server rather than a stdio
|
|
1913
|
+
* protocol, so this adapter owns both halves of the boundary — it spawns the
|
|
1914
|
+
* process through the same injected `spawn` every other adapter uses, then
|
|
1915
|
+
* talks to it through the injected `fetch`. Both are part of the one seam the
|
|
1916
|
+
* tests drive.
|
|
1917
|
+
*
|
|
1918
|
+
* The turn ends when the message POST resolves. Live events stream from the
|
|
1919
|
+
* server's `/event` channel, but the POST is what settles the turn: an event
|
|
1920
|
+
* name that changes between OpenCode versions then costs liveness, not
|
|
1921
|
+
* correctness.
|
|
1922
|
+
*/
|
|
1923
|
+
declare class OpenCodeAdapter implements AgentAdapter {
|
|
1924
|
+
private deps;
|
|
1925
|
+
readonly agentId = "opencode";
|
|
1926
|
+
private process;
|
|
1927
|
+
private baseUrl;
|
|
1928
|
+
private sessionId;
|
|
1929
|
+
private stderrTail;
|
|
1930
|
+
private exited;
|
|
1931
|
+
private disposed;
|
|
1932
|
+
private opts;
|
|
1933
|
+
constructor(deps: AgentProcessDeps);
|
|
1934
|
+
start(opts: AgentStartOptions): Promise<void>;
|
|
1935
|
+
send(message: string, onEvent: (event: AgentEvent) => void, signal?: AbortSignal): Promise<void>;
|
|
1936
|
+
/**
|
|
1937
|
+
* Emit one message part, once. OpenCode reports a tool part repeatedly as it
|
|
1938
|
+
* moves through pending → running → completed, so parts are keyed by id and
|
|
1939
|
+
* only the terminal state produces a result.
|
|
1940
|
+
*/
|
|
1941
|
+
private emitPart;
|
|
1942
|
+
/**
|
|
1943
|
+
* Deny one permission request (T1). OpenCode blocks the turn until the
|
|
1944
|
+
* request is answered, so this must answer — `reject` rather than a silent
|
|
1945
|
+
* drop, which is the same "absent answer is a denial" invariant ADR-0008
|
|
1946
|
+
* states for Ordewell's own tools. The refusal is announced so the timeline
|
|
1947
|
+
* shows the planner reaching for something it may not have.
|
|
1948
|
+
*/
|
|
1949
|
+
private denyPermission;
|
|
1950
|
+
/**
|
|
1951
|
+
* Server-sent events from `/event`. Tool activity on it is liveness only —
|
|
1952
|
+
* the settled POST repeats it — but permission requests arrive nowhere else,
|
|
1953
|
+
* so the stream is load-bearing for {@link denyPermission}.
|
|
1954
|
+
*/
|
|
1955
|
+
private streamEvents;
|
|
1956
|
+
private json;
|
|
1957
|
+
nativeSessionId(): string | null;
|
|
1958
|
+
dispose(): void;
|
|
1959
|
+
private exitMessage;
|
|
1960
|
+
}
|
|
1961
|
+
|
|
1962
|
+
interface MappedTool {
|
|
1963
|
+
tool: ResearchToolType;
|
|
1964
|
+
/** The agent's own name, kept whenever it differs from the member it mapped to. */
|
|
1965
|
+
toolLabel?: string;
|
|
1966
|
+
}
|
|
1967
|
+
/**
|
|
1968
|
+
* Classify one agent tool name. Case- and separator-insensitive, because the
|
|
1969
|
+
* three agents disagree on `Read` vs `read` vs `read_file` for the same thing.
|
|
1970
|
+
*/
|
|
1971
|
+
declare function mapAgentTool(name: string): MappedTool;
|
|
1972
|
+
/**
|
|
1973
|
+
* Normalize an agent's tool arguments into the shapes `summarizeToolCall`
|
|
1974
|
+
* already knows how to read, so a harness `Read` gets a one-line summary
|
|
1975
|
+
* instead of a bare tool name. Unknown keys are preserved — the raw args stay
|
|
1976
|
+
* visible under the surfaces' output chevron.
|
|
1977
|
+
*/
|
|
1978
|
+
declare function normalizeAgentArgs(tool: ResearchToolType, args: Record<string, unknown>): Record<string, unknown>;
|
|
1979
|
+
|
|
1980
|
+
type SerializedTaskStatus = {
|
|
1981
|
+
id: string;
|
|
1982
|
+
status: string;
|
|
1983
|
+
verdict: {
|
|
1984
|
+
outcome: 'pass' | 'fail';
|
|
1985
|
+
reason: string;
|
|
1986
|
+
checks: Verdict['checks'];
|
|
1987
|
+
} | null;
|
|
1988
|
+
};
|
|
1989
|
+
type SerializedTask = {
|
|
1990
|
+
id: string;
|
|
1991
|
+
order: number;
|
|
1992
|
+
title: string;
|
|
1993
|
+
type: string;
|
|
1994
|
+
description: string;
|
|
1995
|
+
dependencies: string[];
|
|
1996
|
+
assignedRunner: RunnerId;
|
|
1997
|
+
assignedModel: Task['assignedModel'] | null;
|
|
1998
|
+
taskMode: string;
|
|
1999
|
+
prompt: string | null;
|
|
2000
|
+
subtasks: SerializedTask[];
|
|
2001
|
+
userSteps: Task['userSteps'];
|
|
2002
|
+
thinkingEffort: Task['thinkingEffort'];
|
|
2003
|
+
autonomy: Task['autonomy'];
|
|
2004
|
+
sliceType: Task['sliceType'];
|
|
2005
|
+
userStoriesCovered: Task['userStoriesCovered'];
|
|
2006
|
+
};
|
|
2007
|
+
type SerializedPlan = {
|
|
2008
|
+
tasks: SerializedTask[];
|
|
2009
|
+
runners: RunnerId[];
|
|
2010
|
+
generatedAt: string;
|
|
2011
|
+
conversationHistory?: LegacyPlanState['conversationHistory'];
|
|
2012
|
+
prdMarkdown?: string;
|
|
2013
|
+
queuedMessages?: QueuedMessage[];
|
|
2014
|
+
};
|
|
2015
|
+
type SessionMessage = {
|
|
2016
|
+
type: 'plan_generated';
|
|
2017
|
+
plan: SerializedPlan;
|
|
2018
|
+
goal: string;
|
|
2019
|
+
runners: RunnerId[];
|
|
2020
|
+
} | {
|
|
2021
|
+
type: 'planner_message';
|
|
2022
|
+
content: string;
|
|
2023
|
+
timestamp: string;
|
|
2024
|
+
} | {
|
|
2025
|
+
type: 'status_update';
|
|
2026
|
+
tasks: SerializedTaskStatus[];
|
|
2027
|
+
} | {
|
|
2028
|
+
type: 'review_needed';
|
|
2029
|
+
tasks: SerializedTask[];
|
|
2030
|
+
} | {
|
|
2031
|
+
type: 'review_approved';
|
|
2032
|
+
} | {
|
|
2033
|
+
type: 'checkpoint';
|
|
2034
|
+
taskId: string;
|
|
2035
|
+
taskTitle: string;
|
|
2036
|
+
summary: string;
|
|
2037
|
+
} | {
|
|
2038
|
+
type: 'execution_complete';
|
|
2039
|
+
summary: {
|
|
2040
|
+
total: number;
|
|
2041
|
+
completed: number;
|
|
2042
|
+
failed: number;
|
|
2043
|
+
};
|
|
2044
|
+
} | {
|
|
2045
|
+
type: 'execution_stopped';
|
|
2046
|
+
} | {
|
|
2047
|
+
type: 'queue_ready';
|
|
2048
|
+
} | {
|
|
2049
|
+
type: 'task_updated';
|
|
2050
|
+
taskId: string;
|
|
2051
|
+
changes: Record<string, unknown>;
|
|
2052
|
+
} | {
|
|
2053
|
+
type: 'task_started';
|
|
2054
|
+
taskId: string;
|
|
2055
|
+
order: number;
|
|
2056
|
+
title: string;
|
|
2057
|
+
runner: RunnerId;
|
|
2058
|
+
modelId?: string;
|
|
2059
|
+
} | {
|
|
2060
|
+
type: 'task_output';
|
|
2061
|
+
taskId: string;
|
|
2062
|
+
text: string;
|
|
2063
|
+
} | {
|
|
2064
|
+
type: 'plan_thinking';
|
|
2065
|
+
text: string;
|
|
2066
|
+
} | {
|
|
2067
|
+
type: 'research_step';
|
|
2068
|
+
tool: string;
|
|
2069
|
+
toolLabel?: string;
|
|
2070
|
+
args: string;
|
|
2071
|
+
subagentId?: string;
|
|
2072
|
+
toolCallId?: string;
|
|
2073
|
+
} | {
|
|
2074
|
+
type: 'plan_token';
|
|
2075
|
+
token: string;
|
|
2076
|
+
} | {
|
|
2077
|
+
type: 'research_step_done';
|
|
2078
|
+
step: ResearchStep;
|
|
2079
|
+
subagentId?: string;
|
|
2080
|
+
} | {
|
|
2081
|
+
type: 'approval_request';
|
|
2082
|
+
id: string;
|
|
2083
|
+
kind: ApprovalKind;
|
|
2084
|
+
subject: string;
|
|
2085
|
+
scope: string;
|
|
2086
|
+
detail?: string;
|
|
2087
|
+
} | {
|
|
2088
|
+
type: 'approval_settled';
|
|
2089
|
+
id: string;
|
|
2090
|
+
granted: boolean;
|
|
2091
|
+
} | {
|
|
2092
|
+
type: 'approval_decided';
|
|
2093
|
+
kind: ApprovalKind;
|
|
2094
|
+
subject: string;
|
|
2095
|
+
scope: string;
|
|
2096
|
+
detail?: string;
|
|
2097
|
+
granted: boolean;
|
|
2098
|
+
source: Exclude<ApprovalSource, 'asked'>;
|
|
2099
|
+
};
|
|
2100
|
+
type SessionBroadcaster = (msg: SessionMessage) => void;
|
|
2101
|
+
declare function serializeTask(t: Task): SerializedTask;
|
|
2102
|
+
declare function serializeTaskStatus(t: Task): SerializedTaskStatus;
|
|
2103
|
+
declare function serializePlan(plan: LegacyPlanState): SerializedPlan;
|
|
2104
|
+
declare function executionSummary(tasks: Task[]): {
|
|
2105
|
+
total: number;
|
|
2106
|
+
completed: number;
|
|
2107
|
+
failed: number;
|
|
2108
|
+
};
|
|
2109
|
+
|
|
2110
|
+
/**
|
|
2111
|
+
* Options for plan generation. Progress is not overridable: every planner
|
|
2112
|
+
* progress event is translated to a SessionMessage inside the Session and
|
|
2113
|
+
* emitted through the broadcast seam, so all surfaces consume one union.
|
|
2114
|
+
*/
|
|
2115
|
+
interface GeneratePlanOptions {
|
|
2116
|
+
signal?: AbortSignal;
|
|
2117
|
+
}
|
|
2118
|
+
/** The slice of Planner the Session drives — the injection seam for tests. */
|
|
2119
|
+
type SessionPlanner = Pick<Planner, 'generate' | 'modify' | 'modifyDuringExecution'>;
|
|
2120
|
+
/** Runtime prefs read live — may toggle between operations. */
|
|
2121
|
+
interface SessionRuntimeSettings {
|
|
2122
|
+
tddEnabled: boolean;
|
|
2123
|
+
grillMeEnabled: boolean;
|
|
2124
|
+
prdEnabled?: boolean;
|
|
2125
|
+
reviewEnabled?: boolean;
|
|
2126
|
+
verificationEnabled?: boolean;
|
|
2127
|
+
researchSubagentsEnabled?: boolean;
|
|
2128
|
+
modelAllowlist?: Record<string, string[]>;
|
|
2129
|
+
}
|
|
2130
|
+
/**
|
|
2131
|
+
* The whole of what a host reads off disk for a Session. Both hosts used to
|
|
2132
|
+
* assemble this by hand, mapping each toggle's settings key to its runtime key
|
|
2133
|
+
* in two blocks nothing kept in step — which is how one toggle came to be
|
|
2134
|
+
* dropped. `MODE_TOGGLES` holds the mapping now; this adds the one field that
|
|
2135
|
+
* is not a toggle.
|
|
2136
|
+
*/
|
|
2137
|
+
declare function sessionRuntimeSettings(settings: UserSettings): SessionRuntimeSettings;
|
|
2138
|
+
/**
|
|
2139
|
+
* Everything a delivery surface constructs to host a session. Structural config
|
|
2140
|
+
* (enabledRunners, orchestratorModel, providerModelLists) is snapshotted inside
|
|
2141
|
+
* `config` at construction and never re-read from the environment. Runtime
|
|
2142
|
+
* settings (tdd, grillMe) are read live via the `settings` callback so a toggle
|
|
2143
|
+
* between generate and execute takes effect.
|
|
2144
|
+
*/
|
|
2145
|
+
interface SessionDeps {
|
|
2146
|
+
config: IConfig;
|
|
2147
|
+
notifications: INotification;
|
|
2148
|
+
runner: ITerminalRunner;
|
|
2149
|
+
registry: RunnerRegistry;
|
|
2150
|
+
/** Resolves the workspace root for the orchestrator (lazy — VS Code can change it). */
|
|
2151
|
+
workspaceRoot: () => string;
|
|
2152
|
+
/** Filesystem adapter for planner research. */
|
|
2153
|
+
fsAdapter: IFileSystem;
|
|
2154
|
+
/** Emits plan-lifecycle events to the surface. Transport-agnostic. */
|
|
2155
|
+
broadcast: SessionBroadcaster;
|
|
2156
|
+
/** Shared across sessions — sole producer of model catalogs and routing lists. */
|
|
2157
|
+
modelResolver: ModelResolver;
|
|
2158
|
+
/** Live runtime settings (tdd, grillMe). Read at each operation that needs them. */
|
|
2159
|
+
settings: () => SessionRuntimeSettings;
|
|
2160
|
+
/**
|
|
2161
|
+
* Host-assigned session id. When set, every persist writes under this id so
|
|
2162
|
+
* the host's REST/UI ids match the saved-session store. When omitted, the
|
|
2163
|
+
* Session mints a fresh id per plan (generatePlan/startPlanning).
|
|
2164
|
+
*/
|
|
2165
|
+
sessionId?: string;
|
|
2166
|
+
/**
|
|
2167
|
+
* Planner-conversation seam. Defaults to the provider service for
|
|
2168
|
+
* `config.aiProvider`; inject a fake to test the conversation half without
|
|
2169
|
+
* an LLM.
|
|
2170
|
+
*/
|
|
2171
|
+
aiService?: IAiService;
|
|
2172
|
+
/** Plan-generation seam. Defaults to a Planner over the session's aiService. */
|
|
2173
|
+
planner?: SessionPlanner;
|
|
2174
|
+
}
|
|
2175
|
+
/**
|
|
2176
|
+
* The per-session execution stack — deepened from a wiring bag into the
|
|
2177
|
+
* lifecycle owner. Owns plan generation, execution, mutation, persistence, and
|
|
2178
|
+
* the orchestrator observer wiring. The orchestrator's observer is subscribed
|
|
2179
|
+
* once for the session's lifetime (not per-operation), which kills the
|
|
2180
|
+
* double-subscribe class of bug. Persistence is an internal seam: every plan
|
|
2181
|
+
* mutation routes through `persist()`, so the obligation has a home instead of
|
|
2182
|
+
* being scattered across 11 call sites.
|
|
2183
|
+
*
|
|
2184
|
+
* The broadcast seam carries {@link SessionMessage} — the 15 plan-lifecycle
|
|
2185
|
+
* events. Catalog/config messages (setModels, setRunnerList, …) stay on the
|
|
2186
|
+
* host; Session never emits them.
|
|
2187
|
+
*/
|
|
2188
|
+
declare class Session {
|
|
2189
|
+
/** Injected by a test; when present it is the service, forever. */
|
|
2190
|
+
private readonly pinnedAiService?;
|
|
2191
|
+
private liveAiService;
|
|
2192
|
+
private liveAiProvider;
|
|
2193
|
+
private readonly workspaceRootFn;
|
|
2194
|
+
private planner;
|
|
2195
|
+
private orchestrator;
|
|
2196
|
+
private store;
|
|
2197
|
+
private config;
|
|
2198
|
+
private registry;
|
|
2199
|
+
private plan;
|
|
2200
|
+
private goal;
|
|
2201
|
+
private workspace;
|
|
2202
|
+
private broadcast;
|
|
2203
|
+
private modelResolver;
|
|
2204
|
+
private fsAdapter;
|
|
2205
|
+
private approvals;
|
|
2206
|
+
private approvalPolicy;
|
|
2207
|
+
private fetcher;
|
|
2208
|
+
private settingsFn;
|
|
2209
|
+
private currentAllowlist;
|
|
2210
|
+
/** Last discovered model catalog — lets sync plan commits clamp thinking efforts to real variants. */
|
|
2211
|
+
private modelsCache;
|
|
2212
|
+
private unsubObserver;
|
|
2213
|
+
private readonly hostSessionId?;
|
|
2214
|
+
private currentSessionId;
|
|
2215
|
+
constructor(deps: SessionDeps);
|
|
2216
|
+
/**
|
|
2217
|
+
* The planner transport for the provider configured *right now* (ADR-0009).
|
|
2218
|
+
*
|
|
2219
|
+
* Resolved on every read rather than once in the constructor, because a
|
|
2220
|
+
* Session outlives the choice: VS Code hosts exactly one for the whole
|
|
2221
|
+
* window, and the webview pills and `/planner` switch backends underneath it.
|
|
2222
|
+
* The model id was already read live, so a service captured at construction
|
|
2223
|
+
* meant a switched planner kept the old backend and got handed the new one's
|
|
2224
|
+
* model — an OpenCode model id spawned as `claude --model opencode/…`, which
|
|
2225
|
+
* the agent rejects as nonexistent.
|
|
2226
|
+
*
|
|
2227
|
+
* Switching releases the outgoing service: a harness planner holds an OS
|
|
2228
|
+
* process, so dropping the reference without `reset()` leaks an agent.
|
|
2229
|
+
*/
|
|
2230
|
+
private get aiService();
|
|
2231
|
+
/**
|
|
2232
|
+
* Answer an outstanding approval. Every surface funnels here — the web
|
|
2233
|
+
* server's HTTP route, the VS Code webview, the TUI prompt — so the decision
|
|
2234
|
+
* path is identical regardless of who is looking.
|
|
2235
|
+
*/
|
|
2236
|
+
resolveApproval(id: string, granted: boolean): boolean;
|
|
2237
|
+
/** Requests still waiting for an answer, replayed to a surface that connects mid-prompt. */
|
|
2238
|
+
outstandingApprovals(): PendingApproval[];
|
|
2239
|
+
/** Scopes the user granted this session — surfaced so a UI can show what is already allowed. */
|
|
2240
|
+
approvedScopes(): string[];
|
|
2241
|
+
/** The stable id this session persists under — matches the host's id when one was provided. */
|
|
2242
|
+
get sessionId(): string;
|
|
2243
|
+
get executionLog(): TaskSnapshot[];
|
|
2244
|
+
/** Tasks always read from PlanStore — the single source of truth. */
|
|
2245
|
+
get planTasks(): Task[];
|
|
2246
|
+
private attachObserver;
|
|
2247
|
+
private buildObserver;
|
|
2248
|
+
private translateProgress;
|
|
2249
|
+
/** Persists PlanStore state to disk. PlanStore is the single authority;
|
|
2250
|
+
* LegacyPlanState.tasks is populated only here, at persist time. */
|
|
2251
|
+
private persist;
|
|
2252
|
+
/** A new plan on a long-lived Session gets its own persisted identity (unless the host fixed one). */
|
|
2253
|
+
private remintSessionId;
|
|
2254
|
+
/**
|
|
2255
|
+
* A new plan starts from zero: drop the live planner conversation and every
|
|
2256
|
+
* task, log, and queued message left over from a previous plan on this
|
|
2257
|
+
* Session. Without this, a long-lived Session (VS Code hosts exactly one)
|
|
2258
|
+
* leaks the previous session's tasks into `planContextBlock()` — the model
|
|
2259
|
+
* is told they are the CURRENT plan and re-presents them as its draft.
|
|
2260
|
+
*/
|
|
2261
|
+
private beginFreshPlan;
|
|
2262
|
+
/**
|
|
2263
|
+
* Return the Session to a blank slate — hosts call this on "new session".
|
|
2264
|
+
* Everything scoped to the old session goes: the live AI conversation, plan,
|
|
2265
|
+
* goal, tasks, execution log, queued messages, model cache, and (unless the
|
|
2266
|
+
* host fixed one) the persisted identity, so nothing can bleed into the next
|
|
2267
|
+
* session.
|
|
2268
|
+
*/
|
|
2269
|
+
reset(): void;
|
|
2270
|
+
private runnerModesFor;
|
|
2271
|
+
get planState(): LegacyPlanState | null;
|
|
2272
|
+
/**
|
|
2273
|
+
* The live plan in the shape a saved session is read back as. The disk
|
|
2274
|
+
* boundary rewrites `in_progress` to `pending` — nothing is running when a
|
|
2275
|
+
* session comes off a file — so a surface that re-reads the plan mid-run must
|
|
2276
|
+
* come here instead, or every task it is watching reads as never started.
|
|
2277
|
+
*/
|
|
2278
|
+
get currentPlanState(): PlanState | null;
|
|
2279
|
+
get currentGoal(): string;
|
|
2280
|
+
get isPlanning(): boolean;
|
|
2281
|
+
get isExecuting(): boolean;
|
|
2282
|
+
get status(): 'approved' | 'running' | 'completed';
|
|
2283
|
+
get sessionConfig(): IConfig;
|
|
2284
|
+
startExecution(): Promise<void>;
|
|
2285
|
+
generatePlan(goal: string, runners: RunnerId[], options?: GeneratePlanOptions): Promise<LegacyPlanState>;
|
|
2286
|
+
/**
|
|
2287
|
+
* Kick off the planner conversation (ADR-0002): research + the first planner
|
|
2288
|
+
* message. The AI service retains the tool-use history; the Session persists
|
|
2289
|
+
* the user/assistant dialogue on the plan state.
|
|
2290
|
+
*/
|
|
2291
|
+
startPlanning(goal: string, runners: RunnerId[], options?: GeneratePlanOptions): Promise<LegacyPlanState>;
|
|
2292
|
+
/**
|
|
2293
|
+
* Every subsequent user reply in the planning conversation — grill-me
|
|
2294
|
+
* answers, PRD accept/adjust, outline confirm. One branch, no phase ladder.
|
|
2295
|
+
*/
|
|
2296
|
+
continueConversation(userMessage: string, options?: GeneratePlanOptions): Promise<LegacyPlanState>;
|
|
2297
|
+
/**
|
|
2298
|
+
* Drive a planner turn to a persisted, broadcast outcome. Task edits apply
|
|
2299
|
+
* atomically; validation failures are fed back to the model for up to 2
|
|
2300
|
+
* silent retries, then surfaced as a message with the plan untouched. The
|
|
2301
|
+
* first turn (startPlanning) and every later turn route through here — one
|
|
2302
|
+
* path, not two.
|
|
2303
|
+
*/
|
|
2304
|
+
private settleTurn;
|
|
2305
|
+
/**
|
|
2306
|
+
* The "you are here" block for post-plan chat: current tasks with stable
|
|
2307
|
+
* references, plus the task-ops protocol. Injected per turn (never
|
|
2308
|
+
* persisted) so the model always sees live statuses — including which tasks
|
|
2309
|
+
* are locked by a running execution.
|
|
2310
|
+
*/
|
|
2311
|
+
private planContextBlock;
|
|
2312
|
+
/** Validate + commit a task_ops turn atomically. Returns the errors on rejection (plan untouched). */
|
|
2313
|
+
private applyTaskOpsTurn;
|
|
2314
|
+
/**
|
|
2315
|
+
* Resume a persisted dialogue onto a fresh AI-service conversation — either
|
|
2316
|
+
* because the in-memory one is gone (session reload, extension restart), or
|
|
2317
|
+
* because {@link continueConversation} found it stale against the planner
|
|
2318
|
+
* config now in effect (a harness planner's model/effort switched mid-chat).
|
|
2319
|
+
* The saved transcript seeds the new conversation; no LLM call happens
|
|
2320
|
+
* before the user's message is sent.
|
|
2321
|
+
*/
|
|
2322
|
+
private resumeConversation;
|
|
2323
|
+
/** Whether the planner conversation is live (started and not yet committed to a plan). */
|
|
2324
|
+
get isConversationActive(): boolean;
|
|
2325
|
+
/** Append one dialogue entry to the persisted transcript. Callers persist via the mutatePlan ritual. */
|
|
2326
|
+
private appendTranscript;
|
|
2327
|
+
/** Commit a settled (non-task_ops) turn. Both branches run the mutatePlan ritual. */
|
|
2328
|
+
private applyConversationTurn;
|
|
2329
|
+
/**
|
|
2330
|
+
* PRD mode: when the planner writes the full markdown PRD, save it to
|
|
2331
|
+
* .scratch/<slug>/PRD.md (Matt Pocock to-prd convention) and keep it on the plan.
|
|
2332
|
+
*/
|
|
2333
|
+
private capturePrd;
|
|
2334
|
+
executePlan(): Promise<void>;
|
|
2335
|
+
approveReview(): Promise<LegacyPlanState>;
|
|
2336
|
+
forceStartTask(taskId: string): Promise<void>;
|
|
2337
|
+
runTask(taskId: string): Promise<void>;
|
|
2338
|
+
retryTask(taskId: string): Promise<void>;
|
|
2339
|
+
cancelTask(taskId: string): Promise<void>;
|
|
2340
|
+
markAiTaskComplete(taskId: string): Promise<void>;
|
|
2341
|
+
markTaskComplete(taskId: string): Promise<void>;
|
|
2342
|
+
markTaskIncomplete(taskId: string): Promise<void>;
|
|
2343
|
+
tick(): Promise<void>;
|
|
2344
|
+
processQueuedMessages(): Promise<void>;
|
|
2345
|
+
approveCheckpoint(taskId: string): void;
|
|
2346
|
+
rejectCheckpoint(taskId: string, reason?: string): void;
|
|
2347
|
+
queueMessage(text: string): void;
|
|
2348
|
+
getQueuedMessages(): QueuedMessage[];
|
|
2349
|
+
setQueuedMessages(msgs: ReturnType<TaskOrchestrator['getQueuedMessages']>): void;
|
|
2350
|
+
processNextQueuedMessage(): QueuedMessage | null;
|
|
2351
|
+
get queuedCount(): number;
|
|
2352
|
+
getTask(taskId: string): Task | undefined;
|
|
2353
|
+
get isReviewApproved(): boolean;
|
|
2354
|
+
stopExecution(): void;
|
|
2355
|
+
/**
|
|
2356
|
+
* The one mutation seam: every structural plan mutation runs the same
|
|
2357
|
+
* ritual — store op → persist (which snapshots tasks from PlanStore) →
|
|
2358
|
+
* broadcast — in that order, once. A store op returning false aborts
|
|
2359
|
+
* before anything is persisted. PlanStore is the single authority for
|
|
2360
|
+
* task state; LegacyPlanState.tasks is populated only at persist time.
|
|
2361
|
+
*/
|
|
2362
|
+
private mutatePlan;
|
|
2363
|
+
updateTask(taskId: string, changes: Partial<Task>): LegacyPlanState | null;
|
|
2364
|
+
/**
|
|
2365
|
+
* Move one task onto a different runner. Distinct from {@link updateTask}
|
|
2366
|
+
* because a runner change is never a single-field edit: the task's model,
|
|
2367
|
+
* thinking effort and mode are all scoped to its runner, so they are
|
|
2368
|
+
* re-derived from the new runner's catalog (see {@link retargetTaskRunner}).
|
|
2369
|
+
* That needs discovery, which is async — hence a method of its own rather
|
|
2370
|
+
* than a branch inside the sync `updateTask`.
|
|
2371
|
+
*
|
|
2372
|
+
* The runner is also admitted into `plan.runners` — see {@link admitRunner}.
|
|
2373
|
+
*/
|
|
2374
|
+
setTaskRunner(taskId: string, runner: RunnerId): Promise<LegacyPlanState | null>;
|
|
2375
|
+
/** What a runner offers, as {@link runnerAssignment} needs it. Spawns the runner's CLI to list models. */
|
|
2376
|
+
private catalogFor;
|
|
2377
|
+
/**
|
|
2378
|
+
* What a *derived* assignment may draw from: the runner's catalog narrowed to
|
|
2379
|
+
* the user's allowlist. Deriving from the full catalog would hand a task the
|
|
2380
|
+
* runner's first model regardless of a restriction the user set — the next
|
|
2381
|
+
* planner turn's `coerceAssignments` would snap it back anyway, so the user
|
|
2382
|
+
* would see their pick silently change instead of never being offered.
|
|
2383
|
+
*
|
|
2384
|
+
* `catalogFor` stays unnarrowed because {@link admitRunner} caches it as what
|
|
2385
|
+
* the runner really offers, which is what effort clamping needs.
|
|
2386
|
+
*/
|
|
2387
|
+
private allowedCatalog;
|
|
2388
|
+
/**
|
|
2389
|
+
* Make a runner a first-class member of this plan. Without this, the next
|
|
2390
|
+
* planner turn's `coerceAssignments` would treat it as disallowed and snap
|
|
2391
|
+
* every task on it back, silently undoing the user's choice; and that same
|
|
2392
|
+
* pass clamps efforts against `modelsCache`, so a catalog missing from there
|
|
2393
|
+
* makes the effort we just derived read as unverifiable.
|
|
2394
|
+
*/
|
|
2395
|
+
private admitRunner;
|
|
2396
|
+
/**
|
|
2397
|
+
* Replace one task's dependency list. Separate from {@link updateTask}
|
|
2398
|
+
* because a hand-edited graph is the one task edit that can leave a plan
|
|
2399
|
+
* unschedulable: `canSetDependencies` is the guard, and it lives behind this
|
|
2400
|
+
* one method so the API and both surfaces' pickers reject the same edits
|
|
2401
|
+
* rather than each carrying a copy of the rule. Throws so a surface can say
|
|
2402
|
+
* why the edit was refused.
|
|
2403
|
+
*/
|
|
2404
|
+
setTaskDependencies(taskId: string, dependencies: string[]): LegacyPlanState | null;
|
|
2405
|
+
completeTask(taskId: string): Promise<void>;
|
|
2406
|
+
removeTask(taskId: string): LegacyPlanState | null;
|
|
2407
|
+
/**
|
|
2408
|
+
* Add one task, filling in whatever the caller left unset. A task with no
|
|
2409
|
+
* runnable assignment is not a lighter task but an unspawnable one, so the
|
|
2410
|
+
* runner falls back to the plan's first and the model, effort and mode are
|
|
2411
|
+
* derived from that runner's catalog — the same derivation a runner change
|
|
2412
|
+
* uses ({@link runnerAssignment}), which is why this is async like
|
|
2413
|
+
* {@link setTaskRunner}. Anything the caller did choose survives when the
|
|
2414
|
+
* runner offers it.
|
|
2415
|
+
*
|
|
2416
|
+
* Dependencies naming tasks that don't exist are dropped rather than rejected:
|
|
2417
|
+
* the caller is a picker over the current plan, so a stale id means the plan
|
|
2418
|
+
* moved on, not that the whole task should be refused.
|
|
2419
|
+
*/
|
|
2420
|
+
addTask(draft: Partial<Task>): Promise<LegacyPlanState | null>;
|
|
2421
|
+
mergeTasks(taskIdA: string, taskIdB: string): LegacyPlanState | null;
|
|
2422
|
+
mergeMultipleTasks(taskIds: string[]): LegacyPlanState | null;
|
|
2423
|
+
splitTask(taskId: string, newTasks: Partial<Task>[]): LegacyPlanState | null;
|
|
2424
|
+
/**
|
|
2425
|
+
* Planner-driven merge: validate compatibility up front, then ask the planner
|
|
2426
|
+
* LLM to produce a single "merge" taskOps op combining the selected tasks.
|
|
2427
|
+
* Goes through the same conversation loop + validated-atomic-edit + corrective
|
|
2428
|
+
* retry flow as every other task_ops edit (ADR-0002). Throws on a
|
|
2429
|
+
* pre-flight compatibility failure so the host surfaces an inline error
|
|
2430
|
+
* before any LLM call.
|
|
2431
|
+
*/
|
|
2432
|
+
requestMerge(taskIds: string[], options?: GeneratePlanOptions): Promise<LegacyPlanState>;
|
|
2433
|
+
/**
|
|
2434
|
+
* Planner-driven split: ask the planner LLM to decompose one task into a
|
|
2435
|
+
* sequence of smaller tasks. The model generates the breakdown (no manual
|
|
2436
|
+
* per-task specs from the user). Same conversation-loop/repair path as merge.
|
|
2437
|
+
*/
|
|
2438
|
+
requestSplit(taskId: string, options?: GeneratePlanOptions): Promise<LegacyPlanState>;
|
|
2439
|
+
loadPlan(plan: LegacyPlanState, goal: string, workspace: string, opts?: {
|
|
2440
|
+
sessionId?: string;
|
|
2441
|
+
persist?: boolean;
|
|
2442
|
+
}): void;
|
|
2443
|
+
modifyPlan(userRequest: string): Promise<Task[]>;
|
|
2444
|
+
destroy(): void;
|
|
2445
|
+
private broadcastPlan;
|
|
2446
|
+
get aiServiceInstance(): IAiService;
|
|
2447
|
+
}
|
|
2448
|
+
|
|
2449
|
+
type VerdictListener = (taskId: string, verdict: Verdict, output: string) => void;
|
|
2450
|
+
type CheckpointListener = (taskId: string, summary: string) => void;
|
|
2451
|
+
declare class VerdictEngine {
|
|
2452
|
+
private markerSeen;
|
|
2453
|
+
private buffers;
|
|
2454
|
+
private checkpointCounts;
|
|
2455
|
+
private pausedSessions;
|
|
2456
|
+
private listeners;
|
|
2457
|
+
private checkpointListeners;
|
|
2458
|
+
/**
|
|
2459
|
+
* Per-task generation counter. Incremented on every watch() and clear().
|
|
2460
|
+
* Stale callbacks (from a prior session whose generation doesn't match
|
|
2461
|
+
* the current one) bail out instead of delivering a verdict for the
|
|
2462
|
+
* wrong session.
|
|
2463
|
+
*/
|
|
2464
|
+
private generations;
|
|
2465
|
+
onVerdict(listener: VerdictListener): void;
|
|
2466
|
+
onCheckpoint(listener: CheckpointListener): void;
|
|
2467
|
+
approveCheckpoint(taskId: string): void;
|
|
2468
|
+
rejectCheckpoint(taskId: string, reason: string): void;
|
|
2469
|
+
/**
|
|
2470
|
+
* Attach to a spawned session: buffer output, scan for the task's completion
|
|
2471
|
+
* marker (delivering a verdict immediately while leaving interactive sessions
|
|
2472
|
+
* open), scan for checkpoint markers, and on exit produce a failed verdict
|
|
2473
|
+
* when the marker was never observed.
|
|
2474
|
+
*/
|
|
2475
|
+
watch(task: Task, session: ITerminalSession): void;
|
|
2476
|
+
/** Manual "Mark complete" override: a pass verdict that bypasses evidence. */
|
|
2477
|
+
markComplete(task: Task): Verdict;
|
|
2478
|
+
/** Clear verification state for a task (used on retry). */
|
|
2479
|
+
clear(task: Task): void;
|
|
2480
|
+
/** Drop all tracking state (used on stop / loadPlan). */
|
|
2481
|
+
reset(): void;
|
|
2482
|
+
private decide;
|
|
2483
|
+
}
|
|
2484
|
+
|
|
2485
|
+
/**
|
|
2486
|
+
* The ids a runner may actually run: the user's allowlist with the ids that
|
|
2487
|
+
* provably belong to a *different* runner dropped. `undefined` means no
|
|
2488
|
+
* restriction.
|
|
2489
|
+
*
|
|
2490
|
+
* Model ids are scoped to the agent that lists them, so an OpenRouter slug
|
|
2491
|
+
* allowlisted for Claude Code does not limit it — it points it at something it
|
|
2492
|
+
* cannot spawn, and the plan only dies once a task is already running. But
|
|
2493
|
+
* "this runner didn't list it" is not enough to call an id wrong: discovery can
|
|
2494
|
+
* be stale, and a plugin runner may have no list at all. What settles it is
|
|
2495
|
+
* whether *another* runner listed the id. So:
|
|
2496
|
+
*
|
|
2497
|
+
* - listed for this runner → keep;
|
|
2498
|
+
* - listed for no runner → unknown, and the user said it explicitly, so keep it
|
|
2499
|
+
* and let the runner validate last (the same call `coerceAssignments` and
|
|
2500
|
+
* `TaskRetarget` make);
|
|
2501
|
+
* - listed only for other runners → provably not this runner's, so drop it.
|
|
2502
|
+
*
|
|
2503
|
+
* When that leaves nothing, the whole allowlist was about some other runner
|
|
2504
|
+
* (a settings file written before a surface scoped its picker to one). No
|
|
2505
|
+
* restriction is the only safe reading — the alternative is handing the planner
|
|
2506
|
+
* a list of models that cannot run.
|
|
2507
|
+
*/
|
|
2508
|
+
declare function effectiveAllowlist(allowlist: string[] | undefined, runner: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>> | undefined): string[] | undefined;
|
|
2509
|
+
declare function filterModelsForPrompt(modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, perRunnerAllowlist: Partial<Record<RunnerId, string[]>>): Partial<Record<RunnerId, DiscoveredModel[]>>;
|
|
2510
|
+
/**
|
|
2511
|
+
* Clamp a planner/user-supplied thinking effort to what the model actually
|
|
2512
|
+
* offers. Invalid values map to the nearest rung the model supports (so a
|
|
2513
|
+
* Claude-style "xhigh" on a low/medium/high model becomes "high") and fall
|
|
2514
|
+
* back to undefined — the runner's own default — when no mapping exists.
|
|
2515
|
+
*/
|
|
2516
|
+
declare function clampThinkingEffort(effort: string | undefined, variants: {
|
|
2517
|
+
id: string;
|
|
2518
|
+
}[]): string | undefined;
|
|
2519
|
+
declare function coerceAssignments(tasks: Task[], perRunnerAllowlist: Partial<Record<RunnerId, string[]>>, allowedRunners?: RunnerId[], modelsByRunner?: Partial<Record<RunnerId, DiscoveredModel[]>>): Task[];
|
|
2520
|
+
|
|
2521
|
+
/** What a runner offers a task: the models discovered for it and the modes its manifest declares. */
|
|
2522
|
+
interface RunnerCatalog {
|
|
2523
|
+
models: DiscoveredModel[];
|
|
2524
|
+
modes: RunnerModeInfo[];
|
|
2525
|
+
}
|
|
2526
|
+
/**
|
|
2527
|
+
* The model, thinking effort and mode a runner offers for a task.
|
|
2528
|
+
*
|
|
2529
|
+
* A task's model, effort and mode are all scoped to its runner:
|
|
2530
|
+
* `claude-sonnet-4-5` on Codex, or `acceptEdits` on OpenCode, are not degraded
|
|
2531
|
+
* choices but unspawnable ones. So this is where any task acquires a runnable
|
|
2532
|
+
* assignment — a runner change (below) and a hand-added task derive it the same
|
|
2533
|
+
* way. Each field in `current` is preserved when the runner also offers it, and
|
|
2534
|
+
* otherwise snapped to that runner's preferred entry (discovery already sorts
|
|
2535
|
+
* models by the manifest's `preferredPatterns`; `modes[0]` is the manifest's own
|
|
2536
|
+
* first choice).
|
|
2537
|
+
*
|
|
2538
|
+
* An empty catalog means discovery failed or the runner is a plugin we have no
|
|
2539
|
+
* list for — not that the runner offers nothing. That field is left out of the
|
|
2540
|
+
* patch and the runner validates last, matching `coerceAssignments`.
|
|
2541
|
+
*/
|
|
2542
|
+
declare function runnerAssignment(catalog: RunnerCatalog, current?: Pick<Task, 'assignedModel' | 'taskMode'>): Partial<Task>;
|
|
2543
|
+
/**
|
|
2544
|
+
* Move a task onto a different runner, carrying its model, thinking effort and
|
|
2545
|
+
* mode over to values that runner actually offers. A runner change is never a
|
|
2546
|
+
* single-field edit — it either brings the other three with it or leaves the
|
|
2547
|
+
* task unrunnable.
|
|
2548
|
+
*
|
|
2549
|
+
* Returns the patch to apply, or `{}` when there is nothing to change.
|
|
2550
|
+
*/
|
|
2551
|
+
declare function retargetTaskRunner(task: Task, runner: RunnerId, catalog: RunnerCatalog): Partial<Task>;
|
|
2552
|
+
|
|
2553
|
+
declare function buildResearchToolsPrompt(subagentsEnabled?: boolean): string;
|
|
2554
|
+
/**
|
|
2555
|
+
* System prompt for one read-only research subagent (issue #34). The digest
|
|
2556
|
+
* contract matters: the reply goes back to the planner as a tool result, so it
|
|
2557
|
+
* must be dense, self-contained, and carry exact file paths — never questions,
|
|
2558
|
+
* never a task plan.
|
|
2559
|
+
*/
|
|
2560
|
+
declare function buildSubagentSystemPrompt(): string;
|
|
2561
|
+
declare function buildConversationSystemPrompt(goal: string, context: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean, grillMeEnabled?: boolean, prdEnabled?: boolean, reviewEnabled?: boolean, verificationEnabled?: boolean,
|
|
2562
|
+
/** Harness planner (ADR-0009): the agent owns its own tools and research budget. */
|
|
2563
|
+
harnessMode?: boolean): string;
|
|
2564
|
+
declare const CORE_PLANNER_PROMPT: string;
|
|
2565
|
+
declare function buildResearchPrompt(userGoal: string, context: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes): string;
|
|
2566
|
+
declare function buildPlanWithResults(userGoal: string, context: string, researchResults: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, modes?: PlannerModes): string;
|
|
2567
|
+
declare function buildModifyPlanPrompt(existingPlan: LegacyPlanState, userRequest: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, aiflowContext?: string, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): string;
|
|
2568
|
+
declare function modelContextBlock(modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): string;
|
|
2569
|
+
declare function executionLogBlock(executionLog: TaskSnapshot[]): string;
|
|
2570
|
+
declare function pendingEditRulesBlock(): string;
|
|
2571
|
+
declare function buildModifyDuringExecutionPrompt(executionLog: TaskSnapshot[], pendingTasks: string, userMessage: string, modelsByRunner: Partial<Record<RunnerId, DiscoveredModel[]>>, runners: RunnerId[], runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): string;
|
|
2572
|
+
/**
|
|
2573
|
+
* User-message prompt for a planner-driven merge: asks the model to combine the
|
|
2574
|
+
* selected tasks into one. Routed through the conversation loop, so the
|
|
2575
|
+
* task-ops protocol (with the "merge" op) is injected alongside it by
|
|
2576
|
+
* `planContextBlock`. The model emits a single taskOps merge op.
|
|
2577
|
+
*/
|
|
2578
|
+
declare function buildMergePrompt(taskIds: string[], tasks: Task[]): string;
|
|
2579
|
+
/**
|
|
2580
|
+
* User-message prompt for a planner-driven split: asks the model to decompose
|
|
2581
|
+
* one task into a sequence of smaller tasks. The model decides the breakdown —
|
|
2582
|
+
* the user does not hand-type the parts.
|
|
2583
|
+
*/
|
|
2584
|
+
declare function buildSplitPrompt(taskId: string, tasks: Task[]): string;
|
|
2585
|
+
|
|
2586
|
+
/**
|
|
2587
|
+
* The one owner of "the model emitted something unusable — correct it and
|
|
2588
|
+
* retry". Every repair path in the planner routes through this module:
|
|
2589
|
+
*
|
|
2590
|
+
* - {@link repairLoop} is the bounded driver (first reply → interpret →
|
|
2591
|
+
* corrective re-send), used by plan generation ({@link generatePlanWithRepair}),
|
|
2592
|
+
* the Session's task-ops settlement, and Planner.modifyDuringExecution.
|
|
2593
|
+
* - {@link classifyPlannerReply} decides what a planner reply *is* — a plan,
|
|
2594
|
+
* targeted task edits, a botched attempt at either (worth a corrective
|
|
2595
|
+
* retry), or prose. The conversation loop in BaseAiService keeps its own
|
|
2596
|
+
* driver (it also runs the tool rounds) but delegates classification here.
|
|
2597
|
+
* - The corrective prompt texts live here, once.
|
|
2598
|
+
*
|
|
2599
|
+
* Policies stay at the call sites (PRD nudge, ops validation, abort guards) —
|
|
2600
|
+
* they are inputs to the loop, not part of it.
|
|
2601
|
+
*/
|
|
2602
|
+
type RepairVerdict<T> = {
|
|
2603
|
+
done: T;
|
|
2604
|
+
} | {
|
|
2605
|
+
retry: {
|
|
2606
|
+
errors: string[];
|
|
2607
|
+
corrective: string;
|
|
2608
|
+
cause?: unknown;
|
|
2609
|
+
};
|
|
2610
|
+
};
|
|
2611
|
+
interface RepairLoopOpts<R, T> {
|
|
2612
|
+
/** Produce the first reply (attempt 0) — often already in hand. */
|
|
2613
|
+
first: () => R | Promise<R>;
|
|
2614
|
+
/** Ask the model again with a corrective prompt after a retryable failure. */
|
|
2615
|
+
resend: (corrective: string) => Promise<R>;
|
|
2616
|
+
/**
|
|
2617
|
+
* Judge one reply: `done` with the settled value, or `retry` with the
|
|
2618
|
+
* errors and the corrective prompt to send. Throw for non-retryable
|
|
2619
|
+
* failures — those propagate immediately.
|
|
2620
|
+
*/
|
|
2621
|
+
interpret: (reply: R) => RepairVerdict<T> | Promise<RepairVerdict<T>>;
|
|
2622
|
+
/** Corrective re-sends allowed after the first attempt (N repairs = N+1 attempts). */
|
|
2623
|
+
maxRepairs: number;
|
|
2624
|
+
/** Budget exhausted: receives the last reply and its errors; return a fallback or throw. */
|
|
2625
|
+
onExhausted: (last: {
|
|
2626
|
+
reply: R;
|
|
2627
|
+
errors: string[];
|
|
2628
|
+
cause?: unknown;
|
|
2629
|
+
}) => T;
|
|
2630
|
+
}
|
|
2631
|
+
declare function repairLoop<R, T>(opts: RepairLoopOpts<R, T>): Promise<T>;
|
|
2632
|
+
/** Instruction appended to a follow-up request asking the model to re-emit strict JSON. */
|
|
2633
|
+
declare const JSON_REPAIR_INSTRUCTION: string;
|
|
2634
|
+
/**
|
|
2635
|
+
* The emission hit the output-token limit: a plain "re-send the JSON" retry
|
|
2636
|
+
* would be cut off at the same point, so this one asks for terser output.
|
|
2637
|
+
*/
|
|
2638
|
+
declare const TRUNCATED_PLAN_REPAIR_INSTRUCTION: string;
|
|
2639
|
+
/** A plan attempt was botched (invalid or truncated JSON): re-emit the whole plan. */
|
|
2640
|
+
declare function reEmitPlanPrompt(detail: string): string;
|
|
2641
|
+
/** A plan emission was truncated mid-JSON; the caller may also have compacted the history to free input context. */
|
|
2642
|
+
declare function truncatedPlanReEmitPrompt(historyCompacted: boolean): string;
|
|
2643
|
+
/** A task-ops attempt was botched (unparseable JSON): re-emit the ops object. */
|
|
2644
|
+
declare function reEmitTaskOpsPrompt(detail: string): string;
|
|
2645
|
+
/** Task ops parsed but failed semantic validation (cycles, unknown refs, locked tasks). */
|
|
2646
|
+
declare function taskOpsRejectedPrompt(errors: string[]): string;
|
|
2647
|
+
/** A modified plan failed validation during execution: full re-prompt feedback block. */
|
|
2648
|
+
declare function modifyValidationFeedback(errors: string[]): string;
|
|
2649
|
+
type PlannerReplyClassification = {
|
|
2650
|
+
kind: 'plan';
|
|
2651
|
+
tasks: Task[];
|
|
2652
|
+
} | {
|
|
2653
|
+
kind: 'task_ops';
|
|
2654
|
+
ops: TaskOp[];
|
|
2655
|
+
}
|
|
2656
|
+
/** Clearly attempted a plan (tasks-keyed object, or JSON cut off mid-stream) and botched it — worth a corrective retry. */
|
|
2657
|
+
| {
|
|
2658
|
+
kind: 'broken_plan';
|
|
2659
|
+
error: PlanParseError;
|
|
2660
|
+
}
|
|
2661
|
+
/** Clearly attempted an ops object and botched it — worth a corrective retry. */
|
|
2662
|
+
| {
|
|
2663
|
+
kind: 'broken_task_ops';
|
|
2664
|
+
error: PlanParseError;
|
|
2665
|
+
} | {
|
|
2666
|
+
kind: 'prose';
|
|
2667
|
+
};
|
|
2668
|
+
/**
|
|
2669
|
+
* Classify one planner reply. Task ops are checked before the plan key — an
|
|
2670
|
+
* ops object never carries a top-level tasks array, but a model may mention
|
|
2671
|
+
* the word in prose around it. "Broken" is deliberately narrower than "failed
|
|
2672
|
+
* to parse": prose that merely mentions the envelope key is left alone.
|
|
2673
|
+
*/
|
|
2674
|
+
declare function classifyPlannerReply(text: string, opts: {
|
|
2675
|
+
runners: RunnerId[];
|
|
2676
|
+
runnerModes?: Record<RunnerId, RunnerModeInfo[]>;
|
|
2677
|
+
autonomousDefault?: boolean;
|
|
2678
|
+
}): PlannerReplyClassification;
|
|
2679
|
+
/**
|
|
2680
|
+
* Generate a plan, retrying on unparseable JSON. `generate` is called with an
|
|
2681
|
+
* optional repair hint (undefined on the first attempt, {@link JSON_REPAIR_INSTRUCTION}
|
|
2682
|
+
* thereafter) and must return the model's raw text. Only {@link PlanParseError} is
|
|
2683
|
+
* retried; transport/other errors propagate immediately. The last parse error is
|
|
2684
|
+
* re-thrown if every attempt fails.
|
|
2685
|
+
*/
|
|
2686
|
+
declare function generatePlanWithRepair(generate: (repairHint?: string) => Promise<string>, runners: RunnerId[], maxAttempts?: number, runnerModes?: Record<RunnerId, RunnerModeInfo[]>, autonomousDefault?: boolean): Promise<Task[]>;
|
|
2687
|
+
|
|
2688
|
+
type RunnerMode = string;
|
|
2689
|
+
|
|
2690
|
+
declare function buildRunnerInvocation(opts: {
|
|
2691
|
+
runner: string;
|
|
2692
|
+
prompt: string;
|
|
2693
|
+
modelId?: string;
|
|
2694
|
+
thinkingEffort?: string;
|
|
2695
|
+
modelVariants?: string[];
|
|
2696
|
+
mode?: RunnerMode;
|
|
2697
|
+
headless?: boolean;
|
|
2698
|
+
registry: RunnerRegistry;
|
|
2699
|
+
}): RunnerInvocation;
|
|
2700
|
+
|
|
2701
|
+
type ExecFileFn = (command: string, args: string[]) => Promise<{
|
|
2702
|
+
stdout: string;
|
|
2703
|
+
stderr: string;
|
|
2704
|
+
}>;
|
|
2705
|
+
/** Test seam: every OS touchpoint is injectable; production uses the defaults. */
|
|
2706
|
+
interface TmuxRunnerDeps {
|
|
2707
|
+
port: number;
|
|
2708
|
+
execFileImpl?: ExecFileFn;
|
|
2709
|
+
resolvePath?: () => Promise<string>;
|
|
2710
|
+
pollIntervalMs?: number;
|
|
2711
|
+
logDir?: string;
|
|
2712
|
+
}
|
|
2713
|
+
declare class TmuxSession extends AbstractTerminalSession {
|
|
2714
|
+
private tmuxSession;
|
|
2715
|
+
private windowName;
|
|
2716
|
+
private execFileImpl;
|
|
2717
|
+
private pollIntervalMs;
|
|
2718
|
+
private logPath;
|
|
2719
|
+
private socket;
|
|
2720
|
+
private outputBuffer;
|
|
2721
|
+
private offset;
|
|
2722
|
+
private timer;
|
|
2723
|
+
constructor(id: string, taskId: string, tmuxSession: string, windowName: string, execFileImpl: ExecFileFn, pollIntervalMs: number, logPath: string, socket: string);
|
|
2724
|
+
private get target();
|
|
2725
|
+
/** Every tmux call must name the daemon's own socket; an unprefixed one
|
|
2726
|
+
* silently targets the shared default server (see `tmuxSocketName`). */
|
|
2727
|
+
private tmux;
|
|
2728
|
+
/**
|
|
2729
|
+
* Runs the invocation in a fresh tmux window, tailed via `pipe-pane` into a
|
|
2730
|
+
* log file rather than polling `capture-pane` snapshots — a byte-exact
|
|
2731
|
+
* stream needs no diff/resync heuristics. The command is wrapped so the
|
|
2732
|
+
* window survives its own exit (an `exec`'d login shell), which is what
|
|
2733
|
+
* lets the user keep poking at a finished task's terminal.
|
|
2734
|
+
*/
|
|
2735
|
+
start(command: string, args: string[], cwd: string, env?: Record<string, string>): Promise<void>;
|
|
2736
|
+
private poll;
|
|
2737
|
+
/**
|
|
2738
|
+
* A task finishing (or being killed) stops observation, but never the
|
|
2739
|
+
* window itself on the sentinel path — only `kill()` closes the window.
|
|
2740
|
+
* Piping is turned off and the log file removed either way, so a finished
|
|
2741
|
+
* task doesn't leave a `cat` process appending to it for the rest of the
|
|
2742
|
+
* daemon's life.
|
|
2743
|
+
*/
|
|
2744
|
+
private finish;
|
|
2745
|
+
kill(): void;
|
|
2746
|
+
getOutput(): string;
|
|
2747
|
+
write(text: string): void;
|
|
2748
|
+
}
|
|
2749
|
+
/**
|
|
2750
|
+
* Runs tasks in a real tmux window instead of a piped subprocess, so a user
|
|
2751
|
+
* can open a genuine, interactive terminal on any task (running or
|
|
2752
|
+
* finished) from outside Ordewell's own process. `ITerminalSession` hides the
|
|
2753
|
+
* transport from `VerdictEngine`/`PoolAwareRunner`, so orchestration is
|
|
2754
|
+
* unchanged — this is a drop-in replacement for `HeadlessRunner`.
|
|
2755
|
+
*/
|
|
2756
|
+
declare class TmuxRunner extends AbstractRunner<TmuxSession> {
|
|
2757
|
+
private sessionName;
|
|
2758
|
+
private socket;
|
|
2759
|
+
private execFileImpl;
|
|
2760
|
+
private resolvePath;
|
|
2761
|
+
private pollIntervalMs;
|
|
2762
|
+
private logDir;
|
|
2763
|
+
/** Retries get a freshly named window so a failed attempt's output stays inspectable. */
|
|
2764
|
+
private attempts;
|
|
2765
|
+
private ready;
|
|
2766
|
+
constructor(deps: TmuxRunnerDeps);
|
|
2767
|
+
/**
|
|
2768
|
+
* Reaps a session orphaned by a crashed prior daemon on the same port, then
|
|
2769
|
+
* creates a fresh one. Memoized: `spawn` awaits it too, so a task spawned
|
|
2770
|
+
* before startup's own call has settled never lands in a missing session —
|
|
2771
|
+
* and a settled failure clears the memo so the next spawn can retry.
|
|
2772
|
+
*/
|
|
2773
|
+
ensureSession(): Promise<void>;
|
|
2774
|
+
/** Every tmux call must name this daemon's socket; see `tmuxSocketName`. */
|
|
2775
|
+
private tmux;
|
|
2776
|
+
private createFreshSession;
|
|
2777
|
+
/**
|
|
2778
|
+
* Makes an attached runner terminal scrollable with the mouse wheel and
|
|
2779
|
+
* Page Up/Down. A fresh tmux session ships with `mouse off`, a 2000-line
|
|
2780
|
+
* scrollback, and no PageUp binding — none of which lets a user page back
|
|
2781
|
+
* through a finished task's output, which is the whole point of opening its
|
|
2782
|
+
* window. Scoping is harmless: the daemon owns a private socket
|
|
2783
|
+
* (`tmuxSocketName`), so these server-wide options touch nothing but its
|
|
2784
|
+
* own session. Best-effort so an ancient or stripped tmux cannot stop tasks
|
|
2785
|
+
* from spawning — the session still works without the scroll comforts.
|
|
2786
|
+
*/
|
|
2787
|
+
private configureScrolling;
|
|
2788
|
+
/**
|
|
2789
|
+
* Called on daemon shutdown — the one guarantee against leaked tmux
|
|
2790
|
+
* processes. Kills the whole server, not just the session: the socket
|
|
2791
|
+
* belongs to this daemon alone, so nothing else can be on it, and a runner
|
|
2792
|
+
* that somehow escaped its session would otherwise keep running (and keep
|
|
2793
|
+
* billing) unattached for as long as the server lived.
|
|
2794
|
+
*/
|
|
2795
|
+
killSession(): Promise<void>;
|
|
2796
|
+
spawn(opts: {
|
|
2797
|
+
taskId: string;
|
|
2798
|
+
runner: string;
|
|
2799
|
+
prompt: string;
|
|
2800
|
+
modelId?: string;
|
|
2801
|
+
thinkingEffort?: string;
|
|
2802
|
+
modelVariants?: string[];
|
|
2803
|
+
mode?: string;
|
|
2804
|
+
headless?: boolean;
|
|
2805
|
+
cwd: string;
|
|
2806
|
+
registry?: RunnerRegistry;
|
|
2807
|
+
planSessionId?: string;
|
|
2808
|
+
}): Promise<ITerminalSession>;
|
|
2809
|
+
}
|
|
2810
|
+
|
|
2811
|
+
declare function stripAnsi(text: string): string;
|
|
2812
|
+
declare function posixShellQuote(s: string): string;
|
|
2813
|
+
/**
|
|
2814
|
+
* Wrap a command for execution inside a POSIX login shell, which is what
|
|
2815
|
+
* resolves runner binaries managed by nvm/volta/asdf.
|
|
2816
|
+
*
|
|
2817
|
+
* Deliberately POSIX-only. This used to take a `platform` and emit
|
|
2818
|
+
* `powershell.exe -Command "'claude' '-p' '…'"` for Windows, which PowerShell
|
|
2819
|
+
* cannot run at all: a quoted string in leading position is parsed in
|
|
2820
|
+
* expression mode, so the invocation died with a parse error before the runner
|
|
2821
|
+
* started — a task that failed instantly, every time, on that platform. Windows
|
|
2822
|
+
* has no login-shell equivalent to emulate (its PATH comes from the registry
|
|
2823
|
+
* and is already inherited), so {@link planShellLaunch} starts the runner
|
|
2824
|
+
* directly there instead of routing it through a shell. Platform choice belongs
|
|
2825
|
+
* to that function; this one only knows how to phrase the POSIX half.
|
|
2826
|
+
*/
|
|
2827
|
+
declare function buildShellInvocation(command: string, args: string[]): {
|
|
2828
|
+
shellPath: string;
|
|
2829
|
+
shellArgs: string[];
|
|
2830
|
+
};
|
|
2831
|
+
/**
|
|
2832
|
+
* Wrap a command in `script` to allocate the PTY some runners require when
|
|
2833
|
+
* headless; `-e` propagates the child's exit code so verification still works.
|
|
2834
|
+
*
|
|
2835
|
+
* POSIX-only by nature — there is no `script` on Windows, which
|
|
2836
|
+
* `HeadlessRunner`'s `hasScriptCmd` probe already discovers, so this is never
|
|
2837
|
+
* reached there.
|
|
2838
|
+
*/
|
|
2839
|
+
declare function wrapWithPty(command: string, args: string[]): {
|
|
2840
|
+
command: string;
|
|
2841
|
+
args: string[];
|
|
2842
|
+
};
|
|
2843
|
+
|
|
2844
|
+
interface KillTreeDeps {
|
|
2845
|
+
platform?: NodeJS.Platform;
|
|
2846
|
+
/** Runs `taskkill`. Injected so the Windows path is testable off Windows. */
|
|
2847
|
+
execFileImpl?: (file: string, args: string[], cb: (err: Error | null) => void) => void;
|
|
2848
|
+
/** Schedules the forced kill. Injected so tests need no timers. */
|
|
2849
|
+
setTimeoutImpl?: (fn: () => void, ms: number) => {
|
|
2850
|
+
unref?: () => void;
|
|
2851
|
+
};
|
|
2852
|
+
clearTimeoutImpl?: (handle: unknown) => void;
|
|
2853
|
+
}
|
|
2854
|
+
/**
|
|
2855
|
+
* Terminate `proc` and everything it started.
|
|
2856
|
+
*
|
|
2857
|
+
* Returns immediately; the forced follow-up (SIGKILL, or `taskkill /F`) is
|
|
2858
|
+
* scheduled and unref'd, so a disposed session is never the reason the host
|
|
2859
|
+
* process refuses to exit.
|
|
2860
|
+
*/
|
|
2861
|
+
declare function killTree(proc: ChildProcess | null, deps?: KillTreeDeps): void;
|
|
2862
|
+
|
|
2863
|
+
type ExecFn = (command: string, options?: {
|
|
2864
|
+
timeout?: number;
|
|
2865
|
+
}) => Promise<{
|
|
2866
|
+
stdout: string;
|
|
2867
|
+
}>;
|
|
2868
|
+
/**
|
|
2869
|
+
* Exported for test: the Windows arm cannot be exercised from a Linux CI box
|
|
2870
|
+
* through {@link augmentedPath}, which reads `process.platform` and the real
|
|
2871
|
+
* environment. Production always calls it through the no-argument form.
|
|
2872
|
+
*/
|
|
2873
|
+
declare function wellKnownBinDirs(deps?: {
|
|
2874
|
+
platform?: NodeJS.Platform;
|
|
2875
|
+
env?: NodeJS.ProcessEnv;
|
|
2876
|
+
home?: string;
|
|
2877
|
+
}): string[];
|
|
2878
|
+
/**
|
|
2879
|
+
* The augmented PATH for spawning user-installed CLI tools. Resolved once per
|
|
2880
|
+
* process and cached (the login shell query costs ~50-200ms).
|
|
2881
|
+
*/
|
|
2882
|
+
declare function augmentedPath(execImpl?: ExecFn): Promise<string>;
|
|
2883
|
+
/** Test seam: drop the per-process cache so the next call re-resolves. */
|
|
2884
|
+
declare function clearAugmentedPathCache(): void;
|
|
2885
|
+
/**
|
|
2886
|
+
* Build a child environment whose PATH is `resolvedPath`, with exactly one
|
|
2887
|
+
* PATH-ish key in it.
|
|
2888
|
+
*
|
|
2889
|
+
* `{ ...process.env, PATH }` is the obvious spelling and it is wrong on
|
|
2890
|
+
* Windows. Node's `process.env` is a case-insensitive proxy, but spreading it
|
|
2891
|
+
* yields the OS's actual casing — `Path` — so adding `PATH` produces an
|
|
2892
|
+
* environment block carrying both. Which one the child sees is undefined, and
|
|
2893
|
+
* the loser is silently discarded. Every existing site passed the same value
|
|
2894
|
+
* under both keys, so the bug was latent rather than live; this makes it
|
|
2895
|
+
* impossible instead of unlikely.
|
|
2896
|
+
*/
|
|
2897
|
+
declare function withPath(base: NodeJS.ProcessEnv, resolvedPath: string, overrides?: Record<string, string>): NodeJS.ProcessEnv;
|
|
2898
|
+
|
|
2899
|
+
/** One shared tmux session per daemon, scoped by port so multiple daemons never collide. */
|
|
2900
|
+
declare function tmuxSessionName(port: number): string;
|
|
2901
|
+
/**
|
|
2902
|
+
* Each daemon gets its own tmux *server*, not just its own session name.
|
|
2903
|
+
*
|
|
2904
|
+
* A plain `tmux new-session` attaches to whatever server already owns the
|
|
2905
|
+
* default socket, and a session created on an existing server inherits that
|
|
2906
|
+
* **server's** environment — not the environment of the process that asked for
|
|
2907
|
+
* it (tmux only refreshes `update-environment` vars, and only on attach). So a
|
|
2908
|
+
* daemon started with, say, a particular provider API key would silently hand
|
|
2909
|
+
* its runners a stale key left behind by whoever started the server first,
|
|
2910
|
+
* possibly hours earlier under a different configuration entirely.
|
|
2911
|
+
*
|
|
2912
|
+
* A private socket makes the server a child of this daemon, so the runner
|
|
2913
|
+
* inherits the daemon's environment the way any child process would, and
|
|
2914
|
+
* `kill-server` at shutdown is authoritative — a runner cannot outlive the
|
|
2915
|
+
* daemon by hiding on a shared server. Users attaching by hand need the same
|
|
2916
|
+
* flag: `tmux -L ordewell-<port> attach -t ordewell-<port>`.
|
|
2917
|
+
*/
|
|
2918
|
+
declare function tmuxSocketName(port: number): string;
|
|
2919
|
+
/**
|
|
2920
|
+
* tmux window targeting breaks on `:` and other punctuation, so ids are
|
|
2921
|
+
* slugged. Task ids are only unique within one plan ("task-1" is every
|
|
2922
|
+
* planner's favourite), so the window is also scoped by the plan session id —
|
|
2923
|
+
* without it, the second plan run in a daemon's lifetime would collide with
|
|
2924
|
+
* the first plan's still-open windows and pipe its output into the wrong one.
|
|
2925
|
+
*/
|
|
2926
|
+
declare function tmuxWindowName(taskId: string, planSessionId?: string): string;
|
|
2927
|
+
type ProbeFn = () => void;
|
|
2928
|
+
/** Feature-detects tmux the same way `HeadlessRunner` detects `script`. */
|
|
2929
|
+
declare function hasTmux(probe?: ProbeFn): boolean;
|
|
2930
|
+
|
|
2931
|
+
declare class FsPluginStore implements IPluginStore {
|
|
2932
|
+
getUserPluginsDir(): string;
|
|
2933
|
+
listUserPluginDirs(): string[];
|
|
2934
|
+
loadManifest(pluginDir: string): RunnerPluginManifest | null;
|
|
2935
|
+
copyDir(sourceDir: string, destDir: string): void;
|
|
2936
|
+
removeDir(dir: string): void;
|
|
2937
|
+
ensureDir(dir: string): void;
|
|
2938
|
+
writeFile(filePath: string, content: string): void;
|
|
2939
|
+
readFile(filePath: string): string | null;
|
|
2940
|
+
dirExists(path: string): boolean;
|
|
2941
|
+
exists(path: string): boolean;
|
|
2942
|
+
}
|
|
2943
|
+
|
|
2944
|
+
declare function resolveArgs(manifest: RunnerPluginManifest, ctx: ResolveContext): RunnerInvocation;
|
|
2945
|
+
|
|
2946
|
+
declare const CLAUDE_CODE_MANIFEST: RunnerPluginManifest;
|
|
2947
|
+
|
|
2948
|
+
declare const OPENCODE_MANIFEST: RunnerPluginManifest;
|
|
2949
|
+
|
|
2950
|
+
declare const STATE_DIR = ".ordewell";
|
|
2951
|
+
declare function ensureDir(dir: string): void;
|
|
2952
|
+
declare function getStateDir(baseDir?: string): string;
|
|
2953
|
+
|
|
2954
|
+
declare function saveState(plan: LegacyPlanState, baseDir?: string): void;
|
|
2955
|
+
declare function loadState(baseDir?: string, logger?: ILogger): LegacyPlanState | null;
|
|
2956
|
+
declare function clearState(baseDir?: string): void;
|
|
2957
|
+
declare function stateExists(baseDir?: string): boolean;
|
|
2958
|
+
|
|
2959
|
+
declare function saveSession(plan: LegacyPlanState, goal: string, baseDir?: string, id?: string): SessionMeta;
|
|
2960
|
+
declare function listSessions(baseDir?: string, logger?: ILogger): SessionMeta[];
|
|
2961
|
+
declare function loadSession(sessionId: string, baseDir?: string, logger?: ILogger): {
|
|
2962
|
+
meta: SessionMeta;
|
|
2963
|
+
plan: LegacyPlanState;
|
|
2964
|
+
} | null;
|
|
2965
|
+
declare function loadSessionPlanState(sessionId: string, baseDir?: string, logger?: ILogger): {
|
|
2966
|
+
meta: SessionMeta;
|
|
2967
|
+
plan: PlanState;
|
|
2968
|
+
} | null;
|
|
2969
|
+
declare function getLatestSession(baseDir?: string, logger?: ILogger): {
|
|
2970
|
+
meta: SessionMeta;
|
|
2971
|
+
plan: LegacyPlanState;
|
|
2972
|
+
} | null;
|
|
2973
|
+
declare function deleteSession(sessionId: string, baseDir?: string, logger?: ILogger): boolean;
|
|
2974
|
+
|
|
2975
|
+
interface PrdBlock {
|
|
2976
|
+
slug: string;
|
|
2977
|
+
markdown: string;
|
|
2978
|
+
}
|
|
2979
|
+
/**
|
|
2980
|
+
* Detect a full markdown PRD in a planner message. The markers are the only
|
|
2981
|
+
* structured artifact left in the conversation loop — they exist so the PRD
|
|
2982
|
+
* can be saved to disk, not to drive any state machine.
|
|
2983
|
+
*/
|
|
2984
|
+
declare function extractPrdBlock(text: string): PrdBlock | null;
|
|
2985
|
+
/** Kebab-case, path-safe feature slug. */
|
|
2986
|
+
declare function sanitizeSlug(raw: string): string;
|
|
2987
|
+
/** Save the PRD to `<workspace>/.scratch/<slug>/PRD.md` (to-prd native convention). */
|
|
2988
|
+
declare function savePrdMarkdown(workspace: string, slug: string, markdown: string): string;
|
|
2989
|
+
|
|
2990
|
+
export { ALL_PROVIDERS, AUTO_COMMANDS, AbstractRunner, AbstractTerminalSession, ActiveTaskSession, type AgentAdapter, type AgentAdapterFactory, type AgentEvent, type AgentProcessDeps, type AgentStartOptions, AiProvider, ApprovalKind, ApprovalMode, ApprovalRequest, ApprovalSource, BaseAiService, BaseConfig, BaseFileSystem, CLAUDE_CODE_MANIFEST, CLI_PROVIDERS, CMD_EXE_MAX_COMMAND_LINE, CORE_PLANNER_PROMPT, type CappedRows, type CheckpointListener, ClaudeCodeAdapter, CliAgentAiService, type CliAgentAiServiceDeps, CodexAdapter, type CommandClassification, CommandLineTooLongError, type CommandTier, ConsoleLogger, ContextCollector, ConversationMessage, type ConversationRequest, type ConversationTurn, DiscoveredModel, EmbeddedNewlineError, EnvConfig, type ExecFileFn, type ExecImpl, FindSymbolOptions, FsPluginStore, GIT_READONLY_SUBCOMMANDS, GeminiService, GlobOptions, type GrepInvocation, GrepOptions, HeadlessRunner, type HeadlessRunnerDeps, type IAiService, IApproval, IConfig, IFileSystem, type ILogger, type INotification, IPluginStore, ITerminalRunner, ITerminalSession, JSON_REPAIR_INSTRUCTION, type KillTreeDeps, type LaunchDeps, type LaunchPlan, LegacyPlanState, LineBuffer, type MappedTool, ModelResolver, type ModelResolverDeps, type ModifyPlanRequest, type NotificationAction, OPENCODE_MANIFEST, OpenAiService, OpenCodeAdapter, type OrchestratorObserver, OrchestratorOption, PROVIDER_DETECT_PRIORITY, PROVIDER_LABEL, PROVIDER_PRIORITY, PROVIDER_SHORT_LABEL, type PendingApproval, PendingApprovals, type PendingApprovalsOptions, PlanParseError, type PlanRequest, PlanState, PlanStatus, PlanStore, Planner, type PlannerReplyClassification, type PlannerRuntimeToggles, type PrdBlock, type ProbeFn, ProviderModelLists, type ProviderRegistration, QueuedMessage, REFUSED_COMMANDS, ReadFileOpts, type RepairLoopOpts, type RepairVerdict, type ResearchChat, ResearchLogEntry, ResearchProgress, type ResearchShell, type ResearchShellDeps, ResearchStep, ResearchToolType, type ResearchTurn, ResolveContext, type RunnerCatalog, RunnerId, RunnerInstallation, RunnerInvocation, type RunnerMode, RunnerModeInfo, RunnerPluginManifest, RunnerRegistry, STATE_DIR, SYMBOL_LANGUAGES, type SerializedPlan, type SerializedTask, type SerializedTaskStatus, Session, type SessionBroadcaster, type SessionData, type SessionDeps, type SessionMessage, type SessionMeta, type SessionPlanner, type SessionRuntimeSettings, SettingsService, type ShellDialect, type SpawnSpec, StdioAgentAdapter, TRUNCATED_PLAN_REPAIR_INSTRUCTION, Task, TaskOp, TaskOrchestrator, TaskSnapshot, TmuxRunner, type TmuxRunnerDeps, type ToolCall, ToolOutcome, type ToolResult, type UserSettings, Verdict, VerdictEngine, type VerdictListener, WINDOWS_MAX_COMMAND_LINE, applyHeadLimit, augmentedPath, buildConversationSystemPrompt, buildFallbackGrepArgs, buildGlobArgs, buildGrepArgs, buildMergePrompt, buildModifyDuringExecutionPrompt, buildModifyPlanPrompt, buildPlanWithResults, buildResearchPrompt, buildResearchToolsPrompt, buildRunnerInvocation, buildShellInvocation, buildSplitPrompt, buildSubagentSystemPrompt, clampThinkingEffort, classifyCommand, classifyPlannerReply, clearAugmentedPathCache, clearResearchShellCache, clearState, coerceAssignments, configuredProviders, createAiService, defaultLogger, definitionPattern, deleteSession, discoverGeminiModels, effectiveAllowlist, ensureDir, executionLogBlock, executionSummary, extractPrdBlock, filterFallbackByAnchoredInclude, filterModelsForPrompt, formatSearchOutput, generatePlanWithRepair, getLatestSession, getProviderMeta, getSettingsPath, getStateDir, grantScopeFor, hasTmux, includeGlobFor, isCliProvider, isOpenAiProvider, killTree, languageForId, listSessions, loadSession, loadSessionPlanState, loadState, mapAgentTool, modelContextBlock, modifyValidationFeedback, normalizeAgentArgs, normalizeGeminiModel, pendingEditRulesBlock, planDirectLaunch, planShellLaunch, posixShellQuote, prefixModelId, providerForRunner, reEmitPlanPrompt, reEmitTaskOpsPrompt, referencePattern, repairLoop, researchShellWarning, researchToolsPath, resolveArgs, resolveProviderFromPrefix, resolveResearchShell, resolveWithin, retargetTaskRunner, runnerAssignment, runnerForProvider, sanitizeSlug, savePrdMarkdown, saveSession, saveState, serializePlan, serializeTask, serializeTaskStatus, sessionRuntimeSettings, stateExists, stripAnsi, stripModelPrefix, taskOpsRejectedPrompt, tmuxSessionName, tmuxSocketName, tmuxWindowName, truncatedPlanReEmitPrompt, wellKnownBinDirs, windowsCommandLine, withPath, wrapWithPty };
|