@diousk/pi-subagents-fast 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +808 -0
- package/CONTRIBUTING.md +72 -0
- package/LICENSE +21 -0
- package/README.md +1034 -0
- package/SECURITY.md +95 -0
- package/dist/abortable.d.ts +12 -0
- package/dist/abortable.js +42 -0
- package/dist/agent-color.d.ts +35 -0
- package/dist/agent-color.js +123 -0
- package/dist/agent-file-toggle.d.ts +125 -0
- package/dist/agent-file-toggle.js +260 -0
- package/dist/agent-manager.d.ts +472 -0
- package/dist/agent-manager.js +1338 -0
- package/dist/agent-runner.d.ts +312 -0
- package/dist/agent-runner.js +1034 -0
- package/dist/agent-types.d.ts +119 -0
- package/dist/agent-types.js +286 -0
- package/dist/child-context.d.ts +2 -0
- package/dist/child-context.js +12 -0
- package/dist/context.d.ts +12 -0
- package/dist/context.js +56 -0
- package/dist/cross-extension-rpc.d.ts +66 -0
- package/dist/cross-extension-rpc.js +138 -0
- package/dist/custom-agents.d.ts +54 -0
- package/dist/custom-agents.js +316 -0
- package/dist/default-agents.d.ts +7 -0
- package/dist/default-agents.js +122 -0
- package/dist/enabled-models.d.ts +49 -0
- package/dist/enabled-models.js +145 -0
- package/dist/env.d.ts +6 -0
- package/dist/env.js +28 -0
- package/dist/group-join.d.ts +32 -0
- package/dist/group-join.js +116 -0
- package/dist/index.d.ts +50 -0
- package/dist/index.js +3682 -0
- package/dist/invocation-config.d.ts +107 -0
- package/dist/invocation-config.js +83 -0
- package/dist/memory.d.ts +53 -0
- package/dist/memory.js +165 -0
- package/dist/mention-clone.d.ts +87 -0
- package/dist/mention-clone.js +153 -0
- package/dist/mention.d.ts +81 -0
- package/dist/mention.js +131 -0
- package/dist/model-resolver.d.ts +36 -0
- package/dist/model-resolver.js +95 -0
- package/dist/model-scope.d.ts +49 -0
- package/dist/model-scope.js +48 -0
- package/dist/nested-tools.d.ts +55 -0
- package/dist/nested-tools.js +299 -0
- package/dist/output-file.d.ts +43 -0
- package/dist/output-file.js +142 -0
- package/dist/prompts.d.ts +55 -0
- package/dist/prompts.js +91 -0
- package/dist/schedule-store.d.ts +38 -0
- package/dist/schedule-store.js +155 -0
- package/dist/schedule.d.ts +109 -0
- package/dist/schedule.js +359 -0
- package/dist/settings.d.ts +360 -0
- package/dist/settings.js +251 -0
- package/dist/skill-loader.d.ts +24 -0
- package/dist/skill-loader.js +93 -0
- package/dist/status-note.d.ts +61 -0
- package/dist/status-note.js +85 -0
- package/dist/structured-output.d.ts +61 -0
- package/dist/structured-output.js +112 -0
- package/dist/types.d.ts +371 -0
- package/dist/types.js +5 -0
- package/dist/ui/agent-mention.d.ts +82 -0
- package/dist/ui/agent-mention.js +187 -0
- package/dist/ui/agent-widget.d.ts +219 -0
- package/dist/ui/agent-widget.js +592 -0
- package/dist/ui/conversation-viewer.d.ts +120 -0
- package/dist/ui/conversation-viewer.js +578 -0
- package/dist/ui/fleet-list.d.ts +195 -0
- package/dist/ui/fleet-list.js +471 -0
- package/dist/ui/schedule-menu.d.ts +16 -0
- package/dist/ui/schedule-menu.js +94 -0
- package/dist/ui/select-item.d.ts +27 -0
- package/dist/ui/select-item.js +34 -0
- package/dist/ui/viewer-keys.d.ts +20 -0
- package/dist/ui/viewer-keys.js +17 -0
- package/dist/ui/workflow-card.d.ts +175 -0
- package/dist/ui/workflow-card.js +332 -0
- package/dist/ui/workflow-dialog.d.ts +305 -0
- package/dist/ui/workflow-dialog.js +843 -0
- package/dist/ui/workflow-menu.d.ts +60 -0
- package/dist/ui/workflow-menu.js +147 -0
- package/dist/usage.d.ts +135 -0
- package/dist/usage.js +120 -0
- package/dist/workflow/collisions.d.ts +95 -0
- package/dist/workflow/collisions.js +88 -0
- package/dist/workflow/entry.d.ts +32 -0
- package/dist/workflow/entry.js +29 -0
- package/dist/workflow/host.d.ts +62 -0
- package/dist/workflow/host.js +362 -0
- package/dist/workflow/journal.d.ts +97 -0
- package/dist/workflow/journal.js +120 -0
- package/dist/workflow/json-schema.d.ts +51 -0
- package/dist/workflow/json-schema.js +111 -0
- package/dist/workflow/meta.d.ts +67 -0
- package/dist/workflow/meta.js +317 -0
- package/dist/workflow/progress.d.ts +224 -0
- package/dist/workflow/progress.js +361 -0
- package/dist/workflow/runtime.d.ts +334 -0
- package/dist/workflow/runtime.js +830 -0
- package/dist/workflow/saved.d.ts +90 -0
- package/dist/workflow/saved.js +203 -0
- package/dist/workflow/task.d.ts +136 -0
- package/dist/workflow/task.js +207 -0
- package/dist/workflow/tool-description.d.ts +38 -0
- package/dist/workflow/tool-description.js +199 -0
- package/dist/workflow/worker-source.d.ts +47 -0
- package/dist/workflow/worker-source.js +778 -0
- package/dist/worktree.d.ts +52 -0
- package/dist/worktree.js +164 -0
- package/dist/xml.d.ts +10 -0
- package/dist/xml.js +12 -0
- package/docs/rpc.md +183 -0
- package/docs/workflows.md +437 -0
- package/examples/agent-tool-description.md +42 -0
- package/examples/workflows/compose.js +51 -0
- package/examples/workflows/fan-out-audit.js +47 -0
- package/examples/workflows/gated-fix.js +60 -0
- package/examples/workflows/lib/count-child.js +27 -0
- package/examples/workflows/review-panel.js +63 -0
- package/examples/workflows/structured-findings.js +78 -0
- package/package.json +68 -0
- package/src/abortable.ts +43 -0
- package/src/agent-color.ts +161 -0
- package/src/agent-file-toggle.ts +270 -0
- package/src/agent-manager.ts +1581 -0
- package/src/agent-runner.ts +1286 -0
- package/src/agent-types.ts +346 -0
- package/src/child-context.ts +15 -0
- package/src/context.ts +58 -0
- package/src/cross-extension-rpc.ts +198 -0
- package/src/custom-agents.ts +333 -0
- package/src/default-agents.ts +126 -0
- package/src/enabled-models.ts +180 -0
- package/src/env.ts +33 -0
- package/src/group-join.ts +141 -0
- package/src/index.ts +3991 -0
- package/src/invocation-config.ts +155 -0
- package/src/memory.ts +179 -0
- package/src/mention-clone.ts +196 -0
- package/src/mention.ts +141 -0
- package/src/model-resolver.ts +118 -0
- package/src/model-scope.ts +70 -0
- package/src/nested-tools.ts +422 -0
- package/src/output-file.ts +155 -0
- package/src/prompts.ts +142 -0
- package/src/schedule-store.ts +153 -0
- package/src/schedule.ts +386 -0
- package/src/settings.ts +587 -0
- package/src/skill-loader.ts +102 -0
- package/src/status-note.ts +90 -0
- package/src/structured-output.ts +130 -0
- package/src/types.ts +384 -0
- package/src/ui/agent-mention.ts +216 -0
- package/src/ui/agent-widget.ts +664 -0
- package/src/ui/conversation-viewer.ts +589 -0
- package/src/ui/fleet-list.ts +543 -0
- package/src/ui/schedule-menu.ts +105 -0
- package/src/ui/select-item.ts +45 -0
- package/src/ui/viewer-keys.ts +39 -0
- package/src/ui/workflow-card.ts +470 -0
- package/src/ui/workflow-dialog.ts +1115 -0
- package/src/ui/workflow-menu.ts +193 -0
- package/src/usage.ts +167 -0
- package/src/workflow/collisions.ts +123 -0
- package/src/workflow/entry.ts +47 -0
- package/src/workflow/host.ts +403 -0
- package/src/workflow/journal.ts +164 -0
- package/src/workflow/json-schema.ts +128 -0
- package/src/workflow/meta.ts +325 -0
- package/src/workflow/progress.ts +550 -0
- package/src/workflow/runtime.ts +1219 -0
- package/src/workflow/saved.ts +217 -0
- package/src/workflow/task.ts +302 -0
- package/src/workflow/tool-description.ts +200 -0
- package/src/workflow/worker-source.ts +781 -0
- package/src/worktree.ts +205 -0
- package/src/xml.ts +13 -0
|
@@ -0,0 +1,1338 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* agent-manager.ts — Tracks agents, background execution, resume support.
|
|
3
|
+
*
|
|
4
|
+
* There are two independent concurrency pools, never one:
|
|
5
|
+
*
|
|
6
|
+
* - Background (`maxConcurrent`, default 10) bounds detached agents.
|
|
7
|
+
* - Foreground (`maxConcurrentForeground`, default 0 = unlimited) bounds
|
|
8
|
+
* agents a caller is blocking on inline — `spawnAndWait`.
|
|
9
|
+
*
|
|
10
|
+
* Independent by design: a foreground agent blocks the parent anyway, so
|
|
11
|
+
* charging it to the background pool would let a saturated pool starve the main
|
|
12
|
+
* session of work it could have done itself. Excess agents in either pool are
|
|
13
|
+
* queued and auto-started as slots free up. Nested children take no slot in
|
|
14
|
+
* either — see `occupiesPoolSlot` / `occupiesForegroundSlot`.
|
|
15
|
+
*/
|
|
16
|
+
import { randomUUID } from "node:crypto";
|
|
17
|
+
import { statSync } from "node:fs";
|
|
18
|
+
import { isAbsolute } from "node:path";
|
|
19
|
+
import { resumeAgent, runAgent } from "./agent-runner.js";
|
|
20
|
+
import { assignHandle, handleBase } from "./mention.js";
|
|
21
|
+
import { describeModel } from "./model-resolver.js";
|
|
22
|
+
import { addUsage } from "./usage.js";
|
|
23
|
+
import { cleanupWorktree, createWorktree, isWorktreeIsolationEnabled, pruneWorktrees, } from "./worktree.js";
|
|
24
|
+
/**
|
|
25
|
+
* Default max concurrent background agents.
|
|
26
|
+
*
|
|
27
|
+
* Raised from 4 when top-level spawns started defaulting to background
|
|
28
|
+
* (`backgroundByDefault`): foreground agents bypass this pool entirely, so
|
|
29
|
+
* while foreground was the default a fan-out of six ran six. With background
|
|
30
|
+
* as the default every top-level agent takes a slot, and a limit of 4 would
|
|
31
|
+
* have silently queued the tail of exactly the parallel fan-outs the `Agent`
|
|
32
|
+
* tool description tells the model to send.
|
|
33
|
+
*/
|
|
34
|
+
const DEFAULT_MAX_CONCURRENT = 10;
|
|
35
|
+
/**
|
|
36
|
+
* Default max concurrent foreground (blocking) agents — `0` = unlimited, the
|
|
37
|
+
* extension's existing convention for "no ceiling" (`defaultMaxTurns`).
|
|
38
|
+
*
|
|
39
|
+
* Off by default because nothing here ever bounded foreground work, and pi
|
|
40
|
+
* dispatches a message's tool calls through `Promise.all`, so an unqualified
|
|
41
|
+
* fan-out of blocking `Agent` calls has always run all at once. Users who want
|
|
42
|
+
* it bounded — chiefly local models, where parallel agents thrash the prompt
|
|
43
|
+
* cache (#253) — opt in; everyone else keeps today's behaviour exactly.
|
|
44
|
+
*/
|
|
45
|
+
const DEFAULT_MAX_CONCURRENT_FOREGROUND = 0;
|
|
46
|
+
/**
|
|
47
|
+
* How many evicted agents stay addressable by name. Only a bound on memory —
|
|
48
|
+
* a session that spawns hundreds of agents shouldn't retain every one — and
|
|
49
|
+
* far above the handful anyone keeps in their head.
|
|
50
|
+
*/
|
|
51
|
+
const MAX_TOMBSTONES = 100;
|
|
52
|
+
/**
|
|
53
|
+
* Validate a caller-supplied SpawnOptions.cwd. `undefined`/`null` mean "unset"
|
|
54
|
+
* (parent cwd). Anything else must be an absolute path to an existing
|
|
55
|
+
* directory — curated errors instead of TypeErrors from path/fs internals
|
|
56
|
+
* (RPC callers send arbitrary JSON: null, numbers, file paths).
|
|
57
|
+
*/
|
|
58
|
+
function assertValidSpawnCwd(cwd) {
|
|
59
|
+
if (cwd == null)
|
|
60
|
+
return;
|
|
61
|
+
if (typeof cwd !== "string" || !isAbsolute(cwd)) {
|
|
62
|
+
throw new Error(`SpawnOptions.cwd must be an absolute path: "${String(cwd)}"`);
|
|
63
|
+
}
|
|
64
|
+
let isDirectory = false;
|
|
65
|
+
try {
|
|
66
|
+
isDirectory = statSync(cwd).isDirectory();
|
|
67
|
+
}
|
|
68
|
+
catch {
|
|
69
|
+
throw new Error(`SpawnOptions.cwd does not exist: "${cwd}"`);
|
|
70
|
+
}
|
|
71
|
+
if (!isDirectory) {
|
|
72
|
+
throw new Error(`SpawnOptions.cwd is not a directory: "${cwd}"`);
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Whether a record occupies one of the `maxConcurrent` background slots.
|
|
77
|
+
* Nested children don't: their parent already holds a slot, so counting (and
|
|
78
|
+
* therefore queueing) them would deadlock a parent that waits on its own child.
|
|
79
|
+
*
|
|
80
|
+
* Note this bounds nothing horizontally — the depth cap limits how DEEP nesting
|
|
81
|
+
* goes, not how WIDE. A parent's only limit on concurrent children is that each
|
|
82
|
+
* spawn costs it a turn, which is unbounded when max turns is unlimited.
|
|
83
|
+
*/
|
|
84
|
+
function occupiesPoolSlot(record) {
|
|
85
|
+
return !!record.isBackground && isTopLevelAgent(record);
|
|
86
|
+
}
|
|
87
|
+
/**
|
|
88
|
+
* Whether a record is one of the session's own agents, rather than something
|
|
89
|
+
* another agent or a workflow owns.
|
|
90
|
+
*
|
|
91
|
+
* The single definition behind every user-facing surface — the fleet list, the
|
|
92
|
+
* widget, the `/agents` menus, `@handle` resolution, and the completion events
|
|
93
|
+
* and session entries. An owned child reports through its owner, so surfacing
|
|
94
|
+
* it separately would double-count the same work in the places a person reads.
|
|
95
|
+
*/
|
|
96
|
+
export function isTopLevelAgent(record) {
|
|
97
|
+
return record.parentAgentId === undefined && record.workflowId === undefined;
|
|
98
|
+
}
|
|
99
|
+
/**
|
|
100
|
+
* Whether a record occupies one of the `maxConcurrentForeground` slots.
|
|
101
|
+
*
|
|
102
|
+
* Keyed on `blocking` — a caller awaiting this record inline — rather than on
|
|
103
|
+
* `isBackground === false`, because `spawn()` is also the funnel for DETACHED
|
|
104
|
+
* starts (cross-extension RPC, `@handle` mentions, the registry) that may pass
|
|
105
|
+
* `isBackground: false` and are documented to run immediately regardless. Those
|
|
106
|
+
* block nobody, so bounding them buys nothing and would park a record with no
|
|
107
|
+
* one waiting to release it.
|
|
108
|
+
*
|
|
109
|
+
* Nested children are excluded for the same reason as `occupiesPoolSlot`, and
|
|
110
|
+
* more sharply: their parent is blocked *awaiting them*, so queueing a child
|
|
111
|
+
* behind its own parent is a guaranteed deadlock rather than a possible one.
|
|
112
|
+
* Enforced here rather than at the call site so no caller can reintroduce it.
|
|
113
|
+
*
|
|
114
|
+
* A workflow's children go out through `spawnAndWait` and so are `blocking`
|
|
115
|
+
* too, and are excluded on the same `isTopLevelAgent` test as the background
|
|
116
|
+
* pool: the run already caps how many of its agents run at once, and charging
|
|
117
|
+
* them here as well would let one fan-out queue behind a limit meant for the
|
|
118
|
+
* session's own work.
|
|
119
|
+
*
|
|
120
|
+
* Like the background pool this bounds width at the top level only — a parent's
|
|
121
|
+
* own fan-out is limited by nothing but its turn budget.
|
|
122
|
+
*/
|
|
123
|
+
function occupiesForegroundSlot(record) {
|
|
124
|
+
return !!record.blocking && isTopLevelAgent(record);
|
|
125
|
+
}
|
|
126
|
+
/** Best-effort ceiling on one child's shutdown handlers, so teardown can't strand a quit. */
|
|
127
|
+
const CHILD_SHUTDOWN_TIMEOUT_MS = 3_000;
|
|
128
|
+
/**
|
|
129
|
+
* Close the extension lifecycle `runAgent` opened with `bindExtensions`, then dispose.
|
|
130
|
+
*
|
|
131
|
+
* `AgentSession.dispose()` only calls `ExtensionRunner.invalidate()` — pi emits the event
|
|
132
|
+
* itself in `AgentSessionRuntime.dispose()` beforehand, and this is the one place that binds
|
|
133
|
+
* extensions onto a session without going through that path. Without the emit, everything an
|
|
134
|
+
* extension armed in `session_start` leaks once per spawn, and its next tick throws
|
|
135
|
+
* `assertActive()` from a bare timer callback — an uncaughtException that kills pi (#242).
|
|
136
|
+
*/
|
|
137
|
+
async function shutdownChildSession(session) {
|
|
138
|
+
try {
|
|
139
|
+
const runner = session?.extensionRunner;
|
|
140
|
+
// Optional all the way down: on a pi without the getter, or a stubbed session from a
|
|
141
|
+
// partial `onSessionCreated`, skip the emit — the same degrade as before this fix.
|
|
142
|
+
if (runner?.hasHandlers?.("session_shutdown")) {
|
|
143
|
+
// Raced, not awaited outright. `emit` runs every handler serially with no timeout of
|
|
144
|
+
// its own, and dispose() is reached from pi's own `session_shutdown` with the TUI
|
|
145
|
+
// already torn down — one hung handler would leave a dead terminal.
|
|
146
|
+
await Promise.race([
|
|
147
|
+
runner.emit({ type: "session_shutdown", reason: "quit" }),
|
|
148
|
+
new Promise(resolve => setTimeout(resolve, CHILD_SHUTDOWN_TIMEOUT_MS).unref()),
|
|
149
|
+
]);
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
catch { /* a partial session must degrade, not take the teardown down with it */ }
|
|
153
|
+
// Always, even on timeout: disposal is what this function ultimately exists to do.
|
|
154
|
+
try {
|
|
155
|
+
session?.dispose?.();
|
|
156
|
+
}
|
|
157
|
+
catch { /* ignore */ }
|
|
158
|
+
}
|
|
159
|
+
export class AgentManager {
|
|
160
|
+
agents = new Map();
|
|
161
|
+
cleanupInterval;
|
|
162
|
+
onComplete;
|
|
163
|
+
onStart;
|
|
164
|
+
onCompact;
|
|
165
|
+
onUsage;
|
|
166
|
+
maxConcurrent;
|
|
167
|
+
maxConcurrentForeground = DEFAULT_MAX_CONCURRENT_FOREGROUND;
|
|
168
|
+
/** Base repos worktrees were created from — so dispose() can prune them all,
|
|
169
|
+
* not just the parent repo (caller-supplied cwd can target other repos). */
|
|
170
|
+
worktreeRepos = new Set();
|
|
171
|
+
/**
|
|
172
|
+
* Startup phases, keyed by agent id. `spawn()` still returns synchronously,
|
|
173
|
+
* but an agent using worktree isolation is not running yet when it does —
|
|
174
|
+
* copying the repo is an awaited git call. This is what `awaitStartup` hands
|
|
175
|
+
* callers that must fail their tool call on a startup failure, and what
|
|
176
|
+
* `waitForAll` waits on while a record is "running" with no `promise` yet.
|
|
177
|
+
* Entries are dropped once the run is underway, and kept (rejected) after a
|
|
178
|
+
* startup failure so a late `awaitStartup` still sees it.
|
|
179
|
+
*/
|
|
180
|
+
startups = new Map();
|
|
181
|
+
/**
|
|
182
|
+
* Evicted agents that can still be reached by name, keyed by handle. Outlives
|
|
183
|
+
* the 10-minute record cleanup — that timer exists to bound memory, not to
|
|
184
|
+
* expire a conversation the user might still want — and is cleared alongside
|
|
185
|
+
* completed records on session start/switch.
|
|
186
|
+
*/
|
|
187
|
+
tombstones = new Map();
|
|
188
|
+
/**
|
|
189
|
+
* Agents waiting to start, tagged with the pool they wait on. One queue for
|
|
190
|
+
* both pools: `drainQueue` picks the earliest entry whose own pool has room,
|
|
191
|
+
* so neither can head-of-line-block the other, and every removal path
|
|
192
|
+
* (`abort`, `abortAll`, `dispose`) stays a single filter.
|
|
193
|
+
*
|
|
194
|
+
* `release` wakes a caller blocked in `spawnAndWait`, and is fired once the
|
|
195
|
+
* entry's `start` has SETTLED rather than at drain time: startup is async
|
|
196
|
+
* now, so releasing earlier would wake the caller before `record.promise`
|
|
197
|
+
* exists and it would read a still-starting agent as one that never ran.
|
|
198
|
+
* Removing an entry from this array MUST release it — a queued record has no
|
|
199
|
+
* promise to await, and pi has no tool-execution timeout to bail the caller
|
|
200
|
+
* out.
|
|
201
|
+
*/
|
|
202
|
+
queue = [];
|
|
203
|
+
/** Number of currently running background agents. */
|
|
204
|
+
runningBackground = 0;
|
|
205
|
+
/** Number of currently running foreground (blocking) agents. */
|
|
206
|
+
runningForeground = 0;
|
|
207
|
+
constructor(onComplete, maxConcurrent = DEFAULT_MAX_CONCURRENT, onStart, onCompact, onUsage) {
|
|
208
|
+
this.onComplete = onComplete;
|
|
209
|
+
this.onStart = onStart;
|
|
210
|
+
this.onCompact = onCompact;
|
|
211
|
+
this.onUsage = onUsage;
|
|
212
|
+
this.maxConcurrent = maxConcurrent;
|
|
213
|
+
// Cleanup completed agents after 10 minutes (but keep sessions for resume)
|
|
214
|
+
this.cleanupInterval = setInterval(() => this.cleanup(), 60_000);
|
|
215
|
+
this.cleanupInterval.unref();
|
|
216
|
+
}
|
|
217
|
+
/** Update the max concurrent background agents limit. */
|
|
218
|
+
setMaxConcurrent(n) {
|
|
219
|
+
this.maxConcurrent = Math.max(1, n);
|
|
220
|
+
// Start queued agents if the new limit allows
|
|
221
|
+
this.drainQueue();
|
|
222
|
+
}
|
|
223
|
+
getMaxConcurrent() {
|
|
224
|
+
return this.maxConcurrent;
|
|
225
|
+
}
|
|
226
|
+
/** Update the max concurrent foreground (blocking) agents limit. 0 = unlimited. */
|
|
227
|
+
setMaxConcurrentForeground(n) {
|
|
228
|
+
// Floor 0, not 1: unlimited is a meaningful value here and the default.
|
|
229
|
+
this.maxConcurrentForeground = Math.max(0, n);
|
|
230
|
+
// Start queued agents if the new limit allows — including everything, when
|
|
231
|
+
// the limit is cleared back to unlimited mid-run.
|
|
232
|
+
this.drainQueue();
|
|
233
|
+
}
|
|
234
|
+
getMaxConcurrentForeground() {
|
|
235
|
+
return this.maxConcurrentForeground;
|
|
236
|
+
}
|
|
237
|
+
/**
|
|
238
|
+
* Which pool a spawn is charged to, or undefined for one that is charged to
|
|
239
|
+
* neither (nested children, detached non-background spawns).
|
|
240
|
+
*
|
|
241
|
+
* Nothing here queues when the limit is unset — `poolHasRoom` reports an
|
|
242
|
+
* unlimited pool as always having room, so that alone is what keeps the
|
|
243
|
+
* default path identical. The `> 0` guard is belt and braces on top: it also
|
|
244
|
+
* keeps the counter from churning and the settle path from calling a drain
|
|
245
|
+
* that would find nothing to do. Both are unobservable, which is why no test
|
|
246
|
+
* pins them; the observable half — that the default start stays synchronous —
|
|
247
|
+
* is pinned in `test/foreground-concurrency.test.ts`.
|
|
248
|
+
*/
|
|
249
|
+
poolFor(record) {
|
|
250
|
+
if (occupiesPoolSlot(record))
|
|
251
|
+
return "background";
|
|
252
|
+
if (this.maxConcurrentForeground > 0 && occupiesForegroundSlot(record))
|
|
253
|
+
return "foreground";
|
|
254
|
+
return undefined;
|
|
255
|
+
}
|
|
256
|
+
poolHasRoom(pool) {
|
|
257
|
+
return pool === "background"
|
|
258
|
+
? this.runningBackground < this.maxConcurrent
|
|
259
|
+
: this.maxConcurrentForeground === 0 || this.runningForeground < this.maxConcurrentForeground;
|
|
260
|
+
}
|
|
261
|
+
/**
|
|
262
|
+
* Spawn an agent and return its ID immediately (for background use).
|
|
263
|
+
* If the concurrency limit is reached, the agent is queued.
|
|
264
|
+
*
|
|
265
|
+
* The id comes back synchronously, but with `isolation: "worktree"` the agent
|
|
266
|
+
* is not running yet when it does — the repo copy is an awaited git call.
|
|
267
|
+
* Callers that must fail a tool call on a startup failure await
|
|
268
|
+
* `awaitStartup(id)`; everyone else sees it on the record (status "error").
|
|
269
|
+
*/
|
|
270
|
+
spawn(pi, ctx, type, prompt, options) {
|
|
271
|
+
// Validate before the queue branch — a queued spawn should fail at the
|
|
272
|
+
// call, not minutes later at drain. Throw (not warn): programmatic callers
|
|
273
|
+
// can fix and retry; the RPC layer converts throws into error envelopes.
|
|
274
|
+
assertValidSpawnCwd(options.cwd);
|
|
275
|
+
const id = randomUUID().slice(0, 17);
|
|
276
|
+
const abortController = new AbortController();
|
|
277
|
+
const record = {
|
|
278
|
+
id,
|
|
279
|
+
type,
|
|
280
|
+
// Owned children — nested, or a workflow's — are filtered out of every
|
|
281
|
+
// top-level surface, so no handle: nothing can address them and they must
|
|
282
|
+
// not consume a name a top-level sibling could otherwise take.
|
|
283
|
+
handle: !isTopLevelAgent(options)
|
|
284
|
+
? undefined
|
|
285
|
+
// A reclaimed handle is used as-is: it belongs to the conversation this
|
|
286
|
+
// spawn is reopening, and re-deriving it would lose the numbering.
|
|
287
|
+
: options.reclaim?.handle ?? assignHandle(handleBase(type), this.takenHandles()),
|
|
288
|
+
description: options.description,
|
|
289
|
+
// Reclaimed here, or filled in below from `name` — in which case it must
|
|
290
|
+
// see the handle this record just took, since both come out of the same
|
|
291
|
+
// namespace.
|
|
292
|
+
alias: isTopLevelAgent(options) ? options.reclaim?.alias : undefined,
|
|
293
|
+
// Overwritten below when the spawn is actually queued; a foreground spawn
|
|
294
|
+
// that queues flips to "queued" there rather than being guessed at here,
|
|
295
|
+
// since the pool decision needs the finished record.
|
|
296
|
+
status: options.isBackground ? "queued" : "running",
|
|
297
|
+
toolUses: 0,
|
|
298
|
+
startedAt: Date.now(),
|
|
299
|
+
abortController,
|
|
300
|
+
lifetimeUsage: { input: 0, output: 0, cacheWrite: 0, cost: 0 },
|
|
301
|
+
compactionCount: 0,
|
|
302
|
+
// Raw tri-state (not coerced to a boolean): true = background, false =
|
|
303
|
+
// foreground (has an inline tool-result surface), undefined = caller never
|
|
304
|
+
// declared it (e.g. a cross-extension RPC spawn). The widget's background-
|
|
305
|
+
// only filter excludes only explicit `false`, so undefined agents — which
|
|
306
|
+
// have no inline surface — stay visible instead of vanishing.
|
|
307
|
+
isBackground: options.isBackground,
|
|
308
|
+
// Whether anyone is awaiting this agent is a property of the agent, not
|
|
309
|
+
// of the call that made it — and both settle paths need it long after
|
|
310
|
+
// `options` has stopped being the interesting object.
|
|
311
|
+
blocking: options.blocking,
|
|
312
|
+
invocation: options.invocation,
|
|
313
|
+
depth: options.depth ?? 1,
|
|
314
|
+
parentAgentId: options.parentAgentId,
|
|
315
|
+
workflowId: options.workflowId,
|
|
316
|
+
maxSubagentDepth: options.maxSubagentDepth,
|
|
317
|
+
rootSessionId: options.rootSessionId,
|
|
318
|
+
};
|
|
319
|
+
this.agents.set(id, record);
|
|
320
|
+
// After the insert, so `takenHandles()` already counts this record's own
|
|
321
|
+
// handle — a spawn named after its own type gets `explore-2`, not a
|
|
322
|
+
// duplicate `explore` that would make resolution ambiguous.
|
|
323
|
+
if (record.handle !== undefined && record.alias === undefined && options.name !== undefined) {
|
|
324
|
+
record.alias = assignHandle(handleBase(options.name), this.takenHandles());
|
|
325
|
+
}
|
|
326
|
+
const args = { pi, ctx, type, prompt, options };
|
|
327
|
+
const pool = this.poolFor(record);
|
|
328
|
+
if (pool !== undefined && !options.bypassQueue && !this.poolHasRoom(pool)) {
|
|
329
|
+
// Queue it — started when a running agent in the same pool completes.
|
|
330
|
+
// Idempotent for background (already "queued"); the flip that matters is
|
|
331
|
+
// a blocking foreground spawn, optimistically marked "running" above.
|
|
332
|
+
record.status = "queued";
|
|
333
|
+
// A queued record never reaches startAgent's signal wiring, so arm the
|
|
334
|
+
// parent abort here or Esc could not release the position.
|
|
335
|
+
if (!this.armQueuedAbort(id, options.signal))
|
|
336
|
+
return id;
|
|
337
|
+
let release;
|
|
338
|
+
record.startGate = new Promise(resolve => { release = resolve; });
|
|
339
|
+
this.queue.push({
|
|
340
|
+
id,
|
|
341
|
+
pool,
|
|
342
|
+
start: () => this.launch(id, record, args, pool),
|
|
343
|
+
release: () => release(),
|
|
344
|
+
});
|
|
345
|
+
options.onQueued?.(id, this.queue.filter(e => e.pool === pool).length - 1);
|
|
346
|
+
return id;
|
|
347
|
+
}
|
|
348
|
+
this.launch(id, record, args, undefined);
|
|
349
|
+
return id;
|
|
350
|
+
}
|
|
351
|
+
/**
|
|
352
|
+
* Wire a parent abort signal for a record that is about to be QUEUED.
|
|
353
|
+
* `startAgent` does this for running agents, and a queued record never gets
|
|
354
|
+
* there, so without this Esc could not release a queue position.
|
|
355
|
+
*
|
|
356
|
+
* Returns false when the signal is ALREADY aborted, in which case the record
|
|
357
|
+
* is stopped here and must not be enqueued: `addEventListener` never fires on
|
|
358
|
+
* an aborted signal, so a `spawnAndWait` on it would wait forever — pi has no
|
|
359
|
+
* tool-execution timeout to bail it out.
|
|
360
|
+
*
|
|
361
|
+
* The listener is left in place when the agent starts. `startAgent` adds its
|
|
362
|
+
* own, so both fire on a later abort, but `abort()` on an already-stopped
|
|
363
|
+
* record is a no-op — so detaching would only be tidiness, and tidiness the
|
|
364
|
+
* `abortAll`/`dispose` paths could not offer anyway.
|
|
365
|
+
*/
|
|
366
|
+
armQueuedAbort(id, signal) {
|
|
367
|
+
if (signal === undefined)
|
|
368
|
+
return true;
|
|
369
|
+
if (signal.aborted) {
|
|
370
|
+
const record = this.agents.get(id);
|
|
371
|
+
if (record) {
|
|
372
|
+
record.status = "stopped";
|
|
373
|
+
record.completedAt = Date.now();
|
|
374
|
+
}
|
|
375
|
+
return false;
|
|
376
|
+
}
|
|
377
|
+
signal.addEventListener("abort", () => this.abort(id), { once: true });
|
|
378
|
+
return true;
|
|
379
|
+
}
|
|
380
|
+
/**
|
|
381
|
+
* Kick off an agent's startup and register it under `startups`. The returned
|
|
382
|
+
* promise never rejects — the failure is delivered through `awaitStartup`,
|
|
383
|
+
* and to the record.
|
|
384
|
+
*
|
|
385
|
+
* @param queuedPool - The pool this start was QUEUED on, or undefined for an
|
|
386
|
+
* immediate start. A queue drain can be minutes after `spawn()` returned,
|
|
387
|
+
* and nobody is awaiting `awaitStartup` by then, so a failure has to live
|
|
388
|
+
* on the record as status "error" — what drainQueue did when the throw was
|
|
389
|
+
* still synchronous. An immediate start instead drops the record, exactly
|
|
390
|
+
* as the throw out of `spawn()` did: no orphan in `listAgents()`, and the
|
|
391
|
+
* handle goes back.
|
|
392
|
+
*/
|
|
393
|
+
launch(id, record, args, queuedPool) {
|
|
394
|
+
const startup = this.startAgent(id, record, args).then(() => { this.startups.delete(id); }, (err) => {
|
|
395
|
+
this.startups.delete(id);
|
|
396
|
+
if (queuedPool !== undefined) {
|
|
397
|
+
// Mirrors settleRun: an inline caller gets this failure as a throw
|
|
398
|
+
// out of spawnAndWait, so an unconsumed record would ALSO nudge the
|
|
399
|
+
// session about it — the same failure reported twice.
|
|
400
|
+
if (queuedPool === "foreground")
|
|
401
|
+
record.resultConsumed = true;
|
|
402
|
+
record.status = "error";
|
|
403
|
+
record.error = err instanceof Error ? err.message : String(err);
|
|
404
|
+
record.completedAt = Date.now();
|
|
405
|
+
this.onComplete?.(record);
|
|
406
|
+
}
|
|
407
|
+
else {
|
|
408
|
+
this.agents.delete(id);
|
|
409
|
+
}
|
|
410
|
+
// The agent never kept its slot (startAgent gives it back on failure),
|
|
411
|
+
// so anything queued behind it can go now.
|
|
412
|
+
this.drainQueue();
|
|
413
|
+
throw err;
|
|
414
|
+
});
|
|
415
|
+
this.startups.set(id, startup);
|
|
416
|
+
// Nothing is obliged to await `startups` — swallow the rejection once here
|
|
417
|
+
// so an unawaited startup can't take the process down, and hand callers
|
|
418
|
+
// (drainQueue) that swallowed promise.
|
|
419
|
+
return startup.catch(() => { });
|
|
420
|
+
}
|
|
421
|
+
/**
|
|
422
|
+
* Resolves once the agent is actually running, and rejects with the startup
|
|
423
|
+
* failure (strict worktree isolation) that `spawn()` used to throw before the
|
|
424
|
+
* repo copy became async. Resolves immediately for an agent that is already
|
|
425
|
+
* running, still queued, or unknown — so callers can await it unconditionally.
|
|
426
|
+
*
|
|
427
|
+
* Call it in the same tick as the `spawn()` it belongs to: a failed startup
|
|
428
|
+
* takes its record (and this entry) with it, exactly as the throw did.
|
|
429
|
+
*/
|
|
430
|
+
awaitStartup(id) {
|
|
431
|
+
return this.startups.get(id) ?? Promise.resolve();
|
|
432
|
+
}
|
|
433
|
+
/** Actually start an agent (called immediately or from queue drain). */
|
|
434
|
+
async startAgent(id, record, { pi, ctx, type, prompt, options }) {
|
|
435
|
+
// Re-validate a caller-supplied cwd: queued spawns can start minutes after
|
|
436
|
+
// spawn()'s check, and the directory may be gone by then (TOCTOU). Same
|
|
437
|
+
// curated errors; drainQueue parks a throw on the record as an error.
|
|
438
|
+
assertValidSpawnCwd(options.cwd);
|
|
439
|
+
// Single resolution point for the caller-supplied cwd — the worktree base
|
|
440
|
+
// repo and both cleanup calls below MUST agree on this value forever.
|
|
441
|
+
const customCwd = options.cwd ?? undefined; // null (RPC "unset") → undefined
|
|
442
|
+
const baseCwd = customCwd ?? ctx.cwd;
|
|
443
|
+
// Take the running state — and with it the concurrency slot — BEFORE the
|
|
444
|
+
// first await. Creating a worktree is an awaited git call, and drainQueue
|
|
445
|
+
// reads the pool counters synchronously in a loop: incrementing after the
|
|
446
|
+
// await would let it start every queued agent at once while the first is
|
|
447
|
+
// still copying its repo. Claiming "running" here also keeps abort() and
|
|
448
|
+
// abortAll() able to reach an agent whose worktree is still being created.
|
|
449
|
+
//
|
|
450
|
+
// The pool is resolved ONCE, here, and carried to `settleRun` below:
|
|
451
|
+
// `poolFor` reads `maxConcurrentForeground`, which the user can change from
|
|
452
|
+
// `/agents → Settings` mid-run, so recomputing it at settle time would
|
|
453
|
+
// decrement a pool this run never charged (counter underflow, limit
|
|
454
|
+
// silently lifted) or skip the decrement for one it did (leaked slot —
|
|
455
|
+
// every later blocking spawn queues forever). The two startup exits below
|
|
456
|
+
// never reach `settleRun`, so they hand the slot back themselves.
|
|
457
|
+
const pool = this.poolFor(record);
|
|
458
|
+
const releaseSlot = () => {
|
|
459
|
+
if (pool === "background")
|
|
460
|
+
this.runningBackground--;
|
|
461
|
+
else if (pool === "foreground")
|
|
462
|
+
this.runningForeground--;
|
|
463
|
+
};
|
|
464
|
+
record.status = "running";
|
|
465
|
+
record.startedAt = Date.now();
|
|
466
|
+
record.startGate = undefined;
|
|
467
|
+
if (pool === "background")
|
|
468
|
+
this.runningBackground++;
|
|
469
|
+
else if (pool === "foreground")
|
|
470
|
+
this.runningForeground++;
|
|
471
|
+
// Worktree isolation: try to create a temporary git worktree. Strict —
|
|
472
|
+
// fail loud if not possible (no silent fallback to main tree). Done BEFORE
|
|
473
|
+
// the run is kicked off so a failure doesn't leave a half-running agent.
|
|
474
|
+
// The project switch is enforced here as well as at the tool boundary
|
|
475
|
+
// because cross-extension RPC forwards its options unvalidated — a schema
|
|
476
|
+
// that omits the field can't stop a caller that never saw the schema.
|
|
477
|
+
let worktreeCwd;
|
|
478
|
+
if (options.isolation === "worktree" && isWorktreeIsolationEnabled()) {
|
|
479
|
+
const wt = await createWorktree(pi, baseCwd, id);
|
|
480
|
+
if (!wt) {
|
|
481
|
+
releaseSlot();
|
|
482
|
+
throw new Error('Cannot run with isolation: "worktree" — not a git repo, no commits yet, or `git worktree add` failed. ' +
|
|
483
|
+
'Initialize git and commit at least once, or omit `isolation`.');
|
|
484
|
+
}
|
|
485
|
+
record.worktree = wt;
|
|
486
|
+
// workPath preserves subdirectory scoping for caller-supplied cwds: a
|
|
487
|
+
// cwd deep in a monorepo maps to the same subdir inside the copy, not
|
|
488
|
+
// the copied repo's root. Plain worktree spawns keep the historical
|
|
489
|
+
// behavior (agent at the copy's root) — moving them to workPath would
|
|
490
|
+
// also move .pi config discovery when the parent session sits in a repo
|
|
491
|
+
// subdirectory, silently dropping extensions/skills.
|
|
492
|
+
worktreeCwd = customCwd !== undefined ? wt.workPath : wt.path;
|
|
493
|
+
this.worktreeRepos.add(baseCwd);
|
|
494
|
+
// No longer "running" means a stop landed while the copy was being made
|
|
495
|
+
// (abort(), abortAll()) — a window that did not exist when creation was
|
|
496
|
+
// synchronous. The record is already terminal, so launching the run would
|
|
497
|
+
// burn tokens on work nobody is waiting for: discard the fresh (and by
|
|
498
|
+
// definition unchanged) worktree instead.
|
|
499
|
+
if (record.status !== "running") {
|
|
500
|
+
releaseSlot();
|
|
501
|
+
record.worktreeResult = await cleanupWorktree(pi, baseCwd, wt, options.description);
|
|
502
|
+
this.drainQueue();
|
|
503
|
+
return;
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
this.onStart?.(record);
|
|
507
|
+
// Wire parent abort signal to stop the subagent when the parent is interrupted
|
|
508
|
+
let detachParentSignal;
|
|
509
|
+
if (options.signal) {
|
|
510
|
+
// A queued spawn can start minutes after the caller handed us its signal,
|
|
511
|
+
// by which time it may already be aborted — and `addEventListener` would
|
|
512
|
+
// never fire, leaving a child the parent can no longer reach.
|
|
513
|
+
if (options.signal.aborted)
|
|
514
|
+
this.abort(id);
|
|
515
|
+
else {
|
|
516
|
+
const onParentAbort = () => this.abort(id);
|
|
517
|
+
options.signal.addEventListener("abort", onParentAbort, { once: true });
|
|
518
|
+
detachParentSignal = () => options.signal.removeEventListener("abort", onParentAbort);
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
const detach = () => { detachParentSignal?.(); detachParentSignal = undefined; };
|
|
522
|
+
const promise = runAgent(ctx, type, prompt, {
|
|
523
|
+
pi,
|
|
524
|
+
agentId: id,
|
|
525
|
+
model: options.model,
|
|
526
|
+
maxTurns: options.maxTurns,
|
|
527
|
+
isolated: options.isolated,
|
|
528
|
+
inheritContext: options.inheritContext,
|
|
529
|
+
thinkingLevel: options.thinkingLevel,
|
|
530
|
+
structuredOutput: options.structuredOutput,
|
|
531
|
+
resumeSessionFile: options.resumeSessionFile,
|
|
532
|
+
nested: options.parentAgentId !== undefined,
|
|
533
|
+
workflow: options.workflowId !== undefined,
|
|
534
|
+
// Worktree wins for the working dir (the agent must run in the copy —
|
|
535
|
+
// which, with a custom cwd, was created from that target). Config stays
|
|
536
|
+
// with the parent project when a caller-supplied cwd is in play; it must
|
|
537
|
+
// stay undefined otherwise so plain worktree runs keep resolving config
|
|
538
|
+
// (incl. relative extension paths and memory) inside the worktree copy.
|
|
539
|
+
cwd: worktreeCwd ?? customCwd,
|
|
540
|
+
// Set iff a worktree was created (see above) — names the directory the
|
|
541
|
+
// copy came from, so the prompt can tell the agent not to work there.
|
|
542
|
+
worktreeBase: worktreeCwd ? baseCwd : undefined,
|
|
543
|
+
configCwd: options.configCwd ?? (customCwd !== undefined ? ctx.cwd : undefined),
|
|
544
|
+
signal: record.abortController.signal,
|
|
545
|
+
onToolActivity: (activity) => {
|
|
546
|
+
if (activity.type === "end")
|
|
547
|
+
record.toolUses++;
|
|
548
|
+
options.onToolActivity?.(activity);
|
|
549
|
+
},
|
|
550
|
+
onTurnEnd: options.onTurnEnd,
|
|
551
|
+
onTextDelta: options.onTextDelta,
|
|
552
|
+
onAssistantUsage: (usage) => {
|
|
553
|
+
addUsage(record.lifetimeUsage, usage);
|
|
554
|
+
this.onUsage?.(record, usage);
|
|
555
|
+
options.onAssistantUsage?.(usage);
|
|
556
|
+
},
|
|
557
|
+
onCompaction: (info) => {
|
|
558
|
+
record.compactionCount++;
|
|
559
|
+
this.onCompact?.(record, info);
|
|
560
|
+
options.onCompaction?.(info);
|
|
561
|
+
},
|
|
562
|
+
nestedRuntime: {
|
|
563
|
+
manager: this,
|
|
564
|
+
parentAgentId: id,
|
|
565
|
+
depth: record.depth ?? 1,
|
|
566
|
+
maxSubagentDepth: record.maxSubagentDepth,
|
|
567
|
+
},
|
|
568
|
+
onSessionCreated: (session) => {
|
|
569
|
+
record.session = session;
|
|
570
|
+
// Capture now, while the session object exists: after eviction this
|
|
571
|
+
// path is the only thing that can reopen the conversation, and an
|
|
572
|
+
// in-memory session reports undefined, which correctly means
|
|
573
|
+
// "nothing to come back to".
|
|
574
|
+
// Optional chaining, not defensiveness for its own sake: this is the
|
|
575
|
+
// only field read off the session at creation, so an older pi or a
|
|
576
|
+
// stubbed session must degrade to "not resumable" rather than throw
|
|
577
|
+
// and take the whole spawn down with it.
|
|
578
|
+
record.sessionFile = session.sessionManager?.getSessionFile?.();
|
|
579
|
+
// Same reason, different field: the model and thinking level are only
|
|
580
|
+
// knowable once pi has resolved its defaults and clamped the level to
|
|
581
|
+
// what the model supports. Writing them back here makes the record
|
|
582
|
+
// authoritative, so every surface reads one place instead of each
|
|
583
|
+
// re-deriving "session, else the request" for itself.
|
|
584
|
+
if (session.model) {
|
|
585
|
+
record.invocation ??= {};
|
|
586
|
+
// Read the kept request first: a caller's level survives being clamped
|
|
587
|
+
// AND, one line later, being replaced by the effective one.
|
|
588
|
+
const requested = record.invocation.requestedThinking ?? record.invocation.thinking;
|
|
589
|
+
Object.assign(record.invocation, describeModel(session.model));
|
|
590
|
+
// Guarded for the reason above: a session that reports no level keeps
|
|
591
|
+
// the request rather than losing it. Overwriting unconditionally would
|
|
592
|
+
// turn an older or stubbed session into a blank `thinking:` tag, which
|
|
593
|
+
// is worse than the stale-but-true value it replaced.
|
|
594
|
+
if (session.thinkingLevel) {
|
|
595
|
+
record.invocation.thinking = session.thinkingLevel;
|
|
596
|
+
if (requested && requested !== session.thinkingLevel) {
|
|
597
|
+
record.invocation.requestedThinking = requested;
|
|
598
|
+
}
|
|
599
|
+
}
|
|
600
|
+
}
|
|
601
|
+
// Flush any steers that arrived before the session was ready
|
|
602
|
+
if (record.pendingSteers?.length) {
|
|
603
|
+
for (const msg of record.pendingSteers) {
|
|
604
|
+
session.steer(msg).catch(() => { });
|
|
605
|
+
}
|
|
606
|
+
record.pendingSteers = undefined;
|
|
607
|
+
}
|
|
608
|
+
options.onSessionCreated?.(session);
|
|
609
|
+
},
|
|
610
|
+
})
|
|
611
|
+
.then(async ({ responseText, session, aborted, steered, failure, structuredJson, structuredRetried }) => {
|
|
612
|
+
// Don't overwrite status if externally stopped via abort()
|
|
613
|
+
if (record.status !== "stopped") {
|
|
614
|
+
// Precedence: a hard abort keeps "aborted"; then a failed final turn
|
|
615
|
+
// (provider error that pi resolved instead of rejecting, #144) is an
|
|
616
|
+
// honest "error" — not a completion with an empty or stale result.
|
|
617
|
+
if (aborted) {
|
|
618
|
+
record.status = "aborted";
|
|
619
|
+
}
|
|
620
|
+
else if (failure) {
|
|
621
|
+
record.status = "error";
|
|
622
|
+
record.error = failure;
|
|
623
|
+
}
|
|
624
|
+
else {
|
|
625
|
+
record.status = steered ? "steered" : "completed";
|
|
626
|
+
}
|
|
627
|
+
}
|
|
628
|
+
record.result = responseText;
|
|
629
|
+
// Kept beside `result`, never inside it: `result` is prose meant for a
|
|
630
|
+
// reader — it is previewed, transcribed, and appended to below — while
|
|
631
|
+
// this is a machine-readable payload one caller asked for by schema.
|
|
632
|
+
record.structuredJson = structuredJson;
|
|
633
|
+
record.structuredRetried = structuredRetried;
|
|
634
|
+
record.session = session;
|
|
635
|
+
record.completedAt ??= Date.now();
|
|
636
|
+
detach();
|
|
637
|
+
// Final flush of streaming output file
|
|
638
|
+
if (record.outputCleanup) {
|
|
639
|
+
try {
|
|
640
|
+
record.outputCleanup();
|
|
641
|
+
}
|
|
642
|
+
catch { /* ignore */ }
|
|
643
|
+
record.outputCleanup = undefined;
|
|
644
|
+
}
|
|
645
|
+
// Clean up worktree if used
|
|
646
|
+
if (record.worktree) {
|
|
647
|
+
// The one moment the child's tree still exists and the child is done
|
|
648
|
+
// writing to it. try/catch, not decoration: a hook that throws must
|
|
649
|
+
// not leave the worktree behind.
|
|
650
|
+
if (options.onBeforeWorktreeCleanup) {
|
|
651
|
+
try {
|
|
652
|
+
await options.onBeforeWorktreeCleanup(record.worktree.path);
|
|
653
|
+
}
|
|
654
|
+
catch { /* ignore — never block cleanup */ }
|
|
655
|
+
}
|
|
656
|
+
const wtResult = await cleanupWorktree(pi, baseCwd, record.worktree, options.description);
|
|
657
|
+
record.worktreeResult = wtResult;
|
|
658
|
+
if (wtResult.hasChanges && wtResult.branch) {
|
|
659
|
+
// With a caller-supplied cwd the branch lives in THAT repo, not the
|
|
660
|
+
// parent session's — say so, or the orchestrator merges in the wrong repo.
|
|
661
|
+
const repoNote = customCwd !== undefined ? ` in \`${baseCwd}\`` : "";
|
|
662
|
+
// Appended to the prose only. A structured child's caller parses
|
|
663
|
+
// `structuredJson`, which stays untouched — but `result` is also
|
|
664
|
+
// what a human reads, so the note still belongs on it.
|
|
665
|
+
record.result = (record.result ?? "") +
|
|
666
|
+
`\n\n---\nChanges saved to branch \`${wtResult.branch}\`${repoNote}. Merge with: \`git merge ${wtResult.branch}\`${customCwd !== undefined ? ` (run in \`${baseCwd}\`)` : ""}`;
|
|
667
|
+
}
|
|
668
|
+
}
|
|
669
|
+
this.abortOwnedChildren(id);
|
|
670
|
+
this.settleRun(record, true, pool);
|
|
671
|
+
return responseText;
|
|
672
|
+
})
|
|
673
|
+
.catch(async (err) => {
|
|
674
|
+
// Don't overwrite status if externally stopped via abort()
|
|
675
|
+
if (record.status !== "stopped") {
|
|
676
|
+
record.status = "error";
|
|
677
|
+
}
|
|
678
|
+
record.error = err instanceof Error ? err.message : String(err);
|
|
679
|
+
record.completedAt ??= Date.now();
|
|
680
|
+
detach();
|
|
681
|
+
// Final flush of streaming output file on error
|
|
682
|
+
if (record.outputCleanup) {
|
|
683
|
+
try {
|
|
684
|
+
record.outputCleanup();
|
|
685
|
+
}
|
|
686
|
+
catch { /* ignore */ }
|
|
687
|
+
record.outputCleanup = undefined;
|
|
688
|
+
}
|
|
689
|
+
// Best-effort worktree cleanup on error
|
|
690
|
+
if (record.worktree) {
|
|
691
|
+
try {
|
|
692
|
+
const wtResult = await cleanupWorktree(pi, baseCwd, record.worktree, options.description);
|
|
693
|
+
record.worktreeResult = wtResult;
|
|
694
|
+
}
|
|
695
|
+
catch { /* ignore cleanup errors */ }
|
|
696
|
+
}
|
|
697
|
+
this.abortOwnedChildren(id);
|
|
698
|
+
this.settleRun(record, false, pool);
|
|
699
|
+
return "";
|
|
700
|
+
});
|
|
701
|
+
record.promise = promise;
|
|
702
|
+
// Notify caller that spawn is complete (record is in the map, promise is set).
|
|
703
|
+
// Called synchronously — onSessionCreated fires asynchronously inside runAgent.
|
|
704
|
+
// Used by spawnAndWait to let the caller set up output files before streaming
|
|
705
|
+
// starts. Read off the options, so a spawn that started from a queue drain
|
|
706
|
+
// still reaches the caller that queued it.
|
|
707
|
+
options.onSpawned?.(id);
|
|
708
|
+
}
|
|
709
|
+
/**
|
|
710
|
+
* The shared tail of both settle paths: release whatever pool slot the run
|
|
711
|
+
* held, notify, and let the queue drain into the freed slot.
|
|
712
|
+
*
|
|
713
|
+
* The decrement lives HERE and nowhere else. `abort()` on a running record
|
|
714
|
+
* only fires its controller and leaves the run to settle normally, so
|
|
715
|
+
* decrementing there too would double-free — permanently lifting the limit.
|
|
716
|
+
*
|
|
717
|
+
* Foreground agents fire `onComplete` for lifecycle symmetry, with
|
|
718
|
+
* `resultConsumed` set so the callback skips notifications the inline result
|
|
719
|
+
* already delivered.
|
|
720
|
+
*
|
|
721
|
+
* @param guardCallback swallow a throwing `onComplete` (the success path does;
|
|
722
|
+
* the error path historically did not, and keeps not doing so).
|
|
723
|
+
* @param pool the pool this run was CHARGED TO at start time — passed in, not
|
|
724
|
+
* recomputed, so a mid-run change to `maxConcurrentForeground` can't make
|
|
725
|
+
* the release disagree with the acquire.
|
|
726
|
+
*/
|
|
727
|
+
settleRun(record, guardCallback, pool) {
|
|
728
|
+
if (!record.isBackground)
|
|
729
|
+
record.resultConsumed = true;
|
|
730
|
+
if (pool === "background")
|
|
731
|
+
this.runningBackground--;
|
|
732
|
+
else if (pool === "foreground")
|
|
733
|
+
this.runningForeground--;
|
|
734
|
+
if (guardCallback) {
|
|
735
|
+
try {
|
|
736
|
+
this.onComplete?.(record);
|
|
737
|
+
}
|
|
738
|
+
catch { /* ignore completion side-effect errors */ }
|
|
739
|
+
}
|
|
740
|
+
else {
|
|
741
|
+
this.onComplete?.(record);
|
|
742
|
+
}
|
|
743
|
+
// The isBackground half reproduces the pre-pool condition exactly — a
|
|
744
|
+
// background settle has always drained, even for a nested child that held
|
|
745
|
+
// no slot — so that path is unchanged whether or not the foreground pool is
|
|
746
|
+
// on. The `pool` half only adds the drain a freed FOREGROUND slot needs.
|
|
747
|
+
// A drain with nothing freed is a no-op anyway, but "no-op" is a claim
|
|
748
|
+
// about reachability, and matching the old condition needs no such claim.
|
|
749
|
+
if (record.isBackground || pool !== undefined)
|
|
750
|
+
this.drainQueue();
|
|
751
|
+
}
|
|
752
|
+
/**
|
|
753
|
+
* Stop the nested children a settled parent owns. Nested records are hidden
|
|
754
|
+
* from the UI and only their owner can consume them, so a child outliving its
|
|
755
|
+
* parent would burn tokens unseen with no way to reach it. Grandchildren are
|
|
756
|
+
* covered transitively — each abort lands in that child's own settle path.
|
|
757
|
+
*/
|
|
758
|
+
abortOwnedChildren(parentId) {
|
|
759
|
+
for (const [id, record] of this.agents) {
|
|
760
|
+
if (record.parentAgentId === parentId)
|
|
761
|
+
this.abort(id);
|
|
762
|
+
}
|
|
763
|
+
}
|
|
764
|
+
/**
|
|
765
|
+
* Start queued agents up to each pool's concurrency limit.
|
|
766
|
+
*
|
|
767
|
+
* `findIndex` on the entry's OWN pool rather than `shift`: with one queue
|
|
768
|
+
* serving two independent limits, a saturated foreground pool at the head
|
|
769
|
+
* would otherwise stall every background agent behind it. Taking the earliest
|
|
770
|
+
* eligible entry keeps FIFO within each pool, which is what callers see.
|
|
771
|
+
*/
|
|
772
|
+
drainQueue() {
|
|
773
|
+
for (;;) {
|
|
774
|
+
const i = this.queue.findIndex(e => this.poolHasRoom(e.pool));
|
|
775
|
+
if (i === -1)
|
|
776
|
+
return;
|
|
777
|
+
const [next] = this.queue.splice(i, 1);
|
|
778
|
+
const record = this.agents.get(next.id);
|
|
779
|
+
// Stale entries (aborted while queued) are not started — but are still
|
|
780
|
+
// released, since nothing else will.
|
|
781
|
+
if (!record || record.status !== "queued") {
|
|
782
|
+
next.release();
|
|
783
|
+
continue;
|
|
784
|
+
}
|
|
785
|
+
// Detached, and never rejects: a late failure (e.g. strict worktree
|
|
786
|
+
// isolation) lands on the record inside `launch`, exactly as the
|
|
787
|
+
// synchronous throw did here before, and draining continues either way.
|
|
788
|
+
//
|
|
789
|
+
// The release waits for that startup to SETTLE rather than firing here.
|
|
790
|
+
// Startup is async now, so a release at drain time would wake a blocked
|
|
791
|
+
// `spawnAndWait` while `record.promise` was still undefined, and it would
|
|
792
|
+
// read a perfectly healthy agent as one that never ran.
|
|
793
|
+
void next.start().then(() => next.release(), () => next.release());
|
|
794
|
+
}
|
|
795
|
+
}
|
|
796
|
+
/**
|
|
797
|
+
* Remove queued entries and wake anyone blocked on them. The single point
|
|
798
|
+
* that enforces "leaving the queue releases the waiter" — a missed release is
|
|
799
|
+
* an unbounded hang, not a failed call.
|
|
800
|
+
*/
|
|
801
|
+
dequeue(pred) {
|
|
802
|
+
const kept = [];
|
|
803
|
+
for (const entry of this.queue) {
|
|
804
|
+
if (pred(entry))
|
|
805
|
+
entry.release();
|
|
806
|
+
else
|
|
807
|
+
kept.push(entry);
|
|
808
|
+
}
|
|
809
|
+
this.queue = kept;
|
|
810
|
+
}
|
|
811
|
+
/**
|
|
812
|
+
* Spawn an agent and wait for completion (foreground use).
|
|
813
|
+
* Charged to the foreground pool (`maxConcurrentForeground`), which is
|
|
814
|
+
* unlimited by default; never to the background one.
|
|
815
|
+
* Returns { id, record } so callers can access the agent ID.
|
|
816
|
+
*
|
|
817
|
+
* @param onSpawned - Called synchronously once the run is kicked off, before
|
|
818
|
+
* onSessionCreated fires. Use this to set record.outputFile so
|
|
819
|
+
* streamToOutputFile can pick it up.
|
|
820
|
+
*/
|
|
821
|
+
async spawnAndWait(pi, ctx, type, prompt, options, onSpawned) {
|
|
822
|
+
// `blocking` is what maxConcurrentForeground bounds, and this is its only
|
|
823
|
+
// source. onSpawned rides on the options rather than on a field of this
|
|
824
|
+
// manager: a queued spawn starts at drain time, long after any install/
|
|
825
|
+
// restore pair around this call would have put the field back — and it now
|
|
826
|
+
// fires after an await (worktree creation) even on the immediate path.
|
|
827
|
+
const id = this.spawn(pi, ctx, type, prompt, {
|
|
828
|
+
...options,
|
|
829
|
+
isBackground: false,
|
|
830
|
+
blocking: true,
|
|
831
|
+
onSpawned,
|
|
832
|
+
});
|
|
833
|
+
const record = this.agents.get(id);
|
|
834
|
+
// Queued: nothing to await yet — the promise appears when the drain starts
|
|
835
|
+
// it. The gate resolves (never rejects) on every path out of the queue,
|
|
836
|
+
// start and abort alike, so a rejection can never escape into the caller's
|
|
837
|
+
// tool `execute` and take down pi's whole Promise.all tool batch.
|
|
838
|
+
if (record.status === "queued")
|
|
839
|
+
await record.startGate;
|
|
840
|
+
// The run promise only exists once startup is past its awaited repo copy —
|
|
841
|
+
// without this the call would return before the agent had started at all.
|
|
842
|
+
// A startup failure (strict worktree isolation) rejects here, which is what
|
|
843
|
+
// the immediate path owes its caller: pi only marks a tool result failed
|
|
844
|
+
// when `execute` throws. A queued spawn's failure landed on the record
|
|
845
|
+
// instead (nobody was awaiting `startups` at drain time) and is rethrown
|
|
846
|
+
// below, so the contract is the same either way.
|
|
847
|
+
await this.awaitStartup(id);
|
|
848
|
+
// undefined when it was aborted while queued, or stopped mid-copy, and so
|
|
849
|
+
// never ran — the record is already terminal with a completedAt, which is
|
|
850
|
+
// what the caller renders.
|
|
851
|
+
if (record.promise)
|
|
852
|
+
await record.promise;
|
|
853
|
+
// A record that ended "error" without ever getting a promise never ran: the
|
|
854
|
+
// same startup failure spawn() rethrows on the immediate path (#179). Keep
|
|
855
|
+
// one contract rather than letting queue pressure decide whether a strict
|
|
856
|
+
// worktree failure throws or returns as a result.
|
|
857
|
+
if (record.promise === undefined && record.status === "error") {
|
|
858
|
+
throw new Error(record.error ?? "Agent failed to start");
|
|
859
|
+
}
|
|
860
|
+
return { id, record };
|
|
861
|
+
}
|
|
862
|
+
/**
|
|
863
|
+
* Resume an existing agent session with a new prompt.
|
|
864
|
+
*/
|
|
865
|
+
async resume(id, prompt, signal, options) {
|
|
866
|
+
const record = this.agents.get(id);
|
|
867
|
+
if (!record?.session)
|
|
868
|
+
return undefined;
|
|
869
|
+
// Background resume: settle asynchronously and notify on completion exactly
|
|
870
|
+
// like a background spawn, returning immediately with the record still
|
|
871
|
+
// "running" — or "queued" when at the concurrency limit. Previously
|
|
872
|
+
// run_in_background was ignored on resume (the Agent tool's resume branch
|
|
873
|
+
// returned before its background branch, and resume() only ever awaited
|
|
874
|
+
// inline), so a resumed agent always blocked the caller until it finished.
|
|
875
|
+
if (options?.isBackground) {
|
|
876
|
+
// Never re-enter a run that is still in flight. Detaching means the caller
|
|
877
|
+
// gets control back while the record stays "running", so nothing stops the
|
|
878
|
+
// model from resuming the same agent again. Starting a second run would
|
|
879
|
+
// overwrite record.abortController — orphaning the live run beyond the
|
|
880
|
+
// reach of `/agents` stop and abortAll() — double-count the pool slot, and
|
|
881
|
+
// then reject from session.prompt() with "Agent is already processing",
|
|
882
|
+
// whose settle path would abort the LIVE run's children and report a
|
|
883
|
+
// failure for a run that is still going. Refuse instead, leaving the
|
|
884
|
+
// record untouched; the caller decides whether to wait or steer.
|
|
885
|
+
if (record.status === "running" || record.status === "queued")
|
|
886
|
+
return undefined;
|
|
887
|
+
record.isBackground = true;
|
|
888
|
+
record.resultConsumed = false;
|
|
889
|
+
record.result = undefined;
|
|
890
|
+
record.error = undefined;
|
|
891
|
+
record.completedAt = undefined;
|
|
892
|
+
record.status = "queued";
|
|
893
|
+
const start = () => this.startResume(id, record, prompt, signal, options);
|
|
894
|
+
if (occupiesPoolSlot(record) && !this.poolHasRoom("background")) {
|
|
895
|
+
// At the concurrency limit — queue it, drains when a slot frees. A
|
|
896
|
+
// detached resume has no inline caller, hence nothing to release. The
|
|
897
|
+
// queue is shared with spawns, whose startup is async, so entries are
|
|
898
|
+
// promise-shaped even though a resume starts synchronously; failures
|
|
899
|
+
// land on the record here, since drainQueue no longer catches.
|
|
900
|
+
this.queue.push({
|
|
901
|
+
id,
|
|
902
|
+
pool: "background",
|
|
903
|
+
start: async () => {
|
|
904
|
+
try {
|
|
905
|
+
start();
|
|
906
|
+
}
|
|
907
|
+
catch (err) {
|
|
908
|
+
record.status = "error";
|
|
909
|
+
record.error = err instanceof Error ? err.message : String(err);
|
|
910
|
+
record.completedAt = Date.now();
|
|
911
|
+
this.onComplete?.(record);
|
|
912
|
+
}
|
|
913
|
+
},
|
|
914
|
+
release: () => { },
|
|
915
|
+
});
|
|
916
|
+
}
|
|
917
|
+
else {
|
|
918
|
+
start();
|
|
919
|
+
}
|
|
920
|
+
return record;
|
|
921
|
+
}
|
|
922
|
+
// Foreground resume: run inline and return the settled record.
|
|
923
|
+
record.status = "running";
|
|
924
|
+
record.startedAt = Date.now();
|
|
925
|
+
record.completedAt = undefined;
|
|
926
|
+
record.result = undefined;
|
|
927
|
+
record.error = undefined;
|
|
928
|
+
try {
|
|
929
|
+
const { text, failure } = await resumeAgent(record.session, prompt, {
|
|
930
|
+
onToolActivity: (activity) => {
|
|
931
|
+
if (activity.type === "end")
|
|
932
|
+
record.toolUses++;
|
|
933
|
+
options?.onToolActivity?.(activity);
|
|
934
|
+
},
|
|
935
|
+
onAssistantUsage: (usage) => {
|
|
936
|
+
addUsage(record.lifetimeUsage, usage);
|
|
937
|
+
this.onUsage?.(record, usage);
|
|
938
|
+
options?.onAssistantUsage?.(usage);
|
|
939
|
+
},
|
|
940
|
+
onCompaction: (info) => {
|
|
941
|
+
record.compactionCount++;
|
|
942
|
+
this.onCompact?.(record, info);
|
|
943
|
+
options?.onCompaction?.(info);
|
|
944
|
+
},
|
|
945
|
+
signal,
|
|
946
|
+
});
|
|
947
|
+
// Same contract as the spawn path (#144): a failed final turn is an
|
|
948
|
+
// error, not a completion — but the resumed text stays available.
|
|
949
|
+
record.status = failure ? "error" : "completed";
|
|
950
|
+
if (failure)
|
|
951
|
+
record.error = failure;
|
|
952
|
+
record.result = text;
|
|
953
|
+
record.completedAt = Date.now();
|
|
954
|
+
}
|
|
955
|
+
catch (err) {
|
|
956
|
+
record.status = "error";
|
|
957
|
+
record.error = err instanceof Error ? err.message : String(err);
|
|
958
|
+
record.completedAt = Date.now();
|
|
959
|
+
}
|
|
960
|
+
// Same contract as the spawn settle paths: children spawned during the
|
|
961
|
+
// resumed turn must not outlive it — nothing else can see or reach them.
|
|
962
|
+
this.abortOwnedChildren(id);
|
|
963
|
+
return record;
|
|
964
|
+
}
|
|
965
|
+
/**
|
|
966
|
+
* Start a background resume run: detached, settling and notifying like
|
|
967
|
+
* startAgent's background path. Invoked immediately, or from drainQueue when
|
|
968
|
+
* a concurrency slot frees. The session already exists (resume reuses it), so
|
|
969
|
+
* there is no onSessionCreated to hang per-run wiring off — callers use
|
|
970
|
+
* `options.onStarted`, which fires on both the immediate and the drained path.
|
|
971
|
+
*/
|
|
972
|
+
startResume(id, record, prompt, parentSignal, options) {
|
|
973
|
+
if (!record.session)
|
|
974
|
+
return;
|
|
975
|
+
record.status = "running";
|
|
976
|
+
record.startedAt = Date.now();
|
|
977
|
+
if (occupiesPoolSlot(record))
|
|
978
|
+
this.runningBackground++;
|
|
979
|
+
this.onStart?.(record);
|
|
980
|
+
// Fresh abort controller so /agents stop and steering target THIS run rather
|
|
981
|
+
// than the previous one's settled controller.
|
|
982
|
+
const abortController = new AbortController();
|
|
983
|
+
record.abortController = abortController;
|
|
984
|
+
// Optional, and NOT what the Agent tool passes for a detached resume: a
|
|
985
|
+
// parent signal aborts on the parent's own interrupt (user Esc), which is
|
|
986
|
+
// right for a foreground run whose result the caller is awaiting, and wrong
|
|
987
|
+
// for a detached one — background spawns omit it for exactly this reason.
|
|
988
|
+
let detachParentSignal;
|
|
989
|
+
if (parentSignal) {
|
|
990
|
+
const onParentAbort = () => this.abort(id);
|
|
991
|
+
parentSignal.addEventListener("abort", onParentAbort, { once: true });
|
|
992
|
+
detachParentSignal = () => parentSignal.removeEventListener("abort", onParentAbort);
|
|
993
|
+
}
|
|
994
|
+
// Per-run side effects (output streaming) — see ResumeOptions.onStarted.
|
|
995
|
+
// After the record is in its running shape, before the run is kicked off.
|
|
996
|
+
try {
|
|
997
|
+
options.onStarted?.();
|
|
998
|
+
}
|
|
999
|
+
catch { /* ignore caller wiring errors */ }
|
|
1000
|
+
const settle = () => {
|
|
1001
|
+
detachParentSignal?.();
|
|
1002
|
+
detachParentSignal = undefined;
|
|
1003
|
+
// Final flush of streaming output file
|
|
1004
|
+
if (record.outputCleanup) {
|
|
1005
|
+
try {
|
|
1006
|
+
record.outputCleanup();
|
|
1007
|
+
}
|
|
1008
|
+
catch { /* ignore */ }
|
|
1009
|
+
record.outputCleanup = undefined;
|
|
1010
|
+
}
|
|
1011
|
+
// Children spawned during the resumed turn must not outlive it.
|
|
1012
|
+
this.abortOwnedChildren(id);
|
|
1013
|
+
if (occupiesPoolSlot(record))
|
|
1014
|
+
this.runningBackground--;
|
|
1015
|
+
try {
|
|
1016
|
+
this.onComplete?.(record);
|
|
1017
|
+
}
|
|
1018
|
+
catch { /* ignore completion side-effect errors */ }
|
|
1019
|
+
this.drainQueue();
|
|
1020
|
+
};
|
|
1021
|
+
const promise = resumeAgent(record.session, prompt, {
|
|
1022
|
+
onToolActivity: (activity) => {
|
|
1023
|
+
if (activity.type === "end")
|
|
1024
|
+
record.toolUses++;
|
|
1025
|
+
options.onToolActivity?.(activity);
|
|
1026
|
+
},
|
|
1027
|
+
onAssistantUsage: (usage) => {
|
|
1028
|
+
addUsage(record.lifetimeUsage, usage);
|
|
1029
|
+
this.onUsage?.(record, usage);
|
|
1030
|
+
options.onAssistantUsage?.(usage);
|
|
1031
|
+
},
|
|
1032
|
+
onCompaction: (info) => {
|
|
1033
|
+
record.compactionCount++;
|
|
1034
|
+
this.onCompact?.(record, info);
|
|
1035
|
+
options.onCompaction?.(info);
|
|
1036
|
+
},
|
|
1037
|
+
signal: abortController.signal,
|
|
1038
|
+
})
|
|
1039
|
+
.then(({ text, failure }) => {
|
|
1040
|
+
// Don't overwrite status if externally stopped via abort().
|
|
1041
|
+
if (record.status !== "stopped") {
|
|
1042
|
+
// Same contract as the spawn path (#144): a failed final turn is an
|
|
1043
|
+
// error, not a completion — but the resumed text stays available.
|
|
1044
|
+
record.status = failure ? "error" : "completed";
|
|
1045
|
+
if (failure)
|
|
1046
|
+
record.error = failure;
|
|
1047
|
+
}
|
|
1048
|
+
record.result = text;
|
|
1049
|
+
record.completedAt ??= Date.now();
|
|
1050
|
+
settle();
|
|
1051
|
+
return text;
|
|
1052
|
+
})
|
|
1053
|
+
.catch((err) => {
|
|
1054
|
+
if (record.status !== "stopped") {
|
|
1055
|
+
record.status = "error";
|
|
1056
|
+
record.error = err instanceof Error ? err.message : String(err);
|
|
1057
|
+
}
|
|
1058
|
+
record.completedAt ??= Date.now();
|
|
1059
|
+
settle();
|
|
1060
|
+
return "";
|
|
1061
|
+
});
|
|
1062
|
+
record.promise = promise;
|
|
1063
|
+
}
|
|
1064
|
+
/**
|
|
1065
|
+
* Send a steering message to an agent from the UI (mirrors the steer_subagent
|
|
1066
|
+
* tool). A live session delivers it now — it interrupts the agent after its
|
|
1067
|
+
* current tool execution and appears as a user message. If the session isn't
|
|
1068
|
+
* ready yet, the message is queued on `pendingSteers` and flushed when the
|
|
1069
|
+
* session is created. Returns false if the agent can't accept steering
|
|
1070
|
+
* (unknown id, or no longer running/queued).
|
|
1071
|
+
*/
|
|
1072
|
+
steer(id, message) {
|
|
1073
|
+
const record = this.agents.get(id);
|
|
1074
|
+
if (!record)
|
|
1075
|
+
return false;
|
|
1076
|
+
if (record.status !== "running" && record.status !== "queued")
|
|
1077
|
+
return false;
|
|
1078
|
+
if (record.session) {
|
|
1079
|
+
record.session.steer(message).catch(() => { });
|
|
1080
|
+
}
|
|
1081
|
+
else {
|
|
1082
|
+
if (!record.pendingSteers)
|
|
1083
|
+
record.pendingSteers = [];
|
|
1084
|
+
record.pendingSteers.push(message);
|
|
1085
|
+
}
|
|
1086
|
+
return true;
|
|
1087
|
+
}
|
|
1088
|
+
getRecord(id) {
|
|
1089
|
+
return this.agents.get(id);
|
|
1090
|
+
}
|
|
1091
|
+
/** Handles already in use, so a fresh spawn can pick an unclaimed one. */
|
|
1092
|
+
takenHandles() {
|
|
1093
|
+
const taken = new Set();
|
|
1094
|
+
for (const record of this.agents.values()) {
|
|
1095
|
+
if (record.handle)
|
|
1096
|
+
taken.add(record.handle);
|
|
1097
|
+
if (record.alias)
|
|
1098
|
+
taken.add(record.alias);
|
|
1099
|
+
}
|
|
1100
|
+
// Tombstones hold their names too: an evicted `@explore` is still
|
|
1101
|
+
// resurrectable, so a later Explore must become `explore-2` rather than
|
|
1102
|
+
// shadowing a conversation the user can still reach.
|
|
1103
|
+
for (const entry of this.tombstones.values()) {
|
|
1104
|
+
taken.add(entry.handle);
|
|
1105
|
+
if (entry.alias)
|
|
1106
|
+
taken.add(entry.alias);
|
|
1107
|
+
}
|
|
1108
|
+
return taken;
|
|
1109
|
+
}
|
|
1110
|
+
/**
|
|
1111
|
+
* Resolve an `@name` from the prompt. Matches a top-level agent's handle
|
|
1112
|
+
* case-insensitively, preferring one that can still be steered and otherwise
|
|
1113
|
+
* the most recently started (which is the one a resume should continue), then
|
|
1114
|
+
* falls back to an exact agent id so `@<agentId>` works too.
|
|
1115
|
+
*/
|
|
1116
|
+
resolveMention(name) {
|
|
1117
|
+
const wanted = name.toLowerCase();
|
|
1118
|
+
let fallback;
|
|
1119
|
+
for (const record of this.agents.values()) {
|
|
1120
|
+
if (record.parentAgentId !== undefined)
|
|
1121
|
+
continue;
|
|
1122
|
+
// Handle and alias share one namespace, so at most one agent answers a
|
|
1123
|
+
// name and it makes no difference which of the two matched.
|
|
1124
|
+
if (record.handle?.toLowerCase() !== wanted && record.alias?.toLowerCase() !== wanted)
|
|
1125
|
+
continue;
|
|
1126
|
+
if (record.status === "running" || record.status === "queued")
|
|
1127
|
+
return { kind: "live", record };
|
|
1128
|
+
if (!fallback || record.startedAt > fallback.startedAt)
|
|
1129
|
+
fallback = record;
|
|
1130
|
+
}
|
|
1131
|
+
if (fallback)
|
|
1132
|
+
return { kind: "live", record: fallback };
|
|
1133
|
+
const byId = this.agents.get(name);
|
|
1134
|
+
if (byId?.parentAgentId === undefined && byId !== undefined)
|
|
1135
|
+
return { kind: "live", record: byId };
|
|
1136
|
+
// Only once nothing live answers: a tombstone is a conversation to reopen,
|
|
1137
|
+
// and reopening one while its record still exists would fork the session.
|
|
1138
|
+
for (const entry of this.tombstones.values()) {
|
|
1139
|
+
if (entry.handle.toLowerCase() === wanted || entry.alias?.toLowerCase() === wanted || entry.id === name) {
|
|
1140
|
+
return { kind: "tombstone", entry };
|
|
1141
|
+
}
|
|
1142
|
+
}
|
|
1143
|
+
return undefined;
|
|
1144
|
+
}
|
|
1145
|
+
/**
|
|
1146
|
+
* Forget an evicted agent, by handle. For the case where its session file has
|
|
1147
|
+
* gone: the entry can then only ever fail, while still holding the name
|
|
1148
|
+
* against the type that would otherwise start a fresh agent under it.
|
|
1149
|
+
*
|
|
1150
|
+
* A *successful* resume does not drop its tombstone — the live record it
|
|
1151
|
+
* creates already wins in `resolveMention`, and overwrites the entry in place
|
|
1152
|
+
* when it is itself evicted.
|
|
1153
|
+
*/
|
|
1154
|
+
dropTombstone(handle) {
|
|
1155
|
+
this.tombstones.delete(handle);
|
|
1156
|
+
}
|
|
1157
|
+
/** Evicted agents whose conversation can still be reopened, newest first. */
|
|
1158
|
+
listTombstones() {
|
|
1159
|
+
return [...this.tombstones.values()].sort((a, b) => b.completedAt - a.completedAt);
|
|
1160
|
+
}
|
|
1161
|
+
listAgents() {
|
|
1162
|
+
return [...this.agents.values()].sort((a, b) => b.startedAt - a.startedAt);
|
|
1163
|
+
}
|
|
1164
|
+
abort(id) {
|
|
1165
|
+
const record = this.agents.get(id);
|
|
1166
|
+
if (!record)
|
|
1167
|
+
return false;
|
|
1168
|
+
// Remove from queue if queued. No decrement — the slot was never taken —
|
|
1169
|
+
// and no onComplete, matching what a queued background abort has always
|
|
1170
|
+
// done; a blocking caller learns of the stop from its own tool result.
|
|
1171
|
+
if (record.status === "queued") {
|
|
1172
|
+
this.dequeue(q => q.id === id);
|
|
1173
|
+
record.status = "stopped";
|
|
1174
|
+
record.completedAt = Date.now();
|
|
1175
|
+
return true;
|
|
1176
|
+
}
|
|
1177
|
+
if (record.status !== "running")
|
|
1178
|
+
return false;
|
|
1179
|
+
record.abortController?.abort();
|
|
1180
|
+
record.status = "stopped";
|
|
1181
|
+
record.completedAt = Date.now();
|
|
1182
|
+
return true;
|
|
1183
|
+
}
|
|
1184
|
+
/** Dispose a record's session and remove it from the map. */
|
|
1185
|
+
removeRecord(id, record) {
|
|
1186
|
+
this.tombstone(record);
|
|
1187
|
+
const session = record.session;
|
|
1188
|
+
// Detached before the shutdown starts, so the record leaves the map at once and
|
|
1189
|
+
// nothing can observe a session that is half torn down.
|
|
1190
|
+
record.session = undefined;
|
|
1191
|
+
this.agents.delete(id);
|
|
1192
|
+
// A failed startup keeps its (rejected) entry so a late awaitStartup still
|
|
1193
|
+
// sees it; drop it with the record so the map can't grow unbounded.
|
|
1194
|
+
this.startups.delete(id);
|
|
1195
|
+
// Fire-and-forget is right here and only here: this runs from the 60s cleanup timer
|
|
1196
|
+
// and from `clearCompleted()` on session boundaries, with the process staying alive,
|
|
1197
|
+
// so handlers get their full window. The quit path awaits instead — see dispose().
|
|
1198
|
+
void shutdownChildSession(session);
|
|
1199
|
+
}
|
|
1200
|
+
/**
|
|
1201
|
+
* Preserve enough of a departing record for `@handle` to reopen its
|
|
1202
|
+
* conversation later. Nothing to keep unless it has both a handle to be
|
|
1203
|
+
* addressed by and a session file to reopen — an in-memory session leaves no
|
|
1204
|
+
* transcript, so the mention would have nothing to continue from.
|
|
1205
|
+
*/
|
|
1206
|
+
tombstone(record) {
|
|
1207
|
+
if (!record.handle || !record.sessionFile)
|
|
1208
|
+
return;
|
|
1209
|
+
this.tombstones.set(record.handle, {
|
|
1210
|
+
handle: record.handle,
|
|
1211
|
+
alias: record.alias,
|
|
1212
|
+
id: record.id,
|
|
1213
|
+
type: record.type,
|
|
1214
|
+
description: record.description,
|
|
1215
|
+
sessionFile: record.sessionFile,
|
|
1216
|
+
completedAt: record.completedAt ?? Date.now(),
|
|
1217
|
+
});
|
|
1218
|
+
// Bound the memory a long session can accumulate. Oldest first, since the
|
|
1219
|
+
// agent someone still wants to reach is the one they used most recently.
|
|
1220
|
+
while (this.tombstones.size > MAX_TOMBSTONES) {
|
|
1221
|
+
const oldest = [...this.tombstones.values()].reduce((a, b) => (a.completedAt <= b.completedAt ? a : b));
|
|
1222
|
+
this.tombstones.delete(oldest.handle);
|
|
1223
|
+
}
|
|
1224
|
+
}
|
|
1225
|
+
cleanup() {
|
|
1226
|
+
const cutoff = Date.now() - 10 * 60_000;
|
|
1227
|
+
for (const [id, record] of this.agents) {
|
|
1228
|
+
if (record.status === "running" || record.status === "queued")
|
|
1229
|
+
continue;
|
|
1230
|
+
if ((record.completedAt ?? 0) >= cutoff)
|
|
1231
|
+
continue;
|
|
1232
|
+
this.removeRecord(id, record);
|
|
1233
|
+
}
|
|
1234
|
+
}
|
|
1235
|
+
/**
|
|
1236
|
+
* Remove all completed/stopped/errored records immediately.
|
|
1237
|
+
* Called on session start/switch so tasks from a prior session don't persist.
|
|
1238
|
+
* Pass skipUnconsumed=true to preserve records the LLM hasn't read yet
|
|
1239
|
+
* (resultConsumed=false) — they will be evicted by the 10-minute cleanup timer instead.
|
|
1240
|
+
*/
|
|
1241
|
+
clearCompleted(skipUnconsumed = false) {
|
|
1242
|
+
for (const [id, record] of this.agents) {
|
|
1243
|
+
if (record.status === "running" || record.status === "queued")
|
|
1244
|
+
continue;
|
|
1245
|
+
if (skipUnconsumed && !record.resultConsumed)
|
|
1246
|
+
continue;
|
|
1247
|
+
this.removeRecord(id, record);
|
|
1248
|
+
}
|
|
1249
|
+
// Unconditional: both callers are session boundaries (`session_start` and
|
|
1250
|
+
// `session_before_switch`), and `skipUnconsumed` only spares records whose
|
|
1251
|
+
// results the LLM has yet to read — it does not make the sweep partial in
|
|
1252
|
+
// the sense that matters here. A new session means new handles, or
|
|
1253
|
+
// `@explore` would silently reach an agent the user never started. Claude
|
|
1254
|
+
// Code resets its registry on `/clear` for the same reason.
|
|
1255
|
+
this.tombstones.clear();
|
|
1256
|
+
}
|
|
1257
|
+
/** Whether any agents are still running or queued. */
|
|
1258
|
+
hasRunning() {
|
|
1259
|
+
return [...this.agents.values()].some(r => r.status === "running" || r.status === "queued");
|
|
1260
|
+
}
|
|
1261
|
+
/** Abort all running and queued agents immediately. */
|
|
1262
|
+
abortAll() {
|
|
1263
|
+
let count = 0;
|
|
1264
|
+
// Clear queued agents first
|
|
1265
|
+
for (const queued of this.queue) {
|
|
1266
|
+
const record = this.agents.get(queued.id);
|
|
1267
|
+
if (record) {
|
|
1268
|
+
record.status = "stopped";
|
|
1269
|
+
record.completedAt = Date.now();
|
|
1270
|
+
count++;
|
|
1271
|
+
}
|
|
1272
|
+
}
|
|
1273
|
+
this.dequeue(() => true);
|
|
1274
|
+
// Abort running agents
|
|
1275
|
+
for (const record of this.agents.values()) {
|
|
1276
|
+
if (record.status === "running") {
|
|
1277
|
+
record.abortController?.abort();
|
|
1278
|
+
record.status = "stopped";
|
|
1279
|
+
record.completedAt = Date.now();
|
|
1280
|
+
count++;
|
|
1281
|
+
}
|
|
1282
|
+
}
|
|
1283
|
+
return count;
|
|
1284
|
+
}
|
|
1285
|
+
/** Wait for all running and queued agents to complete (including queued ones). */
|
|
1286
|
+
async waitForAll() {
|
|
1287
|
+
// Loop because drainQueue respects the concurrency limit — as running
|
|
1288
|
+
// agents finish they start queued ones, which need awaiting too.
|
|
1289
|
+
while (true) {
|
|
1290
|
+
this.drainQueue();
|
|
1291
|
+
const pending = [];
|
|
1292
|
+
for (const record of this.agents.values()) {
|
|
1293
|
+
if (record.status !== "running" && record.status !== "queued")
|
|
1294
|
+
continue;
|
|
1295
|
+
// An agent whose worktree is still being created is "running" with no
|
|
1296
|
+
// `promise` yet — without its startup the wait would return too early.
|
|
1297
|
+
const startup = this.startups.get(record.id);
|
|
1298
|
+
if (startup)
|
|
1299
|
+
pending.push(startup);
|
|
1300
|
+
if (record.promise)
|
|
1301
|
+
pending.push(record.promise);
|
|
1302
|
+
}
|
|
1303
|
+
if (pending.length === 0)
|
|
1304
|
+
break;
|
|
1305
|
+
await Promise.allSettled(pending);
|
|
1306
|
+
}
|
|
1307
|
+
}
|
|
1308
|
+
/**
|
|
1309
|
+
* @param pi - Needed to run `git worktree prune`, which is async now and so
|
|
1310
|
+
* cannot be reached through a stored spawn argument at shutdown. Omitting
|
|
1311
|
+
* it (tests, teardown of a manager that never spawned) skips the prune.
|
|
1312
|
+
*/
|
|
1313
|
+
async dispose(pi) {
|
|
1314
|
+
clearInterval(this.cleanupInterval);
|
|
1315
|
+
// Clear queue — via dequeue, so anyone blocked in spawnAndWait is woken
|
|
1316
|
+
// rather than left awaiting a gate nothing will ever resolve.
|
|
1317
|
+
this.dequeue(() => true);
|
|
1318
|
+
const sessions = [...this.agents.values()].map(record => record.session);
|
|
1319
|
+
this.agents.clear();
|
|
1320
|
+
this.startups.clear();
|
|
1321
|
+
if (pi) {
|
|
1322
|
+
// Prune any orphaned git worktrees (crash recovery). Detached: dispose runs
|
|
1323
|
+
// on the shutdown path, which cannot wait for git. Started before the awaited
|
|
1324
|
+
// shutdown below rather than after it, so the git calls have that window to
|
|
1325
|
+
// finish in instead of racing the process exit that follows.
|
|
1326
|
+
const prune = (repo) => { pruneWorktrees(pi, repo).catch(() => { }); };
|
|
1327
|
+
prune(process.cwd());
|
|
1328
|
+
// Also prune repos that caller-supplied cwds created worktrees in — a clean
|
|
1329
|
+
// exit with in-flight agents would otherwise leave stale registrations there.
|
|
1330
|
+
for (const repo of this.worktreeRepos)
|
|
1331
|
+
prune(repo);
|
|
1332
|
+
}
|
|
1333
|
+
// Awaited, unlike the eviction path: pi awaits this extension's `session_shutdown`
|
|
1334
|
+
// handler and the process exits right after it returns, so anything left unawaited
|
|
1335
|
+
// here never runs at all. Bounded — each call carries its own ceiling, concurrently.
|
|
1336
|
+
await Promise.all(sessions.map(session => shutdownChildSession(session)));
|
|
1337
|
+
}
|
|
1338
|
+
}
|