@signalridge/pi-subagents 1.5.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/README.md +7 -4
- package/package.json +7 -6
- package/src/agent-color.ts +188 -0
- package/src/agent-file-toggle.ts +8 -0
- package/src/agent-manager.ts +793 -62
- package/src/agent-runner.ts +313 -21
- package/src/agent-tiers.ts +82 -3
- package/src/agent-types.ts +3 -0
- package/src/ask-tools.ts +114 -0
- package/src/cross-extension-rpc.ts +16 -5
- package/src/custom-agents.ts +67 -2
- package/src/default-agents.ts +6 -5
- package/src/gate.ts +0 -0
- package/src/index.ts +3165 -1195
- package/src/mention-clone.ts +196 -0
- package/src/mention.ts +141 -0
- package/src/output-file.ts +23 -1
- package/src/settings.ts +159 -3
- package/src/supervisor.ts +115 -0
- package/src/types.ts +59 -2
- package/src/ui/agent-mention.ts +163 -0
- package/src/ui/conversation-viewer.ts +5 -4
- package/src/ui/fleet-list.ts +9 -7
- package/src/worktree.ts +128 -648
package/src/agent-runner.ts
CHANGED
|
@@ -21,10 +21,12 @@ import {
|
|
|
21
21
|
import type { WorkflowTier } from "@signalridge/pi-subagents-protocol";
|
|
22
22
|
import { type AgentTierResolutionSnapshot, resolveAgentTier } from "./agent-tiers.js";
|
|
23
23
|
import { BUILTIN_TOOL_NAMES, getAgentConfig, getConfig, getMemoryToolNames, getReadOnlyMemoryToolNames, getToolNamesForType } from "./agent-types.js";
|
|
24
|
+
import { createAskGate } from "./ask-tools.js";
|
|
24
25
|
import { runInChildSessionContext } from "./child-context.js";
|
|
25
26
|
import { buildParentContext, extractText } from "./context.js";
|
|
26
27
|
import { DEFAULT_AGENTS } from "./default-agents.js";
|
|
27
28
|
import { detectEnv } from "./env.js";
|
|
29
|
+
import { formatGateVerdict, type GateExec, runGate, workspaceFingerprint } from "./gate.js";
|
|
28
30
|
import {
|
|
29
31
|
INTERNAL_AGENT_CONFIG_OVERRIDE,
|
|
30
32
|
type InternalAgentConfigOverride,
|
|
@@ -36,6 +38,7 @@ import { createNestedSubagentTools, getMaxSubagentDepth, type NestedAgentManager
|
|
|
36
38
|
import { buildAgentPrompt, type PromptExtras } from "./prompts.js";
|
|
37
39
|
import { shutdownAndDisposeSession } from "./session-lifecycle.js";
|
|
38
40
|
import { preloadSkills } from "./skill-loader.js";
|
|
41
|
+
import { createSupervisorTool } from "./supervisor.js";
|
|
39
42
|
import type { SubagentType, ThinkingLevel } from "./types.js";
|
|
40
43
|
import type { WorkflowTierResolutionSnapshot } from "./workflow-tiers.js";
|
|
41
44
|
import { resolveWorkflowTier } from "./workflow-tiers.js";
|
|
@@ -246,9 +249,11 @@ export function installExtensionToolScope(
|
|
|
246
249
|
narrowing: Map<string, Set<string>>;
|
|
247
250
|
/** Opt-in nested-delegation tool names to keep active despite the EXCLUDED strip. */
|
|
248
251
|
nestedToolNames: Set<string>;
|
|
252
|
+
/** Per-call approval gate from `ask_tools:`, when the agent declares any. */
|
|
253
|
+
askGate?: (toolName: string, input: unknown) => Promise<{ block: true; reason: string } | undefined>;
|
|
249
254
|
},
|
|
250
255
|
): void {
|
|
251
|
-
const { loader, toolNames, disallowedSet, extNames, narrowing, nestedToolNames } = ctx;
|
|
256
|
+
const { loader, toolNames, disallowedSet, extNames, narrowing, nestedToolNames, askGate } = ctx;
|
|
252
257
|
|
|
253
258
|
// The names allowed right now. Mirrors the `ext:` opt-in flip: when any `ext:`
|
|
254
259
|
// selector is present, extension tools become an explicit allowlist — a loaded
|
|
@@ -300,16 +305,53 @@ export function installExtensionToolScope(
|
|
|
300
305
|
|
|
301
306
|
const priorBeforeToolCall = session.agent.beforeToolCall;
|
|
302
307
|
session.agent.beforeToolCall = async (context, signal) => {
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
+
const run = async () => {
|
|
309
|
+
if (!inScope().has(context.toolCall.name)) {
|
|
310
|
+
return {
|
|
311
|
+
block: true,
|
|
312
|
+
reason: `Tool "${context.toolCall.name}" is not available to this subagent.`,
|
|
313
|
+
} as const;
|
|
314
|
+
}
|
|
315
|
+
// Scope first, then approval: a tool this agent may not use at all is
|
|
316
|
+
// refused without troubling the user about it.
|
|
317
|
+
const gated = await askGate?.(context.toolCall.name, (context.toolCall as { input?: unknown }).input);
|
|
318
|
+
if (gated) return gated;
|
|
319
|
+
return priorBeforeToolCall?.(context, signal);
|
|
320
|
+
};
|
|
321
|
+
// Default per-tool timeout prevents a hung beforeToolCall (slow extension,
|
|
322
|
+
// stuck LLM arbitrator) from stalling the subagent forever. Approval
|
|
323
|
+
// dialogs are wrapped too — a tool that cannot be approved in time is
|
|
324
|
+
// blocked fail-closed rather than left hanging.
|
|
325
|
+
if (defaultToolTimeoutMs > 0) {
|
|
326
|
+
try {
|
|
327
|
+
return await withTimeout(run(), defaultToolTimeoutMs, `beforeToolCall for "${context.toolCall.name}"`);
|
|
328
|
+
} catch (err) {
|
|
329
|
+
return { block: true, reason: (err as Error).message };
|
|
330
|
+
}
|
|
308
331
|
}
|
|
309
|
-
return
|
|
332
|
+
return run();
|
|
310
333
|
};
|
|
311
334
|
}
|
|
312
335
|
|
|
336
|
+
/**
|
|
337
|
+
* Run an agent's `gate:` command and format its verdict for the result text.
|
|
338
|
+
*
|
|
339
|
+
* Failures here are contained: a gate that cannot be fingerprinted simply runs
|
|
340
|
+
* uncached, and one that cannot run at all reports as failed rather than taking
|
|
341
|
+
* the agent's whole result down with it.
|
|
342
|
+
*/
|
|
343
|
+
async function runConfiguredGate(command: string, cwd: string, pi: ExtensionAPI): Promise<string> {
|
|
344
|
+
const exec: GateExec = (file, args, execOptions) =>
|
|
345
|
+
pi.exec(file, args, execOptions as Parameters<ExtensionAPI["exec"]>[2]);
|
|
346
|
+
try {
|
|
347
|
+
const fingerprint = await workspaceFingerprint(cwd, exec);
|
|
348
|
+
const verdict = await runGate({ command, cwd, exec, ...(fingerprint ? { fingerprint } : {}) });
|
|
349
|
+
return formatGateVerdict(command, verdict);
|
|
350
|
+
} catch (error: unknown) {
|
|
351
|
+
return `\n\n---\nAcceptance gate \`${command}\`: could not run (${error instanceof Error ? error.message : String(error)})`;
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
|
|
313
355
|
/** Default max turns. undefined = unlimited (no turn limit). */
|
|
314
356
|
let defaultMaxTurns: number | undefined;
|
|
315
357
|
|
|
@@ -325,6 +367,48 @@ export function getDefaultMaxTurns(): number | undefined { return defaultMaxTurn
|
|
|
325
367
|
export function setDefaultMaxTurns(n: number | undefined): void { defaultMaxTurns = normalizeMaxTurns(n); }
|
|
326
368
|
|
|
327
369
|
/** Additional turns allowed after the soft limit steer message. */
|
|
370
|
+
/**
|
|
371
|
+
* Fraction of a resource budget at which the wrap-up steer is sent.
|
|
372
|
+
*
|
|
373
|
+
* Not 1.0: an agent told to produce its final answer needs allowance left to
|
|
374
|
+
* produce it, so the steer has to arrive while there is still budget to spend
|
|
375
|
+
* on the response.
|
|
376
|
+
*/
|
|
377
|
+
const SOFT_BUDGET_FRACTION = 0.8;
|
|
378
|
+
|
|
379
|
+
/** Project defaults for the per-agent resource budgets. `0` = unlimited. */
|
|
380
|
+
let defaultMaxTokens = 0;
|
|
381
|
+
let defaultMaxToolCalls = 0;
|
|
382
|
+
|
|
383
|
+
/** Token budget for one agent run, from settings. `0` disables the cap. */
|
|
384
|
+
export function getDefaultMaxTokens(): number { return defaultMaxTokens; }
|
|
385
|
+
export function setDefaultMaxTokens(n: number): void { defaultMaxTokens = Math.max(0, Math.floor(n)); }
|
|
386
|
+
|
|
387
|
+
/** Tool-call budget for one agent run, from settings. `0` disables the cap. */
|
|
388
|
+
export function getDefaultMaxToolCalls(): number { return defaultMaxToolCalls; }
|
|
389
|
+
export function setDefaultMaxToolCalls(n: number): void { defaultMaxToolCalls = Math.max(0, Math.floor(n)); }
|
|
390
|
+
|
|
391
|
+
/** Default per-tool timeout; `0` disables (no timeout). Mirrors tintinweb — no per-tool timeout by default; hung tools are reclaimed via session abort/quiescence, not a hard tool cut. Set via settings defaultToolTimeoutMs when needed. */
|
|
392
|
+
const DEFAULT_TOOL_TIMEOUT_MS = 0;
|
|
393
|
+
const TOOL_TIMEOUT_CEILING_MS = 600_000;
|
|
394
|
+
let defaultToolTimeoutMs = DEFAULT_TOOL_TIMEOUT_MS;
|
|
395
|
+
export function getDefaultToolTimeoutMs(): number { return defaultToolTimeoutMs; }
|
|
396
|
+
export function setDefaultToolTimeoutMs(n: number): void {
|
|
397
|
+
defaultToolTimeoutMs = Math.max(0, Math.min(Math.floor(n), TOOL_TIMEOUT_CEILING_MS));
|
|
398
|
+
}
|
|
399
|
+
/** Race a promise against a timeout that rejects with `label timed out after ms`. */
|
|
400
|
+
export function withTimeout<T>(promise: Promise<T>, ms: number, label: string): Promise<T> {
|
|
401
|
+
if (ms <= 0) return promise;
|
|
402
|
+
let handle: ReturnType<typeof setTimeout> | undefined;
|
|
403
|
+
const timeout = new Promise<never>((_, reject) => {
|
|
404
|
+
handle = setTimeout(() => reject(new Error(`${label} timed out after ${ms}ms`)), ms);
|
|
405
|
+
handle.unref?.();
|
|
406
|
+
});
|
|
407
|
+
return Promise.race([promise, timeout]).finally(() => {
|
|
408
|
+
if (handle) clearTimeout(handle);
|
|
409
|
+
});
|
|
410
|
+
}
|
|
411
|
+
|
|
328
412
|
let graceTurns = 5;
|
|
329
413
|
|
|
330
414
|
/** Get the grace turns value. */
|
|
@@ -431,6 +515,12 @@ export interface RunOptions {
|
|
|
431
515
|
* the same precedence and the same fail-closed errors from one place.
|
|
432
516
|
*/
|
|
433
517
|
agentTier?: string;
|
|
518
|
+
/** Optional toolset hint forwarded by managed workflow callers. */
|
|
519
|
+
toolset?: string;
|
|
520
|
+
/** Additional tool names denied by the caller, merged with agent frontmatter. */
|
|
521
|
+
excludeTools?: string[];
|
|
522
|
+
/** Named sequential-thread hint; used for stable session naming. */
|
|
523
|
+
thread?: string;
|
|
434
524
|
/** Parent thinking level used only when a tier profile omits thinking. */
|
|
435
525
|
parentThinking?: ThinkingLevel;
|
|
436
526
|
/** Override working directory (e.g. for worktree isolation). */
|
|
@@ -478,6 +568,27 @@ export interface RunOptions {
|
|
|
478
568
|
* JSON/public spawn field; the generation wizard is the only issuer.
|
|
479
569
|
*/
|
|
480
570
|
readonly [INTERNAL_AGENT_CONFIG_OVERRIDE]?: InternalAgentConfigOverride;
|
|
571
|
+
/**
|
|
572
|
+
* Reopen an existing conversation from this session file instead of starting
|
|
573
|
+
* a new one. Package-internal: the only issuer is the `@handle` mention
|
|
574
|
+
* dispatcher, replaying a path this extension itself wrote to the resumable
|
|
575
|
+
* index. A caller-supplied value would let a spawn read any session on disk,
|
|
576
|
+
* so no public spawn surface forwards it.
|
|
577
|
+
*/
|
|
578
|
+
resumeSessionFile?: string;
|
|
579
|
+
/**
|
|
580
|
+
* Project default for persisting a top-level agent's conversation to disk,
|
|
581
|
+
* from the `rememberAgents` setting. Frontmatter `persist_session:` still
|
|
582
|
+
* wins; nested agents never persist. Without this, only agents that opted in
|
|
583
|
+
* by frontmatter leave a transcript, and `@handle` has nothing to reopen.
|
|
584
|
+
*/
|
|
585
|
+
rememberAgents?: boolean;
|
|
586
|
+
/**
|
|
587
|
+
* Whether this agent may ask its human a question with `contact_supervisor`.
|
|
588
|
+
* Defaults to on wherever there is a UI to ask through; `false` withholds the
|
|
589
|
+
* tool entirely rather than injecting one that always refuses.
|
|
590
|
+
*/
|
|
591
|
+
supervisorQuestions?: boolean;
|
|
481
592
|
/** Runtime bridge for opt-in child-safe nested delegation. */
|
|
482
593
|
nestedRuntime?: {
|
|
483
594
|
manager: NestedAgentManager;
|
|
@@ -951,9 +1062,10 @@ export async function runAgent(
|
|
|
951
1062
|
if (scopeVerdict.kind === "error") throw new Error(scopeVerdict.message);
|
|
952
1063
|
if (scopeVerdict.kind === "warn" && ctx.hasUI) ctx.ui.notify(scopeVerdict.message, "warning");
|
|
953
1064
|
}
|
|
954
|
-
const disallowedSet =
|
|
955
|
-
|
|
956
|
-
: undefined;
|
|
1065
|
+
const disallowedSet = (() => {
|
|
1066
|
+
const names = [...(agentConfig?.disallowedTools ?? []), ...(options.excludeTools ?? [])];
|
|
1067
|
+
return names.length > 0 ? new Set(names) : undefined;
|
|
1068
|
+
})();
|
|
957
1069
|
|
|
958
1070
|
// Nested delegation tools (opt-in, ownership-scoped). Empty unless the agent
|
|
959
1071
|
// set `allowed_subagents` and a nestedRuntime was provided — and never when
|
|
@@ -978,7 +1090,26 @@ export async function runAgent(
|
|
|
978
1090
|
configCwd,
|
|
979
1091
|
})
|
|
980
1092
|
: [];
|
|
981
|
-
|
|
1093
|
+
// `contact_supervisor` is injected separately from the nested tools and under
|
|
1094
|
+
// a different condition. Nesting is gated on `allowedSubagents`, but the agent
|
|
1095
|
+
// that most needs to ask a question is a LEAF one that cannot delegate — so
|
|
1096
|
+
// gating them together would withhold it from exactly those agents. It needs
|
|
1097
|
+
// only a human to answer, which `hasUI` decides. Isolation still suppresses
|
|
1098
|
+
// it: an `isolated: true` agent is defined as built-in tools only.
|
|
1099
|
+
const supervisorTools =
|
|
1100
|
+
ctx.hasUI && !isolated && options.supervisorQuestions !== false
|
|
1101
|
+
? createSupervisorTool({
|
|
1102
|
+
agentLabel: agentConfig?.displayName ?? type,
|
|
1103
|
+
ask: {
|
|
1104
|
+
input: (title, placeholder) => ctx.ui.input(title, placeholder),
|
|
1105
|
+
select: (title, choices) => ctx.ui.select(title, choices),
|
|
1106
|
+
},
|
|
1107
|
+
})
|
|
1108
|
+
: [];
|
|
1109
|
+
// One list from here on: both sets are custom tools this package injects, and
|
|
1110
|
+
// the scoping pass below keeps exactly the names it is given.
|
|
1111
|
+
const injectedTools = [...nestedTools, ...supervisorTools];
|
|
1112
|
+
const nestedToolNames = new Set(injectedTools.map(tool => tool.name));
|
|
982
1113
|
|
|
983
1114
|
// ─── Tool scoping ───────────────────────────────────────────────────────
|
|
984
1115
|
//
|
|
@@ -1035,17 +1166,33 @@ export async function runAgent(
|
|
|
1035
1166
|
for (const name of disallowedSet) denyTools.add(name);
|
|
1036
1167
|
}
|
|
1037
1168
|
sessionExcludeTools = [...denyTools];
|
|
1169
|
+
// Named toolsets are advisory labels; they never widen the configured
|
|
1170
|
+
// allowlist. Concrete tool availability remains owned by the agent config.
|
|
1038
1171
|
}
|
|
1039
1172
|
|
|
1040
1173
|
const settingsManager = SettingsManager.create(configCwd, agentDir);
|
|
1041
1174
|
const configuredSessionDir = resolveConfiguredSessionDir(agentConfig?.sessionDir, effectiveCwd);
|
|
1042
1175
|
const defaultSessionDir = process.env.PI_CODING_AGENT_SESSION_DIR ?? settingsManager.getSessionDir?.();
|
|
1043
|
-
|
|
1044
|
-
|
|
1045
|
-
|
|
1046
|
-
|
|
1047
|
-
|
|
1048
|
-
|
|
1176
|
+
// Frontmatter wins when it says anything; otherwise the project default,
|
|
1177
|
+
// which `rememberAgents` supplies for top-level agents only. A nested agent
|
|
1178
|
+
// is an implementation detail of its parent and is never addressable by
|
|
1179
|
+
// `@handle`, so it has nothing to gain from a transcript on disk.
|
|
1180
|
+
const persistSession = agentConfig?.persistSession ?? (options.nestedRuntime ? false : options.rememberAgents === true);
|
|
1181
|
+
// Optional metadata — it only nests the subagent under its spawner in
|
|
1182
|
+
// `/resume`. Now that `rememberAgents` persists every top-level spawn, a
|
|
1183
|
+
// context without a session manager (a bare programmatic ctx) must still
|
|
1184
|
+
// persist rather than take the whole spawn down.
|
|
1185
|
+
const parentSession = persistSession ? ctx.sessionManager?.getSessionFile?.() : undefined;
|
|
1186
|
+
const sessionManager = options.resumeSessionFile
|
|
1187
|
+
? // Reopening an existing conversation: the file already carries its own
|
|
1188
|
+
// header (cwd, parent) and history, so none of the create-time options
|
|
1189
|
+
// apply. `sessionDir` still matters for a later /new or /branch off it.
|
|
1190
|
+
SessionManager.open(options.resumeSessionFile, configuredSessionDir ?? defaultSessionDir)
|
|
1191
|
+
: persistSession
|
|
1192
|
+
? parentSession
|
|
1193
|
+
? SessionManager.create(effectiveCwd, configuredSessionDir ?? defaultSessionDir, { parentSession })
|
|
1194
|
+
: SessionManager.create(effectiveCwd, configuredSessionDir ?? defaultSessionDir)
|
|
1195
|
+
: SessionManager.inMemory(effectiveCwd);
|
|
1049
1196
|
|
|
1050
1197
|
// Pi 0.80.8 replaced createAgentSession's modelRegistry option with
|
|
1051
1198
|
// modelRuntime, but ExtensionContext still exposes only the registry facade.
|
|
@@ -1063,7 +1210,7 @@ export async function runAgent(
|
|
|
1063
1210
|
...(parentModelRuntime !== undefined && { modelRuntime: parentModelRuntime }),
|
|
1064
1211
|
model,
|
|
1065
1212
|
tools: sessionTools,
|
|
1066
|
-
customTools:
|
|
1213
|
+
customTools: injectedTools,
|
|
1067
1214
|
resourceLoader: loader,
|
|
1068
1215
|
};
|
|
1069
1216
|
if (sessionExcludeTools) {
|
|
@@ -1083,9 +1230,9 @@ export async function runAgent(
|
|
|
1083
1230
|
try {
|
|
1084
1231
|
throwIfAborted(options.signal);
|
|
1085
1232
|
|
|
1086
|
-
const baseSessionName = agentConfig?.name ?? type;
|
|
1233
|
+
const baseSessionName = options.thread ? `workflow-thread:${options.thread}` : (agentConfig?.name ?? type);
|
|
1087
1234
|
session.setSessionName(
|
|
1088
|
-
options.agentId ? `${baseSessionName}#${options.agentId.slice(0, 8)}` : baseSessionName,
|
|
1235
|
+
options.agentId && !options.thread ? `${baseSessionName}#${options.agentId.slice(0, 8)}` : baseSessionName,
|
|
1089
1236
|
);
|
|
1090
1237
|
|
|
1091
1238
|
// Bind extensions so that session_start fires and extensions can initialize
|
|
@@ -1108,6 +1255,15 @@ export async function runAgent(
|
|
|
1108
1255
|
// (we can't deny the name of a tool that hasn't registered yet). Both are
|
|
1109
1256
|
// handled below by re-deriving scope from the loader's live extension maps —
|
|
1110
1257
|
// `registerTool` writes into those same maps, so late arrivals are judged too.
|
|
1258
|
+
// `ask_tools:` gates individual CALLS, which is orthogonal to which tools
|
|
1259
|
+
// exist — so it applies to isolated agents too, where the scope installer
|
|
1260
|
+
// below never runs because the registry is already statically allowlisted.
|
|
1261
|
+
const askGate = createAskGate({
|
|
1262
|
+
askTools: agentConfig?.askTools ?? [],
|
|
1263
|
+
agentLabel: agentConfig?.displayName ?? type,
|
|
1264
|
+
...(ctx.hasUI ? { confirm: (title: string, message: string) => ctx.ui.confirm(title, message) } : {}),
|
|
1265
|
+
});
|
|
1266
|
+
|
|
1111
1267
|
if (!noExtensions) {
|
|
1112
1268
|
installExtensionToolScope(session, {
|
|
1113
1269
|
loader,
|
|
@@ -1116,7 +1272,26 @@ export async function runAgent(
|
|
|
1116
1272
|
extNames,
|
|
1117
1273
|
narrowing,
|
|
1118
1274
|
nestedToolNames,
|
|
1275
|
+
...(askGate ? { askGate } : {}),
|
|
1119
1276
|
});
|
|
1277
|
+
} else if (askGate) {
|
|
1278
|
+
// Same hook, without the scope check the allowlist already performed.
|
|
1279
|
+
const priorBeforeToolCall = session.agent.beforeToolCall;
|
|
1280
|
+
session.agent.beforeToolCall = async (context, signal) => {
|
|
1281
|
+
const run = async () => {
|
|
1282
|
+
const gated = await askGate(context.toolCall.name, (context.toolCall as { input?: unknown }).input);
|
|
1283
|
+
if (gated) return gated;
|
|
1284
|
+
return priorBeforeToolCall?.(context, signal);
|
|
1285
|
+
};
|
|
1286
|
+
if (defaultToolTimeoutMs > 0) {
|
|
1287
|
+
try {
|
|
1288
|
+
return await withTimeout(run(), defaultToolTimeoutMs, `beforeToolCall for "${context.toolCall.name}"`);
|
|
1289
|
+
} catch (err) {
|
|
1290
|
+
return { block: true, reason: (err as Error).message };
|
|
1291
|
+
}
|
|
1292
|
+
}
|
|
1293
|
+
return run();
|
|
1294
|
+
};
|
|
1120
1295
|
}
|
|
1121
1296
|
|
|
1122
1297
|
if (options.onSessionCreated) {
|
|
@@ -1133,7 +1308,76 @@ export async function runAgent(
|
|
|
1133
1308
|
let softLimitReached = false;
|
|
1134
1309
|
let aborted = false;
|
|
1135
1310
|
|
|
1311
|
+
// Resource budgets, in the same shape as max_turns: a wrap-up steer at the
|
|
1312
|
+
// soft threshold, an abort at the hard one. They bound what a single agent
|
|
1313
|
+
// can spend on its own, which turn count does not — one turn can burn an
|
|
1314
|
+
// arbitrary number of tokens or tool calls.
|
|
1315
|
+
//
|
|
1316
|
+
// `0` means unlimited here, matching this package's existing convention for
|
|
1317
|
+
// `maxTurns`. That is the opposite of `maxSubagentSpawnsPerBranch`, and
|
|
1318
|
+
// deliberately: these are opt-in resource caps that ship off, while the
|
|
1319
|
+
// branch spawn budget is a safety valve that ships on.
|
|
1320
|
+
const tokenBudget = normalizeMaxTurns(agentConfig?.maxTokens ?? defaultMaxTokens);
|
|
1321
|
+
const toolCallBudget = normalizeMaxTurns(agentConfig?.maxToolCalls ?? defaultMaxToolCalls);
|
|
1322
|
+
let budgetTokens = 0;
|
|
1323
|
+
let budgetToolCalls = 0;
|
|
1324
|
+
let tokenSoftReached = false;
|
|
1325
|
+
let toolSoftReached = false;
|
|
1326
|
+
|
|
1327
|
+
/**
|
|
1328
|
+
* Apply one budget, returning the next soft-limit state.
|
|
1329
|
+
*
|
|
1330
|
+
* The soft threshold is `SOFT_BUDGET_FRACTION` of the budget so the agent has
|
|
1331
|
+
* room left to actually write its answer after being told to wrap up — a
|
|
1332
|
+
* steer sent at 100% would be asking for a final response with no allowance
|
|
1333
|
+
* to produce it.
|
|
1334
|
+
*/
|
|
1335
|
+
const applyBudget = (used: number, budget: number | undefined, softReached: boolean, label: string): boolean => {
|
|
1336
|
+
if (budget == null) return softReached;
|
|
1337
|
+
// The hard limit is checked FIRST. One message can consume more than the
|
|
1338
|
+
// whole budget, and testing the soft threshold first would answer that with
|
|
1339
|
+
// a wrap-up steer and no abort — leaving an agent already over budget to
|
|
1340
|
+
// run on until its next event.
|
|
1341
|
+
if (used >= budget) {
|
|
1342
|
+
aborted = true;
|
|
1343
|
+
session.abort();
|
|
1344
|
+
return true;
|
|
1345
|
+
}
|
|
1346
|
+
if (!softReached && used >= budget * SOFT_BUDGET_FRACTION) {
|
|
1347
|
+
session.steer(
|
|
1348
|
+
`You are near your ${label} budget for this task. Wrap up immediately — provide your final answer now.`,
|
|
1349
|
+
);
|
|
1350
|
+
return true;
|
|
1351
|
+
}
|
|
1352
|
+
return softReached;
|
|
1353
|
+
};
|
|
1354
|
+
|
|
1136
1355
|
let currentMessageText = "";
|
|
1356
|
+
// Per-tool timeout: a hung bash/MCP call must not stall the subagent forever.
|
|
1357
|
+
// Each tool_execution_start arms a timer; the matching end clears it. On
|
|
1358
|
+
// timeout the session is aborted — which propagates via the tool's
|
|
1359
|
+
// AbortSignal into the hanging execute() and surfaces as an error result.
|
|
1360
|
+
// `contact_supervisor` is excluded: it intentionally waits for a human.
|
|
1361
|
+
const pendingToolTimeouts = new Map<string, ReturnType<typeof setTimeout>>();
|
|
1362
|
+
const clearToolTimeout = (toolCallId: string): void => {
|
|
1363
|
+
const handle = pendingToolTimeouts.get(toolCallId);
|
|
1364
|
+
if (handle) {
|
|
1365
|
+
clearTimeout(handle);
|
|
1366
|
+
pendingToolTimeouts.delete(toolCallId);
|
|
1367
|
+
}
|
|
1368
|
+
};
|
|
1369
|
+
const armToolTimeout = (toolCallId: string, toolName: string): void => {
|
|
1370
|
+
if (defaultToolTimeoutMs <= 0) return;
|
|
1371
|
+
if (toolName === "contact_supervisor") return;
|
|
1372
|
+
const handle = setTimeout(() => {
|
|
1373
|
+
pendingToolTimeouts.delete(toolCallId);
|
|
1374
|
+
try {
|
|
1375
|
+
session.abort();
|
|
1376
|
+
} catch {}
|
|
1377
|
+
}, defaultToolTimeoutMs);
|
|
1378
|
+
handle.unref?.();
|
|
1379
|
+
pendingToolTimeouts.set(toolCallId, handle);
|
|
1380
|
+
};
|
|
1137
1381
|
const unsubTurns = session.subscribe((event: AgentSessionEvent) => {
|
|
1138
1382
|
if (event.type === "turn_end") {
|
|
1139
1383
|
turnCount++;
|
|
@@ -1157,9 +1401,13 @@ export async function runAgent(
|
|
|
1157
1401
|
}
|
|
1158
1402
|
if (event.type === "tool_execution_start") {
|
|
1159
1403
|
options.onToolActivity?.({ type: "start", toolName: event.toolName });
|
|
1404
|
+
armToolTimeout((event as { toolCallId: string }).toolCallId, event.toolName);
|
|
1160
1405
|
}
|
|
1161
1406
|
if (event.type === "tool_execution_end") {
|
|
1407
|
+
clearToolTimeout((event as { toolCallId: string }).toolCallId);
|
|
1162
1408
|
options.onToolActivity?.({ type: "end", toolName: event.toolName });
|
|
1409
|
+
budgetToolCalls++;
|
|
1410
|
+
toolSoftReached = applyBudget(budgetToolCalls, toolCallBudget, toolSoftReached, "tool call");
|
|
1163
1411
|
}
|
|
1164
1412
|
if (event.type === "message_end" && event.message.role === "assistant") {
|
|
1165
1413
|
const u = (event.message as any).usage;
|
|
@@ -1168,6 +1416,12 @@ export async function runAgent(
|
|
|
1168
1416
|
output: u.output ?? 0,
|
|
1169
1417
|
cacheWrite: u.cacheWrite ?? 0,
|
|
1170
1418
|
});
|
|
1419
|
+
if (u) {
|
|
1420
|
+
// Same total as `getLifetimeTotal`: input + output + cacheWrite, with
|
|
1421
|
+
// cacheRead deliberately excluded.
|
|
1422
|
+
budgetTokens += (u.input ?? 0) + (u.output ?? 0) + (u.cacheWrite ?? 0);
|
|
1423
|
+
tokenSoftReached = applyBudget(budgetTokens, tokenBudget, tokenSoftReached, "token");
|
|
1424
|
+
}
|
|
1171
1425
|
}
|
|
1172
1426
|
if (event.type === "compaction_end" && !event.aborted && event.result) {
|
|
1173
1427
|
options.onCompaction?.({ reason: event.reason, tokensBefore: event.result.tokensBefore });
|
|
@@ -1195,9 +1449,18 @@ export async function runAgent(
|
|
|
1195
1449
|
} finally {
|
|
1196
1450
|
unsubTurns();
|
|
1197
1451
|
collector.unsubscribe();
|
|
1452
|
+
for (const handle of pendingToolTimeouts.values()) clearTimeout(handle);
|
|
1453
|
+
pendingToolTimeouts.clear();
|
|
1198
1454
|
}
|
|
1199
1455
|
|
|
1200
|
-
const
|
|
1456
|
+
const baseText = collector.getText().trim() || getLastAssistantText(session, startLen);
|
|
1457
|
+
// The acceptance gate runs AFTER the agent is done and its verdict is
|
|
1458
|
+
// appended to the result, so the parent reads the check and the claim it is
|
|
1459
|
+
// checking side by side. It deliberately does not steer the agent to fix
|
|
1460
|
+
// what failed: the gate is evidence for the caller, not another turn.
|
|
1461
|
+
const responseText = agentConfig?.gate
|
|
1462
|
+
? `${baseText}${await runConfiguredGate(agentConfig.gate, effectiveCwd, options.pi)}`
|
|
1463
|
+
: baseText;
|
|
1201
1464
|
completed = true;
|
|
1202
1465
|
return { responseText, session, aborted, steered: softLimitReached, failure: finalTurnError(session, startLen) };
|
|
1203
1466
|
} finally {
|
|
@@ -1253,6 +1516,32 @@ export async function resumeAgent(
|
|
|
1253
1516
|
}
|
|
1254
1517
|
})
|
|
1255
1518
|
: () => {};
|
|
1519
|
+
// Per-tool timeout for resume as well — same stuck-command fix as runAgent.
|
|
1520
|
+
const resumePendingTimeouts = new Map<string, ReturnType<typeof setTimeout>>();
|
|
1521
|
+
const unsubResumeTimeout =
|
|
1522
|
+
defaultToolTimeoutMs > 0
|
|
1523
|
+
? session.subscribe((event: AgentSessionEvent) => {
|
|
1524
|
+
if (event.type === "tool_execution_start") {
|
|
1525
|
+
const id = (event as { toolCallId: string }).toolCallId;
|
|
1526
|
+
if (event.toolName === "contact_supervisor") return;
|
|
1527
|
+
const handle = setTimeout(() => {
|
|
1528
|
+
resumePendingTimeouts.delete(id);
|
|
1529
|
+
try {
|
|
1530
|
+
session.abort();
|
|
1531
|
+
} catch {}
|
|
1532
|
+
}, defaultToolTimeoutMs);
|
|
1533
|
+
handle.unref?.();
|
|
1534
|
+
resumePendingTimeouts.set(id, handle);
|
|
1535
|
+
} else if (event.type === "tool_execution_end") {
|
|
1536
|
+
const id = (event as { toolCallId: string }).toolCallId;
|
|
1537
|
+
const handle = resumePendingTimeouts.get(id);
|
|
1538
|
+
if (handle) {
|
|
1539
|
+
clearTimeout(handle);
|
|
1540
|
+
resumePendingTimeouts.delete(id);
|
|
1541
|
+
}
|
|
1542
|
+
}
|
|
1543
|
+
})
|
|
1544
|
+
: () => {};
|
|
1256
1545
|
|
|
1257
1546
|
try {
|
|
1258
1547
|
throwIfAborted(options.signal);
|
|
@@ -1260,6 +1549,9 @@ export async function resumeAgent(
|
|
|
1260
1549
|
} finally {
|
|
1261
1550
|
collector.unsubscribe();
|
|
1262
1551
|
unsubEvents();
|
|
1552
|
+
unsubResumeTimeout();
|
|
1553
|
+
for (const handle of resumePendingTimeouts.values()) clearTimeout(handle);
|
|
1554
|
+
resumePendingTimeouts.clear();
|
|
1263
1555
|
cleanupAbort();
|
|
1264
1556
|
}
|
|
1265
1557
|
|
package/src/agent-tiers.ts
CHANGED
|
@@ -36,14 +36,82 @@ function effectiveModelId(model: Model<Api> | undefined): string | undefined {
|
|
|
36
36
|
/** Longest accepted tier key. Long enough for any real name, short enough to render. */
|
|
37
37
|
export const MAX_AGENT_TIER_KEY_LENGTH = 64;
|
|
38
38
|
|
|
39
|
-
|
|
39
|
+
/**
|
|
40
|
+
* The tier every fresh install gets: a cheap, low-thinking tier named `fast`.
|
|
41
|
+
*
|
|
42
|
+
* Explore — the highest-frequency built-in spawn — points its `tier:` at it, so
|
|
43
|
+
* read-only search does not silently inherit the parent session's most
|
|
44
|
+
* expensive model on a machine that never configured `agentTiers`. This is the
|
|
45
|
+
* tier strategy, not a per-agent vendor pin: the shipped profile is
|
|
46
|
+
* provider-neutral (`inherit` model, low thinking), and any user who defines
|
|
47
|
+
* `fast` in `subagents.json` replaces it wholesale.
|
|
48
|
+
*
|
|
49
|
+
* It is not shown as an available tier until settings are loaded; the merge in
|
|
50
|
+
* `setAgentTiersSettings` is where a worktree that explicitly disables/renames
|
|
51
|
+
* `fast` can win.
|
|
52
|
+
*/
|
|
53
|
+
const SHIPPED_FAST_PROFILE: AgentTierProfile = {
|
|
54
|
+
model: "inherit",
|
|
55
|
+
thinking: "low",
|
|
56
|
+
description: "Fast, low-cost tier for cheap read-only work (shipped default)",
|
|
57
|
+
};
|
|
58
|
+
|
|
59
|
+
export const SHIPPED_AGENT_TIER_PROFILES: Readonly<Record<string, AgentTierProfile>> = {
|
|
60
|
+
fast: SHIPPED_FAST_PROFILE,
|
|
61
|
+
};
|
|
40
62
|
|
|
63
|
+
let agentTiersSettings: AgentTiersSettings = {}; // effective view (shipped tiers merged)
|
|
64
|
+
let agentTiersConfigured: AgentTiersSettings = {}; // exactly what the user configured
|
|
65
|
+
|
|
66
|
+
/** Effective catalogue: shipped tiers merged under any user configuration. */
|
|
41
67
|
export function getAgentTiersSettings(): AgentTiersSettings {
|
|
42
68
|
return structuredClone(agentTiersSettings);
|
|
43
69
|
}
|
|
44
70
|
|
|
71
|
+
/** The raw user configuration, without shipped tiers — what snapshotSettings writes back. */
|
|
72
|
+
export function getAgentTiersConfiguredSettings(): AgentTiersSettings {
|
|
73
|
+
return structuredClone(agentTiersConfigured);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** Exactly-equal profile? Used to strip untouched shipped tiers from the configured view. */
|
|
77
|
+
function sameProfile(a: AgentTierProfile, b: AgentTierProfile): boolean {
|
|
78
|
+
return a.model === b.model && a.thinking === b.thinking && (a.description ?? "") === (b.description ?? "");
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* Install the effective tier catalogue.
|
|
83
|
+
*
|
|
84
|
+
* The shipped `fast` tier is merged in unless the caller already defined it or
|
|
85
|
+
* explicitly blocked it — a user catalogue wins over the shipped default, and a
|
|
86
|
+
* tombstone means "do not substitute", which applies to shipped defaults too.
|
|
87
|
+
*
|
|
88
|
+
* The configured view is derived from the same input by stripping profiles that
|
|
89
|
+
* exactly equal a shipped default, so the UI can operate on the effective view
|
|
90
|
+
* and send it back without materializing untouched shipped tiers into
|
|
91
|
+
* `subagents.json`. Editing a shipped tier (changing its model, thinking, or
|
|
92
|
+
* description) makes it a user-owned profile and it is then persisted; deleting
|
|
93
|
+
* one leaves its tombstone, which persists.
|
|
94
|
+
*/
|
|
45
95
|
export function setAgentTiersSettings(settings: AgentTiersSettings): void {
|
|
46
|
-
|
|
96
|
+
const effective = structuredClone(settings);
|
|
97
|
+
const profiles = { ...(effective.profiles ?? {}) };
|
|
98
|
+
const blocked = new Set<string>(effective.blockedProfiles ?? []);
|
|
99
|
+
|
|
100
|
+
const configuredProfiles: Record<string, AgentTierProfile> = {};
|
|
101
|
+
for (const [key, profile] of Object.entries(profiles)) {
|
|
102
|
+
const shipped = SHIPPED_AGENT_TIER_PROFILES[key];
|
|
103
|
+
if (!blocked.has(key) && shipped !== undefined && sameProfile(profile, shipped)) continue;
|
|
104
|
+
configuredProfiles[key] = profile;
|
|
105
|
+
}
|
|
106
|
+
for (const [key, profile] of Object.entries(SHIPPED_AGENT_TIER_PROFILES)) {
|
|
107
|
+
if (!blocked.has(key) && profiles[key] === undefined) profiles[key] = profile;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
agentTiersSettings = { ...effective, profiles };
|
|
111
|
+
const configured: AgentTiersSettings = { ...effective };
|
|
112
|
+
if (Object.keys(configuredProfiles).length > 0) configured.profiles = configuredProfiles;
|
|
113
|
+
else delete configured.profiles;
|
|
114
|
+
agentTiersConfigured = configured;
|
|
47
115
|
}
|
|
48
116
|
|
|
49
117
|
/**
|
|
@@ -173,13 +241,24 @@ export function upsertAgentTierProfile(
|
|
|
173
241
|
* A `defaultTier` pointing at it is cleared in the same step. Leaving it would
|
|
174
242
|
* turn every later spawn that names no tier into a hard refusal, which is a
|
|
175
243
|
* strange thing to get from deleting a tier you had stopped using.
|
|
244
|
+
*
|
|
245
|
+
* Deleting a shipped tier (the default `fast`) tombstones it instead of just
|
|
246
|
+
* dropping it: the shipped merge in `setAgentTiersSettings` would otherwise
|
|
247
|
+
* silently re-add it on the next load, and a user who deletes it means it. The
|
|
248
|
+
* tombstone says "do not substitute", which is exactly the semantics the load
|
|
249
|
+
* path already honors for malformed profiles. Explore still names `fast` in its
|
|
250
|
+
* frontmatter, so the spawn refusal then says so loudly until the agent file or
|
|
251
|
+
* the tier is fixed.
|
|
176
252
|
*/
|
|
177
253
|
export function removeAgentTierProfile(settings: AgentTiersSettings, key: string): AgentTiersSettings {
|
|
178
254
|
const { [key]: _removed, ...profiles } = settings.profiles ?? {};
|
|
255
|
+
const shipped = Object.hasOwn(SHIPPED_AGENT_TIER_PROFILES, key);
|
|
256
|
+
let blocked = withoutBlocked(settings.blockedProfiles, key);
|
|
257
|
+
if (shipped) blocked = [...(blocked ?? []), key];
|
|
179
258
|
return compactTierSettings({
|
|
180
259
|
...settings,
|
|
181
260
|
profiles,
|
|
182
|
-
blockedProfiles:
|
|
261
|
+
blockedProfiles: blocked,
|
|
183
262
|
...(settings.defaultTier === key ? { defaultTier: undefined } : {}),
|
|
184
263
|
});
|
|
185
264
|
}
|
package/src/agent-types.ts
CHANGED
|
@@ -295,6 +295,7 @@ export function getToolNamesForType(type: string): string[] {
|
|
|
295
295
|
/** Get config for a type (case-insensitive, returns a SubagentTypeConfig-compatible object). Falls back to general-purpose. */
|
|
296
296
|
export function getConfig(type: string): {
|
|
297
297
|
displayName: string;
|
|
298
|
+
color?: string;
|
|
298
299
|
description: string;
|
|
299
300
|
builtinToolNames: string[];
|
|
300
301
|
extensions: true | string[] | false;
|
|
@@ -307,6 +308,7 @@ export function getConfig(type: string): {
|
|
|
307
308
|
if (config && config.enabled !== false) {
|
|
308
309
|
return {
|
|
309
310
|
displayName: config.displayName ?? config.name,
|
|
311
|
+
color: config.color,
|
|
310
312
|
description: config.description,
|
|
311
313
|
builtinToolNames: config.builtinToolNames ?? BUILTIN_TOOL_NAMES,
|
|
312
314
|
extensions: config.extensions,
|
|
@@ -321,6 +323,7 @@ export function getConfig(type: string): {
|
|
|
321
323
|
if (gp && gp.enabled !== false) {
|
|
322
324
|
return {
|
|
323
325
|
displayName: gp.displayName ?? gp.name,
|
|
326
|
+
color: gp.color,
|
|
324
327
|
description: gp.description,
|
|
325
328
|
builtinToolNames: gp.builtinToolNames ?? BUILTIN_TOOL_NAMES,
|
|
326
329
|
extensions: gp.extensions,
|