@signalridge/pi-subagents 1.4.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -21,20 +21,24 @@ import {
21
21
  import type { WorkflowTier } from "@signalridge/pi-subagents-protocol";
22
22
  import { type AgentTierResolutionSnapshot, resolveAgentTier } from "./agent-tiers.js";
23
23
  import { BUILTIN_TOOL_NAMES, getAgentConfig, getConfig, getMemoryToolNames, getReadOnlyMemoryToolNames, getToolNamesForType } from "./agent-types.js";
24
+ import { createAskGate } from "./ask-tools.js";
24
25
  import { runInChildSessionContext } from "./child-context.js";
25
26
  import { buildParentContext, extractText } from "./context.js";
26
27
  import { DEFAULT_AGENTS } from "./default-agents.js";
27
28
  import { detectEnv } from "./env.js";
29
+ import { formatGateVerdict, type GateExec, runGate, workspaceFingerprint } from "./gate.js";
28
30
  import {
29
31
  INTERNAL_AGENT_CONFIG_OVERRIDE,
30
32
  type InternalAgentConfigOverride,
31
33
  } from "./internal-run.js";
32
34
  import { buildMemoryBlock, buildReadOnlyMemoryBlock } from "./memory.js";
35
+ import { type ModelRegistry, resolveModel } from "./model-resolver.js";
33
36
  import { checkModelScope } from "./model-scope.js";
34
37
  import { createNestedSubagentTools, getMaxSubagentDepth, type NestedAgentManager } from "./nested-tools.js";
35
38
  import { buildAgentPrompt, type PromptExtras } from "./prompts.js";
36
39
  import { shutdownAndDisposeSession } from "./session-lifecycle.js";
37
40
  import { preloadSkills } from "./skill-loader.js";
41
+ import { createSupervisorTool } from "./supervisor.js";
38
42
  import type { SubagentType, ThinkingLevel } from "./types.js";
39
43
  import type { WorkflowTierResolutionSnapshot } from "./workflow-tiers.js";
40
44
  import { resolveWorkflowTier } from "./workflow-tiers.js";
@@ -245,9 +249,11 @@ export function installExtensionToolScope(
245
249
  narrowing: Map<string, Set<string>>;
246
250
  /** Opt-in nested-delegation tool names to keep active despite the EXCLUDED strip. */
247
251
  nestedToolNames: Set<string>;
252
+ /** Per-call approval gate from `ask_tools:`, when the agent declares any. */
253
+ askGate?: (toolName: string, input: unknown) => Promise<{ block: true; reason: string } | undefined>;
248
254
  },
249
255
  ): void {
250
- const { loader, toolNames, disallowedSet, extNames, narrowing, nestedToolNames } = ctx;
256
+ const { loader, toolNames, disallowedSet, extNames, narrowing, nestedToolNames, askGate } = ctx;
251
257
 
252
258
  // The names allowed right now. Mirrors the `ext:` opt-in flip: when any `ext:`
253
259
  // selector is present, extension tools become an explicit allowlist — a loaded
@@ -299,16 +305,53 @@ export function installExtensionToolScope(
299
305
 
300
306
  const priorBeforeToolCall = session.agent.beforeToolCall;
301
307
  session.agent.beforeToolCall = async (context, signal) => {
302
- if (!inScope().has(context.toolCall.name)) {
303
- return {
304
- block: true,
305
- reason: `Tool "${context.toolCall.name}" is not available to this subagent.`,
306
- };
308
+ const run = async () => {
309
+ if (!inScope().has(context.toolCall.name)) {
310
+ return {
311
+ block: true,
312
+ reason: `Tool "${context.toolCall.name}" is not available to this subagent.`,
313
+ } as const;
314
+ }
315
+ // Scope first, then approval: a tool this agent may not use at all is
316
+ // refused without troubling the user about it.
317
+ const gated = await askGate?.(context.toolCall.name, (context.toolCall as { input?: unknown }).input);
318
+ if (gated) return gated;
319
+ return priorBeforeToolCall?.(context, signal);
320
+ };
321
+ // Default per-tool timeout prevents a hung beforeToolCall (slow extension,
322
+ // stuck LLM arbitrator) from stalling the subagent forever. Approval
323
+ // dialogs are wrapped too — a tool that cannot be approved in time is
324
+ // blocked fail-closed rather than left hanging.
325
+ if (defaultToolTimeoutMs > 0) {
326
+ try {
327
+ return await withTimeout(run(), defaultToolTimeoutMs, `beforeToolCall for "${context.toolCall.name}"`);
328
+ } catch (err) {
329
+ return { block: true, reason: (err as Error).message };
330
+ }
307
331
  }
308
- return priorBeforeToolCall?.(context, signal);
332
+ return run();
309
333
  };
310
334
  }
311
335
 
336
+ /**
337
+ * Run an agent's `gate:` command and format its verdict for the result text.
338
+ *
339
+ * Failures here are contained: a gate that cannot be fingerprinted simply runs
340
+ * uncached, and one that cannot run at all reports as failed rather than taking
341
+ * the agent's whole result down with it.
342
+ */
343
+ async function runConfiguredGate(command: string, cwd: string, pi: ExtensionAPI): Promise<string> {
344
+ const exec: GateExec = (file, args, execOptions) =>
345
+ pi.exec(file, args, execOptions as Parameters<ExtensionAPI["exec"]>[2]);
346
+ try {
347
+ const fingerprint = await workspaceFingerprint(cwd, exec);
348
+ const verdict = await runGate({ command, cwd, exec, ...(fingerprint ? { fingerprint } : {}) });
349
+ return formatGateVerdict(command, verdict);
350
+ } catch (error: unknown) {
351
+ return `\n\n---\nAcceptance gate \`${command}\`: could not run (${error instanceof Error ? error.message : String(error)})`;
352
+ }
353
+ }
354
+
312
355
  /** Default max turns. undefined = unlimited (no turn limit). */
313
356
  let defaultMaxTurns: number | undefined;
314
357
 
@@ -324,6 +367,48 @@ export function getDefaultMaxTurns(): number | undefined { return defaultMaxTurn
324
367
  export function setDefaultMaxTurns(n: number | undefined): void { defaultMaxTurns = normalizeMaxTurns(n); }
325
368
 
326
369
  /** Additional turns allowed after the soft limit steer message. */
370
+ /**
371
+ * Fraction of a resource budget at which the wrap-up steer is sent.
372
+ *
373
+ * Not 1.0: an agent told to produce its final answer needs allowance left to
374
+ * produce it, so the steer has to arrive while there is still budget to spend
375
+ * on the response.
376
+ */
377
+ const SOFT_BUDGET_FRACTION = 0.8;
378
+
379
+ /** Project defaults for the per-agent resource budgets. `0` = unlimited. */
380
+ let defaultMaxTokens = 0;
381
+ let defaultMaxToolCalls = 0;
382
+
383
+ /** Token budget for one agent run, from settings. `0` disables the cap. */
384
+ export function getDefaultMaxTokens(): number { return defaultMaxTokens; }
385
+ export function setDefaultMaxTokens(n: number): void { defaultMaxTokens = Math.max(0, Math.floor(n)); }
386
+
387
+ /** Tool-call budget for one agent run, from settings. `0` disables the cap. */
388
+ export function getDefaultMaxToolCalls(): number { return defaultMaxToolCalls; }
389
+ export function setDefaultMaxToolCalls(n: number): void { defaultMaxToolCalls = Math.max(0, Math.floor(n)); }
390
+
391
+ /** Default per-tool timeout; `0` disables (no timeout). Mirrors tintinweb — no per-tool timeout by default; hung tools are reclaimed via session abort/quiescence, not a hard tool cut. Set via settings defaultToolTimeoutMs when needed. */
392
+ const DEFAULT_TOOL_TIMEOUT_MS = 0;
393
+ const TOOL_TIMEOUT_CEILING_MS = 600_000;
394
+ let defaultToolTimeoutMs = DEFAULT_TOOL_TIMEOUT_MS;
395
+ export function getDefaultToolTimeoutMs(): number { return defaultToolTimeoutMs; }
396
+ export function setDefaultToolTimeoutMs(n: number): void {
397
+ defaultToolTimeoutMs = Math.max(0, Math.min(Math.floor(n), TOOL_TIMEOUT_CEILING_MS));
398
+ }
399
+ /** Race a promise against a timeout that rejects with `label timed out after ms`. */
400
+ export function withTimeout<T>(promise: Promise<T>, ms: number, label: string): Promise<T> {
401
+ if (ms <= 0) return promise;
402
+ let handle: ReturnType<typeof setTimeout> | undefined;
403
+ const timeout = new Promise<never>((_, reject) => {
404
+ handle = setTimeout(() => reject(new Error(`${label} timed out after ${ms}ms`)), ms);
405
+ handle.unref?.();
406
+ });
407
+ return Promise.race([promise, timeout]).finally(() => {
408
+ if (handle) clearTimeout(handle);
409
+ });
410
+ }
411
+
327
412
  let graceTurns = 5;
328
413
 
329
414
  /** Get the grace turns value. */
@@ -331,13 +416,56 @@ export function getGraceTurns(): number { return graceTurns; }
331
416
  /** Set the grace turns value (minimum 1). */
332
417
  export function setGraceTurns(n: number): void { graceTurns = Math.max(1, n); }
333
418
 
419
+ /**
420
+ * Model every subagent falls back to when nothing else picked one.
421
+ *
422
+ * Held here rather than in settings.ts because this is the one module that
423
+ * consumes it, and every spawn path already reaches model resolution through
424
+ * `runAgent`.
425
+ *
426
+ * Stored verbatim, `"inherit"` included, rather than normalized to `undefined`:
427
+ * the two are the same at spawn time but not on disk. A project that must
428
+ * cancel a global `defaultModel` has to write `"inherit"` into its own settings
429
+ * file, and a state that had already collapsed it to `undefined` would persist
430
+ * as an absent key and let the global value win again on the next start.
431
+ */
432
+ let defaultModel: string | undefined;
433
+
434
+ /** The configured fallback model reference — a `provider/model`, `"inherit"`, or unset. */
435
+ export function getDefaultModel(): string | undefined { return defaultModel; }
436
+ /** Set the fallback model reference. `undefined` and blank clear it; `"inherit"` is kept. */
437
+ export function setDefaultModel(ref: string | undefined): void {
438
+ const trimmed = ref?.trim();
439
+ defaultModel = trimmed ? trimmed : undefined;
440
+ }
441
+
442
+ /**
443
+ * The configured default model, resolved against this machine's registry.
444
+ *
445
+ * Resolved with the same fuzzy `resolveModel` the tiers use, so a hand-written
446
+ * `subagents.json` may name a model the way a person would. Unlike a tier it
447
+ * never throws: it is the value nobody chose at the call site, so an
448
+ * unavailable one yields to the parent rather than taking every spawn on this
449
+ * machine down with it.
450
+ *
451
+ * Exported because the callers that compute a spawn's model for the scope check
452
+ * and the UI label have to see the same answer this module will act on.
453
+ */
454
+ export function resolveConfiguredDefaultModel(
455
+ registry: ModelRegistry<Model<any>>,
456
+ ): Model<any> | undefined {
457
+ if (!defaultModel || defaultModel === "inherit") return undefined;
458
+ const resolved = resolveModel(defaultModel, registry);
459
+ return typeof resolved === "string" ? undefined : resolved;
460
+ }
461
+
334
462
  /**
335
463
  * Try to find the right model for an agent type.
336
- * Priority: explicit option > config.model > parent model.
464
+ * Priority: explicit option > config.model > configured default model > parent model.
337
465
  */
338
466
  function resolveDefaultModel(
339
467
  parentModel: Model<any> | undefined,
340
- registry: { find(provider: string, modelId: string): Model<any> | undefined; getAvailable?(): Model<any>[] },
468
+ registry: ModelRegistry<Model<any>>,
341
469
  configModel?: string,
342
470
  ): Model<any> | undefined {
343
471
  if (configModel) {
@@ -349,7 +477,7 @@ function resolveDefaultModel(
349
477
  // Build a set of available model keys for fast lookup
350
478
  const available = registry.getAvailable?.();
351
479
  const availableKeys = available
352
- ? new Set(available.map((m: any) => `${m.provider}/${m.id}`))
480
+ ? new Set(available.map((m) => `${m.provider}/${m.id}`))
353
481
  : undefined;
354
482
  const isAvailable = (p: string, id: string) =>
355
483
  !availableKeys || availableKeys.has(`${p}/${id}`);
@@ -359,7 +487,7 @@ function resolveDefaultModel(
359
487
  }
360
488
  }
361
489
 
362
- return parentModel;
490
+ return resolveConfiguredDefaultModel(registry) ?? parentModel;
363
491
  }
364
492
 
365
493
  /** Info about a tool event in the subagent. */
@@ -387,6 +515,12 @@ export interface RunOptions {
387
515
  * the same precedence and the same fail-closed errors from one place.
388
516
  */
389
517
  agentTier?: string;
518
+ /** Optional toolset hint forwarded by managed workflow callers. */
519
+ toolset?: string;
520
+ /** Additional tool names denied by the caller, merged with agent frontmatter. */
521
+ excludeTools?: string[];
522
+ /** Named sequential-thread hint; used for stable session naming. */
523
+ thread?: string;
390
524
  /** Parent thinking level used only when a tier profile omits thinking. */
391
525
  parentThinking?: ThinkingLevel;
392
526
  /** Override working directory (e.g. for worktree isolation). */
@@ -434,6 +568,27 @@ export interface RunOptions {
434
568
  * JSON/public spawn field; the generation wizard is the only issuer.
435
569
  */
436
570
  readonly [INTERNAL_AGENT_CONFIG_OVERRIDE]?: InternalAgentConfigOverride;
571
+ /**
572
+ * Reopen an existing conversation from this session file instead of starting
573
+ * a new one. Package-internal: the only issuer is the `@handle` mention
574
+ * dispatcher, replaying a path this extension itself wrote to the resumable
575
+ * index. A caller-supplied value would let a spawn read any session on disk,
576
+ * so no public spawn surface forwards it.
577
+ */
578
+ resumeSessionFile?: string;
579
+ /**
580
+ * Project default for persisting a top-level agent's conversation to disk,
581
+ * from the `rememberAgents` setting. Frontmatter `persist_session:` still
582
+ * wins; nested agents never persist. Without this, only agents that opted in
583
+ * by frontmatter leave a transcript, and `@handle` has nothing to reopen.
584
+ */
585
+ rememberAgents?: boolean;
586
+ /**
587
+ * Whether this agent may ask its human a question with `contact_supervisor`.
588
+ * Defaults to on wherever there is a UI to ask through; `false` withholds the
589
+ * tool entirely rather than injecting one that always refuses.
590
+ */
591
+ supervisorQuestions?: boolean;
437
592
  /** Runtime bridge for opt-in child-safe nested delegation. */
438
593
  nestedRuntime?: {
439
594
  manager: NestedAgentManager;
@@ -907,9 +1062,10 @@ export async function runAgent(
907
1062
  if (scopeVerdict.kind === "error") throw new Error(scopeVerdict.message);
908
1063
  if (scopeVerdict.kind === "warn" && ctx.hasUI) ctx.ui.notify(scopeVerdict.message, "warning");
909
1064
  }
910
- const disallowedSet = agentConfig?.disallowedTools
911
- ? new Set(agentConfig.disallowedTools)
912
- : undefined;
1065
+ const disallowedSet = (() => {
1066
+ const names = [...(agentConfig?.disallowedTools ?? []), ...(options.excludeTools ?? [])];
1067
+ return names.length > 0 ? new Set(names) : undefined;
1068
+ })();
913
1069
 
914
1070
  // Nested delegation tools (opt-in, ownership-scoped). Empty unless the agent
915
1071
  // set `allowed_subagents` and a nestedRuntime was provided — and never when
@@ -934,7 +1090,26 @@ export async function runAgent(
934
1090
  configCwd,
935
1091
  })
936
1092
  : [];
937
- const nestedToolNames = new Set(nestedTools.map(tool => tool.name));
1093
+ // `contact_supervisor` is injected separately from the nested tools and under
1094
+ // a different condition. Nesting is gated on `allowedSubagents`, but the agent
1095
+ // that most needs to ask a question is a LEAF one that cannot delegate — so
1096
+ // gating them together would withhold it from exactly those agents. It needs
1097
+ // only a human to answer, which `hasUI` decides. Isolation still suppresses
1098
+ // it: an `isolated: true` agent is defined as built-in tools only.
1099
+ const supervisorTools =
1100
+ ctx.hasUI && !isolated && options.supervisorQuestions !== false
1101
+ ? createSupervisorTool({
1102
+ agentLabel: agentConfig?.displayName ?? type,
1103
+ ask: {
1104
+ input: (title, placeholder) => ctx.ui.input(title, placeholder),
1105
+ select: (title, choices) => ctx.ui.select(title, choices),
1106
+ },
1107
+ })
1108
+ : [];
1109
+ // One list from here on: both sets are custom tools this package injects, and
1110
+ // the scoping pass below keeps exactly the names it is given.
1111
+ const injectedTools = [...nestedTools, ...supervisorTools];
1112
+ const nestedToolNames = new Set(injectedTools.map(tool => tool.name));
938
1113
 
939
1114
  // ─── Tool scoping ───────────────────────────────────────────────────────
940
1115
  //
@@ -991,17 +1166,33 @@ export async function runAgent(
991
1166
  for (const name of disallowedSet) denyTools.add(name);
992
1167
  }
993
1168
  sessionExcludeTools = [...denyTools];
1169
+ // Named toolsets are advisory labels; they never widen the configured
1170
+ // allowlist. Concrete tool availability remains owned by the agent config.
994
1171
  }
995
1172
 
996
1173
  const settingsManager = SettingsManager.create(configCwd, agentDir);
997
1174
  const configuredSessionDir = resolveConfiguredSessionDir(agentConfig?.sessionDir, effectiveCwd);
998
1175
  const defaultSessionDir = process.env.PI_CODING_AGENT_SESSION_DIR ?? settingsManager.getSessionDir?.();
999
- const parentSession = agentConfig?.persistSession ? ctx.sessionManager.getSessionFile?.() : undefined;
1000
- const sessionManager = agentConfig?.persistSession
1001
- ? parentSession
1002
- ? SessionManager.create(effectiveCwd, configuredSessionDir ?? defaultSessionDir, { parentSession })
1003
- : SessionManager.create(effectiveCwd, configuredSessionDir ?? defaultSessionDir)
1004
- : SessionManager.inMemory(effectiveCwd);
1176
+ // Frontmatter wins when it says anything; otherwise the project default,
1177
+ // which `rememberAgents` supplies for top-level agents only. A nested agent
1178
+ // is an implementation detail of its parent and is never addressable by
1179
+ // `@handle`, so it has nothing to gain from a transcript on disk.
1180
+ const persistSession = agentConfig?.persistSession ?? (options.nestedRuntime ? false : options.rememberAgents === true);
1181
+ // Optional metadata — it only nests the subagent under its spawner in
1182
+ // `/resume`. Now that `rememberAgents` persists every top-level spawn, a
1183
+ // context without a session manager (a bare programmatic ctx) must still
1184
+ // persist rather than take the whole spawn down.
1185
+ const parentSession = persistSession ? ctx.sessionManager?.getSessionFile?.() : undefined;
1186
+ const sessionManager = options.resumeSessionFile
1187
+ ? // Reopening an existing conversation: the file already carries its own
1188
+ // header (cwd, parent) and history, so none of the create-time options
1189
+ // apply. `sessionDir` still matters for a later /new or /branch off it.
1190
+ SessionManager.open(options.resumeSessionFile, configuredSessionDir ?? defaultSessionDir)
1191
+ : persistSession
1192
+ ? parentSession
1193
+ ? SessionManager.create(effectiveCwd, configuredSessionDir ?? defaultSessionDir, { parentSession })
1194
+ : SessionManager.create(effectiveCwd, configuredSessionDir ?? defaultSessionDir)
1195
+ : SessionManager.inMemory(effectiveCwd);
1005
1196
 
1006
1197
  // Pi 0.80.8 replaced createAgentSession's modelRegistry option with
1007
1198
  // modelRuntime, but ExtensionContext still exposes only the registry facade.
@@ -1019,7 +1210,7 @@ export async function runAgent(
1019
1210
  ...(parentModelRuntime !== undefined && { modelRuntime: parentModelRuntime }),
1020
1211
  model,
1021
1212
  tools: sessionTools,
1022
- customTools: nestedTools,
1213
+ customTools: injectedTools,
1023
1214
  resourceLoader: loader,
1024
1215
  };
1025
1216
  if (sessionExcludeTools) {
@@ -1039,9 +1230,9 @@ export async function runAgent(
1039
1230
  try {
1040
1231
  throwIfAborted(options.signal);
1041
1232
 
1042
- const baseSessionName = agentConfig?.name ?? type;
1233
+ const baseSessionName = options.thread ? `workflow-thread:${options.thread}` : (agentConfig?.name ?? type);
1043
1234
  session.setSessionName(
1044
- options.agentId ? `${baseSessionName}#${options.agentId.slice(0, 8)}` : baseSessionName,
1235
+ options.agentId && !options.thread ? `${baseSessionName}#${options.agentId.slice(0, 8)}` : baseSessionName,
1045
1236
  );
1046
1237
 
1047
1238
  // Bind extensions so that session_start fires and extensions can initialize
@@ -1064,6 +1255,15 @@ export async function runAgent(
1064
1255
  // (we can't deny the name of a tool that hasn't registered yet). Both are
1065
1256
  // handled below by re-deriving scope from the loader's live extension maps —
1066
1257
  // `registerTool` writes into those same maps, so late arrivals are judged too.
1258
+ // `ask_tools:` gates individual CALLS, which is orthogonal to which tools
1259
+ // exist — so it applies to isolated agents too, where the scope installer
1260
+ // below never runs because the registry is already statically allowlisted.
1261
+ const askGate = createAskGate({
1262
+ askTools: agentConfig?.askTools ?? [],
1263
+ agentLabel: agentConfig?.displayName ?? type,
1264
+ ...(ctx.hasUI ? { confirm: (title: string, message: string) => ctx.ui.confirm(title, message) } : {}),
1265
+ });
1266
+
1067
1267
  if (!noExtensions) {
1068
1268
  installExtensionToolScope(session, {
1069
1269
  loader,
@@ -1072,7 +1272,26 @@ export async function runAgent(
1072
1272
  extNames,
1073
1273
  narrowing,
1074
1274
  nestedToolNames,
1275
+ ...(askGate ? { askGate } : {}),
1075
1276
  });
1277
+ } else if (askGate) {
1278
+ // Same hook, without the scope check the allowlist already performed.
1279
+ const priorBeforeToolCall = session.agent.beforeToolCall;
1280
+ session.agent.beforeToolCall = async (context, signal) => {
1281
+ const run = async () => {
1282
+ const gated = await askGate(context.toolCall.name, (context.toolCall as { input?: unknown }).input);
1283
+ if (gated) return gated;
1284
+ return priorBeforeToolCall?.(context, signal);
1285
+ };
1286
+ if (defaultToolTimeoutMs > 0) {
1287
+ try {
1288
+ return await withTimeout(run(), defaultToolTimeoutMs, `beforeToolCall for "${context.toolCall.name}"`);
1289
+ } catch (err) {
1290
+ return { block: true, reason: (err as Error).message };
1291
+ }
1292
+ }
1293
+ return run();
1294
+ };
1076
1295
  }
1077
1296
 
1078
1297
  if (options.onSessionCreated) {
@@ -1089,7 +1308,76 @@ export async function runAgent(
1089
1308
  let softLimitReached = false;
1090
1309
  let aborted = false;
1091
1310
 
1311
+ // Resource budgets, in the same shape as max_turns: a wrap-up steer at the
1312
+ // soft threshold, an abort at the hard one. They bound what a single agent
1313
+ // can spend on its own, which turn count does not — one turn can burn an
1314
+ // arbitrary number of tokens or tool calls.
1315
+ //
1316
+ // `0` means unlimited here, matching this package's existing convention for
1317
+ // `maxTurns`. That is the opposite of `maxSubagentSpawnsPerBranch`, and
1318
+ // deliberately: these are opt-in resource caps that ship off, while the
1319
+ // branch spawn budget is a safety valve that ships on.
1320
+ const tokenBudget = normalizeMaxTurns(agentConfig?.maxTokens ?? defaultMaxTokens);
1321
+ const toolCallBudget = normalizeMaxTurns(agentConfig?.maxToolCalls ?? defaultMaxToolCalls);
1322
+ let budgetTokens = 0;
1323
+ let budgetToolCalls = 0;
1324
+ let tokenSoftReached = false;
1325
+ let toolSoftReached = false;
1326
+
1327
+ /**
1328
+ * Apply one budget, returning the next soft-limit state.
1329
+ *
1330
+ * The soft threshold is `SOFT_BUDGET_FRACTION` of the budget so the agent has
1331
+ * room left to actually write its answer after being told to wrap up — a
1332
+ * steer sent at 100% would be asking for a final response with no allowance
1333
+ * to produce it.
1334
+ */
1335
+ const applyBudget = (used: number, budget: number | undefined, softReached: boolean, label: string): boolean => {
1336
+ if (budget == null) return softReached;
1337
+ // The hard limit is checked FIRST. One message can consume more than the
1338
+ // whole budget, and testing the soft threshold first would answer that with
1339
+ // a wrap-up steer and no abort — leaving an agent already over budget to
1340
+ // run on until its next event.
1341
+ if (used >= budget) {
1342
+ aborted = true;
1343
+ session.abort();
1344
+ return true;
1345
+ }
1346
+ if (!softReached && used >= budget * SOFT_BUDGET_FRACTION) {
1347
+ session.steer(
1348
+ `You are near your ${label} budget for this task. Wrap up immediately — provide your final answer now.`,
1349
+ );
1350
+ return true;
1351
+ }
1352
+ return softReached;
1353
+ };
1354
+
1092
1355
  let currentMessageText = "";
1356
+ // Per-tool timeout: a hung bash/MCP call must not stall the subagent forever.
1357
+ // Each tool_execution_start arms a timer; the matching end clears it. On
1358
+ // timeout the session is aborted — which propagates via the tool's
1359
+ // AbortSignal into the hanging execute() and surfaces as an error result.
1360
+ // `contact_supervisor` is excluded: it intentionally waits for a human.
1361
+ const pendingToolTimeouts = new Map<string, ReturnType<typeof setTimeout>>();
1362
+ const clearToolTimeout = (toolCallId: string): void => {
1363
+ const handle = pendingToolTimeouts.get(toolCallId);
1364
+ if (handle) {
1365
+ clearTimeout(handle);
1366
+ pendingToolTimeouts.delete(toolCallId);
1367
+ }
1368
+ };
1369
+ const armToolTimeout = (toolCallId: string, toolName: string): void => {
1370
+ if (defaultToolTimeoutMs <= 0) return;
1371
+ if (toolName === "contact_supervisor") return;
1372
+ const handle = setTimeout(() => {
1373
+ pendingToolTimeouts.delete(toolCallId);
1374
+ try {
1375
+ session.abort();
1376
+ } catch {}
1377
+ }, defaultToolTimeoutMs);
1378
+ handle.unref?.();
1379
+ pendingToolTimeouts.set(toolCallId, handle);
1380
+ };
1093
1381
  const unsubTurns = session.subscribe((event: AgentSessionEvent) => {
1094
1382
  if (event.type === "turn_end") {
1095
1383
  turnCount++;
@@ -1113,9 +1401,13 @@ export async function runAgent(
1113
1401
  }
1114
1402
  if (event.type === "tool_execution_start") {
1115
1403
  options.onToolActivity?.({ type: "start", toolName: event.toolName });
1404
+ armToolTimeout((event as { toolCallId: string }).toolCallId, event.toolName);
1116
1405
  }
1117
1406
  if (event.type === "tool_execution_end") {
1407
+ clearToolTimeout((event as { toolCallId: string }).toolCallId);
1118
1408
  options.onToolActivity?.({ type: "end", toolName: event.toolName });
1409
+ budgetToolCalls++;
1410
+ toolSoftReached = applyBudget(budgetToolCalls, toolCallBudget, toolSoftReached, "tool call");
1119
1411
  }
1120
1412
  if (event.type === "message_end" && event.message.role === "assistant") {
1121
1413
  const u = (event.message as any).usage;
@@ -1124,6 +1416,12 @@ export async function runAgent(
1124
1416
  output: u.output ?? 0,
1125
1417
  cacheWrite: u.cacheWrite ?? 0,
1126
1418
  });
1419
+ if (u) {
1420
+ // Same total as `getLifetimeTotal`: input + output + cacheWrite, with
1421
+ // cacheRead deliberately excluded.
1422
+ budgetTokens += (u.input ?? 0) + (u.output ?? 0) + (u.cacheWrite ?? 0);
1423
+ tokenSoftReached = applyBudget(budgetTokens, tokenBudget, tokenSoftReached, "token");
1424
+ }
1127
1425
  }
1128
1426
  if (event.type === "compaction_end" && !event.aborted && event.result) {
1129
1427
  options.onCompaction?.({ reason: event.reason, tokensBefore: event.result.tokensBefore });
@@ -1151,9 +1449,18 @@ export async function runAgent(
1151
1449
  } finally {
1152
1450
  unsubTurns();
1153
1451
  collector.unsubscribe();
1452
+ for (const handle of pendingToolTimeouts.values()) clearTimeout(handle);
1453
+ pendingToolTimeouts.clear();
1154
1454
  }
1155
1455
 
1156
- const responseText = collector.getText().trim() || getLastAssistantText(session, startLen);
1456
+ const baseText = collector.getText().trim() || getLastAssistantText(session, startLen);
1457
+ // The acceptance gate runs AFTER the agent is done and its verdict is
1458
+ // appended to the result, so the parent reads the check and the claim it is
1459
+ // checking side by side. It deliberately does not steer the agent to fix
1460
+ // what failed: the gate is evidence for the caller, not another turn.
1461
+ const responseText = agentConfig?.gate
1462
+ ? `${baseText}${await runConfiguredGate(agentConfig.gate, effectiveCwd, options.pi)}`
1463
+ : baseText;
1157
1464
  completed = true;
1158
1465
  return { responseText, session, aborted, steered: softLimitReached, failure: finalTurnError(session, startLen) };
1159
1466
  } finally {
@@ -1209,6 +1516,32 @@ export async function resumeAgent(
1209
1516
  }
1210
1517
  })
1211
1518
  : () => {};
1519
+ // Per-tool timeout for resume as well — same stuck-command fix as runAgent.
1520
+ const resumePendingTimeouts = new Map<string, ReturnType<typeof setTimeout>>();
1521
+ const unsubResumeTimeout =
1522
+ defaultToolTimeoutMs > 0
1523
+ ? session.subscribe((event: AgentSessionEvent) => {
1524
+ if (event.type === "tool_execution_start") {
1525
+ const id = (event as { toolCallId: string }).toolCallId;
1526
+ if (event.toolName === "contact_supervisor") return;
1527
+ const handle = setTimeout(() => {
1528
+ resumePendingTimeouts.delete(id);
1529
+ try {
1530
+ session.abort();
1531
+ } catch {}
1532
+ }, defaultToolTimeoutMs);
1533
+ handle.unref?.();
1534
+ resumePendingTimeouts.set(id, handle);
1535
+ } else if (event.type === "tool_execution_end") {
1536
+ const id = (event as { toolCallId: string }).toolCallId;
1537
+ const handle = resumePendingTimeouts.get(id);
1538
+ if (handle) {
1539
+ clearTimeout(handle);
1540
+ resumePendingTimeouts.delete(id);
1541
+ }
1542
+ }
1543
+ })
1544
+ : () => {};
1212
1545
 
1213
1546
  try {
1214
1547
  throwIfAborted(options.signal);
@@ -1216,6 +1549,9 @@ export async function resumeAgent(
1216
1549
  } finally {
1217
1550
  collector.unsubscribe();
1218
1551
  unsubEvents();
1552
+ unsubResumeTimeout();
1553
+ for (const handle of resumePendingTimeouts.values()) clearTimeout(handle);
1554
+ resumePendingTimeouts.clear();
1219
1555
  cleanupAbort();
1220
1556
  }
1221
1557