@yusukeshib/pi-babysit 0.3.17 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +30 -18
  2. package/index.ts +703 -163
  3. package/package.json +1 -1
package/index.ts CHANGED
@@ -19,7 +19,7 @@
19
19
  * `pi --mode rpc` worker. Tasks are injected as RPC `prompt`
20
20
  * commands over stdin, completion is detected from the JSONL
21
21
  * event stream (`agent_settled`), NOT process exit; the session
22
- * stays alive for cheap follow-up tasks. Same design as the
22
+ * remains reusable during its configured idle grace. Same design as the
23
23
  * old pi-subagent extension.
24
24
  *
25
25
  * The "profile" is a tool-parameter, not a separate tool set: one small tool
@@ -274,11 +274,14 @@ export function isSupportedBabysitVersion(output: string): boolean {
274
274
  // Cached preflight — probe `babysit --version` exactly once per process.
275
275
  // undefined = not probed, null = supported, string = actionable error.
276
276
  let babysitPreflightError: string | null | undefined;
277
+ let babysitPreflightCheckedAt = 0;
277
278
  async function babysitAvailable(): Promise<boolean> {
278
- // Cache only success. A missing or outdated binary may be installed while pi
279
- // stays open, so subsequent tool calls must be able to recover without a restart.
280
279
  if (babysitPreflightError === null) return true;
280
+ // Briefly negative-cache failures so repeated mistaken calls do not fork a
281
+ // version probe each time, while still recovering quickly after installation.
282
+ if (babysitPreflightError && Date.now() - babysitPreflightCheckedAt < 2_000) return false;
281
283
  const r = await bs(["--version"]);
284
+ babysitPreflightCheckedAt = Date.now();
282
285
  if (r.code !== 0) {
283
286
  babysitPreflightError = INSTALL_HINT;
284
287
  } else if (!isSupportedBabysitVersion(r.stdout)) {
@@ -406,23 +409,61 @@ interface Meta {
406
409
  depth?: number;
407
410
  maxDepth?: number;
408
411
  budget?: SubagentBudget;
412
+ /** Soft-limit warning (80% by default) was accepted for this task. */
413
+ budgetWarnedAt?: number;
414
+ budgetWarningReason?: string;
415
+ /** Hard limit was first observed; grace is measured from observation, not RPC acceptance. */
409
416
  budgetExceededAt?: number;
410
417
  budgetReason?: string;
411
418
  budgetKilled?: boolean;
419
+ /** Prompt offset whose nested usage has already been charged to the parent session. */
420
+ usageReportedOffset?: number;
421
+ /** Prompt offset explicitly collected by foreground mode or babysit_wait. */
422
+ subagentCollectedOffset?: number;
423
+ /** Prompt offset whose ready-to-collect reminder was sent to the parent. */
424
+ subagentNotifiedOffset?: number;
425
+ /** Current task completion first observed by the reminder poller. */
426
+ subagentCompletionObservedOffset?: number;
427
+ subagentCompletionObservedAt?: number;
412
428
  }
413
429
 
414
430
  const metaDir = () => path.join(ROOT, "meta");
415
431
  const logPath = (id: string) => path.join(ROOT, "sessions", id, "output.log");
416
432
 
417
- function writeMeta(id: string, m: Meta): void {
433
+ function writeMeta(id: string, m: Meta): boolean {
434
+ const target = path.join(metaDir(), `${id}.json`);
435
+ const temp = `${target}.${process.pid}.${Date.now()}.${Math.random().toString(16).slice(2)}.tmp`;
418
436
  try {
419
437
  fs.mkdirSync(metaDir(), { recursive: true });
420
- fs.writeFileSync(path.join(metaDir(), `${id}.json`), JSON.stringify(m));
438
+ fs.writeFileSync(temp, JSON.stringify(m));
439
+ fs.renameSync(temp, target);
440
+ return true;
421
441
  } catch {
422
- /* best-effort */
442
+ try {
443
+ fs.rmSync(temp, { force: true });
444
+ } catch {
445
+ /* best-effort */
446
+ }
447
+ return false;
423
448
  }
424
449
  }
425
450
 
451
+ export function claimFileOnce(file: string, payload: string): boolean {
452
+ let fd: number;
453
+ try {
454
+ fs.mkdirSync(path.dirname(file), { recursive: true });
455
+ fd = fs.openSync(file, "wx");
456
+ } catch {
457
+ return false;
458
+ }
459
+ try {
460
+ fs.writeFileSync(fd, payload);
461
+ } finally {
462
+ fs.closeSync(fd);
463
+ }
464
+ return true;
465
+ }
466
+
426
467
  function readMeta(id: string): Meta | null {
427
468
  try {
428
469
  return JSON.parse(fs.readFileSync(path.join(metaDir(), `${id}.json`), "utf-8"));
@@ -448,8 +489,27 @@ function processIsAlive(pid: number): boolean {
448
489
  }
449
490
 
450
491
  const GC_LOCK_FILE = ".pi-babysit-gc.lock";
492
+ const GC_STAMP_FILE = ".pi-babysit-gc.last";
493
+ const AUTOMATIC_GC_INTERVAL_MS = 24 * 60 * 60 * 1_000;
451
494
  const ACTIVE_LEASE_PREFIX = ".pi-babysit-active-";
452
495
 
496
+ function automaticGcDue(now = Date.now()): boolean {
497
+ try {
498
+ return now - fs.statSync(path.join(ROOT_BASE, GC_STAMP_FILE)).mtimeMs >= AUTOMATIC_GC_INTERVAL_MS;
499
+ } catch {
500
+ return true;
501
+ }
502
+ }
503
+
504
+ function markAutomaticGc(now = new Date()): void {
505
+ try {
506
+ fs.mkdirSync(ROOT_BASE, { recursive: true });
507
+ fs.writeFileSync(path.join(ROOT_BASE, GC_STAMP_FILE), now.toISOString());
508
+ } catch {
509
+ /* best-effort; GC safety does not depend on this throttle stamp */
510
+ }
511
+ }
512
+
453
513
  function scanTreeStats(root: string): { bytes: number; newestMtimeMs: number } {
454
514
  let bytes = 0;
455
515
  let newestMtimeMs = 0;
@@ -678,6 +738,16 @@ export function shouldDeliverProcessCompletion(
678
738
  return meta?.kind === "process" && !meta.notified && !meta.notificationPaused;
679
739
  }
680
740
 
741
+ export function shouldDeliverSubagentCompletion(meta: Meta | null): meta is Meta & { kind: "subagent" } {
742
+ if (meta?.kind !== "subagent") return false;
743
+ const offset = meta.promptOffset ?? 0;
744
+ return (
745
+ meta.usageReportedOffset !== offset &&
746
+ meta.subagentCollectedOffset !== offset &&
747
+ meta.subagentNotifiedOffset !== offset
748
+ );
749
+ }
750
+
681
751
  export function isNotificationGroupReady(
682
752
  meta: Meta,
683
753
  sessions: BsSession[],
@@ -699,10 +769,14 @@ export function shouldKeepPolling(
699
769
  sessions: Array<{ id: string; state: string }>,
700
770
  metaFor: (id: string) => Meta | null,
701
771
  ): boolean {
702
- return sessions.some(
703
- (session) =>
704
- session.state === "running" || shouldDeliverProcessCompletion(metaFor(session.id)),
705
- );
772
+ return sessions.some((session) => {
773
+ const meta = metaFor(session.id);
774
+ return (
775
+ session.state === "running" ||
776
+ shouldDeliverProcessCompletion(meta) ||
777
+ (session.state !== "running" && shouldDeliverSubagentCompletion(meta))
778
+ );
779
+ });
706
780
  }
707
781
 
708
782
  export function shouldKeepPollingAfterList(
@@ -818,6 +892,18 @@ export function readLogBytesFrom(file: string, since: number): Buffer {
818
892
  return bytes.subarray(0, read);
819
893
  }
820
894
 
895
+ const RPC_RESPONSE_WINDOW_MAX_BYTES = 1_000_000;
896
+ function readLogWindowFrom(
897
+ file: string,
898
+ since: number,
899
+ maxBytes = RPC_RESPONSE_WINDOW_MAX_BYTES,
900
+ ): { bytes: Buffer; offset: number } {
901
+ const size = fs.statSync(file).size;
902
+ const requested = Math.min(Math.max(0, since), size);
903
+ const offset = Math.max(requested, size - maxBytes);
904
+ return { bytes: readLogBytesFrom(file, offset), offset };
905
+ }
906
+
821
907
  async function rpcResponse(
822
908
  id: string,
823
909
  since: number,
@@ -848,7 +934,8 @@ async function rpcResponse(
848
934
  try {
849
935
  // Long-lived follow-up workers can have very large historical logs. Read
850
936
  // only the response window rather than synchronously loading all history.
851
- structuredError = parseEvents(readLogBytesFrom(logPath(id), since).toString("utf8")).errorMsg ?? "";
937
+ const window = readLogWindowFrom(logPath(id), since);
938
+ structuredError = parseEvents(window.bytes.toString("utf8")).errorMsg ?? "";
852
939
  } catch {
853
940
  /* full log path below remains the diagnostic source */
854
941
  }
@@ -869,7 +956,8 @@ async function rpcResponse(
869
956
  };
870
957
  }
871
958
  try {
872
- return parseRpcResponseBytes(readLogBytesFrom(logPath(id), since), since, command);
959
+ const window = readLogWindowFrom(logPath(id), since);
960
+ return parseRpcResponseBytes(window.bytes, window.offset, command);
873
961
  } catch (error) {
874
962
  return { ok: false, error: `could not read ${command} response: ${String(error)}` };
875
963
  }
@@ -899,7 +987,7 @@ function byteLimitFromEnv(name: string, fallback: number): number {
899
987
  return Number.isSafeInteger(value) && value >= 0 ? value : fallback;
900
988
  }
901
989
 
902
- const TAIL_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_TAIL_MAX_BYTES", 8_000);
990
+ const TAIL_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_TAIL_MAX_BYTES", 4_000);
903
991
  // Direct run/wait results can carry more context because the caller explicitly
904
992
  // requested them. Unsolicited completion notifications default much smaller.
905
993
  const INLINE_OUTPUT_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_INLINE_OUTPUT_MAX_BYTES", 8_000);
@@ -910,6 +998,11 @@ const ANSWER_MAX_BYTES = 24_000; // single subagent answers / structured error m
910
998
  const MAX_MULTI_WAIT_SESSIONS = 32;
911
999
  const SUBAGENT_BUDGET_GRACE_MS =
912
1000
  parseDurMs(process.env.PI_BABYSIT_BUDGET_GRACE ?? "90s") ?? 90_000;
1001
+ const SUBAGENT_REAP_AFTER =
1002
+ process.env.PI_BABYSIT_REAP_AFTER ?? process.env.PI_SUBAGENT_REAP_AFTER ?? "120s";
1003
+ const SUBAGENT_REUSE_HINT = ["0", "off", "none"].includes(SUBAGENT_REAP_AFTER)
1004
+ ? "Session remains available until its absolute timeout."
1005
+ : `Session remains available for follow-ups during the ${SUBAGENT_REAP_AFTER} idle grace.`;
913
1006
 
914
1007
  export function clip(s: string, maxBytes = TAIL_MAX_BYTES): string {
915
1008
  if (maxBytes <= 0) return "";
@@ -940,15 +1033,27 @@ export function clipMultiWaitResult(
940
1033
  return clip(value, Math.min(maxBytes, ANSWER_MAX_BYTES));
941
1034
  }
942
1035
 
1036
+ interface SearchLogCacheEntry {
1037
+ size: number;
1038
+ mtimeMs: number;
1039
+ text: string;
1040
+ }
1041
+ const searchLogCache = new Map<string, SearchLogCacheEntry>();
1042
+
943
1043
  async function searchLog(
944
1044
  id: string,
945
1045
  pattern: string,
946
1046
  maxLines: number,
947
1047
  signal?: AbortSignal,
1048
+ maxBytes = TAIL_MAX_BYTES,
948
1049
  ): Promise<{ text: string; error?: string }> {
949
1050
  const file = logPath(id);
950
1051
  if (!fs.existsSync(file)) return { text: "", error: `Log file is missing: ${file}` };
951
1052
  if (signal?.aborted) return { text: "", error: "Log search was interrupted." };
1053
+ const stat = fs.statSync(file);
1054
+ const cacheKey = `${file}\u0000${pattern}\u0000${maxLines}\u0000${maxBytes}`;
1055
+ const cached = searchLogCache.get(cacheKey);
1056
+ if (cached?.size === stat.size && cached.mtimeMs === stat.mtimeMs) return { text: cached.text };
952
1057
 
953
1058
  // Run regex evaluation out of process so catastrophic backtracking or a huge
954
1059
  // no-newline log cannot freeze or exhaust pi's main Node process. The helper
@@ -996,7 +1101,10 @@ async function searchLog(
996
1101
  } else if (code !== 0) {
997
1102
  finish({ text: "", error: stderr.trim() || `Log search failed (exit ${code ?? "?"}).` });
998
1103
  } else {
999
- finish({ text: clip(stdout.trimEnd()) });
1104
+ const text = clip(stdout.trimEnd(), maxBytes);
1105
+ searchLogCache.set(cacheKey, { size: stat.size, mtimeMs: stat.mtimeMs, text });
1106
+ while (searchLogCache.size > 64) searchLogCache.delete(searchLogCache.keys().next().value as string);
1107
+ finish({ text });
1000
1108
  }
1001
1109
  });
1002
1110
  });
@@ -1030,6 +1138,33 @@ async function inlineOutput(
1030
1138
  return output ? `\n\nOutput:\n${output}` : "";
1031
1139
  }
1032
1140
 
1141
+ interface ProcessOutputSelection {
1142
+ pattern?: string;
1143
+ lines?: number;
1144
+ maxBytes?: number;
1145
+ }
1146
+
1147
+ async function selectedProcessOutput(
1148
+ id: string,
1149
+ status: BsSession,
1150
+ selection?: ProcessOutputSelection,
1151
+ signal?: AbortSignal,
1152
+ ): Promise<string> {
1153
+ if (!selection || (!selection.pattern && selection.lines == null && selection.maxBytes == null)) {
1154
+ return inlineOutput(id, status);
1155
+ }
1156
+ const maxBytes = selection.maxBytes ?? INLINE_OUTPUT_MAX_BYTES;
1157
+ const lines = Math.min(Math.max(1, Math.floor(selection.lines ?? 30)), 200);
1158
+ if (selection.pattern) {
1159
+ const result = await searchLog(id, selection.pattern, lines, signal, maxBytes);
1160
+ if (result.error) return `\nOutput filter failed: ${result.error}`;
1161
+ const body = result.text || `(no output matching /${selection.pattern}/)`;
1162
+ return `\n\nSelected output /${selection.pattern}/:\n${clip(body, maxBytes)}`;
1163
+ }
1164
+ const tail = (await bs(["log", "-s", id, "--tail", String(lines)])).stdout.trimEnd();
1165
+ return tail ? `\n\nSelected tail (${lines} lines max):\n${clip(tail, maxBytes)}` : "";
1166
+ }
1167
+
1033
1168
  export function summarizeNotificationCommand(command: string | undefined): string {
1034
1169
  const preview =
1035
1170
  (command ?? "?")
@@ -1305,7 +1440,10 @@ interface ToolCall {
1305
1440
  }
1306
1441
  export interface Progress {
1307
1442
  turns: number;
1443
+ /** Bounded recent calls for status rendering. */
1308
1444
  toolCalls: ToolCall[];
1445
+ /** Exact count, independent of the bounded recent-call ring. */
1446
+ toolCallCount: number;
1309
1447
  finalText: string;
1310
1448
  /** Best-effort text from the currently streaming assistant message. */
1311
1449
  streamingText: string;
@@ -1320,6 +1458,10 @@ export interface Progress {
1320
1458
  cacheWriteTokens: number;
1321
1459
  reasoningTokens: number;
1322
1460
  cost: number;
1461
+ inputCost: number;
1462
+ outputCost: number;
1463
+ cacheReadCost: number;
1464
+ cacheWriteCost: number;
1323
1465
  errorMsg?: string;
1324
1466
  // RPC lifecycle bookkeeping (computed over the analyzed log slice):
1325
1467
  agentStarts: number;
@@ -1362,6 +1504,7 @@ function emptyProgress(): Progress {
1362
1504
  return {
1363
1505
  turns: 0,
1364
1506
  toolCalls: [],
1507
+ toolCallCount: 0,
1365
1508
  finalText: "",
1366
1509
  streamingText: "",
1367
1510
  modelCalls: 0,
@@ -1372,6 +1515,10 @@ function emptyProgress(): Progress {
1372
1515
  cacheWriteTokens: 0,
1373
1516
  reasoningTokens: 0,
1374
1517
  cost: 0,
1518
+ inputCost: 0,
1519
+ outputCost: 0,
1520
+ cacheReadCost: 0,
1521
+ cacheWriteCost: 0,
1375
1522
  agentStarts: 0,
1376
1523
  agentEnds: 0,
1377
1524
  agentSettled: 0,
@@ -1393,8 +1540,8 @@ export function subagentBudgetViolation(
1393
1540
  if (budget.maxTurns != null && progress.turns >= budget.maxTurns) {
1394
1541
  return `${progress.turns} turns reached maxTurns ${budget.maxTurns}`;
1395
1542
  }
1396
- if (budget.maxToolCalls != null && progress.toolCalls.length >= budget.maxToolCalls) {
1397
- return `${progress.toolCalls.length} tool calls reached maxToolCalls ${budget.maxToolCalls}`;
1543
+ if (budget.maxToolCalls != null && progress.toolCallCount >= budget.maxToolCalls) {
1544
+ return `${progress.toolCallCount} tool calls reached maxToolCalls ${budget.maxToolCalls}`;
1398
1545
  }
1399
1546
  if (budget.maxUsageTokens != null && progress.usageTokens >= budget.maxUsageTokens) {
1400
1547
  return `${progress.usageTokens} usage tokens reached maxUsageTokens ${budget.maxUsageTokens}`;
@@ -1402,6 +1549,33 @@ export function subagentBudgetViolation(
1402
1549
  return null;
1403
1550
  }
1404
1551
 
1552
+ export function subagentBudgetSoftViolation(
1553
+ progress: Progress,
1554
+ budget?: SubagentBudget,
1555
+ ratio = 0.8,
1556
+ ): string | null {
1557
+ if (!budget || ratio <= 0 || ratio >= 1) return null;
1558
+ if (budget.maxCost != null && progress.cost >= budget.maxCost * ratio) {
1559
+ return `cost $${progress.cost.toFixed(4)} reached ${Math.round(ratio * 100)}% of maxCost $${budget.maxCost.toFixed(4)}`;
1560
+ }
1561
+ if (budget.maxTurns != null && progress.turns >= Math.max(1, Math.ceil(budget.maxTurns * ratio))) {
1562
+ return `${progress.turns} turns reached ${Math.round(ratio * 100)}% of maxTurns ${budget.maxTurns}`;
1563
+ }
1564
+ if (
1565
+ budget.maxToolCalls != null &&
1566
+ progress.toolCallCount >= Math.max(1, Math.ceil(budget.maxToolCalls * ratio))
1567
+ ) {
1568
+ return `${progress.toolCallCount} tool calls reached ${Math.round(ratio * 100)}% of maxToolCalls ${budget.maxToolCalls}`;
1569
+ }
1570
+ if (
1571
+ budget.maxUsageTokens != null &&
1572
+ progress.usageTokens >= Math.max(1, Math.ceil(budget.maxUsageTokens * ratio))
1573
+ ) {
1574
+ return `${progress.usageTokens} usage tokens reached ${Math.round(ratio * 100)}% of maxUsageTokens ${budget.maxUsageTokens}`;
1575
+ }
1576
+ return null;
1577
+ }
1578
+
1405
1579
  export function subagentBudgetAction(
1406
1580
  progress: Progress,
1407
1581
  budget: SubagentBudget | undefined,
@@ -1449,16 +1623,19 @@ function parseEventLine(progress: Progress, raw: string): void {
1449
1623
  | { type?: string; delta?: string }
1450
1624
  | undefined;
1451
1625
  if (update?.type === "text_delta" && typeof update.delta === "string") {
1452
- progress.streamingText += update.delta;
1626
+ progress.streamingText = clip(progress.streamingText + update.delta, ANSWER_MAX_BYTES);
1453
1627
  }
1454
1628
  break;
1455
1629
  }
1456
1630
  case "tool_execution_start": {
1457
1631
  const name = String(event.toolName ?? "tool");
1632
+ progress.toolCallCount++;
1458
1633
  progress.toolCalls.push({
1459
1634
  name,
1460
1635
  summary: summarizeToolCall(name, (event.args as Record<string, unknown>) ?? {}),
1461
1636
  });
1637
+ // Open-ended workers must not retain an unbounded tool history in Pi.
1638
+ if (progress.toolCalls.length > 200) progress.toolCalls.splice(0, progress.toolCalls.length - 200);
1462
1639
  break;
1463
1640
  }
1464
1641
  case "message_end": {
@@ -1466,6 +1643,8 @@ function parseEventLine(progress: Progress, raw: string): void {
1466
1643
  | {
1467
1644
  role?: string;
1468
1645
  content?: { type: string; text?: string }[];
1646
+ stopReason?: string;
1647
+ errorMessage?: string;
1469
1648
  usage?: {
1470
1649
  input?: number;
1471
1650
  output?: number;
@@ -1473,7 +1652,13 @@ function parseEventLine(progress: Progress, raw: string): void {
1473
1652
  cacheWrite?: number;
1474
1653
  reasoning?: number;
1475
1654
  totalTokens?: number;
1476
- cost?: { total?: number };
1655
+ cost?: {
1656
+ input?: number;
1657
+ output?: number;
1658
+ cacheRead?: number;
1659
+ cacheWrite?: number;
1660
+ total?: number;
1661
+ };
1477
1662
  };
1478
1663
  }
1479
1664
  | undefined;
@@ -1482,7 +1667,10 @@ function parseEventLine(progress: Progress, raw: string): void {
1482
1667
  .filter((content) => content.type === "text" && content.text)
1483
1668
  .map((content) => content.text)
1484
1669
  .join("");
1485
- if (text.trim()) progress.finalText = text;
1670
+ if (text.trim()) progress.finalText = clip(text, ANSWER_MAX_BYTES);
1671
+ if (message.stopReason === "error") {
1672
+ progress.errorMsg = message.errorMessage || "subagent model request failed";
1673
+ }
1486
1674
  progress.streamingText = "";
1487
1675
  if (message.usage) {
1488
1676
  const finite = (value: number | undefined) =>
@@ -1495,6 +1683,10 @@ function parseEventLine(progress: Progress, raw: string): void {
1495
1683
  progress.cacheReadTokens += finite(message.usage.cacheRead);
1496
1684
  progress.cacheWriteTokens += finite(message.usage.cacheWrite);
1497
1685
  progress.reasoningTokens += finite(message.usage.reasoning);
1686
+ progress.inputCost += finite(message.usage.cost?.input);
1687
+ progress.outputCost += finite(message.usage.cost?.output);
1688
+ progress.cacheReadCost += finite(message.usage.cost?.cacheRead);
1689
+ progress.cacheWriteCost += finite(message.usage.cost?.cacheWrite);
1498
1690
  progress.cost += finite(message.usage.cost?.total);
1499
1691
  }
1500
1692
  }
@@ -1925,6 +2117,33 @@ async function spawnSubagent(
1925
2117
  await bs(["kill", "-s", id]);
1926
2118
  return { error: `subagent ${id} rejected the task: ${resp.error}` };
1927
2119
  }
2120
+ // Prompt acceptance does not guarantee provider authentication: Pi reports
2121
+ // failures that occur after acceptance through the event stream. Probe a short
2122
+ // window so immediate missing-key/config errors fail the spawn instead of
2123
+ // creating a zero-work worker that the caller must discover later.
2124
+ const startupProbe = await bs([
2125
+ "expect",
2126
+ "-s",
2127
+ id,
2128
+ "--since",
2129
+ String(resp.offset),
2130
+ "--timeout",
2131
+ "500ms",
2132
+ '(?m)^\\{"type":"(?:message_end|error|extension_error|agent_settled)"',
2133
+ ]);
2134
+ if (startupProbe.code === 0) {
2135
+ try {
2136
+ const window = readLogWindowFrom(logPath(id), resp.offset);
2137
+ const initialProgress = parseEvents(window.bytes.toString("utf8"));
2138
+ if (initialProgress.errorMsg && initialProgress.modelCalls === 0) {
2139
+ discardDelivery(delivery);
2140
+ await bs(["kill", "-s", id]);
2141
+ return { error: `subagent ${id} failed before its first model response: ${initialProgress.errorMsg}` };
2142
+ }
2143
+ } catch {
2144
+ /* normal startup continues; the full stream remains available to check/wait */
2145
+ }
2146
+ }
1928
2147
  // Report the RESOLVED model (a fuzzy pattern may match something unexpected;
1929
2148
  // null means nothing resolved at all).
1930
2149
  let resolvedModel: string | undefined;
@@ -1985,8 +2204,10 @@ const WIDGET_TAIL_WIDTH = 100;
1985
2204
  // Strip ANSI/control escapes and clamp width so raw PTY output can't wrap or
1986
2205
  // corrupt the widget area.
1987
2206
  function sanitizeTailLine(s: string): string {
1988
- const clean = s
1989
- .replace(/\r/g, "")
2207
+ // PTY progress bars often redraw one logical line with carriage returns.
2208
+ // Keep the latest frame rather than concatenating every historical frame.
2209
+ const terminalFrame = s.split("\r").filter(Boolean).at(-1) ?? "";
2210
+ const clean = terminalFrame
1990
2211
  // CSI / OSC / other escape sequences
1991
2212
  .replace(/\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)/g, "")
1992
2213
  .replace(/\x1b[@-Z\\-_]|\x1b\[[0-?]*[ -/]*[@-~]/g, "")
@@ -1995,9 +2216,34 @@ function sanitizeTailLine(s: string): string {
1995
2216
  return clean.length > WIDGET_TAIL_WIDTH ? `${clean.slice(0, WIDGET_TAIL_WIDTH - 1)}…` : clean;
1996
2217
  }
1997
2218
 
2219
+ function readTailLines(file: string, lines: number, maxBytes = 64_000): string[] {
2220
+ try {
2221
+ const size = fs.statSync(file).size;
2222
+ const start = Math.max(0, size - maxBytes);
2223
+ const length = size - start;
2224
+ const bytes = Buffer.allocUnsafe(length);
2225
+ const fd = fs.openSync(file, "r");
2226
+ let read = 0;
2227
+ try {
2228
+ while (read < length) {
2229
+ const count = fs.readSync(fd, bytes, read, length - read, start + read);
2230
+ if (count === 0) break;
2231
+ read += count;
2232
+ }
2233
+ } finally {
2234
+ fs.closeSync(fd);
2235
+ }
2236
+ const parts = bytes.subarray(0, read).toString("utf8").split("\n");
2237
+ if (start > 0) parts.shift(); // first fragment may begin mid-line
2238
+ return parts.slice(-Math.max(1, lines + 1));
2239
+ } catch {
2240
+ return [];
2241
+ }
2242
+ }
2243
+
1998
2244
  // Trailing lines to show for a running session (sanitized, unprefixed).
1999
- // process → raw log tail; subagent → derived activity (recent tool calls /
2000
- // partial answer), since its log is RPC JSONL, not human-readable.
2245
+ // Process tails are read directly from the bounded end of output.log, avoiding
2246
+ // one `babysit log` subprocess per active process on every poll.
2001
2247
  async function widgetTail(
2002
2248
  id: string,
2003
2249
  isSub: boolean,
@@ -2005,8 +2251,7 @@ async function widgetTail(
2005
2251
  ): Promise<string[]> {
2006
2252
  let raw: string[];
2007
2253
  if (!isSub) {
2008
- const out = (await bs(["log", "-s", id, "--tail", String(WIDGET_TAIL_LINES)])).stdout;
2009
- raw = out.split("\n");
2254
+ raw = readTailLines(logPath(id), WIDGET_TAIL_LINES);
2010
2255
  } else {
2011
2256
  const progress = subagentProgress ?? taskProgressOf(id).progress;
2012
2257
  if (progress.finalText.trim()) {
@@ -2047,6 +2292,90 @@ interface WaitOutcome {
2047
2292
  progress?: Progress;
2048
2293
  }
2049
2294
 
2295
+ export interface NestedUsage {
2296
+ input: number;
2297
+ output: number;
2298
+ cacheRead: number;
2299
+ cacheWrite: number;
2300
+ totalTokens: number;
2301
+ cost: {
2302
+ input: number;
2303
+ output: number;
2304
+ cacheRead: number;
2305
+ cacheWrite: number;
2306
+ total: number;
2307
+ };
2308
+ }
2309
+
2310
+ export function usageFromProgress(progress: Progress): NestedUsage | undefined {
2311
+ if (progress.modelCalls === 0) return undefined;
2312
+ return {
2313
+ input: progress.inputTokens,
2314
+ output: progress.outputTokens,
2315
+ cacheRead: progress.cacheReadTokens,
2316
+ cacheWrite: progress.cacheWriteTokens,
2317
+ totalTokens: progress.usageTokens,
2318
+ cost: {
2319
+ input: progress.inputCost,
2320
+ output: progress.outputCost,
2321
+ cacheRead: progress.cacheReadCost,
2322
+ cacheWrite: progress.cacheWriteCost,
2323
+ total: progress.cost,
2324
+ },
2325
+ };
2326
+ }
2327
+
2328
+ /** Charge one completed task exactly once, even across concurrent wait callers. */
2329
+ function claimOutcomeUsage(outcome: WaitOutcome): NestedUsage | undefined {
2330
+ if (!outcome.progress || (outcome.kind !== "done" && outcome.kind !== "exited")) return undefined;
2331
+ const usage = usageFromProgress(outcome.progress);
2332
+ if (!usage) return undefined;
2333
+ const meta = readMeta(outcome.id);
2334
+ if (meta?.kind !== "subagent") return undefined;
2335
+ const offset = meta.promptOffset ?? 0;
2336
+ if (meta.usageReportedOffset === offset) return undefined;
2337
+
2338
+ // `open(..., "wx")` is the cross-process compare-and-set. A resumed Pi
2339
+ // session can briefly have overlapping extension processes; metadata alone
2340
+ // would let both read the old value and charge the same nested usage.
2341
+ const marker = path.join(metaDir(), `${outcome.id}.usage-${offset}.claimed`);
2342
+ if (!claimFileOnce(marker, JSON.stringify({ pid: process.pid, claimedAt: Date.now() }))) {
2343
+ return undefined;
2344
+ }
2345
+ meta.usageReportedOffset = offset;
2346
+ writeMeta(outcome.id, meta); // compatibility/display hint; marker is authoritative
2347
+ return usage;
2348
+ }
2349
+
2350
+ function sumNestedUsage(values: Array<NestedUsage | undefined>): NestedUsage | undefined {
2351
+ const present = values.filter((value): value is NestedUsage => Boolean(value));
2352
+ if (present.length === 0) return undefined;
2353
+ return present.reduce<NestedUsage>(
2354
+ (total, value) => ({
2355
+ input: total.input + value.input,
2356
+ output: total.output + value.output,
2357
+ cacheRead: total.cacheRead + value.cacheRead,
2358
+ cacheWrite: total.cacheWrite + value.cacheWrite,
2359
+ totalTokens: total.totalTokens + value.totalTokens,
2360
+ cost: {
2361
+ input: total.cost.input + value.cost.input,
2362
+ output: total.cost.output + value.cost.output,
2363
+ cacheRead: total.cost.cacheRead + value.cost.cacheRead,
2364
+ cacheWrite: total.cost.cacheWrite + value.cost.cacheWrite,
2365
+ total: total.cost.total + value.cost.total,
2366
+ },
2367
+ }),
2368
+ {
2369
+ input: 0,
2370
+ output: 0,
2371
+ cacheRead: 0,
2372
+ cacheWrite: 0,
2373
+ totalTokens: 0,
2374
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
2375
+ },
2376
+ );
2377
+ }
2378
+
2050
2379
  // Wait for ONE subagent's current task. Completion = agent_settled without a
2051
2380
  // parked babysit_run/process result (a parked run only means "waiting for
2052
2381
  // process exit; pi will resume itself"). Parse appended bytes incrementally,
@@ -2069,7 +2398,7 @@ async function waitForTask(
2069
2398
  const st = await statusOf(id);
2070
2399
 
2071
2400
  const stats =
2072
- `turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCalls.length}` +
2401
+ `turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}` +
2073
2402
  (prog.tokens != null ? ` ctx=${prog.tokens}` : "") +
2074
2403
  (prog.modelCalls > 0
2075
2404
  ? ` usage=${prog.usageTokens} (in=${prog.inputTokens} out=${prog.outputTokens} cache=${prog.cacheReadTokens}) $${prog.cost.toFixed(4)}`
@@ -2083,7 +2412,7 @@ async function waitForTask(
2083
2412
  ok: completed.ok,
2084
2413
  text:
2085
2414
  `Subagent ${id} finished its task (${stats}).\n` +
2086
- `Session stays alive — follow-up: babysit_send { id: "${id}" }, ` +
2415
+ `${SUBAGENT_REUSE_HINT} Follow-up: babysit_send { id: "${id}" }, ` +
2087
2416
  `or babysit_kill when done.\n\n${completed.body}`,
2088
2417
  status: st,
2089
2418
  progress: prog,
@@ -2149,16 +2478,27 @@ async function waitForTask(
2149
2478
  }
2150
2479
  }
2151
2480
 
2152
- // Mark a process session as already-reported so the exit-notification poller
2153
- // doesn't send a duplicate message for something the agent just observed.
2481
+ // Mark a session as already reported so completion pollers do not send a
2482
+ // duplicate message for something the agent just observed.
2154
2483
  function suppressNotify(id: string, reason: "observed" | "kill" = "observed"): void {
2155
2484
  const meta = readMeta(id);
2156
- if (meta && meta.kind === "process") {
2485
+ if (!meta) return;
2486
+ if (meta.kind === "process") {
2157
2487
  meta.notified = true;
2158
2488
  if (reason === "kill") meta.killNotificationSuppressed = true;
2159
2489
  delete meta.notificationPaused;
2160
- writeMeta(id, meta);
2490
+ } else {
2491
+ meta.subagentCollectedOffset = meta.promptOffset ?? 0;
2161
2492
  }
2493
+ writeMeta(id, meta);
2494
+ }
2495
+
2496
+ function collectSubagentOutcome(outcome: WaitOutcome): void {
2497
+ if (outcome.kind !== "done" && outcome.kind !== "exited") return;
2498
+ const meta = readMeta(outcome.id);
2499
+ if (!meta || meta.kind !== "subagent") return;
2500
+ meta.subagentCollectedOffset = meta.promptOffset ?? 0;
2501
+ writeMeta(outcome.id, meta);
2162
2502
  }
2163
2503
 
2164
2504
  export interface WaitReservationState {
@@ -2227,6 +2567,7 @@ async function waitForExit(
2227
2567
  limitMs: number | null,
2228
2568
  signal?: AbortSignal,
2229
2569
  expectPattern?: string,
2570
+ outputSelection?: ProcessOutputSelection,
2230
2571
  ): Promise<WaitOutcome> {
2231
2572
  const t = limitMs != null ? `${Math.ceil(limitMs / 1000)}s` : "0";
2232
2573
 
@@ -2265,10 +2606,23 @@ async function waitForExit(
2265
2606
  // one concurrent wait timing out cannot re-enable notifications underneath
2266
2607
  // another wait that is still pending.
2267
2608
  updateWaitReservation(id, "reserve");
2268
- const w = await bs(["wait", "-s", id, "--timeout", t], { signal });
2269
- if (signal?.aborted) {
2270
- updateWaitReservation(id, "abandon");
2271
- return { id, kind: "interrupted", ok: false, text: `wait for ${id} was interrupted.` };
2609
+ let w: Awaited<ReturnType<typeof bs>>;
2610
+ let attempt = 0;
2611
+ for (;;) {
2612
+ w = await bs(["wait", "-s", id, "--timeout", t], { signal });
2613
+ if (signal?.aborted) {
2614
+ updateWaitReservation(id, "abandon");
2615
+ return { id, kind: "interrupted", ok: false, text: `wait for ${id} was interrupted.` };
2616
+ }
2617
+ if (w.code === 0 || w.code === 124 || w.code === 130) break;
2618
+
2619
+ // A freshly spawned session can be visible in `list` before the backend's
2620
+ // wait endpoint is ready, especially when sibling foreground tools start
2621
+ // concurrently. Retry that transient startup race instead of reporting the
2622
+ // still-running child as "exited with code ?".
2623
+ const retryStatus = await statusOf(id);
2624
+ if (retryStatus?.state !== "running" || attempt++ >= 3) break;
2625
+ await new Promise((resolve) => setTimeout(resolve, 50 * attempt));
2272
2626
  }
2273
2627
  if (w.code === 130) {
2274
2628
  const interruptedStatus = await statusOf(id);
@@ -2310,12 +2664,24 @@ async function waitForExit(
2310
2664
  if (!expectPattern) updateWaitReservation(id, "claim");
2311
2665
  return { id, kind: "exited", ok: false, text: `No such session: ${id}` };
2312
2666
  }
2667
+ if (st.state === "running") {
2668
+ if (!expectPattern) updateWaitReservation(id, "abandon");
2669
+ return {
2670
+ id,
2671
+ kind: "interrupted",
2672
+ ok: false,
2673
+ text:
2674
+ `The wait backend returned before process ${id} exited; the process is still running. ` +
2675
+ `Use babysit_wait to continue waiting.\nLog: ${logPath(id)}`,
2676
+ status: st,
2677
+ };
2678
+ }
2313
2679
  if (expectPattern) suppressNotify(id);
2314
2680
  else updateWaitReservation(id, "claim"); // the agent sees the exit here; don't notify again
2315
2681
  const meta = readMeta(id);
2316
2682
  const workerDead = st.state === "dead" && st.exit_code == null;
2317
2683
  const ok = st.exit_code === 0;
2318
- const output = await inlineOutput(id, st);
2684
+ const output = await selectedProcessOutput(id, st, outputSelection, signal);
2319
2685
  return {
2320
2686
  id,
2321
2687
  kind: "exited",
@@ -2405,6 +2771,21 @@ export function automaticNotificationGroup(entry: unknown): string | undefined {
2405
2771
  return runs.length >= 2 ? group : undefined;
2406
2772
  }
2407
2773
 
2774
+ export function resolveSubagentSendMode(
2775
+ requested: "auto" | "steer" | "task",
2776
+ streaming?: boolean,
2777
+ currentTaskDone?: boolean,
2778
+ ): { mode: "steer" | "task" } | { error: "busy" | "unknown" | "unsettled" } {
2779
+ if (requested === "steer") return { mode: "steer" };
2780
+ if (requested === "auto") {
2781
+ return { mode: streaming === false && currentTaskDone === true ? "task" : "steer" };
2782
+ }
2783
+ if (streaming === true) return { error: "busy" };
2784
+ if (streaming === undefined || currentTaskDone === undefined) return { error: "unknown" };
2785
+ if (!currentTaskDone) return { error: "unsettled" };
2786
+ return { mode: "task" };
2787
+ }
2788
+
2408
2789
  // ---------------------------------------------------------------------------
2409
2790
  // extension
2410
2791
  // ---------------------------------------------------------------------------
@@ -2459,76 +2840,88 @@ export default function (pi: ExtensionAPI) {
2459
2840
  });
2460
2841
 
2461
2842
  async function enforceSubagentBudgets(sessions: BsSession[]): Promise<void> {
2462
- for (const session of sessions) {
2463
- if (session.state !== "running") continue;
2464
- await withSessionRpcLock(session.id, async () => {
2465
- const meta = readMeta(session.id);
2466
- if (meta?.kind !== "subagent" || !meta.budget || meta.budgetKilled) return;
2467
-
2468
- let progress: Progress;
2469
- try {
2470
- progress = taskProgressOf(session.id).progress;
2471
- } catch {
2472
- return;
2473
- }
2474
- if (progress.done) return;
2475
- const now = Date.now();
2476
- const decision = subagentBudgetAction(
2477
- progress,
2478
- meta.budget,
2479
- meta.budgetExceededAt,
2480
- now,
2481
- SUBAGENT_BUDGET_GRACE_MS,
2482
- );
2483
- if (decision.action === "none") return;
2484
- const reason = decision.reason as string;
2485
-
2486
- // The poll already supplied a running-session snapshot. Revalidate only
2487
- // when a limit requires action; checking every healthy worker made each
2488
- // interval launch N extra `babysit list` subprocesses.
2489
- const latestStatus = await statusOf(session.id);
2490
- if (latestStatus?.state !== "running") return;
2491
-
2492
- if (decision.action === "steer") {
2493
- // Start the grace period only after Pi accepts the steering command.
2494
- const sent = await sendRpc(session.id, {
2495
- type: "steer",
2496
- message: `Budget reached (${reason}). Stop calling tools and provide your final answer now.`,
2497
- });
2498
- if ("error" in sent) return;
2499
- const accepted = await rpcResponse(session.id, sent.offset, "steer", "3s");
2500
- if (!accepted.ok) return;
2501
- const current = readMeta(session.id);
2502
- if (
2503
- current?.kind !== "subagent" ||
2504
- current.promptOffset !== meta.promptOffset ||
2505
- current.budgetExceededAt
2506
- ) {
2507
- return;
2508
- }
2509
- current.budgetExceededAt = now;
2510
- current.budgetReason = reason;
2511
- writeMeta(session.id, current);
2512
- return;
2513
- }
2843
+ // Independent workers must not serialize 3-second RPC probes and delay
2844
+ // unrelated completion notifications. Per-session RPC locks still preserve
2845
+ // ordering within each worker.
2846
+ await Promise.all(
2847
+ sessions
2848
+ .filter((session) => session.state === "running")
2849
+ .map((session) =>
2850
+ withSessionRpcLock(session.id, async () => {
2851
+ const meta = readMeta(session.id);
2852
+ if (meta?.kind !== "subagent" || !meta.budget || meta.budgetKilled) return;
2853
+
2854
+ let progress: Progress;
2855
+ try {
2856
+ progress = taskProgressOf(session.id).progress;
2857
+ } catch {
2858
+ return;
2859
+ }
2860
+ if (progress.done) return;
2861
+ const now = Date.now();
2862
+ const hardReason = subagentBudgetViolation(progress, meta.budget);
2863
+ const softReason = subagentBudgetSoftViolation(progress, meta.budget);
2864
+
2865
+ if (hardReason) {
2866
+ if (meta.budgetExceededAt == null) {
2867
+ // The hard grace begins when the violation is observed, even if a
2868
+ // wedged worker never accepts steering. This makes the cap enforceable.
2869
+ meta.budgetExceededAt = now;
2870
+ meta.budgetReason = hardReason;
2871
+ writeMeta(session.id, meta);
2872
+ const latestStatus = await statusOf(session.id);
2873
+ if (latestStatus?.state !== "running") return;
2874
+ const sent = await sendRpc(session.id, {
2875
+ type: "steer",
2876
+ message: `Hard budget reached (${hardReason}). Stop calling tools and return your best answer now.`,
2877
+ });
2878
+ if (!("error" in sent)) await rpcResponse(session.id, sent.offset, "steer", "3s");
2879
+ return;
2880
+ }
2881
+ if (now - meta.budgetExceededAt < SUBAGENT_BUDGET_GRACE_MS) return;
2882
+ const latestStatus = await statusOf(session.id);
2883
+ if (latestStatus?.state !== "running") return;
2884
+ const killed = await bs(["kill", "-s", session.id, "--json"]);
2885
+ if (killed.code !== 0) return;
2886
+ const terminal = await awaitConfirmedTermination(session.id);
2887
+ const current = readMeta(session.id);
2888
+ if (
2889
+ terminal &&
2890
+ isConfirmedTerminalState(terminal.state) &&
2891
+ current?.kind === "subagent" &&
2892
+ current.promptOffset === meta.promptOffset &&
2893
+ current.budgetExceededAt === meta.budgetExceededAt
2894
+ ) {
2895
+ current.budgetKilled = true;
2896
+ current.budgetReason = current.budgetReason ?? hardReason;
2897
+ writeMeta(session.id, current);
2898
+ }
2899
+ return;
2900
+ }
2514
2901
 
2515
- const killed = await bs(["kill", "-s", session.id, "--json"]);
2516
- if (killed.code !== 0) return;
2517
- const terminal = await awaitConfirmedTermination(session.id);
2518
- const current = readMeta(session.id);
2519
- if (
2520
- terminal &&
2521
- isConfirmedTerminalState(terminal.state) &&
2522
- current?.kind === "subagent" &&
2523
- current.promptOffset === meta.promptOffset &&
2524
- current.budgetExceededAt === meta.budgetExceededAt
2525
- ) {
2526
- current.budgetKilled = true;
2527
- current.budgetReason = current.budgetReason ?? reason;
2528
- writeMeta(session.id, current);
2529
- }
2530
- });
2531
- }
2902
+ if (!softReason || meta.budgetWarnedAt != null) return;
2903
+ const latestStatus = await statusOf(session.id);
2904
+ if (latestStatus?.state !== "running") return;
2905
+ const sent = await sendRpc(session.id, {
2906
+ type: "steer",
2907
+ message: `Budget is approaching its limit (${softReason}). Wrap up now and preserve your best findings.`,
2908
+ });
2909
+ if ("error" in sent) return;
2910
+ const accepted = await rpcResponse(session.id, sent.offset, "steer", "3s");
2911
+ if (!accepted.ok) return;
2912
+ const current = readMeta(session.id);
2913
+ if (
2914
+ current?.kind === "subagent" &&
2915
+ current.promptOffset === meta.promptOffset &&
2916
+ current.budgetWarnedAt == null
2917
+ ) {
2918
+ current.budgetWarnedAt = now;
2919
+ current.budgetWarningReason = softReason;
2920
+ writeMeta(session.id, current);
2921
+ }
2922
+ }),
2923
+ ),
2924
+ );
2532
2925
  }
2533
2926
 
2534
2927
  // Exit notifications for kind=process sessions: the poller detects
@@ -2628,6 +3021,64 @@ export default function (pi: ExtensionAPI) {
2628
3021
  );
2629
3022
  }
2630
3023
 
3024
+ async function notifySettledSubagents(
3025
+ ctx: ExtensionContext,
3026
+ snapshot?: BsSession[],
3027
+ ): Promise<void> {
3028
+ if (shouldDeferCompletionNotification(ctx.isIdle())) return;
3029
+ const sessions = snapshot ?? (await listSessions()).sessions;
3030
+ const ready: Array<{ id: string; offset: number; state: string; summary: string }> = [];
3031
+ for (const session of sessions) {
3032
+ const meta = readMeta(session.id);
3033
+ if (!shouldDeliverSubagentCompletion(meta)) continue;
3034
+ let progress: Progress;
3035
+ try {
3036
+ progress = taskProgressOf(session.id).progress;
3037
+ } catch {
3038
+ progress = emptyProgress();
3039
+ }
3040
+ if (session.state === "running" && !progress.done) continue;
3041
+
3042
+ const offset = meta.promptOffset ?? 0;
3043
+ if (meta.subagentCompletionObservedOffset !== offset) {
3044
+ meta.subagentCompletionObservedOffset = offset;
3045
+ meta.subagentCompletionObservedAt = Date.now();
3046
+ writeMeta(session.id, meta);
3047
+ continue;
3048
+ }
3049
+ if (Date.now() - (meta.subagentCompletionObservedAt ?? 0) < POLL_MS) continue;
3050
+ const summary = progress.done
3051
+ ? `task settled; ${progress.turns} turns, ${progress.toolCallCount} tools, $${progress.cost.toFixed(4)}`
3052
+ : `worker ${session.state} with exit code ${session.exit_code ?? "?"}; partial usage $${progress.cost.toFixed(4)}`;
3053
+ ready.push({ id: session.id, offset, state: session.state, summary });
3054
+ }
3055
+ if (ready.length === 0 || shouldDeferCompletionNotification(ctx.isIdle())) return;
3056
+
3057
+ const deliverable = ready.filter(({ id, offset }) => {
3058
+ const meta = readMeta(id);
3059
+ return shouldDeliverSubagentCompletion(meta) && (meta.promptOffset ?? 0) === offset;
3060
+ });
3061
+ if (deliverable.length === 0) return;
3062
+ pi.sendMessage(
3063
+ {
3064
+ customType: "pi-babysit-subagent-ready",
3065
+ content:
3066
+ `${deliverable.length === 1 ? "A background subagent is" : `${deliverable.length} background subagents are`} ready to collect:\n` +
3067
+ deliverable.map(({ id, summary }) => `- ${id}: ${summary}`).join("\n") +
3068
+ "\nCall babysit_wait now to retrieve the answer and charge nested usage before finishing the parent task.",
3069
+ display: true,
3070
+ details: { subagents: deliverable.map(({ id, state }) => ({ id, state })) },
3071
+ },
3072
+ { triggerTurn: true, deliverAs: "steer" },
3073
+ );
3074
+ for (const { id, offset } of deliverable) {
3075
+ const meta = readMeta(id);
3076
+ if (!meta || meta.kind !== "subagent" || (meta.promptOffset ?? 0) !== offset) continue;
3077
+ meta.subagentNotifiedOffset = offset;
3078
+ writeMeta(id, meta);
3079
+ }
3080
+ }
3081
+
2631
3082
  const refreshWidget = async (ctx: ExtensionContext, snapshot?: BsSession[]) => {
2632
3083
  if (!ctx.hasUI) return;
2633
3084
  const active = (snapshot ?? (await listSessions()).sessions).filter(
@@ -2775,9 +3226,15 @@ export default function (pi: ExtensionAPI) {
2775
3226
  }
2776
3227
  }
2777
3228
  taskProgressCache.clear();
3229
+ searchLogCache.clear();
2778
3230
  pollNeeded = true;
2779
- const retentionDays = Number(process.env.PI_BABYSIT_RETENTION_DAYS);
2780
- if (!automaticGcRan && Number.isFinite(retentionDays) && retentionDays > 0) {
3231
+ const retentionDays = Number(process.env.PI_BABYSIT_RETENTION_DAYS ?? "3");
3232
+ if (
3233
+ !automaticGcRan &&
3234
+ Number.isFinite(retentionDays) &&
3235
+ retentionDays > 0 &&
3236
+ automaticGcDue()
3237
+ ) {
2781
3238
  automaticGcRan = true;
2782
3239
  const gc = gcBabysitRoots({
2783
3240
  rootBase: ROOT_BASE,
@@ -2785,6 +3242,7 @@ export default function (pi: ExtensionAPI) {
2785
3242
  olderThanMs: retentionDays * 86_400_000,
2786
3243
  dryRun: false,
2787
3244
  });
3245
+ markAutomaticGc();
2788
3246
  if (gc.deleted.length > 0 && ctx.hasUI) {
2789
3247
  ctx.ui.notify(
2790
3248
  `pi-babysit GC removed ${gc.deleted.length} roots (${gc.bytes} bytes).`,
@@ -2814,6 +3272,7 @@ export default function (pi: ExtensionAPI) {
2814
3272
  await enforceSubagentBudgets(snapshot);
2815
3273
  await Promise.all([
2816
3274
  notifyEndedProcesses(ctx, snapshot),
3275
+ notifySettledSubagents(ctx, snapshot),
2817
3276
  refreshWidget(ctx, snapshot),
2818
3277
  ]);
2819
3278
  pruneTerminalSessionCache(taskProgressCache, snapshot);
@@ -2880,35 +3339,19 @@ export default function (pi: ExtensionAPI) {
2880
3339
  name: "babysit_run",
2881
3340
  label: "Babysit: run",
2882
3341
  description:
2883
- "Run any shell command in a supervised babysit session. Set `foreground: true` when the next step " +
2884
- "needs the exit result in this tool call. Otherwise commands that finish within a short grace period " +
2885
- "return completion metadata immediately; longer commands continue in the background and trigger an " +
2886
- "automatic notification on exit. Sibling background runs in one assistant message are grouped " +
2887
- "automatically. Complete output is returned inline only when it is small; larger output stays " +
2888
- "in the log path for bounded inspection with babysit_check. " +
2889
- "In non-interactive mode (`pi -p`, no UI), process mode blocks until exit because there is no " +
2890
- "notification loop. Two modes: (1) `command` — run any shell command, including builds, tests, " +
2891
- "dev servers, watchers, and interactive TUIs; you can type into it with babysit_send and read " +
2892
- "its screen with babysit_check. If a worker disappears during startup without recording an exit, " +
2893
- "`retryOnWorkerDeath` can retry one idempotent command once. " +
2894
- "(2) `profile: \"subagent\"` + `task` — spawn a pi subagent that works on the task in the " +
2895
- "background; poll with babysit_check, steer with babysit_send, block with babysit_wait, " +
2896
- "stop with babysit_kill. Subagents cannot recursively spawn more subagents by default; " +
2897
- "the top-level caller must explicitly raise `maxDepth` when creating the first worker.",
2898
- promptSnippet:
2899
- "Run any shell command with context-safe captured output; quick commands return metadata, longer ones continue in background",
3342
+ "Run a supervised shell command, or start a reusable pi subagent with `profile: \"subagent\"`. " +
3343
+ "Use `foreground` for results needed now; otherwise long commands notify on exit. Full logs stay on disk. " +
3344
+ "`returnPattern`/`returnLines` bound foreground output. Sessions support check, wait, send, and kill.",
3345
+ promptSnippet: "Run supervised commands or bounded pi subagents with context-safe logs",
2900
3346
  promptGuidelines: [
2901
- "Use babysit_run as the default for shell commands, not only long-running work. Small output is returned directly; large stdout/stderr stays out of model context in the returned log path. Give every meaningful process or subagent a clear stable `name`.",
2902
- "Use babysit_run { command, foreground: true } when the result is required before the next step; this avoids a separate babysit_wait model turn. Do not use foreground for servers, watchers, or commands of unknown duration without a timeout.",
2903
- "Bundle closely related tiny observations into one babysit_run command when that reduces tool turns without obscuring lifecycle or failure handling.",
2904
- "Inspect a babysit log with babysit_check { id, lines, pattern? }; never read or cat a potentially large log file in full. Prefer a targeted `pattern` search over returning a broad tail.",
2905
- "After babysit_run { command } starts a process, end your response immediately so the automatic process-end notification can resume you; NEVER poll with babysit_check or sleep. Set continueAfterStart: true only when you have immediate, specific, non-polling work to do next. Call babysit_wait when you must consume the result inside the current turn (optionally with `expect` to wait for a readiness line like 'listening on').",
2906
- "If a babysit worker is killed externally, babysit_run reports it as worker-dead rather than hanging. Set retryOnWorkerDeath: true only for safe, idempotent commands; it retries at most once and may otherwise duplicate side effects.",
2907
- "babysit_run gives full PTY control: drive interactive programs (installers, wizards, REPLs) with babysit_send (text or named keys) and read the rendered screen with babysit_check { screen: true }.",
2908
- "Delegate self-contained tasks (codebase recon, a parallelizable subtask, work that would pollute your context) with babysit_run { profile: \"subagent\", task }. Launch several for independent subtasks; they run concurrently.",
2909
- "Set at least one subagent budget (`maxCost`, `maxTurns`, `maxToolCalls`, or `maxUsageTokens`) for bounded recon and review tasks. Omit budgets only for intentionally open-ended work; the absolute timeout remains a separate safety limit.",
2910
- "Subagents cannot create further subagents by default (maximum depth 1). Only the top-level caller can explicitly opt in by setting maxDepth when it creates the first worker; nested workers inherit that limit and cannot raise it.",
2911
- "After spawning subagents, do not idle-wait and do not end your turn to wait for them: keep making progress, then call babysit_wait (ids + mode any/all) when you need their results. Steer or send follow-up tasks with babysit_send; kill runaways with babysit_kill.",
3347
+ "Use babysit_run for shell commands and give meaningful sessions a stable name; bundle tiny related read-only observations into one command.",
3348
+ "Use babysit_run foreground mode for one process or subagent whose result is needed now; never issue sibling foreground runs in parallel. For parallel checks, start background runs with continueAfterStart and collect them with one multi-session babysit_wait.",
3349
+ "Use returnPattern/returnLines for noisy commands. During edit/fix loops run targeted checks first and one full validation suite at the end instead of repeating every full gate.",
3350
+ "After a background process starts, stop the turn for its automatic notification; never poll or sleep. Use continueAfterStart only for specific non-polling work.",
3351
+ "Inspect large logs with a narrow babysit_check pattern and maxBytes rather than broad tails.",
3352
+ "Use retryOnWorkerDeath only once and only for idempotent commands; retries may duplicate side effects.",
3353
+ "Delegate independent work with bounded babysit_run subagents. Prefer foreground for one result needed now; every background subagent must be collected with babysit_wait before the parent task finishes. Size budgets above the worker's initial context and expected tool count.",
3354
+ "Subagent recursion defaults to depth 1; only a top-level caller may explicitly raise maxDepth.",
2912
3355
  ],
2913
3356
  parameters: Type.Object({
2914
3357
  command: Type.Optional(
@@ -2995,10 +3438,18 @@ export default function (pi: ExtensionAPI) {
2995
3438
  ),
2996
3439
  foreground: Type.Optional(
2997
3440
  Type.Boolean({
2998
- description:
2999
- "Process mode: wait for exit and return the result in this tool call. Use when the next step needs the result; avoid for servers/watchers unless bounded by timeout.",
3441
+ description: "Process or subagent: wait for completion and return the result in this tool call.",
3000
3442
  }),
3001
3443
  ),
3444
+ returnPattern: Type.Optional(
3445
+ Type.String({ description: "Foreground/quick process: return only latest regex matches." }),
3446
+ ),
3447
+ returnLines: Type.Optional(
3448
+ Type.Integer({ minimum: 1, maximum: 200, description: "Lines retained by returnPattern/tail (default 30)." }),
3449
+ ),
3450
+ maxBytes: Type.Optional(
3451
+ Type.Integer({ minimum: 1_000, maximum: ANSWER_MAX_BYTES, description: "Returned process-output cap (default 8 KB)." }),
3452
+ ),
3002
3453
  notificationGroup: Type.Optional(
3003
3454
  Type.String({
3004
3455
  description:
@@ -3062,9 +3513,16 @@ export default function (pi: ExtensionAPI) {
3062
3513
  details: {},
3063
3514
  };
3064
3515
  }
3065
- if (isSubagent && params.foreground) {
3516
+ if (isSubagent && (params.returnPattern || params.returnLines != null || params.maxBytes != null)) {
3066
3517
  return {
3067
- content: [{ type: "text", text: "`foreground` is available only in process mode; use babysit_wait for subagent task completion." }],
3518
+ content: [{ type: "text", text: "`returnPattern`, `returnLines`, and `maxBytes` are process-output options." }],
3519
+ isError: true,
3520
+ details: {},
3521
+ };
3522
+ }
3523
+ if (isSubagent && params.continueAfterStart != null) {
3524
+ return {
3525
+ content: [{ type: "text", text: "`continueAfterStart` is available only in process mode." }],
3068
3526
  isError: true,
3069
3527
  details: {},
3070
3528
  };
@@ -3092,6 +3550,21 @@ export default function (pi: ExtensionAPI) {
3092
3550
 
3093
3551
  // --- process mode ---
3094
3552
  if (!isSubagent) {
3553
+ if (params.returnPattern) {
3554
+ try {
3555
+ new RegExp(params.returnPattern);
3556
+ } catch (error) {
3557
+ return {
3558
+ content: [{ type: "text", text: `Invalid returnPattern: ${String(error)}` }],
3559
+ isError: true,
3560
+ details: {},
3561
+ };
3562
+ }
3563
+ }
3564
+ const outputSelection: ProcessOutputSelection | undefined =
3565
+ params.returnPattern || params.returnLines != null || params.maxBytes != null
3566
+ ? { pattern: params.returnPattern, lines: params.returnLines, maxBytes: params.maxBytes }
3567
+ : undefined;
3095
3568
  const spawnOpts: ProcOpts = {
3096
3569
  name: params.name,
3097
3570
  command: params.command as string,
@@ -3122,14 +3595,14 @@ export default function (pi: ExtensionAPI) {
3122
3595
  // the same deadline here races its terminal-state write and can return a
3123
3596
  // false "still running" result at the boundary, so wait for the
3124
3597
  // supervisor's definitive exit instead.
3125
- let outcome = await waitForExit(res.id, null, _signal);
3598
+ let outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
3126
3599
  let retried = false;
3127
3600
  if (params.retryOnWorkerDeath && outcome.status?.state === "dead" && outcome.status.exit_code == null) {
3128
3601
  const retry = await spawnProcess(spawnOpts);
3129
3602
  if (!("error" in retry)) {
3130
3603
  res = retry;
3131
3604
  retried = true;
3132
- outcome = await waitForExit(res.id, null, _signal);
3605
+ outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
3133
3606
  }
3134
3607
  }
3135
3608
  if (ctx.hasUI) await refreshWidget(ctx);
@@ -3167,7 +3640,7 @@ export default function (pi: ExtensionAPI) {
3167
3640
  }
3168
3641
  }
3169
3642
  if (quickStatus && quickStatus.state !== "running") {
3170
- const outcome = await waitForExit(res.id, null, _signal);
3643
+ const outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
3171
3644
  await refreshWidget(ctx);
3172
3645
  return {
3173
3646
  content: [{ type: "text", text: `${retried ? "Retried once after external worker death.\n" : ""}${outcome.text}` }],
@@ -3269,15 +3742,37 @@ export default function (pi: ExtensionAPI) {
3269
3742
 
3270
3743
  pollNeeded = true;
3271
3744
  await refreshWidget(ctx);
3745
+ if (params.foreground || !ctx.hasUI) {
3746
+ const outcome = await waitForTask(res.id, null, _signal);
3747
+ collectSubagentOutcome(outcome);
3748
+ const usage = claimOutcomeUsage(outcome);
3749
+ if (ctx.hasUI) await refreshWidget(ctx);
3750
+ return {
3751
+ content: [{ type: "text", text: outcome.text }],
3752
+ isError: !outcome.ok,
3753
+ usage,
3754
+ details: {
3755
+ id: res.id,
3756
+ kind: "subagent",
3757
+ name: params.name ?? res.id,
3758
+ agent: agent?.name,
3759
+ model: res.model,
3760
+ task: params.task,
3761
+ depth: subagentNesting.childDepth,
3762
+ maxDepth: subagentNesting.maxDepth,
3763
+ status: outcomeStatus(outcome),
3764
+ },
3765
+ };
3766
+ }
3272
3767
  return {
3273
3768
  content: [
3274
3769
  {
3275
3770
  type: "text",
3276
3771
  text:
3277
3772
  `Subagent started (id: ${res.id})${agent ? ` [agent: ${agent.name}]` : ""}${res.model ? ` [model: ${res.model}]` : ""} [depth: ${subagentNesting.childDepth}/${subagentNesting.maxDepth}].\n` +
3278
- `Task accepted — running in the background; keep working (do NOT end your turn just to wait for it).\n` +
3279
- `Poll: babysit_check { id: "${res.id}" }\n` +
3280
- `Wait: babysit_wait { id: "${res.id}" }\n` +
3773
+ `Task accepted — running in the background. You MUST collect it with babysit_wait before finishing the parent task; use foreground: true next time when no independent work is available.\n` +
3774
+ `Progress: babysit_check { id: "${res.id}" } (only when inspection is needed)\n` +
3775
+ `Collect: babysit_wait { id: "${res.id}" }\n` +
3281
3776
  `Human can watch/steer: /babysit (pick ${res.id})`,
3282
3777
  },
3283
3778
  ],
@@ -3369,6 +3864,13 @@ export default function (pi: ExtensionAPI) {
3369
3864
  lines: Type.Optional(
3370
3865
  Type.Number({ description: "How many tail lines or latest matches to show (default 30, max 200)." }),
3371
3866
  ),
3867
+ maxBytes: Type.Optional(
3868
+ Type.Integer({
3869
+ minimum: 1_000,
3870
+ maximum: ANSWER_MAX_BYTES,
3871
+ description: "Maximum returned bytes for this check (default 4 KB).",
3872
+ }),
3873
+ ),
3372
3874
  pattern: Type.Optional(
3373
3875
  Type.String({
3374
3876
  description:
@@ -3456,6 +3958,7 @@ export default function (pi: ExtensionAPI) {
3456
3958
  }
3457
3959
  const meta = readMeta(params.id);
3458
3960
  const nLines = Math.min(Math.max(1, Math.floor(params.lines ?? 30)), 200);
3961
+ const checkMaxBytes = params.maxBytes ?? TAIL_MAX_BYTES;
3459
3962
  if (params.pattern !== undefined) {
3460
3963
  if (params.screen) {
3461
3964
  return {
@@ -3471,7 +3974,7 @@ export default function (pi: ExtensionAPI) {
3471
3974
  details: {},
3472
3975
  };
3473
3976
  }
3474
- const result = await searchLog(params.id, params.pattern, nLines, signal);
3977
+ const result = await searchLog(params.id, params.pattern, nLines, signal, checkMaxBytes);
3475
3978
  if (result.error) {
3476
3979
  return {
3477
3980
  content: [{ type: "text", text: result.error }],
@@ -3485,7 +3988,7 @@ export default function (pi: ExtensionAPI) {
3485
3988
  ? `--- latest matches /${params.pattern}/ ---\n${result.text}`
3486
3989
  : `(no output matching /${params.pattern}/)`;
3487
3990
  return {
3488
- content: [{ type: "text", text: clip(`${header}\n${body}`) }],
3991
+ content: [{ type: "text", text: clip(`${header}\n${body}`, checkMaxBytes) }],
3489
3992
  details: { status: st, kind, logPath: logPath(params.id), pattern: params.pattern },
3490
3993
  };
3491
3994
  }
@@ -3505,15 +4008,16 @@ export default function (pi: ExtensionAPI) {
3505
4008
  parts.push(header);
3506
4009
  if (params.screen) {
3507
4010
  const sc = await bs(["screenshot", "-s", params.id, "--trim"]);
3508
- parts.push(`--- screen ---\n${clip(sc.stdout.trimEnd()) || "(blank screen)"}`);
4011
+ parts.push(`--- screen ---\n${clip(sc.stdout.trimEnd(), checkMaxBytes) || "(blank screen)"}`);
3509
4012
  } else {
3510
4013
  const tail = clip(
3511
4014
  (await bs(["log", "-s", params.id, "--tail", String(nLines)])).stdout.trimEnd(),
4015
+ checkMaxBytes,
3512
4016
  );
3513
4017
  parts.push(tail ? `--- recent output ---\n${tail}` : "(no output yet)");
3514
4018
  }
3515
4019
  return {
3516
- content: [{ type: "text", text: clip(parts.join("\n")) }],
4020
+ content: [{ type: "text", text: clip(parts.join("\n"), checkMaxBytes) }],
3517
4021
  details: { status: st, kind: "process", logPath: logPath(params.id) },
3518
4022
  };
3519
4023
  }
@@ -3536,7 +4040,7 @@ export default function (pi: ExtensionAPI) {
3536
4040
  : " · working";
3537
4041
  }
3538
4042
  if (st.exit_code != null) header += ` exit_code=${st.exit_code}`;
3539
- header += ` turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCalls.length}`;
4043
+ header += ` turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}`;
3540
4044
  if (prog.tokens != null) header += ` ctx=${prog.tokens}`;
3541
4045
  if (prog.modelCalls > 0) header += ` usage=${prog.usageTokens} $${prog.cost.toFixed(4)}`;
3542
4046
  if (st.note) header += ` ⚑ ${st.note}`;
@@ -3545,7 +4049,7 @@ export default function (pi: ExtensionAPI) {
3545
4049
  if (prog.errorMsg) parts.push(`⚠ error: ${clip(prog.errorMsg, ANSWER_MAX_BYTES)}`);
3546
4050
 
3547
4051
  if (recent.length > 0) {
3548
- const skipped = prog.toolCalls.length - recent.length;
4052
+ const skipped = Math.max(0, prog.toolCallCount - recent.length);
3549
4053
  parts.push(
3550
4054
  `--- recent tool calls${skipped > 0 ? ` (+${skipped} earlier)` : ""} ---\n` +
3551
4055
  recent.map((t) => ` ${t.summary}`).join("\n"),
@@ -3554,16 +4058,16 @@ export default function (pi: ExtensionAPI) {
3554
4058
 
3555
4059
  if (prog.finalText.trim()) {
3556
4060
  parts.push(`--- answer so far ---\n${clip(prog.finalText.trim(), ANSWER_MAX_BYTES)}`);
3557
- } else if (prog.toolCalls.length === 0 && st.state !== "running") {
4061
+ } else if (prog.toolCallCount === 0 && st.state !== "running") {
3558
4062
  parts.push(buildSubagentExitDiagnostic(prog, logPath(params.id)));
3559
- } else if (prog.toolCalls.length === 0) {
4063
+ } else if (prog.toolCallCount === 0) {
3560
4064
  parts.push("(starting up… no events yet)");
3561
4065
  } else {
3562
4066
  parts.push("(working… no answer text yet)");
3563
4067
  }
3564
4068
 
3565
4069
  return {
3566
- content: [{ type: "text", text: clip(parts.join("\n")) }],
4070
+ content: [{ type: "text", text: clip(parts.join("\n"), checkMaxBytes) }],
3567
4071
  details: { status: st, progress: prog, kind: "subagent" },
3568
4072
  };
3569
4073
  },
@@ -3577,7 +4081,7 @@ export default function (pi: ExtensionAPI) {
3577
4081
  "Send input to a babysit session. Process: `text` types a line into its stdin (PTY), " +
3578
4082
  "`keys` presses named keys (Enter, Tab, Esc, Up/Down/Left/Right, C-c, F1…) — use with " +
3579
4083
  "babysit_check { screen: true } to drive interactive programs. Subagent: `text` is " +
3580
- "STEERING while it works, or a NEW TASK when it is idle (mode: auto/steer/task) — this " +
4084
+ "STEERING while it works, or a NEW TASK after the current task settles (mode: auto/steer/task) — this " +
3581
4085
  "is how you resume a finished subagent with full context.",
3582
4086
  promptSnippet: "Send text/keys to a process, or steering/follow-up tasks to a subagent",
3583
4087
  parameters: Type.Object({
@@ -3593,7 +4097,7 @@ export default function (pi: ExtensionAPI) {
3593
4097
  mode: Type.Optional(
3594
4098
  StringEnum(["auto", "steer", "task"] as const, {
3595
4099
  description:
3596
- "Subagent only. auto (default): steer if mid-run, otherwise start a new task. steer/task force one behavior.",
4100
+ "Subagent only. auto (default): steer unless the current task is settled. task requires confirmed settlement; steer always sends guidance.",
3597
4101
  }),
3598
4102
  ),
3599
4103
  noNewline: Type.Optional(
@@ -3693,15 +4197,39 @@ export default function (pi: ExtensionAPI) {
3693
4197
  };
3694
4198
  }
3695
4199
  let mode = params.mode ?? "auto";
3696
- if (mode === "auto") {
3697
- // isStreaming tells us whether an agent run is in flight right now.
4200
+ if (mode === "auto" || mode === "task") {
4201
+ // A prompt sent while the current run is streaming can queue behind that
4202
+ // run while immediately replacing our per-task offsets and budget state.
4203
+ // Establish idleness before every new task; auto safely falls back to
4204
+ // steering when state is unknown, while an explicit task fails closed.
3698
4205
  const gs = await sendRpc(params.id, { type: "get_state" });
3699
- let streaming = true; // assume busy when unsure — steering is the safe default
4206
+ let streaming: boolean | undefined;
3700
4207
  if (!("error" in gs)) {
3701
4208
  const r = await rpcResponse(params.id, gs.offset, "get_state", "10s");
3702
4209
  if (r.ok) streaming = Boolean((r.data as { isStreaming?: boolean })?.isStreaming);
3703
4210
  }
3704
- mode = streaming ? "steer" : "task";
4211
+ let currentTaskDone: boolean | undefined;
4212
+ try {
4213
+ currentTaskDone = taskProgressOf(params.id).progress.done;
4214
+ } catch {
4215
+ /* fail closed below rather than replacing unknown task state */
4216
+ }
4217
+ const resolved = resolveSubagentSendMode(mode, streaming, currentTaskDone);
4218
+ if ("error" in resolved) {
4219
+ return {
4220
+ content: [{
4221
+ type: "text",
4222
+ text: resolved.error === "busy"
4223
+ ? `Subagent ${params.id} is still streaming; use mode \"steer\" or wait for the current task to settle before starting another task.`
4224
+ : resolved.error === "unsettled"
4225
+ ? `Subagent ${params.id} has not settled its current task (it may be parked on a background process); wait for completion before starting another task.`
4226
+ : `Could not verify that subagent ${params.id} is idle and settled; retry with mode \"task\" after checking its state.`,
4227
+ }],
4228
+ isError: true,
4229
+ details: { mode: "task" },
4230
+ };
4231
+ }
4232
+ mode = resolved.mode;
3705
4233
  }
3706
4234
  const deliveryCleanupAfter =
3707
4235
  mode === "steer"
@@ -3755,10 +4283,13 @@ export default function (pi: ExtensionAPI) {
3755
4283
  depth: meta?.depth,
3756
4284
  maxDepth: meta?.maxDepth,
3757
4285
  budget: meta?.budget,
3758
- // Each follow-up task receives a fresh budget window.
4286
+ // Each follow-up task receives fresh budget and usage-accounting windows.
4287
+ budgetWarnedAt: undefined,
4288
+ budgetWarningReason: undefined,
3759
4289
  budgetExceededAt: undefined,
3760
4290
  budgetReason: undefined,
3761
4291
  budgetKilled: undefined,
4292
+ usageReportedOffset: undefined,
3762
4293
  });
3763
4294
  } else if (delivery.tempDir && meta) {
3764
4295
  writeMeta(params.id, {
@@ -3793,7 +4324,7 @@ export default function (pi: ExtensionAPI) {
3793
4324
  "Block until babysit session(s) finish, then return the result. A process finishes " +
3794
4325
  "when it EXITS (or, with `expect`, as soon as a regex appears in its output — e.g. wait " +
3795
4326
  "for 'listening on' before hitting a dev server). A subagent finishes when its current " +
3796
- "TASK completes (the session stays alive for follow-ups). Pass `id` for one session, or " +
4327
+ "TASK completes (the worker remains reusable only during its configured idle grace). Pass `id` for one session, or " +
3797
4328
  "`ids` + `mode`: 'all' (default) waits for every one, 'any' returns on the FIRST finisher. " +
3798
4329
  "Multi-session results are capped at the inline-output limit (8 KB by default); use `maxBytes` " +
3799
4330
  "to opt into a larger result up to 24 KB. " +
@@ -3865,9 +4396,12 @@ export default function (pi: ExtensionAPI) {
3865
4396
 
3866
4397
  if (ids.length === 1) {
3867
4398
  const r = await waitFor(ids[0], limitMs, signal, params.expect);
4399
+ collectSubagentOutcome(r);
4400
+ const usage = claimOutcomeUsage(r);
3868
4401
  return {
3869
4402
  content: [{ type: "text", text: r.text }],
3870
4403
  isError: !r.ok,
4404
+ usage,
3871
4405
  details: {
3872
4406
  status: r.status,
3873
4407
  progress: r.progress,
@@ -3883,6 +4417,8 @@ export default function (pi: ExtensionAPI) {
3883
4417
  ids.map((i) => waitFor(i, limitMs, signal, params.expect)),
3884
4418
  );
3885
4419
  const ok = results.every((r) => r.ok);
4420
+ results.forEach(collectSubagentOutcome);
4421
+ const usage = sumNestedUsage(results.map(claimOutcomeUsage));
3886
4422
  return {
3887
4423
  content: [
3888
4424
  {
@@ -3894,6 +4430,7 @@ export default function (pi: ExtensionAPI) {
3894
4430
  },
3895
4431
  ],
3896
4432
  isError: !ok,
4433
+ usage,
3897
4434
  details: {
3898
4435
  results: results.map((r) => ({ id: r.id, kind: r.kind, ok: r.ok })),
3899
4436
  },
@@ -3910,6 +4447,8 @@ export default function (pi: ExtensionAPI) {
3910
4447
  ids.map((i) => waitFor(i, limitMs, ctrl.signal, params.expect)),
3911
4448
  );
3912
4449
  const others = ids.filter((i) => i !== first.id);
4450
+ collectSubagentOutcome(first);
4451
+ const usage = claimOutcomeUsage(first);
3913
4452
  return {
3914
4453
  content: [
3915
4454
  {
@@ -3923,6 +4462,7 @@ export default function (pi: ExtensionAPI) {
3923
4462
  },
3924
4463
  ],
3925
4464
  isError: !first.ok,
4465
+ usage,
3926
4466
  details: { first: { id: first.id, kind: first.kind, ok: first.ok }, remaining: others },
3927
4467
  };
3928
4468
  } finally {
@@ -4073,7 +4613,7 @@ export default function (pi: ExtensionAPI) {
4073
4613
  // Parse the RPC event stream and show the final answer, not raw JSONL.
4074
4614
  const prog = taskProgressOf(picked.id).progress;
4075
4615
  const stats =
4076
- `turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCalls.length}` +
4616
+ `turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}` +
4077
4617
  (prog.tokens != null ? ` ctx=${prog.tokens}` : "") +
4078
4618
  (prog.modelCalls > 0 ? ` usage=${prog.usageTokens} $${prog.cost.toFixed(4)}` : "");
4079
4619
  const body =