@yusukeshib/pi-babysit 0.3.16 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +21 -15
  2. package/index.ts +614 -170
  3. package/package.json +1 -1
package/index.ts CHANGED
@@ -19,7 +19,7 @@
19
19
  * `pi --mode rpc` worker. Tasks are injected as RPC `prompt`
20
20
  * commands over stdin, completion is detected from the JSONL
21
21
  * event stream (`agent_settled`), NOT process exit; the session
22
- * stays alive for cheap follow-up tasks. Same design as the
22
+ * remains reusable during its configured idle grace. Same design as the
23
23
  * old pi-subagent extension.
24
24
  *
25
25
  * The "profile" is a tool-parameter, not a separate tool set: one small tool
@@ -274,11 +274,14 @@ export function isSupportedBabysitVersion(output: string): boolean {
274
274
  // Cached preflight — probe `babysit --version` exactly once per process.
275
275
  // undefined = not probed, null = supported, string = actionable error.
276
276
  let babysitPreflightError: string | null | undefined;
277
+ let babysitPreflightCheckedAt = 0;
277
278
  async function babysitAvailable(): Promise<boolean> {
278
- // Cache only success. A missing or outdated binary may be installed while pi
279
- // stays open, so subsequent tool calls must be able to recover without a restart.
280
279
  if (babysitPreflightError === null) return true;
280
+ // Briefly negative-cache failures so repeated mistaken calls do not fork a
281
+ // version probe each time, while still recovering quickly after installation.
282
+ if (babysitPreflightError && Date.now() - babysitPreflightCheckedAt < 2_000) return false;
281
283
  const r = await bs(["--version"]);
284
+ babysitPreflightCheckedAt = Date.now();
282
285
  if (r.code !== 0) {
283
286
  babysitPreflightError = INSTALL_HINT;
284
287
  } else if (!isSupportedBabysitVersion(r.stdout)) {
@@ -406,23 +409,54 @@ interface Meta {
406
409
  depth?: number;
407
410
  maxDepth?: number;
408
411
  budget?: SubagentBudget;
412
+ /** Soft-limit warning (80% by default) was accepted for this task. */
413
+ budgetWarnedAt?: number;
414
+ budgetWarningReason?: string;
415
+ /** Hard limit was first observed; grace is measured from observation, not RPC acceptance. */
409
416
  budgetExceededAt?: number;
410
417
  budgetReason?: string;
411
418
  budgetKilled?: boolean;
419
+ /** Prompt offset whose nested usage has already been charged to the parent session. */
420
+ usageReportedOffset?: number;
412
421
  }
413
422
 
414
423
  const metaDir = () => path.join(ROOT, "meta");
415
424
  const logPath = (id: string) => path.join(ROOT, "sessions", id, "output.log");
416
425
 
417
- function writeMeta(id: string, m: Meta): void {
426
+ function writeMeta(id: string, m: Meta): boolean {
427
+ const target = path.join(metaDir(), `${id}.json`);
428
+ const temp = `${target}.${process.pid}.${Date.now()}.${Math.random().toString(16).slice(2)}.tmp`;
418
429
  try {
419
430
  fs.mkdirSync(metaDir(), { recursive: true });
420
- fs.writeFileSync(path.join(metaDir(), `${id}.json`), JSON.stringify(m));
431
+ fs.writeFileSync(temp, JSON.stringify(m));
432
+ fs.renameSync(temp, target);
433
+ return true;
421
434
  } catch {
422
- /* best-effort */
435
+ try {
436
+ fs.rmSync(temp, { force: true });
437
+ } catch {
438
+ /* best-effort */
439
+ }
440
+ return false;
423
441
  }
424
442
  }
425
443
 
444
+ export function claimFileOnce(file: string, payload: string): boolean {
445
+ let fd: number;
446
+ try {
447
+ fs.mkdirSync(path.dirname(file), { recursive: true });
448
+ fd = fs.openSync(file, "wx");
449
+ } catch {
450
+ return false;
451
+ }
452
+ try {
453
+ fs.writeFileSync(fd, payload);
454
+ } finally {
455
+ fs.closeSync(fd);
456
+ }
457
+ return true;
458
+ }
459
+
426
460
  function readMeta(id: string): Meta | null {
427
461
  try {
428
462
  return JSON.parse(fs.readFileSync(path.join(metaDir(), `${id}.json`), "utf-8"));
@@ -448,8 +482,27 @@ function processIsAlive(pid: number): boolean {
448
482
  }
449
483
 
450
484
  const GC_LOCK_FILE = ".pi-babysit-gc.lock";
485
+ const GC_STAMP_FILE = ".pi-babysit-gc.last";
486
+ const AUTOMATIC_GC_INTERVAL_MS = 24 * 60 * 60 * 1_000;
451
487
  const ACTIVE_LEASE_PREFIX = ".pi-babysit-active-";
452
488
 
489
+ function automaticGcDue(now = Date.now()): boolean {
490
+ try {
491
+ return now - fs.statSync(path.join(ROOT_BASE, GC_STAMP_FILE)).mtimeMs >= AUTOMATIC_GC_INTERVAL_MS;
492
+ } catch {
493
+ return true;
494
+ }
495
+ }
496
+
497
+ function markAutomaticGc(now = new Date()): void {
498
+ try {
499
+ fs.mkdirSync(ROOT_BASE, { recursive: true });
500
+ fs.writeFileSync(path.join(ROOT_BASE, GC_STAMP_FILE), now.toISOString());
501
+ } catch {
502
+ /* best-effort; GC safety does not depend on this throttle stamp */
503
+ }
504
+ }
505
+
453
506
  function scanTreeStats(root: string): { bytes: number; newestMtimeMs: number } {
454
507
  let bytes = 0;
455
508
  let newestMtimeMs = 0;
@@ -791,6 +844,45 @@ export function parseRpcResponseBytes(
791
844
  // it. Distinguishes: success, explicit failure (success:false → the error
792
845
  // message, e.g. "No API key found for …"), subagent death, and timeout — this
793
846
  // is what makes bad-model/config failures LOUD instead of silent.
847
+ export function rpcResponsePattern(command: string): string {
848
+ const escaped = command.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
849
+ // Match only a complete response record. A loose `"command":"…"` can occur
850
+ // in assistant/tool text, and matching before the terminating newline can
851
+ // race the writer while the JSON record is still partial.
852
+ return `(?m)^\\{(?:"id":"[^"\\n]*",)?"type":"response","command":"${escaped}"[^\\n]*\\}\\r?\\n`;
853
+ }
854
+
855
+ export function readLogBytesFrom(file: string, since: number): Buffer {
856
+ const size = fs.statSync(file).size;
857
+ const offset = Math.min(Math.max(0, since), size);
858
+ const length = size - offset;
859
+ const bytes = Buffer.allocUnsafe(length);
860
+ const fd = fs.openSync(file, "r");
861
+ let read = 0;
862
+ try {
863
+ while (read < length) {
864
+ const count = fs.readSync(fd, bytes, read, length - read, offset + read);
865
+ if (count === 0) break;
866
+ read += count;
867
+ }
868
+ } finally {
869
+ fs.closeSync(fd);
870
+ }
871
+ return bytes.subarray(0, read);
872
+ }
873
+
874
+ const RPC_RESPONSE_WINDOW_MAX_BYTES = 1_000_000;
875
+ function readLogWindowFrom(
876
+ file: string,
877
+ since: number,
878
+ maxBytes = RPC_RESPONSE_WINDOW_MAX_BYTES,
879
+ ): { bytes: Buffer; offset: number } {
880
+ const size = fs.statSync(file).size;
881
+ const requested = Math.min(Math.max(0, since), size);
882
+ const offset = Math.max(requested, size - maxBytes);
883
+ return { bytes: readLogBytesFrom(file, offset), offset };
884
+ }
885
+
794
886
  async function rpcResponse(
795
887
  id: string,
796
888
  since: number,
@@ -802,7 +894,16 @@ async function rpcResponse(
802
894
  | { ok: false; error: string }
803
895
  > {
804
896
  const e = await bs(
805
- ["expect", "-s", id, "--since", String(since), "--timeout", timeout, `"command":"${command}"`],
897
+ [
898
+ "expect",
899
+ "-s",
900
+ id,
901
+ "--since",
902
+ String(since),
903
+ "--timeout",
904
+ timeout,
905
+ rpcResponsePattern(command),
906
+ ],
806
907
  { signal },
807
908
  );
808
909
  if (e.code !== 0) {
@@ -810,8 +911,10 @@ async function rpcResponse(
810
911
  if (st && st.state !== "running") {
811
912
  let structuredError = "";
812
913
  try {
813
- const bytes = fs.readFileSync(logPath(id));
814
- structuredError = parseEvents(bytes.subarray(Math.min(since, bytes.length)).toString("utf8")).errorMsg ?? "";
914
+ // Long-lived follow-up workers can have very large historical logs. Read
915
+ // only the response window rather than synchronously loading all history.
916
+ const window = readLogWindowFrom(logPath(id), since);
917
+ structuredError = parseEvents(window.bytes.toString("utf8")).errorMsg ?? "";
815
918
  } catch {
816
919
  /* full log path below remains the diagnostic source */
817
920
  }
@@ -832,22 +935,8 @@ async function rpcResponse(
832
935
  };
833
936
  }
834
937
  try {
835
- const file = logPath(id);
836
- const size = fs.statSync(file).size;
837
- const length = Math.max(0, size - since);
838
- const bytes = Buffer.allocUnsafe(length);
839
- const fd = fs.openSync(file, "r");
840
- let read = 0;
841
- try {
842
- while (read < length) {
843
- const count = fs.readSync(fd, bytes, read, length - read, since + read);
844
- if (count === 0) break;
845
- read += count;
846
- }
847
- } finally {
848
- fs.closeSync(fd);
849
- }
850
- return parseRpcResponseBytes(bytes.subarray(0, read), since, command);
938
+ const window = readLogWindowFrom(logPath(id), since);
939
+ return parseRpcResponseBytes(window.bytes, window.offset, command);
851
940
  } catch (error) {
852
941
  return { ok: false, error: `could not read ${command} response: ${String(error)}` };
853
942
  }
@@ -877,7 +966,7 @@ function byteLimitFromEnv(name: string, fallback: number): number {
877
966
  return Number.isSafeInteger(value) && value >= 0 ? value : fallback;
878
967
  }
879
968
 
880
- const TAIL_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_TAIL_MAX_BYTES", 8_000);
969
+ const TAIL_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_TAIL_MAX_BYTES", 4_000);
881
970
  // Direct run/wait results can carry more context because the caller explicitly
882
971
  // requested them. Unsolicited completion notifications default much smaller.
883
972
  const INLINE_OUTPUT_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_INLINE_OUTPUT_MAX_BYTES", 8_000);
@@ -887,7 +976,12 @@ const NOTIFY_BATCH_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_NOTIFY_BATCH_MAX_BYT
887
976
  const ANSWER_MAX_BYTES = 24_000; // single subagent answers / structured error messages
888
977
  const MAX_MULTI_WAIT_SESSIONS = 32;
889
978
  const SUBAGENT_BUDGET_GRACE_MS =
890
- parseDurMs(process.env.PI_BABYSIT_BUDGET_GRACE ?? "30s") ?? 30_000;
979
+ parseDurMs(process.env.PI_BABYSIT_BUDGET_GRACE ?? "90s") ?? 90_000;
980
+ const SUBAGENT_REAP_AFTER =
981
+ process.env.PI_BABYSIT_REAP_AFTER ?? process.env.PI_SUBAGENT_REAP_AFTER ?? "120s";
982
+ const SUBAGENT_REUSE_HINT = ["0", "off", "none"].includes(SUBAGENT_REAP_AFTER)
983
+ ? "Session remains available until its absolute timeout."
984
+ : `Session remains available for follow-ups during the ${SUBAGENT_REAP_AFTER} idle grace.`;
891
985
 
892
986
  export function clip(s: string, maxBytes = TAIL_MAX_BYTES): string {
893
987
  if (maxBytes <= 0) return "";
@@ -918,15 +1012,27 @@ export function clipMultiWaitResult(
918
1012
  return clip(value, Math.min(maxBytes, ANSWER_MAX_BYTES));
919
1013
  }
920
1014
 
1015
+ interface SearchLogCacheEntry {
1016
+ size: number;
1017
+ mtimeMs: number;
1018
+ text: string;
1019
+ }
1020
+ const searchLogCache = new Map<string, SearchLogCacheEntry>();
1021
+
921
1022
  async function searchLog(
922
1023
  id: string,
923
1024
  pattern: string,
924
1025
  maxLines: number,
925
1026
  signal?: AbortSignal,
1027
+ maxBytes = TAIL_MAX_BYTES,
926
1028
  ): Promise<{ text: string; error?: string }> {
927
1029
  const file = logPath(id);
928
1030
  if (!fs.existsSync(file)) return { text: "", error: `Log file is missing: ${file}` };
929
1031
  if (signal?.aborted) return { text: "", error: "Log search was interrupted." };
1032
+ const stat = fs.statSync(file);
1033
+ const cacheKey = `${file}\u0000${pattern}\u0000${maxLines}\u0000${maxBytes}`;
1034
+ const cached = searchLogCache.get(cacheKey);
1035
+ if (cached?.size === stat.size && cached.mtimeMs === stat.mtimeMs) return { text: cached.text };
930
1036
 
931
1037
  // Run regex evaluation out of process so catastrophic backtracking or a huge
932
1038
  // no-newline log cannot freeze or exhaust pi's main Node process. The helper
@@ -974,7 +1080,10 @@ async function searchLog(
974
1080
  } else if (code !== 0) {
975
1081
  finish({ text: "", error: stderr.trim() || `Log search failed (exit ${code ?? "?"}).` });
976
1082
  } else {
977
- finish({ text: clip(stdout.trimEnd()) });
1083
+ const text = clip(stdout.trimEnd(), maxBytes);
1084
+ searchLogCache.set(cacheKey, { size: stat.size, mtimeMs: stat.mtimeMs, text });
1085
+ while (searchLogCache.size > 64) searchLogCache.delete(searchLogCache.keys().next().value as string);
1086
+ finish({ text });
978
1087
  }
979
1088
  });
980
1089
  });
@@ -1008,6 +1117,33 @@ async function inlineOutput(
1008
1117
  return output ? `\n\nOutput:\n${output}` : "";
1009
1118
  }
1010
1119
 
1120
+ interface ProcessOutputSelection {
1121
+ pattern?: string;
1122
+ lines?: number;
1123
+ maxBytes?: number;
1124
+ }
1125
+
1126
+ async function selectedProcessOutput(
1127
+ id: string,
1128
+ status: BsSession,
1129
+ selection?: ProcessOutputSelection,
1130
+ signal?: AbortSignal,
1131
+ ): Promise<string> {
1132
+ if (!selection || (!selection.pattern && selection.lines == null && selection.maxBytes == null)) {
1133
+ return inlineOutput(id, status);
1134
+ }
1135
+ const maxBytes = selection.maxBytes ?? INLINE_OUTPUT_MAX_BYTES;
1136
+ const lines = Math.min(Math.max(1, Math.floor(selection.lines ?? 30)), 200);
1137
+ if (selection.pattern) {
1138
+ const result = await searchLog(id, selection.pattern, lines, signal, maxBytes);
1139
+ if (result.error) return `\nOutput filter failed: ${result.error}`;
1140
+ const body = result.text || `(no output matching /${selection.pattern}/)`;
1141
+ return `\n\nSelected output /${selection.pattern}/:\n${clip(body, maxBytes)}`;
1142
+ }
1143
+ const tail = (await bs(["log", "-s", id, "--tail", String(lines)])).stdout.trimEnd();
1144
+ return tail ? `\n\nSelected tail (${lines} lines max):\n${clip(tail, maxBytes)}` : "";
1145
+ }
1146
+
1011
1147
  export function summarizeNotificationCommand(command: string | undefined): string {
1012
1148
  const preview =
1013
1149
  (command ?? "?")
@@ -1283,7 +1419,10 @@ interface ToolCall {
1283
1419
  }
1284
1420
  export interface Progress {
1285
1421
  turns: number;
1422
+ /** Bounded recent calls for status rendering. */
1286
1423
  toolCalls: ToolCall[];
1424
+ /** Exact count, independent of the bounded recent-call ring. */
1425
+ toolCallCount: number;
1287
1426
  finalText: string;
1288
1427
  /** Best-effort text from the currently streaming assistant message. */
1289
1428
  streamingText: string;
@@ -1298,6 +1437,10 @@ export interface Progress {
1298
1437
  cacheWriteTokens: number;
1299
1438
  reasoningTokens: number;
1300
1439
  cost: number;
1440
+ inputCost: number;
1441
+ outputCost: number;
1442
+ cacheReadCost: number;
1443
+ cacheWriteCost: number;
1301
1444
  errorMsg?: string;
1302
1445
  // RPC lifecycle bookkeeping (computed over the analyzed log slice):
1303
1446
  agentStarts: number;
@@ -1340,6 +1483,7 @@ function emptyProgress(): Progress {
1340
1483
  return {
1341
1484
  turns: 0,
1342
1485
  toolCalls: [],
1486
+ toolCallCount: 0,
1343
1487
  finalText: "",
1344
1488
  streamingText: "",
1345
1489
  modelCalls: 0,
@@ -1350,6 +1494,10 @@ function emptyProgress(): Progress {
1350
1494
  cacheWriteTokens: 0,
1351
1495
  reasoningTokens: 0,
1352
1496
  cost: 0,
1497
+ inputCost: 0,
1498
+ outputCost: 0,
1499
+ cacheReadCost: 0,
1500
+ cacheWriteCost: 0,
1353
1501
  agentStarts: 0,
1354
1502
  agentEnds: 0,
1355
1503
  agentSettled: 0,
@@ -1371,8 +1519,8 @@ export function subagentBudgetViolation(
1371
1519
  if (budget.maxTurns != null && progress.turns >= budget.maxTurns) {
1372
1520
  return `${progress.turns} turns reached maxTurns ${budget.maxTurns}`;
1373
1521
  }
1374
- if (budget.maxToolCalls != null && progress.toolCalls.length >= budget.maxToolCalls) {
1375
- return `${progress.toolCalls.length} tool calls reached maxToolCalls ${budget.maxToolCalls}`;
1522
+ if (budget.maxToolCalls != null && progress.toolCallCount >= budget.maxToolCalls) {
1523
+ return `${progress.toolCallCount} tool calls reached maxToolCalls ${budget.maxToolCalls}`;
1376
1524
  }
1377
1525
  if (budget.maxUsageTokens != null && progress.usageTokens >= budget.maxUsageTokens) {
1378
1526
  return `${progress.usageTokens} usage tokens reached maxUsageTokens ${budget.maxUsageTokens}`;
@@ -1380,6 +1528,33 @@ export function subagentBudgetViolation(
1380
1528
  return null;
1381
1529
  }
1382
1530
 
1531
+ export function subagentBudgetSoftViolation(
1532
+ progress: Progress,
1533
+ budget?: SubagentBudget,
1534
+ ratio = 0.8,
1535
+ ): string | null {
1536
+ if (!budget || ratio <= 0 || ratio >= 1) return null;
1537
+ if (budget.maxCost != null && progress.cost >= budget.maxCost * ratio) {
1538
+ return `cost $${progress.cost.toFixed(4)} reached ${Math.round(ratio * 100)}% of maxCost $${budget.maxCost.toFixed(4)}`;
1539
+ }
1540
+ if (budget.maxTurns != null && progress.turns >= Math.max(1, Math.ceil(budget.maxTurns * ratio))) {
1541
+ return `${progress.turns} turns reached ${Math.round(ratio * 100)}% of maxTurns ${budget.maxTurns}`;
1542
+ }
1543
+ if (
1544
+ budget.maxToolCalls != null &&
1545
+ progress.toolCallCount >= Math.max(1, Math.ceil(budget.maxToolCalls * ratio))
1546
+ ) {
1547
+ return `${progress.toolCallCount} tool calls reached ${Math.round(ratio * 100)}% of maxToolCalls ${budget.maxToolCalls}`;
1548
+ }
1549
+ if (
1550
+ budget.maxUsageTokens != null &&
1551
+ progress.usageTokens >= Math.max(1, Math.ceil(budget.maxUsageTokens * ratio))
1552
+ ) {
1553
+ return `${progress.usageTokens} usage tokens reached ${Math.round(ratio * 100)}% of maxUsageTokens ${budget.maxUsageTokens}`;
1554
+ }
1555
+ return null;
1556
+ }
1557
+
1383
1558
  export function subagentBudgetAction(
1384
1559
  progress: Progress,
1385
1560
  budget: SubagentBudget | undefined,
@@ -1427,16 +1602,19 @@ function parseEventLine(progress: Progress, raw: string): void {
1427
1602
  | { type?: string; delta?: string }
1428
1603
  | undefined;
1429
1604
  if (update?.type === "text_delta" && typeof update.delta === "string") {
1430
- progress.streamingText += update.delta;
1605
+ progress.streamingText = clip(progress.streamingText + update.delta, ANSWER_MAX_BYTES);
1431
1606
  }
1432
1607
  break;
1433
1608
  }
1434
1609
  case "tool_execution_start": {
1435
1610
  const name = String(event.toolName ?? "tool");
1611
+ progress.toolCallCount++;
1436
1612
  progress.toolCalls.push({
1437
1613
  name,
1438
1614
  summary: summarizeToolCall(name, (event.args as Record<string, unknown>) ?? {}),
1439
1615
  });
1616
+ // Open-ended workers must not retain an unbounded tool history in Pi.
1617
+ if (progress.toolCalls.length > 200) progress.toolCalls.splice(0, progress.toolCalls.length - 200);
1440
1618
  break;
1441
1619
  }
1442
1620
  case "message_end": {
@@ -1444,6 +1622,8 @@ function parseEventLine(progress: Progress, raw: string): void {
1444
1622
  | {
1445
1623
  role?: string;
1446
1624
  content?: { type: string; text?: string }[];
1625
+ stopReason?: string;
1626
+ errorMessage?: string;
1447
1627
  usage?: {
1448
1628
  input?: number;
1449
1629
  output?: number;
@@ -1451,7 +1631,13 @@ function parseEventLine(progress: Progress, raw: string): void {
1451
1631
  cacheWrite?: number;
1452
1632
  reasoning?: number;
1453
1633
  totalTokens?: number;
1454
- cost?: { total?: number };
1634
+ cost?: {
1635
+ input?: number;
1636
+ output?: number;
1637
+ cacheRead?: number;
1638
+ cacheWrite?: number;
1639
+ total?: number;
1640
+ };
1455
1641
  };
1456
1642
  }
1457
1643
  | undefined;
@@ -1460,7 +1646,10 @@ function parseEventLine(progress: Progress, raw: string): void {
1460
1646
  .filter((content) => content.type === "text" && content.text)
1461
1647
  .map((content) => content.text)
1462
1648
  .join("");
1463
- if (text.trim()) progress.finalText = text;
1649
+ if (text.trim()) progress.finalText = clip(text, ANSWER_MAX_BYTES);
1650
+ if (message.stopReason === "error") {
1651
+ progress.errorMsg = message.errorMessage || "subagent model request failed";
1652
+ }
1464
1653
  progress.streamingText = "";
1465
1654
  if (message.usage) {
1466
1655
  const finite = (value: number | undefined) =>
@@ -1473,6 +1662,10 @@ function parseEventLine(progress: Progress, raw: string): void {
1473
1662
  progress.cacheReadTokens += finite(message.usage.cacheRead);
1474
1663
  progress.cacheWriteTokens += finite(message.usage.cacheWrite);
1475
1664
  progress.reasoningTokens += finite(message.usage.reasoning);
1665
+ progress.inputCost += finite(message.usage.cost?.input);
1666
+ progress.outputCost += finite(message.usage.cost?.output);
1667
+ progress.cacheReadCost += finite(message.usage.cost?.cacheRead);
1668
+ progress.cacheWriteCost += finite(message.usage.cost?.cacheWrite);
1476
1669
  progress.cost += finite(message.usage.cost?.total);
1477
1670
  }
1478
1671
  }
@@ -1561,6 +1754,22 @@ interface TaskProgressCache {
1561
1754
  }
1562
1755
  const taskProgressCache = new Map<string, TaskProgressCache>();
1563
1756
 
1757
+ export function pruneTerminalSessionCache<T>(
1758
+ cache: Map<string, T>,
1759
+ sessions: Array<{ id: string; state: string }>,
1760
+ ): number {
1761
+ const running = new Set(
1762
+ sessions.filter((session) => session.state === "running").map((session) => session.id),
1763
+ );
1764
+ let removed = 0;
1765
+ for (const id of cache.keys()) {
1766
+ if (running.has(id)) continue;
1767
+ cache.delete(id);
1768
+ removed++;
1769
+ }
1770
+ return removed;
1771
+ }
1772
+
1564
1773
  /** Parse only bytes appended since the previous observation of this task. */
1565
1774
  function taskProgressOf(id: string): { progress: Progress; offset: number } {
1566
1775
  const base = readMeta(id)?.promptOffset ?? 0;
@@ -1887,6 +2096,33 @@ async function spawnSubagent(
1887
2096
  await bs(["kill", "-s", id]);
1888
2097
  return { error: `subagent ${id} rejected the task: ${resp.error}` };
1889
2098
  }
2099
+ // Prompt acceptance does not guarantee provider authentication: Pi reports
2100
+ // failures that occur after acceptance through the event stream. Probe a short
2101
+ // window so immediate missing-key/config errors fail the spawn instead of
2102
+ // creating a zero-work worker that the caller must discover later.
2103
+ const startupProbe = await bs([
2104
+ "expect",
2105
+ "-s",
2106
+ id,
2107
+ "--since",
2108
+ String(resp.offset),
2109
+ "--timeout",
2110
+ "500ms",
2111
+ '(?m)^\\{"type":"(?:message_end|error|extension_error|agent_settled)"',
2112
+ ]);
2113
+ if (startupProbe.code === 0) {
2114
+ try {
2115
+ const window = readLogWindowFrom(logPath(id), resp.offset);
2116
+ const initialProgress = parseEvents(window.bytes.toString("utf8"));
2117
+ if (initialProgress.errorMsg && initialProgress.modelCalls === 0) {
2118
+ discardDelivery(delivery);
2119
+ await bs(["kill", "-s", id]);
2120
+ return { error: `subagent ${id} failed before its first model response: ${initialProgress.errorMsg}` };
2121
+ }
2122
+ } catch {
2123
+ /* normal startup continues; the full stream remains available to check/wait */
2124
+ }
2125
+ }
1890
2126
  // Report the RESOLVED model (a fuzzy pattern may match something unexpected;
1891
2127
  // null means nothing resolved at all).
1892
2128
  let resolvedModel: string | undefined;
@@ -1947,8 +2183,10 @@ const WIDGET_TAIL_WIDTH = 100;
1947
2183
  // Strip ANSI/control escapes and clamp width so raw PTY output can't wrap or
1948
2184
  // corrupt the widget area.
1949
2185
  function sanitizeTailLine(s: string): string {
1950
- const clean = s
1951
- .replace(/\r/g, "")
2186
+ // PTY progress bars often redraw one logical line with carriage returns.
2187
+ // Keep the latest frame rather than concatenating every historical frame.
2188
+ const terminalFrame = s.split("\r").filter(Boolean).at(-1) ?? "";
2189
+ const clean = terminalFrame
1952
2190
  // CSI / OSC / other escape sequences
1953
2191
  .replace(/\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)/g, "")
1954
2192
  .replace(/\x1b[@-Z\\-_]|\x1b\[[0-?]*[ -/]*[@-~]/g, "")
@@ -1957,9 +2195,34 @@ function sanitizeTailLine(s: string): string {
1957
2195
  return clean.length > WIDGET_TAIL_WIDTH ? `${clean.slice(0, WIDGET_TAIL_WIDTH - 1)}…` : clean;
1958
2196
  }
1959
2197
 
2198
+ function readTailLines(file: string, lines: number, maxBytes = 64_000): string[] {
2199
+ try {
2200
+ const size = fs.statSync(file).size;
2201
+ const start = Math.max(0, size - maxBytes);
2202
+ const length = size - start;
2203
+ const bytes = Buffer.allocUnsafe(length);
2204
+ const fd = fs.openSync(file, "r");
2205
+ let read = 0;
2206
+ try {
2207
+ while (read < length) {
2208
+ const count = fs.readSync(fd, bytes, read, length - read, start + read);
2209
+ if (count === 0) break;
2210
+ read += count;
2211
+ }
2212
+ } finally {
2213
+ fs.closeSync(fd);
2214
+ }
2215
+ const parts = bytes.subarray(0, read).toString("utf8").split("\n");
2216
+ if (start > 0) parts.shift(); // first fragment may begin mid-line
2217
+ return parts.slice(-Math.max(1, lines + 1));
2218
+ } catch {
2219
+ return [];
2220
+ }
2221
+ }
2222
+
1960
2223
  // Trailing lines to show for a running session (sanitized, unprefixed).
1961
- // process → raw log tail; subagent → derived activity (recent tool calls /
1962
- // partial answer), since its log is RPC JSONL, not human-readable.
2224
+ // Process tails are read directly from the bounded end of output.log, avoiding
2225
+ // one `babysit log` subprocess per active process on every poll.
1963
2226
  async function widgetTail(
1964
2227
  id: string,
1965
2228
  isSub: boolean,
@@ -1967,8 +2230,7 @@ async function widgetTail(
1967
2230
  ): Promise<string[]> {
1968
2231
  let raw: string[];
1969
2232
  if (!isSub) {
1970
- const out = (await bs(["log", "-s", id, "--tail", String(WIDGET_TAIL_LINES)])).stdout;
1971
- raw = out.split("\n");
2233
+ raw = readTailLines(logPath(id), WIDGET_TAIL_LINES);
1972
2234
  } else {
1973
2235
  const progress = subagentProgress ?? taskProgressOf(id).progress;
1974
2236
  if (progress.finalText.trim()) {
@@ -2009,6 +2271,90 @@ interface WaitOutcome {
2009
2271
  progress?: Progress;
2010
2272
  }
2011
2273
 
2274
+ export interface NestedUsage {
2275
+ input: number;
2276
+ output: number;
2277
+ cacheRead: number;
2278
+ cacheWrite: number;
2279
+ totalTokens: number;
2280
+ cost: {
2281
+ input: number;
2282
+ output: number;
2283
+ cacheRead: number;
2284
+ cacheWrite: number;
2285
+ total: number;
2286
+ };
2287
+ }
2288
+
2289
+ export function usageFromProgress(progress: Progress): NestedUsage | undefined {
2290
+ if (progress.modelCalls === 0) return undefined;
2291
+ return {
2292
+ input: progress.inputTokens,
2293
+ output: progress.outputTokens,
2294
+ cacheRead: progress.cacheReadTokens,
2295
+ cacheWrite: progress.cacheWriteTokens,
2296
+ totalTokens: progress.usageTokens,
2297
+ cost: {
2298
+ input: progress.inputCost,
2299
+ output: progress.outputCost,
2300
+ cacheRead: progress.cacheReadCost,
2301
+ cacheWrite: progress.cacheWriteCost,
2302
+ total: progress.cost,
2303
+ },
2304
+ };
2305
+ }
2306
+
2307
+ /** Charge one completed task exactly once, even across concurrent wait callers. */
2308
+ function claimOutcomeUsage(outcome: WaitOutcome): NestedUsage | undefined {
2309
+ if (!outcome.progress || (outcome.kind !== "done" && outcome.kind !== "exited")) return undefined;
2310
+ const usage = usageFromProgress(outcome.progress);
2311
+ if (!usage) return undefined;
2312
+ const meta = readMeta(outcome.id);
2313
+ if (meta?.kind !== "subagent") return undefined;
2314
+ const offset = meta.promptOffset ?? 0;
2315
+ if (meta.usageReportedOffset === offset) return undefined;
2316
+
2317
+ // `open(..., "wx")` is the cross-process compare-and-set. A resumed Pi
2318
+ // session can briefly have overlapping extension processes; metadata alone
2319
+ // would let both read the old value and charge the same nested usage.
2320
+ const marker = path.join(metaDir(), `${outcome.id}.usage-${offset}.claimed`);
2321
+ if (!claimFileOnce(marker, JSON.stringify({ pid: process.pid, claimedAt: Date.now() }))) {
2322
+ return undefined;
2323
+ }
2324
+ meta.usageReportedOffset = offset;
2325
+ writeMeta(outcome.id, meta); // compatibility/display hint; marker is authoritative
2326
+ return usage;
2327
+ }
2328
+
2329
+ function sumNestedUsage(values: Array<NestedUsage | undefined>): NestedUsage | undefined {
2330
+ const present = values.filter((value): value is NestedUsage => Boolean(value));
2331
+ if (present.length === 0) return undefined;
2332
+ return present.reduce<NestedUsage>(
2333
+ (total, value) => ({
2334
+ input: total.input + value.input,
2335
+ output: total.output + value.output,
2336
+ cacheRead: total.cacheRead + value.cacheRead,
2337
+ cacheWrite: total.cacheWrite + value.cacheWrite,
2338
+ totalTokens: total.totalTokens + value.totalTokens,
2339
+ cost: {
2340
+ input: total.cost.input + value.cost.input,
2341
+ output: total.cost.output + value.cost.output,
2342
+ cacheRead: total.cost.cacheRead + value.cost.cacheRead,
2343
+ cacheWrite: total.cost.cacheWrite + value.cost.cacheWrite,
2344
+ total: total.cost.total + value.cost.total,
2345
+ },
2346
+ }),
2347
+ {
2348
+ input: 0,
2349
+ output: 0,
2350
+ cacheRead: 0,
2351
+ cacheWrite: 0,
2352
+ totalTokens: 0,
2353
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
2354
+ },
2355
+ );
2356
+ }
2357
+
2012
2358
  // Wait for ONE subagent's current task. Completion = agent_settled without a
2013
2359
  // parked babysit_run/process result (a parked run only means "waiting for
2014
2360
  // process exit; pi will resume itself"). Parse appended bytes incrementally,
@@ -2031,7 +2377,7 @@ async function waitForTask(
2031
2377
  const st = await statusOf(id);
2032
2378
 
2033
2379
  const stats =
2034
- `turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCalls.length}` +
2380
+ `turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}` +
2035
2381
  (prog.tokens != null ? ` ctx=${prog.tokens}` : "") +
2036
2382
  (prog.modelCalls > 0
2037
2383
  ? ` usage=${prog.usageTokens} (in=${prog.inputTokens} out=${prog.outputTokens} cache=${prog.cacheReadTokens}) $${prog.cost.toFixed(4)}`
@@ -2045,7 +2391,7 @@ async function waitForTask(
2045
2391
  ok: completed.ok,
2046
2392
  text:
2047
2393
  `Subagent ${id} finished its task (${stats}).\n` +
2048
- `Session stays alive — follow-up: babysit_send { id: "${id}" }, ` +
2394
+ `${SUBAGENT_REUSE_HINT} Follow-up: babysit_send { id: "${id}" }, ` +
2049
2395
  `or babysit_kill when done.\n\n${completed.body}`,
2050
2396
  status: st,
2051
2397
  progress: prog,
@@ -2189,6 +2535,7 @@ async function waitForExit(
2189
2535
  limitMs: number | null,
2190
2536
  signal?: AbortSignal,
2191
2537
  expectPattern?: string,
2538
+ outputSelection?: ProcessOutputSelection,
2192
2539
  ): Promise<WaitOutcome> {
2193
2540
  const t = limitMs != null ? `${Math.ceil(limitMs / 1000)}s` : "0";
2194
2541
 
@@ -2277,7 +2624,7 @@ async function waitForExit(
2277
2624
  const meta = readMeta(id);
2278
2625
  const workerDead = st.state === "dead" && st.exit_code == null;
2279
2626
  const ok = st.exit_code === 0;
2280
- const output = await inlineOutput(id, st);
2627
+ const output = await selectedProcessOutput(id, st, outputSelection, signal);
2281
2628
  return {
2282
2629
  id,
2283
2630
  kind: "exited",
@@ -2367,6 +2714,21 @@ export function automaticNotificationGroup(entry: unknown): string | undefined {
2367
2714
  return runs.length >= 2 ? group : undefined;
2368
2715
  }
2369
2716
 
2717
+ export function resolveSubagentSendMode(
2718
+ requested: "auto" | "steer" | "task",
2719
+ streaming?: boolean,
2720
+ currentTaskDone?: boolean,
2721
+ ): { mode: "steer" | "task" } | { error: "busy" | "unknown" | "unsettled" } {
2722
+ if (requested === "steer") return { mode: "steer" };
2723
+ if (requested === "auto") {
2724
+ return { mode: streaming === false && currentTaskDone === true ? "task" : "steer" };
2725
+ }
2726
+ if (streaming === true) return { error: "busy" };
2727
+ if (streaming === undefined || currentTaskDone === undefined) return { error: "unknown" };
2728
+ if (!currentTaskDone) return { error: "unsettled" };
2729
+ return { mode: "task" };
2730
+ }
2731
+
2370
2732
  // ---------------------------------------------------------------------------
2371
2733
  // extension
2372
2734
  // ---------------------------------------------------------------------------
@@ -2421,78 +2783,88 @@ export default function (pi: ExtensionAPI) {
2421
2783
  });
2422
2784
 
2423
2785
  async function enforceSubagentBudgets(sessions: BsSession[]): Promise<void> {
2424
- for (const session of sessions) {
2425
- if (session.state !== "running") continue;
2426
- await withSessionRpcLock(session.id, async () => {
2427
- const latestStatus = await statusOf(session.id);
2428
- const meta = readMeta(session.id);
2429
- if (
2430
- latestStatus?.state !== "running" ||
2431
- meta?.kind !== "subagent" ||
2432
- !meta.budget ||
2433
- meta.budgetKilled
2434
- ) {
2435
- return;
2436
- }
2437
-
2438
- let progress: Progress;
2439
- try {
2440
- progress = taskProgressOf(session.id).progress;
2441
- } catch {
2442
- return;
2443
- }
2444
- if (progress.done) return;
2445
- const now = Date.now();
2446
- const decision = subagentBudgetAction(
2447
- progress,
2448
- meta.budget,
2449
- meta.budgetExceededAt,
2450
- now,
2451
- SUBAGENT_BUDGET_GRACE_MS,
2452
- );
2453
- if (decision.action === "none") return;
2454
- const reason = decision.reason as string;
2455
-
2456
- if (decision.action === "steer") {
2457
- // Start the grace period only after Pi accepts the steering command.
2458
- const sent = await sendRpc(session.id, {
2459
- type: "steer",
2460
- message: `Budget reached (${reason}). Stop calling tools and provide your final answer now.`,
2461
- });
2462
- if ("error" in sent) return;
2463
- const accepted = await rpcResponse(session.id, sent.offset, "steer", "3s");
2464
- if (!accepted.ok) return;
2465
- const current = readMeta(session.id);
2466
- if (
2467
- current?.kind !== "subagent" ||
2468
- current.promptOffset !== meta.promptOffset ||
2469
- current.budgetExceededAt
2470
- ) {
2471
- return;
2472
- }
2473
- current.budgetExceededAt = now;
2474
- current.budgetReason = reason;
2475
- writeMeta(session.id, current);
2476
- return;
2477
- }
2786
+ // Independent workers must not serialize 3-second RPC probes and delay
2787
+ // unrelated completion notifications. Per-session RPC locks still preserve
2788
+ // ordering within each worker.
2789
+ await Promise.all(
2790
+ sessions
2791
+ .filter((session) => session.state === "running")
2792
+ .map((session) =>
2793
+ withSessionRpcLock(session.id, async () => {
2794
+ const meta = readMeta(session.id);
2795
+ if (meta?.kind !== "subagent" || !meta.budget || meta.budgetKilled) return;
2796
+
2797
+ let progress: Progress;
2798
+ try {
2799
+ progress = taskProgressOf(session.id).progress;
2800
+ } catch {
2801
+ return;
2802
+ }
2803
+ if (progress.done) return;
2804
+ const now = Date.now();
2805
+ const hardReason = subagentBudgetViolation(progress, meta.budget);
2806
+ const softReason = subagentBudgetSoftViolation(progress, meta.budget);
2807
+
2808
+ if (hardReason) {
2809
+ if (meta.budgetExceededAt == null) {
2810
+ // The hard grace begins when the violation is observed, even if a
2811
+ // wedged worker never accepts steering. This makes the cap enforceable.
2812
+ meta.budgetExceededAt = now;
2813
+ meta.budgetReason = hardReason;
2814
+ writeMeta(session.id, meta);
2815
+ const latestStatus = await statusOf(session.id);
2816
+ if (latestStatus?.state !== "running") return;
2817
+ const sent = await sendRpc(session.id, {
2818
+ type: "steer",
2819
+ message: `Hard budget reached (${hardReason}). Stop calling tools and return your best answer now.`,
2820
+ });
2821
+ if (!("error" in sent)) await rpcResponse(session.id, sent.offset, "steer", "3s");
2822
+ return;
2823
+ }
2824
+ if (now - meta.budgetExceededAt < SUBAGENT_BUDGET_GRACE_MS) return;
2825
+ const latestStatus = await statusOf(session.id);
2826
+ if (latestStatus?.state !== "running") return;
2827
+ const killed = await bs(["kill", "-s", session.id, "--json"]);
2828
+ if (killed.code !== 0) return;
2829
+ const terminal = await awaitConfirmedTermination(session.id);
2830
+ const current = readMeta(session.id);
2831
+ if (
2832
+ terminal &&
2833
+ isConfirmedTerminalState(terminal.state) &&
2834
+ current?.kind === "subagent" &&
2835
+ current.promptOffset === meta.promptOffset &&
2836
+ current.budgetExceededAt === meta.budgetExceededAt
2837
+ ) {
2838
+ current.budgetKilled = true;
2839
+ current.budgetReason = current.budgetReason ?? hardReason;
2840
+ writeMeta(session.id, current);
2841
+ }
2842
+ return;
2843
+ }
2478
2844
 
2479
- const killed = await bs(["kill", "-s", session.id, "--json"]);
2480
- if (killed.code !== 0) return;
2481
- const terminal = await awaitConfirmedTermination(session.id);
2482
- const current = readMeta(session.id);
2483
- if (
2484
- terminal &&
2485
- isConfirmedTerminalState(terminal.state) &&
2486
- current?.kind === "subagent" &&
2487
- current.promptOffset === meta.promptOffset &&
2488
- current.budgetExceededAt === meta.budgetExceededAt
2489
- ) {
2490
- current.budgetKilled = true;
2491
- current.budgetReason = current.budgetReason ?? reason;
2492
- writeMeta(session.id, current);
2493
- }
2494
- });
2495
- }
2845
+ if (!softReason || meta.budgetWarnedAt != null) return;
2846
+ const latestStatus = await statusOf(session.id);
2847
+ if (latestStatus?.state !== "running") return;
2848
+ const sent = await sendRpc(session.id, {
2849
+ type: "steer",
2850
+ message: `Budget is approaching its limit (${softReason}). Wrap up now and preserve your best findings.`,
2851
+ });
2852
+ if ("error" in sent) return;
2853
+ const accepted = await rpcResponse(session.id, sent.offset, "steer", "3s");
2854
+ if (!accepted.ok) return;
2855
+ const current = readMeta(session.id);
2856
+ if (
2857
+ current?.kind === "subagent" &&
2858
+ current.promptOffset === meta.promptOffset &&
2859
+ current.budgetWarnedAt == null
2860
+ ) {
2861
+ current.budgetWarnedAt = now;
2862
+ current.budgetWarningReason = softReason;
2863
+ writeMeta(session.id, current);
2864
+ }
2865
+ }),
2866
+ ),
2867
+ );
2496
2868
  }
2497
2869
 
2498
2870
  // Exit notifications for kind=process sessions: the poller detects
@@ -2739,9 +3111,15 @@ export default function (pi: ExtensionAPI) {
2739
3111
  }
2740
3112
  }
2741
3113
  taskProgressCache.clear();
3114
+ searchLogCache.clear();
2742
3115
  pollNeeded = true;
2743
- const retentionDays = Number(process.env.PI_BABYSIT_RETENTION_DAYS);
2744
- if (!automaticGcRan && Number.isFinite(retentionDays) && retentionDays > 0) {
3116
+ const retentionDays = Number(process.env.PI_BABYSIT_RETENTION_DAYS ?? "3");
3117
+ if (
3118
+ !automaticGcRan &&
3119
+ Number.isFinite(retentionDays) &&
3120
+ retentionDays > 0 &&
3121
+ automaticGcDue()
3122
+ ) {
2745
3123
  automaticGcRan = true;
2746
3124
  const gc = gcBabysitRoots({
2747
3125
  rootBase: ROOT_BASE,
@@ -2749,6 +3127,7 @@ export default function (pi: ExtensionAPI) {
2749
3127
  olderThanMs: retentionDays * 86_400_000,
2750
3128
  dryRun: false,
2751
3129
  });
3130
+ markAutomaticGc();
2752
3131
  if (gc.deleted.length > 0 && ctx.hasUI) {
2753
3132
  ctx.ui.notify(
2754
3133
  `pi-babysit GC removed ${gc.deleted.length} roots (${gc.bytes} bytes).`,
@@ -2780,6 +3159,7 @@ export default function (pi: ExtensionAPI) {
2780
3159
  notifyEndedProcesses(ctx, snapshot),
2781
3160
  refreshWidget(ctx, snapshot),
2782
3161
  ]);
3162
+ pruneTerminalSessionCache(taskProgressCache, snapshot);
2783
3163
  pollNeeded = shouldKeepPolling(snapshot, readMeta);
2784
3164
  })()
2785
3165
  .catch(() => {
@@ -2843,35 +3223,18 @@ export default function (pi: ExtensionAPI) {
2843
3223
  name: "babysit_run",
2844
3224
  label: "Babysit: run",
2845
3225
  description:
2846
- "Run any shell command in a supervised babysit session. Set `foreground: true` when the next step " +
2847
- "needs the exit result in this tool call. Otherwise commands that finish within a short grace period " +
2848
- "return completion metadata immediately; longer commands continue in the background and trigger an " +
2849
- "automatic notification on exit. Sibling background runs in one assistant message are grouped " +
2850
- "automatically. Complete output is returned inline only when it is small; larger output stays " +
2851
- "in the log path for bounded inspection with babysit_check. " +
2852
- "In non-interactive mode (`pi -p`, no UI), process mode blocks until exit because there is no " +
2853
- "notification loop. Two modes: (1) `command` — run any shell command, including builds, tests, " +
2854
- "dev servers, watchers, and interactive TUIs; you can type into it with babysit_send and read " +
2855
- "its screen with babysit_check. If a worker disappears during startup without recording an exit, " +
2856
- "`retryOnWorkerDeath` can retry one idempotent command once. " +
2857
- "(2) `profile: \"subagent\"` + `task` — spawn a pi subagent that works on the task in the " +
2858
- "background; poll with babysit_check, steer with babysit_send, block with babysit_wait, " +
2859
- "stop with babysit_kill. Subagents cannot recursively spawn more subagents by default; " +
2860
- "the top-level caller must explicitly raise `maxDepth` when creating the first worker.",
2861
- promptSnippet:
2862
- "Run any shell command with context-safe captured output; quick commands return metadata, longer ones continue in background",
3226
+ "Run a supervised shell command, or start a reusable pi subagent with `profile: \"subagent\"`. " +
3227
+ "Use `foreground` for results needed now; otherwise long commands notify on exit. Full logs stay on disk. " +
3228
+ "`returnPattern`/`returnLines` bound foreground output. Sessions support check, wait, send, and kill.",
3229
+ promptSnippet: "Run supervised commands or bounded pi subagents with context-safe logs",
2863
3230
  promptGuidelines: [
2864
- "Use babysit_run as the default for shell commands, not only long-running work. Small output is returned directly; large stdout/stderr stays out of model context in the returned log path. Give every meaningful process or subagent a clear stable `name`.",
2865
- "Use babysit_run { command, foreground: true } when the result is required before the next step; this avoids a separate babysit_wait model turn. Do not use foreground for servers, watchers, or commands of unknown duration without a timeout.",
2866
- "Bundle closely related tiny observations into one babysit_run command when that reduces tool turns without obscuring lifecycle or failure handling.",
2867
- "Inspect a babysit log with babysit_check { id, lines, pattern? }; never read or cat a potentially large log file in full. Prefer a targeted `pattern` search over returning a broad tail.",
2868
- "After babysit_run { command } starts a process, end your response immediately so the automatic process-end notification can resume you; NEVER poll with babysit_check or sleep. Set continueAfterStart: true only when you have immediate, specific, non-polling work to do next. Call babysit_wait when you must consume the result inside the current turn (optionally with `expect` to wait for a readiness line like 'listening on').",
2869
- "If a babysit worker is killed externally, babysit_run reports it as worker-dead rather than hanging. Set retryOnWorkerDeath: true only for safe, idempotent commands; it retries at most once and may otherwise duplicate side effects.",
2870
- "babysit_run gives full PTY control: drive interactive programs (installers, wizards, REPLs) with babysit_send (text or named keys) and read the rendered screen with babysit_check { screen: true }.",
2871
- "Delegate self-contained tasks (codebase recon, a parallelizable subtask, work that would pollute your context) with babysit_run { profile: \"subagent\", task }. Launch several for independent subtasks; they run concurrently.",
2872
- "Set at least one subagent budget (`maxCost`, `maxTurns`, `maxToolCalls`, or `maxUsageTokens`) for bounded recon and review tasks. Omit budgets only for intentionally open-ended work; the absolute timeout remains a separate safety limit.",
2873
- "Subagents cannot create further subagents by default (maximum depth 1). Only the top-level caller can explicitly opt in by setting maxDepth when it creates the first worker; nested workers inherit that limit and cannot raise it.",
2874
- "After spawning subagents, do not idle-wait and do not end your turn to wait for them: keep making progress, then call babysit_wait (ids + mode any/all) when you need their results. Steer or send follow-up tasks with babysit_send; kill runaways with babysit_kill.",
3231
+ "Use babysit_run for shell commands and give meaningful sessions a stable name; bundle tiny related observations.",
3232
+ "Use babysit_run foreground mode when the next step needs the result; use returnPattern/returnLines for noisy commands. Do not foreground unbounded servers.",
3233
+ "After a background process starts, stop the turn for its automatic notification; never poll or sleep. Use continueAfterStart only for specific non-polling work.",
3234
+ "Inspect large logs with a narrow babysit_check pattern and maxBytes rather than broad tails.",
3235
+ "Use retryOnWorkerDeath only once and only for idempotent commands; retries may duplicate side effects.",
3236
+ "Delegate independent work with bounded babysit_run subagents; set at least one cost/turn/tool/token budget and keep making progress before babysit_wait.",
3237
+ "Subagent recursion defaults to depth 1; only a top-level caller may explicitly raise maxDepth.",
2875
3238
  ],
2876
3239
  parameters: Type.Object({
2877
3240
  command: Type.Optional(
@@ -2913,15 +3276,24 @@ export default function (pi: ExtensionAPI) {
2913
3276
  }),
2914
3277
  ),
2915
3278
  maxTurns: Type.Optional(
2916
- Type.Integer({ minimum: 1, description: "Subagent only: maximum turns per task." }),
3279
+ Type.Integer({
3280
+ minimum: 1,
3281
+ description:
3282
+ "Subagent only: observed turn threshold for steering it to wrap up. In-flight work can overshoot before the poller intervenes.",
3283
+ }),
2917
3284
  ),
2918
3285
  maxToolCalls: Type.Optional(
2919
- Type.Integer({ minimum: 1, description: "Subagent only: maximum tool calls per task." }),
3286
+ Type.Integer({
3287
+ minimum: 1,
3288
+ description:
3289
+ "Subagent only: observed tool-call threshold for steering it to wrap up. A parallel in-flight tool batch can overshoot.",
3290
+ }),
2920
3291
  ),
2921
3292
  maxUsageTokens: Type.Optional(
2922
3293
  Type.Integer({
2923
3294
  minimum: 1,
2924
- description: "Subagent only: maximum cumulative reported totalTokens across model calls.",
3295
+ description:
3296
+ "Subagent only: observed cumulative reported totalTokens threshold for steering it to wrap up.",
2925
3297
  }),
2926
3298
  ),
2927
3299
  agentScope: Type.Optional(
@@ -2949,10 +3321,18 @@ export default function (pi: ExtensionAPI) {
2949
3321
  ),
2950
3322
  foreground: Type.Optional(
2951
3323
  Type.Boolean({
2952
- description:
2953
- "Process mode: wait for exit and return the result in this tool call. Use when the next step needs the result; avoid for servers/watchers unless bounded by timeout.",
3324
+ description: "Process: wait for exit and return the result now.",
2954
3325
  }),
2955
3326
  ),
3327
+ returnPattern: Type.Optional(
3328
+ Type.String({ description: "Foreground/quick process: return only latest regex matches." }),
3329
+ ),
3330
+ returnLines: Type.Optional(
3331
+ Type.Integer({ minimum: 1, maximum: 200, description: "Lines retained by returnPattern/tail (default 30)." }),
3332
+ ),
3333
+ maxBytes: Type.Optional(
3334
+ Type.Integer({ minimum: 1_000, maximum: ANSWER_MAX_BYTES, description: "Returned process-output cap (default 8 KB)." }),
3335
+ ),
2956
3336
  notificationGroup: Type.Optional(
2957
3337
  Type.String({
2958
3338
  description:
@@ -3023,6 +3403,13 @@ export default function (pi: ExtensionAPI) {
3023
3403
  details: {},
3024
3404
  };
3025
3405
  }
3406
+ if (isSubagent && (params.returnPattern || params.returnLines != null || params.maxBytes != null)) {
3407
+ return {
3408
+ content: [{ type: "text", text: "`returnPattern`, `returnLines`, and `maxBytes` are process-output options." }],
3409
+ isError: true,
3410
+ details: {},
3411
+ };
3412
+ }
3026
3413
  if (!isSubagent && params.foreground && params.continueAfterStart) {
3027
3414
  return {
3028
3415
  content: [{ type: "text", text: "`foreground` and `continueAfterStart` are mutually exclusive." }],
@@ -3046,6 +3433,21 @@ export default function (pi: ExtensionAPI) {
3046
3433
 
3047
3434
  // --- process mode ---
3048
3435
  if (!isSubagent) {
3436
+ if (params.returnPattern) {
3437
+ try {
3438
+ new RegExp(params.returnPattern);
3439
+ } catch (error) {
3440
+ return {
3441
+ content: [{ type: "text", text: `Invalid returnPattern: ${String(error)}` }],
3442
+ isError: true,
3443
+ details: {},
3444
+ };
3445
+ }
3446
+ }
3447
+ const outputSelection: ProcessOutputSelection | undefined =
3448
+ params.returnPattern || params.returnLines != null || params.maxBytes != null
3449
+ ? { pattern: params.returnPattern, lines: params.returnLines, maxBytes: params.maxBytes }
3450
+ : undefined;
3049
3451
  const spawnOpts: ProcOpts = {
3050
3452
  name: params.name,
3051
3453
  command: params.command as string,
@@ -3076,14 +3478,14 @@ export default function (pi: ExtensionAPI) {
3076
3478
  // the same deadline here races its terminal-state write and can return a
3077
3479
  // false "still running" result at the boundary, so wait for the
3078
3480
  // supervisor's definitive exit instead.
3079
- let outcome = await waitForExit(res.id, null, _signal);
3481
+ let outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
3080
3482
  let retried = false;
3081
3483
  if (params.retryOnWorkerDeath && outcome.status?.state === "dead" && outcome.status.exit_code == null) {
3082
3484
  const retry = await spawnProcess(spawnOpts);
3083
3485
  if (!("error" in retry)) {
3084
3486
  res = retry;
3085
3487
  retried = true;
3086
- outcome = await waitForExit(res.id, null, _signal);
3488
+ outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
3087
3489
  }
3088
3490
  }
3089
3491
  if (ctx.hasUI) await refreshWidget(ctx);
@@ -3121,7 +3523,7 @@ export default function (pi: ExtensionAPI) {
3121
3523
  }
3122
3524
  }
3123
3525
  if (quickStatus && quickStatus.state !== "running") {
3124
- const outcome = await waitForExit(res.id, null, _signal);
3526
+ const outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
3125
3527
  await refreshWidget(ctx);
3126
3528
  return {
3127
3529
  content: [{ type: "text", text: `${retried ? "Retried once after external worker death.\n" : ""}${outcome.text}` }],
@@ -3323,6 +3725,13 @@ export default function (pi: ExtensionAPI) {
3323
3725
  lines: Type.Optional(
3324
3726
  Type.Number({ description: "How many tail lines or latest matches to show (default 30, max 200)." }),
3325
3727
  ),
3728
+ maxBytes: Type.Optional(
3729
+ Type.Integer({
3730
+ minimum: 1_000,
3731
+ maximum: ANSWER_MAX_BYTES,
3732
+ description: "Maximum returned bytes for this check (default 4 KB).",
3733
+ }),
3734
+ ),
3326
3735
  pattern: Type.Optional(
3327
3736
  Type.String({
3328
3737
  description:
@@ -3410,6 +3819,7 @@ export default function (pi: ExtensionAPI) {
3410
3819
  }
3411
3820
  const meta = readMeta(params.id);
3412
3821
  const nLines = Math.min(Math.max(1, Math.floor(params.lines ?? 30)), 200);
3822
+ const checkMaxBytes = params.maxBytes ?? TAIL_MAX_BYTES;
3413
3823
  if (params.pattern !== undefined) {
3414
3824
  if (params.screen) {
3415
3825
  return {
@@ -3425,7 +3835,7 @@ export default function (pi: ExtensionAPI) {
3425
3835
  details: {},
3426
3836
  };
3427
3837
  }
3428
- const result = await searchLog(params.id, params.pattern, nLines, signal);
3838
+ const result = await searchLog(params.id, params.pattern, nLines, signal, checkMaxBytes);
3429
3839
  if (result.error) {
3430
3840
  return {
3431
3841
  content: [{ type: "text", text: result.error }],
@@ -3439,7 +3849,7 @@ export default function (pi: ExtensionAPI) {
3439
3849
  ? `--- latest matches /${params.pattern}/ ---\n${result.text}`
3440
3850
  : `(no output matching /${params.pattern}/)`;
3441
3851
  return {
3442
- content: [{ type: "text", text: clip(`${header}\n${body}`) }],
3852
+ content: [{ type: "text", text: clip(`${header}\n${body}`, checkMaxBytes) }],
3443
3853
  details: { status: st, kind, logPath: logPath(params.id), pattern: params.pattern },
3444
3854
  };
3445
3855
  }
@@ -3453,21 +3863,22 @@ export default function (pi: ExtensionAPI) {
3453
3863
  if (el) header += ` elapsed=${el}`;
3454
3864
  }
3455
3865
  if (st.exit_code != null) header += ` exit_code=${st.exit_code}`;
3456
- if (meta?.command) header += `\ncommand: ${meta.command}`;
3866
+ if (meta?.command) header += `\ncommand: ${summarizeNotificationCommand(meta.command)}`;
3457
3867
  header += `\nlog: ${logPath(params.id)}`;
3458
3868
  if (st.note) header += ` ⚑ ${st.note}`;
3459
3869
  parts.push(header);
3460
3870
  if (params.screen) {
3461
3871
  const sc = await bs(["screenshot", "-s", params.id, "--trim"]);
3462
- parts.push(`--- screen ---\n${clip(sc.stdout.trimEnd()) || "(blank screen)"}`);
3872
+ parts.push(`--- screen ---\n${clip(sc.stdout.trimEnd(), checkMaxBytes) || "(blank screen)"}`);
3463
3873
  } else {
3464
3874
  const tail = clip(
3465
3875
  (await bs(["log", "-s", params.id, "--tail", String(nLines)])).stdout.trimEnd(),
3876
+ checkMaxBytes,
3466
3877
  );
3467
3878
  parts.push(tail ? `--- recent output ---\n${tail}` : "(no output yet)");
3468
3879
  }
3469
3880
  return {
3470
- content: [{ type: "text", text: clip(parts.join("\n")) }],
3881
+ content: [{ type: "text", text: clip(parts.join("\n"), checkMaxBytes) }],
3471
3882
  details: { status: st, kind: "process", logPath: logPath(params.id) },
3472
3883
  };
3473
3884
  }
@@ -3490,7 +3901,7 @@ export default function (pi: ExtensionAPI) {
3490
3901
  : " · working";
3491
3902
  }
3492
3903
  if (st.exit_code != null) header += ` exit_code=${st.exit_code}`;
3493
- header += ` turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCalls.length}`;
3904
+ header += ` turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}`;
3494
3905
  if (prog.tokens != null) header += ` ctx=${prog.tokens}`;
3495
3906
  if (prog.modelCalls > 0) header += ` usage=${prog.usageTokens} $${prog.cost.toFixed(4)}`;
3496
3907
  if (st.note) header += ` ⚑ ${st.note}`;
@@ -3499,7 +3910,7 @@ export default function (pi: ExtensionAPI) {
3499
3910
  if (prog.errorMsg) parts.push(`⚠ error: ${clip(prog.errorMsg, ANSWER_MAX_BYTES)}`);
3500
3911
 
3501
3912
  if (recent.length > 0) {
3502
- const skipped = prog.toolCalls.length - recent.length;
3913
+ const skipped = Math.max(0, prog.toolCallCount - recent.length);
3503
3914
  parts.push(
3504
3915
  `--- recent tool calls${skipped > 0 ? ` (+${skipped} earlier)` : ""} ---\n` +
3505
3916
  recent.map((t) => ` ${t.summary}`).join("\n"),
@@ -3508,16 +3919,16 @@ export default function (pi: ExtensionAPI) {
3508
3919
 
3509
3920
  if (prog.finalText.trim()) {
3510
3921
  parts.push(`--- answer so far ---\n${clip(prog.finalText.trim(), ANSWER_MAX_BYTES)}`);
3511
- } else if (prog.toolCalls.length === 0 && st.state !== "running") {
3922
+ } else if (prog.toolCallCount === 0 && st.state !== "running") {
3512
3923
  parts.push(buildSubagentExitDiagnostic(prog, logPath(params.id)));
3513
- } else if (prog.toolCalls.length === 0) {
3924
+ } else if (prog.toolCallCount === 0) {
3514
3925
  parts.push("(starting up… no events yet)");
3515
3926
  } else {
3516
3927
  parts.push("(working… no answer text yet)");
3517
3928
  }
3518
3929
 
3519
3930
  return {
3520
- content: [{ type: "text", text: clip(parts.join("\n")) }],
3931
+ content: [{ type: "text", text: clip(parts.join("\n"), checkMaxBytes) }],
3521
3932
  details: { status: st, progress: prog, kind: "subagent" },
3522
3933
  };
3523
3934
  },
@@ -3531,7 +3942,7 @@ export default function (pi: ExtensionAPI) {
3531
3942
  "Send input to a babysit session. Process: `text` types a line into its stdin (PTY), " +
3532
3943
  "`keys` presses named keys (Enter, Tab, Esc, Up/Down/Left/Right, C-c, F1…) — use with " +
3533
3944
  "babysit_check { screen: true } to drive interactive programs. Subagent: `text` is " +
3534
- "STEERING while it works, or a NEW TASK when it is idle (mode: auto/steer/task) — this " +
3945
+ "STEERING while it works, or a NEW TASK after the current task settles (mode: auto/steer/task) — this " +
3535
3946
  "is how you resume a finished subagent with full context.",
3536
3947
  promptSnippet: "Send text/keys to a process, or steering/follow-up tasks to a subagent",
3537
3948
  parameters: Type.Object({
@@ -3547,7 +3958,7 @@ export default function (pi: ExtensionAPI) {
3547
3958
  mode: Type.Optional(
3548
3959
  StringEnum(["auto", "steer", "task"] as const, {
3549
3960
  description:
3550
- "Subagent only. auto (default): steer if mid-run, otherwise start a new task. steer/task force one behavior.",
3961
+ "Subagent only. auto (default): steer unless the current task is settled. task requires confirmed settlement; steer always sends guidance.",
3551
3962
  }),
3552
3963
  ),
3553
3964
  noNewline: Type.Optional(
@@ -3647,15 +4058,39 @@ export default function (pi: ExtensionAPI) {
3647
4058
  };
3648
4059
  }
3649
4060
  let mode = params.mode ?? "auto";
3650
- if (mode === "auto") {
3651
- // isStreaming tells us whether an agent run is in flight right now.
4061
+ if (mode === "auto" || mode === "task") {
4062
+ // A prompt sent while the current run is streaming can queue behind that
4063
+ // run while immediately replacing our per-task offsets and budget state.
4064
+ // Establish idleness before every new task; auto safely falls back to
4065
+ // steering when state is unknown, while an explicit task fails closed.
3652
4066
  const gs = await sendRpc(params.id, { type: "get_state" });
3653
- let streaming = true; // assume busy when unsure — steering is the safe default
4067
+ let streaming: boolean | undefined;
3654
4068
  if (!("error" in gs)) {
3655
4069
  const r = await rpcResponse(params.id, gs.offset, "get_state", "10s");
3656
4070
  if (r.ok) streaming = Boolean((r.data as { isStreaming?: boolean })?.isStreaming);
3657
4071
  }
3658
- mode = streaming ? "steer" : "task";
4072
+ let currentTaskDone: boolean | undefined;
4073
+ try {
4074
+ currentTaskDone = taskProgressOf(params.id).progress.done;
4075
+ } catch {
4076
+ /* fail closed below rather than replacing unknown task state */
4077
+ }
4078
+ const resolved = resolveSubagentSendMode(mode, streaming, currentTaskDone);
4079
+ if ("error" in resolved) {
4080
+ return {
4081
+ content: [{
4082
+ type: "text",
4083
+ text: resolved.error === "busy"
4084
+ ? `Subagent ${params.id} is still streaming; use mode \"steer\" or wait for the current task to settle before starting another task.`
4085
+ : resolved.error === "unsettled"
4086
+ ? `Subagent ${params.id} has not settled its current task (it may be parked on a background process); wait for completion before starting another task.`
4087
+ : `Could not verify that subagent ${params.id} is idle and settled; retry with mode \"task\" after checking its state.`,
4088
+ }],
4089
+ isError: true,
4090
+ details: { mode: "task" },
4091
+ };
4092
+ }
4093
+ mode = resolved.mode;
3659
4094
  }
3660
4095
  const deliveryCleanupAfter =
3661
4096
  mode === "steer"
@@ -3709,10 +4144,13 @@ export default function (pi: ExtensionAPI) {
3709
4144
  depth: meta?.depth,
3710
4145
  maxDepth: meta?.maxDepth,
3711
4146
  budget: meta?.budget,
3712
- // Each follow-up task receives a fresh budget window.
4147
+ // Each follow-up task receives fresh budget and usage-accounting windows.
4148
+ budgetWarnedAt: undefined,
4149
+ budgetWarningReason: undefined,
3713
4150
  budgetExceededAt: undefined,
3714
4151
  budgetReason: undefined,
3715
4152
  budgetKilled: undefined,
4153
+ usageReportedOffset: undefined,
3716
4154
  });
3717
4155
  } else if (delivery.tempDir && meta) {
3718
4156
  writeMeta(params.id, {
@@ -3747,7 +4185,7 @@ export default function (pi: ExtensionAPI) {
3747
4185
  "Block until babysit session(s) finish, then return the result. A process finishes " +
3748
4186
  "when it EXITS (or, with `expect`, as soon as a regex appears in its output — e.g. wait " +
3749
4187
  "for 'listening on' before hitting a dev server). A subagent finishes when its current " +
3750
- "TASK completes (the session stays alive for follow-ups). Pass `id` for one session, or " +
4188
+ "TASK completes (the worker remains reusable only during its configured idle grace). Pass `id` for one session, or " +
3751
4189
  "`ids` + `mode`: 'all' (default) waits for every one, 'any' returns on the FIRST finisher. " +
3752
4190
  "Multi-session results are capped at the inline-output limit (8 KB by default); use `maxBytes` " +
3753
4191
  "to opt into a larger result up to 24 KB. " +
@@ -3819,9 +4257,11 @@ export default function (pi: ExtensionAPI) {
3819
4257
 
3820
4258
  if (ids.length === 1) {
3821
4259
  const r = await waitFor(ids[0], limitMs, signal, params.expect);
4260
+ const usage = claimOutcomeUsage(r);
3822
4261
  return {
3823
4262
  content: [{ type: "text", text: r.text }],
3824
4263
  isError: !r.ok,
4264
+ usage,
3825
4265
  details: {
3826
4266
  status: r.status,
3827
4267
  progress: r.progress,
@@ -3837,6 +4277,7 @@ export default function (pi: ExtensionAPI) {
3837
4277
  ids.map((i) => waitFor(i, limitMs, signal, params.expect)),
3838
4278
  );
3839
4279
  const ok = results.every((r) => r.ok);
4280
+ const usage = sumNestedUsage(results.map(claimOutcomeUsage));
3840
4281
  return {
3841
4282
  content: [
3842
4283
  {
@@ -3848,6 +4289,7 @@ export default function (pi: ExtensionAPI) {
3848
4289
  },
3849
4290
  ],
3850
4291
  isError: !ok,
4292
+ usage,
3851
4293
  details: {
3852
4294
  results: results.map((r) => ({ id: r.id, kind: r.kind, ok: r.ok })),
3853
4295
  },
@@ -3864,6 +4306,7 @@ export default function (pi: ExtensionAPI) {
3864
4306
  ids.map((i) => waitFor(i, limitMs, ctrl.signal, params.expect)),
3865
4307
  );
3866
4308
  const others = ids.filter((i) => i !== first.id);
4309
+ const usage = claimOutcomeUsage(first);
3867
4310
  return {
3868
4311
  content: [
3869
4312
  {
@@ -3877,6 +4320,7 @@ export default function (pi: ExtensionAPI) {
3877
4320
  },
3878
4321
  ],
3879
4322
  isError: !first.ok,
4323
+ usage,
3880
4324
  details: { first: { id: first.id, kind: first.kind, ok: first.ok }, remaining: others },
3881
4325
  };
3882
4326
  } finally {
@@ -4027,7 +4471,7 @@ export default function (pi: ExtensionAPI) {
4027
4471
  // Parse the RPC event stream and show the final answer, not raw JSONL.
4028
4472
  const prog = taskProgressOf(picked.id).progress;
4029
4473
  const stats =
4030
- `turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCalls.length}` +
4474
+ `turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}` +
4031
4475
  (prog.tokens != null ? ` ctx=${prog.tokens}` : "") +
4032
4476
  (prog.modelCalls > 0 ? ` usage=${prog.usageTokens} $${prog.cost.toFixed(4)}` : "");
4033
4477
  const body =