@yusukeshib/pi-babysit 0.3.17 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -16
- package/index.ts +544 -146
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -33,7 +33,7 @@ reachable from anywhere (`~/.pi-babysit/<pi-session-id>/`). Two kinds:
|
|
|
33
33
|
| kind | started by | completion | on completion |
|
|
34
34
|
| ---- | ---------- | ---------- | ------------- |
|
|
35
35
|
| **process** | `babysit_run { command }` | process **exit** | automatic notification message (`triggerTurn`), batched for all exits observed in the same poll — the agent may end its turn after starting and is resumed on exit, same contract as the old `process` tool |
|
|
36
|
-
| **subagent** | `babysit_run { profile: "subagent", task }` | `agent_settled` in the RPC event stream (
|
|
36
|
+
| **subagent** | `babysit_run { profile: "subagent", task }` | `agent_settled` in the RPC event stream (worker remains reusable during its idle grace) | none — the agent polls `babysit_check` or blocks on `babysit_wait`; the idle session accepts follow-up tasks until self-reap |
|
|
37
37
|
|
|
38
38
|
The **profile is a tool parameter, not a separate tool set**: domain knowledge
|
|
39
39
|
(RPC bookkeeping, per-task byte offsets, parked-turn detection, PTY-safe
|
|
@@ -56,9 +56,9 @@ programs** (installers, wizards, REPLs): type with `babysit_send`
|
|
|
56
56
|
|
|
57
57
|
| Tool | What it does |
|
|
58
58
|
| ---- | ------------ |
|
|
59
|
-
| `babysit_run` | Run any command (`command`, optional `name`/`pty`/`timeout`/`idleTimeout`/`retryOnWorkerDeath`/`notificationGroup`). Set `foreground: true` when the next step needs the result in the same tool call
|
|
60
|
-
| `babysit_check` | Without an id, list
|
|
61
|
-
| `babysit_send` | Process: type `text` / press `keys` into the PTY. Subagent: steer mid-run, or send a follow-up task when
|
|
59
|
+
| `babysit_run` | Run any command (`command`, optional `name`/`pty`/`timeout`/`idleTimeout`/`retryOnWorkerDeath`/`notificationGroup`). Set `foreground: true` when the next step needs the result in the same tool call; use `returnPattern`/`returnLines`/`maxBytes` to keep noisy output bounded without a second check turn. Or start a named subagent (`profile: "subagent"`, `task`, optional `name`/`agent`/`model`/`tools`/`maxDepth` and budget fields). `maxDepth` defaults to 1. Quick commands return inline; longer ones notify in the background |
|
|
60
|
+
| `babysit_check` | Without an id, list sessions with state/kind filters. With an id, inspect bounded output, search with `pattern`, or capture a TUI with `screen: true`; `maxBytes` overrides the 4 KB default up to 24 KB |
|
|
61
|
+
| `babysit_send` | Process: type `text` / press `keys` into the PTY. Subagent: steer mid-run, or send a follow-up task when confirmed settled (`mode: auto/steer/task`); explicit task mode rejects busy, parked, or unknown state |
|
|
62
62
|
| `babysit_wait` | Block until done: process exit (or `expect: "regex"` readiness marker), subagent task completion. Multi-wait: up to 32 unique `ids` + `mode: "any"\|"all"` |
|
|
63
63
|
| `babysit_kill` | Terminate a session, verify terminal state, then suppress the exit notification |
|
|
64
64
|
|
|
@@ -95,10 +95,12 @@ babysit_check { id: "cargo-test", lines: 50 }
|
|
|
95
95
|
babysit_check { id: "cargo-test", pattern: "FAIL|ERROR", lines: 50 }
|
|
96
96
|
```
|
|
97
97
|
|
|
98
|
-
Tail and search results are capped at 200 lines
|
|
99
|
-
|
|
100
|
-
waited-for subagent answer may use up to 24 KB;
|
|
101
|
-
to the 8 KB inline-output limit and can opt
|
|
98
|
+
Tail and search results are capped at 200 lines and default to a 4 KB total
|
|
99
|
+
result cap; `babysit_check.maxBytes` can raise or lower that per call (up to
|
|
100
|
+
24 KB). A single explicitly waited-for subagent answer may use up to 24 KB;
|
|
101
|
+
multi-session wait results default to the 8 KB inline-output limit and can opt
|
|
102
|
+
into a larger cap with `maxBytes`. Foreground runs can apply `returnPattern` or
|
|
103
|
+
`returnLines` before output enters context, avoiding a follow-up check turn.
|
|
102
104
|
Pattern search returns the latest matching lines with line numbers. Prefer a
|
|
103
105
|
targeted pattern over a broad tail, and do not read a potentially large log file
|
|
104
106
|
in full. Subagent crashes return structured errors plus the full log path, never
|
|
@@ -157,12 +159,14 @@ grace window (`PI_BABYSIT_REAP_AFTER`, default 120s) using the same parked-turn
|
|
|
157
159
|
rule, so a subagent waiting on a long build is never false-killed. Give bounded
|
|
158
160
|
recon/review tasks at least one cost, turn, tool-call, or token budget; omit
|
|
159
161
|
budgets only for intentionally open-ended work. Optional task budgets are
|
|
160
|
-
observed by the parent poller.
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
`PI_BABYSIT_BUDGET_GRACE`, termination is verified before
|
|
165
|
-
budget-killed. Usage shown by check/wait is cumulative
|
|
162
|
+
observed by the parent poller. At 80% of a limit the worker is steered to wrap
|
|
163
|
+
up; reaching the configured limit starts the hard grace immediately, even if a
|
|
164
|
+
wedged worker cannot accept steering. An in-flight model call or parallel tool
|
|
165
|
+
batch can still overshoot before the next poll. If the worker remains active
|
|
166
|
+
after `PI_BABYSIT_BUDGET_GRACE`, termination is verified before it is marked
|
|
167
|
+
budget-killed. Usage shown by check/wait is cumulative, and the first terminal
|
|
168
|
+
wait for each task charges that nested usage exactly once to the parent Pi
|
|
169
|
+
session totals.
|
|
166
170
|
|
|
167
171
|
## Environment overrides
|
|
168
172
|
|
|
@@ -176,8 +180,8 @@ budget-killed. Usage shown by check/wait is cumulative for the task.
|
|
|
176
180
|
| `PI_BABYSIT_REAP_AFTER` | `120s` | idle grace before a finished subagent self-exits (`off`/`none`/`0` disables) |
|
|
177
181
|
| `PI_BABYSIT_BUDGET_GRACE` | `90s` | grace after a subagent budget is exceeded before verified termination |
|
|
178
182
|
| `PI_BABYSIT_RPC_LOG_MODE` | `compact` | `compact` removes duplicate RPC lifecycle payloads; `standard` opts into legacy payloads |
|
|
179
|
-
| `PI_BABYSIT_RETENTION_DAYS` |
|
|
180
|
-
| `PI_BABYSIT_TAIL_MAX_BYTES` | `
|
|
183
|
+
| `PI_BABYSIT_RETENTION_DAYS` | `3` | at most once per day, remove safe terminal roots older than this; set `0` to disable automatic retention |
|
|
184
|
+
| `PI_BABYSIT_TAIL_MAX_BYTES` | `4000` | default cap for explicit log tails/screens returned by `babysit_check`; override per call with `maxBytes` |
|
|
181
185
|
| `PI_BABYSIT_INLINE_OUTPUT_MAX_BYTES` | `8000` | cap for complete process output and aggregate multi-wait results |
|
|
182
186
|
| `PI_BABYSIT_NOTIFY_OUTPUT_MAX_BYTES` | `2000` | per-process output cap for unsolicited completion notifications (`0` omits all output) |
|
|
183
187
|
| `PI_BABYSIT_NOTIFY_COMMAND_MAX_BYTES` | `240` | cap for each command preview in completion notifications |
|
package/index.ts
CHANGED
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
* `pi --mode rpc` worker. Tasks are injected as RPC `prompt`
|
|
20
20
|
* commands over stdin, completion is detected from the JSONL
|
|
21
21
|
* event stream (`agent_settled`), NOT process exit; the session
|
|
22
|
-
*
|
|
22
|
+
* remains reusable during its configured idle grace. Same design as the
|
|
23
23
|
* old pi-subagent extension.
|
|
24
24
|
*
|
|
25
25
|
* The "profile" is a tool-parameter, not a separate tool set: one small tool
|
|
@@ -274,11 +274,14 @@ export function isSupportedBabysitVersion(output: string): boolean {
|
|
|
274
274
|
// Cached preflight — probe `babysit --version` exactly once per process.
|
|
275
275
|
// undefined = not probed, null = supported, string = actionable error.
|
|
276
276
|
let babysitPreflightError: string | null | undefined;
|
|
277
|
+
let babysitPreflightCheckedAt = 0;
|
|
277
278
|
async function babysitAvailable(): Promise<boolean> {
|
|
278
|
-
// Cache only success. A missing or outdated binary may be installed while pi
|
|
279
|
-
// stays open, so subsequent tool calls must be able to recover without a restart.
|
|
280
279
|
if (babysitPreflightError === null) return true;
|
|
280
|
+
// Briefly negative-cache failures so repeated mistaken calls do not fork a
|
|
281
|
+
// version probe each time, while still recovering quickly after installation.
|
|
282
|
+
if (babysitPreflightError && Date.now() - babysitPreflightCheckedAt < 2_000) return false;
|
|
281
283
|
const r = await bs(["--version"]);
|
|
284
|
+
babysitPreflightCheckedAt = Date.now();
|
|
282
285
|
if (r.code !== 0) {
|
|
283
286
|
babysitPreflightError = INSTALL_HINT;
|
|
284
287
|
} else if (!isSupportedBabysitVersion(r.stdout)) {
|
|
@@ -406,21 +409,52 @@ interface Meta {
|
|
|
406
409
|
depth?: number;
|
|
407
410
|
maxDepth?: number;
|
|
408
411
|
budget?: SubagentBudget;
|
|
412
|
+
/** Soft-limit warning (80% by default) was accepted for this task. */
|
|
413
|
+
budgetWarnedAt?: number;
|
|
414
|
+
budgetWarningReason?: string;
|
|
415
|
+
/** Hard limit was first observed; grace is measured from observation, not RPC acceptance. */
|
|
409
416
|
budgetExceededAt?: number;
|
|
410
417
|
budgetReason?: string;
|
|
411
418
|
budgetKilled?: boolean;
|
|
419
|
+
/** Prompt offset whose nested usage has already been charged to the parent session. */
|
|
420
|
+
usageReportedOffset?: number;
|
|
412
421
|
}
|
|
413
422
|
|
|
414
423
|
const metaDir = () => path.join(ROOT, "meta");
|
|
415
424
|
const logPath = (id: string) => path.join(ROOT, "sessions", id, "output.log");
|
|
416
425
|
|
|
417
|
-
function writeMeta(id: string, m: Meta):
|
|
426
|
+
function writeMeta(id: string, m: Meta): boolean {
|
|
427
|
+
const target = path.join(metaDir(), `${id}.json`);
|
|
428
|
+
const temp = `${target}.${process.pid}.${Date.now()}.${Math.random().toString(16).slice(2)}.tmp`;
|
|
418
429
|
try {
|
|
419
430
|
fs.mkdirSync(metaDir(), { recursive: true });
|
|
420
|
-
fs.writeFileSync(
|
|
431
|
+
fs.writeFileSync(temp, JSON.stringify(m));
|
|
432
|
+
fs.renameSync(temp, target);
|
|
433
|
+
return true;
|
|
421
434
|
} catch {
|
|
422
|
-
|
|
435
|
+
try {
|
|
436
|
+
fs.rmSync(temp, { force: true });
|
|
437
|
+
} catch {
|
|
438
|
+
/* best-effort */
|
|
439
|
+
}
|
|
440
|
+
return false;
|
|
441
|
+
}
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
export function claimFileOnce(file: string, payload: string): boolean {
|
|
445
|
+
let fd: number;
|
|
446
|
+
try {
|
|
447
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
448
|
+
fd = fs.openSync(file, "wx");
|
|
449
|
+
} catch {
|
|
450
|
+
return false;
|
|
451
|
+
}
|
|
452
|
+
try {
|
|
453
|
+
fs.writeFileSync(fd, payload);
|
|
454
|
+
} finally {
|
|
455
|
+
fs.closeSync(fd);
|
|
423
456
|
}
|
|
457
|
+
return true;
|
|
424
458
|
}
|
|
425
459
|
|
|
426
460
|
function readMeta(id: string): Meta | null {
|
|
@@ -448,8 +482,27 @@ function processIsAlive(pid: number): boolean {
|
|
|
448
482
|
}
|
|
449
483
|
|
|
450
484
|
const GC_LOCK_FILE = ".pi-babysit-gc.lock";
|
|
485
|
+
const GC_STAMP_FILE = ".pi-babysit-gc.last";
|
|
486
|
+
const AUTOMATIC_GC_INTERVAL_MS = 24 * 60 * 60 * 1_000;
|
|
451
487
|
const ACTIVE_LEASE_PREFIX = ".pi-babysit-active-";
|
|
452
488
|
|
|
489
|
+
function automaticGcDue(now = Date.now()): boolean {
|
|
490
|
+
try {
|
|
491
|
+
return now - fs.statSync(path.join(ROOT_BASE, GC_STAMP_FILE)).mtimeMs >= AUTOMATIC_GC_INTERVAL_MS;
|
|
492
|
+
} catch {
|
|
493
|
+
return true;
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
function markAutomaticGc(now = new Date()): void {
|
|
498
|
+
try {
|
|
499
|
+
fs.mkdirSync(ROOT_BASE, { recursive: true });
|
|
500
|
+
fs.writeFileSync(path.join(ROOT_BASE, GC_STAMP_FILE), now.toISOString());
|
|
501
|
+
} catch {
|
|
502
|
+
/* best-effort; GC safety does not depend on this throttle stamp */
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
|
|
453
506
|
function scanTreeStats(root: string): { bytes: number; newestMtimeMs: number } {
|
|
454
507
|
let bytes = 0;
|
|
455
508
|
let newestMtimeMs = 0;
|
|
@@ -818,6 +871,18 @@ export function readLogBytesFrom(file: string, since: number): Buffer {
|
|
|
818
871
|
return bytes.subarray(0, read);
|
|
819
872
|
}
|
|
820
873
|
|
|
874
|
+
const RPC_RESPONSE_WINDOW_MAX_BYTES = 1_000_000;
|
|
875
|
+
function readLogWindowFrom(
|
|
876
|
+
file: string,
|
|
877
|
+
since: number,
|
|
878
|
+
maxBytes = RPC_RESPONSE_WINDOW_MAX_BYTES,
|
|
879
|
+
): { bytes: Buffer; offset: number } {
|
|
880
|
+
const size = fs.statSync(file).size;
|
|
881
|
+
const requested = Math.min(Math.max(0, since), size);
|
|
882
|
+
const offset = Math.max(requested, size - maxBytes);
|
|
883
|
+
return { bytes: readLogBytesFrom(file, offset), offset };
|
|
884
|
+
}
|
|
885
|
+
|
|
821
886
|
async function rpcResponse(
|
|
822
887
|
id: string,
|
|
823
888
|
since: number,
|
|
@@ -848,7 +913,8 @@ async function rpcResponse(
|
|
|
848
913
|
try {
|
|
849
914
|
// Long-lived follow-up workers can have very large historical logs. Read
|
|
850
915
|
// only the response window rather than synchronously loading all history.
|
|
851
|
-
|
|
916
|
+
const window = readLogWindowFrom(logPath(id), since);
|
|
917
|
+
structuredError = parseEvents(window.bytes.toString("utf8")).errorMsg ?? "";
|
|
852
918
|
} catch {
|
|
853
919
|
/* full log path below remains the diagnostic source */
|
|
854
920
|
}
|
|
@@ -869,7 +935,8 @@ async function rpcResponse(
|
|
|
869
935
|
};
|
|
870
936
|
}
|
|
871
937
|
try {
|
|
872
|
-
|
|
938
|
+
const window = readLogWindowFrom(logPath(id), since);
|
|
939
|
+
return parseRpcResponseBytes(window.bytes, window.offset, command);
|
|
873
940
|
} catch (error) {
|
|
874
941
|
return { ok: false, error: `could not read ${command} response: ${String(error)}` };
|
|
875
942
|
}
|
|
@@ -899,7 +966,7 @@ function byteLimitFromEnv(name: string, fallback: number): number {
|
|
|
899
966
|
return Number.isSafeInteger(value) && value >= 0 ? value : fallback;
|
|
900
967
|
}
|
|
901
968
|
|
|
902
|
-
const TAIL_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_TAIL_MAX_BYTES",
|
|
969
|
+
const TAIL_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_TAIL_MAX_BYTES", 4_000);
|
|
903
970
|
// Direct run/wait results can carry more context because the caller explicitly
|
|
904
971
|
// requested them. Unsolicited completion notifications default much smaller.
|
|
905
972
|
const INLINE_OUTPUT_MAX_BYTES = byteLimitFromEnv("PI_BABYSIT_INLINE_OUTPUT_MAX_BYTES", 8_000);
|
|
@@ -910,6 +977,11 @@ const ANSWER_MAX_BYTES = 24_000; // single subagent answers / structured error m
|
|
|
910
977
|
const MAX_MULTI_WAIT_SESSIONS = 32;
|
|
911
978
|
const SUBAGENT_BUDGET_GRACE_MS =
|
|
912
979
|
parseDurMs(process.env.PI_BABYSIT_BUDGET_GRACE ?? "90s") ?? 90_000;
|
|
980
|
+
const SUBAGENT_REAP_AFTER =
|
|
981
|
+
process.env.PI_BABYSIT_REAP_AFTER ?? process.env.PI_SUBAGENT_REAP_AFTER ?? "120s";
|
|
982
|
+
const SUBAGENT_REUSE_HINT = ["0", "off", "none"].includes(SUBAGENT_REAP_AFTER)
|
|
983
|
+
? "Session remains available until its absolute timeout."
|
|
984
|
+
: `Session remains available for follow-ups during the ${SUBAGENT_REAP_AFTER} idle grace.`;
|
|
913
985
|
|
|
914
986
|
export function clip(s: string, maxBytes = TAIL_MAX_BYTES): string {
|
|
915
987
|
if (maxBytes <= 0) return "";
|
|
@@ -940,15 +1012,27 @@ export function clipMultiWaitResult(
|
|
|
940
1012
|
return clip(value, Math.min(maxBytes, ANSWER_MAX_BYTES));
|
|
941
1013
|
}
|
|
942
1014
|
|
|
1015
|
+
interface SearchLogCacheEntry {
|
|
1016
|
+
size: number;
|
|
1017
|
+
mtimeMs: number;
|
|
1018
|
+
text: string;
|
|
1019
|
+
}
|
|
1020
|
+
const searchLogCache = new Map<string, SearchLogCacheEntry>();
|
|
1021
|
+
|
|
943
1022
|
async function searchLog(
|
|
944
1023
|
id: string,
|
|
945
1024
|
pattern: string,
|
|
946
1025
|
maxLines: number,
|
|
947
1026
|
signal?: AbortSignal,
|
|
1027
|
+
maxBytes = TAIL_MAX_BYTES,
|
|
948
1028
|
): Promise<{ text: string; error?: string }> {
|
|
949
1029
|
const file = logPath(id);
|
|
950
1030
|
if (!fs.existsSync(file)) return { text: "", error: `Log file is missing: ${file}` };
|
|
951
1031
|
if (signal?.aborted) return { text: "", error: "Log search was interrupted." };
|
|
1032
|
+
const stat = fs.statSync(file);
|
|
1033
|
+
const cacheKey = `${file}\u0000${pattern}\u0000${maxLines}\u0000${maxBytes}`;
|
|
1034
|
+
const cached = searchLogCache.get(cacheKey);
|
|
1035
|
+
if (cached?.size === stat.size && cached.mtimeMs === stat.mtimeMs) return { text: cached.text };
|
|
952
1036
|
|
|
953
1037
|
// Run regex evaluation out of process so catastrophic backtracking or a huge
|
|
954
1038
|
// no-newline log cannot freeze or exhaust pi's main Node process. The helper
|
|
@@ -996,7 +1080,10 @@ async function searchLog(
|
|
|
996
1080
|
} else if (code !== 0) {
|
|
997
1081
|
finish({ text: "", error: stderr.trim() || `Log search failed (exit ${code ?? "?"}).` });
|
|
998
1082
|
} else {
|
|
999
|
-
|
|
1083
|
+
const text = clip(stdout.trimEnd(), maxBytes);
|
|
1084
|
+
searchLogCache.set(cacheKey, { size: stat.size, mtimeMs: stat.mtimeMs, text });
|
|
1085
|
+
while (searchLogCache.size > 64) searchLogCache.delete(searchLogCache.keys().next().value as string);
|
|
1086
|
+
finish({ text });
|
|
1000
1087
|
}
|
|
1001
1088
|
});
|
|
1002
1089
|
});
|
|
@@ -1030,6 +1117,33 @@ async function inlineOutput(
|
|
|
1030
1117
|
return output ? `\n\nOutput:\n${output}` : "";
|
|
1031
1118
|
}
|
|
1032
1119
|
|
|
1120
|
+
interface ProcessOutputSelection {
|
|
1121
|
+
pattern?: string;
|
|
1122
|
+
lines?: number;
|
|
1123
|
+
maxBytes?: number;
|
|
1124
|
+
}
|
|
1125
|
+
|
|
1126
|
+
async function selectedProcessOutput(
|
|
1127
|
+
id: string,
|
|
1128
|
+
status: BsSession,
|
|
1129
|
+
selection?: ProcessOutputSelection,
|
|
1130
|
+
signal?: AbortSignal,
|
|
1131
|
+
): Promise<string> {
|
|
1132
|
+
if (!selection || (!selection.pattern && selection.lines == null && selection.maxBytes == null)) {
|
|
1133
|
+
return inlineOutput(id, status);
|
|
1134
|
+
}
|
|
1135
|
+
const maxBytes = selection.maxBytes ?? INLINE_OUTPUT_MAX_BYTES;
|
|
1136
|
+
const lines = Math.min(Math.max(1, Math.floor(selection.lines ?? 30)), 200);
|
|
1137
|
+
if (selection.pattern) {
|
|
1138
|
+
const result = await searchLog(id, selection.pattern, lines, signal, maxBytes);
|
|
1139
|
+
if (result.error) return `\nOutput filter failed: ${result.error}`;
|
|
1140
|
+
const body = result.text || `(no output matching /${selection.pattern}/)`;
|
|
1141
|
+
return `\n\nSelected output /${selection.pattern}/:\n${clip(body, maxBytes)}`;
|
|
1142
|
+
}
|
|
1143
|
+
const tail = (await bs(["log", "-s", id, "--tail", String(lines)])).stdout.trimEnd();
|
|
1144
|
+
return tail ? `\n\nSelected tail (${lines} lines max):\n${clip(tail, maxBytes)}` : "";
|
|
1145
|
+
}
|
|
1146
|
+
|
|
1033
1147
|
export function summarizeNotificationCommand(command: string | undefined): string {
|
|
1034
1148
|
const preview =
|
|
1035
1149
|
(command ?? "?")
|
|
@@ -1305,7 +1419,10 @@ interface ToolCall {
|
|
|
1305
1419
|
}
|
|
1306
1420
|
export interface Progress {
|
|
1307
1421
|
turns: number;
|
|
1422
|
+
/** Bounded recent calls for status rendering. */
|
|
1308
1423
|
toolCalls: ToolCall[];
|
|
1424
|
+
/** Exact count, independent of the bounded recent-call ring. */
|
|
1425
|
+
toolCallCount: number;
|
|
1309
1426
|
finalText: string;
|
|
1310
1427
|
/** Best-effort text from the currently streaming assistant message. */
|
|
1311
1428
|
streamingText: string;
|
|
@@ -1320,6 +1437,10 @@ export interface Progress {
|
|
|
1320
1437
|
cacheWriteTokens: number;
|
|
1321
1438
|
reasoningTokens: number;
|
|
1322
1439
|
cost: number;
|
|
1440
|
+
inputCost: number;
|
|
1441
|
+
outputCost: number;
|
|
1442
|
+
cacheReadCost: number;
|
|
1443
|
+
cacheWriteCost: number;
|
|
1323
1444
|
errorMsg?: string;
|
|
1324
1445
|
// RPC lifecycle bookkeeping (computed over the analyzed log slice):
|
|
1325
1446
|
agentStarts: number;
|
|
@@ -1362,6 +1483,7 @@ function emptyProgress(): Progress {
|
|
|
1362
1483
|
return {
|
|
1363
1484
|
turns: 0,
|
|
1364
1485
|
toolCalls: [],
|
|
1486
|
+
toolCallCount: 0,
|
|
1365
1487
|
finalText: "",
|
|
1366
1488
|
streamingText: "",
|
|
1367
1489
|
modelCalls: 0,
|
|
@@ -1372,6 +1494,10 @@ function emptyProgress(): Progress {
|
|
|
1372
1494
|
cacheWriteTokens: 0,
|
|
1373
1495
|
reasoningTokens: 0,
|
|
1374
1496
|
cost: 0,
|
|
1497
|
+
inputCost: 0,
|
|
1498
|
+
outputCost: 0,
|
|
1499
|
+
cacheReadCost: 0,
|
|
1500
|
+
cacheWriteCost: 0,
|
|
1375
1501
|
agentStarts: 0,
|
|
1376
1502
|
agentEnds: 0,
|
|
1377
1503
|
agentSettled: 0,
|
|
@@ -1393,8 +1519,8 @@ export function subagentBudgetViolation(
|
|
|
1393
1519
|
if (budget.maxTurns != null && progress.turns >= budget.maxTurns) {
|
|
1394
1520
|
return `${progress.turns} turns reached maxTurns ${budget.maxTurns}`;
|
|
1395
1521
|
}
|
|
1396
|
-
if (budget.maxToolCalls != null && progress.
|
|
1397
|
-
return `${progress.
|
|
1522
|
+
if (budget.maxToolCalls != null && progress.toolCallCount >= budget.maxToolCalls) {
|
|
1523
|
+
return `${progress.toolCallCount} tool calls reached maxToolCalls ${budget.maxToolCalls}`;
|
|
1398
1524
|
}
|
|
1399
1525
|
if (budget.maxUsageTokens != null && progress.usageTokens >= budget.maxUsageTokens) {
|
|
1400
1526
|
return `${progress.usageTokens} usage tokens reached maxUsageTokens ${budget.maxUsageTokens}`;
|
|
@@ -1402,6 +1528,33 @@ export function subagentBudgetViolation(
|
|
|
1402
1528
|
return null;
|
|
1403
1529
|
}
|
|
1404
1530
|
|
|
1531
|
+
export function subagentBudgetSoftViolation(
|
|
1532
|
+
progress: Progress,
|
|
1533
|
+
budget?: SubagentBudget,
|
|
1534
|
+
ratio = 0.8,
|
|
1535
|
+
): string | null {
|
|
1536
|
+
if (!budget || ratio <= 0 || ratio >= 1) return null;
|
|
1537
|
+
if (budget.maxCost != null && progress.cost >= budget.maxCost * ratio) {
|
|
1538
|
+
return `cost $${progress.cost.toFixed(4)} reached ${Math.round(ratio * 100)}% of maxCost $${budget.maxCost.toFixed(4)}`;
|
|
1539
|
+
}
|
|
1540
|
+
if (budget.maxTurns != null && progress.turns >= Math.max(1, Math.ceil(budget.maxTurns * ratio))) {
|
|
1541
|
+
return `${progress.turns} turns reached ${Math.round(ratio * 100)}% of maxTurns ${budget.maxTurns}`;
|
|
1542
|
+
}
|
|
1543
|
+
if (
|
|
1544
|
+
budget.maxToolCalls != null &&
|
|
1545
|
+
progress.toolCallCount >= Math.max(1, Math.ceil(budget.maxToolCalls * ratio))
|
|
1546
|
+
) {
|
|
1547
|
+
return `${progress.toolCallCount} tool calls reached ${Math.round(ratio * 100)}% of maxToolCalls ${budget.maxToolCalls}`;
|
|
1548
|
+
}
|
|
1549
|
+
if (
|
|
1550
|
+
budget.maxUsageTokens != null &&
|
|
1551
|
+
progress.usageTokens >= Math.max(1, Math.ceil(budget.maxUsageTokens * ratio))
|
|
1552
|
+
) {
|
|
1553
|
+
return `${progress.usageTokens} usage tokens reached ${Math.round(ratio * 100)}% of maxUsageTokens ${budget.maxUsageTokens}`;
|
|
1554
|
+
}
|
|
1555
|
+
return null;
|
|
1556
|
+
}
|
|
1557
|
+
|
|
1405
1558
|
export function subagentBudgetAction(
|
|
1406
1559
|
progress: Progress,
|
|
1407
1560
|
budget: SubagentBudget | undefined,
|
|
@@ -1449,16 +1602,19 @@ function parseEventLine(progress: Progress, raw: string): void {
|
|
|
1449
1602
|
| { type?: string; delta?: string }
|
|
1450
1603
|
| undefined;
|
|
1451
1604
|
if (update?.type === "text_delta" && typeof update.delta === "string") {
|
|
1452
|
-
progress.streamingText
|
|
1605
|
+
progress.streamingText = clip(progress.streamingText + update.delta, ANSWER_MAX_BYTES);
|
|
1453
1606
|
}
|
|
1454
1607
|
break;
|
|
1455
1608
|
}
|
|
1456
1609
|
case "tool_execution_start": {
|
|
1457
1610
|
const name = String(event.toolName ?? "tool");
|
|
1611
|
+
progress.toolCallCount++;
|
|
1458
1612
|
progress.toolCalls.push({
|
|
1459
1613
|
name,
|
|
1460
1614
|
summary: summarizeToolCall(name, (event.args as Record<string, unknown>) ?? {}),
|
|
1461
1615
|
});
|
|
1616
|
+
// Open-ended workers must not retain an unbounded tool history in Pi.
|
|
1617
|
+
if (progress.toolCalls.length > 200) progress.toolCalls.splice(0, progress.toolCalls.length - 200);
|
|
1462
1618
|
break;
|
|
1463
1619
|
}
|
|
1464
1620
|
case "message_end": {
|
|
@@ -1466,6 +1622,8 @@ function parseEventLine(progress: Progress, raw: string): void {
|
|
|
1466
1622
|
| {
|
|
1467
1623
|
role?: string;
|
|
1468
1624
|
content?: { type: string; text?: string }[];
|
|
1625
|
+
stopReason?: string;
|
|
1626
|
+
errorMessage?: string;
|
|
1469
1627
|
usage?: {
|
|
1470
1628
|
input?: number;
|
|
1471
1629
|
output?: number;
|
|
@@ -1473,7 +1631,13 @@ function parseEventLine(progress: Progress, raw: string): void {
|
|
|
1473
1631
|
cacheWrite?: number;
|
|
1474
1632
|
reasoning?: number;
|
|
1475
1633
|
totalTokens?: number;
|
|
1476
|
-
cost?: {
|
|
1634
|
+
cost?: {
|
|
1635
|
+
input?: number;
|
|
1636
|
+
output?: number;
|
|
1637
|
+
cacheRead?: number;
|
|
1638
|
+
cacheWrite?: number;
|
|
1639
|
+
total?: number;
|
|
1640
|
+
};
|
|
1477
1641
|
};
|
|
1478
1642
|
}
|
|
1479
1643
|
| undefined;
|
|
@@ -1482,7 +1646,10 @@ function parseEventLine(progress: Progress, raw: string): void {
|
|
|
1482
1646
|
.filter((content) => content.type === "text" && content.text)
|
|
1483
1647
|
.map((content) => content.text)
|
|
1484
1648
|
.join("");
|
|
1485
|
-
if (text.trim()) progress.finalText = text;
|
|
1649
|
+
if (text.trim()) progress.finalText = clip(text, ANSWER_MAX_BYTES);
|
|
1650
|
+
if (message.stopReason === "error") {
|
|
1651
|
+
progress.errorMsg = message.errorMessage || "subagent model request failed";
|
|
1652
|
+
}
|
|
1486
1653
|
progress.streamingText = "";
|
|
1487
1654
|
if (message.usage) {
|
|
1488
1655
|
const finite = (value: number | undefined) =>
|
|
@@ -1495,6 +1662,10 @@ function parseEventLine(progress: Progress, raw: string): void {
|
|
|
1495
1662
|
progress.cacheReadTokens += finite(message.usage.cacheRead);
|
|
1496
1663
|
progress.cacheWriteTokens += finite(message.usage.cacheWrite);
|
|
1497
1664
|
progress.reasoningTokens += finite(message.usage.reasoning);
|
|
1665
|
+
progress.inputCost += finite(message.usage.cost?.input);
|
|
1666
|
+
progress.outputCost += finite(message.usage.cost?.output);
|
|
1667
|
+
progress.cacheReadCost += finite(message.usage.cost?.cacheRead);
|
|
1668
|
+
progress.cacheWriteCost += finite(message.usage.cost?.cacheWrite);
|
|
1498
1669
|
progress.cost += finite(message.usage.cost?.total);
|
|
1499
1670
|
}
|
|
1500
1671
|
}
|
|
@@ -1925,6 +2096,33 @@ async function spawnSubagent(
|
|
|
1925
2096
|
await bs(["kill", "-s", id]);
|
|
1926
2097
|
return { error: `subagent ${id} rejected the task: ${resp.error}` };
|
|
1927
2098
|
}
|
|
2099
|
+
// Prompt acceptance does not guarantee provider authentication: Pi reports
|
|
2100
|
+
// failures that occur after acceptance through the event stream. Probe a short
|
|
2101
|
+
// window so immediate missing-key/config errors fail the spawn instead of
|
|
2102
|
+
// creating a zero-work worker that the caller must discover later.
|
|
2103
|
+
const startupProbe = await bs([
|
|
2104
|
+
"expect",
|
|
2105
|
+
"-s",
|
|
2106
|
+
id,
|
|
2107
|
+
"--since",
|
|
2108
|
+
String(resp.offset),
|
|
2109
|
+
"--timeout",
|
|
2110
|
+
"500ms",
|
|
2111
|
+
'(?m)^\\{"type":"(?:message_end|error|extension_error|agent_settled)"',
|
|
2112
|
+
]);
|
|
2113
|
+
if (startupProbe.code === 0) {
|
|
2114
|
+
try {
|
|
2115
|
+
const window = readLogWindowFrom(logPath(id), resp.offset);
|
|
2116
|
+
const initialProgress = parseEvents(window.bytes.toString("utf8"));
|
|
2117
|
+
if (initialProgress.errorMsg && initialProgress.modelCalls === 0) {
|
|
2118
|
+
discardDelivery(delivery);
|
|
2119
|
+
await bs(["kill", "-s", id]);
|
|
2120
|
+
return { error: `subagent ${id} failed before its first model response: ${initialProgress.errorMsg}` };
|
|
2121
|
+
}
|
|
2122
|
+
} catch {
|
|
2123
|
+
/* normal startup continues; the full stream remains available to check/wait */
|
|
2124
|
+
}
|
|
2125
|
+
}
|
|
1928
2126
|
// Report the RESOLVED model (a fuzzy pattern may match something unexpected;
|
|
1929
2127
|
// null means nothing resolved at all).
|
|
1930
2128
|
let resolvedModel: string | undefined;
|
|
@@ -1985,8 +2183,10 @@ const WIDGET_TAIL_WIDTH = 100;
|
|
|
1985
2183
|
// Strip ANSI/control escapes and clamp width so raw PTY output can't wrap or
|
|
1986
2184
|
// corrupt the widget area.
|
|
1987
2185
|
function sanitizeTailLine(s: string): string {
|
|
1988
|
-
|
|
1989
|
-
|
|
2186
|
+
// PTY progress bars often redraw one logical line with carriage returns.
|
|
2187
|
+
// Keep the latest frame rather than concatenating every historical frame.
|
|
2188
|
+
const terminalFrame = s.split("\r").filter(Boolean).at(-1) ?? "";
|
|
2189
|
+
const clean = terminalFrame
|
|
1990
2190
|
// CSI / OSC / other escape sequences
|
|
1991
2191
|
.replace(/\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)/g, "")
|
|
1992
2192
|
.replace(/\x1b[@-Z\\-_]|\x1b\[[0-?]*[ -/]*[@-~]/g, "")
|
|
@@ -1995,9 +2195,34 @@ function sanitizeTailLine(s: string): string {
|
|
|
1995
2195
|
return clean.length > WIDGET_TAIL_WIDTH ? `${clean.slice(0, WIDGET_TAIL_WIDTH - 1)}…` : clean;
|
|
1996
2196
|
}
|
|
1997
2197
|
|
|
2198
|
+
function readTailLines(file: string, lines: number, maxBytes = 64_000): string[] {
|
|
2199
|
+
try {
|
|
2200
|
+
const size = fs.statSync(file).size;
|
|
2201
|
+
const start = Math.max(0, size - maxBytes);
|
|
2202
|
+
const length = size - start;
|
|
2203
|
+
const bytes = Buffer.allocUnsafe(length);
|
|
2204
|
+
const fd = fs.openSync(file, "r");
|
|
2205
|
+
let read = 0;
|
|
2206
|
+
try {
|
|
2207
|
+
while (read < length) {
|
|
2208
|
+
const count = fs.readSync(fd, bytes, read, length - read, start + read);
|
|
2209
|
+
if (count === 0) break;
|
|
2210
|
+
read += count;
|
|
2211
|
+
}
|
|
2212
|
+
} finally {
|
|
2213
|
+
fs.closeSync(fd);
|
|
2214
|
+
}
|
|
2215
|
+
const parts = bytes.subarray(0, read).toString("utf8").split("\n");
|
|
2216
|
+
if (start > 0) parts.shift(); // first fragment may begin mid-line
|
|
2217
|
+
return parts.slice(-Math.max(1, lines + 1));
|
|
2218
|
+
} catch {
|
|
2219
|
+
return [];
|
|
2220
|
+
}
|
|
2221
|
+
}
|
|
2222
|
+
|
|
1998
2223
|
// Trailing lines to show for a running session (sanitized, unprefixed).
|
|
1999
|
-
//
|
|
2000
|
-
//
|
|
2224
|
+
// Process tails are read directly from the bounded end of output.log, avoiding
|
|
2225
|
+
// one `babysit log` subprocess per active process on every poll.
|
|
2001
2226
|
async function widgetTail(
|
|
2002
2227
|
id: string,
|
|
2003
2228
|
isSub: boolean,
|
|
@@ -2005,8 +2230,7 @@ async function widgetTail(
|
|
|
2005
2230
|
): Promise<string[]> {
|
|
2006
2231
|
let raw: string[];
|
|
2007
2232
|
if (!isSub) {
|
|
2008
|
-
|
|
2009
|
-
raw = out.split("\n");
|
|
2233
|
+
raw = readTailLines(logPath(id), WIDGET_TAIL_LINES);
|
|
2010
2234
|
} else {
|
|
2011
2235
|
const progress = subagentProgress ?? taskProgressOf(id).progress;
|
|
2012
2236
|
if (progress.finalText.trim()) {
|
|
@@ -2047,6 +2271,90 @@ interface WaitOutcome {
|
|
|
2047
2271
|
progress?: Progress;
|
|
2048
2272
|
}
|
|
2049
2273
|
|
|
2274
|
+
export interface NestedUsage {
|
|
2275
|
+
input: number;
|
|
2276
|
+
output: number;
|
|
2277
|
+
cacheRead: number;
|
|
2278
|
+
cacheWrite: number;
|
|
2279
|
+
totalTokens: number;
|
|
2280
|
+
cost: {
|
|
2281
|
+
input: number;
|
|
2282
|
+
output: number;
|
|
2283
|
+
cacheRead: number;
|
|
2284
|
+
cacheWrite: number;
|
|
2285
|
+
total: number;
|
|
2286
|
+
};
|
|
2287
|
+
}
|
|
2288
|
+
|
|
2289
|
+
export function usageFromProgress(progress: Progress): NestedUsage | undefined {
|
|
2290
|
+
if (progress.modelCalls === 0) return undefined;
|
|
2291
|
+
return {
|
|
2292
|
+
input: progress.inputTokens,
|
|
2293
|
+
output: progress.outputTokens,
|
|
2294
|
+
cacheRead: progress.cacheReadTokens,
|
|
2295
|
+
cacheWrite: progress.cacheWriteTokens,
|
|
2296
|
+
totalTokens: progress.usageTokens,
|
|
2297
|
+
cost: {
|
|
2298
|
+
input: progress.inputCost,
|
|
2299
|
+
output: progress.outputCost,
|
|
2300
|
+
cacheRead: progress.cacheReadCost,
|
|
2301
|
+
cacheWrite: progress.cacheWriteCost,
|
|
2302
|
+
total: progress.cost,
|
|
2303
|
+
},
|
|
2304
|
+
};
|
|
2305
|
+
}
|
|
2306
|
+
|
|
2307
|
+
/** Charge one completed task exactly once, even across concurrent wait callers. */
|
|
2308
|
+
function claimOutcomeUsage(outcome: WaitOutcome): NestedUsage | undefined {
|
|
2309
|
+
if (!outcome.progress || (outcome.kind !== "done" && outcome.kind !== "exited")) return undefined;
|
|
2310
|
+
const usage = usageFromProgress(outcome.progress);
|
|
2311
|
+
if (!usage) return undefined;
|
|
2312
|
+
const meta = readMeta(outcome.id);
|
|
2313
|
+
if (meta?.kind !== "subagent") return undefined;
|
|
2314
|
+
const offset = meta.promptOffset ?? 0;
|
|
2315
|
+
if (meta.usageReportedOffset === offset) return undefined;
|
|
2316
|
+
|
|
2317
|
+
// `open(..., "wx")` is the cross-process compare-and-set. A resumed Pi
|
|
2318
|
+
// session can briefly have overlapping extension processes; metadata alone
|
|
2319
|
+
// would let both read the old value and charge the same nested usage.
|
|
2320
|
+
const marker = path.join(metaDir(), `${outcome.id}.usage-${offset}.claimed`);
|
|
2321
|
+
if (!claimFileOnce(marker, JSON.stringify({ pid: process.pid, claimedAt: Date.now() }))) {
|
|
2322
|
+
return undefined;
|
|
2323
|
+
}
|
|
2324
|
+
meta.usageReportedOffset = offset;
|
|
2325
|
+
writeMeta(outcome.id, meta); // compatibility/display hint; marker is authoritative
|
|
2326
|
+
return usage;
|
|
2327
|
+
}
|
|
2328
|
+
|
|
2329
|
+
function sumNestedUsage(values: Array<NestedUsage | undefined>): NestedUsage | undefined {
|
|
2330
|
+
const present = values.filter((value): value is NestedUsage => Boolean(value));
|
|
2331
|
+
if (present.length === 0) return undefined;
|
|
2332
|
+
return present.reduce<NestedUsage>(
|
|
2333
|
+
(total, value) => ({
|
|
2334
|
+
input: total.input + value.input,
|
|
2335
|
+
output: total.output + value.output,
|
|
2336
|
+
cacheRead: total.cacheRead + value.cacheRead,
|
|
2337
|
+
cacheWrite: total.cacheWrite + value.cacheWrite,
|
|
2338
|
+
totalTokens: total.totalTokens + value.totalTokens,
|
|
2339
|
+
cost: {
|
|
2340
|
+
input: total.cost.input + value.cost.input,
|
|
2341
|
+
output: total.cost.output + value.cost.output,
|
|
2342
|
+
cacheRead: total.cost.cacheRead + value.cost.cacheRead,
|
|
2343
|
+
cacheWrite: total.cost.cacheWrite + value.cost.cacheWrite,
|
|
2344
|
+
total: total.cost.total + value.cost.total,
|
|
2345
|
+
},
|
|
2346
|
+
}),
|
|
2347
|
+
{
|
|
2348
|
+
input: 0,
|
|
2349
|
+
output: 0,
|
|
2350
|
+
cacheRead: 0,
|
|
2351
|
+
cacheWrite: 0,
|
|
2352
|
+
totalTokens: 0,
|
|
2353
|
+
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
|
2354
|
+
},
|
|
2355
|
+
);
|
|
2356
|
+
}
|
|
2357
|
+
|
|
2050
2358
|
// Wait for ONE subagent's current task. Completion = agent_settled without a
|
|
2051
2359
|
// parked babysit_run/process result (a parked run only means "waiting for
|
|
2052
2360
|
// process exit; pi will resume itself"). Parse appended bytes incrementally,
|
|
@@ -2069,7 +2377,7 @@ async function waitForTask(
|
|
|
2069
2377
|
const st = await statusOf(id);
|
|
2070
2378
|
|
|
2071
2379
|
const stats =
|
|
2072
|
-
`turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.
|
|
2380
|
+
`turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}` +
|
|
2073
2381
|
(prog.tokens != null ? ` ctx=${prog.tokens}` : "") +
|
|
2074
2382
|
(prog.modelCalls > 0
|
|
2075
2383
|
? ` usage=${prog.usageTokens} (in=${prog.inputTokens} out=${prog.outputTokens} cache=${prog.cacheReadTokens}) $${prog.cost.toFixed(4)}`
|
|
@@ -2083,7 +2391,7 @@ async function waitForTask(
|
|
|
2083
2391
|
ok: completed.ok,
|
|
2084
2392
|
text:
|
|
2085
2393
|
`Subagent ${id} finished its task (${stats}).\n` +
|
|
2086
|
-
|
|
2394
|
+
`${SUBAGENT_REUSE_HINT} Follow-up: babysit_send { id: "${id}" }, ` +
|
|
2087
2395
|
`or babysit_kill when done.\n\n${completed.body}`,
|
|
2088
2396
|
status: st,
|
|
2089
2397
|
progress: prog,
|
|
@@ -2227,6 +2535,7 @@ async function waitForExit(
|
|
|
2227
2535
|
limitMs: number | null,
|
|
2228
2536
|
signal?: AbortSignal,
|
|
2229
2537
|
expectPattern?: string,
|
|
2538
|
+
outputSelection?: ProcessOutputSelection,
|
|
2230
2539
|
): Promise<WaitOutcome> {
|
|
2231
2540
|
const t = limitMs != null ? `${Math.ceil(limitMs / 1000)}s` : "0";
|
|
2232
2541
|
|
|
@@ -2315,7 +2624,7 @@ async function waitForExit(
|
|
|
2315
2624
|
const meta = readMeta(id);
|
|
2316
2625
|
const workerDead = st.state === "dead" && st.exit_code == null;
|
|
2317
2626
|
const ok = st.exit_code === 0;
|
|
2318
|
-
const output = await
|
|
2627
|
+
const output = await selectedProcessOutput(id, st, outputSelection, signal);
|
|
2319
2628
|
return {
|
|
2320
2629
|
id,
|
|
2321
2630
|
kind: "exited",
|
|
@@ -2405,6 +2714,21 @@ export function automaticNotificationGroup(entry: unknown): string | undefined {
|
|
|
2405
2714
|
return runs.length >= 2 ? group : undefined;
|
|
2406
2715
|
}
|
|
2407
2716
|
|
|
2717
|
+
export function resolveSubagentSendMode(
|
|
2718
|
+
requested: "auto" | "steer" | "task",
|
|
2719
|
+
streaming?: boolean,
|
|
2720
|
+
currentTaskDone?: boolean,
|
|
2721
|
+
): { mode: "steer" | "task" } | { error: "busy" | "unknown" | "unsettled" } {
|
|
2722
|
+
if (requested === "steer") return { mode: "steer" };
|
|
2723
|
+
if (requested === "auto") {
|
|
2724
|
+
return { mode: streaming === false && currentTaskDone === true ? "task" : "steer" };
|
|
2725
|
+
}
|
|
2726
|
+
if (streaming === true) return { error: "busy" };
|
|
2727
|
+
if (streaming === undefined || currentTaskDone === undefined) return { error: "unknown" };
|
|
2728
|
+
if (!currentTaskDone) return { error: "unsettled" };
|
|
2729
|
+
return { mode: "task" };
|
|
2730
|
+
}
|
|
2731
|
+
|
|
2408
2732
|
// ---------------------------------------------------------------------------
|
|
2409
2733
|
// extension
|
|
2410
2734
|
// ---------------------------------------------------------------------------
|
|
@@ -2459,76 +2783,88 @@ export default function (pi: ExtensionAPI) {
|
|
|
2459
2783
|
});
|
|
2460
2784
|
|
|
2461
2785
|
async function enforceSubagentBudgets(sessions: BsSession[]): Promise<void> {
|
|
2462
|
-
|
|
2463
|
-
|
|
2464
|
-
|
|
2465
|
-
|
|
2466
|
-
|
|
2467
|
-
|
|
2468
|
-
|
|
2469
|
-
|
|
2470
|
-
|
|
2471
|
-
|
|
2472
|
-
|
|
2473
|
-
|
|
2474
|
-
|
|
2475
|
-
|
|
2476
|
-
|
|
2477
|
-
|
|
2478
|
-
|
|
2479
|
-
|
|
2480
|
-
|
|
2481
|
-
|
|
2482
|
-
|
|
2483
|
-
|
|
2484
|
-
|
|
2485
|
-
|
|
2486
|
-
|
|
2487
|
-
|
|
2488
|
-
|
|
2489
|
-
|
|
2490
|
-
|
|
2491
|
-
|
|
2492
|
-
|
|
2493
|
-
|
|
2494
|
-
|
|
2495
|
-
|
|
2496
|
-
|
|
2497
|
-
|
|
2498
|
-
|
|
2499
|
-
|
|
2500
|
-
|
|
2501
|
-
|
|
2502
|
-
|
|
2503
|
-
|
|
2504
|
-
|
|
2505
|
-
|
|
2506
|
-
|
|
2507
|
-
|
|
2508
|
-
|
|
2509
|
-
|
|
2510
|
-
|
|
2511
|
-
|
|
2512
|
-
|
|
2513
|
-
|
|
2786
|
+
// Independent workers must not serialize 3-second RPC probes and delay
|
|
2787
|
+
// unrelated completion notifications. Per-session RPC locks still preserve
|
|
2788
|
+
// ordering within each worker.
|
|
2789
|
+
await Promise.all(
|
|
2790
|
+
sessions
|
|
2791
|
+
.filter((session) => session.state === "running")
|
|
2792
|
+
.map((session) =>
|
|
2793
|
+
withSessionRpcLock(session.id, async () => {
|
|
2794
|
+
const meta = readMeta(session.id);
|
|
2795
|
+
if (meta?.kind !== "subagent" || !meta.budget || meta.budgetKilled) return;
|
|
2796
|
+
|
|
2797
|
+
let progress: Progress;
|
|
2798
|
+
try {
|
|
2799
|
+
progress = taskProgressOf(session.id).progress;
|
|
2800
|
+
} catch {
|
|
2801
|
+
return;
|
|
2802
|
+
}
|
|
2803
|
+
if (progress.done) return;
|
|
2804
|
+
const now = Date.now();
|
|
2805
|
+
const hardReason = subagentBudgetViolation(progress, meta.budget);
|
|
2806
|
+
const softReason = subagentBudgetSoftViolation(progress, meta.budget);
|
|
2807
|
+
|
|
2808
|
+
if (hardReason) {
|
|
2809
|
+
if (meta.budgetExceededAt == null) {
|
|
2810
|
+
// The hard grace begins when the violation is observed, even if a
|
|
2811
|
+
// wedged worker never accepts steering. This makes the cap enforceable.
|
|
2812
|
+
meta.budgetExceededAt = now;
|
|
2813
|
+
meta.budgetReason = hardReason;
|
|
2814
|
+
writeMeta(session.id, meta);
|
|
2815
|
+
const latestStatus = await statusOf(session.id);
|
|
2816
|
+
if (latestStatus?.state !== "running") return;
|
|
2817
|
+
const sent = await sendRpc(session.id, {
|
|
2818
|
+
type: "steer",
|
|
2819
|
+
message: `Hard budget reached (${hardReason}). Stop calling tools and return your best answer now.`,
|
|
2820
|
+
});
|
|
2821
|
+
if (!("error" in sent)) await rpcResponse(session.id, sent.offset, "steer", "3s");
|
|
2822
|
+
return;
|
|
2823
|
+
}
|
|
2824
|
+
if (now - meta.budgetExceededAt < SUBAGENT_BUDGET_GRACE_MS) return;
|
|
2825
|
+
const latestStatus = await statusOf(session.id);
|
|
2826
|
+
if (latestStatus?.state !== "running") return;
|
|
2827
|
+
const killed = await bs(["kill", "-s", session.id, "--json"]);
|
|
2828
|
+
if (killed.code !== 0) return;
|
|
2829
|
+
const terminal = await awaitConfirmedTermination(session.id);
|
|
2830
|
+
const current = readMeta(session.id);
|
|
2831
|
+
if (
|
|
2832
|
+
terminal &&
|
|
2833
|
+
isConfirmedTerminalState(terminal.state) &&
|
|
2834
|
+
current?.kind === "subagent" &&
|
|
2835
|
+
current.promptOffset === meta.promptOffset &&
|
|
2836
|
+
current.budgetExceededAt === meta.budgetExceededAt
|
|
2837
|
+
) {
|
|
2838
|
+
current.budgetKilled = true;
|
|
2839
|
+
current.budgetReason = current.budgetReason ?? hardReason;
|
|
2840
|
+
writeMeta(session.id, current);
|
|
2841
|
+
}
|
|
2842
|
+
return;
|
|
2843
|
+
}
|
|
2514
2844
|
|
|
2515
|
-
|
|
2516
|
-
|
|
2517
|
-
|
|
2518
|
-
|
|
2519
|
-
|
|
2520
|
-
|
|
2521
|
-
|
|
2522
|
-
|
|
2523
|
-
|
|
2524
|
-
|
|
2525
|
-
|
|
2526
|
-
|
|
2527
|
-
|
|
2528
|
-
|
|
2529
|
-
|
|
2530
|
-
|
|
2531
|
-
|
|
2845
|
+
if (!softReason || meta.budgetWarnedAt != null) return;
|
|
2846
|
+
const latestStatus = await statusOf(session.id);
|
|
2847
|
+
if (latestStatus?.state !== "running") return;
|
|
2848
|
+
const sent = await sendRpc(session.id, {
|
|
2849
|
+
type: "steer",
|
|
2850
|
+
message: `Budget is approaching its limit (${softReason}). Wrap up now and preserve your best findings.`,
|
|
2851
|
+
});
|
|
2852
|
+
if ("error" in sent) return;
|
|
2853
|
+
const accepted = await rpcResponse(session.id, sent.offset, "steer", "3s");
|
|
2854
|
+
if (!accepted.ok) return;
|
|
2855
|
+
const current = readMeta(session.id);
|
|
2856
|
+
if (
|
|
2857
|
+
current?.kind === "subagent" &&
|
|
2858
|
+
current.promptOffset === meta.promptOffset &&
|
|
2859
|
+
current.budgetWarnedAt == null
|
|
2860
|
+
) {
|
|
2861
|
+
current.budgetWarnedAt = now;
|
|
2862
|
+
current.budgetWarningReason = softReason;
|
|
2863
|
+
writeMeta(session.id, current);
|
|
2864
|
+
}
|
|
2865
|
+
}),
|
|
2866
|
+
),
|
|
2867
|
+
);
|
|
2532
2868
|
}
|
|
2533
2869
|
|
|
2534
2870
|
// Exit notifications for kind=process sessions: the poller detects
|
|
@@ -2775,9 +3111,15 @@ export default function (pi: ExtensionAPI) {
|
|
|
2775
3111
|
}
|
|
2776
3112
|
}
|
|
2777
3113
|
taskProgressCache.clear();
|
|
3114
|
+
searchLogCache.clear();
|
|
2778
3115
|
pollNeeded = true;
|
|
2779
|
-
const retentionDays = Number(process.env.PI_BABYSIT_RETENTION_DAYS);
|
|
2780
|
-
if (
|
|
3116
|
+
const retentionDays = Number(process.env.PI_BABYSIT_RETENTION_DAYS ?? "3");
|
|
3117
|
+
if (
|
|
3118
|
+
!automaticGcRan &&
|
|
3119
|
+
Number.isFinite(retentionDays) &&
|
|
3120
|
+
retentionDays > 0 &&
|
|
3121
|
+
automaticGcDue()
|
|
3122
|
+
) {
|
|
2781
3123
|
automaticGcRan = true;
|
|
2782
3124
|
const gc = gcBabysitRoots({
|
|
2783
3125
|
rootBase: ROOT_BASE,
|
|
@@ -2785,6 +3127,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2785
3127
|
olderThanMs: retentionDays * 86_400_000,
|
|
2786
3128
|
dryRun: false,
|
|
2787
3129
|
});
|
|
3130
|
+
markAutomaticGc();
|
|
2788
3131
|
if (gc.deleted.length > 0 && ctx.hasUI) {
|
|
2789
3132
|
ctx.ui.notify(
|
|
2790
3133
|
`pi-babysit GC removed ${gc.deleted.length} roots (${gc.bytes} bytes).`,
|
|
@@ -2880,35 +3223,18 @@ export default function (pi: ExtensionAPI) {
|
|
|
2880
3223
|
name: "babysit_run",
|
|
2881
3224
|
label: "Babysit: run",
|
|
2882
3225
|
description:
|
|
2883
|
-
"Run
|
|
2884
|
-
"
|
|
2885
|
-
"
|
|
2886
|
-
|
|
2887
|
-
"automatically. Complete output is returned inline only when it is small; larger output stays " +
|
|
2888
|
-
"in the log path for bounded inspection with babysit_check. " +
|
|
2889
|
-
"In non-interactive mode (`pi -p`, no UI), process mode blocks until exit because there is no " +
|
|
2890
|
-
"notification loop. Two modes: (1) `command` — run any shell command, including builds, tests, " +
|
|
2891
|
-
"dev servers, watchers, and interactive TUIs; you can type into it with babysit_send and read " +
|
|
2892
|
-
"its screen with babysit_check. If a worker disappears during startup without recording an exit, " +
|
|
2893
|
-
"`retryOnWorkerDeath` can retry one idempotent command once. " +
|
|
2894
|
-
"(2) `profile: \"subagent\"` + `task` — spawn a pi subagent that works on the task in the " +
|
|
2895
|
-
"background; poll with babysit_check, steer with babysit_send, block with babysit_wait, " +
|
|
2896
|
-
"stop with babysit_kill. Subagents cannot recursively spawn more subagents by default; " +
|
|
2897
|
-
"the top-level caller must explicitly raise `maxDepth` when creating the first worker.",
|
|
2898
|
-
promptSnippet:
|
|
2899
|
-
"Run any shell command with context-safe captured output; quick commands return metadata, longer ones continue in background",
|
|
3226
|
+
"Run a supervised shell command, or start a reusable pi subagent with `profile: \"subagent\"`. " +
|
|
3227
|
+
"Use `foreground` for results needed now; otherwise long commands notify on exit. Full logs stay on disk. " +
|
|
3228
|
+
"`returnPattern`/`returnLines` bound foreground output. Sessions support check, wait, send, and kill.",
|
|
3229
|
+
promptSnippet: "Run supervised commands or bounded pi subagents with context-safe logs",
|
|
2900
3230
|
promptGuidelines: [
|
|
2901
|
-
"Use babysit_run
|
|
2902
|
-
"Use babysit_run
|
|
2903
|
-
"
|
|
2904
|
-
"Inspect
|
|
2905
|
-
"
|
|
2906
|
-
"
|
|
2907
|
-
"
|
|
2908
|
-
"Delegate self-contained tasks (codebase recon, a parallelizable subtask, work that would pollute your context) with babysit_run { profile: \"subagent\", task }. Launch several for independent subtasks; they run concurrently.",
|
|
2909
|
-
"Set at least one subagent budget (`maxCost`, `maxTurns`, `maxToolCalls`, or `maxUsageTokens`) for bounded recon and review tasks. Omit budgets only for intentionally open-ended work; the absolute timeout remains a separate safety limit.",
|
|
2910
|
-
"Subagents cannot create further subagents by default (maximum depth 1). Only the top-level caller can explicitly opt in by setting maxDepth when it creates the first worker; nested workers inherit that limit and cannot raise it.",
|
|
2911
|
-
"After spawning subagents, do not idle-wait and do not end your turn to wait for them: keep making progress, then call babysit_wait (ids + mode any/all) when you need their results. Steer or send follow-up tasks with babysit_send; kill runaways with babysit_kill.",
|
|
3231
|
+
"Use babysit_run for shell commands and give meaningful sessions a stable name; bundle tiny related observations.",
|
|
3232
|
+
"Use babysit_run foreground mode when the next step needs the result; use returnPattern/returnLines for noisy commands. Do not foreground unbounded servers.",
|
|
3233
|
+
"After a background process starts, stop the turn for its automatic notification; never poll or sleep. Use continueAfterStart only for specific non-polling work.",
|
|
3234
|
+
"Inspect large logs with a narrow babysit_check pattern and maxBytes rather than broad tails.",
|
|
3235
|
+
"Use retryOnWorkerDeath only once and only for idempotent commands; retries may duplicate side effects.",
|
|
3236
|
+
"Delegate independent work with bounded babysit_run subagents; set at least one cost/turn/tool/token budget and keep making progress before babysit_wait.",
|
|
3237
|
+
"Subagent recursion defaults to depth 1; only a top-level caller may explicitly raise maxDepth.",
|
|
2912
3238
|
],
|
|
2913
3239
|
parameters: Type.Object({
|
|
2914
3240
|
command: Type.Optional(
|
|
@@ -2995,10 +3321,18 @@ export default function (pi: ExtensionAPI) {
|
|
|
2995
3321
|
),
|
|
2996
3322
|
foreground: Type.Optional(
|
|
2997
3323
|
Type.Boolean({
|
|
2998
|
-
description:
|
|
2999
|
-
"Process mode: wait for exit and return the result in this tool call. Use when the next step needs the result; avoid for servers/watchers unless bounded by timeout.",
|
|
3324
|
+
description: "Process: wait for exit and return the result now.",
|
|
3000
3325
|
}),
|
|
3001
3326
|
),
|
|
3327
|
+
returnPattern: Type.Optional(
|
|
3328
|
+
Type.String({ description: "Foreground/quick process: return only latest regex matches." }),
|
|
3329
|
+
),
|
|
3330
|
+
returnLines: Type.Optional(
|
|
3331
|
+
Type.Integer({ minimum: 1, maximum: 200, description: "Lines retained by returnPattern/tail (default 30)." }),
|
|
3332
|
+
),
|
|
3333
|
+
maxBytes: Type.Optional(
|
|
3334
|
+
Type.Integer({ minimum: 1_000, maximum: ANSWER_MAX_BYTES, description: "Returned process-output cap (default 8 KB)." }),
|
|
3335
|
+
),
|
|
3002
3336
|
notificationGroup: Type.Optional(
|
|
3003
3337
|
Type.String({
|
|
3004
3338
|
description:
|
|
@@ -3069,6 +3403,13 @@ export default function (pi: ExtensionAPI) {
|
|
|
3069
3403
|
details: {},
|
|
3070
3404
|
};
|
|
3071
3405
|
}
|
|
3406
|
+
if (isSubagent && (params.returnPattern || params.returnLines != null || params.maxBytes != null)) {
|
|
3407
|
+
return {
|
|
3408
|
+
content: [{ type: "text", text: "`returnPattern`, `returnLines`, and `maxBytes` are process-output options." }],
|
|
3409
|
+
isError: true,
|
|
3410
|
+
details: {},
|
|
3411
|
+
};
|
|
3412
|
+
}
|
|
3072
3413
|
if (!isSubagent && params.foreground && params.continueAfterStart) {
|
|
3073
3414
|
return {
|
|
3074
3415
|
content: [{ type: "text", text: "`foreground` and `continueAfterStart` are mutually exclusive." }],
|
|
@@ -3092,6 +3433,21 @@ export default function (pi: ExtensionAPI) {
|
|
|
3092
3433
|
|
|
3093
3434
|
// --- process mode ---
|
|
3094
3435
|
if (!isSubagent) {
|
|
3436
|
+
if (params.returnPattern) {
|
|
3437
|
+
try {
|
|
3438
|
+
new RegExp(params.returnPattern);
|
|
3439
|
+
} catch (error) {
|
|
3440
|
+
return {
|
|
3441
|
+
content: [{ type: "text", text: `Invalid returnPattern: ${String(error)}` }],
|
|
3442
|
+
isError: true,
|
|
3443
|
+
details: {},
|
|
3444
|
+
};
|
|
3445
|
+
}
|
|
3446
|
+
}
|
|
3447
|
+
const outputSelection: ProcessOutputSelection | undefined =
|
|
3448
|
+
params.returnPattern || params.returnLines != null || params.maxBytes != null
|
|
3449
|
+
? { pattern: params.returnPattern, lines: params.returnLines, maxBytes: params.maxBytes }
|
|
3450
|
+
: undefined;
|
|
3095
3451
|
const spawnOpts: ProcOpts = {
|
|
3096
3452
|
name: params.name,
|
|
3097
3453
|
command: params.command as string,
|
|
@@ -3122,14 +3478,14 @@ export default function (pi: ExtensionAPI) {
|
|
|
3122
3478
|
// the same deadline here races its terminal-state write and can return a
|
|
3123
3479
|
// false "still running" result at the boundary, so wait for the
|
|
3124
3480
|
// supervisor's definitive exit instead.
|
|
3125
|
-
let outcome = await waitForExit(res.id, null, _signal);
|
|
3481
|
+
let outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
|
|
3126
3482
|
let retried = false;
|
|
3127
3483
|
if (params.retryOnWorkerDeath && outcome.status?.state === "dead" && outcome.status.exit_code == null) {
|
|
3128
3484
|
const retry = await spawnProcess(spawnOpts);
|
|
3129
3485
|
if (!("error" in retry)) {
|
|
3130
3486
|
res = retry;
|
|
3131
3487
|
retried = true;
|
|
3132
|
-
outcome = await waitForExit(res.id, null, _signal);
|
|
3488
|
+
outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
|
|
3133
3489
|
}
|
|
3134
3490
|
}
|
|
3135
3491
|
if (ctx.hasUI) await refreshWidget(ctx);
|
|
@@ -3167,7 +3523,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3167
3523
|
}
|
|
3168
3524
|
}
|
|
3169
3525
|
if (quickStatus && quickStatus.state !== "running") {
|
|
3170
|
-
const outcome = await waitForExit(res.id, null, _signal);
|
|
3526
|
+
const outcome = await waitForExit(res.id, null, _signal, undefined, outputSelection);
|
|
3171
3527
|
await refreshWidget(ctx);
|
|
3172
3528
|
return {
|
|
3173
3529
|
content: [{ type: "text", text: `${retried ? "Retried once after external worker death.\n" : ""}${outcome.text}` }],
|
|
@@ -3369,6 +3725,13 @@ export default function (pi: ExtensionAPI) {
|
|
|
3369
3725
|
lines: Type.Optional(
|
|
3370
3726
|
Type.Number({ description: "How many tail lines or latest matches to show (default 30, max 200)." }),
|
|
3371
3727
|
),
|
|
3728
|
+
maxBytes: Type.Optional(
|
|
3729
|
+
Type.Integer({
|
|
3730
|
+
minimum: 1_000,
|
|
3731
|
+
maximum: ANSWER_MAX_BYTES,
|
|
3732
|
+
description: "Maximum returned bytes for this check (default 4 KB).",
|
|
3733
|
+
}),
|
|
3734
|
+
),
|
|
3372
3735
|
pattern: Type.Optional(
|
|
3373
3736
|
Type.String({
|
|
3374
3737
|
description:
|
|
@@ -3456,6 +3819,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3456
3819
|
}
|
|
3457
3820
|
const meta = readMeta(params.id);
|
|
3458
3821
|
const nLines = Math.min(Math.max(1, Math.floor(params.lines ?? 30)), 200);
|
|
3822
|
+
const checkMaxBytes = params.maxBytes ?? TAIL_MAX_BYTES;
|
|
3459
3823
|
if (params.pattern !== undefined) {
|
|
3460
3824
|
if (params.screen) {
|
|
3461
3825
|
return {
|
|
@@ -3471,7 +3835,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3471
3835
|
details: {},
|
|
3472
3836
|
};
|
|
3473
3837
|
}
|
|
3474
|
-
const result = await searchLog(params.id, params.pattern, nLines, signal);
|
|
3838
|
+
const result = await searchLog(params.id, params.pattern, nLines, signal, checkMaxBytes);
|
|
3475
3839
|
if (result.error) {
|
|
3476
3840
|
return {
|
|
3477
3841
|
content: [{ type: "text", text: result.error }],
|
|
@@ -3485,7 +3849,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3485
3849
|
? `--- latest matches /${params.pattern}/ ---\n${result.text}`
|
|
3486
3850
|
: `(no output matching /${params.pattern}/)`;
|
|
3487
3851
|
return {
|
|
3488
|
-
content: [{ type: "text", text: clip(`${header}\n${body}
|
|
3852
|
+
content: [{ type: "text", text: clip(`${header}\n${body}`, checkMaxBytes) }],
|
|
3489
3853
|
details: { status: st, kind, logPath: logPath(params.id), pattern: params.pattern },
|
|
3490
3854
|
};
|
|
3491
3855
|
}
|
|
@@ -3505,15 +3869,16 @@ export default function (pi: ExtensionAPI) {
|
|
|
3505
3869
|
parts.push(header);
|
|
3506
3870
|
if (params.screen) {
|
|
3507
3871
|
const sc = await bs(["screenshot", "-s", params.id, "--trim"]);
|
|
3508
|
-
parts.push(`--- screen ---\n${clip(sc.stdout.trimEnd()) || "(blank screen)"}`);
|
|
3872
|
+
parts.push(`--- screen ---\n${clip(sc.stdout.trimEnd(), checkMaxBytes) || "(blank screen)"}`);
|
|
3509
3873
|
} else {
|
|
3510
3874
|
const tail = clip(
|
|
3511
3875
|
(await bs(["log", "-s", params.id, "--tail", String(nLines)])).stdout.trimEnd(),
|
|
3876
|
+
checkMaxBytes,
|
|
3512
3877
|
);
|
|
3513
3878
|
parts.push(tail ? `--- recent output ---\n${tail}` : "(no output yet)");
|
|
3514
3879
|
}
|
|
3515
3880
|
return {
|
|
3516
|
-
content: [{ type: "text", text: clip(parts.join("\n")) }],
|
|
3881
|
+
content: [{ type: "text", text: clip(parts.join("\n"), checkMaxBytes) }],
|
|
3517
3882
|
details: { status: st, kind: "process", logPath: logPath(params.id) },
|
|
3518
3883
|
};
|
|
3519
3884
|
}
|
|
@@ -3536,7 +3901,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3536
3901
|
: " · working";
|
|
3537
3902
|
}
|
|
3538
3903
|
if (st.exit_code != null) header += ` exit_code=${st.exit_code}`;
|
|
3539
|
-
header += ` turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.
|
|
3904
|
+
header += ` turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}`;
|
|
3540
3905
|
if (prog.tokens != null) header += ` ctx=${prog.tokens}`;
|
|
3541
3906
|
if (prog.modelCalls > 0) header += ` usage=${prog.usageTokens} $${prog.cost.toFixed(4)}`;
|
|
3542
3907
|
if (st.note) header += ` ⚑ ${st.note}`;
|
|
@@ -3545,7 +3910,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3545
3910
|
if (prog.errorMsg) parts.push(`⚠ error: ${clip(prog.errorMsg, ANSWER_MAX_BYTES)}`);
|
|
3546
3911
|
|
|
3547
3912
|
if (recent.length > 0) {
|
|
3548
|
-
const skipped = prog.
|
|
3913
|
+
const skipped = Math.max(0, prog.toolCallCount - recent.length);
|
|
3549
3914
|
parts.push(
|
|
3550
3915
|
`--- recent tool calls${skipped > 0 ? ` (+${skipped} earlier)` : ""} ---\n` +
|
|
3551
3916
|
recent.map((t) => ` ${t.summary}`).join("\n"),
|
|
@@ -3554,16 +3919,16 @@ export default function (pi: ExtensionAPI) {
|
|
|
3554
3919
|
|
|
3555
3920
|
if (prog.finalText.trim()) {
|
|
3556
3921
|
parts.push(`--- answer so far ---\n${clip(prog.finalText.trim(), ANSWER_MAX_BYTES)}`);
|
|
3557
|
-
} else if (prog.
|
|
3922
|
+
} else if (prog.toolCallCount === 0 && st.state !== "running") {
|
|
3558
3923
|
parts.push(buildSubagentExitDiagnostic(prog, logPath(params.id)));
|
|
3559
|
-
} else if (prog.
|
|
3924
|
+
} else if (prog.toolCallCount === 0) {
|
|
3560
3925
|
parts.push("(starting up… no events yet)");
|
|
3561
3926
|
} else {
|
|
3562
3927
|
parts.push("(working… no answer text yet)");
|
|
3563
3928
|
}
|
|
3564
3929
|
|
|
3565
3930
|
return {
|
|
3566
|
-
content: [{ type: "text", text: clip(parts.join("\n")) }],
|
|
3931
|
+
content: [{ type: "text", text: clip(parts.join("\n"), checkMaxBytes) }],
|
|
3567
3932
|
details: { status: st, progress: prog, kind: "subagent" },
|
|
3568
3933
|
};
|
|
3569
3934
|
},
|
|
@@ -3577,7 +3942,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3577
3942
|
"Send input to a babysit session. Process: `text` types a line into its stdin (PTY), " +
|
|
3578
3943
|
"`keys` presses named keys (Enter, Tab, Esc, Up/Down/Left/Right, C-c, F1…) — use with " +
|
|
3579
3944
|
"babysit_check { screen: true } to drive interactive programs. Subagent: `text` is " +
|
|
3580
|
-
"STEERING while it works, or a NEW TASK
|
|
3945
|
+
"STEERING while it works, or a NEW TASK after the current task settles (mode: auto/steer/task) — this " +
|
|
3581
3946
|
"is how you resume a finished subagent with full context.",
|
|
3582
3947
|
promptSnippet: "Send text/keys to a process, or steering/follow-up tasks to a subagent",
|
|
3583
3948
|
parameters: Type.Object({
|
|
@@ -3593,7 +3958,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3593
3958
|
mode: Type.Optional(
|
|
3594
3959
|
StringEnum(["auto", "steer", "task"] as const, {
|
|
3595
3960
|
description:
|
|
3596
|
-
"Subagent only. auto (default): steer
|
|
3961
|
+
"Subagent only. auto (default): steer unless the current task is settled. task requires confirmed settlement; steer always sends guidance.",
|
|
3597
3962
|
}),
|
|
3598
3963
|
),
|
|
3599
3964
|
noNewline: Type.Optional(
|
|
@@ -3693,15 +4058,39 @@ export default function (pi: ExtensionAPI) {
|
|
|
3693
4058
|
};
|
|
3694
4059
|
}
|
|
3695
4060
|
let mode = params.mode ?? "auto";
|
|
3696
|
-
if (mode === "auto") {
|
|
3697
|
-
//
|
|
4061
|
+
if (mode === "auto" || mode === "task") {
|
|
4062
|
+
// A prompt sent while the current run is streaming can queue behind that
|
|
4063
|
+
// run while immediately replacing our per-task offsets and budget state.
|
|
4064
|
+
// Establish idleness before every new task; auto safely falls back to
|
|
4065
|
+
// steering when state is unknown, while an explicit task fails closed.
|
|
3698
4066
|
const gs = await sendRpc(params.id, { type: "get_state" });
|
|
3699
|
-
let streaming
|
|
4067
|
+
let streaming: boolean | undefined;
|
|
3700
4068
|
if (!("error" in gs)) {
|
|
3701
4069
|
const r = await rpcResponse(params.id, gs.offset, "get_state", "10s");
|
|
3702
4070
|
if (r.ok) streaming = Boolean((r.data as { isStreaming?: boolean })?.isStreaming);
|
|
3703
4071
|
}
|
|
3704
|
-
|
|
4072
|
+
let currentTaskDone: boolean | undefined;
|
|
4073
|
+
try {
|
|
4074
|
+
currentTaskDone = taskProgressOf(params.id).progress.done;
|
|
4075
|
+
} catch {
|
|
4076
|
+
/* fail closed below rather than replacing unknown task state */
|
|
4077
|
+
}
|
|
4078
|
+
const resolved = resolveSubagentSendMode(mode, streaming, currentTaskDone);
|
|
4079
|
+
if ("error" in resolved) {
|
|
4080
|
+
return {
|
|
4081
|
+
content: [{
|
|
4082
|
+
type: "text",
|
|
4083
|
+
text: resolved.error === "busy"
|
|
4084
|
+
? `Subagent ${params.id} is still streaming; use mode \"steer\" or wait for the current task to settle before starting another task.`
|
|
4085
|
+
: resolved.error === "unsettled"
|
|
4086
|
+
? `Subagent ${params.id} has not settled its current task (it may be parked on a background process); wait for completion before starting another task.`
|
|
4087
|
+
: `Could not verify that subagent ${params.id} is idle and settled; retry with mode \"task\" after checking its state.`,
|
|
4088
|
+
}],
|
|
4089
|
+
isError: true,
|
|
4090
|
+
details: { mode: "task" },
|
|
4091
|
+
};
|
|
4092
|
+
}
|
|
4093
|
+
mode = resolved.mode;
|
|
3705
4094
|
}
|
|
3706
4095
|
const deliveryCleanupAfter =
|
|
3707
4096
|
mode === "steer"
|
|
@@ -3755,10 +4144,13 @@ export default function (pi: ExtensionAPI) {
|
|
|
3755
4144
|
depth: meta?.depth,
|
|
3756
4145
|
maxDepth: meta?.maxDepth,
|
|
3757
4146
|
budget: meta?.budget,
|
|
3758
|
-
// Each follow-up task receives
|
|
4147
|
+
// Each follow-up task receives fresh budget and usage-accounting windows.
|
|
4148
|
+
budgetWarnedAt: undefined,
|
|
4149
|
+
budgetWarningReason: undefined,
|
|
3759
4150
|
budgetExceededAt: undefined,
|
|
3760
4151
|
budgetReason: undefined,
|
|
3761
4152
|
budgetKilled: undefined,
|
|
4153
|
+
usageReportedOffset: undefined,
|
|
3762
4154
|
});
|
|
3763
4155
|
} else if (delivery.tempDir && meta) {
|
|
3764
4156
|
writeMeta(params.id, {
|
|
@@ -3793,7 +4185,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3793
4185
|
"Block until babysit session(s) finish, then return the result. A process finishes " +
|
|
3794
4186
|
"when it EXITS (or, with `expect`, as soon as a regex appears in its output — e.g. wait " +
|
|
3795
4187
|
"for 'listening on' before hitting a dev server). A subagent finishes when its current " +
|
|
3796
|
-
"TASK completes (the
|
|
4188
|
+
"TASK completes (the worker remains reusable only during its configured idle grace). Pass `id` for one session, or " +
|
|
3797
4189
|
"`ids` + `mode`: 'all' (default) waits for every one, 'any' returns on the FIRST finisher. " +
|
|
3798
4190
|
"Multi-session results are capped at the inline-output limit (8 KB by default); use `maxBytes` " +
|
|
3799
4191
|
"to opt into a larger result up to 24 KB. " +
|
|
@@ -3865,9 +4257,11 @@ export default function (pi: ExtensionAPI) {
|
|
|
3865
4257
|
|
|
3866
4258
|
if (ids.length === 1) {
|
|
3867
4259
|
const r = await waitFor(ids[0], limitMs, signal, params.expect);
|
|
4260
|
+
const usage = claimOutcomeUsage(r);
|
|
3868
4261
|
return {
|
|
3869
4262
|
content: [{ type: "text", text: r.text }],
|
|
3870
4263
|
isError: !r.ok,
|
|
4264
|
+
usage,
|
|
3871
4265
|
details: {
|
|
3872
4266
|
status: r.status,
|
|
3873
4267
|
progress: r.progress,
|
|
@@ -3883,6 +4277,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3883
4277
|
ids.map((i) => waitFor(i, limitMs, signal, params.expect)),
|
|
3884
4278
|
);
|
|
3885
4279
|
const ok = results.every((r) => r.ok);
|
|
4280
|
+
const usage = sumNestedUsage(results.map(claimOutcomeUsage));
|
|
3886
4281
|
return {
|
|
3887
4282
|
content: [
|
|
3888
4283
|
{
|
|
@@ -3894,6 +4289,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3894
4289
|
},
|
|
3895
4290
|
],
|
|
3896
4291
|
isError: !ok,
|
|
4292
|
+
usage,
|
|
3897
4293
|
details: {
|
|
3898
4294
|
results: results.map((r) => ({ id: r.id, kind: r.kind, ok: r.ok })),
|
|
3899
4295
|
},
|
|
@@ -3910,6 +4306,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3910
4306
|
ids.map((i) => waitFor(i, limitMs, ctrl.signal, params.expect)),
|
|
3911
4307
|
);
|
|
3912
4308
|
const others = ids.filter((i) => i !== first.id);
|
|
4309
|
+
const usage = claimOutcomeUsage(first);
|
|
3913
4310
|
return {
|
|
3914
4311
|
content: [
|
|
3915
4312
|
{
|
|
@@ -3923,6 +4320,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
3923
4320
|
},
|
|
3924
4321
|
],
|
|
3925
4322
|
isError: !first.ok,
|
|
4323
|
+
usage,
|
|
3926
4324
|
details: { first: { id: first.id, kind: first.kind, ok: first.ok }, remaining: others },
|
|
3927
4325
|
};
|
|
3928
4326
|
} finally {
|
|
@@ -4073,7 +4471,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
4073
4471
|
// Parse the RPC event stream and show the final answer, not raw JSONL.
|
|
4074
4472
|
const prog = taskProgressOf(picked.id).progress;
|
|
4075
4473
|
const stats =
|
|
4076
|
-
`turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.
|
|
4474
|
+
`turns=${prog.turns} calls=${prog.modelCalls} tools=${prog.toolCallCount}` +
|
|
4077
4475
|
(prog.tokens != null ? ` ctx=${prog.tokens}` : "") +
|
|
4078
4476
|
(prog.modelCalls > 0 ? ` usage=${prog.usageTokens} $${prog.cost.toFixed(4)}` : "");
|
|
4079
4477
|
const body =
|