@ryan_nookpi/pi-extension-subagent 0.4.4 → 0.4.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/claude-stream-parser.ts +25 -0
- package/cli.ts +3 -2
- package/commands.ts +17 -0
- package/context-limits.ts +12 -7
- package/failure-telemetry.ts +66 -0
- package/lifecycle.ts +5 -0
- package/package.json +2 -1
- package/runner.ts +18 -1
- package/store.ts +4 -0
- package/tool-execute.ts +57 -7
- package/types.ts +13 -0
- package/widget.ts +2 -1
package/claude-stream-parser.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
/** biome-ignore-all lint/suspicious/noExplicitAny: Claude stream events are dynamic runtime data. */
|
|
2
2
|
import type { Message } from "@earendil-works/pi-ai";
|
|
3
|
+
import { classifySubagentFailure, countTextChars } from "./failure-telemetry.js";
|
|
3
4
|
import { extractActivityPreviewFromTextDelta, extractThoughtText } from "./live-preview.js";
|
|
4
5
|
import type { SingleResult } from "./types.js";
|
|
5
6
|
|
|
@@ -32,6 +33,7 @@ export interface ClaudeStreamState {
|
|
|
32
33
|
completedUsage?: ClaudeUsageTotals;
|
|
33
34
|
currentMessageUsage?: ClaudeUsageTotals;
|
|
34
35
|
lastTurnContextTokens?: number;
|
|
36
|
+
peakContextTokens?: number;
|
|
35
37
|
resultReceived: boolean;
|
|
36
38
|
resultEvent: any | undefined;
|
|
37
39
|
isError: boolean;
|
|
@@ -39,6 +41,8 @@ export interface ClaudeStreamState {
|
|
|
39
41
|
liveActivityPreview: string | undefined;
|
|
40
42
|
currentToolName: string | undefined;
|
|
41
43
|
currentToolInput: string;
|
|
44
|
+
lastToolName?: string;
|
|
45
|
+
lastToolOutputChars?: number;
|
|
42
46
|
}
|
|
43
47
|
|
|
44
48
|
export function createStreamState(): ClaudeStreamState {
|
|
@@ -56,6 +60,7 @@ export function createStreamState(): ClaudeStreamState {
|
|
|
56
60
|
completedUsage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
|
57
61
|
currentMessageUsage: undefined,
|
|
58
62
|
lastTurnContextTokens: 0,
|
|
63
|
+
peakContextTokens: 0,
|
|
59
64
|
resultReceived: false,
|
|
60
65
|
resultEvent: undefined,
|
|
61
66
|
isError: false,
|
|
@@ -63,6 +68,8 @@ export function createStreamState(): ClaudeStreamState {
|
|
|
63
68
|
liveActivityPreview: undefined,
|
|
64
69
|
currentToolName: undefined,
|
|
65
70
|
currentToolInput: "",
|
|
71
|
+
lastToolName: undefined,
|
|
72
|
+
lastToolOutputChars: undefined,
|
|
66
73
|
};
|
|
67
74
|
}
|
|
68
75
|
|
|
@@ -99,6 +106,7 @@ function syncUsageTotals(state: ClaudeStreamState): void {
|
|
|
99
106
|
state.usage.cacheRead = completed.cacheRead + (current?.cacheRead || 0);
|
|
100
107
|
state.usage.cacheWrite = completed.cacheWrite + (current?.cacheWrite || 0);
|
|
101
108
|
state.usage.contextTokens = current ? getContextTokensFromUsage(current) : (state.lastTurnContextTokens ?? 0);
|
|
109
|
+
state.peakContextTokens = Math.max(state.peakContextTokens ?? 0, state.usage.contextTokens);
|
|
102
110
|
}
|
|
103
111
|
|
|
104
112
|
function finalizeCurrentMessageUsage(state: ClaudeStreamState): void {
|
|
@@ -173,6 +181,7 @@ function processStreamEvent(state: ClaudeStreamState, ev: any): boolean {
|
|
|
173
181
|
} else if (block?.type === "tool_use") {
|
|
174
182
|
state.liveToolCalls++;
|
|
175
183
|
state.currentToolName = block.name;
|
|
184
|
+
state.lastToolName = block.name;
|
|
176
185
|
state.currentToolInput = "";
|
|
177
186
|
state.liveActivityPreview = `\u2192 ${block.name}`;
|
|
178
187
|
} else if (block?.type === "thinking") {
|
|
@@ -279,6 +288,10 @@ function processUserEvent(state: ClaudeStreamState, event: any): void {
|
|
|
279
288
|
}
|
|
280
289
|
}
|
|
281
290
|
|
|
291
|
+
if (contentBlocks.length > 0) {
|
|
292
|
+
state.lastToolOutputChars = countTextChars(contentBlocks);
|
|
293
|
+
}
|
|
294
|
+
|
|
282
295
|
const piMessage = {
|
|
283
296
|
role: "user" as const,
|
|
284
297
|
content: contentBlocks.length > 0 ? contentBlocks : [{ type: "text" as const, text: "" }],
|
|
@@ -315,6 +328,7 @@ function processResultEvent(state: ClaudeStreamState, event: any): void {
|
|
|
315
328
|
(state.lastTurnContextTokens ?? 0) > 0 ? (state.lastTurnContextTokens ?? 0) : getContextTokensFromUsage(totals),
|
|
316
329
|
turns: event.num_turns || state.usage.turns,
|
|
317
330
|
};
|
|
331
|
+
state.peakContextTokens = Math.max(state.peakContextTokens ?? 0, state.usage.contextTokens);
|
|
318
332
|
}
|
|
319
333
|
}
|
|
320
334
|
|
|
@@ -372,6 +386,9 @@ export function stateToSingleResult(
|
|
|
372
386
|
model: state.model,
|
|
373
387
|
stopReason: state.stopReason,
|
|
374
388
|
errorMessage: state.errorMessage,
|
|
389
|
+
peakContextTokens: (state.peakContextTokens ?? 0) > 0 ? state.peakContextTokens : undefined,
|
|
390
|
+
lastToolName: state.lastToolName,
|
|
391
|
+
lastToolOutputChars: state.lastToolOutputChars,
|
|
375
392
|
step,
|
|
376
393
|
runtime: "claude",
|
|
377
394
|
claudeSessionId: state.sessionId,
|
|
@@ -396,5 +413,13 @@ export function stateToSingleResult(
|
|
|
396
413
|
result.errorMessage = `Permission denied for tools: ${names}`;
|
|
397
414
|
}
|
|
398
415
|
|
|
416
|
+
result.errorClass = classifySubagentFailure({
|
|
417
|
+
failed: exitCode !== 0 || result.stopReason === "error" || result.stopReason === "aborted" || state.isError,
|
|
418
|
+
stopReason: result.stopReason,
|
|
419
|
+
exitCode,
|
|
420
|
+
errorMessage: result.errorMessage,
|
|
421
|
+
stderr,
|
|
422
|
+
});
|
|
423
|
+
|
|
399
424
|
return result;
|
|
400
425
|
}
|
package/cli.ts
CHANGED
|
@@ -15,7 +15,7 @@ export type SubagentCliParseResult =
|
|
|
15
15
|
| { type: "help" }
|
|
16
16
|
| { type: "agents" }
|
|
17
17
|
| { type: "params"; params: Record<string, unknown> }
|
|
18
|
-
| { type: "error"; message: string };
|
|
18
|
+
| { type: "error"; message: string; showHelp?: boolean };
|
|
19
19
|
|
|
20
20
|
export const SUBAGENT_CLI_HELP_TEXT = [
|
|
21
21
|
"Subagent CLI (LLM interface)",
|
|
@@ -451,7 +451,8 @@ export function parseSubagentToolCommand(
|
|
|
451
451
|
if ("error" in tokenized) {
|
|
452
452
|
return {
|
|
453
453
|
type: "error",
|
|
454
|
-
message: `❌ Syntax error: ${tokenized.error}\
|
|
454
|
+
message: `❌ Syntax error: ${tokenized.error}\nClose the quote or wrap the task after \`--\` in matching quotes.`,
|
|
455
|
+
showHelp: false,
|
|
455
456
|
};
|
|
456
457
|
}
|
|
457
458
|
|
package/commands.ts
CHANGED
|
@@ -640,6 +640,10 @@ function restoreRunsFromSession(store: SubagentStore, ctx: any, pi?: ExtensionAP
|
|
|
640
640
|
contextMode: d.contextMode ?? existing?.contextMode,
|
|
641
641
|
usage: d.usage ?? existing?.usage,
|
|
642
642
|
model: d.model ?? existing?.model,
|
|
643
|
+
errorClass: d.errorClass ?? existing?.errorClass,
|
|
644
|
+
peakContextTokens: d.peakContextTokens ?? existing?.peakContextTokens,
|
|
645
|
+
lastToolName: d.lastToolName ?? existing?.lastToolName,
|
|
646
|
+
lastToolOutputChars: d.lastToolOutputChars ?? existing?.lastToolOutputChars,
|
|
643
647
|
thoughtText: d.thoughtText ?? d.progressText ?? existing?.thoughtText,
|
|
644
648
|
source: restoredSource,
|
|
645
649
|
runtime: d.runtime ?? existing?.runtime,
|
|
@@ -697,6 +701,10 @@ function restoreRunsFromSession(store: SubagentStore, ctx: any, pi?: ExtensionAP
|
|
|
697
701
|
contextMode: d.contextMode ?? existing?.contextMode,
|
|
698
702
|
usage: existing?.usage,
|
|
699
703
|
model: existing?.model,
|
|
704
|
+
errorClass: existing?.errorClass ?? "unknown",
|
|
705
|
+
peakContextTokens: existing?.peakContextTokens,
|
|
706
|
+
lastToolName: existing?.lastToolName,
|
|
707
|
+
lastToolOutputChars: existing?.lastToolOutputChars,
|
|
700
708
|
thoughtText: d.thoughtText ?? d.progressText ?? existing?.thoughtText,
|
|
701
709
|
source: restoredSource,
|
|
702
710
|
runtime: d.runtime ?? existing?.runtime,
|
|
@@ -1514,6 +1522,10 @@ export function registerAll(pi: ExtensionAPI, store: SubagentStore): SubagentReg
|
|
|
1514
1522
|
usage: result.usage,
|
|
1515
1523
|
model: result.model,
|
|
1516
1524
|
source: result.agentSource,
|
|
1525
|
+
errorClass: runState.errorClass,
|
|
1526
|
+
peakContextTokens: runState.peakContextTokens,
|
|
1527
|
+
lastToolName: runState.lastToolName,
|
|
1528
|
+
lastToolOutputChars: runState.lastToolOutputChars,
|
|
1517
1529
|
thoughtText: runState.thoughtText,
|
|
1518
1530
|
retryCount: runState.retryCount,
|
|
1519
1531
|
status: runState.status,
|
|
@@ -1546,6 +1558,7 @@ export function registerAll(pi: ExtensionAPI, store: SubagentStore): SubagentReg
|
|
|
1546
1558
|
} catch (error: any) {
|
|
1547
1559
|
if (runState.removed || store.disposed) return;
|
|
1548
1560
|
runState.status = "error";
|
|
1561
|
+
runState.errorClass = "process_error";
|
|
1549
1562
|
runState.elapsedMs = Date.now() - runState.startedAt;
|
|
1550
1563
|
runState.lastLine =
|
|
1551
1564
|
runState.autoAbortReason ?? (error?.message ? String(error.message) : "Subagent execution failed");
|
|
@@ -1572,6 +1585,10 @@ export function registerAll(pi: ExtensionAPI, store: SubagentStore): SubagentReg
|
|
|
1572
1585
|
elapsedMs: runState.elapsedMs,
|
|
1573
1586
|
lastActivityAt: runState.lastActivityAt,
|
|
1574
1587
|
error: runState.lastLine,
|
|
1588
|
+
errorClass: runState.errorClass,
|
|
1589
|
+
peakContextTokens: runState.peakContextTokens,
|
|
1590
|
+
lastToolName: runState.lastToolName,
|
|
1591
|
+
lastToolOutputChars: runState.lastToolOutputChars,
|
|
1575
1592
|
thoughtText: runState.thoughtText,
|
|
1576
1593
|
status: runState.status,
|
|
1577
1594
|
runtime: runState.runtime,
|
package/context-limits.ts
CHANGED
|
@@ -8,12 +8,12 @@
|
|
|
8
8
|
* partial findings instead of surfacing a raw provider error.
|
|
9
9
|
*
|
|
10
10
|
* 2. Proactive guard (④): some providers register a context window that is
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
* pi
|
|
15
|
-
*
|
|
16
|
-
*
|
|
11
|
+
* unsafe for long internal tool loops. Some `openai-codex` models expose a
|
|
12
|
+
* larger registry window than the backend actually enforces; Spark correctly
|
|
13
|
+
* reports 128k but can still cross the threshold between tool turns because
|
|
14
|
+
* pi checks native compaction only after the whole agent run. We watch
|
|
15
|
+
* reported tokens live and stop just below each known cliff, preserving
|
|
16
|
+
* findings.
|
|
17
17
|
*
|
|
18
18
|
* Overflow patterns are aligned with `@earendil-works/pi-ai`'s OVERFLOW_PATTERNS.
|
|
19
19
|
*/
|
|
@@ -73,6 +73,11 @@ export function isContextOverflowText(text: string | undefined | null): boolean
|
|
|
73
73
|
* Only applied to the pi runtime; the claude runtime handles its own limits.
|
|
74
74
|
*/
|
|
75
75
|
const GUARD_CEILINGS: Array<{ prefix: string; tokens: number }> = [
|
|
76
|
+
// GPT-5.3 Codex Spark has a hard 128k text-only window. Pi checks native
|
|
77
|
+
// threshold compaction only after the full agent run, not between internal
|
|
78
|
+
// tool-use turns, so stop at 115k before one large read/reasoning turn can
|
|
79
|
+
// cross the provider cliff.
|
|
80
|
+
{ prefix: "openai-codex/gpt-5.3-codex-spark", tokens: 115_000 },
|
|
76
81
|
// GPT-5.6 Codex models expose a 372k input window in pi. Keep the same 37k
|
|
77
82
|
// safety margin used by the older 272k-window models so long tool turns stop
|
|
78
83
|
// at 335k while partial findings can still be preserved. The family prefix
|
|
@@ -80,7 +85,7 @@ const GUARD_CEILINGS: Array<{ prefix: string; tokens: number }> = [
|
|
|
80
85
|
{ prefix: "openai-codex/gpt-5.6", tokens: 335_000 },
|
|
81
86
|
// Observed 272k-window codex models hard-error around ~264k with a raw
|
|
82
87
|
// provider error and no compaction. Cut at 235k to preserve the exploration
|
|
83
|
-
// so far.
|
|
88
|
+
// so far. Other unlisted models fall back to overflow detection/recovery (②).
|
|
84
89
|
{ prefix: "openai-codex/gpt-5.5", tokens: 235_000 },
|
|
85
90
|
// gpt-5.4 and gpt-5.4-mini both use a 272k window and are covered via startsWith.
|
|
86
91
|
{ prefix: "openai-codex/gpt-5.4", tokens: 235_000 },
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
import { isContextOverflowText } from "./context-limits.js";
|
|
2
|
+
|
|
3
|
+
export type SubagentErrorClass =
|
|
4
|
+
| "context_overflow"
|
|
5
|
+
| "overloaded"
|
|
6
|
+
| "rate_limit"
|
|
7
|
+
| "tool_error"
|
|
8
|
+
| "aborted"
|
|
9
|
+
| "process_error"
|
|
10
|
+
| "unknown";
|
|
11
|
+
|
|
12
|
+
export interface FailureTelemetryInput {
|
|
13
|
+
failed: boolean;
|
|
14
|
+
stopReason?: string;
|
|
15
|
+
exitCode?: number;
|
|
16
|
+
errorMessage?: string;
|
|
17
|
+
stderr?: string;
|
|
18
|
+
output?: string;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
const OVERLOADED_PATTERNS = [
|
|
22
|
+
/servers? (?:are|is) currently overloaded/i,
|
|
23
|
+
/server overloaded/i,
|
|
24
|
+
/service unavailable/i,
|
|
25
|
+
/\b(?:http\s*)?5(?:03|29)\b/i,
|
|
26
|
+
];
|
|
27
|
+
|
|
28
|
+
const RATE_LIMIT_PATTERNS = [/rate limit/i, /too many requests/i, /\bhttp\s*429\b/i, /\b429\b.*request/i];
|
|
29
|
+
|
|
30
|
+
const TOOL_ERROR_PATTERNS = [
|
|
31
|
+
/tool(?: execution| call)? (?:failed|error)/i,
|
|
32
|
+
/tool_execution_error/i,
|
|
33
|
+
/permission denied for tools?/i,
|
|
34
|
+
/invalid tool/i,
|
|
35
|
+
];
|
|
36
|
+
|
|
37
|
+
export function classifySubagentFailure(input: FailureTelemetryInput): SubagentErrorClass | undefined {
|
|
38
|
+
if (!input.failed) return undefined;
|
|
39
|
+
const text = [input.errorMessage, input.stderr, input.output].filter(Boolean).join("\n");
|
|
40
|
+
|
|
41
|
+
if (isContextOverflowText(text)) return "context_overflow";
|
|
42
|
+
if (OVERLOADED_PATTERNS.some((pattern) => pattern.test(text))) return "overloaded";
|
|
43
|
+
if (RATE_LIMIT_PATTERNS.some((pattern) => pattern.test(text))) return "rate_limit";
|
|
44
|
+
if (input.stopReason === "aborted") return "aborted";
|
|
45
|
+
if (TOOL_ERROR_PATTERNS.some((pattern) => pattern.test(text))) return "tool_error";
|
|
46
|
+
if ((input.exitCode ?? 0) !== 0) return "process_error";
|
|
47
|
+
return "unknown";
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
export function countTextChars(content: unknown): number {
|
|
51
|
+
if (typeof content === "string") return content.length;
|
|
52
|
+
if (!Array.isArray(content)) return 0;
|
|
53
|
+
let chars = 0;
|
|
54
|
+
for (const part of content) {
|
|
55
|
+
if (typeof part === "string") {
|
|
56
|
+
chars += part.length;
|
|
57
|
+
continue;
|
|
58
|
+
}
|
|
59
|
+
if (!part || typeof part !== "object") continue;
|
|
60
|
+
const record = part as Record<string, unknown>;
|
|
61
|
+
if (typeof record.text === "string") chars += record.text.length;
|
|
62
|
+
else if (typeof record.content === "string") chars += record.content.length;
|
|
63
|
+
else if (Array.isArray(record.content)) chars += countTextChars(record.content);
|
|
64
|
+
}
|
|
65
|
+
return chars;
|
|
66
|
+
}
|
package/lifecycle.ts
CHANGED
|
@@ -74,6 +74,7 @@ export function shutdownSubagentRuns(store: SubagentStore, pi: ExtensionAPI, rea
|
|
|
74
74
|
if (run.status !== "running") continue;
|
|
75
75
|
const message = `Aborted because the parent pi session ${reason} is shutting down.`;
|
|
76
76
|
run.status = "error";
|
|
77
|
+
run.errorClass = "aborted";
|
|
77
78
|
run.elapsedMs = Date.now() - run.startedAt;
|
|
78
79
|
run.lastActivityAt = Date.now();
|
|
79
80
|
run.lastLine = message;
|
|
@@ -98,6 +99,10 @@ export function shutdownSubagentRuns(store: SubagentStore, pi: ExtensionAPI, rea
|
|
|
98
99
|
displayTask: run.displayTask,
|
|
99
100
|
status: "error",
|
|
100
101
|
error: message,
|
|
102
|
+
errorClass: run.errorClass,
|
|
103
|
+
peakContextTokens: run.peakContextTokens,
|
|
104
|
+
lastToolName: run.lastToolName,
|
|
105
|
+
lastToolOutputChars: run.lastToolOutputChars,
|
|
101
106
|
startedAt: run.startedAt,
|
|
102
107
|
elapsedMs: run.elapsedMs,
|
|
103
108
|
lastActivityAt: run.lastActivityAt,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ryan_nookpi/pi-extension-subagent",
|
|
3
|
-
"version": "0.4.
|
|
3
|
+
"version": "0.4.7",
|
|
4
4
|
"description": "Asynchronous subagent delegation for pi with run, batch, chain, and continuation workflows.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"repository": {
|
|
@@ -34,6 +34,7 @@
|
|
|
34
34
|
"diagnostics.ts",
|
|
35
35
|
"display-task.ts",
|
|
36
36
|
"escalation.ts",
|
|
37
|
+
"failure-telemetry.ts",
|
|
37
38
|
"format.ts",
|
|
38
39
|
"group-pending.ts",
|
|
39
40
|
"index.ts",
|
package/runner.ts
CHANGED
|
@@ -24,6 +24,7 @@ import {
|
|
|
24
24
|
serializeDiagnosticValue,
|
|
25
25
|
toDiagnosticError,
|
|
26
26
|
} from "./diagnostics.js";
|
|
27
|
+
import { classifySubagentFailure, countTextChars } from "./failure-telemetry.js";
|
|
27
28
|
import { formatToolCallPlain } from "./format.js";
|
|
28
29
|
import {
|
|
29
30
|
extractActivityPreviewFromTextDelta,
|
|
@@ -871,6 +872,7 @@ async function runPiAgent(
|
|
|
871
872
|
if (event.type === "tool_execution_start") {
|
|
872
873
|
currentResult.liveToolCalls = (currentResult.liveToolCalls ?? 0) + 1;
|
|
873
874
|
if (typeof event.toolName === "string") {
|
|
875
|
+
currentResult.lastToolName = event.toolName;
|
|
874
876
|
currentResult.liveActivityPreview = formatPiToolExecutionPreview(event.toolName, event.args);
|
|
875
877
|
}
|
|
876
878
|
emitUpdate();
|
|
@@ -894,6 +896,7 @@ async function runPiAgent(
|
|
|
894
896
|
currentResult.usage.contextTokens = usage.totalTokens || 0;
|
|
895
897
|
}
|
|
896
898
|
peakContextTokens = Math.max(peakContextTokens, currentResult.usage.contextTokens);
|
|
899
|
+
currentResult.peakContextTokens = peakContextTokens;
|
|
897
900
|
// ④ Proactive context guard: stop just below the provider's real
|
|
898
901
|
// ceiling so heavy runs surface partial findings instead of a raw
|
|
899
902
|
// context-overflow error one turn later.
|
|
@@ -944,7 +947,10 @@ async function runPiAgent(
|
|
|
944
947
|
}
|
|
945
948
|
|
|
946
949
|
if (event.type === "tool_result_end" && event.message) {
|
|
947
|
-
|
|
950
|
+
const toolResultMessage = event.message as Message & { toolName?: string };
|
|
951
|
+
currentResult.messages.push(toolResultMessage);
|
|
952
|
+
if (typeof toolResultMessage.toolName === "string") currentResult.lastToolName = toolResultMessage.toolName;
|
|
953
|
+
currentResult.lastToolOutputChars = countTextChars(toolResultMessage.content);
|
|
948
954
|
emitUpdate();
|
|
949
955
|
if (sawAgentEnd) scheduleAgentEndForceResolve();
|
|
950
956
|
return;
|
|
@@ -1112,6 +1118,17 @@ async function runPiAgent(
|
|
|
1112
1118
|
});
|
|
1113
1119
|
|
|
1114
1120
|
currentResult.exitCode = exitCode;
|
|
1121
|
+
if (peakContextTokens > 0) {
|
|
1122
|
+
currentResult.peakContextTokens = Math.max(currentResult.peakContextTokens ?? 0, peakContextTokens);
|
|
1123
|
+
}
|
|
1124
|
+
currentResult.errorClass = classifySubagentFailure({
|
|
1125
|
+
failed: exitCode !== 0 || currentResult.stopReason === "error" || currentResult.stopReason === "aborted",
|
|
1126
|
+
stopReason: wasAborted ? "aborted" : currentResult.stopReason,
|
|
1127
|
+
exitCode,
|
|
1128
|
+
errorMessage: currentResult.errorMessage,
|
|
1129
|
+
stderr: currentResult.stderr,
|
|
1130
|
+
output: getFinalOutput(currentResult.messages),
|
|
1131
|
+
});
|
|
1115
1132
|
if (wasAborted) throw new Error("Subagent was aborted");
|
|
1116
1133
|
return currentResult;
|
|
1117
1134
|
} finally {
|
package/store.ts
CHANGED
|
@@ -127,6 +127,10 @@ export function updateRunFromResult(state: CommandRunState, result: SingleResult
|
|
|
127
127
|
state.toolCalls = Math.max(collectToolCallCount(result.messages), result.liveToolCalls ?? 0);
|
|
128
128
|
state.usage = result.usage;
|
|
129
129
|
state.model = result.model ?? state.model;
|
|
130
|
+
state.errorClass = result.errorClass;
|
|
131
|
+
state.peakContextTokens = Math.max(state.peakContextTokens ?? 0, result.peakContextTokens ?? 0);
|
|
132
|
+
if (result.lastToolName) state.lastToolName = result.lastToolName;
|
|
133
|
+
if (result.lastToolOutputChars != null) state.lastToolOutputChars = result.lastToolOutputChars;
|
|
130
134
|
if (result.usage?.turns != null) state.turnCount = result.usage.turns;
|
|
131
135
|
if (result.thoughtText) state.thoughtText = result.thoughtText;
|
|
132
136
|
|
package/tool-execute.ts
CHANGED
|
@@ -31,6 +31,7 @@ import {
|
|
|
31
31
|
summarizeSubagentDisplayTask,
|
|
32
32
|
} from "./display-task.js";
|
|
33
33
|
import { ESCALATION_EXIT_CODE, readAndConsumeEscalation } from "./escalation.js";
|
|
34
|
+
import { classifySubagentFailure, type SubagentErrorClass } from "./failure-telemetry.js";
|
|
34
35
|
import {
|
|
35
36
|
formatContextUsageBar,
|
|
36
37
|
formatUsageStats,
|
|
@@ -97,6 +98,7 @@ type SessionDetailSummary = {
|
|
|
97
98
|
type ResultFailureDiagnosis = {
|
|
98
99
|
failed: boolean;
|
|
99
100
|
reason?: string;
|
|
101
|
+
errorClass?: SubagentErrorClass;
|
|
100
102
|
/** Set when the failure was caused by exceeding the model context window. */
|
|
101
103
|
contextOverflow?: boolean;
|
|
102
104
|
};
|
|
@@ -283,12 +285,26 @@ function parseSessionDetailSummary(sessionFile?: string): SessionDetailSummary {
|
|
|
283
285
|
}
|
|
284
286
|
|
|
285
287
|
export function diagnoseResultFailure(result: SingleResult): ResultFailureDiagnosis {
|
|
288
|
+
const finalOutput = getFinalOutput(result.messages).trim();
|
|
289
|
+
const failedByStatus = result.exitCode !== 0 || result.stopReason === "error" || result.stopReason === "aborted";
|
|
290
|
+
const errorClass =
|
|
291
|
+
result.errorClass ??
|
|
292
|
+
classifySubagentFailure({
|
|
293
|
+
failed: failedByStatus,
|
|
294
|
+
stopReason: result.stopReason,
|
|
295
|
+
exitCode: result.exitCode,
|
|
296
|
+
errorMessage: result.errorMessage,
|
|
297
|
+
stderr: result.stderr,
|
|
298
|
+
output: finalOutput,
|
|
299
|
+
});
|
|
300
|
+
|
|
286
301
|
if (result.exitCode !== 0 || result.stopReason === "error") {
|
|
287
|
-
const overflowText = result.errorMessage || result.stderr ||
|
|
302
|
+
const overflowText = result.errorMessage || result.stderr || finalOutput;
|
|
288
303
|
if (isContextOverflowText(overflowText)) {
|
|
289
304
|
const turns = result.usage?.turns ?? 0;
|
|
290
305
|
return {
|
|
291
306
|
failed: true,
|
|
307
|
+
errorClass: "context_overflow",
|
|
292
308
|
contextOverflow: true,
|
|
293
309
|
reason:
|
|
294
310
|
`Subagent stopped after exceeding the model context window (${turns} turn(s) completed). ` +
|
|
@@ -296,12 +312,13 @@ export function diagnoseResultFailure(result: SingleResult): ResultFailureDiagno
|
|
|
296
312
|
};
|
|
297
313
|
}
|
|
298
314
|
}
|
|
299
|
-
if (result.exitCode !== 0)
|
|
315
|
+
if (result.exitCode !== 0)
|
|
316
|
+
return { failed: true, errorClass, reason: `Subagent process exited with code ${result.exitCode}.` };
|
|
300
317
|
if (result.stopReason === "error")
|
|
301
|
-
return { failed: true, reason: result.errorMessage || "Subagent reported stopReason=error." };
|
|
302
|
-
if (result.stopReason === "aborted")
|
|
318
|
+
return { failed: true, errorClass, reason: result.errorMessage || "Subagent reported stopReason=error." };
|
|
319
|
+
if (result.stopReason === "aborted")
|
|
320
|
+
return { failed: true, errorClass: errorClass ?? "aborted", reason: "Subagent execution was aborted." };
|
|
303
321
|
|
|
304
|
-
const finalOutput = getFinalOutput(result.messages).trim();
|
|
305
322
|
const hasAssistantText = finalOutput.length > 0;
|
|
306
323
|
if (hasAssistantText) return { failed: false };
|
|
307
324
|
|
|
@@ -309,6 +326,7 @@ export function diagnoseResultFailure(result: SingleResult): ResultFailureDiagno
|
|
|
309
326
|
if (result.messages.length === 0) {
|
|
310
327
|
return {
|
|
311
328
|
failed: true,
|
|
329
|
+
errorClass: errorClass ?? "unknown",
|
|
312
330
|
reason:
|
|
313
331
|
"Subagent returned no messages (turn=0). " +
|
|
314
332
|
(stderr ? `stderr: ${stderr}` : "No stderr captured. Child process may have exited before producing output."),
|
|
@@ -317,6 +335,7 @@ export function diagnoseResultFailure(result: SingleResult): ResultFailureDiagno
|
|
|
317
335
|
|
|
318
336
|
return {
|
|
319
337
|
failed: true,
|
|
338
|
+
errorClass: errorClass ?? "unknown",
|
|
320
339
|
reason: `Subagent finished without assistant text output. ${stderr ? `stderr: ${stderr}` : "No stderr captured."}`,
|
|
321
340
|
};
|
|
322
341
|
}
|
|
@@ -471,6 +490,10 @@ function buildRunCompletionMessage(finalized: FinalizedRun, options?: { display?
|
|
|
471
490
|
usage: result?.usage,
|
|
472
491
|
model: result?.model,
|
|
473
492
|
source: result?.agentSource,
|
|
493
|
+
errorClass: runState.errorClass,
|
|
494
|
+
peakContextTokens: runState.peakContextTokens,
|
|
495
|
+
lastToolName: runState.lastToolName,
|
|
496
|
+
lastToolOutputChars: runState.lastToolOutputChars,
|
|
474
497
|
thoughtText: runState.thoughtText,
|
|
475
498
|
status: runState.status,
|
|
476
499
|
batchId: runState.batchId,
|
|
@@ -506,6 +529,10 @@ function buildEscalationMessage(runState: CommandRunState, escalationMessage: st
|
|
|
506
529
|
exitCode: result.exitCode,
|
|
507
530
|
usage: result.usage,
|
|
508
531
|
model: result.model,
|
|
532
|
+
errorClass: runState.errorClass,
|
|
533
|
+
peakContextTokens: runState.peakContextTokens,
|
|
534
|
+
lastToolName: runState.lastToolName,
|
|
535
|
+
lastToolOutputChars: runState.lastToolOutputChars,
|
|
509
536
|
batchId: runState.batchId,
|
|
510
537
|
pipelineId: runState.pipelineId,
|
|
511
538
|
pipelineStepIndex: runState.pipelineStepIndex,
|
|
@@ -549,6 +576,7 @@ function finalizeRunState(runState: CommandRunState, result: SingleResult): Fina
|
|
|
549
576
|
const failure = diagnoseResultFailure(result);
|
|
550
577
|
const isError = failure.failed;
|
|
551
578
|
runState.status = isError ? "error" : "done";
|
|
579
|
+
runState.errorClass = failure.errorClass;
|
|
552
580
|
runState.elapsedMs = Date.now() - runState.startedAt;
|
|
553
581
|
let rawOutput: string;
|
|
554
582
|
if (isError && failure.contextOverflow) {
|
|
@@ -653,7 +681,19 @@ function toLaunchSummary(
|
|
|
653
681
|
function buildRunAnalyticsSummary(
|
|
654
682
|
runState: Pick<
|
|
655
683
|
CommandRunState,
|
|
656
|
-
|
|
684
|
+
| "id"
|
|
685
|
+
| "agent"
|
|
686
|
+
| "status"
|
|
687
|
+
| "elapsedMs"
|
|
688
|
+
| "model"
|
|
689
|
+
| "batchId"
|
|
690
|
+
| "pipelineId"
|
|
691
|
+
| "pipelineStepIndex"
|
|
692
|
+
| "runtime"
|
|
693
|
+
| "errorClass"
|
|
694
|
+
| "peakContextTokens"
|
|
695
|
+
| "lastToolName"
|
|
696
|
+
| "lastToolOutputChars"
|
|
657
697
|
>,
|
|
658
698
|
): Record<string, unknown> {
|
|
659
699
|
return {
|
|
@@ -662,6 +702,10 @@ function buildRunAnalyticsSummary(
|
|
|
662
702
|
status: runState.status,
|
|
663
703
|
elapsedMs: runState.elapsedMs,
|
|
664
704
|
model: runState.model,
|
|
705
|
+
errorClass: runState.errorClass,
|
|
706
|
+
peakContextTokens: runState.peakContextTokens,
|
|
707
|
+
lastToolName: runState.lastToolName,
|
|
708
|
+
lastToolOutputChars: runState.lastToolOutputChars,
|
|
665
709
|
batchId: runState.batchId,
|
|
666
710
|
pipelineId: runState.pipelineId,
|
|
667
711
|
stepIndex: runState.pipelineStepIndex,
|
|
@@ -683,6 +727,7 @@ function isInteractiveTuiContext(ctx: SubagentToolExecuteContext): boolean {
|
|
|
683
727
|
|
|
684
728
|
function finalizeRunError(runState: CommandRunState, error: unknown): FinalizedRun {
|
|
685
729
|
runState.status = "error";
|
|
730
|
+
runState.errorClass = "process_error";
|
|
686
731
|
runState.elapsedMs = Date.now() - runState.startedAt;
|
|
687
732
|
runState.lastLine =
|
|
688
733
|
runState.autoAbortReason ?? (error instanceof Error ? error.message : "Subagent execution failed");
|
|
@@ -729,8 +774,13 @@ export function createSubagentToolExecute(pi: ExtensionAPI, store: SubagentStore
|
|
|
729
774
|
}
|
|
730
775
|
|
|
731
776
|
if (parsedCommand.type === "error") {
|
|
777
|
+
const errorText =
|
|
778
|
+
parsedCommand.showHelp === false
|
|
779
|
+
? parsedCommand.message
|
|
780
|
+
: `${parsedCommand.message}\n\n${SUBAGENT_CLI_HELP_TEXT}`;
|
|
781
|
+
|
|
732
782
|
return {
|
|
733
|
-
content: [{ type: "text", text:
|
|
783
|
+
content: [{ type: "text", text: errorText }],
|
|
734
784
|
details: createEmptyDetails("single", false, null),
|
|
735
785
|
isError: true,
|
|
736
786
|
};
|
package/types.ts
CHANGED
|
@@ -6,6 +6,7 @@ import type { AgentToolResult } from "@earendil-works/pi-agent-core";
|
|
|
6
6
|
import type { Message } from "@earendil-works/pi-ai";
|
|
7
7
|
import { Type } from "typebox";
|
|
8
8
|
import type { AgentConfig, AgentRuntime } from "./agents.js";
|
|
9
|
+
import type { SubagentErrorClass } from "./failure-telemetry.js";
|
|
9
10
|
|
|
10
11
|
// ─── Interfaces ──────────────────────────────────────────────────────────────
|
|
11
12
|
|
|
@@ -30,6 +31,10 @@ export interface SingleResult {
|
|
|
30
31
|
model?: string;
|
|
31
32
|
stopReason?: string;
|
|
32
33
|
errorMessage?: string;
|
|
34
|
+
errorClass?: SubagentErrorClass;
|
|
35
|
+
peakContextTokens?: number;
|
|
36
|
+
lastToolName?: string;
|
|
37
|
+
lastToolOutputChars?: number;
|
|
33
38
|
step?: number;
|
|
34
39
|
liveText?: string;
|
|
35
40
|
liveThinking?: string;
|
|
@@ -91,6 +96,14 @@ export interface CommandRunState {
|
|
|
91
96
|
retryCount?: number;
|
|
92
97
|
/** Last transient failure reason that triggered an auto-retry. */
|
|
93
98
|
lastRetryReason?: string;
|
|
99
|
+
/** Normalized terminal failure category for analytics. */
|
|
100
|
+
errorClass?: SubagentErrorClass;
|
|
101
|
+
/** Highest reported context usage observed during the run. */
|
|
102
|
+
peakContextTokens?: number;
|
|
103
|
+
/** Most recently completed/started tool name. */
|
|
104
|
+
lastToolName?: string;
|
|
105
|
+
/** Text character count of the most recent tool result. */
|
|
106
|
+
lastToolOutputChars?: number;
|
|
94
107
|
/** Hang detector reason preserved until the normal finalizer emits the sole completion. */
|
|
95
108
|
autoAbortReason?: string;
|
|
96
109
|
runtime?: AgentRuntime;
|
package/widget.ts
CHANGED
|
@@ -119,9 +119,10 @@ function getContextShort(run: CommandRunState, ctx: WidgetRenderCtx, theme: Widg
|
|
|
119
119
|
if (usedContextPercent === undefined) return "";
|
|
120
120
|
const contextBar = formatCompactContextBar(usedContextPercent);
|
|
121
121
|
if (!contextBar) return "";
|
|
122
|
+
const contextLabel = `${contextBar} ${usedContextPercent}%`;
|
|
122
123
|
const contextBarColor =
|
|
123
124
|
remainingContextPercent !== undefined ? getContextBarColorByRemaining(remainingContextPercent) : undefined;
|
|
124
|
-
return contextBarColor ? theme.fg(contextBarColor,
|
|
125
|
+
return contextBarColor ? theme.fg(contextBarColor, contextLabel) : theme.fg("dim", contextLabel);
|
|
125
126
|
}
|
|
126
127
|
|
|
127
128
|
function buildPrimaryLabelText(run: CommandRunState): string {
|