pi-subagents 0.47.1 → 0.49.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +51 -0
- package/README.md +2 -0
- package/docs/agents.md +1 -0
- package/docs/configuration.md +74 -7
- package/docs/missions.md +4 -2
- package/docs/observability.md +27 -3
- package/docs/tool-reference.md +1 -1
- package/package.json +1 -1
- package/src/agents/agents.ts +36 -15
- package/src/api/preflight.ts +1 -1
- package/src/extension/config.ts +26 -0
- package/src/extension/doctor.ts +40 -0
- package/src/extension/index.ts +39 -8
- package/src/extension/public-execution.ts +35 -4
- package/src/extension/rpc.ts +3 -9
- package/src/extension/schemas.ts +10 -9
- package/src/extension/tool-description.ts +10 -8
- package/src/inspectors/herdr/actions.ts +11 -2
- package/src/inspectors/herdr/inspector-runner.ts +16 -3
- package/src/intercom/intercom-bridge.ts +7 -2
- package/src/missions/lifecycle.ts +4 -7
- package/src/missions/store.ts +12 -7
- package/src/missions/workflow-state.ts +6 -2
- package/src/runs/background/active-async-capacity.ts +431 -0
- package/src/runs/background/active-run-index.ts +9 -5
- package/src/runs/background/async-execution.ts +141 -65
- package/src/runs/background/async-job-tracker.ts +4 -0
- package/src/runs/background/async-resume.ts +18 -2
- package/src/runs/background/async-status.ts +11 -5
- package/src/runs/background/chain-append.ts +33 -15
- package/src/runs/background/fleet-view.ts +18 -5
- package/src/runs/background/owned-process-tree.ts +104 -0
- package/src/runs/background/process-terminal.ts +17 -3
- package/src/runs/background/resume-guidance.ts +27 -7
- package/src/runs/background/retained-children.ts +14 -6
- package/src/runs/background/run-status.ts +101 -5
- package/src/runs/background/stale-run-reconciler.ts +3 -3
- package/src/runs/background/subagent-runner.ts +60 -32
- package/src/runs/foreground/chain-execution.ts +37 -2
- package/src/runs/foreground/execution.ts +90 -19
- package/src/runs/foreground/foreground-control.ts +12 -0
- package/src/runs/foreground/prompt-audit.ts +172 -0
- package/src/runs/foreground/subagent-executor.ts +829 -203
- package/src/runs/shared/acceptance.ts +13 -4
- package/src/runs/shared/completion-guard.ts +17 -1
- package/src/runs/shared/llm-intent-arbiter.ts +302 -0
- package/src/runs/shared/parallel-utils.ts +2 -0
- package/src/runs/shared/pi-args.ts +44 -1
- package/src/runs/shared/run-fanout-budget.ts +280 -0
- package/src/runs/shared/single-output.ts +4 -2
- package/src/runs/shared/task-intent.ts +19 -3
- package/src/runs/shared/worktree.ts +17 -5
- package/src/shared/agent-stream-options.ts +5 -0
- package/src/shared/artifacts.ts +2 -6
- package/src/shared/display-text.ts +50 -0
- package/src/shared/node-executable.ts +21 -0
- package/src/shared/types.ts +115 -4
- package/src/shared/utils.ts +3 -1
- package/src/tui/fleet-status.ts +7 -5
- package/src/tui/fleet-transcript.ts +1 -48
- package/src/tui/fleet.ts +228 -13
- package/src/tui/render.ts +86 -39
- package/src/watchdog/permission-arbiter.ts +2 -1
- package/src/watchdog/review.ts +4 -3
- package/src/workflows/chat-progress.ts +8 -2
- package/src/workflows/scripted-workflow.ts +66 -10
|
@@ -23,7 +23,7 @@ import type {
|
|
|
23
23
|
SubagentRunMode,
|
|
24
24
|
} from "../../shared/types.ts";
|
|
25
25
|
import { isAgentContractV1 } from "./agent-contract.ts";
|
|
26
|
-
import { classifyTaskMutationIntent, taskMayMutate } from "./task-intent.ts";
|
|
26
|
+
import { classifyTaskMutationIntent, stripSeverityCompounds, taskMayMutate } from "./task-intent.ts";
|
|
27
27
|
|
|
28
28
|
const LEVEL_RANK: Record<Exclude<AcceptanceLevel, "auto">, number> = {
|
|
29
29
|
none: 0,
|
|
@@ -93,7 +93,7 @@ function inferLevel(input: {
|
|
|
93
93
|
const rolePatchTask = input.acceptanceRole !== undefined
|
|
94
94
|
&& intent.kind !== "read-only"
|
|
95
95
|
&& !/\b(?:do not|don't|must not)\s+patch\b/.test(task)
|
|
96
|
-
&& /\bpatch\s+(?:(?:\.{0,2}[\\/])?(?:[\w.-]+[\\/])+[\w.-]+|[\w.-]+\.[a-z0-9]+\b|(?:the\s+)?parser\b)/.test(task);
|
|
96
|
+
&& /\bpatch\s+(?:(?:\.{0,2}[\\/])?(?:[\w.-]+[\\/])+[\w.-]+|[\w.-]+\.[a-z0-9]+\b|(?:the\s+)?parser\b)/.test(stripSeverityCompounds(task));
|
|
97
97
|
const taskMayWrite = readOnlyTask ? false : taskMayMutate(input.task ?? "") || intent.kind === "implementation" || rolePatchTask;
|
|
98
98
|
const readOnlyAgent = input.acceptanceRole === "read-only"
|
|
99
99
|
|| (input.acceptanceRole === undefined && /\b(?:reviewer|oracle|scout|researcher|analyst)\b/.test(agent));
|
|
@@ -433,6 +433,7 @@ export function formatAcceptancePrompt(acceptance: ResolvedAcceptanceConfig, opt
|
|
|
433
433
|
"",
|
|
434
434
|
"Finish with a fenced JSON block tagged `acceptance-report` in this shape:",
|
|
435
435
|
"Use empty arrays when no items apply; array fields contain strings unless object entries are shown.",
|
|
436
|
+
"Empty-string entries (`[\"\"]`) are ignored; use `[]` when nothing applies.",
|
|
436
437
|
"`criteriaSatisfied[].status` must be exactly one of: satisfied, not-satisfied, not-applicable.",
|
|
437
438
|
"`commandsRun[].result` must be exactly one of: passed, failed, not-run.",
|
|
438
439
|
"`manualNotes` and `notes` are optional strings; an empty string means no note and does not satisfy `manual-notes` evidence.",
|
|
@@ -612,9 +613,17 @@ function normalizeAcceptanceReportValue(value: unknown, pathLabel = ""): { value
|
|
|
612
613
|
case "testsAddedOrUpdated":
|
|
613
614
|
case "validationOutput":
|
|
614
615
|
case "residualRisks":
|
|
615
|
-
case "reviewFindings":
|
|
616
|
-
|
|
616
|
+
case "reviewFindings": {
|
|
617
|
+
// Tolerate empty-string entries ("[\"\"]"): models write them for "no items"
|
|
618
|
+
// and a single empty entry must not reject the whole report. Drop them at
|
|
619
|
+
// parse time; non-string entries are kept so validateStringArrayField still
|
|
620
|
+
// flags structural garbage.
|
|
621
|
+
const items = typeof fieldValue === "string" ? [fieldValue] : fieldValue;
|
|
622
|
+
normalized[canonical] = Array.isArray(items)
|
|
623
|
+
? items.filter((item) => typeof item !== "string" || item.trim().length > 0)
|
|
624
|
+
: items;
|
|
617
625
|
break;
|
|
626
|
+
}
|
|
618
627
|
case "noStagedFiles": {
|
|
619
628
|
const token = typeof fieldValue === "string" ? fieldValue.trim().toLowerCase() : undefined;
|
|
620
629
|
normalized[canonical] = token === "true" ? true : token === "false" ? false : fieldValue;
|
|
@@ -22,6 +22,10 @@ const READ_ONLY_BUILTIN_TOOLS = new Set([
|
|
|
22
22
|
const CURSOR_FILE_MUTATION_THINKING =
|
|
23
23
|
/(?:^|\n)\s*Cursor (?:edit|write)\s*:/i;
|
|
24
24
|
|
|
25
|
+
const IMPLEMENTATION_CHALLENGE_TASK_PATTERN = /^You are reviving a previous subagent conversation\.\n\nOriginal run: .+\nOriginal agent: .+(?:\nOriginal session file: .+)?\n\nUse the stored session context as background\. Answer the orchestrator's follow-up below\. Do not assume the original child process is still alive\.\n\nFollow-up:\nRun implementation challenge pass (?:one|two|\d+) and implement any better current-scope change\.$/;
|
|
26
|
+
const NO_BETTER_CHANGE_NEEDED_PATTERN = /^\s*no (?:better|further|additional) (?:current[- ]scope )?(?:code |source |file )?(?:change|changes|edit|edits|patch|patches) (?:is|are) needed[.!]?\s*$/i;
|
|
27
|
+
const NO_BETTER_CHANGE_QUALIFIER_PATTERN = /\b(?:do\s+not|don't|dont|not|never|cannot|can't|cant|unable|uncertain|unsure|unclear|maybe|might|may|\w+n['’]t)\b/i;
|
|
28
|
+
|
|
25
29
|
interface CompletionMutationGuardInput {
|
|
26
30
|
agent: string;
|
|
27
31
|
task: string;
|
|
@@ -82,14 +86,26 @@ export function hasMutationToolCall(messages: Message[]): boolean {
|
|
|
82
86
|
return false;
|
|
83
87
|
}
|
|
84
88
|
|
|
89
|
+
function reportsNoBetterChallengeChange(messages: Message[]): boolean {
|
|
90
|
+
const report = messages
|
|
91
|
+
.filter((message) => message.role === "assistant")
|
|
92
|
+
.flatMap((message) => message.content)
|
|
93
|
+
.flatMap((part) => part.type === "text" ? [part.text] : [])
|
|
94
|
+
.join("\n");
|
|
95
|
+
return NO_BETTER_CHANGE_NEEDED_PATTERN.test(report)
|
|
96
|
+
&& !NO_BETTER_CHANGE_QUALIFIER_PATTERN.test(report);
|
|
97
|
+
}
|
|
98
|
+
|
|
85
99
|
export function evaluateCompletionMutationGuard(input: CompletionMutationGuardInput): CompletionMutationGuardResult {
|
|
86
100
|
const expectedMutation = hasMutationToolCapability(input.tools, input.mcpDirectTools)
|
|
87
101
|
? expectsImplementationMutation(input.agent, input.task)
|
|
88
102
|
: false;
|
|
89
103
|
const attemptedMutation = hasMutationToolCall(input.messages);
|
|
104
|
+
const noEditChallengeComplete = IMPLEMENTATION_CHALLENGE_TASK_PATTERN.test(input.task)
|
|
105
|
+
&& reportsNoBetterChallengeChange(input.messages);
|
|
90
106
|
return {
|
|
91
107
|
expectedMutation,
|
|
92
108
|
attemptedMutation,
|
|
93
|
-
triggered: expectedMutation && !attemptedMutation,
|
|
109
|
+
triggered: expectedMutation && !attemptedMutation && !noEditChallengeComplete,
|
|
94
110
|
};
|
|
95
111
|
}
|
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { Agent, type AgentTool, type StreamFn } from "@earendil-works/pi-agent-core";
|
|
3
|
+
import { convertToLlm, type ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
4
|
+
import { streamSimple } from "@earendil-works/pi-ai/compat";
|
|
5
|
+
import type { ProviderHeaders } from "@earendil-works/pi-ai";
|
|
6
|
+
import { Type, type Static } from "typebox";
|
|
7
|
+
import { agentStreamOptions } from "../../shared/agent-stream-options.ts";
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* LLM intent arbiter for the completion mutation guard.
|
|
11
|
+
*
|
|
12
|
+
* The regex classifier (task-intent.ts) is deliberately narrow, so exotic
|
|
13
|
+
* review wording can still look like an implementation task ("to fix this,
|
|
14
|
+
* compare the outputs", verbs inside URLs or quoted text). When the guard is
|
|
15
|
+
* about to hard-fail a run that made no edits, this arbiter asks a model
|
|
16
|
+
* whether the task actually instructed file changes. It can only downgrade a
|
|
17
|
+
* failure to a pass; every error, timeout, or non-read-only verdict keeps the
|
|
18
|
+
* guard's original behavior.
|
|
19
|
+
*
|
|
20
|
+
* Enabled by default; set PI_SUBAGENTS_LLM_INTENT_ARBITER=0 to disable.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
const COMPLETION_GUARD_ERROR_PREFIX =
|
|
24
|
+
"Subagent completed without making edits for an implementation task.";
|
|
25
|
+
|
|
26
|
+
export type TaskMutationVerdict = "read-only" | "implementation" | "unavailable";
|
|
27
|
+
|
|
28
|
+
export type TaskMutationArbiter = (task: string) => Promise<TaskMutationVerdict>;
|
|
29
|
+
|
|
30
|
+
const DecisionParams = Type.Object(
|
|
31
|
+
{
|
|
32
|
+
classification: Type.String({ enum: ["read_only", "implementation"] }),
|
|
33
|
+
confidence: Type.String({
|
|
34
|
+
enum: ["low", "medium", "high"],
|
|
35
|
+
description: "Confidence in the classification. Only read_only with high confidence rescues a failed run.",
|
|
36
|
+
}),
|
|
37
|
+
reason: Type.String({ description: "One concise reason for this classification." }),
|
|
38
|
+
},
|
|
39
|
+
{ additionalProperties: false },
|
|
40
|
+
);
|
|
41
|
+
|
|
42
|
+
type DecisionParams = Static<typeof DecisionParams>;
|
|
43
|
+
|
|
44
|
+
/** Map a model decision to a verdict. Only a high-confidence read_only rescues. */
|
|
45
|
+
export function mapArbiterDecision(
|
|
46
|
+
decision: { classification?: string; confidence?: string } | undefined,
|
|
47
|
+
): TaskMutationVerdict {
|
|
48
|
+
if (!decision) return "unavailable";
|
|
49
|
+
if (decision.classification === "read_only" && decision.confidence === "high") return "read-only";
|
|
50
|
+
return "implementation";
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
interface ArbiterRuntime {
|
|
54
|
+
model: NonNullable<RegistryModel>;
|
|
55
|
+
baseStreamFn: StreamFn;
|
|
56
|
+
timeoutMs: number;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
interface ArbiterAuth {
|
|
60
|
+
apiKey?: string;
|
|
61
|
+
headers?: ProviderHeaders;
|
|
62
|
+
env?: Record<string, string>;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
export interface TaskMutationArbiterOptions {
|
|
66
|
+
/** Explicit "provider/id" model override (tests, config). */
|
|
67
|
+
model?: string;
|
|
68
|
+
/** Injectable stream function (tests). */
|
|
69
|
+
streamFn?: StreamFn;
|
|
70
|
+
timeoutMs?: number;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const DEFAULT_ARBITER_TIMEOUT_MS = 10_000;
|
|
74
|
+
|
|
75
|
+
type RegistryModel = ReturnType<NonNullable<ExtensionContext["modelRegistry"]["find"]>>;
|
|
76
|
+
|
|
77
|
+
function resolveArbiterModel(
|
|
78
|
+
ctx: ExtensionContext,
|
|
79
|
+
options?: TaskMutationArbiterOptions,
|
|
80
|
+
): NonNullable<RegistryModel> | null {
|
|
81
|
+
const registry = ctx.modelRegistry as {
|
|
82
|
+
find?: (provider: string, modelId: string) => RegistryModel | undefined;
|
|
83
|
+
getAvailable?: () => Array<{ provider?: string; id?: string }>;
|
|
84
|
+
};
|
|
85
|
+
const explicit = options?.model?.trim();
|
|
86
|
+
if (explicit) {
|
|
87
|
+
const [provider, id] = explicit.split("/");
|
|
88
|
+
if (provider && id) return registry.find?.(provider, id) ?? null;
|
|
89
|
+
return null;
|
|
90
|
+
}
|
|
91
|
+
if (ctx.model) return ctx.model as NonNullable<RegistryModel>;
|
|
92
|
+
const first = registry.getAvailable?.()[0];
|
|
93
|
+
if (first?.provider && first.id) return registry.find?.(first.provider, first.id) ?? null;
|
|
94
|
+
return null;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function resolveArbiterRuntime(
|
|
98
|
+
ctx: ExtensionContext,
|
|
99
|
+
options?: TaskMutationArbiterOptions,
|
|
100
|
+
): ArbiterRuntime | null {
|
|
101
|
+
const model = resolveArbiterModel(ctx, options);
|
|
102
|
+
if (!model?.provider || !model.id) return null;
|
|
103
|
+
const registry = ctx.modelRegistry as {
|
|
104
|
+
getRegisteredProviderConfig?: (provider: string) => { api?: string; streamSimple?: StreamFn } | undefined;
|
|
105
|
+
};
|
|
106
|
+
const modelApi = (model as { api?: string }).api;
|
|
107
|
+
const registered = registry.getRegisteredProviderConfig?.(model.provider);
|
|
108
|
+
const baseStreamFn = options?.streamFn
|
|
109
|
+
?? (registered?.streamSimple && registered.api === modelApi
|
|
110
|
+
? registered.streamSimple
|
|
111
|
+
: streamSimple);
|
|
112
|
+
return {
|
|
113
|
+
model,
|
|
114
|
+
baseStreamFn,
|
|
115
|
+
timeoutMs: options?.timeoutMs ?? DEFAULT_ARBITER_TIMEOUT_MS,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
async function resolveArbiterAuth(
|
|
120
|
+
ctx: ExtensionContext,
|
|
121
|
+
model: RegistryModel,
|
|
122
|
+
): Promise<ArbiterAuth> {
|
|
123
|
+
const registry = ctx.modelRegistry as {
|
|
124
|
+
getApiKeyAndHeaders?: (m: RegistryModel) => Promise<{
|
|
125
|
+
ok: boolean;
|
|
126
|
+
apiKey?: string;
|
|
127
|
+
headers?: ProviderHeaders;
|
|
128
|
+
env?: Record<string, string>;
|
|
129
|
+
error?: string;
|
|
130
|
+
}>;
|
|
131
|
+
};
|
|
132
|
+
// Call as a METHOD on the registry: the host ModelRegistry implementation
|
|
133
|
+
// is a class whose method reads instance state (this.runtime), so a
|
|
134
|
+
// detached call silently fails auth. Same shape as the watchdog.
|
|
135
|
+
if (!registry.getApiKeyAndHeaders) return {};
|
|
136
|
+
try {
|
|
137
|
+
const auth = await registry.getApiKeyAndHeaders(model);
|
|
138
|
+
if (auth.ok === false) return {};
|
|
139
|
+
return {
|
|
140
|
+
...(auth.apiKey ? { apiKey: auth.apiKey } : {}),
|
|
141
|
+
...(auth.headers ? { headers: auth.headers } : {}),
|
|
142
|
+
...(auth.env ? { env: auth.env } : {}),
|
|
143
|
+
};
|
|
144
|
+
} catch {
|
|
145
|
+
return {};
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/** Build the effective stream function with resolved credentials wrapped in (watchdog pattern). */
|
|
150
|
+
function authWrappedStreamFn(
|
|
151
|
+
base: StreamFn,
|
|
152
|
+
auth: ArbiterAuth,
|
|
153
|
+
): StreamFn {
|
|
154
|
+
return (model, context, streamOptions) => base(model, context, {
|
|
155
|
+
...(streamOptions ?? {}),
|
|
156
|
+
...(auth.apiKey ? { apiKey: auth.apiKey } : {}),
|
|
157
|
+
...(auth.env || streamOptions?.env ? { env: { ...(auth.env ?? {}), ...(streamOptions?.env ?? {}) } } : {}),
|
|
158
|
+
headers: { ...(streamOptions?.headers ?? {}), ...(auth.headers ?? {}) },
|
|
159
|
+
});
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
async function runArbitration(
|
|
163
|
+
runtime: ArbiterRuntime,
|
|
164
|
+
auth: ArbiterAuth,
|
|
165
|
+
task: string,
|
|
166
|
+
): Promise<TaskMutationVerdict> {
|
|
167
|
+
let decision: DecisionParams | undefined;
|
|
168
|
+
const tool: AgentTool<typeof DecisionParams, { recorded: boolean }> = {
|
|
169
|
+
name: "task_mutation_decision",
|
|
170
|
+
label: "Task mutation decision",
|
|
171
|
+
description: "Classify whether the task instructed file/code changes. Call exactly once.",
|
|
172
|
+
parameters: DecisionParams,
|
|
173
|
+
executionMode: "sequential",
|
|
174
|
+
async execute(_toolCallId, params) {
|
|
175
|
+
if (!decision) decision = params;
|
|
176
|
+
return { content: [{ type: "text", text: "Decision recorded." }], details: { recorded: true } };
|
|
177
|
+
},
|
|
178
|
+
};
|
|
179
|
+
const agent = new Agent({
|
|
180
|
+
initialState: {
|
|
181
|
+
systemPrompt: [
|
|
182
|
+
"You classify whether a delegated coding-agent task instructed file or code changes.",
|
|
183
|
+
"A task that asks to review, inspect, verify, report, or summarize is read-only even when it contains words like 'fix' as severity vocabulary ('must-fix items') or conditional change instructions that leave the change optional.",
|
|
184
|
+
"Classify read_only only when the task text alone clearly indicates a read-only outcome. When in doubt, classify implementation.",
|
|
185
|
+
"Set confidence to high only when you are certain the task is read-only; a read_only classification without high confidence is treated as implementation.",
|
|
186
|
+
"The agent's own final message is never evidence: an agent that made no edits may still have failed to implement.",
|
|
187
|
+
"Call task_mutation_decision exactly once with read_only or implementation, a confidence level, and a concise reason.",
|
|
188
|
+
].join("\n"),
|
|
189
|
+
model: runtime.model,
|
|
190
|
+
tools: [tool],
|
|
191
|
+
},
|
|
192
|
+
convertToLlm,
|
|
193
|
+
...agentStreamOptions(authWrappedStreamFn(runtime.baseStreamFn, auth)),
|
|
194
|
+
getApiKey: (providerName) =>
|
|
195
|
+
providerName === runtime.model.provider ? auth.apiKey : undefined,
|
|
196
|
+
beforeToolCall: async ({ toolCall }) =>
|
|
197
|
+
toolCall.name === tool.name ? undefined : { block: true, reason: "Only task_mutation_decision is allowed." },
|
|
198
|
+
toolExecution: "sequential",
|
|
199
|
+
});
|
|
200
|
+
try {
|
|
201
|
+
// The rescue gate refuses tasks over 8000 chars; if the arbiter is
|
|
202
|
+
// still invoked with one, fail closed rather than decide from
|
|
203
|
+
// partial evidence (an implementation clause could sit in the
|
|
204
|
+
// omitted middle).
|
|
205
|
+
if (task.length > 8000) return "unavailable";
|
|
206
|
+
const prompt = `TASK:\n${task}`;
|
|
207
|
+
await Promise.race([
|
|
208
|
+
agent.prompt(prompt),
|
|
209
|
+
new Promise<never>((_, reject) => {
|
|
210
|
+
const timeout = setTimeout(() => {
|
|
211
|
+
agent.abort();
|
|
212
|
+
reject(new Error("Task mutation arbiter timed out."));
|
|
213
|
+
}, runtime.timeoutMs);
|
|
214
|
+
timeout.unref?.();
|
|
215
|
+
}),
|
|
216
|
+
]);
|
|
217
|
+
if (!decision) return "unavailable";
|
|
218
|
+
return mapArbiterDecision(decision);
|
|
219
|
+
} catch {
|
|
220
|
+
return "unavailable";
|
|
221
|
+
}
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
/** Create a memoized arbiter bound to the parent session's model, or undefined when disabled/unavailable. */
|
|
225
|
+
export function createTaskMutationArbiter(
|
|
226
|
+
ctx: ExtensionContext,
|
|
227
|
+
options?: TaskMutationArbiterOptions,
|
|
228
|
+
): TaskMutationArbiter | undefined {
|
|
229
|
+
if (process.env.PI_SUBAGENTS_LLM_INTENT_ARBITER === "0") return undefined;
|
|
230
|
+
const runtime = resolveArbiterRuntime(ctx, options);
|
|
231
|
+
if (!runtime) return undefined;
|
|
232
|
+
const cache = new Map<string, TaskMutationVerdict>();
|
|
233
|
+
return async (task) => {
|
|
234
|
+
const key = createHash("sha256").update(task).digest("base64url");
|
|
235
|
+
const cached = cache.get(key);
|
|
236
|
+
if (cached) return cached;
|
|
237
|
+
const auth = await resolveArbiterAuth(ctx, runtime.model);
|
|
238
|
+
const verdict = await runArbitration(runtime, auth, task);
|
|
239
|
+
if (cache.size > 200) cache.clear();
|
|
240
|
+
cache.set(key, verdict);
|
|
241
|
+
return verdict;
|
|
242
|
+
};
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
export function isCompletionGuardFailure(result: { error?: string }): boolean {
|
|
246
|
+
return result.error?.startsWith(COMPLETION_GUARD_ERROR_PREFIX) === true;
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
export async function arbitrateCompletionGuardRescue(input: {
|
|
250
|
+
guardTriggered: boolean;
|
|
251
|
+
task: string;
|
|
252
|
+
arbiter?: TaskMutationArbiter;
|
|
253
|
+
}): Promise<{ triggered: boolean; rescued: boolean }> {
|
|
254
|
+
if (!input.guardTriggered || !input.arbiter) {
|
|
255
|
+
return { triggered: input.guardTriggered, rescued: false };
|
|
256
|
+
}
|
|
257
|
+
// Never decide from partial evidence: refuse tasks over 8000 chars before
|
|
258
|
+
// even consulting the model.
|
|
259
|
+
if (input.task.length > 8000) {
|
|
260
|
+
return { triggered: true, rescued: false };
|
|
261
|
+
}
|
|
262
|
+
try {
|
|
263
|
+
const verdict = await input.arbiter(input.task);
|
|
264
|
+
if (verdict === "read-only") return { triggered: false, rescued: true };
|
|
265
|
+
return { triggered: true, rescued: false };
|
|
266
|
+
} catch {
|
|
267
|
+
return { triggered: true, rescued: false };
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
|
|
271
|
+
/**
|
|
272
|
+
* When the completion guard hard-failed a run and the arbiter classifies the
|
|
273
|
+
* task as read-only, clear the failure. Returns true when rescued. All other
|
|
274
|
+
* verdicts and every failure mode keep the original guard behavior.
|
|
275
|
+
*
|
|
276
|
+
* Classification uses the task text only: the agent's own final message is
|
|
277
|
+
* never evidence, so a child that failed to implement cannot talk its way
|
|
278
|
+
* out of the guard by claiming the work was read-only.
|
|
279
|
+
*/
|
|
280
|
+
export async function maybeRescueCompletionGuardFailure(
|
|
281
|
+
result: { exitCode?: number; error?: string },
|
|
282
|
+
task: string,
|
|
283
|
+
arbiter: TaskMutationArbiter | undefined,
|
|
284
|
+
): Promise<boolean> {
|
|
285
|
+
if (!arbiter || !isCompletionGuardFailure(result)) return false;
|
|
286
|
+
const verdict = await arbitrateWithGuard(arbiter, task);
|
|
287
|
+
if (verdict !== "read-only") return false;
|
|
288
|
+
result.exitCode = 0;
|
|
289
|
+
delete result.error;
|
|
290
|
+
return true;
|
|
291
|
+
}
|
|
292
|
+
|
|
293
|
+
async function arbitrateWithGuard(
|
|
294
|
+
arbiter: TaskMutationArbiter,
|
|
295
|
+
task: string,
|
|
296
|
+
): Promise<TaskMutationVerdict> {
|
|
297
|
+
try {
|
|
298
|
+
return await arbiter(task);
|
|
299
|
+
} catch {
|
|
300
|
+
return "unavailable";
|
|
301
|
+
}
|
|
302
|
+
}
|
|
@@ -61,6 +61,8 @@ export interface RunnerSubagentStep {
|
|
|
61
61
|
toolBudget?: import("../../shared/types.ts").ResolvedToolBudget;
|
|
62
62
|
capabilityCeiling?: import("./capability-ceiling.ts").ResolvedSubagentCapabilityCeiling;
|
|
63
63
|
capabilityAudit?: import("./capability-ceiling.ts").SubagentCapabilityAudit;
|
|
64
|
+
/** Private stable logical-child path for inherited run fan-out accounting. */
|
|
65
|
+
runFanoutPath?: string;
|
|
64
66
|
}
|
|
65
67
|
|
|
66
68
|
export interface RunnerCheckpointStep {
|
|
@@ -23,8 +23,10 @@ import {
|
|
|
23
23
|
type JsonSchemaObject,
|
|
24
24
|
type LaunchResolvedChildExtensionsV1,
|
|
25
25
|
type ResolvedToolBudget,
|
|
26
|
+
type RunFanoutBudgetDescriptor,
|
|
26
27
|
} from "../../shared/types.ts";
|
|
27
28
|
import { THINKING_LEVELS } from "../../shared/model-info.ts";
|
|
29
|
+
import { encodeRunFanoutBudgetDescriptor, RUN_FANOUT_BUDGET_ENV } from "./run-fanout-budget.ts";
|
|
28
30
|
import {
|
|
29
31
|
TOOL_BUDGET_ENV,
|
|
30
32
|
TOOL_BUDGET_ZERO_AUTH_ENV,
|
|
@@ -63,6 +65,32 @@ import {
|
|
|
63
65
|
} from "./capability-ceiling.ts";
|
|
64
66
|
|
|
65
67
|
const TASK_ARG_LIMIT = 8000;
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Env override for how the task text reaches the child process. Endpoint
|
|
71
|
+
* protection (EDR) pre-execution command-line scanning may deny exec of
|
|
72
|
+
* children whose argv embeds a long natural-language task, which surfaces
|
|
73
|
+
* as an immediate zero-activity SIGKILL. File delivery keeps the task out
|
|
74
|
+
* of argv entirely.
|
|
75
|
+
*/
|
|
76
|
+
export const SUBAGENT_TASK_DELIVERY_ENV = "PI_SUBAGENT_TASK_DELIVERY";
|
|
77
|
+
|
|
78
|
+
export type SubagentTaskDelivery = "auto" | "file";
|
|
79
|
+
|
|
80
|
+
export function resolveSubagentTaskDelivery(
|
|
81
|
+
env: NodeJS.ProcessEnv = process.env,
|
|
82
|
+
): SubagentTaskDelivery {
|
|
83
|
+
return env[SUBAGENT_TASK_DELIVERY_ENV]?.trim().toLowerCase() === "file"
|
|
84
|
+
? "file"
|
|
85
|
+
: "auto";
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
function shouldDeliverTaskViaFile(
|
|
89
|
+
task: string,
|
|
90
|
+
delivery: SubagentTaskDelivery,
|
|
91
|
+
): boolean {
|
|
92
|
+
return delivery === "file" || task.length > TASK_ARG_LIMIT;
|
|
93
|
+
}
|
|
66
94
|
const MAX_LAUNCH_RESOLVED_EXTENSION_IDS = 32;
|
|
67
95
|
const PROMPT_RUNTIME_EXTENSION_PATH = path.join(
|
|
68
96
|
path.dirname(fileURLToPath(import.meta.url)),
|
|
@@ -136,6 +164,7 @@ export interface BuildPiArgsInput {
|
|
|
136
164
|
parentDepth?: number;
|
|
137
165
|
parentPath?: NestedPathEntry[];
|
|
138
166
|
parentCapabilityToken?: string;
|
|
167
|
+
runFanoutBudget?: RunFanoutBudgetDescriptor;
|
|
139
168
|
steerInboxDir?: string;
|
|
140
169
|
steerCapabilityPath?: string;
|
|
141
170
|
steerAckDir?: string;
|
|
@@ -149,6 +178,12 @@ export interface BuildPiArgsInput {
|
|
|
149
178
|
permissionRules?: PermissionRules;
|
|
150
179
|
permissionAuditPath?: string;
|
|
151
180
|
childWatchdog?: ChildWatchdogConfig;
|
|
181
|
+
/**
|
|
182
|
+
* Per-launch override of the task delivery mode. Startup-retry paths set
|
|
183
|
+
* this to "file" after an unexplained zero-activity SIGKILL so the retry
|
|
184
|
+
* keeps the task text out of the child's argv.
|
|
185
|
+
*/
|
|
186
|
+
taskDelivery?: SubagentTaskDelivery;
|
|
152
187
|
waitToolEnabled?: boolean;
|
|
153
188
|
capabilityCeiling?: ResolvedSubagentCapabilityCeiling;
|
|
154
189
|
}
|
|
@@ -585,7 +620,12 @@ export function buildPiArgs(input: BuildPiArgsInput): BuildPiArgsResult {
|
|
|
585
620
|
);
|
|
586
621
|
}
|
|
587
622
|
|
|
588
|
-
if (
|
|
623
|
+
if (
|
|
624
|
+
shouldDeliverTaskViaFile(
|
|
625
|
+
input.task,
|
|
626
|
+
input.taskDelivery ?? resolveSubagentTaskDelivery(),
|
|
627
|
+
)
|
|
628
|
+
) {
|
|
589
629
|
if (!tempDir) {
|
|
590
630
|
tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-subagent-"));
|
|
591
631
|
}
|
|
@@ -694,6 +734,9 @@ export function buildPiArgs(input: BuildPiArgsInput): BuildPiArgsResult {
|
|
|
694
734
|
process.env[SUBAGENT_PARENT_CAPABILITY_TOKEN_ENV] ??
|
|
695
735
|
"")
|
|
696
736
|
: "";
|
|
737
|
+
env[RUN_FANOUT_BUDGET_ENV] = toolPlan.fanoutAuthorized
|
|
738
|
+
? (input.runFanoutBudget ? encodeRunFanoutBudgetDescriptor(input.runFanoutBudget) : process.env[RUN_FANOUT_BUDGET_ENV])
|
|
739
|
+
: undefined;
|
|
697
740
|
env.PI_SUBAGENT_INHERIT_PROJECT_CONTEXT = input.inheritProjectContext
|
|
698
741
|
? "1"
|
|
699
742
|
: "0";
|