pi-subagents 0.53.0 → 0.55.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +66 -5
- package/README.md +1 -1
- package/agents/reviewer.md +12 -2
- package/docs/agents.md +4 -4
- package/docs/configuration.md +6 -6
- package/docs/extension-api.md +14 -4
- package/docs/models.md +41 -6
- package/docs/observability.md +14 -3
- package/docs/tool-reference.md +21 -6
- package/docs/workflows.md +7 -1
- package/index.ts +10 -1
- package/package.json +1 -1
- package/prompts/council.md +19 -7
- package/prompts/parallel-review.md +5 -1
- package/prompts/review-loop.md +9 -5
- package/skills/council-mode/SKILL.md +50 -26
- package/skills/pi-subagents/SKILL.md +5 -1
- package/skills/pi-subagents/references/constraints-and-recipes.md +16 -1
- package/skills/pi-subagents/references/execution-controls.md +13 -4
- package/skills/pi-subagents/references/multi-lane-orchestration.md +3 -3
- package/skills/pi-subagents/references/prompting-and-roles.md +14 -7
- package/src/agents/agent-management.ts +115 -14
- package/src/agents/agent-serializer.ts +2 -2
- package/src/agents/agents.ts +195 -47
- package/src/api/external-job-provider.ts +10 -1
- package/src/api/preflight.ts +33 -19
- package/src/api/project-panes.ts +2 -0
- package/src/extension/doctor.ts +10 -0
- package/src/extension/fanout-child.ts +3 -2
- package/src/extension/index.ts +24 -4
- package/src/extension/public-execution.ts +12 -9
- package/src/extension/rpc.ts +77 -3
- package/src/extension/schemas.ts +2 -1
- package/src/extension/tool-description.ts +9 -6
- package/src/extension/tool-result.ts +19 -0
- package/src/inspectors/herdr/client.ts +3 -3
- package/src/inspectors/herdr/focus.ts +55 -0
- package/src/inspectors/herdr/project-panes.ts +228 -44
- package/src/integrations/herdr-status.ts +26 -4
- package/src/runs/background/async-execution.ts +112 -13
- package/src/runs/background/async-job-tracker.ts +33 -14
- package/src/runs/background/async-resume.ts +20 -2
- package/src/runs/background/async-retention.ts +1 -1
- package/src/runs/background/async-status.ts +10 -0
- package/src/runs/background/chain-root-attachment.ts +5 -0
- package/src/runs/background/control-channel.ts +98 -10
- package/src/runs/background/notify.ts +62 -1
- package/src/runs/background/result-watcher.ts +4 -3
- package/src/runs/background/run-status.ts +15 -1
- package/src/runs/background/stale-run-reconciler.ts +3 -0
- package/src/runs/background/subagent-runner.ts +251 -38
- package/src/runs/background/wait-completions.ts +2 -0
- package/src/runs/background/wait-tool.ts +4 -3
- package/src/runs/foreground/async-stop-action.ts +23 -2
- package/src/runs/foreground/execution.ts +110 -8
- package/src/runs/foreground/subagent-executor.ts +361 -45
- package/src/runs/foreground/workflow-detach-reconcile.ts +3 -0
- package/src/runs/shared/acceptance.ts +1 -0
- package/src/runs/shared/child-identity.ts +36 -0
- package/src/runs/shared/completion-guard.ts +50 -1
- package/src/runs/shared/external-job-bridge.ts +53 -37
- package/src/runs/shared/external-job-runner.ts +126 -23
- package/src/runs/shared/model-fallback.ts +46 -15
- package/src/runs/shared/model-scope.ts +106 -39
- package/src/runs/shared/orca-progress-tabs.ts +71 -9
- package/src/runs/shared/parallel-utils.ts +12 -0
- package/src/runs/shared/pi-args.ts +31 -1
- package/src/runs/shared/subagent-prompt-runtime.ts +39 -10
- package/src/runs/shared/tool-availability.ts +1 -3
- package/src/shared/launch-contract.ts +6 -5
- package/src/shared/thinking-ceiling.ts +52 -0
- package/src/shared/types.ts +72 -5
- package/src/tui/fleet-status.ts +66 -13
- package/src/tui/render.ts +5 -4
- package/src/watchdog/permission-arbiter.ts +59 -51
- package/src/workflows/scripted-workflow.ts +107 -10
|
@@ -3,6 +3,7 @@ import { SubagentWaitParams } from "../../extension/schemas.ts";
|
|
|
3
3
|
import type { Details, SubagentState } from "../../shared/types.ts";
|
|
4
4
|
import { resolveWaitToolConfig, waitForSubagents } from "./subagent-wait.ts";
|
|
5
5
|
import type { WaitSubscriptionManager } from "./wait-subscriptions.ts";
|
|
6
|
+
import { finalizeToolResult } from "../../extension/tool-result.ts";
|
|
6
7
|
|
|
7
8
|
export function registerWaitTool(pi: ExtensionAPI, state: SubagentState, enabled = resolveWaitToolConfig().enabled, subscriptions?: Pick<WaitSubscriptionManager, "arm">): void {
|
|
8
9
|
const tool: ToolDefinition<typeof SubagentWaitParams, Details> = {
|
|
@@ -21,14 +22,14 @@ In an interactive chat, do not call this merely to wait: return control to the u
|
|
|
21
22
|
|
|
22
23
|
Non-blocking subscriptions are visible in subagent status and differ from disabling waitTool: waitTool.enabled=false returns immediately without registering any future wake. Provider jobs are session-scoped and identified exactly, so replacing one job with another cannot hide a completion. Provider extensions must be explicitly loaded in this process. In a child agent, keep \`subagent_wait\` in the child tool allowlist and load each provider through the agent's extensions or subagentOnlyExtensions; this tool never loads providers or grants tools itself.${enabled ? "" : "\n\nConfigured behavior: subagent_wait is disabled by config.waitTool or PI_SUBAGENT_WAIT_TOOL_ENABLED and returns immediately without blocking."}`,
|
|
23
24
|
parameters: SubagentWaitParams,
|
|
24
|
-
execute(_id, params, signal, onUpdate, ctx) {
|
|
25
|
-
return waitForSubagents(params, signal, {
|
|
25
|
+
async execute(_id, params, signal, onUpdate, ctx) {
|
|
26
|
+
return finalizeToolResult(await waitForSubagents(params, signal, {
|
|
26
27
|
state,
|
|
27
28
|
events: pi.events,
|
|
28
29
|
enabled,
|
|
29
30
|
onUpdate,
|
|
30
31
|
...(subscriptions && ctx?.hasUI ? { subscribe: (input) => subscriptions.arm(input) } : {}),
|
|
31
|
-
});
|
|
32
|
+
}));
|
|
32
33
|
},
|
|
33
34
|
};
|
|
34
35
|
pi.registerTool(tool);
|
|
@@ -3,6 +3,7 @@ import type { AgentToolResult } from "@earendil-works/pi-agent-core";
|
|
|
3
3
|
import type { Details, SubagentState } from "../../shared/types.ts";
|
|
4
4
|
import { deliverStopRequest } from "../background/control-channel.ts";
|
|
5
5
|
import { reconcileAsyncRun } from "../background/stale-run-reconciler.ts";
|
|
6
|
+
import { isStoppableAsyncStatusStep, resolveAsyncStatusChild, type ResolvedAsyncStatusChild } from "../shared/child-identity.ts";
|
|
6
7
|
|
|
7
8
|
function getAsyncStopTarget(
|
|
8
9
|
state: SubagentState,
|
|
@@ -25,6 +26,7 @@ export function stopAsyncRun(
|
|
|
25
26
|
runId: string | undefined,
|
|
26
27
|
kill?: (pid: number, signal?: NodeJS.Signals | 0) => boolean,
|
|
27
28
|
location?: { asyncDir: string | null; resolvedId?: string },
|
|
29
|
+
childId?: string,
|
|
28
30
|
): AgentToolResult<Details> | null {
|
|
29
31
|
const target = getAsyncStopTarget(state, runId, location);
|
|
30
32
|
if (!target) return null;
|
|
@@ -43,15 +45,34 @@ export function stopAsyncRun(
|
|
|
43
45
|
details: { mode: "management", results: [] },
|
|
44
46
|
};
|
|
45
47
|
}
|
|
48
|
+
let child: ResolvedAsyncStatusChild | undefined;
|
|
49
|
+
if (childId !== undefined) {
|
|
50
|
+
const resolution = resolveAsyncStatusChild(status, childId);
|
|
51
|
+
if (!resolution.ok) {
|
|
52
|
+
return {
|
|
53
|
+
content: [{ type: "text", text: resolution.message }],
|
|
54
|
+
isError: true,
|
|
55
|
+
details: { mode: "management", results: [] },
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
child = resolution.child;
|
|
59
|
+
if (!isStoppableAsyncStatusStep(child.step)) {
|
|
60
|
+
return {
|
|
61
|
+
content: [{ type: "text", text: `Child '${childId}' in async run '${status.runId}' is ${child.step.status}; stop only supports pending or running children.` }],
|
|
62
|
+
isError: true,
|
|
63
|
+
details: { mode: "management", results: [] },
|
|
64
|
+
};
|
|
65
|
+
}
|
|
66
|
+
}
|
|
46
67
|
try {
|
|
47
|
-
deliverStopRequest({ asyncDir: target.asyncDir, pid: typeof status.pid === "number" ? status.pid : undefined, kill, source: "stop-action" });
|
|
68
|
+
deliverStopRequest({ asyncDir: target.asyncDir, pid: typeof status.pid === "number" ? status.pid : undefined, kill, source: "stop-action", targetIndex: child?.index, childId: child?.id ?? childId });
|
|
48
69
|
const tracked = state.asyncJobs.get(target.asyncId);
|
|
49
70
|
if (tracked) {
|
|
50
71
|
tracked.activityState = undefined;
|
|
51
72
|
tracked.updatedAt = Date.now();
|
|
52
73
|
}
|
|
53
74
|
return {
|
|
54
|
-
content: [{ type: "text", text: `Stop requested for async run ${target.asyncId}.` }],
|
|
75
|
+
content: [{ type: "text", text: child ? `Stop requested for child ${child.id} in async run ${target.asyncId}.` : `Stop requested for async run ${target.asyncId}.` }],
|
|
55
76
|
details: { mode: "management", results: [] },
|
|
56
77
|
};
|
|
57
78
|
} catch (error) {
|
|
@@ -55,7 +55,7 @@ import {
|
|
|
55
55
|
import { buildSkillInjection, resolveSkillsWithFallback } from "../../agents/skills.ts";
|
|
56
56
|
import { buildAgentMemoryInjection } from "../../agents/agent-memory.ts";
|
|
57
57
|
import { effectiveToolTimeoutMs, formatToolTimeoutMessage, resolveToolTimeoutMs, toolTimeoutCallKey, toolTimeoutFromEnv } from "../shared/tool-timeout.ts";
|
|
58
|
-
import { evaluateCompletionMutationGuard } from "../shared/completion-guard.ts";
|
|
58
|
+
import { evaluateCompletionMutationGuard, validateImplementationToolContract } from "../shared/completion-guard.ts";
|
|
59
59
|
import { arbitrateCompletionGuardRescue } from "../shared/llm-intent-arbiter.ts";
|
|
60
60
|
import { getPiSpawnCommand } from "../shared/pi-spawn.ts";
|
|
61
61
|
import { createJsonlWriter } from "../../shared/jsonl-writer.ts";
|
|
@@ -66,13 +66,16 @@ import { applyThinkingSuffix, buildPiArgs, cleanupTempDir, projectLaunchResolved
|
|
|
66
66
|
import { readRuntimeAcknowledgedExtensions } from "../shared/runtime-acknowledged-extensions.ts";
|
|
67
67
|
import { assertAgentAllowedByCapabilityCeiling, decodeSubagentCapabilityCeiling, intersectSubagentCapabilityCeilings, resolveCurrentSubagentCapabilityCeiling, SUBAGENT_CAPABILITY_CEILING_ENV } from "../shared/capability-ceiling.ts";
|
|
68
68
|
import { resolveEffectiveThinking } from "../../shared/model-info.ts";
|
|
69
|
+
import { assertThinkingWithinCeiling, decodeThinkingCeiling, intersectThinkingCeilings, SUBAGENT_THINKING_CEILING_ENV } from "../../shared/thinking-ceiling.ts";
|
|
69
70
|
import { MISSING_STRUCTURED_OUTPUT_CALL_ERROR, readStructuredOutput } from "../shared/structured-output.ts";
|
|
70
71
|
import { formatProcessSignalError, isUnexplainedProcessSignal } from "../shared/process-signal.ts";
|
|
71
72
|
import { readChildToolDiagnosticError } from "../shared/tool-availability.ts";
|
|
72
73
|
import { captureSingleOutputSnapshot, extractChildWrittenOutput, finalizeSingleOutput, formatSavedOutputReference, injectOutputPathSystemPrompt, resolveSingleOutput, validateFileOnlyOutputMode, type SingleOutputSnapshot } from "../shared/single-output.ts";
|
|
73
74
|
import {
|
|
74
75
|
buildModelCandidates,
|
|
76
|
+
formatSubagentModelVerificationError,
|
|
75
77
|
formatModelAttemptNote,
|
|
78
|
+
isContextOverflow,
|
|
76
79
|
isRetryableModelFailure,
|
|
77
80
|
recordRetryableModelFailure,
|
|
78
81
|
} from "../shared/model-fallback.ts";
|
|
@@ -308,10 +311,14 @@ async function runSingleAttempt(
|
|
|
308
311
|
originalTask?: string;
|
|
309
312
|
taskDelivery?: SubagentTaskDelivery;
|
|
310
313
|
orcaProgressTab?: OrcaProgressTab;
|
|
314
|
+
launchWarnings: { emitted: boolean };
|
|
315
|
+
verifyModel: boolean;
|
|
311
316
|
},
|
|
312
317
|
): Promise<SingleResult> {
|
|
313
318
|
const effectiveThinking = options.thinkingOverride ?? agent.thinking;
|
|
314
319
|
const modelArg = applyThinkingSuffix(model, effectiveThinking, options.thinkingOverride !== undefined);
|
|
320
|
+
assertThinkingWithinCeiling({ model: modelArg, configThinking: effectiveThinking, ceiling: options.thinkingCeiling, agent: agent.name, runId: options.runId });
|
|
321
|
+
const expectedModelForVerification = shared.verifyModel ? modelArg : undefined;
|
|
315
322
|
const resolvedThinking = resolveEffectiveThinking(modelArg, effectiveThinking);
|
|
316
323
|
const watchdogConfig = resolveWatchdogConfig(options.cwd ?? runtimeCwd);
|
|
317
324
|
const childWatchdog = watchdogConfig.ok
|
|
@@ -326,7 +333,7 @@ async function runSingleAttempt(
|
|
|
326
333
|
const permissionAuditPath = permissionRules && options.artifactsDir
|
|
327
334
|
? path.join(options.artifactsDir, "permission-audit", `${options.runId}-${options.index ?? 0}.jsonl`)
|
|
328
335
|
: undefined;
|
|
329
|
-
const { args, env: sharedEnv, tempDir, toolDiagnosticPath, runtimeAcknowledgedExtensionsPath, capabilityAudit } = buildPiArgs({
|
|
336
|
+
const { args, env: sharedEnv, tempDir, toolDiagnosticPath, runtimeAcknowledgedExtensionsPath, capabilityAudit, warnings } = buildPiArgs({
|
|
330
337
|
baseArgs: ["--mode", "json", "-p"],
|
|
331
338
|
task,
|
|
332
339
|
taskDelivery: shared.taskDelivery,
|
|
@@ -368,7 +375,12 @@ async function runSingleAttempt(
|
|
|
368
375
|
childWatchdog,
|
|
369
376
|
waitToolEnabled: options.waitToolEnabled,
|
|
370
377
|
capabilityCeiling: options.capabilityCeiling,
|
|
378
|
+
thinkingCeiling: options.thinkingCeiling,
|
|
371
379
|
});
|
|
380
|
+
if (!shared.launchWarnings.emitted && warnings.length > 0) {
|
|
381
|
+
for (const warning of warnings) console.warn(`[pi-subagents] ${warning}`);
|
|
382
|
+
shared.launchWarnings.emitted = true;
|
|
383
|
+
}
|
|
372
384
|
|
|
373
385
|
const effectiveSystemPrompt = appendTurnBudgetSystemPrompt(shared.systemPrompt, options.turnBudget);
|
|
374
386
|
const toolPlan = resolvePiLaunchToolPlan({
|
|
@@ -382,7 +394,38 @@ async function runSingleAttempt(
|
|
|
382
394
|
capabilityCeiling: options.capabilityCeiling,
|
|
383
395
|
inheritedCapabilityCeiling: decodeSubagentCapabilityCeiling(process.env[SUBAGENT_CAPABILITY_CEILING_ENV]),
|
|
384
396
|
agentName: agent.name,
|
|
397
|
+
permissionRules,
|
|
385
398
|
});
|
|
399
|
+
const contractTools = toolPlan.explicitToolAllowlist ? toolPlan.effectiveToolAllowlist : undefined;
|
|
400
|
+
const contractError = validateImplementationToolContract({
|
|
401
|
+
agent: agent.name,
|
|
402
|
+
task: shared.originalTask ?? task,
|
|
403
|
+
tools: contractTools,
|
|
404
|
+
mcpDirectTools: toolPlan.effectiveMcpTools,
|
|
405
|
+
configuredExtensions: toolPlan.configuredExtensions,
|
|
406
|
+
requestedTools: toolPlan.requestedBuiltinTools,
|
|
407
|
+
acceptanceRole: agent.acceptanceRole,
|
|
408
|
+
completionGuard: agent.completionGuard,
|
|
409
|
+
});
|
|
410
|
+
if (contractError) {
|
|
411
|
+
cleanupTempDir(tempDir);
|
|
412
|
+
return {
|
|
413
|
+
index: options.index ?? 0,
|
|
414
|
+
agent: agent.name,
|
|
415
|
+
task,
|
|
416
|
+
messages: [],
|
|
417
|
+
finalOutput: "",
|
|
418
|
+
exitCode: 1,
|
|
419
|
+
error: contractError,
|
|
420
|
+
usage: emptyUsage(),
|
|
421
|
+
model: modelArg,
|
|
422
|
+
modelAttempts: [],
|
|
423
|
+
attemptedModels: [],
|
|
424
|
+
progressSummary: { status: "failed", toolCount: 0, tokens: 0, durationMs: 0 },
|
|
425
|
+
...(toolPlan.capabilityCeiling ? { capabilityCeiling: toolPlan.capabilityCeiling } : {}),
|
|
426
|
+
...(toolPlan.capabilityAudit ? { capabilityAudit: toolPlan.capabilityAudit } : {}),
|
|
427
|
+
};
|
|
428
|
+
}
|
|
386
429
|
const launchResolvedExtensions = projectLaunchResolvedChildExtensions(toolPlan);
|
|
387
430
|
const launchContractDigest = launchBindingDigest({
|
|
388
431
|
definitionDigest: agentDefinitionDigest(agent),
|
|
@@ -390,6 +433,7 @@ async function runSingleAttempt(
|
|
|
390
433
|
...(modelArg ? { model: modelArg } : {}),
|
|
391
434
|
modelCandidates: shared.modelCandidates,
|
|
392
435
|
...(resolvedThinking ? { thinking: resolvedThinking } : {}),
|
|
436
|
+
...(options.thinkingCeiling ? { thinkingCeiling: options.thinkingCeiling } : {}),
|
|
393
437
|
systemPrompt: effectiveSystemPrompt,
|
|
394
438
|
systemPromptMode: agent.systemPromptMode,
|
|
395
439
|
inheritProjectContext: agent.inheritProjectContext,
|
|
@@ -483,6 +527,7 @@ async function runSingleAttempt(
|
|
|
483
527
|
let observedMutationAttempt = false;
|
|
484
528
|
let structuredOutputToolInvoked = false;
|
|
485
529
|
let structuredOutputMessageStartIndex: number | undefined;
|
|
530
|
+
let toolAvailabilityError: string | undefined;
|
|
486
531
|
|
|
487
532
|
const exitCode = await new Promise<number>((resolve) => {
|
|
488
533
|
const spawnSpec = getPiSpawnCommand(args);
|
|
@@ -1035,6 +1080,10 @@ async function runSingleAttempt(
|
|
|
1035
1080
|
if (evt.message.model) {
|
|
1036
1081
|
progress.model = evt.message.model;
|
|
1037
1082
|
if (!result.model) result.model = evt.message.model;
|
|
1083
|
+
if (expectedModelForVerification && !hasToolCall) {
|
|
1084
|
+
const modelVerificationError = formatSubagentModelVerificationError(expectedModelForVerification, evt.message.model, options.availableModels);
|
|
1085
|
+
if (modelVerificationError && !result.error) result.error = modelVerificationError;
|
|
1086
|
+
}
|
|
1038
1087
|
}
|
|
1039
1088
|
if (evt.message.errorMessage) assistantError = evt.message.errorMessage;
|
|
1040
1089
|
const assistantText = extractTextFromContent(evt.message.content);
|
|
@@ -1051,6 +1100,20 @@ async function runSingleAttempt(
|
|
|
1051
1100
|
}
|
|
1052
1101
|
|
|
1053
1102
|
if (evt.type === "tool_result_end" && evt.message) {
|
|
1103
|
+
const toolResultCompletion = {
|
|
1104
|
+
toolCallId: (evt.message as { toolCallId?: unknown }).toolCallId ?? (evt as { toolCallId?: unknown }).toolCallId,
|
|
1105
|
+
toolName: (evt.message as { toolName?: unknown }).toolName ?? (evt as { toolName?: unknown }).toolName,
|
|
1106
|
+
};
|
|
1107
|
+
clearActiveToolTimeout(toolResultCompletion);
|
|
1108
|
+
const endedTool = removeActiveToolCall(toolResultCompletion);
|
|
1109
|
+
if (endedTool) {
|
|
1110
|
+
progress.recentTools.push({
|
|
1111
|
+
tool: endedTool.tool,
|
|
1112
|
+
args: endedTool.args,
|
|
1113
|
+
endMs: now,
|
|
1114
|
+
});
|
|
1115
|
+
refreshCurrentTool();
|
|
1116
|
+
}
|
|
1054
1117
|
result.messages!.push(evt.message);
|
|
1055
1118
|
const resultText = extractTextFromContent(evt.message.content);
|
|
1056
1119
|
if (options.toolBudget && pendingToolResult && resultText.includes("Tool budget hard limit reached")) {
|
|
@@ -1246,6 +1309,7 @@ async function runSingleAttempt(
|
|
|
1246
1309
|
// JSONL artifact flush is best effort.
|
|
1247
1310
|
});
|
|
1248
1311
|
const toolDiagnosticError = readChildToolDiagnosticError(toolDiagnosticPath);
|
|
1312
|
+
toolAvailabilityError = toolDiagnosticError;
|
|
1249
1313
|
result.runtimeAcknowledgedExtensions = readRuntimeAcknowledgedExtensions(runtimeAcknowledgedExtensionsPath);
|
|
1250
1314
|
cleanupTempDir(tempDir);
|
|
1251
1315
|
stdoutReader.end();
|
|
@@ -1441,16 +1505,18 @@ async function runSingleAttempt(
|
|
|
1441
1505
|
fullOutput = fullOutput.trim() ? `${note}\n\n${fullOutput}` : note;
|
|
1442
1506
|
}
|
|
1443
1507
|
const completionGuardEnabled = isAgentContractV1(options.agentContract) ? agent.completionGuard === true : agent.completionGuard !== false;
|
|
1444
|
-
const completionGuard = result.exitCode === 0 && !result.error && completionGuardEnabled
|
|
1508
|
+
const completionGuard = ((result.exitCode === 0 && !result.error) || toolAvailabilityError) && completionGuardEnabled
|
|
1445
1509
|
? evaluateCompletionMutationGuard({
|
|
1446
1510
|
agent: agent.name,
|
|
1447
1511
|
task: shared.originalTask ?? task,
|
|
1448
1512
|
messages: result.messages ?? [],
|
|
1449
|
-
tools:
|
|
1450
|
-
mcpDirectTools:
|
|
1513
|
+
tools: contractTools,
|
|
1514
|
+
mcpDirectTools: toolPlan.effectiveMcpTools,
|
|
1515
|
+
toolAvailabilityError,
|
|
1451
1516
|
})
|
|
1452
1517
|
: undefined;
|
|
1453
1518
|
let completionGuardTriggered = completionGuard?.triggered === true && !observedMutationAttempt;
|
|
1519
|
+
const completionGuardBlocked = completionGuard?.blocked === true;
|
|
1454
1520
|
// The classifier is deliberately narrow, so a read-only review task can
|
|
1455
1521
|
// still be misread as implementation. Arbitrate BEFORE any failure side
|
|
1456
1522
|
// effect is published (effects, exit code, progress, notifications,
|
|
@@ -1471,7 +1537,9 @@ async function runSingleAttempt(
|
|
|
1471
1537
|
result.effects = {
|
|
1472
1538
|
...(result.effects ?? {}),
|
|
1473
1539
|
fileMutation: {
|
|
1474
|
-
status:
|
|
1540
|
+
status: completionGuardBlocked
|
|
1541
|
+
? "blocked"
|
|
1542
|
+
: completionGuard.expectedMutation
|
|
1475
1543
|
? completionGuardTriggered
|
|
1476
1544
|
? "missing"
|
|
1477
1545
|
: arbiterRescued
|
|
@@ -1479,7 +1547,8 @@ async function runSingleAttempt(
|
|
|
1479
1547
|
: "observed"
|
|
1480
1548
|
: "not-applicable",
|
|
1481
1549
|
expected: completionGuard.expectedMutation,
|
|
1482
|
-
attempted: completionGuard.attemptedMutation || observedMutationAttempt,
|
|
1550
|
+
attempted: completionGuardBlocked ? false : completionGuard.attemptedMutation || observedMutationAttempt,
|
|
1551
|
+
...(completionGuardBlocked && completionGuard.message ? { message: completionGuard.message } : {}),
|
|
1483
1552
|
...(completionGuardTriggered ? { message: "Subagent completed without making edits for an implementation task." } : {}),
|
|
1484
1553
|
...(arbiterRescued ? { resolvedBy: "llm-intent-arbiter" } : {}),
|
|
1485
1554
|
},
|
|
@@ -1561,6 +1630,14 @@ async function runSyncCompletionInner(
|
|
|
1561
1630
|
error: `Unknown agent: ${agentName}`,
|
|
1562
1631
|
}, options.context));
|
|
1563
1632
|
}
|
|
1633
|
+
options = {
|
|
1634
|
+
...options,
|
|
1635
|
+
thinkingCeiling: intersectThinkingCeilings(
|
|
1636
|
+
options.thinkingCeiling,
|
|
1637
|
+
agent.maxThinking,
|
|
1638
|
+
decodeThinkingCeiling(process.env[SUBAGENT_THINKING_CEILING_ENV]),
|
|
1639
|
+
),
|
|
1640
|
+
};
|
|
1564
1641
|
try {
|
|
1565
1642
|
assertAgentAllowedByCapabilityCeiling(agent.name, options.capabilityCeiling);
|
|
1566
1643
|
} catch (error) {
|
|
@@ -1674,13 +1751,30 @@ async function runSyncCompletionInner(
|
|
|
1674
1751
|
options.modelOverride ?? agent.model,
|
|
1675
1752
|
agent.fallbackModels,
|
|
1676
1753
|
options.availableModels,
|
|
1677
|
-
options.preferredModelProvider,
|
|
1754
|
+
agent.modelProvider ?? options.preferredModelProvider,
|
|
1678
1755
|
{ scope: options.modelScope, primaryModelFromParent: options.modelOverrideFromParent },
|
|
1679
1756
|
);
|
|
1757
|
+
try {
|
|
1758
|
+
for (const candidate of candidates) {
|
|
1759
|
+
const model = applyThinkingSuffix(candidate, options.thinkingOverride ?? agent.thinking, options.thinkingOverride !== undefined);
|
|
1760
|
+
assertThinkingWithinCeiling({ model, configThinking: options.thinkingOverride ?? agent.thinking, ceiling: options.thinkingCeiling, agent: agent.name, runId: options.runId });
|
|
1761
|
+
}
|
|
1762
|
+
} catch (error) {
|
|
1763
|
+
return redactResultPrompt(withRunContext({
|
|
1764
|
+
index: options.index ?? 0,
|
|
1765
|
+
agent: agent.name,
|
|
1766
|
+
task,
|
|
1767
|
+
exitCode: 1,
|
|
1768
|
+
messages: [],
|
|
1769
|
+
usage: emptyUsage(),
|
|
1770
|
+
error: error instanceof Error ? error.message : String(error),
|
|
1771
|
+
}, options.context));
|
|
1772
|
+
}
|
|
1680
1773
|
const attemptedModels: string[] = [];
|
|
1681
1774
|
const modelAttempts: ModelAttempt[] = [];
|
|
1682
1775
|
const aggregateUsage = emptyUsage();
|
|
1683
1776
|
const attemptNotes: string[] = [];
|
|
1777
|
+
const launchWarnings = { emitted: false };
|
|
1684
1778
|
let totalToolCount = 0;
|
|
1685
1779
|
let totalDurationMs = 0;
|
|
1686
1780
|
|
|
@@ -1755,6 +1849,7 @@ async function runSyncCompletionInner(
|
|
|
1755
1849
|
modelAttemptsLoop: for (let modelIndex = 0; modelIndex < modelsToTry.length; modelIndex++) {
|
|
1756
1850
|
const candidate = modelsToTry[modelIndex];
|
|
1757
1851
|
for (let startupAttemptIndex = 0; ; startupAttemptIndex++) {
|
|
1852
|
+
const verifyModel = Boolean(candidate) && !(options.modelOverrideFromParent && modelIndex === 0);
|
|
1758
1853
|
const outputSnapshot = captureSingleOutputSnapshot(options.outputPath);
|
|
1759
1854
|
const result = await runSingleAttempt(runtimeCwd, agent, taskWithAcceptance, candidate, attemptOptions, {
|
|
1760
1855
|
sessionEnabled,
|
|
@@ -1772,6 +1867,8 @@ async function runSyncCompletionInner(
|
|
|
1772
1867
|
originalTask: task,
|
|
1773
1868
|
taskDelivery: taskDeliveryOverride,
|
|
1774
1869
|
orcaProgressTab,
|
|
1870
|
+
launchWarnings,
|
|
1871
|
+
verifyModel,
|
|
1775
1872
|
});
|
|
1776
1873
|
lastResult = result;
|
|
1777
1874
|
if (startupAttemptIndex === 0) {
|
|
@@ -1864,6 +1961,11 @@ async function runSyncCompletionInner(
|
|
|
1864
1961
|
}
|
|
1865
1962
|
const retryableModelFailure = isRetryableModelFailure(result.error);
|
|
1866
1963
|
if (retryableModelFailure) recordRetryableModelFailure(result.model ?? candidate, result.error);
|
|
1964
|
+
if (isContextOverflow(result.error)) {
|
|
1965
|
+
result.contextOverflow = true;
|
|
1966
|
+
attemptNotes.push(`[fallback] ${attempt.model} failed: context overflow — the input exceeds this model's context window. Reduce the task input or use a model with a larger context window.`);
|
|
1967
|
+
break modelAttemptsLoop;
|
|
1968
|
+
}
|
|
1867
1969
|
if (!retryableModelFailure || modelIndex === modelsToTry.length - 1) break modelAttemptsLoop;
|
|
1868
1970
|
attemptNotes.push(formatModelAttemptNote(attempt, modelsToTry[modelIndex + 1]));
|
|
1869
1971
|
break;
|