pi-subagents 0.53.0 → 0.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/CHANGELOG.md +66 -5
  2. package/README.md +1 -1
  3. package/agents/reviewer.md +12 -2
  4. package/docs/agents.md +4 -4
  5. package/docs/configuration.md +6 -6
  6. package/docs/extension-api.md +14 -4
  7. package/docs/models.md +41 -6
  8. package/docs/observability.md +14 -3
  9. package/docs/tool-reference.md +21 -6
  10. package/docs/workflows.md +7 -1
  11. package/index.ts +10 -1
  12. package/package.json +1 -1
  13. package/prompts/council.md +19 -7
  14. package/prompts/parallel-review.md +5 -1
  15. package/prompts/review-loop.md +9 -5
  16. package/skills/council-mode/SKILL.md +50 -26
  17. package/skills/pi-subagents/SKILL.md +5 -1
  18. package/skills/pi-subagents/references/constraints-and-recipes.md +16 -1
  19. package/skills/pi-subagents/references/execution-controls.md +13 -4
  20. package/skills/pi-subagents/references/multi-lane-orchestration.md +3 -3
  21. package/skills/pi-subagents/references/prompting-and-roles.md +14 -7
  22. package/src/agents/agent-management.ts +115 -14
  23. package/src/agents/agent-serializer.ts +2 -2
  24. package/src/agents/agents.ts +195 -47
  25. package/src/api/external-job-provider.ts +10 -1
  26. package/src/api/preflight.ts +33 -19
  27. package/src/api/project-panes.ts +2 -0
  28. package/src/extension/doctor.ts +10 -0
  29. package/src/extension/fanout-child.ts +3 -2
  30. package/src/extension/index.ts +24 -4
  31. package/src/extension/public-execution.ts +12 -9
  32. package/src/extension/rpc.ts +77 -3
  33. package/src/extension/schemas.ts +2 -1
  34. package/src/extension/tool-description.ts +9 -6
  35. package/src/extension/tool-result.ts +19 -0
  36. package/src/inspectors/herdr/client.ts +3 -3
  37. package/src/inspectors/herdr/focus.ts +55 -0
  38. package/src/inspectors/herdr/project-panes.ts +228 -44
  39. package/src/integrations/herdr-status.ts +26 -4
  40. package/src/runs/background/async-execution.ts +112 -13
  41. package/src/runs/background/async-job-tracker.ts +33 -14
  42. package/src/runs/background/async-resume.ts +20 -2
  43. package/src/runs/background/async-retention.ts +1 -1
  44. package/src/runs/background/async-status.ts +10 -0
  45. package/src/runs/background/chain-root-attachment.ts +5 -0
  46. package/src/runs/background/control-channel.ts +98 -10
  47. package/src/runs/background/notify.ts +62 -1
  48. package/src/runs/background/result-watcher.ts +4 -3
  49. package/src/runs/background/run-status.ts +15 -1
  50. package/src/runs/background/stale-run-reconciler.ts +3 -0
  51. package/src/runs/background/subagent-runner.ts +251 -38
  52. package/src/runs/background/wait-completions.ts +2 -0
  53. package/src/runs/background/wait-tool.ts +4 -3
  54. package/src/runs/foreground/async-stop-action.ts +23 -2
  55. package/src/runs/foreground/execution.ts +110 -8
  56. package/src/runs/foreground/subagent-executor.ts +361 -45
  57. package/src/runs/foreground/workflow-detach-reconcile.ts +3 -0
  58. package/src/runs/shared/acceptance.ts +1 -0
  59. package/src/runs/shared/child-identity.ts +36 -0
  60. package/src/runs/shared/completion-guard.ts +50 -1
  61. package/src/runs/shared/external-job-bridge.ts +53 -37
  62. package/src/runs/shared/external-job-runner.ts +126 -23
  63. package/src/runs/shared/model-fallback.ts +46 -15
  64. package/src/runs/shared/model-scope.ts +106 -39
  65. package/src/runs/shared/orca-progress-tabs.ts +71 -9
  66. package/src/runs/shared/parallel-utils.ts +12 -0
  67. package/src/runs/shared/pi-args.ts +31 -1
  68. package/src/runs/shared/subagent-prompt-runtime.ts +39 -10
  69. package/src/runs/shared/tool-availability.ts +1 -3
  70. package/src/shared/launch-contract.ts +6 -5
  71. package/src/shared/thinking-ceiling.ts +52 -0
  72. package/src/shared/types.ts +72 -5
  73. package/src/tui/fleet-status.ts +66 -13
  74. package/src/tui/render.ts +5 -4
  75. package/src/watchdog/permission-arbiter.ts +59 -51
  76. package/src/workflows/scripted-workflow.ts +107 -10
@@ -3,6 +3,7 @@ import { SubagentWaitParams } from "../../extension/schemas.ts";
3
3
  import type { Details, SubagentState } from "../../shared/types.ts";
4
4
  import { resolveWaitToolConfig, waitForSubagents } from "./subagent-wait.ts";
5
5
  import type { WaitSubscriptionManager } from "./wait-subscriptions.ts";
6
+ import { finalizeToolResult } from "../../extension/tool-result.ts";
6
7
 
7
8
  export function registerWaitTool(pi: ExtensionAPI, state: SubagentState, enabled = resolveWaitToolConfig().enabled, subscriptions?: Pick<WaitSubscriptionManager, "arm">): void {
8
9
  const tool: ToolDefinition<typeof SubagentWaitParams, Details> = {
@@ -21,14 +22,14 @@ In an interactive chat, do not call this merely to wait: return control to the u
21
22
 
22
23
  Non-blocking subscriptions are visible in subagent status and differ from disabling waitTool: waitTool.enabled=false returns immediately without registering any future wake. Provider jobs are session-scoped and identified exactly, so replacing one job with another cannot hide a completion. Provider extensions must be explicitly loaded in this process. In a child agent, keep \`subagent_wait\` in the child tool allowlist and load each provider through the agent's extensions or subagentOnlyExtensions; this tool never loads providers or grants tools itself.${enabled ? "" : "\n\nConfigured behavior: subagent_wait is disabled by config.waitTool or PI_SUBAGENT_WAIT_TOOL_ENABLED and returns immediately without blocking."}`,
23
24
  parameters: SubagentWaitParams,
24
- execute(_id, params, signal, onUpdate, ctx) {
25
- return waitForSubagents(params, signal, {
25
+ async execute(_id, params, signal, onUpdate, ctx) {
26
+ return finalizeToolResult(await waitForSubagents(params, signal, {
26
27
  state,
27
28
  events: pi.events,
28
29
  enabled,
29
30
  onUpdate,
30
31
  ...(subscriptions && ctx?.hasUI ? { subscribe: (input) => subscriptions.arm(input) } : {}),
31
- });
32
+ }));
32
33
  },
33
34
  };
34
35
  pi.registerTool(tool);
@@ -3,6 +3,7 @@ import type { AgentToolResult } from "@earendil-works/pi-agent-core";
3
3
  import type { Details, SubagentState } from "../../shared/types.ts";
4
4
  import { deliverStopRequest } from "../background/control-channel.ts";
5
5
  import { reconcileAsyncRun } from "../background/stale-run-reconciler.ts";
6
+ import { isStoppableAsyncStatusStep, resolveAsyncStatusChild, type ResolvedAsyncStatusChild } from "../shared/child-identity.ts";
6
7
 
7
8
  function getAsyncStopTarget(
8
9
  state: SubagentState,
@@ -25,6 +26,7 @@ export function stopAsyncRun(
25
26
  runId: string | undefined,
26
27
  kill?: (pid: number, signal?: NodeJS.Signals | 0) => boolean,
27
28
  location?: { asyncDir: string | null; resolvedId?: string },
29
+ childId?: string,
28
30
  ): AgentToolResult<Details> | null {
29
31
  const target = getAsyncStopTarget(state, runId, location);
30
32
  if (!target) return null;
@@ -43,15 +45,34 @@ export function stopAsyncRun(
43
45
  details: { mode: "management", results: [] },
44
46
  };
45
47
  }
48
+ let child: ResolvedAsyncStatusChild | undefined;
49
+ if (childId !== undefined) {
50
+ const resolution = resolveAsyncStatusChild(status, childId);
51
+ if (!resolution.ok) {
52
+ return {
53
+ content: [{ type: "text", text: resolution.message }],
54
+ isError: true,
55
+ details: { mode: "management", results: [] },
56
+ };
57
+ }
58
+ child = resolution.child;
59
+ if (!isStoppableAsyncStatusStep(child.step)) {
60
+ return {
61
+ content: [{ type: "text", text: `Child '${childId}' in async run '${status.runId}' is ${child.step.status}; stop only supports pending or running children.` }],
62
+ isError: true,
63
+ details: { mode: "management", results: [] },
64
+ };
65
+ }
66
+ }
46
67
  try {
47
- deliverStopRequest({ asyncDir: target.asyncDir, pid: typeof status.pid === "number" ? status.pid : undefined, kill, source: "stop-action" });
68
+ deliverStopRequest({ asyncDir: target.asyncDir, pid: typeof status.pid === "number" ? status.pid : undefined, kill, source: "stop-action", targetIndex: child?.index, childId: child?.id ?? childId });
48
69
  const tracked = state.asyncJobs.get(target.asyncId);
49
70
  if (tracked) {
50
71
  tracked.activityState = undefined;
51
72
  tracked.updatedAt = Date.now();
52
73
  }
53
74
  return {
54
- content: [{ type: "text", text: `Stop requested for async run ${target.asyncId}.` }],
75
+ content: [{ type: "text", text: child ? `Stop requested for child ${child.id} in async run ${target.asyncId}.` : `Stop requested for async run ${target.asyncId}.` }],
55
76
  details: { mode: "management", results: [] },
56
77
  };
57
78
  } catch (error) {
@@ -55,7 +55,7 @@ import {
55
55
  import { buildSkillInjection, resolveSkillsWithFallback } from "../../agents/skills.ts";
56
56
  import { buildAgentMemoryInjection } from "../../agents/agent-memory.ts";
57
57
  import { effectiveToolTimeoutMs, formatToolTimeoutMessage, resolveToolTimeoutMs, toolTimeoutCallKey, toolTimeoutFromEnv } from "../shared/tool-timeout.ts";
58
- import { evaluateCompletionMutationGuard } from "../shared/completion-guard.ts";
58
+ import { evaluateCompletionMutationGuard, validateImplementationToolContract } from "../shared/completion-guard.ts";
59
59
  import { arbitrateCompletionGuardRescue } from "../shared/llm-intent-arbiter.ts";
60
60
  import { getPiSpawnCommand } from "../shared/pi-spawn.ts";
61
61
  import { createJsonlWriter } from "../../shared/jsonl-writer.ts";
@@ -66,13 +66,16 @@ import { applyThinkingSuffix, buildPiArgs, cleanupTempDir, projectLaunchResolved
66
66
  import { readRuntimeAcknowledgedExtensions } from "../shared/runtime-acknowledged-extensions.ts";
67
67
  import { assertAgentAllowedByCapabilityCeiling, decodeSubagentCapabilityCeiling, intersectSubagentCapabilityCeilings, resolveCurrentSubagentCapabilityCeiling, SUBAGENT_CAPABILITY_CEILING_ENV } from "../shared/capability-ceiling.ts";
68
68
  import { resolveEffectiveThinking } from "../../shared/model-info.ts";
69
+ import { assertThinkingWithinCeiling, decodeThinkingCeiling, intersectThinkingCeilings, SUBAGENT_THINKING_CEILING_ENV } from "../../shared/thinking-ceiling.ts";
69
70
  import { MISSING_STRUCTURED_OUTPUT_CALL_ERROR, readStructuredOutput } from "../shared/structured-output.ts";
70
71
  import { formatProcessSignalError, isUnexplainedProcessSignal } from "../shared/process-signal.ts";
71
72
  import { readChildToolDiagnosticError } from "../shared/tool-availability.ts";
72
73
  import { captureSingleOutputSnapshot, extractChildWrittenOutput, finalizeSingleOutput, formatSavedOutputReference, injectOutputPathSystemPrompt, resolveSingleOutput, validateFileOnlyOutputMode, type SingleOutputSnapshot } from "../shared/single-output.ts";
73
74
  import {
74
75
  buildModelCandidates,
76
+ formatSubagentModelVerificationError,
75
77
  formatModelAttemptNote,
78
+ isContextOverflow,
76
79
  isRetryableModelFailure,
77
80
  recordRetryableModelFailure,
78
81
  } from "../shared/model-fallback.ts";
@@ -308,10 +311,14 @@ async function runSingleAttempt(
308
311
  originalTask?: string;
309
312
  taskDelivery?: SubagentTaskDelivery;
310
313
  orcaProgressTab?: OrcaProgressTab;
314
+ launchWarnings: { emitted: boolean };
315
+ verifyModel: boolean;
311
316
  },
312
317
  ): Promise<SingleResult> {
313
318
  const effectiveThinking = options.thinkingOverride ?? agent.thinking;
314
319
  const modelArg = applyThinkingSuffix(model, effectiveThinking, options.thinkingOverride !== undefined);
320
+ assertThinkingWithinCeiling({ model: modelArg, configThinking: effectiveThinking, ceiling: options.thinkingCeiling, agent: agent.name, runId: options.runId });
321
+ const expectedModelForVerification = shared.verifyModel ? modelArg : undefined;
315
322
  const resolvedThinking = resolveEffectiveThinking(modelArg, effectiveThinking);
316
323
  const watchdogConfig = resolveWatchdogConfig(options.cwd ?? runtimeCwd);
317
324
  const childWatchdog = watchdogConfig.ok
@@ -326,7 +333,7 @@ async function runSingleAttempt(
326
333
  const permissionAuditPath = permissionRules && options.artifactsDir
327
334
  ? path.join(options.artifactsDir, "permission-audit", `${options.runId}-${options.index ?? 0}.jsonl`)
328
335
  : undefined;
329
- const { args, env: sharedEnv, tempDir, toolDiagnosticPath, runtimeAcknowledgedExtensionsPath, capabilityAudit } = buildPiArgs({
336
+ const { args, env: sharedEnv, tempDir, toolDiagnosticPath, runtimeAcknowledgedExtensionsPath, capabilityAudit, warnings } = buildPiArgs({
330
337
  baseArgs: ["--mode", "json", "-p"],
331
338
  task,
332
339
  taskDelivery: shared.taskDelivery,
@@ -368,7 +375,12 @@ async function runSingleAttempt(
368
375
  childWatchdog,
369
376
  waitToolEnabled: options.waitToolEnabled,
370
377
  capabilityCeiling: options.capabilityCeiling,
378
+ thinkingCeiling: options.thinkingCeiling,
371
379
  });
380
+ if (!shared.launchWarnings.emitted && warnings.length > 0) {
381
+ for (const warning of warnings) console.warn(`[pi-subagents] ${warning}`);
382
+ shared.launchWarnings.emitted = true;
383
+ }
372
384
 
373
385
  const effectiveSystemPrompt = appendTurnBudgetSystemPrompt(shared.systemPrompt, options.turnBudget);
374
386
  const toolPlan = resolvePiLaunchToolPlan({
@@ -382,7 +394,38 @@ async function runSingleAttempt(
382
394
  capabilityCeiling: options.capabilityCeiling,
383
395
  inheritedCapabilityCeiling: decodeSubagentCapabilityCeiling(process.env[SUBAGENT_CAPABILITY_CEILING_ENV]),
384
396
  agentName: agent.name,
397
+ permissionRules,
385
398
  });
399
+ const contractTools = toolPlan.explicitToolAllowlist ? toolPlan.effectiveToolAllowlist : undefined;
400
+ const contractError = validateImplementationToolContract({
401
+ agent: agent.name,
402
+ task: shared.originalTask ?? task,
403
+ tools: contractTools,
404
+ mcpDirectTools: toolPlan.effectiveMcpTools,
405
+ configuredExtensions: toolPlan.configuredExtensions,
406
+ requestedTools: toolPlan.requestedBuiltinTools,
407
+ acceptanceRole: agent.acceptanceRole,
408
+ completionGuard: agent.completionGuard,
409
+ });
410
+ if (contractError) {
411
+ cleanupTempDir(tempDir);
412
+ return {
413
+ index: options.index ?? 0,
414
+ agent: agent.name,
415
+ task,
416
+ messages: [],
417
+ finalOutput: "",
418
+ exitCode: 1,
419
+ error: contractError,
420
+ usage: emptyUsage(),
421
+ model: modelArg,
422
+ modelAttempts: [],
423
+ attemptedModels: [],
424
+ progressSummary: { status: "failed", toolCount: 0, tokens: 0, durationMs: 0 },
425
+ ...(toolPlan.capabilityCeiling ? { capabilityCeiling: toolPlan.capabilityCeiling } : {}),
426
+ ...(toolPlan.capabilityAudit ? { capabilityAudit: toolPlan.capabilityAudit } : {}),
427
+ };
428
+ }
386
429
  const launchResolvedExtensions = projectLaunchResolvedChildExtensions(toolPlan);
387
430
  const launchContractDigest = launchBindingDigest({
388
431
  definitionDigest: agentDefinitionDigest(agent),
@@ -390,6 +433,7 @@ async function runSingleAttempt(
390
433
  ...(modelArg ? { model: modelArg } : {}),
391
434
  modelCandidates: shared.modelCandidates,
392
435
  ...(resolvedThinking ? { thinking: resolvedThinking } : {}),
436
+ ...(options.thinkingCeiling ? { thinkingCeiling: options.thinkingCeiling } : {}),
393
437
  systemPrompt: effectiveSystemPrompt,
394
438
  systemPromptMode: agent.systemPromptMode,
395
439
  inheritProjectContext: agent.inheritProjectContext,
@@ -483,6 +527,7 @@ async function runSingleAttempt(
483
527
  let observedMutationAttempt = false;
484
528
  let structuredOutputToolInvoked = false;
485
529
  let structuredOutputMessageStartIndex: number | undefined;
530
+ let toolAvailabilityError: string | undefined;
486
531
 
487
532
  const exitCode = await new Promise<number>((resolve) => {
488
533
  const spawnSpec = getPiSpawnCommand(args);
@@ -1035,6 +1080,10 @@ async function runSingleAttempt(
1035
1080
  if (evt.message.model) {
1036
1081
  progress.model = evt.message.model;
1037
1082
  if (!result.model) result.model = evt.message.model;
1083
+ if (expectedModelForVerification && !hasToolCall) {
1084
+ const modelVerificationError = formatSubagentModelVerificationError(expectedModelForVerification, evt.message.model, options.availableModels);
1085
+ if (modelVerificationError && !result.error) result.error = modelVerificationError;
1086
+ }
1038
1087
  }
1039
1088
  if (evt.message.errorMessage) assistantError = evt.message.errorMessage;
1040
1089
  const assistantText = extractTextFromContent(evt.message.content);
@@ -1051,6 +1100,20 @@ async function runSingleAttempt(
1051
1100
  }
1052
1101
 
1053
1102
  if (evt.type === "tool_result_end" && evt.message) {
1103
+ const toolResultCompletion = {
1104
+ toolCallId: (evt.message as { toolCallId?: unknown }).toolCallId ?? (evt as { toolCallId?: unknown }).toolCallId,
1105
+ toolName: (evt.message as { toolName?: unknown }).toolName ?? (evt as { toolName?: unknown }).toolName,
1106
+ };
1107
+ clearActiveToolTimeout(toolResultCompletion);
1108
+ const endedTool = removeActiveToolCall(toolResultCompletion);
1109
+ if (endedTool) {
1110
+ progress.recentTools.push({
1111
+ tool: endedTool.tool,
1112
+ args: endedTool.args,
1113
+ endMs: now,
1114
+ });
1115
+ refreshCurrentTool();
1116
+ }
1054
1117
  result.messages!.push(evt.message);
1055
1118
  const resultText = extractTextFromContent(evt.message.content);
1056
1119
  if (options.toolBudget && pendingToolResult && resultText.includes("Tool budget hard limit reached")) {
@@ -1246,6 +1309,7 @@ async function runSingleAttempt(
1246
1309
  // JSONL artifact flush is best effort.
1247
1310
  });
1248
1311
  const toolDiagnosticError = readChildToolDiagnosticError(toolDiagnosticPath);
1312
+ toolAvailabilityError = toolDiagnosticError;
1249
1313
  result.runtimeAcknowledgedExtensions = readRuntimeAcknowledgedExtensions(runtimeAcknowledgedExtensionsPath);
1250
1314
  cleanupTempDir(tempDir);
1251
1315
  stdoutReader.end();
@@ -1441,16 +1505,18 @@ async function runSingleAttempt(
1441
1505
  fullOutput = fullOutput.trim() ? `${note}\n\n${fullOutput}` : note;
1442
1506
  }
1443
1507
  const completionGuardEnabled = isAgentContractV1(options.agentContract) ? agent.completionGuard === true : agent.completionGuard !== false;
1444
- const completionGuard = result.exitCode === 0 && !result.error && completionGuardEnabled
1508
+ const completionGuard = ((result.exitCode === 0 && !result.error) || toolAvailabilityError) && completionGuardEnabled
1445
1509
  ? evaluateCompletionMutationGuard({
1446
1510
  agent: agent.name,
1447
1511
  task: shared.originalTask ?? task,
1448
1512
  messages: result.messages ?? [],
1449
- tools: agent.tools,
1450
- mcpDirectTools: agent.mcpDirectTools,
1513
+ tools: contractTools,
1514
+ mcpDirectTools: toolPlan.effectiveMcpTools,
1515
+ toolAvailabilityError,
1451
1516
  })
1452
1517
  : undefined;
1453
1518
  let completionGuardTriggered = completionGuard?.triggered === true && !observedMutationAttempt;
1519
+ const completionGuardBlocked = completionGuard?.blocked === true;
1454
1520
  // The classifier is deliberately narrow, so a read-only review task can
1455
1521
  // still be misread as implementation. Arbitrate BEFORE any failure side
1456
1522
  // effect is published (effects, exit code, progress, notifications,
@@ -1471,7 +1537,9 @@ async function runSingleAttempt(
1471
1537
  result.effects = {
1472
1538
  ...(result.effects ?? {}),
1473
1539
  fileMutation: {
1474
- status: completionGuard.expectedMutation
1540
+ status: completionGuardBlocked
1541
+ ? "blocked"
1542
+ : completionGuard.expectedMutation
1475
1543
  ? completionGuardTriggered
1476
1544
  ? "missing"
1477
1545
  : arbiterRescued
@@ -1479,7 +1547,8 @@ async function runSingleAttempt(
1479
1547
  : "observed"
1480
1548
  : "not-applicable",
1481
1549
  expected: completionGuard.expectedMutation,
1482
- attempted: completionGuard.attemptedMutation || observedMutationAttempt,
1550
+ attempted: completionGuardBlocked ? false : completionGuard.attemptedMutation || observedMutationAttempt,
1551
+ ...(completionGuardBlocked && completionGuard.message ? { message: completionGuard.message } : {}),
1483
1552
  ...(completionGuardTriggered ? { message: "Subagent completed without making edits for an implementation task." } : {}),
1484
1553
  ...(arbiterRescued ? { resolvedBy: "llm-intent-arbiter" } : {}),
1485
1554
  },
@@ -1561,6 +1630,14 @@ async function runSyncCompletionInner(
1561
1630
  error: `Unknown agent: ${agentName}`,
1562
1631
  }, options.context));
1563
1632
  }
1633
+ options = {
1634
+ ...options,
1635
+ thinkingCeiling: intersectThinkingCeilings(
1636
+ options.thinkingCeiling,
1637
+ agent.maxThinking,
1638
+ decodeThinkingCeiling(process.env[SUBAGENT_THINKING_CEILING_ENV]),
1639
+ ),
1640
+ };
1564
1641
  try {
1565
1642
  assertAgentAllowedByCapabilityCeiling(agent.name, options.capabilityCeiling);
1566
1643
  } catch (error) {
@@ -1674,13 +1751,30 @@ async function runSyncCompletionInner(
1674
1751
  options.modelOverride ?? agent.model,
1675
1752
  agent.fallbackModels,
1676
1753
  options.availableModels,
1677
- options.preferredModelProvider,
1754
+ agent.modelProvider ?? options.preferredModelProvider,
1678
1755
  { scope: options.modelScope, primaryModelFromParent: options.modelOverrideFromParent },
1679
1756
  );
1757
+ try {
1758
+ for (const candidate of candidates) {
1759
+ const model = applyThinkingSuffix(candidate, options.thinkingOverride ?? agent.thinking, options.thinkingOverride !== undefined);
1760
+ assertThinkingWithinCeiling({ model, configThinking: options.thinkingOverride ?? agent.thinking, ceiling: options.thinkingCeiling, agent: agent.name, runId: options.runId });
1761
+ }
1762
+ } catch (error) {
1763
+ return redactResultPrompt(withRunContext({
1764
+ index: options.index ?? 0,
1765
+ agent: agent.name,
1766
+ task,
1767
+ exitCode: 1,
1768
+ messages: [],
1769
+ usage: emptyUsage(),
1770
+ error: error instanceof Error ? error.message : String(error),
1771
+ }, options.context));
1772
+ }
1680
1773
  const attemptedModels: string[] = [];
1681
1774
  const modelAttempts: ModelAttempt[] = [];
1682
1775
  const aggregateUsage = emptyUsage();
1683
1776
  const attemptNotes: string[] = [];
1777
+ const launchWarnings = { emitted: false };
1684
1778
  let totalToolCount = 0;
1685
1779
  let totalDurationMs = 0;
1686
1780
 
@@ -1755,6 +1849,7 @@ async function runSyncCompletionInner(
1755
1849
  modelAttemptsLoop: for (let modelIndex = 0; modelIndex < modelsToTry.length; modelIndex++) {
1756
1850
  const candidate = modelsToTry[modelIndex];
1757
1851
  for (let startupAttemptIndex = 0; ; startupAttemptIndex++) {
1852
+ const verifyModel = Boolean(candidate) && !(options.modelOverrideFromParent && modelIndex === 0);
1758
1853
  const outputSnapshot = captureSingleOutputSnapshot(options.outputPath);
1759
1854
  const result = await runSingleAttempt(runtimeCwd, agent, taskWithAcceptance, candidate, attemptOptions, {
1760
1855
  sessionEnabled,
@@ -1772,6 +1867,8 @@ async function runSyncCompletionInner(
1772
1867
  originalTask: task,
1773
1868
  taskDelivery: taskDeliveryOverride,
1774
1869
  orcaProgressTab,
1870
+ launchWarnings,
1871
+ verifyModel,
1775
1872
  });
1776
1873
  lastResult = result;
1777
1874
  if (startupAttemptIndex === 0) {
@@ -1864,6 +1961,11 @@ async function runSyncCompletionInner(
1864
1961
  }
1865
1962
  const retryableModelFailure = isRetryableModelFailure(result.error);
1866
1963
  if (retryableModelFailure) recordRetryableModelFailure(result.model ?? candidate, result.error);
1964
+ if (isContextOverflow(result.error)) {
1965
+ result.contextOverflow = true;
1966
+ attemptNotes.push(`[fallback] ${attempt.model} failed: context overflow — the input exceeds this model's context window. Reduce the task input or use a model with a larger context window.`);
1967
+ break modelAttemptsLoop;
1968
+ }
1867
1969
  if (!retryableModelFailure || modelIndex === modelsToTry.length - 1) break modelAttemptsLoop;
1868
1970
  attemptNotes.push(formatModelAttemptNote(attempt, modelsToTry[modelIndex + 1]));
1869
1971
  break;