pi-subagents 0.56.0 → 0.58.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/CHANGELOG.md +92 -0
  2. package/agents/claude-code-writer.md +15 -0
  3. package/agents/claude-code.md +15 -0
  4. package/agents/codex-exec-writer.md +15 -0
  5. package/agents/codex-exec.md +15 -0
  6. package/agents/cursor-agent-writer.md +14 -0
  7. package/agents/cursor-agent.md +14 -0
  8. package/docs/agents.md +124 -21
  9. package/docs/configuration.md +37 -0
  10. package/docs/extension-api.md +41 -2
  11. package/docs/models.md +3 -3
  12. package/docs/observability.md +8 -7
  13. package/docs/tool-reference.md +14 -3
  14. package/docs/workflows.md +44 -0
  15. package/package.json +1 -1
  16. package/skills/pi-subagents/SKILL.md +2 -0
  17. package/skills/pi-subagents/references/execution-controls.md +2 -2
  18. package/skills/pi-subagents/references/management-authoring-rpc.md +1 -0
  19. package/skills/pi-subagents/references/prompting-and-roles.md +2 -2
  20. package/src/agents/agent-management.ts +36 -5
  21. package/src/agents/agent-refinements.ts +4 -4
  22. package/src/agents/agent-serializer.ts +5 -0
  23. package/src/agents/agents.ts +257 -51
  24. package/src/agents/builtin-names.ts +6 -0
  25. package/src/agents/runtime-agent-events.ts +70 -0
  26. package/src/agents/runtime-agent-registry.ts +18 -4
  27. package/src/api/agents.ts +10 -5
  28. package/src/api/preflight.ts +28 -3
  29. package/src/extension/config.ts +70 -0
  30. package/src/extension/doctor.ts +3 -3
  31. package/src/extension/index.ts +33 -8
  32. package/src/extension/public-execution.ts +29 -13
  33. package/src/extension/rpc.ts +55 -19
  34. package/src/extension/schemas.ts +6 -5
  35. package/src/extension/tool-description.ts +18 -12
  36. package/src/inspectors/herdr/actions.ts +2 -1
  37. package/src/inspectors/herdr/inspector-runner.ts +2 -10
  38. package/src/inspectors/herdr/session-roots-codec.ts +42 -0
  39. package/src/integrations/herdr-status.ts +51 -3
  40. package/src/runs/background/active-async-capacity.ts +77 -10
  41. package/src/runs/background/async-execution.ts +127 -19
  42. package/src/runs/background/async-job-tracker.ts +5 -0
  43. package/src/runs/background/async-resume.ts +6 -2
  44. package/src/runs/background/async-retention.ts +20 -3
  45. package/src/runs/background/async-status.ts +7 -0
  46. package/src/runs/background/chain-append.ts +2 -0
  47. package/src/runs/background/chain-root-attachment.ts +15 -1
  48. package/src/runs/background/fleet-view.ts +16 -10
  49. package/src/runs/background/inspect-rpc.ts +8 -8
  50. package/src/runs/background/notify.ts +26 -3
  51. package/src/runs/background/result-delivery-ownership.ts +45 -0
  52. package/src/runs/background/result-files.ts +27 -14
  53. package/src/runs/background/result-watcher.ts +36 -15
  54. package/src/runs/background/run-status.ts +33 -6
  55. package/src/runs/background/scheduled-runs.ts +7 -1
  56. package/src/runs/background/subagent-runner.ts +239 -56
  57. package/src/runs/background/wait-completions.ts +4 -0
  58. package/src/runs/foreground/execution.ts +92 -11
  59. package/src/runs/foreground/foreground-control.ts +6 -0
  60. package/src/runs/foreground/foreground-history.ts +22 -1
  61. package/src/runs/foreground/subagent-executor.ts +322 -102
  62. package/src/runs/foreground/workflow-detach-reconcile.ts +144 -18
  63. package/src/runs/shared/child-protocol.ts +21 -7
  64. package/src/runs/shared/claude-code-adapter.ts +129 -0
  65. package/src/runs/shared/codex-exec-adapter.ts +129 -0
  66. package/src/runs/shared/completion-guard.ts +4 -3
  67. package/src/runs/shared/cursor-agent-adapter.ts +114 -0
  68. package/src/runs/shared/dynamic-fanout.ts +3 -3
  69. package/src/runs/shared/external-cli-contract.ts +167 -0
  70. package/src/runs/shared/external-cli-preflight.ts +122 -0
  71. package/src/runs/shared/external-cli-runner.ts +348 -55
  72. package/src/runs/shared/fast-mode-extension.ts +5 -5
  73. package/src/runs/shared/launch-cwd.ts +16 -0
  74. package/src/runs/shared/long-running-guard.ts +2 -1
  75. package/src/runs/shared/mcp-config-sources.ts +386 -0
  76. package/src/runs/shared/mcp-direct-tool-allowlist.ts +155 -42
  77. package/src/runs/shared/model-exclusions.ts +69 -7
  78. package/src/runs/shared/model-fallback.ts +39 -5
  79. package/src/runs/shared/mutation-evidence.ts +7 -2
  80. package/src/runs/shared/nested-events.ts +3 -1
  81. package/src/runs/shared/nested-render.ts +2 -2
  82. package/src/runs/shared/parallel-utils.ts +8 -1
  83. package/src/runs/shared/pi-args.ts +61 -6
  84. package/src/runs/shared/process-signal.ts +13 -0
  85. package/src/runs/shared/run-history.ts +21 -1
  86. package/src/runs/shared/single-output.ts +17 -0
  87. package/src/runs/shared/subagent-prompt-runtime.ts +85 -9
  88. package/src/shared/fork-context.ts +21 -0
  89. package/src/shared/formatters.ts +13 -1
  90. package/src/shared/launch-contract.ts +4 -0
  91. package/src/shared/pruned-fork.ts +450 -0
  92. package/src/shared/session-file-trust.ts +19 -0
  93. package/src/shared/session-tokens.ts +14 -3
  94. package/src/shared/settings.ts +10 -2
  95. package/src/shared/shortcuts.ts +17 -0
  96. package/src/shared/types.ts +160 -10
  97. package/src/shared/utils.ts +6 -29
  98. package/src/shared/workflow-child-permit.ts +116 -0
  99. package/src/slash/delegation-adapters.ts +0 -1
  100. package/src/slash/slash-commands.ts +8 -6
  101. package/src/slash/subagents-admin.ts +3 -0
  102. package/src/tui/fleet-status.ts +27 -10
  103. package/src/tui/fleet-transcript.ts +11 -5
  104. package/src/tui/fleet.ts +28 -13
  105. package/src/tui/render.ts +55 -21
  106. package/src/workflows/scripted-workflow.ts +299 -31
  107. package/src/workflows/workflow-child-summary.ts +117 -0
  108. package/src/workflows/workflow-receipt.ts +155 -5
@@ -14,6 +14,7 @@ import { createChildTranscriptWriter, type ChildTranscriptWriter } from "../../s
14
14
  import { closeSteerInbox, consumeInterruptRequest, consumeSteerRequests, deliverInterruptRequest, deliverStopRequest, deliverTimeoutRequest, enqueueStepSteer, steerAcksDir, steerCapabilityPath, stepSteerInboxDir, watchAsyncControlInbox, type SteerAck, type SteerCapability, type SteerRequest, type StopRequest } from "./control-channel.ts";
15
15
  import { appendJsonl as appendRawJsonl, formatOutputArtifactContent, getArtifactPaths, writeArtifact, writeMetadata } from "../../shared/artifacts.ts";
16
16
  import { PI_CODING_AGENT_PACKAGE, getPiSpawnCommand, resolveInstalledPiPackageRoot } from "../shared/pi-spawn.ts";
17
+ import { preflightLaunchCwd } from "../shared/launch-cwd.ts";
17
18
  import { captureSingleOutputSnapshot, extractChildWrittenOutput, finalizeSingleOutput, formatSavedOutputReference, injectOutputPathSystemPrompt, injectSingleOutputInstruction, resolveSingleOutput, type SingleOutputSnapshot } from "../shared/single-output.ts";
18
19
  import {
19
20
  type ActivityState,
@@ -49,6 +50,7 @@ import {
49
50
  type SteeringTargetState,
50
51
  type SteeringTargetStatus,
51
52
  type SubagentChildStatusEvent,
53
+ type SettlementDiagnostic,
52
54
  DEFAULT_MAX_OUTPUT,
53
55
  type MaxOutputConfig,
54
56
  SUBAGENT_LIFECYCLE_ARTIFACT_VERSION,
@@ -81,13 +83,13 @@ import { applyThinkingSuffix, buildPiArgs, cleanupTempDir, projectLaunchResolved
81
83
  import { readRuntimeAcknowledgedExtensions } from "../shared/runtime-acknowledged-extensions.ts";
82
84
  import { outputEntryFromAsyncResult, resolveOutputReferences } from "../shared/chain-outputs.ts";
83
85
  import { createStructuredOutputRuntime, MISSING_STRUCTURED_OUTPUT_CALL_ERROR, readStructuredOutput, readStructuredOutputAcceptanceReport } from "../shared/structured-output.ts";
84
- import { formatProcessSignalError, isUnexplainedProcessSignal } from "../shared/process-signal.ts";
86
+ import { formatMidToolExitError, formatProcessSignalError, isOrdinaryToolForMidToolExit, isUnexplainedProcessSignal } from "../shared/process-signal.ts";
85
87
  import { readChildToolDiagnosticError } from "../shared/tool-availability.ts";
86
88
  import { buildTimeoutRecoverySummary, collectTrackedMutationEvidence, snapshotTrackedMutations } from "../shared/mutation-evidence.ts";
87
89
  import { collectDynamicResults, DynamicFanoutError, materializeDynamicParallelStep, validateDynamicCollection } from "../shared/dynamic-fanout.ts";
88
90
  import { claimRunFanoutBatch, getRunFanoutBudgetSnapshot } from "../shared/run-fanout-budget.ts";
89
91
  import { nestedSummaryFromAsyncStatus, projectNestedEvents, resolveNestedAsyncDir, writeNestedEvent } from "../shared/nested-events.ts";
90
- import { formatModelAttemptNote, formatSubagentModelVerificationError, isContextOverflow, isRetryableModelFailure, recordRetryableModelFailure } from "../shared/model-fallback.ts";
92
+ import { formatModelAttemptNote, formatSubagentModelVerificationError, isContextOverflow, isRetryableModelFailureAttempt, recordRetryableModelFailure } from "../shared/model-fallback.ts";
91
93
  import {
92
94
  SUBAGENT_STARTUP_RETRY_DELAYS_MS,
93
95
  formatSubagentExtensionConflictError,
@@ -101,7 +103,7 @@ import { createOwnedProcessTreeController, type OwnedProcessTreeController } fro
101
103
  import { createSteeringStatus, recordSteeringRequest, steeringStatus, terminalSteeringNoticeState, updateSteeringTarget } from "./steering.ts";
102
104
  import { attachPostExitStdioGuard, trySignalChild } from "../../shared/post-exit-stdio-guard.ts";
103
105
  import { PROMPT_REDACTED, detectSubagentError, extractTextFromContent, extractToolArgsPreview, getFinalOutput, hasEmptyTerminalAssistantResponse, readStatus } from "../../shared/utils.ts";
104
- import { evaluateCompletionMutationGuard, validateImplementationToolContract } from "../shared/completion-guard.ts";
106
+ import { evaluateCompletionMutationGuard, expectsImplementationMutation, hasMutationToolCapability, validateImplementationToolContract } from "../shared/completion-guard.ts";
105
107
  import {
106
108
  createMutatingFailureState,
107
109
  didMutatingToolFail,
@@ -124,7 +126,7 @@ import {
124
126
  formatWorktreeTaskCwdConflict,
125
127
  type WorktreeSetup,
126
128
  } from "../shared/worktree.ts";
127
- import { resolveEffectiveThinking } from "../../shared/model-info.ts";
129
+ import { findModelInfo, resolveEffectiveThinking } from "../../shared/model-info.ts";
128
130
  import { assertThinkingWithinCeiling, decodeThinkingCeiling, SUBAGENT_THINKING_CEILING_ENV } from "../../shared/thinking-ceiling.ts";
129
131
  import { launchBindingDigest } from "../../shared/launch-contract.ts";
130
132
  import { writeInitialProgressFile } from "../../shared/settings.ts";
@@ -141,9 +143,13 @@ import { effectiveToolTimeoutMs, formatToolTimeoutMessage, toolTimeoutCallKey }
141
143
  import { usageBudgetExceededMessage, usageBudgetState } from "../shared/usage-budget.ts";
142
144
  import { formatParallelHandoffError, formatParallelHandoffReference, parallelHandoffPath, writeParallelHandoffGroup, writePendingParallelHandoff } from "../shared/parallel-handoff.ts";
143
145
  import { resolveWatchdogConfig } from "../../watchdog/settings.ts";
144
- import { createBoundedByteTail, createBoundedLineReader, formatProtocolOutputLimit, MAX_CHILD_STDERR_BYTES, PI_AGGREGATE_EVENT_PROJECTOR, projectChildLifecycle, type ChildLifecycleAction, type ProtocolOutputLimit } from "../shared/child-protocol.ts";
146
+ import { createBoundedByteTail, createBoundedLineReader, formatProtocolOutputLimit, MAX_CHILD_STDERR_BYTES, PI_AGGREGATE_EVENT_PROJECTOR, projectChildLifecycle, type ChildLifecycleAction, type ChildLifecycleState, type ProtocolOutputLimit } from "../shared/child-protocol.ts";
145
147
  import { acquireSessionLease, type SessionLeaseRequest } from "../shared/session-lease.ts";
146
148
  import { buildExternalCliPrompt, runExternalCli } from "../shared/external-cli-runner.ts";
149
+ import { resolveClaudeCodeLaunch } from "../shared/claude-code-adapter.ts";
150
+ import { resolveCodexExecLaunch } from "../shared/codex-exec-adapter.ts";
151
+ import { resolveCursorAgentLaunch } from "../shared/cursor-agent-adapter.ts";
152
+ import { resolveExternalCliRunnerStatus } from "../shared/external-cli-contract.ts";
147
153
  import { runExternalJob } from "../shared/external-job-runner.ts";
148
154
  import { createOrcaProgressTab, type OrcaProgressTab } from "../shared/orca-progress-tabs.ts";
149
155
  import { decodeSubagentCapabilityCeiling, SUBAGENT_CAPABILITY_CEILING_ENV, type ResolvedSubagentCapabilityCeiling } from "../shared/capability-ceiling.ts";
@@ -206,6 +212,7 @@ interface SubagentRunConfig {
206
212
  launchResolvedExtensions?: LaunchResolvedChildExtensionsV1;
207
213
  runtimeAcknowledgedExtensions?: RuntimeAcknowledgedChildExtensionsV1;
208
214
  runnerProcessInstanceId?: string;
215
+ launchBarrierToken?: string;
209
216
  parentWorkflowRunId?: string;
210
217
  workflowKey?: string;
211
218
  }
@@ -496,6 +503,7 @@ interface ChildUsage {
496
503
  output?: number;
497
504
  outputTokens?: number;
498
505
  cacheRead?: number;
506
+ cacheReadTokens?: number;
499
507
  cacheWrite?: number;
500
508
  cost?: { total?: number };
501
509
  }
@@ -537,6 +545,7 @@ interface RunPiStreamingResult {
537
545
  observedMutationAttempt?: boolean;
538
546
  structuredOutputToolInvoked?: boolean;
539
547
  structuredOutputMessageStartIndex?: number;
548
+ structuredOutput?: unknown;
540
549
  watchdog?: ChildWatchdogStateSnapshot;
541
550
  runtimeAcknowledgedExtensions?: RuntimeAcknowledgedChildExtensionsV1;
542
551
  processInstanceId: string;
@@ -546,6 +555,28 @@ interface RunPiStreamingResult {
546
555
  currentTool?: string;
547
556
  currentToolArgs?: string;
548
557
  currentPath?: string;
558
+ afterCompactionSettlement?: boolean;
559
+ effects?: import("../../shared/types.ts").EffectsProjection;
560
+ }
561
+
562
+ const MAX_CHILD_FAILURE_DIAGNOSTIC_CHARS = 8_192;
563
+
564
+ function formatChildFailureDiagnostic(input: {
565
+ error: string | undefined;
566
+ afterCompactionSettlement?: boolean;
567
+ missingFileOnlyOutput?: string;
568
+ missingStructuredOutput?: string;
569
+ }): string | undefined {
570
+ const notes = [
571
+ input.afterCompactionSettlement ? "Child failure followed session compaction and agent settlement." : undefined,
572
+ input.missingFileOnlyOutput ? `Required file-only output was not produced: ${input.missingFileOnlyOutput.slice(0, 2_048)}` : undefined,
573
+ input.missingStructuredOutput ? `Required structured output was not produced: ${input.missingStructuredOutput.slice(0, 2_048)}` : undefined,
574
+ ].filter((note): note is string => Boolean(note));
575
+ if (notes.length === 0) return input.error;
576
+ const context = notes.join("\n");
577
+ const errorLimit = MAX_CHILD_FAILURE_DIAGNOSTIC_CHARS - (context ? context.length + 1 : 0);
578
+ const baseError = input.error || "Subagent failed.";
579
+ return `${baseError.slice(0, Math.max(0, errorLimit))}${context ? `\n${context}` : ""}`;
549
580
  }
550
581
 
551
582
  function runPiStreaming(
@@ -571,6 +602,7 @@ function runPiStreaming(
571
602
  orcaProgressTab?: OrcaProgressTab,
572
603
  expectedModelForVerification?: string,
573
604
  modelVerificationRegistry?: Array<{ provider: string; id: string; fullId: string }>,
605
+ mutationTools?: readonly string[],
574
606
  ): Promise<RunPiStreamingResult> {
575
607
  return new Promise((resolve) => {
576
608
  const startedAt = Date.now();
@@ -620,8 +652,50 @@ function runPiStreaming(
620
652
  let currentToolArgs: string | undefined;
621
653
  let currentPath: string | undefined;
622
654
  let toolCount = 0;
655
+ type ActiveToolCall = { key: string; tool: string; args?: string; path?: string };
656
+ let activeToolSequence = 0;
657
+ const activeToolCalls = new Map<string, ActiveToolCall>();
658
+ const activeToolKeysByName = new Map<string, string[]>();
659
+ const refreshCurrentTool = (): void => {
660
+ const active = [...activeToolCalls.values()].at(-1);
661
+ currentTool = active?.tool;
662
+ currentToolArgs = active?.args;
663
+ currentPath = active?.path;
664
+ };
665
+ const recordActiveToolCall = (event: { toolCallId?: unknown; toolName: string; args?: Record<string, unknown> }): void => {
666
+ const key = toolTimeoutCallKey(event, ++activeToolSequence);
667
+ const active = omitUndefinedProperties({
668
+ key,
669
+ tool: event.toolName,
670
+ args: extractToolArgsPreview(event.args ?? {}),
671
+ path: resolveCurrentPath(event.toolName, event.args),
672
+ });
673
+ activeToolCalls.set(key, active);
674
+ const keys = activeToolKeysByName.get(active.tool) ?? [];
675
+ keys.push(key);
676
+ activeToolKeysByName.set(active.tool, keys);
677
+ refreshCurrentTool();
678
+ };
679
+ const removeActiveToolCall = (event: { toolCallId?: unknown; toolName?: unknown }): void => {
680
+ const key = typeof event.toolCallId === "string" && event.toolCallId.length > 0
681
+ ? `id:${event.toolCallId}`
682
+ : typeof event.toolName === "string"
683
+ ? activeToolKeysByName.get(event.toolName)?.[0]
684
+ : activeToolCalls.size === 1
685
+ ? [...activeToolCalls.keys()][0]
686
+ : undefined;
687
+ if (!key) return;
688
+ const active = activeToolCalls.get(key);
689
+ if (!active) return;
690
+ activeToolCalls.delete(key);
691
+ const keys = activeToolKeysByName.get(active.tool)?.filter((candidate) => candidate !== key) ?? [];
692
+ if (keys.length > 0) activeToolKeysByName.set(active.tool, keys);
693
+ else activeToolKeysByName.delete(active.tool);
694
+ refreshCurrentTool();
695
+ };
623
696
  const childWatchdogConfig = decodeChildWatchdogConfig(env?.[CHILD_WATCHDOG_CONFIG_ENV]);
624
697
  let childWatchdogState: ChildWatchdogStateSnapshot | undefined;
698
+ const childLifecycleState: ChildLifecycleState = { compactionRetryActive: false };
625
699
  let applyChildLifecycle = (_action: ChildLifecycleAction): void => {};
626
700
  const updateChildWatchdogState = (snapshot: ChildWatchdogStateSnapshot): void => {
627
701
  childWatchdogState = snapshot;
@@ -672,8 +746,10 @@ function runPiStreaming(
672
746
 
673
747
  appendChildEvent(event as unknown as Record<string, unknown>);
674
748
  transcriptWriter?.writeChildEvent(event);
675
- if (event.type === "agent_settled") agentSettledReceived = true;
676
- applyChildLifecycle(projectChildLifecycle(event));
749
+ if (event.type === "compaction_start") compactionStartedReceived = true;
750
+ const lifecycleAction = projectChildLifecycle(event, false, childLifecycleState);
751
+ if (event.type === "agent_settled" && lifecycleAction === "start-drain") agentSettledReceived = true;
752
+ applyChildLifecycle(lifecycleAction);
677
753
 
678
754
  if (isChildWatchdogStatusEvent(event)) {
679
755
  if (!childWatchdogConfig) return;
@@ -710,29 +786,32 @@ function runPiStreaming(
710
786
 
711
787
  if (event.type === "tool_execution_end") {
712
788
  clearActiveToolTimeout(event);
713
- currentTool = undefined;
714
- currentToolArgs = undefined;
715
- currentPath = undefined;
789
+ removeActiveToolCall(event);
716
790
  return;
717
791
  }
718
792
 
719
793
  if (event.type === "tool_execution_start" && event.toolName) {
720
794
  toolCount += 1;
721
795
  armToolTimeout({ toolCallId: (event as { toolCallId?: unknown }).toolCallId, toolName: event.toolName });
722
- currentTool = event.toolName;
723
- currentToolArgs = extractToolArgsPreview(event.args ?? {});
724
- currentPath = resolveCurrentPath(event.toolName, event.args);
796
+ recordActiveToolCall({ toolCallId: (event as { toolCallId?: unknown }).toolCallId, toolName: event.toolName, args: event.args });
725
797
  if (event.toolName === "structured_output") {
726
798
  structuredOutputToolInvoked = true;
727
799
  structuredOutputMessageStartIndex = messages.length;
728
800
  }
729
- observedMutationAttempt = observedMutationAttempt || isMutatingTool(event.toolName, event.args);
801
+ observedMutationAttempt = observedMutationAttempt || isMutatingTool(event.toolName, event.args, mutationTools);
730
802
  const toolArgs = extractToolArgsPreview(event.args ?? {});
731
803
  writeOutputLine(toolArgs ? `${event.toolName}: ${toolArgs}` : event.toolName);
732
804
  return;
733
805
  }
734
806
 
735
807
  if ((event.type === "message_end" || event.type === "tool_result_end") && event.message) {
808
+ if (event.type === "tool_result_end") {
809
+ clearActiveToolTimeout(event);
810
+ removeActiveToolCall({
811
+ toolCallId: (event.message as { toolCallId?: unknown }).toolCallId ?? (event as { toolCallId?: unknown }).toolCallId,
812
+ toolName: (event.message as { toolName?: unknown }).toolName ?? event.toolName,
813
+ });
814
+ }
736
815
  messages.push(event.message);
737
816
  const text = extractTextFromContent(event.message.content);
738
817
  if (text) writeOutputText(text);
@@ -759,7 +838,11 @@ function runPiStreaming(
759
838
  if (isTerminalAssistantStop(event.message)) {
760
839
  if (!event.message.errorMessage && extractTextFromContent(event.message.content).trim()) assistantError = undefined;
761
840
  cleanTerminalAssistantStopReceived ||= !event.message.errorMessage;
762
- applyChildLifecycle(projectChildLifecycle(event, true));
841
+ clearAllToolTimeouts();
842
+ activeToolCalls.clear();
843
+ activeToolKeysByName.clear();
844
+ refreshCurrentTool();
845
+ applyChildLifecycle(projectChildLifecycle(event, true, childLifecycleState));
763
846
  }
764
847
  }
765
848
  };
@@ -772,6 +855,7 @@ function runPiStreaming(
772
855
  let forcedTerminationSignal = false;
773
856
  let cleanTerminalAssistantStopReceived = false;
774
857
  let agentSettledReceived = false;
858
+ let compactionStartedReceived = false;
775
859
  let finalDrainTimer: NodeJS.Timeout | undefined;
776
860
  let finalHardKillTimer: NodeJS.Timeout | undefined;
777
861
  let watchdogTailTimer: NodeJS.Timeout | undefined;
@@ -1047,6 +1131,7 @@ function runPiStreaming(
1047
1131
  currentTool,
1048
1132
  currentToolArgs,
1049
1133
  currentPath,
1134
+ afterCompactionSettlement: (compactionStartedReceived && agentSettledReceived) || undefined,
1050
1135
  }));
1051
1136
  });
1052
1137
 
@@ -1219,7 +1304,7 @@ interface SingleStepContext {
1219
1304
  nestedRoute?: NestedRouteInfo;
1220
1305
  capabilityCeiling?: ResolvedSubagentCapabilityCeiling;
1221
1306
  runFanoutBudget?: RunFanoutBudgetDescriptor;
1222
- onAttemptStart?: (attempt: { model?: string; thinking?: string }) => void;
1307
+ onAttemptStart?: (attempt: { model?: string; thinking?: string; contextLimit?: number }) => void;
1223
1308
  onChildEvent?: (event: ChildEvent) => void;
1224
1309
  onWriterProcess?: (writer: { state: "none" | "spawning" } | { state: "running"; pid: number }) => void;
1225
1310
  onExternalProcess?: (process: ExternalProcessStatus) => void;
@@ -1321,6 +1406,8 @@ async function runSingleStepInner(
1321
1406
  model: step.model,
1322
1407
  modelCandidates: step.modelCandidates,
1323
1408
  mcpDirectTools: step.mcpDirectTools,
1409
+ mcpConfig: step.mcpConfig,
1410
+ runtimeServerNames: step.runtimeServerNames,
1324
1411
  cwd: step.cwd ?? ctx.cwd,
1325
1412
  requireReadTool: Boolean(step.skills?.length),
1326
1413
  structuredOutput: Boolean(effectiveStructuredOutput),
@@ -1381,21 +1468,29 @@ async function runSingleStepInner(
1381
1468
  transcriptWriter?.writeInitialUserMessage(`${PROMPT_REDACTED}; live Prompt Audit only.`);
1382
1469
 
1383
1470
  if (step.runner?.type === "external-cli") {
1384
- const runner: ExternalCliRunnerStatus = {
1385
- type: "external-cli",
1386
- command: step.runner.command,
1387
- args: step.runner.args ?? [],
1388
- promptDelivery: step.runner.promptDelivery ?? "stdin",
1389
- capabilities: { stop: true, steer: false, resume: false, structuredOutput: false, toolEvents: false },
1390
- };
1471
+ const externalCwd = step.cwd ?? ctx.cwd;
1472
+ const adapterLaunch = step.runner.adapter === "codex-exec" || step.runner.adapter === "codex-exec-writer"
1473
+ ? resolveCodexExecLaunch({ adapter: step.runner.adapter, command: step.runner.command, asyncDir: path.dirname(ctx.outputFile), stepIndex: ctx.flatIndex })
1474
+ : step.runner.adapter === "claude-code" || step.runner.adapter === "claude-code-writer"
1475
+ ? resolveClaudeCodeLaunch({ adapter: step.runner.adapter, command: step.runner.command })
1476
+ : step.runner.adapter === "cursor-agent" || step.runner.adapter === "cursor-agent-writer"
1477
+ ? resolveCursorAgentLaunch({ adapter: step.runner.adapter, command: step.runner.command, cwd: externalCwd, asyncDir: path.dirname(ctx.outputFile), stepIndex: ctx.flatIndex })
1478
+ : undefined;
1479
+ const runner = resolveExternalCliRunnerStatus({ ...step.runner, ...(adapterLaunch ? { args: adapterLaunch.args } : {}) });
1391
1480
  const outputSnapshot = captureSingleOutputSnapshot(step.outputPath);
1392
1481
  const external = await runExternalCli(omitUndefinedProperties({
1393
- command: runner.command,
1394
- args: runner.args,
1395
- cwd: step.cwd ?? ctx.cwd,
1482
+ command: adapterLaunch?.command ?? runner.command,
1483
+ args: adapterLaunch?.args ?? runner.args,
1484
+ cwd: externalCwd,
1396
1485
  prompt: buildExternalCliPrompt(step.systemPrompt ?? "", task),
1397
1486
  asyncDir: path.dirname(ctx.outputFile),
1398
1487
  stepIndex: ctx.flatIndex,
1488
+ environment: adapterLaunch?.environment,
1489
+ preflight: adapterLaunch?.preflight,
1490
+ parser: adapterLaunch?.parser,
1491
+ finalOutputPath: adapterLaunch?.finalOutputPath,
1492
+ promptFilePath: adapterLaunch?.promptFilePath,
1493
+ temporaryDirectories: adapterLaunch?.temporaryDirectories,
1399
1494
  registerTimeout: ctx.registerTimeout,
1400
1495
  registerStop: ctx.registerStop,
1401
1496
  timeoutMessage: ctx.timeoutMessage,
@@ -1508,6 +1603,10 @@ async function runSingleStepInner(
1508
1603
  });
1509
1604
  }
1510
1605
 
1606
+ const effectiveCwd = step.cwd ?? ctx.cwd;
1607
+ const cwdError = preflightLaunchCwd(step.requestedCwd ?? effectiveCwd, effectiveCwd);
1608
+ if (cwdError) return { agent: step.agent, output: cwdError, error: cwdError, exitCode: 1, context: step.context };
1609
+
1511
1610
  const candidates = step.modelCandidates !== undefined
1512
1611
  ? step.modelCandidates.length > 0 ? step.modelCandidates : [undefined]
1513
1612
  : step.model
@@ -1550,7 +1649,11 @@ async function runSingleStepInner(
1550
1649
  const message = error instanceof Error ? error.message : String(error);
1551
1650
  return omitUndefinedProperties({ agent: step.agent, output: message, error: message, exitCode: 1, context: step.context, thinkingCeiling: step.thinkingCeiling });
1552
1651
  }
1553
- ctx.onAttemptStart?.(omitUndefinedProperties({ model: candidate, thinking: resolveEffectiveThinking(candidate, step.thinking) }));
1652
+ ctx.onAttemptStart?.(omitUndefinedProperties({
1653
+ model: candidate,
1654
+ thinking: resolveEffectiveThinking(candidate, step.thinking),
1655
+ contextLimit: findModelInfo(candidate, step.modelVerificationRegistry)?.contextWindow,
1656
+ }));
1554
1657
  const outputSnapshot = captureSingleOutputSnapshot(step.outputPath);
1555
1658
  if (effectiveStructuredOutput) {
1556
1659
  try {
@@ -1579,6 +1682,7 @@ async function runSingleStepInner(
1579
1682
  sessionFile: step.sessionFile,
1580
1683
  model: candidate,
1581
1684
  inheritProjectContext: step.inheritProjectContext,
1685
+ inheritGlobalContext: step.inheritGlobalContext,
1582
1686
  inheritSkills: step.inheritSkills,
1583
1687
  requireReadTool: Boolean(step.skills?.length),
1584
1688
  tools: step.tools,
@@ -1589,6 +1693,8 @@ async function runSingleStepInner(
1589
1693
  systemPrompt: appendTurnBudgetSystemPrompt(step.systemPrompt ?? "", ctx.turnBudget),
1590
1694
  systemPromptMode: step.systemPromptMode,
1591
1695
  mcpDirectTools: step.mcpDirectTools,
1696
+ mcpConfig: step.mcpConfig,
1697
+ runtimeServerNames: step.runtimeServerNames,
1592
1698
  capabilityCeiling: step.capabilityCeiling ?? ctx.capabilityCeiling,
1593
1699
  cwd: step.cwd ?? ctx.cwd,
1594
1700
  promptFileStem: step.agent,
@@ -1632,6 +1738,8 @@ async function runSingleStepInner(
1632
1738
  model: step.model,
1633
1739
  modelCandidates: step.modelCandidates,
1634
1740
  mcpDirectTools: step.mcpDirectTools,
1741
+ mcpConfig: step.mcpConfig,
1742
+ runtimeServerNames: step.runtimeServerNames,
1635
1743
  cwd: step.cwd ?? ctx.cwd,
1636
1744
  requireReadTool: Boolean(step.skills?.length),
1637
1745
  structuredOutput: Boolean(effectiveStructuredOutput),
@@ -1651,6 +1759,7 @@ async function runSingleStepInner(
1651
1759
  systemPrompt: appendTurnBudgetSystemPrompt(step.systemPrompt ?? "", ctx.turnBudget),
1652
1760
  systemPromptMode: step.systemPromptMode,
1653
1761
  inheritProjectContext: step.inheritProjectContext,
1762
+ inheritGlobalContext: step.inheritGlobalContext,
1654
1763
  inheritSkills: step.inheritSkills,
1655
1764
  skills: step.skills,
1656
1765
  tools: toolPlan.effectiveToolAllowlist,
@@ -1687,6 +1796,7 @@ async function runSingleStepInner(
1687
1796
  ctx.orcaProgressTab,
1688
1797
  expectedModelForVerification,
1689
1798
  step.modelVerificationRegistry,
1799
+ step.mutationTools,
1690
1800
  );
1691
1801
  if (run.processCloseObservedAt !== undefined) {
1692
1802
  writerProcesses.push({
@@ -1723,11 +1833,25 @@ async function runSingleStepInner(
1723
1833
  : undefined;
1724
1834
  const runtimeAcknowledgedExtensions = readRuntimeAcknowledgedExtensions(runtimeAcknowledgedExtensionsPath);
1725
1835
  cleanupTempDir(tempDir);
1836
+ const midToolExitError = run.currentTool
1837
+ && isOrdinaryToolForMidToolExit(run.currentTool)
1838
+ && !run.interrupted
1839
+ && !run.timedOut
1840
+ && !run.stopped
1841
+ && !run.turnBudgetExceeded
1842
+ && !run.protocolError
1843
+ && !toolAvailabilityError
1844
+ ? formatMidToolExitError({
1845
+ toolName: run.currentTool,
1846
+ exitCode: run.exitCode,
1847
+ processSignal: run.processSignal,
1848
+ })
1849
+ : undefined;
1726
1850
 
1727
1851
  let structuredOutput: unknown;
1728
1852
  let structuredError: string | undefined;
1729
1853
  let validatedStructuredOutput = false;
1730
- if (effectiveStructuredOutput && run.exitCode === 0 && !run.error && !toolAvailabilityError) {
1854
+ if (effectiveStructuredOutput && run.exitCode === 0 && !run.error && !toolAvailabilityError && !midToolExitError) {
1731
1855
  if (!run.structuredOutputToolInvoked) {
1732
1856
  structuredError = MISSING_STRUCTURED_OUTPUT_CALL_ERROR;
1733
1857
  } else {
@@ -1749,7 +1873,7 @@ async function runSingleStepInner(
1749
1873
  const errorMessages = validatedStructuredOutput
1750
1874
  ? run.messages.slice(run.structuredOutputMessageStartIndex ?? run.messages.length)
1751
1875
  : run.messages;
1752
- const hiddenError = run.exitCode === 0 && !run.error && !toolAvailabilityError && !structuredError
1876
+ const hiddenError = run.exitCode === 0 && !run.error && !toolAvailabilityError && !structuredError && !midToolExitError
1753
1877
  ? detectSubagentError(errorMessages)
1754
1878
  : null;
1755
1879
  const emptyOutputError = run.exitCode === 0
@@ -1763,16 +1887,18 @@ async function runSingleStepInner(
1763
1887
  : undefined;
1764
1888
  const completionGuardEnabled = isAgentContractV1(step.agentContract) ? step.completionGuard === true : step.completionGuard !== false;
1765
1889
  const completionToolPlan = resolvedTaskToolPlan;
1890
+ const completionTools = completionToolPlan ? (completionToolPlan.explicitToolAllowlist ? completionToolPlan.effectiveToolAllowlist : undefined) : step.tools;
1766
1891
  const mutationEvidence = collectTrackedMutationEvidence(mutationSnapshot, step.cwd ?? ctx.cwd);
1767
1892
  finalMutationEvidence = mutationEvidence;
1768
1893
  const completionMutationEvidence = ctx.trackedMutationEvidenceForCompletionGuard === false ? undefined : mutationEvidence;
1769
- const completionGuard = run.exitCode === 0 && !run.error && !structuredError && !hiddenError?.hasError && !emptyOutputError && completionGuardEnabled
1894
+ const completionGuard = run.exitCode === 0 && !run.error && !structuredError && !hiddenError?.hasError && !midToolExitError && !emptyOutputError && completionGuardEnabled
1770
1895
  ? evaluateCompletionMutationGuard(omitUndefinedProperties({
1771
1896
  agent: step.agent,
1772
1897
  task: taskForCompletionGuard,
1773
1898
  messages: run.messages,
1774
- tools: completionToolPlan ? (completionToolPlan.explicitToolAllowlist ? completionToolPlan.effectiveToolAllowlist : undefined) : step.tools,
1899
+ tools: completionTools,
1775
1900
  mcpDirectTools: completionToolPlan?.effectiveMcpTools ?? step.mcpDirectTools,
1901
+ mutationTools: step.mutationTools,
1776
1902
  toolAvailabilityError,
1777
1903
  mutationEvidence: completionMutationEvidence,
1778
1904
  }))
@@ -1780,6 +1906,9 @@ async function runSingleStepInner(
1780
1906
  const mutationAttemptObserved = run.observedMutationAttempt === true || completionMutationEvidence?.attemptedMutation === true;
1781
1907
  const completionGuardTriggered = completionGuard?.triggered === true && !mutationAttemptObserved;
1782
1908
  const completionGuardBlocked = completionGuard?.blocked === true;
1909
+ const mutationExpected = completionGuard?.expectedMutation ?? (completionGuardEnabled
1910
+ && hasMutationToolCapability(completionTools, completionToolPlan?.effectiveMcpTools ?? step.mcpDirectTools)
1911
+ && expectsImplementationMutation(step.agent, taskForCompletionGuard));
1783
1912
  const fileMutationEffect = completionGuard
1784
1913
  ? {
1785
1914
  status: completionGuardBlocked ? "blocked" as const : completionGuard.expectedMutation ? completionGuardTriggered ? "missing" as const : "observed" as const : "not-applicable" as const,
@@ -1793,7 +1922,7 @@ async function runSingleStepInner(
1793
1922
  const completionGuardError = completionGuardTriggered && !isAgentContractV1(step.agentContract)
1794
1923
  ? "Subagent completed without making edits for an implementation task.\nIt appears to have returned planning or scratchpad output instead of applying changes."
1795
1924
  : undefined;
1796
- const effectiveExitCode = toolAvailabilityError || (completionGuardTriggered && !isAgentContractV1(step.agentContract)) || structuredError || emptyOutputError
1925
+ const effectiveExitCode = toolAvailabilityError || (completionGuardTriggered && !isAgentContractV1(step.agentContract)) || midToolExitError || structuredError || emptyOutputError
1797
1926
  ? 1
1798
1927
  : hiddenError?.hasError
1799
1928
  ? (hiddenError.exitCode ?? 1)
@@ -1810,6 +1939,7 @@ async function runSingleStepInner(
1810
1939
  const error = formatSubagentExtensionConflictError(
1811
1940
  toolAvailabilityError
1812
1941
  ?? completionGuardError
1942
+ ?? midToolExitError
1813
1943
  ?? structuredError
1814
1944
  ?? emptyOutputError
1815
1945
  ?? (hiddenError?.hasError
@@ -1839,7 +1969,24 @@ async function runSingleStepInner(
1839
1969
  toolBudgetBlocked = Boolean(blockedMessage);
1840
1970
  toolBudget = toolBudgetState(step.toolBudget, toolMessages.length, blockedMessage ? (blockedMessage as { toolName?: string }).toolName : undefined);
1841
1971
  }
1842
- finalResult = { ...run, exitCode: effectiveExitCode, model: candidate ?? run.model, error, structuredOutput, runtimeAcknowledgedExtensions, ...(step.agentContract ? { agentContract: step.agentContract } : {}), ...(fileMutationEffect ? { effects: { fileMutation: fileMutationEffect } } : {}) } as RunPiStreamingResult & { structuredOutput?: unknown; agentContract?: import("../../shared/types.ts").AgentContract; effects?: import("../../shared/types.ts").EffectsProjection };
1972
+ const requiredOutput = step.outputMode === "file-only" && step.outputPath
1973
+ ? { kind: "file-only" as const, path: step.outputPath, missing: !fs.existsSync(step.outputPath) }
1974
+ : effectiveStructuredOutput
1975
+ ? { kind: "structured" as const, path: effectiveStructuredOutput.outputPath, missing: !fs.existsSync(effectiveStructuredOutput.outputPath) }
1976
+ : undefined;
1977
+ const settlementDiagnostic: SettlementDiagnostic | undefined = effectiveExitCode !== 0 || completionGuardTriggered || completionGuardBlocked
1978
+ ? {
1979
+ finalTextPresent: Boolean(stripAcceptanceReport(run.finalOutput).trim()),
1980
+ mutation: {
1981
+ expected: mutationExpected,
1982
+ attempted: completionGuard?.attemptedMutation === true || mutationAttemptObserved,
1983
+ observed: mutationEvidence.attemptedMutation,
1984
+ },
1985
+ ...(requiredOutput ? { requiredOutput } : {}),
1986
+ afterCompactionSettlement: run.afterCompactionSettlement === true,
1987
+ }
1988
+ : undefined;
1989
+ finalResult = { ...run, exitCode: effectiveExitCode, model: candidate ?? run.model, error, structuredOutput, runtimeAcknowledgedExtensions, ...(step.agentContract ? { agentContract: step.agentContract } : {}), ...(fileMutationEffect || settlementDiagnostic ? { effects: { ...(fileMutationEffect ? { fileMutation: fileMutationEffect } : {}), ...(settlementDiagnostic ? { settlementDiagnostic } : {}) } } : {}) } as RunPiStreamingResult;
1843
1990
  if (run.turnBudgetExceeded) break modelAttemptsLoop;
1844
1991
  if (run.stopped || run.timedOut || ctx.timeoutSignal?.aborted || ctx.stopSignal?.aborted || ctx.skipAcceptance?.()) break modelAttemptsLoop;
1845
1992
  if (attempt.success || completionGuardTriggered) break modelAttemptsLoop;
@@ -1889,7 +2036,7 @@ async function runSingleStepInner(
1889
2036
  finalResult.finalOutput = startupError;
1890
2037
  break modelAttemptsLoop;
1891
2038
  }
1892
- const retryableModelFailure = isRetryableModelFailure(error);
2039
+ const retryableModelFailure = isRetryableModelFailureAttempt({ error, messages: run.messages, toolCount: run.toolCount });
1893
2040
  if (retryableModelFailure) recordRetryableModelFailure(candidate ?? run.model ?? step.model, error);
1894
2041
  if (isContextOverflow(error)) {
1895
2042
  contextOverflow = true;
@@ -1993,7 +2140,7 @@ async function runSingleStepInner(
1993
2140
  const acceptanceCanFailRun = acceptanceFailure && effectiveAcceptance?.explicit && (finalResult?.exitCode ?? 1) === 0 && !finalResult?.interrupted && !timedOutAfterAcceptance && !stoppedAfterAcceptance && !turnBudgetExceeded && !isAgentContractV1(step.agentContract);
1994
2141
  const effectiveFinalExitCode = timedOutAfterAcceptance || stoppedAfterAcceptance || turnBudgetExceeded ? 1 : acceptanceCanFailRun ? 1 : finalResult?.exitCode ?? 1;
1995
2142
  const intercomDetachReceipt = finalResult?.finalOutput === INTERCOM_DETACH_RECEIPT;
1996
- const effectiveFinalError = stoppedAfterAcceptance
2143
+ const baseFinalError = stoppedAfterAcceptance
1997
2144
  ? ctx.stopMessage ?? "Subagent stopped by user."
1998
2145
  : timedOutAfterAcceptance
1999
2146
  ? finalResult?.error ?? ctx.timeoutMessage ?? "Subagent timed out."
@@ -2002,6 +2149,16 @@ async function runSingleStepInner(
2002
2149
  : acceptanceCanFailRun
2003
2150
  ? (finalResult?.error ? `${finalResult.error}\n${acceptanceFailure}` : acceptanceFailure)
2004
2151
  : finalResult?.error ?? (intercomDetachReceipt ? INTERCOM_DETACH_RECEIPT : undefined);
2152
+ const effectiveFinalError = formatChildFailureDiagnostic({
2153
+ error: baseFinalError,
2154
+ afterCompactionSettlement: effectiveFinalExitCode !== 0 ? finalResult?.afterCompactionSettlement : undefined,
2155
+ missingFileOnlyOutput: effectiveFinalExitCode !== 0 && finalResult?.effects?.settlementDiagnostic?.requiredOutput?.kind === "file-only" && finalResult.effects.settlementDiagnostic.requiredOutput.missing
2156
+ ? finalResult.effects.settlementDiagnostic.requiredOutput.path
2157
+ : undefined,
2158
+ missingStructuredOutput: effectiveFinalExitCode !== 0 && finalResult?.effects?.settlementDiagnostic?.requiredOutput?.kind === "structured" && finalResult.effects.settlementDiagnostic.requiredOutput.missing
2159
+ ? finalResult.effects.settlementDiagnostic.requiredOutput.path
2160
+ : undefined,
2161
+ });
2005
2162
 
2006
2163
  const artifactErrors = artifactPaths && ctx.artifactConfig?.enabled !== false
2007
2164
  ? persistStepArtifacts({
@@ -2098,13 +2255,7 @@ type RunnerStatusStep = NonNullable<AsyncStatus["steps"]>[number] & {
2098
2255
 
2099
2256
  function externalRunnerStatus(runner: SubagentStep["runner"]): ExternalCliRunnerStatus | ExternalJobRunnerStatus | undefined {
2100
2257
  if (runner?.type === "external-cli") {
2101
- return {
2102
- type: "external-cli",
2103
- command: runner.command,
2104
- args: runner.args ?? [],
2105
- promptDelivery: runner.promptDelivery ?? "stdin",
2106
- capabilities: { stop: true, steer: false, resume: false, structuredOutput: false, toolEvents: false },
2107
- };
2258
+ return resolveExternalCliRunnerStatus(runner);
2108
2259
  }
2109
2260
  if (runner?.type === "external-job") {
2110
2261
  return {
@@ -2408,6 +2559,7 @@ async function runSubagent(
2408
2559
  ...(transcriptPath ? { transcriptPath } : {}),
2409
2560
  skills: task.skills,
2410
2561
  model: task.model,
2562
+ ...(task.contextLimit !== undefined ? { contextLimit: task.contextLimit } : {}),
2411
2563
  thinking: task.thinking,
2412
2564
  attemptedModels: task.modelCandidates && task.modelCandidates.length > 0 ? task.modelCandidates : task.model ? [task.model] : undefined,
2413
2565
  recentTools: [],
@@ -2425,6 +2577,7 @@ async function runSubagent(
2425
2577
  label: step.label ?? step.parallel.label ?? `Dynamic fanout (${step.collect.as})`,
2426
2578
  outputName: step.collect.as,
2427
2579
  structured: Boolean(step.collect.outputSchema),
2580
+ ...(step.parallel.contextLimit !== undefined ? { contextLimit: step.parallel.contextLimit } : {}),
2428
2581
  ...(step.agentContract ? { agentContract: step.agentContract } : {}),
2429
2582
  ...(step.capabilityCeiling ? { capabilityCeiling: step.capabilityCeiling } : {}),
2430
2583
  ...(step.thinkingCeiling ? { thinkingCeiling: step.thinkingCeiling } : {}),
@@ -2457,6 +2610,7 @@ async function runSubagent(
2457
2610
  ...(transcriptPath ? { transcriptPath } : {}),
2458
2611
  skills: step.skills,
2459
2612
  model: step.model,
2613
+ ...(step.contextLimit !== undefined ? { contextLimit: step.contextLimit } : {}),
2460
2614
  thinking: step.thinking,
2461
2615
  attemptedModels: step.modelCandidates && step.modelCandidates.length > 0 ? step.modelCandidates : step.model ? [step.model] : undefined,
2462
2616
  recentTools: [],
@@ -3318,11 +3472,12 @@ async function runSubagent(
3318
3472
  }
3319
3473
  pendingStepSteers.push(...remaining);
3320
3474
  };
3321
- const updateStepModel = (flatIndex: number, model: string | undefined, thinking: string | undefined, now = Date.now()): void => {
3475
+ const updateStepModel = (flatIndex: number, model: string | undefined, thinking: string | undefined, contextLimit?: number, now = Date.now()): void => {
3322
3476
  const step = statusPayload.steps[flatIndex];
3323
3477
  if (!step) return;
3324
3478
  setOptionalProperty(step, "model", model);
3325
3479
  setOptionalProperty(step, "thinking", thinking);
3480
+ setOptionalProperty(step, "contextLimit", contextLimit);
3326
3481
  statusPayload.lastUpdate = now;
3327
3482
  writeStatusPayload();
3328
3483
  };
@@ -3408,7 +3563,7 @@ async function runSubagent(
3408
3563
  return;
3409
3564
  }
3410
3565
  if (event.type === "tool_execution_start" && event.toolName) {
3411
- const mutates = isMutatingTool(event.toolName, event.args);
3566
+ const mutates = isMutatingTool(event.toolName, event.args, flatSteps[flatIndex]?.mutationTools);
3412
3567
  const currentPath = resolveCurrentPath(event.toolName, event.args);
3413
3568
  const argsPreview = extractToolArgsPreview(event.args ?? {});
3414
3569
  const blocksSupervisor = isBlockingSupervisorTool(event.toolName, event.args);
@@ -3516,12 +3671,13 @@ async function runSubagent(
3516
3671
  if (usage) {
3517
3672
  const input = usage.input ?? usage.inputTokens ?? 0;
3518
3673
  const output = usage.output ?? usage.outputTokens ?? 0;
3674
+ const window = input + (usage.cacheRead ?? usage.cacheReadTokens ?? 0);
3519
3675
  const previousInput = step.tokens?.input ?? 0;
3520
3676
  const previousOutput = step.tokens?.output ?? 0;
3521
- step.tokens = { input: previousInput + input, output: previousOutput + output, total: previousInput + previousOutput + input + output };
3677
+ step.tokens = { input: previousInput + input, output: previousOutput + output, total: previousInput + previousOutput + input + output, window, windowPeak: Math.max(step.tokens?.windowPeak ?? 0, window) };
3522
3678
  const totalInput = statusPayload.totalTokens?.input ?? 0;
3523
3679
  const totalOutput = statusPayload.totalTokens?.output ?? 0;
3524
- statusPayload.totalTokens = { input: totalInput + input, output: totalOutput + output, total: totalInput + totalOutput + input + output };
3680
+ statusPayload.totalTokens = { input: totalInput + input, output: totalOutput + output, total: totalInput + totalOutput + input + output, window, windowPeak: Math.max(statusPayload.totalTokens?.windowPeak ?? 0, window) };
3525
3681
  refreshUsageBudget();
3526
3682
  }
3527
3683
  statusPayload.turnCount = Math.max(statusPayload.turnCount ?? 0, step.turnCount);
@@ -3952,6 +4108,7 @@ async function runSubagent(
3952
4108
  ...(transcriptPath ? { transcriptPath } : {}),
3953
4109
  ...(task.skills ? { skills: task.skills } : {}),
3954
4110
  ...(task.model ? { model: task.model } : {}),
4111
+ ...(task.contextLimit !== undefined ? { contextLimit: task.contextLimit } : {}),
3955
4112
  ...(task.thinking ? { thinking: task.thinking } : {}),
3956
4113
  ...(task.thinkingCeiling ? { thinkingCeiling: task.thinkingCeiling } : {}),
3957
4114
  ...(task.modelCandidates && task.modelCandidates.length > 0 ? { attemptedModels: task.modelCandidates } : task.model ? { attemptedModels: [task.model] } : {}),
@@ -4081,7 +4238,7 @@ async function runSubagent(
4081
4238
  stopMessage,
4082
4239
  turnBudget: config.turnBudget,
4083
4240
  toolTimeoutMs: task.toolTimeoutMs ?? config.toolTimeoutMs,
4084
- onAttemptStart: (attempt) => updateStepModel(fi, attempt.model, attempt.thinking),
4241
+ onAttemptStart: (attempt) => updateStepModel(fi, attempt.model, attempt.thinking, attempt.contextLimit),
4085
4242
  onChildEvent: (event) => updateStepFromChildEvent(fi, event),
4086
4243
  onWriterProcess,
4087
4244
  onExternalProcess: (process) => updateExternalProcess(fi, process),
@@ -4477,7 +4634,7 @@ async function runSubagent(
4477
4634
  stopMessage,
4478
4635
  turnBudget: config.turnBudget,
4479
4636
  toolTimeoutMs: taskForRun.toolTimeoutMs ?? config.toolTimeoutMs,
4480
- onAttemptStart: (attempt) => updateStepModel(fi, attempt.model, attempt.thinking),
4637
+ onAttemptStart: (attempt) => updateStepModel(fi, attempt.model, attempt.thinking, attempt.contextLimit),
4481
4638
  onChildEvent: (event) => updateStepFromChildEvent(fi, event),
4482
4639
  onWriterProcess,
4483
4640
  onExternalProcess: (process) => updateExternalProcess(fi, process),
@@ -4580,13 +4737,21 @@ async function runSubagent(
4580
4737
  const sessionTokens = config.sessionDir
4581
4738
  ? parseSessionTokens(path.join(config.sessionDir, `parallel-${t}`))
4582
4739
  : null;
4583
- const taskTokens = sessionTokens ?? tokenUsageFromAttempts(parallelResults[t]?.modelAttempts);
4740
+ const fallbackTokens = tokenUsageFromAttempts(parallelResults[t]?.modelAttempts);
4741
+ const observedTokens = requiredStatusStep(statusPayload, fi).tokens;
4742
+ const taskTokens = sessionTokens ?? (fallbackTokens
4743
+ ? { ...fallbackTokens, ...(observedTokens?.window !== undefined ? { window: observedTokens.window } : {}), ...(observedTokens?.windowPeak !== undefined ? { windowPeak: observedTokens.windowPeak } : {}) }
4744
+ : null);
4584
4745
  if (!taskTokens) continue;
4585
4746
  requiredStatusStep(statusPayload, fi).tokens = taskTokens;
4586
4747
  previousCumulativeTokens = {
4587
4748
  input: previousCumulativeTokens.input + taskTokens.input,
4588
4749
  output: previousCumulativeTokens.output + taskTokens.output,
4589
4750
  total: previousCumulativeTokens.total + taskTokens.total,
4751
+ ...(taskTokens.window !== undefined ? { window: taskTokens.window } : {}),
4752
+ ...(previousCumulativeTokens.windowPeak !== undefined || taskTokens.windowPeak !== undefined
4753
+ ? { windowPeak: Math.max(previousCumulativeTokens.windowPeak ?? 0, taskTokens.windowPeak ?? 0) }
4754
+ : {}),
4590
4755
  };
4591
4756
  }
4592
4757
  statusPayload.totalTokens = { ...previousCumulativeTokens };
@@ -4824,7 +4989,7 @@ async function runSubagent(
4824
4989
  stopMessage,
4825
4990
  turnBudget: config.turnBudget,
4826
4991
  toolTimeoutMs: seqStep.toolTimeoutMs ?? config.toolTimeoutMs,
4827
- onAttemptStart: (attempt) => updateStepModel(flatIndex, attempt.model, attempt.thinking),
4992
+ onAttemptStart: (attempt) => updateStepModel(flatIndex, attempt.model, attempt.thinking, attempt.contextLimit),
4828
4993
  onChildEvent: (event) => updateStepFromChildEvent(flatIndex, event),
4829
4994
  onWriterProcess,
4830
4995
  onExternalProcess: (process) => updateExternalProcess(flatIndex, process),
@@ -4903,17 +5068,27 @@ async function runSubagent(
4903
5068
  input: cumulativeTokens.input - previousCumulativeTokens.input,
4904
5069
  output: cumulativeTokens.output - previousCumulativeTokens.output,
4905
5070
  total: cumulativeTokens.total - previousCumulativeTokens.total,
5071
+ ...(cumulativeTokens.window !== undefined ? { window: cumulativeTokens.window } : {}),
5072
+ ...(cumulativeTokens.windowPeak !== undefined ? { windowPeak: cumulativeTokens.windowPeak } : {}),
4906
5073
  }
4907
5074
  : null;
4908
5075
  if (cumulativeTokens) {
4909
5076
  previousCumulativeTokens = cumulativeTokens;
4910
5077
  } else {
4911
- stepTokens = tokenUsageFromAttempts(singleResult.modelAttempts);
5078
+ const fallbackTokens = tokenUsageFromAttempts(singleResult.modelAttempts);
5079
+ const observedTokens = requiredStatusStep(statusPayload, flatIndex).tokens;
5080
+ stepTokens = fallbackTokens
5081
+ ? { ...fallbackTokens, ...(observedTokens?.window !== undefined ? { window: observedTokens.window } : {}), ...(observedTokens?.windowPeak !== undefined ? { windowPeak: observedTokens.windowPeak } : {}) }
5082
+ : null;
4912
5083
  if (stepTokens) {
4913
5084
  previousCumulativeTokens = {
4914
5085
  input: previousCumulativeTokens.input + stepTokens.input,
4915
5086
  output: previousCumulativeTokens.output + stepTokens.output,
4916
5087
  total: previousCumulativeTokens.total + stepTokens.total,
5088
+ ...(stepTokens.window !== undefined ? { window: stepTokens.window } : {}),
5089
+ ...(previousCumulativeTokens.windowPeak !== undefined || stepTokens.windowPeak !== undefined
5090
+ ? { windowPeak: Math.max(previousCumulativeTokens.windowPeak ?? 0, stepTokens.windowPeak ?? 0) }
5091
+ : {}),
4917
5092
  };
4918
5093
  }
4919
5094
  }
@@ -5048,7 +5223,7 @@ async function runSubagent(
5048
5223
  }
5049
5224
  }
5050
5225
 
5051
- let summary = results.map((r) => `${r.agent}:\n${r.output}`).join("\n\n");
5226
+ let summary = results.map((r) => `${r.agent}:\n${r.output || (r.exitCode !== 0 ? r.error : undefined) || "(no output)"}`).join("\n\n");
5052
5227
  let truncated = false;
5053
5228
 
5054
5229
  if (maxOutput) {
@@ -5359,7 +5534,7 @@ async function waitForStartupControl(
5359
5534
  } catch (error) {
5360
5535
  throw new Error(`Failed to read runner startup control '${controlPath}': ${error instanceof Error ? error.message : String(error)}`);
5361
5536
  }
5362
- if (payload.token !== token) throw new Error("Runner startup control token does not match the acquired session lease.");
5537
+ if (payload.token !== token) throw new Error("Runner startup control token does not match.");
5363
5538
  if (payload.action === action) return;
5364
5539
  if (payload.action !== "ack" && payload.action !== "proceed") throw new Error("Runner startup control action is invalid.");
5365
5540
  }
@@ -5370,7 +5545,7 @@ async function waitForStartupControl(
5370
5545
 
5371
5546
  async function runConfiguredSubagent(config: SubagentRunConfig): Promise<void> {
5372
5547
  let lease: ReturnType<typeof acquireSessionLease> | undefined;
5373
- let startupCommitted = config.revivalLease === undefined;
5548
+ let startupCommitted = config.revivalLease === undefined && config.launchBarrierToken === undefined;
5374
5549
  const startupPath = path.join(config.asyncDir, "runner-startup.json");
5375
5550
  const startupAckPath = path.join(config.asyncDir, "runner-startup-ack.json");
5376
5551
  const startupProceedPath = path.join(config.asyncDir, "runner-startup-proceed.json");
@@ -5383,7 +5558,15 @@ async function runConfiguredSubagent(config: SubagentRunConfig): Promise<void> {
5383
5558
  };
5384
5559
  process.once("exit", releaseOnExit);
5385
5560
  try {
5386
- if (config.revivalLease) {
5561
+ if (config.launchBarrierToken) {
5562
+ await waitForStartupControl(startupProceedPath, config.launchBarrierToken, "proceed");
5563
+ startupCommitted = true;
5564
+ try {
5565
+ fs.rmSync(startupProceedPath, { force: true });
5566
+ } catch {
5567
+ // Startup control cleanup is best effort after the parent commits the run.
5568
+ }
5569
+ } else if (config.revivalLease) {
5387
5570
  lease = acquireSessionLease(config.revivalLease);
5388
5571
  config.revivalLeaseToken = lease.owner.token;
5389
5572
  writeAtomicJson(startupPath, { state: "ready", token: lease.owner.token, pid: process.pid, owner: lease.owner });
@@ -5401,7 +5584,7 @@ async function runConfiguredSubagent(config: SubagentRunConfig): Promise<void> {
5401
5584
  }
5402
5585
  await runSubagent(config, lease ? (writer) => lease!.updateWriter(writer) : undefined);
5403
5586
  } catch (error) {
5404
- if (config.revivalLease && !startupCommitted) {
5587
+ if (!startupCommitted) {
5405
5588
  try {
5406
5589
  writeAtomicJson(startupPath, { state: "error", pid: process.pid, error: error instanceof Error ? error.message : String(error) });
5407
5590
  } catch {