@nexrall/code-core 1.4.62 → 1.4.65

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/dist/agent/agentRegistry.d.ts +14 -3
  2. package/dist/agent/agentRegistry.d.ts.map +1 -1
  3. package/dist/agent/agentRegistry.js +136 -6
  4. package/dist/agent/agentTypes.d.ts +51 -1
  5. package/dist/agent/agentTypes.d.ts.map +1 -1
  6. package/dist/agent/agentTypes.js +194 -11
  7. package/dist/agent/backgroundAgents.d.ts +66 -0
  8. package/dist/agent/backgroundAgents.d.ts.map +1 -0
  9. package/dist/agent/backgroundAgents.js +145 -0
  10. package/dist/agent/loop.d.ts +43 -6
  11. package/dist/agent/loop.d.ts.map +1 -1
  12. package/dist/agent/loop.js +952 -270
  13. package/dist/agent/modelCatalogue.d.ts +15 -0
  14. package/dist/agent/modelCatalogue.d.ts.map +1 -1
  15. package/dist/agent/modelCatalogue.js +46 -0
  16. package/dist/agent/peerTransport.d.ts.map +1 -1
  17. package/dist/agent/peerTransport.js +19 -12
  18. package/dist/agent/readDedupe.d.ts +15 -0
  19. package/dist/agent/readDedupe.d.ts.map +1 -0
  20. package/dist/agent/readDedupe.js +146 -0
  21. package/dist/agent/toolPrefetch.d.ts +44 -0
  22. package/dist/agent/toolPrefetch.d.ts.map +1 -0
  23. package/dist/agent/toolPrefetch.js +101 -0
  24. package/dist/agent/trust.d.ts +0 -5
  25. package/dist/agent/trust.d.ts.map +1 -1
  26. package/dist/agent/trust.js +41 -0
  27. package/dist/api/client.d.ts +38 -0
  28. package/dist/api/client.d.ts.map +1 -1
  29. package/dist/api/client.js +132 -9
  30. package/dist/auth/index.d.ts +26 -18
  31. package/dist/auth/index.d.ts.map +1 -1
  32. package/dist/auth/index.js +45 -44
  33. package/dist/index.d.ts +1 -0
  34. package/dist/index.d.ts.map +1 -1
  35. package/dist/index.js +1 -0
  36. package/dist/permissions/modePolicy.d.ts +10 -0
  37. package/dist/permissions/modePolicy.d.ts.map +1 -1
  38. package/dist/permissions/modePolicy.js +11 -0
  39. package/dist/tools/executor.d.ts +36 -0
  40. package/dist/tools/executor.d.ts.map +1 -1
  41. package/dist/tools/executor.js +416 -89
  42. package/dist/tools/tsLangService.d.ts.map +1 -1
  43. package/dist/tools/tsLangService.js +107 -21
  44. package/dist/types.d.ts +99 -21
  45. package/dist/types.d.ts.map +1 -1
  46. package/dist/types.js +12 -1
  47. package/dist/util/miniYaml.d.ts +10 -0
  48. package/dist/util/miniYaml.d.ts.map +1 -0
  49. package/dist/util/miniYaml.js +149 -0
  50. package/package.json +8 -17
@@ -57,27 +57,37 @@ exports.capSubTaskText = capSubTaskText;
57
57
  exports.summariseSubTaskProgress = summariseSubTaskProgress;
58
58
  exports.lastToolResults = lastToolResults;
59
59
  exports.contextWindowFor = contextWindowFor;
60
+ exports.compactionLimits = compactionLimits;
60
61
  exports.compactionThresholds = compactionThresholds;
62
+ exports.pruneReclaimFloor = pruneReclaimFloor;
61
63
  exports.estimateBodyBytes = estimateBodyBytes;
62
64
  exports.allowsTestOnlyWrite = allowsTestOnlyWrite;
63
65
  exports.findSafeCutIndex = findSafeCutIndex;
66
+ exports.persistRuntimeContext = persistRuntimeContext;
64
67
  exports.transcriptOf = transcriptOf;
65
68
  exports.createLedger = createLedger;
66
69
  exports.ledgerRecord = ledgerRecord;
67
70
  exports.ledgerSummary = ledgerSummary;
68
71
  exports.pruneOldToolResults = pruneOldToolResults;
72
+ exports.makeCachedSummarizer = makeCachedSummarizer;
69
73
  exports.estimateTokensRough = estimateTokensRough;
70
74
  exports.compactMessagesForResume = compactMessagesForResume;
71
75
  exports.dispatchSubAgent = dispatchSubAgent;
76
+ exports.dispatchBackgroundSubAgent = dispatchBackgroundSubAgent;
72
77
  exports.runAgentLoop = runAgentLoop;
73
78
  exports.trimToResumableBoundary = trimToResumableBoundary;
79
+ const readDedupe_1 = require("./readDedupe");
80
+ const toolPrefetch_1 = require("./toolPrefetch");
81
+ const worktree_1 = require("./worktree");
82
+ const backgroundAgents_1 = require("./backgroundAgents");
74
83
  const types_1 = require("../types");
75
84
  const client_1 = require("../api/client");
76
- const index_1 = require("../auth/index");
77
85
  const executor_1 = require("../tools/executor");
78
86
  const agentTypes_1 = require("./agentTypes");
79
87
  const skills_1 = require("./skills");
80
88
  const rules_1 = require("../permissions/rules");
89
+ const modePolicy_1 = require("../permissions/modePolicy");
90
+ const destructive_1 = require("../permissions/destructive");
81
91
  // bashNeedsRepoLock lives in permissions/bashClassify.ts, not here: the permission
82
92
  // gate needs the same "does this command mutate shared state?" answer, and importing
83
93
  // it from loop.ts would make permissions depend on the agent loop (a cycle).
@@ -87,7 +97,7 @@ const planMode_1 = require("./planMode");
87
97
  const worktreeEnforcement_1 = require("./worktreeEnforcement");
88
98
  const agentRegistry_1 = require("./agentRegistry");
89
99
  const sandbox_1 = require("../tools/sandbox");
90
- const index_2 = require("../plugins/index");
100
+ const index_1 = require("../plugins/index");
91
101
  const testIntegrity_1 = require("./testIntegrity");
92
102
  const flaky_1 = require("./flaky");
93
103
  const claimEvidence_1 = require("./claimEvidence");
@@ -99,6 +109,16 @@ const crossProcessLock_1 = require("./crossProcessLock");
99
109
  const fs = __importStar(require("fs"));
100
110
  const path = __importStar(require("path"));
101
111
  const child_process_1 = require("child_process");
112
+ function withAgentHooks(base, agent) {
113
+ if (!agent)
114
+ return base;
115
+ return {
116
+ ...base,
117
+ PreToolUse: [...(base.PreToolUse ?? []), ...(agent.PreToolUse ?? [])],
118
+ PostToolUse: [...(base.PostToolUse ?? []), ...(agent.PostToolUse ?? [])],
119
+ SubagentStop: [...(base.SubagentStop ?? []), ...(agent.Stop ?? [])],
120
+ };
121
+ }
102
122
  function loadHooks(workDir) {
103
123
  let fromSettings = {};
104
124
  try {
@@ -108,7 +128,7 @@ function loadHooks(workDir) {
108
128
  }
109
129
  catch { /* ignore */ }
110
130
  // Merge plugin-provided hooks AFTER the project's own (project hooks run first).
111
- const fromPlugins = (0, index_2.pluginHooks)(workDir);
131
+ const fromPlugins = (0, index_1.pluginHooks)(workDir);
112
132
  const merged = { ...fromSettings };
113
133
  for (const phase of Object.keys(fromPlugins)) {
114
134
  const extra = fromPlugins[phase];
@@ -129,7 +149,60 @@ function loadHooks(workDir) {
129
149
  // • stdout JSON object → { "decision": "block"|"allow", "reason": "...",
130
150
  // "additionalContext": "text to feed the model" }
131
151
  // • any other exit code → non-blocking (stderr logged, tool proceeds)
132
- function runToolHooks(entries, phase, toolName, input, workDir, result) {
152
+ /**
153
+ * Run one hook command WITHOUT blocking the event loop. spawnSync froze everything in
154
+ * the process for up to the hook's timeout — parallel sub-agents' streams, the UI, the
155
+ * stall watchdogs — which a 60 s PostToolUse test hook turned into a visible hang.
156
+ */
157
+ function spawnHook(command, opts) {
158
+ return new Promise((resolve) => {
159
+ let stdout = '';
160
+ let stderr = '';
161
+ let settled = false;
162
+ const done = (status) => {
163
+ if (settled)
164
+ return;
165
+ settled = true;
166
+ clearTimeout(timer);
167
+ resolve({ status, stdout, stderr });
168
+ };
169
+ let child;
170
+ try {
171
+ child = (0, child_process_1.spawn)(command, { shell: true, cwd: opts.cwd, env: opts.env ?? process.env, stdio: ['pipe', 'pipe', 'pipe'] });
172
+ }
173
+ catch {
174
+ resolve({ status: null, stdout: '', stderr: '' });
175
+ return;
176
+ }
177
+ const timer = setTimeout(() => { try {
178
+ child.kill('SIGTERM');
179
+ }
180
+ catch { /* gone */ } done(null); }, opts.timeout);
181
+ const cap = 16 * 1024 * 1024;
182
+ child.stdout?.on('data', (d) => { if (stdout.length < cap)
183
+ stdout += d.toString('utf-8'); });
184
+ child.stderr?.on('data', (d) => { if (stderr.length < cap)
185
+ stderr += d.toString('utf-8'); });
186
+ child.on('error', () => done(null));
187
+ child.on('close', (code) => done(code));
188
+ child.stdin?.on('error', () => { });
189
+ child.stdin?.end(opts.input ?? '');
190
+ });
191
+ }
192
+ /**
193
+ * Does a hook's matcher select this tool? Empty / "*" = every tool. "A|B" = either.
194
+ * Each alternative may be a Claude Code tool name ("Bash", "Edit") or a Nexrall one, and
195
+ * a Nexrall-name alternative keeps the old substring behaviour ("file" matches read_file).
196
+ */
197
+ function hookMatches(matcher, toolName) {
198
+ if (!matcher || matcher === '*')
199
+ return true;
200
+ return matcher.split('|').map((m) => m.trim()).filter(Boolean).some((alt) => {
201
+ const norm = (0, agentTypes_1.normaliseToolName)(alt);
202
+ return norm === toolName || (norm === alt && toolName.includes(alt));
203
+ });
204
+ }
205
+ async function runToolHooks(entries, phase, toolName, input, workDir, result) {
133
206
  const outcome = { block: false };
134
207
  if (!entries?.length)
135
208
  return outcome;
@@ -140,7 +213,7 @@ function runToolHooks(entries, phase, toolName, input, workDir, result) {
140
213
  ...(result ? { result: { output: result.output, error: result.error } } : {}),
141
214
  });
142
215
  for (const entry of entries) {
143
- if (entry.matcher && entry.matcher !== '*' && !toolName.includes(entry.matcher))
216
+ if (!hookMatches(entry.matcher, toolName))
144
217
  continue;
145
218
  for (const hook of entry.hooks ?? []) {
146
219
  if (hook.type !== 'command' || !hook.command)
@@ -151,26 +224,17 @@ function runToolHooks(entries, phase, toolName, input, workDir, result) {
151
224
  const hookTimeout = typeof hook.timeout_ms === 'number' && hook.timeout_ms > 0
152
225
  ? Math.min(hook.timeout_ms, 600000)
153
226
  : 60000;
154
- let r;
155
- try {
156
- r = (0, child_process_1.spawnSync)(hook.command, {
157
- shell: true,
158
- cwd: workDir,
159
- timeout: hookTimeout,
160
- encoding: 'utf-8',
161
- maxBuffer: 16 * 1024 * 1024,
162
- input: payload,
163
- env: {
164
- ...process.env,
165
- NEXRALL_TOOL_NAME: toolName,
166
- NEXRALL_TOOL_INPUT: JSON.stringify(input),
167
- NEXRALL_HOOK_PHASE: phase,
168
- },
169
- });
170
- }
171
- catch {
172
- continue; // hook itself failed to spawn — non-fatal
173
- }
227
+ const r = await spawnHook(hook.command, {
228
+ cwd: workDir,
229
+ timeout: hookTimeout,
230
+ input: payload,
231
+ env: {
232
+ ...process.env,
233
+ NEXRALL_TOOL_NAME: toolName,
234
+ NEXRALL_TOOL_INPUT: JSON.stringify(input),
235
+ NEXRALL_HOOK_PHASE: phase,
236
+ },
237
+ });
174
238
  // Optional JSON directive on stdout
175
239
  const out = (r.stdout ?? '').toString().trim();
176
240
  if (out.startsWith('{')) {
@@ -196,15 +260,12 @@ function runToolHooks(entries, phase, toolName, input, workDir, result) {
196
260
  }
197
261
  return outcome;
198
262
  }
199
- function runSimpleHooks(defs, workDir) {
263
+ async function runSimpleHooks(defs, workDir, extraEnv) {
200
264
  if (!defs?.length)
201
265
  return;
202
266
  for (const hook of defs) {
203
267
  if (hook.type === 'command' && hook.command) {
204
- try {
205
- (0, child_process_1.spawnSync)(hook.command, { shell: true, cwd: workDir, timeout: 10000 });
206
- }
207
- catch { /* non-fatal */ }
268
+ await spawnHook(hook.command, { cwd: workDir, timeout: 10000, env: extraEnv ? { ...process.env, ...extraEnv } : undefined });
208
269
  }
209
270
  }
210
271
  }
@@ -576,29 +637,45 @@ function lockPathsFor(name, input, workDir) {
576
637
  // callers.
577
638
  //
578
639
  // Anthropic hit the same wall and capped Claude Code's concurrent subagents at 20
579
- // (v2.1.217, July 2026); community guidance settles far lower, around 3-5, because
580
- // past that the synthesis overhead cancels the parallelism. We default to 4:
581
- // enough for genuine fan-out (the case sub-agents exist for), low enough that a
582
- // runaway `task` burst degrades into a queue instead of a thundering herd.
640
+ // (v2.1.217, July 2026, CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS); community guidance for
641
+ // everyday work settles around 3-5 because past that the synthesis overhead cancels
642
+ // the parallelism.
643
+ //
644
+ // Default 10, raised from 4 (2026-09-29), with a hard ceiling of 32 (see below). Measured
645
+ // on a 12-core dev box against the Nexrall repo, the per-agent latency of a
646
+ // search_files + glob pair was 116 ms at N=4, 176 ms at N=10, 300 ms at N=20 and 415 ms
647
+ // at N=30. Local CPU is NOT what limits fan-out; the model API is (rounds are seconds
648
+ // long, tools are milliseconds). What did limit it was the SERVER: /api/code was IP
649
+ // rate-limited at 100 req/15 min, which even 4 agents exhausted. That is now per-user
650
+ // (backend shared/utils/codeRateLimit.js), so the old "4" no longer protects anything
651
+ // that 10 does not.
652
+ //
653
+ // Why not default 20/30 like the headline number: the default applies to EVERY user, on
654
+ // every model, including ones with a low per-key TPM, and each concurrent agent is its
655
+ // own bill. Wide fan-out is opt-in (maxConcurrentSubtasks / NEXRALL_MAX_CONCURRENT_SUBTASKS
656
+ // up to 32), narrow fan-out is the safe default.
583
657
  //
584
658
  // This is a QUEUE, not a rejection: every sub-task still runs, just at most N at a
585
659
  // time. Failing the excess would be worse than serialising it.
586
- const DEFAULT_MAX_CONCURRENT_SUBTASKS = 4;
660
+ const DEFAULT_MAX_CONCURRENT_SUBTASKS = 10;
661
+ /** Hard ceiling on maxConcurrentSubtasks, whatever settings.json (repo-controlled) says. */
662
+ const HARD_MAX_CONCURRENT_SUBTASKS = 32;
587
663
  /**
588
664
  * Resolve the fan-out limit: env → settings.json → default.
589
665
  *
590
666
  * Was env-ONLY, the same gap as the sub-task timeout: a project on a small machine that
591
667
  * wanted 2, or a big one that wanted 6, had to set a shell variable, and nobody reading
592
- * settings.json could see what the limit even was. Capped at 16 because this bounds real
593
- * shared resources (CPU, the API rate limit, file handles) and a typo like 400 should
594
- * degrade to "a lot" rather than fork-bomb the machine.
668
+ * settings.json could see what the limit even was. Capped at HARD_MAX_CONCURRENT_SUBTASKS
669
+ * (32 — above Claude Code's default 20, so a user who wants that width can have it)
670
+ * because this bounds real shared resources (CPU, the API rate limit, file handles) and a
671
+ * typo like 400 should degrade to "a lot" rather than fork-bomb the machine.
595
672
  */
596
673
  function resolveMaxConcurrentSubtasks(settingsRaw = {}) {
597
674
  // `Math.max(1, …)` matters: a fractional value like 0.5 passes the `> 0` guard, then floors
598
675
  // to 0, and createLimiter(0) queues every task with nothing left to ever release them — a
599
676
  // silent permanent hang with no timeout and no error. Harmless when only depth 0 used the
600
677
  // limiter; now that every level does, it would wedge the whole tree.
601
- const clamp = (n) => Math.max(1, Math.min(Math.floor(n), 16));
678
+ const clamp = (n) => Math.max(1, Math.min(Math.floor(n), HARD_MAX_CONCURRENT_SUBTASKS));
602
679
  const fromEnv = Number(process.env.NEXRALL_MAX_CONCURRENT_SUBTASKS);
603
680
  if (Number.isFinite(fromEnv) && fromEnv > 0)
604
681
  return clamp(fromEnv);
@@ -704,11 +781,11 @@ function subTaskLimiter(depth, workDir) {
704
781
  if (!run) {
705
782
  // TAPERED per level, not `max` at every level.
706
783
  //
707
- // Giving each depth the full `max` multiplies total concurrency by the depth limit: 4
708
- // becomes 12 by default and 16×5 = 80 at the configured maxima. The value is justified by
784
+ // Giving each depth the full `max` multiplies total concurrency by the depth limit: 10
785
+ // becomes 30 by default and 32×5 = 160 at the configured maxima. The value is justified by
709
786
  // shared resources — CPU, file handles, ONE API rate limit — none of which care which
710
787
  // level a loop is running at, so honouring `4` per level silently abandons the limit the
711
- // user set. Halving per level bounds the total at ~2× `max` (4+2+1 = 7) while keeping the
788
+ // user set. Halving per level bounds the total at ~2× `max` (10+5+3 = 18 by default) while keeping the
712
789
  // per-depth structure that makes the wait-for graph acyclic.
713
790
  //
714
791
  // Never below 1: a level with 0 slots is a permanent hang, not a restriction.
@@ -725,22 +802,19 @@ function _resetSubTaskLimiter() {
725
802
  _subTaskLimitMaxByDepth.clear();
726
803
  _inFlightByDepth.clear();
727
804
  _subTaskLimitMax = 0;
728
- _subAgentsThisSession = 0;
729
- _sessionCapMax = 0;
730
- _sessionCapNotified = false;
805
+ _subAgentBudgets.clear();
806
+ }
807
+ const _subAgentBudgets = new Map();
808
+ const PROCESS_BUDGET_KEY = '\0process';
809
+ function budgetFor(sessionKey) {
810
+ const key = sessionKey || PROCESS_BUDGET_KEY;
811
+ let b = _subAgentBudgets.get(key);
812
+ if (!b) {
813
+ b = { used: 0, max: 0, notified: false };
814
+ _subAgentBudgets.set(key, b);
815
+ }
816
+ return b;
731
817
  }
732
- // Counts sub-agents STARTED, never decremented — that is what makes it a budget rather
733
- // than a concurrency gate.
734
- //
735
- // NOT process-wide, unlike the limiter, and the distinction is load-bearing. A concurrency
736
- // gate is safe to share across a process because it self-drains; a monotonic counter is
737
- // not. In the CLI one process is one session, but VS Code calls runAgentLoop from a
738
- // long-lived extension host, so process-scoped state would accumulate across every
739
- // conversation in the window until delegation died permanently — and the remedy string
740
- // ("start a new session") would be a lie, since only reloading the window would help.
741
- let _subAgentsThisSession = 0;
742
- let _sessionCapMax = 0;
743
- let _sessionCapNotified = false;
744
818
  /**
745
819
  * Start a fresh sub-agent budget. Call when a NEW conversation begins.
746
820
  *
@@ -748,10 +822,9 @@ let _sessionCapNotified = false;
748
822
  * host). A client that never calls it gets process-lifetime semantics, which is correct
749
823
  * for a one-shot CLI invocation.
750
824
  */
751
- function resetSessionSubAgentBudget() {
752
- _subAgentsThisSession = 0;
753
- _sessionCapMax = 0;
754
- _sessionCapNotified = false;
825
+ function resetSessionSubAgentBudget(sessionKey) {
826
+ // No key: the legacy "new conversation" call — reset the process-wide budget only.
827
+ _subAgentBudgets.delete(sessionKey || PROCESS_BUDGET_KEY);
755
828
  }
756
829
  /**
757
830
  * Claim one slot against the session total. Returns an error string when exhausted.
@@ -761,13 +834,15 @@ function resetSessionSubAgentBudget() {
761
834
  * the user never learns delegation was capped, which is the one thing a runaway guard has
762
835
  * to make visible.
763
836
  */
764
- function claimSessionSubAgentSlot(workDir, notify) {
765
- if (!_sessionCapMax) {
766
- _sessionCapMax = resolveMaxSubagentsPerSession(workDir ? (0, rules_1.loadSettings)(workDir).raw : {});
837
+ function claimSessionSubAgentSlot(workDir, notify, sessionKey) {
838
+ const b = budgetFor(sessionKey);
839
+ if (!b.max) {
840
+ b.max = resolveMaxSubagentsPerSession(workDir ? (0, rules_1.loadSettings)(workDir).raw : {});
767
841
  }
768
- if (_subAgentsThisSession >= _sessionCapMax) {
769
- if (!_sessionCapNotified) {
770
- _sessionCapNotified = true;
842
+ const _sessionCapMax = b.max;
843
+ if (b.used >= b.max) {
844
+ if (!b.notified) {
845
+ b.notified = true;
771
846
  notify?.(`\u26a0\ufe0f Sub-agent budget reached (${_sessionCapMax} this session) \u2014 further delegation is ` +
772
847
  'blocked and the agent will continue without it. Raise "maxSubagentsPerSession" in ' +
773
848
  '.nexrall/settings.json if this was legitimate work.');
@@ -776,7 +851,7 @@ function claimSessionSubAgentSlot(workDir, notify) {
776
851
  'runaway-delegation guard, not a per-task limit: do the remaining work directly, and say ' +
777
852
  'in your final message that you hit the delegation cap.');
778
853
  }
779
- _subAgentsThisSession++;
854
+ b.used++;
780
855
  return null;
781
856
  }
782
857
  /**
@@ -795,13 +870,14 @@ function claimSessionSubAgentSlot(workDir, notify) {
795
870
  *
796
871
  * Floored at 0 so a double refund can never manufacture budget.
797
872
  */
798
- function refundSessionSubAgentSlot() {
799
- if (_subAgentsThisSession > 0)
800
- _subAgentsThisSession--;
873
+ function refundSessionSubAgentSlot(sessionKey) {
874
+ const b = budgetFor(sessionKey);
875
+ if (b.used > 0)
876
+ b.used--;
801
877
  }
802
878
  /** Test-only: observe the session counter without exporting the mutable binding. */
803
- function _sessionSubAgentCount() {
804
- return _subAgentsThisSession;
879
+ function _sessionSubAgentCount(sessionKey) {
880
+ return budgetFor(sessionKey).used;
805
881
  }
806
882
  // ─── Peer message rendering ───────────────────────────────────────────────────
807
883
  //
@@ -831,7 +907,7 @@ function humanDescription(name, input) {
831
907
  case 'bash':
832
908
  return `Run: ${input.command ?? '(unknown)'}`;
833
909
  case 'search_files': {
834
- const searchType = input.type === 'filename' ? 'filename' : 'content';
910
+ const searchType = input.type === 'filename' ? 'filename' : input.type === 'content' || input.context_lines ? 'content' : 'files';
835
911
  const inPath = input.path ? ` in ${input.path}` : '';
836
912
  return `Search ${searchType}: "${input.pattern ?? ''}"${inPath}`;
837
913
  }
@@ -1236,6 +1312,72 @@ function toolResultText(block) {
1236
1312
  }
1237
1313
  return '';
1238
1314
  }
1315
+ // ── Hard stop ─────────────────────────────────────────────────────────────────
1316
+ // The stall watchdog and the parent's Stop only SET a flag; runAgentLoop notices it at
1317
+ // the next boundary. A tool that never returns (an MCP server that ignores the abort,
1318
+ // a stuck editor-side call) meant that boundary never came and the parent hung forever.
1319
+ // After the flag is set, the child gets this long to wind down before we stop waiting.
1320
+ const HARD_STOPPED = Symbol('hard-stopped');
1321
+ function hardStopGraceMs() {
1322
+ const v = Number(process.env.NEXRALL_SUBAGENT_HARD_STOP_MS);
1323
+ return Number.isFinite(v) && v > 0 ? v : 30000;
1324
+ }
1325
+ function raceHardStop(run, abort) {
1326
+ return new Promise((resolve, reject) => {
1327
+ let grace = null;
1328
+ const poll = setInterval(() => {
1329
+ if (abort.aborted && !grace) {
1330
+ grace = setTimeout(() => { clearInterval(poll); resolve(HARD_STOPPED); }, hardStopGraceMs());
1331
+ }
1332
+ }, 250);
1333
+ const settle = () => { clearInterval(poll); if (grace)
1334
+ clearTimeout(grace); };
1335
+ run.then((v) => { settle(); resolve(v); }, (e) => { settle(); reject(e); });
1336
+ });
1337
+ }
1338
+ /** Returned instead of waiting forever for a tool that ignored Stop. */
1339
+ const STOP_TIMEOUT_RESULT = {
1340
+ error: 'Interrupted: this tool did not respond to Stop, so the agent stopped waiting for it.',
1341
+ interrupted: true,
1342
+ };
1343
+ const STOP_GRACE_MS = 5000;
1344
+ function stopAwaiting(run, abort) {
1345
+ if (!abort)
1346
+ return run;
1347
+ return new Promise((resolve, reject) => {
1348
+ let grace = null;
1349
+ const poll = setInterval(() => {
1350
+ if (abort.aborted && !grace)
1351
+ grace = setTimeout(() => { clearInterval(poll); resolve(STOP_TIMEOUT_RESULT); }, STOP_GRACE_MS);
1352
+ }, 200);
1353
+ const settle = () => { clearInterval(poll); if (grace)
1354
+ clearTimeout(grace); };
1355
+ run.then((v) => { settle(); resolve(v); }, (e) => { settle(); reject(e); });
1356
+ });
1357
+ }
1358
+ /** Default turn limit for a sub-agent whose definition sets no `maxTurns`. */
1359
+ const DEFAULT_SUBAGENT_MAX_TURNS = 200;
1360
+ /** Stop reasons that mean the sub-agent did NOT deliver a finished report. */
1361
+ const ABNORMAL_SUBAGENT_STOPS = new Set([
1362
+ 'budget', 'stalled', 'stalled-repeat', 'output-limit', 'empty-response',
1363
+ 'no-balance', 'no-team-budget', 'team-unavailable', 'reported-elsewhere',
1364
+ ]);
1365
+ /**
1366
+ * Prepended to EVERY sub-agent's instructions (named or not) — Claude Code's
1367
+ * "you are a sub-agent; your final message is the report" contract.
1368
+ */
1369
+ const SUBAGENT_PREAMBLE = [
1370
+ '# You are a sub-agent',
1371
+ 'The main agent started you for ONE delegated task. You cannot see its conversation with the user — only the task you were given — and you cannot ask the user anything: if something is ambiguous, make the most reasonable assumption and state it.',
1372
+ 'Your FINAL message is the only thing returned to the main agent, and the user does not see it directly. Make it a concise, self-contained report: what you found or changed (with file paths and line numbers), what you verified and how, and anything left undone or uncertain. Do not pad it, and do not paste whole files — cite locations instead.',
1373
+ ].join('\n');
1374
+ function formatTokenCount(n) {
1375
+ return n >= 1000000 ? `${(n / 1000000).toFixed(1)}M` : n >= 1000 ? `${(n / 1000).toFixed(1)}K` : String(n);
1376
+ }
1377
+ function formatElapsed(ms) {
1378
+ const s = Math.round(ms / 1000);
1379
+ return s < 60 ? `${s}s` : `${Math.floor(s / 60)}m ${s % 60}s`;
1380
+ }
1239
1381
  async function runSubTask(input, options, agentTypes,
1240
1382
  // Set to true at the moment an agent loop actually STARTS, so the caller can refund the
1241
1383
  // session slot it claimed for a spawn that turned out never to run.
@@ -1295,7 +1437,7 @@ started) {
1295
1437
  //
1296
1438
  // Reported as "expired" rather than "not yours": a distinct message would confirm the id
1297
1439
  // exists, turning the error into an oracle for enumerating other frames' agents.
1298
- if (resumed && !(0, agentRegistry_1.canResume)(resumed, agentScope)) {
1440
+ if (resumed && !(0, agentRegistry_1.canResume)(resumed, agentScope, options.sessionId)) {
1299
1441
  return {
1300
1442
  error: `No resumable sub-agent with id "${resumeId}" is available to this run. Start a fresh ` +
1301
1443
  'sub-task with a self-contained prompt instead.',
@@ -1342,7 +1484,9 @@ started) {
1342
1484
  'do not create it and do not retry. Do the work yourself, or use a different sub-agent.',
1343
1485
  };
1344
1486
  }
1345
- let agent = (0, agentTypes_1.findAgentType)(agentTypes, requestedType);
1487
+ // An unnamed dispatch IS general-purpose (the tool schema says so): same role prompt,
1488
+ // same "your final message is your report" instructions, same deny rule (denyKey above).
1489
+ let agent = (0, agentTypes_1.findAgentType)(agentTypes, requestedType || 'general-purpose');
1346
1490
  let knownTypes = agentTypes;
1347
1491
  if (requestedType && !agent) {
1348
1492
  // Same `extra` list the top-of-run snapshot used (see runAgentLoop) — otherwise a
@@ -1381,12 +1525,39 @@ started) {
1381
1525
  // `lightPrompt` agents skip the project's nexrall.md — see AgentType.lightPrompt for
1382
1526
  // why. The role prompt and any private notes still apply; only the (potentially very
1383
1527
  // large) project instruction file is dropped.
1384
- const inheritedMd = agent?.lightPrompt ? '' : (options.nexrallMd ?? '');
1385
- const subNexrallMd = agent
1386
- ? `# Sub-agent role: ${agent.name}\n${agent.prompt}` +
1387
- (agentMemoryNotes ? `\n\n---\n\n${agentMemoryNotes}` : '') +
1388
- (inheritedMd ? `\n\n---\n\n${inheritedMd}` : '')
1389
- : options.nexrallMd;
1528
+ //
1529
+ // Composed from the PROJECT's instructions (`_projectNexrallMd`), never from the
1530
+ // parent's own composite prompt: a grandchild used to inherit its parent's role
1531
+ // ("# Sub-agent role: general-purpose … make the changes") on top of its own
1532
+ // ("reviewer … you NEVER modify files"), plus the parent's private memory notes.
1533
+ const projectMd = options._projectNexrallMd ?? options.nexrallMd ?? '';
1534
+ const inheritedMd = agent?.lightPrompt ? '' : projectMd;
1535
+ // `skills:` — preload those skills' instructions (raw body: no !`cmd` expansion runs
1536
+ // just because an agent was spawned). A missing name is stated, not silently dropped.
1537
+ let preloadedSkills = '';
1538
+ if (agent?.skills?.length) {
1539
+ const known = (0, skills_1.loadSkillsWithWarnings)(options.workDir, options._extraSkills ?? []).skills;
1540
+ preloadedSkills = agent.skills.map((n) => {
1541
+ const sk = (0, skills_1.findSkill)(known, n);
1542
+ return sk ? `# Preloaded skill: ${sk.name}\n${sk.body.trim()}` : `# Preloaded skill: ${n}\n(not found in this project — ignore)`;
1543
+ }).join('\n\n');
1544
+ }
1545
+ // Shared-before-specific order: the project's nexrall.md (usually the largest part, and
1546
+ // identical for every non-lightPrompt sibling) goes first; per-agent parts (preamble,
1547
+ // role, skills, private notes) follow — last, where they also carry the most weight
1548
+ // with the model. With the role first, two agents diverged at the first line of this
1549
+ // block. HONEST SCOPE: this only buys cache reuse where everything BEFORE the block
1550
+ // already matches — same tool list and same prompt flags (so e.g. two custom agents
1551
+ // with equal `tools:`, on prefix-cached providers). Types with different allowlists
1552
+ // diverge earlier, at the tools array, whatever the order here. It costs nothing, and
1553
+ // tool enforcement never depended on prompt order (see the allowlist below).
1554
+ const subNexrallMd = [
1555
+ inheritedMd,
1556
+ SUBAGENT_PREAMBLE,
1557
+ agent ? `# Sub-agent role: ${agent.name}\n${agent.prompt}` : '',
1558
+ preloadedSkills,
1559
+ agentMemoryNotes,
1560
+ ].filter(Boolean).join('\n\n---\n\n');
1390
1561
  // Optional tool allowlist — deny anything outside it for this sub-agent.
1391
1562
  //
1392
1563
  // A refusal here is reported through `deniedReason` rather than the generic
@@ -1415,6 +1586,20 @@ started) {
1415
1586
  // it, so the allowlist stays the single source of truth for what this agent can do.
1416
1587
  if (allowed && memoryScope)
1417
1588
  allowed.add(exports.AGENT_MEMORY_TOOL);
1589
+ // `disallowedTools` (Claude Code) — inherited like the allowlist: a child can only add
1590
+ // to its ancestors' refusals, never shed them.
1591
+ const disallowed = new Set([...(options._disallowedTools ?? []), ...(agent?.disallowedTools ?? [])]);
1592
+ if (allowed)
1593
+ for (const t of disallowed)
1594
+ allowed.delete(t);
1595
+ // `mcpServers` — narrows, inherited like the tool allowlist.
1596
+ let mcpAllow = options._mcpServerAllowlist;
1597
+ if (agent?.mcpServers) {
1598
+ const own = new Set(agent.mcpServers);
1599
+ mcpAllow = mcpAllow ? new Set([...mcpAllow].filter((x) => own.has(x))) : own;
1600
+ }
1601
+ // Time spent waiting on a human (permission prompt) is not a stall.
1602
+ let waitingOnUser = 0;
1418
1603
  const gatedPermission = async (req) => {
1419
1604
  // Guard against the tool being reachable without a declared scope — e.g. an agent
1420
1605
  // that lists it in `tools:` by hand, or a general-purpose sub-task with no
@@ -1462,12 +1647,42 @@ started) {
1462
1647
  // allowlist above, a restriction can only ever be narrowed by nesting, never shed.
1463
1648
  // Without the inherited half, `test-writer` could delegate to an unrestricted agent and
1464
1649
  // have production source written on its behalf.
1650
+ if (mcpAllow && req.tool.includes('__') && !mcpAllow.has(req.tool.split('__')[0])) {
1651
+ throw new ToolNotAllowedError(`The "${agent?.name ?? 'general-purpose'}" sub-agent may only use tools from these MCP servers: ` +
1652
+ `${[...mcpAllow].join(', ') || '(none)'}. \`${req.tool}\` is from another server — use a different tool or report back.`);
1653
+ }
1654
+ if (disallowed.has(req.tool)) {
1655
+ throw new ToolNotAllowedError(`The "${agent?.name ?? 'general-purpose'}" sub-agent may not use \`${req.tool}\` (disallowedTools in its ` +
1656
+ 'definition). This is a restriction of the agent definition, NOT a user decision — use another tool or report back.');
1657
+ }
1658
+ if (req.tool === 'memory_write') {
1659
+ throw new ToolNotAllowedError('Sub-agents cannot write the shared memory store. Put anything worth remembering in your final report — ' +
1660
+ 'the main agent decides what to persist.');
1661
+ }
1465
1662
  if (testFilesOnly && !allowsTestOnlyWrite(req.tool, req.input)) {
1466
1663
  throw new ToolNotAllowedError(`The "${agent?.name ?? 'general-purpose'}" sub-agent may only write to TEST files, so ` +
1467
1664
  `\`${req.tool}\` was refused for this path. Do not try to work around it: if production ` +
1468
1665
  'code must change, say so in your report instead.');
1469
1666
  }
1470
- return options.requestPermission(req);
1667
+ // An agent writing its OWN notes (declared `memory:`) is answered here, not by the
1668
+ // ancestors: their gates refuse agent_memory_write unless THEY declared a scope, so a
1669
+ // memory agent spawned by general-purpose could never write. The tool can only touch
1670
+ // this agent's own notes file, so there is nothing for anyone above to approve.
1671
+ if (req.tool === exports.AGENT_MEMORY_TOOL && agent && memoryScope)
1672
+ return true;
1673
+ waitingOnUser++;
1674
+ try {
1675
+ // Say WHICH sub-agent is asking (the innermost one wins as the request bubbles up).
1676
+ const description = typeof input.description === 'string' ? input.description : undefined;
1677
+ return await options.requestPermission({
1678
+ ...req,
1679
+ agent: req.agent ?? { name: agent?.name ?? 'general-purpose', ...(description ? { description } : {}) },
1680
+ });
1681
+ }
1682
+ finally {
1683
+ waitingOnUser--;
1684
+ bumpProgress();
1685
+ }
1471
1686
  };
1472
1687
  // ── Resume: continue a previous sub-agent instead of starting cold ──────────
1473
1688
  //
@@ -1508,6 +1723,10 @@ started) {
1508
1723
  let stalled = false;
1509
1724
  const bumpProgress = () => { lastProgressAt = Date.now(); };
1510
1725
  const stallWatchdog = setInterval(() => {
1726
+ if (waitingOnUser > 0) {
1727
+ lastProgressAt = Date.now();
1728
+ return;
1729
+ }
1511
1730
  if (Date.now() - lastProgressAt > subtaskTimeoutMs) {
1512
1731
  stalled = true;
1513
1732
  subAbort.aborted = true;
@@ -1521,11 +1740,83 @@ started) {
1521
1740
  if (options.abortSignal?.aborted)
1522
1741
  subAbort.aborted = true;
1523
1742
  }, 250);
1743
+ // ── Worktree isolation (`isolation: "worktree"` on the call or in the definition) ──
1744
+ // A fresh worktree per spawn, like Claude Code: the child's writes, bash cwd and git
1745
+ // commands are confined to it (worktreeEnforcement), and it is removed afterwards
1746
+ // unless the child actually changed something.
1747
+ let isoState;
1748
+ if (input.isolation === 'worktree' || agent?.isolation === 'worktree') {
1749
+ const created = (0, worktree_1.createWorktree)(options.workDir);
1750
+ if (!created.ok || !created.state) {
1751
+ clearInterval(stallWatchdog);
1752
+ clearInterval(parentAbortPoll);
1753
+ return {
1754
+ error: `Could not create an isolated worktree for this sub-agent: ${created.error ?? 'unknown error'}. ` +
1755
+ 'Run it without isolation, or fix the repository state first.',
1756
+ };
1757
+ }
1758
+ isoState = created.state;
1759
+ }
1760
+ const childWorkDir = isoState?.worktreePath ?? options.workDir;
1761
+ // Claude Code's Explore/Plan skip git status; so do lightPrompt agents here. An
1762
+ // isolated child is told where it actually is.
1763
+ let childEnv = options.env;
1764
+ if (childEnv && agent?.lightPrompt) {
1765
+ const { gitStatus: _s, gitDiff: _d, recentCommits: _c, ...rest } = childEnv;
1766
+ childEnv = rest;
1767
+ }
1768
+ if (childEnv && isoState)
1769
+ childEnv = { ...childEnv, cwd: isoState.worktreePath, gitBranch: isoState.branch ?? childEnv.gitBranch };
1770
+ // Per-sub-agent accounting, reported in its tool result (Claude Code shows the same).
1771
+ const subStartedAt = Date.now();
1772
+ let subToolCalls = 0;
1773
+ let subTokens = 0;
1774
+ let subCost = 0;
1775
+ let childStop = null;
1776
+ let childNotice = null;
1777
+ let childVerifications = [];
1778
+ const finalize = (r) => {
1779
+ let isoNote = '';
1780
+ if (isoState) {
1781
+ if ((0, worktree_1.worktreeHasWork)(isoState)) {
1782
+ isoNote = `\n[isolated worktree: this sub-agent's changes are in ${isoState.worktreePath}` +
1783
+ `${isoState.branch ? ` (branch ${isoState.branch})` : ''} — NOT in the main checkout. Review them, then merge or discard.]`;
1784
+ }
1785
+ else {
1786
+ (0, worktree_1.removeWorktree)(isoState, { force: true });
1787
+ }
1788
+ isoState = undefined;
1789
+ }
1790
+ const stats = `[sub-agent stats: ${subToolCalls} tool call(s) · ${formatTokenCount(subTokens)} tokens · ` +
1791
+ `$${subCost.toFixed(3)} · ${formatElapsed(Date.now() - subStartedAt)}]`;
1792
+ const out = { ...r, ...(childVerifications.length ? { childVerifications } : {}) };
1793
+ if (out.error !== undefined)
1794
+ out.error = `${out.error}\n\n${stats}${isoNote}`;
1795
+ else
1796
+ out.output = `${out.output ?? ''}\n\n${stats}${isoNote}`;
1797
+ return out;
1798
+ };
1799
+ // Claimed right before `try` (whose finally releases it). runSubTask has no `await`
1800
+ // before this point, so two parallel calls cannot both pass the check.
1801
+ if (resumed && !(0, agentRegistry_1.claimAgentForResume)(resumed.id)) {
1802
+ clearInterval(stallWatchdog);
1803
+ clearInterval(parentAbortPoll);
1804
+ if (isoState)
1805
+ (0, worktree_1.removeWorktree)(isoState, { force: true });
1806
+ return {
1807
+ error: `Sub-agent "${resumed.id}" is already being resumed by another task call that is still running. ` +
1808
+ 'Wait for that result, then resume it again with your follow-up — two parallel resumes of one agent ' +
1809
+ 'would overwrite each other\'s work.',
1810
+ };
1811
+ }
1812
+ const childMeta = (m) => (m
1813
+ ? { ...m, parentId: m.parentId ?? options._taskToolUseId, agentName: m.agentName ?? agent?.name ?? 'general-purpose' }
1814
+ : undefined);
1524
1815
  try {
1525
1816
  // The point of no return: past here a real agent loop exists and the budget slot is spent.
1526
1817
  if (started)
1527
1818
  started.value = true;
1528
- const result = await runAgentLoop(subMessages, {
1819
+ const childRun = runAgentLoop(subMessages, {
1529
1820
  ...options,
1530
1821
  _depth: depth + 1,
1531
1822
  _agentScope: `sub_${++_subTaskCounter}`, // isolated todo store per sub-agent
@@ -1563,16 +1854,48 @@ started) {
1563
1854
  // it, and here that direction is at least safe, whereas forgetting to propagate is not.
1564
1855
  _testFilesOnly: testFilesOnly,
1565
1856
  editorContext: null, // fresh isolated context for sub-agent
1566
- model: agent?.model ?? options.model,
1857
+ model: (0, modelCatalogue_1.resolveSubAgentModel)(agent?.model, options.model),
1858
+ workDir: childWorkDir,
1859
+ env: childEnv,
1860
+ effort: agent?.effort ?? options.effort,
1861
+ // A bounded run that REPORTS when it hits the limit (see childStop below), instead
1862
+ // of inheriting the main agent's 500-step budget plus auto-continue to 2000.
1863
+ maxIterations: agent?.maxTurns ?? DEFAULT_SUBAGENT_MAX_TURNS,
1864
+ autoContinue: false,
1865
+ // ── Session-level channels a child must NEVER consume ────────────────────
1866
+ // Inherited through `...options`, a child drained the user's queued follow-ups
1867
+ // and inbound peer messages into ITS history (the main agent never saw them),
1868
+ // and its onProgress saved the child's transcript AS the session (CLI/desktop).
1869
+ takePendingInput: undefined,
1870
+ onInjectedInput: undefined,
1871
+ drainPeerMessages: undefined,
1872
+ onPeerMessage: undefined,
1873
+ onProgress: undefined,
1874
+ backgroundAgents: undefined,
1875
+ _projectNexrallMd: projectMd,
1876
+ _disallowedTools: disallowed.size ? disallowed : undefined,
1877
+ _mcpServerAllowlist: mcpAllow,
1878
+ _agentHooks: agent?.hooks,
1879
+ _onStopReason: (reason, notice) => { childStop = reason; childNotice = notice; },
1880
+ _onVerifications: (records) => { childVerifications = records; },
1881
+ onUsage: (u, partial, cost, _sub) => {
1882
+ if (!partial) {
1883
+ subTokens += (u.input_tokens ?? 0) + (u.output_tokens ?? 0)
1884
+ + (u.cache_read_input_tokens ?? 0) + (u.cache_creation_input_tokens ?? 0);
1885
+ subCost += cost ?? 0;
1886
+ }
1887
+ options.onUsage(u, partial, cost, true);
1888
+ },
1567
1889
  // Plan mode is inherited, never relaxed. If the main agent could spawn a
1568
1890
  // sub-agent that writes, the lock would be one `task` call from useless.
1569
- planMode: options.planMode,
1891
+ // `permissionMode: plan` in a definition can only ADD the lock.
1892
+ planMode: options.planMode || agent?.permissionMode === 'plan',
1570
1893
  // Same reasoning as planMode directly above: a sub-agent that could reach
1571
1894
  // outside its parent's worktree would defeat the isolation in one `task`
1572
1895
  // call. Already inherited via `...options` above — restated explicitly so
1573
1896
  // it reads the same way as planMode and is never accidentally dropped by
1574
1897
  // a future refactor of this spread.
1575
- worktree: options.worktree,
1898
+ worktree: isoState ?? options.worktree,
1576
1899
  // A sub-agent using message_peer_session/list_peer_sessions should
1577
1900
  // present as the SAME peer identity as its parent — there is one
1578
1901
  // registered peer per SESSION, not per sub-agent, so a sub-agent is
@@ -1582,22 +1905,10 @@ started) {
1582
1905
  abortSignal: subAbort,
1583
1906
  requestPermission: gatedPermission,
1584
1907
  onText: () => { }, // sub-agent text is returned as the tool result, not streamed live
1585
- // Forwarded DELIBERATELY, and it must be a real handler rather than a no-op.
1586
- //
1587
- // A sub-agent streams no text (onText above is a no-op) but it DOES stream thinking
1588
- // through the parent's UI (see onThinking/onThinkingDelta below), and thinking sets
1589
- // `emittedToCaller`. So a sub-agent stream that dies after reasoning genuinely has
1590
- // rendered output to discard — a no-op here would let the restart proceed and then
1591
- // re-stream that reasoning on top of the copy still on screen, the exact duplication
1592
- // the opt-in exists to prevent.
1593
- //
1594
- // Forwarding is safe because the parent's own output is already closed by this
1595
- // point: dispatching the `task` tool goes through options.onToolUse, which finalizes
1596
- // the parent's bubble (VS Code) / flushes the renderer (CLI) before the sub-agent
1597
- // starts. The only live, discardable element at restart time is the sub-agent's own
1598
- // thinking block. Completed tool rows are left alone — those are real side effects
1599
- // that actually happened.
1600
- onStreamRestart: (reason, chars) => options.onStreamRestart?.(reason, chars),
1908
+ // A sub-agent renders nothing live (text and thinking are not streamed — see below),
1909
+ // so a restart of ITS stream has nothing on screen to roll back: a real no-op
1910
+ // handler is exactly right, and it opts the child into post-render restarts.
1911
+ onStreamRestart: () => { },
1601
1912
  // Forward tool events with isSubTask=true so the UI can render a badge
1602
1913
  // instead of prepending "[sub-task]" to the tool name (which caused double-prefix
1603
1914
  // when the name was already labelled, and mixed display concerns into the data layer).
@@ -1605,17 +1916,32 @@ started) {
1605
1916
  // is what the stall watchdog above measures. Bumping on both use and result means a
1606
1917
  // single very slow tool (a long test run) resets the clock when it starts AND when
1607
1918
  // it finishes, so it cannot be mistaken for a hang.
1608
- onToolUse: (n, i) => { bumpProgress(); options.onToolUse(n, i, true); },
1609
- onToolResult: (n, r) => { bumpProgress(); options.onToolResult(n, r, true); },
1610
- onToolStreamChunk: (n, c) => options.onToolStreamChunk?.(n, c, true),
1611
- // Forward thinking so the UI shows the indicator while sub-agent reasons
1612
- // Thinking is progress too — a model reasoning for minutes on a hard problem is
1613
- // working, not stalled. Without this, deep reasoning on an expensive tier would
1614
- // trip the watchdog precisely when the sub-agent was most valuable.
1615
- onThinking: (text) => { bumpProgress(); options.onThinking?.(text); },
1616
- onThinkingDelta: (text) => { bumpProgress(); options.onThinkingDelta?.(text); },
1617
- onThinkingProgress: (tok) => { bumpProgress(); options.onThinkingProgress?.(tok); },
1919
+ // `parentId` = the `task` call that spawned THIS child. Set only if not already set,
1920
+ // so a grandchild's events keep pointing at their own (nearest) task row.
1921
+ onToolUse: (n, i, _s, m) => { bumpProgress(); subToolCalls++; options.onToolUse(n, i, true, childMeta(m)); },
1922
+ onToolResult: (n, r, _s, m) => { bumpProgress(); options.onToolResult(n, r, true, childMeta(m)); },
1923
+ // A long command streaming output is working, not hung.
1924
+ onToolStreamChunk: (n, c, _s, m) => { bumpProgress(); options.onToolStreamChunk?.(n, c, true, childMeta(m)); },
1925
+ // Thinking is progress (a model reasoning for minutes is working, not stalled), but
1926
+ // it is NOT forwarded to the UI: parallel siblings' deltas interleaved into one live
1927
+ // thinking block, and Claude Code does not show sub-agent reasoning either. The
1928
+ // sub-agent's tool rows (grouped under its task row) are its visible progress.
1929
+ onThinking: () => { bumpProgress(); },
1930
+ onThinkingDelta: () => { bumpProgress(); },
1931
+ onThinkingProgress: () => { bumpProgress(); },
1618
1932
  });
1933
+ const raced = await raceHardStop(childRun, subAbort);
1934
+ if (raced === HARD_STOPPED) {
1935
+ // The child was told to stop (stall watchdog, parent Stop, background stop) but a
1936
+ // tool it is running ignored cancellation — an MCP call, an editor-side tool. The
1937
+ // parent must not hang on it forever; the orphaned call is left to finish alone.
1938
+ return finalize({
1939
+ error: `Sub-task did not stop within ${Math.round(hardStopGraceMs() / 1000)}s of being stopped: a tool it was ` +
1940
+ 'running ignored cancellation. Its partial work could not be collected. Do not re-run it as-is — ' +
1941
+ 'narrow the task, or avoid the tool that hung.',
1942
+ });
1943
+ }
1944
+ const result = raced;
1619
1945
  // ── Stall timeout: SALVAGE, don't discard ────────────────────────────────
1620
1946
  //
1621
1947
  // The sub-agent hit its wall-clock cap (rather than finishing, or the parent
@@ -1643,7 +1969,7 @@ started) {
1643
1969
  // do: run the whole task again. Resuming is now POSSIBLE but never implied to be
1644
1970
  // safe: the text below states plainly that the work is unverified, and resumption
1645
1971
  // re-authorises against current permissions exactly as it does for a clean run.
1646
- const partialId = (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope);
1972
+ const partialId = (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope, options.sessionId);
1647
1973
  const sections = [
1648
1974
  `Sub-task STOPPED after ${mins} minutes with NO PROGRESS (it was not making tool calls or ` +
1649
1975
  'producing output) — treat everything below as PARTIAL, unverified work, not a finished answer.',
@@ -1657,27 +1983,49 @@ started) {
1657
1983
  // errored call as a non-effect, which is right — nothing here is verified —
1658
1984
  // and the STALL_LIMIT runaway guard must still see repeated timeouts as
1659
1985
  // failures so a permanently stuck sub-task can't loop forever.
1660
- return { error: sections.join('\n\n') };
1986
+ return finalize({ error: sections.join('\n\n') });
1987
+ }
1988
+ // ── Stopped abnormally (turn limit, repeated failures, truncation, no balance) ──
1989
+ // runAgentLoop RETURNS normally for these, so this used to fall through to the
1990
+ // success path: the parent was handed a fragment as if it were the finished report,
1991
+ // while the explanation went to the user's chat as if the main agent had stopped.
1992
+ // Assigned inside callbacks, so TS narrows them to `null` here without the casts.
1993
+ const stopReasonOfChild = childStop;
1994
+ const noticeOfChild = childNotice;
1995
+ if (stopReasonOfChild && ABNORMAL_SUBAGENT_STOPS.has(stopReasonOfChild) && !subAbort.aborted) {
1996
+ const partial = capSubTaskText(extractSubTaskText(result, false));
1997
+ const progress = summariseSubTaskProgress(result);
1998
+ const partialId = resumed
1999
+ ? ((0, agentRegistry_1.updateAgent)(resumed.id, result, agentScope, options.sessionId), resumed.id)
2000
+ : (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope, options.sessionId);
2001
+ const why = stopReasonOfChild === 'budget'
2002
+ ? `it reached its turn limit (${agent?.maxTurns ?? DEFAULT_SUBAGENT_MAX_TURNS} model round-trips)`
2003
+ : `it stopped early (${stopReasonOfChild})`;
2004
+ const sections = [
2005
+ `Sub-task did NOT finish: ${why}. Treat everything below as PARTIAL, unverified work.`,
2006
+ noticeOfChild ? noticeOfChild.trim() : '',
2007
+ progress,
2008
+ partial ? `Partial output:\n\n${partial}` : '',
2009
+ `To continue it with everything it already read, call task with resume_agent_id="${partialId}".`,
2010
+ ].filter(Boolean);
2011
+ return finalize({ error: sections.join('\n\n') });
1661
2012
  }
1662
2013
  // Normal completion: the final assistant message is the sub-agent's answer.
1663
2014
  const text = capSubTaskText(extractSubTaskText(result, true));
1664
2015
  // Store the transcript so a follow-up can continue this agent rather than
1665
2016
  // re-running it from scratch, and tell the parent the id.
1666
2017
  //
1667
- // Only on NORMAL completion. A timed-out or failed run is deliberately not
1668
- // resumable: its transcript ends mid-thought, often mid-tool-call, and
1669
- // resuming from that state invites the model to build on work whose status
1670
- // it cannot determine. Those paths already salvage their partial output as
1671
- // TEXT, which is the safe way to carry that information forward.
2018
+ // (The stalled and stopped-early paths above register their transcripts too, marked
2019
+ // as partial work in the text they return; only a thrown failure is not resumable.)
1672
2020
  const agentId = resumed
1673
- ? ((0, agentRegistry_1.updateAgent)(resumed.id, result, agentScope), resumed.id)
1674
- : (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope);
2021
+ ? ((0, agentRegistry_1.updateAgent)(resumed.id, result, agentScope, options.sessionId), resumed.id)
2022
+ : (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope, options.sessionId);
1675
2023
  const body = text || '(sub-task completed with no text output)';
1676
- return {
2024
+ return finalize({
1677
2025
  output: `${body}\n\n[resumable: this sub-agent is "${agentId}". To ask IT a follow-up — keeping ` +
1678
2026
  'everything it already read and concluded — call task again with resume_agent_id="' + agentId +
1679
2027
  '" instead of writing a new prompt from scratch.]',
1680
- };
2028
+ });
1681
2029
  }
1682
2030
  catch (err) {
1683
2031
  // Same salvage rule as the timeout path above, for the other way a sub-agent
@@ -1695,13 +2043,18 @@ started) {
1695
2043
  partial ? `Partial output before the failure:\n\n${partial}` : '',
1696
2044
  'Treat the above as PARTIAL, unverified work. Build on it rather than re-running the whole sub-task.',
1697
2045
  ].filter(Boolean);
1698
- return { error: sections.join('\n\n') };
2046
+ return finalize({ error: sections.join('\n\n') });
1699
2047
  }
1700
- return { error: `Sub-task failed: ${err.message}` };
2048
+ return finalize({ error: `Sub-task failed: ${err.message}` });
1701
2049
  }
1702
2050
  finally {
1703
2051
  clearInterval(stallWatchdog);
1704
2052
  clearInterval(parentAbortPoll);
2053
+ // Every return above already ran finalize(); this only catches an unexpected path.
2054
+ if (isoState && !(0, worktree_1.worktreeHasWork)(isoState))
2055
+ (0, worktree_1.removeWorktree)(isoState, { force: true });
2056
+ if (resumed)
2057
+ (0, agentRegistry_1.releaseAgentForResume)(resumed.id);
1705
2058
  }
1706
2059
  }
1707
2060
  // ─── Auto-compact ─────────────────────────────────────────────────────────────
@@ -1739,7 +2092,7 @@ const MODEL_CONTEXT_TOKENS = {
1739
2092
  ultra: 1000000,
1740
2093
  fast: 200000,
1741
2094
  // Anthropic, by real model id.
1742
- 'claude-sonnet-5': 1000000,
2095
+ 'claude-sonnet-5-5': 1000000,
1743
2096
  'claude-opus-5-5': 1000000,
1744
2097
  'claude-fable-5-1': 1000000,
1745
2098
  'claude-haiku-4-5-20251001': 200000,
@@ -1807,13 +2160,13 @@ function contextWindowFor(model) {
1807
2160
  // session that is real money accruing per turn long before the 1M wall.
1808
2161
  // Anthropic's own server-side compaction defaults its trigger to 150K
1809
2162
  // input tokens (docs: compact_20260112 default trigger 150000). We mirror
1810
- // that intent: start shedding already-consumed tool_result bulk at ~35% of
1811
- // a 1M window (~350K tokens) so the per-turn cache-read bill stops growing,
2163
+ // that intent: start shedding already-consumed tool_result bulk at ~120K
2164
+ // tokens (see compactionLimits) so the per-turn cache-read bill stops growing,
1812
2165
  // WITHOUT paying for a summariser model call and WITHOUT dropping any turn
1813
2166
  // (pruneOldToolResults keeps every tool_use/tool_result pair intact).
1814
2167
  //
1815
- // • SUMMARISE threshold (expensive: a real model call, lossy: drops whole
1816
- // turns): stays LATE, because summarise-of-summarise is the main cause of an
2168
+ // • SUMMARISE threshold (a model call, lossy: drops whole turns): Claude Code's
2169
+ // ~167K line (see compactionLimits). Later than prune, because summarise-of-summarise is the main cause of an
1817
2170
  // agent "forgetting" earlier work. Only when cheap pruning can't keep the
1818
2171
  // prompt under this line do we fall through to summarisation.
1819
2172
  //
@@ -1822,17 +2175,67 @@ function envFraction(name, fallback) {
1822
2175
  const v = Number(process.env[name]);
1823
2176
  return Number.isFinite(v) && v > 0 && v < 1 ? v : fallback;
1824
2177
  }
1825
- // Prune early (~35% of window), summarise late (~80% of window).
1826
- const AUTO_PRUNE_THRESHOLD = envFraction('NEXRALL_PRUNE_THRESHOLD', 0.35);
1827
- const AUTO_COMPACT_THRESHOLD = envFraction('NEXRALL_COMPACT_THRESHOLD', 0.8); // summarise when prompt > 80% of the window
2178
+ // Where auto-compaction fires, in TOKENS — Claude Code's rule, not a fraction of a 1M window.
2179
+ //
2180
+ // Claude Code summarises at `window − min(maxOutput, 20K) − 13K buffer`, on a 200K window
2181
+ // (~167K tokens). We used to wait for 80% of a 1M window (~800K): every request of a long
2182
+ // run re-read that whole prefix from cache, so a session cost ~5× more per request than
2183
+ // the same work in Claude Code long before anything was shed. The window used for this is
2184
+ // capped (default 200K, like Claude Code) — the model's real 1M window still bounds what
2185
+ // the backend will accept; this only decides when WE tidy up.
2186
+ //
2187
+ // NEXRALL_COMPACT_WINDOW / settings.json "autoCompactWindow": the cap (tokens). Set it
2188
+ // to e.g. 1000000 to use the whole window before compacting.
2189
+ // NEXRALL_PRUNE_THRESHOLD / NEXRALL_COMPACT_THRESHOLD: fractions of that capped window.
2190
+ const DEFAULT_COMPACT_WINDOW = 200000;
2191
+ const COMPACT_OUTPUT_RESERVE = 20000; // Claude Code: min(model max output, 20K)
2192
+ const COMPACT_BUFFER_TOKENS = 13000; // Claude Code's autocompact buffer
2193
+ const DEFAULT_PRUNE_FRACTION = 0.6; // cheap lossless prune ~120K, before the ~167K summarise
2194
+ function compactWindowCap(settingsRaw) {
2195
+ const fromEnv = Number(process.env.NEXRALL_COMPACT_WINDOW);
2196
+ if (Number.isFinite(fromEnv) && fromEnv >= 50000)
2197
+ return fromEnv;
2198
+ const fromSettings = Number(settingsRaw?.autoCompactWindow);
2199
+ if (Number.isFinite(fromSettings) && fromSettings >= 50000)
2200
+ return fromSettings;
2201
+ return DEFAULT_COMPACT_WINDOW;
2202
+ }
2203
+ /** Token counts at which auto-prune and auto-compact (summarise) fire for a model window. */
2204
+ function compactionLimits(contextWindow, settingsRaw) {
2205
+ const eff = Math.min(contextWindow, compactWindowCap(settingsRaw));
2206
+ const compactFrac = envFraction('NEXRALL_COMPACT_THRESHOLD', 0);
2207
+ const pruneFrac = envFraction('NEXRALL_PRUNE_THRESHOLD', 0);
2208
+ const compact = compactFrac
2209
+ ? eff * compactFrac
2210
+ : Math.max(eff * 0.5, eff - Math.min(COMPACT_OUTPUT_RESERVE, eff * 0.1) - COMPACT_BUFFER_TOKENS);
2211
+ const prune = Math.min(pruneFrac ? eff * pruneFrac : eff * DEFAULT_PRUNE_FRACTION, compact * 0.9);
2212
+ return { prune: Math.floor(prune), compact: Math.floor(compact) };
2213
+ }
1828
2214
  /** Auto-prune / auto-compact thresholds as fractions of the context window — for UI display (e.g. `/context`). */
1829
- function compactionThresholds() {
1830
- return { prune: AUTO_PRUNE_THRESHOLD, compact: AUTO_COMPACT_THRESHOLD };
2215
+ function compactionThresholds(contextWindow = 1000000, settingsRaw) {
2216
+ const { prune, compact } = compactionLimits(contextWindow, settingsRaw);
2217
+ return { prune: prune / contextWindow, compact: compact / contextWindow };
1831
2218
  }
1832
2219
  // Only bother pruning if it reclaims a meaningful amount — a tiny prune busts
1833
2220
  // the message-level prompt cache (the pruned prefix changes) for little gain,
1834
2221
  // so we require at least this many bytes reclaimed before accepting a prune.
1835
- const PRUNE_MIN_RECLAIM_BYTES = 256 * 1024; // 256 KB
2222
+ // Sized for the ~120K-token prune line: at 256 KB a prune could rarely reclaim enough to
2223
+ // qualify, so sessions skipped straight to the lossy summariser.
2224
+ const PRUNE_MIN_RECLAIM_BYTES = 96 * 1024; // 96 KB
2225
+ // Cache-aware prune floor. The 96 KB floor exists ONLY to avoid busting a WARM prompt
2226
+ // cache for a small gain. Once the session has been idle past the provider cache TTL
2227
+ // (Anthropic/OpenAI: 5 min), the whole prefix is re-written on the next request anyway,
2228
+ // so a prune at that moment costs nothing extra — accept a much smaller reclaim then.
2229
+ // Keyed by sessionId (same scheme as _subAgentBudgets) because runAgentLoop runs once
2230
+ // per user turn and the idle gap that matters is BETWEEN turns.
2231
+ const PRUNE_MIN_RECLAIM_BYTES_COLD = 16 * 1024; // 16 KB
2232
+ const CACHE_COLD_AFTER_MS = 5 * 60000;
2233
+ const _lastApiCallEndedAt = new Map();
2234
+ function pruneReclaimFloor(lastCallEndedAt, now = Date.now()) {
2235
+ return lastCallEndedAt !== undefined && now - lastCallEndedAt >= CACHE_COLD_AFTER_MS
2236
+ ? PRUNE_MIN_RECLAIM_BYTES_COLD
2237
+ : PRUNE_MIN_RECLAIM_BYTES;
2238
+ }
1836
2239
  const COMPACT_KEEP_MIN = 6; // always keep at least the last N messages verbatim
1837
2240
  /**
1838
2241
  * Bytes a compaction must reclaim to count as productive.
@@ -1992,10 +2395,27 @@ const MAX_TRANSCRIPT_CHARS = 600000;
1992
2395
  * middle — a middle-out elision that preserves both "what we set out to do" and
1993
2396
  * "where we are now", which is what the continuation summary needs most.
1994
2397
  */
2398
+ /**
2399
+ * Appends the backend-announced runtime-context block to the user message it was
2400
+ * attached to. No-op if that message is not a user turn or already ends with the
2401
+ * identical block. Exported for tests.
2402
+ */
2403
+ function persistRuntimeContext(messages, idx, text) {
2404
+ const m = messages[idx];
2405
+ if (!m || m.role !== 'user' || !Array.isArray(m.content))
2406
+ return false;
2407
+ const lastBlock = m.content[m.content.length - 1];
2408
+ if (lastBlock && lastBlock.type === 'text' && lastBlock.text === text)
2409
+ return false;
2410
+ messages[idx] = { ...m, content: [...m.content, { type: 'text', text }] };
2411
+ return true;
2412
+ }
1995
2413
  function transcriptOf(messages) {
1996
2414
  const parts = [];
1997
2415
  for (const m of messages) {
1998
2416
  for (const b of m.content) {
2417
+ if ((0, types_1.isRuntimeContextBlock)(b))
2418
+ continue; // editor scaffolding, not conversation
1999
2419
  if (b.type === 'text' && b.text) {
2000
2420
  parts.push(`${m.role.toUpperCase()}: ${(0, safeSlice_1.sliceSafeEnd)(b.text, 2000)}`);
2001
2421
  }
@@ -2254,7 +2674,7 @@ function originalTaskText(messages) {
2254
2674
  if (!first || !Array.isArray(first.content))
2255
2675
  return '';
2256
2676
  return first.content
2257
- .filter((b) => b.type === 'text' && b.text)
2677
+ .filter((b) => b.type === 'text' && b.text && !(0, types_1.isRuntimeContextBlock)(b))
2258
2678
  .map((b) => b.text)
2259
2679
  .join('\n')
2260
2680
  .trim();
@@ -2279,7 +2699,32 @@ class CompactionUnavailableError extends Error {
2279
2699
  this.name = 'CompactionUnavailableError';
2280
2700
  }
2281
2701
  }
2282
- async function autoCompactMessages(messages, options, ledger) {
2702
+ function makeCachedSummarizer(messages, requestOptions, abortSignal) {
2703
+ return async (instruction) => {
2704
+ const last = messages[messages.length - 1];
2705
+ if (!last || last.role !== 'user')
2706
+ return '';
2707
+ const lastContent = typeof last.content === 'string'
2708
+ ? [{ type: 'text', text: last.content }]
2709
+ : [...last.content];
2710
+ const probe = [
2711
+ ...messages.slice(0, -1),
2712
+ { ...last, content: [...lastContent, { type: 'text', text: instruction }] },
2713
+ ];
2714
+ const reply = await (0, client_1.streamChat)(probe, { ...requestOptions(), abortSignal, allowRestartAfterRender: true }, () => { });
2715
+ return reply.content
2716
+ .filter((b) => b.type === 'text')
2717
+ .map((b) => b.text ?? '')
2718
+ .join('')
2719
+ .trim();
2720
+ };
2721
+ }
2722
+ const CACHED_SUMMARY_INSTRUCTION = `[Context compaction — this is an automated request from the agent runtime, not the user.]\n` +
2723
+ `Do NOT call any tools and do NOT continue the task. Reply with text only: a concise bullet-point ` +
2724
+ `summary of this whole session so far that you will need to continue the work — the user's goal, ` +
2725
+ `key decisions, files changed (and how), commands run and their outcome, unresolved problems, and ` +
2726
+ `user preferences. Max 400 words.`;
2727
+ async function autoCompactMessages(messages, options, ledger, summarizeCached) {
2283
2728
  const cut = findSafeCutIndex(messages, messages.length - COMPACT_KEEP_MIN);
2284
2729
  if (cut < 2)
2285
2730
  return false; // nothing meaningful to fold
@@ -2297,44 +2742,64 @@ async function autoCompactMessages(messages, options, ledger) {
2297
2742
  `key decisions, files changed (and how), commands run, unresolved problems, and user preferences. Max 400 words.\n\n` +
2298
2743
  transcriptOf(toSummarize);
2299
2744
  let summary = '';
2300
- try {
2301
- const reply = await (0, client_1.streamChat)([{ role: 'user', content: [{ type: 'text', text: summaryPrompt }] }], {
2302
- // Run on the SAME model as the actual conversation. A previous version
2303
- // forced 'turbo' (Claude Sonnet 5) here on the theory that a mechanical
2304
- // "bullet-point this transcript" task doesn't need the user's tier — but
2305
- // that silently billed Anthropic (and made a real network call to a
2306
- // provider the user may not have configured/paid for) even when the
2307
- // whole session was running on OpenAI/DeepSeek/Qwen. Whatever the user
2308
- // is already paying for is used for compaction too, so there is never a
2309
- // surprise charge on a different provider. Falls back to the same
2310
- // default as the main loop (see `runAgentLoop`) only when no model was
2311
- // set at all.
2312
- model: options.model ?? 'turbo',
2313
- mode: 'ask', // summariser must not call tools; ask-mode discourages action
2314
- env: options.env,
2315
- clientType: options.clientType,
2316
- abortSignal: options.abortSignal,
2317
- // The summariser renders NOTHING (onEvent below is a no-op) and its result is
2318
- // read only from the returned message, so a restart has nothing to roll back —
2319
- // always safe. Worth enabling: a blip here used to abandon compaction entirely,
2320
- // which then let the very next turn hit the context wall it was meant to prevent.
2321
- allowRestartAfterRender: true,
2322
- }, () => { });
2323
- summary = reply.content
2324
- .filter((b) => b.type === 'text')
2325
- .map((b) => b.text ?? '')
2326
- .join('')
2327
- .trim();
2328
- }
2329
- catch {
2330
- // Summarisation FAILED — the model call itself threw (network blip, 529,
2331
- // provider quota). Distinguished from "ran fine but didn't help" by the
2332
- // caller, because the two must not feed the same circuit breaker: three
2333
- // transient network errors would otherwise permanently disable compaction
2334
- // for the rest of the run, leaving the context to grow until the turn dies
2335
- // with no recovery left. Leave history as is; the turn may still fit.
2336
- throw new CompactionUnavailableError();
2745
+ // Cache-friendly path first; any failure or empty reply falls back to the standalone
2746
+ // transcript summariser below, which always works but pays for the history uncached.
2747
+ if (summarizeCached) {
2748
+ try {
2749
+ summary = await summarizeCached(CACHED_SUMMARY_INSTRUCTION);
2750
+ }
2751
+ catch (err) {
2752
+ if (options.abortSignal?.aborted || err.name === 'AbortError')
2753
+ throw err;
2754
+ summary = '';
2755
+ }
2337
2756
  }
2757
+ if (!summary)
2758
+ try {
2759
+ const reply = await (0, client_1.streamChat)([{ role: 'user', content: [{ type: 'text', text: summaryPrompt }] }], {
2760
+ // Run on the SAME model as the actual conversation. A previous version
2761
+ // forced 'turbo' (Claude Sonnet 5) here on the theory that a mechanical
2762
+ // "bullet-point this transcript" task doesn't need the user's tier — but
2763
+ // that silently billed Anthropic (and made a real network call to a
2764
+ // provider the user may not have configured/paid for) even when the
2765
+ // whole session was running on OpenAI/DeepSeek/Qwen. Whatever the user
2766
+ // is already paying for is used for compaction too, so there is never a
2767
+ // surprise charge on a different provider. Falls back to the same
2768
+ // default as the main loop (see `runAgentLoop`) only when no model was
2769
+ // set at all.
2770
+ //
2771
+ // Cheapest model of the SAME vendor (Claude Code runs this kind of work on
2772
+ // Haiku). Safe to downgrade HERE because this fallback sends a fresh
2773
+ // transcript — it shares no cached prefix with the conversation, unlike
2774
+ // summarizeCached above, which must stay on the conversation's own model.
2775
+ // Never crosses providers; falls back to the session model if the
2776
+ // catalogue is not loaded.
2777
+ model: (0, modelCatalogue_1.cheapestSameVendorModel)(options.model ?? 'turbo') ?? options.model ?? 'turbo',
2778
+ mode: 'ask', // summariser must not call tools; ask-mode discourages action
2779
+ env: options.env,
2780
+ clientType: options.clientType,
2781
+ abortSignal: options.abortSignal,
2782
+ // The summariser renders NOTHING (onEvent below is a no-op) and its result is
2783
+ // read only from the returned message, so a restart has nothing to roll back —
2784
+ // always safe. Worth enabling: a blip here used to abandon compaction entirely,
2785
+ // which then let the very next turn hit the context wall it was meant to prevent.
2786
+ allowRestartAfterRender: true,
2787
+ }, () => { });
2788
+ summary = reply.content
2789
+ .filter((b) => b.type === 'text')
2790
+ .map((b) => b.text ?? '')
2791
+ .join('')
2792
+ .trim();
2793
+ }
2794
+ catch {
2795
+ // Summarisation FAILED — the model call itself threw (network blip, 529,
2796
+ // provider quota). Distinguished from "ran fine but didn't help" by the
2797
+ // caller, because the two must not feed the same circuit breaker: three
2798
+ // transient network errors would otherwise permanently disable compaction
2799
+ // for the rest of the run, leaving the context to grow until the turn dies
2800
+ // with no recovery left. Leave history as is; the turn may still fit.
2801
+ throw new CompactionUnavailableError();
2802
+ }
2338
2803
  if (!summary)
2339
2804
  return false;
2340
2805
  // Replace the summarized head with a single user summary message. The cut is
@@ -2376,7 +2841,9 @@ async function maybeCompactMemory(scope, options) {
2376
2841
  // Same reasoning as autoCompactMessages' summariser: run on the same
2377
2842
  // model as the actual conversation rather than forcing a fixed tier
2378
2843
  // (which silently billed Anthropic regardless of the user's provider).
2379
- model: options.model ?? 'turbo',
2844
+ // Cheapest SAME-vendor model: a fresh prompt with no shared cache prefix,
2845
+ // and merging a few memory bullets does not need the session's top tier.
2846
+ model: (0, modelCatalogue_1.cheapestSameVendorModel)(options.model ?? 'turbo') ?? options.model ?? 'turbo',
2380
2847
  mode: 'ask',
2381
2848
  env: options.env,
2382
2849
  clientType: options.clientType,
@@ -2447,8 +2914,9 @@ async function compactMessagesForResume(messages, opts) {
2447
2914
  // only at the late one. On resume this matters most: a stored session is resent
2448
2915
  // whole on the first turn, so shedding old tool_result bulk up front is exactly
2449
2916
  // what stops that first message being billed at full size.
2450
- const overPruneThreshold = () => tokenGuess > contextWindow * AUTO_PRUNE_THRESHOLD || bodyBytes > MAX_BODY_BYTES;
2451
- const overCompactThreshold = () => tokenGuess > contextWindow * AUTO_COMPACT_THRESHOLD || bodyBytes > MAX_BODY_BYTES;
2917
+ const limits = compactionLimits(contextWindow, settings.raw);
2918
+ const overPruneThreshold = () => tokenGuess > limits.prune || bodyBytes > MAX_BODY_BYTES;
2919
+ const overCompactThreshold = () => tokenGuess > limits.compact || bodyBytes > MAX_BODY_BYTES;
2452
2920
  if (!overPruneThreshold())
2453
2921
  return false;
2454
2922
  let compacted = false;
@@ -2526,7 +2994,7 @@ async function compactMessagesForResume(messages, opts) {
2526
2994
  */
2527
2995
  async function dispatchSubAgent(input, options, agentTypes) {
2528
2996
  const depth = options._depth ?? 0;
2529
- const overBudget = claimSessionSubAgentSlot(options.workDir, options.onNotice ?? options.onText);
2997
+ const overBudget = claimSessionSubAgentSlot(options.workDir, options.onNotice ?? options.onText, options.sessionId);
2530
2998
  if (overBudget)
2531
2999
  return { error: overBudget };
2532
3000
  const childDepth = depth + 1;
@@ -2544,7 +3012,7 @@ async function dispatchSubAgent(input, options, agentTypes) {
2544
3012
  }
2545
3013
  finally {
2546
3014
  if (!started.value)
2547
- refundSessionSubAgentSlot();
3015
+ refundSessionSubAgentSlot(options.sessionId);
2548
3016
  const n = (_inFlightByDepth.get(childDepth) ?? 1) - 1;
2549
3017
  if (n > 0)
2550
3018
  _inFlightByDepth.set(childDepth, n);
@@ -2552,11 +3020,81 @@ async function dispatchSubAgent(input, options, agentTypes) {
2552
3020
  _inFlightByDepth.delete(childDepth);
2553
3021
  }
2554
3022
  }
3023
+ /**
3024
+ * `task` with run_in_background — start the sub-agent detached and return at once.
3025
+ *
3026
+ * The child runs with its OWN abort signal (the user's Stop on the main turn does not
3027
+ * kill it; the hub's stop() does), renders nothing into the chat (its tool calls are
3028
+ * only counted for the "N agents" badge), and cannot prompt the user: a call that the
3029
+ * project's rules or the session mode do not already allow is refused with an
3030
+ * explanation, which the child then reports (Claude Code auto-denies the same way).
3031
+ * Its final report reaches the main agent as a <task-notification>.
3032
+ */
3033
+ function dispatchBackgroundSubAgent(input, options, agentTypes, hub) {
3034
+ const overBudget = claimSessionSubAgentSlot(options.workDir, options.onNotice ?? options.onText, options.sessionId);
3035
+ if (overBudget)
3036
+ return { error: overBudget };
3037
+ const agentType = (typeof input.subagent_type === 'string' && input.subagent_type) || 'general-purpose';
3038
+ const description = (typeof input.description === 'string' && input.description.trim())
3039
+ || String(input.prompt ?? '').trim().slice(0, 60) || agentType;
3040
+ const mode = (0, modePolicy_1.parseMode)(options.mode);
3041
+ const backgroundPermission = async (req) => {
3042
+ const rule = (0, rules_1.evaluatePermission)((0, rules_1.loadSettings)(options.workDir).permissions, req.tool, req.input, options.workDir);
3043
+ if (rule === 'deny')
3044
+ throw new ToolNotAllowedError(`\`${req.tool}\` is denied by a permission rule in this project.`);
3045
+ if (rule === 'allow')
3046
+ return true;
3047
+ const destructive = !!(0, destructive_1.isDestructiveBash)(req.tool, req.input);
3048
+ if ((0, modePolicy_1.decide)({ tool: req.tool, input: req.input, mode, destructive }) === 'allow')
3049
+ return true;
3050
+ throw new ToolNotAllowedError(`Background agents cannot ask the user for permission, and \`${req.tool}\` is not pre-approved in the ` +
3051
+ `current mode (${mode}). Do not retry it: finish what you can without it and say in your report that ` +
3052
+ 'this step needs the main agent (or the user) to run it.');
3053
+ };
3054
+ const started = hub.start({ description, agentType }, async ({ abort, onToolUse }) => {
3055
+ const didStart = { value: false };
3056
+ try {
3057
+ return await runSubTask(input, {
3058
+ ...options,
3059
+ abortSignal: abort,
3060
+ backgroundAgents: undefined,
3061
+ requestPermission: backgroundPermission,
3062
+ onText: () => { },
3063
+ onNotice: () => { },
3064
+ onToolUse: (name) => onToolUse(name),
3065
+ onToolResult: () => { },
3066
+ onToolStreamChunk: undefined,
3067
+ onThinking: undefined,
3068
+ onThinkingDelta: undefined,
3069
+ onThinkingProgress: undefined,
3070
+ onStreamRestart: undefined,
3071
+ onRetry: undefined,
3072
+ onRetryResolved: undefined,
3073
+ }, agentTypes, didStart);
3074
+ }
3075
+ finally {
3076
+ if (!didStart.value)
3077
+ refundSessionSubAgentSlot(options.sessionId);
3078
+ }
3079
+ });
3080
+ if ('error' in started) {
3081
+ refundSessionSubAgentSlot(options.sessionId);
3082
+ return { error: started.error };
3083
+ }
3084
+ return {
3085
+ output: `Started background agent ${started.id} ("${description}", ${agentType}). It runs independently: its ` +
3086
+ 'report will arrive automatically as a <task-notification> in a later turn. Do not poll for it, do not ' +
3087
+ 'wait for it, and do not duplicate its work — continue with something else, or end your turn if nothing ' +
3088
+ 'else needs doing now.',
3089
+ };
3090
+ }
2555
3091
  // ─── Agent Loop ───────────────────────────────────────────────────────────────
2556
3092
  async function runAgentLoop(initialMessages, options) {
2557
3093
  const messages = [...initialMessages];
2558
3094
  const model = options.model ?? 'turbo';
2559
- const hooks = loadHooks(options.workDir);
3095
+ // An agent definition's own hooks apply only to its own run (appended after the
3096
+ // project's, which run first) — runSubTask sets _agentHooks per child, never inherits it.
3097
+ const hooks = withAgentHooks(loadHooks(options.workDir), options._agentHooks);
2560
3098
  const depth = options._depth ?? 0;
2561
3099
  const agentScope = options._agentScope ?? 'root';
2562
3100
  // ── Audit trail (opt-in) ────────────────────────────────────────────────────
@@ -2628,6 +3166,62 @@ async function runAgentLoop(initialMessages, options) {
2628
3166
  options.worktree ? (0, worktreeEnforcement_1.worktreeInstructions)(options.worktree) : null,
2629
3167
  options.nexrallMd ?? null,
2630
3168
  ].filter((s) => s !== null).join('\n\n---\n\n') || undefined; // '' (nothing at all was present) must behave exactly like the old `: options.nexrallMd` fallback (undefined), not an empty-but-truthy string.
3169
+ // The request options every main-loop call sends. Shared with the compaction
3170
+ // summariser so its request has the SAME system prompt and tools as the conversation
3171
+ // — that is what lets the summary call read the whole history from cache (0.10×)
3172
+ // instead of re-sending it as a fresh, uncached transcript.
3173
+ const chatRequestOptions = () => ({
3174
+ model,
3175
+ mode: options.mode,
3176
+ effort: options.effort,
3177
+ env: options.env,
3178
+ editorContext: options.editorContext,
3179
+ nexrallMd: planAndWorktreeAwareNexrallMd,
3180
+ clientType: options.clientType,
3181
+ abortSignal: options.abortSignal,
3182
+ // MCP tools, plus the per-agent memory tool when (and only when) this run
3183
+ // is a sub-agent that declared a `memory:` scope. Declaring it through
3184
+ // extraTools rather than the backend's static catalogue keeps it invisible
3185
+ // to every other run: a tool the model cannot see is one it cannot try,
3186
+ // which is better than advertising it everywhere and refusing it at the gate.
3187
+ extraTools: [
3188
+ ...(options.mcpManager?.getAnthropicTools() ?? [])
3189
+ .filter((t) => !options._mcpServerAllowlist || options._mcpServerAllowlist.has(String(t.name).split('__')[0])),
3190
+ ...(options._agentMemory ? [exports.AGENT_MEMORY_TOOL_SCHEMA] : []),
3191
+ ],
3192
+ // Derived from the run's ACTUAL allowlist rather than asserted separately,
3193
+ // so the prompt's memory instructions cannot drift from what is permitted.
3194
+ // That drift is the bug being fixed: sub-agents were told they MUST call
3195
+ // memory_write, which no sub-agent allowlist contains.
3196
+ // Sub-agents NEVER write the shared store (the main agent decides what to persist) —
3197
+ // an unnamed/general-purpose child has no allowlist, which used to read as "may write".
3198
+ canWriteSharedMemory: depth > 0
3199
+ ? false
3200
+ : options._allowedTools
3201
+ ? options._allowedTools.has('memory_write')
3202
+ : true, // no allowlist = main agent = may write
3203
+ hasOwnAgentStore: !!options._agentMemory,
3204
+ // Same derivation, same reason, for the network. The base prompt's
3205
+ // "Web search" section tells the run to reach the network, but
3206
+ // READ_ONLY_TOOLS (explorer, security-auditor, ...) contains neither
3207
+ // `fetch_url` nor `web_search`. MEASURED on the 195-trial 2026-08-10
3208
+ // deepseek-v4-pro benchmark: 29 of 51 `fetch_url` failures and 3 of 14
3209
+ // `web_search` failures were the permission gate refusing a sub-agent
3210
+ // that its own system prompt had just invited to browse — every one a
3211
+ // wasted round-trip on a paid turn.
3212
+ canBrowseWeb: options._allowedTools
3213
+ ? options._allowedTools.has('fetch_url') || options._allowedTools.has('web_search')
3214
+ : true, // no allowlist = main agent = may browse
3215
+ // Only the tools this run may use are sent as schemas — explorer no longer pays for
3216
+ // (and gets refused on) write, office and image tools. Sorted so the tools cache
3217
+ // prefix is identical for every run of the same agent type.
3218
+ toolAllowlist: options._allowedTools ? [...options._allowedTools].sort() : undefined,
3219
+ // Withholds the `task` schema and its instructions when this run cannot
3220
+ // delegate — see canSpawnSubAgents.
3221
+ canSpawnSubAgents: maySpawn,
3222
+ agents: agentsCatalogue || undefined,
3223
+ skills: skillsCatalogue || undefined,
3224
+ });
2631
3225
  // Optional OS-level bash sandbox (opt-in via settings.json "sandbox").
2632
3226
  const sandboxCfg = (0, sandbox_1.parseSandboxConfig)(settings.raw.sandbox) ?? undefined;
2633
3227
  // Soft iteration budget + optional auto-continue past it (see resolvers above).
@@ -2638,6 +3232,11 @@ async function runAgentLoop(initialMessages, options) {
2638
3232
  // Live catalogue first (see contextWindowFor's doc comment above for why),
2639
3233
  // same fallback chain this call site always used otherwise.
2640
3234
  const contextWindow = (0, modelCatalogue_1.liveContextWindowFor)(model, MODEL_CONTEXT_TOKENS[model] ?? 200000);
3235
+ const compactLimits = compactionLimits(contextWindow, settings.raw);
3236
+ // Per-run: a sub-agent's own runAgentLoop gets its own, so it never sees its parent's reads.
3237
+ const readDedupe = new readDedupe_1.ReadDedupe(options.workDir);
3238
+ // Only the main agent owns background agents (runSubTask never passes the hub down).
3239
+ const drainBackgroundNotifications = () => depth === 0 ? (options.backgroundAgents?.takeNotifications() ?? []) : [];
2641
3240
  // Live prompt-size estimate, updated from usage events after every stream.
2642
3241
  let lastPromptTokens = 0;
2643
3242
  let compacting = false; // re-entrancy guard — compaction itself calls streamChat
@@ -2759,22 +3358,22 @@ async function runAgentLoop(initialMessages, options) {
2759
3358
  // every single turn on a long task. Runs at a turn boundary only.
2760
3359
  //
2761
3360
  // THREE nested triggers, cheapest/earliest first:
2762
- // 1. PRUNE pressure — prompt crossed AUTO_PRUNE_THRESHOLD (~35% of window)
3361
+ // 1. PRUNE pressure — prompt crossed compactLimits.prune (~120K tokens)
2763
3362
  // OR the body crossed MAX_BODY_BYTES. Handled by the CHEAP, structure-
2764
3363
  // preserving prune (no model call, keeps every turn). This is the big
2765
3364
  // cost win: it fires ~2× earlier than summarisation used to, shedding
2766
3365
  // already-consumed tool_result bulk (read_file/bash/grep output the
2767
3366
  // model has long since acted on) so the per-turn cache-read bill stops
2768
3367
  // compounding well before the old 80% wall.
2769
- // 2. SUMMARISE pressure — prompt crossed AUTO_COMPACT_THRESHOLD (~80%) or
3368
+ // 2. SUMMARISE pressure — prompt crossed compactLimits.compact (~167K, Claude Code's line) or
2770
3369
  // the body is STILL over MAX_BODY_BYTES after pruning. Only then do we
2771
3370
  // pay for a summariser call + drop whole turns (kept late on purpose:
2772
3371
  // summarise-of-summarise is what makes an agent "forget" earlier work).
2773
3372
  // The byte trigger also fires even on turn 0 of a resumed large session,
2774
3373
  // where lastPromptTokens is 0.
2775
3374
  let bodyBytes = estimateBodyBytes(messages);
2776
- const prunePressure = lastPromptTokens > contextWindow * AUTO_PRUNE_THRESHOLD;
2777
- const tokenPressure = lastPromptTokens > contextWindow * AUTO_COMPACT_THRESHOLD;
3375
+ const prunePressure = lastPromptTokens > compactLimits.prune;
3376
+ const tokenPressure = lastPromptTokens > compactLimits.compact;
2778
3377
  let bytePressure = bodyBytes > MAX_BODY_BYTES;
2779
3378
  // Cheap prune first — on token OR byte pressure. Require a meaningful reclaim
2780
3379
  // (PRUNE_MIN_RECLAIM_BYTES): a tiny prune would bust the message-level prompt
@@ -2785,16 +3384,18 @@ async function runAgentLoop(initialMessages, options) {
2785
3384
  // The gate lives INSIDE pruneOldToolResults now (atomic: it measures the
2786
3385
  // total first and mutates nothing if it's below the floor), so a declined
2787
3386
  // prune never invalidates the prompt cache.
2788
- const reclaimed = pruneOldToolResults(messages, PRUNE_MIN_RECLAIM_BYTES);
3387
+ // Main agent only: a sub-agent's own cache is warm while it works, but it would read
3388
+ // the PARENT's last-call time, which goes stale exactly while the parent waits on it.
3389
+ const floor = depth === 0
3390
+ ? pruneReclaimFloor(_lastApiCallEndedAt.get(options.sessionId || PROCESS_BUDGET_KEY))
3391
+ : PRUNE_MIN_RECLAIM_BYTES;
3392
+ const reclaimed = pruneOldToolResults(messages, floor);
2789
3393
  if (reclaimed > 0) {
2790
3394
  bodyBytes = estimateBodyBytes(messages);
2791
3395
  bytePressure = bodyBytes > MAX_BODY_BYTES;
2792
- // System notice, NOT model output — route through onNotice (falls back to
2793
- // onText only if the caller hasn't been updated yet) so the UI can render
2794
- // it as a small dim/system line instead of splicing it into the middle of
2795
- // the assistant's own streamed reply, where it reads like the model itself
2796
- // said "Trimmed ~0.3MB of already-processed tool output…".
2797
- (options.onNotice ?? options.onText)(`\u267b\ufe0f Trimmed ~${(reclaimed / (1024 * 1024)).toFixed(1)}MB of already-processed tool output to keep this chat cheap to continue.`);
3396
+ // Silent by design: routine housekeeping the user can't act on. It used to
3397
+ // print "♻️ Trimmed ~0.1MB of already-processed tool output…" into the chat,
3398
+ // which read as noise (and, before onNotice, as the model's own words).
2798
3399
  }
2799
3400
  }
2800
3401
  // Re-arm the breaker once the conversation has grown materially past the
@@ -2817,7 +3418,7 @@ async function runAgentLoop(initialMessages, options) {
2817
3418
  let did = false;
2818
3419
  let unavailable = false;
2819
3420
  try {
2820
- did = await autoCompactMessages(messages, options, ledger);
3421
+ did = await autoCompactMessages(messages, options, ledger, makeCachedSummarizer(messages, chatRequestOptions, options.abortSignal));
2821
3422
  }
2822
3423
  catch (err) {
2823
3424
  if (!(err instanceof CompactionUnavailableError))
@@ -2831,9 +3432,9 @@ async function runAgentLoop(initialMessages, options) {
2831
3432
  const reason = bytePressure
2832
3433
  ? `body ~${(bodyBytes / (1024 * 1024)).toFixed(1)}MB`
2833
3434
  : 'context window';
2834
- // Same reasoning as above: this is a system notice about housekeeping,
2835
- // not part of the model's answer — keep it out of the text bubble.
2836
- (options.onNotice ?? options.onText)(`\u267b\ufe0f Auto-compacted earlier conversation to stay within the ${reason}.`);
3435
+ // Silent, same as the prune above — routine housekeeping. Only the
3436
+ // actionable "auto-compaction paused" warning below is surfaced.
3437
+ void reason;
2837
3438
  // Refresh local pressure so the rest of THIS iteration sees the new size.
2838
3439
  bodyBytes = bytesAfter;
2839
3440
  bytePressure = bodyBytes > MAX_BODY_BYTES;
@@ -2875,6 +3476,16 @@ async function runAgentLoop(initialMessages, options) {
2875
3476
  if (!autoContinue && iteration === budget - 5 && budget > 5) {
2876
3477
  options.onText(`\n⚠️ Approaching the ${budget}-step limit (step ${iteration + 1}). Please wrap up and summarise what has been done.\n`);
2877
3478
  }
3479
+ // Runtime-context block the backend appended to this request's final user message
3480
+ // (see SSERuntimeContextEvent). Applied only AFTER streamChat succeeds: mutating
3481
+ // `messages` mid-stream would change the request hash a same-turnId retry sends,
3482
+ // and the backend would then run (and bill) it as a brand-new turn.
3483
+ let pendingRuntimeContext = null;
3484
+ const sentUserIdx = messages.length - 1;
3485
+ // Pure-read tools announced mid-stream (tool_use_ready) start here, before the
3486
+ // message is complete. Results are only reused after every gate below passes.
3487
+ const prefetcher = new toolPrefetch_1.ToolPrefetcher((toolName, toolInput) => (0, executor_1.executeTool)(toolName, toolInput, options.abortSignal, sandboxCfg, options.workDir, agentScope, undefined, options.selfPeer));
3488
+ const prefetchOn = (0, toolPrefetch_1.toolPrefetchEnabled)();
2878
3489
  // Build SSE event handler
2879
3490
  const onEvent = (event) => {
2880
3491
  switch (event.type) {
@@ -2944,11 +3555,20 @@ async function runAgentLoop(initialMessages, options) {
2944
3555
  // an assumption: tools are dispatched from `assistantMessage.content` AFTER
2945
3556
  // streamChat() returns, and a restarted attempt threw instead of returning.
2946
3557
  // So no tool ran, no file changed, no checkpoint or hook fired.
3558
+ // The replacement attempt generates new tool_use ids; drop early reads.
3559
+ prefetcher.clear();
2947
3560
  options.onStreamRestart?.(event.reason, event.discardedChars);
2948
3561
  break;
2949
3562
  case 'balance_status':
2950
3563
  options.onBalanceStatus?.(event.balance, event.zero);
2951
3564
  break;
3565
+ case 'runtime_context':
3566
+ pendingRuntimeContext = event.text;
3567
+ break;
3568
+ case 'tool_use_ready':
3569
+ if (prefetchOn && !options.abortSignal?.aborted)
3570
+ prefetcher.start(event.id, event.name, event.input);
3571
+ break;
2952
3572
  case 'done':
2953
3573
  break;
2954
3574
  case 'error':
@@ -2966,47 +3586,7 @@ async function runAgentLoop(initialMessages, options) {
2966
3586
  const _apiCallStartedAt = Date.now();
2967
3587
  try {
2968
3588
  assistantMessage = await (0, client_1.streamChat)(messages, {
2969
- model,
2970
- mode: options.mode,
2971
- effort: options.effort,
2972
- env: options.env,
2973
- editorContext: options.editorContext,
2974
- nexrallMd: planAndWorktreeAwareNexrallMd,
2975
- clientType: options.clientType,
2976
- abortSignal: options.abortSignal,
2977
- // MCP tools, plus the per-agent memory tool when (and only when) this run
2978
- // is a sub-agent that declared a `memory:` scope. Declaring it through
2979
- // extraTools rather than the backend's static catalogue keeps it invisible
2980
- // to every other run: a tool the model cannot see is one it cannot try,
2981
- // which is better than advertising it everywhere and refusing it at the gate.
2982
- extraTools: [
2983
- ...(options.mcpManager?.getAnthropicTools() ?? []),
2984
- ...(options._agentMemory ? [exports.AGENT_MEMORY_TOOL_SCHEMA] : []),
2985
- ],
2986
- // Derived from the run's ACTUAL allowlist rather than asserted separately,
2987
- // so the prompt's memory instructions cannot drift from what is permitted.
2988
- // That drift is the bug being fixed: sub-agents were told they MUST call
2989
- // memory_write, which no sub-agent allowlist contains.
2990
- canWriteSharedMemory: options._allowedTools
2991
- ? options._allowedTools.has('memory_write')
2992
- : true, // no allowlist = main agent = may write
2993
- hasOwnAgentStore: !!options._agentMemory,
2994
- // Same derivation, same reason, for the network. The base prompt's
2995
- // "Web search" section tells the run to reach the network, but
2996
- // READ_ONLY_TOOLS (explorer, security-auditor, ...) contains neither
2997
- // `fetch_url` nor `web_search`. MEASURED on the 195-trial 2026-08-10
2998
- // deepseek-v4-pro benchmark: 29 of 51 `fetch_url` failures and 3 of 14
2999
- // `web_search` failures were the permission gate refusing a sub-agent
3000
- // that its own system prompt had just invited to browse — every one a
3001
- // wasted round-trip on a paid turn.
3002
- canBrowseWeb: options._allowedTools
3003
- ? options._allowedTools.has('fetch_url')
3004
- : true, // no allowlist = main agent = may browse
3005
- // Withholds the `task` schema and its instructions when this run cannot
3006
- // delegate — see canSpawnSubAgents.
3007
- canSpawnSubAgents: maySpawn,
3008
- agents: agentsCatalogue || undefined,
3009
- skills: skillsCatalogue || undefined,
3589
+ ...chatRequestOptions(),
3010
3590
  // Only allow a post-render restart when the caller actually implements the
3011
3591
  // rollback. Without a handler the partial output can't be un-rendered, so we
3012
3592
  // keep the old conservative behaviour (fail the turn) rather than duplicate
@@ -3016,6 +3596,12 @@ async function runAgentLoop(initialMessages, options) {
3016
3596
  // Measured on the SUCCESS path only — a throw below reports its own duration
3017
3597
  // in the catch block, so this can't double-report the same call.
3018
3598
  options.onApiCallDuration?.(Date.now() - _apiCallStartedAt);
3599
+ if (depth === 0)
3600
+ _lastApiCallEndedAt.set(options.sessionId || PROCESS_BUDGET_KEY, Date.now());
3601
+ // Persist the context block where the backend put it, so later rounds carry it
3602
+ // in the cached history and the backend stops re-sending it uncached.
3603
+ if (pendingRuntimeContext)
3604
+ persistRuntimeContext(messages, sentUserIdx, pendingRuntimeContext);
3019
3605
  }
3020
3606
  catch (err) {
3021
3607
  options.onApiCallDuration?.(Date.now() - _apiCallStartedAt);
@@ -3059,19 +3645,19 @@ async function runAgentLoop(initialMessages, options) {
3059
3645
  stopReason = options.onBalanceStatus ? 'reported-elsewhere' : 'no-balance';
3060
3646
  break;
3061
3647
  }
3062
- // A 403 naming a TEAM problem means the "Bill to" choice is no longer usable:
3063
- // the member was removed, the team was suspended/deleted, or the account never
3064
- // had access. Reset the stored choice so the NEXT turn bills the personal wallet
3065
- // instead of failing the same way forever — and SAY so (stopReasonNotice),
3066
- // because a silent downgrade after the user explicitly chose a company is the
3067
- // trust bug docs/TEAMS_ENTERPRISE_BUDGETS_2026-09.md §D4 warns about. The reset
3068
- // is what makes "send continue" work; the notice is what makes it honest.
3648
+ // A 403 naming a TEAM problem means the team that was set to pay cannot —
3649
+ // the member was removed, or the team was suspended. The SAVED choice is
3650
+ // cleaned up by the server (services/teams.js resolveBillingTarget clears
3651
+ // a permanently-unusable preference itself — migration 140), so there is
3652
+ // nothing to clear on this side any more; what remains is telling the
3653
+ // user, because a silent downgrade to their own wallet after they chose a
3654
+ // company is the trust bug docs/TEAMS_ENTERPRISE_BUDGETS_2026-09.md §D4
3655
+ // warns about.
3069
3656
  if (status === 403 && typeof code === 'string' && TEAM_SCOPE_ERROR_CODES.has(code)) {
3070
- (0, index_1.setBillingTeamId)(null);
3071
3657
  stopReason = 'team-unavailable';
3072
3658
  break;
3073
3659
  }
3074
- runSimpleHooks(hooks.OnError, options.workDir);
3660
+ await runSimpleHooks(hooks.OnError, options.workDir);
3075
3661
  // Carry the work already done in this turn out with the error. The loop
3076
3662
  // owns a COPY of the caller's history, so a plain throw would strand every
3077
3663
  // completed tool round inside this function and the caller would fall back
@@ -3095,7 +3681,8 @@ async function runAgentLoop(initialMessages, options) {
3095
3681
  if (assistantMessage.content.length === 0) {
3096
3682
  const queued = options.takePendingInput?.() ?? [];
3097
3683
  const peerMsgs = options.drainPeerMessages?.() ?? [];
3098
- if (queued.length || peerMsgs.length) {
3684
+ const bgDone = drainBackgroundNotifications();
3685
+ if (queued.length || peerMsgs.length || bgDone.length) {
3099
3686
  const parts = [];
3100
3687
  if (queued.length) {
3101
3688
  parts.push(queued.join('\n\n'));
@@ -3105,6 +3692,8 @@ async function runAgentLoop(initialMessages, options) {
3105
3692
  parts.push(renderPeerMessages(peerMsgs));
3106
3693
  options.onPeerMessage?.(peerMsgs);
3107
3694
  }
3695
+ if (bgDone.length)
3696
+ parts.push((0, backgroundAgents_1.formatBackgroundNotifications)(bgDone));
3108
3697
  messages.push({ role: 'user', content: [{ type: 'text', text: parts.join('\n\n') }] });
3109
3698
  continue;
3110
3699
  }
@@ -3148,7 +3737,8 @@ async function runAgentLoop(initialMessages, options) {
3148
3737
  iteration--;
3149
3738
  continue;
3150
3739
  }
3151
- runSimpleHooks(hooks.PostMessageComplete, options.workDir);
3740
+ if (depth === 0)
3741
+ await runSimpleHooks(hooks.PostMessageComplete, options.workDir);
3152
3742
  stopReason = assistantMessage.stopReason === 'max_tokens' ? 'output-limit' : 'empty-response';
3153
3743
  break;
3154
3744
  }
@@ -3201,7 +3791,8 @@ async function runAgentLoop(initialMessages, options) {
3201
3791
  if (toolUseBlocks.length === 0) {
3202
3792
  const queued = options.takePendingInput?.() ?? [];
3203
3793
  const peerMsgs = options.drainPeerMessages?.() ?? [];
3204
- if (queued.length || peerMsgs.length) {
3794
+ const bgDone = drainBackgroundNotifications();
3795
+ if (queued.length || peerMsgs.length || bgDone.length) {
3205
3796
  const parts = [];
3206
3797
  if (queued.length) {
3207
3798
  parts.push(queued.join('\n\n'));
@@ -3211,6 +3802,8 @@ async function runAgentLoop(initialMessages, options) {
3211
3802
  parts.push(renderPeerMessages(peerMsgs));
3212
3803
  options.onPeerMessage?.(peerMsgs);
3213
3804
  }
3805
+ if (bgDone.length)
3806
+ parts.push((0, backgroundAgents_1.formatBackgroundNotifications)(bgDone));
3214
3807
  messages.push({ role: 'user', content: [{ type: 'text', text: parts.join('\n\n') }] });
3215
3808
  continue;
3216
3809
  }
@@ -3308,15 +3901,21 @@ async function runAgentLoop(initialMessages, options) {
3308
3901
  continue;
3309
3902
  }
3310
3903
  }
3311
- runSimpleHooks(hooks.PostMessageComplete, options.workDir);
3904
+ if (depth === 0)
3905
+ await runSimpleHooks(hooks.PostMessageComplete, options.workDir);
3312
3906
  stopReason = 'clean';
3313
3907
  break;
3314
3908
  }
3315
3909
  // 5. Execute all tool uses in parallel
3910
+ // Early reads are only valid if nothing in this message can change the files
3911
+ // they read (a write ordered before them would otherwise be missed).
3912
+ const prefetchSafe = prefetchOn && (0, toolPrefetch_1.isPrefetchSafeMessage)(toolUseBlocks.map((b) => b.name));
3316
3913
  const toolResults = await Promise.all(toolUseBlocks.map(async (block) => {
3317
3914
  const { id, name, input } = block;
3318
- // Notify caller about pending tool use
3319
- options.onToolUse(name, input);
3915
+ // Notify caller about pending tool use. `meta.id` lets the UI pair the result
3916
+ // with THIS row even when parallel sub-agents interleave their events.
3917
+ const toolMeta = { id };
3918
+ options.onToolUse(name, input, undefined, toolMeta);
3320
3919
  let result;
3321
3920
  // Checked BEFORE requesting permission: no point asking the user to
3322
3921
  // approve a diff/preview built from possibly-garbage truncated input
@@ -3339,7 +3938,7 @@ async function runAgentLoop(initialMessages, options) {
3339
3938
  : '') +
3340
3939
  `so the full response fits comfortably under the per-turn output budget.`,
3341
3940
  };
3342
- options.onToolResult(name, result);
3941
+ options.onToolResult(name, result, undefined, toolMeta);
3343
3942
  return { block: { ...block, id }, result };
3344
3943
  }
3345
3944
  // ── Plan mode: a session-wide read-only lock ────────────────────────
@@ -3353,7 +3952,7 @@ async function runAgentLoop(initialMessages, options) {
3353
3952
  const refusal = (0, planMode_1.checkPlanMode)(name, input);
3354
3953
  if (refusal) {
3355
3954
  result = { error: refusal.message };
3356
- options.onToolResult(name, result);
3955
+ options.onToolResult(name, result, undefined, toolMeta);
3357
3956
  return { block: { ...block, id }, result };
3358
3957
  }
3359
3958
  }
@@ -3367,7 +3966,7 @@ async function runAgentLoop(initialMessages, options) {
3367
3966
  const refusal = (0, worktreeEnforcement_1.checkWorktreeIsolation)(name, input, options.worktree, options.workDir);
3368
3967
  if (refusal) {
3369
3968
  result = { error: refusal.message };
3370
- options.onToolResult(name, result);
3969
+ options.onToolResult(name, result, undefined, toolMeta);
3371
3970
  return { block: { ...block, id }, result };
3372
3971
  }
3373
3972
  }
@@ -3398,10 +3997,36 @@ async function runAgentLoop(initialMessages, options) {
3398
3997
  // depth 0) needs its own limiter. Extracted into dispatchSubAgent so
3399
3998
  // `@nexrall/agent`'s delegate()/spawn() drive the SAME path instead of
3400
3999
  // reimplementing depth/budget/concurrency enforcement a second time.
3401
- result = await dispatchSubAgent(input, auditOptions, agentTypes);
4000
+ // Background only for the MAIN agent, and only when the host can deliver the
4001
+ // result later; otherwise it quietly runs in the foreground.
4002
+ const wantsBackground = input.run_in_background === true
4003
+ || (input.run_in_background !== false && !!(0, agentTypes_1.findAgentType)(agentTypes, String(input.subagent_type ?? ''))?.background);
4004
+ const taskOptions = { ...auditOptions, _taskToolUseId: id };
4005
+ // PreToolUse/PostToolUse cover spawns too (Claude Code's matcher "Task"/"Agent"
4006
+ // works the same way): a hook can veto a delegation or audit what came back.
4007
+ const preTask = await runToolHooks(hooks.PreToolUse, 'PreToolUse', name, input, options.workDir);
4008
+ if (preTask.block) {
4009
+ result = { error: `Blocked by PreToolUse hook: ${preTask.reason}` };
4010
+ }
4011
+ else {
4012
+ result = wantsBackground && depth === 0 && options.backgroundAgents && !input.resume_agent_id
4013
+ ? dispatchBackgroundSubAgent(input, taskOptions, agentTypes, options.backgroundAgents)
4014
+ : await dispatchSubAgent(input, taskOptions, agentTypes);
4015
+ const postTask = await runToolHooks(hooks.PostToolUse, 'PostToolUse', name, input, options.workDir, result);
4016
+ const injectedTask = [preTask.context, postTask.context].filter(Boolean).join('\n');
4017
+ if (injectedTask) {
4018
+ if (result.error !== undefined)
4019
+ result.error = `${result.error}\n\n[hook] ${injectedTask}`;
4020
+ else
4021
+ result.output = `${result.output ?? ''}\n\n[hook] ${injectedTask}`;
4022
+ }
4023
+ if (postTask.block) {
4024
+ result.error = `${result.error ? result.error + '\n' : ''}PostToolUse hook flagged: ${postTask.reason}`;
4025
+ }
4026
+ }
3402
4027
  }
3403
4028
  else {
3404
- const pre = runToolHooks(hooks.PreToolUse, 'PreToolUse', name, input, options.workDir);
4029
+ const pre = await runToolHooks(hooks.PreToolUse, 'PreToolUse', name, input, options.workDir);
3405
4030
  if (pre.block) {
3406
4031
  result = { error: `Blocked by PreToolUse hook: ${pre.reason}` };
3407
4032
  }
@@ -3413,27 +4038,32 @@ async function runAgentLoop(initialMessages, options) {
3413
4038
  // could not tell whose notes to write to (and must not be able to).
3414
4039
  if (name === exports.AGENT_MEMORY_TOOL) {
3415
4040
  result = await executeAgentMemoryWrite(input, options._agentMemory, options.workDir);
3416
- options.onToolResult(name, result);
4041
+ options.onToolResult(name, result, undefined, toolMeta);
3417
4042
  return { block: { ...block, id }, result };
3418
4043
  }
3419
4044
  // 1. Try platform-specific tools (e.g. VS Code semantic tools)
4045
+ // Both of these run code we do not control (the editor, a third-party MCP
4046
+ // server) and may ignore Stop; stopAwaiting returns an "interrupted" result
4047
+ // a few seconds after Stop instead of leaving the turn hanging on them.
3420
4048
  const external = options.executeExternalTool
3421
- ? await options.executeExternalTool(name, input)
4049
+ ? await stopAwaiting(options.executeExternalTool(name, input), options.abortSignal)
3422
4050
  : null;
3423
4051
  if (external !== null && external !== undefined) {
3424
4052
  result = external;
3425
4053
  // 2. Try MCP tools (serverName__toolName)
3426
4054
  }
3427
4055
  else if (options.mcpManager?.isMcpTool(name)) {
3428
- const mcpOutput = await options.mcpManager.callTool(name, input, options.abortSignal);
3429
- result = { output: mcpOutput ?? '' };
4056
+ const mcpOutput = await stopAwaiting(options.mcpManager.callTool(name, input, options.abortSignal), options.abortSignal);
4057
+ result = mcpOutput === STOP_TIMEOUT_RESULT
4058
+ ? STOP_TIMEOUT_RESULT
4059
+ : { output: (0, executor_1.capExternalOutput)(mcpOutput ?? '', `mcp-${name}`) };
3430
4060
  // 3. Fall through to built-in executor
3431
4061
  }
3432
4062
  else {
3433
4063
  // Snapshot pre-mutation state so the user can /rewind this turn.
3434
4064
  options.checkpointManager?.recordBeforeMutation(name, input);
3435
4065
  const onStream = options.onToolStreamChunk
3436
- ? (chunk) => options.onToolStreamChunk(name, chunk)
4066
+ ? (chunk) => options.onToolStreamChunk(name, chunk, undefined, toolMeta)
3437
4067
  : undefined;
3438
4068
  const run = () => (0, executor_1.executeTool)(name, input, options.abortSignal, sandboxCfg, options.workDir, agentScope, onStream, options.selfPeer);
3439
4069
  // Serialise anything that would otherwise race. Two sources of races:
@@ -3464,9 +4094,31 @@ async function runAgentLoop(initialMessages, options) {
3464
4094
  // Nesting cross-process OUTSIDE in-process means a process holding
3465
4095
  // the OS lock for a group of sub-agents takes it exactly ONCE for the
3466
4096
  // whole group, not once per sub-agent.
3467
- result = locks.length
3468
- ? await (0, crossProcessLock_1.withCrossProcessLocks)(locks, () => withFileLocks(locks, run), options.workDir)
3469
- : await run();
4097
+ // Unchanged re-read of a range whose earlier result is still in the
4098
+ // history → one-line stub instead of the same content again (readDedupe.ts).
4099
+ const dedupeStub = name === 'read_file' ? readDedupe.check(input, messages) : null;
4100
+ const prefetched = !dedupeStub && prefetchSafe ? prefetcher.take(id, name, input) : null;
4101
+ if (dedupeStub) {
4102
+ result = { output: dedupeStub };
4103
+ }
4104
+ else if (prefetched) {
4105
+ result = await prefetched;
4106
+ if (name === 'read_file' && result.error === undefined && result.output?.startsWith('[File:')) {
4107
+ readDedupe.record(input, id);
4108
+ }
4109
+ }
4110
+ else {
4111
+ result = locks.length
4112
+ ? await (0, crossProcessLock_1.withCrossProcessLocks)(locks, () => withFileLocks(locks, run), options.workDir)
4113
+ : await run();
4114
+ if (name === 'read_file' && result.error === undefined && result.output?.startsWith('[File:')) {
4115
+ readDedupe.record(input, id);
4116
+ }
4117
+ }
4118
+ // A write tool touched these paths — drop their recorded reads even if
4119
+ // mtime happens to be unchanged (same-millisecond edit).
4120
+ if (name !== 'bash' && locks.length)
4121
+ readDedupe.invalidate(locks);
3470
4122
  }
3471
4123
  }
3472
4124
  catch (err) {
@@ -3481,7 +4133,7 @@ async function runAgentLoop(initialMessages, options) {
3481
4133
  void maybeCompactMemory(scope, options);
3482
4134
  }
3483
4135
  // PostToolUse can inject context for the model or flag a problem.
3484
- const post = runToolHooks(hooks.PostToolUse, 'PostToolUse', name, input, options.workDir, result);
4136
+ const post = await runToolHooks(hooks.PostToolUse, 'PostToolUse', name, input, options.workDir, result);
3485
4137
  const injected = [pre.context, post.context].filter(Boolean).join('\n');
3486
4138
  if (injected) {
3487
4139
  if (result.error !== undefined)
@@ -3504,7 +4156,7 @@ async function runAgentLoop(initialMessages, options) {
3504
4156
  result.output = (0, testIntegrity_1.stripTestIntegrityMarker)(result.output);
3505
4157
  }
3506
4158
  // Notify caller about result
3507
- options.onToolResult(name, result);
4159
+ options.onToolResult(name, result, undefined, toolMeta);
3508
4160
  return { block: { ...block, id }, result, rawOutput };
3509
4161
  }));
3510
4162
  // Track whether files were mutated / verified this run, for the one-shot
@@ -3525,6 +4177,20 @@ async function runAgentLoop(initialMessages, options) {
3525
4177
  if (auditCtx) {
3526
4178
  (0, audit_1.recordAudit)(auditCtx, block.name, block.input, ok, result.output, result.error, result.exitCode);
3527
4179
  }
4180
+ // Build/test commands a sub-agent ran count for THIS run too — otherwise the main
4181
+ // agent, reporting "tests pass" from a sub-agent's verified run, was told no test
4182
+ // had been run and re-ran the whole suite. Counted even when the task errored
4183
+ // (a stopped sub-agent may still have run the suite).
4184
+ if (block.name === 'task' && result.childVerifications?.length) {
4185
+ for (const v of result.childVerifications) {
4186
+ ledger.verifications.push({ cmd: `(sub-agent) ${v.cmd}`.slice(0, 120), ok: v.ok, epoch: ledger.epoch });
4187
+ }
4188
+ const lastChild = result.childVerifications[result.childVerifications.length - 1];
4189
+ if (lastChild.ok) {
4190
+ ranVerificationCmd = true;
4191
+ filesMutatedSinceVerify = false;
4192
+ }
4193
+ }
3528
4194
  if (!ok)
3529
4195
  continue; // failed calls don't count either way
3530
4196
  if (exports.WRITE_TOOL_NAMES.has(block.name))
@@ -3576,6 +4242,12 @@ async function runAgentLoop(initialMessages, options) {
3576
4242
  options.onPeerMessage?.(inboundPeerMessages);
3577
4243
  toolResultContent.push({ type: 'text', text: renderPeerMessages(inboundPeerMessages) });
3578
4244
  }
4245
+ // Background agents that finished while this round ran — same boundary, same
4246
+ // reason: a notification must never interrupt an in-flight tool call.
4247
+ const bgFinished = drainBackgroundNotifications();
4248
+ if (bgFinished.length) {
4249
+ toolResultContent.push({ type: 'text', text: (0, backgroundAgents_1.formatBackgroundNotifications)(bgFinished) });
4250
+ }
3579
4251
  const toolResultMessage = {
3580
4252
  role: 'user',
3581
4253
  content: toolResultContent,
@@ -3654,8 +4326,13 @@ async function runAgentLoop(initialMessages, options) {
3654
4326
  // message in this file — these are statements from the harness, not from the model,
3655
4327
  // and splicing them into the assistant's own bubble reads as if it said them.
3656
4328
  const notice = stopReasonNotice(stopReason, { budget, repeatError: stalledRepeatError });
3657
- if (notice)
4329
+ // A sub-agent's stop belongs in its parent's tool result, not in the user's chat as
4330
+ // if the MAIN agent had stopped (runSubTask turns it into an error for the parent).
4331
+ if (options._onStopReason)
4332
+ options._onStopReason(stopReason, notice || null);
4333
+ else if (notice)
3658
4334
  (options.onNotice ?? options.onText)(notice);
4335
+ options._onVerifications?.(ledger.verifications.map((v) => ({ cmd: v.cmd, ok: v.ok })));
3659
4336
  }
3660
4337
  catch (err) {
3661
4338
  // Every other failure path (a tool executor blowing up, a hook throwing, an
@@ -3670,7 +4347,12 @@ async function runAgentLoop(initialMessages, options) {
3670
4347
  throw new types_1.AgentTurnError(err?.message || 'The turn ended unexpectedly', trimToResumableBoundary(messages), completedRounds, err);
3671
4348
  }
3672
4349
  finally {
3673
- runSimpleHooks(hooks.OnStop, options.workDir);
4350
+ // OnStop is the MAIN agent's turn ending. Firing it at the end of every sub-agent
4351
+ // ran the user's "turn finished" hook (notifications, formatters) mid-turn.
4352
+ if (depth === 0)
4353
+ await runSimpleHooks(hooks.OnStop, options.workDir);
4354
+ else
4355
+ await runSimpleHooks(hooks.SubagentStop, options.workDir, { NEXRALL_AGENT_NAME: options._agentTypeName ?? 'general-purpose' });
3674
4356
  }
3675
4357
  return messages;
3676
4358
  }