@nexrall/code-core 1.4.63 → 1.4.65

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/dist/agent/agentRegistry.d.ts +14 -3
  2. package/dist/agent/agentRegistry.d.ts.map +1 -1
  3. package/dist/agent/agentRegistry.js +136 -6
  4. package/dist/agent/agentTypes.d.ts +51 -1
  5. package/dist/agent/agentTypes.d.ts.map +1 -1
  6. package/dist/agent/agentTypes.js +194 -11
  7. package/dist/agent/backgroundAgents.d.ts +66 -0
  8. package/dist/agent/backgroundAgents.d.ts.map +1 -0
  9. package/dist/agent/backgroundAgents.js +145 -0
  10. package/dist/agent/loop.d.ts +43 -6
  11. package/dist/agent/loop.d.ts.map +1 -1
  12. package/dist/agent/loop.js +942 -259
  13. package/dist/agent/modelCatalogue.d.ts +15 -0
  14. package/dist/agent/modelCatalogue.d.ts.map +1 -1
  15. package/dist/agent/modelCatalogue.js +46 -0
  16. package/dist/agent/readDedupe.d.ts +15 -0
  17. package/dist/agent/readDedupe.d.ts.map +1 -0
  18. package/dist/agent/readDedupe.js +146 -0
  19. package/dist/agent/toolPrefetch.d.ts +44 -0
  20. package/dist/agent/toolPrefetch.d.ts.map +1 -0
  21. package/dist/agent/toolPrefetch.js +101 -0
  22. package/dist/agent/trust.d.ts +0 -5
  23. package/dist/agent/trust.d.ts.map +1 -1
  24. package/dist/agent/trust.js +41 -0
  25. package/dist/api/client.d.ts +14 -0
  26. package/dist/api/client.d.ts.map +1 -1
  27. package/dist/api/client.js +88 -8
  28. package/dist/index.d.ts +1 -0
  29. package/dist/index.d.ts.map +1 -1
  30. package/dist/index.js +1 -0
  31. package/dist/permissions/modePolicy.d.ts +10 -0
  32. package/dist/permissions/modePolicy.d.ts.map +1 -1
  33. package/dist/permissions/modePolicy.js +11 -0
  34. package/dist/tools/executor.d.ts +36 -0
  35. package/dist/tools/executor.d.ts.map +1 -1
  36. package/dist/tools/executor.js +408 -85
  37. package/dist/tools/tsLangService.d.ts.map +1 -1
  38. package/dist/tools/tsLangService.js +107 -21
  39. package/dist/types.d.ts +99 -5
  40. package/dist/types.d.ts.map +1 -1
  41. package/dist/types.js +12 -1
  42. package/dist/util/miniYaml.d.ts +10 -0
  43. package/dist/util/miniYaml.d.ts.map +1 -0
  44. package/dist/util/miniYaml.js +149 -0
  45. package/package.json +8 -17
@@ -57,26 +57,37 @@ exports.capSubTaskText = capSubTaskText;
57
57
  exports.summariseSubTaskProgress = summariseSubTaskProgress;
58
58
  exports.lastToolResults = lastToolResults;
59
59
  exports.contextWindowFor = contextWindowFor;
60
+ exports.compactionLimits = compactionLimits;
60
61
  exports.compactionThresholds = compactionThresholds;
62
+ exports.pruneReclaimFloor = pruneReclaimFloor;
61
63
  exports.estimateBodyBytes = estimateBodyBytes;
62
64
  exports.allowsTestOnlyWrite = allowsTestOnlyWrite;
63
65
  exports.findSafeCutIndex = findSafeCutIndex;
66
+ exports.persistRuntimeContext = persistRuntimeContext;
64
67
  exports.transcriptOf = transcriptOf;
65
68
  exports.createLedger = createLedger;
66
69
  exports.ledgerRecord = ledgerRecord;
67
70
  exports.ledgerSummary = ledgerSummary;
68
71
  exports.pruneOldToolResults = pruneOldToolResults;
72
+ exports.makeCachedSummarizer = makeCachedSummarizer;
69
73
  exports.estimateTokensRough = estimateTokensRough;
70
74
  exports.compactMessagesForResume = compactMessagesForResume;
71
75
  exports.dispatchSubAgent = dispatchSubAgent;
76
+ exports.dispatchBackgroundSubAgent = dispatchBackgroundSubAgent;
72
77
  exports.runAgentLoop = runAgentLoop;
73
78
  exports.trimToResumableBoundary = trimToResumableBoundary;
79
+ const readDedupe_1 = require("./readDedupe");
80
+ const toolPrefetch_1 = require("./toolPrefetch");
81
+ const worktree_1 = require("./worktree");
82
+ const backgroundAgents_1 = require("./backgroundAgents");
74
83
  const types_1 = require("../types");
75
84
  const client_1 = require("../api/client");
76
85
  const executor_1 = require("../tools/executor");
77
86
  const agentTypes_1 = require("./agentTypes");
78
87
  const skills_1 = require("./skills");
79
88
  const rules_1 = require("../permissions/rules");
89
+ const modePolicy_1 = require("../permissions/modePolicy");
90
+ const destructive_1 = require("../permissions/destructive");
80
91
  // bashNeedsRepoLock lives in permissions/bashClassify.ts, not here: the permission
81
92
  // gate needs the same "does this command mutate shared state?" answer, and importing
82
93
  // it from loop.ts would make permissions depend on the agent loop (a cycle).
@@ -98,6 +109,16 @@ const crossProcessLock_1 = require("./crossProcessLock");
98
109
  const fs = __importStar(require("fs"));
99
110
  const path = __importStar(require("path"));
100
111
  const child_process_1 = require("child_process");
112
+ function withAgentHooks(base, agent) {
113
+ if (!agent)
114
+ return base;
115
+ return {
116
+ ...base,
117
+ PreToolUse: [...(base.PreToolUse ?? []), ...(agent.PreToolUse ?? [])],
118
+ PostToolUse: [...(base.PostToolUse ?? []), ...(agent.PostToolUse ?? [])],
119
+ SubagentStop: [...(base.SubagentStop ?? []), ...(agent.Stop ?? [])],
120
+ };
121
+ }
101
122
  function loadHooks(workDir) {
102
123
  let fromSettings = {};
103
124
  try {
@@ -128,7 +149,60 @@ function loadHooks(workDir) {
128
149
  // • stdout JSON object → { "decision": "block"|"allow", "reason": "...",
129
150
  // "additionalContext": "text to feed the model" }
130
151
  // • any other exit code → non-blocking (stderr logged, tool proceeds)
131
- function runToolHooks(entries, phase, toolName, input, workDir, result) {
152
+ /**
153
+ * Run one hook command WITHOUT blocking the event loop. spawnSync froze everything in
154
+ * the process for up to the hook's timeout — parallel sub-agents' streams, the UI, the
155
+ * stall watchdogs — which a 60 s PostToolUse test hook turned into a visible hang.
156
+ */
157
+ function spawnHook(command, opts) {
158
+ return new Promise((resolve) => {
159
+ let stdout = '';
160
+ let stderr = '';
161
+ let settled = false;
162
+ const done = (status) => {
163
+ if (settled)
164
+ return;
165
+ settled = true;
166
+ clearTimeout(timer);
167
+ resolve({ status, stdout, stderr });
168
+ };
169
+ let child;
170
+ try {
171
+ child = (0, child_process_1.spawn)(command, { shell: true, cwd: opts.cwd, env: opts.env ?? process.env, stdio: ['pipe', 'pipe', 'pipe'] });
172
+ }
173
+ catch {
174
+ resolve({ status: null, stdout: '', stderr: '' });
175
+ return;
176
+ }
177
+ const timer = setTimeout(() => { try {
178
+ child.kill('SIGTERM');
179
+ }
180
+ catch { /* gone */ } done(null); }, opts.timeout);
181
+ const cap = 16 * 1024 * 1024;
182
+ child.stdout?.on('data', (d) => { if (stdout.length < cap)
183
+ stdout += d.toString('utf-8'); });
184
+ child.stderr?.on('data', (d) => { if (stderr.length < cap)
185
+ stderr += d.toString('utf-8'); });
186
+ child.on('error', () => done(null));
187
+ child.on('close', (code) => done(code));
188
+ child.stdin?.on('error', () => { });
189
+ child.stdin?.end(opts.input ?? '');
190
+ });
191
+ }
192
+ /**
193
+ * Does a hook's matcher select this tool? Empty / "*" = every tool. "A|B" = either.
194
+ * Each alternative may be a Claude Code tool name ("Bash", "Edit") or a Nexrall one, and
195
+ * a Nexrall-name alternative keeps the old substring behaviour ("file" matches read_file).
196
+ */
197
+ function hookMatches(matcher, toolName) {
198
+ if (!matcher || matcher === '*')
199
+ return true;
200
+ return matcher.split('|').map((m) => m.trim()).filter(Boolean).some((alt) => {
201
+ const norm = (0, agentTypes_1.normaliseToolName)(alt);
202
+ return norm === toolName || (norm === alt && toolName.includes(alt));
203
+ });
204
+ }
205
+ async function runToolHooks(entries, phase, toolName, input, workDir, result) {
132
206
  const outcome = { block: false };
133
207
  if (!entries?.length)
134
208
  return outcome;
@@ -139,7 +213,7 @@ function runToolHooks(entries, phase, toolName, input, workDir, result) {
139
213
  ...(result ? { result: { output: result.output, error: result.error } } : {}),
140
214
  });
141
215
  for (const entry of entries) {
142
- if (entry.matcher && entry.matcher !== '*' && !toolName.includes(entry.matcher))
216
+ if (!hookMatches(entry.matcher, toolName))
143
217
  continue;
144
218
  for (const hook of entry.hooks ?? []) {
145
219
  if (hook.type !== 'command' || !hook.command)
@@ -150,26 +224,17 @@ function runToolHooks(entries, phase, toolName, input, workDir, result) {
150
224
  const hookTimeout = typeof hook.timeout_ms === 'number' && hook.timeout_ms > 0
151
225
  ? Math.min(hook.timeout_ms, 600000)
152
226
  : 60000;
153
- let r;
154
- try {
155
- r = (0, child_process_1.spawnSync)(hook.command, {
156
- shell: true,
157
- cwd: workDir,
158
- timeout: hookTimeout,
159
- encoding: 'utf-8',
160
- maxBuffer: 16 * 1024 * 1024,
161
- input: payload,
162
- env: {
163
- ...process.env,
164
- NEXRALL_TOOL_NAME: toolName,
165
- NEXRALL_TOOL_INPUT: JSON.stringify(input),
166
- NEXRALL_HOOK_PHASE: phase,
167
- },
168
- });
169
- }
170
- catch {
171
- continue; // hook itself failed to spawn — non-fatal
172
- }
227
+ const r = await spawnHook(hook.command, {
228
+ cwd: workDir,
229
+ timeout: hookTimeout,
230
+ input: payload,
231
+ env: {
232
+ ...process.env,
233
+ NEXRALL_TOOL_NAME: toolName,
234
+ NEXRALL_TOOL_INPUT: JSON.stringify(input),
235
+ NEXRALL_HOOK_PHASE: phase,
236
+ },
237
+ });
173
238
  // Optional JSON directive on stdout
174
239
  const out = (r.stdout ?? '').toString().trim();
175
240
  if (out.startsWith('{')) {
@@ -195,15 +260,12 @@ function runToolHooks(entries, phase, toolName, input, workDir, result) {
195
260
  }
196
261
  return outcome;
197
262
  }
198
- function runSimpleHooks(defs, workDir) {
263
+ async function runSimpleHooks(defs, workDir, extraEnv) {
199
264
  if (!defs?.length)
200
265
  return;
201
266
  for (const hook of defs) {
202
267
  if (hook.type === 'command' && hook.command) {
203
- try {
204
- (0, child_process_1.spawnSync)(hook.command, { shell: true, cwd: workDir, timeout: 10000 });
205
- }
206
- catch { /* non-fatal */ }
268
+ await spawnHook(hook.command, { cwd: workDir, timeout: 10000, env: extraEnv ? { ...process.env, ...extraEnv } : undefined });
207
269
  }
208
270
  }
209
271
  }
@@ -575,29 +637,45 @@ function lockPathsFor(name, input, workDir) {
575
637
  // callers.
576
638
  //
577
639
  // Anthropic hit the same wall and capped Claude Code's concurrent subagents at 20
578
- // (v2.1.217, July 2026); community guidance settles far lower, around 3-5, because
579
- // past that the synthesis overhead cancels the parallelism. We default to 4:
580
- // enough for genuine fan-out (the case sub-agents exist for), low enough that a
581
- // runaway `task` burst degrades into a queue instead of a thundering herd.
640
+ // (v2.1.217, July 2026, CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS); community guidance for
641
+ // everyday work settles around 3-5 because past that the synthesis overhead cancels
642
+ // the parallelism.
643
+ //
644
+ // Default 10, raised from 4 (2026-09-29), with a hard ceiling of 32 (see below). Measured
645
+ // on a 12-core dev box against the Nexrall repo, the per-agent latency of a
646
+ // search_files + glob pair was 116 ms at N=4, 176 ms at N=10, 300 ms at N=20 and 415 ms
647
+ // at N=30. Local CPU is NOT what limits fan-out; the model API is (rounds are seconds
648
+ // long, tools are milliseconds). What did limit it was the SERVER: /api/code was IP
649
+ // rate-limited at 100 req/15 min, which even 4 agents exhausted. That is now per-user
650
+ // (backend shared/utils/codeRateLimit.js), so the old "4" no longer protects anything
651
+ // that 10 does not.
652
+ //
653
+ // Why not default 20/30 like the headline number: the default applies to EVERY user, on
654
+ // every model, including ones with a low per-key TPM, and each concurrent agent is its
655
+ // own bill. Wide fan-out is opt-in (maxConcurrentSubtasks / NEXRALL_MAX_CONCURRENT_SUBTASKS
656
+ // up to 32), narrow fan-out is the safe default.
582
657
  //
583
658
  // This is a QUEUE, not a rejection: every sub-task still runs, just at most N at a
584
659
  // time. Failing the excess would be worse than serialising it.
585
- const DEFAULT_MAX_CONCURRENT_SUBTASKS = 4;
660
+ const DEFAULT_MAX_CONCURRENT_SUBTASKS = 10;
661
+ /** Hard ceiling on maxConcurrentSubtasks, whatever settings.json (repo-controlled) says. */
662
+ const HARD_MAX_CONCURRENT_SUBTASKS = 32;
586
663
  /**
587
664
  * Resolve the fan-out limit: env → settings.json → default.
588
665
  *
589
666
  * Was env-ONLY, the same gap as the sub-task timeout: a project on a small machine that
590
667
  * wanted 2, or a big one that wanted 6, had to set a shell variable, and nobody reading
591
- * settings.json could see what the limit even was. Capped at 16 because this bounds real
592
- * shared resources (CPU, the API rate limit, file handles) and a typo like 400 should
593
- * degrade to "a lot" rather than fork-bomb the machine.
668
+ * settings.json could see what the limit even was. Capped at HARD_MAX_CONCURRENT_SUBTASKS
669
+ * (32 — above Claude Code's default 20, so a user who wants that width can have it)
670
+ * because this bounds real shared resources (CPU, the API rate limit, file handles) and a
671
+ * typo like 400 should degrade to "a lot" rather than fork-bomb the machine.
594
672
  */
595
673
  function resolveMaxConcurrentSubtasks(settingsRaw = {}) {
596
674
  // `Math.max(1, …)` matters: a fractional value like 0.5 passes the `> 0` guard, then floors
597
675
  // to 0, and createLimiter(0) queues every task with nothing left to ever release them — a
598
676
  // silent permanent hang with no timeout and no error. Harmless when only depth 0 used the
599
677
  // limiter; now that every level does, it would wedge the whole tree.
600
- const clamp = (n) => Math.max(1, Math.min(Math.floor(n), 16));
678
+ const clamp = (n) => Math.max(1, Math.min(Math.floor(n), HARD_MAX_CONCURRENT_SUBTASKS));
601
679
  const fromEnv = Number(process.env.NEXRALL_MAX_CONCURRENT_SUBTASKS);
602
680
  if (Number.isFinite(fromEnv) && fromEnv > 0)
603
681
  return clamp(fromEnv);
@@ -703,11 +781,11 @@ function subTaskLimiter(depth, workDir) {
703
781
  if (!run) {
704
782
  // TAPERED per level, not `max` at every level.
705
783
  //
706
- // Giving each depth the full `max` multiplies total concurrency by the depth limit: 4
707
- // becomes 12 by default and 16×5 = 80 at the configured maxima. The value is justified by
784
+ // Giving each depth the full `max` multiplies total concurrency by the depth limit: 10
785
+ // becomes 30 by default and 32×5 = 160 at the configured maxima. The value is justified by
708
786
  // shared resources — CPU, file handles, ONE API rate limit — none of which care which
709
787
  // level a loop is running at, so honouring `4` per level silently abandons the limit the
710
- // user set. Halving per level bounds the total at ~2× `max` (4+2+1 = 7) while keeping the
788
+ // user set. Halving per level bounds the total at ~2× `max` (10+5+3 = 18 by default) while keeping the
711
789
  // per-depth structure that makes the wait-for graph acyclic.
712
790
  //
713
791
  // Never below 1: a level with 0 slots is a permanent hang, not a restriction.
@@ -724,22 +802,19 @@ function _resetSubTaskLimiter() {
724
802
  _subTaskLimitMaxByDepth.clear();
725
803
  _inFlightByDepth.clear();
726
804
  _subTaskLimitMax = 0;
727
- _subAgentsThisSession = 0;
728
- _sessionCapMax = 0;
729
- _sessionCapNotified = false;
805
+ _subAgentBudgets.clear();
806
+ }
807
+ const _subAgentBudgets = new Map();
808
+ const PROCESS_BUDGET_KEY = '\0process';
809
+ function budgetFor(sessionKey) {
810
+ const key = sessionKey || PROCESS_BUDGET_KEY;
811
+ let b = _subAgentBudgets.get(key);
812
+ if (!b) {
813
+ b = { used: 0, max: 0, notified: false };
814
+ _subAgentBudgets.set(key, b);
815
+ }
816
+ return b;
730
817
  }
731
- // Counts sub-agents STARTED, never decremented — that is what makes it a budget rather
732
- // than a concurrency gate.
733
- //
734
- // NOT process-wide, unlike the limiter, and the distinction is load-bearing. A concurrency
735
- // gate is safe to share across a process because it self-drains; a monotonic counter is
736
- // not. In the CLI one process is one session, but VS Code calls runAgentLoop from a
737
- // long-lived extension host, so process-scoped state would accumulate across every
738
- // conversation in the window until delegation died permanently — and the remedy string
739
- // ("start a new session") would be a lie, since only reloading the window would help.
740
- let _subAgentsThisSession = 0;
741
- let _sessionCapMax = 0;
742
- let _sessionCapNotified = false;
743
818
  /**
744
819
  * Start a fresh sub-agent budget. Call when a NEW conversation begins.
745
820
  *
@@ -747,10 +822,9 @@ let _sessionCapNotified = false;
747
822
  * host). A client that never calls it gets process-lifetime semantics, which is correct
748
823
  * for a one-shot CLI invocation.
749
824
  */
750
- function resetSessionSubAgentBudget() {
751
- _subAgentsThisSession = 0;
752
- _sessionCapMax = 0;
753
- _sessionCapNotified = false;
825
+ function resetSessionSubAgentBudget(sessionKey) {
826
+ // No key: the legacy "new conversation" call — reset the process-wide budget only.
827
+ _subAgentBudgets.delete(sessionKey || PROCESS_BUDGET_KEY);
754
828
  }
755
829
  /**
756
830
  * Claim one slot against the session total. Returns an error string when exhausted.
@@ -760,13 +834,15 @@ function resetSessionSubAgentBudget() {
760
834
  * the user never learns delegation was capped, which is the one thing a runaway guard has
761
835
  * to make visible.
762
836
  */
763
- function claimSessionSubAgentSlot(workDir, notify) {
764
- if (!_sessionCapMax) {
765
- _sessionCapMax = resolveMaxSubagentsPerSession(workDir ? (0, rules_1.loadSettings)(workDir).raw : {});
837
+ function claimSessionSubAgentSlot(workDir, notify, sessionKey) {
838
+ const b = budgetFor(sessionKey);
839
+ if (!b.max) {
840
+ b.max = resolveMaxSubagentsPerSession(workDir ? (0, rules_1.loadSettings)(workDir).raw : {});
766
841
  }
767
- if (_subAgentsThisSession >= _sessionCapMax) {
768
- if (!_sessionCapNotified) {
769
- _sessionCapNotified = true;
842
+ const _sessionCapMax = b.max;
843
+ if (b.used >= b.max) {
844
+ if (!b.notified) {
845
+ b.notified = true;
770
846
  notify?.(`\u26a0\ufe0f Sub-agent budget reached (${_sessionCapMax} this session) \u2014 further delegation is ` +
771
847
  'blocked and the agent will continue without it. Raise "maxSubagentsPerSession" in ' +
772
848
  '.nexrall/settings.json if this was legitimate work.');
@@ -775,7 +851,7 @@ function claimSessionSubAgentSlot(workDir, notify) {
775
851
  'runaway-delegation guard, not a per-task limit: do the remaining work directly, and say ' +
776
852
  'in your final message that you hit the delegation cap.');
777
853
  }
778
- _subAgentsThisSession++;
854
+ b.used++;
779
855
  return null;
780
856
  }
781
857
  /**
@@ -794,13 +870,14 @@ function claimSessionSubAgentSlot(workDir, notify) {
794
870
  *
795
871
  * Floored at 0 so a double refund can never manufacture budget.
796
872
  */
797
- function refundSessionSubAgentSlot() {
798
- if (_subAgentsThisSession > 0)
799
- _subAgentsThisSession--;
873
+ function refundSessionSubAgentSlot(sessionKey) {
874
+ const b = budgetFor(sessionKey);
875
+ if (b.used > 0)
876
+ b.used--;
800
877
  }
801
878
  /** Test-only: observe the session counter without exporting the mutable binding. */
802
- function _sessionSubAgentCount() {
803
- return _subAgentsThisSession;
879
+ function _sessionSubAgentCount(sessionKey) {
880
+ return budgetFor(sessionKey).used;
804
881
  }
805
882
  // ─── Peer message rendering ───────────────────────────────────────────────────
806
883
  //
@@ -830,7 +907,7 @@ function humanDescription(name, input) {
830
907
  case 'bash':
831
908
  return `Run: ${input.command ?? '(unknown)'}`;
832
909
  case 'search_files': {
833
- const searchType = input.type === 'filename' ? 'filename' : 'content';
910
+ const searchType = input.type === 'filename' ? 'filename' : input.type === 'content' || input.context_lines ? 'content' : 'files';
834
911
  const inPath = input.path ? ` in ${input.path}` : '';
835
912
  return `Search ${searchType}: "${input.pattern ?? ''}"${inPath}`;
836
913
  }
@@ -1235,6 +1312,72 @@ function toolResultText(block) {
1235
1312
  }
1236
1313
  return '';
1237
1314
  }
1315
+ // ── Hard stop ─────────────────────────────────────────────────────────────────
1316
+ // The stall watchdog and the parent's Stop only SET a flag; runAgentLoop notices it at
1317
+ // the next boundary. A tool that never returns (an MCP server that ignores the abort,
1318
+ // a stuck editor-side call) meant that boundary never came and the parent hung forever.
1319
+ // After the flag is set, the child gets this long to wind down before we stop waiting.
1320
+ const HARD_STOPPED = Symbol('hard-stopped');
1321
+ function hardStopGraceMs() {
1322
+ const v = Number(process.env.NEXRALL_SUBAGENT_HARD_STOP_MS);
1323
+ return Number.isFinite(v) && v > 0 ? v : 30000;
1324
+ }
1325
+ function raceHardStop(run, abort) {
1326
+ return new Promise((resolve, reject) => {
1327
+ let grace = null;
1328
+ const poll = setInterval(() => {
1329
+ if (abort.aborted && !grace) {
1330
+ grace = setTimeout(() => { clearInterval(poll); resolve(HARD_STOPPED); }, hardStopGraceMs());
1331
+ }
1332
+ }, 250);
1333
+ const settle = () => { clearInterval(poll); if (grace)
1334
+ clearTimeout(grace); };
1335
+ run.then((v) => { settle(); resolve(v); }, (e) => { settle(); reject(e); });
1336
+ });
1337
+ }
1338
+ /** Returned instead of waiting forever for a tool that ignored Stop. */
1339
+ const STOP_TIMEOUT_RESULT = {
1340
+ error: 'Interrupted: this tool did not respond to Stop, so the agent stopped waiting for it.',
1341
+ interrupted: true,
1342
+ };
1343
+ const STOP_GRACE_MS = 5000;
1344
+ function stopAwaiting(run, abort) {
1345
+ if (!abort)
1346
+ return run;
1347
+ return new Promise((resolve, reject) => {
1348
+ let grace = null;
1349
+ const poll = setInterval(() => {
1350
+ if (abort.aborted && !grace)
1351
+ grace = setTimeout(() => { clearInterval(poll); resolve(STOP_TIMEOUT_RESULT); }, STOP_GRACE_MS);
1352
+ }, 200);
1353
+ const settle = () => { clearInterval(poll); if (grace)
1354
+ clearTimeout(grace); };
1355
+ run.then((v) => { settle(); resolve(v); }, (e) => { settle(); reject(e); });
1356
+ });
1357
+ }
1358
+ /** Default turn limit for a sub-agent whose definition sets no `maxTurns`. */
1359
+ const DEFAULT_SUBAGENT_MAX_TURNS = 200;
1360
+ /** Stop reasons that mean the sub-agent did NOT deliver a finished report. */
1361
+ const ABNORMAL_SUBAGENT_STOPS = new Set([
1362
+ 'budget', 'stalled', 'stalled-repeat', 'output-limit', 'empty-response',
1363
+ 'no-balance', 'no-team-budget', 'team-unavailable', 'reported-elsewhere',
1364
+ ]);
1365
+ /**
1366
+ * Prepended to EVERY sub-agent's instructions (named or not) — Claude Code's
1367
+ * "you are a sub-agent; your final message is the report" contract.
1368
+ */
1369
+ const SUBAGENT_PREAMBLE = [
1370
+ '# You are a sub-agent',
1371
+ 'The main agent started you for ONE delegated task. You cannot see its conversation with the user — only the task you were given — and you cannot ask the user anything: if something is ambiguous, make the most reasonable assumption and state it.',
1372
+ 'Your FINAL message is the only thing returned to the main agent, and the user does not see it directly. Make it a concise, self-contained report: what you found or changed (with file paths and line numbers), what you verified and how, and anything left undone or uncertain. Do not pad it, and do not paste whole files — cite locations instead.',
1373
+ ].join('\n');
1374
+ function formatTokenCount(n) {
1375
+ return n >= 1000000 ? `${(n / 1000000).toFixed(1)}M` : n >= 1000 ? `${(n / 1000).toFixed(1)}K` : String(n);
1376
+ }
1377
+ function formatElapsed(ms) {
1378
+ const s = Math.round(ms / 1000);
1379
+ return s < 60 ? `${s}s` : `${Math.floor(s / 60)}m ${s % 60}s`;
1380
+ }
1238
1381
  async function runSubTask(input, options, agentTypes,
1239
1382
  // Set to true at the moment an agent loop actually STARTS, so the caller can refund the
1240
1383
  // session slot it claimed for a spawn that turned out never to run.
@@ -1294,7 +1437,7 @@ started) {
1294
1437
  //
1295
1438
  // Reported as "expired" rather than "not yours": a distinct message would confirm the id
1296
1439
  // exists, turning the error into an oracle for enumerating other frames' agents.
1297
- if (resumed && !(0, agentRegistry_1.canResume)(resumed, agentScope)) {
1440
+ if (resumed && !(0, agentRegistry_1.canResume)(resumed, agentScope, options.sessionId)) {
1298
1441
  return {
1299
1442
  error: `No resumable sub-agent with id "${resumeId}" is available to this run. Start a fresh ` +
1300
1443
  'sub-task with a self-contained prompt instead.',
@@ -1341,7 +1484,9 @@ started) {
1341
1484
  'do not create it and do not retry. Do the work yourself, or use a different sub-agent.',
1342
1485
  };
1343
1486
  }
1344
- let agent = (0, agentTypes_1.findAgentType)(agentTypes, requestedType);
1487
+ // An unnamed dispatch IS general-purpose (the tool schema says so): same role prompt,
1488
+ // same "your final message is your report" instructions, same deny rule (denyKey above).
1489
+ let agent = (0, agentTypes_1.findAgentType)(agentTypes, requestedType || 'general-purpose');
1345
1490
  let knownTypes = agentTypes;
1346
1491
  if (requestedType && !agent) {
1347
1492
  // Same `extra` list the top-of-run snapshot used (see runAgentLoop) — otherwise a
@@ -1380,12 +1525,39 @@ started) {
1380
1525
  // `lightPrompt` agents skip the project's nexrall.md — see AgentType.lightPrompt for
1381
1526
  // why. The role prompt and any private notes still apply; only the (potentially very
1382
1527
  // large) project instruction file is dropped.
1383
- const inheritedMd = agent?.lightPrompt ? '' : (options.nexrallMd ?? '');
1384
- const subNexrallMd = agent
1385
- ? `# Sub-agent role: ${agent.name}\n${agent.prompt}` +
1386
- (agentMemoryNotes ? `\n\n---\n\n${agentMemoryNotes}` : '') +
1387
- (inheritedMd ? `\n\n---\n\n${inheritedMd}` : '')
1388
- : options.nexrallMd;
1528
+ //
1529
+ // Composed from the PROJECT's instructions (`_projectNexrallMd`), never from the
1530
+ // parent's own composite prompt: a grandchild used to inherit its parent's role
1531
+ // ("# Sub-agent role: general-purpose … make the changes") on top of its own
1532
+ // ("reviewer … you NEVER modify files"), plus the parent's private memory notes.
1533
+ const projectMd = options._projectNexrallMd ?? options.nexrallMd ?? '';
1534
+ const inheritedMd = agent?.lightPrompt ? '' : projectMd;
1535
+ // `skills:` — preload those skills' instructions (raw body: no !`cmd` expansion runs
1536
+ // just because an agent was spawned). A missing name is stated, not silently dropped.
1537
+ let preloadedSkills = '';
1538
+ if (agent?.skills?.length) {
1539
+ const known = (0, skills_1.loadSkillsWithWarnings)(options.workDir, options._extraSkills ?? []).skills;
1540
+ preloadedSkills = agent.skills.map((n) => {
1541
+ const sk = (0, skills_1.findSkill)(known, n);
1542
+ return sk ? `# Preloaded skill: ${sk.name}\n${sk.body.trim()}` : `# Preloaded skill: ${n}\n(not found in this project — ignore)`;
1543
+ }).join('\n\n');
1544
+ }
1545
+ // Shared-before-specific order: the project's nexrall.md (usually the largest part, and
1546
+ // identical for every non-lightPrompt sibling) goes first; per-agent parts (preamble,
1547
+ // role, skills, private notes) follow — last, where they also carry the most weight
1548
+ // with the model. With the role first, two agents diverged at the first line of this
1549
+ // block. HONEST SCOPE: this only buys cache reuse where everything BEFORE the block
1550
+ // already matches — same tool list and same prompt flags (so e.g. two custom agents
1551
+ // with equal `tools:`, on prefix-cached providers). Types with different allowlists
1552
+ // diverge earlier, at the tools array, whatever the order here. It costs nothing, and
1553
+ // tool enforcement never depended on prompt order (see the allowlist below).
1554
+ const subNexrallMd = [
1555
+ inheritedMd,
1556
+ SUBAGENT_PREAMBLE,
1557
+ agent ? `# Sub-agent role: ${agent.name}\n${agent.prompt}` : '',
1558
+ preloadedSkills,
1559
+ agentMemoryNotes,
1560
+ ].filter(Boolean).join('\n\n---\n\n');
1389
1561
  // Optional tool allowlist — deny anything outside it for this sub-agent.
1390
1562
  //
1391
1563
  // A refusal here is reported through `deniedReason` rather than the generic
@@ -1414,6 +1586,20 @@ started) {
1414
1586
  // it, so the allowlist stays the single source of truth for what this agent can do.
1415
1587
  if (allowed && memoryScope)
1416
1588
  allowed.add(exports.AGENT_MEMORY_TOOL);
1589
+ // `disallowedTools` (Claude Code) — inherited like the allowlist: a child can only add
1590
+ // to its ancestors' refusals, never shed them.
1591
+ const disallowed = new Set([...(options._disallowedTools ?? []), ...(agent?.disallowedTools ?? [])]);
1592
+ if (allowed)
1593
+ for (const t of disallowed)
1594
+ allowed.delete(t);
1595
+ // `mcpServers` — narrows, inherited like the tool allowlist.
1596
+ let mcpAllow = options._mcpServerAllowlist;
1597
+ if (agent?.mcpServers) {
1598
+ const own = new Set(agent.mcpServers);
1599
+ mcpAllow = mcpAllow ? new Set([...mcpAllow].filter((x) => own.has(x))) : own;
1600
+ }
1601
+ // Time spent waiting on a human (permission prompt) is not a stall.
1602
+ let waitingOnUser = 0;
1417
1603
  const gatedPermission = async (req) => {
1418
1604
  // Guard against the tool being reachable without a declared scope — e.g. an agent
1419
1605
  // that lists it in `tools:` by hand, or a general-purpose sub-task with no
@@ -1461,12 +1647,42 @@ started) {
1461
1647
  // allowlist above, a restriction can only ever be narrowed by nesting, never shed.
1462
1648
  // Without the inherited half, `test-writer` could delegate to an unrestricted agent and
1463
1649
  // have production source written on its behalf.
1650
+ if (mcpAllow && req.tool.includes('__') && !mcpAllow.has(req.tool.split('__')[0])) {
1651
+ throw new ToolNotAllowedError(`The "${agent?.name ?? 'general-purpose'}" sub-agent may only use tools from these MCP servers: ` +
1652
+ `${[...mcpAllow].join(', ') || '(none)'}. \`${req.tool}\` is from another server — use a different tool or report back.`);
1653
+ }
1654
+ if (disallowed.has(req.tool)) {
1655
+ throw new ToolNotAllowedError(`The "${agent?.name ?? 'general-purpose'}" sub-agent may not use \`${req.tool}\` (disallowedTools in its ` +
1656
+ 'definition). This is a restriction of the agent definition, NOT a user decision — use another tool or report back.');
1657
+ }
1658
+ if (req.tool === 'memory_write') {
1659
+ throw new ToolNotAllowedError('Sub-agents cannot write the shared memory store. Put anything worth remembering in your final report — ' +
1660
+ 'the main agent decides what to persist.');
1661
+ }
1464
1662
  if (testFilesOnly && !allowsTestOnlyWrite(req.tool, req.input)) {
1465
1663
  throw new ToolNotAllowedError(`The "${agent?.name ?? 'general-purpose'}" sub-agent may only write to TEST files, so ` +
1466
1664
  `\`${req.tool}\` was refused for this path. Do not try to work around it: if production ` +
1467
1665
  'code must change, say so in your report instead.');
1468
1666
  }
1469
- return options.requestPermission(req);
1667
+ // An agent writing its OWN notes (declared `memory:`) is answered here, not by the
1668
+ // ancestors: their gates refuse agent_memory_write unless THEY declared a scope, so a
1669
+ // memory agent spawned by general-purpose could never write. The tool can only touch
1670
+ // this agent's own notes file, so there is nothing for anyone above to approve.
1671
+ if (req.tool === exports.AGENT_MEMORY_TOOL && agent && memoryScope)
1672
+ return true;
1673
+ waitingOnUser++;
1674
+ try {
1675
+ // Say WHICH sub-agent is asking (the innermost one wins as the request bubbles up).
1676
+ const description = typeof input.description === 'string' ? input.description : undefined;
1677
+ return await options.requestPermission({
1678
+ ...req,
1679
+ agent: req.agent ?? { name: agent?.name ?? 'general-purpose', ...(description ? { description } : {}) },
1680
+ });
1681
+ }
1682
+ finally {
1683
+ waitingOnUser--;
1684
+ bumpProgress();
1685
+ }
1470
1686
  };
1471
1687
  // ── Resume: continue a previous sub-agent instead of starting cold ──────────
1472
1688
  //
@@ -1507,6 +1723,10 @@ started) {
1507
1723
  let stalled = false;
1508
1724
  const bumpProgress = () => { lastProgressAt = Date.now(); };
1509
1725
  const stallWatchdog = setInterval(() => {
1726
+ if (waitingOnUser > 0) {
1727
+ lastProgressAt = Date.now();
1728
+ return;
1729
+ }
1510
1730
  if (Date.now() - lastProgressAt > subtaskTimeoutMs) {
1511
1731
  stalled = true;
1512
1732
  subAbort.aborted = true;
@@ -1520,11 +1740,83 @@ started) {
1520
1740
  if (options.abortSignal?.aborted)
1521
1741
  subAbort.aborted = true;
1522
1742
  }, 250);
1743
+ // ── Worktree isolation (`isolation: "worktree"` on the call or in the definition) ──
1744
+ // A fresh worktree per spawn, like Claude Code: the child's writes, bash cwd and git
1745
+ // commands are confined to it (worktreeEnforcement), and it is removed afterwards
1746
+ // unless the child actually changed something.
1747
+ let isoState;
1748
+ if (input.isolation === 'worktree' || agent?.isolation === 'worktree') {
1749
+ const created = (0, worktree_1.createWorktree)(options.workDir);
1750
+ if (!created.ok || !created.state) {
1751
+ clearInterval(stallWatchdog);
1752
+ clearInterval(parentAbortPoll);
1753
+ return {
1754
+ error: `Could not create an isolated worktree for this sub-agent: ${created.error ?? 'unknown error'}. ` +
1755
+ 'Run it without isolation, or fix the repository state first.',
1756
+ };
1757
+ }
1758
+ isoState = created.state;
1759
+ }
1760
+ const childWorkDir = isoState?.worktreePath ?? options.workDir;
1761
+ // Claude Code's Explore/Plan skip git status; so do lightPrompt agents here. An
1762
+ // isolated child is told where it actually is.
1763
+ let childEnv = options.env;
1764
+ if (childEnv && agent?.lightPrompt) {
1765
+ const { gitStatus: _s, gitDiff: _d, recentCommits: _c, ...rest } = childEnv;
1766
+ childEnv = rest;
1767
+ }
1768
+ if (childEnv && isoState)
1769
+ childEnv = { ...childEnv, cwd: isoState.worktreePath, gitBranch: isoState.branch ?? childEnv.gitBranch };
1770
+ // Per-sub-agent accounting, reported in its tool result (Claude Code shows the same).
1771
+ const subStartedAt = Date.now();
1772
+ let subToolCalls = 0;
1773
+ let subTokens = 0;
1774
+ let subCost = 0;
1775
+ let childStop = null;
1776
+ let childNotice = null;
1777
+ let childVerifications = [];
1778
+ const finalize = (r) => {
1779
+ let isoNote = '';
1780
+ if (isoState) {
1781
+ if ((0, worktree_1.worktreeHasWork)(isoState)) {
1782
+ isoNote = `\n[isolated worktree: this sub-agent's changes are in ${isoState.worktreePath}` +
1783
+ `${isoState.branch ? ` (branch ${isoState.branch})` : ''} — NOT in the main checkout. Review them, then merge or discard.]`;
1784
+ }
1785
+ else {
1786
+ (0, worktree_1.removeWorktree)(isoState, { force: true });
1787
+ }
1788
+ isoState = undefined;
1789
+ }
1790
+ const stats = `[sub-agent stats: ${subToolCalls} tool call(s) · ${formatTokenCount(subTokens)} tokens · ` +
1791
+ `$${subCost.toFixed(3)} · ${formatElapsed(Date.now() - subStartedAt)}]`;
1792
+ const out = { ...r, ...(childVerifications.length ? { childVerifications } : {}) };
1793
+ if (out.error !== undefined)
1794
+ out.error = `${out.error}\n\n${stats}${isoNote}`;
1795
+ else
1796
+ out.output = `${out.output ?? ''}\n\n${stats}${isoNote}`;
1797
+ return out;
1798
+ };
1799
+ // Claimed right before `try` (whose finally releases it). runSubTask has no `await`
1800
+ // before this point, so two parallel calls cannot both pass the check.
1801
+ if (resumed && !(0, agentRegistry_1.claimAgentForResume)(resumed.id)) {
1802
+ clearInterval(stallWatchdog);
1803
+ clearInterval(parentAbortPoll);
1804
+ if (isoState)
1805
+ (0, worktree_1.removeWorktree)(isoState, { force: true });
1806
+ return {
1807
+ error: `Sub-agent "${resumed.id}" is already being resumed by another task call that is still running. ` +
1808
+ 'Wait for that result, then resume it again with your follow-up — two parallel resumes of one agent ' +
1809
+ 'would overwrite each other\'s work.',
1810
+ };
1811
+ }
1812
+ const childMeta = (m) => (m
1813
+ ? { ...m, parentId: m.parentId ?? options._taskToolUseId, agentName: m.agentName ?? agent?.name ?? 'general-purpose' }
1814
+ : undefined);
1523
1815
  try {
1524
1816
  // The point of no return: past here a real agent loop exists and the budget slot is spent.
1525
1817
  if (started)
1526
1818
  started.value = true;
1527
- const result = await runAgentLoop(subMessages, {
1819
+ const childRun = runAgentLoop(subMessages, {
1528
1820
  ...options,
1529
1821
  _depth: depth + 1,
1530
1822
  _agentScope: `sub_${++_subTaskCounter}`, // isolated todo store per sub-agent
@@ -1562,16 +1854,48 @@ started) {
1562
1854
  // it, and here that direction is at least safe, whereas forgetting to propagate is not.
1563
1855
  _testFilesOnly: testFilesOnly,
1564
1856
  editorContext: null, // fresh isolated context for sub-agent
1565
- model: agent?.model ?? options.model,
1857
+ model: (0, modelCatalogue_1.resolveSubAgentModel)(agent?.model, options.model),
1858
+ workDir: childWorkDir,
1859
+ env: childEnv,
1860
+ effort: agent?.effort ?? options.effort,
1861
+ // A bounded run that REPORTS when it hits the limit (see childStop below), instead
1862
+ // of inheriting the main agent's 500-step budget plus auto-continue to 2000.
1863
+ maxIterations: agent?.maxTurns ?? DEFAULT_SUBAGENT_MAX_TURNS,
1864
+ autoContinue: false,
1865
+ // ── Session-level channels a child must NEVER consume ────────────────────
1866
+ // Inherited through `...options`, a child drained the user's queued follow-ups
1867
+ // and inbound peer messages into ITS history (the main agent never saw them),
1868
+ // and its onProgress saved the child's transcript AS the session (CLI/desktop).
1869
+ takePendingInput: undefined,
1870
+ onInjectedInput: undefined,
1871
+ drainPeerMessages: undefined,
1872
+ onPeerMessage: undefined,
1873
+ onProgress: undefined,
1874
+ backgroundAgents: undefined,
1875
+ _projectNexrallMd: projectMd,
1876
+ _disallowedTools: disallowed.size ? disallowed : undefined,
1877
+ _mcpServerAllowlist: mcpAllow,
1878
+ _agentHooks: agent?.hooks,
1879
+ _onStopReason: (reason, notice) => { childStop = reason; childNotice = notice; },
1880
+ _onVerifications: (records) => { childVerifications = records; },
1881
+ onUsage: (u, partial, cost, _sub) => {
1882
+ if (!partial) {
1883
+ subTokens += (u.input_tokens ?? 0) + (u.output_tokens ?? 0)
1884
+ + (u.cache_read_input_tokens ?? 0) + (u.cache_creation_input_tokens ?? 0);
1885
+ subCost += cost ?? 0;
1886
+ }
1887
+ options.onUsage(u, partial, cost, true);
1888
+ },
1566
1889
  // Plan mode is inherited, never relaxed. If the main agent could spawn a
1567
1890
  // sub-agent that writes, the lock would be one `task` call from useless.
1568
- planMode: options.planMode,
1891
+ // `permissionMode: plan` in a definition can only ADD the lock.
1892
+ planMode: options.planMode || agent?.permissionMode === 'plan',
1569
1893
  // Same reasoning as planMode directly above: a sub-agent that could reach
1570
1894
  // outside its parent's worktree would defeat the isolation in one `task`
1571
1895
  // call. Already inherited via `...options` above — restated explicitly so
1572
1896
  // it reads the same way as planMode and is never accidentally dropped by
1573
1897
  // a future refactor of this spread.
1574
- worktree: options.worktree,
1898
+ worktree: isoState ?? options.worktree,
1575
1899
  // A sub-agent using message_peer_session/list_peer_sessions should
1576
1900
  // present as the SAME peer identity as its parent — there is one
1577
1901
  // registered peer per SESSION, not per sub-agent, so a sub-agent is
@@ -1581,22 +1905,10 @@ started) {
1581
1905
  abortSignal: subAbort,
1582
1906
  requestPermission: gatedPermission,
1583
1907
  onText: () => { }, // sub-agent text is returned as the tool result, not streamed live
1584
- // Forwarded DELIBERATELY, and it must be a real handler rather than a no-op.
1585
- //
1586
- // A sub-agent streams no text (onText above is a no-op) but it DOES stream thinking
1587
- // through the parent's UI (see onThinking/onThinkingDelta below), and thinking sets
1588
- // `emittedToCaller`. So a sub-agent stream that dies after reasoning genuinely has
1589
- // rendered output to discard — a no-op here would let the restart proceed and then
1590
- // re-stream that reasoning on top of the copy still on screen, the exact duplication
1591
- // the opt-in exists to prevent.
1592
- //
1593
- // Forwarding is safe because the parent's own output is already closed by this
1594
- // point: dispatching the `task` tool goes through options.onToolUse, which finalizes
1595
- // the parent's bubble (VS Code) / flushes the renderer (CLI) before the sub-agent
1596
- // starts. The only live, discardable element at restart time is the sub-agent's own
1597
- // thinking block. Completed tool rows are left alone — those are real side effects
1598
- // that actually happened.
1599
- onStreamRestart: (reason, chars) => options.onStreamRestart?.(reason, chars),
1908
+ // A sub-agent renders nothing live (text and thinking are not streamed — see below),
1909
+ // so a restart of ITS stream has nothing on screen to roll back: a real no-op
1910
+ // handler is exactly right, and it opts the child into post-render restarts.
1911
+ onStreamRestart: () => { },
1600
1912
  // Forward tool events with isSubTask=true so the UI can render a badge
1601
1913
  // instead of prepending "[sub-task]" to the tool name (which caused double-prefix
1602
1914
  // when the name was already labelled, and mixed display concerns into the data layer).
@@ -1604,17 +1916,32 @@ started) {
1604
1916
  // is what the stall watchdog above measures. Bumping on both use and result means a
1605
1917
  // single very slow tool (a long test run) resets the clock when it starts AND when
1606
1918
  // it finishes, so it cannot be mistaken for a hang.
1607
- onToolUse: (n, i) => { bumpProgress(); options.onToolUse(n, i, true); },
1608
- onToolResult: (n, r) => { bumpProgress(); options.onToolResult(n, r, true); },
1609
- onToolStreamChunk: (n, c) => options.onToolStreamChunk?.(n, c, true),
1610
- // Forward thinking so the UI shows the indicator while sub-agent reasons
1611
- // Thinking is progress too — a model reasoning for minutes on a hard problem is
1612
- // working, not stalled. Without this, deep reasoning on an expensive tier would
1613
- // trip the watchdog precisely when the sub-agent was most valuable.
1614
- onThinking: (text) => { bumpProgress(); options.onThinking?.(text); },
1615
- onThinkingDelta: (text) => { bumpProgress(); options.onThinkingDelta?.(text); },
1616
- onThinkingProgress: (tok) => { bumpProgress(); options.onThinkingProgress?.(tok); },
1919
+ // `parentId` = the `task` call that spawned THIS child. Set only if not already set,
1920
+ // so a grandchild's events keep pointing at their own (nearest) task row.
1921
+ onToolUse: (n, i, _s, m) => { bumpProgress(); subToolCalls++; options.onToolUse(n, i, true, childMeta(m)); },
1922
+ onToolResult: (n, r, _s, m) => { bumpProgress(); options.onToolResult(n, r, true, childMeta(m)); },
1923
+ // A long command streaming output is working, not hung.
1924
+ onToolStreamChunk: (n, c, _s, m) => { bumpProgress(); options.onToolStreamChunk?.(n, c, true, childMeta(m)); },
1925
+ // Thinking is progress (a model reasoning for minutes is working, not stalled), but
1926
+ // it is NOT forwarded to the UI: parallel siblings' deltas interleaved into one live
1927
+ // thinking block, and Claude Code does not show sub-agent reasoning either. The
1928
+ // sub-agent's tool rows (grouped under its task row) are its visible progress.
1929
+ onThinking: () => { bumpProgress(); },
1930
+ onThinkingDelta: () => { bumpProgress(); },
1931
+ onThinkingProgress: () => { bumpProgress(); },
1617
1932
  });
1933
+ const raced = await raceHardStop(childRun, subAbort);
1934
+ if (raced === HARD_STOPPED) {
1935
+ // The child was told to stop (stall watchdog, parent Stop, background stop) but a
1936
+ // tool it is running ignored cancellation — an MCP call, an editor-side tool. The
1937
+ // parent must not hang on it forever; the orphaned call is left to finish alone.
1938
+ return finalize({
1939
+ error: `Sub-task did not stop within ${Math.round(hardStopGraceMs() / 1000)}s of being stopped: a tool it was ` +
1940
+ 'running ignored cancellation. Its partial work could not be collected. Do not re-run it as-is — ' +
1941
+ 'narrow the task, or avoid the tool that hung.',
1942
+ });
1943
+ }
1944
+ const result = raced;
1618
1945
  // ── Stall timeout: SALVAGE, don't discard ────────────────────────────────
1619
1946
  //
1620
1947
  // The sub-agent hit its wall-clock cap (rather than finishing, or the parent
@@ -1642,7 +1969,7 @@ started) {
1642
1969
  // do: run the whole task again. Resuming is now POSSIBLE but never implied to be
1643
1970
  // safe: the text below states plainly that the work is unverified, and resumption
1644
1971
  // re-authorises against current permissions exactly as it does for a clean run.
1645
- const partialId = (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope);
1972
+ const partialId = (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope, options.sessionId);
1646
1973
  const sections = [
1647
1974
  `Sub-task STOPPED after ${mins} minutes with NO PROGRESS (it was not making tool calls or ` +
1648
1975
  'producing output) — treat everything below as PARTIAL, unverified work, not a finished answer.',
@@ -1656,27 +1983,49 @@ started) {
1656
1983
  // errored call as a non-effect, which is right — nothing here is verified —
1657
1984
  // and the STALL_LIMIT runaway guard must still see repeated timeouts as
1658
1985
  // failures so a permanently stuck sub-task can't loop forever.
1659
- return { error: sections.join('\n\n') };
1986
+ return finalize({ error: sections.join('\n\n') });
1987
+ }
1988
+ // ── Stopped abnormally (turn limit, repeated failures, truncation, no balance) ──
1989
+ // runAgentLoop RETURNS normally for these, so this used to fall through to the
1990
+ // success path: the parent was handed a fragment as if it were the finished report,
1991
+ // while the explanation went to the user's chat as if the main agent had stopped.
1992
+ // Assigned inside callbacks, so TS narrows them to `null` here without the casts.
1993
+ const stopReasonOfChild = childStop;
1994
+ const noticeOfChild = childNotice;
1995
+ if (stopReasonOfChild && ABNORMAL_SUBAGENT_STOPS.has(stopReasonOfChild) && !subAbort.aborted) {
1996
+ const partial = capSubTaskText(extractSubTaskText(result, false));
1997
+ const progress = summariseSubTaskProgress(result);
1998
+ const partialId = resumed
1999
+ ? ((0, agentRegistry_1.updateAgent)(resumed.id, result, agentScope, options.sessionId), resumed.id)
2000
+ : (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope, options.sessionId);
2001
+ const why = stopReasonOfChild === 'budget'
2002
+ ? `it reached its turn limit (${agent?.maxTurns ?? DEFAULT_SUBAGENT_MAX_TURNS} model round-trips)`
2003
+ : `it stopped early (${stopReasonOfChild})`;
2004
+ const sections = [
2005
+ `Sub-task did NOT finish: ${why}. Treat everything below as PARTIAL, unverified work.`,
2006
+ noticeOfChild ? noticeOfChild.trim() : '',
2007
+ progress,
2008
+ partial ? `Partial output:\n\n${partial}` : '',
2009
+ `To continue it with everything it already read, call task with resume_agent_id="${partialId}".`,
2010
+ ].filter(Boolean);
2011
+ return finalize({ error: sections.join('\n\n') });
1660
2012
  }
1661
2013
  // Normal completion: the final assistant message is the sub-agent's answer.
1662
2014
  const text = capSubTaskText(extractSubTaskText(result, true));
1663
2015
  // Store the transcript so a follow-up can continue this agent rather than
1664
2016
  // re-running it from scratch, and tell the parent the id.
1665
2017
  //
1666
- // Only on NORMAL completion. A timed-out or failed run is deliberately not
1667
- // resumable: its transcript ends mid-thought, often mid-tool-call, and
1668
- // resuming from that state invites the model to build on work whose status
1669
- // it cannot determine. Those paths already salvage their partial output as
1670
- // TEXT, which is the safe way to carry that information forward.
2018
+ // (The stalled and stopped-early paths above register their transcripts too, marked
2019
+ // as partial work in the text they return; only a thrown failure is not resumable.)
1671
2020
  const agentId = resumed
1672
- ? ((0, agentRegistry_1.updateAgent)(resumed.id, result, agentScope), resumed.id)
1673
- : (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope);
2021
+ ? ((0, agentRegistry_1.updateAgent)(resumed.id, result, agentScope, options.sessionId), resumed.id)
2022
+ : (0, agentRegistry_1.rememberAgent)(agent?.name ?? null, (typeof input.description === 'string' && input.description.trim()) || prompt.slice(0, 80), result, agentScope, options.sessionId);
1674
2023
  const body = text || '(sub-task completed with no text output)';
1675
- return {
2024
+ return finalize({
1676
2025
  output: `${body}\n\n[resumable: this sub-agent is "${agentId}". To ask IT a follow-up — keeping ` +
1677
2026
  'everything it already read and concluded — call task again with resume_agent_id="' + agentId +
1678
2027
  '" instead of writing a new prompt from scratch.]',
1679
- };
2028
+ });
1680
2029
  }
1681
2030
  catch (err) {
1682
2031
  // Same salvage rule as the timeout path above, for the other way a sub-agent
@@ -1694,13 +2043,18 @@ started) {
1694
2043
  partial ? `Partial output before the failure:\n\n${partial}` : '',
1695
2044
  'Treat the above as PARTIAL, unverified work. Build on it rather than re-running the whole sub-task.',
1696
2045
  ].filter(Boolean);
1697
- return { error: sections.join('\n\n') };
2046
+ return finalize({ error: sections.join('\n\n') });
1698
2047
  }
1699
- return { error: `Sub-task failed: ${err.message}` };
2048
+ return finalize({ error: `Sub-task failed: ${err.message}` });
1700
2049
  }
1701
2050
  finally {
1702
2051
  clearInterval(stallWatchdog);
1703
2052
  clearInterval(parentAbortPoll);
2053
+ // Every return above already ran finalize(); this only catches an unexpected path.
2054
+ if (isoState && !(0, worktree_1.worktreeHasWork)(isoState))
2055
+ (0, worktree_1.removeWorktree)(isoState, { force: true });
2056
+ if (resumed)
2057
+ (0, agentRegistry_1.releaseAgentForResume)(resumed.id);
1704
2058
  }
1705
2059
  }
1706
2060
  // ─── Auto-compact ─────────────────────────────────────────────────────────────
@@ -1738,7 +2092,7 @@ const MODEL_CONTEXT_TOKENS = {
1738
2092
  ultra: 1000000,
1739
2093
  fast: 200000,
1740
2094
  // Anthropic, by real model id.
1741
- 'claude-sonnet-5': 1000000,
2095
+ 'claude-sonnet-5-5': 1000000,
1742
2096
  'claude-opus-5-5': 1000000,
1743
2097
  'claude-fable-5-1': 1000000,
1744
2098
  'claude-haiku-4-5-20251001': 200000,
@@ -1806,13 +2160,13 @@ function contextWindowFor(model) {
1806
2160
  // session that is real money accruing per turn long before the 1M wall.
1807
2161
  // Anthropic's own server-side compaction defaults its trigger to 150K
1808
2162
  // input tokens (docs: compact_20260112 default trigger 150000). We mirror
1809
- // that intent: start shedding already-consumed tool_result bulk at ~35% of
1810
- // a 1M window (~350K tokens) so the per-turn cache-read bill stops growing,
2163
+ // that intent: start shedding already-consumed tool_result bulk at ~120K
2164
+ // tokens (see compactionLimits) so the per-turn cache-read bill stops growing,
1811
2165
  // WITHOUT paying for a summariser model call and WITHOUT dropping any turn
1812
2166
  // (pruneOldToolResults keeps every tool_use/tool_result pair intact).
1813
2167
  //
1814
- // • SUMMARISE threshold (expensive: a real model call, lossy: drops whole
1815
- // turns): stays LATE, because summarise-of-summarise is the main cause of an
2168
+ // • SUMMARISE threshold (a model call, lossy: drops whole turns): Claude Code's
2169
+ // ~167K line (see compactionLimits). Later than prune, because summarise-of-summarise is the main cause of an
1816
2170
  // agent "forgetting" earlier work. Only when cheap pruning can't keep the
1817
2171
  // prompt under this line do we fall through to summarisation.
1818
2172
  //
@@ -1821,17 +2175,67 @@ function envFraction(name, fallback) {
1821
2175
  const v = Number(process.env[name]);
1822
2176
  return Number.isFinite(v) && v > 0 && v < 1 ? v : fallback;
1823
2177
  }
1824
- // Prune early (~35% of window), summarise late (~80% of window).
1825
- const AUTO_PRUNE_THRESHOLD = envFraction('NEXRALL_PRUNE_THRESHOLD', 0.35);
1826
- const AUTO_COMPACT_THRESHOLD = envFraction('NEXRALL_COMPACT_THRESHOLD', 0.8); // summarise when prompt > 80% of the window
2178
+ // Where auto-compaction fires, in TOKENS — Claude Code's rule, not a fraction of a 1M window.
2179
+ //
2180
+ // Claude Code summarises at `window − min(maxOutput, 20K) − 13K buffer`, on a 200K window
2181
+ // (~167K tokens). We used to wait for 80% of a 1M window (~800K): every request of a long
2182
+ // run re-read that whole prefix from cache, so a session cost ~5× more per request than
2183
+ // the same work in Claude Code long before anything was shed. The window used for this is
2184
+ // capped (default 200K, like Claude Code) — the model's real 1M window still bounds what
2185
+ // the backend will accept; this only decides when WE tidy up.
2186
+ //
2187
+ // NEXRALL_COMPACT_WINDOW / settings.json "autoCompactWindow": the cap (tokens). Set it
2188
+ // to e.g. 1000000 to use the whole window before compacting.
2189
+ // NEXRALL_PRUNE_THRESHOLD / NEXRALL_COMPACT_THRESHOLD: fractions of that capped window.
2190
+ const DEFAULT_COMPACT_WINDOW = 200000;
2191
+ const COMPACT_OUTPUT_RESERVE = 20000; // Claude Code: min(model max output, 20K)
2192
+ const COMPACT_BUFFER_TOKENS = 13000; // Claude Code's autocompact buffer
2193
+ const DEFAULT_PRUNE_FRACTION = 0.6; // cheap lossless prune ~120K, before the ~167K summarise
2194
+ function compactWindowCap(settingsRaw) {
2195
+ const fromEnv = Number(process.env.NEXRALL_COMPACT_WINDOW);
2196
+ if (Number.isFinite(fromEnv) && fromEnv >= 50000)
2197
+ return fromEnv;
2198
+ const fromSettings = Number(settingsRaw?.autoCompactWindow);
2199
+ if (Number.isFinite(fromSettings) && fromSettings >= 50000)
2200
+ return fromSettings;
2201
+ return DEFAULT_COMPACT_WINDOW;
2202
+ }
2203
+ /** Token counts at which auto-prune and auto-compact (summarise) fire for a model window. */
2204
+ function compactionLimits(contextWindow, settingsRaw) {
2205
+ const eff = Math.min(contextWindow, compactWindowCap(settingsRaw));
2206
+ const compactFrac = envFraction('NEXRALL_COMPACT_THRESHOLD', 0);
2207
+ const pruneFrac = envFraction('NEXRALL_PRUNE_THRESHOLD', 0);
2208
+ const compact = compactFrac
2209
+ ? eff * compactFrac
2210
+ : Math.max(eff * 0.5, eff - Math.min(COMPACT_OUTPUT_RESERVE, eff * 0.1) - COMPACT_BUFFER_TOKENS);
2211
+ const prune = Math.min(pruneFrac ? eff * pruneFrac : eff * DEFAULT_PRUNE_FRACTION, compact * 0.9);
2212
+ return { prune: Math.floor(prune), compact: Math.floor(compact) };
2213
+ }
1827
2214
  /** Auto-prune / auto-compact thresholds as fractions of the context window — for UI display (e.g. `/context`). */
1828
- function compactionThresholds() {
1829
- return { prune: AUTO_PRUNE_THRESHOLD, compact: AUTO_COMPACT_THRESHOLD };
2215
+ function compactionThresholds(contextWindow = 1000000, settingsRaw) {
2216
+ const { prune, compact } = compactionLimits(contextWindow, settingsRaw);
2217
+ return { prune: prune / contextWindow, compact: compact / contextWindow };
1830
2218
  }
1831
2219
  // Only bother pruning if it reclaims a meaningful amount — a tiny prune busts
1832
2220
  // the message-level prompt cache (the pruned prefix changes) for little gain,
1833
2221
  // so we require at least this many bytes reclaimed before accepting a prune.
1834
- const PRUNE_MIN_RECLAIM_BYTES = 256 * 1024; // 256 KB
2222
+ // Sized for the ~120K-token prune line: at 256 KB a prune could rarely reclaim enough to
2223
+ // qualify, so sessions skipped straight to the lossy summariser.
2224
+ const PRUNE_MIN_RECLAIM_BYTES = 96 * 1024; // 96 KB
2225
+ // Cache-aware prune floor. The 96 KB floor exists ONLY to avoid busting a WARM prompt
2226
+ // cache for a small gain. Once the session has been idle past the provider cache TTL
2227
+ // (Anthropic/OpenAI: 5 min), the whole prefix is re-written on the next request anyway,
2228
+ // so a prune at that moment costs nothing extra — accept a much smaller reclaim then.
2229
+ // Keyed by sessionId (same scheme as _subAgentBudgets) because runAgentLoop runs once
2230
+ // per user turn and the idle gap that matters is BETWEEN turns.
2231
+ const PRUNE_MIN_RECLAIM_BYTES_COLD = 16 * 1024; // 16 KB
2232
+ const CACHE_COLD_AFTER_MS = 5 * 60000;
2233
+ const _lastApiCallEndedAt = new Map();
2234
+ function pruneReclaimFloor(lastCallEndedAt, now = Date.now()) {
2235
+ return lastCallEndedAt !== undefined && now - lastCallEndedAt >= CACHE_COLD_AFTER_MS
2236
+ ? PRUNE_MIN_RECLAIM_BYTES_COLD
2237
+ : PRUNE_MIN_RECLAIM_BYTES;
2238
+ }
1835
2239
  const COMPACT_KEEP_MIN = 6; // always keep at least the last N messages verbatim
1836
2240
  /**
1837
2241
  * Bytes a compaction must reclaim to count as productive.
@@ -1991,10 +2395,27 @@ const MAX_TRANSCRIPT_CHARS = 600000;
1991
2395
  * middle — a middle-out elision that preserves both "what we set out to do" and
1992
2396
  * "where we are now", which is what the continuation summary needs most.
1993
2397
  */
2398
+ /**
2399
+ * Appends the backend-announced runtime-context block to the user message it was
2400
+ * attached to. No-op if that message is not a user turn or already ends with the
2401
+ * identical block. Exported for tests.
2402
+ */
2403
+ function persistRuntimeContext(messages, idx, text) {
2404
+ const m = messages[idx];
2405
+ if (!m || m.role !== 'user' || !Array.isArray(m.content))
2406
+ return false;
2407
+ const lastBlock = m.content[m.content.length - 1];
2408
+ if (lastBlock && lastBlock.type === 'text' && lastBlock.text === text)
2409
+ return false;
2410
+ messages[idx] = { ...m, content: [...m.content, { type: 'text', text }] };
2411
+ return true;
2412
+ }
1994
2413
  function transcriptOf(messages) {
1995
2414
  const parts = [];
1996
2415
  for (const m of messages) {
1997
2416
  for (const b of m.content) {
2417
+ if ((0, types_1.isRuntimeContextBlock)(b))
2418
+ continue; // editor scaffolding, not conversation
1998
2419
  if (b.type === 'text' && b.text) {
1999
2420
  parts.push(`${m.role.toUpperCase()}: ${(0, safeSlice_1.sliceSafeEnd)(b.text, 2000)}`);
2000
2421
  }
@@ -2253,7 +2674,7 @@ function originalTaskText(messages) {
2253
2674
  if (!first || !Array.isArray(first.content))
2254
2675
  return '';
2255
2676
  return first.content
2256
- .filter((b) => b.type === 'text' && b.text)
2677
+ .filter((b) => b.type === 'text' && b.text && !(0, types_1.isRuntimeContextBlock)(b))
2257
2678
  .map((b) => b.text)
2258
2679
  .join('\n')
2259
2680
  .trim();
@@ -2278,7 +2699,32 @@ class CompactionUnavailableError extends Error {
2278
2699
  this.name = 'CompactionUnavailableError';
2279
2700
  }
2280
2701
  }
2281
- async function autoCompactMessages(messages, options, ledger) {
2702
+ function makeCachedSummarizer(messages, requestOptions, abortSignal) {
2703
+ return async (instruction) => {
2704
+ const last = messages[messages.length - 1];
2705
+ if (!last || last.role !== 'user')
2706
+ return '';
2707
+ const lastContent = typeof last.content === 'string'
2708
+ ? [{ type: 'text', text: last.content }]
2709
+ : [...last.content];
2710
+ const probe = [
2711
+ ...messages.slice(0, -1),
2712
+ { ...last, content: [...lastContent, { type: 'text', text: instruction }] },
2713
+ ];
2714
+ const reply = await (0, client_1.streamChat)(probe, { ...requestOptions(), abortSignal, allowRestartAfterRender: true }, () => { });
2715
+ return reply.content
2716
+ .filter((b) => b.type === 'text')
2717
+ .map((b) => b.text ?? '')
2718
+ .join('')
2719
+ .trim();
2720
+ };
2721
+ }
2722
+ const CACHED_SUMMARY_INSTRUCTION = `[Context compaction — this is an automated request from the agent runtime, not the user.]\n` +
2723
+ `Do NOT call any tools and do NOT continue the task. Reply with text only: a concise bullet-point ` +
2724
+ `summary of this whole session so far that you will need to continue the work — the user's goal, ` +
2725
+ `key decisions, files changed (and how), commands run and their outcome, unresolved problems, and ` +
2726
+ `user preferences. Max 400 words.`;
2727
+ async function autoCompactMessages(messages, options, ledger, summarizeCached) {
2282
2728
  const cut = findSafeCutIndex(messages, messages.length - COMPACT_KEEP_MIN);
2283
2729
  if (cut < 2)
2284
2730
  return false; // nothing meaningful to fold
@@ -2296,44 +2742,64 @@ async function autoCompactMessages(messages, options, ledger) {
2296
2742
  `key decisions, files changed (and how), commands run, unresolved problems, and user preferences. Max 400 words.\n\n` +
2297
2743
  transcriptOf(toSummarize);
2298
2744
  let summary = '';
2299
- try {
2300
- const reply = await (0, client_1.streamChat)([{ role: 'user', content: [{ type: 'text', text: summaryPrompt }] }], {
2301
- // Run on the SAME model as the actual conversation. A previous version
2302
- // forced 'turbo' (Claude Sonnet 5) here on the theory that a mechanical
2303
- // "bullet-point this transcript" task doesn't need the user's tier — but
2304
- // that silently billed Anthropic (and made a real network call to a
2305
- // provider the user may not have configured/paid for) even when the
2306
- // whole session was running on OpenAI/DeepSeek/Qwen. Whatever the user
2307
- // is already paying for is used for compaction too, so there is never a
2308
- // surprise charge on a different provider. Falls back to the same
2309
- // default as the main loop (see `runAgentLoop`) only when no model was
2310
- // set at all.
2311
- model: options.model ?? 'turbo',
2312
- mode: 'ask', // summariser must not call tools; ask-mode discourages action
2313
- env: options.env,
2314
- clientType: options.clientType,
2315
- abortSignal: options.abortSignal,
2316
- // The summariser renders NOTHING (onEvent below is a no-op) and its result is
2317
- // read only from the returned message, so a restart has nothing to roll back —
2318
- // always safe. Worth enabling: a blip here used to abandon compaction entirely,
2319
- // which then let the very next turn hit the context wall it was meant to prevent.
2320
- allowRestartAfterRender: true,
2321
- }, () => { });
2322
- summary = reply.content
2323
- .filter((b) => b.type === 'text')
2324
- .map((b) => b.text ?? '')
2325
- .join('')
2326
- .trim();
2327
- }
2328
- catch {
2329
- // Summarisation FAILED — the model call itself threw (network blip, 529,
2330
- // provider quota). Distinguished from "ran fine but didn't help" by the
2331
- // caller, because the two must not feed the same circuit breaker: three
2332
- // transient network errors would otherwise permanently disable compaction
2333
- // for the rest of the run, leaving the context to grow until the turn dies
2334
- // with no recovery left. Leave history as is; the turn may still fit.
2335
- throw new CompactionUnavailableError();
2745
+ // Cache-friendly path first; any failure or empty reply falls back to the standalone
2746
+ // transcript summariser below, which always works but pays for the history uncached.
2747
+ if (summarizeCached) {
2748
+ try {
2749
+ summary = await summarizeCached(CACHED_SUMMARY_INSTRUCTION);
2750
+ }
2751
+ catch (err) {
2752
+ if (options.abortSignal?.aborted || err.name === 'AbortError')
2753
+ throw err;
2754
+ summary = '';
2755
+ }
2336
2756
  }
2757
+ if (!summary)
2758
+ try {
2759
+ const reply = await (0, client_1.streamChat)([{ role: 'user', content: [{ type: 'text', text: summaryPrompt }] }], {
2760
+ // Run on the SAME model as the actual conversation. A previous version
2761
+ // forced 'turbo' (Claude Sonnet 5) here on the theory that a mechanical
2762
+ // "bullet-point this transcript" task doesn't need the user's tier — but
2763
+ // that silently billed Anthropic (and made a real network call to a
2764
+ // provider the user may not have configured/paid for) even when the
2765
+ // whole session was running on OpenAI/DeepSeek/Qwen. Whatever the user
2766
+ // is already paying for is used for compaction too, so there is never a
2767
+ // surprise charge on a different provider. Falls back to the same
2768
+ // default as the main loop (see `runAgentLoop`) only when no model was
2769
+ // set at all.
2770
+ //
2771
+ // Cheapest model of the SAME vendor (Claude Code runs this kind of work on
2772
+ // Haiku). Safe to downgrade HERE because this fallback sends a fresh
2773
+ // transcript — it shares no cached prefix with the conversation, unlike
2774
+ // summarizeCached above, which must stay on the conversation's own model.
2775
+ // Never crosses providers; falls back to the session model if the
2776
+ // catalogue is not loaded.
2777
+ model: (0, modelCatalogue_1.cheapestSameVendorModel)(options.model ?? 'turbo') ?? options.model ?? 'turbo',
2778
+ mode: 'ask', // summariser must not call tools; ask-mode discourages action
2779
+ env: options.env,
2780
+ clientType: options.clientType,
2781
+ abortSignal: options.abortSignal,
2782
+ // The summariser renders NOTHING (onEvent below is a no-op) and its result is
2783
+ // read only from the returned message, so a restart has nothing to roll back —
2784
+ // always safe. Worth enabling: a blip here used to abandon compaction entirely,
2785
+ // which then let the very next turn hit the context wall it was meant to prevent.
2786
+ allowRestartAfterRender: true,
2787
+ }, () => { });
2788
+ summary = reply.content
2789
+ .filter((b) => b.type === 'text')
2790
+ .map((b) => b.text ?? '')
2791
+ .join('')
2792
+ .trim();
2793
+ }
2794
+ catch {
2795
+ // Summarisation FAILED — the model call itself threw (network blip, 529,
2796
+ // provider quota). Distinguished from "ran fine but didn't help" by the
2797
+ // caller, because the two must not feed the same circuit breaker: three
2798
+ // transient network errors would otherwise permanently disable compaction
2799
+ // for the rest of the run, leaving the context to grow until the turn dies
2800
+ // with no recovery left. Leave history as is; the turn may still fit.
2801
+ throw new CompactionUnavailableError();
2802
+ }
2337
2803
  if (!summary)
2338
2804
  return false;
2339
2805
  // Replace the summarized head with a single user summary message. The cut is
@@ -2375,7 +2841,9 @@ async function maybeCompactMemory(scope, options) {
2375
2841
  // Same reasoning as autoCompactMessages' summariser: run on the same
2376
2842
  // model as the actual conversation rather than forcing a fixed tier
2377
2843
  // (which silently billed Anthropic regardless of the user's provider).
2378
- model: options.model ?? 'turbo',
2844
+ // Cheapest SAME-vendor model: a fresh prompt with no shared cache prefix,
2845
+ // and merging a few memory bullets does not need the session's top tier.
2846
+ model: (0, modelCatalogue_1.cheapestSameVendorModel)(options.model ?? 'turbo') ?? options.model ?? 'turbo',
2379
2847
  mode: 'ask',
2380
2848
  env: options.env,
2381
2849
  clientType: options.clientType,
@@ -2446,8 +2914,9 @@ async function compactMessagesForResume(messages, opts) {
2446
2914
  // only at the late one. On resume this matters most: a stored session is resent
2447
2915
  // whole on the first turn, so shedding old tool_result bulk up front is exactly
2448
2916
  // what stops that first message being billed at full size.
2449
- const overPruneThreshold = () => tokenGuess > contextWindow * AUTO_PRUNE_THRESHOLD || bodyBytes > MAX_BODY_BYTES;
2450
- const overCompactThreshold = () => tokenGuess > contextWindow * AUTO_COMPACT_THRESHOLD || bodyBytes > MAX_BODY_BYTES;
2917
+ const limits = compactionLimits(contextWindow, settings.raw);
2918
+ const overPruneThreshold = () => tokenGuess > limits.prune || bodyBytes > MAX_BODY_BYTES;
2919
+ const overCompactThreshold = () => tokenGuess > limits.compact || bodyBytes > MAX_BODY_BYTES;
2451
2920
  if (!overPruneThreshold())
2452
2921
  return false;
2453
2922
  let compacted = false;
@@ -2525,7 +2994,7 @@ async function compactMessagesForResume(messages, opts) {
2525
2994
  */
2526
2995
  async function dispatchSubAgent(input, options, agentTypes) {
2527
2996
  const depth = options._depth ?? 0;
2528
- const overBudget = claimSessionSubAgentSlot(options.workDir, options.onNotice ?? options.onText);
2997
+ const overBudget = claimSessionSubAgentSlot(options.workDir, options.onNotice ?? options.onText, options.sessionId);
2529
2998
  if (overBudget)
2530
2999
  return { error: overBudget };
2531
3000
  const childDepth = depth + 1;
@@ -2543,7 +3012,7 @@ async function dispatchSubAgent(input, options, agentTypes) {
2543
3012
  }
2544
3013
  finally {
2545
3014
  if (!started.value)
2546
- refundSessionSubAgentSlot();
3015
+ refundSessionSubAgentSlot(options.sessionId);
2547
3016
  const n = (_inFlightByDepth.get(childDepth) ?? 1) - 1;
2548
3017
  if (n > 0)
2549
3018
  _inFlightByDepth.set(childDepth, n);
@@ -2551,11 +3020,81 @@ async function dispatchSubAgent(input, options, agentTypes) {
2551
3020
  _inFlightByDepth.delete(childDepth);
2552
3021
  }
2553
3022
  }
3023
+ /**
3024
+ * `task` with run_in_background — start the sub-agent detached and return at once.
3025
+ *
3026
+ * The child runs with its OWN abort signal (the user's Stop on the main turn does not
3027
+ * kill it; the hub's stop() does), renders nothing into the chat (its tool calls are
3028
+ * only counted for the "N agents" badge), and cannot prompt the user: a call that the
3029
+ * project's rules or the session mode do not already allow is refused with an
3030
+ * explanation, which the child then reports (Claude Code auto-denies the same way).
3031
+ * Its final report reaches the main agent as a <task-notification>.
3032
+ */
3033
+ function dispatchBackgroundSubAgent(input, options, agentTypes, hub) {
3034
+ const overBudget = claimSessionSubAgentSlot(options.workDir, options.onNotice ?? options.onText, options.sessionId);
3035
+ if (overBudget)
3036
+ return { error: overBudget };
3037
+ const agentType = (typeof input.subagent_type === 'string' && input.subagent_type) || 'general-purpose';
3038
+ const description = (typeof input.description === 'string' && input.description.trim())
3039
+ || String(input.prompt ?? '').trim().slice(0, 60) || agentType;
3040
+ const mode = (0, modePolicy_1.parseMode)(options.mode);
3041
+ const backgroundPermission = async (req) => {
3042
+ const rule = (0, rules_1.evaluatePermission)((0, rules_1.loadSettings)(options.workDir).permissions, req.tool, req.input, options.workDir);
3043
+ if (rule === 'deny')
3044
+ throw new ToolNotAllowedError(`\`${req.tool}\` is denied by a permission rule in this project.`);
3045
+ if (rule === 'allow')
3046
+ return true;
3047
+ const destructive = !!(0, destructive_1.isDestructiveBash)(req.tool, req.input);
3048
+ if ((0, modePolicy_1.decide)({ tool: req.tool, input: req.input, mode, destructive }) === 'allow')
3049
+ return true;
3050
+ throw new ToolNotAllowedError(`Background agents cannot ask the user for permission, and \`${req.tool}\` is not pre-approved in the ` +
3051
+ `current mode (${mode}). Do not retry it: finish what you can without it and say in your report that ` +
3052
+ 'this step needs the main agent (or the user) to run it.');
3053
+ };
3054
+ const started = hub.start({ description, agentType }, async ({ abort, onToolUse }) => {
3055
+ const didStart = { value: false };
3056
+ try {
3057
+ return await runSubTask(input, {
3058
+ ...options,
3059
+ abortSignal: abort,
3060
+ backgroundAgents: undefined,
3061
+ requestPermission: backgroundPermission,
3062
+ onText: () => { },
3063
+ onNotice: () => { },
3064
+ onToolUse: (name) => onToolUse(name),
3065
+ onToolResult: () => { },
3066
+ onToolStreamChunk: undefined,
3067
+ onThinking: undefined,
3068
+ onThinkingDelta: undefined,
3069
+ onThinkingProgress: undefined,
3070
+ onStreamRestart: undefined,
3071
+ onRetry: undefined,
3072
+ onRetryResolved: undefined,
3073
+ }, agentTypes, didStart);
3074
+ }
3075
+ finally {
3076
+ if (!didStart.value)
3077
+ refundSessionSubAgentSlot(options.sessionId);
3078
+ }
3079
+ });
3080
+ if ('error' in started) {
3081
+ refundSessionSubAgentSlot(options.sessionId);
3082
+ return { error: started.error };
3083
+ }
3084
+ return {
3085
+ output: `Started background agent ${started.id} ("${description}", ${agentType}). It runs independently: its ` +
3086
+ 'report will arrive automatically as a <task-notification> in a later turn. Do not poll for it, do not ' +
3087
+ 'wait for it, and do not duplicate its work — continue with something else, or end your turn if nothing ' +
3088
+ 'else needs doing now.',
3089
+ };
3090
+ }
2554
3091
  // ─── Agent Loop ───────────────────────────────────────────────────────────────
2555
3092
  async function runAgentLoop(initialMessages, options) {
2556
3093
  const messages = [...initialMessages];
2557
3094
  const model = options.model ?? 'turbo';
2558
- const hooks = loadHooks(options.workDir);
3095
+ // An agent definition's own hooks apply only to its own run (appended after the
3096
+ // project's, which run first) — runSubTask sets _agentHooks per child, never inherits it.
3097
+ const hooks = withAgentHooks(loadHooks(options.workDir), options._agentHooks);
2559
3098
  const depth = options._depth ?? 0;
2560
3099
  const agentScope = options._agentScope ?? 'root';
2561
3100
  // ── Audit trail (opt-in) ────────────────────────────────────────────────────
@@ -2627,6 +3166,62 @@ async function runAgentLoop(initialMessages, options) {
2627
3166
  options.worktree ? (0, worktreeEnforcement_1.worktreeInstructions)(options.worktree) : null,
2628
3167
  options.nexrallMd ?? null,
2629
3168
  ].filter((s) => s !== null).join('\n\n---\n\n') || undefined; // '' (nothing at all was present) must behave exactly like the old `: options.nexrallMd` fallback (undefined), not an empty-but-truthy string.
3169
+ // The request options every main-loop call sends. Shared with the compaction
3170
+ // summariser so its request has the SAME system prompt and tools as the conversation
3171
+ // — that is what lets the summary call read the whole history from cache (0.10×)
3172
+ // instead of re-sending it as a fresh, uncached transcript.
3173
+ const chatRequestOptions = () => ({
3174
+ model,
3175
+ mode: options.mode,
3176
+ effort: options.effort,
3177
+ env: options.env,
3178
+ editorContext: options.editorContext,
3179
+ nexrallMd: planAndWorktreeAwareNexrallMd,
3180
+ clientType: options.clientType,
3181
+ abortSignal: options.abortSignal,
3182
+ // MCP tools, plus the per-agent memory tool when (and only when) this run
3183
+ // is a sub-agent that declared a `memory:` scope. Declaring it through
3184
+ // extraTools rather than the backend's static catalogue keeps it invisible
3185
+ // to every other run: a tool the model cannot see is one it cannot try,
3186
+ // which is better than advertising it everywhere and refusing it at the gate.
3187
+ extraTools: [
3188
+ ...(options.mcpManager?.getAnthropicTools() ?? [])
3189
+ .filter((t) => !options._mcpServerAllowlist || options._mcpServerAllowlist.has(String(t.name).split('__')[0])),
3190
+ ...(options._agentMemory ? [exports.AGENT_MEMORY_TOOL_SCHEMA] : []),
3191
+ ],
3192
+ // Derived from the run's ACTUAL allowlist rather than asserted separately,
3193
+ // so the prompt's memory instructions cannot drift from what is permitted.
3194
+ // That drift is the bug being fixed: sub-agents were told they MUST call
3195
+ // memory_write, which no sub-agent allowlist contains.
3196
+ // Sub-agents NEVER write the shared store (the main agent decides what to persist) —
3197
+ // an unnamed/general-purpose child has no allowlist, which used to read as "may write".
3198
+ canWriteSharedMemory: depth > 0
3199
+ ? false
3200
+ : options._allowedTools
3201
+ ? options._allowedTools.has('memory_write')
3202
+ : true, // no allowlist = main agent = may write
3203
+ hasOwnAgentStore: !!options._agentMemory,
3204
+ // Same derivation, same reason, for the network. The base prompt's
3205
+ // "Web search" section tells the run to reach the network, but
3206
+ // READ_ONLY_TOOLS (explorer, security-auditor, ...) contains neither
3207
+ // `fetch_url` nor `web_search`. MEASURED on the 195-trial 2026-08-10
3208
+ // deepseek-v4-pro benchmark: 29 of 51 `fetch_url` failures and 3 of 14
3209
+ // `web_search` failures were the permission gate refusing a sub-agent
3210
+ // that its own system prompt had just invited to browse — every one a
3211
+ // wasted round-trip on a paid turn.
3212
+ canBrowseWeb: options._allowedTools
3213
+ ? options._allowedTools.has('fetch_url') || options._allowedTools.has('web_search')
3214
+ : true, // no allowlist = main agent = may browse
3215
+ // Only the tools this run may use are sent as schemas — explorer no longer pays for
3216
+ // (and gets refused on) write, office and image tools. Sorted so the tools cache
3217
+ // prefix is identical for every run of the same agent type.
3218
+ toolAllowlist: options._allowedTools ? [...options._allowedTools].sort() : undefined,
3219
+ // Withholds the `task` schema and its instructions when this run cannot
3220
+ // delegate — see canSpawnSubAgents.
3221
+ canSpawnSubAgents: maySpawn,
3222
+ agents: agentsCatalogue || undefined,
3223
+ skills: skillsCatalogue || undefined,
3224
+ });
2630
3225
  // Optional OS-level bash sandbox (opt-in via settings.json "sandbox").
2631
3226
  const sandboxCfg = (0, sandbox_1.parseSandboxConfig)(settings.raw.sandbox) ?? undefined;
2632
3227
  // Soft iteration budget + optional auto-continue past it (see resolvers above).
@@ -2637,6 +3232,11 @@ async function runAgentLoop(initialMessages, options) {
2637
3232
  // Live catalogue first (see contextWindowFor's doc comment above for why),
2638
3233
  // same fallback chain this call site always used otherwise.
2639
3234
  const contextWindow = (0, modelCatalogue_1.liveContextWindowFor)(model, MODEL_CONTEXT_TOKENS[model] ?? 200000);
3235
+ const compactLimits = compactionLimits(contextWindow, settings.raw);
3236
+ // Per-run: a sub-agent's own runAgentLoop gets its own, so it never sees its parent's reads.
3237
+ const readDedupe = new readDedupe_1.ReadDedupe(options.workDir);
3238
+ // Only the main agent owns background agents (runSubTask never passes the hub down).
3239
+ const drainBackgroundNotifications = () => depth === 0 ? (options.backgroundAgents?.takeNotifications() ?? []) : [];
2640
3240
  // Live prompt-size estimate, updated from usage events after every stream.
2641
3241
  let lastPromptTokens = 0;
2642
3242
  let compacting = false; // re-entrancy guard — compaction itself calls streamChat
@@ -2758,22 +3358,22 @@ async function runAgentLoop(initialMessages, options) {
2758
3358
  // every single turn on a long task. Runs at a turn boundary only.
2759
3359
  //
2760
3360
  // THREE nested triggers, cheapest/earliest first:
2761
- // 1. PRUNE pressure — prompt crossed AUTO_PRUNE_THRESHOLD (~35% of window)
3361
+ // 1. PRUNE pressure — prompt crossed compactLimits.prune (~120K tokens)
2762
3362
  // OR the body crossed MAX_BODY_BYTES. Handled by the CHEAP, structure-
2763
3363
  // preserving prune (no model call, keeps every turn). This is the big
2764
3364
  // cost win: it fires ~2× earlier than summarisation used to, shedding
2765
3365
  // already-consumed tool_result bulk (read_file/bash/grep output the
2766
3366
  // model has long since acted on) so the per-turn cache-read bill stops
2767
3367
  // compounding well before the old 80% wall.
2768
- // 2. SUMMARISE pressure — prompt crossed AUTO_COMPACT_THRESHOLD (~80%) or
3368
+ // 2. SUMMARISE pressure — prompt crossed compactLimits.compact (~167K, Claude Code's line) or
2769
3369
  // the body is STILL over MAX_BODY_BYTES after pruning. Only then do we
2770
3370
  // pay for a summariser call + drop whole turns (kept late on purpose:
2771
3371
  // summarise-of-summarise is what makes an agent "forget" earlier work).
2772
3372
  // The byte trigger also fires even on turn 0 of a resumed large session,
2773
3373
  // where lastPromptTokens is 0.
2774
3374
  let bodyBytes = estimateBodyBytes(messages);
2775
- const prunePressure = lastPromptTokens > contextWindow * AUTO_PRUNE_THRESHOLD;
2776
- const tokenPressure = lastPromptTokens > contextWindow * AUTO_COMPACT_THRESHOLD;
3375
+ const prunePressure = lastPromptTokens > compactLimits.prune;
3376
+ const tokenPressure = lastPromptTokens > compactLimits.compact;
2777
3377
  let bytePressure = bodyBytes > MAX_BODY_BYTES;
2778
3378
  // Cheap prune first — on token OR byte pressure. Require a meaningful reclaim
2779
3379
  // (PRUNE_MIN_RECLAIM_BYTES): a tiny prune would bust the message-level prompt
@@ -2784,16 +3384,18 @@ async function runAgentLoop(initialMessages, options) {
2784
3384
  // The gate lives INSIDE pruneOldToolResults now (atomic: it measures the
2785
3385
  // total first and mutates nothing if it's below the floor), so a declined
2786
3386
  // prune never invalidates the prompt cache.
2787
- const reclaimed = pruneOldToolResults(messages, PRUNE_MIN_RECLAIM_BYTES);
3387
+ // Main agent only: a sub-agent's own cache is warm while it works, but it would read
3388
+ // the PARENT's last-call time, which goes stale exactly while the parent waits on it.
3389
+ const floor = depth === 0
3390
+ ? pruneReclaimFloor(_lastApiCallEndedAt.get(options.sessionId || PROCESS_BUDGET_KEY))
3391
+ : PRUNE_MIN_RECLAIM_BYTES;
3392
+ const reclaimed = pruneOldToolResults(messages, floor);
2788
3393
  if (reclaimed > 0) {
2789
3394
  bodyBytes = estimateBodyBytes(messages);
2790
3395
  bytePressure = bodyBytes > MAX_BODY_BYTES;
2791
- // System notice, NOT model output — route through onNotice (falls back to
2792
- // onText only if the caller hasn't been updated yet) so the UI can render
2793
- // it as a small dim/system line instead of splicing it into the middle of
2794
- // the assistant's own streamed reply, where it reads like the model itself
2795
- // said "Trimmed ~0.3MB of already-processed tool output…".
2796
- (options.onNotice ?? options.onText)(`\u267b\ufe0f Trimmed ~${(reclaimed / (1024 * 1024)).toFixed(1)}MB of already-processed tool output to keep this chat cheap to continue.`);
3396
+ // Silent by design: routine housekeeping the user can't act on. It used to
3397
+ // print "♻️ Trimmed ~0.1MB of already-processed tool output…" into the chat,
3398
+ // which read as noise (and, before onNotice, as the model's own words).
2797
3399
  }
2798
3400
  }
2799
3401
  // Re-arm the breaker once the conversation has grown materially past the
@@ -2816,7 +3418,7 @@ async function runAgentLoop(initialMessages, options) {
2816
3418
  let did = false;
2817
3419
  let unavailable = false;
2818
3420
  try {
2819
- did = await autoCompactMessages(messages, options, ledger);
3421
+ did = await autoCompactMessages(messages, options, ledger, makeCachedSummarizer(messages, chatRequestOptions, options.abortSignal));
2820
3422
  }
2821
3423
  catch (err) {
2822
3424
  if (!(err instanceof CompactionUnavailableError))
@@ -2830,9 +3432,9 @@ async function runAgentLoop(initialMessages, options) {
2830
3432
  const reason = bytePressure
2831
3433
  ? `body ~${(bodyBytes / (1024 * 1024)).toFixed(1)}MB`
2832
3434
  : 'context window';
2833
- // Same reasoning as above: this is a system notice about housekeeping,
2834
- // not part of the model's answer — keep it out of the text bubble.
2835
- (options.onNotice ?? options.onText)(`\u267b\ufe0f Auto-compacted earlier conversation to stay within the ${reason}.`);
3435
+ // Silent, same as the prune above — routine housekeeping. Only the
3436
+ // actionable "auto-compaction paused" warning below is surfaced.
3437
+ void reason;
2836
3438
  // Refresh local pressure so the rest of THIS iteration sees the new size.
2837
3439
  bodyBytes = bytesAfter;
2838
3440
  bytePressure = bodyBytes > MAX_BODY_BYTES;
@@ -2874,6 +3476,16 @@ async function runAgentLoop(initialMessages, options) {
2874
3476
  if (!autoContinue && iteration === budget - 5 && budget > 5) {
2875
3477
  options.onText(`\n⚠️ Approaching the ${budget}-step limit (step ${iteration + 1}). Please wrap up and summarise what has been done.\n`);
2876
3478
  }
3479
+ // Runtime-context block the backend appended to this request's final user message
3480
+ // (see SSERuntimeContextEvent). Applied only AFTER streamChat succeeds: mutating
3481
+ // `messages` mid-stream would change the request hash a same-turnId retry sends,
3482
+ // and the backend would then run (and bill) it as a brand-new turn.
3483
+ let pendingRuntimeContext = null;
3484
+ const sentUserIdx = messages.length - 1;
3485
+ // Pure-read tools announced mid-stream (tool_use_ready) start here, before the
3486
+ // message is complete. Results are only reused after every gate below passes.
3487
+ const prefetcher = new toolPrefetch_1.ToolPrefetcher((toolName, toolInput) => (0, executor_1.executeTool)(toolName, toolInput, options.abortSignal, sandboxCfg, options.workDir, agentScope, undefined, options.selfPeer));
3488
+ const prefetchOn = (0, toolPrefetch_1.toolPrefetchEnabled)();
2877
3489
  // Build SSE event handler
2878
3490
  const onEvent = (event) => {
2879
3491
  switch (event.type) {
@@ -2943,11 +3555,20 @@ async function runAgentLoop(initialMessages, options) {
2943
3555
  // an assumption: tools are dispatched from `assistantMessage.content` AFTER
2944
3556
  // streamChat() returns, and a restarted attempt threw instead of returning.
2945
3557
  // So no tool ran, no file changed, no checkpoint or hook fired.
3558
+ // The replacement attempt generates new tool_use ids; drop early reads.
3559
+ prefetcher.clear();
2946
3560
  options.onStreamRestart?.(event.reason, event.discardedChars);
2947
3561
  break;
2948
3562
  case 'balance_status':
2949
3563
  options.onBalanceStatus?.(event.balance, event.zero);
2950
3564
  break;
3565
+ case 'runtime_context':
3566
+ pendingRuntimeContext = event.text;
3567
+ break;
3568
+ case 'tool_use_ready':
3569
+ if (prefetchOn && !options.abortSignal?.aborted)
3570
+ prefetcher.start(event.id, event.name, event.input);
3571
+ break;
2951
3572
  case 'done':
2952
3573
  break;
2953
3574
  case 'error':
@@ -2965,47 +3586,7 @@ async function runAgentLoop(initialMessages, options) {
2965
3586
  const _apiCallStartedAt = Date.now();
2966
3587
  try {
2967
3588
  assistantMessage = await (0, client_1.streamChat)(messages, {
2968
- model,
2969
- mode: options.mode,
2970
- effort: options.effort,
2971
- env: options.env,
2972
- editorContext: options.editorContext,
2973
- nexrallMd: planAndWorktreeAwareNexrallMd,
2974
- clientType: options.clientType,
2975
- abortSignal: options.abortSignal,
2976
- // MCP tools, plus the per-agent memory tool when (and only when) this run
2977
- // is a sub-agent that declared a `memory:` scope. Declaring it through
2978
- // extraTools rather than the backend's static catalogue keeps it invisible
2979
- // to every other run: a tool the model cannot see is one it cannot try,
2980
- // which is better than advertising it everywhere and refusing it at the gate.
2981
- extraTools: [
2982
- ...(options.mcpManager?.getAnthropicTools() ?? []),
2983
- ...(options._agentMemory ? [exports.AGENT_MEMORY_TOOL_SCHEMA] : []),
2984
- ],
2985
- // Derived from the run's ACTUAL allowlist rather than asserted separately,
2986
- // so the prompt's memory instructions cannot drift from what is permitted.
2987
- // That drift is the bug being fixed: sub-agents were told they MUST call
2988
- // memory_write, which no sub-agent allowlist contains.
2989
- canWriteSharedMemory: options._allowedTools
2990
- ? options._allowedTools.has('memory_write')
2991
- : true, // no allowlist = main agent = may write
2992
- hasOwnAgentStore: !!options._agentMemory,
2993
- // Same derivation, same reason, for the network. The base prompt's
2994
- // "Web search" section tells the run to reach the network, but
2995
- // READ_ONLY_TOOLS (explorer, security-auditor, ...) contains neither
2996
- // `fetch_url` nor `web_search`. MEASURED on the 195-trial 2026-08-10
2997
- // deepseek-v4-pro benchmark: 29 of 51 `fetch_url` failures and 3 of 14
2998
- // `web_search` failures were the permission gate refusing a sub-agent
2999
- // that its own system prompt had just invited to browse — every one a
3000
- // wasted round-trip on a paid turn.
3001
- canBrowseWeb: options._allowedTools
3002
- ? options._allowedTools.has('fetch_url')
3003
- : true, // no allowlist = main agent = may browse
3004
- // Withholds the `task` schema and its instructions when this run cannot
3005
- // delegate — see canSpawnSubAgents.
3006
- canSpawnSubAgents: maySpawn,
3007
- agents: agentsCatalogue || undefined,
3008
- skills: skillsCatalogue || undefined,
3589
+ ...chatRequestOptions(),
3009
3590
  // Only allow a post-render restart when the caller actually implements the
3010
3591
  // rollback. Without a handler the partial output can't be un-rendered, so we
3011
3592
  // keep the old conservative behaviour (fail the turn) rather than duplicate
@@ -3015,6 +3596,12 @@ async function runAgentLoop(initialMessages, options) {
3015
3596
  // Measured on the SUCCESS path only — a throw below reports its own duration
3016
3597
  // in the catch block, so this can't double-report the same call.
3017
3598
  options.onApiCallDuration?.(Date.now() - _apiCallStartedAt);
3599
+ if (depth === 0)
3600
+ _lastApiCallEndedAt.set(options.sessionId || PROCESS_BUDGET_KEY, Date.now());
3601
+ // Persist the context block where the backend put it, so later rounds carry it
3602
+ // in the cached history and the backend stops re-sending it uncached.
3603
+ if (pendingRuntimeContext)
3604
+ persistRuntimeContext(messages, sentUserIdx, pendingRuntimeContext);
3018
3605
  }
3019
3606
  catch (err) {
3020
3607
  options.onApiCallDuration?.(Date.now() - _apiCallStartedAt);
@@ -3070,7 +3657,7 @@ async function runAgentLoop(initialMessages, options) {
3070
3657
  stopReason = 'team-unavailable';
3071
3658
  break;
3072
3659
  }
3073
- runSimpleHooks(hooks.OnError, options.workDir);
3660
+ await runSimpleHooks(hooks.OnError, options.workDir);
3074
3661
  // Carry the work already done in this turn out with the error. The loop
3075
3662
  // owns a COPY of the caller's history, so a plain throw would strand every
3076
3663
  // completed tool round inside this function and the caller would fall back
@@ -3094,7 +3681,8 @@ async function runAgentLoop(initialMessages, options) {
3094
3681
  if (assistantMessage.content.length === 0) {
3095
3682
  const queued = options.takePendingInput?.() ?? [];
3096
3683
  const peerMsgs = options.drainPeerMessages?.() ?? [];
3097
- if (queued.length || peerMsgs.length) {
3684
+ const bgDone = drainBackgroundNotifications();
3685
+ if (queued.length || peerMsgs.length || bgDone.length) {
3098
3686
  const parts = [];
3099
3687
  if (queued.length) {
3100
3688
  parts.push(queued.join('\n\n'));
@@ -3104,6 +3692,8 @@ async function runAgentLoop(initialMessages, options) {
3104
3692
  parts.push(renderPeerMessages(peerMsgs));
3105
3693
  options.onPeerMessage?.(peerMsgs);
3106
3694
  }
3695
+ if (bgDone.length)
3696
+ parts.push((0, backgroundAgents_1.formatBackgroundNotifications)(bgDone));
3107
3697
  messages.push({ role: 'user', content: [{ type: 'text', text: parts.join('\n\n') }] });
3108
3698
  continue;
3109
3699
  }
@@ -3147,7 +3737,8 @@ async function runAgentLoop(initialMessages, options) {
3147
3737
  iteration--;
3148
3738
  continue;
3149
3739
  }
3150
- runSimpleHooks(hooks.PostMessageComplete, options.workDir);
3740
+ if (depth === 0)
3741
+ await runSimpleHooks(hooks.PostMessageComplete, options.workDir);
3151
3742
  stopReason = assistantMessage.stopReason === 'max_tokens' ? 'output-limit' : 'empty-response';
3152
3743
  break;
3153
3744
  }
@@ -3200,7 +3791,8 @@ async function runAgentLoop(initialMessages, options) {
3200
3791
  if (toolUseBlocks.length === 0) {
3201
3792
  const queued = options.takePendingInput?.() ?? [];
3202
3793
  const peerMsgs = options.drainPeerMessages?.() ?? [];
3203
- if (queued.length || peerMsgs.length) {
3794
+ const bgDone = drainBackgroundNotifications();
3795
+ if (queued.length || peerMsgs.length || bgDone.length) {
3204
3796
  const parts = [];
3205
3797
  if (queued.length) {
3206
3798
  parts.push(queued.join('\n\n'));
@@ -3210,6 +3802,8 @@ async function runAgentLoop(initialMessages, options) {
3210
3802
  parts.push(renderPeerMessages(peerMsgs));
3211
3803
  options.onPeerMessage?.(peerMsgs);
3212
3804
  }
3805
+ if (bgDone.length)
3806
+ parts.push((0, backgroundAgents_1.formatBackgroundNotifications)(bgDone));
3213
3807
  messages.push({ role: 'user', content: [{ type: 'text', text: parts.join('\n\n') }] });
3214
3808
  continue;
3215
3809
  }
@@ -3307,15 +3901,21 @@ async function runAgentLoop(initialMessages, options) {
3307
3901
  continue;
3308
3902
  }
3309
3903
  }
3310
- runSimpleHooks(hooks.PostMessageComplete, options.workDir);
3904
+ if (depth === 0)
3905
+ await runSimpleHooks(hooks.PostMessageComplete, options.workDir);
3311
3906
  stopReason = 'clean';
3312
3907
  break;
3313
3908
  }
3314
3909
  // 5. Execute all tool uses in parallel
3910
+ // Early reads are only valid if nothing in this message can change the files
3911
+ // they read (a write ordered before them would otherwise be missed).
3912
+ const prefetchSafe = prefetchOn && (0, toolPrefetch_1.isPrefetchSafeMessage)(toolUseBlocks.map((b) => b.name));
3315
3913
  const toolResults = await Promise.all(toolUseBlocks.map(async (block) => {
3316
3914
  const { id, name, input } = block;
3317
- // Notify caller about pending tool use
3318
- options.onToolUse(name, input);
3915
+ // Notify caller about pending tool use. `meta.id` lets the UI pair the result
3916
+ // with THIS row even when parallel sub-agents interleave their events.
3917
+ const toolMeta = { id };
3918
+ options.onToolUse(name, input, undefined, toolMeta);
3319
3919
  let result;
3320
3920
  // Checked BEFORE requesting permission: no point asking the user to
3321
3921
  // approve a diff/preview built from possibly-garbage truncated input
@@ -3338,7 +3938,7 @@ async function runAgentLoop(initialMessages, options) {
3338
3938
  : '') +
3339
3939
  `so the full response fits comfortably under the per-turn output budget.`,
3340
3940
  };
3341
- options.onToolResult(name, result);
3941
+ options.onToolResult(name, result, undefined, toolMeta);
3342
3942
  return { block: { ...block, id }, result };
3343
3943
  }
3344
3944
  // ── Plan mode: a session-wide read-only lock ────────────────────────
@@ -3352,7 +3952,7 @@ async function runAgentLoop(initialMessages, options) {
3352
3952
  const refusal = (0, planMode_1.checkPlanMode)(name, input);
3353
3953
  if (refusal) {
3354
3954
  result = { error: refusal.message };
3355
- options.onToolResult(name, result);
3955
+ options.onToolResult(name, result, undefined, toolMeta);
3356
3956
  return { block: { ...block, id }, result };
3357
3957
  }
3358
3958
  }
@@ -3366,7 +3966,7 @@ async function runAgentLoop(initialMessages, options) {
3366
3966
  const refusal = (0, worktreeEnforcement_1.checkWorktreeIsolation)(name, input, options.worktree, options.workDir);
3367
3967
  if (refusal) {
3368
3968
  result = { error: refusal.message };
3369
- options.onToolResult(name, result);
3969
+ options.onToolResult(name, result, undefined, toolMeta);
3370
3970
  return { block: { ...block, id }, result };
3371
3971
  }
3372
3972
  }
@@ -3397,10 +3997,36 @@ async function runAgentLoop(initialMessages, options) {
3397
3997
  // depth 0) needs its own limiter. Extracted into dispatchSubAgent so
3398
3998
  // `@nexrall/agent`'s delegate()/spawn() drive the SAME path instead of
3399
3999
  // reimplementing depth/budget/concurrency enforcement a second time.
3400
- result = await dispatchSubAgent(input, auditOptions, agentTypes);
4000
+ // Background only for the MAIN agent, and only when the host can deliver the
4001
+ // result later; otherwise it quietly runs in the foreground.
4002
+ const wantsBackground = input.run_in_background === true
4003
+ || (input.run_in_background !== false && !!(0, agentTypes_1.findAgentType)(agentTypes, String(input.subagent_type ?? ''))?.background);
4004
+ const taskOptions = { ...auditOptions, _taskToolUseId: id };
4005
+ // PreToolUse/PostToolUse cover spawns too (Claude Code's matcher "Task"/"Agent"
4006
+ // works the same way): a hook can veto a delegation or audit what came back.
4007
+ const preTask = await runToolHooks(hooks.PreToolUse, 'PreToolUse', name, input, options.workDir);
4008
+ if (preTask.block) {
4009
+ result = { error: `Blocked by PreToolUse hook: ${preTask.reason}` };
4010
+ }
4011
+ else {
4012
+ result = wantsBackground && depth === 0 && options.backgroundAgents && !input.resume_agent_id
4013
+ ? dispatchBackgroundSubAgent(input, taskOptions, agentTypes, options.backgroundAgents)
4014
+ : await dispatchSubAgent(input, taskOptions, agentTypes);
4015
+ const postTask = await runToolHooks(hooks.PostToolUse, 'PostToolUse', name, input, options.workDir, result);
4016
+ const injectedTask = [preTask.context, postTask.context].filter(Boolean).join('\n');
4017
+ if (injectedTask) {
4018
+ if (result.error !== undefined)
4019
+ result.error = `${result.error}\n\n[hook] ${injectedTask}`;
4020
+ else
4021
+ result.output = `${result.output ?? ''}\n\n[hook] ${injectedTask}`;
4022
+ }
4023
+ if (postTask.block) {
4024
+ result.error = `${result.error ? result.error + '\n' : ''}PostToolUse hook flagged: ${postTask.reason}`;
4025
+ }
4026
+ }
3401
4027
  }
3402
4028
  else {
3403
- const pre = runToolHooks(hooks.PreToolUse, 'PreToolUse', name, input, options.workDir);
4029
+ const pre = await runToolHooks(hooks.PreToolUse, 'PreToolUse', name, input, options.workDir);
3404
4030
  if (pre.block) {
3405
4031
  result = { error: `Blocked by PreToolUse hook: ${pre.reason}` };
3406
4032
  }
@@ -3412,27 +4038,32 @@ async function runAgentLoop(initialMessages, options) {
3412
4038
  // could not tell whose notes to write to (and must not be able to).
3413
4039
  if (name === exports.AGENT_MEMORY_TOOL) {
3414
4040
  result = await executeAgentMemoryWrite(input, options._agentMemory, options.workDir);
3415
- options.onToolResult(name, result);
4041
+ options.onToolResult(name, result, undefined, toolMeta);
3416
4042
  return { block: { ...block, id }, result };
3417
4043
  }
3418
4044
  // 1. Try platform-specific tools (e.g. VS Code semantic tools)
4045
+ // Both of these run code we do not control (the editor, a third-party MCP
4046
+ // server) and may ignore Stop; stopAwaiting returns an "interrupted" result
4047
+ // a few seconds after Stop instead of leaving the turn hanging on them.
3419
4048
  const external = options.executeExternalTool
3420
- ? await options.executeExternalTool(name, input)
4049
+ ? await stopAwaiting(options.executeExternalTool(name, input), options.abortSignal)
3421
4050
  : null;
3422
4051
  if (external !== null && external !== undefined) {
3423
4052
  result = external;
3424
4053
  // 2. Try MCP tools (serverName__toolName)
3425
4054
  }
3426
4055
  else if (options.mcpManager?.isMcpTool(name)) {
3427
- const mcpOutput = await options.mcpManager.callTool(name, input, options.abortSignal);
3428
- result = { output: mcpOutput ?? '' };
4056
+ const mcpOutput = await stopAwaiting(options.mcpManager.callTool(name, input, options.abortSignal), options.abortSignal);
4057
+ result = mcpOutput === STOP_TIMEOUT_RESULT
4058
+ ? STOP_TIMEOUT_RESULT
4059
+ : { output: (0, executor_1.capExternalOutput)(mcpOutput ?? '', `mcp-${name}`) };
3429
4060
  // 3. Fall through to built-in executor
3430
4061
  }
3431
4062
  else {
3432
4063
  // Snapshot pre-mutation state so the user can /rewind this turn.
3433
4064
  options.checkpointManager?.recordBeforeMutation(name, input);
3434
4065
  const onStream = options.onToolStreamChunk
3435
- ? (chunk) => options.onToolStreamChunk(name, chunk)
4066
+ ? (chunk) => options.onToolStreamChunk(name, chunk, undefined, toolMeta)
3436
4067
  : undefined;
3437
4068
  const run = () => (0, executor_1.executeTool)(name, input, options.abortSignal, sandboxCfg, options.workDir, agentScope, onStream, options.selfPeer);
3438
4069
  // Serialise anything that would otherwise race. Two sources of races:
@@ -3463,9 +4094,31 @@ async function runAgentLoop(initialMessages, options) {
3463
4094
  // Nesting cross-process OUTSIDE in-process means a process holding
3464
4095
  // the OS lock for a group of sub-agents takes it exactly ONCE for the
3465
4096
  // whole group, not once per sub-agent.
3466
- result = locks.length
3467
- ? await (0, crossProcessLock_1.withCrossProcessLocks)(locks, () => withFileLocks(locks, run), options.workDir)
3468
- : await run();
4097
+ // Unchanged re-read of a range whose earlier result is still in the
4098
+ // history → one-line stub instead of the same content again (readDedupe.ts).
4099
+ const dedupeStub = name === 'read_file' ? readDedupe.check(input, messages) : null;
4100
+ const prefetched = !dedupeStub && prefetchSafe ? prefetcher.take(id, name, input) : null;
4101
+ if (dedupeStub) {
4102
+ result = { output: dedupeStub };
4103
+ }
4104
+ else if (prefetched) {
4105
+ result = await prefetched;
4106
+ if (name === 'read_file' && result.error === undefined && result.output?.startsWith('[File:')) {
4107
+ readDedupe.record(input, id);
4108
+ }
4109
+ }
4110
+ else {
4111
+ result = locks.length
4112
+ ? await (0, crossProcessLock_1.withCrossProcessLocks)(locks, () => withFileLocks(locks, run), options.workDir)
4113
+ : await run();
4114
+ if (name === 'read_file' && result.error === undefined && result.output?.startsWith('[File:')) {
4115
+ readDedupe.record(input, id);
4116
+ }
4117
+ }
4118
+ // A write tool touched these paths — drop their recorded reads even if
4119
+ // mtime happens to be unchanged (same-millisecond edit).
4120
+ if (name !== 'bash' && locks.length)
4121
+ readDedupe.invalidate(locks);
3469
4122
  }
3470
4123
  }
3471
4124
  catch (err) {
@@ -3480,7 +4133,7 @@ async function runAgentLoop(initialMessages, options) {
3480
4133
  void maybeCompactMemory(scope, options);
3481
4134
  }
3482
4135
  // PostToolUse can inject context for the model or flag a problem.
3483
- const post = runToolHooks(hooks.PostToolUse, 'PostToolUse', name, input, options.workDir, result);
4136
+ const post = await runToolHooks(hooks.PostToolUse, 'PostToolUse', name, input, options.workDir, result);
3484
4137
  const injected = [pre.context, post.context].filter(Boolean).join('\n');
3485
4138
  if (injected) {
3486
4139
  if (result.error !== undefined)
@@ -3503,7 +4156,7 @@ async function runAgentLoop(initialMessages, options) {
3503
4156
  result.output = (0, testIntegrity_1.stripTestIntegrityMarker)(result.output);
3504
4157
  }
3505
4158
  // Notify caller about result
3506
- options.onToolResult(name, result);
4159
+ options.onToolResult(name, result, undefined, toolMeta);
3507
4160
  return { block: { ...block, id }, result, rawOutput };
3508
4161
  }));
3509
4162
  // Track whether files were mutated / verified this run, for the one-shot
@@ -3524,6 +4177,20 @@ async function runAgentLoop(initialMessages, options) {
3524
4177
  if (auditCtx) {
3525
4178
  (0, audit_1.recordAudit)(auditCtx, block.name, block.input, ok, result.output, result.error, result.exitCode);
3526
4179
  }
4180
+ // Build/test commands a sub-agent ran count for THIS run too — otherwise the main
4181
+ // agent, reporting "tests pass" from a sub-agent's verified run, was told no test
4182
+ // had been run and re-ran the whole suite. Counted even when the task errored
4183
+ // (a stopped sub-agent may still have run the suite).
4184
+ if (block.name === 'task' && result.childVerifications?.length) {
4185
+ for (const v of result.childVerifications) {
4186
+ ledger.verifications.push({ cmd: `(sub-agent) ${v.cmd}`.slice(0, 120), ok: v.ok, epoch: ledger.epoch });
4187
+ }
4188
+ const lastChild = result.childVerifications[result.childVerifications.length - 1];
4189
+ if (lastChild.ok) {
4190
+ ranVerificationCmd = true;
4191
+ filesMutatedSinceVerify = false;
4192
+ }
4193
+ }
3527
4194
  if (!ok)
3528
4195
  continue; // failed calls don't count either way
3529
4196
  if (exports.WRITE_TOOL_NAMES.has(block.name))
@@ -3575,6 +4242,12 @@ async function runAgentLoop(initialMessages, options) {
3575
4242
  options.onPeerMessage?.(inboundPeerMessages);
3576
4243
  toolResultContent.push({ type: 'text', text: renderPeerMessages(inboundPeerMessages) });
3577
4244
  }
4245
+ // Background agents that finished while this round ran — same boundary, same
4246
+ // reason: a notification must never interrupt an in-flight tool call.
4247
+ const bgFinished = drainBackgroundNotifications();
4248
+ if (bgFinished.length) {
4249
+ toolResultContent.push({ type: 'text', text: (0, backgroundAgents_1.formatBackgroundNotifications)(bgFinished) });
4250
+ }
3578
4251
  const toolResultMessage = {
3579
4252
  role: 'user',
3580
4253
  content: toolResultContent,
@@ -3653,8 +4326,13 @@ async function runAgentLoop(initialMessages, options) {
3653
4326
  // message in this file — these are statements from the harness, not from the model,
3654
4327
  // and splicing them into the assistant's own bubble reads as if it said them.
3655
4328
  const notice = stopReasonNotice(stopReason, { budget, repeatError: stalledRepeatError });
3656
- if (notice)
4329
+ // A sub-agent's stop belongs in its parent's tool result, not in the user's chat as
4330
+ // if the MAIN agent had stopped (runSubTask turns it into an error for the parent).
4331
+ if (options._onStopReason)
4332
+ options._onStopReason(stopReason, notice || null);
4333
+ else if (notice)
3657
4334
  (options.onNotice ?? options.onText)(notice);
4335
+ options._onVerifications?.(ledger.verifications.map((v) => ({ cmd: v.cmd, ok: v.ok })));
3658
4336
  }
3659
4337
  catch (err) {
3660
4338
  // Every other failure path (a tool executor blowing up, a hook throwing, an
@@ -3669,7 +4347,12 @@ async function runAgentLoop(initialMessages, options) {
3669
4347
  throw new types_1.AgentTurnError(err?.message || 'The turn ended unexpectedly', trimToResumableBoundary(messages), completedRounds, err);
3670
4348
  }
3671
4349
  finally {
3672
- runSimpleHooks(hooks.OnStop, options.workDir);
4350
+ // OnStop is the MAIN agent's turn ending. Firing it at the end of every sub-agent
4351
+ // ran the user's "turn finished" hook (notifications, formatters) mid-turn.
4352
+ if (depth === 0)
4353
+ await runSimpleHooks(hooks.OnStop, options.workDir);
4354
+ else
4355
+ await runSimpleHooks(hooks.SubagentStop, options.workDir, { NEXRALL_AGENT_NAME: options._agentTypeName ?? 'general-purpose' });
3673
4356
  }
3674
4357
  return messages;
3675
4358
  }