aegis-desktop 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,9 +11,20 @@
11
11
  * aegiscodex-dev's src/backend.js runProvider): every turn carries a real
12
12
  * system prompt (prompt.js) and the builtin tool schemas (tools.js), and when
13
13
  * a provider answers with tool calls the loop executes them in-process and
14
- * feeds the results back until the model answers with text or the round cap is
15
- * hit. The window stays contextIsolated + sandboxed: this module runs in the
16
- * MAIN process, so the executor never has to be exposed to the renderer.
14
+ * feeds the results back for as many rounds as the model keeps calling tools
15
+ * — there is no round cap; a turn ends when the model answers with text, or
16
+ * the user cancels it (cancel()/AbortController).
17
+ *
18
+ * Because a provider can signal "finished" with no answer attached, the two
19
+ * exit paths are guarded against the empty turn (see the loop's comment):
20
+ * a `finish_reason: 'length'` completion with no text is retried once with a
21
+ * doubled budget, and a turn that ends with neither text nor a tool call is
22
+ * re-dispatched once with `tools: []` plus a nudge so the model has to write
23
+ * up what it already gathered. A turn still empty after both throws instead
24
+ * of returning a blank completion for the renderer to paint "(empty
25
+ * response)" over. The window stays
26
+ * contextIsolated + sandboxed: this module runs in the MAIN process, so the
27
+ * executor never has to be exposed to the renderer.
17
28
  *
18
29
  * Two turn-scoped resources ride along with the loop, mirroring
19
30
  * aegiscodex-dev's runProvider exactly:
@@ -40,9 +51,6 @@ const { agentSystemPrompt, agentRoleLabel } = require('./agents.js');
40
51
  /** Classes whose transport is a user-supplied endpoint + credential. */
41
52
  const CUSTOM_CLASSES = Object.freeze(['openai-compat', 'anthropic']);
42
53
 
43
- /** Hard cap on tool rounds per turn — mirrors the CLI's bounded loop. */
44
- const MAX_TOOL_ROUNDS = 12;
45
-
46
54
  /**
47
55
  * Depth at which the task tool stops being offered. The main chat (depth 0)
48
56
  * and subagents down to depth MAX_SUBAGENT_DEPTH - 1 can all delegate, so
@@ -59,6 +67,53 @@ const CLASSES = [
59
67
  { class: 'anthropic', label: 'Anthropic-compatible', kind: 'custom' },
60
68
  ];
61
69
 
70
+ /**
71
+ * Mirrors aegiscodex-dev's src/backend.js DEEPSEEK_REASONING_MODEL_RE +
72
+ * EFFORT_TOKEN_BUDGET verbatim. DeepSeek's reasoning models (deepseek-flash,
73
+ * deepseek-v4-pro, the deprecated deepseek-reasoner, and the legacy
74
+ * v4-flash/v4.1-flash aliases some configs still carry) spend part of
75
+ * max_tokens on hidden chain-of-thought before ever emitting visible
76
+ * content — DeepSeek counts reasoning tokens against the same budget as
77
+ * content. At the renderer's 4k default (index.html's max-tokens select),
78
+ * any non-trivial question can burn the whole budget reasoning and finish
79
+ * with empty content: no error, no tool calls, just a turn that "completes"
80
+ * with nothing to show for it (the empty-response bug). A user pointing the
81
+ * Custom OpenAI-compatible class straight at DeepSeek's API hits exactly
82
+ * this, so the request floors to the same effort budget aegiscodex-dev uses
83
+ * for its own direct DeepSeek calls instead of shipping whatever the
84
+ * dropdown happens to have selected.
85
+ */
86
+ const DEEPSEEK_REASONING_MODEL_RE = /^deepseek-(v4(\.\d+)?-(flash|pro)|flash|pro|reasoner)$/;
87
+ const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
88
+
89
+ /**
90
+ * Idle-stream budget for a pooled brain call ("work autonomously"). The
91
+ * generic watchdog in vendor/aegis.js kills a stream that goes 60s without a
92
+ * byte — right default for one provider call answering, wrong for a worker
93
+ * fan-out: pool_brain yields a header chunk, then stays silent until the
94
+ * FIRST worker pass *returns*, and each worker is a full reasoning-model call
95
+ * at roughly 1/(workers+1) of the effort budget. At high effort, 3 workers,
96
+ * that is a multi-thousand-token reasoning pass per worker — easily past a
97
+ * minute. Timing out there aborts a perfectly healthy autonomous turn
98
+ * mid-flight, after the server has already run and billed every worker.
99
+ */
100
+ const AUTONOMOUS_IDLE_TIMEOUT_MS = 5 * 60_000;
101
+
102
+ /**
103
+ * Only ever raises a too-low budget for a DeepSeek reasoning model — never
104
+ * lowers whatever the caller (renderer dropdown, or "adaptive" ceiling)
105
+ * already asked for. Everything else (non-DeepSeek models, non-reasoning
106
+ * DeepSeek ids like deepseek-chat) passes through untouched. Effort defaults
107
+ * to 'high' since custom endpoints have no effort selector of their own
108
+ * (that UI is aegis-class/autonomous-only) — matching aegiscodex-dev's own
109
+ * default effort.
110
+ */
111
+ function deepseekReasoningFloor(model, maxTokens, effort) {
112
+ if (!DEEPSEEK_REASONING_MODEL_RE.test(String(model || ''))) return maxTokens;
113
+ const eff = effort === 'low' || effort === 'medium' ? effort : 'high';
114
+ return Math.max(Number(maxTokens) || 0, EFFORT_TOKEN_BUDGET[eff]);
115
+ }
116
+
62
117
  /** Relay model entries arrive as ids or objects; keep only real model ids. */
63
118
  function normalizeCatalog(models) {
64
119
  if (!Array.isArray(models)) return [];
@@ -131,11 +186,229 @@ function assistantText(res) {
131
186
  return (msg && typeof msg.content === 'string' && msg.content) || '';
132
187
  }
133
188
 
134
- function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBuilder, env }) {
189
+ /**
190
+ * The provider's stop reason for a completion ('' when it sent none). The two
191
+ * wire formats report it in different places and both have to be read: the
192
+ * OpenAI shape carries `choices[0].finish_reason` (this is what providers.js
193
+ * and vendor/aegis.js emit), while providers.js's Anthropic parser puts
194
+ * `stop_reason` on the result and never sets a finish_reason on the choice.
195
+ */
196
+ function finishReasonOf(res) {
197
+ const choice = res && res.choices && res.choices[0];
198
+ return (choice && choice.finish_reason) || (res && res.stop_reason) || '';
199
+ }
200
+
201
+ /**
202
+ * True when the provider cut the answer off at the token budget instead of
203
+ * the model choosing to stop. This is the diagnosis for the most common
204
+ * flavour of the empty turn: DeepSeek (and other reasoning models that bill
205
+ * hidden chain-of-thought against max_tokens) can spend the entire budget
206
+ * before emitting a single visible token, and the completion still arrives
207
+ * as a clean `finish_reason: 'length'` — no error, empty content. A tool
208
+ * call whose JSON was truncated mid-argument lands here too, where
209
+ * parseArgs() would otherwise silently yield `{}` and run the tool with no
210
+ * arguments, which is worse than retrying.
211
+ */
212
+ function isTruncated(res) {
213
+ const reason = finishReasonOf(res);
214
+ // 'length' is OpenAI's wording, 'max_tokens' is Anthropic's — same event.
215
+ return reason === 'length' || reason === 'max_tokens';
216
+ }
217
+
218
+ /**
219
+ * Double a budget for the one-shot truncation retry. A budget the caller set
220
+ * on purpose (the flow lane's deliberate 1024-token cap, say) is merely
221
+ * doubled — the floor only applies when nothing was set at all, so a retry
222
+ * can never silently override a small cap by an order of magnitude.
223
+ */
224
+ function doubledBudget(maxTokens) {
225
+ const n = Number(maxTokens) || 0;
226
+ return n > 0 ? n * 2 : 8192;
227
+ }
228
+
229
+ /**
230
+ * The follow-up shown to a model that ended its turn with neither text nor a
231
+ * tool call. Sent as a plain user message (never as a tool result — there is
232
+ * no pending tool call to answer) so every provider accepts it verbatim.
233
+ */
234
+ const EMPTY_TURN_NUDGE =
235
+ 'Your previous reply came back empty — it contained no answer and no tool call. ' +
236
+ 'Write your answer now, using only the information already gathered above. ' +
237
+ 'No tools are available in this reply, so do not call any: respond in plain ' +
238
+ 'prose or markdown.';
239
+
240
+ /**
241
+ * Last resort for a turn that is still empty after the synthesis pass.
242
+ *
243
+ * Throws rather than returning the blank completion. Returning it is what
244
+ * produces the renderer's undiagnosable "(empty response)" bubble, and
245
+ * synthesising fake assistant text instead would be worse: renderer/app.js
246
+ * persists whatever comes back as the assistant's own message and syncs it
247
+ * to the aegis account, so the notice would re-enter the model's context on
248
+ * the next turn as something it had said. Failing loudly leaves the turn's
249
+ * real tool log on screen, keeps the transcript honest, and names the cause
250
+ * (renderer prints `Error: <message>`).
251
+ */
252
+ function emptyTurnError({ cls, model, maxTokens, finishReason }) {
253
+ const err = new Error(
254
+ `The model returned no answer after ${cls}/${model} was asked to summarise its results ` +
255
+ `(stop reason: ${finishReason || 'none'}, max_tokens: ${maxTokens}). The token budget ` +
256
+ 'was most likely consumed before any visible text — raise the max-tokens setting, ' +
257
+ 'or lower effort.'
258
+ );
259
+ err.status = 502;
260
+ return err;
261
+ }
262
+
263
+ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBuilder, env, getConfirmMode }) {
135
264
  const controllers = new Map(); // sessionId -> AbortController
136
265
  const T = tools || toolsModule;
137
266
  const buildSystemPrompt = (promptBuilder && promptBuilder.buildSystemPrompt) || promptModule.buildSystemPrompt;
138
267
 
268
+ /** "Confirm before running tools" (Settings toggle, persisted by
269
+ * lib/settings.js. Because gatedExecuteTool is called for EVERY tool round
270
+ * it is read per call, not captured once at construction: flipping the
271
+ * switch takes effect on the next tool call, with no restart.
272
+ * An explicit `getConfirmMode` factory arg wins (used by tests); then the
273
+ * settings store's own accessor; then the safe default — ON, i.e. the gate
274
+ * stays up, so a store that predates the toggle can never silently
275
+ * disable it. */
276
+ const confirmModeEnabled = () => {
277
+ if (typeof getConfirmMode === 'function') return getConfirmMode() !== false;
278
+ if (settings && typeof settings.getConfirmMode === 'function') {
279
+ const value = settings.getConfirmMode();
280
+ return value === undefined ? true : Boolean(value);
281
+ }
282
+ return true;
283
+ };
284
+
285
+ // ── Tool-call approval gate (renderer confirms exec/writeFile/editFile
286
+ // before they run) ─────────────────────────────────────────────────────
287
+ //
288
+ // `sessionAllowlists` is keyed by the CONVERSATION's root session id (the
289
+ // one the renderer's `send()` mints once per thread and reuses across
290
+ // turns — see rootSessionId below), never by the per-call sessionId a
291
+ // subagent gets, so "allow for this session" reads the way the user sees
292
+ // it: one decision per open conversation, not per nested tool round.
293
+ // In-memory only, on purpose — never persisted, so a restart (or
294
+ // newChat()'s clearSessionApprovals) always starts from a clean gate.
295
+ const sessionAllowlists = new Map(); // rootSessionId -> Set<toolName>
296
+ const pendingApprovals = new Map(); // approvalId -> { resolve }
297
+
298
+ function sessionAllows(rootId, name) {
299
+ const set = sessionAllowlists.get(rootId);
300
+ return Boolean(set && set.has(name));
301
+ }
302
+
303
+ function allowForSession(rootId, name) {
304
+ if (!sessionAllowlists.has(rootId)) sessionAllowlists.set(rootId, new Set());
305
+ sessionAllowlists.get(rootId).add(name);
306
+ }
307
+
308
+ /** newChat() in the renderer calls this so a fresh conversation never
309
+ * inherits a prior thread's blanket allows. */
310
+ function clearSessionApprovals(rootSessionId) {
311
+ sessionAllowlists.delete(rootSessionId);
312
+ return { ok: true };
313
+ }
314
+
315
+ /** The renderer's approval card resolves the pending requestApproval()
316
+ * promise below. An unknown/already-answered id is a no-op — the card
317
+ * can only be clicked once (it disables itself), but a duplicate or
318
+ * late message must never throw. */
319
+ function respondApproval(approvalId, decision) {
320
+ const pending = pendingApprovals.get(approvalId);
321
+ if (!pending) return { ok: false };
322
+ pendingApprovals.delete(approvalId);
323
+ pending.resolve(decision === 'session' || decision === 'once' ? decision : 'deny');
324
+ return { ok: true };
325
+ }
326
+
327
+ /**
328
+ * Ask the renderer to approve one mutating tool call. Resolves 'once',
329
+ * 'session' or 'deny'. Sent over `rootOnDelta` (see chat()) as an
330
+ * `{ approval }` chunk so it rides the exact same streaming channel as
331
+ * tool-activity chunks — no new IPC surface needed on the push side, only
332
+ * on the reply side (respondApproval). Fails safe: no listener able to
333
+ * ever answer (no onDelta, or the turn was aborted) resolves 'deny'
334
+ * instead of hanging the tool round forever.
335
+ */
336
+ function requestApproval(rootSessionId, rootOnDelta, signal, info) {
337
+ return new Promise((resolve) => {
338
+ if (signal && signal.aborted) {
339
+ resolve('deny');
340
+ return;
341
+ }
342
+ const id = randomUUID();
343
+ let settled = false;
344
+ const onAbort = () => finish('deny');
345
+ const finish = (decision) => {
346
+ if (settled) return;
347
+ settled = true;
348
+ pendingApprovals.delete(id);
349
+ if (signal) signal.removeEventListener('abort', onAbort);
350
+ resolve(decision);
351
+ };
352
+ if (signal) signal.addEventListener('abort', onAbort, { once: true });
353
+ pendingApprovals.set(id, { resolve: finish });
354
+ if (typeof rootOnDelta !== 'function') {
355
+ finish('deny');
356
+ return;
357
+ }
358
+ rootOnDelta({
359
+ delta: '',
360
+ approval: {
361
+ id,
362
+ sessionId: rootSessionId,
363
+ tool: info.tool,
364
+ args: info.args,
365
+ diff: info.diff || null,
366
+ },
367
+ });
368
+ });
369
+ }
370
+
371
+ /**
372
+ * The gate itself: read-only tools and already-session-allowed mutating
373
+ * tools run exactly like executeTool always did. A first-time mutating
374
+ * call previews writeFile/editFile (a preview failure — e.g. old_string
375
+ * not found — is returned as the ordinary tool error, no approval prompt
376
+ * needed for a call that couldn't succeed anyway), asks the renderer, and
377
+ * on approval either applies with a fresh hash check (writeFile/editFile)
378
+ * or runs normally (exec — nothing to hash-check).
379
+ *
380
+ * Confirm mode off (Settings → "Confirm before running tools") short-circuits
381
+ * ALL of that: no preview, no approval card, no requestApproval() — the call
382
+ * runs straight through exactly like a session-allowed one, so "don't ask"
383
+ * is one switch rather than a per-tool blanket allow in every conversation.
384
+ */
385
+ async function gatedExecuteTool(call, { toolCtx, rootSessionId, rootOnDelta, signal }) {
386
+ const { name, args } = call;
387
+ if (!T.MUTATING_TOOLS.has(name)) return T.executeTool(name, args, toolCtx);
388
+ if (!confirmModeEnabled()) return T.executeTool(name, args, toolCtx);
389
+ if (sessionAllows(rootSessionId, name)) return T.executeTool(name, args, toolCtx);
390
+
391
+ let preview = null;
392
+ if (name === 'writeFile' || name === 'editFile') {
393
+ preview = T.previewMutation(name, args);
394
+ if (!preview.ok) return { ok: false, error: preview.error };
395
+ }
396
+
397
+ const decision = await requestApproval(rootSessionId, rootOnDelta, signal, {
398
+ tool: name,
399
+ args,
400
+ diff: preview && preview.diff,
401
+ });
402
+
403
+ if (decision === 'deny') {
404
+ return { ok: false, error: `${name} was not executed — the user denied the request.` };
405
+ }
406
+ if (decision === 'session') allowForSession(rootSessionId, name);
407
+
408
+ if (preview) return T.applyChecked(name, args, preview);
409
+ return T.executeTool(name, args, toolCtx);
410
+ }
411
+
139
412
  /**
140
413
  * Custom endpoints are only usable when they are actually configured:
141
414
  * a base URL is mandatory for both, and Anthropic additionally needs its own
@@ -225,6 +498,12 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
225
498
  maxTokens: opts.maxTokens,
226
499
  stream: true,
227
500
  onStream: opts.onDelta,
501
+ // Extended-reasoning trace (the fan-out's worker findings). Its own
502
+ // channel so it never counts as answer text — see vendor/aegis.js.
503
+ onReasoning: opts.onReasoning,
504
+ // A brain fan-out is silent between passes; give it room (see
505
+ // AUTONOMOUS_IDLE_TIMEOUT_MS). Undefined elsewhere -> 60s default.
506
+ idleTimeoutMs: opts.idleTimeoutMs,
228
507
  signal: opts.signal,
229
508
  // aegis_memory: automatic, no button — the server both reads prior
230
509
  // synced memory into context AND writes this turn back to it, the
@@ -281,7 +560,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
281
560
  async function chat(payload, onDelta) {
282
561
  const cls = payload && payload.class;
283
562
  const model = payload && payload.model;
284
- const maxTokens = payload && payload.maxTokens;
563
+ const maxTokens = deepseekReasoningFloor(model, payload && payload.maxTokens, payload && payload.effort);
285
564
  // "Work autonomously" — routes this call through aegis1's pool_brain
286
565
  // worker fan-out (services/pool_brain.py: N reasoning workers + a
287
566
  // synthesis pass) instead of a single provider call. UI-gated to the
@@ -292,6 +571,27 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
292
571
  // subagent spawned by depth N. Never set by an IPC caller — only by
293
572
  // runSubagent's own recursive chat() call below.
294
573
  const depth = Number.isInteger(payload && payload.depth) ? payload.depth : 0;
574
+ // The approval gate's identity for this whole conversation, regardless of
575
+ // depth: a real user turn defines it (defaults to its own sessionId); a
576
+ // subagent's nested chat() call always receives it explicitly from
577
+ // runSubagent below, so "allow for this session" means the same thing
578
+ // whether the call came from the top-level turn or three subagents deep.
579
+ const rootSessionId = (payload && payload.rootSessionId) || sessionId;
580
+ // Likewise, approval requests must always reach the ORIGINAL caller's
581
+ // stream — a subagent's own chat() call is invoked with a no-op onDelta
582
+ // (its tool activity/text is not streamed to the renderer), so without
583
+ // this a nested approval request would call that no-op and hang forever
584
+ // waiting for a response nobody can ever send.
585
+ const rootOnDelta = (payload && payload.rootOnDelta) || onDelta;
586
+
587
+ // A pooled brain call streams each worker's finding as extended reasoning
588
+ // before the synthesis pass writes the visible answer. Forward it on its
589
+ // own channel so the renderer can show the fan-out working instead of an
590
+ // apparently idle bubble for the whole worker phase. Uses `onDelta` (not
591
+ // rootOnDelta) on purpose: a subagent's reasoning should be suppressed
592
+ // exactly as its text already is.
593
+ const onReasoning =
594
+ typeof onDelta === 'function' ? (text) => text && onDelta({ reasoning: text }) : undefined;
295
595
 
296
596
  const controller = new AbortController();
297
597
  controllers.set(sessionId, controller);
@@ -335,10 +635,42 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
335
635
  cls, model, mode: payload && payload.mode, maxTokens, autonomous, sessionId, signal, onDelta, cfg, apiKey, toolChoice,
336
636
  effort: payload && payload.effort,
337
637
  workers: payload && payload.workers,
638
+ onReasoning,
639
+ idleTimeoutMs: autonomous ? AUTONOMOUS_IDLE_TIMEOUT_MS : undefined,
640
+ };
641
+
642
+ // No round cap: a model that keeps calling tools keeps going for as
643
+ // long as it wants to. The old fixed cap (12 rounds) cut off genuinely
644
+ // long research/exploration turns mid-investigation. A turn ends when
645
+ // the model answers with text, or the user cancels.
646
+ //
647
+ // Both of those exit paths need a guard, because a provider's
648
+ // "I'm finished" signal can arrive with no answer attached — which is
649
+ // exactly how the renderer came to paint "(empty response)" over a turn
650
+ // that had actually done real work:
651
+ // - truncated (finish_reason 'length'): the budget was consumed
652
+ // before any visible text (DeepSeek's hidden reasoning bills
653
+ // against max_tokens), or mid tool-call JSON. One doubled retry.
654
+ // - empty (no tool call AND no text): re-dispatch once with no tools
655
+ // and a nudge, so the model has to write up what it already found.
656
+ // Each guard fires at most once per turn, so a provider that is simply
657
+ // broken still terminates instead of looping.
658
+ let truncationRetried = false;
659
+ let synthesisDone = false;
660
+
661
+ // Round 1's shorthand prompt was sent as `prompt`, not as a message, so
662
+ // any follow-up dispatch in this turn must fold it into the history
663
+ // first or the model would be shown a nudge with no question above it.
664
+ const foldPromptIntoHistory = () => {
665
+ if (prompt === '') return;
666
+ const last = history[history.length - 1];
667
+ if (!(last && last.role === 'user' && last.content === prompt)) {
668
+ history.push({ role: 'user', content: prompt });
669
+ }
670
+ prompt = '';
338
671
  };
339
672
 
340
- let last = null;
341
- for (let round = 0; round <= MAX_TOOL_ROUNDS; round++) {
673
+ for (;;) {
342
674
  const opts = { ...base, system, messages: history, prompt, tools: toolSchemas };
343
675
  let res;
344
676
  try {
@@ -351,23 +683,48 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
351
683
  if (!retriable) throw e;
352
684
  res = await dispatch(cls, { ...opts, tools: [] });
353
685
  }
354
- last = res;
686
+
687
+ // Budget exhausted before the answer was written. Doubling it costs
688
+ // one request and converts a dead turn into a real one; a second
689
+ // 'length' result is accepted as-is so a hard-capped model can't
690
+ // spin here forever.
691
+ //
692
+ // Gated on empty text on purpose. Every transport builds `content`
693
+ // by concatenating the deltas it already forwarded to onDelta, so
694
+ // empty content means nothing was streamed and re-dispatching cannot
695
+ // double up in the renderer's live view. When text *has* arrived the
696
+ // turn is not empty — the answer is merely truncated — and a retry
697
+ // would stream it a second time onto the same bubble.
698
+ if (!truncationRetried && !assistantText(res) && isTruncated(res)) {
699
+ truncationRetried = true;
700
+ res = await dispatch(cls, { ...opts, maxTokens: doubledBudget(opts.maxTokens) });
701
+ }
355
702
 
356
703
  const calls = toolSchemas.length ? extractToolCalls(res) : [];
357
- if (!calls.length) return res;
358
- if (round === MAX_TOOL_ROUNDS) return res; // round cap: hand back what we have
359
-
360
- // Continuing the loop means round 1's shorthand prompt has to become
361
- // part of the history — it was sent as `prompt`, not as a message, so
362
- // without this the model would see a tool result and no question.
363
- if (prompt !== '') {
364
- const last = history[history.length - 1];
365
- if (!(last && last.role === 'user' && last.content === prompt)) {
366
- history.push({ role: 'user', content: prompt });
704
+ if (!calls.length) {
705
+ if (assistantText(res) || synthesisDone || !toolsEnabled) return res;
706
+ // The model stopped without calling a tool and without saying
707
+ // anything. Force the summary out of the context it already holds
708
+ // instead of handing the renderer a blank completion. Skipped when
709
+ // the caller opted out of the agent loop (`tools: false`): there is
710
+ // no gathered context to rescue, so a bare completion is just that.
711
+ synthesisDone = true;
712
+ foldPromptIntoHistory();
713
+ history.push({ role: 'user', content: EMPTY_TURN_NUDGE });
714
+ res = await dispatch(cls, { ...opts, messages: history, prompt: '', tools: [] });
715
+ if (!assistantText(res)) {
716
+ throw emptyTurnError({
717
+ cls,
718
+ model,
719
+ maxTokens: opts.maxTokens,
720
+ finishReason: finishReasonOf(res),
721
+ });
367
722
  }
368
- prompt = '';
723
+ return res;
369
724
  }
370
725
 
726
+ foldPromptIntoHistory();
727
+
371
728
  // Thread the assistant turn (its tool_calls) and each result back in
372
729
  // the shapes both wire formats accept (providers.js normalises them).
373
730
  history.push({
@@ -382,8 +739,10 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
382
739
 
383
740
  for (const call of calls) {
384
741
  const result = call.name === T.SUBAGENT_TOOL
385
- ? await runSubagent(call.args, { cls, model, maxTokens, mode: payload && payload.mode, parentSignal: signal, depth })
386
- : await T.executeTool(call.name, call.args, toolCtx);
742
+ ? await runSubagent(call.args, {
743
+ cls, model, maxTokens, mode: payload && payload.mode, parentSignal: signal, depth, rootSessionId, rootOnDelta,
744
+ })
745
+ : await gatedExecuteTool(call, { toolCtx, rootSessionId, rootOnDelta, signal });
387
746
  if (onDelta) onDelta({ delta: '', tool: { name: call.name, args: call.args, ok: result.ok } });
388
747
  history.push({
389
748
  role: 'tool',
@@ -393,7 +752,6 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
393
752
  });
394
753
  }
395
754
  }
396
- return last;
397
755
  } finally {
398
756
  if (shell) shell.dispose();
399
757
  controllers.delete(sessionId);
@@ -408,7 +766,10 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
408
766
  * throws — resolves { ok, output } or { ok:false, error }, matching
409
767
  * tools.js's executor contract so the caller treats it identically.
410
768
  */
411
- async function runSubagent({ description, subagent_type, prompt: subPrompt } = {}, { cls, model, maxTokens, mode, parentSignal, depth } = {}) {
769
+ async function runSubagent(
770
+ { description, subagent_type, prompt: subPrompt } = {},
771
+ { cls, model, maxTokens, mode, parentSignal, depth, rootSessionId, rootOnDelta } = {}
772
+ ) {
412
773
  const task = String(subPrompt || description || '').trim();
413
774
  if (!task) return { ok: false, error: 'task requires a prompt' };
414
775
  const label = subagent_type && subagent_type !== 'general' ? agentRoleLabel(subagent_type) : 'general';
@@ -425,8 +786,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
425
786
  }
426
787
 
427
788
  try {
789
+ // rootSessionId/rootOnDelta ride along explicitly (see chat()) so the
790
+ // subagent's own mutating tool calls still gate through the SAME
791
+ // approval card the user sees for the top-level turn, instead of
792
+ // silently hanging behind this call's no-op onDelta below.
428
793
  const res = await chat(
429
- { class: cls, model, maxTokens, mode, system, prompt: task, sessionId: subSessionId, depth: (depth || 0) + 1 },
794
+ {
795
+ class: cls, model, maxTokens, mode, system, prompt: task, sessionId: subSessionId, depth: (depth || 0) + 1,
796
+ rootSessionId, rootOnDelta,
797
+ },
430
798
  () => {}
431
799
  );
432
800
  const text = assistantText(res);
@@ -448,13 +816,14 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
448
816
 
449
817
  return {
450
818
  CLASSES,
451
- MAX_TOOL_ROUNDS,
452
819
  listClasses,
453
820
  listModels,
454
821
  chat,
455
822
  cancel,
823
+ respondApproval,
824
+ clearSessionApprovals,
456
825
  settings,
457
826
  };
458
827
  }
459
828
 
460
- module.exports = { CLASSES, MAX_TOOL_ROUNDS, createLocalEngine, extractToolCalls, parseArgs };
829
+ module.exports = { CLASSES, createLocalEngine, extractToolCalls, parseArgs };