aegis-desktop 0.5.1 → 0.5.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -47,6 +47,28 @@ const toolsModule = require('./tools.js');
47
47
  const promptModule = require('./prompt.js');
48
48
  const { ShellSession } = require('./shell.js');
49
49
  const { agentSystemPrompt, agentRoleLabel } = require('./agents.js');
50
+ // Cooperative working-tree sharing (see each module's header). The lock
51
+ // serialises two hosts that start a turn on one checkout close together; the
52
+ // guard is the load-bearing half — it refuses a tree-wide git operation while
53
+ // another session's uncommitted work is present, which is the operation that
54
+ // destroyed a peer's file for real. Both live beside the engine because the
55
+ // CLI vendors this directory wholesale (cli/scripts/predist.mjs), so the GUI
56
+ // and the terminal get one implementation rather than two that drift.
57
+ const { beginTurnGuard, recordWrite, blocksDestructive } = require('./turn-guard.js');
58
+ const { acquireWorktreeLock, releaseWorktreeLock } = require('./worktree-lock.js');
59
+
60
+ /**
61
+ * How long a turn waits for the working-tree lock before running anyway.
62
+ *
63
+ * Short on purpose. This engine serves interactive hosts — a desktop window
64
+ * and a terminal — where stalling for the lock's own 11-minute default would
65
+ * be a worse failure than contending: the user is watching a spinner. So the
66
+ * lock is best-effort mutual exclusion for turns that start near-simultaneously
67
+ * (the common overlap), and the turn guard covers everything after that. A
68
+ * non-interactive caller that can afford to queue should pass
69
+ * `payload.worktreeWaitMs` with the long DEFAULT_WORKTREE_WAIT_MS instead.
70
+ */
71
+ const WORKTREE_LOCK_WAIT_MS = 1500;
50
72
 
51
73
  /** Classes whose transport is a user-supplied endpoint + credential. */
52
74
  const CUSTOM_CLASSES = Object.freeze(['openai-compat', 'anthropic']);
@@ -70,22 +92,30 @@ const CLASSES = [
70
92
  /**
71
93
  * Mirrors aegiscodex-dev's src/backend.js DEEPSEEK_REASONING_MODEL_RE +
72
94
  * EFFORT_TOKEN_BUDGET verbatim. DeepSeek's reasoning models (deepseek-flash,
73
- * deepseek-v4-pro, the deprecated deepseek-reasoner, and the legacy
74
- * v4-flash/v4.1-flash aliases some configs still carry) spend part of
75
- * max_tokens on hidden chain-of-thought before ever emitting visible
76
- * content — DeepSeek counts reasoning tokens against the same budget as
77
- * content. At the renderer's 4k default (index.html's max-tokens select),
78
- * any non-trivial question can burn the whole budget reasoning and finish
79
- * with empty content: no error, no tool calls, just a turn that "completes"
80
- * with nothing to show for it (the empty-response bug). A user pointing the
81
- * Custom OpenAI-compatible class straight at DeepSeek's API hits exactly
82
- * this, so the request floors to the same effort budget aegiscodex-dev uses
83
- * for its own direct DeepSeek calls instead of shipping whatever the
84
- * dropdown happens to have selected.
95
+ * i.e. "Flash 4.1", deepseek-v4-pro, the deprecated deepseek-reasoner, and the
96
+ * legacy v4-flash/v4.1-flash aliases some configs still carry) spend part of
97
+ * the budget on hidden chain-of-thought before ever emitting visible content —
98
+ * DeepSeek counts reasoning tokens against the same budget as content. Under
99
+ * the renderer's removed 4k dropdown default, any non-trivial question could
100
+ * burn the whole budget reasoning and finish with empty content: no error, no
101
+ * tool calls, just a turn that "completes" with nothing to show for it (the
102
+ * empty-response bug). A user pointing the Custom OpenAI-compatible class
103
+ * straight at DeepSeek's API hits exactly this, so the request is sized by the
104
+ * same effort budget aegiscodex-dev uses for its own direct DeepSeek calls.
105
+ *
106
+ * desktop/renderer/budget.js carries the renderer's copy of these two
107
+ * constants plus budgetFor() below; test/budget.test.mjs requires both and
108
+ * asserts they agree, so the mirror cannot drift silently.
85
109
  */
86
110
  const DEEPSEEK_REASONING_MODEL_RE = /^deepseek-(v4(\.\d+)?-(flash|pro)|flash|pro|reasoner)$/;
87
111
  const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
88
112
 
113
+ /** The one class whose wire format REQUIRES a stated `max_tokens`: Anthropic's
114
+ * Messages API 400s without it, so that field is derived from the effort rung
115
+ * rather than invented by the transport (which is what a blanket
116
+ * `max_tokens: maxTokens || 4096` did — see providers.anthropicMessages). */
117
+ const REQUIRES_STATED_BUDGET = new Set(['anthropic']);
118
+
89
119
  /**
90
120
  * Idle-stream budget for a pooled brain call ("work autonomously"). The
91
121
  * generic watchdog in vendor/aegis.js kills a stream that goes 60s without a
@@ -109,18 +139,38 @@ const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
109
139
  const AUTONOMOUS_IDLE_TIMEOUT_MS = 15 * 60_000;
110
140
 
111
141
  /**
112
- * Only ever raises a too-low budget for a DeepSeek reasoning model — never
113
- * lowers whatever the caller (renderer dropdown, or "adaptive" ceiling)
114
- * already asked for. Everything else (non-DeepSeek models, non-reasoning
115
- * DeepSeek ids like deepseek-chat) passes through untouched. Effort defaults
116
- * to 'high' since custom endpoints have no effort selector of their own
117
- * (that UI is aegis-class/autonomous-only) — matching aegiscodex-dev's own
118
- * default effort.
142
+ * The budget a request travels with, resolved from EXACTLY ONE authority per
143
+ * call — the engine-side half of the rule desktop/renderer/budget.js mirrors
144
+ * and test/budget.test.mjs compares the two halves of, so neither can drift.
145
+ *
146
+ * 1. a caller-stated number IS the budget and is returned verbatim. It is a
147
+ * deliberate liability ceiling (aegis1 pass_budgets honours it downward:
148
+ * total = min(ladder, max_tokens x passes)) and no rung may raise it — the
149
+ * old `Math.max(stated, EFFORT_TOKEN_BUDGET[eff])` form did exactly that,
150
+ * so a caller asking for 1024 silently ran on 32768.
151
+ * 2. with nothing stated, a model that reasons against its own output budget
152
+ * (DeepSeek bills hidden chain-of-thought against the SAME budget as the
153
+ * answer) or a class whose wire format REQUIRES the field (Anthropic's
154
+ * Messages API) gets the Effort rung. This is why the renderer's
155
+ * max-tokens dropdown was removed rather than fixed: at its 4k default a
156
+ * reasoning model spent the entire budget thinking and finished empty —
157
+ * no error, no tool call, just a "completed" turn with nothing in it.
158
+ * 3. otherwise `undefined` — no `max_tokens` goes on the wire and the
159
+ * provider's own output limit governs. The transport used to fill this
160
+ * gap with an invented 4096 default, which is the defect in (2).
161
+ *
162
+ * Truncated and empty turns are still handled where they belong — the
163
+ * doubled-budget retry plus emptyTurnError — rather than by inflating the
164
+ * caller's ceiling up front.
119
165
  */
120
- function deepseekReasoningFloor(model, maxTokens, effort) {
121
- if (!DEEPSEEK_REASONING_MODEL_RE.test(String(model || ''))) return maxTokens;
122
- const eff = effort === 'low' || effort === 'medium' ? effort : 'high';
123
- return Math.max(Number(maxTokens) || 0, EFFORT_TOKEN_BUDGET[eff]);
166
+ function reasoningBudget(cls, model, maxTokens, effort) {
167
+ const stated = Number(maxTokens);
168
+ if (Number.isFinite(stated) && stated > 0) return stated;
169
+ if (DEEPSEEK_REASONING_MODEL_RE.test(String(model || '')) || REQUIRES_STATED_BUDGET.has(cls)) {
170
+ const eff = effort === 'low' || effort === 'medium' ? effort : 'high';
171
+ return EFFORT_TOKEN_BUDGET[eff];
172
+ }
173
+ return undefined;
124
174
  }
125
175
 
126
176
  /** Relay model entries arrive as ids or objects; keep only real model ids. */
@@ -259,6 +309,33 @@ function doubledBudget(maxTokens) {
259
309
  return n > 0 ? n * 2 : 8192;
260
310
  }
261
311
 
312
+ /**
313
+ * The cap the CALLER stated, or `undefined` meaning "none was stated".
314
+ *
315
+ * The two branches resolve one question — whose number is this? — and the
316
+ * first is the fix. An internal re-entry (runSubagent's nested chat()) passes
317
+ * `statedMaxTokens` explicitly, because the `maxTokens` it holds is the rung
318
+ * reasoningBudget() derived from `effort`: a number the server is about to
319
+ * derive for itself. Re-reading that value as the caller's own is what put a
320
+ * guessed 32768 on the wire as a *ceiling over* aegis1's ladder
321
+ * (pass_budgets: total = min(ladder, max_tokens x passes)) — so a
322
+ * high-effort subagent silently ran at 8192 and the ladder the UI advertised
323
+ * was not the budget the call used.
324
+ *
325
+ * A non-numeric or non-positive value means "nothing stated" — never a 0- or
326
+ * NaN-token ceiling.
327
+ */
328
+ function statedCapOf(payload) {
329
+ const asCap = (v) => {
330
+ const n = Number(v);
331
+ return Number.isFinite(n) && n > 0 ? n : undefined;
332
+ };
333
+ if (payload && Object.prototype.hasOwnProperty.call(payload, 'statedMaxTokens')) {
334
+ return asCap(payload.statedMaxTokens);
335
+ }
336
+ return asCap(payload && payload.maxTokens);
337
+ }
338
+
262
339
  /**
263
340
  * The follow-up shown to a model that ended its turn with neither text nor a
264
341
  * tool call. Sent as a plain user message (never as a tool result — there is
@@ -285,9 +362,9 @@ const EMPTY_TURN_NUDGE =
285
362
  function emptyTurnError({ cls, model, maxTokens, finishReason }) {
286
363
  const err = new Error(
287
364
  `The model returned no answer after ${cls}/${model} was asked to summarise its results ` +
288
- `(stop reason: ${finishReason || 'none'}, max_tokens: ${maxTokens}). The token budget ` +
289
- 'was most likely consumed before any visible text — raise the max-tokens setting, ' +
290
- 'or lower effort.'
365
+ `(stop reason: ${finishReason || 'none'}, max_tokens: ${maxTokens || 'unstated'}). The token budget ` +
366
+ 'was most likely consumed before any visible text — raise the Effort rung, which is ' +
367
+ 'what sizes this call, or ask for a smaller piece of work.'
291
368
  );
292
369
  err.status = 502;
293
370
  return err;
@@ -326,6 +403,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
326
403
  // In-memory only, on purpose — never persisted, so a restart (or
327
404
  // newChat()'s clearSessionApprovals) always starts from a clean gate.
328
405
  const sessionAllowlists = new Map(); // rootSessionId -> Set<toolName>
406
+ // Denials are remembered for the same reason allows are, mirroring it for
407
+ // the other answer. Without this, a "Deny" was forgotten the instant it was
408
+ // given: the model read `… the user denied the request`, re-planned, called
409
+ // the SAME tool again, and the gate raised a SECOND card — so one "no" cost
410
+ // the user a prompt per round for up to maxRounds (24 chat / 40 autonomous)
411
+ // rounds, each round re-sending the whole conversation to the provider. A
412
+ // gate that only remembers "yes" turns a single click into a retry storm;
413
+ // remembering "no" makes the first answer stick.
414
+ const sessionDenials = new Map(); // rootSessionId -> Set<toolName>
329
415
  const pendingApprovals = new Map(); // approvalId -> { resolve }
330
416
 
331
417
  function sessionAllows(rootId, name) {
@@ -338,10 +424,44 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
338
424
  sessionAllowlists.get(rootId).add(name);
339
425
  }
340
426
 
427
+ /** Has this conversation already refused this tool? Checked before the card
428
+ * is raised, so a repeat call is refused outright instead of re-prompting. */
429
+ function sessionDenies(rootId, name) {
430
+ const set = sessionDenials.get(rootId);
431
+ return Boolean(set && set.has(name));
432
+ }
433
+
434
+ function denyForSession(rootId, name) {
435
+ if (!sessionDenials.has(rootId)) sessionDenials.set(rootId, new Set());
436
+ sessionDenials.get(rootId).add(name);
437
+ }
438
+
439
+ /**
440
+ * The text a refused tool hands back to the model. The old one-line
441
+ * `"<tool> was not executed — the user denied the request."` read to a
442
+ * capable agent as a transient failure to route around: it would apologise,
443
+ * pick a different command that does the same thing, and call the gate
444
+ * again. This says the durable part out loud (the refusal covers the rest of
445
+ * the conversation, not just that call) and asks for the one response that
446
+ * actually helps — say what you need and stop, so the user can re-enable it.
447
+ */
448
+ function denialText(name) {
449
+ return (
450
+ `${name} was not executed — the user denied this tool for this conversation. ` +
451
+ `Do NOT retry it and do NOT attempt the same effect by another route ` +
452
+ `(another command, a writeFile instead of an edit, a subagent). ` +
453
+ `Stop calling tools and reply in plain text: say what you were trying to do, ` +
454
+ `what you need, and that the user can re-enable ${name} to let it proceed.`
455
+ );
456
+ }
457
+
341
458
  /** newChat() in the renderer calls this so a fresh conversation never
342
- * inherits a prior thread's blanket allows. */
459
+ * inherits a prior thread's blanket allows — or its refusals. A new chat is
460
+ * a new gate in both directions: leaving denials behind would silently
461
+ * refuse a tool in a thread where the user never said no. */
343
462
  function clearSessionApprovals(rootSessionId) {
344
463
  sessionAllowlists.delete(rootSessionId);
464
+ sessionDenials.delete(rootSessionId);
345
465
  return { ok: true };
346
466
  }
347
467
 
@@ -359,22 +479,28 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
359
479
 
360
480
  /**
361
481
  * Ask the renderer to approve one mutating tool call. Resolves 'once',
362
- * 'session' or 'deny'. Sent over `rootOnDelta` (see chat()) as an
363
- * `{ approval }` chunk so it rides the exact same streaming channel as
364
- * tool-activity chunks — no new IPC surface needed on the push side, only
365
- * on the reply side (respondApproval). Fails safe: no listener able to
366
- * ever answer (no onDelta, or the turn was aborted) resolves 'deny'
367
- * instead of hanging the tool round forever.
482
+ * 'session' or 'deny' (an explicit choice by the user) — or 'cancel' when
483
+ * nobody ever answered: no listener able to reply, or the turn was aborted.
484
+ * 'cancel' is kept apart from 'deny' on purpose. Both refuse the call, but
485
+ * only 'deny' is a decision the user made, so only 'deny' may be remembered
486
+ * as "this conversation said no" (see sessionDenials). Folding the two
487
+ * together — which is what the old fail-safe did, resolving 'deny' for an
488
+ * abort — would let a cancelled turn permanently refuse a tool the user
489
+ * never ruled on. Sent over `rootOnDelta` (see chat()) as an `{ approval }`
490
+ * chunk so it rides the exact same streaming channel as tool-activity
491
+ * chunks — no new IPC surface needed on the push side, only on the reply
492
+ * side (respondApproval). Fails safe: an unanswered request refuses the
493
+ * call instead of hanging the tool round forever.
368
494
  */
369
495
  function requestApproval(rootSessionId, rootOnDelta, signal, info) {
370
496
  return new Promise((resolve) => {
371
497
  if (signal && signal.aborted) {
372
- resolve('deny');
498
+ resolve('cancel');
373
499
  return;
374
500
  }
375
501
  const id = randomUUID();
376
502
  let settled = false;
377
- const onAbort = () => finish('deny');
503
+ const onAbort = () => finish('cancel');
378
504
  const finish = (decision) => {
379
505
  if (settled) return;
380
506
  settled = true;
@@ -385,7 +511,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
385
511
  if (signal) signal.addEventListener('abort', onAbort, { once: true });
386
512
  pendingApprovals.set(id, { resolve: finish });
387
513
  if (typeof rootOnDelta !== 'function') {
388
- finish('deny');
514
+ finish('cancel');
389
515
  return;
390
516
  }
391
517
  rootOnDelta({
@@ -415,11 +541,51 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
415
541
  * runs straight through exactly like a session-allowed one, so "don't ask"
416
542
  * is one switch rather than a per-tool blanket allow in every conversation.
417
543
  */
544
+ /**
545
+ * The single place a tool executor is called from the chat loop, so the
546
+ * turn guard cannot be bypassed by reaching a different branch (there are
547
+ * four `return`s in gatedExecuteTool below; all of them come through here).
548
+ *
549
+ * exec is *checked* before running and writeFile/editFile are *recorded*
550
+ * after, and the asymmetry is the point: a refusal has to happen before the
551
+ * command runs, while "this path is mine" is only true once the write
552
+ * actually happened — recording a denied or preview-failed write would hide
553
+ * a genuinely foreign file behind a claim of ownership.
554
+ *
555
+ * A refusal is returned as an ordinary tool error, so the model reads it and
556
+ * adapts (name the paths, or don't do it) with no new IPC or renderer
557
+ * channel — the same reason it works identically in the GUI and the CLI.
558
+ */
559
+ async function guardedExecute(name, args, toolCtx) {
560
+ const guard = toolCtx && toolCtx.guard;
561
+ if (guard && name === 'exec') {
562
+ const verdict = blocksDestructive(guard, args && args.command);
563
+ if (!verdict.ok) return { ok: false, error: verdict.reason };
564
+ }
565
+ // Awaited, not merely returned: "this path is mine" is only true once the
566
+ // write actually landed, and `executeTool` is async — inspecting `.ok` on
567
+ // an un-awaited promise would read undefined and record every write,
568
+ // including the ones that failed. Ownership is load-bearing in the other
569
+ // direction too: a path wrongly claimed stops being reported as foreign.
570
+ const result = await T.executeTool(name, args, toolCtx);
571
+ if (guard && (name === 'writeFile' || name === 'editFile') && result && result.ok !== false) {
572
+ recordWrite(guard, args && args.path);
573
+ }
574
+ return result;
575
+ }
576
+
418
577
  async function gatedExecuteTool(call, { toolCtx, rootSessionId, rootOnDelta, signal }) {
419
578
  const { name, args } = call;
420
- if (!T.MUTATING_TOOLS.has(name)) return T.executeTool(name, args, toolCtx);
421
- if (!confirmModeEnabled()) return T.executeTool(name, args, toolCtx);
422
- if (sessionAllows(rootSessionId, name)) return T.executeTool(name, args, toolCtx);
579
+ if (!T.MUTATING_TOOLS.has(name)) return guardedExecute(name, args, toolCtx);
580
+ if (!confirmModeEnabled()) return guardedExecute(name, args, toolCtx);
581
+ if (sessionAllows(rootSessionId, name)) return guardedExecute(name, args, toolCtx);
582
+ // Already refused in this conversation: refuse again WITHOUT raising a
583
+ // second card. Before this, the model's retry after a denial re-prompted
584
+ // the user for the same tool — one "no" produced a card per round for up
585
+ // to maxRounds rounds, each one a billed provider call re-sending the
586
+ // whole conversation. Checked after the confirm-mode short-circuit so
587
+ // turning the gate off still overrides an earlier refusal.
588
+ if (sessionDenies(rootSessionId, name)) return { ok: false, error: denialText(name) };
423
589
 
424
590
  let preview = null;
425
591
  if (name === 'writeFile' || name === 'editFile') {
@@ -433,13 +599,30 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
433
599
  diff: preview && preview.diff,
434
600
  });
435
601
 
602
+ // Only a click is remembered. 'cancel' (aborted turn / nobody able to
603
+ // answer) refuses this call but must not write a durable "no" the user
604
+ // never gave.
436
605
  if (decision === 'deny') {
437
- return { ok: false, error: `${name} was not executed — the user denied the request.` };
606
+ denyForSession(rootSessionId, name);
607
+ return { ok: false, error: denialText(name) };
608
+ }
609
+ if (decision !== 'session' && decision !== 'once') {
610
+ return { ok: false, error: `${name} was not executed — the request was cancelled.` };
438
611
  }
439
612
  if (decision === 'session') allowForSession(rootSessionId, name);
440
613
 
441
- if (preview) return T.applyChecked(name, args, preview);
442
- return T.executeTool(name, args, toolCtx);
614
+ if (preview) {
615
+ const res = T.applyChecked(name, args, preview);
616
+ // Recorded on the same condition as guardedExecute's path: only a write
617
+ // that landed is this session's. (The previous form returned here
618
+ // unconditionally, making every line below it unreachable — the
619
+ // approved-write path recorded nothing at all.)
620
+ if (res && res.ok !== false && toolCtx && toolCtx.guard) {
621
+ recordWrite(toolCtx.guard, args && args.path);
622
+ }
623
+ return res;
624
+ }
625
+ return guardedExecute(name, args, toolCtx);
443
626
  }
444
627
 
445
628
  /**
@@ -545,7 +728,25 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
545
728
  messages: opts.messages,
546
729
  model: opts.model,
547
730
  mode: opts.mode,
548
- maxTokens: opts.maxTokens,
731
+ // Only a cap the caller STATED travels; an effort-derived one does not.
732
+ // The server derives its own budget from `effort` (aegis1 pass_budgets
733
+ // splits its ladder across workers + synthesis), so forwarding the
734
+ // derived number states one decision twice — and the copies had already
735
+ // drifted: 974adc5 doubled aegis1's ladder while this side stood still,
736
+ // leaving the client capping below the budget it displayed. The cap was
737
+ // never the one the renderer showed either (the Max tokens dropdown
738
+ // that used to claim a figure for the pooled class is gone — see
739
+ // desktop/renderer/budget.js). Omitting the field is
740
+ // what tells aegis1 "no cap stated — let effort decide", the same
741
+ // contract aegiscodex-dev sends.
742
+ //
743
+ // A cap the caller DID state is a different thing, and dropping it was
744
+ // a bug: aegis1 reads a body max_tokens as a ceiling over its ladder,
745
+ // so omitting it does not bound the call — it grants the full top rung
746
+ // instead. A deliberate 4096 would have run at 32768, which is the same
747
+ // "only ever raise the caller's ceiling" failure the old
748
+ // Math.max(Number(maxTokens) || 0, EFFORT_TOKEN_BUDGET[eff]) had.
749
+ maxTokens: opts.statedMaxTokens,
549
750
  stream: opts.stream !== false,
550
751
  // The pooled (Nexus) brain is streamed, and an OpenAI-compatible SSE
551
752
  // stream reports no token usage unless asked. Without this the Aegis
@@ -561,12 +762,53 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
561
762
  // AUTONOMOUS_IDLE_TIMEOUT_MS). Undefined elsewhere -> 60s default.
562
763
  idleTimeoutMs: opts.idleTimeoutMs,
563
764
  signal: opts.signal,
564
- // aegis_memory: automatic, no button — the server both reads prior
565
- // synced memory into context AND writes this turn back to it, the
566
- // same flag aegis-online sets. Matches aegiscodex-dev's own
567
- // cross-session memory (auto-indexed, no manual tagging).
765
+ // aegis_recall — the read half of aegis_memory, and the only half a
766
+ // client on this engine should send.
767
+ //
768
+ // aegis_memory is both halves: it injects the account's synced memory
769
+ // into context AND persists this turn back into it, charging sync
770
+ // quota for the write. That pairing is right for aegis-online, whose
771
+ // chat has nowhere else to live. It is wrong here, because this engine
772
+ // is shared by the GUI and the CLI (cli/src/deps.js loads this file):
773
+ // either would be billed for every "hey" and would fill the user's
774
+ // memory with greetings. Verified against aegis1 app.py: every
775
+ // write-back site (the note/upsert calls) is gated on aegis_memory,
776
+ // and aegis_memory implies aegis_recall there — so dropping the write
777
+ // half leaves /online unchanged and costs nothing on the read side.
778
+ //
779
+ // Writing is still available, explicitly: the aegis_memory_save tool
780
+ // (mcp/tools.js; the CLI exposes it as /memory). Recall on every turn,
781
+ // store only what the user asks for — which is what aegiscodex-dev's
782
+ // own cross-session memory does, and what this comment claimed to
783
+ // match while sending both halves.
784
+ //
785
+ // Recall was previously unreachable for a client that would not pay
786
+ // for it: services/tiered_recall.py only ran behind a flag that also
787
+ // bought a write, so the tiered path existed with no caller able to
788
+ // afford it. The split is what makes it reachable. The DEEP tier of the
789
+ // same read is a third flag with its own price — see `opts.recallDeep`
790
+ // below, which is off unless the session opted in.
568
791
  extra: {
569
- aegis_memory: true,
792
+ aegis_recall: true,
793
+ // The DEEP tier of that read — brain corrections plus the semantic
794
+ // answer cache — is not the same price, so it does not ride along.
795
+ // aegis1 app.py:8367 reads `aegis_recall_deep` (or the
796
+ // X-AEGIS-Recall-Deep header) and services/brain_memory.py
797
+ // find_cached_answer embeds the query: one provider embedding per
798
+ // turn, metered. The server deliberately implies it from
799
+ // `aegis_memory` and NOT from `aegis_recall`, so that a terminal
800
+ // client can buy the cheap read without the embedding.
801
+ //
802
+ // This client is that terminal client (the CLI loads this file via
803
+ // cli/src/deps.js), so it must not opt itself in: the flag is sent
804
+ // only when the SESSION asked for it — CLI `/memory-deep on`, a
805
+ // desktop payload with `recallDeep: true` — and it defaults false
806
+ // everywhere. It also travels only on the user's own turn
807
+ // (`opts.recallDeep` is cleared for every other dispatch below): a
808
+ // tool round, the doubled-budget retry and the write-up re-dispatch
809
+ // all re-send a context whose embedding the first round already
810
+ // bought, which would turn one embedding per turn into one per round.
811
+ ...(opts.recallDeep ? { aegis_recall_deep: true } : {}),
570
812
  session: opts.sessionId,
571
813
  // The fan-out is opt-in per dispatch. `brain` is sent EXPLICITLY
572
814
  // whenever this dispatch is not the autonomous one, because the
@@ -638,7 +880,19 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
638
880
  async function chat(payload, onDelta) {
639
881
  const cls = payload && payload.class;
640
882
  const model = payload && payload.model;
641
- const maxTokens = deepseekReasoningFloor(model, payload && payload.maxTokens, payload && payload.effort);
883
+ const maxTokens = reasoningBudget(cls, model, payload && payload.maxTokens, payload && payload.effort);
884
+ // The caller's OWN number, kept apart from `maxTokens` above. That one
885
+ // collapses two different facts into a single value — "the caller stated
886
+ // 4096" and "effort implies 32768" — and the pooled path must treat them
887
+ // differently. A stated cap is a liability ceiling the server honours
888
+ // downward (aegis1 pass_budgets: total = min(ladder, max_tokens x passes));
889
+ // an effort-derived one is the server's own arithmetic stated twice, and
890
+ // sending it is how the two copies came to disagree. So the pooled call
891
+ // forwards only what the caller actually asked for.
892
+ // ...and it is read through statedCapOf(), so a nested chat() that states
893
+ // the key overrides its own derived `maxTokens` instead of being mistaken
894
+ // for a caller who asked for that number (see statedCapOf).
895
+ const statedMaxTokens = statedCapOf(payload);
642
896
  // "Work autonomously" — routes this call through aegis1's pool_brain
643
897
  // worker fan-out (services/pool_brain.py: N reasoning workers + a
644
898
  // synthesis pass) instead of a single provider call. UI-gated to the
@@ -710,9 +964,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
710
964
  }
711
965
 
712
966
  const base = {
713
- cls, model, mode: payload && payload.mode, maxTokens, autonomous, sessionId, signal, onDelta, cfg, apiKey, toolChoice,
967
+ cls, model, mode: payload && payload.mode, maxTokens, statedMaxTokens, autonomous, sessionId, signal, onDelta, cfg, apiKey, toolChoice,
714
968
  effort: payload && payload.effort,
715
969
  workers: payload && payload.workers,
970
+ // Deep recall (`aegis_recall_deep`) is an explicit per-session opt-in
971
+ // and never a default: it costs one provider embedding per turn
972
+ // server-side (aegis1 services/brain_memory.py find_cached_answer), so
973
+ // a client that pays per turn must not turn it on for itself. Only a
974
+ // literal `true` from the caller counts — an absent or `undefined`
975
+ // field is off, which is what keeps every existing caller (the
976
+ // renderer's IPC payloads included) on the cheap read.
977
+ recallDeep: payload && payload.recallDeep === true,
716
978
  onReasoning,
717
979
  idleTimeoutMs: autonomous ? AUTONOMOUS_IDLE_TIMEOUT_MS : undefined,
718
980
  // A caller with no live streaming surface (a `--no-stream` CLI flag, a
@@ -741,6 +1003,26 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
741
1003
  let truncationRetried = false;
742
1004
  let synthesisDone = false;
743
1005
 
1006
+ // Round cap — the bound aegiscodex-dev has had all along and this engine
1007
+ // did not. Removing the old fixed cap (12) was right in spirit and wrong
1008
+ // in effect: it left the turn with NO horizon, and on 2026-09-15 the
1009
+ // question "can you check the plan" ran ~70 rounds, grew the context
1010
+ // from 2,260 to 116,011 tokens, cost about EUR 2, and spent those rounds
1011
+ // writing 400 lines of unrequested code into the source tree. Each round
1012
+ // re-sends the whole conversation, so an unbounded loop gets more
1013
+ // expensive the longer it runs.
1014
+ //
1015
+ // The numbers match aegiscodex-dev's (src/autonomous.js) so both clients
1016
+ // behave the same: 24 rounds for a chat turn, 40 for an autonomous one.
1017
+ // Env-overridable for a deliberately long job.
1018
+ const maxRounds = (() => {
1019
+ const name = autonomous ? 'AEGIS_AUTONOMOUS_MAX_ROUNDS' : 'AEGIS_CHAT_MAX_ROUNDS';
1020
+ const raw = Number.parseInt(process.env[name] || '', 10);
1021
+ if (Number.isFinite(raw) && raw > 0) return raw;
1022
+ return autonomous ? 40 : 24;
1023
+ })();
1024
+ let round = 0;
1025
+
744
1026
  // Token accounting for the whole TURN, not just its last round. An
745
1027
  // agentic turn makes one provider call per tool round, and returning only
746
1028
  // the final round's `usage` (what this did) reported a fraction of what
@@ -789,7 +1071,30 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
789
1071
  };
790
1072
 
791
1073
  for (;;) {
792
- const opts = { ...base, system, messages: history, prompt, tools: toolSchemas };
1074
+ // Stop and SAY so. A turn that reaches its horizon has usually done
1075
+ // real work; ending silently would paint an empty answer over it,
1076
+ // which is the same "(empty response)" failure the guards below exist
1077
+ // to prevent.
1078
+ if (round >= maxRounds) {
1079
+ const note =
1080
+ `[stopped at ${maxRounds} tool rounds` +
1081
+ `${turnUsage.total_tokens ? `, ${turnUsage.total_tokens.toLocaleString()} tokens` : ''}` +
1082
+ `. Ask again to continue, or raise ` +
1083
+ `${autonomous ? 'AEGIS_AUTONOMOUS_MAX_ROUNDS' : 'AEGIS_CHAT_MAX_ROUNDS'}.]`;
1084
+ if (rootOnDelta) rootOnDelta({ delta: `\n\n${note}` });
1085
+ return withTurnUsage({
1086
+ model: base.model,
1087
+ choices: [{ message: { content: note }, finish_reason: 'length' }],
1088
+ stoppedOnRounds: true,
1089
+ });
1090
+ }
1091
+ round += 1;
1092
+ // `round === 1` is the user's own ask, and the ONLY dispatch allowed to
1093
+ // carry the deep-recall opt-in: the deep tier embeds the query once per
1094
+ // dispatch, so leaving it on for an agentic turn would charge one
1095
+ // embedding per tool round instead of one per turn (the retries below
1096
+ // clear it explicitly, being re-dispatches inside round 1).
1097
+ const opts = { ...base, system, messages: history, prompt, tools: toolSchemas, recallDeep: base.recallDeep && round === 1 };
793
1098
  let res;
794
1099
  try {
795
1100
  res = await dispatch(cls, opts);
@@ -833,7 +1138,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
833
1138
  res = await dispatch(cls, {
834
1139
  ...opts,
835
1140
  singlePass: true,
1141
+ // A re-dispatch, not a new ask: the deep tier's embedding was
1142
+ // bought by round 1, and buying it again here would charge a
1143
+ // second one for the same context.
1144
+ recallDeep: false,
836
1145
  maxTokens: doubledBudget(opts.maxTokens),
1146
+ // Doubling applies to the pooled path only when the caller stated a
1147
+ // number. With none stated, the server's effort ladder IS the
1148
+ // budget, and sending doubledBudget's 8192 floor would *lower* it
1149
+ // (aegis1 reads max_tokens as a ceiling over the ladder) — a
1150
+ // "double the budget" retry that halves it at high effort.
1151
+ statedMaxTokens: opts.statedMaxTokens ? doubledBudget(opts.statedMaxTokens) : undefined,
837
1152
  });
838
1153
  addUsage(res);
839
1154
  }
@@ -858,6 +1173,9 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
858
1173
  res = await dispatch(cls, {
859
1174
  ...opts,
860
1175
  singlePass: true,
1176
+ // Same as the truncation retry above: this pass writes up findings
1177
+ // already in `history`, and a fresh embedding buys it nothing.
1178
+ recallDeep: false,
861
1179
  messages: history,
862
1180
  prompt: '',
863
1181
  tools: [],
@@ -891,7 +1209,8 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
891
1209
  for (const call of calls) {
892
1210
  const result = call.name === T.SUBAGENT_TOOL
893
1211
  ? await runSubagent(call.args, {
894
- cls, model, maxTokens, mode: payload && payload.mode, parentSignal: signal, depth, rootSessionId, rootOnDelta,
1212
+ cls, model, statedMaxTokens, effort: payload && payload.effort,
1213
+ mode: payload && payload.mode, parentSignal: signal, depth, rootSessionId, rootOnDelta,
895
1214
  })
896
1215
  : await gatedExecuteTool(call, { toolCtx, rootSessionId, rootOnDelta, signal });
897
1216
  // A subagent's spend rides back on its tool result (see runSubagent).
@@ -921,7 +1240,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
921
1240
  */
922
1241
  async function runSubagent(
923
1242
  { description, subagent_type, prompt: subPrompt } = {},
924
- { cls, model, maxTokens, mode, parentSignal, depth, rootSessionId, rootOnDelta } = {}
1243
+ { cls, model, statedMaxTokens, effort, mode, parentSignal, depth, rootSessionId, rootOnDelta } = {}
925
1244
  ) {
926
1245
  const task = String(subPrompt || description || '').trim();
927
1246
  if (!task) return { ok: false, error: 'task requires a prompt' };
@@ -945,8 +1264,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
945
1264
  // silently hanging behind this call's no-op onDelta below.
946
1265
  const res = await chat(
947
1266
  {
948
- class: cls, model, maxTokens, mode, system, prompt: task, sessionId: subSessionId, depth: (depth || 0) + 1,
1267
+ class: cls, model, mode, system, prompt: task, sessionId: subSessionId, depth: (depth || 0) + 1,
949
1268
  rootSessionId, rootOnDelta,
1269
+ // The parent's budget AUTHORITY is forwarded, not the number it
1270
+ // implies. `maxTokens` is deliberately absent: it is the rung this
1271
+ // turn derived from `effort`, and a re-entry that hands it back gets
1272
+ // read as the caller's own cap (statedCapOf). The rung itself rides
1273
+ // along as `effort`, which is what aegis1 sizes the fan-out and its
1274
+ // ladder from — so a subagent now runs at the effort the user chose
1275
+ // instead of at a number that only meant anything for this model id.
1276
+ statedMaxTokens,
1277
+ effort,
950
1278
  },
951
1279
  () => {}
952
1280
  );
@@ -985,4 +1313,4 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
985
1313
  };
986
1314
  }
987
1315
 
988
- module.exports = { CLASSES, createLocalEngine, extractToolCalls, parseArgs };
1316
+ module.exports = { CLASSES, createLocalEngine, extractToolCalls, parseArgs, reasoningBudget };