aegis-desktop 0.4.4 → 0.4.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -584,13 +584,18 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
584
584
  // model as the fan-out, one sample instead of four — otherwise
585
585
  // dropping the fan-out would have quietly changed the model too.
586
586
  ...(brainFlag === false ? { mode: 'brain' } : {}),
587
- // Only meaningful (and only sent) alongside a running fan-out — aegis1
588
- // services/pool_brain.py parse_brain_request reads `effort`/
589
- // `workers` straight off the body and clamps them itself
590
- // (EFFORT_LEVELS / MAX_WORKERS), so no client-side validation here.
591
- // Keyed on the effective brain flag, not on `autonomous`: an opted-out
592
- // single pass carries no fan-out tuning it cannot use.
593
- ...(brainFlag === true && opts.effort ? { effort: opts.effort } : {}),
587
+ // `effort` is the budget rung, and it is sent whenever the caller has
588
+ // one — not only alongside a fan-out. The fan-out is triggered by the
589
+ // model id (this class sends ``nexus-brain``), so a caller that never
590
+ // ticked "work autonomously" still ran a pooled call and had no way to
591
+ // say how big it should be: the server fell through to its own default
592
+ // rung. That is how a one-word prompt came to reserve the top of the
593
+ // ladder. A single-pass pass carries it too — inert on that path, but
594
+ // it keeps the request honest about what it asked for.
595
+ ...(opts.effort ? { effort: opts.effort } : {}),
596
+ // Only meaningful with a running fan-out — aegis1 services/pool_brain.py
597
+ // parse_brain_request reads `workers` straight off the body and clamps
598
+ // it itself (MAX_WORKERS), so no client-side validation here.
594
599
  ...(brainFlag === true && opts.workers ? { workers: opts.workers } : {}),
595
600
  // The pool forwards `tools` to the provider and returns tool_calls
596
601
  // (aegis1 app.py:7765 → provider, pool_brain synthesis keeps them).
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "aegis-desktop",
3
3
  "productName": "AEGIS Desktop",
4
- "version": "0.4.4",
4
+ "version": "0.4.5",
5
5
  "description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
6
6
  "author": {
7
7
  "name": "AEGIS Code",
package/renderer/app.js CHANGED
@@ -69,7 +69,10 @@ const ELEMENT_IDS = {
69
69
  modelPreset: 'model-preset',
70
70
  modelInput: 'model-input',
71
71
  maxTokens: 'max-tokens',
72
+ maxTokensLabel: 'max-tokens-label',
73
+ maxTokensRow: 'max-tokens-row',
72
74
  maxTokensAdaptive: 'max-tokens-adaptive',
75
+ effortRow: 'effort-row',
73
76
  autonomousToggle: 'autonomous-toggle',
74
77
  autonomousToggleWrap: 'autonomous-toggle-wrap',
75
78
  autonomousControls: 'autonomous-controls',
@@ -351,6 +354,35 @@ function updateAutonomousControlsVisibility() {
351
354
  els.autonomousControls.hidden = !(wrapVisible && autonomousEnabled());
352
355
  }
353
356
 
357
+ /**
358
+ * Which budget control applies to the selected class.
359
+ *
360
+ * Aegis Cloud is sized server-side from `effort`; the other three classes take
361
+ * a per-call token ceiling from the dropdown. Showing both at once is what made
362
+ * the token cap untrustworthy on the pooled class — the dropdown was displayed,
363
+ * read on every send, and then raised by the server's effort ladder, so the
364
+ * number beside it was never the budget the call ran on. Exactly one control is
365
+ * on offer now, and it is the one the request actually travels with.
366
+ */
367
+ function updateBudgetControls(cls) {
368
+ const pooled = cls === AUTONOMOUS_CLASS;
369
+ if (els.effortRow) els.effortRow.hidden = !pooled;
370
+ if (els.maxTokensRow) els.maxTokensRow.hidden = pooled;
371
+ if (els.maxTokensLabel) els.maxTokensLabel.hidden = pooled;
372
+ }
373
+
374
+ /**
375
+ * The effort to send, or undefined for the classes the server does not size
376
+ * from it. `auto` means "let the server infer it from the ask" — the same thing
377
+ * the server already does for a request that names no effort, and a deliberate
378
+ * choice rather than the old silent fall-through to the top rung.
379
+ */
380
+ function effortFor(cls) {
381
+ if (cls !== AUTONOMOUS_CLASS) return undefined;
382
+ const value = els.autonomousEffort && els.autonomousEffort.value;
383
+ return value && value !== 'auto' ? value : undefined;
384
+ }
385
+
354
386
  // ---------------------------------------------------------------- UI helpers
355
387
 
356
388
  function setConn(ok, text) {
@@ -1769,6 +1801,7 @@ async function loadModels(cls) {
1769
1801
  els.autonomousToggleWrap.hidden = cls !== AUTONOMOUS_CLASS;
1770
1802
  }
1771
1803
  updateAutonomousControlsVisibility();
1804
+ updateBudgetControls(cls);
1772
1805
 
1773
1806
  const custom = CUSTOM_CLASSES.has(cls);
1774
1807
  els.modelSelect.hidden = custom;
@@ -2274,12 +2307,21 @@ async function send() {
2274
2307
  addMessage('user', prompt);
2275
2308
 
2276
2309
  const ceiling = applyMaxTokensClamp(model);
2277
- const maxTokens = maxTokensAdaptive() ? ceiling : parseInt(els.maxTokens.value, 10) || 4096;
2310
+ // Aegis Cloud takes no token cap from here at all: the server sizes the call
2311
+ // from `effort`, and a number in this position is a per-pass ceiling *over*
2312
+ // that ladder (aegis1 services/pool_brain.py pass_budgets). Sending the
2313
+ // dropdown's value anyway is what made "Max tokens: 4k" beside a turn a
2314
+ // figure the turn never ran on. The row is hidden for this class as well, so
2315
+ // the two controls can never disagree.
2316
+ const maxTokens = cls === AUTONOMOUS_CLASS
2317
+ ? undefined
2318
+ : maxTokensAdaptive() ? ceiling : parseInt(els.maxTokens.value, 10) || 4096;
2278
2319
  const autonomous = cls === AUTONOMOUS_CLASS && autonomousEnabled();
2279
- // Only meaningful (and only sent) alongside `autonomous` — see
2280
- // aegis1 services/pool_brain.py parse_brain_request for the effort/workers
2281
- // clamping this feeds.
2282
- const effort = autonomous ? els.autonomousEffort.value : undefined;
2320
+ // Sent for the pooled class whether or not the fan-out is ticked: the fan-out
2321
+ // is enabled by the model id this class sends, so a turn that never entered
2322
+ // autonomous mode still ran pooled and had no way to say how big it should
2323
+ // be. `undefined` = "auto" = the server infers it from the ask.
2324
+ const effort = effortFor(cls);
2283
2325
  const workers = autonomous ? parseInt(els.autonomousWorkers.value, 10) || undefined : undefined;
2284
2326
  // Reuse the open thread's session id (minted once, on its first message)
2285
2327
  // instead of a fresh one per send — a new id every turn is what made both
@@ -2378,7 +2420,11 @@ async function send() {
2378
2420
  if (data && data.model) bits.push(`model: ${data.model}`);
2379
2421
  else if (model) bits.push(`model: ${model}`);
2380
2422
  bits.push(classLabel(cls));
2381
- if (autonomous) bits.push(`autonomous (${effort}, ${workers || 3}w)`);
2423
+ if (autonomous) bits.push(`autonomous (${effort || 'auto'}, ${workers || 'auto'}w)`);
2424
+ // The budget rung is worth showing even without the fan-out: on the pooled
2425
+ // class it is what sized the call, and a user cannot tell a 16k turn from a
2426
+ // 64k one by looking at the answer.
2427
+ else if (effort) bits.push(`effort: ${effort}`);
2382
2428
  const turnTokens = usageTokens(data && data.usage);
2383
2429
  if (turnTokens != null) bits.push(`tokens: ${turnTokens}`);
2384
2430
  addMessage('assistant', text, bits.join(' · ') || undefined, sessionId, toolLog);
@@ -2459,14 +2505,23 @@ async function init() {
2459
2505
  updateAutonomousControlsVisibility();
2460
2506
  });
2461
2507
 
2462
- // Effort/worker count for the pool_brain fan-out (aegis1
2463
- // services/pool_brain.py parse_brain_request reads `effort`/`workers` off
2464
- // the request body) — opt-in per machine, remembered across restarts.
2508
+ // Budget control for the pooled class: which rung of the server's effort
2509
+ // ladder sizes the call, or "auto" to let the server infer it from the ask
2510
+ // (aegis1 services/pool_brain.py parse_brain_request). Opt-in per machine,
2511
+ // remembered across restarts; a stored value from a build whose list was
2512
+ // short (low/medium/high, no "auto") falls through to the markup's default
2513
+ // rather than assigning an option that no longer exists.
2465
2514
  const savedEffort = localStorage.getItem(AUTONOMOUS_EFFORT_KEY);
2466
- if (savedEffort) els.autonomousEffort.value = savedEffort;
2515
+ if (savedEffort && Array.from(els.autonomousEffort.options).some((o) => o.value === savedEffort)) {
2516
+ els.autonomousEffort.value = savedEffort;
2517
+ }
2467
2518
  els.autonomousEffort.addEventListener('change', () => {
2468
2519
  localStorage.setItem(AUTONOMOUS_EFFORT_KEY, els.autonomousEffort.value);
2469
2520
  });
2521
+ // Worker count for the fan-out. Left empty by default on purpose: an empty
2522
+ // field is what tells the server to size the fan-out from the ask
2523
+ // (parse_brain_request's auto path) instead of the old client-side default of
2524
+ // 3, which made every turn a 4-pass fan-out.
2470
2525
  const savedWorkers = localStorage.getItem(AUTONOMOUS_WORKERS_KEY);
2471
2526
  if (savedWorkers) els.autonomousWorkers.value = savedWorkers;
2472
2527
  els.autonomousWorkers.addEventListener('change', () => {
@@ -2484,6 +2539,11 @@ async function init() {
2484
2539
 
2485
2540
  els.classSelect.addEventListener('change', () => {
2486
2541
  localStorage.setItem(CLASS_KEY, els.classSelect.value);
2542
+ // Apply the budget-control choice immediately rather than waiting for the
2543
+ // model list: loadModels() is async, and until it resolves the previous
2544
+ // class's control would still be on screen — showing a token cap on a
2545
+ // pooled turn, or hiding the effort rung it runs on.
2546
+ updateBudgetControls(els.classSelect.value);
2487
2547
  loadModels(els.classSelect.value);
2488
2548
  });
2489
2549
 
@@ -76,8 +76,8 @@
76
76
  hidden
77
77
  />
78
78
 
79
- <label for="max-tokens">Max tokens</label>
80
- <div class="max-tokens-row">
79
+ <label for="max-tokens" id="max-tokens-label">Max tokens</label>
80
+ <div class="max-tokens-row" id="max-tokens-row">
81
81
  <select id="max-tokens">
82
82
  <option value="1024">1k</option>
83
83
  <option value="4096" selected>4k</option>
@@ -95,6 +95,23 @@
95
95
  </label>
96
96
  </div>
97
97
 
98
+ <!-- The pooled AEGIS Cloud class sizes a request's token budget
99
+ server-side from this ladder (effort -> low/medium/high ->
100
+ 16384/32768/65536 total across the fan-out), so the max-tokens row
101
+ above is hidden for that class instead of being shown and silently
102
+ overridden: the server treats a body max_tokens as a ceiling, and
103
+ an effort rung as the budget. Shown for Ollama and custom
104
+ endpoints, where a per-call ceiling is the only control there is. -->
105
+ <div class="max-tokens-row" id="effort-row" hidden>
106
+ <label for="autonomous-effort">Effort</label>
107
+ <select id="autonomous-effort">
108
+ <option value="low">low</option>
109
+ <option value="medium">medium</option>
110
+ <option value="high">high</option>
111
+ <option value="auto" selected>auto</option>
112
+ </select>
113
+ </div>
114
+
98
115
  <label
99
116
  class="flow-toggle"
100
117
  for="autonomous-toggle"
@@ -106,14 +123,8 @@
106
123
  <span>work autonomously</span>
107
124
  </label>
108
125
  <div id="autonomous-controls" class="autonomous-controls" hidden>
109
- <label for="autonomous-effort">Effort</label>
110
- <select id="autonomous-effort">
111
- <option value="low">low</option>
112
- <option value="medium">medium</option>
113
- <option value="high" selected>high</option>
114
- </select>
115
126
  <label for="autonomous-workers">Workers</label>
116
- <input type="number" id="autonomous-workers" min="1" max="8" step="1" value="3" />
127
+ <input type="number" id="autonomous-workers" min="1" max="8" step="1" />
117
128
  </div>
118
129
  <p class="hint" id="model-hint"></p>
119
130
  </section>
@@ -1142,6 +1142,17 @@ body.memory-open {
1142
1142
  display: none;
1143
1143
  }
1144
1144
 
1145
+ /* Same trap, one row up: `.max-tokens-row` is display: flex, so the `hidden`
1146
+ attribute did nothing on `#max-tokens-row`/`#effort-row` — which is how the
1147
+ budget-control swap (effort for the pooled class, a token ceiling for the
1148
+ others) would have shown BOTH controls at once, the exact state it exists to
1149
+ prevent. `display: none` is also what lets a `<label>` be hidden, since
1150
+ labels are inline by default and carry no author display rule. */
1151
+ .max-tokens-row[hidden],
1152
+ label[hidden] {
1153
+ display: none;
1154
+ }
1155
+
1145
1156
  .autonomous-controls {
1146
1157
  display: flex;
1147
1158
  align-items: center;
package/vendor/aegis.js CHANGED
@@ -278,6 +278,13 @@ function createClient(opts = {}) {
278
278
  * so a reasoning-only stream still counts as unanswered.
279
279
  * - `idleTimeoutMs` override the stalled-stream watchdog for this call
280
280
  * (a worker fan-out is legitimately silent between passes)
281
+ * - `maxTokens` output ceiling. Omitted from the body when absent so the
282
+ * server's own budget applies — on a pooled request that is
283
+ * the `effort` ladder, and a number sent from here caps it.
284
+ * - `effort` / `workers` pooled-brain budget rung and fan-out size.
285
+ * Forwarded only inside `extra`, as the server reads them off
286
+ * the body (aegis1 services/pool_brain.py parse_brain_request)
287
+ * and sizes the budget from them.
281
288
  */
282
289
  async function chatCompletion({
283
290
  prompt,
@@ -298,9 +305,17 @@ function createClient(opts = {}) {
298
305
  // model the server picks its default (no client-invented tier id). `mode`
299
306
  // is a legacy server-side shorthand — forwarded verbatim only when the
300
307
  // caller supplies it, never defaulted, never built into a model id.
308
+ //
309
+ // `max_tokens` is omitted entirely when the caller states none, rather than
310
+ // defaulted here. The server has its own budget ladder (`mode`/`effort` on
311
+ // the pooled path) and treats a body max_tokens as a *ceiling* over it, so
312
+ // an invented 4096 was not a harmless default: it capped every pass of a
313
+ // pooled-brain fan-out at 4096 and overrode the ladder the caller's effort
314
+ // selected. Omitting it is the documented way to say "server default" (see
315
+ // mcp/tools.js's max_tokens description).
301
316
  const body = {
302
317
  messages: buildMessages(messages, system, prompt),
303
- max_tokens: maxTokens || 4096,
318
+ ...(maxTokens ? { max_tokens: maxTokens } : {}),
304
319
  ...(extra || {}),
305
320
  };
306
321
  if (model) body.model = model;