aegis-desktop 0.4.3 → 0.4.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -473,6 +473,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
473
473
 
474
474
  async function listModels(cls) {
475
475
  if (cls === 'aegis') {
476
+ // The catalog is behind the account's key: GET /api/v1/models answers
477
+ // `401 {"error":{"message":"No API key"}}` without one (verified against
478
+ // aegiscloud.org). Aegis Cloud is the *default* class, so firing the call
479
+ // anyway painted a raw "listModels failed: … 401" over the default model
480
+ // picker for every user who had just installed the app and not yet
481
+ // pasted a key — the one state where the UI must say what unblocks it and
482
+ // not what went wrong. Report the missing key as a state (`needsKey`) and
483
+ // let the renderer invite the user to connect; nothing else about the
484
+ // class changes, and the model dropdown keeps its "server default (auto)"
485
+ // entry so the class is usable the moment a key lands.
486
+ if (!aegis.apiKey) return { class: cls, models: [], needsKey: true };
476
487
  const data = await aegis.listModels();
477
488
  return { class: cls, models: filterAegisCatalog(normalizeCatalog(data && data.models)) };
478
489
  }
@@ -535,7 +546,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
535
546
  model: opts.model,
536
547
  mode: opts.mode,
537
548
  maxTokens: opts.maxTokens,
538
- stream: true,
549
+ stream: opts.stream !== false,
539
550
  // The pooled (Nexus) brain is streamed, and an OpenAI-compatible SSE
540
551
  // stream reports no token usage unless asked. Without this the Aegis
541
552
  // Cloud class — the desktop's default — was the one class that answered
@@ -573,13 +584,18 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
573
584
  // model as the fan-out, one sample instead of four — otherwise
574
585
  // dropping the fan-out would have quietly changed the model too.
575
586
  ...(brainFlag === false ? { mode: 'brain' } : {}),
576
- // Only meaningful (and only sent) alongside a running fan-out — aegis1
577
- // services/pool_brain.py parse_brain_request reads `effort`/
578
- // `workers` straight off the body and clamps them itself
579
- // (EFFORT_LEVELS / MAX_WORKERS), so no client-side validation here.
580
- // Keyed on the effective brain flag, not on `autonomous`: an opted-out
581
- // single pass carries no fan-out tuning it cannot use.
582
- ...(brainFlag === true && opts.effort ? { effort: opts.effort } : {}),
587
+ // `effort` is the budget rung, and it is sent whenever the caller has
588
+ // one — not only alongside a fan-out. The fan-out is triggered by the
589
+ // model id (this class sends ``nexus-brain``), so a caller that never
590
+ // ticked "work autonomously" still ran a pooled call and had no way to
591
+ // say how big it should be: the server fell through to its own default
592
+ // rung. That is how a one-word prompt came to reserve the top of the
593
+ // ladder. A single-pass pass carries it too — inert on that path, but
594
+ // it keeps the request honest about what it asked for.
595
+ ...(opts.effort ? { effort: opts.effort } : {}),
596
+ // Only meaningful with a running fan-out — aegis1 services/pool_brain.py
597
+ // parse_brain_request reads `workers` straight off the body and clamps
598
+ // it itself (MAX_WORKERS), so no client-side validation here.
583
599
  ...(brainFlag === true && opts.workers ? { workers: opts.workers } : {}),
584
600
  // The pool forwards `tools` to the provider and returns tool_calls
585
601
  // (aegis1 app.py:7765 → provider, pool_brain synthesis keeps them).
@@ -699,6 +715,11 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
699
715
  workers: payload && payload.workers,
700
716
  onReasoning,
701
717
  idleTimeoutMs: autonomous ? AUTONOMOUS_IDLE_TIMEOUT_MS : undefined,
718
+ // A caller with no live streaming surface (a `--no-stream` CLI flag, a
719
+ // one-shot script) can ask for the buffered non-stream wire form
720
+ // instead. Undefined/anything but `false` keeps every existing caller
721
+ // (the desktop renderer never sets this) on the streamed path.
722
+ stream: payload && payload.stream === false ? false : true,
702
723
  };
703
724
 
704
725
  // No round cap: a model that keeps calling tools keeps going for as
@@ -782,6 +803,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
782
803
  }
783
804
  addUsage(res);
784
805
 
806
+ // A cancelled turn is over. Both recoveries below exist for "the model
807
+ // said nothing", and an aborted request resolves with exactly that —
808
+ // no text — so without this guard a user pressing Esc to stop a slow
809
+ // turn immediately issued ANOTHER billed provider call (and, because
810
+ // every transport concatenates the deltas it forwards, streamed the
811
+ // partial answer a second time onto the same bubble). Measured before
812
+ // the guard: one interrupted turn == two dispatches.
813
+ if (signal.aborted) return withTurnUsage(res);
814
+
785
815
  // Budget exhausted before the answer was written. Doubling it costs
786
816
  // one request and converts a dead turn into a real one; a second
787
817
  // 'length' result is accepted as-is so a hard-capped model can't
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "aegis-desktop",
3
3
  "productName": "AEGIS Desktop",
4
- "version": "0.4.3",
4
+ "version": "0.4.5",
5
5
  "description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
6
6
  "author": {
7
7
  "name": "AEGIS Code",
package/renderer/app.js CHANGED
@@ -69,7 +69,10 @@ const ELEMENT_IDS = {
69
69
  modelPreset: 'model-preset',
70
70
  modelInput: 'model-input',
71
71
  maxTokens: 'max-tokens',
72
+ maxTokensLabel: 'max-tokens-label',
73
+ maxTokensRow: 'max-tokens-row',
72
74
  maxTokensAdaptive: 'max-tokens-adaptive',
75
+ effortRow: 'effort-row',
73
76
  autonomousToggle: 'autonomous-toggle',
74
77
  autonomousToggleWrap: 'autonomous-toggle-wrap',
75
78
  autonomousControls: 'autonomous-controls',
@@ -171,6 +174,11 @@ const CUSTOM_MODEL_PRESETS = {
171
174
  // already filters out of settings.list(); never render it as a provider row
172
175
  // even if a stale store still surfaces it (defect #1).
173
176
  const RESERVED_PROVIDERS = new Set(['__aegis', 'aegis']);
177
+ // Where a user without a key gets one. The class picker defaults to Aegis Cloud
178
+ // and the catalog is key-gated, so this is the first thing a new install needs;
179
+ // it lives here rather than inline so the Model hint and any future "connect"
180
+ // affordance cannot drift to two different pages.
181
+ const GET_AEGIS_KEY_URL = 'https://aegiscloud.org';
174
182
 
175
183
  let pendingEl = null;
176
184
  let pendingSessionId = null;
@@ -346,6 +354,35 @@ function updateAutonomousControlsVisibility() {
346
354
  els.autonomousControls.hidden = !(wrapVisible && autonomousEnabled());
347
355
  }
348
356
 
357
+ /**
358
+ * Which budget control applies to the selected class.
359
+ *
360
+ * Aegis Cloud is sized server-side from `effort`; the other three classes take
361
+ * a per-call token ceiling from the dropdown. Showing both at once is what made
362
+ * the token cap untrustworthy on the pooled class — the dropdown was displayed,
363
+ * read on every send, and then raised by the server's effort ladder, so the
364
+ * number beside it was never the budget the call ran on. Exactly one control is
365
+ * on offer now, and it is the one the request actually travels with.
366
+ */
367
+ function updateBudgetControls(cls) {
368
+ const pooled = cls === AUTONOMOUS_CLASS;
369
+ if (els.effortRow) els.effortRow.hidden = !pooled;
370
+ if (els.maxTokensRow) els.maxTokensRow.hidden = pooled;
371
+ if (els.maxTokensLabel) els.maxTokensLabel.hidden = pooled;
372
+ }
373
+
374
+ /**
375
+ * The effort to send, or undefined for the classes the server does not size
376
+ * from it. `auto` means "let the server infer it from the ask" — the same thing
377
+ * the server already does for a request that names no effort, and a deliberate
378
+ * choice rather than the old silent fall-through to the top rung.
379
+ */
380
+ function effortFor(cls) {
381
+ if (cls !== AUTONOMOUS_CLASS) return undefined;
382
+ const value = els.autonomousEffort && els.autonomousEffort.value;
383
+ return value && value !== 'auto' ? value : undefined;
384
+ }
385
+
349
386
  // ---------------------------------------------------------------- UI helpers
350
387
 
351
388
  function setConn(ok, text) {
@@ -1764,6 +1801,7 @@ async function loadModels(cls) {
1764
1801
  els.autonomousToggleWrap.hidden = cls !== AUTONOMOUS_CLASS;
1765
1802
  }
1766
1803
  updateAutonomousControlsVisibility();
1804
+ updateBudgetControls(cls);
1767
1805
 
1768
1806
  const custom = CUSTOM_CLASSES.has(cls);
1769
1807
  els.modelSelect.hidden = custom;
@@ -1858,6 +1896,11 @@ async function loadModels(cls) {
1858
1896
  try {
1859
1897
  const data = await models.listModels(cls);
1860
1898
  const list = Array.isArray(data && data.models) ? data.models : [];
1899
+ // No key on the default class: the engine reports the missing credential as
1900
+ // a state instead of letting the catalog call 401 (see engine.listModels).
1901
+ // The hint is where a user finds out they can connect at all — a raw
1902
+ // "listModels failed" told them only that something was broken.
1903
+ const needsKey = Boolean(data && data.needsKey);
1861
1904
  for (const m of list) {
1862
1905
  modelMeta.set(m.id, m);
1863
1906
  const opt = document.createElement('option');
@@ -1866,12 +1909,28 @@ async function loadModels(cls) {
1866
1909
  els.modelSelect.appendChild(opt);
1867
1910
  }
1868
1911
  let hint;
1869
- if (!list.length) {
1912
+ if (needsKey) {
1913
+ hint = null; // carries a link, built below
1914
+ } else if (!list.length) {
1870
1915
  hint = cls === 'ollama' ? 'Ollama not running or no models pulled.' : 'No models listed.';
1871
1916
  } else {
1872
1917
  hint = `${list.length} model${list.length === 1 ? '' : 's'} available.`;
1873
1918
  }
1874
1919
  const ceiling = applyMaxTokensClamp(els.modelSelect.value);
1920
+ if (needsKey) {
1921
+ // The hint elements are bare <p>s, so the link has to be a real child
1922
+ // node — a text assignment would wipe it (same shape as capNotice).
1923
+ els.modelHint.textContent =
1924
+ 'Connect Aegis Cloud to load its models: paste your AEGIS API key in the Status card above (';
1925
+ const a = document.createElement('a');
1926
+ a.href = GET_AEGIS_KEY_URL;
1927
+ a.target = '_blank';
1928
+ a.rel = 'noreferrer noopener';
1929
+ a.textContent = 'free key at aegiscloud.org';
1930
+ els.modelHint.appendChild(a);
1931
+ els.modelHint.appendChild(document.createTextNode('), then Save.'));
1932
+ return;
1933
+ }
1875
1934
  els.modelHint.textContent =
1876
1935
  ceiling < FLAT_CEILING ? `${hint} · max output: ${ceiling.toLocaleString()}` : hint;
1877
1936
  } catch (err) {
@@ -2248,12 +2307,21 @@ async function send() {
2248
2307
  addMessage('user', prompt);
2249
2308
 
2250
2309
  const ceiling = applyMaxTokensClamp(model);
2251
- const maxTokens = maxTokensAdaptive() ? ceiling : parseInt(els.maxTokens.value, 10) || 4096;
2310
+ // Aegis Cloud takes no token cap from here at all: the server sizes the call
2311
+ // from `effort`, and a number in this position is a per-pass ceiling *over*
2312
+ // that ladder (aegis1 services/pool_brain.py pass_budgets). Sending the
2313
+ // dropdown's value anyway is what made "Max tokens: 4k" beside a turn a
2314
+ // figure the turn never ran on. The row is hidden for this class as well, so
2315
+ // the two controls can never disagree.
2316
+ const maxTokens = cls === AUTONOMOUS_CLASS
2317
+ ? undefined
2318
+ : maxTokensAdaptive() ? ceiling : parseInt(els.maxTokens.value, 10) || 4096;
2252
2319
  const autonomous = cls === AUTONOMOUS_CLASS && autonomousEnabled();
2253
- // Only meaningful (and only sent) alongside `autonomous` — see
2254
- // aegis1 services/pool_brain.py parse_brain_request for the effort/workers
2255
- // clamping this feeds.
2256
- const effort = autonomous ? els.autonomousEffort.value : undefined;
2320
+ // Sent for the pooled class whether or not the fan-out is ticked: the fan-out
2321
+ // is enabled by the model id this class sends, so a turn that never entered
2322
+ // autonomous mode still ran pooled and had no way to say how big it should
2323
+ // be. `undefined` = "auto" = the server infers it from the ask.
2324
+ const effort = effortFor(cls);
2257
2325
  const workers = autonomous ? parseInt(els.autonomousWorkers.value, 10) || undefined : undefined;
2258
2326
  // Reuse the open thread's session id (minted once, on its first message)
2259
2327
  // instead of a fresh one per send — a new id every turn is what made both
@@ -2352,7 +2420,11 @@ async function send() {
2352
2420
  if (data && data.model) bits.push(`model: ${data.model}`);
2353
2421
  else if (model) bits.push(`model: ${model}`);
2354
2422
  bits.push(classLabel(cls));
2355
- if (autonomous) bits.push(`autonomous (${effort}, ${workers || 3}w)`);
2423
+ if (autonomous) bits.push(`autonomous (${effort || 'auto'}, ${workers || 'auto'}w)`);
2424
+ // The budget rung is worth showing even without the fan-out: on the pooled
2425
+ // class it is what sized the call, and a user cannot tell a 16k turn from a
2426
+ // 64k one by looking at the answer.
2427
+ else if (effort) bits.push(`effort: ${effort}`);
2356
2428
  const turnTokens = usageTokens(data && data.usage);
2357
2429
  if (turnTokens != null) bits.push(`tokens: ${turnTokens}`);
2358
2430
  addMessage('assistant', text, bits.join(' · ') || undefined, sessionId, toolLog);
@@ -2433,14 +2505,23 @@ async function init() {
2433
2505
  updateAutonomousControlsVisibility();
2434
2506
  });
2435
2507
 
2436
- // Effort/worker count for the pool_brain fan-out (aegis1
2437
- // services/pool_brain.py parse_brain_request reads `effort`/`workers` off
2438
- // the request body) — opt-in per machine, remembered across restarts.
2508
+ // Budget control for the pooled class: which rung of the server's effort
2509
+ // ladder sizes the call, or "auto" to let the server infer it from the ask
2510
+ // (aegis1 services/pool_brain.py parse_brain_request). Opt-in per machine,
2511
+ // remembered across restarts; a stored value from a build whose list was
2512
+ // short (low/medium/high, no "auto") falls through to the markup's default
2513
+ // rather than assigning an option that no longer exists.
2439
2514
  const savedEffort = localStorage.getItem(AUTONOMOUS_EFFORT_KEY);
2440
- if (savedEffort) els.autonomousEffort.value = savedEffort;
2515
+ if (savedEffort && Array.from(els.autonomousEffort.options).some((o) => o.value === savedEffort)) {
2516
+ els.autonomousEffort.value = savedEffort;
2517
+ }
2441
2518
  els.autonomousEffort.addEventListener('change', () => {
2442
2519
  localStorage.setItem(AUTONOMOUS_EFFORT_KEY, els.autonomousEffort.value);
2443
2520
  });
2521
+ // Worker count for the fan-out. Left empty by default on purpose: an empty
2522
+ // field is what tells the server to size the fan-out from the ask
2523
+ // (parse_brain_request's auto path) instead of the old client-side default of
2524
+ // 3, which made every turn a 4-pass fan-out.
2444
2525
  const savedWorkers = localStorage.getItem(AUTONOMOUS_WORKERS_KEY);
2445
2526
  if (savedWorkers) els.autonomousWorkers.value = savedWorkers;
2446
2527
  els.autonomousWorkers.addEventListener('change', () => {
@@ -2458,6 +2539,11 @@ async function init() {
2458
2539
 
2459
2540
  els.classSelect.addEventListener('change', () => {
2460
2541
  localStorage.setItem(CLASS_KEY, els.classSelect.value);
2542
+ // Apply the budget-control choice immediately rather than waiting for the
2543
+ // model list: loadModels() is async, and until it resolves the previous
2544
+ // class's control would still be on screen — showing a token cap on a
2545
+ // pooled turn, or hiding the effort rung it runs on.
2546
+ updateBudgetControls(els.classSelect.value);
2461
2547
  loadModels(els.classSelect.value);
2462
2548
  });
2463
2549
 
@@ -76,8 +76,8 @@
76
76
  hidden
77
77
  />
78
78
 
79
- <label for="max-tokens">Max tokens</label>
80
- <div class="max-tokens-row">
79
+ <label for="max-tokens" id="max-tokens-label">Max tokens</label>
80
+ <div class="max-tokens-row" id="max-tokens-row">
81
81
  <select id="max-tokens">
82
82
  <option value="1024">1k</option>
83
83
  <option value="4096" selected>4k</option>
@@ -95,6 +95,23 @@
95
95
  </label>
96
96
  </div>
97
97
 
98
+ <!-- The pooled AEGIS Cloud class sizes a request's token budget
99
+ server-side from this ladder (effort -> low/medium/high ->
100
+ 16384/32768/65536 total across the fan-out), so the max-tokens row
101
+ above is hidden for that class instead of being shown and silently
102
+ overridden: the server treats a body max_tokens as a ceiling, and
103
+ an effort rung as the budget. Shown for Ollama and custom
104
+ endpoints, where a per-call ceiling is the only control there is. -->
105
+ <div class="max-tokens-row" id="effort-row" hidden>
106
+ <label for="autonomous-effort">Effort</label>
107
+ <select id="autonomous-effort">
108
+ <option value="low">low</option>
109
+ <option value="medium">medium</option>
110
+ <option value="high">high</option>
111
+ <option value="auto" selected>auto</option>
112
+ </select>
113
+ </div>
114
+
98
115
  <label
99
116
  class="flow-toggle"
100
117
  for="autonomous-toggle"
@@ -106,14 +123,8 @@
106
123
  <span>work autonomously</span>
107
124
  </label>
108
125
  <div id="autonomous-controls" class="autonomous-controls" hidden>
109
- <label for="autonomous-effort">Effort</label>
110
- <select id="autonomous-effort">
111
- <option value="low">low</option>
112
- <option value="medium">medium</option>
113
- <option value="high" selected>high</option>
114
- </select>
115
126
  <label for="autonomous-workers">Workers</label>
116
- <input type="number" id="autonomous-workers" min="1" max="8" step="1" value="3" />
127
+ <input type="number" id="autonomous-workers" min="1" max="8" step="1" />
117
128
  </div>
118
129
  <p class="hint" id="model-hint"></p>
119
130
  </section>
@@ -1142,6 +1142,17 @@ body.memory-open {
1142
1142
  display: none;
1143
1143
  }
1144
1144
 
1145
+ /* Same trap, one row up: `.max-tokens-row` is display: flex, so the `hidden`
1146
+ attribute did nothing on `#max-tokens-row`/`#effort-row` — which is how the
1147
+ budget-control swap (effort for the pooled class, a token ceiling for the
1148
+ others) would have shown BOTH controls at once, the exact state it exists to
1149
+ prevent. `display: none` is also what lets a `<label>` be hidden, since
1150
+ labels are inline by default and carry no author display rule. */
1151
+ .max-tokens-row[hidden],
1152
+ label[hidden] {
1153
+ display: none;
1154
+ }
1155
+
1145
1156
  .autonomous-controls {
1146
1157
  display: flex;
1147
1158
  align-items: center;
package/vendor/aegis.js CHANGED
@@ -278,6 +278,13 @@ function createClient(opts = {}) {
278
278
  * so a reasoning-only stream still counts as unanswered.
279
279
  * - `idleTimeoutMs` override the stalled-stream watchdog for this call
280
280
  * (a worker fan-out is legitimately silent between passes)
281
+ * - `maxTokens` output ceiling. Omitted from the body when absent so the
282
+ * server's own budget applies — on a pooled request that is
283
+ * the `effort` ladder, and a number sent from here caps it.
284
+ * - `effort` / `workers` pooled-brain budget rung and fan-out size.
285
+ * Forwarded only inside `extra`, as the server reads them off
286
+ * the body (aegis1 services/pool_brain.py parse_brain_request)
287
+ * and sizes the budget from them.
281
288
  */
282
289
  async function chatCompletion({
283
290
  prompt,
@@ -298,9 +305,17 @@ function createClient(opts = {}) {
298
305
  // model the server picks its default (no client-invented tier id). `mode`
299
306
  // is a legacy server-side shorthand — forwarded verbatim only when the
300
307
  // caller supplies it, never defaulted, never built into a model id.
308
+ //
309
+ // `max_tokens` is omitted entirely when the caller states none, rather than
310
+ // defaulted here. The server has its own budget ladder (`mode`/`effort` on
311
+ // the pooled path) and treats a body max_tokens as a *ceiling* over it, so
312
+ // an invented 4096 was not a harmless default: it capped every pass of a
313
+ // pooled-brain fan-out at 4096 and overrode the ladder the caller's effort
314
+ // selected. Omitting it is the documented way to say "server default" (see
315
+ // mcp/tools.js's max_tokens description).
301
316
  const body = {
302
317
  messages: buildMessages(messages, system, prompt),
303
- max_tokens: maxTokens || 4096,
318
+ ...(maxTokens ? { max_tokens: maxTokens } : {}),
304
319
  ...(extra || {}),
305
320
  };
306
321
  if (model) body.model = model;