aegis-desktop 0.4.2 → 0.4.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -96,8 +96,17 @@ const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
96
96
  * that is a multi-thousand-token reasoning pass per worker — easily past a
97
97
  * minute. Timing out there aborts a perfectly healthy autonomous turn
98
98
  * mid-flight, after the server has already run and billed every worker.
99
+ *
100
+ * 15 minutes deliberately outlasts the server's OWN ceiling for that window
101
+ * (aegis1 services/pool_brain.py: NEXUS_BRAIN_WORKER_TIMEOUT, default 600s),
102
+ * because aborting first leaves the server running and billing a fan-out
103
+ * nobody will ever see. Raising that env var past ~14 minutes means raising
104
+ * this constant too; the shared client applies the same budget from the
105
+ * server's X-AEGIS-Brain response header (see client/aegis.js idleBudgetFor),
106
+ * which is what covers a fan-out the caller did not flag — test/
107
+ * autonomous-mode.test.mjs pins both halves.
99
108
  */
100
- const AUTONOMOUS_IDLE_TIMEOUT_MS = 5 * 60_000;
109
+ const AUTONOMOUS_IDLE_TIMEOUT_MS = 15 * 60_000;
101
110
 
102
111
  /**
103
112
  * Only ever raises a too-low budget for a DeepSeek reasoning model — never
@@ -464,6 +473,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
464
473
 
465
474
  async function listModels(cls) {
466
475
  if (cls === 'aegis') {
476
+ // The catalog is behind the account's key: GET /api/v1/models answers
477
+ // `401 {"error":{"message":"No API key"}}` without one (verified against
478
+ // aegiscloud.org). Aegis Cloud is the *default* class, so firing the call
479
+ // anyway painted a raw "listModels failed: … 401" over the default model
480
+ // picker for every user who had just installed the app and not yet
481
+ // pasted a key — the one state where the UI must say what unblocks it and
482
+ // not what went wrong. Report the missing key as a state (`needsKey`) and
483
+ // let the renderer invite the user to connect; nothing else about the
484
+ // class changes, and the model dropdown keeps its "server default (auto)"
485
+ // entry so the class is usable the moment a key lands.
486
+ if (!aegis.apiKey) return { class: cls, models: [], needsKey: true };
467
487
  const data = await aegis.listModels();
468
488
  return { class: cls, models: filterAegisCatalog(normalizeCatalog(data && data.models)) };
469
489
  }
@@ -513,6 +533,12 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
513
533
  /** One transport round for the chosen class. */
514
534
  async function dispatch(cls, opts) {
515
535
  if (cls === 'aegis') {
536
+ // `undefined` = "leave the model id's own default alone" (a real user
537
+ // turn on the pooled class: the selected Nexus id is what decides, so
538
+ // today's behaviour is unchanged). `false` = an explicit single provider
539
+ // call, for a pass that is a continuation rather than a new
540
+ // investigation. `true` = the autonomous fan-out.
541
+ const brainFlag = opts.singlePass ? false : opts.autonomous ? true : undefined;
516
542
  return aegis.chatCompletion({
517
543
  prompt: opts.prompt,
518
544
  system: opts.system,
@@ -520,7 +546,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
520
546
  model: opts.model,
521
547
  mode: opts.mode,
522
548
  maxTokens: opts.maxTokens,
523
- stream: true,
549
+ stream: opts.stream !== false,
524
550
  // The pooled (Nexus) brain is streamed, and an OpenAI-compatible SSE
525
551
  // stream reports no token usage unless asked. Without this the Aegis
526
552
  // Cloud class — the desktop's default — was the one class that answered
@@ -542,13 +568,30 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
542
568
  extra: {
543
569
  aegis_memory: true,
544
570
  session: opts.sessionId,
545
- ...(opts.autonomous ? { brain: true } : {}),
546
- // Only meaningful (and only sent) alongside `brain` — aegis1
571
+ // The fan-out is opt-in per dispatch. `brain` is sent EXPLICITLY
572
+ // whenever this dispatch is not the autonomous one, because the
573
+ // model id this class sends (``nexus-brain``) enables the pooled
574
+ // brain on its own: without the flag a continuation pass — the
575
+ // doubled-budget retry, or the "write up what you already found"
576
+ // re-dispatch — silently re-ran the whole workers+1 fan-out for a
577
+ // pass whose documented cost is a single request. aegis1
578
+ // services/pool_brain.py parse_brain_request honours the opt-out.
579
+ ...(brainFlag === undefined ? {} : { brain: brainFlag }),
580
+ // An opted-out pass is also told WHICH band to run on. A single call
581
+ // on a brain model id infers its band from the id and lands on
582
+ // "fast" (the cheapest id in the pool); "brain" is the band the
583
+ // workers themselves run on (cheapest model that can think). Same
584
+ // model as the fan-out, one sample instead of four — otherwise
585
+ // dropping the fan-out would have quietly changed the model too.
586
+ ...(brainFlag === false ? { mode: 'brain' } : {}),
587
+ // Only meaningful (and only sent) alongside a running fan-out — aegis1
547
588
  // services/pool_brain.py parse_brain_request reads `effort`/
548
589
  // `workers` straight off the body and clamps them itself
549
590
  // (EFFORT_LEVELS / MAX_WORKERS), so no client-side validation here.
550
- ...(opts.autonomous && opts.effort ? { effort: opts.effort } : {}),
551
- ...(opts.autonomous && opts.workers ? { workers: opts.workers } : {}),
591
+ // Keyed on the effective brain flag, not on `autonomous`: an opted-out
592
+ // single pass carries no fan-out tuning it cannot use.
593
+ ...(brainFlag === true && opts.effort ? { effort: opts.effort } : {}),
594
+ ...(brainFlag === true && opts.workers ? { workers: opts.workers } : {}),
552
595
  // The pool forwards `tools` to the provider and returns tool_calls
553
596
  // (aegis1 app.py:7765 → provider, pool_brain synthesis keeps them).
554
597
  ...(opts.tools.length ? { tools: opts.tools } : {}),
@@ -667,6 +710,11 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
667
710
  workers: payload && payload.workers,
668
711
  onReasoning,
669
712
  idleTimeoutMs: autonomous ? AUTONOMOUS_IDLE_TIMEOUT_MS : undefined,
713
+ // A caller with no live streaming surface (a `--no-stream` CLI flag, a
714
+ // one-shot script) can ask for the buffered non-stream wire form
715
+ // instead. Undefined/anything but `false` keeps every existing caller
716
+ // (the desktop renderer never sets this) on the streamed path.
717
+ stream: payload && payload.stream === false ? false : true,
670
718
  };
671
719
 
672
720
  // No round cap: a model that keeps calling tools keeps going for as
@@ -750,6 +798,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
750
798
  }
751
799
  addUsage(res);
752
800
 
801
+ // A cancelled turn is over. Both recoveries below exist for "the model
802
+ // said nothing", and an aborted request resolves with exactly that —
803
+ // no text — so without this guard a user pressing Esc to stop a slow
804
+ // turn immediately issued ANOTHER billed provider call (and, because
805
+ // every transport concatenates the deltas it forwards, streamed the
806
+ // partial answer a second time onto the same bubble). Measured before
807
+ // the guard: one interrupted turn == two dispatches.
808
+ if (signal.aborted) return withTurnUsage(res);
809
+
753
810
  // Budget exhausted before the answer was written. Doubling it costs
754
811
  // one request and converts a dead turn into a real one; a second
755
812
  // 'length' result is accepted as-is so a hard-capped model can't
@@ -763,7 +820,16 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
763
820
  // would stream it a second time onto the same bubble.
764
821
  if (!truncationRetried && !assistantText(res) && isTruncated(res)) {
765
822
  truncationRetried = true;
766
- res = await dispatch(cls, { ...opts, maxTokens: doubledBudget(opts.maxTokens) });
823
+ // singlePass: this retry buys *budget*, not a second investigation.
824
+ // The fan-out's workers would re-run the whole task from scratch for
825
+ // it — 3 extra reasoning passes + a synthesis — which is the opposite
826
+ // of what "one doubled request" means (and what the note above
827
+ // promises). The pass itself is unchanged apart from that.
828
+ res = await dispatch(cls, {
829
+ ...opts,
830
+ singlePass: true,
831
+ maxTokens: doubledBudget(opts.maxTokens),
832
+ });
767
833
  addUsage(res);
768
834
  }
769
835
 
@@ -778,7 +844,19 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
778
844
  synthesisDone = true;
779
845
  foldPromptIntoHistory();
780
846
  history.push({ role: 'user', content: EMPTY_TURN_NUDGE });
781
- res = await dispatch(cls, { ...opts, messages: history, prompt: '', tools: [] });
847
+ // singlePass: same reasoning as the truncation retry above, and it is
848
+ // the same comment's literal promise ("force the summary out of the
849
+ // context it already holds"). Escalating a write-up back into the
850
+ // worker fan-out asked three fresh workers to redo an investigation
851
+ // whose findings are already in `history`, at 4x the cost, to produce
852
+ // a paragraph the model had all the material for.
853
+ res = await dispatch(cls, {
854
+ ...opts,
855
+ singlePass: true,
856
+ messages: history,
857
+ prompt: '',
858
+ tools: [],
859
+ });
782
860
  addUsage(res);
783
861
  if (!assistantText(res)) {
784
862
  throw emptyTurnError({
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "aegis-desktop",
3
3
  "productName": "AEGIS Desktop",
4
- "version": "0.4.2",
4
+ "version": "0.4.4",
5
5
  "description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
6
6
  "author": {
7
7
  "name": "AEGIS Code",
package/renderer/app.js CHANGED
@@ -171,6 +171,11 @@ const CUSTOM_MODEL_PRESETS = {
171
171
  // already filters out of settings.list(); never render it as a provider row
172
172
  // even if a stale store still surfaces it (defect #1).
173
173
  const RESERVED_PROVIDERS = new Set(['__aegis', 'aegis']);
174
+ // Where a user without a key gets one. The class picker defaults to Aegis Cloud
175
+ // and the catalog is key-gated, so this is the first thing a new install needs;
176
+ // it lives here rather than inline so the Model hint and any future "connect"
177
+ // affordance cannot drift to two different pages.
178
+ const GET_AEGIS_KEY_URL = 'https://aegiscloud.org';
174
179
 
175
180
  let pendingEl = null;
176
181
  let pendingSessionId = null;
@@ -1858,6 +1863,11 @@ async function loadModels(cls) {
1858
1863
  try {
1859
1864
  const data = await models.listModels(cls);
1860
1865
  const list = Array.isArray(data && data.models) ? data.models : [];
1866
+ // No key on the default class: the engine reports the missing credential as
1867
+ // a state instead of letting the catalog call 401 (see engine.listModels).
1868
+ // The hint is where a user finds out they can connect at all — a raw
1869
+ // "listModels failed" told them only that something was broken.
1870
+ const needsKey = Boolean(data && data.needsKey);
1861
1871
  for (const m of list) {
1862
1872
  modelMeta.set(m.id, m);
1863
1873
  const opt = document.createElement('option');
@@ -1866,12 +1876,28 @@ async function loadModels(cls) {
1866
1876
  els.modelSelect.appendChild(opt);
1867
1877
  }
1868
1878
  let hint;
1869
- if (!list.length) {
1879
+ if (needsKey) {
1880
+ hint = null; // carries a link, built below
1881
+ } else if (!list.length) {
1870
1882
  hint = cls === 'ollama' ? 'Ollama not running or no models pulled.' : 'No models listed.';
1871
1883
  } else {
1872
1884
  hint = `${list.length} model${list.length === 1 ? '' : 's'} available.`;
1873
1885
  }
1874
1886
  const ceiling = applyMaxTokensClamp(els.modelSelect.value);
1887
+ if (needsKey) {
1888
+ // The hint elements are bare <p>s, so the link has to be a real child
1889
+ // node — a text assignment would wipe it (same shape as capNotice).
1890
+ els.modelHint.textContent =
1891
+ 'Connect Aegis Cloud to load its models: paste your AEGIS API key in the Status card above (';
1892
+ const a = document.createElement('a');
1893
+ a.href = GET_AEGIS_KEY_URL;
1894
+ a.target = '_blank';
1895
+ a.rel = 'noreferrer noopener';
1896
+ a.textContent = 'free key at aegiscloud.org';
1897
+ els.modelHint.appendChild(a);
1898
+ els.modelHint.appendChild(document.createTextNode('), then Save.'));
1899
+ return;
1900
+ }
1875
1901
  els.modelHint.textContent =
1876
1902
  ceiling < FLAT_CEILING ? `${hint} · max output: ${ceiling.toLocaleString()}` : hint;
1877
1903
  } catch (err) {
package/vendor/aegis.js CHANGED
@@ -466,12 +466,15 @@ function createClient(opts = {}) {
466
466
  // close), reader.read() below waits forever and the whole desktop host
467
467
  // hangs with no way out but force-quit. Cap the gap between chunks —
468
468
  // not the whole response — so a slow-but-alive generation is untouched.
469
- const SSE_IDLE_TIMEOUT_MS = 60_000;
470
- // A pooled brain call may set its own, larger budget: the fan-out yields a
471
- // header chunk and then stays silent until the FIRST worker pass returns,
472
- // which is a full reasoning-model call and can outlast the 60s default.
473
- const idleMs =
474
- Number(idleTimeoutMs) > 0 ? Number(idleTimeoutMs) : SSE_IDLE_TIMEOUT_MS;
469
+ //
470
+ // The budget is per *response*, not per call site: a pooled brain call
471
+ // announces its worker fan-out in the X-AEGIS-Brain response header, and a
472
+ // fan-out is legitimately silent until its first worker returns. Keying the
473
+ // longer budget on a request the caller remembered to flag left the
474
+ // default Nexus turn (brain model id, checkbox off) dying at 60s — with
475
+ // the server already past its own fan-out deadline and every worker
476
+ // billed. See idleBudgetFor().
477
+ const idleMs = idleBudgetFor(res, idleTimeoutMs);
475
478
  async function readWithIdleTimeout() {
476
479
  let timer;
477
480
  const timeout = new Promise((_, reject) => {
@@ -752,7 +755,49 @@ function createClient(opts = {}) {
752
755
  };
753
756
  }
754
757
 
755
- const api = { createClient, envVar, randomUUID, DEFAULT_API_BASE, CLIENT_VERSION };
758
+ /** Gap between SSE chunks that counts as "the stream is dead", in ms. */
759
+ const SSE_IDLE_TIMEOUT_MS = 60_000;
760
+
761
+ /**
762
+ * Gap allowed while a pooled brain call runs its worker fan-out, in ms.
763
+ *
764
+ * The fan-out yields a header chunk and then says nothing until its FIRST
765
+ * worker pass returns — and each worker is a full reasoning-model call at
766
+ * roughly 1/(workers+1) of the effort budget, so the silent window is minutes,
767
+ * not seconds. The server bounds that window itself
768
+ * (``services/pool_brain.py`` ``NEXUS_BRAIN_WORKER_TIMEOUT``, default 600s) and
769
+ * announces it per-response with the ``X-AEGIS-Brain`` header, so this budget
770
+ * has to outlast the *server's* deadline: aborting first kills a healthy turn
771
+ * the server is still running and billing.
772
+ *
773
+ * Raising the server's env var above ~14 minutes requires raising this too.
774
+ * desktop/lib/local/engine.js mirrors the same value for the explicit
775
+ * "work autonomously" path (AUTONOMOUS_IDLE_TIMEOUT_MS); the desktop test
776
+ * test/autonomous-mode.test.mjs fails if either drops below the server default.
777
+ */
778
+ const BRAIN_IDLE_TIMEOUT_MS = 15 * 60_000;
779
+
780
+ /**
781
+ * Idle budget for one SSE response: the caller's override when it asked for
782
+ * one, widened to the fan-out budget when the server says this response *is* a
783
+ * fan-out. The header is the authoritative signal — it is set by the same code
784
+ * that runs the fan-out, so a renamed brain id or a caller that forgot its
785
+ * flag cannot desynchronise the two. A response with no header (an older
786
+ * server, or a single-pass call) keeps the caller's budget or the 60s default.
787
+ */
788
+ function idleBudgetFor(res, requestedMs) {
789
+ const base = Number(requestedMs) > 0 ? Number(requestedMs) : SSE_IDLE_TIMEOUT_MS;
790
+ let header = '';
791
+ try {
792
+ const get = res && res.headers && typeof res.headers.get === 'function' ? res.headers.get.bind(res.headers) : null;
793
+ header = (get && get('X-AEGIS-Brain')) || '';
794
+ } catch {
795
+ header = ''; // an exotic fetch shim without headers: keep the caller's budget
796
+ }
797
+ return header && String(header).trim() ? Math.max(base, BRAIN_IDLE_TIMEOUT_MS) : base;
798
+ }
799
+
800
+ const api = { createClient, envVar, randomUUID, DEFAULT_API_BASE, CLIENT_VERSION, idleBudgetFor, SSE_IDLE_TIMEOUT_MS, BRAIN_IDLE_TIMEOUT_MS };
756
801
 
757
802
  // Node / Electron (CommonJS): the MCP plugin and desktop shell require() this.
758
803
  if (typeof module !== 'undefined' && module.exports) {