aegis-desktop 0.4.2 → 0.4.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/local/engine.js +86 -8
- package/package.json +1 -1
- package/renderer/app.js +27 -1
- package/vendor/aegis.js +52 -7
package/lib/local/engine.js
CHANGED
|
@@ -96,8 +96,17 @@ const EFFORT_TOKEN_BUDGET = { low: 8192, medium: 16384, high: 32768 };
|
|
|
96
96
|
* that is a multi-thousand-token reasoning pass per worker — easily past a
|
|
97
97
|
* minute. Timing out there aborts a perfectly healthy autonomous turn
|
|
98
98
|
* mid-flight, after the server has already run and billed every worker.
|
|
99
|
+
*
|
|
100
|
+
* 15 minutes deliberately outlasts the server's OWN ceiling for that window
|
|
101
|
+
* (aegis1 services/pool_brain.py: NEXUS_BRAIN_WORKER_TIMEOUT, default 600s),
|
|
102
|
+
* because aborting first leaves the server running and billing a fan-out
|
|
103
|
+
* nobody will ever see. Raising that env var past ~14 minutes means raising
|
|
104
|
+
* this constant too; the shared client applies the same budget from the
|
|
105
|
+
* server's X-AEGIS-Brain response header (see client/aegis.js idleBudgetFor),
|
|
106
|
+
* which is what covers a fan-out the caller did not flag — test/
|
|
107
|
+
* autonomous-mode.test.mjs pins both halves.
|
|
99
108
|
*/
|
|
100
|
-
const AUTONOMOUS_IDLE_TIMEOUT_MS =
|
|
109
|
+
const AUTONOMOUS_IDLE_TIMEOUT_MS = 15 * 60_000;
|
|
101
110
|
|
|
102
111
|
/**
|
|
103
112
|
* Only ever raises a too-low budget for a DeepSeek reasoning model — never
|
|
@@ -464,6 +473,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
464
473
|
|
|
465
474
|
async function listModels(cls) {
|
|
466
475
|
if (cls === 'aegis') {
|
|
476
|
+
// The catalog is behind the account's key: GET /api/v1/models answers
|
|
477
|
+
// `401 {"error":{"message":"No API key"}}` without one (verified against
|
|
478
|
+
// aegiscloud.org). Aegis Cloud is the *default* class, so firing the call
|
|
479
|
+
// anyway painted a raw "listModels failed: … 401" over the default model
|
|
480
|
+
// picker for every user who had just installed the app and not yet
|
|
481
|
+
// pasted a key — the one state where the UI must say what unblocks it and
|
|
482
|
+
// not what went wrong. Report the missing key as a state (`needsKey`) and
|
|
483
|
+
// let the renderer invite the user to connect; nothing else about the
|
|
484
|
+
// class changes, and the model dropdown keeps its "server default (auto)"
|
|
485
|
+
// entry so the class is usable the moment a key lands.
|
|
486
|
+
if (!aegis.apiKey) return { class: cls, models: [], needsKey: true };
|
|
467
487
|
const data = await aegis.listModels();
|
|
468
488
|
return { class: cls, models: filterAegisCatalog(normalizeCatalog(data && data.models)) };
|
|
469
489
|
}
|
|
@@ -513,6 +533,12 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
513
533
|
/** One transport round for the chosen class. */
|
|
514
534
|
async function dispatch(cls, opts) {
|
|
515
535
|
if (cls === 'aegis') {
|
|
536
|
+
// `undefined` = "leave the model id's own default alone" (a real user
|
|
537
|
+
// turn on the pooled class: the selected Nexus id is what decides, so
|
|
538
|
+
// today's behaviour is unchanged). `false` = an explicit single provider
|
|
539
|
+
// call, for a pass that is a continuation rather than a new
|
|
540
|
+
// investigation. `true` = the autonomous fan-out.
|
|
541
|
+
const brainFlag = opts.singlePass ? false : opts.autonomous ? true : undefined;
|
|
516
542
|
return aegis.chatCompletion({
|
|
517
543
|
prompt: opts.prompt,
|
|
518
544
|
system: opts.system,
|
|
@@ -520,7 +546,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
520
546
|
model: opts.model,
|
|
521
547
|
mode: opts.mode,
|
|
522
548
|
maxTokens: opts.maxTokens,
|
|
523
|
-
stream:
|
|
549
|
+
stream: opts.stream !== false,
|
|
524
550
|
// The pooled (Nexus) brain is streamed, and an OpenAI-compatible SSE
|
|
525
551
|
// stream reports no token usage unless asked. Without this the Aegis
|
|
526
552
|
// Cloud class — the desktop's default — was the one class that answered
|
|
@@ -542,13 +568,30 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
542
568
|
extra: {
|
|
543
569
|
aegis_memory: true,
|
|
544
570
|
session: opts.sessionId,
|
|
545
|
-
|
|
546
|
-
//
|
|
571
|
+
// The fan-out is opt-in per dispatch. `brain` is sent EXPLICITLY
|
|
572
|
+
// whenever this dispatch is not the autonomous one, because the
|
|
573
|
+
// model id this class sends (``nexus-brain``) enables the pooled
|
|
574
|
+
// brain on its own: without the flag a continuation pass — the
|
|
575
|
+
// doubled-budget retry, or the "write up what you already found"
|
|
576
|
+
// re-dispatch — silently re-ran the whole workers+1 fan-out for a
|
|
577
|
+
// pass whose documented cost is a single request. aegis1
|
|
578
|
+
// services/pool_brain.py parse_brain_request honours the opt-out.
|
|
579
|
+
...(brainFlag === undefined ? {} : { brain: brainFlag }),
|
|
580
|
+
// An opted-out pass is also told WHICH band to run on. A single call
|
|
581
|
+
// on a brain model id infers its band from the id and lands on
|
|
582
|
+
// "fast" (the cheapest id in the pool); "brain" is the band the
|
|
583
|
+
// workers themselves run on (cheapest model that can think). Same
|
|
584
|
+
// model as the fan-out, one sample instead of four — otherwise
|
|
585
|
+
// dropping the fan-out would have quietly changed the model too.
|
|
586
|
+
...(brainFlag === false ? { mode: 'brain' } : {}),
|
|
587
|
+
// Only meaningful (and only sent) alongside a running fan-out — aegis1
|
|
547
588
|
// services/pool_brain.py parse_brain_request reads `effort`/
|
|
548
589
|
// `workers` straight off the body and clamps them itself
|
|
549
590
|
// (EFFORT_LEVELS / MAX_WORKERS), so no client-side validation here.
|
|
550
|
-
|
|
551
|
-
|
|
591
|
+
// Keyed on the effective brain flag, not on `autonomous`: an opted-out
|
|
592
|
+
// single pass carries no fan-out tuning it cannot use.
|
|
593
|
+
...(brainFlag === true && opts.effort ? { effort: opts.effort } : {}),
|
|
594
|
+
...(brainFlag === true && opts.workers ? { workers: opts.workers } : {}),
|
|
552
595
|
// The pool forwards `tools` to the provider and returns tool_calls
|
|
553
596
|
// (aegis1 app.py:7765 → provider, pool_brain synthesis keeps them).
|
|
554
597
|
...(opts.tools.length ? { tools: opts.tools } : {}),
|
|
@@ -667,6 +710,11 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
667
710
|
workers: payload && payload.workers,
|
|
668
711
|
onReasoning,
|
|
669
712
|
idleTimeoutMs: autonomous ? AUTONOMOUS_IDLE_TIMEOUT_MS : undefined,
|
|
713
|
+
// A caller with no live streaming surface (a `--no-stream` CLI flag, a
|
|
714
|
+
// one-shot script) can ask for the buffered non-stream wire form
|
|
715
|
+
// instead. Undefined/anything but `false` keeps every existing caller
|
|
716
|
+
// (the desktop renderer never sets this) on the streamed path.
|
|
717
|
+
stream: payload && payload.stream === false ? false : true,
|
|
670
718
|
};
|
|
671
719
|
|
|
672
720
|
// No round cap: a model that keeps calling tools keeps going for as
|
|
@@ -750,6 +798,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
750
798
|
}
|
|
751
799
|
addUsage(res);
|
|
752
800
|
|
|
801
|
+
// A cancelled turn is over. Both recoveries below exist for "the model
|
|
802
|
+
// said nothing", and an aborted request resolves with exactly that —
|
|
803
|
+
// no text — so without this guard a user pressing Esc to stop a slow
|
|
804
|
+
// turn immediately issued ANOTHER billed provider call (and, because
|
|
805
|
+
// every transport concatenates the deltas it forwards, streamed the
|
|
806
|
+
// partial answer a second time onto the same bubble). Measured before
|
|
807
|
+
// the guard: one interrupted turn == two dispatches.
|
|
808
|
+
if (signal.aborted) return withTurnUsage(res);
|
|
809
|
+
|
|
753
810
|
// Budget exhausted before the answer was written. Doubling it costs
|
|
754
811
|
// one request and converts a dead turn into a real one; a second
|
|
755
812
|
// 'length' result is accepted as-is so a hard-capped model can't
|
|
@@ -763,7 +820,16 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
763
820
|
// would stream it a second time onto the same bubble.
|
|
764
821
|
if (!truncationRetried && !assistantText(res) && isTruncated(res)) {
|
|
765
822
|
truncationRetried = true;
|
|
766
|
-
|
|
823
|
+
// singlePass: this retry buys *budget*, not a second investigation.
|
|
824
|
+
// The fan-out's workers would re-run the whole task from scratch for
|
|
825
|
+
// it — 3 extra reasoning passes + a synthesis — which is the opposite
|
|
826
|
+
// of what "one doubled request" means (and what the note above
|
|
827
|
+
// promises). The pass itself is unchanged apart from that.
|
|
828
|
+
res = await dispatch(cls, {
|
|
829
|
+
...opts,
|
|
830
|
+
singlePass: true,
|
|
831
|
+
maxTokens: doubledBudget(opts.maxTokens),
|
|
832
|
+
});
|
|
767
833
|
addUsage(res);
|
|
768
834
|
}
|
|
769
835
|
|
|
@@ -778,7 +844,19 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
778
844
|
synthesisDone = true;
|
|
779
845
|
foldPromptIntoHistory();
|
|
780
846
|
history.push({ role: 'user', content: EMPTY_TURN_NUDGE });
|
|
781
|
-
|
|
847
|
+
// singlePass: same reasoning as the truncation retry above, and it is
|
|
848
|
+
// the same comment's literal promise ("force the summary out of the
|
|
849
|
+
// context it already holds"). Escalating a write-up back into the
|
|
850
|
+
// worker fan-out asked three fresh workers to redo an investigation
|
|
851
|
+
// whose findings are already in `history`, at 4x the cost, to produce
|
|
852
|
+
// a paragraph the model had all the material for.
|
|
853
|
+
res = await dispatch(cls, {
|
|
854
|
+
...opts,
|
|
855
|
+
singlePass: true,
|
|
856
|
+
messages: history,
|
|
857
|
+
prompt: '',
|
|
858
|
+
tools: [],
|
|
859
|
+
});
|
|
782
860
|
addUsage(res);
|
|
783
861
|
if (!assistantText(res)) {
|
|
784
862
|
throw emptyTurnError({
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "aegis-desktop",
|
|
3
3
|
"productName": "AEGIS Desktop",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.4",
|
|
5
5
|
"description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "AEGIS Code",
|
package/renderer/app.js
CHANGED
|
@@ -171,6 +171,11 @@ const CUSTOM_MODEL_PRESETS = {
|
|
|
171
171
|
// already filters out of settings.list(); never render it as a provider row
|
|
172
172
|
// even if a stale store still surfaces it (defect #1).
|
|
173
173
|
const RESERVED_PROVIDERS = new Set(['__aegis', 'aegis']);
|
|
174
|
+
// Where a user without a key gets one. The class picker defaults to Aegis Cloud
|
|
175
|
+
// and the catalog is key-gated, so this is the first thing a new install needs;
|
|
176
|
+
// it lives here rather than inline so the Model hint and any future "connect"
|
|
177
|
+
// affordance cannot drift to two different pages.
|
|
178
|
+
const GET_AEGIS_KEY_URL = 'https://aegiscloud.org';
|
|
174
179
|
|
|
175
180
|
let pendingEl = null;
|
|
176
181
|
let pendingSessionId = null;
|
|
@@ -1858,6 +1863,11 @@ async function loadModels(cls) {
|
|
|
1858
1863
|
try {
|
|
1859
1864
|
const data = await models.listModels(cls);
|
|
1860
1865
|
const list = Array.isArray(data && data.models) ? data.models : [];
|
|
1866
|
+
// No key on the default class: the engine reports the missing credential as
|
|
1867
|
+
// a state instead of letting the catalog call 401 (see engine.listModels).
|
|
1868
|
+
// The hint is where a user finds out they can connect at all — a raw
|
|
1869
|
+
// "listModels failed" told them only that something was broken.
|
|
1870
|
+
const needsKey = Boolean(data && data.needsKey);
|
|
1861
1871
|
for (const m of list) {
|
|
1862
1872
|
modelMeta.set(m.id, m);
|
|
1863
1873
|
const opt = document.createElement('option');
|
|
@@ -1866,12 +1876,28 @@ async function loadModels(cls) {
|
|
|
1866
1876
|
els.modelSelect.appendChild(opt);
|
|
1867
1877
|
}
|
|
1868
1878
|
let hint;
|
|
1869
|
-
if (
|
|
1879
|
+
if (needsKey) {
|
|
1880
|
+
hint = null; // carries a link, built below
|
|
1881
|
+
} else if (!list.length) {
|
|
1870
1882
|
hint = cls === 'ollama' ? 'Ollama not running or no models pulled.' : 'No models listed.';
|
|
1871
1883
|
} else {
|
|
1872
1884
|
hint = `${list.length} model${list.length === 1 ? '' : 's'} available.`;
|
|
1873
1885
|
}
|
|
1874
1886
|
const ceiling = applyMaxTokensClamp(els.modelSelect.value);
|
|
1887
|
+
if (needsKey) {
|
|
1888
|
+
// The hint elements are bare <p>s, so the link has to be a real child
|
|
1889
|
+
// node — a text assignment would wipe it (same shape as capNotice).
|
|
1890
|
+
els.modelHint.textContent =
|
|
1891
|
+
'Connect Aegis Cloud to load its models: paste your AEGIS API key in the Status card above (';
|
|
1892
|
+
const a = document.createElement('a');
|
|
1893
|
+
a.href = GET_AEGIS_KEY_URL;
|
|
1894
|
+
a.target = '_blank';
|
|
1895
|
+
a.rel = 'noreferrer noopener';
|
|
1896
|
+
a.textContent = 'free key at aegiscloud.org';
|
|
1897
|
+
els.modelHint.appendChild(a);
|
|
1898
|
+
els.modelHint.appendChild(document.createTextNode('), then Save.'));
|
|
1899
|
+
return;
|
|
1900
|
+
}
|
|
1875
1901
|
els.modelHint.textContent =
|
|
1876
1902
|
ceiling < FLAT_CEILING ? `${hint} · max output: ${ceiling.toLocaleString()}` : hint;
|
|
1877
1903
|
} catch (err) {
|
package/vendor/aegis.js
CHANGED
|
@@ -466,12 +466,15 @@ function createClient(opts = {}) {
|
|
|
466
466
|
// close), reader.read() below waits forever and the whole desktop host
|
|
467
467
|
// hangs with no way out but force-quit. Cap the gap between chunks —
|
|
468
468
|
// not the whole response — so a slow-but-alive generation is untouched.
|
|
469
|
-
|
|
470
|
-
//
|
|
471
|
-
//
|
|
472
|
-
//
|
|
473
|
-
|
|
474
|
-
|
|
469
|
+
//
|
|
470
|
+
// The budget is per *response*, not per call site: a pooled brain call
|
|
471
|
+
// announces its worker fan-out in the X-AEGIS-Brain response header, and a
|
|
472
|
+
// fan-out is legitimately silent until its first worker returns. Keying the
|
|
473
|
+
// longer budget on a request the caller remembered to flag left the
|
|
474
|
+
// default Nexus turn (brain model id, checkbox off) dying at 60s — with
|
|
475
|
+
// the server already past its own fan-out deadline and every worker
|
|
476
|
+
// billed. See idleBudgetFor().
|
|
477
|
+
const idleMs = idleBudgetFor(res, idleTimeoutMs);
|
|
475
478
|
async function readWithIdleTimeout() {
|
|
476
479
|
let timer;
|
|
477
480
|
const timeout = new Promise((_, reject) => {
|
|
@@ -752,7 +755,49 @@ function createClient(opts = {}) {
|
|
|
752
755
|
};
|
|
753
756
|
}
|
|
754
757
|
|
|
755
|
-
|
|
758
|
+
/** Gap between SSE chunks that counts as "the stream is dead", in ms. */
|
|
759
|
+
const SSE_IDLE_TIMEOUT_MS = 60_000;
|
|
760
|
+
|
|
761
|
+
/**
|
|
762
|
+
* Gap allowed while a pooled brain call runs its worker fan-out, in ms.
|
|
763
|
+
*
|
|
764
|
+
* The fan-out yields a header chunk and then says nothing until its FIRST
|
|
765
|
+
* worker pass returns — and each worker is a full reasoning-model call at
|
|
766
|
+
* roughly 1/(workers+1) of the effort budget, so the silent window is minutes,
|
|
767
|
+
* not seconds. The server bounds that window itself
|
|
768
|
+
* (``services/pool_brain.py`` ``NEXUS_BRAIN_WORKER_TIMEOUT``, default 600s) and
|
|
769
|
+
* announces it per-response with the ``X-AEGIS-Brain`` header, so this budget
|
|
770
|
+
* has to outlast the *server's* deadline: aborting first kills a healthy turn
|
|
771
|
+
* the server is still running and billing.
|
|
772
|
+
*
|
|
773
|
+
* Raising the server's env var above ~14 minutes requires raising this too.
|
|
774
|
+
* desktop/lib/local/engine.js mirrors the same value for the explicit
|
|
775
|
+
* "work autonomously" path (AUTONOMOUS_IDLE_TIMEOUT_MS); the desktop test
|
|
776
|
+
* test/autonomous-mode.test.mjs fails if either drops below the server default.
|
|
777
|
+
*/
|
|
778
|
+
const BRAIN_IDLE_TIMEOUT_MS = 15 * 60_000;
|
|
779
|
+
|
|
780
|
+
/**
|
|
781
|
+
* Idle budget for one SSE response: the caller's override when it asked for
|
|
782
|
+
* one, widened to the fan-out budget when the server says this response *is* a
|
|
783
|
+
* fan-out. The header is the authoritative signal — it is set by the same code
|
|
784
|
+
* that runs the fan-out, so a renamed brain id or a caller that forgot its
|
|
785
|
+
* flag cannot desynchronise the two. A response with no header (an older
|
|
786
|
+
* server, or a single-pass call) keeps the caller's budget or the 60s default.
|
|
787
|
+
*/
|
|
788
|
+
function idleBudgetFor(res, requestedMs) {
|
|
789
|
+
const base = Number(requestedMs) > 0 ? Number(requestedMs) : SSE_IDLE_TIMEOUT_MS;
|
|
790
|
+
let header = '';
|
|
791
|
+
try {
|
|
792
|
+
const get = res && res.headers && typeof res.headers.get === 'function' ? res.headers.get.bind(res.headers) : null;
|
|
793
|
+
header = (get && get('X-AEGIS-Brain')) || '';
|
|
794
|
+
} catch {
|
|
795
|
+
header = ''; // an exotic fetch shim without headers: keep the caller's budget
|
|
796
|
+
}
|
|
797
|
+
return header && String(header).trim() ? Math.max(base, BRAIN_IDLE_TIMEOUT_MS) : base;
|
|
798
|
+
}
|
|
799
|
+
|
|
800
|
+
const api = { createClient, envVar, randomUUID, DEFAULT_API_BASE, CLIENT_VERSION, idleBudgetFor, SSE_IDLE_TIMEOUT_MS, BRAIN_IDLE_TIMEOUT_MS };
|
|
756
801
|
|
|
757
802
|
// Node / Electron (CommonJS): the MCP plugin and desktop shell require() this.
|
|
758
803
|
if (typeof module !== 'undefined' && module.exports) {
|