aegis-desktop 0.4.4 → 0.4.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/local/engine.js +12 -7
- package/package.json +1 -1
- package/renderer/app.js +70 -10
- package/renderer/index.html +20 -9
- package/renderer/style.css +11 -0
- package/vendor/aegis.js +16 -1
package/lib/local/engine.js
CHANGED
|
@@ -584,13 +584,18 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
584
584
|
// model as the fan-out, one sample instead of four — otherwise
|
|
585
585
|
// dropping the fan-out would have quietly changed the model too.
|
|
586
586
|
...(brainFlag === false ? { mode: 'brain' } : {}),
|
|
587
|
-
//
|
|
588
|
-
//
|
|
589
|
-
//
|
|
590
|
-
//
|
|
591
|
-
//
|
|
592
|
-
//
|
|
593
|
-
|
|
587
|
+
// `effort` is the budget rung, and it is sent whenever the caller has
|
|
588
|
+
// one — not only alongside a fan-out. The fan-out is triggered by the
|
|
589
|
+
// model id (this class sends ``nexus-brain``), so a caller that never
|
|
590
|
+
// ticked "work autonomously" still ran a pooled call and had no way to
|
|
591
|
+
// say how big it should be: the server fell through to its own default
|
|
592
|
+
// rung. That is how a one-word prompt came to reserve the top of the
|
|
593
|
+
// ladder. A single-pass pass carries it too — inert on that path, but
|
|
594
|
+
// it keeps the request honest about what it asked for.
|
|
595
|
+
...(opts.effort ? { effort: opts.effort } : {}),
|
|
596
|
+
// Only meaningful with a running fan-out — aegis1 services/pool_brain.py
|
|
597
|
+
// parse_brain_request reads `workers` straight off the body and clamps
|
|
598
|
+
// it itself (MAX_WORKERS), so no client-side validation here.
|
|
594
599
|
...(brainFlag === true && opts.workers ? { workers: opts.workers } : {}),
|
|
595
600
|
// The pool forwards `tools` to the provider and returns tool_calls
|
|
596
601
|
// (aegis1 app.py:7765 → provider, pool_brain synthesis keeps them).
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "aegis-desktop",
|
|
3
3
|
"productName": "AEGIS Desktop",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.5",
|
|
5
5
|
"description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "AEGIS Code",
|
package/renderer/app.js
CHANGED
|
@@ -69,7 +69,10 @@ const ELEMENT_IDS = {
|
|
|
69
69
|
modelPreset: 'model-preset',
|
|
70
70
|
modelInput: 'model-input',
|
|
71
71
|
maxTokens: 'max-tokens',
|
|
72
|
+
maxTokensLabel: 'max-tokens-label',
|
|
73
|
+
maxTokensRow: 'max-tokens-row',
|
|
72
74
|
maxTokensAdaptive: 'max-tokens-adaptive',
|
|
75
|
+
effortRow: 'effort-row',
|
|
73
76
|
autonomousToggle: 'autonomous-toggle',
|
|
74
77
|
autonomousToggleWrap: 'autonomous-toggle-wrap',
|
|
75
78
|
autonomousControls: 'autonomous-controls',
|
|
@@ -351,6 +354,35 @@ function updateAutonomousControlsVisibility() {
|
|
|
351
354
|
els.autonomousControls.hidden = !(wrapVisible && autonomousEnabled());
|
|
352
355
|
}
|
|
353
356
|
|
|
357
|
+
/**
|
|
358
|
+
* Which budget control applies to the selected class.
|
|
359
|
+
*
|
|
360
|
+
* Aegis Cloud is sized server-side from `effort`; the other three classes take
|
|
361
|
+
* a per-call token ceiling from the dropdown. Showing both at once is what made
|
|
362
|
+
* the token cap untrustworthy on the pooled class — the dropdown was displayed,
|
|
363
|
+
* read on every send, and then raised by the server's effort ladder, so the
|
|
364
|
+
* number beside it was never the budget the call ran on. Exactly one control is
|
|
365
|
+
* on offer now, and it is the one the request actually travels with.
|
|
366
|
+
*/
|
|
367
|
+
function updateBudgetControls(cls) {
|
|
368
|
+
const pooled = cls === AUTONOMOUS_CLASS;
|
|
369
|
+
if (els.effortRow) els.effortRow.hidden = !pooled;
|
|
370
|
+
if (els.maxTokensRow) els.maxTokensRow.hidden = pooled;
|
|
371
|
+
if (els.maxTokensLabel) els.maxTokensLabel.hidden = pooled;
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* The effort to send, or undefined for the classes the server does not size
|
|
376
|
+
* from it. `auto` means "let the server infer it from the ask" — the same thing
|
|
377
|
+
* the server already does for a request that names no effort, and a deliberate
|
|
378
|
+
* choice rather than the old silent fall-through to the top rung.
|
|
379
|
+
*/
|
|
380
|
+
function effortFor(cls) {
|
|
381
|
+
if (cls !== AUTONOMOUS_CLASS) return undefined;
|
|
382
|
+
const value = els.autonomousEffort && els.autonomousEffort.value;
|
|
383
|
+
return value && value !== 'auto' ? value : undefined;
|
|
384
|
+
}
|
|
385
|
+
|
|
354
386
|
// ---------------------------------------------------------------- UI helpers
|
|
355
387
|
|
|
356
388
|
function setConn(ok, text) {
|
|
@@ -1769,6 +1801,7 @@ async function loadModels(cls) {
|
|
|
1769
1801
|
els.autonomousToggleWrap.hidden = cls !== AUTONOMOUS_CLASS;
|
|
1770
1802
|
}
|
|
1771
1803
|
updateAutonomousControlsVisibility();
|
|
1804
|
+
updateBudgetControls(cls);
|
|
1772
1805
|
|
|
1773
1806
|
const custom = CUSTOM_CLASSES.has(cls);
|
|
1774
1807
|
els.modelSelect.hidden = custom;
|
|
@@ -2274,12 +2307,21 @@ async function send() {
|
|
|
2274
2307
|
addMessage('user', prompt);
|
|
2275
2308
|
|
|
2276
2309
|
const ceiling = applyMaxTokensClamp(model);
|
|
2277
|
-
|
|
2310
|
+
// Aegis Cloud takes no token cap from here at all: the server sizes the call
|
|
2311
|
+
// from `effort`, and a number in this position is a per-pass ceiling *over*
|
|
2312
|
+
// that ladder (aegis1 services/pool_brain.py pass_budgets). Sending the
|
|
2313
|
+
// dropdown's value anyway is what made "Max tokens: 4k" beside a turn a
|
|
2314
|
+
// figure the turn never ran on. The row is hidden for this class as well, so
|
|
2315
|
+
// the two controls can never disagree.
|
|
2316
|
+
const maxTokens = cls === AUTONOMOUS_CLASS
|
|
2317
|
+
? undefined
|
|
2318
|
+
: maxTokensAdaptive() ? ceiling : parseInt(els.maxTokens.value, 10) || 4096;
|
|
2278
2319
|
const autonomous = cls === AUTONOMOUS_CLASS && autonomousEnabled();
|
|
2279
|
-
//
|
|
2280
|
-
//
|
|
2281
|
-
//
|
|
2282
|
-
|
|
2320
|
+
// Sent for the pooled class whether or not the fan-out is ticked: the fan-out
|
|
2321
|
+
// is enabled by the model id this class sends, so a turn that never entered
|
|
2322
|
+
// autonomous mode still ran pooled and had no way to say how big it should
|
|
2323
|
+
// be. `undefined` = "auto" = the server infers it from the ask.
|
|
2324
|
+
const effort = effortFor(cls);
|
|
2283
2325
|
const workers = autonomous ? parseInt(els.autonomousWorkers.value, 10) || undefined : undefined;
|
|
2284
2326
|
// Reuse the open thread's session id (minted once, on its first message)
|
|
2285
2327
|
// instead of a fresh one per send — a new id every turn is what made both
|
|
@@ -2378,7 +2420,11 @@ async function send() {
|
|
|
2378
2420
|
if (data && data.model) bits.push(`model: ${data.model}`);
|
|
2379
2421
|
else if (model) bits.push(`model: ${model}`);
|
|
2380
2422
|
bits.push(classLabel(cls));
|
|
2381
|
-
if (autonomous) bits.push(`autonomous (${effort}, ${workers ||
|
|
2423
|
+
if (autonomous) bits.push(`autonomous (${effort || 'auto'}, ${workers || 'auto'}w)`);
|
|
2424
|
+
// The budget rung is worth showing even without the fan-out: on the pooled
|
|
2425
|
+
// class it is what sized the call, and a user cannot tell a 16k turn from a
|
|
2426
|
+
// 64k one by looking at the answer.
|
|
2427
|
+
else if (effort) bits.push(`effort: ${effort}`);
|
|
2382
2428
|
const turnTokens = usageTokens(data && data.usage);
|
|
2383
2429
|
if (turnTokens != null) bits.push(`tokens: ${turnTokens}`);
|
|
2384
2430
|
addMessage('assistant', text, bits.join(' · ') || undefined, sessionId, toolLog);
|
|
@@ -2459,14 +2505,23 @@ async function init() {
|
|
|
2459
2505
|
updateAutonomousControlsVisibility();
|
|
2460
2506
|
});
|
|
2461
2507
|
|
|
2462
|
-
//
|
|
2463
|
-
//
|
|
2464
|
-
//
|
|
2508
|
+
// Budget control for the pooled class: which rung of the server's effort
|
|
2509
|
+
// ladder sizes the call, or "auto" to let the server infer it from the ask
|
|
2510
|
+
// (aegis1 services/pool_brain.py parse_brain_request). Opt-in per machine,
|
|
2511
|
+
// remembered across restarts; a stored value from a build whose list was
|
|
2512
|
+
// short (low/medium/high, no "auto") falls through to the markup's default
|
|
2513
|
+
// rather than assigning an option that no longer exists.
|
|
2465
2514
|
const savedEffort = localStorage.getItem(AUTONOMOUS_EFFORT_KEY);
|
|
2466
|
-
if (savedEffort
|
|
2515
|
+
if (savedEffort && Array.from(els.autonomousEffort.options).some((o) => o.value === savedEffort)) {
|
|
2516
|
+
els.autonomousEffort.value = savedEffort;
|
|
2517
|
+
}
|
|
2467
2518
|
els.autonomousEffort.addEventListener('change', () => {
|
|
2468
2519
|
localStorage.setItem(AUTONOMOUS_EFFORT_KEY, els.autonomousEffort.value);
|
|
2469
2520
|
});
|
|
2521
|
+
// Worker count for the fan-out. Left empty by default on purpose: an empty
|
|
2522
|
+
// field is what tells the server to size the fan-out from the ask
|
|
2523
|
+
// (parse_brain_request's auto path) instead of the old client-side default of
|
|
2524
|
+
// 3, which made every turn a 4-pass fan-out.
|
|
2470
2525
|
const savedWorkers = localStorage.getItem(AUTONOMOUS_WORKERS_KEY);
|
|
2471
2526
|
if (savedWorkers) els.autonomousWorkers.value = savedWorkers;
|
|
2472
2527
|
els.autonomousWorkers.addEventListener('change', () => {
|
|
@@ -2484,6 +2539,11 @@ async function init() {
|
|
|
2484
2539
|
|
|
2485
2540
|
els.classSelect.addEventListener('change', () => {
|
|
2486
2541
|
localStorage.setItem(CLASS_KEY, els.classSelect.value);
|
|
2542
|
+
// Apply the budget-control choice immediately rather than waiting for the
|
|
2543
|
+
// model list: loadModels() is async, and until it resolves the previous
|
|
2544
|
+
// class's control would still be on screen — showing a token cap on a
|
|
2545
|
+
// pooled turn, or hiding the effort rung it runs on.
|
|
2546
|
+
updateBudgetControls(els.classSelect.value);
|
|
2487
2547
|
loadModels(els.classSelect.value);
|
|
2488
2548
|
});
|
|
2489
2549
|
|
package/renderer/index.html
CHANGED
|
@@ -76,8 +76,8 @@
|
|
|
76
76
|
hidden
|
|
77
77
|
/>
|
|
78
78
|
|
|
79
|
-
<label for="max-tokens">Max tokens</label>
|
|
80
|
-
<div class="max-tokens-row">
|
|
79
|
+
<label for="max-tokens" id="max-tokens-label">Max tokens</label>
|
|
80
|
+
<div class="max-tokens-row" id="max-tokens-row">
|
|
81
81
|
<select id="max-tokens">
|
|
82
82
|
<option value="1024">1k</option>
|
|
83
83
|
<option value="4096" selected>4k</option>
|
|
@@ -95,6 +95,23 @@
|
|
|
95
95
|
</label>
|
|
96
96
|
</div>
|
|
97
97
|
|
|
98
|
+
<!-- The pooled AEGIS Cloud class sizes a request's token budget
|
|
99
|
+
server-side from this ladder (effort -> low/medium/high ->
|
|
100
|
+
16384/32768/65536 total across the fan-out), so the max-tokens row
|
|
101
|
+
above is hidden for that class instead of being shown and silently
|
|
102
|
+
overridden: the server treats a body max_tokens as a ceiling, and
|
|
103
|
+
an effort rung as the budget. Shown for Ollama and custom
|
|
104
|
+
endpoints, where a per-call ceiling is the only control there is. -->
|
|
105
|
+
<div class="max-tokens-row" id="effort-row" hidden>
|
|
106
|
+
<label for="autonomous-effort">Effort</label>
|
|
107
|
+
<select id="autonomous-effort">
|
|
108
|
+
<option value="low">low</option>
|
|
109
|
+
<option value="medium">medium</option>
|
|
110
|
+
<option value="high">high</option>
|
|
111
|
+
<option value="auto" selected>auto</option>
|
|
112
|
+
</select>
|
|
113
|
+
</div>
|
|
114
|
+
|
|
98
115
|
<label
|
|
99
116
|
class="flow-toggle"
|
|
100
117
|
for="autonomous-toggle"
|
|
@@ -106,14 +123,8 @@
|
|
|
106
123
|
<span>work autonomously</span>
|
|
107
124
|
</label>
|
|
108
125
|
<div id="autonomous-controls" class="autonomous-controls" hidden>
|
|
109
|
-
<label for="autonomous-effort">Effort</label>
|
|
110
|
-
<select id="autonomous-effort">
|
|
111
|
-
<option value="low">low</option>
|
|
112
|
-
<option value="medium">medium</option>
|
|
113
|
-
<option value="high" selected>high</option>
|
|
114
|
-
</select>
|
|
115
126
|
<label for="autonomous-workers">Workers</label>
|
|
116
|
-
<input type="number" id="autonomous-workers" min="1" max="8" step="1"
|
|
127
|
+
<input type="number" id="autonomous-workers" min="1" max="8" step="1" />
|
|
117
128
|
</div>
|
|
118
129
|
<p class="hint" id="model-hint"></p>
|
|
119
130
|
</section>
|
package/renderer/style.css
CHANGED
|
@@ -1142,6 +1142,17 @@ body.memory-open {
|
|
|
1142
1142
|
display: none;
|
|
1143
1143
|
}
|
|
1144
1144
|
|
|
1145
|
+
/* Same trap, one row up: `.max-tokens-row` is display: flex, so the `hidden`
|
|
1146
|
+
attribute did nothing on `#max-tokens-row`/`#effort-row` — which is how the
|
|
1147
|
+
budget-control swap (effort for the pooled class, a token ceiling for the
|
|
1148
|
+
others) would have shown BOTH controls at once, the exact state it exists to
|
|
1149
|
+
prevent. `display: none` is also what lets a `<label>` be hidden, since
|
|
1150
|
+
labels are inline by default and carry no author display rule. */
|
|
1151
|
+
.max-tokens-row[hidden],
|
|
1152
|
+
label[hidden] {
|
|
1153
|
+
display: none;
|
|
1154
|
+
}
|
|
1155
|
+
|
|
1145
1156
|
.autonomous-controls {
|
|
1146
1157
|
display: flex;
|
|
1147
1158
|
align-items: center;
|
package/vendor/aegis.js
CHANGED
|
@@ -278,6 +278,13 @@ function createClient(opts = {}) {
|
|
|
278
278
|
* so a reasoning-only stream still counts as unanswered.
|
|
279
279
|
* - `idleTimeoutMs` override the stalled-stream watchdog for this call
|
|
280
280
|
* (a worker fan-out is legitimately silent between passes)
|
|
281
|
+
* - `maxTokens` output ceiling. Omitted from the body when absent so the
|
|
282
|
+
* server's own budget applies — on a pooled request that is
|
|
283
|
+
* the `effort` ladder, and a number sent from here caps it.
|
|
284
|
+
* - `effort` / `workers` pooled-brain budget rung and fan-out size.
|
|
285
|
+
* Forwarded only inside `extra`, as the server reads them off
|
|
286
|
+
* the body (aegis1 services/pool_brain.py parse_brain_request)
|
|
287
|
+
* and sizes the budget from them.
|
|
281
288
|
*/
|
|
282
289
|
async function chatCompletion({
|
|
283
290
|
prompt,
|
|
@@ -298,9 +305,17 @@ function createClient(opts = {}) {
|
|
|
298
305
|
// model the server picks its default (no client-invented tier id). `mode`
|
|
299
306
|
// is a legacy server-side shorthand — forwarded verbatim only when the
|
|
300
307
|
// caller supplies it, never defaulted, never built into a model id.
|
|
308
|
+
//
|
|
309
|
+
// `max_tokens` is omitted entirely when the caller states none, rather than
|
|
310
|
+
// defaulted here. The server has its own budget ladder (`mode`/`effort` on
|
|
311
|
+
// the pooled path) and treats a body max_tokens as a *ceiling* over it, so
|
|
312
|
+
// an invented 4096 was not a harmless default: it capped every pass of a
|
|
313
|
+
// pooled-brain fan-out at 4096 and overrode the ladder the caller's effort
|
|
314
|
+
// selected. Omitting it is the documented way to say "server default" (see
|
|
315
|
+
// mcp/tools.js's max_tokens description).
|
|
301
316
|
const body = {
|
|
302
317
|
messages: buildMessages(messages, system, prompt),
|
|
303
|
-
max_tokens: maxTokens
|
|
318
|
+
...(maxTokens ? { max_tokens: maxTokens } : {}),
|
|
304
319
|
...(extra || {}),
|
|
305
320
|
};
|
|
306
321
|
if (model) body.model = model;
|