aegis-desktop 0.4.3 → 0.4.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/local/engine.js +38 -8
- package/package.json +1 -1
- package/renderer/app.js +97 -11
- package/renderer/index.html +20 -9
- package/renderer/style.css +11 -0
- package/vendor/aegis.js +16 -1
package/lib/local/engine.js
CHANGED
|
@@ -473,6 +473,17 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
473
473
|
|
|
474
474
|
async function listModels(cls) {
|
|
475
475
|
if (cls === 'aegis') {
|
|
476
|
+
// The catalog is behind the account's key: GET /api/v1/models answers
|
|
477
|
+
// `401 {"error":{"message":"No API key"}}` without one (verified against
|
|
478
|
+
// aegiscloud.org). Aegis Cloud is the *default* class, so firing the call
|
|
479
|
+
// anyway painted a raw "listModels failed: … 401" over the default model
|
|
480
|
+
// picker for every user who had just installed the app and not yet
|
|
481
|
+
// pasted a key — the one state where the UI must say what unblocks it and
|
|
482
|
+
// not what went wrong. Report the missing key as a state (`needsKey`) and
|
|
483
|
+
// let the renderer invite the user to connect; nothing else about the
|
|
484
|
+
// class changes, and the model dropdown keeps its "server default (auto)"
|
|
485
|
+
// entry so the class is usable the moment a key lands.
|
|
486
|
+
if (!aegis.apiKey) return { class: cls, models: [], needsKey: true };
|
|
476
487
|
const data = await aegis.listModels();
|
|
477
488
|
return { class: cls, models: filterAegisCatalog(normalizeCatalog(data && data.models)) };
|
|
478
489
|
}
|
|
@@ -535,7 +546,7 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
535
546
|
model: opts.model,
|
|
536
547
|
mode: opts.mode,
|
|
537
548
|
maxTokens: opts.maxTokens,
|
|
538
|
-
stream:
|
|
549
|
+
stream: opts.stream !== false,
|
|
539
550
|
// The pooled (Nexus) brain is streamed, and an OpenAI-compatible SSE
|
|
540
551
|
// stream reports no token usage unless asked. Without this the Aegis
|
|
541
552
|
// Cloud class — the desktop's default — was the one class that answered
|
|
@@ -573,13 +584,18 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
573
584
|
// model as the fan-out, one sample instead of four — otherwise
|
|
574
585
|
// dropping the fan-out would have quietly changed the model too.
|
|
575
586
|
...(brainFlag === false ? { mode: 'brain' } : {}),
|
|
576
|
-
//
|
|
577
|
-
//
|
|
578
|
-
//
|
|
579
|
-
//
|
|
580
|
-
//
|
|
581
|
-
//
|
|
582
|
-
|
|
587
|
+
// `effort` is the budget rung, and it is sent whenever the caller has
|
|
588
|
+
// one — not only alongside a fan-out. The fan-out is triggered by the
|
|
589
|
+
// model id (this class sends ``nexus-brain``), so a caller that never
|
|
590
|
+
// ticked "work autonomously" still ran a pooled call and had no way to
|
|
591
|
+
// say how big it should be: the server fell through to its own default
|
|
592
|
+
// rung. That is how a one-word prompt came to reserve the top of the
|
|
593
|
+
// ladder. A single-pass pass carries it too — inert on that path, but
|
|
594
|
+
// it keeps the request honest about what it asked for.
|
|
595
|
+
...(opts.effort ? { effort: opts.effort } : {}),
|
|
596
|
+
// Only meaningful with a running fan-out — aegis1 services/pool_brain.py
|
|
597
|
+
// parse_brain_request reads `workers` straight off the body and clamps
|
|
598
|
+
// it itself (MAX_WORKERS), so no client-side validation here.
|
|
583
599
|
...(brainFlag === true && opts.workers ? { workers: opts.workers } : {}),
|
|
584
600
|
// The pool forwards `tools` to the provider and returns tool_calls
|
|
585
601
|
// (aegis1 app.py:7765 → provider, pool_brain synthesis keeps them).
|
|
@@ -699,6 +715,11 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
699
715
|
workers: payload && payload.workers,
|
|
700
716
|
onReasoning,
|
|
701
717
|
idleTimeoutMs: autonomous ? AUTONOMOUS_IDLE_TIMEOUT_MS : undefined,
|
|
718
|
+
// A caller with no live streaming surface (a `--no-stream` CLI flag, a
|
|
719
|
+
// one-shot script) can ask for the buffered non-stream wire form
|
|
720
|
+
// instead. Undefined/anything but `false` keeps every existing caller
|
|
721
|
+
// (the desktop renderer never sets this) on the streamed path.
|
|
722
|
+
stream: payload && payload.stream === false ? false : true,
|
|
702
723
|
};
|
|
703
724
|
|
|
704
725
|
// No round cap: a model that keeps calling tools keeps going for as
|
|
@@ -782,6 +803,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
782
803
|
}
|
|
783
804
|
addUsage(res);
|
|
784
805
|
|
|
806
|
+
// A cancelled turn is over. Both recoveries below exist for "the model
|
|
807
|
+
// said nothing", and an aborted request resolves with exactly that —
|
|
808
|
+
// no text — so without this guard a user pressing Esc to stop a slow
|
|
809
|
+
// turn immediately issued ANOTHER billed provider call (and, because
|
|
810
|
+
// every transport concatenates the deltas it forwards, streamed the
|
|
811
|
+
// partial answer a second time onto the same bubble). Measured before
|
|
812
|
+
// the guard: one interrupted turn == two dispatches.
|
|
813
|
+
if (signal.aborted) return withTurnUsage(res);
|
|
814
|
+
|
|
785
815
|
// Budget exhausted before the answer was written. Doubling it costs
|
|
786
816
|
// one request and converts a dead turn into a real one; a second
|
|
787
817
|
// 'length' result is accepted as-is so a hard-capped model can't
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "aegis-desktop",
|
|
3
3
|
"productName": "AEGIS Desktop",
|
|
4
|
-
"version": "0.4.
|
|
4
|
+
"version": "0.4.5",
|
|
5
5
|
"description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "AEGIS Code",
|
package/renderer/app.js
CHANGED
|
@@ -69,7 +69,10 @@ const ELEMENT_IDS = {
|
|
|
69
69
|
modelPreset: 'model-preset',
|
|
70
70
|
modelInput: 'model-input',
|
|
71
71
|
maxTokens: 'max-tokens',
|
|
72
|
+
maxTokensLabel: 'max-tokens-label',
|
|
73
|
+
maxTokensRow: 'max-tokens-row',
|
|
72
74
|
maxTokensAdaptive: 'max-tokens-adaptive',
|
|
75
|
+
effortRow: 'effort-row',
|
|
73
76
|
autonomousToggle: 'autonomous-toggle',
|
|
74
77
|
autonomousToggleWrap: 'autonomous-toggle-wrap',
|
|
75
78
|
autonomousControls: 'autonomous-controls',
|
|
@@ -171,6 +174,11 @@ const CUSTOM_MODEL_PRESETS = {
|
|
|
171
174
|
// already filters out of settings.list(); never render it as a provider row
|
|
172
175
|
// even if a stale store still surfaces it (defect #1).
|
|
173
176
|
const RESERVED_PROVIDERS = new Set(['__aegis', 'aegis']);
|
|
177
|
+
// Where a user without a key gets one. The class picker defaults to Aegis Cloud
|
|
178
|
+
// and the catalog is key-gated, so this is the first thing a new install needs;
|
|
179
|
+
// it lives here rather than inline so the Model hint and any future "connect"
|
|
180
|
+
// affordance cannot drift to two different pages.
|
|
181
|
+
const GET_AEGIS_KEY_URL = 'https://aegiscloud.org';
|
|
174
182
|
|
|
175
183
|
let pendingEl = null;
|
|
176
184
|
let pendingSessionId = null;
|
|
@@ -346,6 +354,35 @@ function updateAutonomousControlsVisibility() {
|
|
|
346
354
|
els.autonomousControls.hidden = !(wrapVisible && autonomousEnabled());
|
|
347
355
|
}
|
|
348
356
|
|
|
357
|
+
/**
|
|
358
|
+
* Which budget control applies to the selected class.
|
|
359
|
+
*
|
|
360
|
+
* Aegis Cloud is sized server-side from `effort`; the other three classes take
|
|
361
|
+
* a per-call token ceiling from the dropdown. Showing both at once is what made
|
|
362
|
+
* the token cap untrustworthy on the pooled class — the dropdown was displayed,
|
|
363
|
+
* read on every send, and then raised by the server's effort ladder, so the
|
|
364
|
+
* number beside it was never the budget the call ran on. Exactly one control is
|
|
365
|
+
* on offer now, and it is the one the request actually travels with.
|
|
366
|
+
*/
|
|
367
|
+
function updateBudgetControls(cls) {
|
|
368
|
+
const pooled = cls === AUTONOMOUS_CLASS;
|
|
369
|
+
if (els.effortRow) els.effortRow.hidden = !pooled;
|
|
370
|
+
if (els.maxTokensRow) els.maxTokensRow.hidden = pooled;
|
|
371
|
+
if (els.maxTokensLabel) els.maxTokensLabel.hidden = pooled;
|
|
372
|
+
}
|
|
373
|
+
|
|
374
|
+
/**
|
|
375
|
+
* The effort to send, or undefined for the classes the server does not size
|
|
376
|
+
* from it. `auto` means "let the server infer it from the ask" — the same thing
|
|
377
|
+
* the server already does for a request that names no effort, and a deliberate
|
|
378
|
+
* choice rather than the old silent fall-through to the top rung.
|
|
379
|
+
*/
|
|
380
|
+
function effortFor(cls) {
|
|
381
|
+
if (cls !== AUTONOMOUS_CLASS) return undefined;
|
|
382
|
+
const value = els.autonomousEffort && els.autonomousEffort.value;
|
|
383
|
+
return value && value !== 'auto' ? value : undefined;
|
|
384
|
+
}
|
|
385
|
+
|
|
349
386
|
// ---------------------------------------------------------------- UI helpers
|
|
350
387
|
|
|
351
388
|
function setConn(ok, text) {
|
|
@@ -1764,6 +1801,7 @@ async function loadModels(cls) {
|
|
|
1764
1801
|
els.autonomousToggleWrap.hidden = cls !== AUTONOMOUS_CLASS;
|
|
1765
1802
|
}
|
|
1766
1803
|
updateAutonomousControlsVisibility();
|
|
1804
|
+
updateBudgetControls(cls);
|
|
1767
1805
|
|
|
1768
1806
|
const custom = CUSTOM_CLASSES.has(cls);
|
|
1769
1807
|
els.modelSelect.hidden = custom;
|
|
@@ -1858,6 +1896,11 @@ async function loadModels(cls) {
|
|
|
1858
1896
|
try {
|
|
1859
1897
|
const data = await models.listModels(cls);
|
|
1860
1898
|
const list = Array.isArray(data && data.models) ? data.models : [];
|
|
1899
|
+
// No key on the default class: the engine reports the missing credential as
|
|
1900
|
+
// a state instead of letting the catalog call 401 (see engine.listModels).
|
|
1901
|
+
// The hint is where a user finds out they can connect at all — a raw
|
|
1902
|
+
// "listModels failed" told them only that something was broken.
|
|
1903
|
+
const needsKey = Boolean(data && data.needsKey);
|
|
1861
1904
|
for (const m of list) {
|
|
1862
1905
|
modelMeta.set(m.id, m);
|
|
1863
1906
|
const opt = document.createElement('option');
|
|
@@ -1866,12 +1909,28 @@ async function loadModels(cls) {
|
|
|
1866
1909
|
els.modelSelect.appendChild(opt);
|
|
1867
1910
|
}
|
|
1868
1911
|
let hint;
|
|
1869
|
-
if (
|
|
1912
|
+
if (needsKey) {
|
|
1913
|
+
hint = null; // carries a link, built below
|
|
1914
|
+
} else if (!list.length) {
|
|
1870
1915
|
hint = cls === 'ollama' ? 'Ollama not running or no models pulled.' : 'No models listed.';
|
|
1871
1916
|
} else {
|
|
1872
1917
|
hint = `${list.length} model${list.length === 1 ? '' : 's'} available.`;
|
|
1873
1918
|
}
|
|
1874
1919
|
const ceiling = applyMaxTokensClamp(els.modelSelect.value);
|
|
1920
|
+
if (needsKey) {
|
|
1921
|
+
// The hint elements are bare <p>s, so the link has to be a real child
|
|
1922
|
+
// node — a text assignment would wipe it (same shape as capNotice).
|
|
1923
|
+
els.modelHint.textContent =
|
|
1924
|
+
'Connect Aegis Cloud to load its models: paste your AEGIS API key in the Status card above (';
|
|
1925
|
+
const a = document.createElement('a');
|
|
1926
|
+
a.href = GET_AEGIS_KEY_URL;
|
|
1927
|
+
a.target = '_blank';
|
|
1928
|
+
a.rel = 'noreferrer noopener';
|
|
1929
|
+
a.textContent = 'free key at aegiscloud.org';
|
|
1930
|
+
els.modelHint.appendChild(a);
|
|
1931
|
+
els.modelHint.appendChild(document.createTextNode('), then Save.'));
|
|
1932
|
+
return;
|
|
1933
|
+
}
|
|
1875
1934
|
els.modelHint.textContent =
|
|
1876
1935
|
ceiling < FLAT_CEILING ? `${hint} · max output: ${ceiling.toLocaleString()}` : hint;
|
|
1877
1936
|
} catch (err) {
|
|
@@ -2248,12 +2307,21 @@ async function send() {
|
|
|
2248
2307
|
addMessage('user', prompt);
|
|
2249
2308
|
|
|
2250
2309
|
const ceiling = applyMaxTokensClamp(model);
|
|
2251
|
-
|
|
2310
|
+
// Aegis Cloud takes no token cap from here at all: the server sizes the call
|
|
2311
|
+
// from `effort`, and a number in this position is a per-pass ceiling *over*
|
|
2312
|
+
// that ladder (aegis1 services/pool_brain.py pass_budgets). Sending the
|
|
2313
|
+
// dropdown's value anyway is what made "Max tokens: 4k" beside a turn a
|
|
2314
|
+
// figure the turn never ran on. The row is hidden for this class as well, so
|
|
2315
|
+
// the two controls can never disagree.
|
|
2316
|
+
const maxTokens = cls === AUTONOMOUS_CLASS
|
|
2317
|
+
? undefined
|
|
2318
|
+
: maxTokensAdaptive() ? ceiling : parseInt(els.maxTokens.value, 10) || 4096;
|
|
2252
2319
|
const autonomous = cls === AUTONOMOUS_CLASS && autonomousEnabled();
|
|
2253
|
-
//
|
|
2254
|
-
//
|
|
2255
|
-
//
|
|
2256
|
-
|
|
2320
|
+
// Sent for the pooled class whether or not the fan-out is ticked: the fan-out
|
|
2321
|
+
// is enabled by the model id this class sends, so a turn that never entered
|
|
2322
|
+
// autonomous mode still ran pooled and had no way to say how big it should
|
|
2323
|
+
// be. `undefined` = "auto" = the server infers it from the ask.
|
|
2324
|
+
const effort = effortFor(cls);
|
|
2257
2325
|
const workers = autonomous ? parseInt(els.autonomousWorkers.value, 10) || undefined : undefined;
|
|
2258
2326
|
// Reuse the open thread's session id (minted once, on its first message)
|
|
2259
2327
|
// instead of a fresh one per send — a new id every turn is what made both
|
|
@@ -2352,7 +2420,11 @@ async function send() {
|
|
|
2352
2420
|
if (data && data.model) bits.push(`model: ${data.model}`);
|
|
2353
2421
|
else if (model) bits.push(`model: ${model}`);
|
|
2354
2422
|
bits.push(classLabel(cls));
|
|
2355
|
-
if (autonomous) bits.push(`autonomous (${effort}, ${workers ||
|
|
2423
|
+
if (autonomous) bits.push(`autonomous (${effort || 'auto'}, ${workers || 'auto'}w)`);
|
|
2424
|
+
// The budget rung is worth showing even without the fan-out: on the pooled
|
|
2425
|
+
// class it is what sized the call, and a user cannot tell a 16k turn from a
|
|
2426
|
+
// 64k one by looking at the answer.
|
|
2427
|
+
else if (effort) bits.push(`effort: ${effort}`);
|
|
2356
2428
|
const turnTokens = usageTokens(data && data.usage);
|
|
2357
2429
|
if (turnTokens != null) bits.push(`tokens: ${turnTokens}`);
|
|
2358
2430
|
addMessage('assistant', text, bits.join(' · ') || undefined, sessionId, toolLog);
|
|
@@ -2433,14 +2505,23 @@ async function init() {
|
|
|
2433
2505
|
updateAutonomousControlsVisibility();
|
|
2434
2506
|
});
|
|
2435
2507
|
|
|
2436
|
-
//
|
|
2437
|
-
//
|
|
2438
|
-
//
|
|
2508
|
+
// Budget control for the pooled class: which rung of the server's effort
|
|
2509
|
+
// ladder sizes the call, or "auto" to let the server infer it from the ask
|
|
2510
|
+
// (aegis1 services/pool_brain.py parse_brain_request). Opt-in per machine,
|
|
2511
|
+
// remembered across restarts; a stored value from a build whose list was
|
|
2512
|
+
// short (low/medium/high, no "auto") falls through to the markup's default
|
|
2513
|
+
// rather than assigning an option that no longer exists.
|
|
2439
2514
|
const savedEffort = localStorage.getItem(AUTONOMOUS_EFFORT_KEY);
|
|
2440
|
-
if (savedEffort
|
|
2515
|
+
if (savedEffort && Array.from(els.autonomousEffort.options).some((o) => o.value === savedEffort)) {
|
|
2516
|
+
els.autonomousEffort.value = savedEffort;
|
|
2517
|
+
}
|
|
2441
2518
|
els.autonomousEffort.addEventListener('change', () => {
|
|
2442
2519
|
localStorage.setItem(AUTONOMOUS_EFFORT_KEY, els.autonomousEffort.value);
|
|
2443
2520
|
});
|
|
2521
|
+
// Worker count for the fan-out. Left empty by default on purpose: an empty
|
|
2522
|
+
// field is what tells the server to size the fan-out from the ask
|
|
2523
|
+
// (parse_brain_request's auto path) instead of the old client-side default of
|
|
2524
|
+
// 3, which made every turn a 4-pass fan-out.
|
|
2444
2525
|
const savedWorkers = localStorage.getItem(AUTONOMOUS_WORKERS_KEY);
|
|
2445
2526
|
if (savedWorkers) els.autonomousWorkers.value = savedWorkers;
|
|
2446
2527
|
els.autonomousWorkers.addEventListener('change', () => {
|
|
@@ -2458,6 +2539,11 @@ async function init() {
|
|
|
2458
2539
|
|
|
2459
2540
|
els.classSelect.addEventListener('change', () => {
|
|
2460
2541
|
localStorage.setItem(CLASS_KEY, els.classSelect.value);
|
|
2542
|
+
// Apply the budget-control choice immediately rather than waiting for the
|
|
2543
|
+
// model list: loadModels() is async, and until it resolves the previous
|
|
2544
|
+
// class's control would still be on screen — showing a token cap on a
|
|
2545
|
+
// pooled turn, or hiding the effort rung it runs on.
|
|
2546
|
+
updateBudgetControls(els.classSelect.value);
|
|
2461
2547
|
loadModels(els.classSelect.value);
|
|
2462
2548
|
});
|
|
2463
2549
|
|
package/renderer/index.html
CHANGED
|
@@ -76,8 +76,8 @@
|
|
|
76
76
|
hidden
|
|
77
77
|
/>
|
|
78
78
|
|
|
79
|
-
<label for="max-tokens">Max tokens</label>
|
|
80
|
-
<div class="max-tokens-row">
|
|
79
|
+
<label for="max-tokens" id="max-tokens-label">Max tokens</label>
|
|
80
|
+
<div class="max-tokens-row" id="max-tokens-row">
|
|
81
81
|
<select id="max-tokens">
|
|
82
82
|
<option value="1024">1k</option>
|
|
83
83
|
<option value="4096" selected>4k</option>
|
|
@@ -95,6 +95,23 @@
|
|
|
95
95
|
</label>
|
|
96
96
|
</div>
|
|
97
97
|
|
|
98
|
+
<!-- The pooled AEGIS Cloud class sizes a request's token budget
|
|
99
|
+
server-side from this ladder (effort -> low/medium/high ->
|
|
100
|
+
16384/32768/65536 total across the fan-out), so the max-tokens row
|
|
101
|
+
above is hidden for that class instead of being shown and silently
|
|
102
|
+
overridden: the server treats a body max_tokens as a ceiling, and
|
|
103
|
+
an effort rung as the budget. Shown for Ollama and custom
|
|
104
|
+
endpoints, where a per-call ceiling is the only control there is. -->
|
|
105
|
+
<div class="max-tokens-row" id="effort-row" hidden>
|
|
106
|
+
<label for="autonomous-effort">Effort</label>
|
|
107
|
+
<select id="autonomous-effort">
|
|
108
|
+
<option value="low">low</option>
|
|
109
|
+
<option value="medium">medium</option>
|
|
110
|
+
<option value="high">high</option>
|
|
111
|
+
<option value="auto" selected>auto</option>
|
|
112
|
+
</select>
|
|
113
|
+
</div>
|
|
114
|
+
|
|
98
115
|
<label
|
|
99
116
|
class="flow-toggle"
|
|
100
117
|
for="autonomous-toggle"
|
|
@@ -106,14 +123,8 @@
|
|
|
106
123
|
<span>work autonomously</span>
|
|
107
124
|
</label>
|
|
108
125
|
<div id="autonomous-controls" class="autonomous-controls" hidden>
|
|
109
|
-
<label for="autonomous-effort">Effort</label>
|
|
110
|
-
<select id="autonomous-effort">
|
|
111
|
-
<option value="low">low</option>
|
|
112
|
-
<option value="medium">medium</option>
|
|
113
|
-
<option value="high" selected>high</option>
|
|
114
|
-
</select>
|
|
115
126
|
<label for="autonomous-workers">Workers</label>
|
|
116
|
-
<input type="number" id="autonomous-workers" min="1" max="8" step="1"
|
|
127
|
+
<input type="number" id="autonomous-workers" min="1" max="8" step="1" />
|
|
117
128
|
</div>
|
|
118
129
|
<p class="hint" id="model-hint"></p>
|
|
119
130
|
</section>
|
package/renderer/style.css
CHANGED
|
@@ -1142,6 +1142,17 @@ body.memory-open {
|
|
|
1142
1142
|
display: none;
|
|
1143
1143
|
}
|
|
1144
1144
|
|
|
1145
|
+
/* Same trap, one row up: `.max-tokens-row` is display: flex, so the `hidden`
|
|
1146
|
+
attribute did nothing on `#max-tokens-row`/`#effort-row` — which is how the
|
|
1147
|
+
budget-control swap (effort for the pooled class, a token ceiling for the
|
|
1148
|
+
others) would have shown BOTH controls at once, the exact state it exists to
|
|
1149
|
+
prevent. `display: none` is also what lets a `<label>` be hidden, since
|
|
1150
|
+
labels are inline by default and carry no author display rule. */
|
|
1151
|
+
.max-tokens-row[hidden],
|
|
1152
|
+
label[hidden] {
|
|
1153
|
+
display: none;
|
|
1154
|
+
}
|
|
1155
|
+
|
|
1145
1156
|
.autonomous-controls {
|
|
1146
1157
|
display: flex;
|
|
1147
1158
|
align-items: center;
|
package/vendor/aegis.js
CHANGED
|
@@ -278,6 +278,13 @@ function createClient(opts = {}) {
|
|
|
278
278
|
* so a reasoning-only stream still counts as unanswered.
|
|
279
279
|
* - `idleTimeoutMs` override the stalled-stream watchdog for this call
|
|
280
280
|
* (a worker fan-out is legitimately silent between passes)
|
|
281
|
+
* - `maxTokens` output ceiling. Omitted from the body when absent so the
|
|
282
|
+
* server's own budget applies — on a pooled request that is
|
|
283
|
+
* the `effort` ladder, and a number sent from here caps it.
|
|
284
|
+
* - `effort` / `workers` pooled-brain budget rung and fan-out size.
|
|
285
|
+
* Forwarded only inside `extra`, as the server reads them off
|
|
286
|
+
* the body (aegis1 services/pool_brain.py parse_brain_request)
|
|
287
|
+
* and sizes the budget from them.
|
|
281
288
|
*/
|
|
282
289
|
async function chatCompletion({
|
|
283
290
|
prompt,
|
|
@@ -298,9 +305,17 @@ function createClient(opts = {}) {
|
|
|
298
305
|
// model the server picks its default (no client-invented tier id). `mode`
|
|
299
306
|
// is a legacy server-side shorthand — forwarded verbatim only when the
|
|
300
307
|
// caller supplies it, never defaulted, never built into a model id.
|
|
308
|
+
//
|
|
309
|
+
// `max_tokens` is omitted entirely when the caller states none, rather than
|
|
310
|
+
// defaulted here. The server has its own budget ladder (`mode`/`effort` on
|
|
311
|
+
// the pooled path) and treats a body max_tokens as a *ceiling* over it, so
|
|
312
|
+
// an invented 4096 was not a harmless default: it capped every pass of a
|
|
313
|
+
// pooled-brain fan-out at 4096 and overrode the ladder the caller's effort
|
|
314
|
+
// selected. Omitting it is the documented way to say "server default" (see
|
|
315
|
+
// mcp/tools.js's max_tokens description).
|
|
301
316
|
const body = {
|
|
302
317
|
messages: buildMessages(messages, system, prompt),
|
|
303
|
-
max_tokens: maxTokens
|
|
318
|
+
...(maxTokens ? { max_tokens: maxTokens } : {}),
|
|
304
319
|
...(extra || {}),
|
|
305
320
|
};
|
|
306
321
|
if (model) body.model = model;
|