aegis-desktop 0.6.0 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/local/autonomous.js +165 -20
- package/lib/local/engine.js +62 -7
- package/lib/local/queue.js +16 -1
- package/lib/local/session-rounds.js +188 -0
- package/main.js +18 -0
- package/package.json +2 -2
- package/renderer/app.js +10 -0
- package/renderer/index.html +10 -3
package/lib/local/autonomous.js
CHANGED
|
@@ -73,13 +73,65 @@ function writtenPath(tool) {
|
|
|
73
73
|
|
|
74
74
|
/**
|
|
75
75
|
* The default AEGIS Cloud model for autonomous work: the pooled brain, which
|
|
76
|
-
* is the tier the server
|
|
77
|
-
*
|
|
78
|
-
*
|
|
79
|
-
*
|
|
76
|
+
* is the tier the server can fan out to multiple reasoning workers and
|
|
77
|
+
* synthesise. `nexus-brain` is the canonical id the catalog itself prefers
|
|
78
|
+
* (the other tier spellings are aliases of it — see filterAegisCatalog in
|
|
79
|
+
* engine.js).
|
|
80
|
+
*
|
|
81
|
+
* The model id is only the tier. It is NOT what makes an autonomous task
|
|
82
|
+
* expensive — the fan-out is, and the fan-out is opt-in per task (see
|
|
83
|
+
* resolveFanout below). This used to be documented the other way round ("the
|
|
84
|
+
* pooled brain ... autonomous tasks are exactly the ones worth that spend"),
|
|
85
|
+
* and the worker sent `autonomous: true` on every queued task, so the most
|
|
86
|
+
* expensive shape of the most expensive tier ran on one-line tasks too.
|
|
80
87
|
*/
|
|
81
88
|
const DEFAULT_MODEL = 'nexus-brain';
|
|
82
89
|
|
|
90
|
+
/**
|
|
91
|
+
* THE QUEUE RUNS ON AEGIS CLOUD, AND NOTHING ELSE.
|
|
92
|
+
*
|
|
93
|
+
* A queued task is billed to the AEGIS pool and every turn it makes goes out
|
|
94
|
+
* with `class: 'aegis'` (see runTask below) — the pool is the only backend this
|
|
95
|
+
* worker can reach. So a model id here has to be one the *pool* serves. A
|
|
96
|
+
* direct-provider id is not a cheaper option the queue could fall back to; it
|
|
97
|
+
* is a request the pool cannot honour, or worse, a per-provider spelling
|
|
98
|
+
* (`anthropic`, `groq`, …) that quietly pins one upstream instead of letting
|
|
99
|
+
* the pool auto-route across whichever providers hold a live key.
|
|
100
|
+
*
|
|
101
|
+
* The accept-list is therefore the pooled-brain tier family — the one entry
|
|
102
|
+
* engine.js's filterAegisCatalog offers for the Aegis Cloud class, plus the
|
|
103
|
+
* `-smart`/`-neo` tier spellings the server still serves as aliases of it. This
|
|
104
|
+
* mirrors selectBrainEntry() there rather than re-deriving "anything starting
|
|
105
|
+
* with nexus-": `nexus-fast` is not a tier the catalog has ever served, and
|
|
106
|
+
* accepting a made-up id means a queued task that fails at the server after
|
|
107
|
+
* being picked up, or runs on a tier nobody chose.
|
|
108
|
+
*/
|
|
109
|
+
const AEGIS_MODEL_RE = /^(?:nexus|aegis)-brain(?:-(?:smart|neo))?$/;
|
|
110
|
+
const AEGIS_MODEL_IDS = Object.freeze(['nexus-brain', 'aegis-brain']);
|
|
111
|
+
|
|
112
|
+
/** True when `id` names an AEGIS Cloud pooled-brain tier (the queue's only models). */
|
|
113
|
+
function isAegisModel(id) {
|
|
114
|
+
return AEGIS_MODEL_RE.test(String(id == null ? '' : id).trim());
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Why a STATED model id cannot be queued, or '' when it can. Blank is not a
|
|
119
|
+
* refusal — "no pick" is the default model, which resolveModel supplies.
|
|
120
|
+
*
|
|
121
|
+
* The message names the pool, because the failure it prevents ("queued on
|
|
122
|
+
* claude-sonnet-4, ran on — or was billed to — something else") is invisible
|
|
123
|
+
* otherwise: `class: 'aegis'` would be sent with an id the server does not
|
|
124
|
+
* serve, and the task would come back as an opaque error minutes later.
|
|
125
|
+
*/
|
|
126
|
+
function modelRefusal(id) {
|
|
127
|
+
const stated = String(id == null ? '' : id).trim();
|
|
128
|
+
if (!stated || isAegisModel(stated)) return '';
|
|
129
|
+
return (
|
|
130
|
+
`the autonomous queue runs Aegis Cloud models only (${DEFAULT_MODEL}, ` +
|
|
131
|
+
`${AEGIS_MODEL_IDS.join('/')} aliases); "${stated}" is not one`
|
|
132
|
+
);
|
|
133
|
+
}
|
|
134
|
+
|
|
83
135
|
/** Phrase match for "work autonomously" in a prompt or a queued task. */
|
|
84
136
|
const AUTONOMOUS_REQUEST_RE =
|
|
85
137
|
/\bautonomously\b|\bon your own\b|\bwithout asking\b|\bend[- ]to[- ]end\b|\bno (?:more )?questions\b|\bfully autonomous\b|\bqueue it\b/i;
|
|
@@ -160,23 +212,92 @@ function withRoundHorizon(rounds, env, fn) {
|
|
|
160
212
|
}
|
|
161
213
|
|
|
162
214
|
/**
|
|
163
|
-
* The model an autonomous task runs on: an explicit pick wins, then
|
|
164
|
-
* environment (so a systemd timer can
|
|
165
|
-
* brain. Never the interactive session's model — a queue survives the
|
|
166
|
-
* that queued it, so it cannot inherit that session's choice.
|
|
215
|
+
* The model an autonomous task runs on: an explicit pick wins, then an AEGIS
|
|
216
|
+
* Cloud pin in the environment (so a systemd timer can choose a tier), then the
|
|
217
|
+
* pooled brain. Never the interactive session's model — a queue survives the
|
|
218
|
+
* session that queued it, so it cannot inherit that session's choice.
|
|
219
|
+
*
|
|
220
|
+
* TWO DIFFERENT TREATMENTS FOR TWO DIFFERENT SOURCES, on purpose:
|
|
221
|
+
*
|
|
222
|
+
* - a pick that came from the TASK is returned verbatim, even when it is
|
|
223
|
+
* wrong. Substituting a correct model for a stated one is how a queue
|
|
224
|
+
* "runs on nexus-brain" while the file says otherwise; the caller refuses
|
|
225
|
+
* it out loud instead (modelRefusal, queue.addTask, and the pre-flight in
|
|
226
|
+
* runTask).
|
|
227
|
+
* - a non-Aegis value in the ENVIRONMENT is skipped, because AEGIS_MODEL is
|
|
228
|
+
* shared with the interactive surfaces (which run direct providers), so a
|
|
229
|
+
* stray value there is not a statement about the queue. Refusing every
|
|
230
|
+
* task over it would break drains for a reason that is not the task's
|
|
231
|
+
* fault; a fallback to the pooled brain keeps the drain honest and on-cloud.
|
|
167
232
|
*/
|
|
168
233
|
function resolveModel({ model, env } = {}) {
|
|
169
234
|
const e = env || process.env;
|
|
170
235
|
const picked = String(model || '').trim();
|
|
171
236
|
if (picked) return picked;
|
|
172
237
|
const fromEnv = String(e.AEGIS_AUTONOMOUS_MODEL || e.AEGIS_MODEL || '').trim();
|
|
173
|
-
return fromEnv
|
|
238
|
+
return isAegisModel(fromEnv) ? fromEnv : DEFAULT_MODEL;
|
|
174
239
|
}
|
|
175
240
|
|
|
176
|
-
/**
|
|
177
|
-
|
|
241
|
+
/**
|
|
242
|
+
* The environment pin the queue had to ignore, or '' when there was none.
|
|
243
|
+
*
|
|
244
|
+
* Reported (not silent) because an operator who exported
|
|
245
|
+
* AEGIS_AUTONOMOUS_MODEL=deepseek-v4-flash asked for a model and is not getting
|
|
246
|
+
* it: the queue falls back to the pool, and the one place that says so is this
|
|
247
|
+
* string, which runTask emits as a note and the desktop card shows.
|
|
248
|
+
*/
|
|
249
|
+
function ignoredEnvModel({ model, env } = {}) {
|
|
250
|
+
if (String(model || '').trim()) return ''; // the task's own pick is what counts
|
|
251
|
+
const e = env || process.env;
|
|
252
|
+
const fromEnv = String(e.AEGIS_AUTONOMOUS_MODEL || e.AEGIS_MODEL || '').trim();
|
|
253
|
+
if (!fromEnv || isAegisModel(fromEnv)) return '';
|
|
254
|
+
const which = String(e.AEGIS_AUTONOMOUS_MODEL || '').trim() ? 'AEGIS_AUTONOMOUS_MODEL' : 'AEGIS_MODEL';
|
|
255
|
+
return `${which}="${fromEnv}" is not an Aegis Cloud model — running on ${DEFAULT_MODEL} instead`;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
/**
|
|
259
|
+
* Whether a queued task runs the pooled-brain worker fan-out (aegis1
|
|
260
|
+
* services/pool_brain.py) or one plain turn on the same tier.
|
|
261
|
+
*
|
|
262
|
+
* COST IS THE REASON THIS IS OPT-IN. The fan-out is the single biggest
|
|
263
|
+
* multiplier this app can put on a bill: pool_brain spawns up to `workers`
|
|
264
|
+
* reasoning workers plus a synthesis pass, re-sends the task context to every
|
|
265
|
+
* one of them, and sizes each from the same effort ladder. A 3-worker
|
|
266
|
+
* high-effort task is therefore roughly four full reasoning calls against a
|
|
267
|
+
* 65536-token ladder, where the identical task single-pass is one call on the
|
|
268
|
+
* medium rung. The fan-out earns that on genuinely open-ended investigation
|
|
269
|
+
* ("why did X regress across this repo"); it is pure waste on a task that
|
|
270
|
+
* already names the file to edit.
|
|
271
|
+
*
|
|
272
|
+
* Precedence: the task's own `autonomous: true` (or `singlePass: false`, the
|
|
273
|
+
* explicit "fan me out") wins, then AEGIS_AUTONOMOUS_FANOUT=1 in the
|
|
274
|
+
* environment, else single pass.
|
|
275
|
+
*/
|
|
276
|
+
function resolveFanout(item = {}, env = process.env) {
|
|
277
|
+
const it = item || {};
|
|
278
|
+
if (it.autonomous === true || it.singlePass === false) return true;
|
|
279
|
+
const e = env || process.env;
|
|
280
|
+
return /^(1|true|yes|on)$/i.test(String(e.AEGIS_AUTONOMOUS_FANOUT || '').trim());
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Effort rung for an unattended turn.
|
|
285
|
+
*
|
|
286
|
+
* The rung is a spend knob, not a quality slider: the pooled class sizes its
|
|
287
|
+
* whole budget ladder from it (aegis1 services/pool_brain.py pass_budgets:
|
|
288
|
+
* low/medium/high -> 16384/32768/65536 tokens TOTAL across the fan-out), and
|
|
289
|
+
* the engine uses it for any model that reasons against its own output budget.
|
|
290
|
+
* `high` is the right rung for a fan-out — it is what buys a synthesis pass
|
|
291
|
+
* worth reading — but on a single pass it is a 2x over medium for budget
|
|
292
|
+
* nobody reads, so the default follows the shape of the task rather than
|
|
293
|
+
* always being the most expensive rung. An explicit pick (item.effort) or
|
|
294
|
+
* AEGIS_AUTONOMOUS_EFFORT still wins outright.
|
|
295
|
+
*/
|
|
296
|
+
function resolveEffort({ effort, env, fanout } = {}) {
|
|
178
297
|
const e = env || process.env;
|
|
179
|
-
|
|
298
|
+
const stated = String(effort || e.AEGIS_AUTONOMOUS_EFFORT || '').trim();
|
|
299
|
+
if (stated) return stated;
|
|
300
|
+
return fanout ? 'high' : 'medium';
|
|
180
301
|
}
|
|
181
302
|
|
|
182
303
|
/**
|
|
@@ -290,8 +411,24 @@ function createQueueWorker({ engine, env = process.env, log = () => {}, git = gi
|
|
|
290
411
|
*/
|
|
291
412
|
async function runTask(item, { carry = '', commit } = {}) {
|
|
292
413
|
const cwd = item.cwd || process.cwd();
|
|
414
|
+
// Aegis Cloud or nothing — checked BEFORE the turn, not at the server. An
|
|
415
|
+
// item whose model is not a pooled tier (a hand-edited queue file, a
|
|
416
|
+
// `--model` the CLI accepted before this rule existed, another host's
|
|
417
|
+
// older build) would otherwise go out as `class: 'aegis'` with an id the
|
|
418
|
+
// pool does not serve: billed work if it happens to be a per-provider
|
|
419
|
+
// spelling, an opaque server error otherwise. Failing here names the model
|
|
420
|
+
// and the allowed ones, and costs nothing.
|
|
421
|
+
const refusal = modelRefusal(item.model);
|
|
422
|
+
if (refusal) {
|
|
423
|
+
const failed = { ok: false, error: refusal, model: item.model, ms: 0 };
|
|
424
|
+
emit({ type: 'finish', taskId: item.id, ok: false, result: failed });
|
|
425
|
+
return failed;
|
|
426
|
+
}
|
|
293
427
|
const model = resolveModel({ model: item.model, env });
|
|
294
|
-
const
|
|
428
|
+
const ignored = ignoredEnvModel({ model: item.model, env });
|
|
429
|
+
if (ignored) emit({ type: 'note', taskId: item.id, note: ignored });
|
|
430
|
+
const fanout = resolveFanout(item, env);
|
|
431
|
+
const effort = resolveEffort({ effort: item.effort, env, fanout });
|
|
295
432
|
const rounds = maxRounds(env, item.maxRounds);
|
|
296
433
|
// Approval requests have no one to answer them here; see the header.
|
|
297
434
|
let approvalAsked = null;
|
|
@@ -331,7 +468,7 @@ function createQueueWorker({ engine, env = process.env, log = () => {}, git = gi
|
|
|
331
468
|
const wantCommit = commit === undefined ? Boolean(item.commit) : Boolean(commit);
|
|
332
469
|
const before = wantCommit ? safe(() => git.gitStatusSnapshot(cwd), null) : null;
|
|
333
470
|
|
|
334
|
-
emit({ type: 'start', taskId: item.id, model, cwd, rounds });
|
|
471
|
+
emit({ type: 'start', taskId: item.id, model, cwd, rounds, fanout, effort });
|
|
335
472
|
const started = now();
|
|
336
473
|
let result;
|
|
337
474
|
try {
|
|
@@ -341,13 +478,16 @@ function createQueueWorker({ engine, env = process.env, log = () => {}, git = gi
|
|
|
341
478
|
class: 'aegis',
|
|
342
479
|
model,
|
|
343
480
|
prompt: taskPrompt(item, { carry, rounds }),
|
|
344
|
-
// The pooled
|
|
345
|
-
//
|
|
346
|
-
//
|
|
347
|
-
//
|
|
348
|
-
|
|
481
|
+
// The pooled-brain fan-out ("fan out this turn" in the GUI chat
|
|
482
|
+
// header, and opt-in here): the server fans the round out to
|
|
483
|
+
// several reasoning workers and synthesises. It costs about
|
|
484
|
+
// workers+1 full reasoning calls, so it travels only when the task
|
|
485
|
+
// asked for it — see resolveFanout above. A `singlePass` task is
|
|
486
|
+
// the default for exactly that reason; the retry/write-up passes in
|
|
487
|
+
// the engine send `brain: false` for the same one-call reason.
|
|
488
|
+
autonomous: fanout,
|
|
349
489
|
effort,
|
|
350
|
-
workers: item.workers || undefined,
|
|
490
|
+
workers: fanout ? item.workers || undefined : undefined,
|
|
351
491
|
// The turn's working directory rides on `env`: engine.js reads the
|
|
352
492
|
// tool loop's cwd from envFor(payload), so a top-level `cwd` field
|
|
353
493
|
// is a directory the engine would ignore and every tool would run
|
|
@@ -582,13 +722,18 @@ function commitMessage(item) {
|
|
|
582
722
|
module.exports = {
|
|
583
723
|
DEFAULT_MODEL,
|
|
584
724
|
DEFAULT_ROUNDS,
|
|
725
|
+
AEGIS_MODEL_IDS,
|
|
726
|
+
isAegisModel,
|
|
727
|
+
modelRefusal,
|
|
585
728
|
WRITE_TOOLS,
|
|
586
729
|
writtenPath,
|
|
587
730
|
isAutonomousRequest,
|
|
588
731
|
maxRounds,
|
|
589
732
|
withRoundHorizon,
|
|
590
733
|
resolveModel,
|
|
734
|
+
ignoredEnvModel,
|
|
591
735
|
resolveEffort,
|
|
736
|
+
resolveFanout,
|
|
592
737
|
autonomousDirective,
|
|
593
738
|
taskPrompt,
|
|
594
739
|
appendDigest,
|
package/lib/local/engine.js
CHANGED
|
@@ -383,7 +383,14 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
383
383
|
* settings store's own accessor; then the safe default — ON, i.e. the gate
|
|
384
384
|
* stays up, so a store that predates the toggle can never silently
|
|
385
385
|
* disable it. */
|
|
386
|
+
// Session-scoped round accounting (lib/local/session-rounds.js): the
|
|
387
|
+
// tool-round cap is held against the *session* rather than the single turn,
|
|
388
|
+
// so a turn that reaches its horizon is resumed by the next turn in the same
|
|
389
|
+
// conversation instead of being cut off with the work lost. Module-level
|
|
390
|
+
// state, so it survives `send()` and even a fresh engine instance.
|
|
391
|
+
const sessionRounds = require('./session-rounds.js');
|
|
386
392
|
const confirmModeEnabled = () => {
|
|
393
|
+
|
|
387
394
|
if (typeof getConfirmMode === 'function') return getConfirmMode() !== false;
|
|
388
395
|
if (settings && typeof settings.getConfirmMode === 'function') {
|
|
389
396
|
const value = settings.getConfirmMode();
|
|
@@ -1020,12 +1027,47 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
1020
1027
|
// The numbers match aegiscodex-dev's (src/autonomous.js) so both clients
|
|
1021
1028
|
// behave the same: 24 rounds for a chat turn, 40 for an autonomous one.
|
|
1022
1029
|
// Env-overridable for a deliberately long job.
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1030
|
+
// A stated horizon (env) still wins outright — an explicit number is the
|
|
1031
|
+
// user overriding the engine, not something the ledger may pad.
|
|
1032
|
+
const roundEnvName = autonomous ? 'AEGIS_AUTONOMOUS_MAX_ROUNDS' : 'AEGIS_CHAT_MAX_ROUNDS';
|
|
1033
|
+
const statedRounds = (() => {
|
|
1034
|
+
const raw = Number.parseInt(process.env[roundEnvName] || '', 10);
|
|
1035
|
+
return Number.isFinite(raw) && raw > 0 ? raw : 0;
|
|
1028
1036
|
})();
|
|
1037
|
+
const baseRounds = statedRounds || (autonomous ? 40 : 24);
|
|
1038
|
+
|
|
1039
|
+
// Did the previous turn in this session die at its horizon? If so, this
|
|
1040
|
+
// turn is a continuation of it: same job, unfinished, and the model has
|
|
1041
|
+
// to be told — a cold restart is what made the old cap lose work.
|
|
1042
|
+
let roundSessionKey = sessionRounds.keyFor(payload, history);
|
|
1043
|
+
let resumed = sessionRounds.take(roundSessionKey);
|
|
1044
|
+
if (!resumed) {
|
|
1045
|
+
// The caller minted a fresh key this turn (no stated session id and a
|
|
1046
|
+
// rebuilt history array). Adopt the held work rather than dropping it,
|
|
1047
|
+
// but only within the ledger's recency window.
|
|
1048
|
+
const adopted = sessionRounds.adopt();
|
|
1049
|
+
if (adopted) {
|
|
1050
|
+
resumed = adopted.entry;
|
|
1051
|
+
roundSessionKey = adopted.key;
|
|
1052
|
+
}
|
|
1053
|
+
}
|
|
1054
|
+
if (resumed) {
|
|
1055
|
+
const preamble = sessionRounds.resumePreamble(resumed);
|
|
1056
|
+
// Round 1's ask travels as `prompt`; a follow-up dispatch (and every
|
|
1057
|
+
// turn that skips the shorthand) has to receive it through history.
|
|
1058
|
+
if (prompt) prompt = `${preamble}\n\n${prompt}`;
|
|
1059
|
+
else history.push({ role: 'user', content: preamble });
|
|
1060
|
+
if (rootOnDelta) {
|
|
1061
|
+
rootOnDelta({
|
|
1062
|
+
delta:
|
|
1063
|
+
`\n\n[resuming work this session left unfinished at ${resumed.rounds} tool rounds` +
|
|
1064
|
+
`${resumed.interruptions > 1 ? ` (cut off ${resumed.interruptions}×)` : ''}…]\n`,
|
|
1065
|
+
});
|
|
1066
|
+
}
|
|
1067
|
+
}
|
|
1068
|
+
// A resumed turn gets a small bonus on top of the base horizon so
|
|
1069
|
+
// re-orientation does not consume the new budget before any work lands.
|
|
1070
|
+
const maxRounds = statedRounds || baseRounds + (resumed ? sessionRounds.bonus(baseRounds) : 0);
|
|
1029
1071
|
let round = 0;
|
|
1030
1072
|
|
|
1031
1073
|
// Token accounting for the whole TURN, not just its last round. An
|
|
@@ -1084,13 +1126,26 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
|
|
|
1084
1126
|
const note =
|
|
1085
1127
|
`[stopped at ${maxRounds} tool rounds` +
|
|
1086
1128
|
`${turnUsage.total_tokens ? `, ${turnUsage.total_tokens.toLocaleString()} tokens` : ''}` +
|
|
1087
|
-
`.
|
|
1088
|
-
|
|
1129
|
+
`. This session holds the work: your next message here resumes it automatically` +
|
|
1130
|
+
` (with a continuation preamble), or raise ${roundEnvName} for a longer single turn.]`;
|
|
1131
|
+
// File the interruption before returning, so the horizon is a pause
|
|
1132
|
+
// in the session rather than the end of the job. `chain` carries the
|
|
1133
|
+
// count forward across resumes.
|
|
1134
|
+
const held = sessionRounds.record(roundSessionKey, {
|
|
1135
|
+
rounds: round,
|
|
1136
|
+
tokens: turnUsage.total_tokens || 0,
|
|
1137
|
+
note,
|
|
1138
|
+
chain: resumed ? resumed.interruptions : 0,
|
|
1139
|
+
});
|
|
1089
1140
|
if (rootOnDelta) rootOnDelta({ delta: `\n\n${note}` });
|
|
1090
1141
|
return withTurnUsage({
|
|
1091
1142
|
model: base.model,
|
|
1092
1143
|
choices: [{ message: { content: note }, finish_reason: 'length' }],
|
|
1093
1144
|
stoppedOnRounds: true,
|
|
1145
|
+
heldForSession: true,
|
|
1146
|
+
heldRounds: held.rounds,
|
|
1147
|
+
sessionInterruptions: held.interruptions,
|
|
1148
|
+
resumedFrom: roundSessionKey,
|
|
1094
1149
|
});
|
|
1095
1150
|
}
|
|
1096
1151
|
round += 1;
|
package/lib/local/queue.js
CHANGED
|
@@ -179,17 +179,32 @@ function upsert(items, entry) {
|
|
|
179
179
|
* with it as the tool loop's working directory), `model` is the AEGIS Cloud
|
|
180
180
|
* model id to run it on, and `commit` asks the worker to commit what the task
|
|
181
181
|
* changed (scoped — see autonomous.js).
|
|
182
|
+
*
|
|
183
|
+
* AEGIS CLOUD ONLY, REFUSED AT THE DOOR. The worker sends every task out as
|
|
184
|
+
* `class: 'aegis'`, so a model the pool does not serve cannot run here at all —
|
|
185
|
+
* it is a task that fails minutes into a drain, after being picked up, with an
|
|
186
|
+
* opaque server error. `model: null` stays legal: that is "no pick", and
|
|
187
|
+
* autonomous.resolveModel fills in the pooled brain. Anything else has to pass
|
|
188
|
+
* that module's own accept-list, so the CLI (`aegiscode autonomous add
|
|
189
|
+
* --model`), the desktop card and a hand-written queue file are all held to the
|
|
190
|
+
* same rule. Required lazily: queue.js is a dependency of autonomous.js, and
|
|
191
|
+
* this check must not turn that into a load-order cycle.
|
|
182
192
|
*/
|
|
183
193
|
function addTask(env, opts = {}) {
|
|
184
194
|
const task = String(opts.task == null ? '' : opts.task).trim();
|
|
185
195
|
if (!task) throw new Error('queue: a task needs text');
|
|
196
|
+
const statedModel = String(opts.model == null ? '' : opts.model).trim();
|
|
197
|
+
if (statedModel) {
|
|
198
|
+
const refusal = require('./autonomous.js').modelRefusal(statedModel);
|
|
199
|
+
if (refusal) throw new Error(`queue: ${refusal}`);
|
|
200
|
+
}
|
|
186
201
|
const items = loadQueue(env);
|
|
187
202
|
const now = opts.now || Date.now();
|
|
188
203
|
const entry = {
|
|
189
204
|
id: nextId(items),
|
|
190
205
|
task,
|
|
191
206
|
cwd: opts.cwd || process.cwd(),
|
|
192
|
-
model:
|
|
207
|
+
model: statedModel || null,
|
|
193
208
|
effort: opts.effort || null,
|
|
194
209
|
workers: Number.isInteger(opts.workers) ? opts.workers : null,
|
|
195
210
|
singlePass: Boolean(opts.singlePass),
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Session-scoped tool-round ledger.
|
|
5
|
+
*
|
|
6
|
+
* The engine's tool-round cap used to be a wall. A turn that reached it ended
|
|
7
|
+
* mid-investigation with `[stopped at 24 tool rounds…]`, and because the
|
|
8
|
+
* accounting lived only for the duration of that call, the next turn in the
|
|
9
|
+
* same conversation started from zero: it knew nothing about the horizon that
|
|
10
|
+
* had just cut it off, so a long job could be truncated by the cap forever
|
|
11
|
+
* without ever being finished.
|
|
12
|
+
*
|
|
13
|
+
* This makes the cap *session* state instead of turn state:
|
|
14
|
+
*
|
|
15
|
+
* - a turn that reaches its horizon records the unfinished work against the
|
|
16
|
+
* session key (rounds spent, tokens, the note the user saw);
|
|
17
|
+
* - the next turn in that session takes the record — consuming it, so one
|
|
18
|
+
* resume per interruption and a chain of interruptions is a chain of
|
|
19
|
+
* deliberate asks, never an automatic loop — and gets a continuation
|
|
20
|
+
* preamble plus a small round bonus, so re-orientation does not eat the
|
|
21
|
+
* new horizon before any work happens;
|
|
22
|
+
* - an entry nobody comes back for expires after TTL_MS.
|
|
23
|
+
*
|
|
24
|
+
* State is module-level, so it outlives a turn, a `send()` and whichever
|
|
25
|
+
* engine instance the desktop happens to be using. It never leaves the
|
|
26
|
+
* process and is never written to disk.
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
/** How long an unclaimed interruption stays resumable (30 min). */
|
|
30
|
+
const TTL_MS = 30 * 60 * 1000;
|
|
31
|
+
/** Bound on live entries: a long-lived app must not grow without limit. */
|
|
32
|
+
const MAX_ENTRIES = 64;
|
|
33
|
+
|
|
34
|
+
const entries = new Map(); // key -> { rounds, tokens, note, at, interruptions }
|
|
35
|
+
const historyKeys = new WeakMap(); // history[] -> key
|
|
36
|
+
let historyKeySeq = 0;
|
|
37
|
+
|
|
38
|
+
/** Drop expired entries, then the oldest ones past the bound. */
|
|
39
|
+
function prune(now = Date.now()) {
|
|
40
|
+
for (const [key, entry] of entries) {
|
|
41
|
+
if (now - entry.at > TTL_MS) entries.delete(key);
|
|
42
|
+
}
|
|
43
|
+
while (entries.size > MAX_ENTRIES) {
|
|
44
|
+
// Map iteration is insertion-ordered and `record` re-inserts on write, so
|
|
45
|
+
// the first key is the oldest.
|
|
46
|
+
const oldest = entries.keys().next();
|
|
47
|
+
if (oldest.done) break;
|
|
48
|
+
entries.delete(oldest.value);
|
|
49
|
+
}
|
|
50
|
+
return entries.size;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* The key an interruption is filed under. A caller that states its session
|
|
55
|
+
* (desktop payloads carry an id) gets an exact match; otherwise the working
|
|
56
|
+
* directory, then the identity of the history array, then a single shared
|
|
57
|
+
* bucket. `adopt` covers the remaining case — a caller that mints a fresh key
|
|
58
|
+
* every turn.
|
|
59
|
+
*/
|
|
60
|
+
function keyFor(payload, history) {
|
|
61
|
+
const p = payload && typeof payload === 'object' ? payload : {};
|
|
62
|
+
const stated = p.sessionKey || p.sessionId || p.chatId || p.threadId || p.conversationId;
|
|
63
|
+
if (typeof stated === 'string' && stated.trim()) return `s:${stated.trim()}`;
|
|
64
|
+
if (typeof p.cwd === 'string' && p.cwd.trim()) return `cwd:${p.cwd.trim()}`;
|
|
65
|
+
if (history && typeof history === 'object') {
|
|
66
|
+
let key = historyKeys.get(history);
|
|
67
|
+
if (!key) {
|
|
68
|
+
historyKeySeq += 1;
|
|
69
|
+
key = `hist-${historyKeySeq}`;
|
|
70
|
+
historyKeys.set(history, key);
|
|
71
|
+
}
|
|
72
|
+
return key;
|
|
73
|
+
}
|
|
74
|
+
return 'default';
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Extra rounds granted to a resuming turn. Deliberately small and bounded: the
|
|
79
|
+
* point of resuming is to *continue* work that already exists, not to hand out
|
|
80
|
+
* a second full horizon (every round re-sends the whole conversation, so this
|
|
81
|
+
* is the cheapest way to make the cap non-fatal).
|
|
82
|
+
*/
|
|
83
|
+
function bonus(base) {
|
|
84
|
+
const n = Number.isFinite(base) && base > 0 ? base : 0;
|
|
85
|
+
return Math.max(4, Math.ceil(n / 4));
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
/** Peek without consuming — the entry stays resumable. */
|
|
89
|
+
function peek(key, now = Date.now()) {
|
|
90
|
+
prune(now);
|
|
91
|
+
return entries.get(key) || null;
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/** Claim the interruption for `key`, if any. Consuming: a second call gets null. */
|
|
95
|
+
function take(key, now = Date.now()) {
|
|
96
|
+
prune(now);
|
|
97
|
+
const entry = entries.get(key);
|
|
98
|
+
if (!entry) return null;
|
|
99
|
+
entries.delete(key);
|
|
100
|
+
return entry;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Claim the most recent interruption filed under ANY key. This is the bridge
|
|
105
|
+
* for a caller whose key does not survive between turns: the next turn on this
|
|
106
|
+
* process adopts the held work rather than losing it. Bounded by `maxAgeMs` so
|
|
107
|
+
* an unrelated new conversation is not handed yesterday's job.
|
|
108
|
+
*/
|
|
109
|
+
function adopt({ now = Date.now(), maxAgeMs = 10 * 60 * 1000 } = {}) {
|
|
110
|
+
prune(now);
|
|
111
|
+
let bestKey = null;
|
|
112
|
+
let best = null;
|
|
113
|
+
for (const [key, entry] of entries) {
|
|
114
|
+
if (now - entry.at > maxAgeMs) continue;
|
|
115
|
+
if (!best || entry.at > best.at) {
|
|
116
|
+
best = entry;
|
|
117
|
+
bestKey = key;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
if (!best) return null;
|
|
121
|
+
entries.delete(bestKey);
|
|
122
|
+
return { key: bestKey, entry: best };
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* File an interruption. `chain` carries the count forward from the entry this
|
|
127
|
+
* turn resumed, so `interruptions` reports how many times one job has been cut
|
|
128
|
+
* off — visible in the note and useful when deciding to raise the horizon.
|
|
129
|
+
*/
|
|
130
|
+
function record(key, { rounds, tokens, note, chain } = {}, now = Date.now()) {
|
|
131
|
+
prune(now);
|
|
132
|
+
const prior = entries.get(key);
|
|
133
|
+
const entry = {
|
|
134
|
+
rounds: Number.isFinite(rounds) ? rounds : 0,
|
|
135
|
+
tokens: Number.isFinite(tokens) ? tokens : 0,
|
|
136
|
+
note: typeof note === 'string' ? note : '',
|
|
137
|
+
at: now,
|
|
138
|
+
interruptions: (Number.isFinite(chain) ? chain : prior && prior.interruptions) || 0,
|
|
139
|
+
};
|
|
140
|
+
entry.interruptions += 1;
|
|
141
|
+
entries.delete(key);
|
|
142
|
+
entries.set(key, entry); // re-insert so Map order stays age order
|
|
143
|
+
prune(now);
|
|
144
|
+
return entry;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/** The instruction that turns a cold start into a continuation. */
|
|
148
|
+
function resumePreamble(entry) {
|
|
149
|
+
const rounds = entry && entry.rounds ? entry.rounds : 'the previous turn\'s';
|
|
150
|
+
const tokens = entry && entry.tokens ? ` (${entry.tokens.toLocaleString()} tokens)` : '';
|
|
151
|
+
return (
|
|
152
|
+
`[session resume] The previous turn in this conversation was cut off at ${rounds} tool rounds${tokens} ` +
|
|
153
|
+
'because it reached the round horizon, so the work it started is unfinished. Continue that work now: ' +
|
|
154
|
+
'do not restart, do not re-do steps that already completed, and do not treat the interruption as a ' +
|
|
155
|
+
'failure to report. Pick up the next unfinished step and finish with a written answer.'
|
|
156
|
+
);
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/** Read-only snapshot for the UI (and for tests). */
|
|
160
|
+
function state(now = Date.now()) {
|
|
161
|
+
prune(now);
|
|
162
|
+
return {
|
|
163
|
+
size: entries.size,
|
|
164
|
+
ttlMs: TTL_MS,
|
|
165
|
+
maxEntries: MAX_ENTRIES,
|
|
166
|
+
entries: [...entries].map(([key, entry]) => ({ key, ...entry })),
|
|
167
|
+
};
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
/** Test hook: forget everything. */
|
|
171
|
+
function reset() {
|
|
172
|
+
entries.clear();
|
|
173
|
+
return entries.size;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
module.exports = {
|
|
177
|
+
TTL_MS,
|
|
178
|
+
MAX_ENTRIES,
|
|
179
|
+
keyFor,
|
|
180
|
+
bonus,
|
|
181
|
+
peek,
|
|
182
|
+
take,
|
|
183
|
+
adopt,
|
|
184
|
+
record,
|
|
185
|
+
resumePreamble,
|
|
186
|
+
state,
|
|
187
|
+
reset,
|
|
188
|
+
};
|
package/main.js
CHANGED
|
@@ -1357,6 +1357,11 @@ function createQueueDispatch({
|
|
|
1357
1357
|
draining,
|
|
1358
1358
|
stopping: stopRequested,
|
|
1359
1359
|
defaultCwd: process.cwd(),
|
|
1360
|
+
// What a task with no model of its own actually runs on. The queue is
|
|
1361
|
+
// Aegis Cloud only, so this is a pooled tier id — and the card prints it,
|
|
1362
|
+
// because "which model is this drain spending on" was previously
|
|
1363
|
+
// unanswerable from the UI.
|
|
1364
|
+
defaultModel: autonomous.resolveModel({ env }),
|
|
1360
1365
|
queueFile: queue.queuePath(env),
|
|
1361
1366
|
runs: queue.readRuns(env, { limit: 20 }),
|
|
1362
1367
|
};
|
|
@@ -1385,6 +1390,13 @@ function createQueueDispatch({
|
|
|
1385
1390
|
if (!isExistingDir(cwd)) return { error: `not a directory: ${cwd}` };
|
|
1386
1391
|
|
|
1387
1392
|
const model = String(p.model == null ? '' : p.model).trim().slice(0, 120) || null;
|
|
1393
|
+
// Aegis Cloud only (autonomous.js modelRefusal): the worker's requests go
|
|
1394
|
+
// out as `class: 'aegis'`, so a model the pool does not serve is a task that
|
|
1395
|
+
// fails inside a drain. Refused HERE so the card gets the reason at the
|
|
1396
|
+
// moment of queueing instead of a red row later. queue.addTask enforces the
|
|
1397
|
+
// same rule for the CLI and for hand-written queue files.
|
|
1398
|
+
const modelRefusal = model ? autonomous.modelRefusal(model) : '';
|
|
1399
|
+
if (modelRefusal) return { error: modelRefusal };
|
|
1388
1400
|
const effortRaw = String(p.effort == null ? '' : p.effort).trim().toLowerCase();
|
|
1389
1401
|
return {
|
|
1390
1402
|
opts: {
|
|
@@ -1394,6 +1406,12 @@ function createQueueDispatch({
|
|
|
1394
1406
|
effort: QUEUE_EFFORTS.has(effortRaw) ? effortRaw : null,
|
|
1395
1407
|
workers: clampInt(p.workers, 1, QUEUE_MAX_WORKERS),
|
|
1396
1408
|
maxRounds: clampInt(p.maxRounds, 1, QUEUE_MAX_ROUNDS),
|
|
1409
|
+
// The pooled-brain fan-out is opt-in (autonomous.js resolveFanout):
|
|
1410
|
+
// it costs about workers+1 full reasoning calls on one task, so the
|
|
1411
|
+
// card has to ask for it per task. queue.addTask stores the inverse
|
|
1412
|
+
// (`singlePass`), which is also the field a hand-written queue file
|
|
1413
|
+
// or the CLI uses.
|
|
1414
|
+
singlePass: !(p.fanout === true || p.autonomous === true),
|
|
1397
1415
|
commit: Boolean(p.commit),
|
|
1398
1416
|
source: 'desktop',
|
|
1399
1417
|
},
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "aegis-desktop",
|
|
3
3
|
"productName": "AEGIS Desktop",
|
|
4
|
-
"version": "0.6.
|
|
4
|
+
"version": "0.6.1",
|
|
5
5
|
"description": "Thin Electron host for AEGIS — a local chat UI over the shared client/aegis.js transport. Ships transport + UI only; engine logic stays server-side.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "AEGIS Code",
|
|
@@ -29,7 +29,7 @@
|
|
|
29
29
|
"icon": "node scripts/generate-icon.mjs",
|
|
30
30
|
"dist": "npm run predist && electron-builder",
|
|
31
31
|
"dist:dir": "npm run predist && electron-builder --dir",
|
|
32
|
-
"check": "node --check main.js && node --check preload.js && node --check renderer/app.js && node --check renderer/quick.js && node --check renderer/budget.js && node --check renderer/usage.js && node --check renderer/stream-policy.js && node --check renderer/transcript-view.js && node --check renderer/markdown.js && node --check renderer/vendor/aegis-highlight.js && node --check scripts/predist.mjs && node --check scripts/generate-icon.mjs && node --check lib/local/context.js && node --check lib/local/providers.js && node --check lib/local/ollama.js && node --check lib/local/engine.js && node --check lib/local/tools.js && node --check lib/local/prompt.js && node --check lib/local/shell.js && node --check lib/local/agents.js && node --check lib/local/queue.js && node --check lib/local/autonomous.js && node --check lib/settings.js && node --check lib/sync/sessions.js && node --check lib/sync/memory-queue.js && node --check lib/window-state.js && node --check lib/deep-link.js && node --check lib/quick-launcher.js && node --check bin/aegis.js",
|
|
32
|
+
"check": "node --check main.js && node --check preload.js && node --check renderer/app.js && node --check renderer/quick.js && node --check renderer/budget.js && node --check renderer/usage.js && node --check renderer/stream-policy.js && node --check renderer/transcript-view.js && node --check renderer/markdown.js && node --check renderer/vendor/aegis-highlight.js && node --check scripts/predist.mjs && node --check scripts/generate-icon.mjs && node --check lib/local/context.js && node --check lib/local/providers.js && node --check lib/local/ollama.js && node --check lib/local/engine.js && node --check lib/local/session-rounds.js && node --check lib/local/tools.js && node --check lib/local/prompt.js && node --check lib/local/shell.js && node --check lib/local/agents.js && node --check lib/local/queue.js && node --check lib/local/autonomous.js && node --check lib/settings.js && node --check lib/sync/sessions.js && node --check lib/sync/memory-queue.js && node --check lib/window-state.js && node --check lib/deep-link.js && node --check lib/quick-launcher.js && node --check bin/aegis.js",
|
|
33
33
|
"test:shell": "node ../test/desktop-shell.mjs",
|
|
34
34
|
"test:model": "node ../test/model-dispatch.mjs",
|
|
35
35
|
"test:budget": "node ../test/budget.test.mjs",
|
package/renderer/app.js
CHANGED
|
@@ -891,6 +891,12 @@ function queueStatusMark(status) {
|
|
|
891
891
|
/** The outcome line under a task: what happened, or why it did not. */
|
|
892
892
|
function queueMetaText(item) {
|
|
893
893
|
const bits = [item.status];
|
|
894
|
+
// Which model and which shape, per row. Both are spend facts the card used
|
|
895
|
+
// to hide: an unpinned task runs main.js's defaultModel (a pooled tier), and
|
|
896
|
+
// a task is a single pass unless it was queued with `fanout` — the difference
|
|
897
|
+
// is roughly (workers + 1) full reasoning calls.
|
|
898
|
+
if (item.model) bits.push(item.model);
|
|
899
|
+
bits.push(item.singlePass === false || item.autonomous === true ? 'fan-out' : 'single pass');
|
|
894
900
|
if (item.status === 'error') {
|
|
895
901
|
bits.push(item.error ? String(item.error) : 'failed — no reason recorded');
|
|
896
902
|
} else if (item.status === 'done') {
|
|
@@ -977,6 +983,10 @@ function renderQueueState(state) {
|
|
|
977
983
|
if (state.running) bits.push(`running #${state.running.id}`);
|
|
978
984
|
else if (queueDraining) bits.push('draining…');
|
|
979
985
|
bits.push(`${state.pending || 0} pending`);
|
|
986
|
+
// What the queue can spend on at all: Aegis Cloud, and which tier an
|
|
987
|
+
// unpinned task lands on (main.js snapshot().defaultModel resolves it the
|
|
988
|
+
// same way the worker does, so the card cannot disagree with the bill).
|
|
989
|
+
if (state.defaultModel) bits.push(`Aegis Cloud · ${state.defaultModel}`);
|
|
980
990
|
if (queueDraining && state.stopping) bits.push('stopping after this task');
|
|
981
991
|
if (els.queueHint) {
|
|
982
992
|
els.queueHint.textContent = [queueNote, bits.join(' · ')].filter(Boolean).join(' · ');
|
package/renderer/index.html
CHANGED
|
@@ -142,14 +142,21 @@
|
|
|
142
142
|
for="autonomous-toggle"
|
|
143
143
|
id="autonomous-toggle-wrap"
|
|
144
144
|
hidden
|
|
145
|
-
title="
|
|
145
|
+
title="Fan THIS chat turn out to several Aegis Cloud reasoning workers and synthesize their findings into one answer. Aegis Cloud only, and off by default because it costs about (workers + 1) full model calls per turn. This is a per-turn setting for the thread; it does not govern the unattended queue below, which runs each queued task on a single pass unless that task asks for the fan-out."
|
|
146
146
|
>
|
|
147
147
|
<input type="checkbox" id="autonomous-toggle" />
|
|
148
|
-
<span>
|
|
148
|
+
<span>fan out this turn</span>
|
|
149
149
|
</label>
|
|
150
150
|
<div id="autonomous-controls" class="autonomous-controls" hidden>
|
|
151
151
|
<label for="autonomous-workers">Workers</label>
|
|
152
|
-
<input
|
|
152
|
+
<input
|
|
153
|
+
type="number"
|
|
154
|
+
id="autonomous-workers"
|
|
155
|
+
min="1"
|
|
156
|
+
max="8"
|
|
157
|
+
step="1"
|
|
158
|
+
title="How many reasoning workers this turn fans out to. Every worker is a separate full model call on the same tier, so the turn costs roughly this number plus one times a normal one — higher is not automatically better."
|
|
159
|
+
/>
|
|
153
160
|
</div>
|
|
154
161
|
<p class="hint" id="model-hint"></p>
|
|
155
162
|
</section>
|