aegis-desktop 0.5.5 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/local/autonomous.js +745 -0
- package/lib/local/engine.js +62 -7
- package/lib/local/git-scope.js +163 -9
- package/lib/local/queue.js +549 -0
- package/lib/local/session-rounds.js +188 -0
- package/main.js +852 -29
- package/package.json +2 -2
- package/preload.js +93 -0
- package/renderer/app.js +384 -3
- package/renderer/index.html +55 -3
- package/vendor/update.js +76 -11
|
@@ -0,0 +1,745 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* autonomous.js — the unattended worker: what actually runs a queued task.
|
|
5
|
+
*
|
|
6
|
+
* `queue.js` stores tasks; this module is the half that turns one into work.
|
|
7
|
+
* It owns three things no interactive turn has to think about:
|
|
8
|
+
*
|
|
9
|
+
* 1. THE OPERATING DIRECTIVE. An interactive turn can end on "let me check…"
|
|
10
|
+
* or a question, because a human will read it and answer. A queued task
|
|
11
|
+
* ends there only if nobody ever looks — so every queued turn runs under a
|
|
12
|
+
* directive that says, in the model's own context: nobody is watching,
|
|
13
|
+
* decide, act, verify, report; do not hand the work back. The round horizon
|
|
14
|
+
* is stated too, so the model budgets its exploration instead of meeting
|
|
15
|
+
* the cap by accident.
|
|
16
|
+
*
|
|
17
|
+
* 2. THE CARRY-OVER DIGEST. The AEGIS API is stateless per request, so a
|
|
18
|
+
* drain is not literally one long conversation — but treating ten queued
|
|
19
|
+
* tasks as ten unrelated strangers makes the ninth re-derive what the
|
|
20
|
+
* second already learned. Each finished task appends a one-line digest
|
|
21
|
+
* (what it was, what it touched, whether it was verified) to a carry string
|
|
22
|
+
* that rides in front of the next task's prompt. Bounded and one line per
|
|
23
|
+
* task: it is a briefing, not a transcript, and it must never grow into the
|
|
24
|
+
* context it is meant to save.
|
|
25
|
+
*
|
|
26
|
+
* 3. ATTRIBUTED COMMITS. A queue drain can run in a checkout somebody else is
|
|
27
|
+
* editing. `git add -A && git commit` in that situation commits THEIR
|
|
28
|
+
* half-finished work under our message — the failure `git-scope.js` was
|
|
29
|
+
* written for. So a task's `commit` flag commits only paths that this task's
|
|
30
|
+
* own tool layer wrote (`writeFile`/`editFile` frames, recorded as they
|
|
31
|
+
* land); everything else dirty is left alone, whether it was dirty before
|
|
32
|
+
* the task (a peer's in-flight work) or a path this task never wrote.
|
|
33
|
+
*
|
|
34
|
+
* Two conditions, not one, and the second is not academic: commit 96fb64f
|
|
35
|
+
* swept `desktop/electron-builder.yml` — a file another live session was
|
|
36
|
+
* editing at that moment — into an unrelated commit, because "dirty before
|
|
37
|
+
* and byte-identical now" cannot see a concurrent writer who edits DURING
|
|
38
|
+
* the task window. Only positive attribution can, so a path this task never
|
|
39
|
+
* wrote is never staged, and the paths left behind are reported.
|
|
40
|
+
*
|
|
41
|
+
* It does NOT own the model call itself: the caller passes the local engine
|
|
42
|
+
* (`desktop/lib/local/engine.js`), which is the same tool loop the GUI and the
|
|
43
|
+
* CLI chat in. One tool loop, three callers, no drift.
|
|
44
|
+
*
|
|
45
|
+
* THE APPROVAL GATE IS THE CALLER'S JOB. There is nobody to answer an approval
|
|
46
|
+
* card in an unattended run, so the engine must be constructed with the gate
|
|
47
|
+
* off (`getConfirmMode: () => false`). Rather than trust that, the worker
|
|
48
|
+
* watches the turn's event stream: an approval request means a card nobody can
|
|
49
|
+
* click, so it is reported as a failed task instead of a hang.
|
|
50
|
+
*/
|
|
51
|
+
|
|
52
|
+
const queue = require('./queue.js');
|
|
53
|
+
const gitScope = require('./git-scope.js');
|
|
54
|
+
|
|
55
|
+
/** Default tool-round horizon for an unattended turn (matches the engine's). */
|
|
56
|
+
const DEFAULT_ROUNDS = 40;
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* The engine tools that report the file they write, and the argument holding
|
|
60
|
+
* it. This set is the attribution record for a task's commit: `exec` is absent
|
|
61
|
+
* on purpose — a shell command names no paths, so a file it creates cannot be
|
|
62
|
+
* attributed to this task, and guessing "anything new is ours" is precisely how
|
|
63
|
+
* a concurrent writer's file gets committed.
|
|
64
|
+
*/
|
|
65
|
+
const WRITE_TOOLS = new Set(['writeFile', 'editFile']);
|
|
66
|
+
|
|
67
|
+
/** The path a file-writing tool call names, in either spelling the engine uses. */
|
|
68
|
+
function writtenPath(tool) {
|
|
69
|
+
const args = (tool && tool.args) || {};
|
|
70
|
+
const p = args.file_path || args.path;
|
|
71
|
+
return p ? String(p) : '';
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* The default AEGIS Cloud model for autonomous work: the pooled brain, which
|
|
76
|
+
* is the tier the server can fan out to multiple reasoning workers and
|
|
77
|
+
* synthesise. `nexus-brain` is the canonical id the catalog itself prefers
|
|
78
|
+
* (the other tier spellings are aliases of it — see filterAegisCatalog in
|
|
79
|
+
* engine.js).
|
|
80
|
+
*
|
|
81
|
+
* The model id is only the tier. It is NOT what makes an autonomous task
|
|
82
|
+
* expensive — the fan-out is, and the fan-out is opt-in per task (see
|
|
83
|
+
* resolveFanout below). This used to be documented the other way round ("the
|
|
84
|
+
* pooled brain ... autonomous tasks are exactly the ones worth that spend"),
|
|
85
|
+
* and the worker sent `autonomous: true` on every queued task, so the most
|
|
86
|
+
* expensive shape of the most expensive tier ran on one-line tasks too.
|
|
87
|
+
*/
|
|
88
|
+
const DEFAULT_MODEL = 'nexus-brain';
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* THE QUEUE RUNS ON AEGIS CLOUD, AND NOTHING ELSE.
|
|
92
|
+
*
|
|
93
|
+
* A queued task is billed to the AEGIS pool and every turn it makes goes out
|
|
94
|
+
* with `class: 'aegis'` (see runTask below) — the pool is the only backend this
|
|
95
|
+
* worker can reach. So a model id here has to be one the *pool* serves. A
|
|
96
|
+
* direct-provider id is not a cheaper option the queue could fall back to; it
|
|
97
|
+
* is a request the pool cannot honour, or worse, a per-provider spelling
|
|
98
|
+
* (`anthropic`, `groq`, …) that quietly pins one upstream instead of letting
|
|
99
|
+
* the pool auto-route across whichever providers hold a live key.
|
|
100
|
+
*
|
|
101
|
+
* The accept-list is therefore the pooled-brain tier family — the one entry
|
|
102
|
+
* engine.js's filterAegisCatalog offers for the Aegis Cloud class, plus the
|
|
103
|
+
* `-smart`/`-neo` tier spellings the server still serves as aliases of it. This
|
|
104
|
+
* mirrors selectBrainEntry() there rather than re-deriving "anything starting
|
|
105
|
+
* with nexus-": `nexus-fast` is not a tier the catalog has ever served, and
|
|
106
|
+
* accepting a made-up id means a queued task that fails at the server after
|
|
107
|
+
* being picked up, or runs on a tier nobody chose.
|
|
108
|
+
*/
|
|
109
|
+
const AEGIS_MODEL_RE = /^(?:nexus|aegis)-brain(?:-(?:smart|neo))?$/;
|
|
110
|
+
const AEGIS_MODEL_IDS = Object.freeze(['nexus-brain', 'aegis-brain']);
|
|
111
|
+
|
|
112
|
+
/** True when `id` names an AEGIS Cloud pooled-brain tier (the queue's only models). */
|
|
113
|
+
function isAegisModel(id) {
|
|
114
|
+
return AEGIS_MODEL_RE.test(String(id == null ? '' : id).trim());
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Why a STATED model id cannot be queued, or '' when it can. Blank is not a
|
|
119
|
+
* refusal — "no pick" is the default model, which resolveModel supplies.
|
|
120
|
+
*
|
|
121
|
+
* The message names the pool, because the failure it prevents ("queued on
|
|
122
|
+
* claude-sonnet-4, ran on — or was billed to — something else") is invisible
|
|
123
|
+
* otherwise: `class: 'aegis'` would be sent with an id the server does not
|
|
124
|
+
* serve, and the task would come back as an opaque error minutes later.
|
|
125
|
+
*/
|
|
126
|
+
function modelRefusal(id) {
|
|
127
|
+
const stated = String(id == null ? '' : id).trim();
|
|
128
|
+
if (!stated || isAegisModel(stated)) return '';
|
|
129
|
+
return (
|
|
130
|
+
`the autonomous queue runs Aegis Cloud models only (${DEFAULT_MODEL}, ` +
|
|
131
|
+
`${AEGIS_MODEL_IDS.join('/')} aliases); "${stated}" is not one`
|
|
132
|
+
);
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/** Phrase match for "work autonomously" in a prompt or a queued task. */
|
|
136
|
+
const AUTONOMOUS_REQUEST_RE =
|
|
137
|
+
/\bautonomously\b|\bon your own\b|\bwithout asking\b|\bend[- ]to[- ]end\b|\bno (?:more )?questions\b|\bfully autonomous\b|\bqueue it\b/i;
|
|
138
|
+
|
|
139
|
+
/** True when the user is asking for unattended execution. */
|
|
140
|
+
function isAutonomousRequest(text) {
|
|
141
|
+
return AUTONOMOUS_REQUEST_RE.test(String(text || ''));
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/** The round horizon, honouring AEGIS_AUTONOMOUS_MAX_ROUNDS. */
|
|
145
|
+
function maxRounds(env = process.env, stated) {
|
|
146
|
+
const n = Number.parseInt(stated, 10);
|
|
147
|
+
if (Number.isFinite(n) && n > 0) return n;
|
|
148
|
+
const raw = Number.parseInt((env && env.AEGIS_AUTONOMOUS_MAX_ROUNDS) || '', 10);
|
|
149
|
+
return Number.isFinite(raw) && raw > 0 ? raw : DEFAULT_ROUNDS;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/**
|
|
153
|
+
* Make the horizon real for the engine, then put the environment back.
|
|
154
|
+
*
|
|
155
|
+
* A `maxRounds` field in the chat payload would be dead code: engine.js reads
|
|
156
|
+
* the knob from `process.env` at turn time (engine.js:1024,
|
|
157
|
+
* `AEGIS_AUTONOMOUS_MAX_ROUNDS`). Env, unlike a payload key, is process-wide, so
|
|
158
|
+
* the restore has to wait for a promise to settle (a turn reads the knob many
|
|
159
|
+
* rounds after the call is made) — hence the thenable branch below.
|
|
160
|
+
*/
|
|
161
|
+
function withRoundHorizon(rounds, env, fn) {
|
|
162
|
+
const key = 'AEGIS_AUTONOMOUS_MAX_ROUNDS';
|
|
163
|
+
// Both objects: the caller's env carries the path/model defaults this worker
|
|
164
|
+
// reads, and `process.env` is where engine.js reads the round cap. A test that
|
|
165
|
+
// passes its own env still gets the knob on that object, and a real drain gets
|
|
166
|
+
// it where the engine actually looks.
|
|
167
|
+
const targets = [];
|
|
168
|
+
for (const t of [env, process.env]) {
|
|
169
|
+
if (t && targets.indexOf(t) === -1) targets.push(t);
|
|
170
|
+
}
|
|
171
|
+
const saved = targets.map((t) => ({
|
|
172
|
+
target: t,
|
|
173
|
+
had: Object.prototype.hasOwnProperty.call(t, key),
|
|
174
|
+
prev: t[key],
|
|
175
|
+
}));
|
|
176
|
+
try {
|
|
177
|
+
for (const t of targets) t[key] = String(rounds);
|
|
178
|
+
} catch {
|
|
179
|
+
/* a frozen/sealed env object is the caller's choice; the payload carries it too */
|
|
180
|
+
}
|
|
181
|
+
const restore = () => {
|
|
182
|
+
for (const s of saved) {
|
|
183
|
+
try {
|
|
184
|
+
if (s.had) s.target[key] = s.prev;
|
|
185
|
+
else delete s.target[key];
|
|
186
|
+
} catch {
|
|
187
|
+
/* nothing to restore on an object we could not write */
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
};
|
|
191
|
+
let out;
|
|
192
|
+
try {
|
|
193
|
+
out = fn();
|
|
194
|
+
} catch (e) {
|
|
195
|
+
restore();
|
|
196
|
+
throw e;
|
|
197
|
+
}
|
|
198
|
+
if (out && typeof out.then === 'function') {
|
|
199
|
+
return out.then(
|
|
200
|
+
(value) => {
|
|
201
|
+
restore();
|
|
202
|
+
return value;
|
|
203
|
+
},
|
|
204
|
+
(err) => {
|
|
205
|
+
restore();
|
|
206
|
+
throw err;
|
|
207
|
+
}
|
|
208
|
+
);
|
|
209
|
+
}
|
|
210
|
+
restore();
|
|
211
|
+
return out;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/**
|
|
215
|
+
* The model an autonomous task runs on: an explicit pick wins, then an AEGIS
|
|
216
|
+
* Cloud pin in the environment (so a systemd timer can choose a tier), then the
|
|
217
|
+
* pooled brain. Never the interactive session's model — a queue survives the
|
|
218
|
+
* session that queued it, so it cannot inherit that session's choice.
|
|
219
|
+
*
|
|
220
|
+
* TWO DIFFERENT TREATMENTS FOR TWO DIFFERENT SOURCES, on purpose:
|
|
221
|
+
*
|
|
222
|
+
* - a pick that came from the TASK is returned verbatim, even when it is
|
|
223
|
+
* wrong. Substituting a correct model for a stated one is how a queue
|
|
224
|
+
* "runs on nexus-brain" while the file says otherwise; the caller refuses
|
|
225
|
+
* it out loud instead (modelRefusal, queue.addTask, and the pre-flight in
|
|
226
|
+
* runTask).
|
|
227
|
+
* - a non-Aegis value in the ENVIRONMENT is skipped, because AEGIS_MODEL is
|
|
228
|
+
* shared with the interactive surfaces (which run direct providers), so a
|
|
229
|
+
* stray value there is not a statement about the queue. Refusing every
|
|
230
|
+
* task over it would break drains for a reason that is not the task's
|
|
231
|
+
* fault; a fallback to the pooled brain keeps the drain honest and on-cloud.
|
|
232
|
+
*/
|
|
233
|
+
function resolveModel({ model, env } = {}) {
|
|
234
|
+
const e = env || process.env;
|
|
235
|
+
const picked = String(model || '').trim();
|
|
236
|
+
if (picked) return picked;
|
|
237
|
+
const fromEnv = String(e.AEGIS_AUTONOMOUS_MODEL || e.AEGIS_MODEL || '').trim();
|
|
238
|
+
return isAegisModel(fromEnv) ? fromEnv : DEFAULT_MODEL;
|
|
239
|
+
}
|
|
240
|
+
|
|
241
|
+
/**
|
|
242
|
+
* The environment pin the queue had to ignore, or '' when there was none.
|
|
243
|
+
*
|
|
244
|
+
* Reported (not silent) because an operator who exported
|
|
245
|
+
* AEGIS_AUTONOMOUS_MODEL=deepseek-v4-flash asked for a model and is not getting
|
|
246
|
+
* it: the queue falls back to the pool, and the one place that says so is this
|
|
247
|
+
* string, which runTask emits as a note and the desktop card shows.
|
|
248
|
+
*/
|
|
249
|
+
function ignoredEnvModel({ model, env } = {}) {
|
|
250
|
+
if (String(model || '').trim()) return ''; // the task's own pick is what counts
|
|
251
|
+
const e = env || process.env;
|
|
252
|
+
const fromEnv = String(e.AEGIS_AUTONOMOUS_MODEL || e.AEGIS_MODEL || '').trim();
|
|
253
|
+
if (!fromEnv || isAegisModel(fromEnv)) return '';
|
|
254
|
+
const which = String(e.AEGIS_AUTONOMOUS_MODEL || '').trim() ? 'AEGIS_AUTONOMOUS_MODEL' : 'AEGIS_MODEL';
|
|
255
|
+
return `${which}="${fromEnv}" is not an Aegis Cloud model — running on ${DEFAULT_MODEL} instead`;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
/**
|
|
259
|
+
* Whether a queued task runs the pooled-brain worker fan-out (aegis1
|
|
260
|
+
* services/pool_brain.py) or one plain turn on the same tier.
|
|
261
|
+
*
|
|
262
|
+
* COST IS THE REASON THIS IS OPT-IN. The fan-out is the single biggest
|
|
263
|
+
* multiplier this app can put on a bill: pool_brain spawns up to `workers`
|
|
264
|
+
* reasoning workers plus a synthesis pass, re-sends the task context to every
|
|
265
|
+
* one of them, and sizes each from the same effort ladder. A 3-worker
|
|
266
|
+
* high-effort task is therefore roughly four full reasoning calls against a
|
|
267
|
+
* 65536-token ladder, where the identical task single-pass is one call on the
|
|
268
|
+
* medium rung. The fan-out earns that on genuinely open-ended investigation
|
|
269
|
+
* ("why did X regress across this repo"); it is pure waste on a task that
|
|
270
|
+
* already names the file to edit.
|
|
271
|
+
*
|
|
272
|
+
* Precedence: the task's own `autonomous: true` (or `singlePass: false`, the
|
|
273
|
+
* explicit "fan me out") wins, then AEGIS_AUTONOMOUS_FANOUT=1 in the
|
|
274
|
+
* environment, else single pass.
|
|
275
|
+
*/
|
|
276
|
+
function resolveFanout(item = {}, env = process.env) {
|
|
277
|
+
const it = item || {};
|
|
278
|
+
if (it.autonomous === true || it.singlePass === false) return true;
|
|
279
|
+
const e = env || process.env;
|
|
280
|
+
return /^(1|true|yes|on)$/i.test(String(e.AEGIS_AUTONOMOUS_FANOUT || '').trim());
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Effort rung for an unattended turn.
|
|
285
|
+
*
|
|
286
|
+
* The rung is a spend knob, not a quality slider: the pooled class sizes its
|
|
287
|
+
* whole budget ladder from it (aegis1 services/pool_brain.py pass_budgets:
|
|
288
|
+
* low/medium/high -> 16384/32768/65536 tokens TOTAL across the fan-out), and
|
|
289
|
+
* the engine uses it for any model that reasons against its own output budget.
|
|
290
|
+
* `high` is the right rung for a fan-out — it is what buys a synthesis pass
|
|
291
|
+
* worth reading — but on a single pass it is a 2x over medium for budget
|
|
292
|
+
* nobody reads, so the default follows the shape of the task rather than
|
|
293
|
+
* always being the most expensive rung. An explicit pick (item.effort) or
|
|
294
|
+
* AEGIS_AUTONOMOUS_EFFORT still wins outright.
|
|
295
|
+
*/
|
|
296
|
+
function resolveEffort({ effort, env, fanout } = {}) {
|
|
297
|
+
const e = env || process.env;
|
|
298
|
+
const stated = String(effort || e.AEGIS_AUTONOMOUS_EFFORT || '').trim();
|
|
299
|
+
if (stated) return stated;
|
|
300
|
+
return fanout ? 'high' : 'medium';
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/**
|
|
304
|
+
* The operating directive injected as the autonomous turn's prompt preamble.
|
|
305
|
+
* Ported verbatim in spirit from aegiscodex-dev/src/autonomous.js so both
|
|
306
|
+
* clients behave the same; duplicated rather than vendored because the plugin
|
|
307
|
+
* hosts no ESM build of that module.
|
|
308
|
+
*/
|
|
309
|
+
function autonomousDirective(rounds = DEFAULT_ROUNDS) {
|
|
310
|
+
return [
|
|
311
|
+
'# Autonomous mode',
|
|
312
|
+
`You are running autonomously, not in a conversation: there is no user to answer a question or approve a plan. You have up to ${rounds} tool rounds this turn; use as many as the task needs.`,
|
|
313
|
+
'Work the task end to end:',
|
|
314
|
+
'1. Plan in one line, then start acting in the same turn — never end a turn on a plan.',
|
|
315
|
+
'2. Read only what you need to make the change (no fishing through the repo).',
|
|
316
|
+
'3. Make the change with Write/Edit/Bash, then VERIFY it: re-read the result and run the relevant test/command.',
|
|
317
|
+
'4. If verification fails, fix it and verify again — loop until it passes or you are genuinely blocked.',
|
|
318
|
+
'5. Do not ask for permission, do not hand the work back, do not stop at "let me check…".',
|
|
319
|
+
'6. Finish with a short report: what changed (file paths), the command you ran to verify, its result, and any remaining blocker.',
|
|
320
|
+
].join('\n');
|
|
321
|
+
}
|
|
322
|
+
|
|
323
|
+
/**
|
|
324
|
+
* The digest line one finished task contributes to the next task's briefing.
|
|
325
|
+
* Deliberately tiny: task text clipped, outcome as a mark, files as basenames.
|
|
326
|
+
*/
|
|
327
|
+
function digestLine(item, result) {
|
|
328
|
+
const task = String(item.task || '').replace(/\s+/g, ' ').trim().slice(0, 120);
|
|
329
|
+
const mark = result && result.ok ? '✓' : '✗';
|
|
330
|
+
const files = (result && Array.isArray(result.files) && result.files.slice(0, 6)) || [];
|
|
331
|
+
const where = files.length ? ` — touched: ${files.join(', ')}${result.files.length > files.length ? ', …' : ''}` : '';
|
|
332
|
+
const why = !result || result.ok ? '' : ` — ${String((result && result.error) || 'failed').slice(0, 120)}`;
|
|
333
|
+
return `- #${item.id} ${mark} ${task}${where}${why}`;
|
|
334
|
+
}
|
|
335
|
+
|
|
336
|
+
/** Keep only the last `keep` digest lines, so the briefing cannot grow forever. */
|
|
337
|
+
function appendDigest(carry, line, { keep = 6 } = {}) {
|
|
338
|
+
const lines = String(carry || '')
|
|
339
|
+
.split('\n')
|
|
340
|
+
.filter(Boolean)
|
|
341
|
+
.concat(line);
|
|
342
|
+
return lines.slice(-keep).join('\n');
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/**
|
|
346
|
+
* The prompt a queued task actually sends: the directive, the briefing from
|
|
347
|
+
* earlier tasks in the same drain (when there is one), then the task itself.
|
|
348
|
+
*/
|
|
349
|
+
function taskPrompt(item, { carry = '', rounds = DEFAULT_ROUNDS } = {}) {
|
|
350
|
+
const parts = [autonomousDirective(rounds)];
|
|
351
|
+
if (carry) {
|
|
352
|
+
parts.push(
|
|
353
|
+
'# Earlier tasks in this run\n' +
|
|
354
|
+
'Work already done by this queue — do not redo it, build on it:\n' +
|
|
355
|
+
carry
|
|
356
|
+
);
|
|
357
|
+
}
|
|
358
|
+
parts.push('# Task\n' + String(item.task || '').trim());
|
|
359
|
+
return parts.join('\n\n');
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
/** The assistant's visible text out of an OpenAI-shaped engine result. */
|
|
363
|
+
function assistantText(res) {
|
|
364
|
+
if (!res || typeof res !== 'object') return '';
|
|
365
|
+
const choice = Array.isArray(res.choices) ? res.choices[0] : null;
|
|
366
|
+
const content = choice && choice.message ? choice.message.content : res.content;
|
|
367
|
+
if (typeof content === 'string') return content;
|
|
368
|
+
if (Array.isArray(content)) {
|
|
369
|
+
return content
|
|
370
|
+
.map((p) => (typeof p === 'string' ? p : p && typeof p.text === 'string' ? p.text : ''))
|
|
371
|
+
.join('');
|
|
372
|
+
}
|
|
373
|
+
return '';
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
/** A readable one-liner for whatever a failed turn threw. */
|
|
377
|
+
function errorText(err) {
|
|
378
|
+
if (!err) return 'unknown error';
|
|
379
|
+
if (typeof err === 'string') return err;
|
|
380
|
+
return String(err.message || err.error || err);
|
|
381
|
+
}
|
|
382
|
+
|
|
383
|
+
/**
|
|
384
|
+
* Build the worker.
|
|
385
|
+
*
|
|
386
|
+
* @param {object} opts
|
|
387
|
+
* @param {object} opts.engine The local engine (createLocalEngine()). MUST be
|
|
388
|
+
* constructed with the approval gate off — see the module header.
|
|
389
|
+
* @param {object} [opts.env] Environment for paths/model defaults (tests pass a temp one).
|
|
390
|
+
* @param {function} [opts.log] `(event) => void` progress sink; also forwarded
|
|
391
|
+
* the turn's own `{delta}` / `{tool}` frames.
|
|
392
|
+
* @param {object} [opts.git] Injectable git-scope (tests).
|
|
393
|
+
* @param {function} [opts.now] Clock.
|
|
394
|
+
*/
|
|
395
|
+
function createQueueWorker({ engine, env = process.env, log = () => {}, git = gitScope, now = Date.now } = {}) {
|
|
396
|
+
if (!engine || typeof engine.chat !== 'function') {
|
|
397
|
+
throw new Error('autonomous: a local engine with chat() is required');
|
|
398
|
+
}
|
|
399
|
+
const emit = (event) => {
|
|
400
|
+
try {
|
|
401
|
+
log(event);
|
|
402
|
+
} catch {
|
|
403
|
+
/* a progress sink that throws (a closed stdout, a destroyed window) must
|
|
404
|
+
never take the task down with it — the work is the point, not the log */
|
|
405
|
+
}
|
|
406
|
+
};
|
|
407
|
+
|
|
408
|
+
/**
|
|
409
|
+
* Run ONE queued task. Never throws: a failed task is data (so the drain can
|
|
410
|
+
* decide whether to continue), not an exception that aborts the queue.
|
|
411
|
+
*/
|
|
412
|
+
async function runTask(item, { carry = '', commit } = {}) {
|
|
413
|
+
const cwd = item.cwd || process.cwd();
|
|
414
|
+
// Aegis Cloud or nothing — checked BEFORE the turn, not at the server. An
|
|
415
|
+
// item whose model is not a pooled tier (a hand-edited queue file, a
|
|
416
|
+
// `--model` the CLI accepted before this rule existed, another host's
|
|
417
|
+
// older build) would otherwise go out as `class: 'aegis'` with an id the
|
|
418
|
+
// pool does not serve: billed work if it happens to be a per-provider
|
|
419
|
+
// spelling, an opaque server error otherwise. Failing here names the model
|
|
420
|
+
// and the allowed ones, and costs nothing.
|
|
421
|
+
const refusal = modelRefusal(item.model);
|
|
422
|
+
if (refusal) {
|
|
423
|
+
const failed = { ok: false, error: refusal, model: item.model, ms: 0 };
|
|
424
|
+
emit({ type: 'finish', taskId: item.id, ok: false, result: failed });
|
|
425
|
+
return failed;
|
|
426
|
+
}
|
|
427
|
+
const model = resolveModel({ model: item.model, env });
|
|
428
|
+
const ignored = ignoredEnvModel({ model: item.model, env });
|
|
429
|
+
if (ignored) emit({ type: 'note', taskId: item.id, note: ignored });
|
|
430
|
+
const fanout = resolveFanout(item, env);
|
|
431
|
+
const effort = resolveEffort({ effort: item.effort, env, fanout });
|
|
432
|
+
const rounds = maxRounds(env, item.maxRounds);
|
|
433
|
+
// Approval requests have no one to answer them here; see the header.
|
|
434
|
+
let approvalAsked = null;
|
|
435
|
+
let doneRounds = 0;
|
|
436
|
+
// What THIS task's tool layer wrote. The commit uses it as the only
|
|
437
|
+
// positive attribution available: a path in here that also differs from
|
|
438
|
+
// the pre-task snapshot is ours, and everything else dirty is left alone
|
|
439
|
+
// (see the module header, and git-scope.js's scopedCommit).
|
|
440
|
+
const written = new Set();
|
|
441
|
+
const onDelta = (chunk) => {
|
|
442
|
+
if (!chunk || typeof chunk !== 'object') return;
|
|
443
|
+
if (chunk.approval) {
|
|
444
|
+
approvalAsked = chunk.approval;
|
|
445
|
+
return;
|
|
446
|
+
}
|
|
447
|
+
if (chunk.tool) {
|
|
448
|
+
if (chunk.tool.phase === 'done') {
|
|
449
|
+
doneRounds += 1;
|
|
450
|
+
// Recorded on `done` and only when it succeeded: "this path is mine"
|
|
451
|
+
// is true once the write actually landed, and a refused or failed
|
|
452
|
+
// write must not claim a path a concurrent writer is editing.
|
|
453
|
+
if (WRITE_TOOLS.has(chunk.tool.name) && chunk.tool.ok !== false) {
|
|
454
|
+
const p = writtenPath(chunk.tool);
|
|
455
|
+
if (p) written.add(p);
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
emit({ type: 'tool', taskId: item.id, tool: chunk.tool });
|
|
459
|
+
return;
|
|
460
|
+
}
|
|
461
|
+
if (typeof chunk.delta === 'string') emit({ type: 'delta', taskId: item.id, text: chunk.delta });
|
|
462
|
+
else if (typeof chunk.reasoning === 'string') emit({ type: 'reasoning', taskId: item.id, text: chunk.reasoning });
|
|
463
|
+
};
|
|
464
|
+
|
|
465
|
+
// Snapshot BEFORE the turn: without it there is no way to tell our edits
|
|
466
|
+
// from a concurrent agent's, and scopedCommit refuses to sweep rather than
|
|
467
|
+
// guess.
|
|
468
|
+
const wantCommit = commit === undefined ? Boolean(item.commit) : Boolean(commit);
|
|
469
|
+
const before = wantCommit ? safe(() => git.gitStatusSnapshot(cwd), null) : null;
|
|
470
|
+
|
|
471
|
+
emit({ type: 'start', taskId: item.id, model, cwd, rounds, fanout, effort });
|
|
472
|
+
const started = now();
|
|
473
|
+
let result;
|
|
474
|
+
try {
|
|
475
|
+
const res = await withRoundHorizon(rounds, env, () =>
|
|
476
|
+
engine.chat(
|
|
477
|
+
{
|
|
478
|
+
class: 'aegis',
|
|
479
|
+
model,
|
|
480
|
+
prompt: taskPrompt(item, { carry, rounds }),
|
|
481
|
+
// The pooled-brain fan-out ("fan out this turn" in the GUI chat
|
|
482
|
+
// header, and opt-in here): the server fans the round out to
|
|
483
|
+
// several reasoning workers and synthesises. It costs about
|
|
484
|
+
// workers+1 full reasoning calls, so it travels only when the task
|
|
485
|
+
// asked for it — see resolveFanout above. A `singlePass` task is
|
|
486
|
+
// the default for exactly that reason; the retry/write-up passes in
|
|
487
|
+
// the engine send `brain: false` for the same one-call reason.
|
|
488
|
+
autonomous: fanout,
|
|
489
|
+
effort,
|
|
490
|
+
workers: fanout ? item.workers || undefined : undefined,
|
|
491
|
+
// The turn's working directory rides on `env`: engine.js reads the
|
|
492
|
+
// tool loop's cwd from envFor(payload), so a top-level `cwd` field
|
|
493
|
+
// is a directory the engine would ignore and every tool would run
|
|
494
|
+
// in the host's own process.cwd() instead of the task's.
|
|
495
|
+
env: { cwd },
|
|
496
|
+
maxRounds: rounds,
|
|
497
|
+
sessionId: sessionIdFor(item.id),
|
|
498
|
+
stream: true,
|
|
499
|
+
},
|
|
500
|
+
onDelta
|
|
501
|
+
)
|
|
502
|
+
);
|
|
503
|
+
const output = assistantText(res);
|
|
504
|
+
result = {
|
|
505
|
+
ok: true,
|
|
506
|
+
output,
|
|
507
|
+
usage: (res && res.usage) || null,
|
|
508
|
+
stoppedOnRounds: Boolean(res && res.stoppedOnRounds),
|
|
509
|
+
rounds: doneRounds || undefined,
|
|
510
|
+
};
|
|
511
|
+
if (approvalAsked) {
|
|
512
|
+
// A gate that is up in an unattended run is a bug in the caller, and
|
|
513
|
+
// it is better seen than hung on: the tool round cannot proceed past it,
|
|
514
|
+
// and a card nobody can click is a task that never finishes.
|
|
515
|
+
result.ok = false;
|
|
516
|
+
result.error =
|
|
517
|
+
`tool approval was requested for "${approvalAsked.tool || approvalAsked.name || 'a tool'}" ` +
|
|
518
|
+
'with no one to answer it — construct the queue engine with approvals disabled ' +
|
|
519
|
+
'(getConfirmMode: () => false)';
|
|
520
|
+
}
|
|
521
|
+
} catch (e) {
|
|
522
|
+
result = { ok: false, error: errorText(e) };
|
|
523
|
+
}
|
|
524
|
+
result.ms = Math.max(0, now() - started);
|
|
525
|
+
|
|
526
|
+
if (wantCommit) {
|
|
527
|
+
result.commit = safe(
|
|
528
|
+
() =>
|
|
529
|
+
git.scopedCommit(cwd, {
|
|
530
|
+
message: commitMessage(item),
|
|
531
|
+
before,
|
|
532
|
+
// Positive attribution: the paths this task's tool layer reported
|
|
533
|
+
// writing. Without it scopedCommit refuses to stage anything but
|
|
534
|
+
// the pre-task-dirty-and-still-identical set, which cannot see a
|
|
535
|
+
// concurrent writer editing mid-task (see the module header).
|
|
536
|
+
written: [...written],
|
|
537
|
+
}),
|
|
538
|
+
{ ok: false, error: 'git-scope unavailable' }
|
|
539
|
+
);
|
|
540
|
+
result.files = committedPaths(result.commit, written);
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
emit({ type: 'finish', taskId: item.id, ok: result.ok, result });
|
|
544
|
+
return result;
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
/** `git-scope` is filesystem/git work: a throw there must not lose the task's outcome. */
|
|
548
|
+
function safe(fn, fallback) {
|
|
549
|
+
try {
|
|
550
|
+
return fn();
|
|
551
|
+
} catch {
|
|
552
|
+
return fallback;
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
|
|
556
|
+
/**
|
|
557
|
+
* The paths this task changed: exactly what scopedCommit staged, which is
|
|
558
|
+
* (written by this task) ∩ (differs from the pre-task snapshot). Reported
|
|
559
|
+
* rather than recomputed from a fresh snapshot — a snapshot taken after the
|
|
560
|
+
* commit cannot tell our work from a peer's next edit, and that difference is
|
|
561
|
+
* the whole point of the attribution rule.
|
|
562
|
+
*/
|
|
563
|
+
function committedPaths(commit, written) {
|
|
564
|
+
if (!commit || commit.error) return [];
|
|
565
|
+
if (Array.isArray(commit.paths)) return commit.paths;
|
|
566
|
+
// No commit was made (skipped): report what we know we wrote, so a task's
|
|
567
|
+
// digest still names the work even when the tree had nothing to commit.
|
|
568
|
+
return [...written];
|
|
569
|
+
}
|
|
570
|
+
|
|
571
|
+
/**
|
|
572
|
+
* Drain pending tasks, one at a time, until the queue is empty (or `max` is
|
|
573
|
+
* reached). Rebases on the file between tasks, so a task added mid-drain —
|
|
574
|
+
* by another window, another host, or a `reconcile` — is picked up in the
|
|
575
|
+
* same run instead of waiting for the next one.
|
|
576
|
+
*/
|
|
577
|
+
async function proceed({ commit, max = 0, stopOnError = false, carry = '' } = {}) {
|
|
578
|
+
const lock = queue.acquireLock(env);
|
|
579
|
+
if (!lock.ok) {
|
|
580
|
+
emit({ type: 'locked', holder: lock.holder });
|
|
581
|
+
return { ok: false, locked: true, holder: lock.holder, ran: [], carry };
|
|
582
|
+
}
|
|
583
|
+
const ran = [];
|
|
584
|
+
try {
|
|
585
|
+
for (;;) {
|
|
586
|
+
if (max > 0 && ran.length >= max) break;
|
|
587
|
+
// Re-read the FILE every iteration, never a snapshot taken before the
|
|
588
|
+
// loop: a task added mid-drain (another window, another host, this run's
|
|
589
|
+
// own `reconcile`) has to be picked up by the same drain, and a stale
|
|
590
|
+
// in-memory list is exactly how it would be missed.
|
|
591
|
+
const items = queue.loadQueue(env);
|
|
592
|
+
const recovered = queue.recoverStale(items);
|
|
593
|
+
// Persisted, not just fixed in memory: without this write the items
|
|
594
|
+
// stay `running` in the file, and the next drain re-pends them all over
|
|
595
|
+
// again (a task whose worker was killed would look alive forever to
|
|
596
|
+
// every other reader).
|
|
597
|
+
if (recovered.length) {
|
|
598
|
+
queue.saveQueue(env, items);
|
|
599
|
+
emit({ type: 'recovered', ids: recovered });
|
|
600
|
+
}
|
|
601
|
+
const next = queue.pending(items)[0];
|
|
602
|
+
if (!next) break;
|
|
603
|
+
const outcome = await runOne(next, { commit, carry });
|
|
604
|
+
carry = outcome.carry;
|
|
605
|
+
ran.push(outcome);
|
|
606
|
+
if (!outcome.ok && stopOnError) break;
|
|
607
|
+
}
|
|
608
|
+
} finally {
|
|
609
|
+
queue.releaseLock(env);
|
|
610
|
+
}
|
|
611
|
+
return { ok: true, ran, carry };
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
/** Claim one task from the file, run it, write the outcome back. */
|
|
615
|
+
async function runOne(item, { commit, carry = '' } = {}) {
|
|
616
|
+
const items = queue.loadQueue(env);
|
|
617
|
+
const claimed = queue.markRunning(items, item.id, { now: now() });
|
|
618
|
+
if (!claimed) return { ok: false, error: `unknown task #${item.id}`, carry };
|
|
619
|
+
queue.saveQueue(env, items);
|
|
620
|
+
|
|
621
|
+
const result = await runTask(claimed, { carry, commit });
|
|
622
|
+
|
|
623
|
+
const after = queue.loadQueue(env);
|
|
624
|
+
queue.settle(after, claimed.id, {
|
|
625
|
+
status: result.ok ? 'done' : 'error',
|
|
626
|
+
result: {
|
|
627
|
+
ok: result.ok,
|
|
628
|
+
output: result.output || '',
|
|
629
|
+
usage: result.usage || null,
|
|
630
|
+
ms: result.ms,
|
|
631
|
+
commit: result.commit || null,
|
|
632
|
+
files: result.files || [],
|
|
633
|
+
},
|
|
634
|
+
error: result.ok ? null : result.error,
|
|
635
|
+
now: now(),
|
|
636
|
+
});
|
|
637
|
+
queue.saveQueue(env, after);
|
|
638
|
+
queue.appendRun(env, {
|
|
639
|
+
id: claimed.id,
|
|
640
|
+
task: claimed.task,
|
|
641
|
+
cwd: claimed.cwd,
|
|
642
|
+
model: resolveModel({ model: claimed.model, env }),
|
|
643
|
+
status: result.ok ? 'done' : 'error',
|
|
644
|
+
at: new Date(now()).toISOString(),
|
|
645
|
+
ms: result.ms,
|
|
646
|
+
usage: result.usage || null,
|
|
647
|
+
files: result.files || [],
|
|
648
|
+
error: result.ok ? null : result.error,
|
|
649
|
+
});
|
|
650
|
+
return { ...result, id: claimed.id, carry: appendDigest(carry, digestLine(claimed, result)) };
|
|
651
|
+
}
|
|
652
|
+
|
|
653
|
+
/**
|
|
654
|
+
* Queue the next unfinished PLAN.md phase (and optionally work it straight
|
|
655
|
+
* away). The phase text is looked up rather than pasted, so the task points
|
|
656
|
+
* the model at the spec instead of paraphrasing it — a paraphrase in the
|
|
657
|
+
* prompt is a second spec that can disagree with the file.
|
|
658
|
+
*/
|
|
659
|
+
async function reconcile({ cwd = process.cwd(), auto = false, commit, max = 0, stopOnError = false } = {}) {
|
|
660
|
+
const plan = queue.reconcilePlan(cwd, { write: true });
|
|
661
|
+
if (!plan.ok) return { ok: false, error: plan.reason };
|
|
662
|
+
if (plan.healed && plan.healed.changed) {
|
|
663
|
+
emit({ type: 'healed', phases: plan.healed.added });
|
|
664
|
+
}
|
|
665
|
+
if (plan.next == null) return { ok: true, exhausted: true };
|
|
666
|
+
|
|
667
|
+
const marker = `Work Phase ${plan.next} from PLAN.md`;
|
|
668
|
+
const items = queue.loadQueue(env);
|
|
669
|
+
const existing = items.find(
|
|
670
|
+
(i) => i.cwd === cwd && String(i.task).startsWith(marker) && i.status !== 'done'
|
|
671
|
+
);
|
|
672
|
+
if (existing) {
|
|
673
|
+
emit({ type: 'already-queued', phase: plan.next, id: existing.id, status: existing.status });
|
|
674
|
+
if (!auto) return { ok: true, phase: plan.next, id: existing.id, existing: true };
|
|
675
|
+
} else {
|
|
676
|
+
const item = queue.addTask(env, {
|
|
677
|
+
task:
|
|
678
|
+
`${marker} ("${plan.title}") at ${cwd}. Read the full "## Phase ${plan.next}" section in ` +
|
|
679
|
+
'PLAN.md at the repo root for the spec, exit criteria and constraints, and implement it ' +
|
|
680
|
+
'exactly as scoped there. Keep the repo\'s checks green (`npm run check`, plus that ' +
|
|
681
|
+
`package's tests). When every exit criterion is met, mark the "## Phase ${plan.next}" ` +
|
|
682
|
+
'heading with ✅ in PLAN.md and update its line in the top Status checklist.',
|
|
683
|
+
cwd,
|
|
684
|
+
source: 'reconcile',
|
|
685
|
+
model: resolveModel({ env }),
|
|
686
|
+
});
|
|
687
|
+
emit({ type: 'queued', phase: plan.next, id: item.id, title: plan.title });
|
|
688
|
+
}
|
|
689
|
+
if (!auto) return { ok: true, phase: plan.next, title: plan.title };
|
|
690
|
+
const drained = await proceed({ commit, max, stopOnError });
|
|
691
|
+
return { ok: true, phase: plan.next, title: plan.title, ...drained };
|
|
692
|
+
}
|
|
693
|
+
|
|
694
|
+
/**
|
|
695
|
+
* The engine's session id for one task. Deterministic, because the engine
|
|
696
|
+
* keeps its AbortController under this key (engine.js cancel()), so this is
|
|
697
|
+
* the handle both writing the outcome and stopping the turn go through.
|
|
698
|
+
*/
|
|
699
|
+
const sessionIdFor = (id) => `queue-${id}`;
|
|
700
|
+
|
|
701
|
+
/**
|
|
702
|
+
* Stop the task currently running. The engine owns cancellation through the
|
|
703
|
+
* session id it registered, so this delegates instead of keeping a second
|
|
704
|
+
* controller — a locally-held AbortSignal would be one the engine never reads.
|
|
705
|
+
*/
|
|
706
|
+
function cancel(id) {
|
|
707
|
+
if (id == null) return { ok: false, error: 'cancel needs a task id' };
|
|
708
|
+
if (typeof engine.cancel !== 'function') return { ok: false, error: 'this engine cannot cancel a turn' };
|
|
709
|
+
const res = engine.cancel(sessionIdFor(id));
|
|
710
|
+
return { ok: Boolean(res && res.ok), sessionId: sessionIdFor(id) };
|
|
711
|
+
}
|
|
712
|
+
|
|
713
|
+
return { runTask, runOne, proceed, reconcile, cancel, sessionIdFor, resolveModel: (o) => resolveModel({ ...o, env }) };
|
|
714
|
+
}
|
|
715
|
+
|
|
716
|
+
/** One-line commit message for a task: its first line, clipped, and labelled. */
|
|
717
|
+
function commitMessage(item) {
|
|
718
|
+
const first = String(item.task || 'queued task').split('\n')[0].replace(/\s+/g, ' ').trim();
|
|
719
|
+
return `autonomous: ${first.slice(0, 66)}${first.length > 66 ? '…' : ''}`;
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
module.exports = {
|
|
723
|
+
DEFAULT_MODEL,
|
|
724
|
+
DEFAULT_ROUNDS,
|
|
725
|
+
AEGIS_MODEL_IDS,
|
|
726
|
+
isAegisModel,
|
|
727
|
+
modelRefusal,
|
|
728
|
+
WRITE_TOOLS,
|
|
729
|
+
writtenPath,
|
|
730
|
+
isAutonomousRequest,
|
|
731
|
+
maxRounds,
|
|
732
|
+
withRoundHorizon,
|
|
733
|
+
resolveModel,
|
|
734
|
+
ignoredEnvModel,
|
|
735
|
+
resolveEffort,
|
|
736
|
+
resolveFanout,
|
|
737
|
+
autonomousDirective,
|
|
738
|
+
taskPrompt,
|
|
739
|
+
appendDigest,
|
|
740
|
+
digestLine,
|
|
741
|
+
assistantText,
|
|
742
|
+
errorText,
|
|
743
|
+
commitMessage,
|
|
744
|
+
createQueueWorker,
|
|
745
|
+
};
|