aegis-desktop 0.5.4 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,600 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * autonomous.js — the unattended worker: what actually runs a queued task.
5
+ *
6
+ * `queue.js` stores tasks; this module is the half that turns one into work.
7
+ * It owns three things no interactive turn has to think about:
8
+ *
9
+ * 1. THE OPERATING DIRECTIVE. An interactive turn can end on "let me check…"
10
+ * or a question, because a human will read it and answer. A queued task
11
+ * ends there only if nobody ever looks — so every queued turn runs under a
12
+ * directive that says, in the model's own context: nobody is watching,
13
+ * decide, act, verify, report; do not hand the work back. The round horizon
14
+ * is stated too, so the model budgets its exploration instead of meeting
15
+ * the cap by accident.
16
+ *
17
+ * 2. THE CARRY-OVER DIGEST. The AEGIS API is stateless per request, so a
18
+ * drain is not literally one long conversation — but treating ten queued
19
+ * tasks as ten unrelated strangers makes the ninth re-derive what the
20
+ * second already learned. Each finished task appends a one-line digest
21
+ * (what it was, what it touched, whether it was verified) to a carry string
22
+ * that rides in front of the next task's prompt. Bounded and one line per
23
+ * task: it is a briefing, not a transcript, and it must never grow into the
24
+ * context it is meant to save.
25
+ *
26
+ * 3. ATTRIBUTED COMMITS. A queue drain can run in a checkout somebody else is
27
+ * editing. `git add -A && git commit` in that situation commits THEIR
28
+ * half-finished work under our message — the failure `git-scope.js` was
29
+ * written for. So a task's `commit` flag commits only paths that this task's
30
+ * own tool layer wrote (`writeFile`/`editFile` frames, recorded as they
31
+ * land); everything else dirty is left alone, whether it was dirty before
32
+ * the task (a peer's in-flight work) or a path this task never wrote.
33
+ *
34
+ * Two conditions, not one, and the second is not academic: commit 96fb64f
35
+ * swept `desktop/electron-builder.yml` — a file another live session was
36
+ * editing at that moment — into an unrelated commit, because "dirty before
37
+ * and byte-identical now" cannot see a concurrent writer who edits DURING
38
+ * the task window. Only positive attribution can, so a path this task never
39
+ * wrote is never staged, and the paths left behind are reported.
40
+ *
41
+ * It does NOT own the model call itself: the caller passes the local engine
42
+ * (`desktop/lib/local/engine.js`), which is the same tool loop the GUI and the
43
+ * CLI chat in. One tool loop, three callers, no drift.
44
+ *
45
+ * THE APPROVAL GATE IS THE CALLER'S JOB. There is nobody to answer an approval
46
+ * card in an unattended run, so the engine must be constructed with the gate
47
+ * off (`getConfirmMode: () => false`). Rather than trust that, the worker
48
+ * watches the turn's event stream: an approval request means a card nobody can
49
+ * click, so it is reported as a failed task instead of a hang.
50
+ */
51
+
52
+ const queue = require('./queue.js');
53
+ const gitScope = require('./git-scope.js');
54
+
55
+ /** Default tool-round horizon for an unattended turn (matches the engine's). */
56
+ const DEFAULT_ROUNDS = 40;
57
+
58
+ /**
59
+ * The engine tools that report the file they write, and the argument holding
60
+ * it. This set is the attribution record for a task's commit: `exec` is absent
61
+ * on purpose — a shell command names no paths, so a file it creates cannot be
62
+ * attributed to this task, and guessing "anything new is ours" is precisely how
63
+ * a concurrent writer's file gets committed.
64
+ */
65
+ const WRITE_TOOLS = new Set(['writeFile', 'editFile']);
66
+
67
+ /** The path a file-writing tool call names, in either spelling the engine uses. */
68
+ function writtenPath(tool) {
69
+ const args = (tool && tool.args) || {};
70
+ const p = args.file_path || args.path;
71
+ return p ? String(p) : '';
72
+ }
73
+
74
+ /**
75
+ * The default AEGIS Cloud model for autonomous work: the pooled brain, which
76
+ * is the tier the server fans out to multiple reasoning workers and
77
+ * synthesises. Autonomous tasks are exactly the ones worth that spend, and
78
+ * `nexus-brain` is the canonical id the catalog itself prefers (the other tier
79
+ * spellings are aliases of it — see filterAegisCatalog in engine.js).
80
+ */
81
+ const DEFAULT_MODEL = 'nexus-brain';
82
+
83
+ /** Phrase match for "work autonomously" in a prompt or a queued task. */
84
+ const AUTONOMOUS_REQUEST_RE =
85
+ /\bautonomously\b|\bon your own\b|\bwithout asking\b|\bend[- ]to[- ]end\b|\bno (?:more )?questions\b|\bfully autonomous\b|\bqueue it\b/i;
86
+
87
+ /** True when the user is asking for unattended execution. */
88
+ function isAutonomousRequest(text) {
89
+ return AUTONOMOUS_REQUEST_RE.test(String(text || ''));
90
+ }
91
+
92
+ /** The round horizon, honouring AEGIS_AUTONOMOUS_MAX_ROUNDS. */
93
+ function maxRounds(env = process.env, stated) {
94
+ const n = Number.parseInt(stated, 10);
95
+ if (Number.isFinite(n) && n > 0) return n;
96
+ const raw = Number.parseInt((env && env.AEGIS_AUTONOMOUS_MAX_ROUNDS) || '', 10);
97
+ return Number.isFinite(raw) && raw > 0 ? raw : DEFAULT_ROUNDS;
98
+ }
99
+
100
+ /**
101
+ * Make the horizon real for the engine, then put the environment back.
102
+ *
103
+ * A `maxRounds` field in the chat payload would be dead code: engine.js reads
104
+ * the knob from `process.env` at turn time (engine.js:1024,
105
+ * `AEGIS_AUTONOMOUS_MAX_ROUNDS`). Env, unlike a payload key, is process-wide, so
106
+ * the restore has to wait for a promise to settle (a turn reads the knob many
107
+ * rounds after the call is made) — hence the thenable branch below.
108
+ */
109
+ function withRoundHorizon(rounds, env, fn) {
110
+ const key = 'AEGIS_AUTONOMOUS_MAX_ROUNDS';
111
+ // Both objects: the caller's env carries the path/model defaults this worker
112
+ // reads, and `process.env` is where engine.js reads the round cap. A test that
113
+ // passes its own env still gets the knob on that object, and a real drain gets
114
+ // it where the engine actually looks.
115
+ const targets = [];
116
+ for (const t of [env, process.env]) {
117
+ if (t && targets.indexOf(t) === -1) targets.push(t);
118
+ }
119
+ const saved = targets.map((t) => ({
120
+ target: t,
121
+ had: Object.prototype.hasOwnProperty.call(t, key),
122
+ prev: t[key],
123
+ }));
124
+ try {
125
+ for (const t of targets) t[key] = String(rounds);
126
+ } catch {
127
+ /* a frozen/sealed env object is the caller's choice; the payload carries it too */
128
+ }
129
+ const restore = () => {
130
+ for (const s of saved) {
131
+ try {
132
+ if (s.had) s.target[key] = s.prev;
133
+ else delete s.target[key];
134
+ } catch {
135
+ /* nothing to restore on an object we could not write */
136
+ }
137
+ }
138
+ };
139
+ let out;
140
+ try {
141
+ out = fn();
142
+ } catch (e) {
143
+ restore();
144
+ throw e;
145
+ }
146
+ if (out && typeof out.then === 'function') {
147
+ return out.then(
148
+ (value) => {
149
+ restore();
150
+ return value;
151
+ },
152
+ (err) => {
153
+ restore();
154
+ throw err;
155
+ }
156
+ );
157
+ }
158
+ restore();
159
+ return out;
160
+ }
161
+
162
+ /**
163
+ * The model an autonomous task runs on: an explicit pick wins, then the
164
+ * environment (so a systemd timer can pin a cheap tier), then the pooled
165
+ * brain. Never the interactive session's model — a queue survives the session
166
+ * that queued it, so it cannot inherit that session's choice.
167
+ */
168
+ function resolveModel({ model, env } = {}) {
169
+ const e = env || process.env;
170
+ const picked = String(model || '').trim();
171
+ if (picked) return picked;
172
+ const fromEnv = String(e.AEGIS_AUTONOMOUS_MODEL || e.AEGIS_MODEL || '').trim();
173
+ return fromEnv || DEFAULT_MODEL;
174
+ }
175
+
176
+ /** Effort rung for an unattended turn: high unless the caller/environment says otherwise. */
177
+ function resolveEffort({ effort, env } = {}) {
178
+ const e = env || process.env;
179
+ return String(effort || e.AEGIS_AUTONOMOUS_EFFORT || 'high');
180
+ }
181
+
182
+ /**
183
+ * The operating directive injected as the autonomous turn's prompt preamble.
184
+ * Ported verbatim in spirit from aegiscodex-dev/src/autonomous.js so both
185
+ * clients behave the same; duplicated rather than vendored because the plugin
186
+ * hosts no ESM build of that module.
187
+ */
188
+ function autonomousDirective(rounds = DEFAULT_ROUNDS) {
189
+ return [
190
+ '# Autonomous mode',
191
+ `You are running autonomously, not in a conversation: there is no user to answer a question or approve a plan. You have up to ${rounds} tool rounds this turn; use as many as the task needs.`,
192
+ 'Work the task end to end:',
193
+ '1. Plan in one line, then start acting in the same turn — never end a turn on a plan.',
194
+ '2. Read only what you need to make the change (no fishing through the repo).',
195
+ '3. Make the change with Write/Edit/Bash, then VERIFY it: re-read the result and run the relevant test/command.',
196
+ '4. If verification fails, fix it and verify again — loop until it passes or you are genuinely blocked.',
197
+ '5. Do not ask for permission, do not hand the work back, do not stop at "let me check…".',
198
+ '6. Finish with a short report: what changed (file paths), the command you ran to verify, its result, and any remaining blocker.',
199
+ ].join('\n');
200
+ }
201
+
202
+ /**
203
+ * The digest line one finished task contributes to the next task's briefing.
204
+ * Deliberately tiny: task text clipped, outcome as a mark, files as basenames.
205
+ */
206
+ function digestLine(item, result) {
207
+ const task = String(item.task || '').replace(/\s+/g, ' ').trim().slice(0, 120);
208
+ const mark = result && result.ok ? '✓' : '✗';
209
+ const files = (result && Array.isArray(result.files) && result.files.slice(0, 6)) || [];
210
+ const where = files.length ? ` — touched: ${files.join(', ')}${result.files.length > files.length ? ', …' : ''}` : '';
211
+ const why = !result || result.ok ? '' : ` — ${String((result && result.error) || 'failed').slice(0, 120)}`;
212
+ return `- #${item.id} ${mark} ${task}${where}${why}`;
213
+ }
214
+
215
+ /** Keep only the last `keep` digest lines, so the briefing cannot grow forever. */
216
+ function appendDigest(carry, line, { keep = 6 } = {}) {
217
+ const lines = String(carry || '')
218
+ .split('\n')
219
+ .filter(Boolean)
220
+ .concat(line);
221
+ return lines.slice(-keep).join('\n');
222
+ }
223
+
224
+ /**
225
+ * The prompt a queued task actually sends: the directive, the briefing from
226
+ * earlier tasks in the same drain (when there is one), then the task itself.
227
+ */
228
+ function taskPrompt(item, { carry = '', rounds = DEFAULT_ROUNDS } = {}) {
229
+ const parts = [autonomousDirective(rounds)];
230
+ if (carry) {
231
+ parts.push(
232
+ '# Earlier tasks in this run\n' +
233
+ 'Work already done by this queue — do not redo it, build on it:\n' +
234
+ carry
235
+ );
236
+ }
237
+ parts.push('# Task\n' + String(item.task || '').trim());
238
+ return parts.join('\n\n');
239
+ }
240
+
241
+ /** The assistant's visible text out of an OpenAI-shaped engine result. */
242
+ function assistantText(res) {
243
+ if (!res || typeof res !== 'object') return '';
244
+ const choice = Array.isArray(res.choices) ? res.choices[0] : null;
245
+ const content = choice && choice.message ? choice.message.content : res.content;
246
+ if (typeof content === 'string') return content;
247
+ if (Array.isArray(content)) {
248
+ return content
249
+ .map((p) => (typeof p === 'string' ? p : p && typeof p.text === 'string' ? p.text : ''))
250
+ .join('');
251
+ }
252
+ return '';
253
+ }
254
+
255
+ /** A readable one-liner for whatever a failed turn threw. */
256
+ function errorText(err) {
257
+ if (!err) return 'unknown error';
258
+ if (typeof err === 'string') return err;
259
+ return String(err.message || err.error || err);
260
+ }
261
+
262
+ /**
263
+ * Build the worker.
264
+ *
265
+ * @param {object} opts
266
+ * @param {object} opts.engine The local engine (createLocalEngine()). MUST be
267
+ * constructed with the approval gate off — see the module header.
268
+ * @param {object} [opts.env] Environment for paths/model defaults (tests pass a temp one).
269
+ * @param {function} [opts.log] `(event) => void` progress sink; also forwarded
270
+ * the turn's own `{delta}` / `{tool}` frames.
271
+ * @param {object} [opts.git] Injectable git-scope (tests).
272
+ * @param {function} [opts.now] Clock.
273
+ */
274
+ function createQueueWorker({ engine, env = process.env, log = () => {}, git = gitScope, now = Date.now } = {}) {
275
+ if (!engine || typeof engine.chat !== 'function') {
276
+ throw new Error('autonomous: a local engine with chat() is required');
277
+ }
278
+ const emit = (event) => {
279
+ try {
280
+ log(event);
281
+ } catch {
282
+ /* a progress sink that throws (a closed stdout, a destroyed window) must
283
+ never take the task down with it — the work is the point, not the log */
284
+ }
285
+ };
286
+
287
+ /**
288
+ * Run ONE queued task. Never throws: a failed task is data (so the drain can
289
+ * decide whether to continue), not an exception that aborts the queue.
290
+ */
291
+ async function runTask(item, { carry = '', commit } = {}) {
292
+ const cwd = item.cwd || process.cwd();
293
+ const model = resolveModel({ model: item.model, env });
294
+ const effort = resolveEffort({ effort: item.effort, env });
295
+ const rounds = maxRounds(env, item.maxRounds);
296
+ // Approval requests have no one to answer them here; see the header.
297
+ let approvalAsked = null;
298
+ let doneRounds = 0;
299
+ // What THIS task's tool layer wrote. The commit uses it as the only
300
+ // positive attribution available: a path in here that also differs from
301
+ // the pre-task snapshot is ours, and everything else dirty is left alone
302
+ // (see the module header, and git-scope.js's scopedCommit).
303
+ const written = new Set();
304
+ const onDelta = (chunk) => {
305
+ if (!chunk || typeof chunk !== 'object') return;
306
+ if (chunk.approval) {
307
+ approvalAsked = chunk.approval;
308
+ return;
309
+ }
310
+ if (chunk.tool) {
311
+ if (chunk.tool.phase === 'done') {
312
+ doneRounds += 1;
313
+ // Recorded on `done` and only when it succeeded: "this path is mine"
314
+ // is true once the write actually landed, and a refused or failed
315
+ // write must not claim a path a concurrent writer is editing.
316
+ if (WRITE_TOOLS.has(chunk.tool.name) && chunk.tool.ok !== false) {
317
+ const p = writtenPath(chunk.tool);
318
+ if (p) written.add(p);
319
+ }
320
+ }
321
+ emit({ type: 'tool', taskId: item.id, tool: chunk.tool });
322
+ return;
323
+ }
324
+ if (typeof chunk.delta === 'string') emit({ type: 'delta', taskId: item.id, text: chunk.delta });
325
+ else if (typeof chunk.reasoning === 'string') emit({ type: 'reasoning', taskId: item.id, text: chunk.reasoning });
326
+ };
327
+
328
+ // Snapshot BEFORE the turn: without it there is no way to tell our edits
329
+ // from a concurrent agent's, and scopedCommit refuses to sweep rather than
330
+ // guess.
331
+ const wantCommit = commit === undefined ? Boolean(item.commit) : Boolean(commit);
332
+ const before = wantCommit ? safe(() => git.gitStatusSnapshot(cwd), null) : null;
333
+
334
+ emit({ type: 'start', taskId: item.id, model, cwd, rounds });
335
+ const started = now();
336
+ let result;
337
+ try {
338
+ const res = await withRoundHorizon(rounds, env, () =>
339
+ engine.chat(
340
+ {
341
+ class: 'aegis',
342
+ model,
343
+ prompt: taskPrompt(item, { carry, rounds }),
344
+ // The pooled brain ("work autonomously" in the GUI): the server fans
345
+ // the round out to several reasoning workers and synthesises. A
346
+ // `singlePass` task opts out — the retry/write-up passes in the
347
+ // engine send `brain: false` for exactly this reason.
348
+ autonomous: !item.singlePass,
349
+ effort,
350
+ workers: item.workers || undefined,
351
+ // The turn's working directory rides on `env`: engine.js reads the
352
+ // tool loop's cwd from envFor(payload), so a top-level `cwd` field
353
+ // is a directory the engine would ignore and every tool would run
354
+ // in the host's own process.cwd() instead of the task's.
355
+ env: { cwd },
356
+ maxRounds: rounds,
357
+ sessionId: sessionIdFor(item.id),
358
+ stream: true,
359
+ },
360
+ onDelta
361
+ )
362
+ );
363
+ const output = assistantText(res);
364
+ result = {
365
+ ok: true,
366
+ output,
367
+ usage: (res && res.usage) || null,
368
+ stoppedOnRounds: Boolean(res && res.stoppedOnRounds),
369
+ rounds: doneRounds || undefined,
370
+ };
371
+ if (approvalAsked) {
372
+ // A gate that is up in an unattended run is a bug in the caller, and
373
+ // it is better seen than hung on: the tool round cannot proceed past it,
374
+ // and a card nobody can click is a task that never finishes.
375
+ result.ok = false;
376
+ result.error =
377
+ `tool approval was requested for "${approvalAsked.tool || approvalAsked.name || 'a tool'}" ` +
378
+ 'with no one to answer it — construct the queue engine with approvals disabled ' +
379
+ '(getConfirmMode: () => false)';
380
+ }
381
+ } catch (e) {
382
+ result = { ok: false, error: errorText(e) };
383
+ }
384
+ result.ms = Math.max(0, now() - started);
385
+
386
+ if (wantCommit) {
387
+ result.commit = safe(
388
+ () =>
389
+ git.scopedCommit(cwd, {
390
+ message: commitMessage(item),
391
+ before,
392
+ // Positive attribution: the paths this task's tool layer reported
393
+ // writing. Without it scopedCommit refuses to stage anything but
394
+ // the pre-task-dirty-and-still-identical set, which cannot see a
395
+ // concurrent writer editing mid-task (see the module header).
396
+ written: [...written],
397
+ }),
398
+ { ok: false, error: 'git-scope unavailable' }
399
+ );
400
+ result.files = committedPaths(result.commit, written);
401
+ }
402
+
403
+ emit({ type: 'finish', taskId: item.id, ok: result.ok, result });
404
+ return result;
405
+ }
406
+
407
+ /** `git-scope` is filesystem/git work: a throw there must not lose the task's outcome. */
408
+ function safe(fn, fallback) {
409
+ try {
410
+ return fn();
411
+ } catch {
412
+ return fallback;
413
+ }
414
+ }
415
+
416
+ /**
417
+ * The paths this task changed: exactly what scopedCommit staged, which is
418
+ * (written by this task) ∩ (differs from the pre-task snapshot). Reported
419
+ * rather than recomputed from a fresh snapshot — a snapshot taken after the
420
+ * commit cannot tell our work from a peer's next edit, and that difference is
421
+ * the whole point of the attribution rule.
422
+ */
423
+ function committedPaths(commit, written) {
424
+ if (!commit || commit.error) return [];
425
+ if (Array.isArray(commit.paths)) return commit.paths;
426
+ // No commit was made (skipped): report what we know we wrote, so a task's
427
+ // digest still names the work even when the tree had nothing to commit.
428
+ return [...written];
429
+ }
430
+
431
+ /**
432
+ * Drain pending tasks, one at a time, until the queue is empty (or `max` is
433
+ * reached). Rebases on the file between tasks, so a task added mid-drain —
434
+ * by another window, another host, or a `reconcile` — is picked up in the
435
+ * same run instead of waiting for the next one.
436
+ */
437
+ async function proceed({ commit, max = 0, stopOnError = false, carry = '' } = {}) {
438
+ const lock = queue.acquireLock(env);
439
+ if (!lock.ok) {
440
+ emit({ type: 'locked', holder: lock.holder });
441
+ return { ok: false, locked: true, holder: lock.holder, ran: [], carry };
442
+ }
443
+ const ran = [];
444
+ try {
445
+ for (;;) {
446
+ if (max > 0 && ran.length >= max) break;
447
+ // Re-read the FILE every iteration, never a snapshot taken before the
448
+ // loop: a task added mid-drain (another window, another host, this run's
449
+ // own `reconcile`) has to be picked up by the same drain, and a stale
450
+ // in-memory list is exactly how it would be missed.
451
+ const items = queue.loadQueue(env);
452
+ const recovered = queue.recoverStale(items);
453
+ // Persisted, not just fixed in memory: without this write the items
454
+ // stay `running` in the file, and the next drain re-pends them all over
455
+ // again (a task whose worker was killed would look alive forever to
456
+ // every other reader).
457
+ if (recovered.length) {
458
+ queue.saveQueue(env, items);
459
+ emit({ type: 'recovered', ids: recovered });
460
+ }
461
+ const next = queue.pending(items)[0];
462
+ if (!next) break;
463
+ const outcome = await runOne(next, { commit, carry });
464
+ carry = outcome.carry;
465
+ ran.push(outcome);
466
+ if (!outcome.ok && stopOnError) break;
467
+ }
468
+ } finally {
469
+ queue.releaseLock(env);
470
+ }
471
+ return { ok: true, ran, carry };
472
+ }
473
+
474
+ /** Claim one task from the file, run it, write the outcome back. */
475
+ async function runOne(item, { commit, carry = '' } = {}) {
476
+ const items = queue.loadQueue(env);
477
+ const claimed = queue.markRunning(items, item.id, { now: now() });
478
+ if (!claimed) return { ok: false, error: `unknown task #${item.id}`, carry };
479
+ queue.saveQueue(env, items);
480
+
481
+ const result = await runTask(claimed, { carry, commit });
482
+
483
+ const after = queue.loadQueue(env);
484
+ queue.settle(after, claimed.id, {
485
+ status: result.ok ? 'done' : 'error',
486
+ result: {
487
+ ok: result.ok,
488
+ output: result.output || '',
489
+ usage: result.usage || null,
490
+ ms: result.ms,
491
+ commit: result.commit || null,
492
+ files: result.files || [],
493
+ },
494
+ error: result.ok ? null : result.error,
495
+ now: now(),
496
+ });
497
+ queue.saveQueue(env, after);
498
+ queue.appendRun(env, {
499
+ id: claimed.id,
500
+ task: claimed.task,
501
+ cwd: claimed.cwd,
502
+ model: resolveModel({ model: claimed.model, env }),
503
+ status: result.ok ? 'done' : 'error',
504
+ at: new Date(now()).toISOString(),
505
+ ms: result.ms,
506
+ usage: result.usage || null,
507
+ files: result.files || [],
508
+ error: result.ok ? null : result.error,
509
+ });
510
+ return { ...result, id: claimed.id, carry: appendDigest(carry, digestLine(claimed, result)) };
511
+ }
512
+
513
+ /**
514
+ * Queue the next unfinished PLAN.md phase (and optionally work it straight
515
+ * away). The phase text is looked up rather than pasted, so the task points
516
+ * the model at the spec instead of paraphrasing it — a paraphrase in the
517
+ * prompt is a second spec that can disagree with the file.
518
+ */
519
+ async function reconcile({ cwd = process.cwd(), auto = false, commit, max = 0, stopOnError = false } = {}) {
520
+ const plan = queue.reconcilePlan(cwd, { write: true });
521
+ if (!plan.ok) return { ok: false, error: plan.reason };
522
+ if (plan.healed && plan.healed.changed) {
523
+ emit({ type: 'healed', phases: plan.healed.added });
524
+ }
525
+ if (plan.next == null) return { ok: true, exhausted: true };
526
+
527
+ const marker = `Work Phase ${plan.next} from PLAN.md`;
528
+ const items = queue.loadQueue(env);
529
+ const existing = items.find(
530
+ (i) => i.cwd === cwd && String(i.task).startsWith(marker) && i.status !== 'done'
531
+ );
532
+ if (existing) {
533
+ emit({ type: 'already-queued', phase: plan.next, id: existing.id, status: existing.status });
534
+ if (!auto) return { ok: true, phase: plan.next, id: existing.id, existing: true };
535
+ } else {
536
+ const item = queue.addTask(env, {
537
+ task:
538
+ `${marker} ("${plan.title}") at ${cwd}. Read the full "## Phase ${plan.next}" section in ` +
539
+ 'PLAN.md at the repo root for the spec, exit criteria and constraints, and implement it ' +
540
+ 'exactly as scoped there. Keep the repo\'s checks green (`npm run check`, plus that ' +
541
+ `package's tests). When every exit criterion is met, mark the "## Phase ${plan.next}" ` +
542
+ 'heading with ✅ in PLAN.md and update its line in the top Status checklist.',
543
+ cwd,
544
+ source: 'reconcile',
545
+ model: resolveModel({ env }),
546
+ });
547
+ emit({ type: 'queued', phase: plan.next, id: item.id, title: plan.title });
548
+ }
549
+ if (!auto) return { ok: true, phase: plan.next, title: plan.title };
550
+ const drained = await proceed({ commit, max, stopOnError });
551
+ return { ok: true, phase: plan.next, title: plan.title, ...drained };
552
+ }
553
+
554
+ /**
555
+ * The engine's session id for one task. Deterministic, because the engine
556
+ * keeps its AbortController under this key (engine.js cancel()), so this is
557
+ * the handle both writing the outcome and stopping the turn go through.
558
+ */
559
+ const sessionIdFor = (id) => `queue-${id}`;
560
+
561
+ /**
562
+ * Stop the task currently running. The engine owns cancellation through the
563
+ * session id it registered, so this delegates instead of keeping a second
564
+ * controller — a locally-held AbortSignal would be one the engine never reads.
565
+ */
566
+ function cancel(id) {
567
+ if (id == null) return { ok: false, error: 'cancel needs a task id' };
568
+ if (typeof engine.cancel !== 'function') return { ok: false, error: 'this engine cannot cancel a turn' };
569
+ const res = engine.cancel(sessionIdFor(id));
570
+ return { ok: Boolean(res && res.ok), sessionId: sessionIdFor(id) };
571
+ }
572
+
573
+ return { runTask, runOne, proceed, reconcile, cancel, sessionIdFor, resolveModel: (o) => resolveModel({ ...o, env }) };
574
+ }
575
+
576
+ /** One-line commit message for a task: its first line, clipped, and labelled. */
577
+ function commitMessage(item) {
578
+ const first = String(item.task || 'queued task').split('\n')[0].replace(/\s+/g, ' ').trim();
579
+ return `autonomous: ${first.slice(0, 66)}${first.length > 66 ? '…' : ''}`;
580
+ }
581
+
582
+ module.exports = {
583
+ DEFAULT_MODEL,
584
+ DEFAULT_ROUNDS,
585
+ WRITE_TOOLS,
586
+ writtenPath,
587
+ isAutonomousRequest,
588
+ maxRounds,
589
+ withRoundHorizon,
590
+ resolveModel,
591
+ resolveEffort,
592
+ autonomousDirective,
593
+ taskPrompt,
594
+ appendDigest,
595
+ digestLine,
596
+ assistantText,
597
+ errorText,
598
+ commitMessage,
599
+ createQueueWorker,
600
+ };
@@ -946,6 +946,11 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
946
946
  let shell = null;
947
947
  const getShell = () => shell || (shell = new ShellSession({ cwd: envFor(payload).cwd }));
948
948
  const toolCtx = { getShell, signal };
949
+ // The turn's working directory. It rides on every tool frame because a
950
+ // tool row's args are the command or the path — never the directory the
951
+ // command runs in — which left "where is it working in" unanswerable from
952
+ // the transcript.
953
+ const turnCwd = envFor(payload).cwd;
949
954
 
950
955
  try {
951
956
  const cfg = cls === 'aegis' || cls === 'ollama' ? {} : settings.get(cls) || {};
@@ -1207,6 +1212,18 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
1207
1212
  });
1208
1213
 
1209
1214
  for (const call of calls) {
1215
+ // Open the live row BEFORE the tool runs. The CLI paints
1216
+ // `Running N … · 3s` off a `phase: 'run'` frame and ticks it while
1217
+ // the tool is still going; reporting the tool only once, after it
1218
+ // finished, left the transcript showing a spinner verb and nothing
1219
+ // else for the whole exec timeout (120s, up to 600s) — which reads
1220
+ // as a wedged turn rather than a working one.
1221
+ if (onDelta) {
1222
+ onDelta({
1223
+ delta: '',
1224
+ tool: { name: call.name, args: call.args, id: call.id, cwd: turnCwd, phase: 'run' },
1225
+ });
1226
+ }
1210
1227
  const result = call.name === T.SUBAGENT_TOOL
1211
1228
  ? await runSubagent(call.args, {
1212
1229
  cls, model, statedMaxTokens, effort: payload && payload.effort,
@@ -1215,7 +1232,15 @@ function createLocalEngine({ aegis, settings, ollama, providers, tools, promptBu
1215
1232
  : await gatedExecuteTool(call, { toolCtx, rootSessionId, rootOnDelta, signal });
1216
1233
  // A subagent's spend rides back on its tool result (see runSubagent).
1217
1234
  if (result && result.usage) addUsage({ usage: result.usage });
1218
- if (onDelta) onDelta({ delta: '', tool: { name: call.name, args: call.args, ok: result.ok } });
1235
+ // The id is carried on both frames: the CLI pairs them by id first,
1236
+ // precisely so parallel same-name calls don't collapse into one row
1237
+ // with the wrong plural.
1238
+ if (onDelta) {
1239
+ onDelta({
1240
+ delta: '',
1241
+ tool: { name: call.name, args: call.args, ok: result.ok, id: call.id, cwd: turnCwd, phase: 'done' },
1242
+ });
1243
+ }
1219
1244
  history.push({
1220
1245
  role: 'tool',
1221
1246
  tool_call_id: call.id,