hilos-agent 0.7.0 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/handler.mjs CHANGED
@@ -25,6 +25,7 @@ import {
25
25
  prTitleBody,
26
26
  mentionHandle,
27
27
  detectPrContinuation,
28
+ selfDrivenShipPlan,
28
29
  } from "./daemon.mjs";
29
30
  import {
30
31
  runCli,
@@ -36,17 +37,28 @@ import {
36
37
  scrubHilosEnv,
37
38
  envForCwd,
38
39
  } from "./cli.mjs";
39
- import { makeStreamParser } from "./agent-events.mjs";
40
+ import { makeStreamParser, createUsageFold } from "./agent-events.mjs";
40
41
  import {
41
42
  detectVendor,
42
43
  codeStreamArgs,
43
44
  codeDirArgs,
45
+ codeImageArgs,
44
46
  codeProjectKey,
47
+ imagesNeedReading,
45
48
  attachTarget,
46
49
  createProgressEmitter,
47
50
  fastChatCmd,
48
51
  } from "./progress-emitter.mjs";
52
+ import { createTranscriptTap } from "./transcript.mjs";
53
+ import { imagePromptNote, renderAttachmentLine } from "./attachments.mjs";
49
54
  import { resolveFollowupMode, classifyFollowupCue, normalizeSignal } from "./followup.mjs";
55
+ import {
56
+ anchorIterateDecision,
57
+ confirmAnchorOpen,
58
+ prActionRelayDecision,
59
+ prActionReply,
60
+ staleAnchorNote,
61
+ } from "./thread-pr.mjs";
50
62
  import { buildResumeArgs, resumeDecision, readStateEntry, writeState, HILOS_DIR } from "./resume.mjs";
51
63
  import { createModelArgsResolver } from "./model-resolve.mjs";
52
64
  import {
@@ -58,6 +70,18 @@ import {
58
70
  import { buildMemoryBlock } from "./memory.mjs";
59
71
  import { deployFolder, resolveDeployTarget } from "./deploy.mjs";
60
72
  import { runOpenCodeHttpSession } from "./opencode-session.mjs";
73
+ import { runAcpSession } from "./acp-session.mjs";
74
+ import {
75
+ shouldGateClaudePermissions,
76
+ startClaudePermissionServer,
77
+ } from "./claude-permissions.mjs";
78
+ import {
79
+ codexMcpTransportUnavailable,
80
+ codexSandboxFromArgs,
81
+ runCodexMcpSession,
82
+ shouldGateCodexPermissions,
83
+ } from "./codex-mcp-session.mjs";
84
+ import { createUngatedRunNotice } from "./permission-gate.mjs";
61
85
 
62
86
  /**
63
87
  * The environment for a coding/chat CLI run. runCli always strips HILOS_* on top
@@ -138,6 +162,49 @@ function compactRunMarker(status, branch) {
138
162
  return `Failed:${b}`;
139
163
  }
140
164
 
165
+ /**
166
+ * A `--model x` / `-m x` the operator pinned on the coding command (0787).
167
+ * `modelArgsFor` deliberately returns [] in that case — the hand-pin wins — so
168
+ * this is where a pinned id is recovered for the receipt.
169
+ */
170
+ export function pinnedModelId(codingCmd) {
171
+ const m = /(?:^|\s)(?:--model|-m)(?:\s+|=)("[^"]+"|'[^']+'|\S+)/.exec(String(codingCmd || ""));
172
+ return m ? m[1].replace(/^["']|["']$/g, "") : "";
173
+ }
174
+
175
+ /**
176
+ * The `usage` argument for a report (0787), or `{}` when the CLI told us
177
+ * nothing — an omitted field is the honest answer, and the card degrades to
178
+ * "Tokens unavailable" rather than claiming a free run.
179
+ *
180
+ * TAKES from the fold. Every report writes its OWN ledger row server-side, so a
181
+ * run that reports more than once (a proposal card and then a shipped card, a
182
+ * revision round per loop) must carry only what is new since the last report;
183
+ * a running total would bill the same tokens twice.
184
+ *
185
+ * The model: whatever the stream named wins. Codex names none, so the id
186
+ * resolved at spawn — or pinned on the command — stands in, and failing both,
187
+ * the CLI's own name. The ledger has to say what ran.
188
+ */
189
+ export function reportUsageArgs(fold, { vendor, modelId, runId } = {}) {
190
+ const totals = fold && typeof fold.take === "function" ? fold.take() : null;
191
+ if (!totals) return {};
192
+ const cli = vendor && vendor !== "unknown" ? vendor : null;
193
+ const model = totals.model || modelId || cli || "";
194
+ if (!model) return {};
195
+ const usage = {
196
+ model,
197
+ inputTokens: totals.inputTokens,
198
+ outputTokens: totals.outputTokens,
199
+ cacheReadTokens: totals.cacheReadTokens,
200
+ cacheCreationTokens: totals.cacheCreationTokens,
201
+ };
202
+ if (cli) usage.vendor = cli;
203
+ if (typeof totals.costUsd === "number") usage.costUsd = totals.costUsd;
204
+ if (runId) usage.runId = runId;
205
+ return { usage };
206
+ }
207
+
141
208
  // Every helper below runs a tool in a directory we choose, so each one hands
142
209
  // the child a PWD that matches that directory instead of the daemon's launch
143
210
  // dir (0615). git and gh both use the real cwd, so this is hygiene rather than
@@ -152,6 +219,9 @@ function defaultDeps() {
152
219
  // selected trust boundary without spawning a real model process.
153
220
  runCli: (opts) => runCli(opts),
154
221
  runOpenCodeHttpSession: (opts) => runOpenCodeHttpSession(opts),
222
+ runAcpSession: (opts) => runAcpSession(opts),
223
+ runCodexMcpSession: (opts) => runCodexMcpSession(opts),
224
+ startClaudePermissionServer: (opts) => startClaudePermissionServer(opts),
155
225
  // Does a path exist on disk? Injectable so folder mode's "missing folder"
156
226
  // guard is unit-testable without touching the real filesystem.
157
227
  pathExists: (p) => existsSync(p),
@@ -164,7 +234,7 @@ function defaultDeps() {
164
234
  { cwd, env: cwdEnv(cwd), encoding: "utf8" },
165
235
  );
166
236
  const url = (r.stdout || "").trim().split("\n").filter(Boolean).pop() || null;
167
- return { ok: r.status === 0, url, stderr: r.stderr || "" };
237
+ return { ok: r.status === 0, url, stdout: r.stdout || "", stderr: r.stderr || "" };
168
238
  },
169
239
  // The open PR for a branch, if one already exists — so when the CLI opened a
170
240
  // PR itself we report THAT instead of opening a duplicate.
@@ -189,6 +259,37 @@ function defaultDeps() {
189
259
  const out = (r.stdout || "").trim();
190
260
  return r.status === 0 && out ? out : null;
191
261
  },
262
+ // The live state of a PR (0704): is it still OPEN, what is its head branch,
263
+ // and does that head live in the base repo? The staleness guard for adopting
264
+ // the thread's pull request — a merged/closed anchor must never take a push,
265
+ // and a fork head is another repository's branch. Returns null when `gh`
266
+ // can't answer, which the caller reads as "no proof, no adoption".
267
+ prState: (cwd, ref) => {
268
+ const r = spawnSync(
269
+ "gh",
270
+ [
271
+ "pr",
272
+ "view",
273
+ String(ref),
274
+ "--json",
275
+ "state,headRefName,headRepository,headRepositoryOwner",
276
+ ],
277
+ { cwd, env: cwdEnv(cwd), encoding: "utf8" },
278
+ );
279
+ if (r.status !== 0) return null;
280
+ try {
281
+ const j = JSON.parse(r.stdout || "{}");
282
+ const owner = j.headRepositoryOwner?.login || j.headRepositoryOwner?.name || null;
283
+ const repo = j.headRepository?.name || null;
284
+ return {
285
+ state: typeof j.state === "string" ? j.state : "",
286
+ headRefName: typeof j.headRefName === "string" ? j.headRefName : null,
287
+ headRepoFullName: owner && repo ? `${owner}/${repo}` : null,
288
+ };
289
+ } catch {
290
+ return null;
291
+ }
292
+ },
192
293
  sleep: (ms) => new Promise((r) => setTimeout(r, ms)),
193
294
  now: () => Date.now(),
194
295
  };
@@ -199,7 +300,13 @@ function defaultDeps() {
199
300
  * Both repo and direct-folder runs use this exact callback contract so neither
200
301
  * path can accidentally become the ungated exception.
201
302
  */
202
- function openCodePermissionCallbacks({ tool, channelId, threadRoot, runId = null }) {
303
+ function openCodePermissionCallbacks({
304
+ tool,
305
+ channelId,
306
+ threadRoot,
307
+ runId = null,
308
+ provider = "opencode",
309
+ }) {
203
310
  return {
204
311
  requestPermission: async (request) => {
205
312
  const detail =
@@ -216,7 +323,7 @@ function openCodePermissionCallbacks({ tool, channelId, threadRoot, runId = null
216
323
  channelId,
217
324
  threadRootId: threadRoot,
218
325
  ...(runId ? { runId } : {}),
219
- provider: "opencode",
326
+ provider,
220
327
  vendorSessionId: request.sessionId,
221
328
  vendorRequestId: request.vendorRequestId,
222
329
  action: request.action,
@@ -241,7 +348,7 @@ function openCodePermissionCallbacks({ tool, channelId, threadRoot, runId = null
241
348
  }
242
349
  return tool("get_permission_decision", {
243
350
  requestId,
244
- provider: "opencode",
351
+ provider,
245
352
  vendorSessionId: request.sessionId,
246
353
  vendorRequestId: request.vendorRequestId,
247
354
  ...(failClosed ? { failClosed: true } : {}),
@@ -250,6 +357,145 @@ function openCodePermissionCallbacks({ tool, channelId, threadRoot, runId = null
250
357
  };
251
358
  }
252
359
 
360
+ /**
361
+ * The post_progress sender both run lanes use, with the 0782 stop check folded
362
+ * into the beat they already send.
363
+ *
364
+ * A person can press Stop on the live card from anywhere — the person who
365
+ * delegated the work usually is not the one at this laptop. The server records
366
+ * that on the run's own row and answers the NEXT heartbeat with
367
+ * `stopRequested: true` (plus `stoppedBy` when it knows the name). So the daemon
368
+ * learns it on its existing cadence: no poll tool, no second timer, nothing new
369
+ * to keep alive.
370
+ *
371
+ * `onStopRequested` is called at most once per sender and must never throw into
372
+ * the run — it is the daemon's cue to tear the process group down.
373
+ *
374
+ * @param {{ tool: Function, statusId: string, runId?: string|null,
375
+ * onStopRequested?: (info: {by: string|null}) => any }} o
376
+ */
377
+ export function createProgressSender({ tool, statusId, runId = null, onStopRequested }) {
378
+ let inflight = Promise.resolve();
379
+ let stopSeen = false;
380
+ const takeStop = (res) => {
381
+ if (!res || res.stopRequested !== true) return false;
382
+ if (stopSeen || !onStopRequested) return true;
383
+ stopSeen = true;
384
+ try {
385
+ onStopRequested({ by: typeof res.stoppedBy === "string" ? res.stoppedBy : null });
386
+ } catch {
387
+ /* the stop hand-off must never break the run's teardown */
388
+ }
389
+ return true;
390
+ };
391
+ return {
392
+ /** @param {object} p — the emitter's snapshot. */
393
+ send(p) {
394
+ try {
395
+ const r = tool("post_progress", {
396
+ messageId: statusId,
397
+ ...(runId ? { runId } : {}),
398
+ progress: p,
399
+ });
400
+ if (r && typeof r.then === "function") {
401
+ const done = r.then((res) => void takeStop(res), () => {});
402
+ inflight = Promise.all([inflight, done]).then(
403
+ () => {},
404
+ () => {},
405
+ );
406
+ }
407
+ } catch {
408
+ /* a progress send must never break the run */
409
+ }
410
+ },
411
+ /**
412
+ * Ask the server outright, and WAIT for the answer.
413
+ *
414
+ * The heartbeat above only fires when the CLI says something. A run sitting
415
+ * inside one silent ten-minute command emits nothing, so nothing would carry
416
+ * a stop home — the person would press Stop and watch their laptop keep
417
+ * going. This is the same call, on a timer, and it is also the authoritative
418
+ * pre-ship check: `state` echoes the card's CURRENT lifecycle so asking
419
+ * never repaints it (a stopped run's card is frozen server-side anyway).
420
+ *
421
+ * @param {"working"|"done"|"error"} [state]
422
+ * @returns {Promise<boolean>} true when a person has stopped this run.
423
+ */
424
+ async poll(state = "working") {
425
+ try {
426
+ const res = await tool("post_progress", {
427
+ messageId: statusId,
428
+ ...(runId ? { runId } : {}),
429
+ progress: { state },
430
+ });
431
+ return takeStop(res);
432
+ } catch {
433
+ // A stop check that can't reach the server is not a stop. The run
434
+ // continues; the next tick asks again.
435
+ return false;
436
+ }
437
+ },
438
+ /** Every dispatched write, so a caller can wait them out before settling. */
439
+ drain() {
440
+ return inflight;
441
+ },
442
+ };
443
+ }
444
+
445
+ /** How often a run asks whether it has been stopped while its CLI is silent.
446
+ * Well under the card's own staleness window, far above a chatty write rate. */
447
+ export const STOP_POLL_MS = 15_000;
448
+
449
+ /**
450
+ * Arm the silent-work stop poll for ONE active job. Timers are injectable so
451
+ * this is unit-tested on a fake clock, and the returned `stop()` must be called
452
+ * from the same finally that tears the run down — one timer per job, never a
453
+ * timer that outlives the work it was watching.
454
+ *
455
+ * @param {{ poll: () => Promise<boolean>, intervalMs?: number,
456
+ * setTimer?: Function, clearTimer?: Function }} o
457
+ */
458
+ export function createStopPoller({
459
+ poll,
460
+ intervalMs = STOP_POLL_MS,
461
+ setTimer = (fn, ms) => {
462
+ const id = setInterval(fn, ms);
463
+ if (id && typeof id.unref === "function") id.unref();
464
+ return id;
465
+ },
466
+ clearTimer = (id) => clearInterval(id),
467
+ } = {}) {
468
+ let id = null;
469
+ let asking = false;
470
+ let done = false;
471
+ const tick = async () => {
472
+ // Never stack asks: a slow server must not queue a burst of stop checks.
473
+ if (asking || done) return;
474
+ asking = true;
475
+ try {
476
+ if (await poll()) {
477
+ done = true; // the stop is handed over once; teardown owns the rest
478
+ stop();
479
+ }
480
+ } catch {
481
+ /* a stop check must never break the run */
482
+ } finally {
483
+ asking = false;
484
+ }
485
+ };
486
+ function stop() {
487
+ if (id == null) return;
488
+ try {
489
+ clearTimer(id);
490
+ } catch {
491
+ /* ignore */
492
+ }
493
+ id = null;
494
+ }
495
+ id = setTimer(() => void tick(), intervalMs);
496
+ return { stop, tick };
497
+ }
498
+
253
499
  /** The local HTTP bridge can own only a local OpenCode server. An explicit
254
500
  * `--attach` remains on OpenCode's CLI responder, which rejects unanswered asks
255
501
  * fail closed; taking over a remote server requires a separate authenticated
@@ -268,6 +514,210 @@ export function shouldUseRuntimePermissionBridge({
268
514
  );
269
515
  }
270
516
 
517
+ /** ACP transport (0759; slice 1 opencode, slice 2 cursor). Only runs the
518
+ * workspace already gates with runtime permissions qualify, and only when the
519
+ * vendor's own auto-allow escape hatches are absent — over ACP, hilos must be
520
+ * the one answering asks, so a run configured to never ask has nothing to gate
521
+ * here.
522
+ *
523
+ * 0778 removed two limits that were never about correctness. `acpTransport`
524
+ * now defaults ON (config.mjs) for the vendors whose adapter is proven, and a
525
+ * RESUMED run no longer disqualifies: both live vendors advertise ACP's
526
+ * `loadSession` capability, so acp-session.mjs resumes the session and keeps
527
+ * raising cards. Setting `acpTransport: false` still opts a workspace out. */
528
+ export function shouldUseAcpTransport({
529
+ vendor,
530
+ acpTransport,
531
+ runtimePermissions,
532
+ codeArgs,
533
+ codingCmd,
534
+ }) {
535
+ if (acpTransport !== true) return false;
536
+ if (runtimePermissions !== true) return false;
537
+ if (vendor === "opencode") {
538
+ return shouldUseRuntimePermissionBridge({
539
+ vendor,
540
+ runtimePermissions,
541
+ codeArgs,
542
+ codingCmd,
543
+ });
544
+ }
545
+ if (vendor === "cursor") {
546
+ // -f/--force is cursor's "allow everything" switch — the exact analog of
547
+ // opencode's --auto.
548
+ return !codeArgs.includes("-f") && !codeArgs.includes("--force");
549
+ }
550
+ return false;
551
+ }
552
+
553
+ /**
554
+ * The env a gated claude run gets (0777).
555
+ *
556
+ * A permission card can legitimately wait on a person for minutes — a 150s wait
557
+ * was verified live — so the CLI's own MCP tool timeout must not cut the ask
558
+ * short before hilos's run deadline does. Everything else about the env is
559
+ * unchanged (runCli still strips HILOS_* itself).
560
+ */
561
+ function claudeGateEnv(cfg) {
562
+ const base = codingChildEnv(cfg) || process.env;
563
+ const budget = Math.max(60_000, Number(cfg?.runTimeoutMs) || 0) + 60_000;
564
+ return { ...base, MCP_TOOL_TIMEOUT: String(budget) };
565
+ }
566
+
567
+ /**
568
+ * Run claude_code through its permission-prompt seam (0777).
569
+ *
570
+ * Deliberately NOT a new transport: the ordinary argv run is untouched — same
571
+ * runCli, same resume args, same streaming — and the gate is two extra flags
572
+ * plus a loopback MCP server that lives exactly as long as the run. The prompt
573
+ * stays the final argument.
574
+ */
575
+ async function runClaudeGatedCli({
576
+ deps,
577
+ cfg,
578
+ onGateDropped,
579
+ cmd,
580
+ codeArgs,
581
+ prompt,
582
+ cwd,
583
+ signal,
584
+ onData,
585
+ permissionCallbacks,
586
+ sessionId = "",
587
+ resolveSessionId,
588
+ log = console,
589
+ }) {
590
+ const server = await deps.startClaudePermissionServer({
591
+ ...permissionCallbacks,
592
+ sessionId,
593
+ // A FRESH run has no session id to hand over: claude reveals its own in the
594
+ // init frame of its stream, which the progress emitter is already folding.
595
+ // Pulling from that snapshot means an ask carries the CLI's real session id
596
+ // without a second parser — and hilos rejects an empty one outright.
597
+ resolveSessionId,
598
+ timeoutMs: cfg.runTimeoutMs,
599
+ signal,
600
+ log,
601
+ });
602
+ try {
603
+ return await deps.runCli({
604
+ cmd,
605
+ args: [...codeArgs, ...server.args, prompt],
606
+ cwd,
607
+ timeoutMs: cfg.runTimeoutMs,
608
+ label: "coding",
609
+ signal,
610
+ env: claudeGateEnv(cfg),
611
+ onData,
612
+ // 0785 — a CLI that rejects the gate flags is retried without them. This
613
+ // fires BEFORE that ungated retry is spawned, so the room hears about the
614
+ // downgrade first rather than after the fact.
615
+ onPermissionGateDropped: onGateDropped,
616
+ });
617
+ } finally {
618
+ try {
619
+ await server.close();
620
+ } catch {
621
+ /* tearing the gate down must never fail a finished run */
622
+ }
623
+ }
624
+ }
625
+
626
+ /**
627
+ * Will this codex run take the gated transport (0785)?
628
+ *
629
+ * Answerable before the run's argv exists, because the only input that can
630
+ * change the answer is the operator's OWN command — every flag the daemon
631
+ * appends later (model, dir, image, resume, stream) is ours and none of them is
632
+ * the bypass tier. Which matters because the PROMPT is written before the
633
+ * transport is chosen, and what the prompt may claim about images depends on it.
634
+ */
635
+ function codexRunIsGated(cfg, caps) {
636
+ return shouldGateCodexPermissions({
637
+ vendor: detectVendor(cfg?.codingCmd),
638
+ runtimePermissions: caps?.runtimePermissions,
639
+ codeArgs: String(cfg?.codingCmd || "").split(" ").filter(Boolean).slice(1),
640
+ });
641
+ }
642
+
643
+ /**
644
+ * Run codex through its gated transport, with an honest fallback (0785).
645
+ *
646
+ * A granted workspace ALWAYS gets `codex mcp-server` — that is what the grant
647
+ * means, and it is the only codex transport that can ask (`codex exec` has no
648
+ * approval channel at any flag combination, see codex-mcp-session.mjs). The one
649
+ * thing the grant can't conjure is the subcommand itself: a codex old enough
650
+ * not to have it never answers the MCP handshake, and before this the run just
651
+ * failed. Now it degrades to the plain exec argv — the ungated run every codex
652
+ * did before 0777, never worse.
653
+ *
654
+ * Two rules make that degrade safe. It happens ONLY on positive evidence that
655
+ * the subcommand is absent (codexMcpTransportUnavailable — every other
656
+ * pre-handshake failure is returned as the failed run it is, because a wrongly
657
+ * failed run is recoverable and a wrongly ungated one is not). And the room is
658
+ * told BEFORE the ungated child is spawned, so the warning arrives while the
659
+ * run can still be stopped.
660
+ *
661
+ * The exec argv is the run's real one (model, images, resume, stream flags),
662
+ * so the fallback is the same run the ungated lane would have made.
663
+ */
664
+ async function runCodexGatedSession({
665
+ deps,
666
+ cfg,
667
+ onGateDropped,
668
+ cmd,
669
+ codeArgs,
670
+ prompt,
671
+ cwd,
672
+ model,
673
+ resumeThreadId = null,
674
+ signal,
675
+ onData,
676
+ onEvent,
677
+ permissionCallbacks,
678
+ }) {
679
+ const run = await deps.runCodexMcpSession({
680
+ cmd,
681
+ cwd,
682
+ prompt,
683
+ sandbox: codexSandboxFromArgs(codeArgs),
684
+ // 0785 — the resolved tier (0783) or a hand-pinned id. The gated transport
685
+ // took the account default before this, so a preset was silently ignored
686
+ // exactly where the operator was most likely to have set one.
687
+ model: model || null,
688
+ resumeThreadId,
689
+ timeoutMs: cfg.runTimeoutMs,
690
+ signal,
691
+ env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
692
+ onData,
693
+ onEvent,
694
+ ...permissionCallbacks,
695
+ });
696
+ if (!codexMcpTransportUnavailable(run)) return run;
697
+ console.log(
698
+ " code → this codex has no `mcp-server`; running WITHOUT the hilos permission gate",
699
+ );
700
+ // Warn FIRST — before a single ungated command can run.
701
+ if (typeof onGateDropped === "function") {
702
+ try {
703
+ await onGateDropped();
704
+ } catch {
705
+ /* telling the room must never break the run */
706
+ }
707
+ }
708
+ const fallback = await deps.runCli({
709
+ cmd,
710
+ args: [...codeArgs, prompt],
711
+ cwd,
712
+ timeoutMs: cfg.runTimeoutMs,
713
+ label: "coding",
714
+ signal,
715
+ env: codingChildEnv(cfg),
716
+ onData,
717
+ });
718
+ return { ...fallback, permissionGateDropped: true };
719
+ }
720
+
271
721
  async function awaitDecision({ tool, channelId, reportMessageId, cfg, deps, parentId, signal }) {
272
722
  if (!reportMessageId) return { kind: "timeout" };
273
723
  const deadline = deps.now() + cfg.decisionTimeoutMs;
@@ -318,7 +768,7 @@ async function linkPrFromUrl({ tool, channelId, url }) {
318
768
  await tool("link_pr", { channelId, repoFullName: m[1], prNumber: Number(m[2]) }).catch(() => {});
319
769
  }
320
770
 
321
- async function applyDecision({ decision, repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, existingPrUrl, settleId, runId }) {
771
+ async function applyDecision({ decision, repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, existingPrUrl, settleId, runId, usageArgs = {} }) {
322
772
  const tag = requesterTag(requester);
323
773
  const lead = tag ? `${tag} — ` : "";
324
774
  // `parentId` here is the run's thread root. Terminal outcomes ask the server
@@ -377,13 +827,21 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
377
827
  : `${lead}pushed \`${branch}\`. Open a PR manually — \`gh\` failed.`,
378
828
  prUrl: pr.ok && pr.url ? pr.url : undefined,
379
829
  caveats: pr.ok ? [] : [`gh pr create failed: ${(pr.stderr || "").trim().slice(0, 200)}`],
830
+ // 0787 — what the run cost, when the caller had a receipt to hand over.
831
+ // Ungated runs land here with the whole run's usage; a gated one already
832
+ // spent it on the proposal card and passes {}.
833
+ ...usageArgs,
380
834
  };
381
- await tool(
835
+ // 0792 — keep the id of the card this run settled onto, so the caller can
836
+ // mark it with the run's transcript. `settleId` is that card when the run
837
+ // streamed onto a live status message; otherwise it is the fresh report.
838
+ const reportRes = await tool(
382
839
  "post_report",
383
840
  settleId
384
841
  ? { ...reportArgs, messageId: settleId, broadcast: true }
385
842
  : { ...reportArgs, parentId, broadcast: true },
386
843
  );
844
+ const reportMsgId = settleId || reportRes?.messageId || null;
387
845
  // Work produced a PR → attach it to the channel so its live pill shows up.
388
846
  await linkPrFromUrl({ tool, channelId, url: pr.ok ? pr.url : null });
389
847
  // Bind the PR to the thread's run (0279/0280) so a follow-up continues it.
@@ -393,10 +851,13 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
393
851
  runId,
394
852
  ...(pr.ok && pr.url ? { prUrl: pr.url } : {}),
395
853
  status: "awaiting_review",
396
- ...(settleId ? { reportMsgId: settleId } : {}),
854
+ // 0792: record the card even when it was a fresh report rather than a
855
+ // settle — it is the run's card either way, and the transcript upload
856
+ // falls back to this exact field when it is not handed an id.
857
+ ...(reportMsgId ? { reportMsgId } : {}),
397
858
  }).catch(() => {});
398
859
  }
399
- return { status: "pushed", branch, prUrl: pr.ok ? pr.url : null };
860
+ return { status: "pushed", branch, prUrl: pr.ok ? pr.url : null, reportMsgId };
400
861
  }
401
862
 
402
863
  if (decision.kind === "rejected") {
@@ -431,55 +892,100 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
431
892
  * The coding CLI committed the work itself (autonomous run with skip-permissions
432
893
  * in a repo whose docs prescribe a commit+PR flow), so the working tree is clean
433
894
  * and the daemon's diff is empty. Don't claim "no changes": adopt what it did —
434
- * push the branch it's on (no-op if already pushed), reuse the PR it opened or
435
- * open one, and report it. Used only when NOT gated (bias-to-action).
895
+ * push its HEAD under the daemon-owned task branch, reuse the PR it opened or
896
+ * open one, and report it. Keeping the intended head matters: a child that
897
+ * switched to `main` would otherwise make `gh` try to open main → main. Used
898
+ * only when NOT gated (bias-to-action).
436
899
  */
437
- async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, settleId, runId }) {
900
+ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, settleId, runId, usageArgs = {} }) {
438
901
  const tag = requesterTag(requester);
439
902
  const lead = tag ? `${tag} — ` : "";
440
- // The agent may have committed on the daemon's branch or switched to one of its
441
- // own — push whatever HEAD is on now.
442
- const cur =
443
- (deps.git(repoPath, ["symbolic-ref", "--quiet", "--short", "HEAD"]).stdout || "").trim() || branch;
444
- const push = deps.git(repoPath, ["push", "-u", "origin", cur]);
445
- let prUrl = deps.findPR ? deps.findPR(repoPath, cur) : null;
903
+ const currentBranch =
904
+ (deps.git(repoPath, ["symbolic-ref", "--quiet", "--short", "HEAD"]).stdout || "").trim();
905
+ const ship = selfDrivenShipPlan(currentBranch, branch);
906
+ const push = deps.git(repoPath, ship.pushArgs);
907
+ let prUrl = deps.findPR ? deps.findPR(repoPath, ship.headBranch) : null;
908
+ let prAttempt = null;
446
909
  if (!prUrl && push.status === 0) {
447
- const { title, body } = prTitleBody(task, cur);
448
- const pr = deps.openPR(repoPath, { title, body, branch: cur, base: cfg.defaultBranch });
449
- prUrl = pr.ok && pr.url ? pr.url : null;
910
+ const { title, body } = prTitleBody(task, ship.headBranch);
911
+ prAttempt = deps.openPR(repoPath, {
912
+ title,
913
+ body,
914
+ branch: ship.headBranch,
915
+ base: cfg.defaultBranch,
916
+ });
917
+ prUrl = prAttempt.ok && prAttempt.url ? prAttempt.url : null;
450
918
  }
451
- const { title } = prTitleBody(task, cur);
919
+ const { title } = prTitleBody(task, ship.headBranch);
920
+ const recoveredNote = ship.recoveredFrom
921
+ ? ` The child had switched to \`${ship.recoveredFrom}\`; hilos recovered its HEAD onto \`${ship.headBranch}\`.`
922
+ : "";
923
+ const prFailure =
924
+ prAttempt && !prUrl
925
+ ? oneLine(
926
+ (prAttempt.stderr || prAttempt.stdout || "gh did not return a pull request URL").trim(),
927
+ 200,
928
+ )
929
+ : "";
452
930
  // Seamless single card (0289): settle onto the live status message when one was
453
931
  // streaming this run; the workspace setting decides channel visibility.
454
932
  // Otherwise post a fresh final report.
455
933
  const reportArgs = {
456
934
  channelId,
457
- title: `Shipped: ${title}`,
935
+ title: prUrl
936
+ ? `Shipped: ${title}`
937
+ : push.status === 0
938
+ ? `PR failed: ${title}`
939
+ : `Push failed: ${title}`,
458
940
  summary: prUrl
459
- ? `${lead}the coding agent committed on \`${cur}\` and opened a pull request.`
941
+ ? `${lead}the coding agent committed the work itself and opened a pull request from \`${ship.headBranch}\`.${recoveredNote}`
460
942
  : push.status === 0
461
- ? `${lead}the coding agent committed and pushed \`${cur}\`. Open a PR manually — \`gh\` didn't return one.`
462
- : `${lead}the coding agent committed on \`${cur}\` locally, but pushing it failed.`,
943
+ ? `${lead}the coding agent committed and pushed \`${ship.headBranch}\`, but GitHub rejected PR creation.${recoveredNote}`
944
+ : `${lead}the coding agent committed locally, but pushing \`${ship.headBranch}\` failed.${recoveredNote}`,
463
945
  prUrl: prUrl || undefined,
464
- caveats: push.status === 0 ? [] : [`git push failed: ${(push.stderr || "").trim().slice(0, 200)}`],
946
+ caveats:
947
+ push.status !== 0
948
+ ? [`git push failed: ${(push.stderr || "").trim().slice(0, 200)}`]
949
+ : prFailure
950
+ ? [`gh pr create failed: ${prFailure}`]
951
+ : [],
952
+ // 0787 — an autonomous run's receipt rides its one and only report.
953
+ ...usageArgs,
465
954
  };
466
- await tool(
955
+ const reportRes = await tool(
467
956
  "post_report",
468
957
  settleId
469
958
  ? { ...reportArgs, messageId: settleId, broadcast: true }
470
959
  : { ...reportArgs, parentId, broadcast: true },
471
960
  );
961
+ // 0792 — the card this run settled onto, for the transcript stamp.
962
+ const reportMsgId = settleId || reportRes?.messageId || null;
472
963
  if (prUrl) await linkPrFromUrl({ tool, channelId, url: prUrl });
473
964
  // Bind the PR to the thread's run (0279/0280). Best-effort; only when recorded.
474
965
  if (runId) {
475
966
  await tool("update_run", {
476
967
  runId,
477
968
  ...(prUrl ? { prUrl } : {}),
478
- status: "awaiting_review",
479
- ...(settleId ? { reportMsgId: settleId } : {}),
969
+ status: prUrl ? "awaiting_review" : "failed",
970
+ ...(!prUrl
971
+ ? {
972
+ reason:
973
+ push.status !== 0
974
+ ? oneLine((push.stderr || "git push failed").trim(), 300)
975
+ : prFailure || "GitHub did not return a pull request URL",
976
+ }
977
+ : {}),
978
+ ...(reportMsgId ? { reportMsgId } : {}),
480
979
  }).catch(() => {});
481
980
  }
482
- return { status: push.status === 0 ? "pushed" : "commit-local", branch: cur, prUrl };
981
+ const status =
982
+ push.status !== 0 ? "commit-local" : prUrl ? "pushed" : "pr-failed";
983
+ return {
984
+ status,
985
+ branch: ship.headBranch,
986
+ prUrl,
987
+ reportMsgId,
988
+ };
483
989
  }
484
990
 
485
991
  // Sentinel the router model emits when the latest message is a request to change
@@ -580,16 +1086,21 @@ async function routeIntent({ name, repoFullName, transcript, workspaceMemory, cf
580
1086
  * agent, edit files now" — led by the router's distilled `brief` (what to build),
581
1087
  * with the conversation included only as background. Falls back to the raw
582
1088
  * mention when there's no brief.
583
- * @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, repoFullName?: string }} [o]
1089
+ * @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, repoFullName?: string, images?: {path: string, name: string, type: string}[], imagesReadable?: boolean }} [o]
584
1090
  */
585
1091
  export function codeTaskPrompt(o) {
586
- const { message, context, brief, repoFullName } = o || {};
1092
+ const { message, context, brief, repoFullName, images, imagesReadable } = o || {};
587
1093
  const task = (brief && brief.trim()) || String(message?.body || "").trim();
588
1094
  const transcript = context?.transcript?.trim();
589
1095
  const where = repoFullName ? ` in the git repository ${repoFullName}` : "";
590
1096
  let p =
591
1097
  `You are a coding agent working${where}. Implement the following by EDITING FILES now — ` +
592
1098
  `make the changes directly, do not just describe them, do not ask questions:\n\n${task}`;
1099
+ // 0779: the screenshots the request was about, already on this machine. Right
1100
+ // under the task, because "fix this spacing" only means something next to the
1101
+ // picture of the spacing.
1102
+ const imageNote = imagePromptNote(images, { readable: imagesReadable });
1103
+ if (imageNote) p += `\n\n${imageNote}`;
593
1104
  // The daemon owns git: it stages, commits, pushes, and opens the PR after the
594
1105
  // CLI finishes. An autonomous CLI run with skip-permissions inside a repo whose
595
1106
  // docs prescribe a commit+PR workflow will otherwise do all of that itself,
@@ -612,16 +1123,19 @@ export function codeTaskPrompt(o) {
612
1123
  * imperative "edit files now" framing as codeTaskPrompt, but it tells the agent
613
1124
  * its edits land straight in the folder and bans git/gh (there's nothing for the
614
1125
  * daemon to commit; the changes ARE the deliverable).
615
- * @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, folderPath?: string }} [o]
1126
+ * @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, folderPath?: string, images?: {path: string, name: string, type: string}[], imagesReadable?: boolean }} [o]
616
1127
  */
617
1128
  export function folderTaskPrompt(o) {
618
- const { message, context, brief, folderPath } = o || {};
1129
+ const { message, context, brief, folderPath, images, imagesReadable } = o || {};
619
1130
  const task = (brief && brief.trim()) || String(message?.body || "").trim();
620
1131
  const where = folderPath ? ` in the local folder ${folderPath}` : "";
621
1132
  let p =
622
1133
  `You are a coding agent working directly${where}. Implement the following by EDITING ` +
623
1134
  `FILES now — make the changes directly in this folder, do not just describe them, do not ` +
624
1135
  `ask questions:\n\n${task}`;
1136
+ // 0779 — same image handoff as the repo path.
1137
+ const imageNote = imagePromptNote(images, { readable: imagesReadable });
1138
+ if (imageNote) p += `\n\n${imageNote}`;
625
1139
  p +=
626
1140
  `\n\nIMPORTANT: your edits apply DIRECTLY to the user's folder — there is no branch, no ` +
627
1141
  `commit, and no pull request. Do NOT run git; do NOT stage, commit, push, create branches, ` +
@@ -664,7 +1178,17 @@ async function fetchContext({ channelId, tool, parentId }) {
664
1178
  }));
665
1179
  rows = messages;
666
1180
  }
667
- return { rows, transcript: rows.map((m) => `${m.author}: ${m.body}`).join("\n").slice(-6000) };
1181
+ // 0779: the rows carry resolved `attachments`, and the body carries the
1182
+ // unresolvable `![name](attachment:<uuid>)` pill token. Flatten the token and
1183
+ // name the file, so the daemon's chat replies and routing decisions stop
1184
+ // being handed a reference no model can follow.
1185
+ return {
1186
+ rows,
1187
+ transcript: rows
1188
+ .map((m) => `${m.author}: ${renderAttachmentLine(m.body, m.attachments)}`)
1189
+ .join("\n")
1190
+ .slice(-6000),
1191
+ };
668
1192
  }
669
1193
 
670
1194
  /**
@@ -1242,9 +1766,26 @@ async function handleFolderDeploy({
1242
1766
  * live), request-changes = re-run in place (bounded), reject = revert the run's
1243
1767
  * own changes (git repos only).
1244
1768
  */
1245
- async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps, signal, parentId, folderPath, brief, workspaceMemory, context }) {
1769
+ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps, signal, parentId, folderPath, brief, workspaceMemory, context, onStopRequested }) {
1246
1770
  void me;
1247
1771
  const git = deps.git;
1772
+ // 0779 — screenshots the poll loop already pulled to a temp dir at pickup.
1773
+ // The prompt names them; the argv carries them for a vendor that takes one.
1774
+ const localImages = Array.isArray(message?.images) ? message.images : [];
1775
+ const folderImagePromptArgs = localImages.length
1776
+ ? {
1777
+ images: localImages,
1778
+ imagesReadable: imagesNeedReading(detectVendor(cfg.codingCmd), {
1779
+ gated: codexRunIsGated(cfg, caps),
1780
+ }),
1781
+ }
1782
+ : {};
1783
+ // 0785 — one line per run, whichever CLI turns out to be ungateable.
1784
+ const noticeUngatedRun = createUngatedRunNotice({
1785
+ // Read at post time, not now: the thread root is the ack this run posts.
1786
+ post: (body) => tool("post_message", { channelId, parentId: threadRoot, body }),
1787
+ log: console,
1788
+ });
1248
1789
  const tag = requesterTag(message.author);
1249
1790
  const lead = tag ? `${tag} — ` : "";
1250
1791
 
@@ -1291,6 +1832,33 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1291
1832
  // supports it, else the edit-in-place heartbeat, else nothing. Kept across
1292
1833
  // request-changes re-runs on the same status message.
1293
1834
  let progressId = null;
1835
+ // 0787 — what the run cost, folded across every round of the revise loop and
1836
+ // taken at each report so each round's spend is reported exactly once. The
1837
+ // model id the stream names wins; `resolvedModelId` is the spawn-time
1838
+ // fallback for a CLI (codex) whose stream never says.
1839
+ const runUsage = createUsageFold();
1840
+ const cliVendor = detectVendor(cfg.codingCmd);
1841
+ let resolvedModelId = pinnedModelId(cfg.codingCmd);
1842
+ let lastTerminalState = "done";
1843
+ /**
1844
+ * Accounting-only settlement (0787). Plenty of terminal exits produce no
1845
+ * report at all — nothing changed, the CLI never started, a person pressed
1846
+ * Stop — and the tokens those runs spent were being dropped on the floor.
1847
+ * `post_progress` is the terminal call the daemon already makes on every one
1848
+ * of them, it is already run-scoped, and it already carries `runId`, so the
1849
+ * spend rides home on a call that was happening anyway rather than on a new
1850
+ * tool. Only ever the DELTA, so a later report cannot re-report it; a card is
1851
+ * required because there is no other message to hang a terminal update on.
1852
+ */
1853
+ const settleRunUsage = async () => {
1854
+ const args = reportUsageArgs(runUsage, { vendor: cliVendor, modelId: resolvedModelId });
1855
+ if (!args.usage || !progressId) return;
1856
+ await tool("post_progress", {
1857
+ messageId: progressId,
1858
+ progress: { state: lastTerminalState },
1859
+ usage: args.usage,
1860
+ }).catch(() => {});
1861
+ };
1294
1862
  const folderWorking = (elapsedMs, lastLine) => {
1295
1863
  const base = `Working in \`${folderPath}\` — ${fmtElapsed(elapsedMs)} elapsed. I'll report what changed when it's done.`;
1296
1864
  const tail = oneLine(lastLine);
@@ -1305,7 +1873,10 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1305
1873
  const streamOn = Boolean(caps.postProgress);
1306
1874
  const streamArgs = streamOn ? codeStreamArgs(vendor) : [];
1307
1875
  let emitter = null;
1308
- let progressInflight = Promise.resolve();
1876
+ // Every dispatched progress write, so the settle can wait them out (0289).
1877
+ let drainProgress = async () => {};
1878
+ // 0782 — the silent-work stop poll, torn down in the same finally as the run.
1879
+ let stopPoller = null;
1309
1880
  let stopHeartbeat = () => {};
1310
1881
  let lastLine = "";
1311
1882
  // With stream args the CLI's stdout is NDJSON events, not prose — parse it and
@@ -1317,6 +1888,11 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1317
1888
  const foldResultEvents = (events) => {
1318
1889
  for (const ev of events || []) {
1319
1890
  if (ev && ev.t === "result" && ev.summary) resultText = ev.summary;
1891
+ // 0787 — read the numbers off the SAME parse the summary comes from.
1892
+ // This parser exists whenever stream args do, which is a superset of the
1893
+ // cases where a progress emitter exists, so folder usage never depends
1894
+ // on the status card having been posted.
1895
+ runUsage.push(ev);
1320
1896
  }
1321
1897
  };
1322
1898
  if (streamOn) {
@@ -1330,21 +1906,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1330
1906
  }
1331
1907
  const statusId = progressId;
1332
1908
  if (statusId) {
1909
+ // 0782 — a folder run has no runs row, so the stop check is scoped to
1910
+ // the card's thread by the server; it degrades to "no stop" quietly.
1911
+ const sender = createProgressSender({ tool, statusId, onStopRequested });
1333
1912
  emitter = createProgressEmitter({
1334
1913
  parser: makeStreamParser(vendor),
1335
1914
  now: deps.now,
1336
1915
  throttleMs: cfg.progressMs,
1337
- send: (p) => {
1338
- try {
1339
- const r = tool("post_progress", { messageId: statusId, progress: p });
1340
- if (r && typeof r.then === "function") {
1341
- const done = r.then(() => {}, () => {});
1342
- progressInflight = Promise.all([progressInflight, done]).then(() => {}, () => {});
1343
- }
1344
- } catch {
1345
- /* a progress send must never break the run */
1346
- }
1347
- },
1916
+ send: sender.send,
1917
+ });
1918
+ drainProgress = () => sender.drain();
1919
+ // The heartbeat only fires when the CLI speaks. This asks anyway, so a
1920
+ // run inside one long silent command is still stoppable.
1921
+ stopPoller = createStopPoller({
1922
+ poll: () => sender.poll("working"),
1923
+ intervalMs: cfg.stopPollMs || STOP_POLL_MS,
1348
1924
  });
1349
1925
  }
1350
1926
  } else if (caps.editMessage && cfg.heartbeatMs > 0) {
@@ -1377,14 +1953,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1377
1953
  let run;
1378
1954
  // Model preset (0504): same run-time resolution as the repo path.
1379
1955
  const modelArgs = await modelArgsFor(cfg, vendor);
1956
+ // 0787: the id we actually handed the CLI, kept for the usage receipt when
1957
+ // the vendor's own stream never names a model.
1958
+ if (modelArgs[0] === "--model" && modelArgs[1]) resolvedModelId = modelArgs[1];
1380
1959
  // Project pin (0608): opencode reads its project from PWD, so without this
1381
1960
  // a folder run could edit the daemon's launch directory instead of the
1382
1961
  // folder the channel is linked to. [] for every other vendor.
1383
1962
  const dirArgs = codeDirArgs(vendor, folderPath, cfg.codingCmd);
1963
+ // 0779: [] for every vendor without a verified image flag — their argv is
1964
+ // byte-identical to before, and the prompt note still names the files.
1965
+ const imageArgs = codeImageArgs(vendor, localImages);
1384
1966
  const codeArgs = [
1385
1967
  ...parts.slice(1),
1386
1968
  ...modelArgs,
1387
1969
  ...dirArgs,
1970
+ ...imageArgs,
1388
1971
  ...streamArgs,
1389
1972
  ];
1390
1973
  const handleCliData = (c) => {
@@ -1405,14 +1988,78 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1405
1988
  }
1406
1989
  }
1407
1990
  };
1991
+ // The structured transports (ACP, codex's MCP server) produce events
1992
+ // directly — there is no NDJSON on stdout for `resultParser` to read, so
1993
+ // this is where THEIR usage joins the fold (0787). A stdout run never
1994
+ // reaches here, and a structured run never reaches resultParser, so the two
1995
+ // sources can't double-count the same tokens.
1996
+ const handleCliEvent = (ev) => {
1997
+ try {
1998
+ emitter?.foldEvent(ev);
1999
+ } catch {
2000
+ /* a progress fold must never break the run */
2001
+ }
2002
+ try {
2003
+ runUsage.push(ev);
2004
+ } catch {
2005
+ /* accounting must never break the run */
2006
+ }
2007
+ };
1408
2008
  try {
1409
- const useRuntimePermissionBridge = shouldUseRuntimePermissionBridge({
2009
+ const useAcpTransport = shouldUseAcpTransport({
1410
2010
  vendor,
2011
+ acpTransport: cfg.acpTransport,
1411
2012
  runtimePermissions: caps.runtimePermissions,
1412
2013
  codeArgs,
1413
2014
  codingCmd: cfg.codingCmd,
1414
2015
  });
1415
- if (useRuntimePermissionBridge) {
2016
+ const useRuntimePermissionBridge =
2017
+ !useAcpTransport &&
2018
+ shouldUseRuntimePermissionBridge({
2019
+ vendor,
2020
+ runtimePermissions: caps.runtimePermissions,
2021
+ codeArgs,
2022
+ codingCmd: cfg.codingCmd,
2023
+ });
2024
+ // 0777: the two vendors that had no gate at all. Neither is an ACP
2025
+ // adapter — each CLI turned out to have its own native seam (see the
2026
+ // module headers), so both compose with everything already here.
2027
+ const gateCodexPermissions =
2028
+ !useAcpTransport &&
2029
+ !useRuntimePermissionBridge &&
2030
+ shouldGateCodexPermissions({
2031
+ vendor,
2032
+ runtimePermissions: caps.runtimePermissions,
2033
+ codeArgs,
2034
+ });
2035
+ const gateClaudePermissions =
2036
+ !useAcpTransport &&
2037
+ !useRuntimePermissionBridge &&
2038
+ !gateCodexPermissions &&
2039
+ shouldGateClaudePermissions({
2040
+ vendor,
2041
+ runtimePermissions: caps.runtimePermissions,
2042
+ codeArgs,
2043
+ });
2044
+ if (useAcpTransport) {
2045
+ run = await deps.runAcpSession({
2046
+ cmd: parts[0],
2047
+ vendor,
2048
+ cwd: folderPath,
2049
+ prompt: memoryPreamble(workspaceMemory) + promptText,
2050
+ timeoutMs: cfg.runTimeoutMs,
2051
+ signal,
2052
+ env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
2053
+ onData: handleCliData,
2054
+ onEvent: handleCliEvent,
2055
+ ...openCodePermissionCallbacks({
2056
+ tool,
2057
+ channelId,
2058
+ threadRoot,
2059
+ provider: vendor,
2060
+ }),
2061
+ });
2062
+ } else if (useRuntimePermissionBridge) {
1416
2063
  run = await deps.runOpenCodeHttpSession({
1417
2064
  cmd: parts[0],
1418
2065
  args: codeArgs,
@@ -1428,6 +2075,45 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1428
2075
  threadRoot,
1429
2076
  }),
1430
2077
  });
2078
+ } else if (gateCodexPermissions) {
2079
+ run = await runCodexGatedSession({
2080
+ deps,
2081
+ cfg,
2082
+ onGateDropped: () => noticeUngatedRun(parts[0]),
2083
+ cmd: parts[0],
2084
+ codeArgs,
2085
+ cwd: folderPath,
2086
+ prompt: memoryPreamble(workspaceMemory) + promptText,
2087
+ model: resolvedModelId,
2088
+ signal,
2089
+ onData: handleCliData,
2090
+ onEvent: handleCliEvent,
2091
+ permissionCallbacks: openCodePermissionCallbacks({
2092
+ tool,
2093
+ channelId,
2094
+ threadRoot,
2095
+ provider: vendor,
2096
+ }),
2097
+ });
2098
+ } else if (gateClaudePermissions) {
2099
+ run = await runClaudeGatedCli({
2100
+ deps,
2101
+ cfg,
2102
+ onGateDropped: () => noticeUngatedRun(parts[0]),
2103
+ cmd: parts[0],
2104
+ codeArgs,
2105
+ prompt: memoryPreamble(workspaceMemory) + promptText,
2106
+ cwd: folderPath,
2107
+ signal,
2108
+ onData: handleCliData,
2109
+ resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
2110
+ permissionCallbacks: openCodePermissionCallbacks({
2111
+ tool,
2112
+ channelId,
2113
+ threadRoot,
2114
+ provider: vendor,
2115
+ }),
2116
+ });
1431
2117
  } else {
1432
2118
  run = await deps.runCli({
1433
2119
  cmd: parts[0],
@@ -1443,10 +2129,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1443
2129
  onData: handleCliData,
1444
2130
  });
1445
2131
  }
2132
+ // 0785 backstop. The notice normally goes out from `onGateDropped`,
2133
+ // BEFORE the ungated child is spawned — the room has to be warned while
2134
+ // the run can still be stopped, not told afterwards what it already did.
2135
+ // This catches a degrade that reached us without firing that hook; the
2136
+ // latch makes it a no-op in the ordinary case.
2137
+ if (run?.permissionGateDropped) await noticeUngatedRun(parts[0]);
1446
2138
  } finally {
1447
2139
  stopHeartbeat();
2140
+ stopPoller?.stop();
1448
2141
  if (emitter) {
1449
2142
  const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
2143
+ // Remembered for the accounting-only settlement (0787): an exit with no
2144
+ // report re-sends this same terminal state, so the card is never
2145
+ // repainted into something it wasn't.
2146
+ lastTerminalState = errored ? "error" : "done";
1450
2147
  let reason = "";
1451
2148
  if (errored && run && !run.aborted) {
1452
2149
  const stderrTail = oneLine((run.stderr || "").trim().split("\n").slice(-3).join(" "), 200);
@@ -1461,7 +2158,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1461
2158
  /* ignore */
1462
2159
  }
1463
2160
  try {
1464
- await progressInflight;
2161
+ await drainProgress();
1465
2162
  } catch {
1466
2163
  /* a drain failure must never break the run */
1467
2164
  }
@@ -1529,7 +2226,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1529
2226
  };
1530
2227
 
1531
2228
  // --- Run ---
1532
- let runResult = await runFolderCli(folderTaskPrompt({ message, context, brief, folderPath }));
2229
+ let runResult = await runFolderCli(folderTaskPrompt({ message, context, brief, folderPath, ...folderImagePromptArgs }));
1533
2230
  if (runResult.aborted) {
1534
2231
  await tool("post_message", {
1535
2232
  channelId,
@@ -1537,6 +2234,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1537
2234
  broadcast: Boolean(threadRoot),
1538
2235
  body: `Stopped — I left \`${folderPath}\` as it was.`,
1539
2236
  });
2237
+ await settleRunUsage(); // a stopped run still spent tokens (0787)
1540
2238
  return { status: "cancelled" };
1541
2239
  }
1542
2240
 
@@ -1559,6 +2257,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1559
2257
  body = `The run didn't finish${why}${tail}. ${left} — mention me to retry.`;
1560
2258
  }
1561
2259
  await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body });
2260
+ await settleRunUsage(); // no report on this exit — settle the spend anyway (0787)
1562
2261
  return { status: "run-failed" };
1563
2262
  }
1564
2263
  if (!runResult.failed && isGit && !hasChanges(changed)) {
@@ -1568,13 +2267,22 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1568
2267
  broadcast: Boolean(threadRoot),
1569
2268
  body: `The run finished but nothing changed in \`${folderPath}\`. Mention me to try a different approach.`,
1570
2269
  });
2270
+ await settleRunUsage(); // "nothing changed" is not "nothing spent" (0787)
1571
2271
  return { status: "no-changes" };
1572
2272
  }
1573
2273
 
1574
2274
  // --- Report card + decision loop ---
1575
2275
  const postFolderReport = async (rr, ch) => {
1576
2276
  const report = buildReport(rr, ch);
1577
- const res = await tool("post_report", { channelId, parentId: threadRoot, broadcast: true, ...report });
2277
+ const res = await tool("post_report", {
2278
+ channelId,
2279
+ parentId: threadRoot,
2280
+ broadcast: true,
2281
+ ...report,
2282
+ // 0787 — a folder run has no runs row, so the receipt rides the card and
2283
+ // the ledger row lands unattached to a run. Still the honest number.
2284
+ ...reportUsageArgs(runUsage, { vendor: cliVendor, modelId: resolvedModelId }),
2285
+ });
1578
2286
  return res?.messageId ?? null;
1579
2287
  };
1580
2288
  let reportMessageId = await postFolderReport(runResult, changed);
@@ -1589,11 +2297,12 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1589
2297
  body: `Revising in \`${folderPath}\` with your feedback${decision.note ? `: ${decision.note}` : ""} (round ${round + 1}/${maxRounds}).`,
1590
2298
  });
1591
2299
  runResult = await runFolderCli(
1592
- folderTaskPrompt({ message, context, brief, folderPath }) +
2300
+ folderTaskPrompt({ message, context, brief, folderPath, ...folderImagePromptArgs }) +
1593
2301
  `\n\nReviewer feedback to address: ${decision.note || "(see the channel)"}`,
1594
2302
  );
1595
2303
  if (runResult.aborted) {
1596
2304
  await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body: `Stopped — I left \`${folderPath}\` as it was.` });
2305
+ await settleRunUsage();
1597
2306
  return { status: "cancelled" };
1598
2307
  }
1599
2308
  changed = computeChanged();
@@ -1668,6 +2377,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1668
2377
 
1669
2378
  if (decision.kind === "cancelled") {
1670
2379
  await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body: `Stopped — the changes so far are still in \`${folderPath}\`.` });
2380
+ await settleRunUsage();
1671
2381
  return { status: "cancelled" };
1672
2382
  }
1673
2383
 
@@ -1691,6 +2401,10 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
1691
2401
  const deps = depsOverride ? { ...defaultDeps(), ...depsOverride } : defaultDeps();
1692
2402
  const git = deps.git;
1693
2403
  const signal = opts.signal;
2404
+ // 0782 — the poll loop's hand-off for "a person pressed Stop on the card": it
2405
+ // posts the notice naming them and aborts this job's signal, which is what
2406
+ // tears the coding CLI's process group down (runCli's abort path).
2407
+ const onStopRequested = opts.onStopRequested;
1694
2408
  // When the mention was a thread reply, keep the whole exchange in that thread.
1695
2409
  const parentId = message.parentId ?? null;
1696
2410
 
@@ -1738,6 +2452,43 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
1738
2452
  }
1739
2453
  }
1740
2454
 
2455
+ // MERGE / CLOSE RELAY (0704): "merge it" in a PR thread used to get an answer
2456
+ // describing the button a human should press. Now it is executed — through the
2457
+ // server, which is the authority: merge_pr / close_pr take the id of the human
2458
+ // message that asked, re-classify that text with a forced-tool model call, and
2459
+ // check the author's workspace role before spending the App's write token. The
2460
+ // daemon carries no new trust; it only stops swallowing the request.
2461
+ //
2462
+ // Runs BEFORE the chat/code split for the same reason the server runs it before
2463
+ // the coding gates: a merge instruction must never be read as a request to
2464
+ // start a new run. Every "no" here falls through to today's behavior.
2465
+ {
2466
+ const relay = prActionRelayDecision({
2467
+ body: message.body,
2468
+ threadPr: message.threadPr,
2469
+ mode,
2470
+ // A manager-routed dispatch is a brief to implement, not an instruction to
2471
+ // land something — its body is assembled text, never a person's sentence.
2472
+ canRelay: Boolean(caps.prActions) && !message.dispatch,
2473
+ });
2474
+ if (relay.relay) {
2475
+ const verb = relay.action;
2476
+ console.log(` ${verb} → relaying PR #${relay.anchor.prNumber} to the server (it decides)`);
2477
+ const result = await tool(verb === "merge" ? "merge_pr" : "close_pr", {
2478
+ channelId,
2479
+ // The mention itself is the instruction on record. The server verifies
2480
+ // it is a human's message, in this channel, from an owner or admin, and
2481
+ // that it really asks for THIS action — citing anything else fails.
2482
+ instructionMessageId: message.id,
2483
+ }).catch((e) => ({ ok: false, error: `I couldn't reach hilos to ${verb} that pull request: ${e?.message || e}` }));
2484
+ const reply = prActionReply(result);
2485
+ if (reply.post) {
2486
+ await tool("post_message", { channelId, parentId, body: reply.post }).catch(() => {});
2487
+ }
2488
+ return { status: reply.ok ? `pr-${verb}d` : "pr-action-refused", action: verb };
2489
+ }
2490
+ }
2491
+
1741
2492
  // No repo linked → either FOLDER mode (0322: a channel mapped to a plain local
1742
2493
  // folder can still get coding work done) or, failing that, just reply. A linked
1743
2494
  // repo always wins — the folder map is only consulted when there's no repo link.
@@ -1868,6 +2619,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
1868
2619
  brief: routed.task,
1869
2620
  workspaceMemory,
1870
2621
  context,
2622
+ onStopRequested,
1871
2623
  });
1872
2624
  }
1873
2625
  // Not a coding task → post the router's reply if it produced one, else fall
@@ -2034,22 +2786,65 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2034
2786
  git(repoPath, ["fetch", "origin", cfg.defaultBranch]);
2035
2787
 
2036
2788
  // Same-branch iteration: continue an existing PR on its own branch and push
2037
- // onto it, so the PR updates in place instead of a duplicate opening. Two ways
2789
+ // onto it, so the PR updates in place instead of a duplicate opening. Three ways
2038
2790
  // in, and an explicit token WINS when both apply (they agree — same PR):
2039
2791
  // 1. detectPrContinuation — an explicit "PR <url>" in the text (the server's
2040
2792
  // "request changes" rework ping, or a person naming a PR). The fallback.
2041
2793
  // 2. the run resolver's 'iterate' (0281) — a plain follow-up reply in a thread
2042
2794
  // whose run already owns an OPEN PR. The PRIMARY path.
2795
+ // 3. the THREAD ANCHOR (0704) — no active run owns this thread, but the server
2796
+ // says (on list_mentions, as `threadPr`) that the thread is about a pull
2797
+ // request: a second agent asked here, or a reply long after the first run
2798
+ // settled, still amends THAT PR. Same shape as the server's own anchor
2799
+ // fallback, and it never overrides 1 or 2.
2043
2800
  // Otherwise, a fresh branch off the default (today's behavior).
2044
2801
  const cont = detectPrContinuation(message.body, repoFullName);
2045
- const iterateUrl = cont?.url || (followupMode === "iterate" && activeRun?.prUrl) || null;
2802
+ const runIterateUrl = (followupMode === "iterate" && activeRun?.prUrl) || null;
2803
+ // The anchor is a HYPOTHESIS (its recorded state may be stale or unknown);
2804
+ // `gh` is the proof. No proof, no adoption — the run cuts a fresh branch and
2805
+ // says so on the status card rather than pushing onto a merged PR.
2806
+ let anchorAdopt = null;
2807
+ let anchorNote = "";
2808
+ if (!cont?.url && !runIterateUrl) {
2809
+ const decision = anchorIterateDecision({
2810
+ threadPr: message.threadPr,
2811
+ repoFullName,
2812
+ signal: followupSignal,
2813
+ });
2814
+ if (decision.adopt && deps.prState) {
2815
+ const live = deps.prState(repoPath, decision.anchor.prUrl);
2816
+ const confirmed = confirmAnchorOpen({ live, anchor: decision.anchor });
2817
+ if (confirmed.ok) {
2818
+ anchorAdopt = {
2819
+ prUrl: decision.anchor.prUrl,
2820
+ prNumber: decision.anchor.prNumber,
2821
+ branch: confirmed.branch,
2822
+ };
2823
+ console.log(
2824
+ ` code → thread anchor: PR #${decision.anchor.prNumber} is open on \`${confirmed.branch}\` — adopting it`,
2825
+ );
2826
+ } else {
2827
+ anchorNote = staleAnchorNote({ reason: confirmed.reason, anchor: decision.anchor });
2828
+ console.log(
2829
+ ` ! thread anchor PR #${decision.anchor.prNumber} not adopted (${confirmed.reason}) — new branch`,
2830
+ );
2831
+ }
2832
+ } else if (decision.adopt) {
2833
+ anchorNote = staleAnchorNote({ reason: "unverified", anchor: decision.anchor });
2834
+ } else if (decision.reason === "anchor-merged" || decision.reason === "anchor-closed") {
2835
+ anchorNote = staleAnchorNote({ reason: decision.reason, anchor: decision.anchor });
2836
+ }
2837
+ }
2838
+ const iterateUrl = cont?.url || runIterateUrl || anchorAdopt?.prUrl || null;
2046
2839
  let branch = null;
2047
2840
  let continuingPrUrl = null;
2048
- if (iterateUrl && deps.prHeadRef) {
2841
+ if (iterateUrl && (deps.prHeadRef || anchorAdopt)) {
2049
2842
  // Resolve the PR's head branch (source of truth); for a run-based iterate the
2050
- // run's recorded branch is a fallback if gh can't view the PR.
2843
+ // run's recorded branch is a fallback if gh can't view the PR. An adopted
2844
+ // anchor already has its head branch straight from the live PR.
2051
2845
  const headRef =
2052
- deps.prHeadRef(repoPath, iterateUrl) ||
2846
+ (anchorAdopt?.prUrl === iterateUrl ? anchorAdopt.branch : null) ||
2847
+ deps.prHeadRef?.(repoPath, iterateUrl) ||
2053
2848
  (followupMode === "iterate" ? activeRun?.branch || null : null);
2054
2849
  if (headRef) {
2055
2850
  // Fetch the PR branch, then point a local branch at its tip via FETCH_HEAD
@@ -2067,6 +2862,12 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2067
2862
  console.log(
2068
2863
  ` ! couldn't check out PR branch \`${headRef}\` (falling back to a new branch): ${(sw.stderr || "").trim().slice(0, 200)}`,
2069
2864
  );
2865
+ // An adopted anchor that can't be checked out is the same broken promise
2866
+ // as a stale one — the thread asked about a PR and gets a new branch, so
2867
+ // the status card has to say it.
2868
+ if (anchorAdopt?.prUrl === iterateUrl) {
2869
+ anchorNote = staleAnchorNote({ reason: "checkout-failed", anchor: anchorAdopt });
2870
+ }
2070
2871
  }
2071
2872
  }
2072
2873
  }
@@ -2118,6 +2919,12 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2118
2919
  } else if (effectiveMode === "redirect") {
2119
2920
  const n = activeRun?.prNumber || prNumberFromUrl(activeRun?.prUrl);
2120
2921
  ackBody = `Starting a new PR${n ? ` (separate from #${n})` : ""} on \`${branch}\`.`;
2922
+ } else if (anchorNote) {
2923
+ // The thread is about a pull request this run could NOT adopt (0704). Say it
2924
+ // on the status card itself — silence here would read as "pushing to my PR"
2925
+ // while a fresh branch is being cut. Still one card, per the run-output
2926
+ // contract: no extra chat message.
2927
+ ackBody = `${anchorNote} ${ackBody}`;
2121
2928
  }
2122
2929
  const ack = await tool("post_message", { channelId, parentId, body: ackBody });
2123
2930
  const ackId = ack?.messageId ?? null;
@@ -2171,6 +2978,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2171
2978
  threadRootId: threadRoot,
2172
2979
  taskText: message.body,
2173
2980
  branch,
2981
+ // 0782 — the mention we picked up. The server reads its human author and
2982
+ // records them as the run's requester, which is who (besides workspace
2983
+ // admins) may stop this run from its card. The thread root is NOT that
2984
+ // person: it can be an ack this agent wrote, or someone else's thread.
2985
+ requestedByMessageId: message.id,
2174
2986
  // Map an unrecognized command to null rather than the off-vocabulary
2175
2987
  // "unknown" — provider is documented as
2176
2988
  // claude_code|codex|cursor|opencode|antigravity|hermes|hilos.
@@ -2186,7 +2998,14 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2186
2998
  }
2187
2999
  // Keep the inspectable continuation/redirect statement (don't overwrite it with
2188
3000
  // an LLM plan) so a human can correct the routing before code lands.
2189
- if (ackId && caps.editMessage && chatCmdFor(cfg) && !continuingPrUrl && effectiveMode !== "redirect") {
3001
+ if (
3002
+ ackId &&
3003
+ caps.editMessage &&
3004
+ chatCmdFor(cfg) &&
3005
+ !continuingPrUrl &&
3006
+ !anchorNote &&
3007
+ effectiveMode !== "redirect"
3008
+ ) {
2190
3009
  // Feed the ack the router's distilled brief AND the conversation — not the raw
2191
3010
  // mention — so it states a real plan instead of "what's the task?".
2192
3011
  const plan = await proposePlanAck({
@@ -2210,7 +3029,35 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2210
3029
  const streamOn = Boolean(caps.postProgress);
2211
3030
  const vendor = detectVendor(cfg.codingCmd);
2212
3031
  const streamArgs = codeStreamArgs(vendor);
3032
+ // 0792 — the run's own record, when the operator asked for one. Three gates,
3033
+ // all of which have to say yes: the operator's config, the server offering
3034
+ // the tool, and a durable run row to hang the object off. The tap lives at
3035
+ // TASK scope, not per-CLI-invocation, so a resume retry and the gate:true
3036
+ // revision rounds all land in one transcript — the run is the unit, not the
3037
+ // spawn.
3038
+ const transcriptTap =
3039
+ cfg.uploadTranscripts && caps.uploadTranscript && runId ? createTranscriptTap() : null;
3040
+ // 0779 — screenshots the poll loop pulled to a temp dir at pickup. The prompt
3041
+ // names them; the argv carries them for a vendor with a verified image flag.
3042
+ const localImages = Array.isArray(message?.images) ? message.images : [];
3043
+ const codeImagePromptArgs = localImages.length
3044
+ ? {
3045
+ images: localImages,
3046
+ imagesReadable: imagesNeedReading(vendor, { gated: codexRunIsGated(cfg, caps) }),
3047
+ }
3048
+ : {};
3049
+ // 0785 — one line per run, whichever CLI turns out to be ungateable.
3050
+ const noticeUngatedRun = createUngatedRunNotice({
3051
+ // Read at post time: the thread root is the ack this run posts.
3052
+ post: (body) => tool("post_message", { channelId, parentId: threadRoot, body }),
3053
+ log: console,
3054
+ });
2213
3055
  let runSessionId = null; // captured from the stream for 0282 (resume)
3056
+ // 0787 — what the run cost, folded across every runAndStage call (a gated
3057
+ // iterate runs the CLI more than once) and taken at each report.
3058
+ const runUsage = createUsageFold();
3059
+ let resolvedModelId = pinnedModelId(cfg.codingCmd);
3060
+ let lastTerminalState = "done";
2214
3061
  const machine = hostname();
2215
3062
 
2216
3063
  // Session resume (0282): on an ITERATE we can resume the coding agent's SESSION so
@@ -2221,8 +3068,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2221
3068
  // (get_active_run) with no local match means the session likely lives on another
2222
3069
  // machine/instance — a bad `--resume` id makes claude error → empty diff → a failed
2223
3070
  // run, so we DON'T resume and degrade to today's branch+feedback (never worse).
2224
- // claude_code + cursor + opencode have proven resume flags (0282/0573/0608);
2225
- // codex/unknown → []. The gate itself is pure (resumeDecision in resume.mjs):
3071
+ // claude_code + cursor + opencode + codex have proven resume flags
3072
+ // (0282/0573/0608/0783); unknown → []. Note codex's is a SUBCOMMAND (`exec
3073
+ // resume <id>`), not a flag — it still splices in here, because flags placed
3074
+ // before it are parsed as `exec`'s own (verified live on codex-cli 0.144.1).
3075
+ // The gate itself is pure (resumeDecision in resume.mjs):
2226
3076
  // vendor, machine, project and server agreement all have to line up, and any
2227
3077
  // "no" degrades to the branch+feedback iterate rather than risking a bad id.
2228
3078
  const projectKey = codeProjectKey(vendor, repoPath, cfg.codingCmd);
@@ -2283,6 +3133,25 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2283
3133
  clearInterval(beat);
2284
3134
  };
2285
3135
  };
3136
+ /**
3137
+ * The authoritative check before anything leaves this machine (0782).
3138
+ *
3139
+ * The in-run poll stops when the CLI does, and the git work that follows —
3140
+ * commit, push, `gh pr create` — can take a minute with nothing streaming. A
3141
+ * stop that lands in THAT window would otherwise ship work nobody is waiting
3142
+ * on. So we ask the server one more time, in the shape that changes nothing:
3143
+ * `state` echoes the card's current lifecycle, and a stopped run's card is
3144
+ * frozen server-side regardless. A stop here fires the same hand-off the
3145
+ * heartbeat does (the room hears who stopped it; the job's signal aborts).
3146
+ */
3147
+ const stoppedBeforeShip = async (state = "done") => {
3148
+ // Only the streaming card carries progress metadata. The legacy heartbeat's
3149
+ // message is plain text, and stamping a lifecycle onto it would turn a
3150
+ // status line into a run card — so that path simply has no pre-ship check.
3151
+ if (!streamOn || !progressId) return false;
3152
+ const sender = createProgressSender({ tool, statusId: progressId, runId, onStopRequested });
3153
+ return await sender.poll(state);
3154
+ };
2286
3155
  // Once the run ends, retire the "still working…" progress reply so it doesn't
2287
3156
  // sit there claiming the agent is alive. No-op when no beat ever fired (short
2288
3157
  // run) or without edit_message. Best-effort.
@@ -2292,6 +3161,25 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2292
3161
  }
2293
3162
  };
2294
3163
 
3164
+ /**
3165
+ * Accounting-only settlement (0787) — the repo lane's copy of the folder
3166
+ * lane's. A run that ends with no report (no changes, a CLI that never
3167
+ * started, a person pressing Stop, a self-committing agent under the gate)
3168
+ * still spent tokens, and `post_progress` is the terminal, run-scoped call
3169
+ * the daemon already makes on all of them. Only ever the DELTA, so calling it
3170
+ * on a path that later reports anyway is harmless: the fold is already empty.
3171
+ */
3172
+ const settleRunUsage = async () => {
3173
+ const args = reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId });
3174
+ if (!args.usage || !progressId) return;
3175
+ await tool("post_progress", {
3176
+ messageId: progressId,
3177
+ ...(runId ? { runId } : {}),
3178
+ progress: { state: lastTerminalState },
3179
+ usage: args.usage,
3180
+ }).catch(() => {});
3181
+ };
3182
+
2295
3183
  const parts = cfg.codingCmd.split(" ").filter(Boolean);
2296
3184
  // Run the CLI and stage everything it changed; return the diff stats (no post).
2297
3185
  // Workspace memory (the project's soul) is prepended so the coding agent has
@@ -2311,7 +3199,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2311
3199
  // when post_report settles the report, it would re-write metadata WITHOUT the
2312
3200
  // report (it read the pre-settle snapshot) and clobber it. Draining every send
2313
3201
  // here guarantees no progress write is in flight once we settle.
2314
- let progressInflight = Promise.resolve();
3202
+ let drainProgress = async () => {};
3203
+ // 0782 — the silent-work stop poll, torn down in the same finally as the run.
3204
+ let stopPoller = null;
2315
3205
  if (streamOn) {
2316
3206
  if (!progressId) {
2317
3207
  try {
@@ -2326,27 +3216,35 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2326
3216
  }
2327
3217
  }
2328
3218
  const statusId = progressId;
2329
- if (statusId) {
2330
- emitter = createProgressEmitter({
2331
- parser: makeStreamParser(vendor),
2332
- now: deps.now,
2333
- throttleMs: cfg.progressMs,
2334
- // The module never imports MCP — the send fn is injected here. It must
2335
- // never throw into the run (createProgressEmitter also guards).
2336
- send: (p) => {
2337
- try {
2338
- const r = tool("post_progress", { messageId: statusId, progress: p });
2339
- if (r && typeof r.then === "function") {
2340
- const done = r.then(() => {}, () => {});
2341
- progressInflight = Promise.all([progressInflight, done]).then(
2342
- () => {},
2343
- () => {},
2344
- );
2345
- }
2346
- } catch {
2347
- /* a progress send must never break the run */
2348
- }
2349
- },
3219
+ // The module never imports MCP — the send fn is injected here. It must
3220
+ // never throw into the run (createProgressEmitter also guards). 0782:
3221
+ // the sender also carries the stop signal home on this same beat, bound
3222
+ // to THIS run's id so a stop can't be read off a neighbouring thread.
3223
+ const sender = statusId
3224
+ ? createProgressSender({ tool, statusId, runId, onStopRequested })
3225
+ : null;
3226
+ // The emitter is built whether or not the status card posted (0787). The
3227
+ // stream flags are already on the argv either way, so the CLI is speaking
3228
+ // NDJSON regardless; gating the parser on a card meant one transient
3229
+ // post_message failure silently cost the run its session id (the resume
3230
+ // record's only source on an argv run) and its usage. A card is where
3231
+ // progress is SHOWN, never how the stream is read — this is what the
3232
+ // folder path has always done. With no card the sends are dropped on the
3233
+ // floor rather than skipped, so nothing else in the run changes shape.
3234
+ emitter = createProgressEmitter({
3235
+ parser: makeStreamParser(vendor),
3236
+ now: deps.now,
3237
+ throttleMs: cfg.progressMs,
3238
+ send: sender ? sender.send : () => {},
3239
+ });
3240
+ if (sender) {
3241
+ drainProgress = () => sender.drain();
3242
+ // The heartbeat only fires when the CLI speaks. This asks anyway, so a
3243
+ // run inside one long silent command (a full test suite, a slow install)
3244
+ // is still stoppable from the room.
3245
+ stopPoller = createStopPoller({
3246
+ poll: () => sender.poll("working"),
3247
+ intervalMs: cfg.stopPollMs || STOP_POLL_MS,
2350
3248
  });
2351
3249
  }
2352
3250
  } else {
@@ -2360,10 +3258,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2360
3258
  // fallback) suppresses it. buildResumeArgs is [] unless vendor+session make
2361
3259
  // resume safe, so a non-resume run is byte-identical to the pre-0282 ARGV.
2362
3260
  const resumeArgs = resume ? buildResumeArgs(vendor, resumeSessionId) : [];
2363
- // Model preset (0504): resolved at run time against the CLI's own model
2364
- // list (cursor only today) — [] when unset/unresolvable, so the tool's
2365
- // default stands. Inserted before resume/stream flags, after the base.
3261
+ // Model preset (0504, codex in 0783): resolved at run time against the
3262
+ // CLI's own model list — [] when unset/unresolvable, so the tool's default
3263
+ // stands. Inserted before resume/stream flags, after the base — an order
3264
+ // codex depends on, since its resume is a subcommand and the model flag
3265
+ // has to reach `exec`, i.e. sit BEFORE `resume`.
2366
3266
  const modelArgs = await modelArgsFor(cfg, vendor);
3267
+ // 0787: remember the id we actually handed the CLI — the usage receipt for
3268
+ // a vendor whose stream never names a model (codex) has nothing else to say.
3269
+ if (modelArgs[0] === "--model" && modelArgs[1]) resolvedModelId = modelArgs[1];
2367
3270
  // Project pin (0608, opencode only): the CLI resolves its project from PWD,
2368
3271
  // not the spawn cwd, and its sessions are per project — `--dir` makes both
2369
3272
  // deterministic. [] for every other vendor (and for an `--attach`ed run,
@@ -2371,13 +3274,29 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2371
3274
  // The spawned PWD now matches the cwd too (0615); `--dir` stays as the
2372
3275
  // CLI's own explicit contract, and to keep attach runs off our local path.
2373
3276
  const dirArgs = codeDirArgs(vendor, repoPath, cfg.codingCmd);
3277
+ // 0779: BEFORE resumeArgs, for the same reason modelArgs are — codex's
3278
+ // resume is a SUBCOMMAND (`exec resume <id>`) and `--image=` belongs to
3279
+ // `exec`, so it has to sit ahead of it. [] for every vendor without a
3280
+ // verified flag, leaving their argv byte-identical to before.
3281
+ const imageArgs = codeImageArgs(vendor, localImages);
2374
3282
  const codeArgs = streamOn
2375
- ? [...parts.slice(1), ...modelArgs, ...dirArgs, ...resumeArgs, ...streamArgs]
2376
- : [...parts.slice(1), ...modelArgs, ...dirArgs, ...resumeArgs];
3283
+ ? [...parts.slice(1), ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs, ...streamArgs]
3284
+ : [...parts.slice(1), ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs];
2377
3285
  const handleCliData = (c) => {
2378
3286
  // Keep tracking lastLine as a fallback (legacy heartbeat / honesty).
2379
3287
  const lines = String(c).split("\n").map((s) => s.trim()).filter(Boolean);
2380
3288
  if (lines.length) lastLine = lines[lines.length - 1];
3289
+ // 0792 — the raw tail, before the parser reduces it to eight steps. Every
3290
+ // transport routes its child's stdout through this one callback, so the
3291
+ // tap sees an argv run, a permission-bridged run, and an ACP session
3292
+ // alike; a transport that writes nothing to stdout simply leaves it empty.
3293
+ if (transcriptTap) {
3294
+ try {
3295
+ transcriptTap.push(c);
3296
+ } catch {
3297
+ /* keeping a record must never break the run */
3298
+ }
3299
+ }
2381
3300
  if (emitter) {
2382
3301
  try {
2383
3302
  emitter.feed(c);
@@ -2386,19 +3305,89 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2386
3305
  }
2387
3306
  }
2388
3307
  };
3308
+ // The structured transports (ACP, `codex mcp-server`) put only the
3309
+ // assistant's PROSE on stdout — every tool call travels here instead. The
3310
+ // live card has always read this; the transcript has to as well, or a gated
3311
+ // run uploads a page of narration with no execution anywhere in it.
3312
+ const handleCliEvent = (ev) => {
3313
+ if (transcriptTap) {
3314
+ try {
3315
+ transcriptTap.pushEvent(ev);
3316
+ } catch {
3317
+ /* keeping a record must never break the run */
3318
+ }
3319
+ }
3320
+ try {
3321
+ emitter?.foldEvent(ev);
3322
+ } catch {
3323
+ /* a progress fold must never break the run */
3324
+ }
3325
+ };
2389
3326
  let run;
2390
3327
  try {
2391
3328
  // OpenCode's own non-interactive CLI auto-rejects every permission ask
2392
3329
  // unless --auto is present. For the gated tiers, bypass that responder
2393
3330
  // and own the authenticated HTTP session + SSE stream directly so hilos
2394
3331
  // is the sole authority answering the paused tool call (0593).
2395
- const useRuntimePermissionBridge = shouldUseRuntimePermissionBridge({
3332
+ const useAcpTransport = shouldUseAcpTransport({
2396
3333
  vendor,
3334
+ acpTransport: cfg.acpTransport,
2397
3335
  runtimePermissions: caps.runtimePermissions,
2398
3336
  codeArgs,
2399
3337
  codingCmd: cfg.codingCmd,
2400
3338
  });
2401
- if (useRuntimePermissionBridge) {
3339
+ const useRuntimePermissionBridge =
3340
+ !useAcpTransport &&
3341
+ shouldUseRuntimePermissionBridge({
3342
+ vendor,
3343
+ runtimePermissions: caps.runtimePermissions,
3344
+ codeArgs,
3345
+ codingCmd: cfg.codingCmd,
3346
+ });
3347
+ // 0777, and the reason it composes with resume (0778): claude keeps its
3348
+ // ordinary argv run (resumeArgs included), and codex's gated transport
3349
+ // has its OWN resume — `codex-reply {threadId}` — so a gated iterate
3350
+ // continues the same thread instead of trading continuity for a gate.
3351
+ const gateCodexPermissions =
3352
+ !useAcpTransport &&
3353
+ !useRuntimePermissionBridge &&
3354
+ shouldGateCodexPermissions({
3355
+ vendor,
3356
+ runtimePermissions: caps.runtimePermissions,
3357
+ codeArgs,
3358
+ });
3359
+ const gateClaudePermissions =
3360
+ !useAcpTransport &&
3361
+ !useRuntimePermissionBridge &&
3362
+ !gateCodexPermissions &&
3363
+ shouldGateClaudePermissions({
3364
+ vendor,
3365
+ runtimePermissions: caps.runtimePermissions,
3366
+ codeArgs,
3367
+ });
3368
+ if (useAcpTransport) {
3369
+ run = await deps.runAcpSession({
3370
+ cmd: parts[0],
3371
+ vendor,
3372
+ cwd: repoPath,
3373
+ prompt: memoryPreamble(workspaceMemory) + promptText,
3374
+ // 0778: approvals AND continuity. `resume:false` (the never-worse
3375
+ // retry) drops it exactly like buildResumeArgs does.
3376
+ resumeSessionId: resume ? resumeSessionId : null,
3377
+ timeoutMs: cfg.runTimeoutMs,
3378
+ signal,
3379
+ env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
3380
+ onData: handleCliData,
3381
+ onEvent: handleCliEvent,
3382
+ ...openCodePermissionCallbacks({
3383
+ tool,
3384
+ channelId,
3385
+ threadRoot,
3386
+ runId,
3387
+ provider: vendor,
3388
+ }),
3389
+ });
3390
+ } else if (useRuntimePermissionBridge) {
2402
3391
  run = await deps.runOpenCodeHttpSession({
2403
3392
  cmd: parts[0],
2404
3393
  args: codeArgs,
@@ -2418,8 +3407,59 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2418
3407
  runId,
2419
3408
  }),
2420
3409
  });
3410
+ } else if (gateCodexPermissions) {
3411
+ run = await runCodexGatedSession({
3412
+ deps,
3413
+ cfg,
3414
+ onGateDropped: () => noticeUngatedRun(parts[0]),
3415
+ cmd: parts[0],
3416
+ codeArgs,
3417
+ cwd: repoPath,
3418
+ prompt: memoryPreamble(workspaceMemory) + promptText,
3419
+ model: resolvedModelId,
3420
+ // Codex's own resume over this transport. `resume:false` (the
3421
+ // never-worse retry) drops it exactly like buildResumeArgs does.
3422
+ resumeThreadId: resume ? resumeSessionId : null,
3423
+ signal,
3424
+ onData: handleCliData,
3425
+ onEvent: handleCliEvent,
3426
+ permissionCallbacks: openCodePermissionCallbacks({
3427
+ tool,
3428
+ channelId,
3429
+ threadRoot,
3430
+ runId,
3431
+ provider: vendor,
3432
+ }),
3433
+ });
3434
+ } else if (gateClaudePermissions) {
3435
+ run = await runClaudeGatedCli({
3436
+ deps,
3437
+ cfg,
3438
+ onGateDropped: () => noticeUngatedRun(parts[0]),
3439
+ cmd: parts[0],
3440
+ // codeArgs already carries this run's resume flags, so a gated
3441
+ // iterate resumes AND raises cards — the two never traded off.
3442
+ codeArgs,
3443
+ prompt: memoryPreamble(workspaceMemory) + promptText,
3444
+ cwd: repoPath,
3445
+ signal,
3446
+ onData: handleCliData,
3447
+ sessionId: resume ? resumeSessionId ?? "" : "",
3448
+ resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
3449
+ permissionCallbacks: openCodePermissionCallbacks({
3450
+ tool,
3451
+ channelId,
3452
+ threadRoot,
3453
+ runId,
3454
+ provider: vendor,
3455
+ }),
3456
+ });
2421
3457
  } else {
2422
- run = await runCli({
3458
+ // deps.runCli, not the bare import: `defaultDeps` wraps the very same
3459
+ // function, so production is byte-identical, but the ungated repo run
3460
+ // was the ONE coding path that escaped the injectable runner — which is
3461
+ // why nothing above the unit tests could ever drive it (0787).
3462
+ run = await deps.runCli({
2423
3463
  cmd: parts[0],
2424
3464
  args: [...codeArgs, memoryPreamble(workspaceMemory) + promptText],
2425
3465
  cwd: repoPath,
@@ -2430,8 +3470,12 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2430
3470
  onData: handleCliData,
2431
3471
  });
2432
3472
  }
3473
+ // 0785 — the gate was expected here and the CLI couldn't hold it. Say so
3474
+ // in the room, once per run, rather than only in the daemon's console.
3475
+ if (run?.permissionGateDropped) await noticeUngatedRun(parts[0]);
2433
3476
  } finally {
2434
3477
  stopHeartbeat();
3478
+ stopPoller?.stop();
2435
3479
  if (run?.sessionId) runSessionId = run.sessionId;
2436
3480
  if (emitter) {
2437
3481
  // Terminal state: flip the status card off "working" (state 'done'/'error')
@@ -2440,6 +3484,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2440
3484
  // (0289) — the card cross-fades run → report; for no-changes/failed/gate
2441
3485
  // outcomes this terminal 'done'/'error' is the card's final state.
2442
3486
  const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
3487
+ // Remembered so an exit with no report re-sends this same state (0787)
3488
+ // rather than repainting the card into something it wasn't.
3489
+ lastTerminalState = errored ? "error" : "done";
2443
3490
  // On error, carry an honest reason onto the card (0294) — the same signal
2444
3491
  // the chat note uses: spawn error message, else exit status, plus a short
2445
3492
  // stderr tail. Sanitized again server-side; an old server just drops it.
@@ -2464,7 +3511,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2464
3511
  // BEFORE runAndStage returns, so nothing is in flight when the caller
2465
3512
  // settles the report onto this card — no lost-update clobber (0289).
2466
3513
  try {
2467
- await progressInflight;
3514
+ await drainProgress();
2468
3515
  } catch {
2469
3516
  /* a drain failure must never break the run */
2470
3517
  }
@@ -2474,6 +3521,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2474
3521
  } catch {
2475
3522
  /* ignore */
2476
3523
  }
3524
+ // 0787 — read the run's cost AFTER done() (the terminal usage frame
3525
+ // usually arrives in the parser flush). Folded, not sent yet: the
3526
+ // report call is what carries it home.
3527
+ try {
3528
+ const spend = emitter.usage?.();
3529
+ if (spend) runUsage.push({ t: "usage", ...spend });
3530
+ } catch {
3531
+ /* accounting must never break the run */
3532
+ }
2477
3533
  }
2478
3534
  }
2479
3535
  if (run.aborted || signal?.aborted) {
@@ -2541,8 +3597,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2541
3597
  sessionId: sid,
2542
3598
  machine,
2543
3599
  // The CLI that created the session (a `ses_…` is meaningless to
2544
- // claude's `--resume`) and the project it belongs to — opencode
2545
- // scopes sessions per project. The gate above requires both (0608).
3600
+ // claude's `--resume`) and the project it belongs to — opencode and
3601
+ // codex both scope sessions per project. The gate above requires
3602
+ // both (0608, 0783).
2546
3603
  vendor,
2547
3604
  cwd: projectKey,
2548
3605
  updatedAt: new Date().toISOString(),
@@ -2553,6 +3610,32 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2553
3610
  }
2554
3611
  };
2555
3612
 
3613
+ // 0792 — ship the run's record, AFTER its report. Ordering is the whole
3614
+ // point: the transcript is a fact about finished work, and the card it marks
3615
+ // has to exist before it can be marked. Best-effort in every direction — no
3616
+ // tap, no stream, no server tool, or a refused upload all leave the run
3617
+ // exactly as it would have ended without this feature.
3618
+ let transcriptUploaded = false;
3619
+ const uploadTranscript = async (reportMsgId) => {
3620
+ if (!transcriptTap || !runId || transcriptUploaded) return;
3621
+ let text = "";
3622
+ try {
3623
+ text = transcriptTap.text();
3624
+ } catch {
3625
+ return;
3626
+ }
3627
+ // Nothing to say: a transport that never wrote to stdout (an ACP session, a
3628
+ // vendor with no stream) has no transcript, and an empty upload is worse
3629
+ // than none — it would put an empty object behind a Transcript control.
3630
+ if (!text) return;
3631
+ transcriptUploaded = true;
3632
+ await tool("upload_run_transcript", {
3633
+ runId,
3634
+ transcript: text,
3635
+ ...(reportMsgId ? { messageId: reportMsgId } : {}),
3636
+ }).catch(() => {});
3637
+ };
3638
+
2556
3639
  // Post a proposal card (approve-before-push mode) from a staged change.
2557
3640
  const postProposal = async (staged) => {
2558
3641
  const report = buildProposalReport({
@@ -2566,7 +3649,16 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2566
3649
  stat: staged.stat,
2567
3650
  runFailed: staged.runFailed,
2568
3651
  });
2569
- const res = await tool("post_report", { channelId, parentId: threadRoot, broadcast: true, ...report });
3652
+ const res = await tool("post_report", {
3653
+ channelId,
3654
+ parentId: threadRoot,
3655
+ broadcast: true,
3656
+ ...report,
3657
+ // 0787 — the proposal card is where a gated run's spend lands, because
3658
+ // it is the card the run produced. The later "Shipped" report carries
3659
+ // only what the CLI spent AFTER this one (usually nothing).
3660
+ ...reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
3661
+ });
2570
3662
  return { reportMessageId: res?.messageId ?? null, stat: staged.stat };
2571
3663
  };
2572
3664
 
@@ -2605,6 +3697,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2605
3697
  body,
2606
3698
  });
2607
3699
  await updateChannelMarker(compactRunMarker("cancelled", branch));
3700
+ // Every repo-lane cancel funnels through here, so one settle covers them
3701
+ // all: a stopped run still spent tokens (0787).
3702
+ await settleRunUsage();
2608
3703
  return { status: "cancelled", branch };
2609
3704
  };
2610
3705
 
@@ -2633,7 +3728,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2633
3728
 
2634
3729
  const codePrompt =
2635
3730
  (teamMemoryBlock ? teamMemoryBlock + "\n\n" : "") +
2636
- codeTaskPrompt({ message, context, brief: routed.task, repoFullName });
3731
+ codeTaskPrompt({ message, context, brief: routed.task, repoFullName, ...codeImagePromptArgs });
2637
3732
  let staged = await runAndStage(codePrompt);
2638
3733
  if (staged.aborted) return await postStopped();
2639
3734
  // Never-worse-than-today (0282): if we RESUMED a session and that run FAILED (a
@@ -2672,11 +3767,13 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2672
3767
  });
2673
3768
  await finalizeProgress(`\`${branch}\` has the agent's own commits — needs review.`);
2674
3769
  await updateChannelMarker(compactRunMarker("changes", branch));
3770
+ await settleRunUsage(); // no report on this exit (0787)
2675
3771
  return { status: "self-committed-gated", branch };
2676
3772
  }
2677
3773
  // When a live status card was streaming, settle the report onto it (0289) —
2678
3774
  // the card cross-fades run → report. Skip finalizeProgress then (it edits the
2679
3775
  // card's body, which the settle overwrites with the report anyway).
3776
+ if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
2680
3777
  const selfSettleId = streamOn && progressId ? progressId : null;
2681
3778
  if (!selfSettleId) await finalizeProgress(`Agent shipped \`${branch}\` itself — report below.`);
2682
3779
  const selfResult = await shipSelfDriven({
@@ -2691,8 +3788,10 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2691
3788
  parentId: threadRoot,
2692
3789
  settleId: selfSettleId,
2693
3790
  runId,
3791
+ usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
2694
3792
  });
2695
3793
  await recordSession(selfResult.prUrl);
3794
+ await uploadTranscript(selfResult.reportMsgId);
2696
3795
  await updateChannelMarker(compactRunMarker(selfResult.status, selfResult.branch));
2697
3796
  return selfResult;
2698
3797
  }
@@ -2730,6 +3829,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2730
3829
  await updateChannelMarker(
2731
3830
  compactRunMarker(staged.failed ? "run-failed" : "no-changes", branch),
2732
3831
  );
3832
+ await settleRunUsage(); // "no changes" is not "no spend" (0787)
2733
3833
  return { status: staged.failed ? "run-failed" : "no-changes" };
2734
3834
  }
2735
3835
 
@@ -2738,7 +3838,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2738
3838
  // propose-a-diff-and-wait flow for users who want approve-before-push.
2739
3839
  if (!cfg.gate) {
2740
3840
  // A "stop" that lands between the run finishing and the ship must still win.
2741
- if (signal?.aborted) return await postStopped();
3841
+ // The signal covers a stop we already heard; the server check covers one that
3842
+ // landed while the CLI was silent and nothing was carrying it home.
3843
+ if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
2742
3844
  // Seamless single card (0289): when a live status card streamed this run,
2743
3845
  // settle the report onto it in place instead of posting a separate report
2744
3846
  // message. Only the DEFAULT gate:false report settles in place; gate:true's
@@ -2762,6 +3864,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2762
3864
  existingPrUrl: continuingPrUrl,
2763
3865
  settleId,
2764
3866
  runId,
3867
+ usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
2765
3868
  });
2766
3869
  // The settle already flipped the card to the report (body + metadata); editing
2767
3870
  // the body again would clobber the report summary, so only finalize when we
@@ -2770,6 +3873,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2770
3873
  // Re-record with the now-known PR url so the local session record + the run row
2771
3874
  // carry the PR the session belongs to (best-effort; session id unchanged).
2772
3875
  await recordSession(result.prUrl);
3876
+ await uploadTranscript(result.reportMsgId);
2773
3877
  await updateChannelMarker(compactRunMarker(result.status, result.branch));
2774
3878
  return { ...result, stat: staged.stat, sessionId: runSessionId };
2775
3879
  }
@@ -2806,7 +3910,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2806
3910
  }
2807
3911
 
2808
3912
  // A stop during the final wait, too — don't ship a cancelled run.
2809
- if (signal?.aborted) return await postStopped();
3913
+ if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
2810
3914
  const result = await applyDecision({
2811
3915
  decision,
2812
3916
  repoPath,
@@ -2822,9 +3926,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2822
3926
  parentId: threadRoot,
2823
3927
  existingPrUrl: continuingPrUrl,
2824
3928
  runId,
3929
+ // Normally {} — a gated run already reported its spend on the proposal card
3930
+ // it is now shipping. Anything the CLI spent after that still comes home.
3931
+ usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
2825
3932
  });
2826
3933
  await finalizeProgress(`Done on \`${branch}\` — see the report below.`);
2827
3934
  await recordSession(result.prUrl);
3935
+ // The gated lane's shipped report is a fresh message; `result.reportMsgId`
3936
+ // names it, and the proposal card is the fallback when it never posted.
3937
+ await uploadTranscript(result.reportMsgId || proposal.reportMessageId);
2828
3938
  await updateChannelMarker(compactRunMarker(result.status, result.branch));
2829
3939
  return { ...result, reportMessageId: proposal.reportMessageId, stat: proposal.stat, rounds: round, sessionId: runSessionId };
2830
3940
  } finally {