hilos-agent 0.9.0 → 0.9.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/handler.mjs CHANGED
@@ -37,16 +37,20 @@ import {
37
37
  scrubHilosEnv,
38
38
  envForCwd,
39
39
  } from "./cli.mjs";
40
- import { makeStreamParser } from "./agent-events.mjs";
40
+ import { makeStreamParser, createUsageFold } from "./agent-events.mjs";
41
41
  import {
42
42
  detectVendor,
43
43
  codeStreamArgs,
44
44
  codeDirArgs,
45
+ codeImageArgs,
45
46
  codeProjectKey,
47
+ imagesNeedReading,
46
48
  attachTarget,
47
49
  createProgressEmitter,
48
50
  fastChatCmd,
49
51
  } from "./progress-emitter.mjs";
52
+ import { createTranscriptTap } from "./transcript.mjs";
53
+ import { imagePromptNote, renderAttachmentLine } from "./attachments.mjs";
50
54
  import { resolveFollowupMode, classifyFollowupCue, normalizeSignal } from "./followup.mjs";
51
55
  import {
52
56
  anchorIterateDecision,
@@ -67,6 +71,17 @@ import { buildMemoryBlock } from "./memory.mjs";
67
71
  import { deployFolder, resolveDeployTarget } from "./deploy.mjs";
68
72
  import { runOpenCodeHttpSession } from "./opencode-session.mjs";
69
73
  import { runAcpSession } from "./acp-session.mjs";
74
+ import {
75
+ shouldGateClaudePermissions,
76
+ startClaudePermissionServer,
77
+ } from "./claude-permissions.mjs";
78
+ import {
79
+ codexMcpTransportUnavailable,
80
+ codexSandboxFromArgs,
81
+ runCodexMcpSession,
82
+ shouldGateCodexPermissions,
83
+ } from "./codex-mcp-session.mjs";
84
+ import { createUngatedRunNotice } from "./permission-gate.mjs";
70
85
 
71
86
  /**
72
87
  * The environment for a coding/chat CLI run. runCli always strips HILOS_* on top
@@ -147,6 +162,49 @@ function compactRunMarker(status, branch) {
147
162
  return `Failed:${b}`;
148
163
  }
149
164
 
165
+ /**
166
+ * A `--model x` / `-m x` the operator pinned on the coding command (0787).
167
+ * `modelArgsFor` deliberately returns [] in that case — the hand-pin wins — so
168
+ * this is where a pinned id is recovered for the receipt.
169
+ */
170
+ export function pinnedModelId(codingCmd) {
171
+ const m = /(?:^|\s)(?:--model|-m)(?:\s+|=)("[^"]+"|'[^']+'|\S+)/.exec(String(codingCmd || ""));
172
+ return m ? m[1].replace(/^["']|["']$/g, "") : "";
173
+ }
174
+
175
+ /**
176
+ * The `usage` argument for a report (0787), or `{}` when the CLI told us
177
+ * nothing — an omitted field is the honest answer, and the card degrades to
178
+ * "Tokens unavailable" rather than claiming a free run.
179
+ *
180
+ * TAKES from the fold. Every report writes its OWN ledger row server-side, so a
181
+ * run that reports more than once (a proposal card and then a shipped card, a
182
+ * revision round per loop) must carry only what is new since the last report;
183
+ * a running total would bill the same tokens twice.
184
+ *
185
+ * The model: whatever the stream named wins. Codex names none, so the id
186
+ * resolved at spawn — or pinned on the command — stands in, and failing both,
187
+ * the CLI's own name. The ledger has to say what ran.
188
+ */
189
+ export function reportUsageArgs(fold, { vendor, modelId, runId } = {}) {
190
+ const totals = fold && typeof fold.take === "function" ? fold.take() : null;
191
+ if (!totals) return {};
192
+ const cli = vendor && vendor !== "unknown" ? vendor : null;
193
+ const model = totals.model || modelId || cli || "";
194
+ if (!model) return {};
195
+ const usage = {
196
+ model,
197
+ inputTokens: totals.inputTokens,
198
+ outputTokens: totals.outputTokens,
199
+ cacheReadTokens: totals.cacheReadTokens,
200
+ cacheCreationTokens: totals.cacheCreationTokens,
201
+ };
202
+ if (cli) usage.vendor = cli;
203
+ if (typeof totals.costUsd === "number") usage.costUsd = totals.costUsd;
204
+ if (runId) usage.runId = runId;
205
+ return { usage };
206
+ }
207
+
150
208
  // Every helper below runs a tool in a directory we choose, so each one hands
151
209
  // the child a PWD that matches that directory instead of the daemon's launch
152
210
  // dir (0615). git and gh both use the real cwd, so this is hygiene rather than
@@ -162,6 +220,8 @@ function defaultDeps() {
162
220
  runCli: (opts) => runCli(opts),
163
221
  runOpenCodeHttpSession: (opts) => runOpenCodeHttpSession(opts),
164
222
  runAcpSession: (opts) => runAcpSession(opts),
223
+ runCodexMcpSession: (opts) => runCodexMcpSession(opts),
224
+ startClaudePermissionServer: (opts) => startClaudePermissionServer(opts),
165
225
  // Does a path exist on disk? Injectable so folder mode's "missing folder"
166
226
  // guard is unit-testable without touching the real filesystem.
167
227
  pathExists: (p) => existsSync(p),
@@ -297,6 +357,145 @@ function openCodePermissionCallbacks({
297
357
  };
298
358
  }
299
359
 
360
+ /**
361
+ * The post_progress sender both run lanes use, with the 0782 stop check folded
362
+ * into the beat they already send.
363
+ *
364
+ * A person can press Stop on the live card from anywhere — the person who
365
+ * delegated the work usually is not the one at this laptop. The server records
366
+ * that on the run's own row and answers the NEXT heartbeat with
367
+ * `stopRequested: true` (plus `stoppedBy` when it knows the name). So the daemon
368
+ * learns it on its existing cadence: no poll tool, no second timer, nothing new
369
+ * to keep alive.
370
+ *
371
+ * `onStopRequested` is called at most once per sender and must never throw into
372
+ * the run — it is the daemon's cue to tear the process group down.
373
+ *
374
+ * @param {{ tool: Function, statusId: string, runId?: string|null,
375
+ * onStopRequested?: (info: {by: string|null}) => any }} o
376
+ */
377
+ export function createProgressSender({ tool, statusId, runId = null, onStopRequested }) {
378
+ let inflight = Promise.resolve();
379
+ let stopSeen = false;
380
+ const takeStop = (res) => {
381
+ if (!res || res.stopRequested !== true) return false;
382
+ if (stopSeen || !onStopRequested) return true;
383
+ stopSeen = true;
384
+ try {
385
+ onStopRequested({ by: typeof res.stoppedBy === "string" ? res.stoppedBy : null });
386
+ } catch {
387
+ /* the stop hand-off must never break the run's teardown */
388
+ }
389
+ return true;
390
+ };
391
+ return {
392
+ /** @param {object} p — the emitter's snapshot. */
393
+ send(p) {
394
+ try {
395
+ const r = tool("post_progress", {
396
+ messageId: statusId,
397
+ ...(runId ? { runId } : {}),
398
+ progress: p,
399
+ });
400
+ if (r && typeof r.then === "function") {
401
+ const done = r.then((res) => void takeStop(res), () => {});
402
+ inflight = Promise.all([inflight, done]).then(
403
+ () => {},
404
+ () => {},
405
+ );
406
+ }
407
+ } catch {
408
+ /* a progress send must never break the run */
409
+ }
410
+ },
411
+ /**
412
+ * Ask the server outright, and WAIT for the answer.
413
+ *
414
+ * The heartbeat above only fires when the CLI says something. A run sitting
415
+ * inside one silent ten-minute command emits nothing, so nothing would carry
416
+ * a stop home — the person would press Stop and watch their laptop keep
417
+ * going. This is the same call, on a timer, and it is also the authoritative
418
+ * pre-ship check: `state` echoes the card's CURRENT lifecycle so asking
419
+ * never repaints it (a stopped run's card is frozen server-side anyway).
420
+ *
421
+ * @param {"working"|"done"|"error"} [state]
422
+ * @returns {Promise<boolean>} true when a person has stopped this run.
423
+ */
424
+ async poll(state = "working") {
425
+ try {
426
+ const res = await tool("post_progress", {
427
+ messageId: statusId,
428
+ ...(runId ? { runId } : {}),
429
+ progress: { state },
430
+ });
431
+ return takeStop(res);
432
+ } catch {
433
+ // A stop check that can't reach the server is not a stop. The run
434
+ // continues; the next tick asks again.
435
+ return false;
436
+ }
437
+ },
438
+ /** Every dispatched write, so a caller can wait them out before settling. */
439
+ drain() {
440
+ return inflight;
441
+ },
442
+ };
443
+ }
444
+
445
+ /** How often a run asks whether it has been stopped while its CLI is silent.
446
+ * Well under the card's own staleness window, far above a chatty write rate. */
447
+ export const STOP_POLL_MS = 15_000;
448
+
449
+ /**
450
+ * Arm the silent-work stop poll for ONE active job. Timers are injectable so
451
+ * this is unit-tested on a fake clock, and the returned `stop()` must be called
452
+ * from the same finally that tears the run down — one timer per job, never a
453
+ * timer that outlives the work it was watching.
454
+ *
455
+ * @param {{ poll: () => Promise<boolean>, intervalMs?: number,
456
+ * setTimer?: Function, clearTimer?: Function }} o
457
+ */
458
+ export function createStopPoller({
459
+ poll,
460
+ intervalMs = STOP_POLL_MS,
461
+ setTimer = (fn, ms) => {
462
+ const id = setInterval(fn, ms);
463
+ if (id && typeof id.unref === "function") id.unref();
464
+ return id;
465
+ },
466
+ clearTimer = (id) => clearInterval(id),
467
+ } = {}) {
468
+ let id = null;
469
+ let asking = false;
470
+ let done = false;
471
+ const tick = async () => {
472
+ // Never stack asks: a slow server must not queue a burst of stop checks.
473
+ if (asking || done) return;
474
+ asking = true;
475
+ try {
476
+ if (await poll()) {
477
+ done = true; // the stop is handed over once; teardown owns the rest
478
+ stop();
479
+ }
480
+ } catch {
481
+ /* a stop check must never break the run */
482
+ } finally {
483
+ asking = false;
484
+ }
485
+ };
486
+ function stop() {
487
+ if (id == null) return;
488
+ try {
489
+ clearTimer(id);
490
+ } catch {
491
+ /* ignore */
492
+ }
493
+ id = null;
494
+ }
495
+ id = setTimer(() => void tick(), intervalMs);
496
+ return { stop, tick };
497
+ }
498
+
300
499
  /** The local HTTP bridge can own only a local OpenCode server. An explicit
301
500
  * `--attach` remains on OpenCode's CLI responder, which rejects unanswered asks
302
501
  * fail closed; taking over a remote server requires a separate authenticated
@@ -315,22 +514,25 @@ export function shouldUseRuntimePermissionBridge({
315
514
  );
316
515
  }
317
516
 
318
- /** ACP transport opt-in (0759; slice 1 opencode, slice 2 cursor). Only runs
319
- * the workspace already gates with runtime permissions qualify, PLUS the
320
- * operator's explicit acpTransport flag, and only when the vendor's own
321
- * auto-allow escape hatches are absent — over ACP, hilos must be the one
322
- * answering asks, so a run configured to never ask has nothing to gate here.
323
- * Resumed runs stay on the existing paths: their session continuity contract
324
- * is argv/HTTP-shaped and the ACP path does not carry it yet. */
517
+ /** ACP transport (0759; slice 1 opencode, slice 2 cursor). Only runs the
518
+ * workspace already gates with runtime permissions qualify, and only when the
519
+ * vendor's own auto-allow escape hatches are absent — over ACP, hilos must be
520
+ * the one answering asks, so a run configured to never ask has nothing to gate
521
+ * here.
522
+ *
523
+ * 0778 removed two limits that were never about correctness. `acpTransport`
524
+ * now defaults ON (config.mjs) for the vendors whose adapter is proven, and a
525
+ * RESUMED run no longer disqualifies: both live vendors advertise ACP's
526
+ * `loadSession` capability, so acp-session.mjs resumes the session and keeps
527
+ * raising cards. Setting `acpTransport: false` still opts a workspace out. */
325
528
  export function shouldUseAcpTransport({
326
529
  vendor,
327
530
  acpTransport,
328
531
  runtimePermissions,
329
532
  codeArgs,
330
533
  codingCmd,
331
- resumeSessionId = /** @type {string | null} */ (null),
332
534
  }) {
333
- if (acpTransport !== true || resumeSessionId != null) return false;
535
+ if (acpTransport !== true) return false;
334
536
  if (runtimePermissions !== true) return false;
335
537
  if (vendor === "opencode") {
336
538
  return shouldUseRuntimePermissionBridge({
@@ -348,6 +550,174 @@ export function shouldUseAcpTransport({
348
550
  return false;
349
551
  }
350
552
 
553
+ /**
554
+ * The env a gated claude run gets (0777).
555
+ *
556
+ * A permission card can legitimately wait on a person for minutes — a 150s wait
557
+ * was verified live — so the CLI's own MCP tool timeout must not cut the ask
558
+ * short before hilos's run deadline does. Everything else about the env is
559
+ * unchanged (runCli still strips HILOS_* itself).
560
+ */
561
+ function claudeGateEnv(cfg) {
562
+ const base = codingChildEnv(cfg) || process.env;
563
+ const budget = Math.max(60_000, Number(cfg?.runTimeoutMs) || 0) + 60_000;
564
+ return { ...base, MCP_TOOL_TIMEOUT: String(budget) };
565
+ }
566
+
567
+ /**
568
+ * Run claude_code through its permission-prompt seam (0777).
569
+ *
570
+ * Deliberately NOT a new transport: the ordinary argv run is untouched — same
571
+ * runCli, same resume args, same streaming — and the gate is two extra flags
572
+ * plus a loopback MCP server that lives exactly as long as the run. The prompt
573
+ * stays the final argument.
574
+ */
575
+ async function runClaudeGatedCli({
576
+ deps,
577
+ cfg,
578
+ onGateDropped,
579
+ cmd,
580
+ codeArgs,
581
+ prompt,
582
+ cwd,
583
+ signal,
584
+ onData,
585
+ permissionCallbacks,
586
+ sessionId = "",
587
+ resolveSessionId,
588
+ log = console,
589
+ }) {
590
+ const server = await deps.startClaudePermissionServer({
591
+ ...permissionCallbacks,
592
+ sessionId,
593
+ // A FRESH run has no session id to hand over: claude reveals its own in the
594
+ // init frame of its stream, which the progress emitter is already folding.
595
+ // Pulling from that snapshot means an ask carries the CLI's real session id
596
+ // without a second parser — and hilos rejects an empty one outright.
597
+ resolveSessionId,
598
+ timeoutMs: cfg.runTimeoutMs,
599
+ signal,
600
+ log,
601
+ });
602
+ try {
603
+ return await deps.runCli({
604
+ cmd,
605
+ args: [...codeArgs, ...server.args, prompt],
606
+ cwd,
607
+ timeoutMs: cfg.runTimeoutMs,
608
+ label: "coding",
609
+ signal,
610
+ env: claudeGateEnv(cfg),
611
+ onData,
612
+ // 0785 — a CLI that rejects the gate flags is retried without them. This
613
+ // fires BEFORE that ungated retry is spawned, so the room hears about the
614
+ // downgrade first rather than after the fact.
615
+ onPermissionGateDropped: onGateDropped,
616
+ });
617
+ } finally {
618
+ try {
619
+ await server.close();
620
+ } catch {
621
+ /* tearing the gate down must never fail a finished run */
622
+ }
623
+ }
624
+ }
625
+
626
+ /**
627
+ * Will this codex run take the gated transport (0785)?
628
+ *
629
+ * Answerable before the run's argv exists, because the only input that can
630
+ * change the answer is the operator's OWN command — every flag the daemon
631
+ * appends later (model, dir, image, resume, stream) is ours and none of them is
632
+ * the bypass tier. Which matters because the PROMPT is written before the
633
+ * transport is chosen, and what the prompt may claim about images depends on it.
634
+ */
635
+ function codexRunIsGated(cfg, caps) {
636
+ return shouldGateCodexPermissions({
637
+ vendor: detectVendor(cfg?.codingCmd),
638
+ runtimePermissions: caps?.runtimePermissions,
639
+ codeArgs: String(cfg?.codingCmd || "").split(" ").filter(Boolean).slice(1),
640
+ });
641
+ }
642
+
643
+ /**
644
+ * Run codex through its gated transport, with an honest fallback (0785).
645
+ *
646
+ * A granted workspace ALWAYS gets `codex mcp-server` — that is what the grant
647
+ * means, and it is the only codex transport that can ask (`codex exec` has no
648
+ * approval channel at any flag combination, see codex-mcp-session.mjs). The one
649
+ * thing the grant can't conjure is the subcommand itself: a codex old enough
650
+ * not to have it never answers the MCP handshake, and before this the run just
651
+ * failed. Now it degrades to the plain exec argv — the ungated run every codex
652
+ * did before 0777, never worse.
653
+ *
654
+ * Two rules make that degrade safe. It happens ONLY on positive evidence that
655
+ * the subcommand is absent (codexMcpTransportUnavailable — every other
656
+ * pre-handshake failure is returned as the failed run it is, because a wrongly
657
+ * failed run is recoverable and a wrongly ungated one is not). And the room is
658
+ * told BEFORE the ungated child is spawned, so the warning arrives while the
659
+ * run can still be stopped.
660
+ *
661
+ * The exec argv is the run's real one (model, images, resume, stream flags),
662
+ * so the fallback is the same run the ungated lane would have made.
663
+ */
664
+ async function runCodexGatedSession({
665
+ deps,
666
+ cfg,
667
+ onGateDropped,
668
+ cmd,
669
+ codeArgs,
670
+ prompt,
671
+ cwd,
672
+ model,
673
+ resumeThreadId = null,
674
+ signal,
675
+ onData,
676
+ onEvent,
677
+ permissionCallbacks,
678
+ }) {
679
+ const run = await deps.runCodexMcpSession({
680
+ cmd,
681
+ cwd,
682
+ prompt,
683
+ sandbox: codexSandboxFromArgs(codeArgs),
684
+ // 0785 — the resolved tier (0783) or a hand-pinned id. The gated transport
685
+ // took the account default before this, so a preset was silently ignored
686
+ // exactly where the operator was most likely to have set one.
687
+ model: model || null,
688
+ resumeThreadId,
689
+ timeoutMs: cfg.runTimeoutMs,
690
+ signal,
691
+ env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
692
+ onData,
693
+ onEvent,
694
+ ...permissionCallbacks,
695
+ });
696
+ if (!codexMcpTransportUnavailable(run)) return run;
697
+ console.log(
698
+ " code → this codex has no `mcp-server`; running WITHOUT the hilos permission gate",
699
+ );
700
+ // Warn FIRST — before a single ungated command can run.
701
+ if (typeof onGateDropped === "function") {
702
+ try {
703
+ await onGateDropped();
704
+ } catch {
705
+ /* telling the room must never break the run */
706
+ }
707
+ }
708
+ const fallback = await deps.runCli({
709
+ cmd,
710
+ args: [...codeArgs, prompt],
711
+ cwd,
712
+ timeoutMs: cfg.runTimeoutMs,
713
+ label: "coding",
714
+ signal,
715
+ env: codingChildEnv(cfg),
716
+ onData,
717
+ });
718
+ return { ...fallback, permissionGateDropped: true };
719
+ }
720
+
351
721
  async function awaitDecision({ tool, channelId, reportMessageId, cfg, deps, parentId, signal }) {
352
722
  if (!reportMessageId) return { kind: "timeout" };
353
723
  const deadline = deps.now() + cfg.decisionTimeoutMs;
@@ -398,7 +768,7 @@ async function linkPrFromUrl({ tool, channelId, url }) {
398
768
  await tool("link_pr", { channelId, repoFullName: m[1], prNumber: Number(m[2]) }).catch(() => {});
399
769
  }
400
770
 
401
- async function applyDecision({ decision, repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, existingPrUrl, settleId, runId }) {
771
+ async function applyDecision({ decision, repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, existingPrUrl, settleId, runId, usageArgs = {} }) {
402
772
  const tag = requesterTag(requester);
403
773
  const lead = tag ? `${tag} — ` : "";
404
774
  // `parentId` here is the run's thread root. Terminal outcomes ask the server
@@ -457,13 +827,21 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
457
827
  : `${lead}pushed \`${branch}\`. Open a PR manually — \`gh\` failed.`,
458
828
  prUrl: pr.ok && pr.url ? pr.url : undefined,
459
829
  caveats: pr.ok ? [] : [`gh pr create failed: ${(pr.stderr || "").trim().slice(0, 200)}`],
830
+ // 0787 — what the run cost, when the caller had a receipt to hand over.
831
+ // Ungated runs land here with the whole run's usage; a gated one already
832
+ // spent it on the proposal card and passes {}.
833
+ ...usageArgs,
460
834
  };
461
- await tool(
835
+ // 0792 — keep the id of the card this run settled onto, so the caller can
836
+ // mark it with the run's transcript. `settleId` is that card when the run
837
+ // streamed onto a live status message; otherwise it is the fresh report.
838
+ const reportRes = await tool(
462
839
  "post_report",
463
840
  settleId
464
841
  ? { ...reportArgs, messageId: settleId, broadcast: true }
465
842
  : { ...reportArgs, parentId, broadcast: true },
466
843
  );
844
+ const reportMsgId = settleId || reportRes?.messageId || null;
467
845
  // Work produced a PR → attach it to the channel so its live pill shows up.
468
846
  await linkPrFromUrl({ tool, channelId, url: pr.ok ? pr.url : null });
469
847
  // Bind the PR to the thread's run (0279/0280) so a follow-up continues it.
@@ -473,10 +851,13 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
473
851
  runId,
474
852
  ...(pr.ok && pr.url ? { prUrl: pr.url } : {}),
475
853
  status: "awaiting_review",
476
- ...(settleId ? { reportMsgId: settleId } : {}),
854
+ // 0792: record the card even when it was a fresh report rather than a
855
+ // settle — it is the run's card either way, and the transcript upload
856
+ // falls back to this exact field when it is not handed an id.
857
+ ...(reportMsgId ? { reportMsgId } : {}),
477
858
  }).catch(() => {});
478
859
  }
479
- return { status: "pushed", branch, prUrl: pr.ok ? pr.url : null };
860
+ return { status: "pushed", branch, prUrl: pr.ok ? pr.url : null, reportMsgId };
480
861
  }
481
862
 
482
863
  if (decision.kind === "rejected") {
@@ -516,7 +897,7 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
516
897
  * switched to `main` would otherwise make `gh` try to open main → main. Used
517
898
  * only when NOT gated (bias-to-action).
518
899
  */
519
- async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, settleId, runId }) {
900
+ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, settleId, runId, usageArgs = {} }) {
520
901
  const tag = requesterTag(requester);
521
902
  const lead = tag ? `${tag} — ` : "";
522
903
  const currentBranch =
@@ -568,13 +949,17 @@ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, ch
568
949
  : prFailure
569
950
  ? [`gh pr create failed: ${prFailure}`]
570
951
  : [],
952
+ // 0787 — an autonomous run's receipt rides its one and only report.
953
+ ...usageArgs,
571
954
  };
572
- await tool(
955
+ const reportRes = await tool(
573
956
  "post_report",
574
957
  settleId
575
958
  ? { ...reportArgs, messageId: settleId, broadcast: true }
576
959
  : { ...reportArgs, parentId, broadcast: true },
577
960
  );
961
+ // 0792 — the card this run settled onto, for the transcript stamp.
962
+ const reportMsgId = settleId || reportRes?.messageId || null;
578
963
  if (prUrl) await linkPrFromUrl({ tool, channelId, url: prUrl });
579
964
  // Bind the PR to the thread's run (0279/0280). Best-effort; only when recorded.
580
965
  if (runId) {
@@ -590,7 +975,7 @@ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, ch
590
975
  : prFailure || "GitHub did not return a pull request URL",
591
976
  }
592
977
  : {}),
593
- ...(settleId ? { reportMsgId: settleId } : {}),
978
+ ...(reportMsgId ? { reportMsgId } : {}),
594
979
  }).catch(() => {});
595
980
  }
596
981
  const status =
@@ -599,6 +984,7 @@ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, ch
599
984
  status,
600
985
  branch: ship.headBranch,
601
986
  prUrl,
987
+ reportMsgId,
602
988
  };
603
989
  }
604
990
 
@@ -700,16 +1086,21 @@ async function routeIntent({ name, repoFullName, transcript, workspaceMemory, cf
700
1086
  * agent, edit files now" — led by the router's distilled `brief` (what to build),
701
1087
  * with the conversation included only as background. Falls back to the raw
702
1088
  * mention when there's no brief.
703
- * @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, repoFullName?: string }} [o]
1089
+ * @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, repoFullName?: string, images?: {path: string, name: string, type: string}[], imagesReadable?: boolean }} [o]
704
1090
  */
705
1091
  export function codeTaskPrompt(o) {
706
- const { message, context, brief, repoFullName } = o || {};
1092
+ const { message, context, brief, repoFullName, images, imagesReadable } = o || {};
707
1093
  const task = (brief && brief.trim()) || String(message?.body || "").trim();
708
1094
  const transcript = context?.transcript?.trim();
709
1095
  const where = repoFullName ? ` in the git repository ${repoFullName}` : "";
710
1096
  let p =
711
1097
  `You are a coding agent working${where}. Implement the following by EDITING FILES now — ` +
712
1098
  `make the changes directly, do not just describe them, do not ask questions:\n\n${task}`;
1099
+ // 0779: the screenshots the request was about, already on this machine. Right
1100
+ // under the task, because "fix this spacing" only means something next to the
1101
+ // picture of the spacing.
1102
+ const imageNote = imagePromptNote(images, { readable: imagesReadable });
1103
+ if (imageNote) p += `\n\n${imageNote}`;
713
1104
  // The daemon owns git: it stages, commits, pushes, and opens the PR after the
714
1105
  // CLI finishes. An autonomous CLI run with skip-permissions inside a repo whose
715
1106
  // docs prescribe a commit+PR workflow will otherwise do all of that itself,
@@ -732,16 +1123,19 @@ export function codeTaskPrompt(o) {
732
1123
  * imperative "edit files now" framing as codeTaskPrompt, but it tells the agent
733
1124
  * its edits land straight in the folder and bans git/gh (there's nothing for the
734
1125
  * daemon to commit; the changes ARE the deliverable).
735
- * @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, folderPath?: string }} [o]
1126
+ * @param {{ message?: { body?: string } | null, context?: { transcript?: string } | null, brief?: string, folderPath?: string, images?: {path: string, name: string, type: string}[], imagesReadable?: boolean }} [o]
736
1127
  */
737
1128
  export function folderTaskPrompt(o) {
738
- const { message, context, brief, folderPath } = o || {};
1129
+ const { message, context, brief, folderPath, images, imagesReadable } = o || {};
739
1130
  const task = (brief && brief.trim()) || String(message?.body || "").trim();
740
1131
  const where = folderPath ? ` in the local folder ${folderPath}` : "";
741
1132
  let p =
742
1133
  `You are a coding agent working directly${where}. Implement the following by EDITING ` +
743
1134
  `FILES now — make the changes directly in this folder, do not just describe them, do not ` +
744
1135
  `ask questions:\n\n${task}`;
1136
+ // 0779 — same image handoff as the repo path.
1137
+ const imageNote = imagePromptNote(images, { readable: imagesReadable });
1138
+ if (imageNote) p += `\n\n${imageNote}`;
745
1139
  p +=
746
1140
  `\n\nIMPORTANT: your edits apply DIRECTLY to the user's folder — there is no branch, no ` +
747
1141
  `commit, and no pull request. Do NOT run git; do NOT stage, commit, push, create branches, ` +
@@ -784,7 +1178,17 @@ async function fetchContext({ channelId, tool, parentId }) {
784
1178
  }));
785
1179
  rows = messages;
786
1180
  }
787
- return { rows, transcript: rows.map((m) => `${m.author}: ${m.body}`).join("\n").slice(-6000) };
1181
+ // 0779: the rows carry resolved `attachments`, and the body carries the
1182
+ // unresolvable `![name](attachment:<uuid>)` pill token. Flatten the token and
1183
+ // name the file, so the daemon's chat replies and routing decisions stop
1184
+ // being handed a reference no model can follow.
1185
+ return {
1186
+ rows,
1187
+ transcript: rows
1188
+ .map((m) => `${m.author}: ${renderAttachmentLine(m.body, m.attachments)}`)
1189
+ .join("\n")
1190
+ .slice(-6000),
1191
+ };
788
1192
  }
789
1193
 
790
1194
  /**
@@ -1362,9 +1766,26 @@ async function handleFolderDeploy({
1362
1766
  * live), request-changes = re-run in place (bounded), reject = revert the run's
1363
1767
  * own changes (git repos only).
1364
1768
  */
1365
- async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps, signal, parentId, folderPath, brief, workspaceMemory, context }) {
1769
+ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps, signal, parentId, folderPath, brief, workspaceMemory, context, onStopRequested }) {
1366
1770
  void me;
1367
1771
  const git = deps.git;
1772
+ // 0779 — screenshots the poll loop already pulled to a temp dir at pickup.
1773
+ // The prompt names them; the argv carries them for a vendor that takes one.
1774
+ const localImages = Array.isArray(message?.images) ? message.images : [];
1775
+ const folderImagePromptArgs = localImages.length
1776
+ ? {
1777
+ images: localImages,
1778
+ imagesReadable: imagesNeedReading(detectVendor(cfg.codingCmd), {
1779
+ gated: codexRunIsGated(cfg, caps),
1780
+ }),
1781
+ }
1782
+ : {};
1783
+ // 0785 — one line per run, whichever CLI turns out to be ungateable.
1784
+ const noticeUngatedRun = createUngatedRunNotice({
1785
+ // Read at post time, not now: the thread root is the ack this run posts.
1786
+ post: (body) => tool("post_message", { channelId, parentId: threadRoot, body }),
1787
+ log: console,
1788
+ });
1368
1789
  const tag = requesterTag(message.author);
1369
1790
  const lead = tag ? `${tag} — ` : "";
1370
1791
 
@@ -1411,6 +1832,33 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1411
1832
  // supports it, else the edit-in-place heartbeat, else nothing. Kept across
1412
1833
  // request-changes re-runs on the same status message.
1413
1834
  let progressId = null;
1835
+ // 0787 — what the run cost, folded across every round of the revise loop and
1836
+ // taken at each report so each round's spend is reported exactly once. The
1837
+ // model id the stream names wins; `resolvedModelId` is the spawn-time
1838
+ // fallback for a CLI (codex) whose stream never says.
1839
+ const runUsage = createUsageFold();
1840
+ const cliVendor = detectVendor(cfg.codingCmd);
1841
+ let resolvedModelId = pinnedModelId(cfg.codingCmd);
1842
+ let lastTerminalState = "done";
1843
+ /**
1844
+ * Accounting-only settlement (0787). Plenty of terminal exits produce no
1845
+ * report at all — nothing changed, the CLI never started, a person pressed
1846
+ * Stop — and the tokens those runs spent were being dropped on the floor.
1847
+ * `post_progress` is the terminal call the daemon already makes on every one
1848
+ * of them, it is already run-scoped, and it already carries `runId`, so the
1849
+ * spend rides home on a call that was happening anyway rather than on a new
1850
+ * tool. Only ever the DELTA, so a later report cannot re-report it; a card is
1851
+ * required because there is no other message to hang a terminal update on.
1852
+ */
1853
+ const settleRunUsage = async () => {
1854
+ const args = reportUsageArgs(runUsage, { vendor: cliVendor, modelId: resolvedModelId });
1855
+ if (!args.usage || !progressId) return;
1856
+ await tool("post_progress", {
1857
+ messageId: progressId,
1858
+ progress: { state: lastTerminalState },
1859
+ usage: args.usage,
1860
+ }).catch(() => {});
1861
+ };
1414
1862
  const folderWorking = (elapsedMs, lastLine) => {
1415
1863
  const base = `Working in \`${folderPath}\` — ${fmtElapsed(elapsedMs)} elapsed. I'll report what changed when it's done.`;
1416
1864
  const tail = oneLine(lastLine);
@@ -1425,7 +1873,10 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1425
1873
  const streamOn = Boolean(caps.postProgress);
1426
1874
  const streamArgs = streamOn ? codeStreamArgs(vendor) : [];
1427
1875
  let emitter = null;
1428
- let progressInflight = Promise.resolve();
1876
+ // Every dispatched progress write, so the settle can wait them out (0289).
1877
+ let drainProgress = async () => {};
1878
+ // 0782 — the silent-work stop poll, torn down in the same finally as the run.
1879
+ let stopPoller = null;
1429
1880
  let stopHeartbeat = () => {};
1430
1881
  let lastLine = "";
1431
1882
  // With stream args the CLI's stdout is NDJSON events, not prose — parse it and
@@ -1437,6 +1888,11 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1437
1888
  const foldResultEvents = (events) => {
1438
1889
  for (const ev of events || []) {
1439
1890
  if (ev && ev.t === "result" && ev.summary) resultText = ev.summary;
1891
+ // 0787 — read the numbers off the SAME parse the summary comes from.
1892
+ // This parser exists whenever stream args do, which is a superset of the
1893
+ // cases where a progress emitter exists, so folder usage never depends
1894
+ // on the status card having been posted.
1895
+ runUsage.push(ev);
1440
1896
  }
1441
1897
  };
1442
1898
  if (streamOn) {
@@ -1450,21 +1906,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1450
1906
  }
1451
1907
  const statusId = progressId;
1452
1908
  if (statusId) {
1909
+ // 0782 — a folder run has no runs row, so the stop check is scoped to
1910
+ // the card's thread by the server; it degrades to "no stop" quietly.
1911
+ const sender = createProgressSender({ tool, statusId, onStopRequested });
1453
1912
  emitter = createProgressEmitter({
1454
1913
  parser: makeStreamParser(vendor),
1455
1914
  now: deps.now,
1456
1915
  throttleMs: cfg.progressMs,
1457
- send: (p) => {
1458
- try {
1459
- const r = tool("post_progress", { messageId: statusId, progress: p });
1460
- if (r && typeof r.then === "function") {
1461
- const done = r.then(() => {}, () => {});
1462
- progressInflight = Promise.all([progressInflight, done]).then(() => {}, () => {});
1463
- }
1464
- } catch {
1465
- /* a progress send must never break the run */
1466
- }
1467
- },
1916
+ send: sender.send,
1917
+ });
1918
+ drainProgress = () => sender.drain();
1919
+ // The heartbeat only fires when the CLI speaks. This asks anyway, so a
1920
+ // run inside one long silent command is still stoppable.
1921
+ stopPoller = createStopPoller({
1922
+ poll: () => sender.poll("working"),
1923
+ intervalMs: cfg.stopPollMs || STOP_POLL_MS,
1468
1924
  });
1469
1925
  }
1470
1926
  } else if (caps.editMessage && cfg.heartbeatMs > 0) {
@@ -1497,14 +1953,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1497
1953
  let run;
1498
1954
  // Model preset (0504): same run-time resolution as the repo path.
1499
1955
  const modelArgs = await modelArgsFor(cfg, vendor);
1956
+ // 0787: the id we actually handed the CLI, kept for the usage receipt when
1957
+ // the vendor's own stream never names a model.
1958
+ if (modelArgs[0] === "--model" && modelArgs[1]) resolvedModelId = modelArgs[1];
1500
1959
  // Project pin (0608): opencode reads its project from PWD, so without this
1501
1960
  // a folder run could edit the daemon's launch directory instead of the
1502
1961
  // folder the channel is linked to. [] for every other vendor.
1503
1962
  const dirArgs = codeDirArgs(vendor, folderPath, cfg.codingCmd);
1963
+ // 0779: [] for every vendor without a verified image flag — their argv is
1964
+ // byte-identical to before, and the prompt note still names the files.
1965
+ const imageArgs = codeImageArgs(vendor, localImages);
1504
1966
  const codeArgs = [
1505
1967
  ...parts.slice(1),
1506
1968
  ...modelArgs,
1507
1969
  ...dirArgs,
1970
+ ...imageArgs,
1508
1971
  ...streamArgs,
1509
1972
  ];
1510
1973
  const handleCliData = (c) => {
@@ -1525,6 +1988,23 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1525
1988
  }
1526
1989
  }
1527
1990
  };
1991
+ // The structured transports (ACP, codex's MCP server) produce events
1992
+ // directly — there is no NDJSON on stdout for `resultParser` to read, so
1993
+ // this is where THEIR usage joins the fold (0787). A stdout run never
1994
+ // reaches here, and a structured run never reaches resultParser, so the two
1995
+ // sources can't double-count the same tokens.
1996
+ const handleCliEvent = (ev) => {
1997
+ try {
1998
+ emitter?.foldEvent(ev);
1999
+ } catch {
2000
+ /* a progress fold must never break the run */
2001
+ }
2002
+ try {
2003
+ runUsage.push(ev);
2004
+ } catch {
2005
+ /* accounting must never break the run */
2006
+ }
2007
+ };
1528
2008
  try {
1529
2009
  const useAcpTransport = shouldUseAcpTransport({
1530
2010
  vendor,
@@ -1541,6 +2021,26 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1541
2021
  codeArgs,
1542
2022
  codingCmd: cfg.codingCmd,
1543
2023
  });
2024
+ // 0777: the two vendors that had no gate at all. Neither is an ACP
2025
+ // adapter — each CLI turned out to have its own native seam (see the
2026
+ // module headers), so both compose with everything already here.
2027
+ const gateCodexPermissions =
2028
+ !useAcpTransport &&
2029
+ !useRuntimePermissionBridge &&
2030
+ shouldGateCodexPermissions({
2031
+ vendor,
2032
+ runtimePermissions: caps.runtimePermissions,
2033
+ codeArgs,
2034
+ });
2035
+ const gateClaudePermissions =
2036
+ !useAcpTransport &&
2037
+ !useRuntimePermissionBridge &&
2038
+ !gateCodexPermissions &&
2039
+ shouldGateClaudePermissions({
2040
+ vendor,
2041
+ runtimePermissions: caps.runtimePermissions,
2042
+ codeArgs,
2043
+ });
1544
2044
  if (useAcpTransport) {
1545
2045
  run = await deps.runAcpSession({
1546
2046
  cmd: parts[0],
@@ -1551,6 +2051,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1551
2051
  signal,
1552
2052
  env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
1553
2053
  onData: handleCliData,
2054
+ onEvent: handleCliEvent,
1554
2055
  ...openCodePermissionCallbacks({
1555
2056
  tool,
1556
2057
  channelId,
@@ -1574,6 +2075,45 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1574
2075
  threadRoot,
1575
2076
  }),
1576
2077
  });
2078
+ } else if (gateCodexPermissions) {
2079
+ run = await runCodexGatedSession({
2080
+ deps,
2081
+ cfg,
2082
+ onGateDropped: () => noticeUngatedRun(parts[0]),
2083
+ cmd: parts[0],
2084
+ codeArgs,
2085
+ cwd: folderPath,
2086
+ prompt: memoryPreamble(workspaceMemory) + promptText,
2087
+ model: resolvedModelId,
2088
+ signal,
2089
+ onData: handleCliData,
2090
+ onEvent: handleCliEvent,
2091
+ permissionCallbacks: openCodePermissionCallbacks({
2092
+ tool,
2093
+ channelId,
2094
+ threadRoot,
2095
+ provider: vendor,
2096
+ }),
2097
+ });
2098
+ } else if (gateClaudePermissions) {
2099
+ run = await runClaudeGatedCli({
2100
+ deps,
2101
+ cfg,
2102
+ onGateDropped: () => noticeUngatedRun(parts[0]),
2103
+ cmd: parts[0],
2104
+ codeArgs,
2105
+ prompt: memoryPreamble(workspaceMemory) + promptText,
2106
+ cwd: folderPath,
2107
+ signal,
2108
+ onData: handleCliData,
2109
+ resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
2110
+ permissionCallbacks: openCodePermissionCallbacks({
2111
+ tool,
2112
+ channelId,
2113
+ threadRoot,
2114
+ provider: vendor,
2115
+ }),
2116
+ });
1577
2117
  } else {
1578
2118
  run = await deps.runCli({
1579
2119
  cmd: parts[0],
@@ -1589,10 +2129,21 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1589
2129
  onData: handleCliData,
1590
2130
  });
1591
2131
  }
2132
+ // 0785 backstop. The notice normally goes out from `onGateDropped`,
2133
+ // BEFORE the ungated child is spawned — the room has to be warned while
2134
+ // the run can still be stopped, not told afterwards what it already did.
2135
+ // This catches a degrade that reached us without firing that hook; the
2136
+ // latch makes it a no-op in the ordinary case.
2137
+ if (run?.permissionGateDropped) await noticeUngatedRun(parts[0]);
1592
2138
  } finally {
1593
2139
  stopHeartbeat();
2140
+ stopPoller?.stop();
1594
2141
  if (emitter) {
1595
2142
  const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
2143
+ // Remembered for the accounting-only settlement (0787): an exit with no
2144
+ // report re-sends this same terminal state, so the card is never
2145
+ // repainted into something it wasn't.
2146
+ lastTerminalState = errored ? "error" : "done";
1596
2147
  let reason = "";
1597
2148
  if (errored && run && !run.aborted) {
1598
2149
  const stderrTail = oneLine((run.stderr || "").trim().split("\n").slice(-3).join(" "), 200);
@@ -1607,7 +2158,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1607
2158
  /* ignore */
1608
2159
  }
1609
2160
  try {
1610
- await progressInflight;
2161
+ await drainProgress();
1611
2162
  } catch {
1612
2163
  /* a drain failure must never break the run */
1613
2164
  }
@@ -1675,7 +2226,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1675
2226
  };
1676
2227
 
1677
2228
  // --- Run ---
1678
- let runResult = await runFolderCli(folderTaskPrompt({ message, context, brief, folderPath }));
2229
+ let runResult = await runFolderCli(folderTaskPrompt({ message, context, brief, folderPath, ...folderImagePromptArgs }));
1679
2230
  if (runResult.aborted) {
1680
2231
  await tool("post_message", {
1681
2232
  channelId,
@@ -1683,6 +2234,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1683
2234
  broadcast: Boolean(threadRoot),
1684
2235
  body: `Stopped — I left \`${folderPath}\` as it was.`,
1685
2236
  });
2237
+ await settleRunUsage(); // a stopped run still spent tokens (0787)
1686
2238
  return { status: "cancelled" };
1687
2239
  }
1688
2240
 
@@ -1705,6 +2257,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1705
2257
  body = `The run didn't finish${why}${tail}. ${left} — mention me to retry.`;
1706
2258
  }
1707
2259
  await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body });
2260
+ await settleRunUsage(); // no report on this exit — settle the spend anyway (0787)
1708
2261
  return { status: "run-failed" };
1709
2262
  }
1710
2263
  if (!runResult.failed && isGit && !hasChanges(changed)) {
@@ -1714,13 +2267,22 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1714
2267
  broadcast: Boolean(threadRoot),
1715
2268
  body: `The run finished but nothing changed in \`${folderPath}\`. Mention me to try a different approach.`,
1716
2269
  });
2270
+ await settleRunUsage(); // "nothing changed" is not "nothing spent" (0787)
1717
2271
  return { status: "no-changes" };
1718
2272
  }
1719
2273
 
1720
2274
  // --- Report card + decision loop ---
1721
2275
  const postFolderReport = async (rr, ch) => {
1722
2276
  const report = buildReport(rr, ch);
1723
- const res = await tool("post_report", { channelId, parentId: threadRoot, broadcast: true, ...report });
2277
+ const res = await tool("post_report", {
2278
+ channelId,
2279
+ parentId: threadRoot,
2280
+ broadcast: true,
2281
+ ...report,
2282
+ // 0787 — a folder run has no runs row, so the receipt rides the card and
2283
+ // the ledger row lands unattached to a run. Still the honest number.
2284
+ ...reportUsageArgs(runUsage, { vendor: cliVendor, modelId: resolvedModelId }),
2285
+ });
1724
2286
  return res?.messageId ?? null;
1725
2287
  };
1726
2288
  let reportMessageId = await postFolderReport(runResult, changed);
@@ -1735,11 +2297,12 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1735
2297
  body: `Revising in \`${folderPath}\` with your feedback${decision.note ? `: ${decision.note}` : ""} (round ${round + 1}/${maxRounds}).`,
1736
2298
  });
1737
2299
  runResult = await runFolderCli(
1738
- folderTaskPrompt({ message, context, brief, folderPath }) +
2300
+ folderTaskPrompt({ message, context, brief, folderPath, ...folderImagePromptArgs }) +
1739
2301
  `\n\nReviewer feedback to address: ${decision.note || "(see the channel)"}`,
1740
2302
  );
1741
2303
  if (runResult.aborted) {
1742
2304
  await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body: `Stopped — I left \`${folderPath}\` as it was.` });
2305
+ await settleRunUsage();
1743
2306
  return { status: "cancelled" };
1744
2307
  }
1745
2308
  changed = computeChanged();
@@ -1814,6 +2377,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
1814
2377
 
1815
2378
  if (decision.kind === "cancelled") {
1816
2379
  await tool("post_message", { channelId, parentId: threadRoot, broadcast: Boolean(threadRoot), body: `Stopped — the changes so far are still in \`${folderPath}\`.` });
2380
+ await settleRunUsage();
1817
2381
  return { status: "cancelled" };
1818
2382
  }
1819
2383
 
@@ -1837,6 +2401,10 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
1837
2401
  const deps = depsOverride ? { ...defaultDeps(), ...depsOverride } : defaultDeps();
1838
2402
  const git = deps.git;
1839
2403
  const signal = opts.signal;
2404
+ // 0782 — the poll loop's hand-off for "a person pressed Stop on the card": it
2405
+ // posts the notice naming them and aborts this job's signal, which is what
2406
+ // tears the coding CLI's process group down (runCli's abort path).
2407
+ const onStopRequested = opts.onStopRequested;
1840
2408
  // When the mention was a thread reply, keep the whole exchange in that thread.
1841
2409
  const parentId = message.parentId ?? null;
1842
2410
 
@@ -2051,6 +2619,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2051
2619
  brief: routed.task,
2052
2620
  workspaceMemory,
2053
2621
  context,
2622
+ onStopRequested,
2054
2623
  });
2055
2624
  }
2056
2625
  // Not a coding task → post the router's reply if it produced one, else fall
@@ -2409,6 +2978,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2409
2978
  threadRootId: threadRoot,
2410
2979
  taskText: message.body,
2411
2980
  branch,
2981
+ // 0782 — the mention we picked up. The server reads its human author and
2982
+ // records them as the run's requester, which is who (besides workspace
2983
+ // admins) may stop this run from its card. The thread root is NOT that
2984
+ // person: it can be an ack this agent wrote, or someone else's thread.
2985
+ requestedByMessageId: message.id,
2412
2986
  // Map an unrecognized command to null rather than the off-vocabulary
2413
2987
  // "unknown" — provider is documented as
2414
2988
  // claude_code|codex|cursor|opencode|antigravity|hermes|hilos.
@@ -2455,7 +3029,35 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2455
3029
  const streamOn = Boolean(caps.postProgress);
2456
3030
  const vendor = detectVendor(cfg.codingCmd);
2457
3031
  const streamArgs = codeStreamArgs(vendor);
3032
+ // 0792 — the run's own record, when the operator asked for one. Three gates,
3033
+ // all of which have to say yes: the operator's config, the server offering
3034
+ // the tool, and a durable run row to hang the object off. The tap lives at
3035
+ // TASK scope, not per-CLI-invocation, so a resume retry and the gate:true
3036
+ // revision rounds all land in one transcript — the run is the unit, not the
3037
+ // spawn.
3038
+ const transcriptTap =
3039
+ cfg.uploadTranscripts && caps.uploadTranscript && runId ? createTranscriptTap() : null;
3040
+ // 0779 — screenshots the poll loop pulled to a temp dir at pickup. The prompt
3041
+ // names them; the argv carries them for a vendor with a verified image flag.
3042
+ const localImages = Array.isArray(message?.images) ? message.images : [];
3043
+ const codeImagePromptArgs = localImages.length
3044
+ ? {
3045
+ images: localImages,
3046
+ imagesReadable: imagesNeedReading(vendor, { gated: codexRunIsGated(cfg, caps) }),
3047
+ }
3048
+ : {};
3049
+ // 0785 — one line per run, whichever CLI turns out to be ungateable.
3050
+ const noticeUngatedRun = createUngatedRunNotice({
3051
+ // Read at post time: the thread root is the ack this run posts.
3052
+ post: (body) => tool("post_message", { channelId, parentId: threadRoot, body }),
3053
+ log: console,
3054
+ });
2458
3055
  let runSessionId = null; // captured from the stream for 0282 (resume)
3056
+ // 0787 — what the run cost, folded across every runAndStage call (a gated
3057
+ // iterate runs the CLI more than once) and taken at each report.
3058
+ const runUsage = createUsageFold();
3059
+ let resolvedModelId = pinnedModelId(cfg.codingCmd);
3060
+ let lastTerminalState = "done";
2459
3061
  const machine = hostname();
2460
3062
 
2461
3063
  // Session resume (0282): on an ITERATE we can resume the coding agent's SESSION so
@@ -2466,8 +3068,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2466
3068
  // (get_active_run) with no local match means the session likely lives on another
2467
3069
  // machine/instance — a bad `--resume` id makes claude error → empty diff → a failed
2468
3070
  // run, so we DON'T resume and degrade to today's branch+feedback (never worse).
2469
- // claude_code + cursor + opencode have proven resume flags (0282/0573/0608);
2470
- // codex/unknown → []. The gate itself is pure (resumeDecision in resume.mjs):
3071
+ // claude_code + cursor + opencode + codex have proven resume flags
3072
+ // (0282/0573/0608/0783); unknown → []. Note codex's is a SUBCOMMAND (`exec
3073
+ // resume <id>`), not a flag — it still splices in here, because flags placed
3074
+ // before it are parsed as `exec`'s own (verified live on codex-cli 0.144.1).
3075
+ // The gate itself is pure (resumeDecision in resume.mjs):
2471
3076
  // vendor, machine, project and server agreement all have to line up, and any
2472
3077
  // "no" degrades to the branch+feedback iterate rather than risking a bad id.
2473
3078
  const projectKey = codeProjectKey(vendor, repoPath, cfg.codingCmd);
@@ -2528,6 +3133,25 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2528
3133
  clearInterval(beat);
2529
3134
  };
2530
3135
  };
3136
+ /**
3137
+ * The authoritative check before anything leaves this machine (0782).
3138
+ *
3139
+ * The in-run poll stops when the CLI does, and the git work that follows —
3140
+ * commit, push, `gh pr create` — can take a minute with nothing streaming. A
3141
+ * stop that lands in THAT window would otherwise ship work nobody is waiting
3142
+ * on. So we ask the server one more time, in the shape that changes nothing:
3143
+ * `state` echoes the card's current lifecycle, and a stopped run's card is
3144
+ * frozen server-side regardless. A stop here fires the same hand-off the
3145
+ * heartbeat does (the room hears who stopped it; the job's signal aborts).
3146
+ */
3147
+ const stoppedBeforeShip = async (state = "done") => {
3148
+ // Only the streaming card carries progress metadata. The legacy heartbeat's
3149
+ // message is plain text, and stamping a lifecycle onto it would turn a
3150
+ // status line into a run card — so that path simply has no pre-ship check.
3151
+ if (!streamOn || !progressId) return false;
3152
+ const sender = createProgressSender({ tool, statusId: progressId, runId, onStopRequested });
3153
+ return await sender.poll(state);
3154
+ };
2531
3155
  // Once the run ends, retire the "still working…" progress reply so it doesn't
2532
3156
  // sit there claiming the agent is alive. No-op when no beat ever fired (short
2533
3157
  // run) or without edit_message. Best-effort.
@@ -2537,6 +3161,25 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2537
3161
  }
2538
3162
  };
2539
3163
 
3164
+ /**
3165
+ * Accounting-only settlement (0787) — the repo lane's copy of the folder
3166
+ * lane's. A run that ends with no report (no changes, a CLI that never
3167
+ * started, a person pressing Stop, a self-committing agent under the gate)
3168
+ * still spent tokens, and `post_progress` is the terminal, run-scoped call
3169
+ * the daemon already makes on all of them. Only ever the DELTA, so calling it
3170
+ * on a path that later reports anyway is harmless: the fold is already empty.
3171
+ */
3172
+ const settleRunUsage = async () => {
3173
+ const args = reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId });
3174
+ if (!args.usage || !progressId) return;
3175
+ await tool("post_progress", {
3176
+ messageId: progressId,
3177
+ ...(runId ? { runId } : {}),
3178
+ progress: { state: lastTerminalState },
3179
+ usage: args.usage,
3180
+ }).catch(() => {});
3181
+ };
3182
+
2540
3183
  const parts = cfg.codingCmd.split(" ").filter(Boolean);
2541
3184
  // Run the CLI and stage everything it changed; return the diff stats (no post).
2542
3185
  // Workspace memory (the project's soul) is prepended so the coding agent has
@@ -2556,7 +3199,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2556
3199
  // when post_report settles the report, it would re-write metadata WITHOUT the
2557
3200
  // report (it read the pre-settle snapshot) and clobber it. Draining every send
2558
3201
  // here guarantees no progress write is in flight once we settle.
2559
- let progressInflight = Promise.resolve();
3202
+ let drainProgress = async () => {};
3203
+ // 0782 — the silent-work stop poll, torn down in the same finally as the run.
3204
+ let stopPoller = null;
2560
3205
  if (streamOn) {
2561
3206
  if (!progressId) {
2562
3207
  try {
@@ -2571,27 +3216,35 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2571
3216
  }
2572
3217
  }
2573
3218
  const statusId = progressId;
2574
- if (statusId) {
2575
- emitter = createProgressEmitter({
2576
- parser: makeStreamParser(vendor),
2577
- now: deps.now,
2578
- throttleMs: cfg.progressMs,
2579
- // The module never imports MCP — the send fn is injected here. It must
2580
- // never throw into the run (createProgressEmitter also guards).
2581
- send: (p) => {
2582
- try {
2583
- const r = tool("post_progress", { messageId: statusId, progress: p });
2584
- if (r && typeof r.then === "function") {
2585
- const done = r.then(() => {}, () => {});
2586
- progressInflight = Promise.all([progressInflight, done]).then(
2587
- () => {},
2588
- () => {},
2589
- );
2590
- }
2591
- } catch {
2592
- /* a progress send must never break the run */
2593
- }
2594
- },
3219
+ // The module never imports MCP — the send fn is injected here. It must
3220
+ // never throw into the run (createProgressEmitter also guards). 0782:
3221
+ // the sender also carries the stop signal home on this same beat, bound
3222
+ // to THIS run's id so a stop can't be read off a neighbouring thread.
3223
+ const sender = statusId
3224
+ ? createProgressSender({ tool, statusId, runId, onStopRequested })
3225
+ : null;
3226
+ // The emitter is built whether or not the status card posted (0787). The
3227
+ // stream flags are already on the argv either way, so the CLI is speaking
3228
+ // NDJSON regardless; gating the parser on a card meant one transient
3229
+ // post_message failure silently cost the run its session id (the resume
3230
+ // record's only source on an argv run) and its usage. A card is where
3231
+ // progress is SHOWN, never how the stream is read — this is what the
3232
+ // folder path has always done. With no card the sends are dropped on the
3233
+ // floor rather than skipped, so nothing else in the run changes shape.
3234
+ emitter = createProgressEmitter({
3235
+ parser: makeStreamParser(vendor),
3236
+ now: deps.now,
3237
+ throttleMs: cfg.progressMs,
3238
+ send: sender ? sender.send : () => {},
3239
+ });
3240
+ if (sender) {
3241
+ drainProgress = () => sender.drain();
3242
+ // The heartbeat only fires when the CLI speaks. This asks anyway, so a
3243
+ // run inside one long silent command (a full test suite, a slow install)
3244
+ // is still stoppable from the room.
3245
+ stopPoller = createStopPoller({
3246
+ poll: () => sender.poll("working"),
3247
+ intervalMs: cfg.stopPollMs || STOP_POLL_MS,
2595
3248
  });
2596
3249
  }
2597
3250
  } else {
@@ -2605,10 +3258,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2605
3258
  // fallback) suppresses it. buildResumeArgs is [] unless vendor+session make
2606
3259
  // resume safe, so a non-resume run is byte-identical to the pre-0282 ARGV.
2607
3260
  const resumeArgs = resume ? buildResumeArgs(vendor, resumeSessionId) : [];
2608
- // Model preset (0504): resolved at run time against the CLI's own model
2609
- // list (cursor only today) — [] when unset/unresolvable, so the tool's
2610
- // default stands. Inserted before resume/stream flags, after the base.
3261
+ // Model preset (0504, codex in 0783): resolved at run time against the
3262
+ // CLI's own model list — [] when unset/unresolvable, so the tool's default
3263
+ // stands. Inserted before resume/stream flags, after the base — an order
3264
+ // codex depends on, since its resume is a subcommand and the model flag
3265
+ // has to reach `exec`, i.e. sit BEFORE `resume`.
2611
3266
  const modelArgs = await modelArgsFor(cfg, vendor);
3267
+ // 0787: remember the id we actually handed the CLI — the usage receipt for
3268
+ // a vendor whose stream never names a model (codex) has nothing else to say.
3269
+ if (modelArgs[0] === "--model" && modelArgs[1]) resolvedModelId = modelArgs[1];
2612
3270
  // Project pin (0608, opencode only): the CLI resolves its project from PWD,
2613
3271
  // not the spawn cwd, and its sessions are per project — `--dir` makes both
2614
3272
  // deterministic. [] for every other vendor (and for an `--attach`ed run,
@@ -2616,13 +3274,29 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2616
3274
  // The spawned PWD now matches the cwd too (0615); `--dir` stays as the
2617
3275
  // CLI's own explicit contract, and to keep attach runs off our local path.
2618
3276
  const dirArgs = codeDirArgs(vendor, repoPath, cfg.codingCmd);
3277
+ // 0779: BEFORE resumeArgs, for the same reason modelArgs are — codex's
3278
+ // resume is a SUBCOMMAND (`exec resume <id>`) and `--image=` belongs to
3279
+ // `exec`, so it has to sit ahead of it. [] for every vendor without a
3280
+ // verified flag, leaving their argv byte-identical to before.
3281
+ const imageArgs = codeImageArgs(vendor, localImages);
2619
3282
  const codeArgs = streamOn
2620
- ? [...parts.slice(1), ...modelArgs, ...dirArgs, ...resumeArgs, ...streamArgs]
2621
- : [...parts.slice(1), ...modelArgs, ...dirArgs, ...resumeArgs];
3283
+ ? [...parts.slice(1), ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs, ...streamArgs]
3284
+ : [...parts.slice(1), ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs];
2622
3285
  const handleCliData = (c) => {
2623
3286
  // Keep tracking lastLine as a fallback (legacy heartbeat / honesty).
2624
3287
  const lines = String(c).split("\n").map((s) => s.trim()).filter(Boolean);
2625
3288
  if (lines.length) lastLine = lines[lines.length - 1];
3289
+ // 0792 — the raw tail, before the parser reduces it to eight steps. Every
3290
+ // transport routes its child's stdout through this one callback, so the
3291
+ // tap sees an argv run, a permission-bridged run, and an ACP session
3292
+ // alike; a transport that writes nothing to stdout simply leaves it empty.
3293
+ if (transcriptTap) {
3294
+ try {
3295
+ transcriptTap.push(c);
3296
+ } catch {
3297
+ /* keeping a record must never break the run */
3298
+ }
3299
+ }
2626
3300
  if (emitter) {
2627
3301
  try {
2628
3302
  emitter.feed(c);
@@ -2631,6 +3305,24 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2631
3305
  }
2632
3306
  }
2633
3307
  };
3308
+ // The structured transports (ACP, `codex mcp-server`) put only the
3309
+ // assistant's PROSE on stdout — every tool call travels here instead. The
3310
+ // live card has always read this; the transcript has to as well, or a gated
3311
+ // run uploads a page of narration with no execution anywhere in it.
3312
+ const handleCliEvent = (ev) => {
3313
+ if (transcriptTap) {
3314
+ try {
3315
+ transcriptTap.pushEvent(ev);
3316
+ } catch {
3317
+ /* keeping a record must never break the run */
3318
+ }
3319
+ }
3320
+ try {
3321
+ emitter?.foldEvent(ev);
3322
+ } catch {
3323
+ /* a progress fold must never break the run */
3324
+ }
3325
+ };
2634
3326
  let run;
2635
3327
  try {
2636
3328
  // OpenCode's own non-interactive CLI auto-rejects every permission ask
@@ -2643,7 +3335,6 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2643
3335
  runtimePermissions: caps.runtimePermissions,
2644
3336
  codeArgs,
2645
3337
  codingCmd: cfg.codingCmd,
2646
- resumeSessionId: resume ? resumeSessionId : null,
2647
3338
  });
2648
3339
  const useRuntimePermissionBridge =
2649
3340
  !useAcpTransport &&
@@ -2653,16 +3344,41 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2653
3344
  codeArgs,
2654
3345
  codingCmd: cfg.codingCmd,
2655
3346
  });
3347
+ // 0777, and the reason it composes with resume (0778): claude keeps its
3348
+ // ordinary argv run (resumeArgs included), and codex's gated transport
3349
+ // has its OWN resume — `codex-reply {threadId}` — so a gated iterate
3350
+ // continues the same thread instead of trading continuity for a gate.
3351
+ const gateCodexPermissions =
3352
+ !useAcpTransport &&
3353
+ !useRuntimePermissionBridge &&
3354
+ shouldGateCodexPermissions({
3355
+ vendor,
3356
+ runtimePermissions: caps.runtimePermissions,
3357
+ codeArgs,
3358
+ });
3359
+ const gateClaudePermissions =
3360
+ !useAcpTransport &&
3361
+ !useRuntimePermissionBridge &&
3362
+ !gateCodexPermissions &&
3363
+ shouldGateClaudePermissions({
3364
+ vendor,
3365
+ runtimePermissions: caps.runtimePermissions,
3366
+ codeArgs,
3367
+ });
2656
3368
  if (useAcpTransport) {
2657
3369
  run = await deps.runAcpSession({
2658
3370
  cmd: parts[0],
2659
3371
  vendor,
2660
3372
  cwd: repoPath,
2661
3373
  prompt: memoryPreamble(workspaceMemory) + promptText,
3374
+ // 0778: approvals AND continuity. `resume:false` (the never-worse
3375
+ // retry) drops it exactly like buildResumeArgs does.
3376
+ resumeSessionId: resume ? resumeSessionId : null,
2662
3377
  timeoutMs: cfg.runTimeoutMs,
2663
3378
  signal,
2664
3379
  env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
2665
3380
  onData: handleCliData,
3381
+ onEvent: handleCliEvent,
2666
3382
  ...openCodePermissionCallbacks({
2667
3383
  tool,
2668
3384
  channelId,
@@ -2691,8 +3407,59 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2691
3407
  runId,
2692
3408
  }),
2693
3409
  });
3410
+ } else if (gateCodexPermissions) {
3411
+ run = await runCodexGatedSession({
3412
+ deps,
3413
+ cfg,
3414
+ onGateDropped: () => noticeUngatedRun(parts[0]),
3415
+ cmd: parts[0],
3416
+ codeArgs,
3417
+ cwd: repoPath,
3418
+ prompt: memoryPreamble(workspaceMemory) + promptText,
3419
+ model: resolvedModelId,
3420
+ // Codex's own resume over this transport. `resume:false` (the
3421
+ // never-worse retry) drops it exactly like buildResumeArgs does.
3422
+ resumeThreadId: resume ? resumeSessionId : null,
3423
+ signal,
3424
+ onData: handleCliData,
3425
+ onEvent: handleCliEvent,
3426
+ permissionCallbacks: openCodePermissionCallbacks({
3427
+ tool,
3428
+ channelId,
3429
+ threadRoot,
3430
+ runId,
3431
+ provider: vendor,
3432
+ }),
3433
+ });
3434
+ } else if (gateClaudePermissions) {
3435
+ run = await runClaudeGatedCli({
3436
+ deps,
3437
+ cfg,
3438
+ onGateDropped: () => noticeUngatedRun(parts[0]),
3439
+ cmd: parts[0],
3440
+ // codeArgs already carries this run's resume flags, so a gated
3441
+ // iterate resumes AND raises cards — the two never traded off.
3442
+ codeArgs,
3443
+ prompt: memoryPreamble(workspaceMemory) + promptText,
3444
+ cwd: repoPath,
3445
+ signal,
3446
+ onData: handleCliData,
3447
+ sessionId: resume ? resumeSessionId ?? "" : "",
3448
+ resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
3449
+ permissionCallbacks: openCodePermissionCallbacks({
3450
+ tool,
3451
+ channelId,
3452
+ threadRoot,
3453
+ runId,
3454
+ provider: vendor,
3455
+ }),
3456
+ });
2694
3457
  } else {
2695
- run = await runCli({
3458
+ // deps.runCli, not the bare import: `defaultDeps` wraps the very same
3459
+ // function, so production is byte-identical, but the ungated repo run
3460
+ // was the ONE coding path that escaped the injectable runner — which is
3461
+ // why nothing above the unit tests could ever drive it (0787).
3462
+ run = await deps.runCli({
2696
3463
  cmd: parts[0],
2697
3464
  args: [...codeArgs, memoryPreamble(workspaceMemory) + promptText],
2698
3465
  cwd: repoPath,
@@ -2703,8 +3470,12 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2703
3470
  onData: handleCliData,
2704
3471
  });
2705
3472
  }
3473
+ // 0785 — the gate was expected here and the CLI couldn't hold it. Say so
3474
+ // in the room, once per run, rather than only in the daemon's console.
3475
+ if (run?.permissionGateDropped) await noticeUngatedRun(parts[0]);
2706
3476
  } finally {
2707
3477
  stopHeartbeat();
3478
+ stopPoller?.stop();
2708
3479
  if (run?.sessionId) runSessionId = run.sessionId;
2709
3480
  if (emitter) {
2710
3481
  // Terminal state: flip the status card off "working" (state 'done'/'error')
@@ -2713,6 +3484,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2713
3484
  // (0289) — the card cross-fades run → report; for no-changes/failed/gate
2714
3485
  // outcomes this terminal 'done'/'error' is the card's final state.
2715
3486
  const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
3487
+ // Remembered so an exit with no report re-sends this same state (0787)
3488
+ // rather than repainting the card into something it wasn't.
3489
+ lastTerminalState = errored ? "error" : "done";
2716
3490
  // On error, carry an honest reason onto the card (0294) — the same signal
2717
3491
  // the chat note uses: spawn error message, else exit status, plus a short
2718
3492
  // stderr tail. Sanitized again server-side; an old server just drops it.
@@ -2737,7 +3511,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2737
3511
  // BEFORE runAndStage returns, so nothing is in flight when the caller
2738
3512
  // settles the report onto this card — no lost-update clobber (0289).
2739
3513
  try {
2740
- await progressInflight;
3514
+ await drainProgress();
2741
3515
  } catch {
2742
3516
  /* a drain failure must never break the run */
2743
3517
  }
@@ -2747,6 +3521,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2747
3521
  } catch {
2748
3522
  /* ignore */
2749
3523
  }
3524
+ // 0787 — read the run's cost AFTER done() (the terminal usage frame
3525
+ // usually arrives in the parser flush). Folded, not sent yet: the
3526
+ // report call is what carries it home.
3527
+ try {
3528
+ const spend = emitter.usage?.();
3529
+ if (spend) runUsage.push({ t: "usage", ...spend });
3530
+ } catch {
3531
+ /* accounting must never break the run */
3532
+ }
2750
3533
  }
2751
3534
  }
2752
3535
  if (run.aborted || signal?.aborted) {
@@ -2814,8 +3597,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2814
3597
  sessionId: sid,
2815
3598
  machine,
2816
3599
  // The CLI that created the session (a `ses_…` is meaningless to
2817
- // claude's `--resume`) and the project it belongs to — opencode
2818
- // scopes sessions per project. The gate above requires both (0608).
3600
+ // claude's `--resume`) and the project it belongs to — opencode and
3601
+ // codex both scope sessions per project. The gate above requires
3602
+ // both (0608, 0783).
2819
3603
  vendor,
2820
3604
  cwd: projectKey,
2821
3605
  updatedAt: new Date().toISOString(),
@@ -2826,6 +3610,32 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2826
3610
  }
2827
3611
  };
2828
3612
 
3613
+ // 0792 — ship the run's record, AFTER its report. Ordering is the whole
3614
+ // point: the transcript is a fact about finished work, and the card it marks
3615
+ // has to exist before it can be marked. Best-effort in every direction — no
3616
+ // tap, no stream, no server tool, or a refused upload all leave the run
3617
+ // exactly as it would have ended without this feature.
3618
+ let transcriptUploaded = false;
3619
+ const uploadTranscript = async (reportMsgId) => {
3620
+ if (!transcriptTap || !runId || transcriptUploaded) return;
3621
+ let text = "";
3622
+ try {
3623
+ text = transcriptTap.text();
3624
+ } catch {
3625
+ return;
3626
+ }
3627
+ // Nothing to say: a transport that never wrote to stdout (an ACP session, a
3628
+ // vendor with no stream) has no transcript, and an empty upload is worse
3629
+ // than none — it would put an empty object behind a Transcript control.
3630
+ if (!text) return;
3631
+ transcriptUploaded = true;
3632
+ await tool("upload_run_transcript", {
3633
+ runId,
3634
+ transcript: text,
3635
+ ...(reportMsgId ? { messageId: reportMsgId } : {}),
3636
+ }).catch(() => {});
3637
+ };
3638
+
2829
3639
  // Post a proposal card (approve-before-push mode) from a staged change.
2830
3640
  const postProposal = async (staged) => {
2831
3641
  const report = buildProposalReport({
@@ -2839,7 +3649,16 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2839
3649
  stat: staged.stat,
2840
3650
  runFailed: staged.runFailed,
2841
3651
  });
2842
- const res = await tool("post_report", { channelId, parentId: threadRoot, broadcast: true, ...report });
3652
+ const res = await tool("post_report", {
3653
+ channelId,
3654
+ parentId: threadRoot,
3655
+ broadcast: true,
3656
+ ...report,
3657
+ // 0787 — the proposal card is where a gated run's spend lands, because
3658
+ // it is the card the run produced. The later "Shipped" report carries
3659
+ // only what the CLI spent AFTER this one (usually nothing).
3660
+ ...reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
3661
+ });
2843
3662
  return { reportMessageId: res?.messageId ?? null, stat: staged.stat };
2844
3663
  };
2845
3664
 
@@ -2878,6 +3697,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2878
3697
  body,
2879
3698
  });
2880
3699
  await updateChannelMarker(compactRunMarker("cancelled", branch));
3700
+ // Every repo-lane cancel funnels through here, so one settle covers them
3701
+ // all: a stopped run still spent tokens (0787).
3702
+ await settleRunUsage();
2881
3703
  return { status: "cancelled", branch };
2882
3704
  };
2883
3705
 
@@ -2906,7 +3728,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2906
3728
 
2907
3729
  const codePrompt =
2908
3730
  (teamMemoryBlock ? teamMemoryBlock + "\n\n" : "") +
2909
- codeTaskPrompt({ message, context, brief: routed.task, repoFullName });
3731
+ codeTaskPrompt({ message, context, brief: routed.task, repoFullName, ...codeImagePromptArgs });
2910
3732
  let staged = await runAndStage(codePrompt);
2911
3733
  if (staged.aborted) return await postStopped();
2912
3734
  // Never-worse-than-today (0282): if we RESUMED a session and that run FAILED (a
@@ -2945,11 +3767,13 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2945
3767
  });
2946
3768
  await finalizeProgress(`\`${branch}\` has the agent's own commits — needs review.`);
2947
3769
  await updateChannelMarker(compactRunMarker("changes", branch));
3770
+ await settleRunUsage(); // no report on this exit (0787)
2948
3771
  return { status: "self-committed-gated", branch };
2949
3772
  }
2950
3773
  // When a live status card was streaming, settle the report onto it (0289) —
2951
3774
  // the card cross-fades run → report. Skip finalizeProgress then (it edits the
2952
3775
  // card's body, which the settle overwrites with the report anyway).
3776
+ if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
2953
3777
  const selfSettleId = streamOn && progressId ? progressId : null;
2954
3778
  if (!selfSettleId) await finalizeProgress(`Agent shipped \`${branch}\` itself — report below.`);
2955
3779
  const selfResult = await shipSelfDriven({
@@ -2964,8 +3788,10 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2964
3788
  parentId: threadRoot,
2965
3789
  settleId: selfSettleId,
2966
3790
  runId,
3791
+ usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
2967
3792
  });
2968
3793
  await recordSession(selfResult.prUrl);
3794
+ await uploadTranscript(selfResult.reportMsgId);
2969
3795
  await updateChannelMarker(compactRunMarker(selfResult.status, selfResult.branch));
2970
3796
  return selfResult;
2971
3797
  }
@@ -3003,6 +3829,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3003
3829
  await updateChannelMarker(
3004
3830
  compactRunMarker(staged.failed ? "run-failed" : "no-changes", branch),
3005
3831
  );
3832
+ await settleRunUsage(); // "no changes" is not "no spend" (0787)
3006
3833
  return { status: staged.failed ? "run-failed" : "no-changes" };
3007
3834
  }
3008
3835
 
@@ -3011,7 +3838,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3011
3838
  // propose-a-diff-and-wait flow for users who want approve-before-push.
3012
3839
  if (!cfg.gate) {
3013
3840
  // A "stop" that lands between the run finishing and the ship must still win.
3014
- if (signal?.aborted) return await postStopped();
3841
+ // The signal covers a stop we already heard; the server check covers one that
3842
+ // landed while the CLI was silent and nothing was carrying it home.
3843
+ if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
3015
3844
  // Seamless single card (0289): when a live status card streamed this run,
3016
3845
  // settle the report onto it in place instead of posting a separate report
3017
3846
  // message. Only the DEFAULT gate:false report settles in place; gate:true's
@@ -3035,6 +3864,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3035
3864
  existingPrUrl: continuingPrUrl,
3036
3865
  settleId,
3037
3866
  runId,
3867
+ usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
3038
3868
  });
3039
3869
  // The settle already flipped the card to the report (body + metadata); editing
3040
3870
  // the body again would clobber the report summary, so only finalize when we
@@ -3043,6 +3873,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3043
3873
  // Re-record with the now-known PR url so the local session record + the run row
3044
3874
  // carry the PR the session belongs to (best-effort; session id unchanged).
3045
3875
  await recordSession(result.prUrl);
3876
+ await uploadTranscript(result.reportMsgId);
3046
3877
  await updateChannelMarker(compactRunMarker(result.status, result.branch));
3047
3878
  return { ...result, stat: staged.stat, sessionId: runSessionId };
3048
3879
  }
@@ -3079,7 +3910,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3079
3910
  }
3080
3911
 
3081
3912
  // A stop during the final wait, too — don't ship a cancelled run.
3082
- if (signal?.aborted) return await postStopped();
3913
+ if (signal?.aborted || (await stoppedBeforeShip())) return await postStopped();
3083
3914
  const result = await applyDecision({
3084
3915
  decision,
3085
3916
  repoPath,
@@ -3095,9 +3926,15 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3095
3926
  parentId: threadRoot,
3096
3927
  existingPrUrl: continuingPrUrl,
3097
3928
  runId,
3929
+ // Normally {} — a gated run already reported its spend on the proposal card
3930
+ // it is now shipping. Anything the CLI spent after that still comes home.
3931
+ usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
3098
3932
  });
3099
3933
  await finalizeProgress(`Done on \`${branch}\` — see the report below.`);
3100
3934
  await recordSession(result.prUrl);
3935
+ // The gated lane's shipped report is a fresh message; `result.reportMsgId`
3936
+ // names it, and the proposal card is the fallback when it never posted.
3937
+ await uploadTranscript(result.reportMsgId || proposal.reportMessageId);
3101
3938
  await updateChannelMarker(compactRunMarker(result.status, result.branch));
3102
3939
  return { ...result, reportMessageId: proposal.reportMessageId, stat: proposal.stat, rounds: round, sessionId: runSessionId };
3103
3940
  } finally {