hilos-agent 0.11.13 → 0.11.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/handler.mjs CHANGED
@@ -1,3 +1,4 @@
1
+ import { runCodingTransport, codingChildEnv, codingChildWebEnv, codexRunIsGated } from "./coding-transport.mjs";
1
2
  import { githubArtifactPatch } from "./github-artifacts.mjs";
2
3
  import { AGENT_CODE_RESULT_RULE, AGENT_REPLY_STYLE_RULE, agentResultText } from "./agent-result.mjs";
3
4
  // The handler the published `hilos-agent` package ships and runs. Bias-to-PR by
@@ -28,16 +29,7 @@ import {
28
29
  detectPrContinuation,
29
30
  selfDrivenShipPlan,
30
31
  } from "./daemon.mjs";
31
- import {
32
- runCli,
33
- buildHeartbeat,
34
- ackText,
35
- oneLine,
36
- fmtElapsed,
37
- minimalEnv,
38
- scrubHilosEnv,
39
- envForCwd,
40
- } from "./cli.mjs";
32
+ import { runCli, buildHeartbeat, ackText, oneLine, fmtElapsed, envForCwd } from "./cli.mjs";
41
33
  import { commandArgv } from "./argv.mjs";
42
34
  import { makeStreamParser, createUsageFold } from "./agent-events.mjs";
43
35
  import {
@@ -47,21 +39,7 @@ import {
47
39
  MAX_DIRECTION_TURNS,
48
40
  readPendingInputs,
49
41
  } from "./run-inputs.mjs";
50
- import {
51
- detectVendor,
52
- chatWebArgs,
53
- codeStreamArgs,
54
- codeWebArgs,
55
- codeDirArgs,
56
- codeImageArgs,
57
- codeProjectKey,
58
- imagesNeedReading,
59
- attachTarget,
60
- createProgressEmitter,
61
- fastChatCmd,
62
- nativeWebPrompt,
63
- webToolEnv,
64
- } from "./progress-emitter.mjs";
42
+ import { detectVendor, chatWebArgs, codeStreamArgs, codeWebArgs, codeDirArgs, codeImageArgs, codeProjectKey, imagesNeedReading, createProgressEmitter, fastChatCmd, nativeWebPrompt } from "./progress-emitter.mjs";
65
43
  import { createTranscriptTap } from "./transcript.mjs";
66
44
  import { createLocalHarnessEmitter } from "./local-harness-transport.mjs";
67
45
  import { releaseDaemonIterateClaim } from "./iterate-claim-recovery.mjs";
@@ -86,25 +64,11 @@ import { buildMemoryBlock } from "./memory.mjs";
86
64
  import { deployFolder, resolveDeployTarget } from "./deploy.mjs";
87
65
  import { runOpenCodeHttpSession } from "./opencode-session.mjs";
88
66
  import { runAcpSession } from "./acp-session.mjs";
89
- import {
90
- shouldGateClaudePermissions,
91
- startClaudePermissionServer,
92
- } from "./claude-permissions.mjs";
93
- import {
94
- codexMcpTransportUnavailable,
95
- codexSandboxFromArgs,
96
- runCodexMcpSession,
97
- shouldGateCodexPermissions,
98
- } from "./codex-mcp-session.mjs";
99
- import {
100
- childMcpArgs,
101
- childMcpPlanned,
102
- childToolsNote,
103
- codexChildMcpArgs,
104
- startChildHilosMcp,
105
- } from "./child-mcp.mjs";
67
+ import { startClaudePermissionServer } from "./claude-permissions.mjs";
68
+ import { runCodexMcpSession } from "./codex-mcp-session.mjs";
69
+ import { childMcpPlanned, childToolsNote, startChildHilosMcp } from "./child-mcp.mjs";
106
70
  import { createUngatedRunNotice } from "./permission-gate.mjs";
107
- import { repoEditAllowance } from "./permission-gate.mjs";
71
+
108
72
  import { webMcpAgentPrompt } from "./webmcp-bridge.mjs";
109
73
  import { truncate } from "./util.mjs";
110
74
  import {
@@ -117,28 +81,6 @@ import {
117
81
  validConflictRecovery,
118
82
  } from "./conflicts.mjs";
119
83
 
120
- /**
121
- * The environment for a coding/chat CLI run. runCli always strips HILOS_* on top
122
- * of whatever this returns; here we additionally honor `codingEnv: "minimal"`
123
- * (strict isolation — only what the tool needs to reach its own model provider).
124
- * Returns undefined for the default "inherit" mode so runCli falls back to the
125
- * scrubbed process.env.
126
- */
127
- function codingChildEnv(cfg) {
128
- return cfg?.codingEnv === "minimal"
129
- ? minimalEnv(process.env, cfg.codingEnvAllow)
130
- : undefined;
131
- }
132
-
133
- /** Add only the selected CLI's public-web enablement to its already-isolated
134
- * child environment. Today this changes OpenCode only; keeping it shared makes
135
- * argv, ACP, HTTP, gated, and conversational runs follow one contract. */
136
- function codingChildWebEnv(cfg, vendor) {
137
- return webToolEnv(vendor, codingChildEnv(cfg), {
138
- enabled: cfg?.webSearch !== false,
139
- });
140
- }
141
-
142
84
  /**
143
85
  * The fast chat command for this config: an explicit chatCmd wins, else the
144
86
  * coding vendor's verified non-interactive print mode (fastChatCmd), else the
@@ -165,7 +107,6 @@ const reviewRounds = new Map();
165
107
  // (binary, tier) for the process; a failed lookup emits [] and retries later.
166
108
  const modelArgsFor = createModelArgsResolver({ run: (opts) => runCli(opts) });
167
109
 
168
-
169
110
  // Sentinel the router model emits (only when a thread already owns a run) to
170
111
  // classify a follow-up: change | new-scope | ambiguous. Parsed out of the router
171
112
  // output and stripped from the coding brief.
@@ -359,81 +300,6 @@ function defaultDeps() {
359
300
  };
360
301
  }
361
302
 
362
- /**
363
- * Bind one OpenCode HTTP session to hilos's vendor-neutral permission tools.
364
- * Both repo and direct-folder runs use this exact callback contract so neither
365
- * path can accidentally become the ungated exception.
366
- */
367
- function openCodePermissionCallbacks({
368
- tool,
369
- channelId,
370
- threadRoot,
371
- runId = null,
372
- provider = "opencode",
373
- // 0866 — when the server's get_permission_decision supports a long-poll
374
- // hold, each read blocks up to this many ms and answers the moment a human
375
- // decides, instead of the gate discovering the decision on its next 1s poll.
376
- // 0 (older servers) keeps the plain immediate read.
377
- decisionWaitMs = 0,
378
- // 1256 — the checkout this run edits; an edit under it is allowed on the
379
- // daemon without a card (the pull request is the gate). Empty means "no
380
- // local allowance": every ask still reaches the room.
381
- repoRoot = "",
382
- gateRepoEdits = false,
383
- }) {
384
- return {
385
- localAllow: (request) => repoEditAllowance(request, { repoRoot, gateRepoEdits }),
386
- requestPermission: async (request) => {
387
- const detail =
388
- request.metadata && typeof request.metadata === "object"
389
- ? request.metadata
390
- : {};
391
- const title = [
392
- detail.title,
393
- detail.description,
394
- detail.command,
395
- request.resources?.[0],
396
- ].find((value) => typeof value === "string" && value.trim());
397
- return tool("request_permission", {
398
- channelId,
399
- threadRootId: threadRoot,
400
- ...(runId ? { runId } : {}),
401
- provider,
402
- vendorSessionId: request.sessionId,
403
- vendorRequestId: request.vendorRequestId,
404
- action: request.action,
405
- title: title || `${request.action} permission`,
406
- resources: request.resources,
407
- suggestedSave: request.suggestedSave,
408
- metadata: detail,
409
- ...(request.source ? { source: request.source } : {}),
410
- });
411
- },
412
- getPermissionDecision: async (handle, { request, failClosed = false }) => {
413
- const requestId =
414
- handle &&
415
- typeof handle === "object" &&
416
- typeof handle.requestId === "string"
417
- ? handle.requestId
418
- : null;
419
- if (!requestId) {
420
- throw new Error(
421
- "hilos returned a pending permission without a request id",
422
- );
423
- }
424
- return tool("get_permission_decision", {
425
- requestId,
426
- provider,
427
- vendorSessionId: request.sessionId,
428
- vendorRequestId: request.vendorRequestId,
429
- ...(failClosed ? { failClosed: true } : {}),
430
- // Never on a failClosed settlement — that call exists to act NOW.
431
- ...(decisionWaitMs > 0 && !failClosed ? { waitMs: decisionWaitMs } : {}),
432
- });
433
- },
434
- };
435
- }
436
-
437
303
  /**
438
304
  * The post_progress sender both run lanes use, with the 0782 stop check folded
439
305
  * into the beat they already send.
@@ -591,256 +457,6 @@ export function createStopPoller({
591
457
  return { stop, tick };
592
458
  }
593
459
 
594
- /** The local HTTP bridge can own only a local OpenCode server. An explicit
595
- * `--attach` remains on OpenCode's CLI responder, which rejects unanswered asks
596
- * fail closed; taking over a remote server requires a separate authenticated
597
- * transport contract. Shared by repo and direct-folder runs. */
598
- export function shouldUseRuntimePermissionBridge({
599
- vendor,
600
- runtimePermissions,
601
- codeArgs,
602
- codingCmd,
603
- }) {
604
- return (
605
- vendor === "opencode" &&
606
- runtimePermissions === true &&
607
- !codeArgs.includes("--auto") &&
608
- attachTarget(codingCmd) === null
609
- );
610
- }
611
-
612
- /** ACP transport (0759; slice 1 opencode, slice 2 cursor). Only runs the
613
- * workspace already gates with runtime permissions qualify, and only when the
614
- * vendor's own auto-allow escape hatches are absent — over ACP, hilos must be
615
- * the one answering asks, so a run configured to never ask has nothing to gate
616
- * here.
617
- *
618
- * 0778 removed two limits that were never about correctness. `acpTransport`
619
- * now defaults ON (config.mjs) for the vendors whose adapter is proven, and a
620
- * RESUMED run no longer disqualifies: both live vendors advertise ACP's
621
- * `loadSession` capability, so acp-session.mjs resumes the session and keeps
622
- * raising cards. Setting `acpTransport: false` still opts a workspace out. */
623
- export function shouldUseAcpTransport({
624
- vendor,
625
- acpTransport,
626
- runtimePermissions,
627
- codeArgs,
628
- codingCmd,
629
- }) {
630
- if (acpTransport !== true) return false;
631
- if (runtimePermissions !== true) return false;
632
- if (vendor === "opencode") {
633
- return shouldUseRuntimePermissionBridge({
634
- vendor,
635
- runtimePermissions,
636
- codeArgs,
637
- codingCmd,
638
- });
639
- }
640
- if (vendor === "cursor") {
641
- // -f/--force is cursor's "allow everything" switch — the exact analog of
642
- // opencode's --auto.
643
- return !codeArgs.includes("-f") && !codeArgs.includes("--force");
644
- }
645
- return false;
646
- }
647
-
648
- /**
649
- * The env a gated claude run gets (0777).
650
- *
651
- * A permission card can legitimately wait on a person for minutes — a 150s wait
652
- * was verified live — so the CLI's own MCP tool timeout must not cut the ask
653
- * short before hilos's run deadline does. Everything else about the env is
654
- * unchanged (runCli still strips HILOS_* itself).
655
- */
656
- function claudeGateEnv(cfg) {
657
- const base = codingChildEnv(cfg) || process.env;
658
- const budget = Math.max(60_000, Number(cfg?.runTimeoutMs) || 0) + 60_000;
659
- return { ...base, MCP_TOOL_TIMEOUT: String(budget) };
660
- }
661
-
662
- /**
663
- * Run claude_code through its permission-prompt seam (0777).
664
- *
665
- * Deliberately NOT a new transport: the ordinary argv run is untouched — same
666
- * runCli, same resume args, same streaming — and the gate is two extra flags
667
- * plus a loopback MCP server that lives exactly as long as the run. The prompt
668
- * stays the final argument.
669
- */
670
- async function runClaudeGatedCli({
671
- deps,
672
- cfg,
673
- onGateDropped,
674
- cmd,
675
- codeArgs,
676
- prompt,
677
- cwd,
678
- signal,
679
- onData,
680
- permissionCallbacks,
681
- sessionId = "",
682
- resolveSessionId,
683
- beforeSpawn,
684
- childMcp = null,
685
- log = console,
686
- }) {
687
- const server = await deps.startClaudePermissionServer({
688
- ...permissionCallbacks,
689
- // 1255 — one config file for the run: the gate's permission server and the
690
- // run's own hilos seat, plus the allow entry that keeps the agent's room
691
- // tools off the permission card.
692
- ...(childMcp ? { extraServers: childMcp.servers, allowedTools: childMcp.allowedTools } : {}),
693
- sessionId,
694
- // A FRESH run has no session id to hand over: claude reveals its own in the
695
- // init frame of its stream, which the progress emitter is already folding.
696
- // Pulling from that snapshot means an ask carries the CLI's real session id
697
- // without a second parser — and hilos rejects an empty one outright.
698
- resolveSessionId,
699
- timeoutMs: cfg.runTimeoutMs,
700
- signal,
701
- log,
702
- });
703
- try {
704
- return await deps.runCli({
705
- cmd,
706
- args: [...codeArgs, ...server.args, prompt],
707
- cwd,
708
- timeoutMs: cfg.runTimeoutMs,
709
- label: "coding",
710
- signal,
711
- env: claudeGateEnv(cfg),
712
- onData,
713
- beforeSpawn,
714
- // 0785 — a CLI that rejects the gate flags is retried without them. This
715
- // fires BEFORE that ungated retry is spawned, so the room hears about the
716
- // downgrade first rather than after the fact.
717
- onPermissionGateDropped: onGateDropped,
718
- });
719
- } finally {
720
- try {
721
- await server.close();
722
- } catch {
723
- /* tearing the gate down must never fail a finished run */
724
- }
725
- }
726
- }
727
-
728
- /**
729
- * Will this codex run take the gated transport (0785)?
730
- *
731
- * Answerable before the run's argv exists, because the only input that can
732
- * change the answer is the operator's OWN command — every flag the daemon
733
- * appends later (model, dir, image, resume, stream) is ours and none of them is
734
- * the bypass tier. Which matters because the PROMPT is written before the
735
- * transport is chosen, and what the prompt may claim about images depends on it.
736
- */
737
- function codexRunIsGated(cfg, caps) {
738
- return shouldGateCodexPermissions({
739
- vendor: detectVendor(cfg?.codingCmd),
740
- runtimePermissions: caps?.runtimePermissions,
741
- codeArgs: commandArgv(String(cfg?.codingCmd || "")).slice(1),
742
- });
743
- }
744
-
745
- /**
746
- * Run codex through its gated transport, with an honest fallback (0785).
747
- *
748
- * A granted workspace ALWAYS gets `codex mcp-server` — that is what the grant
749
- * means, and it is the only codex transport that can ask (`codex exec` has no
750
- * approval channel at any flag combination, see codex-mcp-session.mjs). The one
751
- * thing the grant can't conjure is the subcommand itself: a codex old enough
752
- * not to have it never answers the MCP handshake, and before this the run just
753
- * failed. Now it degrades to the plain exec argv — the ungated run every codex
754
- * did before 0777, never worse.
755
- *
756
- * Two rules make that degrade safe. It happens ONLY on positive evidence that
757
- * the subcommand is absent (codexMcpTransportUnavailable — every other
758
- * pre-handshake failure is returned as the failed run it is, because a wrongly
759
- * failed run is recoverable and a wrongly ungated one is not). And the room is
760
- * told BEFORE the ungated child is spawned, so the warning arrives while the
761
- * run can still be stopped.
762
- *
763
- * The exec argv is the run's real one (model, images, resume, stream flags),
764
- * so the fallback is the same run the ungated lane would have made.
765
- */
766
- async function runCodexGatedSession({
767
- deps,
768
- cfg,
769
- onGateDropped,
770
- cmd,
771
- codeArgs,
772
- prompt,
773
- cwd,
774
- model,
775
- resumeThreadId = null,
776
- signal,
777
- onData,
778
- onEvent,
779
- beforeSpawn,
780
- permissionCallbacks,
781
- childMcp = null,
782
- }) {
783
- // 1255 — `codex mcp-server` takes the same global `-c` overrides `codex exec`
784
- // does (verified on codex-cli 0.153.4: `codex mcp-server --help` lists
785
- // `-c, --config <key=value>`), so the seat reaches both codex transports
786
- // through one argv fragment.
787
- const childMcpConfig = childMcp ? codexChildMcpArgs(childMcp.url) : [];
788
- if (typeof beforeSpawn === "function" && (await beforeSpawn()) === false) {
789
- return {
790
- status: null,
791
- stdout: "",
792
- stderr: "",
793
- aborted: true,
794
- authorityLost: true,
795
- error: new Error("execution authority lost before spawn"),
796
- };
797
- }
798
- const run = await deps.runCodexMcpSession({
799
- cmd,
800
- ...(childMcpConfig.length ? { serverArgs: ["mcp-server", ...childMcpConfig] } : {}),
801
- cwd,
802
- prompt,
803
- sandbox: codexSandboxFromArgs(codeArgs),
804
- // 0785 — the resolved tier (0783) or a hand-pinned id. The gated transport
805
- // took the account default before this, so a preset was silently ignored
806
- // exactly where the operator was most likely to have set one.
807
- model: model || null,
808
- webSearch: cfg.webSearch !== false,
809
- resumeThreadId,
810
- timeoutMs: cfg.runTimeoutMs,
811
- signal,
812
- env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
813
- onData,
814
- onEvent,
815
- beforeSpawn,
816
- ...permissionCallbacks,
817
- });
818
- if (!codexMcpTransportUnavailable(run)) return run;
819
- console.log(
820
- " code → this codex has no `mcp-server`; running WITHOUT the hilos permission gate",
821
- );
822
- // Warn FIRST — before a single ungated command can run.
823
- if (typeof onGateDropped === "function") {
824
- try {
825
- await onGateDropped();
826
- } catch {
827
- /* telling the room must never break the run */
828
- }
829
- }
830
- const fallback = await deps.runCli({
831
- cmd,
832
- args: [...codeArgs, ...childMcpConfig, prompt],
833
- cwd,
834
- timeoutMs: cfg.runTimeoutMs,
835
- label: "coding",
836
- signal,
837
- env: codingChildEnv(cfg),
838
- onData,
839
- beforeSpawn,
840
- });
841
- return { ...fallback, permissionGateDropped: true };
842
- }
843
-
844
460
  async function awaitDecision({ tool, reportMessageId, cfg, deps, signal }) {
845
461
  if (!reportMessageId) return { kind: "timeout" };
846
462
  const deadline = deps.now() + cfg.decisionTimeoutMs;
@@ -1740,9 +1356,9 @@ function askedYouTo(message) {
1740
1356
  * The server delivers it as a `kind: "answer"` mention row carrying
1741
1357
  * `askMessageId` — a reply in the thread where this agent asked. Without a lead
1742
1358
  * line the run reads it as a fresh instruction from a stranger; with one it
1743
- * reads as the reply it has been waiting for. Defensive by construction: an
1744
- * older server sends no `kind` at all, this returns "", and the prompt is
1745
- * exactly what it was.
1359
+ * reads as the reply it has been waiting for. Only answer delivery gets this
1360
+ * lead line; every other mention kind leaves the prompt unchanged. Missing
1361
+ * author names likewise produce no lead line.
1746
1362
  * @param {{ kind?: string, author?: string } | null} [message]
1747
1363
  */
1748
1364
  export function answeredYou(message) {
@@ -1976,11 +1592,11 @@ export async function proposePlanAck({
1976
1592
 
1977
1593
  /**
1978
1594
  * React to the wake's message instead of replying (0860). Best-effort and
1979
- * capability-gated: on an older server (no add_reaction) or any failure the
1980
- * agent simply stays quiet, which is what the model chose anyway.
1595
+ * injectable for tests. A write failure leaves the agent quiet, which is
1596
+ * what the model chose anyway.
1981
1597
  */
1982
- async function reactQuietly({ tool, caps, message, emoji }) {
1983
- if (!emoji || !caps?.react || !message?.id) return;
1598
+ async function reactQuietly({ tool, testCapabilities, message, emoji }) {
1599
+ if (!emoji || !testCapabilities?.react || !message?.id) return;
1984
1600
  await tool("add_reaction", { messageId: message.id, emoji }).catch(() => {});
1985
1601
  }
1986
1602
 
@@ -1996,7 +1612,7 @@ async function respondConversationally({
1996
1612
  workspaceMemory,
1997
1613
  signal,
1998
1614
  context,
1999
- caps = {},
1615
+ testCapabilities = {},
2000
1616
  runCliFn = runCli,
2001
1617
  pathExistsFn = existsSync,
2002
1618
  }) {
@@ -2004,8 +1620,8 @@ async function respondConversationally({
2004
1620
  const { transcript } = context || (await fetchContext({ channelId, tool, parentId }));
2005
1621
  // 1249 — the room can ask agents to answer in a thread instead of in the
2006
1622
  // room. The server owns the setting and marks the mention; the daemon only
2007
- // has to post under the message it names. An older server sends neither
2008
- // field, so this stays exactly `parentId` and nothing changes. The context
1623
+ // has to post under the message it names. Other mentions retain their
1624
+ // existing parent, including a top-level reply with no parent. The context
2009
1625
  // above still reads the CHANNEL tail: the ask happened in the room, and the
2010
1626
  // thread we are about to open has nothing in it yet.
2011
1627
  const replyParentId =
@@ -2074,7 +1690,7 @@ async function respondConversationally({
2074
1690
  let beatBusy = false;
2075
1691
  const beatMs = Math.min(cfg.chatTimeoutMs || 90000, 30000);
2076
1692
  const beat =
2077
- caps.editMessage && beatMs > 0
1693
+ testCapabilities.editMessage && beatMs > 0
2078
1694
  ? setInterval(async () => {
2079
1695
  if (beatStopped || beatBusy || thinkingId) return;
2080
1696
  beatBusy = true;
@@ -2131,7 +1747,7 @@ async function respondConversationally({
2131
1747
  if (thinkingId) {
2132
1748
  await deliver(quiet.emoji || "👍");
2133
1749
  } else {
2134
- await reactQuietly({ tool, caps, message, emoji: quiet.emoji });
1750
+ await reactQuietly({ tool, testCapabilities, message, emoji: quiet.emoji });
2135
1751
  }
2136
1752
  return;
2137
1753
  }
@@ -2582,7 +2198,7 @@ async function handleFolderDeploy({
2582
2198
  * live), request-changes = re-run in place (bounded), reject = revert the run's
2583
2199
  * own changes (git repos only).
2584
2200
  */
2585
- async function handleFolderTask({ message, channelId, tool, caps, cfg, deps, signal, parentId, folderPath, brief, workspaceMemory, context, onStopRequested }) {
2201
+ async function handleFolderTask({ message, channelId, tool, testCapabilities, cfg, deps, signal, parentId, folderPath, brief, workspaceMemory, context, onStopRequested }) {
2586
2202
  const git = deps.git;
2587
2203
  // 0779 — screenshots the poll loop already pulled to a temp dir at pickup.
2588
2204
  // The prompt names them; the argv carries them for a vendor that takes one.
@@ -2591,7 +2207,7 @@ async function handleFolderTask({ message, channelId, tool, caps, cfg, deps, sig
2591
2207
  ? {
2592
2208
  images: localImages,
2593
2209
  imagesReadable: imagesNeedReading(detectVendor(cfg.codingCmd), {
2594
- gated: codexRunIsGated(cfg, caps),
2210
+ gated: codexRunIsGated(cfg, testCapabilities),
2595
2211
  }),
2596
2212
  }
2597
2213
  : {};
@@ -2693,7 +2309,7 @@ async function handleFolderTask({ message, channelId, tool, caps, cfg, deps, sig
2693
2309
  const runFolderCli = async (promptText) => {
2694
2310
  const parts = commandArgv(cfg.codingCmd);
2695
2311
  const vendor = detectVendor(cfg.codingCmd);
2696
- const streamOn = Boolean(caps.postProgress);
2312
+ const streamOn = Boolean(testCapabilities.postProgress);
2697
2313
  const streamArgs = streamOn ? codeStreamArgs(vendor) : [];
2698
2314
  let emitter = null;
2699
2315
  // Every dispatched progress write, so the settle can wait them out (0289).
@@ -2746,7 +2362,7 @@ async function handleFolderTask({ message, channelId, tool, caps, cfg, deps, sig
2746
2362
  intervalMs: cfg.stopPollMs || STOP_POLL_MS,
2747
2363
  });
2748
2364
  }
2749
- } else if (caps.editMessage && cfg.heartbeatMs > 0) {
2365
+ } else if (testCapabilities.editMessage && cfg.heartbeatMs > 0) {
2750
2366
  const start = deps.now();
2751
2367
  let stopped = false;
2752
2368
  let busy = false;
@@ -2775,7 +2391,6 @@ async function handleFolderTask({ message, channelId, tool, caps, cfg, deps, sig
2775
2391
  }
2776
2392
  let run;
2777
2393
  // 1255 — the run's own hilos seat, closed in the same finally as the run.
2778
- let childMcp = null;
2779
2394
  // Model preset (0504): same run-time resolution as the repo path.
2780
2395
  const modelArgs = await modelArgsFor(cfg, vendor);
2781
2396
  // 0787: the id we actually handed the CLI, kept for the usage receipt when
@@ -2834,195 +2449,51 @@ async function handleFolderTask({ message, channelId, tool, caps, cfg, deps, sig
2834
2449
  /* accounting must never break the run */
2835
2450
  }
2836
2451
  };
2837
- try {
2838
- const useAcpTransport = shouldUseAcpTransport({
2839
- vendor,
2840
- acpTransport: cfg.acpTransport,
2841
- runtimePermissions: caps.runtimePermissions,
2842
- codeArgs,
2843
- codingCmd: cfg.codingCmd,
2844
- });
2845
- const useRuntimePermissionBridge =
2846
- !useAcpTransport &&
2847
- shouldUseRuntimePermissionBridge({
2848
- vendor,
2849
- runtimePermissions: caps.runtimePermissions,
2850
- codeArgs,
2851
- codingCmd: cfg.codingCmd,
2852
- });
2853
- // 0777: the two vendors that had no gate at all. Neither is an ACP
2854
- // adapter — each CLI turned out to have its own native seam (see the
2855
- // module headers), so both compose with everything already here.
2856
- const gateCodexPermissions =
2857
- !useAcpTransport &&
2858
- !useRuntimePermissionBridge &&
2859
- shouldGateCodexPermissions({
2860
- vendor,
2861
- runtimePermissions: caps.runtimePermissions,
2862
- codeArgs,
2863
- });
2864
- const gateClaudePermissions =
2865
- !useAcpTransport &&
2866
- !useRuntimePermissionBridge &&
2867
- !gateCodexPermissions &&
2868
- shouldGateClaudePermissions({
2869
- vendor,
2870
- runtimePermissions: caps.runtimePermissions,
2871
- codeArgs,
2872
- });
2873
- // 1255 — the agent's seat inside its own run. Turn-scoped: it opens here
2874
- // and closes in the same finally the run does, so no loopback outlives
2875
- // the work it was opened for. Claude and codex only; every other vendor
2876
- // takes the identical argv it took before.
2877
- childMcp = await deps.startChildHilosMcp({
2878
- cfg,
2879
- vendor,
2880
- bindingClaim: message?.bindingClaim,
2881
- // The gate writes the one config when it is on; otherwise this writes it.
2882
- standaloneConfig: vendor === "claude_code" && !gateClaudePermissions,
2883
- });
2884
- if (useAcpTransport) {
2885
- run = await deps.runAcpSession({
2886
- cmd: parts[0],
2887
- vendor,
2888
- cwd: folderPath,
2889
- prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2890
- timeoutMs: cfg.runTimeoutMs,
2891
- signal,
2892
- env: scrubHilosEnv(codingChildWebEnv(cfg, vendor) || process.env),
2893
- onData: handleCliData,
2894
- onEvent: handleCliEvent,
2895
- ...openCodePermissionCallbacks({
2896
- tool,
2897
- channelId,
2898
- threadRoot,
2899
- decisionWaitMs: caps.decisionWaitMs,
2900
- provider: vendor,
2901
- }),
2902
- });
2903
- } else if (useRuntimePermissionBridge) {
2904
- run = await deps.runOpenCodeHttpSession({
2905
- cmd: parts[0],
2906
- args: codeArgs,
2907
- cwd: folderPath,
2908
- prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2909
- timeoutMs: cfg.runTimeoutMs,
2910
- signal,
2911
- env: scrubHilosEnv(codingChildWebEnv(cfg, vendor) || process.env),
2912
- onData: handleCliData,
2913
- ...openCodePermissionCallbacks({
2914
- tool,
2915
- channelId,
2916
- threadRoot,
2917
- decisionWaitMs: caps.decisionWaitMs,
2918
- }),
2919
- });
2920
- } else if (gateCodexPermissions) {
2921
- run = await runCodexGatedSession({
2922
- deps,
2923
- cfg,
2924
- onGateDropped: () => noticeUngatedRun(parts[0]),
2925
- cmd: parts[0],
2926
- codeArgs,
2927
- cwd: folderPath,
2928
- prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2929
- model: resolvedModelId,
2930
- signal,
2931
- onData: handleCliData,
2932
- onEvent: handleCliEvent,
2933
- childMcp,
2934
- permissionCallbacks: openCodePermissionCallbacks({
2935
- tool,
2936
- channelId,
2937
- threadRoot,
2938
- decisionWaitMs: caps.decisionWaitMs,
2939
- provider: vendor,
2940
- repoRoot: folderPath,
2941
- gateRepoEdits: cfg.gateRepoEdits === true,
2942
- }),
2943
- });
2944
- } else if (gateClaudePermissions) {
2945
- run = await runClaudeGatedCli({
2946
- deps,
2947
- cfg,
2948
- onGateDropped: () => noticeUngatedRun(parts[0]),
2949
- cmd: parts[0],
2950
- codeArgs,
2951
- prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2952
- cwd: folderPath,
2953
- signal,
2954
- onData: handleCliData,
2955
- resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
2956
- childMcp,
2957
- permissionCallbacks: openCodePermissionCallbacks({
2958
- tool,
2959
- channelId,
2960
- threadRoot,
2961
- decisionWaitMs: caps.decisionWaitMs,
2962
- provider: vendor,
2963
- repoRoot: folderPath,
2964
- gateRepoEdits: cfg.gateRepoEdits === true,
2965
- }),
2966
- });
2967
- } else {
2968
- run = await deps.runCli({
2969
- cmd: parts[0],
2970
- args: [
2971
- ...codeArgs,
2972
- ...childMcpArgs(vendor, childMcp),
2973
- agentPreamble(workspaceMemory, cfg) + promptText,
2974
- ],
2975
- cwd: folderPath,
2976
- timeoutMs: cfg.runTimeoutMs,
2977
- label: "coding",
2978
- signal,
2979
- env: codingChildWebEnv(cfg, vendor),
2980
- onData: handleCliData,
2981
- });
2982
- }
2983
- // 0785 backstop. The notice normally goes out from `onGateDropped`,
2984
- // BEFORE the ungated child is spawned — the room has to be warned while
2985
- // the run can still be stopped, not told afterwards what it already did.
2986
- // This catches a degrade that reached us without firing that hook; the
2987
- // latch makes it a no-op in the ordinary case.
2988
- if (run?.permissionGateDropped) await noticeUngatedRun(parts[0]);
2989
- } finally {
2990
- // 1255 — the seat dies with the turn. A loopback that outlived its run
2991
- // would be an open credential proxy nobody is using.
2992
- try {
2993
- await childMcp?.close();
2994
- } catch {
2995
- /* tearing the seat down must never fail a finished run */
2996
- }
2997
- childMcp = null;
2998
- stopHeartbeat();
2999
- stopPoller?.stop();
3000
- if (emitter) {
3001
- const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
3002
- // Remembered for the accounting-only settlement (0787): an exit with no
3003
- // report re-sends this same terminal state, so the card is never
3004
- // repainted into something it wasn't.
3005
- lastTerminalState = errored ? "error" : "done";
3006
- let reason = "";
3007
- if (errored && run && !run.aborted) {
3008
- const stderrTail = oneLine((run.stderr || "").trim().split("\n").slice(-3).join(" "), 200);
3009
- if (run.error?.code === "ENOENT") reason = `Couldn't start ${parts[0]} — is it installed and on PATH?`;
3010
- else if (run.error?.message) reason = run.error.message;
3011
- else if (run.status != null) reason = `The CLI exited ${run.status}`;
3012
- if (stderrTail) reason = reason ? `${reason}: ${stderrTail}` : stderrTail;
3013
- }
3014
- try {
3015
- emitter.done(errored ? "error" : "done", reason);
3016
- } catch {
3017
- /* ignore */
3018
- }
3019
- try {
3020
- await drainProgress();
3021
- } catch {
3022
- /* a drain failure must never break the run */
2452
+ run = await runCodingTransport({
2453
+ deps, cfg, cmd: parts[0], vendor, codeArgs, cwd: folderPath,
2454
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2455
+ model: resolvedModelId,
2456
+ runtimePermissions: testCapabilities.runtimePermissions,
2457
+ permissionCallbacks: { tool, channelId, threadRoot, decisionWaitMs: testCapabilities.decisionWaitMs },
2458
+ childMcp: ({ standaloneConfig }) => deps.startChildHilosMcp({
2459
+ cfg, vendor, bindingClaim: message?.bindingClaim, standaloneConfig,
2460
+ }),
2461
+ onData: handleCliData,
2462
+ onEvent: handleCliEvent,
2463
+ signal,
2464
+ onGateDropped: () => noticeUngatedRun(parts[0]),
2465
+ resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
2466
+ onSettle: async (settledRun) => {
2467
+ run = settledRun;
2468
+ stopHeartbeat();
2469
+ stopPoller?.stop();
2470
+ if (emitter) {
2471
+ const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
2472
+ // Remembered for the accounting-only settlement (0787): an exit with no
2473
+ // report re-sends this same terminal state, so the card is never
2474
+ // repainted into something it wasn't.
2475
+ lastTerminalState = errored ? "error" : "done";
2476
+ let reason = "";
2477
+ if (errored && run && !run.aborted) {
2478
+ const stderrTail = oneLine((run.stderr || "").trim().split("\n").slice(-3).join(" "), 200);
2479
+ if (run.error?.code === "ENOENT") reason = `Couldn't start ${parts[0]} — is it installed and on PATH?`;
2480
+ else if (run.error?.message) reason = run.error.message;
2481
+ else if (run.status != null) reason = `The CLI exited ${run.status}`;
2482
+ if (stderrTail) reason = reason ? `${reason}: ${stderrTail}` : stderrTail;
2483
+ }
2484
+ try {
2485
+ emitter.done(errored ? "error" : "done", reason);
2486
+ } catch {
2487
+ /* ignore */
2488
+ }
2489
+ try {
2490
+ await drainProgress();
2491
+ } catch {
2492
+ /* a drain failure must never break the run */
2493
+ }
3023
2494
  }
3024
- }
3025
- }
2495
+ },
2496
+ });
3026
2497
  if (resultParser) {
3027
2498
  try {
3028
2499
  foldResultEvents(resultParser.flush());
@@ -3253,7 +2724,9 @@ async function handleFolderTask({ message, channelId, tool, caps, cfg, deps, sig
3253
2724
 
3254
2725
  /** Handle one task. cfg/deps injectable for tests. `opts.signal` (AbortSignal)
3255
2726
  * cancels an in-flight run — the queue fires it when a human says "stop". */
3256
- export async function handleTask({ message, channelId, tool, me, caps = {}, iterateClaimRecoveryStore = null }, cfg, depsOverride, opts = {}) {
2727
+ // 1290: run() supplies the current hilos.sh contract. testCapabilities is the
2728
+ // missing-tool fixture seam for direct handler tests, not server negotiation.
2729
+ export async function handleTask({ message, channelId, tool, me, testCapabilities = {}, canRecall = true, iterateClaimRecoveryStore = null }, cfg, depsOverride, opts = {}) {
3257
2730
  if (message.dispatch?.briefMarkdown) {
3258
2731
  message = { ...message, body: `${message.body}\n\n${message.dispatch.briefMarkdown}` };
3259
2732
  }
@@ -3311,9 +2784,8 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3311
2784
 
3312
2785
  // REVIEW EXECUTION (0288): a review-request routes to the read-only reviewer
3313
2786
  // BEFORE the chat/code flow — it reads the PR's diff and posts an advisory review
3314
- // that @-tags the author. Gated on caps.review (post_review + get_pr_diff on this
3315
- // server); without it (older deploy) this is skipped and behavior is unchanged.
3316
- if (caps.review && isReviewRequest(message)) {
2787
+ // that @-tags the author. Missing-tool fixtures can disable this branch.
2788
+ if (testCapabilities.review && isReviewRequest(message)) {
3317
2789
  return await reviewTask({ message, channelId, tool, me, cfg, workspaceMemory, context, signal });
3318
2790
  }
3319
2791
 
@@ -3323,7 +2795,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3323
2795
  // agent sees — then fall through to the normal follow-up machinery, which (0281)
3324
2796
  // checks out the thread's PR branch and pushes onto the SAME human-gated PR. Best-
3325
2797
  // effort; a failure just leaves the summary line to drive the iterate.
3326
- if (caps.review && message?.kind === "review") {
2798
+ if (testCapabilities.review && message?.kind === "review") {
3327
2799
  const target =
3328
2800
  message?.reviewOf && (message.reviewOf.prUrl || message.reviewOf.messageId)
3329
2801
  ? message.reviewOf
@@ -3354,7 +2826,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3354
2826
  mode,
3355
2827
  // A manager-routed dispatch is a brief to implement, not an instruction to
3356
2828
  // land something — its body is assembled text, never a person's sentence.
3357
- canRelay: Boolean(caps.prActions) && !message.dispatch && !forcedConflictRecovery,
2829
+ canRelay: Boolean(testCapabilities.prActions) && !message.dispatch && !forcedConflictRecovery,
3358
2830
  });
3359
2831
  if (relay.relay) {
3360
2832
  const verb = relay.action;
@@ -3443,7 +2915,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3443
2915
  routed = { code: false, reply: null };
3444
2916
  } else if (mode === "ship") {
3445
2917
  routed = { code: true, task: message.body };
3446
- } else if (caps.agentIntent) {
2918
+ } else if (testCapabilities.agentIntent) {
3447
2919
  const classifierDeployTarget = deps.resolveDeployTarget({
3448
2920
  cfg,
3449
2921
  channelId,
@@ -3516,7 +2988,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3516
2988
  return { status: "chat" };
3517
2989
  }
3518
2990
  if (routed.noReply) {
3519
- await reactQuietly({ tool, caps, message, emoji: routed.emoji });
2991
+ await reactQuietly({ tool, testCapabilities, message, emoji: routed.emoji });
3520
2992
  return { status: "quiet" };
3521
2993
  }
3522
2994
  if (routed.code) {
@@ -3524,7 +2996,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3524
2996
  message,
3525
2997
  channelId,
3526
2998
  tool,
3527
- caps,
2999
+ testCapabilities,
3528
3000
  cfg,
3529
3001
  deps,
3530
3002
  signal,
@@ -3550,7 +3022,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3550
3022
  workspaceMemory,
3551
3023
  signal,
3552
3024
  context,
3553
- caps,
3025
+ testCapabilities,
3554
3026
  runCliFn: deps.runCli,
3555
3027
  pathExistsFn: deps.pathExists,
3556
3028
  });
@@ -3575,7 +3047,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3575
3047
  workspaceMemory,
3576
3048
  signal,
3577
3049
  context,
3578
- caps,
3050
+ testCapabilities,
3579
3051
  runCliFn: deps.runCli,
3580
3052
  pathExistsFn: deps.pathExists,
3581
3053
  });
@@ -3585,10 +3057,10 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3585
3057
  // Same-PR follow-up (0281): if this mention is a reply in a thread that already
3586
3058
  // owns a run (branch + PR), we may continue THAT run instead of forking a
3587
3059
  // duplicate. Look it up by the thread root — the parentId the reply carried.
3588
- // Best-effort + gated on the server supporting the runs entity (caps.runs); any
3060
+ // Best-effort + injectable for tests (testCapabilities.runs); any
3589
3061
  // failure falls back to a fresh run (mode 'new'), never breaking the task.
3590
3062
  let activeRun = null;
3591
- if (caps.runs && parentId) {
3063
+ if (testCapabilities.runs && parentId) {
3592
3064
  const r = await tool("get_active_run", { channelId, threadRootId: parentId }).catch(() => null);
3593
3065
  if (r && r.found) activeRun = r;
3594
3066
  }
@@ -3635,7 +3107,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3635
3107
  }
3636
3108
  if (!routed.code) {
3637
3109
  if (routed.noReply) {
3638
- await reactQuietly({ tool, caps, message, emoji: routed.emoji });
3110
+ await reactQuietly({ tool, testCapabilities, message, emoji: routed.emoji });
3639
3111
  return { status: "quiet" };
3640
3112
  }
3641
3113
  await respondConversationally({
@@ -3649,7 +3121,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3649
3121
  workspaceMemory,
3650
3122
  signal,
3651
3123
  context,
3652
- caps,
3124
+ testCapabilities,
3653
3125
  runCliFn: deps.runCli,
3654
3126
  pathExistsFn: deps.pathExists,
3655
3127
  });
@@ -3919,8 +3391,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3919
3391
  }
3920
3392
 
3921
3393
  // Instant acknowledgement: post a template in <1s so the channel shows life
3922
- // before any model call. If the server supports edit_message and a fast chatCmd
3923
- // is set, a bounded plan-ack edits in a sentence of specifics ("I'll do X…").
3394
+ // before any model call. If a fast chatCmd is set, a bounded plan-ack edits in a sentence of specifics ("I'll do X…").
3924
3395
  // Best-effort — a slow or failed ack just leaves the template, never blocks.
3925
3396
  //
3926
3397
  // INSPECTABLE ACK (0281): when we're continuing or forking off an existing run,
@@ -3962,8 +3433,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3962
3433
  // redirect → supersede the old run (so a future follow-up no longer points at
3963
3434
  // it), then record a FRESH run for the new, separate PR.
3964
3435
  // new → record a fresh run, exactly as before.
3965
- // Best-effort throughout — a bookkeeping failure (or an older server without the
3966
- // runs tools) must NEVER break the run, so it proceeds without a runId.
3436
+ // A bookkeeping failure must never break the run, so it proceeds without a runId.
3967
3437
  let runId = message.dispatch?.runId ?? (effectiveMode === "iterate" ? activeRun?.runId ?? null : null);
3968
3438
  // Immutable generation for every report attempt in this daemon turn. A reused
3969
3439
  // run claims the next value atomically below; fresh and dispatch-adopted rows
@@ -3975,11 +3445,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
3975
3445
  : 0;
3976
3446
  let iterateClaimId = null;
3977
3447
  let iterateAuthorityLost = false;
3978
- if (message.dispatch?.runId && caps.runs) {
3448
+ if (message.dispatch?.runId && testCapabilities.runs) {
3979
3449
  // Adopt the dispatch's canonical run. If the server already SETTLED it (the
3980
3450
  // reaper/outbox closed a dispatch we were slow to pick up), leave it alone
3981
3451
  // and skip the coding run — resurrecting it would ship work no one is
3982
- // waiting on. Any OTHER failure (network, an older server without the guard)
3452
+ // waiting on. Any OTHER failure (network or the guard)
3983
3453
  // keeps today's best-effort behavior: proceed with the run.
3984
3454
  try {
3985
3455
  const adopted = await tool("update_run", { runId: message.dispatch.runId, status: "running" });
@@ -4002,7 +3472,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4002
3472
  };
4003
3473
  }
4004
3474
  }
4005
- if (caps.runs && effectiveMode === "iterate" && runId && !message.dispatch?.runId) {
3475
+ if (testCapabilities.runs && effectiveMode === "iterate" && runId && !message.dispatch?.runId) {
4006
3476
  try {
4007
3477
  const claimed = await claimDaemonIterate({
4008
3478
  tool,
@@ -4029,7 +3499,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4029
3499
  return { status: "iterate-claim-lost", runId };
4030
3500
  }
4031
3501
  }
4032
- if (caps.runs && threadRoot && effectiveMode !== "iterate" && !message.dispatch?.runId) {
3502
+ if (testCapabilities.runs && threadRoot && effectiveMode !== "iterate" && !message.dispatch?.runId) {
4033
3503
  try {
4034
3504
  if (effectiveMode === "redirect" && activeRun?.runId) {
4035
3505
  await tool("update_run", { runId: activeRun.runId, status: "superseded" }).catch(() => {});
@@ -4055,7 +3525,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4055
3525
  // `pendingInputs` off its heartbeat and runs one more CLI turn with
4056
3526
  // them after the coding turn. Declared only when the server can take
4057
3527
  // the receipt, so the room never offers a composer nothing will read.
4058
- ...(caps.runInputs ? { inputTransport: "next_turn" } : {}),
3528
+ ...(testCapabilities.runInputs ? { inputTransport: "next_turn" } : {}),
4059
3529
  });
4060
3530
  runId = r?.runId ?? null;
4061
3531
  if (Number.isSafeInteger(r?.iterateTurn)) iterateTurn = Number(r.iterateTurn);
@@ -4067,7 +3537,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4067
3537
  // the server knows (start_run answered) and only when it can take the
4068
3538
  // receipt; otherwise there is no door and nothing is ever listed.
4069
3539
  const directions =
4070
- caps.runInputs && runId
3540
+ testCapabilities.runInputs && runId
4071
3541
  ? createDirectionInbox({ tool, runId, log: (line) => console.log(line) })
4072
3542
  : null;
4073
3543
  // All subprocesses launched after an Iterate claim — including the fast plan
@@ -4086,7 +3556,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4086
3556
  return ownsClaim && !signal?.aborted;
4087
3557
  };
4088
3558
  const settleConflictFailure = async (reason) => {
4089
- if (!activeConflictMerge || !runId || !caps.runs) return;
3559
+ if (!activeConflictMerge || !runId || !testCapabilities.runs) return;
4090
3560
  await tool("update_run", {
4091
3561
  runId,
4092
3562
  status: effectiveMode === "iterate" ? "awaiting_review" : "failed",
@@ -4097,7 +3567,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4097
3567
  // an LLM plan) so a human can correct the routing before code lands.
4098
3568
  if (
4099
3569
  ackId &&
4100
- caps.editMessage &&
3570
+ testCapabilities.editMessage &&
4101
3571
  chatCmdFor(cfg) &&
4102
3572
  !continuingPrUrl &&
4103
3573
  !anchorNote &&
@@ -4121,11 +3591,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4121
3591
  // then EDIT it on later beats — alive without thread spam. Needs edit_message;
4122
3592
  // without it we skip rather than post a fresh message every beat. lastLine
4123
3593
  // carries the CLI's latest output into the beat.
4124
- // Live progress (0274): when the server supports post_progress, the code run
4125
- // STREAMS a coalesced "what I'm doing right now" card onto one thread status
4126
- // message (via the 0272 stream parser) instead of the 3-min edit heartbeat. On
4127
- // older servers (no post_progress) we fall back to the exact legacy behavior.
4128
- const streamOn = Boolean(caps.postProgress);
3594
+ // Live progress (0274) coalesces activity onto one thread status message
3595
+ // through the 0272 parser. Tests can inject the heartbeat-only fixture.
3596
+ const streamOn = Boolean(testCapabilities.postProgress);
4129
3597
  const vendor = detectVendor(cfg.codingCmd);
4130
3598
  // 1255 — same pre-run answer as the folder lane, for the same reason.
4131
3599
  const hilosTools = childMcpPlanned({
@@ -4141,14 +3609,14 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4141
3609
  // revision rounds all land in one transcript — the run is the unit, not the
4142
3610
  // spawn.
4143
3611
  const transcriptTap =
4144
- cfg.uploadTranscripts && caps.uploadTranscript && runId ? createTranscriptTap() : null;
3612
+ cfg.uploadTranscripts && testCapabilities.uploadTranscript && runId ? createTranscriptTap() : null;
4145
3613
  // 0779 — screenshots the poll loop pulled to a temp dir at pickup. The prompt
4146
3614
  // names them; the argv carries them for a vendor with a verified image flag.
4147
3615
  const localImages = Array.isArray(message?.images) ? message.images : [];
4148
3616
  const codeImagePromptArgs = localImages.length
4149
3617
  ? {
4150
3618
  images: localImages,
4151
- imagesReadable: imagesNeedReading(vendor, { gated: codexRunIsGated(cfg, caps) }),
3619
+ imagesReadable: imagesNeedReading(vendor, { gated: codexRunIsGated(cfg, testCapabilities) }),
4152
3620
  }
4153
3621
  : {};
4154
3622
  // 0785 — one line per run, whichever CLI turns out to be ungateable.
@@ -4206,7 +3674,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4206
3674
  }
4207
3675
  }
4208
3676
 
4209
- const heartbeatOn = Boolean(caps.editMessage) && cfg.heartbeatMs > 0 && !streamOn;
3677
+ const heartbeatOn = Boolean(testCapabilities.editMessage) && cfg.heartbeatMs > 0 && !streamOn;
4210
3678
  let lastLine = "";
4211
3679
  let progressId = null; // the thread progress/status reply, once one has posted
4212
3680
  const startHeartbeat = () => {
@@ -4282,7 +3750,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4282
3750
  // sit there claiming the agent is alive. No-op when no beat ever fired (short
4283
3751
  // run) or without edit_message. Best-effort.
4284
3752
  const finalizeProgress = async (text) => {
4285
- if (progressId && caps.editMessage) {
3753
+ if (progressId && testCapabilities.editMessage) {
4286
3754
  await tool("edit_message", { messageId: progressId, body: text }).catch(() => {});
4287
3755
  }
4288
3756
  };
@@ -4437,7 +3905,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4437
3905
  // automatic but deliberately best-effort: it is bounded, secret-redacted
4438
3906
  // again on the server, and can never interrupt the coding process.
4439
3907
  const harnessEmitter =
4440
- caps.hepEvents &&
3908
+ testCapabilities.hepEvents &&
4441
3909
  runId &&
4442
3910
  me?.agentId &&
4443
3911
  ["claude_code", "codex", "cursor", "opencode"].includes(vendor)
@@ -4497,255 +3965,89 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4497
3965
  };
4498
3966
  let run;
4499
3967
  // 1255 — the run's own hilos seat, closed in the same finally as the run.
4500
- let childMcp = null;
4501
- try {
4502
- // OpenCode's own non-interactive CLI auto-rejects every permission ask
4503
- // unless --auto is present. For the gated tiers, bypass that responder
4504
- // and own the authenticated HTTP session + SSE stream directly so hilos
4505
- // is the sole authority answering the paused tool call (0593).
4506
- const useAcpTransport = shouldUseAcpTransport({
4507
- vendor,
4508
- acpTransport: cfg.acpTransport,
4509
- runtimePermissions: caps.runtimePermissions,
4510
- codeArgs,
4511
- codingCmd: cfg.codingCmd,
4512
- });
4513
- const useRuntimePermissionBridge =
4514
- !useAcpTransport &&
4515
- shouldUseRuntimePermissionBridge({
4516
- vendor,
4517
- runtimePermissions: caps.runtimePermissions,
4518
- codeArgs,
4519
- codingCmd: cfg.codingCmd,
4520
- });
4521
- // 0777, and the reason it composes with resume (0778): claude keeps its
4522
- // ordinary argv run (resumeArgs included), and codex's gated transport
4523
- // has its OWN resume — `codex-reply {threadId}` — so a gated iterate
4524
- // continues the same thread instead of trading continuity for a gate.
4525
- const gateCodexPermissions =
4526
- !useAcpTransport &&
4527
- !useRuntimePermissionBridge &&
4528
- shouldGateCodexPermissions({
4529
- vendor,
4530
- runtimePermissions: caps.runtimePermissions,
4531
- codeArgs,
4532
- });
4533
- const gateClaudePermissions =
4534
- !useAcpTransport &&
4535
- !useRuntimePermissionBridge &&
4536
- !gateCodexPermissions &&
4537
- shouldGateClaudePermissions({
4538
- vendor,
4539
- runtimePermissions: caps.runtimePermissions,
4540
- codeArgs,
4541
- });
4542
- // 1255 — the agent's seat inside its own run, on the same two vendors and
4543
- // with the same turn-scoped lifetime as the folder lane. Every other
4544
- // vendor's argv is byte-identical to what it was.
4545
- childMcp = await deps.startChildHilosMcp({
4546
- cfg,
4547
- vendor,
4548
- bindingClaim: message?.bindingClaim,
4549
- standaloneConfig: vendor === "claude_code" && !gateClaudePermissions,
4550
- });
4551
- if (useAcpTransport) {
4552
- run = await deps.runAcpSession({
4553
- cmd: parts[0],
4554
- vendor,
4555
- cwd: repoPath,
4556
- prompt: agentPreamble(workspaceMemory, cfg) + promptText,
4557
- // 0778: approvals AND continuity. `resume:false` (the never-worse
4558
- // retry) drops it exactly like buildResumeArgs does.
4559
- resumeSessionId: resume ? resumeSessionId : null,
4560
- timeoutMs: cfg.runTimeoutMs,
4561
- signal,
4562
- env: scrubHilosEnv(codingChildWebEnv(cfg, vendor) || process.env),
4563
- onData: handleCliData,
4564
- onEvent: handleCliEvent,
4565
- beforeSpawn: beforeCodingSpawn,
4566
- ...openCodePermissionCallbacks({
4567
- tool,
4568
- channelId,
4569
- threadRoot,
4570
- decisionWaitMs: caps.decisionWaitMs,
4571
- runId,
4572
- provider: vendor,
4573
- }),
4574
- });
4575
- } else if (useRuntimePermissionBridge) {
4576
- run = await deps.runOpenCodeHttpSession({
4577
- cmd: parts[0],
4578
- args: codeArgs,
4579
- cwd: repoPath,
4580
- prompt: agentPreamble(workspaceMemory, cfg) + promptText,
4581
- timeoutMs: cfg.runTimeoutMs,
4582
- signal,
4583
- // A model running inside the server must never inherit the daemon's
4584
- // hilos bearer token. The bridge's random loopback password is added
4585
- // after this scrub and dies with the process group.
4586
- env: scrubHilosEnv(codingChildWebEnv(cfg, vendor) || process.env),
4587
- onData: handleCliData,
4588
- beforeSpawn: beforeCodingSpawn,
4589
- ...openCodePermissionCallbacks({
4590
- tool,
4591
- channelId,
4592
- threadRoot,
4593
- decisionWaitMs: caps.decisionWaitMs,
4594
- runId,
4595
- }),
4596
- });
4597
- } else if (gateCodexPermissions) {
4598
- run = await runCodexGatedSession({
4599
- deps,
4600
- cfg,
4601
- onGateDropped: () => noticeUngatedRun(parts[0]),
4602
- cmd: parts[0],
4603
- codeArgs,
4604
- cwd: repoPath,
4605
- prompt: agentPreamble(workspaceMemory, cfg) + promptText,
4606
- model: resolvedModelId,
4607
- // Codex's own resume over this transport. `resume:false` (the
4608
- // never-worse retry) drops it exactly like buildResumeArgs does.
4609
- resumeThreadId: resume ? resumeSessionId : null,
4610
- signal,
4611
- onData: handleCliData,
4612
- onEvent: handleCliEvent,
4613
- beforeSpawn: beforeCodingSpawn,
4614
- childMcp,
4615
- permissionCallbacks: openCodePermissionCallbacks({
4616
- tool,
4617
- channelId,
4618
- threadRoot,
4619
- decisionWaitMs: caps.decisionWaitMs,
4620
- runId,
4621
- provider: vendor,
4622
- repoRoot: repoPath,
4623
- gateRepoEdits: cfg.gateRepoEdits === true,
4624
- }),
4625
- });
4626
- } else if (gateClaudePermissions) {
4627
- run = await runClaudeGatedCli({
4628
- deps,
4629
- cfg,
4630
- onGateDropped: () => noticeUngatedRun(parts[0]),
4631
- cmd: parts[0],
4632
- // codeArgs already carries this run's resume flags, so a gated
4633
- // iterate resumes AND raises cards — the two never traded off.
4634
- codeArgs,
4635
- prompt: agentPreamble(workspaceMemory, cfg) + promptText,
4636
- cwd: repoPath,
4637
- signal,
4638
- onData: handleCliData,
4639
- beforeSpawn: beforeCodingSpawn,
4640
- sessionId: resume ? resumeSessionId ?? "" : "",
4641
- resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
4642
- childMcp,
4643
- permissionCallbacks: openCodePermissionCallbacks({
4644
- tool,
4645
- channelId,
4646
- threadRoot,
4647
- decisionWaitMs: caps.decisionWaitMs,
4648
- runId,
4649
- provider: vendor,
4650
- repoRoot: repoPath,
4651
- gateRepoEdits: cfg.gateRepoEdits === true,
4652
- }),
4653
- });
4654
- } else {
4655
- // deps.runCli, not the bare import: `defaultDeps` wraps the very same
4656
- // function, so production is byte-identical, but the ungated repo run
4657
- // was the ONE coding path that escaped the injectable runner — which is
4658
- // why nothing above the unit tests could ever drive it (0787).
4659
- run = await deps.runCli({
4660
- cmd: parts[0],
4661
- args: [
4662
- ...codeArgs,
4663
- ...childMcpArgs(vendor, childMcp),
4664
- agentPreamble(workspaceMemory, cfg) + promptText,
4665
- ],
4666
- cwd: repoPath,
4667
- timeoutMs: cfg.runTimeoutMs,
4668
- label: "coding",
4669
- signal,
4670
- env: codingChildWebEnv(cfg, vendor),
4671
- onData: handleCliData,
4672
- beforeSpawn: beforeCodingSpawn,
4673
- });
4674
- }
4675
- // 0785 — the gate was expected here and the CLI couldn't hold it. Say so
4676
- // in the room, once per run, rather than only in the daemon's console.
4677
- if (run?.permissionGateDropped) await noticeUngatedRun(parts[0]);
4678
- } finally {
4679
- // 1255 — the seat dies with the turn (see the folder lane).
4680
- try {
4681
- await childMcp?.close();
4682
- } catch {
4683
- /* tearing the seat down must never fail a finished run */
4684
- }
4685
- childMcp = null;
4686
- stopHeartbeat();
4687
- stopPoller?.stop();
4688
- if (run?.sessionId) runSessionId = run.sessionId;
4689
- try {
4690
- await harnessEmitter?.drain();
4691
- } catch {
4692
- /* runtime exhaust must never break the run */
4693
- }
4694
- if (emitter) {
4695
- // Terminal state: flip the status card off "working" (state 'done'/'error')
4696
- // so it stops claiming the agent is alive. For a run that produced work,
4697
- // the gate:false path then SETTLES the report onto this same card in place
4698
- // (0289) — the card cross-fades run → report; for no-changes/failed/gate
4699
- // outcomes this terminal 'done'/'error' is the card's final state.
4700
- const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
4701
- // Remembered so an exit with no report re-sends this same state (0787)
4702
- // rather than repainting the card into something it wasn't.
4703
- lastTerminalState = errored ? "error" : "done";
4704
- // On error, carry an honest reason onto the card (0294) — the same signal
4705
- // the chat note uses: spawn error message, else exit status, plus a short
4706
- // stderr tail. Sanitized again server-side; an old server just drops it.
4707
- let reason = "";
4708
- if (errored && run && !run.aborted) {
4709
- const stderrTail = oneLine((run.stderr || "").trim().split("\n").slice(-3).join(" "), 200);
4710
- if (run.error?.code === "ENOENT") {
4711
- reason = `Couldn't start ${parts[0]} — is it installed and on PATH?`;
4712
- } else if (run.error?.message) {
4713
- reason = run.error.message;
4714
- } else if (run.status != null) {
4715
- reason = `The CLI exited ${run.status}`;
4716
- }
4717
- if (stderrTail) reason = reason ? `${reason}: ${stderrTail}` : stderrTail;
4718
- }
4719
- try {
4720
- emitter.done(errored ? "error" : "done", reason);
4721
- } catch {
4722
- /* ignore */
4723
- }
4724
- // Wait for every progress write (including the terminal one above) to land
4725
- // BEFORE runAndStage returns, so nothing is in flight when the caller
4726
- // settles the report onto this card — no lost-update clobber (0289).
4727
- try {
4728
- await drainProgress();
4729
- } catch {
4730
- /* a drain failure must never break the run */
4731
- }
3968
+ run = await runCodingTransport({
3969
+ deps, cfg, cmd: parts[0], vendor, codeArgs, cwd: repoPath,
3970
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
3971
+ model: resolvedModelId,
3972
+ runtimePermissions: testCapabilities.runtimePermissions,
3973
+ permissionCallbacks: { tool, channelId, threadRoot, decisionWaitMs: testCapabilities.decisionWaitMs },
3974
+ childMcp: ({ standaloneConfig }) => deps.startChildHilosMcp({
3975
+ cfg, vendor, bindingClaim: message?.bindingClaim, standaloneConfig,
3976
+ }),
3977
+ onData: handleCliData,
3978
+ onEvent: handleCliEvent,
3979
+ signal,
3980
+ onGateDropped: () => noticeUngatedRun(parts[0]),
3981
+ resolveSessionId: () => emitter?.snapshot()?.sessionId ?? null,
3982
+ beforeSpawn: beforeCodingSpawn,
3983
+ resume: { sessionId: resume ? resumeSessionId : null },
3984
+ runId,
3985
+ onSettle: async (settledRun) => {
3986
+ run = settledRun;
3987
+ stopHeartbeat();
3988
+ stopPoller?.stop();
3989
+ if (run?.sessionId) runSessionId = run.sessionId;
4732
3990
  try {
4733
- const snap = emitter.snapshot();
4734
- if (snap && snap.sessionId) runSessionId = snap.sessionId;
3991
+ await harnessEmitter?.drain();
4735
3992
  } catch {
4736
- /* ignore */
3993
+ /* runtime exhaust must never break the run */
4737
3994
  }
4738
- // 0787 — read the run's cost AFTER done() (the terminal usage frame
4739
- // usually arrives in the parser flush). Folded, not sent yet: the
4740
- // report call is what carries it home.
4741
- try {
4742
- const spend = emitter.usage?.();
4743
- if (spend) runUsage.push({ t: "usage", ...spend });
4744
- } catch {
4745
- /* accounting must never break the run */
3995
+ if (emitter) {
3996
+ // Terminal state: flip the status card off "working" (state 'done'/'error')
3997
+ // so it stops claiming the agent is alive. For a run that produced work,
3998
+ // the gate:false path then SETTLES the report onto this same card in place
3999
+ // (0289) — the card cross-fades run → report; for no-changes/failed/gate
4000
+ // outcomes this terminal 'done'/'error' is the card's final state.
4001
+ const errored = Boolean(run && (run.aborted || run.error || run.status !== 0));
4002
+ // Remembered so an exit with no report re-sends this same state (0787)
4003
+ // rather than repainting the card into something it wasn't.
4004
+ lastTerminalState = errored ? "error" : "done";
4005
+ // On error, carry an honest reason onto the card (0294) — the same signal
4006
+ // the chat note uses: spawn error message, else exit status, plus a short
4007
+ // stderr tail. Sanitized again server-side.
4008
+ let reason = "";
4009
+ if (errored && run && !run.aborted) {
4010
+ const stderrTail = oneLine((run.stderr || "").trim().split("\n").slice(-3).join(" "), 200);
4011
+ if (run.error?.code === "ENOENT") {
4012
+ reason = `Couldn't start ${parts[0]} — is it installed and on PATH?`;
4013
+ } else if (run.error?.message) {
4014
+ reason = run.error.message;
4015
+ } else if (run.status != null) {
4016
+ reason = `The CLI exited ${run.status}`;
4017
+ }
4018
+ if (stderrTail) reason = reason ? `${reason}: ${stderrTail}` : stderrTail;
4019
+ }
4020
+ try {
4021
+ emitter.done(errored ? "error" : "done", reason);
4022
+ } catch {
4023
+ /* ignore */
4024
+ }
4025
+ // Wait for every progress write (including the terminal one above) to land
4026
+ // BEFORE runAndStage returns, so nothing is in flight when the caller
4027
+ // settles the report onto this card — no lost-update clobber (0289).
4028
+ try {
4029
+ await drainProgress();
4030
+ } catch {
4031
+ /* a drain failure must never break the run */
4032
+ }
4033
+ try {
4034
+ const snap = emitter.snapshot();
4035
+ if (snap && snap.sessionId) runSessionId = snap.sessionId;
4036
+ } catch {
4037
+ /* ignore */
4038
+ }
4039
+ // 0787 — read the run's cost AFTER done() (the terminal usage frame
4040
+ // usually arrives in the parser flush). Folded, not sent yet: the
4041
+ // report call is what carries it home.
4042
+ try {
4043
+ const spend = emitter.usage?.();
4044
+ if (spend) runUsage.push({ t: "usage", ...spend });
4045
+ } catch {
4046
+ /* accounting must never break the run */
4047
+ }
4746
4048
  }
4747
- }
4748
- }
4049
+ },
4050
+ });
4749
4051
  if (!resultText && emitter) {
4750
4052
  try {
4751
4053
  resultText = emitter.snapshot()?.lastLine || "";
@@ -4988,7 +4290,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
4988
4290
  await updateChannelMarker(compactRunMarker("cancelled", branch));
4989
4291
  // 1270 — a process interruption has no server-side Stop action to settle
4990
4292
  // the row. Iterate claims keep their fenced recovery path in run().
4991
- if (runId && caps.runs && !iterateClaimId) {
4293
+ if (runId && testCapabilities.runs && !iterateClaimId) {
4992
4294
  await tool("update_run", { runId, status: "failed", reason: "daemon-stopped" }).catch(() => {});
4993
4295
  }
4994
4296
  // Every repo-lane cancel funnels through here, so one settle covers them
@@ -5023,12 +4325,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
5023
4325
 
5024
4326
  // Team memory recall (0297): best-effort fetch accumulated learnings for this
5025
4327
  // channel before the coding run so the coding agent has the team's conventions
5026
- // and gotchas from the start. Capability-gated: older servers without the
5027
- // `recall` tool degrade silently (caps.recall=false → skip, no change in
5028
- // behavior). A recall failure — network hiccup, server error — NEVER fails the
5029
- // task; it is caught and the run proceeds without the block.
4328
+ // and gotchas from the start. Channel-bound tokens cannot read workspace
4329
+ // memory (0513); tests can also omit recall. A failed read leaves the task
4330
+ // running without the memory block.
5030
4331
  let teamMemoryBlock = "";
5031
- if (caps.recall) {
4332
+ if (canRecall && testCapabilities.recall) {
5032
4333
  try {
5033
4334
  // Pass the task's channelId so the server scopes recall to this project AND
5034
4335
  // can apply the guest gate (0298): the daemon's token is long-lived with no
@@ -5275,7 +4576,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
5275
4576
  if (directions) await directions.close();
5276
4577
  // A terminal progress card is presentation; it does not settle this row.
5277
4578
  // Keep an existing PR at its review gate when feedback produces no diff.
5278
- if (runId && caps.runs && !iterateClaimId) {
4579
+ if (runId && testCapabilities.runs && !iterateClaimId) {
5279
4580
  await tool("update_run", {
5280
4581
  runId,
5281
4582
  status: continuingPrUrl ? "awaiting_review" : staged.failed ? "failed" : "succeeded",
@@ -5299,7 +4600,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {}, iter
5299
4600
  // settle the report onto it in place instead of posting a separate report
5300
4601
  // message. Only the DEFAULT gate:false report settles in place; gate:true's
5301
4602
  // proposal → decision → report lifecycle stays on separate messages (a
5302
- // separate ticket). No status card (older server / status post failed) →
4603
+ // separate ticket). No status card (status post failed or missing-tool fixture) →
5303
4604
  // settleId is null and applyDecision posts a fresh final report instead.
5304
4605
  const settleId = streamOn && progressId ? progressId : null;
5305
4606
  // 1182 — the inbox closes BEFORE the report: the report ends the pass on