hilos-agent 0.9.4 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/handler.mjs CHANGED
@@ -41,6 +41,7 @@ import { makeStreamParser, createUsageFold } from "./agent-events.mjs";
41
41
  import {
42
42
  detectVendor,
43
43
  codeStreamArgs,
44
+ codeWebArgs,
44
45
  codeDirArgs,
45
46
  codeImageArgs,
46
47
  codeProjectKey,
@@ -82,6 +83,7 @@ import {
82
83
  shouldGateCodexPermissions,
83
84
  } from "./codex-mcp-session.mjs";
84
85
  import { createUngatedRunNotice } from "./permission-gate.mjs";
86
+ import { webMcpAgentPrompt } from "./webmcp-bridge.mjs";
85
87
 
86
88
  /**
87
89
  * The environment for a coding/chat CLI run. runCli always strips HILOS_* on top
@@ -306,6 +308,11 @@ function openCodePermissionCallbacks({
306
308
  threadRoot,
307
309
  runId = null,
308
310
  provider = "opencode",
311
+ // 0866 — when the server's get_permission_decision supports a long-poll
312
+ // hold, each read blocks up to this many ms and answers the moment a human
313
+ // decides, instead of the gate discovering the decision on its next 1s poll.
314
+ // 0 (older servers) keeps the plain immediate read.
315
+ decisionWaitMs = 0,
309
316
  }) {
310
317
  return {
311
318
  requestPermission: async (request) => {
@@ -352,6 +359,8 @@ function openCodePermissionCallbacks({
352
359
  vendorSessionId: request.sessionId,
353
360
  vendorRequestId: request.vendorRequestId,
354
361
  ...(failClosed ? { failClosed: true } : {}),
362
+ // Never on a failClosed settlement — that call exists to act NOW.
363
+ ...(decisionWaitMs > 0 && !failClosed ? { waitMs: decisionWaitMs } : {}),
355
364
  });
356
365
  },
357
366
  };
@@ -768,7 +777,37 @@ async function linkPrFromUrl({ tool, channelId, url }) {
768
777
  await tool("link_pr", { channelId, repoFullName: m[1], prNumber: Number(m[2]) }).catch(() => {});
769
778
  }
770
779
 
771
- async function applyDecision({ decision, repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, existingPrUrl, settleId, runId, usageArgs = {} }) {
780
+ function githubProvenance({ cfg, me, message, channelId, runId, summary }) {
781
+ let siteUrl = "https://hilos.sh";
782
+ try {
783
+ const parsed = new URL(cfg.url);
784
+ parsed.pathname = parsed.pathname.replace(/\/api\/mcp\/?$/, "");
785
+ parsed.search = "";
786
+ parsed.hash = "";
787
+ siteUrl = parsed.toString().replace(/\/$/, "");
788
+ } catch {
789
+ /* the body helper will keep links absent if configuration is malformed */
790
+ }
791
+ const roomUrl = `${siteUrl}/w/${me?.workspaceId}/c/${channelId}`;
792
+ const messageUrl = message?.id
793
+ ? message.parentId
794
+ ? `${roomUrl}?thread=${message.parentId}&reply=${message.id}`
795
+ : `${roomUrl}#m-${message.id}`
796
+ : undefined;
797
+ return {
798
+ agentName: me?.agentName,
799
+ agentId: me?.agentId,
800
+ personName: message?.author,
801
+ roomName: message?.channel || "project room",
802
+ roomUrl,
803
+ messageId: message?.id,
804
+ messageUrl,
805
+ runId,
806
+ summary,
807
+ };
808
+ }
809
+
810
+ async function applyDecision({ decision, repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, existingPrUrl, settleId, runId, provenance = {}, usageArgs = {} }) {
772
811
  const tag = requesterTag(requester);
773
812
  const lead = tag ? `${tag} — ` : "";
774
813
  // `parentId` here is the run's thread root. Terminal outcomes ask the server
@@ -785,7 +824,7 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
785
824
  });
786
825
  return { status: "nothing-staged", branch };
787
826
  }
788
- const commit = deps.git(repoPath, ["commit", "-m", commitMessage(task)]);
827
+ const commit = deps.git(repoPath, ["commit", "-m", commitMessage(task, provenance)]);
789
828
  if (commit.status !== 0) {
790
829
  await tool("post_message", {
791
830
  channelId,
@@ -805,7 +844,7 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
805
844
  });
806
845
  return { status: "push-failed", branch };
807
846
  }
808
- const { title, body } = prTitleBody(task, branch);
847
+ const { title, body } = prTitleBody(task, branch, provenance);
809
848
  // Continuing an existing PR (request-changes rework): the push already
810
849
  // updated it — reuse its URL instead of opening a duplicate. Otherwise open
811
850
  // a fresh PR.
@@ -897,7 +936,7 @@ async function applyDecision({ decision, repoPath, branch, task, requester, cfg,
897
936
  * switched to `main` would otherwise make `gh` try to open main → main. Used
898
937
  * only when NOT gated (bias-to-action).
899
938
  */
900
- async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, settleId, runId, usageArgs = {} }) {
939
+ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, channelId, deps, parentId, settleId, runId, provenance = {}, usageArgs = {} }) {
901
940
  const tag = requesterTag(requester);
902
941
  const lead = tag ? `${tag} — ` : "";
903
942
  const currentBranch =
@@ -907,7 +946,7 @@ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, ch
907
946
  let prUrl = deps.findPR ? deps.findPR(repoPath, ship.headBranch) : null;
908
947
  let prAttempt = null;
909
948
  if (!prUrl && push.status === 0) {
910
- const { title, body } = prTitleBody(task, ship.headBranch);
949
+ const { title, body } = prTitleBody(task, ship.headBranch, provenance);
911
950
  prAttempt = deps.openPR(repoPath, {
912
951
  title,
913
952
  body,
@@ -916,7 +955,7 @@ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, ch
916
955
  });
917
956
  prUrl = prAttempt.ok && prAttempt.url ? prAttempt.url : null;
918
957
  }
919
- const { title } = prTitleBody(task, ship.headBranch);
958
+ const { title } = prTitleBody(task, ship.headBranch, provenance);
920
959
  const recoveredNote = ship.recoveredFrom
921
960
  ? ` The child had switched to \`${ship.recoveredFrom}\`; hilos recovered its HEAD onto \`${ship.headBranch}\`.`
922
961
  : "";
@@ -991,6 +1030,7 @@ async function shipSelfDriven({ repoPath, branch, task, requester, cfg, tool, ch
991
1030
  // Sentinel the router model emits when the latest message is a request to change
992
1031
  // code. Unusual on purpose so it can't be confused with a real chat reply.
993
1032
  const CODE_SIGNAL = "__CODE__";
1033
+ const CHAT_SIGNAL = "__CHAT__";
994
1034
 
995
1035
  // 0860 — the quiet option, offered ONLY on untagged DM wakes (message.implicit
996
1036
  // === "dm"). In a DM every message reaches the agent without a tag, so some of
@@ -1046,8 +1086,8 @@ export function dmJudgmentBlock(implicitDm) {
1046
1086
 
1047
1087
  /**
1048
1088
  * Decide — with the LLM, not a word list — whether the latest message wants a
1049
- * code change or a conversational reply, and produce the payload in the SAME
1050
- * call. The model reads the whole conversation, so it judges by intent and
1089
+ * code change or a conversational reply. The model reads the whole
1090
+ * conversation, so it judges by intent and
1051
1091
  * context, in any language: "just code it", "dale, hazlo", "yeah go for it" after
1052
1092
  * a request → code; "how does this work?", "thoughts?", "thanks" → chat.
1053
1093
  *
@@ -1059,12 +1099,12 @@ export function dmJudgmentBlock(implicitDm) {
1059
1099
  * Returns one of:
1060
1100
  * { aborted: true }
1061
1101
  * { code: true, task } → run the coding flow with `task` as the spec
1062
- * { code: false, reply, error } → post `reply` as a chat message
1102
+ * { code: false, error } → run the separate conversational responder
1063
1103
  *
1064
1104
  * `error` is set when the model produced nothing (so the caller can be honest
1065
1105
  * about a timeout vs a missing binary instead of inventing a reply).
1066
1106
  */
1067
- async function routeIntent({ name, repoFullName, transcript, workspaceMemory, cfg, signal, hasActiveRun = false, runCliFn, implicitDm = false }) {
1107
+ export async function routeIntent({ name, repoFullName, transcript, workspaceMemory, cfg, signal, hasActiveRun = false, runCliFn, implicitDm = false }) {
1068
1108
  const doRun = runCliFn || runCli; // folder mode injects deps.runCli; repo flow uses the import
1069
1109
  const cmd = chatCmdFor(cfg);
1070
1110
  const parts = cmd.split(" ").filter(Boolean);
@@ -1093,12 +1133,12 @@ async function routeIntent({ name, repoFullName, transcript, workspaceMemory, cf
1093
1133
  `using the conversation. Example:\n${CODE_SIGNAL}\nAdd a hover popover to message reactions ` +
1094
1134
  `that lists who reacted with each emoji.\n\n` +
1095
1135
  `Otherwise — a question, a greeting, general discussion, or they explicitly don't want code ` +
1096
- `yet — just reply to them concisely and directly as a single chat message (no preamble, no ` +
1097
- `headings). When unsure, stay conversational; implementation and PR flow are for clear ` +
1098
- `action language.` +
1136
+ `yet — output ONLY the token ${CHAT_SIGNAL}. When unsure, choose ${CHAT_SIGNAL}; ` +
1137
+ `implementation and PR flow are for clear action language.` +
1099
1138
  `${followupBlock}\n\n` +
1100
- `You are only routing here — output ONLY your text response. Do NOT use any tools, do NOT ` +
1101
- `edit files, do NOT run commands; a separate step does the actual coding.` +
1139
+ `You are only routing here — output ONLY the routing token (plus the imperative spec for ` +
1140
+ `${CODE_SIGNAL}). Do NOT use any tools, edit files, or run commands; separate tool-capable ` +
1141
+ `steps write the conversational answer or do the coding.` +
1102
1142
  `${dmJudgmentBlock(implicitDm)}\n\n` +
1103
1143
  `${memoryPreamble(workspaceMemory)}Conversation so far:\n${transcript}`;
1104
1144
  const run = await doRun({
@@ -1134,7 +1174,10 @@ async function routeIntent({ name, repoFullName, transcript, workspaceMemory, cf
1134
1174
  const quiet = parseNoReply(out);
1135
1175
  if (quiet.noReply) return { code: false, noReply: true, emoji: quiet.emoji, followupSignal };
1136
1176
  }
1137
- return { code: false, reply: out, followupSignal };
1177
+ // A classifier never authors the final answer. Even if a model ignores the
1178
+ // token contract and emits prose, discard it and let the normal responder —
1179
+ // with the vendor's real tools — answer the room.
1180
+ return { code: false, reply: null, followupSignal };
1138
1181
  }
1139
1182
 
1140
1183
  /**
@@ -1221,6 +1264,11 @@ function memoryPreamble(workspaceMemory) {
1221
1264
  return m ? `Workspace context (the project's soul):\n${m}\n\n` : "";
1222
1265
  }
1223
1266
 
1267
+ /** Shared context for a model that may act, not the read-only intent router. */
1268
+ function agentPreamble(workspaceMemory, cfg) {
1269
+ return memoryPreamble(workspaceMemory) + webMcpAgentPrompt(cfg);
1270
+ }
1271
+
1224
1272
  /**
1225
1273
  * Recent conversation the agent should reply within: the thread it was pinged in
1226
1274
  * (threads are where conversations live), else the channel tail. Returns the raw
@@ -1326,7 +1374,7 @@ async function respondConversationally({ message, channelId, tool, me, cfg, repo
1326
1374
  `concisely and directly as a single chat message — no preamble, no headings. ` +
1327
1375
  `${repoLine}` +
1328
1376
  `${dmJudgmentBlock(implicitDm)}\n\n` +
1329
- `${memoryPreamble(workspaceMemory)}` +
1377
+ `${agentPreamble(workspaceMemory, cfg)}` +
1330
1378
  `Conversation so far:\n${transcript}`;
1331
1379
 
1332
1380
  // Chat uses the FAST one-shot command (the coding vendor's own print mode when
@@ -2048,8 +2096,10 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2048
2096
  // 0779: [] for every vendor without a verified image flag — their argv is
2049
2097
  // byte-identical to before, and the prompt note still names the files.
2050
2098
  const imageArgs = codeImageArgs(vendor, localImages);
2099
+ const webArgs = codeWebArgs(vendor, parts.slice(1));
2051
2100
  const codeArgs = [
2052
2101
  ...parts.slice(1),
2102
+ ...webArgs,
2053
2103
  ...modelArgs,
2054
2104
  ...dirArgs,
2055
2105
  ...imageArgs,
@@ -2131,7 +2181,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2131
2181
  cmd: parts[0],
2132
2182
  vendor,
2133
2183
  cwd: folderPath,
2134
- prompt: memoryPreamble(workspaceMemory) + promptText,
2184
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2135
2185
  timeoutMs: cfg.runTimeoutMs,
2136
2186
  signal,
2137
2187
  env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
@@ -2141,6 +2191,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2141
2191
  tool,
2142
2192
  channelId,
2143
2193
  threadRoot,
2194
+ decisionWaitMs: caps.decisionWaitMs,
2144
2195
  provider: vendor,
2145
2196
  }),
2146
2197
  });
@@ -2149,7 +2200,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2149
2200
  cmd: parts[0],
2150
2201
  args: codeArgs,
2151
2202
  cwd: folderPath,
2152
- prompt: memoryPreamble(workspaceMemory) + promptText,
2203
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2153
2204
  timeoutMs: cfg.runTimeoutMs,
2154
2205
  signal,
2155
2206
  env: scrubHilosEnv(codingChildEnv(cfg) || process.env),
@@ -2158,6 +2209,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2158
2209
  tool,
2159
2210
  channelId,
2160
2211
  threadRoot,
2212
+ decisionWaitMs: caps.decisionWaitMs,
2161
2213
  }),
2162
2214
  });
2163
2215
  } else if (gateCodexPermissions) {
@@ -2168,7 +2220,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2168
2220
  cmd: parts[0],
2169
2221
  codeArgs,
2170
2222
  cwd: folderPath,
2171
- prompt: memoryPreamble(workspaceMemory) + promptText,
2223
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2172
2224
  model: resolvedModelId,
2173
2225
  signal,
2174
2226
  onData: handleCliData,
@@ -2177,6 +2229,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2177
2229
  tool,
2178
2230
  channelId,
2179
2231
  threadRoot,
2232
+ decisionWaitMs: caps.decisionWaitMs,
2180
2233
  provider: vendor,
2181
2234
  }),
2182
2235
  });
@@ -2187,7 +2240,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2187
2240
  onGateDropped: () => noticeUngatedRun(parts[0]),
2188
2241
  cmd: parts[0],
2189
2242
  codeArgs,
2190
- prompt: memoryPreamble(workspaceMemory) + promptText,
2243
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
2191
2244
  cwd: folderPath,
2192
2245
  signal,
2193
2246
  onData: handleCliData,
@@ -2196,6 +2249,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2196
2249
  tool,
2197
2250
  channelId,
2198
2251
  threadRoot,
2252
+ decisionWaitMs: caps.decisionWaitMs,
2199
2253
  provider: vendor,
2200
2254
  }),
2201
2255
  });
@@ -2204,7 +2258,7 @@ async function handleFolderTask({ message, channelId, tool, me, caps, cfg, deps,
2204
2258
  cmd: parts[0],
2205
2259
  args: [
2206
2260
  ...codeArgs,
2207
- memoryPreamble(workspaceMemory) + promptText,
2261
+ agentPreamble(workspaceMemory, cfg) + promptText,
2208
2262
  ],
2209
2263
  cwd: folderPath,
2210
2264
  timeoutMs: cfg.runTimeoutMs,
@@ -2715,12 +2769,8 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2715
2769
  onStopRequested,
2716
2770
  });
2717
2771
  }
2718
- // Not a coding task → post the router's reply if it produced one, else fall
2719
- // through to a normal conversational reply.
2720
- if (routed.reply) {
2721
- await tool("post_message", { channelId, parentId, body: routed.reply });
2722
- return { status: "chat" };
2723
- }
2772
+ // Not a coding task → the classifier is finished. The separate responder
2773
+ // below owns the answer and keeps its normal tools.
2724
2774
  }
2725
2775
  await respondConversationally({ message, channelId, tool, me, cfg, repoLink, parentId, workspaceMemory, signal, context, caps });
2726
2776
  return { status: "chat" };
@@ -2780,14 +2830,19 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
2780
2830
  await reactQuietly({ tool, caps, message, emoji: routed.emoji });
2781
2831
  return { status: "quiet" };
2782
2832
  }
2783
- let body = routed.reply;
2784
- if (!body) {
2785
- body =
2786
- routed.error?.code === "ENOENT"
2787
- ? `(my chat command \`${chatCmdFor(cfg)}\` isn't installed or on PATH.)`
2788
- : `Still thinking on this — it's taking longer than usual. I'll follow up shortly.`;
2789
- }
2790
- await tool("post_message", { channelId, parentId, body });
2833
+ await respondConversationally({
2834
+ message,
2835
+ channelId,
2836
+ tool,
2837
+ me,
2838
+ cfg,
2839
+ repoLink,
2840
+ parentId,
2841
+ workspaceMemory,
2842
+ signal,
2843
+ context,
2844
+ caps,
2845
+ });
2791
2846
  return { status: "chat" };
2792
2847
  }
2793
2848
  // routed.code → fall through to the coding flow below.
@@ -3300,6 +3355,11 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3300
3355
  let drainProgress = async () => {};
3301
3356
  // 0782 — the silent-work stop poll, torn down in the same finally as the run.
3302
3357
  let stopPoller = null;
3358
+ // Structured transports expose the final assistant summary as an event;
3359
+ // argv transports fold it into the progress emitter. Keep one local value
3360
+ // so GitHub title/provenance selection never reaches for an out-of-scope
3361
+ // parser variable (0884).
3362
+ let resultText = "";
3303
3363
  if (streamOn) {
3304
3364
  if (!progressId) {
3305
3365
  try {
@@ -3377,9 +3437,10 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3377
3437
  // `exec`, so it has to sit ahead of it. [] for every vendor without a
3378
3438
  // verified flag, leaving their argv byte-identical to before.
3379
3439
  const imageArgs = codeImageArgs(vendor, localImages);
3440
+ const webArgs = codeWebArgs(vendor, parts.slice(1));
3380
3441
  const codeArgs = streamOn
3381
- ? [...parts.slice(1), ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs, ...streamArgs]
3382
- : [...parts.slice(1), ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs];
3442
+ ? [...parts.slice(1), ...webArgs, ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs, ...streamArgs]
3443
+ : [...parts.slice(1), ...webArgs, ...modelArgs, ...dirArgs, ...imageArgs, ...resumeArgs];
3383
3444
  const handleCliData = (c) => {
3384
3445
  // Keep tracking lastLine as a fallback (legacy heartbeat / honesty).
3385
3446
  const lines = String(c).split("\n").map((s) => s.trim()).filter(Boolean);
@@ -3420,6 +3481,9 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3420
3481
  } catch {
3421
3482
  /* a progress fold must never break the run */
3422
3483
  }
3484
+ if (ev?.t === "result" && typeof ev.summary === "string") {
3485
+ resultText = ev.summary;
3486
+ }
3423
3487
  };
3424
3488
  let run;
3425
3489
  try {
@@ -3468,7 +3532,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3468
3532
  cmd: parts[0],
3469
3533
  vendor,
3470
3534
  cwd: repoPath,
3471
- prompt: memoryPreamble(workspaceMemory) + promptText,
3535
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
3472
3536
  // 0778: approvals AND continuity. `resume:false` (the never-worse
3473
3537
  // retry) drops it exactly like buildResumeArgs does.
3474
3538
  resumeSessionId: resume ? resumeSessionId : null,
@@ -3481,6 +3545,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3481
3545
  tool,
3482
3546
  channelId,
3483
3547
  threadRoot,
3548
+ decisionWaitMs: caps.decisionWaitMs,
3484
3549
  runId,
3485
3550
  provider: vendor,
3486
3551
  }),
@@ -3490,7 +3555,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3490
3555
  cmd: parts[0],
3491
3556
  args: codeArgs,
3492
3557
  cwd: repoPath,
3493
- prompt: memoryPreamble(workspaceMemory) + promptText,
3558
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
3494
3559
  timeoutMs: cfg.runTimeoutMs,
3495
3560
  signal,
3496
3561
  // A model running inside the server must never inherit the daemon's
@@ -3502,6 +3567,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3502
3567
  tool,
3503
3568
  channelId,
3504
3569
  threadRoot,
3570
+ decisionWaitMs: caps.decisionWaitMs,
3505
3571
  runId,
3506
3572
  }),
3507
3573
  });
@@ -3513,7 +3579,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3513
3579
  cmd: parts[0],
3514
3580
  codeArgs,
3515
3581
  cwd: repoPath,
3516
- prompt: memoryPreamble(workspaceMemory) + promptText,
3582
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
3517
3583
  model: resolvedModelId,
3518
3584
  // Codex's own resume over this transport. `resume:false` (the
3519
3585
  // never-worse retry) drops it exactly like buildResumeArgs does.
@@ -3525,6 +3591,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3525
3591
  tool,
3526
3592
  channelId,
3527
3593
  threadRoot,
3594
+ decisionWaitMs: caps.decisionWaitMs,
3528
3595
  runId,
3529
3596
  provider: vendor,
3530
3597
  }),
@@ -3538,7 +3605,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3538
3605
  // codeArgs already carries this run's resume flags, so a gated
3539
3606
  // iterate resumes AND raises cards — the two never traded off.
3540
3607
  codeArgs,
3541
- prompt: memoryPreamble(workspaceMemory) + promptText,
3608
+ prompt: agentPreamble(workspaceMemory, cfg) + promptText,
3542
3609
  cwd: repoPath,
3543
3610
  signal,
3544
3611
  onData: handleCliData,
@@ -3548,6 +3615,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3548
3615
  tool,
3549
3616
  channelId,
3550
3617
  threadRoot,
3618
+ decisionWaitMs: caps.decisionWaitMs,
3551
3619
  runId,
3552
3620
  provider: vendor,
3553
3621
  }),
@@ -3559,7 +3627,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3559
3627
  // why nothing above the unit tests could ever drive it (0787).
3560
3628
  run = await deps.runCli({
3561
3629
  cmd: parts[0],
3562
- args: [...codeArgs, memoryPreamble(workspaceMemory) + promptText],
3630
+ args: [...codeArgs, agentPreamble(workspaceMemory, cfg) + promptText],
3563
3631
  cwd: repoPath,
3564
3632
  timeoutMs: cfg.runTimeoutMs,
3565
3633
  label: "coding",
@@ -3630,6 +3698,13 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3630
3698
  }
3631
3699
  }
3632
3700
  }
3701
+ if (!resultText && emitter) {
3702
+ try {
3703
+ resultText = emitter.snapshot()?.lastLine || "";
3704
+ } catch {
3705
+ /* title fallback still has the routed task */
3706
+ }
3707
+ }
3633
3708
  if (run.aborted || signal?.aborted) {
3634
3709
  console.log(" code → cancelled");
3635
3710
  return { aborted: true };
@@ -3661,15 +3736,33 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3661
3736
  Number((git(repoPath, ["rev-list", "--count", `${baseRef}..HEAD`]).stdout || "0").trim()) || 0;
3662
3737
  if (ahead > 0) {
3663
3738
  console.log(` code → agent self-committed (${ahead} commit(s) ahead); reconciling`);
3664
- return { empty: true, failed: false, ahead };
3739
+ return {
3740
+ empty: true,
3741
+ failed: false,
3742
+ ahead,
3743
+ summary: (resultText || (streamArgs.length ? "" : run.stdout) || "").trim(),
3744
+ };
3665
3745
  }
3666
3746
  console.log(" code → no changes produced");
3667
- return { empty: true, failed: false, ahead: 0 };
3747
+ return {
3748
+ empty: true,
3749
+ failed: false,
3750
+ ahead: 0,
3751
+ summary: (resultText || (streamArgs.length ? "" : run.stdout) || "").trim(),
3752
+ };
3668
3753
  }
3669
3754
  console.log(" code → diff captured");
3670
3755
  const stat = parseShortstat(git(repoPath, ["diff", "--cached", "--shortstat"]).stdout);
3671
3756
  const { text: diffText, truncated, omittedLines } = truncateDiff(diff);
3672
- return { empty: false, diffText, truncated, omittedLines, stat, runFailed: run.status !== 0 };
3757
+ return {
3758
+ empty: false,
3759
+ diffText,
3760
+ truncated,
3761
+ omittedLines,
3762
+ stat,
3763
+ runFailed: run.status !== 0,
3764
+ summary: (resultText || (streamArgs.length ? "" : run.stdout) || "").trim(),
3765
+ };
3673
3766
  };
3674
3767
 
3675
3768
  // Persist the run's coding-agent session (0282) so a LATER iterate can resume it.
@@ -3886,6 +3979,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3886
3979
  parentId: threadRoot,
3887
3980
  settleId: selfSettleId,
3888
3981
  runId,
3982
+ provenance: githubProvenance({ cfg, me, message, channelId, runId, summary: staged.summary }),
3889
3983
  usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
3890
3984
  });
3891
3985
  await recordSession(selfResult.prUrl);
@@ -3962,6 +4056,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
3962
4056
  existingPrUrl: continuingPrUrl,
3963
4057
  settleId,
3964
4058
  runId,
4059
+ provenance: githubProvenance({ cfg, me, message, channelId, runId, summary: staged.summary }),
3965
4060
  usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
3966
4061
  });
3967
4062
  // The settle already flipped the card to the report (body + metadata); editing
@@ -4024,6 +4119,7 @@ export async function handleTask({ message, channelId, tool, me, caps = {} }, cf
4024
4119
  parentId: threadRoot,
4025
4120
  existingPrUrl: continuingPrUrl,
4026
4121
  runId,
4122
+ provenance: githubProvenance({ cfg, me, message, channelId, runId, summary: staged.summary }),
4027
4123
  // Normally {} — a gated run already reported its spend on the proposal card
4028
4124
  // it is now shipping. Anything the CLI spent after that still comes home.
4029
4125
  usageArgs: reportUsageArgs(runUsage, { vendor, modelId: resolvedModelId, runId }),
package/src/mcp.mjs CHANGED
@@ -48,13 +48,20 @@ export function makeClient({ url, token }) {
48
48
  }
49
49
  }
50
50
 
51
- async function listToolNames() {
51
+ // Full tool objects, schemas included — capability sniffing reads these
52
+ // (0866: an ARG-level addition like waitMs shows up in a tool's inputSchema,
53
+ // not in the tool list's names).
54
+ async function listTools() {
52
55
  try {
53
- return ((await rpc("tools/list"))?.tools ?? []).map((t) => t.name);
56
+ return (await rpc("tools/list"))?.tools ?? [];
54
57
  } catch {
55
58
  return [];
56
59
  }
57
60
  }
58
61
 
59
- return { rpc, tool, listToolNames };
62
+ async function listToolNames() {
63
+ return (await listTools()).map((t) => t.name);
64
+ }
65
+
66
+ return { rpc, tool, listTools, listToolNames };
60
67
  }
@@ -57,7 +57,8 @@ export function detectVendor(codingCmd) {
57
57
  * Code installed just to answer chat (0521). Every command is the vendor's
58
58
  * verified non-interactive print mode; the daemon appends the prompt as the last
59
59
  * arg. codex carries --skip-git-repo-check because chat (and the read-only
60
- * review sandbox) can run outside a git checkout. cursor carries
60
+ * review sandbox) can run outside a git checkout, plus the explicit web-search
61
+ * config because `codex exec` otherwise leaves that tool off. cursor carries
61
62
  * --output-format text (explicit, so a CLI default change can never post raw
62
63
  * JSONL into the channel) and --trust (its Jan-2026 workspace-trust gate fails
63
64
  * headless runs at spawn in untrusted directories — 0572; pre-2026 CLIs reject
@@ -70,7 +71,9 @@ export function detectVendor(codingCmd) {
70
71
  */
71
72
  export function fastChatCmd(vendor) {
72
73
  if (vendor === "claude_code") return "claude -p --model haiku";
73
- if (vendor === "codex") return "codex exec --skip-git-repo-check";
74
+ if (vendor === "codex") {
75
+ return "codex exec --skip-git-repo-check -c tools.web_search=true";
76
+ }
74
77
  if (vendor === "cursor") return "cursor-agent -p --output-format text --trust";
75
78
  if (vendor === "opencode") return "opencode run";
76
79
  if (vendor === "antigravity") return "agy -p";
@@ -107,6 +110,31 @@ export function codeStreamArgs(vendor) {
107
110
  return [];
108
111
  }
109
112
 
113
+ /**
114
+ * Make Codex's built-in public web search available to code runs. This is a
115
+ * capability flag, not an instruction to browse; Codex decides whether the
116
+ * task needs it. An operator's explicit true/false override wins unchanged.
117
+ * Other vendors already expose their own web tools and receive no guessed
118
+ * flags.
119
+ * @param {'claude_code'|'codex'|'cursor'|'opencode'|'antigravity'|'hermes'|'unknown'} vendor
120
+ * @param {string[]} baseArgs
121
+ * @returns {string[]}
122
+ */
123
+ export function codeWebArgs(vendor, baseArgs = []) {
124
+ if (vendor !== "codex") return [];
125
+ const hasOverride = baseArgs.some((arg, index) => {
126
+ if (/^tools\.web_search=/.test(arg)) return true;
127
+ if (
128
+ (baseArgs[index - 1] === "-c" || baseArgs[index - 1] === "--config") &&
129
+ /^tools\.web_search=/.test(arg)
130
+ ) {
131
+ return true;
132
+ }
133
+ return /^--config=tools\.web_search=/.test(arg);
134
+ });
135
+ return hasOverride ? [] : ["-c", "tools.web_search=true"];
136
+ }
137
+
110
138
  /**
111
139
  * Extra args that hand the code run an IMAGE, appended to the code run's argv
112
140
  * (0779). Verified against the installed binaries, per the 0521 rule — never