openmausbot 0.1.83 → 0.1.85

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/dist/assets/index-Ba9G44HI.js +310 -0
  2. package/dist/assets/{index-nqTFZTxg.js → index-CTNNoqSJ.js} +1 -1
  3. package/dist/assets/index-CkRNp7wX.css +1 -0
  4. package/dist/index.html +2 -2
  5. package/dist-server/companion/src/routes.js +1 -0
  6. package/dist-server/container-mcp.js +212 -7
  7. package/dist-server/drivers/agents-proxy.js +161 -12
  8. package/dist-server/enterprise/server/index.js +1 -0
  9. package/dist-server/hooks/omb-hook.js +74 -0
  10. package/dist-server/index.js +8025 -5661
  11. package/dist-server/local-computer-proxy.js +83 -1
  12. package/dist-server/local-computer.js +14 -2
  13. package/dist-server/openmausbot.js +796 -309
  14. package/dist-server/pair-cli.js +796 -309
  15. package/dist-server/proxy-paths.js +1 -0
  16. package/dist-server/server/agent-tool-policy.js +1 -0
  17. package/dist-server/server/bot-package.js +2 -0
  18. package/dist-server/server/box.js +23 -4
  19. package/dist-server/server/browser-engine.js +87 -10
  20. package/dist-server/server/browser-live.js +7 -5
  21. package/dist-server/server/browser-runtime.js +65 -10
  22. package/dist-server/server/checkpoints.js +71 -23
  23. package/dist-server/server/chief-of-staff.js +3 -0
  24. package/dist-server/server/cli-prompts.js +3 -1
  25. package/dist-server/server/commands.js +27 -0
  26. package/dist-server/server/compaction-summary.js +78 -0
  27. package/dist-server/server/computer-wait.js +33 -0
  28. package/dist-server/server/config.js +84 -3
  29. package/dist-server/server/context-budget.js +29 -0
  30. package/dist-server/server/context-rebuild.js +72 -0
  31. package/dist-server/server/delta-context.js +258 -0
  32. package/dist-server/server/digest.js +130 -0
  33. package/dist-server/server/drivers/acp/core.js +540 -229
  34. package/dist-server/server/drivers/agents-proxy.js +79 -12
  35. package/dist-server/server/drivers/agents-result.js +23 -0
  36. package/dist-server/server/drivers/claude.js +84 -15
  37. package/dist-server/server/drivers/codex.js +19 -5
  38. package/dist-server/server/drivers/openai-chat.js +66 -45
  39. package/dist-server/server/drivers/openai-compat.js +1 -0
  40. package/dist-server/server/hooks/omb-hook.js +103 -0
  41. package/dist-server/server/incidents.js +104 -0
  42. package/dist-server/server/index.js +1382 -318
  43. package/dist-server/server/local-computer.js +21 -2
  44. package/dist-server/server/mcp-bridge.js +10 -2
  45. package/dist-server/server/mcp-tool-images.js +37 -0
  46. package/dist-server/server/mcp-tool-schema.js +108 -0
  47. package/dist-server/server/message-db.js +80 -21
  48. package/dist-server/server/message-file.js +3 -2
  49. package/dist-server/server/notify.js +3 -1
  50. package/dist-server/server/package-export.js +1 -0
  51. package/dist-server/server/peer-roster.js +4 -2
  52. package/dist-server/server/provider-icon.js +20 -0
  53. package/dist-server/server/proxy-paths.js +1 -0
  54. package/dist-server/server/request-auth.js +1 -0
  55. package/dist-server/server/room-handoffs.js +16 -2
  56. package/dist-server/server/routine-requests.js +39 -1
  57. package/dist-server/server/routines.js +43 -9
  58. package/dist-server/server/shared-computer-control.js +31 -4
  59. package/dist-server/server/steer-queue.js +6 -0
  60. package/dist-server/server/store.js +111 -14
  61. package/dist-server/server/system-prompt.js +4 -4
  62. package/dist-server/server/team-backup.js +9 -1
  63. package/dist-server/server/tool-results.js +72 -0
  64. package/dist-server/server/tts/grok.js +74 -0
  65. package/dist-server/server/tts/index.js +19 -1
  66. package/dist-server/server/webhooks.js +28 -0
  67. package/dist-server/shared/approval-mode.js +7 -1
  68. package/dist-server/shared/digest.js +1 -0
  69. package/dist-server/shared/markdown-windows-paths.js +36 -0
  70. package/dist-server/shared/provider-icon.js +110 -0
  71. package/dist-server/shared/team-backup.js +3 -0
  72. package/dist-server/vps-container-mcp.js +217 -12
  73. package/enterprise/server/index.js +1 -0
  74. package/package.json +1 -1
  75. package/dist/assets/index-CKysBq-V.css +0 -1
  76. package/dist/assets/index-DSvqfBOf.js +0 -309
@@ -38,6 +38,7 @@ import { CREDENTIAL_TARGETS, isCredentialTargetId } from "../../shared/credentia
38
38
  import { normalizeCronSchedule } from "../../shared/routine-schedule.js";
39
39
  import { agentToolAnnotations } from "../agent-tool-policy.js";
40
40
  import { peerName } from "../peer-roster.js";
41
+ import { boundedAgentResult } from "./agents-result.js";
41
42
  const HARNESS = process.env.OMB_HARNESS_URL ?? "http://127.0.0.1:8799";
42
43
  const BOT_ID = process.env.OMB_BOT_ID ?? "";
43
44
  const THREAD_ID = process.env.OMB_THREAD_ID ?? "";
@@ -349,8 +350,8 @@ const ROUTINE_FIELDS_SCHEMA = {
349
350
  schedule: ROUTINE_SCHEDULE_SCHEMA,
350
351
  run_on: {
351
352
  type: "string",
352
- enum: ["maus", "cloud"],
353
- description: "Where the routine runs. Defaults to maus (this OpenMausBot setup).",
353
+ enum: ["maus", "box"],
354
+ description: "Default maus keeps the bot's selected model and configured computer, INCLUDING a self-hosted VPS. Omit this field for normal schedules. box explicitly switches the agent to the Box-hosted runner; it requires Box setup and is not the generic cloud/VPS option. Legacy cloud values from list_routines mean box, not VPS.",
354
355
  },
355
356
  timeout_minutes: {
356
357
  type: "integer",
@@ -366,9 +367,26 @@ const ROUTINE_FIELDS_SCHEMA = {
366
367
  type: "boolean",
367
368
  description: "Opt in to using the latest completed run's bounded report as historical context. Defaults to false; set false in an update to start fresh again. Included in the applied result or pending confirmation.",
368
369
  },
370
+ overlap: {
371
+ type: "string",
372
+ enum: ["skip", "queue"],
373
+ description: "While this routine is still working, skip scheduled occurrences (default) or queue at most one run. Queue skips further occurrences until the pending run starts; it never builds an unlimited backlog. Manual and webhook requests are separate.",
374
+ },
369
375
  };
370
376
  const PROPOSAL_OUTCOME = " Read the result: granted Full Access may apply the change immediately. If applied, continue the requested work without another confirmation. Only a pending result requires ending the turn and waiting for the in-app decision. Never claim success from the permission mode alone; report failed or cancelled results honestly. This does not elevate another bot's execution permissions.";
371
377
  const TOOLS = [
378
+ {
379
+ name: "tool_result_read",
380
+ description: "Read a missing portion of an oversized agents-tool result using the saved id and next offset from its notice. Returns at most 16,000 characters, only from this bot in this conversation. Use only when the preview is insufficient; do not load every page by default. Results expire after one hour, on app restart, or under cache pressure. This never reruns the original action.",
381
+ inputSchema: {
382
+ type: "object", additionalProperties: false,
383
+ properties: {
384
+ id: { type: "string", description: "Saved result id copied from the truncation notice." },
385
+ offset: { type: "integer", minimum: 0, description: "Character offset copied from the previous result's notice. Defaults to 0." },
386
+ },
387
+ required: ["id"],
388
+ },
389
+ },
372
390
  {
373
391
  name: "list_shared_computers",
374
392
  description: "List online desktop computers explicitly shared with this workspace, and their allowed folders/capabilities. These are the user's computers, not this server. An offline or unshared computer cannot be accessed. Folder paths use opaque folder IDs and relative paths.",
@@ -412,7 +430,7 @@ const TOOLS = [
412
430
  },
413
431
  {
414
432
  name: "ask_bot",
415
- description: "SYNCHRONOUS consultation: send a short question to another bot and stay blocked until its reply is returned inline. Use only when that reply is required to write your current response. Do not use for assigning work, background tasks, or potentially long work; use delegate_bot for those. Returns promptly with a note if that bot is busy.",
433
+ description: "Brief synchronous consultation: send a short question to another bot. Quick replies return inline; slow replies become asynchronous delegations and return automatically after you finish your turn. Use only when that reply is required to write your current response. Do not use for assigning work, background tasks, or potentially long work; use delegate_bot for those. Returns promptly with a note if that bot is busy.",
416
434
  inputSchema: {
417
435
  type: "object",
418
436
  properties: {
@@ -642,6 +660,20 @@ const TOOLS = [
642
660
  required: ["action"],
643
661
  },
644
662
  },
663
+ {
664
+ name: "retry_thread",
665
+ description: "Chief of Staff only. Resume a teammate's thread whose last run failed, stalled or could not start — the one an incident report named — exactly where it stopped, keeping its conversation and files. The teammate gets a line saying you asked for the retry and why. Use it when the cause looks transient (a crash, a timeout, a busy service). Use delegate_bot with a corrected brief instead when the request itself needs to change, and tell the person instead when only they can fix the cause (a sign-in, a missing credential, an unanswered question). Never retry the same thread more than twice.",
666
+ inputSchema: {
667
+ type: "object",
668
+ additionalProperties: false,
669
+ properties: {
670
+ bot_id: { type: "string", description: "The teammate's id, from the incident report or list_bots." },
671
+ thread_id: { type: "string", description: "The failed thread's id, from the incident report." },
672
+ note: { type: "string", description: "Optional: one sentence for the teammate about what to watch for this time." },
673
+ },
674
+ required: ["bot_id", "thread_id"],
675
+ },
676
+ },
645
677
  {
646
678
  name: "memory_log",
647
679
  description: "Write one line to today's log file, memory/log/YYYY-MM-DD.md, stamped with the time and this conversation: what happened, not what is true. Use it for events worth a trace — a deploy went out, a person decided something, a check failed — that should not shape future sessions. Logs are never loaded into your prompt; the person can read them, and session_search finds them later. A fact that should hold in every session goes to memory_update instead.",
@@ -840,6 +872,8 @@ async function api(path, init) {
840
872
  throw new Error(String(body.error ?? `HTTP ${status}`));
841
873
  return body;
842
874
  }
875
+ const capResult = (text) => boundedAgentResult(text, (retained, truncated) => api("/api/internal/tool-result", { method: "POST", signal: AbortSignal.timeout(3_000),
876
+ body: JSON.stringify({ text: retained, truncated }) }));
843
877
  /** Like api, but a refusal comes back as its body instead of an Error —
844
878
  * for the tools whose refusals carry more than a sentence. */
845
879
  async function apiResponse(path, init) {
@@ -870,16 +904,17 @@ function routineFields(args) {
870
904
  const fields = {};
871
905
  // list_routines returns the harness names. Accept those when a model
872
906
  // copies back a definition, as we already do for interval fields.
873
- if (args.run_on != null && args.runOn != null && args.run_on !== args.runOn) {
907
+ const destination = (value) => value === "box" ? "cloud" : value;
908
+ if (args.run_on != null && args.runOn != null && destination(args.run_on) !== destination(args.runOn)) {
874
909
  return { fields, error: "Choose one run_on destination; run_on and runOn disagree." };
875
910
  }
876
911
  if (args.timeout_minutes != null && args.timeoutMinutes != null && args.timeout_minutes !== args.timeoutMinutes) {
877
912
  return { fields, error: "Choose one timeout_minutes limit; timeout_minutes and timeoutMinutes disagree." };
878
913
  }
879
- const runOn = args.run_on ?? args.runOn;
914
+ const runOn = destination(args.run_on ?? args.runOn);
880
915
  const timeoutMinutes = args.timeout_minutes ?? args.timeoutMinutes;
881
916
  if (runOn != null && runOn !== "maus" && runOn !== "cloud") {
882
- return { fields, error: 'run_on must be "maus" or "cloud".' };
917
+ return { fields, error: 'Use run_on="maus" for the bot’s current model and configured computer (including VPS), or run_on="box" only for the Box-hosted agent. Legacy "cloud" also means Box.' };
883
918
  }
884
919
  if (timeoutMinutes != null && (typeof timeoutMinutes !== "number" || !Number.isInteger(timeoutMinutes) || timeoutMinutes < 5 || timeoutMinutes > 240)) {
885
920
  return { fields, error: "timeout_minutes must be a whole number from 5 to 240. Use clear_timeout to remove a limit." };
@@ -887,6 +922,9 @@ function routineFields(args) {
887
922
  if (args.continuity != null && typeof args.continuity !== "boolean") {
888
923
  return { fields, error: "continuity must be true or false." };
889
924
  }
925
+ if (args.overlap !== undefined && args.overlap !== "skip" && args.overlap !== "queue") {
926
+ return { fields, error: "overlap must be skip or queue." };
927
+ }
890
928
  if (args.clear_timeout != null && typeof args.clear_timeout !== "boolean") {
891
929
  return { fields, error: "clear_timeout must be true or false." };
892
930
  }
@@ -911,6 +949,8 @@ function routineFields(args) {
911
949
  fields.timeoutMinutes = timeoutMinutes;
912
950
  if (typeof args.continuity === "boolean")
913
951
  fields.continuity = args.continuity;
952
+ if (args.overlap !== undefined)
953
+ fields.overlap = args.overlap;
914
954
  return { fields };
915
955
  }
916
956
  /** Full Access is decided by the harness, not inferred from a model claim or
@@ -951,6 +991,17 @@ function recallSpeaker(hit) {
951
991
  return hit.role === "user" ? "user" : "you";
952
992
  }
953
993
  async function callTool(name, args) {
994
+ if (name === "tool_result_read") {
995
+ if (typeof args.id !== "string" || !/^r-[0-9a-f-]{36}$/.test(args.id) ||
996
+ (args.offset !== undefined && (!Number.isSafeInteger(args.offset) || Number(args.offset) < 0))) {
997
+ return { text: "Use the saved result id and a non-negative integer offset from its notice.", isError: true };
998
+ }
999
+ const r = await api(`/api/internal/tool-result?id=${encodeURIComponent(args.id)}&offset=${args.offset ?? 0}`, { signal: AbortSignal.timeout(3_000) });
1000
+ const text = String(r.text ?? "");
1001
+ return { text: `${text}\n\n[${Number(r.nextOffset) < Number(r.length)
1002
+ ? `Read more with tool_result_read id "${args.id}" and offset ${r.nextOffset}.`
1003
+ : `End of retained result.${r.truncated ? " The original tail exceeded the storage limit and was omitted." : ""}`}]` };
1004
+ }
954
1005
  if (name === "list_room_targets") {
955
1006
  const r = await api("/api/internal/room-targets");
956
1007
  return { text: JSON.stringify(r), ...(r.error ? { isError: true } : {}) };
@@ -1067,9 +1118,11 @@ async function callTool(name, args) {
1067
1118
  const taskId = String(r.taskId ?? "").trim();
1068
1119
  if (taskId)
1069
1120
  delegationTaskIdsThisTurn.add(taskId);
1070
- const waitedMinutes = Math.max(1, Math.round((Number(r.waitedMs) || 0) / 60_000));
1121
+ const waitedSeconds = Math.max(1, Math.round((Number(r.waitedMs) || 0) / 1000));
1122
+ const amount = waitedSeconds < 60 ? waitedSeconds : Math.round(waitedSeconds / 60);
1123
+ const unit = waitedSeconds < 60 ? "second" : "minute";
1071
1124
  return {
1072
- text: `${r.toBotName ?? "That bot"} is still working after ${waitedMinutes} minute${waitedMinutes === 1 ? "" : "s"} — the ask was converted to a delegation so the reply is not lost. Task id: ${taskId}. Finish your turn now; the result will be delivered to this conversation automatically. Use check_delegation in a later turn only if the user asks for status.`,
1125
+ text: `${r.toBotName ?? "That bot"} is still working after ${amount} ${unit}${amount === 1 ? "" : "s"} — the ask was converted to a delegation so the reply is not lost. Task id: ${taskId}. Finish your turn now; the result will be delivered to this conversation automatically. Use check_delegation in a later turn only if the user asks for status.`,
1073
1126
  };
1074
1127
  }
1075
1128
  if (r.busy) {
@@ -1507,6 +1560,20 @@ async function callTool(name, args) {
1507
1560
  const entry = typeof r.entry === "string" && r.entry ? ` Entry: ${r.entry}` : "";
1508
1561
  return { text: `Memory updated.${entry}${r.truncated ? " MEMORY.md exceeds the prompt load budget; keep it short and curated." : ""}` };
1509
1562
  }
1563
+ if (name === "retry_thread") {
1564
+ const botId = String(args.bot_id ?? "").trim();
1565
+ const threadId = String(args.thread_id ?? "").trim();
1566
+ const note = typeof args.note === "string" ? args.note.trim() : "";
1567
+ if (!botId || !threadId)
1568
+ return { text: "retry_thread needs bot_id and thread_id — both are in the incident report.", isError: true };
1569
+ const r = await api("/api/internal/retry-thread", {
1570
+ method: "POST",
1571
+ body: JSON.stringify({ fromBotId: BOT_ID, fromThreadId: THREAD_ID, toBotId: botId, toThreadId: threadId, ...(note ? { note } : {}) }),
1572
+ });
1573
+ if (r.error)
1574
+ return { text: `Couldn't retry that thread: ${String(r.error)}`, isError: true };
1575
+ return { text: typeof r.message === "string" ? r.message : "The thread is running again. Its result stays in that thread; you are not woken for it — check it later with session_search or list_threads if you need to." };
1576
+ }
1510
1577
  if (name === "memory_log") {
1511
1578
  if (typeof args.text !== "string" || !args.text.trim()) {
1512
1579
  return { text: "memory_log needs text: one line about what happened.", isError: true };
@@ -1706,7 +1773,7 @@ async function handle(msg) {
1706
1773
  return;
1707
1774
  }
1708
1775
  if (name === "list_shared_computers") {
1709
- textResult(id, JSON.stringify(await api("/api/internal/shared-computers")));
1776
+ textResult(id, await capResult(JSON.stringify(await api("/api/internal/shared-computers"))));
1710
1777
  return;
1711
1778
  }
1712
1779
  if (name === "shared_computer") {
@@ -1715,14 +1782,14 @@ async function handle(msg) {
1715
1782
  if (Array.isArray(result?.content))
1716
1783
  ok(id, result);
1717
1784
  else
1718
- textResult(id, JSON.stringify(result));
1785
+ textResult(id, await capResult(JSON.stringify(result)));
1719
1786
  return;
1720
1787
  }
1721
1788
  const { text, isError } = await callTool(name, (params.arguments ?? {}));
1722
- textResult(id, text, isError);
1789
+ textResult(id, name === "tool_result_read" ? text : await capResult(text), isError);
1723
1790
  }
1724
1791
  catch (e) {
1725
- textResult(id, e.message, true);
1792
+ textResult(id, await capResult(e.message), true);
1726
1793
  }
1727
1794
  return;
1728
1795
  }
@@ -0,0 +1,23 @@
1
+ import { redactSecretsInText } from "../../shared/redact.js";
2
+ import { TOOL_RESULT_MAX_CHARS, TOOL_RESULT_PREVIEW_CHARS, toolResultPrefix } from "../tool-results.js";
3
+ /** The operation already happened. Saving overflow must never retry it or
4
+ * turn a successful operation into a failed MCP call. Only cache I/O is timed. */
5
+ export async function boundedAgentResult(text, save) {
6
+ if (text.length <= 24_000)
7
+ return text;
8
+ const redacted = redactSecretsInText(text);
9
+ const prefix = toolResultPrefix(redacted, TOOL_RESULT_PREVIEW_CHARS);
10
+ const retained = toolResultPrefix(redacted, TOOL_RESULT_MAX_CHARS);
11
+ const truncated = retained.length < redacted.length;
12
+ try {
13
+ const saved = await save(retained, truncated);
14
+ if (!saved || typeof saved.id !== "string" || !/^r-[0-9a-f-]{36}$/.test(saved.id))
15
+ throw new Error("Invalid saved result");
16
+ return `${prefix}\n\n[Large tool result: showing the first ${prefix.length} characters. ${truncated || saved.truncated
17
+ ? "Only a bounded portion was retained; the remaining tail was omitted."
18
+ : "The remaining redacted result is temporarily saved."} If a missing detail is needed, call tool_result_read with id "${saved.id}" and offset ${prefix.length}. Saved results expire after one hour, on app restart, or under cache pressure. Do not repeat an action just to retrieve its output.]`;
19
+ }
20
+ catch {
21
+ return `${prefix}\n\n[Large tool result: showing the first ${prefix.length} characters. The remaining output could not be saved. The original operation was not retried. Do not repeat an action just to retrieve its output.]`;
22
+ }
23
+ }
@@ -9,11 +9,12 @@
9
9
  // - the bot's cloud computer (box.ascii.dev) via server/computer-proxy.ts
10
10
  // — screenshot/exec/open_url, the CUA-on-the-box bridge
11
11
  import { createHash, randomBytes } from "node:crypto";
12
- import { chmodSync, mkdtempSync, readFileSync, rmSync, unlinkSync, writeFileSync } from "node:fs";
12
+ import { chmodSync, mkdirSync, mkdtempSync, readFileSync, rmSync, unlinkSync, writeFileSync } from "node:fs";
13
13
  import { createServer as createNetServer } from "node:net";
14
14
  import { homedir, tmpdir } from "node:os";
15
15
  import { join, dirname, isAbsolute, normalize } from "node:path";
16
16
  import { DATA_DIR, stripWorkspaceCredentialEnv } from "../config.js";
17
+ import { writeFileAtomic } from "../atomic.js";
17
18
  import { augmentedPath } from "../env-path.js";
18
19
  import { brokerSocketPath, describeSpawnFailure, execCli, killCliTree, spawnCli } from "../procs.js";
19
20
  import { classifyResumeFailure, mayReplay, recoveryPromptFor } from "../resume-recovery.js";
@@ -25,6 +26,7 @@ import { classifyError, computeBackoff, interruptibleDelay, RETRY_MAX_ATTEMPTS }
25
26
  import { applyClaudeInject, decodeInjectId, mergeLocalInject, probeLocalInjects, resolveInjectId, } from "./local-inject.js";
26
27
  import { appendNative } from "./native.js";
27
28
  import { SPAWNED_PROXIES } from "../proxy-paths.js";
29
+ import { extractMcpImages } from "../mcp-tool-images.js";
28
30
  import { ASK_USER_QUESTION_TOOL, askQuestionSummary, parseAskQuestions, parseChoices, questionChoices, } from "../../shared/ask-question.js";
29
31
  /** Whether `claude` has been signed in.
30
32
  *
@@ -392,6 +394,7 @@ export function readClaudeModelCatalog(env = process.env) {
392
394
  // far. See server/proxy-paths.ts.
393
395
  const PERM_PROXY_PATH = SPAWNED_PROXIES.permission;
394
396
  const DWEB_PROXY_PATH = SPAWNED_PROXIES.dweb;
397
+ const HOOK_HELPER_PATH = SPAWNED_PROXIES.hook;
395
398
  // in the packaged app process.execPath is the Electron binary — this env
396
399
  // makes it behave as plain node for the spawned MCP proxies (harmless in dev)
397
400
  const NODE_ENV_FLAG = { ELECTRON_RUN_AS_NODE: "1" };
@@ -430,6 +433,26 @@ function askSummary(ask) {
430
433
  return askQuestionSummary(questions).slice(0, 300);
431
434
  return askInputSummary(ask.input) ?? ask.tool ?? "tool";
432
435
  }
436
+ /** Where the hook helper reads this thread's current turn token. Stable per
437
+ * thread (so the CLI's environment can name it once) and private. */
438
+ export function hookTokenFile(threadId, botId) {
439
+ const digest = createHash("sha256").update(`${botId ?? ""}\0${threadId}`).digest("hex").slice(0, 24);
440
+ return join(DATA_DIR, "hook-tokens", `${digest}.token`);
441
+ }
442
+ /** The `hooks` block for the private --settings file: one command for each
443
+ * event the harness observes. Claude Code runs it with the event JSON on
444
+ * stdin and applies any hookSpecificOutput it prints. The command string is
445
+ * a shell line, so both paths are quoted (this repo's own path has a space). */
446
+ export function claudeHookSettings(helperPath) {
447
+ // JSON quoting is not shell quoting: $(), backticks and $names still
448
+ // expand inside double quotes on POSIX. Windows paths come through env
449
+ // variables so their backslashes are not JSON-escaped into the command.
450
+ const command = process.platform === "win32"
451
+ ? '"%OMB_HOOK_NODE%" "%OMB_HOOK_HELPER%"'
452
+ : [process.execPath, helperPath].map(path => `'${path.replace(/'/g, "'\\''")}'`).join(" ");
453
+ const entry = [{ matcher: "", hooks: [{ type: "command", command, timeout: 5 }] }];
454
+ return { PostToolUse: entry, PreCompact: entry, SessionStart: entry, Stop: entry };
455
+ }
433
456
  export function permissionSocketPath(threadId, botId) {
434
457
  // A readable prefix alone is not unique: ids that agree on their first
435
458
  // characters ("t-perm-dup-1", "t-perm-dup-2") would share a socket. POSIX
@@ -945,13 +968,15 @@ export const ClaudeDriver = {
945
968
  const retry = retryState.get(threadId) ?? { attempt: 0, cancelled: false };
946
969
  // A fresh user turn starts un-cancelled. A relaunch must keep a Stop
947
970
  // that landed while it was being scheduled.
948
- if (!relaunch)
971
+ if (!relaunch) {
949
972
  retry.cancelled = false;
973
+ retry.rebuilt = false;
974
+ }
950
975
  retryState.set(threadId, retry);
951
976
  // a retry relaunches the whole CLI; the backoff is scaled down in tests
952
977
  // so a fake's transient failures don't stall real seconds
953
978
  const retryScale = Number(process.env.FAKE_CLAUDE_RETRY_SCALE ?? "1");
954
- const sessionId = typeof turn.resumeCursor === "string" ? turn.resumeCursor : null;
979
+ const sessionId = !turn.sessionReset && typeof turn.resumeCursor === "string" ? turn.resumeCursor : null;
955
980
  const newSessionId = sessionId ? null : newId();
956
981
  const args = [
957
982
  "-p",
@@ -1138,7 +1163,28 @@ export const ClaudeDriver = {
1138
1163
  const env = environment(turnModel);
1139
1164
  const authSettings = isolated && !injected.injected
1140
1165
  ? readClaudeAuthSettings(env, input.environment) : {};
1141
- const authSettingsPath = mcpConfigPath && Object.keys(authSettings).length
1166
+ // Harness hooks (item 0.2): one helper command for the events the
1167
+ // harness observes. The helper reads its bearer from a per-thread file
1168
+ // the harness rewrites every turn, so a long-lived CLI process never
1169
+ // presents a stale token. Registered through the same private
1170
+ // --settings file as the auth override; both are 0600 and per launch.
1171
+ const hooks = turn.integrations?.hooks;
1172
+ const hookTokenPath = hooks ? hookTokenFile(threadId, botId) : null;
1173
+ if (hooks && hookTokenPath) {
1174
+ mkdirSync(dirname(hookTokenPath), { recursive: true, mode: 0o700 });
1175
+ writeFileAtomic(hookTokenPath, hooks.token, { mode: 0o600 });
1176
+ env.OMB_HOOK_URL = hooks.url;
1177
+ env.OMB_HOOK_TOKEN_FILE = hookTokenPath;
1178
+ env.OMB_HOOK_NODE = process.execPath;
1179
+ env.OMB_HOOK_HELPER = HOOK_HELPER_PATH;
1180
+ // in the packaged app process.execPath is Electron — run the helper as node
1181
+ if (process.versions.electron)
1182
+ env.ELECTRON_RUN_AS_NODE = "1";
1183
+ }
1184
+ const settings = { ...authSettings };
1185
+ if (hooks)
1186
+ settings.hooks = claudeHookSettings(HOOK_HELPER_PATH);
1187
+ const authSettingsPath = mcpConfigPath && Object.keys(settings).length
1142
1188
  ? join(dirname(mcpConfigPath), "auth-settings.json") : null;
1143
1189
  if (authSettingsPath)
1144
1190
  args.push("--settings", authSettingsPath);
@@ -1161,6 +1207,8 @@ export const ClaudeDriver = {
1161
1207
  model: injected.model ?? null,
1162
1208
  base: env.ANTHROPIC_BASE_URL ?? null,
1163
1209
  configDir: env.CLAUDE_CONFIG_DIR ?? null,
1210
+ // hooks on/off changes the settings file the process was launched with
1211
+ hooks: Boolean(hooks),
1164
1212
  // Rotating an account's key/helper must not reuse the old process.
1165
1213
  auth: createHash("sha256").update(JSON.stringify({
1166
1214
  settings: authSettings,
@@ -1168,10 +1216,10 @@ export const ClaudeDriver = {
1168
1216
  })).digest("hex"),
1169
1217
  });
1170
1218
  // Reuse the live process when it is idle, unchanged, and is the session
1171
- // the harness wants resumed. Anything else: close it and spawn fresh
1172
- // (with --resume, so the conversation continues in the new process).
1219
+ // the harness wants resumed. Clearing a cursor alone does not opt out
1220
+ // of legacy reuse: an explicit rebuild must discard the idle context.
1173
1221
  const live = sessions.get(threadId);
1174
- if (live && !live.turn && !live.closing && live.child.exitCode === null && live.argsKey === argsKey && (!sessionId || sessionId === live.sessionId)) {
1222
+ if (!turn.sessionReset && live && !live.turn && !live.closing && live.child.exitCode === null && live.argsKey === argsKey && (!sessionId || sessionId === live.sessionId)) {
1175
1223
  if (live.idleTimer)
1176
1224
  clearTimeout(live.idleTimer);
1177
1225
  live.turn = { turnId, input: turn, retryAbort, settled: false, sawStreamDelta: false };
@@ -1211,7 +1259,7 @@ export const ClaudeDriver = {
1211
1259
  return { turnId };
1212
1260
  }
1213
1261
  if (live)
1214
- closeSession(threadId, "spawn contract changed");
1262
+ closeSession(threadId, turn.sessionReset ? "context reset" : "spawn contract changed");
1215
1263
  // Until sessions.set() below, this turn owns every launch resource.
1216
1264
  // Any bind, private-config or synchronous spawn failure must release
1217
1265
  // them here rather than leave a live listener or credential temp file.
@@ -1306,7 +1354,7 @@ export const ClaudeDriver = {
1306
1354
  writeFileSync(mcpConfigPath, JSON.stringify({ mcpServers }), { mode: 0o600 });
1307
1355
  }
1308
1356
  if (authSettingsPath) {
1309
- writeFileSync(authSettingsPath, JSON.stringify(authSettings), { mode: 0o600 });
1357
+ writeFileSync(authSettingsPath, JSON.stringify(settings), { mode: 0o600 });
1310
1358
  }
1311
1359
  if (sessionId)
1312
1360
  args.push("--resume", sessionId);
@@ -1407,7 +1455,7 @@ export const ClaudeDriver = {
1407
1455
  session.nativePermissionMode = typeof o.permissionMode === "string" ? o.permissionMode : null;
1408
1456
  if (typeof o.session_id === "string")
1409
1457
  session.sessionId = o.session_id;
1410
- emit({ ...base(threadId, currentTurnId()), type: "session.started", sessionId: o.session_id, model: o.model });
1458
+ emit({ ...base(threadId, currentTurnId()), type: "session.started", sessionId: o.session_id, model: o.model, ...(retry.rebuilt ? { rebuilt: true } : {}) });
1411
1459
  }
1412
1460
  else if (o.subtype === "thinking_tokens") {
1413
1461
  emit({ ...base(threadId, currentTurnId()), type: "item.updated", itemType: "reasoning", tokens: o.estimated_tokens });
@@ -1447,13 +1495,16 @@ export const ClaudeDriver = {
1447
1495
  break;
1448
1496
  }
1449
1497
  if (text.trim()) {
1498
+ // The CLI's own report of any other API error is still shown,
1499
+ // but marked: the model never produced it.
1500
+ const synthetic = o.is_api_error_message === true || typeof o.error === "string" ? { synthetic: true } : {};
1450
1501
  // fallback delta for CLIs/paths that never streamed the block
1451
1502
  if (!session.turn?.sawStreamDelta) {
1452
- emit({ ...base(threadId, currentTurnId()), type: "content.delta", streamKind: "assistant_text", delta: text });
1503
+ emit({ ...base(threadId, currentTurnId()), ...synthetic, type: "content.delta", streamKind: "assistant_text", delta: text });
1453
1504
  }
1454
1505
  if (session.turn)
1455
1506
  session.turn.sawStreamDelta = false;
1456
- emit({ ...base(threadId, currentTurnId()), type: "item.completed", itemType: "assistant_text", text });
1507
+ emit({ ...base(threadId, currentTurnId()), ...synthetic, type: "item.completed", itemType: "assistant_text", text });
1457
1508
  }
1458
1509
  for (const b of Array.isArray(msg.content) ? msg.content : []) {
1459
1510
  if (b.type === "tool_use") {
@@ -1488,6 +1539,9 @@ export const ClaudeDriver = {
1488
1539
  for (const b of Array.isArray(o.message?.content) ? o.message.content : []) {
1489
1540
  if (b.type === "tool_result") {
1490
1541
  emit({ ...base(threadId, currentTurnId()), type: "item.completed", itemType: "tool", itemId: b.tool_use_id, ok: !b.is_error, output: toolDetailPreview(b.content) });
1542
+ for (const img of extractMcpImages(b.content)) {
1543
+ emit({ ...base(threadId, currentTurnId()), type: "item.completed", itemType: "assistant_image", data: img.data });
1544
+ }
1491
1545
  }
1492
1546
  }
1493
1547
  break;
@@ -1622,7 +1676,9 @@ export const ClaudeDriver = {
1622
1676
  active.set(threadId, { stop: () => { retry.cancelled = true; retryAbort.abort(); }, turnId });
1623
1677
  try {
1624
1678
  const cursor = session.sessionId ?? sessionId ?? undefined;
1625
- await sendTurn({ ...turn, resumeCursor: cursor }, turnId);
1679
+ // The reset was consumed by the initial launch. Retry the
1680
+ // new session, never the context that launch replaced.
1681
+ await sendTurn({ ...turn, sessionReset: false, resumeCursor: cursor }, turnId);
1626
1682
  }
1627
1683
  catch (e) {
1628
1684
  if (active.get(threadId)?.turnId === turnId)
@@ -1680,7 +1736,10 @@ export const ClaudeDriver = {
1680
1736
  }
1681
1737
  sessions.delete(threadId);
1682
1738
  session.turn = null;
1683
- // Same relaunch handle as the transient-retry path above.
1739
+ // Same relaunch handle as the transient-retry path above. The new
1740
+ // session is announced as rebuilt only when it is actually given
1741
+ // the replay: with nothing to replay it gets the turn text alone.
1742
+ retry.rebuilt = recovery.replayed;
1684
1743
  retryState.set(threadId, retry);
1685
1744
  active.set(threadId, { stop: () => { retry.cancelled = true; retryAbort.abort(); }, turnId });
1686
1745
  emit({
@@ -1867,9 +1926,19 @@ export const ClaudeDriver = {
1867
1926
  nativeImageInput: true,
1868
1927
  effortLevels: ["low", "medium", "high", "xhigh", "max"],
1869
1928
  queueing: true,
1929
+ // Only while this CLI can be told to refresh a resumed session's
1930
+ // recorded system prompt (--system-prompt-snapshot). Keeping a
1931
+ // session across an update from outside it means the harness keeps
1932
+ // its prompt too; an older CLI would answer a delegated return with
1933
+ // the instructions of the turn that started the session, where a
1934
+ // fresh session rebuilt them. Unknown version: not yet.
1935
+ get strictResume() {
1936
+ return cliVersionChecked && cliVersion !== null && claudeCliSupports(cliVersion, "--system-prompt-snapshot");
1937
+ },
1870
1938
  // Harness turns reassert a per-bot mode and restore the broker even
1871
1939
  // when an old instance was configured with bypassPermissions.
1872
1940
  localComputerMcp: true,
1941
+ hooks: true,
1873
1942
  },
1874
1943
  sendTurn,
1875
1944
  steer,
@@ -1897,7 +1966,7 @@ export const ClaudeDriver = {
1897
1966
  return () => listeners.delete(listener);
1898
1967
  },
1899
1968
  },
1900
- generateText: (prompt) => generateReview(prompt),
1969
+ generateText: (prompt, options) => generateReview(prompt, options?.signal),
1901
1970
  reviewPermission: generateReview,
1902
1971
  dispose: async () => {
1903
1972
  try {
@@ -25,6 +25,7 @@ import { codexDeveloperInstructions, syncCodexInstructions } from "./codex-instr
25
25
  import { CodexDeviceAuthController } from "./codex-device-auth.js";
26
26
  import { codexAccountEmail } from "./codex-identity.js";
27
27
  import { classifyResumeFailure, mayReplay, recoveryPromptFor } from "../resume-recovery.js";
28
+ import { extractMcpImages } from "../mcp-tool-images.js";
28
29
  export { decodeCodexSelection, readCodexModelCatalog, STATIC_CODEX_MODELS } from "./codex-catalog.js";
29
30
  const DRIVER_KIND = "codex";
30
31
  const ASTRA_MODEL_ID = "gpt-6-astra";
@@ -1039,6 +1040,11 @@ export const CodexDriver = {
1039
1040
  ok: item.status !== "failed" && item.status !== "declined",
1040
1041
  output: toolDetailPreview(item.type === "commandExecution" ? { output: item.aggregatedOutput, exitCode: item.exitCode } : item.type === "mcpToolCall" ? item.error ?? item.result : item.type === "fileChange" ? item.changes : item.action),
1041
1042
  });
1043
+ if (item.type === "mcpToolCall") {
1044
+ for (const img of extractMcpImages(item.result)) {
1045
+ emit({ ...base(threadId, turnId), type: "item.completed", itemType: "assistant_image", data: img.data });
1046
+ }
1047
+ }
1042
1048
  }
1043
1049
  else if (item.type === "reasoning") {
1044
1050
  emit({ ...base(threadId, turnId), type: "item.updated", itemType: "reasoning", tokens: null });
@@ -1309,6 +1315,7 @@ export const CodexDriver = {
1309
1315
  const cursor = typeof turn.resumeCursor === "string" ? turn.resumeCursor : null;
1310
1316
  let startedModel = null;
1311
1317
  let resumedNativeThread = false;
1318
+ let rebuiltFromReplay = false;
1312
1319
  let promptText = turn.text;
1313
1320
  if (cursor) {
1314
1321
  const resumeThread = () => request("thread/resume", {
@@ -1339,13 +1346,19 @@ export const CodexDriver = {
1339
1346
  promptSubmitted,
1340
1347
  producedOutput: state.sawStreamDelta,
1341
1348
  });
1342
- if (!config.managed || recoveredMissingSession || stopRequested || state.settled ||
1349
+ if ((!config.managed && !turn.recoveryIsReplay) || recoveredMissingSession || stopRequested || state.settled ||
1343
1350
  !turn.recoveryText?.trim() || !missingNativeCodexThread(error, cursor) || !mayReplay(failure))
1344
1351
  throw error;
1345
- // The prompt has never been submitted. Rebuild only missing Company
1346
- // histories, once, through the same approved model/provider below.
1352
+ // The prompt has never been submitted. Rebuild missing Company
1353
+ // histories, and a personal thread only for a turn whose recovery
1354
+ // text is the replay it would have had anyway; once, through the
1355
+ // same approved model/provider below.
1347
1356
  recoveredMissingSession = true;
1348
- promptText = recoveryPromptFor({ recoveryText: turn.recoveryText, currentText: turn.text, failure }).text;
1357
+ const rebuild = recoveryPromptFor({ recoveryText: turn.recoveryText, currentText: turn.text, failure });
1358
+ // Announced as rebuilt only when the replacement really carries the
1359
+ // replay; otherwise it holds no more than the turn text.
1360
+ rebuiltFromReplay = rebuild.replayed;
1361
+ promptText = rebuild.text;
1349
1362
  }
1350
1363
  }
1351
1364
  if (!codexThreadId) {
@@ -1374,7 +1387,7 @@ export const CodexDriver = {
1374
1387
  if (!codexThreadId)
1375
1388
  throw new Error("Codex did not return a native thread id");
1376
1389
  await syncCodexInstructions(threadId, codexThreadId, developerInstructions, resumedNativeThread, request);
1377
- emit({ ...base(threadId, turnId), type: "session.started", sessionId: codexThreadId, model: startedModel ?? turn.model ?? null });
1390
+ emit({ ...base(threadId, turnId), type: "session.started", sessionId: codexThreadId, model: startedModel ?? turn.model ?? null, ...(rebuiltFromReplay ? { rebuilt: true } : {}) });
1378
1391
  const turnInput = [
1379
1392
  ...(promptText ? [{ type: "text", text: promptText }] : []),
1380
1393
  ...(turn.images ?? []).map((image) => ({ type: "localImage", path: image.path })),
@@ -1512,6 +1525,7 @@ export const CodexDriver = {
1512
1525
  images: true,
1513
1526
  nativeImageInput: true,
1514
1527
  effortLevels: ["low", "medium", "high", "xhigh", "max"],
1528
+ strictResume: true,
1515
1529
  },
1516
1530
  sendTurn,
1517
1531
  interruptTurn: async (threadId) => {