openmausbot 0.1.78 → 0.1.79

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/dist/assets/{index-BqDJqf2C.js → index-C2Je5jdF.js} +1 -1
  2. package/dist/assets/index-D_TFZLlo.js +305 -0
  3. package/dist/index.html +1 -1
  4. package/dist-server/container-mcp.js +11 -5
  5. package/dist-server/enterprise/server/index.js +0 -1
  6. package/dist-server/index.js +15451 -6812
  7. package/dist-server/local-computer.js +0 -1
  8. package/dist-server/mcp-server.js +8 -1
  9. package/dist-server/openmausbot.js +9190 -758
  10. package/dist-server/pair-cli.js +9190 -758
  11. package/dist-server/proxy-paths.js +0 -1
  12. package/dist-server/server/box.js +98 -41
  13. package/dist-server/server/browser-engine.js +1 -1
  14. package/dist-server/server/browser-runtime.js +11 -3
  15. package/dist-server/server/browser-tool-shape.js +72 -0
  16. package/dist-server/server/claude-accounts.js +1 -0
  17. package/dist-server/server/cli-setup.js +1 -1
  18. package/dist-server/server/composio.js +18 -0
  19. package/dist-server/server/config.js +16 -4
  20. package/dist-server/server/container-computer.js +0 -12
  21. package/dist-server/server/drivers/acp/core.js +4 -19
  22. package/dist-server/server/drivers/boxagent.js +48 -4
  23. package/dist-server/server/drivers/chat-mcp-tools.js +368 -0
  24. package/dist-server/server/drivers/chat-tool-approval.js +42 -0
  25. package/dist-server/server/drivers/claude.js +87 -32
  26. package/dist-server/server/drivers/codex.js +69 -31
  27. package/dist-server/server/drivers/grok.js +4 -0
  28. package/dist-server/server/drivers/minimax.js +11 -3
  29. package/dist-server/server/drivers/openai-chat-protocol.js +92 -0
  30. package/dist-server/server/drivers/openai-chat.js +332 -88
  31. package/dist-server/server/drivers/openai-compat.js +4 -0
  32. package/dist-server/server/drivers/pi.js +1 -9
  33. package/dist-server/server/index.js +64 -38
  34. package/dist-server/server/proxy-paths.js +0 -1
  35. package/dist-server/server/room-handoffs.js +47 -9
  36. package/dist-server/server/tts/chatterbox.js +83 -0
  37. package/dist-server/server/tts/index.js +45 -12
  38. package/dist-server/server/workspace.js +12 -2
  39. package/dist-server/shared/ask-question.js +28 -0
  40. package/dist-server/shared/computer-contention.js +17 -0
  41. package/dist-server/vps-container-mcp.js +11 -5
  42. package/enterprise/server/index.js +0 -1
  43. package/package.json +1 -1
  44. package/dist/assets/index-DjwAroIZ.js +0 -305
  45. package/dist-server/computer-proxy.js +0 -1196
  46. package/dist-server/server/computer-proxy.js +0 -1070
  47. package/dist-server/server/remote-computer.js +0 -170
@@ -11,6 +11,7 @@ import { SharedComputerControl } from "./shared-computer-control.js";
11
11
  import { RoomHandoffs } from "./room-handoffs.js";
12
12
  import { botAvatarUrlFromStoredPath } from "../shared/bot-avatar.js";
13
13
  import { BOT_PROFILE_LIMITS } from "../shared/bot-profile.js";
14
+ import { CLOUD_COMPUTER_BUSY_ERROR } from "../shared/computer-contention.js";
14
15
  import { approvalModeFor, supportsApprovalMode, modelSwitchNeedsAsk, isEmergencyApprovalDowngrade, isApprovalMode, } from "../shared/approval-mode.js";
15
16
  import { escapeAttribute } from "../src/lib/composer-attachments.js";
16
17
  import { CREDENTIAL_TARGETS, credentialResumeOutcome, credentialIsConfigured, isReusableCredentialRequest, isCredentialTargetId, } from "../shared/credential-request.js";
@@ -38,7 +39,6 @@ import { openMausStatusSystemPrompt } from "./openmaus-status-capsule.js";
38
39
  import { containerComputerAction, containerComputerExists, containerComputerFrame, containerComputerMcp, containerComputerScreenshot, containerComputerStatus, containerRuntimeStatus, localVmRecreatableOnDemand, perBotLocalVmTarget, SHARED_LOCAL_VM_TARGET, setupCommands, } from "./container-computer.js";
39
40
  import { ensureDirs, instanceConfigs, loadConfig, providerReloadKeys, localVmMaxInstances, localVmMode, parseConfigPatch, roomTurnTimeoutMinutes, maxConcurrentBotThreads, saveConfig, showToolCallsEnabled, skillAuthoringEnabled, sharedComputersEnabled, builtInBrowserEnabled, browserProfileReplacementConflict, browserProfilePartitionTarget, syncCredentialEnv, withInstanceCli, persistableInstanceConfigs, vpsSshAlias, DATA_DIR, EVENTS_DIR, NATIVE_DIR, customMcpServers, } from "./config.js";
40
41
  import { ComputerControl } from "./computer-control.js";
41
- import { MAX_REMOTE_COMMAND_LENGTH } from "./remote-computer.js";
42
42
  import { augmentedPath, findCliCandidates, resetPathCache } from "./env-path.js";
43
43
  import { registerEnginesBinDir } from "./engine-install.js";
44
44
  import { appendUsage, parseUsageRange, readUsage, summarizeUsage, usageCsv, USAGE_GROUPINGS, flushUsageLedger } from "./usage-ledger.js";
@@ -80,7 +80,7 @@ import { buildRecoveryText, buildTurnContext, engineIsFresh } from "./turn-conte
80
80
  import { extractTurnImages } from "./turn-images.js";
81
81
  import { TurnWatchdog } from "./turn-watchdog.js";
82
82
  import { TurnResources, workspaceResource } from "./turn-resources.js";
83
- import { ensureWorkspace, ensureTaskWorkspace, workspaceLocationsPrompt, updateMemory, appendMemoryLog, isMemoryTopicName, memorySystemPrompt, memorySourceLabel, searchMemoryFiles, SESSION_SEARCH_SYSTEM_PROMPT, workspaceDir, } from "./workspace.js";
83
+ import { ensureWorkspace, ensureTaskWorkspace, workspaceLocationsPrompt, supportsWorkspaceFiles, updateMemory, appendMemoryLog, isMemoryTopicName, memorySystemPrompt, memorySourceLabel, searchMemoryFiles, SESSION_SEARCH_SYSTEM_PROMPT, workspaceDir, } from "./workspace.js";
84
84
  import { readMemoryTopic } from "./workspace.js";
85
85
  import { MEMORY_INDEX, MemoryStoreError, memoryCapacity, memoryOverview, openMemoryLocation, readMemoryDoc, } from "./memory-store.js";
86
86
  import { beginMemoryTurn, endMemoryTurn, flushAllMemoryJournals, flushMemoryJournal, journalMemoryDelete, journalMemoryWrite, readMemoryJournal, revertMemoryChange, } from "./memory-journal.js";
@@ -613,7 +613,7 @@ async function interruptDirectThread(botId, threadId) {
613
613
  const owner = botForThread(botId, threadId);
614
614
  cancelDirectTurnDispatch(botId, threadId);
615
615
  revokeInternalCapabilitiesForThread(threadId);
616
- await (owner ? registry.get(owner.modelSelection.instanceId) : undefined)?.adapter.interruptTurn(threadId);
616
+ await (owner ? runningTurnInstance(owner, threadId) : null)?.adapter.interruptTurn(threadId);
617
617
  closeOpenApprovals(threadId);
618
618
  }
619
619
  /** Stop left teammates mid-turn: say so in the transcript, name them, and
@@ -1222,7 +1222,7 @@ function previewSystemPrompt(bot) {
1222
1222
  destination: previewComputer,
1223
1223
  browserOn: caps?.browserMcp === true && builtInBrowserEnabled(cfg) && bot.browser !== false,
1224
1224
  });
1225
- const privateWorkspace = instance && !["grok", "boxAgent"].includes(instance.driverKind);
1225
+ const privateWorkspace = instance && supportsWorkspaceFiles(instance.driverKind);
1226
1226
  const built = buildSystemPrompt(persona, bot.soul ?? "", [
1227
1227
  {
1228
1228
  id: "setup",
@@ -1243,7 +1243,7 @@ function previewSystemPrompt(bot) {
1243
1243
  { id: "routine", label: "Routines", text: agentsMounted ? ROUTINE_PROMPT : "" },
1244
1244
  { id: "profile", label: "Profile changes", text: agentsMounted ? PROFILE_PROMPT : "" },
1245
1245
  { id: "section-context", label: "Section context", text: sectionContextSystemPrompt(bot.section) },
1246
- { id: "memory", label: "Memory", text: privateWorkspace ? memorySystemPrompt(bot.id) : "" },
1246
+ { id: "memory", label: "Memory", text: memorySystemPrompt(bot.id, { managedWrites: agentsMounted, fileTools: Boolean(privateWorkspace) }) },
1247
1247
  { id: "skills", label: "Skills index", text: privateWorkspace ? skillsSystemPrompt(bot.id) : "" },
1248
1248
  ]);
1249
1249
  const totalBytes = built.sections.reduce((n, s) => n + s.bytes, 0);
@@ -2794,9 +2794,9 @@ const watchdog = new TurnWatchdog({
2794
2794
  repeats.settle(turn.threadId);
2795
2795
  const bot = botForThread(turn.botId, turn.threadId);
2796
2796
  const routineRun = activeRoutineRunForThread(turn.threadId);
2797
- const instance = routineRun?.runOn === "cloud"
2798
- ? registry.instances().find((candidate) => candidate.driverKind === "boxAgent")
2799
- : bot ? registry.get(bot.modelSelection.instanceId) : null;
2797
+ const instance = bot
2798
+ ? runningTurnInstance(bot, turn.threadId, routineRun?.runOn)
2799
+ : routineRun?.runOn === "cloud" ? registry.instances().find((candidate) => candidate.driverKind === "boxAgent") : null;
2800
2800
  void instance?.adapter.interruptTurn(turn.threadId).catch(() => { });
2801
2801
  const minutes = Math.round(TURN_STALL_MS / 60_000);
2802
2802
  if (routineRun?.target === "bot") {
@@ -2898,8 +2898,10 @@ bus.subscribe((event) => {
2898
2898
  bus.subscribe((event) => {
2899
2899
  if (shouldIgnoreProviderEvent(event))
2900
2900
  return;
2901
- if (event.type === "turn.completed" || event.type === "session.exited")
2901
+ if (event.type === "turn.completed" || event.type === "session.exited") {
2902
+ runningTurnEngines.delete(event.threadId);
2902
2903
  endMemoryTurn(event.threadId);
2904
+ }
2903
2905
  });
2904
2906
  // Bots currently working with nobody at the keyboard — a webhook turn, or a
2905
2907
  // turn a webhook-driven bot handed to a teammate. Auto mode is a decision
@@ -3157,6 +3159,23 @@ function turnProvider(bot, runOn) {
3157
3159
  return null;
3158
3160
  return bot.cloudBackend === "vps" ? "vps" : "box";
3159
3161
  }
3162
+ /** A turn on the cloud computer runs ON the cloud computer: the Box runs the
3163
+ * bot's own harness there with the computer tools built in, so nothing on this
3164
+ * machine relays clicks and screenshots. Every start/interrupt of a turn asks
3165
+ * here which engine owns it. */
3166
+ function turnInstance(bot, runOn) {
3167
+ const onBox = runOn === "cloud"
3168
+ || turnProvider(bot, runOn) === "box" && (bot.computer === "cloud" || inheritedTeamComputer(bot) !== undefined);
3169
+ return onBox
3170
+ ? registry.instances().find((candidate) => candidate.driverKind === "boxAgent") ?? null
3171
+ : registry.get(bot.modelSelection.instanceId);
3172
+ }
3173
+ /** The engine that dispatched each live turn. A bot's settings may change
3174
+ * mid-turn; an interrupt or steer must reach the engine actually running. */
3175
+ const runningTurnEngines = new Map();
3176
+ function runningTurnInstance(bot, threadId, runOn) {
3177
+ return runningTurnEngines.get(threadId) ?? turnInstance(bot, runOn);
3178
+ }
3160
3179
  function providerTransitionForTurn(bot, runOn) {
3161
3180
  const provider = turnProvider(bot, runOn);
3162
3181
  return provider && computerProviderConfigTransitions.has(provider)
@@ -3648,7 +3667,7 @@ bus.subscribe((event) => {
3648
3667
  pushMessage({
3649
3668
  role: "bot",
3650
3669
  kind: "activity",
3651
- tool: { name: `error: ${event.message.slice(0, 160)}`, ok: false, setup: event.setup },
3670
+ tool: { name: `error: ${event.message.slice(0, 160)}`, ok: false, setup: event.setup, ...(event.terminal ? { terminal: true } : {}) },
3652
3671
  });
3653
3672
  // a setup error means the engine could not even start: the bot is
3654
3673
  // dead until something changes, not merely idle. The next successful
@@ -4576,11 +4595,9 @@ async function startTurn(botId, text, opts) {
4576
4595
  const task = store.taskByThread(bot.id, threadId);
4577
4596
  if (!task)
4578
4597
  throw Object.assign(new Error("no such task"), { status: 404 });
4579
- const instance = opts?.runOn === "cloud"
4580
- ? registry.instances().find((candidate) => candidate.driverKind === "boxAgent") ?? null
4581
- : registry.get(bot.modelSelection.instanceId);
4598
+ const instance = turnInstance(bot, opts?.runOn);
4582
4599
  if (!instance) {
4583
- throw Object.assign(new Error(opts?.runOn === "cloud"
4600
+ throw Object.assign(new Error(opts?.runOn === "cloud" || bot.computer === "cloud" || inheritedTeamComputer(bot)
4584
4601
  ? "the Cloud VM runner is unavailable — configure Box in App Settings"
4585
4602
  : `provider instance "${bot.modelSelection.instanceId}" is unavailable — pick another model in settings`), { status: 409 });
4586
4603
  }
@@ -4791,7 +4808,7 @@ async function startTurn(botId, text, opts) {
4791
4808
  // than the user's home: a bot with file tools and acceptEdits gets a
4792
4809
  // desk, not the whole house — and the workspace is where its
4793
4810
  // MEMORY.md lives. API/box engines have no local filesystem story.
4794
- const worksInWorkspace = instance.driverKind !== "grok" && instance.driverKind !== "boxAgent";
4811
+ const worksInWorkspace = supportsWorkspaceFiles(instance.driverKind);
4795
4812
  if (worksInWorkspace) {
4796
4813
  ensureWorkspace(bot.id);
4797
4814
  // baseline for the journal's turn-boundary diff (see the bus hook)
@@ -5175,13 +5192,14 @@ async function startTurn(botId, text, opts) {
5175
5192
  { id: "profile", label: "Profile changes", text: profilePrompt },
5176
5193
  { id: "learn", label: "Skill authoring", text: learnPrompt },
5177
5194
  { id: "section-context", label: "Section context", text: sectionContextSystemPrompt(bot.section) },
5178
- { id: "memory", label: "Memory", text: privateWorkspace ? memorySystemPrompt(bot.id, { managedWrites: Boolean(integrations.agents) }) : "" },
5195
+ { id: "memory", label: "Memory", text: memorySystemPrompt(bot.id, { managedWrites: Boolean(integrations.agents), fileTools: worksInWorkspace }) },
5179
5196
  { id: "skills", label: "Skills index", text: privateWorkspace ? skillsSystemPrompt(bot.id) : "" },
5180
5197
  { id: "skill-instructions", label: "Skill instructions", text: skillInstructions },
5181
5198
  { id: "playbooks", label: "Playbooks", text: packagePlaybooks },
5182
5199
  { id: "webhook", label: "Webhook provenance", text: opts?.automationSource === "webhook" ? WEBHOOK_PROMPT : "" },
5183
5200
  { id: "mentions", label: "Mentions", text: boundedCoordination && tagged.length ? `The user named these existing teammates: ${tagged.map(b => `${peerName(b.name)} (${b.id})`).join(", ")}. Use coordinate_bots when their contribution is needed; do not substitute native helper agents for these bots.` : mentionPrompt(tagged) },
5184
5201
  ]);
5202
+ runningTurnEngines.set(threadId, instance);
5185
5203
  const dispatch = await guardTurnDispatch(instance.adapter.sendTurn({
5186
5204
  threadId,
5187
5205
  botId: bot.id,
@@ -5458,7 +5476,7 @@ async function interruptRoutineGroupGoal(groupId, threadId, outcome) {
5458
5476
  const bot = speaker ? store.bot(speaker.botId) : undefined;
5459
5477
  cancelGroupTurnOperations(groupId, threadId, outcome);
5460
5478
  revokeInternalCapabilitiesForThread(threadId);
5461
- await (bot ? registry.get(bot.modelSelection.instanceId) : undefined)
5479
+ await (bot ? runningTurnInstance(bot, threadId) : null)
5462
5480
  ?.adapter.interruptTurn(threadId)
5463
5481
  .catch(() => { });
5464
5482
  closeOpenApprovals(threadId);
@@ -5494,7 +5512,7 @@ async function stopBotForEmergencyApprovalDowngrade(botId) {
5494
5512
  cancelGroupTurnOperations(groupTurn.group.id, groupTurn.threadId);
5495
5513
  const results = await Promise.allSettled([
5496
5514
  directStop,
5497
- registry.get(bot.modelSelection.instanceId)?.adapter.interruptTurn(groupTurn.threadId),
5515
+ runningTurnInstance(bot, groupTurn.threadId)?.adapter.interruptTurn(groupTurn.threadId),
5498
5516
  ]);
5499
5517
  closeOpenApprovals(groupTurn.threadId);
5500
5518
  const failure = results.find((result) => result.status === "rejected");
@@ -5573,11 +5591,9 @@ routines = new RoutineManager({
5573
5591
  discardDelegations(commsBus, threadId);
5574
5592
  cancelDirectTurnDispatch(botId, threadId);
5575
5593
  revokeInternalCapabilitiesForThread(threadId);
5576
- const instance = runOn === "cloud"
5577
- ? registry.instances().find((candidate) => candidate.driverKind === "boxAgent") ?? null
5578
- : bot
5579
- ? registry.get(bot.modelSelection.instanceId)
5580
- : null;
5594
+ const instance = bot
5595
+ ? runningTurnInstance(bot, threadId, runOn)
5596
+ : runOn === "cloud" ? registry.instances().find((candidate) => candidate.driverKind === "boxAgent") ?? null : null;
5581
5597
  try {
5582
5598
  await instance?.adapter.interruptTurn(threadId);
5583
5599
  }
@@ -6323,7 +6339,7 @@ setupRetry = 0) {
6323
6339
  const preparedApprovalMode = approvalModeForTurn(bot, Boolean(orchestration?.roomHandoffId));
6324
6340
  const preparedSelection = { ...bot.modelSelection };
6325
6341
  const preparedComposio = bot.composio;
6326
- const instance = registry.get(bot.modelSelection.instanceId);
6342
+ const instance = turnInstance(bot);
6327
6343
  const userName = cfg.profile?.name?.trim() || "User";
6328
6344
  if (providerInstancesChanging.has(bot.modelSelection.instanceId)) {
6329
6345
  onDispatchError?.(`${bot.name}'s provider account is being updated — try again shortly`);
@@ -6446,7 +6462,7 @@ setupRetry = 0) {
6446
6462
  : Boolean(readyGroup && store.groupTaskByThread(readyGroup.id, threadId));
6447
6463
  if (!readyGroup || !stillOwnsThread || !readyGroup.memberIds.includes(readyBot.id))
6448
6464
  return false;
6449
- const setupChanged = registry.get(preparedSelection.instanceId) !== instance ||
6465
+ const setupChanged = turnInstance(readyBot) !== instance ||
6450
6466
  approvalModeForTurn(readyBot, Boolean(orchestration?.roomHandoffId)) !== preparedApprovalMode ||
6451
6467
  readyBot.modelSelection.instanceId !== preparedSelection.instanceId ||
6452
6468
  readyBot.modelSelection.model !== preparedSelection.model ||
@@ -6637,7 +6653,7 @@ setupRetry = 0) {
6637
6653
  const text = `${roomContext}\n\n(Reply to the conversation above as ${bot.name}.)${learnBlock}${cardContinuation ? `\n\n${cardContinuation}` : ""}`;
6638
6654
  // same workspace + memory as a 1:1 turn — the room is a different
6639
6655
  // conversation, not a different bot
6640
- const worksInWorkspace = instance.driverKind !== "grok" && instance.driverKind !== "boxAgent";
6656
+ const worksInWorkspace = supportsWorkspaceFiles(instance.driverKind);
6641
6657
  const workspace = worksInWorkspace ? ensureWorkspace(bot.id) : undefined;
6642
6658
  // a room member's memory writes are journaled the same as a 1:1 turn's
6643
6659
  if (workspace)
@@ -6654,6 +6670,7 @@ setupRetry = 0) {
6654
6670
  if (drift.drift !== Boolean(bot.soulDrift))
6655
6671
  store.patchBot(bot.id, { soulDrift: drift.drift });
6656
6672
  }
6673
+ const roomMemory = memorySystemPrompt(bot.id, { managedWrites: Boolean(integrations.agents), fileTools: worksInWorkspace });
6657
6674
  const roomSystem = buildSystemPrompt(system, store.bot(bot.id)?.soul ?? bot.soul ?? "", [
6658
6675
  { id: "files", label: "File locations", text: workspace ? workspaceLocationsPrompt(bot.id, cwd, readyBot.cwd) : "" },
6659
6676
  { id: "mcp", label: "MCP servers", text: customMcpPrompt(Object.keys(integrations.custom ?? {})) },
@@ -6668,7 +6685,7 @@ setupRetry = 0) {
6668
6685
  // byte-identical. The write guidance follows the tools actually
6669
6686
  // mounted, exactly as the 1:1 path decides it: memory_update is on the
6670
6687
  // agents server, so a room turn with it must be told to use it too.
6671
- { id: "memory", label: "Memory", text: workspace ? `\n${memorySystemPrompt(bot.id, { managedWrites: Boolean(integrations.agents) }).trim()}` : "" },
6688
+ { id: "memory", label: "Memory", text: roomMemory ? `\n${roomMemory.trim()}` : "" },
6672
6689
  { id: "skills", label: "Skills index", text: workspace ? skillsSystemPrompt(bot.id) : "" },
6673
6690
  { id: "skill-instructions", label: "Skill instructions", text: renderSkillInstructions(selectedSkills, { includeRoot: Boolean(workspace) }) },
6674
6691
  { id: "playbooks", label: "Playbooks", text: installedPlaybookInstructions(text, bot.playbooks) },
@@ -6770,6 +6787,7 @@ setupRetry = 0) {
6770
6787
  watchdog.watch(threadId, bot.id);
6771
6788
  onProviderHandshakeStarted?.();
6772
6789
  providerDispatched = true;
6790
+ runningTurnEngines.set(threadId, instance);
6773
6791
  guardTurnDispatch(instance.adapter.sendTurn({
6774
6792
  threadId,
6775
6793
  botId: readyBot.id,
@@ -11976,7 +11994,7 @@ const handleRequest = async (req, res) => {
11976
11994
  : undefined;
11977
11995
  return {
11978
11996
  threadId,
11979
- instance: busy ? registry.get(busy.modelSelection.instanceId) : undefined,
11997
+ instance: busy ? runningTurnInstance(busy, threadId) : undefined,
11980
11998
  };
11981
11999
  });
11982
12000
  // Abort every queued operation before the first provider round trip;
@@ -12646,7 +12664,7 @@ const handleRequest = async (req, res) => {
12646
12664
  if (groupTurn) {
12647
12665
  cancelGroupTurnOperations(groupTurn.group.id, groupTurn.threadId);
12648
12666
  revokeInternalCapabilitiesForThread(groupTurn.threadId);
12649
- await registry.get(existingBot.modelSelection.instanceId)?.adapter.interruptTurn(groupTurn.threadId).catch(() => { });
12667
+ await runningTurnInstance(existingBot, groupTurn.threadId)?.adapter.interruptTurn(groupTurn.threadId).catch(() => { });
12650
12668
  closeOpenApprovals(groupTurn.threadId);
12651
12669
  }
12652
12670
  }
@@ -12746,12 +12764,11 @@ const handleRequest = async (req, res) => {
12746
12764
  }
12747
12765
  await routines.cancelRun(routineRun.id);
12748
12766
  }
12749
- const instance = registry.get(bot.modelSelection.instanceId);
12750
12767
  const groupTurn = activeGroupTurnForBot(bot.id);
12751
12768
  if (groupTurn) {
12752
12769
  cancelGroupTurnOperations(groupTurn.group.id, groupTurn.threadId);
12753
12770
  revokeInternalCapabilitiesForThread(groupTurn.threadId);
12754
- await instance?.adapter.interruptTurn(groupTurn.threadId).catch(() => { });
12771
+ await runningTurnInstance(bot, groupTurn.threadId)?.adapter.interruptTurn(groupTurn.threadId).catch(() => { });
12755
12772
  closeOpenApprovals(groupTurn.threadId);
12756
12773
  }
12757
12774
  }));
@@ -13296,7 +13313,7 @@ const handleRequest = async (req, res) => {
13296
13313
  // loses a race with turn settlement, or the engine cannot steer, the
13297
13314
  // existing server-side queue records it atomically for the next turn.
13298
13315
  if (currentAtStart.busy) {
13299
- const instance = registry.get(currentAtStart.modelSelection.instanceId);
13316
+ const instance = runningTurnInstance(currentAtStart, threadId);
13300
13317
  let steered = false;
13301
13318
  // A live text steer has no image side channel. Keep an attachment
13302
13319
  // message intact for the next ordinary turn, where central image
@@ -14492,7 +14509,7 @@ const handleRequest = async (req, res) => {
14492
14509
  claudeUpdatesInFlight.delete(target.cli);
14493
14510
  }
14494
14511
  }
14495
- // ── per-instance CLI path override (custom builds / versioned bins) ──
14512
+ // ── per-instance settings (CLI/account or API tool support) ──
14496
14513
  // PATCH /api/instances/:id {cli: "/path/to/cli" | ""} — "" reverts to the
14497
14514
  // driver default. Only this idle instance is replaced; siblings keep running.
14498
14515
  const instancePatch = /^\/api\/instances\/([\w.-]+)$/.exec(path);
@@ -14503,7 +14520,7 @@ const handleRequest = async (req, res) => {
14503
14520
  }
14504
14521
  const parsed = instanceSettingsSchema.safeParse(await readBody(req, 16384));
14505
14522
  if (!parsed.success)
14506
- return json(res, 400, { error: "Supply a valid CLI path, account name or configuration directory." });
14523
+ return json(res, 400, { error: "Supply a valid CLI path, account name, configuration directory or boolean tools setting." });
14507
14524
  const body = parsed.data;
14508
14525
  const instanceId = instancePatch[1];
14509
14526
  if (providerConfigBusy)
@@ -14522,6 +14539,12 @@ const handleRequest = async (req, res) => {
14522
14539
  if ((body.displayName !== undefined || body.configDir !== undefined) && entry.driver !== "claudeAgent") {
14523
14540
  return json(res, 400, { error: "Account settings are currently available for Claude only." });
14524
14541
  }
14542
+ if (body.tools !== undefined) {
14543
+ if (!["openai-compat", "grok", "minimax"].includes(entry.driver)) {
14544
+ return json(res, 400, { error: "The tools setting is available for OpenAI-compatible, Grok API and MiniMax API instances only." });
14545
+ }
14546
+ entry.config = { ...entry.config, tools: body.tools };
14547
+ }
14525
14548
  if (body.displayName !== undefined)
14526
14549
  entry.displayName = body.displayName;
14527
14550
  if (body.configDir !== undefined) {
@@ -14939,7 +14962,10 @@ const handleRequest = async (req, res) => {
14939
14962
  // patch SELECTS, not the one already saved, or pasting a Cartesia key
14940
14963
  // while switching from ElevenLabs validates against the wrong service
14941
14964
  const newTts = patch.tts;
14942
- if (newTts?.key?.trim()) {
14965
+ // Chatterbox has no key at all — its credential is a local server
14966
+ // address, and an ElevenLabs round trip would be the wrong test
14967
+ const selectedVoiceProvider = newTts?.provider ?? cfg.tts?.provider ?? "elevenlabs";
14968
+ if (newTts?.key?.trim() && selectedVoiceProvider !== "chatterbox") {
14943
14969
  const check = await tts.verifyKey(newTts.key.trim());
14944
14970
  if (!check.ok)
14945
14971
  return json(res, 400, { error: check.message });
@@ -15525,7 +15551,7 @@ const handleRequest = async (req, res) => {
15525
15551
  const activeBoxTurn = botHasActiveTurn(botId);
15526
15552
  if (["provision", "sleep"].includes(m[2]) && activeBoxTurn) {
15527
15553
  return json(res, 409, {
15528
- error: "this bot's cloud computer is being used by an active turn — interrupt it first",
15554
+ error: CLOUD_COMPUTER_BUSY_ERROR,
15529
15555
  });
15530
15556
  }
15531
15557
  // Input validity is independent of destination authorization. Preserve
@@ -15535,9 +15561,9 @@ const handleRequest = async (req, res) => {
15535
15561
  if (m[2] === "exec") {
15536
15562
  const body = await readBody(req);
15537
15563
  boxCommand = String(body?.command ?? "");
15538
- if (boxCommand.length > MAX_REMOTE_COMMAND_LENGTH) {
15564
+ if (boxCommand.length > box.MAX_REMOTE_COMMAND_LENGTH) {
15539
15565
  return json(res, 400, {
15540
- error: `command is too long (maximum ${MAX_REMOTE_COMMAND_LENGTH} characters)`,
15566
+ error: `command is too long (maximum ${box.MAX_REMOTE_COMMAND_LENGTH} characters)`,
15541
15567
  });
15542
15568
  }
15543
15569
  }
@@ -32,7 +32,6 @@ export function resolveProxy(relative) {
32
32
  * the check that would have caught the 0.1.24 breakage. */
33
33
  export const SPAWNED_PROXIES = {
34
34
  browser: resolveProxy("browser-proxy"),
35
- computer: resolveProxy("computer-proxy"),
36
35
  localComputer: resolveProxy("local-computer-proxy"),
37
36
  permission: resolveProxy("permission-proxy"),
38
37
  containerMcp: resolveProxy("container-mcp"),
@@ -8,11 +8,13 @@ const nodeSchema = z.object({
8
8
  key: z.string(), text: z.string(), createdAt: z.number(),
9
9
  status: z.enum(["source", "queued", "running", "waiting", "resume", "completed", "failed", "cancelled"]),
10
10
  result: z.string().default(""), reported: z.boolean().default(false),
11
- executions: z.number().int().nonnegative().default(0),
11
+ executions: z.number().int().nonnegative().default(0), startedAt: z.number().optional(),
12
12
  approvalGranted: z.boolean().default(false),
13
13
  kind: z.enum(["work", "assignment"]).default("work"),
14
14
  });
15
- export const ROOM_HANDOFF_LIMITS = { depth: 4, requests: 24, executions: 48, lifetimeMs: 30 * 60_000 };
15
+ export const ROOM_HANDOFF_LIMITS = { depth: 4, requests: 24, executions: 48, lifetimeMs: 30 * 60_000, minRunwayMs: 10 * 60_000 };
16
+ /** Renders elapsed milliseconds as whole minutes, or seconds under one minute. */
17
+ const duration = (ms) => ms >= 60_000 ? `${Math.floor(ms / 60_000)}m` : `${Math.floor(ms / 1000)}s`;
16
18
  const terminal = (n) => ["completed", "failed", "cancelled"].includes(n.status);
17
19
  /** A bounded tree of addressed room turns. Waiting for children never holds a
18
20
  * room/provider queue; reporting is data, and only the named parent is resumed.
@@ -65,6 +67,31 @@ export class RoomHandoffs {
65
67
  }
66
68
  return path;
67
69
  }
70
+ /** The earliest moment this node may be failed for lifetime: the tree
71
+ * ceiling or, for a running node, its own start plus a minimum runway. */
72
+ deadline(n) {
73
+ const anchor = n.status === "running" ? n.startedAt ?? n.createdAt : n.createdAt;
74
+ return Math.max(this.root(n).createdAt + ROOM_HANDOFF_LIMITS.lifetimeMs, anchor + ROOM_HANDOFF_LIMITS.minRunwayMs);
75
+ }
76
+ /** An ancestor past its ceiling is not failed while a descendant is still
77
+ * running inside its own runway; cancelling would cascade into that work. */
78
+ protectsRunner(n) {
79
+ return this.children(n.id).some(c => !terminal(c) && ((c.status === "running" && this.now() <= this.deadline(c)) || this.protectsRunner(c)));
80
+ }
81
+ /** A parent still owes the follow-up execution that decides on its
82
+ * children's results; the ceiling defers to that execution's own runway. */
83
+ owesFollowUp(n) {
84
+ if (n.status === "resume")
85
+ return true;
86
+ if (n.status !== "waiting")
87
+ return false;
88
+ const children = this.children(n.id);
89
+ return (children.length > 0 && children.every(c => terminal(c))) || children.some(c => this.owesFollowUp(c));
90
+ }
91
+ /** Names the budget, the node's status, and the elapsed tree time. */
92
+ lifetimeError(n) {
93
+ return `Room handoff lifetime budget exhausted: node was ${n.status} after ${duration(this.now() - this.root(n).createdAt)} of the ${duration(ROOM_HANDOFF_LIMITS.lifetimeMs)} tree lifetime`;
94
+ }
68
95
  enqueue(source, generation, parentId, target, key, text, approvalGranted = false, rework = false, sourceText = "") {
69
96
  if (this.loadError)
70
97
  throw new Error(this.loadError);
@@ -100,8 +127,14 @@ export class RoomHandoffs {
100
127
  throw new Error("Room handoff depth limit reached");
101
128
  const root = fresh ? parent : this.root(parent);
102
129
  const count = [...this.nodes.values()].filter(n => n.rootId === parent.rootId && n.parentId).length;
103
- if (count >= ROOM_HANDOFF_LIMITS.requests || this.now() - root.createdAt > ROOM_HANDOFF_LIMITS.lifetimeMs)
130
+ if (count >= ROOM_HANDOFF_LIMITS.requests)
104
131
  throw new Error("Room handoff budget exhausted");
132
+ // Refuse work the tree's lifetime budget cannot honestly serve: a node
133
+ // accepted in the root's last minutes would be doomed at enqueue time.
134
+ const remaining = ROOM_HANDOFF_LIMITS.lifetimeMs - (this.now() - root.createdAt);
135
+ if (remaining < ROOM_HANDOFF_LIMITS.minRunwayMs) {
136
+ throw new Error(`Room handoff budget exhausted: only ${duration(Math.max(remaining, 0))} of the ${duration(ROOM_HANDOFF_LIMITS.lifetimeMs)} tree lifetime remains`);
137
+ }
105
138
  // Retain a bounded audit history without evicting active requests.
106
139
  if (this.nodes.size >= 1000) {
107
140
  const oldRoots = [...this.nodes.values()].filter(n => !n.parentId && terminal(n)).sort((a, b) => a.createdAt - b.createdAt);
@@ -210,14 +243,18 @@ export class RoomHandoffs {
210
243
  tick() {
211
244
  if (this.loadError)
212
245
  return;
246
+ // Validate and expire deepest nodes first so each one is failed with its
247
+ // own status; an ancestor's cancellation then only sweeps what is left.
248
+ for (const n of [...this.nodes.values()].reverse()) {
249
+ if (terminal(n))
250
+ continue;
251
+ const error = this.hooks.validate(n, n.parentId ? this.nodes.get(n.parentId) : undefined);
252
+ if (error || (this.now() > this.deadline(n) && !this.protectsRunner(n) && !this.owesFollowUp(n))) {
253
+ this.cancelTree(n, error ?? this.lifetimeError(n), "failed");
254
+ }
255
+ }
213
256
  for (const n of this.nodes.values()) {
214
257
  const parent = n.parentId ? this.nodes.get(n.parentId) : undefined;
215
- if (!terminal(n)) {
216
- const error = this.hooks.validate(n, parent);
217
- if (error || this.now() - this.root(n).createdAt > ROOM_HANDOFF_LIMITS.lifetimeMs) {
218
- this.cancelTree(n, error ?? "Room request timed out", "failed");
219
- }
220
- }
221
258
  if (terminal(n) && parent && !n.reported) {
222
259
  this.hooks.report(n, parent);
223
260
  n.reported = true;
@@ -253,6 +290,7 @@ export class RoomHandoffs {
253
290
  const childCount = this.children(n.id).length;
254
291
  root.executions += executionCost;
255
292
  n.status = "running";
293
+ n.startedAt = this.now();
256
294
  this.publish(n, root);
257
295
  const controller = new AbortController();
258
296
  this.controllers.set(n.id, controller);
@@ -0,0 +1,83 @@
1
+ export const DEFAULT_MODEL = "chatterbox-turbo";
2
+ /** What the picker falls back to when the server has no /v1/models. Most
3
+ * Chatterbox servers speak with one built-in voice unless told otherwise,
4
+ * so one honest entry beats a list of names the server may reject. */
5
+ export const FALLBACK_VOICES = [
6
+ { id: "default", label: "Default", description: "the server's built-in Chatterbox voice" },
7
+ ];
8
+ /** Accept "http://127.0.0.1:4123", the same with a trailing slash, and a
9
+ * base that already ends in /v1 — the address is typed by hand, so meet
10
+ * it halfway rather than 404ing over a slash. */
11
+ export function apiRoot(baseUrl) {
12
+ const trimmed = baseUrl.trim().replace(/\/+$/, "");
13
+ return /\/v1$/i.test(trimmed) ? trimmed : `${trimmed}/v1`;
14
+ }
15
+ async function safeJson(res) {
16
+ try {
17
+ return await res.json();
18
+ }
19
+ catch {
20
+ return null;
21
+ }
22
+ }
23
+ /** Prefer the server's own words over anything we can invent — wrapper
24
+ * services vary, and their "model not loaded" says more than a 500. */
25
+ function message(status, what, body) {
26
+ const theirs = (typeof body?.error === "string" && body.error.trim()) ||
27
+ (typeof body?.detail === "string" && body.detail.trim()) ||
28
+ (typeof body?.message === "string" && body.message.trim()) ||
29
+ "";
30
+ return theirs ? `${what} failed: ${theirs}` : `${what} failed (${status})`;
31
+ }
32
+ /** One bounded request shape for both calls: a dead server is the common
33
+ * failure, and it must be named in the user's words, quickly. */
34
+ async function request(url, init, what, baseUrl) {
35
+ try {
36
+ return await fetch(url, init);
37
+ }
38
+ catch (e) {
39
+ if (e instanceof Error && (e.name === "TimeoutError" || e.name === "AbortError")) {
40
+ throw new Error(`${what} timed out — the Chatterbox server at ${baseUrl.trim()} did not answer`);
41
+ }
42
+ throw new Error(`couldn't reach the Chatterbox server at ${baseUrl.trim()} — check that it is running`);
43
+ }
44
+ }
45
+ export async function listChatterboxVoices(baseUrl) {
46
+ try {
47
+ const res = await request(`${apiRoot(baseUrl)}/models`, { headers: { accept: "application/json" }, signal: AbortSignal.timeout(10_000) }, "listing voices", baseUrl);
48
+ if (res.ok) {
49
+ const body = await safeJson(res);
50
+ const voices = (body?.data ?? [])
51
+ .map((m) => ({ id: String(m?.id ?? ""), label: String(m?.id ?? "") }))
52
+ .filter((v) => v.id);
53
+ if (voices.length)
54
+ return voices;
55
+ }
56
+ }
57
+ catch {
58
+ // fall through: the built-in entry below beats an empty picker
59
+ }
60
+ return FALLBACK_VOICES;
61
+ }
62
+ /** Ask for wav because every browser plays it and every OpenAI-compatible
63
+ * server can produce it; the server's own content-type wins when it says
64
+ * something else, so returned bytes are never mislabeled. */
65
+ export async function synthesizeChatterbox(text, voiceId, baseUrl, model) {
66
+ const res = await request(`${apiRoot(baseUrl)}/audio/speech`, {
67
+ method: "POST",
68
+ headers: { "content-type": "application/json", accept: "audio/wav" },
69
+ body: JSON.stringify({ model: model?.trim() || DEFAULT_MODEL, input: text, voice: voiceId, response_format: "wav" }),
70
+ signal: AbortSignal.timeout(60_000),
71
+ }, "speaking", baseUrl);
72
+ if (!res.ok)
73
+ throw new Error(message(res.status, "speaking", await safeJson(res)));
74
+ const header = (res.headers.get("content-type") ?? "").split(";")[0].trim().toLowerCase();
75
+ const mime = header === "audio/mpeg" || header === "audio/mp3"
76
+ ? "audio/mpeg"
77
+ : header === "audio/x-wav"
78
+ ? "audio/wav"
79
+ : header.startsWith("audio/")
80
+ ? header
81
+ : "audio/wav";
82
+ return { bytes: new Uint8Array(await res.arrayBuffer()), mime };
83
+ }
@@ -1,3 +1,4 @@
1
+ import * as chatterbox from "./chatterbox.js";
1
2
  import * as elevenlabs from "./elevenlabs.js";
2
3
  import * as systemVoices from "./system-voices.js";
3
4
  export class NoVoiceConfigured extends Error {
@@ -5,52 +6,73 @@ export class NoVoiceConfigured extends Error {
5
6
  // runs under `node --experimental-strip-types`, which is strip-ONLY, so a
6
7
  // parameter property is rejected at load time even though it typechecks
7
8
  reason;
8
- constructor(reason) {
9
- super(reason === "key"
10
- ? "Add an ElevenLabs key in Settings on the computer to turn on voice."
11
- : "Pick a voice in the agent profile.");
9
+ constructor(reason, hint) {
10
+ super(hint ??
11
+ (reason === "key"
12
+ ? "Add an ElevenLabs key in Settings on the computer to turn on voice."
13
+ : "Pick a voice in the agent profile."));
12
14
  this.reason = reason;
13
15
  }
14
16
  }
15
17
  export function voiceProvider(cfg) {
16
- return cfg.tts?.provider === "system" ? "system" : "elevenlabs";
18
+ return cfg.tts?.provider === "system" || cfg.tts?.provider === "chatterbox" ? cfg.tts.provider : "elevenlabs";
17
19
  }
18
20
  /** The system provider needs no credential — it is only ever offered where
19
21
  * the platform actually has it, so "configured" means "this engine can
20
22
  * speak", not "a key is on file". */
21
23
  export function providerConfigured(cfg) {
22
- return voiceProvider(cfg) === "system" ? systemVoices.systemVoicesAvailable() : Boolean(cfg.tts?.key);
24
+ const provider = voiceProvider(cfg);
25
+ if (provider === "system")
26
+ return systemVoices.systemVoicesAvailable();
27
+ if (provider === "chatterbox")
28
+ return Boolean(cfg.tts?.baseUrl?.trim());
29
+ return Boolean(cfg.tts?.key);
23
30
  }
24
31
  export function voiceConfigured(cfg) {
25
- if (voiceProvider(cfg) === "system") {
32
+ const provider = voiceProvider(cfg);
33
+ if (provider === "system") {
26
34
  return systemVoices.systemVoicesAvailable() && Boolean(cfg.tts?.voice);
27
35
  }
36
+ if (provider === "chatterbox")
37
+ return Boolean(cfg.tts?.baseUrl?.trim() && cfg.tts?.voice);
28
38
  return Boolean(cfg.tts?.key && cfg.tts?.voice);
29
39
  }
30
40
  /** A per-bot voice is a complete choice too; it should not be blocked just
31
41
  * because the app-wide fallback has not been selected yet. */
32
42
  export function voiceReady(cfg, voiceId) {
33
- if (voiceProvider(cfg) === "system") {
43
+ const provider = voiceProvider(cfg);
44
+ if (provider === "system") {
34
45
  return systemVoices.systemVoicesAvailable() && Boolean(voiceId || cfg.tts?.voice);
35
46
  }
47
+ if (provider === "chatterbox")
48
+ return Boolean(cfg.tts?.baseUrl?.trim() && (voiceId || cfg.tts?.voice));
36
49
  return Boolean(cfg.tts?.key && (voiceId || cfg.tts?.voice));
37
50
  }
38
51
  /** What the settings panel needs. Never includes the key — same write-only
39
- * rule as every other credential. */
52
+ * rule as every other credential. baseUrl and model are Chatterbox
53
+ * settings, not credentials, so they come back in full. */
40
54
  export function describeVoice(cfg) {
55
+ const provider = voiceProvider(cfg);
41
56
  return {
42
57
  configured: providerConfigured(cfg),
43
58
  ready: voiceConfigured(cfg),
44
59
  voice: cfg.tts?.voice ?? "",
45
- provider: voiceProvider(cfg),
60
+ provider,
61
+ baseUrl: provider === "chatterbox" ? (cfg.tts?.baseUrl ?? "") : "",
62
+ model: provider === "chatterbox" ? (cfg.tts?.model ?? "") : "",
46
63
  };
47
64
  }
48
65
  export function verifyKey(key) {
49
66
  return elevenlabs.verifyKey(key);
50
67
  }
51
68
  export async function listVoices(cfg, run) {
52
- if (voiceProvider(cfg) === "system")
69
+ const provider = voiceProvider(cfg);
70
+ if (provider === "system")
53
71
  return systemVoices.listSystemVoices(run);
72
+ if (provider === "chatterbox") {
73
+ const baseUrl = cfg.tts?.baseUrl?.trim();
74
+ return baseUrl ? chatterbox.listChatterboxVoices(baseUrl) : [];
75
+ }
54
76
  const key = cfg.tts?.key;
55
77
  if (!key)
56
78
  return [];
@@ -59,7 +81,8 @@ export async function listVoices(cfg, run) {
59
81
  /** Synthesize one utterance. Throws NoVoiceConfigured when there is nothing
60
82
  * to speak with, which the route turns into a 409 the client can explain. */
61
83
  export function speak(cfg, text, voiceId, run) {
62
- if (voiceProvider(cfg) === "system") {
84
+ const provider = voiceProvider(cfg);
85
+ if (provider === "system") {
63
86
  const voice = voiceId || cfg.tts?.voice;
64
87
  // An injected runner is the cross-platform test seam for `/usr/bin/say`;
65
88
  // production calls omit it and remain strictly Darwin-gated.
@@ -69,6 +92,16 @@ export function speak(cfg, text, voiceId, run) {
69
92
  throw new NoVoiceConfigured("voice");
70
93
  return systemVoices.synthesizeSystem(text, voice, run);
71
94
  }
95
+ if (provider === "chatterbox") {
96
+ const baseUrl = cfg.tts?.baseUrl?.trim();
97
+ if (!baseUrl) {
98
+ throw new NoVoiceConfigured("key", "Add the address of your Chatterbox server in Settings on the computer to turn on voice.");
99
+ }
100
+ const voice = voiceId || cfg.tts?.voice;
101
+ if (!voice)
102
+ throw new NoVoiceConfigured("voice");
103
+ return chatterbox.synthesizeChatterbox(text, voice, baseUrl, cfg.tts?.model);
104
+ }
72
105
  const key = cfg.tts?.key;
73
106
  if (!key)
74
107
  throw new NoVoiceConfigured("key");