openmausbot 0.1.82 → 0.1.84

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/dist/assets/index-BbU5REzd.js +310 -0
  2. package/dist/assets/{index-CKysBq-V.css → index-CLGfYlx_.css} +1 -1
  3. package/dist/assets/{index-5HVwp5m2.js → index-DNa2umw-.js} +1 -1
  4. package/dist/index.html +2 -2
  5. package/dist-server/container-mcp.js +20 -4
  6. package/dist-server/drivers/agents-proxy.js +30 -4
  7. package/dist-server/index.js +1716 -581
  8. package/dist-server/mcp-gate.js +2 -1
  9. package/dist-server/openmausbot.js +224 -63
  10. package/dist-server/pair-cli.js +224 -63
  11. package/dist-server/server/auto-approve.js +15 -1
  12. package/dist-server/server/auto-vm-claims.js +22 -0
  13. package/dist-server/server/browser-bundle-release.js +8 -7
  14. package/dist-server/server/browser-engine.js +1 -1
  15. package/dist-server/server/channel-queue.js +58 -0
  16. package/dist-server/server/checkpoints.js +9 -5
  17. package/dist-server/server/chief-of-staff.js +5 -2
  18. package/dist-server/server/computer-wait.js +33 -0
  19. package/dist-server/server/config.js +58 -3
  20. package/dist-server/server/contracts.js +0 -7
  21. package/dist-server/server/delta-context.js +258 -0
  22. package/dist-server/server/drivers/acp/gemini.js +11 -1
  23. package/dist-server/server/drivers/acp/qwen.js +16 -1
  24. package/dist-server/server/drivers/agents-proxy.js +32 -4
  25. package/dist-server/server/drivers/claude.js +26 -8
  26. package/dist-server/server/drivers/codex.js +79 -6
  27. package/dist-server/server/drivers/pi.js +2 -1
  28. package/dist-server/server/env-path.js +6 -1
  29. package/dist-server/server/incidents.js +104 -0
  30. package/dist-server/server/index.js +886 -134
  31. package/dist-server/server/local-routing.js +10 -4
  32. package/dist-server/server/notify.js +3 -1
  33. package/dist-server/server/peer-roster.js +37 -1
  34. package/dist-server/server/recent-work.js +4 -1
  35. package/dist-server/server/redact.js +5 -48
  36. package/dist-server/server/room-handoffs.js +79 -16
  37. package/dist-server/server/skill-fetch.js +1 -1
  38. package/dist-server/server/steer-queue.js +29 -0
  39. package/dist-server/server/store.js +43 -3
  40. package/dist-server/server/thread-retention.js +50 -0
  41. package/dist-server/server/turn-context.js +7 -0
  42. package/dist-server/server/turn-resources.js +6 -0
  43. package/dist-server/shared/approval-mode.js +8 -5
  44. package/dist-server/shared/inspector.js +1 -0
  45. package/dist-server/shared/json.js +1 -0
  46. package/dist-server/shared/notification.js +1 -0
  47. package/dist-server/shared/redact.js +61 -0
  48. package/dist-server/shared/routines.js +1 -0
  49. package/dist-server/shared/runtime-events.js +1 -0
  50. package/dist-server/shared/webhooks.js +1 -0
  51. package/dist-server/shared/wire.js +8 -0
  52. package/dist-server/vps-container-mcp.js +20 -4
  53. package/package.json +1 -1
  54. package/dist/assets/index-BLLsDK2F.js +0 -309
@@ -2,7 +2,7 @@
2
2
  // (upstream rule): the React app dispatches typed commands over HTTP and
3
3
  // folds one SSE event stream; every provider process runs here.
4
4
  import { createHash, randomBytes, randomUUID, timingSafeEqual } from "node:crypto";
5
- import { existsSync, readFileSync, rmSync, unlinkSync } from "node:fs";
5
+ import { existsSync, readFileSync, rmSync } from "node:fs";
6
6
  import { rm as removeDirectory } from "node:fs/promises";
7
7
  import { createServer } from "node:http";
8
8
  import { extname, join } from "node:path";
@@ -16,7 +16,7 @@ import { CLOUD_COMPUTER_BUSY_ERROR } from "../shared/computer-contention.js";
16
16
  import { approvalModeFor, supportsApprovalMode, modelSwitchNeedsAsk, isEmergencyApprovalDowngrade, isApprovalMode, } from "../shared/approval-mode.js";
17
17
  import { escapeAttribute } from "../src/lib/composer-attachments.js";
18
18
  import { CREDENTIAL_TARGETS, credentialResumeOutcome, credentialIsConfigured, isReusableCredentialRequest, isCredentialTargetId, } from "../shared/credential-request.js";
19
- import { approvalHeldNote, approvalHeldReason, approvalModeForOrigin, autoVerdict, deliverFullAccessApproval } from "./auto-approve.js";
19
+ import { approvalHeldNote, approvalHeldReason, approvalModeForOrigin, autoVerdict, deliverFullAccessApproval, delegationInheritsFullAccess } from "./auto-approve.js";
20
20
  import { updateClaudeCli } from "./claude-update.js";
21
21
  import { configuredAccountDirectory, assertSeparateClaudeAccount, claudeAccountInfo, createClaudeAccountSchema, instanceSettingsSchema, newClaudeAccount } from "./claude-accounts.js";
22
22
  import { BrowserCleanupCoordinator, finalizeBrowserCleanupMutation, requireBrowserCleanupAcknowledged, } from "./browser-lifecycle-cleanup.js";
@@ -31,15 +31,17 @@ import { groupTurnCwd } from "./room-cwd.js";
31
31
  import { RoomTurnDeadline, RoomTurnStallRegistry, roomTurnTimeoutMessage } from "./room-turn-timeout.js";
32
32
  import * as box from "./box.js";
33
33
  import { TeamComputers, teamComputerAssignment, teamComputerCreate, teamComputerOwner } from "./team-computers.js";
34
+ import { isEffortLevel } from "../shared/wire.js";
34
35
  import { boxCreateRecoverySnapshot, retireDeletedBoxCreate } from "./box-create-idempotency.js";
35
36
  import { boxDeletionSnapshot } from "./box-delete-journal.js";
36
37
  import { boxAccountResourceChangeError, cloudBackendChangeError, vpsAliasResourceChangeError, } from "./cloud-backend.js";
37
38
  import * as composio from "./composio.js";
38
39
  import { chiefOfStaffSystemPrompt } from "./chief-of-staff.js";
39
- import { canAccessTeam, canReachPeer, peerAllowed, peerName, peerRosterSystemPrompt, peerStatus, peerStatusWords, reachablePeers, roomPeerRosterSystemPrompt, roomRosterLine } from "./peer-roster.js";
40
+ import { canAccessTeam, canReachPeer, peerAllowed, peerName, peerRosterSystemPrompt, peerStatus, peerStatusWords, reachablePeers, resolveTeammate, roomPeerRosterSystemPrompt, roomRosterLine } from "./peer-roster.js";
40
41
  import { openMausStatusSystemPrompt } from "./openmaus-status-capsule.js";
41
42
  import { containerComputerAction, containerComputerExists, containerComputerFrame, containerComputerMcp, containerComputerScreenshot, containerComputerStatus, containerRuntimeStatus, localVmRecreatableOnDemand, perBotLocalVmTarget, SHARED_LOCAL_VM_TARGET, setupCommands, } from "./container-computer.js";
42
- import { ensureDirs, instanceConfigs, loadConfig, providerReloadKeys, localVmMaxInstances, localVmMode, parseConfigPatch, roomTurnTimeoutMinutes, maxConcurrentBotThreads, threadEventLogMaxBytes, saveConfig, showToolCallsEnabled, claudeUserMcpEnabled, skillAuthoringEnabled, sharedComputersEnabled, builtInBrowserEnabled, browserProfileReplacementConflict, browserProfilePartitionTarget, syncCredentialEnv, withInstanceCli, persistableInstanceConfigs, vpsSshAlias, DATA_DIR, EVENTS_DIR, NATIVE_DIR, customMcpServers, } from "./config.js";
43
+ import { ensureDirs, instanceConfigs, loadConfig, providerReloadKeys, localVmMaxInstances, localVmMode, parseConfigPatch, roomTurnTimeoutMinutes, threadEventLogMaxBytes, maxConcurrentBotThreads, threadEventLogRetentionDays, saveConfig, showToolCallsEnabled, claudeUserMcpEnabled, skillAuthoringEnabled, sharedComputersEnabled, builtInBrowserEnabled, browserProfileReplacementConflict, browserProfilePartitionTarget, syncCredentialEnv, withInstanceCli, persistableInstanceConfigs, vpsSshAlias, DATA_DIR, EVENTS_DIR, NATIVE_DIR, customMcpServers, roomHandoffLimits, } from "./config.js";
44
+ import { sweepThreadEventLogs } from "./thread-retention.js";
43
45
  import { ComputerControl } from "./computer-control.js";
44
46
  import { augmentedPath, findCliCandidates, resetPathCache } from "./env-path.js";
45
47
  import { registerEnginesBinDir } from "./engine-install.js";
@@ -51,7 +53,7 @@ import { entitled } from "./enterprise.js";
51
53
  import { HOSTED_CONTRACT_HEADER, HOSTED_CONTRACT_METADATA, HOSTED_CONTRACT_VERSION } from "./hosted-contract.js";
52
54
  import { describeSpawnFailure, execCli } from "./procs.js";
53
55
  import { blockedTarget, buildNotification } from "./notify.js";
54
- import { isEffortLevel, isModelVariant, newId, } from "./contracts.js";
56
+ import { isModelVariant, newId, } from "./contracts.js";
55
57
  import { RETRY_MAX_ATTEMPTS } from "./drivers/retry.js";
56
58
  import { decodeGeneratedImage } from "./generated-image.js";
57
59
  import { MAX_MCP_SERVERS, listMcpServers, mcpServerNameError, parseMcpServerMutation, parseMcpServersImport, parseStoredMcpServer, } from "./mcp-registry.js";
@@ -62,13 +64,14 @@ import { getOrCreateChannel, mirrorActivity, mirrorExchange, mirrorReply } from
62
64
  import { readMessageText, recallMessages, recentMessages, searchMessages, closeMessageDb, chatFollowups, cancelledChatFollowup, settleChatFollowups } from "./message-db.js";
63
65
  import { briefCrossingLabel, claimRecallCrossings, recallCrossingLabel } from "./recall-disclosure.js";
64
66
  import { parseSince, recentWork, recentWorkPrompt, turnOutcomeLine } from "./recent-work.js";
67
+ import { chiefForBot, INCIDENTS_THREAD_TITLE, IncidentLedger, incidentChip, incidentText } from "./incidents.js";
65
68
  /** A session_read answer competes with the transcript for the context
66
69
  * window; a computer-use turn's output can run to hundreds of KB. */
67
70
  const SESSION_READ_MAX_CHARS = 8_000;
68
71
  import { promptWithReply, transcriptText } from "./replies.js";
69
72
  import { _loadPending, buildDelegationFailurePrompt, buildDelegationRevivalPrompt, DELEGATION_TTL_MS, DelegationWakeBudget, discardDelegations, drainDelegations, expireStaleDelegations, findDelegationReceipt, pendingDelegationInfo, pendingDelegationSnapshot, pendingThreads, queueDelegation, recordDelegationReceipt, releaseDelegationsWaitingOn, summarizeDelegatedActivity } from "./delegations.js";
70
- import { cancelSteeredMessage, drainSteeredMessages, onSteeredQueueChange, queuedSteerSnapshot, queuedSteeredMessage, queuedThreadPosition, queueSteeredMessage, restoreSteeredMessages, } from "./steer-queue.js";
71
- import { cancelChannelMessage, drainChannelMessages, queuedChannelMessage, queueChannelMessage, restoreChannelMessages, } from "./channel-queue.js";
73
+ import { cancelSteeredMessage, drainSteeredMessages, holdSteeredQueue, onSteeredQueueChange, queuedSteerSnapshot, queuedSteeredMessage, queuedThreadPosition, queueSteeredMessage, restoreHeldSteeredQueue, restoreSteeredMessages, settleHeldSteeredQueue, } from "./steer-queue.js";
74
+ import { cancelChannelMessage, drainChannelMessages, holdChannelQueue, queuedChannelMessage, queueChannelMessage, restoreChannelMessages, restoreHeldChannelQueue, resolveHeldReplyTarget, settleHeldChannelQueueHead, } from "./channel-queue.js";
72
75
  import { acceptedSendMatch, parseSendId, sendFingerprint, SendSequencer, } from "./send-idempotency.js";
73
76
  import { EventBus } from "./harness/bus.js";
74
77
  import { ProviderRegistry } from "./harness/registry.js";
@@ -77,10 +80,11 @@ import { selectDefaultModelSelection } from "./default-model-selection.js";
77
80
  import { cancelPeerApprovalsFor, cancelPeerApprovalsForThread, dismissStalePeerCards, requestPeerApproval, resolvePeerComms } from "./peer-approval.js";
78
81
  import { peerProvenanceNote, withPeerProvenance } from "./peer-provenance.js";
79
82
  import { decideRoomPost, emptyRoomPostBudget } from "./room-post-budget.js";
80
- import { isProjectEmoji, mentionedBots, roomResponders, sectionKey, Store, } from "./store.js";
83
+ import { isProjectEmoji, mentionedBots, roomResponders, sectionKey, Store, toWireTask, } from "./store.js";
81
84
  import * as tts from "./tts/index.js";
82
85
  import { narrateTool, toUtterances } from "./tts/speech-text.js";
83
- import { buildRecoveryText, buildTurnContext, engineIsFresh } from "./turn-context.js";
86
+ import { buildRecoveryText, buildTurnContext, engineIsFresh, peerMessageText } from "./turn-context.js";
87
+ import { Handoffs, handedStateUsable, recordHanded, renderUnseen, sessionStart, unseenMessages, withUnseenMessages } from "./delta-context.js";
84
88
  import { extractTurnImages } from "./turn-images.js";
85
89
  import { TurnWatchdog } from "./turn-watchdog.js";
86
90
  import { TurnResources, workspaceResource } from "./turn-resources.js";
@@ -135,6 +139,8 @@ import { createTeamBackup, importTeamBackup } from "./team-backup.js";
135
139
  import { MAX_TEAM_BACKUP_BYTES } from "../shared/team-backup.js";
136
140
  import { shouldMountLocalComputer } from "./local-routing.js";
137
141
  import { autoLocalVmAttachable } from "./container-computer.js";
142
+ import { startAutoVmClaim } from "./auto-vm-claims.js";
143
+ import { computerFreeText, computerStillBusyText, computerWaitEndedText, computerWaitingText } from "./computer-wait.js";
138
144
  import { modelContextWindow } from "./model-context-window.js";
139
145
  import { parseSurface, resolveSurface, surfaceLabel, surfaceOfComputerKind, surfacePrompt } from "./surface.js";
140
146
  import { PendingTurnCancellations, ProviderTurnGenerationRegistry, RetiredTurnRegistry, guardTurnDispatch, isTurnAdmissionBlocked, isTurnEventQuarantined, } from "./turn-dispatch-guard.js";
@@ -603,6 +609,8 @@ function claimTurnResource(owner, resource) {
603
609
  function releaseTurnResources(owner) {
604
610
  if (!owner)
605
611
  return;
612
+ if (autoVmClaims.get(owner.threadId)?.owner.generation === owner.generation)
613
+ autoVmClaims.delete(owner.threadId);
606
614
  if (settlingResourceOwners.get(owner.threadId) === owner.generation)
607
615
  settlingResourceOwners.delete(owner.threadId);
608
616
  turnResources.release(owner);
@@ -620,6 +628,9 @@ async function bindTurnComputer(owner, resource, exclusive = false) {
620
628
  const active = () => activeInternalGenerationByThread.get(owner.threadId) === owner.generation &&
621
629
  turnResourceOwners.get(owner.threadId)?.generation === owner.generation;
622
630
  let waitingMessage;
631
+ // Who holds the desktop, as the chip and the give-up error name them: a
632
+ // bot running a titled thread, or a room. Read once, when the wait begins.
633
+ let holder;
623
634
  const deadline = Date.now() + GROUP_GOAL_WAIT_MAX_MS;
624
635
  try {
625
636
  while (true) {
@@ -631,23 +642,25 @@ async function bindTurnComputer(owner, resource, exclusive = false) {
631
642
  const blocker = turnResources.blocker(resource, owner);
632
643
  const holderBot = blocker && store.botByThread(blocker.threadId);
633
644
  const holderTask = holderBot && blocker && store.taskByThread(holderBot.id, blocker.threadId);
634
- const holder = holderBot ? `${holderBot.name}${holderTask?.title ? ` / ${holderTask.title}` : ""}`
635
- : blocker && store.groupByThread(blocker.threadId)?.name;
645
+ const holderRoom = !holderBot && blocker ? store.groupByThread(blocker.threadId) : null;
646
+ holder = holderBot
647
+ ? { name: holderBot.name, ...(holderTask?.title ? { task: holderTask.title } : {}) }
648
+ : holderRoom ? { name: holderRoom.name } : undefined;
636
649
  waitingMessage = store.appendMessage(owner.threadId, {
637
650
  role: "bot", kind: "activity",
638
- tool: { name: `Waiting for computer${holder ? ` — ${holder} is using it` : ""}; will continue automatically` },
651
+ tool: { name: computerWaitingText(holder) },
639
652
  ...(holderBot && holderTask ? { threadRef: { botId: holderBot.id, threadId: holderTask.threadId, title: holderTask.title } } : {}),
640
653
  });
641
654
  }
642
655
  if (Date.now() >= deadline)
643
- throw new Error("Computer is still busy. Stop the turn using it, then retry.");
656
+ throw new Error(computerStillBusyText(holder, GROUP_GOAL_WAIT_MAX_MS));
644
657
  await new Promise(resolve => setTimeout(resolve, 100));
645
658
  }
646
659
  }
647
660
  finally {
648
661
  if (waitingMessage)
649
662
  store.patchMessage(owner.threadId, waitingMessage.id, {
650
- tool: { name: active() && turnResources.owns(resource, owner) ? "Computer available — continuing" : "Computer wait ended", ok: true },
663
+ tool: { name: active() && turnResources.owns(resource, owner) ? computerFreeText() : computerWaitEndedText(), ok: true },
651
664
  });
652
665
  }
653
666
  turnResourceOwners.set(owner.threadId, owner);
@@ -676,6 +689,14 @@ function botAtThreadCapacity(botId) {
676
689
  function hasDirectDispatch(botId) {
677
690
  return [...directTurnDispatchClaims.values()].some((claim) => claim.botId === botId);
678
691
  }
692
+ /** Routine and webhook dispatch shares startTurn's admission preconditions
693
+ * instead of waiting for whole-bot idleness: a free thread slot and no
694
+ * active group turn. A group turn blocks scheduled starts the same way it
695
+ * blocks every other turn kind; it does not consume a capacity slot. */
696
+ function unattendedDispatchState(botId) {
697
+ const bot = store.bot(botId);
698
+ return !bot ? "missing" : botAtThreadCapacity(botId) || activeGroupTurnForBot(botId) ? "busy" : "ready";
699
+ }
679
700
  function requestedTaskBot(botId, rawThreadId) {
680
701
  const profile = store.bot(botId);
681
702
  if (!profile)
@@ -1265,7 +1286,9 @@ if (browserCleanupReferencesReconciled)
1265
1286
  * than the desktop window did. Stripped here rather than at each call site
1266
1287
  * so a new broadcast cannot forget. */
1267
1288
  let activeCoordinationForThread = (_threadId) => false;
1268
- const wireTask = ({ resumeCursors: _resumeCursors, lastInstanceId: _lastInstanceId, ...task }) => activeCoordinationForThread(task.threadId) && !task.busy ? { ...task, busy: true, activity: "working" } : task;
1289
+ const wireTask = (task) => activeCoordinationForThread(task.threadId) && !task.busy
1290
+ ? { ...toWireTask(task), busy: true, activity: "working" }
1291
+ : toWireTask(task);
1269
1292
  const wireBot = (bot) => {
1270
1293
  const { resumeCursors: _resumeCursors, tasks, approvalGrant, lastProfileRequestId: _lastProfileRequestId, lastTeamSetupReceipt: _lastTeamSetupReceipt, ...rest } = bot;
1271
1294
  // An elevated selection is inert until the desktop confirms its exact
@@ -1433,7 +1456,10 @@ async function botOverview(bot) {
1433
1456
  * approval semantics require an implemented provider mapping. The trusted transition enforces
1434
1457
  * this too, but no provider dispatch or later permission callback relies on
1435
1458
  * persistence having been produced exclusively by that route. Delegation
1436
- * uses the receiving bot's grant, never the sender's — see approvalModeForOrigin. */
1459
+ * uses the receiving bot's grant, never the sender's (approvalModeForOrigin) —
1460
+ * with one deliberate exception: a Chief of Staff with Full access makes the
1461
+ * threads it delegates Full too (delegatedFullAccess), so the grant the
1462
+ * person gave the Chief covers the work the Chief hands out. */
1437
1463
  const approvalModeForTurn = (bot, peerInitiated = false) => {
1438
1464
  const mode = approvalModeForOrigin(approvalModeFor(bot), { peerInitiated });
1439
1465
  if (!supportsApprovalMode(registry.cliTarget(bot.modelSelection.instanceId)?.driverKind, mode)) {
@@ -1454,6 +1480,48 @@ function fullAccessForSource(botId, threadId) {
1454
1480
  function peerReviewRequired(bot, threadId) {
1455
1481
  return Boolean(bot.approvePeerComms && !fullAccessForSource(bot.id, threadId));
1456
1482
  }
1483
+ /** Full access flows down a Chief of Staff's delegation. The person gave the
1484
+ * Chief Full access so its work runs without prompts; a teammate stopping
1485
+ * that same work to ask defeats the grant — and in practice the person was
1486
+ * answering every one of those cards, all day, for the whole team. So a
1487
+ * teammate a Full-access Chief delegates to runs Full for that work: the
1488
+ * recipient switches, whatever its own level says. The recipient's engine
1489
+ * has to implement Full (supportsApprovalMode); otherwise the work keeps the
1490
+ * recipient's own level, as before. Only a Chief passes access on — an
1491
+ * ordinary bot's delegation still uses the recipient's setting. */
1492
+ function delegatedFullAccess(from, fromThreadId, target) {
1493
+ return delegationInheritsFullAccess({
1494
+ senderIsChief: Boolean(from.chiefOfStaff),
1495
+ senderHasFullAccess: fullAccessForSource(from.id, fromThreadId),
1496
+ sameBot: from.id === target.id,
1497
+ recipientDriverKind: registry.cliTarget(target.modelSelection.instanceId)?.driverKind,
1498
+ });
1499
+ }
1500
+ /** Make a delegated thread Full and say so in it once, so the level the
1501
+ * chip shows and the level the turns run at agree, and the person can see
1502
+ * where the access came from. Idempotent: a pair conversation is reused
1503
+ * across delegations and must not collect a chip per request. */
1504
+ function grantDelegatedFullAccess(from, target, threadId) {
1505
+ if (store.taskByThread(target.id, threadId)?.approvalMode === "full")
1506
+ return;
1507
+ store.patchTask(target.id, threadId, { approvalMode: "full", autoApprove: false, alwaysAllow: [] });
1508
+ store.appendMessage(threadId, {
1509
+ role: "bot",
1510
+ kind: "activity",
1511
+ tool: { name: `Full access — delegated by ${from.name}, a Chief of Staff with Full access`, ok: true },
1512
+ });
1513
+ }
1514
+ /** A room member's level for one turn. Work a Full-access Chief hands out
1515
+ * in a room runs Full for that turn: the room thread is shared, so the
1516
+ * level is not stored on it — it rides the handoff. */
1517
+ function roomTurnApprovalMode(bot, orchestration) {
1518
+ const handoff = orchestration?.roomHandoffId ? roomHandoffs.nodes.get(orchestration.roomHandoffId) : undefined;
1519
+ const source = handoff?.parentId ? roomHandoffs.nodes.get(handoff.parentId) : undefined;
1520
+ const from = source ? store.bot(source.botId) : undefined;
1521
+ if (from && source && delegatedFullAccess(from, source.threadId, bot))
1522
+ return "full";
1523
+ return approvalModeForTurn(bot, Boolean(orchestration?.roomHandoffId));
1524
+ }
1457
1525
  /** Privileged approval-mode transitions are deliberately absent from the
1458
1526
  * loopback HTTP authority model: a bot with shell access can curl that
1459
1527
  * surface itself. Only Electron's private utility-process channel can deliver
@@ -2260,7 +2328,7 @@ const roomHandoffs = new RoomHandoffs(join(DATA_DIR, "room-handoffs.json"), {
2260
2328
  await tracked;
2261
2329
  return { ok: result.outcome === "settled", text: result.stopReason || result.replyText || result.outcome || "The addressed agent could not run" };
2262
2330
  },
2263
- });
2331
+ }, Date.now, roomHandoffLimits(cfg));
2264
2332
  activeCoordinationForThread = threadId => roomHandoffs.activeDirect(threadId);
2265
2333
  function publicGroupState(group) {
2266
2334
  return { ...group, working: groupIsWorking(group) || [...roomHandoffs.nodes.values()].some(n => n.groupId === group.id && !["completed", "failed", "cancelled"].includes(n.status)) };
@@ -2941,6 +3009,10 @@ const watchdog = new TurnWatchdog({
2941
3009
  kind: "activity",
2942
3010
  tool: { name: `error: no activity for ${minutes} minutes — the turn was stopped`, ok: false },
2943
3011
  });
3012
+ // a routine's stall reports through its own failure path
3013
+ if (bot && routineRun?.target !== "bot") {
3014
+ reportIncident({ kind: "stalled", bot, threadId: turn.threadId, detail: `no activity for ${minutes} minutes — the turn was stopped` });
3015
+ }
2944
3016
  settleDirectFollowup(stalledGeneration);
2945
3017
  finalizeDelegationWatch(turn.threadId, false, "", "Delegated turn stalled and was stopped");
2946
3018
  turnUsage.delete(turn.threadId);
@@ -2979,8 +3051,7 @@ const watchdog = new TurnWatchdog({
2979
3051
  const currentBot = store.bot(turn.botId);
2980
3052
  if (currentBot?.busy) {
2981
3053
  stopScreenPoller(currentBot.id, turn.threadId);
2982
- if (activeVpsThreads.get(currentBot.id) === turn.threadId)
2983
- activeVpsThreads.delete(currentBot.id);
3054
+ vpsThreadEnded(currentBot.id, turn.threadId);
2984
3055
  if (store.taskByThread(currentBot.id, turn.threadId))
2985
3056
  store.setTaskActivity(currentBot.id, turn.threadId, "idle");
2986
3057
  else
@@ -3152,6 +3223,15 @@ const localVmLeases = new LocalVmLeasePool(30 * 60_000);
3152
3223
  const localVmLifecycleBusy = new Set();
3153
3224
  const localVmThreadTargets = new Map();
3154
3225
  const localVmActiveThreads = new Map();
3226
+ // Lazy Auto-VM claims (issue #1361): a thread whose auto-resolved Local VM
3227
+ // attach deferred the exclusive claim registers here so the first screen
3228
+ // tools/call gate can fire it. Dispatch claims eagerly today, so entries
3229
+ // exist only as a no-op handoff to the gate.
3230
+ const autoVmClaims = new Map();
3231
+ /** How long the computer-control gate lets a lazy claim land before it
3232
+ * answers held. A free, ready VM claims in the time of one container
3233
+ * inspect; only a claim queued behind another holder outlives this. */
3234
+ const LAZY_VM_CLAIM_GRACE_MS = 5_000;
3155
3235
  /** Local VM targets this process has seen ready or recreatable — at boot, in
3156
3236
  * the inventory, or in a turn. Auto probes the container runtime for a VM
3157
3237
  * only when one of these exists, so an ordinary Auto turn on a machine with
@@ -3164,7 +3244,25 @@ function noteLocalVmSeen(target, status) {
3164
3244
  let localVmImageBusy = false;
3165
3245
  let localVmProvisionBusy = false;
3166
3246
  let localVmModeChangeBusy = false;
3247
+ /** Threads running with a bot's VPS computer mounted, per bot. Several run
3248
+ * at once — only the desktop lease is exclusive, and it is claimed on the
3249
+ * first screen call (see the VPS mount in dispatch) — so the alias and
3250
+ * backend guards ask "any thread?", and a settling thread removes only
3251
+ * itself, never a sibling still running. */
3167
3252
  const activeVpsThreads = new Map();
3253
+ function vpsThreadStarted(botId, threadId) {
3254
+ const threads = activeVpsThreads.get(botId) ?? new Set();
3255
+ threads.add(threadId);
3256
+ activeVpsThreads.set(botId, threads);
3257
+ }
3258
+ function vpsThreadEnded(botId, threadId) {
3259
+ const threads = activeVpsThreads.get(botId);
3260
+ if (!threads)
3261
+ return;
3262
+ threads.delete(threadId);
3263
+ if (!threads.size)
3264
+ activeVpsThreads.delete(botId);
3265
+ }
3168
3266
  const boxLifecycleBusyBots = new Set();
3169
3267
  // A refresh is a reader, not a lifecycle change. Keep its reservation until
3170
3268
  // the provider settles even if the HTTP client leaves, and share it on retry.
@@ -3577,6 +3675,10 @@ function localVmIdleFor(target) {
3577
3675
  return idle;
3578
3676
  }
3579
3677
  function releaseLocalVmThread(threadId) {
3678
+ // Covers lazy claims too (issue #1361): every settle path funnels through
3679
+ // here or through releaseTurnResources, so a turn that ends before its
3680
+ // first screen call leaves no claim slot behind.
3681
+ autoVmClaims.delete(threadId);
3580
3682
  const target = localVmThreadTargets.get(threadId);
3581
3683
  if (!target)
3582
3684
  return;
@@ -3712,6 +3814,8 @@ bus.subscribe((event) => {
3712
3814
  pushMessage({ role: "bot", kind: "text", text: coordinatorVisibleText, turnId: completedTurnId });
3713
3815
  lastReply.set(event.threadId, coordinatorVisibleText);
3714
3816
  }
3817
+ if (bot)
3818
+ handoffs.onEvent(event);
3715
3819
  switch (event.type) {
3716
3820
  case "session.started":
3717
3821
  if (bot && event.sessionId && event.providerInstanceId) {
@@ -4036,6 +4140,13 @@ bus.subscribe((event) => {
4036
4140
  store.markTerminalAssistantMessage(event.threadId, completedTurnId);
4037
4141
  const reply = lastReply.get(event.threadId) ?? "";
4038
4142
  lastReply.delete(event.threadId);
4143
+ // A run that broke — not one the person stopped, and not a routine's,
4144
+ // which reports through its own failure path — is the Chief's to see.
4145
+ if (!event.ok && event.stopReason !== "interrupted" && !routines?.runForThread(event.threadId)) {
4146
+ const broken = bot ?? (speaker ? store.bot(speaker.botId) : undefined);
4147
+ if (broken)
4148
+ reportIncident({ kind: "failed", bot: broken, threadId: event.threadId, detail: event.stopReason?.trim() || "the run ended without a result" });
4149
+ }
4039
4150
  const lastReported = turnUsage.get(event.threadId);
4040
4151
  turnUsage.delete(event.threadId);
4041
4152
  turnContext.delete(event.threadId);
@@ -4053,8 +4164,7 @@ bus.subscribe((event) => {
4053
4164
  const ownsResources = resourceOwner &&
4054
4165
  turnResourceOwners.get(event.threadId)?.generation === resourceOwner.generation;
4055
4166
  if (ownsResources) {
4056
- if (activeVpsThreads.get(bot.id) === event.threadId)
4057
- activeVpsThreads.delete(bot.id);
4167
+ vpsThreadEnded(bot.id, event.threadId);
4058
4168
  releaseLocalVmThread(event.threadId);
4059
4169
  }
4060
4170
  releaseTurnResources(resourceOwner);
@@ -4283,6 +4393,93 @@ function wakeDelegationSource(source, threadId, targetName, failureReason, routi
4283
4393
  }
4284
4394
  dispatchDelegationWake(source.id, threadId, targetName, failureReason, routineRunId);
4285
4395
  }
4396
+ // ── incidents: a broken run reaches the Chief of Staff ──────────────────
4397
+ // A failed, stalled or unstartable run used to leave one chip in the thread
4398
+ // it died in and nothing anywhere else; the person found it hours later,
4399
+ // from a phone, by opening the desktop and reading every thread. The team
4400
+ // already has a role for this — the Chief coordinates the section — so the
4401
+ // incident becomes a turn of the Chief's, in its "Team incidents" thread,
4402
+ // with a link to the broken thread and retry_thread to act on it. The person
4403
+ // reads one place. Policy in server/incidents.ts.
4404
+ const incidentLedger = new IncidentLedger();
4405
+ /** What the broken thread was about: the last line the person (or the
4406
+ * requester) sent there, and the last thing the bot said. */
4407
+ function incidentContext(threadId) {
4408
+ const messages = [...store.messagesFor(threadId)].reverse();
4409
+ return {
4410
+ lastRequest: messages.find((message) => message.role === "user" && message.kind === "text" && message.text)?.text ?? null,
4411
+ lastReply: messages.find((message) => message.role === "bot" && message.kind === "text" && message.text)?.text ?? null,
4412
+ };
4413
+ }
4414
+ function reportIncident(input) {
4415
+ const { bot, threadId } = input;
4416
+ const task = store.taskByThread(bot.id, threadId);
4417
+ // A thread another bot opened and is watching is that bot's to handle:
4418
+ // the delegator is woken with the failure already (wakeDelegationSource).
4419
+ if (task?.openedBy?.delegationId || delegationWatch.has(threadId) || roomHandoffs.activeDirect(threadId))
4420
+ return;
4421
+ const group = store.groupByThread(threadId);
4422
+ const incident = {
4423
+ kind: input.kind,
4424
+ bot,
4425
+ threadId,
4426
+ title: task?.title ?? null,
4427
+ room: group?.name ?? null,
4428
+ detail: redactSecretsInText(input.detail),
4429
+ ...incidentContext(threadId),
4430
+ };
4431
+ const count = incidentLedger.note(threadId);
4432
+ // a crash loop is one incident, not a storm
4433
+ if (count.muted)
4434
+ return;
4435
+ const chief = chiefForBot(store.bots, bot);
4436
+ // A run that could not start and a failed routine have already buzzed
4437
+ // the person (turn-failed, routine-failed) by the time they get here; a
4438
+ // failure or stall mid-run has not. One notification per failure, never two.
4439
+ const alreadyNotified = input.kind === "could-not-start" || input.kind === "routine-failed";
4440
+ const tellThePerson = () => {
4441
+ if (alreadyNotified)
4442
+ return;
4443
+ notify(buildNotification("incident", bot, threadId, incidentChip(incident), {
4444
+ avatarUrl: bot.avatarUrl,
4445
+ ...(group ? { group: { id: group.id, name: group.name } } : {}),
4446
+ }));
4447
+ };
4448
+ // no Chief on duty, or the Chief itself broke: the person is next
4449
+ if (!chief) {
4450
+ tellThePerson();
4451
+ return;
4452
+ }
4453
+ const incidents = store.tasks(chief.id).find((candidate) => candidate.title === INCIDENTS_THREAD_TITLE && !candidate.archivedAt)
4454
+ ?? store.createTask(chief.id, INCIDENTS_THREAD_TITLE, false, undefined, { botId: chief.id, name: chief.name, at: Date.now() });
4455
+ if (!incidents || incidents.threadId === threadId) {
4456
+ tellThePerson();
4457
+ return;
4458
+ }
4459
+ store.appendMessage(incidents.threadId, {
4460
+ role: "bot",
4461
+ kind: "activity",
4462
+ tool: { name: incidentChip(incident), ok: false },
4463
+ threadRef: { botId: bot.id, threadId, title: task?.title ?? (group ? group.name : `${bot.name}'s conversation`) },
4464
+ });
4465
+ const text = incidentText(incident, count);
4466
+ // the report carries the broken bot's name as its provenance: it is about
4467
+ // that bot's work and nobody was at the keyboard
4468
+ const peerAsk = { botId: bot.id, name: bot.name, unattended: true };
4469
+ if (botAtThreadCapacity(chief.id) || activeGroupTurnForBot(chief.id)) {
4470
+ queueSteeredMessage(chief.id, incidents.threadId, text, { reason: "capacity", unattended: true, peerAsk });
4471
+ return;
4472
+ }
4473
+ void startTurn(chief.id, text, { threadId: incidents.threadId, unattended: true, peerAsk }).catch((error) => {
4474
+ const why = error instanceof Error ? error.message : String(error);
4475
+ store.appendMessage(incidents.threadId, {
4476
+ role: "bot",
4477
+ kind: "activity",
4478
+ tool: { name: `error: the incident could not reach ${chief.name} — ${why.slice(0, 120)}`, ok: false },
4479
+ });
4480
+ tellThePerson();
4481
+ });
4482
+ }
4286
4483
  function drainDelegationWakes() {
4287
4484
  for (const [threadId, entry] of pendingDelegationWakes) {
4288
4485
  if (!store.taskByThread(entry.botId, threadId) || !routineDelegationCanResume(threadId, entry.routineRunId)) {
@@ -4317,12 +4514,36 @@ function markTaskContextExternallyUpdated(bot, threadId) {
4317
4514
  const task = store.taskByThread(bot.id, threadId);
4318
4515
  if (!task)
4319
4516
  return;
4517
+ // An engine that records which messages its current session was handed
4518
+ // keeps that session: its next turn is sent what it has not seen. Without a
4519
+ // record for that exact session (another engine, one switched in since, a
4520
+ // replaced session, or a task from before records existed) the next turn
4521
+ // replays once, as before.
4522
+ const owner = task.lastInstanceId;
4523
+ const record = owner ? task.handedMessages?.[owner] : undefined;
4524
+ if (owner && record?.session !== undefined && record.session === task.resumeCursors[owner] &&
4525
+ turnInstance(botForThread(bot.id, threadId) ?? bot, undefined, threadId)?.instanceId === owner &&
4526
+ registry.get(owner)?.adapter.capabilities.strictResume) {
4527
+ store.patchTask(bot.id, threadId, { unread: true });
4528
+ return;
4529
+ }
4320
4530
  store.patchTask(bot.id, threadId, {
4321
4531
  resumeCursors: {},
4322
4532
  lastInstanceId: `${EXTERNAL_CONTEXT_MARKER_PREFIX}${randomUUID()}`,
4323
4533
  unread: true,
4324
4534
  });
4325
4535
  }
4536
+ /** Active-branch messages a provider reads as conversation context. */
4537
+ function isContextMessage(m) {
4538
+ return Boolean((m.kind === "text" && m.text) || m.roomRequest?.phase === "result");
4539
+ }
4540
+ const handoffs = new Handoffs({
4541
+ order: (threadId) => store.activePath(threadId).filter(isContextMessage).map((m) => m.id),
4542
+ read: (botId, threadId, instanceId) => store.taskByThread(botId, threadId)?.handedMessages?.[instanceId],
4543
+ write: (botId, threadId, instanceId, state) => store.setHandedMessages(botId, threadId, instanceId, state),
4544
+ replies: (threadId, turnId) => store.activePath(threadId)
4545
+ .filter((m) => m.role === "bot" && m.kind === "text" && !m.from && m.turnId === turnId).map((m) => m.id),
4546
+ });
4326
4547
  /** Consume one delegated-turn watch and mirror exactly one terminal state.
4327
4548
  * Some harness paths settle a busy bot without a provider turn.completed
4328
4549
  * event, so they call this same finalizer explicitly. */
@@ -5007,18 +5228,33 @@ async function startTurn(botId, text, opts) {
5007
5228
  // branch only — abandoned forks never reach the model
5008
5229
  const skipTranscript = new Set([userMessage.id, ...(opts?.excludeMessageIds ?? [])]);
5009
5230
  const activeMessages = store.activePath(threadId);
5231
+ // This turn's own text carries the addressed request and, when resuming,
5232
+ // every child result as JSON: neither is repeated from the transcript.
5233
+ const coordinationChildren = opts?.coordination?.resumed
5234
+ ? new Set(roomHandoffs.children(opts.coordination.id).map((child) => child.id)) : undefined;
5235
+ for (const m of activeMessages) {
5236
+ if (!opts?.coordination || !m.roomRequest)
5237
+ continue;
5238
+ if ((m.roomRequest.phase === "request" && m.roomRequest.id === opts.coordination.id) ||
5239
+ (m.roomRequest.phase === "result" && coordinationChildren?.has(m.roomRequest.id)))
5240
+ skipTranscript.add(m.id);
5241
+ }
5010
5242
  // A flat reply may deliberately point across a fork in the same thread.
5011
5243
  // Resolve its quote from full storage, while the replay itself remains
5012
5244
  // strictly limited to the selected branch below.
5013
5245
  const messagesById = new Map(store.messagesFor(threadId).map((message) => [message.id, message]));
5014
- const transcript = activeMessages
5015
- .filter((m) => ((m.kind === "text" && m.text) || m.roomRequest?.phase === "result") && !skipTranscript.has(m.id))
5016
- .slice(-40)
5017
- .map((m) => ({
5246
+ const context = activeMessages.filter(isContextMessage).map((m) => ({
5247
+ id: m.id,
5018
5248
  role: m.role === "user" ? "user" : "assistant",
5019
5249
  text: m.roomRequest?.phase === "result" ? teammateReportContext(m.roomRequest.id, bot.id)
5020
- : transcriptText(m, messagesById, cfg.profile?.name?.trim() || "User"),
5250
+ : m.role !== "user" && m.from ? peerMessageText(m.from.name, transcriptText(m, messagesById, cfg.profile?.name?.trim() || "User"))
5251
+ : transcriptText(m, messagesById, cfg.profile?.name?.trim() || "User"),
5252
+ keep: m.roomRequest?.phase === "result" || (m.role !== "user" && Boolean(m.from)),
5253
+ ...(m.steered ? { steered: true } : {}),
5021
5254
  }));
5255
+ const contextOrder = context.map((m) => m.id);
5256
+ const replayable = context.filter((m) => !skipTranscript.has(m.id));
5257
+ const transcript = replayable.slice(-40).map((m) => ({ role: m.role, text: m.text }));
5022
5258
  // After a rewind (edit / branch switch) the provider's native session
5023
5259
  // still contains the abandoned branch: start a fresh session instead of
5024
5260
  // resuming, and for cursor-resuming drivers replay the surviving path
@@ -5037,6 +5273,28 @@ async function startTurn(botId, text, opts) {
5037
5273
  const fresh = !rewound &&
5038
5274
  !externalContextMarker &&
5039
5275
  engineIsFresh({ instanceId, lastInstanceId: task.lastInstanceId, resumeCursors: task.resumeCursors, transcript });
5276
+ // An engine that records what its session was handed resumes it with only
5277
+ // the context messages outside that record. A record of another session
5278
+ // (the one it replaced) or one that no longer lines up with the branch is
5279
+ // not trusted: the session is rebuilt by the same replay as any other.
5280
+ const strictResume = instance.adapter.capabilities.strictResume === true;
5281
+ const cursor = task.resumeCursors[instanceId];
5282
+ const handed = strictResume && !rewound && !fresh && !externalContextMarker && cursor !== undefined
5283
+ ? task.handedMessages?.[instanceId] : undefined;
5284
+ const unseen = handed && handedStateUsable(handed, cursor, contextOrder) ? unseenMessages(replayable, contextOrder, handed) : undefined;
5285
+ // A teammate's result or reply: without records this turn would replay.
5286
+ const externalUpdate = Boolean(opts?.coordination?.resumed || unseen?.some((m) => m.keep));
5287
+ // What a resumed session keeps from its launch: the standing instructions
5288
+ // (tools, servers and — for Claude — the model are passed on every launch),
5289
+ // plus whatever this engine can only set when a session starts. Codex's
5290
+ // thread/resume sends no model selection, and an effort it is not sent stays
5291
+ // at the thread's last value, so both belong to the session there. An
5292
+ // external update that finds any of it changed since the session started
5293
+ // gets the fresh session and replay it always got, rather than a resume.
5294
+ const persistentConfig = [bot.name, bot.title, bot.description, sectionContextSystemPrompt(bot.section),
5295
+ ...(instance.driverKind === "codex" ? [model, effort ?? null] : [])];
5296
+ const sessionConfig = (soul) => createHash("sha256").update(JSON.stringify([...persistentConfig, soul])).digest("hex").slice(0, 16);
5297
+ const plannedConfig = sessionConfig(bot.soul);
5040
5298
  // Agent-tool gate shared by skill authoring, the /setup turn-text rewrite,
5041
5299
  // the setup prompt block, and the peer-comms integration below: a driver
5042
5300
  // that never mounts agent tools (or a turn already at the comms-depth cap)
@@ -5050,23 +5308,50 @@ async function startTurn(botId, text, opts) {
5050
5308
  // which also depends on the bot's soul/description — is decided below,
5051
5309
  // from the same bot snapshot the prompt's soul is built from.
5052
5310
  const setupText = agentsMounted ? expandSetupTurnText(providerText) : providerText;
5053
- const { turnText, resume } = buildTurnContext({
5054
- text: promptWithReply(skillAuthoring ? expandLearnTurnText(setupText) : setupText, opts?.replyTo, cfg.profile?.name?.trim() || "User"),
5055
- transcript,
5056
- rewound,
5057
- fresh,
5058
- externallyUpdated: Boolean(externalContextMarker),
5059
- replaysNatively: instance.driverKind === "grok",
5060
- });
5061
- // Snapshot the cursor alongside the context decision. An external result
5062
- // can arrive during async computer/setup work and clear the task cursor;
5063
- // this already-built turn must either keep its old session or replay on the
5064
- // following turn, never start a blank session with no transcript.
5065
- const resumeCursor = resume ? task.resumeCursors[instanceId] : undefined;
5066
- // A cursor the provider no longer honours must not brick the thread: the
5067
- // driver may fall back to ONE fresh session, and this is what it sends
5068
- // there, so the new session is not blank (server/resume-recovery.ts).
5069
- const recoveryText = resumeCursor !== undefined ? buildRecoveryText({ text: turnText, transcript }) : undefined;
5311
+ const userTurnText = promptWithReply(skillAuthoring ? expandLearnTurnText(setupText) : setupText, opts?.replyTo, cfg.profile?.name?.trim() || "User");
5312
+ // Decided again at dispatch when setup outlasted a soul edit (config).
5313
+ const decideContext = (config) => {
5314
+ const handedStale = Boolean(handed && (!unseen || (externalUpdate && handed.config !== config)));
5315
+ const { block: unseenBlock, placed } = unseen && !handedStale ? renderUnseen(unseen) : { block: "", placed: [] };
5316
+ const { turnText: contextTurnText, resume } = buildTurnContext({
5317
+ text: userTurnText,
5318
+ transcript,
5319
+ rewound,
5320
+ fresh,
5321
+ externallyUpdated: Boolean(externalContextMarker) || handedStale,
5322
+ replaysNatively: instance.driverKind === "grok",
5323
+ });
5324
+ // Snapshot the cursor alongside the context decision. An external result
5325
+ // can arrive during async computer/setup work and clear the task cursor;
5326
+ // this already-built turn must either keep its old session or replay on the
5327
+ // following turn, never start a blank session with no transcript.
5328
+ const resumeCursor = resume ? task.resumeCursors[instanceId] : undefined;
5329
+ // A cursor the provider no longer honours must not brick the thread: the
5330
+ // driver may fall back to ONE fresh session, and this is what it sends
5331
+ // there, so the new session is not blank (server/resume-recovery.ts). A
5332
+ // turn carrying an external update gets the replay it would have had.
5333
+ const recoveryIsReplay = resumeCursor !== undefined && externalUpdate && transcript.length > 0;
5334
+ const recoveryText = resumeCursor === undefined ? undefined : recoveryIsReplay
5335
+ ? buildTurnContext({ text: userTurnText, transcript, rewound: false, fresh: false, externallyUpdated: true, replaysNatively: false }).turnText
5336
+ : buildRecoveryText({ text: userTurnText, transcript });
5337
+ // What this turn puts in front of the provider, for each session it can end
5338
+ // up in (server/delta-context.ts). buildTurnContext prepends a replay only
5339
+ // when it replays, so a changed text means the transcript was sent.
5340
+ const carried = [...skipTranscript];
5341
+ const windowIds = replayable.slice(-40).map((m) => m.id);
5342
+ return {
5343
+ turnText: withUnseenMessages(unseenBlock, contextTurnText),
5344
+ resumeCursor, recoveryText, recoveryIsReplay,
5345
+ handoff: strictResume ? {
5346
+ botId: bot.id, instanceId, config, resumeCursor: typeof resumeCursor === "string" ? resumeCursor : undefined,
5347
+ started: sessionStart(contextOrder, contextTurnText !== userTurnText ? windowIds : [], carried),
5348
+ recovery: sessionStart(contextOrder, recoveryText !== undefined ? windowIds : [], carried),
5349
+ resumed: sessionStart(contextOrder, [], [...placed, ...carried]),
5350
+ placed, carried, own: [userMessage.id, ...(opts?.excludeMessageIds ?? [])],
5351
+ } : undefined,
5352
+ };
5353
+ };
5354
+ let dispatchContext = decideContext(plannedConfig);
5070
5355
  const persona = [
5071
5356
  `You are ${bot.name}, a personal bot in OpenMausBot.`,
5072
5357
  bot.title && `Role: ${bot.title}.`,
@@ -5094,6 +5379,8 @@ async function startTurn(botId, text, opts) {
5094
5379
  // follow-ups: any normal turn may ask teammates to coordinate work.
5095
5380
  directFollowupSettlers.set(dispatchClaimId, { threadId, settle: opts?.onTurnSettled });
5096
5381
  directTurnDispatchClaims.set(threadId, { id: dispatchClaimId, botId, threadId, phase: "setup" });
5382
+ if (dispatchContext.handoff)
5383
+ handoffs.begin(threadId, dispatchClaimId, dispatchContext.handoff);
5097
5384
  directTurnBots.set(threadId, bot);
5098
5385
  beginInternalCapabilityGeneration(threadId, dispatchClaimId);
5099
5386
  if (!opts?.computerSelectionContinuation && !opts?.cardContinuation && !opts?.automationSource && !opts?.unattended &&
@@ -5215,6 +5502,79 @@ async function startTurn(botId, text, opts) {
5215
5502
  let browserCapture = null;
5216
5503
  let computerKind = null;
5217
5504
  let autoVpsProblem = null;
5505
+ /** The Local VM frame capture for the poller and the settled transcript
5506
+ * screenshot. The shared desktop outlives the turn: once another thread
5507
+ * owns it, a capture still in flight would picture ITS work under this
5508
+ * bot's name — live and in the settled frame, which is taken after the
5509
+ * lease is already released. No owner means the desktop is simply
5510
+ * idle: that final frame is ours to keep. */
5511
+ const localVmPreviewFor = (localVmTarget, claimThreadId) => () => {
5512
+ const owner = localVmLeaseFor(localVmTarget).current(localVmOwnerBusy);
5513
+ if (owner && owner.threadId !== claimThreadId) {
5514
+ throw new Error("the Local VM moved on to another turn");
5515
+ }
5516
+ return containerComputerFrame(undefined, undefined, localVmTarget);
5517
+ };
5518
+ /** The exclusive Local VM claim sequence, verbatim from the old inline
5519
+ * attach path, shared by dispatch (eager) and the first-screen-call
5520
+ * gate (issue #1361). Idempotent per turn: the resource claim and the
5521
+ * lease both re-assert the same owner, so a re-entrant call from the
5522
+ * gate no-ops once dispatch has already claimed. A lazy attach pins
5523
+ * the target it mounted the tools against: the claim must lease that
5524
+ * desktop, not whatever localVmTargetForBot resolves to by the time
5525
+ * the first screen call arrives. */
5526
+ const claimAutoLocalVm = async (claimThreadId, pinnedTarget) => {
5527
+ const localVmTarget = pinnedTarget ?? localVmTargetForBot(bot.id);
5528
+ await bindTurnComputer(resourceOwner, `computer:vm:${localVmTarget.key}`, true);
5529
+ if (localVmImageBusy || localVmModeChangeBusy || localVmLifecycleBusy.has(localVmTarget.key)) {
5530
+ throw new Error("this Local VM is being started, stopped, or replaced — wait for setup to finish");
5531
+ }
5532
+ // Claim before the first await. The lifecycle route performs its
5533
+ // matching check synchronously, so neither side can enter while the
5534
+ // other is between inspection and mutation.
5535
+ if (!localVmLeaseFor(localVmTarget).claim(claimThreadId, bot.id, localVmOwnerBusy)) {
5536
+ throw new Error("this Local VM is already being used by another turn — wait for that turn to finish");
5537
+ }
5538
+ localVmThreadTargets.set(claimThreadId, localVmTarget);
5539
+ localVmActiveThreads.set(localVmTarget.key, claimThreadId);
5540
+ localVmIdleFor(localVmTarget).touch();
5541
+ // The lease is held from here. An eager attach that fails below
5542
+ // fails the turn and settle releases it; a lazy claim's rejection is
5543
+ // swallowed into the slot's failed flag and the turn carries on, so
5544
+ // without this the exclusive lease would sit held for the rest of a
5545
+ // turn that never got the VM — the very serialisation #1361 removes.
5546
+ const dropLease = () => {
5547
+ localVmLeaseFor(localVmTarget).release(claimThreadId);
5548
+ if (localVmActiveThreads.get(localVmTarget.key) === claimThreadId)
5549
+ localVmActiveThreads.delete(localVmTarget.key);
5550
+ localVmThreadTargets.delete(claimThreadId);
5551
+ // bindTurnComputer above also took the turn-level resource; a
5552
+ // later turn's exclusive bind queues behind it just the same.
5553
+ const resource = `computer:vm:${localVmTarget.key}`;
5554
+ turnResources.releaseOne(resource, resourceOwner);
5555
+ if (turnComputerResources.get(resourceOwner.threadId)?.resource === resource)
5556
+ turnComputerResources.delete(resourceOwner.threadId);
5557
+ };
5558
+ let localVm;
5559
+ try {
5560
+ localVm = await readyLocalVmForTurn(bot.id, localVmTarget);
5561
+ }
5562
+ catch (error) {
5563
+ dropLease();
5564
+ throw error;
5565
+ }
5566
+ if (!localVm.ready || !localVm.runtime) {
5567
+ dropLease();
5568
+ throw new Error(`${localVm.problem ?? "the Local VM is not ready"} (App Settings → Computers)`);
5569
+ }
5570
+ // Same contract as the Box and VPS branches below: without this the
5571
+ // poller never starts, so the Local VM publishes no `screen` events
5572
+ // and every client that only has the stream (the phone) waits
5573
+ // forever. The web panel hid the gap by polling the screenshot
5574
+ // route itself.
5575
+ previewCapture = localVmPreviewFor(localVmTarget, claimThreadId);
5576
+ return { target: localVmTarget, runtime: localVm.runtime };
5577
+ };
5218
5578
  // Explicit destinations are strict. In particular, Local VM must never
5219
5579
  // fall through to host CUA and accidentally click on the user's Mac.
5220
5580
  // The Local VM attach, shared by explicit "Local VM" and by Auto. Explicit
@@ -5229,6 +5589,7 @@ async function startTurn(botId, text, opts) {
5229
5589
  throw new Error("this model engine cannot use the Local VM — choose Claude or an ACP engine, or select another computer destination");
5230
5590
  }
5231
5591
  const localVmTarget = localVmTargetForBot(bot.id);
5592
+ let lazyReadyVm = null;
5232
5593
  if (!strict) {
5233
5594
  // Nothing this process has ever seen for this target, and nobody is
5234
5595
  // relying on an unattended run: do not pay for a runtime probe.
@@ -5239,43 +5600,52 @@ async function startTurn(botId, text, opts) {
5239
5600
  return false;
5240
5601
  if (localVmImageBusy || localVmModeChangeBusy || localVmLifecycleBusy.has(localVmTarget.key))
5241
5602
  return false;
5603
+ if (seen.ready && seen.runtime)
5604
+ lazyReadyVm = { runtime: seen.runtime };
5242
5605
  }
5243
5606
  try {
5244
- await bindTurnComputer(resourceOwner, `computer:vm:${localVmTarget.key}`, true);
5245
- if (localVmImageBusy || localVmModeChangeBusy || localVmLifecycleBusy.has(localVmTarget.key)) {
5246
- throw new Error("this Local VM is being started, stopped, or replaced — wait for setup to finish");
5247
- }
5248
- // Claim before the first await. The lifecycle route performs its
5249
- // matching check synchronously, so neither side can enter while the
5250
- // other is between inspection and mutation.
5251
- if (!localVmLeaseFor(localVmTarget).claim(threadId, bot.id, localVmOwnerBusy)) {
5252
- throw new Error("this Local VM is already being used by another turn — wait for that turn to finish");
5253
- }
5254
- localVmThreadTargets.set(threadId, localVmTarget);
5255
- localVmActiveThreads.set(localVmTarget.key, threadId);
5256
- localVmIdleFor(localVmTarget).touch();
5257
- const localVm = await readyLocalVmForTurn(bot.id, localVmTarget);
5258
- if (!localVm.ready || !localVm.runtime) {
5259
- throw new Error(`${localVm.problem ?? "the Local VM is not ready"} (App Settings → Computers)`);
5607
+ if (lazyReadyVm) {
5608
+ // Lazy exclusivity (issue #1361): a VM that is ready right now
5609
+ // mounts without claiming — screen-less Auto turns never touch
5610
+ // the lease, and the first screen tools/call fires the claim
5611
+ // through the computer-control gate. A VM that must be created
5612
+ // or recreated first keeps the eager claim below: the bridge
5613
+ // child needs the container to exist, and readyLocalVmForTurn
5614
+ // is what boots it.
5615
+ integrations.localComputer = containerComputerMcp(lazyReadyVm.runtime, controlIntegration(bot.id, threadId, dispatchClaimId), localVmTarget);
5616
+ autoVmClaims.set(threadId, {
5617
+ owner: resourceOwner,
5618
+ lazy: true,
5619
+ label: "the Local VM",
5620
+ claim: async () => {
5621
+ await claimAutoLocalVm(threadId, localVmTarget);
5622
+ // The dispatch-site poller start saw a null previewCapture
5623
+ // (this lazy mount runs before any claim exists), so this
5624
+ // turn would publish no live `screen` events and settle no
5625
+ // final computer frame. Restart the poller with the now-live
5626
+ // computer capture, keeping any browser capture and whether
5627
+ // this turn already touched its screen. Same still-running
5628
+ // guard as dispatch: a poller started after its own
5629
+ // turn.completed would never be torn down.
5630
+ if (previewCapture && threadBusy(bot.id, threadId)) {
5631
+ const touched = screenPollers.get(threadId)?.touched ?? instance.driverKind === "boxAgent";
5632
+ stopScreenPoller(bot.id, threadId);
5633
+ startScreenPoller(bot.id, threadId, { computer: previewCapture, ...(browserCapture ? { browser: browserCapture } : {}) }, { screenIsTheWork: touched });
5634
+ }
5635
+ },
5636
+ });
5637
+ return true;
5260
5638
  }
5261
- integrations.localComputer = containerComputerMcp(localVm.runtime, controlIntegration(bot.id, threadId, dispatchClaimId), localVmTarget);
5262
- // Same contract as the Box and VPS branches below: without this the
5263
- // poller never starts, so the Local VM publishes no `screen` events
5264
- // and every client that only has the stream (the phone) waits
5265
- // forever. The web panel hid the gap by polling the screenshot
5266
- // route itself.
5267
- previewCapture = () => {
5268
- // The shared desktop outlives the turn. Once another thread owns
5269
- // it, a capture still in flight would picture ITS work under this
5270
- // bot's name — live and in the settled transcript frame, which is
5271
- // taken after the lease is already released. No owner means the
5272
- // desktop is simply idle: that final frame is ours to keep.
5273
- const owner = localVmLeaseFor(localVmTarget).current(localVmOwnerBusy);
5274
- if (owner && owner.threadId !== threadId) {
5275
- throw new Error("the Local VM moved on to another turn");
5276
- }
5277
- return containerComputerFrame(undefined, undefined, localVmTarget);
5278
- };
5639
+ const claimed = await claimAutoLocalVm(threadId);
5640
+ integrations.localComputer = containerComputerMcp(claimed.runtime, controlIntegration(bot.id, threadId, dispatchClaimId), claimed.target);
5641
+ // Hand the same claim to the first-screen-call gate. This eager
5642
+ // path has already claimed, so the gate's fire-once call can only
5643
+ // re-assert the same owner — a no-op (issue #1361).
5644
+ autoVmClaims.set(threadId, {
5645
+ owner: resourceOwner,
5646
+ label: "the Local VM",
5647
+ claim: async () => { await claimAutoLocalVm(threadId); },
5648
+ });
5279
5649
  return true;
5280
5650
  }
5281
5651
  catch (error) {
@@ -5295,7 +5665,11 @@ async function startTurn(botId, text, opts) {
5295
5665
  hostPlatform: process.platform,
5296
5666
  providerSupportsLocal: mountsLocalComputer,
5297
5667
  })) {
5298
- throw new Error("this model engine cannot control this computer — choose Claude or an ACP engine, or select another destination");
5668
+ // Name the condition that actually failed: a person told "choose an
5669
+ // ACP engine" while already on one has nowhere to go.
5670
+ throw new Error(mountsLocalComputer
5671
+ ? `local computer control is not available on ${process.platform} — select another destination`
5672
+ : "this model engine cannot control this computer — choose Claude or an ACP engine, or select another destination");
5299
5673
  }
5300
5674
  const cua = readCuaConnection();
5301
5675
  if (!cua)
@@ -5314,10 +5688,17 @@ async function startTurn(botId, text, opts) {
5314
5688
  if (unsupported && wants === undefined)
5315
5689
  autoVpsProblem = unsupported;
5316
5690
  if (!unsupported) {
5317
- // The remote lifecycle and container are shared by this bot. Keep
5318
- // its explicit computer turns serialized; ordinary threads still run.
5319
- await bindTurnComputer(resourceOwner, `computer:vps:${vpsSshAlias(cfg)}:${bot.id}`, true);
5320
- activeVpsThreads.set(bot.id, threadId);
5691
+ // The VPS "computer" is the desktop inside this bot's managed
5692
+ // container, and only screen work needs that desktop to itself.
5693
+ // So the lease is claimed on the first computer call, through the
5694
+ // computer-control gate (the Local VM's seam, #1361), never at
5695
+ // mount: a bot's turns that never touch the computer tools run
5696
+ // side by side, and its 3-hourly routine no longer queues behind
5697
+ // — or fails after 30 minutes behind — its own long-running task.
5698
+ // Container lifecycle (provision, start) is serialized by the
5699
+ // runner's per-container lock, not by this turn.
5700
+ const vpsResource = `computer:vps:${vpsSshAlias(cfg)}:${bot.id}`;
5701
+ vpsThreadStarted(bot.id, threadId);
5321
5702
  let remote;
5322
5703
  remote = vps.vpsStartsForTurn({ wants, autoStartVps: bot.autoStartVps, automationSource: opts?.automationSource })
5323
5704
  ? await vps.vpsComputerAction("provision", cfg, bot.id)
@@ -5331,10 +5712,28 @@ async function startTurn(botId, text, opts) {
5331
5712
  env: { ...vpsMcp.env, OMB_CONTROL_URL: vpsControl.url, OMB_CONTROL_TOKEN: vpsControl.token },
5332
5713
  };
5333
5714
  computerKind = "vps";
5334
- previewCapture = () => vps.vpsComputerScreenshot(targetCfg, bot.id);
5715
+ // Live frames only once this turn holds the desktop: a poller on
5716
+ // a desktop another turn is driving would publish that turn's
5717
+ // screen as this one's. The claim restarts the poller with the
5718
+ // capture, the way the Local VM's lazy claim does.
5719
+ const vpsCapture = () => vps.vpsComputerScreenshot(targetCfg, bot.id);
5720
+ autoVmClaims.set(threadId, {
5721
+ owner: resourceOwner,
5722
+ lazy: true,
5723
+ label: "the VPS computer",
5724
+ claim: async () => {
5725
+ await bindTurnComputer(resourceOwner, vpsResource, true);
5726
+ previewCapture = vpsCapture;
5727
+ if (threadBusy(bot.id, threadId)) {
5728
+ const touched = screenPollers.get(threadId)?.touched ?? false;
5729
+ stopScreenPoller(bot.id, threadId);
5730
+ startScreenPoller(bot.id, threadId, { computer: vpsCapture, ...(browserCapture ? { browser: browserCapture } : {}) }, { screenIsTheWork: touched });
5731
+ }
5732
+ },
5733
+ });
5335
5734
  }
5336
5735
  else {
5337
- activeVpsThreads.delete(bot.id);
5736
+ vpsThreadEnded(bot.id, threadId);
5338
5737
  if (wants === "cloud") {
5339
5738
  throw new Error(remote?.problem ?? "the VPS computer could not be created or reached");
5340
5739
  }
@@ -5616,10 +6015,17 @@ async function startTurn(botId, text, opts) {
5616
6015
  { id: "mentions", label: "Mentions", text: boundedCoordination && tagged.length ? `The user named these existing teammates: ${tagged.map(b => `${peerName(b.name)} (${b.id})`).join(", ")}. Use coordinate_bots when their contribution is needed; do not substitute native helper agents for these bots.` : mentionPrompt(tagged) },
5617
6016
  ]);
5618
6017
  runningTurnEngines.set(threadId, instance);
6018
+ // The prompt carries the soul as saved now. If it changed during setup,
6019
+ // decide again from what is actually sent.
6020
+ const dispatchedConfig = sessionConfig(liveBot?.soul ?? bot.soul);
6021
+ if (strictResume && dispatchedConfig !== plannedConfig)
6022
+ dispatchContext = decideContext(dispatchedConfig);
6023
+ // Before sendTurn: an adapter may emit the whole turn before it resolves.
6024
+ handoffs.dispatching(threadId, dispatchClaimId, dispatchContext.handoff);
5619
6025
  const dispatch = await guardTurnDispatch(instance.adapter.sendTurn({
5620
6026
  threadId,
5621
6027
  botId: bot.id,
5622
- text: turnText,
6028
+ text: dispatchContext.turnText,
5623
6029
  refreshSystemPrompt: true,
5624
6030
  images: turnImages,
5625
6031
  approvalMode: approvalModeForTurn(bot, commsDepth > 0),
@@ -5629,8 +6035,9 @@ async function startTurn(botId, text, opts) {
5629
6035
  // a rewound thread never resumes the abandoned branch's session
5630
6036
  // the active task's own session — another task's cursor would
5631
6037
  // resume the wrong conversation and defeat the context bubble
5632
- resumeCursor,
5633
- ...(recoveryText !== undefined ? { recoveryText } : {}),
6038
+ resumeCursor: dispatchContext.resumeCursor,
6039
+ ...(dispatchContext.recoveryText !== undefined ? { recoveryText: dispatchContext.recoveryText } : {}),
6040
+ ...(dispatchContext.recoveryIsReplay ? { recoveryIsReplay: true } : {}),
5634
6041
  transcript,
5635
6042
  system: prompt.text,
5636
6043
  systemStable: prompt.stable,
@@ -5646,6 +6053,7 @@ async function startTurn(botId, text, opts) {
5646
6053
  throw new DirectTurnSetupCancelled("turn stopped during provider setup");
5647
6054
  }
5648
6055
  bindInternalCapabilityToProviderTurn(threadId, dispatchClaimId, dispatch.value.turnId);
6056
+ handoffs.bindTurn(threadId, dispatchClaimId, dispatch.value.turnId);
5649
6057
  if (directFollowupSettlers.has(dispatchClaimId) && dispatch.value.turnId &&
5650
6058
  !directFollowupTurns.bind(threadId, dispatchClaimId, dispatch.value.turnId)) {
5651
6059
  // This exact queued turn completed before its dispatch ACK arrived.
@@ -5688,6 +6096,7 @@ async function startTurn(botId, text, opts) {
5688
6096
  }
5689
6097
  }
5690
6098
  catch (e) {
6099
+ handoffs.abandon(threadId, dispatchClaimId);
5691
6100
  if (computerSelectionTurns.get(threadId)?.generation === dispatchClaimId)
5692
6101
  computerSelectionTurns.delete(threadId);
5693
6102
  settleDirectFollowup(dispatchClaimId);
@@ -5698,8 +6107,7 @@ async function startTurn(botId, text, opts) {
5698
6107
  releaseTurnResources(resourceOwner);
5699
6108
  if (ownsLatestGeneration) {
5700
6109
  releaseLocalVmThread(threadId);
5701
- if (activeVpsThreads.get(bot.id) === threadId)
5702
- activeVpsThreads.delete(bot.id);
6110
+ vpsThreadEnded(bot.id, threadId);
5703
6111
  watchdog.settle(threadId);
5704
6112
  turnUsage.delete(threadId);
5705
6113
  turnContext.delete(threadId);
@@ -5740,6 +6148,7 @@ async function startTurn(botId, text, opts) {
5740
6148
  // for it, in its own thread, so it does not need a second channel.
5741
6149
  if (opts?.automationSource === undefined && !opts?.commsDepth && !opts?.cardContinuation) {
5742
6150
  notify(buildNotification("turn-failed", bot, threadId, redactSecretsInText(message), { avatarUrl: bot.avatarUrl }));
6151
+ reportIncident({ kind: "could-not-start", bot, threadId, detail: message });
5743
6152
  }
5744
6153
  store.setTaskActivity(bot.id, threadId, "idle");
5745
6154
  directTurnBots.delete(threadId);
@@ -5958,10 +6367,7 @@ routines = new RoutineManager({
5958
6367
  hasPendingDelegations: (threadId) => pendingThreads().includes(threadId) ||
5959
6368
  [...delegationWatch.values()].some((watch) => watch.sourceThreadId === threadId) ||
5960
6369
  pendingDelegationWakes.has(threadId),
5961
- botState: (botId) => {
5962
- const bot = store.bot(botId);
5963
- return !bot ? "missing" : bot.busy ? "busy" : "ready";
5964
- },
6370
+ botState: unattendedDispatchState,
5965
6371
  goalState: (groupId, coordinatorBotId) => {
5966
6372
  const group = store.group(groupId);
5967
6373
  const coordinator = store.bot(coordinatorBotId);
@@ -6000,6 +6406,7 @@ routines = new RoutineManager({
6000
6406
  const task = store.taskByThread(botId, threadId);
6001
6407
  if (task && !task.busy && !task.routineRunId && store.messagesFor(threadId).length === 0) {
6002
6408
  store.deleteTask(botId, threadId);
6409
+ handoffs.forget(threadId);
6003
6410
  }
6004
6411
  },
6005
6412
  startTurn: (botId, threadId, prompt, runOn, triggerSource, onDispatchError) => startTurn(botId, prompt, { threadId, runOn, automationSource: triggerSource, onDispatchError })
@@ -6040,6 +6447,7 @@ routines = new RoutineManager({
6040
6447
  const detail = run.error ? `${run.routineName}: ${run.error}` : run.routineName;
6041
6448
  const notificationBot = routineSourceOwner(run)?.bot ?? bot;
6042
6449
  notify(buildNotification("routine-failed", notificationBot, routineSourceThread(run) ?? run.threadId ?? bot.threadId, detail));
6450
+ reportIncident({ kind: "routine-failed", bot, threadId: run.threadId ?? bot.threadId, detail });
6043
6451
  },
6044
6452
  onRunDeferred: (run) => {
6045
6453
  const bot = store.bot(run.botId);
@@ -6342,12 +6750,6 @@ async function deleteBotWithLifecycle(botId, revalidate = () => { }, setupReques
6342
6750
  const acknowledged = await browserCleanup.ensure(committedCleanup);
6343
6751
  requireBrowserCleanupAcknowledged(acknowledged, `Browser data for ${bot.name}`);
6344
6752
  }
6345
- for (const dir of [EVENTS_DIR, NATIVE_DIR]) {
6346
- try {
6347
- unlinkSync(join(dir, `${bot.threadId}.ndjson`));
6348
- }
6349
- catch { }
6350
- }
6351
6753
  return deletionResponse(200, { ok: true });
6352
6754
  }
6353
6755
  finally {
@@ -6656,10 +7058,7 @@ function resolveAndSendProfile(res, args) {
6656
7058
  // ordered behind a busy MAUS and gives webhook runs the same durable receipts.
6657
7059
  const webhooks = new WebhookManager({
6658
7060
  emit: broadcast,
6659
- botState: (botId) => {
6660
- const bot = store.bot(botId);
6661
- return !bot ? "missing" : bot.busy ? "busy" : "ready";
6662
- },
7061
+ botState: unattendedDispatchState,
6663
7062
  enqueue: (input) => routines.enqueueWebhook(input),
6664
7063
  findRun: (webhookId, deliveryId) => routines.webhookRunReceipt(webhookId, deliveryId),
6665
7064
  cancelQueued: (webhookId, message) => routines.cancelQueuedWebhook(webhookId, message),
@@ -6794,7 +7193,11 @@ setupRetry = 0) {
6794
7193
  }
6795
7194
  revokeInternalCapabilitiesForThread(threadId);
6796
7195
  spoken.add(botId);
6797
- const preparedApprovalMode = approvalModeForTurn(bot, Boolean(orchestration?.roomHandoffId));
7196
+ // Must be the SAME resolver the readiness re-check uses below, or a Chief's
7197
+ // delegated Full elevation makes the two disagree by construction: every
7198
+ // such room turn then reads as "settings changed", retries once, and
7199
+ // settles as busy without ever dispatching.
7200
+ const preparedApprovalMode = roomTurnApprovalMode(bot, orchestration);
6798
7201
  const preparedSelection = { ...bot.modelSelection };
6799
7202
  const preparedComposio = bot.composio;
6800
7203
  const instance = turnInstance(bot);
@@ -6922,7 +7325,7 @@ setupRetry = 0) {
6922
7325
  if (!readyGroup || !stillOwnsThread || !readyGroup.memberIds.includes(readyBot.id))
6923
7326
  return false;
6924
7327
  const setupChanged = turnInstance(readyBot) !== instance ||
6925
- approvalModeForTurn(readyBot, Boolean(orchestration?.roomHandoffId)) !== preparedApprovalMode ||
7328
+ roomTurnApprovalMode(readyBot, orchestration) !== preparedApprovalMode ||
6926
7329
  readyBot.modelSelection.instanceId !== preparedSelection.instanceId ||
6927
7330
  readyBot.modelSelection.model !== preparedSelection.model ||
6928
7331
  readyBot.modelSelection.effort !== preparedSelection.effort ||
@@ -7295,7 +7698,7 @@ setupRetry = 0) {
7295
7698
  text,
7296
7699
  refreshSystemPrompt: true,
7297
7700
  images: turnImages,
7298
- approvalMode: approvalModeForTurn(readyBot, Boolean(orchestration?.roomHandoffId)),
7701
+ approvalMode: roomTurnApprovalMode(readyBot, orchestration),
7299
7702
  system: roomSystem.text,
7300
7703
  systemStable: roomSystem.stable,
7301
7704
  systemVolatile: roomSystem.volatile,
@@ -9111,8 +9514,7 @@ async function reloadProviders() {
9111
9514
  stopScreenPoller(botId, threadId);
9112
9515
  releaseLocalVmThread(threadId);
9113
9516
  releaseTurnResources(owner);
9114
- if (activeVpsThreads.get(botId) === threadId)
9115
- activeVpsThreads.delete(botId);
9517
+ vpsThreadEnded(botId, threadId);
9116
9518
  watchdog.settle(threadId);
9117
9519
  closeOpenApprovals(threadId);
9118
9520
  directTurnBots.delete(threadId);
@@ -10408,19 +10810,27 @@ const handleRequest = async (req, res) => {
10408
10810
  if (method === "POST" && path === "/api/internal/ask-bot") {
10409
10811
  const body = await readInternalBody();
10410
10812
  const fromBotId = internalSender.id;
10411
- const toBotId = String(body.toBotId ?? "");
10813
+ const toBotRef = String(body.toBotId ?? "");
10412
10814
  const message = String(body.message ?? "").trim();
10413
10815
  if (body.depth !== undefined &&
10414
10816
  (!Number.isInteger(body.depth) || body.depth < 0 || body.depth !== internalCapability.depth)) {
10415
10817
  return json(res, 403, { error: "the recursion depth does not match this turn" });
10416
10818
  }
10417
10819
  const depth = internalCapability.depth;
10418
- if (!toBotId || !message)
10820
+ if (!toBotRef || !message)
10419
10821
  return json(res, 400, { error: "toBotId and message required" });
10420
- if (toBotId === fromBotId)
10822
+ if (toBotRef === fromBotId)
10421
10823
  return json(res, 400, { error: "a bot cannot message itself" });
10422
10824
  if (depth >= MAX_COMMS_DEPTH)
10423
10825
  return json(res, 200, { error: "message chains are limited to one hop" });
10826
+ // A unique reachable teammate name is accepted where an id is
10827
+ // expected; see resolveTeammate for why.
10828
+ const resolvedTo = resolveTeammate(store.bots, internalSender, toBotRef);
10829
+ if ("error" in resolvedTo)
10830
+ return json(res, 404, { error: `no such bot: ${resolvedTo.error}` });
10831
+ if (resolvedTo.id === fromBotId)
10832
+ return json(res, 400, { error: "a bot cannot message itself" });
10833
+ const toBotId = resolvedTo.id;
10424
10834
  const target = store.bot(toBotId);
10425
10835
  if (!target)
10426
10836
  return json(res, 404, { error: "no such bot" });
@@ -10621,9 +11031,55 @@ const handleRequest = async (req, res) => {
10621
11031
  await new Promise((wake) => setTimeout(wake, 500));
10622
11032
  }
10623
11033
  }
11034
+ // A Chief resumes a teammate's broken thread (server/incidents.ts): the
11035
+ // same thread, its conversation and files, one more turn, with a line
11036
+ // saying who asked and why. Chief-only, for a teammate it can reach,
11037
+ // never a room (coordinate there) and never a thread still running.
11038
+ if (method === "POST" && path === "/api/internal/retry-thread") {
11039
+ const body = await readInternalBody();
11040
+ const from = internalSender;
11041
+ const fromThreadId = internalCapability.threadId;
11042
+ if (!from.chiefOfStaff || from.hidden)
11043
+ return json(res, 403, { error: "only a Chief of Staff can retry a teammate's thread" });
11044
+ // `toBotId`/`toThreadId`: the guard above reads bare botId/threadId as
11045
+ // the caller's own identity, the way every internal route does.
11046
+ const botId = typeof body.toBotId === "string" ? body.toBotId : "";
11047
+ const threadId = typeof body.toThreadId === "string" ? body.toThreadId : "";
11048
+ const note = typeof body.note === "string" ? body.note.trim().slice(0, 300) : "";
11049
+ const target = store.bot(botId);
11050
+ if (!target || target.id === from.id)
11051
+ return json(res, 404, { error: "no such teammate" });
11052
+ if (target.hidden || !canAccessTeam(from, target.section) || !peerAllowed(from, target.id)) {
11053
+ return json(res, 403, { error: "that bot is not on this Chief's team — call list_bots for the ones you can reach" });
11054
+ }
11055
+ if (store.groupByThread(threadId))
11056
+ return json(res, 400, { error: "that is a room thread — use coordinate_bots in the room instead" });
11057
+ const task = store.taskByThread(target.id, threadId);
11058
+ if (!task)
11059
+ return json(res, 404, { error: "no such thread on that bot" });
11060
+ if (threadBusy(target.id, threadId) || queuedThreadPosition(target.id, threadId) !== null) {
11061
+ return json(res, 409, { error: "that thread is still running — wait for it to settle before retrying" });
11062
+ }
11063
+ requireActiveInternalCapability();
11064
+ const unattended = isUnattended(from.id, fromThreadId);
11065
+ const text = `[Retry requested by ${from.name}, your Chief of Staff, after this thread's last run stopped.${note ? ` Note from ${from.name}: ${note}` : ""} Continue the request above from where it stopped and finish it. If the same problem comes back, say exactly what is blocking and stop.]`;
11066
+ try {
11067
+ await startTurn(target.id, text, { threadId, unattended, peerAsk: { botId: from.id, name: from.name, ...(unattended ? { unattended: true } : {}) } });
11068
+ }
11069
+ catch (error) {
11070
+ return json(res, 409, { error: error instanceof Error ? error.message : String(error) });
11071
+ }
11072
+ store.appendMessage(fromThreadId, {
11073
+ role: "bot",
11074
+ kind: "activity",
11075
+ tool: { name: `Retried ${target.name}'s thread #${task.title}`, ok: true },
11076
+ threadRef: { botId: target.id, threadId, title: task.title },
11077
+ });
11078
+ return json(res, 200, { started: true, message: `${target.name}'s thread #${task.title} is running again. Its result stays in that thread; you are not woken for it — check later with list_threads or session_search if you need to.` });
11079
+ }
10624
11080
  if (method === "POST" && path === "/api/internal/delegate-bot") {
10625
11081
  const body = await readInternalBody();
10626
- const toBotId = String(body.toBotId ?? "");
11082
+ const toBotRef = String(body.toBotId ?? "");
10627
11083
  const message = String(body.message ?? "").trim();
10628
11084
  const reason = typeof body.reason === "string" && body.reason.trim() ? body.reason.trim() : undefined;
10629
11085
  if (body.depth !== undefined &&
@@ -10631,9 +11087,13 @@ const handleRequest = async (req, res) => {
10631
11087
  return json(res, 403, { error: "the recursion depth does not match this turn" });
10632
11088
  }
10633
11089
  const depth = internalCapability.depth;
10634
- if (!toBotId || !message)
11090
+ if (!toBotRef || !message)
10635
11091
  return json(res, 400, { error: "toBotId and message required" });
10636
11092
  const from = internalSender;
11093
+ const resolvedTo = resolveTeammate(store.bots, from, toBotRef);
11094
+ if ("error" in resolvedTo)
11095
+ return json(res, 404, { error: `no such bot: ${resolvedTo.error}` });
11096
+ const toBotId = resolvedTo.id;
10637
11097
  const target = store.bot(toBotId);
10638
11098
  if (!target)
10639
11099
  return json(res, 404, { error: "no such bot" });
@@ -10704,11 +11164,35 @@ const handleRequest = async (req, res) => {
10704
11164
  const destination = groupId ? store.group(groupId) : undefined;
10705
11165
  if (groupId && !destination)
10706
11166
  return json(res, 404, { error: "No such room; use list_room_targets." });
10707
- const targets = parsed.data.botIds.map(botId => ({ groupId: destination?.id,
11167
+ // A slot may carry a teammate's name instead of its id — the
11168
+ // roster shows both, list_bots shows both, and a Chief reading its
11169
+ // prompt reaches for the name. A unique reachable name resolves;
11170
+ // anything else is refused with the id or name the caller sent
11171
+ // and the way to the real ids (peer-roster.ts).
11172
+ const botIds = [];
11173
+ for (const raw of parsed.data.botIds) {
11174
+ const resolved = resolveTeammate(store.bots, internalSender, raw);
11175
+ if ("error" in resolved)
11176
+ return json(res, 403, { error: resolved.error });
11177
+ botIds.push(resolved.id);
11178
+ }
11179
+ if (new Set(botIds).size !== botIds.length)
11180
+ return json(res, 400, { error: "bot_ids name the same teammate twice — send each teammate once" });
11181
+ const targets = botIds.map(botId => ({ groupId: destination?.id,
10708
11182
  threadId: destination ? destination.id === source?.id ? address.threadId : destination.threadId : store.bot(botId)?.threadId ?? "", botId,
10709
11183
  }));
10710
11184
  for (const target of targets) {
10711
- const eligibility = target.botId === internalSender.id ? "Choose a teammate, not yourself" : roomHandoffProblem(target, address);
11185
+ // roomHandoffProblem's "no longer exists" is written for a route
11186
+ // that was valid and went away. Here the id is the model's own
11187
+ // argument — usually a display name dropped into a bot_ids slot —
11188
+ // so say which id failed and where the real ones are, instead of
11189
+ // telling the model a teammate it can still reach is gone. Only
11190
+ // the id the caller sent is echoed back, never a bot's name.
11191
+ const addressed = store.bot(target.botId);
11192
+ const eligibility = target.botId === internalSender.id ? "Choose a teammate, not yourself"
11193
+ : !addressed ? `No bot with id "${target.botId}" — call list_bots and copy the exact id from the result`
11194
+ : addressed.hidden ? `The bot with id "${target.botId}" is no longer available — call list_bots for the ones you can reach`
11195
+ : roomHandoffProblem(target, address);
10712
11196
  if (eligibility)
10713
11197
  return json(res, 403, { error: eligibility });
10714
11198
  }
@@ -10744,6 +11228,9 @@ const handleRequest = async (req, res) => {
10744
11228
  target.threadId = resolved.task.threadId;
10745
11229
  if (resolved.created)
10746
11230
  createdThread = resolved.task.threadId;
11231
+ if (delegatedFullAccess(internalSender, internalCapability.threadId, store.bot(target.botId))) {
11232
+ grantDelegatedFullAccess(internalSender, store.bot(target.botId), target.threadId);
11233
+ }
10747
11234
  }
10748
11235
  const { node, duplicate } = roomHandoffs.enqueue(address, internalCapability.generation, internalCapability.roomHandoffId, target, parsed.data.requestKey + ":" + target.botId, parsed.data.message, approvalGranted, parsed.data.rework, [...store.messagesFor(address.threadId)].reverse().find(m => m.role === "user" && m.kind === "text")?.text ?? "");
10749
11236
  // A re-dispatched request_key is answered by the request it
@@ -11012,6 +11499,8 @@ const handleRequest = async (req, res) => {
11012
11499
  const task = store.createTask(target.id, title, false, projectId, { botId: from.id, name: from.name, at: Date.now() });
11013
11500
  if (!task)
11014
11501
  return json(res, 500, { error: "couldn't create that thread" });
11502
+ if (delegatedFullAccess(from, fromThreadId, target))
11503
+ grantDelegatedFullAccess(from, target, task.threadId);
11015
11504
  const queued = queueDelegation(commsBus, from, { toBotId: target.id, message, depth, targetThreadId: task.threadId }, MAX_COMMS_DEPTH, fromThreadId);
11016
11505
  if (queued.result !== "ok" || !queued.id) {
11017
11506
  // nothing will ever run there: take the row back before the person
@@ -11258,6 +11747,48 @@ const handleRequest = async (req, res) => {
11258
11747
  return json(res, 404, { error: "no such bot" });
11259
11748
  if (method === "GET") {
11260
11749
  const snapshot = botComputerControlSnapshot(botId, internalCapability.teamComputerId);
11750
+ const slot = autoVmClaims.get(internalCapability.threadId);
11751
+ const lazyClaim = slot && slot.owner.generation === internalCapability.generation ? slot : undefined;
11752
+ if (!snapshot.held && lazyClaim?.lazy && !lazyClaim.begin) {
11753
+ // First screen tools/call on a lazily-attached Auto VM (issue
11754
+ // #1361): fire the exclusive claim — once — and give it a moment
11755
+ // to land. A free, ready VM claims in the time of one container
11756
+ // inspect, so this call then proceeds with an honest answer;
11757
+ // only a claim still queued behind another holder answers held
11758
+ // below, and then the contention text is true. Keyed on the
11759
+ // slot, never on the thread's turn-computer entry: a bind this
11760
+ // turn abandoned earlier (a VPS that turned out to be asleep)
11761
+ // must not hide the unclaimed VM and let the call through.
11762
+ startAutoVmClaim(autoVmClaims, internalCapability.threadId, internalCapability.generation);
11763
+ await Promise.race([
11764
+ lazyClaim.begin ?? Promise.resolve(),
11765
+ new Promise((resolve) => setTimeout(resolve, LAZY_VM_CLAIM_GRACE_MS)),
11766
+ ]);
11767
+ }
11768
+ if (!snapshot.held && lazyClaim?.failed === true) {
11769
+ // A rejected lazy claim (gate finding F1, issue #1361): the
11770
+ // computer MCP mounted at dispatch is still live, and the claim
11771
+ // may even have left a turn-computer entry behind (it can reject
11772
+ // after bindTurnComputer succeeded — lease lost to a person,
11773
+ // lifecycle busy, boot failure). Either way this turn owns no
11774
+ // usable VM, so keep refusing every screen call for the rest of
11775
+ // the generation; the bridge must never forward one onto a VM
11776
+ // this turn never claimed. Turn settle GC clears the slot. Say
11777
+ // why, and say not to retry: the contention text would send the
11778
+ // model into a screenshot loop against a claim that cannot land.
11779
+ return json(res, 200, {
11780
+ held: true, helpOpen: false,
11781
+ blockedReason: `This turn could not claim ${lazyClaim.label ?? "this computer"}${lazyClaim.failure ? ` (${lazyClaim.failure})` : ""}. This call was not performed. Do not retry computer work in this turn; tell the person what you could not do.`,
11782
+ });
11783
+ }
11784
+ if (!snapshot.held && lazyClaim?.lazy && lazyClaim.begin && !lazyClaim.claimed) {
11785
+ // The claim fired and is still waiting on the exclusive bind:
11786
+ // another turn genuinely holds this desktop right now.
11787
+ return json(res, 200, {
11788
+ held: true, helpOpen: false,
11789
+ blockedReason: "Another thread is using this computer. This call was not performed. Pause computer work until that thread finishes, then take a fresh screenshot before acting.",
11790
+ });
11791
+ }
11261
11792
  const computer = turnComputerResources.get(internalCapability.threadId);
11262
11793
  if (!snapshot.held && computer && computer.owner.generation === internalCapability.generation &&
11263
11794
  !claimTurnResource(computer.owner, computer.resource)) {
@@ -12567,14 +13098,6 @@ const handleRequest = async (req, res) => {
12567
13098
  routines.disableForGroup(group.id);
12568
13099
  store.deleteGroup(group.id);
12569
13100
  rejectDeletedThreadSkillStages(stagedSkillCleanups);
12570
- for (const threadId of threadIds) {
12571
- for (const dir of [EVENTS_DIR, NATIVE_DIR]) {
12572
- try {
12573
- unlinkSync(join(dir, `${threadId}.ndjson`));
12574
- }
12575
- catch { }
12576
- }
12577
- }
12578
13101
  return json(res, 200, { ok: true });
12579
13102
  }
12580
13103
  m = path.match(/^\/api\/groups\/([\w-]+)\/messages$/);
@@ -12682,6 +13205,112 @@ const handleRequest = async (req, res) => {
12682
13205
  }
12683
13206
  return json(res, 200, { ok: true });
12684
13207
  }
13208
+ // Steer a queued room message into the RUNNING room turn (no interrupt).
13209
+ // Only the head steers — room queues drain one item at a time — and the
13210
+ // engine that receives it is the thread's live speaker. A room whose
13211
+ // running driver cannot steer keeps its queue, exactly like an incapable
13212
+ // 1:1 engine; this never ends the running turn.
13213
+ m = path.match(/^\/api\/groups\/([\w-]+)\/queue\/([\w-]+)\/steer$/);
13214
+ if (m && method === "POST") {
13215
+ const body = await readBody(req);
13216
+ if (body !== null && (typeof body !== "object" || Array.isArray(body))) {
13217
+ return json(res, 400, { error: "body must be a JSON object" });
13218
+ }
13219
+ const threadId = typeof body?.threadId === "string" ? body.threadId : undefined;
13220
+ if (threadId !== undefined && !/^[\w-]+$/.test(threadId)) {
13221
+ return json(res, 400, { error: "threadId must be a task id" });
13222
+ }
13223
+ const group = store.group(m[1]);
13224
+ if (!group)
13225
+ return json(res, 404, { error: "no such room" });
13226
+ const targetThreadId = threadId ?? group.threadId;
13227
+ const ownsThread = group.dm
13228
+ ? group.threadId === targetThreadId
13229
+ : Boolean(store.groupTaskByThread(group.id, targetThreadId));
13230
+ if (!ownsThread) {
13231
+ return json(res, 409, { error: "the channel switched tasks before it could receive the message" });
13232
+ }
13233
+ noteTurnTrigger(targetThreadId, auth);
13234
+ const current = store.group(group.id);
13235
+ if (!current)
13236
+ return json(res, 404, { error: "no such room" });
13237
+ // Which engine owns the running room turn on this thread? The same
13238
+ // resolution the room's own Stop uses: the live speaker, else the busy
13239
+ // bot on the channel's main thread.
13240
+ const speakerBotId = groupSpeakers.get(targetThreadId)?.botId ??
13241
+ (targetThreadId === current.threadId ? current.busyBotId : undefined);
13242
+ const speaker = speakerBotId ? store.bot(speakerBotId) : undefined;
13243
+ const instance = speaker ? runningTurnInstance(speaker, targetThreadId) : undefined;
13244
+ // Lift the queue atomically: the room settling can drain it as the
13245
+ // next follow-up, or this request can steer its head into the live
13246
+ // turn — never both for the same words.
13247
+ const held = holdChannelQueue(current.id, targetThreadId, m[2]);
13248
+ if (!held)
13249
+ return json(res, 404, { error: "no such queued message" });
13250
+ if (!speaker || !instance?.adapter.capabilities.queueing || !instance.adapter.steer) {
13251
+ restoreHeldChannelQueue(held);
13252
+ return json(res, 200, { ok: true, queued: true, threadId: targetThreadId });
13253
+ }
13254
+ const [head] = held.items;
13255
+ if (!head || head.id !== m[2]) {
13256
+ restoreHeldChannelQueue(held);
13257
+ return json(res, 409, { error: "only the first queued message can steer" });
13258
+ }
13259
+ // A reply target that cannot be resolved restores the held queue
13260
+ // before the request fails — the room's normal drain keeps the head.
13261
+ const replyTo = resolveHeldReplyTarget(held, resolveReplyTarget);
13262
+ const steered = await instance.adapter
13263
+ .steer(targetThreadId, promptWithReply(head.text, replyTo, cfg.profile?.name?.trim() || "User"))
13264
+ .catch(() => "indeterminate");
13265
+ // The steer was awaited adapter work: re-read every ownership
13266
+ // invariant before writing anything, exactly like the 1:1 path. A
13267
+ // speaker change, a channel switch, or a settled room restores the
13268
+ // queue instead of recording words the new turn never saw.
13269
+ const after = store.group(current.id);
13270
+ const afterSpeakerBotId = after
13271
+ ? groupSpeakers.get(targetThreadId)?.botId ??
13272
+ (targetThreadId === after.threadId ? after.busyBotId : undefined)
13273
+ : undefined;
13274
+ // "indeterminate" (timeout after delivery, lost transport, a settle
13275
+ // race) never restores: the words may already be folded into the turn
13276
+ // that was live when they were sent, and replaying them into a new
13277
+ // turn would run them twice. Record them once — even under a new
13278
+ // speaker — and settle the head.
13279
+ const delivered = steered !== "refused";
13280
+ if (after && delivered && (steered === "indeterminate" || afterSpeakerBotId === speakerBotId)) {
13281
+ const message = store.appendMessage(targetThreadId, {
13282
+ role: "user",
13283
+ kind: "text",
13284
+ text: head.text,
13285
+ replyToId: head.replyToId,
13286
+ sendId: head.sendId,
13287
+ channelMode: head.mode,
13288
+ queueId: head.id,
13289
+ via: head.via,
13290
+ steered: true,
13291
+ });
13292
+ settleHeldChannelQueueHead(held);
13293
+ return json(res, 200, {
13294
+ ok: true,
13295
+ steered: true,
13296
+ threadId: targetThreadId,
13297
+ messages: [message],
13298
+ queueIds: [head.id],
13299
+ });
13300
+ }
13301
+ if (steered === "indeterminate" && !after) {
13302
+ // The room vanished while the answer was lost: settle the head so a
13303
+ // restart cannot replay words the dead turn may already have run.
13304
+ settleHeldChannelQueueHead(held);
13305
+ return json(res, 404, { error: "no such room" });
13306
+ }
13307
+ restoreHeldChannelQueue(held);
13308
+ // The room may have settled while the steer was refused; a queue that
13309
+ // is now drainable must not strand behind a missed settle.
13310
+ if (after && !groupIsWorking(after))
13311
+ drainQueuedChannelSends();
13312
+ return json(res, 200, { ok: true, queued: true, threadId: targetThreadId });
13313
+ }
12685
13314
  m = path.match(/^\/api\/groups\/([\w-]+)\/interrupt$/);
12686
13315
  if (m && method === "POST") {
12687
13316
  const group = store.group(m[1]);
@@ -14055,15 +14684,16 @@ const handleRequest = async (req, res) => {
14055
14684
  // existing server-side queue records it atomically for the next turn.
14056
14685
  if (currentAtStart.busy) {
14057
14686
  const instance = runningTurnInstance(currentAtStart, threadId);
14058
- let steered = false;
14687
+ let steered = "refused";
14059
14688
  // A live text steer has no image side channel. Keep an attachment
14060
14689
  // message intact for the next ordinary turn, where central image
14061
14690
  // admission can hand it to the provider natively.
14062
14691
  const carriesImages = extractTurnImages(text).images.length > 0;
14692
+ const steerTarget = handoffs.current(threadId);
14063
14693
  if (!carriesImages && !computerSelectionTurns.get(threadId)?.selected && instance?.adapter.capabilities.queueing && instance.adapter.steer) {
14064
14694
  steered = await instance.adapter
14065
14695
  .steer(threadId, promptWithReply(text, replyTo, cfg.profile?.name?.trim() || "User"))
14066
- .catch(() => false);
14696
+ .catch(() => "indeterminate");
14067
14697
  }
14068
14698
  // steer() is awaited adapter work. The turn can settle, the task can
14069
14699
  // switch, or the whole bot can be deleted before its acknowledgement
@@ -14077,10 +14707,15 @@ const handleRequest = async (req, res) => {
14077
14707
  if (!store.taskByThread(bot.id, threadId)) {
14078
14708
  throw Object.assign(new Error("the target task no longer exists"), { status: 409 });
14079
14709
  }
14080
- if (steered) {
14081
- if (!current.busy) {
14710
+ const delivered = steered !== "refused";
14711
+ if (delivered) {
14712
+ if (steered === "steered" && !current.busy) {
14082
14713
  throw Object.assign(new Error("the running turn ended before the steered message could be recorded"), { status: 409 });
14083
14714
  }
14715
+ // "indeterminate" falls through to the same record: the words
14716
+ // may already be folded into a turn whose acknowledgement was
14717
+ // lost, and handing them back for a resend could run them
14718
+ // twice. Recording them once is the honest outcome.
14084
14719
  // A person steering a webhook turn is present, and auto mode may
14085
14720
  // follow them again. But this route is also reachable from the
14086
14721
  // bot's own shell on a headless server (loopback is the owner
@@ -14099,6 +14734,8 @@ const handleRequest = async (req, res) => {
14099
14734
  sendId,
14100
14735
  steered: true,
14101
14736
  });
14737
+ // Offered to the next turn again unless the person stops this one.
14738
+ handoffs.steered(threadId, steerTarget, instance?.instanceId, message.id);
14102
14739
  return { ok: true, steered: true, threadId, message };
14103
14740
  }
14104
14741
  if (!current.busy) {
@@ -14126,6 +14763,81 @@ const handleRequest = async (req, res) => {
14126
14763
  }
14127
14764
  return json(res, 200, { ok: true });
14128
14765
  }
14766
+ // Steer a queued message into the RUNNING turn (no interrupt). Engines
14767
+ // without a live steer keep the queue; this never ends the current turn.
14768
+ m = path.match(/^\/api\/bots\/([\w-]+)\/queue\/([\w-]+)\/steer$/);
14769
+ if (m && method === "POST") {
14770
+ const body = await readBody(req);
14771
+ requirePinnedClientThread(m[1], body?.threadId);
14772
+ const bot = requestedTaskBot(m[1], body?.threadId);
14773
+ noteTurnTrigger(bot.threadId, auth);
14774
+ // Lift the whole queue atomically: a settle racing this request can
14775
+ // drain it as a follow-up, or this request can steer it into the live
14776
+ // turn — never both for the same words.
14777
+ const held = holdSteeredQueue(bot.id, bot.threadId, m[2]);
14778
+ if (!held)
14779
+ return json(res, 404, { error: "no such queued message" });
14780
+ // A live steer has no image side channel. Attachment words wait for a
14781
+ // real turn where central admission can hand the images to the engine.
14782
+ if (held.items.some((item) => extractTurnImages(item.text).images.length > 0)) {
14783
+ restoreHeldSteeredQueue(held);
14784
+ return json(res, 200, { ok: true, queued: true, threadId: bot.threadId });
14785
+ }
14786
+ const currentAtStart = store.projectBotForTask(bot.id, bot.threadId);
14787
+ const instance = currentAtStart?.busy ? runningTurnInstance(currentAtStart, bot.threadId) : undefined;
14788
+ const prompt = held.items.map((item) => item.prompt).join("\n\n");
14789
+ const steerTarget = handoffs.current(bot.threadId);
14790
+ let steered = "refused";
14791
+ if (currentAtStart?.busy && instance?.adapter.capabilities.queueing && instance.adapter.steer) {
14792
+ steered = await instance.adapter
14793
+ .steer(bot.threadId, prompt)
14794
+ .catch(() => "indeterminate");
14795
+ }
14796
+ // The steer was awaited adapter work: re-read every ownership
14797
+ // invariant before writing anything, exactly like the live-send path.
14798
+ const current = store.projectBotForTask(bot.id, bot.threadId);
14799
+ // "indeterminate" never restores: the words may already be folded into
14800
+ // the turn that was live when they were sent, and replaying them into
14801
+ // a fresh follow-up turn would run them twice. Record them whenever
14802
+ // the destination still exists, busy or not.
14803
+ if ((steered === "steered" && current?.busy ||
14804
+ steered === "indeterminate" && current) &&
14805
+ store.taskByThread(bot.id, bot.threadId)) {
14806
+ if (auth.kind === "session" || DESKTOP_MANAGED)
14807
+ clearUnattended(bot.threadId);
14808
+ const messages = held.items.map((item) => store.appendMessage(bot.threadId, {
14809
+ role: "user",
14810
+ kind: "text",
14811
+ text: item.text,
14812
+ replyToId: item.replyToId,
14813
+ sendId: item.sendId,
14814
+ queueId: item.messageId,
14815
+ peerAsk: item.peerAsk,
14816
+ steered: true,
14817
+ }));
14818
+ // Offered to the next turn again unless the person stops this one.
14819
+ for (const message of messages)
14820
+ handoffs.steered(bot.threadId, steerTarget, instance?.instanceId, message.id);
14821
+ const queueIds = held.items.map((item) => item.messageId);
14822
+ settleHeldSteeredQueue(held);
14823
+ return json(res, 200, { ok: true, steered: true, threadId: bot.threadId, messages, queueIds });
14824
+ }
14825
+ if (steered === "indeterminate") {
14826
+ // No destination is left: settle so a restart cannot replay words a
14827
+ // dead turn may already have run.
14828
+ settleHeldSteeredQueue(held);
14829
+ if (!store.taskByThread(bot.id, bot.threadId)) {
14830
+ throw Object.assign(new Error("the target task no longer exists"), { status: 409 });
14831
+ }
14832
+ throw Object.assign(new Error("no such bot"), { status: 404 });
14833
+ }
14834
+ restoreHeldSteeredQueue(held);
14835
+ // The turn may have settled while the steer was refused; a queue that
14836
+ // is now drainable must not strand behind a missed settle.
14837
+ if (current && !current.busy)
14838
+ drainQueuedSends();
14839
+ return json(res, 200, { ok: true, queued: true, threadId: bot.threadId });
14840
+ }
14129
14841
  // edit a user message → fork the conversation there and rerun the turn.
14130
14842
  // Rewinding a live thread is refused, exactly like switching versions
14131
14843
  // below: interrupting mid-flight and branching under the dying turn is
@@ -14342,8 +15054,10 @@ const handleRequest = async (req, res) => {
14342
15054
  const routine = routines.activeBotRunForBot(bot.id);
14343
15055
  if (routine?.threadId === expectedThreadId)
14344
15056
  await routines.cancelRun(routine.id);
14345
- else
15057
+ else {
15058
+ handoffs.stoppedByPerson(expectedThreadId);
14346
15059
  await interruptDirectThread(bot.id, expectedThreadId);
15060
+ }
14347
15061
  return json(res, 200, { ok: true });
14348
15062
  }
14349
15063
  const directClaim = directTurnDispatchClaims.get(bot.threadId);
@@ -14379,6 +15093,7 @@ const handleRequest = async (req, res) => {
14379
15093
  directClaim?.threadId !== expectedThreadId) {
14380
15094
  return json(res, 409, { error: "the bot switched tasks before it could be interrupted" });
14381
15095
  }
15096
+ handoffs.stoppedByPerson(expectedThreadId ?? bot.threadId);
14382
15097
  await interruptDirectThread(bot.id, expectedThreadId ?? bot.threadId);
14383
15098
  return json(res, 200, { ok: true });
14384
15099
  }
@@ -14611,6 +15326,7 @@ const handleRequest = async (req, res) => {
14611
15326
  const updated = store.deleteTask(m[1], m[2]);
14612
15327
  if (!updated)
14613
15328
  return json(res, 404, { error: "no such task" });
15329
+ handoffs.forget(m[2]);
14614
15330
  settleDirectFollowup(directTurnGenerationByThread.get(m[2]));
14615
15331
  rejectDeletedThreadSkillStages(stagedSkillCleanups);
14616
15332
  const fresh = botWithThread(updated);
@@ -16464,6 +17180,37 @@ try {
16464
17180
  catch (error) {
16465
17181
  console.warn(`attachments: startup partial cleanup failed: ${error instanceof Error ? error.message : String(error)}`);
16466
17182
  }
17183
+ // #1280: retention for per-thread event logs. Off unless configured, and
17184
+ // even then it only removes log files — transcripts, thread records, and
17185
+ // workspace state stay untouched. A thread qualifies only when its newest
17186
+ // close or archive stamp is older than the window and it is not busy,
17187
+ // unread, or carrying an open direct handoff.
17188
+ const THREAD_LOG_RETENTION_SWEEP_MS = 24 * 60 * 60 * 1000;
17189
+ function sweepThreadEventLogsNow() {
17190
+ const retentionDays = threadEventLogRetentionDays(cfg);
17191
+ if (retentionDays === null)
17192
+ return;
17193
+ const candidates = store.bots.flatMap((bot) => (bot.tasks ?? []).map((task) => ({
17194
+ threadId: task.threadId,
17195
+ closedAt: task.closedBy?.at ?? null,
17196
+ archivedAt: task.archivedAt ?? null,
17197
+ unread: task.unread === true,
17198
+ busy: threadBusy(bot.id, task.threadId),
17199
+ openDirectHandoff: roomHandoffs.activeDirect(task.threadId),
17200
+ })));
17201
+ const swept = sweepThreadEventLogs(candidates, retentionDays);
17202
+ if (swept > 0)
17203
+ console.log(`[retention] removed event logs for ${swept} idle thread(s) past ${retentionDays} day(s)`);
17204
+ }
17205
+ try {
17206
+ sweepThreadEventLogsNow();
17207
+ }
17208
+ catch (error) {
17209
+ console.warn(`thread event log retention sweep failed: ${error instanceof Error ? error.message : String(error)}`);
17210
+ }
17211
+ // A days-scale window needs no tighter cadence; unref so the timer never
17212
+ // holds the process open.
17213
+ setInterval(sweepThreadEventLogsNow, THREAD_LOG_RETENTION_SWEEP_MS).unref();
16467
17214
  // A dispatch claim is deliberately committed before transcript/provider work.
16468
17215
  // If we died after that point, its outcome is unknown: recover the user's words
16469
17216
  // and a review notice, never hand them to a model for a second execution.
@@ -16479,12 +17226,17 @@ for (const row of chatFollowups()) {
16479
17226
  }
16480
17227
  settleChatFollowups([row.id], "interrupted");
16481
17228
  const messages = store.messagesFor(row.threadId);
16482
- if (!messages.some((message) => message.queueId === row.id && message.role === "user")) {
16483
- store.appendMessage(row.threadId, {
16484
- role: "user", kind: "text", text: row.payload.text, replyToId: row.payload.replyToId,
16485
- sendId: row.payload.sendId, queueId: row.id,
16486
- ...(row.kind === "channel" ? { channelMode: row.payload.mode, via: row.payload.via } : {}),
16487
- });
17229
+ const recovered = messages.find((message) => message.queueId === row.id && message.role === "user") ?? store.appendMessage(row.threadId, {
17230
+ role: "user", kind: "text", text: row.payload.text, replyToId: row.payload.replyToId,
17231
+ sendId: row.payload.sendId, queueId: row.id,
17232
+ ...(row.kind === "channel" ? { channelMode: row.payload.mode, via: row.payload.via } : {}),
17233
+ });
17234
+ // Nor as a message a resumed session has not seen: count it as handed.
17235
+ const recoveredTask = row.kind === "bot" ? store.taskByThread(row.ownerId, row.threadId) : undefined;
17236
+ const order = store.activePath(row.threadId).filter(isContextMessage).map((m) => m.id);
17237
+ for (const [instanceId, state] of Object.entries(recoveredTask?.handedMessages ?? {})) {
17238
+ if (state.session !== undefined)
17239
+ store.setHandedMessages(row.ownerId, row.threadId, instanceId, recordHanded(state, order, [recovered.id]));
16488
17240
  }
16489
17241
  if (!messages.some((message) => message.queueId === row.id && message.kind === "activity")) {
16490
17242
  store.appendMessage(row.threadId, {