@monoes/monomindcli 2.10.10 → 2.10.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (243) hide show
  1. package/.claude/commands/mastermind/brain.md +14 -14
  2. package/.claude/commands/mastermind/help.md +2 -2
  3. package/.claude/commands/mastermind/master.md +24 -19
  4. package/.claude/commands/mastermind/memory.md +8 -8
  5. package/.claude/commands/mastermind/monoswarm.md +4 -4
  6. package/.claude/commands/mastermind.md +7 -7
  7. package/.claude/commands/truth/start.md +3 -3
  8. package/.claude/helpers/handlers/gates-handler.cjs +13 -1
  9. package/.claude/skills/mastermind/SKILL.md +7 -16
  10. package/.claude/skills/mastermind-debug/SKILL.md +3 -3
  11. package/.claude/skills/mastermind-design/SKILL.md +2 -0
  12. package/.claude/skills/mastermind-execute/SKILL.md +66 -11
  13. package/.claude/skills/mastermind-idea/SKILL.md +9 -2
  14. package/.claude/skills/mastermind-intake/SKILL.md +31 -7
  15. package/.claude/skills/mastermind-issue-detail/SKILL.md +70 -16
  16. package/.claude/skills/mastermind-issues/SKILL.md +111 -16
  17. package/.claude/skills/mastermind-liveness/SKILL.md +96 -26
  18. package/.claude/skills/mastermind-my-issues/SKILL.md +40 -8
  19. package/.claude/skills/mastermind-org/SKILL.md +2 -0
  20. package/.claude/skills/mastermind-plan/SKILL.md +7 -16
  21. package/.claude/skills/mastermind-plan-to-tasks/SKILL.md +132 -24
  22. package/.claude/skills/mastermind-protocol/SKILL.md +33 -22
  23. package/.claude/skills/mastermind-runorg/SKILL.md +22 -3
  24. package/.claude/skills/mastermind-skill-builder/SKILL.md +1 -1
  25. package/.claude/skills/mastermind-tasks/SKILL.md +5 -0
  26. package/.claude/skills/mastermind-techport/SKILL.md +1 -1
  27. package/.claude/skills/monodesign/scripts/detector/engines/browser/drivers.mjs +56 -15
  28. package/.claude/skills/performance-analysis/SKILL.md +1 -1
  29. package/.claude/skills/verification-quality/SKILL.md +2 -3
  30. package/README.md +2 -2
  31. package/dist/src/commands/doc.js +2 -2
  32. package/dist/src/commands/doc.js.map +1 -1
  33. package/dist/src/commands/doctor-project-checks.d.ts.map +1 -1
  34. package/dist/src/commands/doctor-project-checks.js +45 -2
  35. package/dist/src/commands/doctor-project-checks.js.map +1 -1
  36. package/dist/src/commands/memory-crud.d.ts.map +1 -1
  37. package/dist/src/commands/memory-crud.js +13 -2
  38. package/dist/src/commands/memory-crud.js.map +1 -1
  39. package/dist/src/commands/memory.js +1 -1
  40. package/dist/src/commands/memory.js.map +1 -1
  41. package/dist/src/commands/monograph.d.ts.map +1 -1
  42. package/dist/src/commands/monograph.js +11 -4
  43. package/dist/src/commands/monograph.js.map +1 -1
  44. package/dist/src/commands/org-observe.d.ts.map +1 -1
  45. package/dist/src/commands/org-observe.js +56 -6
  46. package/dist/src/commands/org-observe.js.map +1 -1
  47. package/dist/src/commands/org.d.ts +26 -0
  48. package/dist/src/commands/org.d.ts.map +1 -1
  49. package/dist/src/commands/org.js +144 -28
  50. package/dist/src/commands/org.js.map +1 -1
  51. package/dist/src/init/executor.d.ts.map +1 -1
  52. package/dist/src/init/executor.js +10 -9
  53. package/dist/src/init/executor.js.map +1 -1
  54. package/dist/src/init/settings-generator.d.ts.map +1 -1
  55. package/dist/src/init/settings-generator.js.map +1 -1
  56. package/dist/src/init/upgrade.d.ts.map +1 -1
  57. package/dist/src/init/upgrade.js +29 -0
  58. package/dist/src/init/upgrade.js.map +1 -1
  59. package/dist/src/init/write-codex.d.ts.map +1 -1
  60. package/dist/src/init/write-codex.js.map +1 -1
  61. package/dist/src/knowledge/document-pipeline.d.ts +5 -0
  62. package/dist/src/knowledge/document-pipeline.d.ts.map +1 -1
  63. package/dist/src/knowledge/document-pipeline.js +32 -16
  64. package/dist/src/knowledge/document-pipeline.js.map +1 -1
  65. package/dist/src/mcp-tools/hooks-routing.d.ts +9 -0
  66. package/dist/src/mcp-tools/hooks-routing.d.ts.map +1 -1
  67. package/dist/src/mcp-tools/hooks-routing.js +12 -1
  68. package/dist/src/mcp-tools/hooks-routing.js.map +1 -1
  69. package/dist/src/mcp-tools/knowledge-tools.d.ts.map +1 -1
  70. package/dist/src/mcp-tools/knowledge-tools.js +105 -13
  71. package/dist/src/mcp-tools/knowledge-tools.js.map +1 -1
  72. package/dist/src/mcp-tools/memory-tools.d.ts +13 -0
  73. package/dist/src/mcp-tools/memory-tools.d.ts.map +1 -1
  74. package/dist/src/mcp-tools/memory-tools.js +257 -31
  75. package/dist/src/mcp-tools/memory-tools.js.map +1 -1
  76. package/dist/src/mcp-tools/monograph/health-tools.d.ts.map +1 -1
  77. package/dist/src/mcp-tools/monograph/health-tools.js +99 -42
  78. package/dist/src/mcp-tools/monograph/health-tools.js.map +1 -1
  79. package/dist/src/mcp-tools/monograph/impact-tools.d.ts.map +1 -1
  80. package/dist/src/mcp-tools/monograph/impact-tools.js +123 -51
  81. package/dist/src/mcp-tools/monograph/impact-tools.js.map +1 -1
  82. package/dist/src/mcp-tools/monograph/query-tools.d.ts.map +1 -1
  83. package/dist/src/mcp-tools/monograph/query-tools.js +113 -96
  84. package/dist/src/mcp-tools/monograph/query-tools.js.map +1 -1
  85. package/dist/src/mcp-tools/monograph/shared.d.ts +37 -4
  86. package/dist/src/mcp-tools/monograph/shared.d.ts.map +1 -1
  87. package/dist/src/mcp-tools/monograph/shared.js +75 -56
  88. package/dist/src/mcp-tools/monograph/shared.js.map +1 -1
  89. package/dist/src/memory/memory-bridge.d.ts +60 -1
  90. package/dist/src/memory/memory-bridge.d.ts.map +1 -1
  91. package/dist/src/memory/memory-bridge.js +172 -42
  92. package/dist/src/memory/memory-bridge.js.map +1 -1
  93. package/dist/src/memory/memory-kg.d.ts +495 -28
  94. package/dist/src/memory/memory-kg.d.ts.map +1 -1
  95. package/dist/src/memory/memory-kg.js +2186 -251
  96. package/dist/src/memory/memory-kg.js.map +1 -1
  97. package/dist/src/memory/query-router.d.ts +51 -0
  98. package/dist/src/memory/query-router.d.ts.map +1 -1
  99. package/dist/src/memory/query-router.js +38 -2
  100. package/dist/src/memory/query-router.js.map +1 -1
  101. package/dist/src/orgrt/agent-exec.d.ts.map +1 -1
  102. package/dist/src/orgrt/agent-exec.js +7 -0
  103. package/dist/src/orgrt/agent-exec.js.map +1 -1
  104. package/dist/src/orgrt/agent-runner.d.ts +26 -1
  105. package/dist/src/orgrt/agent-runner.d.ts.map +1 -1
  106. package/dist/src/orgrt/agent-runner.js +73 -22
  107. package/dist/src/orgrt/agent-runner.js.map +1 -1
  108. package/dist/src/orgrt/antigravity-runner.d.ts +1 -1
  109. package/dist/src/orgrt/antigravity-runner.d.ts.map +1 -1
  110. package/dist/src/orgrt/antigravity-runner.js +36 -19
  111. package/dist/src/orgrt/antigravity-runner.js.map +1 -1
  112. package/dist/src/orgrt/approvals.d.ts +26 -2
  113. package/dist/src/orgrt/approvals.d.ts.map +1 -1
  114. package/dist/src/orgrt/approvals.js +67 -7
  115. package/dist/src/orgrt/approvals.js.map +1 -1
  116. package/dist/src/orgrt/broker.d.ts +21 -2
  117. package/dist/src/orgrt/broker.d.ts.map +1 -1
  118. package/dist/src/orgrt/broker.js +56 -8
  119. package/dist/src/orgrt/broker.js.map +1 -1
  120. package/dist/src/orgrt/bus.d.ts +8 -0
  121. package/dist/src/orgrt/bus.d.ts.map +1 -1
  122. package/dist/src/orgrt/bus.js +27 -0
  123. package/dist/src/orgrt/bus.js.map +1 -1
  124. package/dist/src/orgrt/checkpoint-ops.d.ts.map +1 -1
  125. package/dist/src/orgrt/checkpoint-ops.js +15 -5
  126. package/dist/src/orgrt/checkpoint-ops.js.map +1 -1
  127. package/dist/src/orgrt/checkpoint.d.ts +42 -4
  128. package/dist/src/orgrt/checkpoint.d.ts.map +1 -1
  129. package/dist/src/orgrt/checkpoint.js +75 -4
  130. package/dist/src/orgrt/checkpoint.js.map +1 -1
  131. package/dist/src/orgrt/codex-runner.d.ts +1 -1
  132. package/dist/src/orgrt/codex-runner.d.ts.map +1 -1
  133. package/dist/src/orgrt/codex-runner.js +50 -23
  134. package/dist/src/orgrt/codex-runner.js.map +1 -1
  135. package/dist/src/orgrt/copilot-runner.d.ts +24 -2
  136. package/dist/src/orgrt/copilot-runner.d.ts.map +1 -1
  137. package/dist/src/orgrt/copilot-runner.js +257 -157
  138. package/dist/src/orgrt/copilot-runner.js.map +1 -1
  139. package/dist/src/orgrt/cross-org.d.ts +10 -3
  140. package/dist/src/orgrt/cross-org.d.ts.map +1 -1
  141. package/dist/src/orgrt/cross-org.js +225 -10
  142. package/dist/src/orgrt/cross-org.js.map +1 -1
  143. package/dist/src/orgrt/crush-runner.d.ts +22 -2
  144. package/dist/src/orgrt/crush-runner.d.ts.map +1 -1
  145. package/dist/src/orgrt/crush-runner.js +227 -120
  146. package/dist/src/orgrt/crush-runner.js.map +1 -1
  147. package/dist/src/orgrt/daemon.d.ts +77 -7
  148. package/dist/src/orgrt/daemon.d.ts.map +1 -1
  149. package/dist/src/orgrt/daemon.js +989 -385
  150. package/dist/src/orgrt/daemon.js.map +1 -1
  151. package/dist/src/orgrt/decisions.d.ts +2 -2
  152. package/dist/src/orgrt/decisions.d.ts.map +1 -1
  153. package/dist/src/orgrt/decisions.js +90 -18
  154. package/dist/src/orgrt/decisions.js.map +1 -1
  155. package/dist/src/orgrt/grok-runner.d.ts +29 -3
  156. package/dist/src/orgrt/grok-runner.d.ts.map +1 -1
  157. package/dist/src/orgrt/grok-runner.js +286 -150
  158. package/dist/src/orgrt/grok-runner.js.map +1 -1
  159. package/dist/src/orgrt/kimicode-runner.d.ts +5 -5
  160. package/dist/src/orgrt/kimicode-runner.d.ts.map +1 -1
  161. package/dist/src/orgrt/kimicode-runner.js +53 -32
  162. package/dist/src/orgrt/kimicode-runner.js.map +1 -1
  163. package/dist/src/orgrt/mailbox.d.ts +15 -0
  164. package/dist/src/orgrt/mailbox.d.ts.map +1 -1
  165. package/dist/src/orgrt/mailbox.js +29 -1
  166. package/dist/src/orgrt/mailbox.js.map +1 -1
  167. package/dist/src/orgrt/migrate.d.ts.map +1 -1
  168. package/dist/src/orgrt/migrate.js +8 -5
  169. package/dist/src/orgrt/migrate.js.map +1 -1
  170. package/dist/src/orgrt/opencode-runner.d.ts +1 -1
  171. package/dist/src/orgrt/opencode-runner.d.ts.map +1 -1
  172. package/dist/src/orgrt/opencode-runner.js +29 -3
  173. package/dist/src/orgrt/opencode-runner.js.map +1 -1
  174. package/dist/src/orgrt/org-memory.d.ts +19 -3
  175. package/dist/src/orgrt/org-memory.d.ts.map +1 -1
  176. package/dist/src/orgrt/org-memory.js +110 -38
  177. package/dist/src/orgrt/org-memory.js.map +1 -1
  178. package/dist/src/orgrt/pi-rpc-runner.d.ts +3 -1
  179. package/dist/src/orgrt/pi-rpc-runner.d.ts.map +1 -1
  180. package/dist/src/orgrt/pi-rpc-runner.js +35 -3
  181. package/dist/src/orgrt/pi-rpc-runner.js.map +1 -1
  182. package/dist/src/orgrt/pi-runner.d.ts +25 -2
  183. package/dist/src/orgrt/pi-runner.d.ts.map +1 -1
  184. package/dist/src/orgrt/pi-runner.js +271 -152
  185. package/dist/src/orgrt/pi-runner.js.map +1 -1
  186. package/dist/src/orgrt/policy.d.ts +1 -0
  187. package/dist/src/orgrt/policy.d.ts.map +1 -1
  188. package/dist/src/orgrt/policy.js +199 -39
  189. package/dist/src/orgrt/policy.js.map +1 -1
  190. package/dist/src/orgrt/provider.d.ts +4 -0
  191. package/dist/src/orgrt/provider.d.ts.map +1 -1
  192. package/dist/src/orgrt/provider.js +16 -0
  193. package/dist/src/orgrt/provider.js.map +1 -1
  194. package/dist/src/orgrt/qwen-rpc-runner.d.ts +3 -1
  195. package/dist/src/orgrt/qwen-rpc-runner.d.ts.map +1 -1
  196. package/dist/src/orgrt/qwen-rpc-runner.js +35 -3
  197. package/dist/src/orgrt/qwen-rpc-runner.js.map +1 -1
  198. package/dist/src/orgrt/qwen-runner.d.ts +30 -3
  199. package/dist/src/orgrt/qwen-runner.d.ts.map +1 -1
  200. package/dist/src/orgrt/qwen-runner.js +282 -142
  201. package/dist/src/orgrt/qwen-runner.js.map +1 -1
  202. package/dist/src/orgrt/role-slot.d.ts +88 -0
  203. package/dist/src/orgrt/role-slot.d.ts.map +1 -0
  204. package/dist/src/orgrt/role-slot.js +133 -0
  205. package/dist/src/orgrt/role-slot.js.map +1 -0
  206. package/dist/src/orgrt/runtime-options.d.ts +17 -0
  207. package/dist/src/orgrt/runtime-options.d.ts.map +1 -0
  208. package/dist/src/orgrt/runtime-options.js +32 -0
  209. package/dist/src/orgrt/runtime-options.js.map +1 -0
  210. package/dist/src/orgrt/scheduler-integration.d.ts +11 -0
  211. package/dist/src/orgrt/scheduler-integration.d.ts.map +1 -1
  212. package/dist/src/orgrt/scheduler-integration.js +75 -17
  213. package/dist/src/orgrt/scheduler-integration.js.map +1 -1
  214. package/dist/src/orgrt/scheduler.d.ts +4 -0
  215. package/dist/src/orgrt/scheduler.d.ts.map +1 -1
  216. package/dist/src/orgrt/scheduler.js +7 -1
  217. package/dist/src/orgrt/scheduler.js.map +1 -1
  218. package/dist/src/orgrt/server.d.ts +8 -3
  219. package/dist/src/orgrt/server.d.ts.map +1 -1
  220. package/dist/src/orgrt/server.js +48 -13
  221. package/dist/src/orgrt/server.js.map +1 -1
  222. package/dist/src/orgrt/session.d.ts +26 -2
  223. package/dist/src/orgrt/session.d.ts.map +1 -1
  224. package/dist/src/orgrt/session.js +98 -4
  225. package/dist/src/orgrt/session.js.map +1 -1
  226. package/dist/src/orgrt/task-dag.d.ts +5 -0
  227. package/dist/src/orgrt/task-dag.d.ts.map +1 -1
  228. package/dist/src/orgrt/task-dag.js +41 -0
  229. package/dist/src/orgrt/task-dag.js.map +1 -1
  230. package/dist/src/orgrt/test-loop.js +2 -2
  231. package/dist/src/orgrt/test-loop.js.map +1 -1
  232. package/dist/src/orgrt/types.d.ts +14 -2
  233. package/dist/src/orgrt/types.d.ts.map +1 -1
  234. package/dist/src/orgrt/types.js +15 -1
  235. package/dist/src/orgrt/types.js.map +1 -1
  236. package/dist/src/orgrt/vercel-runner.d.ts.map +1 -1
  237. package/dist/src/orgrt/vercel-runner.js +4 -0
  238. package/dist/src/orgrt/vercel-runner.js.map +1 -1
  239. package/dist/src/ui/routes-org.mjs +27 -7
  240. package/dist/src/ui/server.mjs +82 -48
  241. package/dist/tsconfig.tsbuildinfo +1 -1
  242. package/package.json +3 -2
  243. package/dist/src/ui/data/mastermind-sessions.json +0 -1
@@ -1,6 +1,7 @@
1
1
  // packages/@monomind/cli/src/orgrt/daemon.ts
2
2
  // monolean: single-process inter-org — upgrade path = daemon-to-daemon HTTP when multi-host is real
3
3
  import { execFileSync } from 'node:child_process';
4
+ import { randomUUID } from 'node:crypto';
4
5
  import { existsSync, mkdirSync, readFileSync, unlinkSync } from 'node:fs';
5
6
  import { isAbsolute, join } from 'node:path';
6
7
  import { writeJsonFileAtomic } from '../utils/json-file.js';
@@ -10,7 +11,7 @@ import { AntigravityAgentRunner } from './antigravity-runner.js';
10
11
  import * as approvalOps from './approvals.js';
11
12
  import { BrokerLease, normalizeCredential } from './broker.js';
12
13
  import { OrgBus } from './bus.js';
13
- import { captureCheckpoint, generateChecksum, isCheckpointExpired, restoreMailboxQueue, validateCheckpoint, } from './checkpoint.js';
14
+ import { captureCheckpoint, generateChecksum, isCheckpointExpired, migrateCheckpoint, restoreMailboxQueue, validateCheckpoint, } from './checkpoint.js';
14
15
  import * as checkpointOps from './checkpoint-ops.js';
15
16
  import { CodexAgentRunner } from './codex-runner.js';
16
17
  import { CopilotAgentRunner } from './copilot-runner.js';
@@ -20,7 +21,7 @@ import * as decisionOps from './decisions.js';
20
21
  import { createFenceForRole, loadGlobalFenceConfig, mergeFenceConfigs, } from './fence.js';
21
22
  import { attachForwarder } from './forwarder.js';
22
23
  import { GrokAgentRunner } from './grok-runner.js';
23
- import { drainInbox } from './inbox.js';
24
+ import { drainInbox, queueMessage } from './inbox.js';
24
25
  import { KimiCodeAgentRunner } from './kimicode-runner.js';
25
26
  import { isRecoverableCloseReason, Mailbox } from './mailbox.js';
26
27
  import { OpencodeAgentRunner } from './opencode-runner.js';
@@ -28,9 +29,12 @@ import * as orgMemory from './org-memory.js';
28
29
  import { PiRpcAgentRunner } from './pi-rpc-runner.js';
29
30
  import { PiAgentRunner } from './pi-runner.js';
30
31
  import { PolicyEngine } from './policy.js';
32
+ import { resolveRoleProvider } from './provider.js';
31
33
  import * as questionOps from './questions.js';
32
34
  import { QwenRpcAgentRunner } from './qwen-rpc-runner.js';
33
35
  import { QwenAgentRunner } from './qwen-runner.js';
36
+ import { buildRespawnReceipt, computeReplacementBudget, mergeEffectiveRoleConfig, redactRoleConfig, validateRespawnInput, } from './role-slot.js';
37
+ import { buildRuntimeOptions } from './runtime-options.js';
34
38
  import { historyFile, readHistory, readRunEvents, summarizeRun, } from './reporting.js';
35
39
  import * as scheduler from './scheduler-integration.js';
36
40
  import { runAgentSession } from './session.js';
@@ -183,6 +187,17 @@ export class ScrollbackBuffer {
183
187
  this.lines.length = 0;
184
188
  }
185
189
  }
190
+ /** Bug 4: number of roles for this org that are actually spawned and running
191
+ * right now — the live count run_config.max_concurrent_agents caps. A role
192
+ * that crashed or ended no longer counts, so a slot frees up automatically
193
+ * the moment that happens; nothing needs to explicitly decrement a counter. */
194
+ export function activeRoleCount(org) {
195
+ let n = 0;
196
+ for (const rt of org.agents.values())
197
+ if (rt.status === 'running')
198
+ n++;
199
+ return n;
200
+ }
186
201
  export class OrgDaemon {
187
202
  root;
188
203
  opts;
@@ -193,6 +208,12 @@ export class OrgDaemon {
193
208
  /** @internal */ forwarders = new Map();
194
209
  /** @internal */ watchdogs = new Map();
195
210
  /** @internal */ stopping = new Map();
211
+ /** @internal Bug 2 (TOCTOU race): names currently reserved by an in-flight
212
+ * startOrg() call, from the synchronous existence check through
213
+ * registration in `orgs`. Closes the window where two concurrent
214
+ * startOrg(name) calls could both pass the `orgs.has(name)` check before
215
+ * either registered and spawn duplicate runs. */
216
+ /** @internal */ startingOrgs = new Set();
196
217
  /** @internal */ approvals = new Map();
197
218
  /** @internal */ approvalLocks = new Map();
198
219
  /** @internal */ gatesLocks = new Map();
@@ -215,10 +236,10 @@ export class OrgDaemon {
215
236
  this.opts = opts;
216
237
  }
217
238
  /** Publish this daemon's inbox so orgs started AFTER this call register with the broker. */
218
- setInboxUrl(url, credential) {
239
+ setInboxUrl(url, operatorCredential) {
219
240
  this.opts.inboxUrl = url;
220
- if (credential !== undefined)
221
- this.opts.inboxCredential = credential;
241
+ if (operatorCredential !== undefined)
242
+ this.opts.operatorCredential = operatorCredential;
222
243
  }
223
244
  /** subscribe to events from ALL running orgs (dashboard server uses this) */
224
245
  subscribe(fn) {
@@ -333,8 +354,48 @@ export class OrgDaemon {
333
354
  const inflightStop = this.stopping.get(name);
334
355
  if (inflightStop)
335
356
  await inflightStop;
357
+ // Bug 2 (TOCTOU race): this existence check is synchronous, but the real
358
+ // registration into `this.orgs` doesn't happen until deep inside
359
+ // startOrgInner, after several genuine `await` points (the provider
360
+ // validation dynamic import, `git worktree add` for workspace:
361
+ // 'worktree'). Two concurrent startOrg(name) calls — e.g. the
362
+ // scheduler's tick, the runfile poll loop, and autoWake firing close
363
+ // together — could each pass this check before either registered,
364
+ // spawning two duplicate runs with separate budget/policy counters that
365
+ // both write the same shared per-org files. Reserve the name in
366
+ // `startingOrgs` synchronously, in the same tick as the check, so a
367
+ // second concurrent call sees the reservation and is rejected instead of
368
+ // racing ahead to spawn a duplicate.
336
369
  if (this.orgs.has(name))
337
370
  throw new Error(`org ${name} already running`);
371
+ if (this.startingOrgs.has(name))
372
+ throw new Error(`org ${name} already starting`);
373
+ this.startingOrgs.add(name);
374
+ try {
375
+ return await this.startOrgInner(name, taskOverride, options);
376
+ }
377
+ catch (err) {
378
+ // startOrgInner registers the org in `this.orgs` (and spawns the boss,
379
+ // installs the exit listener, starts the broker lease) well before it
380
+ // returns; persistState (ENOSPC/EACCES) and BrokerLease.start() can
381
+ // still throw after that. Left alone, that was a live, unreachable org:
382
+ // sessions running, `this.orgs` still holding it, every later startOrg
383
+ // rejected with "already running", and nothing ever calling stopOrg.
384
+ // Only this call can have registered the name (the reservation above
385
+ // holds until `finally`), so anything in the map is ours to tear down.
386
+ if (this.orgs.has(name)) {
387
+ await this.stopOrg(name).catch((stopErr) => console.error(`org ${name}: teardown after failed start failed:`, stopErr instanceof Error ? stopErr.message : stopErr));
388
+ }
389
+ throw err;
390
+ }
391
+ finally {
392
+ this.startingOrgs.delete(name);
393
+ }
394
+ }
395
+ /** The actual startOrg implementation. Split out of startOrg() so the
396
+ * reservation guard above runs synchronously, before any `await` in here —
397
+ * see the bug 2 comment in startOrg(). */
398
+ async startOrgInner(name, taskOverride, options) {
338
399
  const defPath = join(this.root, ORG_DIR, `${name}.json`);
339
400
  const def = OrgDefSchema.parse(JSON.parse(readFileSync(defPath, 'utf8')));
340
401
  let run;
@@ -348,8 +409,13 @@ export class OrgDaemon {
348
409
  throw new Error(`cannot resume org "${name}": no valid checkpoint found`);
349
410
  if (isCheckpointExpired(rt.checkpoint))
350
411
  throw new Error(`cannot resume org "${name}": checkpoint expired`);
351
- if (!validateCheckpoint(rt.checkpoint))
412
+ // Migrate an older-schema checkpoint (verifying ITS OWN stored checksum
413
+ // first) before validating it against CHECKPOINT_VERSION — see
414
+ // migrateCheckpoint's doc comment in checkpoint.ts.
415
+ const migrated = migrateCheckpoint(rt.checkpoint);
416
+ if (!migrated || !validateCheckpoint(migrated))
352
417
  throw new Error(`cannot resume org "${name}": checkpoint validation failed`);
418
+ rt.checkpoint = migrated;
353
419
  run = rt.run;
354
420
  checkpoint = rt.checkpoint;
355
421
  if (rt.abandonedRoles) {
@@ -476,6 +542,10 @@ export class OrgDaemon {
476
542
  // boss that's genuinely out of ideas can loop forever making zero
477
543
  // progress without ever tripping the watchdog.
478
544
  let lastToolActivity = 0;
545
+ // Org-wide budget ceiling (bug 1): tracks whether run_config.budget_tokens
546
+ // has already been enforced this run, so the close-all-mailboxes sweep
547
+ // below only fires once instead of on every subsequent usage event.
548
+ let orgBudgetClosed = false;
479
549
  bus.subscribe((e) => {
480
550
  const slim = e.data?.content != null ? { ...e, data: { ...e.data, content: undefined } } : e;
481
551
  collected.push(slim);
@@ -512,6 +582,42 @@ export class OrgDaemon {
512
582
  }
513
583
  }
514
584
  }
585
+ // Bug 1: run_config.budget_tokens is an org-wide ceiling, not just a
586
+ // per-role one — a role's explicit budget_tokens override lets IT spend
587
+ // more without raising what every other role can spend, so nothing
588
+ // upstream of this ever summed real usage across the whole roster and
589
+ // stopped the org when the declared total was reached. Mirror the
590
+ // per-role budget-exhausted handling (session.ts's mailbox.close('token-budget'))
591
+ // at the org level: once the sum of every role's PolicyEngine.usage
592
+ // reaches the ceiling, close every mailbox and stop lazy-spawning new
593
+ // ones so the org can't keep spending past its declared cap.
594
+ if (e.type === 'usage' && !orgBudgetClosed) {
595
+ const orgBudget = def.run_config.budget_tokens;
596
+ if (orgBudget != null) {
597
+ let orgUsage = 0;
598
+ for (const rt of running.agents.values())
599
+ orgUsage += rt.policy.usage;
600
+ // Mid-run role replacement retires a policy engine's usage into the
601
+ // slot instead of discarding it (see role-slot.ts / respawnRole) -
602
+ // include it here or a replacement could silently reset spend and
603
+ // let the org exceed its declared ceiling.
604
+ for (const slot of running.roleSlots.values())
605
+ orgUsage += slot.retiredUsage.tokens;
606
+ if (orgUsage >= orgBudget) {
607
+ orgBudgetClosed = true;
608
+ running.pendingRoles?.clear(); // prevent lazy spawns after the org budget is exhausted
609
+ for (const rt of running.agents.values()) {
610
+ if (!rt.mailbox.isClosed)
611
+ rt.mailbox.close('token-budget');
612
+ }
613
+ bus.emit({
614
+ type: 'status',
615
+ reason: 'org-budget-exhausted',
616
+ msg: `org-wide token budget exhausted (${orgUsage}/${orgBudget}) — closing all roles`,
617
+ });
618
+ }
619
+ }
620
+ }
515
621
  // Track last message ID for threading responses
516
622
  if ((e.type === 'message' || e.type === 'xorg') && e.from) {
517
623
  const runtime = running.agents.get(e.from);
@@ -537,8 +643,13 @@ export class OrgDaemon {
537
643
  run,
538
644
  bus,
539
645
  agents: new Map(),
646
+ roleSlots: new Map(),
647
+ bossRoleId: '', // set below, once bossRole is computed
648
+ glossary: [],
649
+ respawning: new Set(),
540
650
  busEvents: () => [...collected],
541
651
  workdir: cwd,
652
+ credential: randomUUID(),
542
653
  };
543
654
  this.orgs.set(name, running);
544
655
  // ── MonoFence guardrail: pre-create per-role instances ────────────────
@@ -569,402 +680,72 @@ export class OrgDaemon {
569
680
  if (roleFences.size > 0)
570
681
  running.fences = roleFences;
571
682
  // Even-split budget; a role's own budget_tokens overrides it (roleTokenBudget).
572
- const perRoleBudget = Math.floor((def.run_config.budget_tokens ?? 1_000_000) / def.roles.length);
683
+ // Bug 1: roles WITH an explicit override spend on top of the even split
684
+ // rather than out of it, so the roster's ceilings could sum to well over
685
+ // the declared org-wide budget (e.g. 4 roles @ 250k + one role overridden
686
+ // to 2M = 2.75M achievable against a declared 1M cap). Subtract the sum of
687
+ // every role's explicit override from the org-wide budget first, then
688
+ // split only the remainder among the roles WITHOUT an override, so the
689
+ // static split is honest about what's left. (Live usage is still tracked
690
+ // and enforced as a real ceiling above, independent of this static split.)
691
+ const orgBudgetTokens = def.run_config.budget_tokens ?? 1_000_000;
692
+ const overriddenTokenSum = def.roles.reduce((sum, r) => sum + (r.budget_tokens ?? 0), 0);
693
+ const unoverriddenRoleCount = def.roles.filter((r) => r.budget_tokens == null).length;
694
+ const perRoleBudget = unoverriddenRoleCount > 0
695
+ ? Math.max(0, Math.floor((orgBudgetTokens - overriddenTokenSum) / unoverriddenRoleCount))
696
+ : 0;
573
697
  // Single boss-selection rule for kickoff AND org_complete gating — the
574
698
  // session layer previously keyed the tool on reports_to===null while the
575
699
  // kickoff went to (type==='boss' || reports_to===null || roles[0]), so a
576
700
  // fallback-selected boss could be told to call org_complete without having
577
701
  // the tool.
578
702
  const bossRole = def.roles.find((r) => r.type === 'boss' || r.reports_to === null) ?? def.roles[0];
579
- // Canonical entity names from the org KG — injected into the coordinator
703
+ running.bossRoleId = bossRole.id;
704
+ // Canonical entity names from THIS org's KG — injected into the coordinator
580
705
  // prompt so org_learn extractions reuse them instead of minting duplicates.
706
+ // Scoped: an unscoped glossary handed every org's entity names to every
707
+ // coordinator, which is how one org's claims got merged into another's.
581
708
  const glossary = await (async () => {
582
709
  try {
583
710
  if (!(await this.orgMemoryUsable()))
584
711
  return [];
585
712
  const kg = await import('../memory/memory-kg.js');
586
- return await kg.kgGlossary({ dbPath: this.orgMemoryDbPath() });
713
+ return await kg.kgGlossary({
714
+ dbPath: this.orgMemoryDbPath(),
715
+ scope: orgMemory.orgKgScope(name),
716
+ });
587
717
  }
588
718
  catch {
589
719
  return [];
590
720
  }
591
721
  })();
722
+ running.glossary = glossary;
592
723
  // Resource-gated staggered spawn: check memory/process limits before each
593
724
  // NON-BOSS agent, wait if under pressure. The boss always spawns immediately
594
725
  // and ungated — the org has no coordinator at all without it, so gating it
595
726
  // behind host memory pressure would make the whole org fail to start over a
596
727
  // condition workers are specifically designed to ride out.
597
- const _limits = getResourceLimits();
598
728
  // Extracted so a role that fails its gate check can be spawned later by
599
729
  // scheduleDeferredSpawn() once resources free up, without re-running the
600
730
  // gate logic or duplicating the session-wiring below.
601
731
  const spawnRole = (role, roleCheckpoint) => {
602
732
  if (running.agents.has(role.id))
603
733
  return;
604
- let roleCwd = cwd;
605
- if (ws === 'worktree-per-role' && role.id !== bossRole.id) {
606
- const wtPath = join(this.root, ORG_DIR, name, `worktree-${role.id}`);
607
- try {
608
- // Q7: top-level `import { execFileSync }` replaces the inlined
609
- // `require('node:child_process')` that broke ESM at runtime —
610
- // vitest's CJS shim masked it in tests but the built package
611
- // threw "require is not defined" in real Node ESM execution.
612
- // SEC-5: argv-array form, no shell.
613
- if (existsSync(wtPath)) {
614
- try {
615
- execFileSync('git', ['worktree', 'remove', '--force', wtPath], {
616
- cwd: this.root,
617
- stdio: 'ignore',
618
- timeout: 30_000,
619
- });
620
- }
621
- catch {
622
- /* best-effort */
623
- }
624
- }
625
- execFileSync('git', ['worktree', 'add', wtPath, 'HEAD', '--detach'], {
626
- cwd: this.root,
627
- stdio: 'ignore',
628
- timeout: 30_000,
629
- });
630
- roleCwd = wtPath;
631
- }
632
- catch {
633
- /* fallback to shared cwd if git worktree fails */
634
- }
635
- }
636
- const mailbox = new Mailbox();
637
- if (roleCheckpoint?.mailboxQueue?.length) {
638
- restoreMailboxQueue({ mailbox }, roleCheckpoint.mailboxQueue);
639
- }
640
- // A recoverable close (budget exhaustion) is left open on resume — see
641
- // isRecoverableCloseReason's doc comment. Re-closing it here would
642
- // make the idle watchdog's "raise the budget and resume" remedy a
643
- // no-op, since nothing in this codebase ever reopens a closed mailbox.
644
- if (roleCheckpoint?.mailboxClosed &&
645
- !isRecoverableCloseReason(roleCheckpoint.mailboxCloseReason)) {
646
- mailbox.close(roleCheckpoint.mailboxCloseReason);
647
- }
648
- const policy = new PolicyEngine(role.id, {
649
- maxTokens: role.budget_tokens ?? perRoleBudget,
650
- maxUsd: role.budget_usd,
651
- ...(role.policy ?? {}),
652
- }, bus, roleCwd);
653
- if (roleCheckpoint?.tokensUsed) {
654
- policy.setUsage(roleCheckpoint.tokensUsed);
655
- }
656
- // ORG-7: restore accumulated USD spend across resume so a stop/resume
657
- // cycle can't reset a role's USD budget back to zero.
658
- if (roleCheckpoint?.costUsd) {
659
- policy.setUsageUsd(roleCheckpoint.costUsd);
660
- }
661
- const runtime = {
662
- mailbox,
663
- policy,
664
- status: roleCheckpoint?.status ?? 'running',
665
- done: Promise.resolve(),
666
- metrics: { tokens: roleCheckpoint?.tokensUsed ?? 0, costUsd: roleCheckpoint?.costUsd ?? 0 },
667
- lastMessageId: roleCheckpoint?.lastMessageId,
668
- error: roleCheckpoint?.error,
669
- sessionId: roleCheckpoint?.sessionId,
670
- worktreePath: roleCwd !== cwd ? roleCwd : undefined,
671
- scrollback: new ScrollbackBuffer(),
672
- };
673
- if (roleCheckpoint?.scrollback?.length) {
674
- for (const line of roleCheckpoint.scrollback)
675
- runtime.scrollback.push(line);
676
- }
677
- const sessionOpts = {
678
- org: name,
679
- role,
680
- bus,
681
- policy,
682
- mailbox,
683
- cwd: roleCwd,
684
- def,
685
- // Pass the org state directory so runners that persist per-role state
686
- // (VercelAgentRunner session files) write under .monomind/orgs/<name>
687
- // instead of polluting the workspace cwd.
688
- orgDir: join(this.root, ORG_DIR, name),
689
- // Project root for named-provider (`adapter_config.provider`) config
690
- // lookup — role cwd may be an isolated workspace with no config file.
691
- orgRoot: this.root,
692
- maxTurns: role.max_turns_per_message ?? def.run_config.max_turns_per_message,
693
- resumeSessionId: roleCheckpoint?.sessionId,
694
- lastMessageId: () => runtime.lastMessageId,
695
- onOutput: (line) => runtime.scrollback.push(line),
696
- onSessionId: (id) => {
697
- runtime.sessionId = id;
698
- },
699
- deliver: (from, to, subject, body) => this.deliver(name, from, to, subject, body),
700
- askHuman: (r, question) => this.askHuman(name, r, question),
701
- onGate: (r, gateName, gateDesc) => this.createGate(name, r, gateName, gateDesc),
702
- circuitBreaker: (() => {
703
- const cb = def.run_config.circuit_breaker;
704
- if (!cb)
705
- return undefined;
706
- return { threshold: cb.failure_threshold ?? 5, state: { failures: 0, tripped: false } };
707
- })(),
708
- beforeTool: (r, toolName) => this.checkApproval(name, r, toolName),
709
- fence: roleFences.get(role.id),
710
- // ORG-1: gatedCanUseTool denials are a natural decision point — record them so
711
- // `org decisions` shows real traces instead of always reporting none.
712
- onDecision: (r, toolName, message) => {
713
- this.recordDecision(name, r, {
714
- type: 'tool',
715
- context: `tool call: ${toolName}`,
716
- reasoning: message,
717
- outcome: 'denied',
718
- });
719
- },
720
- // ORG-9: decision gates are documented as "hard-blocking" — make that
721
- // true by actually denying tool use while this role has a pending gate,
722
- // the same way pending approvals already do.
723
- hasPendingGate: () => this.listGates(name, 'pending').some((g) => g.roleId === role.id),
724
- onComplete: role.id === bossRole.id
725
- ? (r, outcome, summary) => {
726
- bus.emit({
727
- type: 'status',
728
- from: r,
729
- reason: 'org-complete',
730
- msg: `run outcome: ${outcome}`,
731
- data: { outcome, summary },
732
- });
733
- }
734
- : undefined,
735
- // #11: a boss that overflows its context window isn't a crash (it keeps
736
- // returning +0-token errors forever), so without this the idle watchdog
737
- // just nudges it for ~30 min before idle-stopping. Restart the whole org
738
- // with fresh sessions instead — bounded by MAX_BOSS_RESTARTS.
739
- onContextLimit: role.id === bossRole.id ? () => this.scheduleBossRestart(name) : undefined,
740
- recall: async (r, q) => {
741
- const answer = await this.recallOrgMemory(name, def, q, r);
742
- bus.emit({
743
- type: 'status',
744
- from: r,
745
- reason: 'org-recall',
746
- msg: `recall: ${q.slice(0, 80)}`,
747
- data: { hits: answer.hits },
748
- });
749
- return answer.text;
750
- },
751
- searchKnowledge: async (r, q) => {
752
- const answer = await this.searchProjectKnowledge(q);
753
- bus.emit({
754
- type: 'status',
755
- from: r,
756
- reason: 'knowledge-search',
757
- msg: `knowledge: ${q.slice(0, 80)}`,
758
- data: { hits: answer.hits },
759
- });
760
- return answer.text;
761
- },
762
- glossary,
763
- remember: async (r, content, scope) => {
764
- const text = await this.rememberOrgMemory(name, def, r, content, scope, run);
765
- bus.emit({
766
- type: 'status',
767
- from: r,
768
- reason: 'org-remember',
769
- msg: `remember (${scope}): ${content.slice(0, 80)}`,
770
- data: { scope },
771
- });
772
- return text;
773
- },
774
- learn: async (r, payload) => {
775
- const text = await this.learnOrgKnowledge(name, run, payload);
776
- bus.emit({
777
- type: 'status',
778
- from: r,
779
- reason: 'org-learn',
780
- msg: `learn: ${text.slice(0, 120)}`,
781
- data: {
782
- nodes: payload.nodes?.length ?? 0,
783
- edges: payload.edges?.length ?? 0,
784
- rules: payload.rules?.length ?? 0,
785
- },
786
- });
787
- return text;
788
- },
789
- createTask: (r, title, assignee, deps) => {
790
- return this.dagCreateTask(name, r, title, assignee, deps);
791
- },
792
- completeTask: (r, taskId, result) => {
793
- return this.dagCompleteTask(name, r, taskId, result);
794
- },
795
- listTasks: () => {
796
- const running = this.orgs.get(name);
797
- return JSON.stringify(running?.taskDag?.all() ?? [], null, 2);
798
- },
799
- splitTask: (r, parentId, children) => {
800
- return this.dagSplitTask(name, r, parentId, children);
801
- },
802
- mergeTask: (r, sourceId, targetId) => {
803
- return this.dagMergeTask(name, r, sourceId, targetId);
804
- },
805
- cancelTask: (r, taskId, reason) => {
806
- return this.dagCancelTask(name, r, taskId, reason);
807
- },
808
- blockTask: (r, taskId, untilIso, reason) => {
809
- return this.dagBlockTask(name, r, taskId, untilIso, reason);
810
- },
811
- planGraph: (r, specs) => {
812
- return this.dagPlanGraph(name, r, specs);
813
- },
814
- queryFn: this.opts.queryFn,
815
- // Runner resolution: explicit opts.runner > role `runtime` field >
816
- // org def `runtime` field > MONOMIND_RUNTIME env (opencode/kimicode) >
817
- // undefined (session.ts falls back to ClaudeAgentRunner via queryFn).
818
- // Leaving it undefined for the default path is what keeps
819
- // Claude/Antigravity orgs byte-for-byte unchanged. Session opts are
820
- // built per role here in spawnRole, so each role gets its own runner.
821
- runner: this.opts.runner ??
822
- resolveRoleRunner(role.runtime, def.runtime, role.provider?.kind, undefined, role.provider),
823
- };
824
- // Supervised session: transient crashes (provider blips, network) restart
825
- // with backoff; a crash with the mailbox already closed, or one that
826
- // exhausts the retry budget, is terminal. runAgentSession already emits a
827
- // 'status' event for the raw error; the terminal 'audit' event is for
828
- // dashboards/alerts that filter on actionable failures (not routine
829
- // status chatter) so a dead agent surfaces instead of a run that
830
- // silently never progresses.
831
- const BACKOFFS_MS = this.opts.crashBackoffsMs ?? [1000, 5000, 15000];
832
- if (!mailbox.isClosed && runtime.status !== 'crashed') {
833
- runtime.done = (async () => {
834
- for (let attempt = 0;; attempt++) {
835
- try {
836
- await runAgentSession(sessionOpts);
837
- runtime.status = 'ended';
838
- return;
839
- }
840
- catch (err) {
841
- // Drop the crashed session's stale waker immediately: a push()
842
- // during the backoff window must queue for the NEXT session, not
843
- // wake the dead generator to swallow it.
844
- mailbox.detach();
845
- // #203: if the crashed session's mailbox generator was abandoned
846
- // mid-yield (message already shift()ed for it, turn never
847
- // finished), put that message back on the queue — otherwise the
848
- // replacement session's stream() finds an empty queue and parks
849
- // forever, since the "delivered" message is gone for good.
850
- mailbox.reclaimInFlight();
851
- const message = err instanceof Error ? err.message : String(err);
852
- const isTurnLimit = /Reached maximum number of turns|error_max_turns/i.test(message);
853
- // Bounded like every other recovery: attempt counts every pass
854
- // through this loop, so a role that keeps surfacing max-turns
855
- // errors here (session.ts already swallows the normal ones)
856
- // falls through to crash handling instead of looping forever.
857
- if (isTurnLimit && !mailbox.isClosed && attempt < BACKOFFS_MS.length) {
858
- sessionOpts.resumeSessionId = undefined;
859
- mailbox.push(`${Mailbox.CONTINUE_PREFIX} You reached the turn limit on your task. Continue your in-progress work from where you left off; if finished, end your turn.`);
860
- bus.emit({
861
- type: 'status',
862
- from: role.id,
863
- reason: 'turn-limit-recover',
864
- msg: `agent "${role.id}" hit turn limit error — continuing with fresh session`,
865
- });
866
- continue;
867
- }
868
- // Exit 143 = SIGTERM. If the mailbox is already closed, we
869
- // sent the signal ourselves during stop — not a crash.
870
- const killedByStop = mailbox.isClosed && /exit(?:ed)? with code 143/.test(message);
871
- const crash = () => {
872
- if (killedByStop) {
873
- runtime.status = 'ended';
874
- bus.emit({
875
- type: 'status',
876
- from: role.id,
877
- msg: `agent "${role.id}" terminated by stop (was still working when drain window expired)`,
878
- reason: 'terminated-by-stop',
879
- });
880
- return;
881
- }
882
- runtime.status = 'crashed';
883
- runtime.error = message;
884
- // Close the mailbox so deliver()/receiveRemote() report a real
885
- // error instead of pushing into a queue no session will read
886
- // (and returning a false "delivered" receipt to the sender).
887
- mailbox.close();
888
- const isContextLimit = OrgDaemon.CONTEXT_LIMIT_RE.test(message);
889
- bus.emit({
890
- type: 'audit',
891
- from: role.id,
892
- msg: `agent "${role.id}" crashed: ${message}`,
893
- reason: isContextLimit ? 'agent-context-limit' : 'agent-session-crash',
894
- data: {
895
- agentId: role.id,
896
- error: message,
897
- restarts: attempt,
898
- contextLimit: isContextLimit,
899
- },
900
- });
901
- if (role.id !== bossRole.id) {
902
- // #2/#3: a worker is gone for the rest of this run. Without this
903
- // notice the coordinator keeps messaging a corpse (observed: four
904
- // unanswered org_send calls to a developer that had crashed on a
905
- // context-window limit). Tell the boss to reassign — and if the
906
- // crash was a context overflow, tell it to chunk smaller, since
907
- // re-dispatching the same task verbatim fails the same way.
908
- const bossRt = running.agents.get(bossRole.id);
909
- if (bossRt && !bossRt.mailbox.isClosed) {
910
- const guidance = isContextLimit
911
- ? ' This was a context-window overflow — re-dispatching the same task verbatim will fail identically. Break the work into smaller pieces (one file or section at a time) and do not paste large file contents in a single message.'
912
- : '';
913
- bossRt.mailbox.push(`[system] Worker "${role.id}" crashed and will not recover this run (${message}). It can no longer receive messages — stop messaging it. Reassign its outstanding work to another agent or take it on yourself.${guidance}`);
914
- bus.emit({
915
- type: 'audit',
916
- from: bossRole.id,
917
- reason: 'worker-crashed',
918
- msg: `worker "${role.id}" crashed (contextLimit=${isContextLimit}); coordinator notified to reassign`,
919
- });
920
- }
921
- }
922
- else {
923
- // #4: the coordinator itself died. Don't go silent and wait for a
924
- // human — attempt a bounded whole-org restart with fresh sessions
925
- // (which also sheds whatever bloated context caused the crash).
926
- this.scheduleBossRestart(name);
927
- }
928
- };
929
- // Fatal errors (provider auth/quota/billing — tagged with
930
- // err.fatal by the runner) can NEVER be fixed by a restart: the
931
- // same call fails identically or hangs. Skip the backoff loop
932
- // and go straight to terminal crash handling instead of burning
933
- // the retry budget and wall-clock on a guaranteed failure.
934
- const fatal = err?.fatal === true;
935
- if (fatal) {
936
- bus.emit({
937
- type: 'status',
938
- from: role.id,
939
- reason: 'agent-fatal',
940
- msg: `agent "${role.id}" hit a fatal (non-retryable) error — not restarting`,
941
- });
942
- crash();
943
- return;
944
- }
945
- if (mailbox.isClosed || attempt >= BACKOFFS_MS.length) {
946
- crash();
947
- return;
948
- }
949
- bus.emit({
950
- type: 'status',
951
- from: role.id,
952
- reason: 'agent-restart',
953
- msg: `agent "${role.id}" crashed (${message}) — restarting in ${BACKOFFS_MS[attempt]}ms (attempt ${attempt + 1}/${BACKOFFS_MS.length})`,
954
- });
955
- await new Promise((r) => {
956
- const t = setTimeout(r, BACKOFFS_MS[attempt]);
957
- t.unref?.();
958
- });
959
- if (mailbox.isClosed) {
960
- crash();
961
- return;
962
- } // org stopped during backoff — never recovered
963
- }
964
- }
965
- })();
966
- }
734
+ const { runtime, abort } = this.spawnRoleIncarnation(name, running, role, roleCheckpoint?.generation ?? 0, { roleCheckpoint });
967
735
  running.agents.set(role.id, runtime);
736
+ running.roleSlots.set(role.id, {
737
+ generation: roleCheckpoint?.generation ?? 0,
738
+ phase: 'running',
739
+ runtime,
740
+ abort,
741
+ effectiveRole: roleCheckpoint?.effectiveRoleOverrides &&
742
+ Object.keys(roleCheckpoint.effectiveRoleOverrides).length > 0
743
+ ? mergeEffectiveRoleConfig(role, roleCheckpoint.effectiveRoleOverrides)
744
+ : role,
745
+ respawnCount: roleCheckpoint?.respawnCount ?? 0,
746
+ queuedDuringSwap: roleCheckpoint?.queuedDuringSwap ?? [],
747
+ retiredUsage: roleCheckpoint?.retiredUsage ?? { tokens: 0, costUsd: 0 },
748
+ });
968
749
  };
969
750
  if (options?.resume && checkpoint) {
970
751
  const restoredRoles = new Set(Object.keys(checkpoint.roleState));
@@ -981,7 +762,23 @@ export class OrgDaemon {
981
762
  }
982
763
  running.pendingRoles = pendingRoles;
983
764
  running.spawnRole = spawnRole;
984
- running.taskDag = new TaskDag();
765
+ running.taskDag =
766
+ checkpoint.tasks && checkpoint.tasks.length > 0
767
+ ? TaskDag.fromJSON(checkpoint.tasks)
768
+ : new TaskDag();
769
+ // A 'running' task's "[task:…]" message was consumed by the session that
770
+ // was working it. If that role's SDK session is resumed (checkpointed
771
+ // sessionId) the task is still in its context; otherwise — role not
772
+ // restored at all, or restored into a fresh session — nothing knows
773
+ // about the task, so put it back to 'ready' and re-dispatch.
774
+ for (const task of running.taskDag.all()) {
775
+ if (task.status !== 'running')
776
+ continue;
777
+ if (checkpoint.roleState[task.assignee]?.sessionId)
778
+ continue;
779
+ running.taskDag.requeue(task.id);
780
+ }
781
+ decisionOps.dispatchReadyTasks(this, name, running);
985
782
  if (worktreePath)
986
783
  running.worktreePath = worktreePath;
987
784
  }
@@ -1103,6 +900,18 @@ export class OrgDaemon {
1103
900
  const pendingGates = this.readGates(name).gates.filter((g) => g.status === 'pending');
1104
901
  if (pendingGates.length > 0)
1105
902
  return;
903
+ // Bug 3: a pending ask_human question is the same kind of legitimate
904
+ // wait as a pending gate — askHuman()'s receipt tells the role to end
905
+ // its turn and wait for the resolution, so a role that follows that
906
+ // instruction and goes quiet looks identical to a genuinely stalled
907
+ // agent. Without this check the watchdog nudges (and, after enough
908
+ // nudges, idle-stops) an org that's simply waiting on a human answer
909
+ // that's already on its way.
910
+ const pendingQuestions = questionOps
911
+ .readQuestions(this.root, name)
912
+ .questions.filter((q) => q.answer === null);
913
+ if (pendingQuestions.length > 0)
914
+ return;
1106
915
  // Auto-resume any task whose org_task_block time has passed: flip it
1107
916
  // back to 'running' and re-push it into the assignee's mailbox, same
1108
917
  // as a fresh dispatch. This IS real activity, so fall through to the
@@ -1177,7 +986,8 @@ export class OrgDaemon {
1177
986
  this.watchdogs.set(name, wd);
1178
987
  }
1179
988
  if (this.opts.crossProcess && this.opts.inboxUrl) {
1180
- const lease = new BrokerLease(name, this.opts.inboxUrl, this.opts.brokerDir, undefined, normalizeCredential(this.opts.inboxCredential));
989
+ const operatorCred = normalizeCredential(this.opts.operatorCredential);
990
+ const lease = new BrokerLease(name, this.opts.inboxUrl, this.opts.brokerDir, undefined, running.credential, operatorCred ? { credential: operatorCred, dir: this.opts.operatorDir } : undefined);
1181
991
  lease.start();
1182
992
  this.leases.set(name, lease);
1183
993
  }
@@ -1191,8 +1001,19 @@ export class OrgDaemon {
1191
1001
  // queueMessage had already reported them accepted.
1192
1002
  if (!running.agents.has(msg.toRole) && running.pendingRoles?.has(msg.toRole)) {
1193
1003
  const pending = running.pendingRoles.get(msg.toRole);
1194
- running.pendingRoles.delete(msg.toRole);
1195
- running.spawnRole?.(pending);
1004
+ // Bug 4: don't spawn past run_config.max_concurrent_agents. Requeue
1005
+ // this message (queueMessage, not a silent drop) and defer the spawn
1006
+ // the same way a concurrency-gated lazy spawn defers elsewhere.
1007
+ const concurrencyLimit = def.run_config.max_concurrent_agents;
1008
+ if (concurrencyLimit != null && activeRoleCount(running) >= concurrencyLimit) {
1009
+ running.pendingRoles.delete(msg.toRole);
1010
+ queueMessage(this.root, name, msg);
1011
+ this.scheduleConcurrencyDeferredSpawn(name, running, pending, running.spawnRole);
1012
+ }
1013
+ else {
1014
+ running.pendingRoles.delete(msg.toRole);
1015
+ running.spawnRole?.(pending);
1016
+ }
1196
1017
  }
1197
1018
  const agent = running.agents.get(msg.toRole);
1198
1019
  if (agent && !agent.mailbox.isClosed) {
@@ -1203,13 +1024,751 @@ export class OrgDaemon {
1203
1024
  subject: msg.subject,
1204
1025
  msg: msg.body,
1205
1026
  });
1206
- agent.mailbox.push(this.mailBody(name, running, `[message from ${msg.fromQualified}] subject: ${msg.subject}`, msg.body, `inbox-${msg.ts}-${Math.random().toString(36).slice(2, 8)}`));
1027
+ await crossOrg.pushMessage(this, name, running, msg.toRole, msg.fromQualified, msg.subject, msg.body, `inbox-${msg.ts}-${Math.random().toString(36).slice(2, 8)}`);
1207
1028
  }
1208
1029
  }
1209
1030
  if (queued.length)
1210
1031
  bus.emit({ type: 'status', msg: `drained ${queued.length} queued message(s) from inbox` });
1211
1032
  return running;
1212
1033
  }
1034
+ /** Build one role incarnation: mailbox, policy, AgentRuntime, sessionOpts,
1035
+ * and the supervised crash-retry loop. Used by BOTH the startup lazy-spawn
1036
+ * path (generation 0, via the `spawnRole` closure inside startOrgInner)
1037
+ * and respawnRole() (generation N+1). Does not touch running.agents or
1038
+ * running.roleSlots — callers publish the result themselves. */
1039
+ spawnRoleIncarnation(name, running, role, generation, opts = {}) {
1040
+ const { roleCheckpoint } = opts;
1041
+ const abort = opts.abort ?? new AbortController();
1042
+ const { def, bus, run } = running;
1043
+ const cwd = running.workdir;
1044
+ const ws = this.workspaceSetting(def);
1045
+ const perRoleBudget = opts.budgetTokensOverride ?? computeReplacementBudget(def, role.id);
1046
+ let roleCwd = cwd;
1047
+ const existingSlot = running.roleSlots.get(role.id);
1048
+ if (ws === 'worktree-per-role' && role.id !== running.bossRoleId) {
1049
+ const wtPath = join(this.root, ORG_DIR, name, `worktree-${role.id}`);
1050
+ if (existingSlot?.runtime?.worktreePath === wtPath && existsSync(wtPath)) {
1051
+ // A replacement (generation > 0) reuses the SAME worktree path —
1052
+ // recreating it here would delete any uncommitted work the old
1053
+ // incarnation left behind (design constraint #5).
1054
+ roleCwd = wtPath;
1055
+ }
1056
+ else {
1057
+ try {
1058
+ // Q7: top-level `import { execFileSync }` replaces the inlined
1059
+ // `require('node:child_process')` that broke ESM at runtime —
1060
+ // vitest's CJS shim masked it in tests but the built package
1061
+ // threw "require is not defined" in real Node ESM execution.
1062
+ // SEC-5: argv-array form, no shell.
1063
+ if (existsSync(wtPath)) {
1064
+ try {
1065
+ execFileSync('git', ['worktree', 'remove', '--force', wtPath], {
1066
+ cwd: this.root,
1067
+ stdio: 'ignore',
1068
+ timeout: 30_000,
1069
+ });
1070
+ }
1071
+ catch {
1072
+ /* best-effort */
1073
+ }
1074
+ }
1075
+ execFileSync('git', ['worktree', 'add', wtPath, 'HEAD', '--detach'], {
1076
+ cwd: this.root,
1077
+ stdio: 'ignore',
1078
+ timeout: 30_000,
1079
+ });
1080
+ roleCwd = wtPath;
1081
+ }
1082
+ catch {
1083
+ /* fallback to shared cwd if git worktree fails */
1084
+ }
1085
+ }
1086
+ }
1087
+ const mailbox = new Mailbox();
1088
+ if (roleCheckpoint?.mailboxQueue?.length) {
1089
+ restoreMailboxQueue({ mailbox }, roleCheckpoint.mailboxQueue);
1090
+ }
1091
+ // A recoverable close (budget exhaustion) is left open on resume — see
1092
+ // isRecoverableCloseReason's doc comment. Re-closing it here would
1093
+ // make the idle watchdog's "raise the budget and resume" remedy a
1094
+ // no-op, since nothing in this codebase ever reopens a closed mailbox.
1095
+ if (roleCheckpoint?.mailboxClosed &&
1096
+ !isRecoverableCloseReason(roleCheckpoint.mailboxCloseReason)) {
1097
+ mailbox.close(roleCheckpoint.mailboxCloseReason);
1098
+ }
1099
+ const policy = new PolicyEngine(role.id, {
1100
+ maxTokens: role.budget_tokens ?? perRoleBudget,
1101
+ maxUsd: role.budget_usd,
1102
+ ...(role.policy ?? {}),
1103
+ }, bus, roleCwd);
1104
+ if (roleCheckpoint?.tokensUsed) {
1105
+ policy.setUsage(roleCheckpoint.tokensUsed);
1106
+ }
1107
+ // ORG-7: restore accumulated USD spend across resume so a stop/resume
1108
+ // cycle can't reset a role's USD budget back to zero.
1109
+ if (roleCheckpoint?.costUsd) {
1110
+ policy.setUsageUsd(roleCheckpoint.costUsd);
1111
+ }
1112
+ const runtime = {
1113
+ mailbox,
1114
+ policy,
1115
+ status: roleCheckpoint?.status ?? 'running',
1116
+ done: Promise.resolve(),
1117
+ metrics: { tokens: roleCheckpoint?.tokensUsed ?? 0, costUsd: roleCheckpoint?.costUsd ?? 0 },
1118
+ lastMessageId: roleCheckpoint?.lastMessageId,
1119
+ error: roleCheckpoint?.error,
1120
+ sessionId: roleCheckpoint?.sessionId,
1121
+ worktreePath: roleCwd !== cwd ? roleCwd : undefined,
1122
+ scrollback: new ScrollbackBuffer(),
1123
+ };
1124
+ if (roleCheckpoint?.scrollback?.length) {
1125
+ for (const line of roleCheckpoint.scrollback)
1126
+ runtime.scrollback.push(line);
1127
+ }
1128
+ const sessionOpts = {
1129
+ org: name,
1130
+ role,
1131
+ bus,
1132
+ policy,
1133
+ mailbox,
1134
+ cwd: roleCwd,
1135
+ def,
1136
+ // Pass the org state directory so runners that persist per-role state
1137
+ // (VercelAgentRunner session files) write under .monomind/orgs/<name>
1138
+ // instead of polluting the workspace cwd.
1139
+ orgDir: join(this.root, ORG_DIR, name),
1140
+ // Project root for named-provider (`adapter_config.provider`) config
1141
+ // lookup — role cwd may be an isolated workspace with no config file.
1142
+ orgRoot: this.root,
1143
+ maxTurns: role.max_turns_per_message ?? def.run_config.max_turns_per_message,
1144
+ resumeSessionId: roleCheckpoint?.sessionId,
1145
+ lastMessageId: () => runtime.lastMessageId,
1146
+ onOutput: (line) => runtime.scrollback.push(line),
1147
+ onSessionId: (id) => {
1148
+ runtime.sessionId = id;
1149
+ },
1150
+ deliver: (from, to, subject, body) => this.deliver(name, from, to, subject, body),
1151
+ askHuman: (r, question) => this.askHuman(name, r, question),
1152
+ onGate: (r, gateName, gateDesc) => this.createGate(name, r, gateName, gateDesc),
1153
+ circuitBreaker: (() => {
1154
+ const cb = def.run_config.circuit_breaker;
1155
+ if (!cb)
1156
+ return undefined;
1157
+ return { threshold: cb.failure_threshold ?? 5, state: { failures: 0, tripped: false } };
1158
+ })(),
1159
+ beforeTool: (r, toolName, input) => this.checkApproval(name, r, toolName, input),
1160
+ fence: running.fences?.get(role.id),
1161
+ // ORG-1: gatedCanUseTool denials are a natural decision point — record them so
1162
+ // `org decisions` shows real traces instead of always reporting none.
1163
+ onDecision: (r, toolName, message) => {
1164
+ this.recordDecision(name, r, {
1165
+ type: 'tool',
1166
+ context: `tool call: ${toolName}`,
1167
+ reasoning: message,
1168
+ outcome: 'denied',
1169
+ });
1170
+ },
1171
+ // ORG-9: decision gates are documented as "hard-blocking" — make that
1172
+ // true by actually denying tool use while this role has a pending gate,
1173
+ // the same way pending approvals already do.
1174
+ hasPendingGate: () => this.listGates(name, 'pending').some((g) => g.roleId === role.id),
1175
+ onComplete: role.id === running.bossRoleId
1176
+ ? (r, outcome, summary) => {
1177
+ bus.emit({
1178
+ type: 'status',
1179
+ from: r,
1180
+ reason: 'org-complete',
1181
+ msg: `run outcome: ${outcome}`,
1182
+ data: { outcome, summary },
1183
+ });
1184
+ }
1185
+ : undefined,
1186
+ // #11: a boss that overflows its context window isn't a crash (it keeps
1187
+ // returning +0-token errors forever), so without this the idle watchdog
1188
+ // just nudges it for ~30 min before idle-stopping. Restart the whole org
1189
+ // with fresh sessions instead — bounded by MAX_BOSS_RESTARTS.
1190
+ onContextLimit: role.id === running.bossRoleId ? () => this.scheduleBossRestart(name) : undefined,
1191
+ onListRuntimeOptions: role.id === running.bossRoleId && (def.run_config.max_role_respawns ?? 0) > 0
1192
+ ? () => this.listRuntimeOptions()
1193
+ : undefined,
1194
+ onRespawnRole: role.id === running.bossRoleId && (def.run_config.max_role_respawns ?? 0) > 0
1195
+ ? (callerId, args) => this.respawnRole(name, callerId, args)
1196
+ : undefined,
1197
+ recall: async (r, q) => {
1198
+ const answer = await this.recallOrgMemory(name, def, q, r);
1199
+ bus.emit({
1200
+ type: 'status',
1201
+ from: r,
1202
+ reason: 'org-recall',
1203
+ msg: `recall: ${q.slice(0, 80)}`,
1204
+ data: { hits: answer.hits },
1205
+ });
1206
+ return answer.text;
1207
+ },
1208
+ searchKnowledge: async (r, q) => {
1209
+ const answer = await this.searchProjectKnowledge(q);
1210
+ bus.emit({
1211
+ type: 'status',
1212
+ from: r,
1213
+ reason: 'knowledge-search',
1214
+ msg: `knowledge: ${q.slice(0, 80)}`,
1215
+ data: { hits: answer.hits },
1216
+ });
1217
+ return answer.text;
1218
+ },
1219
+ glossary: running.glossary,
1220
+ remember: async (r, content, scope) => {
1221
+ const text = await this.rememberOrgMemory(name, def, r, content, scope, run);
1222
+ bus.emit({
1223
+ type: 'status',
1224
+ from: r,
1225
+ reason: 'org-remember',
1226
+ msg: `remember (${scope}): ${content.slice(0, 80)}`,
1227
+ data: { scope },
1228
+ });
1229
+ return text;
1230
+ },
1231
+ learn: async (r, payload) => {
1232
+ const text = await this.learnOrgKnowledge(name, run, payload);
1233
+ bus.emit({
1234
+ type: 'status',
1235
+ from: r,
1236
+ reason: 'org-learn',
1237
+ msg: `learn: ${text.slice(0, 120)}`,
1238
+ data: {
1239
+ nodes: payload.nodes?.length ?? 0,
1240
+ edges: payload.edges?.length ?? 0,
1241
+ rules: payload.rules?.length ?? 0,
1242
+ },
1243
+ });
1244
+ return text;
1245
+ },
1246
+ createTask: (r, title, assignee, deps) => {
1247
+ return this.dagCreateTask(name, r, title, assignee, deps);
1248
+ },
1249
+ completeTask: (r, taskId, result) => {
1250
+ return this.dagCompleteTask(name, r, taskId, result);
1251
+ },
1252
+ listTasks: () => {
1253
+ const running = this.orgs.get(name);
1254
+ return JSON.stringify(running?.taskDag?.all() ?? [], null, 2);
1255
+ },
1256
+ splitTask: (r, parentId, children) => {
1257
+ return this.dagSplitTask(name, r, parentId, children);
1258
+ },
1259
+ mergeTask: (r, sourceId, targetId) => {
1260
+ return this.dagMergeTask(name, r, sourceId, targetId);
1261
+ },
1262
+ cancelTask: (r, taskId, reason) => {
1263
+ return this.dagCancelTask(name, r, taskId, reason);
1264
+ },
1265
+ blockTask: (r, taskId, untilIso, reason) => {
1266
+ return this.dagBlockTask(name, r, taskId, untilIso, reason);
1267
+ },
1268
+ planGraph: (r, specs) => {
1269
+ return this.dagPlanGraph(name, r, specs);
1270
+ },
1271
+ queryFn: this.opts.queryFn,
1272
+ // Runner resolution: explicit opts.runner > role `runtime` field >
1273
+ // org def `runtime` field > MONOMIND_RUNTIME env (opencode/kimicode) >
1274
+ // undefined (session.ts falls back to ClaudeAgentRunner via queryFn).
1275
+ // Leaving it undefined for the default path is what keeps
1276
+ // Claude/Antigravity orgs byte-for-byte unchanged. Session opts are
1277
+ // built per role here, so each role gets its own runner.
1278
+ runner: this.opts.runner ??
1279
+ resolveRoleRunner(role.runtime, def.runtime, role.provider?.kind, undefined, role.provider),
1280
+ // Lets respawnRole force-stop THIS specific incarnation (mid-run role
1281
+ // replacement's forced-stop step) without reaching into runAgentSession's
1282
+ // internals.
1283
+ externalAbort: abort,
1284
+ };
1285
+ // Supervised session: transient crashes (provider blips, network) restart
1286
+ // with backoff; a crash with the mailbox already closed, or one that
1287
+ // exhausts the retry budget, is terminal. runAgentSession already emits a
1288
+ // 'status' event for the raw error; the terminal 'audit' event is for
1289
+ // dashboards/alerts that filter on actionable failures (not routine
1290
+ // status chatter) so a dead agent surfaces instead of a run that
1291
+ // silently never progresses.
1292
+ const BACKOFFS_MS = this.opts.crashBackoffsMs ?? [1000, 5000, 15000];
1293
+ const myGeneration = generation;
1294
+ const isStaleGeneration = () => (running.roleSlots.get(role.id)?.generation ?? 0) !== myGeneration;
1295
+ if (!mailbox.isClosed && runtime.status !== 'crashed') {
1296
+ runtime.done = (async () => {
1297
+ for (let attempt = 0;; attempt++) {
1298
+ try {
1299
+ await runAgentSession(sessionOpts);
1300
+ runtime.status = 'ended';
1301
+ return;
1302
+ }
1303
+ catch (err) {
1304
+ // A deliberate respawn (see respawnRole) bumps the slot's
1305
+ // generation and force-stops this incarnation via its
1306
+ // externalAbort - that abort makes runAgentSession reject here
1307
+ // exactly like a real crash would. Recognize supersession
1308
+ // FIRST: this generation's retry loop must never restart,
1309
+ // never run terminal crash handling, and never notify the
1310
+ // boss - the replacement (a new generation, spawned
1311
+ // separately) already owns this role id.
1312
+ if (isStaleGeneration())
1313
+ return;
1314
+ // Drop the crashed session's stale waker immediately: a push()
1315
+ // during the backoff window must queue for the NEXT session, not
1316
+ // wake the dead generator to swallow it.
1317
+ mailbox.detach();
1318
+ // #203: if the crashed session's mailbox generator was abandoned
1319
+ // mid-yield (message already shift()ed for it, turn never
1320
+ // finished), put that message back on the queue — otherwise the
1321
+ // replacement session's stream() finds an empty queue and parks
1322
+ // forever, since the "delivered" message is gone for good.
1323
+ mailbox.reclaimInFlight();
1324
+ const message = err instanceof Error ? err.message : String(err);
1325
+ const isTurnLimit = /Reached maximum number of turns|error_max_turns/i.test(message);
1326
+ // Bounded like every other recovery: attempt counts every pass
1327
+ // through this loop, so a role that keeps surfacing max-turns
1328
+ // errors here (session.ts already swallows the normal ones)
1329
+ // falls through to crash handling instead of looping forever.
1330
+ if (isTurnLimit && !mailbox.isClosed && attempt < BACKOFFS_MS.length) {
1331
+ sessionOpts.resumeSessionId = undefined;
1332
+ mailbox.push(`${Mailbox.CONTINUE_PREFIX} You reached the turn limit on your task. Continue your in-progress work from where you left off; if finished, end your turn.`);
1333
+ bus.emit({
1334
+ type: 'status',
1335
+ from: role.id,
1336
+ reason: 'turn-limit-recover',
1337
+ msg: `agent "${role.id}" hit turn limit error — continuing with fresh session`,
1338
+ });
1339
+ continue;
1340
+ }
1341
+ // Exit 143 = SIGTERM. If the mailbox is already closed, we
1342
+ // sent the signal ourselves during stop — not a crash.
1343
+ const killedByStop = mailbox.isClosed && /exit(?:ed)? with code 143/.test(message);
1344
+ const crash = () => {
1345
+ if (killedByStop) {
1346
+ runtime.status = 'ended';
1347
+ bus.emit({
1348
+ type: 'status',
1349
+ from: role.id,
1350
+ msg: `agent "${role.id}" terminated by stop (was still working when drain window expired)`,
1351
+ reason: 'terminated-by-stop',
1352
+ });
1353
+ return;
1354
+ }
1355
+ runtime.status = 'crashed';
1356
+ runtime.error = message;
1357
+ // Close the mailbox so deliver()/receiveRemote() report a real
1358
+ // error instead of pushing into a queue no session will read
1359
+ // (and returning a false "delivered" receipt to the sender).
1360
+ mailbox.close();
1361
+ const isContextLimit = OrgDaemon.CONTEXT_LIMIT_RE.test(message);
1362
+ bus.emit({
1363
+ type: 'audit',
1364
+ from: role.id,
1365
+ msg: `agent "${role.id}" crashed: ${message}`,
1366
+ reason: isContextLimit ? 'agent-context-limit' : 'agent-session-crash',
1367
+ data: {
1368
+ agentId: role.id,
1369
+ error: message,
1370
+ restarts: attempt,
1371
+ contextLimit: isContextLimit,
1372
+ },
1373
+ });
1374
+ if (role.id !== running.bossRoleId) {
1375
+ // #2/#3: a worker is gone for the rest of this run. Without this
1376
+ // notice the coordinator keeps messaging a corpse (observed: four
1377
+ // unanswered org_send calls to a developer that had crashed on a
1378
+ // context-window limit). Tell the boss to reassign — and if the
1379
+ // crash was a context overflow, tell it to chunk smaller, since
1380
+ // re-dispatching the same task verbatim fails the same way.
1381
+ const bossRt = running.agents.get(running.bossRoleId);
1382
+ if (bossRt && !bossRt.mailbox.isClosed) {
1383
+ const guidance = isContextLimit
1384
+ ? ' This was a context-window overflow — re-dispatching the same task verbatim will fail identically. Break the work into smaller pieces (one file or section at a time) and do not paste large file contents in a single message.'
1385
+ : '';
1386
+ bossRt.mailbox.push(`[system] Worker "${role.id}" crashed and will not recover this run (${message}). It can no longer receive messages — stop messaging it. Reassign its outstanding work to another agent or take it on yourself.${guidance}`);
1387
+ bus.emit({
1388
+ type: 'audit',
1389
+ from: running.bossRoleId,
1390
+ reason: 'worker-crashed',
1391
+ msg: `worker "${role.id}" crashed (contextLimit=${isContextLimit}); coordinator notified to reassign`,
1392
+ });
1393
+ }
1394
+ }
1395
+ else {
1396
+ // #4: the coordinator itself died. Don't go silent and wait for a
1397
+ // human — attempt a bounded whole-org restart with fresh sessions
1398
+ // (which also sheds whatever bloated context caused the crash).
1399
+ this.scheduleBossRestart(name);
1400
+ }
1401
+ };
1402
+ // Fatal errors (provider auth/quota/billing — tagged with
1403
+ // err.fatal by the runner) can NEVER be fixed by a restart: the
1404
+ // same call fails identically or hangs. Skip the backoff loop
1405
+ // and go straight to terminal crash handling instead of burning
1406
+ // the retry budget and wall-clock on a guaranteed failure.
1407
+ const fatal = err?.fatal === true;
1408
+ if (fatal) {
1409
+ bus.emit({
1410
+ type: 'status',
1411
+ from: role.id,
1412
+ reason: 'agent-fatal',
1413
+ msg: `agent "${role.id}" hit a fatal (non-retryable) error — not restarting`,
1414
+ });
1415
+ crash();
1416
+ return;
1417
+ }
1418
+ if (mailbox.isClosed || attempt >= BACKOFFS_MS.length) {
1419
+ crash();
1420
+ return;
1421
+ }
1422
+ bus.emit({
1423
+ type: 'status',
1424
+ from: role.id,
1425
+ reason: 'agent-restart',
1426
+ msg: `agent "${role.id}" crashed (${message}) — restarting in ${BACKOFFS_MS[attempt]}ms (attempt ${attempt + 1}/${BACKOFFS_MS.length})`,
1427
+ });
1428
+ await new Promise((r) => {
1429
+ const t = setTimeout(r, BACKOFFS_MS[attempt]);
1430
+ t.unref?.();
1431
+ // Org stop (finishStop) aborts every active slot's controller —
1432
+ // without racing it here, this wait wouldn't notice for up to
1433
+ // BACKOFFS_MS[attempt] (default up to 15s), well past finishStop's
1434
+ // own bounded drain window. That let this loop's crash() —
1435
+ // and the bus.emit() it triggers — fire AFTER finishStop had
1436
+ // already declared the org stopped and returned, capable of
1437
+ // recreating files in a run directory a caller was already
1438
+ // deleting.
1439
+ if (abort.signal.aborted) {
1440
+ clearTimeout(t);
1441
+ r();
1442
+ return;
1443
+ }
1444
+ abort.signal.addEventListener('abort', () => {
1445
+ clearTimeout(t);
1446
+ r();
1447
+ }, { once: true });
1448
+ });
1449
+ if (isStaleGeneration())
1450
+ return; // superseded during the backoff wait
1451
+ if (mailbox.isClosed) {
1452
+ crash();
1453
+ return;
1454
+ } // org stopped during backoff — never recovered
1455
+ }
1456
+ }
1457
+ })();
1458
+ }
1459
+ return { runtime, abort };
1460
+ }
1461
+ /** org_respawn_role's daemon-owned implementation. See the design doc's
1462
+ * "Replacement algorithm" (13 steps) — this method's body follows those
1463
+ * steps in order, numbered in comments. */
1464
+ async respawnRole(name, callerId, rawInput) {
1465
+ const running = this.orgs.get(name);
1466
+ if (!running) {
1467
+ return {
1468
+ success: false,
1469
+ roleId: '',
1470
+ generation: 0,
1471
+ respawnCount: 0,
1472
+ respawnsRemaining: 0,
1473
+ error: `org "${name}" is not running`,
1474
+ };
1475
+ }
1476
+ // Step 1: authorize (defense in depth — buildOrgTools only ever wires
1477
+ // onRespawnRole for the selected coordinator, but re-check here too).
1478
+ if (callerId !== running.bossRoleId) {
1479
+ return {
1480
+ success: false,
1481
+ roleId: '',
1482
+ generation: 0,
1483
+ respawnCount: 0,
1484
+ respawnsRemaining: 0,
1485
+ error: 'only the selected coordinator may call org_respawn_role',
1486
+ };
1487
+ }
1488
+ if (this.stopping.has(name)) {
1489
+ return {
1490
+ success: false,
1491
+ roleId: '',
1492
+ generation: 0,
1493
+ respawnCount: 0,
1494
+ respawnsRemaining: 0,
1495
+ error: `org "${name}" is stopping`,
1496
+ };
1497
+ }
1498
+ const validated = validateRespawnInput(rawInput);
1499
+ if (!validated.ok) {
1500
+ return {
1501
+ success: false,
1502
+ roleId: '',
1503
+ generation: 0,
1504
+ respawnCount: 0,
1505
+ respawnsRemaining: 0,
1506
+ error: validated.error,
1507
+ };
1508
+ }
1509
+ const input = validated.value;
1510
+ if (input.roleId === running.bossRoleId) {
1511
+ return {
1512
+ success: false,
1513
+ roleId: input.roleId,
1514
+ generation: 0,
1515
+ respawnCount: 0,
1516
+ respawnsRemaining: 0,
1517
+ error: 'cannot replace the selected coordinator',
1518
+ };
1519
+ }
1520
+ const slot = running.roleSlots.get(input.roleId);
1521
+ if (!slot) {
1522
+ return {
1523
+ success: false,
1524
+ roleId: input.roleId,
1525
+ generation: 0,
1526
+ respawnCount: 0,
1527
+ respawnsRemaining: 0,
1528
+ error: `unknown or not-yet-started role "${input.roleId}"`,
1529
+ };
1530
+ }
1531
+ const maxRespawns = running.def.run_config.max_role_respawns ?? 0;
1532
+ if (slot.phase === 'removed') {
1533
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1534
+ roleId: input.roleId,
1535
+ error: `role "${input.roleId}" was removed from this org`,
1536
+ });
1537
+ }
1538
+ // Step 2: acquire the role slot (reject a concurrent replacement).
1539
+ if (running.respawning.has(input.roleId) || slot.respawnPromise) {
1540
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1541
+ roleId: input.roleId,
1542
+ error: `role "${input.roleId}" is already undergoing replacement`,
1543
+ });
1544
+ }
1545
+ if (slot.respawnCount >= maxRespawns) {
1546
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1547
+ roleId: input.roleId,
1548
+ error: `role "${input.roleId}" has reached its respawn limit (${slot.respawnCount}/${maxRespawns}) for this run`,
1549
+ });
1550
+ }
1551
+ // Step 3: resolve the candidate configuration.
1552
+ if (input.providerName !== undefined && slot.effectiveRole.provider) {
1553
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1554
+ roleId: input.roleId,
1555
+ error: `role "${input.roleId}" has an inline provider, which always takes precedence over adapter_config.provider — replacing an inline provider is a separate design`,
1556
+ });
1557
+ }
1558
+ const candidateRole = mergeEffectiveRoleConfig(slot.effectiveRole, {
1559
+ runtime: input.runtime,
1560
+ model: input.model,
1561
+ providerName: input.providerName,
1562
+ });
1563
+ const budgetTokens = input.budgetTokens ?? computeReplacementBudget(running.def, input.roleId);
1564
+ // Step 4: preflight — must not mutate the old runtime.
1565
+ try {
1566
+ resolveRoleProvider(candidateRole, this.root);
1567
+ }
1568
+ catch (err) {
1569
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1570
+ roleId: input.roleId,
1571
+ error: `preflight failed: ${err instanceof Error ? err.message : String(err)}`,
1572
+ });
1573
+ }
1574
+ // resolveRoleRunner's undefined return is the valid Claude default, not
1575
+ // an error — nothing further to validate for the runtime dimension here.
1576
+ resolveRoleRunner(candidateRole.runtime, running.def.runtime, candidateRole.provider?.kind, undefined, candidateRole.provider);
1577
+ // Step 5: consume one attempt — only after validation/preflight succeed.
1578
+ running.respawning.add(input.roleId);
1579
+ slot.respawnCount++;
1580
+ running.bus.emit({
1581
+ type: 'audit',
1582
+ from: callerId,
1583
+ reason: 'role-respawn-started',
1584
+ msg: `replacing role "${input.roleId}": ${input.reason}`,
1585
+ data: {
1586
+ roleId: input.roleId,
1587
+ from: redactRoleConfig(slot.effectiveRole),
1588
+ to: redactRoleConfig(candidateRole),
1589
+ generation: slot.generation,
1590
+ caller: callerId,
1591
+ },
1592
+ });
1593
+ // Step 6: quiesce the old incarnation. Bump the generation NOW, before
1594
+ // draining starts — not at the final publish (step 11-13) — so the OLD
1595
+ // generation's crash-retry loop (spawnRoleIncarnation's isStaleGeneration
1596
+ // check) recognizes supersession immediately. Without this, a backoff
1597
+ // timer firing during the drain/force-stop window, or the forced abort's
1598
+ // own rejection, would still see itself as the current generation:
1599
+ // the abort's rejection doesn't match killedByStop's SIGTERM-only regex,
1600
+ // so it would run full terminal crash handling — a duplicate live runner
1601
+ // (mid-backoff restart) or a false worker-crashed notification, exactly
1602
+ // what the guard exists to prevent.
1603
+ const newGeneration = slot.generation + 1;
1604
+ slot.generation = newGeneration;
1605
+ slot.phase = 'draining';
1606
+ // Every await from here on can race a stop/restart of this org — verify
1607
+ // ownership before EVERY subsequent step, not just once before the final
1608
+ // publish, so a stale operation can never mutate accounting, force-stop
1609
+ // a runtime, or spawn into an org that's no longer the live one.
1610
+ const stillOwned = () => this.orgs.get(name) === running && running.roleSlots.get(input.roleId) === slot;
1611
+ const abandonedReceipt = () => {
1612
+ running.respawning.delete(input.roleId);
1613
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1614
+ roleId: input.roleId,
1615
+ error: `org "${name}" stopped or restarted during replacement`,
1616
+ });
1617
+ };
1618
+ const oldRuntime = slot.runtime;
1619
+ const sweptQueue = oldRuntime.mailbox.beginDrain();
1620
+ slot.queuedDuringSwap.push(...sweptQueue);
1621
+ const drainTimeoutMs = running.def.run_config.respawn_drain_timeout_ms ?? 30_000;
1622
+ const drained = await Promise.race([
1623
+ oldRuntime.done.then(() => true),
1624
+ new Promise((r) => setTimeout(() => r(false), drainTimeoutMs)),
1625
+ ]);
1626
+ if (!stillOwned())
1627
+ return abandonedReceipt();
1628
+ let drainTimedOut = false;
1629
+ if (!drained) {
1630
+ drainTimedOut = true;
1631
+ // Step 7: force stop.
1632
+ slot.abort?.abort();
1633
+ const forceStopMs = running.def.run_config.respawn_force_stop_timeout_ms ?? 5_000;
1634
+ const stopped = await Promise.race([
1635
+ oldRuntime.done.then(() => true).catch(() => true),
1636
+ new Promise((r) => setTimeout(() => r(false), forceStopMs)),
1637
+ ]);
1638
+ if (!stillOwned())
1639
+ return abandonedReceipt();
1640
+ if (!stopped) {
1641
+ slot.phase = 'stuck';
1642
+ running.respawning.delete(input.roleId);
1643
+ running.bus.emit({
1644
+ type: 'audit',
1645
+ from: callerId,
1646
+ reason: 'role-respawn-failed',
1647
+ msg: `role "${input.roleId}" forced stop did not confirm termination — refusing to spawn a replacement`,
1648
+ });
1649
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1650
+ roleId: input.roleId,
1651
+ error: `role "${input.roleId}" could not be confirmed stopped; not replaced`,
1652
+ });
1653
+ }
1654
+ }
1655
+ // Step 8: preserve durable role state (worktree path, task ownership, and
1656
+ // the DAG survive untouched — they live outside AgentRuntime/Mailbox
1657
+ // entirely, keyed by role.id, which never changes). Reclaim any message
1658
+ // abandoned mid-yield by a forced stop for at-least-once redelivery.
1659
+ oldRuntime.mailbox.reclaimInFlight();
1660
+ const reclaimedQueue = oldRuntime.mailbox.serialize().queue;
1661
+ slot.queuedDuringSwap.push(...reclaimedQueue);
1662
+ // Step 9: retire accounting BEFORE replacing the runtime.
1663
+ slot.retiredUsage = {
1664
+ tokens: slot.retiredUsage.tokens + oldRuntime.policy.usage,
1665
+ costUsd: slot.retiredUsage.costUsd + oldRuntime.metrics.costUsd,
1666
+ };
1667
+ // Step 10: spawn generation N+1 (generation already bumped in step 6).
1668
+ const { runtime: newRuntime, abort: newAbort } = this.spawnRoleIncarnation(name, running, candidateRole, newGeneration, { budgetTokensOverride: budgetTokens });
1669
+ // Seed the new mailbox with everything swapped/reclaimed, delivered
1670
+ // FIFO, plus a delimited coordinator briefing appended last so it reads
1671
+ // as the newest context once the replacement starts its first turn.
1672
+ for (const queued of slot.queuedDuringSwap)
1673
+ newRuntime.mailbox.push(queued);
1674
+ newRuntime.mailbox.push(`[system: role replacement briefing — not a system prompt] You are a fresh session replacing the previous incarnation of role "${input.roleId}". Reason: ${input.reason}\n\n${input.briefing}`);
1675
+ // Seed USD accounting from retained totals so a respawn cannot reset
1676
+ // role.budget_usd.
1677
+ if (running.def.roles.find((r) => r.id === input.roleId)?.budget_usd !== undefined) {
1678
+ newRuntime.policy.setUsageUsd(slot.retiredUsage.costUsd);
1679
+ }
1680
+ // "Ready" here means "did not crash within the startup window" — a
1681
+ // silent-but-healthy runner (one that never emits a chat/tool/usage
1682
+ // event, e.g. because it hasn't finished its first turn yet) must not be
1683
+ // misreported as a startup failure, so this does NOT wait for a positive
1684
+ // signal. It races the new incarnation's own crash-retry loop (which
1685
+ // shares this generation, so it is NOT superseded and behaves normally)
1686
+ // against the timeout: a config that fails immediately (bad model,
1687
+ // missing runtime binary, auth failure) crashes fast and newRuntime.done
1688
+ // resolves with status 'crashed' well before startTimeoutMs, correctly
1689
+ // failing readiness and triggering rollback.
1690
+ const startTimeoutMs = running.def.run_config.respawn_start_timeout_ms ?? 60_000;
1691
+ const ready = await Promise.race([
1692
+ newRuntime.done.then(() => newRuntime.status !== 'crashed'),
1693
+ new Promise((r) => setTimeout(() => r(true), startTimeoutMs)),
1694
+ ]);
1695
+ // Step 11: publish atomically — verify ownership is still current.
1696
+ if (!stillOwned()) {
1697
+ newAbort.abort();
1698
+ return abandonedReceipt();
1699
+ }
1700
+ if (!ready) {
1701
+ // Step 12: rollback — one attempt with the prior effective config.
1702
+ newAbort.abort();
1703
+ running.bus.emit({
1704
+ type: 'audit',
1705
+ from: callerId,
1706
+ reason: 'role-respawn-failed',
1707
+ msg: `role "${input.roleId}" replacement did not become ready within ${startTimeoutMs}ms — attempting rollback`,
1708
+ });
1709
+ try {
1710
+ const { runtime: rolledBack, abort: rolledBackAbort } = this.spawnRoleIncarnation(name, running, slot.effectiveRole, newGeneration + 1, {});
1711
+ for (const queued of slot.queuedDuringSwap)
1712
+ rolledBack.mailbox.push(queued);
1713
+ running.agents.set(input.roleId, rolledBack);
1714
+ slot.runtime = rolledBack;
1715
+ slot.abort = rolledBackAbort;
1716
+ slot.generation = newGeneration + 1;
1717
+ slot.phase = 'running';
1718
+ slot.queuedDuringSwap = [];
1719
+ running.respawning.delete(input.roleId);
1720
+ running.bus.emit({
1721
+ type: 'audit',
1722
+ from: callerId,
1723
+ reason: 'role-respawn-failed',
1724
+ msg: `role "${input.roleId}" replacement failed; rolled back to prior config`,
1725
+ });
1726
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1727
+ roleId: input.roleId,
1728
+ drainTimedOut,
1729
+ error: `replacement failed to start; rolled back to prior configuration`,
1730
+ });
1731
+ }
1732
+ catch (rollbackErr) {
1733
+ slot.phase = 'crashed';
1734
+ running.respawning.delete(input.roleId);
1735
+ running.bus.emit({
1736
+ type: 'audit',
1737
+ from: callerId,
1738
+ reason: 'role-respawn-rollback-failed',
1739
+ msg: `role "${input.roleId}" replacement AND rollback both failed: ${rollbackErr instanceof Error ? rollbackErr.message : String(rollbackErr)}`,
1740
+ });
1741
+ return buildRespawnReceipt(slot, maxRespawns, false, {
1742
+ roleId: input.roleId,
1743
+ drainTimedOut,
1744
+ error: `replacement and rollback both failed; role "${input.roleId}" is unavailable`,
1745
+ });
1746
+ }
1747
+ }
1748
+ running.agents.set(input.roleId, newRuntime);
1749
+ slot.runtime = newRuntime;
1750
+ slot.abort = newAbort;
1751
+ slot.generation = newGeneration;
1752
+ slot.effectiveRole = candidateRole;
1753
+ slot.phase = 'running';
1754
+ slot.queuedDuringSwap = [];
1755
+ running.respawning.delete(input.roleId);
1756
+ // Step 13: audit and persist.
1757
+ running.bus.emit({
1758
+ type: 'audit',
1759
+ from: callerId,
1760
+ reason: 'role-respawned',
1761
+ msg: `role "${input.roleId}" replaced (generation ${newGeneration})`,
1762
+ data: {
1763
+ roleId: input.roleId,
1764
+ generation: newGeneration,
1765
+ respawnCount: slot.respawnCount,
1766
+ drainTimedOut,
1767
+ },
1768
+ });
1769
+ this.persistState(name, 'running', running.run);
1770
+ return buildRespawnReceipt(slot, maxRespawns, true, { roleId: input.roleId, drainTimedOut });
1771
+ }
1213
1772
  /** @internal */
1214
1773
  hasOrgDef(name) {
1215
1774
  if (!/^[a-z0-9][a-z0-9_-]{0,63}$/i.test(name))
@@ -1275,6 +1834,14 @@ export class OrgDaemon {
1275
1834
  this.leases.delete(name);
1276
1835
  for (const a of org.agents.values())
1277
1836
  a.mailbox.close();
1837
+ // Closing the mailbox stops new work being handed to a session, but does
1838
+ // NOT cancel a turn already in flight (e.g. mid provider call) — that
1839
+ // session can keep running, and eventually crash/finish, well past this
1840
+ // function's own bounded drain below. Abort each slot's live incarnation
1841
+ // too, reusing respawnRole's existing force-stop handle, so in-flight
1842
+ // work is told to stop now instead of merely being denied new input.
1843
+ for (const slot of org.roleSlots.values())
1844
+ slot.abort?.abort();
1278
1845
  // Bounded: a genuinely hung agent session (stuck mid-tool-call, not just
1279
1846
  // idle) must not make stopOrg() hang forever — callers like the scheduler
1280
1847
  // already race their own timeout around a run, and this wait re-blocking
@@ -1286,10 +1853,20 @@ export class OrgDaemon {
1286
1853
  // session ends, so a long drain is a ceiling, not a delay.
1287
1854
  const stopWaitMs = drainMs ?? this.opts.stopWaitMs ?? 15_000;
1288
1855
  const allDone = Promise.allSettled([...org.agents.values()].map((a) => a.done)).then(() => false);
1856
+ // Clear the ceiling timer once the sessions win the race: left pending, a
1857
+ // COMPLETE_DRAIN_MS stop kept `org run` (which returns without
1858
+ // process.exit on a clean completion) alive for up to five minutes after
1859
+ // every session had already ended. Deliberately NOT unref'd — on the
1860
+ // timed-out path this timer may be the only thing keeping the loop alive
1861
+ // long enough to write 'stopped' to runtime.json and flush the bus.
1862
+ let drainTimer;
1289
1863
  const timedOut = await Promise.race([
1290
1864
  allDone,
1291
- new Promise((r) => setTimeout(() => r(true), stopWaitMs)),
1865
+ new Promise((r) => {
1866
+ drainTimer = setTimeout(() => r(true), stopWaitMs);
1867
+ }),
1292
1868
  ]);
1869
+ clearTimeout(drainTimer);
1293
1870
  if (timedOut) {
1294
1871
  // #152: "proceeding anyway" alone didn't say WHO got cut off — a run
1295
1872
  // reviewer had no way to tell whether real, in-progress work (a
@@ -1325,6 +1902,13 @@ export class OrgDaemon {
1325
1902
  }
1326
1903
  org.bus.emit({ type: 'status', msg: 'org stopped' });
1327
1904
  await org.bus.flush();
1905
+ // flush() only awaits a snapshot of writes queued at call time (see its
1906
+ // own doc comment) — it has no visibility into a session that crashes
1907
+ // after the abort signal above but before this function returns. Seal
1908
+ // the bus now so any such late bus.emit() still reaches in-memory
1909
+ // listeners but can never schedule a new disk write into a run
1910
+ // directory a caller (e.g. a test's afterEach) may already be deleting.
1911
+ await org.bus.seal();
1328
1912
  // Append this run's summary to <org>/history.jsonl — read back from the
1329
1913
  // flushed bus.jsonl (the full durable record) rather than the bounded
1330
1914
  // in-memory buffer, so long runs summarize completely.
@@ -1461,6 +2045,17 @@ export class OrgDaemon {
1461
2045
  for (const [name, org] of this.orgs) {
1462
2046
  try {
1463
2047
  const p = join(this.root, ORG_DIR, name, 'runtime.json');
2048
+ // Capture separately from the write below: a throw here (e.g. a
2049
+ // cyclic structure in roleState reaching generateChecksum) must not
2050
+ // suppress the base crash record, which is the actually-important
2051
+ // best-effort write this method exists for.
2052
+ let checkpoint;
2053
+ try {
2054
+ checkpoint = captureCheckpoint(org, 'crashed');
2055
+ }
2056
+ catch {
2057
+ /* best effort — proceed without a checkpoint */
2058
+ }
1464
2059
  // C4: atomic write — crash handler is the most likely place to hit
1465
2060
  // a partial write since the process is mid-teardown.
1466
2061
  writeJsonFileAtomic(p, {
@@ -1469,6 +2064,7 @@ export class OrgDaemon {
1469
2064
  pid: process.pid,
1470
2065
  updated: new Date().toISOString(),
1471
2066
  closedBy: 'crash-handler',
2067
+ ...(checkpoint ? { checkpoint } : {}),
1472
2068
  ...(error ? { error } : {}),
1473
2069
  });
1474
2070
  }
@@ -1508,8 +2104,8 @@ export class OrgDaemon {
1508
2104
  }
1509
2105
  // ── Delegated methods — extracted to focused modules ──────────────────
1510
2106
  // approvals.ts
1511
- checkApproval(org, role, action) {
1512
- return approvalOps.checkApproval(this, org, role, action);
2107
+ checkApproval(org, role, action, input) {
2108
+ return approvalOps.checkApproval(this, org, role, action, input);
1513
2109
  }
1514
2110
  async setApproval(org, role, action, approved) {
1515
2111
  return approvalOps.setApproval(this, org, role, action, approved);
@@ -1562,11 +2158,12 @@ export class OrgDaemon {
1562
2158
  async deliver(fromOrg, fromRole, to, subject, body) {
1563
2159
  return crossOrg.deliver(this, fromOrg, fromRole, to, subject, body);
1564
2160
  }
1565
- receiveRemote(toOrg, toRole, fromQualified, subject, body) {
1566
- return crossOrg.receiveRemote(this, toOrg, toRole, fromQualified, subject, body);
2161
+ receiveRemote(toOrg, toRole, fromQualified, subject, body, fromCredential) {
2162
+ return crossOrg.receiveRemote(this, toOrg, toRole, fromQualified, subject, body, fromCredential);
1567
2163
  }
1568
- mailBody(orgName, org, header, body, id) {
1569
- return crossOrg.mailBody(this.root, orgName, org, header, body, id);
2164
+ // runtime-options.ts
2165
+ listRuntimeOptions() {
2166
+ return buildRuntimeOptions(this.root);
1570
2167
  }
1571
2168
  // scheduler-integration.ts
1572
2169
  /** @internal */
@@ -1580,6 +2177,13 @@ export class OrgDaemon {
1580
2177
  scheduleDeferredSpawn(name, running, role, spawnRole) {
1581
2178
  scheduler.scheduleDeferredSpawn(this, name, running, role, spawnRole);
1582
2179
  }
2180
+ /** Bug 4: mirrors scheduleDeferredSpawn, but for a role deferred because the
2181
+ * org is already at run_config.max_concurrent_agents rather than under host
2182
+ * resource pressure — see scheduleConcurrencyDeferredSpawn's doc comment. */
2183
+ /** @internal */
2184
+ scheduleConcurrencyDeferredSpawn(name, running, role, spawnRole) {
2185
+ scheduler.scheduleConcurrencyDeferredSpawn(this, name, running, role, spawnRole);
2186
+ }
1583
2187
  // org-memory.ts
1584
2188
  orgMemoryNamespace(name, def) {
1585
2189
  return orgMemory.orgMemoryNamespace(name, def);