@monoes/monomindcli 2.10.10 → 2.10.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/mastermind/brain.md +14 -14
- package/.claude/commands/mastermind/help.md +2 -2
- package/.claude/commands/mastermind/master.md +24 -19
- package/.claude/commands/mastermind/memory.md +8 -8
- package/.claude/commands/mastermind/monoswarm.md +4 -4
- package/.claude/commands/mastermind.md +7 -7
- package/.claude/commands/truth/start.md +3 -3
- package/.claude/helpers/handlers/gates-handler.cjs +13 -1
- package/.claude/skills/mastermind/SKILL.md +7 -16
- package/.claude/skills/mastermind-debug/SKILL.md +3 -3
- package/.claude/skills/mastermind-design/SKILL.md +2 -0
- package/.claude/skills/mastermind-execute/SKILL.md +66 -11
- package/.claude/skills/mastermind-idea/SKILL.md +9 -2
- package/.claude/skills/mastermind-intake/SKILL.md +31 -7
- package/.claude/skills/mastermind-issue-detail/SKILL.md +70 -16
- package/.claude/skills/mastermind-issues/SKILL.md +111 -16
- package/.claude/skills/mastermind-liveness/SKILL.md +96 -26
- package/.claude/skills/mastermind-my-issues/SKILL.md +40 -8
- package/.claude/skills/mastermind-org/SKILL.md +2 -0
- package/.claude/skills/mastermind-plan/SKILL.md +7 -16
- package/.claude/skills/mastermind-plan-to-tasks/SKILL.md +132 -24
- package/.claude/skills/mastermind-protocol/SKILL.md +33 -22
- package/.claude/skills/mastermind-runorg/SKILL.md +22 -3
- package/.claude/skills/mastermind-skill-builder/SKILL.md +1 -1
- package/.claude/skills/mastermind-tasks/SKILL.md +5 -0
- package/.claude/skills/mastermind-techport/SKILL.md +1 -1
- package/.claude/skills/monodesign/scripts/detector/engines/browser/drivers.mjs +56 -15
- package/.claude/skills/performance-analysis/SKILL.md +1 -1
- package/.claude/skills/verification-quality/SKILL.md +2 -3
- package/README.md +2 -2
- package/dist/src/commands/doc.js +2 -2
- package/dist/src/commands/doc.js.map +1 -1
- package/dist/src/commands/doctor-project-checks.d.ts.map +1 -1
- package/dist/src/commands/doctor-project-checks.js +45 -2
- package/dist/src/commands/doctor-project-checks.js.map +1 -1
- package/dist/src/commands/memory-crud.d.ts.map +1 -1
- package/dist/src/commands/memory-crud.js +13 -2
- package/dist/src/commands/memory-crud.js.map +1 -1
- package/dist/src/commands/memory.js +1 -1
- package/dist/src/commands/memory.js.map +1 -1
- package/dist/src/commands/monograph.d.ts.map +1 -1
- package/dist/src/commands/monograph.js +11 -4
- package/dist/src/commands/monograph.js.map +1 -1
- package/dist/src/commands/org-observe.d.ts.map +1 -1
- package/dist/src/commands/org-observe.js +56 -6
- package/dist/src/commands/org-observe.js.map +1 -1
- package/dist/src/commands/org.d.ts +26 -0
- package/dist/src/commands/org.d.ts.map +1 -1
- package/dist/src/commands/org.js +144 -28
- package/dist/src/commands/org.js.map +1 -1
- package/dist/src/init/executor.d.ts.map +1 -1
- package/dist/src/init/executor.js +10 -9
- package/dist/src/init/executor.js.map +1 -1
- package/dist/src/init/settings-generator.d.ts.map +1 -1
- package/dist/src/init/settings-generator.js.map +1 -1
- package/dist/src/init/upgrade.d.ts.map +1 -1
- package/dist/src/init/upgrade.js +29 -0
- package/dist/src/init/upgrade.js.map +1 -1
- package/dist/src/init/write-codex.d.ts.map +1 -1
- package/dist/src/init/write-codex.js.map +1 -1
- package/dist/src/knowledge/document-pipeline.d.ts +5 -0
- package/dist/src/knowledge/document-pipeline.d.ts.map +1 -1
- package/dist/src/knowledge/document-pipeline.js +32 -16
- package/dist/src/knowledge/document-pipeline.js.map +1 -1
- package/dist/src/mcp-tools/hooks-routing.d.ts +9 -0
- package/dist/src/mcp-tools/hooks-routing.d.ts.map +1 -1
- package/dist/src/mcp-tools/hooks-routing.js +12 -1
- package/dist/src/mcp-tools/hooks-routing.js.map +1 -1
- package/dist/src/mcp-tools/knowledge-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/knowledge-tools.js +105 -13
- package/dist/src/mcp-tools/knowledge-tools.js.map +1 -1
- package/dist/src/mcp-tools/memory-tools.d.ts +13 -0
- package/dist/src/mcp-tools/memory-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/memory-tools.js +257 -31
- package/dist/src/mcp-tools/memory-tools.js.map +1 -1
- package/dist/src/mcp-tools/monograph/health-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/monograph/health-tools.js +99 -42
- package/dist/src/mcp-tools/monograph/health-tools.js.map +1 -1
- package/dist/src/mcp-tools/monograph/impact-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/monograph/impact-tools.js +123 -51
- package/dist/src/mcp-tools/monograph/impact-tools.js.map +1 -1
- package/dist/src/mcp-tools/monograph/query-tools.d.ts.map +1 -1
- package/dist/src/mcp-tools/monograph/query-tools.js +113 -96
- package/dist/src/mcp-tools/monograph/query-tools.js.map +1 -1
- package/dist/src/mcp-tools/monograph/shared.d.ts +37 -4
- package/dist/src/mcp-tools/monograph/shared.d.ts.map +1 -1
- package/dist/src/mcp-tools/monograph/shared.js +75 -56
- package/dist/src/mcp-tools/monograph/shared.js.map +1 -1
- package/dist/src/memory/memory-bridge.d.ts +60 -1
- package/dist/src/memory/memory-bridge.d.ts.map +1 -1
- package/dist/src/memory/memory-bridge.js +172 -42
- package/dist/src/memory/memory-bridge.js.map +1 -1
- package/dist/src/memory/memory-kg.d.ts +495 -28
- package/dist/src/memory/memory-kg.d.ts.map +1 -1
- package/dist/src/memory/memory-kg.js +2186 -251
- package/dist/src/memory/memory-kg.js.map +1 -1
- package/dist/src/memory/query-router.d.ts +51 -0
- package/dist/src/memory/query-router.d.ts.map +1 -1
- package/dist/src/memory/query-router.js +38 -2
- package/dist/src/memory/query-router.js.map +1 -1
- package/dist/src/orgrt/agent-exec.d.ts.map +1 -1
- package/dist/src/orgrt/agent-exec.js +7 -0
- package/dist/src/orgrt/agent-exec.js.map +1 -1
- package/dist/src/orgrt/agent-runner.d.ts +26 -1
- package/dist/src/orgrt/agent-runner.d.ts.map +1 -1
- package/dist/src/orgrt/agent-runner.js +73 -22
- package/dist/src/orgrt/agent-runner.js.map +1 -1
- package/dist/src/orgrt/antigravity-runner.d.ts +1 -1
- package/dist/src/orgrt/antigravity-runner.d.ts.map +1 -1
- package/dist/src/orgrt/antigravity-runner.js +36 -19
- package/dist/src/orgrt/antigravity-runner.js.map +1 -1
- package/dist/src/orgrt/approvals.d.ts +26 -2
- package/dist/src/orgrt/approvals.d.ts.map +1 -1
- package/dist/src/orgrt/approvals.js +67 -7
- package/dist/src/orgrt/approvals.js.map +1 -1
- package/dist/src/orgrt/broker.d.ts +21 -2
- package/dist/src/orgrt/broker.d.ts.map +1 -1
- package/dist/src/orgrt/broker.js +56 -8
- package/dist/src/orgrt/broker.js.map +1 -1
- package/dist/src/orgrt/bus.d.ts +8 -0
- package/dist/src/orgrt/bus.d.ts.map +1 -1
- package/dist/src/orgrt/bus.js +27 -0
- package/dist/src/orgrt/bus.js.map +1 -1
- package/dist/src/orgrt/checkpoint-ops.d.ts.map +1 -1
- package/dist/src/orgrt/checkpoint-ops.js +15 -5
- package/dist/src/orgrt/checkpoint-ops.js.map +1 -1
- package/dist/src/orgrt/checkpoint.d.ts +42 -4
- package/dist/src/orgrt/checkpoint.d.ts.map +1 -1
- package/dist/src/orgrt/checkpoint.js +75 -4
- package/dist/src/orgrt/checkpoint.js.map +1 -1
- package/dist/src/orgrt/codex-runner.d.ts +1 -1
- package/dist/src/orgrt/codex-runner.d.ts.map +1 -1
- package/dist/src/orgrt/codex-runner.js +50 -23
- package/dist/src/orgrt/codex-runner.js.map +1 -1
- package/dist/src/orgrt/copilot-runner.d.ts +24 -2
- package/dist/src/orgrt/copilot-runner.d.ts.map +1 -1
- package/dist/src/orgrt/copilot-runner.js +257 -157
- package/dist/src/orgrt/copilot-runner.js.map +1 -1
- package/dist/src/orgrt/cross-org.d.ts +10 -3
- package/dist/src/orgrt/cross-org.d.ts.map +1 -1
- package/dist/src/orgrt/cross-org.js +225 -10
- package/dist/src/orgrt/cross-org.js.map +1 -1
- package/dist/src/orgrt/crush-runner.d.ts +22 -2
- package/dist/src/orgrt/crush-runner.d.ts.map +1 -1
- package/dist/src/orgrt/crush-runner.js +227 -120
- package/dist/src/orgrt/crush-runner.js.map +1 -1
- package/dist/src/orgrt/daemon.d.ts +77 -7
- package/dist/src/orgrt/daemon.d.ts.map +1 -1
- package/dist/src/orgrt/daemon.js +989 -385
- package/dist/src/orgrt/daemon.js.map +1 -1
- package/dist/src/orgrt/decisions.d.ts +2 -2
- package/dist/src/orgrt/decisions.d.ts.map +1 -1
- package/dist/src/orgrt/decisions.js +90 -18
- package/dist/src/orgrt/decisions.js.map +1 -1
- package/dist/src/orgrt/grok-runner.d.ts +29 -3
- package/dist/src/orgrt/grok-runner.d.ts.map +1 -1
- package/dist/src/orgrt/grok-runner.js +286 -150
- package/dist/src/orgrt/grok-runner.js.map +1 -1
- package/dist/src/orgrt/kimicode-runner.d.ts +5 -5
- package/dist/src/orgrt/kimicode-runner.d.ts.map +1 -1
- package/dist/src/orgrt/kimicode-runner.js +53 -32
- package/dist/src/orgrt/kimicode-runner.js.map +1 -1
- package/dist/src/orgrt/mailbox.d.ts +15 -0
- package/dist/src/orgrt/mailbox.d.ts.map +1 -1
- package/dist/src/orgrt/mailbox.js +29 -1
- package/dist/src/orgrt/mailbox.js.map +1 -1
- package/dist/src/orgrt/migrate.d.ts.map +1 -1
- package/dist/src/orgrt/migrate.js +8 -5
- package/dist/src/orgrt/migrate.js.map +1 -1
- package/dist/src/orgrt/opencode-runner.d.ts +1 -1
- package/dist/src/orgrt/opencode-runner.d.ts.map +1 -1
- package/dist/src/orgrt/opencode-runner.js +29 -3
- package/dist/src/orgrt/opencode-runner.js.map +1 -1
- package/dist/src/orgrt/org-memory.d.ts +19 -3
- package/dist/src/orgrt/org-memory.d.ts.map +1 -1
- package/dist/src/orgrt/org-memory.js +110 -38
- package/dist/src/orgrt/org-memory.js.map +1 -1
- package/dist/src/orgrt/pi-rpc-runner.d.ts +3 -1
- package/dist/src/orgrt/pi-rpc-runner.d.ts.map +1 -1
- package/dist/src/orgrt/pi-rpc-runner.js +35 -3
- package/dist/src/orgrt/pi-rpc-runner.js.map +1 -1
- package/dist/src/orgrt/pi-runner.d.ts +25 -2
- package/dist/src/orgrt/pi-runner.d.ts.map +1 -1
- package/dist/src/orgrt/pi-runner.js +271 -152
- package/dist/src/orgrt/pi-runner.js.map +1 -1
- package/dist/src/orgrt/policy.d.ts +1 -0
- package/dist/src/orgrt/policy.d.ts.map +1 -1
- package/dist/src/orgrt/policy.js +199 -39
- package/dist/src/orgrt/policy.js.map +1 -1
- package/dist/src/orgrt/provider.d.ts +4 -0
- package/dist/src/orgrt/provider.d.ts.map +1 -1
- package/dist/src/orgrt/provider.js +16 -0
- package/dist/src/orgrt/provider.js.map +1 -1
- package/dist/src/orgrt/qwen-rpc-runner.d.ts +3 -1
- package/dist/src/orgrt/qwen-rpc-runner.d.ts.map +1 -1
- package/dist/src/orgrt/qwen-rpc-runner.js +35 -3
- package/dist/src/orgrt/qwen-rpc-runner.js.map +1 -1
- package/dist/src/orgrt/qwen-runner.d.ts +30 -3
- package/dist/src/orgrt/qwen-runner.d.ts.map +1 -1
- package/dist/src/orgrt/qwen-runner.js +282 -142
- package/dist/src/orgrt/qwen-runner.js.map +1 -1
- package/dist/src/orgrt/role-slot.d.ts +88 -0
- package/dist/src/orgrt/role-slot.d.ts.map +1 -0
- package/dist/src/orgrt/role-slot.js +133 -0
- package/dist/src/orgrt/role-slot.js.map +1 -0
- package/dist/src/orgrt/runtime-options.d.ts +17 -0
- package/dist/src/orgrt/runtime-options.d.ts.map +1 -0
- package/dist/src/orgrt/runtime-options.js +32 -0
- package/dist/src/orgrt/runtime-options.js.map +1 -0
- package/dist/src/orgrt/scheduler-integration.d.ts +11 -0
- package/dist/src/orgrt/scheduler-integration.d.ts.map +1 -1
- package/dist/src/orgrt/scheduler-integration.js +75 -17
- package/dist/src/orgrt/scheduler-integration.js.map +1 -1
- package/dist/src/orgrt/scheduler.d.ts +4 -0
- package/dist/src/orgrt/scheduler.d.ts.map +1 -1
- package/dist/src/orgrt/scheduler.js +7 -1
- package/dist/src/orgrt/scheduler.js.map +1 -1
- package/dist/src/orgrt/server.d.ts +8 -3
- package/dist/src/orgrt/server.d.ts.map +1 -1
- package/dist/src/orgrt/server.js +48 -13
- package/dist/src/orgrt/server.js.map +1 -1
- package/dist/src/orgrt/session.d.ts +26 -2
- package/dist/src/orgrt/session.d.ts.map +1 -1
- package/dist/src/orgrt/session.js +98 -4
- package/dist/src/orgrt/session.js.map +1 -1
- package/dist/src/orgrt/task-dag.d.ts +5 -0
- package/dist/src/orgrt/task-dag.d.ts.map +1 -1
- package/dist/src/orgrt/task-dag.js +41 -0
- package/dist/src/orgrt/task-dag.js.map +1 -1
- package/dist/src/orgrt/test-loop.js +2 -2
- package/dist/src/orgrt/test-loop.js.map +1 -1
- package/dist/src/orgrt/types.d.ts +14 -2
- package/dist/src/orgrt/types.d.ts.map +1 -1
- package/dist/src/orgrt/types.js +15 -1
- package/dist/src/orgrt/types.js.map +1 -1
- package/dist/src/orgrt/vercel-runner.d.ts.map +1 -1
- package/dist/src/orgrt/vercel-runner.js +4 -0
- package/dist/src/orgrt/vercel-runner.js.map +1 -1
- package/dist/src/ui/routes-org.mjs +27 -7
- package/dist/src/ui/server.mjs +82 -48
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/package.json +3 -2
- package/dist/src/ui/data/mastermind-sessions.json +0 -1
package/dist/src/orgrt/daemon.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
// packages/@monomind/cli/src/orgrt/daemon.ts
|
|
2
2
|
// monolean: single-process inter-org — upgrade path = daemon-to-daemon HTTP when multi-host is real
|
|
3
3
|
import { execFileSync } from 'node:child_process';
|
|
4
|
+
import { randomUUID } from 'node:crypto';
|
|
4
5
|
import { existsSync, mkdirSync, readFileSync, unlinkSync } from 'node:fs';
|
|
5
6
|
import { isAbsolute, join } from 'node:path';
|
|
6
7
|
import { writeJsonFileAtomic } from '../utils/json-file.js';
|
|
@@ -10,7 +11,7 @@ import { AntigravityAgentRunner } from './antigravity-runner.js';
|
|
|
10
11
|
import * as approvalOps from './approvals.js';
|
|
11
12
|
import { BrokerLease, normalizeCredential } from './broker.js';
|
|
12
13
|
import { OrgBus } from './bus.js';
|
|
13
|
-
import { captureCheckpoint, generateChecksum, isCheckpointExpired, restoreMailboxQueue, validateCheckpoint, } from './checkpoint.js';
|
|
14
|
+
import { captureCheckpoint, generateChecksum, isCheckpointExpired, migrateCheckpoint, restoreMailboxQueue, validateCheckpoint, } from './checkpoint.js';
|
|
14
15
|
import * as checkpointOps from './checkpoint-ops.js';
|
|
15
16
|
import { CodexAgentRunner } from './codex-runner.js';
|
|
16
17
|
import { CopilotAgentRunner } from './copilot-runner.js';
|
|
@@ -20,7 +21,7 @@ import * as decisionOps from './decisions.js';
|
|
|
20
21
|
import { createFenceForRole, loadGlobalFenceConfig, mergeFenceConfigs, } from './fence.js';
|
|
21
22
|
import { attachForwarder } from './forwarder.js';
|
|
22
23
|
import { GrokAgentRunner } from './grok-runner.js';
|
|
23
|
-
import { drainInbox } from './inbox.js';
|
|
24
|
+
import { drainInbox, queueMessage } from './inbox.js';
|
|
24
25
|
import { KimiCodeAgentRunner } from './kimicode-runner.js';
|
|
25
26
|
import { isRecoverableCloseReason, Mailbox } from './mailbox.js';
|
|
26
27
|
import { OpencodeAgentRunner } from './opencode-runner.js';
|
|
@@ -28,9 +29,12 @@ import * as orgMemory from './org-memory.js';
|
|
|
28
29
|
import { PiRpcAgentRunner } from './pi-rpc-runner.js';
|
|
29
30
|
import { PiAgentRunner } from './pi-runner.js';
|
|
30
31
|
import { PolicyEngine } from './policy.js';
|
|
32
|
+
import { resolveRoleProvider } from './provider.js';
|
|
31
33
|
import * as questionOps from './questions.js';
|
|
32
34
|
import { QwenRpcAgentRunner } from './qwen-rpc-runner.js';
|
|
33
35
|
import { QwenAgentRunner } from './qwen-runner.js';
|
|
36
|
+
import { buildRespawnReceipt, computeReplacementBudget, mergeEffectiveRoleConfig, redactRoleConfig, validateRespawnInput, } from './role-slot.js';
|
|
37
|
+
import { buildRuntimeOptions } from './runtime-options.js';
|
|
34
38
|
import { historyFile, readHistory, readRunEvents, summarizeRun, } from './reporting.js';
|
|
35
39
|
import * as scheduler from './scheduler-integration.js';
|
|
36
40
|
import { runAgentSession } from './session.js';
|
|
@@ -183,6 +187,17 @@ export class ScrollbackBuffer {
|
|
|
183
187
|
this.lines.length = 0;
|
|
184
188
|
}
|
|
185
189
|
}
|
|
190
|
+
/** Bug 4: number of roles for this org that are actually spawned and running
|
|
191
|
+
* right now — the live count run_config.max_concurrent_agents caps. A role
|
|
192
|
+
* that crashed or ended no longer counts, so a slot frees up automatically
|
|
193
|
+
* the moment that happens; nothing needs to explicitly decrement a counter. */
|
|
194
|
+
export function activeRoleCount(org) {
|
|
195
|
+
let n = 0;
|
|
196
|
+
for (const rt of org.agents.values())
|
|
197
|
+
if (rt.status === 'running')
|
|
198
|
+
n++;
|
|
199
|
+
return n;
|
|
200
|
+
}
|
|
186
201
|
export class OrgDaemon {
|
|
187
202
|
root;
|
|
188
203
|
opts;
|
|
@@ -193,6 +208,12 @@ export class OrgDaemon {
|
|
|
193
208
|
/** @internal */ forwarders = new Map();
|
|
194
209
|
/** @internal */ watchdogs = new Map();
|
|
195
210
|
/** @internal */ stopping = new Map();
|
|
211
|
+
/** @internal Bug 2 (TOCTOU race): names currently reserved by an in-flight
|
|
212
|
+
* startOrg() call, from the synchronous existence check through
|
|
213
|
+
* registration in `orgs`. Closes the window where two concurrent
|
|
214
|
+
* startOrg(name) calls could both pass the `orgs.has(name)` check before
|
|
215
|
+
* either registered and spawn duplicate runs. */
|
|
216
|
+
/** @internal */ startingOrgs = new Set();
|
|
196
217
|
/** @internal */ approvals = new Map();
|
|
197
218
|
/** @internal */ approvalLocks = new Map();
|
|
198
219
|
/** @internal */ gatesLocks = new Map();
|
|
@@ -215,10 +236,10 @@ export class OrgDaemon {
|
|
|
215
236
|
this.opts = opts;
|
|
216
237
|
}
|
|
217
238
|
/** Publish this daemon's inbox so orgs started AFTER this call register with the broker. */
|
|
218
|
-
setInboxUrl(url,
|
|
239
|
+
setInboxUrl(url, operatorCredential) {
|
|
219
240
|
this.opts.inboxUrl = url;
|
|
220
|
-
if (
|
|
221
|
-
this.opts.
|
|
241
|
+
if (operatorCredential !== undefined)
|
|
242
|
+
this.opts.operatorCredential = operatorCredential;
|
|
222
243
|
}
|
|
223
244
|
/** subscribe to events from ALL running orgs (dashboard server uses this) */
|
|
224
245
|
subscribe(fn) {
|
|
@@ -333,8 +354,48 @@ export class OrgDaemon {
|
|
|
333
354
|
const inflightStop = this.stopping.get(name);
|
|
334
355
|
if (inflightStop)
|
|
335
356
|
await inflightStop;
|
|
357
|
+
// Bug 2 (TOCTOU race): this existence check is synchronous, but the real
|
|
358
|
+
// registration into `this.orgs` doesn't happen until deep inside
|
|
359
|
+
// startOrgInner, after several genuine `await` points (the provider
|
|
360
|
+
// validation dynamic import, `git worktree add` for workspace:
|
|
361
|
+
// 'worktree'). Two concurrent startOrg(name) calls — e.g. the
|
|
362
|
+
// scheduler's tick, the runfile poll loop, and autoWake firing close
|
|
363
|
+
// together — could each pass this check before either registered,
|
|
364
|
+
// spawning two duplicate runs with separate budget/policy counters that
|
|
365
|
+
// both write the same shared per-org files. Reserve the name in
|
|
366
|
+
// `startingOrgs` synchronously, in the same tick as the check, so a
|
|
367
|
+
// second concurrent call sees the reservation and is rejected instead of
|
|
368
|
+
// racing ahead to spawn a duplicate.
|
|
336
369
|
if (this.orgs.has(name))
|
|
337
370
|
throw new Error(`org ${name} already running`);
|
|
371
|
+
if (this.startingOrgs.has(name))
|
|
372
|
+
throw new Error(`org ${name} already starting`);
|
|
373
|
+
this.startingOrgs.add(name);
|
|
374
|
+
try {
|
|
375
|
+
return await this.startOrgInner(name, taskOverride, options);
|
|
376
|
+
}
|
|
377
|
+
catch (err) {
|
|
378
|
+
// startOrgInner registers the org in `this.orgs` (and spawns the boss,
|
|
379
|
+
// installs the exit listener, starts the broker lease) well before it
|
|
380
|
+
// returns; persistState (ENOSPC/EACCES) and BrokerLease.start() can
|
|
381
|
+
// still throw after that. Left alone, that was a live, unreachable org:
|
|
382
|
+
// sessions running, `this.orgs` still holding it, every later startOrg
|
|
383
|
+
// rejected with "already running", and nothing ever calling stopOrg.
|
|
384
|
+
// Only this call can have registered the name (the reservation above
|
|
385
|
+
// holds until `finally`), so anything in the map is ours to tear down.
|
|
386
|
+
if (this.orgs.has(name)) {
|
|
387
|
+
await this.stopOrg(name).catch((stopErr) => console.error(`org ${name}: teardown after failed start failed:`, stopErr instanceof Error ? stopErr.message : stopErr));
|
|
388
|
+
}
|
|
389
|
+
throw err;
|
|
390
|
+
}
|
|
391
|
+
finally {
|
|
392
|
+
this.startingOrgs.delete(name);
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
/** The actual startOrg implementation. Split out of startOrg() so the
|
|
396
|
+
* reservation guard above runs synchronously, before any `await` in here —
|
|
397
|
+
* see the bug 2 comment in startOrg(). */
|
|
398
|
+
async startOrgInner(name, taskOverride, options) {
|
|
338
399
|
const defPath = join(this.root, ORG_DIR, `${name}.json`);
|
|
339
400
|
const def = OrgDefSchema.parse(JSON.parse(readFileSync(defPath, 'utf8')));
|
|
340
401
|
let run;
|
|
@@ -348,8 +409,13 @@ export class OrgDaemon {
|
|
|
348
409
|
throw new Error(`cannot resume org "${name}": no valid checkpoint found`);
|
|
349
410
|
if (isCheckpointExpired(rt.checkpoint))
|
|
350
411
|
throw new Error(`cannot resume org "${name}": checkpoint expired`);
|
|
351
|
-
|
|
412
|
+
// Migrate an older-schema checkpoint (verifying ITS OWN stored checksum
|
|
413
|
+
// first) before validating it against CHECKPOINT_VERSION — see
|
|
414
|
+
// migrateCheckpoint's doc comment in checkpoint.ts.
|
|
415
|
+
const migrated = migrateCheckpoint(rt.checkpoint);
|
|
416
|
+
if (!migrated || !validateCheckpoint(migrated))
|
|
352
417
|
throw new Error(`cannot resume org "${name}": checkpoint validation failed`);
|
|
418
|
+
rt.checkpoint = migrated;
|
|
353
419
|
run = rt.run;
|
|
354
420
|
checkpoint = rt.checkpoint;
|
|
355
421
|
if (rt.abandonedRoles) {
|
|
@@ -476,6 +542,10 @@ export class OrgDaemon {
|
|
|
476
542
|
// boss that's genuinely out of ideas can loop forever making zero
|
|
477
543
|
// progress without ever tripping the watchdog.
|
|
478
544
|
let lastToolActivity = 0;
|
|
545
|
+
// Org-wide budget ceiling (bug 1): tracks whether run_config.budget_tokens
|
|
546
|
+
// has already been enforced this run, so the close-all-mailboxes sweep
|
|
547
|
+
// below only fires once instead of on every subsequent usage event.
|
|
548
|
+
let orgBudgetClosed = false;
|
|
479
549
|
bus.subscribe((e) => {
|
|
480
550
|
const slim = e.data?.content != null ? { ...e, data: { ...e.data, content: undefined } } : e;
|
|
481
551
|
collected.push(slim);
|
|
@@ -512,6 +582,42 @@ export class OrgDaemon {
|
|
|
512
582
|
}
|
|
513
583
|
}
|
|
514
584
|
}
|
|
585
|
+
// Bug 1: run_config.budget_tokens is an org-wide ceiling, not just a
|
|
586
|
+
// per-role one — a role's explicit budget_tokens override lets IT spend
|
|
587
|
+
// more without raising what every other role can spend, so nothing
|
|
588
|
+
// upstream of this ever summed real usage across the whole roster and
|
|
589
|
+
// stopped the org when the declared total was reached. Mirror the
|
|
590
|
+
// per-role budget-exhausted handling (session.ts's mailbox.close('token-budget'))
|
|
591
|
+
// at the org level: once the sum of every role's PolicyEngine.usage
|
|
592
|
+
// reaches the ceiling, close every mailbox and stop lazy-spawning new
|
|
593
|
+
// ones so the org can't keep spending past its declared cap.
|
|
594
|
+
if (e.type === 'usage' && !orgBudgetClosed) {
|
|
595
|
+
const orgBudget = def.run_config.budget_tokens;
|
|
596
|
+
if (orgBudget != null) {
|
|
597
|
+
let orgUsage = 0;
|
|
598
|
+
for (const rt of running.agents.values())
|
|
599
|
+
orgUsage += rt.policy.usage;
|
|
600
|
+
// Mid-run role replacement retires a policy engine's usage into the
|
|
601
|
+
// slot instead of discarding it (see role-slot.ts / respawnRole) -
|
|
602
|
+
// include it here or a replacement could silently reset spend and
|
|
603
|
+
// let the org exceed its declared ceiling.
|
|
604
|
+
for (const slot of running.roleSlots.values())
|
|
605
|
+
orgUsage += slot.retiredUsage.tokens;
|
|
606
|
+
if (orgUsage >= orgBudget) {
|
|
607
|
+
orgBudgetClosed = true;
|
|
608
|
+
running.pendingRoles?.clear(); // prevent lazy spawns after the org budget is exhausted
|
|
609
|
+
for (const rt of running.agents.values()) {
|
|
610
|
+
if (!rt.mailbox.isClosed)
|
|
611
|
+
rt.mailbox.close('token-budget');
|
|
612
|
+
}
|
|
613
|
+
bus.emit({
|
|
614
|
+
type: 'status',
|
|
615
|
+
reason: 'org-budget-exhausted',
|
|
616
|
+
msg: `org-wide token budget exhausted (${orgUsage}/${orgBudget}) — closing all roles`,
|
|
617
|
+
});
|
|
618
|
+
}
|
|
619
|
+
}
|
|
620
|
+
}
|
|
515
621
|
// Track last message ID for threading responses
|
|
516
622
|
if ((e.type === 'message' || e.type === 'xorg') && e.from) {
|
|
517
623
|
const runtime = running.agents.get(e.from);
|
|
@@ -537,8 +643,13 @@ export class OrgDaemon {
|
|
|
537
643
|
run,
|
|
538
644
|
bus,
|
|
539
645
|
agents: new Map(),
|
|
646
|
+
roleSlots: new Map(),
|
|
647
|
+
bossRoleId: '', // set below, once bossRole is computed
|
|
648
|
+
glossary: [],
|
|
649
|
+
respawning: new Set(),
|
|
540
650
|
busEvents: () => [...collected],
|
|
541
651
|
workdir: cwd,
|
|
652
|
+
credential: randomUUID(),
|
|
542
653
|
};
|
|
543
654
|
this.orgs.set(name, running);
|
|
544
655
|
// ── MonoFence guardrail: pre-create per-role instances ────────────────
|
|
@@ -569,402 +680,72 @@ export class OrgDaemon {
|
|
|
569
680
|
if (roleFences.size > 0)
|
|
570
681
|
running.fences = roleFences;
|
|
571
682
|
// Even-split budget; a role's own budget_tokens overrides it (roleTokenBudget).
|
|
572
|
-
|
|
683
|
+
// Bug 1: roles WITH an explicit override spend on top of the even split
|
|
684
|
+
// rather than out of it, so the roster's ceilings could sum to well over
|
|
685
|
+
// the declared org-wide budget (e.g. 4 roles @ 250k + one role overridden
|
|
686
|
+
// to 2M = 2.75M achievable against a declared 1M cap). Subtract the sum of
|
|
687
|
+
// every role's explicit override from the org-wide budget first, then
|
|
688
|
+
// split only the remainder among the roles WITHOUT an override, so the
|
|
689
|
+
// static split is honest about what's left. (Live usage is still tracked
|
|
690
|
+
// and enforced as a real ceiling above, independent of this static split.)
|
|
691
|
+
const orgBudgetTokens = def.run_config.budget_tokens ?? 1_000_000;
|
|
692
|
+
const overriddenTokenSum = def.roles.reduce((sum, r) => sum + (r.budget_tokens ?? 0), 0);
|
|
693
|
+
const unoverriddenRoleCount = def.roles.filter((r) => r.budget_tokens == null).length;
|
|
694
|
+
const perRoleBudget = unoverriddenRoleCount > 0
|
|
695
|
+
? Math.max(0, Math.floor((orgBudgetTokens - overriddenTokenSum) / unoverriddenRoleCount))
|
|
696
|
+
: 0;
|
|
573
697
|
// Single boss-selection rule for kickoff AND org_complete gating — the
|
|
574
698
|
// session layer previously keyed the tool on reports_to===null while the
|
|
575
699
|
// kickoff went to (type==='boss' || reports_to===null || roles[0]), so a
|
|
576
700
|
// fallback-selected boss could be told to call org_complete without having
|
|
577
701
|
// the tool.
|
|
578
702
|
const bossRole = def.roles.find((r) => r.type === 'boss' || r.reports_to === null) ?? def.roles[0];
|
|
579
|
-
|
|
703
|
+
running.bossRoleId = bossRole.id;
|
|
704
|
+
// Canonical entity names from THIS org's KG — injected into the coordinator
|
|
580
705
|
// prompt so org_learn extractions reuse them instead of minting duplicates.
|
|
706
|
+
// Scoped: an unscoped glossary handed every org's entity names to every
|
|
707
|
+
// coordinator, which is how one org's claims got merged into another's.
|
|
581
708
|
const glossary = await (async () => {
|
|
582
709
|
try {
|
|
583
710
|
if (!(await this.orgMemoryUsable()))
|
|
584
711
|
return [];
|
|
585
712
|
const kg = await import('../memory/memory-kg.js');
|
|
586
|
-
return await kg.kgGlossary({
|
|
713
|
+
return await kg.kgGlossary({
|
|
714
|
+
dbPath: this.orgMemoryDbPath(),
|
|
715
|
+
scope: orgMemory.orgKgScope(name),
|
|
716
|
+
});
|
|
587
717
|
}
|
|
588
718
|
catch {
|
|
589
719
|
return [];
|
|
590
720
|
}
|
|
591
721
|
})();
|
|
722
|
+
running.glossary = glossary;
|
|
592
723
|
// Resource-gated staggered spawn: check memory/process limits before each
|
|
593
724
|
// NON-BOSS agent, wait if under pressure. The boss always spawns immediately
|
|
594
725
|
// and ungated — the org has no coordinator at all without it, so gating it
|
|
595
726
|
// behind host memory pressure would make the whole org fail to start over a
|
|
596
727
|
// condition workers are specifically designed to ride out.
|
|
597
|
-
const _limits = getResourceLimits();
|
|
598
728
|
// Extracted so a role that fails its gate check can be spawned later by
|
|
599
729
|
// scheduleDeferredSpawn() once resources free up, without re-running the
|
|
600
730
|
// gate logic or duplicating the session-wiring below.
|
|
601
731
|
const spawnRole = (role, roleCheckpoint) => {
|
|
602
732
|
if (running.agents.has(role.id))
|
|
603
733
|
return;
|
|
604
|
-
|
|
605
|
-
if (ws === 'worktree-per-role' && role.id !== bossRole.id) {
|
|
606
|
-
const wtPath = join(this.root, ORG_DIR, name, `worktree-${role.id}`);
|
|
607
|
-
try {
|
|
608
|
-
// Q7: top-level `import { execFileSync }` replaces the inlined
|
|
609
|
-
// `require('node:child_process')` that broke ESM at runtime —
|
|
610
|
-
// vitest's CJS shim masked it in tests but the built package
|
|
611
|
-
// threw "require is not defined" in real Node ESM execution.
|
|
612
|
-
// SEC-5: argv-array form, no shell.
|
|
613
|
-
if (existsSync(wtPath)) {
|
|
614
|
-
try {
|
|
615
|
-
execFileSync('git', ['worktree', 'remove', '--force', wtPath], {
|
|
616
|
-
cwd: this.root,
|
|
617
|
-
stdio: 'ignore',
|
|
618
|
-
timeout: 30_000,
|
|
619
|
-
});
|
|
620
|
-
}
|
|
621
|
-
catch {
|
|
622
|
-
/* best-effort */
|
|
623
|
-
}
|
|
624
|
-
}
|
|
625
|
-
execFileSync('git', ['worktree', 'add', wtPath, 'HEAD', '--detach'], {
|
|
626
|
-
cwd: this.root,
|
|
627
|
-
stdio: 'ignore',
|
|
628
|
-
timeout: 30_000,
|
|
629
|
-
});
|
|
630
|
-
roleCwd = wtPath;
|
|
631
|
-
}
|
|
632
|
-
catch {
|
|
633
|
-
/* fallback to shared cwd if git worktree fails */
|
|
634
|
-
}
|
|
635
|
-
}
|
|
636
|
-
const mailbox = new Mailbox();
|
|
637
|
-
if (roleCheckpoint?.mailboxQueue?.length) {
|
|
638
|
-
restoreMailboxQueue({ mailbox }, roleCheckpoint.mailboxQueue);
|
|
639
|
-
}
|
|
640
|
-
// A recoverable close (budget exhaustion) is left open on resume — see
|
|
641
|
-
// isRecoverableCloseReason's doc comment. Re-closing it here would
|
|
642
|
-
// make the idle watchdog's "raise the budget and resume" remedy a
|
|
643
|
-
// no-op, since nothing in this codebase ever reopens a closed mailbox.
|
|
644
|
-
if (roleCheckpoint?.mailboxClosed &&
|
|
645
|
-
!isRecoverableCloseReason(roleCheckpoint.mailboxCloseReason)) {
|
|
646
|
-
mailbox.close(roleCheckpoint.mailboxCloseReason);
|
|
647
|
-
}
|
|
648
|
-
const policy = new PolicyEngine(role.id, {
|
|
649
|
-
maxTokens: role.budget_tokens ?? perRoleBudget,
|
|
650
|
-
maxUsd: role.budget_usd,
|
|
651
|
-
...(role.policy ?? {}),
|
|
652
|
-
}, bus, roleCwd);
|
|
653
|
-
if (roleCheckpoint?.tokensUsed) {
|
|
654
|
-
policy.setUsage(roleCheckpoint.tokensUsed);
|
|
655
|
-
}
|
|
656
|
-
// ORG-7: restore accumulated USD spend across resume so a stop/resume
|
|
657
|
-
// cycle can't reset a role's USD budget back to zero.
|
|
658
|
-
if (roleCheckpoint?.costUsd) {
|
|
659
|
-
policy.setUsageUsd(roleCheckpoint.costUsd);
|
|
660
|
-
}
|
|
661
|
-
const runtime = {
|
|
662
|
-
mailbox,
|
|
663
|
-
policy,
|
|
664
|
-
status: roleCheckpoint?.status ?? 'running',
|
|
665
|
-
done: Promise.resolve(),
|
|
666
|
-
metrics: { tokens: roleCheckpoint?.tokensUsed ?? 0, costUsd: roleCheckpoint?.costUsd ?? 0 },
|
|
667
|
-
lastMessageId: roleCheckpoint?.lastMessageId,
|
|
668
|
-
error: roleCheckpoint?.error,
|
|
669
|
-
sessionId: roleCheckpoint?.sessionId,
|
|
670
|
-
worktreePath: roleCwd !== cwd ? roleCwd : undefined,
|
|
671
|
-
scrollback: new ScrollbackBuffer(),
|
|
672
|
-
};
|
|
673
|
-
if (roleCheckpoint?.scrollback?.length) {
|
|
674
|
-
for (const line of roleCheckpoint.scrollback)
|
|
675
|
-
runtime.scrollback.push(line);
|
|
676
|
-
}
|
|
677
|
-
const sessionOpts = {
|
|
678
|
-
org: name,
|
|
679
|
-
role,
|
|
680
|
-
bus,
|
|
681
|
-
policy,
|
|
682
|
-
mailbox,
|
|
683
|
-
cwd: roleCwd,
|
|
684
|
-
def,
|
|
685
|
-
// Pass the org state directory so runners that persist per-role state
|
|
686
|
-
// (VercelAgentRunner session files) write under .monomind/orgs/<name>
|
|
687
|
-
// instead of polluting the workspace cwd.
|
|
688
|
-
orgDir: join(this.root, ORG_DIR, name),
|
|
689
|
-
// Project root for named-provider (`adapter_config.provider`) config
|
|
690
|
-
// lookup — role cwd may be an isolated workspace with no config file.
|
|
691
|
-
orgRoot: this.root,
|
|
692
|
-
maxTurns: role.max_turns_per_message ?? def.run_config.max_turns_per_message,
|
|
693
|
-
resumeSessionId: roleCheckpoint?.sessionId,
|
|
694
|
-
lastMessageId: () => runtime.lastMessageId,
|
|
695
|
-
onOutput: (line) => runtime.scrollback.push(line),
|
|
696
|
-
onSessionId: (id) => {
|
|
697
|
-
runtime.sessionId = id;
|
|
698
|
-
},
|
|
699
|
-
deliver: (from, to, subject, body) => this.deliver(name, from, to, subject, body),
|
|
700
|
-
askHuman: (r, question) => this.askHuman(name, r, question),
|
|
701
|
-
onGate: (r, gateName, gateDesc) => this.createGate(name, r, gateName, gateDesc),
|
|
702
|
-
circuitBreaker: (() => {
|
|
703
|
-
const cb = def.run_config.circuit_breaker;
|
|
704
|
-
if (!cb)
|
|
705
|
-
return undefined;
|
|
706
|
-
return { threshold: cb.failure_threshold ?? 5, state: { failures: 0, tripped: false } };
|
|
707
|
-
})(),
|
|
708
|
-
beforeTool: (r, toolName) => this.checkApproval(name, r, toolName),
|
|
709
|
-
fence: roleFences.get(role.id),
|
|
710
|
-
// ORG-1: gatedCanUseTool denials are a natural decision point — record them so
|
|
711
|
-
// `org decisions` shows real traces instead of always reporting none.
|
|
712
|
-
onDecision: (r, toolName, message) => {
|
|
713
|
-
this.recordDecision(name, r, {
|
|
714
|
-
type: 'tool',
|
|
715
|
-
context: `tool call: ${toolName}`,
|
|
716
|
-
reasoning: message,
|
|
717
|
-
outcome: 'denied',
|
|
718
|
-
});
|
|
719
|
-
},
|
|
720
|
-
// ORG-9: decision gates are documented as "hard-blocking" — make that
|
|
721
|
-
// true by actually denying tool use while this role has a pending gate,
|
|
722
|
-
// the same way pending approvals already do.
|
|
723
|
-
hasPendingGate: () => this.listGates(name, 'pending').some((g) => g.roleId === role.id),
|
|
724
|
-
onComplete: role.id === bossRole.id
|
|
725
|
-
? (r, outcome, summary) => {
|
|
726
|
-
bus.emit({
|
|
727
|
-
type: 'status',
|
|
728
|
-
from: r,
|
|
729
|
-
reason: 'org-complete',
|
|
730
|
-
msg: `run outcome: ${outcome}`,
|
|
731
|
-
data: { outcome, summary },
|
|
732
|
-
});
|
|
733
|
-
}
|
|
734
|
-
: undefined,
|
|
735
|
-
// #11: a boss that overflows its context window isn't a crash (it keeps
|
|
736
|
-
// returning +0-token errors forever), so without this the idle watchdog
|
|
737
|
-
// just nudges it for ~30 min before idle-stopping. Restart the whole org
|
|
738
|
-
// with fresh sessions instead — bounded by MAX_BOSS_RESTARTS.
|
|
739
|
-
onContextLimit: role.id === bossRole.id ? () => this.scheduleBossRestart(name) : undefined,
|
|
740
|
-
recall: async (r, q) => {
|
|
741
|
-
const answer = await this.recallOrgMemory(name, def, q, r);
|
|
742
|
-
bus.emit({
|
|
743
|
-
type: 'status',
|
|
744
|
-
from: r,
|
|
745
|
-
reason: 'org-recall',
|
|
746
|
-
msg: `recall: ${q.slice(0, 80)}`,
|
|
747
|
-
data: { hits: answer.hits },
|
|
748
|
-
});
|
|
749
|
-
return answer.text;
|
|
750
|
-
},
|
|
751
|
-
searchKnowledge: async (r, q) => {
|
|
752
|
-
const answer = await this.searchProjectKnowledge(q);
|
|
753
|
-
bus.emit({
|
|
754
|
-
type: 'status',
|
|
755
|
-
from: r,
|
|
756
|
-
reason: 'knowledge-search',
|
|
757
|
-
msg: `knowledge: ${q.slice(0, 80)}`,
|
|
758
|
-
data: { hits: answer.hits },
|
|
759
|
-
});
|
|
760
|
-
return answer.text;
|
|
761
|
-
},
|
|
762
|
-
glossary,
|
|
763
|
-
remember: async (r, content, scope) => {
|
|
764
|
-
const text = await this.rememberOrgMemory(name, def, r, content, scope, run);
|
|
765
|
-
bus.emit({
|
|
766
|
-
type: 'status',
|
|
767
|
-
from: r,
|
|
768
|
-
reason: 'org-remember',
|
|
769
|
-
msg: `remember (${scope}): ${content.slice(0, 80)}`,
|
|
770
|
-
data: { scope },
|
|
771
|
-
});
|
|
772
|
-
return text;
|
|
773
|
-
},
|
|
774
|
-
learn: async (r, payload) => {
|
|
775
|
-
const text = await this.learnOrgKnowledge(name, run, payload);
|
|
776
|
-
bus.emit({
|
|
777
|
-
type: 'status',
|
|
778
|
-
from: r,
|
|
779
|
-
reason: 'org-learn',
|
|
780
|
-
msg: `learn: ${text.slice(0, 120)}`,
|
|
781
|
-
data: {
|
|
782
|
-
nodes: payload.nodes?.length ?? 0,
|
|
783
|
-
edges: payload.edges?.length ?? 0,
|
|
784
|
-
rules: payload.rules?.length ?? 0,
|
|
785
|
-
},
|
|
786
|
-
});
|
|
787
|
-
return text;
|
|
788
|
-
},
|
|
789
|
-
createTask: (r, title, assignee, deps) => {
|
|
790
|
-
return this.dagCreateTask(name, r, title, assignee, deps);
|
|
791
|
-
},
|
|
792
|
-
completeTask: (r, taskId, result) => {
|
|
793
|
-
return this.dagCompleteTask(name, r, taskId, result);
|
|
794
|
-
},
|
|
795
|
-
listTasks: () => {
|
|
796
|
-
const running = this.orgs.get(name);
|
|
797
|
-
return JSON.stringify(running?.taskDag?.all() ?? [], null, 2);
|
|
798
|
-
},
|
|
799
|
-
splitTask: (r, parentId, children) => {
|
|
800
|
-
return this.dagSplitTask(name, r, parentId, children);
|
|
801
|
-
},
|
|
802
|
-
mergeTask: (r, sourceId, targetId) => {
|
|
803
|
-
return this.dagMergeTask(name, r, sourceId, targetId);
|
|
804
|
-
},
|
|
805
|
-
cancelTask: (r, taskId, reason) => {
|
|
806
|
-
return this.dagCancelTask(name, r, taskId, reason);
|
|
807
|
-
},
|
|
808
|
-
blockTask: (r, taskId, untilIso, reason) => {
|
|
809
|
-
return this.dagBlockTask(name, r, taskId, untilIso, reason);
|
|
810
|
-
},
|
|
811
|
-
planGraph: (r, specs) => {
|
|
812
|
-
return this.dagPlanGraph(name, r, specs);
|
|
813
|
-
},
|
|
814
|
-
queryFn: this.opts.queryFn,
|
|
815
|
-
// Runner resolution: explicit opts.runner > role `runtime` field >
|
|
816
|
-
// org def `runtime` field > MONOMIND_RUNTIME env (opencode/kimicode) >
|
|
817
|
-
// undefined (session.ts falls back to ClaudeAgentRunner via queryFn).
|
|
818
|
-
// Leaving it undefined for the default path is what keeps
|
|
819
|
-
// Claude/Antigravity orgs byte-for-byte unchanged. Session opts are
|
|
820
|
-
// built per role here in spawnRole, so each role gets its own runner.
|
|
821
|
-
runner: this.opts.runner ??
|
|
822
|
-
resolveRoleRunner(role.runtime, def.runtime, role.provider?.kind, undefined, role.provider),
|
|
823
|
-
};
|
|
824
|
-
// Supervised session: transient crashes (provider blips, network) restart
|
|
825
|
-
// with backoff; a crash with the mailbox already closed, or one that
|
|
826
|
-
// exhausts the retry budget, is terminal. runAgentSession already emits a
|
|
827
|
-
// 'status' event for the raw error; the terminal 'audit' event is for
|
|
828
|
-
// dashboards/alerts that filter on actionable failures (not routine
|
|
829
|
-
// status chatter) so a dead agent surfaces instead of a run that
|
|
830
|
-
// silently never progresses.
|
|
831
|
-
const BACKOFFS_MS = this.opts.crashBackoffsMs ?? [1000, 5000, 15000];
|
|
832
|
-
if (!mailbox.isClosed && runtime.status !== 'crashed') {
|
|
833
|
-
runtime.done = (async () => {
|
|
834
|
-
for (let attempt = 0;; attempt++) {
|
|
835
|
-
try {
|
|
836
|
-
await runAgentSession(sessionOpts);
|
|
837
|
-
runtime.status = 'ended';
|
|
838
|
-
return;
|
|
839
|
-
}
|
|
840
|
-
catch (err) {
|
|
841
|
-
// Drop the crashed session's stale waker immediately: a push()
|
|
842
|
-
// during the backoff window must queue for the NEXT session, not
|
|
843
|
-
// wake the dead generator to swallow it.
|
|
844
|
-
mailbox.detach();
|
|
845
|
-
// #203: if the crashed session's mailbox generator was abandoned
|
|
846
|
-
// mid-yield (message already shift()ed for it, turn never
|
|
847
|
-
// finished), put that message back on the queue — otherwise the
|
|
848
|
-
// replacement session's stream() finds an empty queue and parks
|
|
849
|
-
// forever, since the "delivered" message is gone for good.
|
|
850
|
-
mailbox.reclaimInFlight();
|
|
851
|
-
const message = err instanceof Error ? err.message : String(err);
|
|
852
|
-
const isTurnLimit = /Reached maximum number of turns|error_max_turns/i.test(message);
|
|
853
|
-
// Bounded like every other recovery: attempt counts every pass
|
|
854
|
-
// through this loop, so a role that keeps surfacing max-turns
|
|
855
|
-
// errors here (session.ts already swallows the normal ones)
|
|
856
|
-
// falls through to crash handling instead of looping forever.
|
|
857
|
-
if (isTurnLimit && !mailbox.isClosed && attempt < BACKOFFS_MS.length) {
|
|
858
|
-
sessionOpts.resumeSessionId = undefined;
|
|
859
|
-
mailbox.push(`${Mailbox.CONTINUE_PREFIX} You reached the turn limit on your task. Continue your in-progress work from where you left off; if finished, end your turn.`);
|
|
860
|
-
bus.emit({
|
|
861
|
-
type: 'status',
|
|
862
|
-
from: role.id,
|
|
863
|
-
reason: 'turn-limit-recover',
|
|
864
|
-
msg: `agent "${role.id}" hit turn limit error — continuing with fresh session`,
|
|
865
|
-
});
|
|
866
|
-
continue;
|
|
867
|
-
}
|
|
868
|
-
// Exit 143 = SIGTERM. If the mailbox is already closed, we
|
|
869
|
-
// sent the signal ourselves during stop — not a crash.
|
|
870
|
-
const killedByStop = mailbox.isClosed && /exit(?:ed)? with code 143/.test(message);
|
|
871
|
-
const crash = () => {
|
|
872
|
-
if (killedByStop) {
|
|
873
|
-
runtime.status = 'ended';
|
|
874
|
-
bus.emit({
|
|
875
|
-
type: 'status',
|
|
876
|
-
from: role.id,
|
|
877
|
-
msg: `agent "${role.id}" terminated by stop (was still working when drain window expired)`,
|
|
878
|
-
reason: 'terminated-by-stop',
|
|
879
|
-
});
|
|
880
|
-
return;
|
|
881
|
-
}
|
|
882
|
-
runtime.status = 'crashed';
|
|
883
|
-
runtime.error = message;
|
|
884
|
-
// Close the mailbox so deliver()/receiveRemote() report a real
|
|
885
|
-
// error instead of pushing into a queue no session will read
|
|
886
|
-
// (and returning a false "delivered" receipt to the sender).
|
|
887
|
-
mailbox.close();
|
|
888
|
-
const isContextLimit = OrgDaemon.CONTEXT_LIMIT_RE.test(message);
|
|
889
|
-
bus.emit({
|
|
890
|
-
type: 'audit',
|
|
891
|
-
from: role.id,
|
|
892
|
-
msg: `agent "${role.id}" crashed: ${message}`,
|
|
893
|
-
reason: isContextLimit ? 'agent-context-limit' : 'agent-session-crash',
|
|
894
|
-
data: {
|
|
895
|
-
agentId: role.id,
|
|
896
|
-
error: message,
|
|
897
|
-
restarts: attempt,
|
|
898
|
-
contextLimit: isContextLimit,
|
|
899
|
-
},
|
|
900
|
-
});
|
|
901
|
-
if (role.id !== bossRole.id) {
|
|
902
|
-
// #2/#3: a worker is gone for the rest of this run. Without this
|
|
903
|
-
// notice the coordinator keeps messaging a corpse (observed: four
|
|
904
|
-
// unanswered org_send calls to a developer that had crashed on a
|
|
905
|
-
// context-window limit). Tell the boss to reassign — and if the
|
|
906
|
-
// crash was a context overflow, tell it to chunk smaller, since
|
|
907
|
-
// re-dispatching the same task verbatim fails the same way.
|
|
908
|
-
const bossRt = running.agents.get(bossRole.id);
|
|
909
|
-
if (bossRt && !bossRt.mailbox.isClosed) {
|
|
910
|
-
const guidance = isContextLimit
|
|
911
|
-
? ' This was a context-window overflow — re-dispatching the same task verbatim will fail identically. Break the work into smaller pieces (one file or section at a time) and do not paste large file contents in a single message.'
|
|
912
|
-
: '';
|
|
913
|
-
bossRt.mailbox.push(`[system] Worker "${role.id}" crashed and will not recover this run (${message}). It can no longer receive messages — stop messaging it. Reassign its outstanding work to another agent or take it on yourself.${guidance}`);
|
|
914
|
-
bus.emit({
|
|
915
|
-
type: 'audit',
|
|
916
|
-
from: bossRole.id,
|
|
917
|
-
reason: 'worker-crashed',
|
|
918
|
-
msg: `worker "${role.id}" crashed (contextLimit=${isContextLimit}); coordinator notified to reassign`,
|
|
919
|
-
});
|
|
920
|
-
}
|
|
921
|
-
}
|
|
922
|
-
else {
|
|
923
|
-
// #4: the coordinator itself died. Don't go silent and wait for a
|
|
924
|
-
// human — attempt a bounded whole-org restart with fresh sessions
|
|
925
|
-
// (which also sheds whatever bloated context caused the crash).
|
|
926
|
-
this.scheduleBossRestart(name);
|
|
927
|
-
}
|
|
928
|
-
};
|
|
929
|
-
// Fatal errors (provider auth/quota/billing — tagged with
|
|
930
|
-
// err.fatal by the runner) can NEVER be fixed by a restart: the
|
|
931
|
-
// same call fails identically or hangs. Skip the backoff loop
|
|
932
|
-
// and go straight to terminal crash handling instead of burning
|
|
933
|
-
// the retry budget and wall-clock on a guaranteed failure.
|
|
934
|
-
const fatal = err?.fatal === true;
|
|
935
|
-
if (fatal) {
|
|
936
|
-
bus.emit({
|
|
937
|
-
type: 'status',
|
|
938
|
-
from: role.id,
|
|
939
|
-
reason: 'agent-fatal',
|
|
940
|
-
msg: `agent "${role.id}" hit a fatal (non-retryable) error — not restarting`,
|
|
941
|
-
});
|
|
942
|
-
crash();
|
|
943
|
-
return;
|
|
944
|
-
}
|
|
945
|
-
if (mailbox.isClosed || attempt >= BACKOFFS_MS.length) {
|
|
946
|
-
crash();
|
|
947
|
-
return;
|
|
948
|
-
}
|
|
949
|
-
bus.emit({
|
|
950
|
-
type: 'status',
|
|
951
|
-
from: role.id,
|
|
952
|
-
reason: 'agent-restart',
|
|
953
|
-
msg: `agent "${role.id}" crashed (${message}) — restarting in ${BACKOFFS_MS[attempt]}ms (attempt ${attempt + 1}/${BACKOFFS_MS.length})`,
|
|
954
|
-
});
|
|
955
|
-
await new Promise((r) => {
|
|
956
|
-
const t = setTimeout(r, BACKOFFS_MS[attempt]);
|
|
957
|
-
t.unref?.();
|
|
958
|
-
});
|
|
959
|
-
if (mailbox.isClosed) {
|
|
960
|
-
crash();
|
|
961
|
-
return;
|
|
962
|
-
} // org stopped during backoff — never recovered
|
|
963
|
-
}
|
|
964
|
-
}
|
|
965
|
-
})();
|
|
966
|
-
}
|
|
734
|
+
const { runtime, abort } = this.spawnRoleIncarnation(name, running, role, roleCheckpoint?.generation ?? 0, { roleCheckpoint });
|
|
967
735
|
running.agents.set(role.id, runtime);
|
|
736
|
+
running.roleSlots.set(role.id, {
|
|
737
|
+
generation: roleCheckpoint?.generation ?? 0,
|
|
738
|
+
phase: 'running',
|
|
739
|
+
runtime,
|
|
740
|
+
abort,
|
|
741
|
+
effectiveRole: roleCheckpoint?.effectiveRoleOverrides &&
|
|
742
|
+
Object.keys(roleCheckpoint.effectiveRoleOverrides).length > 0
|
|
743
|
+
? mergeEffectiveRoleConfig(role, roleCheckpoint.effectiveRoleOverrides)
|
|
744
|
+
: role,
|
|
745
|
+
respawnCount: roleCheckpoint?.respawnCount ?? 0,
|
|
746
|
+
queuedDuringSwap: roleCheckpoint?.queuedDuringSwap ?? [],
|
|
747
|
+
retiredUsage: roleCheckpoint?.retiredUsage ?? { tokens: 0, costUsd: 0 },
|
|
748
|
+
});
|
|
968
749
|
};
|
|
969
750
|
if (options?.resume && checkpoint) {
|
|
970
751
|
const restoredRoles = new Set(Object.keys(checkpoint.roleState));
|
|
@@ -981,7 +762,23 @@ export class OrgDaemon {
|
|
|
981
762
|
}
|
|
982
763
|
running.pendingRoles = pendingRoles;
|
|
983
764
|
running.spawnRole = spawnRole;
|
|
984
|
-
running.taskDag =
|
|
765
|
+
running.taskDag =
|
|
766
|
+
checkpoint.tasks && checkpoint.tasks.length > 0
|
|
767
|
+
? TaskDag.fromJSON(checkpoint.tasks)
|
|
768
|
+
: new TaskDag();
|
|
769
|
+
// A 'running' task's "[task:…]" message was consumed by the session that
|
|
770
|
+
// was working it. If that role's SDK session is resumed (checkpointed
|
|
771
|
+
// sessionId) the task is still in its context; otherwise — role not
|
|
772
|
+
// restored at all, or restored into a fresh session — nothing knows
|
|
773
|
+
// about the task, so put it back to 'ready' and re-dispatch.
|
|
774
|
+
for (const task of running.taskDag.all()) {
|
|
775
|
+
if (task.status !== 'running')
|
|
776
|
+
continue;
|
|
777
|
+
if (checkpoint.roleState[task.assignee]?.sessionId)
|
|
778
|
+
continue;
|
|
779
|
+
running.taskDag.requeue(task.id);
|
|
780
|
+
}
|
|
781
|
+
decisionOps.dispatchReadyTasks(this, name, running);
|
|
985
782
|
if (worktreePath)
|
|
986
783
|
running.worktreePath = worktreePath;
|
|
987
784
|
}
|
|
@@ -1103,6 +900,18 @@ export class OrgDaemon {
|
|
|
1103
900
|
const pendingGates = this.readGates(name).gates.filter((g) => g.status === 'pending');
|
|
1104
901
|
if (pendingGates.length > 0)
|
|
1105
902
|
return;
|
|
903
|
+
// Bug 3: a pending ask_human question is the same kind of legitimate
|
|
904
|
+
// wait as a pending gate — askHuman()'s receipt tells the role to end
|
|
905
|
+
// its turn and wait for the resolution, so a role that follows that
|
|
906
|
+
// instruction and goes quiet looks identical to a genuinely stalled
|
|
907
|
+
// agent. Without this check the watchdog nudges (and, after enough
|
|
908
|
+
// nudges, idle-stops) an org that's simply waiting on a human answer
|
|
909
|
+
// that's already on its way.
|
|
910
|
+
const pendingQuestions = questionOps
|
|
911
|
+
.readQuestions(this.root, name)
|
|
912
|
+
.questions.filter((q) => q.answer === null);
|
|
913
|
+
if (pendingQuestions.length > 0)
|
|
914
|
+
return;
|
|
1106
915
|
// Auto-resume any task whose org_task_block time has passed: flip it
|
|
1107
916
|
// back to 'running' and re-push it into the assignee's mailbox, same
|
|
1108
917
|
// as a fresh dispatch. This IS real activity, so fall through to the
|
|
@@ -1177,7 +986,8 @@ export class OrgDaemon {
|
|
|
1177
986
|
this.watchdogs.set(name, wd);
|
|
1178
987
|
}
|
|
1179
988
|
if (this.opts.crossProcess && this.opts.inboxUrl) {
|
|
1180
|
-
const
|
|
989
|
+
const operatorCred = normalizeCredential(this.opts.operatorCredential);
|
|
990
|
+
const lease = new BrokerLease(name, this.opts.inboxUrl, this.opts.brokerDir, undefined, running.credential, operatorCred ? { credential: operatorCred, dir: this.opts.operatorDir } : undefined);
|
|
1181
991
|
lease.start();
|
|
1182
992
|
this.leases.set(name, lease);
|
|
1183
993
|
}
|
|
@@ -1191,8 +1001,19 @@ export class OrgDaemon {
|
|
|
1191
1001
|
// queueMessage had already reported them accepted.
|
|
1192
1002
|
if (!running.agents.has(msg.toRole) && running.pendingRoles?.has(msg.toRole)) {
|
|
1193
1003
|
const pending = running.pendingRoles.get(msg.toRole);
|
|
1194
|
-
|
|
1195
|
-
|
|
1004
|
+
// Bug 4: don't spawn past run_config.max_concurrent_agents. Requeue
|
|
1005
|
+
// this message (queueMessage, not a silent drop) and defer the spawn
|
|
1006
|
+
// the same way a concurrency-gated lazy spawn defers elsewhere.
|
|
1007
|
+
const concurrencyLimit = def.run_config.max_concurrent_agents;
|
|
1008
|
+
if (concurrencyLimit != null && activeRoleCount(running) >= concurrencyLimit) {
|
|
1009
|
+
running.pendingRoles.delete(msg.toRole);
|
|
1010
|
+
queueMessage(this.root, name, msg);
|
|
1011
|
+
this.scheduleConcurrencyDeferredSpawn(name, running, pending, running.spawnRole);
|
|
1012
|
+
}
|
|
1013
|
+
else {
|
|
1014
|
+
running.pendingRoles.delete(msg.toRole);
|
|
1015
|
+
running.spawnRole?.(pending);
|
|
1016
|
+
}
|
|
1196
1017
|
}
|
|
1197
1018
|
const agent = running.agents.get(msg.toRole);
|
|
1198
1019
|
if (agent && !agent.mailbox.isClosed) {
|
|
@@ -1203,13 +1024,751 @@ export class OrgDaemon {
|
|
|
1203
1024
|
subject: msg.subject,
|
|
1204
1025
|
msg: msg.body,
|
|
1205
1026
|
});
|
|
1206
|
-
|
|
1027
|
+
await crossOrg.pushMessage(this, name, running, msg.toRole, msg.fromQualified, msg.subject, msg.body, `inbox-${msg.ts}-${Math.random().toString(36).slice(2, 8)}`);
|
|
1207
1028
|
}
|
|
1208
1029
|
}
|
|
1209
1030
|
if (queued.length)
|
|
1210
1031
|
bus.emit({ type: 'status', msg: `drained ${queued.length} queued message(s) from inbox` });
|
|
1211
1032
|
return running;
|
|
1212
1033
|
}
|
|
1034
|
+
/** Build one role incarnation: mailbox, policy, AgentRuntime, sessionOpts,
|
|
1035
|
+
* and the supervised crash-retry loop. Used by BOTH the startup lazy-spawn
|
|
1036
|
+
* path (generation 0, via the `spawnRole` closure inside startOrgInner)
|
|
1037
|
+
* and respawnRole() (generation N+1). Does not touch running.agents or
|
|
1038
|
+
* running.roleSlots — callers publish the result themselves. */
|
|
1039
|
+
spawnRoleIncarnation(name, running, role, generation, opts = {}) {
|
|
1040
|
+
const { roleCheckpoint } = opts;
|
|
1041
|
+
const abort = opts.abort ?? new AbortController();
|
|
1042
|
+
const { def, bus, run } = running;
|
|
1043
|
+
const cwd = running.workdir;
|
|
1044
|
+
const ws = this.workspaceSetting(def);
|
|
1045
|
+
const perRoleBudget = opts.budgetTokensOverride ?? computeReplacementBudget(def, role.id);
|
|
1046
|
+
let roleCwd = cwd;
|
|
1047
|
+
const existingSlot = running.roleSlots.get(role.id);
|
|
1048
|
+
if (ws === 'worktree-per-role' && role.id !== running.bossRoleId) {
|
|
1049
|
+
const wtPath = join(this.root, ORG_DIR, name, `worktree-${role.id}`);
|
|
1050
|
+
if (existingSlot?.runtime?.worktreePath === wtPath && existsSync(wtPath)) {
|
|
1051
|
+
// A replacement (generation > 0) reuses the SAME worktree path —
|
|
1052
|
+
// recreating it here would delete any uncommitted work the old
|
|
1053
|
+
// incarnation left behind (design constraint #5).
|
|
1054
|
+
roleCwd = wtPath;
|
|
1055
|
+
}
|
|
1056
|
+
else {
|
|
1057
|
+
try {
|
|
1058
|
+
// Q7: top-level `import { execFileSync }` replaces the inlined
|
|
1059
|
+
// `require('node:child_process')` that broke ESM at runtime —
|
|
1060
|
+
// vitest's CJS shim masked it in tests but the built package
|
|
1061
|
+
// threw "require is not defined" in real Node ESM execution.
|
|
1062
|
+
// SEC-5: argv-array form, no shell.
|
|
1063
|
+
if (existsSync(wtPath)) {
|
|
1064
|
+
try {
|
|
1065
|
+
execFileSync('git', ['worktree', 'remove', '--force', wtPath], {
|
|
1066
|
+
cwd: this.root,
|
|
1067
|
+
stdio: 'ignore',
|
|
1068
|
+
timeout: 30_000,
|
|
1069
|
+
});
|
|
1070
|
+
}
|
|
1071
|
+
catch {
|
|
1072
|
+
/* best-effort */
|
|
1073
|
+
}
|
|
1074
|
+
}
|
|
1075
|
+
execFileSync('git', ['worktree', 'add', wtPath, 'HEAD', '--detach'], {
|
|
1076
|
+
cwd: this.root,
|
|
1077
|
+
stdio: 'ignore',
|
|
1078
|
+
timeout: 30_000,
|
|
1079
|
+
});
|
|
1080
|
+
roleCwd = wtPath;
|
|
1081
|
+
}
|
|
1082
|
+
catch {
|
|
1083
|
+
/* fallback to shared cwd if git worktree fails */
|
|
1084
|
+
}
|
|
1085
|
+
}
|
|
1086
|
+
}
|
|
1087
|
+
const mailbox = new Mailbox();
|
|
1088
|
+
if (roleCheckpoint?.mailboxQueue?.length) {
|
|
1089
|
+
restoreMailboxQueue({ mailbox }, roleCheckpoint.mailboxQueue);
|
|
1090
|
+
}
|
|
1091
|
+
// A recoverable close (budget exhaustion) is left open on resume — see
|
|
1092
|
+
// isRecoverableCloseReason's doc comment. Re-closing it here would
|
|
1093
|
+
// make the idle watchdog's "raise the budget and resume" remedy a
|
|
1094
|
+
// no-op, since nothing in this codebase ever reopens a closed mailbox.
|
|
1095
|
+
if (roleCheckpoint?.mailboxClosed &&
|
|
1096
|
+
!isRecoverableCloseReason(roleCheckpoint.mailboxCloseReason)) {
|
|
1097
|
+
mailbox.close(roleCheckpoint.mailboxCloseReason);
|
|
1098
|
+
}
|
|
1099
|
+
const policy = new PolicyEngine(role.id, {
|
|
1100
|
+
maxTokens: role.budget_tokens ?? perRoleBudget,
|
|
1101
|
+
maxUsd: role.budget_usd,
|
|
1102
|
+
...(role.policy ?? {}),
|
|
1103
|
+
}, bus, roleCwd);
|
|
1104
|
+
if (roleCheckpoint?.tokensUsed) {
|
|
1105
|
+
policy.setUsage(roleCheckpoint.tokensUsed);
|
|
1106
|
+
}
|
|
1107
|
+
// ORG-7: restore accumulated USD spend across resume so a stop/resume
|
|
1108
|
+
// cycle can't reset a role's USD budget back to zero.
|
|
1109
|
+
if (roleCheckpoint?.costUsd) {
|
|
1110
|
+
policy.setUsageUsd(roleCheckpoint.costUsd);
|
|
1111
|
+
}
|
|
1112
|
+
const runtime = {
|
|
1113
|
+
mailbox,
|
|
1114
|
+
policy,
|
|
1115
|
+
status: roleCheckpoint?.status ?? 'running',
|
|
1116
|
+
done: Promise.resolve(),
|
|
1117
|
+
metrics: { tokens: roleCheckpoint?.tokensUsed ?? 0, costUsd: roleCheckpoint?.costUsd ?? 0 },
|
|
1118
|
+
lastMessageId: roleCheckpoint?.lastMessageId,
|
|
1119
|
+
error: roleCheckpoint?.error,
|
|
1120
|
+
sessionId: roleCheckpoint?.sessionId,
|
|
1121
|
+
worktreePath: roleCwd !== cwd ? roleCwd : undefined,
|
|
1122
|
+
scrollback: new ScrollbackBuffer(),
|
|
1123
|
+
};
|
|
1124
|
+
if (roleCheckpoint?.scrollback?.length) {
|
|
1125
|
+
for (const line of roleCheckpoint.scrollback)
|
|
1126
|
+
runtime.scrollback.push(line);
|
|
1127
|
+
}
|
|
1128
|
+
const sessionOpts = {
|
|
1129
|
+
org: name,
|
|
1130
|
+
role,
|
|
1131
|
+
bus,
|
|
1132
|
+
policy,
|
|
1133
|
+
mailbox,
|
|
1134
|
+
cwd: roleCwd,
|
|
1135
|
+
def,
|
|
1136
|
+
// Pass the org state directory so runners that persist per-role state
|
|
1137
|
+
// (VercelAgentRunner session files) write under .monomind/orgs/<name>
|
|
1138
|
+
// instead of polluting the workspace cwd.
|
|
1139
|
+
orgDir: join(this.root, ORG_DIR, name),
|
|
1140
|
+
// Project root for named-provider (`adapter_config.provider`) config
|
|
1141
|
+
// lookup — role cwd may be an isolated workspace with no config file.
|
|
1142
|
+
orgRoot: this.root,
|
|
1143
|
+
maxTurns: role.max_turns_per_message ?? def.run_config.max_turns_per_message,
|
|
1144
|
+
resumeSessionId: roleCheckpoint?.sessionId,
|
|
1145
|
+
lastMessageId: () => runtime.lastMessageId,
|
|
1146
|
+
onOutput: (line) => runtime.scrollback.push(line),
|
|
1147
|
+
onSessionId: (id) => {
|
|
1148
|
+
runtime.sessionId = id;
|
|
1149
|
+
},
|
|
1150
|
+
deliver: (from, to, subject, body) => this.deliver(name, from, to, subject, body),
|
|
1151
|
+
askHuman: (r, question) => this.askHuman(name, r, question),
|
|
1152
|
+
onGate: (r, gateName, gateDesc) => this.createGate(name, r, gateName, gateDesc),
|
|
1153
|
+
circuitBreaker: (() => {
|
|
1154
|
+
const cb = def.run_config.circuit_breaker;
|
|
1155
|
+
if (!cb)
|
|
1156
|
+
return undefined;
|
|
1157
|
+
return { threshold: cb.failure_threshold ?? 5, state: { failures: 0, tripped: false } };
|
|
1158
|
+
})(),
|
|
1159
|
+
beforeTool: (r, toolName, input) => this.checkApproval(name, r, toolName, input),
|
|
1160
|
+
fence: running.fences?.get(role.id),
|
|
1161
|
+
// ORG-1: gatedCanUseTool denials are a natural decision point — record them so
|
|
1162
|
+
// `org decisions` shows real traces instead of always reporting none.
|
|
1163
|
+
onDecision: (r, toolName, message) => {
|
|
1164
|
+
this.recordDecision(name, r, {
|
|
1165
|
+
type: 'tool',
|
|
1166
|
+
context: `tool call: ${toolName}`,
|
|
1167
|
+
reasoning: message,
|
|
1168
|
+
outcome: 'denied',
|
|
1169
|
+
});
|
|
1170
|
+
},
|
|
1171
|
+
// ORG-9: decision gates are documented as "hard-blocking" — make that
|
|
1172
|
+
// true by actually denying tool use while this role has a pending gate,
|
|
1173
|
+
// the same way pending approvals already do.
|
|
1174
|
+
hasPendingGate: () => this.listGates(name, 'pending').some((g) => g.roleId === role.id),
|
|
1175
|
+
onComplete: role.id === running.bossRoleId
|
|
1176
|
+
? (r, outcome, summary) => {
|
|
1177
|
+
bus.emit({
|
|
1178
|
+
type: 'status',
|
|
1179
|
+
from: r,
|
|
1180
|
+
reason: 'org-complete',
|
|
1181
|
+
msg: `run outcome: ${outcome}`,
|
|
1182
|
+
data: { outcome, summary },
|
|
1183
|
+
});
|
|
1184
|
+
}
|
|
1185
|
+
: undefined,
|
|
1186
|
+
// #11: a boss that overflows its context window isn't a crash (it keeps
|
|
1187
|
+
// returning +0-token errors forever), so without this the idle watchdog
|
|
1188
|
+
// just nudges it for ~30 min before idle-stopping. Restart the whole org
|
|
1189
|
+
// with fresh sessions instead — bounded by MAX_BOSS_RESTARTS.
|
|
1190
|
+
onContextLimit: role.id === running.bossRoleId ? () => this.scheduleBossRestart(name) : undefined,
|
|
1191
|
+
onListRuntimeOptions: role.id === running.bossRoleId && (def.run_config.max_role_respawns ?? 0) > 0
|
|
1192
|
+
? () => this.listRuntimeOptions()
|
|
1193
|
+
: undefined,
|
|
1194
|
+
onRespawnRole: role.id === running.bossRoleId && (def.run_config.max_role_respawns ?? 0) > 0
|
|
1195
|
+
? (callerId, args) => this.respawnRole(name, callerId, args)
|
|
1196
|
+
: undefined,
|
|
1197
|
+
recall: async (r, q) => {
|
|
1198
|
+
const answer = await this.recallOrgMemory(name, def, q, r);
|
|
1199
|
+
bus.emit({
|
|
1200
|
+
type: 'status',
|
|
1201
|
+
from: r,
|
|
1202
|
+
reason: 'org-recall',
|
|
1203
|
+
msg: `recall: ${q.slice(0, 80)}`,
|
|
1204
|
+
data: { hits: answer.hits },
|
|
1205
|
+
});
|
|
1206
|
+
return answer.text;
|
|
1207
|
+
},
|
|
1208
|
+
searchKnowledge: async (r, q) => {
|
|
1209
|
+
const answer = await this.searchProjectKnowledge(q);
|
|
1210
|
+
bus.emit({
|
|
1211
|
+
type: 'status',
|
|
1212
|
+
from: r,
|
|
1213
|
+
reason: 'knowledge-search',
|
|
1214
|
+
msg: `knowledge: ${q.slice(0, 80)}`,
|
|
1215
|
+
data: { hits: answer.hits },
|
|
1216
|
+
});
|
|
1217
|
+
return answer.text;
|
|
1218
|
+
},
|
|
1219
|
+
glossary: running.glossary,
|
|
1220
|
+
remember: async (r, content, scope) => {
|
|
1221
|
+
const text = await this.rememberOrgMemory(name, def, r, content, scope, run);
|
|
1222
|
+
bus.emit({
|
|
1223
|
+
type: 'status',
|
|
1224
|
+
from: r,
|
|
1225
|
+
reason: 'org-remember',
|
|
1226
|
+
msg: `remember (${scope}): ${content.slice(0, 80)}`,
|
|
1227
|
+
data: { scope },
|
|
1228
|
+
});
|
|
1229
|
+
return text;
|
|
1230
|
+
},
|
|
1231
|
+
learn: async (r, payload) => {
|
|
1232
|
+
const text = await this.learnOrgKnowledge(name, run, payload);
|
|
1233
|
+
bus.emit({
|
|
1234
|
+
type: 'status',
|
|
1235
|
+
from: r,
|
|
1236
|
+
reason: 'org-learn',
|
|
1237
|
+
msg: `learn: ${text.slice(0, 120)}`,
|
|
1238
|
+
data: {
|
|
1239
|
+
nodes: payload.nodes?.length ?? 0,
|
|
1240
|
+
edges: payload.edges?.length ?? 0,
|
|
1241
|
+
rules: payload.rules?.length ?? 0,
|
|
1242
|
+
},
|
|
1243
|
+
});
|
|
1244
|
+
return text;
|
|
1245
|
+
},
|
|
1246
|
+
createTask: (r, title, assignee, deps) => {
|
|
1247
|
+
return this.dagCreateTask(name, r, title, assignee, deps);
|
|
1248
|
+
},
|
|
1249
|
+
completeTask: (r, taskId, result) => {
|
|
1250
|
+
return this.dagCompleteTask(name, r, taskId, result);
|
|
1251
|
+
},
|
|
1252
|
+
listTasks: () => {
|
|
1253
|
+
const running = this.orgs.get(name);
|
|
1254
|
+
return JSON.stringify(running?.taskDag?.all() ?? [], null, 2);
|
|
1255
|
+
},
|
|
1256
|
+
splitTask: (r, parentId, children) => {
|
|
1257
|
+
return this.dagSplitTask(name, r, parentId, children);
|
|
1258
|
+
},
|
|
1259
|
+
mergeTask: (r, sourceId, targetId) => {
|
|
1260
|
+
return this.dagMergeTask(name, r, sourceId, targetId);
|
|
1261
|
+
},
|
|
1262
|
+
cancelTask: (r, taskId, reason) => {
|
|
1263
|
+
return this.dagCancelTask(name, r, taskId, reason);
|
|
1264
|
+
},
|
|
1265
|
+
blockTask: (r, taskId, untilIso, reason) => {
|
|
1266
|
+
return this.dagBlockTask(name, r, taskId, untilIso, reason);
|
|
1267
|
+
},
|
|
1268
|
+
planGraph: (r, specs) => {
|
|
1269
|
+
return this.dagPlanGraph(name, r, specs);
|
|
1270
|
+
},
|
|
1271
|
+
queryFn: this.opts.queryFn,
|
|
1272
|
+
// Runner resolution: explicit opts.runner > role `runtime` field >
|
|
1273
|
+
// org def `runtime` field > MONOMIND_RUNTIME env (opencode/kimicode) >
|
|
1274
|
+
// undefined (session.ts falls back to ClaudeAgentRunner via queryFn).
|
|
1275
|
+
// Leaving it undefined for the default path is what keeps
|
|
1276
|
+
// Claude/Antigravity orgs byte-for-byte unchanged. Session opts are
|
|
1277
|
+
// built per role here, so each role gets its own runner.
|
|
1278
|
+
runner: this.opts.runner ??
|
|
1279
|
+
resolveRoleRunner(role.runtime, def.runtime, role.provider?.kind, undefined, role.provider),
|
|
1280
|
+
// Lets respawnRole force-stop THIS specific incarnation (mid-run role
|
|
1281
|
+
// replacement's forced-stop step) without reaching into runAgentSession's
|
|
1282
|
+
// internals.
|
|
1283
|
+
externalAbort: abort,
|
|
1284
|
+
};
|
|
1285
|
+
// Supervised session: transient crashes (provider blips, network) restart
|
|
1286
|
+
// with backoff; a crash with the mailbox already closed, or one that
|
|
1287
|
+
// exhausts the retry budget, is terminal. runAgentSession already emits a
|
|
1288
|
+
// 'status' event for the raw error; the terminal 'audit' event is for
|
|
1289
|
+
// dashboards/alerts that filter on actionable failures (not routine
|
|
1290
|
+
// status chatter) so a dead agent surfaces instead of a run that
|
|
1291
|
+
// silently never progresses.
|
|
1292
|
+
const BACKOFFS_MS = this.opts.crashBackoffsMs ?? [1000, 5000, 15000];
|
|
1293
|
+
const myGeneration = generation;
|
|
1294
|
+
const isStaleGeneration = () => (running.roleSlots.get(role.id)?.generation ?? 0) !== myGeneration;
|
|
1295
|
+
if (!mailbox.isClosed && runtime.status !== 'crashed') {
|
|
1296
|
+
runtime.done = (async () => {
|
|
1297
|
+
for (let attempt = 0;; attempt++) {
|
|
1298
|
+
try {
|
|
1299
|
+
await runAgentSession(sessionOpts);
|
|
1300
|
+
runtime.status = 'ended';
|
|
1301
|
+
return;
|
|
1302
|
+
}
|
|
1303
|
+
catch (err) {
|
|
1304
|
+
// A deliberate respawn (see respawnRole) bumps the slot's
|
|
1305
|
+
// generation and force-stops this incarnation via its
|
|
1306
|
+
// externalAbort - that abort makes runAgentSession reject here
|
|
1307
|
+
// exactly like a real crash would. Recognize supersession
|
|
1308
|
+
// FIRST: this generation's retry loop must never restart,
|
|
1309
|
+
// never run terminal crash handling, and never notify the
|
|
1310
|
+
// boss - the replacement (a new generation, spawned
|
|
1311
|
+
// separately) already owns this role id.
|
|
1312
|
+
if (isStaleGeneration())
|
|
1313
|
+
return;
|
|
1314
|
+
// Drop the crashed session's stale waker immediately: a push()
|
|
1315
|
+
// during the backoff window must queue for the NEXT session, not
|
|
1316
|
+
// wake the dead generator to swallow it.
|
|
1317
|
+
mailbox.detach();
|
|
1318
|
+
// #203: if the crashed session's mailbox generator was abandoned
|
|
1319
|
+
// mid-yield (message already shift()ed for it, turn never
|
|
1320
|
+
// finished), put that message back on the queue — otherwise the
|
|
1321
|
+
// replacement session's stream() finds an empty queue and parks
|
|
1322
|
+
// forever, since the "delivered" message is gone for good.
|
|
1323
|
+
mailbox.reclaimInFlight();
|
|
1324
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
1325
|
+
const isTurnLimit = /Reached maximum number of turns|error_max_turns/i.test(message);
|
|
1326
|
+
// Bounded like every other recovery: attempt counts every pass
|
|
1327
|
+
// through this loop, so a role that keeps surfacing max-turns
|
|
1328
|
+
// errors here (session.ts already swallows the normal ones)
|
|
1329
|
+
// falls through to crash handling instead of looping forever.
|
|
1330
|
+
if (isTurnLimit && !mailbox.isClosed && attempt < BACKOFFS_MS.length) {
|
|
1331
|
+
sessionOpts.resumeSessionId = undefined;
|
|
1332
|
+
mailbox.push(`${Mailbox.CONTINUE_PREFIX} You reached the turn limit on your task. Continue your in-progress work from where you left off; if finished, end your turn.`);
|
|
1333
|
+
bus.emit({
|
|
1334
|
+
type: 'status',
|
|
1335
|
+
from: role.id,
|
|
1336
|
+
reason: 'turn-limit-recover',
|
|
1337
|
+
msg: `agent "${role.id}" hit turn limit error — continuing with fresh session`,
|
|
1338
|
+
});
|
|
1339
|
+
continue;
|
|
1340
|
+
}
|
|
1341
|
+
// Exit 143 = SIGTERM. If the mailbox is already closed, we
|
|
1342
|
+
// sent the signal ourselves during stop — not a crash.
|
|
1343
|
+
const killedByStop = mailbox.isClosed && /exit(?:ed)? with code 143/.test(message);
|
|
1344
|
+
const crash = () => {
|
|
1345
|
+
if (killedByStop) {
|
|
1346
|
+
runtime.status = 'ended';
|
|
1347
|
+
bus.emit({
|
|
1348
|
+
type: 'status',
|
|
1349
|
+
from: role.id,
|
|
1350
|
+
msg: `agent "${role.id}" terminated by stop (was still working when drain window expired)`,
|
|
1351
|
+
reason: 'terminated-by-stop',
|
|
1352
|
+
});
|
|
1353
|
+
return;
|
|
1354
|
+
}
|
|
1355
|
+
runtime.status = 'crashed';
|
|
1356
|
+
runtime.error = message;
|
|
1357
|
+
// Close the mailbox so deliver()/receiveRemote() report a real
|
|
1358
|
+
// error instead of pushing into a queue no session will read
|
|
1359
|
+
// (and returning a false "delivered" receipt to the sender).
|
|
1360
|
+
mailbox.close();
|
|
1361
|
+
const isContextLimit = OrgDaemon.CONTEXT_LIMIT_RE.test(message);
|
|
1362
|
+
bus.emit({
|
|
1363
|
+
type: 'audit',
|
|
1364
|
+
from: role.id,
|
|
1365
|
+
msg: `agent "${role.id}" crashed: ${message}`,
|
|
1366
|
+
reason: isContextLimit ? 'agent-context-limit' : 'agent-session-crash',
|
|
1367
|
+
data: {
|
|
1368
|
+
agentId: role.id,
|
|
1369
|
+
error: message,
|
|
1370
|
+
restarts: attempt,
|
|
1371
|
+
contextLimit: isContextLimit,
|
|
1372
|
+
},
|
|
1373
|
+
});
|
|
1374
|
+
if (role.id !== running.bossRoleId) {
|
|
1375
|
+
// #2/#3: a worker is gone for the rest of this run. Without this
|
|
1376
|
+
// notice the coordinator keeps messaging a corpse (observed: four
|
|
1377
|
+
// unanswered org_send calls to a developer that had crashed on a
|
|
1378
|
+
// context-window limit). Tell the boss to reassign — and if the
|
|
1379
|
+
// crash was a context overflow, tell it to chunk smaller, since
|
|
1380
|
+
// re-dispatching the same task verbatim fails the same way.
|
|
1381
|
+
const bossRt = running.agents.get(running.bossRoleId);
|
|
1382
|
+
if (bossRt && !bossRt.mailbox.isClosed) {
|
|
1383
|
+
const guidance = isContextLimit
|
|
1384
|
+
? ' This was a context-window overflow — re-dispatching the same task verbatim will fail identically. Break the work into smaller pieces (one file or section at a time) and do not paste large file contents in a single message.'
|
|
1385
|
+
: '';
|
|
1386
|
+
bossRt.mailbox.push(`[system] Worker "${role.id}" crashed and will not recover this run (${message}). It can no longer receive messages — stop messaging it. Reassign its outstanding work to another agent or take it on yourself.${guidance}`);
|
|
1387
|
+
bus.emit({
|
|
1388
|
+
type: 'audit',
|
|
1389
|
+
from: running.bossRoleId,
|
|
1390
|
+
reason: 'worker-crashed',
|
|
1391
|
+
msg: `worker "${role.id}" crashed (contextLimit=${isContextLimit}); coordinator notified to reassign`,
|
|
1392
|
+
});
|
|
1393
|
+
}
|
|
1394
|
+
}
|
|
1395
|
+
else {
|
|
1396
|
+
// #4: the coordinator itself died. Don't go silent and wait for a
|
|
1397
|
+
// human — attempt a bounded whole-org restart with fresh sessions
|
|
1398
|
+
// (which also sheds whatever bloated context caused the crash).
|
|
1399
|
+
this.scheduleBossRestart(name);
|
|
1400
|
+
}
|
|
1401
|
+
};
|
|
1402
|
+
// Fatal errors (provider auth/quota/billing — tagged with
|
|
1403
|
+
// err.fatal by the runner) can NEVER be fixed by a restart: the
|
|
1404
|
+
// same call fails identically or hangs. Skip the backoff loop
|
|
1405
|
+
// and go straight to terminal crash handling instead of burning
|
|
1406
|
+
// the retry budget and wall-clock on a guaranteed failure.
|
|
1407
|
+
const fatal = err?.fatal === true;
|
|
1408
|
+
if (fatal) {
|
|
1409
|
+
bus.emit({
|
|
1410
|
+
type: 'status',
|
|
1411
|
+
from: role.id,
|
|
1412
|
+
reason: 'agent-fatal',
|
|
1413
|
+
msg: `agent "${role.id}" hit a fatal (non-retryable) error — not restarting`,
|
|
1414
|
+
});
|
|
1415
|
+
crash();
|
|
1416
|
+
return;
|
|
1417
|
+
}
|
|
1418
|
+
if (mailbox.isClosed || attempt >= BACKOFFS_MS.length) {
|
|
1419
|
+
crash();
|
|
1420
|
+
return;
|
|
1421
|
+
}
|
|
1422
|
+
bus.emit({
|
|
1423
|
+
type: 'status',
|
|
1424
|
+
from: role.id,
|
|
1425
|
+
reason: 'agent-restart',
|
|
1426
|
+
msg: `agent "${role.id}" crashed (${message}) — restarting in ${BACKOFFS_MS[attempt]}ms (attempt ${attempt + 1}/${BACKOFFS_MS.length})`,
|
|
1427
|
+
});
|
|
1428
|
+
await new Promise((r) => {
|
|
1429
|
+
const t = setTimeout(r, BACKOFFS_MS[attempt]);
|
|
1430
|
+
t.unref?.();
|
|
1431
|
+
// Org stop (finishStop) aborts every active slot's controller —
|
|
1432
|
+
// without racing it here, this wait wouldn't notice for up to
|
|
1433
|
+
// BACKOFFS_MS[attempt] (default up to 15s), well past finishStop's
|
|
1434
|
+
// own bounded drain window. That let this loop's crash() —
|
|
1435
|
+
// and the bus.emit() it triggers — fire AFTER finishStop had
|
|
1436
|
+
// already declared the org stopped and returned, capable of
|
|
1437
|
+
// recreating files in a run directory a caller was already
|
|
1438
|
+
// deleting.
|
|
1439
|
+
if (abort.signal.aborted) {
|
|
1440
|
+
clearTimeout(t);
|
|
1441
|
+
r();
|
|
1442
|
+
return;
|
|
1443
|
+
}
|
|
1444
|
+
abort.signal.addEventListener('abort', () => {
|
|
1445
|
+
clearTimeout(t);
|
|
1446
|
+
r();
|
|
1447
|
+
}, { once: true });
|
|
1448
|
+
});
|
|
1449
|
+
if (isStaleGeneration())
|
|
1450
|
+
return; // superseded during the backoff wait
|
|
1451
|
+
if (mailbox.isClosed) {
|
|
1452
|
+
crash();
|
|
1453
|
+
return;
|
|
1454
|
+
} // org stopped during backoff — never recovered
|
|
1455
|
+
}
|
|
1456
|
+
}
|
|
1457
|
+
})();
|
|
1458
|
+
}
|
|
1459
|
+
return { runtime, abort };
|
|
1460
|
+
}
|
|
1461
|
+
/** org_respawn_role's daemon-owned implementation. See the design doc's
|
|
1462
|
+
* "Replacement algorithm" (13 steps) — this method's body follows those
|
|
1463
|
+
* steps in order, numbered in comments. */
|
|
1464
|
+
async respawnRole(name, callerId, rawInput) {
|
|
1465
|
+
const running = this.orgs.get(name);
|
|
1466
|
+
if (!running) {
|
|
1467
|
+
return {
|
|
1468
|
+
success: false,
|
|
1469
|
+
roleId: '',
|
|
1470
|
+
generation: 0,
|
|
1471
|
+
respawnCount: 0,
|
|
1472
|
+
respawnsRemaining: 0,
|
|
1473
|
+
error: `org "${name}" is not running`,
|
|
1474
|
+
};
|
|
1475
|
+
}
|
|
1476
|
+
// Step 1: authorize (defense in depth — buildOrgTools only ever wires
|
|
1477
|
+
// onRespawnRole for the selected coordinator, but re-check here too).
|
|
1478
|
+
if (callerId !== running.bossRoleId) {
|
|
1479
|
+
return {
|
|
1480
|
+
success: false,
|
|
1481
|
+
roleId: '',
|
|
1482
|
+
generation: 0,
|
|
1483
|
+
respawnCount: 0,
|
|
1484
|
+
respawnsRemaining: 0,
|
|
1485
|
+
error: 'only the selected coordinator may call org_respawn_role',
|
|
1486
|
+
};
|
|
1487
|
+
}
|
|
1488
|
+
if (this.stopping.has(name)) {
|
|
1489
|
+
return {
|
|
1490
|
+
success: false,
|
|
1491
|
+
roleId: '',
|
|
1492
|
+
generation: 0,
|
|
1493
|
+
respawnCount: 0,
|
|
1494
|
+
respawnsRemaining: 0,
|
|
1495
|
+
error: `org "${name}" is stopping`,
|
|
1496
|
+
};
|
|
1497
|
+
}
|
|
1498
|
+
const validated = validateRespawnInput(rawInput);
|
|
1499
|
+
if (!validated.ok) {
|
|
1500
|
+
return {
|
|
1501
|
+
success: false,
|
|
1502
|
+
roleId: '',
|
|
1503
|
+
generation: 0,
|
|
1504
|
+
respawnCount: 0,
|
|
1505
|
+
respawnsRemaining: 0,
|
|
1506
|
+
error: validated.error,
|
|
1507
|
+
};
|
|
1508
|
+
}
|
|
1509
|
+
const input = validated.value;
|
|
1510
|
+
if (input.roleId === running.bossRoleId) {
|
|
1511
|
+
return {
|
|
1512
|
+
success: false,
|
|
1513
|
+
roleId: input.roleId,
|
|
1514
|
+
generation: 0,
|
|
1515
|
+
respawnCount: 0,
|
|
1516
|
+
respawnsRemaining: 0,
|
|
1517
|
+
error: 'cannot replace the selected coordinator',
|
|
1518
|
+
};
|
|
1519
|
+
}
|
|
1520
|
+
const slot = running.roleSlots.get(input.roleId);
|
|
1521
|
+
if (!slot) {
|
|
1522
|
+
return {
|
|
1523
|
+
success: false,
|
|
1524
|
+
roleId: input.roleId,
|
|
1525
|
+
generation: 0,
|
|
1526
|
+
respawnCount: 0,
|
|
1527
|
+
respawnsRemaining: 0,
|
|
1528
|
+
error: `unknown or not-yet-started role "${input.roleId}"`,
|
|
1529
|
+
};
|
|
1530
|
+
}
|
|
1531
|
+
const maxRespawns = running.def.run_config.max_role_respawns ?? 0;
|
|
1532
|
+
if (slot.phase === 'removed') {
|
|
1533
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1534
|
+
roleId: input.roleId,
|
|
1535
|
+
error: `role "${input.roleId}" was removed from this org`,
|
|
1536
|
+
});
|
|
1537
|
+
}
|
|
1538
|
+
// Step 2: acquire the role slot (reject a concurrent replacement).
|
|
1539
|
+
if (running.respawning.has(input.roleId) || slot.respawnPromise) {
|
|
1540
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1541
|
+
roleId: input.roleId,
|
|
1542
|
+
error: `role "${input.roleId}" is already undergoing replacement`,
|
|
1543
|
+
});
|
|
1544
|
+
}
|
|
1545
|
+
if (slot.respawnCount >= maxRespawns) {
|
|
1546
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1547
|
+
roleId: input.roleId,
|
|
1548
|
+
error: `role "${input.roleId}" has reached its respawn limit (${slot.respawnCount}/${maxRespawns}) for this run`,
|
|
1549
|
+
});
|
|
1550
|
+
}
|
|
1551
|
+
// Step 3: resolve the candidate configuration.
|
|
1552
|
+
if (input.providerName !== undefined && slot.effectiveRole.provider) {
|
|
1553
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1554
|
+
roleId: input.roleId,
|
|
1555
|
+
error: `role "${input.roleId}" has an inline provider, which always takes precedence over adapter_config.provider — replacing an inline provider is a separate design`,
|
|
1556
|
+
});
|
|
1557
|
+
}
|
|
1558
|
+
const candidateRole = mergeEffectiveRoleConfig(slot.effectiveRole, {
|
|
1559
|
+
runtime: input.runtime,
|
|
1560
|
+
model: input.model,
|
|
1561
|
+
providerName: input.providerName,
|
|
1562
|
+
});
|
|
1563
|
+
const budgetTokens = input.budgetTokens ?? computeReplacementBudget(running.def, input.roleId);
|
|
1564
|
+
// Step 4: preflight — must not mutate the old runtime.
|
|
1565
|
+
try {
|
|
1566
|
+
resolveRoleProvider(candidateRole, this.root);
|
|
1567
|
+
}
|
|
1568
|
+
catch (err) {
|
|
1569
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1570
|
+
roleId: input.roleId,
|
|
1571
|
+
error: `preflight failed: ${err instanceof Error ? err.message : String(err)}`,
|
|
1572
|
+
});
|
|
1573
|
+
}
|
|
1574
|
+
// resolveRoleRunner's undefined return is the valid Claude default, not
|
|
1575
|
+
// an error — nothing further to validate for the runtime dimension here.
|
|
1576
|
+
resolveRoleRunner(candidateRole.runtime, running.def.runtime, candidateRole.provider?.kind, undefined, candidateRole.provider);
|
|
1577
|
+
// Step 5: consume one attempt — only after validation/preflight succeed.
|
|
1578
|
+
running.respawning.add(input.roleId);
|
|
1579
|
+
slot.respawnCount++;
|
|
1580
|
+
running.bus.emit({
|
|
1581
|
+
type: 'audit',
|
|
1582
|
+
from: callerId,
|
|
1583
|
+
reason: 'role-respawn-started',
|
|
1584
|
+
msg: `replacing role "${input.roleId}": ${input.reason}`,
|
|
1585
|
+
data: {
|
|
1586
|
+
roleId: input.roleId,
|
|
1587
|
+
from: redactRoleConfig(slot.effectiveRole),
|
|
1588
|
+
to: redactRoleConfig(candidateRole),
|
|
1589
|
+
generation: slot.generation,
|
|
1590
|
+
caller: callerId,
|
|
1591
|
+
},
|
|
1592
|
+
});
|
|
1593
|
+
// Step 6: quiesce the old incarnation. Bump the generation NOW, before
|
|
1594
|
+
// draining starts — not at the final publish (step 11-13) — so the OLD
|
|
1595
|
+
// generation's crash-retry loop (spawnRoleIncarnation's isStaleGeneration
|
|
1596
|
+
// check) recognizes supersession immediately. Without this, a backoff
|
|
1597
|
+
// timer firing during the drain/force-stop window, or the forced abort's
|
|
1598
|
+
// own rejection, would still see itself as the current generation:
|
|
1599
|
+
// the abort's rejection doesn't match killedByStop's SIGTERM-only regex,
|
|
1600
|
+
// so it would run full terminal crash handling — a duplicate live runner
|
|
1601
|
+
// (mid-backoff restart) or a false worker-crashed notification, exactly
|
|
1602
|
+
// what the guard exists to prevent.
|
|
1603
|
+
const newGeneration = slot.generation + 1;
|
|
1604
|
+
slot.generation = newGeneration;
|
|
1605
|
+
slot.phase = 'draining';
|
|
1606
|
+
// Every await from here on can race a stop/restart of this org — verify
|
|
1607
|
+
// ownership before EVERY subsequent step, not just once before the final
|
|
1608
|
+
// publish, so a stale operation can never mutate accounting, force-stop
|
|
1609
|
+
// a runtime, or spawn into an org that's no longer the live one.
|
|
1610
|
+
const stillOwned = () => this.orgs.get(name) === running && running.roleSlots.get(input.roleId) === slot;
|
|
1611
|
+
const abandonedReceipt = () => {
|
|
1612
|
+
running.respawning.delete(input.roleId);
|
|
1613
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1614
|
+
roleId: input.roleId,
|
|
1615
|
+
error: `org "${name}" stopped or restarted during replacement`,
|
|
1616
|
+
});
|
|
1617
|
+
};
|
|
1618
|
+
const oldRuntime = slot.runtime;
|
|
1619
|
+
const sweptQueue = oldRuntime.mailbox.beginDrain();
|
|
1620
|
+
slot.queuedDuringSwap.push(...sweptQueue);
|
|
1621
|
+
const drainTimeoutMs = running.def.run_config.respawn_drain_timeout_ms ?? 30_000;
|
|
1622
|
+
const drained = await Promise.race([
|
|
1623
|
+
oldRuntime.done.then(() => true),
|
|
1624
|
+
new Promise((r) => setTimeout(() => r(false), drainTimeoutMs)),
|
|
1625
|
+
]);
|
|
1626
|
+
if (!stillOwned())
|
|
1627
|
+
return abandonedReceipt();
|
|
1628
|
+
let drainTimedOut = false;
|
|
1629
|
+
if (!drained) {
|
|
1630
|
+
drainTimedOut = true;
|
|
1631
|
+
// Step 7: force stop.
|
|
1632
|
+
slot.abort?.abort();
|
|
1633
|
+
const forceStopMs = running.def.run_config.respawn_force_stop_timeout_ms ?? 5_000;
|
|
1634
|
+
const stopped = await Promise.race([
|
|
1635
|
+
oldRuntime.done.then(() => true).catch(() => true),
|
|
1636
|
+
new Promise((r) => setTimeout(() => r(false), forceStopMs)),
|
|
1637
|
+
]);
|
|
1638
|
+
if (!stillOwned())
|
|
1639
|
+
return abandonedReceipt();
|
|
1640
|
+
if (!stopped) {
|
|
1641
|
+
slot.phase = 'stuck';
|
|
1642
|
+
running.respawning.delete(input.roleId);
|
|
1643
|
+
running.bus.emit({
|
|
1644
|
+
type: 'audit',
|
|
1645
|
+
from: callerId,
|
|
1646
|
+
reason: 'role-respawn-failed',
|
|
1647
|
+
msg: `role "${input.roleId}" forced stop did not confirm termination — refusing to spawn a replacement`,
|
|
1648
|
+
});
|
|
1649
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1650
|
+
roleId: input.roleId,
|
|
1651
|
+
error: `role "${input.roleId}" could not be confirmed stopped; not replaced`,
|
|
1652
|
+
});
|
|
1653
|
+
}
|
|
1654
|
+
}
|
|
1655
|
+
// Step 8: preserve durable role state (worktree path, task ownership, and
|
|
1656
|
+
// the DAG survive untouched — they live outside AgentRuntime/Mailbox
|
|
1657
|
+
// entirely, keyed by role.id, which never changes). Reclaim any message
|
|
1658
|
+
// abandoned mid-yield by a forced stop for at-least-once redelivery.
|
|
1659
|
+
oldRuntime.mailbox.reclaimInFlight();
|
|
1660
|
+
const reclaimedQueue = oldRuntime.mailbox.serialize().queue;
|
|
1661
|
+
slot.queuedDuringSwap.push(...reclaimedQueue);
|
|
1662
|
+
// Step 9: retire accounting BEFORE replacing the runtime.
|
|
1663
|
+
slot.retiredUsage = {
|
|
1664
|
+
tokens: slot.retiredUsage.tokens + oldRuntime.policy.usage,
|
|
1665
|
+
costUsd: slot.retiredUsage.costUsd + oldRuntime.metrics.costUsd,
|
|
1666
|
+
};
|
|
1667
|
+
// Step 10: spawn generation N+1 (generation already bumped in step 6).
|
|
1668
|
+
const { runtime: newRuntime, abort: newAbort } = this.spawnRoleIncarnation(name, running, candidateRole, newGeneration, { budgetTokensOverride: budgetTokens });
|
|
1669
|
+
// Seed the new mailbox with everything swapped/reclaimed, delivered
|
|
1670
|
+
// FIFO, plus a delimited coordinator briefing appended last so it reads
|
|
1671
|
+
// as the newest context once the replacement starts its first turn.
|
|
1672
|
+
for (const queued of slot.queuedDuringSwap)
|
|
1673
|
+
newRuntime.mailbox.push(queued);
|
|
1674
|
+
newRuntime.mailbox.push(`[system: role replacement briefing — not a system prompt] You are a fresh session replacing the previous incarnation of role "${input.roleId}". Reason: ${input.reason}\n\n${input.briefing}`);
|
|
1675
|
+
// Seed USD accounting from retained totals so a respawn cannot reset
|
|
1676
|
+
// role.budget_usd.
|
|
1677
|
+
if (running.def.roles.find((r) => r.id === input.roleId)?.budget_usd !== undefined) {
|
|
1678
|
+
newRuntime.policy.setUsageUsd(slot.retiredUsage.costUsd);
|
|
1679
|
+
}
|
|
1680
|
+
// "Ready" here means "did not crash within the startup window" — a
|
|
1681
|
+
// silent-but-healthy runner (one that never emits a chat/tool/usage
|
|
1682
|
+
// event, e.g. because it hasn't finished its first turn yet) must not be
|
|
1683
|
+
// misreported as a startup failure, so this does NOT wait for a positive
|
|
1684
|
+
// signal. It races the new incarnation's own crash-retry loop (which
|
|
1685
|
+
// shares this generation, so it is NOT superseded and behaves normally)
|
|
1686
|
+
// against the timeout: a config that fails immediately (bad model,
|
|
1687
|
+
// missing runtime binary, auth failure) crashes fast and newRuntime.done
|
|
1688
|
+
// resolves with status 'crashed' well before startTimeoutMs, correctly
|
|
1689
|
+
// failing readiness and triggering rollback.
|
|
1690
|
+
const startTimeoutMs = running.def.run_config.respawn_start_timeout_ms ?? 60_000;
|
|
1691
|
+
const ready = await Promise.race([
|
|
1692
|
+
newRuntime.done.then(() => newRuntime.status !== 'crashed'),
|
|
1693
|
+
new Promise((r) => setTimeout(() => r(true), startTimeoutMs)),
|
|
1694
|
+
]);
|
|
1695
|
+
// Step 11: publish atomically — verify ownership is still current.
|
|
1696
|
+
if (!stillOwned()) {
|
|
1697
|
+
newAbort.abort();
|
|
1698
|
+
return abandonedReceipt();
|
|
1699
|
+
}
|
|
1700
|
+
if (!ready) {
|
|
1701
|
+
// Step 12: rollback — one attempt with the prior effective config.
|
|
1702
|
+
newAbort.abort();
|
|
1703
|
+
running.bus.emit({
|
|
1704
|
+
type: 'audit',
|
|
1705
|
+
from: callerId,
|
|
1706
|
+
reason: 'role-respawn-failed',
|
|
1707
|
+
msg: `role "${input.roleId}" replacement did not become ready within ${startTimeoutMs}ms — attempting rollback`,
|
|
1708
|
+
});
|
|
1709
|
+
try {
|
|
1710
|
+
const { runtime: rolledBack, abort: rolledBackAbort } = this.spawnRoleIncarnation(name, running, slot.effectiveRole, newGeneration + 1, {});
|
|
1711
|
+
for (const queued of slot.queuedDuringSwap)
|
|
1712
|
+
rolledBack.mailbox.push(queued);
|
|
1713
|
+
running.agents.set(input.roleId, rolledBack);
|
|
1714
|
+
slot.runtime = rolledBack;
|
|
1715
|
+
slot.abort = rolledBackAbort;
|
|
1716
|
+
slot.generation = newGeneration + 1;
|
|
1717
|
+
slot.phase = 'running';
|
|
1718
|
+
slot.queuedDuringSwap = [];
|
|
1719
|
+
running.respawning.delete(input.roleId);
|
|
1720
|
+
running.bus.emit({
|
|
1721
|
+
type: 'audit',
|
|
1722
|
+
from: callerId,
|
|
1723
|
+
reason: 'role-respawn-failed',
|
|
1724
|
+
msg: `role "${input.roleId}" replacement failed; rolled back to prior config`,
|
|
1725
|
+
});
|
|
1726
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1727
|
+
roleId: input.roleId,
|
|
1728
|
+
drainTimedOut,
|
|
1729
|
+
error: `replacement failed to start; rolled back to prior configuration`,
|
|
1730
|
+
});
|
|
1731
|
+
}
|
|
1732
|
+
catch (rollbackErr) {
|
|
1733
|
+
slot.phase = 'crashed';
|
|
1734
|
+
running.respawning.delete(input.roleId);
|
|
1735
|
+
running.bus.emit({
|
|
1736
|
+
type: 'audit',
|
|
1737
|
+
from: callerId,
|
|
1738
|
+
reason: 'role-respawn-rollback-failed',
|
|
1739
|
+
msg: `role "${input.roleId}" replacement AND rollback both failed: ${rollbackErr instanceof Error ? rollbackErr.message : String(rollbackErr)}`,
|
|
1740
|
+
});
|
|
1741
|
+
return buildRespawnReceipt(slot, maxRespawns, false, {
|
|
1742
|
+
roleId: input.roleId,
|
|
1743
|
+
drainTimedOut,
|
|
1744
|
+
error: `replacement and rollback both failed; role "${input.roleId}" is unavailable`,
|
|
1745
|
+
});
|
|
1746
|
+
}
|
|
1747
|
+
}
|
|
1748
|
+
running.agents.set(input.roleId, newRuntime);
|
|
1749
|
+
slot.runtime = newRuntime;
|
|
1750
|
+
slot.abort = newAbort;
|
|
1751
|
+
slot.generation = newGeneration;
|
|
1752
|
+
slot.effectiveRole = candidateRole;
|
|
1753
|
+
slot.phase = 'running';
|
|
1754
|
+
slot.queuedDuringSwap = [];
|
|
1755
|
+
running.respawning.delete(input.roleId);
|
|
1756
|
+
// Step 13: audit and persist.
|
|
1757
|
+
running.bus.emit({
|
|
1758
|
+
type: 'audit',
|
|
1759
|
+
from: callerId,
|
|
1760
|
+
reason: 'role-respawned',
|
|
1761
|
+
msg: `role "${input.roleId}" replaced (generation ${newGeneration})`,
|
|
1762
|
+
data: {
|
|
1763
|
+
roleId: input.roleId,
|
|
1764
|
+
generation: newGeneration,
|
|
1765
|
+
respawnCount: slot.respawnCount,
|
|
1766
|
+
drainTimedOut,
|
|
1767
|
+
},
|
|
1768
|
+
});
|
|
1769
|
+
this.persistState(name, 'running', running.run);
|
|
1770
|
+
return buildRespawnReceipt(slot, maxRespawns, true, { roleId: input.roleId, drainTimedOut });
|
|
1771
|
+
}
|
|
1213
1772
|
/** @internal */
|
|
1214
1773
|
hasOrgDef(name) {
|
|
1215
1774
|
if (!/^[a-z0-9][a-z0-9_-]{0,63}$/i.test(name))
|
|
@@ -1275,6 +1834,14 @@ export class OrgDaemon {
|
|
|
1275
1834
|
this.leases.delete(name);
|
|
1276
1835
|
for (const a of org.agents.values())
|
|
1277
1836
|
a.mailbox.close();
|
|
1837
|
+
// Closing the mailbox stops new work being handed to a session, but does
|
|
1838
|
+
// NOT cancel a turn already in flight (e.g. mid provider call) — that
|
|
1839
|
+
// session can keep running, and eventually crash/finish, well past this
|
|
1840
|
+
// function's own bounded drain below. Abort each slot's live incarnation
|
|
1841
|
+
// too, reusing respawnRole's existing force-stop handle, so in-flight
|
|
1842
|
+
// work is told to stop now instead of merely being denied new input.
|
|
1843
|
+
for (const slot of org.roleSlots.values())
|
|
1844
|
+
slot.abort?.abort();
|
|
1278
1845
|
// Bounded: a genuinely hung agent session (stuck mid-tool-call, not just
|
|
1279
1846
|
// idle) must not make stopOrg() hang forever — callers like the scheduler
|
|
1280
1847
|
// already race their own timeout around a run, and this wait re-blocking
|
|
@@ -1286,10 +1853,20 @@ export class OrgDaemon {
|
|
|
1286
1853
|
// session ends, so a long drain is a ceiling, not a delay.
|
|
1287
1854
|
const stopWaitMs = drainMs ?? this.opts.stopWaitMs ?? 15_000;
|
|
1288
1855
|
const allDone = Promise.allSettled([...org.agents.values()].map((a) => a.done)).then(() => false);
|
|
1856
|
+
// Clear the ceiling timer once the sessions win the race: left pending, a
|
|
1857
|
+
// COMPLETE_DRAIN_MS stop kept `org run` (which returns without
|
|
1858
|
+
// process.exit on a clean completion) alive for up to five minutes after
|
|
1859
|
+
// every session had already ended. Deliberately NOT unref'd — on the
|
|
1860
|
+
// timed-out path this timer may be the only thing keeping the loop alive
|
|
1861
|
+
// long enough to write 'stopped' to runtime.json and flush the bus.
|
|
1862
|
+
let drainTimer;
|
|
1289
1863
|
const timedOut = await Promise.race([
|
|
1290
1864
|
allDone,
|
|
1291
|
-
new Promise((r) =>
|
|
1865
|
+
new Promise((r) => {
|
|
1866
|
+
drainTimer = setTimeout(() => r(true), stopWaitMs);
|
|
1867
|
+
}),
|
|
1292
1868
|
]);
|
|
1869
|
+
clearTimeout(drainTimer);
|
|
1293
1870
|
if (timedOut) {
|
|
1294
1871
|
// #152: "proceeding anyway" alone didn't say WHO got cut off — a run
|
|
1295
1872
|
// reviewer had no way to tell whether real, in-progress work (a
|
|
@@ -1325,6 +1902,13 @@ export class OrgDaemon {
|
|
|
1325
1902
|
}
|
|
1326
1903
|
org.bus.emit({ type: 'status', msg: 'org stopped' });
|
|
1327
1904
|
await org.bus.flush();
|
|
1905
|
+
// flush() only awaits a snapshot of writes queued at call time (see its
|
|
1906
|
+
// own doc comment) — it has no visibility into a session that crashes
|
|
1907
|
+
// after the abort signal above but before this function returns. Seal
|
|
1908
|
+
// the bus now so any such late bus.emit() still reaches in-memory
|
|
1909
|
+
// listeners but can never schedule a new disk write into a run
|
|
1910
|
+
// directory a caller (e.g. a test's afterEach) may already be deleting.
|
|
1911
|
+
await org.bus.seal();
|
|
1328
1912
|
// Append this run's summary to <org>/history.jsonl — read back from the
|
|
1329
1913
|
// flushed bus.jsonl (the full durable record) rather than the bounded
|
|
1330
1914
|
// in-memory buffer, so long runs summarize completely.
|
|
@@ -1461,6 +2045,17 @@ export class OrgDaemon {
|
|
|
1461
2045
|
for (const [name, org] of this.orgs) {
|
|
1462
2046
|
try {
|
|
1463
2047
|
const p = join(this.root, ORG_DIR, name, 'runtime.json');
|
|
2048
|
+
// Capture separately from the write below: a throw here (e.g. a
|
|
2049
|
+
// cyclic structure in roleState reaching generateChecksum) must not
|
|
2050
|
+
// suppress the base crash record, which is the actually-important
|
|
2051
|
+
// best-effort write this method exists for.
|
|
2052
|
+
let checkpoint;
|
|
2053
|
+
try {
|
|
2054
|
+
checkpoint = captureCheckpoint(org, 'crashed');
|
|
2055
|
+
}
|
|
2056
|
+
catch {
|
|
2057
|
+
/* best effort — proceed without a checkpoint */
|
|
2058
|
+
}
|
|
1464
2059
|
// C4: atomic write — crash handler is the most likely place to hit
|
|
1465
2060
|
// a partial write since the process is mid-teardown.
|
|
1466
2061
|
writeJsonFileAtomic(p, {
|
|
@@ -1469,6 +2064,7 @@ export class OrgDaemon {
|
|
|
1469
2064
|
pid: process.pid,
|
|
1470
2065
|
updated: new Date().toISOString(),
|
|
1471
2066
|
closedBy: 'crash-handler',
|
|
2067
|
+
...(checkpoint ? { checkpoint } : {}),
|
|
1472
2068
|
...(error ? { error } : {}),
|
|
1473
2069
|
});
|
|
1474
2070
|
}
|
|
@@ -1508,8 +2104,8 @@ export class OrgDaemon {
|
|
|
1508
2104
|
}
|
|
1509
2105
|
// ── Delegated methods — extracted to focused modules ──────────────────
|
|
1510
2106
|
// approvals.ts
|
|
1511
|
-
checkApproval(org, role, action) {
|
|
1512
|
-
return approvalOps.checkApproval(this, org, role, action);
|
|
2107
|
+
checkApproval(org, role, action, input) {
|
|
2108
|
+
return approvalOps.checkApproval(this, org, role, action, input);
|
|
1513
2109
|
}
|
|
1514
2110
|
async setApproval(org, role, action, approved) {
|
|
1515
2111
|
return approvalOps.setApproval(this, org, role, action, approved);
|
|
@@ -1562,11 +2158,12 @@ export class OrgDaemon {
|
|
|
1562
2158
|
async deliver(fromOrg, fromRole, to, subject, body) {
|
|
1563
2159
|
return crossOrg.deliver(this, fromOrg, fromRole, to, subject, body);
|
|
1564
2160
|
}
|
|
1565
|
-
receiveRemote(toOrg, toRole, fromQualified, subject, body) {
|
|
1566
|
-
return crossOrg.receiveRemote(this, toOrg, toRole, fromQualified, subject, body);
|
|
2161
|
+
receiveRemote(toOrg, toRole, fromQualified, subject, body, fromCredential) {
|
|
2162
|
+
return crossOrg.receiveRemote(this, toOrg, toRole, fromQualified, subject, body, fromCredential);
|
|
1567
2163
|
}
|
|
1568
|
-
|
|
1569
|
-
|
|
2164
|
+
// runtime-options.ts
|
|
2165
|
+
listRuntimeOptions() {
|
|
2166
|
+
return buildRuntimeOptions(this.root);
|
|
1570
2167
|
}
|
|
1571
2168
|
// scheduler-integration.ts
|
|
1572
2169
|
/** @internal */
|
|
@@ -1580,6 +2177,13 @@ export class OrgDaemon {
|
|
|
1580
2177
|
scheduleDeferredSpawn(name, running, role, spawnRole) {
|
|
1581
2178
|
scheduler.scheduleDeferredSpawn(this, name, running, role, spawnRole);
|
|
1582
2179
|
}
|
|
2180
|
+
/** Bug 4: mirrors scheduleDeferredSpawn, but for a role deferred because the
|
|
2181
|
+
* org is already at run_config.max_concurrent_agents rather than under host
|
|
2182
|
+
* resource pressure — see scheduleConcurrencyDeferredSpawn's doc comment. */
|
|
2183
|
+
/** @internal */
|
|
2184
|
+
scheduleConcurrencyDeferredSpawn(name, running, role, spawnRole) {
|
|
2185
|
+
scheduler.scheduleConcurrencyDeferredSpawn(this, name, running, role, spawnRole);
|
|
2186
|
+
}
|
|
1583
2187
|
// org-memory.ts
|
|
1584
2188
|
orgMemoryNamespace(name, def) {
|
|
1585
2189
|
return orgMemory.orgMemoryNamespace(name, def);
|