@cohortapp/agent-sdk 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (731) hide show
  1. package/.claude/commands/init-agent.md +104 -0
  2. package/.claude/commands/init-maestro.md +1187 -0
  3. package/.claude/settings.json +161 -0
  4. package/.env.example +216 -0
  5. package/README.md +632 -0
  6. package/agents/browser-operator/agent.md +52 -0
  7. package/agents/calendar-ops/agent.md +50 -0
  8. package/agents/communications/agent.md +96 -0
  9. package/agents/decision-log/agent.md +65 -0
  10. package/agents/desktop-operator/agent.md +59 -0
  11. package/agents/gmail-operator/agent.md +62 -0
  12. package/agents/inbound-dispatcher/agent.md +66 -0
  13. package/agents/inbox-processor/agent.md +39 -0
  14. package/agents/pmo-execution/agent.md +60 -0
  15. package/agents/session-spawner/agent.md +64 -0
  16. package/agents/slack-operator/agent.md +60 -0
  17. package/agents/whatsapp-operator/agent.md +60 -0
  18. package/agents/workflow-automation/agent.md +61 -0
  19. package/archetypes/altitudes/c-suite.yaml +58 -0
  20. package/archetypes/altitudes/founder.yaml +68 -0
  21. package/archetypes/altitudes/senior-manager.yaml +63 -0
  22. package/archetypes/altitudes/svp.yaml +60 -0
  23. package/archetypes/altitudes/vp.yaml +50 -0
  24. package/archetypes/archetype.schema.json +77 -0
  25. package/archetypes/base.yaml +47 -0
  26. package/archetypes/capabilities/commercial-leader.yaml +159 -0
  27. package/archetypes/capabilities/compliance-officer.yaml +159 -0
  28. package/archetypes/capabilities/executive-operator.yaml +169 -0
  29. package/archetypes/capabilities/finance-leader.yaml +162 -0
  30. package/archetypes/capabilities/operations-leader.yaml +154 -0
  31. package/archetypes/capabilities/product-leader.yaml +148 -0
  32. package/archetypes/capabilities/technical-leader.yaml +146 -0
  33. package/archetypes/functions/commercial-leader.yaml +62 -0
  34. package/archetypes/functions/compliance-officer.yaml +64 -0
  35. package/archetypes/functions/executive-operator.yaml +70 -0
  36. package/archetypes/functions/finance-leader.yaml +70 -0
  37. package/archetypes/functions/operations-leader.yaml +62 -0
  38. package/archetypes/functions/product-leader.yaml +61 -0
  39. package/archetypes/functions/technical-leader.yaml +57 -0
  40. package/bin/cohort-mcp.mjs +81 -0
  41. package/bin/maestro.mjs +3516 -0
  42. package/bin/maestro.test.mjs +1015 -0
  43. package/desktop-control/README.md +56 -0
  44. package/desktop-control/app-profiles/gmail.yaml +120 -0
  45. package/desktop-control/app-profiles/slack.yaml +315 -0
  46. package/desktop-control/app-profiles/whatsapp.yaml +107 -0
  47. package/docs/architecture/agent-topology.md +2239 -0
  48. package/docs/architecture/archetype-agent-factory.md +110 -0
  49. package/docs/architecture/collective-memory-and-org-mesh.md +115 -0
  50. package/docs/architecture/continuous-monitoring.md +221 -0
  51. package/docs/architecture/mcp-capability-map.md +585 -0
  52. package/docs/architecture/system-architecture.md +1272 -0
  53. package/docs/company-context/README.md +40 -0
  54. package/docs/guides/agent-persona-setup.md +600 -0
  55. package/docs/guides/agents-observe-setup.md +64 -0
  56. package/docs/guides/billing-console-keys.md +88 -0
  57. package/docs/guides/ccxray-diagnostics.md +65 -0
  58. package/docs/guides/channel-bus.md +127 -0
  59. package/docs/guides/claude-mem-setup.md +79 -0
  60. package/docs/guides/claude-pace-setup.md +56 -0
  61. package/docs/guides/claudraband-sessions.md +98 -0
  62. package/docs/guides/clawteam-swarm.md +116 -0
  63. package/docs/guides/code-review-graph-setup.md +86 -0
  64. package/docs/guides/email-setup.md +431 -0
  65. package/docs/guides/mac-mini.md +119 -0
  66. package/docs/guides/media-generation-setup.md +349 -0
  67. package/docs/guides/model-routing.md +162 -0
  68. package/docs/guides/observability-otel.md +265 -0
  69. package/docs/guides/org-onboarding.md +132 -0
  70. package/docs/guides/outbound-governance-setup.md +437 -0
  71. package/docs/guides/pdf-generation-setup.md +315 -0
  72. package/docs/guides/poller-daemon-setup.md +563 -0
  73. package/docs/guides/rag-context-setup.md +459 -0
  74. package/docs/guides/self-optimization-pattern.md +82 -0
  75. package/docs/guides/setup-wizard.md +178 -0
  76. package/docs/guides/slack-setup.md +350 -0
  77. package/docs/guides/telegram-setup.md +227 -0
  78. package/docs/guides/twilio-subaccounts-setup.md +223 -0
  79. package/docs/guides/verification.md +128 -0
  80. package/docs/guides/voice-mode.md +188 -0
  81. package/docs/guides/voice-sms-setup.md +698 -0
  82. package/docs/guides/webhook-relay-setup.md +349 -0
  83. package/docs/guides/whatsapp-setup.md +288 -0
  84. package/docs/prompts/board-pack-cover-template.md +36 -0
  85. package/docs/prompts/decision-recommendation-template.md +88 -0
  86. package/docs/prompts/followup-message-template.md +141 -0
  87. package/docs/prompts/investor-letter-template.md +52 -0
  88. package/docs/prompts/morning-brief-template.md +82 -0
  89. package/docs/prompts/presentation-template.md +58 -0
  90. package/docs/prompts/weekly-strategic-memo-template.md +104 -0
  91. package/docs/research/hallucinated-tool-output-investigation.md +151 -0
  92. package/docs/runbooks/backup-restore.md +205 -0
  93. package/docs/runbooks/cohort-cutover.md +129 -0
  94. package/docs/runbooks/fleet-operations.md +200 -0
  95. package/docs/runbooks/incident-response.md +226 -0
  96. package/docs/runbooks/mac-mini-bootstrap.md +431 -0
  97. package/docs/runbooks/perpetual-operations.md +509 -0
  98. package/docs/runbooks/recovery-and-failover.md +260 -0
  99. package/framework-features.json +267 -0
  100. package/ingest/README.md +87 -0
  101. package/lib/action-executor.js +689 -0
  102. package/lib/action-executor.test.mjs +871 -0
  103. package/lib/agent-root.mjs +37 -0
  104. package/lib/archetype.mjs +236 -0
  105. package/lib/archetype.test.mjs +132 -0
  106. package/lib/autonomy.mjs +114 -0
  107. package/lib/autonomy.test.mjs +66 -0
  108. package/lib/backlog.mjs +358 -0
  109. package/lib/backlog.test.mjs +266 -0
  110. package/lib/budget-guard.mjs +279 -0
  111. package/lib/budget-guard.test.mjs +291 -0
  112. package/lib/cadence-bus-schedule.test.mjs +194 -0
  113. package/lib/cadence-bus.mjs +1120 -0
  114. package/lib/cadence-bus.test.mjs +720 -0
  115. package/lib/cadences.mjs +205 -0
  116. package/lib/cadences.test.mjs +125 -0
  117. package/lib/capability.mjs +154 -0
  118. package/lib/capability.test.mjs +78 -0
  119. package/lib/channels/base-adapter.mjs +719 -0
  120. package/lib/channels/base-adapter.test.mjs +590 -0
  121. package/lib/channels/channel.mjs +128 -0
  122. package/lib/channels/channels.test.mjs +371 -0
  123. package/lib/channels/contract.mjs +215 -0
  124. package/lib/channels/contract.test.mjs +137 -0
  125. package/lib/channels/conversation-resolver.mjs +95 -0
  126. package/lib/channels/gmail/adapter.mjs +87 -0
  127. package/lib/channels/inbox-item.mjs +255 -0
  128. package/lib/channels/inbox-item.test.mjs +335 -0
  129. package/lib/channels/index.mjs +94 -0
  130. package/lib/channels/orgmail/adapter.mjs +353 -0
  131. package/lib/channels/orgmail/adapter.test.mjs +311 -0
  132. package/lib/channels/pairing.mjs +363 -0
  133. package/lib/channels/pairing.test.mjs +270 -0
  134. package/lib/channels/registry.mjs +164 -0
  135. package/lib/channels/slack/adapter.mjs +317 -0
  136. package/lib/channels/slack-adapter.test.mjs +212 -0
  137. package/lib/channels/sms/adapter.mjs +43 -0
  138. package/lib/channels/telegram/adapter.mjs +432 -0
  139. package/lib/channels/telegram-adapter.test.mjs +306 -0
  140. package/lib/channels/voice/adapter.mjs +301 -0
  141. package/lib/channels/voice/adapter.test.mjs +278 -0
  142. package/lib/channels/whatsapp/adapter-baileys.mjs +587 -0
  143. package/lib/channels/whatsapp/adapter-baileys.test.mjs +359 -0
  144. package/lib/channels/whatsapp/adapter-twilio.mjs +65 -0
  145. package/lib/channels/whatsapp/baileys-typing.test.mjs +154 -0
  146. package/lib/charter.mjs +256 -0
  147. package/lib/charter.test.mjs +89 -0
  148. package/lib/claude-bin.mjs +134 -0
  149. package/lib/claude-bin.test.mjs +75 -0
  150. package/lib/collective/capture.mjs +185 -0
  151. package/lib/collective/capture.test.mjs +121 -0
  152. package/lib/collective/cards.mjs +201 -0
  153. package/lib/collective/cards.test.mjs +114 -0
  154. package/lib/collective/config.mjs +186 -0
  155. package/lib/collective/config.test.mjs +123 -0
  156. package/lib/collective/global-config.mjs +113 -0
  157. package/lib/collective/global-config.test.mjs +75 -0
  158. package/lib/collective/presence.mjs +201 -0
  159. package/lib/collective/presence.test.mjs +95 -0
  160. package/lib/collective/recall.mjs +215 -0
  161. package/lib/collective/recall.test.mjs +116 -0
  162. package/lib/comms/send-gate.mjs +554 -0
  163. package/lib/comms/send-gate.test.mjs +577 -0
  164. package/lib/comms.mjs +67 -0
  165. package/lib/comms.test.mjs +41 -0
  166. package/lib/diagnostics/alerts.mjs +424 -0
  167. package/lib/diagnostics/alerts.test.mjs +318 -0
  168. package/lib/diagnostics/backup-freshness.mjs +188 -0
  169. package/lib/diagnostics/backup-freshness.test.mjs +185 -0
  170. package/lib/diagnostics/counters.mjs +269 -0
  171. package/lib/diagnostics/counters.test.mjs +206 -0
  172. package/lib/diagnostics/events.mjs +188 -0
  173. package/lib/diagnostics/events.test.mjs +290 -0
  174. package/lib/diagnostics/otel.mjs +237 -0
  175. package/lib/diagnostics/otel.test.mjs +196 -0
  176. package/lib/diagnostics/trace.mjs +216 -0
  177. package/lib/diagnostics/trace.test.mjs +251 -0
  178. package/lib/env-compat.mjs +74 -0
  179. package/lib/env-compat.test.mjs +104 -0
  180. package/lib/feature-init.mjs +331 -0
  181. package/lib/fs-atomic.mjs +112 -0
  182. package/lib/fs-atomic.test.mjs +72 -0
  183. package/lib/fs-ownership.mjs +111 -0
  184. package/lib/fs-ownership.test.mjs +158 -0
  185. package/lib/hooks/bus.mjs +347 -0
  186. package/lib/hooks/bus.test.mjs +387 -0
  187. package/lib/index.js +16 -0
  188. package/lib/learning/config.mjs +106 -0
  189. package/lib/learning/config.test.mjs +75 -0
  190. package/lib/learning/counters.mjs +156 -0
  191. package/lib/learning/counters.test.mjs +69 -0
  192. package/lib/learning/curator-consolidate.test.mjs +238 -0
  193. package/lib/learning/curator.mjs +453 -0
  194. package/lib/learning/curator.test.mjs +106 -0
  195. package/lib/learning/log.mjs +40 -0
  196. package/lib/learning/reflect.mjs +534 -0
  197. package/lib/learning/reflect.test.mjs +0 -0
  198. package/lib/learning/session-index.mjs +352 -0
  199. package/lib/learning/session-index.test.mjs +125 -0
  200. package/lib/learning/skill-writer.mjs +474 -0
  201. package/lib/learning/skill-writer.test.mjs +210 -0
  202. package/lib/mcp/server.mjs +328 -0
  203. package/lib/mcp/server.test.mjs +400 -0
  204. package/lib/model-router/auth-profiles.mjs +758 -0
  205. package/lib/model-router/auth-profiles.test.mjs +580 -0
  206. package/lib/model-router/catalog/anthropic.yaml +153 -0
  207. package/lib/model-router/catalog/deepseek.yaml +86 -0
  208. package/lib/model-router/catalog/moonshot.yaml +81 -0
  209. package/lib/model-router/catalog/qwen.yaml +114 -0
  210. package/lib/model-router/catalog.mjs +925 -0
  211. package/lib/model-router/catalog.test.mjs +385 -0
  212. package/lib/model-router/economics.mjs +564 -0
  213. package/lib/model-router/economics.test.mjs +344 -0
  214. package/lib/model-router/failover.mjs +298 -0
  215. package/lib/model-router/failover.test.mjs +439 -0
  216. package/lib/model-router/health.mjs +453 -0
  217. package/lib/model-router/health.test.mjs +338 -0
  218. package/lib/model-router/integration-coverage.test.mjs +829 -0
  219. package/lib/model-router/integration.test.mjs +564 -0
  220. package/lib/model-router/ledger.mjs +402 -0
  221. package/lib/model-router/ledger.test.mjs +382 -0
  222. package/lib/model-router/llm-task.mjs +515 -0
  223. package/lib/model-router/llm-task.test.mjs +392 -0
  224. package/lib/model-router/org-credentials.mjs +260 -0
  225. package/lib/model-router/org-credentials.test.mjs +265 -0
  226. package/lib/model-router/pricing-refresh.mjs +463 -0
  227. package/lib/model-router/pricing-refresh.test.mjs +286 -0
  228. package/lib/model-router/reconcile.mjs +429 -0
  229. package/lib/model-router/reconcile.test.mjs +316 -0
  230. package/lib/model-router/repair.mjs +471 -0
  231. package/lib/model-router/repair.test.mjs +180 -0
  232. package/lib/model-router/resolve.mjs +1206 -0
  233. package/lib/model-router/spawn.mjs +497 -0
  234. package/lib/model-router/spawn.test.mjs +425 -0
  235. package/lib/model-router/taxonomy.mjs +893 -0
  236. package/lib/model-router/taxonomy.test.mjs +410 -0
  237. package/lib/model-router.mjs +677 -0
  238. package/lib/model-router.test.mjs +907 -0
  239. package/lib/org/activity.mjs +211 -0
  240. package/lib/org/activity.test.mjs +134 -0
  241. package/lib/org/approvals.mjs +448 -0
  242. package/lib/org/approvals.test.mjs +216 -0
  243. package/lib/org/awareness.mjs +222 -0
  244. package/lib/org/awareness.test.mjs +159 -0
  245. package/lib/org/board.mjs +229 -0
  246. package/lib/org/board.test.mjs +177 -0
  247. package/lib/org/bootstrap-context.mjs +169 -0
  248. package/lib/org/bootstrap-context.test.mjs +153 -0
  249. package/lib/org/client.mjs +1628 -0
  250. package/lib/org/client.test.mjs +1107 -0
  251. package/lib/org/cohort-client.mjs +67 -0
  252. package/lib/org/cohort-client.test.mjs +126 -0
  253. package/lib/org/cost-sync.mjs +227 -0
  254. package/lib/org/cost-sync.test.mjs +153 -0
  255. package/lib/org/doctor.mjs +212 -0
  256. package/lib/org/doctor.test.mjs +212 -0
  257. package/lib/org/handoff.mjs +293 -0
  258. package/lib/org/handoff.test.mjs +269 -0
  259. package/lib/org/integration-tools.mjs +182 -0
  260. package/lib/org/integration-tools.test.mjs +160 -0
  261. package/lib/org/keys.mjs +131 -0
  262. package/lib/org/keys.test.mjs +92 -0
  263. package/lib/org/knowledge.mjs +463 -0
  264. package/lib/org/knowledge.test.mjs +319 -0
  265. package/lib/org/leases.mjs +335 -0
  266. package/lib/org/leases.test.mjs +235 -0
  267. package/lib/org/mesh-integration.test.mjs +127 -0
  268. package/lib/org/mesh.mjs +459 -0
  269. package/lib/org/mesh.test.mjs +345 -0
  270. package/lib/org/messaging.mjs +503 -0
  271. package/lib/org/messaging.test.mjs +238 -0
  272. package/lib/org/policy.mjs +345 -0
  273. package/lib/org/policy.test.mjs +237 -0
  274. package/lib/org/protocol.checksum +1 -0
  275. package/lib/org/protocol.checksum.test.mjs +90 -0
  276. package/lib/org/protocol.mjs +967 -0
  277. package/lib/org/protocol.test.mjs +264 -0
  278. package/lib/org/registry.mjs +194 -0
  279. package/lib/org/registry.test.mjs +100 -0
  280. package/lib/org/tool-surface-integration.test.mjs +120 -0
  281. package/lib/org/tool-surface.mjs +2535 -0
  282. package/lib/org/tool-surface.test.mjs +589 -0
  283. package/lib/org/ui-parity.mjs +3236 -0
  284. package/lib/org/ui-parity.test.mjs +348 -0
  285. package/lib/org/verify.mjs +176 -0
  286. package/lib/org/verify.test.mjs +194 -0
  287. package/lib/rag/embed.mjs +188 -0
  288. package/lib/rag/indexer.mjs +425 -0
  289. package/lib/rag/rag.test.mjs +505 -0
  290. package/lib/rag/search.mjs +475 -0
  291. package/lib/rate-guard.mjs +246 -0
  292. package/lib/rate-guard.test.mjs +201 -0
  293. package/lib/render.mjs +112 -0
  294. package/lib/render.test.mjs +68 -0
  295. package/lib/resource-governor.mjs +297 -0
  296. package/lib/resource-governor.test.mjs +262 -0
  297. package/lib/scheduling/dynamic-jobs.mjs +675 -0
  298. package/lib/scheduling/dynamic-jobs.test.mjs +344 -0
  299. package/lib/scheduling/jitter.mjs +0 -0
  300. package/lib/scheduling/jitter.test.mjs +140 -0
  301. package/lib/secrets/broker.mjs +315 -0
  302. package/lib/secrets/broker.test.mjs +280 -0
  303. package/lib/secrets/providers.mjs +461 -0
  304. package/lib/secrets/providers.test.mjs +274 -0
  305. package/lib/security/audit-engine.mjs +684 -0
  306. package/lib/security/audit-engine.test.mjs +389 -0
  307. package/lib/security/coerce-args.mjs +552 -0
  308. package/lib/security/coerce-args.test.mjs +281 -0
  309. package/lib/security/dangerous-tools.mjs +97 -0
  310. package/lib/security/dangerous-tools.test.mjs +68 -0
  311. package/lib/security/external-content.mjs +145 -0
  312. package/lib/security/external-content.test.mjs +67 -0
  313. package/lib/security/redact.mjs +592 -0
  314. package/lib/security/redact.test.mjs +441 -0
  315. package/lib/security/secret-equal.mjs +73 -0
  316. package/lib/security/secret-equal.test.mjs +55 -0
  317. package/lib/session-permissions.mjs +101 -0
  318. package/lib/session-permissions.test.mjs +100 -0
  319. package/lib/setup/claude-probe.mjs +74 -0
  320. package/lib/setup/completeness.mjs +175 -0
  321. package/lib/setup/completeness.test.mjs +110 -0
  322. package/lib/setup/context-pack.mjs +173 -0
  323. package/lib/setup/context-pack.test.mjs +89 -0
  324. package/lib/setup/enrich.mjs +277 -0
  325. package/lib/setup/enrich.test.mjs +115 -0
  326. package/lib/setup/enroll-from-cohort.mjs +441 -0
  327. package/lib/setup/enroll-from-cohort.test.mjs +233 -0
  328. package/lib/setup/integration.test.mjs +162 -0
  329. package/lib/setup/io.mjs +360 -0
  330. package/lib/setup/io.test.mjs +77 -0
  331. package/lib/setup/run-generator.mjs +81 -0
  332. package/lib/setup/runner.mjs +244 -0
  333. package/lib/setup/runner.test.mjs +132 -0
  334. package/lib/setup/sections/comms.mjs +173 -0
  335. package/lib/setup/sections/company.mjs +120 -0
  336. package/lib/setup/sections/enrich.mjs +138 -0
  337. package/lib/setup/sections/identity.mjs +182 -0
  338. package/lib/setup/sections/identity.test.mjs +140 -0
  339. package/lib/setup/sections/learning.mjs +153 -0
  340. package/lib/setup/sections/learning.test.mjs +81 -0
  341. package/lib/setup/sections/messaging.mjs +219 -0
  342. package/lib/setup/sections/messaging.test.mjs +127 -0
  343. package/lib/setup/sections/model.mjs +102 -0
  344. package/lib/setup/sections/operating-model.mjs +78 -0
  345. package/lib/setup/sections/org.mjs +475 -0
  346. package/lib/setup/sections/org.test.mjs +313 -0
  347. package/lib/setup/sections/orgmail.mjs +173 -0
  348. package/lib/setup/sections/orgmail.test.mjs +118 -0
  349. package/lib/setup/sections/recovery.mjs +159 -0
  350. package/lib/setup/sections/recovery.test.mjs +98 -0
  351. package/lib/setup/sections/tools.mjs +132 -0
  352. package/lib/setup/sections/verify.mjs +97 -0
  353. package/lib/setup/sot.mjs +205 -0
  354. package/lib/setup/sot.test.mjs +81 -0
  355. package/lib/setup/state.mjs +151 -0
  356. package/lib/setup/state.test.mjs +92 -0
  357. package/lib/singleton.js +229 -0
  358. package/lib/singleton.test.mjs +135 -0
  359. package/lib/telemetry/alerts.mjs +216 -0
  360. package/lib/telemetry/alerts.test.mjs +109 -0
  361. package/lib/telemetry/collect.mjs +512 -0
  362. package/lib/telemetry/collect.test.mjs +202 -0
  363. package/lib/tool-definitions-integration.test.mjs +83 -0
  364. package/lib/tool-definitions.js +738 -0
  365. package/lib/tool-definitions.test.mjs +437 -0
  366. package/lib/util/fetch-timeout.mjs +136 -0
  367. package/lib/util/fetch-timeout.test.mjs +202 -0
  368. package/lib/util/reconnect.mjs +343 -0
  369. package/lib/util/reconnect.test.mjs +369 -0
  370. package/lib/util/unhandled.mjs +205 -0
  371. package/lib/util/unhandled.test.mjs +216 -0
  372. package/lib/voice/context-loader.mjs +466 -0
  373. package/lib/voice/index.mjs +100 -0
  374. package/lib/voice/openai-realtime.mjs +510 -0
  375. package/lib/voice/outbound.mjs +542 -0
  376. package/lib/voice/outbound.test.mjs +69 -0
  377. package/lib/voice/post-call-brief.mjs +428 -0
  378. package/lib/voice/provider.mjs +52 -0
  379. package/lib/voice/session-rotation.mjs +257 -0
  380. package/lib/voice/stt.mjs +161 -0
  381. package/lib/voice/stt.test.mjs +226 -0
  382. package/lib/voice/tool-bridge.mjs +370 -0
  383. package/lib/voice/tts.mjs +104 -0
  384. package/lib/voice/twilio-sip-bridge.mjs +288 -0
  385. package/lib/voice/voice.test.mjs +990 -0
  386. package/mcp/README.md +80 -0
  387. package/package.json +151 -0
  388. package/plugins/maestro-skills/plugin.json +139 -0
  389. package/plugins/maestro-skills/skills/agents-observe.md +110 -0
  390. package/plugins/maestro-skills/skills/board-deck.md +68 -0
  391. package/plugins/maestro-skills/skills/books-close.md +77 -0
  392. package/plugins/maestro-skills/skills/brand-steward.md +121 -0
  393. package/plugins/maestro-skills/skills/calendar-plan.md +57 -0
  394. package/plugins/maestro-skills/skills/call-working-sessions.md +124 -0
  395. package/plugins/maestro-skills/skills/ccxray-diagnostics.md +91 -0
  396. package/plugins/maestro-skills/skills/claude-pace.md +61 -0
  397. package/plugins/maestro-skills/skills/code-review-graph.md +99 -0
  398. package/plugins/maestro-skills/skills/crm-pipeline.md +65 -0
  399. package/plugins/maestro-skills/skills/decision-brief.md +89 -0
  400. package/plugins/maestro-skills/skills/directory-hygiene.md +125 -0
  401. package/plugins/maestro-skills/skills/draft-comms.md +84 -0
  402. package/plugins/maestro-skills/skills/evening-wrap.md +53 -0
  403. package/plugins/maestro-skills/skills/files-find.md +65 -0
  404. package/plugins/maestro-skills/skills/generative-ui.md +228 -0
  405. package/plugins/maestro-skills/skills/hiring-triage.md +74 -0
  406. package/plugins/maestro-skills/skills/inbox-triage.md +61 -0
  407. package/plugins/maestro-skills/skills/mail-triage.md +86 -0
  408. package/plugins/maestro-skills/skills/morning-brief.md +54 -0
  409. package/plugins/maestro-skills/skills/native-artifacts.md +157 -0
  410. package/plugins/maestro-skills/skills/org-board.md +133 -0
  411. package/plugins/maestro-skills/skills/org-credential.md +68 -0
  412. package/plugins/maestro-skills/skills/org-recall.md +81 -0
  413. package/plugins/maestro-skills/skills/pipeline-review.md +76 -0
  414. package/plugins/maestro-skills/skills/regulatory-status.md +81 -0
  415. package/plugins/maestro-skills/skills/router-why.md +78 -0
  416. package/plugins/maestro-skills/skills/schedule-meeting.md +91 -0
  417. package/plugins/maestro-skills/skills/session-search.md +71 -0
  418. package/plugins/maestro-skills/skills/set-reminder.md +93 -0
  419. package/plugins/maestro-skills/skills/slack-followup.md +64 -0
  420. package/plugins/maestro-skills/skills/team-activity.md +86 -0
  421. package/plugins/maestro-skills/skills/weekly-memo.md +70 -0
  422. package/policies/action-classification.yaml +114 -0
  423. package/policies/ai-disclosure.yaml +294 -0
  424. package/policies/communication-style.md +139 -0
  425. package/policies/information-barriers.yaml +118 -0
  426. package/policies/prompt-injection-defence.yaml +138 -0
  427. package/public/assets/icon-dark.png +0 -0
  428. package/public/assets/icon-dark.svg +9 -0
  429. package/public/assets/icon-light.svg +9 -0
  430. package/public/assets/logo-dark.svg +15 -0
  431. package/public/assets/logo-light.svg +15 -0
  432. package/scaffold/.mcp.json +7 -0
  433. package/scaffold/CLAUDE.md +368 -0
  434. package/scaffold/config/agent.json +55 -0
  435. package/scaffold/config/agent.ts +76 -0
  436. package/scaffold/config/agent.ts.example +89 -0
  437. package/scaffold/config/alerts.yaml +23 -0
  438. package/scaffold/config/allowlist.yaml.example +25 -0
  439. package/scaffold/config/caller-id-map.yaml +46 -0
  440. package/scaffold/config/collective.yaml +49 -0
  441. package/scaffold/config/company.json +20 -0
  442. package/scaffold/config/known-agents.json +6 -0
  443. package/scaffold/config/learning.yaml +55 -0
  444. package/scaffold/config/model-routing.yaml.example +104 -0
  445. package/scaffold/config/org.yaml +25 -0
  446. package/scaffold/config/orgmail.yaml.example +19 -0
  447. package/scaffold/config/recovery.yaml +72 -0
  448. package/scaffold/config/secrets.yaml +27 -0
  449. package/scaffold/config/slack.yaml.example +35 -0
  450. package/scaffold/config/telegram.yaml.example +38 -0
  451. package/scaffold/config/voice.yaml.example +89 -0
  452. package/scaffold/config/whatsapp.yaml.example +39 -0
  453. package/schedules/README.md +49 -0
  454. package/schedules/triggers/backlog-executor.md +102 -0
  455. package/schedules/triggers/brand-steward.md +72 -0
  456. package/schedules/triggers/daily-evening-wrap.md +159 -0
  457. package/schedules/triggers/daily-midday-sweep.md +58 -0
  458. package/schedules/triggers/daily-morning-brief.md +55 -0
  459. package/schedules/triggers/directory-hygiene.md +81 -0
  460. package/schedules/triggers/dynamic-jobs.md +40 -0
  461. package/schedules/triggers/inbox-processor.md +115 -0
  462. package/schedules/triggers/meeting-action-capture.md +60 -0
  463. package/schedules/triggers/meeting-prep.md +69 -0
  464. package/schedules/triggers/messaging-inbound.md +50 -0
  465. package/schedules/triggers/org-pulse.md +24 -0
  466. package/schedules/triggers/quarterly-self-assessment.md +54 -0
  467. package/schedules/triggers/weekly-engineering-health.md +37 -0
  468. package/schedules/triggers/weekly-execution.md +65 -0
  469. package/schedules/triggers/weekly-hiring.md +53 -0
  470. package/schedules/triggers/weekly-priorities.md +38 -0
  471. package/schedules/triggers/weekly-strategic-memo.md +124 -0
  472. package/scripts/archive-email.sh +55 -0
  473. package/scripts/cadence/cadence-status.mjs +36 -0
  474. package/scripts/cadence/enqueue-cadence-tick.mjs +174 -0
  475. package/scripts/cadence/enqueue-cadence-tick.test.mjs +187 -0
  476. package/scripts/cadence/launchd-cadence-wrapper.sh +85 -0
  477. package/scripts/cadence/launchd-cloud-relay-wrapper.sh +95 -0
  478. package/scripts/cadence/launchd-socket-mode-wrapper.sh +95 -0
  479. package/scripts/ci/check-docs-accuracy.mjs +493 -0
  480. package/scripts/ci/check-docs-accuracy.test.mjs +409 -0
  481. package/scripts/ci/check-exports-exist.mjs +140 -0
  482. package/scripts/ci/check-files-exist.mjs +107 -0
  483. package/scripts/ci/check-no-build-artifacts.mjs +111 -0
  484. package/scripts/ci/check-no-build-artifacts.test.mjs +71 -0
  485. package/scripts/ci/check-no-confidential.mjs +198 -0
  486. package/scripts/ci/check-no-conflict-markers.mjs +169 -0
  487. package/scripts/ci/check-no-residual-identity.mjs +163 -0
  488. package/scripts/ci/check-no-residual-identity.test.mjs +89 -0
  489. package/scripts/ci/check-tarball-fidelity.mjs +205 -0
  490. package/scripts/ci/check-unresolved-tokens.mjs +83 -0
  491. package/scripts/ci/check.mjs +109 -0
  492. package/scripts/ci/check.test.mjs +194 -0
  493. package/scripts/ci/run-coverage.mjs +82 -0
  494. package/scripts/ci/run-tests.mjs +71 -0
  495. package/scripts/cloud-relay/README.md +59 -0
  496. package/scripts/cloud-relay/index.mjs +233 -0
  497. package/scripts/cloud-relay/package.json +15 -0
  498. package/scripts/cloud-relay/railway.json +13 -0
  499. package/scripts/cloud-relay/voice/README.md +94 -0
  500. package/scripts/cloud-relay/voice/package-lock.json +39 -0
  501. package/scripts/cloud-relay/voice/package.json +16 -0
  502. package/scripts/cloud-relay/voice/railway.json +13 -0
  503. package/scripts/cloud-relay/voice/server.mjs +532 -0
  504. package/scripts/collective/hook-runner.mjs +211 -0
  505. package/scripts/collective/hook-runner.test.mjs +90 -0
  506. package/scripts/collective/org-pulse.mjs +72 -0
  507. package/scripts/collective/org-sync.mjs +61 -0
  508. package/scripts/collective/recall.mjs +45 -0
  509. package/scripts/collective/who.mjs +30 -0
  510. package/scripts/comms-monitor.sh +288 -0
  511. package/scripts/configure-whatsapp-sandbox.sh +201 -0
  512. package/scripts/continuous-monitor.sh +91 -0
  513. package/scripts/cost/fleet-digest.mjs +407 -0
  514. package/scripts/cost/fleet-digest.test.mjs +207 -0
  515. package/scripts/cost/track-claude-usage.mjs +169 -0
  516. package/scripts/daemon/agent-daemon.mjs +989 -0
  517. package/scripts/daemon/agent-daemon.test.mjs +525 -0
  518. package/scripts/daemon/cadence-consumer-governance.test.mjs +220 -0
  519. package/scripts/daemon/cadence-consumer.mjs +1080 -0
  520. package/scripts/daemon/cadence-consumer.test.mjs +770 -0
  521. package/scripts/daemon/cadence-handlers.mjs +1121 -0
  522. package/scripts/daemon/cadence-handlers.test.mjs +617 -0
  523. package/scripts/daemon/classifier.mjs +704 -0
  524. package/scripts/daemon/classifier.test.mjs +238 -0
  525. package/scripts/daemon/classify-kind.mjs +54 -0
  526. package/scripts/daemon/classify-kind.test.mjs +40 -0
  527. package/scripts/daemon/context-compiler.mjs +605 -0
  528. package/scripts/daemon/context-compiler.test.mjs +300 -0
  529. package/scripts/daemon/dispatcher-cooldown.test.mjs +122 -0
  530. package/scripts/daemon/dispatcher-governance.test.mjs +886 -0
  531. package/scripts/daemon/dispatcher.mjs +1516 -0
  532. package/scripts/daemon/health.mjs +72 -0
  533. package/scripts/daemon/inbox-deferral.mjs +210 -0
  534. package/scripts/daemon/inbox-deferral.test.mjs +242 -0
  535. package/scripts/daemon/integration.test.mjs +149 -0
  536. package/scripts/daemon/launchd-wrapper-generic.sh +96 -0
  537. package/scripts/daemon/launchd-wrapper-slack-events.sh +37 -0
  538. package/scripts/daemon/launchd-wrapper.sh +91 -0
  539. package/scripts/daemon/lib/session-router.mjs +274 -0
  540. package/scripts/daemon/lib/session-router.test.mjs +295 -0
  541. package/scripts/daemon/maestro-daemon.mjs +275 -0
  542. package/scripts/daemon/prompt-builder.mjs +685 -0
  543. package/scripts/daemon/prompt-builder.test.mjs +213 -0
  544. package/scripts/daemon/responder.mjs +854 -0
  545. package/scripts/daemon/session-lock.mjs +721 -0
  546. package/scripts/daemon/session-lock.test.mjs +252 -0
  547. package/scripts/daemon/session-outcomes.mjs +640 -0
  548. package/scripts/daemon/session-outcomes.test.mjs +533 -0
  549. package/scripts/daemon/typing-registry.mjs +90 -0
  550. package/scripts/daemon/typing-registry.test.mjs +77 -0
  551. package/scripts/daemon/voice-webhook-server.mjs +804 -0
  552. package/scripts/decisions/capture-decision.mjs +116 -0
  553. package/scripts/disclosure_assessment.py +873 -0
  554. package/scripts/disclosure_boundaries.py +562 -0
  555. package/scripts/email-signature-principal.html +52 -0
  556. package/scripts/email-signature.html +60 -0
  557. package/scripts/email_quote_thread.py +167 -0
  558. package/scripts/email_thread_dedup.py +362 -0
  559. package/scripts/emergency-stop.sh +81 -0
  560. package/scripts/healthcheck.sh +116 -0
  561. package/scripts/hooks/block-mcp-cohort-send.sh +15 -0
  562. package/scripts/hooks/block-mcp-slack-send.sh +7 -0
  563. package/scripts/hooks/post-action-log.sh +126 -0
  564. package/scripts/hooks/pre-send-audit.sh +174 -0
  565. package/scripts/hooks/pre-send-audit.test.mjs +215 -0
  566. package/scripts/hooks/session-end-log.sh +27 -0
  567. package/scripts/hooks/session-start-banner.sh +115 -0
  568. package/scripts/huddle/audio-bridge.mjs +664 -0
  569. package/scripts/huddle/boot-slack-cdp.sh +102 -0
  570. package/scripts/huddle/huddle-controller.mjs +942 -0
  571. package/scripts/huddle/huddle-server.mjs +1229 -0
  572. package/scripts/huddle/launch-slack.sh +232 -0
  573. package/scripts/huddle/openai-realtime-bridge.mjs +462 -0
  574. package/scripts/huddle/package-lock.json +62 -0
  575. package/scripts/huddle/package.json +22 -0
  576. package/scripts/huddle/setup-audio.sh +239 -0
  577. package/scripts/huddle/start-call.mjs +318 -0
  578. package/scripts/huddle/test-pipeline.mjs +263 -0
  579. package/scripts/learning/consolidate-skills.mjs +72 -0
  580. package/scripts/learning/session-search.mjs +125 -0
  581. package/scripts/llm_email_dedup.py +442 -0
  582. package/scripts/local-triggers/generate-plists.sh +432 -0
  583. package/scripts/local-triggers/generate-plists.test.mjs +413 -0
  584. package/scripts/local-triggers/install-all.sh +49 -0
  585. package/scripts/local-triggers/plists/.gitkeep +0 -0
  586. package/scripts/local-triggers/run-trigger.sh +63 -0
  587. package/scripts/local-triggers/templates/rag-reindex.plist.template +47 -0
  588. package/scripts/local-triggers/templates/voice-relay-poller.plist.template +54 -0
  589. package/scripts/local-triggers/templates/voice-tunnel.plist.template +55 -0
  590. package/scripts/local-triggers/templates/voice-webhook.plist.template +51 -0
  591. package/scripts/maintenance/backup-to-cloud.sh +124 -0
  592. package/scripts/maintenance/health-check.sh +377 -0
  593. package/scripts/media-generation/README.md +105 -0
  594. package/scripts/media-generation/gemini-image-client.mjs +173 -0
  595. package/scripts/media-generation/generate-assets.mjs +289 -0
  596. package/scripts/media-generation/veo-video-client.mjs +219 -0
  597. package/scripts/org/send-orgmail.mjs +227 -0
  598. package/scripts/outbound-dedup-cleanup.sh +43 -0
  599. package/scripts/outbound-dedup.sh +477 -0
  600. package/scripts/outbound_dedup.py +115 -0
  601. package/scripts/parse-voice-transcript.mjs +481 -0
  602. package/scripts/pdf-generation/README.md +63 -0
  603. package/scripts/pdf-generation/build-document.mjs +247 -0
  604. package/scripts/pdf-generation/templates/board-pack.latex +136 -0
  605. package/scripts/pdf-generation/templates/corporate-letter.latex +126 -0
  606. package/scripts/pdf-generation/templates/memo.latex +114 -0
  607. package/scripts/poll-slack-events.sh +35 -0
  608. package/scripts/poller/calendar-poller.mjs +12 -0
  609. package/scripts/poller/gmail-poller.mjs +192 -0
  610. package/scripts/poller/imap-client.mjs +289 -0
  611. package/scripts/poller/inbox-scan-poller.mjs +156 -0
  612. package/scripts/poller/inbox-scan-poller.test.mjs +231 -0
  613. package/scripts/poller/index.mjs +73 -0
  614. package/scripts/poller/intra-session-check.mjs +285 -0
  615. package/scripts/poller/lib/cloud-relay-dedup.mjs +88 -0
  616. package/scripts/poller/lib/cloud-relay-dedup.test.mjs +133 -0
  617. package/scripts/poller/lib/slash-command-handlers.mjs +177 -0
  618. package/scripts/poller/secondary-gmail-poller.mjs +132 -0
  619. package/scripts/poller/slack-cloud-relay-client.mjs +368 -0
  620. package/scripts/poller/slack-poller.mjs +854 -0
  621. package/scripts/poller/slack-socket-mode.mjs +917 -0
  622. package/scripts/poller/slack-socket-mode.test.mjs +753 -0
  623. package/scripts/poller/trigger.mjs +75 -0
  624. package/scripts/poller/utils.mjs +371 -0
  625. package/scripts/poller/voice-cloud-relay-client.mjs +179 -0
  626. package/scripts/poller/voice-poller.mjs +236 -0
  627. package/scripts/poller-launchd/install.sh +66 -0
  628. package/scripts/poller-launchd/poller.plist.template +40 -0
  629. package/scripts/poller-launchd/whatsapp-handler.plist.template +39 -0
  630. package/scripts/post-interaction-indexer.py +1598 -0
  631. package/scripts/pre-draft-context.py +994 -0
  632. package/scripts/pre_draft_lookup.py +258 -0
  633. package/scripts/rag/build-index.mjs +47 -0
  634. package/scripts/rag/ingest.mjs +111 -0
  635. package/scripts/rag/search.mjs +119 -0
  636. package/scripts/rag-indexer.py +629 -0
  637. package/scripts/restore-from-backup.sh +248 -0
  638. package/scripts/restore-from-backup.test.mjs +178 -0
  639. package/scripts/resume-operations.sh +80 -0
  640. package/scripts/search-secondary-inbox.py +181 -0
  641. package/scripts/secondary-inbox-poller.py +437 -0
  642. package/scripts/self-optimization/compute-metrics.py +398 -0
  643. package/scripts/send-email-as-principal.py +369 -0
  644. package/scripts/send-email-threaded.py +392 -0
  645. package/scripts/send-email-with-attachment.py +377 -0
  646. package/scripts/send-email.sh +131 -0
  647. package/scripts/send-sms.sh +175 -0
  648. package/scripts/send-whatsapp.sh +292 -0
  649. package/scripts/session-start.sh +106 -0
  650. package/scripts/setup/boot-claude-session.sh +94 -0
  651. package/scripts/setup/configure-macos.sh +674 -0
  652. package/scripts/setup/configure-twilio-sip-trunk.mjs +207 -0
  653. package/scripts/setup/configure-voice-tunnel.mjs +182 -0
  654. package/scripts/setup/generate-agent-env.mjs +92 -0
  655. package/scripts/setup/generate-agent-package-json.mjs +222 -0
  656. package/scripts/setup/generate-agent-package-json.test.mjs +143 -0
  657. package/scripts/setup/generate-autonomy.mjs +60 -0
  658. package/scripts/setup/generate-backlog.mjs +101 -0
  659. package/scripts/setup/generate-cadences.mjs +92 -0
  660. package/scripts/setup/generate-capability.mjs +162 -0
  661. package/scripts/setup/generate-charter.mjs +76 -0
  662. package/scripts/setup/generate-comms.mjs +59 -0
  663. package/scripts/setup/generate-company.mjs +92 -0
  664. package/scripts/setup/init-agent.sh +547 -0
  665. package/scripts/setup/init-agent.test.mjs +151 -0
  666. package/scripts/setup/init-archetype.mjs +74 -0
  667. package/scripts/setup/init-backup.mjs +54 -0
  668. package/scripts/setup/init-cadence-bus.mjs +60 -0
  669. package/scripts/setup/init-channel-bus.mjs +46 -0
  670. package/scripts/setup/init-cost-tracking.mjs +45 -0
  671. package/scripts/setup/init-decision-capture.mjs +66 -0
  672. package/scripts/setup/init-known-agents.mjs +57 -0
  673. package/scripts/setup/init-learning.mjs +70 -0
  674. package/scripts/setup/init-memory-executive.mjs +45 -0
  675. package/scripts/setup/init-model-router.mjs +124 -0
  676. package/scripts/setup/init-rag.mjs +174 -0
  677. package/scripts/setup/init-session-router.mjs +38 -0
  678. package/scripts/setup/init-slack-socket-mode.mjs +260 -0
  679. package/scripts/setup/init-telegram.mjs +165 -0
  680. package/scripts/setup/init-voice-realtime.mjs +204 -0
  681. package/scripts/setup/init-whatsapp-baileys.mjs +77 -0
  682. package/scripts/setup/install-dev-tools.sh +150 -0
  683. package/scripts/setup/lib/install-plist.mjs +95 -0
  684. package/scripts/setup/migrate-agent-to-sot.mjs +192 -0
  685. package/scripts/setup/render-environment-yaml.mjs +133 -0
  686. package/scripts/slack-events-ctl.sh +177 -0
  687. package/scripts/slack-events-server.mjs +1045 -0
  688. package/scripts/slack-react.mjs +89 -0
  689. package/scripts/slack-responded.sh +232 -0
  690. package/scripts/slack-send.sh +287 -0
  691. package/scripts/slack-typing.mjs +196 -0
  692. package/scripts/slack-upload-v2.py +95 -0
  693. package/scripts/sms-handler.mjs +450 -0
  694. package/scripts/spawn-session.sh +120 -0
  695. package/scripts/sync-protocol.mjs +217 -0
  696. package/scripts/system-verify.sh +184 -0
  697. package/scripts/test-email-thread-dedup.py +239 -0
  698. package/scripts/test-information-barriers.py +484 -0
  699. package/scripts/test-llm-email-dedup.py +251 -0
  700. package/scripts/test-pre-draft-integration.py +203 -0
  701. package/scripts/test-rag-phase2.sh +442 -0
  702. package/scripts/test-rag-search.sh +251 -0
  703. package/scripts/test-voice-parser.mjs +316 -0
  704. package/scripts/user-context-search.py +659 -0
  705. package/scripts/validate_outbound.py +1504 -0
  706. package/scripts/watchdog/ai.maestro.memory-watchdog.plist +41 -0
  707. package/scripts/watchdog/force-reboot.sh +157 -0
  708. package/scripts/watchdog/memory-watchdog.sh +473 -0
  709. package/scripts/whatsapp-handler.mjs +538 -0
  710. package/teams/desktop-operations.yaml +34 -0
  711. package/teams/executive-office.yaml +27 -0
  712. package/teams/legal-and-regulatory.yaml +24 -0
  713. package/teams/platform-and-engineering.yaml +23 -0
  714. package/teams/strategy-and-growth.yaml +29 -0
  715. package/workflows/continuous/backlog-executor.yaml +141 -0
  716. package/workflows/continuous/inbound-monitor.yaml +168 -0
  717. package/workflows/daily/applicant-triage.yaml +197 -0
  718. package/workflows/daily/comms-triage.yaml +80 -0
  719. package/workflows/daily/evening-wrap.yaml +105 -0
  720. package/workflows/daily/morning-brief.yaml +164 -0
  721. package/workflows/daily/slack-followup-sweep.yaml +87 -0
  722. package/workflows/event-driven/README.md +50 -0
  723. package/workflows/event-driven/agent-failure-investigation.yaml +137 -0
  724. package/workflows/event-driven/pr-review.yaml +107 -0
  725. package/workflows/monthly/board-readiness.yaml +76 -0
  726. package/workflows/quarterly/strategic-scenario-analysis.yaml +85 -0
  727. package/workflows/session-protocol.md +171 -0
  728. package/workflows/weekly/engineering-health.yaml +154 -0
  729. package/workflows/weekly/hiring-review.yaml +169 -0
  730. package/workflows/weekly/rollup-pipeline-review.yaml +76 -0
  731. package/workflows/weekly/strategic-memo.yaml +79 -0
@@ -0,0 +1,1516 @@
1
+ #!/usr/bin/env node
2
+ // Dispatcher — manages concurrent claude --print sessions
3
+ // Spawns child processes, manages queue, enforces concurrency cap
4
+
5
+ import { spawn } from "child_process";
6
+ import { appendFileSync, mkdirSync, writeFileSync, readFileSync, renameSync, existsSync, readdirSync, unlinkSync } from "fs";
7
+ import { randomUUID } from "crypto";
8
+ import { join, dirname } from "path";
9
+ import { releaseLock, releaseThreadLock, releaseRequestClaim, claimItem, releaseItemClaim } from "./session-lock.mjs";
10
+ import { promoteDeferred } from "./inbox-deferral.mjs";
11
+ import { recordSession } from "./health.mjs";
12
+ import { startTyping, stopTyping } from "./typing-registry.mjs";
13
+ // Permission scoping (security CRITICAL / audit H1). sessionPermissionArgs()
14
+ // returns ["--dangerously-skip-permissions"] by default (byte-for-byte the prior
15
+ // hardcoded literal) OR a scoped ["--allowedTools", "<list>"] when the operator
16
+ // sets MAESTRO_SCOPED_PERMISSIONS=1. Spawns route through it so opted-in scoping
17
+ // is actually honoured instead of being bypassed on the user-facing path.
18
+ import { sessionPermissionArgs } from "../../lib/session-permissions.mjs";
19
+
20
+ // WS4 — resource governance + cross-session 429 breaker + daily budget cap.
21
+ // Every spawn is gated on admission (ADMIT/QUEUE/DEFER) and the shared rate
22
+ // breaker so the agent throttles instead of bricking and never exceeds the
23
+ // host. All three are injectable (setGovernanceForTests) so the dispatcher
24
+ // stays hermetically testable.
25
+ import * as resourceGovernor from "../../lib/resource-governor.mjs";
26
+ import * as rateGuardModule from "../../lib/rate-guard.mjs";
27
+ import * as budgetGuardModule from "../../lib/budget-guard.mjs";
28
+
29
+ let governor = resourceGovernor;
30
+ let rateGuard = rateGuardModule;
31
+ let budgetGuard = budgetGuardModule;
32
+ const RATE_PROVIDER = process.env.MAESTRO_RATE_PROVIDER || "anthropic";
33
+
34
+ /**
35
+ * Test seam: swap the governance modules for fakes. Each arg is optional;
36
+ * omitted modules keep the real implementation. Returns a restore fn.
37
+ */
38
+ export function setGovernanceForTests({ governor: g, rateGuard: r, budgetGuard: b } = {}) {
39
+ const prev = { governor, rateGuard, budgetGuard };
40
+ if (g) governor = g;
41
+ if (r) rateGuard = r;
42
+ if (b) budgetGuard = b;
43
+ return () => { governor = prev.governor; rateGuard = prev.rateGuard; budgetGuard = prev.budgetGuard; };
44
+ }
45
+
46
+ // Indirection so a test can capture the EXACT argv handed to `claude` by the
47
+ // real resume spawner (defaultSpawnResume) without launching a child process.
48
+ // Production default is the imported `spawn`, so behaviour is unchanged.
49
+ let _spawn = spawn;
50
+ /** Test seam: replace child_process.spawn. Returns a restore fn. */
51
+ export function setSpawnForTests(fn) {
52
+ const prev = _spawn;
53
+ _spawn = typeof fn === "function" ? fn : spawn;
54
+ return () => { _spawn = prev; };
55
+ }
56
+
57
+ /**
58
+ * Resolve the current admission decision for a (source) under live host load,
59
+ * folding in the daily-budget essential-only mode. Best-effort: any throw is
60
+ * treated as ADMIT (fail-open-for-work — Invariant: never drop/stall work on a
61
+ * governance bug).
62
+ */
63
+ function admitFor(source, priority) {
64
+ try {
65
+ let mode = null;
66
+ try {
67
+ if (budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR }).essentialOnly) mode = "essential-only";
68
+ } catch { /* budget read best-effort */ }
69
+ return governor.admit({ source, priority, mode }, governor.defaultDeps({ agentRoot: AGENT_REPO_DIR }));
70
+ } catch {
71
+ return { decision: "ADMIT", reason: "governor-error-fail-open", snapshot: {} };
72
+ }
73
+ }
74
+
75
+ /** Is the shared 429 breaker currently open for our provider? */
76
+ function rateBlocked() {
77
+ try {
78
+ const r = rateGuard.checkRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR });
79
+ return r.allowed ? null : r;
80
+ } catch {
81
+ return null; // fail-open-for-work
82
+ }
83
+ }
84
+
85
+ const AGENT_REPO_DIR = process.env.AGENT_DIR || join(new URL(".", import.meta.url).pathname, "../..");
86
+ // Resolve the claude binary against the agent's PATH (not launchd's bare
87
+ // env). Without this, every daemon-spawned `claude --print` exits ENOENT.
88
+ import { resolveClaudeBin, augmentedPath, daemonClaudeArgs } from "../../lib/claude-bin.mjs";
89
+ const CLAUDE_BIN = resolveClaudeBin();
90
+
91
+ // Model router — opt-in. When config/model-routing.yaml is present in the
92
+ // agent repo, each spawn is routed to the cheapest backend that satisfies
93
+ // the request's capability needs. When absent, the resolver returns null
94
+ // and we preserve the current Claude-CLI-on-Max-subscription behaviour.
95
+ //
96
+ // Two routing paths now coexist (SPEC §3 — additive, byte-compatible):
97
+ // - v1 (legacy): resolveBackend → modelFlagFor → envForSpawn. Preserved
98
+ // verbatim for v1 configs (no schema_version) and as the safety net.
99
+ // - v2 (schema_version: 2): resolveChain → RouteDecision. The decision carries
100
+ // the chosen backend, the session-retarget env, the spawn knobs, the
101
+ // ordered failover chain, an estimated cost, and a decision_id that joins
102
+ // the ledger row ↔ resume marker ↔ routing audit. We build the argv + child
103
+ // env from the decision via the execution layer's pure builders
104
+ // (buildSpawnArgs / buildChildEnv from lib/model-router/spawn.mjs) so the
105
+ // §7.3 child-env scrubbing + argv contract match spawnRouted exactly.
106
+ //
107
+ // NOTE on spawnRouted: the daemon's spawnSession() must register the child
108
+ // process SYNCHRONOUSLY (tests + getStatus() observe active_sessions right after
109
+ // dispatch, and the child's close/error/timeout handlers own cooldowns, resume
110
+ // markers, lock release, and the cost ledger). spawnRouted owns an ASYNC
111
+ // fast-fail failover loop that resolves a Promise — incompatible with that
112
+ // synchronous, externally-driven-close contract here without a rewrite that
113
+ // risks regressions. We therefore reuse spawnRouted's pure argv/env BUILDERS at
114
+ // this seam and keep the existing synchronous _spawn + handlers; the async
115
+ // failover loop is the right fit for the scripted one-shot/cadence sites (which
116
+ // already drive their own close). Cross-backend failover at the dispatcher is a
117
+ // precise follow-up (see the spawn-site checklist in the WS summary).
118
+ import {
119
+ loadRoutingConfig,
120
+ resolveBackend,
121
+ requestFromClassifierResult,
122
+ modelFlagFor,
123
+ resolveChain,
124
+ } from "../../lib/model-router.mjs";
125
+ import { buildSpawnArgs, buildChildEnv } from "../../lib/model-router/spawn.mjs";
126
+ import { budgetLadder, spawnKnobsFor } from "../../lib/model-router/economics.mjs";
127
+ // Observability spine (WS — diagnostics). The dispatcher emits `dispatched` at
128
+ // spawn and `session_opened`/`session_closed` from the existing close hook,
129
+ // attaching the interaction's trace_id (handed off explicitly on item.trace_id —
130
+ // AsyncLocalStorage cannot cross the proc-event boundary) + the v2 decision_id.
131
+ // emitEvent is fail-open (never throws), so this is purely additive telemetry.
132
+ import { emitEvent, EVENT_TYPES } from "../../lib/diagnostics/events.mjs";
133
+ // Lazy + cached so a misconfigured YAML doesn't break agents that didn't
134
+ // opt in. The cache is invalidated only on daemon restart.
135
+ let _routingConfigCache;
136
+ function getRoutingConfig() {
137
+ if (_routingConfigCache === undefined) {
138
+ try {
139
+ _routingConfigCache = loadRoutingConfig(AGENT_REPO_DIR) || null;
140
+ } catch (err) {
141
+ console.warn(`[dispatcher] model-router config invalid, falling back to Anthropic CLI: ${err.message}`);
142
+ _routingConfigCache = null;
143
+ }
144
+ }
145
+ return _routingConfigCache;
146
+ }
147
+ const MAX_CONCURRENT = parseInt(process.env.DAEMON_MAX_CONCURRENT || "10", 10);
148
+ const RESERVED_INBOX_SLOTS = 3; // Always keep 3 slots free for real-time inbox items
149
+
150
+ // Timeouts — tiered by model and source
151
+ const SONNET_INBOX_TIMEOUT = 10 * 60 * 1000; // 10 min (inbox — user expects a reply)
152
+ const SONNET_BACKLOG_TIMEOUT = 30 * 60 * 1000; // 30 min (backlog tasks need more time)
153
+ const OPUS_INBOX_TIMEOUT = 45 * 60 * 1000; // 45 min (complex inbox — CEO requests, research)
154
+ const OPUS_BACKLOG_TIMEOUT = 12 * 60 * 60 * 1000; // 12 hours (ultra-complex backlog — deep work)
155
+
156
+ // Legacy aliases for compatibility
157
+ const SONNET_TIMEOUT = SONNET_INBOX_TIMEOUT;
158
+ const OPUS_TIMEOUT = OPUS_INBOX_TIMEOUT;
159
+
160
+ const activeSessions = new Map(); // sessionId -> { process, item, startTime, model, source }
161
+ const priorityQueue = []; // critical/high items
162
+ const normalQueue = []; // normal items
163
+ let sessionCounter = 0;
164
+
165
+ // Tracks sessions whose proc.on("error") handler has already fired.
166
+ // Prevents double-counting + double-cleanup when a spawn failure (ENOENT,
167
+ // EACCES, ETIMEDOUT) triggers both "error" and a trailing "close" event.
168
+ // See ib-20260416-daemon-etimedout-failed-event + cycle 135 memo.
169
+ const spawnErrorHandled = new Set();
170
+
171
+ // Backlog dedup: track which items have active sessions to prevent retry storms
172
+ const activeBacklogKeys = new Set(); // backlog item key -> true (while session is running)
173
+ const backlogRetryCount = new Map(); // backlog item key -> number of times dispatched
174
+ const MAX_BACKLOG_RETRIES = 6; // Max retries before skipping (was 3 — too aggressive)
175
+
176
+ // Post-completion cooldown — once a session has run on a backlog item, don't
177
+ // re-dispatch it until N hours later. Without this, every 2-min backlog
178
+ // sweep re-dispatches the same items because the daemon has no signal
179
+ // that the underlying work was actually completed (sessions exit 0 even
180
+ // when they only "looked at" the item). 53 redundant spawns/day per item
181
+ // was the observed rate before this fix.
182
+ const SUCCESS_COOLDOWN_MS = 4 * 60 * 60 * 1000; // 4h after exit 0
183
+ const FAILURE_COOLDOWN_MS = 30 * 60 * 1000; // 30m after non-zero exit
184
+ const backlogCooldownUntil = new Map(); // key -> epoch ms
185
+ const COOLDOWN_STATE_PATH = join(AGENT_REPO_DIR, "state/sessions/backlog-cooldowns.json");
186
+
187
+ // Persist cooldown state across daemon restarts so a freshly-started
188
+ // daemon doesn't immediately re-dispatch items it just completed.
189
+ function loadCooldowns() {
190
+ try {
191
+ const body = readFileSync(COOLDOWN_STATE_PATH, "utf-8");
192
+ const data = JSON.parse(body);
193
+ const now = Date.now();
194
+ for (const [key, until] of Object.entries(data || {})) {
195
+ if (typeof until === "number" && until > now) {
196
+ backlogCooldownUntil.set(key, until);
197
+ }
198
+ }
199
+ } catch { /* file missing or malformed — start fresh */ }
200
+ }
201
+ function saveCooldowns() {
202
+ try {
203
+ const obj = {};
204
+ for (const [k, v] of backlogCooldownUntil) obj[k] = v;
205
+ mkdirSync(dirname(COOLDOWN_STATE_PATH), { recursive: true });
206
+ writeFileSync(COOLDOWN_STATE_PATH, JSON.stringify(obj, null, 2) + "\n");
207
+ } catch { /* best-effort */ }
208
+ }
209
+ loadCooldowns();
210
+
211
+ /**
212
+ * Parse the `claude --print --output-format json` result object out of a
213
+ * captured stdout STRING (the dispatcher buffers stdout in memory) to recover
214
+ * the run's REAL token usage + authoritative cost for the ledger (recovery C1).
215
+ * Returns { ok, inputTokens, outputTokens, cacheReadTokens?, totalCostUsd?,
216
+ * model? } or { ok:false, reason } — never throws, never fabricates counts.
217
+ */
218
+ export function parseUsageFromText(text) {
219
+ const trimmed = (text || "").trim();
220
+ if (!trimmed) return { ok: false, reason: "empty-stdout" };
221
+ let obj = null;
222
+ try { obj = JSON.parse(trimmed); }
223
+ catch {
224
+ // Tolerant path: a leading banner/log line can precede the JSON tail. The
225
+ // CLI's result object is the FIRST balanced top-level object, so slice from
226
+ // the first "{" to the last "}". (Using lastIndexOf("{") would wrongly grab
227
+ // a nested object's opening brace.)
228
+ const start = trimmed.indexOf("{");
229
+ const end = trimmed.lastIndexOf("}");
230
+ if (start !== -1 && end !== -1 && end > start) {
231
+ try { obj = JSON.parse(trimmed.slice(start, end + 1)); } catch { obj = null; }
232
+ }
233
+ }
234
+ if (!obj || typeof obj !== "object") return { ok: false, reason: "no-json-object" };
235
+ const usage = obj.usage && typeof obj.usage === "object" ? obj.usage : null;
236
+ if (!usage) return { ok: false, reason: "no-usage-field" };
237
+ const inputTokens = Number(usage.input_tokens);
238
+ const outputTokens = Number(usage.output_tokens);
239
+ if (!Number.isFinite(inputTokens) || !Number.isFinite(outputTokens)) {
240
+ return { ok: false, reason: "non-numeric-tokens" };
241
+ }
242
+ const out = { ok: true, inputTokens, outputTokens };
243
+ const cacheRead = Number(usage.cache_read_input_tokens);
244
+ if (Number.isFinite(cacheRead)) out.cacheReadTokens = cacheRead;
245
+ const totalCost = Number(obj.total_cost_usd);
246
+ if (Number.isFinite(totalCost) && totalCost >= 0) out.totalCostUsd = totalCost;
247
+ if (typeof obj.model === "string" && obj.model) {
248
+ out.model = /opus/i.test(obj.model) ? "opus" : /haiku/i.test(obj.model) ? "haiku" : "sonnet";
249
+ }
250
+ return out;
251
+ }
252
+
253
+ /**
254
+ * Append a TRUTHFUL cost-ledger row for a finished dispatcher session via
255
+ * scripts/cost/track-claude-usage.mjs (source "dispatcher"). Real token counts
256
+ * are parsed from the run's --output-format json stdout; on a parse failure we
257
+ * pass NO token flags (tracker records 0) rather than fabricate zeros, and the
258
+ * caller logs the gap. Best-effort + detached; never blocks the close path.
259
+ */
260
+ function recordDispatcherCost({ stdout, model, durationMs, exitCode, decisionId }) {
261
+ const usage = parseUsageFromText(stdout);
262
+ try {
263
+ const trackerPath = join(AGENT_REPO_DIR, "scripts/cost/track-claude-usage.mjs");
264
+ if (!existsSync(trackerPath)) return usage;
265
+ const trackerArgs = [
266
+ trackerPath, "record",
267
+ "--cadence", "inbox",
268
+ "--source", "dispatcher",
269
+ "--model", usage.model || model || "sonnet",
270
+ "--duration-ms", String(durationMs),
271
+ "--exit", String(exitCode),
272
+ ];
273
+ // Carry the v2 RouteDecision id onto the ledger row so a spend row joins its
274
+ // routing decision + resume marker (SPEC §4.7). The tracker captures it as a
275
+ // generic flag; absent on the v1 / no-config paths.
276
+ if (decisionId) trackerArgs.push("--decision-id", String(decisionId));
277
+ if (usage.ok) {
278
+ trackerArgs.push("--input-tokens", String(usage.inputTokens));
279
+ trackerArgs.push("--output-tokens", String(usage.outputTokens));
280
+ if (usage.cacheReadTokens != null) trackerArgs.push("--cache-read-tokens", String(usage.cacheReadTokens));
281
+ if (usage.totalCostUsd != null) trackerArgs.push("--total-cost-usd", String(usage.totalCostUsd));
282
+ }
283
+ spawn(process.execPath, trackerArgs, {
284
+ stdio: "ignore",
285
+ env: { ...process.env, AGENT_ROOT: AGENT_REPO_DIR, AGENT_DIR: AGENT_REPO_DIR },
286
+ }).unref();
287
+ } catch { /* cost tracking is best-effort */ }
288
+ return usage;
289
+ }
290
+
291
+ function logDir() {
292
+ const dir = join(AGENT_REPO_DIR, "logs", "daemon");
293
+ mkdirSync(dir, { recursive: true });
294
+ return dir;
295
+ }
296
+
297
+ function sessionLogDir() {
298
+ const dir = join(AGENT_REPO_DIR, "logs", "daemon", "sessions");
299
+ mkdirSync(dir, { recursive: true });
300
+ return dir;
301
+ }
302
+
303
+ function today() {
304
+ return new Date().toISOString().split("T")[0];
305
+ }
306
+
307
+ function logSession(entry) {
308
+ const path = join(logDir(), `${today()}-sessions.jsonl`);
309
+ appendFileSync(path, JSON.stringify({ timestamp: new Date().toISOString(), ...entry }) + "\n");
310
+ }
311
+
312
+ const ACTIVE_PATH = join(AGENT_REPO_DIR, "state", "sessions", "active.json");
313
+
314
+ // ---------------------------------------------------------------------------
315
+ // WS4 — in-flight session resume markers
316
+ // ---------------------------------------------------------------------------
317
+ // A marker is written to state/sessions/resume-pending/<sessionId>.json right
318
+ // after spawn and DELETED on a clean close. Its presence after a crash/reboot
319
+ // means "this session was mid-flight" — resetActiveSessions() reconciles them
320
+ // (re-dispatching `claude --print --session-id <claudeSessionId> <prompt>`,
321
+ // the SAME continuation mechanism responder.mjs uses — NOT `--resume`) within a
322
+ // freshness window, instead of blindly wiping the slate. 3 strikes → the queue
323
+ // item is marked status:blocked. This is the "never brick / never drop work"
324
+ // recovery path for the dispatcher.
325
+ const RESUME_PENDING_DIR = join(AGENT_REPO_DIR, "state", "sessions", "resume-pending");
326
+ // Sessions whose marker is older than this are too stale to resume meaningfully
327
+ // (the work context has moved on); reconcile blocks them rather than re-running.
328
+ const RESUME_FRESHNESS_MS = parseInt(process.env.MAESTRO_RESUME_FRESHNESS_MS || String(6 * 60 * 60 * 1000), 10);
329
+ const RESUME_MAX_ATTEMPTS = 3;
330
+
331
+ function resumePendingPath(sessionId) {
332
+ return join(RESUME_PENDING_DIR, `${sessionId}.json`);
333
+ }
334
+
335
+ /**
336
+ * Write the resume marker for a freshly-spawned session. Best-effort: a marker
337
+ * we can't write just means that session won't be auto-resumed after a crash
338
+ * (it still completes normally) — never throws into the spawn path.
339
+ */
340
+ function writeResumePending(sessionId, entry, claudeSessionId, modelFlag, decisionId) {
341
+ try {
342
+ mkdirSync(RESUME_PENDING_DIR, { recursive: true });
343
+ const marker = {
344
+ sessionId,
345
+ claudeSessionId: claudeSessionId || null,
346
+ itemRef: entry.item?.raw_ref || entry.item?.id || entry.item?.title || null,
347
+ itemId: entry.item?.id || null,
348
+ sourceFile: entry.item?.source_file || null,
349
+ source: entry.source,
350
+ transcriptPath: entry.transcriptPath || null,
351
+ summary: entry.classResult?.summary || null,
352
+ model: entry.classResult?.model || "sonnet",
353
+ // The exact `--model` flag value the original spawn used (post model-
354
+ // router resolution). Persisted so a reboot-resume re-spawns against the
355
+ // SAME backend, not just the coarse sonnet/opus class. (H1)
356
+ modelFlag: modelFlag || entry.classResult?.model || "sonnet",
357
+ // The v2 RouteDecision id (when routed through resolveChain). Persisted so
358
+ // a reboot-resume can correlate the resumed run with the original routing
359
+ // decision + ledger row (SPEC §4.8). Null on the v1 / no-config paths.
360
+ decision_id: decisionId || null,
361
+ // The original prompt is what makes a true resume possible: the resume
362
+ // path re-spawns `claude --print --session-id <id> <prompt>` (NOT
363
+ // `--resume`), mirroring responder.mjs's blessed continuation pattern.
364
+ // Without it a resumed spawn would have no work to do and exit 0, which
365
+ // the close handler would mistake for "done". (H1)
366
+ prompt: typeof entry.prompt === "string" ? entry.prompt : null,
367
+ priority: entry.classResult?.priority || "normal",
368
+ startedAt: Date.now(),
369
+ recoveryAttempts: typeof entry.recoveryAttempts === "number" ? entry.recoveryAttempts : 0,
370
+ };
371
+ const tmp = resumePendingPath(sessionId) + ".tmp";
372
+ writeFileSync(tmp, JSON.stringify(marker, null, 2));
373
+ renameSync(tmp, resumePendingPath(sessionId));
374
+ } catch (err) {
375
+ console.warn(`[dispatcher] Failed to write resume-pending marker for ${sessionId}: ${err.message}`);
376
+ }
377
+ }
378
+
379
+ /** Delete the resume marker on a clean close. Best-effort. */
380
+ function clearResumePending(sessionId) {
381
+ try { unlinkSync(resumePendingPath(sessionId)); } catch { /* already gone */ }
382
+ }
383
+
384
+ /**
385
+ * Mark a queue item blocked (after 3 failed resume attempts) so the backlog
386
+ * sweep stops re-dispatching it and an operator can see why. Edits the queue
387
+ * YAML in place by flipping the item's status to `blocked` and appending a
388
+ * reason comment. Best-effort + behavior-preserving: if we can't locate the
389
+ * item we no-op (the cooldown/retry caps still bound re-dispatch).
390
+ */
391
+ function markItemBlocked(marker, reason) {
392
+ try {
393
+ if (!marker.itemId || !marker.sourceFile) return false;
394
+ const queuePath = join(AGENT_REPO_DIR, "state", "queues", marker.sourceFile);
395
+ if (!existsSync(queuePath)) return false;
396
+ let body = readFileSync(queuePath, "utf-8");
397
+ // Find the item's id line and flip the nearest status: within its block.
398
+ const idRe = new RegExp(`(^|\\n)(\\s*)(-?\\s*)"?id"?:\\s*["']?${marker.itemId.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}["']?`, "m");
399
+ const m = idRe.exec(body);
400
+ if (!m) return false;
401
+ // Replace the first status: line after the id with status: blocked.
402
+ const after = body.slice(m.index);
403
+ const replacedAfter = after.replace(/("?status"?:\s*)["']?(open|in_progress|pending)["']?/, `$1blocked # WS4 recovery: ${reason}`);
404
+ if (replacedAfter === after) return false;
405
+ body = body.slice(0, m.index) + replacedAfter;
406
+ const tmp = queuePath + ".tmp";
407
+ writeFileSync(tmp, body);
408
+ renameSync(tmp, queuePath);
409
+ logSession({ event: "item_blocked_after_resume_strikes", item_id: marker.itemId, source_file: marker.sourceFile, reason });
410
+ return true;
411
+ } catch (err) {
412
+ console.warn(`[dispatcher] markItemBlocked failed: ${err.message}`);
413
+ return false;
414
+ }
415
+ }
416
+
417
+ /**
418
+ * Reset active.json + (on startup) reconcile in-flight session resume markers.
419
+ *
420
+ * Previous daemon instances may have left stale active.json entries (killed
421
+ * sessions whose close handlers never fired). The in-memory session map starts
422
+ * empty, so active.json is always cleared.
423
+ *
424
+ * WS4 — instead of *also* blindly discarding any work that was mid-flight when
425
+ * the box rebooted/lost power, when `opts.reconcile` is set we walk
426
+ * state/sessions/resume-pending/ and, for each marker still inside the
427
+ * freshness window, re-spawn `claude --print --session-id <claudeSessionId>
428
+ * <prompt>` (the responder's continuation pattern — NOT `--resume`) so the work
429
+ * continues where it left off. After RESUME_MAX_ATTEMPTS (3) the item is marked
430
+ * status:blocked with a reason rather than retried forever.
431
+ *
432
+ * Graceful shutdown passes no opts (just clears active.json); only startup
433
+ * reconciles, so we never re-spawn work we're deliberately stopping.
434
+ *
435
+ * @param {object} [opts] { reconcile?: boolean, spawnResume?: fn (tests) }
436
+ */
437
+ export function resetActiveSessions(opts = {}) {
438
+ try {
439
+ let staleCount = 0;
440
+ try {
441
+ const old = JSON.parse(readFileSync(ACTIVE_PATH, "utf-8"));
442
+ staleCount = Object.keys(old).length;
443
+ } catch {}
444
+ writeFileSync(ACTIVE_PATH, "{}");
445
+ if (staleCount > 0) {
446
+ console.log(`[dispatcher] Cleared ${staleCount} stale entries from active.json`);
447
+ }
448
+ } catch (err) {
449
+ console.warn(`[dispatcher] Failed to reset active.json: ${err.message}`);
450
+ }
451
+ if (opts.reconcile) {
452
+ try { return reconcileResumePending(opts); }
453
+ catch (err) { console.warn(`[dispatcher] resume reconcile failed: ${err.message}`); }
454
+ }
455
+ }
456
+
457
+ /**
458
+ * Walk resume-pending markers and re-dispatch / block each as appropriate.
459
+ * Exported so a test can drive it deterministically with an injected
460
+ * `spawnResume`. Returns a summary { resumed, blocked, expired, scanned }.
461
+ *
462
+ * @param {object} [opts] { spawnResume?: ({marker})=>void, now?: ()=>number }
463
+ */
464
+ export function reconcileResumePending(opts = {}) {
465
+ const now = (typeof opts.now === "function" ? opts.now : Date.now)();
466
+ const stats = { scanned: 0, resumed: 0, blocked: 0, expired: 0 };
467
+ let files = [];
468
+ try {
469
+ if (!existsSync(RESUME_PENDING_DIR)) return stats;
470
+ files = readdirSync(RESUME_PENDING_DIR).filter((f) => f.endsWith(".json") && !f.endsWith(".tmp"));
471
+ } catch { return stats; }
472
+
473
+ for (const file of files) {
474
+ const path = join(RESUME_PENDING_DIR, file);
475
+ stats.scanned++;
476
+ let marker;
477
+ try { marker = JSON.parse(readFileSync(path, "utf-8")); }
478
+ catch { try { unlinkSync(path); } catch { /* */ } continue; }
479
+
480
+ const age = now - (marker.startedAt || 0);
481
+ const attempts = (marker.recoveryAttempts || 0);
482
+
483
+ // 3 strikes → block the item, drop the marker.
484
+ if (attempts >= RESUME_MAX_ATTEMPTS) {
485
+ markItemBlocked(marker, `exceeded ${RESUME_MAX_ATTEMPTS} resume attempts`);
486
+ logSession({ event: "resume_blocked", sessionId: marker.sessionId, item_id: marker.itemId, recovery_attempts: attempts });
487
+ try { unlinkSync(path); } catch { /* */ }
488
+ stats.blocked++;
489
+ continue;
490
+ }
491
+
492
+ // Too stale to resume meaningfully → block (work context has moved on).
493
+ if (age > RESUME_FRESHNESS_MS) {
494
+ markItemBlocked(marker, `resume marker stale (${Math.round(age / 60000)}m old)`);
495
+ logSession({ event: "resume_expired", sessionId: marker.sessionId, item_id: marker.itemId, age_min: Math.round(age / 60000) });
496
+ try { unlinkSync(path); } catch { /* */ }
497
+ stats.expired++;
498
+ continue;
499
+ }
500
+
501
+ // No claude session id to re-spawn against → can't continue; block.
502
+ if (!marker.claudeSessionId) {
503
+ logSession({ event: "resume_no_session_id", sessionId: marker.sessionId, item_id: marker.itemId });
504
+ try { unlinkSync(path); } catch { /* */ }
505
+ stats.expired++;
506
+ continue;
507
+ }
508
+
509
+ // No original prompt persisted → there's nothing to re-spawn with (the
510
+ // resume re-issues `--session-id <id> <prompt>`, NOT `--resume`). A marker
511
+ // without a prompt is either a legacy marker (pre-H1) or one whose write
512
+ // dropped the field; we can't truly resume it, so retire it rather than
513
+ // spawn a no-op that the close handler would mistake for "done". (H1)
514
+ if (typeof marker.prompt !== "string" || !marker.prompt) {
515
+ logSession({ event: "resume_no_prompt", sessionId: marker.sessionId, item_id: marker.itemId });
516
+ try { unlinkSync(path); } catch { /* */ }
517
+ stats.expired++;
518
+ continue;
519
+ }
520
+
521
+ // Bump the attempt count on the marker BEFORE re-dispatch so a crash mid-
522
+ // resume still advances toward the 3-strike cap (no infinite loop).
523
+ marker.recoveryAttempts = attempts + 1;
524
+ try {
525
+ const tmp = path + ".tmp";
526
+ writeFileSync(tmp, JSON.stringify(marker, null, 2));
527
+ renameSync(tmp, path);
528
+ } catch { /* best-effort */ }
529
+
530
+ const spawnResume = typeof opts.spawnResume === "function" ? opts.spawnResume : defaultSpawnResume;
531
+ try {
532
+ spawnResume({ marker });
533
+ logSession({ event: "resume_dispatched", sessionId: marker.sessionId, claudeSessionId: marker.claudeSessionId, item_id: marker.itemId, attempt: marker.recoveryAttempts });
534
+ stats.resumed++;
535
+ } catch (err) {
536
+ console.warn(`[dispatcher] resume spawn failed for ${marker.sessionId}: ${err.message}`);
537
+ }
538
+ }
539
+ if (stats.scanned > 0) {
540
+ console.log(`[dispatcher] resume reconcile: ${stats.resumed} resumed, ${stats.blocked} blocked, ${stats.expired} expired (of ${stats.scanned})`);
541
+ }
542
+ return stats;
543
+ }
544
+
545
+ /**
546
+ * Real resume spawner. Mirrors responder.mjs's blessed session-continuation
547
+ * pattern (see runClaudeCLI ~L116-122 + generateResponse ~L461-465): a
548
+ * continuation re-spawns `claude --print --session-id <sessionId> <prompt>`
549
+ * with the SAME model/flags the original used — it does NOT use `--resume`.
550
+ *
551
+ * Why NOT `--resume` (the previous bug): `--resume <id>` with no prompt and
552
+ * stdin ignored either (a) exits 0 having done nothing — the close handler
553
+ * then clears the marker and the interrupted work is silently lost — or (b)
554
+ * errors, force-`blocked`ing every in-flight item after a reboot. Pre-minting
555
+ * a stable session id and RE-SPAWNING against it with the original prompt is
556
+ * the actual resume mechanism in `--print` mode (per the b1 flag-verification
557
+ * report cited in responder.mjs). The CLI rehydrates the session-id's history
558
+ * and the fresh prompt continues the work.
559
+ *
560
+ * The resumed session inherits the same permission/env posture as a fresh
561
+ * spawn. Best-effort; failures are logged, the marker stays (next reconcile
562
+ * retries up to the strike cap). On a clean close the resumed session's own
563
+ * close handler clears the marker.
564
+ */
565
+ function defaultSpawnResume({ marker }) {
566
+ const args = [
567
+ "--print",
568
+ ...sessionPermissionArgs({ source: "dispatcher-resume" }),
569
+ ...daemonClaudeArgs(),
570
+ "--session-id", marker.claudeSessionId,
571
+ "--model", marker.modelFlag || marker.model || "sonnet",
572
+ marker.prompt,
573
+ ];
574
+ const spawnEnv = {
575
+ ...process.env,
576
+ PATH: augmentedPath(),
577
+ ANTHROPIC_API_KEY: "",
578
+ ANTHROPIC_AUTH_TOKEN: "",
579
+ };
580
+ const proc = _spawn(CLAUDE_BIN, args, { cwd: AGENT_REPO_DIR, env: spawnEnv, stdio: ["ignore", "pipe", "pipe"] });
581
+ let stderr = "";
582
+ proc.stderr.on("data", (c) => { stderr += c.toString(); });
583
+ proc.on("close", (code) => {
584
+ if (code === 0) {
585
+ // Resume succeeded — the underlying work is done; clear the marker.
586
+ clearResumePending(marker.sessionId);
587
+ logSession({ event: "resume_completed", sessionId: marker.sessionId, item_id: marker.itemId });
588
+ } else {
589
+ // A 429 during resume must open the breaker so subsequent spawns gate.
590
+ if (rateGuard.classifyStderr(stderr)) {
591
+ try { rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
592
+ }
593
+ logSession({ event: "resume_exit_nonzero", sessionId: marker.sessionId, item_id: marker.itemId, exit_code: code });
594
+ }
595
+ });
596
+ proc.on("error", (err) => {
597
+ logSession({ event: "resume_spawn_error", sessionId: marker.sessionId, error: err.code || err.message });
598
+ });
599
+ }
600
+
601
+ function writeActiveSession(sessionId, entry) {
602
+ try {
603
+ let active = {};
604
+ try { active = JSON.parse(readFileSync(ACTIVE_PATH, "utf-8")); } catch {}
605
+ active[sessionId] = {
606
+ sender: entry.item?.sender || null,
607
+ channel: entry.item?.channel || entry.item?.channel_id || null,
608
+ summary: entry.classResult?.summary || "unknown",
609
+ model: entry.classResult?.model || "sonnet",
610
+ startTime: Date.now(),
611
+ source: entry.source,
612
+ };
613
+ const tmpPath = ACTIVE_PATH + ".tmp";
614
+ writeFileSync(tmpPath, JSON.stringify(active, null, 2));
615
+ renameSync(tmpPath, ACTIVE_PATH);
616
+ } catch (err) {
617
+ console.warn(`[dispatcher] Failed to write active.json: ${err.message}`);
618
+ }
619
+ }
620
+
621
+ function removeActiveSession(sessionId) {
622
+ try {
623
+ let active = {};
624
+ try { active = JSON.parse(readFileSync(ACTIVE_PATH, "utf-8")); } catch {}
625
+ delete active[sessionId];
626
+ const tmpPath = ACTIVE_PATH + ".tmp";
627
+ writeFileSync(tmpPath, JSON.stringify(active, null, 2));
628
+ renameSync(tmpPath, ACTIVE_PATH);
629
+ } catch (err) {
630
+ console.warn(`[dispatcher] Failed to update active.json: ${err.message}`);
631
+ }
632
+ }
633
+
634
+ /**
635
+ * Count active sessions by source type.
636
+ */
637
+ function countBySource(source) {
638
+ let count = 0;
639
+ for (const [, s] of activeSessions) {
640
+ if (s.source === source) count++;
641
+ }
642
+ return count;
643
+ }
644
+
645
+ /**
646
+ * Evict the lowest-priority, longest-running backlog session to make room
647
+ * for a high-priority inbox item. Returns true if a session was evicted.
648
+ */
649
+ function evictForPriority() {
650
+ const priorityRank = { low: 0, normal: 1, high: 2, critical: 3 };
651
+ let worst = null;
652
+ let worstId = null;
653
+
654
+ for (const [id, s] of activeSessions) {
655
+ if (s.source !== "backlog") continue;
656
+ const rank = priorityRank[s.classResult.priority] || 1;
657
+ if (!worst) {
658
+ worst = s; worstId = id;
659
+ continue;
660
+ }
661
+ const worstRank = priorityRank[worst.classResult.priority] || 1;
662
+ // Prefer to evict lower priority; break ties by oldest session
663
+ if (rank < worstRank || (rank === worstRank && s.startTime < worst.startTime)) {
664
+ worst = s; worstId = id;
665
+ }
666
+ }
667
+
668
+ if (worst) {
669
+ console.log(`[dispatcher] PREEMPT: Evicting backlog session ${worstId} (${worst.classResult.summary}) for priority inbox item`);
670
+ logSession({ event: "evicted", sessionId: worstId, reason: "priority_preemption", summary: worst.classResult.summary });
671
+ // M2: clear the resume-pending marker BEFORE the SIGTERM. Eviction is a
672
+ // deliberate preemption, not a crash — the item stays on disk and the
673
+ // backlog sweep re-dispatches it normally (acquiring the item-claim).
674
+ // If we left the marker, a reboot's reconcileResumePending would re-spawn
675
+ // it WITHOUT the claim while sweepBacklog independently claim+dispatched
676
+ // the same item → two concurrent runs. The non-zero SIGTERM close would
677
+ // otherwise keep the marker, so we must retire it here.
678
+ clearResumePending(worstId);
679
+ worst.process.kill("SIGTERM");
680
+ return true;
681
+ }
682
+ return false;
683
+ }
684
+
685
+ /**
686
+ * Generate a stable key for a backlog item to track in-flight status.
687
+ */
688
+ function backlogKey(item) {
689
+ return item.id || item.title || item.summary || JSON.stringify(item).substring(0, 100);
690
+ }
691
+
692
+ /**
693
+ * Check if a backlog item already has an active session or has exceeded retry limit.
694
+ * Returns { allowed: boolean, reason?: string }
695
+ */
696
+ export function canDispatchBacklog(item) {
697
+ const key = backlogKey(item);
698
+ if (activeBacklogKeys.has(key)) {
699
+ return { allowed: false, reason: "session_already_active" };
700
+ }
701
+ const retries = backlogRetryCount.get(key) || 0;
702
+ if (retries >= MAX_BACKLOG_RETRIES) {
703
+ return { allowed: false, reason: "max_retries_exceeded", retries };
704
+ }
705
+ // Post-completion cooldown — prevent every-2-min re-dispatch of items
706
+ // that completed (success or failure) within the recent window.
707
+ const cooldownUntil = backlogCooldownUntil.get(key) || 0;
708
+ if (cooldownUntil > Date.now()) {
709
+ const remaining_min = Math.ceil((cooldownUntil - Date.now()) / 60000);
710
+ return { allowed: false, reason: "post_completion_cooldown", remaining_min };
711
+ }
712
+ return { allowed: true };
713
+ }
714
+
715
+ /**
716
+ * Dispatch an item for processing.
717
+ *
718
+ * @param {string} prompt
719
+ * @param {object} item
720
+ * @param {object} classResult
721
+ * @param {string} source - "inbox" or "backlog"
722
+ * @param {object} [opts]
723
+ * @param {(result:{ok:boolean, sessionId:string|null, item:object, code:number|null, stdout:string})=>void} [opts.onClose]
724
+ * Invoked EXACTLY ONCE when the spawned session reaches a terminal state
725
+ * (clean close → ok:true; non-zero close OR spawn error → ok:false). Lets the
726
+ * caller defer finalizing the inbox item (markProcessed) to spawn-close success
727
+ * so a crash between dispatch and close leaves the item re-deliverable rather
728
+ * than prematurely `.processed` (F1/H2 message-loss window). The callback
729
+ * survives queueing (it's carried on the entry, so a QUEUE/DEFER'd item still
730
+ * fires onClose once it is later drained and spawned). It does NOT fire on the
731
+ * pre-spawn drop paths (backlog dedup / claim-denied / governor-DEFER backlog),
732
+ * which leave the durable on-disk item untouched for the next sweep anyway.
733
+ */
734
+ export function dispatch(prompt, item, classResult, source = "inbox", opts = {}) {
735
+ const entry = { prompt, item, classResult, source, onClose: typeof opts.onClose === "function" ? opts.onClose : null };
736
+ const priority = classResult.priority;
737
+ const isPriorityInbox = source === "inbox" && (priority === "critical" || priority === "high");
738
+
739
+ // Backlog items: check dedup and retry limits
740
+ if (source === "backlog") {
741
+ const check = canDispatchBacklog(item);
742
+ if (!check.allowed) {
743
+ logSession({ event: "skipped", reason: check.reason, summary: classResult.summary, retries: check.retries });
744
+ return;
745
+ }
746
+ }
747
+
748
+ // ── WS4 governance gate ──────────────────────────────────────────────────
749
+ // Consult the shared 429 breaker + the resource governor BEFORE claiming a
750
+ // backlog item, so a DEFER leaves the item untouched on disk (the 10-min
751
+ // sweep retries it — no work lost, no retry budget burned). For inbox, a
752
+ // non-ADMIT routes to the in-memory queue instead of spawning (the user's
753
+ // message is never dropped). Priority inbox can still preempt a backlog
754
+ // session below. `forceQueue` carries the decision to the spawn branch.
755
+ let forceQueue = false;
756
+ let queueReason = null;
757
+ let queueRetryAt = null; // M1: when the breaker says "retry after T", arm a re-drain then.
758
+ {
759
+ const rb = rateBlocked();
760
+ if (rb) {
761
+ if (source === "backlog") {
762
+ logSession({ event: "deferred", reason: "rate_limited", retry_at: rb.retryAt, summary: classResult.summary });
763
+ return; // leave on disk; sweep retries after the breaker closes
764
+ }
765
+ forceQueue = true; queueReason = "rate_limited"; queueRetryAt = rb.retryAt || null;
766
+ }
767
+ if (!forceQueue || source !== "inbox") {
768
+ const adm = admitFor(source, priority);
769
+ if (adm.decision !== "ADMIT") {
770
+ if (source === "backlog") {
771
+ // QUEUE and DEFER both mean "not now" for backlog — the item stays in
772
+ // its queue file (re-derivable, durable) and the sweep retries. We do
773
+ // NOT claim it and we do NOT burn retry budget.
774
+ logSession({ event: adm.decision === "DEFER" ? "deferred" : "queued_skip", reason: adm.reason, decision: adm.decision, source, summary: classResult.summary });
775
+ return;
776
+ }
777
+ // inbox: never drop — queue it (DEFER(inbox) → queue anyway).
778
+ forceQueue = true; queueReason = queueReason || adm.reason;
779
+ }
780
+ }
781
+ }
782
+
783
+ // Item-claim acquisition: file-based claim visible across daemon restarts
784
+ // and concurrent launchd triggers. Complements the in-memory activeBacklogKeys.
785
+ // (ib-20260407-001b: concurrent session coordination)
786
+ if (source === "backlog" && item.id) {
787
+ const sessionId = `s-${Date.now()}-${sessionCounter + 1}`;
788
+ const claim = claimItem(item.id, {
789
+ session_id: sessionId,
790
+ agent_description: classResult.summary || item.title || "",
791
+ ttl_minutes: classResult.model === "opus" ? 120 : 30,
792
+ source: "backlog",
793
+ queue_file: item.source_file || "",
794
+ pid: process.pid, // daemon PID; child PID not yet known
795
+ });
796
+ if (!claim.claimed) {
797
+ console.log(`[dispatcher] Item claim denied for ${item.id}: ${claim.reason} (holder: ${claim.holder || "unknown"})`);
798
+ logSession({ event: "skipped", reason: `item_claim_denied: ${claim.reason}`, summary: classResult.summary, holder: claim.holder });
799
+ return;
800
+ }
801
+ }
802
+
803
+ // Backlog items respect the reserved slot cap
804
+ if (source === "backlog") {
805
+ const backlogCount = countBySource("backlog");
806
+ const backlogCap = MAX_CONCURRENT - RESERVED_INBOX_SLOTS;
807
+ if (backlogCount >= backlogCap) {
808
+ // No room for more backlog — silently skip (don't queue indefinitely)
809
+ return;
810
+ }
811
+ }
812
+
813
+ // forceQueue (WS4): the governor/breaker said "not now" for this inbox item,
814
+ // OR all slots are full. Either way we don't spawn directly — but a priority
815
+ // inbox item may still preempt a running backlog session.
816
+ if (!forceQueue && activeSessions.size < MAX_CONCURRENT) {
817
+ spawnSession(entry);
818
+ } else if (isPriorityInbox) {
819
+ // Priority inbox item but no slots (or governor-gated) — try to evict a
820
+ // backlog session. (Eviction is only useful under capacity pressure; under
821
+ // a rate-limit/budget gate there may be nothing to evict, in which case it
822
+ // simply queues — still never dropped.)
823
+ if (evictForPriority()) {
824
+ // Session evicted; it will call drainQueue on close. Queue this item at front.
825
+ priorityQueue.unshift(entry);
826
+ logSession({ event: "queued_after_eviction", priority: classResult.priority, summary: classResult.summary, reason: queueReason });
827
+ } else {
828
+ // No backlog sessions to evict — queue normally
829
+ priorityQueue.push(entry);
830
+ logSession({ event: "queued", priority: classResult.priority, summary: classResult.summary, queue_position: priorityQueue.length, reason: queueReason });
831
+ }
832
+ } else if (classResult.priority === "critical" || classResult.priority === "high") {
833
+ priorityQueue.push(entry);
834
+ logSession({ event: "queued", priority: classResult.priority, summary: classResult.summary, queue_position: priorityQueue.length, reason: queueReason });
835
+ } else {
836
+ normalQueue.push(entry);
837
+ logSession({ event: "queued", priority: classResult.priority, summary: classResult.summary, queue_position: normalQueue.length, reason: queueReason });
838
+ }
839
+
840
+ // M1: a governor/budget/rate DEFER can now force-queue an inbox item while
841
+ // NO session is running (pre-WS4 an item was only ever queued behind a
842
+ // running session, whose close drains). With zero active sessions nothing
843
+ // wakes the queue until the next external dispatch — so the item could stall
844
+ // indefinitely. Arm a single debounced re-drain timer to self-heal. The
845
+ // 60s inbox re-scan can't double-dispatch this: the item-claim / sent-message
846
+ // dedupe already guards that, and drainQueue itself re-checks admission.
847
+ if (activeSessions.size === 0 && (priorityQueue.length > 0 || normalQueue.length > 0)) {
848
+ armReDrain(queueRetryAt);
849
+ }
850
+ }
851
+
852
+ // ── M1: re-drain timer (single, debounced) ──────────────────────────────────
853
+ // Only ONE timer is ever pending. Arming again while one is live is a no-op
854
+ // (we never stack timers). Cleared whenever a real drain runs.
855
+ const RE_DRAIN_MIN_MS = 15_000; // floor: don't busy-poll the governor
856
+ const RE_DRAIN_MAX_MS = 30_000; // ceiling: bound worst-case stall to 30s
857
+ let reDrainTimer = null;
858
+
859
+ /**
860
+ * Arm a one-shot re-drain. `retryAt` (epoch ms, optional) is the breaker's
861
+ * "safe to retry after" hint; we clamp the delay into [15s, 30s] so we neither
862
+ * hammer the governor nor leave an item parked too long. Debounced: a second
863
+ * call while a timer is pending does nothing.
864
+ */
865
+ function armReDrain(retryAt) {
866
+ if (reDrainTimer) return; // already armed — debounce
867
+ let delay = RE_DRAIN_MAX_MS;
868
+ if (typeof retryAt === "number" && retryAt > 0) {
869
+ delay = Math.max(RE_DRAIN_MIN_MS, Math.min(RE_DRAIN_MAX_MS, retryAt - Date.now()));
870
+ }
871
+ reDrainTimer = setTimeout(() => {
872
+ reDrainTimer = null;
873
+ try { drainQueue(); } catch (err) { console.warn(`[dispatcher] re-drain failed: ${err.message}`); }
874
+ }, delay);
875
+ // Don't let this timer pin the event loop alive on its own (mirrors the
876
+ // socket-mode supervisor's unref pattern).
877
+ if (reDrainTimer && typeof reDrainTimer.unref === "function") reDrainTimer.unref();
878
+ }
879
+
880
+ /** Test seam: report whether a re-drain timer is currently armed. */
881
+ export function _hasReDrainArmed() {
882
+ return reDrainTimer !== null;
883
+ }
884
+
885
+ /**
886
+ * Current budget band (0|75|90|100) from budget-guard's daily spend vs the
887
+ * config/recovery.yaml cap, mapped onto the economics ladder. Fed into
888
+ * resolveChain so a chain degrades (frontier→default→fast→cheap) as the day's
889
+ * spend climbs (SPEC §6.4). Best-effort: any read failure → band 0 (policy as
890
+ * written) so a budget-read bug never changes routing under us.
891
+ */
892
+ function currentBudgetBand() {
893
+ try {
894
+ const st = budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR });
895
+ return budgetLadder(st.spentUSD, st.capUSD).band;
896
+ } catch {
897
+ return 0;
898
+ }
899
+ }
900
+
901
+ /**
902
+ * Resolve the spawn target through the v2 router (resolveChain) when the agent
903
+ * has opted into a `schema_version: 2` config, else fall back to the v1
904
+ * resolveBackend path (byte-compatible). Returns a normalised shape the spawn
905
+ * path consumes regardless of which router produced it:
906
+ *
907
+ * { modelFlag, envForSpawn, decisionId, backend, model, transport,
908
+ * maxTurns, effort, agentsJson, explain, decision }
909
+ *
910
+ * NEVER throws — any router error degrades to the v1 path (or null → stock CLI).
911
+ * The kill switch (MAESTRO_ROUTER_FORCE_ANTHROPIC) is honoured INSIDE both
912
+ * resolveChain and resolveBackend, so it short-circuits to stock Anthropic
913
+ * either way.
914
+ *
915
+ * @param {object} routingConfig loaded config (v1 or v2) or null
916
+ * @param {object} routingRequest v1 AgentRequest (from requestFromClassifierResult)
917
+ * @param {object} classResult
918
+ * @param {string} source
919
+ * @param {string} fallbackModel the coarse sonnet/opus class for the no-config path
920
+ */
921
+ function resolveSpawnTarget(routingConfig, routingRequest, classResult, source, fallbackModel) {
922
+ // No config at all → stock Claude-CLI-on-Max behaviour (unchanged).
923
+ if (!routingConfig) {
924
+ return { modelFlag: fallbackModel, envForSpawn: {}, decisionId: null, backend: null, model: null, transport: "anthropic-cli", maxTurns: null, effort: null, agentsJson: null, explain: null, decision: null };
925
+ }
926
+
927
+ // v2 path — resolveChain produces a full RouteDecision.
928
+ if (routingConfig.schema_version === 2) {
929
+ try {
930
+ const taskClass = "session.responder";
931
+ const req = {
932
+ ...routingRequest,
933
+ task_class: taskClass,
934
+ // data_class is DERIVED from the channel/source, never model-chosen
935
+ // (SPEC §7.6): inbox traffic is sensitive; backlog is internal work.
936
+ data_class: source === "inbox" ? "sensitive" : "internal",
937
+ budget_band: currentBudgetBand(),
938
+ harness_hint: "session",
939
+ };
940
+ const decision = resolveChain(req, { config: routingConfig, agentRoot: AGENT_REPO_DIR });
941
+ if (decision && decision.chosen) {
942
+ // Spawn knobs default per task_class; a rule may have overridden them on
943
+ // the decision (decision.spawnArgs wins where present).
944
+ const knobs = spawnKnobsFor(taskClass);
945
+ const sa = decision.spawnArgs || {};
946
+ return {
947
+ modelFlag: sa.modelFlag || decision.chosen.model || fallbackModel,
948
+ envForSpawn: decision.envForSpawn || {},
949
+ decisionId: decision.decision_id || null,
950
+ backend: decision.chosen.provider || null,
951
+ model: decision.chosen.model || null,
952
+ transport: decision.chosen.transport || null,
953
+ maxTurns: sa.maxTurns != null ? sa.maxTurns : knobs.maxTurns,
954
+ effort: sa.effort != null ? sa.effort : knobs.effort,
955
+ agentsJson: sa.agentsJson != null ? sa.agentsJson : knobs.agentsJson,
956
+ explain: decision.explain || null,
957
+ decision,
958
+ };
959
+ }
960
+ } catch (err) {
961
+ console.warn(`[dispatcher] resolveChain failed, falling back to v1 resolver: ${err.message}`);
962
+ }
963
+ // resolveChain degraded — fall through to v1 below.
964
+ }
965
+
966
+ // v1 path — resolveBackend → modelFlagFor (preserved verbatim).
967
+ const resolved = resolveBackend(routingRequest, { config: routingConfig });
968
+ if (!resolved) {
969
+ return { modelFlag: fallbackModel, envForSpawn: {}, decisionId: null, backend: null, model: null, transport: "anthropic-cli", maxTurns: null, effort: null, agentsJson: null, explain: null, decision: null };
970
+ }
971
+ return {
972
+ modelFlag: modelFlagFor(resolved, routingRequest),
973
+ envForSpawn: resolved.envForSpawn || {},
974
+ decisionId: null,
975
+ backend: resolved.name || null,
976
+ model: resolved.model || null,
977
+ transport: resolved.transport || null,
978
+ maxTurns: null,
979
+ effort: null,
980
+ agentsJson: null,
981
+ explain: null,
982
+ decision: null,
983
+ _v1Resolved: resolved,
984
+ };
985
+ }
986
+
987
+ function spawnSession(entry) {
988
+ const { prompt, item, classResult, source } = entry;
989
+ const sessionId = `s-${Date.now()}-${++sessionCounter}`;
990
+
991
+ // F1/H2: notify the dispatch caller exactly once when this session reaches a
992
+ // terminal state, so it can finalize the inbox item on success and re-deliver
993
+ // on crash. Single-fire guard — both proc.on("close") (clean + non-zero) and
994
+ // proc.on("error") (spawn failure) route here, and the close handler can run
995
+ // after error on the ENOENT/ETIMEDOUT path; we must invoke onClose only once.
996
+ //
997
+ // The callback also receives the run's raw `stdout` (already buffered here for
998
+ // the cost ledger) so the caller can apply POST-SESSION OUTCOME DISCIPLINE —
999
+ // reading the turn's commitments off its final text and landing them through
1000
+ // the org protocol (scripts/daemon/session-outcomes.mjs). Purely additive: the
1001
+ // field is new, existing callers destructure what they already used.
1002
+ let onCloseFired = false;
1003
+ const fireOnClose = (ok, code, closeStdout = "") => {
1004
+ if (onCloseFired || !entry.onClose) return;
1005
+ onCloseFired = true;
1006
+ try { entry.onClose({ ok, sessionId, item, code, stdout: closeStdout }); }
1007
+ catch (err) { console.warn(`[dispatcher] onClose callback threw for ${sessionId}: ${err.message}`); }
1008
+ };
1009
+ const model = classResult.model || "sonnet";
1010
+ const timeout = model === "opus"
1011
+ ? (source === "backlog" ? OPUS_BACKLOG_TIMEOUT : OPUS_INBOX_TIMEOUT)
1012
+ : (source === "backlog" ? SONNET_BACKLOG_TIMEOUT : SONNET_INBOX_TIMEOUT);
1013
+
1014
+ // Track backlog items to prevent retry storms
1015
+ if (source === "backlog") {
1016
+ const key = backlogKey(item);
1017
+ activeBacklogKeys.add(key);
1018
+ backlogRetryCount.set(key, (backlogRetryCount.get(key) || 0) + 1);
1019
+ }
1020
+
1021
+ // Resolve the spawn target via the model router. v2 configs go through
1022
+ // resolveChain (full RouteDecision: model + retarget env + spawn knobs +
1023
+ // decision_id + estimated cost + failover chain); v1 configs keep the legacy
1024
+ // resolveBackend path; no config preserves the historical Claude-CLI-on-Max
1025
+ // default exactly. NEVER throws (resolveSpawnTarget degrades internally).
1026
+ const routingConfig = getRoutingConfig();
1027
+ const routingRequest = requestFromClassifierResult(classResult, { source, role: "responder" });
1028
+ const target = resolveSpawnTarget(routingConfig, routingRequest, classResult, source, model);
1029
+ const effectiveModelFlag = target.modelFlag || model;
1030
+
1031
+ // WS4: pre-mint a stable Claude session id so a crash/reboot mid-flight can
1032
+ // be resumed deterministically with `claude --print --resume <id>`. (Same
1033
+ // mechanism the responder already uses; safe in --print text mode.)
1034
+ const claudeSessionId = randomUUID();
1035
+
1036
+ // The v2 router supplies spawn knobs (SPEC §4.4 / §6.5): --max-turns bounds
1037
+ // the session, --effort tunes reasoning where supported, --agents attaches the
1038
+ // cheap-subagent (Haiku-Explore) fan-out map — the sanctioned intra-session
1039
+ // cost lever that keeps the main loop's cache intact. v1 / no-config spawns
1040
+ // carry none of these (knobs are null), so this is purely additive.
1041
+ const knobArgs = [];
1042
+ if (target.maxTurns != null && Number.isFinite(Number(target.maxTurns))) {
1043
+ knobArgs.push("--max-turns", String(target.maxTurns));
1044
+ }
1045
+ if (target.effort) knobArgs.push("--effort", String(target.effort));
1046
+ if (target.agentsJson) {
1047
+ knobArgs.push("--agents", typeof target.agentsJson === "string" ? target.agentsJson : JSON.stringify(target.agentsJson));
1048
+ }
1049
+
1050
+ const args = [
1051
+ "--print",
1052
+ // --output-format json so this run's stdout carries the REAL token usage we
1053
+ // record to the cost ledger (recovery C1). The dispatcher's stdout is only
1054
+ // tee'd to a per-session log file — it's never streamed to the user (the
1055
+ // session sends its own user-facing messages via the Slack/Gmail APIs), so
1056
+ // switching the format does not affect any reply.
1057
+ "--output-format", "json",
1058
+ ...sessionPermissionArgs({ source: "dispatcher", priority: classResult?.priority }),
1059
+ ...daemonClaudeArgs(),
1060
+ ...knobArgs,
1061
+ "--session-id", claudeSessionId,
1062
+ "--model", effectiveModelFlag,
1063
+ prompt,
1064
+ ];
1065
+
1066
+ // Build the spawn env.
1067
+ // Default (no router): strip ANTHROPIC_API_KEY/ANTHROPIC_AUTH_TOKEN so
1068
+ // `claude` falls through to the keychain OAuth (Max subscription) per
1069
+ // CEO directive 2026-04-27.
1070
+ // Router active: merge the resolved target's envForSpawn — this points the
1071
+ // CLI at the chosen backend (Moonshot, OpenRouter, NIM, etc.) and sets
1072
+ // ANTHROPIC_API_KEY="" explicitly so Claude Code doesn't fall back to
1073
+ // OAuth against api.anthropic.com.
1074
+ //
1075
+ // For a v2 retarget (envForSpawn carries ANTHROPIC_BASE_URL), we build the
1076
+ // child env through the execution layer's buildChildEnv so the §7.3 allowlist
1077
+ // scrub applies (no foreign *_API_KEY / *_AUTH_TOKEN leaks into a third-party
1078
+ // session). buildChildEnv already injects ANTHROPIC_API_KEY="" for a retarget.
1079
+ // For the common Anthropic-session case (envForSpawn === {}) and the v1 path
1080
+ // we keep the prior explicit empty-key env verbatim (byte-compatible).
1081
+ const retargeting = !!(target.envForSpawn && target.envForSpawn.ANTHROPIC_BASE_URL);
1082
+ const spawnEnv = retargeting
1083
+ ? { ...buildChildEnv({ ...process.env, PATH: augmentedPath() }, target.envForSpawn), PATH: augmentedPath() }
1084
+ : {
1085
+ ...process.env,
1086
+ PATH: augmentedPath(),
1087
+ ANTHROPIC_API_KEY: "",
1088
+ ANTHROPIC_AUTH_TOKEN: "",
1089
+ ...(target.envForSpawn || {}),
1090
+ };
1091
+
1092
+ const proc = _spawn(CLAUDE_BIN, args, {
1093
+ cwd: AGENT_REPO_DIR,
1094
+ env: spawnEnv,
1095
+ stdio: ["ignore", "pipe", "pipe"],
1096
+ });
1097
+
1098
+ // Log the routing outcome whenever a backend was actually chosen (v1 or v2)
1099
+ // and it isn't the stock Anthropic CLI no-op. The v2 path also carries the
1100
+ // decision_id + the one-line explain so `maestro router why` can join this
1101
+ // session to its routing-audit + ledger rows.
1102
+ if (target.backend && target.transport !== "anthropic-cli") {
1103
+ logSession({
1104
+ event: "routed",
1105
+ sessionId,
1106
+ decision_id: target.decisionId,
1107
+ backend: target.backend,
1108
+ transport: target.transport,
1109
+ model: target.model,
1110
+ tried: target.decision ? target.decision.tried : target._v1Resolved?.tried,
1111
+ fallback_reason: target.decision ? target.decision.audit?.fallback_reason : target._v1Resolved?.fallback_reason,
1112
+ explain: target.explain,
1113
+ });
1114
+ }
1115
+
1116
+ let stdout = "";
1117
+ let stderr = "";
1118
+ const logFile = join(sessionLogDir(), `${sessionId}.log`);
1119
+
1120
+ proc.stdout.on("data", (chunk) => { stdout += chunk.toString(); });
1121
+ proc.stderr.on("data", (chunk) => { stderr += chunk.toString(); });
1122
+
1123
+ const timer = setTimeout(() => {
1124
+ console.error(`[dispatcher] Session ${sessionId} timed out (${timeout / 1000}s), killing`);
1125
+ proc.kill("SIGTERM");
1126
+ const killTimer = setTimeout(() => { if (!proc.killed) proc.kill("SIGKILL"); }, 5000);
1127
+ if (killTimer && typeof killTimer.unref === "function") killTimer.unref();
1128
+ }, timeout);
1129
+ // A kill-watchdog must never be the sole handle pinning the event loop open
1130
+ // (mirrors the re-drain timer above). It is cleared on every close path below;
1131
+ // unref only ensures a leaked/never-closed session (e.g. a test fake-spawn that
1132
+ // never emits "close") can't hold the process — otherwise the test suite would
1133
+ // wait out the full session timeout before exiting.
1134
+ if (timer && typeof timer.unref === "function") timer.unref();
1135
+
1136
+ const startTime = Date.now();
1137
+ activeSessions.set(sessionId, { process: proc, item, classResult, startTime, model, source, claudeSessionId });
1138
+ writeActiveSession(sessionId, entry);
1139
+
1140
+ // WS4: drop the in-flight resume marker. Present-after-crash = mid-flight, so
1141
+ // a startup reconcile can re-dispatch `claude --print --resume`. Cleared on a
1142
+ // clean close below. Inbox sessions are short and user-facing; backlog
1143
+ // sessions are the ones that most benefit from surviving a reboot, but we mark
1144
+ // both uniformly — the freshness window + 3-strike cap bound any churn.
1145
+ writeResumePending(sessionId, { ...entry, recoveryAttempts: entry.recoveryAttempts }, claudeSessionId, effectiveModelFlag, target.decisionId);
1146
+
1147
+ // WS2: show a typing indicator on the originating channel while the session
1148
+ // works (inbox items only — backlog work has no waiting human). Best-effort
1149
+ // via the typing registry; channels without a live adapter no-op.
1150
+ if (source === "inbox") {
1151
+ try { startTyping(item); } catch { /* fail-open */ }
1152
+ }
1153
+
1154
+ logSession({
1155
+ event: "spawned",
1156
+ sessionId,
1157
+ model,
1158
+ source,
1159
+ priority: classResult.priority,
1160
+ summary: classResult.summary,
1161
+ active_count: activeSessions.size,
1162
+ });
1163
+
1164
+ // Observability: the interaction's trace_id rode in on item.trace_id (set by
1165
+ // the daemon at item_received); read it explicitly here because the close/error
1166
+ // handlers below fire on a separate event-loop tick, outside any withTrace
1167
+ // scope. `dispatched` marks the spawn decision; `session_opened` the sub-session
1168
+ // start. Both carry trace_id + the v2 decision_id so the audit can join the
1169
+ // routing decision ↔ ledger row ↔ diagnostic stream.
1170
+ const traceId = (item && typeof item.trace_id === "string") ? item.trace_id : null;
1171
+ emitEvent({
1172
+ type: EVENT_TYPES.DISPATCHED,
1173
+ trace_id: traceId,
1174
+ attrs: { session_id: sessionId, source, model, decision_id: target.decisionId || null, priority: classResult.priority },
1175
+ });
1176
+ emitEvent({
1177
+ type: EVENT_TYPES.SESSION_OPENED,
1178
+ trace_id: traceId,
1179
+ attrs: { session_id: sessionId, source, model, decision_id: target.decisionId || null },
1180
+ });
1181
+
1182
+ console.log(`[dispatcher] Spawned ${sessionId} (${model}) — ${classResult.summary} [${activeSessions.size}/${MAX_CONCURRENT}]`);
1183
+
1184
+ proc.on("close", (code) => {
1185
+ clearTimeout(timer);
1186
+ // WS2: stop the typing heartbeat for this conversation (mirrors the
1187
+ // promoteDeferred close-hook). Best-effort; fail-open.
1188
+ if (source === "inbox") { try { stopTyping(item); } catch { /* */ } }
1189
+ // If proc.on("error") already fired for this session (spawn failure path
1190
+ // — ENOENT, EACCES, ETIMEDOUT), cleanup + metric + lock release already
1191
+ // happened. Skip to avoid double-count and double-release.
1192
+ if (spawnErrorHandled.has(sessionId)) {
1193
+ spawnErrorHandled.delete(sessionId);
1194
+ clearResumePending(sessionId); // error path already terminal — no resume
1195
+ // The error handler already fired onClose(ok:false) AND emitted
1196
+ // session_closed; the single-fire guard makes the onClose a no-op, and we
1197
+ // do NOT re-emit session_closed here to keep it exactly-once per session.
1198
+ fireOnClose(false, code);
1199
+ drainQueue();
1200
+ return;
1201
+ }
1202
+ activeSessions.delete(sessionId);
1203
+ removeActiveSession(sessionId);
1204
+ recordSession(true, code === 0);
1205
+ const duration = ((Date.now() - startTime) / 1000).toFixed(1);
1206
+
1207
+ // WS4: a clean exit retires the resume marker (work finished — nothing to
1208
+ // resume) and closes the shared 429 breaker. A non-zero exit whose stderr
1209
+ // looks like a rate limit opens the breaker so ALL spawn sources back off
1210
+ // together. The marker is deliberately LEFT in place on a non-zero exit so
1211
+ // a subsequent reboot can still resume the work; the cooldown/strike caps
1212
+ // bound re-dispatch.
1213
+ if (code === 0) {
1214
+ clearResumePending(sessionId);
1215
+ try { rateGuard.recordSuccess(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
1216
+ } else if (rateGuard.classifyStderr(stderr)) {
1217
+ try {
1218
+ const rec = rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR });
1219
+ logSession({ event: "rate_limit_recorded", sessionId, open_until: rec.openUntil, consecutive: rec.consecutive429 });
1220
+ } catch { /* */ }
1221
+ }
1222
+
1223
+ // Release item lock — MUST use same key order as acquireLock in daemon
1224
+ // (raw_ref first, then id). Previously this used id || raw_ref, creating
1225
+ // a different sanitised filename so the release silently missed the lock.
1226
+ const itemId = item.raw_ref || item.id || item.title;
1227
+ if (itemId) releaseLock(itemId);
1228
+
1229
+ // Release thread/channel lock so new messages in this thread/DM can
1230
+ // be processed. We must release unconditionally when a channel is
1231
+ // known — the previous `if (item.thread_id)` gate skipped DMs (which
1232
+ // always have empty thread_id), so DM-channel locks never cleared
1233
+ // until the 60min TTL fired. acquireThreadLock normalises DM
1234
+ // channels to `dm-channel` internally; releaseThreadLock mirrors
1235
+ // that normalisation, so calling it with an empty thread_id on a
1236
+ // DM channel does the right thing.
1237
+ {
1238
+ const channel = item.channel_id || (item.raw_ref ? (item.raw_ref.match(/slack:([^:]+):/) || [])[1] : null) || item.channel;
1239
+ if (channel) {
1240
+ releaseThreadLock(channel, item.thread_id);
1241
+ // Now that the lock is gone, promote any messages that were
1242
+ // deferred behind it. Latest-wins: a burst of N messages
1243
+ // collapses into ONE re-dispatch carrying the most recent
1244
+ // message (its thread_context already includes the earlier
1245
+ // ones), so the user gets a single coherent reply rather than
1246
+ // N replies serialised over N session-durations.
1247
+ const promo = promoteDeferred(channel, AGENT_REPO_DIR);
1248
+ if (promo.promoted > 0) {
1249
+ logSession({
1250
+ event: "deferred_promoted",
1251
+ channel,
1252
+ promoted: promo.promoted,
1253
+ bundled: promo.bundled,
1254
+ service: promo.service,
1255
+ });
1256
+ }
1257
+ }
1258
+ }
1259
+
1260
+ // Release request claim so the same type of request can be processed again
1261
+ if (classResult && classResult.summary) {
1262
+ releaseRequestClaim({
1263
+ recipient: item.channel_id || item.channel || item.sender || "unknown",
1264
+ subject: classResult.summary || item.subject || "",
1265
+ action_type: classResult.action || "respond",
1266
+ });
1267
+ }
1268
+
1269
+ // Release backlog tracking + item claim
1270
+ if (source === "backlog") {
1271
+ const key = backlogKey(item);
1272
+ activeBacklogKeys.delete(key);
1273
+ const retries = backlogRetryCount.get(key) || 0;
1274
+
1275
+ // Apply post-completion cooldown — different for success vs failure.
1276
+ // This is the fix for the every-2-min re-dispatch loop: once a
1277
+ // session has touched an item, we wait before touching it again.
1278
+ const cooldownMs = code === 0 ? SUCCESS_COOLDOWN_MS : FAILURE_COOLDOWN_MS;
1279
+ backlogCooldownUntil.set(key, Date.now() + cooldownMs);
1280
+ saveCooldowns();
1281
+ logSession({
1282
+ event: "cooldown_set",
1283
+ summary: classResult.summary,
1284
+ exit_code: code,
1285
+ cooldown_minutes: Math.round(cooldownMs / 60000),
1286
+ });
1287
+
1288
+ // Release file-based item claim (ib-20260407-001b)
1289
+ if (item.id) releaseItemClaim(item.id);
1290
+
1291
+ // If session timed out (143=SIGTERM) and hit retry limit, log it
1292
+ if (code === 143 && retries >= MAX_BACKLOG_RETRIES) {
1293
+ console.warn(`[dispatcher] Backlog item "${classResult.summary}" exhausted ${MAX_BACKLOG_RETRIES} retries — will not retry`);
1294
+ logSession({ event: "retries_exhausted", summary: classResult.summary, retries });
1295
+ }
1296
+ }
1297
+
1298
+ // Write session log
1299
+ writeFileSync(logFile, `# Session ${sessionId}\n# Model: ${model}\n# Duration: ${duration}s\n# Exit: ${code}\n\n## STDOUT\n${stdout}\n\n## STDERR\n${stderr}\n`);
1300
+
1301
+ // Record a TRUTHFUL cost-ledger row from this run's real token usage
1302
+ // (recovery C1) — the dispatcher is the user-facing spawn source and was
1303
+ // previously invisible to budget-guard. Parse failures are surfaced, not
1304
+ // backfilled with fake zeros.
1305
+ const costUsage = recordDispatcherCost({
1306
+ stdout,
1307
+ model: effectiveModelFlag,
1308
+ durationMs: Date.now() - startTime,
1309
+ exitCode: code,
1310
+ decisionId: target.decisionId,
1311
+ });
1312
+ if (!costUsage.ok) {
1313
+ logSession({ event: "cost_usage_parse_failed", sessionId, reason: costUsage.reason });
1314
+ }
1315
+
1316
+ logSession({
1317
+ event: "completed",
1318
+ sessionId,
1319
+ model,
1320
+ source,
1321
+ exit_code: code,
1322
+ duration_s: parseFloat(duration),
1323
+ summary: classResult.summary,
1324
+ active_count: activeSessions.size,
1325
+ });
1326
+
1327
+ // Observability: the sub-session ended. Carry the interaction's trace_id +
1328
+ // decision_id so item_received → dispatched → session_opened → session_closed
1329
+ // (→ sent, emitted by the daemon's onClose) all group by one id.
1330
+ emitEvent({
1331
+ type: EVENT_TYPES.SESSION_CLOSED,
1332
+ trace_id: traceId,
1333
+ attrs: { session_id: sessionId, source, exit_code: code, duration_s: parseFloat(duration), decision_id: target.decisionId || null },
1334
+ });
1335
+
1336
+ console.log(`[dispatcher] Session ${sessionId} done (${duration}s, exit ${code}) [${activeSessions.size}/${MAX_CONCURRENT}]`);
1337
+
1338
+ // F1/H2: tell the caller this session reached a terminal state. A clean exit
1339
+ // (code 0) finalizes the inbox item (markProcessed); any non-zero exit leaves
1340
+ // it re-deliverable. Fired AFTER lock/claim release above so the caller's
1341
+ // re-delivery path (which may release the per-item lock) doesn't race them.
1342
+ fireOnClose(code === 0, code, stdout);
1343
+
1344
+ // Dispatch next queued item
1345
+ drainQueue();
1346
+ });
1347
+
1348
+ proc.on("error", (err) => {
1349
+ clearTimeout(timer);
1350
+ // WS2: stop the typing heartbeat (the session never really ran).
1351
+ if (source === "inbox") { try { stopTyping(item); } catch { /* */ } }
1352
+ // Mark so the trailing proc.on("close") doesn't double-process.
1353
+ spawnErrorHandled.add(sessionId);
1354
+ activeSessions.delete(sessionId);
1355
+ removeActiveSession(sessionId);
1356
+ // Spawn never ran — there is nothing to resume; retire the marker so the
1357
+ // next startup reconcile doesn't try to --resume a session that never began.
1358
+ clearResumePending(sessionId);
1359
+ recordSession(true, false);
1360
+ const duration = ((Date.now() - startTime) / 1000).toFixed(1);
1361
+ const errorCode = err.code || "unknown";
1362
+
1363
+ // Release item lock — mirror of close-handler logic.
1364
+ const itemId = item.raw_ref || item.id || item.title;
1365
+ if (itemId) releaseLock(itemId);
1366
+
1367
+ // Release thread/channel lock so new messages can be processed.
1368
+ // Mirror of close-handler logic — unconditional when channel is
1369
+ // known, since releaseThreadLock handles the DM normalization.
1370
+ {
1371
+ const channel = item.channel_id || (item.raw_ref ? (item.raw_ref.match(/slack:([^:]+):/) || [])[1] : null) || item.channel;
1372
+ if (channel) {
1373
+ releaseThreadLock(channel, item.thread_id);
1374
+ const promo = promoteDeferred(channel, AGENT_REPO_DIR);
1375
+ if (promo.promoted > 0) {
1376
+ logSession({
1377
+ event: "deferred_promoted_on_error",
1378
+ channel,
1379
+ promoted: promo.promoted,
1380
+ bundled: promo.bundled,
1381
+ service: promo.service,
1382
+ });
1383
+ }
1384
+ }
1385
+ }
1386
+
1387
+ // Release request claim + emit explicit claim_released event so
1388
+ // reconciliation audits (cycle 124 Agent C pattern) can distinguish
1389
+ // genuine in-flight sessions from silent-exit ETIMEDOUT failures
1390
+ // without cross-referencing logs/daemon/responses.jsonl.
1391
+ let claimReleased = false;
1392
+ if (classResult && classResult.summary) {
1393
+ releaseRequestClaim({
1394
+ recipient: item.channel_id || item.channel || item.sender || "unknown",
1395
+ subject: classResult.summary || item.subject || "",
1396
+ action_type: classResult.action || "respond",
1397
+ });
1398
+ claimReleased = true;
1399
+ }
1400
+
1401
+ // Release backlog tracking + item claim. Apply failure cooldown so the
1402
+ // same item isn't re-spawned on the next backlog sweep.
1403
+ if (source === "backlog") {
1404
+ const key = backlogKey(item);
1405
+ activeBacklogKeys.delete(key);
1406
+ backlogCooldownUntil.set(key, Date.now() + FAILURE_COOLDOWN_MS);
1407
+ saveCooldowns();
1408
+ // Release file-based item claim (ib-20260407-001b)
1409
+ if (item.id) releaseItemClaim(item.id);
1410
+ }
1411
+
1412
+ console.error(`[dispatcher] Session ${sessionId} failed: ${errorCode} (${err.message})`);
1413
+
1414
+ // Rich "failed" event — see ib-20260416-daemon-etimedout-failed-event.
1415
+ logSession({
1416
+ event: "failed",
1417
+ sessionId,
1418
+ error: errorCode,
1419
+ error_message: err.message,
1420
+ model,
1421
+ source,
1422
+ priority: classResult?.priority,
1423
+ summary: classResult?.summary,
1424
+ duration_s: parseFloat(duration),
1425
+ active_count: activeSessions.size,
1426
+ });
1427
+
1428
+ if (claimReleased) {
1429
+ logSession({
1430
+ event: "claim_released",
1431
+ sessionId,
1432
+ reason: `spawn_failed_${errorCode}`,
1433
+ });
1434
+ }
1435
+
1436
+ // Observability: a spawn failure is a terminal close too. Emit session_closed
1437
+ // (exit_code null, error code carried) so the audit sees every session reach a
1438
+ // terminal hop. Exactly-once: the trailing proc.on("close") early-returns via
1439
+ // the spawnErrorHandled guard WITHOUT re-emitting.
1440
+ emitEvent({
1441
+ type: EVENT_TYPES.SESSION_CLOSED,
1442
+ trace_id: traceId,
1443
+ attrs: { session_id: sessionId, source, exit_code: null, error: errorCode, decision_id: target.decisionId || null },
1444
+ });
1445
+
1446
+ // F1/H2: a spawn failure is terminal and the work never ran — tell the caller
1447
+ // (ok:false) so the inbox item is left re-deliverable, never `.processed`.
1448
+ fireOnClose(false, null);
1449
+
1450
+ drainQueue();
1451
+ });
1452
+ }
1453
+
1454
+ function drainQueue() {
1455
+ // A real drain supersedes any pending re-drain timer; clear it so we don't
1456
+ // also fire a redundant one. If we bail out under pressure below with items
1457
+ // still queued + no active session, we re-arm before returning (M1).
1458
+ if (reDrainTimer) { clearTimeout(reDrainTimer); reDrainTimer = null; }
1459
+
1460
+ // WS4: the shared 429 breaker gates the whole drain — if it's open, leave
1461
+ // everything queued (re-derivable; nothing dropped) and let a later close /
1462
+ // drain retry once the breaker closes.
1463
+ const rb = rateBlocked();
1464
+ if (rb) {
1465
+ // M1: with no session to drive a future close-drain, re-arm so the queued
1466
+ // item isn't stranded until the next external dispatch.
1467
+ if (activeSessions.size === 0 && (priorityQueue.length > 0 || normalQueue.length > 0)) {
1468
+ armReDrain(rb.retryAt);
1469
+ }
1470
+ return;
1471
+ }
1472
+
1473
+ while (activeSessions.size < MAX_CONCURRENT) {
1474
+ // Peek the head without removing it so a governor QUEUE/DEFER leaves the
1475
+ // item in place (never dropped).
1476
+ const next = priorityQueue[0] || normalQueue[0];
1477
+ if (!next) break;
1478
+
1479
+ const adm = admitFor(next.source, next.classResult?.priority);
1480
+ if (adm.decision !== "ADMIT") {
1481
+ // Under pressure: stop draining. Inbox items stay queued; the next close
1482
+ // event (a freed slot) or the breaker closing will resume the drain.
1483
+ logSession({ event: "drain_paused", reason: adm.reason, decision: adm.decision, source: next.source });
1484
+ // M1: if nothing is running, no close event will re-drain — arm a timer.
1485
+ if (activeSessions.size === 0) armReDrain(adm.retryAt);
1486
+ break;
1487
+ }
1488
+
1489
+ // Commit: remove from whichever queue it sat at the head of, then spawn.
1490
+ if (priorityQueue[0] === next) priorityQueue.shift();
1491
+ else normalQueue.shift();
1492
+ spawnSession(next);
1493
+ }
1494
+ }
1495
+
1496
+ /** Get current status for health checks */
1497
+ export function getStatus() {
1498
+ return {
1499
+ active_sessions: activeSessions.size,
1500
+ max_concurrent: MAX_CONCURRENT,
1501
+ priority_queue_length: priorityQueue.length,
1502
+ normal_queue_length: normalQueue.length,
1503
+ sessions: Array.from(activeSessions.entries()).map(([id, s]) => ({
1504
+ id,
1505
+ model: s.model,
1506
+ priority: s.classResult.priority,
1507
+ summary: s.classResult.summary,
1508
+ duration_s: ((Date.now() - s.startTime) / 1000).toFixed(0),
1509
+ })),
1510
+ };
1511
+ }
1512
+
1513
+ /** Get available slots count */
1514
+ export function availableSlots() {
1515
+ return MAX_CONCURRENT - activeSessions.size;
1516
+ }