@cohortapp/agent-sdk 2.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/commands/init-agent.md +104 -0
- package/.claude/commands/init-maestro.md +1187 -0
- package/.claude/settings.json +161 -0
- package/.env.example +216 -0
- package/README.md +632 -0
- package/agents/browser-operator/agent.md +52 -0
- package/agents/calendar-ops/agent.md +50 -0
- package/agents/communications/agent.md +96 -0
- package/agents/decision-log/agent.md +65 -0
- package/agents/desktop-operator/agent.md +59 -0
- package/agents/gmail-operator/agent.md +62 -0
- package/agents/inbound-dispatcher/agent.md +66 -0
- package/agents/inbox-processor/agent.md +39 -0
- package/agents/pmo-execution/agent.md +60 -0
- package/agents/session-spawner/agent.md +64 -0
- package/agents/slack-operator/agent.md +60 -0
- package/agents/whatsapp-operator/agent.md +60 -0
- package/agents/workflow-automation/agent.md +61 -0
- package/archetypes/altitudes/c-suite.yaml +58 -0
- package/archetypes/altitudes/founder.yaml +68 -0
- package/archetypes/altitudes/senior-manager.yaml +63 -0
- package/archetypes/altitudes/svp.yaml +60 -0
- package/archetypes/altitudes/vp.yaml +50 -0
- package/archetypes/archetype.schema.json +77 -0
- package/archetypes/base.yaml +47 -0
- package/archetypes/capabilities/commercial-leader.yaml +159 -0
- package/archetypes/capabilities/compliance-officer.yaml +159 -0
- package/archetypes/capabilities/executive-operator.yaml +169 -0
- package/archetypes/capabilities/finance-leader.yaml +162 -0
- package/archetypes/capabilities/operations-leader.yaml +154 -0
- package/archetypes/capabilities/product-leader.yaml +148 -0
- package/archetypes/capabilities/technical-leader.yaml +146 -0
- package/archetypes/functions/commercial-leader.yaml +62 -0
- package/archetypes/functions/compliance-officer.yaml +64 -0
- package/archetypes/functions/executive-operator.yaml +70 -0
- package/archetypes/functions/finance-leader.yaml +70 -0
- package/archetypes/functions/operations-leader.yaml +62 -0
- package/archetypes/functions/product-leader.yaml +61 -0
- package/archetypes/functions/technical-leader.yaml +57 -0
- package/bin/cohort-mcp.mjs +81 -0
- package/bin/maestro.mjs +3516 -0
- package/bin/maestro.test.mjs +1015 -0
- package/desktop-control/README.md +56 -0
- package/desktop-control/app-profiles/gmail.yaml +120 -0
- package/desktop-control/app-profiles/slack.yaml +315 -0
- package/desktop-control/app-profiles/whatsapp.yaml +107 -0
- package/docs/architecture/agent-topology.md +2239 -0
- package/docs/architecture/archetype-agent-factory.md +110 -0
- package/docs/architecture/collective-memory-and-org-mesh.md +115 -0
- package/docs/architecture/continuous-monitoring.md +221 -0
- package/docs/architecture/mcp-capability-map.md +585 -0
- package/docs/architecture/system-architecture.md +1272 -0
- package/docs/company-context/README.md +40 -0
- package/docs/guides/agent-persona-setup.md +600 -0
- package/docs/guides/agents-observe-setup.md +64 -0
- package/docs/guides/billing-console-keys.md +88 -0
- package/docs/guides/ccxray-diagnostics.md +65 -0
- package/docs/guides/channel-bus.md +127 -0
- package/docs/guides/claude-mem-setup.md +79 -0
- package/docs/guides/claude-pace-setup.md +56 -0
- package/docs/guides/claudraband-sessions.md +98 -0
- package/docs/guides/clawteam-swarm.md +116 -0
- package/docs/guides/code-review-graph-setup.md +86 -0
- package/docs/guides/email-setup.md +431 -0
- package/docs/guides/mac-mini.md +119 -0
- package/docs/guides/media-generation-setup.md +349 -0
- package/docs/guides/model-routing.md +162 -0
- package/docs/guides/observability-otel.md +265 -0
- package/docs/guides/org-onboarding.md +132 -0
- package/docs/guides/outbound-governance-setup.md +437 -0
- package/docs/guides/pdf-generation-setup.md +315 -0
- package/docs/guides/poller-daemon-setup.md +563 -0
- package/docs/guides/rag-context-setup.md +459 -0
- package/docs/guides/self-optimization-pattern.md +82 -0
- package/docs/guides/setup-wizard.md +178 -0
- package/docs/guides/slack-setup.md +350 -0
- package/docs/guides/telegram-setup.md +227 -0
- package/docs/guides/twilio-subaccounts-setup.md +223 -0
- package/docs/guides/verification.md +128 -0
- package/docs/guides/voice-mode.md +188 -0
- package/docs/guides/voice-sms-setup.md +698 -0
- package/docs/guides/webhook-relay-setup.md +349 -0
- package/docs/guides/whatsapp-setup.md +288 -0
- package/docs/prompts/board-pack-cover-template.md +36 -0
- package/docs/prompts/decision-recommendation-template.md +88 -0
- package/docs/prompts/followup-message-template.md +141 -0
- package/docs/prompts/investor-letter-template.md +52 -0
- package/docs/prompts/morning-brief-template.md +82 -0
- package/docs/prompts/presentation-template.md +58 -0
- package/docs/prompts/weekly-strategic-memo-template.md +104 -0
- package/docs/research/hallucinated-tool-output-investigation.md +151 -0
- package/docs/runbooks/backup-restore.md +205 -0
- package/docs/runbooks/cohort-cutover.md +129 -0
- package/docs/runbooks/fleet-operations.md +200 -0
- package/docs/runbooks/incident-response.md +226 -0
- package/docs/runbooks/mac-mini-bootstrap.md +431 -0
- package/docs/runbooks/perpetual-operations.md +509 -0
- package/docs/runbooks/recovery-and-failover.md +260 -0
- package/framework-features.json +267 -0
- package/ingest/README.md +87 -0
- package/lib/action-executor.js +689 -0
- package/lib/action-executor.test.mjs +871 -0
- package/lib/agent-root.mjs +37 -0
- package/lib/archetype.mjs +236 -0
- package/lib/archetype.test.mjs +132 -0
- package/lib/autonomy.mjs +114 -0
- package/lib/autonomy.test.mjs +66 -0
- package/lib/backlog.mjs +358 -0
- package/lib/backlog.test.mjs +266 -0
- package/lib/budget-guard.mjs +279 -0
- package/lib/budget-guard.test.mjs +291 -0
- package/lib/cadence-bus-schedule.test.mjs +194 -0
- package/lib/cadence-bus.mjs +1120 -0
- package/lib/cadence-bus.test.mjs +720 -0
- package/lib/cadences.mjs +205 -0
- package/lib/cadences.test.mjs +125 -0
- package/lib/capability.mjs +154 -0
- package/lib/capability.test.mjs +78 -0
- package/lib/channels/base-adapter.mjs +719 -0
- package/lib/channels/base-adapter.test.mjs +590 -0
- package/lib/channels/channel.mjs +128 -0
- package/lib/channels/channels.test.mjs +371 -0
- package/lib/channels/contract.mjs +215 -0
- package/lib/channels/contract.test.mjs +137 -0
- package/lib/channels/conversation-resolver.mjs +95 -0
- package/lib/channels/gmail/adapter.mjs +87 -0
- package/lib/channels/inbox-item.mjs +255 -0
- package/lib/channels/inbox-item.test.mjs +335 -0
- package/lib/channels/index.mjs +94 -0
- package/lib/channels/orgmail/adapter.mjs +353 -0
- package/lib/channels/orgmail/adapter.test.mjs +311 -0
- package/lib/channels/pairing.mjs +363 -0
- package/lib/channels/pairing.test.mjs +270 -0
- package/lib/channels/registry.mjs +164 -0
- package/lib/channels/slack/adapter.mjs +317 -0
- package/lib/channels/slack-adapter.test.mjs +212 -0
- package/lib/channels/sms/adapter.mjs +43 -0
- package/lib/channels/telegram/adapter.mjs +432 -0
- package/lib/channels/telegram-adapter.test.mjs +306 -0
- package/lib/channels/voice/adapter.mjs +301 -0
- package/lib/channels/voice/adapter.test.mjs +278 -0
- package/lib/channels/whatsapp/adapter-baileys.mjs +587 -0
- package/lib/channels/whatsapp/adapter-baileys.test.mjs +359 -0
- package/lib/channels/whatsapp/adapter-twilio.mjs +65 -0
- package/lib/channels/whatsapp/baileys-typing.test.mjs +154 -0
- package/lib/charter.mjs +256 -0
- package/lib/charter.test.mjs +89 -0
- package/lib/claude-bin.mjs +134 -0
- package/lib/claude-bin.test.mjs +75 -0
- package/lib/collective/capture.mjs +185 -0
- package/lib/collective/capture.test.mjs +121 -0
- package/lib/collective/cards.mjs +201 -0
- package/lib/collective/cards.test.mjs +114 -0
- package/lib/collective/config.mjs +186 -0
- package/lib/collective/config.test.mjs +123 -0
- package/lib/collective/global-config.mjs +113 -0
- package/lib/collective/global-config.test.mjs +75 -0
- package/lib/collective/presence.mjs +201 -0
- package/lib/collective/presence.test.mjs +95 -0
- package/lib/collective/recall.mjs +215 -0
- package/lib/collective/recall.test.mjs +116 -0
- package/lib/comms/send-gate.mjs +554 -0
- package/lib/comms/send-gate.test.mjs +577 -0
- package/lib/comms.mjs +67 -0
- package/lib/comms.test.mjs +41 -0
- package/lib/diagnostics/alerts.mjs +424 -0
- package/lib/diagnostics/alerts.test.mjs +318 -0
- package/lib/diagnostics/backup-freshness.mjs +188 -0
- package/lib/diagnostics/backup-freshness.test.mjs +185 -0
- package/lib/diagnostics/counters.mjs +269 -0
- package/lib/diagnostics/counters.test.mjs +206 -0
- package/lib/diagnostics/events.mjs +188 -0
- package/lib/diagnostics/events.test.mjs +290 -0
- package/lib/diagnostics/otel.mjs +237 -0
- package/lib/diagnostics/otel.test.mjs +196 -0
- package/lib/diagnostics/trace.mjs +216 -0
- package/lib/diagnostics/trace.test.mjs +251 -0
- package/lib/env-compat.mjs +74 -0
- package/lib/env-compat.test.mjs +104 -0
- package/lib/feature-init.mjs +331 -0
- package/lib/fs-atomic.mjs +112 -0
- package/lib/fs-atomic.test.mjs +72 -0
- package/lib/fs-ownership.mjs +111 -0
- package/lib/fs-ownership.test.mjs +158 -0
- package/lib/hooks/bus.mjs +347 -0
- package/lib/hooks/bus.test.mjs +387 -0
- package/lib/index.js +16 -0
- package/lib/learning/config.mjs +106 -0
- package/lib/learning/config.test.mjs +75 -0
- package/lib/learning/counters.mjs +156 -0
- package/lib/learning/counters.test.mjs +69 -0
- package/lib/learning/curator-consolidate.test.mjs +238 -0
- package/lib/learning/curator.mjs +453 -0
- package/lib/learning/curator.test.mjs +106 -0
- package/lib/learning/log.mjs +40 -0
- package/lib/learning/reflect.mjs +534 -0
- package/lib/learning/reflect.test.mjs +0 -0
- package/lib/learning/session-index.mjs +352 -0
- package/lib/learning/session-index.test.mjs +125 -0
- package/lib/learning/skill-writer.mjs +474 -0
- package/lib/learning/skill-writer.test.mjs +210 -0
- package/lib/mcp/server.mjs +328 -0
- package/lib/mcp/server.test.mjs +400 -0
- package/lib/model-router/auth-profiles.mjs +758 -0
- package/lib/model-router/auth-profiles.test.mjs +580 -0
- package/lib/model-router/catalog/anthropic.yaml +153 -0
- package/lib/model-router/catalog/deepseek.yaml +86 -0
- package/lib/model-router/catalog/moonshot.yaml +81 -0
- package/lib/model-router/catalog/qwen.yaml +114 -0
- package/lib/model-router/catalog.mjs +925 -0
- package/lib/model-router/catalog.test.mjs +385 -0
- package/lib/model-router/economics.mjs +564 -0
- package/lib/model-router/economics.test.mjs +344 -0
- package/lib/model-router/failover.mjs +298 -0
- package/lib/model-router/failover.test.mjs +439 -0
- package/lib/model-router/health.mjs +453 -0
- package/lib/model-router/health.test.mjs +338 -0
- package/lib/model-router/integration-coverage.test.mjs +829 -0
- package/lib/model-router/integration.test.mjs +564 -0
- package/lib/model-router/ledger.mjs +402 -0
- package/lib/model-router/ledger.test.mjs +382 -0
- package/lib/model-router/llm-task.mjs +515 -0
- package/lib/model-router/llm-task.test.mjs +392 -0
- package/lib/model-router/org-credentials.mjs +260 -0
- package/lib/model-router/org-credentials.test.mjs +265 -0
- package/lib/model-router/pricing-refresh.mjs +463 -0
- package/lib/model-router/pricing-refresh.test.mjs +286 -0
- package/lib/model-router/reconcile.mjs +429 -0
- package/lib/model-router/reconcile.test.mjs +316 -0
- package/lib/model-router/repair.mjs +471 -0
- package/lib/model-router/repair.test.mjs +180 -0
- package/lib/model-router/resolve.mjs +1206 -0
- package/lib/model-router/spawn.mjs +497 -0
- package/lib/model-router/spawn.test.mjs +425 -0
- package/lib/model-router/taxonomy.mjs +893 -0
- package/lib/model-router/taxonomy.test.mjs +410 -0
- package/lib/model-router.mjs +677 -0
- package/lib/model-router.test.mjs +907 -0
- package/lib/org/activity.mjs +211 -0
- package/lib/org/activity.test.mjs +134 -0
- package/lib/org/approvals.mjs +448 -0
- package/lib/org/approvals.test.mjs +216 -0
- package/lib/org/awareness.mjs +222 -0
- package/lib/org/awareness.test.mjs +159 -0
- package/lib/org/board.mjs +229 -0
- package/lib/org/board.test.mjs +177 -0
- package/lib/org/bootstrap-context.mjs +169 -0
- package/lib/org/bootstrap-context.test.mjs +153 -0
- package/lib/org/client.mjs +1628 -0
- package/lib/org/client.test.mjs +1107 -0
- package/lib/org/cohort-client.mjs +67 -0
- package/lib/org/cohort-client.test.mjs +126 -0
- package/lib/org/cost-sync.mjs +227 -0
- package/lib/org/cost-sync.test.mjs +153 -0
- package/lib/org/doctor.mjs +212 -0
- package/lib/org/doctor.test.mjs +212 -0
- package/lib/org/handoff.mjs +293 -0
- package/lib/org/handoff.test.mjs +269 -0
- package/lib/org/integration-tools.mjs +182 -0
- package/lib/org/integration-tools.test.mjs +160 -0
- package/lib/org/keys.mjs +131 -0
- package/lib/org/keys.test.mjs +92 -0
- package/lib/org/knowledge.mjs +463 -0
- package/lib/org/knowledge.test.mjs +319 -0
- package/lib/org/leases.mjs +335 -0
- package/lib/org/leases.test.mjs +235 -0
- package/lib/org/mesh-integration.test.mjs +127 -0
- package/lib/org/mesh.mjs +459 -0
- package/lib/org/mesh.test.mjs +345 -0
- package/lib/org/messaging.mjs +503 -0
- package/lib/org/messaging.test.mjs +238 -0
- package/lib/org/policy.mjs +345 -0
- package/lib/org/policy.test.mjs +237 -0
- package/lib/org/protocol.checksum +1 -0
- package/lib/org/protocol.checksum.test.mjs +90 -0
- package/lib/org/protocol.mjs +967 -0
- package/lib/org/protocol.test.mjs +264 -0
- package/lib/org/registry.mjs +194 -0
- package/lib/org/registry.test.mjs +100 -0
- package/lib/org/tool-surface-integration.test.mjs +120 -0
- package/lib/org/tool-surface.mjs +2535 -0
- package/lib/org/tool-surface.test.mjs +589 -0
- package/lib/org/ui-parity.mjs +3236 -0
- package/lib/org/ui-parity.test.mjs +348 -0
- package/lib/org/verify.mjs +176 -0
- package/lib/org/verify.test.mjs +194 -0
- package/lib/rag/embed.mjs +188 -0
- package/lib/rag/indexer.mjs +425 -0
- package/lib/rag/rag.test.mjs +505 -0
- package/lib/rag/search.mjs +475 -0
- package/lib/rate-guard.mjs +246 -0
- package/lib/rate-guard.test.mjs +201 -0
- package/lib/render.mjs +112 -0
- package/lib/render.test.mjs +68 -0
- package/lib/resource-governor.mjs +297 -0
- package/lib/resource-governor.test.mjs +262 -0
- package/lib/scheduling/dynamic-jobs.mjs +675 -0
- package/lib/scheduling/dynamic-jobs.test.mjs +344 -0
- package/lib/scheduling/jitter.mjs +0 -0
- package/lib/scheduling/jitter.test.mjs +140 -0
- package/lib/secrets/broker.mjs +315 -0
- package/lib/secrets/broker.test.mjs +280 -0
- package/lib/secrets/providers.mjs +461 -0
- package/lib/secrets/providers.test.mjs +274 -0
- package/lib/security/audit-engine.mjs +684 -0
- package/lib/security/audit-engine.test.mjs +389 -0
- package/lib/security/coerce-args.mjs +552 -0
- package/lib/security/coerce-args.test.mjs +281 -0
- package/lib/security/dangerous-tools.mjs +97 -0
- package/lib/security/dangerous-tools.test.mjs +68 -0
- package/lib/security/external-content.mjs +145 -0
- package/lib/security/external-content.test.mjs +67 -0
- package/lib/security/redact.mjs +592 -0
- package/lib/security/redact.test.mjs +441 -0
- package/lib/security/secret-equal.mjs +73 -0
- package/lib/security/secret-equal.test.mjs +55 -0
- package/lib/session-permissions.mjs +101 -0
- package/lib/session-permissions.test.mjs +100 -0
- package/lib/setup/claude-probe.mjs +74 -0
- package/lib/setup/completeness.mjs +175 -0
- package/lib/setup/completeness.test.mjs +110 -0
- package/lib/setup/context-pack.mjs +173 -0
- package/lib/setup/context-pack.test.mjs +89 -0
- package/lib/setup/enrich.mjs +277 -0
- package/lib/setup/enrich.test.mjs +115 -0
- package/lib/setup/enroll-from-cohort.mjs +441 -0
- package/lib/setup/enroll-from-cohort.test.mjs +233 -0
- package/lib/setup/integration.test.mjs +162 -0
- package/lib/setup/io.mjs +360 -0
- package/lib/setup/io.test.mjs +77 -0
- package/lib/setup/run-generator.mjs +81 -0
- package/lib/setup/runner.mjs +244 -0
- package/lib/setup/runner.test.mjs +132 -0
- package/lib/setup/sections/comms.mjs +173 -0
- package/lib/setup/sections/company.mjs +120 -0
- package/lib/setup/sections/enrich.mjs +138 -0
- package/lib/setup/sections/identity.mjs +182 -0
- package/lib/setup/sections/identity.test.mjs +140 -0
- package/lib/setup/sections/learning.mjs +153 -0
- package/lib/setup/sections/learning.test.mjs +81 -0
- package/lib/setup/sections/messaging.mjs +219 -0
- package/lib/setup/sections/messaging.test.mjs +127 -0
- package/lib/setup/sections/model.mjs +102 -0
- package/lib/setup/sections/operating-model.mjs +78 -0
- package/lib/setup/sections/org.mjs +475 -0
- package/lib/setup/sections/org.test.mjs +313 -0
- package/lib/setup/sections/orgmail.mjs +173 -0
- package/lib/setup/sections/orgmail.test.mjs +118 -0
- package/lib/setup/sections/recovery.mjs +159 -0
- package/lib/setup/sections/recovery.test.mjs +98 -0
- package/lib/setup/sections/tools.mjs +132 -0
- package/lib/setup/sections/verify.mjs +97 -0
- package/lib/setup/sot.mjs +205 -0
- package/lib/setup/sot.test.mjs +81 -0
- package/lib/setup/state.mjs +151 -0
- package/lib/setup/state.test.mjs +92 -0
- package/lib/singleton.js +229 -0
- package/lib/singleton.test.mjs +135 -0
- package/lib/telemetry/alerts.mjs +216 -0
- package/lib/telemetry/alerts.test.mjs +109 -0
- package/lib/telemetry/collect.mjs +512 -0
- package/lib/telemetry/collect.test.mjs +202 -0
- package/lib/tool-definitions-integration.test.mjs +83 -0
- package/lib/tool-definitions.js +738 -0
- package/lib/tool-definitions.test.mjs +437 -0
- package/lib/util/fetch-timeout.mjs +136 -0
- package/lib/util/fetch-timeout.test.mjs +202 -0
- package/lib/util/reconnect.mjs +343 -0
- package/lib/util/reconnect.test.mjs +369 -0
- package/lib/util/unhandled.mjs +205 -0
- package/lib/util/unhandled.test.mjs +216 -0
- package/lib/voice/context-loader.mjs +466 -0
- package/lib/voice/index.mjs +100 -0
- package/lib/voice/openai-realtime.mjs +510 -0
- package/lib/voice/outbound.mjs +542 -0
- package/lib/voice/outbound.test.mjs +69 -0
- package/lib/voice/post-call-brief.mjs +428 -0
- package/lib/voice/provider.mjs +52 -0
- package/lib/voice/session-rotation.mjs +257 -0
- package/lib/voice/stt.mjs +161 -0
- package/lib/voice/stt.test.mjs +226 -0
- package/lib/voice/tool-bridge.mjs +370 -0
- package/lib/voice/tts.mjs +104 -0
- package/lib/voice/twilio-sip-bridge.mjs +288 -0
- package/lib/voice/voice.test.mjs +990 -0
- package/mcp/README.md +80 -0
- package/package.json +151 -0
- package/plugins/maestro-skills/plugin.json +139 -0
- package/plugins/maestro-skills/skills/agents-observe.md +110 -0
- package/plugins/maestro-skills/skills/board-deck.md +68 -0
- package/plugins/maestro-skills/skills/books-close.md +77 -0
- package/plugins/maestro-skills/skills/brand-steward.md +121 -0
- package/plugins/maestro-skills/skills/calendar-plan.md +57 -0
- package/plugins/maestro-skills/skills/call-working-sessions.md +124 -0
- package/plugins/maestro-skills/skills/ccxray-diagnostics.md +91 -0
- package/plugins/maestro-skills/skills/claude-pace.md +61 -0
- package/plugins/maestro-skills/skills/code-review-graph.md +99 -0
- package/plugins/maestro-skills/skills/crm-pipeline.md +65 -0
- package/plugins/maestro-skills/skills/decision-brief.md +89 -0
- package/plugins/maestro-skills/skills/directory-hygiene.md +125 -0
- package/plugins/maestro-skills/skills/draft-comms.md +84 -0
- package/plugins/maestro-skills/skills/evening-wrap.md +53 -0
- package/plugins/maestro-skills/skills/files-find.md +65 -0
- package/plugins/maestro-skills/skills/generative-ui.md +228 -0
- package/plugins/maestro-skills/skills/hiring-triage.md +74 -0
- package/plugins/maestro-skills/skills/inbox-triage.md +61 -0
- package/plugins/maestro-skills/skills/mail-triage.md +86 -0
- package/plugins/maestro-skills/skills/morning-brief.md +54 -0
- package/plugins/maestro-skills/skills/native-artifacts.md +157 -0
- package/plugins/maestro-skills/skills/org-board.md +133 -0
- package/plugins/maestro-skills/skills/org-credential.md +68 -0
- package/plugins/maestro-skills/skills/org-recall.md +81 -0
- package/plugins/maestro-skills/skills/pipeline-review.md +76 -0
- package/plugins/maestro-skills/skills/regulatory-status.md +81 -0
- package/plugins/maestro-skills/skills/router-why.md +78 -0
- package/plugins/maestro-skills/skills/schedule-meeting.md +91 -0
- package/plugins/maestro-skills/skills/session-search.md +71 -0
- package/plugins/maestro-skills/skills/set-reminder.md +93 -0
- package/plugins/maestro-skills/skills/slack-followup.md +64 -0
- package/plugins/maestro-skills/skills/team-activity.md +86 -0
- package/plugins/maestro-skills/skills/weekly-memo.md +70 -0
- package/policies/action-classification.yaml +114 -0
- package/policies/ai-disclosure.yaml +294 -0
- package/policies/communication-style.md +139 -0
- package/policies/information-barriers.yaml +118 -0
- package/policies/prompt-injection-defence.yaml +138 -0
- package/public/assets/icon-dark.png +0 -0
- package/public/assets/icon-dark.svg +9 -0
- package/public/assets/icon-light.svg +9 -0
- package/public/assets/logo-dark.svg +15 -0
- package/public/assets/logo-light.svg +15 -0
- package/scaffold/.mcp.json +7 -0
- package/scaffold/CLAUDE.md +368 -0
- package/scaffold/config/agent.json +55 -0
- package/scaffold/config/agent.ts +76 -0
- package/scaffold/config/agent.ts.example +89 -0
- package/scaffold/config/alerts.yaml +23 -0
- package/scaffold/config/allowlist.yaml.example +25 -0
- package/scaffold/config/caller-id-map.yaml +46 -0
- package/scaffold/config/collective.yaml +49 -0
- package/scaffold/config/company.json +20 -0
- package/scaffold/config/known-agents.json +6 -0
- package/scaffold/config/learning.yaml +55 -0
- package/scaffold/config/model-routing.yaml.example +104 -0
- package/scaffold/config/org.yaml +25 -0
- package/scaffold/config/orgmail.yaml.example +19 -0
- package/scaffold/config/recovery.yaml +72 -0
- package/scaffold/config/secrets.yaml +27 -0
- package/scaffold/config/slack.yaml.example +35 -0
- package/scaffold/config/telegram.yaml.example +38 -0
- package/scaffold/config/voice.yaml.example +89 -0
- package/scaffold/config/whatsapp.yaml.example +39 -0
- package/schedules/README.md +49 -0
- package/schedules/triggers/backlog-executor.md +102 -0
- package/schedules/triggers/brand-steward.md +72 -0
- package/schedules/triggers/daily-evening-wrap.md +159 -0
- package/schedules/triggers/daily-midday-sweep.md +58 -0
- package/schedules/triggers/daily-morning-brief.md +55 -0
- package/schedules/triggers/directory-hygiene.md +81 -0
- package/schedules/triggers/dynamic-jobs.md +40 -0
- package/schedules/triggers/inbox-processor.md +115 -0
- package/schedules/triggers/meeting-action-capture.md +60 -0
- package/schedules/triggers/meeting-prep.md +69 -0
- package/schedules/triggers/messaging-inbound.md +50 -0
- package/schedules/triggers/org-pulse.md +24 -0
- package/schedules/triggers/quarterly-self-assessment.md +54 -0
- package/schedules/triggers/weekly-engineering-health.md +37 -0
- package/schedules/triggers/weekly-execution.md +65 -0
- package/schedules/triggers/weekly-hiring.md +53 -0
- package/schedules/triggers/weekly-priorities.md +38 -0
- package/schedules/triggers/weekly-strategic-memo.md +124 -0
- package/scripts/archive-email.sh +55 -0
- package/scripts/cadence/cadence-status.mjs +36 -0
- package/scripts/cadence/enqueue-cadence-tick.mjs +174 -0
- package/scripts/cadence/enqueue-cadence-tick.test.mjs +187 -0
- package/scripts/cadence/launchd-cadence-wrapper.sh +85 -0
- package/scripts/cadence/launchd-cloud-relay-wrapper.sh +95 -0
- package/scripts/cadence/launchd-socket-mode-wrapper.sh +95 -0
- package/scripts/ci/check-docs-accuracy.mjs +493 -0
- package/scripts/ci/check-docs-accuracy.test.mjs +409 -0
- package/scripts/ci/check-exports-exist.mjs +140 -0
- package/scripts/ci/check-files-exist.mjs +107 -0
- package/scripts/ci/check-no-build-artifacts.mjs +111 -0
- package/scripts/ci/check-no-build-artifacts.test.mjs +71 -0
- package/scripts/ci/check-no-confidential.mjs +198 -0
- package/scripts/ci/check-no-conflict-markers.mjs +169 -0
- package/scripts/ci/check-no-residual-identity.mjs +163 -0
- package/scripts/ci/check-no-residual-identity.test.mjs +89 -0
- package/scripts/ci/check-tarball-fidelity.mjs +205 -0
- package/scripts/ci/check-unresolved-tokens.mjs +83 -0
- package/scripts/ci/check.mjs +109 -0
- package/scripts/ci/check.test.mjs +194 -0
- package/scripts/ci/run-coverage.mjs +82 -0
- package/scripts/ci/run-tests.mjs +71 -0
- package/scripts/cloud-relay/README.md +59 -0
- package/scripts/cloud-relay/index.mjs +233 -0
- package/scripts/cloud-relay/package.json +15 -0
- package/scripts/cloud-relay/railway.json +13 -0
- package/scripts/cloud-relay/voice/README.md +94 -0
- package/scripts/cloud-relay/voice/package-lock.json +39 -0
- package/scripts/cloud-relay/voice/package.json +16 -0
- package/scripts/cloud-relay/voice/railway.json +13 -0
- package/scripts/cloud-relay/voice/server.mjs +532 -0
- package/scripts/collective/hook-runner.mjs +211 -0
- package/scripts/collective/hook-runner.test.mjs +90 -0
- package/scripts/collective/org-pulse.mjs +72 -0
- package/scripts/collective/org-sync.mjs +61 -0
- package/scripts/collective/recall.mjs +45 -0
- package/scripts/collective/who.mjs +30 -0
- package/scripts/comms-monitor.sh +288 -0
- package/scripts/configure-whatsapp-sandbox.sh +201 -0
- package/scripts/continuous-monitor.sh +91 -0
- package/scripts/cost/fleet-digest.mjs +407 -0
- package/scripts/cost/fleet-digest.test.mjs +207 -0
- package/scripts/cost/track-claude-usage.mjs +169 -0
- package/scripts/daemon/agent-daemon.mjs +989 -0
- package/scripts/daemon/agent-daemon.test.mjs +525 -0
- package/scripts/daemon/cadence-consumer-governance.test.mjs +220 -0
- package/scripts/daemon/cadence-consumer.mjs +1080 -0
- package/scripts/daemon/cadence-consumer.test.mjs +770 -0
- package/scripts/daemon/cadence-handlers.mjs +1121 -0
- package/scripts/daemon/cadence-handlers.test.mjs +617 -0
- package/scripts/daemon/classifier.mjs +704 -0
- package/scripts/daemon/classifier.test.mjs +238 -0
- package/scripts/daemon/classify-kind.mjs +54 -0
- package/scripts/daemon/classify-kind.test.mjs +40 -0
- package/scripts/daemon/context-compiler.mjs +605 -0
- package/scripts/daemon/context-compiler.test.mjs +300 -0
- package/scripts/daemon/dispatcher-cooldown.test.mjs +122 -0
- package/scripts/daemon/dispatcher-governance.test.mjs +886 -0
- package/scripts/daemon/dispatcher.mjs +1516 -0
- package/scripts/daemon/health.mjs +72 -0
- package/scripts/daemon/inbox-deferral.mjs +210 -0
- package/scripts/daemon/inbox-deferral.test.mjs +242 -0
- package/scripts/daemon/integration.test.mjs +149 -0
- package/scripts/daemon/launchd-wrapper-generic.sh +96 -0
- package/scripts/daemon/launchd-wrapper-slack-events.sh +37 -0
- package/scripts/daemon/launchd-wrapper.sh +91 -0
- package/scripts/daemon/lib/session-router.mjs +274 -0
- package/scripts/daemon/lib/session-router.test.mjs +295 -0
- package/scripts/daemon/maestro-daemon.mjs +275 -0
- package/scripts/daemon/prompt-builder.mjs +685 -0
- package/scripts/daemon/prompt-builder.test.mjs +213 -0
- package/scripts/daemon/responder.mjs +854 -0
- package/scripts/daemon/session-lock.mjs +721 -0
- package/scripts/daemon/session-lock.test.mjs +252 -0
- package/scripts/daemon/session-outcomes.mjs +640 -0
- package/scripts/daemon/session-outcomes.test.mjs +533 -0
- package/scripts/daemon/typing-registry.mjs +90 -0
- package/scripts/daemon/typing-registry.test.mjs +77 -0
- package/scripts/daemon/voice-webhook-server.mjs +804 -0
- package/scripts/decisions/capture-decision.mjs +116 -0
- package/scripts/disclosure_assessment.py +873 -0
- package/scripts/disclosure_boundaries.py +562 -0
- package/scripts/email-signature-principal.html +52 -0
- package/scripts/email-signature.html +60 -0
- package/scripts/email_quote_thread.py +167 -0
- package/scripts/email_thread_dedup.py +362 -0
- package/scripts/emergency-stop.sh +81 -0
- package/scripts/healthcheck.sh +116 -0
- package/scripts/hooks/block-mcp-cohort-send.sh +15 -0
- package/scripts/hooks/block-mcp-slack-send.sh +7 -0
- package/scripts/hooks/post-action-log.sh +126 -0
- package/scripts/hooks/pre-send-audit.sh +174 -0
- package/scripts/hooks/pre-send-audit.test.mjs +215 -0
- package/scripts/hooks/session-end-log.sh +27 -0
- package/scripts/hooks/session-start-banner.sh +115 -0
- package/scripts/huddle/audio-bridge.mjs +664 -0
- package/scripts/huddle/boot-slack-cdp.sh +102 -0
- package/scripts/huddle/huddle-controller.mjs +942 -0
- package/scripts/huddle/huddle-server.mjs +1229 -0
- package/scripts/huddle/launch-slack.sh +232 -0
- package/scripts/huddle/openai-realtime-bridge.mjs +462 -0
- package/scripts/huddle/package-lock.json +62 -0
- package/scripts/huddle/package.json +22 -0
- package/scripts/huddle/setup-audio.sh +239 -0
- package/scripts/huddle/start-call.mjs +318 -0
- package/scripts/huddle/test-pipeline.mjs +263 -0
- package/scripts/learning/consolidate-skills.mjs +72 -0
- package/scripts/learning/session-search.mjs +125 -0
- package/scripts/llm_email_dedup.py +442 -0
- package/scripts/local-triggers/generate-plists.sh +432 -0
- package/scripts/local-triggers/generate-plists.test.mjs +413 -0
- package/scripts/local-triggers/install-all.sh +49 -0
- package/scripts/local-triggers/plists/.gitkeep +0 -0
- package/scripts/local-triggers/run-trigger.sh +63 -0
- package/scripts/local-triggers/templates/rag-reindex.plist.template +47 -0
- package/scripts/local-triggers/templates/voice-relay-poller.plist.template +54 -0
- package/scripts/local-triggers/templates/voice-tunnel.plist.template +55 -0
- package/scripts/local-triggers/templates/voice-webhook.plist.template +51 -0
- package/scripts/maintenance/backup-to-cloud.sh +124 -0
- package/scripts/maintenance/health-check.sh +377 -0
- package/scripts/media-generation/README.md +105 -0
- package/scripts/media-generation/gemini-image-client.mjs +173 -0
- package/scripts/media-generation/generate-assets.mjs +289 -0
- package/scripts/media-generation/veo-video-client.mjs +219 -0
- package/scripts/org/send-orgmail.mjs +227 -0
- package/scripts/outbound-dedup-cleanup.sh +43 -0
- package/scripts/outbound-dedup.sh +477 -0
- package/scripts/outbound_dedup.py +115 -0
- package/scripts/parse-voice-transcript.mjs +481 -0
- package/scripts/pdf-generation/README.md +63 -0
- package/scripts/pdf-generation/build-document.mjs +247 -0
- package/scripts/pdf-generation/templates/board-pack.latex +136 -0
- package/scripts/pdf-generation/templates/corporate-letter.latex +126 -0
- package/scripts/pdf-generation/templates/memo.latex +114 -0
- package/scripts/poll-slack-events.sh +35 -0
- package/scripts/poller/calendar-poller.mjs +12 -0
- package/scripts/poller/gmail-poller.mjs +192 -0
- package/scripts/poller/imap-client.mjs +289 -0
- package/scripts/poller/inbox-scan-poller.mjs +156 -0
- package/scripts/poller/inbox-scan-poller.test.mjs +231 -0
- package/scripts/poller/index.mjs +73 -0
- package/scripts/poller/intra-session-check.mjs +285 -0
- package/scripts/poller/lib/cloud-relay-dedup.mjs +88 -0
- package/scripts/poller/lib/cloud-relay-dedup.test.mjs +133 -0
- package/scripts/poller/lib/slash-command-handlers.mjs +177 -0
- package/scripts/poller/secondary-gmail-poller.mjs +132 -0
- package/scripts/poller/slack-cloud-relay-client.mjs +368 -0
- package/scripts/poller/slack-poller.mjs +854 -0
- package/scripts/poller/slack-socket-mode.mjs +917 -0
- package/scripts/poller/slack-socket-mode.test.mjs +753 -0
- package/scripts/poller/trigger.mjs +75 -0
- package/scripts/poller/utils.mjs +371 -0
- package/scripts/poller/voice-cloud-relay-client.mjs +179 -0
- package/scripts/poller/voice-poller.mjs +236 -0
- package/scripts/poller-launchd/install.sh +66 -0
- package/scripts/poller-launchd/poller.plist.template +40 -0
- package/scripts/poller-launchd/whatsapp-handler.plist.template +39 -0
- package/scripts/post-interaction-indexer.py +1598 -0
- package/scripts/pre-draft-context.py +994 -0
- package/scripts/pre_draft_lookup.py +258 -0
- package/scripts/rag/build-index.mjs +47 -0
- package/scripts/rag/ingest.mjs +111 -0
- package/scripts/rag/search.mjs +119 -0
- package/scripts/rag-indexer.py +629 -0
- package/scripts/restore-from-backup.sh +248 -0
- package/scripts/restore-from-backup.test.mjs +178 -0
- package/scripts/resume-operations.sh +80 -0
- package/scripts/search-secondary-inbox.py +181 -0
- package/scripts/secondary-inbox-poller.py +437 -0
- package/scripts/self-optimization/compute-metrics.py +398 -0
- package/scripts/send-email-as-principal.py +369 -0
- package/scripts/send-email-threaded.py +392 -0
- package/scripts/send-email-with-attachment.py +377 -0
- package/scripts/send-email.sh +131 -0
- package/scripts/send-sms.sh +175 -0
- package/scripts/send-whatsapp.sh +292 -0
- package/scripts/session-start.sh +106 -0
- package/scripts/setup/boot-claude-session.sh +94 -0
- package/scripts/setup/configure-macos.sh +674 -0
- package/scripts/setup/configure-twilio-sip-trunk.mjs +207 -0
- package/scripts/setup/configure-voice-tunnel.mjs +182 -0
- package/scripts/setup/generate-agent-env.mjs +92 -0
- package/scripts/setup/generate-agent-package-json.mjs +222 -0
- package/scripts/setup/generate-agent-package-json.test.mjs +143 -0
- package/scripts/setup/generate-autonomy.mjs +60 -0
- package/scripts/setup/generate-backlog.mjs +101 -0
- package/scripts/setup/generate-cadences.mjs +92 -0
- package/scripts/setup/generate-capability.mjs +162 -0
- package/scripts/setup/generate-charter.mjs +76 -0
- package/scripts/setup/generate-comms.mjs +59 -0
- package/scripts/setup/generate-company.mjs +92 -0
- package/scripts/setup/init-agent.sh +547 -0
- package/scripts/setup/init-agent.test.mjs +151 -0
- package/scripts/setup/init-archetype.mjs +74 -0
- package/scripts/setup/init-backup.mjs +54 -0
- package/scripts/setup/init-cadence-bus.mjs +60 -0
- package/scripts/setup/init-channel-bus.mjs +46 -0
- package/scripts/setup/init-cost-tracking.mjs +45 -0
- package/scripts/setup/init-decision-capture.mjs +66 -0
- package/scripts/setup/init-known-agents.mjs +57 -0
- package/scripts/setup/init-learning.mjs +70 -0
- package/scripts/setup/init-memory-executive.mjs +45 -0
- package/scripts/setup/init-model-router.mjs +124 -0
- package/scripts/setup/init-rag.mjs +174 -0
- package/scripts/setup/init-session-router.mjs +38 -0
- package/scripts/setup/init-slack-socket-mode.mjs +260 -0
- package/scripts/setup/init-telegram.mjs +165 -0
- package/scripts/setup/init-voice-realtime.mjs +204 -0
- package/scripts/setup/init-whatsapp-baileys.mjs +77 -0
- package/scripts/setup/install-dev-tools.sh +150 -0
- package/scripts/setup/lib/install-plist.mjs +95 -0
- package/scripts/setup/migrate-agent-to-sot.mjs +192 -0
- package/scripts/setup/render-environment-yaml.mjs +133 -0
- package/scripts/slack-events-ctl.sh +177 -0
- package/scripts/slack-events-server.mjs +1045 -0
- package/scripts/slack-react.mjs +89 -0
- package/scripts/slack-responded.sh +232 -0
- package/scripts/slack-send.sh +287 -0
- package/scripts/slack-typing.mjs +196 -0
- package/scripts/slack-upload-v2.py +95 -0
- package/scripts/sms-handler.mjs +450 -0
- package/scripts/spawn-session.sh +120 -0
- package/scripts/sync-protocol.mjs +217 -0
- package/scripts/system-verify.sh +184 -0
- package/scripts/test-email-thread-dedup.py +239 -0
- package/scripts/test-information-barriers.py +484 -0
- package/scripts/test-llm-email-dedup.py +251 -0
- package/scripts/test-pre-draft-integration.py +203 -0
- package/scripts/test-rag-phase2.sh +442 -0
- package/scripts/test-rag-search.sh +251 -0
- package/scripts/test-voice-parser.mjs +316 -0
- package/scripts/user-context-search.py +659 -0
- package/scripts/validate_outbound.py +1504 -0
- package/scripts/watchdog/ai.maestro.memory-watchdog.plist +41 -0
- package/scripts/watchdog/force-reboot.sh +157 -0
- package/scripts/watchdog/memory-watchdog.sh +473 -0
- package/scripts/whatsapp-handler.mjs +538 -0
- package/teams/desktop-operations.yaml +34 -0
- package/teams/executive-office.yaml +27 -0
- package/teams/legal-and-regulatory.yaml +24 -0
- package/teams/platform-and-engineering.yaml +23 -0
- package/teams/strategy-and-growth.yaml +29 -0
- package/workflows/continuous/backlog-executor.yaml +141 -0
- package/workflows/continuous/inbound-monitor.yaml +168 -0
- package/workflows/daily/applicant-triage.yaml +197 -0
- package/workflows/daily/comms-triage.yaml +80 -0
- package/workflows/daily/evening-wrap.yaml +105 -0
- package/workflows/daily/morning-brief.yaml +164 -0
- package/workflows/daily/slack-followup-sweep.yaml +87 -0
- package/workflows/event-driven/README.md +50 -0
- package/workflows/event-driven/agent-failure-investigation.yaml +137 -0
- package/workflows/event-driven/pr-review.yaml +107 -0
- package/workflows/monthly/board-readiness.yaml +76 -0
- package/workflows/quarterly/strategic-scenario-analysis.yaml +85 -0
- package/workflows/session-protocol.md +171 -0
- package/workflows/weekly/engineering-health.yaml +154 -0
- package/workflows/weekly/hiring-review.yaml +169 -0
- package/workflows/weekly/rollup-pipeline-review.yaml +76 -0
- package/workflows/weekly/strategic-memo.yaml +79 -0
|
@@ -0,0 +1,1516 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// Dispatcher — manages concurrent claude --print sessions
|
|
3
|
+
// Spawns child processes, manages queue, enforces concurrency cap
|
|
4
|
+
|
|
5
|
+
import { spawn } from "child_process";
|
|
6
|
+
import { appendFileSync, mkdirSync, writeFileSync, readFileSync, renameSync, existsSync, readdirSync, unlinkSync } from "fs";
|
|
7
|
+
import { randomUUID } from "crypto";
|
|
8
|
+
import { join, dirname } from "path";
|
|
9
|
+
import { releaseLock, releaseThreadLock, releaseRequestClaim, claimItem, releaseItemClaim } from "./session-lock.mjs";
|
|
10
|
+
import { promoteDeferred } from "./inbox-deferral.mjs";
|
|
11
|
+
import { recordSession } from "./health.mjs";
|
|
12
|
+
import { startTyping, stopTyping } from "./typing-registry.mjs";
|
|
13
|
+
// Permission scoping (security CRITICAL / audit H1). sessionPermissionArgs()
|
|
14
|
+
// returns ["--dangerously-skip-permissions"] by default (byte-for-byte the prior
|
|
15
|
+
// hardcoded literal) OR a scoped ["--allowedTools", "<list>"] when the operator
|
|
16
|
+
// sets MAESTRO_SCOPED_PERMISSIONS=1. Spawns route through it so opted-in scoping
|
|
17
|
+
// is actually honoured instead of being bypassed on the user-facing path.
|
|
18
|
+
import { sessionPermissionArgs } from "../../lib/session-permissions.mjs";
|
|
19
|
+
|
|
20
|
+
// WS4 — resource governance + cross-session 429 breaker + daily budget cap.
|
|
21
|
+
// Every spawn is gated on admission (ADMIT/QUEUE/DEFER) and the shared rate
|
|
22
|
+
// breaker so the agent throttles instead of bricking and never exceeds the
|
|
23
|
+
// host. All three are injectable (setGovernanceForTests) so the dispatcher
|
|
24
|
+
// stays hermetically testable.
|
|
25
|
+
import * as resourceGovernor from "../../lib/resource-governor.mjs";
|
|
26
|
+
import * as rateGuardModule from "../../lib/rate-guard.mjs";
|
|
27
|
+
import * as budgetGuardModule from "../../lib/budget-guard.mjs";
|
|
28
|
+
|
|
29
|
+
let governor = resourceGovernor;
|
|
30
|
+
let rateGuard = rateGuardModule;
|
|
31
|
+
let budgetGuard = budgetGuardModule;
|
|
32
|
+
const RATE_PROVIDER = process.env.MAESTRO_RATE_PROVIDER || "anthropic";
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Test seam: swap the governance modules for fakes. Each arg is optional;
|
|
36
|
+
* omitted modules keep the real implementation. Returns a restore fn.
|
|
37
|
+
*/
|
|
38
|
+
export function setGovernanceForTests({ governor: g, rateGuard: r, budgetGuard: b } = {}) {
|
|
39
|
+
const prev = { governor, rateGuard, budgetGuard };
|
|
40
|
+
if (g) governor = g;
|
|
41
|
+
if (r) rateGuard = r;
|
|
42
|
+
if (b) budgetGuard = b;
|
|
43
|
+
return () => { governor = prev.governor; rateGuard = prev.rateGuard; budgetGuard = prev.budgetGuard; };
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// Indirection so a test can capture the EXACT argv handed to `claude` by the
|
|
47
|
+
// real resume spawner (defaultSpawnResume) without launching a child process.
|
|
48
|
+
// Production default is the imported `spawn`, so behaviour is unchanged.
|
|
49
|
+
let _spawn = spawn;
|
|
50
|
+
/** Test seam: replace child_process.spawn. Returns a restore fn. */
|
|
51
|
+
export function setSpawnForTests(fn) {
|
|
52
|
+
const prev = _spawn;
|
|
53
|
+
_spawn = typeof fn === "function" ? fn : spawn;
|
|
54
|
+
return () => { _spawn = prev; };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Resolve the current admission decision for a (source) under live host load,
|
|
59
|
+
* folding in the daily-budget essential-only mode. Best-effort: any throw is
|
|
60
|
+
* treated as ADMIT (fail-open-for-work — Invariant: never drop/stall work on a
|
|
61
|
+
* governance bug).
|
|
62
|
+
*/
|
|
63
|
+
function admitFor(source, priority) {
|
|
64
|
+
try {
|
|
65
|
+
let mode = null;
|
|
66
|
+
try {
|
|
67
|
+
if (budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR }).essentialOnly) mode = "essential-only";
|
|
68
|
+
} catch { /* budget read best-effort */ }
|
|
69
|
+
return governor.admit({ source, priority, mode }, governor.defaultDeps({ agentRoot: AGENT_REPO_DIR }));
|
|
70
|
+
} catch {
|
|
71
|
+
return { decision: "ADMIT", reason: "governor-error-fail-open", snapshot: {} };
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** Is the shared 429 breaker currently open for our provider? */
|
|
76
|
+
function rateBlocked() {
|
|
77
|
+
try {
|
|
78
|
+
const r = rateGuard.checkRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR });
|
|
79
|
+
return r.allowed ? null : r;
|
|
80
|
+
} catch {
|
|
81
|
+
return null; // fail-open-for-work
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const AGENT_REPO_DIR = process.env.AGENT_DIR || join(new URL(".", import.meta.url).pathname, "../..");
|
|
86
|
+
// Resolve the claude binary against the agent's PATH (not launchd's bare
|
|
87
|
+
// env). Without this, every daemon-spawned `claude --print` exits ENOENT.
|
|
88
|
+
import { resolveClaudeBin, augmentedPath, daemonClaudeArgs } from "../../lib/claude-bin.mjs";
|
|
89
|
+
const CLAUDE_BIN = resolveClaudeBin();
|
|
90
|
+
|
|
91
|
+
// Model router — opt-in. When config/model-routing.yaml is present in the
|
|
92
|
+
// agent repo, each spawn is routed to the cheapest backend that satisfies
|
|
93
|
+
// the request's capability needs. When absent, the resolver returns null
|
|
94
|
+
// and we preserve the current Claude-CLI-on-Max-subscription behaviour.
|
|
95
|
+
//
|
|
96
|
+
// Two routing paths now coexist (SPEC §3 — additive, byte-compatible):
|
|
97
|
+
// - v1 (legacy): resolveBackend → modelFlagFor → envForSpawn. Preserved
|
|
98
|
+
// verbatim for v1 configs (no schema_version) and as the safety net.
|
|
99
|
+
// - v2 (schema_version: 2): resolveChain → RouteDecision. The decision carries
|
|
100
|
+
// the chosen backend, the session-retarget env, the spawn knobs, the
|
|
101
|
+
// ordered failover chain, an estimated cost, and a decision_id that joins
|
|
102
|
+
// the ledger row ↔ resume marker ↔ routing audit. We build the argv + child
|
|
103
|
+
// env from the decision via the execution layer's pure builders
|
|
104
|
+
// (buildSpawnArgs / buildChildEnv from lib/model-router/spawn.mjs) so the
|
|
105
|
+
// §7.3 child-env scrubbing + argv contract match spawnRouted exactly.
|
|
106
|
+
//
|
|
107
|
+
// NOTE on spawnRouted: the daemon's spawnSession() must register the child
|
|
108
|
+
// process SYNCHRONOUSLY (tests + getStatus() observe active_sessions right after
|
|
109
|
+
// dispatch, and the child's close/error/timeout handlers own cooldowns, resume
|
|
110
|
+
// markers, lock release, and the cost ledger). spawnRouted owns an ASYNC
|
|
111
|
+
// fast-fail failover loop that resolves a Promise — incompatible with that
|
|
112
|
+
// synchronous, externally-driven-close contract here without a rewrite that
|
|
113
|
+
// risks regressions. We therefore reuse spawnRouted's pure argv/env BUILDERS at
|
|
114
|
+
// this seam and keep the existing synchronous _spawn + handlers; the async
|
|
115
|
+
// failover loop is the right fit for the scripted one-shot/cadence sites (which
|
|
116
|
+
// already drive their own close). Cross-backend failover at the dispatcher is a
|
|
117
|
+
// precise follow-up (see the spawn-site checklist in the WS summary).
|
|
118
|
+
import {
|
|
119
|
+
loadRoutingConfig,
|
|
120
|
+
resolveBackend,
|
|
121
|
+
requestFromClassifierResult,
|
|
122
|
+
modelFlagFor,
|
|
123
|
+
resolveChain,
|
|
124
|
+
} from "../../lib/model-router.mjs";
|
|
125
|
+
import { buildSpawnArgs, buildChildEnv } from "../../lib/model-router/spawn.mjs";
|
|
126
|
+
import { budgetLadder, spawnKnobsFor } from "../../lib/model-router/economics.mjs";
|
|
127
|
+
// Observability spine (WS — diagnostics). The dispatcher emits `dispatched` at
|
|
128
|
+
// spawn and `session_opened`/`session_closed` from the existing close hook,
|
|
129
|
+
// attaching the interaction's trace_id (handed off explicitly on item.trace_id —
|
|
130
|
+
// AsyncLocalStorage cannot cross the proc-event boundary) + the v2 decision_id.
|
|
131
|
+
// emitEvent is fail-open (never throws), so this is purely additive telemetry.
|
|
132
|
+
import { emitEvent, EVENT_TYPES } from "../../lib/diagnostics/events.mjs";
|
|
133
|
+
// Lazy + cached so a misconfigured YAML doesn't break agents that didn't
|
|
134
|
+
// opt in. The cache is invalidated only on daemon restart.
|
|
135
|
+
let _routingConfigCache;
|
|
136
|
+
function getRoutingConfig() {
|
|
137
|
+
if (_routingConfigCache === undefined) {
|
|
138
|
+
try {
|
|
139
|
+
_routingConfigCache = loadRoutingConfig(AGENT_REPO_DIR) || null;
|
|
140
|
+
} catch (err) {
|
|
141
|
+
console.warn(`[dispatcher] model-router config invalid, falling back to Anthropic CLI: ${err.message}`);
|
|
142
|
+
_routingConfigCache = null;
|
|
143
|
+
}
|
|
144
|
+
}
|
|
145
|
+
return _routingConfigCache;
|
|
146
|
+
}
|
|
147
|
+
const MAX_CONCURRENT = parseInt(process.env.DAEMON_MAX_CONCURRENT || "10", 10);
|
|
148
|
+
const RESERVED_INBOX_SLOTS = 3; // Always keep 3 slots free for real-time inbox items
|
|
149
|
+
|
|
150
|
+
// Timeouts — tiered by model and source
|
|
151
|
+
const SONNET_INBOX_TIMEOUT = 10 * 60 * 1000; // 10 min (inbox — user expects a reply)
|
|
152
|
+
const SONNET_BACKLOG_TIMEOUT = 30 * 60 * 1000; // 30 min (backlog tasks need more time)
|
|
153
|
+
const OPUS_INBOX_TIMEOUT = 45 * 60 * 1000; // 45 min (complex inbox — CEO requests, research)
|
|
154
|
+
const OPUS_BACKLOG_TIMEOUT = 12 * 60 * 60 * 1000; // 12 hours (ultra-complex backlog — deep work)
|
|
155
|
+
|
|
156
|
+
// Legacy aliases for compatibility
|
|
157
|
+
const SONNET_TIMEOUT = SONNET_INBOX_TIMEOUT;
|
|
158
|
+
const OPUS_TIMEOUT = OPUS_INBOX_TIMEOUT;
|
|
159
|
+
|
|
160
|
+
const activeSessions = new Map(); // sessionId -> { process, item, startTime, model, source }
|
|
161
|
+
const priorityQueue = []; // critical/high items
|
|
162
|
+
const normalQueue = []; // normal items
|
|
163
|
+
let sessionCounter = 0;
|
|
164
|
+
|
|
165
|
+
// Tracks sessions whose proc.on("error") handler has already fired.
|
|
166
|
+
// Prevents double-counting + double-cleanup when a spawn failure (ENOENT,
|
|
167
|
+
// EACCES, ETIMEDOUT) triggers both "error" and a trailing "close" event.
|
|
168
|
+
// See ib-20260416-daemon-etimedout-failed-event + cycle 135 memo.
|
|
169
|
+
const spawnErrorHandled = new Set();
|
|
170
|
+
|
|
171
|
+
// Backlog dedup: track which items have active sessions to prevent retry storms
|
|
172
|
+
const activeBacklogKeys = new Set(); // backlog item key -> true (while session is running)
|
|
173
|
+
const backlogRetryCount = new Map(); // backlog item key -> number of times dispatched
|
|
174
|
+
const MAX_BACKLOG_RETRIES = 6; // Max retries before skipping (was 3 — too aggressive)
|
|
175
|
+
|
|
176
|
+
// Post-completion cooldown — once a session has run on a backlog item, don't
|
|
177
|
+
// re-dispatch it until N hours later. Without this, every 2-min backlog
|
|
178
|
+
// sweep re-dispatches the same items because the daemon has no signal
|
|
179
|
+
// that the underlying work was actually completed (sessions exit 0 even
|
|
180
|
+
// when they only "looked at" the item). 53 redundant spawns/day per item
|
|
181
|
+
// was the observed rate before this fix.
|
|
182
|
+
const SUCCESS_COOLDOWN_MS = 4 * 60 * 60 * 1000; // 4h after exit 0
|
|
183
|
+
const FAILURE_COOLDOWN_MS = 30 * 60 * 1000; // 30m after non-zero exit
|
|
184
|
+
const backlogCooldownUntil = new Map(); // key -> epoch ms
|
|
185
|
+
const COOLDOWN_STATE_PATH = join(AGENT_REPO_DIR, "state/sessions/backlog-cooldowns.json");
|
|
186
|
+
|
|
187
|
+
// Persist cooldown state across daemon restarts so a freshly-started
|
|
188
|
+
// daemon doesn't immediately re-dispatch items it just completed.
|
|
189
|
+
function loadCooldowns() {
|
|
190
|
+
try {
|
|
191
|
+
const body = readFileSync(COOLDOWN_STATE_PATH, "utf-8");
|
|
192
|
+
const data = JSON.parse(body);
|
|
193
|
+
const now = Date.now();
|
|
194
|
+
for (const [key, until] of Object.entries(data || {})) {
|
|
195
|
+
if (typeof until === "number" && until > now) {
|
|
196
|
+
backlogCooldownUntil.set(key, until);
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
} catch { /* file missing or malformed — start fresh */ }
|
|
200
|
+
}
|
|
201
|
+
function saveCooldowns() {
|
|
202
|
+
try {
|
|
203
|
+
const obj = {};
|
|
204
|
+
for (const [k, v] of backlogCooldownUntil) obj[k] = v;
|
|
205
|
+
mkdirSync(dirname(COOLDOWN_STATE_PATH), { recursive: true });
|
|
206
|
+
writeFileSync(COOLDOWN_STATE_PATH, JSON.stringify(obj, null, 2) + "\n");
|
|
207
|
+
} catch { /* best-effort */ }
|
|
208
|
+
}
|
|
209
|
+
loadCooldowns();
|
|
210
|
+
|
|
211
|
+
/**
|
|
212
|
+
* Parse the `claude --print --output-format json` result object out of a
|
|
213
|
+
* captured stdout STRING (the dispatcher buffers stdout in memory) to recover
|
|
214
|
+
* the run's REAL token usage + authoritative cost for the ledger (recovery C1).
|
|
215
|
+
* Returns { ok, inputTokens, outputTokens, cacheReadTokens?, totalCostUsd?,
|
|
216
|
+
* model? } or { ok:false, reason } — never throws, never fabricates counts.
|
|
217
|
+
*/
|
|
218
|
+
export function parseUsageFromText(text) {
|
|
219
|
+
const trimmed = (text || "").trim();
|
|
220
|
+
if (!trimmed) return { ok: false, reason: "empty-stdout" };
|
|
221
|
+
let obj = null;
|
|
222
|
+
try { obj = JSON.parse(trimmed); }
|
|
223
|
+
catch {
|
|
224
|
+
// Tolerant path: a leading banner/log line can precede the JSON tail. The
|
|
225
|
+
// CLI's result object is the FIRST balanced top-level object, so slice from
|
|
226
|
+
// the first "{" to the last "}". (Using lastIndexOf("{") would wrongly grab
|
|
227
|
+
// a nested object's opening brace.)
|
|
228
|
+
const start = trimmed.indexOf("{");
|
|
229
|
+
const end = trimmed.lastIndexOf("}");
|
|
230
|
+
if (start !== -1 && end !== -1 && end > start) {
|
|
231
|
+
try { obj = JSON.parse(trimmed.slice(start, end + 1)); } catch { obj = null; }
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
if (!obj || typeof obj !== "object") return { ok: false, reason: "no-json-object" };
|
|
235
|
+
const usage = obj.usage && typeof obj.usage === "object" ? obj.usage : null;
|
|
236
|
+
if (!usage) return { ok: false, reason: "no-usage-field" };
|
|
237
|
+
const inputTokens = Number(usage.input_tokens);
|
|
238
|
+
const outputTokens = Number(usage.output_tokens);
|
|
239
|
+
if (!Number.isFinite(inputTokens) || !Number.isFinite(outputTokens)) {
|
|
240
|
+
return { ok: false, reason: "non-numeric-tokens" };
|
|
241
|
+
}
|
|
242
|
+
const out = { ok: true, inputTokens, outputTokens };
|
|
243
|
+
const cacheRead = Number(usage.cache_read_input_tokens);
|
|
244
|
+
if (Number.isFinite(cacheRead)) out.cacheReadTokens = cacheRead;
|
|
245
|
+
const totalCost = Number(obj.total_cost_usd);
|
|
246
|
+
if (Number.isFinite(totalCost) && totalCost >= 0) out.totalCostUsd = totalCost;
|
|
247
|
+
if (typeof obj.model === "string" && obj.model) {
|
|
248
|
+
out.model = /opus/i.test(obj.model) ? "opus" : /haiku/i.test(obj.model) ? "haiku" : "sonnet";
|
|
249
|
+
}
|
|
250
|
+
return out;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* Append a TRUTHFUL cost-ledger row for a finished dispatcher session via
|
|
255
|
+
* scripts/cost/track-claude-usage.mjs (source "dispatcher"). Real token counts
|
|
256
|
+
* are parsed from the run's --output-format json stdout; on a parse failure we
|
|
257
|
+
* pass NO token flags (tracker records 0) rather than fabricate zeros, and the
|
|
258
|
+
* caller logs the gap. Best-effort + detached; never blocks the close path.
|
|
259
|
+
*/
|
|
260
|
+
function recordDispatcherCost({ stdout, model, durationMs, exitCode, decisionId }) {
|
|
261
|
+
const usage = parseUsageFromText(stdout);
|
|
262
|
+
try {
|
|
263
|
+
const trackerPath = join(AGENT_REPO_DIR, "scripts/cost/track-claude-usage.mjs");
|
|
264
|
+
if (!existsSync(trackerPath)) return usage;
|
|
265
|
+
const trackerArgs = [
|
|
266
|
+
trackerPath, "record",
|
|
267
|
+
"--cadence", "inbox",
|
|
268
|
+
"--source", "dispatcher",
|
|
269
|
+
"--model", usage.model || model || "sonnet",
|
|
270
|
+
"--duration-ms", String(durationMs),
|
|
271
|
+
"--exit", String(exitCode),
|
|
272
|
+
];
|
|
273
|
+
// Carry the v2 RouteDecision id onto the ledger row so a spend row joins its
|
|
274
|
+
// routing decision + resume marker (SPEC §4.7). The tracker captures it as a
|
|
275
|
+
// generic flag; absent on the v1 / no-config paths.
|
|
276
|
+
if (decisionId) trackerArgs.push("--decision-id", String(decisionId));
|
|
277
|
+
if (usage.ok) {
|
|
278
|
+
trackerArgs.push("--input-tokens", String(usage.inputTokens));
|
|
279
|
+
trackerArgs.push("--output-tokens", String(usage.outputTokens));
|
|
280
|
+
if (usage.cacheReadTokens != null) trackerArgs.push("--cache-read-tokens", String(usage.cacheReadTokens));
|
|
281
|
+
if (usage.totalCostUsd != null) trackerArgs.push("--total-cost-usd", String(usage.totalCostUsd));
|
|
282
|
+
}
|
|
283
|
+
spawn(process.execPath, trackerArgs, {
|
|
284
|
+
stdio: "ignore",
|
|
285
|
+
env: { ...process.env, AGENT_ROOT: AGENT_REPO_DIR, AGENT_DIR: AGENT_REPO_DIR },
|
|
286
|
+
}).unref();
|
|
287
|
+
} catch { /* cost tracking is best-effort */ }
|
|
288
|
+
return usage;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
function logDir() {
|
|
292
|
+
const dir = join(AGENT_REPO_DIR, "logs", "daemon");
|
|
293
|
+
mkdirSync(dir, { recursive: true });
|
|
294
|
+
return dir;
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
function sessionLogDir() {
|
|
298
|
+
const dir = join(AGENT_REPO_DIR, "logs", "daemon", "sessions");
|
|
299
|
+
mkdirSync(dir, { recursive: true });
|
|
300
|
+
return dir;
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
function today() {
|
|
304
|
+
return new Date().toISOString().split("T")[0];
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
function logSession(entry) {
|
|
308
|
+
const path = join(logDir(), `${today()}-sessions.jsonl`);
|
|
309
|
+
appendFileSync(path, JSON.stringify({ timestamp: new Date().toISOString(), ...entry }) + "\n");
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
const ACTIVE_PATH = join(AGENT_REPO_DIR, "state", "sessions", "active.json");
|
|
313
|
+
|
|
314
|
+
// ---------------------------------------------------------------------------
|
|
315
|
+
// WS4 — in-flight session resume markers
|
|
316
|
+
// ---------------------------------------------------------------------------
|
|
317
|
+
// A marker is written to state/sessions/resume-pending/<sessionId>.json right
|
|
318
|
+
// after spawn and DELETED on a clean close. Its presence after a crash/reboot
|
|
319
|
+
// means "this session was mid-flight" — resetActiveSessions() reconciles them
|
|
320
|
+
// (re-dispatching `claude --print --session-id <claudeSessionId> <prompt>`,
|
|
321
|
+
// the SAME continuation mechanism responder.mjs uses — NOT `--resume`) within a
|
|
322
|
+
// freshness window, instead of blindly wiping the slate. 3 strikes → the queue
|
|
323
|
+
// item is marked status:blocked. This is the "never brick / never drop work"
|
|
324
|
+
// recovery path for the dispatcher.
|
|
325
|
+
const RESUME_PENDING_DIR = join(AGENT_REPO_DIR, "state", "sessions", "resume-pending");
|
|
326
|
+
// Sessions whose marker is older than this are too stale to resume meaningfully
|
|
327
|
+
// (the work context has moved on); reconcile blocks them rather than re-running.
|
|
328
|
+
const RESUME_FRESHNESS_MS = parseInt(process.env.MAESTRO_RESUME_FRESHNESS_MS || String(6 * 60 * 60 * 1000), 10);
|
|
329
|
+
const RESUME_MAX_ATTEMPTS = 3;
|
|
330
|
+
|
|
331
|
+
function resumePendingPath(sessionId) {
|
|
332
|
+
return join(RESUME_PENDING_DIR, `${sessionId}.json`);
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
/**
|
|
336
|
+
* Write the resume marker for a freshly-spawned session. Best-effort: a marker
|
|
337
|
+
* we can't write just means that session won't be auto-resumed after a crash
|
|
338
|
+
* (it still completes normally) — never throws into the spawn path.
|
|
339
|
+
*/
|
|
340
|
+
function writeResumePending(sessionId, entry, claudeSessionId, modelFlag, decisionId) {
|
|
341
|
+
try {
|
|
342
|
+
mkdirSync(RESUME_PENDING_DIR, { recursive: true });
|
|
343
|
+
const marker = {
|
|
344
|
+
sessionId,
|
|
345
|
+
claudeSessionId: claudeSessionId || null,
|
|
346
|
+
itemRef: entry.item?.raw_ref || entry.item?.id || entry.item?.title || null,
|
|
347
|
+
itemId: entry.item?.id || null,
|
|
348
|
+
sourceFile: entry.item?.source_file || null,
|
|
349
|
+
source: entry.source,
|
|
350
|
+
transcriptPath: entry.transcriptPath || null,
|
|
351
|
+
summary: entry.classResult?.summary || null,
|
|
352
|
+
model: entry.classResult?.model || "sonnet",
|
|
353
|
+
// The exact `--model` flag value the original spawn used (post model-
|
|
354
|
+
// router resolution). Persisted so a reboot-resume re-spawns against the
|
|
355
|
+
// SAME backend, not just the coarse sonnet/opus class. (H1)
|
|
356
|
+
modelFlag: modelFlag || entry.classResult?.model || "sonnet",
|
|
357
|
+
// The v2 RouteDecision id (when routed through resolveChain). Persisted so
|
|
358
|
+
// a reboot-resume can correlate the resumed run with the original routing
|
|
359
|
+
// decision + ledger row (SPEC §4.8). Null on the v1 / no-config paths.
|
|
360
|
+
decision_id: decisionId || null,
|
|
361
|
+
// The original prompt is what makes a true resume possible: the resume
|
|
362
|
+
// path re-spawns `claude --print --session-id <id> <prompt>` (NOT
|
|
363
|
+
// `--resume`), mirroring responder.mjs's blessed continuation pattern.
|
|
364
|
+
// Without it a resumed spawn would have no work to do and exit 0, which
|
|
365
|
+
// the close handler would mistake for "done". (H1)
|
|
366
|
+
prompt: typeof entry.prompt === "string" ? entry.prompt : null,
|
|
367
|
+
priority: entry.classResult?.priority || "normal",
|
|
368
|
+
startedAt: Date.now(),
|
|
369
|
+
recoveryAttempts: typeof entry.recoveryAttempts === "number" ? entry.recoveryAttempts : 0,
|
|
370
|
+
};
|
|
371
|
+
const tmp = resumePendingPath(sessionId) + ".tmp";
|
|
372
|
+
writeFileSync(tmp, JSON.stringify(marker, null, 2));
|
|
373
|
+
renameSync(tmp, resumePendingPath(sessionId));
|
|
374
|
+
} catch (err) {
|
|
375
|
+
console.warn(`[dispatcher] Failed to write resume-pending marker for ${sessionId}: ${err.message}`);
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
/** Delete the resume marker on a clean close. Best-effort. */
|
|
380
|
+
function clearResumePending(sessionId) {
|
|
381
|
+
try { unlinkSync(resumePendingPath(sessionId)); } catch { /* already gone */ }
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
/**
|
|
385
|
+
* Mark a queue item blocked (after 3 failed resume attempts) so the backlog
|
|
386
|
+
* sweep stops re-dispatching it and an operator can see why. Edits the queue
|
|
387
|
+
* YAML in place by flipping the item's status to `blocked` and appending a
|
|
388
|
+
* reason comment. Best-effort + behavior-preserving: if we can't locate the
|
|
389
|
+
* item we no-op (the cooldown/retry caps still bound re-dispatch).
|
|
390
|
+
*/
|
|
391
|
+
function markItemBlocked(marker, reason) {
|
|
392
|
+
try {
|
|
393
|
+
if (!marker.itemId || !marker.sourceFile) return false;
|
|
394
|
+
const queuePath = join(AGENT_REPO_DIR, "state", "queues", marker.sourceFile);
|
|
395
|
+
if (!existsSync(queuePath)) return false;
|
|
396
|
+
let body = readFileSync(queuePath, "utf-8");
|
|
397
|
+
// Find the item's id line and flip the nearest status: within its block.
|
|
398
|
+
const idRe = new RegExp(`(^|\\n)(\\s*)(-?\\s*)"?id"?:\\s*["']?${marker.itemId.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}["']?`, "m");
|
|
399
|
+
const m = idRe.exec(body);
|
|
400
|
+
if (!m) return false;
|
|
401
|
+
// Replace the first status: line after the id with status: blocked.
|
|
402
|
+
const after = body.slice(m.index);
|
|
403
|
+
const replacedAfter = after.replace(/("?status"?:\s*)["']?(open|in_progress|pending)["']?/, `$1blocked # WS4 recovery: ${reason}`);
|
|
404
|
+
if (replacedAfter === after) return false;
|
|
405
|
+
body = body.slice(0, m.index) + replacedAfter;
|
|
406
|
+
const tmp = queuePath + ".tmp";
|
|
407
|
+
writeFileSync(tmp, body);
|
|
408
|
+
renameSync(tmp, queuePath);
|
|
409
|
+
logSession({ event: "item_blocked_after_resume_strikes", item_id: marker.itemId, source_file: marker.sourceFile, reason });
|
|
410
|
+
return true;
|
|
411
|
+
} catch (err) {
|
|
412
|
+
console.warn(`[dispatcher] markItemBlocked failed: ${err.message}`);
|
|
413
|
+
return false;
|
|
414
|
+
}
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
/**
|
|
418
|
+
* Reset active.json + (on startup) reconcile in-flight session resume markers.
|
|
419
|
+
*
|
|
420
|
+
* Previous daemon instances may have left stale active.json entries (killed
|
|
421
|
+
* sessions whose close handlers never fired). The in-memory session map starts
|
|
422
|
+
* empty, so active.json is always cleared.
|
|
423
|
+
*
|
|
424
|
+
* WS4 — instead of *also* blindly discarding any work that was mid-flight when
|
|
425
|
+
* the box rebooted/lost power, when `opts.reconcile` is set we walk
|
|
426
|
+
* state/sessions/resume-pending/ and, for each marker still inside the
|
|
427
|
+
* freshness window, re-spawn `claude --print --session-id <claudeSessionId>
|
|
428
|
+
* <prompt>` (the responder's continuation pattern — NOT `--resume`) so the work
|
|
429
|
+
* continues where it left off. After RESUME_MAX_ATTEMPTS (3) the item is marked
|
|
430
|
+
* status:blocked with a reason rather than retried forever.
|
|
431
|
+
*
|
|
432
|
+
* Graceful shutdown passes no opts (just clears active.json); only startup
|
|
433
|
+
* reconciles, so we never re-spawn work we're deliberately stopping.
|
|
434
|
+
*
|
|
435
|
+
* @param {object} [opts] { reconcile?: boolean, spawnResume?: fn (tests) }
|
|
436
|
+
*/
|
|
437
|
+
export function resetActiveSessions(opts = {}) {
|
|
438
|
+
try {
|
|
439
|
+
let staleCount = 0;
|
|
440
|
+
try {
|
|
441
|
+
const old = JSON.parse(readFileSync(ACTIVE_PATH, "utf-8"));
|
|
442
|
+
staleCount = Object.keys(old).length;
|
|
443
|
+
} catch {}
|
|
444
|
+
writeFileSync(ACTIVE_PATH, "{}");
|
|
445
|
+
if (staleCount > 0) {
|
|
446
|
+
console.log(`[dispatcher] Cleared ${staleCount} stale entries from active.json`);
|
|
447
|
+
}
|
|
448
|
+
} catch (err) {
|
|
449
|
+
console.warn(`[dispatcher] Failed to reset active.json: ${err.message}`);
|
|
450
|
+
}
|
|
451
|
+
if (opts.reconcile) {
|
|
452
|
+
try { return reconcileResumePending(opts); }
|
|
453
|
+
catch (err) { console.warn(`[dispatcher] resume reconcile failed: ${err.message}`); }
|
|
454
|
+
}
|
|
455
|
+
}
|
|
456
|
+
|
|
457
|
+
/**
|
|
458
|
+
* Walk resume-pending markers and re-dispatch / block each as appropriate.
|
|
459
|
+
* Exported so a test can drive it deterministically with an injected
|
|
460
|
+
* `spawnResume`. Returns a summary { resumed, blocked, expired, scanned }.
|
|
461
|
+
*
|
|
462
|
+
* @param {object} [opts] { spawnResume?: ({marker})=>void, now?: ()=>number }
|
|
463
|
+
*/
|
|
464
|
+
export function reconcileResumePending(opts = {}) {
|
|
465
|
+
const now = (typeof opts.now === "function" ? opts.now : Date.now)();
|
|
466
|
+
const stats = { scanned: 0, resumed: 0, blocked: 0, expired: 0 };
|
|
467
|
+
let files = [];
|
|
468
|
+
try {
|
|
469
|
+
if (!existsSync(RESUME_PENDING_DIR)) return stats;
|
|
470
|
+
files = readdirSync(RESUME_PENDING_DIR).filter((f) => f.endsWith(".json") && !f.endsWith(".tmp"));
|
|
471
|
+
} catch { return stats; }
|
|
472
|
+
|
|
473
|
+
for (const file of files) {
|
|
474
|
+
const path = join(RESUME_PENDING_DIR, file);
|
|
475
|
+
stats.scanned++;
|
|
476
|
+
let marker;
|
|
477
|
+
try { marker = JSON.parse(readFileSync(path, "utf-8")); }
|
|
478
|
+
catch { try { unlinkSync(path); } catch { /* */ } continue; }
|
|
479
|
+
|
|
480
|
+
const age = now - (marker.startedAt || 0);
|
|
481
|
+
const attempts = (marker.recoveryAttempts || 0);
|
|
482
|
+
|
|
483
|
+
// 3 strikes → block the item, drop the marker.
|
|
484
|
+
if (attempts >= RESUME_MAX_ATTEMPTS) {
|
|
485
|
+
markItemBlocked(marker, `exceeded ${RESUME_MAX_ATTEMPTS} resume attempts`);
|
|
486
|
+
logSession({ event: "resume_blocked", sessionId: marker.sessionId, item_id: marker.itemId, recovery_attempts: attempts });
|
|
487
|
+
try { unlinkSync(path); } catch { /* */ }
|
|
488
|
+
stats.blocked++;
|
|
489
|
+
continue;
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
// Too stale to resume meaningfully → block (work context has moved on).
|
|
493
|
+
if (age > RESUME_FRESHNESS_MS) {
|
|
494
|
+
markItemBlocked(marker, `resume marker stale (${Math.round(age / 60000)}m old)`);
|
|
495
|
+
logSession({ event: "resume_expired", sessionId: marker.sessionId, item_id: marker.itemId, age_min: Math.round(age / 60000) });
|
|
496
|
+
try { unlinkSync(path); } catch { /* */ }
|
|
497
|
+
stats.expired++;
|
|
498
|
+
continue;
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
// No claude session id to re-spawn against → can't continue; block.
|
|
502
|
+
if (!marker.claudeSessionId) {
|
|
503
|
+
logSession({ event: "resume_no_session_id", sessionId: marker.sessionId, item_id: marker.itemId });
|
|
504
|
+
try { unlinkSync(path); } catch { /* */ }
|
|
505
|
+
stats.expired++;
|
|
506
|
+
continue;
|
|
507
|
+
}
|
|
508
|
+
|
|
509
|
+
// No original prompt persisted → there's nothing to re-spawn with (the
|
|
510
|
+
// resume re-issues `--session-id <id> <prompt>`, NOT `--resume`). A marker
|
|
511
|
+
// without a prompt is either a legacy marker (pre-H1) or one whose write
|
|
512
|
+
// dropped the field; we can't truly resume it, so retire it rather than
|
|
513
|
+
// spawn a no-op that the close handler would mistake for "done". (H1)
|
|
514
|
+
if (typeof marker.prompt !== "string" || !marker.prompt) {
|
|
515
|
+
logSession({ event: "resume_no_prompt", sessionId: marker.sessionId, item_id: marker.itemId });
|
|
516
|
+
try { unlinkSync(path); } catch { /* */ }
|
|
517
|
+
stats.expired++;
|
|
518
|
+
continue;
|
|
519
|
+
}
|
|
520
|
+
|
|
521
|
+
// Bump the attempt count on the marker BEFORE re-dispatch so a crash mid-
|
|
522
|
+
// resume still advances toward the 3-strike cap (no infinite loop).
|
|
523
|
+
marker.recoveryAttempts = attempts + 1;
|
|
524
|
+
try {
|
|
525
|
+
const tmp = path + ".tmp";
|
|
526
|
+
writeFileSync(tmp, JSON.stringify(marker, null, 2));
|
|
527
|
+
renameSync(tmp, path);
|
|
528
|
+
} catch { /* best-effort */ }
|
|
529
|
+
|
|
530
|
+
const spawnResume = typeof opts.spawnResume === "function" ? opts.spawnResume : defaultSpawnResume;
|
|
531
|
+
try {
|
|
532
|
+
spawnResume({ marker });
|
|
533
|
+
logSession({ event: "resume_dispatched", sessionId: marker.sessionId, claudeSessionId: marker.claudeSessionId, item_id: marker.itemId, attempt: marker.recoveryAttempts });
|
|
534
|
+
stats.resumed++;
|
|
535
|
+
} catch (err) {
|
|
536
|
+
console.warn(`[dispatcher] resume spawn failed for ${marker.sessionId}: ${err.message}`);
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
if (stats.scanned > 0) {
|
|
540
|
+
console.log(`[dispatcher] resume reconcile: ${stats.resumed} resumed, ${stats.blocked} blocked, ${stats.expired} expired (of ${stats.scanned})`);
|
|
541
|
+
}
|
|
542
|
+
return stats;
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
/**
|
|
546
|
+
* Real resume spawner. Mirrors responder.mjs's blessed session-continuation
|
|
547
|
+
* pattern (see runClaudeCLI ~L116-122 + generateResponse ~L461-465): a
|
|
548
|
+
* continuation re-spawns `claude --print --session-id <sessionId> <prompt>`
|
|
549
|
+
* with the SAME model/flags the original used — it does NOT use `--resume`.
|
|
550
|
+
*
|
|
551
|
+
* Why NOT `--resume` (the previous bug): `--resume <id>` with no prompt and
|
|
552
|
+
* stdin ignored either (a) exits 0 having done nothing — the close handler
|
|
553
|
+
* then clears the marker and the interrupted work is silently lost — or (b)
|
|
554
|
+
* errors, force-`blocked`ing every in-flight item after a reboot. Pre-minting
|
|
555
|
+
* a stable session id and RE-SPAWNING against it with the original prompt is
|
|
556
|
+
* the actual resume mechanism in `--print` mode (per the b1 flag-verification
|
|
557
|
+
* report cited in responder.mjs). The CLI rehydrates the session-id's history
|
|
558
|
+
* and the fresh prompt continues the work.
|
|
559
|
+
*
|
|
560
|
+
* The resumed session inherits the same permission/env posture as a fresh
|
|
561
|
+
* spawn. Best-effort; failures are logged, the marker stays (next reconcile
|
|
562
|
+
* retries up to the strike cap). On a clean close the resumed session's own
|
|
563
|
+
* close handler clears the marker.
|
|
564
|
+
*/
|
|
565
|
+
function defaultSpawnResume({ marker }) {
|
|
566
|
+
const args = [
|
|
567
|
+
"--print",
|
|
568
|
+
...sessionPermissionArgs({ source: "dispatcher-resume" }),
|
|
569
|
+
...daemonClaudeArgs(),
|
|
570
|
+
"--session-id", marker.claudeSessionId,
|
|
571
|
+
"--model", marker.modelFlag || marker.model || "sonnet",
|
|
572
|
+
marker.prompt,
|
|
573
|
+
];
|
|
574
|
+
const spawnEnv = {
|
|
575
|
+
...process.env,
|
|
576
|
+
PATH: augmentedPath(),
|
|
577
|
+
ANTHROPIC_API_KEY: "",
|
|
578
|
+
ANTHROPIC_AUTH_TOKEN: "",
|
|
579
|
+
};
|
|
580
|
+
const proc = _spawn(CLAUDE_BIN, args, { cwd: AGENT_REPO_DIR, env: spawnEnv, stdio: ["ignore", "pipe", "pipe"] });
|
|
581
|
+
let stderr = "";
|
|
582
|
+
proc.stderr.on("data", (c) => { stderr += c.toString(); });
|
|
583
|
+
proc.on("close", (code) => {
|
|
584
|
+
if (code === 0) {
|
|
585
|
+
// Resume succeeded — the underlying work is done; clear the marker.
|
|
586
|
+
clearResumePending(marker.sessionId);
|
|
587
|
+
logSession({ event: "resume_completed", sessionId: marker.sessionId, item_id: marker.itemId });
|
|
588
|
+
} else {
|
|
589
|
+
// A 429 during resume must open the breaker so subsequent spawns gate.
|
|
590
|
+
if (rateGuard.classifyStderr(stderr)) {
|
|
591
|
+
try { rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
592
|
+
}
|
|
593
|
+
logSession({ event: "resume_exit_nonzero", sessionId: marker.sessionId, item_id: marker.itemId, exit_code: code });
|
|
594
|
+
}
|
|
595
|
+
});
|
|
596
|
+
proc.on("error", (err) => {
|
|
597
|
+
logSession({ event: "resume_spawn_error", sessionId: marker.sessionId, error: err.code || err.message });
|
|
598
|
+
});
|
|
599
|
+
}
|
|
600
|
+
|
|
601
|
+
function writeActiveSession(sessionId, entry) {
|
|
602
|
+
try {
|
|
603
|
+
let active = {};
|
|
604
|
+
try { active = JSON.parse(readFileSync(ACTIVE_PATH, "utf-8")); } catch {}
|
|
605
|
+
active[sessionId] = {
|
|
606
|
+
sender: entry.item?.sender || null,
|
|
607
|
+
channel: entry.item?.channel || entry.item?.channel_id || null,
|
|
608
|
+
summary: entry.classResult?.summary || "unknown",
|
|
609
|
+
model: entry.classResult?.model || "sonnet",
|
|
610
|
+
startTime: Date.now(),
|
|
611
|
+
source: entry.source,
|
|
612
|
+
};
|
|
613
|
+
const tmpPath = ACTIVE_PATH + ".tmp";
|
|
614
|
+
writeFileSync(tmpPath, JSON.stringify(active, null, 2));
|
|
615
|
+
renameSync(tmpPath, ACTIVE_PATH);
|
|
616
|
+
} catch (err) {
|
|
617
|
+
console.warn(`[dispatcher] Failed to write active.json: ${err.message}`);
|
|
618
|
+
}
|
|
619
|
+
}
|
|
620
|
+
|
|
621
|
+
function removeActiveSession(sessionId) {
|
|
622
|
+
try {
|
|
623
|
+
let active = {};
|
|
624
|
+
try { active = JSON.parse(readFileSync(ACTIVE_PATH, "utf-8")); } catch {}
|
|
625
|
+
delete active[sessionId];
|
|
626
|
+
const tmpPath = ACTIVE_PATH + ".tmp";
|
|
627
|
+
writeFileSync(tmpPath, JSON.stringify(active, null, 2));
|
|
628
|
+
renameSync(tmpPath, ACTIVE_PATH);
|
|
629
|
+
} catch (err) {
|
|
630
|
+
console.warn(`[dispatcher] Failed to update active.json: ${err.message}`);
|
|
631
|
+
}
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
/**
|
|
635
|
+
* Count active sessions by source type.
|
|
636
|
+
*/
|
|
637
|
+
function countBySource(source) {
|
|
638
|
+
let count = 0;
|
|
639
|
+
for (const [, s] of activeSessions) {
|
|
640
|
+
if (s.source === source) count++;
|
|
641
|
+
}
|
|
642
|
+
return count;
|
|
643
|
+
}
|
|
644
|
+
|
|
645
|
+
/**
|
|
646
|
+
* Evict the lowest-priority, longest-running backlog session to make room
|
|
647
|
+
* for a high-priority inbox item. Returns true if a session was evicted.
|
|
648
|
+
*/
|
|
649
|
+
function evictForPriority() {
|
|
650
|
+
const priorityRank = { low: 0, normal: 1, high: 2, critical: 3 };
|
|
651
|
+
let worst = null;
|
|
652
|
+
let worstId = null;
|
|
653
|
+
|
|
654
|
+
for (const [id, s] of activeSessions) {
|
|
655
|
+
if (s.source !== "backlog") continue;
|
|
656
|
+
const rank = priorityRank[s.classResult.priority] || 1;
|
|
657
|
+
if (!worst) {
|
|
658
|
+
worst = s; worstId = id;
|
|
659
|
+
continue;
|
|
660
|
+
}
|
|
661
|
+
const worstRank = priorityRank[worst.classResult.priority] || 1;
|
|
662
|
+
// Prefer to evict lower priority; break ties by oldest session
|
|
663
|
+
if (rank < worstRank || (rank === worstRank && s.startTime < worst.startTime)) {
|
|
664
|
+
worst = s; worstId = id;
|
|
665
|
+
}
|
|
666
|
+
}
|
|
667
|
+
|
|
668
|
+
if (worst) {
|
|
669
|
+
console.log(`[dispatcher] PREEMPT: Evicting backlog session ${worstId} (${worst.classResult.summary}) for priority inbox item`);
|
|
670
|
+
logSession({ event: "evicted", sessionId: worstId, reason: "priority_preemption", summary: worst.classResult.summary });
|
|
671
|
+
// M2: clear the resume-pending marker BEFORE the SIGTERM. Eviction is a
|
|
672
|
+
// deliberate preemption, not a crash — the item stays on disk and the
|
|
673
|
+
// backlog sweep re-dispatches it normally (acquiring the item-claim).
|
|
674
|
+
// If we left the marker, a reboot's reconcileResumePending would re-spawn
|
|
675
|
+
// it WITHOUT the claim while sweepBacklog independently claim+dispatched
|
|
676
|
+
// the same item → two concurrent runs. The non-zero SIGTERM close would
|
|
677
|
+
// otherwise keep the marker, so we must retire it here.
|
|
678
|
+
clearResumePending(worstId);
|
|
679
|
+
worst.process.kill("SIGTERM");
|
|
680
|
+
return true;
|
|
681
|
+
}
|
|
682
|
+
return false;
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
/**
|
|
686
|
+
* Generate a stable key for a backlog item to track in-flight status.
|
|
687
|
+
*/
|
|
688
|
+
function backlogKey(item) {
|
|
689
|
+
return item.id || item.title || item.summary || JSON.stringify(item).substring(0, 100);
|
|
690
|
+
}
|
|
691
|
+
|
|
692
|
+
/**
|
|
693
|
+
* Check if a backlog item already has an active session or has exceeded retry limit.
|
|
694
|
+
* Returns { allowed: boolean, reason?: string }
|
|
695
|
+
*/
|
|
696
|
+
export function canDispatchBacklog(item) {
|
|
697
|
+
const key = backlogKey(item);
|
|
698
|
+
if (activeBacklogKeys.has(key)) {
|
|
699
|
+
return { allowed: false, reason: "session_already_active" };
|
|
700
|
+
}
|
|
701
|
+
const retries = backlogRetryCount.get(key) || 0;
|
|
702
|
+
if (retries >= MAX_BACKLOG_RETRIES) {
|
|
703
|
+
return { allowed: false, reason: "max_retries_exceeded", retries };
|
|
704
|
+
}
|
|
705
|
+
// Post-completion cooldown — prevent every-2-min re-dispatch of items
|
|
706
|
+
// that completed (success or failure) within the recent window.
|
|
707
|
+
const cooldownUntil = backlogCooldownUntil.get(key) || 0;
|
|
708
|
+
if (cooldownUntil > Date.now()) {
|
|
709
|
+
const remaining_min = Math.ceil((cooldownUntil - Date.now()) / 60000);
|
|
710
|
+
return { allowed: false, reason: "post_completion_cooldown", remaining_min };
|
|
711
|
+
}
|
|
712
|
+
return { allowed: true };
|
|
713
|
+
}
|
|
714
|
+
|
|
715
|
+
/**
|
|
716
|
+
* Dispatch an item for processing.
|
|
717
|
+
*
|
|
718
|
+
* @param {string} prompt
|
|
719
|
+
* @param {object} item
|
|
720
|
+
* @param {object} classResult
|
|
721
|
+
* @param {string} source - "inbox" or "backlog"
|
|
722
|
+
* @param {object} [opts]
|
|
723
|
+
* @param {(result:{ok:boolean, sessionId:string|null, item:object, code:number|null, stdout:string})=>void} [opts.onClose]
|
|
724
|
+
* Invoked EXACTLY ONCE when the spawned session reaches a terminal state
|
|
725
|
+
* (clean close → ok:true; non-zero close OR spawn error → ok:false). Lets the
|
|
726
|
+
* caller defer finalizing the inbox item (markProcessed) to spawn-close success
|
|
727
|
+
* so a crash between dispatch and close leaves the item re-deliverable rather
|
|
728
|
+
* than prematurely `.processed` (F1/H2 message-loss window). The callback
|
|
729
|
+
* survives queueing (it's carried on the entry, so a QUEUE/DEFER'd item still
|
|
730
|
+
* fires onClose once it is later drained and spawned). It does NOT fire on the
|
|
731
|
+
* pre-spawn drop paths (backlog dedup / claim-denied / governor-DEFER backlog),
|
|
732
|
+
* which leave the durable on-disk item untouched for the next sweep anyway.
|
|
733
|
+
*/
|
|
734
|
+
export function dispatch(prompt, item, classResult, source = "inbox", opts = {}) {
|
|
735
|
+
const entry = { prompt, item, classResult, source, onClose: typeof opts.onClose === "function" ? opts.onClose : null };
|
|
736
|
+
const priority = classResult.priority;
|
|
737
|
+
const isPriorityInbox = source === "inbox" && (priority === "critical" || priority === "high");
|
|
738
|
+
|
|
739
|
+
// Backlog items: check dedup and retry limits
|
|
740
|
+
if (source === "backlog") {
|
|
741
|
+
const check = canDispatchBacklog(item);
|
|
742
|
+
if (!check.allowed) {
|
|
743
|
+
logSession({ event: "skipped", reason: check.reason, summary: classResult.summary, retries: check.retries });
|
|
744
|
+
return;
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
|
|
748
|
+
// ── WS4 governance gate ──────────────────────────────────────────────────
|
|
749
|
+
// Consult the shared 429 breaker + the resource governor BEFORE claiming a
|
|
750
|
+
// backlog item, so a DEFER leaves the item untouched on disk (the 10-min
|
|
751
|
+
// sweep retries it — no work lost, no retry budget burned). For inbox, a
|
|
752
|
+
// non-ADMIT routes to the in-memory queue instead of spawning (the user's
|
|
753
|
+
// message is never dropped). Priority inbox can still preempt a backlog
|
|
754
|
+
// session below. `forceQueue` carries the decision to the spawn branch.
|
|
755
|
+
let forceQueue = false;
|
|
756
|
+
let queueReason = null;
|
|
757
|
+
let queueRetryAt = null; // M1: when the breaker says "retry after T", arm a re-drain then.
|
|
758
|
+
{
|
|
759
|
+
const rb = rateBlocked();
|
|
760
|
+
if (rb) {
|
|
761
|
+
if (source === "backlog") {
|
|
762
|
+
logSession({ event: "deferred", reason: "rate_limited", retry_at: rb.retryAt, summary: classResult.summary });
|
|
763
|
+
return; // leave on disk; sweep retries after the breaker closes
|
|
764
|
+
}
|
|
765
|
+
forceQueue = true; queueReason = "rate_limited"; queueRetryAt = rb.retryAt || null;
|
|
766
|
+
}
|
|
767
|
+
if (!forceQueue || source !== "inbox") {
|
|
768
|
+
const adm = admitFor(source, priority);
|
|
769
|
+
if (adm.decision !== "ADMIT") {
|
|
770
|
+
if (source === "backlog") {
|
|
771
|
+
// QUEUE and DEFER both mean "not now" for backlog — the item stays in
|
|
772
|
+
// its queue file (re-derivable, durable) and the sweep retries. We do
|
|
773
|
+
// NOT claim it and we do NOT burn retry budget.
|
|
774
|
+
logSession({ event: adm.decision === "DEFER" ? "deferred" : "queued_skip", reason: adm.reason, decision: adm.decision, source, summary: classResult.summary });
|
|
775
|
+
return;
|
|
776
|
+
}
|
|
777
|
+
// inbox: never drop — queue it (DEFER(inbox) → queue anyway).
|
|
778
|
+
forceQueue = true; queueReason = queueReason || adm.reason;
|
|
779
|
+
}
|
|
780
|
+
}
|
|
781
|
+
}
|
|
782
|
+
|
|
783
|
+
// Item-claim acquisition: file-based claim visible across daemon restarts
|
|
784
|
+
// and concurrent launchd triggers. Complements the in-memory activeBacklogKeys.
|
|
785
|
+
// (ib-20260407-001b: concurrent session coordination)
|
|
786
|
+
if (source === "backlog" && item.id) {
|
|
787
|
+
const sessionId = `s-${Date.now()}-${sessionCounter + 1}`;
|
|
788
|
+
const claim = claimItem(item.id, {
|
|
789
|
+
session_id: sessionId,
|
|
790
|
+
agent_description: classResult.summary || item.title || "",
|
|
791
|
+
ttl_minutes: classResult.model === "opus" ? 120 : 30,
|
|
792
|
+
source: "backlog",
|
|
793
|
+
queue_file: item.source_file || "",
|
|
794
|
+
pid: process.pid, // daemon PID; child PID not yet known
|
|
795
|
+
});
|
|
796
|
+
if (!claim.claimed) {
|
|
797
|
+
console.log(`[dispatcher] Item claim denied for ${item.id}: ${claim.reason} (holder: ${claim.holder || "unknown"})`);
|
|
798
|
+
logSession({ event: "skipped", reason: `item_claim_denied: ${claim.reason}`, summary: classResult.summary, holder: claim.holder });
|
|
799
|
+
return;
|
|
800
|
+
}
|
|
801
|
+
}
|
|
802
|
+
|
|
803
|
+
// Backlog items respect the reserved slot cap
|
|
804
|
+
if (source === "backlog") {
|
|
805
|
+
const backlogCount = countBySource("backlog");
|
|
806
|
+
const backlogCap = MAX_CONCURRENT - RESERVED_INBOX_SLOTS;
|
|
807
|
+
if (backlogCount >= backlogCap) {
|
|
808
|
+
// No room for more backlog — silently skip (don't queue indefinitely)
|
|
809
|
+
return;
|
|
810
|
+
}
|
|
811
|
+
}
|
|
812
|
+
|
|
813
|
+
// forceQueue (WS4): the governor/breaker said "not now" for this inbox item,
|
|
814
|
+
// OR all slots are full. Either way we don't spawn directly — but a priority
|
|
815
|
+
// inbox item may still preempt a running backlog session.
|
|
816
|
+
if (!forceQueue && activeSessions.size < MAX_CONCURRENT) {
|
|
817
|
+
spawnSession(entry);
|
|
818
|
+
} else if (isPriorityInbox) {
|
|
819
|
+
// Priority inbox item but no slots (or governor-gated) — try to evict a
|
|
820
|
+
// backlog session. (Eviction is only useful under capacity pressure; under
|
|
821
|
+
// a rate-limit/budget gate there may be nothing to evict, in which case it
|
|
822
|
+
// simply queues — still never dropped.)
|
|
823
|
+
if (evictForPriority()) {
|
|
824
|
+
// Session evicted; it will call drainQueue on close. Queue this item at front.
|
|
825
|
+
priorityQueue.unshift(entry);
|
|
826
|
+
logSession({ event: "queued_after_eviction", priority: classResult.priority, summary: classResult.summary, reason: queueReason });
|
|
827
|
+
} else {
|
|
828
|
+
// No backlog sessions to evict — queue normally
|
|
829
|
+
priorityQueue.push(entry);
|
|
830
|
+
logSession({ event: "queued", priority: classResult.priority, summary: classResult.summary, queue_position: priorityQueue.length, reason: queueReason });
|
|
831
|
+
}
|
|
832
|
+
} else if (classResult.priority === "critical" || classResult.priority === "high") {
|
|
833
|
+
priorityQueue.push(entry);
|
|
834
|
+
logSession({ event: "queued", priority: classResult.priority, summary: classResult.summary, queue_position: priorityQueue.length, reason: queueReason });
|
|
835
|
+
} else {
|
|
836
|
+
normalQueue.push(entry);
|
|
837
|
+
logSession({ event: "queued", priority: classResult.priority, summary: classResult.summary, queue_position: normalQueue.length, reason: queueReason });
|
|
838
|
+
}
|
|
839
|
+
|
|
840
|
+
// M1: a governor/budget/rate DEFER can now force-queue an inbox item while
|
|
841
|
+
// NO session is running (pre-WS4 an item was only ever queued behind a
|
|
842
|
+
// running session, whose close drains). With zero active sessions nothing
|
|
843
|
+
// wakes the queue until the next external dispatch — so the item could stall
|
|
844
|
+
// indefinitely. Arm a single debounced re-drain timer to self-heal. The
|
|
845
|
+
// 60s inbox re-scan can't double-dispatch this: the item-claim / sent-message
|
|
846
|
+
// dedupe already guards that, and drainQueue itself re-checks admission.
|
|
847
|
+
if (activeSessions.size === 0 && (priorityQueue.length > 0 || normalQueue.length > 0)) {
|
|
848
|
+
armReDrain(queueRetryAt);
|
|
849
|
+
}
|
|
850
|
+
}
|
|
851
|
+
|
|
852
|
+
// ── M1: re-drain timer (single, debounced) ──────────────────────────────────
|
|
853
|
+
// Only ONE timer is ever pending. Arming again while one is live is a no-op
|
|
854
|
+
// (we never stack timers). Cleared whenever a real drain runs.
|
|
855
|
+
const RE_DRAIN_MIN_MS = 15_000; // floor: don't busy-poll the governor
|
|
856
|
+
const RE_DRAIN_MAX_MS = 30_000; // ceiling: bound worst-case stall to 30s
|
|
857
|
+
let reDrainTimer = null;
|
|
858
|
+
|
|
859
|
+
/**
|
|
860
|
+
* Arm a one-shot re-drain. `retryAt` (epoch ms, optional) is the breaker's
|
|
861
|
+
* "safe to retry after" hint; we clamp the delay into [15s, 30s] so we neither
|
|
862
|
+
* hammer the governor nor leave an item parked too long. Debounced: a second
|
|
863
|
+
* call while a timer is pending does nothing.
|
|
864
|
+
*/
|
|
865
|
+
function armReDrain(retryAt) {
|
|
866
|
+
if (reDrainTimer) return; // already armed — debounce
|
|
867
|
+
let delay = RE_DRAIN_MAX_MS;
|
|
868
|
+
if (typeof retryAt === "number" && retryAt > 0) {
|
|
869
|
+
delay = Math.max(RE_DRAIN_MIN_MS, Math.min(RE_DRAIN_MAX_MS, retryAt - Date.now()));
|
|
870
|
+
}
|
|
871
|
+
reDrainTimer = setTimeout(() => {
|
|
872
|
+
reDrainTimer = null;
|
|
873
|
+
try { drainQueue(); } catch (err) { console.warn(`[dispatcher] re-drain failed: ${err.message}`); }
|
|
874
|
+
}, delay);
|
|
875
|
+
// Don't let this timer pin the event loop alive on its own (mirrors the
|
|
876
|
+
// socket-mode supervisor's unref pattern).
|
|
877
|
+
if (reDrainTimer && typeof reDrainTimer.unref === "function") reDrainTimer.unref();
|
|
878
|
+
}
|
|
879
|
+
|
|
880
|
+
/** Test seam: report whether a re-drain timer is currently armed. */
|
|
881
|
+
export function _hasReDrainArmed() {
|
|
882
|
+
return reDrainTimer !== null;
|
|
883
|
+
}
|
|
884
|
+
|
|
885
|
+
/**
|
|
886
|
+
* Current budget band (0|75|90|100) from budget-guard's daily spend vs the
|
|
887
|
+
* config/recovery.yaml cap, mapped onto the economics ladder. Fed into
|
|
888
|
+
* resolveChain so a chain degrades (frontier→default→fast→cheap) as the day's
|
|
889
|
+
* spend climbs (SPEC §6.4). Best-effort: any read failure → band 0 (policy as
|
|
890
|
+
* written) so a budget-read bug never changes routing under us.
|
|
891
|
+
*/
|
|
892
|
+
function currentBudgetBand() {
|
|
893
|
+
try {
|
|
894
|
+
const st = budgetGuard.dailyStatus({ agentRoot: AGENT_REPO_DIR });
|
|
895
|
+
return budgetLadder(st.spentUSD, st.capUSD).band;
|
|
896
|
+
} catch {
|
|
897
|
+
return 0;
|
|
898
|
+
}
|
|
899
|
+
}
|
|
900
|
+
|
|
901
|
+
/**
|
|
902
|
+
* Resolve the spawn target through the v2 router (resolveChain) when the agent
|
|
903
|
+
* has opted into a `schema_version: 2` config, else fall back to the v1
|
|
904
|
+
* resolveBackend path (byte-compatible). Returns a normalised shape the spawn
|
|
905
|
+
* path consumes regardless of which router produced it:
|
|
906
|
+
*
|
|
907
|
+
* { modelFlag, envForSpawn, decisionId, backend, model, transport,
|
|
908
|
+
* maxTurns, effort, agentsJson, explain, decision }
|
|
909
|
+
*
|
|
910
|
+
* NEVER throws — any router error degrades to the v1 path (or null → stock CLI).
|
|
911
|
+
* The kill switch (MAESTRO_ROUTER_FORCE_ANTHROPIC) is honoured INSIDE both
|
|
912
|
+
* resolveChain and resolveBackend, so it short-circuits to stock Anthropic
|
|
913
|
+
* either way.
|
|
914
|
+
*
|
|
915
|
+
* @param {object} routingConfig loaded config (v1 or v2) or null
|
|
916
|
+
* @param {object} routingRequest v1 AgentRequest (from requestFromClassifierResult)
|
|
917
|
+
* @param {object} classResult
|
|
918
|
+
* @param {string} source
|
|
919
|
+
* @param {string} fallbackModel the coarse sonnet/opus class for the no-config path
|
|
920
|
+
*/
|
|
921
|
+
function resolveSpawnTarget(routingConfig, routingRequest, classResult, source, fallbackModel) {
|
|
922
|
+
// No config at all → stock Claude-CLI-on-Max behaviour (unchanged).
|
|
923
|
+
if (!routingConfig) {
|
|
924
|
+
return { modelFlag: fallbackModel, envForSpawn: {}, decisionId: null, backend: null, model: null, transport: "anthropic-cli", maxTurns: null, effort: null, agentsJson: null, explain: null, decision: null };
|
|
925
|
+
}
|
|
926
|
+
|
|
927
|
+
// v2 path — resolveChain produces a full RouteDecision.
|
|
928
|
+
if (routingConfig.schema_version === 2) {
|
|
929
|
+
try {
|
|
930
|
+
const taskClass = "session.responder";
|
|
931
|
+
const req = {
|
|
932
|
+
...routingRequest,
|
|
933
|
+
task_class: taskClass,
|
|
934
|
+
// data_class is DERIVED from the channel/source, never model-chosen
|
|
935
|
+
// (SPEC §7.6): inbox traffic is sensitive; backlog is internal work.
|
|
936
|
+
data_class: source === "inbox" ? "sensitive" : "internal",
|
|
937
|
+
budget_band: currentBudgetBand(),
|
|
938
|
+
harness_hint: "session",
|
|
939
|
+
};
|
|
940
|
+
const decision = resolveChain(req, { config: routingConfig, agentRoot: AGENT_REPO_DIR });
|
|
941
|
+
if (decision && decision.chosen) {
|
|
942
|
+
// Spawn knobs default per task_class; a rule may have overridden them on
|
|
943
|
+
// the decision (decision.spawnArgs wins where present).
|
|
944
|
+
const knobs = spawnKnobsFor(taskClass);
|
|
945
|
+
const sa = decision.spawnArgs || {};
|
|
946
|
+
return {
|
|
947
|
+
modelFlag: sa.modelFlag || decision.chosen.model || fallbackModel,
|
|
948
|
+
envForSpawn: decision.envForSpawn || {},
|
|
949
|
+
decisionId: decision.decision_id || null,
|
|
950
|
+
backend: decision.chosen.provider || null,
|
|
951
|
+
model: decision.chosen.model || null,
|
|
952
|
+
transport: decision.chosen.transport || null,
|
|
953
|
+
maxTurns: sa.maxTurns != null ? sa.maxTurns : knobs.maxTurns,
|
|
954
|
+
effort: sa.effort != null ? sa.effort : knobs.effort,
|
|
955
|
+
agentsJson: sa.agentsJson != null ? sa.agentsJson : knobs.agentsJson,
|
|
956
|
+
explain: decision.explain || null,
|
|
957
|
+
decision,
|
|
958
|
+
};
|
|
959
|
+
}
|
|
960
|
+
} catch (err) {
|
|
961
|
+
console.warn(`[dispatcher] resolveChain failed, falling back to v1 resolver: ${err.message}`);
|
|
962
|
+
}
|
|
963
|
+
// resolveChain degraded — fall through to v1 below.
|
|
964
|
+
}
|
|
965
|
+
|
|
966
|
+
// v1 path — resolveBackend → modelFlagFor (preserved verbatim).
|
|
967
|
+
const resolved = resolveBackend(routingRequest, { config: routingConfig });
|
|
968
|
+
if (!resolved) {
|
|
969
|
+
return { modelFlag: fallbackModel, envForSpawn: {}, decisionId: null, backend: null, model: null, transport: "anthropic-cli", maxTurns: null, effort: null, agentsJson: null, explain: null, decision: null };
|
|
970
|
+
}
|
|
971
|
+
return {
|
|
972
|
+
modelFlag: modelFlagFor(resolved, routingRequest),
|
|
973
|
+
envForSpawn: resolved.envForSpawn || {},
|
|
974
|
+
decisionId: null,
|
|
975
|
+
backend: resolved.name || null,
|
|
976
|
+
model: resolved.model || null,
|
|
977
|
+
transport: resolved.transport || null,
|
|
978
|
+
maxTurns: null,
|
|
979
|
+
effort: null,
|
|
980
|
+
agentsJson: null,
|
|
981
|
+
explain: null,
|
|
982
|
+
decision: null,
|
|
983
|
+
_v1Resolved: resolved,
|
|
984
|
+
};
|
|
985
|
+
}
|
|
986
|
+
|
|
987
|
+
function spawnSession(entry) {
|
|
988
|
+
const { prompt, item, classResult, source } = entry;
|
|
989
|
+
const sessionId = `s-${Date.now()}-${++sessionCounter}`;
|
|
990
|
+
|
|
991
|
+
// F1/H2: notify the dispatch caller exactly once when this session reaches a
|
|
992
|
+
// terminal state, so it can finalize the inbox item on success and re-deliver
|
|
993
|
+
// on crash. Single-fire guard — both proc.on("close") (clean + non-zero) and
|
|
994
|
+
// proc.on("error") (spawn failure) route here, and the close handler can run
|
|
995
|
+
// after error on the ENOENT/ETIMEDOUT path; we must invoke onClose only once.
|
|
996
|
+
//
|
|
997
|
+
// The callback also receives the run's raw `stdout` (already buffered here for
|
|
998
|
+
// the cost ledger) so the caller can apply POST-SESSION OUTCOME DISCIPLINE —
|
|
999
|
+
// reading the turn's commitments off its final text and landing them through
|
|
1000
|
+
// the org protocol (scripts/daemon/session-outcomes.mjs). Purely additive: the
|
|
1001
|
+
// field is new, existing callers destructure what they already used.
|
|
1002
|
+
let onCloseFired = false;
|
|
1003
|
+
const fireOnClose = (ok, code, closeStdout = "") => {
|
|
1004
|
+
if (onCloseFired || !entry.onClose) return;
|
|
1005
|
+
onCloseFired = true;
|
|
1006
|
+
try { entry.onClose({ ok, sessionId, item, code, stdout: closeStdout }); }
|
|
1007
|
+
catch (err) { console.warn(`[dispatcher] onClose callback threw for ${sessionId}: ${err.message}`); }
|
|
1008
|
+
};
|
|
1009
|
+
const model = classResult.model || "sonnet";
|
|
1010
|
+
const timeout = model === "opus"
|
|
1011
|
+
? (source === "backlog" ? OPUS_BACKLOG_TIMEOUT : OPUS_INBOX_TIMEOUT)
|
|
1012
|
+
: (source === "backlog" ? SONNET_BACKLOG_TIMEOUT : SONNET_INBOX_TIMEOUT);
|
|
1013
|
+
|
|
1014
|
+
// Track backlog items to prevent retry storms
|
|
1015
|
+
if (source === "backlog") {
|
|
1016
|
+
const key = backlogKey(item);
|
|
1017
|
+
activeBacklogKeys.add(key);
|
|
1018
|
+
backlogRetryCount.set(key, (backlogRetryCount.get(key) || 0) + 1);
|
|
1019
|
+
}
|
|
1020
|
+
|
|
1021
|
+
// Resolve the spawn target via the model router. v2 configs go through
|
|
1022
|
+
// resolveChain (full RouteDecision: model + retarget env + spawn knobs +
|
|
1023
|
+
// decision_id + estimated cost + failover chain); v1 configs keep the legacy
|
|
1024
|
+
// resolveBackend path; no config preserves the historical Claude-CLI-on-Max
|
|
1025
|
+
// default exactly. NEVER throws (resolveSpawnTarget degrades internally).
|
|
1026
|
+
const routingConfig = getRoutingConfig();
|
|
1027
|
+
const routingRequest = requestFromClassifierResult(classResult, { source, role: "responder" });
|
|
1028
|
+
const target = resolveSpawnTarget(routingConfig, routingRequest, classResult, source, model);
|
|
1029
|
+
const effectiveModelFlag = target.modelFlag || model;
|
|
1030
|
+
|
|
1031
|
+
// WS4: pre-mint a stable Claude session id so a crash/reboot mid-flight can
|
|
1032
|
+
// be resumed deterministically with `claude --print --resume <id>`. (Same
|
|
1033
|
+
// mechanism the responder already uses; safe in --print text mode.)
|
|
1034
|
+
const claudeSessionId = randomUUID();
|
|
1035
|
+
|
|
1036
|
+
// The v2 router supplies spawn knobs (SPEC §4.4 / §6.5): --max-turns bounds
|
|
1037
|
+
// the session, --effort tunes reasoning where supported, --agents attaches the
|
|
1038
|
+
// cheap-subagent (Haiku-Explore) fan-out map — the sanctioned intra-session
|
|
1039
|
+
// cost lever that keeps the main loop's cache intact. v1 / no-config spawns
|
|
1040
|
+
// carry none of these (knobs are null), so this is purely additive.
|
|
1041
|
+
const knobArgs = [];
|
|
1042
|
+
if (target.maxTurns != null && Number.isFinite(Number(target.maxTurns))) {
|
|
1043
|
+
knobArgs.push("--max-turns", String(target.maxTurns));
|
|
1044
|
+
}
|
|
1045
|
+
if (target.effort) knobArgs.push("--effort", String(target.effort));
|
|
1046
|
+
if (target.agentsJson) {
|
|
1047
|
+
knobArgs.push("--agents", typeof target.agentsJson === "string" ? target.agentsJson : JSON.stringify(target.agentsJson));
|
|
1048
|
+
}
|
|
1049
|
+
|
|
1050
|
+
const args = [
|
|
1051
|
+
"--print",
|
|
1052
|
+
// --output-format json so this run's stdout carries the REAL token usage we
|
|
1053
|
+
// record to the cost ledger (recovery C1). The dispatcher's stdout is only
|
|
1054
|
+
// tee'd to a per-session log file — it's never streamed to the user (the
|
|
1055
|
+
// session sends its own user-facing messages via the Slack/Gmail APIs), so
|
|
1056
|
+
// switching the format does not affect any reply.
|
|
1057
|
+
"--output-format", "json",
|
|
1058
|
+
...sessionPermissionArgs({ source: "dispatcher", priority: classResult?.priority }),
|
|
1059
|
+
...daemonClaudeArgs(),
|
|
1060
|
+
...knobArgs,
|
|
1061
|
+
"--session-id", claudeSessionId,
|
|
1062
|
+
"--model", effectiveModelFlag,
|
|
1063
|
+
prompt,
|
|
1064
|
+
];
|
|
1065
|
+
|
|
1066
|
+
// Build the spawn env.
|
|
1067
|
+
// Default (no router): strip ANTHROPIC_API_KEY/ANTHROPIC_AUTH_TOKEN so
|
|
1068
|
+
// `claude` falls through to the keychain OAuth (Max subscription) per
|
|
1069
|
+
// CEO directive 2026-04-27.
|
|
1070
|
+
// Router active: merge the resolved target's envForSpawn — this points the
|
|
1071
|
+
// CLI at the chosen backend (Moonshot, OpenRouter, NIM, etc.) and sets
|
|
1072
|
+
// ANTHROPIC_API_KEY="" explicitly so Claude Code doesn't fall back to
|
|
1073
|
+
// OAuth against api.anthropic.com.
|
|
1074
|
+
//
|
|
1075
|
+
// For a v2 retarget (envForSpawn carries ANTHROPIC_BASE_URL), we build the
|
|
1076
|
+
// child env through the execution layer's buildChildEnv so the §7.3 allowlist
|
|
1077
|
+
// scrub applies (no foreign *_API_KEY / *_AUTH_TOKEN leaks into a third-party
|
|
1078
|
+
// session). buildChildEnv already injects ANTHROPIC_API_KEY="" for a retarget.
|
|
1079
|
+
// For the common Anthropic-session case (envForSpawn === {}) and the v1 path
|
|
1080
|
+
// we keep the prior explicit empty-key env verbatim (byte-compatible).
|
|
1081
|
+
const retargeting = !!(target.envForSpawn && target.envForSpawn.ANTHROPIC_BASE_URL);
|
|
1082
|
+
const spawnEnv = retargeting
|
|
1083
|
+
? { ...buildChildEnv({ ...process.env, PATH: augmentedPath() }, target.envForSpawn), PATH: augmentedPath() }
|
|
1084
|
+
: {
|
|
1085
|
+
...process.env,
|
|
1086
|
+
PATH: augmentedPath(),
|
|
1087
|
+
ANTHROPIC_API_KEY: "",
|
|
1088
|
+
ANTHROPIC_AUTH_TOKEN: "",
|
|
1089
|
+
...(target.envForSpawn || {}),
|
|
1090
|
+
};
|
|
1091
|
+
|
|
1092
|
+
const proc = _spawn(CLAUDE_BIN, args, {
|
|
1093
|
+
cwd: AGENT_REPO_DIR,
|
|
1094
|
+
env: spawnEnv,
|
|
1095
|
+
stdio: ["ignore", "pipe", "pipe"],
|
|
1096
|
+
});
|
|
1097
|
+
|
|
1098
|
+
// Log the routing outcome whenever a backend was actually chosen (v1 or v2)
|
|
1099
|
+
// and it isn't the stock Anthropic CLI no-op. The v2 path also carries the
|
|
1100
|
+
// decision_id + the one-line explain so `maestro router why` can join this
|
|
1101
|
+
// session to its routing-audit + ledger rows.
|
|
1102
|
+
if (target.backend && target.transport !== "anthropic-cli") {
|
|
1103
|
+
logSession({
|
|
1104
|
+
event: "routed",
|
|
1105
|
+
sessionId,
|
|
1106
|
+
decision_id: target.decisionId,
|
|
1107
|
+
backend: target.backend,
|
|
1108
|
+
transport: target.transport,
|
|
1109
|
+
model: target.model,
|
|
1110
|
+
tried: target.decision ? target.decision.tried : target._v1Resolved?.tried,
|
|
1111
|
+
fallback_reason: target.decision ? target.decision.audit?.fallback_reason : target._v1Resolved?.fallback_reason,
|
|
1112
|
+
explain: target.explain,
|
|
1113
|
+
});
|
|
1114
|
+
}
|
|
1115
|
+
|
|
1116
|
+
let stdout = "";
|
|
1117
|
+
let stderr = "";
|
|
1118
|
+
const logFile = join(sessionLogDir(), `${sessionId}.log`);
|
|
1119
|
+
|
|
1120
|
+
proc.stdout.on("data", (chunk) => { stdout += chunk.toString(); });
|
|
1121
|
+
proc.stderr.on("data", (chunk) => { stderr += chunk.toString(); });
|
|
1122
|
+
|
|
1123
|
+
const timer = setTimeout(() => {
|
|
1124
|
+
console.error(`[dispatcher] Session ${sessionId} timed out (${timeout / 1000}s), killing`);
|
|
1125
|
+
proc.kill("SIGTERM");
|
|
1126
|
+
const killTimer = setTimeout(() => { if (!proc.killed) proc.kill("SIGKILL"); }, 5000);
|
|
1127
|
+
if (killTimer && typeof killTimer.unref === "function") killTimer.unref();
|
|
1128
|
+
}, timeout);
|
|
1129
|
+
// A kill-watchdog must never be the sole handle pinning the event loop open
|
|
1130
|
+
// (mirrors the re-drain timer above). It is cleared on every close path below;
|
|
1131
|
+
// unref only ensures a leaked/never-closed session (e.g. a test fake-spawn that
|
|
1132
|
+
// never emits "close") can't hold the process — otherwise the test suite would
|
|
1133
|
+
// wait out the full session timeout before exiting.
|
|
1134
|
+
if (timer && typeof timer.unref === "function") timer.unref();
|
|
1135
|
+
|
|
1136
|
+
const startTime = Date.now();
|
|
1137
|
+
activeSessions.set(sessionId, { process: proc, item, classResult, startTime, model, source, claudeSessionId });
|
|
1138
|
+
writeActiveSession(sessionId, entry);
|
|
1139
|
+
|
|
1140
|
+
// WS4: drop the in-flight resume marker. Present-after-crash = mid-flight, so
|
|
1141
|
+
// a startup reconcile can re-dispatch `claude --print --resume`. Cleared on a
|
|
1142
|
+
// clean close below. Inbox sessions are short and user-facing; backlog
|
|
1143
|
+
// sessions are the ones that most benefit from surviving a reboot, but we mark
|
|
1144
|
+
// both uniformly — the freshness window + 3-strike cap bound any churn.
|
|
1145
|
+
writeResumePending(sessionId, { ...entry, recoveryAttempts: entry.recoveryAttempts }, claudeSessionId, effectiveModelFlag, target.decisionId);
|
|
1146
|
+
|
|
1147
|
+
// WS2: show a typing indicator on the originating channel while the session
|
|
1148
|
+
// works (inbox items only — backlog work has no waiting human). Best-effort
|
|
1149
|
+
// via the typing registry; channels without a live adapter no-op.
|
|
1150
|
+
if (source === "inbox") {
|
|
1151
|
+
try { startTyping(item); } catch { /* fail-open */ }
|
|
1152
|
+
}
|
|
1153
|
+
|
|
1154
|
+
logSession({
|
|
1155
|
+
event: "spawned",
|
|
1156
|
+
sessionId,
|
|
1157
|
+
model,
|
|
1158
|
+
source,
|
|
1159
|
+
priority: classResult.priority,
|
|
1160
|
+
summary: classResult.summary,
|
|
1161
|
+
active_count: activeSessions.size,
|
|
1162
|
+
});
|
|
1163
|
+
|
|
1164
|
+
// Observability: the interaction's trace_id rode in on item.trace_id (set by
|
|
1165
|
+
// the daemon at item_received); read it explicitly here because the close/error
|
|
1166
|
+
// handlers below fire on a separate event-loop tick, outside any withTrace
|
|
1167
|
+
// scope. `dispatched` marks the spawn decision; `session_opened` the sub-session
|
|
1168
|
+
// start. Both carry trace_id + the v2 decision_id so the audit can join the
|
|
1169
|
+
// routing decision ↔ ledger row ↔ diagnostic stream.
|
|
1170
|
+
const traceId = (item && typeof item.trace_id === "string") ? item.trace_id : null;
|
|
1171
|
+
emitEvent({
|
|
1172
|
+
type: EVENT_TYPES.DISPATCHED,
|
|
1173
|
+
trace_id: traceId,
|
|
1174
|
+
attrs: { session_id: sessionId, source, model, decision_id: target.decisionId || null, priority: classResult.priority },
|
|
1175
|
+
});
|
|
1176
|
+
emitEvent({
|
|
1177
|
+
type: EVENT_TYPES.SESSION_OPENED,
|
|
1178
|
+
trace_id: traceId,
|
|
1179
|
+
attrs: { session_id: sessionId, source, model, decision_id: target.decisionId || null },
|
|
1180
|
+
});
|
|
1181
|
+
|
|
1182
|
+
console.log(`[dispatcher] Spawned ${sessionId} (${model}) — ${classResult.summary} [${activeSessions.size}/${MAX_CONCURRENT}]`);
|
|
1183
|
+
|
|
1184
|
+
proc.on("close", (code) => {
|
|
1185
|
+
clearTimeout(timer);
|
|
1186
|
+
// WS2: stop the typing heartbeat for this conversation (mirrors the
|
|
1187
|
+
// promoteDeferred close-hook). Best-effort; fail-open.
|
|
1188
|
+
if (source === "inbox") { try { stopTyping(item); } catch { /* */ } }
|
|
1189
|
+
// If proc.on("error") already fired for this session (spawn failure path
|
|
1190
|
+
// — ENOENT, EACCES, ETIMEDOUT), cleanup + metric + lock release already
|
|
1191
|
+
// happened. Skip to avoid double-count and double-release.
|
|
1192
|
+
if (spawnErrorHandled.has(sessionId)) {
|
|
1193
|
+
spawnErrorHandled.delete(sessionId);
|
|
1194
|
+
clearResumePending(sessionId); // error path already terminal — no resume
|
|
1195
|
+
// The error handler already fired onClose(ok:false) AND emitted
|
|
1196
|
+
// session_closed; the single-fire guard makes the onClose a no-op, and we
|
|
1197
|
+
// do NOT re-emit session_closed here to keep it exactly-once per session.
|
|
1198
|
+
fireOnClose(false, code);
|
|
1199
|
+
drainQueue();
|
|
1200
|
+
return;
|
|
1201
|
+
}
|
|
1202
|
+
activeSessions.delete(sessionId);
|
|
1203
|
+
removeActiveSession(sessionId);
|
|
1204
|
+
recordSession(true, code === 0);
|
|
1205
|
+
const duration = ((Date.now() - startTime) / 1000).toFixed(1);
|
|
1206
|
+
|
|
1207
|
+
// WS4: a clean exit retires the resume marker (work finished — nothing to
|
|
1208
|
+
// resume) and closes the shared 429 breaker. A non-zero exit whose stderr
|
|
1209
|
+
// looks like a rate limit opens the breaker so ALL spawn sources back off
|
|
1210
|
+
// together. The marker is deliberately LEFT in place on a non-zero exit so
|
|
1211
|
+
// a subsequent reboot can still resume the work; the cooldown/strike caps
|
|
1212
|
+
// bound re-dispatch.
|
|
1213
|
+
if (code === 0) {
|
|
1214
|
+
clearResumePending(sessionId);
|
|
1215
|
+
try { rateGuard.recordSuccess(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR }); } catch { /* */ }
|
|
1216
|
+
} else if (rateGuard.classifyStderr(stderr)) {
|
|
1217
|
+
try {
|
|
1218
|
+
const rec = rateGuard.recordRateLimit(RATE_PROVIDER, { agentRoot: AGENT_REPO_DIR });
|
|
1219
|
+
logSession({ event: "rate_limit_recorded", sessionId, open_until: rec.openUntil, consecutive: rec.consecutive429 });
|
|
1220
|
+
} catch { /* */ }
|
|
1221
|
+
}
|
|
1222
|
+
|
|
1223
|
+
// Release item lock — MUST use same key order as acquireLock in daemon
|
|
1224
|
+
// (raw_ref first, then id). Previously this used id || raw_ref, creating
|
|
1225
|
+
// a different sanitised filename so the release silently missed the lock.
|
|
1226
|
+
const itemId = item.raw_ref || item.id || item.title;
|
|
1227
|
+
if (itemId) releaseLock(itemId);
|
|
1228
|
+
|
|
1229
|
+
// Release thread/channel lock so new messages in this thread/DM can
|
|
1230
|
+
// be processed. We must release unconditionally when a channel is
|
|
1231
|
+
// known — the previous `if (item.thread_id)` gate skipped DMs (which
|
|
1232
|
+
// always have empty thread_id), so DM-channel locks never cleared
|
|
1233
|
+
// until the 60min TTL fired. acquireThreadLock normalises DM
|
|
1234
|
+
// channels to `dm-channel` internally; releaseThreadLock mirrors
|
|
1235
|
+
// that normalisation, so calling it with an empty thread_id on a
|
|
1236
|
+
// DM channel does the right thing.
|
|
1237
|
+
{
|
|
1238
|
+
const channel = item.channel_id || (item.raw_ref ? (item.raw_ref.match(/slack:([^:]+):/) || [])[1] : null) || item.channel;
|
|
1239
|
+
if (channel) {
|
|
1240
|
+
releaseThreadLock(channel, item.thread_id);
|
|
1241
|
+
// Now that the lock is gone, promote any messages that were
|
|
1242
|
+
// deferred behind it. Latest-wins: a burst of N messages
|
|
1243
|
+
// collapses into ONE re-dispatch carrying the most recent
|
|
1244
|
+
// message (its thread_context already includes the earlier
|
|
1245
|
+
// ones), so the user gets a single coherent reply rather than
|
|
1246
|
+
// N replies serialised over N session-durations.
|
|
1247
|
+
const promo = promoteDeferred(channel, AGENT_REPO_DIR);
|
|
1248
|
+
if (promo.promoted > 0) {
|
|
1249
|
+
logSession({
|
|
1250
|
+
event: "deferred_promoted",
|
|
1251
|
+
channel,
|
|
1252
|
+
promoted: promo.promoted,
|
|
1253
|
+
bundled: promo.bundled,
|
|
1254
|
+
service: promo.service,
|
|
1255
|
+
});
|
|
1256
|
+
}
|
|
1257
|
+
}
|
|
1258
|
+
}
|
|
1259
|
+
|
|
1260
|
+
// Release request claim so the same type of request can be processed again
|
|
1261
|
+
if (classResult && classResult.summary) {
|
|
1262
|
+
releaseRequestClaim({
|
|
1263
|
+
recipient: item.channel_id || item.channel || item.sender || "unknown",
|
|
1264
|
+
subject: classResult.summary || item.subject || "",
|
|
1265
|
+
action_type: classResult.action || "respond",
|
|
1266
|
+
});
|
|
1267
|
+
}
|
|
1268
|
+
|
|
1269
|
+
// Release backlog tracking + item claim
|
|
1270
|
+
if (source === "backlog") {
|
|
1271
|
+
const key = backlogKey(item);
|
|
1272
|
+
activeBacklogKeys.delete(key);
|
|
1273
|
+
const retries = backlogRetryCount.get(key) || 0;
|
|
1274
|
+
|
|
1275
|
+
// Apply post-completion cooldown — different for success vs failure.
|
|
1276
|
+
// This is the fix for the every-2-min re-dispatch loop: once a
|
|
1277
|
+
// session has touched an item, we wait before touching it again.
|
|
1278
|
+
const cooldownMs = code === 0 ? SUCCESS_COOLDOWN_MS : FAILURE_COOLDOWN_MS;
|
|
1279
|
+
backlogCooldownUntil.set(key, Date.now() + cooldownMs);
|
|
1280
|
+
saveCooldowns();
|
|
1281
|
+
logSession({
|
|
1282
|
+
event: "cooldown_set",
|
|
1283
|
+
summary: classResult.summary,
|
|
1284
|
+
exit_code: code,
|
|
1285
|
+
cooldown_minutes: Math.round(cooldownMs / 60000),
|
|
1286
|
+
});
|
|
1287
|
+
|
|
1288
|
+
// Release file-based item claim (ib-20260407-001b)
|
|
1289
|
+
if (item.id) releaseItemClaim(item.id);
|
|
1290
|
+
|
|
1291
|
+
// If session timed out (143=SIGTERM) and hit retry limit, log it
|
|
1292
|
+
if (code === 143 && retries >= MAX_BACKLOG_RETRIES) {
|
|
1293
|
+
console.warn(`[dispatcher] Backlog item "${classResult.summary}" exhausted ${MAX_BACKLOG_RETRIES} retries — will not retry`);
|
|
1294
|
+
logSession({ event: "retries_exhausted", summary: classResult.summary, retries });
|
|
1295
|
+
}
|
|
1296
|
+
}
|
|
1297
|
+
|
|
1298
|
+
// Write session log
|
|
1299
|
+
writeFileSync(logFile, `# Session ${sessionId}\n# Model: ${model}\n# Duration: ${duration}s\n# Exit: ${code}\n\n## STDOUT\n${stdout}\n\n## STDERR\n${stderr}\n`);
|
|
1300
|
+
|
|
1301
|
+
// Record a TRUTHFUL cost-ledger row from this run's real token usage
|
|
1302
|
+
// (recovery C1) — the dispatcher is the user-facing spawn source and was
|
|
1303
|
+
// previously invisible to budget-guard. Parse failures are surfaced, not
|
|
1304
|
+
// backfilled with fake zeros.
|
|
1305
|
+
const costUsage = recordDispatcherCost({
|
|
1306
|
+
stdout,
|
|
1307
|
+
model: effectiveModelFlag,
|
|
1308
|
+
durationMs: Date.now() - startTime,
|
|
1309
|
+
exitCode: code,
|
|
1310
|
+
decisionId: target.decisionId,
|
|
1311
|
+
});
|
|
1312
|
+
if (!costUsage.ok) {
|
|
1313
|
+
logSession({ event: "cost_usage_parse_failed", sessionId, reason: costUsage.reason });
|
|
1314
|
+
}
|
|
1315
|
+
|
|
1316
|
+
logSession({
|
|
1317
|
+
event: "completed",
|
|
1318
|
+
sessionId,
|
|
1319
|
+
model,
|
|
1320
|
+
source,
|
|
1321
|
+
exit_code: code,
|
|
1322
|
+
duration_s: parseFloat(duration),
|
|
1323
|
+
summary: classResult.summary,
|
|
1324
|
+
active_count: activeSessions.size,
|
|
1325
|
+
});
|
|
1326
|
+
|
|
1327
|
+
// Observability: the sub-session ended. Carry the interaction's trace_id +
|
|
1328
|
+
// decision_id so item_received → dispatched → session_opened → session_closed
|
|
1329
|
+
// (→ sent, emitted by the daemon's onClose) all group by one id.
|
|
1330
|
+
emitEvent({
|
|
1331
|
+
type: EVENT_TYPES.SESSION_CLOSED,
|
|
1332
|
+
trace_id: traceId,
|
|
1333
|
+
attrs: { session_id: sessionId, source, exit_code: code, duration_s: parseFloat(duration), decision_id: target.decisionId || null },
|
|
1334
|
+
});
|
|
1335
|
+
|
|
1336
|
+
console.log(`[dispatcher] Session ${sessionId} done (${duration}s, exit ${code}) [${activeSessions.size}/${MAX_CONCURRENT}]`);
|
|
1337
|
+
|
|
1338
|
+
// F1/H2: tell the caller this session reached a terminal state. A clean exit
|
|
1339
|
+
// (code 0) finalizes the inbox item (markProcessed); any non-zero exit leaves
|
|
1340
|
+
// it re-deliverable. Fired AFTER lock/claim release above so the caller's
|
|
1341
|
+
// re-delivery path (which may release the per-item lock) doesn't race them.
|
|
1342
|
+
fireOnClose(code === 0, code, stdout);
|
|
1343
|
+
|
|
1344
|
+
// Dispatch next queued item
|
|
1345
|
+
drainQueue();
|
|
1346
|
+
});
|
|
1347
|
+
|
|
1348
|
+
proc.on("error", (err) => {
|
|
1349
|
+
clearTimeout(timer);
|
|
1350
|
+
// WS2: stop the typing heartbeat (the session never really ran).
|
|
1351
|
+
if (source === "inbox") { try { stopTyping(item); } catch { /* */ } }
|
|
1352
|
+
// Mark so the trailing proc.on("close") doesn't double-process.
|
|
1353
|
+
spawnErrorHandled.add(sessionId);
|
|
1354
|
+
activeSessions.delete(sessionId);
|
|
1355
|
+
removeActiveSession(sessionId);
|
|
1356
|
+
// Spawn never ran — there is nothing to resume; retire the marker so the
|
|
1357
|
+
// next startup reconcile doesn't try to --resume a session that never began.
|
|
1358
|
+
clearResumePending(sessionId);
|
|
1359
|
+
recordSession(true, false);
|
|
1360
|
+
const duration = ((Date.now() - startTime) / 1000).toFixed(1);
|
|
1361
|
+
const errorCode = err.code || "unknown";
|
|
1362
|
+
|
|
1363
|
+
// Release item lock — mirror of close-handler logic.
|
|
1364
|
+
const itemId = item.raw_ref || item.id || item.title;
|
|
1365
|
+
if (itemId) releaseLock(itemId);
|
|
1366
|
+
|
|
1367
|
+
// Release thread/channel lock so new messages can be processed.
|
|
1368
|
+
// Mirror of close-handler logic — unconditional when channel is
|
|
1369
|
+
// known, since releaseThreadLock handles the DM normalization.
|
|
1370
|
+
{
|
|
1371
|
+
const channel = item.channel_id || (item.raw_ref ? (item.raw_ref.match(/slack:([^:]+):/) || [])[1] : null) || item.channel;
|
|
1372
|
+
if (channel) {
|
|
1373
|
+
releaseThreadLock(channel, item.thread_id);
|
|
1374
|
+
const promo = promoteDeferred(channel, AGENT_REPO_DIR);
|
|
1375
|
+
if (promo.promoted > 0) {
|
|
1376
|
+
logSession({
|
|
1377
|
+
event: "deferred_promoted_on_error",
|
|
1378
|
+
channel,
|
|
1379
|
+
promoted: promo.promoted,
|
|
1380
|
+
bundled: promo.bundled,
|
|
1381
|
+
service: promo.service,
|
|
1382
|
+
});
|
|
1383
|
+
}
|
|
1384
|
+
}
|
|
1385
|
+
}
|
|
1386
|
+
|
|
1387
|
+
// Release request claim + emit explicit claim_released event so
|
|
1388
|
+
// reconciliation audits (cycle 124 Agent C pattern) can distinguish
|
|
1389
|
+
// genuine in-flight sessions from silent-exit ETIMEDOUT failures
|
|
1390
|
+
// without cross-referencing logs/daemon/responses.jsonl.
|
|
1391
|
+
let claimReleased = false;
|
|
1392
|
+
if (classResult && classResult.summary) {
|
|
1393
|
+
releaseRequestClaim({
|
|
1394
|
+
recipient: item.channel_id || item.channel || item.sender || "unknown",
|
|
1395
|
+
subject: classResult.summary || item.subject || "",
|
|
1396
|
+
action_type: classResult.action || "respond",
|
|
1397
|
+
});
|
|
1398
|
+
claimReleased = true;
|
|
1399
|
+
}
|
|
1400
|
+
|
|
1401
|
+
// Release backlog tracking + item claim. Apply failure cooldown so the
|
|
1402
|
+
// same item isn't re-spawned on the next backlog sweep.
|
|
1403
|
+
if (source === "backlog") {
|
|
1404
|
+
const key = backlogKey(item);
|
|
1405
|
+
activeBacklogKeys.delete(key);
|
|
1406
|
+
backlogCooldownUntil.set(key, Date.now() + FAILURE_COOLDOWN_MS);
|
|
1407
|
+
saveCooldowns();
|
|
1408
|
+
// Release file-based item claim (ib-20260407-001b)
|
|
1409
|
+
if (item.id) releaseItemClaim(item.id);
|
|
1410
|
+
}
|
|
1411
|
+
|
|
1412
|
+
console.error(`[dispatcher] Session ${sessionId} failed: ${errorCode} (${err.message})`);
|
|
1413
|
+
|
|
1414
|
+
// Rich "failed" event — see ib-20260416-daemon-etimedout-failed-event.
|
|
1415
|
+
logSession({
|
|
1416
|
+
event: "failed",
|
|
1417
|
+
sessionId,
|
|
1418
|
+
error: errorCode,
|
|
1419
|
+
error_message: err.message,
|
|
1420
|
+
model,
|
|
1421
|
+
source,
|
|
1422
|
+
priority: classResult?.priority,
|
|
1423
|
+
summary: classResult?.summary,
|
|
1424
|
+
duration_s: parseFloat(duration),
|
|
1425
|
+
active_count: activeSessions.size,
|
|
1426
|
+
});
|
|
1427
|
+
|
|
1428
|
+
if (claimReleased) {
|
|
1429
|
+
logSession({
|
|
1430
|
+
event: "claim_released",
|
|
1431
|
+
sessionId,
|
|
1432
|
+
reason: `spawn_failed_${errorCode}`,
|
|
1433
|
+
});
|
|
1434
|
+
}
|
|
1435
|
+
|
|
1436
|
+
// Observability: a spawn failure is a terminal close too. Emit session_closed
|
|
1437
|
+
// (exit_code null, error code carried) so the audit sees every session reach a
|
|
1438
|
+
// terminal hop. Exactly-once: the trailing proc.on("close") early-returns via
|
|
1439
|
+
// the spawnErrorHandled guard WITHOUT re-emitting.
|
|
1440
|
+
emitEvent({
|
|
1441
|
+
type: EVENT_TYPES.SESSION_CLOSED,
|
|
1442
|
+
trace_id: traceId,
|
|
1443
|
+
attrs: { session_id: sessionId, source, exit_code: null, error: errorCode, decision_id: target.decisionId || null },
|
|
1444
|
+
});
|
|
1445
|
+
|
|
1446
|
+
// F1/H2: a spawn failure is terminal and the work never ran — tell the caller
|
|
1447
|
+
// (ok:false) so the inbox item is left re-deliverable, never `.processed`.
|
|
1448
|
+
fireOnClose(false, null);
|
|
1449
|
+
|
|
1450
|
+
drainQueue();
|
|
1451
|
+
});
|
|
1452
|
+
}
|
|
1453
|
+
|
|
1454
|
+
function drainQueue() {
|
|
1455
|
+
// A real drain supersedes any pending re-drain timer; clear it so we don't
|
|
1456
|
+
// also fire a redundant one. If we bail out under pressure below with items
|
|
1457
|
+
// still queued + no active session, we re-arm before returning (M1).
|
|
1458
|
+
if (reDrainTimer) { clearTimeout(reDrainTimer); reDrainTimer = null; }
|
|
1459
|
+
|
|
1460
|
+
// WS4: the shared 429 breaker gates the whole drain — if it's open, leave
|
|
1461
|
+
// everything queued (re-derivable; nothing dropped) and let a later close /
|
|
1462
|
+
// drain retry once the breaker closes.
|
|
1463
|
+
const rb = rateBlocked();
|
|
1464
|
+
if (rb) {
|
|
1465
|
+
// M1: with no session to drive a future close-drain, re-arm so the queued
|
|
1466
|
+
// item isn't stranded until the next external dispatch.
|
|
1467
|
+
if (activeSessions.size === 0 && (priorityQueue.length > 0 || normalQueue.length > 0)) {
|
|
1468
|
+
armReDrain(rb.retryAt);
|
|
1469
|
+
}
|
|
1470
|
+
return;
|
|
1471
|
+
}
|
|
1472
|
+
|
|
1473
|
+
while (activeSessions.size < MAX_CONCURRENT) {
|
|
1474
|
+
// Peek the head without removing it so a governor QUEUE/DEFER leaves the
|
|
1475
|
+
// item in place (never dropped).
|
|
1476
|
+
const next = priorityQueue[0] || normalQueue[0];
|
|
1477
|
+
if (!next) break;
|
|
1478
|
+
|
|
1479
|
+
const adm = admitFor(next.source, next.classResult?.priority);
|
|
1480
|
+
if (adm.decision !== "ADMIT") {
|
|
1481
|
+
// Under pressure: stop draining. Inbox items stay queued; the next close
|
|
1482
|
+
// event (a freed slot) or the breaker closing will resume the drain.
|
|
1483
|
+
logSession({ event: "drain_paused", reason: adm.reason, decision: adm.decision, source: next.source });
|
|
1484
|
+
// M1: if nothing is running, no close event will re-drain — arm a timer.
|
|
1485
|
+
if (activeSessions.size === 0) armReDrain(adm.retryAt);
|
|
1486
|
+
break;
|
|
1487
|
+
}
|
|
1488
|
+
|
|
1489
|
+
// Commit: remove from whichever queue it sat at the head of, then spawn.
|
|
1490
|
+
if (priorityQueue[0] === next) priorityQueue.shift();
|
|
1491
|
+
else normalQueue.shift();
|
|
1492
|
+
spawnSession(next);
|
|
1493
|
+
}
|
|
1494
|
+
}
|
|
1495
|
+
|
|
1496
|
+
/** Get current status for health checks */
|
|
1497
|
+
export function getStatus() {
|
|
1498
|
+
return {
|
|
1499
|
+
active_sessions: activeSessions.size,
|
|
1500
|
+
max_concurrent: MAX_CONCURRENT,
|
|
1501
|
+
priority_queue_length: priorityQueue.length,
|
|
1502
|
+
normal_queue_length: normalQueue.length,
|
|
1503
|
+
sessions: Array.from(activeSessions.entries()).map(([id, s]) => ({
|
|
1504
|
+
id,
|
|
1505
|
+
model: s.model,
|
|
1506
|
+
priority: s.classResult.priority,
|
|
1507
|
+
summary: s.classResult.summary,
|
|
1508
|
+
duration_s: ((Date.now() - s.startTime) / 1000).toFixed(0),
|
|
1509
|
+
})),
|
|
1510
|
+
};
|
|
1511
|
+
}
|
|
1512
|
+
|
|
1513
|
+
/** Get available slots count */
|
|
1514
|
+
export function availableSlots() {
|
|
1515
|
+
return MAX_CONCURRENT - activeSessions.size;
|
|
1516
|
+
}
|