@cohortapp/agent-sdk 2.17.0 → 2.18.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/settings.json +18 -0
- package/.env.example +18 -5
- package/README.md +1 -0
- package/bin/maestro.mjs +62 -0
- package/docs/guides/billing-console-keys.md +60 -0
- package/docs/guides/front-door-session.md +54 -9
- package/docs/guides/mac-mini.md +20 -25
- package/docs/guides/setup-wizard.md +1 -1
- package/docs/runbooks/fleet-rollout.md +156 -0
- package/docs/runbooks/mac-mini-bootstrap.md +12 -14
- package/lib/action-executor.js +19 -3
- package/lib/budget-guard.mjs +279 -3
- package/lib/channels/base-adapter.mjs +3 -1
- package/lib/channels/contract.mjs +2 -1
- package/lib/channels/inbox-item.mjs +8 -0
- package/lib/claude-bin.mjs +5 -6
- package/lib/cli/doctor-checks.mjs +141 -10
- package/lib/cli/global-setup-extras.mjs +5 -1
- package/lib/cli/inbox.mjs +100 -15
- package/lib/cli/seat-auth.mjs +463 -0
- package/lib/cli/session.mjs +80 -12
- package/lib/collective/capture.mjs +8 -6
- package/lib/collective/global-config.mjs +63 -1
- package/lib/collective/presence.mjs +142 -5
- package/lib/comms/send-gate.mjs +559 -1
- package/lib/diagnostics/alerts.mjs +49 -0
- package/lib/diagnostics/cadence-output-freshness.mjs +288 -0
- package/lib/engine/agents/definitions.mjs +343 -0
- package/lib/engine/agents/persist.mjs +275 -0
- package/lib/engine/agents/runtime.mjs +748 -0
- package/lib/engine/agents/usage.mjs +95 -0
- package/lib/engine/auth-status.mjs +139 -0
- package/lib/engine/budget.mjs +194 -0
- package/lib/engine/cli.mjs +1204 -0
- package/lib/engine/commands/index.mjs +269 -0
- package/lib/engine/context/budget.mjs +219 -0
- package/lib/engine/context/cache.mjs +125 -0
- package/lib/engine/context/child-env.mjs +215 -0
- package/lib/engine/context/compaction.mjs +342 -0
- package/lib/engine/context/images.mjs +90 -0
- package/lib/engine/context/instructions.mjs +327 -0
- package/lib/engine/context/lazy-instructions.mjs +169 -0
- package/lib/engine/context/manager.mjs +182 -0
- package/lib/engine/context/real-path.mjs +91 -0
- package/lib/engine/context/secret-values.mjs +163 -0
- package/lib/engine/context/settings.mjs +274 -0
- package/lib/engine/context/stream-input.mjs +159 -0
- package/lib/engine/guard.mjs +152 -0
- package/lib/engine/hooks.mjs +713 -0
- package/lib/engine/loop.mjs +560 -0
- package/lib/engine/mcp/client.mjs +254 -0
- package/lib/engine/mcp/config.mjs +301 -0
- package/lib/engine/mcp/http.mjs +201 -0
- package/lib/engine/mcp/index.mjs +146 -0
- package/lib/engine/mcp/jsonrpc.mjs +147 -0
- package/lib/engine/mcp/naming.mjs +66 -0
- package/lib/engine/mcp/resources.mjs +89 -0
- package/lib/engine/mcp/results.mjs +133 -0
- package/lib/engine/mcp/stdio.mjs +137 -0
- package/lib/engine/mcp/supervisor.mjs +116 -0
- package/lib/engine/messages.mjs +104 -0
- package/lib/engine/output/json.mjs +164 -0
- package/lib/engine/output/stream-json.mjs +266 -0
- package/lib/engine/permissions.mjs +845 -0
- package/lib/engine/process-identity.mjs +164 -0
- package/lib/engine/process-tree.mjs +551 -0
- package/lib/engine/prompt.mjs +60 -0
- package/lib/engine/session/store.mjs +299 -0
- package/lib/engine/session-runtime/args.mjs +97 -0
- package/lib/engine/session-runtime/host.mjs +143 -0
- package/lib/engine/session-runtime/inbox.mjs +122 -0
- package/lib/engine/session-runtime/notifications.mjs +129 -0
- package/lib/engine/session-runtime/registry.mjs +328 -0
- package/lib/engine/session-runtime/runner.mjs +344 -0
- package/lib/engine/session-runtime/socket.mjs +212 -0
- package/lib/engine/session-runtime/wakeup.mjs +115 -0
- package/lib/engine/skills/index.mjs +321 -0
- package/lib/engine/tools/bash-background.mjs +533 -0
- package/lib/engine/tools/bash.mjs +216 -0
- package/lib/engine/tools/edit.mjs +97 -0
- package/lib/engine/tools/glob.mjs +81 -0
- package/lib/engine/tools/grep.mjs +224 -0
- package/lib/engine/tools/index.mjs +84 -0
- package/lib/engine/tools/list-agents.mjs +32 -0
- package/lib/engine/tools/ls.mjs +127 -0
- package/lib/engine/tools/monitor.mjs +82 -0
- package/lib/engine/tools/notebook-edit.mjs +218 -0
- package/lib/engine/tools/read.mjs +103 -0
- package/lib/engine/tools/schedule-wakeup.mjs +45 -0
- package/lib/engine/tools/schema.mjs +144 -0
- package/lib/engine/tools/send-message.mjs +77 -0
- package/lib/engine/tools/session.mjs +70 -0
- package/lib/engine/tools/todo.mjs +144 -0
- package/lib/engine/tools/toolsearch.mjs +217 -0
- package/lib/engine/tools/walk.mjs +193 -0
- package/lib/engine/tools/web-switch.mjs +31 -0
- package/lib/engine/tools/webfetch-html.mjs +387 -0
- package/lib/engine/tools/webfetch-net.mjs +340 -0
- package/lib/engine/tools/webfetch.mjs +198 -0
- package/lib/engine/tools/websearch.mjs +91 -0
- package/lib/engine/tools/workflow.mjs +95 -0
- package/lib/engine/tools/write.mjs +76 -0
- package/lib/engine/tui/line-editor.mjs +137 -0
- package/lib/engine/tui/render.mjs +86 -0
- package/lib/engine/tui/tui.mjs +274 -0
- package/lib/engine/wire/anthropic-messages.mjs +263 -0
- package/lib/engine/wire/effort.mjs +36 -0
- package/lib/engine/wire/errors.mjs +496 -0
- package/lib/engine/wire/http.mjs +441 -0
- package/lib/engine/wire/index.mjs +76 -0
- package/lib/engine/wire/openai-chat.mjs +332 -0
- package/lib/engine/wire/prompt-cache.mjs +79 -0
- package/lib/engine/wire/search.mjs +140 -0
- package/lib/engine/wire/sse.mjs +114 -0
- package/lib/engine/wire/stall.mjs +349 -0
- package/lib/engine/wire/token-provider.mjs +175 -0
- package/lib/engine/wire/usage.mjs +192 -0
- package/lib/engine/workflow/host.mjs +524 -0
- package/lib/engine/workflow/journal.mjs +188 -0
- package/lib/engine/workflow/json-schema.mjs +171 -0
- package/lib/engine/workflow/meta.mjs +329 -0
- package/lib/engine/workflow/notifications.mjs +52 -0
- package/lib/engine/workflow/runtime.mjs +447 -0
- package/lib/engine/workflow/sandbox.mjs +534 -0
- package/lib/engine/workflow/worker.mjs +141 -0
- package/lib/engine/workflow/worktree.mjs +74 -0
- package/lib/execution/disposition.mjs +1 -1
- package/lib/execution/intake.mjs +10 -0
- package/lib/execution/surface-policy.mjs +15 -0
- package/lib/learning/curator.mjs +8 -6
- package/lib/learning/reflect.mjs +8 -6
- package/lib/model-router/catalog/cohort.yaml +137 -0
- package/lib/model-router/catalog.mjs +118 -1
- package/lib/model-router/failover.mjs +67 -16
- package/lib/model-router/llm-task.mjs +39 -3
- package/lib/model-router/resolve.mjs +89 -3
- package/lib/model-router/spawn.mjs +46 -47
- package/lib/model-router/taxonomy.mjs +126 -4
- package/lib/org/cost-sync.mjs +141 -11
- package/lib/org/inbound/broadcast.mjs +289 -0
- package/lib/org/inbound/collective.mjs +375 -0
- package/lib/org/inbound/directedness.mjs +96 -8
- package/lib/org/inbound/facts.mjs +78 -2
- package/lib/org/inbound/project.mjs +22 -0
- package/lib/org/inbound/surfaces.mjs +14 -0
- package/lib/org/llm-token.mjs +879 -0
- package/lib/org/mesh.mjs +61 -0
- package/lib/org/messaging.mjs +3 -1
- package/lib/org/protocol.checksum +1 -1
- package/lib/org/protocol.mjs +15 -0
- package/lib/org/quota.mjs +520 -0
- package/lib/org/tool-surface.mjs +104 -16
- package/lib/org/ui-parity.mjs +16 -1
- package/lib/org/work-ledger.mjs +37 -6
- package/lib/rate-guard.mjs +114 -1
- package/lib/resource-governor.mjs +41 -6
- package/lib/runtime/adapter.mjs +823 -0
- package/lib/runtime/child-env.mjs +191 -0
- package/lib/runtime/legacy-shell-guard.mjs +97 -0
- package/lib/runtime/seat-engine.mjs +162 -0
- package/lib/session/ask-ledger.mjs +271 -0
- package/lib/session/current-work.mjs +676 -0
- package/lib/session/feed-core.mjs +40 -3
- package/lib/session/launch-args.mjs +56 -4
- package/lib/session/status-summary.mjs +26 -9
- package/lib/session/upgrade-notice.mjs +42 -0
- package/lib/setup/claude-probe.mjs +117 -13
- package/lib/setup/enrich.mjs +13 -10
- package/lib/setup/sections/model.mjs +39 -13
- package/lib/telemetry/collect.mjs +208 -9
- package/lib/upgrade/ignored-drift.mjs +105 -0
- package/lib/voice/post-call-brief.mjs +30 -17
- package/package.json +13 -3
- package/plugins/maestro-skills/skills/board-work.md +5 -0
- package/plugins/maestro-skills/skills/inbound-triage.md +56 -15
- package/plugins/maestro-skills/skills/main-session.md +18 -7
- package/scripts/ci/check-tarball-fidelity.mjs +126 -2
- package/scripts/ci/run-tests.mjs +47 -19
- package/scripts/cohort-llm/api-key-helper.mjs +92 -0
- package/scripts/collective/hook-runner.mjs +29 -2
- package/scripts/continuous-monitor.sh +13 -0
- package/scripts/cost/track-claude-usage.mjs +15 -0
- package/scripts/daemon/agent-daemon.mjs +408 -20
- package/scripts/daemon/assurance.mjs +48 -12
- package/scripts/daemon/cadence-consumer.mjs +218 -68
- package/scripts/daemon/cadence-handlers.mjs +73 -4
- package/scripts/daemon/classifier.mjs +75 -26
- package/scripts/daemon/context-compiler.mjs +51 -37
- package/scripts/daemon/deliver.mjs +30 -1
- package/scripts/daemon/dispatcher.mjs +595 -149
- package/scripts/daemon/health.mjs +14 -1
- package/scripts/daemon/maestro-daemon.mjs +11 -0
- package/scripts/daemon/prompt-builder.mjs +24 -0
- package/scripts/daemon/responder.mjs +246 -79
- package/scripts/daemon/sdk-version.mjs +98 -16
- package/scripts/eval/probe-gateway.mjs +635 -0
- package/scripts/eval/replay/extract.mjs +270 -0
- package/scripts/eval/replay/grade.mjs +260 -0
- package/scripts/eval/replay/lib/config.mjs +50 -0
- package/scripts/eval/replay/lib/effects.mjs +65 -0
- package/scripts/eval/replay/lib/fixture.mjs +188 -0
- package/scripts/eval/replay/lib/judge.mjs +72 -0
- package/scripts/eval/replay/lib/redact.mjs +136 -0
- package/scripts/eval/replay/lib/sandbox.mjs +170 -0
- package/scripts/eval/replay/lib/schema-check.mjs +63 -0
- package/scripts/eval/replay/lib/transcript.mjs +76 -0
- package/scripts/eval/replay/mcp-replay-stub.mjs +101 -0
- package/scripts/eval/replay/report.mjs +185 -0
- package/scripts/eval/replay/run.mjs +404 -0
- package/scripts/fleet/rollout.mjs +1094 -0
- package/scripts/hooks/pre-send-audit.sh +36 -245
- package/scripts/hooks/pre-write-yaml-validate.mjs +275 -0
- package/scripts/hooks/validate-state-yaml.sh +190 -0
- package/scripts/huddle/huddle-llm.mjs +361 -0
- package/scripts/huddle/huddle-server.mjs +46 -121
- package/scripts/local-triggers/autoupdate.sh +448 -78
- package/scripts/local-triggers/run-trigger.sh +13 -0
- package/scripts/maintenance/pin-integrity.mjs +364 -0
- package/scripts/poll-slack-events.sh +41 -9
- package/scripts/poller/slack-socket-mode.mjs +28 -3
- package/scripts/session/supervisor.mjs +80 -13
- package/scripts/spawn-session.sh +13 -0
- package/bin/maestro.test.mjs +0 -1574
- package/lib/action-executor.test.mjs +0 -871
- package/lib/archetype.test.mjs +0 -132
- package/lib/assurance/plan-note.test.mjs +0 -234
- package/lib/assurance/room-budget.test.mjs +0 -486
- package/lib/assurance/tier.test.mjs +0 -174
- package/lib/autonomy.test.mjs +0 -66
- package/lib/backlog.test.mjs +0 -302
- package/lib/backup/policy.test.mjs +0 -305
- package/lib/budget-escalate.test.mjs +0 -232
- package/lib/budget-guard.envelope.test.mjs +0 -476
- package/lib/budget-guard.test.mjs +0 -427
- package/lib/cadence-bus-requeue.test.mjs +0 -83
- package/lib/cadence-bus-schedule.test.mjs +0 -194
- package/lib/cadence-bus.test.mjs +0 -720
- package/lib/cadences.test.mjs +0 -230
- package/lib/capability/inventory.test.mjs +0 -232
- package/lib/capability.test.mjs +0 -78
- package/lib/channels/base-adapter.test.mjs +0 -590
- package/lib/channels/channels.test.mjs +0 -371
- package/lib/channels/contract.test.mjs +0 -162
- package/lib/channels/inbox-item.test.mjs +0 -368
- package/lib/channels/orgmail/adapter.test.mjs +0 -448
- package/lib/channels/pairing.test.mjs +0 -270
- package/lib/channels/repeat-suppressor.test.mjs +0 -134
- package/lib/channels/slack-adapter.test.mjs +0 -212
- package/lib/channels/telegram-adapter.test.mjs +0 -306
- package/lib/channels/voice/adapter.test.mjs +0 -278
- package/lib/channels/whatsapp/adapter-baileys.test.mjs +0 -359
- package/lib/channels/whatsapp/baileys-typing.test.mjs +0 -154
- package/lib/charter.test.mjs +0 -89
- package/lib/claude-bin.test.mjs +0 -131
- package/lib/cli/board.test.mjs +0 -227
- package/lib/cli/design.test.mjs +0 -270
- package/lib/cli/doctor-checks.test.mjs +0 -336
- package/lib/cli/global-setup-extras.test.mjs +0 -462
- package/lib/cli/inbox.test.mjs +0 -230
- package/lib/cli/session-ack.test.mjs +0 -63
- package/lib/cli/session.test.mjs +0 -613
- package/lib/collective/capture.test.mjs +0 -121
- package/lib/collective/cards.test.mjs +0 -114
- package/lib/collective/config.test.mjs +0 -123
- package/lib/collective/global-config.test.mjs +0 -220
- package/lib/collective/global-skills.test.mjs +0 -126
- package/lib/collective/presence.test.mjs +0 -95
- package/lib/collective/recall.test.mjs +0 -116
- package/lib/collective/vendor-skills.test.mjs +0 -306
- package/lib/comms/send-gate.test.mjs +0 -770
- package/lib/comms.test.mjs +0 -41
- package/lib/context/budget.test.mjs +0 -252
- package/lib/context/history-scope.test.mjs +0 -79
- package/lib/cost/ledger-row.test.mjs +0 -183
- package/lib/design/design-md.test.mjs +0 -318
- package/lib/design/fixtures/DESIGN.golden.md +0 -238
- package/lib/design/fixtures/PRODUCT.golden.md +0 -67
- package/lib/design/fixtures/foundation.json +0 -133
- package/lib/design/refresh-gate.test.mjs +0 -144
- package/lib/design/write.test.mjs +0 -241
- package/lib/diagnostics/alerts.test.mjs +0 -318
- package/lib/diagnostics/backup-freshness.test.mjs +0 -185
- package/lib/diagnostics/counters.test.mjs +0 -206
- package/lib/diagnostics/events.test.mjs +0 -290
- package/lib/diagnostics/otel.test.mjs +0 -196
- package/lib/diagnostics/trace.test.mjs +0 -251
- package/lib/env-compat.test.mjs +0 -104
- package/lib/execution/disposition.test.mjs +0 -553
- package/lib/execution/drive.test.mjs +0 -270
- package/lib/execution/effects.test.mjs +0 -344
- package/lib/execution/intake.test.mjs +0 -389
- package/lib/execution/journal.test.mjs +0 -261
- package/lib/execution/match.test.mjs +0 -235
- package/lib/execution/pipeline.test.mjs +0 -392
- package/lib/execution/route.test.mjs +0 -186
- package/lib/execution/surface-policy.test.mjs +0 -162
- package/lib/fs-atomic.test.mjs +0 -72
- package/lib/fs-ownership.test.mjs +0 -158
- package/lib/goals/admission.test.mjs +0 -164
- package/lib/goals/classify.test.mjs +0 -167
- package/lib/goals/collaborate.test.mjs +0 -336
- package/lib/goals/gaps.test.mjs +0 -284
- package/lib/goals/loop.test.mjs +0 -845
- package/lib/hooks/bus.test.mjs +0 -387
- package/lib/identity/persona.test.mjs +0 -142
- package/lib/kpi-sensors.test.mjs +0 -278
- package/lib/kpi.test.mjs +0 -244
- package/lib/learning/config.test.mjs +0 -75
- package/lib/learning/counters.test.mjs +0 -69
- package/lib/learning/curator-consolidate.test.mjs +0 -238
- package/lib/learning/curator.test.mjs +0 -106
- package/lib/learning/reflect.test.mjs +0 -0
- package/lib/learning/session-index.test.mjs +0 -125
- package/lib/learning/skill-writer.test.mjs +0 -210
- package/lib/mandate/audit.test.mjs +0 -195
- package/lib/mandate/contract.test.mjs +0 -185
- package/lib/mandate/derive.test.mjs +0 -274
- package/lib/mandate/model.test.mjs +0 -164
- package/lib/mandate/refresh.test.mjs +0 -389
- package/lib/mcp/server.test.mjs +0 -426
- package/lib/model-router/auth-profiles.test.mjs +0 -580
- package/lib/model-router/catalog.test.mjs +0 -385
- package/lib/model-router/economics.test.mjs +0 -438
- package/lib/model-router/failover.test.mjs +0 -439
- package/lib/model-router/health.test.mjs +0 -338
- package/lib/model-router/integration-coverage.test.mjs +0 -831
- package/lib/model-router/integration.test.mjs +0 -564
- package/lib/model-router/ledger.test.mjs +0 -415
- package/lib/model-router/llm-task.test.mjs +0 -392
- package/lib/model-router/org-credentials.test.mjs +0 -265
- package/lib/model-router/pricing-refresh.test.mjs +0 -286
- package/lib/model-router/reconcile.test.mjs +0 -316
- package/lib/model-router/repair.test.mjs +0 -180
- package/lib/model-router/spawn.test.mjs +0 -446
- package/lib/model-router/taxonomy.test.mjs +0 -410
- package/lib/model-router.test.mjs +0 -1207
- package/lib/org/activity.test.mjs +0 -134
- package/lib/org/approvals.test.mjs +0 -216
- package/lib/org/awareness.test.mjs +0 -159
- package/lib/org/board-mine-cache.test.mjs +0 -53
- package/lib/org/board.test.mjs +0 -187
- package/lib/org/bootstrap-context.test.mjs +0 -153
- package/lib/org/client.test.mjs +0 -1206
- package/lib/org/cohort-client.test.mjs +0 -126
- package/lib/org/cost-sync.test.mjs +0 -153
- package/lib/org/doctor.test.mjs +0 -346
- package/lib/org/engagement-ledger.test.mjs +0 -112
- package/lib/org/engagement.test.mjs +0 -739
- package/lib/org/handoff.test.mjs +0 -269
- package/lib/org/inbound/directedness.test.mjs +0 -668
- package/lib/org/inbound/facts.test.mjs +0 -471
- package/lib/org/inbound/hydrate.test.mjs +0 -908
- package/lib/org/inbound/index.test.mjs +0 -429
- package/lib/org/inbound/project.test.mjs +0 -287
- package/lib/org/integration-tools.test.mjs +0 -160
- package/lib/org/keys.test.mjs +0 -92
- package/lib/org/knowledge.test.mjs +0 -326
- package/lib/org/leases.test.mjs +0 -235
- package/lib/org/mesh-directives.test.mjs +0 -110
- package/lib/org/mesh-integration.test.mjs +0 -127
- package/lib/org/mesh.test.mjs +0 -400
- package/lib/org/messaging.test.mjs +0 -471
- package/lib/org/param-contract.test.mjs +0 -477
- package/lib/org/policy.test.mjs +0 -237
- package/lib/org/protocol.checksum.test.mjs +0 -90
- package/lib/org/protocol.test.mjs +0 -323
- package/lib/org/push.test.mjs +0 -792
- package/lib/org/registry.test.mjs +0 -100
- package/lib/org/resource-tools.test.mjs +0 -361
- package/lib/org/tool-access.test.mjs +0 -144
- package/lib/org/tool-surface-integration.test.mjs +0 -120
- package/lib/org/tool-surface.test.mjs +0 -1268
- package/lib/org/typing.test.mjs +0 -291
- package/lib/org/ui-parity.test.mjs +0 -560
- package/lib/org/verify.test.mjs +0 -194
- package/lib/org/work-ledger.test.mjs +0 -273
- package/lib/plan/adoption-e2e.test.mjs +0 -366
- package/lib/plan/budget-enforcement.test.mjs +0 -400
- package/lib/plan/compile.test.mjs +0 -382
- package/lib/plan/emit.test.mjs +0 -269
- package/lib/plan/explain.test.mjs +0 -188
- package/lib/prompts/parallelism.test.mjs +0 -177
- package/lib/rag/rag.test.mjs +0 -505
- package/lib/rate-guard.test.mjs +0 -272
- package/lib/reactive-gate.test.mjs +0 -57
- package/lib/render.test.mjs +0 -68
- package/lib/resource-governor.test.mjs +0 -488
- package/lib/scheduling/dynamic-jobs.test.mjs +0 -344
- package/lib/scheduling/jitter.test.mjs +0 -140
- package/lib/secrets/broker.test.mjs +0 -280
- package/lib/secrets/providers.test.mjs +0 -274
- package/lib/security/audit-engine.test.mjs +0 -424
- package/lib/security/coerce-args.test.mjs +0 -281
- package/lib/security/dangerous-tools.test.mjs +0 -68
- package/lib/security/external-content.test.mjs +0 -84
- package/lib/security/redact.test.mjs +0 -441
- package/lib/security/secret-equal.test.mjs +0 -55
- package/lib/session/config.test.mjs +0 -92
- package/lib/session/feed-core.test.mjs +0 -198
- package/lib/session/first-run.test.mjs +0 -121
- package/lib/session/frontdoor.test.mjs +0 -205
- package/lib/session/handoffs.test.mjs +0 -183
- package/lib/session/identity.test.mjs +0 -180
- package/lib/session/inbox-claims.test.mjs +0 -286
- package/lib/session/launch-args.test.mjs +0 -157
- package/lib/session/liveness.test.mjs +0 -100
- package/lib/session/status-summary.test.mjs +0 -118
- package/lib/session-permissions.test.mjs +0 -120
- package/lib/setup/claude-probe.test.mjs +0 -187
- package/lib/setup/completeness.test.mjs +0 -110
- package/lib/setup/context-pack.test.mjs +0 -89
- package/lib/setup/enrich.test.mjs +0 -115
- package/lib/setup/enroll-from-cohort.test.mjs +0 -300
- package/lib/setup/integration.test.mjs +0 -162
- package/lib/setup/io.test.mjs +0 -77
- package/lib/setup/runner.test.mjs +0 -132
- package/lib/setup/sections/identity.test.mjs +0 -234
- package/lib/setup/sections/inventory.test.mjs +0 -198
- package/lib/setup/sections/learning.test.mjs +0 -81
- package/lib/setup/sections/mandate.test.mjs +0 -388
- package/lib/setup/sections/messaging.test.mjs +0 -127
- package/lib/setup/sections/model.test.mjs +0 -240
- package/lib/setup/sections/org.test.mjs +0 -346
- package/lib/setup/sections/orgmail.test.mjs +0 -118
- package/lib/setup/sections/recovery.test.mjs +0 -98
- package/lib/setup/sections/subagents.test.mjs +0 -429
- package/lib/setup/sections/verify.test.mjs +0 -175
- package/lib/setup/sot.test.mjs +0 -81
- package/lib/setup/state.test.mjs +0 -115
- package/lib/singleton.test.mjs +0 -151
- package/lib/subagents/cli.test.mjs +0 -389
- package/lib/subagents/client.test.mjs +0 -309
- package/lib/subagents/gap.test.mjs +0 -234
- package/lib/subagents/lock.test.mjs +0 -248
- package/lib/subagents/manifest.test.mjs +0 -175
- package/lib/subagents/refs.test.mjs +0 -204
- package/lib/subagents/resolve.test.mjs +0 -422
- package/lib/subagents/schema.test.mjs +0 -328
- package/lib/telemetry/alerts.test.mjs +0 -109
- package/lib/telemetry/collect.test.mjs +0 -1274
- package/lib/tool-definitions-integration.test.mjs +0 -83
- package/lib/tool-definitions.test.mjs +0 -437
- package/lib/upgrade/global-refresh.test.mjs +0 -65
- package/lib/upgrade/launchd-reconcile.test.mjs +0 -272
- package/lib/upgrade/post-steps.test.mjs +0 -200
- package/lib/upgrade/verify.test.mjs +0 -164
- package/lib/util/fetch-timeout.test.mjs +0 -202
- package/lib/util/reconnect.test.mjs +0 -369
- package/lib/util/unhandled.test.mjs +0 -216
- package/lib/voice/outbound.test.mjs +0 -69
- package/lib/voice/session-rotation.test.mjs +0 -114
- package/lib/voice/stt.test.mjs +0 -226
- package/lib/voice/voice.test.mjs +0 -990
- package/scripts/cadence/enqueue-cadence-tick.test.mjs +0 -187
- package/scripts/ci/check-docs-accuracy.test.mjs +0 -409
- package/scripts/ci/check-durable-write-seam.test.mjs +0 -90
- package/scripts/ci/check-no-build-artifacts.test.mjs +0 -71
- package/scripts/ci/check-no-residual-identity.test.mjs +0 -202
- package/scripts/ci/check-skill-packs.test.mjs +0 -495
- package/scripts/ci/check-subagent-frontmatter.test.mjs +0 -124
- package/scripts/ci/check.test.mjs +0 -194
- package/scripts/ci/conformance-org-api.test.mjs +0 -425
- package/scripts/cloud-relay/voice/relay-identity.test.mjs +0 -96
- package/scripts/collective/hook-runner.test.mjs +0 -173
- package/scripts/cost/fleet-digest.test.mjs +0 -207
- package/scripts/cost/track-claude-usage-pricing.test.mjs +0 -183
- package/scripts/cost/track-claude-usage.test.mjs +0 -148
- package/scripts/daemon/agent-daemon-board-mine.test.mjs +0 -96
- package/scripts/daemon/agent-daemon-design.test.mjs +0 -238
- package/scripts/daemon/agent-daemon-frontdoor.test.mjs +0 -60
- package/scripts/daemon/agent-daemon.test.mjs +0 -995
- package/scripts/daemon/assurance-e2e.test.mjs +0 -613
- package/scripts/daemon/assurance.test.mjs +0 -1791
- package/scripts/daemon/board-mirror.test.mjs +0 -165
- package/scripts/daemon/cadence-consumer-frontdoor.test.mjs +0 -393
- package/scripts/daemon/cadence-consumer-governance.test.mjs +0 -276
- package/scripts/daemon/cadence-consumer.test.mjs +0 -776
- package/scripts/daemon/cadence-handlers.test.mjs +0 -837
- package/scripts/daemon/classifier-identity.test.mjs +0 -137
- package/scripts/daemon/classifier.test.mjs +0 -266
- package/scripts/daemon/classify-kind.test.mjs +0 -40
- package/scripts/daemon/context-compiler.test.mjs +0 -406
- package/scripts/daemon/deliver.test.mjs +0 -564
- package/scripts/daemon/dispatcher-cooldown.test.mjs +0 -122
- package/scripts/daemon/dispatcher-governance.test.mjs +0 -1013
- package/scripts/daemon/dispatcher-resume.test.mjs +0 -166
- package/scripts/daemon/dispatcher-session-continuity.test.mjs +0 -365
- package/scripts/daemon/execution-ladder.test.mjs +0 -470
- package/scripts/daemon/goal-steward-cadence.test.mjs +0 -312
- package/scripts/daemon/inbox-deferral-session.test.mjs +0 -49
- package/scripts/daemon/inbox-deferral.test.mjs +0 -336
- package/scripts/daemon/inbox-wake.test.mjs +0 -199
- package/scripts/daemon/integration.test.mjs +0 -149
- package/scripts/daemon/lib/self-echo.test.mjs +0 -153
- package/scripts/daemon/lib/session-router.test.mjs +0 -554
- package/scripts/daemon/prompt-builder-preamble.test.mjs +0 -210
- package/scripts/daemon/prompt-builder.test.mjs +0 -556
- package/scripts/daemon/responder-cost.test.mjs +0 -68
- package/scripts/daemon/responder-history.test.mjs +0 -221
- package/scripts/daemon/sdk-version.test.mjs +0 -31
- package/scripts/daemon/session-lock.test.mjs +0 -252
- package/scripts/daemon/session-outcomes.test.mjs +0 -533
- package/scripts/daemon/typing-registry.test.mjs +0 -102
- package/scripts/hooks/pre-send-audit.test.mjs +0 -354
- package/scripts/huddle/huddle-prompt.test.mjs +0 -176
- package/scripts/local-triggers/autoupdate.test.mjs +0 -518
- package/scripts/local-triggers/generate-plists.test.mjs +0 -456
- package/scripts/media-generation/brand-clause.test.mjs +0 -135
- package/scripts/org/send-orgmail.first-contact.test.mjs +0 -102
- package/scripts/poller/inbox-privilege-injection.test.mjs +0 -167
- package/scripts/poller/inbox-scan-poller.test.mjs +0 -295
- package/scripts/poller/lib/cloud-relay-dedup.test.mjs +0 -133
- package/scripts/poller/slack-socket-mode.test.mjs +0 -805
- package/scripts/poller-launchd/install.test.mjs +0 -243
- package/scripts/restore-from-backup.test.mjs +0 -181
- package/scripts/session/feed.test.mjs +0 -196
- package/scripts/session/supervisor-sh.test.mjs +0 -218
- package/scripts/session/supervisor.test.mjs +0 -482
- package/scripts/setup/configure-macos.test.mjs +0 -306
- package/scripts/setup/gen-subagent-manifest.test.mjs +0 -124
- package/scripts/setup/generate-agent-package-json.test.mjs +0 -143
- package/scripts/setup/generate-capability.test.mjs +0 -134
- package/scripts/setup/init-agent.test.mjs +0 -370
- package/scripts/setup/init-skill-marketplace.test.mjs +0 -193
- package/scripts/vendor/sync-skill-packs.test.mjs +0 -103
- package/scripts/watchdog/memory-watchdog.test.mjs +0 -64
|
@@ -1,1207 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* model-router.test.mjs — node:test coverage for the model router.
|
|
3
|
-
*
|
|
4
|
-
* Tests write configs to a temp dir + load them via loadRoutingConfig so
|
|
5
|
-
* we exercise the full file-system path. No real network, no real Claude.
|
|
6
|
-
*/
|
|
7
|
-
|
|
8
|
-
import { test } from "node:test";
|
|
9
|
-
import assert from "node:assert/strict";
|
|
10
|
-
import { promises as fsp } from "fs";
|
|
11
|
-
import { mkdirSync, rmSync, writeFileSync } from "node:fs";
|
|
12
|
-
import { tmpdir } from "os";
|
|
13
|
-
import { join } from "path";
|
|
14
|
-
import { fileURLToPath } from "node:url";
|
|
15
|
-
|
|
16
|
-
import {
|
|
17
|
-
loadRoutingConfig,
|
|
18
|
-
defaultRoutingConfig,
|
|
19
|
-
resolveBackend,
|
|
20
|
-
estimateCost,
|
|
21
|
-
listBackends,
|
|
22
|
-
describeBackend,
|
|
23
|
-
modelFlagFor,
|
|
24
|
-
requestFromClassifierResult,
|
|
25
|
-
CONFIG_RELATIVE_PATH,
|
|
26
|
-
resolveChain,
|
|
27
|
-
validateRoutingConfig,
|
|
28
|
-
} from "./model-router.mjs";
|
|
29
|
-
import { loadCatalog } from "./model-router/catalog.mjs";
|
|
30
|
-
// Read-only, to PROVE (not assume) that a provider the SDK does not bundle
|
|
31
|
-
// inherits credential pooling from the catalog rather than needing new code.
|
|
32
|
-
import { loadAuthProfiles, authEnvMapFromCatalog } from "./model-router/auth-profiles.mjs";
|
|
33
|
-
|
|
34
|
-
async function makeAgentRoot() {
|
|
35
|
-
const path = join(
|
|
36
|
-
tmpdir(),
|
|
37
|
-
`model-router-test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`
|
|
38
|
-
);
|
|
39
|
-
await fsp.mkdir(join(path, "config"), { recursive: true });
|
|
40
|
-
return path;
|
|
41
|
-
}
|
|
42
|
-
|
|
43
|
-
async function rmRoot(path) {
|
|
44
|
-
try { await fsp.rm(path, { recursive: true, force: true }); } catch { /* */ }
|
|
45
|
-
}
|
|
46
|
-
|
|
47
|
-
function writeJsonConfig(root, body) {
|
|
48
|
-
// We write JSON (not YAML) so the tests don't depend on js-yaml being
|
|
49
|
-
// installed in CI. The loader treats .json identically.
|
|
50
|
-
writeFileSync(join(root, "config/model-routing.json"), JSON.stringify(body, null, 2));
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
const ANTHROPIC_BACKEND = {
|
|
54
|
-
transport: "anthropic-cli",
|
|
55
|
-
base_url: "https://api.anthropic.com",
|
|
56
|
-
capabilities: ["thinking", "vision", "prompt_cache_1h", "tool_use", "parallel_tools", "long_context_1m"],
|
|
57
|
-
pricing: { input_per_m: 3.0, output_per_m: 15.0 },
|
|
58
|
-
// A CONFIG-DECLARED model map. Deliberately NOT the same ids as
|
|
59
|
-
// ANTHROPIC_DEFAULT: these tests assert that a config's own map is what
|
|
60
|
-
// resolveBackend serves, so pinning them to the built-in defaults would make
|
|
61
|
-
// the assertions vacuous (they would pass even if the config map were being
|
|
62
|
-
// ignored entirely). The built-in defaults get their own test below.
|
|
63
|
-
models: {
|
|
64
|
-
classifier: "claude-haiku-4-5-20251001",
|
|
65
|
-
default: "config-declared-sonnet",
|
|
66
|
-
premium: "config-declared-opus",
|
|
67
|
-
fast: "claude-haiku-4-5-20251001",
|
|
68
|
-
},
|
|
69
|
-
};
|
|
70
|
-
|
|
71
|
-
const MOONSHOT_BACKEND = {
|
|
72
|
-
transport: "anthropic-native",
|
|
73
|
-
base_url: "https://api.moonshot.ai/anthropic",
|
|
74
|
-
auth_env: "MOONSHOT_API_KEY",
|
|
75
|
-
capabilities: ["tool_use", "prompt_cache_hit", "long_context_262k"],
|
|
76
|
-
pricing: { input_per_m: 0.95, output_per_m: 4.0, cache_hit_per_m: 0.16 },
|
|
77
|
-
models: { default: "kimi-k2.6", fast: "kimi-k2.6" },
|
|
78
|
-
};
|
|
79
|
-
|
|
80
|
-
const QWEN_BACKEND = {
|
|
81
|
-
transport: "openai-compat",
|
|
82
|
-
base_url: "https://openrouter.ai/api/v1",
|
|
83
|
-
auth_env: "OPENROUTER_API_KEY",
|
|
84
|
-
capabilities: ["tool_use", "long_context_262k", "degraded_tool_use"], // mimic real-world OR bug
|
|
85
|
-
pricing: { input_per_m: 0.22, output_per_m: 1.8 },
|
|
86
|
-
models: { default: "qwen/qwen3-coder", fast: "qwen/qwen3-coder-flash" },
|
|
87
|
-
};
|
|
88
|
-
|
|
89
|
-
// ---------------------------------------------------------------------------
|
|
90
|
-
// loadRoutingConfig
|
|
91
|
-
// ---------------------------------------------------------------------------
|
|
92
|
-
|
|
93
|
-
test("loadRoutingConfig returns null when no config exists", async () => {
|
|
94
|
-
const root = await makeAgentRoot();
|
|
95
|
-
try {
|
|
96
|
-
assert.equal(loadRoutingConfig(root), null);
|
|
97
|
-
} finally { await rmRoot(root); }
|
|
98
|
-
});
|
|
99
|
-
|
|
100
|
-
test("loadRoutingConfig parses JSON and synthesizes anthropic fallback", async () => {
|
|
101
|
-
const root = await makeAgentRoot();
|
|
102
|
-
try {
|
|
103
|
-
writeJsonConfig(root, {
|
|
104
|
-
backends: { moonshot: MOONSHOT_BACKEND },
|
|
105
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
106
|
-
});
|
|
107
|
-
const cfg = loadRoutingConfig(root);
|
|
108
|
-
assert.ok(cfg, "config should load");
|
|
109
|
-
// Anthropic synthesised because fallback_to_anthropic defaulted to true.
|
|
110
|
-
assert.ok(cfg.backends.anthropic, "anthropic synthesised");
|
|
111
|
-
assert.equal(cfg.backends.moonshot.base_url, "https://api.moonshot.ai/anthropic");
|
|
112
|
-
assert.equal(cfg.routing_policy[0].backend, "moonshot");
|
|
113
|
-
assert.equal(cfg.fallback_to_anthropic, true);
|
|
114
|
-
} finally { await rmRoot(root); }
|
|
115
|
-
});
|
|
116
|
-
|
|
117
|
-
test("loadRoutingConfig respects explicit fallback_to_anthropic: false", async () => {
|
|
118
|
-
const root = await makeAgentRoot();
|
|
119
|
-
try {
|
|
120
|
-
writeJsonConfig(root, {
|
|
121
|
-
backends: { moonshot: MOONSHOT_BACKEND },
|
|
122
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
123
|
-
fallback_to_anthropic: false,
|
|
124
|
-
});
|
|
125
|
-
const cfg = loadRoutingConfig(root);
|
|
126
|
-
assert.equal(cfg.backends.anthropic, undefined, "anthropic NOT synthesised");
|
|
127
|
-
assert.equal(cfg.fallback_to_anthropic, false);
|
|
128
|
-
} finally { await rmRoot(root); }
|
|
129
|
-
});
|
|
130
|
-
|
|
131
|
-
test("loadRoutingConfig throws on malformed JSON", async () => {
|
|
132
|
-
const root = await makeAgentRoot();
|
|
133
|
-
try {
|
|
134
|
-
writeFileSync(join(root, "config/model-routing.json"), "{ not valid json");
|
|
135
|
-
assert.throws(() => loadRoutingConfig(root), /not valid|JSON|parse/);
|
|
136
|
-
} finally { await rmRoot(root); }
|
|
137
|
-
});
|
|
138
|
-
|
|
139
|
-
test("loadRoutingConfig accepts $MAESTRO_ROUTING_CONFIG override", async () => {
|
|
140
|
-
const root = await makeAgentRoot();
|
|
141
|
-
const alt = await makeAgentRoot();
|
|
142
|
-
try {
|
|
143
|
-
const altPath = join(alt, "config/model-routing.json");
|
|
144
|
-
writeFileSync(altPath, JSON.stringify({
|
|
145
|
-
backends: { moonshot: MOONSHOT_BACKEND },
|
|
146
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
147
|
-
}));
|
|
148
|
-
const prev = process.env.MAESTRO_ROUTING_CONFIG;
|
|
149
|
-
process.env.MAESTRO_ROUTING_CONFIG = altPath;
|
|
150
|
-
try {
|
|
151
|
-
const cfg = loadRoutingConfig(root);
|
|
152
|
-
assert.ok(cfg.backends.moonshot);
|
|
153
|
-
} finally {
|
|
154
|
-
if (prev === undefined) delete process.env.MAESTRO_ROUTING_CONFIG;
|
|
155
|
-
else process.env.MAESTRO_ROUTING_CONFIG = prev;
|
|
156
|
-
}
|
|
157
|
-
} finally {
|
|
158
|
-
await rmRoot(root);
|
|
159
|
-
await rmRoot(alt);
|
|
160
|
-
}
|
|
161
|
-
});
|
|
162
|
-
|
|
163
|
-
// ---------------------------------------------------------------------------
|
|
164
|
-
// resolveBackend
|
|
165
|
-
// ---------------------------------------------------------------------------
|
|
166
|
-
|
|
167
|
-
test("resolveBackend returns null when no config is loaded", () => {
|
|
168
|
-
const r = resolveBackend({ agent_role: "responder" }, { config: null });
|
|
169
|
-
assert.equal(r, null, "no config → no resolution (callers preserve current behaviour)");
|
|
170
|
-
});
|
|
171
|
-
|
|
172
|
-
test("resolveBackend picks default rule when nothing else matches", () => {
|
|
173
|
-
const config = {
|
|
174
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
|
|
175
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
176
|
-
fallback_to_anthropic: true,
|
|
177
|
-
strip_attribution_header: true,
|
|
178
|
-
disable_experimental_betas: true,
|
|
179
|
-
};
|
|
180
|
-
const r = resolveBackend({ agent_role: "responder", tier: "default" }, { config, env: { MOONSHOT_API_KEY: "msk-test" } });
|
|
181
|
-
assert.equal(r.name, "moonshot");
|
|
182
|
-
assert.equal(r.model, "kimi-k2.6");
|
|
183
|
-
assert.equal(r.envForSpawn.ANTHROPIC_BASE_URL, "https://api.moonshot.ai/anthropic");
|
|
184
|
-
assert.equal(r.envForSpawn.ANTHROPIC_AUTH_TOKEN, "msk-test");
|
|
185
|
-
// Critical: API_KEY must be explicit empty string, not unset.
|
|
186
|
-
assert.equal(r.envForSpawn.ANTHROPIC_API_KEY, "");
|
|
187
|
-
assert.equal(r.envForSpawn.ANTHROPIC_MODEL, "kimi-k2.6");
|
|
188
|
-
assert.equal(r.envForSpawn.CLAUDE_CODE_ATTRIBUTION_HEADER, "0");
|
|
189
|
-
assert.equal(r.envForSpawn.CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS, "1");
|
|
190
|
-
});
|
|
191
|
-
|
|
192
|
-
test("resolveBackend forces anthropic for sensitive roles via agent_role_in", () => {
|
|
193
|
-
const config = {
|
|
194
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
|
|
195
|
-
routing_policy: [
|
|
196
|
-
{ match: { agent_role_in: ["ceo_pre_pass", "audit", "decision_writer"] }, backend: "anthropic" },
|
|
197
|
-
{ default: true, backend: "qwen" },
|
|
198
|
-
],
|
|
199
|
-
};
|
|
200
|
-
const sensitive = resolveBackend({ agent_role: "ceo_pre_pass" }, { config });
|
|
201
|
-
assert.equal(sensitive.name, "anthropic");
|
|
202
|
-
assert.equal(sensitive.transport, "anthropic-cli");
|
|
203
|
-
// anthropic-cli mode injects no env — current CLI behaviour preserved.
|
|
204
|
-
assert.deepEqual(sensitive.envForSpawn, {});
|
|
205
|
-
|
|
206
|
-
const routine = resolveBackend({ agent_role: "log_triage" }, { config, env: { OPENROUTER_API_KEY: "or-test" } });
|
|
207
|
-
assert.equal(routine.name, "qwen");
|
|
208
|
-
assert.equal(routine.envForSpawn.ANTHROPIC_AUTH_TOKEN, "or-test");
|
|
209
|
-
});
|
|
210
|
-
|
|
211
|
-
test("resolveBackend falls through when backend lacks capability (no_thinking)", () => {
|
|
212
|
-
const config = {
|
|
213
|
-
backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
|
|
214
|
-
routing_policy: [
|
|
215
|
-
{ match: { needs_thinking: true }, backend: "qwen", fallback: ["anthropic"] },
|
|
216
|
-
{ default: true, backend: "qwen" },
|
|
217
|
-
],
|
|
218
|
-
fallback_to_anthropic: true,
|
|
219
|
-
};
|
|
220
|
-
const r = resolveBackend({ agent_role: "responder", needs_thinking: true }, { config });
|
|
221
|
-
// Qwen has no `thinking` cap → falls back to anthropic.
|
|
222
|
-
assert.equal(r.name, "anthropic");
|
|
223
|
-
assert.equal(r.tried[0].backend, "qwen");
|
|
224
|
-
assert.equal(r.tried[0].reason, "no_thinking");
|
|
225
|
-
});
|
|
226
|
-
|
|
227
|
-
test("resolveBackend rejects degraded_tool_use when caller needs reliable tools", () => {
|
|
228
|
-
const config = {
|
|
229
|
-
backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
|
|
230
|
-
routing_policy: [
|
|
231
|
-
{ match: { needs_tool_use: true }, backend: "qwen", fallback: ["anthropic"] },
|
|
232
|
-
],
|
|
233
|
-
fallback_to_anthropic: true,
|
|
234
|
-
};
|
|
235
|
-
const r = resolveBackend({ agent_role: "responder", needs_tool_use: true }, { config });
|
|
236
|
-
assert.equal(r.name, "anthropic");
|
|
237
|
-
assert.equal(r.tried[0].backend, "qwen");
|
|
238
|
-
assert.equal(r.tried[0].reason, "degraded_tool_use");
|
|
239
|
-
});
|
|
240
|
-
|
|
241
|
-
test("resolveBackend honours token_estimate_gte for long-context routing", () => {
|
|
242
|
-
const config = {
|
|
243
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
|
|
244
|
-
routing_policy: [
|
|
245
|
-
{ match: { token_estimate_gte: 60000 }, backend: "moonshot" },
|
|
246
|
-
{ default: true, backend: "anthropic" },
|
|
247
|
-
],
|
|
248
|
-
};
|
|
249
|
-
const short = resolveBackend({ token_estimate: 20000 }, { config });
|
|
250
|
-
const long = resolveBackend({ token_estimate: 120000 }, { config, env: { MOONSHOT_API_KEY: "msk" } });
|
|
251
|
-
assert.equal(short.name, "anthropic");
|
|
252
|
-
assert.equal(long.name, "moonshot");
|
|
253
|
-
});
|
|
254
|
-
|
|
255
|
-
test("resolveBackend honours model_hint legacy values (opus/sonnet/haiku)", () => {
|
|
256
|
-
const config = {
|
|
257
|
-
backends: { anthropic: ANTHROPIC_BACKEND },
|
|
258
|
-
routing_policy: [{ default: true, backend: "anthropic" }],
|
|
259
|
-
};
|
|
260
|
-
const opus = resolveBackend({ model_hint: "opus" }, { config });
|
|
261
|
-
const sonnet = resolveBackend({ model_hint: "sonnet" }, { config });
|
|
262
|
-
const haiku = resolveBackend({ model_hint: "haiku" }, { config });
|
|
263
|
-
assert.equal(opus.model, "config-declared-opus");
|
|
264
|
-
assert.equal(sonnet.model, "config-declared-sonnet");
|
|
265
|
-
assert.equal(haiku.model, "claude-haiku-4-5-20251001");
|
|
266
|
-
assert.equal(modelFlagFor(opus, { model_hint: "opus" }), "opus");
|
|
267
|
-
assert.equal(modelFlagFor(sonnet, { model_hint: "sonnet" }), "sonnet");
|
|
268
|
-
assert.equal(modelFlagFor(haiku, { model_hint: "haiku" }), "haiku");
|
|
269
|
-
});
|
|
270
|
-
|
|
271
|
-
// ---------------------------------------------------------------------------
|
|
272
|
-
// Model freshness — the built-in Anthropic defaults (see the provenance block
|
|
273
|
-
// above ANTHROPIC_DEFAULT in model-router.mjs).
|
|
274
|
-
// ---------------------------------------------------------------------------
|
|
275
|
-
|
|
276
|
-
test("defaultRoutingConfig ships the CURRENT Anthropic ids, not last quarter's", () => {
|
|
277
|
-
// The floor an agent with no routing config lands on. A stale id here never
|
|
278
|
-
// reached the wire (anthropic-cli sends the "opus"/"sonnet" shorthand), but it
|
|
279
|
-
// DID reach resolved.model, which the cost ledger and telemetry store — so a
|
|
280
|
-
// stale id mis-prices and mis-attributes every unrouted session.
|
|
281
|
-
const models = defaultRoutingConfig().backends.anthropic.models;
|
|
282
|
-
assert.equal(models.premium, "claude-opus-5");
|
|
283
|
-
assert.equal(models.default, "claude-sonnet-5");
|
|
284
|
-
// The fast/classifier tier did NOT move in the 2026-08-13 refresh.
|
|
285
|
-
assert.equal(models.fast, "claude-haiku-4-5-20251001");
|
|
286
|
-
assert.equal(models.classifier, "claude-haiku-4-5-20251001");
|
|
287
|
-
// Guard the specific regression: no 4.x id survives anywhere in the map.
|
|
288
|
-
for (const [tier, id] of Object.entries(models)) {
|
|
289
|
-
assert.ok(!/^claude-(opus|sonnet)-4/.test(id), `${tier} still pinned to a retired id: ${id}`);
|
|
290
|
-
}
|
|
291
|
-
});
|
|
292
|
-
|
|
293
|
-
test("resolveBackend serves the current ids through the no-config default path", () => {
|
|
294
|
-
// End-to-end through the resolver, not just the constant: an agent with no
|
|
295
|
-
// config file (useDefault) resolves premium/default to the -5 pair.
|
|
296
|
-
const config = defaultRoutingConfig();
|
|
297
|
-
assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-5");
|
|
298
|
-
assert.equal(resolveBackend({ model_hint: "sonnet" }, { config }).model, "claude-sonnet-5");
|
|
299
|
-
// …while the CLI flag stays version-agnostic, which is WHY the stale ids were
|
|
300
|
-
// survivable on the wire and only corrupted the ledger.
|
|
301
|
-
assert.equal(modelFlagFor(resolveBackend({ model_hint: "opus" }, { config }), { model_hint: "opus" }), "opus");
|
|
302
|
-
});
|
|
303
|
-
|
|
304
|
-
test("a config-declared models map still outranks the built-in defaults (the no-PR refresh path)", () => {
|
|
305
|
-
// Refresh path #2 in the provenance block: when claude-opus-6 ships, a v1
|
|
306
|
-
// agent adds backends.anthropic.models to config/model-routing.yaml and is
|
|
307
|
-
// current WITHOUT an SDK release. normaliseConfig only injects the built-in
|
|
308
|
-
// map when backends.anthropic is absent, so this must win outright.
|
|
309
|
-
const config = {
|
|
310
|
-
backends: { anthropic: { ...ANTHROPIC_BACKEND, models: { ...ANTHROPIC_BACKEND.models, premium: "claude-opus-99" } } },
|
|
311
|
-
routing_policy: [{ default: true, backend: "anthropic" }],
|
|
312
|
-
};
|
|
313
|
-
assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-99");
|
|
314
|
-
});
|
|
315
|
-
|
|
316
|
-
test("resolveBackend returns null when no backend satisfies request and fallback disabled", () => {
|
|
317
|
-
const config = {
|
|
318
|
-
backends: { qwen: QWEN_BACKEND },
|
|
319
|
-
routing_policy: [{ default: true, backend: "qwen" }],
|
|
320
|
-
fallback_to_anthropic: false,
|
|
321
|
-
};
|
|
322
|
-
const r = resolveBackend({ needs_thinking: true }, { config });
|
|
323
|
-
assert.equal(r, null, "no fallback and no compatible backend → null");
|
|
324
|
-
});
|
|
325
|
-
|
|
326
|
-
test("resolveBackend tags fallback_reason when falling through to anthropic safety net", () => {
|
|
327
|
-
const config = {
|
|
328
|
-
backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
|
|
329
|
-
routing_policy: [{ default: true, backend: "qwen" }],
|
|
330
|
-
fallback_to_anthropic: true,
|
|
331
|
-
};
|
|
332
|
-
const r = resolveBackend({ needs_thinking: true }, { config });
|
|
333
|
-
assert.equal(r.name, "anthropic");
|
|
334
|
-
assert.equal(r.fallback_reason, "no_compatible_backend");
|
|
335
|
-
});
|
|
336
|
-
|
|
337
|
-
// ---------------------------------------------------------------------------
|
|
338
|
-
// estimateCost
|
|
339
|
-
// ---------------------------------------------------------------------------
|
|
340
|
-
|
|
341
|
-
test("estimateCost returns null without pricing data", () => {
|
|
342
|
-
const fake = { pricing: {} };
|
|
343
|
-
assert.equal(estimateCost({ token_estimate: 1000 }, fake), null);
|
|
344
|
-
});
|
|
345
|
-
|
|
346
|
-
test("estimateCost computes per-million-token cost without cache hits", () => {
|
|
347
|
-
const fake = { pricing: { input_per_m: 3.0, output_per_m: 15.0 } };
|
|
348
|
-
const cost = estimateCost({ token_estimate: 1_000_000, token_estimate_out: 500_000 }, fake);
|
|
349
|
-
// 1M input @ $3 + 0.5M output @ $15 = $3 + $7.5 = $10.5
|
|
350
|
-
assert.equal(cost, 10.5);
|
|
351
|
-
});
|
|
352
|
-
|
|
353
|
-
test("estimateCost honours cache_hit_per_m for Moonshot-style caching", () => {
|
|
354
|
-
const fake = { pricing: { input_per_m: 0.95, output_per_m: 4.0, cache_hit_per_m: 0.16 } };
|
|
355
|
-
// 1M tokens, 80% cache hit, no explicit output → defaults to 0.5M output
|
|
356
|
-
const cost = estimateCost({ token_estimate: 1_000_000, cache_hit_ratio: 0.8 }, fake);
|
|
357
|
-
// input: 200k @ 0.95 + 800k @ 0.16 = 0.19 + 0.128 = 0.318
|
|
358
|
-
// output: 500k @ 4.0 = 2.0
|
|
359
|
-
// total: 2.318
|
|
360
|
-
assert.ok(Math.abs(cost - 2.318) < 0.001, `got ${cost}`);
|
|
361
|
-
});
|
|
362
|
-
|
|
363
|
-
// ---------------------------------------------------------------------------
|
|
364
|
-
// listBackends / describeBackend
|
|
365
|
-
// ---------------------------------------------------------------------------
|
|
366
|
-
|
|
367
|
-
test("listBackends defaults to ['anthropic'] when no config", () => {
|
|
368
|
-
assert.deepEqual(listBackends({ config: null }), ["anthropic"]);
|
|
369
|
-
});
|
|
370
|
-
|
|
371
|
-
test("listBackends returns all configured backend keys", () => {
|
|
372
|
-
const config = {
|
|
373
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
|
|
374
|
-
routing_policy: [],
|
|
375
|
-
};
|
|
376
|
-
assert.deepEqual(listBackends({ config }).sort(), ["anthropic", "moonshot", "qwen"]);
|
|
377
|
-
});
|
|
378
|
-
|
|
379
|
-
test("describeBackend returns model/transport/pricing for a backend", () => {
|
|
380
|
-
const config = {
|
|
381
|
-
backends: { moonshot: MOONSHOT_BACKEND },
|
|
382
|
-
routing_policy: [],
|
|
383
|
-
};
|
|
384
|
-
const d = describeBackend("moonshot", { config });
|
|
385
|
-
assert.equal(d.name, "moonshot");
|
|
386
|
-
assert.equal(d.transport, "anthropic-native");
|
|
387
|
-
assert.equal(d.base_url, "https://api.moonshot.ai/anthropic");
|
|
388
|
-
assert.equal(d.models.default, "kimi-k2.6");
|
|
389
|
-
});
|
|
390
|
-
|
|
391
|
-
// ---------------------------------------------------------------------------
|
|
392
|
-
// requestFromClassifierResult
|
|
393
|
-
// ---------------------------------------------------------------------------
|
|
394
|
-
|
|
395
|
-
test("requestFromClassifierResult preserves priority + model hint", () => {
|
|
396
|
-
const req = requestFromClassifierResult(
|
|
397
|
-
{ priority: "critical", model: "opus", summary: "CEO request" },
|
|
398
|
-
{ role: "responder", source: "inbox" }
|
|
399
|
-
);
|
|
400
|
-
assert.equal(req.priority, "critical");
|
|
401
|
-
assert.equal(req.model_hint, "opus");
|
|
402
|
-
assert.equal(req.needs_thinking, true);
|
|
403
|
-
assert.equal(req.needs_tool_use, true);
|
|
404
|
-
assert.equal(req.agent_role, "responder");
|
|
405
|
-
});
|
|
406
|
-
|
|
407
|
-
test("requestFromClassifierResult does NOT set needs_thinking for routine sonnet items", () => {
|
|
408
|
-
const req = requestFromClassifierResult(
|
|
409
|
-
{ priority: "normal", model: "sonnet" },
|
|
410
|
-
{ role: "responder" }
|
|
411
|
-
);
|
|
412
|
-
assert.equal(req.needs_thinking, undefined);
|
|
413
|
-
assert.equal(req.model_hint, "sonnet");
|
|
414
|
-
});
|
|
415
|
-
|
|
416
|
-
// ---------------------------------------------------------------------------
|
|
417
|
-
// Foot-gun fixes (audit W5 / W8 + kill switch)
|
|
418
|
-
// ---------------------------------------------------------------------------
|
|
419
|
-
|
|
420
|
-
test("W5: session work defaults needs_tool_use=true even for routine items", () => {
|
|
421
|
-
// A normal-priority backlog item is still a full agent session — it must
|
|
422
|
-
// declare tool use so a degraded_tool_use backend is rejected.
|
|
423
|
-
const req = requestFromClassifierResult(
|
|
424
|
-
{ priority: "normal", model: "sonnet" },
|
|
425
|
-
{ role: "responder", source: "backlog" }
|
|
426
|
-
);
|
|
427
|
-
assert.equal(req.needs_tool_use, true, "session work needs tool use by default");
|
|
428
|
-
});
|
|
429
|
-
|
|
430
|
-
test("W5: one-shot lookups may opt out of needs_tool_use", () => {
|
|
431
|
-
const req = requestFromClassifierResult(
|
|
432
|
-
{ priority: "normal", model: "sonnet" },
|
|
433
|
-
{ role: "lookup", oneShot: true }
|
|
434
|
-
);
|
|
435
|
-
assert.equal(req.needs_tool_use, false, "one-shot lookups opt out");
|
|
436
|
-
});
|
|
437
|
-
|
|
438
|
-
test("W5: shipped-example backlog item is kept off the degraded_tool_use backend", () => {
|
|
439
|
-
// Mirrors scaffold/config/model-routing.yaml.example: a normal backlog item
|
|
440
|
-
// matched `source: backlog → openrouter_qwen` (degraded_tool_use). With
|
|
441
|
-
// needs_tool_use defaulting true, the capability gate now fires and the
|
|
442
|
-
// request falls through to a tool-capable backend.
|
|
443
|
-
const config = {
|
|
444
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
|
|
445
|
-
routing_policy: [
|
|
446
|
-
{ match: { source: "backlog" }, backend: "qwen", fallback: ["moonshot", "anthropic"] },
|
|
447
|
-
{ default: true, backend: "moonshot", fallback: ["anthropic"] },
|
|
448
|
-
],
|
|
449
|
-
fallback_to_anthropic: true,
|
|
450
|
-
};
|
|
451
|
-
const req = requestFromClassifierResult(
|
|
452
|
-
{ priority: "normal", model: "sonnet" },
|
|
453
|
-
{ role: "responder", source: "backlog" }
|
|
454
|
-
);
|
|
455
|
-
const r = resolveBackend(req, { config, env: { MOONSHOT_API_KEY: "msk" } });
|
|
456
|
-
assert.notEqual(r.name, "qwen", "must NOT route a tool-using session to degraded_tool_use backend");
|
|
457
|
-
assert.equal(r.name, "moonshot");
|
|
458
|
-
assert.equal(r.tried[0].backend, "qwen");
|
|
459
|
-
assert.equal(r.tried[0].reason, "degraded_tool_use");
|
|
460
|
-
});
|
|
461
|
-
|
|
462
|
-
test("W8: candidate with missing auth env is skipped, not spawned with empty token", () => {
|
|
463
|
-
const config = {
|
|
464
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
|
|
465
|
-
routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
|
|
466
|
-
fallback_to_anthropic: true,
|
|
467
|
-
};
|
|
468
|
-
// MOONSHOT_API_KEY deliberately absent from env.
|
|
469
|
-
const r = resolveBackend({ agent_role: "responder", tier: "default" }, { config, env: {} });
|
|
470
|
-
assert.equal(r.name, "anthropic", "skips moonshot, lands on anthropic safety net");
|
|
471
|
-
assert.equal(r.tried[0].backend, "moonshot");
|
|
472
|
-
assert.equal(r.tried[0].reason, "missing_auth_env");
|
|
473
|
-
// The anthropic-cli safety net injects no empty credential.
|
|
474
|
-
assert.deepEqual(r.envForSpawn, {});
|
|
475
|
-
});
|
|
476
|
-
|
|
477
|
-
test("W8: empty/whitespace auth env counts as absent and is skipped", () => {
|
|
478
|
-
const config = {
|
|
479
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
|
|
480
|
-
routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
|
|
481
|
-
fallback_to_anthropic: true,
|
|
482
|
-
};
|
|
483
|
-
const r = resolveBackend({ agent_role: "responder" }, { config, env: { MOONSHOT_API_KEY: " " } });
|
|
484
|
-
assert.equal(r.name, "anthropic");
|
|
485
|
-
assert.equal(r.tried[0].reason, "missing_auth_env");
|
|
486
|
-
});
|
|
487
|
-
|
|
488
|
-
test("W8: all candidates missing auth + no fallback → null (never empty spawn)", () => {
|
|
489
|
-
const config = {
|
|
490
|
-
backends: { moonshot: MOONSHOT_BACKEND },
|
|
491
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
492
|
-
fallback_to_anthropic: false,
|
|
493
|
-
};
|
|
494
|
-
const r = resolveBackend({ agent_role: "responder" }, { config, env: {} });
|
|
495
|
-
assert.equal(r, null, "no creds anywhere and no fallback → no resolution");
|
|
496
|
-
});
|
|
497
|
-
|
|
498
|
-
test("W8: present auth env still resolves normally (no regression)", () => {
|
|
499
|
-
const config = {
|
|
500
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
|
|
501
|
-
routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
|
|
502
|
-
fallback_to_anthropic: true,
|
|
503
|
-
};
|
|
504
|
-
const r = resolveBackend({ agent_role: "responder" }, { config, env: { MOONSHOT_API_KEY: "msk-ok" } });
|
|
505
|
-
assert.equal(r.name, "moonshot");
|
|
506
|
-
assert.equal(r.envForSpawn.ANTHROPIC_AUTH_TOKEN, "msk-ok");
|
|
507
|
-
});
|
|
508
|
-
|
|
509
|
-
test("kill switch: MAESTRO_ROUTER_FORCE_ANTHROPIC=1 short-circuits to stock behaviour", () => {
|
|
510
|
-
const config = {
|
|
511
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
|
|
512
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
513
|
-
fallback_to_anthropic: true,
|
|
514
|
-
};
|
|
515
|
-
// Even with a valid Moonshot key and a default-moonshot policy, the kill
|
|
516
|
-
// switch returns null — the same path as no config, so callers preserve
|
|
517
|
-
// the legacy keychain-OAuth `claude --print` behaviour.
|
|
518
|
-
const r = resolveBackend(
|
|
519
|
-
{ agent_role: "responder" },
|
|
520
|
-
{ config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1", MOONSHOT_API_KEY: "msk" } }
|
|
521
|
-
);
|
|
522
|
-
assert.equal(r, null, "kill switch → null (stock Anthropic CLI behaviour)");
|
|
523
|
-
});
|
|
524
|
-
|
|
525
|
-
test("kill switch: accepts 'true' as well as '1'", () => {
|
|
526
|
-
const config = {
|
|
527
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
|
|
528
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
529
|
-
};
|
|
530
|
-
const r = resolveBackend(
|
|
531
|
-
{ agent_role: "responder" },
|
|
532
|
-
{ config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "true", MOONSHOT_API_KEY: "msk" } }
|
|
533
|
-
);
|
|
534
|
-
assert.equal(r, null);
|
|
535
|
-
});
|
|
536
|
-
|
|
537
|
-
test("kill switch: any other value does NOT short-circuit", () => {
|
|
538
|
-
const config = {
|
|
539
|
-
backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
|
|
540
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
541
|
-
};
|
|
542
|
-
const r = resolveBackend(
|
|
543
|
-
{ agent_role: "responder" },
|
|
544
|
-
{ config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "0", MOONSHOT_API_KEY: "msk" } }
|
|
545
|
-
);
|
|
546
|
-
assert.equal(r.name, "moonshot", "0/unset → routing still active");
|
|
547
|
-
});
|
|
548
|
-
|
|
549
|
-
// ---------------------------------------------------------------------------
|
|
550
|
-
// defaultRoutingConfig wiring (audit L21)
|
|
551
|
-
// ---------------------------------------------------------------------------
|
|
552
|
-
|
|
553
|
-
test("defaultRoutingConfig returns a usable Anthropic-only config", () => {
|
|
554
|
-
const cfg = defaultRoutingConfig();
|
|
555
|
-
assert.ok(cfg, "default config should be an object");
|
|
556
|
-
assert.ok(cfg.backends.anthropic, "default config has an anthropic backend");
|
|
557
|
-
assert.equal(cfg.backends.anthropic.transport, "anthropic-cli");
|
|
558
|
-
assert.equal(cfg.fallback_to_anthropic, true);
|
|
559
|
-
// resolveBackend must accept it and preserve current CLI behaviour (no env).
|
|
560
|
-
const r = resolveBackend({ agent_role: "responder" }, { config: cfg });
|
|
561
|
-
assert.equal(r.name, "anthropic");
|
|
562
|
-
assert.equal(r.transport, "anthropic-cli");
|
|
563
|
-
assert.deepEqual(r.envForSpawn, {}, "anthropic-cli default injects no env");
|
|
564
|
-
});
|
|
565
|
-
|
|
566
|
-
test("defaultRoutingConfig returns a fresh, mutation-safe copy each call", () => {
|
|
567
|
-
const a = defaultRoutingConfig();
|
|
568
|
-
const b = defaultRoutingConfig();
|
|
569
|
-
assert.notEqual(a, b, "each call returns a distinct object");
|
|
570
|
-
a.backends.anthropic.models.default = "MUTATED";
|
|
571
|
-
assert.notEqual(
|
|
572
|
-
b.backends.anthropic.models.default,
|
|
573
|
-
"MUTATED",
|
|
574
|
-
"mutating one copy must not affect another",
|
|
575
|
-
);
|
|
576
|
-
});
|
|
577
|
-
|
|
578
|
-
test("loadRoutingConfig still returns null with no config and no useDefault flag", async () => {
|
|
579
|
-
const root = await makeAgentRoot();
|
|
580
|
-
try {
|
|
581
|
-
assert.equal(loadRoutingConfig(root), null, "default contract preserved");
|
|
582
|
-
} finally { await rmRoot(root); }
|
|
583
|
-
});
|
|
584
|
-
|
|
585
|
-
test("loadRoutingConfig({ useDefault: true }) returns the default when no file exists", async () => {
|
|
586
|
-
const root = await makeAgentRoot();
|
|
587
|
-
try {
|
|
588
|
-
const cfg = loadRoutingConfig(root, { useDefault: true });
|
|
589
|
-
assert.ok(cfg, "useDefault yields a config instead of null");
|
|
590
|
-
assert.ok(cfg.backends.anthropic, "it is the anthropic default");
|
|
591
|
-
assert.equal(cfg.fallback_to_anthropic, true);
|
|
592
|
-
} finally { await rmRoot(root); }
|
|
593
|
-
});
|
|
594
|
-
|
|
595
|
-
test("loadRoutingConfig prefers a real file over the useDefault fallback", async () => {
|
|
596
|
-
const root = await makeAgentRoot();
|
|
597
|
-
try {
|
|
598
|
-
writeJsonConfig(root, {
|
|
599
|
-
backends: { moonshot: MOONSHOT_BACKEND },
|
|
600
|
-
routing_policy: [{ default: true, backend: "moonshot" }],
|
|
601
|
-
});
|
|
602
|
-
const cfg = loadRoutingConfig(root, { useDefault: true });
|
|
603
|
-
assert.ok(cfg.backends.moonshot, "real config wins over the default fallback");
|
|
604
|
-
} finally { await rmRoot(root); }
|
|
605
|
-
});
|
|
606
|
-
|
|
607
|
-
// ---------------------------------------------------------------------------
|
|
608
|
-
// Constants exported
|
|
609
|
-
// ---------------------------------------------------------------------------
|
|
610
|
-
|
|
611
|
-
test("CONFIG_RELATIVE_PATH points at config/model-routing.yaml", () => {
|
|
612
|
-
assert.equal(CONFIG_RELATIVE_PATH, "config/model-routing.yaml");
|
|
613
|
-
});
|
|
614
|
-
|
|
615
|
-
// ===========================================================================
|
|
616
|
-
// resolveChain — the v2 policy brain (additive; v1 above stays byte-compatible)
|
|
617
|
-
// ===========================================================================
|
|
618
|
-
|
|
619
|
-
// One shared bundled catalog snapshot (read-only) for the resolve tests. Loaded
|
|
620
|
-
// from the framework's own lib/model-router/catalog/*.yaml so the refs are real.
|
|
621
|
-
// This file lives at <root>/lib/, so the framework root is one dir up.
|
|
622
|
-
const MAESTRO_ROOT_FOR_TESTS = () =>
|
|
623
|
-
join(fileURLToPath(new URL(".", import.meta.url)), "..");
|
|
624
|
-
const CATALOG = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {});
|
|
625
|
-
|
|
626
|
-
const V2_CONFIG = Object.freeze({
|
|
627
|
-
schema_version: 2,
|
|
628
|
-
aliases: {
|
|
629
|
-
frontier: "anthropic/claude-opus-4-8",
|
|
630
|
-
default: "anthropic/claude-sonnet-4-6",
|
|
631
|
-
fast: "anthropic/claude-haiku-4-5",
|
|
632
|
-
cheap: "deepseek/deepseek-v4-flash",
|
|
633
|
-
"cheap-session": "moonshot/kimi-k2.6",
|
|
634
|
-
},
|
|
635
|
-
defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
|
|
636
|
-
backends: {
|
|
637
|
-
anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
|
|
638
|
-
deepseek: { allowed_data_classes: ["public"] },
|
|
639
|
-
moonshot: { allowed_data_classes: ["public"] },
|
|
640
|
-
},
|
|
641
|
-
routing_policy: [
|
|
642
|
-
{ match: { agent_role_in: ["regulatory", "audit"] }, chain: ["frontier", "default"], pin: true },
|
|
643
|
-
{ match: { task_class: "classify.inbox" }, harness: "direct", chain: ["fast", "cheap", "rules"], needs_tool_use: false },
|
|
644
|
-
{ match: { task_class: "lookup.gmail" }, harness: "direct", chain: ["fast", "cheap"], needs_tool_use: false },
|
|
645
|
-
{ match: { source: "backlog", data_class: "public" }, chain: ["cheap", "default"] },
|
|
646
|
-
{ match: { token_estimate_gte: 250000 }, chain: ["default", "frontier"] },
|
|
647
|
-
{ default: true, chain: ["default", "fast"] },
|
|
648
|
-
],
|
|
649
|
-
budget_ladder: {
|
|
650
|
-
"75": { downgrade_tiers: 1 },
|
|
651
|
-
"90": { force_alias: "cheap_or_fast" },
|
|
652
|
-
"100": { essential_only: true },
|
|
653
|
-
},
|
|
654
|
-
fallback_to_anthropic: true,
|
|
655
|
-
});
|
|
656
|
-
|
|
657
|
-
const CLOCK = () => 1_718_000_000_000;
|
|
658
|
-
function rc(req, extra = {}) {
|
|
659
|
-
return resolveChain(req, { catalog: CATALOG, config: V2_CONFIG, env: {}, now: CLOCK, ...extra });
|
|
660
|
-
}
|
|
661
|
-
|
|
662
|
-
// ── rule match + chain build ───────────────────────────────────────────────
|
|
663
|
-
|
|
664
|
-
test("resolveChain: default rule routes a session to the default workhorse", () => {
|
|
665
|
-
const d = rc({ task_class: "session.responder", source: "inbox", data_class: "sensitive", token_estimate: 5000 });
|
|
666
|
-
assert.equal(d.chosen.provider, "anthropic");
|
|
667
|
-
assert.equal(d.chosen.model, "claude-sonnet-4-6");
|
|
668
|
-
assert.equal(d.chosen.harness, "session");
|
|
669
|
-
assert.ok(d.decision_id, "every decision has an id");
|
|
670
|
-
assert.ok(d.explain && typeof d.explain === "string", "explain is mandatory");
|
|
671
|
-
});
|
|
672
|
-
|
|
673
|
-
test("resolveChain: first-match picks the regulatory pin rule (frontier)", () => {
|
|
674
|
-
const d = rc({ agent_role: "regulatory", task_class: "decision.writer", data_class: "sensitive", token_estimate: 1000 });
|
|
675
|
-
assert.equal(d.chosen.model, "claude-opus-4-8", "regulatory → frontier");
|
|
676
|
-
// The visible chain leads with the matched rule's first alias.
|
|
677
|
-
assert.equal(d.chain[0].ref, "anthropic/claude-opus-4-8");
|
|
678
|
-
});
|
|
679
|
-
|
|
680
|
-
test("resolveChain: classify.inbox resolves direct harness + tool-less", () => {
|
|
681
|
-
// Anthropic direct needs the key; with a key present, fast (haiku) is chosen.
|
|
682
|
-
const d = resolveChain(
|
|
683
|
-
{ task_class: "classify.inbox", source: "inbox", data_class: "public", token_estimate: 1500 },
|
|
684
|
-
{ catalog: CATALOG, config: V2_CONFIG, env: { ANTHROPIC_API_KEY: "ak" }, now: CLOCK }
|
|
685
|
-
);
|
|
686
|
-
assert.equal(d.chosen.harness, "direct");
|
|
687
|
-
assert.equal(d.chosen.model, "claude-haiku-4-5");
|
|
688
|
-
});
|
|
689
|
-
|
|
690
|
-
test("resolveChain: the chain[] carries the failover tail, not just the winner", () => {
|
|
691
|
-
const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 });
|
|
692
|
-
// default chain is [default, fast]; both Anthropic session rows survive.
|
|
693
|
-
const refs = d.chain.map((c) => c.ref);
|
|
694
|
-
assert.deepEqual(refs, ["anthropic/claude-sonnet-4-6", "anthropic/claude-haiku-4-5"]);
|
|
695
|
-
});
|
|
696
|
-
|
|
697
|
-
// ── credential gate (key-absence-as-enforcement, §7.1) ──────────────────────
|
|
698
|
-
|
|
699
|
-
test("resolveChain: a third-party row without its key is skipped (missing_credential)", () => {
|
|
700
|
-
// lookup.gmail is harness:direct, needs_tool_use:false so deepseek (grade B)
|
|
701
|
-
// clears the capability gate — the ONLY reason left to skip it is the absent
|
|
702
|
-
// DEEPSEEK_API_KEY (key-absence-as-enforcement, §7.1). Anthropic haiku direct
|
|
703
|
-
// also needs its key (absent) so the chain ends on the Anthropic safety net.
|
|
704
|
-
const d = rc({ task_class: "lookup.gmail", data_class: "public", token_estimate: 1000 });
|
|
705
|
-
const skipped = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
|
|
706
|
-
assert.ok(skipped, "deepseek recorded in tried[]");
|
|
707
|
-
assert.equal(skipped.reason, "missing_credential");
|
|
708
|
-
});
|
|
709
|
-
|
|
710
|
-
test("resolveChain: present third-party key still gated by grade for sessions (B<A)", () => {
|
|
711
|
-
// deepseek is grade B; a session needing tools requires grade A (§6.6), so even
|
|
712
|
-
// WITH the key it is skipped grade_too_low and we fall to Anthropic.
|
|
713
|
-
const d = resolveChain(
|
|
714
|
-
{ task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
|
|
715
|
-
{ catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
|
|
716
|
-
);
|
|
717
|
-
assert.equal(d.chosen.provider, "anthropic");
|
|
718
|
-
const skipped = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
|
|
719
|
-
assert.equal(skipped.reason, "grade_too_low");
|
|
720
|
-
});
|
|
721
|
-
|
|
722
|
-
test("resolveChain: grade-B row IS reachable on the direct (tool-less) lane", () => {
|
|
723
|
-
// lookup.gmail is harness:direct, needs_tool_use:false; deepseek (B) is fine.
|
|
724
|
-
// Anthropic haiku (direct) needs ANTHROPIC_API_KEY (absent) so it is skipped,
|
|
725
|
-
// landing on deepseek with its key present.
|
|
726
|
-
const d = resolveChain(
|
|
727
|
-
{ task_class: "lookup.gmail", data_class: "public", token_estimate: 1000 },
|
|
728
|
-
{ catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
|
|
729
|
-
);
|
|
730
|
-
assert.equal(d.chosen.provider, "deepseek");
|
|
731
|
-
assert.equal(d.chosen.harness, "direct");
|
|
732
|
-
});
|
|
733
|
-
|
|
734
|
-
// ── data_class gate (deny-by-default, §7.6) ─────────────────────────────────
|
|
735
|
-
|
|
736
|
-
test("resolveChain: sensitive data never routes to a public-only backend", () => {
|
|
737
|
-
// Even with the key AND a direct lane, deepseek (allowed: public) is denied for
|
|
738
|
-
// a sensitive request and recorded as data_class_denied.
|
|
739
|
-
const cfg = { ...V2_CONFIG, routing_policy: [{ match: { task_class: "lookup.gmail" }, harness: "direct", chain: ["cheap", "fast"], needs_tool_use: false }, { default: true, chain: ["default"] }] };
|
|
740
|
-
const d = resolveChain(
|
|
741
|
-
{ task_class: "lookup.gmail", data_class: "sensitive", token_estimate: 1000 },
|
|
742
|
-
{ catalog: CATALOG, config: cfg, env: { DEEPSEEK_API_KEY: "dk", ANTHROPIC_API_KEY: "ak" }, now: CLOCK }
|
|
743
|
-
);
|
|
744
|
-
assert.equal(d.chosen.provider, "anthropic", "sensitive falls back to anthropic");
|
|
745
|
-
const denied = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
|
|
746
|
-
assert.equal(denied.reason, "data_class_denied");
|
|
747
|
-
});
|
|
748
|
-
|
|
749
|
-
// ── breaker gate (health.isOpen) ────────────────────────────────────────────
|
|
750
|
-
|
|
751
|
-
test("resolveChain: an open breaker skips the candidate (breaker_open in tried[])", () => {
|
|
752
|
-
// Inject an isOpen that reports the sonnet model breaker open; chain falls to
|
|
753
|
-
// the next survivor (haiku).
|
|
754
|
-
const isOpen = (key) => ({
|
|
755
|
-
open: key === "anthropic:claude-sonnet-4-6",
|
|
756
|
-
until: 1_718_000_100_000,
|
|
757
|
-
reason: "rate_limit",
|
|
758
|
-
strikes: 1,
|
|
759
|
-
});
|
|
760
|
-
const d = resolveChain(
|
|
761
|
-
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
762
|
-
{ catalog: CATALOG, config: V2_CONFIG, env: {}, now: CLOCK, isOpen }
|
|
763
|
-
);
|
|
764
|
-
assert.equal(d.chosen.model, "claude-haiku-4-5", "skips open sonnet, lands on haiku");
|
|
765
|
-
const skipped = d.tried.find((t) => t.ref === "anthropic/claude-sonnet-4-6");
|
|
766
|
-
assert.equal(skipped.reason, "breaker_open");
|
|
767
|
-
});
|
|
768
|
-
|
|
769
|
-
// ── context gate ────────────────────────────────────────────────────────────
|
|
770
|
-
|
|
771
|
-
test("resolveChain: a too-large request skips a small-context row (context_overflow)", () => {
|
|
772
|
-
// haiku context_tokens is 190000; a 250k request matches the long-context rule
|
|
773
|
-
// [default, frontier] (both 950k) and never even tries a small row. To assert
|
|
774
|
-
// the gate directly, force a chain through haiku for a huge prompt.
|
|
775
|
-
const cfg = { ...V2_CONFIG, routing_policy: [{ default: true, chain: ["fast", "default"] }] };
|
|
776
|
-
const d = resolveChain(
|
|
777
|
-
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 500000 },
|
|
778
|
-
{ catalog: CATALOG, config: cfg, env: {}, now: CLOCK }
|
|
779
|
-
);
|
|
780
|
-
assert.equal(d.chosen.model, "claude-sonnet-4-6", "haiku too small → sonnet");
|
|
781
|
-
const skipped = d.tried.find((t) => t.ref === "anthropic/claude-haiku-4-5");
|
|
782
|
-
assert.equal(skipped.reason, "context_overflow");
|
|
783
|
-
});
|
|
784
|
-
|
|
785
|
-
test("resolveChain: long-context rule routes 250k+ to a 1M-ctx row", () => {
|
|
786
|
-
const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 300000 });
|
|
787
|
-
assert.equal(d.chosen.model, "claude-sonnet-4-6");
|
|
788
|
-
});
|
|
789
|
-
|
|
790
|
-
// ── envForSpawn (session retarget for Kimi/DeepSeek) ─────────────────────────
|
|
791
|
-
|
|
792
|
-
test("resolveChain: a third-party SESSION row builds the ANTHROPIC_BASE_URL retarget env", () => {
|
|
793
|
-
// Allow a grade override so kimi (B) can host a session, and route public work
|
|
794
|
-
// to it with the key present.
|
|
795
|
-
const cfg = {
|
|
796
|
-
...V2_CONFIG,
|
|
797
|
-
catalog_overrides: undefined,
|
|
798
|
-
routing_policy: [{ match: { source: "backlog" }, chain: ["cheap-session", "default"] }, { default: true, chain: ["default"] }],
|
|
799
|
-
};
|
|
800
|
-
// Override kimi grade to A via catalog_overrides so the session grade gate passes.
|
|
801
|
-
const cat = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
|
|
802
|
-
agentConfig: { catalog_overrides: [{ ref: "moonshot/kimi-k2.6", tool_reliability: "A" }] },
|
|
803
|
-
});
|
|
804
|
-
const d = resolveChain(
|
|
805
|
-
{ task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
|
|
806
|
-
{ catalog: cat, config: cfg, env: { MOONSHOT_API_KEY: "mk-test" }, now: CLOCK }
|
|
807
|
-
);
|
|
808
|
-
assert.equal(d.chosen.provider, "moonshot");
|
|
809
|
-
assert.equal(d.chosen.transport, "anthropic-native");
|
|
810
|
-
assert.equal(d.envForSpawn.ANTHROPIC_BASE_URL, "https://api.moonshot.ai/anthropic");
|
|
811
|
-
assert.equal(d.envForSpawn.ANTHROPIC_AUTH_TOKEN, "mk-test");
|
|
812
|
-
assert.equal(d.envForSpawn.ANTHROPIC_API_KEY, "", "empty string, NOT unset (keychain-fallthrough guard)");
|
|
813
|
-
assert.equal(d.envForSpawn.ANTHROPIC_MODEL, "kimi-k2.6");
|
|
814
|
-
});
|
|
815
|
-
|
|
816
|
-
test("resolveChain: an Anthropic session injects NO retarget env (stock CLI)", () => {
|
|
817
|
-
const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 });
|
|
818
|
-
assert.deepEqual(d.envForSpawn, {});
|
|
819
|
-
});
|
|
820
|
-
|
|
821
|
-
// ── budget ladder ────────────────────────────────────────────────────────────
|
|
822
|
-
|
|
823
|
-
test("resolveChain: band ≥75 downgrades one tier (default→fast) for non-pinned work", () => {
|
|
824
|
-
const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, budget_band: 75 });
|
|
825
|
-
assert.equal(d.chosen.model, "claude-haiku-4-5", "default downgraded to fast at band 75");
|
|
826
|
-
assert.match(d.explain, /band 75/);
|
|
827
|
-
});
|
|
828
|
-
|
|
829
|
-
test("resolveChain: a pinned rule is exempt from the budget ladder", () => {
|
|
830
|
-
// regulatory rule sets pin:true on the request indirectly via the rule pin flag;
|
|
831
|
-
// pass pin on the request to exercise the exemption path.
|
|
832
|
-
const d = rc({ agent_role: "regulatory", task_class: "decision", data_class: "sensitive", token_estimate: 1000, budget_band: 90, pin: { ref: "anthropic/claude-opus-4-8" } });
|
|
833
|
-
assert.equal(d.chosen.model, "claude-opus-4-8", "pin beats the band-90 downgrade");
|
|
834
|
-
});
|
|
835
|
-
|
|
836
|
-
// ── affinity pin ─────────────────────────────────────────────────────────────
|
|
837
|
-
|
|
838
|
-
test("resolveChain: a live affinity pin is moved to the chain head when it passes gates", () => {
|
|
839
|
-
const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, session_key: "thread-1", pin: { ref: "anthropic/claude-haiku-4-5" } });
|
|
840
|
-
assert.equal(d.chosen.model, "claude-haiku-4-5");
|
|
841
|
-
assert.equal(d.chain[0].ref, "anthropic/claude-haiku-4-5", "pin at head");
|
|
842
|
-
assert.match(d.explain, /pin /);
|
|
843
|
-
});
|
|
844
|
-
|
|
845
|
-
test("resolveChain: a pin that fails a gate emits pin_overridden and routes normally", () => {
|
|
846
|
-
// Pin a public-only deepseek for sensitive work → data_class_denied → overridden.
|
|
847
|
-
const d = resolveChain(
|
|
848
|
-
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, pin: { ref: "deepseek/deepseek-v4-flash" } },
|
|
849
|
-
{ catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
|
|
850
|
-
);
|
|
851
|
-
assert.equal(d.chosen.provider, "anthropic", "overridden pin falls to normal routing");
|
|
852
|
-
const overridden = d.tried.find((t) => String(t.reason).startsWith("pin_overridden"));
|
|
853
|
-
assert.ok(overridden, "pin_overridden recorded in tried[]");
|
|
854
|
-
});
|
|
855
|
-
|
|
856
|
-
// ── kill switch ──────────────────────────────────────────────────────────────
|
|
857
|
-
|
|
858
|
-
test("resolveChain: MAESTRO_ROUTER_FORCE_ANTHROPIC forces an Anthropic-only chain", () => {
|
|
859
|
-
const d = resolveChain(
|
|
860
|
-
{ task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
|
|
861
|
-
{ catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1", DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
|
|
862
|
-
);
|
|
863
|
-
assert.equal(d.chosen.provider, "anthropic");
|
|
864
|
-
assert.equal(d.audit.fallback_reason, "kill_switch");
|
|
865
|
-
assert.match(d.explain, /kill switch/);
|
|
866
|
-
// No third-party retarget env under the kill switch.
|
|
867
|
-
assert.deepEqual(d.envForSpawn, {});
|
|
868
|
-
});
|
|
869
|
-
|
|
870
|
-
test("resolveChain: kill switch picks frontier for critical/thinking work", () => {
|
|
871
|
-
const d = resolveChain(
|
|
872
|
-
{ task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
|
|
873
|
-
{ catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "true" }, now: CLOCK }
|
|
874
|
-
);
|
|
875
|
-
assert.equal(d.chosen.model, "claude-opus-4-8");
|
|
876
|
-
});
|
|
877
|
-
|
|
878
|
-
// ── inert / no-config collapse ───────────────────────────────────────────────
|
|
879
|
-
|
|
880
|
-
test("resolveChain: no v2 config collapses to the Anthropic safety net", () => {
|
|
881
|
-
const d = resolveChain(
|
|
882
|
-
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
883
|
-
{ catalog: CATALOG, config: null, env: {}, now: CLOCK }
|
|
884
|
-
);
|
|
885
|
-
assert.equal(d.chosen.provider, "anthropic");
|
|
886
|
-
assert.equal(d.audit.fallback_reason, "no_v2_config");
|
|
887
|
-
});
|
|
888
|
-
|
|
889
|
-
// ── safety-net model freshness (pickAnthropicRow's ordered preference) ───────
|
|
890
|
-
//
|
|
891
|
-
// The safety net fires on the three paths that have no chain to walk: the kill
|
|
892
|
-
// switch, "no v2 config", and "nothing survived the gates". It is the ONE place
|
|
893
|
-
// resolve.mjs still names Anthropic model ids, so it is the one place that can
|
|
894
|
-
// go stale. lookupModel is an exact byRef hit (no replaced_by chasing), so the
|
|
895
|
-
// preference list carries the current -5 ids AND the 4-x succession fallback.
|
|
896
|
-
|
|
897
|
-
/** The bundled catalog plus synthetic claude-*-5 rows (what ships next). */
|
|
898
|
-
const CATALOG_WITH_5 = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
|
|
899
|
-
agentConfig: {
|
|
900
|
-
catalog_overrides: [
|
|
901
|
-
{
|
|
902
|
-
ref: "anthropic/claude-opus-5",
|
|
903
|
-
status: "available",
|
|
904
|
-
context_tokens: 950000,
|
|
905
|
-
max_tokens: 128000,
|
|
906
|
-
tool_reliability: "A",
|
|
907
|
-
harness: { session: true, direct: true, batch: true },
|
|
908
|
-
cost: { input: 5.0, output: 25.0, cache_read: 0.5, cache_write: 10.0 },
|
|
909
|
-
},
|
|
910
|
-
{
|
|
911
|
-
ref: "anthropic/claude-sonnet-5",
|
|
912
|
-
status: "available",
|
|
913
|
-
context_tokens: 950000,
|
|
914
|
-
max_tokens: 64000,
|
|
915
|
-
tool_reliability: "A",
|
|
916
|
-
harness: { session: true, direct: true, batch: true },
|
|
917
|
-
cost: { input: 3.0, output: 15.0, cache_read: 0.3, cache_write: 6.0 },
|
|
918
|
-
},
|
|
919
|
-
],
|
|
920
|
-
},
|
|
921
|
-
});
|
|
922
|
-
|
|
923
|
-
test("resolveChain: the safety net prefers claude-opus-5 / claude-sonnet-5 when the catalog has them", () => {
|
|
924
|
-
const frontier = resolveChain(
|
|
925
|
-
{ task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
|
|
926
|
-
{ catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
927
|
-
);
|
|
928
|
-
assert.equal(frontier.chosen.model, "claude-opus-5", "thinking/critical → current frontier, not opus-4-8");
|
|
929
|
-
|
|
930
|
-
const workhorse = resolveChain(
|
|
931
|
-
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
932
|
-
{ catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
933
|
-
);
|
|
934
|
-
assert.equal(workhorse.chosen.model, "claude-sonnet-5", "ordinary work → current workhorse, not sonnet-4-6");
|
|
935
|
-
});
|
|
936
|
-
|
|
937
|
-
test("resolveChain: the safety net falls back through succession when the catalog has no -5 rows", () => {
|
|
938
|
-
// This is why the 4-x refs stay in the preference list. Delete them and this
|
|
939
|
-
// does NOT fail loudly — it degrades to `models.find()` over the bundled rows,
|
|
940
|
-
// i.e. whatever sits first in anthropic.yaml (opus-4-8), which would hand
|
|
941
|
-
// ORDINARY work the frontier model and quietly quadruple its cost.
|
|
942
|
-
const d = resolveChain(
|
|
943
|
-
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
944
|
-
{ catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
|
|
945
|
-
);
|
|
946
|
-
assert.equal(d.chosen.model, "claude-sonnet-4-6", "succession fallback, ordered — NOT the first row in the file");
|
|
947
|
-
// Flip this to claude-sonnet-5 (and drop the 4-x refs from pickAnthropicRow)
|
|
948
|
-
// the day lib/model-router/catalog/anthropic.yaml carries the -5 rows.
|
|
949
|
-
});
|
|
950
|
-
|
|
951
|
-
test("the SHIPPED catalog is what decides the wire value — the -5 refresh is NOT done", () => {
|
|
952
|
-
// THE HONEST STATE OF THIS REFRESH, pinned so it cannot be mistaken for
|
|
953
|
-
// finished. The test above asserts `chosen.model`, which reads as bookkeeping.
|
|
954
|
-
// This one asserts `spawnArgs.modelFlag` — the string spawn.mjs pushes as
|
|
955
|
-
// `--model` — against the catalog the SDK actually ships, with `config: null`,
|
|
956
|
-
// which is the DEFAULT state of every agent because this repo contains no
|
|
957
|
-
// config/model-routing.yaml at all.
|
|
958
|
-
//
|
|
959
|
-
// Unlike the v1 lane (model-router.mjs's modelFlagFor sends the version-
|
|
960
|
-
// agnostic "opus"/"sonnet" shorthand), resolve.mjs's modelFlagFor returns
|
|
961
|
-
// row.id verbatim. So on v2 a stale catalog row is a stale id ON THE WIRE, not
|
|
962
|
-
// merely a mis-stamped ledger entry.
|
|
963
|
-
//
|
|
964
|
-
// WHEN THIS FAILS: someone added the -5 rows to anthropic.yaml. Good — that is
|
|
965
|
-
// the missing step. Update the expectations here, and DELETE the 4-x refs from
|
|
966
|
-
// pickAnthropicRow's preference lists in the same change, or the museum the
|
|
967
|
-
// provenance block warns about starts accumulating.
|
|
968
|
-
const cases = [
|
|
969
|
-
["session.responder", {}, "claude-sonnet-4-6"],
|
|
970
|
-
["decision", { needs_thinking: true }, "claude-opus-4-8"],
|
|
971
|
-
];
|
|
972
|
-
for (const [task_class, extra, expected] of cases) {
|
|
973
|
-
const req = { task_class, data_class: "sensitive", token_estimate: 1000, ...extra };
|
|
974
|
-
for (const [label, env] of [
|
|
975
|
-
["kill switch", { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }],
|
|
976
|
-
["no v2 config", {}],
|
|
977
|
-
]) {
|
|
978
|
-
const d = resolveChain(req, { catalog: CATALOG, config: null, env, now: CLOCK });
|
|
979
|
-
assert.equal(
|
|
980
|
-
d.spawnArgs.modelFlag,
|
|
981
|
-
expected,
|
|
982
|
-
`${label}/${task_class}: --model is the literal catalog id, and the bundled catalog is still 4-x`
|
|
983
|
-
);
|
|
984
|
-
// Not the shorthand: nothing downstream re-resolves this to a current build.
|
|
985
|
-
assert.ok(!["opus", "sonnet", "haiku"].includes(d.spawnArgs.modelFlag));
|
|
986
|
-
}
|
|
987
|
-
}
|
|
988
|
-
});
|
|
989
|
-
|
|
990
|
-
test("resolveChain: with NO catalog rows at all, the synthetic row's id and ref agree", () => {
|
|
991
|
-
// The last-ditch path: no catalog, so the id goes straight onto `claude
|
|
992
|
-
// --model`. It used to emit {model: "claude-opus-4-8", ref:
|
|
993
|
-
// "anthropic/claude-sonnet-4-6"} for a thinking request — and ref is what
|
|
994
|
-
// feeds cacheAffinityKey and chain[0].ref, so the pin and the audit trail
|
|
995
|
-
// both named a different model than the one that ran.
|
|
996
|
-
const emptyDir = join(tmpdir(), `model-router-empty-catalog-${process.pid}-${Date.now()}`);
|
|
997
|
-
mkdirSync(emptyDir, { recursive: true });
|
|
998
|
-
try {
|
|
999
|
-
const EMPTY = loadCatalog(emptyDir, { bundledDir: emptyDir });
|
|
1000
|
-
assert.equal(EMPTY.models.length, 0, "fixture really is an empty catalog");
|
|
1001
|
-
|
|
1002
|
-
const d = resolveChain(
|
|
1003
|
-
{ task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
|
|
1004
|
-
{ catalog: EMPTY, config: null, env: {}, now: CLOCK }
|
|
1005
|
-
);
|
|
1006
|
-
assert.equal(d.chosen.model, "claude-opus-5");
|
|
1007
|
-
assert.equal(d.chain[0].ref, "anthropic/claude-opus-5", "ref tracks id");
|
|
1008
|
-
assert.match(d.cacheAffinityKey, /anthropic\/claude-opus-5$/);
|
|
1009
|
-
|
|
1010
|
-
const ordinary = resolveChain(
|
|
1011
|
-
{ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
|
|
1012
|
-
{ catalog: EMPTY, config: null, env: {}, now: CLOCK }
|
|
1013
|
-
);
|
|
1014
|
-
assert.equal(ordinary.chosen.model, "claude-sonnet-5");
|
|
1015
|
-
assert.equal(ordinary.chain[0].ref, "anthropic/claude-sonnet-5");
|
|
1016
|
-
} finally {
|
|
1017
|
-
rmSync(emptyDir, { recursive: true, force: true });
|
|
1018
|
-
}
|
|
1019
|
-
});
|
|
1020
|
-
|
|
1021
|
-
// ── cheap tier: an UNBUNDLED provider (xAI/Grok) is routable end to end ──────
|
|
1022
|
-
//
|
|
1023
|
-
// The SDK bundles catalog rows for anthropic/deepseek/moonshot/qwen only. Grok
|
|
1024
|
-
// is NOT bundled — but "not bundled" is not "not routable": catalog.mjs's
|
|
1025
|
-
// highest authority layer (config/model-routing.yaml `catalog_overrides:`) may
|
|
1026
|
-
// introduce a whole provider, and resolveChain gates it exactly like a bundled
|
|
1027
|
-
// one. These tests prove the full path — row → alias → chain → credential gate
|
|
1028
|
-
// → chosen — so the remaining work is a bundled YAML file, not plumbing.
|
|
1029
|
-
|
|
1030
|
-
const XAI_PROVIDER_DOC = {
|
|
1031
|
-
provider: "xai",
|
|
1032
|
-
auth_env: "XAI_API_KEY",
|
|
1033
|
-
endpoints: { openai: "https://api.x.ai/v1", anthropic: "https://api.x.ai/anthropic" },
|
|
1034
|
-
data_residency: "us",
|
|
1035
|
-
models: [
|
|
1036
|
-
{
|
|
1037
|
-
id: "grok-4-fast",
|
|
1038
|
-
status: "available",
|
|
1039
|
-
context_window: 2000000,
|
|
1040
|
-
context_tokens: 1900000,
|
|
1041
|
-
max_tokens: 30000,
|
|
1042
|
-
cost: { input: 0.2, output: 0.5 },
|
|
1043
|
-
cost_provenance: { source: "unverified", fetched: "2026-08-13", volatile: true },
|
|
1044
|
-
compat: { thinking_format: "openai", cache_control: "implicit" },
|
|
1045
|
-
// Grade B like every other third-party row: good enough for the direct
|
|
1046
|
-
// lane, structurally barred from hosting a tool-using session (§6.6).
|
|
1047
|
-
tool_reliability: "B",
|
|
1048
|
-
harness: { session: true, direct: true, batch: false },
|
|
1049
|
-
},
|
|
1050
|
-
],
|
|
1051
|
-
};
|
|
1052
|
-
|
|
1053
|
-
const CATALOG_WITH_XAI = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
|
|
1054
|
-
agentConfig: { catalog_overrides: XAI_PROVIDER_DOC },
|
|
1055
|
-
});
|
|
1056
|
-
|
|
1057
|
-
const XAI_CONFIG = Object.freeze({
|
|
1058
|
-
schema_version: 2,
|
|
1059
|
-
aliases: { cheap: "xai/grok-4-fast", default: "anthropic/claude-sonnet-4-6" },
|
|
1060
|
-
defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
|
|
1061
|
-
backends: {
|
|
1062
|
-
anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
|
|
1063
|
-
xai: { allowed_data_classes: ["public"] },
|
|
1064
|
-
},
|
|
1065
|
-
routing_policy: [
|
|
1066
|
-
{ match: { task_class: "lookup.web" }, harness: "direct", chain: ["cheap", "default"], needs_tool_use: false },
|
|
1067
|
-
{ default: true, chain: ["default"] },
|
|
1068
|
-
],
|
|
1069
|
-
fallback_to_anthropic: true,
|
|
1070
|
-
});
|
|
1071
|
-
|
|
1072
|
-
test("catalog_overrides can introduce xAI/Grok — loudly, as a warning, never silently", () => {
|
|
1073
|
-
const row = CATALOG_WITH_XAI.byRef.get("xai/grok-4-fast");
|
|
1074
|
-
assert.ok(row, "grok row present after the agent-config merge");
|
|
1075
|
-
assert.equal(row.provider, "xai");
|
|
1076
|
-
assert.equal(row.auth_env, "XAI_API_KEY", "carries its OWN auth env, not a borrowed one");
|
|
1077
|
-
assert.equal(row.endpoints.anthropic, "https://api.x.ai/anthropic");
|
|
1078
|
-
assert.ok(
|
|
1079
|
-
CATALOG_WITH_XAI.warnings.some((w) => /non-bundled provider "xai"/.test(w.warning)),
|
|
1080
|
-
"introducing an unbundled provider warns (typo guard) — it does not fail"
|
|
1081
|
-
);
|
|
1082
|
-
});
|
|
1083
|
-
|
|
1084
|
-
test("resolveChain routes a cheap-tier task to Grok when its own key is present", () => {
|
|
1085
|
-
const d = resolveChain(
|
|
1086
|
-
{ task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
|
|
1087
|
-
{ catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
|
|
1088
|
-
);
|
|
1089
|
-
assert.equal(d.chosen.provider, "xai");
|
|
1090
|
-
assert.equal(d.chosen.model, "grok-4-fast");
|
|
1091
|
-
assert.equal(d.chosen.wire, "openai", "third-party direct call speaks the openai wire");
|
|
1092
|
-
});
|
|
1093
|
-
|
|
1094
|
-
test("resolveChain skips Grok without XAI_API_KEY (key-absence-as-enforcement, §7.1)", () => {
|
|
1095
|
-
const d = resolveChain(
|
|
1096
|
-
{ task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
|
|
1097
|
-
{ catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: {}, now: CLOCK }
|
|
1098
|
-
);
|
|
1099
|
-
const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
|
|
1100
|
-
assert.ok(skipped, "grok recorded in tried[]");
|
|
1101
|
-
assert.equal(skipped.reason, "missing_credential");
|
|
1102
|
-
assert.notEqual(d.chosen.provider, "xai");
|
|
1103
|
-
});
|
|
1104
|
-
|
|
1105
|
-
test("Grok inherits auth-profile key rotation from the catalog, with no per-provider code", () => {
|
|
1106
|
-
// auth-profiles derives provider → auth_env from the CATALOG, so a provider
|
|
1107
|
-
// the SDK has never heard of gets pooled multi-key rotation for free. This is
|
|
1108
|
-
// the "their own auth profiles" half of the cheap-tier requirement.
|
|
1109
|
-
const profiles = loadAuthProfiles(null, {
|
|
1110
|
-
configDoc: null, // bypass the FS
|
|
1111
|
-
env: { XAI_API_KEY: "k1", XAI_API_KEY_2: "k2", DEEPSEEK_API_KEY: "d1" },
|
|
1112
|
-
authEnvByProvider: authEnvMapFromCatalog(CATALOG_WITH_XAI),
|
|
1113
|
-
});
|
|
1114
|
-
assert.deepEqual(profiles.keysFor("xai"), ["k1", "k2"]);
|
|
1115
|
-
assert.equal(profiles.isPooled("xai"), true, "two keys ⇒ rotation is live");
|
|
1116
|
-
assert.equal(profiles.isPooled("deepseek"), false, "single key ⇒ unchanged single-key behaviour");
|
|
1117
|
-
});
|
|
1118
|
-
|
|
1119
|
-
test("a session on Grok is still barred by the grade gate, exactly like DeepSeek/Kimi", () => {
|
|
1120
|
-
// Not a bug to fix: §6.6 says a tool-using session needs grade A, and no
|
|
1121
|
-
// third-party row is graded A until the probe ledger earns it. Cheap-tier
|
|
1122
|
-
// reachability must NOT quietly become cheap-tier session hosting.
|
|
1123
|
-
const sessionCfg = {
|
|
1124
|
-
...XAI_CONFIG,
|
|
1125
|
-
routing_policy: [{ default: true, chain: ["cheap", "default"] }],
|
|
1126
|
-
};
|
|
1127
|
-
const d = resolveChain(
|
|
1128
|
-
{ task_class: "session.responder", data_class: "public", token_estimate: 2000 },
|
|
1129
|
-
{ catalog: CATALOG_WITH_XAI, config: sessionCfg, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
|
|
1130
|
-
);
|
|
1131
|
-
const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
|
|
1132
|
-
assert.equal(skipped?.reason, "grade_too_low");
|
|
1133
|
-
assert.equal(d.chosen.provider, "anthropic");
|
|
1134
|
-
});
|
|
1135
|
-
|
|
1136
|
-
// ── estCostUSD + ledger join fields ──────────────────────────────────────────
|
|
1137
|
-
|
|
1138
|
-
test("resolveChain: stamps a finite estCostUSD and a decision_id for the ledger join", () => {
|
|
1139
|
-
const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 100000, token_estimate_out: 20000 });
|
|
1140
|
-
assert.ok(typeof d.estCostUSD === "number" && d.estCostUSD > 0, "priced from the catalog row");
|
|
1141
|
-
assert.ok(/^[0-9a-z]{18}$/.test(d.decision_id), "ulid-ish decision_id");
|
|
1142
|
-
assert.equal(d.spawnArgs.bare, true);
|
|
1143
|
-
assert.ok(d.spawnArgs.maxTurns > 0);
|
|
1144
|
-
});
|
|
1145
|
-
|
|
1146
|
-
// ── deterministic "rules" fallback ───────────────────────────────────────────
|
|
1147
|
-
|
|
1148
|
-
test("resolveChain: falls to the deterministic rules member when no model survives", () => {
|
|
1149
|
-
// classify.inbox chain is [fast, cheap, rules]; with no keys at all, both
|
|
1150
|
-
// model rows are unreachable and the caller's deterministic fallback wins.
|
|
1151
|
-
const d = rc({ task_class: "classify.inbox", source: "inbox", data_class: "public", token_estimate: 1000 });
|
|
1152
|
-
assert.equal(d.chosen.provider, "rules");
|
|
1153
|
-
assert.equal(d.chosen.harness, "direct");
|
|
1154
|
-
});
|
|
1155
|
-
|
|
1156
|
-
// ===========================================================================
|
|
1157
|
-
// validateRoutingConfig — strict v2 config validation
|
|
1158
|
-
// ===========================================================================
|
|
1159
|
-
|
|
1160
|
-
test("validateRoutingConfig: a clean v2 config has zero errors", () => {
|
|
1161
|
-
const { errors } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, env: {} });
|
|
1162
|
-
assert.equal(errors.length, 0, JSON.stringify(errors));
|
|
1163
|
-
});
|
|
1164
|
-
|
|
1165
|
-
test("validateRoutingConfig: unknown match key is an error (capability typo)", () => {
|
|
1166
|
-
const cfg = { ...V2_CONFIG, routing_policy: [{ match: { taks_class: "x" }, chain: ["default"] }, { default: true, chain: ["default"] }] };
|
|
1167
|
-
const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
|
|
1168
|
-
assert.ok(errors.some((e) => /unknown match key "taks_class"/.test(e.error)));
|
|
1169
|
-
});
|
|
1170
|
-
|
|
1171
|
-
test("validateRoutingConfig: a chain ref not in the catalog is an error", () => {
|
|
1172
|
-
const cfg = { ...V2_CONFIG, aliases: { ...V2_CONFIG.aliases, ghost: "ghost/model-x" }, routing_policy: [{ default: true, chain: ["ghost"] }] };
|
|
1173
|
-
const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
|
|
1174
|
-
assert.ok(errors.some((e) => /not in catalog/.test(e.error)));
|
|
1175
|
-
});
|
|
1176
|
-
|
|
1177
|
-
test("validateRoutingConfig: a chain member that is neither alias nor ref is an error", () => {
|
|
1178
|
-
const cfg = { ...V2_CONFIG, routing_policy: [{ default: true, chain: ["notanalias"] }] };
|
|
1179
|
-
const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
|
|
1180
|
-
assert.ok(errors.some((e) => /neither a known alias nor/.test(e.error)));
|
|
1181
|
-
});
|
|
1182
|
-
|
|
1183
|
-
test("validateRoutingConfig: an unknown harness is an error", () => {
|
|
1184
|
-
const cfg = { ...V2_CONFIG, routing_policy: [{ match: { task_class: "x" }, harness: "telepathy", chain: ["default"] }, { default: true, chain: ["default"] }] };
|
|
1185
|
-
const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
|
|
1186
|
-
assert.ok(errors.some((e) => /unknown harness "telepathy"/.test(e.error)));
|
|
1187
|
-
});
|
|
1188
|
-
|
|
1189
|
-
test("validateRoutingConfig: an overlay-excluded backend in config is an error (tighten-only)", () => {
|
|
1190
|
-
const overlay = { version: "t1", constraints: { backends_allowed: ["anthropic", "deepseek"] } };
|
|
1191
|
-
// config declares a moonshot backend the overlay does not permit.
|
|
1192
|
-
const { errors } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, overlay });
|
|
1193
|
-
assert.ok(errors.some((e) => /not in the org overlay's backends_allowed/.test(e.error)));
|
|
1194
|
-
});
|
|
1195
|
-
|
|
1196
|
-
test("validateRoutingConfig: a missing third-party key is a WARNING, not an error", () => {
|
|
1197
|
-
const { errors, warnings } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, env: {} });
|
|
1198
|
-
assert.equal(errors.length, 0);
|
|
1199
|
-
assert.ok(warnings.some((w) => /DEEPSEEK_API_KEY/.test(w.warning)));
|
|
1200
|
-
// Anthropic's key is never warned (session rides keychain OAuth).
|
|
1201
|
-
assert.ok(!warnings.some((w) => /ANTHROPIC_API_KEY/.test(w.warning)));
|
|
1202
|
-
});
|
|
1203
|
-
|
|
1204
|
-
test("validateRoutingConfig: a v1 config is not strict-validated (returns clean)", () => {
|
|
1205
|
-
const { errors } = validateRoutingConfig({ backends: {}, routing_policy: [] }, { catalog: CATALOG });
|
|
1206
|
-
assert.equal(errors.length, 0);
|
|
1207
|
-
});
|