@cohortapp/agent-sdk 2.16.0 → 2.18.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (529) hide show
  1. package/.claude/settings.json +18 -0
  2. package/.env.example +23 -7
  3. package/README.md +1 -0
  4. package/bin/maestro.mjs +62 -0
  5. package/docs/guides/billing-console-keys.md +60 -0
  6. package/docs/guides/front-door-session.md +54 -9
  7. package/docs/guides/mac-mini.md +20 -25
  8. package/docs/guides/poller-daemon-setup.md +4 -1
  9. package/docs/guides/setup-wizard.md +1 -1
  10. package/docs/runbooks/fleet-rollout.md +156 -0
  11. package/docs/runbooks/mac-mini-bootstrap.md +12 -14
  12. package/lib/action-executor.js +19 -3
  13. package/lib/budget-guard.mjs +279 -3
  14. package/lib/channels/base-adapter.mjs +3 -1
  15. package/lib/channels/contract.mjs +2 -1
  16. package/lib/channels/inbox-item.mjs +8 -0
  17. package/lib/claude-bin.mjs +5 -6
  18. package/lib/cli/doctor-checks.mjs +141 -10
  19. package/lib/cli/global-setup-extras.mjs +5 -1
  20. package/lib/cli/inbox.mjs +100 -15
  21. package/lib/cli/seat-auth.mjs +463 -0
  22. package/lib/cli/session.mjs +80 -12
  23. package/lib/collective/capture.mjs +8 -6
  24. package/lib/collective/global-config.mjs +63 -1
  25. package/lib/collective/presence.mjs +142 -5
  26. package/lib/comms/send-gate.mjs +559 -1
  27. package/lib/context/budget.mjs +327 -0
  28. package/lib/context/history-scope.mjs +138 -0
  29. package/lib/diagnostics/alerts.mjs +49 -0
  30. package/lib/diagnostics/cadence-output-freshness.mjs +288 -0
  31. package/lib/engine/agents/definitions.mjs +343 -0
  32. package/lib/engine/agents/persist.mjs +275 -0
  33. package/lib/engine/agents/runtime.mjs +748 -0
  34. package/lib/engine/agents/usage.mjs +95 -0
  35. package/lib/engine/auth-status.mjs +139 -0
  36. package/lib/engine/budget.mjs +194 -0
  37. package/lib/engine/cli.mjs +1204 -0
  38. package/lib/engine/commands/index.mjs +269 -0
  39. package/lib/engine/context/budget.mjs +219 -0
  40. package/lib/engine/context/cache.mjs +125 -0
  41. package/lib/engine/context/child-env.mjs +215 -0
  42. package/lib/engine/context/compaction.mjs +342 -0
  43. package/lib/engine/context/images.mjs +90 -0
  44. package/lib/engine/context/instructions.mjs +327 -0
  45. package/lib/engine/context/lazy-instructions.mjs +169 -0
  46. package/lib/engine/context/manager.mjs +182 -0
  47. package/lib/engine/context/real-path.mjs +91 -0
  48. package/lib/engine/context/secret-values.mjs +163 -0
  49. package/lib/engine/context/settings.mjs +274 -0
  50. package/lib/engine/context/stream-input.mjs +159 -0
  51. package/lib/engine/guard.mjs +152 -0
  52. package/lib/engine/hooks.mjs +713 -0
  53. package/lib/engine/loop.mjs +560 -0
  54. package/lib/engine/mcp/client.mjs +254 -0
  55. package/lib/engine/mcp/config.mjs +301 -0
  56. package/lib/engine/mcp/http.mjs +201 -0
  57. package/lib/engine/mcp/index.mjs +146 -0
  58. package/lib/engine/mcp/jsonrpc.mjs +147 -0
  59. package/lib/engine/mcp/naming.mjs +66 -0
  60. package/lib/engine/mcp/resources.mjs +89 -0
  61. package/lib/engine/mcp/results.mjs +133 -0
  62. package/lib/engine/mcp/stdio.mjs +137 -0
  63. package/lib/engine/mcp/supervisor.mjs +116 -0
  64. package/lib/engine/messages.mjs +104 -0
  65. package/lib/engine/output/json.mjs +164 -0
  66. package/lib/engine/output/stream-json.mjs +266 -0
  67. package/lib/engine/permissions.mjs +845 -0
  68. package/lib/engine/process-identity.mjs +164 -0
  69. package/lib/engine/process-tree.mjs +551 -0
  70. package/lib/engine/prompt.mjs +60 -0
  71. package/lib/engine/session/store.mjs +299 -0
  72. package/lib/engine/session-runtime/args.mjs +97 -0
  73. package/lib/engine/session-runtime/host.mjs +143 -0
  74. package/lib/engine/session-runtime/inbox.mjs +122 -0
  75. package/lib/engine/session-runtime/notifications.mjs +129 -0
  76. package/lib/engine/session-runtime/registry.mjs +328 -0
  77. package/lib/engine/session-runtime/runner.mjs +344 -0
  78. package/lib/engine/session-runtime/socket.mjs +212 -0
  79. package/lib/engine/session-runtime/wakeup.mjs +115 -0
  80. package/lib/engine/skills/index.mjs +321 -0
  81. package/lib/engine/tools/bash-background.mjs +533 -0
  82. package/lib/engine/tools/bash.mjs +216 -0
  83. package/lib/engine/tools/edit.mjs +97 -0
  84. package/lib/engine/tools/glob.mjs +81 -0
  85. package/lib/engine/tools/grep.mjs +224 -0
  86. package/lib/engine/tools/index.mjs +84 -0
  87. package/lib/engine/tools/list-agents.mjs +32 -0
  88. package/lib/engine/tools/ls.mjs +127 -0
  89. package/lib/engine/tools/monitor.mjs +82 -0
  90. package/lib/engine/tools/notebook-edit.mjs +218 -0
  91. package/lib/engine/tools/read.mjs +103 -0
  92. package/lib/engine/tools/schedule-wakeup.mjs +45 -0
  93. package/lib/engine/tools/schema.mjs +144 -0
  94. package/lib/engine/tools/send-message.mjs +77 -0
  95. package/lib/engine/tools/session.mjs +70 -0
  96. package/lib/engine/tools/todo.mjs +144 -0
  97. package/lib/engine/tools/toolsearch.mjs +217 -0
  98. package/lib/engine/tools/walk.mjs +193 -0
  99. package/lib/engine/tools/web-switch.mjs +31 -0
  100. package/lib/engine/tools/webfetch-html.mjs +387 -0
  101. package/lib/engine/tools/webfetch-net.mjs +340 -0
  102. package/lib/engine/tools/webfetch.mjs +198 -0
  103. package/lib/engine/tools/websearch.mjs +91 -0
  104. package/lib/engine/tools/workflow.mjs +95 -0
  105. package/lib/engine/tools/write.mjs +76 -0
  106. package/lib/engine/tui/line-editor.mjs +137 -0
  107. package/lib/engine/tui/render.mjs +86 -0
  108. package/lib/engine/tui/tui.mjs +274 -0
  109. package/lib/engine/wire/anthropic-messages.mjs +263 -0
  110. package/lib/engine/wire/effort.mjs +36 -0
  111. package/lib/engine/wire/errors.mjs +496 -0
  112. package/lib/engine/wire/http.mjs +441 -0
  113. package/lib/engine/wire/index.mjs +76 -0
  114. package/lib/engine/wire/openai-chat.mjs +332 -0
  115. package/lib/engine/wire/prompt-cache.mjs +79 -0
  116. package/lib/engine/wire/search.mjs +140 -0
  117. package/lib/engine/wire/sse.mjs +114 -0
  118. package/lib/engine/wire/stall.mjs +349 -0
  119. package/lib/engine/wire/token-provider.mjs +175 -0
  120. package/lib/engine/wire/usage.mjs +192 -0
  121. package/lib/engine/workflow/host.mjs +524 -0
  122. package/lib/engine/workflow/journal.mjs +188 -0
  123. package/lib/engine/workflow/json-schema.mjs +171 -0
  124. package/lib/engine/workflow/meta.mjs +329 -0
  125. package/lib/engine/workflow/notifications.mjs +52 -0
  126. package/lib/engine/workflow/runtime.mjs +447 -0
  127. package/lib/engine/workflow/sandbox.mjs +534 -0
  128. package/lib/engine/workflow/worker.mjs +141 -0
  129. package/lib/engine/workflow/worktree.mjs +74 -0
  130. package/lib/execution/disposition.mjs +1 -1
  131. package/lib/execution/intake.mjs +10 -0
  132. package/lib/execution/surface-policy.mjs +15 -0
  133. package/lib/learning/curator.mjs +8 -6
  134. package/lib/learning/reflect.mjs +8 -6
  135. package/lib/model-router/catalog/cohort.yaml +137 -0
  136. package/lib/model-router/catalog.mjs +118 -1
  137. package/lib/model-router/economics.mjs +9 -0
  138. package/lib/model-router/failover.mjs +67 -16
  139. package/lib/model-router/llm-task.mjs +39 -3
  140. package/lib/model-router/resolve.mjs +95 -3
  141. package/lib/model-router/spawn.mjs +46 -47
  142. package/lib/model-router/taxonomy.mjs +126 -4
  143. package/lib/org/cost-sync.mjs +141 -11
  144. package/lib/org/inbound/broadcast.mjs +289 -0
  145. package/lib/org/inbound/collective.mjs +375 -0
  146. package/lib/org/inbound/directedness.mjs +96 -8
  147. package/lib/org/inbound/facts.mjs +82 -4
  148. package/lib/org/inbound/hydrate.mjs +555 -51
  149. package/lib/org/inbound/project.mjs +22 -0
  150. package/lib/org/inbound/surfaces.mjs +14 -0
  151. package/lib/org/llm-token.mjs +879 -0
  152. package/lib/org/mesh.mjs +61 -0
  153. package/lib/org/messaging.mjs +3 -1
  154. package/lib/org/protocol.checksum +1 -1
  155. package/lib/org/protocol.mjs +15 -0
  156. package/lib/org/quota.mjs +520 -0
  157. package/lib/org/tool-surface.mjs +104 -16
  158. package/lib/org/ui-parity.mjs +16 -1
  159. package/lib/org/work-ledger.mjs +37 -6
  160. package/lib/rate-guard.mjs +114 -1
  161. package/lib/resource-governor.mjs +41 -6
  162. package/lib/runtime/adapter.mjs +823 -0
  163. package/lib/runtime/child-env.mjs +191 -0
  164. package/lib/runtime/legacy-shell-guard.mjs +97 -0
  165. package/lib/runtime/seat-engine.mjs +162 -0
  166. package/lib/session/ask-ledger.mjs +271 -0
  167. package/lib/session/current-work.mjs +676 -0
  168. package/lib/session/feed-core.mjs +40 -3
  169. package/lib/session/launch-args.mjs +56 -4
  170. package/lib/session/status-summary.mjs +26 -9
  171. package/lib/session/upgrade-notice.mjs +42 -0
  172. package/lib/setup/claude-probe.mjs +117 -13
  173. package/lib/setup/enrich.mjs +13 -10
  174. package/lib/setup/sections/model.mjs +39 -13
  175. package/lib/telemetry/collect.mjs +208 -9
  176. package/lib/upgrade/ignored-drift.mjs +105 -0
  177. package/lib/voice/post-call-brief.mjs +30 -17
  178. package/package.json +15 -3
  179. package/plugins/maestro-skills/skills/board-work.md +5 -0
  180. package/plugins/maestro-skills/skills/inbound-triage.md +56 -15
  181. package/plugins/maestro-skills/skills/main-session.md +18 -7
  182. package/scripts/ci/check-tarball-fidelity.mjs +126 -2
  183. package/scripts/ci/run-tests.mjs +47 -19
  184. package/scripts/cohort-llm/api-key-helper.mjs +92 -0
  185. package/scripts/collective/hook-runner.mjs +29 -2
  186. package/scripts/continuous-monitor.sh +13 -0
  187. package/scripts/cost/track-claude-usage.mjs +15 -0
  188. package/scripts/daemon/agent-daemon.mjs +408 -20
  189. package/scripts/daemon/assurance.mjs +48 -12
  190. package/scripts/daemon/cadence-consumer.mjs +218 -68
  191. package/scripts/daemon/cadence-handlers.mjs +73 -4
  192. package/scripts/daemon/classifier.mjs +75 -26
  193. package/scripts/daemon/context-compiler.mjs +104 -59
  194. package/scripts/daemon/deliver.mjs +30 -1
  195. package/scripts/daemon/dispatcher.mjs +804 -157
  196. package/scripts/daemon/health.mjs +14 -1
  197. package/scripts/daemon/lib/session-router.mjs +310 -42
  198. package/scripts/daemon/maestro-daemon.mjs +11 -0
  199. package/scripts/daemon/prompt-builder.mjs +121 -12
  200. package/scripts/daemon/responder.mjs +315 -146
  201. package/scripts/daemon/sdk-version.mjs +98 -16
  202. package/scripts/eval/probe-gateway.mjs +635 -0
  203. package/scripts/eval/replay/extract.mjs +270 -0
  204. package/scripts/eval/replay/grade.mjs +260 -0
  205. package/scripts/eval/replay/lib/config.mjs +50 -0
  206. package/scripts/eval/replay/lib/effects.mjs +65 -0
  207. package/scripts/eval/replay/lib/fixture.mjs +188 -0
  208. package/scripts/eval/replay/lib/judge.mjs +72 -0
  209. package/scripts/eval/replay/lib/redact.mjs +136 -0
  210. package/scripts/eval/replay/lib/sandbox.mjs +170 -0
  211. package/scripts/eval/replay/lib/schema-check.mjs +63 -0
  212. package/scripts/eval/replay/lib/transcript.mjs +76 -0
  213. package/scripts/eval/replay/mcp-replay-stub.mjs +101 -0
  214. package/scripts/eval/replay/report.mjs +185 -0
  215. package/scripts/eval/replay/run.mjs +404 -0
  216. package/scripts/fleet/rollout.mjs +1094 -0
  217. package/scripts/hooks/pre-send-audit.sh +36 -245
  218. package/scripts/hooks/pre-write-yaml-validate.mjs +275 -0
  219. package/scripts/hooks/validate-state-yaml.sh +190 -0
  220. package/scripts/huddle/huddle-llm.mjs +361 -0
  221. package/scripts/huddle/huddle-server.mjs +46 -121
  222. package/scripts/local-triggers/autoupdate.sh +448 -78
  223. package/scripts/local-triggers/run-trigger.sh +13 -0
  224. package/scripts/maintenance/pin-integrity.mjs +364 -0
  225. package/scripts/poll-slack-events.sh +41 -9
  226. package/scripts/poller/slack-socket-mode.mjs +28 -3
  227. package/scripts/session/supervisor.mjs +80 -13
  228. package/scripts/spawn-session.sh +13 -0
  229. package/bin/maestro.test.mjs +0 -1574
  230. package/lib/action-executor.test.mjs +0 -871
  231. package/lib/archetype.test.mjs +0 -132
  232. package/lib/assurance/plan-note.test.mjs +0 -234
  233. package/lib/assurance/room-budget.test.mjs +0 -486
  234. package/lib/assurance/tier.test.mjs +0 -174
  235. package/lib/autonomy.test.mjs +0 -66
  236. package/lib/backlog.test.mjs +0 -302
  237. package/lib/backup/policy.test.mjs +0 -305
  238. package/lib/budget-escalate.test.mjs +0 -232
  239. package/lib/budget-guard.envelope.test.mjs +0 -476
  240. package/lib/budget-guard.test.mjs +0 -427
  241. package/lib/cadence-bus-requeue.test.mjs +0 -83
  242. package/lib/cadence-bus-schedule.test.mjs +0 -194
  243. package/lib/cadence-bus.test.mjs +0 -720
  244. package/lib/cadences.test.mjs +0 -230
  245. package/lib/capability/inventory.test.mjs +0 -232
  246. package/lib/capability.test.mjs +0 -78
  247. package/lib/channels/base-adapter.test.mjs +0 -590
  248. package/lib/channels/channels.test.mjs +0 -371
  249. package/lib/channels/contract.test.mjs +0 -162
  250. package/lib/channels/inbox-item.test.mjs +0 -368
  251. package/lib/channels/orgmail/adapter.test.mjs +0 -448
  252. package/lib/channels/pairing.test.mjs +0 -270
  253. package/lib/channels/repeat-suppressor.test.mjs +0 -134
  254. package/lib/channels/slack-adapter.test.mjs +0 -212
  255. package/lib/channels/telegram-adapter.test.mjs +0 -306
  256. package/lib/channels/voice/adapter.test.mjs +0 -278
  257. package/lib/channels/whatsapp/adapter-baileys.test.mjs +0 -359
  258. package/lib/channels/whatsapp/baileys-typing.test.mjs +0 -154
  259. package/lib/charter.test.mjs +0 -89
  260. package/lib/claude-bin.test.mjs +0 -131
  261. package/lib/cli/board.test.mjs +0 -227
  262. package/lib/cli/design.test.mjs +0 -270
  263. package/lib/cli/doctor-checks.test.mjs +0 -336
  264. package/lib/cli/global-setup-extras.test.mjs +0 -462
  265. package/lib/cli/inbox.test.mjs +0 -230
  266. package/lib/cli/session-ack.test.mjs +0 -63
  267. package/lib/cli/session.test.mjs +0 -613
  268. package/lib/collective/capture.test.mjs +0 -121
  269. package/lib/collective/cards.test.mjs +0 -114
  270. package/lib/collective/config.test.mjs +0 -123
  271. package/lib/collective/global-config.test.mjs +0 -220
  272. package/lib/collective/global-skills.test.mjs +0 -126
  273. package/lib/collective/presence.test.mjs +0 -95
  274. package/lib/collective/recall.test.mjs +0 -116
  275. package/lib/collective/vendor-skills.test.mjs +0 -306
  276. package/lib/comms/send-gate.test.mjs +0 -770
  277. package/lib/comms.test.mjs +0 -41
  278. package/lib/cost/ledger-row.test.mjs +0 -183
  279. package/lib/design/design-md.test.mjs +0 -318
  280. package/lib/design/fixtures/DESIGN.golden.md +0 -238
  281. package/lib/design/fixtures/PRODUCT.golden.md +0 -67
  282. package/lib/design/fixtures/foundation.json +0 -133
  283. package/lib/design/refresh-gate.test.mjs +0 -144
  284. package/lib/design/write.test.mjs +0 -241
  285. package/lib/diagnostics/alerts.test.mjs +0 -318
  286. package/lib/diagnostics/backup-freshness.test.mjs +0 -185
  287. package/lib/diagnostics/counters.test.mjs +0 -206
  288. package/lib/diagnostics/events.test.mjs +0 -290
  289. package/lib/diagnostics/otel.test.mjs +0 -196
  290. package/lib/diagnostics/trace.test.mjs +0 -251
  291. package/lib/env-compat.test.mjs +0 -104
  292. package/lib/execution/disposition.test.mjs +0 -553
  293. package/lib/execution/drive.test.mjs +0 -270
  294. package/lib/execution/effects.test.mjs +0 -344
  295. package/lib/execution/intake.test.mjs +0 -389
  296. package/lib/execution/journal.test.mjs +0 -261
  297. package/lib/execution/match.test.mjs +0 -235
  298. package/lib/execution/pipeline.test.mjs +0 -392
  299. package/lib/execution/route.test.mjs +0 -186
  300. package/lib/execution/surface-policy.test.mjs +0 -162
  301. package/lib/fs-atomic.test.mjs +0 -72
  302. package/lib/fs-ownership.test.mjs +0 -158
  303. package/lib/goals/admission.test.mjs +0 -164
  304. package/lib/goals/classify.test.mjs +0 -167
  305. package/lib/goals/collaborate.test.mjs +0 -336
  306. package/lib/goals/gaps.test.mjs +0 -284
  307. package/lib/goals/loop.test.mjs +0 -845
  308. package/lib/hooks/bus.test.mjs +0 -387
  309. package/lib/identity/persona.test.mjs +0 -142
  310. package/lib/kpi-sensors.test.mjs +0 -278
  311. package/lib/kpi.test.mjs +0 -244
  312. package/lib/learning/config.test.mjs +0 -75
  313. package/lib/learning/counters.test.mjs +0 -69
  314. package/lib/learning/curator-consolidate.test.mjs +0 -238
  315. package/lib/learning/curator.test.mjs +0 -106
  316. package/lib/learning/reflect.test.mjs +0 -0
  317. package/lib/learning/session-index.test.mjs +0 -125
  318. package/lib/learning/skill-writer.test.mjs +0 -210
  319. package/lib/mandate/audit.test.mjs +0 -195
  320. package/lib/mandate/contract.test.mjs +0 -185
  321. package/lib/mandate/derive.test.mjs +0 -274
  322. package/lib/mandate/model.test.mjs +0 -164
  323. package/lib/mandate/refresh.test.mjs +0 -389
  324. package/lib/mcp/server.test.mjs +0 -426
  325. package/lib/model-router/auth-profiles.test.mjs +0 -580
  326. package/lib/model-router/catalog.test.mjs +0 -385
  327. package/lib/model-router/economics.test.mjs +0 -438
  328. package/lib/model-router/failover.test.mjs +0 -439
  329. package/lib/model-router/health.test.mjs +0 -338
  330. package/lib/model-router/integration-coverage.test.mjs +0 -831
  331. package/lib/model-router/integration.test.mjs +0 -564
  332. package/lib/model-router/ledger.test.mjs +0 -415
  333. package/lib/model-router/llm-task.test.mjs +0 -392
  334. package/lib/model-router/org-credentials.test.mjs +0 -265
  335. package/lib/model-router/pricing-refresh.test.mjs +0 -286
  336. package/lib/model-router/reconcile.test.mjs +0 -316
  337. package/lib/model-router/repair.test.mjs +0 -180
  338. package/lib/model-router/spawn.test.mjs +0 -446
  339. package/lib/model-router/taxonomy.test.mjs +0 -410
  340. package/lib/model-router.test.mjs +0 -1207
  341. package/lib/org/activity.test.mjs +0 -134
  342. package/lib/org/approvals.test.mjs +0 -216
  343. package/lib/org/awareness.test.mjs +0 -159
  344. package/lib/org/board-mine-cache.test.mjs +0 -53
  345. package/lib/org/board.test.mjs +0 -187
  346. package/lib/org/bootstrap-context.test.mjs +0 -153
  347. package/lib/org/client.test.mjs +0 -1206
  348. package/lib/org/cohort-client.test.mjs +0 -126
  349. package/lib/org/cost-sync.test.mjs +0 -153
  350. package/lib/org/doctor.test.mjs +0 -346
  351. package/lib/org/engagement-ledger.test.mjs +0 -112
  352. package/lib/org/engagement.test.mjs +0 -739
  353. package/lib/org/handoff.test.mjs +0 -269
  354. package/lib/org/inbound/directedness.test.mjs +0 -668
  355. package/lib/org/inbound/facts.test.mjs +0 -471
  356. package/lib/org/inbound/hydrate.test.mjs +0 -453
  357. package/lib/org/inbound/index.test.mjs +0 -429
  358. package/lib/org/inbound/project.test.mjs +0 -287
  359. package/lib/org/integration-tools.test.mjs +0 -160
  360. package/lib/org/keys.test.mjs +0 -92
  361. package/lib/org/knowledge.test.mjs +0 -326
  362. package/lib/org/leases.test.mjs +0 -235
  363. package/lib/org/mesh-directives.test.mjs +0 -110
  364. package/lib/org/mesh-integration.test.mjs +0 -127
  365. package/lib/org/mesh.test.mjs +0 -400
  366. package/lib/org/messaging.test.mjs +0 -471
  367. package/lib/org/param-contract.test.mjs +0 -477
  368. package/lib/org/policy.test.mjs +0 -237
  369. package/lib/org/protocol.checksum.test.mjs +0 -90
  370. package/lib/org/protocol.test.mjs +0 -323
  371. package/lib/org/push.test.mjs +0 -792
  372. package/lib/org/registry.test.mjs +0 -100
  373. package/lib/org/resource-tools.test.mjs +0 -361
  374. package/lib/org/tool-access.test.mjs +0 -144
  375. package/lib/org/tool-surface-integration.test.mjs +0 -120
  376. package/lib/org/tool-surface.test.mjs +0 -1268
  377. package/lib/org/typing.test.mjs +0 -291
  378. package/lib/org/ui-parity.test.mjs +0 -560
  379. package/lib/org/verify.test.mjs +0 -194
  380. package/lib/org/work-ledger.test.mjs +0 -273
  381. package/lib/plan/adoption-e2e.test.mjs +0 -366
  382. package/lib/plan/budget-enforcement.test.mjs +0 -400
  383. package/lib/plan/compile.test.mjs +0 -382
  384. package/lib/plan/emit.test.mjs +0 -269
  385. package/lib/plan/explain.test.mjs +0 -188
  386. package/lib/prompts/parallelism.test.mjs +0 -177
  387. package/lib/rag/rag.test.mjs +0 -505
  388. package/lib/rate-guard.test.mjs +0 -272
  389. package/lib/reactive-gate.test.mjs +0 -57
  390. package/lib/render.test.mjs +0 -68
  391. package/lib/resource-governor.test.mjs +0 -488
  392. package/lib/scheduling/dynamic-jobs.test.mjs +0 -344
  393. package/lib/scheduling/jitter.test.mjs +0 -140
  394. package/lib/secrets/broker.test.mjs +0 -280
  395. package/lib/secrets/providers.test.mjs +0 -274
  396. package/lib/security/audit-engine.test.mjs +0 -424
  397. package/lib/security/coerce-args.test.mjs +0 -281
  398. package/lib/security/dangerous-tools.test.mjs +0 -68
  399. package/lib/security/external-content.test.mjs +0 -84
  400. package/lib/security/redact.test.mjs +0 -441
  401. package/lib/security/secret-equal.test.mjs +0 -55
  402. package/lib/session/config.test.mjs +0 -92
  403. package/lib/session/feed-core.test.mjs +0 -198
  404. package/lib/session/first-run.test.mjs +0 -121
  405. package/lib/session/frontdoor.test.mjs +0 -205
  406. package/lib/session/handoffs.test.mjs +0 -183
  407. package/lib/session/identity.test.mjs +0 -180
  408. package/lib/session/inbox-claims.test.mjs +0 -286
  409. package/lib/session/launch-args.test.mjs +0 -157
  410. package/lib/session/liveness.test.mjs +0 -100
  411. package/lib/session/status-summary.test.mjs +0 -118
  412. package/lib/session-permissions.test.mjs +0 -120
  413. package/lib/setup/claude-probe.test.mjs +0 -187
  414. package/lib/setup/completeness.test.mjs +0 -110
  415. package/lib/setup/context-pack.test.mjs +0 -89
  416. package/lib/setup/enrich.test.mjs +0 -115
  417. package/lib/setup/enroll-from-cohort.test.mjs +0 -300
  418. package/lib/setup/integration.test.mjs +0 -162
  419. package/lib/setup/io.test.mjs +0 -77
  420. package/lib/setup/runner.test.mjs +0 -132
  421. package/lib/setup/sections/identity.test.mjs +0 -234
  422. package/lib/setup/sections/inventory.test.mjs +0 -198
  423. package/lib/setup/sections/learning.test.mjs +0 -81
  424. package/lib/setup/sections/mandate.test.mjs +0 -388
  425. package/lib/setup/sections/messaging.test.mjs +0 -127
  426. package/lib/setup/sections/model.test.mjs +0 -240
  427. package/lib/setup/sections/org.test.mjs +0 -346
  428. package/lib/setup/sections/orgmail.test.mjs +0 -118
  429. package/lib/setup/sections/recovery.test.mjs +0 -98
  430. package/lib/setup/sections/subagents.test.mjs +0 -429
  431. package/lib/setup/sections/verify.test.mjs +0 -175
  432. package/lib/setup/sot.test.mjs +0 -81
  433. package/lib/setup/state.test.mjs +0 -115
  434. package/lib/singleton.test.mjs +0 -151
  435. package/lib/subagents/cli.test.mjs +0 -389
  436. package/lib/subagents/client.test.mjs +0 -309
  437. package/lib/subagents/gap.test.mjs +0 -234
  438. package/lib/subagents/lock.test.mjs +0 -248
  439. package/lib/subagents/manifest.test.mjs +0 -175
  440. package/lib/subagents/refs.test.mjs +0 -204
  441. package/lib/subagents/resolve.test.mjs +0 -422
  442. package/lib/subagents/schema.test.mjs +0 -328
  443. package/lib/telemetry/alerts.test.mjs +0 -109
  444. package/lib/telemetry/collect.test.mjs +0 -1274
  445. package/lib/tool-definitions-integration.test.mjs +0 -83
  446. package/lib/tool-definitions.test.mjs +0 -437
  447. package/lib/upgrade/global-refresh.test.mjs +0 -65
  448. package/lib/upgrade/launchd-reconcile.test.mjs +0 -272
  449. package/lib/upgrade/post-steps.test.mjs +0 -200
  450. package/lib/upgrade/verify.test.mjs +0 -164
  451. package/lib/util/fetch-timeout.test.mjs +0 -202
  452. package/lib/util/reconnect.test.mjs +0 -369
  453. package/lib/util/unhandled.test.mjs +0 -216
  454. package/lib/voice/outbound.test.mjs +0 -69
  455. package/lib/voice/session-rotation.test.mjs +0 -114
  456. package/lib/voice/stt.test.mjs +0 -226
  457. package/lib/voice/voice.test.mjs +0 -990
  458. package/scripts/cadence/enqueue-cadence-tick.test.mjs +0 -187
  459. package/scripts/ci/check-docs-accuracy.test.mjs +0 -409
  460. package/scripts/ci/check-durable-write-seam.test.mjs +0 -90
  461. package/scripts/ci/check-no-build-artifacts.test.mjs +0 -71
  462. package/scripts/ci/check-no-residual-identity.test.mjs +0 -202
  463. package/scripts/ci/check-skill-packs.test.mjs +0 -495
  464. package/scripts/ci/check-subagent-frontmatter.test.mjs +0 -124
  465. package/scripts/ci/check.test.mjs +0 -194
  466. package/scripts/ci/conformance-org-api.test.mjs +0 -425
  467. package/scripts/cloud-relay/voice/relay-identity.test.mjs +0 -96
  468. package/scripts/collective/hook-runner.test.mjs +0 -173
  469. package/scripts/cost/fleet-digest.test.mjs +0 -207
  470. package/scripts/cost/track-claude-usage-pricing.test.mjs +0 -183
  471. package/scripts/cost/track-claude-usage.test.mjs +0 -148
  472. package/scripts/daemon/agent-daemon-board-mine.test.mjs +0 -96
  473. package/scripts/daemon/agent-daemon-design.test.mjs +0 -238
  474. package/scripts/daemon/agent-daemon-frontdoor.test.mjs +0 -60
  475. package/scripts/daemon/agent-daemon.test.mjs +0 -995
  476. package/scripts/daemon/assurance-e2e.test.mjs +0 -613
  477. package/scripts/daemon/assurance.test.mjs +0 -1791
  478. package/scripts/daemon/board-mirror.test.mjs +0 -165
  479. package/scripts/daemon/cadence-consumer-frontdoor.test.mjs +0 -393
  480. package/scripts/daemon/cadence-consumer-governance.test.mjs +0 -276
  481. package/scripts/daemon/cadence-consumer.test.mjs +0 -776
  482. package/scripts/daemon/cadence-handlers.test.mjs +0 -837
  483. package/scripts/daemon/classifier-identity.test.mjs +0 -137
  484. package/scripts/daemon/classifier.test.mjs +0 -266
  485. package/scripts/daemon/classify-kind.test.mjs +0 -40
  486. package/scripts/daemon/context-compiler.test.mjs +0 -300
  487. package/scripts/daemon/deliver.test.mjs +0 -564
  488. package/scripts/daemon/dispatcher-cooldown.test.mjs +0 -122
  489. package/scripts/daemon/dispatcher-governance.test.mjs +0 -1013
  490. package/scripts/daemon/dispatcher-resume.test.mjs +0 -166
  491. package/scripts/daemon/execution-ladder.test.mjs +0 -470
  492. package/scripts/daemon/goal-steward-cadence.test.mjs +0 -312
  493. package/scripts/daemon/inbox-deferral-session.test.mjs +0 -49
  494. package/scripts/daemon/inbox-deferral.test.mjs +0 -336
  495. package/scripts/daemon/inbox-wake.test.mjs +0 -199
  496. package/scripts/daemon/integration.test.mjs +0 -149
  497. package/scripts/daemon/lib/self-echo.test.mjs +0 -153
  498. package/scripts/daemon/lib/session-router.test.mjs +0 -295
  499. package/scripts/daemon/prompt-builder-preamble.test.mjs +0 -210
  500. package/scripts/daemon/prompt-builder.test.mjs +0 -344
  501. package/scripts/daemon/responder-cost.test.mjs +0 -68
  502. package/scripts/daemon/responder-history.test.mjs +0 -185
  503. package/scripts/daemon/sdk-version.test.mjs +0 -31
  504. package/scripts/daemon/session-lock.test.mjs +0 -252
  505. package/scripts/daemon/session-outcomes.test.mjs +0 -533
  506. package/scripts/daemon/typing-registry.test.mjs +0 -102
  507. package/scripts/hooks/pre-send-audit.test.mjs +0 -354
  508. package/scripts/huddle/huddle-prompt.test.mjs +0 -176
  509. package/scripts/local-triggers/autoupdate.test.mjs +0 -518
  510. package/scripts/local-triggers/generate-plists.test.mjs +0 -456
  511. package/scripts/media-generation/brand-clause.test.mjs +0 -135
  512. package/scripts/org/send-orgmail.first-contact.test.mjs +0 -102
  513. package/scripts/poller/inbox-privilege-injection.test.mjs +0 -167
  514. package/scripts/poller/inbox-scan-poller.test.mjs +0 -295
  515. package/scripts/poller/lib/cloud-relay-dedup.test.mjs +0 -133
  516. package/scripts/poller/slack-socket-mode.test.mjs +0 -805
  517. package/scripts/poller-launchd/install.test.mjs +0 -243
  518. package/scripts/restore-from-backup.test.mjs +0 -181
  519. package/scripts/session/feed.test.mjs +0 -196
  520. package/scripts/session/supervisor-sh.test.mjs +0 -218
  521. package/scripts/session/supervisor.test.mjs +0 -482
  522. package/scripts/setup/configure-macos.test.mjs +0 -306
  523. package/scripts/setup/gen-subagent-manifest.test.mjs +0 -124
  524. package/scripts/setup/generate-agent-package-json.test.mjs +0 -143
  525. package/scripts/setup/generate-capability.test.mjs +0 -134
  526. package/scripts/setup/init-agent.test.mjs +0 -370
  527. package/scripts/setup/init-skill-marketplace.test.mjs +0 -193
  528. package/scripts/vendor/sync-skill-packs.test.mjs +0 -103
  529. package/scripts/watchdog/memory-watchdog.test.mjs +0 -64
@@ -1,1207 +0,0 @@
1
- /**
2
- * model-router.test.mjs — node:test coverage for the model router.
3
- *
4
- * Tests write configs to a temp dir + load them via loadRoutingConfig so
5
- * we exercise the full file-system path. No real network, no real Claude.
6
- */
7
-
8
- import { test } from "node:test";
9
- import assert from "node:assert/strict";
10
- import { promises as fsp } from "fs";
11
- import { mkdirSync, rmSync, writeFileSync } from "node:fs";
12
- import { tmpdir } from "os";
13
- import { join } from "path";
14
- import { fileURLToPath } from "node:url";
15
-
16
- import {
17
- loadRoutingConfig,
18
- defaultRoutingConfig,
19
- resolveBackend,
20
- estimateCost,
21
- listBackends,
22
- describeBackend,
23
- modelFlagFor,
24
- requestFromClassifierResult,
25
- CONFIG_RELATIVE_PATH,
26
- resolveChain,
27
- validateRoutingConfig,
28
- } from "./model-router.mjs";
29
- import { loadCatalog } from "./model-router/catalog.mjs";
30
- // Read-only, to PROVE (not assume) that a provider the SDK does not bundle
31
- // inherits credential pooling from the catalog rather than needing new code.
32
- import { loadAuthProfiles, authEnvMapFromCatalog } from "./model-router/auth-profiles.mjs";
33
-
34
- async function makeAgentRoot() {
35
- const path = join(
36
- tmpdir(),
37
- `model-router-test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`
38
- );
39
- await fsp.mkdir(join(path, "config"), { recursive: true });
40
- return path;
41
- }
42
-
43
- async function rmRoot(path) {
44
- try { await fsp.rm(path, { recursive: true, force: true }); } catch { /* */ }
45
- }
46
-
47
- function writeJsonConfig(root, body) {
48
- // We write JSON (not YAML) so the tests don't depend on js-yaml being
49
- // installed in CI. The loader treats .json identically.
50
- writeFileSync(join(root, "config/model-routing.json"), JSON.stringify(body, null, 2));
51
- }
52
-
53
- const ANTHROPIC_BACKEND = {
54
- transport: "anthropic-cli",
55
- base_url: "https://api.anthropic.com",
56
- capabilities: ["thinking", "vision", "prompt_cache_1h", "tool_use", "parallel_tools", "long_context_1m"],
57
- pricing: { input_per_m: 3.0, output_per_m: 15.0 },
58
- // A CONFIG-DECLARED model map. Deliberately NOT the same ids as
59
- // ANTHROPIC_DEFAULT: these tests assert that a config's own map is what
60
- // resolveBackend serves, so pinning them to the built-in defaults would make
61
- // the assertions vacuous (they would pass even if the config map were being
62
- // ignored entirely). The built-in defaults get their own test below.
63
- models: {
64
- classifier: "claude-haiku-4-5-20251001",
65
- default: "config-declared-sonnet",
66
- premium: "config-declared-opus",
67
- fast: "claude-haiku-4-5-20251001",
68
- },
69
- };
70
-
71
- const MOONSHOT_BACKEND = {
72
- transport: "anthropic-native",
73
- base_url: "https://api.moonshot.ai/anthropic",
74
- auth_env: "MOONSHOT_API_KEY",
75
- capabilities: ["tool_use", "prompt_cache_hit", "long_context_262k"],
76
- pricing: { input_per_m: 0.95, output_per_m: 4.0, cache_hit_per_m: 0.16 },
77
- models: { default: "kimi-k2.6", fast: "kimi-k2.6" },
78
- };
79
-
80
- const QWEN_BACKEND = {
81
- transport: "openai-compat",
82
- base_url: "https://openrouter.ai/api/v1",
83
- auth_env: "OPENROUTER_API_KEY",
84
- capabilities: ["tool_use", "long_context_262k", "degraded_tool_use"], // mimic real-world OR bug
85
- pricing: { input_per_m: 0.22, output_per_m: 1.8 },
86
- models: { default: "qwen/qwen3-coder", fast: "qwen/qwen3-coder-flash" },
87
- };
88
-
89
- // ---------------------------------------------------------------------------
90
- // loadRoutingConfig
91
- // ---------------------------------------------------------------------------
92
-
93
- test("loadRoutingConfig returns null when no config exists", async () => {
94
- const root = await makeAgentRoot();
95
- try {
96
- assert.equal(loadRoutingConfig(root), null);
97
- } finally { await rmRoot(root); }
98
- });
99
-
100
- test("loadRoutingConfig parses JSON and synthesizes anthropic fallback", async () => {
101
- const root = await makeAgentRoot();
102
- try {
103
- writeJsonConfig(root, {
104
- backends: { moonshot: MOONSHOT_BACKEND },
105
- routing_policy: [{ default: true, backend: "moonshot" }],
106
- });
107
- const cfg = loadRoutingConfig(root);
108
- assert.ok(cfg, "config should load");
109
- // Anthropic synthesised because fallback_to_anthropic defaulted to true.
110
- assert.ok(cfg.backends.anthropic, "anthropic synthesised");
111
- assert.equal(cfg.backends.moonshot.base_url, "https://api.moonshot.ai/anthropic");
112
- assert.equal(cfg.routing_policy[0].backend, "moonshot");
113
- assert.equal(cfg.fallback_to_anthropic, true);
114
- } finally { await rmRoot(root); }
115
- });
116
-
117
- test("loadRoutingConfig respects explicit fallback_to_anthropic: false", async () => {
118
- const root = await makeAgentRoot();
119
- try {
120
- writeJsonConfig(root, {
121
- backends: { moonshot: MOONSHOT_BACKEND },
122
- routing_policy: [{ default: true, backend: "moonshot" }],
123
- fallback_to_anthropic: false,
124
- });
125
- const cfg = loadRoutingConfig(root);
126
- assert.equal(cfg.backends.anthropic, undefined, "anthropic NOT synthesised");
127
- assert.equal(cfg.fallback_to_anthropic, false);
128
- } finally { await rmRoot(root); }
129
- });
130
-
131
- test("loadRoutingConfig throws on malformed JSON", async () => {
132
- const root = await makeAgentRoot();
133
- try {
134
- writeFileSync(join(root, "config/model-routing.json"), "{ not valid json");
135
- assert.throws(() => loadRoutingConfig(root), /not valid|JSON|parse/);
136
- } finally { await rmRoot(root); }
137
- });
138
-
139
- test("loadRoutingConfig accepts $MAESTRO_ROUTING_CONFIG override", async () => {
140
- const root = await makeAgentRoot();
141
- const alt = await makeAgentRoot();
142
- try {
143
- const altPath = join(alt, "config/model-routing.json");
144
- writeFileSync(altPath, JSON.stringify({
145
- backends: { moonshot: MOONSHOT_BACKEND },
146
- routing_policy: [{ default: true, backend: "moonshot" }],
147
- }));
148
- const prev = process.env.MAESTRO_ROUTING_CONFIG;
149
- process.env.MAESTRO_ROUTING_CONFIG = altPath;
150
- try {
151
- const cfg = loadRoutingConfig(root);
152
- assert.ok(cfg.backends.moonshot);
153
- } finally {
154
- if (prev === undefined) delete process.env.MAESTRO_ROUTING_CONFIG;
155
- else process.env.MAESTRO_ROUTING_CONFIG = prev;
156
- }
157
- } finally {
158
- await rmRoot(root);
159
- await rmRoot(alt);
160
- }
161
- });
162
-
163
- // ---------------------------------------------------------------------------
164
- // resolveBackend
165
- // ---------------------------------------------------------------------------
166
-
167
- test("resolveBackend returns null when no config is loaded", () => {
168
- const r = resolveBackend({ agent_role: "responder" }, { config: null });
169
- assert.equal(r, null, "no config → no resolution (callers preserve current behaviour)");
170
- });
171
-
172
- test("resolveBackend picks default rule when nothing else matches", () => {
173
- const config = {
174
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
175
- routing_policy: [{ default: true, backend: "moonshot" }],
176
- fallback_to_anthropic: true,
177
- strip_attribution_header: true,
178
- disable_experimental_betas: true,
179
- };
180
- const r = resolveBackend({ agent_role: "responder", tier: "default" }, { config, env: { MOONSHOT_API_KEY: "msk-test" } });
181
- assert.equal(r.name, "moonshot");
182
- assert.equal(r.model, "kimi-k2.6");
183
- assert.equal(r.envForSpawn.ANTHROPIC_BASE_URL, "https://api.moonshot.ai/anthropic");
184
- assert.equal(r.envForSpawn.ANTHROPIC_AUTH_TOKEN, "msk-test");
185
- // Critical: API_KEY must be explicit empty string, not unset.
186
- assert.equal(r.envForSpawn.ANTHROPIC_API_KEY, "");
187
- assert.equal(r.envForSpawn.ANTHROPIC_MODEL, "kimi-k2.6");
188
- assert.equal(r.envForSpawn.CLAUDE_CODE_ATTRIBUTION_HEADER, "0");
189
- assert.equal(r.envForSpawn.CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS, "1");
190
- });
191
-
192
- test("resolveBackend forces anthropic for sensitive roles via agent_role_in", () => {
193
- const config = {
194
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
195
- routing_policy: [
196
- { match: { agent_role_in: ["ceo_pre_pass", "audit", "decision_writer"] }, backend: "anthropic" },
197
- { default: true, backend: "qwen" },
198
- ],
199
- };
200
- const sensitive = resolveBackend({ agent_role: "ceo_pre_pass" }, { config });
201
- assert.equal(sensitive.name, "anthropic");
202
- assert.equal(sensitive.transport, "anthropic-cli");
203
- // anthropic-cli mode injects no env — current CLI behaviour preserved.
204
- assert.deepEqual(sensitive.envForSpawn, {});
205
-
206
- const routine = resolveBackend({ agent_role: "log_triage" }, { config, env: { OPENROUTER_API_KEY: "or-test" } });
207
- assert.equal(routine.name, "qwen");
208
- assert.equal(routine.envForSpawn.ANTHROPIC_AUTH_TOKEN, "or-test");
209
- });
210
-
211
- test("resolveBackend falls through when backend lacks capability (no_thinking)", () => {
212
- const config = {
213
- backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
214
- routing_policy: [
215
- { match: { needs_thinking: true }, backend: "qwen", fallback: ["anthropic"] },
216
- { default: true, backend: "qwen" },
217
- ],
218
- fallback_to_anthropic: true,
219
- };
220
- const r = resolveBackend({ agent_role: "responder", needs_thinking: true }, { config });
221
- // Qwen has no `thinking` cap → falls back to anthropic.
222
- assert.equal(r.name, "anthropic");
223
- assert.equal(r.tried[0].backend, "qwen");
224
- assert.equal(r.tried[0].reason, "no_thinking");
225
- });
226
-
227
- test("resolveBackend rejects degraded_tool_use when caller needs reliable tools", () => {
228
- const config = {
229
- backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
230
- routing_policy: [
231
- { match: { needs_tool_use: true }, backend: "qwen", fallback: ["anthropic"] },
232
- ],
233
- fallback_to_anthropic: true,
234
- };
235
- const r = resolveBackend({ agent_role: "responder", needs_tool_use: true }, { config });
236
- assert.equal(r.name, "anthropic");
237
- assert.equal(r.tried[0].backend, "qwen");
238
- assert.equal(r.tried[0].reason, "degraded_tool_use");
239
- });
240
-
241
- test("resolveBackend honours token_estimate_gte for long-context routing", () => {
242
- const config = {
243
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
244
- routing_policy: [
245
- { match: { token_estimate_gte: 60000 }, backend: "moonshot" },
246
- { default: true, backend: "anthropic" },
247
- ],
248
- };
249
- const short = resolveBackend({ token_estimate: 20000 }, { config });
250
- const long = resolveBackend({ token_estimate: 120000 }, { config, env: { MOONSHOT_API_KEY: "msk" } });
251
- assert.equal(short.name, "anthropic");
252
- assert.equal(long.name, "moonshot");
253
- });
254
-
255
- test("resolveBackend honours model_hint legacy values (opus/sonnet/haiku)", () => {
256
- const config = {
257
- backends: { anthropic: ANTHROPIC_BACKEND },
258
- routing_policy: [{ default: true, backend: "anthropic" }],
259
- };
260
- const opus = resolveBackend({ model_hint: "opus" }, { config });
261
- const sonnet = resolveBackend({ model_hint: "sonnet" }, { config });
262
- const haiku = resolveBackend({ model_hint: "haiku" }, { config });
263
- assert.equal(opus.model, "config-declared-opus");
264
- assert.equal(sonnet.model, "config-declared-sonnet");
265
- assert.equal(haiku.model, "claude-haiku-4-5-20251001");
266
- assert.equal(modelFlagFor(opus, { model_hint: "opus" }), "opus");
267
- assert.equal(modelFlagFor(sonnet, { model_hint: "sonnet" }), "sonnet");
268
- assert.equal(modelFlagFor(haiku, { model_hint: "haiku" }), "haiku");
269
- });
270
-
271
- // ---------------------------------------------------------------------------
272
- // Model freshness — the built-in Anthropic defaults (see the provenance block
273
- // above ANTHROPIC_DEFAULT in model-router.mjs).
274
- // ---------------------------------------------------------------------------
275
-
276
- test("defaultRoutingConfig ships the CURRENT Anthropic ids, not last quarter's", () => {
277
- // The floor an agent with no routing config lands on. A stale id here never
278
- // reached the wire (anthropic-cli sends the "opus"/"sonnet" shorthand), but it
279
- // DID reach resolved.model, which the cost ledger and telemetry store — so a
280
- // stale id mis-prices and mis-attributes every unrouted session.
281
- const models = defaultRoutingConfig().backends.anthropic.models;
282
- assert.equal(models.premium, "claude-opus-5");
283
- assert.equal(models.default, "claude-sonnet-5");
284
- // The fast/classifier tier did NOT move in the 2026-08-13 refresh.
285
- assert.equal(models.fast, "claude-haiku-4-5-20251001");
286
- assert.equal(models.classifier, "claude-haiku-4-5-20251001");
287
- // Guard the specific regression: no 4.x id survives anywhere in the map.
288
- for (const [tier, id] of Object.entries(models)) {
289
- assert.ok(!/^claude-(opus|sonnet)-4/.test(id), `${tier} still pinned to a retired id: ${id}`);
290
- }
291
- });
292
-
293
- test("resolveBackend serves the current ids through the no-config default path", () => {
294
- // End-to-end through the resolver, not just the constant: an agent with no
295
- // config file (useDefault) resolves premium/default to the -5 pair.
296
- const config = defaultRoutingConfig();
297
- assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-5");
298
- assert.equal(resolveBackend({ model_hint: "sonnet" }, { config }).model, "claude-sonnet-5");
299
- // …while the CLI flag stays version-agnostic, which is WHY the stale ids were
300
- // survivable on the wire and only corrupted the ledger.
301
- assert.equal(modelFlagFor(resolveBackend({ model_hint: "opus" }, { config }), { model_hint: "opus" }), "opus");
302
- });
303
-
304
- test("a config-declared models map still outranks the built-in defaults (the no-PR refresh path)", () => {
305
- // Refresh path #2 in the provenance block: when claude-opus-6 ships, a v1
306
- // agent adds backends.anthropic.models to config/model-routing.yaml and is
307
- // current WITHOUT an SDK release. normaliseConfig only injects the built-in
308
- // map when backends.anthropic is absent, so this must win outright.
309
- const config = {
310
- backends: { anthropic: { ...ANTHROPIC_BACKEND, models: { ...ANTHROPIC_BACKEND.models, premium: "claude-opus-99" } } },
311
- routing_policy: [{ default: true, backend: "anthropic" }],
312
- };
313
- assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-99");
314
- });
315
-
316
- test("resolveBackend returns null when no backend satisfies request and fallback disabled", () => {
317
- const config = {
318
- backends: { qwen: QWEN_BACKEND },
319
- routing_policy: [{ default: true, backend: "qwen" }],
320
- fallback_to_anthropic: false,
321
- };
322
- const r = resolveBackend({ needs_thinking: true }, { config });
323
- assert.equal(r, null, "no fallback and no compatible backend → null");
324
- });
325
-
326
- test("resolveBackend tags fallback_reason when falling through to anthropic safety net", () => {
327
- const config = {
328
- backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
329
- routing_policy: [{ default: true, backend: "qwen" }],
330
- fallback_to_anthropic: true,
331
- };
332
- const r = resolveBackend({ needs_thinking: true }, { config });
333
- assert.equal(r.name, "anthropic");
334
- assert.equal(r.fallback_reason, "no_compatible_backend");
335
- });
336
-
337
- // ---------------------------------------------------------------------------
338
- // estimateCost
339
- // ---------------------------------------------------------------------------
340
-
341
- test("estimateCost returns null without pricing data", () => {
342
- const fake = { pricing: {} };
343
- assert.equal(estimateCost({ token_estimate: 1000 }, fake), null);
344
- });
345
-
346
- test("estimateCost computes per-million-token cost without cache hits", () => {
347
- const fake = { pricing: { input_per_m: 3.0, output_per_m: 15.0 } };
348
- const cost = estimateCost({ token_estimate: 1_000_000, token_estimate_out: 500_000 }, fake);
349
- // 1M input @ $3 + 0.5M output @ $15 = $3 + $7.5 = $10.5
350
- assert.equal(cost, 10.5);
351
- });
352
-
353
- test("estimateCost honours cache_hit_per_m for Moonshot-style caching", () => {
354
- const fake = { pricing: { input_per_m: 0.95, output_per_m: 4.0, cache_hit_per_m: 0.16 } };
355
- // 1M tokens, 80% cache hit, no explicit output → defaults to 0.5M output
356
- const cost = estimateCost({ token_estimate: 1_000_000, cache_hit_ratio: 0.8 }, fake);
357
- // input: 200k @ 0.95 + 800k @ 0.16 = 0.19 + 0.128 = 0.318
358
- // output: 500k @ 4.0 = 2.0
359
- // total: 2.318
360
- assert.ok(Math.abs(cost - 2.318) < 0.001, `got ${cost}`);
361
- });
362
-
363
- // ---------------------------------------------------------------------------
364
- // listBackends / describeBackend
365
- // ---------------------------------------------------------------------------
366
-
367
- test("listBackends defaults to ['anthropic'] when no config", () => {
368
- assert.deepEqual(listBackends({ config: null }), ["anthropic"]);
369
- });
370
-
371
- test("listBackends returns all configured backend keys", () => {
372
- const config = {
373
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
374
- routing_policy: [],
375
- };
376
- assert.deepEqual(listBackends({ config }).sort(), ["anthropic", "moonshot", "qwen"]);
377
- });
378
-
379
- test("describeBackend returns model/transport/pricing for a backend", () => {
380
- const config = {
381
- backends: { moonshot: MOONSHOT_BACKEND },
382
- routing_policy: [],
383
- };
384
- const d = describeBackend("moonshot", { config });
385
- assert.equal(d.name, "moonshot");
386
- assert.equal(d.transport, "anthropic-native");
387
- assert.equal(d.base_url, "https://api.moonshot.ai/anthropic");
388
- assert.equal(d.models.default, "kimi-k2.6");
389
- });
390
-
391
- // ---------------------------------------------------------------------------
392
- // requestFromClassifierResult
393
- // ---------------------------------------------------------------------------
394
-
395
- test("requestFromClassifierResult preserves priority + model hint", () => {
396
- const req = requestFromClassifierResult(
397
- { priority: "critical", model: "opus", summary: "CEO request" },
398
- { role: "responder", source: "inbox" }
399
- );
400
- assert.equal(req.priority, "critical");
401
- assert.equal(req.model_hint, "opus");
402
- assert.equal(req.needs_thinking, true);
403
- assert.equal(req.needs_tool_use, true);
404
- assert.equal(req.agent_role, "responder");
405
- });
406
-
407
- test("requestFromClassifierResult does NOT set needs_thinking for routine sonnet items", () => {
408
- const req = requestFromClassifierResult(
409
- { priority: "normal", model: "sonnet" },
410
- { role: "responder" }
411
- );
412
- assert.equal(req.needs_thinking, undefined);
413
- assert.equal(req.model_hint, "sonnet");
414
- });
415
-
416
- // ---------------------------------------------------------------------------
417
- // Foot-gun fixes (audit W5 / W8 + kill switch)
418
- // ---------------------------------------------------------------------------
419
-
420
- test("W5: session work defaults needs_tool_use=true even for routine items", () => {
421
- // A normal-priority backlog item is still a full agent session — it must
422
- // declare tool use so a degraded_tool_use backend is rejected.
423
- const req = requestFromClassifierResult(
424
- { priority: "normal", model: "sonnet" },
425
- { role: "responder", source: "backlog" }
426
- );
427
- assert.equal(req.needs_tool_use, true, "session work needs tool use by default");
428
- });
429
-
430
- test("W5: one-shot lookups may opt out of needs_tool_use", () => {
431
- const req = requestFromClassifierResult(
432
- { priority: "normal", model: "sonnet" },
433
- { role: "lookup", oneShot: true }
434
- );
435
- assert.equal(req.needs_tool_use, false, "one-shot lookups opt out");
436
- });
437
-
438
- test("W5: shipped-example backlog item is kept off the degraded_tool_use backend", () => {
439
- // Mirrors scaffold/config/model-routing.yaml.example: a normal backlog item
440
- // matched `source: backlog → openrouter_qwen` (degraded_tool_use). With
441
- // needs_tool_use defaulting true, the capability gate now fires and the
442
- // request falls through to a tool-capable backend.
443
- const config = {
444
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
445
- routing_policy: [
446
- { match: { source: "backlog" }, backend: "qwen", fallback: ["moonshot", "anthropic"] },
447
- { default: true, backend: "moonshot", fallback: ["anthropic"] },
448
- ],
449
- fallback_to_anthropic: true,
450
- };
451
- const req = requestFromClassifierResult(
452
- { priority: "normal", model: "sonnet" },
453
- { role: "responder", source: "backlog" }
454
- );
455
- const r = resolveBackend(req, { config, env: { MOONSHOT_API_KEY: "msk" } });
456
- assert.notEqual(r.name, "qwen", "must NOT route a tool-using session to degraded_tool_use backend");
457
- assert.equal(r.name, "moonshot");
458
- assert.equal(r.tried[0].backend, "qwen");
459
- assert.equal(r.tried[0].reason, "degraded_tool_use");
460
- });
461
-
462
- test("W8: candidate with missing auth env is skipped, not spawned with empty token", () => {
463
- const config = {
464
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
465
- routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
466
- fallback_to_anthropic: true,
467
- };
468
- // MOONSHOT_API_KEY deliberately absent from env.
469
- const r = resolveBackend({ agent_role: "responder", tier: "default" }, { config, env: {} });
470
- assert.equal(r.name, "anthropic", "skips moonshot, lands on anthropic safety net");
471
- assert.equal(r.tried[0].backend, "moonshot");
472
- assert.equal(r.tried[0].reason, "missing_auth_env");
473
- // The anthropic-cli safety net injects no empty credential.
474
- assert.deepEqual(r.envForSpawn, {});
475
- });
476
-
477
- test("W8: empty/whitespace auth env counts as absent and is skipped", () => {
478
- const config = {
479
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
480
- routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
481
- fallback_to_anthropic: true,
482
- };
483
- const r = resolveBackend({ agent_role: "responder" }, { config, env: { MOONSHOT_API_KEY: " " } });
484
- assert.equal(r.name, "anthropic");
485
- assert.equal(r.tried[0].reason, "missing_auth_env");
486
- });
487
-
488
- test("W8: all candidates missing auth + no fallback → null (never empty spawn)", () => {
489
- const config = {
490
- backends: { moonshot: MOONSHOT_BACKEND },
491
- routing_policy: [{ default: true, backend: "moonshot" }],
492
- fallback_to_anthropic: false,
493
- };
494
- const r = resolveBackend({ agent_role: "responder" }, { config, env: {} });
495
- assert.equal(r, null, "no creds anywhere and no fallback → no resolution");
496
- });
497
-
498
- test("W8: present auth env still resolves normally (no regression)", () => {
499
- const config = {
500
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
501
- routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
502
- fallback_to_anthropic: true,
503
- };
504
- const r = resolveBackend({ agent_role: "responder" }, { config, env: { MOONSHOT_API_KEY: "msk-ok" } });
505
- assert.equal(r.name, "moonshot");
506
- assert.equal(r.envForSpawn.ANTHROPIC_AUTH_TOKEN, "msk-ok");
507
- });
508
-
509
- test("kill switch: MAESTRO_ROUTER_FORCE_ANTHROPIC=1 short-circuits to stock behaviour", () => {
510
- const config = {
511
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
512
- routing_policy: [{ default: true, backend: "moonshot" }],
513
- fallback_to_anthropic: true,
514
- };
515
- // Even with a valid Moonshot key and a default-moonshot policy, the kill
516
- // switch returns null — the same path as no config, so callers preserve
517
- // the legacy keychain-OAuth `claude --print` behaviour.
518
- const r = resolveBackend(
519
- { agent_role: "responder" },
520
- { config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1", MOONSHOT_API_KEY: "msk" } }
521
- );
522
- assert.equal(r, null, "kill switch → null (stock Anthropic CLI behaviour)");
523
- });
524
-
525
- test("kill switch: accepts 'true' as well as '1'", () => {
526
- const config = {
527
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
528
- routing_policy: [{ default: true, backend: "moonshot" }],
529
- };
530
- const r = resolveBackend(
531
- { agent_role: "responder" },
532
- { config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "true", MOONSHOT_API_KEY: "msk" } }
533
- );
534
- assert.equal(r, null);
535
- });
536
-
537
- test("kill switch: any other value does NOT short-circuit", () => {
538
- const config = {
539
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
540
- routing_policy: [{ default: true, backend: "moonshot" }],
541
- };
542
- const r = resolveBackend(
543
- { agent_role: "responder" },
544
- { config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "0", MOONSHOT_API_KEY: "msk" } }
545
- );
546
- assert.equal(r.name, "moonshot", "0/unset → routing still active");
547
- });
548
-
549
- // ---------------------------------------------------------------------------
550
- // defaultRoutingConfig wiring (audit L21)
551
- // ---------------------------------------------------------------------------
552
-
553
- test("defaultRoutingConfig returns a usable Anthropic-only config", () => {
554
- const cfg = defaultRoutingConfig();
555
- assert.ok(cfg, "default config should be an object");
556
- assert.ok(cfg.backends.anthropic, "default config has an anthropic backend");
557
- assert.equal(cfg.backends.anthropic.transport, "anthropic-cli");
558
- assert.equal(cfg.fallback_to_anthropic, true);
559
- // resolveBackend must accept it and preserve current CLI behaviour (no env).
560
- const r = resolveBackend({ agent_role: "responder" }, { config: cfg });
561
- assert.equal(r.name, "anthropic");
562
- assert.equal(r.transport, "anthropic-cli");
563
- assert.deepEqual(r.envForSpawn, {}, "anthropic-cli default injects no env");
564
- });
565
-
566
- test("defaultRoutingConfig returns a fresh, mutation-safe copy each call", () => {
567
- const a = defaultRoutingConfig();
568
- const b = defaultRoutingConfig();
569
- assert.notEqual(a, b, "each call returns a distinct object");
570
- a.backends.anthropic.models.default = "MUTATED";
571
- assert.notEqual(
572
- b.backends.anthropic.models.default,
573
- "MUTATED",
574
- "mutating one copy must not affect another",
575
- );
576
- });
577
-
578
- test("loadRoutingConfig still returns null with no config and no useDefault flag", async () => {
579
- const root = await makeAgentRoot();
580
- try {
581
- assert.equal(loadRoutingConfig(root), null, "default contract preserved");
582
- } finally { await rmRoot(root); }
583
- });
584
-
585
- test("loadRoutingConfig({ useDefault: true }) returns the default when no file exists", async () => {
586
- const root = await makeAgentRoot();
587
- try {
588
- const cfg = loadRoutingConfig(root, { useDefault: true });
589
- assert.ok(cfg, "useDefault yields a config instead of null");
590
- assert.ok(cfg.backends.anthropic, "it is the anthropic default");
591
- assert.equal(cfg.fallback_to_anthropic, true);
592
- } finally { await rmRoot(root); }
593
- });
594
-
595
- test("loadRoutingConfig prefers a real file over the useDefault fallback", async () => {
596
- const root = await makeAgentRoot();
597
- try {
598
- writeJsonConfig(root, {
599
- backends: { moonshot: MOONSHOT_BACKEND },
600
- routing_policy: [{ default: true, backend: "moonshot" }],
601
- });
602
- const cfg = loadRoutingConfig(root, { useDefault: true });
603
- assert.ok(cfg.backends.moonshot, "real config wins over the default fallback");
604
- } finally { await rmRoot(root); }
605
- });
606
-
607
- // ---------------------------------------------------------------------------
608
- // Constants exported
609
- // ---------------------------------------------------------------------------
610
-
611
- test("CONFIG_RELATIVE_PATH points at config/model-routing.yaml", () => {
612
- assert.equal(CONFIG_RELATIVE_PATH, "config/model-routing.yaml");
613
- });
614
-
615
- // ===========================================================================
616
- // resolveChain — the v2 policy brain (additive; v1 above stays byte-compatible)
617
- // ===========================================================================
618
-
619
- // One shared bundled catalog snapshot (read-only) for the resolve tests. Loaded
620
- // from the framework's own lib/model-router/catalog/*.yaml so the refs are real.
621
- // This file lives at <root>/lib/, so the framework root is one dir up.
622
- const MAESTRO_ROOT_FOR_TESTS = () =>
623
- join(fileURLToPath(new URL(".", import.meta.url)), "..");
624
- const CATALOG = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {});
625
-
626
- const V2_CONFIG = Object.freeze({
627
- schema_version: 2,
628
- aliases: {
629
- frontier: "anthropic/claude-opus-4-8",
630
- default: "anthropic/claude-sonnet-4-6",
631
- fast: "anthropic/claude-haiku-4-5",
632
- cheap: "deepseek/deepseek-v4-flash",
633
- "cheap-session": "moonshot/kimi-k2.6",
634
- },
635
- defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
636
- backends: {
637
- anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
638
- deepseek: { allowed_data_classes: ["public"] },
639
- moonshot: { allowed_data_classes: ["public"] },
640
- },
641
- routing_policy: [
642
- { match: { agent_role_in: ["regulatory", "audit"] }, chain: ["frontier", "default"], pin: true },
643
- { match: { task_class: "classify.inbox" }, harness: "direct", chain: ["fast", "cheap", "rules"], needs_tool_use: false },
644
- { match: { task_class: "lookup.gmail" }, harness: "direct", chain: ["fast", "cheap"], needs_tool_use: false },
645
- { match: { source: "backlog", data_class: "public" }, chain: ["cheap", "default"] },
646
- { match: { token_estimate_gte: 250000 }, chain: ["default", "frontier"] },
647
- { default: true, chain: ["default", "fast"] },
648
- ],
649
- budget_ladder: {
650
- "75": { downgrade_tiers: 1 },
651
- "90": { force_alias: "cheap_or_fast" },
652
- "100": { essential_only: true },
653
- },
654
- fallback_to_anthropic: true,
655
- });
656
-
657
- const CLOCK = () => 1_718_000_000_000;
658
- function rc(req, extra = {}) {
659
- return resolveChain(req, { catalog: CATALOG, config: V2_CONFIG, env: {}, now: CLOCK, ...extra });
660
- }
661
-
662
- // ── rule match + chain build ───────────────────────────────────────────────
663
-
664
- test("resolveChain: default rule routes a session to the default workhorse", () => {
665
- const d = rc({ task_class: "session.responder", source: "inbox", data_class: "sensitive", token_estimate: 5000 });
666
- assert.equal(d.chosen.provider, "anthropic");
667
- assert.equal(d.chosen.model, "claude-sonnet-4-6");
668
- assert.equal(d.chosen.harness, "session");
669
- assert.ok(d.decision_id, "every decision has an id");
670
- assert.ok(d.explain && typeof d.explain === "string", "explain is mandatory");
671
- });
672
-
673
- test("resolveChain: first-match picks the regulatory pin rule (frontier)", () => {
674
- const d = rc({ agent_role: "regulatory", task_class: "decision.writer", data_class: "sensitive", token_estimate: 1000 });
675
- assert.equal(d.chosen.model, "claude-opus-4-8", "regulatory → frontier");
676
- // The visible chain leads with the matched rule's first alias.
677
- assert.equal(d.chain[0].ref, "anthropic/claude-opus-4-8");
678
- });
679
-
680
- test("resolveChain: classify.inbox resolves direct harness + tool-less", () => {
681
- // Anthropic direct needs the key; with a key present, fast (haiku) is chosen.
682
- const d = resolveChain(
683
- { task_class: "classify.inbox", source: "inbox", data_class: "public", token_estimate: 1500 },
684
- { catalog: CATALOG, config: V2_CONFIG, env: { ANTHROPIC_API_KEY: "ak" }, now: CLOCK }
685
- );
686
- assert.equal(d.chosen.harness, "direct");
687
- assert.equal(d.chosen.model, "claude-haiku-4-5");
688
- });
689
-
690
- test("resolveChain: the chain[] carries the failover tail, not just the winner", () => {
691
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 });
692
- // default chain is [default, fast]; both Anthropic session rows survive.
693
- const refs = d.chain.map((c) => c.ref);
694
- assert.deepEqual(refs, ["anthropic/claude-sonnet-4-6", "anthropic/claude-haiku-4-5"]);
695
- });
696
-
697
- // ── credential gate (key-absence-as-enforcement, §7.1) ──────────────────────
698
-
699
- test("resolveChain: a third-party row without its key is skipped (missing_credential)", () => {
700
- // lookup.gmail is harness:direct, needs_tool_use:false so deepseek (grade B)
701
- // clears the capability gate — the ONLY reason left to skip it is the absent
702
- // DEEPSEEK_API_KEY (key-absence-as-enforcement, §7.1). Anthropic haiku direct
703
- // also needs its key (absent) so the chain ends on the Anthropic safety net.
704
- const d = rc({ task_class: "lookup.gmail", data_class: "public", token_estimate: 1000 });
705
- const skipped = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
706
- assert.ok(skipped, "deepseek recorded in tried[]");
707
- assert.equal(skipped.reason, "missing_credential");
708
- });
709
-
710
- test("resolveChain: present third-party key still gated by grade for sessions (B<A)", () => {
711
- // deepseek is grade B; a session needing tools requires grade A (§6.6), so even
712
- // WITH the key it is skipped grade_too_low and we fall to Anthropic.
713
- const d = resolveChain(
714
- { task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
715
- { catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
716
- );
717
- assert.equal(d.chosen.provider, "anthropic");
718
- const skipped = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
719
- assert.equal(skipped.reason, "grade_too_low");
720
- });
721
-
722
- test("resolveChain: grade-B row IS reachable on the direct (tool-less) lane", () => {
723
- // lookup.gmail is harness:direct, needs_tool_use:false; deepseek (B) is fine.
724
- // Anthropic haiku (direct) needs ANTHROPIC_API_KEY (absent) so it is skipped,
725
- // landing on deepseek with its key present.
726
- const d = resolveChain(
727
- { task_class: "lookup.gmail", data_class: "public", token_estimate: 1000 },
728
- { catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
729
- );
730
- assert.equal(d.chosen.provider, "deepseek");
731
- assert.equal(d.chosen.harness, "direct");
732
- });
733
-
734
- // ── data_class gate (deny-by-default, §7.6) ─────────────────────────────────
735
-
736
- test("resolveChain: sensitive data never routes to a public-only backend", () => {
737
- // Even with the key AND a direct lane, deepseek (allowed: public) is denied for
738
- // a sensitive request and recorded as data_class_denied.
739
- const cfg = { ...V2_CONFIG, routing_policy: [{ match: { task_class: "lookup.gmail" }, harness: "direct", chain: ["cheap", "fast"], needs_tool_use: false }, { default: true, chain: ["default"] }] };
740
- const d = resolveChain(
741
- { task_class: "lookup.gmail", data_class: "sensitive", token_estimate: 1000 },
742
- { catalog: CATALOG, config: cfg, env: { DEEPSEEK_API_KEY: "dk", ANTHROPIC_API_KEY: "ak" }, now: CLOCK }
743
- );
744
- assert.equal(d.chosen.provider, "anthropic", "sensitive falls back to anthropic");
745
- const denied = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
746
- assert.equal(denied.reason, "data_class_denied");
747
- });
748
-
749
- // ── breaker gate (health.isOpen) ────────────────────────────────────────────
750
-
751
- test("resolveChain: an open breaker skips the candidate (breaker_open in tried[])", () => {
752
- // Inject an isOpen that reports the sonnet model breaker open; chain falls to
753
- // the next survivor (haiku).
754
- const isOpen = (key) => ({
755
- open: key === "anthropic:claude-sonnet-4-6",
756
- until: 1_718_000_100_000,
757
- reason: "rate_limit",
758
- strikes: 1,
759
- });
760
- const d = resolveChain(
761
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
762
- { catalog: CATALOG, config: V2_CONFIG, env: {}, now: CLOCK, isOpen }
763
- );
764
- assert.equal(d.chosen.model, "claude-haiku-4-5", "skips open sonnet, lands on haiku");
765
- const skipped = d.tried.find((t) => t.ref === "anthropic/claude-sonnet-4-6");
766
- assert.equal(skipped.reason, "breaker_open");
767
- });
768
-
769
- // ── context gate ────────────────────────────────────────────────────────────
770
-
771
- test("resolveChain: a too-large request skips a small-context row (context_overflow)", () => {
772
- // haiku context_tokens is 190000; a 250k request matches the long-context rule
773
- // [default, frontier] (both 950k) and never even tries a small row. To assert
774
- // the gate directly, force a chain through haiku for a huge prompt.
775
- const cfg = { ...V2_CONFIG, routing_policy: [{ default: true, chain: ["fast", "default"] }] };
776
- const d = resolveChain(
777
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 500000 },
778
- { catalog: CATALOG, config: cfg, env: {}, now: CLOCK }
779
- );
780
- assert.equal(d.chosen.model, "claude-sonnet-4-6", "haiku too small → sonnet");
781
- const skipped = d.tried.find((t) => t.ref === "anthropic/claude-haiku-4-5");
782
- assert.equal(skipped.reason, "context_overflow");
783
- });
784
-
785
- test("resolveChain: long-context rule routes 250k+ to a 1M-ctx row", () => {
786
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 300000 });
787
- assert.equal(d.chosen.model, "claude-sonnet-4-6");
788
- });
789
-
790
- // ── envForSpawn (session retarget for Kimi/DeepSeek) ─────────────────────────
791
-
792
- test("resolveChain: a third-party SESSION row builds the ANTHROPIC_BASE_URL retarget env", () => {
793
- // Allow a grade override so kimi (B) can host a session, and route public work
794
- // to it with the key present.
795
- const cfg = {
796
- ...V2_CONFIG,
797
- catalog_overrides: undefined,
798
- routing_policy: [{ match: { source: "backlog" }, chain: ["cheap-session", "default"] }, { default: true, chain: ["default"] }],
799
- };
800
- // Override kimi grade to A via catalog_overrides so the session grade gate passes.
801
- const cat = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
802
- agentConfig: { catalog_overrides: [{ ref: "moonshot/kimi-k2.6", tool_reliability: "A" }] },
803
- });
804
- const d = resolveChain(
805
- { task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
806
- { catalog: cat, config: cfg, env: { MOONSHOT_API_KEY: "mk-test" }, now: CLOCK }
807
- );
808
- assert.equal(d.chosen.provider, "moonshot");
809
- assert.equal(d.chosen.transport, "anthropic-native");
810
- assert.equal(d.envForSpawn.ANTHROPIC_BASE_URL, "https://api.moonshot.ai/anthropic");
811
- assert.equal(d.envForSpawn.ANTHROPIC_AUTH_TOKEN, "mk-test");
812
- assert.equal(d.envForSpawn.ANTHROPIC_API_KEY, "", "empty string, NOT unset (keychain-fallthrough guard)");
813
- assert.equal(d.envForSpawn.ANTHROPIC_MODEL, "kimi-k2.6");
814
- });
815
-
816
- test("resolveChain: an Anthropic session injects NO retarget env (stock CLI)", () => {
817
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 });
818
- assert.deepEqual(d.envForSpawn, {});
819
- });
820
-
821
- // ── budget ladder ────────────────────────────────────────────────────────────
822
-
823
- test("resolveChain: band ≥75 downgrades one tier (default→fast) for non-pinned work", () => {
824
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, budget_band: 75 });
825
- assert.equal(d.chosen.model, "claude-haiku-4-5", "default downgraded to fast at band 75");
826
- assert.match(d.explain, /band 75/);
827
- });
828
-
829
- test("resolveChain: a pinned rule is exempt from the budget ladder", () => {
830
- // regulatory rule sets pin:true on the request indirectly via the rule pin flag;
831
- // pass pin on the request to exercise the exemption path.
832
- const d = rc({ agent_role: "regulatory", task_class: "decision", data_class: "sensitive", token_estimate: 1000, budget_band: 90, pin: { ref: "anthropic/claude-opus-4-8" } });
833
- assert.equal(d.chosen.model, "claude-opus-4-8", "pin beats the band-90 downgrade");
834
- });
835
-
836
- // ── affinity pin ─────────────────────────────────────────────────────────────
837
-
838
- test("resolveChain: a live affinity pin is moved to the chain head when it passes gates", () => {
839
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, session_key: "thread-1", pin: { ref: "anthropic/claude-haiku-4-5" } });
840
- assert.equal(d.chosen.model, "claude-haiku-4-5");
841
- assert.equal(d.chain[0].ref, "anthropic/claude-haiku-4-5", "pin at head");
842
- assert.match(d.explain, /pin /);
843
- });
844
-
845
- test("resolveChain: a pin that fails a gate emits pin_overridden and routes normally", () => {
846
- // Pin a public-only deepseek for sensitive work → data_class_denied → overridden.
847
- const d = resolveChain(
848
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, pin: { ref: "deepseek/deepseek-v4-flash" } },
849
- { catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
850
- );
851
- assert.equal(d.chosen.provider, "anthropic", "overridden pin falls to normal routing");
852
- const overridden = d.tried.find((t) => String(t.reason).startsWith("pin_overridden"));
853
- assert.ok(overridden, "pin_overridden recorded in tried[]");
854
- });
855
-
856
- // ── kill switch ──────────────────────────────────────────────────────────────
857
-
858
- test("resolveChain: MAESTRO_ROUTER_FORCE_ANTHROPIC forces an Anthropic-only chain", () => {
859
- const d = resolveChain(
860
- { task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
861
- { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1", DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
862
- );
863
- assert.equal(d.chosen.provider, "anthropic");
864
- assert.equal(d.audit.fallback_reason, "kill_switch");
865
- assert.match(d.explain, /kill switch/);
866
- // No third-party retarget env under the kill switch.
867
- assert.deepEqual(d.envForSpawn, {});
868
- });
869
-
870
- test("resolveChain: kill switch picks frontier for critical/thinking work", () => {
871
- const d = resolveChain(
872
- { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
873
- { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "true" }, now: CLOCK }
874
- );
875
- assert.equal(d.chosen.model, "claude-opus-4-8");
876
- });
877
-
878
- // ── inert / no-config collapse ───────────────────────────────────────────────
879
-
880
- test("resolveChain: no v2 config collapses to the Anthropic safety net", () => {
881
- const d = resolveChain(
882
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
883
- { catalog: CATALOG, config: null, env: {}, now: CLOCK }
884
- );
885
- assert.equal(d.chosen.provider, "anthropic");
886
- assert.equal(d.audit.fallback_reason, "no_v2_config");
887
- });
888
-
889
- // ── safety-net model freshness (pickAnthropicRow's ordered preference) ───────
890
- //
891
- // The safety net fires on the three paths that have no chain to walk: the kill
892
- // switch, "no v2 config", and "nothing survived the gates". It is the ONE place
893
- // resolve.mjs still names Anthropic model ids, so it is the one place that can
894
- // go stale. lookupModel is an exact byRef hit (no replaced_by chasing), so the
895
- // preference list carries the current -5 ids AND the 4-x succession fallback.
896
-
897
- /** The bundled catalog plus synthetic claude-*-5 rows (what ships next). */
898
- const CATALOG_WITH_5 = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
899
- agentConfig: {
900
- catalog_overrides: [
901
- {
902
- ref: "anthropic/claude-opus-5",
903
- status: "available",
904
- context_tokens: 950000,
905
- max_tokens: 128000,
906
- tool_reliability: "A",
907
- harness: { session: true, direct: true, batch: true },
908
- cost: { input: 5.0, output: 25.0, cache_read: 0.5, cache_write: 10.0 },
909
- },
910
- {
911
- ref: "anthropic/claude-sonnet-5",
912
- status: "available",
913
- context_tokens: 950000,
914
- max_tokens: 64000,
915
- tool_reliability: "A",
916
- harness: { session: true, direct: true, batch: true },
917
- cost: { input: 3.0, output: 15.0, cache_read: 0.3, cache_write: 6.0 },
918
- },
919
- ],
920
- },
921
- });
922
-
923
- test("resolveChain: the safety net prefers claude-opus-5 / claude-sonnet-5 when the catalog has them", () => {
924
- const frontier = resolveChain(
925
- { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
926
- { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
927
- );
928
- assert.equal(frontier.chosen.model, "claude-opus-5", "thinking/critical → current frontier, not opus-4-8");
929
-
930
- const workhorse = resolveChain(
931
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
932
- { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
933
- );
934
- assert.equal(workhorse.chosen.model, "claude-sonnet-5", "ordinary work → current workhorse, not sonnet-4-6");
935
- });
936
-
937
- test("resolveChain: the safety net falls back through succession when the catalog has no -5 rows", () => {
938
- // This is why the 4-x refs stay in the preference list. Delete them and this
939
- // does NOT fail loudly — it degrades to `models.find()` over the bundled rows,
940
- // i.e. whatever sits first in anthropic.yaml (opus-4-8), which would hand
941
- // ORDINARY work the frontier model and quietly quadruple its cost.
942
- const d = resolveChain(
943
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
944
- { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
945
- );
946
- assert.equal(d.chosen.model, "claude-sonnet-4-6", "succession fallback, ordered — NOT the first row in the file");
947
- // Flip this to claude-sonnet-5 (and drop the 4-x refs from pickAnthropicRow)
948
- // the day lib/model-router/catalog/anthropic.yaml carries the -5 rows.
949
- });
950
-
951
- test("the SHIPPED catalog is what decides the wire value — the -5 refresh is NOT done", () => {
952
- // THE HONEST STATE OF THIS REFRESH, pinned so it cannot be mistaken for
953
- // finished. The test above asserts `chosen.model`, which reads as bookkeeping.
954
- // This one asserts `spawnArgs.modelFlag` — the string spawn.mjs pushes as
955
- // `--model` — against the catalog the SDK actually ships, with `config: null`,
956
- // which is the DEFAULT state of every agent because this repo contains no
957
- // config/model-routing.yaml at all.
958
- //
959
- // Unlike the v1 lane (model-router.mjs's modelFlagFor sends the version-
960
- // agnostic "opus"/"sonnet" shorthand), resolve.mjs's modelFlagFor returns
961
- // row.id verbatim. So on v2 a stale catalog row is a stale id ON THE WIRE, not
962
- // merely a mis-stamped ledger entry.
963
- //
964
- // WHEN THIS FAILS: someone added the -5 rows to anthropic.yaml. Good — that is
965
- // the missing step. Update the expectations here, and DELETE the 4-x refs from
966
- // pickAnthropicRow's preference lists in the same change, or the museum the
967
- // provenance block warns about starts accumulating.
968
- const cases = [
969
- ["session.responder", {}, "claude-sonnet-4-6"],
970
- ["decision", { needs_thinking: true }, "claude-opus-4-8"],
971
- ];
972
- for (const [task_class, extra, expected] of cases) {
973
- const req = { task_class, data_class: "sensitive", token_estimate: 1000, ...extra };
974
- for (const [label, env] of [
975
- ["kill switch", { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }],
976
- ["no v2 config", {}],
977
- ]) {
978
- const d = resolveChain(req, { catalog: CATALOG, config: null, env, now: CLOCK });
979
- assert.equal(
980
- d.spawnArgs.modelFlag,
981
- expected,
982
- `${label}/${task_class}: --model is the literal catalog id, and the bundled catalog is still 4-x`
983
- );
984
- // Not the shorthand: nothing downstream re-resolves this to a current build.
985
- assert.ok(!["opus", "sonnet", "haiku"].includes(d.spawnArgs.modelFlag));
986
- }
987
- }
988
- });
989
-
990
- test("resolveChain: with NO catalog rows at all, the synthetic row's id and ref agree", () => {
991
- // The last-ditch path: no catalog, so the id goes straight onto `claude
992
- // --model`. It used to emit {model: "claude-opus-4-8", ref:
993
- // "anthropic/claude-sonnet-4-6"} for a thinking request — and ref is what
994
- // feeds cacheAffinityKey and chain[0].ref, so the pin and the audit trail
995
- // both named a different model than the one that ran.
996
- const emptyDir = join(tmpdir(), `model-router-empty-catalog-${process.pid}-${Date.now()}`);
997
- mkdirSync(emptyDir, { recursive: true });
998
- try {
999
- const EMPTY = loadCatalog(emptyDir, { bundledDir: emptyDir });
1000
- assert.equal(EMPTY.models.length, 0, "fixture really is an empty catalog");
1001
-
1002
- const d = resolveChain(
1003
- { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
1004
- { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1005
- );
1006
- assert.equal(d.chosen.model, "claude-opus-5");
1007
- assert.equal(d.chain[0].ref, "anthropic/claude-opus-5", "ref tracks id");
1008
- assert.match(d.cacheAffinityKey, /anthropic\/claude-opus-5$/);
1009
-
1010
- const ordinary = resolveChain(
1011
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
1012
- { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1013
- );
1014
- assert.equal(ordinary.chosen.model, "claude-sonnet-5");
1015
- assert.equal(ordinary.chain[0].ref, "anthropic/claude-sonnet-5");
1016
- } finally {
1017
- rmSync(emptyDir, { recursive: true, force: true });
1018
- }
1019
- });
1020
-
1021
- // ── cheap tier: an UNBUNDLED provider (xAI/Grok) is routable end to end ──────
1022
- //
1023
- // The SDK bundles catalog rows for anthropic/deepseek/moonshot/qwen only. Grok
1024
- // is NOT bundled — but "not bundled" is not "not routable": catalog.mjs's
1025
- // highest authority layer (config/model-routing.yaml `catalog_overrides:`) may
1026
- // introduce a whole provider, and resolveChain gates it exactly like a bundled
1027
- // one. These tests prove the full path — row → alias → chain → credential gate
1028
- // → chosen — so the remaining work is a bundled YAML file, not plumbing.
1029
-
1030
- const XAI_PROVIDER_DOC = {
1031
- provider: "xai",
1032
- auth_env: "XAI_API_KEY",
1033
- endpoints: { openai: "https://api.x.ai/v1", anthropic: "https://api.x.ai/anthropic" },
1034
- data_residency: "us",
1035
- models: [
1036
- {
1037
- id: "grok-4-fast",
1038
- status: "available",
1039
- context_window: 2000000,
1040
- context_tokens: 1900000,
1041
- max_tokens: 30000,
1042
- cost: { input: 0.2, output: 0.5 },
1043
- cost_provenance: { source: "unverified", fetched: "2026-08-13", volatile: true },
1044
- compat: { thinking_format: "openai", cache_control: "implicit" },
1045
- // Grade B like every other third-party row: good enough for the direct
1046
- // lane, structurally barred from hosting a tool-using session (§6.6).
1047
- tool_reliability: "B",
1048
- harness: { session: true, direct: true, batch: false },
1049
- },
1050
- ],
1051
- };
1052
-
1053
- const CATALOG_WITH_XAI = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
1054
- agentConfig: { catalog_overrides: XAI_PROVIDER_DOC },
1055
- });
1056
-
1057
- const XAI_CONFIG = Object.freeze({
1058
- schema_version: 2,
1059
- aliases: { cheap: "xai/grok-4-fast", default: "anthropic/claude-sonnet-4-6" },
1060
- defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
1061
- backends: {
1062
- anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
1063
- xai: { allowed_data_classes: ["public"] },
1064
- },
1065
- routing_policy: [
1066
- { match: { task_class: "lookup.web" }, harness: "direct", chain: ["cheap", "default"], needs_tool_use: false },
1067
- { default: true, chain: ["default"] },
1068
- ],
1069
- fallback_to_anthropic: true,
1070
- });
1071
-
1072
- test("catalog_overrides can introduce xAI/Grok — loudly, as a warning, never silently", () => {
1073
- const row = CATALOG_WITH_XAI.byRef.get("xai/grok-4-fast");
1074
- assert.ok(row, "grok row present after the agent-config merge");
1075
- assert.equal(row.provider, "xai");
1076
- assert.equal(row.auth_env, "XAI_API_KEY", "carries its OWN auth env, not a borrowed one");
1077
- assert.equal(row.endpoints.anthropic, "https://api.x.ai/anthropic");
1078
- assert.ok(
1079
- CATALOG_WITH_XAI.warnings.some((w) => /non-bundled provider "xai"/.test(w.warning)),
1080
- "introducing an unbundled provider warns (typo guard) — it does not fail"
1081
- );
1082
- });
1083
-
1084
- test("resolveChain routes a cheap-tier task to Grok when its own key is present", () => {
1085
- const d = resolveChain(
1086
- { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1087
- { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1088
- );
1089
- assert.equal(d.chosen.provider, "xai");
1090
- assert.equal(d.chosen.model, "grok-4-fast");
1091
- assert.equal(d.chosen.wire, "openai", "third-party direct call speaks the openai wire");
1092
- });
1093
-
1094
- test("resolveChain skips Grok without XAI_API_KEY (key-absence-as-enforcement, §7.1)", () => {
1095
- const d = resolveChain(
1096
- { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1097
- { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: {}, now: CLOCK }
1098
- );
1099
- const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1100
- assert.ok(skipped, "grok recorded in tried[]");
1101
- assert.equal(skipped.reason, "missing_credential");
1102
- assert.notEqual(d.chosen.provider, "xai");
1103
- });
1104
-
1105
- test("Grok inherits auth-profile key rotation from the catalog, with no per-provider code", () => {
1106
- // auth-profiles derives provider → auth_env from the CATALOG, so a provider
1107
- // the SDK has never heard of gets pooled multi-key rotation for free. This is
1108
- // the "their own auth profiles" half of the cheap-tier requirement.
1109
- const profiles = loadAuthProfiles(null, {
1110
- configDoc: null, // bypass the FS
1111
- env: { XAI_API_KEY: "k1", XAI_API_KEY_2: "k2", DEEPSEEK_API_KEY: "d1" },
1112
- authEnvByProvider: authEnvMapFromCatalog(CATALOG_WITH_XAI),
1113
- });
1114
- assert.deepEqual(profiles.keysFor("xai"), ["k1", "k2"]);
1115
- assert.equal(profiles.isPooled("xai"), true, "two keys ⇒ rotation is live");
1116
- assert.equal(profiles.isPooled("deepseek"), false, "single key ⇒ unchanged single-key behaviour");
1117
- });
1118
-
1119
- test("a session on Grok is still barred by the grade gate, exactly like DeepSeek/Kimi", () => {
1120
- // Not a bug to fix: §6.6 says a tool-using session needs grade A, and no
1121
- // third-party row is graded A until the probe ledger earns it. Cheap-tier
1122
- // reachability must NOT quietly become cheap-tier session hosting.
1123
- const sessionCfg = {
1124
- ...XAI_CONFIG,
1125
- routing_policy: [{ default: true, chain: ["cheap", "default"] }],
1126
- };
1127
- const d = resolveChain(
1128
- { task_class: "session.responder", data_class: "public", token_estimate: 2000 },
1129
- { catalog: CATALOG_WITH_XAI, config: sessionCfg, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1130
- );
1131
- const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1132
- assert.equal(skipped?.reason, "grade_too_low");
1133
- assert.equal(d.chosen.provider, "anthropic");
1134
- });
1135
-
1136
- // ── estCostUSD + ledger join fields ──────────────────────────────────────────
1137
-
1138
- test("resolveChain: stamps a finite estCostUSD and a decision_id for the ledger join", () => {
1139
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 100000, token_estimate_out: 20000 });
1140
- assert.ok(typeof d.estCostUSD === "number" && d.estCostUSD > 0, "priced from the catalog row");
1141
- assert.ok(/^[0-9a-z]{18}$/.test(d.decision_id), "ulid-ish decision_id");
1142
- assert.equal(d.spawnArgs.bare, true);
1143
- assert.ok(d.spawnArgs.maxTurns > 0);
1144
- });
1145
-
1146
- // ── deterministic "rules" fallback ───────────────────────────────────────────
1147
-
1148
- test("resolveChain: falls to the deterministic rules member when no model survives", () => {
1149
- // classify.inbox chain is [fast, cheap, rules]; with no keys at all, both
1150
- // model rows are unreachable and the caller's deterministic fallback wins.
1151
- const d = rc({ task_class: "classify.inbox", source: "inbox", data_class: "public", token_estimate: 1000 });
1152
- assert.equal(d.chosen.provider, "rules");
1153
- assert.equal(d.chosen.harness, "direct");
1154
- });
1155
-
1156
- // ===========================================================================
1157
- // validateRoutingConfig — strict v2 config validation
1158
- // ===========================================================================
1159
-
1160
- test("validateRoutingConfig: a clean v2 config has zero errors", () => {
1161
- const { errors } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, env: {} });
1162
- assert.equal(errors.length, 0, JSON.stringify(errors));
1163
- });
1164
-
1165
- test("validateRoutingConfig: unknown match key is an error (capability typo)", () => {
1166
- const cfg = { ...V2_CONFIG, routing_policy: [{ match: { taks_class: "x" }, chain: ["default"] }, { default: true, chain: ["default"] }] };
1167
- const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
1168
- assert.ok(errors.some((e) => /unknown match key "taks_class"/.test(e.error)));
1169
- });
1170
-
1171
- test("validateRoutingConfig: a chain ref not in the catalog is an error", () => {
1172
- const cfg = { ...V2_CONFIG, aliases: { ...V2_CONFIG.aliases, ghost: "ghost/model-x" }, routing_policy: [{ default: true, chain: ["ghost"] }] };
1173
- const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
1174
- assert.ok(errors.some((e) => /not in catalog/.test(e.error)));
1175
- });
1176
-
1177
- test("validateRoutingConfig: a chain member that is neither alias nor ref is an error", () => {
1178
- const cfg = { ...V2_CONFIG, routing_policy: [{ default: true, chain: ["notanalias"] }] };
1179
- const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
1180
- assert.ok(errors.some((e) => /neither a known alias nor/.test(e.error)));
1181
- });
1182
-
1183
- test("validateRoutingConfig: an unknown harness is an error", () => {
1184
- const cfg = { ...V2_CONFIG, routing_policy: [{ match: { task_class: "x" }, harness: "telepathy", chain: ["default"] }, { default: true, chain: ["default"] }] };
1185
- const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
1186
- assert.ok(errors.some((e) => /unknown harness "telepathy"/.test(e.error)));
1187
- });
1188
-
1189
- test("validateRoutingConfig: an overlay-excluded backend in config is an error (tighten-only)", () => {
1190
- const overlay = { version: "t1", constraints: { backends_allowed: ["anthropic", "deepseek"] } };
1191
- // config declares a moonshot backend the overlay does not permit.
1192
- const { errors } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, overlay });
1193
- assert.ok(errors.some((e) => /not in the org overlay's backends_allowed/.test(e.error)));
1194
- });
1195
-
1196
- test("validateRoutingConfig: a missing third-party key is a WARNING, not an error", () => {
1197
- const { errors, warnings } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, env: {} });
1198
- assert.equal(errors.length, 0);
1199
- assert.ok(warnings.some((w) => /DEEPSEEK_API_KEY/.test(w.warning)));
1200
- // Anthropic's key is never warned (session rides keychain OAuth).
1201
- assert.ok(!warnings.some((w) => /ANTHROPIC_API_KEY/.test(w.warning)));
1202
- });
1203
-
1204
- test("validateRoutingConfig: a v1 config is not strict-validated (returns clean)", () => {
1205
- const { errors } = validateRoutingConfig({ backends: {}, routing_policy: [] }, { catalog: CATALOG });
1206
- assert.equal(errors.length, 0);
1207
- });