@cohortapp/agent-sdk 2.17.0 → 2.18.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (526) hide show
  1. package/.claude/settings.json +18 -0
  2. package/.env.example +18 -5
  3. package/README.md +1 -0
  4. package/bin/maestro.mjs +62 -0
  5. package/docs/guides/billing-console-keys.md +60 -0
  6. package/docs/guides/front-door-session.md +54 -9
  7. package/docs/guides/mac-mini.md +20 -25
  8. package/docs/guides/setup-wizard.md +1 -1
  9. package/docs/runbooks/fleet-rollout.md +156 -0
  10. package/docs/runbooks/mac-mini-bootstrap.md +12 -14
  11. package/lib/action-executor.js +19 -3
  12. package/lib/budget-guard.mjs +279 -3
  13. package/lib/channels/base-adapter.mjs +3 -1
  14. package/lib/channels/contract.mjs +2 -1
  15. package/lib/channels/inbox-item.mjs +8 -0
  16. package/lib/claude-bin.mjs +5 -6
  17. package/lib/cli/doctor-checks.mjs +141 -10
  18. package/lib/cli/global-setup-extras.mjs +5 -1
  19. package/lib/cli/inbox.mjs +100 -15
  20. package/lib/cli/seat-auth.mjs +463 -0
  21. package/lib/cli/session.mjs +80 -12
  22. package/lib/collective/capture.mjs +8 -6
  23. package/lib/collective/global-config.mjs +63 -1
  24. package/lib/collective/presence.mjs +142 -5
  25. package/lib/comms/send-gate.mjs +559 -1
  26. package/lib/diagnostics/alerts.mjs +49 -0
  27. package/lib/diagnostics/cadence-output-freshness.mjs +288 -0
  28. package/lib/engine/agents/definitions.mjs +343 -0
  29. package/lib/engine/agents/persist.mjs +275 -0
  30. package/lib/engine/agents/runtime.mjs +748 -0
  31. package/lib/engine/agents/usage.mjs +95 -0
  32. package/lib/engine/auth-status.mjs +139 -0
  33. package/lib/engine/budget.mjs +194 -0
  34. package/lib/engine/cli.mjs +1204 -0
  35. package/lib/engine/commands/index.mjs +269 -0
  36. package/lib/engine/context/budget.mjs +219 -0
  37. package/lib/engine/context/cache.mjs +125 -0
  38. package/lib/engine/context/child-env.mjs +215 -0
  39. package/lib/engine/context/compaction.mjs +342 -0
  40. package/lib/engine/context/images.mjs +90 -0
  41. package/lib/engine/context/instructions.mjs +327 -0
  42. package/lib/engine/context/lazy-instructions.mjs +169 -0
  43. package/lib/engine/context/manager.mjs +182 -0
  44. package/lib/engine/context/real-path.mjs +91 -0
  45. package/lib/engine/context/secret-values.mjs +163 -0
  46. package/lib/engine/context/settings.mjs +274 -0
  47. package/lib/engine/context/stream-input.mjs +159 -0
  48. package/lib/engine/guard.mjs +152 -0
  49. package/lib/engine/hooks.mjs +713 -0
  50. package/lib/engine/loop.mjs +560 -0
  51. package/lib/engine/mcp/client.mjs +254 -0
  52. package/lib/engine/mcp/config.mjs +301 -0
  53. package/lib/engine/mcp/http.mjs +201 -0
  54. package/lib/engine/mcp/index.mjs +146 -0
  55. package/lib/engine/mcp/jsonrpc.mjs +147 -0
  56. package/lib/engine/mcp/naming.mjs +66 -0
  57. package/lib/engine/mcp/resources.mjs +89 -0
  58. package/lib/engine/mcp/results.mjs +133 -0
  59. package/lib/engine/mcp/stdio.mjs +137 -0
  60. package/lib/engine/mcp/supervisor.mjs +116 -0
  61. package/lib/engine/messages.mjs +104 -0
  62. package/lib/engine/output/json.mjs +164 -0
  63. package/lib/engine/output/stream-json.mjs +266 -0
  64. package/lib/engine/permissions.mjs +845 -0
  65. package/lib/engine/process-identity.mjs +164 -0
  66. package/lib/engine/process-tree.mjs +551 -0
  67. package/lib/engine/prompt.mjs +60 -0
  68. package/lib/engine/session/store.mjs +299 -0
  69. package/lib/engine/session-runtime/args.mjs +97 -0
  70. package/lib/engine/session-runtime/host.mjs +143 -0
  71. package/lib/engine/session-runtime/inbox.mjs +122 -0
  72. package/lib/engine/session-runtime/notifications.mjs +129 -0
  73. package/lib/engine/session-runtime/registry.mjs +328 -0
  74. package/lib/engine/session-runtime/runner.mjs +344 -0
  75. package/lib/engine/session-runtime/socket.mjs +212 -0
  76. package/lib/engine/session-runtime/wakeup.mjs +115 -0
  77. package/lib/engine/skills/index.mjs +321 -0
  78. package/lib/engine/tools/bash-background.mjs +533 -0
  79. package/lib/engine/tools/bash.mjs +216 -0
  80. package/lib/engine/tools/edit.mjs +97 -0
  81. package/lib/engine/tools/glob.mjs +81 -0
  82. package/lib/engine/tools/grep.mjs +224 -0
  83. package/lib/engine/tools/index.mjs +84 -0
  84. package/lib/engine/tools/list-agents.mjs +32 -0
  85. package/lib/engine/tools/ls.mjs +127 -0
  86. package/lib/engine/tools/monitor.mjs +82 -0
  87. package/lib/engine/tools/notebook-edit.mjs +218 -0
  88. package/lib/engine/tools/read.mjs +103 -0
  89. package/lib/engine/tools/schedule-wakeup.mjs +45 -0
  90. package/lib/engine/tools/schema.mjs +144 -0
  91. package/lib/engine/tools/send-message.mjs +77 -0
  92. package/lib/engine/tools/session.mjs +70 -0
  93. package/lib/engine/tools/todo.mjs +144 -0
  94. package/lib/engine/tools/toolsearch.mjs +217 -0
  95. package/lib/engine/tools/walk.mjs +193 -0
  96. package/lib/engine/tools/web-switch.mjs +31 -0
  97. package/lib/engine/tools/webfetch-html.mjs +387 -0
  98. package/lib/engine/tools/webfetch-net.mjs +340 -0
  99. package/lib/engine/tools/webfetch.mjs +198 -0
  100. package/lib/engine/tools/websearch.mjs +91 -0
  101. package/lib/engine/tools/workflow.mjs +95 -0
  102. package/lib/engine/tools/write.mjs +76 -0
  103. package/lib/engine/tui/line-editor.mjs +137 -0
  104. package/lib/engine/tui/render.mjs +86 -0
  105. package/lib/engine/tui/tui.mjs +274 -0
  106. package/lib/engine/wire/anthropic-messages.mjs +263 -0
  107. package/lib/engine/wire/effort.mjs +36 -0
  108. package/lib/engine/wire/errors.mjs +496 -0
  109. package/lib/engine/wire/http.mjs +441 -0
  110. package/lib/engine/wire/index.mjs +76 -0
  111. package/lib/engine/wire/openai-chat.mjs +332 -0
  112. package/lib/engine/wire/prompt-cache.mjs +79 -0
  113. package/lib/engine/wire/search.mjs +140 -0
  114. package/lib/engine/wire/sse.mjs +114 -0
  115. package/lib/engine/wire/stall.mjs +349 -0
  116. package/lib/engine/wire/token-provider.mjs +175 -0
  117. package/lib/engine/wire/usage.mjs +192 -0
  118. package/lib/engine/workflow/host.mjs +524 -0
  119. package/lib/engine/workflow/journal.mjs +188 -0
  120. package/lib/engine/workflow/json-schema.mjs +171 -0
  121. package/lib/engine/workflow/meta.mjs +329 -0
  122. package/lib/engine/workflow/notifications.mjs +52 -0
  123. package/lib/engine/workflow/runtime.mjs +447 -0
  124. package/lib/engine/workflow/sandbox.mjs +534 -0
  125. package/lib/engine/workflow/worker.mjs +141 -0
  126. package/lib/engine/workflow/worktree.mjs +74 -0
  127. package/lib/execution/disposition.mjs +1 -1
  128. package/lib/execution/intake.mjs +10 -0
  129. package/lib/execution/surface-policy.mjs +15 -0
  130. package/lib/learning/curator.mjs +8 -6
  131. package/lib/learning/reflect.mjs +8 -6
  132. package/lib/model-router/catalog/cohort.yaml +137 -0
  133. package/lib/model-router/catalog.mjs +118 -1
  134. package/lib/model-router/failover.mjs +67 -16
  135. package/lib/model-router/llm-task.mjs +39 -3
  136. package/lib/model-router/resolve.mjs +89 -3
  137. package/lib/model-router/spawn.mjs +46 -47
  138. package/lib/model-router/taxonomy.mjs +126 -4
  139. package/lib/org/cost-sync.mjs +141 -11
  140. package/lib/org/inbound/broadcast.mjs +289 -0
  141. package/lib/org/inbound/collective.mjs +375 -0
  142. package/lib/org/inbound/directedness.mjs +96 -8
  143. package/lib/org/inbound/facts.mjs +78 -2
  144. package/lib/org/inbound/project.mjs +22 -0
  145. package/lib/org/inbound/surfaces.mjs +14 -0
  146. package/lib/org/llm-token.mjs +879 -0
  147. package/lib/org/mesh.mjs +61 -0
  148. package/lib/org/messaging.mjs +3 -1
  149. package/lib/org/protocol.checksum +1 -1
  150. package/lib/org/protocol.mjs +15 -0
  151. package/lib/org/quota.mjs +520 -0
  152. package/lib/org/tool-surface.mjs +104 -16
  153. package/lib/org/ui-parity.mjs +16 -1
  154. package/lib/org/work-ledger.mjs +37 -6
  155. package/lib/rate-guard.mjs +114 -1
  156. package/lib/resource-governor.mjs +41 -6
  157. package/lib/runtime/adapter.mjs +823 -0
  158. package/lib/runtime/child-env.mjs +191 -0
  159. package/lib/runtime/legacy-shell-guard.mjs +97 -0
  160. package/lib/runtime/seat-engine.mjs +162 -0
  161. package/lib/session/ask-ledger.mjs +271 -0
  162. package/lib/session/current-work.mjs +676 -0
  163. package/lib/session/feed-core.mjs +40 -3
  164. package/lib/session/launch-args.mjs +56 -4
  165. package/lib/session/status-summary.mjs +26 -9
  166. package/lib/session/upgrade-notice.mjs +42 -0
  167. package/lib/setup/claude-probe.mjs +117 -13
  168. package/lib/setup/enrich.mjs +13 -10
  169. package/lib/setup/sections/model.mjs +39 -13
  170. package/lib/telemetry/collect.mjs +208 -9
  171. package/lib/upgrade/ignored-drift.mjs +105 -0
  172. package/lib/voice/post-call-brief.mjs +30 -17
  173. package/package.json +13 -3
  174. package/plugins/maestro-skills/skills/board-work.md +5 -0
  175. package/plugins/maestro-skills/skills/inbound-triage.md +56 -15
  176. package/plugins/maestro-skills/skills/main-session.md +18 -7
  177. package/scripts/ci/check-tarball-fidelity.mjs +126 -2
  178. package/scripts/ci/run-tests.mjs +47 -19
  179. package/scripts/cohort-llm/api-key-helper.mjs +92 -0
  180. package/scripts/collective/hook-runner.mjs +29 -2
  181. package/scripts/continuous-monitor.sh +13 -0
  182. package/scripts/cost/track-claude-usage.mjs +15 -0
  183. package/scripts/daemon/agent-daemon.mjs +408 -20
  184. package/scripts/daemon/assurance.mjs +48 -12
  185. package/scripts/daemon/cadence-consumer.mjs +218 -68
  186. package/scripts/daemon/cadence-handlers.mjs +73 -4
  187. package/scripts/daemon/classifier.mjs +75 -26
  188. package/scripts/daemon/context-compiler.mjs +51 -37
  189. package/scripts/daemon/deliver.mjs +30 -1
  190. package/scripts/daemon/dispatcher.mjs +595 -149
  191. package/scripts/daemon/health.mjs +14 -1
  192. package/scripts/daemon/maestro-daemon.mjs +11 -0
  193. package/scripts/daemon/prompt-builder.mjs +24 -0
  194. package/scripts/daemon/responder.mjs +246 -79
  195. package/scripts/daemon/sdk-version.mjs +98 -16
  196. package/scripts/eval/probe-gateway.mjs +635 -0
  197. package/scripts/eval/replay/extract.mjs +270 -0
  198. package/scripts/eval/replay/grade.mjs +260 -0
  199. package/scripts/eval/replay/lib/config.mjs +50 -0
  200. package/scripts/eval/replay/lib/effects.mjs +65 -0
  201. package/scripts/eval/replay/lib/fixture.mjs +188 -0
  202. package/scripts/eval/replay/lib/judge.mjs +72 -0
  203. package/scripts/eval/replay/lib/redact.mjs +136 -0
  204. package/scripts/eval/replay/lib/sandbox.mjs +170 -0
  205. package/scripts/eval/replay/lib/schema-check.mjs +63 -0
  206. package/scripts/eval/replay/lib/transcript.mjs +76 -0
  207. package/scripts/eval/replay/mcp-replay-stub.mjs +101 -0
  208. package/scripts/eval/replay/report.mjs +185 -0
  209. package/scripts/eval/replay/run.mjs +404 -0
  210. package/scripts/fleet/rollout.mjs +1094 -0
  211. package/scripts/hooks/pre-send-audit.sh +36 -245
  212. package/scripts/hooks/pre-write-yaml-validate.mjs +275 -0
  213. package/scripts/hooks/validate-state-yaml.sh +190 -0
  214. package/scripts/huddle/huddle-llm.mjs +361 -0
  215. package/scripts/huddle/huddle-server.mjs +46 -121
  216. package/scripts/local-triggers/autoupdate.sh +448 -78
  217. package/scripts/local-triggers/run-trigger.sh +13 -0
  218. package/scripts/maintenance/pin-integrity.mjs +364 -0
  219. package/scripts/poll-slack-events.sh +41 -9
  220. package/scripts/poller/slack-socket-mode.mjs +28 -3
  221. package/scripts/session/supervisor.mjs +80 -13
  222. package/scripts/spawn-session.sh +13 -0
  223. package/bin/maestro.test.mjs +0 -1574
  224. package/lib/action-executor.test.mjs +0 -871
  225. package/lib/archetype.test.mjs +0 -132
  226. package/lib/assurance/plan-note.test.mjs +0 -234
  227. package/lib/assurance/room-budget.test.mjs +0 -486
  228. package/lib/assurance/tier.test.mjs +0 -174
  229. package/lib/autonomy.test.mjs +0 -66
  230. package/lib/backlog.test.mjs +0 -302
  231. package/lib/backup/policy.test.mjs +0 -305
  232. package/lib/budget-escalate.test.mjs +0 -232
  233. package/lib/budget-guard.envelope.test.mjs +0 -476
  234. package/lib/budget-guard.test.mjs +0 -427
  235. package/lib/cadence-bus-requeue.test.mjs +0 -83
  236. package/lib/cadence-bus-schedule.test.mjs +0 -194
  237. package/lib/cadence-bus.test.mjs +0 -720
  238. package/lib/cadences.test.mjs +0 -230
  239. package/lib/capability/inventory.test.mjs +0 -232
  240. package/lib/capability.test.mjs +0 -78
  241. package/lib/channels/base-adapter.test.mjs +0 -590
  242. package/lib/channels/channels.test.mjs +0 -371
  243. package/lib/channels/contract.test.mjs +0 -162
  244. package/lib/channels/inbox-item.test.mjs +0 -368
  245. package/lib/channels/orgmail/adapter.test.mjs +0 -448
  246. package/lib/channels/pairing.test.mjs +0 -270
  247. package/lib/channels/repeat-suppressor.test.mjs +0 -134
  248. package/lib/channels/slack-adapter.test.mjs +0 -212
  249. package/lib/channels/telegram-adapter.test.mjs +0 -306
  250. package/lib/channels/voice/adapter.test.mjs +0 -278
  251. package/lib/channels/whatsapp/adapter-baileys.test.mjs +0 -359
  252. package/lib/channels/whatsapp/baileys-typing.test.mjs +0 -154
  253. package/lib/charter.test.mjs +0 -89
  254. package/lib/claude-bin.test.mjs +0 -131
  255. package/lib/cli/board.test.mjs +0 -227
  256. package/lib/cli/design.test.mjs +0 -270
  257. package/lib/cli/doctor-checks.test.mjs +0 -336
  258. package/lib/cli/global-setup-extras.test.mjs +0 -462
  259. package/lib/cli/inbox.test.mjs +0 -230
  260. package/lib/cli/session-ack.test.mjs +0 -63
  261. package/lib/cli/session.test.mjs +0 -613
  262. package/lib/collective/capture.test.mjs +0 -121
  263. package/lib/collective/cards.test.mjs +0 -114
  264. package/lib/collective/config.test.mjs +0 -123
  265. package/lib/collective/global-config.test.mjs +0 -220
  266. package/lib/collective/global-skills.test.mjs +0 -126
  267. package/lib/collective/presence.test.mjs +0 -95
  268. package/lib/collective/recall.test.mjs +0 -116
  269. package/lib/collective/vendor-skills.test.mjs +0 -306
  270. package/lib/comms/send-gate.test.mjs +0 -770
  271. package/lib/comms.test.mjs +0 -41
  272. package/lib/context/budget.test.mjs +0 -252
  273. package/lib/context/history-scope.test.mjs +0 -79
  274. package/lib/cost/ledger-row.test.mjs +0 -183
  275. package/lib/design/design-md.test.mjs +0 -318
  276. package/lib/design/fixtures/DESIGN.golden.md +0 -238
  277. package/lib/design/fixtures/PRODUCT.golden.md +0 -67
  278. package/lib/design/fixtures/foundation.json +0 -133
  279. package/lib/design/refresh-gate.test.mjs +0 -144
  280. package/lib/design/write.test.mjs +0 -241
  281. package/lib/diagnostics/alerts.test.mjs +0 -318
  282. package/lib/diagnostics/backup-freshness.test.mjs +0 -185
  283. package/lib/diagnostics/counters.test.mjs +0 -206
  284. package/lib/diagnostics/events.test.mjs +0 -290
  285. package/lib/diagnostics/otel.test.mjs +0 -196
  286. package/lib/diagnostics/trace.test.mjs +0 -251
  287. package/lib/env-compat.test.mjs +0 -104
  288. package/lib/execution/disposition.test.mjs +0 -553
  289. package/lib/execution/drive.test.mjs +0 -270
  290. package/lib/execution/effects.test.mjs +0 -344
  291. package/lib/execution/intake.test.mjs +0 -389
  292. package/lib/execution/journal.test.mjs +0 -261
  293. package/lib/execution/match.test.mjs +0 -235
  294. package/lib/execution/pipeline.test.mjs +0 -392
  295. package/lib/execution/route.test.mjs +0 -186
  296. package/lib/execution/surface-policy.test.mjs +0 -162
  297. package/lib/fs-atomic.test.mjs +0 -72
  298. package/lib/fs-ownership.test.mjs +0 -158
  299. package/lib/goals/admission.test.mjs +0 -164
  300. package/lib/goals/classify.test.mjs +0 -167
  301. package/lib/goals/collaborate.test.mjs +0 -336
  302. package/lib/goals/gaps.test.mjs +0 -284
  303. package/lib/goals/loop.test.mjs +0 -845
  304. package/lib/hooks/bus.test.mjs +0 -387
  305. package/lib/identity/persona.test.mjs +0 -142
  306. package/lib/kpi-sensors.test.mjs +0 -278
  307. package/lib/kpi.test.mjs +0 -244
  308. package/lib/learning/config.test.mjs +0 -75
  309. package/lib/learning/counters.test.mjs +0 -69
  310. package/lib/learning/curator-consolidate.test.mjs +0 -238
  311. package/lib/learning/curator.test.mjs +0 -106
  312. package/lib/learning/reflect.test.mjs +0 -0
  313. package/lib/learning/session-index.test.mjs +0 -125
  314. package/lib/learning/skill-writer.test.mjs +0 -210
  315. package/lib/mandate/audit.test.mjs +0 -195
  316. package/lib/mandate/contract.test.mjs +0 -185
  317. package/lib/mandate/derive.test.mjs +0 -274
  318. package/lib/mandate/model.test.mjs +0 -164
  319. package/lib/mandate/refresh.test.mjs +0 -389
  320. package/lib/mcp/server.test.mjs +0 -426
  321. package/lib/model-router/auth-profiles.test.mjs +0 -580
  322. package/lib/model-router/catalog.test.mjs +0 -385
  323. package/lib/model-router/economics.test.mjs +0 -438
  324. package/lib/model-router/failover.test.mjs +0 -439
  325. package/lib/model-router/health.test.mjs +0 -338
  326. package/lib/model-router/integration-coverage.test.mjs +0 -831
  327. package/lib/model-router/integration.test.mjs +0 -564
  328. package/lib/model-router/ledger.test.mjs +0 -415
  329. package/lib/model-router/llm-task.test.mjs +0 -392
  330. package/lib/model-router/org-credentials.test.mjs +0 -265
  331. package/lib/model-router/pricing-refresh.test.mjs +0 -286
  332. package/lib/model-router/reconcile.test.mjs +0 -316
  333. package/lib/model-router/repair.test.mjs +0 -180
  334. package/lib/model-router/spawn.test.mjs +0 -446
  335. package/lib/model-router/taxonomy.test.mjs +0 -410
  336. package/lib/model-router.test.mjs +0 -1207
  337. package/lib/org/activity.test.mjs +0 -134
  338. package/lib/org/approvals.test.mjs +0 -216
  339. package/lib/org/awareness.test.mjs +0 -159
  340. package/lib/org/board-mine-cache.test.mjs +0 -53
  341. package/lib/org/board.test.mjs +0 -187
  342. package/lib/org/bootstrap-context.test.mjs +0 -153
  343. package/lib/org/client.test.mjs +0 -1206
  344. package/lib/org/cohort-client.test.mjs +0 -126
  345. package/lib/org/cost-sync.test.mjs +0 -153
  346. package/lib/org/doctor.test.mjs +0 -346
  347. package/lib/org/engagement-ledger.test.mjs +0 -112
  348. package/lib/org/engagement.test.mjs +0 -739
  349. package/lib/org/handoff.test.mjs +0 -269
  350. package/lib/org/inbound/directedness.test.mjs +0 -668
  351. package/lib/org/inbound/facts.test.mjs +0 -471
  352. package/lib/org/inbound/hydrate.test.mjs +0 -908
  353. package/lib/org/inbound/index.test.mjs +0 -429
  354. package/lib/org/inbound/project.test.mjs +0 -287
  355. package/lib/org/integration-tools.test.mjs +0 -160
  356. package/lib/org/keys.test.mjs +0 -92
  357. package/lib/org/knowledge.test.mjs +0 -326
  358. package/lib/org/leases.test.mjs +0 -235
  359. package/lib/org/mesh-directives.test.mjs +0 -110
  360. package/lib/org/mesh-integration.test.mjs +0 -127
  361. package/lib/org/mesh.test.mjs +0 -400
  362. package/lib/org/messaging.test.mjs +0 -471
  363. package/lib/org/param-contract.test.mjs +0 -477
  364. package/lib/org/policy.test.mjs +0 -237
  365. package/lib/org/protocol.checksum.test.mjs +0 -90
  366. package/lib/org/protocol.test.mjs +0 -323
  367. package/lib/org/push.test.mjs +0 -792
  368. package/lib/org/registry.test.mjs +0 -100
  369. package/lib/org/resource-tools.test.mjs +0 -361
  370. package/lib/org/tool-access.test.mjs +0 -144
  371. package/lib/org/tool-surface-integration.test.mjs +0 -120
  372. package/lib/org/tool-surface.test.mjs +0 -1268
  373. package/lib/org/typing.test.mjs +0 -291
  374. package/lib/org/ui-parity.test.mjs +0 -560
  375. package/lib/org/verify.test.mjs +0 -194
  376. package/lib/org/work-ledger.test.mjs +0 -273
  377. package/lib/plan/adoption-e2e.test.mjs +0 -366
  378. package/lib/plan/budget-enforcement.test.mjs +0 -400
  379. package/lib/plan/compile.test.mjs +0 -382
  380. package/lib/plan/emit.test.mjs +0 -269
  381. package/lib/plan/explain.test.mjs +0 -188
  382. package/lib/prompts/parallelism.test.mjs +0 -177
  383. package/lib/rag/rag.test.mjs +0 -505
  384. package/lib/rate-guard.test.mjs +0 -272
  385. package/lib/reactive-gate.test.mjs +0 -57
  386. package/lib/render.test.mjs +0 -68
  387. package/lib/resource-governor.test.mjs +0 -488
  388. package/lib/scheduling/dynamic-jobs.test.mjs +0 -344
  389. package/lib/scheduling/jitter.test.mjs +0 -140
  390. package/lib/secrets/broker.test.mjs +0 -280
  391. package/lib/secrets/providers.test.mjs +0 -274
  392. package/lib/security/audit-engine.test.mjs +0 -424
  393. package/lib/security/coerce-args.test.mjs +0 -281
  394. package/lib/security/dangerous-tools.test.mjs +0 -68
  395. package/lib/security/external-content.test.mjs +0 -84
  396. package/lib/security/redact.test.mjs +0 -441
  397. package/lib/security/secret-equal.test.mjs +0 -55
  398. package/lib/session/config.test.mjs +0 -92
  399. package/lib/session/feed-core.test.mjs +0 -198
  400. package/lib/session/first-run.test.mjs +0 -121
  401. package/lib/session/frontdoor.test.mjs +0 -205
  402. package/lib/session/handoffs.test.mjs +0 -183
  403. package/lib/session/identity.test.mjs +0 -180
  404. package/lib/session/inbox-claims.test.mjs +0 -286
  405. package/lib/session/launch-args.test.mjs +0 -157
  406. package/lib/session/liveness.test.mjs +0 -100
  407. package/lib/session/status-summary.test.mjs +0 -118
  408. package/lib/session-permissions.test.mjs +0 -120
  409. package/lib/setup/claude-probe.test.mjs +0 -187
  410. package/lib/setup/completeness.test.mjs +0 -110
  411. package/lib/setup/context-pack.test.mjs +0 -89
  412. package/lib/setup/enrich.test.mjs +0 -115
  413. package/lib/setup/enroll-from-cohort.test.mjs +0 -300
  414. package/lib/setup/integration.test.mjs +0 -162
  415. package/lib/setup/io.test.mjs +0 -77
  416. package/lib/setup/runner.test.mjs +0 -132
  417. package/lib/setup/sections/identity.test.mjs +0 -234
  418. package/lib/setup/sections/inventory.test.mjs +0 -198
  419. package/lib/setup/sections/learning.test.mjs +0 -81
  420. package/lib/setup/sections/mandate.test.mjs +0 -388
  421. package/lib/setup/sections/messaging.test.mjs +0 -127
  422. package/lib/setup/sections/model.test.mjs +0 -240
  423. package/lib/setup/sections/org.test.mjs +0 -346
  424. package/lib/setup/sections/orgmail.test.mjs +0 -118
  425. package/lib/setup/sections/recovery.test.mjs +0 -98
  426. package/lib/setup/sections/subagents.test.mjs +0 -429
  427. package/lib/setup/sections/verify.test.mjs +0 -175
  428. package/lib/setup/sot.test.mjs +0 -81
  429. package/lib/setup/state.test.mjs +0 -115
  430. package/lib/singleton.test.mjs +0 -151
  431. package/lib/subagents/cli.test.mjs +0 -389
  432. package/lib/subagents/client.test.mjs +0 -309
  433. package/lib/subagents/gap.test.mjs +0 -234
  434. package/lib/subagents/lock.test.mjs +0 -248
  435. package/lib/subagents/manifest.test.mjs +0 -175
  436. package/lib/subagents/refs.test.mjs +0 -204
  437. package/lib/subagents/resolve.test.mjs +0 -422
  438. package/lib/subagents/schema.test.mjs +0 -328
  439. package/lib/telemetry/alerts.test.mjs +0 -109
  440. package/lib/telemetry/collect.test.mjs +0 -1274
  441. package/lib/tool-definitions-integration.test.mjs +0 -83
  442. package/lib/tool-definitions.test.mjs +0 -437
  443. package/lib/upgrade/global-refresh.test.mjs +0 -65
  444. package/lib/upgrade/launchd-reconcile.test.mjs +0 -272
  445. package/lib/upgrade/post-steps.test.mjs +0 -200
  446. package/lib/upgrade/verify.test.mjs +0 -164
  447. package/lib/util/fetch-timeout.test.mjs +0 -202
  448. package/lib/util/reconnect.test.mjs +0 -369
  449. package/lib/util/unhandled.test.mjs +0 -216
  450. package/lib/voice/outbound.test.mjs +0 -69
  451. package/lib/voice/session-rotation.test.mjs +0 -114
  452. package/lib/voice/stt.test.mjs +0 -226
  453. package/lib/voice/voice.test.mjs +0 -990
  454. package/scripts/cadence/enqueue-cadence-tick.test.mjs +0 -187
  455. package/scripts/ci/check-docs-accuracy.test.mjs +0 -409
  456. package/scripts/ci/check-durable-write-seam.test.mjs +0 -90
  457. package/scripts/ci/check-no-build-artifacts.test.mjs +0 -71
  458. package/scripts/ci/check-no-residual-identity.test.mjs +0 -202
  459. package/scripts/ci/check-skill-packs.test.mjs +0 -495
  460. package/scripts/ci/check-subagent-frontmatter.test.mjs +0 -124
  461. package/scripts/ci/check.test.mjs +0 -194
  462. package/scripts/ci/conformance-org-api.test.mjs +0 -425
  463. package/scripts/cloud-relay/voice/relay-identity.test.mjs +0 -96
  464. package/scripts/collective/hook-runner.test.mjs +0 -173
  465. package/scripts/cost/fleet-digest.test.mjs +0 -207
  466. package/scripts/cost/track-claude-usage-pricing.test.mjs +0 -183
  467. package/scripts/cost/track-claude-usage.test.mjs +0 -148
  468. package/scripts/daemon/agent-daemon-board-mine.test.mjs +0 -96
  469. package/scripts/daemon/agent-daemon-design.test.mjs +0 -238
  470. package/scripts/daemon/agent-daemon-frontdoor.test.mjs +0 -60
  471. package/scripts/daemon/agent-daemon.test.mjs +0 -995
  472. package/scripts/daemon/assurance-e2e.test.mjs +0 -613
  473. package/scripts/daemon/assurance.test.mjs +0 -1791
  474. package/scripts/daemon/board-mirror.test.mjs +0 -165
  475. package/scripts/daemon/cadence-consumer-frontdoor.test.mjs +0 -393
  476. package/scripts/daemon/cadence-consumer-governance.test.mjs +0 -276
  477. package/scripts/daemon/cadence-consumer.test.mjs +0 -776
  478. package/scripts/daemon/cadence-handlers.test.mjs +0 -837
  479. package/scripts/daemon/classifier-identity.test.mjs +0 -137
  480. package/scripts/daemon/classifier.test.mjs +0 -266
  481. package/scripts/daemon/classify-kind.test.mjs +0 -40
  482. package/scripts/daemon/context-compiler.test.mjs +0 -406
  483. package/scripts/daemon/deliver.test.mjs +0 -564
  484. package/scripts/daemon/dispatcher-cooldown.test.mjs +0 -122
  485. package/scripts/daemon/dispatcher-governance.test.mjs +0 -1013
  486. package/scripts/daemon/dispatcher-resume.test.mjs +0 -166
  487. package/scripts/daemon/dispatcher-session-continuity.test.mjs +0 -365
  488. package/scripts/daemon/execution-ladder.test.mjs +0 -470
  489. package/scripts/daemon/goal-steward-cadence.test.mjs +0 -312
  490. package/scripts/daemon/inbox-deferral-session.test.mjs +0 -49
  491. package/scripts/daemon/inbox-deferral.test.mjs +0 -336
  492. package/scripts/daemon/inbox-wake.test.mjs +0 -199
  493. package/scripts/daemon/integration.test.mjs +0 -149
  494. package/scripts/daemon/lib/self-echo.test.mjs +0 -153
  495. package/scripts/daemon/lib/session-router.test.mjs +0 -554
  496. package/scripts/daemon/prompt-builder-preamble.test.mjs +0 -210
  497. package/scripts/daemon/prompt-builder.test.mjs +0 -556
  498. package/scripts/daemon/responder-cost.test.mjs +0 -68
  499. package/scripts/daemon/responder-history.test.mjs +0 -221
  500. package/scripts/daemon/sdk-version.test.mjs +0 -31
  501. package/scripts/daemon/session-lock.test.mjs +0 -252
  502. package/scripts/daemon/session-outcomes.test.mjs +0 -533
  503. package/scripts/daemon/typing-registry.test.mjs +0 -102
  504. package/scripts/hooks/pre-send-audit.test.mjs +0 -354
  505. package/scripts/huddle/huddle-prompt.test.mjs +0 -176
  506. package/scripts/local-triggers/autoupdate.test.mjs +0 -518
  507. package/scripts/local-triggers/generate-plists.test.mjs +0 -456
  508. package/scripts/media-generation/brand-clause.test.mjs +0 -135
  509. package/scripts/org/send-orgmail.first-contact.test.mjs +0 -102
  510. package/scripts/poller/inbox-privilege-injection.test.mjs +0 -167
  511. package/scripts/poller/inbox-scan-poller.test.mjs +0 -295
  512. package/scripts/poller/lib/cloud-relay-dedup.test.mjs +0 -133
  513. package/scripts/poller/slack-socket-mode.test.mjs +0 -805
  514. package/scripts/poller-launchd/install.test.mjs +0 -243
  515. package/scripts/restore-from-backup.test.mjs +0 -181
  516. package/scripts/session/feed.test.mjs +0 -196
  517. package/scripts/session/supervisor-sh.test.mjs +0 -218
  518. package/scripts/session/supervisor.test.mjs +0 -482
  519. package/scripts/setup/configure-macos.test.mjs +0 -306
  520. package/scripts/setup/gen-subagent-manifest.test.mjs +0 -124
  521. package/scripts/setup/generate-agent-package-json.test.mjs +0 -143
  522. package/scripts/setup/generate-capability.test.mjs +0 -134
  523. package/scripts/setup/init-agent.test.mjs +0 -370
  524. package/scripts/setup/init-skill-marketplace.test.mjs +0 -193
  525. package/scripts/vendor/sync-skill-packs.test.mjs +0 -103
  526. package/scripts/watchdog/memory-watchdog.test.mjs +0 -64
@@ -1,1207 +0,0 @@
1
- /**
2
- * model-router.test.mjs — node:test coverage for the model router.
3
- *
4
- * Tests write configs to a temp dir + load them via loadRoutingConfig so
5
- * we exercise the full file-system path. No real network, no real Claude.
6
- */
7
-
8
- import { test } from "node:test";
9
- import assert from "node:assert/strict";
10
- import { promises as fsp } from "fs";
11
- import { mkdirSync, rmSync, writeFileSync } from "node:fs";
12
- import { tmpdir } from "os";
13
- import { join } from "path";
14
- import { fileURLToPath } from "node:url";
15
-
16
- import {
17
- loadRoutingConfig,
18
- defaultRoutingConfig,
19
- resolveBackend,
20
- estimateCost,
21
- listBackends,
22
- describeBackend,
23
- modelFlagFor,
24
- requestFromClassifierResult,
25
- CONFIG_RELATIVE_PATH,
26
- resolveChain,
27
- validateRoutingConfig,
28
- } from "./model-router.mjs";
29
- import { loadCatalog } from "./model-router/catalog.mjs";
30
- // Read-only, to PROVE (not assume) that a provider the SDK does not bundle
31
- // inherits credential pooling from the catalog rather than needing new code.
32
- import { loadAuthProfiles, authEnvMapFromCatalog } from "./model-router/auth-profiles.mjs";
33
-
34
- async function makeAgentRoot() {
35
- const path = join(
36
- tmpdir(),
37
- `model-router-test-${process.pid}-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`
38
- );
39
- await fsp.mkdir(join(path, "config"), { recursive: true });
40
- return path;
41
- }
42
-
43
- async function rmRoot(path) {
44
- try { await fsp.rm(path, { recursive: true, force: true }); } catch { /* */ }
45
- }
46
-
47
- function writeJsonConfig(root, body) {
48
- // We write JSON (not YAML) so the tests don't depend on js-yaml being
49
- // installed in CI. The loader treats .json identically.
50
- writeFileSync(join(root, "config/model-routing.json"), JSON.stringify(body, null, 2));
51
- }
52
-
53
- const ANTHROPIC_BACKEND = {
54
- transport: "anthropic-cli",
55
- base_url: "https://api.anthropic.com",
56
- capabilities: ["thinking", "vision", "prompt_cache_1h", "tool_use", "parallel_tools", "long_context_1m"],
57
- pricing: { input_per_m: 3.0, output_per_m: 15.0 },
58
- // A CONFIG-DECLARED model map. Deliberately NOT the same ids as
59
- // ANTHROPIC_DEFAULT: these tests assert that a config's own map is what
60
- // resolveBackend serves, so pinning them to the built-in defaults would make
61
- // the assertions vacuous (they would pass even if the config map were being
62
- // ignored entirely). The built-in defaults get their own test below.
63
- models: {
64
- classifier: "claude-haiku-4-5-20251001",
65
- default: "config-declared-sonnet",
66
- premium: "config-declared-opus",
67
- fast: "claude-haiku-4-5-20251001",
68
- },
69
- };
70
-
71
- const MOONSHOT_BACKEND = {
72
- transport: "anthropic-native",
73
- base_url: "https://api.moonshot.ai/anthropic",
74
- auth_env: "MOONSHOT_API_KEY",
75
- capabilities: ["tool_use", "prompt_cache_hit", "long_context_262k"],
76
- pricing: { input_per_m: 0.95, output_per_m: 4.0, cache_hit_per_m: 0.16 },
77
- models: { default: "kimi-k2.6", fast: "kimi-k2.6" },
78
- };
79
-
80
- const QWEN_BACKEND = {
81
- transport: "openai-compat",
82
- base_url: "https://openrouter.ai/api/v1",
83
- auth_env: "OPENROUTER_API_KEY",
84
- capabilities: ["tool_use", "long_context_262k", "degraded_tool_use"], // mimic real-world OR bug
85
- pricing: { input_per_m: 0.22, output_per_m: 1.8 },
86
- models: { default: "qwen/qwen3-coder", fast: "qwen/qwen3-coder-flash" },
87
- };
88
-
89
- // ---------------------------------------------------------------------------
90
- // loadRoutingConfig
91
- // ---------------------------------------------------------------------------
92
-
93
- test("loadRoutingConfig returns null when no config exists", async () => {
94
- const root = await makeAgentRoot();
95
- try {
96
- assert.equal(loadRoutingConfig(root), null);
97
- } finally { await rmRoot(root); }
98
- });
99
-
100
- test("loadRoutingConfig parses JSON and synthesizes anthropic fallback", async () => {
101
- const root = await makeAgentRoot();
102
- try {
103
- writeJsonConfig(root, {
104
- backends: { moonshot: MOONSHOT_BACKEND },
105
- routing_policy: [{ default: true, backend: "moonshot" }],
106
- });
107
- const cfg = loadRoutingConfig(root);
108
- assert.ok(cfg, "config should load");
109
- // Anthropic synthesised because fallback_to_anthropic defaulted to true.
110
- assert.ok(cfg.backends.anthropic, "anthropic synthesised");
111
- assert.equal(cfg.backends.moonshot.base_url, "https://api.moonshot.ai/anthropic");
112
- assert.equal(cfg.routing_policy[0].backend, "moonshot");
113
- assert.equal(cfg.fallback_to_anthropic, true);
114
- } finally { await rmRoot(root); }
115
- });
116
-
117
- test("loadRoutingConfig respects explicit fallback_to_anthropic: false", async () => {
118
- const root = await makeAgentRoot();
119
- try {
120
- writeJsonConfig(root, {
121
- backends: { moonshot: MOONSHOT_BACKEND },
122
- routing_policy: [{ default: true, backend: "moonshot" }],
123
- fallback_to_anthropic: false,
124
- });
125
- const cfg = loadRoutingConfig(root);
126
- assert.equal(cfg.backends.anthropic, undefined, "anthropic NOT synthesised");
127
- assert.equal(cfg.fallback_to_anthropic, false);
128
- } finally { await rmRoot(root); }
129
- });
130
-
131
- test("loadRoutingConfig throws on malformed JSON", async () => {
132
- const root = await makeAgentRoot();
133
- try {
134
- writeFileSync(join(root, "config/model-routing.json"), "{ not valid json");
135
- assert.throws(() => loadRoutingConfig(root), /not valid|JSON|parse/);
136
- } finally { await rmRoot(root); }
137
- });
138
-
139
- test("loadRoutingConfig accepts $MAESTRO_ROUTING_CONFIG override", async () => {
140
- const root = await makeAgentRoot();
141
- const alt = await makeAgentRoot();
142
- try {
143
- const altPath = join(alt, "config/model-routing.json");
144
- writeFileSync(altPath, JSON.stringify({
145
- backends: { moonshot: MOONSHOT_BACKEND },
146
- routing_policy: [{ default: true, backend: "moonshot" }],
147
- }));
148
- const prev = process.env.MAESTRO_ROUTING_CONFIG;
149
- process.env.MAESTRO_ROUTING_CONFIG = altPath;
150
- try {
151
- const cfg = loadRoutingConfig(root);
152
- assert.ok(cfg.backends.moonshot);
153
- } finally {
154
- if (prev === undefined) delete process.env.MAESTRO_ROUTING_CONFIG;
155
- else process.env.MAESTRO_ROUTING_CONFIG = prev;
156
- }
157
- } finally {
158
- await rmRoot(root);
159
- await rmRoot(alt);
160
- }
161
- });
162
-
163
- // ---------------------------------------------------------------------------
164
- // resolveBackend
165
- // ---------------------------------------------------------------------------
166
-
167
- test("resolveBackend returns null when no config is loaded", () => {
168
- const r = resolveBackend({ agent_role: "responder" }, { config: null });
169
- assert.equal(r, null, "no config → no resolution (callers preserve current behaviour)");
170
- });
171
-
172
- test("resolveBackend picks default rule when nothing else matches", () => {
173
- const config = {
174
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
175
- routing_policy: [{ default: true, backend: "moonshot" }],
176
- fallback_to_anthropic: true,
177
- strip_attribution_header: true,
178
- disable_experimental_betas: true,
179
- };
180
- const r = resolveBackend({ agent_role: "responder", tier: "default" }, { config, env: { MOONSHOT_API_KEY: "msk-test" } });
181
- assert.equal(r.name, "moonshot");
182
- assert.equal(r.model, "kimi-k2.6");
183
- assert.equal(r.envForSpawn.ANTHROPIC_BASE_URL, "https://api.moonshot.ai/anthropic");
184
- assert.equal(r.envForSpawn.ANTHROPIC_AUTH_TOKEN, "msk-test");
185
- // Critical: API_KEY must be explicit empty string, not unset.
186
- assert.equal(r.envForSpawn.ANTHROPIC_API_KEY, "");
187
- assert.equal(r.envForSpawn.ANTHROPIC_MODEL, "kimi-k2.6");
188
- assert.equal(r.envForSpawn.CLAUDE_CODE_ATTRIBUTION_HEADER, "0");
189
- assert.equal(r.envForSpawn.CLAUDE_CODE_DISABLE_EXPERIMENTAL_BETAS, "1");
190
- });
191
-
192
- test("resolveBackend forces anthropic for sensitive roles via agent_role_in", () => {
193
- const config = {
194
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
195
- routing_policy: [
196
- { match: { agent_role_in: ["ceo_pre_pass", "audit", "decision_writer"] }, backend: "anthropic" },
197
- { default: true, backend: "qwen" },
198
- ],
199
- };
200
- const sensitive = resolveBackend({ agent_role: "ceo_pre_pass" }, { config });
201
- assert.equal(sensitive.name, "anthropic");
202
- assert.equal(sensitive.transport, "anthropic-cli");
203
- // anthropic-cli mode injects no env — current CLI behaviour preserved.
204
- assert.deepEqual(sensitive.envForSpawn, {});
205
-
206
- const routine = resolveBackend({ agent_role: "log_triage" }, { config, env: { OPENROUTER_API_KEY: "or-test" } });
207
- assert.equal(routine.name, "qwen");
208
- assert.equal(routine.envForSpawn.ANTHROPIC_AUTH_TOKEN, "or-test");
209
- });
210
-
211
- test("resolveBackend falls through when backend lacks capability (no_thinking)", () => {
212
- const config = {
213
- backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
214
- routing_policy: [
215
- { match: { needs_thinking: true }, backend: "qwen", fallback: ["anthropic"] },
216
- { default: true, backend: "qwen" },
217
- ],
218
- fallback_to_anthropic: true,
219
- };
220
- const r = resolveBackend({ agent_role: "responder", needs_thinking: true }, { config });
221
- // Qwen has no `thinking` cap → falls back to anthropic.
222
- assert.equal(r.name, "anthropic");
223
- assert.equal(r.tried[0].backend, "qwen");
224
- assert.equal(r.tried[0].reason, "no_thinking");
225
- });
226
-
227
- test("resolveBackend rejects degraded_tool_use when caller needs reliable tools", () => {
228
- const config = {
229
- backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
230
- routing_policy: [
231
- { match: { needs_tool_use: true }, backend: "qwen", fallback: ["anthropic"] },
232
- ],
233
- fallback_to_anthropic: true,
234
- };
235
- const r = resolveBackend({ agent_role: "responder", needs_tool_use: true }, { config });
236
- assert.equal(r.name, "anthropic");
237
- assert.equal(r.tried[0].backend, "qwen");
238
- assert.equal(r.tried[0].reason, "degraded_tool_use");
239
- });
240
-
241
- test("resolveBackend honours token_estimate_gte for long-context routing", () => {
242
- const config = {
243
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
244
- routing_policy: [
245
- { match: { token_estimate_gte: 60000 }, backend: "moonshot" },
246
- { default: true, backend: "anthropic" },
247
- ],
248
- };
249
- const short = resolveBackend({ token_estimate: 20000 }, { config });
250
- const long = resolveBackend({ token_estimate: 120000 }, { config, env: { MOONSHOT_API_KEY: "msk" } });
251
- assert.equal(short.name, "anthropic");
252
- assert.equal(long.name, "moonshot");
253
- });
254
-
255
- test("resolveBackend honours model_hint legacy values (opus/sonnet/haiku)", () => {
256
- const config = {
257
- backends: { anthropic: ANTHROPIC_BACKEND },
258
- routing_policy: [{ default: true, backend: "anthropic" }],
259
- };
260
- const opus = resolveBackend({ model_hint: "opus" }, { config });
261
- const sonnet = resolveBackend({ model_hint: "sonnet" }, { config });
262
- const haiku = resolveBackend({ model_hint: "haiku" }, { config });
263
- assert.equal(opus.model, "config-declared-opus");
264
- assert.equal(sonnet.model, "config-declared-sonnet");
265
- assert.equal(haiku.model, "claude-haiku-4-5-20251001");
266
- assert.equal(modelFlagFor(opus, { model_hint: "opus" }), "opus");
267
- assert.equal(modelFlagFor(sonnet, { model_hint: "sonnet" }), "sonnet");
268
- assert.equal(modelFlagFor(haiku, { model_hint: "haiku" }), "haiku");
269
- });
270
-
271
- // ---------------------------------------------------------------------------
272
- // Model freshness — the built-in Anthropic defaults (see the provenance block
273
- // above ANTHROPIC_DEFAULT in model-router.mjs).
274
- // ---------------------------------------------------------------------------
275
-
276
- test("defaultRoutingConfig ships the CURRENT Anthropic ids, not last quarter's", () => {
277
- // The floor an agent with no routing config lands on. A stale id here never
278
- // reached the wire (anthropic-cli sends the "opus"/"sonnet" shorthand), but it
279
- // DID reach resolved.model, which the cost ledger and telemetry store — so a
280
- // stale id mis-prices and mis-attributes every unrouted session.
281
- const models = defaultRoutingConfig().backends.anthropic.models;
282
- assert.equal(models.premium, "claude-opus-5");
283
- assert.equal(models.default, "claude-sonnet-5");
284
- // The fast/classifier tier did NOT move in the 2026-08-13 refresh.
285
- assert.equal(models.fast, "claude-haiku-4-5-20251001");
286
- assert.equal(models.classifier, "claude-haiku-4-5-20251001");
287
- // Guard the specific regression: no 4.x id survives anywhere in the map.
288
- for (const [tier, id] of Object.entries(models)) {
289
- assert.ok(!/^claude-(opus|sonnet)-4/.test(id), `${tier} still pinned to a retired id: ${id}`);
290
- }
291
- });
292
-
293
- test("resolveBackend serves the current ids through the no-config default path", () => {
294
- // End-to-end through the resolver, not just the constant: an agent with no
295
- // config file (useDefault) resolves premium/default to the -5 pair.
296
- const config = defaultRoutingConfig();
297
- assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-5");
298
- assert.equal(resolveBackend({ model_hint: "sonnet" }, { config }).model, "claude-sonnet-5");
299
- // …while the CLI flag stays version-agnostic, which is WHY the stale ids were
300
- // survivable on the wire and only corrupted the ledger.
301
- assert.equal(modelFlagFor(resolveBackend({ model_hint: "opus" }, { config }), { model_hint: "opus" }), "opus");
302
- });
303
-
304
- test("a config-declared models map still outranks the built-in defaults (the no-PR refresh path)", () => {
305
- // Refresh path #2 in the provenance block: when claude-opus-6 ships, a v1
306
- // agent adds backends.anthropic.models to config/model-routing.yaml and is
307
- // current WITHOUT an SDK release. normaliseConfig only injects the built-in
308
- // map when backends.anthropic is absent, so this must win outright.
309
- const config = {
310
- backends: { anthropic: { ...ANTHROPIC_BACKEND, models: { ...ANTHROPIC_BACKEND.models, premium: "claude-opus-99" } } },
311
- routing_policy: [{ default: true, backend: "anthropic" }],
312
- };
313
- assert.equal(resolveBackend({ model_hint: "opus" }, { config }).model, "claude-opus-99");
314
- });
315
-
316
- test("resolveBackend returns null when no backend satisfies request and fallback disabled", () => {
317
- const config = {
318
- backends: { qwen: QWEN_BACKEND },
319
- routing_policy: [{ default: true, backend: "qwen" }],
320
- fallback_to_anthropic: false,
321
- };
322
- const r = resolveBackend({ needs_thinking: true }, { config });
323
- assert.equal(r, null, "no fallback and no compatible backend → null");
324
- });
325
-
326
- test("resolveBackend tags fallback_reason when falling through to anthropic safety net", () => {
327
- const config = {
328
- backends: { anthropic: ANTHROPIC_BACKEND, qwen: QWEN_BACKEND },
329
- routing_policy: [{ default: true, backend: "qwen" }],
330
- fallback_to_anthropic: true,
331
- };
332
- const r = resolveBackend({ needs_thinking: true }, { config });
333
- assert.equal(r.name, "anthropic");
334
- assert.equal(r.fallback_reason, "no_compatible_backend");
335
- });
336
-
337
- // ---------------------------------------------------------------------------
338
- // estimateCost
339
- // ---------------------------------------------------------------------------
340
-
341
- test("estimateCost returns null without pricing data", () => {
342
- const fake = { pricing: {} };
343
- assert.equal(estimateCost({ token_estimate: 1000 }, fake), null);
344
- });
345
-
346
- test("estimateCost computes per-million-token cost without cache hits", () => {
347
- const fake = { pricing: { input_per_m: 3.0, output_per_m: 15.0 } };
348
- const cost = estimateCost({ token_estimate: 1_000_000, token_estimate_out: 500_000 }, fake);
349
- // 1M input @ $3 + 0.5M output @ $15 = $3 + $7.5 = $10.5
350
- assert.equal(cost, 10.5);
351
- });
352
-
353
- test("estimateCost honours cache_hit_per_m for Moonshot-style caching", () => {
354
- const fake = { pricing: { input_per_m: 0.95, output_per_m: 4.0, cache_hit_per_m: 0.16 } };
355
- // 1M tokens, 80% cache hit, no explicit output → defaults to 0.5M output
356
- const cost = estimateCost({ token_estimate: 1_000_000, cache_hit_ratio: 0.8 }, fake);
357
- // input: 200k @ 0.95 + 800k @ 0.16 = 0.19 + 0.128 = 0.318
358
- // output: 500k @ 4.0 = 2.0
359
- // total: 2.318
360
- assert.ok(Math.abs(cost - 2.318) < 0.001, `got ${cost}`);
361
- });
362
-
363
- // ---------------------------------------------------------------------------
364
- // listBackends / describeBackend
365
- // ---------------------------------------------------------------------------
366
-
367
- test("listBackends defaults to ['anthropic'] when no config", () => {
368
- assert.deepEqual(listBackends({ config: null }), ["anthropic"]);
369
- });
370
-
371
- test("listBackends returns all configured backend keys", () => {
372
- const config = {
373
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
374
- routing_policy: [],
375
- };
376
- assert.deepEqual(listBackends({ config }).sort(), ["anthropic", "moonshot", "qwen"]);
377
- });
378
-
379
- test("describeBackend returns model/transport/pricing for a backend", () => {
380
- const config = {
381
- backends: { moonshot: MOONSHOT_BACKEND },
382
- routing_policy: [],
383
- };
384
- const d = describeBackend("moonshot", { config });
385
- assert.equal(d.name, "moonshot");
386
- assert.equal(d.transport, "anthropic-native");
387
- assert.equal(d.base_url, "https://api.moonshot.ai/anthropic");
388
- assert.equal(d.models.default, "kimi-k2.6");
389
- });
390
-
391
- // ---------------------------------------------------------------------------
392
- // requestFromClassifierResult
393
- // ---------------------------------------------------------------------------
394
-
395
- test("requestFromClassifierResult preserves priority + model hint", () => {
396
- const req = requestFromClassifierResult(
397
- { priority: "critical", model: "opus", summary: "CEO request" },
398
- { role: "responder", source: "inbox" }
399
- );
400
- assert.equal(req.priority, "critical");
401
- assert.equal(req.model_hint, "opus");
402
- assert.equal(req.needs_thinking, true);
403
- assert.equal(req.needs_tool_use, true);
404
- assert.equal(req.agent_role, "responder");
405
- });
406
-
407
- test("requestFromClassifierResult does NOT set needs_thinking for routine sonnet items", () => {
408
- const req = requestFromClassifierResult(
409
- { priority: "normal", model: "sonnet" },
410
- { role: "responder" }
411
- );
412
- assert.equal(req.needs_thinking, undefined);
413
- assert.equal(req.model_hint, "sonnet");
414
- });
415
-
416
- // ---------------------------------------------------------------------------
417
- // Foot-gun fixes (audit W5 / W8 + kill switch)
418
- // ---------------------------------------------------------------------------
419
-
420
- test("W5: session work defaults needs_tool_use=true even for routine items", () => {
421
- // A normal-priority backlog item is still a full agent session — it must
422
- // declare tool use so a degraded_tool_use backend is rejected.
423
- const req = requestFromClassifierResult(
424
- { priority: "normal", model: "sonnet" },
425
- { role: "responder", source: "backlog" }
426
- );
427
- assert.equal(req.needs_tool_use, true, "session work needs tool use by default");
428
- });
429
-
430
- test("W5: one-shot lookups may opt out of needs_tool_use", () => {
431
- const req = requestFromClassifierResult(
432
- { priority: "normal", model: "sonnet" },
433
- { role: "lookup", oneShot: true }
434
- );
435
- assert.equal(req.needs_tool_use, false, "one-shot lookups opt out");
436
- });
437
-
438
- test("W5: shipped-example backlog item is kept off the degraded_tool_use backend", () => {
439
- // Mirrors scaffold/config/model-routing.yaml.example: a normal backlog item
440
- // matched `source: backlog → openrouter_qwen` (degraded_tool_use). With
441
- // needs_tool_use defaulting true, the capability gate now fires and the
442
- // request falls through to a tool-capable backend.
443
- const config = {
444
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND, qwen: QWEN_BACKEND },
445
- routing_policy: [
446
- { match: { source: "backlog" }, backend: "qwen", fallback: ["moonshot", "anthropic"] },
447
- { default: true, backend: "moonshot", fallback: ["anthropic"] },
448
- ],
449
- fallback_to_anthropic: true,
450
- };
451
- const req = requestFromClassifierResult(
452
- { priority: "normal", model: "sonnet" },
453
- { role: "responder", source: "backlog" }
454
- );
455
- const r = resolveBackend(req, { config, env: { MOONSHOT_API_KEY: "msk" } });
456
- assert.notEqual(r.name, "qwen", "must NOT route a tool-using session to degraded_tool_use backend");
457
- assert.equal(r.name, "moonshot");
458
- assert.equal(r.tried[0].backend, "qwen");
459
- assert.equal(r.tried[0].reason, "degraded_tool_use");
460
- });
461
-
462
- test("W8: candidate with missing auth env is skipped, not spawned with empty token", () => {
463
- const config = {
464
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
465
- routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
466
- fallback_to_anthropic: true,
467
- };
468
- // MOONSHOT_API_KEY deliberately absent from env.
469
- const r = resolveBackend({ agent_role: "responder", tier: "default" }, { config, env: {} });
470
- assert.equal(r.name, "anthropic", "skips moonshot, lands on anthropic safety net");
471
- assert.equal(r.tried[0].backend, "moonshot");
472
- assert.equal(r.tried[0].reason, "missing_auth_env");
473
- // The anthropic-cli safety net injects no empty credential.
474
- assert.deepEqual(r.envForSpawn, {});
475
- });
476
-
477
- test("W8: empty/whitespace auth env counts as absent and is skipped", () => {
478
- const config = {
479
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
480
- routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
481
- fallback_to_anthropic: true,
482
- };
483
- const r = resolveBackend({ agent_role: "responder" }, { config, env: { MOONSHOT_API_KEY: " " } });
484
- assert.equal(r.name, "anthropic");
485
- assert.equal(r.tried[0].reason, "missing_auth_env");
486
- });
487
-
488
- test("W8: all candidates missing auth + no fallback → null (never empty spawn)", () => {
489
- const config = {
490
- backends: { moonshot: MOONSHOT_BACKEND },
491
- routing_policy: [{ default: true, backend: "moonshot" }],
492
- fallback_to_anthropic: false,
493
- };
494
- const r = resolveBackend({ agent_role: "responder" }, { config, env: {} });
495
- assert.equal(r, null, "no creds anywhere and no fallback → no resolution");
496
- });
497
-
498
- test("W8: present auth env still resolves normally (no regression)", () => {
499
- const config = {
500
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
501
- routing_policy: [{ default: true, backend: "moonshot", fallback: ["anthropic"] }],
502
- fallback_to_anthropic: true,
503
- };
504
- const r = resolveBackend({ agent_role: "responder" }, { config, env: { MOONSHOT_API_KEY: "msk-ok" } });
505
- assert.equal(r.name, "moonshot");
506
- assert.equal(r.envForSpawn.ANTHROPIC_AUTH_TOKEN, "msk-ok");
507
- });
508
-
509
- test("kill switch: MAESTRO_ROUTER_FORCE_ANTHROPIC=1 short-circuits to stock behaviour", () => {
510
- const config = {
511
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
512
- routing_policy: [{ default: true, backend: "moonshot" }],
513
- fallback_to_anthropic: true,
514
- };
515
- // Even with a valid Moonshot key and a default-moonshot policy, the kill
516
- // switch returns null — the same path as no config, so callers preserve
517
- // the legacy keychain-OAuth `claude --print` behaviour.
518
- const r = resolveBackend(
519
- { agent_role: "responder" },
520
- { config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1", MOONSHOT_API_KEY: "msk" } }
521
- );
522
- assert.equal(r, null, "kill switch → null (stock Anthropic CLI behaviour)");
523
- });
524
-
525
- test("kill switch: accepts 'true' as well as '1'", () => {
526
- const config = {
527
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
528
- routing_policy: [{ default: true, backend: "moonshot" }],
529
- };
530
- const r = resolveBackend(
531
- { agent_role: "responder" },
532
- { config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "true", MOONSHOT_API_KEY: "msk" } }
533
- );
534
- assert.equal(r, null);
535
- });
536
-
537
- test("kill switch: any other value does NOT short-circuit", () => {
538
- const config = {
539
- backends: { anthropic: ANTHROPIC_BACKEND, moonshot: MOONSHOT_BACKEND },
540
- routing_policy: [{ default: true, backend: "moonshot" }],
541
- };
542
- const r = resolveBackend(
543
- { agent_role: "responder" },
544
- { config, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "0", MOONSHOT_API_KEY: "msk" } }
545
- );
546
- assert.equal(r.name, "moonshot", "0/unset → routing still active");
547
- });
548
-
549
- // ---------------------------------------------------------------------------
550
- // defaultRoutingConfig wiring (audit L21)
551
- // ---------------------------------------------------------------------------
552
-
553
- test("defaultRoutingConfig returns a usable Anthropic-only config", () => {
554
- const cfg = defaultRoutingConfig();
555
- assert.ok(cfg, "default config should be an object");
556
- assert.ok(cfg.backends.anthropic, "default config has an anthropic backend");
557
- assert.equal(cfg.backends.anthropic.transport, "anthropic-cli");
558
- assert.equal(cfg.fallback_to_anthropic, true);
559
- // resolveBackend must accept it and preserve current CLI behaviour (no env).
560
- const r = resolveBackend({ agent_role: "responder" }, { config: cfg });
561
- assert.equal(r.name, "anthropic");
562
- assert.equal(r.transport, "anthropic-cli");
563
- assert.deepEqual(r.envForSpawn, {}, "anthropic-cli default injects no env");
564
- });
565
-
566
- test("defaultRoutingConfig returns a fresh, mutation-safe copy each call", () => {
567
- const a = defaultRoutingConfig();
568
- const b = defaultRoutingConfig();
569
- assert.notEqual(a, b, "each call returns a distinct object");
570
- a.backends.anthropic.models.default = "MUTATED";
571
- assert.notEqual(
572
- b.backends.anthropic.models.default,
573
- "MUTATED",
574
- "mutating one copy must not affect another",
575
- );
576
- });
577
-
578
- test("loadRoutingConfig still returns null with no config and no useDefault flag", async () => {
579
- const root = await makeAgentRoot();
580
- try {
581
- assert.equal(loadRoutingConfig(root), null, "default contract preserved");
582
- } finally { await rmRoot(root); }
583
- });
584
-
585
- test("loadRoutingConfig({ useDefault: true }) returns the default when no file exists", async () => {
586
- const root = await makeAgentRoot();
587
- try {
588
- const cfg = loadRoutingConfig(root, { useDefault: true });
589
- assert.ok(cfg, "useDefault yields a config instead of null");
590
- assert.ok(cfg.backends.anthropic, "it is the anthropic default");
591
- assert.equal(cfg.fallback_to_anthropic, true);
592
- } finally { await rmRoot(root); }
593
- });
594
-
595
- test("loadRoutingConfig prefers a real file over the useDefault fallback", async () => {
596
- const root = await makeAgentRoot();
597
- try {
598
- writeJsonConfig(root, {
599
- backends: { moonshot: MOONSHOT_BACKEND },
600
- routing_policy: [{ default: true, backend: "moonshot" }],
601
- });
602
- const cfg = loadRoutingConfig(root, { useDefault: true });
603
- assert.ok(cfg.backends.moonshot, "real config wins over the default fallback");
604
- } finally { await rmRoot(root); }
605
- });
606
-
607
- // ---------------------------------------------------------------------------
608
- // Constants exported
609
- // ---------------------------------------------------------------------------
610
-
611
- test("CONFIG_RELATIVE_PATH points at config/model-routing.yaml", () => {
612
- assert.equal(CONFIG_RELATIVE_PATH, "config/model-routing.yaml");
613
- });
614
-
615
- // ===========================================================================
616
- // resolveChain — the v2 policy brain (additive; v1 above stays byte-compatible)
617
- // ===========================================================================
618
-
619
- // One shared bundled catalog snapshot (read-only) for the resolve tests. Loaded
620
- // from the framework's own lib/model-router/catalog/*.yaml so the refs are real.
621
- // This file lives at <root>/lib/, so the framework root is one dir up.
622
- const MAESTRO_ROOT_FOR_TESTS = () =>
623
- join(fileURLToPath(new URL(".", import.meta.url)), "..");
624
- const CATALOG = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {});
625
-
626
- const V2_CONFIG = Object.freeze({
627
- schema_version: 2,
628
- aliases: {
629
- frontier: "anthropic/claude-opus-4-8",
630
- default: "anthropic/claude-sonnet-4-6",
631
- fast: "anthropic/claude-haiku-4-5",
632
- cheap: "deepseek/deepseek-v4-flash",
633
- "cheap-session": "moonshot/kimi-k2.6",
634
- },
635
- defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
636
- backends: {
637
- anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
638
- deepseek: { allowed_data_classes: ["public"] },
639
- moonshot: { allowed_data_classes: ["public"] },
640
- },
641
- routing_policy: [
642
- { match: { agent_role_in: ["regulatory", "audit"] }, chain: ["frontier", "default"], pin: true },
643
- { match: { task_class: "classify.inbox" }, harness: "direct", chain: ["fast", "cheap", "rules"], needs_tool_use: false },
644
- { match: { task_class: "lookup.gmail" }, harness: "direct", chain: ["fast", "cheap"], needs_tool_use: false },
645
- { match: { source: "backlog", data_class: "public" }, chain: ["cheap", "default"] },
646
- { match: { token_estimate_gte: 250000 }, chain: ["default", "frontier"] },
647
- { default: true, chain: ["default", "fast"] },
648
- ],
649
- budget_ladder: {
650
- "75": { downgrade_tiers: 1 },
651
- "90": { force_alias: "cheap_or_fast" },
652
- "100": { essential_only: true },
653
- },
654
- fallback_to_anthropic: true,
655
- });
656
-
657
- const CLOCK = () => 1_718_000_000_000;
658
- function rc(req, extra = {}) {
659
- return resolveChain(req, { catalog: CATALOG, config: V2_CONFIG, env: {}, now: CLOCK, ...extra });
660
- }
661
-
662
- // ── rule match + chain build ───────────────────────────────────────────────
663
-
664
- test("resolveChain: default rule routes a session to the default workhorse", () => {
665
- const d = rc({ task_class: "session.responder", source: "inbox", data_class: "sensitive", token_estimate: 5000 });
666
- assert.equal(d.chosen.provider, "anthropic");
667
- assert.equal(d.chosen.model, "claude-sonnet-4-6");
668
- assert.equal(d.chosen.harness, "session");
669
- assert.ok(d.decision_id, "every decision has an id");
670
- assert.ok(d.explain && typeof d.explain === "string", "explain is mandatory");
671
- });
672
-
673
- test("resolveChain: first-match picks the regulatory pin rule (frontier)", () => {
674
- const d = rc({ agent_role: "regulatory", task_class: "decision.writer", data_class: "sensitive", token_estimate: 1000 });
675
- assert.equal(d.chosen.model, "claude-opus-4-8", "regulatory → frontier");
676
- // The visible chain leads with the matched rule's first alias.
677
- assert.equal(d.chain[0].ref, "anthropic/claude-opus-4-8");
678
- });
679
-
680
- test("resolveChain: classify.inbox resolves direct harness + tool-less", () => {
681
- // Anthropic direct needs the key; with a key present, fast (haiku) is chosen.
682
- const d = resolveChain(
683
- { task_class: "classify.inbox", source: "inbox", data_class: "public", token_estimate: 1500 },
684
- { catalog: CATALOG, config: V2_CONFIG, env: { ANTHROPIC_API_KEY: "ak" }, now: CLOCK }
685
- );
686
- assert.equal(d.chosen.harness, "direct");
687
- assert.equal(d.chosen.model, "claude-haiku-4-5");
688
- });
689
-
690
- test("resolveChain: the chain[] carries the failover tail, not just the winner", () => {
691
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 });
692
- // default chain is [default, fast]; both Anthropic session rows survive.
693
- const refs = d.chain.map((c) => c.ref);
694
- assert.deepEqual(refs, ["anthropic/claude-sonnet-4-6", "anthropic/claude-haiku-4-5"]);
695
- });
696
-
697
- // ── credential gate (key-absence-as-enforcement, §7.1) ──────────────────────
698
-
699
- test("resolveChain: a third-party row without its key is skipped (missing_credential)", () => {
700
- // lookup.gmail is harness:direct, needs_tool_use:false so deepseek (grade B)
701
- // clears the capability gate — the ONLY reason left to skip it is the absent
702
- // DEEPSEEK_API_KEY (key-absence-as-enforcement, §7.1). Anthropic haiku direct
703
- // also needs its key (absent) so the chain ends on the Anthropic safety net.
704
- const d = rc({ task_class: "lookup.gmail", data_class: "public", token_estimate: 1000 });
705
- const skipped = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
706
- assert.ok(skipped, "deepseek recorded in tried[]");
707
- assert.equal(skipped.reason, "missing_credential");
708
- });
709
-
710
- test("resolveChain: present third-party key still gated by grade for sessions (B<A)", () => {
711
- // deepseek is grade B; a session needing tools requires grade A (§6.6), so even
712
- // WITH the key it is skipped grade_too_low and we fall to Anthropic.
713
- const d = resolveChain(
714
- { task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
715
- { catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
716
- );
717
- assert.equal(d.chosen.provider, "anthropic");
718
- const skipped = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
719
- assert.equal(skipped.reason, "grade_too_low");
720
- });
721
-
722
- test("resolveChain: grade-B row IS reachable on the direct (tool-less) lane", () => {
723
- // lookup.gmail is harness:direct, needs_tool_use:false; deepseek (B) is fine.
724
- // Anthropic haiku (direct) needs ANTHROPIC_API_KEY (absent) so it is skipped,
725
- // landing on deepseek with its key present.
726
- const d = resolveChain(
727
- { task_class: "lookup.gmail", data_class: "public", token_estimate: 1000 },
728
- { catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
729
- );
730
- assert.equal(d.chosen.provider, "deepseek");
731
- assert.equal(d.chosen.harness, "direct");
732
- });
733
-
734
- // ── data_class gate (deny-by-default, §7.6) ─────────────────────────────────
735
-
736
- test("resolveChain: sensitive data never routes to a public-only backend", () => {
737
- // Even with the key AND a direct lane, deepseek (allowed: public) is denied for
738
- // a sensitive request and recorded as data_class_denied.
739
- const cfg = { ...V2_CONFIG, routing_policy: [{ match: { task_class: "lookup.gmail" }, harness: "direct", chain: ["cheap", "fast"], needs_tool_use: false }, { default: true, chain: ["default"] }] };
740
- const d = resolveChain(
741
- { task_class: "lookup.gmail", data_class: "sensitive", token_estimate: 1000 },
742
- { catalog: CATALOG, config: cfg, env: { DEEPSEEK_API_KEY: "dk", ANTHROPIC_API_KEY: "ak" }, now: CLOCK }
743
- );
744
- assert.equal(d.chosen.provider, "anthropic", "sensitive falls back to anthropic");
745
- const denied = d.tried.find((t) => t.ref === "deepseek/deepseek-v4-flash");
746
- assert.equal(denied.reason, "data_class_denied");
747
- });
748
-
749
- // ── breaker gate (health.isOpen) ────────────────────────────────────────────
750
-
751
- test("resolveChain: an open breaker skips the candidate (breaker_open in tried[])", () => {
752
- // Inject an isOpen that reports the sonnet model breaker open; chain falls to
753
- // the next survivor (haiku).
754
- const isOpen = (key) => ({
755
- open: key === "anthropic:claude-sonnet-4-6",
756
- until: 1_718_000_100_000,
757
- reason: "rate_limit",
758
- strikes: 1,
759
- });
760
- const d = resolveChain(
761
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
762
- { catalog: CATALOG, config: V2_CONFIG, env: {}, now: CLOCK, isOpen }
763
- );
764
- assert.equal(d.chosen.model, "claude-haiku-4-5", "skips open sonnet, lands on haiku");
765
- const skipped = d.tried.find((t) => t.ref === "anthropic/claude-sonnet-4-6");
766
- assert.equal(skipped.reason, "breaker_open");
767
- });
768
-
769
- // ── context gate ────────────────────────────────────────────────────────────
770
-
771
- test("resolveChain: a too-large request skips a small-context row (context_overflow)", () => {
772
- // haiku context_tokens is 190000; a 250k request matches the long-context rule
773
- // [default, frontier] (both 950k) and never even tries a small row. To assert
774
- // the gate directly, force a chain through haiku for a huge prompt.
775
- const cfg = { ...V2_CONFIG, routing_policy: [{ default: true, chain: ["fast", "default"] }] };
776
- const d = resolveChain(
777
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 500000 },
778
- { catalog: CATALOG, config: cfg, env: {}, now: CLOCK }
779
- );
780
- assert.equal(d.chosen.model, "claude-sonnet-4-6", "haiku too small → sonnet");
781
- const skipped = d.tried.find((t) => t.ref === "anthropic/claude-haiku-4-5");
782
- assert.equal(skipped.reason, "context_overflow");
783
- });
784
-
785
- test("resolveChain: long-context rule routes 250k+ to a 1M-ctx row", () => {
786
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 300000 });
787
- assert.equal(d.chosen.model, "claude-sonnet-4-6");
788
- });
789
-
790
- // ── envForSpawn (session retarget for Kimi/DeepSeek) ─────────────────────────
791
-
792
- test("resolveChain: a third-party SESSION row builds the ANTHROPIC_BASE_URL retarget env", () => {
793
- // Allow a grade override so kimi (B) can host a session, and route public work
794
- // to it with the key present.
795
- const cfg = {
796
- ...V2_CONFIG,
797
- catalog_overrides: undefined,
798
- routing_policy: [{ match: { source: "backlog" }, chain: ["cheap-session", "default"] }, { default: true, chain: ["default"] }],
799
- };
800
- // Override kimi grade to A via catalog_overrides so the session grade gate passes.
801
- const cat = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
802
- agentConfig: { catalog_overrides: [{ ref: "moonshot/kimi-k2.6", tool_reliability: "A" }] },
803
- });
804
- const d = resolveChain(
805
- { task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
806
- { catalog: cat, config: cfg, env: { MOONSHOT_API_KEY: "mk-test" }, now: CLOCK }
807
- );
808
- assert.equal(d.chosen.provider, "moonshot");
809
- assert.equal(d.chosen.transport, "anthropic-native");
810
- assert.equal(d.envForSpawn.ANTHROPIC_BASE_URL, "https://api.moonshot.ai/anthropic");
811
- assert.equal(d.envForSpawn.ANTHROPIC_AUTH_TOKEN, "mk-test");
812
- assert.equal(d.envForSpawn.ANTHROPIC_API_KEY, "", "empty string, NOT unset (keychain-fallthrough guard)");
813
- assert.equal(d.envForSpawn.ANTHROPIC_MODEL, "kimi-k2.6");
814
- });
815
-
816
- test("resolveChain: an Anthropic session injects NO retarget env (stock CLI)", () => {
817
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 });
818
- assert.deepEqual(d.envForSpawn, {});
819
- });
820
-
821
- // ── budget ladder ────────────────────────────────────────────────────────────
822
-
823
- test("resolveChain: band ≥75 downgrades one tier (default→fast) for non-pinned work", () => {
824
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, budget_band: 75 });
825
- assert.equal(d.chosen.model, "claude-haiku-4-5", "default downgraded to fast at band 75");
826
- assert.match(d.explain, /band 75/);
827
- });
828
-
829
- test("resolveChain: a pinned rule is exempt from the budget ladder", () => {
830
- // regulatory rule sets pin:true on the request indirectly via the rule pin flag;
831
- // pass pin on the request to exercise the exemption path.
832
- const d = rc({ agent_role: "regulatory", task_class: "decision", data_class: "sensitive", token_estimate: 1000, budget_band: 90, pin: { ref: "anthropic/claude-opus-4-8" } });
833
- assert.equal(d.chosen.model, "claude-opus-4-8", "pin beats the band-90 downgrade");
834
- });
835
-
836
- // ── affinity pin ─────────────────────────────────────────────────────────────
837
-
838
- test("resolveChain: a live affinity pin is moved to the chain head when it passes gates", () => {
839
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, session_key: "thread-1", pin: { ref: "anthropic/claude-haiku-4-5" } });
840
- assert.equal(d.chosen.model, "claude-haiku-4-5");
841
- assert.equal(d.chain[0].ref, "anthropic/claude-haiku-4-5", "pin at head");
842
- assert.match(d.explain, /pin /);
843
- });
844
-
845
- test("resolveChain: a pin that fails a gate emits pin_overridden and routes normally", () => {
846
- // Pin a public-only deepseek for sensitive work → data_class_denied → overridden.
847
- const d = resolveChain(
848
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000, pin: { ref: "deepseek/deepseek-v4-flash" } },
849
- { catalog: CATALOG, config: V2_CONFIG, env: { DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
850
- );
851
- assert.equal(d.chosen.provider, "anthropic", "overridden pin falls to normal routing");
852
- const overridden = d.tried.find((t) => String(t.reason).startsWith("pin_overridden"));
853
- assert.ok(overridden, "pin_overridden recorded in tried[]");
854
- });
855
-
856
- // ── kill switch ──────────────────────────────────────────────────────────────
857
-
858
- test("resolveChain: MAESTRO_ROUTER_FORCE_ANTHROPIC forces an Anthropic-only chain", () => {
859
- const d = resolveChain(
860
- { task_class: "backlog.work", source: "backlog", data_class: "public", token_estimate: 1000 },
861
- { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1", DEEPSEEK_API_KEY: "dk" }, now: CLOCK }
862
- );
863
- assert.equal(d.chosen.provider, "anthropic");
864
- assert.equal(d.audit.fallback_reason, "kill_switch");
865
- assert.match(d.explain, /kill switch/);
866
- // No third-party retarget env under the kill switch.
867
- assert.deepEqual(d.envForSpawn, {});
868
- });
869
-
870
- test("resolveChain: kill switch picks frontier for critical/thinking work", () => {
871
- const d = resolveChain(
872
- { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
873
- { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "true" }, now: CLOCK }
874
- );
875
- assert.equal(d.chosen.model, "claude-opus-4-8");
876
- });
877
-
878
- // ── inert / no-config collapse ───────────────────────────────────────────────
879
-
880
- test("resolveChain: no v2 config collapses to the Anthropic safety net", () => {
881
- const d = resolveChain(
882
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
883
- { catalog: CATALOG, config: null, env: {}, now: CLOCK }
884
- );
885
- assert.equal(d.chosen.provider, "anthropic");
886
- assert.equal(d.audit.fallback_reason, "no_v2_config");
887
- });
888
-
889
- // ── safety-net model freshness (pickAnthropicRow's ordered preference) ───────
890
- //
891
- // The safety net fires on the three paths that have no chain to walk: the kill
892
- // switch, "no v2 config", and "nothing survived the gates". It is the ONE place
893
- // resolve.mjs still names Anthropic model ids, so it is the one place that can
894
- // go stale. lookupModel is an exact byRef hit (no replaced_by chasing), so the
895
- // preference list carries the current -5 ids AND the 4-x succession fallback.
896
-
897
- /** The bundled catalog plus synthetic claude-*-5 rows (what ships next). */
898
- const CATALOG_WITH_5 = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
899
- agentConfig: {
900
- catalog_overrides: [
901
- {
902
- ref: "anthropic/claude-opus-5",
903
- status: "available",
904
- context_tokens: 950000,
905
- max_tokens: 128000,
906
- tool_reliability: "A",
907
- harness: { session: true, direct: true, batch: true },
908
- cost: { input: 5.0, output: 25.0, cache_read: 0.5, cache_write: 10.0 },
909
- },
910
- {
911
- ref: "anthropic/claude-sonnet-5",
912
- status: "available",
913
- context_tokens: 950000,
914
- max_tokens: 64000,
915
- tool_reliability: "A",
916
- harness: { session: true, direct: true, batch: true },
917
- cost: { input: 3.0, output: 15.0, cache_read: 0.3, cache_write: 6.0 },
918
- },
919
- ],
920
- },
921
- });
922
-
923
- test("resolveChain: the safety net prefers claude-opus-5 / claude-sonnet-5 when the catalog has them", () => {
924
- const frontier = resolveChain(
925
- { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
926
- { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
927
- );
928
- assert.equal(frontier.chosen.model, "claude-opus-5", "thinking/critical → current frontier, not opus-4-8");
929
-
930
- const workhorse = resolveChain(
931
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
932
- { catalog: CATALOG_WITH_5, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
933
- );
934
- assert.equal(workhorse.chosen.model, "claude-sonnet-5", "ordinary work → current workhorse, not sonnet-4-6");
935
- });
936
-
937
- test("resolveChain: the safety net falls back through succession when the catalog has no -5 rows", () => {
938
- // This is why the 4-x refs stay in the preference list. Delete them and this
939
- // does NOT fail loudly — it degrades to `models.find()` over the bundled rows,
940
- // i.e. whatever sits first in anthropic.yaml (opus-4-8), which would hand
941
- // ORDINARY work the frontier model and quietly quadruple its cost.
942
- const d = resolveChain(
943
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
944
- { catalog: CATALOG, config: V2_CONFIG, env: { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }, now: CLOCK }
945
- );
946
- assert.equal(d.chosen.model, "claude-sonnet-4-6", "succession fallback, ordered — NOT the first row in the file");
947
- // Flip this to claude-sonnet-5 (and drop the 4-x refs from pickAnthropicRow)
948
- // the day lib/model-router/catalog/anthropic.yaml carries the -5 rows.
949
- });
950
-
951
- test("the SHIPPED catalog is what decides the wire value — the -5 refresh is NOT done", () => {
952
- // THE HONEST STATE OF THIS REFRESH, pinned so it cannot be mistaken for
953
- // finished. The test above asserts `chosen.model`, which reads as bookkeeping.
954
- // This one asserts `spawnArgs.modelFlag` — the string spawn.mjs pushes as
955
- // `--model` — against the catalog the SDK actually ships, with `config: null`,
956
- // which is the DEFAULT state of every agent because this repo contains no
957
- // config/model-routing.yaml at all.
958
- //
959
- // Unlike the v1 lane (model-router.mjs's modelFlagFor sends the version-
960
- // agnostic "opus"/"sonnet" shorthand), resolve.mjs's modelFlagFor returns
961
- // row.id verbatim. So on v2 a stale catalog row is a stale id ON THE WIRE, not
962
- // merely a mis-stamped ledger entry.
963
- //
964
- // WHEN THIS FAILS: someone added the -5 rows to anthropic.yaml. Good — that is
965
- // the missing step. Update the expectations here, and DELETE the 4-x refs from
966
- // pickAnthropicRow's preference lists in the same change, or the museum the
967
- // provenance block warns about starts accumulating.
968
- const cases = [
969
- ["session.responder", {}, "claude-sonnet-4-6"],
970
- ["decision", { needs_thinking: true }, "claude-opus-4-8"],
971
- ];
972
- for (const [task_class, extra, expected] of cases) {
973
- const req = { task_class, data_class: "sensitive", token_estimate: 1000, ...extra };
974
- for (const [label, env] of [
975
- ["kill switch", { MAESTRO_ROUTER_FORCE_ANTHROPIC: "1" }],
976
- ["no v2 config", {}],
977
- ]) {
978
- const d = resolveChain(req, { catalog: CATALOG, config: null, env, now: CLOCK });
979
- assert.equal(
980
- d.spawnArgs.modelFlag,
981
- expected,
982
- `${label}/${task_class}: --model is the literal catalog id, and the bundled catalog is still 4-x`
983
- );
984
- // Not the shorthand: nothing downstream re-resolves this to a current build.
985
- assert.ok(!["opus", "sonnet", "haiku"].includes(d.spawnArgs.modelFlag));
986
- }
987
- }
988
- });
989
-
990
- test("resolveChain: with NO catalog rows at all, the synthetic row's id and ref agree", () => {
991
- // The last-ditch path: no catalog, so the id goes straight onto `claude
992
- // --model`. It used to emit {model: "claude-opus-4-8", ref:
993
- // "anthropic/claude-sonnet-4-6"} for a thinking request — and ref is what
994
- // feeds cacheAffinityKey and chain[0].ref, so the pin and the audit trail
995
- // both named a different model than the one that ran.
996
- const emptyDir = join(tmpdir(), `model-router-empty-catalog-${process.pid}-${Date.now()}`);
997
- mkdirSync(emptyDir, { recursive: true });
998
- try {
999
- const EMPTY = loadCatalog(emptyDir, { bundledDir: emptyDir });
1000
- assert.equal(EMPTY.models.length, 0, "fixture really is an empty catalog");
1001
-
1002
- const d = resolveChain(
1003
- { task_class: "decision", needs_thinking: true, data_class: "sensitive", token_estimate: 1000 },
1004
- { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1005
- );
1006
- assert.equal(d.chosen.model, "claude-opus-5");
1007
- assert.equal(d.chain[0].ref, "anthropic/claude-opus-5", "ref tracks id");
1008
- assert.match(d.cacheAffinityKey, /anthropic\/claude-opus-5$/);
1009
-
1010
- const ordinary = resolveChain(
1011
- { task_class: "session.responder", data_class: "sensitive", token_estimate: 1000 },
1012
- { catalog: EMPTY, config: null, env: {}, now: CLOCK }
1013
- );
1014
- assert.equal(ordinary.chosen.model, "claude-sonnet-5");
1015
- assert.equal(ordinary.chain[0].ref, "anthropic/claude-sonnet-5");
1016
- } finally {
1017
- rmSync(emptyDir, { recursive: true, force: true });
1018
- }
1019
- });
1020
-
1021
- // ── cheap tier: an UNBUNDLED provider (xAI/Grok) is routable end to end ──────
1022
- //
1023
- // The SDK bundles catalog rows for anthropic/deepseek/moonshot/qwen only. Grok
1024
- // is NOT bundled — but "not bundled" is not "not routable": catalog.mjs's
1025
- // highest authority layer (config/model-routing.yaml `catalog_overrides:`) may
1026
- // introduce a whole provider, and resolveChain gates it exactly like a bundled
1027
- // one. These tests prove the full path — row → alias → chain → credential gate
1028
- // → chosen — so the remaining work is a bundled YAML file, not plumbing.
1029
-
1030
- const XAI_PROVIDER_DOC = {
1031
- provider: "xai",
1032
- auth_env: "XAI_API_KEY",
1033
- endpoints: { openai: "https://api.x.ai/v1", anthropic: "https://api.x.ai/anthropic" },
1034
- data_residency: "us",
1035
- models: [
1036
- {
1037
- id: "grok-4-fast",
1038
- status: "available",
1039
- context_window: 2000000,
1040
- context_tokens: 1900000,
1041
- max_tokens: 30000,
1042
- cost: { input: 0.2, output: 0.5 },
1043
- cost_provenance: { source: "unverified", fetched: "2026-08-13", volatile: true },
1044
- compat: { thinking_format: "openai", cache_control: "implicit" },
1045
- // Grade B like every other third-party row: good enough for the direct
1046
- // lane, structurally barred from hosting a tool-using session (§6.6).
1047
- tool_reliability: "B",
1048
- harness: { session: true, direct: true, batch: false },
1049
- },
1050
- ],
1051
- };
1052
-
1053
- const CATALOG_WITH_XAI = loadCatalog(MAESTRO_ROOT_FOR_TESTS(), {
1054
- agentConfig: { catalog_overrides: XAI_PROVIDER_DOC },
1055
- });
1056
-
1057
- const XAI_CONFIG = Object.freeze({
1058
- schema_version: 2,
1059
- aliases: { cheap: "xai/grok-4-fast", default: "anthropic/claude-sonnet-4-6" },
1060
- defaults: { needs_tool_use_for_sessions: true, data_class: "sensitive", cache_ttl: "1h" },
1061
- backends: {
1062
- anthropic: { allowed_data_classes: ["public", "internal", "sensitive"] },
1063
- xai: { allowed_data_classes: ["public"] },
1064
- },
1065
- routing_policy: [
1066
- { match: { task_class: "lookup.web" }, harness: "direct", chain: ["cheap", "default"], needs_tool_use: false },
1067
- { default: true, chain: ["default"] },
1068
- ],
1069
- fallback_to_anthropic: true,
1070
- });
1071
-
1072
- test("catalog_overrides can introduce xAI/Grok — loudly, as a warning, never silently", () => {
1073
- const row = CATALOG_WITH_XAI.byRef.get("xai/grok-4-fast");
1074
- assert.ok(row, "grok row present after the agent-config merge");
1075
- assert.equal(row.provider, "xai");
1076
- assert.equal(row.auth_env, "XAI_API_KEY", "carries its OWN auth env, not a borrowed one");
1077
- assert.equal(row.endpoints.anthropic, "https://api.x.ai/anthropic");
1078
- assert.ok(
1079
- CATALOG_WITH_XAI.warnings.some((w) => /non-bundled provider "xai"/.test(w.warning)),
1080
- "introducing an unbundled provider warns (typo guard) — it does not fail"
1081
- );
1082
- });
1083
-
1084
- test("resolveChain routes a cheap-tier task to Grok when its own key is present", () => {
1085
- const d = resolveChain(
1086
- { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1087
- { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1088
- );
1089
- assert.equal(d.chosen.provider, "xai");
1090
- assert.equal(d.chosen.model, "grok-4-fast");
1091
- assert.equal(d.chosen.wire, "openai", "third-party direct call speaks the openai wire");
1092
- });
1093
-
1094
- test("resolveChain skips Grok without XAI_API_KEY (key-absence-as-enforcement, §7.1)", () => {
1095
- const d = resolveChain(
1096
- { task_class: "lookup.web", data_class: "public", token_estimate: 2000 },
1097
- { catalog: CATALOG_WITH_XAI, config: XAI_CONFIG, env: {}, now: CLOCK }
1098
- );
1099
- const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1100
- assert.ok(skipped, "grok recorded in tried[]");
1101
- assert.equal(skipped.reason, "missing_credential");
1102
- assert.notEqual(d.chosen.provider, "xai");
1103
- });
1104
-
1105
- test("Grok inherits auth-profile key rotation from the catalog, with no per-provider code", () => {
1106
- // auth-profiles derives provider → auth_env from the CATALOG, so a provider
1107
- // the SDK has never heard of gets pooled multi-key rotation for free. This is
1108
- // the "their own auth profiles" half of the cheap-tier requirement.
1109
- const profiles = loadAuthProfiles(null, {
1110
- configDoc: null, // bypass the FS
1111
- env: { XAI_API_KEY: "k1", XAI_API_KEY_2: "k2", DEEPSEEK_API_KEY: "d1" },
1112
- authEnvByProvider: authEnvMapFromCatalog(CATALOG_WITH_XAI),
1113
- });
1114
- assert.deepEqual(profiles.keysFor("xai"), ["k1", "k2"]);
1115
- assert.equal(profiles.isPooled("xai"), true, "two keys ⇒ rotation is live");
1116
- assert.equal(profiles.isPooled("deepseek"), false, "single key ⇒ unchanged single-key behaviour");
1117
- });
1118
-
1119
- test("a session on Grok is still barred by the grade gate, exactly like DeepSeek/Kimi", () => {
1120
- // Not a bug to fix: §6.6 says a tool-using session needs grade A, and no
1121
- // third-party row is graded A until the probe ledger earns it. Cheap-tier
1122
- // reachability must NOT quietly become cheap-tier session hosting.
1123
- const sessionCfg = {
1124
- ...XAI_CONFIG,
1125
- routing_policy: [{ default: true, chain: ["cheap", "default"] }],
1126
- };
1127
- const d = resolveChain(
1128
- { task_class: "session.responder", data_class: "public", token_estimate: 2000 },
1129
- { catalog: CATALOG_WITH_XAI, config: sessionCfg, env: { XAI_API_KEY: "xai-k" }, now: CLOCK }
1130
- );
1131
- const skipped = d.tried.find((t) => t.ref === "xai/grok-4-fast");
1132
- assert.equal(skipped?.reason, "grade_too_low");
1133
- assert.equal(d.chosen.provider, "anthropic");
1134
- });
1135
-
1136
- // ── estCostUSD + ledger join fields ──────────────────────────────────────────
1137
-
1138
- test("resolveChain: stamps a finite estCostUSD and a decision_id for the ledger join", () => {
1139
- const d = rc({ task_class: "session.responder", data_class: "sensitive", token_estimate: 100000, token_estimate_out: 20000 });
1140
- assert.ok(typeof d.estCostUSD === "number" && d.estCostUSD > 0, "priced from the catalog row");
1141
- assert.ok(/^[0-9a-z]{18}$/.test(d.decision_id), "ulid-ish decision_id");
1142
- assert.equal(d.spawnArgs.bare, true);
1143
- assert.ok(d.spawnArgs.maxTurns > 0);
1144
- });
1145
-
1146
- // ── deterministic "rules" fallback ───────────────────────────────────────────
1147
-
1148
- test("resolveChain: falls to the deterministic rules member when no model survives", () => {
1149
- // classify.inbox chain is [fast, cheap, rules]; with no keys at all, both
1150
- // model rows are unreachable and the caller's deterministic fallback wins.
1151
- const d = rc({ task_class: "classify.inbox", source: "inbox", data_class: "public", token_estimate: 1000 });
1152
- assert.equal(d.chosen.provider, "rules");
1153
- assert.equal(d.chosen.harness, "direct");
1154
- });
1155
-
1156
- // ===========================================================================
1157
- // validateRoutingConfig — strict v2 config validation
1158
- // ===========================================================================
1159
-
1160
- test("validateRoutingConfig: a clean v2 config has zero errors", () => {
1161
- const { errors } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, env: {} });
1162
- assert.equal(errors.length, 0, JSON.stringify(errors));
1163
- });
1164
-
1165
- test("validateRoutingConfig: unknown match key is an error (capability typo)", () => {
1166
- const cfg = { ...V2_CONFIG, routing_policy: [{ match: { taks_class: "x" }, chain: ["default"] }, { default: true, chain: ["default"] }] };
1167
- const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
1168
- assert.ok(errors.some((e) => /unknown match key "taks_class"/.test(e.error)));
1169
- });
1170
-
1171
- test("validateRoutingConfig: a chain ref not in the catalog is an error", () => {
1172
- const cfg = { ...V2_CONFIG, aliases: { ...V2_CONFIG.aliases, ghost: "ghost/model-x" }, routing_policy: [{ default: true, chain: ["ghost"] }] };
1173
- const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
1174
- assert.ok(errors.some((e) => /not in catalog/.test(e.error)));
1175
- });
1176
-
1177
- test("validateRoutingConfig: a chain member that is neither alias nor ref is an error", () => {
1178
- const cfg = { ...V2_CONFIG, routing_policy: [{ default: true, chain: ["notanalias"] }] };
1179
- const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
1180
- assert.ok(errors.some((e) => /neither a known alias nor/.test(e.error)));
1181
- });
1182
-
1183
- test("validateRoutingConfig: an unknown harness is an error", () => {
1184
- const cfg = { ...V2_CONFIG, routing_policy: [{ match: { task_class: "x" }, harness: "telepathy", chain: ["default"] }, { default: true, chain: ["default"] }] };
1185
- const { errors } = validateRoutingConfig(cfg, { catalog: CATALOG });
1186
- assert.ok(errors.some((e) => /unknown harness "telepathy"/.test(e.error)));
1187
- });
1188
-
1189
- test("validateRoutingConfig: an overlay-excluded backend in config is an error (tighten-only)", () => {
1190
- const overlay = { version: "t1", constraints: { backends_allowed: ["anthropic", "deepseek"] } };
1191
- // config declares a moonshot backend the overlay does not permit.
1192
- const { errors } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, overlay });
1193
- assert.ok(errors.some((e) => /not in the org overlay's backends_allowed/.test(e.error)));
1194
- });
1195
-
1196
- test("validateRoutingConfig: a missing third-party key is a WARNING, not an error", () => {
1197
- const { errors, warnings } = validateRoutingConfig(V2_CONFIG, { catalog: CATALOG, env: {} });
1198
- assert.equal(errors.length, 0);
1199
- assert.ok(warnings.some((w) => /DEEPSEEK_API_KEY/.test(w.warning)));
1200
- // Anthropic's key is never warned (session rides keychain OAuth).
1201
- assert.ok(!warnings.some((w) => /ANTHROPIC_API_KEY/.test(w.warning)));
1202
- });
1203
-
1204
- test("validateRoutingConfig: a v1 config is not strict-validated (returns clean)", () => {
1205
- const { errors } = validateRoutingConfig({ backends: {}, routing_policy: [] }, { catalog: CATALOG });
1206
- assert.equal(errors.length, 0);
1207
- });