@vellumai/assistant 0.11.5 → 0.11.6-staging.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (233) hide show
  1. package/AGENTS.md +5 -1
  2. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
  3. package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
  4. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
  5. package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
  6. package/node_modules/@vellumai/gateway-client/src/gateway-ipc-contracts.ts +70 -0
  7. package/node_modules/@vellumai/gateway-client/src/inbound-contract.ts +16 -1
  8. package/node_modules/@vellumai/gateway-client/src/index.ts +6 -2
  9. package/node_modules/@vellumai/gateway-client/src/outbound-contract.ts +121 -61
  10. package/node_modules/@vellumai/service-contracts/src/__tests__/ingress.test.ts +118 -0
  11. package/node_modules/@vellumai/service-contracts/src/ingress.ts +103 -0
  12. package/openapi.yaml +421 -15
  13. package/package.json +1 -1
  14. package/scripts/sync-web-search-catalog.ts +6 -0
  15. package/src/__tests__/app-pin-store.test.ts +149 -0
  16. package/src/__tests__/channel-availability-routes.test.ts +23 -1
  17. package/src/__tests__/channel-readiness-discord.test.ts +231 -0
  18. package/src/__tests__/channel-readiness-service.test.ts +126 -0
  19. package/src/__tests__/channel-readiness-slack-remote.test.ts +141 -0
  20. package/src/__tests__/channel-reply-delivery.test.ts +4 -4
  21. package/src/__tests__/client-os-metadata-persistence.test.ts +23 -10
  22. package/src/__tests__/conversation-delete-watch-timeline.test.ts +231 -0
  23. package/src/__tests__/conversation-error.test.ts +17 -0
  24. package/src/__tests__/conversation-seed-composer.test.ts +8 -0
  25. package/src/__tests__/conversation-slash-commands.test.ts +8 -0
  26. package/src/__tests__/disk-pressure-policy.test.ts +6 -0
  27. package/src/__tests__/gemini-provider.test.ts +138 -0
  28. package/src/__tests__/history-repair.test.ts +105 -3
  29. package/src/__tests__/identity-routes.test.ts +1 -0
  30. package/src/__tests__/llm-catalog-parity.test.ts +45 -0
  31. package/src/__tests__/migration-import-from-path.test.ts +349 -0
  32. package/src/__tests__/notification-telegram-adapter.test.ts +102 -0
  33. package/src/__tests__/oauth-commands-routes.test.ts +89 -0
  34. package/src/__tests__/oauth-provider-profiles.test.ts +7 -6
  35. package/src/__tests__/openai-provider.test.ts +18 -0
  36. package/src/__tests__/openai-responses-provider.test.ts +18 -0
  37. package/src/__tests__/platform-callback-registration.test.ts +184 -0
  38. package/src/__tests__/plugin-api-store-credential.test.ts +71 -3
  39. package/src/__tests__/pricing.test.ts +2 -2
  40. package/src/__tests__/public-ingress-urls.test.ts +36 -0
  41. package/src/__tests__/resolve-trust-class.test.ts +0 -48
  42. package/src/__tests__/sanitize-config-for-transfer.test.ts +28 -0
  43. package/src/__tests__/secret-routes-platform-proxy.test.ts +49 -0
  44. package/src/__tests__/settings-routes.test.ts +85 -3
  45. package/src/__tests__/web-search-catalog-parity.test.ts +8 -0
  46. package/src/agent/history-repair/history-repair.ts +45 -14
  47. package/src/agent/loop.ts +4 -1
  48. package/src/api/constants/profile-config-validation.ts +60 -0
  49. package/src/api/events/tool-result.ts +6 -1
  50. package/src/api/events/watch-retro-completed.ts +52 -0
  51. package/src/api/index.ts +11 -0
  52. package/src/apps/app-pin-reconciler.ts +92 -0
  53. package/src/apps/app-pin-store.ts +125 -0
  54. package/src/channels/gateway-channel-socket-health.ts +32 -0
  55. package/src/channels/gateway-discord-admission.ts +32 -0
  56. package/src/channels/types.ts +20 -0
  57. package/src/cli/commands/__tests__/conversations-slack.test.ts +1 -1
  58. package/src/cli/commands/__tests__/inference-profiles.test.ts +16 -4
  59. package/src/cli/commands/__tests__/inference-providers.test.ts +67 -2
  60. package/src/cli/commands/channels/__tests__/channels.test.ts +85 -0
  61. package/src/cli/commands/channels/index.ts +45 -31
  62. package/src/cli/commands/inference-profiles.ts +56 -3
  63. package/src/cli/commands/inference-providers.ts +28 -2
  64. package/src/cli/commands/oauth/index.help.ts +7 -1
  65. package/src/cli/commands/oauth/request.test.ts +290 -0
  66. package/src/cli/commands/oauth/request.ts +57 -41
  67. package/src/cli/lib/bundled-marketplace.json +14 -1
  68. package/src/cli/lib/open-browser.test.ts +67 -0
  69. package/src/cli/lib/open-browser.ts +24 -5
  70. package/src/config/__tests__/profile-materialization.test.ts +26 -0
  71. package/src/config/bundled-skills/phone-calls/references/TROUBLESHOOTING.md +6 -0
  72. package/src/config/bundled-skills/schedule/SKILL.md +1 -1
  73. package/src/config/feature-flag-registry.json +17 -1
  74. package/src/config/profile-materialization.ts +29 -0
  75. package/src/config/sanitize-for-transfer.ts +16 -0
  76. package/src/config/schemas/llm.ts +7 -0
  77. package/src/config/schemas/services.ts +6 -0
  78. package/src/context/outbound-sanitize.ts +6 -0
  79. package/src/daemon/__tests__/lifecycle-watch-timeline-sweep.test.ts +98 -0
  80. package/src/daemon/conversation-error.ts +24 -2
  81. package/src/daemon/conversation-slash.ts +6 -15
  82. package/src/daemon/daemon-control.ts +1 -0
  83. package/src/daemon/disk-pressure-policy.ts +7 -1
  84. package/src/daemon/handlers/__tests__/config-ingress-tunnel-records.test.ts +208 -0
  85. package/src/daemon/handlers/config-ingress.ts +115 -5
  86. package/src/daemon/lifecycle.ts +26 -0
  87. package/src/daemon/message-types/web-activity.ts +3 -2
  88. package/src/daemon/trust-context.ts +0 -37
  89. package/src/inbound/__tests__/tunnel-probe.test.ts +448 -0
  90. package/src/inbound/platform-callback-registration.ts +28 -2
  91. package/src/inbound/public-ingress-urls.ts +12 -0
  92. package/src/inbound/tunnel-probe.ts +261 -0
  93. package/src/live-voice/__tests__/live-voice-connection.test.ts +25 -0
  94. package/src/live-voice/__tests__/live-voice-flux-turn-end.test.ts +8 -2
  95. package/src/live-voice/__tests__/live-voice-session-manager.test.ts +212 -10
  96. package/src/live-voice/__tests__/live-voice-session-telemetry.test.ts +5 -2
  97. package/src/live-voice/live-voice-connection.ts +46 -8
  98. package/src/live-voice/live-voice-manager.ts +25 -0
  99. package/src/live-voice/live-voice-session-manager.ts +318 -2
  100. package/src/live-voice/live-voice-session.ts +52 -2
  101. package/src/messaging/providers/__tests__/transport-dispatch.test.ts +126 -68
  102. package/src/messaging/providers/channel-transport.ts +64 -47
  103. package/src/messaging/providers/discord/send.test.ts +46 -1
  104. package/src/messaging/providers/discord/send.ts +51 -0
  105. package/src/messaging/providers/discord/transport.ts +26 -3
  106. package/src/messaging/providers/index.ts +22 -47
  107. package/src/messaging/providers/slack/send.test.ts +83 -26
  108. package/src/messaging/providers/slack/send.ts +120 -51
  109. package/src/messaging/providers/slack/stream-tasks.test.ts +26 -0
  110. package/src/messaging/providers/slack/stream-tasks.ts +39 -0
  111. package/src/messaging/providers/slack/transport.ts +24 -22
  112. package/src/messaging/providers/telegram-bot/send.test.ts +109 -12
  113. package/src/messaging/providers/telegram-bot/send.ts +43 -0
  114. package/src/messaging/providers/telegram-bot/transport.ts +25 -8
  115. package/src/notifications/__tests__/assistant-reply-producer.test.ts +30 -7
  116. package/src/notifications/adapters/telegram.ts +48 -1
  117. package/src/notifications/assistant-reply-producer.ts +7 -7
  118. package/src/notifications/conversation-seed-composer.ts +7 -2
  119. package/src/oauth/byo-connection.test.ts +63 -0
  120. package/src/oauth/byo-connection.ts +16 -15
  121. package/src/oauth/connection.test.ts +111 -0
  122. package/src/oauth/connection.ts +142 -1
  123. package/src/oauth/platform-connection.test.ts +34 -0
  124. package/src/oauth/platform-connection.ts +28 -5
  125. package/src/oauth/seed-providers.ts +15 -1
  126. package/src/permissions/types.ts +3 -1
  127. package/src/persistence/conversation-crud.ts +46 -0
  128. package/src/persistence/conversation-types.ts +11 -9
  129. package/src/persistence/db-async-query.ts +2 -1
  130. package/src/persistence/db-maintenance.ts +15 -0
  131. package/src/persistence/embeddings/qdrant-manager.ts +1 -0
  132. package/src/persistence/migrations/367-create-watch-timeline-entries.ts +46 -0
  133. package/src/persistence/migrations/368-watch-timeline-screenshot-blob.ts +33 -0
  134. package/src/persistence/migrations/369-create-app-pins.ts +37 -0
  135. package/src/persistence/migrations/__tests__/367-create-watch-timeline-entries.test.ts +98 -0
  136. package/src/persistence/migrations/__tests__/368-watch-timeline-screenshot-blob.test.ts +98 -0
  137. package/src/persistence/schema/index.ts +1 -0
  138. package/src/persistence/schema/infrastructure.ts +17 -0
  139. package/src/persistence/schema/watch.ts +29 -0
  140. package/src/persistence/steps.ts +6 -0
  141. package/src/plugins/mtime-cache.ts +11 -0
  142. package/src/providers/__tests__/retry-network-error.test.ts +84 -0
  143. package/src/providers/connection-resolution.ts +23 -1
  144. package/src/providers/content-blocks.ts +9 -0
  145. package/src/providers/fetch-provider-catalog.ts +19 -0
  146. package/src/providers/gemini/client.ts +13 -5
  147. package/src/providers/inference/__tests__/endpoint-probe.test.ts +92 -0
  148. package/src/providers/inference/__tests__/profile-config-validation.test.ts +39 -0
  149. package/src/providers/inference/__tests__/profile-probe-classify.test.ts +66 -0
  150. package/src/providers/inference/adapter-factory.ts +0 -9
  151. package/src/providers/inference/credential-rotation.ts +61 -0
  152. package/src/providers/inference/endpoint-probe.ts +115 -0
  153. package/src/providers/inference/profile-probe.ts +256 -0
  154. package/src/providers/model-catalog.ts +170 -125
  155. package/src/providers/openai/__tests__/api-error-normalization.test.ts +17 -1
  156. package/src/providers/openai/__tests__/chat-completions-provider-reasoning.test.ts +42 -60
  157. package/src/providers/openai/__tests__/connection-error-wrap.test.ts +44 -0
  158. package/src/providers/openai/__tests__/orphan-tool-result-guard.test.ts +34 -2
  159. package/src/providers/openai/api-error-normalization.ts +16 -2
  160. package/src/providers/openai/chat-completions-provider.ts +75 -29
  161. package/src/providers/openai/responses-provider.ts +5 -2
  162. package/src/providers/openrouter/client.ts +0 -1
  163. package/src/providers/provider-send-message.ts +11 -0
  164. package/src/providers/retry.ts +6 -0
  165. package/src/providers/search-provider-catalog.ts +20 -0
  166. package/src/providers/vercel-ai-gateway/client.ts +0 -1
  167. package/src/runtime/AGENTS.md +1 -0
  168. package/src/runtime/__tests__/desktop-presence.test.ts +27 -4
  169. package/src/runtime/__tests__/host-observe.test.ts +302 -0
  170. package/src/runtime/channel-readiness-service.ts +214 -14
  171. package/src/runtime/channel-readiness-types.ts +49 -2
  172. package/src/runtime/channel-reply-delivery.ts +2 -2
  173. package/src/runtime/desktop-presence.ts +24 -21
  174. package/src/runtime/host-observe.ts +246 -0
  175. package/src/runtime/http-server.ts +181 -1
  176. package/src/runtime/migrations/__tests__/staged-import-path.test.ts +104 -0
  177. package/src/runtime/migrations/staged-import-path.ts +116 -0
  178. package/src/runtime/routes/__tests__/app-pin-routes.test.ts +383 -0
  179. package/src/runtime/routes/__tests__/conversation-query-routes.test.ts +80 -0
  180. package/src/runtime/routes/__tests__/inference-profiles-routes.test.ts +118 -0
  181. package/src/runtime/routes/__tests__/inference-provider-connection-routes.test.ts +20 -0
  182. package/src/runtime/routes/__tests__/ingress-status-routes.test.ts +508 -0
  183. package/src/runtime/routes/__tests__/plugins-routes.test.ts +35 -56
  184. package/src/runtime/routes/__tests__/watch-routes-guardian-cache.test.ts +139 -0
  185. package/src/runtime/routes/__tests__/watch-routes.test.ts +598 -0
  186. package/src/runtime/routes/app-management-routes.ts +140 -29
  187. package/src/runtime/routes/channel-availability-routes.ts +1 -0
  188. package/src/runtime/routes/channel-readiness-routes.ts +14 -2
  189. package/src/runtime/routes/conversation-query-routes.ts +10 -0
  190. package/src/runtime/routes/guardian-approval-interception.ts +24 -33
  191. package/src/runtime/routes/host-cu-routes.ts +18 -0
  192. package/src/runtime/routes/identity-routes.ts +2 -0
  193. package/src/runtime/routes/inbound-message-handler.ts +10 -7
  194. package/src/runtime/routes/inbound-stages/background-dispatch.test.ts +166 -308
  195. package/src/runtime/routes/inbound-stages/background-dispatch.ts +158 -335
  196. package/src/runtime/routes/index.ts +2 -0
  197. package/src/runtime/routes/inference-profiles-routes.ts +232 -31
  198. package/src/runtime/routes/inference-provider-connection-routes.ts +24 -4
  199. package/src/runtime/routes/ingress-status-routes.ts +180 -0
  200. package/src/runtime/routes/live-voice-routes.test.ts +40 -1
  201. package/src/runtime/routes/live-voice-routes.ts +34 -0
  202. package/src/runtime/routes/migration-routes.ts +218 -10
  203. package/src/runtime/routes/oauth-commands-routes.ts +23 -16
  204. package/src/runtime/routes/plugins-routes.ts +12 -28
  205. package/src/runtime/routes/question-routes.ts +6 -0
  206. package/src/runtime/routes/secret-routes.ts +7 -27
  207. package/src/runtime/routes/settings-routes.ts +9 -6
  208. package/src/runtime/routes/watch-routes.ts +807 -0
  209. package/src/runtime/slack-reply-session.test.ts +230 -121
  210. package/src/runtime/slack-reply-session.ts +113 -81
  211. package/src/runtime/{slack-task-progress.test.ts → task-progress.test.ts} +1 -28
  212. package/src/runtime/{slack-task-progress.ts → task-progress.ts} +30 -51
  213. package/src/security/__tests__/untrusted-content.test.ts +42 -0
  214. package/src/security/untrusted-content.ts +28 -9
  215. package/src/telemetry/__tests__/live-voice-funnel.test.ts +108 -0
  216. package/src/telemetry/live-voice-funnel.ts +75 -8
  217. package/src/tools/credentials/store.ts +18 -6
  218. package/src/tools/network/__tests__/firecrawl-compat.test.ts +77 -0
  219. package/src/tools/network/__tests__/web-fetch-fastcrw.test.ts +169 -0
  220. package/src/tools/network/__tests__/web-search.test.ts +97 -2
  221. package/src/tools/network/firecrawl-compat.ts +90 -0
  222. package/src/tools/network/web-fetch.ts +142 -62
  223. package/src/tools/network/web-search.ts +141 -55
  224. package/src/tools/types.ts +2 -1
  225. package/src/util/oauth-request-body.test.ts +74 -0
  226. package/src/util/oauth-request-body.ts +60 -0
  227. package/src/util/worker-process.ts +1 -0
  228. package/src/watch/__tests__/watch-retro.test.ts +665 -0
  229. package/src/watch/__tests__/watch-session-manager.test.ts +566 -0
  230. package/src/watch/__tests__/watch-timeline.test.ts +670 -0
  231. package/src/watch/watch-retro.ts +480 -0
  232. package/src/watch/watch-session-manager.ts +575 -0
  233. package/src/watch/watch-timeline.ts +848 -0
@@ -1196,9 +1196,10 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1196
1196
  // / 1.5x output / 2x cache-read+write for the whole request, per
1197
1197
  // OpenAI's model cards.
1198
1198
  //
1199
- // Rates are OpenRouter's own card (https://openrouter.ai/api/v1/models),
1200
- // which discounts Terra and Luna below OpenAI's direct list price; Sol
1201
- // matches direct pricing.
1199
+ // Rates are OpenRouter's own card (https://openrouter.ai/api/v1/models)
1200
+ // and can differ from OpenAI's direct list. cacheWrite is 1.25x input.
1201
+ // Long-context (>272K input) is 2x input / 1.5x output / 2x
1202
+ // cache-read+write for the whole request, per OpenAI's model cards.
1202
1203
  {
1203
1204
  id: "openai/gpt-5.6-sol",
1204
1205
  displayName: "GPT-5.6 Sol",
@@ -1212,17 +1213,17 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1212
1213
  supportsToolUse: true,
1213
1214
  supportsPromptCacheBreakpoints: true,
1214
1215
  pricing: {
1215
- inputPer1mTokens: 5.0,
1216
- outputPer1mTokens: 30.0,
1217
- cacheWritePer1mTokens: 6.25,
1218
- cacheReadPer1mTokens: 0.5,
1216
+ inputPer1mTokens: 2.0,
1217
+ outputPer1mTokens: 10.0,
1218
+ cacheWritePer1mTokens: 2.5,
1219
+ cacheReadPer1mTokens: 0.2,
1219
1220
  tiers: [
1220
1221
  {
1221
1222
  inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1222
- inputPer1mTokens: 10,
1223
- outputPer1mTokens: 45,
1224
- cacheWritePer1mTokens: 12.5,
1225
- cacheReadPer1mTokens: 1,
1223
+ inputPer1mTokens: 4,
1224
+ outputPer1mTokens: 15,
1225
+ cacheWritePer1mTokens: 5,
1226
+ cacheReadPer1mTokens: 0.4,
1226
1227
  },
1227
1228
  ],
1228
1229
  },
@@ -1240,17 +1241,17 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1240
1241
  supportsToolUse: true,
1241
1242
  supportsPromptCacheBreakpoints: true,
1242
1243
  pricing: {
1243
- inputPer1mTokens: 5.0,
1244
- outputPer1mTokens: 30.0,
1245
- cacheWritePer1mTokens: 6.25,
1246
- cacheReadPer1mTokens: 0.5,
1244
+ inputPer1mTokens: 2.0,
1245
+ outputPer1mTokens: 10.0,
1246
+ cacheWritePer1mTokens: 2.5,
1247
+ cacheReadPer1mTokens: 0.2,
1247
1248
  tiers: [
1248
1249
  {
1249
1250
  inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1250
- inputPer1mTokens: 10,
1251
- outputPer1mTokens: 45,
1252
- cacheWritePer1mTokens: 12.5,
1253
- cacheReadPer1mTokens: 1,
1251
+ inputPer1mTokens: 4,
1252
+ outputPer1mTokens: 15,
1253
+ cacheWritePer1mTokens: 5,
1254
+ cacheReadPer1mTokens: 0.4,
1254
1255
  },
1255
1256
  ],
1256
1257
  },
@@ -1268,17 +1269,17 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1268
1269
  supportsToolUse: true,
1269
1270
  supportsPromptCacheBreakpoints: true,
1270
1271
  pricing: {
1271
- inputPer1mTokens: 1.0,
1272
- outputPer1mTokens: 6.0,
1273
- cacheWritePer1mTokens: 1.25,
1274
- cacheReadPer1mTokens: 0.1,
1272
+ inputPer1mTokens: 2.0,
1273
+ outputPer1mTokens: 12.0,
1274
+ cacheWritePer1mTokens: 2.5,
1275
+ cacheReadPer1mTokens: 0.2,
1275
1276
  tiers: [
1276
1277
  {
1277
1278
  inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1278
- inputPer1mTokens: 2,
1279
- outputPer1mTokens: 9,
1280
- cacheWritePer1mTokens: 2.5,
1281
- cacheReadPer1mTokens: 0.2,
1279
+ inputPer1mTokens: 4,
1280
+ outputPer1mTokens: 18,
1281
+ cacheWritePer1mTokens: 5,
1282
+ cacheReadPer1mTokens: 0.4,
1282
1283
  },
1283
1284
  ],
1284
1285
  },
@@ -1296,17 +1297,17 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1296
1297
  supportsToolUse: true,
1297
1298
  supportsPromptCacheBreakpoints: true,
1298
1299
  pricing: {
1299
- inputPer1mTokens: 1.0,
1300
- outputPer1mTokens: 6.0,
1301
- cacheWritePer1mTokens: 1.25,
1302
- cacheReadPer1mTokens: 0.1,
1300
+ inputPer1mTokens: 2.0,
1301
+ outputPer1mTokens: 12.0,
1302
+ cacheWritePer1mTokens: 2.5,
1303
+ cacheReadPer1mTokens: 0.2,
1303
1304
  tiers: [
1304
1305
  {
1305
1306
  inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1306
- inputPer1mTokens: 2,
1307
- outputPer1mTokens: 9,
1308
- cacheWritePer1mTokens: 2.5,
1309
- cacheReadPer1mTokens: 0.2,
1307
+ inputPer1mTokens: 4,
1308
+ outputPer1mTokens: 18,
1309
+ cacheWritePer1mTokens: 5,
1310
+ cacheReadPer1mTokens: 0.4,
1310
1311
  },
1311
1312
  ],
1312
1313
  },
@@ -1324,17 +1325,17 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1324
1325
  supportsToolUse: true,
1325
1326
  supportsPromptCacheBreakpoints: true,
1326
1327
  pricing: {
1327
- inputPer1mTokens: 0.1,
1328
- outputPer1mTokens: 0.6,
1329
- cacheWritePer1mTokens: 0.125,
1330
- cacheReadPer1mTokens: 0.01,
1328
+ inputPer1mTokens: 0.2,
1329
+ outputPer1mTokens: 1.2,
1330
+ cacheWritePer1mTokens: 0.25,
1331
+ cacheReadPer1mTokens: 0.02,
1331
1332
  tiers: [
1332
1333
  {
1333
1334
  inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1334
- inputPer1mTokens: 0.2,
1335
- outputPer1mTokens: 0.9,
1336
- cacheWritePer1mTokens: 0.25,
1337
- cacheReadPer1mTokens: 0.02,
1335
+ inputPer1mTokens: 0.4,
1336
+ outputPer1mTokens: 1.8,
1337
+ cacheWritePer1mTokens: 0.5,
1338
+ cacheReadPer1mTokens: 0.04,
1338
1339
  },
1339
1340
  ],
1340
1341
  },
@@ -1352,17 +1353,17 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1352
1353
  supportsToolUse: true,
1353
1354
  supportsPromptCacheBreakpoints: true,
1354
1355
  pricing: {
1355
- inputPer1mTokens: 0.1,
1356
- outputPer1mTokens: 0.6,
1357
- cacheWritePer1mTokens: 0.125,
1358
- cacheReadPer1mTokens: 0.01,
1356
+ inputPer1mTokens: 0.2,
1357
+ outputPer1mTokens: 1.2,
1358
+ cacheWritePer1mTokens: 0.25,
1359
+ cacheReadPer1mTokens: 0.02,
1359
1360
  tiers: [
1360
1361
  {
1361
1362
  inputTokenThreshold: OPENAI_LONG_CONTEXT_PRICING_THRESHOLD_TOKENS,
1362
- inputPer1mTokens: 0.2,
1363
- outputPer1mTokens: 0.9,
1364
- cacheWritePer1mTokens: 0.25,
1365
- cacheReadPer1mTokens: 0.02,
1363
+ inputPer1mTokens: 0.4,
1364
+ outputPer1mTokens: 1.8,
1365
+ cacheWritePer1mTokens: 0.5,
1366
+ cacheReadPer1mTokens: 0.04,
1366
1367
  },
1367
1368
  ],
1368
1369
  },
@@ -1390,7 +1391,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1390
1391
  pricing: {
1391
1392
  inputPer1mTokens: 2,
1392
1393
  outputPer1mTokens: 6,
1393
- cacheReadPer1mTokens: 0.5,
1394
+ cacheReadPer1mTokens: 0.3,
1394
1395
  },
1395
1396
  },
1396
1397
  {
@@ -1430,10 +1431,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1430
1431
  contextWindowTokens: 163840,
1431
1432
  maxOutputTokens: 32000,
1432
1433
  supportsThinking: true,
1433
- supportsCaching: false,
1434
+ supportsCaching: true,
1434
1435
  supportsVision: false,
1435
1436
  supportsToolUse: true,
1436
- pricing: { inputPer1mTokens: 0.55, outputPer1mTokens: 2.19 },
1437
+ pricing: {
1438
+ inputPer1mTokens: 0.5,
1439
+ outputPer1mTokens: 2.15,
1440
+ cacheReadPer1mTokens: 0.35,
1441
+ },
1437
1442
  },
1438
1443
  {
1439
1444
  id: "deepseek/deepseek-chat-v3-0324",
@@ -1444,7 +1449,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1444
1449
  supportsCaching: false,
1445
1450
  supportsVision: false,
1446
1451
  supportsToolUse: true,
1447
- pricing: { inputPer1mTokens: 0.27, outputPer1mTokens: 1.1 },
1452
+ pricing: { inputPer1mTokens: 0.25, outputPer1mTokens: 1.0 },
1448
1453
  },
1449
1454
  {
1450
1455
  id: "deepseek/deepseek-v4-pro",
@@ -1452,10 +1457,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1452
1457
  contextWindowTokens: 1048576,
1453
1458
  maxOutputTokens: 384000,
1454
1459
  supportsThinking: true,
1455
- supportsCaching: false,
1460
+ supportsCaching: true,
1456
1461
  supportsVision: false,
1457
1462
  supportsToolUse: true,
1458
- pricing: { inputPer1mTokens: 0.435, outputPer1mTokens: 0.87 },
1463
+ pricing: {
1464
+ inputPer1mTokens: 0.579072,
1465
+ outputPer1mTokens: 1.158144,
1466
+ cacheReadPer1mTokens: 0.048256,
1467
+ },
1459
1468
  },
1460
1469
  {
1461
1470
  id: "deepseek/deepseek-v4-flash",
@@ -1463,21 +1472,18 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1463
1472
  contextWindowTokens: 1048576,
1464
1473
  maxOutputTokens: 384000,
1465
1474
  supportsThinking: true,
1466
- supportsCaching: false,
1475
+ supportsCaching: true,
1467
1476
  supportsVision: false,
1468
1477
  supportsToolUse: true,
1469
- pricing: { inputPer1mTokens: 0.14, outputPer1mTokens: 0.28 },
1470
- },
1471
- {
1472
- id: "deepseek/deepseek-v3.2-speciale",
1473
- displayName: "DeepSeek V3.2 Speciale",
1474
- contextWindowTokens: 163840,
1475
- maxOutputTokens: 163840,
1476
- supportsThinking: true,
1477
- supportsCaching: false,
1478
- supportsVision: false,
1479
- supportsToolUse: false,
1480
- pricing: { inputPer1mTokens: 0.287, outputPer1mTokens: 0.431 },
1478
+ // Reseller list rate, matching the `vercel-ai-gateway` entry for this
1479
+ // model. DeepSeek serves no OpenRouter endpoint of its own, so the
1480
+ // card carries whichever reseller holds the default route rather than
1481
+ // a first-party rate. An estimate, not a quote.
1482
+ pricing: {
1483
+ inputPer1mTokens: 0.14,
1484
+ outputPer1mTokens: 0.28,
1485
+ cacheReadPer1mTokens: 0.028,
1486
+ },
1481
1487
  },
1482
1488
  // Qwen
1483
1489
  {
@@ -1489,7 +1495,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1489
1495
  supportsCaching: false,
1490
1496
  supportsVision: false,
1491
1497
  supportsToolUse: true,
1492
- pricing: { inputPer1mTokens: 0.8, outputPer1mTokens: 2.4 },
1498
+ pricing: { inputPer1mTokens: 0.26, outputPer1mTokens: 1.56 },
1493
1499
  },
1494
1500
  {
1495
1501
  id: "qwen/qwen3.5-397b-a17b",
@@ -1497,10 +1503,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1497
1503
  contextWindowTokens: 131072,
1498
1504
  maxOutputTokens: 8192,
1499
1505
  supportsThinking: true,
1500
- supportsCaching: false,
1506
+ supportsCaching: true,
1501
1507
  supportsVision: false,
1502
1508
  supportsToolUse: true,
1503
- pricing: { inputPer1mTokens: 0.9, outputPer1mTokens: 2.7 },
1509
+ pricing: {
1510
+ inputPer1mTokens: 0.5,
1511
+ outputPer1mTokens: 3.6,
1512
+ cacheReadPer1mTokens: 0.3,
1513
+ },
1504
1514
  },
1505
1515
  {
1506
1516
  id: "qwen/qwen3.5-flash-02-23",
@@ -1511,7 +1521,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1511
1521
  supportsCaching: false,
1512
1522
  supportsVision: false,
1513
1523
  supportsToolUse: true,
1514
- pricing: { inputPer1mTokens: 0.2, outputPer1mTokens: 0.6 },
1524
+ pricing: { inputPer1mTokens: 0.065, outputPer1mTokens: 0.26 },
1515
1525
  },
1516
1526
  {
1517
1527
  id: "qwen/qwen3-coder-next",
@@ -1519,10 +1529,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1519
1529
  contextWindowTokens: 131072,
1520
1530
  maxOutputTokens: 8192,
1521
1531
  supportsThinking: false,
1522
- supportsCaching: false,
1532
+ supportsCaching: true,
1523
1533
  supportsVision: false,
1524
1534
  supportsToolUse: true,
1525
- pricing: { inputPer1mTokens: 0.5, outputPer1mTokens: 1.5 },
1535
+ pricing: {
1536
+ inputPer1mTokens: 0.12,
1537
+ outputPer1mTokens: 0.8,
1538
+ cacheReadPer1mTokens: 0.07,
1539
+ },
1526
1540
  },
1527
1541
  // Moonshot
1528
1542
  {
@@ -1532,10 +1546,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1532
1546
  maxOutputTokens: 131072,
1533
1547
  supportsThinking: true,
1534
1548
  adaptiveThinkingOnly: true,
1535
- supportsCaching: false,
1549
+ supportsCaching: true,
1536
1550
  supportsVision: true,
1537
1551
  supportsToolUse: true,
1538
- pricing: { inputPer1mTokens: 3, outputPer1mTokens: 15 },
1552
+ pricing: {
1553
+ inputPer1mTokens: 3,
1554
+ outputPer1mTokens: 15,
1555
+ cacheReadPer1mTokens: 0.3,
1556
+ },
1539
1557
  },
1540
1558
  {
1541
1559
  id: "moonshotai/kimi-k2.6",
@@ -1543,10 +1561,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1543
1561
  contextWindowTokens: 262144,
1544
1562
  maxOutputTokens: 32768,
1545
1563
  supportsThinking: true,
1546
- supportsCaching: false,
1564
+ supportsCaching: true,
1547
1565
  supportsVision: true,
1548
1566
  supportsToolUse: true,
1549
- pricing: { inputPer1mTokens: 0.6, outputPer1mTokens: 2.8 },
1567
+ pricing: {
1568
+ inputPer1mTokens: 0.95,
1569
+ outputPer1mTokens: 4.0,
1570
+ cacheReadPer1mTokens: 0.16,
1571
+ },
1550
1572
  },
1551
1573
  {
1552
1574
  id: "moonshotai/kimi-k2.5",
@@ -1554,10 +1576,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1554
1576
  contextWindowTokens: 256000,
1555
1577
  maxOutputTokens: 32768,
1556
1578
  supportsThinking: false,
1557
- supportsCaching: false,
1579
+ supportsCaching: true,
1558
1580
  supportsVision: false,
1559
1581
  supportsToolUse: true,
1560
- pricing: { inputPer1mTokens: 0.6, outputPer1mTokens: 2.5 },
1582
+ pricing: {
1583
+ inputPer1mTokens: 0.6,
1584
+ outputPer1mTokens: 3.0,
1585
+ cacheReadPer1mTokens: 0.1,
1586
+ },
1561
1587
  },
1562
1588
  // MiniMax
1563
1589
  {
@@ -1568,10 +1594,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1568
1594
  contextWindowTokens: 524288,
1569
1595
  maxOutputTokens: 512000,
1570
1596
  supportsThinking: true,
1571
- supportsCaching: false,
1597
+ supportsCaching: true,
1572
1598
  supportsVision: true,
1573
1599
  supportsToolUse: true,
1574
- pricing: { inputPer1mTokens: 0.3, outputPer1mTokens: 1.2 },
1600
+ pricing: {
1601
+ inputPer1mTokens: 0.3,
1602
+ outputPer1mTokens: 1.2,
1603
+ cacheReadPer1mTokens: 0.06,
1604
+ },
1575
1605
  },
1576
1606
  {
1577
1607
  id: "minimax/minimax-m2.7",
@@ -1579,10 +1609,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1579
1609
  contextWindowTokens: 196608,
1580
1610
  maxOutputTokens: 131072,
1581
1611
  supportsThinking: true,
1582
- supportsCaching: false,
1612
+ supportsCaching: true,
1583
1613
  supportsVision: false,
1584
1614
  supportsToolUse: true,
1585
- pricing: { inputPer1mTokens: 0.279, outputPer1mTokens: 1.2 },
1615
+ pricing: {
1616
+ inputPer1mTokens: 0.3,
1617
+ outputPer1mTokens: 1.2,
1618
+ cacheReadPer1mTokens: 0.06,
1619
+ },
1586
1620
  },
1587
1621
  {
1588
1622
  id: "minimax/minimax-m2.5",
@@ -1590,10 +1624,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1590
1624
  contextWindowTokens: 196608,
1591
1625
  maxOutputTokens: 196608,
1592
1626
  supportsThinking: true,
1593
- supportsCaching: false,
1627
+ supportsCaching: true,
1594
1628
  supportsVision: false,
1595
1629
  supportsToolUse: true,
1596
- pricing: { inputPer1mTokens: 0.15, outputPer1mTokens: 1.15 },
1630
+ pricing: {
1631
+ inputPer1mTokens: 0.27,
1632
+ outputPer1mTokens: 1.08,
1633
+ cacheReadPer1mTokens: 0.027,
1634
+ },
1597
1635
  },
1598
1636
  {
1599
1637
  id: "minimax/minimax-m2.1",
@@ -1601,10 +1639,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1601
1639
  contextWindowTokens: 196608,
1602
1640
  maxOutputTokens: 196608,
1603
1641
  supportsThinking: true,
1604
- supportsCaching: false,
1642
+ supportsCaching: true,
1605
1643
  supportsVision: false,
1606
1644
  supportsToolUse: true,
1607
- pricing: { inputPer1mTokens: 0.29, outputPer1mTokens: 0.95 },
1645
+ pricing: {
1646
+ inputPer1mTokens: 0.3,
1647
+ outputPer1mTokens: 1.2,
1648
+ cacheReadPer1mTokens: 0.03,
1649
+ },
1608
1650
  },
1609
1651
  {
1610
1652
  id: "minimax/minimax-m2",
@@ -1615,7 +1657,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1615
1657
  supportsCaching: false,
1616
1658
  supportsVision: false,
1617
1659
  supportsToolUse: true,
1618
- pricing: { inputPer1mTokens: 0.255, outputPer1mTokens: 1.0 },
1660
+ pricing: { inputPer1mTokens: 0.255, outputPer1mTokens: 1.02 },
1619
1661
  },
1620
1662
  {
1621
1663
  id: "minimax/minimax-m2-her",
@@ -1623,10 +1665,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1623
1665
  contextWindowTokens: 65536,
1624
1666
  maxOutputTokens: 2048,
1625
1667
  supportsThinking: false,
1626
- supportsCaching: false,
1668
+ supportsCaching: true,
1627
1669
  supportsVision: false,
1628
1670
  supportsToolUse: false,
1629
- pricing: { inputPer1mTokens: 0.3, outputPer1mTokens: 1.2 },
1671
+ pricing: {
1672
+ inputPer1mTokens: 0.3,
1673
+ outputPer1mTokens: 1.2,
1674
+ cacheReadPer1mTokens: 0.03,
1675
+ },
1630
1676
  },
1631
1677
  {
1632
1678
  id: "minimax/minimax-m1",
@@ -1637,7 +1683,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1637
1683
  supportsCaching: false,
1638
1684
  supportsVision: false,
1639
1685
  supportsToolUse: true,
1640
- pricing: { inputPer1mTokens: 0.4, outputPer1mTokens: 2.2 },
1686
+ pricing: { inputPer1mTokens: 0.55, outputPer1mTokens: 2.2 },
1641
1687
  },
1642
1688
  {
1643
1689
  id: "minimax/minimax-01",
@@ -1657,10 +1703,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1657
1703
  contextWindowTokens: 1048576,
1658
1704
  maxOutputTokens: 131072,
1659
1705
  supportsThinking: true,
1660
- supportsCaching: false,
1706
+ supportsCaching: true,
1661
1707
  supportsVision: false,
1662
1708
  supportsToolUse: true,
1663
- pricing: { inputPer1mTokens: 1.4, outputPer1mTokens: 4.4 },
1709
+ pricing: {
1710
+ inputPer1mTokens: 1.19,
1711
+ outputPer1mTokens: 3.74,
1712
+ cacheReadPer1mTokens: 0.221,
1713
+ },
1664
1714
  },
1665
1715
  // Mistral
1666
1716
  {
@@ -1669,10 +1719,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1669
1719
  contextWindowTokens: 131072,
1670
1720
  maxOutputTokens: 16000,
1671
1721
  supportsThinking: false,
1672
- supportsCaching: false,
1722
+ supportsCaching: true,
1673
1723
  supportsVision: false,
1674
1724
  supportsToolUse: true,
1675
- pricing: { inputPer1mTokens: 0.4, outputPer1mTokens: 2.0 },
1725
+ pricing: {
1726
+ inputPer1mTokens: 0.4,
1727
+ outputPer1mTokens: 2.0,
1728
+ cacheReadPer1mTokens: 0.04,
1729
+ },
1676
1730
  },
1677
1731
  {
1678
1732
  id: "mistralai/mistral-small-2603",
@@ -1680,21 +1734,14 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1680
1734
  contextWindowTokens: 131072,
1681
1735
  maxOutputTokens: 16000,
1682
1736
  supportsThinking: false,
1683
- supportsCaching: false,
1684
- supportsVision: false,
1685
- supportsToolUse: true,
1686
- pricing: { inputPer1mTokens: 0.2, outputPer1mTokens: 0.6 },
1687
- },
1688
- {
1689
- id: "mistralai/devstral-2512",
1690
- displayName: "Devstral 2",
1691
- contextWindowTokens: 131072,
1692
- maxOutputTokens: 16000,
1693
- supportsThinking: false,
1694
- supportsCaching: false,
1737
+ supportsCaching: true,
1695
1738
  supportsVision: false,
1696
1739
  supportsToolUse: true,
1697
- pricing: { inputPer1mTokens: 0.1, outputPer1mTokens: 0.3 },
1740
+ pricing: {
1741
+ inputPer1mTokens: 0.15,
1742
+ outputPer1mTokens: 0.6,
1743
+ cacheReadPer1mTokens: 0.015,
1744
+ },
1698
1745
  },
1699
1746
  // Meta
1700
1747
  {
@@ -1706,7 +1753,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1706
1753
  supportsCaching: false,
1707
1754
  supportsVision: true,
1708
1755
  supportsToolUse: true,
1709
- pricing: { inputPer1mTokens: 0.27, outputPer1mTokens: 0.85 },
1756
+ pricing: { inputPer1mTokens: 0.2, outputPer1mTokens: 0.8 },
1710
1757
  },
1711
1758
  {
1712
1759
  id: "meta-llama/llama-4-scout",
@@ -1717,7 +1764,7 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1717
1764
  supportsCaching: false,
1718
1765
  supportsVision: true,
1719
1766
  supportsToolUse: true,
1720
- pricing: { inputPer1mTokens: 0.11, outputPer1mTokens: 0.34 },
1767
+ pricing: { inputPer1mTokens: 0.1, outputPer1mTokens: 0.3 },
1721
1768
  },
1722
1769
  // Amazon
1723
1770
  {
@@ -1731,18 +1778,6 @@ const RAW_PROVIDER_CATALOG: ProviderCatalogEntry[] = [
1731
1778
  supportsToolUse: true,
1732
1779
  pricing: { inputPer1mTokens: 0.8, outputPer1mTokens: 3.2 },
1733
1780
  },
1734
- // Owl (OpenRouter first-party)
1735
- {
1736
- id: "openrouter/owl-alpha",
1737
- displayName: "Owl Alpha",
1738
- contextWindowTokens: 1048576,
1739
- maxOutputTokens: 262144,
1740
- supportsThinking: false,
1741
- supportsCaching: false,
1742
- supportsVision: false,
1743
- supportsToolUse: true,
1744
- pricing: { inputPer1mTokens: 0, outputPer1mTokens: 0 },
1745
- },
1746
1781
  ],
1747
1782
  defaultModel: "x-ai/grok-4.20",
1748
1783
  apiKeyUrl: "https://openrouter.ai/keys",
@@ -2190,6 +2225,16 @@ export function catalogMaxOutputTokens(
2190
2225
  )?.maxOutputTokens;
2191
2226
  }
2192
2227
 
2228
+ /** The `contextWindowTokens` declared for a (provider, model) catalog entry, if any. */
2229
+ export function catalogContextWindowTokens(
2230
+ provider: string,
2231
+ modelId: string,
2232
+ ): number | undefined {
2233
+ return PROVIDER_CATALOG.find((p) => p.id === provider)?.models.find(
2234
+ (m) => m.id === modelId,
2235
+ )?.contextWindowTokens;
2236
+ }
2237
+
2193
2238
  /**
2194
2239
  * Model IDs (across all catalog providers) flagged
2195
2240
  * `supportsPromptCacheBreakpoints`. Consumed by the OpenAI Responses
@@ -1,6 +1,6 @@
1
1
  import { describe, expect, test } from "bun:test";
2
2
 
3
- import type OpenAI from "openai";
3
+ import OpenAI from "openai";
4
4
 
5
5
  import type { NormalizedOpenAIAPIError } from "../api-error-normalization.js";
6
6
  import {
@@ -44,6 +44,22 @@ describe("normalizeOpenAIAPIError", () => {
44
44
  );
45
45
  });
46
46
 
47
+ test("stamps network_error for SDK connection failures", () => {
48
+ const cause = Object.assign(new Error("connect ECONNREFUSED 127.0.0.1:9"), {
49
+ code: "ECONNREFUSED",
50
+ });
51
+ const n = normalizeOpenAIAPIError(new OpenAI.APIConnectionError({ cause }));
52
+ expect(n.reason).toBe("network_error");
53
+ });
54
+
55
+ test("does not stamp network_error for user aborts", () => {
56
+ // APIUserAbortError covers caller cancellation and inner stream
57
+ // deadlines; classifying it as transient would make retry re-run
58
+ // 30-minute deadline failures.
59
+ const n = normalizeOpenAIAPIError(new OpenAI.APIUserAbortError());
60
+ expect(n.reason).not.toBe("network_error");
61
+ });
62
+
47
63
  test("extracts OpenAI-shaped error metadata", () => {
48
64
  const n = normalizeOpenAIAPIError(
49
65
  apiError(401),