jeopi-ai 16.2.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (598) hide show
  1. package/CHANGELOG.md +4347 -0
  2. package/README.md +1193 -0
  3. package/dist/types/api-registry.d.ts +30 -0
  4. package/dist/types/auth-broker/client.d.ts +73 -0
  5. package/dist/types/auth-broker/discover.d.ts +35 -0
  6. package/dist/types/auth-broker/index.d.ts +7 -0
  7. package/dist/types/auth-broker/refresher.d.ts +25 -0
  8. package/dist/types/auth-broker/remote-store.d.ts +102 -0
  9. package/dist/types/auth-broker/server.d.ts +43 -0
  10. package/dist/types/auth-broker/snapshot-cache.d.ts +17 -0
  11. package/dist/types/auth-broker/types.d.ts +107 -0
  12. package/dist/types/auth-broker/wire-schemas.d.ts +411 -0
  13. package/dist/types/auth-gateway/http.d.ts +39 -0
  14. package/dist/types/auth-gateway/index.d.ts +3 -0
  15. package/dist/types/auth-gateway/server.d.ts +36 -0
  16. package/dist/types/auth-gateway/types.d.ts +123 -0
  17. package/dist/types/auth-retry.d.ts +124 -0
  18. package/dist/types/auth-storage.d.ts +1026 -0
  19. package/dist/types/dialect/anthropic.d.ts +15 -0
  20. package/dist/types/dialect/catalog.d.ts +3 -0
  21. package/dist/types/dialect/coercion.d.ts +23 -0
  22. package/dist/types/dialect/deepseek.d.ts +14 -0
  23. package/dist/types/dialect/demotion.d.ts +23 -0
  24. package/dist/types/dialect/examples.d.ts +2 -0
  25. package/dist/types/dialect/factory.d.ts +3 -0
  26. package/dist/types/dialect/fenced-thinking.d.ts +53 -0
  27. package/dist/types/dialect/gemini.d.ts +17 -0
  28. package/dist/types/dialect/gemma.d.ts +15 -0
  29. package/dist/types/dialect/glm.d.ts +9 -0
  30. package/dist/types/dialect/harmony.d.ts +8 -0
  31. package/dist/types/dialect/hermes.d.ts +9 -0
  32. package/dist/types/dialect/history.d.ts +3 -0
  33. package/dist/types/dialect/index.d.ts +11 -0
  34. package/dist/types/dialect/inventory.d.ts +12 -0
  35. package/dist/types/dialect/kimi.d.ts +14 -0
  36. package/dist/types/dialect/minimax.d.ts +3 -0
  37. package/dist/types/dialect/owned-stream.d.ts +4 -0
  38. package/dist/types/dialect/qwen3.d.ts +9 -0
  39. package/dist/types/dialect/rendering.d.ts +45 -0
  40. package/dist/types/dialect/thinking.d.ts +6 -0
  41. package/dist/types/dialect/types.d.ts +69 -0
  42. package/dist/types/dialect/xml.d.ts +9 -0
  43. package/dist/types/error/abort.d.ts +14 -0
  44. package/dist/types/error/auth-classify.d.ts +16 -0
  45. package/dist/types/error/auth.d.ts +27 -0
  46. package/dist/types/error/aws.d.ts +23 -0
  47. package/dist/types/error/classes.d.ts +102 -0
  48. package/dist/types/error/finalize.d.ts +39 -0
  49. package/dist/types/error/flags.d.ts +79 -0
  50. package/dist/types/error/format.d.ts +20 -0
  51. package/dist/types/error/gateway.d.ts +20 -0
  52. package/dist/types/error/index.d.ts +13 -0
  53. package/dist/types/error/oauth.d.ts +43 -0
  54. package/dist/types/error/provider.d.ts +42 -0
  55. package/dist/types/error/rate-limit.d.ts +59 -0
  56. package/dist/types/error/retryable.d.ts +27 -0
  57. package/dist/types/error/validation.d.ts +32 -0
  58. package/dist/types/index.d.ts +49 -0
  59. package/dist/types/provider-details.d.ts +24 -0
  60. package/dist/types/providers/__tests__/google-auth.test.d.ts +1 -0
  61. package/dist/types/providers/__tests__/kimi-code-thinking.test.d.ts +1 -0
  62. package/dist/types/providers/__tests__/openai-codex-error.test.d.ts +1 -0
  63. package/dist/types/providers/amazon-bedrock.d.ts +39 -0
  64. package/dist/types/providers/anthropic-client.d.ts +94 -0
  65. package/dist/types/providers/anthropic-messages-server-schema.d.ts +497 -0
  66. package/dist/types/providers/anthropic-messages-server.d.ts +17 -0
  67. package/dist/types/providers/anthropic-wire.d.ts +318 -0
  68. package/dist/types/providers/anthropic.d.ts +248 -0
  69. package/dist/types/providers/aws-credentials.d.ts +53 -0
  70. package/dist/types/providers/aws-eventstream.d.ts +39 -0
  71. package/dist/types/providers/aws-sigv4.d.ts +55 -0
  72. package/dist/types/providers/azure-openai-responses.d.ts +16 -0
  73. package/dist/types/providers/cursor.d.ts +91 -0
  74. package/dist/types/providers/devin.d.ts +12 -0
  75. package/dist/types/providers/error-message.d.ts +27 -0
  76. package/dist/types/providers/github-copilot-headers.d.ts +40 -0
  77. package/dist/types/providers/gitlab-duo-workflow.d.ts +254 -0
  78. package/dist/types/providers/gitlab-duo.d.ts +27 -0
  79. package/dist/types/providers/google-auth.d.ts +32 -0
  80. package/dist/types/providers/google-gemini-cli.d.ts +118 -0
  81. package/dist/types/providers/google-interactions.d.ts +65 -0
  82. package/dist/types/providers/google-shared.d.ts +203 -0
  83. package/dist/types/providers/google-types.d.ts +155 -0
  84. package/dist/types/providers/google-vertex.d.ts +7 -0
  85. package/dist/types/providers/google.d.ts +4 -0
  86. package/dist/types/providers/grammar.d.ts +1 -0
  87. package/dist/types/providers/kimi.d.ts +27 -0
  88. package/dist/types/providers/mock.d.ts +178 -0
  89. package/dist/types/providers/ollama.d.ts +7 -0
  90. package/dist/types/providers/openai-anthropic-shim.d.ts +31 -0
  91. package/dist/types/providers/openai-chat-server-schema.d.ts +695 -0
  92. package/dist/types/providers/openai-chat-server.d.ts +16 -0
  93. package/dist/types/providers/openai-chat-wire.d.ts +644 -0
  94. package/dist/types/providers/openai-codex/request-transformer.d.ts +54 -0
  95. package/dist/types/providers/openai-codex/response-handler.d.ts +26 -0
  96. package/dist/types/providers/openai-codex-responses.d.ts +108 -0
  97. package/dist/types/providers/openai-completions.d.ts +45 -0
  98. package/dist/types/providers/openai-reasoning-fallback.d.ts +25 -0
  99. package/dist/types/providers/openai-responses-server-schema.d.ts +349 -0
  100. package/dist/types/providers/openai-responses-server.d.ts +17 -0
  101. package/dist/types/providers/openai-responses-wire.d.ts +6065 -0
  102. package/dist/types/providers/openai-responses.d.ts +126 -0
  103. package/dist/types/providers/openai-shared.d.ts +506 -0
  104. package/dist/types/providers/pi-native-client.d.ts +13 -0
  105. package/dist/types/providers/pi-native-server.d.ts +69 -0
  106. package/dist/types/providers/register-builtins.d.ts +32 -0
  107. package/dist/types/providers/synthetic.d.ts +26 -0
  108. package/dist/types/providers/transform-messages.d.ts +11 -0
  109. package/dist/types/providers/vision-guard.d.ts +20 -0
  110. package/dist/types/registry/aimlapi.d.ts +4 -0
  111. package/dist/types/registry/alibaba-coding-plan.d.ts +8 -0
  112. package/dist/types/registry/amazon-bedrock.d.ts +5 -0
  113. package/dist/types/registry/anthropic.d.ts +10 -0
  114. package/dist/types/registry/api-key-login.d.ts +42 -0
  115. package/dist/types/registry/api-key-validation.d.ts +43 -0
  116. package/dist/types/registry/azure.d.ts +4 -0
  117. package/dist/types/registry/cerebras.d.ts +7 -0
  118. package/dist/types/registry/cloudflare-ai-gateway.d.ts +13 -0
  119. package/dist/types/registry/coreweave.d.ts +7 -0
  120. package/dist/types/registry/cursor.d.ts +7 -0
  121. package/dist/types/registry/deepseek.d.ts +8 -0
  122. package/dist/types/registry/derived.d.ts +5 -0
  123. package/dist/types/registry/devin.d.ts +8 -0
  124. package/dist/types/registry/firepass.d.ts +16 -0
  125. package/dist/types/registry/fireworks.d.ts +7 -0
  126. package/dist/types/registry/github-copilot.d.ts +7 -0
  127. package/dist/types/registry/gitlab-duo-workflow.d.ts +10 -0
  128. package/dist/types/registry/gitlab-duo.d.ts +9 -0
  129. package/dist/types/registry/google-antigravity.d.ts +9 -0
  130. package/dist/types/registry/google-gemini-cli.d.ts +9 -0
  131. package/dist/types/registry/google-vertex.d.ts +5 -0
  132. package/dist/types/registry/google.d.ts +4 -0
  133. package/dist/types/registry/groq.d.ts +4 -0
  134. package/dist/types/registry/huggingface.d.ts +7 -0
  135. package/dist/types/registry/index.d.ts +4 -0
  136. package/dist/types/registry/kagi.d.ts +14 -0
  137. package/dist/types/registry/kilo.d.ts +7 -0
  138. package/dist/types/registry/kimi-code.d.ts +7 -0
  139. package/dist/types/registry/litellm.d.ts +13 -0
  140. package/dist/types/registry/llama-cpp.d.ts +8 -0
  141. package/dist/types/registry/lm-studio.d.ts +8 -0
  142. package/dist/types/registry/minimax-code-cn.d.ts +6 -0
  143. package/dist/types/registry/minimax-code.d.ts +6 -0
  144. package/dist/types/registry/minimax.d.ts +4 -0
  145. package/dist/types/registry/mistral.d.ts +4 -0
  146. package/dist/types/registry/moonshot.d.ts +7 -0
  147. package/dist/types/registry/nanogpt.d.ts +7 -0
  148. package/dist/types/registry/nvidia.d.ts +7 -0
  149. package/dist/types/registry/oauth/__tests__/xai-oauth.test.d.ts +1 -0
  150. package/dist/types/registry/oauth/anthropic.d.ts +23 -0
  151. package/dist/types/registry/oauth/callback-server.d.ts +72 -0
  152. package/dist/types/registry/oauth/cursor.d.ts +15 -0
  153. package/dist/types/registry/oauth/devin.d.ts +5 -0
  154. package/dist/types/registry/oauth/github-copilot.d.ts +30 -0
  155. package/dist/types/registry/oauth/gitlab-duo-workflow.d.ts +6 -0
  156. package/dist/types/registry/oauth/gitlab-duo.d.ts +3 -0
  157. package/dist/types/registry/oauth/google-antigravity.d.ts +11 -0
  158. package/dist/types/registry/oauth/google-gemini-cli.d.ts +10 -0
  159. package/dist/types/registry/oauth/google-oauth-shared.d.ts +22 -0
  160. package/dist/types/registry/oauth/index.d.ts +64 -0
  161. package/dist/types/registry/oauth/kimi.d.ts +21 -0
  162. package/dist/types/registry/oauth/minimax-code.d.ts +27 -0
  163. package/dist/types/registry/oauth/openai-codex.d.ts +33 -0
  164. package/dist/types/registry/oauth/opencode.d.ts +18 -0
  165. package/dist/types/registry/oauth/perplexity.d.ts +9 -0
  166. package/dist/types/registry/oauth/pkce.d.ts +8 -0
  167. package/dist/types/registry/oauth/types.d.ts +56 -0
  168. package/dist/types/registry/oauth/wafer.d.ts +1 -0
  169. package/dist/types/registry/oauth/xai-oauth.d.ts +52 -0
  170. package/dist/types/registry/oauth/xiaomi.d.ts +25 -0
  171. package/dist/types/registry/ollama-cloud.d.ts +7 -0
  172. package/dist/types/registry/ollama.d.ts +12 -0
  173. package/dist/types/registry/openai-codex-device.d.ts +8 -0
  174. package/dist/types/registry/openai-codex.d.ts +9 -0
  175. package/dist/types/registry/openai.d.ts +4 -0
  176. package/dist/types/registry/opencode-go.d.ts +6 -0
  177. package/dist/types/registry/opencode-zen.d.ts +6 -0
  178. package/dist/types/registry/openrouter.d.ts +13 -0
  179. package/dist/types/registry/parallel.d.ts +14 -0
  180. package/dist/types/registry/perplexity.d.ts +7 -0
  181. package/dist/types/registry/qianfan.d.ts +7 -0
  182. package/dist/types/registry/qwen-portal.d.ts +7 -0
  183. package/dist/types/registry/registry.d.ts +303 -0
  184. package/dist/types/registry/sakana.d.ts +7 -0
  185. package/dist/types/registry/synthetic.d.ts +6 -0
  186. package/dist/types/registry/tavily.d.ts +14 -0
  187. package/dist/types/registry/together.d.ts +6 -0
  188. package/dist/types/registry/types.d.ts +51 -0
  189. package/dist/types/registry/umans.d.ts +7 -0
  190. package/dist/types/registry/venice.d.ts +13 -0
  191. package/dist/types/registry/vercel-ai-gateway.d.ts +7 -0
  192. package/dist/types/registry/vllm.d.ts +7 -0
  193. package/dist/types/registry/wafer-serverless.d.ts +6 -0
  194. package/dist/types/registry/xai-oauth.d.ts +7 -0
  195. package/dist/types/registry/xai.d.ts +4 -0
  196. package/dist/types/registry/xiaomi-token-plan-ams.d.ts +6 -0
  197. package/dist/types/registry/xiaomi-token-plan-cn.d.ts +6 -0
  198. package/dist/types/registry/xiaomi-token-plan-sgp.d.ts +6 -0
  199. package/dist/types/registry/xiaomi.d.ts +6 -0
  200. package/dist/types/registry/zai.d.ts +7 -0
  201. package/dist/types/registry/zenmux.d.ts +7 -0
  202. package/dist/types/registry/zhipu-coding-plan.d.ts +7 -0
  203. package/dist/types/stream.d.ts +44 -0
  204. package/dist/types/types.d.ts +715 -0
  205. package/dist/types/usage/claude.d.ts +4 -0
  206. package/dist/types/usage/gemini.d.ts +2 -0
  207. package/dist/types/usage/github-copilot.d.ts +7 -0
  208. package/dist/types/usage/google-antigravity.d.ts +15 -0
  209. package/dist/types/usage/kimi.d.ts +2 -0
  210. package/dist/types/usage/minimax-code.d.ts +2 -0
  211. package/dist/types/usage/ollama.d.ts +5 -0
  212. package/dist/types/usage/openai-codex-base-url.d.ts +18 -0
  213. package/dist/types/usage/openai-codex-reset.d.ts +79 -0
  214. package/dist/types/usage/openai-codex.d.ts +3 -0
  215. package/dist/types/usage/opencode-go.d.ts +2 -0
  216. package/dist/types/usage/shared.d.ts +1 -0
  217. package/dist/types/usage/zai.d.ts +2 -0
  218. package/dist/types/usage.d.ts +346 -0
  219. package/dist/types/utils/abort.d.ts +25 -0
  220. package/dist/types/utils/anthropic-auth.d.ts +35 -0
  221. package/dist/types/utils/block-symbols.d.ts +20 -0
  222. package/dist/types/utils/deterministic-id.d.ts +16 -0
  223. package/dist/types/utils/empty-completion-retry.d.ts +21 -0
  224. package/dist/types/utils/event-stream.d.ts +30 -0
  225. package/dist/types/utils/foundry.d.ts +1 -0
  226. package/dist/types/utils/google-validation.d.ts +2 -0
  227. package/dist/types/utils/harmony-leak.d.ts +118 -0
  228. package/dist/types/utils/http-inspector.d.ts +30 -0
  229. package/dist/types/utils/idle-iterator.d.ts +137 -0
  230. package/dist/types/utils/leaked-thinking-stream.d.ts +29 -0
  231. package/dist/types/utils/openai-http.d.ts +54 -0
  232. package/dist/types/utils/openrouter-headers.d.ts +1 -0
  233. package/dist/types/utils/parse-bind.d.ts +23 -0
  234. package/dist/types/utils/provider-response.d.ts +3 -0
  235. package/dist/types/utils/proxy.d.ts +29 -0
  236. package/dist/types/utils/request-debug.d.ts +29 -0
  237. package/dist/types/utils/retry-after.d.ts +4 -0
  238. package/dist/types/utils/retry.d.ts +14 -0
  239. package/dist/types/utils/schema/adapt.d.ts +24 -0
  240. package/dist/types/utils/schema/compatibility.d.ts +30 -0
  241. package/dist/types/utils/schema/dereference.d.ts +11 -0
  242. package/dist/types/utils/schema/draft.d.ts +10 -0
  243. package/dist/types/utils/schema/equality.d.ts +4 -0
  244. package/dist/types/utils/schema/fields.d.ts +54 -0
  245. package/dist/types/utils/schema/index.d.ts +15 -0
  246. package/dist/types/utils/schema/json-schema-validator.d.ts +20 -0
  247. package/dist/types/utils/schema/meta-validator.d.ts +2 -0
  248. package/dist/types/utils/schema/normalize.d.ts +124 -0
  249. package/dist/types/utils/schema/spill.d.ts +8 -0
  250. package/dist/types/utils/schema/stamps.d.ts +17 -0
  251. package/dist/types/utils/schema/strict-tool-validation.d.ts +16 -0
  252. package/dist/types/utils/schema/types.d.ts +4 -0
  253. package/dist/types/utils/schema/typescript.d.ts +18 -0
  254. package/dist/types/utils/schema/wire.d.ts +92 -0
  255. package/dist/types/utils/schema/zod-decontaminate.d.ts +31 -0
  256. package/dist/types/utils/sdk-stream-timeout.d.ts +33 -0
  257. package/dist/types/utils/sse-debug.d.ts +5 -0
  258. package/dist/types/utils/stream-markup-healing.d.ts +87 -0
  259. package/dist/types/utils/thinking-loop.d.ts +102 -0
  260. package/dist/types/utils/tool-call-loop-guard.d.ts +26 -0
  261. package/dist/types/utils/tool-choice.d.ts +50 -0
  262. package/dist/types/utils/validation.d.ts +42 -0
  263. package/dist/types/utils.d.ts +24 -0
  264. package/package.json +139 -0
  265. package/src/api-registry.ts +109 -0
  266. package/src/auth-broker/client.ts +359 -0
  267. package/src/auth-broker/discover.ts +222 -0
  268. package/src/auth-broker/index.ts +7 -0
  269. package/src/auth-broker/refresher.ts +117 -0
  270. package/src/auth-broker/remote-store.ts +657 -0
  271. package/src/auth-broker/server.ts +646 -0
  272. package/src/auth-broker/snapshot-cache.ts +191 -0
  273. package/src/auth-broker/types.ts +130 -0
  274. package/src/auth-broker/wire-schemas.ts +249 -0
  275. package/src/auth-gateway/http.ts +194 -0
  276. package/src/auth-gateway/index.ts +3 -0
  277. package/src/auth-gateway/server.ts +802 -0
  278. package/src/auth-gateway/types.ts +151 -0
  279. package/src/auth-retry.ts +250 -0
  280. package/src/auth-storage.ts +5576 -0
  281. package/src/dialect/anthropic.md +31 -0
  282. package/src/dialect/anthropic.ts +608 -0
  283. package/src/dialect/catalog.ts +29 -0
  284. package/src/dialect/coercion.ts +136 -0
  285. package/src/dialect/deepseek.md +24 -0
  286. package/src/dialect/deepseek.ts +609 -0
  287. package/src/dialect/demotion.ts +36 -0
  288. package/src/dialect/examples.ts +33 -0
  289. package/src/dialect/factory.ts +34 -0
  290. package/src/dialect/fenced-thinking.ts +184 -0
  291. package/src/dialect/gemini.md +44 -0
  292. package/src/dialect/gemini.ts +597 -0
  293. package/src/dialect/gemma.md +33 -0
  294. package/src/dialect/gemma.ts +387 -0
  295. package/src/dialect/glm.md +32 -0
  296. package/src/dialect/glm.ts +456 -0
  297. package/src/dialect/harmony.md +31 -0
  298. package/src/dialect/harmony.ts +346 -0
  299. package/src/dialect/hermes.md +25 -0
  300. package/src/dialect/hermes.ts +206 -0
  301. package/src/dialect/history.ts +81 -0
  302. package/src/dialect/index.ts +15 -0
  303. package/src/dialect/inventory.ts +73 -0
  304. package/src/dialect/kimi.md +24 -0
  305. package/src/dialect/kimi.ts +340 -0
  306. package/src/dialect/minimax.md +31 -0
  307. package/src/dialect/minimax.ts +95 -0
  308. package/src/dialect/owned-stream.ts +470 -0
  309. package/src/dialect/prompt-template.md +12 -0
  310. package/src/dialect/qwen3.md +28 -0
  311. package/src/dialect/qwen3.ts +240 -0
  312. package/src/dialect/rendering.ts +249 -0
  313. package/src/dialect/thinking.ts +122 -0
  314. package/src/dialect/types.ts +57 -0
  315. package/src/dialect/xml.md +22 -0
  316. package/src/dialect/xml.ts +90 -0
  317. package/src/error/abort.ts +18 -0
  318. package/src/error/auth-classify.ts +30 -0
  319. package/src/error/auth.ts +48 -0
  320. package/src/error/aws.ts +31 -0
  321. package/src/error/classes.ts +186 -0
  322. package/src/error/finalize.ts +69 -0
  323. package/src/error/flags.ts +506 -0
  324. package/src/error/format.ts +45 -0
  325. package/src/error/gateway.ts +96 -0
  326. package/src/error/index.ts +13 -0
  327. package/src/error/oauth.ts +58 -0
  328. package/src/error/provider.ts +62 -0
  329. package/src/error/rate-limit.ts +161 -0
  330. package/src/error/retryable.ts +70 -0
  331. package/src/error/validation.ts +44 -0
  332. package/src/index.ts +49 -0
  333. package/src/provider-details.ts +90 -0
  334. package/src/providers/__tests__/google-auth.test.ts +144 -0
  335. package/src/providers/__tests__/kimi-code-thinking.test.ts +112 -0
  336. package/src/providers/__tests__/openai-codex-error.test.ts +84 -0
  337. package/src/providers/amazon-bedrock.ts +1042 -0
  338. package/src/providers/anthropic-client.ts +295 -0
  339. package/src/providers/anthropic-messages-server-schema.ts +252 -0
  340. package/src/providers/anthropic-messages-server.ts +756 -0
  341. package/src/providers/anthropic-wire.ts +318 -0
  342. package/src/providers/anthropic.ts +4078 -0
  343. package/src/providers/aws-credentials.ts +586 -0
  344. package/src/providers/aws-eventstream.ts +181 -0
  345. package/src/providers/aws-sigv4.ts +218 -0
  346. package/src/providers/azure-openai-responses.ts +382 -0
  347. package/src/providers/cursor/proto/agent.proto +3526 -0
  348. package/src/providers/cursor/proto/buf.gen.yaml +6 -0
  349. package/src/providers/cursor/proto/buf.yaml +17 -0
  350. package/src/providers/cursor.ts +2695 -0
  351. package/src/providers/devin/proto/buf/validate/validate.proto +468 -0
  352. package/src/providers/devin/proto/buf.gen.yaml +33 -0
  353. package/src/providers/devin/proto/buf.yaml +17 -0
  354. package/src/providers/devin/proto/cel/expr/checked.proto +103 -0
  355. package/src/providers/devin/proto/cel/expr/eval.proto +38 -0
  356. package/src/providers/devin/proto/cel/expr/explain.proto +15 -0
  357. package/src/providers/devin/proto/cel/expr/syntax.proto +113 -0
  358. package/src/providers/devin/proto/cel/expr/value.proto +41 -0
  359. package/src/providers/devin/proto/connectext/grpc/status/v1/status.proto +11 -0
  360. package/src/providers/devin/proto/errorspb/errors.proto +56 -0
  361. package/src/providers/devin/proto/errorspb/hintdetail.proto +7 -0
  362. package/src/providers/devin/proto/errorspb/markers.proto +10 -0
  363. package/src/providers/devin/proto/errorspb/tags.proto +12 -0
  364. package/src/providers/devin/proto/errorspb/testing.proto +6 -0
  365. package/src/providers/devin/proto/exa/analytics_pb/analytics.proto +188 -0
  366. package/src/providers/devin/proto/exa/api_server_pb/api_server.proto +2461 -0
  367. package/src/providers/devin/proto/exa/auth_pb/auth.proto +19 -0
  368. package/src/providers/devin/proto/exa/auto_cascade_common_pb/auto_cascade_common.proto +79 -0
  369. package/src/providers/devin/proto/exa/browser_preview_pb/browser_preview.proto +32 -0
  370. package/src/providers/devin/proto/exa/bug_checker_pb/bug_checker.proto +22 -0
  371. package/src/providers/devin/proto/exa/cascade_plugins_pb/cascade_plugins.proto +262 -0
  372. package/src/providers/devin/proto/exa/chat_client_server_pb/chat_client_server.proto +57 -0
  373. package/src/providers/devin/proto/exa/chat_pb/chat.proto +449 -0
  374. package/src/providers/devin/proto/exa/code_edit/code_edit_pb/code_edit.proto +186 -0
  375. package/src/providers/devin/proto/exa/codeium_common_pb/codeium_common.proto +4157 -0
  376. package/src/providers/devin/proto/exa/context_module_pb/context_module.proto +175 -0
  377. package/src/providers/devin/proto/exa/cortex_pb/cortex.proto +3268 -0
  378. package/src/providers/devin/proto/exa/dev_pb/dev.proto +26 -0
  379. package/src/providers/devin/proto/exa/diff_action_pb/diff_action.proto +75 -0
  380. package/src/providers/devin/proto/exa/eval/pr_eval/datasets_pb/datasets.proto +103 -0
  381. package/src/providers/devin/proto/exa/eval_pb/eval.proto +1315 -0
  382. package/src/providers/devin/proto/exa/extension_server_pb/extension_server.proto +556 -0
  383. package/src/providers/devin/proto/exa/file_system_provider_pb/file_system_provider.proto +75 -0
  384. package/src/providers/devin/proto/exa/index_pb/index.proto +461 -0
  385. package/src/providers/devin/proto/exa/knowledge_base_pb/knowledge_base.proto +144 -0
  386. package/src/providers/devin/proto/exa/language_server_pb/language_server.proto +2385 -0
  387. package/src/providers/devin/proto/exa/model_management_pb/model_management.proto +186 -0
  388. package/src/providers/devin/proto/exa/opensearch_clients_pb/opensearch_clients.proto +503 -0
  389. package/src/providers/devin/proto/exa/product_analytics_pb/product_analytics.proto +37 -0
  390. package/src/providers/devin/proto/exa/prompt_pb/prompt.proto +92 -0
  391. package/src/providers/devin/proto/exa/reactive_component_pb/reactive_component.proto +96 -0
  392. package/src/providers/devin/proto/exa/seat_management_pb/seat_management.proto +2680 -0
  393. package/src/providers/devin/proto/exa/tokenizer_pb/tokenizer.proto +37 -0
  394. package/src/providers/devin/proto/exa/trainer_pb/config.proto +647 -0
  395. package/src/providers/devin/proto/exa/tree_sitter/language_data_pb/language_data.proto +14 -0
  396. package/src/providers/devin/proto/exa/trust_pb/trust.proto +157 -0
  397. package/src/providers/devin/proto/exa/user_analytics_pb/user_analytics.proto +519 -0
  398. package/src/providers/devin/proto/google.golang.org/appengine/internal/base/api_base.proto +28 -0
  399. package/src/providers/devin/proto/google.golang.org/appengine/internal/datastore/datastore_v3.proto +484 -0
  400. package/src/providers/devin/proto/google.golang.org/appengine/internal/log/log_service.proto +136 -0
  401. package/src/providers/devin/proto/google.golang.org/appengine/internal/remote_api/remote_api.proto +42 -0
  402. package/src/providers/devin/proto/google.golang.org/appengine/internal/urlfetch/urlfetch_service.proto +61 -0
  403. package/src/providers/devin/proto/grpc/binlog/v1/binarylog.proto +84 -0
  404. package/src/providers/devin/proto/io/prometheus/client/metrics.proto +98 -0
  405. package/src/providers/devin.ts +577 -0
  406. package/src/providers/error-message.ts +21 -0
  407. package/src/providers/github-copilot-headers.ts +141 -0
  408. package/src/providers/gitlab-duo-workflow-chatml-note.md +1 -0
  409. package/src/providers/gitlab-duo-workflow.ts +3058 -0
  410. package/src/providers/gitlab-duo.ts +395 -0
  411. package/src/providers/google-auth.ts +350 -0
  412. package/src/providers/google-gemini-cli.ts +1362 -0
  413. package/src/providers/google-interactions.ts +753 -0
  414. package/src/providers/google-shared.ts +1103 -0
  415. package/src/providers/google-types.ts +180 -0
  416. package/src/providers/google-vertex.ts +183 -0
  417. package/src/providers/google.ts +87 -0
  418. package/src/providers/grammar.ts +70 -0
  419. package/src/providers/kimi.ts +52 -0
  420. package/src/providers/mock.ts +507 -0
  421. package/src/providers/ollama.ts +773 -0
  422. package/src/providers/openai-anthropic-shim.ts +152 -0
  423. package/src/providers/openai-chat-server-schema.ts +242 -0
  424. package/src/providers/openai-chat-server.ts +715 -0
  425. package/src/providers/openai-chat-wire.ts +847 -0
  426. package/src/providers/openai-codex/request-transformer.ts +295 -0
  427. package/src/providers/openai-codex/response-handler.ts +102 -0
  428. package/src/providers/openai-codex-responses.ts +3468 -0
  429. package/src/providers/openai-completions.ts +2173 -0
  430. package/src/providers/openai-reasoning-fallback.ts +269 -0
  431. package/src/providers/openai-responses-reasoning-suppression.md +1 -0
  432. package/src/providers/openai-responses-server-schema.ts +282 -0
  433. package/src/providers/openai-responses-server.ts +1280 -0
  434. package/src/providers/openai-responses-wire.ts +6391 -0
  435. package/src/providers/openai-responses.ts +1022 -0
  436. package/src/providers/openai-shared.ts +2648 -0
  437. package/src/providers/pi-native-client.ts +266 -0
  438. package/src/providers/pi-native-server.ts +242 -0
  439. package/src/providers/register-builtins.ts +475 -0
  440. package/src/providers/synthetic.ts +50 -0
  441. package/src/providers/transform-messages.ts +787 -0
  442. package/src/providers/vision-guard.ts +54 -0
  443. package/src/registry/aimlapi.ts +6 -0
  444. package/src/registry/alibaba-coding-plan.ts +95 -0
  445. package/src/registry/amazon-bedrock.ts +22 -0
  446. package/src/registry/anthropic.ts +26 -0
  447. package/src/registry/api-key-login.ts +112 -0
  448. package/src/registry/api-key-validation.ts +161 -0
  449. package/src/registry/azure.ts +6 -0
  450. package/src/registry/cerebras.ts +23 -0
  451. package/src/registry/cloudflare-ai-gateway.ts +45 -0
  452. package/src/registry/coreweave.ts +40 -0
  453. package/src/registry/cursor.ts +20 -0
  454. package/src/registry/deepseek.ts +46 -0
  455. package/src/registry/derived.ts +9 -0
  456. package/src/registry/devin.ts +15 -0
  457. package/src/registry/firepass.ts +32 -0
  458. package/src/registry/fireworks.ts +28 -0
  459. package/src/registry/github-copilot.ts +22 -0
  460. package/src/registry/gitlab-duo-workflow.ts +20 -0
  461. package/src/registry/gitlab-duo.ts +19 -0
  462. package/src/registry/google-antigravity.ts +22 -0
  463. package/src/registry/google-gemini-cli.ts +22 -0
  464. package/src/registry/google-vertex.ts +38 -0
  465. package/src/registry/google.ts +6 -0
  466. package/src/registry/groq.ts +6 -0
  467. package/src/registry/huggingface.ts +29 -0
  468. package/src/registry/index.ts +4 -0
  469. package/src/registry/kagi.ts +46 -0
  470. package/src/registry/kilo.ts +114 -0
  471. package/src/registry/kimi-code.ts +17 -0
  472. package/src/registry/litellm.ts +45 -0
  473. package/src/registry/llama-cpp.ts +35 -0
  474. package/src/registry/lm-studio.ts +31 -0
  475. package/src/registry/minimax-code-cn.ts +12 -0
  476. package/src/registry/minimax-code.ts +12 -0
  477. package/src/registry/minimax.ts +6 -0
  478. package/src/registry/mistral.ts +6 -0
  479. package/src/registry/moonshot.ts +22 -0
  480. package/src/registry/nanogpt.ts +22 -0
  481. package/src/registry/nvidia.ts +61 -0
  482. package/src/registry/oauth/__tests__/xai-oauth.test.ts +104 -0
  483. package/src/registry/oauth/anthropic.ts +311 -0
  484. package/src/registry/oauth/callback-server.ts +315 -0
  485. package/src/registry/oauth/cursor.ts +171 -0
  486. package/src/registry/oauth/devin.ts +124 -0
  487. package/src/registry/oauth/github-copilot.ts +369 -0
  488. package/src/registry/oauth/gitlab-duo-workflow.ts +146 -0
  489. package/src/registry/oauth/gitlab-duo.ts +222 -0
  490. package/src/registry/oauth/google-antigravity.ts +209 -0
  491. package/src/registry/oauth/google-gemini-cli.ts +273 -0
  492. package/src/registry/oauth/google-oauth-shared.ts +125 -0
  493. package/src/registry/oauth/index.ts +269 -0
  494. package/src/registry/oauth/kimi.ts +289 -0
  495. package/src/registry/oauth/minimax-code.ts +53 -0
  496. package/src/registry/oauth/oauth.html +311 -0
  497. package/src/registry/oauth/openai-codex.ts +364 -0
  498. package/src/registry/oauth/opencode.ts +50 -0
  499. package/src/registry/oauth/perplexity.ts +228 -0
  500. package/src/registry/oauth/pkce.ts +18 -0
  501. package/src/registry/oauth/types.ts +65 -0
  502. package/src/registry/oauth/wafer.ts +24 -0
  503. package/src/registry/oauth/xai-oauth.ts +394 -0
  504. package/src/registry/oauth/xiaomi.ts +211 -0
  505. package/src/registry/ollama-cloud.ts +36 -0
  506. package/src/registry/ollama.ts +43 -0
  507. package/src/registry/openai-codex-device.ts +18 -0
  508. package/src/registry/openai-codex.ts +19 -0
  509. package/src/registry/openai.ts +6 -0
  510. package/src/registry/opencode-go.ts +12 -0
  511. package/src/registry/opencode-zen.ts +12 -0
  512. package/src/registry/openrouter.ts +28 -0
  513. package/src/registry/parallel.ts +45 -0
  514. package/src/registry/perplexity.ts +13 -0
  515. package/src/registry/qianfan.ts +27 -0
  516. package/src/registry/qwen-portal.ts +50 -0
  517. package/src/registry/registry.ts +161 -0
  518. package/src/registry/sakana.ts +22 -0
  519. package/src/registry/synthetic.ts +21 -0
  520. package/src/registry/tavily.ts +45 -0
  521. package/src/registry/together.ts +22 -0
  522. package/src/registry/types.ts +56 -0
  523. package/src/registry/umans.ts +23 -0
  524. package/src/registry/venice.ts +33 -0
  525. package/src/registry/vercel-ai-gateway.ts +38 -0
  526. package/src/registry/vllm.ts +34 -0
  527. package/src/registry/wafer-serverless.ts +12 -0
  528. package/src/registry/xai-oauth.ts +17 -0
  529. package/src/registry/xai.ts +6 -0
  530. package/src/registry/xiaomi-token-plan-ams.ts +12 -0
  531. package/src/registry/xiaomi-token-plan-cn.ts +12 -0
  532. package/src/registry/xiaomi-token-plan-sgp.ts +12 -0
  533. package/src/registry/xiaomi.ts +12 -0
  534. package/src/registry/zai.ts +27 -0
  535. package/src/registry/zenmux.ts +22 -0
  536. package/src/registry/zhipu-coding-plan.ts +27 -0
  537. package/src/stream.ts +1778 -0
  538. package/src/types.ts +856 -0
  539. package/src/usage/claude.ts +485 -0
  540. package/src/usage/gemini.ts +258 -0
  541. package/src/usage/github-copilot.ts +424 -0
  542. package/src/usage/google-antigravity.ts +497 -0
  543. package/src/usage/kimi.ts +271 -0
  544. package/src/usage/minimax-code.ts +30 -0
  545. package/src/usage/ollama.ts +41 -0
  546. package/src/usage/openai-codex-base-url.ts +35 -0
  547. package/src/usage/openai-codex-reset.ts +174 -0
  548. package/src/usage/openai-codex.ts +535 -0
  549. package/src/usage/opencode-go.ts +89 -0
  550. package/src/usage/shared.ts +10 -0
  551. package/src/usage/zai.ts +321 -0
  552. package/src/usage.ts +333 -0
  553. package/src/utils/abort.ts +67 -0
  554. package/src/utils/anthropic-auth.ts +93 -0
  555. package/src/utils/block-symbols.ts +32 -0
  556. package/src/utils/deterministic-id.ts +20 -0
  557. package/src/utils/empty-completion-retry.ts +159 -0
  558. package/src/utils/event-stream.ts +171 -0
  559. package/src/utils/foundry.ts +8 -0
  560. package/src/utils/google-validation.ts +25 -0
  561. package/src/utils/harmony-leak.ts +456 -0
  562. package/src/utils/http-inspector.ts +168 -0
  563. package/src/utils/idle-iterator.ts +473 -0
  564. package/src/utils/leaked-thinking-stream.ts +294 -0
  565. package/src/utils/openai-http.ts +122 -0
  566. package/src/utils/openrouter-headers.ts +12 -0
  567. package/src/utils/parse-bind.ts +56 -0
  568. package/src/utils/provider-response.ts +30 -0
  569. package/src/utils/proxy.ts +240 -0
  570. package/src/utils/request-debug.ts +351 -0
  571. package/src/utils/retry-after.ts +110 -0
  572. package/src/utils/retry.ts +59 -0
  573. package/src/utils/schema/CONSTRAINTS.md +166 -0
  574. package/src/utils/schema/adapt.ts +36 -0
  575. package/src/utils/schema/compatibility.ts +435 -0
  576. package/src/utils/schema/dereference.ts +98 -0
  577. package/src/utils/schema/draft.ts +341 -0
  578. package/src/utils/schema/equality.ts +97 -0
  579. package/src/utils/schema/fields.ts +207 -0
  580. package/src/utils/schema/index.ts +15 -0
  581. package/src/utils/schema/json-schema-validator.ts +595 -0
  582. package/src/utils/schema/meta-validator.ts +167 -0
  583. package/src/utils/schema/normalize.ts +1901 -0
  584. package/src/utils/schema/spill.ts +43 -0
  585. package/src/utils/schema/stamps.ts +109 -0
  586. package/src/utils/schema/strict-tool-validation.ts +117 -0
  587. package/src/utils/schema/types.ts +10 -0
  588. package/src/utils/schema/typescript.ts +198 -0
  589. package/src/utils/schema/wire.ts +789 -0
  590. package/src/utils/schema/zod-decontaminate.ts +331 -0
  591. package/src/utils/sdk-stream-timeout.ts +43 -0
  592. package/src/utils/sse-debug.ts +18 -0
  593. package/src/utils/stream-markup-healing.ts +247 -0
  594. package/src/utils/thinking-loop.ts +552 -0
  595. package/src/utils/tool-call-loop-guard.ts +107 -0
  596. package/src/utils/tool-choice.ts +99 -0
  597. package/src/utils/validation.ts +1507 -0
  598. package/src/utils.ts +171 -0
package/src/stream.ts ADDED
@@ -0,0 +1,1778 @@
1
+ import * as crypto from "node:crypto";
2
+ import * as fsSync from "node:fs";
3
+ import * as fs from "node:fs/promises";
4
+ import * as path from "node:path";
5
+ import { scheduler } from "node:timers/promises";
6
+ import type { Effort } from "jeopi-catalog/effort";
7
+ import { isVertexExpressOpenAIUrl, isVertexRawPredictUrl } from "jeopi-catalog/hosts";
8
+ import {
9
+ mapEffortToAnthropicAdaptiveEffort,
10
+ mapEffortToGoogleThinkingLevel,
11
+ minimumSupportedEffort,
12
+ requireSupportedEffort,
13
+ resolveWireModelId,
14
+ } from "jeopi-catalog/model-thinking";
15
+ import { CATALOG_PROVIDERS, type ProviderCatalogEntry } from "jeopi-catalog/provider-models";
16
+ import { $env, $pickenv, getConfigRootDir, isEnoent, logger, withExtraCaFetch } from "jeopi-utils";
17
+ import { getCustomApi } from "./api-registry";
18
+ import { AUTH_RETRY_STEPS, isApiKeyResolver, resolveRetryKey } from "./auth-retry";
19
+ import * as AIError from "./error";
20
+ import { ProviderHttpError } from "./error";
21
+ import { isUsageLimitOutcome } from "./error/rate-limit";
22
+ import type { BedrockOptions } from "./providers/amazon-bedrock";
23
+ import type { AnthropicOptions } from "./providers/anthropic";
24
+ import type { CursorOptions } from "./providers/cursor";
25
+ import type { DevinOptions } from "./providers/devin";
26
+ import { isGitLabDuoModel, streamGitLabDuo } from "./providers/gitlab-duo";
27
+ import { type GitLabDuoWorkflowOptions, streamGitLabDuoWorkflow } from "./providers/gitlab-duo-workflow";
28
+ import type { GoogleOptions } from "./providers/google";
29
+ import { getVertexAccessToken } from "./providers/google-auth";
30
+ import type { GoogleGeminiCliOptions } from "./providers/google-gemini-cli";
31
+ import type { GoogleVertexOptions } from "./providers/google-vertex";
32
+ import { isKimiModel, streamKimi } from "./providers/kimi";
33
+ import type { OllamaChatOptions } from "./providers/ollama";
34
+ import type { OpenAICompletionsOptions } from "./providers/openai-completions";
35
+ import { streamPiNative } from "./providers/pi-native-client";
36
+ // Heavy provider stream functions are imported lazily via register-builtins,
37
+ // which wraps each provider module in a dynamic import. This keeps the
38
+ // AWS SDK, google-auth-library, @google/genai, @bufbuild/protobuf, and
39
+ // other provider SDKs out of the CLI startup parse graph. The
40
+ // gitlab-duo / kimi / synthetic providers stay eager because their modules
41
+ // export routing predicates (isGitLabDuoModel, isKimiModel, isSyntheticModel)
42
+ // that must be callable synchronously before streaming begins, and their
43
+ // modules are thin wrappers with no heavy SDK dependencies.
44
+ import {
45
+ streamAnthropic,
46
+ streamAzureOpenAIResponses,
47
+ streamBedrock,
48
+ streamCursor,
49
+ streamDevin,
50
+ streamGoogle,
51
+ streamGoogleGeminiCli,
52
+ streamGoogleVertex,
53
+ streamOllama,
54
+ streamOpenAICodexResponses,
55
+ streamOpenAICompletions,
56
+ streamOpenAIResponses,
57
+ } from "./providers/register-builtins";
58
+ import { isSyntheticModel, streamSynthetic } from "./providers/synthetic";
59
+ import { PROVIDER_REGISTRY } from "./registry";
60
+ import type {
61
+ Api,
62
+ AssistantMessage,
63
+ AssistantMessageEvent,
64
+ Context,
65
+ FetchImpl,
66
+ Model,
67
+ OptionsForApi,
68
+ SimpleStreamOptions,
69
+ StreamOptions,
70
+ ThinkingBudgets,
71
+ ToolChoice,
72
+ } from "./types";
73
+ import { AssistantMessageEventStream } from "./utils/event-stream";
74
+ import { wrapLeakedThinkingStream } from "./utils/leaked-thinking-stream";
75
+ import { wrapFetchForProxy } from "./utils/proxy";
76
+ import { withRequestDebugFetch } from "./utils/request-debug";
77
+ import { withGeminiThinkingLoopGuard } from "./utils/thinking-loop";
78
+
79
+ function isGoogleVertexAuthenticatedModel(model: Model<Api>): boolean {
80
+ return (
81
+ model.provider === "google-vertex" &&
82
+ ((model.api === "openai-completions" && isVertexExpressOpenAIUrl(model.baseUrl)) ||
83
+ (model.api === "anthropic-messages" && isVertexRawPredictUrl(model.baseUrl)))
84
+ );
85
+ }
86
+
87
+ type ProviderInFlightLease = {
88
+ path: string;
89
+ heartbeat: NodeJS.Timeout;
90
+ flushHeartbeat: () => Promise<void>;
91
+ };
92
+
93
+ type ProviderInFlightLeaseInfo = {
94
+ pid: number;
95
+ timestamp: number;
96
+ token: string;
97
+ };
98
+ type ProviderInFlightStaleLock = { token: string } | { mtimeMs: number };
99
+ type ProviderInFlightLockIdentity = { dev: number; ino: number; birthtimeMs: number };
100
+
101
+ const PROVIDER_INFLIGHT_LOCK_STALE_MS = 10_000;
102
+ const PROVIDER_INFLIGHT_LEASE_STALE_MS = 30_000;
103
+ const PROVIDER_INFLIGHT_HEARTBEAT_MS = 5_000;
104
+ const PROVIDER_INFLIGHT_SIGNAL_FALLBACK_MS = 250;
105
+
106
+ let configuredProviderMaxInFlightRequests: Record<string, number> = {};
107
+ let providerInFlightRootOverride: string | undefined;
108
+
109
+ export function configureProviderMaxInFlightRequests(limits: Record<string, number> | undefined): void {
110
+ configuredProviderMaxInFlightRequests = limits ?? {};
111
+ }
112
+
113
+ function resolveProviderInFlightLimit(
114
+ provider: string,
115
+ options?: Pick<StreamOptions, "maxInFlightRequests">,
116
+ ): number | undefined {
117
+ const limits = options?.maxInFlightRequests ?? configuredProviderMaxInFlightRequests;
118
+ const value = limits[provider];
119
+ if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) return undefined;
120
+ return Math.max(1, Math.floor(value));
121
+ }
122
+
123
+ function providerInFlightRoot(): string {
124
+ if (providerInFlightRootOverride) return providerInFlightRootOverride;
125
+ return path.join(getConfigRootDir(), "run", "provider-inflight");
126
+ }
127
+
128
+ function providerInFlightSegment(provider: string): string {
129
+ return crypto.createHash("sha256").update(provider).digest("base64url");
130
+ }
131
+
132
+ function providerInFlightDir(provider: string): string {
133
+ return path.join(providerInFlightRoot(), providerInFlightSegment(provider));
134
+ }
135
+
136
+ function providerInFlightSignalPath(provider: string): string {
137
+ return path.join(providerInFlightDir(provider), ".wakeup");
138
+ }
139
+
140
+ function providerInFlightLockDir(provider: string): string {
141
+ return `${providerInFlightDir(provider)}.lock`;
142
+ }
143
+
144
+ // `process.kill(pid, 0)` may throw for permission/sandbox reasons even when a
145
+ // process exists. Treat non-ESRCH failures as alive; timestamp expiry still
146
+ // reaps leases whose heartbeat stopped.
147
+ function isProcessAlive(pid: number): boolean {
148
+ try {
149
+ process.kill(pid, 0);
150
+ return true;
151
+ } catch (error) {
152
+ return (error as NodeJS.ErrnoException).code !== "ESRCH";
153
+ }
154
+ }
155
+
156
+ async function readProviderInFlightInfo(infoPath: string): Promise<ProviderInFlightLeaseInfo | null> {
157
+ try {
158
+ const content = await fs.readFile(infoPath, "utf-8");
159
+ const parsed = JSON.parse(content) as Partial<ProviderInFlightLeaseInfo>;
160
+ if (typeof parsed.pid !== "number" || typeof parsed.timestamp !== "number" || typeof parsed.token !== "string") {
161
+ return null;
162
+ }
163
+ return { pid: parsed.pid, timestamp: parsed.timestamp, token: parsed.token };
164
+ } catch {
165
+ return null;
166
+ }
167
+ }
168
+
169
+ async function writeProviderInFlightInfo(dir: string, token: string): Promise<void> {
170
+ const info: ProviderInFlightLeaseInfo = { pid: process.pid, timestamp: Date.now(), token };
171
+ const infoPath = path.join(dir, "info.json");
172
+ const tempPath = path.join(dir, `.info-${process.pid}-${crypto.randomUUID()}.tmp`);
173
+ try {
174
+ await Bun.write(tempPath, JSON.stringify(info));
175
+ await fs.rename(tempPath, infoPath);
176
+ } catch (error) {
177
+ await fs.rm(tempPath, { force: true }).catch(() => {});
178
+ throw error;
179
+ }
180
+ }
181
+
182
+ async function isProviderInFlightDirStale(dir: string, staleMs: number): Promise<boolean> {
183
+ const info = await readProviderInFlightInfo(path.join(dir, "info.json"));
184
+ if (info) {
185
+ if (!isProcessAlive(info.pid)) return true;
186
+ return Date.now() - info.timestamp > staleMs;
187
+ }
188
+
189
+ try {
190
+ const stat = await fs.stat(path.join(dir, "info.json"));
191
+ return Date.now() - stat.mtimeMs > staleMs;
192
+ } catch (error) {
193
+ if (!isEnoent(error)) throw error;
194
+ }
195
+
196
+ try {
197
+ const stat = await fs.stat(dir);
198
+ return Date.now() - stat.mtimeMs > staleMs;
199
+ } catch (error) {
200
+ if (isEnoent(error)) return false;
201
+ throw error;
202
+ }
203
+ }
204
+
205
+ async function readProviderInFlightStaleLock(lockDir: string): Promise<ProviderInFlightStaleLock | null> {
206
+ const infoPath = path.join(lockDir, "info.json");
207
+ const info = await readProviderInFlightInfo(infoPath);
208
+ if (info) return isProcessAlive(info.pid) ? null : { token: info.token };
209
+
210
+ try {
211
+ const stat = await fs.stat(lockDir);
212
+ return Date.now() - stat.mtimeMs > PROVIDER_INFLIGHT_LOCK_STALE_MS ? { mtimeMs: stat.mtimeMs } : null;
213
+ } catch (error) {
214
+ if (isEnoent(error)) return null;
215
+ throw error;
216
+ }
217
+ }
218
+
219
+ async function readProviderInFlightLockIdentity(lockDir: string): Promise<ProviderInFlightLockIdentity> {
220
+ const stat = await fs.stat(lockDir);
221
+ return { dev: stat.dev, ino: stat.ino, birthtimeMs: stat.birthtimeMs };
222
+ }
223
+
224
+ function isSameProviderInFlightLock(
225
+ current: ProviderInFlightLockIdentity,
226
+ expected: ProviderInFlightLockIdentity,
227
+ ): boolean {
228
+ if (current.dev !== expected.dev) return false;
229
+ if (current.ino !== 0 || expected.ino !== 0) return current.ino === expected.ino;
230
+ return current.birthtimeMs === expected.birthtimeMs;
231
+ }
232
+
233
+ async function releaseProviderInFlightStaleLock(lockDir: string, stale: ProviderInFlightStaleLock): Promise<void> {
234
+ if ("token" in stale) {
235
+ await releaseProviderInFlightLock(lockDir, stale.token);
236
+ return;
237
+ }
238
+
239
+ const infoPath = path.join(lockDir, "info.json");
240
+ if (await readProviderInFlightInfo(infoPath)) return;
241
+ try {
242
+ const stat = await fs.stat(lockDir);
243
+ if (stat.mtimeMs !== stale.mtimeMs || Date.now() - stat.mtimeMs <= PROVIDER_INFLIGHT_LOCK_STALE_MS) return;
244
+ await fs.rm(lockDir, { recursive: true, force: true });
245
+ } catch {}
246
+ }
247
+
248
+ // Best-effort token-checked release. A token mismatch means another process has
249
+ // already replaced the lock, so the fresh lock must be left intact.
250
+ async function releaseProviderInFlightLock(lockDir: string, token: string): Promise<void> {
251
+ try {
252
+ const info = await readProviderInFlightInfo(path.join(lockDir, "info.json"));
253
+ if (!info || info.token !== token) return;
254
+ await fs.rm(lockDir, { recursive: true, force: true });
255
+ } catch {}
256
+ }
257
+
258
+ async function releaseProviderInFlightLockDirIfSame(
259
+ lockDir: string,
260
+ identity: ProviderInFlightLockIdentity,
261
+ ): Promise<void> {
262
+ try {
263
+ if (await readProviderInFlightInfo(path.join(lockDir, "info.json"))) return;
264
+ const current = await readProviderInFlightLockIdentity(lockDir);
265
+ if (!isSameProviderInFlightLock(current, identity)) return;
266
+ await fs.rm(lockDir, { recursive: true, force: true });
267
+ } catch {}
268
+ }
269
+
270
+ async function acquireProviderInFlightLock(provider: string, signal?: AbortSignal): Promise<() => Promise<void>> {
271
+ const lockDir = providerInFlightLockDir(provider);
272
+ await fs.mkdir(path.dirname(lockDir), { recursive: true });
273
+
274
+ while (true) {
275
+ if (signal?.aborted) throw signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch");
276
+ try {
277
+ await fs.mkdir(lockDir);
278
+ const lockIdentity = await readProviderInFlightLockIdentity(lockDir);
279
+ const token = crypto.randomUUID();
280
+ try {
281
+ await writeProviderInFlightInfo(lockDir, token);
282
+ } catch (error) {
283
+ await releaseProviderInFlightLockDirIfSame(lockDir, lockIdentity);
284
+ throw error;
285
+ }
286
+ return async () => {
287
+ await releaseProviderInFlightLock(lockDir, token);
288
+ };
289
+ } catch (error) {
290
+ if ((error as NodeJS.ErrnoException).code !== "EEXIST") throw error;
291
+ }
292
+
293
+ const staleLock = await readProviderInFlightStaleLock(lockDir);
294
+ if (staleLock) {
295
+ await releaseProviderInFlightStaleLock(lockDir, staleLock);
296
+ await signalProviderInFlightWaiters(provider);
297
+ continue;
298
+ }
299
+
300
+ await waitForProviderInFlightSignal(provider, signal);
301
+ }
302
+ }
303
+
304
+ async function cleanupProviderInFlightLeases(providerDir: string): Promise<number> {
305
+ let active = 0;
306
+ let entries: string[];
307
+ try {
308
+ entries = await fs.readdir(providerDir);
309
+ } catch (error) {
310
+ if (isEnoent(error)) return 0;
311
+ throw error;
312
+ }
313
+
314
+ for (const entry of entries) {
315
+ const leaseDir = path.join(providerDir, entry);
316
+ let isDirectory = false;
317
+ try {
318
+ isDirectory = (await fs.stat(leaseDir)).isDirectory();
319
+ } catch (error) {
320
+ if (isEnoent(error)) continue;
321
+ throw error;
322
+ }
323
+ if (!isDirectory) continue;
324
+ if (await isProviderInFlightDirStale(leaseDir, PROVIDER_INFLIGHT_LEASE_STALE_MS)) {
325
+ await fs.rm(leaseDir, { recursive: true, force: true });
326
+ continue;
327
+ }
328
+ active++;
329
+ }
330
+ return active;
331
+ }
332
+
333
+ async function tryAcquireProviderInFlightLease(
334
+ provider: string,
335
+ limit: number,
336
+ signal?: AbortSignal,
337
+ ): Promise<ProviderInFlightLease | null> {
338
+ const releaseLock = await acquireProviderInFlightLock(provider, signal);
339
+ try {
340
+ const dir = providerInFlightDir(provider);
341
+ await fs.mkdir(dir, { recursive: true });
342
+ const active = await cleanupProviderInFlightLeases(dir);
343
+ if (active >= limit) return null;
344
+
345
+ const leaseDir = path.join(dir, `${process.pid}-${Date.now()}-${crypto.randomUUID()}`);
346
+ const token = crypto.randomUUID();
347
+ try {
348
+ await fs.mkdir(leaseDir);
349
+ await writeProviderInFlightInfo(leaseDir, token);
350
+ } catch (error) {
351
+ await removeProviderInFlightLeaseDir(leaseDir).catch(() => {});
352
+ throw error;
353
+ }
354
+ let heartbeatFlush = Promise.resolve();
355
+ const touchHeartbeat = () => {
356
+ heartbeatFlush = heartbeatFlush
357
+ .then(
358
+ () => writeProviderInFlightInfo(leaseDir, token),
359
+ () => writeProviderInFlightInfo(leaseDir, token),
360
+ )
361
+ .catch(() => {});
362
+ };
363
+ const heartbeat = setInterval(touchHeartbeat, PROVIDER_INFLIGHT_HEARTBEAT_MS);
364
+ heartbeat.unref?.();
365
+ return { path: leaseDir, heartbeat, flushHeartbeat: () => heartbeatFlush };
366
+ } finally {
367
+ await releaseLock();
368
+ }
369
+ }
370
+
371
+ async function signalProviderInFlightWaitersInDir(dir: string): Promise<void> {
372
+ try {
373
+ await fs.mkdir(dir, { recursive: true });
374
+ await Bun.write(path.join(dir, ".wakeup"), String(Date.now()));
375
+ } catch {}
376
+ }
377
+
378
+ async function signalProviderInFlightWaiters(provider: string): Promise<void> {
379
+ await signalProviderInFlightWaitersInDir(providerInFlightDir(provider));
380
+ }
381
+
382
+ function waitForProviderInFlightSignal(provider: string, signal?: AbortSignal): Promise<void> {
383
+ if (signal?.aborted)
384
+ return Promise.reject(signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch"));
385
+ const signalPath = providerInFlightSignalPath(provider);
386
+ const waitStarted = Date.now();
387
+ const { promise, resolve, reject } = Promise.withResolvers<void>();
388
+ let settled = false;
389
+ let watcher: fsSync.FSWatcher | undefined;
390
+ const timer = setTimeout(() => finish(resolve), PROVIDER_INFLIGHT_SIGNAL_FALLBACK_MS);
391
+ const finish = (settle: () => void) => {
392
+ if (settled) return;
393
+ settled = true;
394
+ clearTimeout(timer);
395
+ watcher?.close();
396
+ signal?.removeEventListener("abort", onAbort);
397
+ settle();
398
+ };
399
+ const onAbort = () => {
400
+ finish(() => reject(signal?.reason ?? new AIError.AbortError("Provider request aborted before dispatch")));
401
+ };
402
+ signal?.addEventListener("abort", onAbort, { once: true });
403
+ try {
404
+ watcher = fsSync.watch(providerInFlightDir(provider), (_event, filename) => {
405
+ if (filename === ".wakeup" || filename === null) {
406
+ finish(resolve);
407
+ }
408
+ });
409
+ void fs.stat(signalPath).then(
410
+ stat => {
411
+ if (stat.mtimeMs >= waitStarted) finish(resolve);
412
+ },
413
+ error => {
414
+ if (!isEnoent(error)) finish(resolve);
415
+ },
416
+ );
417
+ } catch {
418
+ // Filesystem notifications are best-effort across platforms; the fallback
419
+ // timer keeps stale-lock/lease cleanup progressing if an event is dropped.
420
+ }
421
+ return promise;
422
+ }
423
+
424
+ async function removeProviderInFlightLeaseDir(leasePath: string): Promise<void> {
425
+ for (let attempt = 0; attempt < 3; attempt++) {
426
+ try {
427
+ await fs.rm(leasePath, { recursive: true, force: true });
428
+ return;
429
+ } catch (error) {
430
+ if (isEnoent(error)) return;
431
+ const code = (error as NodeJS.ErrnoException).code;
432
+ if (attempt < 2 && (code === "EBUSY" || code === "ENOTEMPTY" || code === "EPERM")) {
433
+ await Bun.sleep(25);
434
+ continue;
435
+ }
436
+ throw error;
437
+ }
438
+ }
439
+ }
440
+
441
+ // Signal into the lease's OWN provider directory (derived from `lease.path`)
442
+ // rather than recomputing it from the current root. A release that lands after
443
+ // the in-flight root has been repointed (only the test seam does that) must not
444
+ // write `.wakeup` into an unrelated provider directory.
445
+ async function releaseProviderInFlightLease(lease: ProviderInFlightLease): Promise<void> {
446
+ clearInterval(lease.heartbeat);
447
+ await lease.flushHeartbeat();
448
+ await removeProviderInFlightLeaseDir(lease.path);
449
+ await signalProviderInFlightWaitersInDir(path.dirname(lease.path));
450
+ }
451
+
452
+ async function acquireProviderInFlightSlot(
453
+ provider: string,
454
+ limit: number | undefined,
455
+ signal?: AbortSignal,
456
+ ): Promise<() => Promise<void>> {
457
+ if (limit === undefined) return async () => {};
458
+ let loggedWait = false;
459
+ while (true) {
460
+ if (signal?.aborted) throw signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch");
461
+ const lease = await tryAcquireProviderInFlightLease(provider, limit, signal);
462
+ if (lease) return () => releaseProviderInFlightLease(lease);
463
+ if (!loggedWait) {
464
+ loggedWait = true;
465
+ logger.debug("Provider in-flight limit blocked request", { provider, limit });
466
+ }
467
+ await waitForProviderInFlightSignal(provider, signal);
468
+ }
469
+ }
470
+
471
+ export const __providerInFlightForTesting = {
472
+ setRoot(root: string | undefined): void {
473
+ providerInFlightRootOverride = root;
474
+ },
475
+ providerDir(provider: string): string {
476
+ return providerInFlightDir(provider);
477
+ },
478
+ lockDir(provider: string): string {
479
+ return providerInFlightLockDir(provider);
480
+ },
481
+ async captureStaleLockRelease(provider: string): Promise<(() => Promise<void>) | null> {
482
+ const lockDir = providerInFlightLockDir(provider);
483
+ const stale = await readProviderInFlightStaleLock(lockDir);
484
+ if (!stale) return null;
485
+ return () => releaseProviderInFlightStaleLock(lockDir, stale);
486
+ },
487
+ async captureLockDirRelease(provider: string): Promise<(() => Promise<void>) | null> {
488
+ const lockDir = providerInFlightLockDir(provider);
489
+ try {
490
+ const identity = await readProviderInFlightLockIdentity(lockDir);
491
+ return () => releaseProviderInFlightLockDirIfSame(lockDir, identity);
492
+ } catch {
493
+ return null;
494
+ }
495
+ },
496
+ };
497
+
498
+ function withProviderInFlightLimit<TOptions extends Pick<StreamOptions, "signal" | "maxInFlightRequests">>(
499
+ model: Model<Api>,
500
+ options: TOptions | undefined,
501
+ dispatch: () => AssistantMessageEventStream,
502
+ ): AssistantMessageEventStream {
503
+ // Leaked-thinking healing folds in here — the one shared provider-dispatch
504
+ // chokepoint — so the loop guard (which wraps this) sees healed events and all
505
+ // six provider exits are covered by one wrap. Healing is idempotent.
506
+ const limit = resolveProviderInFlightLimit(model.provider, options);
507
+ if (limit === undefined) return wrapLeakedThinkingStream(dispatch());
508
+
509
+ const outer = new AssistantMessageEventStream();
510
+ void (async () => {
511
+ let release: (() => Promise<void>) | undefined;
512
+ let released = false;
513
+ const releaseOnce = async () => {
514
+ if (!release || released) return;
515
+ released = true;
516
+ await release();
517
+ };
518
+ try {
519
+ const startedWaitingAt = Date.now();
520
+ release = await acquireProviderInFlightSlot(model.provider, limit, options?.signal);
521
+ if (Date.now() - startedWaitingAt >= PROVIDER_INFLIGHT_SIGNAL_FALLBACK_MS) {
522
+ logger.debug("Provider in-flight limit wait completed", { provider: model.provider, limit });
523
+ }
524
+ if (options?.signal?.aborted) {
525
+ throw options.signal.reason ?? new AIError.AbortError("Provider request aborted before dispatch");
526
+ }
527
+ const inner = wrapLeakedThinkingStream(dispatch());
528
+ try {
529
+ for await (const event of inner) {
530
+ outer.push(event);
531
+ if (outer.done) return;
532
+ }
533
+ if (!outer.done) outer.end(await inner.result());
534
+ } finally {
535
+ await releaseOnce();
536
+ }
537
+ } catch (error) {
538
+ await releaseOnce();
539
+ if (!outer.done) outer.fail(error);
540
+ }
541
+ })();
542
+ return outer;
543
+ }
544
+
545
+ function createVertexAuthenticatedFetch(options: StreamOptions | undefined): FetchImpl {
546
+ const baseFetch = options?.fetch ?? fetch;
547
+ const vertexFetch = async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
548
+ const token = await getVertexAccessToken({ signal: options?.signal, fetch: baseFetch });
549
+ const headers = new Headers(init?.headers);
550
+ headers.set("Authorization", `Bearer ${token}`);
551
+ const rewritten = resolveVertexRequest(input);
552
+ const url = rewritten instanceof Request ? rewritten.url : rewritten.toString();
553
+ if (isVertexRawPredictUrl(url)) {
554
+ const bodyText = await readVertexRequestBody(rewritten, init);
555
+ const transformed = transformVertexAnthropicBody(bodyText);
556
+ return baseFetch(url, {
557
+ ...init,
558
+ method: init?.method ?? (rewritten instanceof Request ? rewritten.method : "POST"),
559
+ headers,
560
+ body: transformed,
561
+ });
562
+ }
563
+ return baseFetch(rewritten, { ...init, headers });
564
+ };
565
+ return Object.assign(vertexFetch, baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {});
566
+ }
567
+
568
+ async function readVertexRequestBody(input: string | URL | Request, init: RequestInit | undefined): Promise<string> {
569
+ if (input instanceof Request) return input.clone().text();
570
+ const body = init?.body;
571
+ if (typeof body === "string") return body;
572
+ if (body instanceof Uint8Array) return new TextDecoder().decode(body);
573
+ if (body instanceof ArrayBuffer) return new TextDecoder().decode(body);
574
+ return "";
575
+ }
576
+
577
+ // Vertex Claude rejects the standard Anthropic body shape: the `model` field
578
+ // is encoded in the URL path and `anthropic_version: "vertex-2023-10-16"` is
579
+ // required in the JSON body instead of the `anthropic-version` HTTP header.
580
+ function transformVertexAnthropicBody(bodyText: string): string {
581
+ if (!bodyText) return bodyText;
582
+ try {
583
+ const payload = JSON.parse(bodyText) as Record<string, unknown>;
584
+ delete payload.model;
585
+ payload.anthropic_version = "vertex-2023-10-16";
586
+ return JSON.stringify(payload);
587
+ } catch {
588
+ return bodyText;
589
+ }
590
+ }
591
+
592
+ function resolveVertexRequest(input: string | URL | Request): string | URL | Request {
593
+ const project = $env.GOOGLE_CLOUD_PROJECT || $env.GCP_PROJECT || $env.GCLOUD_PROJECT;
594
+ const location = $env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION;
595
+ if (!project || !location) return input;
596
+
597
+ const rewriteUrl = (url: string): string => {
598
+ const hasPlaceholder =
599
+ url.includes("{project}") ||
600
+ url.includes("{location}") ||
601
+ url.includes("%7Bproject%7D") ||
602
+ url.includes("%7Blocation%7D");
603
+ const host = location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`;
604
+ const rewritten = hasPlaceholder
605
+ ? url
606
+ .replace("https://{location}-aiplatform.googleapis.com", `https://${host}`)
607
+ .replace("https://%7Blocation%7D-aiplatform.googleapis.com", `https://${host}`)
608
+ .replaceAll("{project}", encodeURIComponent(project))
609
+ .replaceAll("%7Bproject%7D", encodeURIComponent(project))
610
+ .replaceAll("{location}", encodeURIComponent(location))
611
+ .replaceAll("%7Blocation%7D", encodeURIComponent(location))
612
+ : url;
613
+ return rewritten.replace(":streamRawPredict/v1/messages", ":streamRawPredict");
614
+ };
615
+
616
+ if (input instanceof Request) {
617
+ const rewrittenUrl = rewriteUrl(input.url);
618
+ return rewrittenUrl === input.url ? input : new Request(rewrittenUrl, input);
619
+ }
620
+ if (input instanceof URL) {
621
+ const rewrittenUrl = rewriteUrl(input.toString());
622
+ return rewrittenUrl === input.toString() ? input : new URL(rewrittenUrl);
623
+ }
624
+ return rewriteUrl(input);
625
+ }
626
+
627
+ type KeyResolver = string | (() => string | undefined);
628
+
629
+ const LEGACY_ENV_KEYS: Record<string, KeyResolver> = {
630
+ // Non-provider / search-tool keys and API-name keys not modeled as registry provider defs.
631
+ "azure-openai-responses": "AZURE_OPENAI_API_KEY",
632
+ exa: "EXA_API_KEY",
633
+ jina: "JINA_API_KEY",
634
+ brave: "BRAVE_API_KEY",
635
+ tinyfish: "TINYFISH_API_KEY",
636
+ firecrawl: "FIRECRAWL_API_KEY",
637
+ };
638
+
639
+ /**
640
+ * Env fallbacks derived from the catalog table — the single source for plain
641
+ * provider env-var names. Registry defs override with computed resolvers
642
+ * (Foundry/ADC/Bedrock probes); legacy non-provider keys merge last.
643
+ */
644
+ const CATALOG_ENTRY_ENV_KEYS = (CATALOG_PROVIDERS as readonly ProviderCatalogEntry[]).flatMap(provider => {
645
+ const envVars = provider.envVars;
646
+ if (!envVars || envVars.length === 0) return [];
647
+ const resolver: KeyResolver = envVars.length === 1 ? envVars[0] : () => $pickenv(...envVars);
648
+ return [[provider.id, resolver] as [string, KeyResolver]];
649
+ });
650
+
651
+ const serviceProviderMap: Record<string, KeyResolver> = {
652
+ ...Object.fromEntries(CATALOG_ENTRY_ENV_KEYS),
653
+ ...Object.fromEntries(
654
+ PROVIDER_REGISTRY.flatMap(provider =>
655
+ provider.envKeys != null ? [[provider.id, provider.envKeys] as [string, KeyResolver]] : [],
656
+ ),
657
+ ),
658
+ ...LEGACY_ENV_KEYS,
659
+ };
660
+
661
+ /**
662
+ * Get API key for provider from known environment variables, e.g. OPENAI_API_KEY.
663
+ *
664
+ * Will not return API keys for providers that require OAuth tokens.
665
+ * Checks Bun.env, then cwd/.env, then ~/.env.
666
+ */
667
+ export function getEnvApiKey(provider: string): string | undefined {
668
+ const resolver = serviceProviderMap[provider];
669
+ if (typeof resolver === "string") {
670
+ return $env[resolver];
671
+ }
672
+ return resolver?.();
673
+ }
674
+
675
+ /**
676
+ * Name of the environment variable that backs `getEnvApiKey` for a provider,
677
+ * when that provider maps to a single named variable (e.g. `github-copilot` →
678
+ * `COPILOT_GITHUB_TOKEN`). Returns undefined for providers whose env fallback
679
+ * is computed (multi-var pickers, Vertex ADC / Bedrock probes, …) since no
680
+ * single variable name describes the source.
681
+ */
682
+ export function getEnvApiKeyName(provider: string): string | undefined {
683
+ const resolver = serviceProviderMap[provider];
684
+ return typeof resolver === "string" ? resolver : undefined;
685
+ }
686
+
687
+ /**
688
+ * Enumerate every provider that has an env-var fallback for `getEnvApiKey`.
689
+ * Used by `omp auth-broker migrate --include-env` to discover env-sourced keys
690
+ * that should be uploaded to the broker.
691
+ */
692
+ export function listProvidersWithEnvKey(): string[] {
693
+ return Object.keys(serviceProviderMap);
694
+ }
695
+
696
+ export function stream<TApi extends Api>(
697
+ model: Model<TApi>,
698
+ context: Context,
699
+ options?: OptionsForApi<TApi>,
700
+ ): AssistantMessageEventStream {
701
+ return withGeminiThinkingLoopGuard(model, options, opts =>
702
+ withProviderInFlightLimit(model, opts, () => streamDispatch(model, context, opts)),
703
+ );
704
+ }
705
+
706
+ function streamDispatch<TApi extends Api>(
707
+ model: Model<TApi>,
708
+ context: Context,
709
+ options?: OptionsForApi<TApi>,
710
+ ): AssistantMessageEventStream {
711
+ const baseOptions = (options || {}) as StreamOptions;
712
+ const debugOptions = withExtraCaFetch(withRequestDebugFetch(baseOptions));
713
+ const requestOptions = {
714
+ ...debugOptions,
715
+ fetch: wrapFetchForProxy(debugOptions.fetch ?? (globalThis.fetch as FetchImpl), model.provider),
716
+ } as OptionsForApi<TApi>;
717
+
718
+ // Check custom API registry first (extension-provided APIs like "vertex-claude-api")
719
+ const customApiProvider = getCustomApi(model.api);
720
+ if (customApiProvider) {
721
+ return customApiProvider.stream(model, context, requestOptions as StreamOptions);
722
+ }
723
+
724
+ if (isGitLabDuoModel(model)) {
725
+ const apiKey = requestOptions.apiKey || getEnvApiKey(model.provider);
726
+ if (!apiKey) {
727
+ throw new AIError.MissingApiKeyError(model.provider);
728
+ }
729
+ return streamGitLabDuo(model, context, {
730
+ ...(requestOptions as SimpleStreamOptions),
731
+ apiKey,
732
+ });
733
+ }
734
+
735
+ if (model.api === "gitlab-duo-agent") {
736
+ const apiKey = (requestOptions as StreamOptions | undefined)?.apiKey || getEnvApiKey(model.provider);
737
+ if (!apiKey) {
738
+ throw new AIError.MissingApiKeyError(model.provider);
739
+ }
740
+ return streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, {
741
+ ...(requestOptions as StreamOptions | undefined),
742
+ apiKey,
743
+ } as GitLabDuoWorkflowOptions);
744
+ }
745
+
746
+ // Vertex AI uses Application Default Credentials, not API keys
747
+ if (model.api === "google-vertex") {
748
+ return streamGoogleVertex(model as Model<"google-vertex">, context, requestOptions as GoogleVertexOptions);
749
+ } else if (model.api === "bedrock-converse-stream") {
750
+ // Bedrock doesn't have any API keys instead it sources credentials from standard AWS env variables or from given AWS profile.
751
+ return streamBedrock(model as Model<"bedrock-converse-stream">, context, requestOptions as BedrockOptions);
752
+ }
753
+
754
+ const apiKey = requestOptions.apiKey || getEnvApiKey(model.provider);
755
+ if (!apiKey) {
756
+ throw new AIError.MissingApiKeyError(model.provider);
757
+ }
758
+ const providerOptions = isGoogleVertexAuthenticatedModel(model)
759
+ ? {
760
+ ...requestOptions,
761
+ apiKey: "vertex-adc",
762
+ fetch: createVertexAuthenticatedFetch(requestOptions),
763
+ }
764
+ : { ...requestOptions, apiKey };
765
+
766
+ const api: Api = model.api;
767
+ switch (api) {
768
+ case "anthropic-messages": {
769
+ const anthropicOptions = providerOptions as AnthropicOptions;
770
+ return streamAnthropic(model as Model<"anthropic-messages">, context, {
771
+ ...anthropicOptions,
772
+ isOAuth: anthropicOptions.isOAuth ?? model.isOAuth,
773
+ });
774
+ }
775
+
776
+ case "openrouter": {
777
+ const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0";
778
+ if (useResponses) {
779
+ return streamOpenAIResponses(
780
+ model as Model<"openai-responses">,
781
+ context,
782
+ providerOptions as OptionsForApi<"openai-responses">,
783
+ );
784
+ }
785
+ return streamOpenAICompletions(
786
+ model as Model<"openai-completions">,
787
+ context,
788
+ providerOptions as OptionsForApi<"openai-completions">,
789
+ );
790
+ }
791
+
792
+ case "openai-completions":
793
+ return streamOpenAICompletions(
794
+ model as Model<"openai-completions">,
795
+ context,
796
+ providerOptions as OptionsForApi<"openai-completions">,
797
+ );
798
+
799
+ case "openai-responses":
800
+ return streamOpenAIResponses(
801
+ model as Model<"openai-responses">,
802
+ context,
803
+ providerOptions as OptionsForApi<"openai-responses">,
804
+ );
805
+
806
+ case "azure-openai-responses":
807
+ return streamAzureOpenAIResponses(
808
+ model as Model<"azure-openai-responses">,
809
+ context,
810
+ providerOptions as OptionsForApi<"azure-openai-responses">,
811
+ );
812
+
813
+ case "openai-codex-responses":
814
+ return streamOpenAICodexResponses(
815
+ model as Model<"openai-codex-responses">,
816
+ context,
817
+ providerOptions as OptionsForApi<"openai-codex-responses">,
818
+ );
819
+
820
+ case "google-generative-ai":
821
+ return streamGoogle(model as Model<"google-generative-ai">, context, providerOptions);
822
+
823
+ case "google-gemini-cli":
824
+ return streamGoogleGeminiCli(
825
+ model as Model<"google-gemini-cli">,
826
+ context,
827
+ providerOptions as GoogleGeminiCliOptions,
828
+ );
829
+
830
+ case "ollama-chat":
831
+ return streamOllama(model as Model<"ollama-chat">, context, providerOptions as OllamaChatOptions);
832
+
833
+ case "cursor-agent":
834
+ return streamCursor(model as Model<"cursor-agent">, context, providerOptions as CursorOptions);
835
+
836
+ case "devin-agent":
837
+ return streamDevin(model as Model<"devin-agent">, context, providerOptions as DevinOptions);
838
+
839
+ default:
840
+ throw new AIError.ConfigurationError(`Unhandled API: ${api}`);
841
+ }
842
+ }
843
+
844
+ /** Thinking-loop re-samples spent before {@link resolveWithThinkingLoopCook} cooks. */
845
+ const THINKING_LOOP_MAX_ABORTS = 3;
846
+ const THINKING_LOOP_RETRY_BASE_DELAY_MS = 500;
847
+ const THINKING_LOOP_RETRY_MAX_DELAY_MS = 8_000;
848
+
849
+ /**
850
+ * Resolve a completion, re-sampling a thinking-loop stall up to
851
+ * {@link THINKING_LOOP_MAX_ABORTS} times before letting it cook. The loop guard
852
+ * raises an empty `stopReason: "error"` stall on each guarded attempt; this
853
+ * result-path consumer re-dispatches a fresh request per stall and, once the abort
854
+ * budget is spent, runs one final pass with the guard disabled so a stubborn loop
855
+ * returns the model's raw output instead of a fatal stall. Non-stall results —
856
+ * including genuine errors — return immediately; a caller abort during backoff
857
+ * propagates so cancellation surfaces as an abort, never a stale stall result.
858
+ */
859
+ async function resolveWithThinkingLoopCook(
860
+ signal: AbortSignal | undefined,
861
+ dispatch: () => AssistantMessageEventStream,
862
+ cook: () => AssistantMessageEventStream,
863
+ ): Promise<AssistantMessage> {
864
+ let message = await dispatch().result();
865
+ let thinkingLoopRetry = AIError.is(message.errorId, AIError.Flag.ThinkingLoop);
866
+ for (let attempt = 0; thinkingLoopRetry && attempt < THINKING_LOOP_MAX_ABORTS - 1; attempt += 1) {
867
+ // A caller abort surfaces as a thrown abort (never the stall, which would
868
+ // misclassify as a 502): throwIfAborted before backoff, and scheduler.wait
869
+ // rejects if the abort lands mid-delay.
870
+ signal?.throwIfAborted();
871
+ const delay = Math.min(THINKING_LOOP_RETRY_BASE_DELAY_MS * 2 ** attempt, THINKING_LOOP_RETRY_MAX_DELAY_MS);
872
+ await scheduler.wait(delay, { signal });
873
+ message = await dispatch().result();
874
+ thinkingLoopRetry =
875
+ message.stopReason === "error" &&
876
+ message.content.length === 0 &&
877
+ AIError.is(message.errorId, AIError.Flag.ThinkingLoop);
878
+ }
879
+ if (!thinkingLoopRetry) return message;
880
+ signal?.throwIfAborted();
881
+ // Abort budget spent and still looping: let it cook with the guard disabled.
882
+ return cook().result();
883
+ }
884
+
885
+ export async function complete<TApi extends Api>(
886
+ model: Model<TApi>,
887
+ context: Context,
888
+ options?: OptionsForApi<TApi>,
889
+ ): Promise<AssistantMessage> {
890
+ return resolveWithThinkingLoopCook(
891
+ options?.signal,
892
+ () => stream(model, context, options),
893
+ () => stream(model, context, { ...options, loopGuard: { ...options?.loopGuard, enabled: false } }),
894
+ );
895
+ }
896
+
897
+ type AuthRetryFailure = {
898
+ error: unknown;
899
+ bufferedEvents: AssistantMessageEvent[];
900
+ terminalEvent?: Extract<AssistantMessageEvent, { type: "error" }>;
901
+ };
902
+
903
+ function extractStatusFromAssistantError(message: AssistantMessage): number | undefined {
904
+ if (message.errorStatus !== undefined) return message.errorStatus;
905
+ if (!message.errorMessage) return undefined;
906
+ return AIError.status({ message: message.errorMessage });
907
+ }
908
+
909
+ function isRetryableUpstreamError(error: unknown, status: number | undefined, message: string | undefined): boolean {
910
+ // 401 means the credential is bad. Usage-limit phrasing (Codex's
911
+ // "You have hit your ChatGPT usage limit", Anthropic's "usage_limit_reached",
912
+ // Google's "resource_exhausted", OpenAI's "insufficient_quota") and 429s
913
+ // without transient rate-limit wording mean this account is parked but a
914
+ // sibling credential can usually pick the request up. Both are rotatable
915
+ // via `onAuthError` — the auth-gateway maps the former to
916
+ // `invalidateCredentialMatching` and the latter to
917
+ // `markUsageLimitReached`. Transient 429s ("Too many requests",
918
+ // per-minute caps) classify as RATE_LIMIT_EXCEEDED in
919
+ // `parseRateLimitReason` and stay in the provider's own backoff layer
920
+ // instead of burning siblings.
921
+ if (status === 401) return true;
922
+ void error;
923
+ return isUsageLimitOutcome(status, message);
924
+ }
925
+
926
+ function createAssistantAuthError(message: AssistantMessage): Error {
927
+ const text = message.errorMessage ?? "Provider authentication failed";
928
+ const status = extractStatusFromAssistantError(message);
929
+ return status === undefined
930
+ ? new AIError.ProviderResponseError(text, { kind: "runtime" })
931
+ : new ProviderHttpError(text, status);
932
+ }
933
+
934
+ function emitBufferedEvents(stream: AssistantMessageEventStream, events: AssistantMessageEvent[]): void {
935
+ for (const event of events) {
936
+ stream.push(event);
937
+ }
938
+ }
939
+
940
+ export function streamSimple<TApi extends Api>(
941
+ model: Model<TApi>,
942
+ context: Context,
943
+ options?: SimpleStreamOptions,
944
+ ): AssistantMessageEventStream {
945
+ const baseOptions = (options || {}) as SimpleStreamOptions;
946
+ const debugOptions = withExtraCaFetch(withRequestDebugFetch(baseOptions));
947
+ const requestOptions = {
948
+ ...debugOptions,
949
+ fetch: wrapFetchForProxy(debugOptions.fetch ?? (globalThis.fetch as FetchImpl), model.provider),
950
+ } as SimpleStreamOptions;
951
+ const apiKeyResolver = isApiKeyResolver(requestOptions?.apiKey) ? requestOptions.apiKey : undefined;
952
+ if (apiKeyResolver) {
953
+ const outer = new AssistantMessageEventStream();
954
+ const signal = requestOptions?.signal;
955
+ // One inner attempt against a resolved string key. When
956
+ // `captureAuthFailure` is set, a retryable auth error that arrives before
957
+ // any replay-unsafe event is buffered and returned (so the caller can
958
+ // retry with a fresh key) instead of surfaced. The terminal attempt
959
+ // clears the flag and emits whatever it gets.
960
+ const runAttempt = async (apiKey: string, captureAuthFailure: boolean): Promise<AuthRetryFailure | undefined> => {
961
+ const bufferedEvents: AssistantMessageEvent[] = [];
962
+ let emittedReplayUnsafeEvent = false;
963
+ const flushBuffered = (): void => {
964
+ emitBufferedEvents(outer, bufferedEvents);
965
+ bufferedEvents.length = 0;
966
+ };
967
+
968
+ try {
969
+ const inner = streamSimple(model, context, { ...requestOptions, apiKey });
970
+ for await (const event of inner) {
971
+ if (!emittedReplayUnsafeEvent && event.type === "start") {
972
+ bufferedEvents.push(event);
973
+ continue;
974
+ }
975
+ if (
976
+ !emittedReplayUnsafeEvent &&
977
+ captureAuthFailure &&
978
+ event.type === "error" &&
979
+ isRetryableUpstreamError(
980
+ event.error,
981
+ extractStatusFromAssistantError(event.error),
982
+ event.error.errorMessage,
983
+ )
984
+ ) {
985
+ return { error: createAssistantAuthError(event.error), bufferedEvents, terminalEvent: event };
986
+ }
987
+ flushBuffered();
988
+ emittedReplayUnsafeEvent = true;
989
+ outer.push(event);
990
+ if (outer.done) return undefined;
991
+ }
992
+ flushBuffered();
993
+ if (!outer.done) outer.end(await inner.result());
994
+ } catch (error) {
995
+ if (
996
+ !emittedReplayUnsafeEvent &&
997
+ captureAuthFailure &&
998
+ isRetryableUpstreamError(
999
+ error,
1000
+ AIError.status(error),
1001
+ error instanceof Error ? error.message : undefined,
1002
+ )
1003
+ ) {
1004
+ return { error, bufferedEvents };
1005
+ }
1006
+ flushBuffered();
1007
+ outer.fail(error);
1008
+ }
1009
+ return undefined;
1010
+ };
1011
+ const emitFailure = (failure: AuthRetryFailure): void => {
1012
+ emitBufferedEvents(outer, failure.bufferedEvents);
1013
+ if (failure.terminalEvent) {
1014
+ outer.push(failure.terminalEvent);
1015
+ } else {
1016
+ outer.fail(failure.error);
1017
+ }
1018
+ };
1019
+
1020
+ void (async () => {
1021
+ let lastKey: string | undefined;
1022
+ try {
1023
+ lastKey = (await apiKeyResolver({ lastChance: false, error: undefined, signal })) || undefined;
1024
+ } catch (error) {
1025
+ // A thrown resolver is a broker/OAuth/network failure, not a missing
1026
+ // key — surface the cause instead of masking it as "No API key".
1027
+ outer.fail(
1028
+ new AIError.ConfigurationError(
1029
+ `Failed to resolve API key for provider ${model.provider}: ${error instanceof Error ? error.message : String(error)}`,
1030
+ { cause: error },
1031
+ ),
1032
+ );
1033
+ return;
1034
+ }
1035
+ if (lastKey === undefined) {
1036
+ outer.fail(new AIError.MissingApiKeyError(model.provider));
1037
+ return;
1038
+ }
1039
+ let failure = await runAttempt(lastKey, true);
1040
+ if (!failure) return;
1041
+ // a/b/c policy: refresh the same account (lastChance=false), then
1042
+ // switch to a sibling (lastChance=true). A step is skipped when the
1043
+ // resolver yields the same key it just tried or `undefined`; the
1044
+ // final step's attempt clears the capture flag so it emits directly.
1045
+ for (let step = 0; step < AUTH_RETRY_STEPS.length; step++) {
1046
+ // Caller aborted between attempts: don't mint a fresh token or fire
1047
+ // another doomed request — emit the captured failure instead.
1048
+ if (signal?.aborted) break;
1049
+ const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal);
1050
+ if (nextKey === undefined || nextKey === lastKey) continue;
1051
+ lastKey = nextKey;
1052
+ const isLastStep = step === AUTH_RETRY_STEPS.length - 1;
1053
+ const next = await runAttempt(nextKey, !isLastStep);
1054
+ if (!next) return;
1055
+ failure = next;
1056
+ }
1057
+ emitFailure(failure);
1058
+ })();
1059
+ return outer;
1060
+ }
1061
+
1062
+ // Pi-native transport short-circuits the per-provider dispatch entirely:
1063
+ // the gateway resolves provider + credential server-side, so we don't
1064
+ // need an `apiKey` from `getEnvApiKey` here — `options.apiKey` carries
1065
+ // the gateway bearer instead. Comes BEFORE the custom-API check so
1066
+ // extension-registered APIs can't accidentally override a configured
1067
+ // pi-native transport.
1068
+ if (model.transport === "pi-native") {
1069
+ return withGeminiThinkingLoopGuard(model, requestOptions, opts =>
1070
+ withProviderInFlightLimit(model, opts, () => streamPiNative(model, context, opts)),
1071
+ );
1072
+ }
1073
+
1074
+ // Check custom API registry (extension-provided APIs)
1075
+ const customApiProvider = getCustomApi(model.api);
1076
+ if (customApiProvider) {
1077
+ return withGeminiThinkingLoopGuard(model, requestOptions, opts =>
1078
+ withProviderInFlightLimit(model, opts, () => customApiProvider.streamSimple(model, context, opts)),
1079
+ );
1080
+ }
1081
+
1082
+ // Vertex AI uses Application Default Credentials, not API keys
1083
+ if (model.api === "google-vertex") {
1084
+ const providerOptions = mapOptionsForApi(model, requestOptions, undefined);
1085
+ return stream(model, context, providerOptions);
1086
+ } else if (model.api === "bedrock-converse-stream") {
1087
+ // Bedrock doesn't have any API keys instead it sources credentials from standard AWS env variables or from given AWS profile.
1088
+ const providerOptions = mapOptionsForApi(model, requestOptions, undefined);
1089
+ return stream(model, context, providerOptions);
1090
+ }
1091
+
1092
+ // The resolver form is handled by the wrapper above; only a static string
1093
+ // key reaches this point.
1094
+ const apiKey =
1095
+ (typeof requestOptions?.apiKey === "string" ? requestOptions.apiKey : undefined) || getEnvApiKey(model.provider);
1096
+ if (!apiKey) {
1097
+ throw new AIError.MissingApiKeyError(model.provider);
1098
+ }
1099
+
1100
+ // GitLab Duo - wraps Anthropic/OpenAI behind GitLab AI Gateway direct access tokens
1101
+ if (isGitLabDuoModel(model)) {
1102
+ return withProviderInFlightLimit(model, requestOptions, () =>
1103
+ streamGitLabDuo(model, context, {
1104
+ ...requestOptions,
1105
+ apiKey,
1106
+ }),
1107
+ );
1108
+ }
1109
+
1110
+ // GitLab Duo Workflow - IDE workflow protocol + WebSocket action bridge
1111
+ if (model.api === "gitlab-duo-agent") {
1112
+ // Does not route through withProviderInFlightLimit, so heal explicitly.
1113
+ return wrapLeakedThinkingStream(
1114
+ streamGitLabDuoWorkflow(model as Model<"gitlab-duo-agent">, context, {
1115
+ ...requestOptions,
1116
+ apiKey,
1117
+ }),
1118
+ );
1119
+ }
1120
+
1121
+ // Kimi Code - route to dedicated handler that wraps OpenAI or Anthropic API
1122
+ if (isKimiModel(model)) {
1123
+ // Pass raw SimpleStreamOptions - streamKimi handles mapping internally
1124
+ return withProviderInFlightLimit(model, requestOptions, () =>
1125
+ streamKimi(model as Model<"openai-completions">, context, {
1126
+ ...requestOptions,
1127
+ apiKey,
1128
+ format: requestOptions?.kimiApiFormat ?? "anthropic",
1129
+ }),
1130
+ );
1131
+ }
1132
+
1133
+ // Synthetic - route to dedicated handler that wraps OpenAI or Anthropic API
1134
+ if (isSyntheticModel(model)) {
1135
+ // Pass raw SimpleStreamOptions - streamSynthetic handles mapping internally
1136
+ return withProviderInFlightLimit(model, requestOptions, () =>
1137
+ streamSynthetic(model as Model<"openai-completions">, context, {
1138
+ ...requestOptions,
1139
+ apiKey,
1140
+ format: requestOptions?.syntheticApiFormat ?? "openai", // Default to OpenAI format
1141
+ }),
1142
+ );
1143
+ }
1144
+ const providerOptions = mapOptionsForApi(model, requestOptions, apiKey);
1145
+ return stream(model, context, providerOptions);
1146
+ }
1147
+
1148
+ export async function completeSimple<TApi extends Api>(
1149
+ model: Model<TApi>,
1150
+ context: Context,
1151
+ options?: SimpleStreamOptions,
1152
+ ): Promise<AssistantMessage> {
1153
+ return resolveWithThinkingLoopCook(
1154
+ options?.signal,
1155
+ () => streamSimple(model, context, options),
1156
+ () => streamSimple(model, context, { ...options, loopGuard: { ...options?.loopGuard, enabled: false } }),
1157
+ );
1158
+ }
1159
+
1160
+ const MIN_OUTPUT_TOKENS = 1024;
1161
+ // Fallback total output cap for models whose catalog entry has no maxTokens.
1162
+ const OUTPUT_CAP_WHEN_UNKNOWN = 64_000;
1163
+ function maxTokensWithThinkingBudget(
1164
+ baseMaxTokens: number | undefined,
1165
+ modelMaxTokens: number | null,
1166
+ thinkingBudget: number,
1167
+ ): number {
1168
+ const uncappedMaxTokens = baseMaxTokens === undefined ? OUTPUT_CAP_WHEN_UNKNOWN : baseMaxTokens + thinkingBudget;
1169
+ return Math.min(uncappedMaxTokens, modelMaxTokens ?? Number.POSITIVE_INFINITY);
1170
+ }
1171
+ export const OUTPUT_FALLBACK_BUFFER = 4000;
1172
+ const ANTHROPIC_USE_INTERLEAVED_THINKING = Bun.env.PI_NO_INTERLEAVED_THINKING !== "1";
1173
+
1174
+ export const ANTHROPIC_THINKING: Record<Effort, number> = {
1175
+ minimal: 1024,
1176
+ low: 4096,
1177
+ medium: 8192,
1178
+ high: 16384,
1179
+ xhigh: 32768,
1180
+ };
1181
+
1182
+ const GOOGLE_THINKING: Record<Effort, number> = {
1183
+ minimal: 1024,
1184
+ low: 4096,
1185
+ medium: 8192,
1186
+ high: 16384,
1187
+ xhigh: 24575,
1188
+ };
1189
+
1190
+ const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
1191
+ minimal: 1024,
1192
+ low: 2048,
1193
+ medium: 8192,
1194
+ high: 16384,
1195
+ xhigh: 16384,
1196
+ };
1197
+
1198
+ function resolveBedrockThinkingBudget(
1199
+ model: Model<"bedrock-converse-stream">,
1200
+ options?: SimpleStreamOptions,
1201
+ ): { budget: number; level: Effort } | null {
1202
+ if (!options?.reasoning || !model.reasoning) return null;
1203
+ const level = requireSupportedEffort(model, options.reasoning);
1204
+ const budget = options.thinkingBudgets?.[level] ?? BEDROCK_CLAUDE_THINKING[level];
1205
+ return { budget, level };
1206
+ }
1207
+
1208
+ export function mapAnthropicToolChoice(choice?: ToolChoice): AnthropicOptions["toolChoice"] {
1209
+ if (!choice) return undefined;
1210
+ if (typeof choice === "string") {
1211
+ if (choice === "required") return "any";
1212
+ if (choice === "auto" || choice === "none" || choice === "any") return choice;
1213
+ return undefined;
1214
+ }
1215
+ if (choice.type === "tool") {
1216
+ return choice.name ? { type: "tool", name: choice.name } : undefined;
1217
+ }
1218
+ if (choice.type === "function") {
1219
+ const name = "function" in choice ? choice.function?.name : choice.name;
1220
+ return name ? { type: "tool", name } : undefined;
1221
+ }
1222
+ return undefined;
1223
+ }
1224
+
1225
+ export function mapGoogleToolChoice(
1226
+ choice?: ToolChoice,
1227
+ ): GoogleOptions["toolChoice"] | GoogleGeminiCliOptions["toolChoice"] | GoogleVertexOptions["toolChoice"] {
1228
+ if (!choice) return undefined;
1229
+ if (typeof choice === "string") {
1230
+ if (choice === "required") return "any";
1231
+ if (choice === "auto" || choice === "none" || choice === "any") return choice;
1232
+ return undefined;
1233
+ }
1234
+ // Named-tool routing on Google: emit an `ANY`-mode allow-list of one entry,
1235
+ // mirroring the Anthropic mapper that returns `{type: "tool", name}`.
1236
+ if (choice.type === "tool") {
1237
+ return choice.name ? { mode: "ANY", allowedFunctionNames: [choice.name] } : undefined;
1238
+ }
1239
+ if (choice.type === "function") {
1240
+ const name = "function" in choice ? choice.function?.name : choice.name;
1241
+ return name ? { mode: "ANY", allowedFunctionNames: [name] } : undefined;
1242
+ }
1243
+ return undefined;
1244
+ }
1245
+
1246
+ function mapOpenAiToolChoice(choice?: ToolChoice): OpenAICompletionsOptions["toolChoice"] {
1247
+ if (!choice) return undefined;
1248
+ if (typeof choice === "string") {
1249
+ if (choice === "any") return "required";
1250
+ if (choice === "auto" || choice === "none" || choice === "required") return choice;
1251
+ return undefined;
1252
+ }
1253
+ if (choice.type === "tool") {
1254
+ return choice.name ? { type: "function", function: { name: choice.name } } : undefined;
1255
+ }
1256
+ if (choice.type === "function") {
1257
+ const name = "function" in choice ? choice.function?.name : choice.name;
1258
+ return name ? { type: "function", function: { name } } : undefined;
1259
+ }
1260
+ return undefined;
1261
+ }
1262
+
1263
+ type ReasoningEffortMapCompat = {
1264
+ reasoningEffortMap?: Partial<Record<Effort, string>>;
1265
+ };
1266
+
1267
+ function getCompatReasoningEffortMap<TApi extends Api>(
1268
+ model: Model<TApi>,
1269
+ ): Partial<Record<Effort, string>> | undefined {
1270
+ const compat = model.compat;
1271
+ if (compat === undefined || typeof compat !== "object" || !("reasoningEffortMap" in compat)) {
1272
+ return undefined;
1273
+ }
1274
+ return (compat as ReasoningEffortMapCompat).reasoningEffortMap;
1275
+ }
1276
+
1277
+ function resolveSupportedMappedReasoningEffort<TApi extends Api>(
1278
+ model: Model<TApi>,
1279
+ reasoning: Effort,
1280
+ ): Effort | undefined {
1281
+ const mapped = getCompatReasoningEffortMap(model)?.[reasoning];
1282
+ if (!mapped) return undefined;
1283
+ const mappedEffort = mapped as Effort;
1284
+ return model.thinking?.efforts.includes(mappedEffort) ? mappedEffort : undefined;
1285
+ }
1286
+
1287
+ function resolveOpenAiReasoningEffort<TApi extends Api>(
1288
+ model: Model<TApi>,
1289
+ options?: SimpleStreamOptions,
1290
+ ): Effort | undefined {
1291
+ const reasoning = options?.reasoning;
1292
+ if (!reasoning || !model.reasoning) return undefined;
1293
+ // Models that reason natively but expose no effort dial carry
1294
+ // `thinking: undefined` (baked at build time from
1295
+ // `compat.supportsReasoningEffort: false` on openai-responses*). The
1296
+ // wire-side omitReasoningEffort gate (stream.ts) is the actual strip; returning
1297
+ // undefined here avoids a redundant requireSupportedEffort throw that would
1298
+ // defeat the gate and surface a confusing "Compaction failed: Thinking effort
1299
+ // high is not supported by..." to the user.
1300
+ if (!model.thinking) return undefined;
1301
+ if (model.thinking.efforts.includes(reasoning)) return reasoning;
1302
+ const mappedReasoning = resolveSupportedMappedReasoningEffort(model, reasoning);
1303
+ if (mappedReasoning) return mappedReasoning;
1304
+ if (getCompatReasoningEffortMap(model)?.[reasoning] !== undefined) return reasoning;
1305
+ if (model.thinking.effortMap?.[reasoning] !== undefined) return reasoning;
1306
+ return requireSupportedEffort(model, reasoning);
1307
+ }
1308
+
1309
+ const castApi = <TApi extends Api>(api: OptionsForApi<TApi>): OptionsForApi<Api> => api as OptionsForApi<Api>;
1310
+
1311
+ /**
1312
+ * Mandatory-reasoning endpoints (`thinking.requiresEffort`) reject disabled
1313
+ * or omitted thinking ("Reasoning is mandatory for this endpoint and cannot
1314
+ * be disabled") — clamp to the lowest supported effort instead.
1315
+ * `suppressWhenOff` models handle off provider-side via explicit wire
1316
+ * suppression. Collapsed pairs interplay: pair derivation strips member
1317
+ * flags (off routes to a bare SKU that CAN disable), while identity backfill
1318
+ * re-flags pairs whose logical id is itself mandatory (Gemini 3.x) — there
1319
+ * the clamp wins and the floored effort routes to the thinking SKU.
1320
+ */
1321
+ function normalizeMandatoryReasoningOptions<TApi extends Api>(
1322
+ model: Model<TApi>,
1323
+ options?: SimpleStreamOptions,
1324
+ ): SimpleStreamOptions | undefined {
1325
+ if (
1326
+ !model.reasoning ||
1327
+ !model.thinking?.requiresEffort ||
1328
+ model.thinking.suppressWhenOff ||
1329
+ (options?.reasoning !== undefined && !options.disableReasoning)
1330
+ ) {
1331
+ return options;
1332
+ }
1333
+ const floor = minimumSupportedEffort(model);
1334
+ if (floor === undefined) return options;
1335
+ return { ...options, reasoning: floor, disableReasoning: undefined };
1336
+ }
1337
+
1338
+ function mapOptionsForApi<TApi extends Api>(
1339
+ model: Model<TApi>,
1340
+ rawOptions?: SimpleStreamOptions,
1341
+ apiKey?: string,
1342
+ ): OptionsForApi<TApi> {
1343
+ const options = normalizeMandatoryReasoningOptions(model, rawOptions);
1344
+ const base = {
1345
+ temperature: options?.temperature,
1346
+ topP: options?.topP,
1347
+ topK: options?.topK,
1348
+ minP: options?.minP,
1349
+ presencePenalty: options?.presencePenalty,
1350
+ repetitionPenalty: options?.repetitionPenalty,
1351
+ maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
1352
+ signal: options?.signal,
1353
+ apiKey: apiKey ?? (typeof options?.apiKey === "string" ? options.apiKey : undefined),
1354
+ cacheRetention: options?.cacheRetention,
1355
+ headers: options?.headers,
1356
+ initiatorOverride: options?.initiatorOverride,
1357
+ maxRetryDelayMs: options?.maxRetryDelayMs,
1358
+ metadata: options?.metadata,
1359
+ taskBudget: options?.taskBudget,
1360
+ sessionId: options?.sessionId,
1361
+ promptCacheKey: options?.promptCacheKey,
1362
+ streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs,
1363
+ streamIdleTimeoutMs: options?.streamIdleTimeoutMs,
1364
+ providerSessionState: options?.providerSessionState,
1365
+ useInteractionsApi: options?.useInteractionsApi,
1366
+ storeInteraction: options?.storeInteraction,
1367
+ previousInteractionId: options?.previousInteractionId,
1368
+ maxInFlightRequests: options?.maxInFlightRequests,
1369
+ onPayload: options?.onPayload,
1370
+ onResponse: options?.onResponse,
1371
+ onSseEvent: options?.onSseEvent,
1372
+ execHandlers: options?.execHandlers,
1373
+ fetch: options?.fetch,
1374
+ fallbacks: options?.fallbacks,
1375
+ };
1376
+
1377
+ switch (model.api) {
1378
+ case "anthropic-messages": {
1379
+ // Explicitly disable thinking when reasoning is not specified or model doesn't support it
1380
+ const reasoning = options?.reasoning;
1381
+ if (!reasoning || !model.reasoning) {
1382
+ return castApi<"anthropic-messages">({
1383
+ ...base,
1384
+ requestModelId: resolveWireModelId(model, undefined),
1385
+ thinkingEnabled: false,
1386
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
1387
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
1388
+ serviceTier: options?.serviceTier,
1389
+ });
1390
+ }
1391
+
1392
+ let thinkingBudget = options.thinkingBudgets?.[reasoning] ?? ANTHROPIC_THINKING[reasoning];
1393
+ if (thinkingBudget <= 0) {
1394
+ return castApi<"anthropic-messages">({
1395
+ ...base,
1396
+ requestModelId: resolveWireModelId(model, undefined),
1397
+ thinkingEnabled: false,
1398
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
1399
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
1400
+ serviceTier: options?.serviceTier,
1401
+ });
1402
+ }
1403
+
1404
+ const thinkingMode = model.thinking?.mode;
1405
+ const effort =
1406
+ thinkingMode === "anthropic-adaptive" || thinkingMode === "anthropic-budget-effort"
1407
+ ? mapEffortToAnthropicAdaptiveEffort(model, reasoning)
1408
+ : undefined;
1409
+
1410
+ // For Opus 4.6+ and Sonnet 4.6+: use adaptive thinking with effort level
1411
+ // For older models: use budget-based thinking
1412
+ if (thinkingMode === "anthropic-adaptive") {
1413
+ return castApi<"anthropic-messages">({
1414
+ ...base,
1415
+ requestModelId: resolveWireModelId(model, reasoning),
1416
+ thinkingEnabled: true,
1417
+ effort,
1418
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
1419
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
1420
+ serviceTier: options?.serviceTier,
1421
+ });
1422
+ }
1423
+
1424
+ if (ANTHROPIC_USE_INTERLEAVED_THINKING) {
1425
+ return castApi<"anthropic-messages">({
1426
+ ...base,
1427
+ requestModelId: resolveWireModelId(model, reasoning),
1428
+ thinkingEnabled: true,
1429
+ thinkingBudgetTokens: thinkingBudget,
1430
+ effort,
1431
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
1432
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
1433
+ serviceTier: options?.serviceTier,
1434
+ });
1435
+ }
1436
+
1437
+ // Caller's maxTokens is desired output, so add thinking budget on top. With no caller/model cap, use a finite total fallback.
1438
+ const maxTokens = maxTokensWithThinkingBudget(base.maxTokens, model.maxTokens, thinkingBudget);
1439
+
1440
+ // If not enough room for thinking + output, reduce thinking budget
1441
+ if (maxTokens <= thinkingBudget) {
1442
+ thinkingBudget = maxTokens - MIN_OUTPUT_TOKENS;
1443
+ }
1444
+
1445
+ // If thinking budget is too low, disable thinking
1446
+ if (thinkingBudget <= 0) {
1447
+ return castApi<"anthropic-messages">({
1448
+ ...base,
1449
+ requestModelId: resolveWireModelId(model, undefined),
1450
+ thinkingEnabled: false,
1451
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
1452
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
1453
+ serviceTier: options?.serviceTier,
1454
+ });
1455
+ } else {
1456
+ return castApi<"anthropic-messages">({
1457
+ ...base,
1458
+ maxTokens,
1459
+ requestModelId: resolveWireModelId(model, reasoning),
1460
+ thinkingEnabled: true,
1461
+ thinkingBudgetTokens: thinkingBudget,
1462
+ effort,
1463
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
1464
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
1465
+ serviceTier: options?.serviceTier,
1466
+ });
1467
+ }
1468
+ }
1469
+
1470
+ case "bedrock-converse-stream": {
1471
+ const bedrockBase: BedrockOptions = {
1472
+ ...base,
1473
+ reasoning: options?.reasoning,
1474
+ thinkingBudgets: options?.thinkingBudgets,
1475
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
1476
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
1477
+ };
1478
+ // Adaptive mode sends effort directly, no budget_tokens — skip budget inflation.
1479
+ if (model.thinking?.mode === "anthropic-adaptive") {
1480
+ return castApi<"bedrock-converse-stream">(bedrockBase);
1481
+ }
1482
+ const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
1483
+ if (!budgetInfo) return bedrockBase as OptionsForApi<TApi>;
1484
+ let maxTokens = bedrockBase.maxTokens ?? model.maxTokens ?? OUTPUT_CAP_WHEN_UNKNOWN;
1485
+ let thinkingBudgets = bedrockBase.thinkingBudgets;
1486
+ if (maxTokens <= budgetInfo.budget) {
1487
+ const desiredMaxTokens = Math.min(
1488
+ model.maxTokens ?? Number.POSITIVE_INFINITY,
1489
+ budgetInfo.budget + MIN_OUTPUT_TOKENS,
1490
+ );
1491
+ if (desiredMaxTokens > maxTokens) {
1492
+ maxTokens = desiredMaxTokens;
1493
+ }
1494
+ }
1495
+ if (maxTokens <= budgetInfo.budget) {
1496
+ const adjustedBudget = Math.max(0, maxTokens - MIN_OUTPUT_TOKENS);
1497
+ thinkingBudgets = { ...(thinkingBudgets ?? {}), [budgetInfo.level]: adjustedBudget };
1498
+ }
1499
+ return castApi<"bedrock-converse-stream">({ ...bedrockBase, maxTokens, thinkingBudgets });
1500
+ }
1501
+
1502
+ case "openrouter": {
1503
+ const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0";
1504
+ if (useResponses) {
1505
+ return castApi<"openai-responses">({
1506
+ ...base,
1507
+ reasoning: resolveOpenAiReasoningEffort(model, options),
1508
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
1509
+ serviceTier: options?.serviceTier,
1510
+ reasoningSummary: options?.hideThinkingSummary ? null : undefined,
1511
+ openrouterVariant: options?.openrouterVariant,
1512
+ maxTokensExplicit: rawOptions?.maxTokens !== undefined,
1513
+ disableReasoning: options?.disableReasoning,
1514
+ textVerbosity: options?.textVerbosity,
1515
+ });
1516
+ }
1517
+ return castApi<"openai-completions">({
1518
+ ...base,
1519
+ reasoning: resolveOpenAiReasoningEffort(model, options),
1520
+ disableReasoning: options?.disableReasoning,
1521
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
1522
+ serviceTier: options?.serviceTier,
1523
+ openrouterVariant: options?.openrouterVariant,
1524
+ maxTokensExplicit: rawOptions?.maxTokens !== undefined,
1525
+ });
1526
+ }
1527
+
1528
+ case "openai-completions":
1529
+ return castApi<"openai-completions">({
1530
+ ...base,
1531
+ reasoning: resolveOpenAiReasoningEffort(model, options),
1532
+ disableReasoning: options?.disableReasoning,
1533
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
1534
+ serviceTier: options?.serviceTier,
1535
+ openrouterVariant: options?.openrouterVariant,
1536
+ maxTokensExplicit: rawOptions?.maxTokens !== undefined,
1537
+ });
1538
+
1539
+ case "openai-responses":
1540
+ return castApi<"openai-responses">({
1541
+ ...base,
1542
+ reasoning: resolveOpenAiReasoningEffort(model, options),
1543
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
1544
+ serviceTier: options?.serviceTier,
1545
+ reasoningSummary: options?.hideThinkingSummary ? null : undefined,
1546
+ openrouterVariant: options?.openrouterVariant,
1547
+ maxTokensExplicit: rawOptions?.maxTokens !== undefined,
1548
+ disableReasoning: options?.disableReasoning,
1549
+ textVerbosity: options?.textVerbosity,
1550
+ });
1551
+
1552
+ case "azure-openai-responses":
1553
+ return castApi<"azure-openai-responses">({
1554
+ ...base,
1555
+ reasoning: resolveOpenAiReasoningEffort(model, options),
1556
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
1557
+ serviceTier: options?.serviceTier,
1558
+ reasoningSummary: options?.hideThinkingSummary ? null : undefined,
1559
+ });
1560
+
1561
+ case "openai-codex-responses":
1562
+ return castApi<"openai-codex-responses">({
1563
+ ...base,
1564
+ reasoning: resolveOpenAiReasoningEffort(model, options),
1565
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
1566
+ serviceTier: options?.serviceTier,
1567
+ preferWebsockets: options?.preferWebsockets,
1568
+ reasoningSummary: options?.hideThinkingSummary ? null : "detailed",
1569
+ textVerbosity: options?.textVerbosity,
1570
+ });
1571
+
1572
+ case "google-generative-ai": {
1573
+ // Explicitly disable thinking when reasoning is not specified or model doesn't support it
1574
+ // This is needed because Gemini has "dynamic thinking" enabled by default
1575
+ const reasoning = options?.reasoning;
1576
+ if (!reasoning || !model.reasoning) {
1577
+ return castApi<"google-generative-ai">({
1578
+ ...base,
1579
+ serviceTier: options?.serviceTier,
1580
+ thinking: { enabled: false },
1581
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
1582
+ });
1583
+ }
1584
+
1585
+ const googleModel = model as Model<"google-generative-ai">;
1586
+ const effort = requireSupportedEffort(googleModel, reasoning);
1587
+
1588
+ // Gemini 3+ models use thinkingLevel exclusively instead of thinkingBudget.
1589
+ // https://ai.google.dev/gemini-api/docs/thinking#set-budget
1590
+ if (googleModel.thinking?.mode === "google-level") {
1591
+ return castApi<"google-generative-ai">({
1592
+ ...base,
1593
+ serviceTier: options?.serviceTier,
1594
+ thinking: {
1595
+ enabled: true,
1596
+ level: mapEffortToGoogleThinkingLevel(effort),
1597
+ },
1598
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
1599
+ });
1600
+ }
1601
+
1602
+ return castApi<"google-gemini-cli">({
1603
+ ...base,
1604
+ thinking: {
1605
+ enabled: true,
1606
+ budgetTokens: getGoogleBudget(googleModel, effort, options?.thinkingBudgets),
1607
+ },
1608
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
1609
+ });
1610
+ }
1611
+
1612
+ case "google-gemini-cli": {
1613
+ const reasoning = options?.reasoning;
1614
+ const toolChoice = mapGoogleToolChoice(options?.toolChoice);
1615
+ if (reasoning && model.reasoning) {
1616
+ const effort = requireSupportedEffort(model, reasoning);
1617
+
1618
+ // Gemini 3+ models use thinkingLevel instead of thinkingBudget
1619
+ if (model.thinking?.mode === "google-level") {
1620
+ return castApi<"google-gemini-cli">({
1621
+ ...base,
1622
+ requestModelId: resolveWireModelId(model, effort),
1623
+ thinking: {
1624
+ enabled: true,
1625
+ level: mapEffortToGoogleThinkingLevel(effort),
1626
+ },
1627
+ toolChoice,
1628
+ antigravityEndpointMode: options?.antigravityEndpointMode,
1629
+ });
1630
+ }
1631
+
1632
+ let thinkingBudget =
1633
+ options.thinkingBudgets?.[effort] ?? model.thinking?.effortBudgets?.[effort] ?? GOOGLE_THINKING[effort];
1634
+
1635
+ // Caller's maxTokens is desired output, so add thinking budget on top. With no caller/model cap, use a finite total fallback.
1636
+ const maxTokens = maxTokensWithThinkingBudget(base.maxTokens, model.maxTokens, thinkingBudget);
1637
+
1638
+ // If not enough room for thinking + output, reduce thinking budget
1639
+ if (maxTokens <= thinkingBudget) {
1640
+ thinkingBudget = Math.max(0, maxTokens - MIN_OUTPUT_TOKENS);
1641
+ }
1642
+
1643
+ if (thinkingBudget > 0) {
1644
+ return castApi<"google-gemini-cli">({
1645
+ ...base,
1646
+ maxTokens,
1647
+ requestModelId: resolveWireModelId(model, effort),
1648
+ thinking: { enabled: true, budgetTokens: thinkingBudget },
1649
+ toolChoice,
1650
+ antigravityEndpointMode: options?.antigravityEndpointMode,
1651
+ });
1652
+ }
1653
+ // Budget clamped to zero — fall through to the thinking-off path.
1654
+ }
1655
+
1656
+ const thinking: GoogleGeminiCliOptions["thinking"] = { enabled: false };
1657
+ if (model.reasoning && model.thinking?.suppressWhenOff) {
1658
+ // CCA re-applies the per-id baked server default when the config
1659
+ // is omitted; suppression must be explicit on the wire.
1660
+ thinking.suppress = model.thinking.mode === "google-level" ? { level: "MINIMAL" } : { budget: 0 };
1661
+ }
1662
+ return castApi<"google-gemini-cli">({
1663
+ ...base,
1664
+ requestModelId: resolveWireModelId(model, undefined),
1665
+ thinking,
1666
+ toolChoice,
1667
+ antigravityEndpointMode: options?.antigravityEndpointMode,
1668
+ });
1669
+ }
1670
+
1671
+ case "google-vertex": {
1672
+ // Explicitly disable thinking when reasoning is not specified or model doesn't support it
1673
+ const reasoning = options?.reasoning;
1674
+ if (!reasoning || !model.reasoning) {
1675
+ return castApi<"google-vertex">({
1676
+ ...base,
1677
+ serviceTier: options?.serviceTier,
1678
+ thinking: { enabled: false },
1679
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
1680
+ });
1681
+ }
1682
+
1683
+ const vertexModel = model as Model<"google-vertex">;
1684
+ const effort = requireSupportedEffort(vertexModel, reasoning);
1685
+ const geminiModel = vertexModel as unknown as Model<"google-generative-ai">;
1686
+
1687
+ if (geminiModel.thinking?.mode === "google-level") {
1688
+ return castApi<"google-vertex">({
1689
+ ...base,
1690
+ serviceTier: options?.serviceTier,
1691
+ thinking: {
1692
+ enabled: true,
1693
+ level: mapEffortToGoogleThinkingLevel(effort),
1694
+ },
1695
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
1696
+ });
1697
+ }
1698
+
1699
+ return castApi<"google-vertex">({
1700
+ ...base,
1701
+ serviceTier: options?.serviceTier,
1702
+ thinking: {
1703
+ enabled: true,
1704
+ budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets),
1705
+ },
1706
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
1707
+ });
1708
+ }
1709
+
1710
+ case "ollama-chat":
1711
+ return castApi<"ollama-chat">({
1712
+ ...base,
1713
+ reasoning: resolveOpenAiReasoningEffort(model, options),
1714
+ disableReasoning: options?.disableReasoning,
1715
+ toolChoice: options?.toolChoice,
1716
+ });
1717
+
1718
+ case "cursor-agent": {
1719
+ const execHandlers = options?.cursorExecHandlers ?? options?.execHandlers;
1720
+ const onToolResult = options?.cursorOnToolResult ?? execHandlers?.onToolResult;
1721
+ return castApi<"cursor-agent">({
1722
+ ...base,
1723
+ execHandlers,
1724
+ onToolResult,
1725
+ });
1726
+ }
1727
+
1728
+ case "gitlab-duo-agent":
1729
+ return castApi<"gitlab-duo-agent">({
1730
+ ...base,
1731
+ cwd: options?.cwd,
1732
+ toolChoice: options?.toolChoice,
1733
+ });
1734
+ case "devin-agent": {
1735
+ const devinModel = model as Model<"devin-agent">;
1736
+ const effort =
1737
+ options?.reasoning && !options.disableReasoning
1738
+ ? requireSupportedEffort(devinModel, options.reasoning)
1739
+ : undefined;
1740
+ return castApi<"devin-agent">({
1741
+ ...base,
1742
+ chatModelUid: resolveWireModelId(devinModel, effort),
1743
+ });
1744
+ }
1745
+ default:
1746
+ throw new AIError.ConfigurationError(`Unhandled API in mapOptionsForApi: ${model.api}`);
1747
+ }
1748
+ }
1749
+
1750
+ function getGoogleBudget(
1751
+ model: Model<"google-generative-ai">,
1752
+ effort: Effort,
1753
+ customBudgets?: ThinkingBudgets,
1754
+ ): number {
1755
+ requireSupportedEffort(model, effort);
1756
+
1757
+ // Custom budgets take precedence if provided for this level
1758
+ if (customBudgets?.[effort] !== undefined) {
1759
+ return customBudgets[effort]!;
1760
+ }
1761
+
1762
+ // See https://ai.google.dev/gemini-api/docs/thinking#set-budget
1763
+ if (model.id.includes("2.5-")) {
1764
+ switch (effort) {
1765
+ case "minimal":
1766
+ return 128;
1767
+ case "low":
1768
+ return 2048;
1769
+ case "medium":
1770
+ return 8192;
1771
+ default:
1772
+ return model.id.includes("2.5-flash") ? 24576 : 32768;
1773
+ }
1774
+ }
1775
+
1776
+ // Unknown model - use dynamic
1777
+ return -1;
1778
+ }