@sayknow-cli/ai 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (361) hide show
  1. package/CHANGELOG.md +2788 -0
  2. package/README.md +1183 -0
  3. package/dist/types/api-registry.d.ts +30 -0
  4. package/dist/types/auth-broker/client.d.ts +66 -0
  5. package/dist/types/auth-broker/index.d.ts +5 -0
  6. package/dist/types/auth-broker/refresher.d.ts +25 -0
  7. package/dist/types/auth-broker/remote-store.d.ts +96 -0
  8. package/dist/types/auth-broker/server.d.ts +32 -0
  9. package/dist/types/auth-broker/types.d.ts +105 -0
  10. package/dist/types/auth-broker/wire-schemas.d.ts +412 -0
  11. package/dist/types/auth-gateway/http.d.ts +39 -0
  12. package/dist/types/auth-gateway/index.d.ts +3 -0
  13. package/dist/types/auth-gateway/server.d.ts +17 -0
  14. package/dist/types/auth-gateway/types.d.ts +115 -0
  15. package/dist/types/auth-storage.d.ts +660 -0
  16. package/dist/types/cli.d.ts +2 -0
  17. package/dist/types/index.d.ts +51 -0
  18. package/dist/types/model-cache.d.ts +17 -0
  19. package/dist/types/model-manager.d.ts +62 -0
  20. package/dist/types/model-thinking.d.ts +74 -0
  21. package/dist/types/models.d.ts +12 -0
  22. package/dist/types/provider-details.d.ts +24 -0
  23. package/dist/types/provider-models/bundled-references.d.ts +4 -0
  24. package/dist/types/provider-models/descriptors.d.ts +48 -0
  25. package/dist/types/provider-models/google.d.ts +20 -0
  26. package/dist/types/provider-models/index.d.ts +5 -0
  27. package/dist/types/provider-models/ollama.d.ts +7 -0
  28. package/dist/types/provider-models/openai-compat.d.ts +244 -0
  29. package/dist/types/provider-models/special.d.ts +16 -0
  30. package/dist/types/providers/amazon-bedrock.d.ts +60 -0
  31. package/dist/types/providers/anthropic-messages-server-schema.d.ts +450 -0
  32. package/dist/types/providers/anthropic-messages-server.d.ts +17 -0
  33. package/dist/types/providers/anthropic.d.ts +198 -0
  34. package/dist/types/providers/aws-credentials.d.ts +43 -0
  35. package/dist/types/providers/aws-eventstream.d.ts +38 -0
  36. package/dist/types/providers/aws-sigv4.d.ts +55 -0
  37. package/dist/types/providers/azure-openai-responses.d.ts +15 -0
  38. package/dist/types/providers/composer-discipline.d.ts +26 -0
  39. package/dist/types/providers/cursor/gen/agent_pb.d.ts +13022 -0
  40. package/dist/types/providers/cursor.d.ts +44 -0
  41. package/dist/types/providers/error-message.d.ts +27 -0
  42. package/dist/types/providers/github-copilot-headers.d.ts +40 -0
  43. package/dist/types/providers/gitlab-duo.d.ts +27 -0
  44. package/dist/types/providers/google-auth.d.ts +24 -0
  45. package/dist/types/providers/google-gemini-cli.d.ts +72 -0
  46. package/dist/types/providers/google-gemini-headers.d.ts +18 -0
  47. package/dist/types/providers/google-shared.d.ts +173 -0
  48. package/dist/types/providers/google-types.d.ts +138 -0
  49. package/dist/types/providers/google-vertex.d.ts +7 -0
  50. package/dist/types/providers/google.d.ts +4 -0
  51. package/dist/types/providers/grammar.d.ts +1 -0
  52. package/dist/types/providers/kimi.d.ts +27 -0
  53. package/dist/types/providers/mock.d.ts +175 -0
  54. package/dist/types/providers/ollama.d.ts +41 -0
  55. package/dist/types/providers/openai-anthropic-shim.d.ts +31 -0
  56. package/dist/types/providers/openai-chat-server-schema.d.ts +815 -0
  57. package/dist/types/providers/openai-chat-server.d.ts +16 -0
  58. package/dist/types/providers/openai-codex/constants.d.ts +26 -0
  59. package/dist/types/providers/openai-codex/request-transformer.d.ts +49 -0
  60. package/dist/types/providers/openai-codex/response-handler.d.ts +17 -0
  61. package/dist/types/providers/openai-codex-responses.d.ts +67 -0
  62. package/dist/types/providers/openai-completions-compat.d.ts +27 -0
  63. package/dist/types/providers/openai-completions.d.ts +33 -0
  64. package/dist/types/providers/openai-request-transform.d.ts +4 -0
  65. package/dist/types/providers/openai-responses-server-schema.d.ts +392 -0
  66. package/dist/types/providers/openai-responses-server.d.ts +17 -0
  67. package/dist/types/providers/openai-responses-shared.d.ts +89 -0
  68. package/dist/types/providers/openai-responses.d.ts +32 -0
  69. package/dist/types/providers/pi-native-client.d.ts +13 -0
  70. package/dist/types/providers/pi-native-server.d.ts +68 -0
  71. package/dist/types/providers/register-builtins.d.ts +31 -0
  72. package/dist/types/providers/synthetic.d.ts +26 -0
  73. package/dist/types/providers/transform-messages.d.ts +14 -0
  74. package/dist/types/providers/vision-guard.d.ts +8 -0
  75. package/dist/types/rate-limit-utils.d.ts +19 -0
  76. package/dist/types/stream.d.ts +43 -0
  77. package/dist/types/types.d.ts +811 -0
  78. package/dist/types/usage/claude.d.ts +3 -0
  79. package/dist/types/usage/gemini.d.ts +2 -0
  80. package/dist/types/usage/github-copilot.d.ts +7 -0
  81. package/dist/types/usage/google-antigravity.d.ts +2 -0
  82. package/dist/types/usage/grok-cli.d.ts +10 -0
  83. package/dist/types/usage/kimi.d.ts +2 -0
  84. package/dist/types/usage/minimax-code.d.ts +2 -0
  85. package/dist/types/usage/openai-codex.d.ts +3 -0
  86. package/dist/types/usage/shared.d.ts +1 -0
  87. package/dist/types/usage/zai.d.ts +2 -0
  88. package/dist/types/usage.d.ts +258 -0
  89. package/dist/types/utils/abort.d.ts +19 -0
  90. package/dist/types/utils/anthropic-auth.d.ts +31 -0
  91. package/dist/types/utils/discovery/antigravity.d.ts +61 -0
  92. package/dist/types/utils/discovery/codex.d.ts +38 -0
  93. package/dist/types/utils/discovery/cursor.d.ts +23 -0
  94. package/dist/types/utils/discovery/gemini.d.ts +25 -0
  95. package/dist/types/utils/discovery/index.d.ts +4 -0
  96. package/dist/types/utils/discovery/openai-compatible.d.ts +74 -0
  97. package/dist/types/utils/event-stream.d.ts +33 -0
  98. package/dist/types/utils/fireworks-model-id.d.ts +10 -0
  99. package/dist/types/utils/foundry.d.ts +1 -0
  100. package/dist/types/utils/h2-fetch.d.ts +22 -0
  101. package/dist/types/utils/http-inspector.d.ts +35 -0
  102. package/dist/types/utils/idle-iterator.d.ts +67 -0
  103. package/dist/types/utils/json-parse.d.ts +10 -0
  104. package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +18 -0
  105. package/dist/types/utils/oauth/anthropic.d.ts +22 -0
  106. package/dist/types/utils/oauth/api-key-login.d.ts +35 -0
  107. package/dist/types/utils/oauth/api-key-validation.d.ts +27 -0
  108. package/dist/types/utils/oauth/callback-server.d.ts +60 -0
  109. package/dist/types/utils/oauth/cerebras.d.ts +1 -0
  110. package/dist/types/utils/oauth/cloudflare-ai-gateway.d.ts +18 -0
  111. package/dist/types/utils/oauth/cursor.d.ts +15 -0
  112. package/dist/types/utils/oauth/deepseek.d.ts +10 -0
  113. package/dist/types/utils/oauth/firepass.d.ts +1 -0
  114. package/dist/types/utils/oauth/fireworks.d.ts +1 -0
  115. package/dist/types/utils/oauth/github-copilot.d.ts +38 -0
  116. package/dist/types/utils/oauth/gitlab-duo.d.ts +3 -0
  117. package/dist/types/utils/oauth/google-antigravity.d.ts +11 -0
  118. package/dist/types/utils/oauth/google-gemini-cli.d.ts +10 -0
  119. package/dist/types/utils/oauth/google-oauth-shared.d.ts +28 -0
  120. package/dist/types/utils/oauth/huggingface.d.ts +19 -0
  121. package/dist/types/utils/oauth/index.d.ts +38 -0
  122. package/dist/types/utils/oauth/kagi.d.ts +17 -0
  123. package/dist/types/utils/oauth/kilo.d.ts +5 -0
  124. package/dist/types/utils/oauth/kimi.d.ts +21 -0
  125. package/dist/types/utils/oauth/litellm.d.ts +18 -0
  126. package/dist/types/utils/oauth/lm-studio.d.ts +17 -0
  127. package/dist/types/utils/oauth/minimax-code.d.ts +28 -0
  128. package/dist/types/utils/oauth/moonshot.d.ts +1 -0
  129. package/dist/types/utils/oauth/nanogpt.d.ts +1 -0
  130. package/dist/types/utils/oauth/nvidia.d.ts +18 -0
  131. package/dist/types/utils/oauth/ollama-cloud.d.ts +2 -0
  132. package/dist/types/utils/oauth/ollama.d.ts +18 -0
  133. package/dist/types/utils/oauth/openai-codex.d.ts +21 -0
  134. package/dist/types/utils/oauth/opencode.d.ts +18 -0
  135. package/dist/types/utils/oauth/parallel.d.ts +17 -0
  136. package/dist/types/utils/oauth/perplexity.d.ts +9 -0
  137. package/dist/types/utils/oauth/pkce.d.ts +8 -0
  138. package/dist/types/utils/oauth/qianfan.d.ts +17 -0
  139. package/dist/types/utils/oauth/qwen-portal.d.ts +19 -0
  140. package/dist/types/utils/oauth/synthetic.d.ts +1 -0
  141. package/dist/types/utils/oauth/tavily.d.ts +17 -0
  142. package/dist/types/utils/oauth/together.d.ts +1 -0
  143. package/dist/types/utils/oauth/types.d.ts +45 -0
  144. package/dist/types/utils/oauth/venice.d.ts +18 -0
  145. package/dist/types/utils/oauth/vercel-ai-gateway.d.ts +18 -0
  146. package/dist/types/utils/oauth/vllm.d.ts +16 -0
  147. package/dist/types/utils/oauth/xai.d.ts +30 -0
  148. package/dist/types/utils/oauth/xiaomi.d.ts +25 -0
  149. package/dist/types/utils/oauth/zai.d.ts +18 -0
  150. package/dist/types/utils/oauth/zenmux.d.ts +1 -0
  151. package/dist/types/utils/overflow.d.ts +54 -0
  152. package/dist/types/utils/parse-bind.d.ts +23 -0
  153. package/dist/types/utils/provider-response.d.ts +3 -0
  154. package/dist/types/utils/retry-after.d.ts +3 -0
  155. package/dist/types/utils/retry-budget.d.ts +1 -0
  156. package/dist/types/utils/retry.d.ts +26 -0
  157. package/dist/types/utils/schema/adapt.d.ts +24 -0
  158. package/dist/types/utils/schema/compatibility.d.ts +30 -0
  159. package/dist/types/utils/schema/dereference.d.ts +11 -0
  160. package/dist/types/utils/schema/draft.d.ts +10 -0
  161. package/dist/types/utils/schema/equality.d.ts +4 -0
  162. package/dist/types/utils/schema/fields.d.ts +49 -0
  163. package/dist/types/utils/schema/index.d.ts +13 -0
  164. package/dist/types/utils/schema/json-schema-validator.d.ts +12 -0
  165. package/dist/types/utils/schema/meta-validator.d.ts +2 -0
  166. package/dist/types/utils/schema/normalize.d.ts +93 -0
  167. package/dist/types/utils/schema/spill.d.ts +8 -0
  168. package/dist/types/utils/schema/stamps.d.ts +25 -0
  169. package/dist/types/utils/schema/types.d.ts +4 -0
  170. package/dist/types/utils/schema/wire.d.ts +54 -0
  171. package/dist/types/utils/schema/zod-decontaminate.d.ts +31 -0
  172. package/dist/types/utils/sse-debug.d.ts +10 -0
  173. package/dist/types/utils/tool-call-healing.d.ts +71 -0
  174. package/dist/types/utils/tool-choice-capability.d.ts +41 -0
  175. package/dist/types/utils/tool-choice.d.ts +50 -0
  176. package/dist/types/utils/validation.d.ts +17 -0
  177. package/dist/types/utils.d.ts +34 -0
  178. package/package.json +146 -0
  179. package/src/api-registry.ts +96 -0
  180. package/src/auth-broker/client.ts +358 -0
  181. package/src/auth-broker/index.ts +5 -0
  182. package/src/auth-broker/refresher.ts +127 -0
  183. package/src/auth-broker/remote-store.ts +623 -0
  184. package/src/auth-broker/server.ts +644 -0
  185. package/src/auth-broker/types.ts +127 -0
  186. package/src/auth-broker/wire-schemas.ts +200 -0
  187. package/src/auth-gateway/http.ts +194 -0
  188. package/src/auth-gateway/index.ts +3 -0
  189. package/src/auth-gateway/server.ts +717 -0
  190. package/src/auth-gateway/types.ts +134 -0
  191. package/src/auth-storage.ts +4179 -0
  192. package/src/cli.ts +263 -0
  193. package/src/index.ts +56 -0
  194. package/src/model-cache.ts +129 -0
  195. package/src/model-manager.ts +486 -0
  196. package/src/model-thinking.ts +772 -0
  197. package/src/models.json +75437 -0
  198. package/src/models.json.d.ts +9 -0
  199. package/src/models.ts +82 -0
  200. package/src/prompts/turn-aborted-guidance.md +4 -0
  201. package/src/provider-details.ts +90 -0
  202. package/src/provider-models/bundled-references.ts +38 -0
  203. package/src/provider-models/descriptors.ts +327 -0
  204. package/src/provider-models/google.ts +91 -0
  205. package/src/provider-models/index.ts +5 -0
  206. package/src/provider-models/ollama.ts +153 -0
  207. package/src/provider-models/openai-compat.ts +2352 -0
  208. package/src/provider-models/special.ts +67 -0
  209. package/src/providers/amazon-bedrock.ts +937 -0
  210. package/src/providers/anthropic-messages-server-schema.ts +229 -0
  211. package/src/providers/anthropic-messages-server.ts +677 -0
  212. package/src/providers/anthropic.ts +2940 -0
  213. package/src/providers/aws-credentials.ts +501 -0
  214. package/src/providers/aws-eventstream.ts +185 -0
  215. package/src/providers/aws-sigv4.ts +218 -0
  216. package/src/providers/azure-openai-responses.ts +379 -0
  217. package/src/providers/composer-discipline.ts +41 -0
  218. package/src/providers/cursor/gen/agent_pb.ts +15274 -0
  219. package/src/providers/cursor/proto/agent.proto +3526 -0
  220. package/src/providers/cursor/proto/buf.gen.yaml +6 -0
  221. package/src/providers/cursor/proto/buf.yaml +17 -0
  222. package/src/providers/cursor.ts +2671 -0
  223. package/src/providers/error-message.ts +21 -0
  224. package/src/providers/github-copilot-headers.ts +140 -0
  225. package/src/providers/gitlab-duo.ts +372 -0
  226. package/src/providers/google-auth.ts +252 -0
  227. package/src/providers/google-gemini-cli.ts +856 -0
  228. package/src/providers/google-gemini-headers.ts +41 -0
  229. package/src/providers/google-shared.ts +951 -0
  230. package/src/providers/google-types.ts +167 -0
  231. package/src/providers/google-vertex.ts +88 -0
  232. package/src/providers/google.ts +41 -0
  233. package/src/providers/grammar.ts +70 -0
  234. package/src/providers/kimi.ts +52 -0
  235. package/src/providers/mock.ts +500 -0
  236. package/src/providers/ollama.ts +603 -0
  237. package/src/providers/openai-anthropic-shim.ts +138 -0
  238. package/src/providers/openai-chat-server-schema.ts +243 -0
  239. package/src/providers/openai-chat-server.ts +635 -0
  240. package/src/providers/openai-codex/constants.ts +43 -0
  241. package/src/providers/openai-codex/request-transformer.ts +161 -0
  242. package/src/providers/openai-codex/response-handler.ts +81 -0
  243. package/src/providers/openai-codex-responses.ts +2774 -0
  244. package/src/providers/openai-completions-compat.ts +289 -0
  245. package/src/providers/openai-completions.ts +1935 -0
  246. package/src/providers/openai-request-transform.ts +136 -0
  247. package/src/providers/openai-responses-server-schema.ts +290 -0
  248. package/src/providers/openai-responses-server.ts +1190 -0
  249. package/src/providers/openai-responses-shared.ts +800 -0
  250. package/src/providers/openai-responses.ts +738 -0
  251. package/src/providers/pi-native-client.ts +227 -0
  252. package/src/providers/pi-native-server.ts +210 -0
  253. package/src/providers/register-builtins.ts +411 -0
  254. package/src/providers/synthetic.ts +50 -0
  255. package/src/providers/transform-messages.ts +319 -0
  256. package/src/providers/vision-guard.ts +31 -0
  257. package/src/rate-limit-utils.ts +93 -0
  258. package/src/stream.ts +960 -0
  259. package/src/types.ts +967 -0
  260. package/src/usage/claude.ts +431 -0
  261. package/src/usage/gemini.ts +250 -0
  262. package/src/usage/github-copilot.ts +421 -0
  263. package/src/usage/google-antigravity.ts +201 -0
  264. package/src/usage/grok-cli.ts +163 -0
  265. package/src/usage/kimi.ts +271 -0
  266. package/src/usage/minimax-code.ts +31 -0
  267. package/src/usage/openai-codex.ts +503 -0
  268. package/src/usage/shared.ts +10 -0
  269. package/src/usage/zai.ts +247 -0
  270. package/src/usage.ts +183 -0
  271. package/src/utils/abort.ts +51 -0
  272. package/src/utils/anthropic-auth.ts +87 -0
  273. package/src/utils/discovery/antigravity.ts +261 -0
  274. package/src/utils/discovery/codex.ts +371 -0
  275. package/src/utils/discovery/cursor.ts +306 -0
  276. package/src/utils/discovery/gemini.ts +248 -0
  277. package/src/utils/discovery/index.ts +4 -0
  278. package/src/utils/discovery/openai-compatible.ts +230 -0
  279. package/src/utils/event-stream.ts +172 -0
  280. package/src/utils/fireworks-model-id.ts +30 -0
  281. package/src/utils/foundry.ts +8 -0
  282. package/src/utils/h2-fetch.ts +60 -0
  283. package/src/utils/http-inspector.ts +255 -0
  284. package/src/utils/idle-iterator.ts +257 -0
  285. package/src/utils/json-parse.ts +148 -0
  286. package/src/utils/oauth/alibaba-coding-plan.ts +59 -0
  287. package/src/utils/oauth/anthropic.ts +200 -0
  288. package/src/utils/oauth/api-key-login.ts +87 -0
  289. package/src/utils/oauth/api-key-validation.ts +92 -0
  290. package/src/utils/oauth/callback-server.ts +281 -0
  291. package/src/utils/oauth/cerebras.ts +16 -0
  292. package/src/utils/oauth/cloudflare-ai-gateway.ts +48 -0
  293. package/src/utils/oauth/cursor.ts +157 -0
  294. package/src/utils/oauth/deepseek.ts +53 -0
  295. package/src/utils/oauth/firepass.ts +24 -0
  296. package/src/utils/oauth/fireworks.ts +15 -0
  297. package/src/utils/oauth/github-copilot.ts +362 -0
  298. package/src/utils/oauth/gitlab-duo.ts +123 -0
  299. package/src/utils/oauth/google-antigravity.ts +200 -0
  300. package/src/utils/oauth/google-gemini-cli.ts +256 -0
  301. package/src/utils/oauth/google-oauth-shared.ts +110 -0
  302. package/src/utils/oauth/huggingface.ts +62 -0
  303. package/src/utils/oauth/index.ts +469 -0
  304. package/src/utils/oauth/kagi.ts +47 -0
  305. package/src/utils/oauth/kilo.ts +87 -0
  306. package/src/utils/oauth/kimi.ts +254 -0
  307. package/src/utils/oauth/litellm.ts +47 -0
  308. package/src/utils/oauth/lm-studio.ts +38 -0
  309. package/src/utils/oauth/minimax-code.ts +78 -0
  310. package/src/utils/oauth/moonshot.ts +16 -0
  311. package/src/utils/oauth/nanogpt.ts +15 -0
  312. package/src/utils/oauth/nvidia.ts +70 -0
  313. package/src/utils/oauth/oauth.html +199 -0
  314. package/src/utils/oauth/ollama-cloud.ts +28 -0
  315. package/src/utils/oauth/ollama.ts +47 -0
  316. package/src/utils/oauth/openai-codex.ts +299 -0
  317. package/src/utils/oauth/opencode.ts +49 -0
  318. package/src/utils/oauth/parallel.ts +46 -0
  319. package/src/utils/oauth/perplexity.ts +206 -0
  320. package/src/utils/oauth/pkce.ts +18 -0
  321. package/src/utils/oauth/qianfan.ts +58 -0
  322. package/src/utils/oauth/qwen-portal.ts +60 -0
  323. package/src/utils/oauth/synthetic.ts +16 -0
  324. package/src/utils/oauth/tavily.ts +46 -0
  325. package/src/utils/oauth/together.ts +16 -0
  326. package/src/utils/oauth/types.ts +99 -0
  327. package/src/utils/oauth/venice.ts +59 -0
  328. package/src/utils/oauth/vercel-ai-gateway.ts +47 -0
  329. package/src/utils/oauth/vllm.ts +40 -0
  330. package/src/utils/oauth/xai.ts +246 -0
  331. package/src/utils/oauth/xiaomi.ts +199 -0
  332. package/src/utils/oauth/zai.ts +60 -0
  333. package/src/utils/oauth/zenmux.ts +15 -0
  334. package/src/utils/overflow.ts +137 -0
  335. package/src/utils/parse-bind.ts +54 -0
  336. package/src/utils/provider-response.ts +30 -0
  337. package/src/utils/retry-after.ts +110 -0
  338. package/src/utils/retry-budget.ts +4 -0
  339. package/src/utils/retry.ts +54 -0
  340. package/src/utils/schema/CONSTRAINTS.md +164 -0
  341. package/src/utils/schema/adapt.ts +36 -0
  342. package/src/utils/schema/compatibility.ts +435 -0
  343. package/src/utils/schema/dereference.ts +98 -0
  344. package/src/utils/schema/draft.ts +341 -0
  345. package/src/utils/schema/equality.ts +97 -0
  346. package/src/utils/schema/fields.ts +190 -0
  347. package/src/utils/schema/index.ts +13 -0
  348. package/src/utils/schema/json-schema-validator.ts +577 -0
  349. package/src/utils/schema/meta-validator.ts +167 -0
  350. package/src/utils/schema/normalize.ts +1588 -0
  351. package/src/utils/schema/spill.ts +43 -0
  352. package/src/utils/schema/stamps.ts +97 -0
  353. package/src/utils/schema/types.ts +11 -0
  354. package/src/utils/schema/wire.ts +213 -0
  355. package/src/utils/schema/zod-decontaminate.ts +331 -0
  356. package/src/utils/sse-debug.ts +289 -0
  357. package/src/utils/tool-call-healing.ts +271 -0
  358. package/src/utils/tool-choice-capability.ts +220 -0
  359. package/src/utils/tool-choice.ts +99 -0
  360. package/src/utils/validation.ts +1019 -0
  361. package/src/utils.ts +178 -0
package/src/stream.ts ADDED
@@ -0,0 +1,960 @@
1
+ import * as fs from "node:fs";
2
+ import * as os from "node:os";
3
+ import * as path from "node:path";
4
+ import { $credentialEnv, $env, $pickCredentialEnv, extractHttpStatusFromError } from "@sayknow-cli/utils";
5
+ import { getCustomApi } from "./api-registry";
6
+ import type { Effort } from "./model-thinking";
7
+ import {
8
+ mapEffortToAnthropicAdaptiveEffort,
9
+ mapEffortToGoogleThinkingLevel,
10
+ requireSupportedEffort,
11
+ } from "./model-thinking";
12
+ import type { BedrockOptions } from "./providers/amazon-bedrock";
13
+ import type { AnthropicOptions } from "./providers/anthropic";
14
+ import type { CursorOptions } from "./providers/cursor";
15
+ import { isGitLabDuoModel, streamGitLabDuo } from "./providers/gitlab-duo";
16
+ import type { GoogleOptions } from "./providers/google";
17
+ import type { GoogleGeminiCliOptions } from "./providers/google-gemini-cli";
18
+ import type { GoogleVertexOptions } from "./providers/google-vertex";
19
+ import { isKimiModel, streamKimi } from "./providers/kimi";
20
+ import type { OllamaChatOptions } from "./providers/ollama";
21
+ import type { OpenAICompletionsOptions } from "./providers/openai-completions";
22
+ import { streamPiNative } from "./providers/pi-native-client";
23
+ // Heavy provider stream functions are imported lazily via register-builtins,
24
+ // which wraps each provider module in a dynamic import. This keeps the
25
+ // AWS SDK, google-auth-library, @google/genai, @bufbuild/protobuf, and
26
+ // other provider SDKs out of the CLI startup parse graph. The
27
+ // gitlab-duo / kimi / synthetic providers stay eager because their modules
28
+ // export routing predicates (isGitLabDuoModel, isKimiModel, isSyntheticModel)
29
+ // that must be callable synchronously before streaming begins, and their
30
+ // modules are thin wrappers with no heavy SDK dependencies.
31
+ import {
32
+ streamAnthropic,
33
+ streamAzureOpenAIResponses,
34
+ streamBedrock,
35
+ streamCursor,
36
+ streamGoogle,
37
+ streamGoogleGeminiCli,
38
+ streamGoogleVertex,
39
+ streamOllama,
40
+ streamOpenAICodexResponses,
41
+ streamOpenAICompletions,
42
+ streamOpenAIResponses,
43
+ } from "./providers/register-builtins";
44
+ import { isSyntheticModel, streamSynthetic } from "./providers/synthetic";
45
+ import type {
46
+ Api,
47
+ AssistantMessage,
48
+ AssistantMessageEvent,
49
+ Context,
50
+ Model,
51
+ OptionsForApi,
52
+ SimpleStreamOptions,
53
+ StreamOptions,
54
+ ThinkingBudgets,
55
+ ToolChoice,
56
+ } from "./types";
57
+ import { AssistantMessageEventStream } from "./utils/event-stream";
58
+ import { isFoundryEnabled } from "./utils/foundry";
59
+
60
+ let cachedVertexAdcCredentialsExists: boolean | null = null;
61
+
62
+ function hasVertexAdcCredentials(): boolean {
63
+ if (cachedVertexAdcCredentialsExists === null) {
64
+ const gacPath = $credentialEnv("GOOGLE_APPLICATION_CREDENTIALS");
65
+ if (gacPath) {
66
+ cachedVertexAdcCredentialsExists = fs.existsSync(gacPath);
67
+ } else {
68
+ cachedVertexAdcCredentialsExists = fs.existsSync(
69
+ path.join(os.homedir(), ".config", "gcloud", "application_default_credentials.json"),
70
+ );
71
+ }
72
+ }
73
+ return cachedVertexAdcCredentialsExists;
74
+ }
75
+
76
+ type KeyResolver = string | (() => string | undefined);
77
+
78
+ const serviceProviderMap: Record<string, KeyResolver> = {
79
+ "alibaba-coding-plan": "ALIBABA_CODING_PLAN_API_KEY",
80
+ openai: () => $credentialEnv("OPENAI_API_KEY"),
81
+ google: "GEMINI_API_KEY",
82
+ groq: "GROQ_API_KEY",
83
+ cerebras: "CEREBRAS_API_KEY",
84
+ xai: "XAI_API_KEY",
85
+ fireworks: "FIREWORKS_API_KEY",
86
+ firepass: "FIREPASS_API_KEY",
87
+ openrouter: "OPENROUTER_API_KEY",
88
+ kilo: "KILO_API_KEY",
89
+ "vercel-ai-gateway": "AI_GATEWAY_API_KEY",
90
+ zai: "ZAI_API_KEY",
91
+ mistral: "MISTRAL_API_KEY",
92
+ minimax: "MINIMAX_API_KEY",
93
+ "minimax-code": "MINIMAX_CODE_API_KEY",
94
+ "minimax-code-cn": "MINIMAX_CODE_CN_API_KEY",
95
+ "opencode-go": "OPENCODE_API_KEY",
96
+ "opencode-zen": "OPENCODE_API_KEY",
97
+ cursor: "CURSOR_ACCESS_TOKEN",
98
+ deepseek: "DEEPSEEK_API_KEY",
99
+ "openai-codex": "OPENAI_CODEX_OAUTH_TOKEN",
100
+ "azure-openai": "AZURE_OPENAI_API_KEY",
101
+ "azure-openai-responses": "AZURE_OPENAI_API_KEY",
102
+ exa: "EXA_API_KEY",
103
+ jina: "JINA_API_KEY",
104
+ brave: "BRAVE_API_KEY",
105
+ perplexity: "PERPLEXITY_API_KEY",
106
+ tavily: "TAVILY_API_KEY",
107
+ parallel: "PARALLEL_API_KEY",
108
+ kagi: "KAGI_API_KEY",
109
+ // GitHub Copilot uses GitHub personal access token
110
+ "github-copilot": () => $pickCredentialEnv("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN"),
111
+ // Foundry mode optionally switches Anthropic auth to enterprise gateway credentials.
112
+ anthropic: () =>
113
+ isFoundryEnabled()
114
+ ? $pickCredentialEnv("ANTHROPIC_FOUNDRY_API_KEY", "ANTHROPIC_OAUTH_TOKEN", "ANTHROPIC_API_KEY")
115
+ : $pickCredentialEnv("ANTHROPIC_OAUTH_TOKEN", "ANTHROPIC_API_KEY"),
116
+ "gitlab-duo": "GITLAB_TOKEN",
117
+ // Vertex AI supports either GOOGLE_CLOUD_API_KEY or Application Default Credentials.
118
+ "google-vertex": () => {
119
+ const googleCloudApiKey = $credentialEnv("GOOGLE_CLOUD_API_KEY");
120
+ if (googleCloudApiKey) return googleCloudApiKey;
121
+
122
+ const hasCredentials = hasVertexAdcCredentials();
123
+ const hasProject = !!($env.GOOGLE_CLOUD_PROJECT || $env.GCLOUD_PROJECT);
124
+ const hasLocation = !!$env.GOOGLE_CLOUD_LOCATION;
125
+ if (hasCredentials && hasProject && hasLocation) {
126
+ return "<authenticated>";
127
+ }
128
+ },
129
+ // Amazon Bedrock supports multiple credential sources:
130
+ // 1. AWS_PROFILE - named profile from ~/.aws/credentials
131
+ // 2. AWS_ACCESS_KEY_ID + AWS_SECRET_ACCESS_KEY - standard IAM keys
132
+ // 3. AWS_BEARER_TOKEN_BEDROCK - Bedrock API keys (bearer token)
133
+ // 4. AWS_CONTAINER_CREDENTIALS_* - ECS/Task IAM role credentials
134
+ // 5. AWS_WEB_IDENTITY_TOKEN_FILE + AWS_ROLE_ARN - IRSA (EKS) web identity
135
+ "amazon-bedrock": () => {
136
+ const awsProfile = $credentialEnv("AWS_PROFILE");
137
+ const awsAccessKeyId = $credentialEnv("AWS_ACCESS_KEY_ID");
138
+ const awsSecretAccessKey = $credentialEnv("AWS_SECRET_ACCESS_KEY");
139
+ const awsBearerToken = $credentialEnv("AWS_BEARER_TOKEN_BEDROCK");
140
+ const hasEcsCredentials =
141
+ !!$credentialEnv("AWS_CONTAINER_CREDENTIALS_RELATIVE_URI") ||
142
+ !!$credentialEnv("AWS_CONTAINER_CREDENTIALS_FULL_URI");
143
+ const hasWebIdentity = !!$credentialEnv("AWS_WEB_IDENTITY_TOKEN_FILE") && !!$credentialEnv("AWS_ROLE_ARN");
144
+ if (
145
+ awsProfile ||
146
+ (awsAccessKeyId && awsSecretAccessKey) ||
147
+ awsBearerToken ||
148
+ hasEcsCredentials ||
149
+ hasWebIdentity
150
+ ) {
151
+ return "<authenticated>";
152
+ }
153
+ },
154
+ synthetic: "SYNTHETIC_API_KEY",
155
+ "cloudflare-ai-gateway": "CLOUDFLARE_AI_GATEWAY_API_KEY",
156
+ huggingface: () => $pickCredentialEnv("HUGGINGFACE_HUB_TOKEN", "HF_TOKEN"),
157
+ litellm: "LITELLM_API_KEY",
158
+ moonshot: "MOONSHOT_API_KEY",
159
+ nvidia: "NVIDIA_API_KEY",
160
+ nanogpt: "NANO_GPT_API_KEY",
161
+ "lm-studio": "LM_STUDIO_API_KEY",
162
+ ollama: "OLLAMA_API_KEY",
163
+ "ollama-cloud": "OLLAMA_CLOUD_API_KEY",
164
+ "llama.cpp": "LLAMA_CPP_API_KEY",
165
+ qianfan: "QIANFAN_API_KEY",
166
+ "qwen-portal": () => $pickCredentialEnv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
167
+ together: "TOGETHER_API_KEY",
168
+ zenmux: "ZENMUX_API_KEY",
169
+ venice: "VENICE_API_KEY",
170
+ vllm: "VLLM_API_KEY",
171
+ xiaomi: "XIAOMI_API_KEY",
172
+ };
173
+
174
+ /**
175
+ * Get API key for provider from known environment variables, e.g. OPENAI_API_KEY.
176
+ *
177
+ * Provider authentication intentionally excludes cwd/.env values. Project dotenv files are
178
+ * loaded into $env for app/tool execution, but must not silently fund SKC model requests.
179
+ */
180
+ export function getEnvApiKey(provider: string): string | undefined {
181
+ const resolver = serviceProviderMap[provider];
182
+ if (typeof resolver === "string") {
183
+ return $credentialEnv(resolver);
184
+ }
185
+ return resolver?.();
186
+ }
187
+
188
+ /**
189
+ * Enumerate every provider that has an env-var fallback for `getEnvApiKey`.
190
+ * Used by `skc auth-broker migrate --include-env` to discover env-sourced keys
191
+ * that should be uploaded to the broker.
192
+ */
193
+ export function listProvidersWithEnvKey(): string[] {
194
+ return Object.keys(serviceProviderMap);
195
+ }
196
+
197
+ /**
198
+ * Subscription-style providers whose "subscription" is delivered as an API key
199
+ * (created at https://opencode.ai/auth), not a separate OAuth/session token.
200
+ * Used to give OpenCode users an accurate headless auth diagnostic (#755).
201
+ */
202
+ const OPENCODE_SUBSCRIPTION_PROVIDERS = new Set(["opencode-go", "opencode-zen"]);
203
+
204
+ /**
205
+ * Provider-specific credential guidance appended to "no credential" errors.
206
+ *
207
+ * Headless SKC has no interactive `/login` TUI, so a bare "No API key" /
208
+ * "No credentials" error left users — OpenCode Go subscribers especially
209
+ * (#755) — unsure what signal SKC actually reads. OpenCode subscriptions are
210
+ * themselves API keys, so this names the env var SKC reads for the provider,
211
+ * warns that a project `.env` is intentionally ignored for provider
212
+ * credentials, and points OpenCode users at one-time interactive CLI credential capture.
213
+ *
214
+ * Returns an empty string when the provider has no env-var key and no special
215
+ * handling, so callers can append it unconditionally.
216
+ */
217
+ export function formatProviderCredentialHint(provider: string): string {
218
+ const resolver = serviceProviderMap[provider];
219
+ const envVar = typeof resolver === "string" ? resolver : undefined;
220
+ const isOpenCodeSubscription = OPENCODE_SUBSCRIPTION_PROVIDERS.has(provider);
221
+ const parts: string[] = [];
222
+ if (isOpenCodeSubscription) {
223
+ parts.push(
224
+ "OpenCode subscriptions authenticate with an API key (created at https://opencode.ai/auth), not a separate session/OAuth token.",
225
+ );
226
+ }
227
+ if (envVar) {
228
+ parts.push(
229
+ `Headless SKC reads this provider's key from ${envVar} (exported in your shell or set in ~/.skc/.env).`,
230
+ );
231
+ parts.push("A value set only in a project .env is intentionally ignored for provider credentials.");
232
+ }
233
+ if (isOpenCodeSubscription) {
234
+ parts.push(
235
+ `Or run \`skc auth-broker login ${provider}\` once before headless/print mode to store the key interactively.`,
236
+ );
237
+ }
238
+ return parts.join(" ");
239
+ }
240
+
241
+ /**
242
+ * Build an actionable "missing API key" error for a provider, used by the
243
+ * low-level `stream`/`complete` entry points (#755).
244
+ */
245
+ export function formatMissingApiKeyError(provider: string): string {
246
+ const base = `No API key for provider: ${provider}.`;
247
+ const hint = formatProviderCredentialHint(provider);
248
+ return hint ? `${base} ${hint}` : base;
249
+ }
250
+
251
+ export function stream<TApi extends Api>(
252
+ model: Model<TApi>,
253
+ context: Context,
254
+ options?: OptionsForApi<TApi>,
255
+ ): AssistantMessageEventStream {
256
+ // Check custom API registry first (extension-provided APIs like "vertex-Anthropic model-api")
257
+ const customApiProvider = getCustomApi(model.api);
258
+ if (customApiProvider) {
259
+ return customApiProvider.stream(model, context, options as StreamOptions);
260
+ }
261
+
262
+ if (isGitLabDuoModel(model)) {
263
+ const apiKey = (options as StreamOptions | undefined)?.apiKey || getEnvApiKey(model.provider);
264
+ if (!apiKey) {
265
+ throw new Error(formatMissingApiKeyError(model.provider));
266
+ }
267
+ return streamGitLabDuo(model, context, {
268
+ ...(options as SimpleStreamOptions | undefined),
269
+ apiKey,
270
+ });
271
+ }
272
+
273
+ // Vertex AI uses Application Default Credentials, not API keys
274
+ if (model.api === "google-vertex") {
275
+ return streamGoogleVertex(model as Model<"google-vertex">, context, options as GoogleVertexOptions);
276
+ } else if (model.api === "bedrock-converse-stream") {
277
+ // Bedrock doesn't have any API keys instead it sources credentials from standard AWS env variables or from given AWS profile.
278
+ return streamBedrock(model as Model<"bedrock-converse-stream">, context, (options || {}) as BedrockOptions);
279
+ }
280
+
281
+ const apiKey = options?.apiKey || getEnvApiKey(model.provider);
282
+ if (!apiKey) {
283
+ throw new Error(formatMissingApiKeyError(model.provider));
284
+ }
285
+ const providerOptions = { ...options, apiKey };
286
+
287
+ const api: Api = model.api;
288
+ switch (api) {
289
+ case "anthropic-messages": {
290
+ const anthropicOptions = providerOptions as AnthropicOptions;
291
+ return streamAnthropic(model as Model<"anthropic-messages">, context, {
292
+ ...anthropicOptions,
293
+ isOAuth: anthropicOptions.isOAuth ?? model.isOAuth,
294
+ });
295
+ }
296
+
297
+ case "openai-completions":
298
+ return streamOpenAICompletions(model as Model<"openai-completions">, context, providerOptions as any);
299
+
300
+ case "openai-responses":
301
+ return streamOpenAIResponses(model as Model<"openai-responses">, context, providerOptions as any);
302
+
303
+ case "azure-openai-responses":
304
+ return streamAzureOpenAIResponses(model as Model<"azure-openai-responses">, context, providerOptions as any);
305
+
306
+ case "openai-codex-responses":
307
+ return streamOpenAICodexResponses(model as Model<"openai-codex-responses">, context, providerOptions as any);
308
+
309
+ case "google-generative-ai":
310
+ return streamGoogle(model as Model<"google-generative-ai">, context, providerOptions);
311
+
312
+ case "google-gemini-cli":
313
+ return streamGoogleGeminiCli(
314
+ model as Model<"google-gemini-cli">,
315
+ context,
316
+ providerOptions as GoogleGeminiCliOptions,
317
+ );
318
+
319
+ case "ollama-chat":
320
+ return streamOllama(model as Model<"ollama-chat">, context, providerOptions as OllamaChatOptions);
321
+
322
+ case "cursor-agent":
323
+ return streamCursor(model as Model<"cursor-agent">, context, providerOptions as CursorOptions);
324
+
325
+ default:
326
+ throw new Error(`Unhandled API: ${api}`);
327
+ }
328
+ }
329
+
330
+ export async function complete<TApi extends Api>(
331
+ model: Model<TApi>,
332
+ context: Context,
333
+ options?: OptionsForApi<TApi>,
334
+ ): Promise<AssistantMessage> {
335
+ const s = stream(model, context, options);
336
+ return s.result();
337
+ }
338
+
339
+ type AuthRetryFailure = {
340
+ error: unknown;
341
+ bufferedEvents: AssistantMessageEvent[];
342
+ terminalEvent?: Extract<AssistantMessageEvent, { type: "error" }>;
343
+ };
344
+
345
+ function extractStatusFromAssistantError(message: AssistantMessage): number | undefined {
346
+ if (message.errorStatus !== undefined) return message.errorStatus;
347
+ if (!message.errorMessage) return undefined;
348
+ return extractHttpStatusFromError({ message: message.errorMessage });
349
+ }
350
+
351
+ function createAssistantAuthError(message: AssistantMessage): Error & { status?: number } {
352
+ const error: Error & { status?: number } = new Error(message.errorMessage ?? "Provider authentication failed");
353
+ const status = extractStatusFromAssistantError(message);
354
+ if (status !== undefined) error.status = status;
355
+ return error;
356
+ }
357
+
358
+ function emitBufferedEvents(stream: AssistantMessageEventStream, events: AssistantMessageEvent[]): void {
359
+ for (const event of events) {
360
+ stream.push(event);
361
+ }
362
+ }
363
+
364
+ export function streamSimple<TApi extends Api>(
365
+ model: Model<TApi>,
366
+ context: Context,
367
+ options?: SimpleStreamOptions,
368
+ ): AssistantMessageEventStream {
369
+ const retryApiKey = options?.onAuthError ? (options.apiKey ?? getEnvApiKey(model.provider)) : undefined;
370
+ if (retryApiKey) {
371
+ const outer = new AssistantMessageEventStream();
372
+ const onAuthError = options!.onAuthError!;
373
+ const runAttempt = async (apiKey: string, captureAuthFailure: boolean): Promise<AuthRetryFailure | undefined> => {
374
+ const bufferedEvents: AssistantMessageEvent[] = [];
375
+ let emittedReplayUnsafeEvent = false;
376
+ const flushBuffered = (): void => {
377
+ emitBufferedEvents(outer, bufferedEvents);
378
+ bufferedEvents.length = 0;
379
+ };
380
+
381
+ try {
382
+ const inner = streamSimple(model, context, { ...options, apiKey, onAuthError: undefined });
383
+ for await (const event of inner) {
384
+ if (!emittedReplayUnsafeEvent && event.type === "start") {
385
+ bufferedEvents.push(event);
386
+ continue;
387
+ }
388
+ if (
389
+ !emittedReplayUnsafeEvent &&
390
+ captureAuthFailure &&
391
+ event.type === "error" &&
392
+ extractStatusFromAssistantError(event.error) === 401
393
+ ) {
394
+ return { error: createAssistantAuthError(event.error), bufferedEvents, terminalEvent: event };
395
+ }
396
+ flushBuffered();
397
+ emittedReplayUnsafeEvent = true;
398
+ outer.push(event);
399
+ if (outer.done) return undefined;
400
+ }
401
+ flushBuffered();
402
+ if (!outer.done) outer.end(await inner.result());
403
+ } catch (error) {
404
+ if (!emittedReplayUnsafeEvent && captureAuthFailure && extractHttpStatusFromError(error) === 401) {
405
+ return { error, bufferedEvents };
406
+ }
407
+ flushBuffered();
408
+ outer.fail(error);
409
+ }
410
+ return undefined;
411
+ };
412
+ const emitFailure = (failure: AuthRetryFailure): void => {
413
+ emitBufferedEvents(outer, failure.bufferedEvents);
414
+ if (failure.terminalEvent) {
415
+ outer.push(failure.terminalEvent);
416
+ } else {
417
+ outer.fail(failure.error);
418
+ }
419
+ };
420
+
421
+ void (async () => {
422
+ const failure = await runAttempt(retryApiKey, true);
423
+ if (!failure) return;
424
+ let nextKey: string | undefined;
425
+ try {
426
+ nextKey = await onAuthError(model.provider, retryApiKey, failure.error);
427
+ } catch {
428
+ nextKey = undefined;
429
+ }
430
+ if (!nextKey || nextKey === retryApiKey) {
431
+ emitFailure(failure);
432
+ return;
433
+ }
434
+ await runAttempt(nextKey, false);
435
+ })();
436
+ return outer;
437
+ }
438
+
439
+ // Pi-native transport short-circuits the per-provider dispatch entirely:
440
+ // the gateway resolves provider + credential server-side, so we don't
441
+ // need an `apiKey` from `getEnvApiKey` here — `options.apiKey` carries
442
+ // the gateway bearer instead. Comes BEFORE the custom-API check so
443
+ // extension-registered APIs can't accidentally override a configured
444
+ // pi-native transport.
445
+ if (model.transport === "pi-native") {
446
+ return streamPiNative(model, context, options);
447
+ }
448
+
449
+ // Check custom API registry (extension-provided APIs)
450
+ const customApiProvider = getCustomApi(model.api);
451
+ if (customApiProvider) {
452
+ return customApiProvider.streamSimple(model, context, options);
453
+ }
454
+
455
+ // Vertex AI uses Application Default Credentials, not API keys
456
+ if (model.api === "google-vertex") {
457
+ const providerOptions = mapOptionsForApi(model, options, undefined);
458
+ return stream(model, context, providerOptions);
459
+ } else if (model.api === "bedrock-converse-stream") {
460
+ // Bedrock doesn't have any API keys instead it sources credentials from standard AWS env variables or from given AWS profile.
461
+ const providerOptions = mapOptionsForApi(model, options, undefined);
462
+ return stream(model, context, providerOptions);
463
+ }
464
+
465
+ const apiKey = options?.apiKey || getEnvApiKey(model.provider);
466
+ if (!apiKey) {
467
+ throw new Error(formatMissingApiKeyError(model.provider));
468
+ }
469
+
470
+ // GitLab Duo - wraps Anthropic/OpenAI behind GitLab AI Gateway direct access tokens
471
+ if (isGitLabDuoModel(model)) {
472
+ return streamGitLabDuo(model, context, {
473
+ ...options,
474
+ apiKey,
475
+ });
476
+ }
477
+
478
+ // Kimi Code - route to dedicated handler that wraps OpenAI or Anthropic API
479
+ if (isKimiModel(model)) {
480
+ // Pass raw SimpleStreamOptions - streamKimi handles mapping internally
481
+ return streamKimi(model as Model<"openai-completions">, context, {
482
+ ...options,
483
+ apiKey,
484
+ format: options?.kimiApiFormat ?? "anthropic",
485
+ });
486
+ }
487
+
488
+ // Synthetic - route to dedicated handler that wraps OpenAI or Anthropic API
489
+ if (isSyntheticModel(model)) {
490
+ // Pass raw SimpleStreamOptions - streamSynthetic handles mapping internally
491
+ return streamSynthetic(model as Model<"openai-completions">, context, {
492
+ ...options,
493
+ apiKey,
494
+ format: options?.syntheticApiFormat ?? "openai", // Default to OpenAI format
495
+ });
496
+ }
497
+
498
+ const providerOptions = mapOptionsForApi(model, options, apiKey);
499
+ return stream(model, context, providerOptions);
500
+ }
501
+
502
+ export async function completeSimple<TApi extends Api>(
503
+ model: Model<TApi>,
504
+ context: Context,
505
+ options?: SimpleStreamOptions,
506
+ ): Promise<AssistantMessage> {
507
+ const s = streamSimple(model, context, options);
508
+ return s.result();
509
+ }
510
+
511
+ const MIN_OUTPUT_TOKENS = 1024;
512
+ export const OUTPUT_FALLBACK_BUFFER = 4000;
513
+ const ANTHROPIC_USE_INTERLEAVED_THINKING = Bun.env.PI_NO_INTERLEAVED_THINKING !== "1";
514
+
515
+ export const ANTHROPIC_THINKING: Record<Effort, number> = {
516
+ minimal: 1024,
517
+ low: 4096,
518
+ medium: 8192,
519
+ high: 16384,
520
+ xhigh: 32768,
521
+ max: 65536,
522
+ };
523
+
524
+ const GOOGLE_THINKING: Record<Effort, number> = {
525
+ minimal: 1024,
526
+ low: 4096,
527
+ medium: 8192,
528
+ high: 16384,
529
+ xhigh: 24575,
530
+ max: 24575,
531
+ };
532
+
533
+ const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
534
+ minimal: 1024,
535
+ low: 2048,
536
+ medium: 8192,
537
+ high: 16384,
538
+ xhigh: 16384,
539
+ max: 32768,
540
+ };
541
+
542
+ function resolveBedrockThinkingBudget(
543
+ model: Model<"bedrock-converse-stream">,
544
+ options?: SimpleStreamOptions,
545
+ ): { budget: number; level: Effort } | null {
546
+ if (!options?.reasoning || !model.reasoning) return null;
547
+ const level = requireSupportedEffort(model, options.reasoning);
548
+ const budget = options.thinkingBudgets?.[level] ?? BEDROCK_CLAUDE_THINKING[level];
549
+ return { budget, level };
550
+ }
551
+
552
+ export function mapAnthropicToolChoice(choice?: ToolChoice): AnthropicOptions["toolChoice"] {
553
+ if (!choice) return undefined;
554
+ if (typeof choice === "string") {
555
+ if (choice === "required") return "any";
556
+ if (choice === "auto" || choice === "none" || choice === "any") return choice;
557
+ return undefined;
558
+ }
559
+ if (choice.type === "tool") {
560
+ return choice.name ? { type: "tool", name: choice.name } : undefined;
561
+ }
562
+ if (choice.type === "function") {
563
+ const name = "function" in choice ? choice.function?.name : choice.name;
564
+ return name ? { type: "tool", name } : undefined;
565
+ }
566
+ return undefined;
567
+ }
568
+
569
+ function mapGoogleToolChoice(
570
+ choice?: ToolChoice,
571
+ ): GoogleOptions["toolChoice"] | GoogleGeminiCliOptions["toolChoice"] | GoogleVertexOptions["toolChoice"] {
572
+ if (!choice) return undefined;
573
+ if (typeof choice === "string") {
574
+ if (choice === "required") return "any";
575
+ if (choice === "auto" || choice === "none" || choice === "any") return choice;
576
+ return undefined;
577
+ }
578
+ return "any";
579
+ }
580
+
581
+ function mapOpenAiToolChoice(choice?: ToolChoice): OpenAICompletionsOptions["toolChoice"] {
582
+ if (!choice) return undefined;
583
+ if (typeof choice === "string") {
584
+ if (choice === "any") return "required";
585
+ if (choice === "auto" || choice === "none" || choice === "required") return choice;
586
+ return undefined;
587
+ }
588
+ if (choice.type === "tool") {
589
+ return choice.name ? { type: "function", function: { name: choice.name } } : undefined;
590
+ }
591
+ if (choice.type === "function") {
592
+ const name = "function" in choice ? choice.function?.name : choice.name;
593
+ return name ? { type: "function", function: { name } } : undefined;
594
+ }
595
+ return undefined;
596
+ }
597
+
598
+ function resolveOpenAiReasoningEffort<TApi extends Api>(
599
+ model: Model<TApi>,
600
+ options?: SimpleStreamOptions,
601
+ ): Effort | undefined {
602
+ const reasoning = options?.reasoning;
603
+ if (!reasoning || !model.reasoning) return undefined;
604
+ return requireSupportedEffort(model, reasoning);
605
+ }
606
+
607
+ const castApi = <TApi extends Api>(api: OptionsForApi<TApi>): OptionsForApi<Api> => api as OptionsForApi<Api>;
608
+
609
+ function mapOptionsForApi<TApi extends Api>(
610
+ model: Model<TApi>,
611
+ options?: SimpleStreamOptions,
612
+ apiKey?: string,
613
+ ): OptionsForApi<TApi> {
614
+ const base = {
615
+ temperature: options?.temperature,
616
+ topP: options?.topP,
617
+ topK: options?.topK,
618
+ minP: options?.minP,
619
+ presencePenalty: options?.presencePenalty,
620
+ repetitionPenalty: options?.repetitionPenalty,
621
+ maxTokens: options?.maxTokens || Math.min(model.maxTokens, 32000),
622
+ signal: options?.signal,
623
+ apiKey: apiKey || options?.apiKey,
624
+ cacheRetention: options?.cacheRetention ?? model.cacheRetention,
625
+ headers: options?.headers,
626
+ initiatorOverride: options?.initiatorOverride,
627
+ maxRetryDelayMs: options?.maxRetryDelayMs,
628
+ requestMaxRetries: options?.requestMaxRetries,
629
+ streamMaxRetries: options?.streamMaxRetries,
630
+ metadata: options?.metadata,
631
+ sessionId: options?.sessionId,
632
+ providerSessionState: options?.providerSessionState,
633
+ onPayload: options?.onPayload,
634
+ onResponse: options?.onResponse,
635
+ onSseEvent: options?.onSseEvent,
636
+ execHandlers: options?.execHandlers,
637
+ };
638
+
639
+ switch (model.api) {
640
+ case "anthropic-messages": {
641
+ // Explicitly disable thinking when reasoning is not specified or model doesn't support it
642
+ const reasoning = options?.reasoning;
643
+ if (!reasoning || !model.reasoning) {
644
+ return castApi<"anthropic-messages">({
645
+ ...base,
646
+ thinkingEnabled: false,
647
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
648
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
649
+ serviceTier: options?.serviceTier,
650
+ });
651
+ }
652
+
653
+ let thinkingBudget = options.thinkingBudgets?.[reasoning] ?? ANTHROPIC_THINKING[reasoning];
654
+ if (thinkingBudget <= 0) {
655
+ return castApi<"anthropic-messages">({
656
+ ...base,
657
+ thinkingEnabled: false,
658
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
659
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
660
+ serviceTier: options?.serviceTier,
661
+ });
662
+ }
663
+
664
+ // For Opus 4.6+ and Sonnet 4.6+: use adaptive thinking with effort level
665
+ // For older models: use budget-based thinking
666
+ if (model.thinking?.mode === "anthropic-adaptive") {
667
+ const effort = mapEffortToAnthropicAdaptiveEffort(model, reasoning);
668
+ return castApi<"anthropic-messages">({
669
+ ...base,
670
+ thinkingEnabled: true,
671
+ effort,
672
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
673
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
674
+ serviceTier: options?.serviceTier,
675
+ });
676
+ }
677
+
678
+ if (ANTHROPIC_USE_INTERLEAVED_THINKING) {
679
+ return castApi<"anthropic-messages">({
680
+ ...base,
681
+ thinkingEnabled: true,
682
+ thinkingBudgetTokens: thinkingBudget,
683
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
684
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
685
+ serviceTier: options?.serviceTier,
686
+ });
687
+ }
688
+
689
+ // Caller's maxTokens is the desired output; add thinking budget on top, capped at model limit
690
+ const maxTokens = Math.min((base.maxTokens || 0) + thinkingBudget, model.maxTokens);
691
+
692
+ // If not enough room for thinking + output, reduce thinking budget
693
+ if (maxTokens <= thinkingBudget) {
694
+ thinkingBudget = maxTokens - MIN_OUTPUT_TOKENS;
695
+ }
696
+
697
+ // If thinking budget is too low, disable thinking
698
+ if (thinkingBudget <= 0) {
699
+ return castApi<"anthropic-messages">({
700
+ ...base,
701
+ thinkingEnabled: false,
702
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
703
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
704
+ serviceTier: options?.serviceTier,
705
+ });
706
+ } else {
707
+ return castApi<"anthropic-messages">({
708
+ ...base,
709
+ maxTokens,
710
+ thinkingEnabled: true,
711
+ thinkingBudgetTokens: thinkingBudget,
712
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
713
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
714
+ serviceTier: options?.serviceTier,
715
+ });
716
+ }
717
+ }
718
+
719
+ case "bedrock-converse-stream": {
720
+ const bedrockBase: BedrockOptions = {
721
+ ...base,
722
+ reasoning: options?.reasoning,
723
+ thinkingBudgets: options?.thinkingBudgets,
724
+ toolChoice: mapAnthropicToolChoice(options?.toolChoice),
725
+ thinkingDisplay: options?.hideThinkingSummary ? "omitted" : undefined,
726
+ };
727
+ // Adaptive mode sends effort directly, no budget_tokens — skip budget inflation.
728
+ if (model.thinking?.mode === "anthropic-adaptive") {
729
+ return castApi<"bedrock-converse-stream">(bedrockBase);
730
+ }
731
+ const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
732
+ if (!budgetInfo) return bedrockBase as OptionsForApi<TApi>;
733
+ let maxTokens = bedrockBase.maxTokens ?? model.maxTokens;
734
+ let thinkingBudgets = bedrockBase.thinkingBudgets;
735
+ if (maxTokens <= budgetInfo.budget) {
736
+ const desiredMaxTokens = Math.min(model.maxTokens, budgetInfo.budget + MIN_OUTPUT_TOKENS);
737
+ if (desiredMaxTokens > maxTokens) {
738
+ maxTokens = desiredMaxTokens;
739
+ }
740
+ }
741
+ if (maxTokens <= budgetInfo.budget) {
742
+ const adjustedBudget = Math.max(0, maxTokens - MIN_OUTPUT_TOKENS);
743
+ thinkingBudgets = { ...(thinkingBudgets ?? {}), [budgetInfo.level]: adjustedBudget };
744
+ }
745
+ return castApi<"bedrock-converse-stream">({ ...bedrockBase, maxTokens, thinkingBudgets });
746
+ }
747
+
748
+ case "openai-completions":
749
+ return castApi<"openai-completions">({
750
+ ...base,
751
+ reasoning: resolveOpenAiReasoningEffort(model, options),
752
+ disableReasoning: options?.disableReasoning,
753
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
754
+ serviceTier: options?.serviceTier,
755
+ });
756
+
757
+ case "openai-responses":
758
+ return castApi<"openai-responses">({
759
+ ...base,
760
+ reasoning: resolveOpenAiReasoningEffort(model, options),
761
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
762
+ serviceTier: options?.serviceTier,
763
+ reasoningSummary: options?.hideThinkingSummary ? null : undefined,
764
+ });
765
+
766
+ case "azure-openai-responses":
767
+ return castApi<"azure-openai-responses">({
768
+ ...base,
769
+ reasoning: resolveOpenAiReasoningEffort(model, options),
770
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
771
+ serviceTier: options?.serviceTier,
772
+ reasoningSummary: options?.hideThinkingSummary ? null : undefined,
773
+ });
774
+
775
+ case "openai-codex-responses":
776
+ return castApi<"openai-codex-responses">({
777
+ ...base,
778
+ reasoning: resolveOpenAiReasoningEffort(model, options),
779
+ toolChoice: mapOpenAiToolChoice(options?.toolChoice),
780
+ serviceTier: options?.serviceTier,
781
+ preferWebsockets: options?.preferWebsockets,
782
+ reasoningSummary: options?.hideThinkingSummary ? null : undefined,
783
+ });
784
+
785
+ case "google-generative-ai": {
786
+ // Explicitly disable thinking when reasoning is not specified or model doesn't support it
787
+ // This is needed because Gemini has "dynamic thinking" enabled by default
788
+ const reasoning = options?.reasoning;
789
+ if (!reasoning || !model.reasoning) {
790
+ return castApi<"google-generative-ai">({
791
+ ...base,
792
+ thinking: { enabled: false },
793
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
794
+ });
795
+ }
796
+
797
+ const googleModel = model as Model<"google-generative-ai">;
798
+ const effort = requireSupportedEffort(googleModel, reasoning);
799
+
800
+ // Gemini 3+ models use thinkingLevel exclusively instead of thinkingBudget.
801
+ // https://ai.google.dev/gemini-api/docs/thinking#set-budget
802
+ if (googleModel.thinking?.mode === "google-level") {
803
+ return castApi<"google-generative-ai">({
804
+ ...base,
805
+ thinking: {
806
+ enabled: true,
807
+ level: mapEffortToGoogleThinkingLevel(googleModel, effort),
808
+ },
809
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
810
+ });
811
+ }
812
+
813
+ return castApi<"google-gemini-cli">({
814
+ ...base,
815
+ thinking: {
816
+ enabled: true,
817
+ budgetTokens: getGoogleBudget(googleModel, effort, options?.thinkingBudgets),
818
+ },
819
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
820
+ });
821
+ }
822
+
823
+ case "google-gemini-cli": {
824
+ const reasoning = options?.reasoning;
825
+ if (!reasoning || !model.reasoning) {
826
+ return castApi<"google-gemini-cli">({
827
+ ...base,
828
+ thinking: { enabled: false },
829
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
830
+ });
831
+ }
832
+
833
+ const effort = requireSupportedEffort(model, reasoning);
834
+
835
+ // Gemini 3+ models use thinkingLevel instead of thinkingBudget
836
+ if (model.thinking?.mode === "google-level") {
837
+ return castApi<"google-gemini-cli">({
838
+ ...base,
839
+ thinking: {
840
+ enabled: true,
841
+ level: mapEffortToGoogleThinkingLevel(model, effort),
842
+ },
843
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
844
+ });
845
+ }
846
+
847
+ let thinkingBudget = options.thinkingBudgets?.[effort] ?? GOOGLE_THINKING[effort];
848
+
849
+ // Caller's maxTokens is the desired output; add thinking budget on top, capped at model limit
850
+ const maxTokens = Math.min((base.maxTokens || 0) + thinkingBudget, model.maxTokens);
851
+
852
+ // If not enough room for thinking + output, reduce thinking budget
853
+ if (maxTokens <= thinkingBudget) {
854
+ thinkingBudget = Math.max(0, maxTokens - MIN_OUTPUT_TOKENS) ?? 0;
855
+ }
856
+
857
+ // If thinking budget is too low, disable thinking
858
+ if (thinkingBudget <= 0) {
859
+ return castApi<"google-gemini-cli">({
860
+ ...base,
861
+ thinking: { enabled: false },
862
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
863
+ });
864
+ } else {
865
+ return castApi<"google-gemini-cli">({
866
+ ...base,
867
+ maxTokens,
868
+ thinking: { enabled: true, budgetTokens: thinkingBudget },
869
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
870
+ });
871
+ }
872
+ }
873
+
874
+ case "google-vertex": {
875
+ // Explicitly disable thinking when reasoning is not specified or model doesn't support it
876
+ const reasoning = options?.reasoning;
877
+ if (!reasoning || !model.reasoning) {
878
+ return castApi<"google-vertex">({
879
+ ...base,
880
+ thinking: { enabled: false },
881
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
882
+ });
883
+ }
884
+
885
+ const vertexModel = model as Model<"google-vertex">;
886
+ const effort = requireSupportedEffort(vertexModel, reasoning);
887
+ const geminiModel = vertexModel as unknown as Model<"google-generative-ai">;
888
+
889
+ if (geminiModel.thinking?.mode === "google-level") {
890
+ return castApi<"google-vertex">({
891
+ ...base,
892
+ thinking: {
893
+ enabled: true,
894
+ level: mapEffortToGoogleThinkingLevel(geminiModel, effort),
895
+ },
896
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
897
+ });
898
+ }
899
+
900
+ return castApi<"google-vertex">({
901
+ ...base,
902
+ thinking: {
903
+ enabled: true,
904
+ budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets),
905
+ },
906
+ toolChoice: mapGoogleToolChoice(options?.toolChoice),
907
+ });
908
+ }
909
+
910
+ case "ollama-chat":
911
+ return castApi<"ollama-chat">({
912
+ ...base,
913
+ reasoning: resolveOpenAiReasoningEffort(model, options),
914
+ toolChoice: options?.toolChoice,
915
+ });
916
+
917
+ case "cursor-agent": {
918
+ const execHandlers = options?.cursorExecHandlers ?? options?.execHandlers;
919
+ const onToolResult = options?.cursorOnToolResult ?? execHandlers?.onToolResult;
920
+ return castApi<"cursor-agent">({
921
+ ...base,
922
+ execHandlers,
923
+ onToolResult,
924
+ });
925
+ }
926
+
927
+ default:
928
+ throw new Error(`Unhandled API in mapOptionsForApi: ${model.api}`);
929
+ }
930
+ }
931
+
932
+ function getGoogleBudget(
933
+ model: Model<"google-generative-ai">,
934
+ effort: Effort,
935
+ customBudgets?: ThinkingBudgets,
936
+ ): number {
937
+ requireSupportedEffort(model, effort);
938
+
939
+ // Custom budgets take precedence if provided for this level
940
+ if (customBudgets?.[effort] !== undefined) {
941
+ return customBudgets[effort]!;
942
+ }
943
+
944
+ // See https://ai.google.dev/gemini-api/docs/thinking#set-budget
945
+ if (model.id.includes("2.5-")) {
946
+ switch (effort) {
947
+ case "minimal":
948
+ return 128;
949
+ case "low":
950
+ return 2048;
951
+ case "medium":
952
+ return 8192;
953
+ default:
954
+ return model.id.includes("2.5-flash") ? 24576 : 32768;
955
+ }
956
+ }
957
+
958
+ // Unknown model - use dynamic
959
+ return -1;
960
+ }