@sayknow-cli/ai 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (361) hide show
  1. package/CHANGELOG.md +2788 -0
  2. package/README.md +1183 -0
  3. package/dist/types/api-registry.d.ts +30 -0
  4. package/dist/types/auth-broker/client.d.ts +66 -0
  5. package/dist/types/auth-broker/index.d.ts +5 -0
  6. package/dist/types/auth-broker/refresher.d.ts +25 -0
  7. package/dist/types/auth-broker/remote-store.d.ts +96 -0
  8. package/dist/types/auth-broker/server.d.ts +32 -0
  9. package/dist/types/auth-broker/types.d.ts +105 -0
  10. package/dist/types/auth-broker/wire-schemas.d.ts +412 -0
  11. package/dist/types/auth-gateway/http.d.ts +39 -0
  12. package/dist/types/auth-gateway/index.d.ts +3 -0
  13. package/dist/types/auth-gateway/server.d.ts +17 -0
  14. package/dist/types/auth-gateway/types.d.ts +115 -0
  15. package/dist/types/auth-storage.d.ts +660 -0
  16. package/dist/types/cli.d.ts +2 -0
  17. package/dist/types/index.d.ts +51 -0
  18. package/dist/types/model-cache.d.ts +17 -0
  19. package/dist/types/model-manager.d.ts +62 -0
  20. package/dist/types/model-thinking.d.ts +74 -0
  21. package/dist/types/models.d.ts +12 -0
  22. package/dist/types/provider-details.d.ts +24 -0
  23. package/dist/types/provider-models/bundled-references.d.ts +4 -0
  24. package/dist/types/provider-models/descriptors.d.ts +48 -0
  25. package/dist/types/provider-models/google.d.ts +20 -0
  26. package/dist/types/provider-models/index.d.ts +5 -0
  27. package/dist/types/provider-models/ollama.d.ts +7 -0
  28. package/dist/types/provider-models/openai-compat.d.ts +244 -0
  29. package/dist/types/provider-models/special.d.ts +16 -0
  30. package/dist/types/providers/amazon-bedrock.d.ts +60 -0
  31. package/dist/types/providers/anthropic-messages-server-schema.d.ts +450 -0
  32. package/dist/types/providers/anthropic-messages-server.d.ts +17 -0
  33. package/dist/types/providers/anthropic.d.ts +198 -0
  34. package/dist/types/providers/aws-credentials.d.ts +43 -0
  35. package/dist/types/providers/aws-eventstream.d.ts +38 -0
  36. package/dist/types/providers/aws-sigv4.d.ts +55 -0
  37. package/dist/types/providers/azure-openai-responses.d.ts +15 -0
  38. package/dist/types/providers/composer-discipline.d.ts +26 -0
  39. package/dist/types/providers/cursor/gen/agent_pb.d.ts +13022 -0
  40. package/dist/types/providers/cursor.d.ts +44 -0
  41. package/dist/types/providers/error-message.d.ts +27 -0
  42. package/dist/types/providers/github-copilot-headers.d.ts +40 -0
  43. package/dist/types/providers/gitlab-duo.d.ts +27 -0
  44. package/dist/types/providers/google-auth.d.ts +24 -0
  45. package/dist/types/providers/google-gemini-cli.d.ts +72 -0
  46. package/dist/types/providers/google-gemini-headers.d.ts +18 -0
  47. package/dist/types/providers/google-shared.d.ts +173 -0
  48. package/dist/types/providers/google-types.d.ts +138 -0
  49. package/dist/types/providers/google-vertex.d.ts +7 -0
  50. package/dist/types/providers/google.d.ts +4 -0
  51. package/dist/types/providers/grammar.d.ts +1 -0
  52. package/dist/types/providers/kimi.d.ts +27 -0
  53. package/dist/types/providers/mock.d.ts +175 -0
  54. package/dist/types/providers/ollama.d.ts +41 -0
  55. package/dist/types/providers/openai-anthropic-shim.d.ts +31 -0
  56. package/dist/types/providers/openai-chat-server-schema.d.ts +815 -0
  57. package/dist/types/providers/openai-chat-server.d.ts +16 -0
  58. package/dist/types/providers/openai-codex/constants.d.ts +26 -0
  59. package/dist/types/providers/openai-codex/request-transformer.d.ts +49 -0
  60. package/dist/types/providers/openai-codex/response-handler.d.ts +17 -0
  61. package/dist/types/providers/openai-codex-responses.d.ts +67 -0
  62. package/dist/types/providers/openai-completions-compat.d.ts +27 -0
  63. package/dist/types/providers/openai-completions.d.ts +33 -0
  64. package/dist/types/providers/openai-request-transform.d.ts +4 -0
  65. package/dist/types/providers/openai-responses-server-schema.d.ts +392 -0
  66. package/dist/types/providers/openai-responses-server.d.ts +17 -0
  67. package/dist/types/providers/openai-responses-shared.d.ts +89 -0
  68. package/dist/types/providers/openai-responses.d.ts +32 -0
  69. package/dist/types/providers/pi-native-client.d.ts +13 -0
  70. package/dist/types/providers/pi-native-server.d.ts +68 -0
  71. package/dist/types/providers/register-builtins.d.ts +31 -0
  72. package/dist/types/providers/synthetic.d.ts +26 -0
  73. package/dist/types/providers/transform-messages.d.ts +14 -0
  74. package/dist/types/providers/vision-guard.d.ts +8 -0
  75. package/dist/types/rate-limit-utils.d.ts +19 -0
  76. package/dist/types/stream.d.ts +43 -0
  77. package/dist/types/types.d.ts +811 -0
  78. package/dist/types/usage/claude.d.ts +3 -0
  79. package/dist/types/usage/gemini.d.ts +2 -0
  80. package/dist/types/usage/github-copilot.d.ts +7 -0
  81. package/dist/types/usage/google-antigravity.d.ts +2 -0
  82. package/dist/types/usage/grok-cli.d.ts +10 -0
  83. package/dist/types/usage/kimi.d.ts +2 -0
  84. package/dist/types/usage/minimax-code.d.ts +2 -0
  85. package/dist/types/usage/openai-codex.d.ts +3 -0
  86. package/dist/types/usage/shared.d.ts +1 -0
  87. package/dist/types/usage/zai.d.ts +2 -0
  88. package/dist/types/usage.d.ts +258 -0
  89. package/dist/types/utils/abort.d.ts +19 -0
  90. package/dist/types/utils/anthropic-auth.d.ts +31 -0
  91. package/dist/types/utils/discovery/antigravity.d.ts +61 -0
  92. package/dist/types/utils/discovery/codex.d.ts +38 -0
  93. package/dist/types/utils/discovery/cursor.d.ts +23 -0
  94. package/dist/types/utils/discovery/gemini.d.ts +25 -0
  95. package/dist/types/utils/discovery/index.d.ts +4 -0
  96. package/dist/types/utils/discovery/openai-compatible.d.ts +74 -0
  97. package/dist/types/utils/event-stream.d.ts +33 -0
  98. package/dist/types/utils/fireworks-model-id.d.ts +10 -0
  99. package/dist/types/utils/foundry.d.ts +1 -0
  100. package/dist/types/utils/h2-fetch.d.ts +22 -0
  101. package/dist/types/utils/http-inspector.d.ts +35 -0
  102. package/dist/types/utils/idle-iterator.d.ts +67 -0
  103. package/dist/types/utils/json-parse.d.ts +10 -0
  104. package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +18 -0
  105. package/dist/types/utils/oauth/anthropic.d.ts +22 -0
  106. package/dist/types/utils/oauth/api-key-login.d.ts +35 -0
  107. package/dist/types/utils/oauth/api-key-validation.d.ts +27 -0
  108. package/dist/types/utils/oauth/callback-server.d.ts +60 -0
  109. package/dist/types/utils/oauth/cerebras.d.ts +1 -0
  110. package/dist/types/utils/oauth/cloudflare-ai-gateway.d.ts +18 -0
  111. package/dist/types/utils/oauth/cursor.d.ts +15 -0
  112. package/dist/types/utils/oauth/deepseek.d.ts +10 -0
  113. package/dist/types/utils/oauth/firepass.d.ts +1 -0
  114. package/dist/types/utils/oauth/fireworks.d.ts +1 -0
  115. package/dist/types/utils/oauth/github-copilot.d.ts +38 -0
  116. package/dist/types/utils/oauth/gitlab-duo.d.ts +3 -0
  117. package/dist/types/utils/oauth/google-antigravity.d.ts +11 -0
  118. package/dist/types/utils/oauth/google-gemini-cli.d.ts +10 -0
  119. package/dist/types/utils/oauth/google-oauth-shared.d.ts +28 -0
  120. package/dist/types/utils/oauth/huggingface.d.ts +19 -0
  121. package/dist/types/utils/oauth/index.d.ts +38 -0
  122. package/dist/types/utils/oauth/kagi.d.ts +17 -0
  123. package/dist/types/utils/oauth/kilo.d.ts +5 -0
  124. package/dist/types/utils/oauth/kimi.d.ts +21 -0
  125. package/dist/types/utils/oauth/litellm.d.ts +18 -0
  126. package/dist/types/utils/oauth/lm-studio.d.ts +17 -0
  127. package/dist/types/utils/oauth/minimax-code.d.ts +28 -0
  128. package/dist/types/utils/oauth/moonshot.d.ts +1 -0
  129. package/dist/types/utils/oauth/nanogpt.d.ts +1 -0
  130. package/dist/types/utils/oauth/nvidia.d.ts +18 -0
  131. package/dist/types/utils/oauth/ollama-cloud.d.ts +2 -0
  132. package/dist/types/utils/oauth/ollama.d.ts +18 -0
  133. package/dist/types/utils/oauth/openai-codex.d.ts +21 -0
  134. package/dist/types/utils/oauth/opencode.d.ts +18 -0
  135. package/dist/types/utils/oauth/parallel.d.ts +17 -0
  136. package/dist/types/utils/oauth/perplexity.d.ts +9 -0
  137. package/dist/types/utils/oauth/pkce.d.ts +8 -0
  138. package/dist/types/utils/oauth/qianfan.d.ts +17 -0
  139. package/dist/types/utils/oauth/qwen-portal.d.ts +19 -0
  140. package/dist/types/utils/oauth/synthetic.d.ts +1 -0
  141. package/dist/types/utils/oauth/tavily.d.ts +17 -0
  142. package/dist/types/utils/oauth/together.d.ts +1 -0
  143. package/dist/types/utils/oauth/types.d.ts +45 -0
  144. package/dist/types/utils/oauth/venice.d.ts +18 -0
  145. package/dist/types/utils/oauth/vercel-ai-gateway.d.ts +18 -0
  146. package/dist/types/utils/oauth/vllm.d.ts +16 -0
  147. package/dist/types/utils/oauth/xai.d.ts +30 -0
  148. package/dist/types/utils/oauth/xiaomi.d.ts +25 -0
  149. package/dist/types/utils/oauth/zai.d.ts +18 -0
  150. package/dist/types/utils/oauth/zenmux.d.ts +1 -0
  151. package/dist/types/utils/overflow.d.ts +54 -0
  152. package/dist/types/utils/parse-bind.d.ts +23 -0
  153. package/dist/types/utils/provider-response.d.ts +3 -0
  154. package/dist/types/utils/retry-after.d.ts +3 -0
  155. package/dist/types/utils/retry-budget.d.ts +1 -0
  156. package/dist/types/utils/retry.d.ts +26 -0
  157. package/dist/types/utils/schema/adapt.d.ts +24 -0
  158. package/dist/types/utils/schema/compatibility.d.ts +30 -0
  159. package/dist/types/utils/schema/dereference.d.ts +11 -0
  160. package/dist/types/utils/schema/draft.d.ts +10 -0
  161. package/dist/types/utils/schema/equality.d.ts +4 -0
  162. package/dist/types/utils/schema/fields.d.ts +49 -0
  163. package/dist/types/utils/schema/index.d.ts +13 -0
  164. package/dist/types/utils/schema/json-schema-validator.d.ts +12 -0
  165. package/dist/types/utils/schema/meta-validator.d.ts +2 -0
  166. package/dist/types/utils/schema/normalize.d.ts +93 -0
  167. package/dist/types/utils/schema/spill.d.ts +8 -0
  168. package/dist/types/utils/schema/stamps.d.ts +25 -0
  169. package/dist/types/utils/schema/types.d.ts +4 -0
  170. package/dist/types/utils/schema/wire.d.ts +54 -0
  171. package/dist/types/utils/schema/zod-decontaminate.d.ts +31 -0
  172. package/dist/types/utils/sse-debug.d.ts +10 -0
  173. package/dist/types/utils/tool-call-healing.d.ts +71 -0
  174. package/dist/types/utils/tool-choice-capability.d.ts +41 -0
  175. package/dist/types/utils/tool-choice.d.ts +50 -0
  176. package/dist/types/utils/validation.d.ts +17 -0
  177. package/dist/types/utils.d.ts +34 -0
  178. package/package.json +146 -0
  179. package/src/api-registry.ts +96 -0
  180. package/src/auth-broker/client.ts +358 -0
  181. package/src/auth-broker/index.ts +5 -0
  182. package/src/auth-broker/refresher.ts +127 -0
  183. package/src/auth-broker/remote-store.ts +623 -0
  184. package/src/auth-broker/server.ts +644 -0
  185. package/src/auth-broker/types.ts +127 -0
  186. package/src/auth-broker/wire-schemas.ts +200 -0
  187. package/src/auth-gateway/http.ts +194 -0
  188. package/src/auth-gateway/index.ts +3 -0
  189. package/src/auth-gateway/server.ts +717 -0
  190. package/src/auth-gateway/types.ts +134 -0
  191. package/src/auth-storage.ts +4179 -0
  192. package/src/cli.ts +263 -0
  193. package/src/index.ts +56 -0
  194. package/src/model-cache.ts +129 -0
  195. package/src/model-manager.ts +486 -0
  196. package/src/model-thinking.ts +772 -0
  197. package/src/models.json +75437 -0
  198. package/src/models.json.d.ts +9 -0
  199. package/src/models.ts +82 -0
  200. package/src/prompts/turn-aborted-guidance.md +4 -0
  201. package/src/provider-details.ts +90 -0
  202. package/src/provider-models/bundled-references.ts +38 -0
  203. package/src/provider-models/descriptors.ts +327 -0
  204. package/src/provider-models/google.ts +91 -0
  205. package/src/provider-models/index.ts +5 -0
  206. package/src/provider-models/ollama.ts +153 -0
  207. package/src/provider-models/openai-compat.ts +2352 -0
  208. package/src/provider-models/special.ts +67 -0
  209. package/src/providers/amazon-bedrock.ts +937 -0
  210. package/src/providers/anthropic-messages-server-schema.ts +229 -0
  211. package/src/providers/anthropic-messages-server.ts +677 -0
  212. package/src/providers/anthropic.ts +2940 -0
  213. package/src/providers/aws-credentials.ts +501 -0
  214. package/src/providers/aws-eventstream.ts +185 -0
  215. package/src/providers/aws-sigv4.ts +218 -0
  216. package/src/providers/azure-openai-responses.ts +379 -0
  217. package/src/providers/composer-discipline.ts +41 -0
  218. package/src/providers/cursor/gen/agent_pb.ts +15274 -0
  219. package/src/providers/cursor/proto/agent.proto +3526 -0
  220. package/src/providers/cursor/proto/buf.gen.yaml +6 -0
  221. package/src/providers/cursor/proto/buf.yaml +17 -0
  222. package/src/providers/cursor.ts +2671 -0
  223. package/src/providers/error-message.ts +21 -0
  224. package/src/providers/github-copilot-headers.ts +140 -0
  225. package/src/providers/gitlab-duo.ts +372 -0
  226. package/src/providers/google-auth.ts +252 -0
  227. package/src/providers/google-gemini-cli.ts +856 -0
  228. package/src/providers/google-gemini-headers.ts +41 -0
  229. package/src/providers/google-shared.ts +951 -0
  230. package/src/providers/google-types.ts +167 -0
  231. package/src/providers/google-vertex.ts +88 -0
  232. package/src/providers/google.ts +41 -0
  233. package/src/providers/grammar.ts +70 -0
  234. package/src/providers/kimi.ts +52 -0
  235. package/src/providers/mock.ts +500 -0
  236. package/src/providers/ollama.ts +603 -0
  237. package/src/providers/openai-anthropic-shim.ts +138 -0
  238. package/src/providers/openai-chat-server-schema.ts +243 -0
  239. package/src/providers/openai-chat-server.ts +635 -0
  240. package/src/providers/openai-codex/constants.ts +43 -0
  241. package/src/providers/openai-codex/request-transformer.ts +161 -0
  242. package/src/providers/openai-codex/response-handler.ts +81 -0
  243. package/src/providers/openai-codex-responses.ts +2774 -0
  244. package/src/providers/openai-completions-compat.ts +289 -0
  245. package/src/providers/openai-completions.ts +1935 -0
  246. package/src/providers/openai-request-transform.ts +136 -0
  247. package/src/providers/openai-responses-server-schema.ts +290 -0
  248. package/src/providers/openai-responses-server.ts +1190 -0
  249. package/src/providers/openai-responses-shared.ts +800 -0
  250. package/src/providers/openai-responses.ts +738 -0
  251. package/src/providers/pi-native-client.ts +227 -0
  252. package/src/providers/pi-native-server.ts +210 -0
  253. package/src/providers/register-builtins.ts +411 -0
  254. package/src/providers/synthetic.ts +50 -0
  255. package/src/providers/transform-messages.ts +319 -0
  256. package/src/providers/vision-guard.ts +31 -0
  257. package/src/rate-limit-utils.ts +93 -0
  258. package/src/stream.ts +960 -0
  259. package/src/types.ts +967 -0
  260. package/src/usage/claude.ts +431 -0
  261. package/src/usage/gemini.ts +250 -0
  262. package/src/usage/github-copilot.ts +421 -0
  263. package/src/usage/google-antigravity.ts +201 -0
  264. package/src/usage/grok-cli.ts +163 -0
  265. package/src/usage/kimi.ts +271 -0
  266. package/src/usage/minimax-code.ts +31 -0
  267. package/src/usage/openai-codex.ts +503 -0
  268. package/src/usage/shared.ts +10 -0
  269. package/src/usage/zai.ts +247 -0
  270. package/src/usage.ts +183 -0
  271. package/src/utils/abort.ts +51 -0
  272. package/src/utils/anthropic-auth.ts +87 -0
  273. package/src/utils/discovery/antigravity.ts +261 -0
  274. package/src/utils/discovery/codex.ts +371 -0
  275. package/src/utils/discovery/cursor.ts +306 -0
  276. package/src/utils/discovery/gemini.ts +248 -0
  277. package/src/utils/discovery/index.ts +4 -0
  278. package/src/utils/discovery/openai-compatible.ts +230 -0
  279. package/src/utils/event-stream.ts +172 -0
  280. package/src/utils/fireworks-model-id.ts +30 -0
  281. package/src/utils/foundry.ts +8 -0
  282. package/src/utils/h2-fetch.ts +60 -0
  283. package/src/utils/http-inspector.ts +255 -0
  284. package/src/utils/idle-iterator.ts +257 -0
  285. package/src/utils/json-parse.ts +148 -0
  286. package/src/utils/oauth/alibaba-coding-plan.ts +59 -0
  287. package/src/utils/oauth/anthropic.ts +200 -0
  288. package/src/utils/oauth/api-key-login.ts +87 -0
  289. package/src/utils/oauth/api-key-validation.ts +92 -0
  290. package/src/utils/oauth/callback-server.ts +281 -0
  291. package/src/utils/oauth/cerebras.ts +16 -0
  292. package/src/utils/oauth/cloudflare-ai-gateway.ts +48 -0
  293. package/src/utils/oauth/cursor.ts +157 -0
  294. package/src/utils/oauth/deepseek.ts +53 -0
  295. package/src/utils/oauth/firepass.ts +24 -0
  296. package/src/utils/oauth/fireworks.ts +15 -0
  297. package/src/utils/oauth/github-copilot.ts +362 -0
  298. package/src/utils/oauth/gitlab-duo.ts +123 -0
  299. package/src/utils/oauth/google-antigravity.ts +200 -0
  300. package/src/utils/oauth/google-gemini-cli.ts +256 -0
  301. package/src/utils/oauth/google-oauth-shared.ts +110 -0
  302. package/src/utils/oauth/huggingface.ts +62 -0
  303. package/src/utils/oauth/index.ts +469 -0
  304. package/src/utils/oauth/kagi.ts +47 -0
  305. package/src/utils/oauth/kilo.ts +87 -0
  306. package/src/utils/oauth/kimi.ts +254 -0
  307. package/src/utils/oauth/litellm.ts +47 -0
  308. package/src/utils/oauth/lm-studio.ts +38 -0
  309. package/src/utils/oauth/minimax-code.ts +78 -0
  310. package/src/utils/oauth/moonshot.ts +16 -0
  311. package/src/utils/oauth/nanogpt.ts +15 -0
  312. package/src/utils/oauth/nvidia.ts +70 -0
  313. package/src/utils/oauth/oauth.html +199 -0
  314. package/src/utils/oauth/ollama-cloud.ts +28 -0
  315. package/src/utils/oauth/ollama.ts +47 -0
  316. package/src/utils/oauth/openai-codex.ts +299 -0
  317. package/src/utils/oauth/opencode.ts +49 -0
  318. package/src/utils/oauth/parallel.ts +46 -0
  319. package/src/utils/oauth/perplexity.ts +206 -0
  320. package/src/utils/oauth/pkce.ts +18 -0
  321. package/src/utils/oauth/qianfan.ts +58 -0
  322. package/src/utils/oauth/qwen-portal.ts +60 -0
  323. package/src/utils/oauth/synthetic.ts +16 -0
  324. package/src/utils/oauth/tavily.ts +46 -0
  325. package/src/utils/oauth/together.ts +16 -0
  326. package/src/utils/oauth/types.ts +99 -0
  327. package/src/utils/oauth/venice.ts +59 -0
  328. package/src/utils/oauth/vercel-ai-gateway.ts +47 -0
  329. package/src/utils/oauth/vllm.ts +40 -0
  330. package/src/utils/oauth/xai.ts +246 -0
  331. package/src/utils/oauth/xiaomi.ts +199 -0
  332. package/src/utils/oauth/zai.ts +60 -0
  333. package/src/utils/oauth/zenmux.ts +15 -0
  334. package/src/utils/overflow.ts +137 -0
  335. package/src/utils/parse-bind.ts +54 -0
  336. package/src/utils/provider-response.ts +30 -0
  337. package/src/utils/retry-after.ts +110 -0
  338. package/src/utils/retry-budget.ts +4 -0
  339. package/src/utils/retry.ts +54 -0
  340. package/src/utils/schema/CONSTRAINTS.md +164 -0
  341. package/src/utils/schema/adapt.ts +36 -0
  342. package/src/utils/schema/compatibility.ts +435 -0
  343. package/src/utils/schema/dereference.ts +98 -0
  344. package/src/utils/schema/draft.ts +341 -0
  345. package/src/utils/schema/equality.ts +97 -0
  346. package/src/utils/schema/fields.ts +190 -0
  347. package/src/utils/schema/index.ts +13 -0
  348. package/src/utils/schema/json-schema-validator.ts +577 -0
  349. package/src/utils/schema/meta-validator.ts +167 -0
  350. package/src/utils/schema/normalize.ts +1588 -0
  351. package/src/utils/schema/spill.ts +43 -0
  352. package/src/utils/schema/stamps.ts +97 -0
  353. package/src/utils/schema/types.ts +11 -0
  354. package/src/utils/schema/wire.ts +213 -0
  355. package/src/utils/schema/zod-decontaminate.ts +331 -0
  356. package/src/utils/sse-debug.ts +289 -0
  357. package/src/utils/tool-call-healing.ts +271 -0
  358. package/src/utils/tool-choice-capability.ts +220 -0
  359. package/src/utils/tool-choice.ts +99 -0
  360. package/src/utils/validation.ts +1019 -0
  361. package/src/utils.ts +178 -0
package/CHANGELOG.md ADDED
@@ -0,0 +1,2788 @@
1
+ # Changelog
2
+
3
+ ## [Unreleased]
4
+
5
+ ## [0.6.0] - 2026-06-18
6
+ ### Fixed
7
+
8
+ - Corrected the bundled `zai/glm-5.2` context window from 200K to its true 1M (1,000,000-token) lossless window. GLM-5.2 was added in #579 by copying GLM-5.1's 200K entry, and that stale value survived every `generate-models` run because provider-scoped models bypass the models.dev refresh in `applyGlobalModelsDevFallback`. Added a regen-safe pin in `applyGeneratedModelPolicy` (mirroring the Bedrock-Opus-4.6 precedent) so the 1M value persists, and updated the bundled catalog entry. The wrong 200K tripped auto-compaction / context-cap thresholds ~5x early for GLM-5.2 sessions.
9
+ - Corrected the bundled `minimax-m3` context window from 512K to its true 1M (1,000,000-token) window per the official MiniMax docs (platform.minimax.io documents MiniMax-M3 as a 1M-context frontier coding model). All four provider copies (`minimax`, `minimax-cn`, `minimax-code`, `minimax-code-cn`) carried the stale 512K, which survived `generate-models` (provider-scoped models bypass the models.dev refresh in `applyGlobalModelsDevFallback`) and tripped auto-compaction / context-cap thresholds 2x early on MiniMax sessions. Added a regen-safe pin in `applyGeneratedModelPolicy` keyed on `model.id === "minimax-m3"` and updated the bundled entries. `minimax-v3` (an undocumented catalog alias) is intentionally left untouched.
10
+
11
+ ## [0.5.4] - 2026-06-17
12
+
13
+ ### Fixed
14
+
15
+ - Fixed Anthropic tool schema compatibility for discriminated-union tool inputs by flattening only the model-facing input-schema root, avoiding top-level `oneOf`/`anyOf`/`allOf` request rejections while preserving nested combinators and runtime validation authority.
16
+
17
+ - Made the "No API key for provider" error from `stream`/`complete` actionable for OpenCode Go/Zen subscription providers in headless runs (#755). The subscription is itself an API key (`OPENCODE_API_KEY`, created at https://opencode.ai/auth), not a separate OAuth/session token; the new `formatProviderCredentialHint` helper (composed into `formatMissingApiKeyError`) names the env var SKC reads, warns that a project `.env` is intentionally ignored for provider credentials, and points OpenCode users at the one-time interactive `skc auth-broker login <provider>` credential capture to run before headless/print mode. No auth behavior changed.
18
+
19
+ ## [0.5.3] - 2026-06-16
20
+
21
+ ### Added
22
+
23
+ - Added opt-in `AuthStorageOptions.credentialRankingMode` (`balanced` (default) | `earliest-reset`) for multi-account OAuth credential selection. `earliest-reset` ranks non-blocked credentials earliest-expiry-first — draining the soonest-to-reset account before its perishable tumbling-window quota (e.g. Claude 5h/7d) is lost at reset — keeping the existing drain-rate/used-fraction metrics as tiebreakers. `balanced` is byte-identical to prior behavior, and ranking only runs at session start (or when the session's preferred credential is blocked), so this never thrashes accounts mid-session.
24
+
25
+ ### Fixed
26
+
27
+ - Allowed `openai-codex-responses` custom backends to use opaque `apiKey` bearer tokens by omitting `chatgpt-account-id` when the token does not expose a Codex account id.
28
+ - Fixed OpenAI code websocket continuations to treat codex-lb's `codex_previous_response_stale` response failures as expired `previous_response_id` anchors and retry with full context instead of surfacing the transient failure.
29
+ - Bounded the Cursor provider's conversation cache with an LRU(64) + 1h TTL and added `disposeCursorConversation`, so long-running sessions no longer retain Cursor conversation state without limit (#717).
30
+
31
+ ## [0.5.2] - 2026-06-15
32
+
33
+ ### Changed
34
+
35
+ - Changed the Anthropic provider's default prompt-cache retention to `long` (`ttl: "1h"`) when a request and model omit `cacheRetention`. The previous default (~5m) was too fragile for long-running Codex/Sayknow-CLI subagent workflows, where the cached prefix was frequently evicted between turns. The 1h `ttl` marker is only emitted on the canonical Anthropic API (`api.anthropic.com`) for models advertising `supportsLongCacheRetention`; proxies, gateways, and models without that capability still fall back to the default ephemeral breakpoint (Anthropic services it at ~5m). Explicit request/model `cacheRetention` and the `SKC_CACHE_RETENTION`/`PI_CACHE_RETENTION` env overrides continue to win, and `resolveCacheRetention` now accepts a `fallback` argument (defaulting to `"short"`) so non-Anthropic providers are unaffected.
36
+
37
+ ## [0.5.1] - 2026-06-14
38
+
39
+ ### Fixed
40
+
41
+ - Classified model/message limit exhaustion as persistent usage-limit errors so hosts fail fast or switch credentials instead of leaving sessions in an unbounded retry/working state.
42
+
43
+ ## [0.5.0] - 2026-06-13
44
+
45
+ ### Added
46
+
47
+ - Added a generic tool-choice capability model: `toolChoiceSupport` compat enum (`none`/`auto`/`required`/`named`) available on every forced-choice-capable API, derived from the legacy `supportsToolChoice`/`supportsForcedToolChoice` booleans when absent, with a shared `resolveToolChoice` helper that clamps requested tool choices (`named` → `required` → omit) and returns structured degradation metadata.
48
+ - Added a transparent one-shot fallback for forced `tool_choice` 400s ("tool_choice forces tool use is not compatible with this model" and equivalents): transports retry once without the forced field at a pre-content streaming boundary, record the discovery in an in-memory per-process incapability registry, and emit an internal non-rendered `toolChoiceIncapability` event. Applies to Anthropic, OpenAI Completions/Responses, Azure Responses, OpenAI code Responses, Bedrock (including event-stream `validationException`), Ollama, Google, and Gemini CLI transports.
49
+ - Added bundled catalog entries for `kimi-code/kimi-k2.7-code`, `minimax-code/minimax-v3`, and `xai/grok-composer-2.5-fast`.
50
+ - Added composer-harness anchor/edit discipline injection for Cursor Composer and Grok Composer models so provider-specific coding harness priors do not override SKC hashline/edit contracts.
51
+
52
+ ### Removed
53
+
54
+ - Removed the retired `anthropic/claude-fable-5` bundled catalog entry.
55
+
56
+ ### Changed
57
+
58
+ - Moved the Claude Mythos forced-tool-use incapability knowledge out of Anthropic request code into catalog compat defaults (`toolChoiceSupport: "auto"`), applied during catalog generation, dynamic discovery, and bundled-model loading via a shared predicate.
59
+ - Google `toolConfig` mapping now sends `FunctionCallingConfig` mode `ANY` for both `required` and `any` requests instead of silently relaxing `required` to `AUTO`.
60
+ - Optimized `EventStream` queue draining with a head-indexed queue to avoid repeated array shifts in hot streaming paths.
61
+ - Clarified lazy builtin provider registration as the main provider loading path.
62
+
63
+ ### Fixed
64
+
65
+ - Stripped `OpenAI-Beta` in the `openai-proxy` request transform profile so OpenAI-compatible proxies do not receive SDK beta headers.
66
+
67
+ ## [0.4.5] - 2026-06-12
68
+
69
+ ### Changed
70
+
71
+ - Bumped the spoofed Gemini CLI User-Agent version to 0.46.0 to track the upstream release.
72
+
73
+ ### Fixed
74
+
75
+ - Fixed direct Anthropic requests for Claude Mythos-style models that support tools but reject forced tool use by omitting forced `tool_choice` while preserving `auto`/`none` choices.
76
+ - Preserved catalog transport metadata for opencode-go `qwen3.7-max` model resolution.
77
+ - Set SQLite auth-store `busy_timeout` before enabling WAL so initialization is reliable under contention.
78
+ - Resolved provider credentials from inherited or SKC-owned environment sources instead of trusting the caller project's `.env` overlays.
79
+ - Rendered and executed Cursor-native tool calls without dropping provider-specific call details.
80
+
81
+ ## [0.4.4] - 2026-06-10
82
+
83
+ - Version aligned with the 0.4.4 monorepo release; no functional changes in this package.
84
+
85
+ ## [0.4.2] - 2026-06-09
86
+
87
+ ### Fixed
88
+
89
+ - Treated `gpt-5.5` as a 400K-context model wherever context caps / auto-promote thresholds are resolved, so a ~272K session is no longer considered over-cap and no longer demotes to `gpt-5.4`. Pinned the OpenAI Codex `gpt-5.5` context window to 400K and removed its `gpt-5.4` promotion target (the smaller window made it a demotion) ([#428](https://github.com/jaybeyond/sayknow-cli/issues/428)).
90
+
91
+ ## [0.4.0] - 2026-06-06
92
+
93
+ ### Added
94
+
95
+ - Added minimax-m3 model support across MiniMax providers.
96
+ - Honored the `SKC_CACHE_RETENTION` environment variable and `cacheRetention` model config so hosts can control provider prompt-cache retention (#379/#381).
97
+ - Added an Opus max reasoning preset to the model thinking presets (#372).
98
+ - Refreshed the generated models schema for the new model/config surface (#382).
99
+
100
+ ### Changed
101
+
102
+ - Pinned the OpenAI Codex provider default to GPT-5.5 at `xhigh` reasoning effort (#352). This changes the default model and effort for Codex users (latency/cost/quality impact) and is a behavior change, not an API break; pass an explicit model/effort to override.
103
+ - Bumped the spoofed Gemini CLI user-agent version to 0.45.2 to track the upstream release.
104
+
105
+ ## [0.3.0] - 2026-06-03
106
+
107
+ ### Added
108
+
109
+ - Added xAI to the `/login` provider catalog as a Grok OAuth login with PKCE, refresh-token storage, and mocked login/refresh coverage.
110
+
111
+ ## [0.2.4] - 2026-06-02
112
+
113
+ ### Added
114
+
115
+ - Added configurable provider request and stream retry budgets so hosts can bound transient upstream/server retry behavior separately from session-level retries.
116
+
117
+ ## [0.2.2] - 2026-05-31
118
+
119
+ ### Fixed
120
+
121
+ - Fixed Anthropic extended-thinking replay after aborted turns by dropping partial `thinking`/`redacted_thinking` blocks before the next request, preserving/synthesizing matching tool results, and retrying once with repaired latest-assistant thinking when Anthropic rejects a replay with the immutable-thinking HTTP 400 ([#107](https://github.com/jaybeyond/sayknow-cli/issues/107)).
122
+ - Fixed first-class `azure-openai` catalog models so normal provider auth resolution reads `AZURE_OPENAI_API_KEY` when streaming `azure-openai/gpt-*` models.
123
+
124
+ ## [0.2.1] - 2026-05-30
125
+
126
+ ### Changed
127
+
128
+ - Refreshed AI package metadata for the SKC 0.2.1 release.
129
+
130
+ ## [0.2.0] - 2026-05-28
131
+
132
+ ### Fixed
133
+
134
+ - Fixed OpenAI-compatible base URL handling so configured proxy URLs and inherited environment overrides are respected at model discovery, completions, responses, and streaming call sites.
135
+ - Fixed OpenAI direct-provider feature gates so prompt-cache/session behavior uses the resolved Responses base URL and only treats exact default `api.openai.com` hosts/paths as direct OpenAI.
136
+
137
+ ## [0.1.3] - 2026-05-28
138
+
139
+ ### Changed
140
+
141
+ - Released the current dev branch fixes with refreshed 0.1.3 package metadata.
142
+
143
+ ## [0.1.2] - 2026-05-28
144
+
145
+ ### Changed
146
+
147
+ - Updated package metadata for the Sayknow-CLI npm publication.
148
+
149
+ ## [0.1.1] - 2026-05-28
150
+ ### Breaking Changes
151
+
152
+ - Removed `findAnthropicAuth` from `anthropic-auth` and replaced store-driven auth discovery with `buildAnthropicAuthConfig`, requiring callers to provide an already-resolved API key before building Anthropic auth config
153
+
154
+ ### Added
155
+
156
+ - Added `AuthStorage.getOAuthAccess` to return a refreshed OAuth access token with identity metadata (`accountId`, `email`, `projectId`, `enterpriseUrl`) for callers that need bearer-token headers together
157
+
158
+ ### Changed
159
+
160
+ - Changed OAuth selection in `AuthStorage` to treat credentials as stale when they are within 60 seconds of expiry and rotate them preemptively
161
+ - Changed Google Gemini CLI, Google Gemini usage, Antigravity usage, and Kimi usage flows to stop refreshing OAuth tokens directly and rely on `AuthStorage` for token rotation
162
+
163
+ ### Removed
164
+
165
+ - Removed provider-local OAuth refresh helpers from Google Gemini CLI and Google/Kimi/Antigravity usage probes, preventing direct refresh calls from those usage paths
166
+
167
+ ### Fixed
168
+
169
+ - Fixed expired OAuth handling so provider-level paths no longer attempt direct token refresh calls for expired credentials and instead rely on `AuthStorage` for rotation
170
+ - Fixed `google-gemini-cli` / `google-antigravity` aborting heavy reasoning runs with "Provider stream timed out while waiting for the first event" before the upstream had a chance to emit its first SSE frame. Cloud Code Assist routinely takes >100s on Gemini 3.x Pro at high thinking levels; the lazy-stream wrapper now floors the first-event watchdog at 5 minutes for these two providers when neither `StreamOptions.streamFirstEventTimeoutMs` nor `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` pins a value. Other providers keep the 100s default. Internally, `getStreamIdleTimeoutMs` and `getStreamFirstEventTimeoutMs` now accept an optional per-provider `fallbackMs` so other slow-first-token providers can opt into the same widening without leaking through to the global default.
171
+ - Fixed Anthropic model Opus 4.7 on Amazon Bedrock streaming no reasoning output (and appearing to hang on long reasoning runs) because Anthropic silently switched the adaptive-thinking display default to `"omitted"`. The Bedrock provider now sends `thinking.display = "summarized"` by default on Opus 4.7+ adaptive models and on budget-based Anthropic model models, mirroring the existing direct-Anthropic behavior. `BedrockOptions.thinkingDisplay` (`"summarized" | "omitted"`) is exposed for callers that want to opt out, and `hideThinkingSummary` now wires through to the Bedrock case ([#1373](https://github.com/jaybeyond/sayknow-cli/issues/1373)).
172
+ - Fixed Cursor Composer resume/tool-continuation turns failing with `Cannot send empty user message to Cursor API`. Empty current user turns now use Cursor's `resumeAction` instead of constructing an invalid `userMessageAction` ([#1376](https://github.com/jaybeyond/sayknow-cli/issues/1376)).
173
+
174
+ ## [15.3.2] - 2026-05-25
175
+ ### Added
176
+
177
+ - Added `GET /v1/snapshot/stream` for live auth-broker snapshot updates via SSE with `snapshot`, `entry`, and `removed` event frames
178
+ - Added `AuthBrokerClient.openSnapshotStream()` for consuming SSE snapshot streams from `/v1/snapshot/stream`
179
+ - Added `streamSnapshots` option to `RemoteAuthCredentialStore` (default `true`) to enable or disable SSE-based snapshot synchronization
180
+ - Added `streamKeepaliveMs` to `startAuthBroker()` to tune heartbeat frequency for the SSE stream
181
+ - Added `AuthStorage.checkCredentials({ signal?, timeoutMs?, baseUrlResolver? })` that returns a per-credential `CredentialHealthResult` with tri-state `ok` (`true` / `false` / `null`-unverifiable), the credential's identity (provider, type, email/accountId, broker-refresh flag), and the upstream error string when the probe fails. Iterates sequentially over `listAuthCredentials()`, exercises OAuth refresh on expiry, then calls the per-provider `UsageProvider.fetchUsage` without swallowing errors — so callers can identify which row in a multi-account broker is producing 401s instead of getting a silently-deduplicated `fetchUsageReports` list.
182
+ - Added `GET /v1/credentials/check` to `startAuthGateway()` that forwards to `AuthStorage.checkCredentials` and returns `{ generatedAt, credentials }`. Gated by the same bearer as the rest of the gateway.
183
+
184
+ ### Changed
185
+
186
+ - Changed `RemoteAuthCredentialStore` to prefer SSE snapshot streaming and automatically fall back to long-polling when a broker returns 404 for `/v1/snapshot/stream`
187
+ - Changed snapshot write-refresh flow so `RemoteAuthCredentialStore` skips immediate `/v1/snapshot` refreshes when SSE streaming is active
188
+ - Changed broker SSE stream behavior to keep connections open with periodic keepalives and an increased server idle timeout
189
+
190
+ ## [15.3.0] - 2026-05-25
191
+
192
+ ### Added
193
+
194
+ - Added DeepSeek to the built-in API-key login provider catalog so `skc login deepseek` stores a reusable `DEEPSEEK_API_KEY` credential for the bundled DeepSeek models.
195
+
196
+ ### Fixed
197
+
198
+ - Fixed `openai-responses` requests intermittently 400ing with `No tool call found for function call output with call_id …` after an aborted turn or a locally-rejected tool call (e.g. argument-validation failure). `convertConversationMessages` now folds orphan `function_call_output` / `custom_tool_call_output` items — those whose matching `function_call` was wiped by an earlier `dt: false` snapshot splice or never landed in any persisted provider payload — into assistant text notes, preserving the payload while keeping the request grammatically valid ([#1351](https://github.com/jaybeyond/sayknow-cli/issues/1351)).
199
+
200
+ ## [15.2.4] - 2026-05-22
201
+
202
+ ### Fixed
203
+
204
+ - Fixed ChatGPT Plus/Pro (OpenAI code) OAuth login returning `Token exchange failed: 403` on Windows. When port 1455 was in use, the callback server silently fell back to a random port; OpenAI's authorization endpoint accepts any localhost redirect URI (loose validation), so the browser callback succeeds and shows "Authentication Successful", but the token endpoint rejects the non-registered port with 403. The `OpenAIOpenAI codeOAuthFlow` now enforces a fixed `redirectUri` option so a busy port immediately surfaces as "port unavailable" instead of producing a confusing 403 ([#1277](https://github.com/jaybeyond/sayknow-cli/issues/1277)).
205
+ - Improved `exchangeCodeForToken` error diagnostics: the 403 response body (`error` / `error_description` fields) is now included in the thrown message, matching the existing `refreshOpenAIOpenAI codeToken` behaviour.
206
+
207
+ ### Added
208
+
209
+ - Added `ChatGPT Plus/Pro (OpenAI code, headless/device)` (`openai-code-device`) as an alternative login method for the OpenAI code provider. Uses OpenAI's device-code flow (`/api/accounts/deviceauth/usercode` → poll `/api/accounts/deviceauth/token`), which avoids a local callback server and port 1455 entirely. Credentials are stored under the existing `openai-code` provider key so all models and tooling continue to work without reconfiguration ([#1277](https://github.com/jaybeyond/sayknow-cli/issues/1277)).
210
+
211
+ ## [15.2.2] - 2026-05-22
212
+
213
+ ### Fixed
214
+
215
+ - Fixed `gemini-3.1-pro-high` and `gemini-3.1-pro-low` on the `google-antigravity` provider always returning HTTP 400 from Cloud Code Assist. The `ANTIGRAVITY_SYSTEM_INSTRUCTION` identity header was not injected for these models because the internal check matched the string `"gemini-3-pro-high"` (hyphen) instead of the versioned `"gemini-3.1-pro-..."` form. The guard now matches all `gemini-3` model variants ([#1274](https://github.com/jaybeyond/sayknow-cli/issues/1274)).
216
+
217
+ ## [15.2.0] - 2026-05-21
218
+
219
+ ### Fixed
220
+
221
+ - Fixed `/login` (and `/logout`, plus any `AuthStorage.set` / `remove` call) against a remote auth-broker throwing `RemoteAuthCredentialStore is read-only on the client. Use 'skc auth-broker login <provider>' to mutate credentials.` Added three optional async write hooks to `AuthCredentialStore` (`upsertAuthCredentialRemote`, `replaceAuthCredentialsRemote`, `deleteAuthCredentialsRemote`); `RemoteAuthCredentialStore` implements them via the broker's `POST /v1/credential` and `POST /v1/credential/:id/disable` endpoints and applies the broker's authoritative post-write entries to the local snapshot. `AuthStorage` routes through the hooks when present, so OAuth and API-key logins (and logouts) initiated from a broker-backed client now persist server-side and surface immediately without waiting for the long-poll snapshot tick.
222
+
223
+ ## [15.1.9] - 2026-05-21
224
+
225
+ ### Fixed
226
+
227
+ - Fixed Ollama named tool forcing to send only the requested tool when the caller passes a named `toolChoice`, preserving `tool_choice: "required"` while preventing local models from selecting a different tool. ([#1236](https://github.com/jaybeyond/sayknow-cli/issues/1236))
228
+ - Fixed `/btw` (and IRC background replies) returning a `BedrockException` 400 (`The toolConfig field must be defined when using toolUse and toolResult content blocks.`) on LiteLLM → Bedrock once the session has tool-call history. Two source fixes in `buildParams`: (1) `if (context.tools)` → `if (context.tools?.length)` so an explicit `context.tools = []` (the /btw opt-out) never routes through `convertTools` and never emits an empty `"tools"` array; (2) `else if (hasToolHistory(...))` → `else if (context.tools === undefined && hasToolHistory(...))` so the Anthropic-proxy sentinel that injects `tools: []` for tool-history turns is suppressed when the caller explicitly opted out, preventing it from re-introducing the empty array. As defence-in-depth, `tool_choice: "none"` is also dropped when the resolved tools list is missing or empty. ([#1227](https://github.com/jaybeyond/sayknow-cli/issues/1227))
229
+
230
+ ## [15.1.8] - 2026-05-20
231
+ ### Added
232
+
233
+ - Added Fireworks Fire Pass as a separate `firepass` provider with API-key login flow, bundled `kimi-k2.6-turbo` model entry (Kimi K2.6 Turbo), and wire-id translation from the friendly catalog id to the `accounts/fireworks/routers/kimi-k2p6-turbo` router endpoint. Fire Pass keys (`fpk_…`) authorize only the dedicated router and reject `/v1/models`, so login validation pings chat completions against the router id directly. Extended the openai-completions Kimi-family safety net so the firepass entry inherits the per-Fireworks-docs "always send `max_tokens`" default ([Kimi K2 guide](https://docs.fireworks.ai/models/kimi-k2)); the router's accepted `reasoning_effort` set includes `xhigh`, so it is forwarded verbatim rather than remapped. See https://docs.fireworks.ai/firepass.
234
+
235
+ ### Fixed
236
+
237
+ - Fixed DeepSeek V4 direct API requests with tools to keep documented thinking mode instead of dropping reasoning: lower SKC efforts now map to DeepSeek's supported `high`, `tool_choice` is omitted, `thinking: { type: "enabled" }` and `max_tokens` are sent, and partial user `reasoningEffortMap` overrides merge with DeepSeek defaults. ([#1207](https://github.com/jaybeyond/sayknow-cli/issues/1207))
238
+ - Fixed model cache schema v2 databases so offline refreshes preserve cached provider discoveries after upgrading to schema v3 and subsequent online refreshes can overwrite the cache. ([#1219](https://github.com/jaybeyond/sayknow-cli/issues/1219))
239
+ - Fixed Perplexity OAuth credentials being treated as expired one hour after login. `getJwtExpiry` was fabricating `expires = now + 1h` whenever the JWT had no `exp` claim (the common case — Perplexity sessions are server-side). Once the hour elapsed, `getOAuthApiKey` would mark the cred expired and the search provider's loader would silently skip it, surfacing as "logged out". Logins with no `exp` now persist a far-future sentinel; `getOAuthApiKey` also normalizes any stale `expires` written by older builds.
240
+
241
+ ## [15.1.7] - 2026-05-19
242
+ ### Added
243
+
244
+ - Added Anthropic realization of `serviceTier: "priority"`. The anthropic-messages provider now sets `speed: "fast"` on the request and appends the `fast-mode-2026-02-01` beta to `Anthropic-Beta` whenever the caller passes `serviceTier: "priority"`. When the server rejects an unsupported model with `invalid_request_error`, the provider transparently retries the same turn without the fast-mode signal (mirroring the strict-tools fallback pattern), persists the disable via a new `providerSessionState.fastModeDisabled` flag so subsequent requests in the session skip the field, and surfaces the action via the new `AssistantMessage.disabledFeatures` array (id `"priority"`) so callers can sync user-facing toggles. A new `clearAnthropicFastModeFallback(providerSessionState)` helper lets callers re-arm priority after the auto-fallback fired.
245
+ - Added scoped `ServiceTier` values: `"openai-only"` (priority on `openai`/`openai-code`, ignored elsewhere) and `"anthropic-model-only"` (priority on direct `anthropic`, ignored on Bedrock/Vertex Anthropic model and elsewhere). A new `resolveServiceTier(serviceTier, provider)` helper computes the effective tier for the provider; existing OpenAI/Anthropic provider code routes through it, so `service_tier` and Anthropic fast-mode emission both respect scope. `getPriorityPremiumRequests` now counts Anthropic+priority as one premium request (previously zero) and continues to ignore providers that drop the field on the wire.
246
+
247
+ ### Fixed
248
+
249
+ - Fixed Anthropic fast mode (`serviceTier: "priority"`) looping on 429 `rate_limit_error: "Extra usage is required for fast mode."` for accounts without the extra-usage entitlement. `isAnthropicFastModeUnsupportedError` now matches the 429 phrasing in addition to the 400 `invalid_request_error` "does not support the `speed` parameter" case, so the provider drops `speed: "fast"` on the in-turn retry, sets `providerSessionState.fastModeDisabled` for the remainder of the session, and surfaces `disabledFeatures: ["priority"]` to the caller instead of retrying with the same payload until `PROVIDER_MAX_RETRIES` is exhausted.
250
+ - Fixed MiniMax Coding Plan CN streaming `<think>...</think>` reasoning as visible assistant text. The OpenAI-compatible stream parser now enables the existing MiniMax tag parser for both `minimax-code` and `minimax-code-cn`, so CN responses become structured `thinking` blocks instead of raw text. ([#1203](https://github.com/jaybeyond/sayknow-cli/issues/1203))
251
+
252
+ ## [15.1.6] - 2026-05-19
253
+
254
+ ### Fixed
255
+
256
+ - Fixed `{}` (empty JSON Schema, the wire representation of `z.unknown()`) being passed verbatim to grammar-constrained samplers (llama.cpp, etc.) in `additionalProperties`, `items`, and other schema-valued positions across **every provider** (OpenAI, Anthropic, Google, Ollama, Bedrock, Cursor). Grammar builders treat `{}` as "generate an empty object" rather than "any JSON value", causing open-typed fields (e.g. `extra.title` from `z.record(z.string(), z.unknown())`) to always emit `{}` instead of the intended string/number/etc. `toolWireSchema` now applies a new `normalizeEmptySchemas` pass (exported) to both the Zod and TypeBox/raw-JSON-Schema branches, converting `{}` → `true` (semantically identical per JSON Schema draft 2020-12 §4.3.1) in all schema-valued positions. Strict-mode opt-out is preserved across all providers: OpenAI's `hasUnrepresentableStrictObjectMap` hits the `=== true` branch instead of the `isJsonObject({})` branch (same result); Anthropic's `normalizeAnthropicStrictSchemaNode` opts out via `additionalProperties !== false` (still true for `true`); Google's `normalizeSchemaForGoogle` strips `additionalProperties` regardless (pre-existing). ([#1179](https://github.com/jaybeyond/sayknow-cli/issues/1179))
257
+ - Fixed `pi-ai login <provider>` crashing with `Unknown provider` for providers that only the `auth-storage` `login()` switch knew about (perplexity, alibaba-coding-plan, gitlab-duo, huggingface, opencode-zen/go, lm-studio, ollama, cerebras, fireworks, qianfan, synthetic, venice, litellm, moonshot, together, cloudflare/vercel ai gateways, vllm, qwen-portal, nvidia, xiaomi, and any custom OAuth provider). The CLI now delegates to `SqliteAuthCredentialStore.login()` instead of duplicating a smaller switch, so the auth-broker `skc auth-broker login <provider>` flow works for every registered OAuth provider.
258
+
259
+ ## [15.1.4] - 2026-05-19
260
+ ### Changed
261
+
262
+ - Updated auth-gateway format and pi-native request handling to invalidate the failed API key and retry the provider request with a replacement key when authentication fails
263
+
264
+ ### Fixed
265
+
266
+ - Fixed OpenCode-Go and OpenCode-Zen chat-completions replay to omit stored reasoning fields on Kimi assistant tool-call messages, avoiding provider 400s for rejected `messages[].reasoning` payloads. ([#1157](https://github.com/jaybeyond/sayknow-cli/issues/1157))
267
+ - Fixed OpenAI Responses and OpenAI code tool schema normalization to emit `properties: {}` for no-argument object schemas without rewriting literal payloads. ([#1147](https://github.com/jaybeyond/sayknow-cli/issues/1147))
268
+ - Fixed Anthropic 400 (`unexpected tool_use_id found in tool_result blocks ... Each tool_result block must have a corresponding tool_use block in the previous message`) when handoff/compaction folds an assistant `tool_use` into the handoff summary string but leaves the matching user-side `tool_result` message in the history. `transformMessages` now indexes every `tool_use` id surviving the first pass and drops orphan `tool_result` messages whose originator was compacted away, preserving the text payload as a user-level `<stale-tool-result>` note so the model still sees what the tool returned. The note is emitted with `role: "user"` rather than `role: "developer"` so providers that elevate developer-role messages (Ollama: `developer` → `system`; OpenAI chat-completions reasoning models: `developer` → `developer`) cannot lift stale tool output to an instruction-priority tier above the surrounding user/developer messages.
269
+ - Fixed streaming authentication retry to trigger when a provider emits a 401 `error` event after a `start` event but before any replay-unsafe content is emitted
270
+ - Added `credential_process` support to the Bedrock provider's AWS credential resolver so profiles delegating to external brokers (`aws-vault`, `granted`, in-house tools) resolve instead of falling through to `Unable to resolve AWS credentials`. Parses the AWS SDK `Version: 1` JSON envelope, honors `Expiration` in the per-profile cache, propagates `AbortSignal` to the spawned helper, routes Windows `.cmd`/`.bat` helpers through `cmd.exe /c`, and ships a POSIX-shell-style tokenizer that preserves backslashes inside double quotes so Windows paths survive ([#1142](https://github.com/jaybeyond/sayknow-cli/issues/1142))
271
+
272
+ ## [15.1.3] - 2026-05-17
273
+ ### Breaking Changes
274
+
275
+ - Changed `AuthBrokerClient.fetchSnapshot()` to return status-based results (`200` or `304`) instead of always returning a raw snapshot body, so callers now need to branch on `status`
276
+ - Renamed public schema utilities in `@sayknow-cli/ai/utils/schema` by replacing `sanitizeSchemaForGoogle`, `sanitizeSchemaForCCA`, `prepareSchemaForCCA`, and `sanitizeSchemaForMCP` with `normalizeSchemaForGoogle`, `normalizeSchemaForCCA`, and `normalizeSchemaForMCP`
277
+ - Added MCP schema normalization via `normalizeSchemaForMCP` for compatibility checks
278
+ - Removed the `StringEnum` helper from `@sayknow-cli/ai/utils/schema`. Use `z.enum([...])` directly; Zod's emitted JSON Schema is already wire-compatible with Google and other providers.
279
+ - Renamed the concrete SQLite credential store class from `AuthCredentialStore` to `SqliteAuthCredentialStore`. `AuthCredentialStore` is now the persistence interface implemented by both the SQLite store and the new `RemoteAuthCredentialStore`. Update `new AuthCredentialStore(db)` / `AuthCredentialStore.open(...)` call-sites to `SqliteAuthCredentialStore`; type-position uses (`store: AuthCredentialStore`) continue to work unchanged.
280
+
281
+ ### Added
282
+
283
+ - Added `onAuthError` to `StreamOptions` and wired `streamSimple()` to retry once with a replacement API key when the first provider response is a 401 before any assistant events are emitted
284
+ - Added generation-aware snapshot metadata (`generation`, `serverNowMs`, `refresher`, and `rotatesInMs`) to auth-broker snapshot responses to support client-side credential-rotation planning
285
+ - Added `transport: "pi-native"` on `Model` and the matching `streamPiNative` client. When `model.transport === "pi-native"`, `streamSimple` short-circuits the per-provider dispatch and POSTs the canonical `Context` to the auth-gateway's `POST /v1/pi/stream` endpoint. The response is SSE-framed `AssistantMessageEvent`s parsed by `readSseJson` and pushed verbatim into the local `AssistantMessageEventStream` — no wire-format translation, no partial-stripping reconstruction. Used by containerized skc installs (roboskc slots, swarm extension, etc.) to route every LLM call through a credential-holding sidecar; the slot itself never sees the real provider tokens. Server-controlled fields (`apiKey`, `signal`, `fetch`, lifecycle callbacks, the provider-session map) are stripped from the wire body — `apiKey` rides in the `Authorization` header as the gateway bearer.
286
+ - Added `POST /v1/pi/stream` to the auth-gateway. Same auth + abort + model-resolution + openai-code-compat + prefix-cache plumbing as the foreign-wire routes; only the wire-format translation is skipped. Request body is `{ modelId, context, options?, stream? }` where `context` is the canonical pi-ai `Context` and `options` is `SimpleStreamOptions` with non-serializable fields stripped. Response is SSE-framed `AssistantMessageEvent` (terminated by `data: [DONE]`) when streaming, or `{ message: AssistantMessage }` JSON when `stream: false`.
287
+ - Added Vertex AI authentication via Google Application Default Credentials from `GOOGLE_APPLICATION_CREDENTIALS`, `~/.config/gcloud/application_default_credentials.json`, or metadata server tokens, with token caching and refresh skew control via `GOOGLE_VERTEX_REFRESH_SKEW_MS`
288
+ - Added support for Anthropic image message parts with `type: "url"` and `type: "file"` sources
289
+ - Added `stopSequences` and `frequencyPenalty` to shared stream options and wired them through to OpenAI request translation
290
+ - Added optional request cancellation support to auth-broker interactions by propagating `AbortSignal` into health, snapshot, usage, and refresh calls
291
+ - Added `AuthStorage.setConfigApiKey` / `removeConfigApiKey` / `clearConfigApiKeys` for config-sourced per-provider bearers (e.g. `models.yml` `providers.<name>.apiKey`). The new tier sits between runtime `--api-key` and stored credentials in `getApiKey`/`peekApiKey` resolution, so a bearer pinned in config now beats the broker's OAuth access token. Also suppresses OAuth `account_uuid` attribution when active, since outbound auth is the explicit config bearer, not OAuth. `describeCredentialSource` reports `"config override (models.yml)"` for visibility.
292
+ - Added per-model `additional_rate_limits` parsing to `openaiOpenAI codeUsageProvider`. The OpenAI code `wham/usage` endpoint surfaces a separate `GPT-5.3-OpenAI code-Spark` rate limit (`metered_feature: openai-code_bengalfox`) on Pro accounts; these now emit dedicated `openai-code:spark:{primary,secondary}` `UsageLimit` entries with `scope.tier = "spark"`, mirroring how Anthropic exposes `anthropic:7d:sonnet` separately from the umbrella `anthropic:7d` bucket. The osx-widgets client already keyed spark detection off `limit.id.includes("spark")`; this populates that contract end-to-end.
293
+ - Added `GET /v1/usage` to the auth-broker API to expose aggregated usage reports from `AuthStorage.fetchUsageReports`
294
+ - Added auth-broker usage polling response handling that returns normalized usage reports plus generation timestamp for clients (5-min per-credential cache via `AuthStorage`)
295
+ - Added the auth-broker subsystem (`@sayknow-cli/ai/auth-broker`) for sharing OAuth credentials across machines without leaking refresh tokens.
296
+ - `startAuthBroker(...)` boots a `Bun.serve` HTTP server exposing `GET /v1/healthz`, `GET /v1/snapshot`, `POST /v1/credential` (upsert), `POST /v1/credential/:id/refresh`, and `POST /v1/credential/:id/disable`.
297
+ - `AuthBrokerClient` is the matching HTTP client used by remote clients.
298
+ - `RemoteAuthCredentialStore` is a client-side `AuthCredentialStore` that mirrors a broker snapshot in memory; mutating methods (`replace*`, `upsert*`, `delete*ForProvider`) throw because writes are server-side only.
299
+ - `AuthBrokerRefresher` is the background refresh loop that pre-refreshes credentials within `refreshSkewMs` and disables on definitive failure (`invalid_grant` / non-network 401-403).
300
+ - Added `AuthStorage.exportSnapshot()`, `AuthStorage.upsertCredential(provider, credential)`, `AuthStorage.forceRefreshCredentialById(id)`, and `AuthStorage.disableCredentialById(id, cause)` public methods consumed by the auth-broker server.
301
+ - Added `AuthStorageOptions.refreshOAuthCredential` override so a remote-store client can route every OAuth refresh through the broker instead of the local OAuth endpoint.
302
+ - Added `REMOTE_REFRESH_SENTINEL` (`"__remote__"`) — the wire placeholder substituted for OAuth refresh tokens in broker snapshots; clients never see the real refresh token.
303
+ - Exposed the OAuth provider catalog (`getOAuthProviders`, `OAuthProvider`, `OAuthProviderInfo`) and `refreshOAuthToken` through the package barrel so the coding-agent CLI can target them without reaching into `utils/oauth`.
304
+ - Added the auth-gateway subsystem (`@sayknow-cli/ai/auth-gateway`) — a forward-proxy that sits between unauthenticated clients (the macOS usage widget, llm-git, roboskc containers, …) and the broker. Clients send standard provider-format requests; the gateway parses them into skc's canonical `Context`, dispatches through pi-ai's `streamSimple()`, and translates the canonical event stream back to the matching wire format. `Authorization` is injected server-side so access tokens never leave the gateway host. Wire surface:
305
+ - `GET /healthz` — unauth liveness.
306
+ - `GET /v1/usage` — aggregated provider usage; 5-min per-credential cache via `AuthStorage.fetchUsageReports`.
307
+ - `GET /v1/models` — model catalog (scoped to providers with credentials).
308
+ - `POST /v1/chat/completions` — OpenAI chat-completions in/out.
309
+ - `POST /v1/messages` — Anthropic messages in/out (text + thinking + tool_use blocks, SSE event taxonomy preserved).
310
+ - `POST /v1/responses` — OpenAI Responses in/out (reasoning items + function_call output items, SSE pass-through).
311
+ - Added exports from `@sayknow-cli/ai/auth-gateway`: `startAuthGateway`, `AuthGatewayServerOptions`, `AuthGatewayBootOptions`, `AuthGatewayServerHandle`, `ModelResolver`, `DEFAULT_AUTH_GATEWAY_BIND`. Per-format `parseRequest` / `encodeResponse` / `encodeStream` triples are reachable via the `./providers/*` subpath as `openai-chat-server`, `anthropic-messages-server`, and `openai-responses-server`.
312
+ - Added `listProvidersWithEnvKey()` to enumerate every provider with an env-var fallback (used by the new migrate command in coding-agent).
313
+
314
+ ### Changed
315
+
316
+ - Changed `GET /v1/snapshot` to support generation-based polling with `If-None-Match` and `wait` for long-poll updates and to return `304` when no snapshot changes are available
317
+ - Changed Bedrock credential resolution for streaming calls to prefer environment keys, AWS profile/SSO credentials, and IMDSv2 fallback when available
318
+ - Changed auth-gateway parsing for OpenAI chat-completions and Responses to ignore unsupported SDK-only fields instead of rejecting requests
319
+ - Changed auth-gateway protocol handling to include CORS headers on responses and support browser-origin requests
320
+ - Changed prompt-cache handling to resolve cache keys from request metadata and headers and preserve them through protocol translation
321
+ - Changed Anthropic messages parsing to forward request `metadata` through to downstream execution
322
+ - Changed usage report caching to use a 5-minute per-credential TTL with jittered refresh timing to reduce usage endpoint rate-limit collisions
323
+ - Changed usage polling failure handling so transient errors continue serving the last known report instead of returning null and dropping the credential from usage aggregates after cache expiry
324
+ - Changed `sanitizeSchemaForGoogle` to normalize snake_case schema keys (such as `any_of` and `additional_properties`) to camelCase and auto-generate `propertyOrdering` for multi-property objects
325
+ - Changed strict-mode sanitization to resolve `$ref` nodes with sibling keys by inlining and merging referenced local definitions
326
+ - Changed strict-mode sanitization to flatten single-entry `allOf` nodes and remove the `allOf` wrapper
327
+ - Changed Anthropic tool schema normalization to preserve supported metadata keywords such as `$ref`, `$defs`, `$schema`, `enum`, `const`, `default`, `title`, and `nullable` instead of stripping them
328
+ - Changed string schema processing to retain only supported `format` values (`date-time`, `time`, `date`, `duration`, `email`, `hostname`, `uri`, `ipv4`, `ipv6`, `uuid`) and demote unsupported `format` values to `description` hints
329
+
330
+ ### Fixed
331
+
332
+ - Fixed OAuth credential refresh flow so concurrent manual and background refreshes now share one in-flight attempt per credential, and `RemoteAuthCredentialStore` now re-synchronizes before using near-expiring OAuth credentials
333
+ - Fixed stale-credential handling after auth failures by waiting for updated broker snapshots and refreshing suspect credentials through broker endpoints before continuing
334
+ - Fixed Google Generative AI startup behavior to throw a clear API-key-required error when no key is configured
335
+ - Fixed AWS Bedrock image message serialization to preserve base64 `source.bytes` payloads instead of decoding and rebuilding them
336
+ - Fixed Google provider error handling to extract the API-reported `error.message` from JSON response bodies when available
337
+ - Fixed `RemoteAuthCredentialStore.getUsageReport` to return the matching credential-specific usage report and coalesce parallel callers into one broker `/v1/usage` fetch
338
+ - Fixed auth-broker credential upload validation to reject the remote refresh-token sentinel and prevent storing a non-refresh value
339
+ - Fixed OpenAI Responses streaming output to emit `reasoning_summary_text` events and parse/send `summary_text` reasoning payloads
340
+ - Fixed Anthropic stop-sequence handling by trimming requests to the API limit of four entries before forwarding
341
+ - Fixed prompt caching behavior across protocol translations so cached-token usage is preserved when Anthropic and OpenAI requests are routed through each other
342
+ - Fixed Anthropic model usage fetching to retry transient `429` and `5xx` responses with exponential backoff, respecting `Retry-After` before returning failure
343
+ - Fixed auth-gateway request translation to preserve OpenAI Responses string/system message content, reasoning replay payloads, completed item text in stream item-done events, Anthropic tool-result ordering, and OpenAI Chat/Responses cached-token usage totals
344
+ - Fixed auth-gateway failure handling so unsupported request controls, upstream terminal errors, non-streaming aborts, and already-aborted client requests fail explicitly instead of being accepted, ignored, or encoded as successful HTTP 200 responses
345
+ - Fixed Gemini CLI / Antigravity tool schema normalization to run the full Cloud Code Assist pipeline, matching shared Google schema handling for union/object merging and nullable extraction
346
+ - Fixed stripped validation hints to be preserved as description spill text (`{key: value}` blocks) when `normalizeSchemaForGoogle` and `normalizeSchemaForCCA` drop unsupported schema keywords
347
+ - Fixed `sanitizeSchemaForGoogle` to collapse nullability forms (`type:'null'` and null-bearing `anyOf` variants) into `nullable` while preserving remaining variants
348
+ - Fixed `sanitizeSchemaForGoogle` to inline local `$defs` references instead of dropping `$ref`/`$defs` structure during Google schema sanitization
349
+ - Fixed `normalizeAnthropicToolSchema` to handle self-referential schemas without infinite recursion
350
+ - Fixed object schema normalization so explicit open-map declarations (`additionalProperties: true` and schema-valued `additionalProperties`) are preserved instead of being converted to closed objects
351
+ - Fixed unsupported schema constraints on arrays and strings (`maxItems`, `uniqueItems`, `pattern`, `minLength`, `maxLength`, and `minItems` when greater than 1) by demoting them into `description` rather than dropping them
352
+
353
+ ### Security
354
+
355
+ - Hardened auth-gateway bearer-token checks with constant-time comparison to avoid timing-side-channel leaks
356
+
357
+ ## [15.1.2] - 2026-05-15
358
+ ### Breaking Changes
359
+
360
+ - Rejected draft-07 tuple and dependency keywords (`items` arrays, `dependencies`, `additionalItems`) in JSON Schema validation
361
+
362
+ ### Added
363
+
364
+ - Added `responseHeaders`, `responseStatus`, and `responseRequestId` fields to `MockResponse` so mock providers can provide synthetic `ProviderResponseMetadata`
365
+ - Added `onResponse` metadata emission for mocks that sends lowercased headers and a default status of 200 before streaming when response headers are configured
366
+ - Added recursive strict-mode sanitization for array `prefixItems` entries so tuple schemas now enforce object constraints per item
367
+
368
+ ### Changed
369
+
370
+ - Normalized legacy draft-07 JSON Schema constructs used in tool parameters (`items` arrays, `additionalItems`, `definitions`, `dependencies`) to draft 2020-12 before OpenAI/Google/CCA sanitization, wire conversion, and argument validation
371
+ - Reworked OpenAI response schema adaptation to rewrite `oneOf` into `anyOf` while preserving existing `anyOf` branches
372
+ - Changed tuple array validation to validate per-index schemas from `prefixItems` and apply `items` only to remaining elements
373
+
374
+ ### Fixed
375
+
376
+ - Fixed validation of plain JSON Schema tool arguments that omitted a `$schema` URI so draft-07-shaped schemas now pass validation instead of being rejected
377
+ - Fixed tuple-array validation for legacy JSON Schema tool schemas to enforce `additionalItems: false` and per-position constraints after automatic draft upgrade
378
+ - Fixed Anthropic tool schema normalization to recurse into `prefixItems` so unsupported constraints inside tuple items are stripped in the generated input schema
379
+ - Fixed Anthropic tool-schema normalization stripping the body of explicit open `additionalProperties` (e.g. Zod's `z.record(z.string(), z.unknown())` compiling to `additionalProperties: {}`) by unconditionally overwriting it with `false`, which closed record-style fields and prevented models from supplying any key. The coding-agent's `resolve` tool exposes plan-approval titles via such a field, so Kimi K2 (and any other Anthropic-shaped provider) could not pass `extra: { title }`, blocking plan mode entirely ([#1104](https://github.com/jaybeyond/sayknow-cli/issues/1104))
380
+ - Fixed Anthropic strict tool planning to leave tools with open `additionalProperties` maps non-strict instead of sending schemas Anthropic rejects.
381
+
382
+ ## [15.1.0] - 2026-05-15
383
+
384
+ ### Breaking Changes
385
+
386
+ - Removed TypeBox root exports (`Type`, `Static`, and `TSchema`) from the package entrypoint, so callers importing those symbols from `@sayknow-cli/ai` must migrate to `zod` or `@sayknow-cli/ai/types`
387
+
388
+ ### Added
389
+
390
+ - Added support for defining tool schemas with Zod (`z.object`, `z.string`, etc.) by allowing `Tool.parameters` to be either Zod schemas or legacy JSON Schema objects and converting them to provider wire format automatically
391
+ - Added package-level schema helpers in the `zod/v4` style by exporting `z` and `ZodType` from the root entrypoint
392
+ - Added a `mock` API provider via `createMockModel` to build `Model<"mock">` instances for fully in-memory, deterministic assistant streams in tests
393
+ - Added `streamMock` and `registerMockApi` so mock responses can be consumed through `stream()` and the global custom API registry without an external model backend
394
+ - Added async/sync response scripting with optional context-based handlers, and new `push()`/`reset()` controls to drive multi-turn mock interactions and inspect per-call invocation state
395
+ - Added support in mock responses for simulating tool calls, usage metadata, custom stop reasons, delayed emissions, and terminal error/aborted outcomes
396
+
397
+ ### Changed
398
+
399
+ - Changed Azure OpenAI Responses tool schema conversion to sanitize tool parameter schemas and rewrite `oneOf` branches as `anyOf` so tool calls remain compatible with Azure's schema expectations
400
+ - Changed `Static<S>` to extract a schema object’s `static` type when present, improving inferred tool argument types for non-Zod parameter definitions
401
+ - Changed `Static` typing behavior so it now infers argument types from Zod schemas and defaults to `unknown` for non-Zod JSON Schema parameter definitions
402
+ - Restored the default steady-state stream idle timeout to 120s (regressed in 15.0.0). 30s was too aggressive for reasoning models, slow proxies, and tool-call planning gaps, surfacing as repeated `Provider stream stalled while waiting for the next event` errors. Existing `PI_STREAM_IDLE_TIMEOUT_MS` / `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` overrides are unchanged.
403
+
404
+ ### Fixed
405
+
406
+ - Preserved top-level unknown fields in validated tool-call arguments so extra root properties are retained after schema coercion
407
+ - Fixed coercion for Zod `record` fields by parsing JSON-stringified record arguments into objects
408
+ - Validated legacy draft-07 JSON Schema tool parameters directly instead of converting through Zod, improving support for features like `$ref`, `definitions`, `nullable`, and `uniqueItems`
409
+ - Fixed Cloud Code Assist schema preparation to strip unsupported `propertyNames` and fall back to a minimal tool schema when schema meta-validation detects malformed keywords
410
+ - Fixed OpenAI Completions streaming to avoid treating non-output chunks (including role-only preambles) as progress events so idle-timeout watchdog behavior no longer hangs on no-op streamed chunks
411
+ - Fixed Cloud Code Assist schema compatibility checks by replacing strict AJV meta-schema validation with structural JSON Schema validation to avoid rejecting structurally valid tool schemas
412
+ - Fixed lazy built-in provider streams (`anthropic-messages`, `bedrock-converse-stream`, `cursor-agent`, `google-*`, `ollama-chat`, `openai-*`) prematurely aborting slow first-token responses with `Provider stream stalled while waiting for the next event`. The lazy-stream watchdog wrapper was treating the synthetic `start` event (yielded immediately by every provider before the model emits any tokens) as the first real item, which caused the watchdog to drop from `firstItemTimeoutMs` (100s) to `idleTimeoutMs` (30s) before the upstream model had produced anything. The shared `iterateWithIdleTimeout` now keeps `awaitingFirstItem` true until a real progress item arrives, and the lazy-stream wrapper marks `start` as a non-progress keepalive ([#1073](https://github.com/jaybeyond/sayknow-cli/pull/1073) regression).
413
+ - Heal leaked Kimi K2 chat-template tool-call tokens (`<|tool_calls_section_begin|>` … `<|tool_call_argument_begin|>` … `<|tool_calls_section_end|>`) that some hosts (native `kimi-code` API, OpenRouter, Fireworks, etc.) emit into `delta.content` instead of structured `tool_calls`. The OpenAI-completions stream consumer now strips the markers from visible text, reconstructs the embedded calls as proper `toolCall` content blocks (stream-aware, token-boundary-safe), and promotes `finish_reason: stop` to `toolUse` when calls were healed.
414
+ - Fixed OpenAI-completions Kimi K2 healed-call promotion clobbering non-stop terminal finish reasons (`error`, `length`, `aborted`); promotion now only fires when the prior stop reason is the natural-completion `stop`
415
+ - Fixed OpenAI-completions duplicate Kimi tool calls when a single chunk delivers both leaked markers and a structured `delta.tool_calls`; the healer now strips visible markers but discards its synthesized calls so structured payloads remain the single source of truth
416
+ - Fixed Kimi tool-call healer synthesizing a bogus empty call when assistant text mentions a literal `<|tool_call_end|>` (or `<|tool_call_begin|>` / `<|tool_call_argument_begin|>`) outside an active `<|tool_calls_section_begin|>…<|tool_calls_section_end|>` section; the tokens now survive as text
417
+ - Fixed OpenAI-completions ignoring per-request `StreamOptions.streamFirstEventTimeoutMs` when configuring the underlying OpenAI SDK HTTP timeout, causing slow-before-headers providers to be aborted at the env default before the wrapping watchdog armed
418
+ - Fixed JSON Schema validator silently accepting values that violate `propertyNames`, `patternProperties`, `dependentRequired`, `dependencies`, `if`/`then`/`else`, `contains`, and `prefixItems`; the in-tree validator now enforces these keywords instead of falling through. `unevaluatedProperties`/`unevaluatedItems` remain permissive but log a one-time warning so tool authors are not surprised.
419
+ - Fixed recursive `$ref` schemas being treated as universally valid: the validator previously short-circuited on the second occurrence of any ref it had already seen, so nested values violating the referenced sub-schema passed. Cycle detection now keys on (ref, value-identity) pairs with a depth cap for primitive values, so genuine sub-tree violations are still caught.
420
+ - Fixed JSON Schema meta-validator accepting malformed `if`/`then`/`else` and `dependencies` keywords; each conditional sub-schema is now structurally validated and draft-07 `dependencies` accepts either a schema or a string array of dependent keys.
421
+ - Fixed Zod-emitted wire schemas dropping null-valued unknown root fields before `preserveUnknownRootFields` could snapshot them, so callers like `task.simple` no longer lose a `schema: null` argument and downstream rejection paths fire as intended.
422
+ - Fixed mock provider partial `Usage` to recompute `totalTokens` (and `cost.total` when cost components are supplied) when omitted, instead of reporting 0
423
+ - Fixed mock provider auto-generated tool-call IDs to use a per-instance counter (now reset by `reset()`), so test order no longer affects IDs across `createMockModel()` instances
424
+
425
+ ## [15.0.2] - 2026-05-15
426
+ ### Fixed
427
+
428
+ - Fixed `StreamOptions.fetch` typing to accept fetch-compatible override functions that do not expose `preconnect`, allowing custom fetch implementations to be used without type errors across runtimes
429
+ - Fixed Moonshot Kimi K2.6 forced tool calls to send `thinking: { type: "disabled" }`, avoiding `tool_choice 'specified' is incompatible with thinking enabled` 400s while preserving the requested named tool ([#1077](https://github.com/jaybeyond/sayknow-cli/issues/1077)).
430
+
431
+ ## [15.0.1] - 2026-05-14
432
+ ### Breaking Changes
433
+
434
+ - Increased the minimum Bun runtime version to `>=1.3.14` for the `@aws-?` package
435
+
436
+ ### Added
437
+
438
+ - Added `installH2Fetch` to patch `globalThis.fetch` so HTTPS requests attempt HTTP/2 over ALPN with automatic HTTP/1.1 fallback when HTTP/2 is unsupported
439
+ - Added priority service-tier traffic to the `premiumRequests` accounting on OpenAI and OpenAI code provider providers. Sending `serviceTier: "priority"` now increments `usage.premiumRequests` by 1 per request, matching the existing GitHub Copilot premium-request budget semantics so downstream consumers (e.g. the `skc stats` "Premium Reqs" card and `/usage`) reflect priority traffic alongside Copilot premium calls.
440
+
441
+ ## [15.0.0] - 2026-05-13
442
+
443
+ ### Added
444
+
445
+ - Added `AuthStorage.onCredentialDisabled(listener)` — a multi-subscriber `on/off` API for `credential_disabled` events. Returns an unsubscribe function; calling it more than once is a no-op. Multiple subscribers all receive every disable event, with synchronous and async exceptions isolated per-listener so a misbehaving subscriber cannot starve the rest of the chain. Buffer-and-replay semantics are preserved: events emitted while no listener is subscribed are buffered (FIFO, capped at 32) and replayed once to the listener that triggers the empty→non-empty transition. After every subscriber unsubscribes, subsequent disable events buffer again until the next subscribe.
446
+
447
+ ### Fixed
448
+
449
+ - Fixed OAuth credentials being silently disabled when two skc processes (or any two `AuthStorage` instances sharing a `agent.db`) race on token refresh. Anthropic rotates refresh tokens on every use, so the loser's `invalid_grant` response previously soft-deleted the row that the winner just rotated, forcing the user to `/login` again. `#tryOAuthCredential` now re-reads the row from disk before declaring a definitive failure: if the persisted `refresh` differs from the snapshot it tried, the peer-rotated credential is reloaded and the request retries against the fresh token instead of disabling the live row.
450
+ - Closed a remaining race window in OAuth refresh-failure handling: between re-reading the credential row to check for peer rotation and the subsequent soft-delete, another process could still complete a refresh and rotate the row, leaving us to disable the freshly-rotated credential by `id`. The disable now runs as a single CAS update conditioned on the row's `data` still matching the snapshot we tried to refresh, and on `disabled_cause IS NULL`. If the CAS reports 0 rows changed (peer rotation, or row already disabled by a concurrent failure on the same snapshot), we reload from disk and retry instead of mutating the wrong row or emitting a spurious `credential_disabled` event.
451
+ ### Changed
452
+ - Lowered the default steady-state stream idle timeout from 120s to 30s while preserving the existing environment overrides.
453
+
454
+ ### Fixed
455
+ - Lazy built-in provider streams now enforce the shared idle watchdog and abort stalled provider requests, so session auto-retry can continue after transient network drops instead of remaining stuck. Caller aborts still terminate as aborted.
456
+
457
+ ## [14.9.3] - 2026-05-10
458
+
459
+ ### Fixed
460
+ - Anthropic provider now retries generic transient connect failures (`unable to connect`, `fetch failed`, `connection error`, etc.) by falling back to the shared `isRetryableError` allowlist after the provider-specific patterns. Previously these errors bypassed the hand-curated regex in `isProviderRetryableError` and aborted the stream on the first attempt, while the OpenAI SDK and OpenAI code `fetchWithRetry` paths already handled them.
461
+
462
+ ## [14.9.0] - 2026-05-10
463
+
464
+ ### Added
465
+
466
+ ### Fixed
467
+ - Fixed silent forwarding of image content (for example Python plot output rendered in the terminal) to models without vision support, which produced opaque 404 errors from upstream. Image blocks are now stripped and replaced with a `[image omitted: model does not support vision]` placeholder for non-vision models, including tool-result payloads ([#967](https://github.com/jaybeyond/sayknow-cli/issues/967), [#968](https://github.com/jaybeyond/sayknow-cli/issues/968)).
468
+
469
+ - Added `AuthStorage` `onCredentialDisabled` callback (sync or async) so embedders can react when a credential is automatically disabled (e.g. OAuth refresh fails with `invalid_grant`) — useful for surfacing a banner or auto-launching a re-login flow instead of letting the credential silently disappear. Sync throws and async rejections are both caught and logged so a misbehaving subscriber cannot break the disable path.
470
+ - Added Anthropic OAuth `account.uuid` and `account.email_address` extraction from the `/v1/oauth/token` exchange and refresh responses; both `AnthropicOAuthFlow.exchangeToken()` and `refreshAnthropicToken()` now populate `OAuthCredentials.{accountId, email}` so downstream consumers can attribute requests to the authenticated account without a separate `/api/oauth/profile` round-trip.
471
+ - Added `onSseEvent` stream diagnostics so HTTP SSE providers can expose raw SSE frames without changing parsed model output.
472
+ - Added `streamIdleTimeoutMs` option (and `PI_STREAM_IDLE_TIMEOUT_MS` env override; `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` remains a backward-compatible alias) for a steady-state inter-event watchdog. Set to `0` to disable.
473
+ - Added a semantic-progress predicate to OpenAI Responses and OpenAI code SSE/WebSocket transports so `response.in_progress`-style keepalives no longer reset the idle deadline on stalled tool calls.
474
+
475
+ ### Changed
476
+
477
+ - Anthropic streams now enforce a steady-state idle timeout (defaults to 120s, same control as `PI_STREAM_IDLE_TIMEOUT_MS`) in addition to the first-event watchdog. Long-running responses that go fully silent between events will now surface as `Anthropic stream stalled while waiting for the next event` instead of hanging.
478
+ - Fixed `resolveAnthropicMetadataUserId()` to accept JSON-format `user_id` values that match real Anthropic Code's payload shape (`{ device_id, account_uuid, session_id, ... }` from `services/api/anthropic-model.ts:getAPIMetadata`). Previously only the synthetic `user_<hex>_account_<uuid>_session_<uuid>` cloaking format was accepted on OAuth, which caused stable session-keyed metadata supplied by callers to be discarded and replaced with fresh random entropy on every request — defeating session-count attribution on the Anthropic model OAuth path.
479
+
480
+ ## [14.8.0] - 2026-05-09
481
+
482
+ ### Fixed
483
+ - Fixed Gemini 3 Pro thinking metadata so `medium` effort is rejected with the expected error instead of being silently accepted: `ThinkingConfig` now carries an optional explicit `levels` list that survives `expandEffortRange`, letting non-contiguous supported sets (e.g. `[low, high]`) round-trip through enrichment.
484
+ - Fixed Kimi Code OAuth expiry handling to refresh access tokens 5 minutes before server expiry, avoiding daily 401s from using tokens right up to the cutoff.
485
+ - Fixed OpenAI Responses custom tool replay to preserve custom tool call item IDs with the `ctc_` prefix instead of rewriting them as `fc_` function-call IDs ([#977](https://github.com/jaybeyond/sayknow-cli/issues/977)).
486
+
487
+ ## [14.7.6] - 2026-05-07
488
+
489
+ ### Added
490
+
491
+ - Added `hideThinkingSummary` option to `SimpleStreamOptions`. When true, `streamSimple` requests that the underlying provider omit reasoning/thinking summaries: Anthropic receives `thinking.display = "omitted"` (where supported), and OpenAI Responses / Azure / OpenAI code providers leave `reasoning.summary` unset so the server skips emitting the human-readable summary stream entirely.
492
+
493
+ ### Changed
494
+
495
+ - Changed OpenAI Responses, Azure OpenAI Responses, and OpenAI code provider providers to omit `reasoning.summary` from requests when `reasoningSummary` is explicitly `null` (previously fell back to `"auto"`).
496
+ ## [14.7.5] - 2026-05-07
497
+
498
+ ### Added
499
+
500
+ - Added `OpenAICompat.supportsMultipleSystemMessages` so chat-completions hosts can opt out of separate leading system blocks. Auto-detected as `true` for OpenAI, Azure, OpenRouter, Cerebras, Together, Fireworks, Groq, DeepSeek, Mistral, xAI, Z.ai, GitHub Copilot, and Zenmux; `false` for MiniMax, Alibaba Dashscope, and Qwen Portal whose chat templates reject follow-up system messages. Unknown OpenAI-compatible hosts (custom vLLM/local) default to `false`; users can opt back in via `compat.supportsMultipleSystemMessages: true`.
501
+
502
+ ### Fixed
503
+
504
+ - Fixed strict-template OpenAI-compatible hosts (e.g. Qwen 3.5+ via vLLM, MiniMax) rejecting follow-up `system`/`developer` messages by coalescing ordered system prompts into a single block joined by `\n\n` when `compat.supportsMultipleSystemMessages` is false. Canonical hosts continue to receive separate blocks so KV-cache reuse stays effective when only the trailing prompt changes ([#958](https://github.com/jaybeyond/sayknow-cli/issues/958)).
505
+
506
+ ## [14.7.2] - 2026-05-06
507
+
508
+ ### Fixed
509
+
510
+ - Fixed VLLM model discovery to use `max_model_len` as the context window when the endpoint reports it.
511
+ - Fixed custom Ollama Cloud/local-proxy model aliases (for example `deepseek-v4-pro:cloud`) to inherit bundled cache-pricing metadata when the upstream model is known ([#937](https://github.com/jaybeyond/sayknow-cli/issues/937)).
512
+ - Fixed local Ollama model discovery to apply `/api/show` thinking and vision capabilities in addition to native context windows ([#928](https://github.com/jaybeyond/sayknow-cli/issues/928)).
513
+
514
+ ## [14.7.0] - 2026-05-04
515
+ ### Breaking Changes
516
+
517
+ - Changed `Context.systemPrompt` from a string to `string[]`, so callers must now pass an array of prompts instead of a single string
518
+ - Changed behavior will throw at runtime for non-array system prompts because request builders now normalize system prompts as an array
519
+
520
+ ### Added
521
+
522
+ - Added support for multiple system prompts by changing `Context.systemPrompt` to an ordered string array and preserving provider-appropriate instruction precedence
523
+
524
+ ### Changed
525
+
526
+ - Changed request builders for Anthropic, OpenAI, Bedrock, Azure, Cursor, Google, and Ollama to propagate every non-empty system prompt entry without demoting durable instructions into ordinary conversation turns
527
+
528
+ ### Fixed
529
+
530
+ - Filtered out empty normalized system prompts so blank entries are no longer sent to providers
531
+ - Removed blank system prompt strings from provider payloads to avoid unnecessary empty instruction messages
532
+
533
+ ## [14.6.6] - 2026-05-04
534
+
535
+ ### Added
536
+
537
+ - Added always-on OpenRouter response caching (1h TTL) by sending `X-OpenRouter-Cache: true` and `X-OpenRouter-Cache-TTL: 3600` on every OpenRouter request — identical requests replay from OpenRouter's edge cache for free. https://openrouter.ai/docs/features/response-caching
538
+
539
+ ## [14.6.4] - 2026-05-03
540
+
541
+ ### Fixed
542
+
543
+ - Fixed OpenAI code provider websocket continuations to retry with full context when `previous_response_id` expires server-side instead of surfacing `previous_response_not_found`.
544
+
545
+ ## [14.6.2] - 2026-05-03
546
+ ### Added
547
+
548
+ - Added `EventStream.fail(err)` method to terminate the async iterator with an error, enabling consumers to catch stream-level failures via `for await` without hanging
549
+
550
+ ### Fixed
551
+
552
+ - Fixed OpenAI Responses tool schema conversion to rewrite non-strict `oneOf` unions to `anyOf` before sending tools to the Responses API ([#920](https://github.com/jaybeyond/sayknow-cli/issues/920))
553
+
554
+ ## [14.6.0] - 2026-05-02
555
+
556
+ ### Added
557
+
558
+ - Added `disableReasoning` to stream and OpenAI completion options to force reasoning off for models that support it, sending `reasoning: { enabled: false }` for OpenRouter-compatible requests
559
+ - Added `thinkingDisplay` option to Anthropic options to control whether adaptive and explicit reasoning is returned as `summarized` or `omitted`
560
+ - Added Anthropic model compatibility flags `supportsEagerToolInputStreaming` and `supportsLongCacheRetention` for API-capability-specific request behavior
561
+
562
+ ### Changed
563
+
564
+ - Changed Anthropic request payloads to send `thinking: { type: "disabled" }` when `thinkingEnabled` is explicitly `false` on reasoning-enabled models
565
+ - Changed Anthropic cache retention handling so `cacheRetention: "long"` now uses `ttl: "1h"` only for canonical Anthropic endpoints with long-cache support
566
+ - Changed Anthropic tool schema generation to include `eager_input_streaming` only on models that advertise support
567
+ - Changed Anthropic OAuth login flow to include browser fallback guidance and richer error context when token exchange or refresh fails
568
+
569
+ ### Fixed
570
+
571
+ - Fixed Anthropic non-thinking requests to include the caller-provided `temperature` value in request payloads
572
+ - Fixed Anthropic `anthropic-model-opus-4-7` non-thinking payloads to omit sampling fields (`temperature`, `top_p`, and `top_k`)
573
+ - Fixed OpenAI code provider base URL normalization so configured base URLs with or without `/openai-code` or `/openai-code/responses` now resolve to `/openai-code/responses`
574
+ - Fixed OpenAI code provider websocket handling to parse JSON from non-string message payloads including `ArrayBuffer`, typed arrays, and `Blob` values
575
+ - Fixed OpenAI code provider websocket handshakes to replace stale `openai-beta` values with the websocket beta and avoid sending request-body headers over websocket transport
576
+ - Fixed abort tracking so caller-initiated cancellations are treated as user aborts even after local watchdog timeouts, preventing unintended automatic retries
577
+ - Fixed Anthropic stream handling to parse raw SSE envelopes directly, ignore unrelated events, and repair malformed JSON in SSE payloads
578
+ - Fixed Anthropic streaming to emit an explicit error when the SSE stream ends without a `message_stop` event
579
+ - Fixed OpenAI code provider websocket continuations to send true `previous_response_id` deltas for `store: false` transcripts, expose request stats, and default text verbosity to `low` unless explicitly overridden.
580
+ - Fixed OpenAI code provider websocket append reuse after `response.completed` terminal events.
581
+
582
+ ## [14.5.14] - 2026-05-01
583
+ ### Added
584
+
585
+ - Added package-level `google-gemini-headers` exports (`getGeminiCliHeaders`, `getGeminiCliUserAgent`, `getAntigravityHeaders`, `extractRetryDelay`, and `ANTIGRAVITY_SYSTEM_INSTRUCTION`) for header and retry handling reuse without importing full Google providers
586
+
587
+ ### Changed
588
+
589
+ - Changed package exports and streaming/provider wiring to load heavy Google/Kimi/GitLab/synthetic provider modules lazily through `register-builtins`, reducing startup import overhead from optional provider SDKs
590
+
591
+ ### Fixed
592
+
593
+ - Fixed DeepSeek V4 tool-call follow-up 400 errors from three root causes:
594
+ - Mapped `reasoning_effort` "xhigh" to "max" for DeepSeek-family models on any provider (NVIDIA, OpenCode-Go, etc.), not just `deepseek`
595
+ - Recovered `reasoning_content` from thinking blocks with valid signatures that were filtered by the non-empty-text check
596
+ - Added empty-string fallback when `reasoning_content` is genuinely absent (e.g. proxy-stripped) but the provider requires the field
597
+
598
+ ## [14.5.13] - 2026-05-01
599
+
600
+ ### Breaking Changes
601
+
602
+ - Removed `utils/oauth` re-exports from the package entrypoint, so OAuth helper imports from the root module must be updated
603
+
604
+ ## [14.5.10] - 2026-04-30
605
+
606
+ ### Added
607
+
608
+ - Added provider response metadata callbacks for Anthropic and OpenAI streaming requests.
609
+
610
+ ## [14.5.9] - 2026-04-30
611
+
612
+ ### Added
613
+
614
+ - Added `usage.reasoningTokens` to OpenAI and Google usage output when providers report reasoning/thinking tokens
615
+ - Added `usage.cttl.ephemeral5m` and `usage.cttl.ephemeral1h` to report Anthropic cache-write TTL token buckets
616
+ - Added `usage.server.webSearch` and `usage.server.webFetch` to report Anthropic server tool-call request counts
617
+
618
+ ### Fixed
619
+
620
+ - Fixed OpenAI usage attribution to avoid double-counting `reasoning_tokens` in output totals
621
+ - Fixed Anthropic streaming usage handling so a previously populated cache TTL breakdown is preserved when later events omit `cache_creation`
622
+
623
+ ## [14.5.4] - 2026-04-28
624
+
625
+ ### Changed
626
+
627
+ - Changed OpenAI custom Lark grammar payloads to strip comments and blank lines before sending provider requests.
628
+
629
+ ### Fixed
630
+
631
+ - Fixed OpenAI code provider GPT model pricing by inheriting matching OpenAI catalog rates for zero-priced discovered OpenAI code entries.
632
+
633
+ ## [14.5.3] - 2026-04-27
634
+
635
+ ### Added
636
+
637
+ - Added `fireworks` as a supported provider with API key login flow and credential storage
638
+ - Added Fireworks model catalog support with `fireworks`-scoped openai-completions models `glm-5`, `glm-5.1`, `kimi-k2.5`, `kimi-k2.6`, and `minimax-m2.7`
639
+ - Added built-in discovery wiring so providers with base URL `api.fireworks.ai` are recognized as OpenAI-compatible and can use streaming token control
640
+
641
+ ### Changed
642
+
643
+ - Updated the built-in model catalog to use corrected `contextWindow` and `maxTokens` values for many existing models instead of placeholder limits
644
+ - Updated several model cost entries, including cache-read pricing, to corrected values
645
+
646
+ ### Fixed
647
+
648
+ - Fixed Fireworks request formatting by translating between public model IDs and API wire IDs when sending OpenAI-completions requests
649
+ - Fixed OpenAI-compatible model parameter handling for Fireworks by allowing `max_tokens` to be sent during requests
650
+
651
+ ## [14.5.1] - 2026-04-26
652
+
653
+ ### Fixed
654
+
655
+ - Fixed NVIDIA NIM DeepSeek-V4 models leaking chat-template tool-call markers (e.g. `<|DSML|tool_calls|>`) into visible response text by stripping the special tokens from streamed `delta.content` ([#798](https://github.com/jaybeyond/sayknow-cli/issues/798))
656
+
657
+ ## [14.4.0] - 2026-04-26
658
+
659
+ ### Added
660
+
661
+ - Added an `examples` option to `StringEnum` to include example values in the generated schema
662
+
663
+ ### Changed
664
+
665
+ - Changed Anthropic tool schema generation to strip unsupported schema fields (including `patternProperties`), add `additionalProperties: false` for object types, and apply Anthropic strict-mode limits when marking tools as strict
666
+ - Changed Anthropic strict tool planning to cap strict `tools` at twenty entries and convert excess optional/union parameters to nullable schemas to stay within provider constraints
667
+
668
+ ### Fixed
669
+
670
+ - Fixed Anthropic tool schema compilation failures by keeping the `write` tool out of the strict-tool allowlist when the full coding-agent tool set is active
671
+ - Fixed Anthropic 400 `tools.*.custom: For 'object' type, property 'minItems' is not supported` by stripping `minItems` from object-shaped JSON schema nodes (array nodes still keep supported `minItems` values)
672
+ - Fixed Anthropic tool schemas that used tuple-style arrays by stripping unsupported `maxItems` and only preserving provider-supported `minItems` values
673
+ - Fixed Anthropic and OpenRouter Anthropic tool calls that previously failed with `compiled grammar is too large` by retrying automatically without strict tool schemas and reusing non-strict mode for subsequent requests in the same provider session
674
+ - Fixed parsing of JSON tool arguments containing raw control characters inside string values (such as embedded newlines) by escaping them before JSON parsing
675
+ - Fixed `validateToolArguments` to accept stringified objects and arrays that include literal control characters inside string fields
676
+ - Fixed OpenAI code provider Spark OAuth selection to fall back to non-Pro accounts when no ChatGPT Pro account is connected, so users without a Pro account can still attempt Spark requests in case the server permits access.
677
+
678
+ ## [14.3.0] - 2026-04-25
679
+
680
+ ### Added
681
+
682
+ - Added support for Anthropic model Opus 4.7 (`anthropic-model-opus-4-7`) model ([#726](https://github.com/jaybeyond/sayknow-cli/issues/726))
683
+ - Suppresses sampling parameters (temperature/top_p/top_k) that Opus 4.7 rejects
684
+ - Enables `display: "summarized"` for adaptive thinking to restore visible thinking content
685
+
686
+ ### Fixed
687
+
688
+ - Fixed Cursor provider losing conversation history on follow-up turns (model responding "this appears to be the start of our session") by populating `ConversationStateStructure.rootPromptMessagesJson` with JSON blob IDs for the system prompt plus prior user/assistant/tool-result messages. Cursor's server builds the model prompt from `rootPromptMessagesJson`, not from the protobuf `turns[]` tree, so sending only the system prompt there caused prior turns to be dropped
689
+ - Fixed Cursor provider multi-turn conversations failing with `Connect error internal: Blob not found` on the second message by storing `ConversationStateStructure.turns`, `AgentConversationTurnStructure.user_message`, and `AgentConversationTurnStructure.steps` as content-addressed blob IDs in the KV store (matching the existing handling for `rootPromptMessagesJson`) rather than sending the raw serialized bytes inline ([#678](https://github.com/jaybeyond/sayknow-cli/issues/678))
690
+
691
+ ## [14.2.1] - 2026-04-24
692
+
693
+ ### Fixed
694
+
695
+ - Fixed OpenAI code provider Spark OAuth selection to require a verified ChatGPT Pro account instead of falling back to Plus or unknown-plan accounts.
696
+
697
+ ## [14.2.0] - 2026-04-23
698
+
699
+ ### Added
700
+
701
+ - Added `gpt-5.5` to the built-in model catalog for both OpenAI Responses (`openai`) and local `litellm` (`openai-completions`) providers
702
+ - Added `gpt-image-2` to the `litellm` built-in model catalog
703
+ - Added `isCopilotTransientModelError()` and `callWithCopilotModelRetry()` helpers in `utils/retry` that detect GitHub Copilot's intermittent `HTTP 400 model_not_supported` responses for preview models (`gpt-5.3-openai-code`, `gpt-5.4`, `gpt-5.4-mini`, ...) and retry the request up to three times with backoff. OpenAI Responses, OpenAI Completions, and Anthropic provider paths now participate in this retry when the model is served through Copilot.
704
+ - Added OpenAI Responses custom-tool grammar support for patch-envelope `apply_patch` calls, including freeform streaming, history replay, and forced tool-choice mapping to the custom wire name.
705
+
706
+ ### Changed
707
+
708
+ - Updated built-in model metadata with revised `contextWindow`, `maxTokens`, and pricing values for existing entries
709
+ - Changed generated model policies to assign `applyPatchToolType: "freeform"` for first-party GPT-5 OpenAI Responses and OpenAI code models, so regenerated `models.json` preserves the `apply_patch` custom-tool metadata.
710
+ - Renamed `rewriteCopilotAuthError` to `rewriteCopilotError` and extended it to rewrite `HTTP 400 model_not_supported` after retries are exhausted with guidance about Copilot's OAuth-client-specific rollout gap (see opencode#13313).
711
+
712
+ ### Fixed
713
+
714
+ - Fixed Amazon Bedrock proxy handling to honor lowercase `http_proxy`, `https_proxy`, and `all_proxy` environment variables when using HTTP/1 fallback
715
+ - Fixed Amazon Bedrock streaming behind corporate HTTP proxies by using a proxy-aware HTTP/1 transport when `HTTPS_PROXY`, `HTTP_PROXY`, or `ALL_PROXY` is configured, including AWS SSO credential calls.
716
+ - Fixed Amazon Bedrock requests to retry once with HTTP/1 when the AWS SDK's default HTTP/2 transport fails before streaming begins.
717
+ - Fixed OpenAI Responses streaming to display thinking tokens from local providers (llama.cpp, etc.) that send raw `reasoning_text.delta` events and empty `summary` arrays in `output_item.done`. Previously, thinking content was silently dropped during streaming while non-streaming mode worked correctly.
718
+ - Synced the bundled OpenCode Go catalog with the current docs so `kimi-k2.6`, `mimo-v2.5`, and `mimo-v2.5-pro` appear in offline/default model lists.
719
+
720
+ ## [14.1.3] - 2026-04-17
721
+
722
+ ### Fixed
723
+
724
+ - Preserved user-provided `session_id` and `x-client-request-id` headers in OpenAI Responses requests instead of overriding them with automatic session-derived values
725
+ - Stopped sending `session_id` and `x-client-request-id` headers for OpenAI Responses requests when `cacheRetention` is set to `none`
726
+ - Fixed direct OpenAI Responses requests to send `session_id` and `x-client-request-id` from the same session-derived value as `prompt_cache_key`, improving prompt cache affinity for append-only sessions
727
+
728
+ ## [14.1.1] - 2026-04-14
729
+
730
+ ### Added
731
+
732
+ - Added `toolStrictMode` compatibility option (`"all_strict"` or `"none"`) to OpenAI-compatible model config to force tool schemas to be sent uniformly strict, uniformly non-strict, or keep mixed per-tool behavior
733
+
734
+ ### Changed
735
+
736
+ - Changed Cerebras OpenAI-compatible providers to default `toolStrictMode` to `"all_strict"` unless explicitly overridden
737
+
738
+ ### Fixed
739
+
740
+ - Fixed OpenAI Completions handling for providers that reject mixed `strict` flags by automatically retrying with non-strict tool schemas when an initial all-strict tool request fails with strict-format 400/422 errors
741
+ - Fixed OpenAI-completions error reporting by including captured JSON error body details such as type, param, and code when a request fails without a body in the thrown SDK error
742
+ - Fixed shell execution failure responses to preserve all result fields when sanitizing, preventing truncated metadata in stream results
743
+ - Fixed context overflow detection to recognize `model_context_window_exceeded` from z.ai / GLM providers, preventing infinite retry loops when context window is exceeded ([#638](https://github.com/jaybeyond/sayknow-cli/issues/638))
744
+ - Fixed strict tool schema enforcement to preserve `additionalProperties: false` and required keys for reused nested object schemas, preventing invalid `todo_write` function schemas in OpenAI code/OpenAI requests
745
+ - Fixed GitHub Copilot reasoning regressions by preserving GPT-5.x / Anthropic model 4.x reasoning controls instead of stripping them from requests ([#773](https://github.com/jaybeyond/sayknow-cli/issues/773))
746
+
747
+ ## [14.1.0] - 2026-04-11
748
+
749
+ ### Added
750
+
751
+ - Added `accountId` to usage report metadata
752
+
753
+ ### Changed
754
+
755
+ - Changed usage parsing to emit a usage report with available fields when parsing fails, rather than returning null
756
+
757
+ ### Fixed
758
+
759
+ - Fixed `planType` resolution to fall back to the raw payload `plan_type` when parsed value is absent
760
+ - Fixed usage metadata `raw` fallback to preserve the original payload when parsed raw output is missing
761
+
762
+ ## [14.0.5] - 2026-04-11
763
+
764
+ ### Changed
765
+
766
+ - Replaced GitHub Copilot authentication from VSCode extension impersonation to the opencode OAuth flow, eliminating TOS concerns. Existing users will need to re-authenticate once with `/login github-copilot`.
767
+ - Simplified Copilot token handling: GitHub OAuth token is used directly for all API requests (no JWT exchange or refresh cycle).
768
+ - Changed GitHub Copilot API base URL from `api.individual.githubcopilot.com` to `api.githubcopilot.com`.
769
+ - Updated default OpenAI stream idle timeout to 120,000 milliseconds to keep stream generation alive longer
770
+
771
+ ### Fixed
772
+
773
+ - Fixed duplicate synthetic tool results being generated when a real tool result appears later in message history
774
+ - Fixed GitHub Copilot `/models` discovery to unwrap structured OAuth credentials before sending the bearer token, preserving dynamic catalog refresh for OAuth-backed callers.
775
+
776
+ ### Removed
777
+
778
+ - Removed Copilot JWT proxy-ep base URL resolution (no longer needed with opencode auth).
779
+
780
+ ## [14.0.3] - 2026-04-09
781
+
782
+ ### Fixed
783
+
784
+ - Fixed Ollama discovery cache normalization so cached models upgrade to the OpenAI Responses transport after the provider change
785
+
786
+ ## [14.0.0] - 2026-04-08
787
+
788
+ ### Breaking Changes
789
+
790
+ - Removed `coerceNullStrings` function and its automatic null-string coercion behavior from JSON parsing
791
+
792
+ ### Added
793
+
794
+ - Added support for OpenRouter provider with strict mode detection
795
+ - Added automatic cleaning of literal escape sequences (`\n`, `\t`, `\r`) in JSON parsing to handle LLM encoding confusion
796
+ - Added support for healing JSON with trailing junk after balanced containers (e.g., `]\n</invoke>`)
797
+ - Added `OPENAI_CODE_STARTUP_EVENT_CHANNEL` constant and `OpenAI codeStartupEvent` type for monitoring OpenAI code provider initialization status
798
+ - Added automatic healing of malformed JSON with single-character bracket errors at the end of strings, improving LLM tool argument parsing robustness
799
+
800
+ ## [13.19.0] - 2026-04-05
801
+
802
+ ### Fixed
803
+
804
+ - Fixed GitHub Copilot model context window detection by correcting fallback priority for maxContextWindowTokens and maxPromptTokens
805
+ - Fixed Gemini 2.5 Pro context window detection in GitHub Copilot model limits test
806
+ - Fixed Anthropic model Opus 4.6 context window detection in GitHub Copilot model limits test
807
+ - Fixed Anthropic streaming to suppress transient SDK console errors for malformed SSE keep-alive frames so the TUI only shows surfaced provider errors
808
+
809
+ - Added environment-based credential fallback for the OpenAI code provider provider.
810
+
811
+ ## [13.17.6] - 2026-04-01
812
+
813
+ ### Fixed
814
+
815
+ - Fixed Anthropic first-event timeouts to exclude stream connection setup from the watchdog, preserve timeout-specific retry classification after local aborts, and reset retry state cleanly between attempts
816
+
817
+ ## [13.17.5] - 2026-04-01
818
+
819
+ ### Changed
820
+
821
+ - Increased default first-event timeout from 15s to 45s to better accommodate longer request setup times
822
+ - Modified first-event watchdog to inherit idle timeout when it exceeds the default, ensuring consistent timeout behavior across different configurations
823
+
824
+ ### Fixed
825
+
826
+ - Fixed first-event watchdog initialization timing so it no longer starts before the actual stream request is created, preventing premature timeouts during request setup
827
+ - Fixed first-event watchdog timing so OpenAI-family providers no longer count slow request setup against the first streamed event timeout, and raised the default first-event timeout to avoid false aborts after long tool turns
828
+
829
+ ## [13.17.2] - 2026-04-01
830
+
831
+ ### Fixed
832
+
833
+ - Fixed OpenAI-family first-event timeouts to preserve provider-specific timeout errors for retry classification instead of flattening them to generic aborts ([#591](https://github.com/jaybeyond/sayknow-cli/issues/591))
834
+
835
+ ## [13.17.1] - 2026-04-01
836
+
837
+ ### Added
838
+
839
+ - Added `thinkingSignature` field to thinking content blocks to preserve the original reasoning field name (e.g., `reasoning_text`, `reasoning_content`) for accurate follow-up requests
840
+ - Added first-event timeout detection for streaming responses to abort stuck requests before user-visible content arrives
841
+ - Added `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` environment variable to configure first-event timeout (defaults to 15 seconds or idle timeout, whichever is lower)
842
+
843
+ ### Changed
844
+
845
+ - Changed thinking block handling to track and distinguish between different reasoning field types, enabling proper field name preservation across multiple turns
846
+
847
+ ### Fixed
848
+
849
+ - Fixed Anthropic stream timeout errors to be properly retried by recognizing first-event timeout messages
850
+ - Fixed stream stall detection to distinguish between first-event timeouts and idle timeouts, enabling faster recovery for stuck connections
851
+
852
+ ### Added
853
+
854
+ - Added Vercel AI Gateway to `/login` providers for interactive API key setup
855
+
856
+ ### Fixed
857
+
858
+ - Fixed `skc commit` failing with HTTP 400 errors when using reasoning-enabled models on OpenAI-compatible endpoints that don't support the `developer` role (e.g., GitHub Copilot, custom proxies). Now falls back to `system` role when `developer` is unsupported.
859
+
860
+ ## [13.17.0] - 2026-03-30
861
+
862
+ ### Changed
863
+
864
+ - Bumped zai provider default model from glm-4.6 to glm-5.1
865
+
866
+ ## [13.16.5] - 2026-03-29
867
+
868
+ ### Added
869
+
870
+ - Added Gemma 3 27B model support for Google Generative AI
871
+
872
+ ### Changed
873
+
874
+ - Updated Kwaipilot KAT-Coder-Pro V2 model display name and pricing information
875
+ - Updated Kwaipilot KAT-Coder-Pro V2 context window from 222,222 to 256,000 tokens and max tokens from 8,888 to 80,000
876
+
877
+ ### Fixed
878
+
879
+ - Fixed normalizeAnthropicBaseUrl returning empty string instead of undefined when baseUrl is empty
880
+
881
+ ## [13.16.4] - 2026-03-28
882
+
883
+ ### Added
884
+
885
+ - Added support for Groq Compound and Compound Mini models with extended context window (131K tokens) and configurable thinking levels
886
+ - Added support for OpenAI GPT-OSS-Safeguard-20B model with reasoning capabilities across multiple providers
887
+ - Added support for Kwaipilot KAT-Coder-Pro V2 model across Kilo, NanoGPT, and OpenRouter providers
888
+ - Added support for GLM-5.1 model with extended context window (200K tokens) and max output of 131K tokens
889
+ - Added support for Qwen3.5-27B-Musica-v1 model
890
+ - Added support for zai-org/glm-5.1 model with reasoning capabilities
891
+ - Added support for Sapiens AI Agnes-1.5-Lite model with multimodal input (text and image) and reasoning
892
+ - Added support for Venice openai-gpt-54-mini model
893
+
894
+ ### Changed
895
+
896
+ - Updated Qwen QwQ 32B max tokens from 16,384 to 40,960 across multiple providers
897
+ - Updated OpenAI GPT-OSS-Safeguard-20B model name to 'Safety GPT OSS 20B' and enabled reasoning capabilities
898
+ - Updated OpenAI GPT-OSS-Safeguard-20B context window from 222,222 to 131,072 tokens and max tokens from 8,888 to 65,536
899
+ - Updated OpenRouter Qwen QwQ 32B pricing: input from 0.2 to 0.19, output from 1.17 to 1.15, cache read from 0.1 to 0.095
900
+ - Updated OpenRouter Anthropic model 3.5 Sonnet pricing: input from 0.45 to 0.42, cache read from 0.225 to 0.21
901
+
902
+ ## [13.16.3] - 2026-03-28
903
+
904
+ ### Changed
905
+
906
+ - Modified OAuth credential saving to preserve unrelated identities instead of replacing all credentials for a provider
907
+ - Updated credential identity resolution to use provider context for more accurate email deduplication
908
+
909
+ ### Fixed
910
+
911
+ - Fixed OAuth credential updates to replace matching credentials in-place rather than creating disabled rows, preventing unbounded accumulation of soft-deleted credentials
912
+
913
+ ## [13.15.0] - 2026-03-23
914
+
915
+ ### Added
916
+
917
+ - Added `isUsageLimitError()` to `rate-limit-utils` as a single source of truth for detecting usage/quota limit errors across all providers
918
+
919
+ ### Fixed
920
+
921
+ - Fixed lazy stream forwarding to properly handle final results from source streams with `result()` methods
922
+ - Fixed lazy stream error handling to convert iterator failures into terminal error results instead of silently failing
923
+ - Fixed `parseRateLimitReason` to recognize "usage limit" in error messages and correctly classify them as `QUOTA_EXHAUSTED`
924
+ - Fixed OpenAI code `fetchWithRetry` retrying 429 responses for `usage_limit_reached` errors for up to 5 minutes instead of returning immediately for credential switching
925
+ - Removed `usage.?limit` from `TRANSIENT_MESSAGE_PATTERN` in retry utils since usage limits are not transient and require credential rotation
926
+ - Fixed `parseRateLimitReason` not recognizing "usage limit" in OpenAI code error messages, causing incorrect fallback to `UNKNOWN` classification instead of `QUOTA_EXHAUSTED`
927
+
928
+ ## [13.14.2] - 2026-03-21
929
+
930
+ ### Changed
931
+
932
+ - Updated thinking configuration format from `levels` array to `minLevel` and `maxLevel` properties for improved clarity
933
+ - Corrected context window from 400000 to 272000 tokens for GPT-5.4 mini and nano variants on OpenAI code transport
934
+ - Normalized GPT-5.4 variant priority handling to use parsed variant instead of special-casing raw model IDs
935
+ - Added support for `mini` variant in OpenAI model parsing regex
936
+
937
+ ### Fixed
938
+
939
+ - Fixed inconsistent thinking level configuration across multiple model definitions
940
+
941
+ ## [13.14.0] - 2026-03-20
942
+
943
+ ### Fixed
944
+
945
+ - Fixed resumed OpenAI Responses sessions to avoid replaying stale same-provider native history on the first follow-up after process restart ([#488](https://github.com/jaybeyond/sayknow-cli/issues/488))
946
+
947
+ ### Added
948
+
949
+ - Added bundled GPT-5.4 mini model metadata for OpenAI, OpenAI code provider, and GitHub Copilot, including low-to-xhigh thinking support and GitHub Copilot premium multiplier metadata
950
+ - Added bundled GPT-5.4 nano model metadata for OpenAI and OpenAI code provider, including low-to-xhigh thinking support
951
+
952
+ ## [13.13.2] - 2026-03-18
953
+
954
+ ### Changed
955
+
956
+ - Modified tool result handling for aborted assistant messages to preserve existing tool results when already recorded, instead of always replacing them with synthetic 'aborted' results
957
+
958
+ ## [13.13.0] - 2026-03-18
959
+
960
+ ### Changed
961
+
962
+ - Changed tool argument validation to always normalize optional null values before type coercion, ensuring consistent handling of LLM-generated 'null' strings
963
+
964
+ ### Fixed
965
+
966
+ - Fixed tool argument validation to properly handle string 'null' values from LLMs on optional fields by stripping them during normalization
967
+ - Improved type safety of `validateToolCall` and `validateToolArguments` functions by returning properly typed `ToolCall["arguments"]` instead of `any`
968
+
969
+ ## [13.12.9] - 2026-03-17
970
+
971
+ ### Changed
972
+
973
+ - Extracted OpenAI compatibility detection and resolution logic into dedicated `openai-completions-compat` module for improved maintainability and reusability
974
+
975
+ ### Fixed
976
+
977
+ - Fixed `openai-responses` manual history replay to strip replay-only item IDs and preserve normalized tool `call_id` values for GitHub Copilot follow-up turns ([#457](https://github.com/jaybeyond/sayknow-cli/issues/457))
978
+
979
+ ## [13.12.0] - 2026-03-14
980
+
981
+ ### Added
982
+
983
+ - Added support for `qwen-chat-template` thinking format to enable reasoning via `chat_template_kwargs.enable_thinking`
984
+ - Added `reasoningEffortMap` option to `OpenAICompat` for mapping pi-ai reasoning levels to provider-specific `reasoning_effort` values
985
+ - Added `extraBody` to `OpenAICompat` to support provider-specific request body routing fields in OpenAI-completions requests
986
+ - Added support for reading token usage from choice-level `usage` field as fallback when root-level usage is unavailable
987
+ - Added new models: DeepSeek-V3.2 (Bedrock), Llama 3.1 405B Instruct, Magistral Small 1.2, Ministral 3 3B, Mistral Large 3, Pixtral Large (25.02), NVIDIA Nemotron Nano 3 30B, and Qwen3-5-9b
988
+ - Added `close()` method to `AuthStorage` for properly closing the underlying credential store
989
+ - Added `initiatorOverride` option in OpenAI and Anthropic providers to customize message attribution
990
+
991
+ ### Changed
992
+
993
+ - Changed assistant message content serialization to always use plain string format instead of text block arrays to prevent recursive nesting in OpenAI-compatible backends
994
+ - Changed Bedrock Opus 4.6 context window from 1M to 1M and added max tokens limit of 128K
995
+ - Changed OpenCode Zen/Go Sonnet 4.0/4.5 context window from 1M to 200K
996
+ - Changed GitHub Copilot context windows from 200K to 128K for both gpt-4o and gpt-4o-mini
997
+ - Changed Anthropic model 3.5 Sonnet (Anthropic API) pricing: input from $0.5 to $0.25, output from $3 to $1.5, cache read from $0.05 to $0.025, cache write from $0 to $1
998
+ - Changed Devstral 2 model name from '135B' to '123B'
999
+ - Changed ByteDance Seed 2.0-Lite to support reasoning with effort-based thinking mode and image inputs
1000
+ - Changed Qwen3-32b (Groq) reasoning effort mapping to normalize all levels to 'default'
1001
+ - Changed finish_reason 'end' to map to 'stop' for improved compatibility with additional providers
1002
+ - Changed Anthropic reference model merging to prioritize bundled metadata for known models while using models.dev for newly discovered IDs
1003
+
1004
+ ### Fixed
1005
+
1006
+ - Fixed reasoning_effort parameter handling to use provider-specific mappings instead of raw effort values
1007
+ - Fixed assistant content serialization for GitHub Copilot and other OpenAI-compatible backends that mirror array payloads
1008
+ - Fixed token usage calculation to properly extract cached tokens from both root and nested `prompt_tokens_details` fields
1009
+ - Fixed stop reason mapping to handle string values and unknown finish reasons gracefully
1010
+ - Fixed resource cleanup in `AuthCredentialStore.close()` to properly finalize all prepared statements before closing the database
1011
+
1012
+ ## [13.11.1] - 2026-03-13
1013
+
1014
+ ### Fixed
1015
+
1016
+ - Added `llama.cpp` as local provider
1017
+ - Fixed auth schema V0-to-V1 migration crash when the V0 table lacks a `disabled` column
1018
+
1019
+ ## [13.11.0] - 2026-03-12
1020
+
1021
+ ### Added
1022
+
1023
+ - Added support for Parallel AI provider with API key authentication
1024
+ - Added `PARALLEL_API_KEY` environment variable support for Parallel provider configuration
1025
+ - Added automatic websocket reconnection handling for connection limit errors, with fallback to SSE replay when content has already been emitted
1026
+
1027
+ ### Changed
1028
+
1029
+ - Enhanced `OpenAI codeProviderStreamError` to include an optional error code field for better error categorization and handling
1030
+
1031
+ ### Fixed
1032
+
1033
+ - Improved retry logic to handle HTTP/2 stream errors and internal_error responses from Anthropic API
1034
+
1035
+ ## [13.9.16] - 2026-03-10
1036
+
1037
+ ### Added
1038
+
1039
+ - Support for `onPayload` callback to replace provider request payloads before sending, enabling request interception and modification
1040
+ - Support for structured text signature metadata with phase information (commentary/final_answer) in OpenAI and Azure OpenAI Responses providers
1041
+ - Support for OpenAI code provider Spark model selection with plan-based account prioritization
1042
+ - Added `modelId` option to `getApiKey()` to enable model-specific credential ranking
1043
+
1044
+ ### Changed
1045
+
1046
+ - Enhanced `onPayload` callback signature to accept model parameter and support async payload replacement
1047
+ - Improved error messages for `response.failed` events to include detailed error codes, messages, and incomplete reasons
1048
+ - Refactored OpenAI code provider response streaming to improve code organization and maintainability with extracted helper functions and type definitions
1049
+ - Enhanced websocket fallback logic to safely replay buffered output over SSE when websocket connections fail mid-stream
1050
+ - Improved error recovery for websocket streams by distinguishing between fatal connection errors and retryable stream errors
1051
+ - Updated credential ranking strategy to prioritize Pro plan accounts when requesting OpenAI code provider Spark models
1052
+
1053
+ ### Fixed
1054
+
1055
+ - Fixed websocket stream recovery to properly reset output state and clear buffered items when falling back to SSE after partial output
1056
+ - Fixed handling of malformed JSON messages in websocket streams to trigger immediate fallback to SSE without retry attempts
1057
+
1058
+ ## [13.9.13] - 2026-03-10
1059
+
1060
+ ### Added
1061
+
1062
+ - Added `isSpecialServiceTier` utility function to validate OpenAI service tier values
1063
+
1064
+ ## [13.9.12] - 2026-03-09
1065
+
1066
+ ### Added
1067
+
1068
+ - Added Tavily web search provider support with API key authentication
1069
+
1070
+ ### Fixed
1071
+
1072
+ - Fixed OpenAI-family streaming transports to fail with an explicit idle-timeout error instead of hanging indefinitely when the provider stops sending events mid-response
1073
+ - Fixed OpenAI code provider OAuth refresh and usage-limit lookups to respect request timeouts instead of waiting indefinitely during account selection or rotation
1074
+ - Fixed OpenAI code provider prewarmed websocket requests to fall back quickly when the socket connects but never starts the response stream
1075
+
1076
+ ## [13.9.10] - 2026-03-08
1077
+
1078
+ ### Added
1079
+
1080
+ - Added `identity_key` column to auth credentials storage for improved credential deduplication
1081
+ - Added schema versioning system to auth credentials database for safer migrations
1082
+ - Added automatic backfilling of identity keys during database schema migrations
1083
+
1084
+ ### Changed
1085
+
1086
+ - Changed credential deduplication logic to use single identity key instead of multiple identifiers for better performance
1087
+ - Changed database schema to store normalized identity keys alongside credentials
1088
+ - Changed auth schema migration to support upgrading from legacy database versions with automatic data backfill
1089
+
1090
+ ### Fixed
1091
+
1092
+ - Fixed API key credential matching to correctly identify when the same key is re-stored, preventing unnecessary row duplication on re-login
1093
+ - Fixed credential deduplication to correctly handle OAuth accounts with matching emails but different account IDs
1094
+ - Fixed API key replacement to reuse existing stored rows instead of accumulating disabled duplicates
1095
+ - Fixed auth storage to preserve newer recorded schema versions when opened by older binaries
1096
+
1097
+ ## [13.9.8] - 2026-03-08
1098
+
1099
+ ### Fixed
1100
+
1101
+ - Fixed WebSocket stream fallback logic to safely replay buffered output over SSE when WebSocket fails after partial content has been streamed
1102
+
1103
+ ## [13.9.4] - 2026-03-07
1104
+
1105
+ ### Changed
1106
+
1107
+ - Simplified API key credential storage to always replace existing credentials on re-login instead of accumulating multiple keys
1108
+ - Updated Kagi API key placeholder from `kagi_...` to `KG_...` to match current API key format
1109
+ - Updated Kagi login instructions to clarify Search API access is beta-only and provide support contact
1110
+ - Disabled usage reporting in streaming responses for Cerebras models due to compatibility issues
1111
+
1112
+ ### Fixed
1113
+
1114
+ - Fixed Cerebras model compatibility by preventing `stream_options` usage requests in chat completions
1115
+
1116
+ ## [13.9.3] - 2026-03-07
1117
+
1118
+ ### Breaking Changes
1119
+
1120
+ - Changed `reasoning` parameter from `ThinkingLevel | undefined` to `Effort | undefined` in `SimpleStreamOptions`; 'off' is no longer valid (omit the field instead)
1121
+ - Removed `supportsXhigh()` function; check `model.thinking?.maxLevel` instead
1122
+ - Removed `ThinkingLevel` and `ThinkingEffort` types; use `Effort` enum
1123
+ - Removed `getAvailableThinkingLevels()` and `getAvailableThinkingEfforts()` functions
1124
+ - Changed `transformRequestBody()` signature to require `Model` parameter as second argument for effort validation
1125
+ - Removed `thinking.ts` module export; import from `model-thinking.ts` instead
1126
+
1127
+ ### Added
1128
+
1129
+ - Added `incremental` flag to `OpenAIResponsesHistoryPayload` to support building conversation history from multiple assistant messages instead of replacing it
1130
+ - Added `dt` flag to `OpenAIResponsesHistoryPayload` for transport-level metadata
1131
+ - Added `ThinkingConfig` interface to models for canonical thinking transport metadata with min/max effort levels and provider-specific mode
1132
+ - Added `thinking` field to `Model` type containing per-model thinking capabilities used to clamp and map user-facing effort levels
1133
+ - Added `Effort` enum (minimal, low, medium, high, xhigh) as canonical user-facing thinking levels replacing `ThinkingLevel`
1134
+ - Added `enrichModelThinking()` function to automatically populate thinking metadata on models based on their capabilities
1135
+ - Added `mapEffortToAnthropicAdaptiveEffort()` function to map user effort levels to Anthropic adaptive thinking effort
1136
+ - Added `mapEffortToGoogleThinkingLevel()` function to map user effort levels to Google thinking levels
1137
+ - Added `requireSupportedEffort()` function to validate and clamp effort levels per model, throwing errors for unsupported combinations
1138
+ - Added `clampThinkingLevelForModel()` function to clamp thinking levels to model-supported range
1139
+ - Added `applyGeneratedModelPolicies()` and `linkSparkPromotionTargets()` exports from model-thinking module
1140
+ - Added `serviceTier` option to control OpenAI processing priority and cost (auto, default, flex, scale, priority)
1141
+ - Added `providerPayload` field to messages and responses for reconstructing transport-native history
1142
+ - Added Gemini usage provider for tracking quota and tier information
1143
+ - Added `getOpenAI codeAccountId()` utility to extract account ID from OpenAI code JWT tokens
1144
+ - Added email extraction from OpenAI code provider OAuth tokens for credential deduplication
1145
+
1146
+ ### Changed
1147
+
1148
+ - Changed credential disabling mechanism from boolean `disabled` flag to `disabled_cause` text field for tracking why credentials were disabled
1149
+ - Changed `deleteAuthCredential()` and `deleteAuthCredentialsForProvider()` methods to require a `disabledCause` parameter explaining the reason for disabling
1150
+ - Changed Gemini model parsing to strip `-preview` suffix for consistent model identification
1151
+ - Changed OpenAI code provider websocket error handling to detect fatal connection errors and immediately fall back to SSE without retrying
1152
+ - Changed OpenAI code provider to always use websockets v2 protocol (removed v1 support)
1153
+ - Changed `reasoning` parameter type from `ThinkingLevel` to `Effort` in `SimpleStreamOptions`, removing 'off' value (callers should omit the field instead)
1154
+ - Changed thinking configuration to use model-specific metadata instead of hardcoded provider logic for effort mapping
1155
+ - Changed OpenAI code provider request transformer to accept `Model` parameter for effort validation instead of string model ID
1156
+ - Changed Anthropic provider to use model thinking metadata for determining adaptive thinking support instead of model ID pattern matching
1157
+ - Changed Google Vertex and Google providers to use shorter variable names for thinking config construction
1158
+ - Moved thinking-related utilities from `thinking.ts` to new `model-thinking.ts` module with expanded functionality
1159
+ - Moved model policy functions from `provider-models/model-policies.ts` to `model-thinking.ts`
1160
+ - Moved `googleGeminiCliUsageProvider` from `providers/google-gemini-cli-usage.ts` to `usage/gemini.ts`
1161
+ - Changed default OpenAI model from gpt-5.1-openai-code to gpt-5.4 across all providers
1162
+ - Changed `UsageFetchContext` to remove cache and now() dependencies—usage fetchers now use Date.now() directly
1163
+ - Removed `resetInMs` field from usage windows; consumers should calculate from `resetsAt` timestamp
1164
+ - Changed OpenAI code provider credential ranking to deduplicate by email when accountId matches
1165
+ - Improved OpenAI code provider error handling with retryable error detection
1166
+
1167
+ ### Removed
1168
+
1169
+ - Removed `thinking.ts` module; use `model-thinking.ts` instead
1170
+ - Removed `provider-models/model-policies.ts` module; functionality moved to `model-thinking.ts`
1171
+ - Removed `supportsXhigh()` function from models.ts; use model.thinking metadata instead
1172
+ - Removed `ThinkingLevel` and `ThinkingEffort` types; use `Effort` enum instead
1173
+ - Removed `getAvailableThinkingLevels()` and `getAvailableThinkingEfforts()` functions
1174
+ - Removed `model-policies` export from `provider-models/index.ts`
1175
+ - Removed hardcoded thinking level clamping logic from OpenAI code provider request transformer; now uses model metadata
1176
+ - Removed `UsageCache` and `UsageCacheEntry` interfaces—caching is now handled internally by AuthStorage
1177
+ - Removed `google-gemini-cli-usage` export; use new `gemini` usage provider instead
1178
+ - Removed `resetInMs` computation from all usage providers
1179
+ - Removed cache TTL constants and cache management from usage fetchers (anthropic-model, github-copilot, google-antigravity, kimi, openai-code, zai)
1180
+
1181
+ ### Fixed
1182
+
1183
+ - Fixed credential purging to respect disabled credentials when deduplicating by email, preventing re-enablement of intentionally disabled credentials
1184
+ - Fixed OpenAI code provider websocket error reporting to include detailed error messages from error events
1185
+ - Fixed conversation history reconstruction to support incremental updates from multiple assistant messages while maintaining backward compatibility with full-snapshot payloads
1186
+ - Fixed OpenAI code provider to reject unsupported effort levels instead of silently clamping them, providing clear error messages about supported efforts
1187
+ - Fixed model cache normalization to properly apply thinking enrichment when loading cached models
1188
+ - Fixed dynamic model merging to apply thinking enrichment to merged model results
1189
+ - Fixed OpenAI code provider streaming to properly include service_tier in SSE payloads
1190
+ - Fixed type safety in OpenAI responses by removing unsafe type casts on image content blocks
1191
+ - Fixed credential purging to respect disabled credentials when deduplicating by email
1192
+ - Fixed API-key provider re-login to replace the active stored key instead of appending stale credentials that were still selected first
1193
+ - Fixed Kagi login guidance to use the correct `KG_...` key format and mention Search API beta access requirements
1194
+
1195
+ ## [13.9.2] - 2026-03-05
1196
+
1197
+ ### Added
1198
+
1199
+ - Support for redacted thinking blocks in Anthropic messages, enabling secure handling of encrypted reasoning content
1200
+ - Preservation of latest Anthropic thinking blocks and redacted thinking content during message transformation, even when switching between Anthropic models
1201
+
1202
+ ### Changed
1203
+
1204
+ - Assistant message content now includes `RedactedThinkingContent` type alongside existing text, thinking, and tool call blocks
1205
+ - Message transformation logic now preserves signed thinking blocks and redacted thinking for the latest assistant message in Anthropic conversations
1206
+
1207
+ ### Fixed
1208
+
1209
+ - Fixed Unicode normalization to consistently apply `toWellFormed()` to all text content, including thinking blocks, ensuring proper handling of malformed UTF-16 sequences
1210
+
1211
+ ## [13.9.1] - 2026-03-05
1212
+
1213
+ ### Breaking Changes
1214
+
1215
+ - Removed `THINKING_LEVELS`, `ALL_THINKING_LEVELS`, `ALL_THINKING_MODES`, `THINKING_MODE_DESCRIPTIONS`, and `THINKING_MODE_LABELS` exports
1216
+ - Renamed `formatThinking()` to `getThinkingMetadata()` with changed return type from string to `ThinkingMetadata` object
1217
+ - Renamed `getAvailableThinkingLevel()` to `getAvailableThinkingLevels()` and added default parameter
1218
+ - Renamed `getAvailableEffort()` to `getAvailableEfforts()` and added default parameter
1219
+
1220
+ ### Added
1221
+
1222
+ - Added `ThinkingMetadata` type to provide structured access to thinking mode information (value, label, description)
1223
+
1224
+ ## [13.9.0] - 2026-03-05
1225
+
1226
+ ### Added
1227
+
1228
+ - Exported new thinking module with `Effort`, `ThinkingLevel`, and `ThinkingMode` types for managing reasoning effort levels
1229
+ - Added `getAvailableEffort()` function to determine supported thinking effort levels based on model capabilities
1230
+ - Added `parseEffort()`, `parseThinkingLevel()`, and `parseThinkingMode()` functions for parsing thinking configuration strings
1231
+ - Added `THINKING_LEVELS`, `ALL_THINKING_LEVELS`, and `ALL_THINKING_MODES` constants for iterating over available thinking options
1232
+ - Added `THINKING_MODE_DESCRIPTIONS` and `THINKING_MODE_LABELS` for displaying thinking modes in user interfaces
1233
+ - Added `formatThinking()` function to format thinking modes as compact display labels
1234
+
1235
+ ### Changed
1236
+
1237
+ - Refactored thinking level handling to distinguish between `Effort` (provider-level, no "off") and `ThinkingLevel` (user-facing, includes "off")
1238
+ - Updated `ThinkingBudgets` type to use `Effort` instead of `ThinkingLevel` for more precise token budget configuration
1239
+ - Improved reasoning option handling to explicitly support "off" value for disabling reasoning across all providers
1240
+ - Simplified thinking effort mapping logic by centralizing provider-specific clamping behavior
1241
+
1242
+ ## [13.7.8] - 2026-03-04
1243
+
1244
+ ### Added
1245
+
1246
+ - Added ZenMux provider support with mixed API routing: Anthropic-owned models discovered from `https://zenmux.ai/api/v1/models` now use the Anthropic transport (`https://zenmux.ai/api/anthropic`), while other ZenMux models use the OpenAI-compatible transport.
1247
+
1248
+ ## [13.7.7] - 2026-03-04
1249
+
1250
+ ### Changed
1251
+
1252
+ - Modified response ID normalization to preserve existing item ID prefixes when truncating oversized IDs
1253
+ - Updated tool call ID normalization to use `fc_` prefix for generated item IDs instead of `item_` prefix
1254
+
1255
+ ### Fixed
1256
+
1257
+ - Fixed handling of reasoning item IDs to remain untouched during response normalization while function call IDs are properly normalized
1258
+
1259
+ ## [13.7.2] - 2026-03-04
1260
+
1261
+ ### Added
1262
+
1263
+ - Added support for Kagi API key authentication via `login kagi` command
1264
+ - Added Kagi to the list of available OAuth providers
1265
+
1266
+ ### Fixed
1267
+
1268
+ - MCP tool schemas with `$ref`/`$defs` are now dereferenced before being sent to LLM providers, fixing dangling references that left models without type definitions
1269
+ - Ajv schema validation no longer emits `console.warn()` for non-standard format keywords (e.g. `"uint"`) from MCP servers, preventing TUI corruption
1270
+ - Tool schema compilation is now cached per schema identity, eliminating redundant recompilation on every tool call
1271
+
1272
+ ## [13.6.0] - 2026-03-03
1273
+
1274
+ ### Added
1275
+
1276
+ - Added Anthropic Foundry gateway mode controlled by `ANTHROPIC_MODEL_CODE_USE_FOUNDRY`, with support for `FOUNDRY_BASE_URL`, `ANTHROPIC_FOUNDRY_API_KEY`, `ANTHROPIC_CUSTOM_HEADERS`, and optional mTLS material (`ANTHROPIC_MODEL_CODE_CLIENT_CERT`, `ANTHROPIC_MODEL_CODE_CLIENT_KEY`, `NODE_EXTRA_CA_CERTS`)
1277
+ - Added LM Studio provider support with OpenAI-compatible model discovery and OAuth login.
1278
+ - Added support for `LM_STUDIO_API_KEY` and `LM_STUDIO_BASE_URL` environment variables for authentication and custom host configuration.
1279
+
1280
+ ### Changed
1281
+
1282
+ - Anthropic key resolution now prefers `ANTHROPIC_FOUNDRY_API_KEY` over `ANTHROPIC_OAUTH_TOKEN` and `ANTHROPIC_API_KEY` when Foundry mode is enabled
1283
+ - Anthropic auth base-URL fallback now prefers `FOUNDRY_BASE_URL` when `ANTHROPIC_MODEL_CODE_USE_FOUNDRY` is enabled
1284
+
1285
+ ## [13.5.8] - 2026-03-02
1286
+
1287
+ ### Fixed
1288
+
1289
+ - Fixed schema compatibility issue where patternProperties in tool parameters caused failures when converting to legacy Antigravity format
1290
+
1291
+ ## [13.5.5] - 2026-03-01
1292
+
1293
+ ### Changed
1294
+
1295
+ - Anthropic Anthropic model system-block cloaking now leaves the agent identity block uncached and applies `cache_control: { type: "ephemeral" }` to injected user system blocks without forcing `ttl: "1h"`
1296
+
1297
+ ### Fixed
1298
+
1299
+ - Anthropic request payload construction now enforces a maximum of 4 `cache_control` breakpoints (tools/system/messages priority order) before dispatch
1300
+ - Anthropic cache-control normalization now removes later `ttl: "1h"` entries when a default/5m block has already appeared earlier in evaluation order
1301
+
1302
+ ## [13.5.3] - 2026-03-01
1303
+
1304
+ ### Fixed
1305
+
1306
+ - Fixed tool argument coercion to handle malformed JSON with trailing wrapper braces by parsing leading JSON containers
1307
+
1308
+ ## [13.4.0] - 2026-03-01
1309
+
1310
+ ### Breaking Changes
1311
+
1312
+ - Removed `TInput` generic parameter from `ToolResultMessage` interface and removed `$normative` property
1313
+
1314
+ ### Added
1315
+
1316
+ - `hasUnrepresentableStrictObjectMap()` pre-flight check in `tryEnforceStrictSchema`: schemas with `patternProperties` or schema-valued `additionalProperties` now degrade gracefully to non-strict mode instead of throwing during enforcement
1317
+ - `generateAnthropic modelCloakingUserId()` generates structured user IDs for Anthropic OAuth metadata (`user_{hex64}_account_{uuid}_session_{uuid}`)
1318
+ - `isAnthropic modelCloakingUserId()` validates whether a string matches the cloaking user-ID format
1319
+ - `mapStainlessOs()` and `mapStainlessArch()` map `process.platform`/`process.arch` to Stainless header values; X-Stainless-Os and X-Stainless-Arch in `anthropic-modelCodeHeaders` are now runtime-computed
1320
+ - `buildAnthropic modelCodeTlsFetchOptions()` attaches SNI and default TLS ciphers for direct `api.anthropic.com` connections
1321
+ - `createAnthropic modelBillingHeader()` generates the `x-anthropic-billing-header` block (SHA-256 payload fingerprint + random build hash)
1322
+ - `buildAnthropicSystemBlocks()` now injects a billing header block and the Anthropic model Agent SDK identity block with `ephemeral` 1h cache-control when `includeAnthropic modelCodeInstruction` is set
1323
+ - `resolveAnthropicMetadataUserId()` auto-generates a cloaking user ID for OAuth requests when `metadata.user_id` is absent or invalid
1324
+ - `AnthropicOAuthFlow` is now exported for direct use
1325
+ - OAuth callback server timeout extended from 2 min to 5 min
1326
+ - `parseGeminiCliCredentials()` parses Google Cloud credential JSON with support for legacy (`{token,projectId}`), alias (`project_id`/`refresh`/`expires`), and enriched formats
1327
+ - `shouldRefreshGeminiCliCredentials()` and proactive token refresh before requests for both Gemini CLI and Antigravity providers (60s pre-expiry buffer)
1328
+ - `normalizeAntigravityTools()` converts `parametersJsonSchema` → `parameters` in function declarations for Antigravity compatibility
1329
+ - `ANTIGRAVITY_SYSTEM_INSTRUCTION` is now exported for use by search and other consumers
1330
+ - `ANTIGRAVITY_LOAD_CODE_ASSIST_METADATA` constant exported from OAuth module with `ANTIGRAVITY` ideType
1331
+ - Antigravity project onboarding: `onboardProjectWithRetries()` provisions a new project via `onboardUser` LRO when `loadCodeAssist` returns no existing project (up to 5 attempts, 2s interval)
1332
+ - `getOAuthApiKey` now includes `refreshToken`, `expiresAt`, `email`, and `accountId` in the Gemini/Antigravity JSON credential payload to enable proactive refresh
1333
+ - Antigravity model discovery now tries the production daily endpoint first, with sandbox as fallback
1334
+ - `ANTIGRAVITY_DISCOVERY_DENYLIST` filters low-quality/internal models from discovery results
1335
+
1336
+ ### Changed
1337
+
1338
+ - Replaced `sanitizeSurrogates()` utility with native `String.prototype.toWellFormed()` for handling unpaired Unicode surrogates across all providers
1339
+ - Extended `ANTHROPIC_OAUTH_BETA` constant in the OpenAI-compat Anthropic route with `interleaved-thinking-2025-05-14`, `context-management-2025-06-27`, and `prompt-caching-scope-2026-01-05` beta flags
1340
+ - `anthropic-modelCodeVersion` bumped to `2.1.63`; `anthropic-modelCodeSystemInstruction` updated to identify as Anthropic model Agent SDK
1341
+ - `anthropic-modelCodeHeaders`: removed `X-Stainless-Helper-Method`, updated package version to `0.74.0`, runtime version to `v24.3.0`
1342
+ - `applyAnthropic modelToolPrefix` / `stripAnthropic modelToolPrefix` now accept an optional prefix override and skip Anthropic built-in tool names (`web_search`, `code_execution`, `text_editor`, `computer`)
1343
+ - Accept-Encoding header updated to `gzip, deflate, br, zstd`
1344
+ - Non-Anthropic base URLs now receive `Authorization: Bearer` regardless of OAuth status
1345
+ - Prompt-caching logic now skips applying breakpoints when any block already carries `cache_control`, instead of stripping then re-applying
1346
+ - `fine-grained-tool-streaming-2025-05-14` removed from default beta set
1347
+ - Anthropic OAuth token URL changed from `platform.anthropic-model.com` to `api.anthropic.com`
1348
+ - Anthropic OAuth scopes reduced to `org:create_api_key user:profile user:inference`
1349
+ - OAuth code exchange now strips URL fragment from callback code, using the fragment as state override when present
1350
+ - Anthropic model usage headers aligned: user-agent updated to `anthropic-model-cli/2.1.63 (external, cli)`, anthropic-beta extended with full beta set
1351
+ - Antigravity session ID format changed to signed decimal (negative int63 derived from SHA-256 of first user message, or random bounded int63)
1352
+ - Antigravity `requestId` now uses `agent-{uuid}` format; non-Antigravity requests no longer include requestId/userAgent/requestType in the payload
1353
+ - `ANTIGRAVITY_DAILY_ENDPOINT` corrected to `daily-cloudcode-pa.googleapis.com`; sandbox endpoint kept as fallback only
1354
+ - Antigravity discovery: removed `recommended`/`agentModelSorts` filter; now includes all non-internal, non-denylisted models
1355
+ - Antigravity discovery no longer sends `project` in the request body
1356
+ - Gemini/Antigravity OAuth flows no longer use PKCE (code_challenge removed)
1357
+ - Antigravity `loadCodeAssist` metadata ideType changed from `IDE_UNSPECIFIED` to `ANTIGRAVITY`
1358
+ - Antigravity `discoverProject` now uses a single canonical production endpoint; falls back to project onboarding instead of a hardcoded default project ID
1359
+ - `VALIDATED` tool calling config applied to Antigravity requests with Anthropic model models
1360
+ - `maxOutputTokens` removed from Antigravity generation config for non-Anthropic model models
1361
+ - System instruction injection for Antigravity scoped to Anthropic model and `gemini-3-pro-high` models only
1362
+
1363
+ ### Removed
1364
+
1365
+ - Removed `sanitizeSurrogates()` utility function; use native `String.prototype.toWellFormed()` instead
1366
+
1367
+ ## [13.3.14] - 2026-02-28
1368
+
1369
+ ### Added
1370
+
1371
+ - Exported schema utilities from new `./utils/schema` module, consolidating JSON Schema handling across providers
1372
+ - Added `CredentialRankingStrategy` interface for providers to implement usage-based credential selection
1373
+ - Added `anthropic-modelRankingStrategy` for Anthropic OAuth credentials to enable smart multi-account selection based on usage windows
1374
+ - Added `openai-codeRankingStrategy` for OpenAI code provider OAuth credentials with priority boost for fresh 5-hour window starts
1375
+ - Added `adaptSchemaForStrict()` helper for unified OpenAI strict schema enforcement across providers
1376
+ - Added schema equality and merging utilities: `areJsonValuesEqual()`, `mergeCompatibleEnumSchemas()`, `mergePropertySchemas()`
1377
+ - Added Cloud Code Assist schema normalization: `copySchemaWithout()`, `stripResidualCombiners()`, `prepareSchemaForCCA()`
1378
+ - Added `sanitizeSchemaForGoogle()` and `sanitizeSchemaForCCA()` for provider-specific schema sanitization
1379
+ - Added `StringEnum()` helper for creating string enum schemas compatible with Google and other providers
1380
+ - Added `enforceStrictSchema()` and `sanitizeSchemaForStrictMode()` for OpenAI strict mode schema validation
1381
+ - Added package exports for `./utils/schema` and `./utils/schema/*` subpaths
1382
+ - Added `validateSchemaCompatibility()` to statically audit a JSON Schema against provider-specific rules (`openai-strict`, `google`, `cloud-code-assist-anthropic-model`) and return structured violations
1383
+ - Added `validateStrictSchemaEnforcement()` to verify the strict-fail-open contract: enforced schemas pass strict validation, failed schemas return the original object identity
1384
+ - Added `COMBINATOR_KEYS` (`anyOf`, `allOf`, `oneOf`) and `CCA_UNSUPPORTED_SCHEMA_FIELDS` as exported constants in `fields.ts` to eliminate duplication across modules
1385
+ - Added `tryEnforceStrictSchema` result cache (`WeakMap`) to avoid redundant sanitize + enforce work for the same schema object
1386
+ - Added comprehensive schema normalization test suite (`schema-normalization.test.ts`) covering strict mode, Google, and Cloud Code Assist normalization paths
1387
+ - Added schema compatibility validation test suite (`schema-compatibility.test.ts`) covering all three provider targets
1388
+
1389
+ ### Changed
1390
+
1391
+ - Moved schema utilities from `./utils/typebox-helpers` to new `./utils/schema` module with expanded functionality
1392
+ - Refactored OpenAI provider tool conversion to use unified `adaptSchemaForStrict()` helper across openai-code, completions, and responses
1393
+ - Updated `AuthStorage` to support generic credential ranking via `CredentialRankingStrategy` instead of OpenAI code-only logic
1394
+ - Moved Google schema sanitization functions from `google-shared.ts` to `./utils/schema` module
1395
+ - Changed export path: `./utils/typebox-helpers` → `./utils/schema` in main index
1396
+ - `sanitizeSchemaForGoogle()` / `sanitizeSchemaForCCA()` now accept a parameterized `unsupportedFields` set internally, enabling code reuse between the two sanitizers
1397
+ - `copySchemaWithout()` rewritten using object-rest destructuring for clarity
1398
+
1399
+ ### Fixed
1400
+
1401
+ - Fixed cycle detection: `WeakSet` guards added to all recursive schema traversals (`sanitizeSchemaForStrictMode`, `enforceStrictSchema`, `normalizeSchemaForCCA`, `normalizeNullablePropertiesForCloudCodeAssist`, `stripResidualCombiners`, `sanitizeSchemaImpl`, `hasResidualCloudCodeAssistIncompatibilities`) — circular schemas no longer cause infinite loops or stack overflows
1402
+ - Fixed `hasResidualCloudCodeAssistIncompatibilities`: cycle detection now returns `false` (not `true`) for already-visited nodes, eliminating false positives that forced the CCA fallback schema on valid recursive inputs
1403
+ - Fixed `stripResidualCombiners` to iterate to a fixpoint rather than making a single pass, ensuring chained combiner reductions (where one reduction enables another) are fully resolved
1404
+ - Fixed `mergeObjectCombinerVariants` required-field computation: the flattened object now takes the intersection of all variants' `required` arrays (unioned with own-level required properties that exist in the merged schema), preventing required fields from being silently dropped or over-included
1405
+ - Fixed `mergeCompatibleEnumSchemas` to use deep structural equality (`areJsonValuesEqual`) instead of `Object.is` when deduplicating object-valued enum members
1406
+ - Fixed `sanitizeSchemaForGoogle` const-to-enum deduplication to use deep equality instead of reference equality
1407
+ - Fixed `sanitizeSchemaForGoogle` type inference for `anyOf`/`oneOf`-flattened const enums: type is now derived from all variants (must agree), falling back to inference from enum values; mixed null/non-null infers the non-null type and sets `nullable`
1408
+ - Fixed `sanitizeSchemaForGoogle` recursion to spread options when descending (previously only `insideProperties`, `normalizeTypeArrayToNullable`, `stripNullableKeyword` were forwarded; new fields `unsupportedFields` and `seen` were silently dropped)
1409
+ - Fixed `sanitizeSchemaForGoogle` array-valued `type` filtering to exclude non-string entries before processing
1410
+ - Removed incorrect `additionalProperties: false` stripping from `sanitizeSchemaForGoogle` (the field is valid in Google schemas when `false`)
1411
+ - Fixed `sanitizeSchemaForStrictMode` to strip the `nullable` keyword and expand it into `anyOf: [schema, {type: "null"}]` in the output, matching what OpenAI strict mode actually expects
1412
+ - Fixed `sanitizeSchemaForStrictMode` to infer `type: "array"` when `items` is present but `type` is absent
1413
+ - Fixed `sanitizeSchemaForStrictMode` to infer a scalar `type` from uniform `enum` values when `type` is not explicitly set
1414
+ - Fixed `sanitizeSchemaForStrictMode` const-to-enum merge to use deep equality, preventing duplicate enum entries when `const` and `enum` both exist with the same value
1415
+ - Fixed `enforceStrictSchema` to drop `additionalProperties` unconditionally (previously only object-valued `additionalProperties` was recursed into; non-object values were passed through, violating strict schema requirements)
1416
+ - Fixed `enforceStrictSchema` to recurse into `$defs` and `definitions` blocks so referenced sub-schemas are also made strict-compliant
1417
+ - Fixed `enforceStrictSchema` to handle tuple-style `items` arrays (previously only single-schema `items` objects were recursed)
1418
+ - Fixed `enforceStrictSchema` double-wrapping: optional properties already expressed as `anyOf: [..., {type: "null"}]` are not wrapped again
1419
+ - Fixed `enforceStrictSchema` `Array.isArray` type-narrowing for `type` field to filter non-string entries before checking for `"object"`
1420
+
1421
+ ## [13.3.8] - 2026-02-28
1422
+
1423
+ ### Fixed
1424
+
1425
+ - Fixed response body reuse error when handling 429 rate limit responses with retry logic
1426
+
1427
+ ## [13.3.7] - 2026-02-27
1428
+
1429
+ ### Added
1430
+
1431
+ - Added `tryEnforceStrictSchema` function that gracefully downgrades to non-strict mode when schema enforcement fails, enabling better compatibility with malformed or circular schemas
1432
+ - Added `sanitizeSchemaForStrictMode` function to normalize JSON schemas by stripping non-structural keywords, converting `const` to `enum`, and expanding type arrays into `anyOf` variants
1433
+ - Added Kilo Gateway provider support with OpenAI-compatible model discovery, OAuth `/login kilo`, and `KILO_API_KEY` environment variable support ([#193](https://github.com/jaybeyond/sayknow-cli/issues/193))
1434
+
1435
+ ### Changed
1436
+
1437
+ - Changed strict mode handling in OpenAI providers to use `tryEnforceStrictSchema` for safer schema enforcement with automatic fallback to non-strict mode
1438
+ - Enhanced `enforceStrictSchema` to properly handle schemas with type arrays containing `object` (e.g., `type: ["object", "null"]`)
1439
+
1440
+ ### Fixed
1441
+
1442
+ - Fixed `enforceStrictSchema` to properly handle malformed object schemas with required keys but missing properties
1443
+ - Fixed `enforceStrictSchema` to correctly process nested object schemas within `anyOf`, `allOf`, and `oneOf` combinators
1444
+
1445
+ ## [13.3.1] - 2026-02-26
1446
+
1447
+ ### Added
1448
+
1449
+ - Added `topP`, `topK`, `minP`, `presencePenalty`, and `repetitionPenalty` options to `StreamOptions` for fine-grained control over model sampling behavior
1450
+
1451
+ ## [13.3.0] - 2026-02-26
1452
+
1453
+ ### Changed
1454
+
1455
+ - Allowed OAuth provider logins to supply a manual authorization code handler with a default prompt when none is provided
1456
+
1457
+ ## [13.2.0] - 2026-02-23
1458
+
1459
+ ### Added
1460
+
1461
+ - Added support for GitHub Copilot provider in strict mode for both openai-completions and openai-responses tool schemas
1462
+
1463
+ ### Fixed
1464
+
1465
+ - Fixed tool descriptions being rejected when undefined by providing empty string fallback across all providers
1466
+
1467
+ ## [12.19.1] - 2026-02-22
1468
+
1469
+ ### Added
1470
+
1471
+ - Exported `isProviderRetryableError` function for detecting rate-limit and transient stream errors
1472
+ - Support for retrying malformed JSON stream-envelope parse errors from Anthropic-compatible proxy endpoints
1473
+
1474
+ ### Changed
1475
+
1476
+ - Expanded retry detection to include JSON parse errors (unterminated strings, unexpected end of input) in addition to rate-limit errors
1477
+
1478
+ ## [12.19.0] - 2026-02-22
1479
+
1480
+ ### Added
1481
+
1482
+ - Added GitLab Duo provider with support for Anthropic model, GPT-5, and other models via GitLab AI Gateway
1483
+ - Added OAuth authentication for GitLab Duo with automatic token refresh and direct access caching
1484
+ - Added 16 new GitLab Duo models including Anthropic model Opus/Sonnet/Haiku variants and GPT-5 series models
1485
+ - Added `isOAuth` option to Anthropic provider to force OAuth bearer auth mode for proxy tokens
1486
+ - Added `streamGitLabDuo` function to route requests through GitLab AI Gateway with direct access tokens
1487
+ - Added `getGitLabDuoModels` function to retrieve available GitLab Duo model configurations
1488
+ - Added `clearGitLabDuoDirectAccessCache` function to manually clear cached direct access tokens
1489
+
1490
+ ### Changed
1491
+
1492
+ - Enhanced `getModelMapping()` to support both GitLab Duo alias IDs (e.g., `duo-chat-gpt-5-openai-code`) and canonical model IDs (e.g., `gpt-5-openai-code`) for improved model resolution flexibility
1493
+ - Migrated `AuthCredentialStore` and `AuthStorage` into `@sayknow-cli/ai` as shared credential primitives for downstream packages
1494
+ - Moved Anthropic auth helpers (`findAnthropicAuth`, `isOAuthToken`, `buildAnthropicSearchHeaders`, `buildAnthropicUrl`) into shared AI utilities for reuse across providers
1495
+ - Replaced `CliAuthStorage` with `AuthCredentialStore` for improved credential management with multiple credentials per provider
1496
+ - Updated models.json pricing for Anthropic model 3.5 Sonnet (input: 0.23→0.45, output: 3→2.2, added cache read: 0.225) and Anthropic model 3 Opus (input: 0.3→0.95)
1497
+ - Moved `mapAnthropicToolChoice` function from gitlab-duo provider to stream module for broader reusability
1498
+ - Enhanced HTTP status code extraction to handle string-formatted status codes in error objects
1499
+
1500
+ ### Removed
1501
+
1502
+ - Removed `CliAuthStorage` class in favor of new `AuthCredentialStore` with enhanced functionality
1503
+
1504
+ ## [12.17.2] - 2026-02-21
1505
+
1506
+ ### Added
1507
+
1508
+ - Exported `getAntigravityUserAgent()` function for constructing Antigravity User-Agent headers
1509
+
1510
+ ### Changed
1511
+
1512
+ - Updated default Antigravity version from 1.15.8 to 1.18.3
1513
+ - Unified User-Agent header generation across Antigravity API calls to use centralized `getAntigravityUserAgent()` function
1514
+
1515
+ ## [12.17.1] - 2026-02-21
1516
+
1517
+ ### Added
1518
+
1519
+ - Added new export paths for provider models via `./provider-models` and `./provider-models/*`
1520
+ - Added new export paths for Cursor and OpenAI code provider providers via `./providers/cursor/gen/*` and `./providers/openai-code/*`
1521
+ - Added new export paths for usage utilities via `./usage/*`
1522
+ - Added new export paths for discovery and OAuth utilities via `./utils/discovery` and `./utils/oauth` with subpath exports
1523
+
1524
+ ### Changed
1525
+
1526
+ - Simplified main export path to use wildcard pattern `./src/*.ts` for broader module access
1527
+ - Updated `models.json` export to include TypeScript declaration file at `./src/models.json.d.ts`
1528
+ - Reorganized package.json field ordering for improved readability
1529
+
1530
+ ## [12.17.0] - 2026-02-21
1531
+
1532
+ ### Fixed
1533
+
1534
+ - Cursor provider: bind `execHandlers` when passing handler methods to the exec protocol so handlers receive correct `this` context (fixes "undefined is not an object (evaluating 'this.options')" when using exec tools such as web search with Cursor)
1535
+
1536
+ ## [12.16.0] - 2026-02-21
1537
+
1538
+ ### Added
1539
+
1540
+ - Exported `readModelCache` and `writeModelCache` functions for direct SQLite-backed model cache access
1541
+ - Added `<turn_aborted>` guidance marker as synthetic user message when assistant messages are aborted or errored, informing the model that tools may have partially executed
1542
+ - Added support for Sonnet 4.6 models in adaptive thinking detection
1543
+
1544
+ ### Changed
1545
+
1546
+ - Updated model cache schema version to support improved global model fallback resolution
1547
+ - Improved GitHub Copilot model resolution to prefer provider-specific model definitions over global references when context window is larger, ensuring optimal model capabilities
1548
+ - Migrated model cache from per-provider JSON files to unified SQLite database (models.db) for atomic cross-process access
1549
+ - Renamed `cachePath` option to `cacheDbPath` in ModelManagerOptions to reflect database-backed storage
1550
+ - Improved non-authoritative cache handling with 5-minute retry backoff instead of retrying on every startup
1551
+ - Modified handling of aborted/errored assistant messages to preserve tool call structure instead of converting to text summaries, with synthetic 'aborted' tool results injected
1552
+ - Updated tool call tracking to use status map (Resolved/Aborted) instead of separate sets for better handling of duplicate and aborted tool results
1553
+
1554
+ ## [12.15.0] - 2026-02-20
1555
+
1556
+ ### Fixed
1557
+
1558
+ - Improved error messages for OAuth token refresh failures by including detailed error information from the provider
1559
+ - Separated rate limit and usage limit error handling to provide distinct user-friendly messages for ChatGPT rate limits vs subscription usage limits
1560
+
1561
+ ### Changed
1562
+
1563
+ - Increased SDK retry attempts to 5 for OpenAI, Azure OpenAI, and Anthropic clients (was SDK default of 2)
1564
+ - Changed 429 retry strategy for OpenAI code provider and Google Gemini CLI to use a 5-minute time budget when the server provides a retry delay, instead of a fixed attempt cap
1565
+
1566
+ ## [12.14.0] - 2026-02-19
1567
+
1568
+ ### Added
1569
+
1570
+ - Added `gemini-3.1-pro` model to opencode provider with text and image input support
1571
+ - Added `trinity-large-preview-free` model to opencode provider
1572
+ - Added `google/gemini-3.1-pro-preview` model to nanogpt provider
1573
+ - Added `google/gemini-3.1-pro-preview` model to openrouter provider with text and image input support
1574
+ - Added `gemini-3.1-pro` model to cursor provider
1575
+ - Added optional `intent` field to `ToolCall` interface for harness-level intent metadata
1576
+
1577
+ ### Changed
1578
+
1579
+ - Changed `big-pickle` model API from `openai-completions` to `anthropic-messages`
1580
+ - Changed `big-pickle` model baseUrl from `https://opencode.ai/zen/v1` to `https://opencode.ai/zen`
1581
+ - Changed `minimax-m2.5-free` model API from `openai-completions` to `anthropic-messages`
1582
+ - Changed `minimax-m2.5-free` model baseUrl from `https://opencode.ai/zen/v1` to `https://opencode.ai/zen`
1583
+
1584
+ ### Fixed
1585
+
1586
+ - Fixed tool argument validation to iteratively coerce nested JSON strings across multiple passes, enabling proper handling of deeply nested JSON-serialized objects and arrays
1587
+
1588
+ ## [12.13.0] - 2026-02-19
1589
+
1590
+ ### Added
1591
+
1592
+ - Added NanoGPT provider support with API-key login, dynamic model discovery from `https://nano-gpt.com/api/v1/models`, and text-model filtering for catalog/runtime discovery ([#111](https://github.com/jaybeyond/sayknow-cli/issues/111))
1593
+
1594
+ ## [12.12.3] - 2026-02-19
1595
+
1596
+ ### Fixed
1597
+
1598
+ - Fixed retry logic to recognize 'unable to connect' errors as transient failures
1599
+
1600
+ ## [12.11.3] - 2026-02-19
1601
+
1602
+ ### Fixed
1603
+
1604
+ - Fixed OpenAI code provider streaming to fail truncated responses that end without a terminal completion event, preventing partial outputs from being treated as successful completions.
1605
+ - Fixed OpenAI code websocket append fallback by resetting stale turn-state/model-etag session metadata when request shape diverges from appendable history.
1606
+
1607
+ ## [12.11.1] - 2026-02-19
1608
+
1609
+ ### Added
1610
+
1611
+ - Added support for Anthropic model 4.6 Opus and Sonnet models via Cursor API
1612
+ - Added support for Composer 1.5 model via Cursor API
1613
+ - Added support for GPT-5.1 OpenAI code Mini and GPT-5.1 High models via Cursor API
1614
+ - Added support for GPT-5.2 and GPT-5.3 OpenAI code variants (Fast, High, Low, Extra High) via Cursor API
1615
+ - Added HTTP/2 transport support for Cursor API requests (required by Cursor API)
1616
+
1617
+ ### Changed
1618
+
1619
+ - Updated pricing for Anthropic model 3.5 Sonnet model
1620
+ - Updated Anthropic model 3.5 Sonnet context window from 262,144 to 131,072 tokens
1621
+ - Simplified Cursor model display names by removing '(Cursor)' suffix
1622
+ - Changed Cursor API timeout from 15 seconds to 5 seconds
1623
+ - Switched Cursor API transport from HTTP/1.1 to HTTP/2
1624
+
1625
+ ## [12.11.0] - 2026-02-19
1626
+
1627
+ ### Added
1628
+
1629
+ - Added `priority` field to Model interface for provider-assigned model prioritization
1630
+ - Added `CatalogDiscoveryConfig` interface to standardize catalog discovery configuration across providers
1631
+ - Added type guards `isCatalogDescriptor()` and `allowsUnauthenticatedCatalogDiscovery()` for safer descriptor handling
1632
+ - Added `DEFAULT_MODEL_PER_PROVIDER` export from descriptors module for centralized default model management
1633
+ - Support for 11 new AI providers: Cloudflare AI Gateway, Hugging Face Inference, LiteLLM, Moonshot, NVIDIA, Ollama, Qianfan, Qwen Portal, Together, Venice, vLLM, and Xiaomi MiMo
1634
+ - Login flows for new providers with API key validation and OAuth token support
1635
+ - Extended `KnownProvider` type to include all newly supported providers
1636
+ - API key environment variable mappings for all new providers in service provider map
1637
+ - Model discovery and configuration for Cloudflare AI Gateway, Hugging Face, LiteLLM, Moonshot, NVIDIA, Ollama, Qianfan, Qwen Portal, Together, Venice, vLLM, and Xiaomi MiMo
1638
+
1639
+ ### Changed
1640
+
1641
+ - Refactored OAuth credential retrieval to simplify storage lifecycle management in model generation script
1642
+ - Parallelized special model discovery sources (Antigravity, OpenAI code) for improved generation performance
1643
+ - Reorganized model JSON structure to place `contextWindow` and `maxTokens` before `compat` field for consistency
1644
+ - Added `priority` field to OpenAI code provider models for provider-assigned model prioritization
1645
+ - Refactored provider descriptors to use helper functions (`descriptor`, `catalog`, `catalogDescriptor`) for reduced code duplication
1646
+ - Refactored models.dev provider descriptors to use helper functions (`simpleModelsDevDescriptor`, `openAiCompletionsDescriptor`, `anthropicMessagesDescriptor`) for improved maintainability
1647
+ - Unified provider descriptors into single source of truth in `descriptors.ts` for both runtime model discovery and catalog generation, improving maintainability
1648
+ - Refactored model generation script to use declarative `CatalogProviderDescriptor` interface instead of separate descriptor types, reducing code duplication
1649
+ - Reorganized models.dev provider descriptors into logical groups (Bedrock, Core, Coding Plans, Specialized) for better code organization
1650
+ - Simplified API resolution for OpenCode and GitHub Copilot providers using rule-based matching instead of inline conditionals
1651
+ - Refactored model generation script to use declarative provider descriptors instead of inline provider-specific logic, improving maintainability and reducing code duplication
1652
+ - Extracted model post-processing policies (cache pricing corrections, context window normalization) into dedicated `model-policies.ts` module for better testability and clarity
1653
+ - Removed static bundled models for Ollama and vLLM from `models.json` to rely on dynamic discovery instead, reducing static catalog size
1654
+ - Updated `OAuthProvider` type to include new provider identifiers
1655
+ - Expanded model registry (models.json) with thousands of new model entries across all new providers
1656
+ - Modified environment variable resolution to use `$pickenv` for providers with multiple possible env var names
1657
+ - Updated README documentation to list all newly supported providers and their authentication requirements
1658
+
1659
+ ## [12.10.1] - 2026-02-18
1660
+
1661
+ - Added Synthetic provider
1662
+ - Added API-key login helpers for Synthetic and Cerebras providers
1663
+
1664
+ ## [12.10.0] - 2026-02-18
1665
+
1666
+ ### Breaking Changes
1667
+
1668
+ - Renamed public API functions: `getModel()` → `getBundledModel()`, `getModels()` → `getBundledModels()`, `getProviders()` → `getBundledProviders()`
1669
+
1670
+ ### Added
1671
+
1672
+ - Exported `ModelManager` API for runtime-aware model resolution with dynamic endpoint discovery
1673
+ - Exported provider-specific model manager configuration helpers for Google, OpenAI-compatible, OpenAI code, and Cursor providers
1674
+ - Exported discovery utilities for fetching models from Antigravity, OpenAI code, Cursor, Gemini, and OpenAI-compatible endpoints
1675
+ - Added `createModelManager()` function to manage bundled and dynamically discovered models with configurable refresh strategies
1676
+ - Added support for on-disk model caching with TTL-based invalidation
1677
+ - Added `resolveProviderModels()` function for runtime model resolution across multiple providers
1678
+ - Added EU cross-region inference variants for Anthropic model Haiku 3.5 on Bedrock
1679
+ - Added Anthropic model Sonnet 4.6 and Anthropic model Sonnet 4.6 Thinking models to Antigravity provider
1680
+ - Added GLM-5 Free model via OpenCode provider
1681
+ - Added GLM-4.7-FlashX model via ZAI provider
1682
+ - Added MiniMax-M2.5-highspeed model across multiple providers (minimax-code, minimax-code-cn, minimax, minimax-cn)
1683
+ - Added Anthropic model Sonnet 4.6 model to OpenRouter provider
1684
+ - Added Qwen 3.5 Plus model to Vercel AI Gateway provider
1685
+ - Added Anthropic model Sonnet 4.6 model to Vercel AI Gateway provider
1686
+
1687
+ ### Changed
1688
+
1689
+ - Renamed `getModel()` to `getBundledModel()` to clarify it returns compile-time bundled models only
1690
+ - Renamed `getModels()` to `getBundledModels()` for consistency
1691
+ - Renamed `getProviders()` to `getBundledProviders()` for consistency
1692
+ - Refactored model generation script to use modular discovery functions instead of monolithic provider-specific logic
1693
+ - Updated models.json with new model entries and pricing updates across multiple providers
1694
+ - Updated pricing for deepseek/deepseek-v3 model on OpenRouter
1695
+ - Updated maxTokens from 65536 to 4096 for deepseek/deepseek-v3 on OpenRouter
1696
+ - Updated pricing and maxTokens for mistralai/mistral-large-2411 on OpenRouter
1697
+ - Updated pricing for qwen/qwen-max on Together AI
1698
+ - Updated pricing for qwen/qwen-vl-plus on Together AI
1699
+ - Updated pricing for qwen/qwen-plus on Together AI
1700
+ - Updated pricing for qwen/qwen-turbo on Together AI
1701
+ - Expanded EU cross-region inference variant support to all Anthropic model models on Bedrock (previously limited to Haiku, Sonnet, and Opus 4.5)
1702
+
1703
+ ## [12.8.0] - 2026-02-16
1704
+
1705
+ ### Added
1706
+
1707
+ - Added `contextPromotionTarget` model property to specify preferred fallback model when context promotion is triggered
1708
+ - Added automatic context promotion target assignment for Spark models to their base model equivalents
1709
+ - Added support for Brave search provider with BRAVE_API_KEY environment variable
1710
+
1711
+ ### Changed
1712
+
1713
+ - Updated Qwen model context window and max token limits for improved accuracy
1714
+
1715
+ ## [12.7.0] - 2026-02-16
1716
+
1717
+ ### Added
1718
+
1719
+ - Added DeepSeek-V3.2 model support via Amazon Bedrock
1720
+ - Added GLM-5 model support via OpenCode
1721
+ - Added MiniMax M2.5 model support via OpenCode
1722
+
1723
+ ### Changed
1724
+
1725
+ - Updated GLM-4.5, GLM-4.5-Air, GLM-4.5-Flash, GLM-4.5V, GLM-4.6, GLM-4.6V, GLM-4.7, GLM-4.7-Flash, and GLM-5 models to use anthropic-messages API instead of openai-completions
1726
+ - Updated GLM models base URL from https://api.z.ai/api/coding/paas/v4 to https://api.z.ai/api/anthropic
1727
+ - Updated pricing for multiple models including Mistral, Moonshot, and Qwen variants
1728
+ - Updated context window and max tokens for several models to reflect accurate specifications
1729
+
1730
+ ### Removed
1731
+
1732
+ - Removed compat field with supportsDeveloperRole and thinkingFormat properties from GLM models
1733
+
1734
+ ## [12.6.0] - 2026-02-16
1735
+
1736
+ ### Added
1737
+
1738
+ - Added source-scoped custom API and OAuth provider registration helpers for extension-defined providers.
1739
+
1740
+ ### Changed
1741
+
1742
+ - Expanded `Api` typing to allow extension-defined API identifiers while preserving built-in API exhaustiveness checks.
1743
+
1744
+ ### Fixed
1745
+
1746
+ - Fixed custom API registration to reject built-in API identifiers and prevent accidental provider overrides.
1747
+
1748
+ ## [12.2.0] - 2026-02-13
1749
+
1750
+ ### Added
1751
+
1752
+ - Added automatic retry logic for WebSocket stream closures before response completion, with configurable retry budget to improve reliability on flaky connections
1753
+ - Added `providerSessionState` option to enable provider-scoped mutable state persistence across agent turns
1754
+ - Added WebSocket retry logic with configurable retry budget and delay via `PI_OPENAI_CODE_WEBSOCKET_RETRY_BUDGET` and `PI_OPENAI_CODE_WEBSOCKET_RETRY_DELAY_MS` environment variables
1755
+ - Added WebSocket idle timeout detection via `PI_OPENAI_CODE_WEBSOCKET_IDLE_TIMEOUT_MS` environment variable to fail stalled connections
1756
+ - Added WebSocket v2 beta header support via `PI_OPENAI_CODE_WEBSOCKET_V2` environment variable for newer OpenAI API versions
1757
+ - Added WebSocket handshake header capture to extract and replay session metadata (turn state, models etag, reasoning flags) across SSE fallback requests
1758
+ - Added `preferWebsockets` option to enable WebSocket transport for OpenAI code provider responses when supported
1759
+ - Added `prewarmOpenAIOpenAI codeResponses()` function to establish and reuse WebSocket connections across multiple requests
1760
+ - Added `getOpenAIOpenAI codeTransportDetails()` function to inspect transport layer details including WebSocket status and fallback information
1761
+ - Added `getProviderDetails()` function to retrieve formatted provider configuration and transport information
1762
+ - Added automatic fallback from WebSocket to SSE when connection fails, with transparent retry logic
1763
+ - Added session state management to reuse WebSocket connections and enable request appending across turns
1764
+ - Added support for x-openai-code-turn-state header to maintain conversation state across SSE requests
1765
+
1766
+ ### Changed
1767
+
1768
+ - Changed WebSocket session state storage from global maps to provider-scoped session state for multi-agent isolation
1769
+ - Changed WebSocket connection initialization to accept idle timeout configuration and handshake header callbacks
1770
+ - Changed WebSocket error handling to use standardized transport error messages with `OpenAI code websocket transport error` prefix
1771
+ - Changed WebSocket retry behavior to retry transient failures before activating sticky fallback, improving reliability on flaky connections
1772
+ - Changed OpenAI code provider model configuration to prefer WebSocket transport by default with `preferWebsockets: true`
1773
+ - Changed header handling to use appropriate OpenAI-Beta header values for WebSocket vs SSE transports
1774
+ - Perplexity OAuth token refresh now uses JWT expiry extraction instead of Socket.IO RPC, improving reliability when server is unreachable
1775
+ - Removed Socket.IO client implementation for Perplexity token refresh; tokens are now validated using embedded JWT expiry claims
1776
+
1777
+ ### Removed
1778
+
1779
+ - Removed `refreshPerplexityToken` export; token refresh is now handled internally via JWT expiry detection
1780
+
1781
+ ### Fixed
1782
+
1783
+ - Fixed WebSocket stream retry logic to properly handle mid-stream connection closures and retry before falling back to SSE transport
1784
+ - Fixed `preferWebsockets` option handling to correctly respect explicit `false` values when determining transport preference
1785
+ - Fixed WebSocket append state not being reset after aborted requests, preventing stale state from affecting subsequent turns
1786
+ - Fixed WebSocket append state not being reset after stream errors, preventing failed append attempts from blocking future requests
1787
+ - Fixed OpenAI code model context window metadata to use 272000 input tokens (instead of 400000 total budget) for non-Spark OpenAI code variants
1788
+
1789
+ ## [12.0.0] - 2026-02-12
1790
+
1791
+ ### Added
1792
+
1793
+ - Added GPT-5.3 OpenAI code Spark model with 128K context window and extended reasoning capabilities
1794
+ - Added MiniMax M2.5 and M2.5 Lightning models via OpenAI-compatible API (minimax-code provider)
1795
+ - Added MiniMax M2.5 and M2.5 Lightning models via OpenAI-compatible API (minimax-code-cn provider for China region)
1796
+ - Added MiniMax M2.5 and M2.5 Lightning models via Anthropic API (minimax and minimax-cn providers)
1797
+ - Added Llama 3.1 8B model via Cerebras API
1798
+ - Added MiniMax M2.5 model via OpenRouter
1799
+ - Added MiniMax M2.5 model via Vercel AI Gateway
1800
+ - Added MiniMax M2.5 Free model via OpenCode
1801
+ - Added Qwen3 VL 32B Instruct multimodal model via OpenRouter
1802
+
1803
+ ### Changed
1804
+
1805
+ - Updated Z.ai GLM-5 pricing and context window configuration on OpenRouter
1806
+ - Updated Qwen3 Max Thinking max tokens from 32768 to 65536 on OpenRouter
1807
+ - Updated OpenAI GPT-5 Image Mini pricing on OpenRouter
1808
+ - Updated OpenAI GPT-5 Pro pricing and context window on OpenRouter
1809
+ - Updated OpenAI o4-mini pricing and context window on OpenRouter
1810
+ - Updated Anthropic model Opus 4.5 Thinking model name formatting (removed parentheses)
1811
+ - Updated Anthropic model Opus 4.6 Thinking model name formatting (removed parentheses)
1812
+ - Updated Anthropic model Sonnet 4.5 Thinking model name formatting (removed parentheses)
1813
+ - Updated Gemini 2.5 Flash Thinking model name formatting (removed parentheses)
1814
+ - Updated Gemini 3 Pro High and Low model name formatting (removed parentheses)
1815
+ - Updated GPT-OSS 120B Medium model name formatting (removed parentheses) and context window to 131072
1816
+
1817
+ ### Removed
1818
+
1819
+ - Removed GLM-5 model from Z.ai provider
1820
+ - Removed Trinity Large Preview Free model from OpenCode provider
1821
+ - Removed MiniMax M2.1 Free model from OpenCode provider
1822
+ - Removed deprecated Anthropic model entries: `anthropic-model-3-5-haiku-latest`, `anthropic-model-3-5-haiku-20241022`, `anthropic-model-3-7-sonnet-20250219`, `anthropic-model-3-7-sonnet-latest`, `anthropic-model-3-opus-20240229`, `anthropic-model-3-sonnet-20240229` ([#33](https://github.com/jaybeyond/sayknow-cli/issues/33))
1823
+
1824
+ ### Fixed
1825
+
1826
+ - Added deprecation filter in model generation script to prevent re-adding deprecated Anthropic models ([#33](https://github.com/jaybeyond/sayknow-cli/issues/33))
1827
+
1828
+ ## [11.14.1] - 2026-02-12
1829
+
1830
+ ### Added
1831
+
1832
+ - Added prompt-caching-scope-2026-01-05 beta feature support
1833
+
1834
+ ### Changed
1835
+
1836
+ - Updated Anthropic Code version header to 2.1.39
1837
+ - Updated runtime version header to v24.13.1 and package version to 0.73.0
1838
+ - Increased request timeout from 60s to 600s
1839
+ - Reordered Accept-Encoding header values for compression preference
1840
+ - Updated OAuth authorization and token endpoints to use platform.anthropic-model.com
1841
+ - Expanded OAuth scopes to include user:sessions:anthropic-model_code and user:mcp_servers
1842
+
1843
+ ### Removed
1844
+
1845
+ - Removed anthropic-model-code-20250219 beta feature from default models
1846
+ - Removed fine-grained-tool-streaming-2025-05-14 beta feature
1847
+
1848
+ ## [11.13.1] - 2026-02-12
1849
+
1850
+ ### Added
1851
+
1852
+ - Added Perplexity (Pro/Max) OAuth login support via native macOS app extraction or email OTP authentication
1853
+ - Added `loginPerplexity` and `refreshPerplexityToken` functions for Perplexity account integration
1854
+ - Added Socket.IO v4 client implementation for authenticated WebSocket communication with Perplexity API
1855
+
1856
+ ## [11.12.0] - 2026-02-11
1857
+
1858
+ ### Changed
1859
+
1860
+ - Increased maximum retry attempts for OpenAI code requests from 2 to 5 to improve reliability on transient failures
1861
+
1862
+ ### Fixed
1863
+
1864
+ - Fixed tool result content handling in Anthropic provider to provide fallback error message when content is empty
1865
+ - Improved retry delay calculation to parse delay values from error response bodies (e.g., 'Please try again in 225ms')
1866
+
1867
+ ## [11.11.0] - 2026-02-10
1868
+
1869
+ ### Breaking Changes
1870
+
1871
+ - Replaced `./models.generated` export with `./models.json` - update imports from `import { MODELS } from './models.generated'` to `import MODELS from './models.json' with { type: 'json' }`
1872
+
1873
+ ### Added
1874
+
1875
+ - Added TypeScript type declarations for `models.json` to enable proper type inference when importing the JSON file
1876
+
1877
+ ### Changed
1878
+
1879
+ - Updated available models in google-antigravity provider with new model variants and updated context window/token limits
1880
+ - Simplified type signatures for `getModel()` and `getModels()` functions for improved usability
1881
+ - Changed models export from TypeScript module to JSON format for improved performance and reduced bundle size
1882
+ - Updated `@anthropic-ai/sdk` dependency from ^0.72.1 to ^0.74.0
1883
+
1884
+ ## [11.10.0] - 2026-02-10
1885
+
1886
+ ### Added
1887
+
1888
+ - Added support for Kimi K2, K2 Turbo Preview, and K2.5 models with reasoning capabilities
1889
+
1890
+ ### Fixed
1891
+
1892
+ - Fixed Anthropic model Opus 4.6 context window to 200K across all providers (was incorrectly set to 1M)
1893
+ - Fixed Anthropic model Sonnet 4 context window to 200K across multiple providers (was incorrectly set to 1M)
1894
+
1895
+ ## [11.8.0] - 2026-02-10
1896
+
1897
+ ### Added
1898
+
1899
+ - Added `auto` model alias for OpenRouter with automatic model routing
1900
+ - Added `openrouter/aurora-alpha` model with reasoning capabilities
1901
+ - Added `qwen/qwen3-max-thinking` model with extended context window support
1902
+ - Added support for `parametersJsonSchema` in Google Gemini tool definitions for improved JSON Schema compatibility
1903
+
1904
+ ### Changed
1905
+
1906
+ - Updated Anthropic model Sonnet 4 and 4.5 context window from 1M to 200K tokens to reflect actual limits
1907
+ - Updated Anthropic model Opus 4.6 context window to 200K tokens across providers
1908
+ - Changed default `reasoningSummary` for OpenAI code provider from `undefined` to `auto`
1909
+ - Updated Qwen model pricing and context window specifications across multiple variants
1910
+ - Modified Google Gemini CLI system instruction to use compact format
1911
+ - Changed tool parameter handling for Anthropic model models on Google Cloud Code Assist to use legacy `parameters` field for API translation
1912
+
1913
+ ### Removed
1914
+
1915
+ - Removed `glm-4.7-free` model from OpenCode provider
1916
+ - Removed `qwen3-coder` model from OpenCode provider
1917
+ - Removed `ai21/jamba-mini-1.7` model from OpenRouter
1918
+ - Removed `stepfun-ai/step3` model from OpenRouter
1919
+ - Removed duplicate test suite for Google Antigravity Provider with `gemini-3-pro-high`
1920
+
1921
+ ### Fixed
1922
+
1923
+ - Fixed Amazon Bedrock HTTP/1.1 handler import to use direct import instead of dynamic import
1924
+ - Fixed Qwen model context window and pricing inconsistencies across OpenRouter
1925
+ - Fixed cache read pricing for multiple Qwen models
1926
+ - Fixed OpenAI code provider reasoning effort clamping for `gpt-5.3-openai-code` model
1927
+
1928
+ ## [11.7.1] - 2026-02-07
1929
+
1930
+ ### Added
1931
+
1932
+ - Added Anthropic model Opus 4.6 Thinking model for Antigravity provider
1933
+ - Added Gemini 2.5 Flash, Gemini 2.5 Flash Thinking, and Gemini 2.5 Pro models for Antigravity provider
1934
+ - Added Pony Alpha model via OpenRouter
1935
+
1936
+ ### Changed
1937
+
1938
+ - Updated Antigravity models to use free tier pricing (0 cost) across all models
1939
+ - Changed Antigravity model fetching to dynamically load from API when credentials are available, with hardcoded fallback models
1940
+ - Updated Anthropic model Opus 4.6 context window from 200,000 to 1,000,000 tokens across Bedrock regions
1941
+ - Updated Anthropic model Opus 4.6 cache pricing from 1.5/18.75 to 0.5/6.25 for EU and US regions
1942
+ - Updated Antigravity model pricing to free tier (0 cost) for Anthropic model Opus 4.5 Thinking, Anthropic model Sonnet 4.5 Thinking, Gemini 3 Flash, Gemini 3 Pro variants, and GPT-OSS 120B Medium
1943
+ - Updated GPT-OSS 120B Medium reasoning capability from false to true
1944
+ - Updated Gemini 3 Flash max tokens from 65,535 to 65,536
1945
+ - Updated Anthropic model Opus 4.5 Thinking display name formatting to include parentheses
1946
+ - Updated various model pricing and context window parameters across OpenRouter and other providers
1947
+ - Removed Anthropic model Opus 4.6 20260205 model from Anthropic provider
1948
+
1949
+ ### Fixed
1950
+
1951
+ - Fixed Anthropic model Opus 4.6 model ID format by removing version suffix (:0) in Bedrock configurations
1952
+ - Fixed Llama 3.1 70B Instruct pricing and context window parameters
1953
+ - Fixed Mistral model pricing and cache read costs
1954
+ - Fixed DeepSeek and other model pricing inconsistencies
1955
+ - Fixed Qwen model pricing and token limits
1956
+ - Fixed GLM model pricing and context window specifications
1957
+
1958
+ ## [11.6.0] - 2026-02-07
1959
+
1960
+ ### Added
1961
+
1962
+ - Added Bedrock cache retention support with `PI_CACHE_RETENTION` env var and per-request `cacheRetention` option
1963
+ - Added adaptive thinking support for Bedrock Opus 4.6+ models
1964
+ - Added `AWS_BEDROCK_SKIP_AUTH` env var to support unauthenticated Bedrock proxies
1965
+ - Added `AWS_BEDROCK_FORCE_HTTP1` env var to force HTTP/1.1 for custom Bedrock endpoints
1966
+ - Re-exported `Static`, `TSchema`, and `Type` from `@sinclair/typebox`
1967
+
1968
+ ### Fixed
1969
+
1970
+ - Fixed OpenAI Responses storage disabled by default (`store: false`)
1971
+ - Fixed reasoning effort clamping for gpt-5.3 OpenAI code models (minimal -> low)
1972
+ - Fixed Bedrock `supportsPromptCaching` to also check model cost fields
1973
+
1974
+ ## [11.5.1] - 2026-02-07
1975
+
1976
+ ### Fixed
1977
+
1978
+ - Fixed schema normalization to handle array-valued `type` fields by converting them to a single type with nullable flag for Google provider compatibility
1979
+
1980
+ ## [11.3.0] - 2026-02-06
1981
+
1982
+ ### Added
1983
+
1984
+ - Added `cacheRetention` option to control prompt cache retention preference ('none', 'short', 'long') across providers
1985
+ - Added `maxRetryDelayMs` option to cap server-requested retry delays and fail fast when delays exceed the limit
1986
+ - Added `effort` option for Anthropic Opus 4.6+ models to control adaptive thinking effort levels ('low', 'medium', 'high', 'max')
1987
+ - Added support for Anthropic Opus 4.6+ adaptive thinking mode that lets Anthropic model decide when and how much to think
1988
+ - Added `PI_AI_ANTIGRAVITY_VERSION` environment variable to customize Antigravity sandbox endpoint version
1989
+ - Exported `convertAnthropicMessages` function for converting message formats to Anthropic API
1990
+ - Automatic fallback for Anthropic assistant-prefill requests: appends synthetic user "Continue." message when conversation ends with assistant turn to maintain API compatibility
1991
+
1992
+ ### Changed
1993
+
1994
+ - Changed `supportsXhigh()` to include GPT-5.1 OpenAI code Max and broaden Anthropic support to all Anthropic Messages API models with budget-based thinking capability
1995
+ - Changed Anthropic thinking mode to use adaptive thinking for Opus 4.6+ models instead of budget-based thinking
1996
+ - Changed `supportsXhigh()` to support GPT-5.2/5.3 and Anthropic Opus 4.6+ models with adaptive thinking
1997
+ - Changed prompt caching to respect `cacheRetention` option and support TTL configuration for Anthropic
1998
+ - Changed OpenAI tool definitions to conditionally include `strict` field only when provider supports it
1999
+ - Changed Qwen model support to use `enable_thinking` boolean parameter instead of OpenAI-style reasoning_effort
2000
+
2001
+ ### Fixed
2002
+
2003
+ - Fixed indentation and formatting in `convertAnthropicMessages` function
2004
+ - Fixed handling of conversations ending with assistant messages on Anthropic-routed models that reject assistant prefill requests
2005
+
2006
+ ## [11.2.3] - 2026-02-05
2007
+
2008
+ ### Added
2009
+
2010
+ - Added Anthropic model Opus 4.6 model support across multiple providers (Anthropic, Amazon Bedrock, GitHub Copilot, OpenRouter, OpenCode, Vercel AI Gateway)
2011
+ - Added GPT-5.3 OpenAI code model support for OpenAI
2012
+ - Added `readSseJson` utility import for improved SSE stream handling in Google Gemini CLI provider
2013
+
2014
+ ### Changed
2015
+
2016
+ - Updated Google Gemini CLI provider to use `readSseJson` utility for cleaner SSE stream parsing
2017
+ - Updated pricing for Llama 3.1 405B model on Vercel AI Gateway (cache read rate adjusted)
2018
+ - Updated Llama 3.1 405B context window and max tokens on Vercel AI Gateway (256000 for both)
2019
+
2020
+ ### Removed
2021
+
2022
+ - Removed Kimi K2, Kimi K2 Turbo Preview, and Kimi K2.5 models
2023
+ - Removed Deep Cogito Cogito V2 Preview models from OpenRouter
2024
+
2025
+ ## [11.0.0] - 2026-02-05
2026
+
2027
+ ### Changed
2028
+
2029
+ - Replaced direct `Bun.env` access with `getEnv()` utility from `@sayknow-cli/utils` for consistent environment variable handling across all providers
2030
+ - Updated environment variable names from `SKC_*` prefix to `PI_*` prefix for consistency (e.g., `SKC_CODING_AGENT_DIR` → `PI_CODING_AGENT_DIR`)
2031
+
2032
+ ### Removed
2033
+
2034
+ - Removed automatic environment variable migration from `PI_*` to `SKC_*` prefixes via `migrate-env.ts` module
2035
+
2036
+ ## [10.5.0] - 2026-02-04
2037
+
2038
+ ### Changed
2039
+
2040
+ - Updated @anthropic-ai/sdk to ^0.72.1
2041
+ - Updated @aws-sdk/client-bedrock-runtime to ^3.982.0
2042
+ - Updated @google/genai to ^1.39.0
2043
+ - Updated @smithy/node-http-handler to ^4.4.9
2044
+ - Updated openai to ^6.17.0
2045
+ - Updated @types/node to ^25.2.0
2046
+
2047
+ ### Removed
2048
+
2049
+ - Removed proxy-agent dependency
2050
+ - Removed undici dependency
2051
+
2052
+ ## [9.4.0] - 2026-01-31
2053
+
2054
+ ### Added
2055
+
2056
+ - Added `getEnv()` function to retrieve environment variables from Bun.env, cwd/.env, or ~/.env
2057
+ - Added support for reading .env files from home directory and current working directory
2058
+ - Added support for `exa` and `perplexity` as known providers in `getEnvApiKey()`
2059
+
2060
+ ### Changed
2061
+
2062
+ - Changed `getEnvApiKey()` to check Bun.env, cwd/.env, and ~/.env files in order of precedence
2063
+ - Refactored provider API key resolution to use a declarative service provider map
2064
+
2065
+ ## [9.2.2] - 2026-01-31
2066
+
2067
+ ### Added
2068
+
2069
+ - Added OpenCode Zen provider with API key authentication for accessing multiple AI models
2070
+ - Added 4 new free models via OpenCode: glm-4.7-free, kimi-k2.5-free, minimax-m2.1-free, trinity-large-preview-free
2071
+ - Added glm-4.7-flash model via Zai provider
2072
+ - Added Kimi Code provider with OpenAI and Anthropic API format support
2073
+ - Added prompt cache retention support with PI_CACHE_RETENTION env var
2074
+ - Added overflow patterns for Bedrock, MiniMax, Kimi; reclassified 429 as rate limiting
2075
+ - Added profile endpoint integration to resolve user emails with 24-hour caching
2076
+ - Added automatic token refresh for expired Kimi OAuth credentials
2077
+ - Added Kimi Code OAuth handler with device authorization flow
2078
+ - Added Kimi Code usage provider with quota caching
2079
+ - Added 4 new Kimi Code models (kimi-for-coding, kimi-k2, kimi-k2-turbo-preview, kimi-k2.5)
2080
+ - Added Kimi Code provider integration with OAuth and token management
2081
+ - Added tool-choice utility for mapping unified ToolChoice to provider-specific formats
2082
+ - Added ToolChoice type for controlling tool selection (auto, none, any, required, function)
2083
+
2084
+ ### Changed
2085
+
2086
+ - Updated Kimi K2.5 cache read pricing from 0.1 to 0.08
2087
+ - Updated MiniMax M2 pricing: input 0.6→0.6, output 3→3, cache read 0.1→0.09999999999999999
2088
+ - Updated OpenRouter DeepSeek V3.1 pricing and max tokens: input 0.6→0.5, output 3→2.8, maxTokens 262144→4096
2089
+ - Updated OpenRouter DeepSeek R1 pricing and max tokens: input 0.06→0.049999999999999996, output 0.24→0.19999999999999998, maxTokens 262144→4096
2090
+ - Updated Anthropic Anthropic model 3.5 Sonnet max tokens from 256000 to 65536 on OpenRouter
2091
+ - Updated Vercel AI Gateway Anthropic model 3.5 Sonnet cache read pricing from 0.125 to 0.13
2092
+ - Updated Vercel AI Gateway Anthropic model 3.5 Sonnet New cache read pricing from 0.125 to 0.13
2093
+ - Updated Vercel AI Gateway GPT-5.2 cache read pricing from 0.175 to 0.18 and display name to 'GPT 5.2'
2094
+ - Updated Zai GLM-4.6 cache read pricing from 0.024999999999999998 to 0.03
2095
+ - Updated Zai Qwen QwQ max tokens from 66000 to 16384
2096
+ - Added delta event batching and throttling (50ms, 20 updates/sec max) to AssistantMessageEventStream
2097
+ - Updated MiniMax-M2 pricing: input 1.2→0.6, output 1.2→3, cacheRead 0.6→0.1
2098
+
2099
+ ### Removed
2100
+
2101
+ - Removed OpenRouter google/gemini-2.0-flash-exp:free model
2102
+ - Removed Vercel AI Gateway stealth/sonoma-dusk-alpha and stealth/sonoma-sky-alpha models
2103
+
2104
+ ### Fixed
2105
+
2106
+ - Fixed rate limit issues with Kimi models by always sending max_tokens
2107
+ - Added handling for sensitive stop reason from Anthropic API safety filters
2108
+ - Added optional chaining for safer JSON schema property access in Anthropic provider
2109
+
2110
+ ## [8.6.0] - 2026-01-27
2111
+
2112
+ ### Changed
2113
+
2114
+ - Replaced JSON5 dependency with Bun.JSON5 parsing
2115
+
2116
+ ### Fixed
2117
+
2118
+ - Filtered empty user text blocks for OpenAI-compatible completions and normalized Kimi reasoning_content for OpenRouter tool-call messages
2119
+
2120
+ ## [8.4.0] - 2026-01-25
2121
+
2122
+ ### Added
2123
+
2124
+ - Added Azure OpenAI Responses provider with deployment mapping and resource-based base URL support
2125
+
2126
+ ### Changed
2127
+
2128
+ - Added OpenRouter routing preferences for OpenAI-compatible completions
2129
+
2130
+ ### Fixed
2131
+
2132
+ - Defaulted Google tool call arguments to empty objects when providers omit args
2133
+ - Guarded Responses/OpenAI code streaming deltas against missing content parts and handled arguments.done events
2134
+
2135
+ ## [8.2.1] - 2026-01-24
2136
+
2137
+ ### Fixed
2138
+
2139
+ - Fixed handling of streaming function call arguments in OpenAI responses to properly parse arguments when sent via `response.function_call_arguments.done` events
2140
+
2141
+ ## [8.2.0] - 2026-01-24
2142
+
2143
+ ### Changed
2144
+
2145
+ - Migrated node module imports from named to namespace imports across all packages for consistency with project guidelines
2146
+
2147
+ ## [8.0.0] - 2026-01-23
2148
+
2149
+ ### Fixed
2150
+
2151
+ - Fixed OpenAI Responses API 400 error "function_call without required reasoning item" when switching between models (same provider, different model). The fix omits the `id` field for function_calls from different models to avoid triggering OpenAI's reasoning/function_call pairing validation
2152
+ - Fixed 400 errors when reading multiple images via GitHub Copilot's Anthropic model models. Anthropic model requires tool_use -> tool_result adjacency with no user messages interleaved. Images from consecutive tool results are now batched into a single user message
2153
+
2154
+ ## [7.0.0] - 2026-01-21
2155
+
2156
+ ### Added
2157
+
2158
+ - Added usage tracking system with normalized schema for provider quota/limit endpoints
2159
+ - Added Anthropic model usage provider for 5-hour and 7-day quota windows
2160
+ - Added GitHub Copilot usage provider for chat, completions, and premium requests
2161
+ - Added Google Antigravity usage provider for model quota tracking
2162
+ - Added Google Gemini CLI usage provider for tier-based quota monitoring
2163
+ - Added OpenAI code provider usage provider for primary and secondary rate limit windows
2164
+ - Added ZAI usage provider for token and request quota tracking
2165
+
2166
+ ### Changed
2167
+
2168
+ - Updated Anthropic model usage provider to extract account identifiers from response headers
2169
+ - Updated GitHub Copilot usage provider to include account identifiers in usage reports
2170
+ - Updated Google Gemini CLI usage provider to handle missing reset time gracefully
2171
+
2172
+ ### Fixed
2173
+
2174
+ - Fixed GitHub Copilot usage provider to simplify token handling and improve reliability
2175
+ - Fixed GitHub Copilot usage provider to properly resolve account identifiers for OAuth credentials
2176
+ - Fixed API validation errors when sending empty user messages (resume with `.`) across all providers:
2177
+ - Google Cloud Code Assist (google-shared.ts)
2178
+ - OpenAI Responses API (openai-responses.ts)
2179
+ - OpenAI code provider Responses API (openai-code-responses.ts)
2180
+ - Cursor (cursor.ts)
2181
+ - Amazon Bedrock (amazon-bedrock.ts)
2182
+ - Clamped OpenAI code provider reasoning effort "minimal" to "low" for gpt-5.2 models to avoid API errors
2183
+ - Fixed GitHub Copilot usage fallback to internal quota endpoints when billing usage is unavailable
2184
+ - Fixed GitHub Copilot usage metadata to include account identifiers for report dedupe
2185
+ - Fixed Anthropic usage metadata extraction to include account identifiers when provided by the usage endpoint
2186
+ - Fixed Gemini CLI usage windows to consistently label quota windows for display suppression
2187
+
2188
+ ## [6.9.69] - 2026-01-21
2189
+
2190
+ ### Added
2191
+
2192
+ - Added duration and time-to-first-token (ttft) metrics to all AI provider responses
2193
+ - Added performance tracking for streaming responses across all providers
2194
+
2195
+ ## [6.9.0] - 2026-01-21
2196
+
2197
+ ### Removed
2198
+
2199
+ - Removed openai-code provider exports from main package index
2200
+ - Removed openai-code prompt utilities and moved them inline
2201
+ - Removed vitest configuration file
2202
+
2203
+ ## [6.8.4] - 2026-01-21
2204
+
2205
+ ### Changed
2206
+
2207
+ - Updated prompt caching strategy to follow Anthropic's recommended hierarchy
2208
+ - Fixed token usage tracking to properly handle cumulative output tokens from message_delta events
2209
+ - Improved message validation to filter out empty or invalid content blocks
2210
+ - Increased OAuth callback timeout from 120 seconds to 120,000 milliseconds
2211
+
2212
+ ## [6.8.3] - 2026-01-21
2213
+
2214
+ ### Added
2215
+
2216
+ - Added `headers` option to all providers for custom request headers
2217
+ - Added `onPayload` hook to observe provider request payloads before sending
2218
+ - Added `strictResponsesPairing` option for Azure OpenAI Responses API compatibility
2219
+ - Added `originator` option to `loginOpenAIOpenAI code` for custom OAuth flow identification
2220
+ - Added per-request `headers` and `onPayload` hooks to `StreamOptions`
2221
+ - Added `originator` option to `loginOpenAIOpenAI code`
2222
+
2223
+ ### Fixed
2224
+
2225
+ - Fixed tool call ID normalization for OpenAI Responses API cross-provider handoffs
2226
+ - Skipped errored or aborted assistant messages during cross-provider transforms
2227
+ - Detected AWS ECS/IRSA credentials for Bedrock authentication checks
2228
+ - Detected AWS ECS/IRSA credentials for Bedrock authentication checks
2229
+ - Normalized Responses API tool call IDs during handoffs and refreshed handoff tests
2230
+ - Enforced strict tool call/result pairing for Azure OpenAI Responses API
2231
+ - Skipped errored or aborted assistant messages during cross-provider transforms
2232
+
2233
+ ### Security
2234
+
2235
+ - Enhanced AWS credential detection to support ECS task roles and IRSA web identity tokens
2236
+
2237
+ ## [6.8.2] - 2026-01-21
2238
+
2239
+ ### Fixed
2240
+
2241
+ - Improved error handling for aborted requests in Google Gemini CLI provider
2242
+ - Enhanced OAuth callback flow to handle manual input errors gracefully
2243
+ - Fixed login cancellation handling in GitHub Copilot OAuth flow
2244
+ - Removed fallback manual input from OpenAI code provider OAuth flow
2245
+
2246
+ ### Security
2247
+
2248
+ - Hardened database file permissions to prevent credential leakage
2249
+ - Set secure directory permissions (0o700) for credential storage
2250
+
2251
+ ## [6.8.0] - 2026-01-20
2252
+
2253
+ ### Added
2254
+
2255
+ - Added `logout` command to CLI for OAuth provider logout
2256
+ - Added `status` command to show logged-in providers and token expiry
2257
+ - Added persistent credential storage using SQLite database
2258
+ - Added OAuth callback server with automatic port fallback
2259
+ - Added HTML callback page with success/error states
2260
+ - Added support for Cursor OAuth provider
2261
+
2262
+ ### Changed
2263
+
2264
+ - Updated Promise.withResolvers usage for better compatibility
2265
+ - Replaced custom sleep implementations with Bun.sleep and abortableSleep
2266
+ - Simplified SSE stream parsing using readLines utility
2267
+ - Updated test framework from vitest to bun:test
2268
+ - Replaced temp directory creation with TempDir API
2269
+ - Changed credential storage from auth.json to ~/.skc/agent/agent.db
2270
+ - Changed CLI command examples from npx to bunx
2271
+ - Refactored OAuth flows to use common callback server base class
2272
+ - Updated OAuth provider interfaces to use controller pattern
2273
+
2274
+ ### Fixed
2275
+
2276
+ - Fixed OAuth callback handling with improved error states
2277
+ - Fixed token refresh for all OAuth providers
2278
+
2279
+ ## [6.7.670] - 2026-01-19
2280
+
2281
+ ### Changed
2282
+
2283
+ - Updated Anthropic Code compatibility headers and version
2284
+ - Improved OAuth token handling with proper state generation
2285
+ - Enhanced cache control for tool and user message blocks
2286
+ - Simplified tool name prefixing for OAuth traffic
2287
+ - Updated PKCE verifier generation for better security
2288
+
2289
+ ## [5.7.67] - 2026-01-18
2290
+
2291
+ ### Fixed
2292
+
2293
+ - Added error handling for unknown OAuth providers
2294
+
2295
+ ## [5.6.77] - 2026-01-18
2296
+
2297
+ ### Fixed
2298
+
2299
+ - Prevented duplicate tool results for errored or aborted messages when results already exist
2300
+
2301
+ ## [5.6.7] - 2026-01-18
2302
+
2303
+ ### Added
2304
+
2305
+ - Added automatic retry logic for OpenAI code provider responses with configurable delay and max retries
2306
+ - Added tool call ID sanitization for Amazon Bedrock to ensure valid characters
2307
+ - Added tool argument validation that coerces JSON-encoded strings for expected non-string types
2308
+
2309
+ ### Changed
2310
+
2311
+ - Updated environment variable prefix from PI* to SKC* for better consistency
2312
+ - Added automatic migration for legacy PI* environment variables to SKC* equivalents
2313
+ - Adjusted Bedrock Anthropic model thinking budgets to reserve output tokens when maxTokens is too low
2314
+
2315
+ ### Fixed
2316
+
2317
+ - Fixed orphaned tool call handling to ensure proper tool_use/tool_result pairing for all assistant messages
2318
+ - Fixed message transformation to insert synthetic tool results for errored/aborted assistant messages with tool calls
2319
+ - Fixed tool prefix handling in Anthropic model provider to use case-insensitive comparison
2320
+ - Fixed Gemini 3 model handling to treat unsigned tool calls as context-only with anti-mimicry context
2321
+ - Fixed message transformation to filter out empty error messages from conversation history
2322
+ - Fixed OpenAI completions provider compatibility detection to use provider metadata
2323
+ - Fixed OpenAI completions provider to avoid using developer role for opencode provider
2324
+ - Fixed orphaned tool call handling to skip synthetic results for errored assistant messages
2325
+
2326
+ ## [5.5.0] - 2026-01-18
2327
+
2328
+ ### Changed
2329
+
2330
+ - Updated User-Agent header from 'opencode' to 'pi' for OpenAI code provider requests
2331
+ - Simplified OpenAI code system prompt instructions
2332
+ - Removed bridge text override from OpenAI code system prompt builder
2333
+
2334
+ ## [5.3.0] - 2026-01-15
2335
+
2336
+ ### Changed
2337
+
2338
+ - Replaced detailed OpenAI code system instructions with simplified pi assistant instructions
2339
+ - Updated internal documentation references to use pi-internal:// protocol
2340
+
2341
+ ## [5.1.0] - 2026-01-14
2342
+
2343
+ ### Added
2344
+
2345
+ - Added Amazon Bedrock provider with `bedrock-converse-stream` API for Anthropic model models via AWS
2346
+ - Added MiniMax provider with OpenAI-compatible API
2347
+ - Added EU cross-region inference model variants for Anthropic model models on Bedrock
2348
+
2349
+ ### Fixed
2350
+
2351
+ - Fixed Gemini CLI provider retries with proper error handling, retry delays from headers, and empty stream retry logic
2352
+ - Fixed numbered list items showing "1." for all items when code blocks break list continuity (via `start` property)
2353
+
2354
+ ## [5.0.0] - 2026-01-12
2355
+
2356
+ ### Added
2357
+
2358
+ - Added support for `xhigh` thinking level in `thinkingBudgets` configuration
2359
+
2360
+ ### Changed
2361
+
2362
+ - Changed Anthropic thinking token budgets: minimal (1024→3072), low (2048→6144), medium (8192→12288), high (16384→24576)
2363
+ - Changed Google thinking token budgets: minimal (1024), low (2048→4096), medium (8192), high (16384), xhigh (24575)
2364
+ - Changed `supportsXhigh()` to return true for all Anthropic models
2365
+
2366
+ ## [4.6.0] - 2026-01-12
2367
+
2368
+ ### Fixed
2369
+
2370
+ - Fixed incorrect classification of thought signatures in Google Gemini responses—thought signatures are now correctly treated as metadata rather than thinking content indicators
2371
+ - Fixed thought signature handling in Google Gemini CLI and Vertex AI streaming to properly preserve signatures across text deltas
2372
+ - Fixed Google schema sanitization stripping property names that match schema keywords (e.g., "pattern", "format") from tool definitions
2373
+
2374
+ ## [4.4.9] - 2026-01-12
2375
+
2376
+ ### Fixed
2377
+
2378
+ - Fixed Google provider schema sanitization to strip additional unsupported JSON Schema fields (patternProperties, additionalProperties, min/max constraints, pattern, format)
2379
+
2380
+ ## [4.4.8] - 2026-01-12
2381
+
2382
+ ### Fixed
2383
+
2384
+ - Fixed Google provider schema sanitization to properly collapse `anyOf`/`oneOf` with const values into enum arrays
2385
+ - Fixed const-to-enum conversion to infer type from the const value when type is not specified
2386
+
2387
+ ## [4.4.6] - 2026-01-11
2388
+
2389
+ ### Fixed
2390
+
2391
+ - Fixed tool parameter schema sanitization to only apply Google-specific transformations for Gemini models, preserving original schemas for other model types
2392
+
2393
+ ## [4.4.5] - 2026-01-11
2394
+
2395
+ ### Changed
2396
+
2397
+ - Exported `sanitizeSchemaForGoogle` utility function for external use
2398
+
2399
+ ### Fixed
2400
+
2401
+ - Fixed Google provider schema sanitization to strip additional unsupported JSON Schema fields ($schema, $ref, $defs, format, examples, and others)
2402
+ - Fixed Google provider to ignore `additionalProperties: false` which is unsupported by the API
2403
+
2404
+ ## [4.4.4] - 2026-01-11
2405
+
2406
+ ### Fixed
2407
+
2408
+ - Fixed Cursor todo updates to bridge update_todos tool calls to the local todo_write tool
2409
+
2410
+ ## [4.3.0] - 2026-01-11
2411
+
2412
+ ### Added
2413
+
2414
+ - Added debug log filtering and display script for Cursor JSONL logs with follow mode and coalescing support
2415
+ - Added protobuf definition extractor script to reconstruct .proto files from bundled JavaScript
2416
+ - Added conversation state caching to persist context across multiple Cursor API requests in the same session
2417
+ - Added shell streaming support for real-time stdout/stderr output during command execution
2418
+ - Added JSON5 parsing for MCP tool arguments with Python-style boolean and None value normalization
2419
+ - Added Cursor provider with support for Anthropic model, GPT, and Gemini models via Cursor's agent API
2420
+ - Added OAuth authentication flow for Cursor including login, token refresh, and expiry detection
2421
+ - Added `cursor-agent` API type with streaming support and tool execution handlers
2422
+ - Added Cursor model definitions including Anthropic model 4.5, GPT-5.x, Gemini 3, and Grok variants
2423
+ - Added model generation script to automatically fetch and update AI model definitions from models.dev and OpenRouter APIs
2424
+
2425
+ ### Changed
2426
+
2427
+ - Changed Cursor debug logging to use structured JSONL format with automatic MCP argument decoding
2428
+ - Changed MCP tool argument decoding to use protobuf Value schema for improved type handling
2429
+ - Changed tool advertisement to filter Cursor native tools (bash, read, write, delete, ls, grep, lsp) instead of only exposing mcp\_ prefixed tools
2430
+
2431
+ ### Fixed
2432
+
2433
+ - Fixed Cursor conversation history serialization so subagents retain task context and can call complete
2434
+
2435
+ ## [4.2.1] - 2026-01-11
2436
+
2437
+ ### Changed
2438
+
2439
+ - Updated `reasoningSummary` option to accept only `"auto"`, `"concise"`, `"detailed"`, or `null` (removed `"off"` and `"on"` values)
2440
+ - Changed default `reasoningSummary` from `"auto"` to `"detailed"`
2441
+ - OpenAI code provider: switched to bundled system prompt matching opencode, changed originator to "opencode", simplified prompt handling
2442
+
2443
+ ### Fixed
2444
+
2445
+ - Fixed Cloud Code Assist tool schema conversion to avoid unsupported `const` fields
2446
+
2447
+ ## [4.0.0] - 2026-01-10
2448
+
2449
+ ### Added
2450
+
2451
+ - Added `betas` option in `AnthropicOptions` for passing custom Anthropic beta feature flags
2452
+ - OpenCode Zen provider support with 26 models (Anthropic model, GPT, Gemini, Grok, Kimi, GLM, Qwen, etc.). Set `OPENCODE_API_KEY` env var to use.
2453
+ - `thinkingBudgets` option in `SimpleStreamOptions` for customizing token budgets per thinking level on token-based providers
2454
+ - `sessionId` option in `StreamOptions` for providers that support session-based caching. OpenAI code provider provider uses this to set `prompt_cache_key` and routing headers.
2455
+ - `supportsUsageInStreaming` compatibility flag for OpenAI-compatible providers that reject `stream_options: { include_usage: true }`. Defaults to `true`. Set to `false` in model config for providers like gatewayz.ai.
2456
+ - `GOOGLE_APPLICATION_CREDENTIALS` env var support for Vertex AI credential detection (standard for CI/production)
2457
+ - Exported OpenAI code provider utilities: `CacheMetadata`, `getOpenAI codeInstructions`, `getModelFamily`, `ModelFamily`, `buildOpenAI codePiBridge`, `buildOpenAI codeSystemPrompt`, `OpenAI codeSystemPrompt`
2458
+ - Headless OAuth support for all callback-server providers (Google Gemini CLI, Antigravity, OpenAI code provider): paste redirect URL when browser callback is unreachable
2459
+ - Cancellable GitHub Copilot device code polling via AbortSignal
2460
+ - Improved error messages for OpenRouter providers by including raw metadata from upstream errors
2461
+
2462
+ ### Changed
2463
+
2464
+ - Changed Anthropic provider to include Anthropic Code system instruction for all API key types, not just OAuth tokens (except Haiku models)
2465
+ - Changed Anthropic OAuth tool naming to use `proxy_` prefix instead of mapping to Anthropic Code tool names, avoiding potential name collisions
2466
+ - Changed Anthropic provider to include Anthropic Code headers for all requests, not just OAuth tokens
2467
+ - Anthropic provider now maps tool names to Anthropic Code's exact tool names (Read, Write, Edit, Bash, Grep, Glob) instead of using prefixed names
2468
+ - OpenAI Completions provider now disables strict mode on tools to allow optional parameters without null unions
2469
+
2470
+ ### Fixed
2471
+
2472
+ - Fixed Anthropic OAuth code parsing to accept full redirect URLs in addition to raw authorization codes
2473
+ - Fixed Anthropic token refresh to preserve existing refresh token when server doesn't return a new one
2474
+ - Fixed thinking mode being enabled when tool_choice forces a specific tool, which is unsupported
2475
+ - Fixed max_tokens being too low when thinking budget is set, now auto-adjusts to model's maxTokens
2476
+ - Google Cloud Code Assist OAuth for paid subscriptions: properly handles long-running operations for project provisioning, supports `GOOGLE_CLOUD_PROJECT` / `GOOGLE_CLOUD_PROJECT_ID` env vars for paid tiers
2477
+ - `os.homedir()` calls at module load time; now resolved lazily when needed
2478
+ - OpenAI Responses tool strict flag to use a boolean for LM Studio compatibility
2479
+ - Gemini CLI abort handling: detect native `AbortError` in retry catch block, cancel SSE reader when abort signal fires
2480
+ - Antigravity provider 429 errors by aligning request payload with CLIProxyAPI v6.6.89
2481
+ - Thinking block handling for cross-model conversations: thinking blocks are now converted to plain text when switching models
2482
+ - OpenAI code provider context window from 400,000 to 272,000 tokens to match OpenAI code CLI defaults
2483
+ - OpenAI code SSE error events to surface message, code, and status
2484
+ - Context overflow detection for `context_length_exceeded` error codes
2485
+ - OpenAI code provider now always includes `reasoning.encrypted_content` even when custom `include` options are passed
2486
+ - OpenAI code requests now omit the `reasoning` field entirely when thinking is off
2487
+ - Crash when pasting text with trailing whitespace exceeding terminal width
2488
+
2489
+ ## [3.37.1] - 2026-01-10
2490
+
2491
+ ### Added
2492
+
2493
+ - Added automatic type coercion for tool arguments when LLMs return JSON-encoded strings instead of native types (numbers, booleans, arrays, objects)
2494
+
2495
+ ### Changed
2496
+
2497
+ - Changed tool argument validation to attempt JSON parsing and type coercion before rejecting mismatched types
2498
+ - Changed validation error messages to include both original and normalized arguments when coercion was attempted
2499
+
2500
+ ## [3.37.0] - 2026-01-10
2501
+
2502
+ ### Changed
2503
+
2504
+ - Enabled type coercion in JSON schema validation to automatically convert compatible types
2505
+
2506
+ ## [3.35.0] - 2026-01-09
2507
+
2508
+ ### Added
2509
+
2510
+ - Enhanced error messages to include retry-after timing information from API rate limit headers
2511
+
2512
+ ## [0.42.0] - 2026-01-09
2513
+
2514
+ ### Added
2515
+
2516
+ - Added OpenCode Zen provider support with 26 models (Anthropic model, GPT, Gemini, Grok, Kimi, GLM, Qwen, etc.). Set `OPENCODE_API_KEY` env var to use.
2517
+
2518
+ ## [0.39.0] - 2026-01-08
2519
+
2520
+ ### Fixed
2521
+
2522
+ - Fixed Gemini CLI abort handling: detect native `AbortError` in retry catch block, cancel SSE reader when abort signal fires ([#568](https://github.com/badlogic/pi-mono/pull/568) by [@tmustier](https://github.com/tmustier))
2523
+ - Fixed Antigravity provider 429 errors by aligning request payload with CLIProxyAPI v6.6.89: inject Antigravity system instruction with `role: "user"`, set `requestType: "agent"`, and use `antigravity` userAgent. Added bridge prompt to override Antigravity behavior (identity, paths, web dev guidelines) with Pi defaults. ([#571](https://github.com/badlogic/pi-mono/pull/571) by [@ben-vargas](https://github.com/ben-vargas))
2524
+ - Fixed thinking block handling for cross-model conversations: thinking blocks are now converted to plain text (no `<thinking>` tags) when switching models. Previously, `<thinking>` tags caused models to mimic the pattern and output literal tags. Also fixed empty thinking blocks causing API errors. ([#561](https://github.com/badlogic/pi-mono/issues/561))
2525
+
2526
+ ## [0.38.0] - 2026-01-08
2527
+
2528
+ ### Added
2529
+
2530
+ - `thinkingBudgets` option in `SimpleStreamOptions` for customizing token budgets per thinking level on token-based providers ([#529](https://github.com/badlogic/pi-mono/pull/529) by [@melihmucuk](https://github.com/melihmucuk))
2531
+
2532
+ ### Breaking Changes
2533
+
2534
+ - Removed OpenAI code provider model aliases (`gpt-5`, `gpt-5-mini`, `gpt-5-nano`, `openai-code-mini-latest`, `gpt-5-openai-code`, `gpt-5.1-openai-code`, `gpt-5.1-chat-latest`). Use canonical model IDs: `gpt-5.1`, `gpt-5.1-openai-code-max`, `gpt-5.1-openai-code-mini`, `gpt-5.2`, `gpt-5.2-openai-code`. ([#536](https://github.com/badlogic/pi-mono/pull/536) by [@ghoulr](https://github.com/ghoulr))
2535
+
2536
+ ### Fixed
2537
+
2538
+ - Fixed OpenAI code provider context window from 400,000 to 272,000 tokens to match OpenAI code CLI defaults and prevent 400 errors. ([#536](https://github.com/badlogic/pi-mono/pull/536) by [@ghoulr](https://github.com/ghoulr))
2539
+ - Fixed OpenAI code SSE error events to surface message, code, and status. ([#551](https://github.com/badlogic/pi-mono/pull/551) by [@tmustier](https://github.com/tmustier))
2540
+ - Fixed context overflow detection for `context_length_exceeded` error codes.
2541
+
2542
+ ## [0.37.6] - 2026-01-06
2543
+
2544
+ ### Added
2545
+
2546
+ - Exported OpenAI code provider utilities: `CacheMetadata`, `getOpenAI codeInstructions`, `getModelFamily`, `ModelFamily`, `buildOpenAI codePiBridge`, `buildOpenAI codeSystemPrompt`, `OpenAI codeSystemPrompt` ([#510](https://github.com/badlogic/pi-mono/pull/510) by [@mitsuhiko](https://github.com/mitsuhiko))
2547
+
2548
+ ## [0.37.3] - 2026-01-06
2549
+
2550
+ ### Added
2551
+
2552
+ - `sessionId` option in `StreamOptions` for providers that support session-based caching. OpenAI code provider provider uses this to set `prompt_cache_key` and routing headers.
2553
+
2554
+ ## [0.37.2] - 2026-01-05
2555
+
2556
+ ### Fixed
2557
+
2558
+ - OpenAI code provider now always includes `reasoning.encrypted_content` even when custom `include` options are passed ([#484](https://github.com/badlogic/pi-mono/pull/484) by [@kim0](https://github.com/kim0))
2559
+
2560
+ ## [0.37.0] - 2026-01-05
2561
+
2562
+ ### Breaking Changes
2563
+
2564
+ - OpenAI code provider models no longer have per-thinking-level variants (e.g., `gpt-5.2-openai-code-high`). Use the base model ID and set thinking level separately. The OpenAI code provider clamps reasoning effort to what each model supports internally. (initial implementation by [@ben-vargas](https://github.com/ben-vargas) in [#472](https://github.com/badlogic/pi-mono/pull/472))
2565
+
2566
+ ### Added
2567
+
2568
+ - Headless OAuth support for all callback-server providers (Google Gemini CLI, Antigravity, OpenAI code provider): paste redirect URL when browser callback is unreachable ([#428](https://github.com/badlogic/pi-mono/pull/428) by [@ben-vargas](https://github.com/ben-vargas), [#468](https://github.com/badlogic/pi-mono/pull/468) by [@crcatala](https://github.com/crcatala))
2569
+ - Cancellable GitHub Copilot device code polling via AbortSignal
2570
+
2571
+ ### Fixed
2572
+
2573
+ - OpenAI code requests now omit the `reasoning` field entirely when thinking is off, letting the backend use its default instead of forcing a value. ([#472](https://github.com/badlogic/pi-mono/pull/472))
2574
+
2575
+ ## [0.36.0] - 2026-01-05
2576
+
2577
+ ### Added
2578
+
2579
+ - OpenAI code provider OAuth provider with Responses API streaming support: `openai-code-responses` streaming provider with SSE parsing, tool-call handling, usage/cost tracking, and PKCE OAuth flow ([#451](https://github.com/badlogic/pi-mono/pull/451) by [@kim0](https://github.com/kim0))
2580
+
2581
+ ### Fixed
2582
+
2583
+ - Vertex AI dummy value for `getEnvApiKey()`: Returns `"<authenticated>"` when Application Default Credentials are configured (`~/.config/gcloud/application_default_credentials.json` exists) and both `GOOGLE_CLOUD_PROJECT` (or `GCLOUD_PROJECT`) and `GOOGLE_CLOUD_LOCATION` are set. This allows `streamSimple()` to work with Vertex AI without explicit `apiKey` option. The ADC credentials file existence check is cached per-process to avoid repeated filesystem access.
2584
+
2585
+ ## [0.32.3] - 2026-01-03
2586
+
2587
+ ### Fixed
2588
+
2589
+ - Google Vertex AI models no longer appear in available models list without explicit authentication. Previously, `getEnvApiKey()` returned a dummy value for `google-vertex`, causing models to show up even when Google Cloud ADC was not configured.
2590
+
2591
+ ## [0.32.0] - 2026-01-03
2592
+
2593
+ ### Added
2594
+
2595
+ - Vertex AI provider with ADC (Application Default Credentials) support. Authenticate with `gcloud auth application-default login`, set `GOOGLE_CLOUD_PROJECT` and `GOOGLE_CLOUD_LOCATION`, and access Gemini models via Vertex AI. ([#300](https://github.com/badlogic/pi-mono/pull/300) by [@default-anton](https://github.com/default-anton))
2596
+
2597
+ ### Fixed
2598
+
2599
+ - **Gemini CLI rate limit handling**: Added automatic retry with server-provided delay for 429 errors. Parses delay from error messages like "Your quota will reset after 39s" and waits accordingly. Falls back to exponential backoff for other transient errors. ([#370](https://github.com/badlogic/pi-mono/issues/370))
2600
+
2601
+ ## [0.31.0] - 2026-01-02
2602
+
2603
+ ### Breaking Changes
2604
+
2605
+ - **Agent API moved**: All agent functionality (`agentLoop`, `agentLoopContinue`, `AgentContext`, `AgentEvent`, `AgentTool`, `AgentToolResult`, etc.) has moved to `@mariozechner/pi-agent-core`. Import from that package instead of `@sayknow-cli/ai`.
2606
+
2607
+ ### Added
2608
+
2609
+ - **`GoogleThinkingLevel` type**: Exported type that mirrors Google's `ThinkingLevel` enum values (`"THINKING_LEVEL_UNSPECIFIED" | "MINIMAL" | "LOW" | "MEDIUM" | "HIGH"`). Allows configuring Gemini thinking levels without importing from `@google/genai`.
2610
+ - **`ANTHROPIC_OAUTH_TOKEN` env var**: Now checked before `ANTHROPIC_API_KEY` in `getEnvApiKey()`, allowing OAuth tokens to take precedence.
2611
+ - **`event-stream.js` export**: `AssistantMessageEventStream` utility now exported from package index.
2612
+
2613
+ ### Changed
2614
+
2615
+ - **OAuth uses Web Crypto API**: PKCE generation and OAuth flows now use Web Crypto API (`crypto.subtle`) instead of Node.js `crypto` module. This improves browser compatibility while still working in Node.js 20+.
2616
+ - **Deterministic model generation**: `generate-models.ts` now sorts providers and models alphabetically for consistent output across runs. ([#332](https://github.com/badlogic/pi-mono/pull/332) by [@mrexodia](https://github.com/mrexodia))
2617
+
2618
+ ### Fixed
2619
+
2620
+ - **OpenAI completions empty content blocks**: Empty text or thinking blocks in assistant messages are now filtered out before sending to the OpenAI completions API, preventing validation errors. ([#344](https://github.com/badlogic/pi-mono/pull/344) by [@default-anton](https://github.com/default-anton))
2621
+ - **Thinking token duplication**: Fixed thinking content duplication with chutes.ai provider. The provider was returning thinking content in both `reasoning_content` and `reasoning` fields, causing each chunk to be processed twice. Now only the first non-empty reasoning field is used.
2622
+ - **zAi provider API mapping**: Fixed zAi models to use `openai-completions` API with correct base URL (`https://api.z.ai/api/coding/paas/v4`) instead of incorrect Anthropic API mapping. ([#344](https://github.com/badlogic/pi-mono/pull/344), [#358](https://github.com/badlogic/pi-mono/pull/358) by [@default-anton](https://github.com/default-anton))
2623
+
2624
+ ## [0.28.0] - 2025-12-25
2625
+
2626
+ ### Breaking Changes
2627
+
2628
+ - **OAuth storage removed** ([#296](https://github.com/badlogic/pi-mono/issues/296)): All storage functions (`loadOAuthCredentials`, `saveOAuthCredentials`, `setOAuthStorage`, etc.) removed. Callers are responsible for storing credentials.
2629
+ - **OAuth login functions**: `loginAnthropic`, `loginGitHubCopilot`, `loginGeminiCli`, `loginAntigravity` now return `OAuthCredentials` instead of saving to disk.
2630
+ - **refreshOAuthToken**: Now takes `(provider, credentials)` and returns new `OAuthCredentials` instead of saving.
2631
+ - **getOAuthApiKey**: Now takes `(provider, credentials)` and returns `{ newCredentials, apiKey }` or null.
2632
+ - **OAuthCredentials type**: No longer includes `type: "oauth"` discriminator. Callers add discriminator when storing.
2633
+ - **setApiKey, resolveApiKey**: Removed. Callers must manage their own API key storage/resolution.
2634
+ - **getApiKey**: Renamed to `getEnvApiKey`. Only checks environment variables for known providers.
2635
+
2636
+ ## [0.27.7] - 2025-12-24
2637
+
2638
+ ### Fixed
2639
+
2640
+ - **Thinking tag leakage**: Fixed Anthropic model mimicking literal `</thinking>` tags in responses. Unsigned thinking blocks (from aborted streams) are now converted to plain text without `<thinking>` tags. The TUI still displays them as thinking blocks. ([#302](https://github.com/badlogic/pi-mono/pull/302) by [@nicobailon](https://github.com/nicobailon))
2641
+
2642
+ ## [0.25.1] - 2025-12-21
2643
+
2644
+ ### Added
2645
+
2646
+ - **xhigh thinking level support**: Added `supportsXhigh()` function to check if a model supports xhigh reasoning level. Also clamps xhigh to high for OpenAI models that don't support it. ([#236](https://github.com/badlogic/pi-mono/pull/236) by [@theBucky](https://github.com/theBucky))
2647
+
2648
+ ### Fixed
2649
+
2650
+ - **Gemini multimodal tool results**: Fixed images in tool results causing flaky/broken responses with Gemini models. For Gemini 3, images are now nested inside `functionResponse.parts` per the [docs](https://ai.google.dev/gemini-api/docs/function-calling#multimodal). For older models (which don't support multimodal function responses), images are sent in a separate user message.
2651
+
2652
+ - **Queued message steering**: When `getQueuedMessages` is provided, the agent loop now checks for queued user messages after each tool call and skips remaining tool calls in the current assistant message when a queued message arrives (emitting error tool results).
2653
+
2654
+ - **Double API version path in Google provider URL**: Fixed Gemini API calls returning 404 after baseUrl support was added. The SDK was appending its default apiVersion to baseUrl which already included the version path. ([#251](https://github.com/badlogic/pi-mono/pull/251) by [@shellfyred](https://github.com/shellfyred))
2655
+
2656
+ - **Anthropic SDK retries disabled**: Re-enabled SDK-level retries (default 2) for transient HTTP failures. ([#252](https://github.com/badlogic/pi-mono/issues/252))
2657
+
2658
+ ## [0.23.5] - 2025-12-19
2659
+
2660
+ ### Added
2661
+
2662
+ - **Gemini 3 Flash thinking support**: Extended thinking level support for Gemini 3 Flash models (MINIMAL, LOW, MEDIUM, HIGH) to match Pro models' capabilities. ([#212](https://github.com/badlogic/pi-mono/pull/212) by [@markusylisiurunen](https://github.com/markusylisiurunen))
2663
+
2664
+ - **GitHub Copilot thinking models**: Added thinking support for additional Copilot models (o3-mini, o1-mini, o1-preview). ([#234](https://github.com/badlogic/pi-mono/pull/234) by [@aadishv](https://github.com/aadishv))
2665
+
2666
+ ### Fixed
2667
+
2668
+ - **Gemini tool result format**: Fixed tool result format for Gemini 3 Flash Preview which strictly requires `{ output: value }` for success and `{ error: value }` for errors. Previous format using `{ result, isError }` was rejected by newer Gemini models. Also improved type safety by removing `as any` casts. ([#213](https://github.com/badlogic/pi-mono/issues/213), [#220](https://github.com/badlogic/pi-mono/pull/220))
2669
+
2670
+ - **Google baseUrl configuration**: Google provider now respects `baseUrl` configuration for custom endpoints or API proxies. ([#216](https://github.com/badlogic/pi-mono/issues/216), [#221](https://github.com/badlogic/pi-mono/pull/221) by [@theBucky](https://github.com/theBucky))
2671
+
2672
+ - **GitHub Copilot vision requests**: Added `Copilot-Vision-Request` header when sending images to GitHub Copilot models. ([#222](https://github.com/badlogic/pi-mono/issues/222))
2673
+
2674
+ - **GitHub Copilot X-Initiator header**: Fixed X-Initiator logic to check last message role instead of any message in history. This ensures proper billing when users send follow-up messages. ([#209](https://github.com/badlogic/pi-mono/issues/209))
2675
+
2676
+ ## [0.22.3] - 2025-12-16
2677
+
2678
+ ### Added
2679
+
2680
+ - **Image limits test suite**: Added comprehensive tests for provider-specific image limitations (max images, max size, max dimensions). Discovered actual limits: Anthropic (100 images, 5MB, 8000px), OpenAI (500 images, ≥25MB), Gemini (~2500 images, ≥40MB), Mistral (8 images, ~15MB), OpenRouter (~40 images context-limited, ~15MB). ([#120](https://github.com/badlogic/pi-mono/pull/120))
2681
+
2682
+ - **Tool result streaming**: Added `tool_execution_update` event and optional `onUpdate` callback to `AgentTool.execute()` for streaming tool output during execution. Tools can now emit partial results (e.g., bash stdout) that are forwarded to subscribers. ([#44](https://github.com/badlogic/pi-mono/issues/44))
2683
+
2684
+ - **X-Initiator header for GitHub Copilot**: Added X-Initiator header handling for GitHub Copilot provider to ensure correct call accounting (agent calls are not deducted from quota). Sets initiator based on last message role. ([#200](https://github.com/badlogic/pi-mono/pull/200) by [@kim0](https://github.com/kim0))
2685
+
2686
+ ### Changed
2687
+
2688
+ - **Normalized tool_execution_end result**: `tool_execution_end` event now always contains `AgentToolResult` (no longer `AgentToolResult | string`). Errors are wrapped in the standard result format.
2689
+
2690
+ ### Fixed
2691
+
2692
+ - **Reasoning disabled by default**: When `reasoning` option is not specified, thinking is now explicitly disabled for all providers. Previously, some providers like Gemini with "dynamic thinking" would use their default (thinking ON), causing unexpected token usage. This was the original intended behavior. ([#180](https://github.com/badlogic/pi-mono/pull/180) by [@markusylisiurunen](https://github.com/markusylisiurunen))
2693
+
2694
+ ## [0.22.2] - 2025-12-15
2695
+
2696
+ ### Added
2697
+
2698
+ - **Interleaved thinking for Anthropic**: Added `interleavedThinking` option to `AnthropicOptions`. When enabled, Anthropic model 4 models can think between tool calls and reason after receiving tool results. Enabled by default (no extra token cost, just unlocks the capability). Set `interleavedThinking: false` to disable.
2699
+
2700
+ ## [0.22.1] - 2025-12-15
2701
+
2702
+ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_
2703
+
2704
+ ### Added
2705
+
2706
+ - **Interleaved thinking for Anthropic**: Enabled interleaved thinking in the Anthropic provider, allowing Anthropic model models to output thinking blocks interspersed with text responses.
2707
+
2708
+ ## [0.22.0] - 2025-12-15
2709
+
2710
+ ### Added
2711
+
2712
+ - **GitHub Copilot provider**: Added `github-copilot` as a known provider with models sourced from models.dev. Includes Anthropic model, GPT, Gemini, Grok, and other models available through GitHub Copilot. ([#191](https://github.com/badlogic/pi-mono/pull/191) by [@cau1k](https://github.com/cau1k))
2713
+
2714
+ ### Fixed
2715
+
2716
+ - **GitHub Copilot gpt-5 models**: Fixed API selection for gpt-5 models to use `openai-responses` instead of `openai-completions` (gpt-5 models are not accessible via completions endpoint)
2717
+
2718
+ - **GitHub Copilot cross-model context handoff**: Fixed context handoff failing when switching between GitHub Copilot models using different APIs (e.g., gpt-5 to anthropic-model-sonnet-4). Tool call IDs from OpenAI Responses API were incompatible with other models. ([#198](https://github.com/badlogic/pi-mono/issues/198))
2719
+
2720
+ - **Gemini 3 Pro thinking levels**: Thinking level configuration now works correctly for Gemini 3 Pro models. Previously all levels mapped to -1 (minimal thinking). Now LOW/MEDIUM/HIGH properly control test-time computation. ([#176](https://github.com/badlogic/pi-mono/pull/176) by [@markusylisiurunen](https://github.com/markusylisiurunen))
2721
+
2722
+ ## [0.18.2] - 2025-12-11
2723
+
2724
+ ### Changed
2725
+
2726
+ - **Anthropic SDK retries disabled**: Set `maxRetries: 0` on Anthropic client to allow application-level retry handling. The SDK's built-in retries were interfering with coding-agent's retry logic. ([#157](https://github.com/badlogic/pi-mono/issues/157))
2727
+
2728
+ ## [0.18.1] - 2025-12-10
2729
+
2730
+ ### Added
2731
+
2732
+ - **Mistral provider**: Added support for Mistral AI models via the OpenAI-compatible API. Includes automatic handling of Mistral-specific requirements (tool call ID format). Set `MISTRAL_API_KEY` environment variable to use.
2733
+
2734
+ ### Fixed
2735
+
2736
+ - Fixed Mistral 400 errors after aborted assistant messages by skipping empty assistant messages (no content, no tool calls) ([#165](https://github.com/badlogic/pi-mono/issues/165))
2737
+
2738
+ - Removed synthetic assistant bridge message after tool results for Mistral (no longer required as of Dec 2025) ([#165](https://github.com/badlogic/pi-mono/issues/165))
2739
+
2740
+ - Fixed bug where `ANTHROPIC_API_KEY` environment variable was deleted globally after first OAuth token usage, causing subsequent prompts to fail ([#164](https://github.com/badlogic/pi-mono/pull/164))
2741
+
2742
+ ## [0.17.0] - 2025-12-09
2743
+
2744
+ ### Added
2745
+
2746
+ - **`agentLoopContinue` function**: Continue an agent loop from existing context without adding a new user message. Validates that the last message is `user` or `toolResult`. Useful for retry after context overflow or resuming from manually-added tool results.
2747
+
2748
+ ### Breaking Changes
2749
+
2750
+ - Removed provider-level tool argument validation. Validation now happens in `agentLoop` via `executeToolCalls`, allowing models to retry on validation errors. For manual tool execution, use `validateToolCall(tools, toolCall)` or `validateToolArguments(tool, toolCall)`.
2751
+
2752
+ ### Added
2753
+
2754
+ - Added `validateToolCall(tools, toolCall)` helper that finds the tool by name and validates arguments.
2755
+
2756
+ - **OpenAI compatibility overrides**: Added `compat` field to `Model` for `openai-completions` API, allowing explicit configuration of provider quirks (`supportsStore`, `supportsDeveloperRole`, `supportsReasoningEffort`, `maxTokensField`). Falls back to URL-based detection if not set. Useful for LiteLLM, custom proxies, and other non-standard endpoints. ([#133](https://github.com/badlogic/pi-mono/issues/133), thanks @fink-andreas for the initial idea and PR)
2757
+
2758
+ - **xhigh reasoning level**: Added `xhigh` to `ReasoningEffort` type for OpenAI openai-code-max models. For non-OpenAI providers (Anthropic, Google), `xhigh` is automatically mapped to `high`. ([#143](https://github.com/badlogic/pi-mono/issues/143))
2759
+
2760
+ ### Changed
2761
+
2762
+ - **Updated SDK versions**: OpenAI SDK 5.21.0 → 6.10.0, Anthropic SDK 0.61.0 → 0.71.2, Google GenAI SDK 1.30.0 → 1.31.0
2763
+
2764
+ ## [0.13.0] - 2025-12-06
2765
+
2766
+ ### Breaking Changes
2767
+
2768
+ - **Added `totalTokens` field to `Usage` type**: All code that constructs `Usage` objects must now include the `totalTokens` field. This field represents the total tokens processed by the LLM (input + output + cache). For OpenAI and Google, this uses native API values (`total_tokens`, `totalTokenCount`). For Anthropic, it's computed as `input + output + cacheRead + cacheWrite`.
2769
+
2770
+ ## [0.12.10] - 2025-12-04
2771
+
2772
+ ### Added
2773
+
2774
+ - Added `gpt-5.1-openai-code-max` model support
2775
+
2776
+ ### Fixed
2777
+
2778
+ - **OpenAI Token Counting**: Fixed `usage.input` to exclude cached tokens for OpenAI providers. Previously, `input` included cached tokens, causing double-counting when calculating total context size via `input + cacheRead`. Now `input` represents non-cached input tokens across all providers, making `input + output + cacheRead + cacheWrite` the correct formula for total context size.
2779
+
2780
+ - **Fixed Anthropic model Opus 4.5 cache pricing** (was 3x too expensive)
2781
+ - Corrected cache_read: $1.50 → $0.50 per MTok
2782
+ - Corrected cache_write: $18.75 → $6.25 per MTok
2783
+ - Added manual override in `scripts/generate-models.ts` until upstream fix is merged
2784
+ - Submitted PR to models.dev: https://github.com/sst/models.dev/pull/439
2785
+
2786
+ ## [0.9.4] - 2025-11-26
2787
+
2788
+ Initial release with multi-provider LLM support.