@bastani/pi-ai 0.9.20-alpha.2 → 0.9.20-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (264) hide show
  1. package/CHANGELOG.md +34 -1
  2. package/NOTICE.md +1 -1
  3. package/README.md +28 -5
  4. package/dist/api/anthropic-messages.d.ts.map +1 -1
  5. package/dist/api/anthropic-messages.js +167 -85
  6. package/dist/api/anthropic-messages.js.map +1 -1
  7. package/dist/api/azure-openai-responses.d.ts.map +1 -1
  8. package/dist/api/azure-openai-responses.js +17 -5
  9. package/dist/api/azure-openai-responses.js.map +1 -1
  10. package/dist/api/bedrock-converse-stream.d.ts.map +1 -1
  11. package/dist/api/bedrock-converse-stream.js +15 -12
  12. package/dist/api/bedrock-converse-stream.js.map +1 -1
  13. package/dist/api/google-generative-ai.d.ts.map +1 -1
  14. package/dist/api/google-generative-ai.js +22 -80
  15. package/dist/api/google-generative-ai.js.map +1 -1
  16. package/dist/api/google-shared.d.ts +13 -4
  17. package/dist/api/google-shared.d.ts.map +1 -1
  18. package/dist/api/google-shared.js +54 -4
  19. package/dist/api/google-shared.js.map +1 -1
  20. package/dist/api/google-vertex.d.ts.map +1 -1
  21. package/dist/api/google-vertex.js +25 -65
  22. package/dist/api/google-vertex.js.map +1 -1
  23. package/dist/api/mistral-conversations.d.ts.map +1 -1
  24. package/dist/api/mistral-conversations.js +15 -11
  25. package/dist/api/mistral-conversations.js.map +1 -1
  26. package/dist/api/openai-codex-responses.d.ts.map +1 -1
  27. package/dist/api/openai-codex-responses.js +20 -22
  28. package/dist/api/openai-codex-responses.js.map +1 -1
  29. package/dist/api/openai-completions.d.ts +6 -4
  30. package/dist/api/openai-completions.d.ts.map +1 -1
  31. package/dist/api/openai-completions.js +39 -57
  32. package/dist/api/openai-completions.js.map +1 -1
  33. package/dist/api/openai-responses-shared.d.ts +7 -5
  34. package/dist/api/openai-responses-shared.d.ts.map +1 -1
  35. package/dist/api/openai-responses-shared.js +56 -56
  36. package/dist/api/openai-responses-shared.js.map +1 -1
  37. package/dist/api/openai-responses.d.ts.map +1 -1
  38. package/dist/api/openai-responses.js +14 -19
  39. package/dist/api/openai-responses.js.map +1 -1
  40. package/dist/api/pi-messages.d.ts +3 -3
  41. package/dist/api/pi-messages.d.ts.map +1 -1
  42. package/dist/api/pi-messages.js +2 -2
  43. package/dist/api/pi-messages.js.map +1 -1
  44. package/dist/api/simple-options.d.ts +3 -3
  45. package/dist/api/simple-options.d.ts.map +1 -1
  46. package/dist/api/simple-options.js.map +1 -1
  47. package/dist/api/transform-messages.d.ts.map +1 -1
  48. package/dist/api/transform-messages.js +21 -7
  49. package/dist/api/transform-messages.js.map +1 -1
  50. package/dist/auth/credential-store.d.ts.map +1 -1
  51. package/dist/auth/credential-store.js +3 -2
  52. package/dist/auth/credential-store.js.map +1 -1
  53. package/dist/auth/resolve.d.ts +4 -0
  54. package/dist/auth/resolve.d.ts.map +1 -1
  55. package/dist/auth/resolve.js +27 -8
  56. package/dist/auth/resolve.js.map +1 -1
  57. package/dist/compat.d.ts +3 -3
  58. package/dist/compat.d.ts.map +1 -1
  59. package/dist/compat.js +9 -6
  60. package/dist/compat.js.map +1 -1
  61. package/dist/env-api-keys.d.ts +2 -0
  62. package/dist/env-api-keys.d.ts.map +1 -1
  63. package/dist/env-api-keys.js +59 -43
  64. package/dist/env-api-keys.js.map +1 -1
  65. package/dist/image-models.generated.d.ts +30 -0
  66. package/dist/image-models.generated.d.ts.map +1 -1
  67. package/dist/image-models.generated.js +34 -4
  68. package/dist/image-models.generated.js.map +1 -1
  69. package/dist/index.d.ts +3 -1
  70. package/dist/index.d.ts.map +1 -1
  71. package/dist/index.js +3 -1
  72. package/dist/index.js.map +1 -1
  73. package/dist/models.d.ts +5 -4
  74. package/dist/models.d.ts.map +1 -1
  75. package/dist/models.generated.d.ts +2 -0
  76. package/dist/models.generated.d.ts.map +1 -1
  77. package/dist/models.generated.js +2 -0
  78. package/dist/models.generated.js.map +1 -1
  79. package/dist/models.js +6 -3
  80. package/dist/models.js.map +1 -1
  81. package/dist/providers/amazon-bedrock.models.d.ts +1 -2
  82. package/dist/providers/amazon-bedrock.models.d.ts.map +1 -1
  83. package/dist/providers/amazon-bedrock.models.js.map +1 -1
  84. package/dist/providers/ant-ling.models.d.ts +1 -2
  85. package/dist/providers/ant-ling.models.d.ts.map +1 -1
  86. package/dist/providers/ant-ling.models.js.map +1 -1
  87. package/dist/providers/anthropic.models.d.ts +1 -2
  88. package/dist/providers/anthropic.models.d.ts.map +1 -1
  89. package/dist/providers/anthropic.models.js.map +1 -1
  90. package/dist/providers/azure-openai-responses.models.d.ts +1 -2
  91. package/dist/providers/azure-openai-responses.models.d.ts.map +1 -1
  92. package/dist/providers/azure-openai-responses.models.js.map +1 -1
  93. package/dist/providers/baseten.models.d.ts +1 -2
  94. package/dist/providers/baseten.models.d.ts.map +1 -1
  95. package/dist/providers/baseten.models.js.map +1 -1
  96. package/dist/providers/cerebras.models.d.ts +1 -2
  97. package/dist/providers/cerebras.models.d.ts.map +1 -1
  98. package/dist/providers/cerebras.models.js.map +1 -1
  99. package/dist/providers/cloudflare-ai-gateway.models.d.ts +1 -2
  100. package/dist/providers/cloudflare-ai-gateway.models.d.ts.map +1 -1
  101. package/dist/providers/cloudflare-ai-gateway.models.js.map +1 -1
  102. package/dist/providers/cloudflare-workers-ai.models.d.ts +1 -2
  103. package/dist/providers/cloudflare-workers-ai.models.d.ts.map +1 -1
  104. package/dist/providers/cloudflare-workers-ai.models.js.map +1 -1
  105. package/dist/providers/data/.manifest.json +1 -1
  106. package/dist/providers/data/amazon-bedrock.json +1 -1
  107. package/dist/providers/data/anthropic.json +1 -1
  108. package/dist/providers/data/baseten.json +1 -1
  109. package/dist/providers/data/cerebras.json +1 -1
  110. package/dist/providers/data/cloudflare-ai-gateway.json +1 -1
  111. package/dist/providers/data/deepseek.json +1 -1
  112. package/dist/providers/data/fireworks.json +1 -1
  113. package/dist/providers/data/github-copilot.json +1 -1
  114. package/dist/providers/data/google-vertex.json +1 -1
  115. package/dist/providers/data/google.json +1 -1
  116. package/dist/providers/data/mistral.json +1 -1
  117. package/dist/providers/data/moonshotai-cn.json +1 -1
  118. package/dist/providers/data/moonshotai.json +1 -1
  119. package/dist/providers/data/nvidia.json +1 -1
  120. package/dist/providers/data/openai-codex.json +1 -1
  121. package/dist/providers/data/openai.json +1 -1
  122. package/dist/providers/data/opencode-go.json +1 -1
  123. package/dist/providers/data/opencode.json +1 -1
  124. package/dist/providers/data/openrouter.json +1 -1
  125. package/dist/providers/data/qwen-token-plan-cn.json +1 -1
  126. package/dist/providers/data/qwen-token-plan.json +1 -1
  127. package/dist/providers/data/radius.json +1 -0
  128. package/dist/providers/data/vercel-ai-gateway.json +1 -1
  129. package/dist/providers/data/zai-coding-cn.json +1 -1
  130. package/dist/providers/data/zai.json +1 -1
  131. package/dist/providers/deepseek.models.d.ts +1 -2
  132. package/dist/providers/deepseek.models.d.ts.map +1 -1
  133. package/dist/providers/deepseek.models.js.map +1 -1
  134. package/dist/providers/faux.d.ts +2 -2
  135. package/dist/providers/faux.d.ts.map +1 -1
  136. package/dist/providers/faux.js +11 -11
  137. package/dist/providers/faux.js.map +1 -1
  138. package/dist/providers/fireworks.models.d.ts +1 -2
  139. package/dist/providers/fireworks.models.d.ts.map +1 -1
  140. package/dist/providers/fireworks.models.js.map +1 -1
  141. package/dist/providers/github-copilot.models.d.ts +1 -2
  142. package/dist/providers/github-copilot.models.d.ts.map +1 -1
  143. package/dist/providers/github-copilot.models.js.map +1 -1
  144. package/dist/providers/google-vertex.models.d.ts +1 -2
  145. package/dist/providers/google-vertex.models.d.ts.map +1 -1
  146. package/dist/providers/google-vertex.models.js.map +1 -1
  147. package/dist/providers/google.models.d.ts +1 -2
  148. package/dist/providers/google.models.d.ts.map +1 -1
  149. package/dist/providers/google.models.js.map +1 -1
  150. package/dist/providers/groq.models.d.ts +1 -2
  151. package/dist/providers/groq.models.d.ts.map +1 -1
  152. package/dist/providers/groq.models.js.map +1 -1
  153. package/dist/providers/huggingface.models.d.ts +1 -2
  154. package/dist/providers/huggingface.models.d.ts.map +1 -1
  155. package/dist/providers/huggingface.models.js.map +1 -1
  156. package/dist/providers/kimi-coding.models.d.ts +1 -2
  157. package/dist/providers/kimi-coding.models.d.ts.map +1 -1
  158. package/dist/providers/kimi-coding.models.js.map +1 -1
  159. package/dist/providers/minimax-cn.models.d.ts +1 -2
  160. package/dist/providers/minimax-cn.models.d.ts.map +1 -1
  161. package/dist/providers/minimax-cn.models.js.map +1 -1
  162. package/dist/providers/minimax.models.d.ts +1 -2
  163. package/dist/providers/minimax.models.d.ts.map +1 -1
  164. package/dist/providers/minimax.models.js.map +1 -1
  165. package/dist/providers/mistral.models.d.ts +1 -2
  166. package/dist/providers/mistral.models.d.ts.map +1 -1
  167. package/dist/providers/mistral.models.js.map +1 -1
  168. package/dist/providers/moonshotai-cn.models.d.ts +1 -2
  169. package/dist/providers/moonshotai-cn.models.d.ts.map +1 -1
  170. package/dist/providers/moonshotai-cn.models.js.map +1 -1
  171. package/dist/providers/moonshotai.models.d.ts +1 -2
  172. package/dist/providers/moonshotai.models.d.ts.map +1 -1
  173. package/dist/providers/moonshotai.models.js.map +1 -1
  174. package/dist/providers/nvidia.models.d.ts +1 -2
  175. package/dist/providers/nvidia.models.d.ts.map +1 -1
  176. package/dist/providers/nvidia.models.js.map +1 -1
  177. package/dist/providers/openai-codex.models.d.ts +1 -2
  178. package/dist/providers/openai-codex.models.d.ts.map +1 -1
  179. package/dist/providers/openai-codex.models.js.map +1 -1
  180. package/dist/providers/openai.models.d.ts +1 -2
  181. package/dist/providers/openai.models.d.ts.map +1 -1
  182. package/dist/providers/openai.models.js.map +1 -1
  183. package/dist/providers/opencode-go.models.d.ts +1 -2
  184. package/dist/providers/opencode-go.models.d.ts.map +1 -1
  185. package/dist/providers/opencode-go.models.js.map +1 -1
  186. package/dist/providers/opencode.models.d.ts +1 -2
  187. package/dist/providers/opencode.models.d.ts.map +1 -1
  188. package/dist/providers/opencode.models.js.map +1 -1
  189. package/dist/providers/openrouter.models.d.ts +1 -2
  190. package/dist/providers/openrouter.models.d.ts.map +1 -1
  191. package/dist/providers/openrouter.models.js.map +1 -1
  192. package/dist/providers/qwen-token-plan-cn.models.d.ts +1 -2
  193. package/dist/providers/qwen-token-plan-cn.models.d.ts.map +1 -1
  194. package/dist/providers/qwen-token-plan-cn.models.js.map +1 -1
  195. package/dist/providers/qwen-token-plan-individual.models.d.ts +1 -2
  196. package/dist/providers/qwen-token-plan-individual.models.d.ts.map +1 -1
  197. package/dist/providers/qwen-token-plan-individual.models.js.map +1 -1
  198. package/dist/providers/qwen-token-plan.models.d.ts +1 -2
  199. package/dist/providers/qwen-token-plan.models.d.ts.map +1 -1
  200. package/dist/providers/qwen-token-plan.models.js.map +1 -1
  201. package/dist/providers/radius.d.ts.map +1 -1
  202. package/dist/providers/radius.js +19 -5
  203. package/dist/providers/radius.js.map +1 -1
  204. package/dist/providers/radius.models.d.ts +3 -0
  205. package/dist/providers/radius.models.d.ts.map +1 -0
  206. package/dist/providers/radius.models.js +6 -0
  207. package/dist/providers/radius.models.js.map +1 -0
  208. package/dist/providers/together.models.d.ts +1 -2
  209. package/dist/providers/together.models.d.ts.map +1 -1
  210. package/dist/providers/together.models.js.map +1 -1
  211. package/dist/providers/vercel-ai-gateway.models.d.ts +1 -2
  212. package/dist/providers/vercel-ai-gateway.models.d.ts.map +1 -1
  213. package/dist/providers/vercel-ai-gateway.models.js.map +1 -1
  214. package/dist/providers/xai.models.d.ts +1 -2
  215. package/dist/providers/xai.models.d.ts.map +1 -1
  216. package/dist/providers/xai.models.js.map +1 -1
  217. package/dist/providers/xiaomi-token-plan-ams.models.d.ts +1 -2
  218. package/dist/providers/xiaomi-token-plan-ams.models.d.ts.map +1 -1
  219. package/dist/providers/xiaomi-token-plan-ams.models.js.map +1 -1
  220. package/dist/providers/xiaomi-token-plan-cn.models.d.ts +1 -2
  221. package/dist/providers/xiaomi-token-plan-cn.models.d.ts.map +1 -1
  222. package/dist/providers/xiaomi-token-plan-cn.models.js.map +1 -1
  223. package/dist/providers/xiaomi-token-plan-sgp.models.d.ts +1 -2
  224. package/dist/providers/xiaomi-token-plan-sgp.models.d.ts.map +1 -1
  225. package/dist/providers/xiaomi-token-plan-sgp.models.js.map +1 -1
  226. package/dist/providers/xiaomi.models.d.ts +1 -2
  227. package/dist/providers/xiaomi.models.d.ts.map +1 -1
  228. package/dist/providers/xiaomi.models.js.map +1 -1
  229. package/dist/providers/zai-coding-cn.models.d.ts +1 -2
  230. package/dist/providers/zai-coding-cn.models.d.ts.map +1 -1
  231. package/dist/providers/zai-coding-cn.models.js.map +1 -1
  232. package/dist/providers/zai.models.d.ts +1 -2
  233. package/dist/providers/zai.models.d.ts.map +1 -1
  234. package/dist/providers/zai.models.js.map +1 -1
  235. package/dist/types.d.ts +95 -26
  236. package/dist/types.d.ts.map +1 -1
  237. package/dist/types.js.map +1 -1
  238. package/dist/utils/diagnostics.d.ts +3 -2
  239. package/dist/utils/diagnostics.d.ts.map +1 -1
  240. package/dist/utils/diagnostics.js.map +1 -1
  241. package/dist/utils/estimate.d.ts +2 -2
  242. package/dist/utils/estimate.d.ts.map +1 -1
  243. package/dist/utils/estimate.js +8 -29
  244. package/dist/utils/estimate.js.map +1 -1
  245. package/dist/utils/overflow.d.ts +1 -0
  246. package/dist/utils/overflow.d.ts.map +1 -1
  247. package/dist/utils/overflow.js +17 -5
  248. package/dist/utils/overflow.js.map +1 -1
  249. package/dist/utils/retry.d.ts.map +1 -1
  250. package/dist/utils/retry.js +2 -0
  251. package/dist/utils/retry.js.map +1 -1
  252. package/dist/utils/text.d.ts +9 -1
  253. package/dist/utils/text.d.ts.map +1 -1
  254. package/dist/utils/text.js +26 -0
  255. package/dist/utils/text.js.map +1 -1
  256. package/dist/utils/transcript.d.ts +84 -0
  257. package/dist/utils/transcript.d.ts.map +1 -0
  258. package/dist/utils/transcript.js +204 -0
  259. package/dist/utils/transcript.js.map +1 -0
  260. package/package.json +3 -3
  261. package/dist/utils/deferred-tools.d.ts +0 -9
  262. package/dist/utils/deferred-tools.d.ts.map +0 -1
  263. package/dist/utils/deferred-tools.js +0 -36
  264. package/dist/utils/deferred-tools.js.map +0 -1
package/dist/types.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"types.js","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"","sourcesContent":["import type { TelemetryContext } from \"@earendil-works/pi-telemetry\";\nimport type { AnthropicOptions } from \"./api/anthropic-messages.ts\";\nimport type { AzureOpenAIResponsesOptions } from \"./api/azure-openai-responses.ts\";\nimport type { BedrockOptions } from \"./api/bedrock-converse-stream.ts\";\nimport type { GoogleOptions } from \"./api/google-generative-ai.ts\";\nimport type { GoogleVertexOptions } from \"./api/google-vertex.ts\";\nimport type { MistralOptions } from \"./api/mistral-conversations.ts\";\nimport type { OpenAICodexResponsesOptions } from \"./api/openai-codex-responses.ts\";\nimport type { OpenAICompletionsOptions } from \"./api/openai-completions.ts\";\nimport type { OpenAIResponsesOptions } from \"./api/openai-responses.ts\";\nimport type { PiMessagesOptions } from \"./api/pi-messages.ts\";\nimport type { AssistantMessageDiagnostic } from \"./utils/diagnostics.ts\";\nimport type { AssistantMessageEventStream } from \"./utils/event-stream.ts\";\n\nexport type { AssistantMessageEventStream } from \"./utils/event-stream.ts\";\n\nexport type KnownApi =\n\t| \"openai-completions\"\n\t| \"mistral-conversations\"\n\t| \"openai-responses\"\n\t| \"azure-openai-responses\"\n\t| \"openai-codex-responses\"\n\t| \"anthropic-messages\"\n\t| \"bedrock-converse-stream\"\n\t| \"google-generative-ai\"\n\t| \"google-vertex\"\n\t| \"pi-messages\";\n\nexport type Api = KnownApi | (string & {});\n\nexport type KnownImagesApi = \"openrouter-images\";\n\nexport type ImagesApi = KnownImagesApi | (string & {});\n\nexport type KnownProvider =\n\t| \"amazon-bedrock\"\n\t| \"ant-ling\"\n\t| \"anthropic\"\n\t| \"google\"\n\t| \"google-vertex\"\n\t| \"openai\"\n\t| \"azure-openai-responses\"\n\t| \"openai-codex\"\n\t| \"radius\"\n\t| \"nvidia\"\n\t| \"deepseek\"\n\t| \"github-copilot\"\n\t| \"xai\"\n\t| \"groq\"\n\t| \"cerebras\"\n\t| \"openrouter\"\n\t| \"vercel-ai-gateway\"\n\t| \"zai\"\n\t| \"zai-coding-cn\"\n\t| \"mistral\"\n\t| \"minimax\"\n\t| \"minimax-cn\"\n\t| \"moonshotai\"\n\t| \"moonshotai-cn\"\n\t| \"huggingface\"\n\t| \"fireworks\"\n\t| \"together\"\n\t| \"baseten\"\n\t| \"opencode\"\n\t| \"opencode-go\"\n\t| \"kimi-coding\"\n\t| \"cloudflare-workers-ai\"\n\t| \"cloudflare-ai-gateway\"\n\t| \"qwen-token-plan\"\n\t| \"qwen-token-plan-cn\"\n\t| \"qwen-token-plan-individual\"\n\t| \"xiaomi\"\n\t| \"xiaomi-token-plan-cn\"\n\t| \"xiaomi-token-plan-ams\"\n\t| \"xiaomi-token-plan-sgp\";\nexport type ProviderId = KnownProvider | string;\n\nexport type KnownImagesProvider = \"openrouter\";\n\nexport type ImagesProviderId = KnownImagesProvider | string;\n\nexport type ToolChoice = \"auto\" | \"none\";\nexport type ThinkingLevel = \"minimal\" | \"low\" | \"medium\" | \"high\" | \"xhigh\" | \"max\";\nexport type ModelThinkingLevel = \"off\" | ThinkingLevel;\nexport type ThinkingLevelMap = Partial<Record<ModelThinkingLevel, string | null>>;\nexport type ChatTemplateKwargValue =\n\t| string\n\t| number\n\t| boolean\n\t| null\n\t| {\n\t\t\t$var: \"thinking.enabled\" | \"thinking.effort\" | \"thinking.budget\";\n\t\t\tomitWhenOff?: boolean;\n\t };\n\n/** Top-level request field used to cap reasoning tokens on OpenAI-compatible servers. */\nexport type ThinkingTokenBudgetField = \"thinking_token_budget\" | \"thinking_budget\" | \"thinking_budget_tokens\";\n\n/** Token budgets for each thinking level (token-based providers only) */\nexport interface ThinkingBudgets {\n\tminimal?: number;\n\tlow?: number;\n\tmedium?: number;\n\thigh?: number;\n}\n\n// Base options all providers share\nexport type CacheRetention = \"none\" | \"short\" | \"long\";\n\nexport type Transport = \"sse\" | \"websocket\" | \"websocket-cached\" | \"auto\";\n\n/** Provider-scoped environment overrides. Values take precedence over process.env. */\nexport type ProviderEnv = Record<string, string>;\nexport type ProviderHeaders = Record<string, string | null>;\nexport type FetchFunction = typeof globalThis.fetch;\nexport type SessionAffinityFormat = \"openai\" | \"openai-nosession\" | \"openrouter\";\n\nexport interface ProviderResponse {\n\tstatus: number;\n\theaders: Record<string, string>;\n}\n\n/** Authentication, HTTP transport, and lifecycle callbacks shared by provider requests. */\nexport interface ProviderRequestOptions<TModel = Model<Api>> {\n\tsignal?: AbortSignal;\n\t/** Explicit parent context for telemetry produced by this logical request. */\n\ttelemetryContext?: TelemetryContext;\n\tapiKey?: string;\n\t/**\n\t * Optional fetch implementation for provider HTTP requests.\n\t * Defaults to `globalThis.fetch`. Provider adapters that cannot inject a custom implementation may reject it.\n\t * This does not affect WebSocket transports.\n\t */\n\tfetch?: FetchFunction;\n\t/**\n\t * Provider-scoped environment values. These take precedence over process.env for\n\t * provider configuration such as regional settings, endpoint placeholders, and\n\t * proxy variables.\n\t */\n\tenv?: ProviderEnv;\n\t/**\n\t * Optional callback for inspecting or replacing provider payloads before sending.\n\t * Return undefined to keep the payload unchanged.\n\t */\n\tonPayload?: (payload: unknown, model: TModel) => unknown | undefined | Promise<unknown | undefined>;\n\t/**\n\t * Optional callback invoked after an HTTP response is received.\n\t */\n\tonResponse?: (response: ProviderResponse, model: TModel) => void | Promise<void>;\n\t/**\n\t * Optional custom HTTP headers to include in API requests.\n\t * Merged with provider defaults; caller values override default headers.\n\t * On AWS Bedrock these are injected via a Smithy `build`-step middleware so\n\t * they are covered by SigV4 signing; reserved headers (`x-amz-*`,\n\t * `authorization`, `host`) are silently ignored to preserve SigV4 / bearer auth.\n\t * A null value suppresses a provider/API default header with the same name.\n\t */\n\theaders?: ProviderHeaders;\n\t/**\n\t * HTTP request timeout in milliseconds for providers/SDKs that support it.\n\t * For example, OpenAI and Anthropic SDK clients default to 10 minutes.\n\t */\n\ttimeoutMs?: number;\n\t/**\n\t * Maximum retry attempts for providers/SDKs that support client-side retries.\n\t * For example, OpenAI and Anthropic SDK clients default to 2.\n\t */\n\tmaxRetries?: number;\n\t/**\n\t * Maximum delay in milliseconds to wait for a retry when the server requests a long wait.\n\t * If the server's requested delay exceeds this value, the request fails immediately\n\t * with an error containing the requested delay, allowing higher-level retry logic\n\t * to handle it with user visibility.\n\t * Default: 60000 (60 seconds). Set to 0 to disable the cap.\n\t */\n\tmaxRetryDelayMs?: number;\n}\n\nexport interface StreamOptions extends ProviderRequestOptions<Model<Api>> {\n\t/**\n\t * Optional callback invoked after an HTTP response is received and before\n\t * its body stream is consumed.\n\t */\n\tonResponse?: (response: ProviderResponse, model: Model<Api>) => void | Promise<void>;\n\ttemperature?: number;\n\t/**\n\t * Arbitrary sampling parameters merged into the request body as-is, after the named request\n\t * fields, so keys here override them. Lets custom OpenAI-compatible servers (llama.cpp, vLLM,\n\t * SGLang, ...) receive parameters pi does not model, e.g. `top_p`, `top_k`, `min_p`,\n\t * `repetition_penalty`. Merged over `Model.samplingParams` per key. Only applied by\n\t * OpenAI-compatible adapters (completions, responses, Azure responses); other APIs ignore it.\n\t */\n\tsamplingParams?: Record<string, unknown>;\n\tmaxTokens?: number;\n\t/**\n\t * Preferred transport for providers that support multiple transports.\n\t * Providers that do not support this option ignore it.\n\t */\n\ttransport?: Transport;\n\t/**\n\t * Prompt cache retention preference. Providers map this to their supported values.\n\t * Default: \"short\".\n\t */\n\tcacheRetention?: CacheRetention;\n\t/**\n\t * Optional session identifier for providers that support session-based caching.\n\t * Providers can use this to enable prompt caching, request routing, or other\n\t * session-aware features. Ignored by providers that don't support it.\n\t */\n\tsessionId?: string;\n\t/**\n\t * WebSocket connect timeout in milliseconds for providers that support\n\t * WebSocket transports. This covers the connection/open handshake only;\n\t * HTTP/SDK idleness uses `timeoutMs`, while gaps between decoded provider\n\t * stream events use `streamDeadlineMs`.\n\t */\n\twebsocketConnectTimeoutMs?: number;\n\t/**\n\t * Maximum idle gap in milliseconds between two provider stream events.\n\t *\n\t * Enforced below the HTTP layer, so a response body that stalls without ever\n\t * rejecting the adapter's async iterator (for example a decompression failure\n\t * that destroys the stream silently, #2553) still settles the attempt as a\n\t * retryable transport error instead of hanging forever. The timer restarts on\n\t * every event, so it never caps total stream duration.\n\t *\n\t * Defaults to 300000 ms. `0` (or any non-positive value) disables it.\n\t */\n\tstreamDeadlineMs?: number;\n\t/**\n\t * Optional metadata to include in API requests.\n\t * Providers extract the fields they understand and ignore the rest.\n\t * For example, Anthropic uses `user_id` for abuse tracking and rate limiting.\n\t */\n\tmetadata?: Record<string, unknown>;\n}\n\nexport type ProviderStreamOptions = StreamOptions & Record<string, unknown>;\n\nexport interface DeferredFetchOptions extends ProviderRequestOptions<Model<Api>> {\n\t/**\n\t * Maximum provider long-poll duration in milliseconds.\n\t * Defaults to 0, which performs one status check.\n\t */\n\twait?: number;\n}\n\n/** Request options for best-effort deferred-response cancellation. */\nexport type DeferredCancelOptions = ProviderRequestOptions<Model<Api>>;\n\n/**\n * Maps known APIs to their full provider-specific stream option types.\n * Type-only imports from API implementation modules are erased at emit, so\n * this is tree-shake safe.\n */\nexport interface ApiOptionsMap {\n\t\"anthropic-messages\": AnthropicOptions;\n\t\"openai-completions\": OpenAICompletionsOptions;\n\t\"openai-responses\": OpenAIResponsesOptions;\n\t\"openai-codex-responses\": OpenAICodexResponsesOptions;\n\t\"azure-openai-responses\": AzureOpenAIResponsesOptions;\n\t\"google-generative-ai\": GoogleOptions;\n\t\"google-vertex\": GoogleVertexOptions;\n\t\"mistral-conversations\": MistralOptions;\n\t\"bedrock-converse-stream\": BedrockOptions;\n\t\"pi-messages\": PiMessagesOptions;\n}\n\n/**\n * Full stream options for an API. Known APIs resolve to their concrete option\n * type; custom API strings fall back to the generic shape.\n */\nexport type ApiStreamOptions<TApi extends Api> = TApi extends keyof ApiOptionsMap\n\t? ApiOptionsMap[TApi]\n\t: StreamOptions & Record<string, unknown>;\n\n/**\n * The uniform stream contract of an API implementation module: every module\n * under `src/api/` exports `stream` and `streamSimple`; capable modules may also\n * export deferred-response methods. Lazy wrappers (`lazyApi()`) and provider\n * factories pass these around as values. This is the untyped dispatch shape;\n * per-API option typing lives on the implementation modules themselves and on\n * `Provider.stream()` via `ApiStreamOptions`.\n */\nexport interface ProviderStreams {\n\tstream(model: Model<Api>, context: Context, options?: StreamOptions): AssistantMessageEventStream;\n\tstreamSimple(model: Model<Api>, context: Context, options?: SimpleStreamOptions): AssistantMessageEventStream;\n\tfetchDeferred?(\n\t\tmodel: Model<Api>,\n\t\thandle: DeferredHandle,\n\t\toptions?: DeferredFetchOptions,\n\t): AssistantMessageEventStream;\n\tcancelDeferred?(model: Model<Api>, handle: DeferredHandle, options?: DeferredCancelOptions): Promise<void>;\n}\n\n/**\n * The uniform contract of an image-generation API implementation module:\n * every image API module under `src/api/` exports exactly `generateImages`,\n * so the module itself satisfies this interface. Lazy wrappers and image\n * provider factories pass these around as values.\n */\nexport interface ProviderImages {\n\tgenerateImages(\n\t\tmodel: ImagesModel<ImagesApi>,\n\t\tcontext: ImagesContext,\n\t\toptions?: ImagesOptions,\n\t): Promise<AssistantImages>;\n}\n\nexport interface ImagesOptions extends ProviderRequestOptions<ImagesModel<ImagesApi>> {\n\t/**\n\t * Optional metadata to include in API requests.\n\t * Providers extract the fields they understand and ignore the rest.\n\t */\n\tmetadata?: Record<string, unknown>;\n}\n\nexport type ProviderImagesOptions = ImagesOptions & Record<string, unknown>;\n\nexport interface AnthropicAllowedFallbackModel {\n\tprovider: ProviderId;\n\tmodel: string;\n\tcost: ModelCost;\n}\n\n// Unified options with reasoning passed to streamSimple() and completeSimple()\nexport interface SimpleStreamOptions extends StreamOptions {\n\t/** Provider-neutral tool selection for simple requests. When omitted, adapters use provider-specific behavior. */\n\ttoolChoice?: ToolChoice;\n\treasoning?: ThinkingLevel;\n\t/** Ask a capable provider to return a durable handle and continue the request asynchronously. */\n\tdeferred?: boolean | { window?: \"15m\" | \"1h\" | \"24h\" };\n\t/** Custom token budgets for thinking levels (token-based providers only) */\n\tthinkingBudgets?: ThinkingBudgets;\n}\n\n// Generic StreamFunction with typed options.\n//\n// Contract:\n// - Must return an AssistantMessageEventStream.\n// - Direct streamSimple() calls may throw synchronously when request auth is\n// missing. Once a stream is returned, request/model/runtime failures should\n// be encoded in the stream, not thrown.\n// - Error termination must produce an AssistantMessage with stopReason\n// \"error\" or \"aborted\" and errorMessage, emitted via the stream protocol.\nexport type StreamFunction<TApi extends Api = Api, TOptions extends StreamOptions = StreamOptions> = (\n\tmodel: Model<TApi>,\n\tcontext: Context,\n\toptions?: TOptions,\n) => AssistantMessageEventStream;\n\nexport type ImagesFunction<TApi extends ImagesApi = ImagesApi, TOptions extends ImagesOptions = ImagesOptions> = (\n\tmodel: ImagesModel<TApi>,\n\tcontext: ImagesContext,\n\toptions?: TOptions,\n) => Promise<AssistantImages>;\n\nexport interface TextSignatureV1 {\n\tv: 1;\n\tid: string;\n\tphase?: \"commentary\" | \"final_answer\";\n}\n\nexport interface TextContent {\n\ttype: \"text\";\n\ttext: string;\n\ttextSignature?: string; // e.g., for OpenAI responses, message metadata (legacy id string or TextSignatureV1 JSON)\n}\n\nexport interface ThinkingContent {\n\ttype: \"thinking\";\n\tthinking: string;\n\tthinkingSignature?: string; // Provider-specific opaque or serialized reasoning replay data\n\t/** When true, the thinking content was redacted by safety filters. The opaque\n\t * encrypted payload is stored in `thinkingSignature` so it can be passed back\n\t * to the API for multi-turn continuity. */\n\tredacted?: boolean;\n}\n\n/**\n * Boundary marker emitted by Anthropic server-side fallback, where one model's output gives way\n * to the next after a classifier refusal.\n *\n * This is public content rather than a stream-only event because it must survive the round trip:\n * \"Keep it exactly where it appeared. The API uses its position to validate the thinking blocks\n * around it, so a request that echoes thinking blocks from both sides of the boundary is rejected\n * if the block is omitted or moved.\"\n * https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback\n */\nexport interface FallbackContent {\n\ttype: \"fallback\";\n\t/** The model that declined. Echoes the model string sent when the declining hop is the request's own model. */\n\tfromModel: string;\n\t/** Resolved id of the model that continues. Always present. */\n\ttoModel: string;\n}\n\nexport interface ImageContent {\n\ttype: \"image\";\n\tdata: string; // base64 encoded image data\n\tmimeType: string; // e.g., \"image/jpeg\", \"image/png\"\n}\n\n/**\n * A document sent as model input, currently PDF only.\n *\n * Anthropic documents PDF as a platform capability — \"All active models support PDF processing\" —\n * routed through the same vision path as images, so it is not a per-model modality. A model\n * advertises it through `Model.input` containing `\"pdf\"`, which generated metadata sets only for\n * the runtimes that can actually serialize a document block. Providers that cannot are sent a\n * visible placeholder instead, exactly as they are for images they cannot accept.\n * https://platform.claude.com/docs/en/build-with-claude/pdf-support\n */\nexport interface DocumentContent {\n\ttype: \"document\";\n\t/** Base64-encoded document bytes. Amazon Bedrock takes raw bytes and decodes this itself. */\n\tdata: string;\n\t/**\n\t * The only media type either serializer implements. It is a literal rather than `string`\n\t * because both paths hardcode PDF — the Anthropic block emits `BetaBase64PDFSource`, whose\n\t * `media_type` is itself a fixed literal, and Bedrock emits `DocumentFormat.PDF` — so a wider\n\t * type would let a caller label arbitrary bytes as a PDF. Widening this union later is\n\t * non-breaking; narrowing it after release would not be. Note that adding a format is not a\n\t * media-type swap on the Anthropic side: plain text needs a structurally different source\n\t * variant (`BetaPlainTextSource`, `type: \"text\"`), and Bedrock's `doc`/`docx`/`xls`/`xlsx`/\n\t * `csv`/`html`/`md` formats have no Anthropic source variant at all.\n\t */\n\tmimeType: \"application/pdf\";\n\t/**\n\t * Optional caller-supplied label. Amazon Bedrock requires a name and warns that it \"is\n\t * vulnerable to prompt injections\", so the Bedrock path sanitizes this to the characters AWS\n\t * permits and falls back to a neutral generated name.\n\t */\n\tname?: string;\n}\n\nexport interface ToolCall {\n\ttype: \"toolCall\";\n\tid: string;\n\tname: string;\n\targuments: Record<string, any>;\n\tthoughtSignature?: string; // Google-specific: opaque signature for reusing thought context\n\t/** OpenAI Responses namespace for calls to dynamically loaded or namespaced tools. */\n\tnamespace?: string;\n}\n\nexport interface Usage {\n\tinput: number;\n\toutput: number;\n\tcacheRead: number;\n\tcacheWrite: number;\n\t/** Subset of `cacheWrite` written with 1h retention. Only Anthropic reports this split. */\n\tcacheWrite1h?: number;\n\t/**\n\t * Reasoning/thinking tokens, when the provider reports them. This is a subset of\n\t * `output`: `output` already includes these tokens. Set to a number (possibly 0) by\n\t * providers that expose a reasoning breakdown; left undefined by providers that don't.\n\t */\n\treasoning?: number;\n\ttotalTokens: number;\n\tcost: {\n\t\tinput: number;\n\t\toutput: number;\n\t\tcacheRead: number;\n\t\tcacheWrite: number;\n\t\ttotal: number;\n\t};\n}\n\nexport type StopReason = \"pending\" | \"stop\" | \"length\" | \"toolUse\" | \"error\" | \"aborted\" | \"deferred\";\n\nexport type JsonValue = string | number | boolean | null | JsonValue[] | { [key: string]: JsonValue };\n\nexport interface DeferredHandle {\n\tprovider: string;\n\tmodelId: string;\n\tapi: string;\n\t/** Provider token, such as a response id or batch id plus row id. */\n\tid: string;\n\texpiresAt?: number;\n\tpollAfterMs?: number;\n\t/** Provider conversion data required to reconstruct the final assistant message. */\n\tdata?: JsonValue;\n}\n\nexport interface UserMessage {\n\trole: \"user\";\n\tcontent: string | (TextContent | ImageContent | DocumentContent)[];\n\ttimestamp: number; // Unix timestamp in milliseconds\n}\n\nexport interface AssistantMessage {\n\trole: \"assistant\";\n\tcontent: (TextContent | ThinkingContent | ToolCall | FallbackContent)[];\n\tapi: Api;\n\tprovider: ProviderId;\n\tmodel: string;\n\tresponseModel?: string; // Concrete `chunk.model` when different from the requested `model` (e.g. OpenRouter `auto` -> `anthropic/...`)\n\tresponseId?: string; // Provider-specific response/message identifier when the upstream API exposes one\n\t/** Exact provider-native effort level used for this response. Absent for legacy or unmanaged responses. */\n\tproviderThinkingLevel?: string;\n\tdiagnostics?: AssistantMessageDiagnostic[]; // Redacted provider/runtime diagnostics for failures and recoveries.\n\tusage: Usage;\n\tstopReason: StopReason;\n\tdeferred?: DeferredHandle;\n\terrorMessage?: string;\n\trawStopReason?: string;\n\t/**\n\t * Provider indication of whether the model explicitly ended its turn.\n\t * Preserved for debugging and does not currently affect agent control flow.\n\t */\n\tendTurn?: boolean;\n\ttimestamp: number; // Unix timestamp in milliseconds\n}\n\nexport interface ToolResultMessage<TDetails = any> {\n\trole: \"toolResult\";\n\ttoolCallId: string;\n\ttoolName: string;\n\tcontent: (TextContent | ImageContent)[]; // Supports text and images\n\tdetails?: TDetails;\n\t/** Usage from the tool execution itself, if available. Not part of main LLM context accounting. */\n\tusage?: Usage;\n\t/**\n\t * Names from `Context.tools` that became available after this result.\n\t * Providers with native deferred tool loading use this as the load point;\n\t * other providers ignore it and use `Context.tools` normally.\n\t */\n\taddedToolNames?: string[];\n\tisError: boolean;\n\ttimestamp: number; // Unix timestamp in milliseconds\n}\n\nexport type Message = UserMessage | AssistantMessage | ToolResultMessage;\n\nexport type ImagesInputContent = TextContent | ImageContent;\nexport type ImagesOutputContent = TextContent | ImageContent;\n\nexport interface ImagesContext {\n\tinput: ImagesInputContent[];\n}\n\nexport type ImagesStopReason = \"stop\" | \"error\" | \"aborted\";\n\nexport interface AssistantImages {\n\tapi: ImagesApi;\n\tprovider: ImagesProviderId;\n\tmodel: string;\n\toutput: ImagesOutputContent[];\n\tresponseId?: string;\n\tusage?: Usage;\n\tstopReason: ImagesStopReason;\n\terrorMessage?: string;\n\ttimestamp: number; // Unix timestamp in milliseconds\n}\n\nimport type { TSchema } from \"typebox\";\n\n/** OpenAI grammar variants for constrained sampling. */\nexport type GrammarFormat = \"openai_lark\" | \"openai_regex\";\n\nexport type GrammarVariants = Partial<Record<GrammarFormat, string>>;\n\n/**\n * Optional provider-side constrained sampling configs for a tool.\n *\n * The `json_schema` value roughly maps to the concept of `strict` in APIs which is\n * implemented as json-schema constrained sampling by APIs. Grammar variants let\n * callers provide provider-specific encodings of the same intended language.\n */\nexport type ConstrainedSamplingConfig =\n\t| {\n\t\t\ttype: \"json_schema\";\n\t\t\tstrict: \"prefer\" | \"require\";\n\t }\n\t| {\n\t\t\ttype: \"grammar\";\n\t\t\tvariants: GrammarVariants;\n\t };\n\nexport interface Tool<TParameters extends TSchema = TSchema> {\n\tname: string;\n\tdescription: string;\n\tparameters: TParameters;\n\tconstrainedSampling?: false | ConstrainedSamplingConfig;\n}\n\nexport interface Context {\n\tsystemPrompt?: string;\n\tmessages: Message[];\n\ttools?: Tool[];\n}\n\n/**\n * Event protocol for AssistantMessageEventStream.\n *\n * Successful streams emit `start` before partial updates and terminate with\n * `done`. A stream may terminate directly with `error` when request setup fails\n * before generation starts; after `start`, failures also terminate with `error`.\n * Direct `streamSimple()` calls throw synchronously when request auth is missing.\n * Updates and `done` must never appear before `start`.\n *\n * Tool-call arguments at `toolcall_start` are provider-specific;\n * `toolcall_delta` carries subsequent JSON updates.\n */\nexport type AssistantMessageEvent =\n\t| { type: \"start\"; partial: AssistantMessage }\n\t| { type: \"text_start\"; contentIndex: number; partial: AssistantMessage }\n\t| { type: \"text_delta\"; contentIndex: number; delta: string; partial: AssistantMessage }\n\t| { type: \"text_end\"; contentIndex: number; content: string; partial: AssistantMessage }\n\t| { type: \"thinking_start\"; contentIndex: number; partial: AssistantMessage }\n\t| { type: \"thinking_delta\"; contentIndex: number; delta: string; partial: AssistantMessage }\n\t| { type: \"thinking_end\"; contentIndex: number; content: string; partial: AssistantMessage }\n\t| { type: \"toolcall_start\"; contentIndex: number; partial: AssistantMessage }\n\t| { type: \"toolcall_delta\"; contentIndex: number; delta: string; partial: AssistantMessage }\n\t| { type: \"toolcall_end\"; contentIndex: number; toolCall: ToolCall; partial: AssistantMessage }\n\t| {\n\t\t\ttype: \"done\";\n\t\t\treason: Extract<StopReason, \"stop\" | \"length\" | \"toolUse\" | \"deferred\">;\n\t\t\tmessage: AssistantMessage;\n\t }\n\t| { type: \"error\"; reason: Extract<StopReason, \"aborted\" | \"error\">; error: AssistantMessage };\n\n/**\n * Compatibility settings for OpenAI-compatible completions APIs.\n * Use this to override URL-based auto-detection for custom providers.\n */\nexport interface OpenAICompletionsCompat {\n\t/** Whether the provider supports the `store` field. Default: auto-detected from URL. */\n\tsupportsStore?: boolean;\n\t/** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */\n\tsupportsDeveloperRole?: boolean;\n\t/** Whether the provider supports `reasoning_effort`. Default: auto-detected from URL. */\n\tsupportsReasoningEffort?: boolean;\n\t/**\n\t * Whether the model accepts the `temperature` request field. Claude Fable 5.1 rejects\n\t * non-default `temperature`, `top_p`, and `top_k` on every request, and OpenRouter's own\n\t * `supported_parameters` for that model omits `temperature`. When false, the provider\n\t * omits `temperature` and also strips `temperature`, `top_p`, and `top_k` from\n\t * `samplingParams`, which is otherwise merged last and would reopen them.\n\t * Default: true.\n\t */\n\tsupportsTemperature?: boolean;\n\t/**\n\t * Whether the model accepts forced tool use (`tool_choice` `\"required\"` or\n\t * `{ type: \"function\", ... }`). Claude Fable 5.1 rejects it on every request, whichever\n\t * platform serves the model. When false, the provider rejects a forced choice with an error\n\t * rather than sending a request the model cannot honor. `\"auto\"` and `\"none\"` are never\n\t * altered. https://platform.claude.com/docs/en/build-with-claude/thinking\n\t * Default: true.\n\t */\n\tsupportsForcedToolChoice?: boolean;\n\t/** Whether the provider supports `stream_options: { include_usage: true }` for token usage in streaming responses. Default: true. */\n\tsupportsUsageInStreaming?: boolean;\n\t/** Whether streamed responses include `finish_reason`. When false, pi infers `stop` or `toolUse` when the stream ends. Default: true. */\n\tsupportsFinishReason?: boolean;\n\t/** Which field to use for max tokens. Default: auto-detected from URL. */\n\tmaxTokensField?: \"max_completion_tokens\" | \"max_tokens\";\n\t/** Whether tool results require the `name` field. Default: auto-detected from URL. */\n\trequiresToolResultName?: boolean;\n\t/** Whether a user message after tool results requires an assistant message in between. Default: auto-detected from URL. */\n\trequiresAssistantAfterToolResult?: boolean;\n\t/** Whether thinking blocks must be converted to text blocks with <thinking> delimiters. Default: auto-detected from URL. */\n\trequiresThinkingAsText?: boolean;\n\t/** Whether all replayed assistant messages must include an empty reasoning_content field when reasoning is enabled. Default: auto-detected from URL. */\n\trequiresReasoningContentOnAssistantMessages?: boolean;\n\t/** Format for reasoning/thinking parameter. \"openai\" uses reasoning_effort, \"openrouter\" uses reasoning: { effort }, \"deepseek\" uses thinking: { type } plus reasoning_effort when supported, \"together\" uses reasoning: { enabled } plus reasoning_effort when supported, \"baseten\" uses configurable chat_template_args plus reasoning_effort when supported, \"zai\" uses thinking: { type }, \"qwen\" uses top-level enable_thinking: boolean, \"qwen-chat-template\" uses chat_template_kwargs.enable_thinking and preserve_thinking, \"chat-template\" uses configurable chat_template_kwargs, \"string-thinking\" uses top-level thinking: string, and \"ant-ling\" uses reasoning: { effort } only when the mapped effort is non-null. Default: \"openai\". */\n\tthinkingFormat?:\n\t\t| \"openai\"\n\t\t| \"openrouter\"\n\t\t| \"deepseek\"\n\t\t| \"together\"\n\t\t| \"baseten\"\n\t\t| \"zai\"\n\t\t| \"qwen\"\n\t\t| \"chat-template\"\n\t\t| \"qwen-chat-template\"\n\t\t| \"string-thinking\"\n\t\t| \"ant-ling\";\n\t/** Kwargs to send as `chat_template_kwargs` when `thinkingFormat` is `chat-template`. Use `{ \"$var\": \"thinking.enabled\" }`, `{ \"$var\": \"thinking.effort\" }`, or `{ \"$var\": \"thinking.budget\" }` for pi-controlled thinking values. */\n\tchatTemplateKwargs?: Record<string, ChatTemplateKwargValue>;\n\t/** Arguments to send as `chat_template_args` when `thinkingFormat` is `baseten`. Use `{ \"$var\": \"thinking.enabled\" }`, `{ \"$var\": \"thinking.effort\" }`, or `{ \"$var\": \"thinking.budget\" }` for pi-controlled thinking values. */\n\tchatTemplateArgs?: Record<string, ChatTemplateKwargValue>;\n\t/** OpenRouter-compatible routing preferences sent as the `provider` request field. */\n\topenRouterRouting?: OpenRouterRouting;\n\t/** Vercel AI Gateway routing preferences. Only used when baseUrl points to Vercel AI Gateway. */\n\tvercelGatewayRouting?: VercelGatewayRouting;\n\t/** Whether z.ai supports top-level `tool_stream: true` for streaming tool call deltas. Default: false. */\n\tzaiToolStream?: boolean;\n\t/**\n\t * Top-level request field used to cap reasoning tokens from `thinkingBudgets`.\n\t * Reasoning and the answer share `max_tokens` on these endpoints, so without a budget a\n\t * reasoning-heavy turn can consume the whole response and emit no answer.\n\t * `\"thinking_token_budget\"` is vLLM, `\"thinking_budget\"` is Qwen/DashScope/SGLang,\n\t * `\"thinking_budget_tokens\"` is llama.cpp. Off by default; not set on the generated catalog.\n\t */\n\tthinkingTokenBudgetField?: ThinkingTokenBudgetField;\n\t/** Alias for `thinkingTokenBudgetField: \"thinking_token_budget\"` (vLLM). Prefer `thinkingTokenBudgetField`. Default: false. */\n\tsupportsThinkingTokenBudget?: boolean;\n\t/** Whether the provider supports OpenAI custom tools with Lark/regex grammar formats. When false, grammar-constrained tools fall back to normal function tools. Default: false; the generated model catalog enables it for capable models. */\n\tsupportsOpenAIGrammarTools?: boolean;\n\t/** Whether the provider supports the `strict` field in tool definitions. Default: true. */\n\tsupportsStrictMode?: boolean;\n\t/** Cache control convention for prompt caching. \"anthropic\" applies Anthropic-style `cache_control` markers to the system prompt, last tool definition, and last user, assistant, or tool-result text content. */\n\tcacheControlFormat?: \"anthropic\";\n\t/** Whether to send session-affinity data from `options.sessionId`. Default: true for OpenRouter endpoints, false otherwise. */\n\tsendSessionAffinityHeaders?: boolean;\n\t/** Provider-specific deferred tool serialization mode. */\n\tdeferredToolsMode?: \"kimi\";\n\t/** Session-affinity header format: `openai` sends `session_id`, `x-client-request-id`, and `x-session-affinity`; `openai-nosession` sends `x-client-request-id` and `x-session-affinity`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */\n\tsessionAffinityFormat?: SessionAffinityFormat;\n\t/** Whether the provider supports long prompt cache retention (`prompt_cache_retention: \"24h\"` or Anthropic-style `cache_control.ttl: \"1h\"`, depending on format). Default: true. */\n\tsupportsLongCacheRetention?: boolean;\n\t/**\n\t * vLLM scheduler priority sent as the top-level `priority` request field (lower values are\n\t * handled earlier; server default 0). Only meaningful when vLLM runs with\n\t * `--scheduling-policy priority`; useful for keeping background/batch work from stalling\n\t * interactive sessions. Off by default; not set on the generated catalog.\n\t */\n\tvllmPriority?: number;\n}\n\n/** Compatibility settings for OpenAI Responses APIs. */\nexport interface OpenAIResponsesCompat {\n\t/** Whether the provider supports the `developer` role (vs `system`). Default: true. */\n\tsupportsDeveloperRole?: boolean;\n\t/** Session-affinity header format: `openai` sends `session_id` and `x-client-request-id`; `openai-nosession` sends `x-client-request-id`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */\n\tsessionAffinityFormat?: SessionAffinityFormat;\n\t/** Whether the provider supports long prompt cache retention. This uses `prompt_cache_options.ttl: \"30m\"` on GPT-5.6+ and `prompt_cache_retention: \"24h\"` on earlier models. Default: true. */\n\tsupportsLongCacheRetention?: boolean;\n\t/** Whether the provider supports strict JSON-schema function tools. Defaults are API-specific; generated OpenAI models enable it explicitly. */\n\tsupportsStrictMode?: boolean;\n\t/** Whether to emit OpenAI custom tools with Lark/regex grammar formats. When false, grammar-constrained tools fall back to normal function tools. Default: false; the generated model catalog enables it for capable models. */\n\tsupportsOpenAIGrammarTools?: boolean;\n\t/** Whether the model supports message-anchored `additional_tools` input items. Default: false. */\n\tsupportsAdditionalTools?: boolean;\n\t/** Whether the model supports client-executed tool search for deferred tools. Default: false. */\n\tsupportsToolSearch?: boolean;\n\t/** Whether the model accepts `prompt_cache_options` (OpenAI GPT-5.6+ prompt caching). Older OpenAI models reject the parameter. Default: false. */\n\tsupportsExplicitPromptCacheMode?: boolean;\n\t/** Whether the provider accepts the `max_output_tokens` parameter. Some Codex-protocol gateways reject it. Default: true. */\n\tsupportsMaxOutputTokens?: boolean;\n}\n\n/** Compatibility settings for Anthropic Messages-compatible APIs. */\nexport interface AnthropicMessagesCompat {\n\t/**\n\t * Whether the provider accepts per-tool `eager_input_streaming`.\n\t * When false, the Anthropic provider omits `tools[].eager_input_streaming`\n\t * and sends the legacy `fine-grained-tool-streaming-2025-05-14` beta header\n\t * for tool-enabled requests.\n\t * Default: true.\n\t */\n\tsupportsEagerToolInputStreaming?: boolean;\n\t/** Whether the provider supports Anthropic long cache retention (`cache_control.ttl: \"1h\"`). Default: true. */\n\tsupportsLongCacheRetention?: boolean;\n\t/**\n\t * Whether to send a session-affinity header from `options.sessionId`\n\t * when caching is enabled. Required for providers like Fireworks that use\n\t * session affinity for prompt cache routing (requests to the same replica\n\t * maximize cache hits).\n\t * Default: true for OpenRouter endpoints, false otherwise.\n\t */\n\tsendSessionAffinityHeaders?: boolean;\n\t/** Session-affinity format. `\"openrouter\"` sends `x-session-id`; defaults to that format for OpenRouter endpoints and `x-session-affinity` elsewhere. */\n\tsessionAffinityFormat?: \"openrouter\";\n\t/**\n\t * Whether the provider supports Anthropic-style `cache_control` markers on\n\t * tool definitions. When false, `cache_control` is omitted from tool params.\n\t * Some Anthropic-compatible providers (e.g., Fireworks) do not support this\n\t * field on tools and may reject or ignore it.\n\t * Default: true.\n\t */\n\tsupportsCacheControlOnTools?: boolean;\n\t/**\n\t * Whether the model accepts the Anthropic `temperature` request field.\n\t * Claude Opus 4.7+ rejects non-default temperature values.\n\t * Default: true.\n\t */\n\tsupportsTemperature?: boolean;\n\t/**\n\t * Whether to force adaptive thinking (`thinking.type: \"adaptive\"` plus\n\t * `output_config.effort`) regardless of the model id. Built-in models that\n\t * require adaptive thinking set this in generated metadata. Custom\n\t * Anthropic-compatible providers can set this to `true` for any model whose\n\t * upstream requires the adaptive format. Set to `false` to\n\t * opt out on overridden built-in models.\n\t * Default: false.\n\t */\n\tforceAdaptiveThinking?: boolean;\n\t/** Whether to replay empty thinking signatures as `signature: \"\"` instead of converting thinking to text. Default: false. */\n\tallowEmptySignature?: boolean;\n\t/** Whether the provider supports Anthropic strict tool schemas. Default: false; generated Anthropic models enable it explicitly. */\n\tsupportsStrictTools?: boolean;\n\t/** Whether the exact model transport supports effort-only system messages and thinking binding controls. Default: false. */\n\tsupportsMidConvoEffort?: boolean;\n\t/**\n\t * Models Anthropic accepts in `fallbacks` for server-side refusal fallback,\n\t * with local pricing metadata for returned fallback responses. When absent or\n\t * empty, callers must omit `fallbacks`; Anthropic rejects the field for models\n\t * with no permitted fallback targets.\n\t */\n\tallowedFallbackModels?: AnthropicAllowedFallbackModel[];\n\t/**\n\t * Whether the provider supports deferred tools loaded by `tool_reference`\n\t * blocks in tool results. Default: true for first-party Anthropic models\n\t * except Haiku and models older than Claude 4.5; false for other providers.\n\t */\n\tsupportsToolReferences?: boolean;\n\t/**\n\t * Whether the model runs Anthropic's preserved-thinking *conversation* check, which\n\t * rejects a request whose `system` prompt, `tools`, or earlier messages changed since a\n\t * replayed thinking block was produced. When true the provider sends the\n\t * `thinking-binding-controls-2026-08-01` beta header and\n\t * `thinking.block_binding.prefix_mismatch_behavior: \"drop_block\"`, so a changed prefix\n\t * drops the affected thinking blocks instead of failing the request with a 400.\n\t * Enforced by default for Anthropic accounts created on or after 2026-08-31.\n\t * Default: false.\n\t */\n\tenforcesPreservedThinkingBinding?: boolean;\n\t/**\n\t * Whether the API adjudicates thinking-block *model* binding itself, always dropping a\n\t * block the target model cannot read. When true, an assistant turn produced by a\n\t * different model on the same provider and API replays its signed `thinking` and\n\t * `redacted_thinking` blocks unchanged rather than being rewritten into plain text, so a\n\t * mid-conversation model switch keeps reasoning the target model is allowed to read.\n\t * Default: false.\n\t */\n\tdelegatesThinkingModelBinding?: boolean;\n\t/**\n\t * Whether the model accepts forced tool use (`tool_choice: {\"type\": \"any\"}` or\n\t * `{\"type\": \"tool\", ...}`). Claude Fable 5.1 and Claude Mythos 5.1 reject it on every\n\t * request with a 400; Anthropic's guidance for those models is to use\n\t * `tool_choice: {\"type\": \"auto\"}` with strict tool use or structured outputs instead.\n\t * When false, the provider rejects a forced choice with an error rather than sending a\n\t * request the model is guaranteed to refuse, or silently substituting a different one.\n\t * `auto` and `none` are never altered.\n\t * https://platform.claude.com/docs/en/build-with-claude/thinking\n\t * Default: true.\n\t */\n\tsupportsForcedToolChoice?: boolean;\n}\n\n/** Compatibility settings for Amazon Bedrock models. */\nexport interface BedrockCompat {\n\t/** Whether the model supports Bedrock strict tool schemas. Default: false. */\n\tsupportsStrictMode?: boolean;\n\t/**\n\t * Whether the model accepts forced tool use (`toolChoice` `\"any\"` or `{ type: \"tool\" }`).\n\t * Claude Fable 5.1 rejects it on every request with a 400, whichever platform serves the\n\t * model. When false, the provider rejects a forced choice with an error rather than sending\n\t * a request that is guaranteed to fail. `auto` and `none` are never altered.\n\t * https://platform.claude.com/docs/en/build-with-claude/thinking\n\t * Default: true.\n\t */\n\tsupportsForcedToolChoice?: boolean;\n\t/**\n\t * Whether the model accepts `inferenceConfig.temperature`. Claude Fable 5.1 rejects\n\t * non-default `temperature`, `top_p`, and `top_k` on every request. When false, the\n\t * provider omits the field.\n\t * Default: true.\n\t */\n\tsupportsTemperature?: boolean;\n}\n\n/**\n * OpenRouter provider routing preferences.\n * Controls which upstream providers OpenRouter routes requests to.\n * Sent as the `provider` field in the OpenRouter API request body.\n * @see https://openrouter.ai/docs/guides/routing/provider-selection\n */\nexport interface OpenRouterRouting {\n\t/** Whether to allow backup providers to serve requests. Default: true. */\n\tallow_fallbacks?: boolean;\n\t/** Whether to filter providers to only those that support all parameters in the request. Default: false. */\n\trequire_parameters?: boolean;\n\t/** Data collection setting. \"allow\" (default): allow providers that may store/train on data. \"deny\": only use providers that don't collect user data. */\n\tdata_collection?: \"deny\" | \"allow\";\n\t/** Whether to restrict routing to only ZDR (Zero Data Retention) endpoints. */\n\tzdr?: boolean;\n\t/** Whether to restrict routing to only models that allow text distillation. */\n\tenforce_distillable_text?: boolean;\n\t/** An ordered list of provider names/slugs to try in sequence, falling back to the next if unavailable. */\n\torder?: string[];\n\t/** List of provider names/slugs to exclusively allow for this request. */\n\tonly?: string[];\n\t/** List of provider names/slugs to skip for this request. */\n\tignore?: string[];\n\t/** A list of quantization levels to filter providers by (e.g., [\"fp16\", \"bf16\", \"fp8\", \"fp6\", \"int8\", \"int4\", \"fp4\", \"fp32\"]). */\n\tquantizations?: string[];\n\t/** Sorting strategy. Can be a string (e.g., \"price\", \"throughput\", \"latency\") or an object with `by` and `partition`. */\n\tsort?:\n\t\t| string\n\t\t| {\n\t\t\t\t/** The sorting metric: \"price\", \"throughput\", \"latency\". */\n\t\t\t\tby?: string;\n\t\t\t\t/** Partitioning strategy: \"model\" (default) or \"none\". */\n\t\t\t\tpartition?: string | null;\n\t\t };\n\t/** Maximum price per million tokens (USD). */\n\tmax_price?: {\n\t\t/** Price per million prompt tokens. */\n\t\tprompt?: number | string;\n\t\t/** Price per million completion tokens. */\n\t\tcompletion?: number | string;\n\t\t/** Price per image. */\n\t\timage?: number | string;\n\t\t/** Price per audio unit. */\n\t\taudio?: number | string;\n\t\t/** Price per request. */\n\t\trequest?: number | string;\n\t};\n\t/** Preferred minimum throughput (tokens/second). Can be a number (applies to p50) or an object with percentile-specific cutoffs. */\n\tpreferred_min_throughput?:\n\t\t| number\n\t\t| {\n\t\t\t\t/** Minimum tokens/second at the 50th percentile. */\n\t\t\t\tp50?: number;\n\t\t\t\t/** Minimum tokens/second at the 75th percentile. */\n\t\t\t\tp75?: number;\n\t\t\t\t/** Minimum tokens/second at the 90th percentile. */\n\t\t\t\tp90?: number;\n\t\t\t\t/** Minimum tokens/second at the 99th percentile. */\n\t\t\t\tp99?: number;\n\t\t };\n\t/** Preferred maximum latency (seconds). Can be a number (applies to p50) or an object with percentile-specific cutoffs. */\n\tpreferred_max_latency?:\n\t\t| number\n\t\t| {\n\t\t\t\t/** Maximum latency in seconds at the 50th percentile. */\n\t\t\t\tp50?: number;\n\t\t\t\t/** Maximum latency in seconds at the 75th percentile. */\n\t\t\t\tp75?: number;\n\t\t\t\t/** Maximum latency in seconds at the 90th percentile. */\n\t\t\t\tp90?: number;\n\t\t\t\t/** Maximum latency in seconds at the 99th percentile. */\n\t\t\t\tp99?: number;\n\t\t };\n}\n\n/**\n * Vercel AI Gateway routing preferences.\n * Controls which upstream providers the gateway routes requests to.\n * @see https://vercel.com/docs/ai-gateway/models-and-providers/provider-options\n */\nexport interface VercelGatewayRouting {\n\t/** List of provider slugs to exclusively use for this request (e.g., [\"bedrock\", \"anthropic\"]). */\n\tonly?: string[];\n\t/** List of provider slugs to try in order (e.g., [\"anthropic\", \"openai\"]). */\n\torder?: string[];\n}\n\nexport interface ModelCostRates {\n\tinput: number; // $/million tokens\n\toutput: number; // $/million tokens\n\tcacheRead: number; // $/million tokens\n\tcacheWrite: number; // $/million tokens\n}\n\nexport interface ModelCostTier extends ModelCostRates {\n\t/** Use this tier for requests whose total input usage exceeds this token count. */\n\tinputTokensAbove: number;\n}\n\nexport interface ModelCost extends ModelCostRates {\n\t/** Request-wide pricing tiers. The highest matching input threshold applies to the full request. */\n\ttiers?: ModelCostTier[];\n}\n\n/**\n * Explicit routing metadata that marks a model as the fast-inference variant of another model.\n *\n * Presence of this field — never a `-fast` name suffix — is what gives a model fast semantics.\n * A provider, `models.json` custom model, or extension may own an exact `<base>-fast` ID without\n * this metadata, in which case it is an ordinary model and routes exactly as it is declared.\n */\nexport interface ModelFastRoute {\n\t/** Canonical ID of the normal-speed model this variant pairs with, in the same provider. */\n\tbaseModelId: string;\n\t/** Model ID to send upstream. OpenAI-style routing keeps the base ID; a provider with real fast siblings sends its own ID. */\n\tupstreamModelId: string;\n\t/** Service tier to send with the request. Set only for providers that route fast traffic through an OpenAI-style tier. */\n\tserviceTier?: \"priority\";\n}\n\n// Model interface for the unified model system\nexport interface Model<TApi extends Api> {\n\tid: string;\n\tname: string;\n\tapi: TApi;\n\tprovider: ProviderId;\n\tbaseUrl: string;\n\treasoning: boolean;\n\t/**\n\t * Maps pi thinking levels to provider/model-specific values.\n\t * Missing keys use provider defaults. null marks a level as unsupported.\n\t */\n\tthinkingLevelMap?: ThinkingLevelMap;\n\t/**\n\t * Modalities Atomic can send to this model. `\"pdf\"` is set only where a runtime can serialize\n\t * a document block — the Anthropic Messages and Amazon Bedrock Converse paths — so a provider\n\t * that publishes PDF support but has no such path here stays at `[\"text\", \"image\"]`.\n\t */\n\tinput: (\"text\" | \"image\" | \"pdf\")[];\n\tcost: ModelCost;\n\tcontextWindow: number;\n\tmaxTokens: number;\n\t/** Default sampling parameters for this model. See {@link StreamOptions.samplingParams}; per-request keys override these. */\n\tsamplingParams?: Record<string, unknown>;\n\theaders?: Record<string, string>;\n\t/**\n\t * Marks this model as the fast-inference variant of {@link ModelFastRoute.baseModelId} and carries the\n\t * upstream routing it needs. Absent on every normal model.\n\t */\n\tfastRoute?: ModelFastRoute;\n\t/** Compatibility overrides for OpenAI-compatible APIs. If not set, auto-detected from baseUrl. */\n\tcompat?: TApi extends \"openai-completions\"\n\t\t? OpenAICompletionsCompat\n\t\t: TApi extends \"openai-responses\" | \"azure-openai-responses\" | \"openai-codex-responses\"\n\t\t\t? OpenAIResponsesCompat\n\t\t\t: TApi extends \"anthropic-messages\"\n\t\t\t\t? AnthropicMessagesCompat\n\t\t\t\t: TApi extends \"bedrock-converse-stream\"\n\t\t\t\t\t? BedrockCompat\n\t\t\t\t\t: never;\n}\n\nexport interface ImagesModel<TApi extends ImagesApi>\n\textends Omit<Model<Api>, \"api\" | \"provider\" | \"reasoning\" | \"contextWindow\" | \"maxTokens\" | \"compat\"> {\n\tapi: TApi;\n\tprovider: ImagesProviderId;\n\toutput: (\"text\" | \"image\")[];\n}\n"]}
1
+ {"version":3,"file":"types.js","sourceRoot":"","sources":["../src/types.ts"],"names":[],"mappings":"","sourcesContent":["import type { TelemetryContext } from \"@earendil-works/pi-telemetry\";\nimport type { AnthropicOptions } from \"./api/anthropic-messages.ts\";\nimport type { AzureOpenAIResponsesOptions } from \"./api/azure-openai-responses.ts\";\nimport type { BedrockOptions } from \"./api/bedrock-converse-stream.ts\";\nimport type { GoogleOptions } from \"./api/google-generative-ai.ts\";\nimport type { GoogleVertexOptions } from \"./api/google-vertex.ts\";\nimport type { MistralOptions } from \"./api/mistral-conversations.ts\";\nimport type { OpenAICodexResponsesOptions } from \"./api/openai-codex-responses.ts\";\nimport type { OpenAICompletionsOptions } from \"./api/openai-completions.ts\";\nimport type { OpenAIResponsesOptions } from \"./api/openai-responses.ts\";\nimport type { PiMessagesOptions } from \"./api/pi-messages.ts\";\nimport type { AssistantMessageDiagnostic } from \"./utils/diagnostics.ts\";\nimport type { AssistantMessageEventStream } from \"./utils/event-stream.ts\";\n\nexport type { AssistantMessageEventStream } from \"./utils/event-stream.ts\";\n\nexport type KnownApi =\n\t| \"openai-completions\"\n\t| \"mistral-conversations\"\n\t| \"openai-responses\"\n\t| \"azure-openai-responses\"\n\t| \"openai-codex-responses\"\n\t| \"anthropic-messages\"\n\t| \"bedrock-converse-stream\"\n\t| \"google-generative-ai\"\n\t| \"google-vertex\"\n\t| \"pi-messages\";\n\nexport type Api = KnownApi | (string & {});\n\nexport type KnownImagesApi = \"openrouter-images\";\n\nexport type ImagesApi = KnownImagesApi | (string & {});\n\nexport type KnownProvider =\n\t| \"amazon-bedrock\"\n\t| \"ant-ling\"\n\t| \"anthropic\"\n\t| \"google\"\n\t| \"google-vertex\"\n\t| \"openai\"\n\t| \"azure-openai-responses\"\n\t| \"openai-codex\"\n\t| \"radius\"\n\t| \"nvidia\"\n\t| \"deepseek\"\n\t| \"github-copilot\"\n\t| \"xai\"\n\t| \"groq\"\n\t| \"cerebras\"\n\t| \"openrouter\"\n\t| \"vercel-ai-gateway\"\n\t| \"zai\"\n\t| \"zai-coding-cn\"\n\t| \"mistral\"\n\t| \"minimax\"\n\t| \"minimax-cn\"\n\t| \"moonshotai\"\n\t| \"moonshotai-cn\"\n\t| \"huggingface\"\n\t| \"fireworks\"\n\t| \"together\"\n\t| \"baseten\"\n\t| \"opencode\"\n\t| \"opencode-go\"\n\t| \"kimi-coding\"\n\t| \"cloudflare-workers-ai\"\n\t| \"cloudflare-ai-gateway\"\n\t| \"qwen-token-plan\"\n\t| \"qwen-token-plan-cn\"\n\t| \"qwen-token-plan-individual\"\n\t| \"xiaomi\"\n\t| \"xiaomi-token-plan-cn\"\n\t| \"xiaomi-token-plan-ams\"\n\t| \"xiaomi-token-plan-sgp\";\nexport type ProviderId = KnownProvider | string;\n\nexport type KnownImagesProvider = \"openrouter\";\n\nexport type ImagesProviderId = KnownImagesProvider | string;\n\nexport type ToolChoice = \"auto\" | \"none\";\nexport type ThinkingLevel = \"minimal\" | \"low\" | \"medium\" | \"high\" | \"xhigh\" | \"max\";\nexport type ModelThinkingLevel = \"off\" | ThinkingLevel;\nexport type ThinkingLevelMap = Partial<Record<ModelThinkingLevel, string | null>>;\nexport type ChatTemplateKwargValue =\n\t| string\n\t| number\n\t| boolean\n\t| null\n\t| {\n\t\t\t$var: \"thinking.enabled\" | \"thinking.effort\" | \"thinking.budget\";\n\t\t\tomitWhenOff?: boolean;\n\t };\n\n/** Top-level request field used to cap reasoning tokens on OpenAI-compatible servers. */\nexport type ThinkingTokenBudgetField = \"thinking_token_budget\" | \"thinking_budget\" | \"thinking_budget_tokens\";\n\n/** Token budgets for each thinking level (token-based providers only) */\nexport interface ThinkingBudgets {\n\tminimal?: number;\n\tlow?: number;\n\tmedium?: number;\n\thigh?: number;\n}\n\n// Base options all providers share\nexport type CacheRetention = \"none\" | \"short\" | \"long\";\n\n/**\n * Best-effort prompt cache lifetime in seconds for each retention tier a request can ask for.\n * A missing tier means the lifetime is unknown; pi does not warm such caches.\n */\nexport type ModelPromptCache = Partial<Record<Exclude<CacheRetention, \"none\">, number>>;\n\nexport type Transport = \"sse\" | \"websocket\" | \"websocket-cached\" | \"auto\";\n\n/** Provider-scoped environment overrides. Values take precedence over process.env. */\nexport type ProviderEnv = Record<string, string>;\nexport type ProviderHeaders = Record<string, string | null>;\nexport type FetchFunction = typeof globalThis.fetch;\nexport type SessionAffinityFormat = \"openai\" | \"openai-nosession\" | \"openrouter\";\n\nexport interface ProviderResponse {\n\tstatus: number;\n\theaders: Record<string, string>;\n}\n\n/** Authentication, HTTP transport, and lifecycle callbacks shared by provider requests. */\nexport interface ProviderRequestOptions<TModel = Model<Api>> {\n\tsignal?: AbortSignal;\n\t/** Explicit parent context for telemetry produced by this logical request. */\n\ttelemetryContext?: TelemetryContext;\n\tapiKey?: string;\n\t/**\n\t * Optional fetch implementation for provider HTTP requests.\n\t * Defaults to `globalThis.fetch`. Provider adapters that cannot inject a custom implementation may reject it.\n\t * This does not affect WebSocket transports.\n\t */\n\tfetch?: FetchFunction;\n\t/**\n\t * Provider-scoped environment values. These take precedence over process.env for\n\t * provider configuration such as regional settings, endpoint placeholders, and\n\t * proxy variables.\n\t */\n\tenv?: ProviderEnv;\n\t/**\n\t * Optional callback for inspecting or replacing provider payloads before sending.\n\t * Return undefined to keep the payload unchanged.\n\t */\n\tonPayload?: (payload: unknown, model: TModel) => unknown | undefined | Promise<unknown | undefined>;\n\t/**\n\t * Optional callback invoked after an HTTP response is received.\n\t */\n\tonResponse?: (response: ProviderResponse, model: TModel) => void | Promise<void>;\n\t/**\n\t * Optional custom HTTP headers to include in API requests.\n\t * Merged with provider defaults; caller values override default headers.\n\t * On AWS Bedrock these are injected via a Smithy `build`-step middleware so\n\t * they are covered by SigV4 signing; reserved headers (`x-amz-*`,\n\t * `authorization`, `host`) are silently ignored to preserve SigV4 / bearer auth.\n\t * A null value suppresses a provider/API default header with the same name.\n\t */\n\theaders?: ProviderHeaders;\n\t/**\n\t * HTTP request timeout in milliseconds for providers/SDKs that support it.\n\t * For example, OpenAI and Anthropic SDK clients default to 10 minutes.\n\t */\n\ttimeoutMs?: number;\n\t/**\n\t * Maximum retry attempts for providers/SDKs that support client-side retries.\n\t * For example, OpenAI and Anthropic SDK clients default to 2.\n\t */\n\tmaxRetries?: number;\n\t/**\n\t * Maximum delay in milliseconds to wait for a retry when the server requests a long wait.\n\t * If the server's requested delay exceeds this value, the request fails immediately\n\t * with an error containing the requested delay, allowing higher-level retry logic\n\t * to handle it with user visibility.\n\t * Default: 60000 (60 seconds). Set to 0 to disable the cap.\n\t */\n\tmaxRetryDelayMs?: number;\n}\n\nexport interface StreamOptions extends ProviderRequestOptions<Model<Api>> {\n\t/**\n\t * Optional callback invoked after an HTTP response is received and before\n\t * its body stream is consumed.\n\t */\n\tonResponse?: (response: ProviderResponse, model: Model<Api>) => void | Promise<void>;\n\ttemperature?: number;\n\t/**\n\t * Arbitrary sampling parameters merged into the request body as-is, after the named request\n\t * fields, so keys here override them. Lets custom OpenAI-compatible servers (llama.cpp, vLLM,\n\t * SGLang, ...) receive parameters pi does not model, e.g. `top_p`, `top_k`, `min_p`,\n\t * `repetition_penalty`. Merged over `Model.samplingParams` per key. Only applied by\n\t * OpenAI-compatible adapters (completions, responses, Azure responses); other APIs ignore it.\n\t */\n\tsamplingParams?: Record<string, unknown>;\n\tmaxTokens?: number;\n\t/**\n\t * Preferred transport for providers that support multiple transports.\n\t * Providers that do not support this option ignore it.\n\t */\n\ttransport?: Transport;\n\t/**\n\t * Prompt cache retention preference. Providers map this to their supported values.\n\t * Default: \"short\".\n\t */\n\tcacheRetention?: CacheRetention;\n\t/**\n\t * Optional session identifier for providers that support session-based caching.\n\t * Providers can use this to enable prompt caching, request routing, or other\n\t * session-aware features. Ignored by providers that don't support it.\n\t */\n\tsessionId?: string;\n\t/**\n\t * WebSocket connect timeout in milliseconds for providers that support\n\t * WebSocket transports. This covers the connection/open handshake only;\n\t * HTTP/SDK idleness uses `timeoutMs`, while gaps between decoded provider\n\t * stream events use `streamDeadlineMs`.\n\t */\n\twebsocketConnectTimeoutMs?: number;\n\t/**\n\t * Maximum idle gap in milliseconds between two provider stream events.\n\t *\n\t * Enforced below the HTTP layer, so a response body that stalls without ever\n\t * rejecting the adapter's async iterator (for example a decompression failure\n\t * that destroys the stream silently, #2553) still settles the attempt as a\n\t * retryable transport error instead of hanging forever. The timer restarts on\n\t * every event, so it never caps total stream duration.\n\t *\n\t * Defaults to 300000 ms. `0` (or any non-positive value) disables it.\n\t */\n\tstreamDeadlineMs?: number;\n\t/**\n\t * Optional metadata to include in API requests.\n\t * Providers extract the fields they understand and ignore the rest.\n\t * For example, Anthropic uses `user_id` for abuse tracking and rate limiting.\n\t */\n\tmetadata?: Record<string, unknown>;\n}\n\nexport type ProviderStreamOptions = StreamOptions & Record<string, unknown>;\n\nexport interface DeferredFetchOptions extends ProviderRequestOptions<Model<Api>> {\n\t/**\n\t * Maximum provider long-poll duration in milliseconds.\n\t * Defaults to 0, which performs one status check.\n\t */\n\twait?: number;\n}\n\n/** Request options for best-effort deferred-response cancellation. */\nexport type DeferredCancelOptions = ProviderRequestOptions<Model<Api>>;\n\n/**\n * Maps known APIs to their full provider-specific stream option types.\n * Type-only imports from API implementation modules are erased at emit, so\n * this is tree-shake safe.\n */\nexport interface ApiOptionsMap {\n\t\"anthropic-messages\": AnthropicOptions;\n\t\"openai-completions\": OpenAICompletionsOptions;\n\t\"openai-responses\": OpenAIResponsesOptions;\n\t\"openai-codex-responses\": OpenAICodexResponsesOptions;\n\t\"azure-openai-responses\": AzureOpenAIResponsesOptions;\n\t\"google-generative-ai\": GoogleOptions;\n\t\"google-vertex\": GoogleVertexOptions;\n\t\"mistral-conversations\": MistralOptions;\n\t\"bedrock-converse-stream\": BedrockOptions;\n\t\"pi-messages\": PiMessagesOptions;\n}\n\n/**\n * Full stream options for an API. Known APIs resolve to their concrete option\n * type; custom API strings fall back to the generic shape.\n */\nexport type ApiStreamOptions<TApi extends Api> = TApi extends keyof ApiOptionsMap\n\t? ApiOptionsMap[TApi]\n\t: StreamOptions & Record<string, unknown>;\n\n/**\n * The uniform stream contract of an API implementation module: every module\n * under `src/api/` exports `stream` and `streamSimple`; capable modules may also\n * export deferred-response methods. Lazy wrappers (`lazyApi()`) and provider\n * factories pass these around as values. This is the untyped dispatch shape;\n * per-API option typing lives on the implementation modules themselves and on\n * `Provider.stream()` via `ApiStreamOptions`.\n */\nexport interface ProviderStreams {\n\tstream(model: Model<Api>, context: TranscriptContext, options?: StreamOptions): AssistantMessageEventStream;\n\tstreamSimple(\n\t\tmodel: Model<Api>,\n\t\tcontext: TranscriptContext,\n\t\toptions?: SimpleStreamOptions,\n\t): AssistantMessageEventStream;\n\tfetchDeferred?(\n\t\tmodel: Model<Api>,\n\t\thandle: DeferredHandle,\n\t\toptions?: DeferredFetchOptions,\n\t): AssistantMessageEventStream;\n\tcancelDeferred?(model: Model<Api>, handle: DeferredHandle, options?: DeferredCancelOptions): Promise<void>;\n}\n\n/**\n * The uniform contract of an image-generation API implementation module:\n * every image API module under `src/api/` exports exactly `generateImages`,\n * so the module itself satisfies this interface. Lazy wrappers and image\n * provider factories pass these around as values.\n */\nexport interface ProviderImages {\n\tgenerateImages(\n\t\tmodel: ImagesModel<ImagesApi>,\n\t\tcontext: ImagesContext,\n\t\toptions?: ImagesOptions,\n\t): Promise<AssistantImages>;\n}\n\nexport interface ImagesOptions extends ProviderRequestOptions<ImagesModel<ImagesApi>> {\n\t/**\n\t * Optional metadata to include in API requests.\n\t * Providers extract the fields they understand and ignore the rest.\n\t */\n\tmetadata?: Record<string, unknown>;\n}\n\nexport type ProviderImagesOptions = ImagesOptions & Record<string, unknown>;\n\nexport interface AnthropicAllowedFallbackModel {\n\tprovider: ProviderId;\n\tmodel: string;\n\tcost: ModelCost;\n}\n\n// Unified options with reasoning passed to streamSimple() and completeSimple()\nexport interface SimpleStreamOptions extends StreamOptions {\n\t/** Provider-neutral tool selection for simple requests. When omitted, adapters use provider-specific behavior. */\n\ttoolChoice?: ToolChoice;\n\treasoning?: ThinkingLevel;\n\t/** Ask a capable provider to return a durable handle and continue the request asynchronously. */\n\tdeferred?: boolean | { window?: \"15m\" | \"1h\" | \"24h\" };\n\t/** Custom token budgets for thinking levels (token-based providers only) */\n\tthinkingBudgets?: ThinkingBudgets;\n}\n\n// Generic StreamFunction with typed options.\n//\n// Contract:\n// - Receives a normalized transcript: the system prompt and tools live in the\n// leading system message, never on the context itself.\n// - Must return an AssistantMessageEventStream.\n// - Direct streamSimple() calls may throw synchronously when request auth is\n// missing. Once a stream is returned, request/model/runtime failures should\n// be encoded in the stream, not thrown.\n// - Error termination must produce an AssistantMessage with stopReason\n// \"error\" or \"aborted\" and errorMessage, emitted via the stream protocol.\nexport type StreamFunction<TApi extends Api = Api, TOptions extends StreamOptions = StreamOptions> = (\n\tmodel: Model<TApi>,\n\tcontext: TranscriptContext,\n\toptions?: TOptions,\n) => AssistantMessageEventStream;\n\nexport type ImagesFunction<TApi extends ImagesApi = ImagesApi, TOptions extends ImagesOptions = ImagesOptions> = (\n\tmodel: ImagesModel<TApi>,\n\tcontext: ImagesContext,\n\toptions?: TOptions,\n) => Promise<AssistantImages>;\n\nexport interface TextSignatureV1 {\n\tv: 1;\n\tid: string;\n\tphase?: \"commentary\" | \"final_answer\";\n}\n\nexport interface TextContent {\n\ttype: \"text\";\n\ttext: string;\n\ttextSignature?: string; // e.g., for OpenAI responses, message metadata (legacy id string or TextSignatureV1 JSON)\n}\n\nexport interface ThinkingContent {\n\ttype: \"thinking\";\n\tthinking: string;\n\tthinkingSignature?: string; // Provider-specific opaque or serialized reasoning replay data\n\t/** When true, the thinking content was redacted by safety filters. The opaque\n\t * encrypted payload is stored in `thinkingSignature` so it can be passed back\n\t * to the API for multi-turn continuity. */\n\tredacted?: boolean;\n}\n\n/**\n * Boundary marker emitted by Anthropic server-side fallback, where one model's output gives way\n * to the next after a classifier refusal.\n *\n * This is public content rather than a stream-only event because it must survive the round trip:\n * \"Keep it exactly where it appeared. The API uses its position to validate the thinking blocks\n * around it, so a request that echoes thinking blocks from both sides of the boundary is rejected\n * if the block is omitted or moved.\"\n * https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback\n */\nexport interface FallbackContent {\n\ttype: \"fallback\";\n\t/** The model that declined. Echoes the model string sent when the declining hop is the request's own model. */\n\tfromModel: string;\n\t/** Resolved id of the model that continues. Always present. */\n\ttoModel: string;\n}\n\nexport interface ImageContent {\n\ttype: \"image\";\n\tdata: string; // base64 encoded image data\n\tmimeType: string; // e.g., \"image/jpeg\", \"image/png\"\n}\n\n/**\n * A document sent as model input, currently PDF only.\n *\n * Anthropic documents PDF as a platform capability — \"All active models support PDF processing\" —\n * routed through the same vision path as images, so it is not a per-model modality. A model\n * advertises it through `Model.input` containing `\"pdf\"`, which generated metadata sets only for\n * the runtimes that can actually serialize a document block. Providers that cannot are sent a\n * visible placeholder instead, exactly as they are for images they cannot accept.\n * https://platform.claude.com/docs/en/build-with-claude/pdf-support\n */\nexport interface DocumentContent {\n\ttype: \"document\";\n\t/** Base64-encoded document bytes. Amazon Bedrock takes raw bytes and decodes this itself. */\n\tdata: string;\n\t/**\n\t * The only media type either serializer implements. It is a literal rather than `string`\n\t * because both paths hardcode PDF — the Anthropic block emits `BetaBase64PDFSource`, whose\n\t * `media_type` is itself a fixed literal, and Bedrock emits `DocumentFormat.PDF` — so a wider\n\t * type would let a caller label arbitrary bytes as a PDF. Widening this union later is\n\t * non-breaking; narrowing it after release would not be. Note that adding a format is not a\n\t * media-type swap on the Anthropic side: plain text needs a structurally different source\n\t * variant (`BetaPlainTextSource`, `type: \"text\"`), and Bedrock's `doc`/`docx`/`xls`/`xlsx`/\n\t * `csv`/`html`/`md` formats have no Anthropic source variant at all.\n\t */\n\tmimeType: \"application/pdf\";\n\t/**\n\t * Optional caller-supplied label. Amazon Bedrock requires a name and warns that it \"is\n\t * vulnerable to prompt injections\", so the Bedrock path sanitizes this to the characters AWS\n\t * permits and falls back to a neutral generated name.\n\t */\n\tname?: string;\n}\n\nexport interface ToolCall {\n\ttype: \"toolCall\";\n\tid: string;\n\tname: string;\n\targuments: JsonObject;\n\tthoughtSignature?: string; // Google-specific: opaque signature for reusing thought context\n\t/** OpenAI Responses namespace for calls to dynamically loaded or namespaced tools. */\n\tnamespace?: string;\n}\n\nexport interface Usage {\n\tinput: number;\n\toutput: number;\n\tcacheRead: number;\n\tcacheWrite: number;\n\t/** Subset of `cacheWrite` written with 1h retention. Anthropic and Bedrock report this split. */\n\tcacheWrite1h?: number;\n\t/**\n\t * Reasoning/thinking tokens, when the provider reports them. This is a subset of\n\t * `output`: `output` already includes these tokens. Set to a number (possibly 0) by\n\t * providers that expose a reasoning breakdown; left undefined by providers that don't.\n\t */\n\treasoning?: number;\n\ttotalTokens: number;\n\tcost: {\n\t\tinput: number;\n\t\toutput: number;\n\t\tcacheRead: number;\n\t\tcacheWrite: number;\n\t\ttotal: number;\n\t};\n}\n\nexport type StopReason = \"pending\" | \"stop\" | \"length\" | \"toolUse\" | \"error\" | \"aborted\" | \"deferred\";\n\nexport type JsonValue = null | boolean | number | string | readonly JsonValue[] | JsonObject;\nexport type JsonObject = { [key: string]: JsonValue };\n\ntype IsAny<T> = 0 extends 1 & T ? true : false;\ntype IsExactlyJsonValue<T> = [T] extends [JsonValue] ? ([JsonValue] extends [T] ? true : false) : false;\ntype IsJsonProperty<T> =\n\tIsAny<T> extends true\n\t\t? false\n\t\t: unknown extends T\n\t\t\t? false\n\t\t\t: [Exclude<T, undefined>] extends [never]\n\t\t\t\t? true\n\t\t\t\t: IsJsonCompatible<Exclude<T, undefined>>;\ntype InvalidJsonKeys<T extends object> = {\n\t[TKey in keyof T]-?: TKey extends string | number ? (IsJsonProperty<T[TKey]> extends true ? never : TKey) : TKey;\n}[keyof T];\ntype IsJsonCompatible<T> =\n\tIsAny<T> extends true\n\t\t? false\n\t\t: unknown extends T\n\t\t\t? false\n\t\t\t: IsExactlyJsonValue<T> extends true\n\t\t\t\t? true\n\t\t\t\t: T extends null | boolean | number | string\n\t\t\t\t\t? true\n\t\t\t\t\t: T extends undefined\n\t\t\t\t\t\t? false\n\t\t\t\t\t\t: T extends readonly (infer TItem)[]\n\t\t\t\t\t\t\t? IsJsonCompatible<TItem>\n\t\t\t\t\t\t\t: T extends (...args: never[]) => unknown\n\t\t\t\t\t\t\t\t? false\n\t\t\t\t\t\t\t\t: T extends object\n\t\t\t\t\t\t\t\t\t? [InvalidJsonKeys<T>] extends [never]\n\t\t\t\t\t\t\t\t\t\t? true\n\t\t\t\t\t\t\t\t\t\t: false\n\t\t\t\t\t\t\t\t\t: false;\n\n/** The JSON representation of a typed in-memory value. Optional object properties remain optional. */\nexport type JsonRepresentation<T> =\n\tIsAny<T> extends true\n\t\t? JsonValue\n\t\t: unknown extends T\n\t\t\t? JsonValue\n\t\t\t: [T] extends [JsonValue]\n\t\t\t\t? T\n\t\t\t\t: T extends readonly unknown[]\n\t\t\t\t\t? { [TKey in keyof T]: JsonRepresentation<Exclude<T[TKey], undefined>> }\n\t\t\t\t\t: T extends object\n\t\t\t\t\t\t? { [TKey in keyof T]: JsonRepresentation<Exclude<T[TKey], undefined>> }\n\t\t\t\t\t\t: never;\n\nexport interface DeferredHandle {\n\tprovider: string;\n\tmodelId: string;\n\tapi: string;\n\t/** Provider token, such as a response id or batch id plus row id. */\n\tid: string;\n\texpiresAt?: number;\n\tpollAfterMs?: number;\n\t/** Provider conversion data required to reconstruct the final assistant message. */\n\tdata?: JsonValue;\n}\n\n/**\n * System instructions and tool declarations at one point in the transcript.\n *\n * The leading system message is the system prompt. Later system messages change it:\n * `content` adds instructions from that point on, `sections` replace or remove named\n * prompt sections, and `toolsAdded`/`toolsRemoved` change the tool set. Replaying\n * every system message in order yields the current prompt and tools. Providers that\n * accept system messages mid-conversation send each one in place; other providers\n * rebuild the leading system message from the replayed state.\n */\nexport interface SystemMessage {\n\trole: \"system\";\n\t/** Instruction text. On the leading message this is the base prompt; later, additional instructions. */\n\tcontent: string | TextContent[];\n\t/**\n\t * Named, ordered prompt sections rendered verbatim after `content`. The leading message\n\t * declares them; later messages replace sections by name, and `null` removes one. Keep\n\t * each section self-delimiting (a tag, a heading) so the model can relate an update to\n\t * the original. Avoid integer-like names; JSON objects reorder those.\n\t */\n\tsections?: Record<string, string | null>;\n\t/** Complete definitions of tools that become available at this point. */\n\ttoolsAdded?: Tool[];\n\t/** Tools that stop being available at this point. */\n\ttoolsRemoved?: ToolReference[];\n\ttimestamp: number; // Unix timestamp in milliseconds\n}\n\nexport interface UserMessage {\n\trole: \"user\";\n\tcontent: string | (TextContent | ImageContent | DocumentContent)[];\n\ttimestamp: number; // Unix timestamp in milliseconds\n}\n\nexport interface AssistantMessage {\n\trole: \"assistant\";\n\tcontent: (TextContent | ThinkingContent | ToolCall | FallbackContent)[];\n\tapi: Api;\n\tprovider: ProviderId;\n\tmodel: string;\n\tresponseModel?: string; // Concrete model reported by the provider when different from the requested `model`\n\tresponseId?: string; // Provider-specific response/message identifier when the upstream API exposes one\n\t/** Exact provider-native effort level used for this response. Absent for legacy or unmanaged responses. */\n\tproviderThinkingLevel?: string;\n\tdiagnostics?: AssistantMessageDiagnostic[]; // Redacted provider/runtime diagnostics for failures and recoveries.\n\tusage: Usage;\n\tstopReason: StopReason;\n\tdeferred?: DeferredHandle;\n\terrorMessage?: string;\n\trawStopReason?: string;\n\t/**\n\t * Provider indication of whether the model explicitly ended its turn.\n\t * Preserved for debugging and does not currently affect agent control flow.\n\t */\n\tendTurn?: boolean;\n\ttimestamp: number; // Unix timestamp in milliseconds\n}\n\nexport type ToolResultMessage<TDetails = JsonValue> =\n\tIsJsonCompatible<TDetails> extends true\n\t\t? {\n\t\t\t\trole: \"toolResult\";\n\t\t\t\ttoolCallId: string;\n\t\t\t\ttoolName: string;\n\t\t\t\tcontent: (TextContent | ImageContent)[]; // Supports text and images\n\t\t\t\tdetails?: JsonRepresentation<TDetails>;\n\t\t\t\t/** Usage from the tool execution itself, if available. Not part of main LLM context accounting. */\n\t\t\t\tusage?: Usage;\n\t\t\t\tisError: boolean;\n\t\t\t\ttimestamp: number; // Unix timestamp in milliseconds\n\t\t\t}\n\t\t: never;\n\nexport type Message = SystemMessage | UserMessage | AssistantMessage | ToolResultMessage;\n\nexport type ImagesInputContent = TextContent | ImageContent;\nexport type ImagesOutputContent = TextContent | ImageContent;\n\nexport interface ImagesContext {\n\tinput: ImagesInputContent[];\n}\n\nexport type ImagesStopReason = \"stop\" | \"error\" | \"aborted\";\n\nexport interface AssistantImages {\n\tapi: ImagesApi;\n\tprovider: ImagesProviderId;\n\tmodel: string;\n\toutput: ImagesOutputContent[];\n\tresponseId?: string;\n\tusage?: Usage;\n\tstopReason: ImagesStopReason;\n\terrorMessage?: string;\n\ttimestamp: number; // Unix timestamp in milliseconds\n}\n\nimport type { TSchema } from \"typebox\";\n\n/** OpenAI grammar variants for constrained sampling. */\nexport type GrammarFormat = \"openai_lark\" | \"openai_regex\";\n\nexport type GrammarVariants = Partial<Record<GrammarFormat, string>>;\n\n/**\n * Optional provider-side constrained sampling configs for a tool.\n *\n * The `json_schema` value roughly maps to the concept of `strict` in APIs which is\n * implemented as json-schema constrained sampling by APIs. Grammar variants let\n * callers provide provider-specific encodings of the same intended language.\n */\nexport type ConstrainedSamplingConfig =\n\t| {\n\t\t\ttype: \"json_schema\";\n\t\t\tstrict: \"prefer\" | \"require\";\n\t }\n\t| {\n\t\t\ttype: \"grammar\";\n\t\t\tvariants: GrammarVariants;\n\t };\n\nexport interface Tool<TParameters extends TSchema = TSchema> {\n\tname: string;\n\tdescription: string;\n\tparameters: TParameters;\n\tconstrainedSampling?: false | ConstrainedSamplingConfig;\n}\n\nexport interface ToolReference {\n\tname: string;\n}\n\n/**\n * Request input accepted by the public stream entry points (`Models.stream()`,\n * `streamSimple()`, ...). `systemPrompt` and `tools` are shorthand for a leading\n * system message; `normalizeContext()` folds them into one before the request\n * reaches a provider.\n */\nexport interface Context {\n\tsystemPrompt?: string;\n\tmessages: Message[];\n\ttools?: Tool[];\n}\n\ndeclare const transcriptContextBrand: unique symbol;\n\n/**\n * Normalized request context passed to providers and API implementations. The\n * prompt and tool declarations are carried by the transcript's system messages.\n * Only `normalizeContext()` produces this type, so a raw `Context` cannot reach\n * provider code by accident.\n */\nexport type TranscriptContext = {\n\tmessages: Message[];\n\treadonly [transcriptContextBrand]: true;\n};\n\n/**\n * Event protocol for AssistantMessageEventStream.\n *\n * Successful streams emit `start` before partial updates and terminate with\n * `done`. A stream may terminate directly with `error` when request setup fails\n * before generation starts; after `start`, failures also terminate with `error`.\n * Direct `streamSimple()` calls throw synchronously when request auth is missing.\n * Updates and `done` must never appear before `start`.\n *\n * Tool-call arguments at `toolcall_start` are provider-specific;\n * `toolcall_delta` carries subsequent JSON updates.\n */\nexport type AssistantMessageEvent =\n\t| { type: \"start\"; partial: AssistantMessage }\n\t| { type: \"text_start\"; contentIndex: number; partial: AssistantMessage }\n\t| { type: \"text_delta\"; contentIndex: number; delta: string; partial: AssistantMessage }\n\t| { type: \"text_end\"; contentIndex: number; content: string; partial: AssistantMessage }\n\t| { type: \"thinking_start\"; contentIndex: number; partial: AssistantMessage }\n\t| { type: \"thinking_delta\"; contentIndex: number; delta: string; partial: AssistantMessage }\n\t| { type: \"thinking_end\"; contentIndex: number; content: string; partial: AssistantMessage }\n\t| { type: \"toolcall_start\"; contentIndex: number; partial: AssistantMessage }\n\t| { type: \"toolcall_delta\"; contentIndex: number; delta: string; partial: AssistantMessage }\n\t| { type: \"toolcall_end\"; contentIndex: number; toolCall: ToolCall; partial: AssistantMessage }\n\t| {\n\t\t\ttype: \"done\";\n\t\t\treason: Extract<StopReason, \"stop\" | \"length\" | \"toolUse\" | \"deferred\">;\n\t\t\tmessage: AssistantMessage;\n\t }\n\t| { type: \"error\"; reason: Extract<StopReason, \"aborted\" | \"error\">; error: AssistantMessage };\n\n/**\n * Compatibility settings for OpenAI-compatible completions APIs.\n * Use this to override URL-based auto-detection for custom providers.\n */\nexport interface OpenAICompletionsCompat {\n\t/** Whether the provider supports the `store` field. Default: auto-detected from URL. */\n\tsupportsStore?: boolean;\n\t/** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */\n\tsupportsDeveloperRole?: boolean;\n\t/** Whether the provider supports `reasoning_effort`. Default: auto-detected from URL. */\n\tsupportsReasoningEffort?: boolean;\n\t/**\n\t * Whether the model accepts the `temperature` request field. Claude Fable 5.1 rejects\n\t * non-default `temperature`, `top_p`, and `top_k` on every request, and OpenRouter's own\n\t * `supported_parameters` for that model omits `temperature`. When false, the provider\n\t * omits `temperature` and also strips `temperature`, `top_p`, and `top_k` from\n\t * `samplingParams`, which is otherwise merged last and would reopen them.\n\t * Default: true.\n\t */\n\tsupportsTemperature?: boolean;\n\t/**\n\t * Whether the model accepts forced tool use (`tool_choice` `\"required\"` or\n\t * `{ type: \"function\", ... }`). Claude Fable 5.1 rejects it on every request, whichever\n\t * platform serves the model. When false, the provider rejects a forced choice with an error\n\t * rather than sending a request the model cannot honor. `\"auto\"` and `\"none\"` are never\n\t * altered. https://platform.claude.com/docs/en/build-with-claude/thinking\n\t * Default: true.\n\t */\n\tsupportsForcedToolChoice?: boolean;\n\t/** Whether the provider supports `stream_options: { include_usage: true }` for token usage in streaming responses. Default: true. */\n\tsupportsUsageInStreaming?: boolean;\n\t/** Whether streamed responses include `finish_reason`. When false, pi infers `stop` or `toolUse` when the stream ends. Default: true. */\n\tsupportsFinishReason?: boolean;\n\t/** Which field to use for max tokens. Default: auto-detected from URL. */\n\tmaxTokensField?: \"max_completion_tokens\" | \"max_tokens\";\n\t/** Whether tool results require the `name` field. Default: auto-detected from URL. */\n\trequiresToolResultName?: boolean;\n\t/** Whether a user message after tool results requires an assistant message in between. Default: auto-detected from URL. */\n\trequiresAssistantAfterToolResult?: boolean;\n\t/** Whether thinking blocks must be converted to text blocks with <thinking> delimiters. Default: auto-detected from URL. */\n\trequiresThinkingAsText?: boolean;\n\t/** Whether all replayed assistant messages must include an empty reasoning_content field when reasoning is enabled. Default: auto-detected from URL. */\n\trequiresReasoningContentOnAssistantMessages?: boolean;\n\t/** Format for reasoning/thinking parameter. \"openai\" uses reasoning_effort, \"openrouter\" uses reasoning: { effort }, \"deepseek\" uses thinking: { type } plus reasoning_effort when supported, \"together\" uses reasoning: { enabled } plus reasoning_effort when supported, \"baseten\" uses configurable chat_template_args plus reasoning_effort when supported, \"zai\" uses thinking: { type }, \"qwen\" uses top-level enable_thinking: boolean, \"qwen-chat-template\" uses chat_template_kwargs.enable_thinking and preserve_thinking, \"chat-template\" uses configurable chat_template_kwargs, \"string-thinking\" uses top-level thinking: string, and \"ant-ling\" uses reasoning: { effort } only when the mapped effort is non-null. Default: \"openai\". */\n\tthinkingFormat?:\n\t\t| \"openai\"\n\t\t| \"openrouter\"\n\t\t| \"deepseek\"\n\t\t| \"together\"\n\t\t| \"baseten\"\n\t\t| \"zai\"\n\t\t| \"qwen\"\n\t\t| \"chat-template\"\n\t\t| \"qwen-chat-template\"\n\t\t| \"string-thinking\"\n\t\t| \"ant-ling\";\n\t/** Kwargs to send as `chat_template_kwargs` when `thinkingFormat` is `chat-template`. Use `{ \"$var\": \"thinking.enabled\" }`, `{ \"$var\": \"thinking.effort\" }`, or `{ \"$var\": \"thinking.budget\" }` for pi-controlled thinking values. */\n\tchatTemplateKwargs?: Record<string, ChatTemplateKwargValue>;\n\t/** Arguments to send as `chat_template_args` when `thinkingFormat` is `baseten`. Use `{ \"$var\": \"thinking.enabled\" }`, `{ \"$var\": \"thinking.effort\" }`, or `{ \"$var\": \"thinking.budget\" }` for pi-controlled thinking values. */\n\tchatTemplateArgs?: Record<string, ChatTemplateKwargValue>;\n\t/** OpenRouter-compatible routing preferences sent as the `provider` request field. */\n\topenRouterRouting?: OpenRouterRouting;\n\t/** Vercel AI Gateway routing preferences. Only used when baseUrl points to Vercel AI Gateway. */\n\tvercelGatewayRouting?: VercelGatewayRouting;\n\t/** Whether z.ai supports top-level `tool_stream: true` for streaming tool call deltas. Default: false. */\n\tzaiToolStream?: boolean;\n\t/**\n\t * Top-level request field used to cap reasoning tokens from `thinkingBudgets`.\n\t * Reasoning and the answer share `max_tokens` on these endpoints, so without a budget a\n\t * reasoning-heavy turn can consume the whole response and emit no answer.\n\t * `\"thinking_token_budget\"` is vLLM, `\"thinking_budget\"` is Qwen/DashScope/SGLang,\n\t * `\"thinking_budget_tokens\"` is llama.cpp. Off by default; not set on the generated catalog.\n\t */\n\tthinkingTokenBudgetField?: ThinkingTokenBudgetField;\n\t/** Alias for `thinkingTokenBudgetField: \"thinking_token_budget\"` (vLLM). Prefer `thinkingTokenBudgetField`. Default: false. */\n\tsupportsThinkingTokenBudget?: boolean;\n\t/** Whether the provider supports OpenAI custom tools with Lark/regex grammar formats. When false, grammar-constrained tools fall back to normal function tools. Default: false; the generated model catalog enables it for capable models. */\n\tsupportsOpenAIGrammarTools?: boolean;\n\t/** Whether the exact model accepts system or developer messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false; the generated model catalog enables it for verified models. */\n\tsupportsMidConvoSystemMessages?: boolean;\n\t/** Whether system messages can introduce additional tools mid-conversation. Requires `supportsMidConvoSystemMessages`. Default: false; the generated model catalog enables it for capable models. */\n\tsupportsMidConvoToolAdditions?: boolean;\n\t/** Whether the provider supports the `strict` field in tool definitions. Default: true. */\n\tsupportsStrictMode?: boolean;\n\t/** Cache control convention for prompt caching. \"anthropic\" applies Anthropic-style `cache_control` markers to the system prompt, last tool definition, and last user, assistant, or tool-result text content. */\n\tcacheControlFormat?: \"anthropic\";\n\t/** Whether to send session-affinity data from `options.sessionId`. Default: true for OpenRouter endpoints, false otherwise. */\n\tsendSessionAffinityHeaders?: boolean;\n\t/** Session-affinity header format: `openai` sends `session_id`, `x-client-request-id`, and `x-session-affinity`; `openai-nosession` sends `x-client-request-id` and `x-session-affinity`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */\n\tsessionAffinityFormat?: SessionAffinityFormat;\n\t/** Whether the provider supports long prompt cache retention (`prompt_cache_retention: \"24h\"` or Anthropic-style `cache_control.ttl: \"1h\"`, depending on format). Default: true. */\n\tsupportsLongCacheRetention?: boolean;\n\t/**\n\t * vLLM scheduler priority sent as the top-level `priority` request field (lower values are\n\t * handled earlier; server default 0). Only meaningful when vLLM runs with\n\t * `--scheduling-policy priority`; useful for keeping background/batch work from stalling\n\t * interactive sessions. Off by default; not set on the generated catalog.\n\t */\n\tvllmPriority?: number;\n}\n\n/** Compatibility settings for OpenAI Responses APIs. */\nexport interface OpenAIResponsesCompat {\n\t/** Whether the provider supports the `developer` role (vs `system`). Default: true. */\n\tsupportsDeveloperRole?: boolean;\n\t/** Whether the exact model accepts developer or system messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false; the generated model catalog enables it for verified models. */\n\tsupportsMidConvoSystemMessages?: boolean;\n\t/** Session-affinity header format: `openai` sends `session_id` and `x-client-request-id`; `openai-nosession` sends `x-client-request-id`; `openrouter` sends `x-session-id`. Does not affect the `prompt_cache_key` body param, which is governed by cache retention. Default: auto-detected. */\n\tsessionAffinityFormat?: SessionAffinityFormat;\n\t/** Whether the provider supports long prompt cache retention. This uses `prompt_cache_options.ttl: \"30m\"` on GPT-5.6+ and `prompt_cache_retention: \"24h\"` on earlier models. Default: true. */\n\tsupportsLongCacheRetention?: boolean;\n\t/** Whether the provider supports strict JSON-schema function tools. Defaults are API-specific; generated OpenAI models enable it explicitly. */\n\tsupportsStrictMode?: boolean;\n\t/** Whether to emit OpenAI custom tools with Lark/regex grammar formats. When false, grammar-constrained tools fall back to normal function tools. Default: false; the generated model catalog enables it for capable models. */\n\tsupportsOpenAIGrammarTools?: boolean;\n\t/** Whether the model supports message-anchored `additional_tools` input items. Default: false. */\n\tsupportsAdditionalTools?: boolean;\n\t/** Whether the model supports client-executed tool search for transcript-anchored additions. Default: false. */\n\tsupportsToolSearch?: boolean;\n\t/** Whether the model accepts `prompt_cache_options` (OpenAI GPT-5.6+ prompt caching). Older OpenAI models reject the parameter. Default: false. */\n\tsupportsExplicitPromptCacheMode?: boolean;\n\t/** Whether the provider accepts the `max_output_tokens` parameter. Some Codex-protocol gateways reject it. Default: true. */\n\tsupportsMaxOutputTokens?: boolean;\n}\n\n/** Compatibility settings for Anthropic Messages-compatible APIs. */\nexport interface AnthropicMessagesCompat {\n\t/**\n\t * Whether the provider accepts per-tool `eager_input_streaming`.\n\t * When false, the Anthropic provider omits `tools[].eager_input_streaming`\n\t * and sends the legacy `fine-grained-tool-streaming-2025-05-14` beta header\n\t * for tool-enabled requests.\n\t * Default: true.\n\t */\n\tsupportsEagerToolInputStreaming?: boolean;\n\t/** Whether the provider supports Anthropic long cache retention (`cache_control.ttl: \"1h\"`). Default: true. */\n\tsupportsLongCacheRetention?: boolean;\n\t/**\n\t * Whether to send a session-affinity header from `options.sessionId`\n\t * when caching is enabled. Required for providers like Fireworks that use\n\t * session affinity for prompt cache routing (requests to the same replica\n\t * maximize cache hits).\n\t * Default: true for OpenRouter endpoints, false otherwise.\n\t */\n\tsendSessionAffinityHeaders?: boolean;\n\t/** Session-affinity format. `\"openrouter\"` sends `x-session-id`; defaults to that format for OpenRouter endpoints and `x-session-affinity` elsewhere. */\n\tsessionAffinityFormat?: \"openrouter\";\n\t/**\n\t * Whether the provider supports Anthropic-style `cache_control` markers on\n\t * tool definitions. When false, `cache_control` is omitted from tool params.\n\t * Some Anthropic-compatible providers (e.g., Fireworks) do not support this\n\t * field on tools and may reject or ignore it.\n\t * Default: true.\n\t */\n\tsupportsCacheControlOnTools?: boolean;\n\t/**\n\t * Whether the model accepts the Anthropic `temperature` request field.\n\t * Claude Opus 4.7+ rejects non-default temperature values.\n\t * Default: true.\n\t */\n\tsupportsTemperature?: boolean;\n\t/**\n\t * Whether to force adaptive thinking (`thinking.type: \"adaptive\"` plus\n\t * `output_config.effort`) regardless of the model id. Built-in models that\n\t * require adaptive thinking set this in generated metadata. Custom\n\t * Anthropic-compatible providers can set this to `true` for any model whose\n\t * upstream requires the adaptive format. Set to `false` to\n\t * opt out on overridden built-in models.\n\t * Default: false.\n\t */\n\tforceAdaptiveThinking?: boolean;\n\t/** Whether to replay empty thinking signatures as `signature: \"\"` instead of converting thinking to text. Default: false. */\n\tallowEmptySignature?: boolean;\n\t/** Whether the provider supports Anthropic strict tool schemas. Default: false; generated Anthropic models enable it explicitly. */\n\tsupportsStrictTools?: boolean;\n\t/** Whether the exact model transport supports effort-only system messages and thinking binding controls. Default: false. */\n\tsupportsMidConvoEffort?: boolean;\n\t/** Whether the exact model accepts system-role messages inside the conversation. When false, later system messages are folded into the top-level system prompt. Default: false. */\n\tsupportsMidConvoSystemMessages?: boolean;\n\t/** Whether the exact model accepts mid-conversation `tool_addition` and `tool_removal` blocks. Requires `supportsMidConvoSystemMessages`. Default: false. */\n\tsupportsMidConvoToolChanges?: boolean;\n\t/**\n\t * Models Anthropic accepts in `fallbacks` for server-side refusal fallback,\n\t * with local pricing metadata for returned fallback responses. When absent or\n\t * empty, callers must omit `fallbacks`; Anthropic rejects the field for models\n\t * with no permitted fallback targets.\n\t */\n\tallowedFallbackModels?: AnthropicAllowedFallbackModel[];\n\t/**\n\t * Whether the model runs Anthropic's preserved-thinking *conversation* check, which\n\t * rejects a request whose `system` prompt, `tools`, or earlier messages changed since a\n\t * replayed thinking block was produced. When true the provider sends the\n\t * `thinking-binding-controls-2026-08-01` beta header and\n\t * `thinking.block_binding.prefix_mismatch_behavior: \"drop_block\"`, so a changed prefix\n\t * drops the affected thinking blocks instead of failing the request with a 400.\n\t * Enforced by default for Anthropic accounts created on or after 2026-08-31.\n\t * Default: false.\n\t */\n\tenforcesPreservedThinkingBinding?: boolean;\n\t/**\n\t * Whether the API adjudicates thinking-block *model* binding itself, always dropping a\n\t * block the target model cannot read. When true, an assistant turn produced by a\n\t * different model on the same provider and API replays its signed `thinking` and\n\t * `redacted_thinking` blocks unchanged rather than being rewritten into plain text, so a\n\t * mid-conversation model switch keeps reasoning the target model is allowed to read.\n\t * Default: false.\n\t */\n\tdelegatesThinkingModelBinding?: boolean;\n\t/**\n\t * Whether the model accepts forced tool use (`tool_choice: {\"type\": \"any\"}` or\n\t * `{\"type\": \"tool\", ...}`). Claude Fable 5.1 and Claude Mythos 5.1 reject it on every\n\t * request with a 400; Anthropic's guidance for those models is to use\n\t * `tool_choice: {\"type\": \"auto\"}` with strict tool use or structured outputs instead.\n\t * When false, the provider rejects a forced choice with an error rather than sending a\n\t * request the model is guaranteed to refuse, or silently substituting a different one.\n\t * `auto` and `none` are never altered.\n\t * https://platform.claude.com/docs/en/build-with-claude/thinking\n\t * Default: true.\n\t */\n\tsupportsForcedToolChoice?: boolean;\n}\n\n/** Compatibility settings for Amazon Bedrock models. */\nexport interface BedrockCompat {\n\t/** Whether the model supports Bedrock strict tool schemas. Default: false. */\n\tsupportsStrictMode?: boolean;\n\t/**\n\t * Whether the model accepts forced tool use (`toolChoice` `\"any\"` or `{ type: \"tool\" }`).\n\t * Claude Fable 5.1 rejects it on every request with a 400, whichever platform serves the\n\t * model. When false, the provider rejects a forced choice with an error rather than sending\n\t * a request that is guaranteed to fail. `auto` and `none` are never altered.\n\t * https://platform.claude.com/docs/en/build-with-claude/thinking\n\t * Default: true.\n\t */\n\tsupportsForcedToolChoice?: boolean;\n\t/**\n\t * Whether the model accepts `inferenceConfig.temperature`. Claude Fable 5.1 rejects\n\t * non-default `temperature`, `top_p`, and `top_k` on every request. When false, the\n\t * provider omits the field.\n\t * Default: true.\n\t */\n\tsupportsTemperature?: boolean;\n}\n\n/** Compatibility settings for the Mistral chat API. */\nexport interface MistralConversationsCompat {\n\t/** Whether the exact model accepts system messages after the conversation has started. When false, later system messages are folded into the leading system message. Default: false. */\n\tsupportsMidConvoSystemMessages?: boolean;\n}\n\n/**\n * OpenRouter provider routing preferences.\n * Controls which upstream providers OpenRouter routes requests to.\n * Sent as the `provider` field in the OpenRouter API request body.\n * @see https://openrouter.ai/docs/guides/routing/provider-selection\n */\nexport interface OpenRouterRouting {\n\t/** Whether to allow backup providers to serve requests. Default: true. */\n\tallow_fallbacks?: boolean;\n\t/** Whether to filter providers to only those that support all parameters in the request. Default: false. */\n\trequire_parameters?: boolean;\n\t/** Data collection setting. \"allow\" (default): allow providers that may store/train on data. \"deny\": only use providers that don't collect user data. */\n\tdata_collection?: \"deny\" | \"allow\";\n\t/** Whether to restrict routing to only ZDR (Zero Data Retention) endpoints. */\n\tzdr?: boolean;\n\t/** Whether to restrict routing to only models that allow text distillation. */\n\tenforce_distillable_text?: boolean;\n\t/** An ordered list of provider names/slugs to try in sequence, falling back to the next if unavailable. */\n\torder?: string[];\n\t/** List of provider names/slugs to exclusively allow for this request. */\n\tonly?: string[];\n\t/** List of provider names/slugs to skip for this request. */\n\tignore?: string[];\n\t/** A list of quantization levels to filter providers by (e.g., [\"fp16\", \"bf16\", \"fp8\", \"fp6\", \"int8\", \"int4\", \"fp4\", \"fp32\"]). */\n\tquantizations?: string[];\n\t/** Sorting strategy. Can be a string (e.g., \"price\", \"throughput\", \"latency\") or an object with `by` and `partition`. */\n\tsort?:\n\t\t| string\n\t\t| {\n\t\t\t\t/** The sorting metric: \"price\", \"throughput\", \"latency\". */\n\t\t\t\tby?: string;\n\t\t\t\t/** Partitioning strategy: \"model\" (default) or \"none\". */\n\t\t\t\tpartition?: string | null;\n\t\t };\n\t/** Maximum price per million tokens (USD). */\n\tmax_price?: {\n\t\t/** Price per million prompt tokens. */\n\t\tprompt?: number | string;\n\t\t/** Price per million completion tokens. */\n\t\tcompletion?: number | string;\n\t\t/** Price per image. */\n\t\timage?: number | string;\n\t\t/** Price per audio unit. */\n\t\taudio?: number | string;\n\t\t/** Price per request. */\n\t\trequest?: number | string;\n\t};\n\t/** Preferred minimum throughput (tokens/second). Can be a number (applies to p50) or an object with percentile-specific cutoffs. */\n\tpreferred_min_throughput?:\n\t\t| number\n\t\t| {\n\t\t\t\t/** Minimum tokens/second at the 50th percentile. */\n\t\t\t\tp50?: number;\n\t\t\t\t/** Minimum tokens/second at the 75th percentile. */\n\t\t\t\tp75?: number;\n\t\t\t\t/** Minimum tokens/second at the 90th percentile. */\n\t\t\t\tp90?: number;\n\t\t\t\t/** Minimum tokens/second at the 99th percentile. */\n\t\t\t\tp99?: number;\n\t\t };\n\t/** Preferred maximum latency (seconds). Can be a number (applies to p50) or an object with percentile-specific cutoffs. */\n\tpreferred_max_latency?:\n\t\t| number\n\t\t| {\n\t\t\t\t/** Maximum latency in seconds at the 50th percentile. */\n\t\t\t\tp50?: number;\n\t\t\t\t/** Maximum latency in seconds at the 75th percentile. */\n\t\t\t\tp75?: number;\n\t\t\t\t/** Maximum latency in seconds at the 90th percentile. */\n\t\t\t\tp90?: number;\n\t\t\t\t/** Maximum latency in seconds at the 99th percentile. */\n\t\t\t\tp99?: number;\n\t\t };\n}\n\n/**\n * Vercel AI Gateway routing preferences.\n * Controls which upstream providers the gateway routes requests to.\n * @see https://vercel.com/docs/ai-gateway/models-and-providers/provider-options\n */\nexport interface VercelGatewayRouting {\n\t/** List of provider slugs to exclusively use for this request (e.g., [\"bedrock\", \"anthropic\"]). */\n\tonly?: string[];\n\t/** List of provider slugs to try in order (e.g., [\"anthropic\", \"openai\"]). */\n\torder?: string[];\n}\n\nexport interface ModelCostRates {\n\tinput: number; // $/million tokens\n\toutput: number; // $/million tokens\n\tcacheRead: number; // $/million tokens\n\tcacheWrite: number; // $/million tokens\n}\n\nexport interface ModelCostTier extends ModelCostRates {\n\t/** Use this tier for requests whose total input usage exceeds this token count. */\n\tinputTokensAbove: number;\n}\n\nexport interface ModelCost extends ModelCostRates {\n\t/** Request-wide pricing tiers. The highest matching input threshold applies to the full request. */\n\ttiers?: ModelCostTier[];\n}\n\n/**\n * Explicit routing metadata that marks a model as the fast-inference variant of another model.\n *\n * Presence of this field — never a `-fast` name suffix — is what gives a model fast semantics.\n * A provider, `models.json` custom model, or extension may own an exact `<base>-fast` ID without\n * this metadata, in which case it is an ordinary model and routes exactly as it is declared.\n */\nexport interface ModelFastRoute {\n\t/** Canonical ID of the normal-speed model this variant pairs with, in the same provider. */\n\tbaseModelId: string;\n\t/** Model ID to send upstream. OpenAI-style routing keeps the base ID; a provider with real fast siblings sends its own ID. */\n\tupstreamModelId: string;\n\t/** Service tier to send with the request. Set only for providers that route fast traffic through an OpenAI-style tier. */\n\tserviceTier?: \"priority\";\n}\n\n// Model interface for the unified model system\nexport interface Model<TApi extends Api> {\n\tid: string;\n\tname: string;\n\tapi: TApi;\n\tprovider: ProviderId;\n\tbaseUrl: string;\n\treasoning: boolean;\n\t/**\n\t * Maps pi thinking levels to provider/model-specific values.\n\t * Missing keys use provider defaults. null marks a level as unsupported.\n\t */\n\tthinkingLevelMap?: ThinkingLevelMap;\n\t/**\n\t * Modalities Atomic can send to this model. `\"pdf\"` is set only where a runtime can serialize\n\t * a document block — the Anthropic Messages and Amazon Bedrock Converse paths — so a provider\n\t * that publishes PDF support but has no such path here stays at `[\"text\", \"image\"]`.\n\t */\n\tinput: (\"text\" | \"image\" | \"pdf\")[];\n\tcost: ModelCost;\n\t/** Prompt cache lifetimes per retention tier. Unset when the provider's cache behavior is unknown. */\n\tpromptCache?: ModelPromptCache;\n\tcontextWindow: number;\n\tmaxTokens: number;\n\t/** Default sampling parameters for this model. See {@link StreamOptions.samplingParams}; per-request keys override these. */\n\tsamplingParams?: Record<string, unknown>;\n\theaders?: Record<string, string>;\n\t/**\n\t * Marks this model as the fast-inference variant of {@link ModelFastRoute.baseModelId} and carries the\n\t * upstream routing it needs. Absent on every normal model.\n\t */\n\tfastRoute?: ModelFastRoute;\n\t/** Compatibility overrides for OpenAI-compatible APIs. If not set, auto-detected from baseUrl. */\n\tcompat?: TApi extends \"openai-completions\"\n\t\t? OpenAICompletionsCompat\n\t\t: TApi extends \"openai-responses\" | \"azure-openai-responses\" | \"openai-codex-responses\"\n\t\t\t? OpenAIResponsesCompat\n\t\t\t: TApi extends \"anthropic-messages\"\n\t\t\t\t? AnthropicMessagesCompat\n\t\t\t\t: TApi extends \"bedrock-converse-stream\"\n\t\t\t\t\t? BedrockCompat\n\t\t\t\t\t: TApi extends \"mistral-conversations\"\n\t\t\t\t\t\t? MistralConversationsCompat\n\t\t\t\t\t\t: never;\n}\n\nexport interface ImagesModel<TApi extends ImagesApi>\n\textends Omit<Model<Api>, \"api\" | \"provider\" | \"reasoning\" | \"contextWindow\" | \"maxTokens\" | \"compat\"> {\n\tapi: TApi;\n\tprovider: ImagesProviderId;\n\toutput: (\"text\" | \"image\")[];\n}\n"]}
@@ -1,3 +1,4 @@
1
+ import type { JsonObject } from "../types.ts";
1
2
  export interface DiagnosticErrorInfo {
2
3
  name?: string;
3
4
  message: string;
@@ -8,11 +9,11 @@ export interface AssistantMessageDiagnostic {
8
9
  type: string;
9
10
  timestamp: number;
10
11
  error?: DiagnosticErrorInfo;
11
- details?: Record<string, unknown>;
12
+ details?: JsonObject;
12
13
  }
13
14
  export declare function formatThrownValue(value: unknown): string;
14
15
  export declare function extractDiagnosticError(error: unknown): DiagnosticErrorInfo;
15
- export declare function createAssistantMessageDiagnostic(type: string, error: unknown, details?: Record<string, unknown>): AssistantMessageDiagnostic;
16
+ export declare function createAssistantMessageDiagnostic(type: string, error: unknown, details?: JsonObject): AssistantMessageDiagnostic;
16
17
  export declare function appendAssistantMessageDiagnostic<T extends {
17
18
  diagnostics?: AssistantMessageDiagnostic[];
18
19
  }>(message: T, diagnostic: AssistantMessageDiagnostic): void;
@@ -1 +1 @@
1
- {"version":3,"file":"diagnostics.d.ts","sourceRoot":"","sources":["../../src/utils/diagnostics.ts"],"names":[],"mappings":"AAAA,MAAM,WAAW,mBAAmB;IACnC,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,OAAO,EAAE,MAAM,CAAC;IAChB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,IAAI,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC;CACvB;AAED,MAAM,WAAW,0BAA0B;IAC1C,IAAI,EAAE,MAAM,CAAC;IACb,SAAS,EAAE,MAAM,CAAC;IAClB,KAAK,CAAC,EAAE,mBAAmB,CAAC;IAC5B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAC;CAClC;AAED,wBAAgB,iBAAiB,CAAC,KAAK,EAAE,OAAO,GAAG,MAAM,CAIxD;AAED,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,OAAO,GAAG,mBAAmB,CAS1E;AAED,wBAAgB,gCAAgC,CAC/C,IAAI,EAAE,MAAM,EACZ,KAAK,EAAE,OAAO,EACd,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,GAC/B,0BAA0B,CAE5B;AAED,wBAAgB,gCAAgC,CAAC,CAAC,SAAS;IAAE,WAAW,CAAC,EAAE,0BAA0B,EAAE,CAAA;CAAE,EACxG,OAAO,EAAE,CAAC,EACV,UAAU,EAAE,0BAA0B,GACpC,IAAI,CAEN","sourcesContent":["export interface DiagnosticErrorInfo {\n\tname?: string;\n\tmessage: string;\n\tstack?: string;\n\tcode?: string | number;\n}\n\nexport interface AssistantMessageDiagnostic {\n\ttype: string;\n\ttimestamp: number;\n\terror?: DiagnosticErrorInfo;\n\tdetails?: Record<string, unknown>;\n}\n\nexport function formatThrownValue(value: unknown): string {\n\tif (value instanceof Error) return value.message || value.name;\n\tif (typeof value === \"string\") return value;\n\treturn String(value);\n}\n\nexport function extractDiagnosticError(error: unknown): DiagnosticErrorInfo {\n\tif (!(error instanceof Error)) return { name: \"ThrownValue\", message: formatThrownValue(error) };\n\tconst code = (error as Error & { code?: unknown }).code;\n\treturn {\n\t\tname: error.name || undefined,\n\t\tmessage: error.message || error.name,\n\t\tstack: error.stack,\n\t\tcode: typeof code === \"string\" || typeof code === \"number\" ? code : undefined,\n\t};\n}\n\nexport function createAssistantMessageDiagnostic(\n\ttype: string,\n\terror: unknown,\n\tdetails?: Record<string, unknown>,\n): AssistantMessageDiagnostic {\n\treturn { type, timestamp: Date.now(), error: extractDiagnosticError(error), details };\n}\n\nexport function appendAssistantMessageDiagnostic<T extends { diagnostics?: AssistantMessageDiagnostic[] }>(\n\tmessage: T,\n\tdiagnostic: AssistantMessageDiagnostic,\n): void {\n\tmessage.diagnostics = [...(message.diagnostics ?? []), diagnostic];\n}\n"]}
1
+ {"version":3,"file":"diagnostics.d.ts","sourceRoot":"","sources":["../../src/utils/diagnostics.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AAE9C,MAAM,WAAW,mBAAmB;IACnC,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,OAAO,EAAE,MAAM,CAAC;IAChB,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,IAAI,CAAC,EAAE,MAAM,GAAG,MAAM,CAAC;CACvB;AAED,MAAM,WAAW,0BAA0B;IAC1C,IAAI,EAAE,MAAM,CAAC;IACb,SAAS,EAAE,MAAM,CAAC;IAClB,KAAK,CAAC,EAAE,mBAAmB,CAAC;IAC5B,OAAO,CAAC,EAAE,UAAU,CAAC;CACrB;AAED,wBAAgB,iBAAiB,CAAC,KAAK,EAAE,OAAO,GAAG,MAAM,CAIxD;AAED,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,OAAO,GAAG,mBAAmB,CAS1E;AAED,wBAAgB,gCAAgC,CAC/C,IAAI,EAAE,MAAM,EACZ,KAAK,EAAE,OAAO,EACd,OAAO,CAAC,EAAE,UAAU,GAClB,0BAA0B,CAE5B;AAED,wBAAgB,gCAAgC,CAAC,CAAC,SAAS;IAAE,WAAW,CAAC,EAAE,0BAA0B,EAAE,CAAA;CAAE,EACxG,OAAO,EAAE,CAAC,EACV,UAAU,EAAE,0BAA0B,GACpC,IAAI,CAEN","sourcesContent":["import type { JsonObject } from \"../types.ts\";\n\nexport interface DiagnosticErrorInfo {\n\tname?: string;\n\tmessage: string;\n\tstack?: string;\n\tcode?: string | number;\n}\n\nexport interface AssistantMessageDiagnostic {\n\ttype: string;\n\ttimestamp: number;\n\terror?: DiagnosticErrorInfo;\n\tdetails?: JsonObject;\n}\n\nexport function formatThrownValue(value: unknown): string {\n\tif (value instanceof Error) return value.message || value.name;\n\tif (typeof value === \"string\") return value;\n\treturn String(value);\n}\n\nexport function extractDiagnosticError(error: unknown): DiagnosticErrorInfo {\n\tif (!(error instanceof Error)) return { name: \"ThrownValue\", message: formatThrownValue(error) };\n\tconst code = (error as Error & { code?: unknown }).code;\n\treturn {\n\t\tname: error.name || undefined,\n\t\tmessage: error.message || error.name,\n\t\tstack: error.stack,\n\t\tcode: typeof code === \"string\" || typeof code === \"number\" ? code : undefined,\n\t};\n}\n\nexport function createAssistantMessageDiagnostic(\n\ttype: string,\n\terror: unknown,\n\tdetails?: JsonObject,\n): AssistantMessageDiagnostic {\n\treturn { type, timestamp: Date.now(), error: extractDiagnosticError(error), details };\n}\n\nexport function appendAssistantMessageDiagnostic<T extends { diagnostics?: AssistantMessageDiagnostic[] }>(\n\tmessage: T,\n\tdiagnostic: AssistantMessageDiagnostic,\n): void {\n\tmessage.diagnostics = [...(message.diagnostics ?? []), diagnostic];\n}\n"]}
@@ -1 +1 @@
1
- {"version":3,"file":"diagnostics.js","sourceRoot":"","sources":["../../src/utils/diagnostics.ts"],"names":[],"mappings":"AAcA,MAAM,UAAU,iBAAiB,CAAC,KAAc,EAAU;IACzD,IAAI,KAAK,YAAY,KAAK;QAAE,OAAO,KAAK,CAAC,OAAO,IAAI,KAAK,CAAC,IAAI,CAAC;IAC/D,IAAI,OAAO,KAAK,KAAK,QAAQ;QAAE,OAAO,KAAK,CAAC;IAC5C,OAAO,MAAM,CAAC,KAAK,CAAC,CAAC;AAAA,CACrB;AAED,MAAM,UAAU,sBAAsB,CAAC,KAAc,EAAuB;IAC3E,IAAI,CAAC,CAAC,KAAK,YAAY,KAAK,CAAC;QAAE,OAAO,EAAE,IAAI,EAAE,aAAa,EAAE,OAAO,EAAE,iBAAiB,CAAC,KAAK,CAAC,EAAE,CAAC;IACjG,MAAM,IAAI,GAAI,KAAoC,CAAC,IAAI,CAAC;IACxD,OAAO;QACN,IAAI,EAAE,KAAK,CAAC,IAAI,IAAI,SAAS;QAC7B,OAAO,EAAE,KAAK,CAAC,OAAO,IAAI,KAAK,CAAC,IAAI;QACpC,KAAK,EAAE,KAAK,CAAC,KAAK;QAClB,IAAI,EAAE,OAAO,IAAI,KAAK,QAAQ,IAAI,OAAO,IAAI,KAAK,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,SAAS;KAC7E,CAAC;AAAA,CACF;AAED,MAAM,UAAU,gCAAgC,CAC/C,IAAY,EACZ,KAAc,EACd,OAAiC,EACJ;IAC7B,OAAO,EAAE,IAAI,EAAE,SAAS,EAAE,IAAI,CAAC,GAAG,EAAE,EAAE,KAAK,EAAE,sBAAsB,CAAC,KAAK,CAAC,EAAE,OAAO,EAAE,CAAC;AAAA,CACtF;AAED,MAAM,UAAU,gCAAgC,CAC/C,OAAU,EACV,UAAsC,EAC/B;IACP,OAAO,CAAC,WAAW,GAAG,CAAC,GAAG,CAAC,OAAO,CAAC,WAAW,IAAI,EAAE,CAAC,EAAE,UAAU,CAAC,CAAC;AAAA,CACnE","sourcesContent":["export interface DiagnosticErrorInfo {\n\tname?: string;\n\tmessage: string;\n\tstack?: string;\n\tcode?: string | number;\n}\n\nexport interface AssistantMessageDiagnostic {\n\ttype: string;\n\ttimestamp: number;\n\terror?: DiagnosticErrorInfo;\n\tdetails?: Record<string, unknown>;\n}\n\nexport function formatThrownValue(value: unknown): string {\n\tif (value instanceof Error) return value.message || value.name;\n\tif (typeof value === \"string\") return value;\n\treturn String(value);\n}\n\nexport function extractDiagnosticError(error: unknown): DiagnosticErrorInfo {\n\tif (!(error instanceof Error)) return { name: \"ThrownValue\", message: formatThrownValue(error) };\n\tconst code = (error as Error & { code?: unknown }).code;\n\treturn {\n\t\tname: error.name || undefined,\n\t\tmessage: error.message || error.name,\n\t\tstack: error.stack,\n\t\tcode: typeof code === \"string\" || typeof code === \"number\" ? code : undefined,\n\t};\n}\n\nexport function createAssistantMessageDiagnostic(\n\ttype: string,\n\terror: unknown,\n\tdetails?: Record<string, unknown>,\n): AssistantMessageDiagnostic {\n\treturn { type, timestamp: Date.now(), error: extractDiagnosticError(error), details };\n}\n\nexport function appendAssistantMessageDiagnostic<T extends { diagnostics?: AssistantMessageDiagnostic[] }>(\n\tmessage: T,\n\tdiagnostic: AssistantMessageDiagnostic,\n): void {\n\tmessage.diagnostics = [...(message.diagnostics ?? []), diagnostic];\n}\n"]}
1
+ {"version":3,"file":"diagnostics.js","sourceRoot":"","sources":["../../src/utils/diagnostics.ts"],"names":[],"mappings":"AAgBA,MAAM,UAAU,iBAAiB,CAAC,KAAc,EAAU;IACzD,IAAI,KAAK,YAAY,KAAK;QAAE,OAAO,KAAK,CAAC,OAAO,IAAI,KAAK,CAAC,IAAI,CAAC;IAC/D,IAAI,OAAO,KAAK,KAAK,QAAQ;QAAE,OAAO,KAAK,CAAC;IAC5C,OAAO,MAAM,CAAC,KAAK,CAAC,CAAC;AAAA,CACrB;AAED,MAAM,UAAU,sBAAsB,CAAC,KAAc,EAAuB;IAC3E,IAAI,CAAC,CAAC,KAAK,YAAY,KAAK,CAAC;QAAE,OAAO,EAAE,IAAI,EAAE,aAAa,EAAE,OAAO,EAAE,iBAAiB,CAAC,KAAK,CAAC,EAAE,CAAC;IACjG,MAAM,IAAI,GAAI,KAAoC,CAAC,IAAI,CAAC;IACxD,OAAO;QACN,IAAI,EAAE,KAAK,CAAC,IAAI,IAAI,SAAS;QAC7B,OAAO,EAAE,KAAK,CAAC,OAAO,IAAI,KAAK,CAAC,IAAI;QACpC,KAAK,EAAE,KAAK,CAAC,KAAK;QAClB,IAAI,EAAE,OAAO,IAAI,KAAK,QAAQ,IAAI,OAAO,IAAI,KAAK,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,SAAS;KAC7E,CAAC;AAAA,CACF;AAED,MAAM,UAAU,gCAAgC,CAC/C,IAAY,EACZ,KAAc,EACd,OAAoB,EACS;IAC7B,OAAO,EAAE,IAAI,EAAE,SAAS,EAAE,IAAI,CAAC,GAAG,EAAE,EAAE,KAAK,EAAE,sBAAsB,CAAC,KAAK,CAAC,EAAE,OAAO,EAAE,CAAC;AAAA,CACtF;AAED,MAAM,UAAU,gCAAgC,CAC/C,OAAU,EACV,UAAsC,EAC/B;IACP,OAAO,CAAC,WAAW,GAAG,CAAC,GAAG,CAAC,OAAO,CAAC,WAAW,IAAI,EAAE,CAAC,EAAE,UAAU,CAAC,CAAC;AAAA,CACnE","sourcesContent":["import type { JsonObject } from \"../types.ts\";\n\nexport interface DiagnosticErrorInfo {\n\tname?: string;\n\tmessage: string;\n\tstack?: string;\n\tcode?: string | number;\n}\n\nexport interface AssistantMessageDiagnostic {\n\ttype: string;\n\ttimestamp: number;\n\terror?: DiagnosticErrorInfo;\n\tdetails?: JsonObject;\n}\n\nexport function formatThrownValue(value: unknown): string {\n\tif (value instanceof Error) return value.message || value.name;\n\tif (typeof value === \"string\") return value;\n\treturn String(value);\n}\n\nexport function extractDiagnosticError(error: unknown): DiagnosticErrorInfo {\n\tif (!(error instanceof Error)) return { name: \"ThrownValue\", message: formatThrownValue(error) };\n\tconst code = (error as Error & { code?: unknown }).code;\n\treturn {\n\t\tname: error.name || undefined,\n\t\tmessage: error.message || error.name,\n\t\tstack: error.stack,\n\t\tcode: typeof code === \"string\" || typeof code === \"number\" ? code : undefined,\n\t};\n}\n\nexport function createAssistantMessageDiagnostic(\n\ttype: string,\n\terror: unknown,\n\tdetails?: JsonObject,\n): AssistantMessageDiagnostic {\n\treturn { type, timestamp: Date.now(), error: extractDiagnosticError(error), details };\n}\n\nexport function appendAssistantMessageDiagnostic<T extends { diagnostics?: AssistantMessageDiagnostic[] }>(\n\tmessage: T,\n\tdiagnostic: AssistantMessageDiagnostic,\n): void {\n\tmessage.diagnostics = [...(message.diagnostics ?? []), diagnostic];\n}\n"]}
@@ -1,4 +1,4 @@
1
- import type { Context, DocumentContent, ImageContent, Message, TextContent, Usage } from "../types.ts";
1
+ import type { DocumentContent, ImageContent, Message, TextContent, TranscriptContext, Usage } from "../types.ts";
2
2
  export interface ContextUsageEstimate {
3
3
  /** Estimated total context tokens. */
4
4
  tokens: number;
@@ -13,5 +13,5 @@ export declare function calculateContextTokens(usage: Usage): number;
13
13
  export declare function estimateTextTokens(text: string): number;
14
14
  export declare function estimateTextAndImageContentTokens(content: string | Array<TextContent | ImageContent | DocumentContent>): number;
15
15
  export declare function estimateMessageTokens(message: Message): number;
16
- export declare function estimateContextTokens(context: Context | readonly Message[]): ContextUsageEstimate;
16
+ export declare function estimateContextTokens(context: TranscriptContext | readonly Message[]): ContextUsageEstimate;
17
17
  //# sourceMappingURL=estimate.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"estimate.d.ts","sourceRoot":"","sources":["../../src/utils/estimate.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAEX,OAAO,EACP,eAAe,EACf,YAAY,EACZ,OAAO,EACP,WAAW,EAEX,KAAK,EACL,MAAM,aAAa,CAAC;AAErB,MAAM,WAAW,oBAAoB;IACpC,sCAAsC;IACtC,MAAM,EAAE,MAAM,CAAC;IACf,2EAA2E;IAC3E,WAAW,EAAE,MAAM,CAAC;IACpB,+EAA+E;IAC/E,cAAc,EAAE,MAAM,CAAC;IACvB,qFAAqF;IACrF,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;CAC9B;AAKD,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,CAE3D;AA8BD,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEvD;AAED,wBAAgB,iCAAiC,CAChD,OAAO,EAAE,MAAM,GAAG,KAAK,CAAC,WAAW,GAAG,YAAY,GAAG,eAAe,CAAC,GACnE,MAAM,CAER;AAED,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,OAAO,GAAG,MAAM,CAiB9D;AAqDD,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,OAAO,GAAG,SAAS,OAAO,EAAE,GAAG,oBAAoB,CA6BjG","sourcesContent":["import type {\n\tAssistantMessage,\n\tContext,\n\tDocumentContent,\n\tImageContent,\n\tMessage,\n\tTextContent,\n\tTool,\n\tUsage,\n} from \"../types.ts\";\n\nexport interface ContextUsageEstimate {\n\t/** Estimated total context tokens. */\n\ttokens: number;\n\t/** Tokens reported by the most recent applicable assistant usage block. */\n\tusageTokens: number;\n\t/** Estimated tokens after the most recent applicable assistant usage block. */\n\ttrailingTokens: number;\n\t/** Index of the applicable message that provided usage, or null when none exists. */\n\tlastUsageIndex: number | null;\n}\n\nconst CHARS_PER_TOKEN = 4;\nconst ESTIMATED_IMAGE_CHARS = 4800;\n\nexport function calculateContextTokens(usage: Usage): number {\n\treturn usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;\n}\n\nfunction safeJsonStringify(value: unknown): string {\n\ttry {\n\t\treturn JSON.stringify(value) ?? \"undefined\";\n\t} catch {\n\t\treturn \"[unserializable]\";\n\t}\n}\n\nfunction estimateTextAndImageContentChars(\n\tcontent: string | Array<TextContent | ImageContent | DocumentContent>,\n): number {\n\tif (typeof content === \"string\") return content.length;\n\n\tlet chars = 0;\n\tfor (const block of content) {\n\t\tif (block.type === \"text\") {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"document\") {\n\t\t\t// A base64 PDF is far larger than an image and must not be counted as one, or context\n\t\t\t// accounting under-reports badly. Estimate from the encoded payload actually sent.\n\t\t\tchars += block.data.length;\n\t\t} else {\n\t\t\tchars += ESTIMATED_IMAGE_CHARS;\n\t\t}\n\t}\n\treturn chars;\n}\n\nexport function estimateTextTokens(text: string): number {\n\treturn Math.ceil(text.length / CHARS_PER_TOKEN);\n}\n\nexport function estimateTextAndImageContentTokens(\n\tcontent: string | Array<TextContent | ImageContent | DocumentContent>,\n): number {\n\treturn Math.ceil(estimateTextAndImageContentChars(content) / CHARS_PER_TOKEN);\n}\n\nexport function estimateMessageTokens(message: Message): number {\n\tlet chars = 0;\n\n\tif (message.role === \"user\") return estimateTextAndImageContentTokens(message.content);\n\tif (message.role === \"toolResult\") return estimateTextAndImageContentTokens(message.content);\n\n\tfor (const block of message.content) {\n\t\tif (block.type === \"text\") {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"thinking\") {\n\t\t\tchars += block.thinking.length;\n\t\t} else if (block.type === \"toolCall\") {\n\t\t\tchars += block.name.length + safeJsonStringify(block.arguments).length;\n\t\t}\n\t\t// Fallback boundary markers carry no text and contribute nothing to the estimate.\n\t}\n\treturn Math.ceil(chars / CHARS_PER_TOKEN);\n}\n\nfunction getLastAssistantUsageInfo(messages: readonly Message[]): { usage: Usage; index: number } | undefined {\n\tlet latestPrefixTimestamp = Number.NEGATIVE_INFINITY;\n\tlet usageInfo: { usage: Usage; index: number } | undefined;\n\n\tfor (let i = 0; i < messages.length; i++) {\n\t\tconst message = messages[i];\n\t\tif (message.role === \"assistant\") {\n\t\t\tconst assistant = message as AssistantMessage;\n\t\t\t// A newer prefix message was inserted after this response (for example, a\n\t\t\t// compaction summary), so its usage cannot describe the current prefix.\n\t\t\tconst usageAppliesToPrefix = assistant.timestamp >= latestPrefixTimestamp;\n\t\t\tif (\n\t\t\t\tusageAppliesToPrefix &&\n\t\t\t\tassistant.stopReason !== \"aborted\" &&\n\t\t\t\tassistant.stopReason !== \"error\" &&\n\t\t\t\tcalculateContextTokens(assistant.usage) > 0\n\t\t\t) {\n\t\t\t\tusageInfo = { usage: assistant.usage, index: i };\n\t\t\t}\n\t\t}\n\t\tlatestPrefixTimestamp = Math.max(latestPrefixTimestamp, message.timestamp);\n\t}\n\n\treturn usageInfo;\n}\n\nfunction estimateMessages(messages: readonly Message[]): ContextUsageEstimate {\n\tconst usageInfo = getLastAssistantUsageInfo(messages);\n\tif (usageInfo) {\n\t\tconst usageTokens = calculateContextTokens(usageInfo.usage);\n\t\tlet trailingTokens = 0;\n\t\tfor (let i = usageInfo.index + 1; i < messages.length; i++) {\n\t\t\ttrailingTokens += estimateMessageTokens(messages[i]);\n\t\t}\n\t\treturn { tokens: usageTokens + trailingTokens, usageTokens, trailingTokens, lastUsageIndex: usageInfo.index };\n\t}\n\n\tlet tokens = 0;\n\tfor (const message of messages) tokens += estimateMessageTokens(message);\n\treturn { tokens, usageTokens: 0, trailingTokens: tokens, lastUsageIndex: null };\n}\n\nfunction estimateToolsTokens(tools: readonly Tool[] | undefined): number {\n\tif (!tools || tools.length === 0) return 0;\n\treturn estimateTextTokens(safeJsonStringify(tools));\n}\n\nfunction isMessageArray(value: Context | readonly Message[]): value is readonly Message[] {\n\treturn Array.isArray(value);\n}\n\nexport function estimateContextTokens(context: Context | readonly Message[]): ContextUsageEstimate {\n\tif (isMessageArray(context)) return estimateMessages(context);\n\n\tconst estimate = estimateMessages(context.messages);\n\tif (estimate.lastUsageIndex !== null) {\n\t\tconst addedNames = new Set(\n\t\t\tcontext.messages\n\t\t\t\t.slice(estimate.lastUsageIndex + 1)\n\t\t\t\t.filter((message) => message.role === \"toolResult\")\n\t\t\t\t.flatMap((message) => message.addedToolNames ?? []),\n\t\t);\n\t\tconst addedToolTokens = estimateToolsTokens(context.tools?.filter((tool) => addedNames.has(tool.name)));\n\t\treturn {\n\t\t\ttokens: estimate.tokens + addedToolTokens,\n\t\t\tusageTokens: estimate.usageTokens,\n\t\t\ttrailingTokens: estimate.trailingTokens + addedToolTokens,\n\t\t\tlastUsageIndex: estimate.lastUsageIndex,\n\t\t};\n\t}\n\n\tconst prefixTokens =\n\t\t(context.systemPrompt ? estimateTextTokens(context.systemPrompt) : 0) + estimateToolsTokens(context.tools);\n\n\treturn {\n\t\ttokens: estimate.tokens + prefixTokens,\n\t\tusageTokens: estimate.usageTokens,\n\t\ttrailingTokens: estimate.trailingTokens + prefixTokens,\n\t\tlastUsageIndex: estimate.lastUsageIndex,\n\t};\n}\n"]}
1
+ {"version":3,"file":"estimate.d.ts","sourceRoot":"","sources":["../../src/utils/estimate.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAEX,eAAe,EACf,YAAY,EACZ,OAAO,EACP,WAAW,EACX,iBAAiB,EACjB,KAAK,EACL,MAAM,aAAa,CAAC;AAGrB,MAAM,WAAW,oBAAoB;IACpC,sCAAsC;IACtC,MAAM,EAAE,MAAM,CAAC;IACf,2EAA2E;IAC3E,WAAW,EAAE,MAAM,CAAC;IACpB,+EAA+E;IAC/E,cAAc,EAAE,MAAM,CAAC;IACvB,qFAAqF;IACrF,cAAc,EAAE,MAAM,GAAG,IAAI,CAAC;CAC9B;AAKD,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,KAAK,GAAG,MAAM,CAE3D;AA8BD,wBAAgB,kBAAkB,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEvD;AAED,wBAAgB,iCAAiC,CAChD,OAAO,EAAE,MAAM,GAAG,KAAK,CAAC,WAAW,GAAG,YAAY,GAAG,eAAe,CAAC,GACnE,MAAM,CAER;AAED,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,OAAO,GAAG,MAAM,CAwB9D;AA4BD,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,iBAAiB,GAAG,SAAS,OAAO,EAAE,GAAG,oBAAoB,CAe3G","sourcesContent":["import type {\n\tAssistantMessage,\n\tDocumentContent,\n\tImageContent,\n\tMessage,\n\tTextContent,\n\tTranscriptContext,\n\tUsage,\n} from \"../types.ts\";\nimport { getSystemMessageText } from \"./text.ts\";\n\nexport interface ContextUsageEstimate {\n\t/** Estimated total context tokens. */\n\ttokens: number;\n\t/** Tokens reported by the most recent applicable assistant usage block. */\n\tusageTokens: number;\n\t/** Estimated tokens after the most recent applicable assistant usage block. */\n\ttrailingTokens: number;\n\t/** Index of the applicable message that provided usage, or null when none exists. */\n\tlastUsageIndex: number | null;\n}\n\nconst CHARS_PER_TOKEN = 4;\nconst ESTIMATED_IMAGE_CHARS = 4800;\n\nexport function calculateContextTokens(usage: Usage): number {\n\treturn usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;\n}\n\nfunction safeJsonStringify(value: unknown): string {\n\ttry {\n\t\treturn JSON.stringify(value) ?? \"undefined\";\n\t} catch {\n\t\treturn \"[unserializable]\";\n\t}\n}\n\nfunction estimateTextAndImageContentChars(\n\tcontent: string | Array<TextContent | ImageContent | DocumentContent>,\n): number {\n\tif (typeof content === \"string\") return content.length;\n\n\tlet chars = 0;\n\tfor (const block of content) {\n\t\tif (block.type === \"text\") {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"document\") {\n\t\t\t// A base64 PDF is far larger than an image and must not be counted as one, or context\n\t\t\t// accounting under-reports badly. Estimate from the encoded payload actually sent.\n\t\t\tchars += block.data.length;\n\t\t} else {\n\t\t\tchars += ESTIMATED_IMAGE_CHARS;\n\t\t}\n\t}\n\treturn chars;\n}\n\nexport function estimateTextTokens(text: string): number {\n\treturn Math.ceil(text.length / CHARS_PER_TOKEN);\n}\n\nexport function estimateTextAndImageContentTokens(\n\tcontent: string | Array<TextContent | ImageContent | DocumentContent>,\n): number {\n\treturn Math.ceil(estimateTextAndImageContentChars(content) / CHARS_PER_TOKEN);\n}\n\nexport function estimateMessageTokens(message: Message): number {\n\tlet chars = 0;\n\n\tif (message.role === \"system\") {\n\t\treturn (\n\t\t\testimateTextTokens(getSystemMessageText(message)) +\n\t\t\testimateToolsTokens(message.toolsAdded) +\n\t\t\testimateToolsTokens(message.toolsRemoved)\n\t\t);\n\t}\n\tif (message.role === \"user\") return estimateTextAndImageContentTokens(message.content);\n\tif (message.role === \"toolResult\") return estimateTextAndImageContentTokens(message.content);\n\n\tfor (const block of message.content) {\n\t\tif (block.type === \"text\") {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"thinking\") {\n\t\t\tchars += block.thinking.length;\n\t\t} else if (block.type === \"toolCall\") {\n\t\t\tchars += block.name.length + safeJsonStringify(block.arguments).length;\n\t\t}\n\t\t// Fallback boundary markers carry no text and contribute nothing to the estimate.\n\t}\n\treturn Math.ceil(chars / CHARS_PER_TOKEN);\n}\n\nfunction getLastAssistantUsageInfo(messages: readonly Message[]): { usage: Usage; index: number } | undefined {\n\tlet latestPrefixTimestamp = Number.NEGATIVE_INFINITY;\n\tlet usageInfo: { usage: Usage; index: number } | undefined;\n\n\tfor (let i = 0; i < messages.length; i++) {\n\t\tconst message = messages[i];\n\t\tif (message.role === \"assistant\") {\n\t\t\tconst assistant = message as AssistantMessage;\n\t\t\t// A newer prefix message was inserted after this response (for example, a\n\t\t\t// compaction summary), so its usage cannot describe the current prefix.\n\t\t\tconst usageAppliesToPrefix = assistant.timestamp >= latestPrefixTimestamp;\n\t\t\tif (\n\t\t\t\tusageAppliesToPrefix &&\n\t\t\t\tassistant.stopReason !== \"aborted\" &&\n\t\t\t\tassistant.stopReason !== \"error\" &&\n\t\t\t\tcalculateContextTokens(assistant.usage) > 0\n\t\t\t) {\n\t\t\t\tusageInfo = { usage: assistant.usage, index: i };\n\t\t\t}\n\t\t}\n\t\tlatestPrefixTimestamp = Math.max(latestPrefixTimestamp, message.timestamp);\n\t}\n\n\treturn usageInfo;\n}\n\nexport function estimateContextTokens(context: TranscriptContext | readonly Message[]): ContextUsageEstimate {\n\tconst messages = \"messages\" in context ? context.messages : context;\n\tconst usageInfo = getLastAssistantUsageInfo(messages);\n\tif (usageInfo) {\n\t\tconst usageTokens = calculateContextTokens(usageInfo.usage);\n\t\tlet trailingTokens = 0;\n\t\tfor (let i = usageInfo.index + 1; i < messages.length; i++) {\n\t\t\ttrailingTokens += estimateMessageTokens(messages[i]);\n\t\t}\n\t\treturn { tokens: usageTokens + trailingTokens, usageTokens, trailingTokens, lastUsageIndex: usageInfo.index };\n\t}\n\n\tlet tokens = 0;\n\tfor (const message of messages) tokens += estimateMessageTokens(message);\n\treturn { tokens, usageTokens: 0, trailingTokens: tokens, lastUsageIndex: null };\n}\n\nfunction estimateToolsTokens(tools: readonly unknown[] | undefined): number {\n\tif (!tools || tools.length === 0) return 0;\n\treturn estimateTextTokens(safeJsonStringify(tools));\n}\n"]}
@@ -1,3 +1,4 @@
1
+ import { getSystemMessageText } from "./text.js";
1
2
  const CHARS_PER_TOKEN = 4;
2
3
  const ESTIMATED_IMAGE_CHARS = 4800;
3
4
  export function calculateContextTokens(usage) {
@@ -38,6 +39,11 @@ export function estimateTextAndImageContentTokens(content) {
38
39
  }
39
40
  export function estimateMessageTokens(message) {
40
41
  let chars = 0;
42
+ if (message.role === "system") {
43
+ return (estimateTextTokens(getSystemMessageText(message)) +
44
+ estimateToolsTokens(message.toolsAdded) +
45
+ estimateToolsTokens(message.toolsRemoved));
46
+ }
41
47
  if (message.role === "user")
42
48
  return estimateTextAndImageContentTokens(message.content);
43
49
  if (message.role === "toolResult")
@@ -77,7 +83,8 @@ function getLastAssistantUsageInfo(messages) {
77
83
  }
78
84
  return usageInfo;
79
85
  }
80
- function estimateMessages(messages) {
86
+ export function estimateContextTokens(context) {
87
+ const messages = "messages" in context ? context.messages : context;
81
88
  const usageInfo = getLastAssistantUsageInfo(messages);
82
89
  if (usageInfo) {
83
90
  const usageTokens = calculateContextTokens(usageInfo.usage);
@@ -97,32 +104,4 @@ function estimateToolsTokens(tools) {
97
104
  return 0;
98
105
  return estimateTextTokens(safeJsonStringify(tools));
99
106
  }
100
- function isMessageArray(value) {
101
- return Array.isArray(value);
102
- }
103
- export function estimateContextTokens(context) {
104
- if (isMessageArray(context))
105
- return estimateMessages(context);
106
- const estimate = estimateMessages(context.messages);
107
- if (estimate.lastUsageIndex !== null) {
108
- const addedNames = new Set(context.messages
109
- .slice(estimate.lastUsageIndex + 1)
110
- .filter((message) => message.role === "toolResult")
111
- .flatMap((message) => message.addedToolNames ?? []));
112
- const addedToolTokens = estimateToolsTokens(context.tools?.filter((tool) => addedNames.has(tool.name)));
113
- return {
114
- tokens: estimate.tokens + addedToolTokens,
115
- usageTokens: estimate.usageTokens,
116
- trailingTokens: estimate.trailingTokens + addedToolTokens,
117
- lastUsageIndex: estimate.lastUsageIndex,
118
- };
119
- }
120
- const prefixTokens = (context.systemPrompt ? estimateTextTokens(context.systemPrompt) : 0) + estimateToolsTokens(context.tools);
121
- return {
122
- tokens: estimate.tokens + prefixTokens,
123
- usageTokens: estimate.usageTokens,
124
- trailingTokens: estimate.trailingTokens + prefixTokens,
125
- lastUsageIndex: estimate.lastUsageIndex,
126
- };
127
- }
128
107
  //# sourceMappingURL=estimate.js.map
@@ -1 +1 @@
1
- {"version":3,"file":"estimate.js","sourceRoot":"","sources":["../../src/utils/estimate.ts"],"names":[],"mappings":"AAsBA,MAAM,eAAe,GAAG,CAAC,CAAC;AAC1B,MAAM,qBAAqB,GAAG,IAAI,CAAC;AAEnC,MAAM,UAAU,sBAAsB,CAAC,KAAY,EAAU;IAC5D,OAAO,KAAK,CAAC,WAAW,IAAI,KAAK,CAAC,KAAK,GAAG,KAAK,CAAC,MAAM,GAAG,KAAK,CAAC,SAAS,GAAG,KAAK,CAAC,UAAU,CAAC;AAAA,CAC5F;AAED,SAAS,iBAAiB,CAAC,KAAc,EAAU;IAClD,IAAI,CAAC;QACJ,OAAO,IAAI,CAAC,SAAS,CAAC,KAAK,CAAC,IAAI,WAAW,CAAC;IAC7C,CAAC;IAAC,MAAM,CAAC;QACR,OAAO,kBAAkB,CAAC;IAC3B,CAAC;AAAA,CACD;AAED,SAAS,gCAAgC,CACxC,OAAqE,EAC5D;IACT,IAAI,OAAO,OAAO,KAAK,QAAQ;QAAE,OAAO,OAAO,CAAC,MAAM,CAAC;IAEvD,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,KAAK,IAAI,OAAO,EAAE,CAAC;QAC7B,IAAI,KAAK,CAAC,IAAI,KAAK,MAAM,EAAE,CAAC;YAC3B,KAAK,IAAI,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC;QAC5B,CAAC;aAAM,IAAI,KAAK,CAAC,IAAI,KAAK,UAAU,EAAE,CAAC;YACtC,sFAAsF;YACtF,mFAAmF;YACnF,KAAK,IAAI,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC;QAC5B,CAAC;aAAM,CAAC;YACP,KAAK,IAAI,qBAAqB,CAAC;QAChC,CAAC;IACF,CAAC;IACD,OAAO,KAAK,CAAC;AAAA,CACb;AAED,MAAM,UAAU,kBAAkB,CAAC,IAAY,EAAU;IACxD,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,eAAe,CAAC,CAAC;AAAA,CAChD;AAED,MAAM,UAAU,iCAAiC,CAChD,OAAqE,EAC5D;IACT,OAAO,IAAI,CAAC,IAAI,CAAC,gCAAgC,CAAC,OAAO,CAAC,GAAG,eAAe,CAAC,CAAC;AAAA,CAC9E;AAED,MAAM,UAAU,qBAAqB,CAAC,OAAgB,EAAU;IAC/D,IAAI,KAAK,GAAG,CAAC,CAAC;IAEd,IAAI,OAAO,CAAC,IAAI,KAAK,MAAM;QAAE,OAAO,iCAAiC,CAAC,OAAO,CAAC,OAAO,CAAC,CAAC;IACvF,IAAI,OAAO,CAAC,IAAI,KAAK,YAAY;QAAE,OAAO,iCAAiC,CAAC,OAAO,CAAC,OAAO,CAAC,CAAC;IAE7F,KAAK,MAAM,KAAK,IAAI,OAAO,CAAC,OAAO,EAAE,CAAC;QACrC,IAAI,KAAK,CAAC,IAAI,KAAK,MAAM,EAAE,CAAC;YAC3B,KAAK,IAAI,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC;QAC5B,CAAC;aAAM,IAAI,KAAK,CAAC,IAAI,KAAK,UAAU,EAAE,CAAC;YACtC,KAAK,IAAI,KAAK,CAAC,QAAQ,CAAC,MAAM,CAAC;QAChC,CAAC;aAAM,IAAI,KAAK,CAAC,IAAI,KAAK,UAAU,EAAE,CAAC;YACtC,KAAK,IAAI,KAAK,CAAC,IAAI,CAAC,MAAM,GAAG,iBAAiB,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC,MAAM,CAAC;QACxE,CAAC;QACD,kFAAkF;IACnF,CAAC;IACD,OAAO,IAAI,CAAC,IAAI,CAAC,KAAK,GAAG,eAAe,CAAC,CAAC;AAAA,CAC1C;AAED,SAAS,yBAAyB,CAAC,QAA4B,EAA+C;IAC7G,IAAI,qBAAqB,GAAG,MAAM,CAAC,iBAAiB,CAAC;IACrD,IAAI,SAAsD,CAAC;IAE3D,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QAC1C,MAAM,OAAO,GAAG,QAAQ,CAAC,CAAC,CAAC,CAAC;QAC5B,IAAI,OAAO,CAAC,IAAI,KAAK,WAAW,EAAE,CAAC;YAClC,MAAM,SAAS,GAAG,OAA2B,CAAC;YAC9C,0EAA0E;YAC1E,wEAAwE;YACxE,MAAM,oBAAoB,GAAG,SAAS,CAAC,SAAS,IAAI,qBAAqB,CAAC;YAC1E,IACC,oBAAoB;gBACpB,SAAS,CAAC,UAAU,KAAK,SAAS;gBAClC,SAAS,CAAC,UAAU,KAAK,OAAO;gBAChC,sBAAsB,CAAC,SAAS,CAAC,KAAK,CAAC,GAAG,CAAC,EAC1C,CAAC;gBACF,SAAS,GAAG,EAAE,KAAK,EAAE,SAAS,CAAC,KAAK,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;YAClD,CAAC;QACF,CAAC;QACD,qBAAqB,GAAG,IAAI,CAAC,GAAG,CAAC,qBAAqB,EAAE,OAAO,CAAC,SAAS,CAAC,CAAC;IAC5E,CAAC;IAED,OAAO,SAAS,CAAC;AAAA,CACjB;AAED,SAAS,gBAAgB,CAAC,QAA4B,EAAwB;IAC7E,MAAM,SAAS,GAAG,yBAAyB,CAAC,QAAQ,CAAC,CAAC;IACtD,IAAI,SAAS,EAAE,CAAC;QACf,MAAM,WAAW,GAAG,sBAAsB,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QAC5D,IAAI,cAAc,GAAG,CAAC,CAAC;QACvB,KAAK,IAAI,CAAC,GAAG,SAAS,CAAC,KAAK,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5D,cAAc,IAAI,qBAAqB,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC;QACtD,CAAC;QACD,OAAO,EAAE,MAAM,EAAE,WAAW,GAAG,cAAc,EAAE,WAAW,EAAE,cAAc,EAAE,cAAc,EAAE,SAAS,CAAC,KAAK,EAAE,CAAC;IAC/G,CAAC;IAED,IAAI,MAAM,GAAG,CAAC,CAAC;IACf,KAAK,MAAM,OAAO,IAAI,QAAQ;QAAE,MAAM,IAAI,qBAAqB,CAAC,OAAO,CAAC,CAAC;IACzE,OAAO,EAAE,MAAM,EAAE,WAAW,EAAE,CAAC,EAAE,cAAc,EAAE,MAAM,EAAE,cAAc,EAAE,IAAI,EAAE,CAAC;AAAA,CAChF;AAED,SAAS,mBAAmB,CAAC,KAAkC,EAAU;IACxE,IAAI,CAAC,KAAK,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,CAAC,CAAC;IAC3C,OAAO,kBAAkB,CAAC,iBAAiB,CAAC,KAAK,CAAC,CAAC,CAAC;AAAA,CACpD;AAED,SAAS,cAAc,CAAC,KAAmC,EAA+B;IACzF,OAAO,KAAK,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC;AAAA,CAC5B;AAED,MAAM,UAAU,qBAAqB,CAAC,OAAqC,EAAwB;IAClG,IAAI,cAAc,CAAC,OAAO,CAAC;QAAE,OAAO,gBAAgB,CAAC,OAAO,CAAC,CAAC;IAE9D,MAAM,QAAQ,GAAG,gBAAgB,CAAC,OAAO,CAAC,QAAQ,CAAC,CAAC;IACpD,IAAI,QAAQ,CAAC,cAAc,KAAK,IAAI,EAAE,CAAC;QACtC,MAAM,UAAU,GAAG,IAAI,GAAG,CACzB,OAAO,CAAC,QAAQ;aACd,KAAK,CAAC,QAAQ,CAAC,cAAc,GAAG,CAAC,CAAC;aAClC,MAAM,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,OAAO,CAAC,IAAI,KAAK,YAAY,CAAC;aAClD,OAAO,CAAC,CAAC,OAAO,EAAE,EAAE,CAAC,OAAO,CAAC,cAAc,IAAI,EAAE,CAAC,CACpD,CAAC;QACF,MAAM,eAAe,GAAG,mBAAmB,CAAC,OAAO,CAAC,KAAK,EAAE,MAAM,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,UAAU,CAAC,GAAG,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;QACxG,OAAO;YACN,MAAM,EAAE,QAAQ,CAAC,MAAM,GAAG,eAAe;YACzC,WAAW,EAAE,QAAQ,CAAC,WAAW;YACjC,cAAc,EAAE,QAAQ,CAAC,cAAc,GAAG,eAAe;YACzD,cAAc,EAAE,QAAQ,CAAC,cAAc;SACvC,CAAC;IACH,CAAC;IAED,MAAM,YAAY,GACjB,CAAC,OAAO,CAAC,YAAY,CAAC,CAAC,CAAC,kBAAkB,CAAC,OAAO,CAAC,YAAY,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,GAAG,mBAAmB,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC;IAE5G,OAAO;QACN,MAAM,EAAE,QAAQ,CAAC,MAAM,GAAG,YAAY;QACtC,WAAW,EAAE,QAAQ,CAAC,WAAW;QACjC,cAAc,EAAE,QAAQ,CAAC,cAAc,GAAG,YAAY;QACtD,cAAc,EAAE,QAAQ,CAAC,cAAc;KACvC,CAAC;AAAA,CACF","sourcesContent":["import type {\n\tAssistantMessage,\n\tContext,\n\tDocumentContent,\n\tImageContent,\n\tMessage,\n\tTextContent,\n\tTool,\n\tUsage,\n} from \"../types.ts\";\n\nexport interface ContextUsageEstimate {\n\t/** Estimated total context tokens. */\n\ttokens: number;\n\t/** Tokens reported by the most recent applicable assistant usage block. */\n\tusageTokens: number;\n\t/** Estimated tokens after the most recent applicable assistant usage block. */\n\ttrailingTokens: number;\n\t/** Index of the applicable message that provided usage, or null when none exists. */\n\tlastUsageIndex: number | null;\n}\n\nconst CHARS_PER_TOKEN = 4;\nconst ESTIMATED_IMAGE_CHARS = 4800;\n\nexport function calculateContextTokens(usage: Usage): number {\n\treturn usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;\n}\n\nfunction safeJsonStringify(value: unknown): string {\n\ttry {\n\t\treturn JSON.stringify(value) ?? \"undefined\";\n\t} catch {\n\t\treturn \"[unserializable]\";\n\t}\n}\n\nfunction estimateTextAndImageContentChars(\n\tcontent: string | Array<TextContent | ImageContent | DocumentContent>,\n): number {\n\tif (typeof content === \"string\") return content.length;\n\n\tlet chars = 0;\n\tfor (const block of content) {\n\t\tif (block.type === \"text\") {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"document\") {\n\t\t\t// A base64 PDF is far larger than an image and must not be counted as one, or context\n\t\t\t// accounting under-reports badly. Estimate from the encoded payload actually sent.\n\t\t\tchars += block.data.length;\n\t\t} else {\n\t\t\tchars += ESTIMATED_IMAGE_CHARS;\n\t\t}\n\t}\n\treturn chars;\n}\n\nexport function estimateTextTokens(text: string): number {\n\treturn Math.ceil(text.length / CHARS_PER_TOKEN);\n}\n\nexport function estimateTextAndImageContentTokens(\n\tcontent: string | Array<TextContent | ImageContent | DocumentContent>,\n): number {\n\treturn Math.ceil(estimateTextAndImageContentChars(content) / CHARS_PER_TOKEN);\n}\n\nexport function estimateMessageTokens(message: Message): number {\n\tlet chars = 0;\n\n\tif (message.role === \"user\") return estimateTextAndImageContentTokens(message.content);\n\tif (message.role === \"toolResult\") return estimateTextAndImageContentTokens(message.content);\n\n\tfor (const block of message.content) {\n\t\tif (block.type === \"text\") {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"thinking\") {\n\t\t\tchars += block.thinking.length;\n\t\t} else if (block.type === \"toolCall\") {\n\t\t\tchars += block.name.length + safeJsonStringify(block.arguments).length;\n\t\t}\n\t\t// Fallback boundary markers carry no text and contribute nothing to the estimate.\n\t}\n\treturn Math.ceil(chars / CHARS_PER_TOKEN);\n}\n\nfunction getLastAssistantUsageInfo(messages: readonly Message[]): { usage: Usage; index: number } | undefined {\n\tlet latestPrefixTimestamp = Number.NEGATIVE_INFINITY;\n\tlet usageInfo: { usage: Usage; index: number } | undefined;\n\n\tfor (let i = 0; i < messages.length; i++) {\n\t\tconst message = messages[i];\n\t\tif (message.role === \"assistant\") {\n\t\t\tconst assistant = message as AssistantMessage;\n\t\t\t// A newer prefix message was inserted after this response (for example, a\n\t\t\t// compaction summary), so its usage cannot describe the current prefix.\n\t\t\tconst usageAppliesToPrefix = assistant.timestamp >= latestPrefixTimestamp;\n\t\t\tif (\n\t\t\t\tusageAppliesToPrefix &&\n\t\t\t\tassistant.stopReason !== \"aborted\" &&\n\t\t\t\tassistant.stopReason !== \"error\" &&\n\t\t\t\tcalculateContextTokens(assistant.usage) > 0\n\t\t\t) {\n\t\t\t\tusageInfo = { usage: assistant.usage, index: i };\n\t\t\t}\n\t\t}\n\t\tlatestPrefixTimestamp = Math.max(latestPrefixTimestamp, message.timestamp);\n\t}\n\n\treturn usageInfo;\n}\n\nfunction estimateMessages(messages: readonly Message[]): ContextUsageEstimate {\n\tconst usageInfo = getLastAssistantUsageInfo(messages);\n\tif (usageInfo) {\n\t\tconst usageTokens = calculateContextTokens(usageInfo.usage);\n\t\tlet trailingTokens = 0;\n\t\tfor (let i = usageInfo.index + 1; i < messages.length; i++) {\n\t\t\ttrailingTokens += estimateMessageTokens(messages[i]);\n\t\t}\n\t\treturn { tokens: usageTokens + trailingTokens, usageTokens, trailingTokens, lastUsageIndex: usageInfo.index };\n\t}\n\n\tlet tokens = 0;\n\tfor (const message of messages) tokens += estimateMessageTokens(message);\n\treturn { tokens, usageTokens: 0, trailingTokens: tokens, lastUsageIndex: null };\n}\n\nfunction estimateToolsTokens(tools: readonly Tool[] | undefined): number {\n\tif (!tools || tools.length === 0) return 0;\n\treturn estimateTextTokens(safeJsonStringify(tools));\n}\n\nfunction isMessageArray(value: Context | readonly Message[]): value is readonly Message[] {\n\treturn Array.isArray(value);\n}\n\nexport function estimateContextTokens(context: Context | readonly Message[]): ContextUsageEstimate {\n\tif (isMessageArray(context)) return estimateMessages(context);\n\n\tconst estimate = estimateMessages(context.messages);\n\tif (estimate.lastUsageIndex !== null) {\n\t\tconst addedNames = new Set(\n\t\t\tcontext.messages\n\t\t\t\t.slice(estimate.lastUsageIndex + 1)\n\t\t\t\t.filter((message) => message.role === \"toolResult\")\n\t\t\t\t.flatMap((message) => message.addedToolNames ?? []),\n\t\t);\n\t\tconst addedToolTokens = estimateToolsTokens(context.tools?.filter((tool) => addedNames.has(tool.name)));\n\t\treturn {\n\t\t\ttokens: estimate.tokens + addedToolTokens,\n\t\t\tusageTokens: estimate.usageTokens,\n\t\t\ttrailingTokens: estimate.trailingTokens + addedToolTokens,\n\t\t\tlastUsageIndex: estimate.lastUsageIndex,\n\t\t};\n\t}\n\n\tconst prefixTokens =\n\t\t(context.systemPrompt ? estimateTextTokens(context.systemPrompt) : 0) + estimateToolsTokens(context.tools);\n\n\treturn {\n\t\ttokens: estimate.tokens + prefixTokens,\n\t\tusageTokens: estimate.usageTokens,\n\t\ttrailingTokens: estimate.trailingTokens + prefixTokens,\n\t\tlastUsageIndex: estimate.lastUsageIndex,\n\t};\n}\n"]}
1
+ {"version":3,"file":"estimate.js","sourceRoot":"","sources":["../../src/utils/estimate.ts"],"names":[],"mappings":"AASA,OAAO,EAAE,oBAAoB,EAAE,MAAM,WAAW,CAAC;AAajD,MAAM,eAAe,GAAG,CAAC,CAAC;AAC1B,MAAM,qBAAqB,GAAG,IAAI,CAAC;AAEnC,MAAM,UAAU,sBAAsB,CAAC,KAAY,EAAU;IAC5D,OAAO,KAAK,CAAC,WAAW,IAAI,KAAK,CAAC,KAAK,GAAG,KAAK,CAAC,MAAM,GAAG,KAAK,CAAC,SAAS,GAAG,KAAK,CAAC,UAAU,CAAC;AAAA,CAC5F;AAED,SAAS,iBAAiB,CAAC,KAAc,EAAU;IAClD,IAAI,CAAC;QACJ,OAAO,IAAI,CAAC,SAAS,CAAC,KAAK,CAAC,IAAI,WAAW,CAAC;IAC7C,CAAC;IAAC,MAAM,CAAC;QACR,OAAO,kBAAkB,CAAC;IAC3B,CAAC;AAAA,CACD;AAED,SAAS,gCAAgC,CACxC,OAAqE,EAC5D;IACT,IAAI,OAAO,OAAO,KAAK,QAAQ;QAAE,OAAO,OAAO,CAAC,MAAM,CAAC;IAEvD,IAAI,KAAK,GAAG,CAAC,CAAC;IACd,KAAK,MAAM,KAAK,IAAI,OAAO,EAAE,CAAC;QAC7B,IAAI,KAAK,CAAC,IAAI,KAAK,MAAM,EAAE,CAAC;YAC3B,KAAK,IAAI,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC;QAC5B,CAAC;aAAM,IAAI,KAAK,CAAC,IAAI,KAAK,UAAU,EAAE,CAAC;YACtC,sFAAsF;YACtF,mFAAmF;YACnF,KAAK,IAAI,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC;QAC5B,CAAC;aAAM,CAAC;YACP,KAAK,IAAI,qBAAqB,CAAC;QAChC,CAAC;IACF,CAAC;IACD,OAAO,KAAK,CAAC;AAAA,CACb;AAED,MAAM,UAAU,kBAAkB,CAAC,IAAY,EAAU;IACxD,OAAO,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,eAAe,CAAC,CAAC;AAAA,CAChD;AAED,MAAM,UAAU,iCAAiC,CAChD,OAAqE,EAC5D;IACT,OAAO,IAAI,CAAC,IAAI,CAAC,gCAAgC,CAAC,OAAO,CAAC,GAAG,eAAe,CAAC,CAAC;AAAA,CAC9E;AAED,MAAM,UAAU,qBAAqB,CAAC,OAAgB,EAAU;IAC/D,IAAI,KAAK,GAAG,CAAC,CAAC;IAEd,IAAI,OAAO,CAAC,IAAI,KAAK,QAAQ,EAAE,CAAC;QAC/B,OAAO,CACN,kBAAkB,CAAC,oBAAoB,CAAC,OAAO,CAAC,CAAC;YACjD,mBAAmB,CAAC,OAAO,CAAC,UAAU,CAAC;YACvC,mBAAmB,CAAC,OAAO,CAAC,YAAY,CAAC,CACzC,CAAC;IACH,CAAC;IACD,IAAI,OAAO,CAAC,IAAI,KAAK,MAAM;QAAE,OAAO,iCAAiC,CAAC,OAAO,CAAC,OAAO,CAAC,CAAC;IACvF,IAAI,OAAO,CAAC,IAAI,KAAK,YAAY;QAAE,OAAO,iCAAiC,CAAC,OAAO,CAAC,OAAO,CAAC,CAAC;IAE7F,KAAK,MAAM,KAAK,IAAI,OAAO,CAAC,OAAO,EAAE,CAAC;QACrC,IAAI,KAAK,CAAC,IAAI,KAAK,MAAM,EAAE,CAAC;YAC3B,KAAK,IAAI,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC;QAC5B,CAAC;aAAM,IAAI,KAAK,CAAC,IAAI,KAAK,UAAU,EAAE,CAAC;YACtC,KAAK,IAAI,KAAK,CAAC,QAAQ,CAAC,MAAM,CAAC;QAChC,CAAC;aAAM,IAAI,KAAK,CAAC,IAAI,KAAK,UAAU,EAAE,CAAC;YACtC,KAAK,IAAI,KAAK,CAAC,IAAI,CAAC,MAAM,GAAG,iBAAiB,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC,MAAM,CAAC;QACxE,CAAC;QACD,kFAAkF;IACnF,CAAC;IACD,OAAO,IAAI,CAAC,IAAI,CAAC,KAAK,GAAG,eAAe,CAAC,CAAC;AAAA,CAC1C;AAED,SAAS,yBAAyB,CAAC,QAA4B,EAA+C;IAC7G,IAAI,qBAAqB,GAAG,MAAM,CAAC,iBAAiB,CAAC;IACrD,IAAI,SAAsD,CAAC;IAE3D,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QAC1C,MAAM,OAAO,GAAG,QAAQ,CAAC,CAAC,CAAC,CAAC;QAC5B,IAAI,OAAO,CAAC,IAAI,KAAK,WAAW,EAAE,CAAC;YAClC,MAAM,SAAS,GAAG,OAA2B,CAAC;YAC9C,0EAA0E;YAC1E,wEAAwE;YACxE,MAAM,oBAAoB,GAAG,SAAS,CAAC,SAAS,IAAI,qBAAqB,CAAC;YAC1E,IACC,oBAAoB;gBACpB,SAAS,CAAC,UAAU,KAAK,SAAS;gBAClC,SAAS,CAAC,UAAU,KAAK,OAAO;gBAChC,sBAAsB,CAAC,SAAS,CAAC,KAAK,CAAC,GAAG,CAAC,EAC1C,CAAC;gBACF,SAAS,GAAG,EAAE,KAAK,EAAE,SAAS,CAAC,KAAK,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;YAClD,CAAC;QACF,CAAC;QACD,qBAAqB,GAAG,IAAI,CAAC,GAAG,CAAC,qBAAqB,EAAE,OAAO,CAAC,SAAS,CAAC,CAAC;IAC5E,CAAC;IAED,OAAO,SAAS,CAAC;AAAA,CACjB;AAED,MAAM,UAAU,qBAAqB,CAAC,OAA+C,EAAwB;IAC5G,MAAM,QAAQ,GAAG,UAAU,IAAI,OAAO,CAAC,CAAC,CAAC,OAAO,CAAC,QAAQ,CAAC,CAAC,CAAC,OAAO,CAAC;IACpE,MAAM,SAAS,GAAG,yBAAyB,CAAC,QAAQ,CAAC,CAAC;IACtD,IAAI,SAAS,EAAE,CAAC;QACf,MAAM,WAAW,GAAG,sBAAsB,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;QAC5D,IAAI,cAAc,GAAG,CAAC,CAAC;QACvB,KAAK,IAAI,CAAC,GAAG,SAAS,CAAC,KAAK,GAAG,CAAC,EAAE,CAAC,GAAG,QAAQ,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;YAC5D,cAAc,IAAI,qBAAqB,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC;QACtD,CAAC;QACD,OAAO,EAAE,MAAM,EAAE,WAAW,GAAG,cAAc,EAAE,WAAW,EAAE,cAAc,EAAE,cAAc,EAAE,SAAS,CAAC,KAAK,EAAE,CAAC;IAC/G,CAAC;IAED,IAAI,MAAM,GAAG,CAAC,CAAC;IACf,KAAK,MAAM,OAAO,IAAI,QAAQ;QAAE,MAAM,IAAI,qBAAqB,CAAC,OAAO,CAAC,CAAC;IACzE,OAAO,EAAE,MAAM,EAAE,WAAW,EAAE,CAAC,EAAE,cAAc,EAAE,MAAM,EAAE,cAAc,EAAE,IAAI,EAAE,CAAC;AAAA,CAChF;AAED,SAAS,mBAAmB,CAAC,KAAqC,EAAU;IAC3E,IAAI,CAAC,KAAK,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;QAAE,OAAO,CAAC,CAAC;IAC3C,OAAO,kBAAkB,CAAC,iBAAiB,CAAC,KAAK,CAAC,CAAC,CAAC;AAAA,CACpD","sourcesContent":["import type {\n\tAssistantMessage,\n\tDocumentContent,\n\tImageContent,\n\tMessage,\n\tTextContent,\n\tTranscriptContext,\n\tUsage,\n} from \"../types.ts\";\nimport { getSystemMessageText } from \"./text.ts\";\n\nexport interface ContextUsageEstimate {\n\t/** Estimated total context tokens. */\n\ttokens: number;\n\t/** Tokens reported by the most recent applicable assistant usage block. */\n\tusageTokens: number;\n\t/** Estimated tokens after the most recent applicable assistant usage block. */\n\ttrailingTokens: number;\n\t/** Index of the applicable message that provided usage, or null when none exists. */\n\tlastUsageIndex: number | null;\n}\n\nconst CHARS_PER_TOKEN = 4;\nconst ESTIMATED_IMAGE_CHARS = 4800;\n\nexport function calculateContextTokens(usage: Usage): number {\n\treturn usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;\n}\n\nfunction safeJsonStringify(value: unknown): string {\n\ttry {\n\t\treturn JSON.stringify(value) ?? \"undefined\";\n\t} catch {\n\t\treturn \"[unserializable]\";\n\t}\n}\n\nfunction estimateTextAndImageContentChars(\n\tcontent: string | Array<TextContent | ImageContent | DocumentContent>,\n): number {\n\tif (typeof content === \"string\") return content.length;\n\n\tlet chars = 0;\n\tfor (const block of content) {\n\t\tif (block.type === \"text\") {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"document\") {\n\t\t\t// A base64 PDF is far larger than an image and must not be counted as one, or context\n\t\t\t// accounting under-reports badly. Estimate from the encoded payload actually sent.\n\t\t\tchars += block.data.length;\n\t\t} else {\n\t\t\tchars += ESTIMATED_IMAGE_CHARS;\n\t\t}\n\t}\n\treturn chars;\n}\n\nexport function estimateTextTokens(text: string): number {\n\treturn Math.ceil(text.length / CHARS_PER_TOKEN);\n}\n\nexport function estimateTextAndImageContentTokens(\n\tcontent: string | Array<TextContent | ImageContent | DocumentContent>,\n): number {\n\treturn Math.ceil(estimateTextAndImageContentChars(content) / CHARS_PER_TOKEN);\n}\n\nexport function estimateMessageTokens(message: Message): number {\n\tlet chars = 0;\n\n\tif (message.role === \"system\") {\n\t\treturn (\n\t\t\testimateTextTokens(getSystemMessageText(message)) +\n\t\t\testimateToolsTokens(message.toolsAdded) +\n\t\t\testimateToolsTokens(message.toolsRemoved)\n\t\t);\n\t}\n\tif (message.role === \"user\") return estimateTextAndImageContentTokens(message.content);\n\tif (message.role === \"toolResult\") return estimateTextAndImageContentTokens(message.content);\n\n\tfor (const block of message.content) {\n\t\tif (block.type === \"text\") {\n\t\t\tchars += block.text.length;\n\t\t} else if (block.type === \"thinking\") {\n\t\t\tchars += block.thinking.length;\n\t\t} else if (block.type === \"toolCall\") {\n\t\t\tchars += block.name.length + safeJsonStringify(block.arguments).length;\n\t\t}\n\t\t// Fallback boundary markers carry no text and contribute nothing to the estimate.\n\t}\n\treturn Math.ceil(chars / CHARS_PER_TOKEN);\n}\n\nfunction getLastAssistantUsageInfo(messages: readonly Message[]): { usage: Usage; index: number } | undefined {\n\tlet latestPrefixTimestamp = Number.NEGATIVE_INFINITY;\n\tlet usageInfo: { usage: Usage; index: number } | undefined;\n\n\tfor (let i = 0; i < messages.length; i++) {\n\t\tconst message = messages[i];\n\t\tif (message.role === \"assistant\") {\n\t\t\tconst assistant = message as AssistantMessage;\n\t\t\t// A newer prefix message was inserted after this response (for example, a\n\t\t\t// compaction summary), so its usage cannot describe the current prefix.\n\t\t\tconst usageAppliesToPrefix = assistant.timestamp >= latestPrefixTimestamp;\n\t\t\tif (\n\t\t\t\tusageAppliesToPrefix &&\n\t\t\t\tassistant.stopReason !== \"aborted\" &&\n\t\t\t\tassistant.stopReason !== \"error\" &&\n\t\t\t\tcalculateContextTokens(assistant.usage) > 0\n\t\t\t) {\n\t\t\t\tusageInfo = { usage: assistant.usage, index: i };\n\t\t\t}\n\t\t}\n\t\tlatestPrefixTimestamp = Math.max(latestPrefixTimestamp, message.timestamp);\n\t}\n\n\treturn usageInfo;\n}\n\nexport function estimateContextTokens(context: TranscriptContext | readonly Message[]): ContextUsageEstimate {\n\tconst messages = \"messages\" in context ? context.messages : context;\n\tconst usageInfo = getLastAssistantUsageInfo(messages);\n\tif (usageInfo) {\n\t\tconst usageTokens = calculateContextTokens(usageInfo.usage);\n\t\tlet trailingTokens = 0;\n\t\tfor (let i = usageInfo.index + 1; i < messages.length; i++) {\n\t\t\ttrailingTokens += estimateMessageTokens(messages[i]);\n\t\t}\n\t\treturn { tokens: usageTokens + trailingTokens, usageTokens, trailingTokens, lastUsageIndex: usageInfo.index };\n\t}\n\n\tlet tokens = 0;\n\tfor (const message of messages) tokens += estimateMessageTokens(message);\n\treturn { tokens, usageTokens: 0, trailingTokens: tokens, lastUsageIndex: null };\n}\n\nfunction estimateToolsTokens(tools: readonly unknown[] | undefined): number {\n\tif (!tools || tools.length === 0) return 0;\n\treturn estimateTextTokens(safeJsonStringify(tools));\n}\n"]}
@@ -28,6 +28,7 @@ import type { AssistantMessage } from "../types.ts";
28
28
  * - Kimi For Coding: "exceeded model token limit: X (requested: Y)"
29
29
  * - DS4: "Prompt has X tokens, but the configured context size is Y tokens"
30
30
  * - DashScope/Qwen: "Range of input length should be [1, X]"
31
+ * - z.ai: "Prompt too long"
31
32
  *
32
33
  * **Unreliable detection:**
33
34
  * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
@@ -1 +1 @@
1
- {"version":3,"file":"overflow.d.ts","sourceRoot":"","sources":["../../src/utils/overflow.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,aAAa,CAAC;AA+EpD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAqDG;AACH,wBAAgB,iBAAiB,CAAC,OAAO,EAAE,gBAAgB,EAAE,aAAa,CAAC,EAAE,MAAM,GAAG,OAAO,CA6B5F;AAED;;;;;GAKG;AACH,wBAAgB,mBAAmB,CAAC,OAAO,EAAE,gBAAgB,EAAE,gBAAgB,EAAE,MAAM,GAAG,OAAO,CAEhG;AAED;;GAEG;AACH,wBAAgB,mBAAmB,IAAI,MAAM,EAAE,CAE9C","sourcesContent":["import type { AssistantMessage } from \"../types.ts\";\n\n/**\n * Regex patterns to detect context overflow errors from different providers.\n *\n * These patterns match error messages returned when the input exceeds\n * the model's context window.\n *\n * Provider-specific patterns (with example error messages):\n *\n * - Anthropic: \"prompt is too long: 213462 tokens > 200000 maximum\"\n * - Anthropic: \"413 {\\\"error\\\":{\\\"type\\\":\\\"request_too_large\\\",\\\"message\\\":\\\"Request exceeds the maximum size\\\"}}\"\n * - OpenAI: \"Your input exceeds the context window of this model\"\n * - OpenAI/LiteLLM: \"Requested token count exceeds the model's maximum context length of 131072 tokens\"\n * - OpenAI-compatible: \"Input length (265330) exceeds model's maximum context length (262144).\"\n * - Google: \"The input token count (1196265) exceeds the maximum number of tokens allowed (1048575)\"\n * - xAI: \"This model's maximum prompt length is 131072 but the request contains 537812 tokens\"\n * - Groq: \"Please reduce the length of the messages or completion\"\n * - OpenRouter: \"This endpoint's maximum context length is X tokens. However, you requested about Y tokens\"\n * - OpenRouter/Poolside: \"Input length X exceeds the maximum allowed input length of Y tokens.\"\n * - Together AI: \"The input (X tokens) is longer than the model's context length (Y tokens).\"\n * - llama.cpp: \"the request exceeds the available context size, try increasing it\"\n * - LM Studio: \"tokens to keep from the initial prompt is greater than the context length\"\n * - GitHub Copilot: \"prompt token count of X exceeds the limit of Y\"\n * - MiniMax: \"invalid params, context window exceeds limit\"\n * - Kimi For Coding: \"Your request exceeded model token limit: X (requested: Y)\"\n * - DS4: \"Prompt has X tokens, but the configured context size is Y tokens\"\n * - Cerebras: \"400/413 status code (no body)\"\n * - Mistral: \"Prompt contains X tokens ... too large for model with Y maximum context length\"\n * - z.ai: Does NOT error, accepts overflow silently - handled via usage.input > contextWindow\n * - Xiaomi MiMo: Truncates input to fill contextWindow exactly, then returns finish_reason \"length\"\n * with output=0 (no room left to generate). Detected via stopReason \"length\" + zero output +\n * input filling the context window.\n * - DashScope/Qwen: \"Range of input length should be [1, X]\" (HTTP 400 invalid_parameter_error)\n * - Ollama: Some deployments truncate silently, others return errors like \"prompt too long; exceeded max context length by X tokens\"\n */\nconst OVERFLOW_PATTERNS = [\n\t/prompt is too long/i, // Anthropic token overflow\n\t/request_too_large/i, // Anthropic request byte-size overflow (HTTP 413)\n\t/input is too long for requested model/i, // Amazon Bedrock\n\t/exceeds the context window/i, // OpenAI (Completions & Responses API)\n\t/exceeds (?:the )?(?:model'?s )?maximum context length(?: of [\\d,]+ tokens?|\\s*\\([\\d,]+\\))/i, // OpenAI-compatible proxies (LiteLLM)\n\t/input token count.*exceeds the maximum/i, // Google (Gemini)\n\t/maximum prompt length is \\d+/i, // xAI (Grok)\n\t/reduce the length of the messages/i, // Groq\n\t/maximum context length is \\d+ tokens/i, // OpenRouter (most backends)\n\t/exceeds (?:the )?maximum allowed input length of [\\d,]+ tokens?/i, // OpenRouter/Poolside\n\t/input \\(\\d+ tokens\\) is longer than the model'?s context length \\(\\d+ tokens\\)/i, // Together AI\n\t/exceeds the limit of \\d+/i, // GitHub Copilot\n\t/exceeds the available context size/i, // llama.cpp server\n\t/greater than the context length/i, // LM Studio\n\t/context window exceeds limit/i, // MiniMax\n\t/exceeded model token limit/i, // Kimi For Coding\n\t/too large for model with \\d+ maximum context length/i, // Mistral\n\t/prompt has [\\d,]+ tokens?, but the configured context size is [\\d,]+ tokens?/i, // DS4 server\n\t/model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text\n\t/prompt too long; exceeded (?:max )?context length/i, // Ollama explicit overflow error\n\t/range of input length should be/i, // DashScope / Qwen Token Plan\n\t/context[_ ]length[_ ]exceeded/i, // Generic fallback\n\t/too many tokens/i, // Generic fallback\n\t/token limit exceeded/i, // Generic fallback\n\t/^4(?:00|13)\\s*(?:status code)?\\s*\\(no body\\)/i, // Cerebras: 400/413 with no body\n];\n\n/**\n * Patterns that indicate non-overflow errors (e.g. rate limiting, server errors).\n * Error messages matching any of these are excluded from overflow detection\n * even if they also match an OVERFLOW_PATTERN.\n *\n * Example: Bedrock formats throttling errors as \"ThrottlingException: Too many tokens,\n * please wait before trying again.\" which would match the /too many tokens/i overflow\n * pattern without this exclusion.\n */\nconst NON_OVERFLOW_PATTERNS = [\n\t/^(Throttling error|Service unavailable):/i, // AWS Bedrock non-overflow errors (human-readable prefixes from formatBedrockError)\n\t/rate limit/i, // Generic rate limiting\n\t/too many requests/i, // Generic HTTP 429 style\n];\n\n/**\n * Check if an assistant message represents a context overflow error.\n *\n * This handles three cases:\n * 1. Error-based overflow: Most providers return stopReason \"error\" with a\n * specific error message pattern.\n * 2. Silent overflow: Some providers accept overflow requests and return\n * successfully. For these, we check if usage.input exceeds the context window.\n * 3. Length-stop overflow: Xiaomi MiMo can return \"length\" with zero output when\n * the input fills the context window.\n *\n * ## Reliability by Provider\n *\n * **Reliable detection (returns error with detectable message):**\n * - Anthropic: \"prompt is too long: X tokens > Y maximum\" or \"request_too_large\"\n * - OpenAI (Completions & Responses): \"exceeds the context window\", \"exceeds the model's maximum context length of X tokens\", or \"exceeds model's maximum context length (X)\"\n * - Google Gemini: \"input token count exceeds the maximum\"\n * - xAI (Grok): \"maximum prompt length is X but request contains Y\"\n * - Groq: \"reduce the length of the messages\"\n * - Cerebras: 400/413 status code (no body)\n * - Mistral: \"Prompt contains X tokens ... too large for model with Y maximum context length\"\n * - OpenRouter (most backends): \"maximum context length is X tokens\"\n * - OpenRouter/Poolside: \"Input length X exceeds the maximum allowed input length of Y tokens.\"\n * - Together AI: \"The input (X tokens) is longer than the model's context length (Y tokens).\"\n * - llama.cpp: \"exceeds the available context size\"\n * - LM Studio: \"greater than the context length\"\n * - Kimi For Coding: \"exceeded model token limit: X (requested: Y)\"\n * - DS4: \"Prompt has X tokens, but the configured context size is Y tokens\"\n * - DashScope/Qwen: \"Range of input length should be [1, X]\"\n *\n * **Unreliable detection:**\n * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),\n * sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.\n * - Xiaomi MiMo: Truncates input to fit contextWindow then returns stopReason \"length\" with\n * output=0. Pass contextWindow param to detect via the \"filled context + zero output\" signal.\n * - Ollama: May truncate input silently for some setups, but may also return explicit\n * overflow errors that match the patterns above. Silent truncation still cannot be\n * detected here because we do not know the expected token count.\n *\n * ## Custom Providers\n *\n * If you've added custom models via settings.json, this function may not detect\n * overflow errors from those providers. To add support:\n *\n * 1. Send a request that exceeds the model's context window\n * 2. Check the errorMessage in the response\n * 3. Create a regex pattern that matches the error\n * 4. The pattern should be added to OVERFLOW_PATTERNS in this file, or\n * check the errorMessage yourself before calling this function\n *\n * @param message - The assistant message to check\n * @param contextWindow - Optional context window size for detecting silent overflow (z.ai)\n * @returns true if the message indicates a context overflow\n */\nexport function isContextOverflow(message: AssistantMessage, contextWindow?: number): boolean {\n\t// Case 1: Check error message patterns\n\tif (message.stopReason === \"error\" && message.errorMessage) {\n\t\t// Skip messages matching known non-overflow patterns (e.g. throttling / rate-limit)\n\t\tconst isNonOverflow = NON_OVERFLOW_PATTERNS.some((p) => p.test(message.errorMessage!));\n\t\tif (!isNonOverflow && OVERFLOW_PATTERNS.some((p) => p.test(message.errorMessage!))) {\n\t\t\treturn true;\n\t\t}\n\t}\n\n\t// Case 2: Silent overflow (z.ai style) - successful but usage exceeds context\n\tif (contextWindow && message.stopReason === \"stop\") {\n\t\tconst inputTokens = message.usage.input + message.usage.cacheRead;\n\t\tif (inputTokens > contextWindow) {\n\t\t\treturn true;\n\t\t}\n\t}\n\n\t// Case 3: Length-stop overflow (Xiaomi MiMo style) - server truncates oversized input\n\t// to fit the context window, leaving no room for output. Returns stopReason \"length\"\n\t// with output=0 and input+cacheRead filling the context window.\n\tif (contextWindow && message.stopReason === \"length\" && message.usage.output === 0) {\n\t\tconst inputTokens = message.usage.input + message.usage.cacheRead;\n\t\tif (inputTokens >= contextWindow * 0.99) {\n\t\t\treturn true;\n\t\t}\n\t}\n\n\treturn false;\n}\n\n/**\n * Check whether a length stop ended below the caller or model's intended output limit.\n * Such responses may be caused by context pressure or provider-side truncation, so callers\n * can make one bounded compact-and-retry attempt. `desiredMaxOutput` must be the original\n * limit before any context-based clamping.\n */\nexport function isRecoverableLength(message: AssistantMessage, desiredMaxOutput: number): boolean {\n\treturn message.stopReason === \"length\" && desiredMaxOutput > 0 && message.usage.output < desiredMaxOutput;\n}\n\n/**\n * Get the overflow patterns for testing purposes.\n */\nexport function getOverflowPatterns(): RegExp[] {\n\treturn [...OVERFLOW_PATTERNS];\n}\n"]}
1
+ {"version":3,"file":"overflow.d.ts","sourceRoot":"","sources":["../../src/utils/overflow.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,aAAa,CAAC;AAwFpD;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAsDG;AACH,wBAAgB,iBAAiB,CAAC,OAAO,EAAE,gBAAgB,EAAE,aAAa,CAAC,EAAE,MAAM,GAAG,OAAO,CAkC5F;AAED;;;;;GAKG;AACH,wBAAgB,mBAAmB,CAAC,OAAO,EAAE,gBAAgB,EAAE,gBAAgB,EAAE,MAAM,GAAG,OAAO,CAEhG;AAED;;GAEG;AACH,wBAAgB,mBAAmB,IAAI,MAAM,EAAE,CAE9C","sourcesContent":["import type { AssistantMessage } from \"../types.ts\";\n\n/**\n * Regex patterns to detect context overflow errors from different providers.\n *\n * These patterns match error messages returned when the input exceeds\n * the model's context window.\n *\n * Provider-specific patterns (with example error messages):\n *\n * - Anthropic: \"prompt is too long: 213462 tokens > 200000 maximum\"\n * - Anthropic: \"413 {\\\"error\\\":{\\\"type\\\":\\\"request_too_large\\\",\\\"message\\\":\\\"Request exceeds the maximum size\\\"}}\"\n * - OpenAI: \"Your input exceeds the context window of this model\"\n * - OpenAI/LiteLLM: \"Requested token count exceeds the model's maximum context length of 131072 tokens\"\n * - OpenAI-compatible: \"Input length (265330) exceeds model's maximum context length (262144).\"\n * - Google: \"The input token count (1196265) exceeds the maximum number of tokens allowed (1048575)\"\n * - xAI: \"This model's maximum prompt length is 131072 but the request contains 537812 tokens\"\n * - Groq: \"Please reduce the length of the messages or completion\"\n * - OpenRouter: \"This endpoint's maximum context length is X tokens. However, you requested about Y tokens\"\n * - OpenRouter/Poolside: \"Input length X exceeds the maximum allowed input length of Y tokens.\"\n * - Together AI: \"The input (X tokens) is longer than the model's context length (Y tokens).\"\n * - llama.cpp: \"the request exceeds the available context size, try increasing it\"\n * - LM Studio: \"tokens to keep from the initial prompt is greater than the context length\"\n * - GitHub Copilot: \"prompt token count of X exceeds the limit of Y\"\n * - MiniMax: \"invalid params, context window exceeds limit\"\n * - Kimi For Coding: \"Your request exceeded model token limit: X (requested: Y)\"\n * - DS4: \"Prompt has X tokens, but the configured context size is Y tokens\"\n * - Cerebras: \"400/413 status code (no body)\"\n * - Mistral: \"Prompt contains X tokens ... too large for model with Y maximum context length\"\n * - z.ai: `{\"code\":\"1261\",\"message\":\"Prompt too long\"}` or silent overflow via usage.input > contextWindow\n * - Xiaomi MiMo: Truncates input to fill contextWindow exactly, then returns finish_reason \"length\"\n * with output=0 (no room left to generate). Detected via stopReason \"length\" + zero output +\n * input filling the context window.\n * - DashScope/Qwen: \"Range of input length should be [1, X]\" (HTTP 400 invalid_parameter_error)\n * - Ollama: Some deployments truncate silently, others return errors like \"prompt too long; exceeded max context length by X tokens\"\n */\nconst OVERFLOW_PATTERNS = [\n\t/prompt (?:is )?too long/i, // Anthropic and z.ai token overflow\n\t/request_too_large/i, // Anthropic request byte-size overflow (HTTP 413)\n\t/input is too long for requested model/i, // Amazon Bedrock\n\t/exceeds the context window/i, // OpenAI (Completions & Responses API)\n\t/exceeds (?:the )?(?:model'?s )?maximum context length(?: of [\\d,]+ tokens?|\\s*\\([\\d,]+\\))/i, // OpenAI-compatible proxies (LiteLLM)\n\t/input token count.*exceeds the maximum/i, // Google (Gemini)\n\t/maximum prompt length is \\d+/i, // xAI (Grok)\n\t/reduce the length of the messages/i, // Groq\n\t/maximum context length is \\d+ tokens/i, // OpenRouter (most backends)\n\t/exceeds (?:the )?maximum allowed input length of [\\d,]+ tokens?/i, // OpenRouter/Poolside\n\t/input \\(\\d+ tokens\\) is longer than the model'?s context length \\(\\d+ tokens\\)/i, // Together AI\n\t/exceeds the limit of \\d+/i, // GitHub Copilot\n\t/exceeds the available context size/i, // llama.cpp server\n\t/greater than the context length/i, // LM Studio\n\t/context window exceeds limit/i, // MiniMax\n\t/exceeded model token limit/i, // Kimi For Coding\n\t/too large for model with \\d+ maximum context length/i, // Mistral\n\t/prompt has [\\d,]+ tokens?, but the configured context size is [\\d,]+ tokens?/i, // DS4 server\n\t/model_context_window_exceeded/i, // z.ai non-standard finish_reason surfaced as error text\n\t/prompt too long; exceeded (?:max )?context length/i, // Ollama explicit overflow error\n\t/range of input length should be/i, // DashScope / Qwen Token Plan\n\t/context[_ ]length[_ ]exceeded/i, // Generic fallback\n\t/too many tokens/i, // Generic fallback\n\t/token limit exceeded/i, // Generic fallback\n];\n\nfunction isCerebrasBodylessOverflow(errorMessage: string): boolean {\n\tconst normalized = errorMessage.trim().toLowerCase();\n\treturn (\n\t\tnormalized === \"400 status code (no body)\" ||\n\t\tnormalized === \"413 status code (no body)\" ||\n\t\tnormalized === \"400 (no body)\" ||\n\t\tnormalized === \"413 (no body)\"\n\t);\n}\n\n/**\n * Patterns that indicate non-overflow errors (e.g. rate limiting, server errors).\n * Error messages matching any of these are excluded from overflow detection\n * even if they also match an OVERFLOW_PATTERN.\n *\n * Example: Bedrock formats throttling errors as \"ThrottlingException: Too many tokens,\n * please wait before trying again.\" which would match the /too many tokens/i overflow\n * pattern without this exclusion.\n */\nconst NON_OVERFLOW_PATTERNS = [\n\t/^(Throttling error|Service unavailable):/i, // AWS Bedrock non-overflow errors (human-readable prefixes from formatBedrockError)\n\t/rate limit/i, // Generic rate limiting\n\t/too many requests/i, // Generic HTTP 429 style\n];\n\n/**\n * Check if an assistant message represents a context overflow error.\n *\n * This handles three cases:\n * 1. Error-based overflow: Most providers return stopReason \"error\" with a\n * specific error message pattern.\n * 2. Silent overflow: Some providers accept overflow requests and return\n * successfully. For these, we check if usage.input exceeds the context window.\n * 3. Length-stop overflow: Xiaomi MiMo can return \"length\" with zero output when\n * the input fills the context window.\n *\n * ## Reliability by Provider\n *\n * **Reliable detection (returns error with detectable message):**\n * - Anthropic: \"prompt is too long: X tokens > Y maximum\" or \"request_too_large\"\n * - OpenAI (Completions & Responses): \"exceeds the context window\", \"exceeds the model's maximum context length of X tokens\", or \"exceeds model's maximum context length (X)\"\n * - Google Gemini: \"input token count exceeds the maximum\"\n * - xAI (Grok): \"maximum prompt length is X but request contains Y\"\n * - Groq: \"reduce the length of the messages\"\n * - Cerebras: 400/413 status code (no body)\n * - Mistral: \"Prompt contains X tokens ... too large for model with Y maximum context length\"\n * - OpenRouter (most backends): \"maximum context length is X tokens\"\n * - OpenRouter/Poolside: \"Input length X exceeds the maximum allowed input length of Y tokens.\"\n * - Together AI: \"The input (X tokens) is longer than the model's context length (Y tokens).\"\n * - llama.cpp: \"exceeds the available context size\"\n * - LM Studio: \"greater than the context length\"\n * - Kimi For Coding: \"exceeded model token limit: X (requested: Y)\"\n * - DS4: \"Prompt has X tokens, but the configured context size is Y tokens\"\n * - DashScope/Qwen: \"Range of input length should be [1, X]\"\n * - z.ai: \"Prompt too long\"\n *\n * **Unreliable detection:**\n * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),\n * sometimes returns rate limit errors. Pass contextWindow param to detect silent overflow.\n * - Xiaomi MiMo: Truncates input to fit contextWindow then returns stopReason \"length\" with\n * output=0. Pass contextWindow param to detect via the \"filled context + zero output\" signal.\n * - Ollama: May truncate input silently for some setups, but may also return explicit\n * overflow errors that match the patterns above. Silent truncation still cannot be\n * detected here because we do not know the expected token count.\n *\n * ## Custom Providers\n *\n * If you've added custom models via settings.json, this function may not detect\n * overflow errors from those providers. To add support:\n *\n * 1. Send a request that exceeds the model's context window\n * 2. Check the errorMessage in the response\n * 3. Create a regex pattern that matches the error\n * 4. The pattern should be added to OVERFLOW_PATTERNS in this file, or\n * check the errorMessage yourself before calling this function\n *\n * @param message - The assistant message to check\n * @param contextWindow - Optional context window size for detecting silent overflow (z.ai)\n * @returns true if the message indicates a context overflow\n */\nexport function isContextOverflow(message: AssistantMessage, contextWindow?: number): boolean {\n\t// Case 1: Check error message patterns\n\tif (message.stopReason === \"error\" && message.errorMessage) {\n\t\t// Skip messages matching known non-overflow patterns (e.g. throttling / rate-limit)\n\t\tconst isNonOverflow = NON_OVERFLOW_PATTERNS.some((p) => p.test(message.errorMessage!));\n\t\tif (!isNonOverflow) {\n\t\t\tif (OVERFLOW_PATTERNS.some((p) => p.test(message.errorMessage!))) {\n\t\t\t\treturn true;\n\t\t\t}\n\t\t\tif (message.provider === \"cerebras\" && isCerebrasBodylessOverflow(message.errorMessage)) {\n\t\t\t\treturn true;\n\t\t\t}\n\t\t}\n\t}\n\n\t// Case 2: Silent overflow (z.ai style) - successful but usage exceeds context\n\tif (contextWindow && message.stopReason === \"stop\") {\n\t\tconst inputTokens = message.usage.input + message.usage.cacheRead;\n\t\tif (inputTokens > contextWindow) {\n\t\t\treturn true;\n\t\t}\n\t}\n\n\t// Case 3: Length-stop overflow (Xiaomi MiMo style) - server truncates oversized input\n\t// to fit the context window, leaving no room for output. Returns stopReason \"length\"\n\t// with output=0 and input+cacheRead filling the context window.\n\tif (contextWindow && message.stopReason === \"length\" && message.usage.output === 0) {\n\t\tconst inputTokens = message.usage.input + message.usage.cacheRead;\n\t\tif (inputTokens >= contextWindow * 0.99) {\n\t\t\treturn true;\n\t\t}\n\t}\n\n\treturn false;\n}\n\n/**\n * Check whether a length stop ended below the caller or model's intended output limit.\n * Such responses may be caused by context pressure or provider-side truncation, so callers\n * can make one bounded compact-and-retry attempt. `desiredMaxOutput` must be the original\n * limit before any context-based clamping.\n */\nexport function isRecoverableLength(message: AssistantMessage, desiredMaxOutput: number): boolean {\n\treturn message.stopReason === \"length\" && desiredMaxOutput > 0 && message.usage.output < desiredMaxOutput;\n}\n\n/**\n * Get the overflow patterns for testing purposes.\n */\nexport function getOverflowPatterns(): RegExp[] {\n\treturn [...OVERFLOW_PATTERNS];\n}\n"]}
@@ -25,7 +25,7 @@
25
25
  * - DS4: "Prompt has X tokens, but the configured context size is Y tokens"
26
26
  * - Cerebras: "400/413 status code (no body)"
27
27
  * - Mistral: "Prompt contains X tokens ... too large for model with Y maximum context length"
28
- * - z.ai: Does NOT error, accepts overflow silently - handled via usage.input > contextWindow
28
+ * - z.ai: `{"code":"1261","message":"Prompt too long"}` or silent overflow via usage.input > contextWindow
29
29
  * - Xiaomi MiMo: Truncates input to fill contextWindow exactly, then returns finish_reason "length"
30
30
  * with output=0 (no room left to generate). Detected via stopReason "length" + zero output +
31
31
  * input filling the context window.
@@ -33,7 +33,7 @@
33
33
  * - Ollama: Some deployments truncate silently, others return errors like "prompt too long; exceeded max context length by X tokens"
34
34
  */
35
35
  const OVERFLOW_PATTERNS = [
36
- /prompt is too long/i, // Anthropic token overflow
36
+ /prompt (?:is )?too long/i, // Anthropic and z.ai token overflow
37
37
  /request_too_large/i, // Anthropic request byte-size overflow (HTTP 413)
38
38
  /input is too long for requested model/i, // Amazon Bedrock
39
39
  /exceeds the context window/i, // OpenAI (Completions & Responses API)
@@ -57,8 +57,14 @@ const OVERFLOW_PATTERNS = [
57
57
  /context[_ ]length[_ ]exceeded/i, // Generic fallback
58
58
  /too many tokens/i, // Generic fallback
59
59
  /token limit exceeded/i, // Generic fallback
60
- /^4(?:00|13)\s*(?:status code)?\s*\(no body\)/i, // Cerebras: 400/413 with no body
61
60
  ];
61
+ function isCerebrasBodylessOverflow(errorMessage) {
62
+ const normalized = errorMessage.trim().toLowerCase();
63
+ return (normalized === "400 status code (no body)" ||
64
+ normalized === "413 status code (no body)" ||
65
+ normalized === "400 (no body)" ||
66
+ normalized === "413 (no body)");
67
+ }
62
68
  /**
63
69
  * Patterns that indicate non-overflow errors (e.g. rate limiting, server errors).
64
70
  * Error messages matching any of these are excluded from overflow detection
@@ -102,6 +108,7 @@ const NON_OVERFLOW_PATTERNS = [
102
108
  * - Kimi For Coding: "exceeded model token limit: X (requested: Y)"
103
109
  * - DS4: "Prompt has X tokens, but the configured context size is Y tokens"
104
110
  * - DashScope/Qwen: "Range of input length should be [1, X]"
111
+ * - z.ai: "Prompt too long"
105
112
  *
106
113
  * **Unreliable detection:**
107
114
  * - z.ai: Sometimes accepts overflow silently (detectable via usage.input > contextWindow),
@@ -132,8 +139,13 @@ export function isContextOverflow(message, contextWindow) {
132
139
  if (message.stopReason === "error" && message.errorMessage) {
133
140
  // Skip messages matching known non-overflow patterns (e.g. throttling / rate-limit)
134
141
  const isNonOverflow = NON_OVERFLOW_PATTERNS.some((p) => p.test(message.errorMessage));
135
- if (!isNonOverflow && OVERFLOW_PATTERNS.some((p) => p.test(message.errorMessage))) {
136
- return true;
142
+ if (!isNonOverflow) {
143
+ if (OVERFLOW_PATTERNS.some((p) => p.test(message.errorMessage))) {
144
+ return true;
145
+ }
146
+ if (message.provider === "cerebras" && isCerebrasBodylessOverflow(message.errorMessage)) {
147
+ return true;
148
+ }
137
149
  }
138
150
  }
139
151
  // Case 2: Silent overflow (z.ai style) - successful but usage exceeds context