@sayknow-cli/ai 0.3.6 → 0.3.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (198) hide show
  1. package/dist/types/api-registry.d.ts +30 -0
  2. package/dist/types/auth-broker/client.d.ts +67 -0
  3. package/dist/types/auth-broker/index.d.ts +5 -0
  4. package/dist/types/auth-broker/refresher.d.ts +25 -0
  5. package/dist/types/auth-broker/remote-store.d.ts +99 -0
  6. package/dist/types/auth-broker/server.d.ts +32 -0
  7. package/dist/types/auth-broker/types.d.ts +110 -0
  8. package/dist/types/auth-broker/wire-schemas.d.ts +443 -0
  9. package/dist/types/auth-gateway/http.d.ts +40 -0
  10. package/dist/types/auth-gateway/index.d.ts +3 -0
  11. package/dist/types/auth-gateway/server.d.ts +17 -0
  12. package/dist/types/auth-gateway/types.d.ts +115 -0
  13. package/dist/types/auth-storage.d.ts +679 -0
  14. package/dist/types/cli.d.ts +2 -0
  15. package/dist/types/index.d.ts +51 -0
  16. package/dist/types/model-cache.d.ts +17 -0
  17. package/dist/types/model-manager.d.ts +62 -0
  18. package/dist/types/model-thinking.d.ts +74 -0
  19. package/dist/types/models.d.ts +12 -0
  20. package/dist/types/provider-details.d.ts +24 -0
  21. package/dist/types/provider-models/bundled-references.d.ts +4 -0
  22. package/dist/types/provider-models/descriptors.d.ts +48 -0
  23. package/dist/types/provider-models/google.d.ts +20 -0
  24. package/dist/types/provider-models/index.d.ts +5 -0
  25. package/dist/types/provider-models/ollama.d.ts +7 -0
  26. package/dist/types/provider-models/openai-compat.d.ts +251 -0
  27. package/dist/types/provider-models/special.d.ts +19 -0
  28. package/dist/types/providers/amazon-bedrock.d.ts +60 -0
  29. package/dist/types/providers/anthropic-messages-server-schema.d.ts +450 -0
  30. package/dist/types/providers/anthropic-messages-server.d.ts +17 -0
  31. package/dist/types/providers/anthropic.d.ts +208 -0
  32. package/dist/types/providers/aws-credentials.d.ts +43 -0
  33. package/dist/types/providers/aws-eventstream.d.ts +38 -0
  34. package/dist/types/providers/aws-sigv4.d.ts +55 -0
  35. package/dist/types/providers/azure-openai-responses.d.ts +15 -0
  36. package/dist/types/providers/composer-discipline.d.ts +26 -0
  37. package/dist/types/providers/cursor/gen/agent_pb.d.ts +13022 -0
  38. package/dist/types/providers/cursor.d.ts +44 -0
  39. package/dist/types/providers/error-message.d.ts +27 -0
  40. package/dist/types/providers/github-copilot-headers.d.ts +40 -0
  41. package/dist/types/providers/gitlab-duo.d.ts +27 -0
  42. package/dist/types/providers/google-auth.d.ts +24 -0
  43. package/dist/types/providers/google-gemini-cli.d.ts +75 -0
  44. package/dist/types/providers/google-gemini-headers.d.ts +40 -0
  45. package/dist/types/providers/google-shared.d.ts +173 -0
  46. package/dist/types/providers/google-types.d.ts +138 -0
  47. package/dist/types/providers/google-vertex.d.ts +7 -0
  48. package/dist/types/providers/google.d.ts +4 -0
  49. package/dist/types/providers/grammar.d.ts +1 -0
  50. package/dist/types/providers/kimi.d.ts +27 -0
  51. package/dist/types/providers/mock.d.ts +177 -0
  52. package/dist/types/providers/ollama.d.ts +41 -0
  53. package/dist/types/providers/openai-anthropic-shim.d.ts +31 -0
  54. package/dist/types/providers/openai-bounded-rate-limits.d.ts +3 -0
  55. package/dist/types/providers/openai-chat-server-schema.d.ts +815 -0
  56. package/dist/types/providers/openai-chat-server.d.ts +16 -0
  57. package/dist/types/providers/openai-codex/constants.d.ts +26 -0
  58. package/dist/types/providers/openai-codex/request-transformer.d.ts +50 -0
  59. package/dist/types/providers/openai-codex/response-handler.d.ts +17 -0
  60. package/dist/types/providers/openai-codex-responses.d.ts +67 -0
  61. package/dist/types/providers/openai-completions-compat.d.ts +27 -0
  62. package/dist/types/providers/openai-completions.d.ts +33 -0
  63. package/dist/types/providers/openai-request-transform.d.ts +4 -0
  64. package/dist/types/providers/openai-responses-server-schema.d.ts +392 -0
  65. package/dist/types/providers/openai-responses-server.d.ts +17 -0
  66. package/dist/types/providers/openai-responses-shared.d.ts +104 -0
  67. package/dist/types/providers/openai-responses.d.ts +32 -0
  68. package/dist/types/providers/pi-native-client.d.ts +13 -0
  69. package/dist/types/providers/pi-native-server.d.ts +68 -0
  70. package/dist/types/providers/register-builtins.d.ts +31 -0
  71. package/dist/types/providers/synthetic.d.ts +26 -0
  72. package/dist/types/providers/transform-messages.d.ts +14 -0
  73. package/dist/types/providers/vision-guard.d.ts +8 -0
  74. package/dist/types/rate-limit-utils.d.ts +19 -0
  75. package/dist/types/stream.d.ts +43 -0
  76. package/dist/types/types.d.ts +830 -0
  77. package/dist/types/usage/claude.d.ts +3 -0
  78. package/dist/types/usage/gemini.d.ts +2 -0
  79. package/dist/types/usage/github-copilot.d.ts +7 -0
  80. package/dist/types/usage/google-antigravity.d.ts +2 -0
  81. package/dist/types/usage/grok-cli.d.ts +10 -0
  82. package/dist/types/usage/kimi.d.ts +2 -0
  83. package/dist/types/usage/minimax-code.d.ts +2 -0
  84. package/dist/types/usage/openai-codex.d.ts +3 -0
  85. package/dist/types/usage/shared.d.ts +1 -0
  86. package/dist/types/usage/zai.d.ts +2 -0
  87. package/dist/types/usage.d.ts +258 -0
  88. package/dist/types/utils/abort.d.ts +19 -0
  89. package/dist/types/utils/anthropic-auth.d.ts +31 -0
  90. package/dist/types/utils/discovery/antigravity.d.ts +61 -0
  91. package/dist/types/utils/discovery/codex.d.ts +38 -0
  92. package/dist/types/utils/discovery/cursor.d.ts +23 -0
  93. package/dist/types/utils/discovery/gemini.d.ts +25 -0
  94. package/dist/types/utils/discovery/index.d.ts +4 -0
  95. package/dist/types/utils/discovery/openai-compatible.d.ts +74 -0
  96. package/dist/types/utils/event-stream.d.ts +33 -0
  97. package/dist/types/utils/fireworks-model-id.d.ts +10 -0
  98. package/dist/types/utils/foundry.d.ts +1 -0
  99. package/dist/types/utils/h2-fetch.d.ts +22 -0
  100. package/dist/types/utils/http-inspector.d.ts +35 -0
  101. package/dist/types/utils/idle-iterator.d.ts +67 -0
  102. package/dist/types/utils/json-parse.d.ts +18 -0
  103. package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +18 -0
  104. package/dist/types/utils/oauth/anthropic.d.ts +22 -0
  105. package/dist/types/utils/oauth/api-key-login.d.ts +35 -0
  106. package/dist/types/utils/oauth/api-key-validation.d.ts +27 -0
  107. package/dist/types/utils/oauth/callback-server.d.ts +60 -0
  108. package/dist/types/utils/oauth/cerebras.d.ts +1 -0
  109. package/dist/types/utils/oauth/cloudflare-ai-gateway.d.ts +18 -0
  110. package/dist/types/utils/oauth/cursor.d.ts +15 -0
  111. package/dist/types/utils/oauth/deepinfra.d.ts +1 -0
  112. package/dist/types/utils/oauth/deepseek.d.ts +10 -0
  113. package/dist/types/utils/oauth/firepass.d.ts +1 -0
  114. package/dist/types/utils/oauth/fireworks.d.ts +1 -0
  115. package/dist/types/utils/oauth/fugu.d.ts +1 -0
  116. package/dist/types/utils/oauth/github-copilot.d.ts +38 -0
  117. package/dist/types/utils/oauth/gitlab-duo.d.ts +3 -0
  118. package/dist/types/utils/oauth/glm-zcode.d.ts +71 -0
  119. package/dist/types/utils/oauth/google-antigravity.d.ts +11 -0
  120. package/dist/types/utils/oauth/google-gemini-cli.d.ts +10 -0
  121. package/dist/types/utils/oauth/google-oauth-shared.d.ts +28 -0
  122. package/dist/types/utils/oauth/huggingface.d.ts +19 -0
  123. package/dist/types/utils/oauth/index.d.ts +39 -0
  124. package/dist/types/utils/oauth/kagi.d.ts +17 -0
  125. package/dist/types/utils/oauth/kilo.d.ts +5 -0
  126. package/dist/types/utils/oauth/kimi.d.ts +21 -0
  127. package/dist/types/utils/oauth/litellm.d.ts +18 -0
  128. package/dist/types/utils/oauth/lm-studio.d.ts +17 -0
  129. package/dist/types/utils/oauth/minimax-code.d.ts +28 -0
  130. package/dist/types/utils/oauth/moonshot.d.ts +1 -0
  131. package/dist/types/utils/oauth/nanogpt.d.ts +1 -0
  132. package/dist/types/utils/oauth/nvidia.d.ts +18 -0
  133. package/dist/types/utils/oauth/ollama-cloud.d.ts +2 -0
  134. package/dist/types/utils/oauth/ollama.d.ts +18 -0
  135. package/dist/types/utils/oauth/openai-codex.d.ts +21 -0
  136. package/dist/types/utils/oauth/opencode.d.ts +18 -0
  137. package/dist/types/utils/oauth/parallel.d.ts +17 -0
  138. package/dist/types/utils/oauth/perplexity.d.ts +9 -0
  139. package/dist/types/utils/oauth/pkce.d.ts +8 -0
  140. package/dist/types/utils/oauth/qianfan.d.ts +17 -0
  141. package/dist/types/utils/oauth/qwen-portal.d.ts +19 -0
  142. package/dist/types/utils/oauth/synthetic.d.ts +1 -0
  143. package/dist/types/utils/oauth/tavily.d.ts +17 -0
  144. package/dist/types/utils/oauth/together.d.ts +1 -0
  145. package/dist/types/utils/oauth/types.d.ts +45 -0
  146. package/dist/types/utils/oauth/venice.d.ts +18 -0
  147. package/dist/types/utils/oauth/vercel-ai-gateway.d.ts +18 -0
  148. package/dist/types/utils/oauth/vllm.d.ts +16 -0
  149. package/dist/types/utils/oauth/xai.d.ts +30 -0
  150. package/dist/types/utils/oauth/xiaomi.d.ts +25 -0
  151. package/dist/types/utils/oauth/zai.d.ts +18 -0
  152. package/dist/types/utils/oauth/zenmux.d.ts +1 -0
  153. package/dist/types/utils/overflow.d.ts +64 -0
  154. package/dist/types/utils/parse-bind.d.ts +23 -0
  155. package/dist/types/utils/provider-response.d.ts +3 -0
  156. package/dist/types/utils/retry-after.d.ts +3 -0
  157. package/dist/types/utils/retry-budget.d.ts +1 -0
  158. package/dist/types/utils/retry.d.ts +26 -0
  159. package/dist/types/utils/schema/adapt.d.ts +24 -0
  160. package/dist/types/utils/schema/compatibility.d.ts +30 -0
  161. package/dist/types/utils/schema/dereference.d.ts +11 -0
  162. package/dist/types/utils/schema/draft.d.ts +10 -0
  163. package/dist/types/utils/schema/equality.d.ts +4 -0
  164. package/dist/types/utils/schema/fields.d.ts +49 -0
  165. package/dist/types/utils/schema/index.d.ts +14 -0
  166. package/dist/types/utils/schema/json-schema-validator.d.ts +12 -0
  167. package/dist/types/utils/schema/meta-validator.d.ts +2 -0
  168. package/dist/types/utils/schema/normalize.d.ts +93 -0
  169. package/dist/types/utils/schema/root-combinator.d.ts +12 -0
  170. package/dist/types/utils/schema/spill.d.ts +8 -0
  171. package/dist/types/utils/schema/stamps.d.ts +25 -0
  172. package/dist/types/utils/schema/types.d.ts +4 -0
  173. package/dist/types/utils/schema/wire.d.ts +54 -0
  174. package/dist/types/utils/schema/zod-decontaminate.d.ts +31 -0
  175. package/dist/types/utils/sse-debug.d.ts +10 -0
  176. package/dist/types/utils/tool-call-healing.d.ts +71 -0
  177. package/dist/types/utils/tool-choice-capability.d.ts +41 -0
  178. package/dist/types/utils/tool-choice.d.ts +50 -0
  179. package/dist/types/utils/validation.d.ts +17 -0
  180. package/dist/types/utils.d.ts +35 -0
  181. package/package.json +26 -25
  182. package/src/auth-broker/remote-store.ts +17 -2
  183. package/src/auth-broker/wire-schemas.ts +1 -0
  184. package/src/auth-storage.ts +80 -7
  185. package/src/providers/amazon-bedrock.ts +2 -2
  186. package/src/providers/anthropic-messages-server.ts +2 -1
  187. package/src/providers/anthropic.ts +107 -11
  188. package/src/providers/azure-openai-responses.ts +2 -1
  189. package/src/providers/google-shared.ts +3 -3
  190. package/src/providers/openai-bounded-rate-limits.ts +57 -0
  191. package/src/providers/openai-chat-server.ts +3 -2
  192. package/src/providers/openai-codex/request-transformer.ts +5 -0
  193. package/src/providers/openai-completions-compat.ts +26 -13
  194. package/src/providers/openai-responses-server.ts +6 -3
  195. package/src/providers/openai-responses-shared.ts +3 -3
  196. package/src/providers/openai-responses.ts +5 -1
  197. package/src/rate-limit-utils.ts +9 -2
  198. package/src/utils.ts +49 -1
@@ -0,0 +1,57 @@
1
+ import type { FetchImpl } from "../types";
2
+ import { getRetryAfterMsFromHeaders } from "../utils/retry-after";
3
+
4
+ const OPENAI_RETRY_DELAY_CAP_MS = 60_000;
5
+
6
+ // Mirror of `wrapAnthropicFetchForBoundedRateLimits`: OpenAI-compatible providers
7
+ // (e.g. opencode-go) return HTTP 429 for *permanent* usage/quota exhaustion — a
8
+ // monthly-cap reset that can be days away. The OpenAI SDK treats 429 as transient
9
+ // and retries up to `maxRetries`, honoring an out-of-range `Retry-After`; the
10
+ // `create()` call then hangs before the error can surface to the agent loop, so
11
+ // no assistant error is produced and the session-level retry/fallback never runs.
12
+ // Detect exhaustion and set `x-should-retry: false` so the SDK gives up at once
13
+ // and the session retry layer applies its own fail-fast (retry-after > maxDelayMs).
14
+ //
15
+ // Shared by every adapter that drives a raw OpenAI SDK client — openai-completions,
16
+ // openai-responses, and azure-openai-responses. Adapters that route through
17
+ // `fetchWithRetry` (codex, bedrock, ollama, gemini-cli) already bound 429 retries
18
+ // themselves and do not need this wrapper.
19
+ export function isOpenAIUsageExhaustionResponse(
20
+ bodyText: string,
21
+ retryAfterMs: number | undefined,
22
+ retryDelayCapMs: number,
23
+ ): boolean {
24
+ if (retryAfterMs !== undefined && retryAfterMs > retryDelayCapMs) return true;
25
+ return /monthly usage limit|usage limit reached|usage_limit_reached|out_of_credits|insufficient_quota|quota[ _]?exceeded/i.test(
26
+ bodyText,
27
+ );
28
+ }
29
+
30
+ export function wrapOpenAIFetchForBoundedRateLimits(
31
+ baseFetch: FetchImpl,
32
+ maxRetryDelayMs: number | undefined,
33
+ ): FetchImpl {
34
+ const retryDelayCapMs = maxRetryDelayMs ?? OPENAI_RETRY_DELAY_CAP_MS;
35
+ return Object.assign(
36
+ async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
37
+ const response = await baseFetch(input, init);
38
+ if (response.status !== 429 || retryDelayCapMs === 0) return response;
39
+
40
+ const headers = new Headers(response.headers);
41
+ const retryAfterMs = getRetryAfterMsFromHeaders(headers);
42
+ const bodyText = await response
43
+ .clone()
44
+ .text()
45
+ .catch(() => "");
46
+ if (!isOpenAIUsageExhaustionResponse(bodyText, retryAfterMs, retryDelayCapMs)) return response;
47
+
48
+ headers.set("x-should-retry", "false");
49
+ return new Response(bodyText, {
50
+ status: response.status,
51
+ statusText: response.statusText,
52
+ headers,
53
+ });
54
+ },
55
+ baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
56
+ );
57
+ }
@@ -19,6 +19,7 @@ import type {
19
19
  ToolResultMessage,
20
20
  TSchema,
21
21
  } from "../types";
22
+ import { sanitizeJsonStrings } from "../utils";
22
23
  import {
23
24
  type OpenAIChatContentPart,
24
25
  type OpenAIChatMessage,
@@ -453,9 +454,9 @@ function isOnlyRaw(args: Record<string, unknown>): boolean {
453
454
 
454
455
  function stringifyArgs(args: Record<string, unknown>): string {
455
456
  // `__raw` is our fallback marker for un-parseable inbound args; preserve it verbatim on the way out.
456
- if (typeof args.__raw === "string" && isOnlyRaw(args)) return args.__raw;
457
+ if (typeof args.__raw === "string" && isOnlyRaw(args)) return args.__raw.toWellFormed();
457
458
  try {
458
- return JSON.stringify(args);
459
+ return JSON.stringify(sanitizeJsonStrings(args));
459
460
  } catch {
460
461
  return "{}";
461
462
  }
@@ -1,6 +1,7 @@
1
1
  import type { Effort } from "../../model-thinking";
2
2
  import { requireSupportedEffort } from "../../model-thinking";
3
3
  import type { Api, Model } from "../../types";
4
+ import { sanitizeJsonStrings } from "../../utils";
4
5
 
5
6
  export interface ReasoningConfig {
6
7
  effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
@@ -109,6 +110,10 @@ function normalizeInputTextPartFields(input: InputItem[] | undefined): InputItem
109
110
  }
110
111
  if (normalizedItem.type === "message") {
111
112
  normalizedItem.content = normalizeTextPartFields(normalizedItem.content, `input[${itemIndex}].content`);
113
+ } else if (normalizedItem.type === "function_call" && "arguments" in itemRecord) {
114
+ itemRecord.arguments = sanitizeJsonStrings(itemRecord.arguments);
115
+ } else if (normalizedItem.type === "custom_tool_call" && typeof itemRecord.input === "string") {
116
+ itemRecord.input = itemRecord.input.toWellFormed();
112
117
  }
113
118
  return normalizedItem;
114
119
  });
@@ -104,6 +104,10 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
104
104
  baseUrl.includes("opencode.ai");
105
105
  const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
106
106
  const isOpenCodeGoReasoning = provider === "opencode-go" && Boolean(model.reasoning);
107
+ const isOpenCodeGoKimiReasoning = provider === "opencode-go" && isKimiModel && Boolean(model.reasoning);
108
+ const isOpenCodeGoKimi25Reasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.5";
109
+ const isOpenCodeGoKimi27CodeReasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.7-code";
110
+ const needsOpenCodeGoKimiEffortMap = isOpenCodeGoKimi25Reasoning || isOpenCodeGoKimi27CodeReasoning;
107
111
 
108
112
  const useMaxTokens =
109
113
  provider === "mistral" ||
@@ -170,22 +174,31 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
170
174
  xhigh: "default",
171
175
  max: "default",
172
176
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
173
- : isDeepseekFamily && model.reasoning
177
+ : needsOpenCodeGoKimiEffortMap
174
178
  ? ({
175
- minimal: "high",
176
- low: "high",
177
- medium: "high",
178
- high: "high",
179
- xhigh: "max",
180
- max: "max",
179
+ // Live Go probes (2026-07-06) showed model-specific effort gaps:
180
+ // kimi-k2.5 rejects "minimal", while kimi-k2.7-code rejects
181
+ // OpenAI-style "xhigh" and "max"; all other Kimi efforts tested
182
+ // successfully and should pass through unchanged.
183
+ ...(isOpenCodeGoKimi25Reasoning ? { minimal: "low" } : {}),
184
+ ...(isOpenCodeGoKimi27CodeReasoning ? { xhigh: "high", max: "high" } : {}),
181
185
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
182
- : isFireworks
186
+ : isDeepseekFamily && model.reasoning
183
187
  ? ({
184
- // Fireworks' OpenAI-compatible endpoint rejects OpenAI's
185
- // `minimal` literal but accepts `none` for the lowest setting.
186
- minimal: "none",
188
+ minimal: "high",
189
+ low: "high",
190
+ medium: "high",
191
+ high: "high",
192
+ xhigh: "max",
193
+ max: "max",
187
194
  } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
188
- : {};
195
+ : isFireworks
196
+ ? ({
197
+ // Fireworks' OpenAI-compatible endpoint rejects OpenAI's
198
+ // `minimal` literal but accepts `none` for the lowest setting.
199
+ minimal: "none",
200
+ } satisfies Partial<Record<OpenAIReasoningEffort, string>>)
201
+ : {};
189
202
 
190
203
  return {
191
204
  supportsStore: !isNonStandard,
@@ -198,7 +211,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
198
211
  disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel || isOpenCodeGoReasoning,
199
212
  disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
200
213
  supportsToolChoice: !isDirectDeepseekReasoning,
201
- supportsForcedToolChoice: true,
214
+ supportsForcedToolChoice: !isOpenCodeGoKimiReasoning,
202
215
  maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
203
216
  requiresToolResultName: isMistral,
204
217
  requiresAssistantAfterToolResult: false,
@@ -22,6 +22,7 @@ import type {
22
22
  Tool,
23
23
  ToolCall,
24
24
  } from "../types";
25
+ import { sanitizeJsonStrings } from "../utils";
25
26
  import {
26
27
  type OpenAIResponsesFunctionCallItem,
27
28
  type OpenAIResponsesFunctionCallOutputItem,
@@ -612,7 +613,8 @@ function buildOutputItems(message: AssistantMessage): OutputItem[] {
612
613
  } else if (part.type === "toolCall") {
613
614
  flushMessage();
614
615
  if (part.customWireName) {
615
- const rawInput = typeof part.arguments?.input === "string" ? (part.arguments.input as string) : "";
616
+ const rawInput =
617
+ typeof part.arguments?.input === "string" ? (part.arguments.input as string).toWellFormed() : "";
616
618
  out.push({
617
619
  type: "custom_tool_call",
618
620
  id: part.thoughtSignature ?? makeCustomCallId(),
@@ -627,7 +629,7 @@ function buildOutputItems(message: AssistantMessage): OutputItem[] {
627
629
  id: part.thoughtSignature ?? makeFuncCallId(),
628
630
  call_id: part.id,
629
631
  name: part.name,
630
- arguments: JSON.stringify(part.arguments ?? {}),
632
+ arguments: JSON.stringify(sanitizeJsonStrings(part.arguments ?? {})),
631
633
  status: "completed",
632
634
  });
633
635
  }
@@ -1085,7 +1087,8 @@ export function encodeStream(
1085
1087
  } else {
1086
1088
  // Standard JSON tool: arguments object on the skc side, the
1087
1089
  // wire wants the JSON string the model emitted (= streamed deltas).
1088
- const argsJson = cur.argsText || JSON.stringify(tc.arguments ?? {});
1090
+ const argsJson =
1091
+ cur.argsText.toWellFormed() || JSON.stringify(sanitizeJsonStrings(tc.arguments ?? {}));
1089
1092
  cur.argsText = argsJson;
1090
1093
  emit("response.function_call_arguments.done", {
1091
1094
  item_id: cur.itemId,
@@ -28,7 +28,7 @@ import {
28
28
  type ToolCall,
29
29
  type ToolResultMessage,
30
30
  } from "../types";
31
- import { normalizeResponsesToolCallId } from "../utils";
31
+ import { normalizeResponsesToolCallId, sanitizeJsonStrings } from "../utils";
32
32
  import type { AssistantMessageEventStream } from "../utils/event-stream";
33
33
  import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
34
34
  import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
@@ -275,7 +275,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
275
275
  }
276
276
  knownCallIds.add(normalized.callId);
277
277
  if (block.customWireName) {
278
- const rawInput = typeof block.arguments?.input === "string" ? block.arguments.input : "";
278
+ const rawInput = typeof block.arguments?.input === "string" ? block.arguments.input.toWellFormed() : "";
279
279
  customCallIds?.add(normalized.callId);
280
280
  outputItems.push({
281
281
  type: "custom_tool_call",
@@ -291,7 +291,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
291
291
  id: itemId,
292
292
  call_id: normalized.callId,
293
293
  name: block.name,
294
- arguments: JSON.stringify(block.arguments),
294
+ arguments: JSON.stringify(sanitizeJsonStrings(block.arguments ?? {})),
295
295
  });
296
296
  }
297
297
 
@@ -71,6 +71,7 @@ import {
71
71
  resolveGitHubCopilotBaseUrl,
72
72
  } from "./github-copilot-headers";
73
73
  import { compactGrammarDefinition } from "./grammar";
74
+ import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
74
75
  import {
75
76
  applyOpenAIRequestTransformBody,
76
77
  applyOpenAIRequestTransformHeaders,
@@ -274,6 +275,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
274
275
  options?.fetch,
275
276
  options?.authCredentialType,
276
277
  options?.requestMaxRetries,
278
+ options?.maxRetryDelayMs,
277
279
  );
278
280
  const premiumRequestsTotal = copilotPremiumRequests;
279
281
  const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
@@ -400,6 +402,7 @@ function createClient(
400
402
  fetchOverride?: FetchImpl,
401
403
  authCredentialType?: OpenAIResponsesOptions["authCredentialType"],
402
404
  requestMaxRetries?: number,
405
+ maxRetryDelayMs?: number,
403
406
  ): {
404
407
  client: OpenAI;
405
408
  copilotPremiumRequests: number | undefined;
@@ -446,8 +449,9 @@ function createClient(
446
449
  headers["x-client-request-id"] ??= sessionId;
447
450
  }
448
451
  const baseFetch = fetchOverride ?? fetch;
452
+ const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(baseFetch, maxRetryDelayMs);
449
453
  const transformedFetch = wrapFetchForOpenAIRequestTransform(
450
- baseFetch,
454
+ boundedFetch,
451
455
  model.requestTransform,
452
456
  `Sayknow-CLI/${packageJson.version}`,
453
457
  );
@@ -36,6 +36,14 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
36
36
  return "MODEL_CAPACITY_EXHAUSTED";
37
37
  }
38
38
 
39
+ if (
40
+ lower.includes("out_of_credits") ||
41
+ lower.includes("request would exceed your account's rate limit") ||
42
+ lower.includes("request would exceed your accounts rate limit")
43
+ ) {
44
+ return "QUOTA_EXHAUSTED";
45
+ }
46
+
39
47
  if (
40
48
  lower.includes("per minute") ||
41
49
  lower.includes("rate limit") ||
@@ -86,8 +94,7 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
86
94
 
87
95
  /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
88
96
  const USAGE_LIMIT_PATTERN =
89
- /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|resource has been exhausted[^\n]*(?:quota|limit)/i;
90
-
97
+ /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
91
98
  export function isUsageLimitError(errorMessage: string): boolean {
92
99
  return USAGE_LIMIT_PATTERN.test(errorMessage);
93
100
  }
package/src/utils.ts CHANGED
@@ -11,6 +11,34 @@ export function normalizeSystemPrompts(systemPrompt: readonly string[] | string
11
11
  return prompts.map(prompt => prompt.toWellFormed()).filter(prompt => prompt.length > 0);
12
12
  }
13
13
 
14
+ export function sanitizeJsonStrings(value: unknown): unknown {
15
+ return sanitizeJsonStringsInner(value, new WeakMap<object, unknown>());
16
+ }
17
+
18
+ function sanitizeJsonStringsInner(value: unknown, seen: WeakMap<object, unknown>): unknown {
19
+ if (typeof value === "string") return value.toWellFormed();
20
+ if (!value || typeof value !== "object") return value;
21
+
22
+ const cached = seen.get(value);
23
+ if (cached !== undefined) return cached;
24
+
25
+ if (Array.isArray(value)) {
26
+ const sanitized: unknown[] = [];
27
+ seen.set(value, sanitized);
28
+ for (const item of value) {
29
+ sanitized.push(sanitizeJsonStringsInner(item, seen));
30
+ }
31
+ return sanitized;
32
+ }
33
+
34
+ const sanitized: Record<string, unknown> = {};
35
+ seen.set(value, sanitized);
36
+ for (const [key, nestedValue] of Object.entries(value)) {
37
+ sanitized[key.toWellFormed()] = sanitizeJsonStringsInner(nestedValue, seen);
38
+ }
39
+ return sanitized;
40
+ }
41
+
14
42
  export function toNumber(value: unknown): number | undefined {
15
43
  if (typeof value === "number" && Number.isFinite(value)) return value;
16
44
  if (typeof value === "string" && value.trim()) {
@@ -187,7 +215,11 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
187
215
  // providerPayload stores raw output items; replay strips fields that are output-only.
188
216
  const { id: _id, ...itemWithoutId } = item;
189
217
  const sanitizedItem =
190
- item.type === "computer_call" ? sanitizeComputerCallForResponsesInput(itemWithoutId) : itemWithoutId;
218
+ item.type === "computer_call"
219
+ ? sanitizeComputerCallForResponsesInput(itemWithoutId)
220
+ : item.type === "image_generation_call"
221
+ ? sanitizeImageGenerationCallForResponsesInput(itemWithoutId)
222
+ : itemWithoutId;
191
223
  if (typeof item.call_id === "string") {
192
224
  sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
193
225
  }
@@ -203,6 +235,22 @@ function sanitizeComputerCallForResponsesInput(item: Record<string, unknown>): R
203
235
  return inputSafeItem;
204
236
  }
205
237
 
238
+ function sanitizeImageGenerationCallForResponsesInput(item: Record<string, unknown>): Record<string, unknown> {
239
+ // Image generation output items include request-time knobs that are not part of
240
+ // the Responses input replay schema. Replaying them verbatim makes OpenAI-compatible
241
+ // endpoints reject the next turn, e.g. `Unknown parameter: input[n].action`.
242
+ const {
243
+ action: _action,
244
+ background: _background,
245
+ output_format: _outputFormat,
246
+ quality: _quality,
247
+ revised_prompt: _revisedPrompt,
248
+ size: _size,
249
+ ...inputSafeItem
250
+ } = item;
251
+ return inputSafeItem;
252
+ }
253
+
206
254
  function normalizeReplayedResponsesHistoryCallId(value: string, normalizedValues: Map<string, string>): string {
207
255
  const normalized = normalizedValues.get(value);
208
256
  if (normalized) return normalized;