@sayknow-cli/ai 0.3.6 → 0.3.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/types/api-registry.d.ts +30 -0
- package/dist/types/auth-broker/client.d.ts +67 -0
- package/dist/types/auth-broker/index.d.ts +5 -0
- package/dist/types/auth-broker/refresher.d.ts +25 -0
- package/dist/types/auth-broker/remote-store.d.ts +99 -0
- package/dist/types/auth-broker/server.d.ts +32 -0
- package/dist/types/auth-broker/types.d.ts +110 -0
- package/dist/types/auth-broker/wire-schemas.d.ts +443 -0
- package/dist/types/auth-gateway/http.d.ts +40 -0
- package/dist/types/auth-gateway/index.d.ts +3 -0
- package/dist/types/auth-gateway/server.d.ts +17 -0
- package/dist/types/auth-gateway/types.d.ts +115 -0
- package/dist/types/auth-storage.d.ts +679 -0
- package/dist/types/cli.d.ts +2 -0
- package/dist/types/index.d.ts +51 -0
- package/dist/types/model-cache.d.ts +17 -0
- package/dist/types/model-manager.d.ts +62 -0
- package/dist/types/model-thinking.d.ts +74 -0
- package/dist/types/models.d.ts +12 -0
- package/dist/types/provider-details.d.ts +24 -0
- package/dist/types/provider-models/bundled-references.d.ts +4 -0
- package/dist/types/provider-models/descriptors.d.ts +48 -0
- package/dist/types/provider-models/google.d.ts +20 -0
- package/dist/types/provider-models/index.d.ts +5 -0
- package/dist/types/provider-models/ollama.d.ts +7 -0
- package/dist/types/provider-models/openai-compat.d.ts +251 -0
- package/dist/types/provider-models/special.d.ts +19 -0
- package/dist/types/providers/amazon-bedrock.d.ts +60 -0
- package/dist/types/providers/anthropic-messages-server-schema.d.ts +450 -0
- package/dist/types/providers/anthropic-messages-server.d.ts +17 -0
- package/dist/types/providers/anthropic.d.ts +208 -0
- package/dist/types/providers/aws-credentials.d.ts +43 -0
- package/dist/types/providers/aws-eventstream.d.ts +38 -0
- package/dist/types/providers/aws-sigv4.d.ts +55 -0
- package/dist/types/providers/azure-openai-responses.d.ts +15 -0
- package/dist/types/providers/composer-discipline.d.ts +26 -0
- package/dist/types/providers/cursor/gen/agent_pb.d.ts +13022 -0
- package/dist/types/providers/cursor.d.ts +44 -0
- package/dist/types/providers/error-message.d.ts +27 -0
- package/dist/types/providers/github-copilot-headers.d.ts +40 -0
- package/dist/types/providers/gitlab-duo.d.ts +27 -0
- package/dist/types/providers/google-auth.d.ts +24 -0
- package/dist/types/providers/google-gemini-cli.d.ts +75 -0
- package/dist/types/providers/google-gemini-headers.d.ts +40 -0
- package/dist/types/providers/google-shared.d.ts +173 -0
- package/dist/types/providers/google-types.d.ts +138 -0
- package/dist/types/providers/google-vertex.d.ts +7 -0
- package/dist/types/providers/google.d.ts +4 -0
- package/dist/types/providers/grammar.d.ts +1 -0
- package/dist/types/providers/kimi.d.ts +27 -0
- package/dist/types/providers/mock.d.ts +177 -0
- package/dist/types/providers/ollama.d.ts +41 -0
- package/dist/types/providers/openai-anthropic-shim.d.ts +31 -0
- package/dist/types/providers/openai-bounded-rate-limits.d.ts +3 -0
- package/dist/types/providers/openai-chat-server-schema.d.ts +815 -0
- package/dist/types/providers/openai-chat-server.d.ts +16 -0
- package/dist/types/providers/openai-codex/constants.d.ts +26 -0
- package/dist/types/providers/openai-codex/request-transformer.d.ts +50 -0
- package/dist/types/providers/openai-codex/response-handler.d.ts +17 -0
- package/dist/types/providers/openai-codex-responses.d.ts +67 -0
- package/dist/types/providers/openai-completions-compat.d.ts +27 -0
- package/dist/types/providers/openai-completions.d.ts +33 -0
- package/dist/types/providers/openai-request-transform.d.ts +4 -0
- package/dist/types/providers/openai-responses-server-schema.d.ts +392 -0
- package/dist/types/providers/openai-responses-server.d.ts +17 -0
- package/dist/types/providers/openai-responses-shared.d.ts +104 -0
- package/dist/types/providers/openai-responses.d.ts +32 -0
- package/dist/types/providers/pi-native-client.d.ts +13 -0
- package/dist/types/providers/pi-native-server.d.ts +68 -0
- package/dist/types/providers/register-builtins.d.ts +31 -0
- package/dist/types/providers/synthetic.d.ts +26 -0
- package/dist/types/providers/transform-messages.d.ts +14 -0
- package/dist/types/providers/vision-guard.d.ts +8 -0
- package/dist/types/rate-limit-utils.d.ts +19 -0
- package/dist/types/stream.d.ts +43 -0
- package/dist/types/types.d.ts +830 -0
- package/dist/types/usage/claude.d.ts +3 -0
- package/dist/types/usage/gemini.d.ts +2 -0
- package/dist/types/usage/github-copilot.d.ts +7 -0
- package/dist/types/usage/google-antigravity.d.ts +2 -0
- package/dist/types/usage/grok-cli.d.ts +10 -0
- package/dist/types/usage/kimi.d.ts +2 -0
- package/dist/types/usage/minimax-code.d.ts +2 -0
- package/dist/types/usage/openai-codex.d.ts +3 -0
- package/dist/types/usage/shared.d.ts +1 -0
- package/dist/types/usage/zai.d.ts +2 -0
- package/dist/types/usage.d.ts +258 -0
- package/dist/types/utils/abort.d.ts +19 -0
- package/dist/types/utils/anthropic-auth.d.ts +31 -0
- package/dist/types/utils/discovery/antigravity.d.ts +61 -0
- package/dist/types/utils/discovery/codex.d.ts +38 -0
- package/dist/types/utils/discovery/cursor.d.ts +23 -0
- package/dist/types/utils/discovery/gemini.d.ts +25 -0
- package/dist/types/utils/discovery/index.d.ts +4 -0
- package/dist/types/utils/discovery/openai-compatible.d.ts +74 -0
- package/dist/types/utils/event-stream.d.ts +33 -0
- package/dist/types/utils/fireworks-model-id.d.ts +10 -0
- package/dist/types/utils/foundry.d.ts +1 -0
- package/dist/types/utils/h2-fetch.d.ts +22 -0
- package/dist/types/utils/http-inspector.d.ts +35 -0
- package/dist/types/utils/idle-iterator.d.ts +67 -0
- package/dist/types/utils/json-parse.d.ts +18 -0
- package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +18 -0
- package/dist/types/utils/oauth/anthropic.d.ts +22 -0
- package/dist/types/utils/oauth/api-key-login.d.ts +35 -0
- package/dist/types/utils/oauth/api-key-validation.d.ts +27 -0
- package/dist/types/utils/oauth/callback-server.d.ts +60 -0
- package/dist/types/utils/oauth/cerebras.d.ts +1 -0
- package/dist/types/utils/oauth/cloudflare-ai-gateway.d.ts +18 -0
- package/dist/types/utils/oauth/cursor.d.ts +15 -0
- package/dist/types/utils/oauth/deepinfra.d.ts +1 -0
- package/dist/types/utils/oauth/deepseek.d.ts +10 -0
- package/dist/types/utils/oauth/firepass.d.ts +1 -0
- package/dist/types/utils/oauth/fireworks.d.ts +1 -0
- package/dist/types/utils/oauth/fugu.d.ts +1 -0
- package/dist/types/utils/oauth/github-copilot.d.ts +38 -0
- package/dist/types/utils/oauth/gitlab-duo.d.ts +3 -0
- package/dist/types/utils/oauth/glm-zcode.d.ts +71 -0
- package/dist/types/utils/oauth/google-antigravity.d.ts +11 -0
- package/dist/types/utils/oauth/google-gemini-cli.d.ts +10 -0
- package/dist/types/utils/oauth/google-oauth-shared.d.ts +28 -0
- package/dist/types/utils/oauth/huggingface.d.ts +19 -0
- package/dist/types/utils/oauth/index.d.ts +39 -0
- package/dist/types/utils/oauth/kagi.d.ts +17 -0
- package/dist/types/utils/oauth/kilo.d.ts +5 -0
- package/dist/types/utils/oauth/kimi.d.ts +21 -0
- package/dist/types/utils/oauth/litellm.d.ts +18 -0
- package/dist/types/utils/oauth/lm-studio.d.ts +17 -0
- package/dist/types/utils/oauth/minimax-code.d.ts +28 -0
- package/dist/types/utils/oauth/moonshot.d.ts +1 -0
- package/dist/types/utils/oauth/nanogpt.d.ts +1 -0
- package/dist/types/utils/oauth/nvidia.d.ts +18 -0
- package/dist/types/utils/oauth/ollama-cloud.d.ts +2 -0
- package/dist/types/utils/oauth/ollama.d.ts +18 -0
- package/dist/types/utils/oauth/openai-codex.d.ts +21 -0
- package/dist/types/utils/oauth/opencode.d.ts +18 -0
- package/dist/types/utils/oauth/parallel.d.ts +17 -0
- package/dist/types/utils/oauth/perplexity.d.ts +9 -0
- package/dist/types/utils/oauth/pkce.d.ts +8 -0
- package/dist/types/utils/oauth/qianfan.d.ts +17 -0
- package/dist/types/utils/oauth/qwen-portal.d.ts +19 -0
- package/dist/types/utils/oauth/synthetic.d.ts +1 -0
- package/dist/types/utils/oauth/tavily.d.ts +17 -0
- package/dist/types/utils/oauth/together.d.ts +1 -0
- package/dist/types/utils/oauth/types.d.ts +45 -0
- package/dist/types/utils/oauth/venice.d.ts +18 -0
- package/dist/types/utils/oauth/vercel-ai-gateway.d.ts +18 -0
- package/dist/types/utils/oauth/vllm.d.ts +16 -0
- package/dist/types/utils/oauth/xai.d.ts +30 -0
- package/dist/types/utils/oauth/xiaomi.d.ts +25 -0
- package/dist/types/utils/oauth/zai.d.ts +18 -0
- package/dist/types/utils/oauth/zenmux.d.ts +1 -0
- package/dist/types/utils/overflow.d.ts +64 -0
- package/dist/types/utils/parse-bind.d.ts +23 -0
- package/dist/types/utils/provider-response.d.ts +3 -0
- package/dist/types/utils/retry-after.d.ts +3 -0
- package/dist/types/utils/retry-budget.d.ts +1 -0
- package/dist/types/utils/retry.d.ts +26 -0
- package/dist/types/utils/schema/adapt.d.ts +24 -0
- package/dist/types/utils/schema/compatibility.d.ts +30 -0
- package/dist/types/utils/schema/dereference.d.ts +11 -0
- package/dist/types/utils/schema/draft.d.ts +10 -0
- package/dist/types/utils/schema/equality.d.ts +4 -0
- package/dist/types/utils/schema/fields.d.ts +49 -0
- package/dist/types/utils/schema/index.d.ts +14 -0
- package/dist/types/utils/schema/json-schema-validator.d.ts +12 -0
- package/dist/types/utils/schema/meta-validator.d.ts +2 -0
- package/dist/types/utils/schema/normalize.d.ts +93 -0
- package/dist/types/utils/schema/root-combinator.d.ts +12 -0
- package/dist/types/utils/schema/spill.d.ts +8 -0
- package/dist/types/utils/schema/stamps.d.ts +25 -0
- package/dist/types/utils/schema/types.d.ts +4 -0
- package/dist/types/utils/schema/wire.d.ts +54 -0
- package/dist/types/utils/schema/zod-decontaminate.d.ts +31 -0
- package/dist/types/utils/sse-debug.d.ts +10 -0
- package/dist/types/utils/tool-call-healing.d.ts +71 -0
- package/dist/types/utils/tool-choice-capability.d.ts +41 -0
- package/dist/types/utils/tool-choice.d.ts +50 -0
- package/dist/types/utils/validation.d.ts +17 -0
- package/dist/types/utils.d.ts +35 -0
- package/package.json +26 -25
- package/src/auth-broker/remote-store.ts +17 -2
- package/src/auth-broker/wire-schemas.ts +1 -0
- package/src/auth-storage.ts +80 -7
- package/src/providers/amazon-bedrock.ts +2 -2
- package/src/providers/anthropic-messages-server.ts +2 -1
- package/src/providers/anthropic.ts +107 -11
- package/src/providers/azure-openai-responses.ts +2 -1
- package/src/providers/google-shared.ts +3 -3
- package/src/providers/openai-bounded-rate-limits.ts +57 -0
- package/src/providers/openai-chat-server.ts +3 -2
- package/src/providers/openai-codex/request-transformer.ts +5 -0
- package/src/providers/openai-completions-compat.ts +26 -13
- package/src/providers/openai-responses-server.ts +6 -3
- package/src/providers/openai-responses-shared.ts +3 -3
- package/src/providers/openai-responses.ts +5 -1
- package/src/rate-limit-utils.ts +9 -2
- package/src/utils.ts +49 -1
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import type { FetchImpl } from "../types";
|
|
2
|
+
import { getRetryAfterMsFromHeaders } from "../utils/retry-after";
|
|
3
|
+
|
|
4
|
+
const OPENAI_RETRY_DELAY_CAP_MS = 60_000;
|
|
5
|
+
|
|
6
|
+
// Mirror of `wrapAnthropicFetchForBoundedRateLimits`: OpenAI-compatible providers
|
|
7
|
+
// (e.g. opencode-go) return HTTP 429 for *permanent* usage/quota exhaustion — a
|
|
8
|
+
// monthly-cap reset that can be days away. The OpenAI SDK treats 429 as transient
|
|
9
|
+
// and retries up to `maxRetries`, honoring an out-of-range `Retry-After`; the
|
|
10
|
+
// `create()` call then hangs before the error can surface to the agent loop, so
|
|
11
|
+
// no assistant error is produced and the session-level retry/fallback never runs.
|
|
12
|
+
// Detect exhaustion and set `x-should-retry: false` so the SDK gives up at once
|
|
13
|
+
// and the session retry layer applies its own fail-fast (retry-after > maxDelayMs).
|
|
14
|
+
//
|
|
15
|
+
// Shared by every adapter that drives a raw OpenAI SDK client — openai-completions,
|
|
16
|
+
// openai-responses, and azure-openai-responses. Adapters that route through
|
|
17
|
+
// `fetchWithRetry` (codex, bedrock, ollama, gemini-cli) already bound 429 retries
|
|
18
|
+
// themselves and do not need this wrapper.
|
|
19
|
+
export function isOpenAIUsageExhaustionResponse(
|
|
20
|
+
bodyText: string,
|
|
21
|
+
retryAfterMs: number | undefined,
|
|
22
|
+
retryDelayCapMs: number,
|
|
23
|
+
): boolean {
|
|
24
|
+
if (retryAfterMs !== undefined && retryAfterMs > retryDelayCapMs) return true;
|
|
25
|
+
return /monthly usage limit|usage limit reached|usage_limit_reached|out_of_credits|insufficient_quota|quota[ _]?exceeded/i.test(
|
|
26
|
+
bodyText,
|
|
27
|
+
);
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export function wrapOpenAIFetchForBoundedRateLimits(
|
|
31
|
+
baseFetch: FetchImpl,
|
|
32
|
+
maxRetryDelayMs: number | undefined,
|
|
33
|
+
): FetchImpl {
|
|
34
|
+
const retryDelayCapMs = maxRetryDelayMs ?? OPENAI_RETRY_DELAY_CAP_MS;
|
|
35
|
+
return Object.assign(
|
|
36
|
+
async (input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
|
37
|
+
const response = await baseFetch(input, init);
|
|
38
|
+
if (response.status !== 429 || retryDelayCapMs === 0) return response;
|
|
39
|
+
|
|
40
|
+
const headers = new Headers(response.headers);
|
|
41
|
+
const retryAfterMs = getRetryAfterMsFromHeaders(headers);
|
|
42
|
+
const bodyText = await response
|
|
43
|
+
.clone()
|
|
44
|
+
.text()
|
|
45
|
+
.catch(() => "");
|
|
46
|
+
if (!isOpenAIUsageExhaustionResponse(bodyText, retryAfterMs, retryDelayCapMs)) return response;
|
|
47
|
+
|
|
48
|
+
headers.set("x-should-retry", "false");
|
|
49
|
+
return new Response(bodyText, {
|
|
50
|
+
status: response.status,
|
|
51
|
+
statusText: response.statusText,
|
|
52
|
+
headers,
|
|
53
|
+
});
|
|
54
|
+
},
|
|
55
|
+
baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {},
|
|
56
|
+
);
|
|
57
|
+
}
|
|
@@ -19,6 +19,7 @@ import type {
|
|
|
19
19
|
ToolResultMessage,
|
|
20
20
|
TSchema,
|
|
21
21
|
} from "../types";
|
|
22
|
+
import { sanitizeJsonStrings } from "../utils";
|
|
22
23
|
import {
|
|
23
24
|
type OpenAIChatContentPart,
|
|
24
25
|
type OpenAIChatMessage,
|
|
@@ -453,9 +454,9 @@ function isOnlyRaw(args: Record<string, unknown>): boolean {
|
|
|
453
454
|
|
|
454
455
|
function stringifyArgs(args: Record<string, unknown>): string {
|
|
455
456
|
// `__raw` is our fallback marker for un-parseable inbound args; preserve it verbatim on the way out.
|
|
456
|
-
if (typeof args.__raw === "string" && isOnlyRaw(args)) return args.__raw;
|
|
457
|
+
if (typeof args.__raw === "string" && isOnlyRaw(args)) return args.__raw.toWellFormed();
|
|
457
458
|
try {
|
|
458
|
-
return JSON.stringify(args);
|
|
459
|
+
return JSON.stringify(sanitizeJsonStrings(args));
|
|
459
460
|
} catch {
|
|
460
461
|
return "{}";
|
|
461
462
|
}
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import type { Effort } from "../../model-thinking";
|
|
2
2
|
import { requireSupportedEffort } from "../../model-thinking";
|
|
3
3
|
import type { Api, Model } from "../../types";
|
|
4
|
+
import { sanitizeJsonStrings } from "../../utils";
|
|
4
5
|
|
|
5
6
|
export interface ReasoningConfig {
|
|
6
7
|
effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max";
|
|
@@ -109,6 +110,10 @@ function normalizeInputTextPartFields(input: InputItem[] | undefined): InputItem
|
|
|
109
110
|
}
|
|
110
111
|
if (normalizedItem.type === "message") {
|
|
111
112
|
normalizedItem.content = normalizeTextPartFields(normalizedItem.content, `input[${itemIndex}].content`);
|
|
113
|
+
} else if (normalizedItem.type === "function_call" && "arguments" in itemRecord) {
|
|
114
|
+
itemRecord.arguments = sanitizeJsonStrings(itemRecord.arguments);
|
|
115
|
+
} else if (normalizedItem.type === "custom_tool_call" && typeof itemRecord.input === "string") {
|
|
116
|
+
itemRecord.input = itemRecord.input.toWellFormed();
|
|
112
117
|
}
|
|
113
118
|
return normalizedItem;
|
|
114
119
|
});
|
|
@@ -104,6 +104,10 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
104
104
|
baseUrl.includes("opencode.ai");
|
|
105
105
|
const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
|
|
106
106
|
const isOpenCodeGoReasoning = provider === "opencode-go" && Boolean(model.reasoning);
|
|
107
|
+
const isOpenCodeGoKimiReasoning = provider === "opencode-go" && isKimiModel && Boolean(model.reasoning);
|
|
108
|
+
const isOpenCodeGoKimi25Reasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.5";
|
|
109
|
+
const isOpenCodeGoKimi27CodeReasoning = isOpenCodeGoKimiReasoning && model.id === "kimi-k2.7-code";
|
|
110
|
+
const needsOpenCodeGoKimiEffortMap = isOpenCodeGoKimi25Reasoning || isOpenCodeGoKimi27CodeReasoning;
|
|
107
111
|
|
|
108
112
|
const useMaxTokens =
|
|
109
113
|
provider === "mistral" ||
|
|
@@ -170,22 +174,31 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
170
174
|
xhigh: "default",
|
|
171
175
|
max: "default",
|
|
172
176
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
173
|
-
:
|
|
177
|
+
: needsOpenCodeGoKimiEffortMap
|
|
174
178
|
? ({
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
179
|
+
// Live Go probes (2026-07-06) showed model-specific effort gaps:
|
|
180
|
+
// kimi-k2.5 rejects "minimal", while kimi-k2.7-code rejects
|
|
181
|
+
// OpenAI-style "xhigh" and "max"; all other Kimi efforts tested
|
|
182
|
+
// successfully and should pass through unchanged.
|
|
183
|
+
...(isOpenCodeGoKimi25Reasoning ? { minimal: "low" } : {}),
|
|
184
|
+
...(isOpenCodeGoKimi27CodeReasoning ? { xhigh: "high", max: "high" } : {}),
|
|
181
185
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
182
|
-
:
|
|
186
|
+
: isDeepseekFamily && model.reasoning
|
|
183
187
|
? ({
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
188
|
+
minimal: "high",
|
|
189
|
+
low: "high",
|
|
190
|
+
medium: "high",
|
|
191
|
+
high: "high",
|
|
192
|
+
xhigh: "max",
|
|
193
|
+
max: "max",
|
|
187
194
|
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
188
|
-
:
|
|
195
|
+
: isFireworks
|
|
196
|
+
? ({
|
|
197
|
+
// Fireworks' OpenAI-compatible endpoint rejects OpenAI's
|
|
198
|
+
// `minimal` literal but accepts `none` for the lowest setting.
|
|
199
|
+
minimal: "none",
|
|
200
|
+
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
|
201
|
+
: {};
|
|
189
202
|
|
|
190
203
|
return {
|
|
191
204
|
supportsStore: !isNonStandard,
|
|
@@ -198,7 +211,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
198
211
|
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel || isOpenCodeGoReasoning,
|
|
199
212
|
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
|
|
200
213
|
supportsToolChoice: !isDirectDeepseekReasoning,
|
|
201
|
-
supportsForcedToolChoice:
|
|
214
|
+
supportsForcedToolChoice: !isOpenCodeGoKimiReasoning,
|
|
202
215
|
maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
|
|
203
216
|
requiresToolResultName: isMistral,
|
|
204
217
|
requiresAssistantAfterToolResult: false,
|
|
@@ -22,6 +22,7 @@ import type {
|
|
|
22
22
|
Tool,
|
|
23
23
|
ToolCall,
|
|
24
24
|
} from "../types";
|
|
25
|
+
import { sanitizeJsonStrings } from "../utils";
|
|
25
26
|
import {
|
|
26
27
|
type OpenAIResponsesFunctionCallItem,
|
|
27
28
|
type OpenAIResponsesFunctionCallOutputItem,
|
|
@@ -612,7 +613,8 @@ function buildOutputItems(message: AssistantMessage): OutputItem[] {
|
|
|
612
613
|
} else if (part.type === "toolCall") {
|
|
613
614
|
flushMessage();
|
|
614
615
|
if (part.customWireName) {
|
|
615
|
-
const rawInput =
|
|
616
|
+
const rawInput =
|
|
617
|
+
typeof part.arguments?.input === "string" ? (part.arguments.input as string).toWellFormed() : "";
|
|
616
618
|
out.push({
|
|
617
619
|
type: "custom_tool_call",
|
|
618
620
|
id: part.thoughtSignature ?? makeCustomCallId(),
|
|
@@ -627,7 +629,7 @@ function buildOutputItems(message: AssistantMessage): OutputItem[] {
|
|
|
627
629
|
id: part.thoughtSignature ?? makeFuncCallId(),
|
|
628
630
|
call_id: part.id,
|
|
629
631
|
name: part.name,
|
|
630
|
-
arguments: JSON.stringify(part.arguments ?? {}),
|
|
632
|
+
arguments: JSON.stringify(sanitizeJsonStrings(part.arguments ?? {})),
|
|
631
633
|
status: "completed",
|
|
632
634
|
});
|
|
633
635
|
}
|
|
@@ -1085,7 +1087,8 @@ export function encodeStream(
|
|
|
1085
1087
|
} else {
|
|
1086
1088
|
// Standard JSON tool: arguments object on the skc side, the
|
|
1087
1089
|
// wire wants the JSON string the model emitted (= streamed deltas).
|
|
1088
|
-
const argsJson =
|
|
1090
|
+
const argsJson =
|
|
1091
|
+
cur.argsText.toWellFormed() || JSON.stringify(sanitizeJsonStrings(tc.arguments ?? {}));
|
|
1089
1092
|
cur.argsText = argsJson;
|
|
1090
1093
|
emit("response.function_call_arguments.done", {
|
|
1091
1094
|
item_id: cur.itemId,
|
|
@@ -28,7 +28,7 @@ import {
|
|
|
28
28
|
type ToolCall,
|
|
29
29
|
type ToolResultMessage,
|
|
30
30
|
} from "../types";
|
|
31
|
-
import { normalizeResponsesToolCallId } from "../utils";
|
|
31
|
+
import { normalizeResponsesToolCallId, sanitizeJsonStrings } from "../utils";
|
|
32
32
|
import type { AssistantMessageEventStream } from "../utils/event-stream";
|
|
33
33
|
import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
|
|
34
34
|
import { joinTextWithImagePlaceholder, NON_VISION_IMAGE_PLACEHOLDER, partitionVisionContent } from "./vision-guard";
|
|
@@ -275,7 +275,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
|
|
275
275
|
}
|
|
276
276
|
knownCallIds.add(normalized.callId);
|
|
277
277
|
if (block.customWireName) {
|
|
278
|
-
const rawInput = typeof block.arguments?.input === "string" ? block.arguments.input : "";
|
|
278
|
+
const rawInput = typeof block.arguments?.input === "string" ? block.arguments.input.toWellFormed() : "";
|
|
279
279
|
customCallIds?.add(normalized.callId);
|
|
280
280
|
outputItems.push({
|
|
281
281
|
type: "custom_tool_call",
|
|
@@ -291,7 +291,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
|
|
291
291
|
id: itemId,
|
|
292
292
|
call_id: normalized.callId,
|
|
293
293
|
name: block.name,
|
|
294
|
-
arguments: JSON.stringify(block.arguments),
|
|
294
|
+
arguments: JSON.stringify(sanitizeJsonStrings(block.arguments ?? {})),
|
|
295
295
|
});
|
|
296
296
|
}
|
|
297
297
|
|
|
@@ -71,6 +71,7 @@ import {
|
|
|
71
71
|
resolveGitHubCopilotBaseUrl,
|
|
72
72
|
} from "./github-copilot-headers";
|
|
73
73
|
import { compactGrammarDefinition } from "./grammar";
|
|
74
|
+
import { wrapOpenAIFetchForBoundedRateLimits } from "./openai-bounded-rate-limits";
|
|
74
75
|
import {
|
|
75
76
|
applyOpenAIRequestTransformBody,
|
|
76
77
|
applyOpenAIRequestTransformHeaders,
|
|
@@ -274,6 +275,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
274
275
|
options?.fetch,
|
|
275
276
|
options?.authCredentialType,
|
|
276
277
|
options?.requestMaxRetries,
|
|
278
|
+
options?.maxRetryDelayMs,
|
|
277
279
|
);
|
|
278
280
|
const premiumRequestsTotal = copilotPremiumRequests;
|
|
279
281
|
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
|
|
@@ -400,6 +402,7 @@ function createClient(
|
|
|
400
402
|
fetchOverride?: FetchImpl,
|
|
401
403
|
authCredentialType?: OpenAIResponsesOptions["authCredentialType"],
|
|
402
404
|
requestMaxRetries?: number,
|
|
405
|
+
maxRetryDelayMs?: number,
|
|
403
406
|
): {
|
|
404
407
|
client: OpenAI;
|
|
405
408
|
copilotPremiumRequests: number | undefined;
|
|
@@ -446,8 +449,9 @@ function createClient(
|
|
|
446
449
|
headers["x-client-request-id"] ??= sessionId;
|
|
447
450
|
}
|
|
448
451
|
const baseFetch = fetchOverride ?? fetch;
|
|
452
|
+
const boundedFetch = wrapOpenAIFetchForBoundedRateLimits(baseFetch, maxRetryDelayMs);
|
|
449
453
|
const transformedFetch = wrapFetchForOpenAIRequestTransform(
|
|
450
|
-
|
|
454
|
+
boundedFetch,
|
|
451
455
|
model.requestTransform,
|
|
452
456
|
`Sayknow-CLI/${packageJson.version}`,
|
|
453
457
|
);
|
package/src/rate-limit-utils.ts
CHANGED
|
@@ -36,6 +36,14 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason {
|
|
|
36
36
|
return "MODEL_CAPACITY_EXHAUSTED";
|
|
37
37
|
}
|
|
38
38
|
|
|
39
|
+
if (
|
|
40
|
+
lower.includes("out_of_credits") ||
|
|
41
|
+
lower.includes("request would exceed your account's rate limit") ||
|
|
42
|
+
lower.includes("request would exceed your accounts rate limit")
|
|
43
|
+
) {
|
|
44
|
+
return "QUOTA_EXHAUSTED";
|
|
45
|
+
}
|
|
46
|
+
|
|
39
47
|
if (
|
|
40
48
|
lower.includes("per minute") ||
|
|
41
49
|
lower.includes("rate limit") ||
|
|
@@ -86,8 +94,7 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number {
|
|
|
86
94
|
|
|
87
95
|
/** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */
|
|
88
96
|
const USAGE_LIMIT_PATTERN =
|
|
89
|
-
/usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|resource has been exhausted[^\n]*(?:quota|limit)/i;
|
|
90
|
-
|
|
97
|
+
/usage.?limit|usage_limit_reached|usage_not_included|limit_reached|model.?limit|model_limit_reached|message.?limit|message_limit_reached|limit for this model|quota.?exceeded|out_of_credits|request would exceed your account.?s rate limit|resource has been exhausted[^\n]*(?:quota|limit)/i;
|
|
91
98
|
export function isUsageLimitError(errorMessage: string): boolean {
|
|
92
99
|
return USAGE_LIMIT_PATTERN.test(errorMessage);
|
|
93
100
|
}
|
package/src/utils.ts
CHANGED
|
@@ -11,6 +11,34 @@ export function normalizeSystemPrompts(systemPrompt: readonly string[] | string
|
|
|
11
11
|
return prompts.map(prompt => prompt.toWellFormed()).filter(prompt => prompt.length > 0);
|
|
12
12
|
}
|
|
13
13
|
|
|
14
|
+
export function sanitizeJsonStrings(value: unknown): unknown {
|
|
15
|
+
return sanitizeJsonStringsInner(value, new WeakMap<object, unknown>());
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
function sanitizeJsonStringsInner(value: unknown, seen: WeakMap<object, unknown>): unknown {
|
|
19
|
+
if (typeof value === "string") return value.toWellFormed();
|
|
20
|
+
if (!value || typeof value !== "object") return value;
|
|
21
|
+
|
|
22
|
+
const cached = seen.get(value);
|
|
23
|
+
if (cached !== undefined) return cached;
|
|
24
|
+
|
|
25
|
+
if (Array.isArray(value)) {
|
|
26
|
+
const sanitized: unknown[] = [];
|
|
27
|
+
seen.set(value, sanitized);
|
|
28
|
+
for (const item of value) {
|
|
29
|
+
sanitized.push(sanitizeJsonStringsInner(item, seen));
|
|
30
|
+
}
|
|
31
|
+
return sanitized;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const sanitized: Record<string, unknown> = {};
|
|
35
|
+
seen.set(value, sanitized);
|
|
36
|
+
for (const [key, nestedValue] of Object.entries(value)) {
|
|
37
|
+
sanitized[key.toWellFormed()] = sanitizeJsonStringsInner(nestedValue, seen);
|
|
38
|
+
}
|
|
39
|
+
return sanitized;
|
|
40
|
+
}
|
|
41
|
+
|
|
14
42
|
export function toNumber(value: unknown): number | undefined {
|
|
15
43
|
if (typeof value === "number" && Number.isFinite(value)) return value;
|
|
16
44
|
if (typeof value === "string" && value.trim()) {
|
|
@@ -187,7 +215,11 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
|
|
|
187
215
|
// providerPayload stores raw output items; replay strips fields that are output-only.
|
|
188
216
|
const { id: _id, ...itemWithoutId } = item;
|
|
189
217
|
const sanitizedItem =
|
|
190
|
-
item.type === "computer_call"
|
|
218
|
+
item.type === "computer_call"
|
|
219
|
+
? sanitizeComputerCallForResponsesInput(itemWithoutId)
|
|
220
|
+
: item.type === "image_generation_call"
|
|
221
|
+
? sanitizeImageGenerationCallForResponsesInput(itemWithoutId)
|
|
222
|
+
: itemWithoutId;
|
|
191
223
|
if (typeof item.call_id === "string") {
|
|
192
224
|
sanitizedItem.call_id = normalizeReplayedResponsesHistoryCallId(item.call_id, normalizedCallIds);
|
|
193
225
|
}
|
|
@@ -203,6 +235,22 @@ function sanitizeComputerCallForResponsesInput(item: Record<string, unknown>): R
|
|
|
203
235
|
return inputSafeItem;
|
|
204
236
|
}
|
|
205
237
|
|
|
238
|
+
function sanitizeImageGenerationCallForResponsesInput(item: Record<string, unknown>): Record<string, unknown> {
|
|
239
|
+
// Image generation output items include request-time knobs that are not part of
|
|
240
|
+
// the Responses input replay schema. Replaying them verbatim makes OpenAI-compatible
|
|
241
|
+
// endpoints reject the next turn, e.g. `Unknown parameter: input[n].action`.
|
|
242
|
+
const {
|
|
243
|
+
action: _action,
|
|
244
|
+
background: _background,
|
|
245
|
+
output_format: _outputFormat,
|
|
246
|
+
quality: _quality,
|
|
247
|
+
revised_prompt: _revisedPrompt,
|
|
248
|
+
size: _size,
|
|
249
|
+
...inputSafeItem
|
|
250
|
+
} = item;
|
|
251
|
+
return inputSafeItem;
|
|
252
|
+
}
|
|
253
|
+
|
|
206
254
|
function normalizeReplayedResponsesHistoryCallId(value: string, normalizedValues: Map<string, string>): string {
|
|
207
255
|
const normalized = normalizedValues.get(value);
|
|
208
256
|
if (normalized) return normalized;
|