@sayknow-cli/ai 0.4.7 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -0
- package/dist/types/auth-broker/client.d.ts +2 -1
- package/dist/types/auth-broker/remote-store.d.ts +2 -1
- package/dist/types/auth-broker/types.d.ts +3 -1
- package/dist/types/auth-broker/wire-schemas.d.ts +68 -0
- package/dist/types/auth-storage.d.ts +21 -1
- package/dist/types/provider-models/openai-compat.d.ts +19 -2
- package/dist/types/providers/anthropic.d.ts +11 -1
- package/dist/types/providers/azure-openai-responses.d.ts +6 -1
- package/dist/types/providers/google-auth.d.ts +2 -0
- package/dist/types/providers/google-gemini-headers.d.ts +1 -1
- package/dist/types/providers/google-vertex.d.ts +2 -0
- package/dist/types/providers/openai-codex-responses.d.ts +4 -0
- package/dist/types/providers/openai-completions.d.ts +2 -0
- package/dist/types/providers/openai-responses.d.ts +2 -0
- package/dist/types/providers/register-builtins.d.ts +8 -0
- package/dist/types/providers/transform-messages.d.ts +1 -0
- package/dist/types/types.d.ts +3 -1
- package/dist/types/usage/grok-cli.d.ts +3 -1
- package/dist/types/usage/kimi.d.ts +2 -0
- package/dist/types/utils/anthropic-auth.d.ts +8 -0
- package/dist/types/utils/foundry.d.ts +10 -0
- package/dist/types/utils/http-inspector.d.ts +13 -0
- package/dist/types/utils/idle-iterator.d.ts +3 -2
- package/dist/types/utils/oauth/alibaba-token-plan.d.ts +19 -0
- package/dist/types/utils/oauth/bizrouter.d.ts +1 -0
- package/dist/types/utils/oauth/opengateway.d.ts +1 -0
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +2 -2
- package/src/auth-broker/client.ts +13 -0
- package/src/auth-broker/refresher.ts +1 -0
- package/src/auth-broker/remote-store.ts +25 -0
- package/src/auth-broker/server.ts +10 -2
- package/src/auth-broker/types.ts +4 -0
- package/src/auth-broker/wire-schemas.ts +17 -1
- package/src/auth-storage.ts +234 -45
- package/src/cli.ts +2 -0
- package/src/model-thinking.ts +11 -3
- package/src/models.json +3289 -486
- package/src/provider-models/descriptors.ts +19 -6
- package/src/provider-models/openai-compat.ts +99 -18
- package/src/providers/amazon-bedrock.ts +4 -0
- package/src/providers/anthropic.ts +131 -28
- package/src/providers/azure-openai-responses.ts +16 -3
- package/src/providers/google-auth.ts +13 -2
- package/src/providers/google-gemini-headers.ts +1 -1
- package/src/providers/google-vertex.ts +7 -2
- package/src/providers/openai-anthropic-shim.ts +4 -0
- package/src/providers/openai-codex-responses.ts +52 -10
- package/src/providers/openai-completions-compat.ts +2 -2
- package/src/providers/openai-completions.ts +20 -3
- package/src/providers/openai-responses.ts +17 -10
- package/src/providers/register-builtins.ts +21 -2
- package/src/providers/transform-messages.ts +25 -6
- package/src/stream.ts +3 -1
- package/src/types.ts +9 -2
- package/src/usage/claude.ts +21 -3
- package/src/usage/grok-cli.ts +12 -1
- package/src/usage/kimi.ts +16 -2
- package/src/utils/anthropic-auth.ts +11 -3
- package/src/utils/foundry.ts +12 -2
- package/src/utils/http-inspector.ts +77 -0
- package/src/utils/idle-iterator.ts +20 -7
- package/src/utils/oauth/{alibaba-coding-plan.ts → alibaba-token-plan.ts} +12 -11
- package/src/utils/oauth/bizrouter.ts +15 -0
- package/src/utils/oauth/index.ts +15 -2
- package/src/utils/oauth/opengateway.ts +15 -0
- package/src/utils/oauth/types.ts +3 -1
- package/src/utils/validation.ts +17 -2
- package/src/utils.ts +41 -4
- package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +0 -18
|
@@ -2,7 +2,7 @@ import * as os from "node:os";
|
|
|
2
2
|
import { scheduler } from "node:timers/promises";
|
|
3
3
|
import {
|
|
4
4
|
$env,
|
|
5
|
-
$
|
|
5
|
+
$pickflag,
|
|
6
6
|
asRecord,
|
|
7
7
|
extractHttpStatusFromError,
|
|
8
8
|
fetchWithRetry,
|
|
@@ -98,7 +98,7 @@ export interface OpenAICodexResponsesOptions extends StreamOptions {
|
|
|
98
98
|
serviceTier?: ServiceTier;
|
|
99
99
|
}
|
|
100
100
|
|
|
101
|
-
const CODEX_DEBUG = $
|
|
101
|
+
const CODEX_DEBUG = $pickflag("SKC_OPENAI_CODE_DEBUG", "PI_CODEX_DEBUG");
|
|
102
102
|
const CODEX_MAX_RETRIES = 5;
|
|
103
103
|
const CODEX_RETRY_DELAY_MS = 500;
|
|
104
104
|
const CODEX_WEBSOCKET_CONNECT_TIMEOUT_MS = 10000;
|
|
@@ -133,6 +133,31 @@ const CODEX_WEBSOCKET_FATAL_PATTERNS = ["websocket error:", "websocket closed be
|
|
|
133
133
|
/** Max total time to spend retrying 429s with server-provided delays (5 minutes). */
|
|
134
134
|
const CODEX_RATE_LIMIT_BUDGET_MS = 5 * 60 * 1000;
|
|
135
135
|
|
|
136
|
+
/**
|
|
137
|
+
* Tool names the Codex backend reserves for its own namespaces. Sending a
|
|
138
|
+
* function tool under one of these names is rejected with
|
|
139
|
+
* `Function 'computer.computer' not allowed in namespace 'computer'`.
|
|
140
|
+
* These are renamed on the wire and mapped back on receive so the internal
|
|
141
|
+
* tool name stays canonical everywhere else in the harness.
|
|
142
|
+
*/
|
|
143
|
+
const CODEX_RESERVED_TOOL_WIRE_NAMES: ReadonlyMap<string, string> = new Map([
|
|
144
|
+
["browser", "browser_tool"],
|
|
145
|
+
["computer", "computer_tool"],
|
|
146
|
+
]);
|
|
147
|
+
const CODEX_CANONICAL_TOOL_NAMES: ReadonlyMap<string, string> = new Map(
|
|
148
|
+
Array.from(CODEX_RESERVED_TOOL_WIRE_NAMES, ([canonical, wire]) => [wire, canonical]),
|
|
149
|
+
);
|
|
150
|
+
|
|
151
|
+
/** Maps a canonical tool name to the name Codex accepts on the wire. */
|
|
152
|
+
export function codexToolWireName(name: string): string {
|
|
153
|
+
return CODEX_RESERVED_TOOL_WIRE_NAMES.get(name) ?? name;
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/** Maps a Codex wire tool name back to the canonical harness tool name. */
|
|
157
|
+
export function codexToolCanonicalName(wireName: string): string {
|
|
158
|
+
return CODEX_CANONICAL_TOOL_NAMES.get(wireName) ?? wireName;
|
|
159
|
+
}
|
|
160
|
+
|
|
136
161
|
const CODEX_PROGRESS_EVENT_TYPES = new Set([
|
|
137
162
|
"response.created",
|
|
138
163
|
"response.output_item.added",
|
|
@@ -299,24 +324,34 @@ function parseCodexPositiveInteger(value: string | undefined, fallback: number):
|
|
|
299
324
|
}
|
|
300
325
|
|
|
301
326
|
function isCodexWebSocketEnvEnabled(): boolean {
|
|
302
|
-
return $
|
|
327
|
+
return $pickflag("SKC_OPENAI_CODE_WEBSOCKET", "PI_CODEX_WEBSOCKET");
|
|
303
328
|
}
|
|
304
329
|
|
|
305
330
|
function getCodexWebSocketRetryBudget(options?: Pick<OpenAICodexResponsesOptions, "streamMaxRetries">): number {
|
|
306
331
|
if (options?.streamMaxRetries !== undefined) {
|
|
307
332
|
return resolveRetryBudget(options.streamMaxRetries, CODEX_WEBSOCKET_RETRY_BUDGET);
|
|
308
333
|
}
|
|
309
|
-
return parseCodexNonNegativeInteger(
|
|
334
|
+
return parseCodexNonNegativeInteger(
|
|
335
|
+
$env.SKC_OPENAI_CODE_WEBSOCKET_RETRY_BUDGET ?? $env.PI_CODEX_WEBSOCKET_RETRY_BUDGET,
|
|
336
|
+
CODEX_WEBSOCKET_RETRY_BUDGET,
|
|
337
|
+
);
|
|
310
338
|
}
|
|
311
339
|
|
|
312
340
|
function getCodexWebSocketRetryDelayMs(retry: number): number {
|
|
313
|
-
const baseDelay = parseCodexPositiveInteger(
|
|
341
|
+
const baseDelay = parseCodexPositiveInteger(
|
|
342
|
+
$env.SKC_OPENAI_CODE_WEBSOCKET_RETRY_DELAY_MS ?? $env.PI_CODEX_WEBSOCKET_RETRY_DELAY_MS,
|
|
343
|
+
CODEX_RETRY_DELAY_MS,
|
|
344
|
+
);
|
|
314
345
|
return baseDelay * Math.max(1, retry);
|
|
315
346
|
}
|
|
316
347
|
|
|
317
348
|
function getCodexWebSocketIdleTimeoutMs(overrideMs?: number): number {
|
|
318
349
|
return (
|
|
319
|
-
overrideMs ??
|
|
350
|
+
overrideMs ??
|
|
351
|
+
parseCodexPositiveInteger(
|
|
352
|
+
$env.SKC_OPENAI_CODE_WEBSOCKET_IDLE_TIMEOUT_MS ?? $env.PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS,
|
|
353
|
+
CODEX_WEBSOCKET_IDLE_TIMEOUT_MS,
|
|
354
|
+
)
|
|
320
355
|
);
|
|
321
356
|
}
|
|
322
357
|
|
|
@@ -467,7 +502,7 @@ export function normalizeCodexToolChoice(
|
|
|
467
502
|
: undefined;
|
|
468
503
|
return customTool
|
|
469
504
|
? { type: "custom", name: customTool.customWireName ?? customTool.name }
|
|
470
|
-
: { type: "function", name };
|
|
505
|
+
: { type: "function", name: codexToolWireName(name) };
|
|
471
506
|
};
|
|
472
507
|
if (choice.type === "function") {
|
|
473
508
|
if ("function" in choice && choice.function?.name) {
|
|
@@ -1090,7 +1125,7 @@ function createOutputBlockForItem(item: CodexEventItem): CodexOutputBlock | null
|
|
|
1090
1125
|
return {
|
|
1091
1126
|
type: "toolCall",
|
|
1092
1127
|
id: encodeResponsesToolCallId(item.call_id, item.id),
|
|
1093
|
-
name: item.name,
|
|
1128
|
+
name: codexToolCanonicalName(item.name),
|
|
1094
1129
|
arguments: {},
|
|
1095
1130
|
partialJson: item.arguments || "",
|
|
1096
1131
|
};
|
|
@@ -1348,7 +1383,7 @@ function handleOutputItemDone(
|
|
|
1348
1383
|
const toolCall: ToolCall = {
|
|
1349
1384
|
type: "toolCall",
|
|
1350
1385
|
id,
|
|
1351
|
-
name: item.name,
|
|
1386
|
+
name: codexToolCanonicalName(item.name),
|
|
1352
1387
|
arguments: parseStreamingJson(item.arguments || "{}"),
|
|
1353
1388
|
};
|
|
1354
1389
|
runtime.canSafelyReplayWebsocketOverSse = false;
|
|
@@ -2690,6 +2725,13 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
|
|
|
2690
2725
|
true,
|
|
2691
2726
|
customCallIds,
|
|
2692
2727
|
);
|
|
2728
|
+
for (const item of outputItems) {
|
|
2729
|
+
// Reconstructed (non-raw) history carries canonical tool names; the
|
|
2730
|
+
// wire form has to match the renamed `tools` entries.
|
|
2731
|
+
if (item.type === "function_call" && typeof item.name === "string") {
|
|
2732
|
+
item.name = codexToolWireName(item.name);
|
|
2733
|
+
}
|
|
2734
|
+
}
|
|
2693
2735
|
if (outputItems.length > 0) {
|
|
2694
2736
|
messages.push(...outputItems);
|
|
2695
2737
|
}
|
|
@@ -2776,7 +2818,7 @@ export function convertOpenAICodexResponsesTools(
|
|
|
2776
2818
|
const { schema: parameters, strict: effectiveStrict } = adaptSchemaForStrict(baseParameters, strict);
|
|
2777
2819
|
return {
|
|
2778
2820
|
type: "function",
|
|
2779
|
-
name: tool.name,
|
|
2821
|
+
name: codexToolWireName(tool.name),
|
|
2780
2822
|
description: tool.description || "",
|
|
2781
2823
|
parameters,
|
|
2782
2824
|
...(effectiveStrict && { strict: true }),
|
|
@@ -69,7 +69,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
69
69
|
baseUrl.includes("api.anthropic.com") ||
|
|
70
70
|
/(^|\/)claude[-.]/i.test(model.id) ||
|
|
71
71
|
/(^|\/)anthropic\//i.test(model.id);
|
|
72
|
-
const isAlibaba =
|
|
72
|
+
const isAlibaba = baseUrl.includes("dashscope");
|
|
73
73
|
const isQwen = model.id.toLowerCase().includes("qwen");
|
|
74
74
|
// DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
|
|
75
75
|
// thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
|
|
@@ -244,7 +244,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
244
244
|
requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
|
|
245
245
|
openRouterRouting: undefined,
|
|
246
246
|
vercelGatewayRouting: undefined,
|
|
247
|
-
supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
|
|
247
|
+
supportsStrictMode: detectStrictModeSupport(provider, baseUrl) && !(isDeepseekFamily && isOpenRouter),
|
|
248
248
|
extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
|
|
249
249
|
toolStrictMode: isCerebras ? "all_strict" : "mixed",
|
|
250
250
|
};
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $credentialEnv, $env,
|
|
1
|
+
import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@sayknow-cli/utils";
|
|
2
2
|
import OpenAI from "openai";
|
|
3
3
|
import type {
|
|
4
4
|
ChatCompletionAssistantMessageParam,
|
|
@@ -48,6 +48,7 @@ import {
|
|
|
48
48
|
import {
|
|
49
49
|
createWatchdog,
|
|
50
50
|
getOpenAIStreamIdleTimeoutMs,
|
|
51
|
+
getProviderFirstEventTimeoutFallbackMs,
|
|
51
52
|
getStreamFirstEventTimeoutMs,
|
|
52
53
|
iterateWithIdleTimeout,
|
|
53
54
|
} from "../utils/idle-iterator";
|
|
@@ -99,7 +100,9 @@ function resolveOpenAIProviderBaseUrl(
|
|
|
99
100
|
authCredentialType: "api_key" | "oauth" | undefined,
|
|
100
101
|
): string {
|
|
101
102
|
if (authCredentialType === "oauth") return OPENAI_DEFAULT_BASE_URL;
|
|
102
|
-
|
|
103
|
+
// Trusted sources only: this base URL becomes the request endpoint that carries
|
|
104
|
+
// the OpenAI credential, and `$env` merges the caller's `cwd/.env`.
|
|
105
|
+
const envBaseUrl = $credentialEnv("OPENAI_BASE_URL");
|
|
103
106
|
const configuredBaseUrl = baseUrl?.trim();
|
|
104
107
|
if (envBaseUrl && (!configuredBaseUrl || isDefaultOpenAIBaseUrl(configuredBaseUrl))) {
|
|
105
108
|
return envBaseUrl;
|
|
@@ -107,6 +110,14 @@ function resolveOpenAIProviderBaseUrl(
|
|
|
107
110
|
return configuredBaseUrl || envBaseUrl || OPENAI_DEFAULT_BASE_URL;
|
|
108
111
|
}
|
|
109
112
|
|
|
113
|
+
/** Test seam: the provider base URL as resolved from trusted env. */
|
|
114
|
+
export function resolveOpenAICompletionsBaseUrlForTest(
|
|
115
|
+
baseUrl: string | undefined,
|
|
116
|
+
authCredentialType: "api_key" | "oauth" | undefined,
|
|
117
|
+
): string {
|
|
118
|
+
return resolveOpenAIProviderBaseUrl(baseUrl, authCredentialType);
|
|
119
|
+
}
|
|
120
|
+
|
|
110
121
|
/**
|
|
111
122
|
* Normalize tool call ID for Mistral.
|
|
112
123
|
* Mistral requires tool IDs to be exactly 9 alphanumeric characters (a-z, A-Z, 0-9).
|
|
@@ -411,6 +422,8 @@ function getTrailingPartialDeepseekToken(text: string): string {
|
|
|
411
422
|
return tail;
|
|
412
423
|
}
|
|
413
424
|
|
|
425
|
+
const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
|
|
426
|
+
|
|
414
427
|
const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
|
|
415
428
|
"OpenAI completions stream timed out while waiting for the first event";
|
|
416
429
|
|
|
@@ -537,8 +550,12 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
537
550
|
openaiStream = await createCompletionsStream("none");
|
|
538
551
|
}
|
|
539
552
|
}
|
|
553
|
+
const firstEventFallbackMs =
|
|
554
|
+
model.provider === "alibaba-token-plan"
|
|
555
|
+
? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS
|
|
556
|
+
: getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
540
557
|
const firstEventWatchdog = createWatchdog(
|
|
541
|
-
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
|
558
|
+
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
542
559
|
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
543
560
|
);
|
|
544
561
|
if (premiumRequestsTotal !== undefined) {
|
|
@@ -1,11 +1,4 @@
|
|
|
1
|
-
import {
|
|
2
|
-
$credentialEnv,
|
|
3
|
-
$env,
|
|
4
|
-
$inheritedEnv,
|
|
5
|
-
extractHttpStatusFromError,
|
|
6
|
-
logger,
|
|
7
|
-
structuredCloneJSON,
|
|
8
|
-
} from "@sayknow-cli/utils";
|
|
1
|
+
import { $credentialEnv, extractHttpStatusFromError, logger, structuredCloneJSON } from "@sayknow-cli/utils";
|
|
9
2
|
import OpenAI from "openai";
|
|
10
3
|
import type {
|
|
11
4
|
Tool as OpenAITool,
|
|
@@ -130,6 +123,7 @@ export interface OpenAIResponsesOptions extends StreamOptions {
|
|
|
130
123
|
}
|
|
131
124
|
|
|
132
125
|
const OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX = "openai-responses:";
|
|
126
|
+
const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
|
|
133
127
|
const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
|
|
134
128
|
"OpenAI responses stream timed out while waiting for the first event";
|
|
135
129
|
const OPENAI_DEFAULT_BASE_URL = "https://api.openai.com/v1";
|
|
@@ -158,7 +152,10 @@ function resolveOpenAIProviderBaseUrl(
|
|
|
158
152
|
authCredentialType: "api_key" | "oauth" | undefined,
|
|
159
153
|
): string {
|
|
160
154
|
if (authCredentialType === "oauth") return OPENAI_DEFAULT_BASE_URL;
|
|
161
|
-
|
|
155
|
+
// Trusted sources only: this base URL becomes the request endpoint that carries
|
|
156
|
+
// the OpenAI credential, and `$env` merges the caller's `cwd/.env`, so reading it
|
|
157
|
+
// there would let repository content redirect authenticated traffic.
|
|
158
|
+
const envBaseUrl = $credentialEnv("OPENAI_BASE_URL");
|
|
162
159
|
const configuredBaseUrl = baseUrl?.trim();
|
|
163
160
|
if (envBaseUrl && (!configuredBaseUrl || isDefaultOpenAIBaseUrl(configuredBaseUrl))) {
|
|
164
161
|
return envBaseUrl;
|
|
@@ -166,6 +163,14 @@ function resolveOpenAIProviderBaseUrl(
|
|
|
166
163
|
return configuredBaseUrl || envBaseUrl || OPENAI_DEFAULT_BASE_URL;
|
|
167
164
|
}
|
|
168
165
|
|
|
166
|
+
/** Test seam: the provider base URL as resolved from trusted env. */
|
|
167
|
+
export function resolveOpenAIProviderBaseUrlForTest(
|
|
168
|
+
baseUrl: string | undefined,
|
|
169
|
+
authCredentialType: "api_key" | "oauth" | undefined,
|
|
170
|
+
): string {
|
|
171
|
+
return resolveOpenAIProviderBaseUrl(baseUrl, authCredentialType);
|
|
172
|
+
}
|
|
173
|
+
|
|
169
174
|
const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
|
|
170
175
|
"response.created",
|
|
171
176
|
"response.output_item.added",
|
|
@@ -330,8 +335,10 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
330
335
|
await notifyProviderResponse(options, response, model, request_id);
|
|
331
336
|
return data;
|
|
332
337
|
});
|
|
338
|
+
const firstEventFallbackMs =
|
|
339
|
+
model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
|
|
333
340
|
const firstEventWatchdog = createWatchdog(
|
|
334
|
-
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
|
341
|
+
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
335
342
|
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
336
343
|
);
|
|
337
344
|
if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
|
|
@@ -190,6 +190,22 @@ interface LazyStreamLimits {
|
|
|
190
190
|
const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
|
|
191
191
|
defaultFirstEventTimeoutMs: 300_000,
|
|
192
192
|
};
|
|
193
|
+
const SLOW_FIRST_EVENT_PROVIDERS = new Set(["alibaba-token-plan", "kimi-code"]);
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
|
|
197
|
+
* A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
|
|
198
|
+
* otherwise providers known to have slow first events get a five-minute floor
|
|
199
|
+
* matching their inner provider-level override. Returns `undefined` for
|
|
200
|
+
* providers that should use the shared default.
|
|
201
|
+
*/
|
|
202
|
+
export function resolveLazyStreamFirstEventFallbackMs(
|
|
203
|
+
provider: string,
|
|
204
|
+
configuredFallbackMs?: number,
|
|
205
|
+
): number | undefined {
|
|
206
|
+
if (configuredFallbackMs !== undefined) return configuredFallbackMs;
|
|
207
|
+
return SLOW_FIRST_EVENT_PROVIDERS.has(provider) ? 300_000 : undefined;
|
|
208
|
+
}
|
|
193
209
|
|
|
194
210
|
function forwardStream<TApi extends Api>(
|
|
195
211
|
target: EventStreamImpl,
|
|
@@ -202,11 +218,14 @@ function forwardStream<TApi extends Api>(
|
|
|
202
218
|
(async () => {
|
|
203
219
|
try {
|
|
204
220
|
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
|
|
221
|
+
const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
|
|
222
|
+
model.provider,
|
|
223
|
+
limits?.defaultFirstEventTimeoutMs,
|
|
224
|
+
);
|
|
205
225
|
const watchedSource = iterateWithIdleTimeout(source, {
|
|
206
226
|
idleTimeoutMs,
|
|
207
227
|
firstItemTimeoutMs:
|
|
208
|
-
options.streamFirstEventTimeoutMs ??
|
|
209
|
-
getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs),
|
|
228
|
+
options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
210
229
|
errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
|
|
211
230
|
firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
|
|
212
231
|
onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
|
|
@@ -31,7 +31,7 @@ export function transformMessages<TApi extends Api>(
|
|
|
31
31
|
messages: Message[],
|
|
32
32
|
model: Model<TApi>,
|
|
33
33
|
normalizeToolCallId?: (id: string, model: Model<TApi>, source: AssistantMessage) => string,
|
|
34
|
-
options?: { repairLatestAssistantThinking?: boolean },
|
|
34
|
+
options?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
|
|
35
35
|
): Message[] {
|
|
36
36
|
// Build a map of original tool call IDs to normalized IDs
|
|
37
37
|
const toolCallIdMap = new Map<string, string>();
|
|
@@ -73,16 +73,29 @@ export function transformMessages<TApi extends Api>(
|
|
|
73
73
|
// are kept so the second pass can either preserve real results or synthesize
|
|
74
74
|
// an explicit aborted result without leaving dangling tool_use blocks.
|
|
75
75
|
const hasPartialThinking = assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error";
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
76
|
+
// One-shot Anthropic replay repair. `repairLatestAssistantThinking` targets the
|
|
77
|
+
// "latest assistant message ... cannot be modified" 400; `repairAllAssistantThinking`
|
|
78
|
+
// targets the "Invalid `signature` in `thinking` block" 400, which can cite a block
|
|
79
|
+
// anywhere in the replayed history (e.g. after compaction/pruning rewrote an earlier
|
|
80
|
+
// turn), so the drop must apply to every assistant message. Within each
|
|
81
|
+
// message only blocks that would replay as native thinking/redacted_thinking
|
|
82
|
+
// are dropped; cross-model reasoning degrades to text and is preserved.
|
|
83
|
+
const dropAssistantThinkingForRepair =
|
|
84
|
+
(options?.repairAllAssistantThinking === true ||
|
|
85
|
+
(options?.repairLatestAssistantThinking === true && index === latestAssistantIndex)) &&
|
|
79
86
|
model.api === "anthropic-messages" &&
|
|
80
87
|
assistantMsg.api === "anthropic-messages";
|
|
81
88
|
|
|
82
89
|
const transformedContent = assistantMsg.content.flatMap(block => {
|
|
83
90
|
if (block.type === "thinking") {
|
|
84
|
-
if (hasPartialThinking
|
|
91
|
+
if (hasPartialThinking) return [];
|
|
85
92
|
const sanitized = block;
|
|
93
|
+
// Repair must only drop blocks that would otherwise replay as native
|
|
94
|
+
// thinking. Cross-model/provider reasoning degrades to unsigned text
|
|
95
|
+
// below and was never replayed as a signed block, so it cannot be the
|
|
96
|
+
// signature failure — dropping it would silently lose valid context.
|
|
97
|
+
const replaysAsNativeThinking = mustPreserveLatestAnthropicThinking || isSameModel;
|
|
98
|
+
if (dropAssistantThinkingForRepair && replaysAsNativeThinking) return [];
|
|
86
99
|
if (mustPreserveLatestAnthropicThinking) return sanitized;
|
|
87
100
|
// For same model: keep thinking blocks with signatures (needed for replay)
|
|
88
101
|
// even if the thinking text is empty (OpenAI encrypted reasoning)
|
|
@@ -97,7 +110,13 @@ export function transformMessages<TApi extends Api>(
|
|
|
97
110
|
}
|
|
98
111
|
|
|
99
112
|
if (block.type === "redactedThinking") {
|
|
100
|
-
if (hasPartialThinking
|
|
113
|
+
if (hasPartialThinking) return [];
|
|
114
|
+
// Same restriction as thinking blocks: cross-model/provider redacted
|
|
115
|
+
// blocks already drop below, so repair only needs to cover blocks that
|
|
116
|
+
// would replay as native redacted_thinking.
|
|
117
|
+
if (dropAssistantThinkingForRepair && (mustPreserveLatestAnthropicThinking || isSameModel)) {
|
|
118
|
+
return [];
|
|
119
|
+
}
|
|
101
120
|
if (mustPreserveLatestAnthropicThinking) return block;
|
|
102
121
|
if (isSameModel) return block;
|
|
103
122
|
return [];
|
package/src/stream.ts
CHANGED
|
@@ -76,7 +76,7 @@ function hasVertexAdcCredentials(): boolean {
|
|
|
76
76
|
type KeyResolver = string | (() => string | undefined);
|
|
77
77
|
|
|
78
78
|
const serviceProviderMap: Record<string, KeyResolver> = {
|
|
79
|
-
"alibaba-
|
|
79
|
+
"alibaba-token-plan": "ALIBABA_TOKEN_PLAN_API_KEY",
|
|
80
80
|
openai: () => $credentialEnv("OPENAI_API_KEY"),
|
|
81
81
|
google: "GEMINI_API_KEY",
|
|
82
82
|
groq: "GROQ_API_KEY",
|
|
@@ -169,6 +169,8 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
169
169
|
"qwen-portal": () => $pickCredentialEnv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
|
|
170
170
|
together: "TOGETHER_API_KEY",
|
|
171
171
|
zenmux: "ZENMUX_API_KEY",
|
|
172
|
+
opengateway: "OPENGATEWAY_API_KEY",
|
|
173
|
+
bizrouter: "BIZROUTER_API_KEY",
|
|
172
174
|
venice: "VENICE_API_KEY",
|
|
173
175
|
vllm: "VLLM_API_KEY",
|
|
174
176
|
xiaomi: "XIAOMI_API_KEY",
|
package/src/types.ts
CHANGED
|
@@ -114,7 +114,7 @@ export interface ThinkingConfig {
|
|
|
114
114
|
}
|
|
115
115
|
|
|
116
116
|
export type KnownProvider =
|
|
117
|
-
| "alibaba-
|
|
117
|
+
| "alibaba-token-plan"
|
|
118
118
|
| "amazon-bedrock"
|
|
119
119
|
| "azure-openai"
|
|
120
120
|
| "anthropic"
|
|
@@ -147,6 +147,8 @@ export type KnownProvider =
|
|
|
147
147
|
| "minimax"
|
|
148
148
|
| "opencode-go"
|
|
149
149
|
| "opencode-zen"
|
|
150
|
+
| "opengateway"
|
|
151
|
+
| "bizrouter"
|
|
150
152
|
| "synthetic"
|
|
151
153
|
| "cloudflare-ai-gateway"
|
|
152
154
|
| "huggingface"
|
|
@@ -709,10 +711,15 @@ export type TSchema = ZodType | TJsonSchema;
|
|
|
709
711
|
/** Resolve parameter types for tool execution / handlers. */
|
|
710
712
|
export type Static<S> = S extends ZodType ? z.infer<S> : S extends { static: infer T } ? T : unknown;
|
|
711
713
|
|
|
714
|
+
export type RawArgumentRejectionCode =
|
|
715
|
+
| "ask-intent-review-requires-positive-round"
|
|
716
|
+
| "ask-intent-contract-requires-non-empty-authority"
|
|
717
|
+
| "ask-deep-interview-metadata-requires-deep-interview-gate";
|
|
718
|
+
|
|
712
719
|
export type RawArgumentValidationResult =
|
|
713
720
|
| { outcome: "passthrough" }
|
|
714
721
|
| { outcome: "accept"; arguments: ToolCall["arguments"] }
|
|
715
|
-
| { outcome: "reject" };
|
|
722
|
+
| { outcome: "reject"; code?: RawArgumentRejectionCode };
|
|
716
723
|
|
|
717
724
|
export interface Tool<TParameters extends TSchema = TSchema> {
|
|
718
725
|
name: string;
|
package/src/usage/claude.ts
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { scheduler } from "node:timers/promises";
|
|
2
|
+
import { claudeCodeVersion } from "../providers/anthropic";
|
|
2
3
|
import type {
|
|
3
4
|
CredentialRankingStrategy,
|
|
4
5
|
UsageAmount,
|
|
@@ -17,6 +18,13 @@ const FIVE_HOURS_MS = 5 * 60 * 60 * 1000;
|
|
|
17
18
|
const SEVEN_DAYS_MS = 7 * 24 * 60 * 60 * 1000;
|
|
18
19
|
const MAX_ATTEMPTS = 3;
|
|
19
20
|
const BASE_RETRY_DELAY_MS = 500;
|
|
21
|
+
/**
|
|
22
|
+
* Ceiling for a server-supplied `Retry-After`. Matches `OPENAI_RETRY_DELAY_CAP_MS`
|
|
23
|
+
* and `fetchWithRetry`'s `DEFAULT_MAX_DELAY_MS`. Without it a hostile or
|
|
24
|
+
* misconfigured endpoint stalls the usage fetch for as long as it likes
|
|
25
|
+
* (`Retry-After: 86400` previously produced a 24h sleep).
|
|
26
|
+
*/
|
|
27
|
+
const MAX_RETRY_DELAY_MS = 60_000;
|
|
20
28
|
|
|
21
29
|
const CLAUDE_HEADERS = {
|
|
22
30
|
accept: "application/json, text/plain, */*",
|
|
@@ -24,7 +32,7 @@ const CLAUDE_HEADERS = {
|
|
|
24
32
|
"anthropic-beta":
|
|
25
33
|
"claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05",
|
|
26
34
|
"content-type": "application/json",
|
|
27
|
-
"user-agent":
|
|
35
|
+
"user-agent": `claude-cli/${claudeCodeVersion} (external, cli)`,
|
|
28
36
|
connection: "keep-alive",
|
|
29
37
|
} as const;
|
|
30
38
|
|
|
@@ -140,13 +148,23 @@ function isAbortError(error: unknown, signal?: AbortSignal): boolean {
|
|
|
140
148
|
return error.name === "AbortError" || error.name === "TimeoutError";
|
|
141
149
|
}
|
|
142
150
|
|
|
151
|
+
/**
|
|
152
|
+
* Honour the server hint but never exceed `MAX_RETRY_DELAY_MS`, and never
|
|
153
|
+
* return a negative/non-finite delay. Keeps the sleep bounded so an abort has
|
|
154
|
+
* an upper bound to fire within.
|
|
155
|
+
*/
|
|
156
|
+
function clampRetryDelay(baseline: number, hintMs: number): number {
|
|
157
|
+
const hint = Number.isFinite(hintMs) ? Math.max(0, hintMs) : 0;
|
|
158
|
+
return Math.min(Math.max(baseline, hint), MAX_RETRY_DELAY_MS);
|
|
159
|
+
}
|
|
160
|
+
|
|
143
161
|
function retryDelayMs(attempt: number, retryAfter: string | null): number {
|
|
144
162
|
const baseline = BASE_RETRY_DELAY_MS * 2 ** attempt;
|
|
145
163
|
if (!retryAfter?.trim()) return baseline;
|
|
146
164
|
const seconds = Number.parseFloat(retryAfter);
|
|
147
|
-
if (Number.isFinite(seconds)) return
|
|
165
|
+
if (Number.isFinite(seconds)) return clampRetryDelay(baseline, seconds * 1000);
|
|
148
166
|
const dateDelay = Date.parse(retryAfter) - Date.now();
|
|
149
|
-
return Number.isFinite(dateDelay) ?
|
|
167
|
+
return Number.isFinite(dateDelay) ? clampRetryDelay(baseline, dateDelay) : baseline;
|
|
150
168
|
}
|
|
151
169
|
|
|
152
170
|
async function waitBeforeRetry(
|
package/src/usage/grok-cli.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { $credentialEnv } from "@sayknow-cli/utils";
|
|
1
2
|
import type {
|
|
2
3
|
CredentialRankingStrategy,
|
|
3
4
|
UsageFetchContext,
|
|
@@ -64,10 +65,20 @@ function isUnsafeGrokBaseUrlOverride(baseUrl?: string): boolean {
|
|
|
64
65
|
}
|
|
65
66
|
|
|
66
67
|
function resolveAccessToken(params: UsageFetchParams): string | undefined {
|
|
67
|
-
|
|
68
|
+
// Trusted sources only for the env fallback: this token authenticates the
|
|
69
|
+
// billing/usage call, so whatever can set it decides which account is queried
|
|
70
|
+
// with it. `$env` merges the caller's `cwd/.env` into `process.env`, so
|
|
71
|
+
// reading it there would let repository content supply the credential.
|
|
72
|
+
// Stored credentials keep precedence.
|
|
73
|
+
const token = params.credential.accessToken ?? params.credential.apiKey ?? $credentialEnv("GROK_CLI_OAUTH_TOKEN");
|
|
68
74
|
return token?.trim() || undefined;
|
|
69
75
|
}
|
|
70
76
|
|
|
77
|
+
/** Test seam: the usage access token as resolved from a credential plus trusted env. */
|
|
78
|
+
export function resolveGrokAccessTokenForTest(params: UsageFetchParams): string | undefined {
|
|
79
|
+
return resolveAccessToken(params);
|
|
80
|
+
}
|
|
81
|
+
|
|
71
82
|
function buildMonthlyUsageLimit(usage: BillingUsage, nowMs: number): UsageLimit {
|
|
72
83
|
const usedFraction = usage.monthlyLimit > 0 ? usage.used / usage.monthlyLimit : 0;
|
|
73
84
|
const percent = usedFraction * 100;
|
package/src/usage/kimi.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $
|
|
1
|
+
import { $credentialEnv } from "@sayknow-cli/utils";
|
|
2
2
|
import type {
|
|
3
3
|
UsageAmount,
|
|
4
4
|
UsageFetchContext,
|
|
@@ -31,12 +31,26 @@ type KimiUsageRow = {
|
|
|
31
31
|
window?: UsageWindow;
|
|
32
32
|
};
|
|
33
33
|
|
|
34
|
+
/**
|
|
35
|
+
* Usage endpoint base, with the environment override resolved from trusted
|
|
36
|
+
* sources only.
|
|
37
|
+
*
|
|
38
|
+
* The result becomes the usage URL that the request sends
|
|
39
|
+
* `Authorization: Bearer <accessToken>` to, so whatever can set it receives the
|
|
40
|
+
* user's Kimi access token. `$env` merges the caller's `cwd/.env`, so reading it
|
|
41
|
+
* there would let repository content collect that token.
|
|
42
|
+
*/
|
|
34
43
|
function normalizeBaseUrl(baseUrl?: string): string {
|
|
35
|
-
const envBase = $
|
|
44
|
+
const envBase = $credentialEnv("KIMI_CODE_BASE_URL");
|
|
36
45
|
const candidate = baseUrl?.trim() || envBase || DEFAULT_BASE_URL;
|
|
37
46
|
return candidate.replace(/\/+$/, "");
|
|
38
47
|
}
|
|
39
48
|
|
|
49
|
+
/** Test seam: the usage base URL as resolved from a caller value plus trusted env. */
|
|
50
|
+
export function normalizeKimiUsageBaseUrlForTest(baseUrl?: string): string {
|
|
51
|
+
return normalizeBaseUrl(baseUrl);
|
|
52
|
+
}
|
|
53
|
+
|
|
40
54
|
function buildUsageUrl(baseUrl: string): string {
|
|
41
55
|
const normalized = baseUrl.endsWith("/") ? baseUrl : `${baseUrl}/`;
|
|
42
56
|
return `${normalized}${USAGE_PATH}`;
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* `authStorage.getApiKey("anthropic", sessionId)` first, then pass the result
|
|
9
9
|
* through {@link buildAnthropicAuthConfig} for header/URL shaping.
|
|
10
10
|
*/
|
|
11
|
-
import { $
|
|
11
|
+
import { $credentialEnv } from "@sayknow-cli/utils";
|
|
12
12
|
import {
|
|
13
13
|
buildAnthropicHeaders as buildProviderAnthropicHeaders,
|
|
14
14
|
normalizeAnthropicBaseUrl,
|
|
@@ -29,12 +29,20 @@ function normalizeBaseUrl(baseUrl: string | undefined): string | undefined {
|
|
|
29
29
|
return trimmed ? trimmed.replace(/\/+$/, "") : undefined;
|
|
30
30
|
}
|
|
31
31
|
|
|
32
|
+
/**
|
|
33
|
+
* Resolve the Anthropic base URL from the environment.
|
|
34
|
+
*
|
|
35
|
+
* Trusted sources only: the result becomes the request URL that carries the
|
|
36
|
+
* Anthropic API key / OAuth token, so whatever can set it can redirect
|
|
37
|
+
* authenticated traffic. `$env` merges the caller's `cwd/.env`, so reading it
|
|
38
|
+
* there would let repository content choose where credentials are sent.
|
|
39
|
+
*/
|
|
32
40
|
export function resolveAnthropicBaseUrlFromEnv(): string | undefined {
|
|
33
41
|
if (isFoundryEnabled()) {
|
|
34
|
-
const foundryBaseUrl = normalizeBaseUrl($
|
|
42
|
+
const foundryBaseUrl = normalizeBaseUrl($credentialEnv("FOUNDRY_BASE_URL"));
|
|
35
43
|
if (foundryBaseUrl) return foundryBaseUrl;
|
|
36
44
|
}
|
|
37
|
-
const anthropicBaseUrl = normalizeBaseUrl($
|
|
45
|
+
const anthropicBaseUrl = normalizeBaseUrl($credentialEnv("ANTHROPIC_BASE_URL"));
|
|
38
46
|
return anthropicBaseUrl || undefined;
|
|
39
47
|
}
|
|
40
48
|
|
package/src/utils/foundry.ts
CHANGED
|
@@ -1,7 +1,17 @@
|
|
|
1
|
-
import { $
|
|
1
|
+
import { $credentialEnv } from "@sayknow-cli/utils";
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* Whether Anthropic requests run in Foundry gateway mode.
|
|
5
|
+
*
|
|
6
|
+
* Resolved from trusted environment sources only. Enabling Foundry switches the
|
|
7
|
+
* request base URL and injects TLS client material, so whatever can set this
|
|
8
|
+
* redirects authenticated traffic. `$env` merges the caller's `cwd/.env`, so
|
|
9
|
+
* reading it there would let repository content flip the mode; resolve it the
|
|
10
|
+
* same way the credentials themselves are (launching shell plus SKC/user-owned
|
|
11
|
+
* `.env` files, never the project `.env`).
|
|
12
|
+
*/
|
|
3
13
|
export function isFoundryEnabled(): boolean {
|
|
4
|
-
const value = $
|
|
14
|
+
const value = $credentialEnv("CLAUDE_CODE_USE_FOUNDRY");
|
|
5
15
|
if (!value) return false;
|
|
6
16
|
const normalized = value.trim().toLowerCase();
|
|
7
17
|
return normalized === "1" || normalized === "true" || normalized === "yes" || normalized === "on";
|