@gajae-code/ai 0.11.6 → 0.11.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/dist/types/auth-broker/client.d.ts +2 -1
- package/dist/types/auth-broker/remote-store.d.ts +2 -1
- package/dist/types/auth-broker/types.d.ts +3 -1
- package/dist/types/auth-broker/wire-schemas.d.ts +68 -0
- package/dist/types/auth-storage.d.ts +21 -1
- package/dist/types/provider-models/openai-compat.d.ts +2 -2
- package/dist/types/providers/anthropic.d.ts +9 -0
- package/dist/types/providers/transform-messages.d.ts +1 -0
- package/dist/types/types.d.ts +3 -1
- package/dist/types/utils/oauth/alibaba-token-plan.d.ts +19 -0
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +2 -2
- package/src/auth-broker/client.ts +13 -0
- package/src/auth-broker/refresher.ts +1 -0
- package/src/auth-broker/remote-store.ts +25 -0
- package/src/auth-broker/server.ts +10 -2
- package/src/auth-broker/types.ts +4 -0
- package/src/auth-broker/wire-schemas.ts +17 -1
- package/src/auth-storage.ts +199 -30
- package/src/model-thinking.ts +11 -3
- package/src/models.json +5362 -1219
- package/src/provider-models/descriptors.ts +5 -6
- package/src/provider-models/openai-compat.ts +12 -12
- package/src/providers/amazon-bedrock.ts +4 -0
- package/src/providers/anthropic.ts +78 -22
- package/src/providers/openai-completions-compat.ts +2 -2
- package/src/providers/openai-completions.ts +4 -1
- package/src/providers/openai-responses.ts +4 -1
- package/src/providers/transform-messages.ts +25 -6
- package/src/stream.ts +1 -1
- package/src/types.ts +7 -2
- package/src/utils/oauth/{alibaba-coding-plan.ts → alibaba-token-plan.ts} +12 -11
- package/src/utils/oauth/index.ts +3 -2
- package/src/utils/oauth/types.ts +1 -1
- package/src/utils/validation.ts +17 -2
- package/src/utils.ts +41 -4
- package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +0 -18
|
@@ -9,7 +9,7 @@ import type { OAuthProvider } from "../utils/oauth/types";
|
|
|
9
9
|
import { googleModelManagerOptions } from "./google";
|
|
10
10
|
import { ollamaCloudModelManagerOptions } from "./ollama";
|
|
11
11
|
import {
|
|
12
|
-
|
|
12
|
+
alibabaTokenPlanModelManagerOptions,
|
|
13
13
|
anthropicModelManagerOptions,
|
|
14
14
|
cerebrasModelManagerOptions,
|
|
15
15
|
cloudflareAiGatewayModelManagerOptions,
|
|
@@ -130,10 +130,10 @@ function catalogDescriptor(
|
|
|
130
130
|
export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
|
|
131
131
|
descriptor("anthropic", "claude-sonnet-5", config => anthropicModelManagerOptions(config)),
|
|
132
132
|
catalogDescriptor(
|
|
133
|
-
"alibaba-
|
|
134
|
-
"
|
|
135
|
-
config =>
|
|
136
|
-
catalog("Alibaba
|
|
133
|
+
"alibaba-token-plan",
|
|
134
|
+
"deepseek-v4-pro",
|
|
135
|
+
config => alibabaTokenPlanModelManagerOptions(config),
|
|
136
|
+
catalog("Alibaba Token Plan", ["ALIBABA_TOKEN_PLAN_API_KEY"], { oauthProvider: "alibaba-token-plan" }),
|
|
137
137
|
),
|
|
138
138
|
descriptor("openai", "gpt-5.4", config => openaiModelManagerOptions(config)),
|
|
139
139
|
descriptor("groq", "openai/gpt-oss-120b", config => groqModelManagerOptions(config)),
|
|
@@ -334,7 +334,6 @@ export const DEFAULT_MODEL_PER_PROVIDER: Record<KnownProvider, string> = {
|
|
|
334
334
|
...Object.fromEntries(PROVIDER_DESCRIPTORS.map(d => [d.providerId, d.defaultModel])),
|
|
335
335
|
// Providers not in PROVIDER_DESCRIPTORS (special auth or no standard discovery)
|
|
336
336
|
"azure-openai": "gpt-4.1",
|
|
337
|
-
"alibaba-coding-plan": "qwen3.5-plus",
|
|
338
337
|
"amazon-bedrock": "us.anthropic.claude-opus-4-6-v1",
|
|
339
338
|
"google-antigravity": "gemini-3-pro-high",
|
|
340
339
|
"google-gemini-cli": "gemini-2.5-pro",
|
|
@@ -1124,26 +1124,26 @@ export function kiloModelManagerOptions(config?: KiloModelManagerConfig): ModelM
|
|
|
1124
1124
|
}
|
|
1125
1125
|
|
|
1126
1126
|
// ---------------------------------------------------------------------------
|
|
1127
|
-
// Alibaba
|
|
1127
|
+
// Alibaba Token Plan
|
|
1128
1128
|
// ---------------------------------------------------------------------------
|
|
1129
1129
|
|
|
1130
|
-
export interface
|
|
1130
|
+
export interface AlibabaTokenPlanModelManagerConfig {
|
|
1131
1131
|
apiKey?: string;
|
|
1132
1132
|
baseUrl?: string;
|
|
1133
1133
|
}
|
|
1134
1134
|
|
|
1135
|
-
export function
|
|
1136
|
-
config?:
|
|
1135
|
+
export function alibabaTokenPlanModelManagerOptions(
|
|
1136
|
+
config?: AlibabaTokenPlanModelManagerConfig,
|
|
1137
1137
|
): ModelManagerOptions<"openai-completions"> {
|
|
1138
1138
|
const apiKey = config?.apiKey;
|
|
1139
|
-
const baseUrl = config?.baseUrl ?? "https://
|
|
1140
|
-
const references = createBundledReferenceMap<"openai-completions">("alibaba-
|
|
1139
|
+
const baseUrl = config?.baseUrl ?? "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1";
|
|
1140
|
+
const references = createBundledReferenceMap<"openai-completions">("alibaba-token-plan");
|
|
1141
1141
|
return {
|
|
1142
|
-
providerId: "alibaba-
|
|
1142
|
+
providerId: "alibaba-token-plan",
|
|
1143
1143
|
fetchDynamicModels: () =>
|
|
1144
1144
|
fetchOpenAICompatibleModels({
|
|
1145
1145
|
api: "openai-completions",
|
|
1146
|
-
provider: "alibaba-
|
|
1146
|
+
provider: "alibaba-token-plan",
|
|
1147
1147
|
baseUrl,
|
|
1148
1148
|
apiKey,
|
|
1149
1149
|
mapModel: (entry, defaults) => {
|
|
@@ -2356,11 +2356,11 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDe
|
|
|
2356
2356
|
reasoningContentField: "reasoning_content",
|
|
2357
2357
|
},
|
|
2358
2358
|
}),
|
|
2359
|
-
// --- Alibaba
|
|
2359
|
+
// --- Alibaba Token Plan ---
|
|
2360
2360
|
openAiCompletionsDescriptor(
|
|
2361
|
-
"alibaba-
|
|
2362
|
-
"alibaba-
|
|
2363
|
-
"https://
|
|
2361
|
+
"alibaba-token-plan",
|
|
2362
|
+
"alibaba-token-plan",
|
|
2363
|
+
"https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
|
|
2364
2364
|
{
|
|
2365
2365
|
compat: {
|
|
2366
2366
|
supportsDeveloperRole: false,
|
|
@@ -910,11 +910,15 @@ function buildAdditionalModelRequestFields(
|
|
|
910
910
|
/**
|
|
911
911
|
* Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
|
|
912
912
|
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
913
|
+
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
914
|
+
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
915
|
+
* must opt in like Opus 4.7+ (issue #2791).
|
|
913
916
|
* Bedrock model ids are prefixed with region/inference-profile slugs (e.g.
|
|
914
917
|
* `eu.anthropic.Anthropic model-opus-4-7-...`); the regex matches the `Anthropic model-opus-X-Y`
|
|
915
918
|
* fragment regardless of prefix.
|
|
916
919
|
*/
|
|
917
920
|
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
921
|
+
if (/claude-fable-\d/.test(modelId)) return true;
|
|
918
922
|
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
|
|
919
923
|
if (!match) return false;
|
|
920
924
|
const major = Number(match[1]);
|
|
@@ -306,8 +306,12 @@ let warnedStopSequencesTrim = false;
|
|
|
306
306
|
/**
|
|
307
307
|
* Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
|
|
308
308
|
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
309
|
+
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
310
|
+
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
311
|
+
* must opt in like Opus 4.7+ (issue #2791).
|
|
309
312
|
*/
|
|
310
313
|
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
314
|
+
if (/claude-fable-\d/.test(modelId)) return true;
|
|
311
315
|
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
|
|
312
316
|
if (!match) return false;
|
|
313
317
|
const major = Number(match[1]);
|
|
@@ -407,6 +411,23 @@ export function isAnthropicThinkingBlockMutationError(error: unknown): boolean {
|
|
|
407
411
|
);
|
|
408
412
|
}
|
|
409
413
|
|
|
414
|
+
/**
|
|
415
|
+
* 400 shape where a replayed `thinking`/`redacted_thinking` block fails signature
|
|
416
|
+
* validation, e.g. `messages.5.content.24: Invalid \`signature\` in \`thinking\` block`.
|
|
417
|
+
* Unlike the latest-assistant mutation error above, the cited block can sit anywhere
|
|
418
|
+
* in the replayed history, so recovery must repair every assistant message rather
|
|
419
|
+
* than only the latest one.
|
|
420
|
+
*/
|
|
421
|
+
export function isAnthropicThinkingSignatureInvalidError(error: unknown): boolean {
|
|
422
|
+
if (extractHttpStatusFromError(error) !== 400) return false;
|
|
423
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
424
|
+
return (
|
|
425
|
+
/invalid_request_error/i.test(message) &&
|
|
426
|
+
/thinking|redacted_thinking/i.test(message) &&
|
|
427
|
+
/invalid\s+`?signature`?/i.test(message)
|
|
428
|
+
);
|
|
429
|
+
}
|
|
430
|
+
|
|
410
431
|
function hasStrictAnthropicTools(params: MessageCreateParamsStreaming): boolean {
|
|
411
432
|
const tools = params.tools as Array<{ strict?: unknown }> | undefined;
|
|
412
433
|
return tools?.some(tool => tool.strict === true) ?? false;
|
|
@@ -600,6 +621,24 @@ export const stripClaudeToolPrefix = (name: string, prefixOverride: string = cla
|
|
|
600
621
|
return name.slice(prefixOverride.length);
|
|
601
622
|
};
|
|
602
623
|
|
|
624
|
+
// Anthropic requires image `data` to be standard (RFC 4648) base64: the standard
|
|
625
|
+
// alphabet only, correct quartet grouping, and padding (when present) confined to
|
|
626
|
+
// a trailing `=`/`==`. A resident image whose blob went missing bakes a
|
|
627
|
+
// human-readable placeholder into `data` (e.g. "[Session resident imageData blob
|
|
628
|
+
// missing: …]"), and other callers can pass whitespace, data URLs, or URL-safe
|
|
629
|
+
// variants — all of which the API rejects with a 400 `invalid base64 data` that
|
|
630
|
+
// fails the *entire* request and bricks the session. Validate the wire format
|
|
631
|
+
// strictly and degrade anything that is not standard base64 to text.
|
|
632
|
+
//
|
|
633
|
+
// Accepts canonical padded forms and their unpadded equivalents; rejects
|
|
634
|
+
// length % 4 === 1, misplaced/overlong padding, whitespace, data URLs, URL-safe
|
|
635
|
+
// (`-`/`_`) alphabets, prose, and empty input. The pattern has no nested
|
|
636
|
+
// quantifier, so even oversized inputs are rejected in linear time.
|
|
637
|
+
const ANTHROPIC_BASE64_IMAGE_DATA = /^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}(?:==)?|[A-Za-z0-9+/]{3}=?)?$/;
|
|
638
|
+
function isAnthropicBase64ImageData(data: string): boolean {
|
|
639
|
+
return data.length > 0 && data.length % 4 !== 1 && ANTHROPIC_BASE64_IMAGE_DATA.test(data);
|
|
640
|
+
}
|
|
641
|
+
|
|
603
642
|
/**
|
|
604
643
|
* Convert content blocks to Anthropic API format
|
|
605
644
|
*/
|
|
@@ -623,7 +662,18 @@ function convertContentBlocks(
|
|
|
623
662
|
.filter((block): block is TextContent => block.type === "text")
|
|
624
663
|
.map(block => block.text.toWellFormed())
|
|
625
664
|
.filter(text => text.trim().length > 0);
|
|
626
|
-
const imageBlocks
|
|
665
|
+
const imageBlocks: ImageContent[] = [];
|
|
666
|
+
for (const block of content) {
|
|
667
|
+
if (block.type !== "image") continue;
|
|
668
|
+
if (isAnthropicBase64ImageData(block.data)) {
|
|
669
|
+
imageBlocks.push(block);
|
|
670
|
+
continue;
|
|
671
|
+
}
|
|
672
|
+
// Non-base64 image payload (e.g. a missing-blob placeholder): degrade to
|
|
673
|
+
// text so one lost image cannot invalidate the entire request.
|
|
674
|
+
const text = block.data.toWellFormed().trim();
|
|
675
|
+
if (text.length > 0) textBlocks.push(text);
|
|
676
|
+
}
|
|
627
677
|
const omittedImages = !supportsImages && imageBlocks.length > 0;
|
|
628
678
|
if (imageBlocks.length === 0 || !supportsImages) {
|
|
629
679
|
if (omittedImages) {
|
|
@@ -1283,20 +1333,19 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1283
1333
|
let strictFallbackErrorMessage: string | undefined;
|
|
1284
1334
|
let dropFastMode = providerSessionState?.fastModeDisabled ?? false;
|
|
1285
1335
|
let droppedForcedToolChoice = false;
|
|
1286
|
-
|
|
1287
|
-
|
|
1288
|
-
|
|
1289
|
-
|
|
1290
|
-
|
|
1291
|
-
|
|
1292
|
-
|
|
1293
|
-
|
|
1294
|
-
|
|
1295
|
-
|
|
1296
|
-
|
|
1297
|
-
|
|
1298
|
-
)
|
|
1299
|
-
if (paramsOptions?.dropForcedToolChoice === true) {
|
|
1336
|
+
let repairLatestAssistantThinking = false;
|
|
1337
|
+
let repairAllAssistantThinking = false;
|
|
1338
|
+
const prepareParams = async (): Promise<MessageCreateParamsStreaming> => {
|
|
1339
|
+
// Degradation state is cumulative: every fallback rebuild must merge all
|
|
1340
|
+
// repairs activated so far. Rebuilding from only the immediate call lets
|
|
1341
|
+
// a later strict/forced-tool/fast-mode fallback reintroduce the rejected
|
|
1342
|
+
// shape (e.g. invalid thinking signatures or forced tool_choice), and
|
|
1343
|
+
// the one-shot thinking-repair guard then blocks recovery.
|
|
1344
|
+
let nextParams = buildParams(model, baseUrl, context, isOAuthToken, options, disableStrictTools, {
|
|
1345
|
+
repairLatestAssistantThinking,
|
|
1346
|
+
repairAllAssistantThinking,
|
|
1347
|
+
});
|
|
1348
|
+
if (droppedForcedToolChoice) {
|
|
1300
1349
|
delete nextParams.tool_choice;
|
|
1301
1350
|
}
|
|
1302
1351
|
if (disableStrictTools) {
|
|
@@ -1730,23 +1779,30 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1730
1779
|
registryKey: resolveToolChoice(model, options?.toolChoice).registryKey,
|
|
1731
1780
|
});
|
|
1732
1781
|
droppedForcedToolChoice = true;
|
|
1733
|
-
params = await prepareParams(
|
|
1782
|
+
params = await prepareParams();
|
|
1734
1783
|
providerRetryAttempt = 0;
|
|
1735
1784
|
resetOutputForRetry();
|
|
1736
1785
|
continue;
|
|
1737
1786
|
}
|
|
1787
|
+
const thinkingSignatureInvalid = isAnthropicThinkingSignatureInvalidError(streamFailure);
|
|
1738
1788
|
if (
|
|
1739
1789
|
!options?.fallbackManaged &&
|
|
1740
1790
|
!thinkingRepairAttempted &&
|
|
1741
1791
|
firstTokenTime === undefined &&
|
|
1742
|
-
isAnthropicThinkingBlockMutationError(streamFailure)
|
|
1792
|
+
(thinkingSignatureInvalid || isAnthropicThinkingBlockMutationError(streamFailure))
|
|
1743
1793
|
) {
|
|
1744
|
-
logger.debug("anthropic: repairing
|
|
1794
|
+
logger.debug("anthropic: repairing assistant thinking replay after provider rejection", {
|
|
1745
1795
|
model: model.id,
|
|
1796
|
+
scope: thinkingSignatureInvalid ? "all" : "latest",
|
|
1746
1797
|
error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure),
|
|
1747
1798
|
});
|
|
1748
1799
|
thinkingRepairAttempted = true;
|
|
1749
|
-
|
|
1800
|
+
if (thinkingSignatureInvalid) {
|
|
1801
|
+
repairAllAssistantThinking = true;
|
|
1802
|
+
} else {
|
|
1803
|
+
repairLatestAssistantThinking = true;
|
|
1804
|
+
}
|
|
1805
|
+
params = await prepareParams();
|
|
1750
1806
|
providerRetryAttempt = 0;
|
|
1751
1807
|
resetOutputForRetry();
|
|
1752
1808
|
continue;
|
|
@@ -2210,13 +2266,13 @@ function buildParams(
|
|
|
2210
2266
|
isOAuthToken: boolean,
|
|
2211
2267
|
options?: AnthropicOptions,
|
|
2212
2268
|
disableStrictTools = false,
|
|
2213
|
-
repairLatestAssistantThinking
|
|
2269
|
+
thinkingRepair?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
|
|
2214
2270
|
): MessageCreateParamsStreaming {
|
|
2215
2271
|
const { mode: cacheMode, cacheControl } = getCacheControl(model, baseUrl, options?.cacheRetention);
|
|
2216
2272
|
|
|
2217
2273
|
const params: AnthropicSamplingParams = {
|
|
2218
2274
|
model: model.id,
|
|
2219
|
-
messages: convertAnthropicMessages(context.messages, model, isOAuthToken,
|
|
2275
|
+
messages: convertAnthropicMessages(context.messages, model, isOAuthToken, thinkingRepair),
|
|
2220
2276
|
max_tokens: options?.maxTokens || (model.maxTokens / 3) | 0,
|
|
2221
2277
|
stream: true,
|
|
2222
2278
|
};
|
|
@@ -2407,7 +2463,7 @@ export function convertAnthropicMessages(
|
|
|
2407
2463
|
messages: Message[],
|
|
2408
2464
|
model: Model<"anthropic-messages">,
|
|
2409
2465
|
isOAuthToken: boolean,
|
|
2410
|
-
options?: { repairLatestAssistantThinking?: boolean },
|
|
2466
|
+
options?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
|
|
2411
2467
|
): MessageParam[] {
|
|
2412
2468
|
const params: MessageParam[] = [];
|
|
2413
2469
|
|
|
@@ -69,7 +69,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
69
69
|
baseUrl.includes("api.anthropic.com") ||
|
|
70
70
|
/(^|\/)claude[-.]/i.test(model.id) ||
|
|
71
71
|
/(^|\/)anthropic\//i.test(model.id);
|
|
72
|
-
const isAlibaba =
|
|
72
|
+
const isAlibaba = baseUrl.includes("dashscope");
|
|
73
73
|
const isQwen = model.id.toLowerCase().includes("qwen");
|
|
74
74
|
// DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
|
|
75
75
|
// thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
|
|
@@ -244,7 +244,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
244
244
|
requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
|
|
245
245
|
openRouterRouting: undefined,
|
|
246
246
|
vercelGatewayRouting: undefined,
|
|
247
|
-
supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
|
|
247
|
+
supportsStrictMode: detectStrictModeSupport(provider, baseUrl) && !(isDeepseekFamily && isOpenRouter),
|
|
248
248
|
extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
|
|
249
249
|
toolStrictMode: isCerebras ? "all_strict" : "mixed",
|
|
250
250
|
};
|
|
@@ -426,6 +426,7 @@ function getTrailingPartialDeepseekToken(text: string): string {
|
|
|
426
426
|
return tail;
|
|
427
427
|
}
|
|
428
428
|
|
|
429
|
+
const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
|
|
429
430
|
const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
|
|
430
431
|
"OpenAI completions stream timed out while waiting for the first event";
|
|
431
432
|
|
|
@@ -562,8 +563,10 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
562
563
|
openaiStream = await createCompletionsStream("none");
|
|
563
564
|
}
|
|
564
565
|
}
|
|
566
|
+
const firstEventFallbackMs =
|
|
567
|
+
model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
|
|
565
568
|
const firstEventWatchdog = createWatchdog(
|
|
566
|
-
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
|
569
|
+
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
567
570
|
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
568
571
|
);
|
|
569
572
|
if (premiumRequestsTotal !== undefined) {
|
|
@@ -130,6 +130,7 @@ export interface OpenAIResponsesOptions extends StreamOptions {
|
|
|
130
130
|
}
|
|
131
131
|
|
|
132
132
|
const OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX = "openai-responses:";
|
|
133
|
+
const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
|
|
133
134
|
const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
|
|
134
135
|
"OpenAI responses stream timed out while waiting for the first event";
|
|
135
136
|
const OPENAI_DEFAULT_BASE_URL = "https://api.openai.com/v1";
|
|
@@ -330,8 +331,10 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
330
331
|
await notifyProviderResponse(options, response, model, request_id);
|
|
331
332
|
return data;
|
|
332
333
|
});
|
|
334
|
+
const firstEventFallbackMs =
|
|
335
|
+
model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
|
|
333
336
|
const firstEventWatchdog = createWatchdog(
|
|
334
|
-
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
|
337
|
+
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
|
|
335
338
|
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
336
339
|
);
|
|
337
340
|
if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
|
|
@@ -31,7 +31,7 @@ export function transformMessages<TApi extends Api>(
|
|
|
31
31
|
messages: Message[],
|
|
32
32
|
model: Model<TApi>,
|
|
33
33
|
normalizeToolCallId?: (id: string, model: Model<TApi>, source: AssistantMessage) => string,
|
|
34
|
-
options?: { repairLatestAssistantThinking?: boolean },
|
|
34
|
+
options?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
|
|
35
35
|
): Message[] {
|
|
36
36
|
// Build a map of original tool call IDs to normalized IDs
|
|
37
37
|
const toolCallIdMap = new Map<string, string>();
|
|
@@ -73,16 +73,29 @@ export function transformMessages<TApi extends Api>(
|
|
|
73
73
|
// are kept so the second pass can either preserve real results or synthesize
|
|
74
74
|
// an explicit aborted result without leaving dangling tool_use blocks.
|
|
75
75
|
const hasPartialThinking = assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error";
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
76
|
+
// One-shot Anthropic replay repair. `repairLatestAssistantThinking` targets the
|
|
77
|
+
// "latest assistant message ... cannot be modified" 400; `repairAllAssistantThinking`
|
|
78
|
+
// targets the "Invalid `signature` in `thinking` block" 400, which can cite a block
|
|
79
|
+
// anywhere in the replayed history (e.g. after compaction/pruning rewrote an earlier
|
|
80
|
+
// turn), so the drop must apply to every assistant message. Within each
|
|
81
|
+
// message only blocks that would replay as native thinking/redacted_thinking
|
|
82
|
+
// are dropped; cross-model reasoning degrades to text and is preserved.
|
|
83
|
+
const dropAssistantThinkingForRepair =
|
|
84
|
+
(options?.repairAllAssistantThinking === true ||
|
|
85
|
+
(options?.repairLatestAssistantThinking === true && index === latestAssistantIndex)) &&
|
|
79
86
|
model.api === "anthropic-messages" &&
|
|
80
87
|
assistantMsg.api === "anthropic-messages";
|
|
81
88
|
|
|
82
89
|
const transformedContent = assistantMsg.content.flatMap(block => {
|
|
83
90
|
if (block.type === "thinking") {
|
|
84
|
-
if (hasPartialThinking
|
|
91
|
+
if (hasPartialThinking) return [];
|
|
85
92
|
const sanitized = block;
|
|
93
|
+
// Repair must only drop blocks that would otherwise replay as native
|
|
94
|
+
// thinking. Cross-model/provider reasoning degrades to unsigned text
|
|
95
|
+
// below and was never replayed as a signed block, so it cannot be the
|
|
96
|
+
// signature failure — dropping it would silently lose valid context.
|
|
97
|
+
const replaysAsNativeThinking = mustPreserveLatestAnthropicThinking || isSameModel;
|
|
98
|
+
if (dropAssistantThinkingForRepair && replaysAsNativeThinking) return [];
|
|
86
99
|
if (mustPreserveLatestAnthropicThinking) return sanitized;
|
|
87
100
|
// For same model: keep thinking blocks with signatures (needed for replay)
|
|
88
101
|
// even if the thinking text is empty (OpenAI encrypted reasoning)
|
|
@@ -97,7 +110,13 @@ export function transformMessages<TApi extends Api>(
|
|
|
97
110
|
}
|
|
98
111
|
|
|
99
112
|
if (block.type === "redactedThinking") {
|
|
100
|
-
if (hasPartialThinking
|
|
113
|
+
if (hasPartialThinking) return [];
|
|
114
|
+
// Same restriction as thinking blocks: cross-model/provider redacted
|
|
115
|
+
// blocks already drop below, so repair only needs to cover blocks that
|
|
116
|
+
// would replay as native redacted_thinking.
|
|
117
|
+
if (dropAssistantThinkingForRepair && (mustPreserveLatestAnthropicThinking || isSameModel)) {
|
|
118
|
+
return [];
|
|
119
|
+
}
|
|
101
120
|
if (mustPreserveLatestAnthropicThinking) return block;
|
|
102
121
|
if (isSameModel) return block;
|
|
103
122
|
return [];
|
package/src/stream.ts
CHANGED
|
@@ -85,7 +85,7 @@ function hasVertexAdcCredentials(): boolean {
|
|
|
85
85
|
type KeyResolver = string | (() => string | undefined);
|
|
86
86
|
|
|
87
87
|
const serviceProviderMap: Record<string, KeyResolver> = {
|
|
88
|
-
"alibaba-
|
|
88
|
+
"alibaba-token-plan": "ALIBABA_TOKEN_PLAN_API_KEY",
|
|
89
89
|
openai: () => $credentialEnv("OPENAI_API_KEY"),
|
|
90
90
|
google: "GEMINI_API_KEY",
|
|
91
91
|
groq: "GROQ_API_KEY",
|
package/src/types.ts
CHANGED
|
@@ -114,7 +114,7 @@ export interface ThinkingConfig {
|
|
|
114
114
|
}
|
|
115
115
|
|
|
116
116
|
export type KnownProvider =
|
|
117
|
-
| "alibaba-
|
|
117
|
+
| "alibaba-token-plan"
|
|
118
118
|
| "amazon-bedrock"
|
|
119
119
|
| "azure-openai"
|
|
120
120
|
| "anthropic"
|
|
@@ -709,10 +709,15 @@ export type TSchema = ZodType | TJsonSchema;
|
|
|
709
709
|
/** Resolve parameter types for tool execution / handlers. */
|
|
710
710
|
export type Static<S> = S extends ZodType ? z.infer<S> : S extends { static: infer T } ? T : unknown;
|
|
711
711
|
|
|
712
|
+
export type RawArgumentRejectionCode =
|
|
713
|
+
| "ask-intent-review-requires-positive-round"
|
|
714
|
+
| "ask-intent-contract-requires-non-empty-authority"
|
|
715
|
+
| "ask-deep-interview-metadata-requires-deep-interview-gate";
|
|
716
|
+
|
|
712
717
|
export type RawArgumentValidationResult =
|
|
713
718
|
| { outcome: "passthrough" }
|
|
714
719
|
| { outcome: "accept"; arguments: ToolCall["arguments"] }
|
|
715
|
-
| { outcome: "reject" };
|
|
720
|
+
| { outcome: "reject"; code?: RawArgumentRejectionCode };
|
|
716
721
|
|
|
717
722
|
export interface Tool<TParameters extends TSchema = TSchema> {
|
|
718
723
|
name: string;
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Alibaba
|
|
2
|
+
* Alibaba Token Plan login flow.
|
|
3
3
|
*
|
|
4
|
-
* Alibaba
|
|
4
|
+
* Alibaba Token Plan provides OpenAI-compatible models via
|
|
5
|
+
* https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1.
|
|
5
6
|
*
|
|
6
7
|
* This is not OAuth - it's a simple API key flow:
|
|
7
|
-
* 1. Open browser to Alibaba Cloud
|
|
8
|
+
* 1. Open browser to Alibaba Cloud Model Studio console
|
|
8
9
|
* 2. User copies their API key
|
|
9
10
|
* 3. User pastes the API key into the CLI
|
|
10
11
|
*/
|
|
@@ -13,27 +14,27 @@ import { validateOpenAICompatibleApiKey } from "./api-key-validation";
|
|
|
13
14
|
import type { OAuthController } from "./types";
|
|
14
15
|
|
|
15
16
|
const AUTH_URL = "https://modelstudio.console.alibabacloud.com/";
|
|
16
|
-
const API_BASE_URL = "https://
|
|
17
|
-
const VALIDATION_MODEL = "
|
|
17
|
+
const API_BASE_URL = "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1";
|
|
18
|
+
const VALIDATION_MODEL = "deepseek-v4-pro";
|
|
18
19
|
|
|
19
20
|
/**
|
|
20
|
-
* Login to Alibaba
|
|
21
|
+
* Login to Alibaba Token Plan.
|
|
21
22
|
*
|
|
22
23
|
* Opens browser to API keys page, prompts user to paste their API key.
|
|
23
24
|
* Returns the API key directly (not OAuthCredentials - this isn't OAuth).
|
|
24
25
|
*/
|
|
25
|
-
export async function
|
|
26
|
+
export async function loginAlibabaTokenPlan(options: OAuthController): Promise<string> {
|
|
26
27
|
if (!options.onPrompt) {
|
|
27
|
-
throw new Error("Alibaba
|
|
28
|
+
throw new Error("Alibaba Token Plan login requires onPrompt callback");
|
|
28
29
|
}
|
|
29
30
|
|
|
30
31
|
options.onAuth?.({
|
|
31
32
|
url: AUTH_URL,
|
|
32
|
-
instructions: "Copy your API key from the Alibaba Cloud
|
|
33
|
+
instructions: "Copy your API key from the Alibaba Cloud Model Studio console",
|
|
33
34
|
});
|
|
34
35
|
|
|
35
36
|
const apiKey = await options.onPrompt({
|
|
36
|
-
message: "Paste your Alibaba
|
|
37
|
+
message: "Paste your Alibaba Token Plan API key",
|
|
37
38
|
placeholder: "sk-...",
|
|
38
39
|
});
|
|
39
40
|
|
|
@@ -48,7 +49,7 @@ export async function loginAlibabaCodingPlan(options: OAuthController): Promise<
|
|
|
48
49
|
|
|
49
50
|
options.onProgress?.("Validating API key...");
|
|
50
51
|
await validateOpenAICompatibleApiKey({
|
|
51
|
-
provider: "Alibaba
|
|
52
|
+
provider: "Alibaba Token Plan",
|
|
52
53
|
apiKey: trimmed,
|
|
53
54
|
baseUrl: API_BASE_URL,
|
|
54
55
|
model: VALIDATION_MODEL,
|
package/src/utils/oauth/index.ts
CHANGED
|
@@ -16,8 +16,8 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
|
|
|
16
16
|
available: true,
|
|
17
17
|
},
|
|
18
18
|
{
|
|
19
|
-
id: "alibaba-
|
|
20
|
-
name: "Alibaba
|
|
19
|
+
id: "alibaba-token-plan",
|
|
20
|
+
name: "Alibaba Token Plan",
|
|
21
21
|
available: true,
|
|
22
22
|
},
|
|
23
23
|
{
|
|
@@ -371,6 +371,7 @@ export async function refreshOAuthToken(
|
|
|
371
371
|
case "together":
|
|
372
372
|
case "litellm":
|
|
373
373
|
case "lm-studio":
|
|
374
|
+
case "alibaba-token-plan":
|
|
374
375
|
case "ollama":
|
|
375
376
|
case "ollama-cloud":
|
|
376
377
|
case "xiaomi":
|
package/src/utils/oauth/types.ts
CHANGED
package/src/utils/validation.ts
CHANGED
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
import { structuredCloneJSON } from "@gajae-code/utils";
|
|
26
26
|
import type { ZodType } from "zod/v4";
|
|
27
27
|
import type { $ZodIssue as ZodIssue } from "zod/v4/core";
|
|
28
|
-
import type { Tool, ToolCall } from "../types";
|
|
28
|
+
import type { RawArgumentRejectionCode, Tool, ToolCall } from "../types";
|
|
29
29
|
import { upgradeJsonSchemaTo202012 } from "./schema/draft";
|
|
30
30
|
import {
|
|
31
31
|
isJsonSchemaValueValid,
|
|
@@ -958,6 +958,15 @@ export function validateToolCall(tools: Tool[], toolCall: ToolCall): ToolCall["a
|
|
|
958
958
|
return validateToolArguments(tool, toolCall);
|
|
959
959
|
}
|
|
960
960
|
|
|
961
|
+
const RAW_ARGUMENT_REJECTION_MESSAGES: Record<RawArgumentRejectionCode, string> = {
|
|
962
|
+
"ask-intent-review-requires-positive-round":
|
|
963
|
+
"deepInterview.intent_review is post-Round-0 only and requires a positive round",
|
|
964
|
+
"ask-intent-contract-requires-non-empty-authority":
|
|
965
|
+
"deepInterview.intent_contract requires non-empty items and confirmation_options",
|
|
966
|
+
"ask-deep-interview-metadata-requires-deep-interview-gate":
|
|
967
|
+
"deepInterview metadata cannot be combined with a non-deep-interview workflowGate",
|
|
968
|
+
};
|
|
969
|
+
|
|
961
970
|
/**
|
|
962
971
|
* Validates tool call arguments against the tool's schema (Zod or plain JSON
|
|
963
972
|
* Schema). Applies LLM-quirk coercions (numeric strings, JSON-string
|
|
@@ -969,7 +978,13 @@ export function validateToolArguments(tool: Tool, toolCall: ToolCall): ToolCall[
|
|
|
969
978
|
const originalArgs = toolCall.arguments;
|
|
970
979
|
const rawValidation = tool.rawArgumentValidation?.(originalArgs);
|
|
971
980
|
if (rawValidation?.outcome === "reject") {
|
|
972
|
-
|
|
981
|
+
const base = `Validation failed for tool "${toolCall.name}": raw arguments rejected before coercion`;
|
|
982
|
+
const code = rawValidation.code;
|
|
983
|
+
const correction =
|
|
984
|
+
typeof code === "string" && Object.hasOwn(RAW_ARGUMENT_REJECTION_MESSAGES, code)
|
|
985
|
+
? RAW_ARGUMENT_REJECTION_MESSAGES[code as RawArgumentRejectionCode]
|
|
986
|
+
: undefined;
|
|
987
|
+
throw new Error(correction ? `${base}; ${correction}` : base);
|
|
973
988
|
}
|
|
974
989
|
const rawArgs = rawValidation?.outcome === "accept" ? rawValidation.arguments : originalArgs;
|
|
975
990
|
const ctx = getValidationContext(tool);
|
package/src/utils.ts
CHANGED
|
@@ -271,17 +271,53 @@ function normalizeResponsesImageUrlForReplay(value: unknown): NormalizedResponse
|
|
|
271
271
|
return { imageUrl: stringifyResponsesStringParamForReplay(value) };
|
|
272
272
|
}
|
|
273
273
|
|
|
274
|
+
/**
|
|
275
|
+
* OpenAI Responses `input_image.image_url` must be a fetchable HTTP(S) URL or an
|
|
276
|
+
* image data URI. Session resident-blob materialization may leave a human-readable
|
|
277
|
+
* placeholder like `[Session resident imageUrl blob missing: sha256:…; …]` in this
|
|
278
|
+
* field; replaying that string as `image_url` makes Codex reject the entire turn
|
|
279
|
+
* with `invalid_value` (#2924).
|
|
280
|
+
*/
|
|
281
|
+
function isProviderSafeResponsesImageUrl(value: string): boolean {
|
|
282
|
+
const url = value.trim();
|
|
283
|
+
if (url.length === 0) return false;
|
|
284
|
+
if (url.startsWith("https://") || url.startsWith("http://")) return true;
|
|
285
|
+
// Accept only image data URIs — other data: schemes are not valid image inputs.
|
|
286
|
+
if (url.startsWith("data:image/")) return true;
|
|
287
|
+
return false;
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
function hasNonEmptyResponsesFileId(part: Record<string, unknown>): boolean {
|
|
291
|
+
return typeof part.file_id === "string" && part.file_id.trim().length > 0;
|
|
292
|
+
}
|
|
293
|
+
|
|
274
294
|
function sanitizeResponsesMessageContentForReplay(content: unknown): unknown {
|
|
275
295
|
if (typeof content === "string") return neutralizeReservedControlTokens(content.toWellFormed());
|
|
276
296
|
if (!Array.isArray(content)) return content;
|
|
277
|
-
|
|
278
|
-
|
|
297
|
+
const sanitizedContent: unknown[] = [];
|
|
298
|
+
for (const part of content) {
|
|
299
|
+
if (!part || typeof part !== "object") {
|
|
300
|
+
sanitizedContent.push(part);
|
|
301
|
+
continue;
|
|
302
|
+
}
|
|
279
303
|
const sanitizedPart = { ...(part as Record<string, unknown>) };
|
|
280
304
|
if ("text" in sanitizedPart) {
|
|
281
305
|
sanitizedPart.text = normalizeResponsesMessageTextForReplay(sanitizedPart.text);
|
|
282
306
|
}
|
|
283
307
|
if ("image_url" in sanitizedPart) {
|
|
284
308
|
const normalizedImageUrl = normalizeResponsesImageUrlForReplay(sanitizedPart.image_url);
|
|
309
|
+
if (!isProviderSafeResponsesImageUrl(normalizedImageUrl.imageUrl)) {
|
|
310
|
+
// Keep the part when a provider file_id can stand alone; otherwise drop
|
|
311
|
+
// only this image part so neighboring text/history still replays.
|
|
312
|
+
if (!hasNonEmptyResponsesFileId(sanitizedPart)) continue;
|
|
313
|
+
delete sanitizedPart.image_url;
|
|
314
|
+
if (sanitizedPart.type === "image_url") sanitizedPart.type = "input_image";
|
|
315
|
+
if ("detail" in sanitizedPart && !isResponsesImageDetail(sanitizedPart.detail)) {
|
|
316
|
+
delete sanitizedPart.detail;
|
|
317
|
+
}
|
|
318
|
+
sanitizedContent.push(sanitizedPart);
|
|
319
|
+
continue;
|
|
320
|
+
}
|
|
285
321
|
sanitizedPart.image_url = normalizedImageUrl.imageUrl;
|
|
286
322
|
if (sanitizedPart.type === "image_url") {
|
|
287
323
|
sanitizedPart.type = "input_image";
|
|
@@ -292,8 +328,9 @@ function sanitizeResponsesMessageContentForReplay(content: unknown): unknown {
|
|
|
292
328
|
delete sanitizedPart.detail;
|
|
293
329
|
}
|
|
294
330
|
}
|
|
295
|
-
|
|
296
|
-
}
|
|
331
|
+
sanitizedContent.push(sanitizedPart);
|
|
332
|
+
}
|
|
333
|
+
return sanitizedContent;
|
|
297
334
|
}
|
|
298
335
|
|
|
299
336
|
function sanitizeResponsesStringFieldsForReplay(item: Record<string, unknown>): void {
|
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Alibaba Coding Plan login flow.
|
|
3
|
-
*
|
|
4
|
-
* Alibaba Coding Plan provides OpenAI-compatible models via https://coding-intl.dashscope.aliyuncs.com/v1.
|
|
5
|
-
*
|
|
6
|
-
* This is not OAuth - it's a simple API key flow:
|
|
7
|
-
* 1. Open browser to Alibaba Cloud DashScope API key settings
|
|
8
|
-
* 2. User copies their API key
|
|
9
|
-
* 3. User pastes the API key into the CLI
|
|
10
|
-
*/
|
|
11
|
-
import type { OAuthController } from "./types";
|
|
12
|
-
/**
|
|
13
|
-
* Login to Alibaba Coding Plan.
|
|
14
|
-
*
|
|
15
|
-
* Opens browser to API keys page, prompts user to paste their API key.
|
|
16
|
-
* Returns the API key directly (not OAuthCredentials - this isn't OAuth).
|
|
17
|
-
*/
|
|
18
|
-
export declare function loginAlibabaCodingPlan(options: OAuthController): Promise<string>;
|