@gajae-code/ai 0.11.6 → 0.11.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/CHANGELOG.md +22 -0
  2. package/dist/types/auth-broker/client.d.ts +2 -1
  3. package/dist/types/auth-broker/remote-store.d.ts +2 -1
  4. package/dist/types/auth-broker/types.d.ts +3 -1
  5. package/dist/types/auth-broker/wire-schemas.d.ts +68 -0
  6. package/dist/types/auth-storage.d.ts +21 -1
  7. package/dist/types/provider-models/openai-compat.d.ts +2 -2
  8. package/dist/types/providers/anthropic.d.ts +9 -0
  9. package/dist/types/providers/transform-messages.d.ts +1 -0
  10. package/dist/types/types.d.ts +3 -1
  11. package/dist/types/utils/oauth/alibaba-token-plan.d.ts +19 -0
  12. package/dist/types/utils/oauth/types.d.ts +1 -1
  13. package/package.json +2 -2
  14. package/src/auth-broker/client.ts +13 -0
  15. package/src/auth-broker/refresher.ts +1 -0
  16. package/src/auth-broker/remote-store.ts +25 -0
  17. package/src/auth-broker/server.ts +10 -2
  18. package/src/auth-broker/types.ts +4 -0
  19. package/src/auth-broker/wire-schemas.ts +17 -1
  20. package/src/auth-storage.ts +199 -30
  21. package/src/model-thinking.ts +11 -3
  22. package/src/models.json +5362 -1219
  23. package/src/provider-models/descriptors.ts +5 -6
  24. package/src/provider-models/openai-compat.ts +12 -12
  25. package/src/providers/amazon-bedrock.ts +4 -0
  26. package/src/providers/anthropic.ts +78 -22
  27. package/src/providers/openai-completions-compat.ts +2 -2
  28. package/src/providers/openai-completions.ts +4 -1
  29. package/src/providers/openai-responses.ts +4 -1
  30. package/src/providers/transform-messages.ts +25 -6
  31. package/src/stream.ts +1 -1
  32. package/src/types.ts +7 -2
  33. package/src/utils/oauth/{alibaba-coding-plan.ts → alibaba-token-plan.ts} +12 -11
  34. package/src/utils/oauth/index.ts +3 -2
  35. package/src/utils/oauth/types.ts +1 -1
  36. package/src/utils/validation.ts +17 -2
  37. package/src/utils.ts +41 -4
  38. package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +0 -18
@@ -9,7 +9,7 @@ import type { OAuthProvider } from "../utils/oauth/types";
9
9
  import { googleModelManagerOptions } from "./google";
10
10
  import { ollamaCloudModelManagerOptions } from "./ollama";
11
11
  import {
12
- alibabaCodingPlanModelManagerOptions,
12
+ alibabaTokenPlanModelManagerOptions,
13
13
  anthropicModelManagerOptions,
14
14
  cerebrasModelManagerOptions,
15
15
  cloudflareAiGatewayModelManagerOptions,
@@ -130,10 +130,10 @@ function catalogDescriptor(
130
130
  export const PROVIDER_DESCRIPTORS: readonly ProviderDescriptor[] = [
131
131
  descriptor("anthropic", "claude-sonnet-5", config => anthropicModelManagerOptions(config)),
132
132
  catalogDescriptor(
133
- "alibaba-coding-plan",
134
- "qwen3.5-plus",
135
- config => alibabaCodingPlanModelManagerOptions(config),
136
- catalog("Alibaba Coding Plan", ["ALIBABA_CODING_PLAN_API_KEY"]),
133
+ "alibaba-token-plan",
134
+ "deepseek-v4-pro",
135
+ config => alibabaTokenPlanModelManagerOptions(config),
136
+ catalog("Alibaba Token Plan", ["ALIBABA_TOKEN_PLAN_API_KEY"], { oauthProvider: "alibaba-token-plan" }),
137
137
  ),
138
138
  descriptor("openai", "gpt-5.4", config => openaiModelManagerOptions(config)),
139
139
  descriptor("groq", "openai/gpt-oss-120b", config => groqModelManagerOptions(config)),
@@ -334,7 +334,6 @@ export const DEFAULT_MODEL_PER_PROVIDER: Record<KnownProvider, string> = {
334
334
  ...Object.fromEntries(PROVIDER_DESCRIPTORS.map(d => [d.providerId, d.defaultModel])),
335
335
  // Providers not in PROVIDER_DESCRIPTORS (special auth or no standard discovery)
336
336
  "azure-openai": "gpt-4.1",
337
- "alibaba-coding-plan": "qwen3.5-plus",
338
337
  "amazon-bedrock": "us.anthropic.claude-opus-4-6-v1",
339
338
  "google-antigravity": "gemini-3-pro-high",
340
339
  "google-gemini-cli": "gemini-2.5-pro",
@@ -1124,26 +1124,26 @@ export function kiloModelManagerOptions(config?: KiloModelManagerConfig): ModelM
1124
1124
  }
1125
1125
 
1126
1126
  // ---------------------------------------------------------------------------
1127
- // Alibaba Coding Plan
1127
+ // Alibaba Token Plan
1128
1128
  // ---------------------------------------------------------------------------
1129
1129
 
1130
- export interface AlibabaCodingPlanModelManagerConfig {
1130
+ export interface AlibabaTokenPlanModelManagerConfig {
1131
1131
  apiKey?: string;
1132
1132
  baseUrl?: string;
1133
1133
  }
1134
1134
 
1135
- export function alibabaCodingPlanModelManagerOptions(
1136
- config?: AlibabaCodingPlanModelManagerConfig,
1135
+ export function alibabaTokenPlanModelManagerOptions(
1136
+ config?: AlibabaTokenPlanModelManagerConfig,
1137
1137
  ): ModelManagerOptions<"openai-completions"> {
1138
1138
  const apiKey = config?.apiKey;
1139
- const baseUrl = config?.baseUrl ?? "https://coding-intl.dashscope.aliyuncs.com/v1";
1140
- const references = createBundledReferenceMap<"openai-completions">("alibaba-coding-plan");
1139
+ const baseUrl = config?.baseUrl ?? "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1";
1140
+ const references = createBundledReferenceMap<"openai-completions">("alibaba-token-plan");
1141
1141
  return {
1142
- providerId: "alibaba-coding-plan",
1142
+ providerId: "alibaba-token-plan",
1143
1143
  fetchDynamicModels: () =>
1144
1144
  fetchOpenAICompatibleModels({
1145
1145
  api: "openai-completions",
1146
- provider: "alibaba-coding-plan",
1146
+ provider: "alibaba-token-plan",
1147
1147
  baseUrl,
1148
1148
  apiKey,
1149
1149
  mapModel: (entry, defaults) => {
@@ -2356,11 +2356,11 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDe
2356
2356
  reasoningContentField: "reasoning_content",
2357
2357
  },
2358
2358
  }),
2359
- // --- Alibaba Coding Plan ---
2359
+ // --- Alibaba Token Plan ---
2360
2360
  openAiCompletionsDescriptor(
2361
- "alibaba-coding-plan",
2362
- "alibaba-coding-plan",
2363
- "https://coding-intl.dashscope.aliyuncs.com/v1",
2361
+ "alibaba-token-plan",
2362
+ "alibaba-token-plan",
2363
+ "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1",
2364
2364
  {
2365
2365
  compat: {
2366
2366
  supportsDeveloperRole: false,
@@ -910,11 +910,15 @@ function buildAdditionalModelRequestFields(
910
910
  /**
911
911
  * Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
912
912
  * Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
913
+ * Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
914
+ * "omitted" — thinking tokens are billed but no content streams back — so it
915
+ * must opt in like Opus 4.7+ (issue #2791).
913
916
  * Bedrock model ids are prefixed with region/inference-profile slugs (e.g.
914
917
  * `eu.anthropic.Anthropic model-opus-4-7-...`); the regex matches the `Anthropic model-opus-X-Y`
915
918
  * fragment regardless of prefix.
916
919
  */
917
920
  function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
921
+ if (/claude-fable-\d/.test(modelId)) return true;
918
922
  const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
919
923
  if (!match) return false;
920
924
  const major = Number(match[1]);
@@ -306,8 +306,12 @@ let warnedStopSequencesTrim = false;
306
306
  /**
307
307
  * Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
308
308
  * Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
309
+ * Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
310
+ * "omitted" — thinking tokens are billed but no content streams back — so it
311
+ * must opt in like Opus 4.7+ (issue #2791).
309
312
  */
310
313
  function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
314
+ if (/claude-fable-\d/.test(modelId)) return true;
311
315
  const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
312
316
  if (!match) return false;
313
317
  const major = Number(match[1]);
@@ -407,6 +411,23 @@ export function isAnthropicThinkingBlockMutationError(error: unknown): boolean {
407
411
  );
408
412
  }
409
413
 
414
+ /**
415
+ * 400 shape where a replayed `thinking`/`redacted_thinking` block fails signature
416
+ * validation, e.g. `messages.5.content.24: Invalid \`signature\` in \`thinking\` block`.
417
+ * Unlike the latest-assistant mutation error above, the cited block can sit anywhere
418
+ * in the replayed history, so recovery must repair every assistant message rather
419
+ * than only the latest one.
420
+ */
421
+ export function isAnthropicThinkingSignatureInvalidError(error: unknown): boolean {
422
+ if (extractHttpStatusFromError(error) !== 400) return false;
423
+ const message = error instanceof Error ? error.message : String(error);
424
+ return (
425
+ /invalid_request_error/i.test(message) &&
426
+ /thinking|redacted_thinking/i.test(message) &&
427
+ /invalid\s+`?signature`?/i.test(message)
428
+ );
429
+ }
430
+
410
431
  function hasStrictAnthropicTools(params: MessageCreateParamsStreaming): boolean {
411
432
  const tools = params.tools as Array<{ strict?: unknown }> | undefined;
412
433
  return tools?.some(tool => tool.strict === true) ?? false;
@@ -600,6 +621,24 @@ export const stripClaudeToolPrefix = (name: string, prefixOverride: string = cla
600
621
  return name.slice(prefixOverride.length);
601
622
  };
602
623
 
624
+ // Anthropic requires image `data` to be standard (RFC 4648) base64: the standard
625
+ // alphabet only, correct quartet grouping, and padding (when present) confined to
626
+ // a trailing `=`/`==`. A resident image whose blob went missing bakes a
627
+ // human-readable placeholder into `data` (e.g. "[Session resident imageData blob
628
+ // missing: …]"), and other callers can pass whitespace, data URLs, or URL-safe
629
+ // variants — all of which the API rejects with a 400 `invalid base64 data` that
630
+ // fails the *entire* request and bricks the session. Validate the wire format
631
+ // strictly and degrade anything that is not standard base64 to text.
632
+ //
633
+ // Accepts canonical padded forms and their unpadded equivalents; rejects
634
+ // length % 4 === 1, misplaced/overlong padding, whitespace, data URLs, URL-safe
635
+ // (`-`/`_`) alphabets, prose, and empty input. The pattern has no nested
636
+ // quantifier, so even oversized inputs are rejected in linear time.
637
+ const ANTHROPIC_BASE64_IMAGE_DATA = /^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}(?:==)?|[A-Za-z0-9+/]{3}=?)?$/;
638
+ function isAnthropicBase64ImageData(data: string): boolean {
639
+ return data.length > 0 && data.length % 4 !== 1 && ANTHROPIC_BASE64_IMAGE_DATA.test(data);
640
+ }
641
+
603
642
  /**
604
643
  * Convert content blocks to Anthropic API format
605
644
  */
@@ -623,7 +662,18 @@ function convertContentBlocks(
623
662
  .filter((block): block is TextContent => block.type === "text")
624
663
  .map(block => block.text.toWellFormed())
625
664
  .filter(text => text.trim().length > 0);
626
- const imageBlocks = content.filter((block): block is ImageContent => block.type === "image");
665
+ const imageBlocks: ImageContent[] = [];
666
+ for (const block of content) {
667
+ if (block.type !== "image") continue;
668
+ if (isAnthropicBase64ImageData(block.data)) {
669
+ imageBlocks.push(block);
670
+ continue;
671
+ }
672
+ // Non-base64 image payload (e.g. a missing-blob placeholder): degrade to
673
+ // text so one lost image cannot invalidate the entire request.
674
+ const text = block.data.toWellFormed().trim();
675
+ if (text.length > 0) textBlocks.push(text);
676
+ }
627
677
  const omittedImages = !supportsImages && imageBlocks.length > 0;
628
678
  if (imageBlocks.length === 0 || !supportsImages) {
629
679
  if (omittedImages) {
@@ -1283,20 +1333,19 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1283
1333
  let strictFallbackErrorMessage: string | undefined;
1284
1334
  let dropFastMode = providerSessionState?.fastModeDisabled ?? false;
1285
1335
  let droppedForcedToolChoice = false;
1286
- const prepareParams = async (paramsOptions?: {
1287
- repairLatestAssistantThinking?: boolean;
1288
- dropForcedToolChoice?: boolean;
1289
- }): Promise<MessageCreateParamsStreaming> => {
1290
- let nextParams = buildParams(
1291
- model,
1292
- baseUrl,
1293
- context,
1294
- isOAuthToken,
1295
- options,
1296
- disableStrictTools,
1297
- paramsOptions?.repairLatestAssistantThinking === true,
1298
- );
1299
- if (paramsOptions?.dropForcedToolChoice === true) {
1336
+ let repairLatestAssistantThinking = false;
1337
+ let repairAllAssistantThinking = false;
1338
+ const prepareParams = async (): Promise<MessageCreateParamsStreaming> => {
1339
+ // Degradation state is cumulative: every fallback rebuild must merge all
1340
+ // repairs activated so far. Rebuilding from only the immediate call lets
1341
+ // a later strict/forced-tool/fast-mode fallback reintroduce the rejected
1342
+ // shape (e.g. invalid thinking signatures or forced tool_choice), and
1343
+ // the one-shot thinking-repair guard then blocks recovery.
1344
+ let nextParams = buildParams(model, baseUrl, context, isOAuthToken, options, disableStrictTools, {
1345
+ repairLatestAssistantThinking,
1346
+ repairAllAssistantThinking,
1347
+ });
1348
+ if (droppedForcedToolChoice) {
1300
1349
  delete nextParams.tool_choice;
1301
1350
  }
1302
1351
  if (disableStrictTools) {
@@ -1730,23 +1779,30 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1730
1779
  registryKey: resolveToolChoice(model, options?.toolChoice).registryKey,
1731
1780
  });
1732
1781
  droppedForcedToolChoice = true;
1733
- params = await prepareParams({ dropForcedToolChoice: true });
1782
+ params = await prepareParams();
1734
1783
  providerRetryAttempt = 0;
1735
1784
  resetOutputForRetry();
1736
1785
  continue;
1737
1786
  }
1787
+ const thinkingSignatureInvalid = isAnthropicThinkingSignatureInvalidError(streamFailure);
1738
1788
  if (
1739
1789
  !options?.fallbackManaged &&
1740
1790
  !thinkingRepairAttempted &&
1741
1791
  firstTokenTime === undefined &&
1742
- isAnthropicThinkingBlockMutationError(streamFailure)
1792
+ (thinkingSignatureInvalid || isAnthropicThinkingBlockMutationError(streamFailure))
1743
1793
  ) {
1744
- logger.debug("anthropic: repairing latest assistant thinking replay after provider rejection", {
1794
+ logger.debug("anthropic: repairing assistant thinking replay after provider rejection", {
1745
1795
  model: model.id,
1796
+ scope: thinkingSignatureInvalid ? "all" : "latest",
1746
1797
  error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure),
1747
1798
  });
1748
1799
  thinkingRepairAttempted = true;
1749
- params = await prepareParams({ repairLatestAssistantThinking: true });
1800
+ if (thinkingSignatureInvalid) {
1801
+ repairAllAssistantThinking = true;
1802
+ } else {
1803
+ repairLatestAssistantThinking = true;
1804
+ }
1805
+ params = await prepareParams();
1750
1806
  providerRetryAttempt = 0;
1751
1807
  resetOutputForRetry();
1752
1808
  continue;
@@ -2210,13 +2266,13 @@ function buildParams(
2210
2266
  isOAuthToken: boolean,
2211
2267
  options?: AnthropicOptions,
2212
2268
  disableStrictTools = false,
2213
- repairLatestAssistantThinking = false,
2269
+ thinkingRepair?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
2214
2270
  ): MessageCreateParamsStreaming {
2215
2271
  const { mode: cacheMode, cacheControl } = getCacheControl(model, baseUrl, options?.cacheRetention);
2216
2272
 
2217
2273
  const params: AnthropicSamplingParams = {
2218
2274
  model: model.id,
2219
- messages: convertAnthropicMessages(context.messages, model, isOAuthToken, { repairLatestAssistantThinking }),
2275
+ messages: convertAnthropicMessages(context.messages, model, isOAuthToken, thinkingRepair),
2220
2276
  max_tokens: options?.maxTokens || (model.maxTokens / 3) | 0,
2221
2277
  stream: true,
2222
2278
  };
@@ -2407,7 +2463,7 @@ export function convertAnthropicMessages(
2407
2463
  messages: Message[],
2408
2464
  model: Model<"anthropic-messages">,
2409
2465
  isOAuthToken: boolean,
2410
- options?: { repairLatestAssistantThinking?: boolean },
2466
+ options?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
2411
2467
  ): MessageParam[] {
2412
2468
  const params: MessageParam[] = [];
2413
2469
 
@@ -69,7 +69,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
69
69
  baseUrl.includes("api.anthropic.com") ||
70
70
  /(^|\/)claude[-.]/i.test(model.id) ||
71
71
  /(^|\/)anthropic\//i.test(model.id);
72
- const isAlibaba = provider === "alibaba-coding-plan" || baseUrl.includes("dashscope");
72
+ const isAlibaba = baseUrl.includes("dashscope");
73
73
  const isQwen = model.id.toLowerCase().includes("qwen");
74
74
  // DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
75
75
  // thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
@@ -244,7 +244,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
244
244
  requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
245
245
  openRouterRouting: undefined,
246
246
  vercelGatewayRouting: undefined,
247
- supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
247
+ supportsStrictMode: detectStrictModeSupport(provider, baseUrl) && !(isDeepseekFamily && isOpenRouter),
248
248
  extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
249
249
  toolStrictMode: isCerebras ? "all_strict" : "mixed",
250
250
  };
@@ -426,6 +426,7 @@ function getTrailingPartialDeepseekToken(text: string): string {
426
426
  return tail;
427
427
  }
428
428
 
429
+ const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
429
430
  const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
430
431
  "OpenAI completions stream timed out while waiting for the first event";
431
432
 
@@ -562,8 +563,10 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
562
563
  openaiStream = await createCompletionsStream("none");
563
564
  }
564
565
  }
566
+ const firstEventFallbackMs =
567
+ model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
565
568
  const firstEventWatchdog = createWatchdog(
566
- options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
569
+ options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
567
570
  () => abortTracker.abortLocally(firstEventTimeoutAbortError),
568
571
  );
569
572
  if (premiumRequestsTotal !== undefined) {
@@ -130,6 +130,7 @@ export interface OpenAIResponsesOptions extends StreamOptions {
130
130
  }
131
131
 
132
132
  const OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX = "openai-responses:";
133
+ const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
133
134
  const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
134
135
  "OpenAI responses stream timed out while waiting for the first event";
135
136
  const OPENAI_DEFAULT_BASE_URL = "https://api.openai.com/v1";
@@ -330,8 +331,10 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
330
331
  await notifyProviderResponse(options, response, model, request_id);
331
332
  return data;
332
333
  });
334
+ const firstEventFallbackMs =
335
+ model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
333
336
  const firstEventWatchdog = createWatchdog(
334
- options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
337
+ options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
335
338
  () => abortTracker.abortLocally(firstEventTimeoutAbortError),
336
339
  );
337
340
  if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
@@ -31,7 +31,7 @@ export function transformMessages<TApi extends Api>(
31
31
  messages: Message[],
32
32
  model: Model<TApi>,
33
33
  normalizeToolCallId?: (id: string, model: Model<TApi>, source: AssistantMessage) => string,
34
- options?: { repairLatestAssistantThinking?: boolean },
34
+ options?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
35
35
  ): Message[] {
36
36
  // Build a map of original tool call IDs to normalized IDs
37
37
  const toolCallIdMap = new Map<string, string>();
@@ -73,16 +73,29 @@ export function transformMessages<TApi extends Api>(
73
73
  // are kept so the second pass can either preserve real results or synthesize
74
74
  // an explicit aborted result without leaving dangling tool_use blocks.
75
75
  const hasPartialThinking = assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error";
76
- const dropLatestAssistantThinking =
77
- options?.repairLatestAssistantThinking === true &&
78
- index === latestAssistantIndex &&
76
+ // One-shot Anthropic replay repair. `repairLatestAssistantThinking` targets the
77
+ // "latest assistant message ... cannot be modified" 400; `repairAllAssistantThinking`
78
+ // targets the "Invalid `signature` in `thinking` block" 400, which can cite a block
79
+ // anywhere in the replayed history (e.g. after compaction/pruning rewrote an earlier
80
+ // turn), so the drop must apply to every assistant message. Within each
81
+ // message only blocks that would replay as native thinking/redacted_thinking
82
+ // are dropped; cross-model reasoning degrades to text and is preserved.
83
+ const dropAssistantThinkingForRepair =
84
+ (options?.repairAllAssistantThinking === true ||
85
+ (options?.repairLatestAssistantThinking === true && index === latestAssistantIndex)) &&
79
86
  model.api === "anthropic-messages" &&
80
87
  assistantMsg.api === "anthropic-messages";
81
88
 
82
89
  const transformedContent = assistantMsg.content.flatMap(block => {
83
90
  if (block.type === "thinking") {
84
- if (hasPartialThinking || dropLatestAssistantThinking) return [];
91
+ if (hasPartialThinking) return [];
85
92
  const sanitized = block;
93
+ // Repair must only drop blocks that would otherwise replay as native
94
+ // thinking. Cross-model/provider reasoning degrades to unsigned text
95
+ // below and was never replayed as a signed block, so it cannot be the
96
+ // signature failure — dropping it would silently lose valid context.
97
+ const replaysAsNativeThinking = mustPreserveLatestAnthropicThinking || isSameModel;
98
+ if (dropAssistantThinkingForRepair && replaysAsNativeThinking) return [];
86
99
  if (mustPreserveLatestAnthropicThinking) return sanitized;
87
100
  // For same model: keep thinking blocks with signatures (needed for replay)
88
101
  // even if the thinking text is empty (OpenAI encrypted reasoning)
@@ -97,7 +110,13 @@ export function transformMessages<TApi extends Api>(
97
110
  }
98
111
 
99
112
  if (block.type === "redactedThinking") {
100
- if (hasPartialThinking || dropLatestAssistantThinking) return [];
113
+ if (hasPartialThinking) return [];
114
+ // Same restriction as thinking blocks: cross-model/provider redacted
115
+ // blocks already drop below, so repair only needs to cover blocks that
116
+ // would replay as native redacted_thinking.
117
+ if (dropAssistantThinkingForRepair && (mustPreserveLatestAnthropicThinking || isSameModel)) {
118
+ return [];
119
+ }
101
120
  if (mustPreserveLatestAnthropicThinking) return block;
102
121
  if (isSameModel) return block;
103
122
  return [];
package/src/stream.ts CHANGED
@@ -85,7 +85,7 @@ function hasVertexAdcCredentials(): boolean {
85
85
  type KeyResolver = string | (() => string | undefined);
86
86
 
87
87
  const serviceProviderMap: Record<string, KeyResolver> = {
88
- "alibaba-coding-plan": "ALIBABA_CODING_PLAN_API_KEY",
88
+ "alibaba-token-plan": "ALIBABA_TOKEN_PLAN_API_KEY",
89
89
  openai: () => $credentialEnv("OPENAI_API_KEY"),
90
90
  google: "GEMINI_API_KEY",
91
91
  groq: "GROQ_API_KEY",
package/src/types.ts CHANGED
@@ -114,7 +114,7 @@ export interface ThinkingConfig {
114
114
  }
115
115
 
116
116
  export type KnownProvider =
117
- | "alibaba-coding-plan"
117
+ | "alibaba-token-plan"
118
118
  | "amazon-bedrock"
119
119
  | "azure-openai"
120
120
  | "anthropic"
@@ -709,10 +709,15 @@ export type TSchema = ZodType | TJsonSchema;
709
709
  /** Resolve parameter types for tool execution / handlers. */
710
710
  export type Static<S> = S extends ZodType ? z.infer<S> : S extends { static: infer T } ? T : unknown;
711
711
 
712
+ export type RawArgumentRejectionCode =
713
+ | "ask-intent-review-requires-positive-round"
714
+ | "ask-intent-contract-requires-non-empty-authority"
715
+ | "ask-deep-interview-metadata-requires-deep-interview-gate";
716
+
712
717
  export type RawArgumentValidationResult =
713
718
  | { outcome: "passthrough" }
714
719
  | { outcome: "accept"; arguments: ToolCall["arguments"] }
715
- | { outcome: "reject" };
720
+ | { outcome: "reject"; code?: RawArgumentRejectionCode };
716
721
 
717
722
  export interface Tool<TParameters extends TSchema = TSchema> {
718
723
  name: string;
@@ -1,10 +1,11 @@
1
1
  /**
2
- * Alibaba Coding Plan login flow.
2
+ * Alibaba Token Plan login flow.
3
3
  *
4
- * Alibaba Coding Plan provides OpenAI-compatible models via https://coding-intl.dashscope.aliyuncs.com/v1.
4
+ * Alibaba Token Plan provides OpenAI-compatible models via
5
+ * https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1.
5
6
  *
6
7
  * This is not OAuth - it's a simple API key flow:
7
- * 1. Open browser to Alibaba Cloud DashScope API key settings
8
+ * 1. Open browser to Alibaba Cloud Model Studio console
8
9
  * 2. User copies their API key
9
10
  * 3. User pastes the API key into the CLI
10
11
  */
@@ -13,27 +14,27 @@ import { validateOpenAICompatibleApiKey } from "./api-key-validation";
13
14
  import type { OAuthController } from "./types";
14
15
 
15
16
  const AUTH_URL = "https://modelstudio.console.alibabacloud.com/";
16
- const API_BASE_URL = "https://coding-intl.dashscope.aliyuncs.com/v1";
17
- const VALIDATION_MODEL = "qwen3.5-plus";
17
+ const API_BASE_URL = "https://token-plan.ap-southeast-1.maas.aliyuncs.com/compatible-mode/v1";
18
+ const VALIDATION_MODEL = "deepseek-v4-pro";
18
19
 
19
20
  /**
20
- * Login to Alibaba Coding Plan.
21
+ * Login to Alibaba Token Plan.
21
22
  *
22
23
  * Opens browser to API keys page, prompts user to paste their API key.
23
24
  * Returns the API key directly (not OAuthCredentials - this isn't OAuth).
24
25
  */
25
- export async function loginAlibabaCodingPlan(options: OAuthController): Promise<string> {
26
+ export async function loginAlibabaTokenPlan(options: OAuthController): Promise<string> {
26
27
  if (!options.onPrompt) {
27
- throw new Error("Alibaba Coding Plan login requires onPrompt callback");
28
+ throw new Error("Alibaba Token Plan login requires onPrompt callback");
28
29
  }
29
30
 
30
31
  options.onAuth?.({
31
32
  url: AUTH_URL,
32
- instructions: "Copy your API key from the Alibaba Cloud DashScope console",
33
+ instructions: "Copy your API key from the Alibaba Cloud Model Studio console",
33
34
  });
34
35
 
35
36
  const apiKey = await options.onPrompt({
36
- message: "Paste your Alibaba Coding Plan API key",
37
+ message: "Paste your Alibaba Token Plan API key",
37
38
  placeholder: "sk-...",
38
39
  });
39
40
 
@@ -48,7 +49,7 @@ export async function loginAlibabaCodingPlan(options: OAuthController): Promise<
48
49
 
49
50
  options.onProgress?.("Validating API key...");
50
51
  await validateOpenAICompatibleApiKey({
51
- provider: "Alibaba Coding Plan",
52
+ provider: "Alibaba Token Plan",
52
53
  apiKey: trimmed,
53
54
  baseUrl: API_BASE_URL,
54
55
  model: VALIDATION_MODEL,
@@ -16,8 +16,8 @@ const builtInOAuthProviders: OAuthProviderInfo[] = [
16
16
  available: true,
17
17
  },
18
18
  {
19
- id: "alibaba-coding-plan",
20
- name: "Alibaba Coding Plan",
19
+ id: "alibaba-token-plan",
20
+ name: "Alibaba Token Plan",
21
21
  available: true,
22
22
  },
23
23
  {
@@ -371,6 +371,7 @@ export async function refreshOAuthToken(
371
371
  case "together":
372
372
  case "litellm":
373
373
  case "lm-studio":
374
+ case "alibaba-token-plan":
374
375
  case "ollama":
375
376
  case "ollama-cloud":
376
377
  case "xiaomi":
@@ -9,7 +9,7 @@ export type OAuthCredentials = {
9
9
  };
10
10
 
11
11
  export type OAuthProvider =
12
- | "alibaba-coding-plan"
12
+ | "alibaba-token-plan"
13
13
  | "anthropic"
14
14
  | "cerebras"
15
15
  | "cloudflare-ai-gateway"
@@ -25,7 +25,7 @@
25
25
  import { structuredCloneJSON } from "@gajae-code/utils";
26
26
  import type { ZodType } from "zod/v4";
27
27
  import type { $ZodIssue as ZodIssue } from "zod/v4/core";
28
- import type { Tool, ToolCall } from "../types";
28
+ import type { RawArgumentRejectionCode, Tool, ToolCall } from "../types";
29
29
  import { upgradeJsonSchemaTo202012 } from "./schema/draft";
30
30
  import {
31
31
  isJsonSchemaValueValid,
@@ -958,6 +958,15 @@ export function validateToolCall(tools: Tool[], toolCall: ToolCall): ToolCall["a
958
958
  return validateToolArguments(tool, toolCall);
959
959
  }
960
960
 
961
+ const RAW_ARGUMENT_REJECTION_MESSAGES: Record<RawArgumentRejectionCode, string> = {
962
+ "ask-intent-review-requires-positive-round":
963
+ "deepInterview.intent_review is post-Round-0 only and requires a positive round",
964
+ "ask-intent-contract-requires-non-empty-authority":
965
+ "deepInterview.intent_contract requires non-empty items and confirmation_options",
966
+ "ask-deep-interview-metadata-requires-deep-interview-gate":
967
+ "deepInterview metadata cannot be combined with a non-deep-interview workflowGate",
968
+ };
969
+
961
970
  /**
962
971
  * Validates tool call arguments against the tool's schema (Zod or plain JSON
963
972
  * Schema). Applies LLM-quirk coercions (numeric strings, JSON-string
@@ -969,7 +978,13 @@ export function validateToolArguments(tool: Tool, toolCall: ToolCall): ToolCall[
969
978
  const originalArgs = toolCall.arguments;
970
979
  const rawValidation = tool.rawArgumentValidation?.(originalArgs);
971
980
  if (rawValidation?.outcome === "reject") {
972
- throw new Error(`Validation failed for tool "${toolCall.name}": raw arguments rejected before coercion`);
981
+ const base = `Validation failed for tool "${toolCall.name}": raw arguments rejected before coercion`;
982
+ const code = rawValidation.code;
983
+ const correction =
984
+ typeof code === "string" && Object.hasOwn(RAW_ARGUMENT_REJECTION_MESSAGES, code)
985
+ ? RAW_ARGUMENT_REJECTION_MESSAGES[code as RawArgumentRejectionCode]
986
+ : undefined;
987
+ throw new Error(correction ? `${base}; ${correction}` : base);
973
988
  }
974
989
  const rawArgs = rawValidation?.outcome === "accept" ? rawValidation.arguments : originalArgs;
975
990
  const ctx = getValidationContext(tool);
package/src/utils.ts CHANGED
@@ -271,17 +271,53 @@ function normalizeResponsesImageUrlForReplay(value: unknown): NormalizedResponse
271
271
  return { imageUrl: stringifyResponsesStringParamForReplay(value) };
272
272
  }
273
273
 
274
+ /**
275
+ * OpenAI Responses `input_image.image_url` must be a fetchable HTTP(S) URL or an
276
+ * image data URI. Session resident-blob materialization may leave a human-readable
277
+ * placeholder like `[Session resident imageUrl blob missing: sha256:…; …]` in this
278
+ * field; replaying that string as `image_url` makes Codex reject the entire turn
279
+ * with `invalid_value` (#2924).
280
+ */
281
+ function isProviderSafeResponsesImageUrl(value: string): boolean {
282
+ const url = value.trim();
283
+ if (url.length === 0) return false;
284
+ if (url.startsWith("https://") || url.startsWith("http://")) return true;
285
+ // Accept only image data URIs — other data: schemes are not valid image inputs.
286
+ if (url.startsWith("data:image/")) return true;
287
+ return false;
288
+ }
289
+
290
+ function hasNonEmptyResponsesFileId(part: Record<string, unknown>): boolean {
291
+ return typeof part.file_id === "string" && part.file_id.trim().length > 0;
292
+ }
293
+
274
294
  function sanitizeResponsesMessageContentForReplay(content: unknown): unknown {
275
295
  if (typeof content === "string") return neutralizeReservedControlTokens(content.toWellFormed());
276
296
  if (!Array.isArray(content)) return content;
277
- return content.map(part => {
278
- if (!part || typeof part !== "object") return part;
297
+ const sanitizedContent: unknown[] = [];
298
+ for (const part of content) {
299
+ if (!part || typeof part !== "object") {
300
+ sanitizedContent.push(part);
301
+ continue;
302
+ }
279
303
  const sanitizedPart = { ...(part as Record<string, unknown>) };
280
304
  if ("text" in sanitizedPart) {
281
305
  sanitizedPart.text = normalizeResponsesMessageTextForReplay(sanitizedPart.text);
282
306
  }
283
307
  if ("image_url" in sanitizedPart) {
284
308
  const normalizedImageUrl = normalizeResponsesImageUrlForReplay(sanitizedPart.image_url);
309
+ if (!isProviderSafeResponsesImageUrl(normalizedImageUrl.imageUrl)) {
310
+ // Keep the part when a provider file_id can stand alone; otherwise drop
311
+ // only this image part so neighboring text/history still replays.
312
+ if (!hasNonEmptyResponsesFileId(sanitizedPart)) continue;
313
+ delete sanitizedPart.image_url;
314
+ if (sanitizedPart.type === "image_url") sanitizedPart.type = "input_image";
315
+ if ("detail" in sanitizedPart && !isResponsesImageDetail(sanitizedPart.detail)) {
316
+ delete sanitizedPart.detail;
317
+ }
318
+ sanitizedContent.push(sanitizedPart);
319
+ continue;
320
+ }
285
321
  sanitizedPart.image_url = normalizedImageUrl.imageUrl;
286
322
  if (sanitizedPart.type === "image_url") {
287
323
  sanitizedPart.type = "input_image";
@@ -292,8 +328,9 @@ function sanitizeResponsesMessageContentForReplay(content: unknown): unknown {
292
328
  delete sanitizedPart.detail;
293
329
  }
294
330
  }
295
- return sanitizedPart;
296
- });
331
+ sanitizedContent.push(sanitizedPart);
332
+ }
333
+ return sanitizedContent;
297
334
  }
298
335
 
299
336
  function sanitizeResponsesStringFieldsForReplay(item: Record<string, unknown>): void {
@@ -1,18 +0,0 @@
1
- /**
2
- * Alibaba Coding Plan login flow.
3
- *
4
- * Alibaba Coding Plan provides OpenAI-compatible models via https://coding-intl.dashscope.aliyuncs.com/v1.
5
- *
6
- * This is not OAuth - it's a simple API key flow:
7
- * 1. Open browser to Alibaba Cloud DashScope API key settings
8
- * 2. User copies their API key
9
- * 3. User pastes the API key into the CLI
10
- */
11
- import type { OAuthController } from "./types";
12
- /**
13
- * Login to Alibaba Coding Plan.
14
- *
15
- * Opens browser to API keys page, prompts user to paste their API key.
16
- * Returns the API key directly (not OAuthCredentials - this isn't OAuth).
17
- */
18
- export declare function loginAlibabaCodingPlan(options: OAuthController): Promise<string>;