@sayknow-cli/ai 0.4.6 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +6 -0
  2. package/dist/types/auth-broker/client.d.ts +2 -1
  3. package/dist/types/auth-broker/remote-store.d.ts +2 -1
  4. package/dist/types/auth-broker/types.d.ts +3 -1
  5. package/dist/types/auth-broker/wire-schemas.d.ts +68 -0
  6. package/dist/types/auth-storage.d.ts +21 -1
  7. package/dist/types/provider-models/openai-compat.d.ts +19 -2
  8. package/dist/types/providers/anthropic.d.ts +11 -1
  9. package/dist/types/providers/azure-openai-responses.d.ts +6 -1
  10. package/dist/types/providers/google-auth.d.ts +2 -0
  11. package/dist/types/providers/google-gemini-headers.d.ts +1 -1
  12. package/dist/types/providers/google-vertex.d.ts +2 -0
  13. package/dist/types/providers/openai-codex-responses.d.ts +4 -0
  14. package/dist/types/providers/openai-completions.d.ts +2 -0
  15. package/dist/types/providers/openai-responses.d.ts +2 -0
  16. package/dist/types/providers/register-builtins.d.ts +8 -0
  17. package/dist/types/providers/transform-messages.d.ts +1 -0
  18. package/dist/types/types.d.ts +3 -1
  19. package/dist/types/usage/grok-cli.d.ts +3 -1
  20. package/dist/types/usage/kimi.d.ts +2 -0
  21. package/dist/types/utils/anthropic-auth.d.ts +8 -0
  22. package/dist/types/utils/foundry.d.ts +10 -0
  23. package/dist/types/utils/http-inspector.d.ts +13 -0
  24. package/dist/types/utils/idle-iterator.d.ts +3 -2
  25. package/dist/types/utils/oauth/alibaba-token-plan.d.ts +19 -0
  26. package/dist/types/utils/oauth/bizrouter.d.ts +1 -0
  27. package/dist/types/utils/oauth/opengateway.d.ts +1 -0
  28. package/dist/types/utils/oauth/types.d.ts +1 -1
  29. package/package.json +2 -2
  30. package/src/auth-broker/client.ts +13 -0
  31. package/src/auth-broker/refresher.ts +1 -0
  32. package/src/auth-broker/remote-store.ts +25 -0
  33. package/src/auth-broker/server.ts +10 -2
  34. package/src/auth-broker/types.ts +4 -0
  35. package/src/auth-broker/wire-schemas.ts +17 -1
  36. package/src/auth-storage.ts +234 -45
  37. package/src/cli.ts +2 -0
  38. package/src/model-thinking.ts +11 -3
  39. package/src/models.json +3289 -486
  40. package/src/provider-models/descriptors.ts +19 -6
  41. package/src/provider-models/openai-compat.ts +99 -18
  42. package/src/providers/amazon-bedrock.ts +4 -0
  43. package/src/providers/anthropic.ts +131 -28
  44. package/src/providers/azure-openai-responses.ts +16 -3
  45. package/src/providers/google-auth.ts +13 -2
  46. package/src/providers/google-gemini-headers.ts +1 -1
  47. package/src/providers/google-vertex.ts +7 -2
  48. package/src/providers/openai-anthropic-shim.ts +4 -0
  49. package/src/providers/openai-codex-responses.ts +52 -10
  50. package/src/providers/openai-completions-compat.ts +2 -2
  51. package/src/providers/openai-completions.ts +20 -3
  52. package/src/providers/openai-responses.ts +17 -10
  53. package/src/providers/register-builtins.ts +21 -2
  54. package/src/providers/transform-messages.ts +25 -6
  55. package/src/stream.ts +3 -1
  56. package/src/types.ts +9 -2
  57. package/src/usage/claude.ts +21 -3
  58. package/src/usage/grok-cli.ts +12 -1
  59. package/src/usage/kimi.ts +16 -2
  60. package/src/utils/anthropic-auth.ts +11 -3
  61. package/src/utils/foundry.ts +12 -2
  62. package/src/utils/http-inspector.ts +77 -0
  63. package/src/utils/idle-iterator.ts +20 -7
  64. package/src/utils/oauth/{alibaba-coding-plan.ts → alibaba-token-plan.ts} +12 -11
  65. package/src/utils/oauth/bizrouter.ts +15 -0
  66. package/src/utils/oauth/index.ts +15 -2
  67. package/src/utils/oauth/opengateway.ts +15 -0
  68. package/src/utils/oauth/types.ts +3 -1
  69. package/src/utils/validation.ts +17 -2
  70. package/src/utils.ts +41 -4
  71. package/dist/types/utils/oauth/alibaba-coding-plan.d.ts +0 -18
@@ -2,7 +2,7 @@ import * as os from "node:os";
2
2
  import { scheduler } from "node:timers/promises";
3
3
  import {
4
4
  $env,
5
- $flag,
5
+ $pickflag,
6
6
  asRecord,
7
7
  extractHttpStatusFromError,
8
8
  fetchWithRetry,
@@ -98,7 +98,7 @@ export interface OpenAICodexResponsesOptions extends StreamOptions {
98
98
  serviceTier?: ServiceTier;
99
99
  }
100
100
 
101
- const CODEX_DEBUG = $flag("PI_CODEX_DEBUG");
101
+ const CODEX_DEBUG = $pickflag("SKC_OPENAI_CODE_DEBUG", "PI_CODEX_DEBUG");
102
102
  const CODEX_MAX_RETRIES = 5;
103
103
  const CODEX_RETRY_DELAY_MS = 500;
104
104
  const CODEX_WEBSOCKET_CONNECT_TIMEOUT_MS = 10000;
@@ -133,6 +133,31 @@ const CODEX_WEBSOCKET_FATAL_PATTERNS = ["websocket error:", "websocket closed be
133
133
  /** Max total time to spend retrying 429s with server-provided delays (5 minutes). */
134
134
  const CODEX_RATE_LIMIT_BUDGET_MS = 5 * 60 * 1000;
135
135
 
136
+ /**
137
+ * Tool names the Codex backend reserves for its own namespaces. Sending a
138
+ * function tool under one of these names is rejected with
139
+ * `Function 'computer.computer' not allowed in namespace 'computer'`.
140
+ * These are renamed on the wire and mapped back on receive so the internal
141
+ * tool name stays canonical everywhere else in the harness.
142
+ */
143
+ const CODEX_RESERVED_TOOL_WIRE_NAMES: ReadonlyMap<string, string> = new Map([
144
+ ["browser", "browser_tool"],
145
+ ["computer", "computer_tool"],
146
+ ]);
147
+ const CODEX_CANONICAL_TOOL_NAMES: ReadonlyMap<string, string> = new Map(
148
+ Array.from(CODEX_RESERVED_TOOL_WIRE_NAMES, ([canonical, wire]) => [wire, canonical]),
149
+ );
150
+
151
+ /** Maps a canonical tool name to the name Codex accepts on the wire. */
152
+ export function codexToolWireName(name: string): string {
153
+ return CODEX_RESERVED_TOOL_WIRE_NAMES.get(name) ?? name;
154
+ }
155
+
156
+ /** Maps a Codex wire tool name back to the canonical harness tool name. */
157
+ export function codexToolCanonicalName(wireName: string): string {
158
+ return CODEX_CANONICAL_TOOL_NAMES.get(wireName) ?? wireName;
159
+ }
160
+
136
161
  const CODEX_PROGRESS_EVENT_TYPES = new Set([
137
162
  "response.created",
138
163
  "response.output_item.added",
@@ -299,24 +324,34 @@ function parseCodexPositiveInteger(value: string | undefined, fallback: number):
299
324
  }
300
325
 
301
326
  function isCodexWebSocketEnvEnabled(): boolean {
302
- return $flag("PI_CODEX_WEBSOCKET");
327
+ return $pickflag("SKC_OPENAI_CODE_WEBSOCKET", "PI_CODEX_WEBSOCKET");
303
328
  }
304
329
 
305
330
  function getCodexWebSocketRetryBudget(options?: Pick<OpenAICodexResponsesOptions, "streamMaxRetries">): number {
306
331
  if (options?.streamMaxRetries !== undefined) {
307
332
  return resolveRetryBudget(options.streamMaxRetries, CODEX_WEBSOCKET_RETRY_BUDGET);
308
333
  }
309
- return parseCodexNonNegativeInteger($env.PI_CODEX_WEBSOCKET_RETRY_BUDGET, CODEX_WEBSOCKET_RETRY_BUDGET);
334
+ return parseCodexNonNegativeInteger(
335
+ $env.SKC_OPENAI_CODE_WEBSOCKET_RETRY_BUDGET ?? $env.PI_CODEX_WEBSOCKET_RETRY_BUDGET,
336
+ CODEX_WEBSOCKET_RETRY_BUDGET,
337
+ );
310
338
  }
311
339
 
312
340
  function getCodexWebSocketRetryDelayMs(retry: number): number {
313
- const baseDelay = parseCodexPositiveInteger($env.PI_CODEX_WEBSOCKET_RETRY_DELAY_MS, CODEX_RETRY_DELAY_MS);
341
+ const baseDelay = parseCodexPositiveInteger(
342
+ $env.SKC_OPENAI_CODE_WEBSOCKET_RETRY_DELAY_MS ?? $env.PI_CODEX_WEBSOCKET_RETRY_DELAY_MS,
343
+ CODEX_RETRY_DELAY_MS,
344
+ );
314
345
  return baseDelay * Math.max(1, retry);
315
346
  }
316
347
 
317
348
  function getCodexWebSocketIdleTimeoutMs(overrideMs?: number): number {
318
349
  return (
319
- overrideMs ?? parseCodexPositiveInteger($env.PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS, CODEX_WEBSOCKET_IDLE_TIMEOUT_MS)
350
+ overrideMs ??
351
+ parseCodexPositiveInteger(
352
+ $env.SKC_OPENAI_CODE_WEBSOCKET_IDLE_TIMEOUT_MS ?? $env.PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS,
353
+ CODEX_WEBSOCKET_IDLE_TIMEOUT_MS,
354
+ )
320
355
  );
321
356
  }
322
357
 
@@ -467,7 +502,7 @@ export function normalizeCodexToolChoice(
467
502
  : undefined;
468
503
  return customTool
469
504
  ? { type: "custom", name: customTool.customWireName ?? customTool.name }
470
- : { type: "function", name };
505
+ : { type: "function", name: codexToolWireName(name) };
471
506
  };
472
507
  if (choice.type === "function") {
473
508
  if ("function" in choice && choice.function?.name) {
@@ -1090,7 +1125,7 @@ function createOutputBlockForItem(item: CodexEventItem): CodexOutputBlock | null
1090
1125
  return {
1091
1126
  type: "toolCall",
1092
1127
  id: encodeResponsesToolCallId(item.call_id, item.id),
1093
- name: item.name,
1128
+ name: codexToolCanonicalName(item.name),
1094
1129
  arguments: {},
1095
1130
  partialJson: item.arguments || "",
1096
1131
  };
@@ -1348,7 +1383,7 @@ function handleOutputItemDone(
1348
1383
  const toolCall: ToolCall = {
1349
1384
  type: "toolCall",
1350
1385
  id,
1351
- name: item.name,
1386
+ name: codexToolCanonicalName(item.name),
1352
1387
  arguments: parseStreamingJson(item.arguments || "{}"),
1353
1388
  };
1354
1389
  runtime.canSafelyReplayWebsocketOverSse = false;
@@ -2690,6 +2725,13 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
2690
2725
  true,
2691
2726
  customCallIds,
2692
2727
  );
2728
+ for (const item of outputItems) {
2729
+ // Reconstructed (non-raw) history carries canonical tool names; the
2730
+ // wire form has to match the renamed `tools` entries.
2731
+ if (item.type === "function_call" && typeof item.name === "string") {
2732
+ item.name = codexToolWireName(item.name);
2733
+ }
2734
+ }
2693
2735
  if (outputItems.length > 0) {
2694
2736
  messages.push(...outputItems);
2695
2737
  }
@@ -2776,7 +2818,7 @@ export function convertOpenAICodexResponsesTools(
2776
2818
  const { schema: parameters, strict: effectiveStrict } = adaptSchemaForStrict(baseParameters, strict);
2777
2819
  return {
2778
2820
  type: "function",
2779
- name: tool.name,
2821
+ name: codexToolWireName(tool.name),
2780
2822
  description: tool.description || "",
2781
2823
  parameters,
2782
2824
  ...(effectiveStrict && { strict: true }),
@@ -69,7 +69,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
69
69
  baseUrl.includes("api.anthropic.com") ||
70
70
  /(^|\/)claude[-.]/i.test(model.id) ||
71
71
  /(^|\/)anthropic\//i.test(model.id);
72
- const isAlibaba = provider === "alibaba-coding-plan" || baseUrl.includes("dashscope");
72
+ const isAlibaba = baseUrl.includes("dashscope");
73
73
  const isQwen = model.id.toLowerCase().includes("qwen");
74
74
  // DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
75
75
  // thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
@@ -244,7 +244,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
244
244
  requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning,
245
245
  openRouterRouting: undefined,
246
246
  vercelGatewayRouting: undefined,
247
- supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
247
+ supportsStrictMode: detectStrictModeSupport(provider, baseUrl) && !(isDeepseekFamily && isOpenRouter),
248
248
  extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
249
249
  toolStrictMode: isCerebras ? "all_strict" : "mixed",
250
250
  };
@@ -1,4 +1,4 @@
1
- import { $credentialEnv, $env, $inheritedEnv, extractHttpStatusFromError, logger } from "@sayknow-cli/utils";
1
+ import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@sayknow-cli/utils";
2
2
  import OpenAI from "openai";
3
3
  import type {
4
4
  ChatCompletionAssistantMessageParam,
@@ -48,6 +48,7 @@ import {
48
48
  import {
49
49
  createWatchdog,
50
50
  getOpenAIStreamIdleTimeoutMs,
51
+ getProviderFirstEventTimeoutFallbackMs,
51
52
  getStreamFirstEventTimeoutMs,
52
53
  iterateWithIdleTimeout,
53
54
  } from "../utils/idle-iterator";
@@ -99,7 +100,9 @@ function resolveOpenAIProviderBaseUrl(
99
100
  authCredentialType: "api_key" | "oauth" | undefined,
100
101
  ): string {
101
102
  if (authCredentialType === "oauth") return OPENAI_DEFAULT_BASE_URL;
102
- const envBaseUrl = $inheritedEnv("OPENAI_BASE_URL") ?? $env.OPENAI_BASE_URL?.trim();
103
+ // Trusted sources only: this base URL becomes the request endpoint that carries
104
+ // the OpenAI credential, and `$env` merges the caller's `cwd/.env`.
105
+ const envBaseUrl = $credentialEnv("OPENAI_BASE_URL");
103
106
  const configuredBaseUrl = baseUrl?.trim();
104
107
  if (envBaseUrl && (!configuredBaseUrl || isDefaultOpenAIBaseUrl(configuredBaseUrl))) {
105
108
  return envBaseUrl;
@@ -107,6 +110,14 @@ function resolveOpenAIProviderBaseUrl(
107
110
  return configuredBaseUrl || envBaseUrl || OPENAI_DEFAULT_BASE_URL;
108
111
  }
109
112
 
113
+ /** Test seam: the provider base URL as resolved from trusted env. */
114
+ export function resolveOpenAICompletionsBaseUrlForTest(
115
+ baseUrl: string | undefined,
116
+ authCredentialType: "api_key" | "oauth" | undefined,
117
+ ): string {
118
+ return resolveOpenAIProviderBaseUrl(baseUrl, authCredentialType);
119
+ }
120
+
110
121
  /**
111
122
  * Normalize tool call ID for Mistral.
112
123
  * Mistral requires tool IDs to be exactly 9 alphanumeric characters (a-z, A-Z, 0-9).
@@ -411,6 +422,8 @@ function getTrailingPartialDeepseekToken(text: string): string {
411
422
  return tail;
412
423
  }
413
424
 
425
+ const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
426
+
414
427
  const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
415
428
  "OpenAI completions stream timed out while waiting for the first event";
416
429
 
@@ -537,8 +550,12 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
537
550
  openaiStream = await createCompletionsStream("none");
538
551
  }
539
552
  }
553
+ const firstEventFallbackMs =
554
+ model.provider === "alibaba-token-plan"
555
+ ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS
556
+ : getProviderFirstEventTimeoutFallbackMs(model.provider);
540
557
  const firstEventWatchdog = createWatchdog(
541
- options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
558
+ options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
542
559
  () => abortTracker.abortLocally(firstEventTimeoutAbortError),
543
560
  );
544
561
  if (premiumRequestsTotal !== undefined) {
@@ -1,11 +1,4 @@
1
- import {
2
- $credentialEnv,
3
- $env,
4
- $inheritedEnv,
5
- extractHttpStatusFromError,
6
- logger,
7
- structuredCloneJSON,
8
- } from "@sayknow-cli/utils";
1
+ import { $credentialEnv, extractHttpStatusFromError, logger, structuredCloneJSON } from "@sayknow-cli/utils";
9
2
  import OpenAI from "openai";
10
3
  import type {
11
4
  Tool as OpenAITool,
@@ -130,6 +123,7 @@ export interface OpenAIResponsesOptions extends StreamOptions {
130
123
  }
131
124
 
132
125
  const OPENAI_RESPONSES_PROVIDER_SESSION_STATE_PREFIX = "openai-responses:";
126
+ const ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS = 300_000;
133
127
  const OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
134
128
  "OpenAI responses stream timed out while waiting for the first event";
135
129
  const OPENAI_DEFAULT_BASE_URL = "https://api.openai.com/v1";
@@ -158,7 +152,10 @@ function resolveOpenAIProviderBaseUrl(
158
152
  authCredentialType: "api_key" | "oauth" | undefined,
159
153
  ): string {
160
154
  if (authCredentialType === "oauth") return OPENAI_DEFAULT_BASE_URL;
161
- const envBaseUrl = $inheritedEnv("OPENAI_BASE_URL") ?? $env.OPENAI_BASE_URL?.trim();
155
+ // Trusted sources only: this base URL becomes the request endpoint that carries
156
+ // the OpenAI credential, and `$env` merges the caller's `cwd/.env`, so reading it
157
+ // there would let repository content redirect authenticated traffic.
158
+ const envBaseUrl = $credentialEnv("OPENAI_BASE_URL");
162
159
  const configuredBaseUrl = baseUrl?.trim();
163
160
  if (envBaseUrl && (!configuredBaseUrl || isDefaultOpenAIBaseUrl(configuredBaseUrl))) {
164
161
  return envBaseUrl;
@@ -166,6 +163,14 @@ function resolveOpenAIProviderBaseUrl(
166
163
  return configuredBaseUrl || envBaseUrl || OPENAI_DEFAULT_BASE_URL;
167
164
  }
168
165
 
166
+ /** Test seam: the provider base URL as resolved from trusted env. */
167
+ export function resolveOpenAIProviderBaseUrlForTest(
168
+ baseUrl: string | undefined,
169
+ authCredentialType: "api_key" | "oauth" | undefined,
170
+ ): string {
171
+ return resolveOpenAIProviderBaseUrl(baseUrl, authCredentialType);
172
+ }
173
+
169
174
  const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES = new Set([
170
175
  "response.created",
171
176
  "response.output_item.added",
@@ -330,8 +335,10 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
330
335
  await notifyProviderResponse(options, response, model, request_id);
331
336
  return data;
332
337
  });
338
+ const firstEventFallbackMs =
339
+ model.provider === "alibaba-token-plan" ? ALIBABA_TOKEN_PLAN_FIRST_EVENT_TIMEOUT_MS : undefined;
333
340
  const firstEventWatchdog = createWatchdog(
334
- options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
341
+ options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
335
342
  () => abortTracker.abortLocally(firstEventTimeoutAbortError),
336
343
  );
337
344
  if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
@@ -190,6 +190,22 @@ interface LazyStreamLimits {
190
190
  const GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS: LazyStreamLimits = {
191
191
  defaultFirstEventTimeoutMs: 300_000,
192
192
  };
193
+ const SLOW_FIRST_EVENT_PROVIDERS = new Set(["alibaba-token-plan", "kimi-code"]);
194
+
195
+ /**
196
+ * Resolves the first-event timeout fallback for the outer lazy-stream watchdog.
197
+ * A configured wrapper-specific fallback (from `LazyStreamLimits`) always wins;
198
+ * otherwise providers known to have slow first events get a five-minute floor
199
+ * matching their inner provider-level override. Returns `undefined` for
200
+ * providers that should use the shared default.
201
+ */
202
+ export function resolveLazyStreamFirstEventFallbackMs(
203
+ provider: string,
204
+ configuredFallbackMs?: number,
205
+ ): number | undefined {
206
+ if (configuredFallbackMs !== undefined) return configuredFallbackMs;
207
+ return SLOW_FIRST_EVENT_PROVIDERS.has(provider) ? 300_000 : undefined;
208
+ }
193
209
 
194
210
  function forwardStream<TApi extends Api>(
195
211
  target: EventStreamImpl,
@@ -202,11 +218,14 @@ function forwardStream<TApi extends Api>(
202
218
  (async () => {
203
219
  try {
204
220
  const idleTimeoutMs = options.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs(limits?.defaultIdleTimeoutMs);
221
+ const firstEventFallbackMs = resolveLazyStreamFirstEventFallbackMs(
222
+ model.provider,
223
+ limits?.defaultFirstEventTimeoutMs,
224
+ );
205
225
  const watchedSource = iterateWithIdleTimeout(source, {
206
226
  idleTimeoutMs,
207
227
  firstItemTimeoutMs:
208
- options.streamFirstEventTimeoutMs ??
209
- getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs),
228
+ options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs, firstEventFallbackMs),
210
229
  errorMessage: LAZY_STREAM_IDLE_TIMEOUT_ERROR,
211
230
  firstItemErrorMessage: LAZY_STREAM_FIRST_EVENT_TIMEOUT_ERROR,
212
231
  onIdle: () => abortTracker.abortLocally(new Error(LAZY_STREAM_IDLE_TIMEOUT_ERROR)),
@@ -31,7 +31,7 @@ export function transformMessages<TApi extends Api>(
31
31
  messages: Message[],
32
32
  model: Model<TApi>,
33
33
  normalizeToolCallId?: (id: string, model: Model<TApi>, source: AssistantMessage) => string,
34
- options?: { repairLatestAssistantThinking?: boolean },
34
+ options?: { repairLatestAssistantThinking?: boolean; repairAllAssistantThinking?: boolean },
35
35
  ): Message[] {
36
36
  // Build a map of original tool call IDs to normalized IDs
37
37
  const toolCallIdMap = new Map<string, string>();
@@ -73,16 +73,29 @@ export function transformMessages<TApi extends Api>(
73
73
  // are kept so the second pass can either preserve real results or synthesize
74
74
  // an explicit aborted result without leaving dangling tool_use blocks.
75
75
  const hasPartialThinking = assistantMsg.stopReason === "aborted" || assistantMsg.stopReason === "error";
76
- const dropLatestAssistantThinking =
77
- options?.repairLatestAssistantThinking === true &&
78
- index === latestAssistantIndex &&
76
+ // One-shot Anthropic replay repair. `repairLatestAssistantThinking` targets the
77
+ // "latest assistant message ... cannot be modified" 400; `repairAllAssistantThinking`
78
+ // targets the "Invalid `signature` in `thinking` block" 400, which can cite a block
79
+ // anywhere in the replayed history (e.g. after compaction/pruning rewrote an earlier
80
+ // turn), so the drop must apply to every assistant message. Within each
81
+ // message only blocks that would replay as native thinking/redacted_thinking
82
+ // are dropped; cross-model reasoning degrades to text and is preserved.
83
+ const dropAssistantThinkingForRepair =
84
+ (options?.repairAllAssistantThinking === true ||
85
+ (options?.repairLatestAssistantThinking === true && index === latestAssistantIndex)) &&
79
86
  model.api === "anthropic-messages" &&
80
87
  assistantMsg.api === "anthropic-messages";
81
88
 
82
89
  const transformedContent = assistantMsg.content.flatMap(block => {
83
90
  if (block.type === "thinking") {
84
- if (hasPartialThinking || dropLatestAssistantThinking) return [];
91
+ if (hasPartialThinking) return [];
85
92
  const sanitized = block;
93
+ // Repair must only drop blocks that would otherwise replay as native
94
+ // thinking. Cross-model/provider reasoning degrades to unsigned text
95
+ // below and was never replayed as a signed block, so it cannot be the
96
+ // signature failure — dropping it would silently lose valid context.
97
+ const replaysAsNativeThinking = mustPreserveLatestAnthropicThinking || isSameModel;
98
+ if (dropAssistantThinkingForRepair && replaysAsNativeThinking) return [];
86
99
  if (mustPreserveLatestAnthropicThinking) return sanitized;
87
100
  // For same model: keep thinking blocks with signatures (needed for replay)
88
101
  // even if the thinking text is empty (OpenAI encrypted reasoning)
@@ -97,7 +110,13 @@ export function transformMessages<TApi extends Api>(
97
110
  }
98
111
 
99
112
  if (block.type === "redactedThinking") {
100
- if (hasPartialThinking || dropLatestAssistantThinking) return [];
113
+ if (hasPartialThinking) return [];
114
+ // Same restriction as thinking blocks: cross-model/provider redacted
115
+ // blocks already drop below, so repair only needs to cover blocks that
116
+ // would replay as native redacted_thinking.
117
+ if (dropAssistantThinkingForRepair && (mustPreserveLatestAnthropicThinking || isSameModel)) {
118
+ return [];
119
+ }
101
120
  if (mustPreserveLatestAnthropicThinking) return block;
102
121
  if (isSameModel) return block;
103
122
  return [];
package/src/stream.ts CHANGED
@@ -76,7 +76,7 @@ function hasVertexAdcCredentials(): boolean {
76
76
  type KeyResolver = string | (() => string | undefined);
77
77
 
78
78
  const serviceProviderMap: Record<string, KeyResolver> = {
79
- "alibaba-coding-plan": "ALIBABA_CODING_PLAN_API_KEY",
79
+ "alibaba-token-plan": "ALIBABA_TOKEN_PLAN_API_KEY",
80
80
  openai: () => $credentialEnv("OPENAI_API_KEY"),
81
81
  google: "GEMINI_API_KEY",
82
82
  groq: "GROQ_API_KEY",
@@ -169,6 +169,8 @@ const serviceProviderMap: Record<string, KeyResolver> = {
169
169
  "qwen-portal": () => $pickCredentialEnv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
170
170
  together: "TOGETHER_API_KEY",
171
171
  zenmux: "ZENMUX_API_KEY",
172
+ opengateway: "OPENGATEWAY_API_KEY",
173
+ bizrouter: "BIZROUTER_API_KEY",
172
174
  venice: "VENICE_API_KEY",
173
175
  vllm: "VLLM_API_KEY",
174
176
  xiaomi: "XIAOMI_API_KEY",
package/src/types.ts CHANGED
@@ -114,7 +114,7 @@ export interface ThinkingConfig {
114
114
  }
115
115
 
116
116
  export type KnownProvider =
117
- | "alibaba-coding-plan"
117
+ | "alibaba-token-plan"
118
118
  | "amazon-bedrock"
119
119
  | "azure-openai"
120
120
  | "anthropic"
@@ -147,6 +147,8 @@ export type KnownProvider =
147
147
  | "minimax"
148
148
  | "opencode-go"
149
149
  | "opencode-zen"
150
+ | "opengateway"
151
+ | "bizrouter"
150
152
  | "synthetic"
151
153
  | "cloudflare-ai-gateway"
152
154
  | "huggingface"
@@ -709,10 +711,15 @@ export type TSchema = ZodType | TJsonSchema;
709
711
  /** Resolve parameter types for tool execution / handlers. */
710
712
  export type Static<S> = S extends ZodType ? z.infer<S> : S extends { static: infer T } ? T : unknown;
711
713
 
714
+ export type RawArgumentRejectionCode =
715
+ | "ask-intent-review-requires-positive-round"
716
+ | "ask-intent-contract-requires-non-empty-authority"
717
+ | "ask-deep-interview-metadata-requires-deep-interview-gate";
718
+
712
719
  export type RawArgumentValidationResult =
713
720
  | { outcome: "passthrough" }
714
721
  | { outcome: "accept"; arguments: ToolCall["arguments"] }
715
- | { outcome: "reject" };
722
+ | { outcome: "reject"; code?: RawArgumentRejectionCode };
716
723
 
717
724
  export interface Tool<TParameters extends TSchema = TSchema> {
718
725
  name: string;
@@ -1,4 +1,5 @@
1
1
  import { scheduler } from "node:timers/promises";
2
+ import { claudeCodeVersion } from "../providers/anthropic";
2
3
  import type {
3
4
  CredentialRankingStrategy,
4
5
  UsageAmount,
@@ -17,6 +18,13 @@ const FIVE_HOURS_MS = 5 * 60 * 60 * 1000;
17
18
  const SEVEN_DAYS_MS = 7 * 24 * 60 * 60 * 1000;
18
19
  const MAX_ATTEMPTS = 3;
19
20
  const BASE_RETRY_DELAY_MS = 500;
21
+ /**
22
+ * Ceiling for a server-supplied `Retry-After`. Matches `OPENAI_RETRY_DELAY_CAP_MS`
23
+ * and `fetchWithRetry`'s `DEFAULT_MAX_DELAY_MS`. Without it a hostile or
24
+ * misconfigured endpoint stalls the usage fetch for as long as it likes
25
+ * (`Retry-After: 86400` previously produced a 24h sleep).
26
+ */
27
+ const MAX_RETRY_DELAY_MS = 60_000;
20
28
 
21
29
  const CLAUDE_HEADERS = {
22
30
  accept: "application/json, text/plain, */*",
@@ -24,7 +32,7 @@ const CLAUDE_HEADERS = {
24
32
  "anthropic-beta":
25
33
  "claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,context-management-2025-06-27,prompt-caching-scope-2026-01-05",
26
34
  "content-type": "application/json",
27
- "user-agent": "claude-cli/2.1.63 (external, cli)",
35
+ "user-agent": `claude-cli/${claudeCodeVersion} (external, cli)`,
28
36
  connection: "keep-alive",
29
37
  } as const;
30
38
 
@@ -140,13 +148,23 @@ function isAbortError(error: unknown, signal?: AbortSignal): boolean {
140
148
  return error.name === "AbortError" || error.name === "TimeoutError";
141
149
  }
142
150
 
151
+ /**
152
+ * Honour the server hint but never exceed `MAX_RETRY_DELAY_MS`, and never
153
+ * return a negative/non-finite delay. Keeps the sleep bounded so an abort has
154
+ * an upper bound to fire within.
155
+ */
156
+ function clampRetryDelay(baseline: number, hintMs: number): number {
157
+ const hint = Number.isFinite(hintMs) ? Math.max(0, hintMs) : 0;
158
+ return Math.min(Math.max(baseline, hint), MAX_RETRY_DELAY_MS);
159
+ }
160
+
143
161
  function retryDelayMs(attempt: number, retryAfter: string | null): number {
144
162
  const baseline = BASE_RETRY_DELAY_MS * 2 ** attempt;
145
163
  if (!retryAfter?.trim()) return baseline;
146
164
  const seconds = Number.parseFloat(retryAfter);
147
- if (Number.isFinite(seconds)) return Math.max(baseline, Math.max(0, seconds * 1000));
165
+ if (Number.isFinite(seconds)) return clampRetryDelay(baseline, seconds * 1000);
148
166
  const dateDelay = Date.parse(retryAfter) - Date.now();
149
- return Number.isFinite(dateDelay) ? Math.max(baseline, Math.max(0, dateDelay)) : baseline;
167
+ return Number.isFinite(dateDelay) ? clampRetryDelay(baseline, dateDelay) : baseline;
150
168
  }
151
169
 
152
170
  async function waitBeforeRetry(
@@ -1,3 +1,4 @@
1
+ import { $credentialEnv } from "@sayknow-cli/utils";
1
2
  import type {
2
3
  CredentialRankingStrategy,
3
4
  UsageFetchContext,
@@ -64,10 +65,20 @@ function isUnsafeGrokBaseUrlOverride(baseUrl?: string): boolean {
64
65
  }
65
66
 
66
67
  function resolveAccessToken(params: UsageFetchParams): string | undefined {
67
- const token = params.credential.accessToken ?? params.credential.apiKey ?? process.env.GROK_CLI_OAUTH_TOKEN;
68
+ // Trusted sources only for the env fallback: this token authenticates the
69
+ // billing/usage call, so whatever can set it decides which account is queried
70
+ // with it. `$env` merges the caller's `cwd/.env` into `process.env`, so
71
+ // reading it there would let repository content supply the credential.
72
+ // Stored credentials keep precedence.
73
+ const token = params.credential.accessToken ?? params.credential.apiKey ?? $credentialEnv("GROK_CLI_OAUTH_TOKEN");
68
74
  return token?.trim() || undefined;
69
75
  }
70
76
 
77
+ /** Test seam: the usage access token as resolved from a credential plus trusted env. */
78
+ export function resolveGrokAccessTokenForTest(params: UsageFetchParams): string | undefined {
79
+ return resolveAccessToken(params);
80
+ }
81
+
71
82
  function buildMonthlyUsageLimit(usage: BillingUsage, nowMs: number): UsageLimit {
72
83
  const usedFraction = usage.monthlyLimit > 0 ? usage.used / usage.monthlyLimit : 0;
73
84
  const percent = usedFraction * 100;
package/src/usage/kimi.ts CHANGED
@@ -1,4 +1,4 @@
1
- import { $env } from "@sayknow-cli/utils";
1
+ import { $credentialEnv } from "@sayknow-cli/utils";
2
2
  import type {
3
3
  UsageAmount,
4
4
  UsageFetchContext,
@@ -31,12 +31,26 @@ type KimiUsageRow = {
31
31
  window?: UsageWindow;
32
32
  };
33
33
 
34
+ /**
35
+ * Usage endpoint base, with the environment override resolved from trusted
36
+ * sources only.
37
+ *
38
+ * The result becomes the usage URL that the request sends
39
+ * `Authorization: Bearer <accessToken>` to, so whatever can set it receives the
40
+ * user's Kimi access token. `$env` merges the caller's `cwd/.env`, so reading it
41
+ * there would let repository content collect that token.
42
+ */
34
43
  function normalizeBaseUrl(baseUrl?: string): string {
35
- const envBase = $env.KIMI_CODE_BASE_URL?.trim();
44
+ const envBase = $credentialEnv("KIMI_CODE_BASE_URL");
36
45
  const candidate = baseUrl?.trim() || envBase || DEFAULT_BASE_URL;
37
46
  return candidate.replace(/\/+$/, "");
38
47
  }
39
48
 
49
+ /** Test seam: the usage base URL as resolved from a caller value plus trusted env. */
50
+ export function normalizeKimiUsageBaseUrlForTest(baseUrl?: string): string {
51
+ return normalizeBaseUrl(baseUrl);
52
+ }
53
+
40
54
  function buildUsageUrl(baseUrl: string): string {
41
55
  const normalized = baseUrl.endsWith("/") ? baseUrl : `${baseUrl}/`;
42
56
  return `${normalized}${USAGE_PATH}`;
@@ -8,7 +8,7 @@
8
8
  * `authStorage.getApiKey("anthropic", sessionId)` first, then pass the result
9
9
  * through {@link buildAnthropicAuthConfig} for header/URL shaping.
10
10
  */
11
- import { $env } from "@sayknow-cli/utils";
11
+ import { $credentialEnv } from "@sayknow-cli/utils";
12
12
  import {
13
13
  buildAnthropicHeaders as buildProviderAnthropicHeaders,
14
14
  normalizeAnthropicBaseUrl,
@@ -29,12 +29,20 @@ function normalizeBaseUrl(baseUrl: string | undefined): string | undefined {
29
29
  return trimmed ? trimmed.replace(/\/+$/, "") : undefined;
30
30
  }
31
31
 
32
+ /**
33
+ * Resolve the Anthropic base URL from the environment.
34
+ *
35
+ * Trusted sources only: the result becomes the request URL that carries the
36
+ * Anthropic API key / OAuth token, so whatever can set it can redirect
37
+ * authenticated traffic. `$env` merges the caller's `cwd/.env`, so reading it
38
+ * there would let repository content choose where credentials are sent.
39
+ */
32
40
  export function resolveAnthropicBaseUrlFromEnv(): string | undefined {
33
41
  if (isFoundryEnabled()) {
34
- const foundryBaseUrl = normalizeBaseUrl($env.FOUNDRY_BASE_URL);
42
+ const foundryBaseUrl = normalizeBaseUrl($credentialEnv("FOUNDRY_BASE_URL"));
35
43
  if (foundryBaseUrl) return foundryBaseUrl;
36
44
  }
37
- const anthropicBaseUrl = normalizeBaseUrl($env.ANTHROPIC_BASE_URL);
45
+ const anthropicBaseUrl = normalizeBaseUrl($credentialEnv("ANTHROPIC_BASE_URL"));
38
46
  return anthropicBaseUrl || undefined;
39
47
  }
40
48
 
@@ -1,7 +1,17 @@
1
- import { $env } from "@sayknow-cli/utils";
1
+ import { $credentialEnv } from "@sayknow-cli/utils";
2
2
 
3
+ /**
4
+ * Whether Anthropic requests run in Foundry gateway mode.
5
+ *
6
+ * Resolved from trusted environment sources only. Enabling Foundry switches the
7
+ * request base URL and injects TLS client material, so whatever can set this
8
+ * redirects authenticated traffic. `$env` merges the caller's `cwd/.env`, so
9
+ * reading it there would let repository content flip the mode; resolve it the
10
+ * same way the credentials themselves are (launching shell plus SKC/user-owned
11
+ * `.env` files, never the project `.env`).
12
+ */
3
13
  export function isFoundryEnabled(): boolean {
4
- const value = $env.CLAUDE_CODE_USE_FOUNDRY;
14
+ const value = $credentialEnv("CLAUDE_CODE_USE_FOUNDRY");
5
15
  if (!value) return false;
6
16
  const normalized = value.trim().toLowerCase();
7
17
  return normalized === "1" || normalized === "true" || normalized === "yes" || normalized === "on";