@gajae-code/ai 0.11.11 → 0.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +42 -0
  2. package/README.md +3 -0
  3. package/dist/types/model-thinking.d.ts +15 -0
  4. package/dist/types/provider-models/openai-compat.d.ts +7 -0
  5. package/dist/types/providers/anthropic.d.ts +2 -1
  6. package/dist/types/providers/azure-openai-responses.d.ts +8 -1
  7. package/dist/types/providers/google-auth.d.ts +2 -0
  8. package/dist/types/providers/google-gemini-headers.d.ts +1 -1
  9. package/dist/types/providers/google-vertex.d.ts +4 -0
  10. package/dist/types/providers/openai-codex-responses.d.ts +4 -0
  11. package/dist/types/providers/openai-completions.d.ts +2 -0
  12. package/dist/types/providers/openai-responses.d.ts +2 -0
  13. package/dist/types/types.d.ts +1 -1
  14. package/dist/types/usage/grok-cli.d.ts +3 -1
  15. package/dist/types/usage/kimi.d.ts +2 -0
  16. package/dist/types/utils/anthropic-auth.d.ts +8 -0
  17. package/dist/types/utils/fallback-transport.d.ts +2 -0
  18. package/dist/types/utils/foundry.d.ts +10 -0
  19. package/dist/types/utils/http-inspector.d.ts +23 -0
  20. package/dist/types/utils/idle-iterator.d.ts +4 -0
  21. package/dist/types/utils/oauth/bizrouter.d.ts +1 -0
  22. package/dist/types/utils/oauth/kimi.d.ts +2 -0
  23. package/dist/types/utils/oauth/perplexity.d.ts +2 -7
  24. package/dist/types/utils/oauth/types.d.ts +1 -1
  25. package/package.json +2 -2
  26. package/src/auth-storage.ts +6 -0
  27. package/src/cli.ts +1 -0
  28. package/src/model-thinking.ts +19 -0
  29. package/src/models.json +72 -0
  30. package/src/provider-models/descriptors.ts +7 -0
  31. package/src/provider-models/openai-compat.ts +67 -6
  32. package/src/providers/amazon-bedrock.ts +5 -20
  33. package/src/providers/anthropic.ts +103 -48
  34. package/src/providers/azure-openai-responses.ts +44 -19
  35. package/src/providers/google-auth.ts +13 -2
  36. package/src/providers/google-gemini-headers.ts +1 -1
  37. package/src/providers/google-vertex.ts +41 -5
  38. package/src/providers/ollama.ts +32 -4
  39. package/src/providers/openai-codex-responses.ts +97 -38
  40. package/src/providers/openai-completions.ts +25 -14
  41. package/src/providers/openai-responses.ts +22 -20
  42. package/src/providers/register-builtins.ts +12 -2
  43. package/src/stream.ts +1 -0
  44. package/src/types.ts +1 -0
  45. package/src/usage/claude.ts +2 -1
  46. package/src/usage/grok-cli.ts +12 -1
  47. package/src/usage/kimi.ts +16 -2
  48. package/src/utils/anthropic-auth.ts +11 -3
  49. package/src/utils/fallback-transport.ts +17 -10
  50. package/src/utils/foundry.ts +12 -2
  51. package/src/utils/http-inspector.ts +124 -1
  52. package/src/utils/idle-iterator.ts +12 -3
  53. package/src/utils/oauth/bizrouter.ts +15 -0
  54. package/src/utils/oauth/index.ts +6 -0
  55. package/src/utils/oauth/kimi.ts +17 -2
  56. package/src/utils/oauth/perplexity.ts +21 -2
  57. package/src/utils/oauth/types.ts +1 -0
  58. package/src/utils/schema/adapt.ts +2 -2
  59. package/src/utils/tool-choice-capability.ts +2 -1
@@ -1,4 +1,4 @@
1
- import { $env, $inheritedEnv } from "@gajae-code/utils";
1
+ import { $credentialEnv } from "@gajae-code/utils";
2
2
  import type { ModelManagerOptions } from "../model-manager";
3
3
  import { Effort } from "../model-thinking";
4
4
  import { getBundledModels } from "../models";
@@ -553,13 +553,19 @@ export interface OpenAIModelManagerConfig {
553
553
  baseUrl?: string;
554
554
  }
555
555
 
556
+ /** Base URL for the OpenAI model manager, from trusted env only (`$env` merges the caller's `cwd/.env`). */
557
+ function resolveOpenAIModelManagerBaseUrl(config?: OpenAIModelManagerConfig): string {
558
+ return config?.baseUrl?.trim() || $credentialEnv("OPENAI_BASE_URL") || OPENAI_DEFAULT_BASE_URL;
559
+ }
560
+
561
+ /** Test seam: the model-manager base URL as resolved from trusted env. */
562
+ export function resolveOpenAIModelManagerBaseUrlForTest(config?: OpenAIModelManagerConfig): string {
563
+ return resolveOpenAIModelManagerBaseUrl(config);
564
+ }
565
+
556
566
  export function openaiModelManagerOptions(config?: OpenAIModelManagerConfig): ModelManagerOptions<"openai-responses"> {
557
567
  const apiKey = config?.apiKey;
558
- const baseUrl =
559
- config?.baseUrl?.trim() ||
560
- $inheritedEnv("OPENAI_BASE_URL") ||
561
- $env.OPENAI_BASE_URL?.trim() ||
562
- OPENAI_DEFAULT_BASE_URL;
568
+ const baseUrl = resolveOpenAIModelManagerBaseUrl(config);
563
569
  const references = createBundledReferenceMap<"openai-responses">("openai");
564
570
  return {
565
571
  providerId: "openai",
@@ -1119,6 +1125,61 @@ export function opengatewayModelManagerOptions(
1119
1125
  return createSimpleOpenAICompletionsOptions("opengateway", "https://apis.opengateway.ai/v1", config);
1120
1126
  }
1121
1127
 
1128
+ // ---------------------------------------------------------------------------
1129
+ // 10.5.2 BizRouter
1130
+ // ---------------------------------------------------------------------------
1131
+
1132
+ const BIZROUTER_BASE_URL = "https://api.bizrouter.ai/v1";
1133
+
1134
+ function toBizRouterPrice(value: unknown, fallback: number): number {
1135
+ const parsed = toNumber(value);
1136
+ return parsed === undefined || parsed < 0 ? fallback : parsed;
1137
+ }
1138
+
1139
+ export interface BizRouterModelManagerConfig {
1140
+ apiKey?: string;
1141
+ baseUrl?: string;
1142
+ }
1143
+
1144
+ export function bizrouterModelManagerOptions(
1145
+ config?: BizRouterModelManagerConfig,
1146
+ ): ModelManagerOptions<"openai-completions"> {
1147
+ const apiKey = config?.apiKey;
1148
+ const baseUrl = config?.baseUrl ?? BIZROUTER_BASE_URL;
1149
+ const references = createBundledReferenceMap<"openai-completions">("bizrouter");
1150
+ return {
1151
+ providerId: "bizrouter",
1152
+ ...(apiKey && {
1153
+ fetchDynamicModels: () =>
1154
+ fetchOpenAICompatibleModels({
1155
+ api: "openai-completions",
1156
+ provider: "bizrouter",
1157
+ baseUrl,
1158
+ apiKey,
1159
+ mapModel: (entry, defaults) => {
1160
+ const mapped = mapWithBundledReference(entry, defaults, references.get(defaults.id));
1161
+ return {
1162
+ ...mapped,
1163
+ name: toModelName(entry.display_name, mapped.name),
1164
+ contextWindow: toPositiveNumber(entry.context_length, mapped.contextWindow),
1165
+ maxTokens: toPositiveNumber(entry.max_output_tokens, mapped.maxTokens),
1166
+ input: toInputCapabilities(entry.input_modalities),
1167
+ cost: {
1168
+ input: toBizRouterPrice(entry.input_price_per_1m_usd, mapped.cost.input),
1169
+ output: toBizRouterPrice(entry.output_price_per_1m_usd, mapped.cost.output),
1170
+ cacheRead: mapped.cost.cacheRead,
1171
+ cacheWrite: mapped.cost.cacheWrite,
1172
+ },
1173
+ api: "openai-completions",
1174
+ provider: "bizrouter",
1175
+ baseUrl,
1176
+ };
1177
+ },
1178
+ }),
1179
+ }),
1180
+ };
1181
+ }
1182
+
1122
1183
  // ---------------------------------------------------------------------------
1123
1184
  // 10.6 Kilo Gateway
1124
1185
  // ---------------------------------------------------------------------------
@@ -9,7 +9,11 @@
9
9
 
10
10
  import { $credentialEnv, $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@gajae-code/utils";
11
11
  import type { Effort } from "../model-thinking";
12
- import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "../model-thinking";
12
+ import {
13
+ mapEffortToAnthropicAdaptiveEffort,
14
+ requireSupportedEffort,
15
+ supportsAnthropicAdaptiveThinkingDisplay as supportsAdaptiveThinkingDisplay,
16
+ } from "../model-thinking";
13
17
  import { calculateCost } from "../models";
14
18
  import type {
15
19
  Api,
@@ -907,25 +911,6 @@ function buildAdditionalModelRequestFields(
907
911
  return result;
908
912
  }
909
913
 
910
- /**
911
- * Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
912
- * Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
913
- * Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
914
- * "omitted" — thinking tokens are billed but no content streams back — so it
915
- * must opt in like Opus 4.7+ (issue #2791).
916
- * Bedrock model ids are prefixed with region/inference-profile slugs (e.g.
917
- * `eu.anthropic.Anthropic model-opus-4-7-...`); the regex matches the `Anthropic model-opus-X-Y`
918
- * fragment regardless of prefix.
919
- */
920
- function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
921
- if (/claude-fable-\d/.test(modelId)) return true;
922
- const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
923
- if (!match) return false;
924
- const major = Number(match[1]);
925
- const minor = Number(match[2]);
926
- return major > 4 || (major === 4 && minor >= 7);
927
- }
928
-
929
914
  /**
930
915
  * Bedrock's wire format expects the image as `{ source: { bytes: <base64-string> }, format }`.
931
916
  * The caller already passes base64-encoded data, so no decode/re-encode round-trip is needed.
@@ -11,6 +11,7 @@ import type {
11
11
  RawMessageStreamEvent,
12
12
  } from "@anthropic-ai/sdk/resources/messages";
13
13
  import {
14
+ $credentialEnv,
14
15
  $env,
15
16
  extractHttpStatusFromError,
16
17
  isEnoent,
@@ -19,7 +20,11 @@ import {
19
20
  logger,
20
21
  readSseEvents,
21
22
  } from "@gajae-code/utils";
22
- import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "../model-thinking";
23
+ import {
24
+ hasOpus47ApiRestrictions,
25
+ mapEffortToAnthropicAdaptiveEffort,
26
+ supportsAnthropicAdaptiveThinkingDisplay as supportsAdaptiveThinkingDisplay,
27
+ } from "../model-thinking";
23
28
  import { calculateCost } from "../models";
24
29
  import { isUsageLimitError } from "../rate-limit-utils";
25
30
  import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
@@ -61,12 +66,13 @@ import { transportFailureFacts } from "../utils/fallback-transport";
61
66
  import { isFoundryEnabled } from "../utils/foundry";
62
67
  import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
63
68
  import {
69
+ FirstEventTimeoutError,
64
70
  getProviderFirstEventTimeoutFallbackMs,
65
71
  getStreamFirstEventTimeoutMs,
66
72
  getStreamIdleTimeoutMs,
67
73
  iterateWithIdleTimeout,
68
74
  } from "../utils/idle-iterator";
69
- import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
75
+ import { isCompleteJson, parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
70
76
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
71
77
  import { notifyProviderResponse } from "../utils/provider-response";
72
78
  import { isCopilotTransientModelError } from "../utils/retry";
@@ -308,22 +314,6 @@ type AnthropicSamplingParams = MessageCreateParamsStreaming & {
308
314
  const ANTHROPIC_STOP_SEQUENCES_MAX = 4;
309
315
  let warnedStopSequencesTrim = false;
310
316
 
311
- /**
312
- * Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
313
- * Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
314
- * Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
315
- * "omitted" — thinking tokens are billed but no content streams back — so it
316
- * must opt in like Opus 4.7+ (issue #2791).
317
- */
318
- function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
319
- if (/claude-fable-\d/.test(modelId)) return true;
320
- const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
321
- if (!match) return false;
322
- const major = Number(match[1]);
323
- const minor = Number(match[2]);
324
- return major > 4 || (major === 4 && minor >= 7);
325
- }
326
-
327
317
  const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages";
328
318
 
329
319
  type AnthropicProviderSessionState = ProviderSessionState & {
@@ -490,7 +480,8 @@ function getCacheControl(
490
480
  }
491
481
 
492
482
  // Stealth mode: Mimic Anthropic Code headers and tool prefixing.
493
- export const claudeCodeVersion = "2.1.63";
483
+ export const claudeCodeVersion = "2.1.219";
484
+ export const claudeCodeEntrypoint = "sdk-cli";
494
485
  export const claudeToolPrefix: string = "proxy_";
495
486
  export const claudeCodeSystemInstruction = "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
496
487
 
@@ -566,7 +557,7 @@ function createClaudeBillingHeader(payload: unknown): string {
566
557
  const buildHash = Array.from(randomBytes, byte => byte.toString(16).padStart(2, "0"))
567
558
  .join("")
568
559
  .slice(0, 3);
569
- return `${CLAUDE_BILLING_HEADER_PREFIX} cc_version=${claudeCodeVersion}.${buildHash}; cc_entrypoint=cli; cch=${cch};`;
560
+ return `${CLAUDE_BILLING_HEADER_PREFIX} cc_version=${claudeCodeVersion}.${buildHash}; cc_entrypoint=${claudeCodeEntrypoint}; cch=${cch};`;
570
561
  }
571
562
 
572
563
  const CLAUDE_CLOAKING_USER_ID_REGEX =
@@ -816,10 +807,12 @@ function resolveAnthropicBaseUrl(model: Model<"anthropic-messages">, apiKey?: st
816
807
  // calls api.z.ai directly (no zcode.z.ai gateway, no captcha). Pin the base so dynamic
817
808
  // discovery / stale bundled catalogs / model cache can't redirect it elsewhere.
818
809
  if (model.provider === "glm-zcode") {
819
- return normalizeAnthropicBaseUrl(process.env.ZCODE_PLAN_ANTHROPIC_BASE_URL) ?? "https://api.z.ai/api/anthropic";
810
+ return (
811
+ normalizeAnthropicBaseUrl($credentialEnv("ZCODE_PLAN_ANTHROPIC_BASE_URL")) ?? "https://api.z.ai/api/anthropic"
812
+ );
820
813
  }
821
814
  if (model.provider === "anthropic" && isFoundryEnabled()) {
822
- const foundryBaseUrl = normalizeAnthropicBaseUrl($env.FOUNDRY_BASE_URL);
815
+ const foundryBaseUrl = normalizeAnthropicBaseUrl($credentialEnv("FOUNDRY_BASE_URL"));
823
816
  if (foundryBaseUrl) {
824
817
  return foundryBaseUrl;
825
818
  }
@@ -1418,6 +1411,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1418
1411
  ) & { index: number };
1419
1412
  const blocks = output.content as Block[];
1420
1413
  const blocksByAnthropicIndex = new Map<number, Block>();
1414
+ const truncatedToolCalls = new Set<ToolCall>();
1415
+ let sawTerminalStopReason = false;
1421
1416
  // Derive from the ACTUAL request shape, not the option default: the request
1422
1417
  // only sends `display: "summarized"` on specific paths (adaptive display is
1423
1418
  // omitted for models where supportsAdaptiveThinkingDisplay is false). Defaulting
@@ -1435,8 +1430,14 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1435
1430
  // finalize the orphaned block so no internal stream fields leak into output.
1436
1431
  const orphaned = blocksByAnthropicIndex.get(anthropicIndex);
1437
1432
  if (orphaned) {
1438
- if (orphaned.type === "toolCall" && orphaned.partialJson.trim()) {
1439
- orphaned.arguments = parseStreamingJson(orphaned.partialJson);
1433
+ if (orphaned.type === "toolCall") {
1434
+ if (!isCompleteJson(orphaned.partialJson)) {
1435
+ orphaned.incompleteArguments = true;
1436
+ truncatedToolCalls.add(orphaned);
1437
+ }
1438
+ if (orphaned.partialJson.trim()) {
1439
+ orphaned.arguments = parseStreamingJson(orphaned.partialJson);
1440
+ }
1440
1441
  }
1441
1442
  delete (orphaned as { index?: number }).index;
1442
1443
  delete (orphaned as { partialJson?: string }).partialJson;
@@ -1453,6 +1454,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1453
1454
  output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
1454
1455
  output.stopReason = "stop";
1455
1456
  firstTokenTime = undefined;
1457
+ truncatedToolCalls.clear();
1458
+ sawTerminalStopReason = false;
1456
1459
  };
1457
1460
  const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
1458
1461
  const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
@@ -1463,12 +1466,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1463
1466
  // Provider-level transport/rate-limit failures: only before any streamed content starts.
1464
1467
  // Malformed envelopes/JSON: only before replay-unsafe text/tool events are visible on this stream.
1465
1468
  let providerRetryAttempt = 0;
1466
- let thinkingRepairAttempted = false;
1467
1469
  while (true) {
1468
1470
  // Retries reset output.content; drop stale block correlations from the aborted attempt.
1469
1471
  blocksByAnthropicIndex.clear();
1472
+ truncatedToolCalls.clear();
1473
+ sawTerminalStopReason = false;
1470
1474
  activeAbortTracker = createAbortSourceTracker(options?.signal);
1471
- const firstEventTimeoutAbortError = new Error(
1475
+ const firstEventTimeoutAbortError = new FirstEventTimeoutError(
1472
1476
  "Anthropic stream timed out while waiting for the first event",
1473
1477
  );
1474
1478
  const idleTimeoutAbortError = new Error("Anthropic stream stalled while waiting for the next event");
@@ -1491,6 +1495,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1491
1495
  let sawEvent = false;
1492
1496
  let sawMessageStart = false;
1493
1497
  let sawTerminalEnvelope = false;
1498
+ let sawMessageStop = false;
1494
1499
  const isProgressEvent = createAnthropicStreamProgressPredicate();
1495
1500
 
1496
1501
  for await (const event of iterateWithIdleTimeout(anthropicStream, {
@@ -1504,9 +1509,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1504
1509
  isProgressItem: isProgressEvent,
1505
1510
  })) {
1506
1511
  sawEvent = true;
1512
+ if (sawMessageStop) {
1513
+ throw createAnthropicStreamEnvelopeError("received event after message_stop");
1514
+ }
1507
1515
  if (sawProviderSafetyStop) {
1508
1516
  if (event.type === "message_stop") {
1509
1517
  sawTerminalEnvelope = true;
1518
+ sawMessageStop = true;
1510
1519
  }
1511
1520
  continue;
1512
1521
  }
@@ -1694,6 +1703,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1694
1703
  partial: output,
1695
1704
  });
1696
1705
  } else if (block.type === "toolCall") {
1706
+ if (!isCompleteJson(block.partialJson)) truncatedToolCalls.add(block);
1697
1707
  if (block.partialJson.trim()) {
1698
1708
  block.arguments = parseStreamingJson(block.partialJson);
1699
1709
  }
@@ -1714,6 +1724,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1714
1724
  if (rawStopReason) {
1715
1725
  output.stopReason = isProviderSafetyStop ? "error" : mapStopReason(rawStopReason);
1716
1726
  sawTerminalEnvelope = true;
1727
+ sawTerminalStopReason = true;
1717
1728
  }
1718
1729
  if (isProviderSafetyStop) {
1719
1730
  sawProviderSafetyStop = true;
@@ -1755,6 +1766,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1755
1766
  calculateCost(model, output.usage);
1756
1767
  } else if (event.type === "message_stop") {
1757
1768
  sawTerminalEnvelope = true;
1769
+ sawMessageStop = true;
1758
1770
  }
1759
1771
  }
1760
1772
 
@@ -1777,8 +1789,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1777
1789
  }
1778
1790
  break;
1779
1791
  } catch (streamError) {
1780
- const streamFailure = activeAbortTracker.getLocalAbortReason() ?? streamError;
1781
- if (sawProviderSafetyStop) {
1792
+ const localAbortReason = activeAbortTracker.getLocalAbortReason();
1793
+ const streamFailure = localAbortReason ?? streamError;
1794
+ if (localAbortReason || sawProviderSafetyStop) {
1782
1795
  throw streamFailure;
1783
1796
  }
1784
1797
  if (
@@ -1830,21 +1843,22 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1830
1843
  const thinkingSignatureInvalid = isAnthropicThinkingSignatureInvalidError(streamFailure);
1831
1844
  if (
1832
1845
  !options?.fallbackManaged &&
1833
- !thinkingRepairAttempted &&
1846
+ !repairAllAssistantThinking &&
1834
1847
  firstTokenTime === undefined &&
1835
1848
  (thinkingSignatureInvalid || isAnthropicThinkingBlockMutationError(streamFailure))
1836
1849
  ) {
1850
+ // The mutation 400 blames the "latest assistant message", but its cited
1851
+ // `messages.N.content.M` path can point at an EARLIER replayed turn, so the
1852
+ // latest-only repair gets rejected identically. Escalate to the full-history
1853
+ // repair instead of burning the single retry on one scope.
1854
+ const escalateToAll: boolean = thinkingSignatureInvalid || repairLatestAssistantThinking;
1837
1855
  logger.debug("anthropic: repairing assistant thinking replay after provider rejection", {
1838
1856
  model: model.id,
1839
- scope: thinkingSignatureInvalid ? "all" : "latest",
1857
+ scope: escalateToAll ? "all" : "latest",
1840
1858
  error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure),
1841
1859
  });
1842
- thinkingRepairAttempted = true;
1843
- if (thinkingSignatureInvalid) {
1844
- repairAllAssistantThinking = true;
1845
- } else {
1846
- repairLatestAssistantThinking = true;
1847
- }
1860
+ repairLatestAssistantThinking = !escalateToAll;
1861
+ repairAllAssistantThinking = escalateToAll;
1848
1862
  params = await prepareParams();
1849
1863
  providerRetryAttempt = 0;
1850
1864
  resetOutputForRetry();
@@ -1893,6 +1907,24 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1893
1907
  }
1894
1908
  }
1895
1909
 
1910
+ for (const block of blocksByAnthropicIndex.values()) {
1911
+ delete (block as { index?: number }).index;
1912
+ if (block.type === "toolCall") {
1913
+ truncatedToolCalls.add(block);
1914
+ if (block.partialJson.trim()) {
1915
+ block.arguments = parseStreamingJson(block.partialJson);
1916
+ }
1917
+ delete (block as { partialJson?: string }).partialJson;
1918
+ }
1919
+ }
1920
+ blocksByAnthropicIndex.clear();
1921
+ if (output.stopReason === "length" || !sawTerminalStopReason) {
1922
+ for (const block of output.content) {
1923
+ if (block.type === "toolCall" && truncatedToolCalls.has(block)) {
1924
+ block.incompleteArguments = true;
1925
+ }
1926
+ }
1927
+ }
1896
1928
  output.duration = Date.now() - startTime;
1897
1929
  if (firstTokenTime) output.ttft = firstTokenTime - startTime;
1898
1930
  if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
@@ -1905,13 +1937,12 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1905
1937
  delete (block as { index?: number }).index;
1906
1938
  delete (block as { partialJson?: string }).partialJson;
1907
1939
  }
1908
- const firstEventTimeoutError = activeAbortTracker.getLocalAbortReason();
1940
+ const localAbortReason = activeAbortTracker.getLocalAbortReason();
1909
1941
  output.stopReason = activeAbortTracker.wasCallerAbort() ? "aborted" : "error";
1910
- output.errorStatus = extractHttpStatusFromError(error);
1911
- output.transportFailure = transportFailureFacts(error);
1942
+ output.errorStatus = extractHttpStatusFromError(localAbortReason ?? error);
1943
+ output.transportFailure = transportFailureFacts(localAbortReason ?? error);
1912
1944
  if (output.errorKind !== "provider_safety_stop" || !output.errorMessage) {
1913
- output.errorMessage =
1914
- firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
1945
+ output.errorMessage = localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
1915
1946
  }
1916
1947
  output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
1917
1948
  output.duration = Date.now() - startTime;
@@ -2096,13 +2127,26 @@ function createClient(
2096
2127
  return { client, isOAuthToken: oauthToken };
2097
2128
  }
2098
2129
 
2099
- function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming): void {
2130
+ /**
2131
+ * Anthropic rejects extended thinking combined with a forced tool choice, so such a
2132
+ * request drops `thinking`/`output_config`. Reports whether the forced-choice branch
2133
+ * applied so the caller can keep the replayed history consistent with it.
2134
+ */
2135
+ function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming): boolean {
2100
2136
  const toolChoice = params.tool_choice;
2101
- if (!toolChoice) return;
2102
- if (toolChoice.type === "any" || toolChoice.type === "tool") {
2103
- delete params.thinking;
2104
- delete params.output_config;
2105
- }
2137
+ if (!toolChoice) return false;
2138
+ if (toolChoice.type !== "any" && toolChoice.type !== "tool") return false;
2139
+ delete params.thinking;
2140
+ delete params.output_config;
2141
+ return true;
2142
+ }
2143
+
2144
+ function hasNativeThinkingBlocks(messages: MessageParam[]): boolean {
2145
+ return messages.some(
2146
+ message =>
2147
+ Array.isArray(message.content) &&
2148
+ message.content.some(block => block.type === "thinking" || block.type === "redacted_thinking"),
2149
+ );
2106
2150
  }
2107
2151
 
2108
2152
  function mapAnthropicToolChoice(
@@ -2426,6 +2470,18 @@ function buildParams(
2426
2470
  }
2427
2471
  }
2428
2472
 
2473
+ // A forced tool choice strips `thinking` from the request. Signed thinking blocks
2474
+ // replayed from history belong to a thinking-enabled request, and Anthropic rejects
2475
+ // that pair with `thinking`/`redacted_thinking` blocks "cannot be modified", so the
2476
+ // replay has to degrade in the same rebuild. Runs before the billing/system payload
2477
+ // snapshot so the attribution hash covers the messages actually sent.
2478
+ if (disableThinkingIfToolChoiceForced(params) && hasNativeThinkingBlocks(params.messages)) {
2479
+ params.messages = convertAnthropicMessages(context.messages, model, isOAuthToken, {
2480
+ ...thinkingRepair,
2481
+ repairAllAssistantThinking: true,
2482
+ });
2483
+ }
2484
+
2429
2485
  const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku");
2430
2486
  const billingSystemPrompts = normalizeSystemPrompts(context.systemPrompt);
2431
2487
  const billingPayload = shouldInjectClaudeCodeInstruction
@@ -2441,7 +2497,6 @@ function buildParams(
2441
2497
  if (systemBlocks) {
2442
2498
  params.system = systemBlocks;
2443
2499
  }
2444
- disableThinkingIfToolChoiceForced(params);
2445
2500
  ensureMaxTokensForThinking(params, model);
2446
2501
  applyPromptCaching(params as AnthropicCacheParams, cacheMode, cacheControl);
2447
2502
  enforceCacheControlLimit(params, 4);
@@ -1,4 +1,4 @@
1
- import { $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
1
+ import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
2
2
  import { AzureOpenAI } from "openai";
3
3
  import type {
4
4
  Tool as OpenAITool,
@@ -22,7 +22,6 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
22
22
  import { transportFailureFacts } from "../utils/fallback-transport";
23
23
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
24
24
  import {
25
- createWatchdog,
26
25
  getOpenAIStreamIdleTimeoutMs,
27
26
  getStreamFirstEventTimeoutMs,
28
27
  iterateWithIdleTimeout,
@@ -118,7 +117,6 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
118
117
  );
119
118
  let rawRequestDump: RawHttpRequestDump | undefined;
120
119
  const abortTracker = createAbortSourceTracker(options?.signal);
121
- const firstEventTimeoutAbortError = new Error(AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE);
122
120
  const { requestAbortController, requestSignal } = abortTracker;
123
121
 
124
122
  try {
@@ -127,7 +125,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
127
125
  const client = createClient(model, apiKey, options);
128
126
  const { baseUrl } = resolveAzureConfig(model, options);
129
127
  const params = buildParams(model, context, options, deploymentName, baseUrl);
130
- const idleTimeoutMs = getOpenAIStreamIdleTimeoutMs();
128
+ const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
131
129
  options?.onPayload?.(params);
132
130
  rawRequestDump = {
133
131
  provider: model.provider,
@@ -164,18 +162,17 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
164
162
  rawRequestDump = { ...rawRequestDump, body: params };
165
163
  openaiStream = await client.responses.create(params, { signal: requestSignal });
166
164
  }
167
- const firstEventWatchdog = createWatchdog(
168
- options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
169
- () => abortTracker.abortLocally(firstEventTimeoutAbortError),
170
- );
165
+ const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
171
166
  stream.push({ type: "start", partial: output });
172
167
 
173
168
  await processResponsesStream(
174
169
  iterateWithIdleTimeout(openaiStream, {
175
- watchdog: firstEventWatchdog,
170
+ firstItemTimeoutMs: firstEventTimeoutMs,
171
+ firstItemErrorMessage: AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE,
176
172
  idleTimeoutMs,
177
173
  errorMessage: "Azure OpenAI responses stream stalled while waiting for the next event",
178
174
  onIdle: () => requestAbortController.abort(),
175
+ onFirstItemTimeout: () => requestAbortController.abort(),
179
176
  }),
180
177
  output,
181
178
  stream,
@@ -234,8 +231,13 @@ function resolveAzureConfig(
234
231
  ): { baseUrl: string; apiVersion: string } {
235
232
  const apiVersion = options?.azureApiVersion || $env.AZURE_OPENAI_API_VERSION || DEFAULT_AZURE_API_VERSION;
236
233
 
237
- const baseUrl = options?.azureBaseUrl?.trim() || $env.AZURE_OPENAI_BASE_URL?.trim() || undefined;
238
- const resourceName = options?.azureResourceName || $env.AZURE_OPENAI_RESOURCE_NAME;
234
+ // Trusted sources only: both of these decide the request endpoint that carries
235
+ // the Azure credential, and `$env` merges the caller's `cwd/.env`. The resource
236
+ // name is the alternate constructor for the same host
237
+ // (`https://<resource>.openai.azure.com/openai/v1`), so it needs the same
238
+ // boundary as the explicit base URL.
239
+ const baseUrl = options?.azureBaseUrl?.trim() || $credentialEnv("AZURE_OPENAI_BASE_URL") || undefined;
240
+ const resourceName = options?.azureResourceName || $credentialEnv("AZURE_OPENAI_RESOURCE_NAME");
239
241
 
240
242
  let resolvedBaseUrl = baseUrl;
241
243
 
@@ -259,16 +261,39 @@ function resolveAzureConfig(
259
261
  };
260
262
  }
261
263
 
264
+ /** Test seam: the Azure endpoint config as resolved from trusted env. */
265
+ export function resolveAzureConfigForTest(
266
+ model: Model<"azure-openai-responses">,
267
+ options?: AzureOpenAIResponsesOptions,
268
+ ): { baseUrl: string; apiVersion: string } {
269
+ return resolveAzureConfig(model, options);
270
+ }
271
+
272
+ /**
273
+ * Azure API key for the client, from trusted environment sources only.
274
+ *
275
+ * `$env` merges the caller's `cwd/.env`, so reading the key there would let
276
+ * repository content supply the credential this client authenticates with.
277
+ * Provider credentials are resolved from the launching shell plus GJC/user-owned
278
+ * `.env` files, never the project `.env` — this fallback now matches that rule.
279
+ */
280
+ function resolveAzureClientApiKey(apiKey: string): string | undefined {
281
+ if (apiKey) return apiKey;
282
+ return $credentialEnv("AZURE_OPENAI_API_KEY");
283
+ }
284
+
285
+ /** Test seam: the client API key as resolved from a caller value plus trusted env. */
286
+ export function resolveAzureClientApiKeyForTest(apiKey: string): string | undefined {
287
+ return resolveAzureClientApiKey(apiKey);
288
+ }
262
289
  function createClient(model: Model<"azure-openai-responses">, apiKey: string, options?: AzureOpenAIResponsesOptions) {
263
- if (!apiKey) {
264
- const envKey = $env.AZURE_OPENAI_API_KEY;
265
- if (!envKey) {
266
- throw new Error(
267
- "Azure OpenAI API key is required. Set AZURE_OPENAI_API_KEY environment variable or pass it as an argument.",
268
- );
269
- }
270
- apiKey = envKey;
290
+ const resolvedApiKey = resolveAzureClientApiKey(apiKey);
291
+ if (!resolvedApiKey) {
292
+ throw new Error(
293
+ "Azure OpenAI API key is required. Set AZURE_OPENAI_API_KEY environment variable or pass it as an argument.",
294
+ );
271
295
  }
296
+ apiKey = resolvedApiKey;
272
297
 
273
298
  const headers = { ...(model.headers ?? {}) };
274
299
 
@@ -15,7 +15,7 @@
15
15
  import { Buffer } from "node:buffer";
16
16
  import * as os from "node:os";
17
17
  import * as path from "node:path";
18
- import { $envpos, isEnoent, logger } from "@gajae-code/utils";
18
+ import { $credentialEnv, $envpos, isEnoent, logger } from "@gajae-code/utils";
19
19
  import type { FetchImpl } from "../types";
20
20
 
21
21
  const OAUTH_TOKEN_URL = "https://oauth2.googleapis.com/token";
@@ -70,8 +70,19 @@ async function readJsonFile<T>(filePath: string): Promise<T | undefined> {
70
70
  }
71
71
  }
72
72
 
73
+ /** Test seam: the ADC credentials file path as resolved from trusted env. */
74
+ export function resolveAdcCredentialsPathForTest(): string | undefined {
75
+ return $credentialEnv("GOOGLE_APPLICATION_CREDENTIALS");
76
+ }
77
+
73
78
  async function loadAdcCredentials(): Promise<{ source: string; creds: AdcFileCredentials } | undefined> {
74
- const gacPath = Bun.env.GOOGLE_APPLICATION_CREDENTIALS;
79
+ // Trusted sources only: this path is read as service-account / authorized-user
80
+ // credentials and exchanged for a Google access token, so whatever can set it
81
+ // chooses the identity the agent authenticates as. `Bun.env` is `process.env`
82
+ // and the env module merges the caller's `cwd/.env` into it, so reading it
83
+ // there would let repository content point this at a key file it ships.
84
+ // `stream.ts` already resolves the same variable through `$credentialEnv`.
85
+ const gacPath = $credentialEnv("GOOGLE_APPLICATION_CREDENTIALS");
75
86
  if (gacPath) {
76
87
  const creds = await readJsonFile<AdcFileCredentials>(gacPath);
77
88
  if (!creds) {
@@ -5,7 +5,7 @@
5
5
  */
6
6
  export const GEMINI_CLI_VERSION_ENV = "GJC_AI_GEMINI_CLI_VERSION";
7
7
  export const LEGACY_GEMINI_CLI_VERSION_ENV = "PI_AI_GEMINI_CLI_VERSION";
8
- export const DEFAULT_GEMINI_CLI_VERSION = "0.50.0";
8
+ export const DEFAULT_GEMINI_CLI_VERSION = "0.52.0";
9
9
 
10
10
  export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
11
11
  const version =