@gajae-code/ai 0.12.0 → 0.12.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -20,7 +20,11 @@ import {
20
20
  logger,
21
21
  readSseEvents,
22
22
  } from "@gajae-code/utils";
23
- import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "../model-thinking";
23
+ import {
24
+ hasOpus47ApiRestrictions,
25
+ mapEffortToAnthropicAdaptiveEffort,
26
+ supportsAnthropicAdaptiveThinkingDisplay as supportsAdaptiveThinkingDisplay,
27
+ } from "../model-thinking";
24
28
  import { calculateCost } from "../models";
25
29
  import { isUsageLimitError } from "../rate-limit-utils";
26
30
  import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
@@ -62,12 +66,13 @@ import { transportFailureFacts } from "../utils/fallback-transport";
62
66
  import { isFoundryEnabled } from "../utils/foundry";
63
67
  import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
64
68
  import {
69
+ FirstEventTimeoutError,
65
70
  getProviderFirstEventTimeoutFallbackMs,
66
71
  getStreamFirstEventTimeoutMs,
67
72
  getStreamIdleTimeoutMs,
68
73
  iterateWithIdleTimeout,
69
74
  } from "../utils/idle-iterator";
70
- import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
75
+ import { isCompleteJson, parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
71
76
  import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
72
77
  import { notifyProviderResponse } from "../utils/provider-response";
73
78
  import { isCopilotTransientModelError } from "../utils/retry";
@@ -309,22 +314,6 @@ type AnthropicSamplingParams = MessageCreateParamsStreaming & {
309
314
  const ANTHROPIC_STOP_SEQUENCES_MAX = 4;
310
315
  let warnedStopSequencesTrim = false;
311
316
 
312
- /**
313
- * Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
314
- * Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
315
- * Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
316
- * "omitted" — thinking tokens are billed but no content streams back — so it
317
- * must opt in like Opus 4.7+ (issue #2791).
318
- */
319
- function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
320
- if (/claude-fable-\d/.test(modelId)) return true;
321
- const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
322
- if (!match) return false;
323
- const major = Number(match[1]);
324
- const minor = Number(match[2]);
325
- return major > 4 || (major === 4 && minor >= 7);
326
- }
327
-
328
317
  const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages";
329
318
 
330
319
  type AnthropicProviderSessionState = ProviderSessionState & {
@@ -1422,6 +1411,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1422
1411
  ) & { index: number };
1423
1412
  const blocks = output.content as Block[];
1424
1413
  const blocksByAnthropicIndex = new Map<number, Block>();
1414
+ const truncatedToolCalls = new Set<ToolCall>();
1415
+ let sawTerminalStopReason = false;
1425
1416
  // Derive from the ACTUAL request shape, not the option default: the request
1426
1417
  // only sends `display: "summarized"` on specific paths (adaptive display is
1427
1418
  // omitted for models where supportsAdaptiveThinkingDisplay is false). Defaulting
@@ -1439,8 +1430,14 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1439
1430
  // finalize the orphaned block so no internal stream fields leak into output.
1440
1431
  const orphaned = blocksByAnthropicIndex.get(anthropicIndex);
1441
1432
  if (orphaned) {
1442
- if (orphaned.type === "toolCall" && orphaned.partialJson.trim()) {
1443
- orphaned.arguments = parseStreamingJson(orphaned.partialJson);
1433
+ if (orphaned.type === "toolCall") {
1434
+ if (!isCompleteJson(orphaned.partialJson)) {
1435
+ orphaned.incompleteArguments = true;
1436
+ truncatedToolCalls.add(orphaned);
1437
+ }
1438
+ if (orphaned.partialJson.trim()) {
1439
+ orphaned.arguments = parseStreamingJson(orphaned.partialJson);
1440
+ }
1444
1441
  }
1445
1442
  delete (orphaned as { index?: number }).index;
1446
1443
  delete (orphaned as { partialJson?: string }).partialJson;
@@ -1457,6 +1454,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1457
1454
  output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
1458
1455
  output.stopReason = "stop";
1459
1456
  firstTokenTime = undefined;
1457
+ truncatedToolCalls.clear();
1458
+ sawTerminalStopReason = false;
1460
1459
  };
1461
1460
  const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
1462
1461
  const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
@@ -1467,12 +1466,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1467
1466
  // Provider-level transport/rate-limit failures: only before any streamed content starts.
1468
1467
  // Malformed envelopes/JSON: only before replay-unsafe text/tool events are visible on this stream.
1469
1468
  let providerRetryAttempt = 0;
1470
- let thinkingRepairAttempted = false;
1471
1469
  while (true) {
1472
1470
  // Retries reset output.content; drop stale block correlations from the aborted attempt.
1473
1471
  blocksByAnthropicIndex.clear();
1472
+ truncatedToolCalls.clear();
1473
+ sawTerminalStopReason = false;
1474
1474
  activeAbortTracker = createAbortSourceTracker(options?.signal);
1475
- const firstEventTimeoutAbortError = new Error(
1475
+ const firstEventTimeoutAbortError = new FirstEventTimeoutError(
1476
1476
  "Anthropic stream timed out while waiting for the first event",
1477
1477
  );
1478
1478
  const idleTimeoutAbortError = new Error("Anthropic stream stalled while waiting for the next event");
@@ -1495,6 +1495,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1495
1495
  let sawEvent = false;
1496
1496
  let sawMessageStart = false;
1497
1497
  let sawTerminalEnvelope = false;
1498
+ let sawMessageStop = false;
1498
1499
  const isProgressEvent = createAnthropicStreamProgressPredicate();
1499
1500
 
1500
1501
  for await (const event of iterateWithIdleTimeout(anthropicStream, {
@@ -1508,9 +1509,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1508
1509
  isProgressItem: isProgressEvent,
1509
1510
  })) {
1510
1511
  sawEvent = true;
1512
+ if (sawMessageStop) {
1513
+ throw createAnthropicStreamEnvelopeError("received event after message_stop");
1514
+ }
1511
1515
  if (sawProviderSafetyStop) {
1512
1516
  if (event.type === "message_stop") {
1513
1517
  sawTerminalEnvelope = true;
1518
+ sawMessageStop = true;
1514
1519
  }
1515
1520
  continue;
1516
1521
  }
@@ -1698,6 +1703,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1698
1703
  partial: output,
1699
1704
  });
1700
1705
  } else if (block.type === "toolCall") {
1706
+ if (!isCompleteJson(block.partialJson)) truncatedToolCalls.add(block);
1701
1707
  if (block.partialJson.trim()) {
1702
1708
  block.arguments = parseStreamingJson(block.partialJson);
1703
1709
  }
@@ -1718,6 +1724,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1718
1724
  if (rawStopReason) {
1719
1725
  output.stopReason = isProviderSafetyStop ? "error" : mapStopReason(rawStopReason);
1720
1726
  sawTerminalEnvelope = true;
1727
+ sawTerminalStopReason = true;
1721
1728
  }
1722
1729
  if (isProviderSafetyStop) {
1723
1730
  sawProviderSafetyStop = true;
@@ -1759,6 +1766,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1759
1766
  calculateCost(model, output.usage);
1760
1767
  } else if (event.type === "message_stop") {
1761
1768
  sawTerminalEnvelope = true;
1769
+ sawMessageStop = true;
1762
1770
  }
1763
1771
  }
1764
1772
 
@@ -1781,8 +1789,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1781
1789
  }
1782
1790
  break;
1783
1791
  } catch (streamError) {
1784
- const streamFailure = activeAbortTracker.getLocalAbortReason() ?? streamError;
1785
- if (sawProviderSafetyStop) {
1792
+ const localAbortReason = activeAbortTracker.getLocalAbortReason();
1793
+ const streamFailure = localAbortReason ?? streamError;
1794
+ if (localAbortReason || sawProviderSafetyStop) {
1786
1795
  throw streamFailure;
1787
1796
  }
1788
1797
  if (
@@ -1834,21 +1843,22 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1834
1843
  const thinkingSignatureInvalid = isAnthropicThinkingSignatureInvalidError(streamFailure);
1835
1844
  if (
1836
1845
  !options?.fallbackManaged &&
1837
- !thinkingRepairAttempted &&
1846
+ !repairAllAssistantThinking &&
1838
1847
  firstTokenTime === undefined &&
1839
1848
  (thinkingSignatureInvalid || isAnthropicThinkingBlockMutationError(streamFailure))
1840
1849
  ) {
1850
+ // The mutation 400 blames the "latest assistant message", but its cited
1851
+ // `messages.N.content.M` path can point at an EARLIER replayed turn, so the
1852
+ // latest-only repair gets rejected identically. Escalate to the full-history
1853
+ // repair instead of burning the single retry on one scope.
1854
+ const escalateToAll: boolean = thinkingSignatureInvalid || repairLatestAssistantThinking;
1841
1855
  logger.debug("anthropic: repairing assistant thinking replay after provider rejection", {
1842
1856
  model: model.id,
1843
- scope: thinkingSignatureInvalid ? "all" : "latest",
1857
+ scope: escalateToAll ? "all" : "latest",
1844
1858
  error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure),
1845
1859
  });
1846
- thinkingRepairAttempted = true;
1847
- if (thinkingSignatureInvalid) {
1848
- repairAllAssistantThinking = true;
1849
- } else {
1850
- repairLatestAssistantThinking = true;
1851
- }
1860
+ repairLatestAssistantThinking = !escalateToAll;
1861
+ repairAllAssistantThinking = escalateToAll;
1852
1862
  params = await prepareParams();
1853
1863
  providerRetryAttempt = 0;
1854
1864
  resetOutputForRetry();
@@ -1897,6 +1907,24 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1897
1907
  }
1898
1908
  }
1899
1909
 
1910
+ for (const block of blocksByAnthropicIndex.values()) {
1911
+ delete (block as { index?: number }).index;
1912
+ if (block.type === "toolCall") {
1913
+ truncatedToolCalls.add(block);
1914
+ if (block.partialJson.trim()) {
1915
+ block.arguments = parseStreamingJson(block.partialJson);
1916
+ }
1917
+ delete (block as { partialJson?: string }).partialJson;
1918
+ }
1919
+ }
1920
+ blocksByAnthropicIndex.clear();
1921
+ if (output.stopReason === "length" || !sawTerminalStopReason) {
1922
+ for (const block of output.content) {
1923
+ if (block.type === "toolCall" && truncatedToolCalls.has(block)) {
1924
+ block.incompleteArguments = true;
1925
+ }
1926
+ }
1927
+ }
1900
1928
  output.duration = Date.now() - startTime;
1901
1929
  if (firstTokenTime) output.ttft = firstTokenTime - startTime;
1902
1930
  if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
@@ -1909,13 +1937,12 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1909
1937
  delete (block as { index?: number }).index;
1910
1938
  delete (block as { partialJson?: string }).partialJson;
1911
1939
  }
1912
- const firstEventTimeoutError = activeAbortTracker.getLocalAbortReason();
1940
+ const localAbortReason = activeAbortTracker.getLocalAbortReason();
1913
1941
  output.stopReason = activeAbortTracker.wasCallerAbort() ? "aborted" : "error";
1914
- output.errorStatus = extractHttpStatusFromError(error);
1915
- output.transportFailure = transportFailureFacts(error);
1942
+ output.errorStatus = extractHttpStatusFromError(localAbortReason ?? error);
1943
+ output.transportFailure = transportFailureFacts(localAbortReason ?? error);
1916
1944
  if (output.errorKind !== "provider_safety_stop" || !output.errorMessage) {
1917
- output.errorMessage =
1918
- firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
1945
+ output.errorMessage = localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
1919
1946
  }
1920
1947
  output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
1921
1948
  output.duration = Date.now() - startTime;
@@ -2100,13 +2127,26 @@ function createClient(
2100
2127
  return { client, isOAuthToken: oauthToken };
2101
2128
  }
2102
2129
 
2103
- function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming): void {
2130
+ /**
2131
+ * Anthropic rejects extended thinking combined with a forced tool choice, so such a
2132
+ * request drops `thinking`/`output_config`. Reports whether the forced-choice branch
2133
+ * applied so the caller can keep the replayed history consistent with it.
2134
+ */
2135
+ function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming): boolean {
2104
2136
  const toolChoice = params.tool_choice;
2105
- if (!toolChoice) return;
2106
- if (toolChoice.type === "any" || toolChoice.type === "tool") {
2107
- delete params.thinking;
2108
- delete params.output_config;
2109
- }
2137
+ if (!toolChoice) return false;
2138
+ if (toolChoice.type !== "any" && toolChoice.type !== "tool") return false;
2139
+ delete params.thinking;
2140
+ delete params.output_config;
2141
+ return true;
2142
+ }
2143
+
2144
+ function hasNativeThinkingBlocks(messages: MessageParam[]): boolean {
2145
+ return messages.some(
2146
+ message =>
2147
+ Array.isArray(message.content) &&
2148
+ message.content.some(block => block.type === "thinking" || block.type === "redacted_thinking"),
2149
+ );
2110
2150
  }
2111
2151
 
2112
2152
  function mapAnthropicToolChoice(
@@ -2430,6 +2470,18 @@ function buildParams(
2430
2470
  }
2431
2471
  }
2432
2472
 
2473
+ // A forced tool choice strips `thinking` from the request. Signed thinking blocks
2474
+ // replayed from history belong to a thinking-enabled request, and Anthropic rejects
2475
+ // that pair with `thinking`/`redacted_thinking` blocks "cannot be modified", so the
2476
+ // replay has to degrade in the same rebuild. Runs before the billing/system payload
2477
+ // snapshot so the attribution hash covers the messages actually sent.
2478
+ if (disableThinkingIfToolChoiceForced(params) && hasNativeThinkingBlocks(params.messages)) {
2479
+ params.messages = convertAnthropicMessages(context.messages, model, isOAuthToken, {
2480
+ ...thinkingRepair,
2481
+ repairAllAssistantThinking: true,
2482
+ });
2483
+ }
2484
+
2433
2485
  const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku");
2434
2486
  const billingSystemPrompts = normalizeSystemPrompts(context.systemPrompt);
2435
2487
  const billingPayload = shouldInjectClaudeCodeInstruction
@@ -2445,7 +2497,6 @@ function buildParams(
2445
2497
  if (systemBlocks) {
2446
2498
  params.system = systemBlocks;
2447
2499
  }
2448
- disableThinkingIfToolChoiceForced(params);
2449
2500
  ensureMaxTokensForThinking(params, model);
2450
2501
  applyPromptCaching(params as AnthropicCacheParams, cacheMode, cacheControl);
2451
2502
  enforceCacheControlLimit(params, 4);
@@ -22,7 +22,6 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
22
22
  import { transportFailureFacts } from "../utils/fallback-transport";
23
23
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
24
24
  import {
25
- createWatchdog,
26
25
  getOpenAIStreamIdleTimeoutMs,
27
26
  getStreamFirstEventTimeoutMs,
28
27
  iterateWithIdleTimeout,
@@ -118,7 +117,6 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
118
117
  );
119
118
  let rawRequestDump: RawHttpRequestDump | undefined;
120
119
  const abortTracker = createAbortSourceTracker(options?.signal);
121
- const firstEventTimeoutAbortError = new Error(AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE);
122
120
  const { requestAbortController, requestSignal } = abortTracker;
123
121
 
124
122
  try {
@@ -127,7 +125,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
127
125
  const client = createClient(model, apiKey, options);
128
126
  const { baseUrl } = resolveAzureConfig(model, options);
129
127
  const params = buildParams(model, context, options, deploymentName, baseUrl);
130
- const idleTimeoutMs = getOpenAIStreamIdleTimeoutMs();
128
+ const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
131
129
  options?.onPayload?.(params);
132
130
  rawRequestDump = {
133
131
  provider: model.provider,
@@ -164,18 +162,17 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
164
162
  rawRequestDump = { ...rawRequestDump, body: params };
165
163
  openaiStream = await client.responses.create(params, { signal: requestSignal });
166
164
  }
167
- const firstEventWatchdog = createWatchdog(
168
- options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
169
- () => abortTracker.abortLocally(firstEventTimeoutAbortError),
170
- );
165
+ const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
171
166
  stream.push({ type: "start", partial: output });
172
167
 
173
168
  await processResponsesStream(
174
169
  iterateWithIdleTimeout(openaiStream, {
175
- watchdog: firstEventWatchdog,
170
+ firstItemTimeoutMs: firstEventTimeoutMs,
171
+ firstItemErrorMessage: AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE,
176
172
  idleTimeoutMs,
177
173
  errorMessage: "Azure OpenAI responses stream stalled while waiting for the next event",
178
174
  onIdle: () => requestAbortController.abort(),
175
+ onFirstItemTimeout: () => requestAbortController.abort(),
179
176
  }),
180
177
  output,
181
178
  stream,
@@ -272,16 +269,31 @@ export function resolveAzureConfigForTest(
272
269
  return resolveAzureConfig(model, options);
273
270
  }
274
271
 
272
+ /**
273
+ * Azure API key for the client, from trusted environment sources only.
274
+ *
275
+ * `$env` merges the caller's `cwd/.env`, so reading the key there would let
276
+ * repository content supply the credential this client authenticates with.
277
+ * Provider credentials are resolved from the launching shell plus GJC/user-owned
278
+ * `.env` files, never the project `.env` — this fallback now matches that rule.
279
+ */
280
+ function resolveAzureClientApiKey(apiKey: string): string | undefined {
281
+ if (apiKey) return apiKey;
282
+ return $credentialEnv("AZURE_OPENAI_API_KEY");
283
+ }
284
+
285
+ /** Test seam: the client API key as resolved from a caller value plus trusted env. */
286
+ export function resolveAzureClientApiKeyForTest(apiKey: string): string | undefined {
287
+ return resolveAzureClientApiKey(apiKey);
288
+ }
275
289
  function createClient(model: Model<"azure-openai-responses">, apiKey: string, options?: AzureOpenAIResponsesOptions) {
276
- if (!apiKey) {
277
- const envKey = $env.AZURE_OPENAI_API_KEY;
278
- if (!envKey) {
279
- throw new Error(
280
- "Azure OpenAI API key is required. Set AZURE_OPENAI_API_KEY environment variable or pass it as an argument.",
281
- );
282
- }
283
- apiKey = envKey;
290
+ const resolvedApiKey = resolveAzureClientApiKey(apiKey);
291
+ if (!resolvedApiKey) {
292
+ throw new Error(
293
+ "Azure OpenAI API key is required. Set AZURE_OPENAI_API_KEY environment variable or pass it as an argument.",
294
+ );
284
295
  }
296
+ apiKey = resolvedApiKey;
285
297
 
286
298
  const headers = { ...(model.headers ?? {}) };
287
299
 
@@ -1,4 +1,4 @@
1
- import { $credentialEnv, $env } from "@gajae-code/utils";
1
+ import { $credentialEnv, $pickCredentialEnv } from "@gajae-code/utils";
2
2
  import type { Context, Model, StreamFunction } from "../types";
3
3
  import type { AssistantMessageEventStream } from "../utils/event-stream";
4
4
  import { getVertexAccessToken } from "./google-auth";
@@ -72,7 +72,7 @@ function resolveApiKey(options?: GoogleVertexOptions): string | undefined {
72
72
  }
73
73
 
74
74
  function resolveProject(options?: GoogleVertexOptions): string {
75
- const project = options?.project || $env.GOOGLE_CLOUD_PROJECT || $env.GCLOUD_PROJECT;
75
+ const project = options?.project || $pickCredentialEnv("GOOGLE_CLOUD_PROJECT", "GCLOUD_PROJECT");
76
76
  if (!project) {
77
77
  throw new Error(
78
78
  "Vertex AI requires a project ID. Set GOOGLE_CLOUD_PROJECT/GCLOUD_PROJECT or pass project in options.",
@@ -84,10 +84,41 @@ function resolveProject(options?: GoogleVertexOptions): string {
84
84
  function resolveEndpointHost(location: string): string {
85
85
  return location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`;
86
86
  }
87
+ /**
88
+ * Vertex location, from trusted environment sources only and constrained to a
89
+ * region label.
90
+ *
91
+ * The location is interpolated into the request **host**
92
+ * (`${location}-aiplatform.googleapis.com`) as well as the path, and the request
93
+ * carries `Authorization: Bearer <accessToken>`. A value containing `/`
94
+ * terminates the authority component, so `evil.example.com/` resolves to origin
95
+ * `https://evil.example.com` and the Google access token leaves Google entirely.
96
+ * `$env` merges the caller's `cwd/.env`, so this was reachable from repository
97
+ * content.
98
+ *
99
+ * Both halves are needed: trusted resolution keeps a repository from setting it,
100
+ * and the shape check keeps any source from turning a region into an authority.
101
+ */
102
+ const VERTEX_LOCATION_RE = /^[a-z0-9-]+$/;
103
+
104
+ function assertVertexLocation(location: string): string {
105
+ if (!VERTEX_LOCATION_RE.test(location)) {
106
+ throw new Error(
107
+ `Invalid Vertex AI location ${JSON.stringify(location)}. Expected a region label such as "us-central1" or "global".`,
108
+ );
109
+ }
110
+ return location;
111
+ }
112
+
87
113
  function resolveLocation(options?: GoogleVertexOptions): string {
88
- const location = options?.location || $env.GOOGLE_CLOUD_LOCATION;
114
+ const location = options?.location || $credentialEnv("GOOGLE_CLOUD_LOCATION");
89
115
  if (!location) {
90
116
  throw new Error("Vertex AI requires a location. Set GOOGLE_CLOUD_LOCATION or pass location in options.");
91
117
  }
92
- return location;
118
+ return assertVertexLocation(location);
119
+ }
120
+
121
+ /** Test seam: the Vertex location as resolved from options plus trusted env. */
122
+ export function resolveVertexLocationForTest(options?: GoogleVertexOptions): string {
123
+ return resolveLocation(options);
93
124
  }
@@ -18,7 +18,7 @@ import { normalizeSystemPrompts } from "../utils";
18
18
  import { AssistantMessageEventStream } from "../utils/event-stream";
19
19
  import { transportFailureFacts } from "../utils/fallback-transport";
20
20
  import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
21
- import { parseStreamingJson } from "../utils/json-parse";
21
+ import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
22
22
  import { resolveRetryBudget } from "../utils/retry-budget";
23
23
  import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
24
24
  import {
@@ -26,6 +26,7 @@ import {
26
26
  markToolChoiceIncapability,
27
27
  resolveToolChoice,
28
28
  } from "../utils/tool-choice-capability";
29
+ import { flagTruncatedToolCalls } from "./openai-responses-shared";
29
30
  import { transformMessages } from "./transform-messages";
30
31
 
31
32
  export interface OllamaChatOptions extends StreamOptions {
@@ -357,8 +358,10 @@ function endToolCallBlock(stream: AssistantMessageEventStream, output: Assistant
357
358
  return;
358
359
  }
359
360
  const toolCall = block as InternalToolCallBlock;
360
- if (toolCall.partialJson) {
361
- toolCall.arguments = parseStreamingJson<Record<string, unknown>>(toolCall.partialJson);
361
+ if (toolCall.partialJson !== undefined) {
362
+ if (toolCall.partialJson.trim()) {
363
+ toolCall.arguments = parseStreamingJson<Record<string, unknown>>(toolCall.partialJson);
364
+ }
362
365
  delete toolCall.partialJson;
363
366
  }
364
367
  stream.push({ type: "toolcall_end", contentIndex: index, toolCall, partial: output });
@@ -393,6 +396,8 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
393
396
  let activeThinkingIndex: number | undefined;
394
397
  let activeTextIndex: number | undefined;
395
398
  const activeToolIndices = new Set<number>();
399
+ const unverifiableArgumentToolCallIds = new Set<string>();
400
+ let sawTerminalChunk = false;
396
401
  try {
397
402
  const apiKey = options.apiKey || getEnvApiKey(model.provider);
398
403
  if (!apiKey) {
@@ -537,6 +542,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
537
542
  for (const call of chunk.message.tool_calls) {
538
543
  const name = call.function?.name ?? "unknown_tool";
539
544
  const rawArgs = call.function?.arguments;
545
+ const unverifiableArguments = typeof rawArgs !== "string";
540
546
  const partialJson = typeof rawArgs === "string" ? rawArgs : JSON.stringify(rawArgs ?? {});
541
547
  const toolCall: InternalToolCallBlock = {
542
548
  type: "toolCall",
@@ -545,6 +551,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
545
551
  arguments: parseStreamingJson<Record<string, unknown>>(partialJson),
546
552
  partialJson,
547
553
  };
554
+ if (unverifiableArguments) unverifiableArgumentToolCallIds.add(toolCall.id);
548
555
  output.content.push(toolCall);
549
556
  const index = output.content.length - 1;
550
557
  activeToolIndices.add(index);
@@ -561,6 +568,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
561
568
  }
562
569
  }
563
570
  if (chunk.done) {
571
+ sawTerminalChunk = true;
564
572
  if (activeThinkingIndex !== undefined) {
565
573
  endThinkingBlock(stream, output, activeThinkingIndex);
566
574
  activeThinkingIndex = undefined;
@@ -569,16 +577,36 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
569
577
  endTextBlock(stream, output, activeTextIndex);
570
578
  activeTextIndex = undefined;
571
579
  }
580
+ output.stopReason = mapDoneReason(chunk.done_reason, output);
581
+ // Ollama still owns every partialJson buffer here; use the helper's
582
+ // finalized-call branch before endToolCallBlock deletes those buffers.
583
+ // Non-string arguments have no raw completion evidence, so a length stop
584
+ // must fail closed rather than trusting normalized or re-serialized values.
585
+ flagTruncatedToolCalls(
586
+ output,
587
+ output.stopReason,
588
+ block => !unverifiableArgumentToolCallIds.has(block.id),
589
+ );
590
+ if (chunk.done_reason === undefined) {
591
+ for (const block of output.content) {
592
+ if (block.type !== "toolCall") continue;
593
+ const partialJson = (block as InternalToolCallBlock).partialJson;
594
+ if (partialJson !== undefined && !isCompleteJson(partialJson)) block.incompleteArguments = true;
595
+ }
596
+ }
572
597
  for (const index of activeToolIndices) {
573
598
  endToolCallBlock(stream, output, index);
574
599
  }
575
600
  activeToolIndices.clear();
576
- output.stopReason = mapDoneReason(chunk.done_reason, output);
577
601
  output.usage.input = chunk.prompt_eval_count ?? 0;
578
602
  output.usage.output = chunk.eval_count ?? 0;
579
603
  output.usage.totalTokens = output.usage.input + output.usage.output;
604
+ break;
580
605
  }
581
606
  }
607
+ if (!sawTerminalChunk) {
608
+ throw new Error("Ollama stream ended before terminal done chunk");
609
+ }
582
610
  output.duration = Date.now() - startTime;
583
611
  if (firstTokenTime) {
584
612
  output.ttft = firstTokenTime - startTime;