@gajae-code/ai 0.12.0 → 0.12.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/dist/types/model-thinking.d.ts +15 -0
- package/dist/types/providers/azure-openai-responses.d.ts +2 -0
- package/dist/types/providers/google-vertex.d.ts +2 -0
- package/dist/types/utils/fallback-transport.d.ts +2 -0
- package/dist/types/utils/http-inspector.d.ts +10 -0
- package/dist/types/utils/idle-iterator.d.ts +4 -0
- package/dist/types/utils/oauth/kimi.d.ts +2 -0
- package/dist/types/utils/oauth/perplexity.d.ts +2 -7
- package/package.json +2 -2
- package/src/auth-storage.ts +92 -15
- package/src/model-thinking.ts +19 -0
- package/src/providers/amazon-bedrock.ts +5 -20
- package/src/providers/anthropic.ts +95 -44
- package/src/providers/azure-openai-responses.ts +28 -16
- package/src/providers/google-vertex.ts +35 -4
- package/src/providers/ollama.ts +32 -4
- package/src/providers/openai-codex-responses.ts +48 -31
- package/src/providers/openai-completions.ts +13 -12
- package/src/providers/openai-responses.ts +9 -11
- package/src/providers/register-builtins.ts +12 -2
- package/src/utils/fallback-transport.ts +17 -10
- package/src/utils/http-inspector.ts +47 -1
- package/src/utils/idle-iterator.ts +12 -3
- package/src/utils/oauth/kimi.ts +17 -2
- package/src/utils/oauth/perplexity.ts +21 -2
- package/src/utils/schema/adapt.ts +2 -2
- package/src/utils/tool-choice-capability.ts +2 -1
|
@@ -20,7 +20,11 @@ import {
|
|
|
20
20
|
logger,
|
|
21
21
|
readSseEvents,
|
|
22
22
|
} from "@gajae-code/utils";
|
|
23
|
-
import {
|
|
23
|
+
import {
|
|
24
|
+
hasOpus47ApiRestrictions,
|
|
25
|
+
mapEffortToAnthropicAdaptiveEffort,
|
|
26
|
+
supportsAnthropicAdaptiveThinkingDisplay as supportsAdaptiveThinkingDisplay,
|
|
27
|
+
} from "../model-thinking";
|
|
24
28
|
import { calculateCost } from "../models";
|
|
25
29
|
import { isUsageLimitError } from "../rate-limit-utils";
|
|
26
30
|
import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
|
|
@@ -62,12 +66,13 @@ import { transportFailureFacts } from "../utils/fallback-transport";
|
|
|
62
66
|
import { isFoundryEnabled } from "../utils/foundry";
|
|
63
67
|
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
|
|
64
68
|
import {
|
|
69
|
+
FirstEventTimeoutError,
|
|
65
70
|
getProviderFirstEventTimeoutFallbackMs,
|
|
66
71
|
getStreamFirstEventTimeoutMs,
|
|
67
72
|
getStreamIdleTimeoutMs,
|
|
68
73
|
iterateWithIdleTimeout,
|
|
69
74
|
} from "../utils/idle-iterator";
|
|
70
|
-
import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
|
|
75
|
+
import { isCompleteJson, parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
|
|
71
76
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
72
77
|
import { notifyProviderResponse } from "../utils/provider-response";
|
|
73
78
|
import { isCopilotTransientModelError } from "../utils/retry";
|
|
@@ -309,22 +314,6 @@ type AnthropicSamplingParams = MessageCreateParamsStreaming & {
|
|
|
309
314
|
const ANTHROPIC_STOP_SEQUENCES_MAX = 4;
|
|
310
315
|
let warnedStopSequencesTrim = false;
|
|
311
316
|
|
|
312
|
-
/**
|
|
313
|
-
* Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
|
|
314
|
-
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
315
|
-
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
316
|
-
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
317
|
-
* must opt in like Opus 4.7+ (issue #2791).
|
|
318
|
-
*/
|
|
319
|
-
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
320
|
-
if (/claude-fable-\d/.test(modelId)) return true;
|
|
321
|
-
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
|
|
322
|
-
if (!match) return false;
|
|
323
|
-
const major = Number(match[1]);
|
|
324
|
-
const minor = Number(match[2]);
|
|
325
|
-
return major > 4 || (major === 4 && minor >= 7);
|
|
326
|
-
}
|
|
327
|
-
|
|
328
317
|
const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages";
|
|
329
318
|
|
|
330
319
|
type AnthropicProviderSessionState = ProviderSessionState & {
|
|
@@ -1422,6 +1411,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1422
1411
|
) & { index: number };
|
|
1423
1412
|
const blocks = output.content as Block[];
|
|
1424
1413
|
const blocksByAnthropicIndex = new Map<number, Block>();
|
|
1414
|
+
const truncatedToolCalls = new Set<ToolCall>();
|
|
1415
|
+
let sawTerminalStopReason = false;
|
|
1425
1416
|
// Derive from the ACTUAL request shape, not the option default: the request
|
|
1426
1417
|
// only sends `display: "summarized"` on specific paths (adaptive display is
|
|
1427
1418
|
// omitted for models where supportsAdaptiveThinkingDisplay is false). Defaulting
|
|
@@ -1439,8 +1430,14 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1439
1430
|
// finalize the orphaned block so no internal stream fields leak into output.
|
|
1440
1431
|
const orphaned = blocksByAnthropicIndex.get(anthropicIndex);
|
|
1441
1432
|
if (orphaned) {
|
|
1442
|
-
if (orphaned.type === "toolCall"
|
|
1443
|
-
|
|
1433
|
+
if (orphaned.type === "toolCall") {
|
|
1434
|
+
if (!isCompleteJson(orphaned.partialJson)) {
|
|
1435
|
+
orphaned.incompleteArguments = true;
|
|
1436
|
+
truncatedToolCalls.add(orphaned);
|
|
1437
|
+
}
|
|
1438
|
+
if (orphaned.partialJson.trim()) {
|
|
1439
|
+
orphaned.arguments = parseStreamingJson(orphaned.partialJson);
|
|
1440
|
+
}
|
|
1444
1441
|
}
|
|
1445
1442
|
delete (orphaned as { index?: number }).index;
|
|
1446
1443
|
delete (orphaned as { partialJson?: string }).partialJson;
|
|
@@ -1457,6 +1454,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1457
1454
|
output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
|
|
1458
1455
|
output.stopReason = "stop";
|
|
1459
1456
|
firstTokenTime = undefined;
|
|
1457
|
+
truncatedToolCalls.clear();
|
|
1458
|
+
sawTerminalStopReason = false;
|
|
1460
1459
|
};
|
|
1461
1460
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
|
|
1462
1461
|
const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
@@ -1467,12 +1466,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1467
1466
|
// Provider-level transport/rate-limit failures: only before any streamed content starts.
|
|
1468
1467
|
// Malformed envelopes/JSON: only before replay-unsafe text/tool events are visible on this stream.
|
|
1469
1468
|
let providerRetryAttempt = 0;
|
|
1470
|
-
let thinkingRepairAttempted = false;
|
|
1471
1469
|
while (true) {
|
|
1472
1470
|
// Retries reset output.content; drop stale block correlations from the aborted attempt.
|
|
1473
1471
|
blocksByAnthropicIndex.clear();
|
|
1472
|
+
truncatedToolCalls.clear();
|
|
1473
|
+
sawTerminalStopReason = false;
|
|
1474
1474
|
activeAbortTracker = createAbortSourceTracker(options?.signal);
|
|
1475
|
-
const firstEventTimeoutAbortError = new
|
|
1475
|
+
const firstEventTimeoutAbortError = new FirstEventTimeoutError(
|
|
1476
1476
|
"Anthropic stream timed out while waiting for the first event",
|
|
1477
1477
|
);
|
|
1478
1478
|
const idleTimeoutAbortError = new Error("Anthropic stream stalled while waiting for the next event");
|
|
@@ -1495,6 +1495,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1495
1495
|
let sawEvent = false;
|
|
1496
1496
|
let sawMessageStart = false;
|
|
1497
1497
|
let sawTerminalEnvelope = false;
|
|
1498
|
+
let sawMessageStop = false;
|
|
1498
1499
|
const isProgressEvent = createAnthropicStreamProgressPredicate();
|
|
1499
1500
|
|
|
1500
1501
|
for await (const event of iterateWithIdleTimeout(anthropicStream, {
|
|
@@ -1508,9 +1509,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1508
1509
|
isProgressItem: isProgressEvent,
|
|
1509
1510
|
})) {
|
|
1510
1511
|
sawEvent = true;
|
|
1512
|
+
if (sawMessageStop) {
|
|
1513
|
+
throw createAnthropicStreamEnvelopeError("received event after message_stop");
|
|
1514
|
+
}
|
|
1511
1515
|
if (sawProviderSafetyStop) {
|
|
1512
1516
|
if (event.type === "message_stop") {
|
|
1513
1517
|
sawTerminalEnvelope = true;
|
|
1518
|
+
sawMessageStop = true;
|
|
1514
1519
|
}
|
|
1515
1520
|
continue;
|
|
1516
1521
|
}
|
|
@@ -1698,6 +1703,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1698
1703
|
partial: output,
|
|
1699
1704
|
});
|
|
1700
1705
|
} else if (block.type === "toolCall") {
|
|
1706
|
+
if (!isCompleteJson(block.partialJson)) truncatedToolCalls.add(block);
|
|
1701
1707
|
if (block.partialJson.trim()) {
|
|
1702
1708
|
block.arguments = parseStreamingJson(block.partialJson);
|
|
1703
1709
|
}
|
|
@@ -1718,6 +1724,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1718
1724
|
if (rawStopReason) {
|
|
1719
1725
|
output.stopReason = isProviderSafetyStop ? "error" : mapStopReason(rawStopReason);
|
|
1720
1726
|
sawTerminalEnvelope = true;
|
|
1727
|
+
sawTerminalStopReason = true;
|
|
1721
1728
|
}
|
|
1722
1729
|
if (isProviderSafetyStop) {
|
|
1723
1730
|
sawProviderSafetyStop = true;
|
|
@@ -1759,6 +1766,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1759
1766
|
calculateCost(model, output.usage);
|
|
1760
1767
|
} else if (event.type === "message_stop") {
|
|
1761
1768
|
sawTerminalEnvelope = true;
|
|
1769
|
+
sawMessageStop = true;
|
|
1762
1770
|
}
|
|
1763
1771
|
}
|
|
1764
1772
|
|
|
@@ -1781,8 +1789,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1781
1789
|
}
|
|
1782
1790
|
break;
|
|
1783
1791
|
} catch (streamError) {
|
|
1784
|
-
const
|
|
1785
|
-
|
|
1792
|
+
const localAbortReason = activeAbortTracker.getLocalAbortReason();
|
|
1793
|
+
const streamFailure = localAbortReason ?? streamError;
|
|
1794
|
+
if (localAbortReason || sawProviderSafetyStop) {
|
|
1786
1795
|
throw streamFailure;
|
|
1787
1796
|
}
|
|
1788
1797
|
if (
|
|
@@ -1834,21 +1843,22 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1834
1843
|
const thinkingSignatureInvalid = isAnthropicThinkingSignatureInvalidError(streamFailure);
|
|
1835
1844
|
if (
|
|
1836
1845
|
!options?.fallbackManaged &&
|
|
1837
|
-
!
|
|
1846
|
+
!repairAllAssistantThinking &&
|
|
1838
1847
|
firstTokenTime === undefined &&
|
|
1839
1848
|
(thinkingSignatureInvalid || isAnthropicThinkingBlockMutationError(streamFailure))
|
|
1840
1849
|
) {
|
|
1850
|
+
// The mutation 400 blames the "latest assistant message", but its cited
|
|
1851
|
+
// `messages.N.content.M` path can point at an EARLIER replayed turn, so the
|
|
1852
|
+
// latest-only repair gets rejected identically. Escalate to the full-history
|
|
1853
|
+
// repair instead of burning the single retry on one scope.
|
|
1854
|
+
const escalateToAll: boolean = thinkingSignatureInvalid || repairLatestAssistantThinking;
|
|
1841
1855
|
logger.debug("anthropic: repairing assistant thinking replay after provider rejection", {
|
|
1842
1856
|
model: model.id,
|
|
1843
|
-
scope:
|
|
1857
|
+
scope: escalateToAll ? "all" : "latest",
|
|
1844
1858
|
error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure),
|
|
1845
1859
|
});
|
|
1846
|
-
|
|
1847
|
-
|
|
1848
|
-
repairAllAssistantThinking = true;
|
|
1849
|
-
} else {
|
|
1850
|
-
repairLatestAssistantThinking = true;
|
|
1851
|
-
}
|
|
1860
|
+
repairLatestAssistantThinking = !escalateToAll;
|
|
1861
|
+
repairAllAssistantThinking = escalateToAll;
|
|
1852
1862
|
params = await prepareParams();
|
|
1853
1863
|
providerRetryAttempt = 0;
|
|
1854
1864
|
resetOutputForRetry();
|
|
@@ -1897,6 +1907,24 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1897
1907
|
}
|
|
1898
1908
|
}
|
|
1899
1909
|
|
|
1910
|
+
for (const block of blocksByAnthropicIndex.values()) {
|
|
1911
|
+
delete (block as { index?: number }).index;
|
|
1912
|
+
if (block.type === "toolCall") {
|
|
1913
|
+
truncatedToolCalls.add(block);
|
|
1914
|
+
if (block.partialJson.trim()) {
|
|
1915
|
+
block.arguments = parseStreamingJson(block.partialJson);
|
|
1916
|
+
}
|
|
1917
|
+
delete (block as { partialJson?: string }).partialJson;
|
|
1918
|
+
}
|
|
1919
|
+
}
|
|
1920
|
+
blocksByAnthropicIndex.clear();
|
|
1921
|
+
if (output.stopReason === "length" || !sawTerminalStopReason) {
|
|
1922
|
+
for (const block of output.content) {
|
|
1923
|
+
if (block.type === "toolCall" && truncatedToolCalls.has(block)) {
|
|
1924
|
+
block.incompleteArguments = true;
|
|
1925
|
+
}
|
|
1926
|
+
}
|
|
1927
|
+
}
|
|
1900
1928
|
output.duration = Date.now() - startTime;
|
|
1901
1929
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
1902
1930
|
if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
|
|
@@ -1909,13 +1937,12 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1909
1937
|
delete (block as { index?: number }).index;
|
|
1910
1938
|
delete (block as { partialJson?: string }).partialJson;
|
|
1911
1939
|
}
|
|
1912
|
-
const
|
|
1940
|
+
const localAbortReason = activeAbortTracker.getLocalAbortReason();
|
|
1913
1941
|
output.stopReason = activeAbortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
1914
|
-
output.errorStatus = extractHttpStatusFromError(error);
|
|
1915
|
-
output.transportFailure = transportFailureFacts(error);
|
|
1942
|
+
output.errorStatus = extractHttpStatusFromError(localAbortReason ?? error);
|
|
1943
|
+
output.transportFailure = transportFailureFacts(localAbortReason ?? error);
|
|
1916
1944
|
if (output.errorKind !== "provider_safety_stop" || !output.errorMessage) {
|
|
1917
|
-
output.errorMessage =
|
|
1918
|
-
firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
1945
|
+
output.errorMessage = localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
1919
1946
|
}
|
|
1920
1947
|
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
|
|
1921
1948
|
output.duration = Date.now() - startTime;
|
|
@@ -2100,13 +2127,26 @@ function createClient(
|
|
|
2100
2127
|
return { client, isOAuthToken: oauthToken };
|
|
2101
2128
|
}
|
|
2102
2129
|
|
|
2103
|
-
|
|
2130
|
+
/**
|
|
2131
|
+
* Anthropic rejects extended thinking combined with a forced tool choice, so such a
|
|
2132
|
+
* request drops `thinking`/`output_config`. Reports whether the forced-choice branch
|
|
2133
|
+
* applied so the caller can keep the replayed history consistent with it.
|
|
2134
|
+
*/
|
|
2135
|
+
function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming): boolean {
|
|
2104
2136
|
const toolChoice = params.tool_choice;
|
|
2105
|
-
if (!toolChoice) return;
|
|
2106
|
-
if (toolChoice.type
|
|
2107
|
-
|
|
2108
|
-
|
|
2109
|
-
|
|
2137
|
+
if (!toolChoice) return false;
|
|
2138
|
+
if (toolChoice.type !== "any" && toolChoice.type !== "tool") return false;
|
|
2139
|
+
delete params.thinking;
|
|
2140
|
+
delete params.output_config;
|
|
2141
|
+
return true;
|
|
2142
|
+
}
|
|
2143
|
+
|
|
2144
|
+
function hasNativeThinkingBlocks(messages: MessageParam[]): boolean {
|
|
2145
|
+
return messages.some(
|
|
2146
|
+
message =>
|
|
2147
|
+
Array.isArray(message.content) &&
|
|
2148
|
+
message.content.some(block => block.type === "thinking" || block.type === "redacted_thinking"),
|
|
2149
|
+
);
|
|
2110
2150
|
}
|
|
2111
2151
|
|
|
2112
2152
|
function mapAnthropicToolChoice(
|
|
@@ -2430,6 +2470,18 @@ function buildParams(
|
|
|
2430
2470
|
}
|
|
2431
2471
|
}
|
|
2432
2472
|
|
|
2473
|
+
// A forced tool choice strips `thinking` from the request. Signed thinking blocks
|
|
2474
|
+
// replayed from history belong to a thinking-enabled request, and Anthropic rejects
|
|
2475
|
+
// that pair with `thinking`/`redacted_thinking` blocks "cannot be modified", so the
|
|
2476
|
+
// replay has to degrade in the same rebuild. Runs before the billing/system payload
|
|
2477
|
+
// snapshot so the attribution hash covers the messages actually sent.
|
|
2478
|
+
if (disableThinkingIfToolChoiceForced(params) && hasNativeThinkingBlocks(params.messages)) {
|
|
2479
|
+
params.messages = convertAnthropicMessages(context.messages, model, isOAuthToken, {
|
|
2480
|
+
...thinkingRepair,
|
|
2481
|
+
repairAllAssistantThinking: true,
|
|
2482
|
+
});
|
|
2483
|
+
}
|
|
2484
|
+
|
|
2433
2485
|
const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku");
|
|
2434
2486
|
const billingSystemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
|
2435
2487
|
const billingPayload = shouldInjectClaudeCodeInstruction
|
|
@@ -2445,7 +2497,6 @@ function buildParams(
|
|
|
2445
2497
|
if (systemBlocks) {
|
|
2446
2498
|
params.system = systemBlocks;
|
|
2447
2499
|
}
|
|
2448
|
-
disableThinkingIfToolChoiceForced(params);
|
|
2449
2500
|
ensureMaxTokensForThinking(params, model);
|
|
2450
2501
|
applyPromptCaching(params as AnthropicCacheParams, cacheMode, cacheControl);
|
|
2451
2502
|
enforceCacheControlLimit(params, 4);
|
|
@@ -22,7 +22,6 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
22
22
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
23
23
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
24
24
|
import {
|
|
25
|
-
createWatchdog,
|
|
26
25
|
getOpenAIStreamIdleTimeoutMs,
|
|
27
26
|
getStreamFirstEventTimeoutMs,
|
|
28
27
|
iterateWithIdleTimeout,
|
|
@@ -118,7 +117,6 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
118
117
|
);
|
|
119
118
|
let rawRequestDump: RawHttpRequestDump | undefined;
|
|
120
119
|
const abortTracker = createAbortSourceTracker(options?.signal);
|
|
121
|
-
const firstEventTimeoutAbortError = new Error(AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE);
|
|
122
120
|
const { requestAbortController, requestSignal } = abortTracker;
|
|
123
121
|
|
|
124
122
|
try {
|
|
@@ -127,7 +125,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
127
125
|
const client = createClient(model, apiKey, options);
|
|
128
126
|
const { baseUrl } = resolveAzureConfig(model, options);
|
|
129
127
|
const params = buildParams(model, context, options, deploymentName, baseUrl);
|
|
130
|
-
const idleTimeoutMs = getOpenAIStreamIdleTimeoutMs();
|
|
128
|
+
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
|
131
129
|
options?.onPayload?.(params);
|
|
132
130
|
rawRequestDump = {
|
|
133
131
|
provider: model.provider,
|
|
@@ -164,18 +162,17 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
164
162
|
rawRequestDump = { ...rawRequestDump, body: params };
|
|
165
163
|
openaiStream = await client.responses.create(params, { signal: requestSignal });
|
|
166
164
|
}
|
|
167
|
-
const
|
|
168
|
-
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
|
169
|
-
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
170
|
-
);
|
|
165
|
+
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
|
|
171
166
|
stream.push({ type: "start", partial: output });
|
|
172
167
|
|
|
173
168
|
await processResponsesStream(
|
|
174
169
|
iterateWithIdleTimeout(openaiStream, {
|
|
175
|
-
|
|
170
|
+
firstItemTimeoutMs: firstEventTimeoutMs,
|
|
171
|
+
firstItemErrorMessage: AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE,
|
|
176
172
|
idleTimeoutMs,
|
|
177
173
|
errorMessage: "Azure OpenAI responses stream stalled while waiting for the next event",
|
|
178
174
|
onIdle: () => requestAbortController.abort(),
|
|
175
|
+
onFirstItemTimeout: () => requestAbortController.abort(),
|
|
179
176
|
}),
|
|
180
177
|
output,
|
|
181
178
|
stream,
|
|
@@ -272,16 +269,31 @@ export function resolveAzureConfigForTest(
|
|
|
272
269
|
return resolveAzureConfig(model, options);
|
|
273
270
|
}
|
|
274
271
|
|
|
272
|
+
/**
|
|
273
|
+
* Azure API key for the client, from trusted environment sources only.
|
|
274
|
+
*
|
|
275
|
+
* `$env` merges the caller's `cwd/.env`, so reading the key there would let
|
|
276
|
+
* repository content supply the credential this client authenticates with.
|
|
277
|
+
* Provider credentials are resolved from the launching shell plus GJC/user-owned
|
|
278
|
+
* `.env` files, never the project `.env` — this fallback now matches that rule.
|
|
279
|
+
*/
|
|
280
|
+
function resolveAzureClientApiKey(apiKey: string): string | undefined {
|
|
281
|
+
if (apiKey) return apiKey;
|
|
282
|
+
return $credentialEnv("AZURE_OPENAI_API_KEY");
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/** Test seam: the client API key as resolved from a caller value plus trusted env. */
|
|
286
|
+
export function resolveAzureClientApiKeyForTest(apiKey: string): string | undefined {
|
|
287
|
+
return resolveAzureClientApiKey(apiKey);
|
|
288
|
+
}
|
|
275
289
|
function createClient(model: Model<"azure-openai-responses">, apiKey: string, options?: AzureOpenAIResponsesOptions) {
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
);
|
|
282
|
-
}
|
|
283
|
-
apiKey = envKey;
|
|
290
|
+
const resolvedApiKey = resolveAzureClientApiKey(apiKey);
|
|
291
|
+
if (!resolvedApiKey) {
|
|
292
|
+
throw new Error(
|
|
293
|
+
"Azure OpenAI API key is required. Set AZURE_OPENAI_API_KEY environment variable or pass it as an argument.",
|
|
294
|
+
);
|
|
284
295
|
}
|
|
296
|
+
apiKey = resolvedApiKey;
|
|
285
297
|
|
|
286
298
|
const headers = { ...(model.headers ?? {}) };
|
|
287
299
|
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $credentialEnv, $
|
|
1
|
+
import { $credentialEnv, $pickCredentialEnv } from "@gajae-code/utils";
|
|
2
2
|
import type { Context, Model, StreamFunction } from "../types";
|
|
3
3
|
import type { AssistantMessageEventStream } from "../utils/event-stream";
|
|
4
4
|
import { getVertexAccessToken } from "./google-auth";
|
|
@@ -72,7 +72,7 @@ function resolveApiKey(options?: GoogleVertexOptions): string | undefined {
|
|
|
72
72
|
}
|
|
73
73
|
|
|
74
74
|
function resolveProject(options?: GoogleVertexOptions): string {
|
|
75
|
-
const project = options?.project || $
|
|
75
|
+
const project = options?.project || $pickCredentialEnv("GOOGLE_CLOUD_PROJECT", "GCLOUD_PROJECT");
|
|
76
76
|
if (!project) {
|
|
77
77
|
throw new Error(
|
|
78
78
|
"Vertex AI requires a project ID. Set GOOGLE_CLOUD_PROJECT/GCLOUD_PROJECT or pass project in options.",
|
|
@@ -84,10 +84,41 @@ function resolveProject(options?: GoogleVertexOptions): string {
|
|
|
84
84
|
function resolveEndpointHost(location: string): string {
|
|
85
85
|
return location === "global" ? "aiplatform.googleapis.com" : `${location}-aiplatform.googleapis.com`;
|
|
86
86
|
}
|
|
87
|
+
/**
|
|
88
|
+
* Vertex location, from trusted environment sources only and constrained to a
|
|
89
|
+
* region label.
|
|
90
|
+
*
|
|
91
|
+
* The location is interpolated into the request **host**
|
|
92
|
+
* (`${location}-aiplatform.googleapis.com`) as well as the path, and the request
|
|
93
|
+
* carries `Authorization: Bearer <accessToken>`. A value containing `/`
|
|
94
|
+
* terminates the authority component, so `evil.example.com/` resolves to origin
|
|
95
|
+
* `https://evil.example.com` and the Google access token leaves Google entirely.
|
|
96
|
+
* `$env` merges the caller's `cwd/.env`, so this was reachable from repository
|
|
97
|
+
* content.
|
|
98
|
+
*
|
|
99
|
+
* Both halves are needed: trusted resolution keeps a repository from setting it,
|
|
100
|
+
* and the shape check keeps any source from turning a region into an authority.
|
|
101
|
+
*/
|
|
102
|
+
const VERTEX_LOCATION_RE = /^[a-z0-9-]+$/;
|
|
103
|
+
|
|
104
|
+
function assertVertexLocation(location: string): string {
|
|
105
|
+
if (!VERTEX_LOCATION_RE.test(location)) {
|
|
106
|
+
throw new Error(
|
|
107
|
+
`Invalid Vertex AI location ${JSON.stringify(location)}. Expected a region label such as "us-central1" or "global".`,
|
|
108
|
+
);
|
|
109
|
+
}
|
|
110
|
+
return location;
|
|
111
|
+
}
|
|
112
|
+
|
|
87
113
|
function resolveLocation(options?: GoogleVertexOptions): string {
|
|
88
|
-
const location = options?.location || $
|
|
114
|
+
const location = options?.location || $credentialEnv("GOOGLE_CLOUD_LOCATION");
|
|
89
115
|
if (!location) {
|
|
90
116
|
throw new Error("Vertex AI requires a location. Set GOOGLE_CLOUD_LOCATION or pass location in options.");
|
|
91
117
|
}
|
|
92
|
-
return location;
|
|
118
|
+
return assertVertexLocation(location);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** Test seam: the Vertex location as resolved from options plus trusted env. */
|
|
122
|
+
export function resolveVertexLocationForTest(options?: GoogleVertexOptions): string {
|
|
123
|
+
return resolveLocation(options);
|
|
93
124
|
}
|
package/src/providers/ollama.ts
CHANGED
|
@@ -18,7 +18,7 @@ import { normalizeSystemPrompts } from "../utils";
|
|
|
18
18
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
19
19
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
20
20
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
21
|
-
import { parseStreamingJson } from "../utils/json-parse";
|
|
21
|
+
import { isCompleteJson, parseStreamingJson } from "../utils/json-parse";
|
|
22
22
|
import { resolveRetryBudget } from "../utils/retry-budget";
|
|
23
23
|
import { flattenToolRootCombinators, toolWireSchema } from "../utils/schema";
|
|
24
24
|
import {
|
|
@@ -26,6 +26,7 @@ import {
|
|
|
26
26
|
markToolChoiceIncapability,
|
|
27
27
|
resolveToolChoice,
|
|
28
28
|
} from "../utils/tool-choice-capability";
|
|
29
|
+
import { flagTruncatedToolCalls } from "./openai-responses-shared";
|
|
29
30
|
import { transformMessages } from "./transform-messages";
|
|
30
31
|
|
|
31
32
|
export interface OllamaChatOptions extends StreamOptions {
|
|
@@ -357,8 +358,10 @@ function endToolCallBlock(stream: AssistantMessageEventStream, output: Assistant
|
|
|
357
358
|
return;
|
|
358
359
|
}
|
|
359
360
|
const toolCall = block as InternalToolCallBlock;
|
|
360
|
-
if (toolCall.partialJson) {
|
|
361
|
-
|
|
361
|
+
if (toolCall.partialJson !== undefined) {
|
|
362
|
+
if (toolCall.partialJson.trim()) {
|
|
363
|
+
toolCall.arguments = parseStreamingJson<Record<string, unknown>>(toolCall.partialJson);
|
|
364
|
+
}
|
|
362
365
|
delete toolCall.partialJson;
|
|
363
366
|
}
|
|
364
367
|
stream.push({ type: "toolcall_end", contentIndex: index, toolCall, partial: output });
|
|
@@ -393,6 +396,8 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
|
|
393
396
|
let activeThinkingIndex: number | undefined;
|
|
394
397
|
let activeTextIndex: number | undefined;
|
|
395
398
|
const activeToolIndices = new Set<number>();
|
|
399
|
+
const unverifiableArgumentToolCallIds = new Set<string>();
|
|
400
|
+
let sawTerminalChunk = false;
|
|
396
401
|
try {
|
|
397
402
|
const apiKey = options.apiKey || getEnvApiKey(model.provider);
|
|
398
403
|
if (!apiKey) {
|
|
@@ -537,6 +542,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
|
|
537
542
|
for (const call of chunk.message.tool_calls) {
|
|
538
543
|
const name = call.function?.name ?? "unknown_tool";
|
|
539
544
|
const rawArgs = call.function?.arguments;
|
|
545
|
+
const unverifiableArguments = typeof rawArgs !== "string";
|
|
540
546
|
const partialJson = typeof rawArgs === "string" ? rawArgs : JSON.stringify(rawArgs ?? {});
|
|
541
547
|
const toolCall: InternalToolCallBlock = {
|
|
542
548
|
type: "toolCall",
|
|
@@ -545,6 +551,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
|
|
545
551
|
arguments: parseStreamingJson<Record<string, unknown>>(partialJson),
|
|
546
552
|
partialJson,
|
|
547
553
|
};
|
|
554
|
+
if (unverifiableArguments) unverifiableArgumentToolCallIds.add(toolCall.id);
|
|
548
555
|
output.content.push(toolCall);
|
|
549
556
|
const index = output.content.length - 1;
|
|
550
557
|
activeToolIndices.add(index);
|
|
@@ -561,6 +568,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
|
|
561
568
|
}
|
|
562
569
|
}
|
|
563
570
|
if (chunk.done) {
|
|
571
|
+
sawTerminalChunk = true;
|
|
564
572
|
if (activeThinkingIndex !== undefined) {
|
|
565
573
|
endThinkingBlock(stream, output, activeThinkingIndex);
|
|
566
574
|
activeThinkingIndex = undefined;
|
|
@@ -569,16 +577,36 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
|
|
569
577
|
endTextBlock(stream, output, activeTextIndex);
|
|
570
578
|
activeTextIndex = undefined;
|
|
571
579
|
}
|
|
580
|
+
output.stopReason = mapDoneReason(chunk.done_reason, output);
|
|
581
|
+
// Ollama still owns every partialJson buffer here; use the helper's
|
|
582
|
+
// finalized-call branch before endToolCallBlock deletes those buffers.
|
|
583
|
+
// Non-string arguments have no raw completion evidence, so a length stop
|
|
584
|
+
// must fail closed rather than trusting normalized or re-serialized values.
|
|
585
|
+
flagTruncatedToolCalls(
|
|
586
|
+
output,
|
|
587
|
+
output.stopReason,
|
|
588
|
+
block => !unverifiableArgumentToolCallIds.has(block.id),
|
|
589
|
+
);
|
|
590
|
+
if (chunk.done_reason === undefined) {
|
|
591
|
+
for (const block of output.content) {
|
|
592
|
+
if (block.type !== "toolCall") continue;
|
|
593
|
+
const partialJson = (block as InternalToolCallBlock).partialJson;
|
|
594
|
+
if (partialJson !== undefined && !isCompleteJson(partialJson)) block.incompleteArguments = true;
|
|
595
|
+
}
|
|
596
|
+
}
|
|
572
597
|
for (const index of activeToolIndices) {
|
|
573
598
|
endToolCallBlock(stream, output, index);
|
|
574
599
|
}
|
|
575
600
|
activeToolIndices.clear();
|
|
576
|
-
output.stopReason = mapDoneReason(chunk.done_reason, output);
|
|
577
601
|
output.usage.input = chunk.prompt_eval_count ?? 0;
|
|
578
602
|
output.usage.output = chunk.eval_count ?? 0;
|
|
579
603
|
output.usage.totalTokens = output.usage.input + output.usage.output;
|
|
604
|
+
break;
|
|
580
605
|
}
|
|
581
606
|
}
|
|
607
|
+
if (!sawTerminalChunk) {
|
|
608
|
+
throw new Error("Ollama stream ended before terminal done chunk");
|
|
609
|
+
}
|
|
582
610
|
output.duration = Date.now() - startTime;
|
|
583
611
|
if (firstTokenTime) {
|
|
584
612
|
output.ttft = firstTokenTime - startTime;
|