@gajae-code/ai 0.11.11 → 0.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +42 -0
- package/README.md +3 -0
- package/dist/types/model-thinking.d.ts +15 -0
- package/dist/types/provider-models/openai-compat.d.ts +7 -0
- package/dist/types/providers/anthropic.d.ts +2 -1
- package/dist/types/providers/azure-openai-responses.d.ts +8 -1
- package/dist/types/providers/google-auth.d.ts +2 -0
- package/dist/types/providers/google-gemini-headers.d.ts +1 -1
- package/dist/types/providers/google-vertex.d.ts +4 -0
- package/dist/types/providers/openai-codex-responses.d.ts +4 -0
- package/dist/types/providers/openai-completions.d.ts +2 -0
- package/dist/types/providers/openai-responses.d.ts +2 -0
- package/dist/types/types.d.ts +1 -1
- package/dist/types/usage/grok-cli.d.ts +3 -1
- package/dist/types/usage/kimi.d.ts +2 -0
- package/dist/types/utils/anthropic-auth.d.ts +8 -0
- package/dist/types/utils/fallback-transport.d.ts +2 -0
- package/dist/types/utils/foundry.d.ts +10 -0
- package/dist/types/utils/http-inspector.d.ts +23 -0
- package/dist/types/utils/idle-iterator.d.ts +4 -0
- package/dist/types/utils/oauth/bizrouter.d.ts +1 -0
- package/dist/types/utils/oauth/kimi.d.ts +2 -0
- package/dist/types/utils/oauth/perplexity.d.ts +2 -7
- package/dist/types/utils/oauth/types.d.ts +1 -1
- package/package.json +2 -2
- package/src/auth-storage.ts +6 -0
- package/src/cli.ts +1 -0
- package/src/model-thinking.ts +19 -0
- package/src/models.json +72 -0
- package/src/provider-models/descriptors.ts +7 -0
- package/src/provider-models/openai-compat.ts +67 -6
- package/src/providers/amazon-bedrock.ts +5 -20
- package/src/providers/anthropic.ts +103 -48
- package/src/providers/azure-openai-responses.ts +44 -19
- package/src/providers/google-auth.ts +13 -2
- package/src/providers/google-gemini-headers.ts +1 -1
- package/src/providers/google-vertex.ts +41 -5
- package/src/providers/ollama.ts +32 -4
- package/src/providers/openai-codex-responses.ts +97 -38
- package/src/providers/openai-completions.ts +25 -14
- package/src/providers/openai-responses.ts +22 -20
- package/src/providers/register-builtins.ts +12 -2
- package/src/stream.ts +1 -0
- package/src/types.ts +1 -0
- package/src/usage/claude.ts +2 -1
- package/src/usage/grok-cli.ts +12 -1
- package/src/usage/kimi.ts +16 -2
- package/src/utils/anthropic-auth.ts +11 -3
- package/src/utils/fallback-transport.ts +17 -10
- package/src/utils/foundry.ts +12 -2
- package/src/utils/http-inspector.ts +124 -1
- package/src/utils/idle-iterator.ts +12 -3
- package/src/utils/oauth/bizrouter.ts +15 -0
- package/src/utils/oauth/index.ts +6 -0
- package/src/utils/oauth/kimi.ts +17 -2
- package/src/utils/oauth/perplexity.ts +21 -2
- package/src/utils/oauth/types.ts +1 -0
- package/src/utils/schema/adapt.ts +2 -2
- package/src/utils/tool-choice-capability.ts +2 -1
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $
|
|
1
|
+
import { $credentialEnv } from "@gajae-code/utils";
|
|
2
2
|
import type { ModelManagerOptions } from "../model-manager";
|
|
3
3
|
import { Effort } from "../model-thinking";
|
|
4
4
|
import { getBundledModels } from "../models";
|
|
@@ -553,13 +553,19 @@ export interface OpenAIModelManagerConfig {
|
|
|
553
553
|
baseUrl?: string;
|
|
554
554
|
}
|
|
555
555
|
|
|
556
|
+
/** Base URL for the OpenAI model manager, from trusted env only (`$env` merges the caller's `cwd/.env`). */
|
|
557
|
+
function resolveOpenAIModelManagerBaseUrl(config?: OpenAIModelManagerConfig): string {
|
|
558
|
+
return config?.baseUrl?.trim() || $credentialEnv("OPENAI_BASE_URL") || OPENAI_DEFAULT_BASE_URL;
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
/** Test seam: the model-manager base URL as resolved from trusted env. */
|
|
562
|
+
export function resolveOpenAIModelManagerBaseUrlForTest(config?: OpenAIModelManagerConfig): string {
|
|
563
|
+
return resolveOpenAIModelManagerBaseUrl(config);
|
|
564
|
+
}
|
|
565
|
+
|
|
556
566
|
export function openaiModelManagerOptions(config?: OpenAIModelManagerConfig): ModelManagerOptions<"openai-responses"> {
|
|
557
567
|
const apiKey = config?.apiKey;
|
|
558
|
-
const baseUrl =
|
|
559
|
-
config?.baseUrl?.trim() ||
|
|
560
|
-
$inheritedEnv("OPENAI_BASE_URL") ||
|
|
561
|
-
$env.OPENAI_BASE_URL?.trim() ||
|
|
562
|
-
OPENAI_DEFAULT_BASE_URL;
|
|
568
|
+
const baseUrl = resolveOpenAIModelManagerBaseUrl(config);
|
|
563
569
|
const references = createBundledReferenceMap<"openai-responses">("openai");
|
|
564
570
|
return {
|
|
565
571
|
providerId: "openai",
|
|
@@ -1119,6 +1125,61 @@ export function opengatewayModelManagerOptions(
|
|
|
1119
1125
|
return createSimpleOpenAICompletionsOptions("opengateway", "https://apis.opengateway.ai/v1", config);
|
|
1120
1126
|
}
|
|
1121
1127
|
|
|
1128
|
+
// ---------------------------------------------------------------------------
|
|
1129
|
+
// 10.5.2 BizRouter
|
|
1130
|
+
// ---------------------------------------------------------------------------
|
|
1131
|
+
|
|
1132
|
+
const BIZROUTER_BASE_URL = "https://api.bizrouter.ai/v1";
|
|
1133
|
+
|
|
1134
|
+
function toBizRouterPrice(value: unknown, fallback: number): number {
|
|
1135
|
+
const parsed = toNumber(value);
|
|
1136
|
+
return parsed === undefined || parsed < 0 ? fallback : parsed;
|
|
1137
|
+
}
|
|
1138
|
+
|
|
1139
|
+
export interface BizRouterModelManagerConfig {
|
|
1140
|
+
apiKey?: string;
|
|
1141
|
+
baseUrl?: string;
|
|
1142
|
+
}
|
|
1143
|
+
|
|
1144
|
+
export function bizrouterModelManagerOptions(
|
|
1145
|
+
config?: BizRouterModelManagerConfig,
|
|
1146
|
+
): ModelManagerOptions<"openai-completions"> {
|
|
1147
|
+
const apiKey = config?.apiKey;
|
|
1148
|
+
const baseUrl = config?.baseUrl ?? BIZROUTER_BASE_URL;
|
|
1149
|
+
const references = createBundledReferenceMap<"openai-completions">("bizrouter");
|
|
1150
|
+
return {
|
|
1151
|
+
providerId: "bizrouter",
|
|
1152
|
+
...(apiKey && {
|
|
1153
|
+
fetchDynamicModels: () =>
|
|
1154
|
+
fetchOpenAICompatibleModels({
|
|
1155
|
+
api: "openai-completions",
|
|
1156
|
+
provider: "bizrouter",
|
|
1157
|
+
baseUrl,
|
|
1158
|
+
apiKey,
|
|
1159
|
+
mapModel: (entry, defaults) => {
|
|
1160
|
+
const mapped = mapWithBundledReference(entry, defaults, references.get(defaults.id));
|
|
1161
|
+
return {
|
|
1162
|
+
...mapped,
|
|
1163
|
+
name: toModelName(entry.display_name, mapped.name),
|
|
1164
|
+
contextWindow: toPositiveNumber(entry.context_length, mapped.contextWindow),
|
|
1165
|
+
maxTokens: toPositiveNumber(entry.max_output_tokens, mapped.maxTokens),
|
|
1166
|
+
input: toInputCapabilities(entry.input_modalities),
|
|
1167
|
+
cost: {
|
|
1168
|
+
input: toBizRouterPrice(entry.input_price_per_1m_usd, mapped.cost.input),
|
|
1169
|
+
output: toBizRouterPrice(entry.output_price_per_1m_usd, mapped.cost.output),
|
|
1170
|
+
cacheRead: mapped.cost.cacheRead,
|
|
1171
|
+
cacheWrite: mapped.cost.cacheWrite,
|
|
1172
|
+
},
|
|
1173
|
+
api: "openai-completions",
|
|
1174
|
+
provider: "bizrouter",
|
|
1175
|
+
baseUrl,
|
|
1176
|
+
};
|
|
1177
|
+
},
|
|
1178
|
+
}),
|
|
1179
|
+
}),
|
|
1180
|
+
};
|
|
1181
|
+
}
|
|
1182
|
+
|
|
1122
1183
|
// ---------------------------------------------------------------------------
|
|
1123
1184
|
// 10.6 Kilo Gateway
|
|
1124
1185
|
// ---------------------------------------------------------------------------
|
|
@@ -9,7 +9,11 @@
|
|
|
9
9
|
|
|
10
10
|
import { $credentialEnv, $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@gajae-code/utils";
|
|
11
11
|
import type { Effort } from "../model-thinking";
|
|
12
|
-
import {
|
|
12
|
+
import {
|
|
13
|
+
mapEffortToAnthropicAdaptiveEffort,
|
|
14
|
+
requireSupportedEffort,
|
|
15
|
+
supportsAnthropicAdaptiveThinkingDisplay as supportsAdaptiveThinkingDisplay,
|
|
16
|
+
} from "../model-thinking";
|
|
13
17
|
import { calculateCost } from "../models";
|
|
14
18
|
import type {
|
|
15
19
|
Api,
|
|
@@ -907,25 +911,6 @@ function buildAdditionalModelRequestFields(
|
|
|
907
911
|
return result;
|
|
908
912
|
}
|
|
909
913
|
|
|
910
|
-
/**
|
|
911
|
-
* Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
|
|
912
|
-
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
913
|
-
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
914
|
-
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
915
|
-
* must opt in like Opus 4.7+ (issue #2791).
|
|
916
|
-
* Bedrock model ids are prefixed with region/inference-profile slugs (e.g.
|
|
917
|
-
* `eu.anthropic.Anthropic model-opus-4-7-...`); the regex matches the `Anthropic model-opus-X-Y`
|
|
918
|
-
* fragment regardless of prefix.
|
|
919
|
-
*/
|
|
920
|
-
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
921
|
-
if (/claude-fable-\d/.test(modelId)) return true;
|
|
922
|
-
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
|
|
923
|
-
if (!match) return false;
|
|
924
|
-
const major = Number(match[1]);
|
|
925
|
-
const minor = Number(match[2]);
|
|
926
|
-
return major > 4 || (major === 4 && minor >= 7);
|
|
927
|
-
}
|
|
928
|
-
|
|
929
914
|
/**
|
|
930
915
|
* Bedrock's wire format expects the image as `{ source: { bytes: <base64-string> }, format }`.
|
|
931
916
|
* The caller already passes base64-encoded data, so no decode/re-encode round-trip is needed.
|
|
@@ -11,6 +11,7 @@ import type {
|
|
|
11
11
|
RawMessageStreamEvent,
|
|
12
12
|
} from "@anthropic-ai/sdk/resources/messages";
|
|
13
13
|
import {
|
|
14
|
+
$credentialEnv,
|
|
14
15
|
$env,
|
|
15
16
|
extractHttpStatusFromError,
|
|
16
17
|
isEnoent,
|
|
@@ -19,7 +20,11 @@ import {
|
|
|
19
20
|
logger,
|
|
20
21
|
readSseEvents,
|
|
21
22
|
} from "@gajae-code/utils";
|
|
22
|
-
import {
|
|
23
|
+
import {
|
|
24
|
+
hasOpus47ApiRestrictions,
|
|
25
|
+
mapEffortToAnthropicAdaptiveEffort,
|
|
26
|
+
supportsAnthropicAdaptiveThinkingDisplay as supportsAdaptiveThinkingDisplay,
|
|
27
|
+
} from "../model-thinking";
|
|
23
28
|
import { calculateCost } from "../models";
|
|
24
29
|
import { isUsageLimitError } from "../rate-limit-utils";
|
|
25
30
|
import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
|
|
@@ -61,12 +66,13 @@ import { transportFailureFacts } from "../utils/fallback-transport";
|
|
|
61
66
|
import { isFoundryEnabled } from "../utils/foundry";
|
|
62
67
|
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
|
|
63
68
|
import {
|
|
69
|
+
FirstEventTimeoutError,
|
|
64
70
|
getProviderFirstEventTimeoutFallbackMs,
|
|
65
71
|
getStreamFirstEventTimeoutMs,
|
|
66
72
|
getStreamIdleTimeoutMs,
|
|
67
73
|
iterateWithIdleTimeout,
|
|
68
74
|
} from "../utils/idle-iterator";
|
|
69
|
-
import { parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
|
|
75
|
+
import { isCompleteJson, parseJsonWithRepair, parseStreamingJson } from "../utils/json-parse";
|
|
70
76
|
import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot";
|
|
71
77
|
import { notifyProviderResponse } from "../utils/provider-response";
|
|
72
78
|
import { isCopilotTransientModelError } from "../utils/retry";
|
|
@@ -308,22 +314,6 @@ type AnthropicSamplingParams = MessageCreateParamsStreaming & {
|
|
|
308
314
|
const ANTHROPIC_STOP_SEQUENCES_MAX = 4;
|
|
309
315
|
let warnedStopSequencesTrim = false;
|
|
310
316
|
|
|
311
|
-
/**
|
|
312
|
-
* Adaptive thinking `display` is supported starting with Anthropic model Opus 4.7.
|
|
313
|
-
* Older adaptive-thinking models (Opus 4.6, Sonnet 4.6+) reject the field.
|
|
314
|
-
* Fable (5+) postdates Opus 4.7, accepts `display`, and defaults it to
|
|
315
|
-
* "omitted" — thinking tokens are billed but no content streams back — so it
|
|
316
|
-
* must opt in like Opus 4.7+ (issue #2791).
|
|
317
|
-
*/
|
|
318
|
-
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
|
319
|
-
if (/claude-fable-\d/.test(modelId)) return true;
|
|
320
|
-
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
|
|
321
|
-
if (!match) return false;
|
|
322
|
-
const major = Number(match[1]);
|
|
323
|
-
const minor = Number(match[2]);
|
|
324
|
-
return major > 4 || (major === 4 && minor >= 7);
|
|
325
|
-
}
|
|
326
|
-
|
|
327
317
|
const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages";
|
|
328
318
|
|
|
329
319
|
type AnthropicProviderSessionState = ProviderSessionState & {
|
|
@@ -490,7 +480,8 @@ function getCacheControl(
|
|
|
490
480
|
}
|
|
491
481
|
|
|
492
482
|
// Stealth mode: Mimic Anthropic Code headers and tool prefixing.
|
|
493
|
-
export const claudeCodeVersion = "2.1.
|
|
483
|
+
export const claudeCodeVersion = "2.1.219";
|
|
484
|
+
export const claudeCodeEntrypoint = "sdk-cli";
|
|
494
485
|
export const claudeToolPrefix: string = "proxy_";
|
|
495
486
|
export const claudeCodeSystemInstruction = "You are a Claude agent, built on Anthropic's Claude Agent SDK.";
|
|
496
487
|
|
|
@@ -566,7 +557,7 @@ function createClaudeBillingHeader(payload: unknown): string {
|
|
|
566
557
|
const buildHash = Array.from(randomBytes, byte => byte.toString(16).padStart(2, "0"))
|
|
567
558
|
.join("")
|
|
568
559
|
.slice(0, 3);
|
|
569
|
-
return `${CLAUDE_BILLING_HEADER_PREFIX} cc_version=${claudeCodeVersion}.${buildHash}; cc_entrypoint
|
|
560
|
+
return `${CLAUDE_BILLING_HEADER_PREFIX} cc_version=${claudeCodeVersion}.${buildHash}; cc_entrypoint=${claudeCodeEntrypoint}; cch=${cch};`;
|
|
570
561
|
}
|
|
571
562
|
|
|
572
563
|
const CLAUDE_CLOAKING_USER_ID_REGEX =
|
|
@@ -816,10 +807,12 @@ function resolveAnthropicBaseUrl(model: Model<"anthropic-messages">, apiKey?: st
|
|
|
816
807
|
// calls api.z.ai directly (no zcode.z.ai gateway, no captcha). Pin the base so dynamic
|
|
817
808
|
// discovery / stale bundled catalogs / model cache can't redirect it elsewhere.
|
|
818
809
|
if (model.provider === "glm-zcode") {
|
|
819
|
-
return
|
|
810
|
+
return (
|
|
811
|
+
normalizeAnthropicBaseUrl($credentialEnv("ZCODE_PLAN_ANTHROPIC_BASE_URL")) ?? "https://api.z.ai/api/anthropic"
|
|
812
|
+
);
|
|
820
813
|
}
|
|
821
814
|
if (model.provider === "anthropic" && isFoundryEnabled()) {
|
|
822
|
-
const foundryBaseUrl = normalizeAnthropicBaseUrl($
|
|
815
|
+
const foundryBaseUrl = normalizeAnthropicBaseUrl($credentialEnv("FOUNDRY_BASE_URL"));
|
|
823
816
|
if (foundryBaseUrl) {
|
|
824
817
|
return foundryBaseUrl;
|
|
825
818
|
}
|
|
@@ -1418,6 +1411,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1418
1411
|
) & { index: number };
|
|
1419
1412
|
const blocks = output.content as Block[];
|
|
1420
1413
|
const blocksByAnthropicIndex = new Map<number, Block>();
|
|
1414
|
+
const truncatedToolCalls = new Set<ToolCall>();
|
|
1415
|
+
let sawTerminalStopReason = false;
|
|
1421
1416
|
// Derive from the ACTUAL request shape, not the option default: the request
|
|
1422
1417
|
// only sends `display: "summarized"` on specific paths (adaptive display is
|
|
1423
1418
|
// omitted for models where supportsAdaptiveThinkingDisplay is false). Defaulting
|
|
@@ -1435,8 +1430,14 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1435
1430
|
// finalize the orphaned block so no internal stream fields leak into output.
|
|
1436
1431
|
const orphaned = blocksByAnthropicIndex.get(anthropicIndex);
|
|
1437
1432
|
if (orphaned) {
|
|
1438
|
-
if (orphaned.type === "toolCall"
|
|
1439
|
-
|
|
1433
|
+
if (orphaned.type === "toolCall") {
|
|
1434
|
+
if (!isCompleteJson(orphaned.partialJson)) {
|
|
1435
|
+
orphaned.incompleteArguments = true;
|
|
1436
|
+
truncatedToolCalls.add(orphaned);
|
|
1437
|
+
}
|
|
1438
|
+
if (orphaned.partialJson.trim()) {
|
|
1439
|
+
orphaned.arguments = parseStreamingJson(orphaned.partialJson);
|
|
1440
|
+
}
|
|
1440
1441
|
}
|
|
1441
1442
|
delete (orphaned as { index?: number }).index;
|
|
1442
1443
|
delete (orphaned as { partialJson?: string }).partialJson;
|
|
@@ -1453,6 +1454,8 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1453
1454
|
output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
|
|
1454
1455
|
output.stopReason = "stop";
|
|
1455
1456
|
firstTokenTime = undefined;
|
|
1457
|
+
truncatedToolCalls.clear();
|
|
1458
|
+
sawTerminalStopReason = false;
|
|
1456
1459
|
};
|
|
1457
1460
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getStreamIdleTimeoutMs();
|
|
1458
1461
|
const firstEventFallbackMs = getProviderFirstEventTimeoutFallbackMs(model.provider);
|
|
@@ -1463,12 +1466,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1463
1466
|
// Provider-level transport/rate-limit failures: only before any streamed content starts.
|
|
1464
1467
|
// Malformed envelopes/JSON: only before replay-unsafe text/tool events are visible on this stream.
|
|
1465
1468
|
let providerRetryAttempt = 0;
|
|
1466
|
-
let thinkingRepairAttempted = false;
|
|
1467
1469
|
while (true) {
|
|
1468
1470
|
// Retries reset output.content; drop stale block correlations from the aborted attempt.
|
|
1469
1471
|
blocksByAnthropicIndex.clear();
|
|
1472
|
+
truncatedToolCalls.clear();
|
|
1473
|
+
sawTerminalStopReason = false;
|
|
1470
1474
|
activeAbortTracker = createAbortSourceTracker(options?.signal);
|
|
1471
|
-
const firstEventTimeoutAbortError = new
|
|
1475
|
+
const firstEventTimeoutAbortError = new FirstEventTimeoutError(
|
|
1472
1476
|
"Anthropic stream timed out while waiting for the first event",
|
|
1473
1477
|
);
|
|
1474
1478
|
const idleTimeoutAbortError = new Error("Anthropic stream stalled while waiting for the next event");
|
|
@@ -1491,6 +1495,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1491
1495
|
let sawEvent = false;
|
|
1492
1496
|
let sawMessageStart = false;
|
|
1493
1497
|
let sawTerminalEnvelope = false;
|
|
1498
|
+
let sawMessageStop = false;
|
|
1494
1499
|
const isProgressEvent = createAnthropicStreamProgressPredicate();
|
|
1495
1500
|
|
|
1496
1501
|
for await (const event of iterateWithIdleTimeout(anthropicStream, {
|
|
@@ -1504,9 +1509,13 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1504
1509
|
isProgressItem: isProgressEvent,
|
|
1505
1510
|
})) {
|
|
1506
1511
|
sawEvent = true;
|
|
1512
|
+
if (sawMessageStop) {
|
|
1513
|
+
throw createAnthropicStreamEnvelopeError("received event after message_stop");
|
|
1514
|
+
}
|
|
1507
1515
|
if (sawProviderSafetyStop) {
|
|
1508
1516
|
if (event.type === "message_stop") {
|
|
1509
1517
|
sawTerminalEnvelope = true;
|
|
1518
|
+
sawMessageStop = true;
|
|
1510
1519
|
}
|
|
1511
1520
|
continue;
|
|
1512
1521
|
}
|
|
@@ -1694,6 +1703,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1694
1703
|
partial: output,
|
|
1695
1704
|
});
|
|
1696
1705
|
} else if (block.type === "toolCall") {
|
|
1706
|
+
if (!isCompleteJson(block.partialJson)) truncatedToolCalls.add(block);
|
|
1697
1707
|
if (block.partialJson.trim()) {
|
|
1698
1708
|
block.arguments = parseStreamingJson(block.partialJson);
|
|
1699
1709
|
}
|
|
@@ -1714,6 +1724,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1714
1724
|
if (rawStopReason) {
|
|
1715
1725
|
output.stopReason = isProviderSafetyStop ? "error" : mapStopReason(rawStopReason);
|
|
1716
1726
|
sawTerminalEnvelope = true;
|
|
1727
|
+
sawTerminalStopReason = true;
|
|
1717
1728
|
}
|
|
1718
1729
|
if (isProviderSafetyStop) {
|
|
1719
1730
|
sawProviderSafetyStop = true;
|
|
@@ -1755,6 +1766,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1755
1766
|
calculateCost(model, output.usage);
|
|
1756
1767
|
} else if (event.type === "message_stop") {
|
|
1757
1768
|
sawTerminalEnvelope = true;
|
|
1769
|
+
sawMessageStop = true;
|
|
1758
1770
|
}
|
|
1759
1771
|
}
|
|
1760
1772
|
|
|
@@ -1777,8 +1789,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1777
1789
|
}
|
|
1778
1790
|
break;
|
|
1779
1791
|
} catch (streamError) {
|
|
1780
|
-
const
|
|
1781
|
-
|
|
1792
|
+
const localAbortReason = activeAbortTracker.getLocalAbortReason();
|
|
1793
|
+
const streamFailure = localAbortReason ?? streamError;
|
|
1794
|
+
if (localAbortReason || sawProviderSafetyStop) {
|
|
1782
1795
|
throw streamFailure;
|
|
1783
1796
|
}
|
|
1784
1797
|
if (
|
|
@@ -1830,21 +1843,22 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1830
1843
|
const thinkingSignatureInvalid = isAnthropicThinkingSignatureInvalidError(streamFailure);
|
|
1831
1844
|
if (
|
|
1832
1845
|
!options?.fallbackManaged &&
|
|
1833
|
-
!
|
|
1846
|
+
!repairAllAssistantThinking &&
|
|
1834
1847
|
firstTokenTime === undefined &&
|
|
1835
1848
|
(thinkingSignatureInvalid || isAnthropicThinkingBlockMutationError(streamFailure))
|
|
1836
1849
|
) {
|
|
1850
|
+
// The mutation 400 blames the "latest assistant message", but its cited
|
|
1851
|
+
// `messages.N.content.M` path can point at an EARLIER replayed turn, so the
|
|
1852
|
+
// latest-only repair gets rejected identically. Escalate to the full-history
|
|
1853
|
+
// repair instead of burning the single retry on one scope.
|
|
1854
|
+
const escalateToAll: boolean = thinkingSignatureInvalid || repairLatestAssistantThinking;
|
|
1837
1855
|
logger.debug("anthropic: repairing assistant thinking replay after provider rejection", {
|
|
1838
1856
|
model: model.id,
|
|
1839
|
-
scope:
|
|
1857
|
+
scope: escalateToAll ? "all" : "latest",
|
|
1840
1858
|
error: streamFailure instanceof Error ? streamFailure.message : String(streamFailure),
|
|
1841
1859
|
});
|
|
1842
|
-
|
|
1843
|
-
|
|
1844
|
-
repairAllAssistantThinking = true;
|
|
1845
|
-
} else {
|
|
1846
|
-
repairLatestAssistantThinking = true;
|
|
1847
|
-
}
|
|
1860
|
+
repairLatestAssistantThinking = !escalateToAll;
|
|
1861
|
+
repairAllAssistantThinking = escalateToAll;
|
|
1848
1862
|
params = await prepareParams();
|
|
1849
1863
|
providerRetryAttempt = 0;
|
|
1850
1864
|
resetOutputForRetry();
|
|
@@ -1893,6 +1907,24 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1893
1907
|
}
|
|
1894
1908
|
}
|
|
1895
1909
|
|
|
1910
|
+
for (const block of blocksByAnthropicIndex.values()) {
|
|
1911
|
+
delete (block as { index?: number }).index;
|
|
1912
|
+
if (block.type === "toolCall") {
|
|
1913
|
+
truncatedToolCalls.add(block);
|
|
1914
|
+
if (block.partialJson.trim()) {
|
|
1915
|
+
block.arguments = parseStreamingJson(block.partialJson);
|
|
1916
|
+
}
|
|
1917
|
+
delete (block as { partialJson?: string }).partialJson;
|
|
1918
|
+
}
|
|
1919
|
+
}
|
|
1920
|
+
blocksByAnthropicIndex.clear();
|
|
1921
|
+
if (output.stopReason === "length" || !sawTerminalStopReason) {
|
|
1922
|
+
for (const block of output.content) {
|
|
1923
|
+
if (block.type === "toolCall" && truncatedToolCalls.has(block)) {
|
|
1924
|
+
block.incompleteArguments = true;
|
|
1925
|
+
}
|
|
1926
|
+
}
|
|
1927
|
+
}
|
|
1896
1928
|
output.duration = Date.now() - startTime;
|
|
1897
1929
|
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
1898
1930
|
if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
|
|
@@ -1905,13 +1937,12 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1905
1937
|
delete (block as { index?: number }).index;
|
|
1906
1938
|
delete (block as { partialJson?: string }).partialJson;
|
|
1907
1939
|
}
|
|
1908
|
-
const
|
|
1940
|
+
const localAbortReason = activeAbortTracker.getLocalAbortReason();
|
|
1909
1941
|
output.stopReason = activeAbortTracker.wasCallerAbort() ? "aborted" : "error";
|
|
1910
|
-
output.errorStatus = extractHttpStatusFromError(error);
|
|
1911
|
-
output.transportFailure = transportFailureFacts(error);
|
|
1942
|
+
output.errorStatus = extractHttpStatusFromError(localAbortReason ?? error);
|
|
1943
|
+
output.transportFailure = transportFailureFacts(localAbortReason ?? error);
|
|
1912
1944
|
if (output.errorKind !== "provider_safety_stop" || !output.errorMessage) {
|
|
1913
|
-
output.errorMessage =
|
|
1914
|
-
firstEventTimeoutError?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
1945
|
+
output.errorMessage = localAbortReason?.message ?? (await finalizeErrorMessage(error, rawRequestDump));
|
|
1915
1946
|
}
|
|
1916
1947
|
output.errorMessage = rewriteCopilotError(output.errorMessage, error, model.provider);
|
|
1917
1948
|
output.duration = Date.now() - startTime;
|
|
@@ -2096,13 +2127,26 @@ function createClient(
|
|
|
2096
2127
|
return { client, isOAuthToken: oauthToken };
|
|
2097
2128
|
}
|
|
2098
2129
|
|
|
2099
|
-
|
|
2130
|
+
/**
|
|
2131
|
+
* Anthropic rejects extended thinking combined with a forced tool choice, so such a
|
|
2132
|
+
* request drops `thinking`/`output_config`. Reports whether the forced-choice branch
|
|
2133
|
+
* applied so the caller can keep the replayed history consistent with it.
|
|
2134
|
+
*/
|
|
2135
|
+
function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming): boolean {
|
|
2100
2136
|
const toolChoice = params.tool_choice;
|
|
2101
|
-
if (!toolChoice) return;
|
|
2102
|
-
if (toolChoice.type
|
|
2103
|
-
|
|
2104
|
-
|
|
2105
|
-
|
|
2137
|
+
if (!toolChoice) return false;
|
|
2138
|
+
if (toolChoice.type !== "any" && toolChoice.type !== "tool") return false;
|
|
2139
|
+
delete params.thinking;
|
|
2140
|
+
delete params.output_config;
|
|
2141
|
+
return true;
|
|
2142
|
+
}
|
|
2143
|
+
|
|
2144
|
+
function hasNativeThinkingBlocks(messages: MessageParam[]): boolean {
|
|
2145
|
+
return messages.some(
|
|
2146
|
+
message =>
|
|
2147
|
+
Array.isArray(message.content) &&
|
|
2148
|
+
message.content.some(block => block.type === "thinking" || block.type === "redacted_thinking"),
|
|
2149
|
+
);
|
|
2106
2150
|
}
|
|
2107
2151
|
|
|
2108
2152
|
function mapAnthropicToolChoice(
|
|
@@ -2426,6 +2470,18 @@ function buildParams(
|
|
|
2426
2470
|
}
|
|
2427
2471
|
}
|
|
2428
2472
|
|
|
2473
|
+
// A forced tool choice strips `thinking` from the request. Signed thinking blocks
|
|
2474
|
+
// replayed from history belong to a thinking-enabled request, and Anthropic rejects
|
|
2475
|
+
// that pair with `thinking`/`redacted_thinking` blocks "cannot be modified", so the
|
|
2476
|
+
// replay has to degrade in the same rebuild. Runs before the billing/system payload
|
|
2477
|
+
// snapshot so the attribution hash covers the messages actually sent.
|
|
2478
|
+
if (disableThinkingIfToolChoiceForced(params) && hasNativeThinkingBlocks(params.messages)) {
|
|
2479
|
+
params.messages = convertAnthropicMessages(context.messages, model, isOAuthToken, {
|
|
2480
|
+
...thinkingRepair,
|
|
2481
|
+
repairAllAssistantThinking: true,
|
|
2482
|
+
});
|
|
2483
|
+
}
|
|
2484
|
+
|
|
2429
2485
|
const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku");
|
|
2430
2486
|
const billingSystemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
|
2431
2487
|
const billingPayload = shouldInjectClaudeCodeInstruction
|
|
@@ -2441,7 +2497,6 @@ function buildParams(
|
|
|
2441
2497
|
if (systemBlocks) {
|
|
2442
2498
|
params.system = systemBlocks;
|
|
2443
2499
|
}
|
|
2444
|
-
disableThinkingIfToolChoiceForced(params);
|
|
2445
2500
|
ensureMaxTokensForThinking(params, model);
|
|
2446
2501
|
applyPromptCaching(params as AnthropicCacheParams, cacheMode, cacheControl);
|
|
2447
2502
|
enforceCacheControlLimit(params, 4);
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
|
|
1
|
+
import { $credentialEnv, $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
|
|
2
2
|
import { AzureOpenAI } from "openai";
|
|
3
3
|
import type {
|
|
4
4
|
Tool as OpenAITool,
|
|
@@ -22,7 +22,6 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
|
22
22
|
import { transportFailureFacts } from "../utils/fallback-transport";
|
|
23
23
|
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
|
24
24
|
import {
|
|
25
|
-
createWatchdog,
|
|
26
25
|
getOpenAIStreamIdleTimeoutMs,
|
|
27
26
|
getStreamFirstEventTimeoutMs,
|
|
28
27
|
iterateWithIdleTimeout,
|
|
@@ -118,7 +117,6 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
118
117
|
);
|
|
119
118
|
let rawRequestDump: RawHttpRequestDump | undefined;
|
|
120
119
|
const abortTracker = createAbortSourceTracker(options?.signal);
|
|
121
|
-
const firstEventTimeoutAbortError = new Error(AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE);
|
|
122
120
|
const { requestAbortController, requestSignal } = abortTracker;
|
|
123
121
|
|
|
124
122
|
try {
|
|
@@ -127,7 +125,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
127
125
|
const client = createClient(model, apiKey, options);
|
|
128
126
|
const { baseUrl } = resolveAzureConfig(model, options);
|
|
129
127
|
const params = buildParams(model, context, options, deploymentName, baseUrl);
|
|
130
|
-
const idleTimeoutMs = getOpenAIStreamIdleTimeoutMs();
|
|
128
|
+
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
|
131
129
|
options?.onPayload?.(params);
|
|
132
130
|
rawRequestDump = {
|
|
133
131
|
provider: model.provider,
|
|
@@ -164,18 +162,17 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
164
162
|
rawRequestDump = { ...rawRequestDump, body: params };
|
|
165
163
|
openaiStream = await client.responses.create(params, { signal: requestSignal });
|
|
166
164
|
}
|
|
167
|
-
const
|
|
168
|
-
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
|
169
|
-
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
170
|
-
);
|
|
165
|
+
const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs);
|
|
171
166
|
stream.push({ type: "start", partial: output });
|
|
172
167
|
|
|
173
168
|
await processResponsesStream(
|
|
174
169
|
iterateWithIdleTimeout(openaiStream, {
|
|
175
|
-
|
|
170
|
+
firstItemTimeoutMs: firstEventTimeoutMs,
|
|
171
|
+
firstItemErrorMessage: AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE,
|
|
176
172
|
idleTimeoutMs,
|
|
177
173
|
errorMessage: "Azure OpenAI responses stream stalled while waiting for the next event",
|
|
178
174
|
onIdle: () => requestAbortController.abort(),
|
|
175
|
+
onFirstItemTimeout: () => requestAbortController.abort(),
|
|
179
176
|
}),
|
|
180
177
|
output,
|
|
181
178
|
stream,
|
|
@@ -234,8 +231,13 @@ function resolveAzureConfig(
|
|
|
234
231
|
): { baseUrl: string; apiVersion: string } {
|
|
235
232
|
const apiVersion = options?.azureApiVersion || $env.AZURE_OPENAI_API_VERSION || DEFAULT_AZURE_API_VERSION;
|
|
236
233
|
|
|
237
|
-
|
|
238
|
-
|
|
234
|
+
// Trusted sources only: both of these decide the request endpoint that carries
|
|
235
|
+
// the Azure credential, and `$env` merges the caller's `cwd/.env`. The resource
|
|
236
|
+
// name is the alternate constructor for the same host
|
|
237
|
+
// (`https://<resource>.openai.azure.com/openai/v1`), so it needs the same
|
|
238
|
+
// boundary as the explicit base URL.
|
|
239
|
+
const baseUrl = options?.azureBaseUrl?.trim() || $credentialEnv("AZURE_OPENAI_BASE_URL") || undefined;
|
|
240
|
+
const resourceName = options?.azureResourceName || $credentialEnv("AZURE_OPENAI_RESOURCE_NAME");
|
|
239
241
|
|
|
240
242
|
let resolvedBaseUrl = baseUrl;
|
|
241
243
|
|
|
@@ -259,16 +261,39 @@ function resolveAzureConfig(
|
|
|
259
261
|
};
|
|
260
262
|
}
|
|
261
263
|
|
|
264
|
+
/** Test seam: the Azure endpoint config as resolved from trusted env. */
|
|
265
|
+
export function resolveAzureConfigForTest(
|
|
266
|
+
model: Model<"azure-openai-responses">,
|
|
267
|
+
options?: AzureOpenAIResponsesOptions,
|
|
268
|
+
): { baseUrl: string; apiVersion: string } {
|
|
269
|
+
return resolveAzureConfig(model, options);
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* Azure API key for the client, from trusted environment sources only.
|
|
274
|
+
*
|
|
275
|
+
* `$env` merges the caller's `cwd/.env`, so reading the key there would let
|
|
276
|
+
* repository content supply the credential this client authenticates with.
|
|
277
|
+
* Provider credentials are resolved from the launching shell plus GJC/user-owned
|
|
278
|
+
* `.env` files, never the project `.env` — this fallback now matches that rule.
|
|
279
|
+
*/
|
|
280
|
+
function resolveAzureClientApiKey(apiKey: string): string | undefined {
|
|
281
|
+
if (apiKey) return apiKey;
|
|
282
|
+
return $credentialEnv("AZURE_OPENAI_API_KEY");
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/** Test seam: the client API key as resolved from a caller value plus trusted env. */
|
|
286
|
+
export function resolveAzureClientApiKeyForTest(apiKey: string): string | undefined {
|
|
287
|
+
return resolveAzureClientApiKey(apiKey);
|
|
288
|
+
}
|
|
262
289
|
function createClient(model: Model<"azure-openai-responses">, apiKey: string, options?: AzureOpenAIResponsesOptions) {
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
);
|
|
269
|
-
}
|
|
270
|
-
apiKey = envKey;
|
|
290
|
+
const resolvedApiKey = resolveAzureClientApiKey(apiKey);
|
|
291
|
+
if (!resolvedApiKey) {
|
|
292
|
+
throw new Error(
|
|
293
|
+
"Azure OpenAI API key is required. Set AZURE_OPENAI_API_KEY environment variable or pass it as an argument.",
|
|
294
|
+
);
|
|
271
295
|
}
|
|
296
|
+
apiKey = resolvedApiKey;
|
|
272
297
|
|
|
273
298
|
const headers = { ...(model.headers ?? {}) };
|
|
274
299
|
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
import { Buffer } from "node:buffer";
|
|
16
16
|
import * as os from "node:os";
|
|
17
17
|
import * as path from "node:path";
|
|
18
|
-
import { $envpos, isEnoent, logger } from "@gajae-code/utils";
|
|
18
|
+
import { $credentialEnv, $envpos, isEnoent, logger } from "@gajae-code/utils";
|
|
19
19
|
import type { FetchImpl } from "../types";
|
|
20
20
|
|
|
21
21
|
const OAUTH_TOKEN_URL = "https://oauth2.googleapis.com/token";
|
|
@@ -70,8 +70,19 @@ async function readJsonFile<T>(filePath: string): Promise<T | undefined> {
|
|
|
70
70
|
}
|
|
71
71
|
}
|
|
72
72
|
|
|
73
|
+
/** Test seam: the ADC credentials file path as resolved from trusted env. */
|
|
74
|
+
export function resolveAdcCredentialsPathForTest(): string | undefined {
|
|
75
|
+
return $credentialEnv("GOOGLE_APPLICATION_CREDENTIALS");
|
|
76
|
+
}
|
|
77
|
+
|
|
73
78
|
async function loadAdcCredentials(): Promise<{ source: string; creds: AdcFileCredentials } | undefined> {
|
|
74
|
-
|
|
79
|
+
// Trusted sources only: this path is read as service-account / authorized-user
|
|
80
|
+
// credentials and exchanged for a Google access token, so whatever can set it
|
|
81
|
+
// chooses the identity the agent authenticates as. `Bun.env` is `process.env`
|
|
82
|
+
// and the env module merges the caller's `cwd/.env` into it, so reading it
|
|
83
|
+
// there would let repository content point this at a key file it ships.
|
|
84
|
+
// `stream.ts` already resolves the same variable through `$credentialEnv`.
|
|
85
|
+
const gacPath = $credentialEnv("GOOGLE_APPLICATION_CREDENTIALS");
|
|
75
86
|
if (gacPath) {
|
|
76
87
|
const creds = await readJsonFile<AdcFileCredentials>(gacPath);
|
|
77
88
|
if (!creds) {
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
*/
|
|
6
6
|
export const GEMINI_CLI_VERSION_ENV = "GJC_AI_GEMINI_CLI_VERSION";
|
|
7
7
|
export const LEGACY_GEMINI_CLI_VERSION_ENV = "PI_AI_GEMINI_CLI_VERSION";
|
|
8
|
-
export const DEFAULT_GEMINI_CLI_VERSION = "0.
|
|
8
|
+
export const DEFAULT_GEMINI_CLI_VERSION = "0.52.0";
|
|
9
9
|
|
|
10
10
|
export function getGeminiCliUserAgent(modelId = "gemini-3.1-pro-preview"): string {
|
|
11
11
|
const version =
|