@gajae-code/ai 0.4.4 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +42 -0
- package/dist/types/index.d.ts +1 -0
- package/dist/types/providers/amazon-bedrock.d.ts +29 -5
- package/dist/types/providers/composer-discipline.d.ts +27 -0
- package/dist/types/providers/cursor.d.ts +1 -1
- package/dist/types/providers/google-gemini-cli.d.ts +1 -1
- package/dist/types/providers/google-shared.d.ts +11 -1
- package/dist/types/providers/ollama.d.ts +36 -1
- package/dist/types/providers/openai-completions-compat.d.ts +3 -1
- package/dist/types/providers/register-builtins.d.ts +3 -3
- package/dist/types/stream.d.ts +2 -2
- package/dist/types/types.d.ts +29 -3
- package/dist/types/utils/event-stream.d.ts +6 -1
- package/dist/types/utils/tool-choice-capability.d.ts +41 -0
- package/package.json +2 -2
- package/src/auth-storage.ts +5 -1
- package/src/index.ts +1 -0
- package/src/model-manager.ts +33 -1
- package/src/model-thinking.ts +9 -0
- package/src/models.json +92 -0
- package/src/models.ts +33 -7
- package/src/provider-models/openai-compat.ts +9 -1
- package/src/providers/amazon-bedrock.ts +145 -60
- package/src/providers/anthropic.ts +89 -10
- package/src/providers/azure-openai-responses.ts +44 -3
- package/src/providers/composer-discipline.ts +38 -0
- package/src/providers/cursor.ts +80 -4
- package/src/providers/google-gemini-cli.ts +69 -10
- package/src/providers/google-gemini-headers.ts +1 -1
- package/src/providers/google-shared.ts +61 -12
- package/src/providers/ollama.ts +60 -4
- package/src/providers/openai-codex-responses.ts +151 -2
- package/src/providers/openai-completions-compat.ts +9 -1
- package/src/providers/openai-completions.ts +46 -6
- package/src/providers/openai-request-transform.ts +1 -0
- package/src/providers/openai-responses.ts +54 -5
- package/src/providers/register-builtins.ts +5 -6
- package/src/stream.ts +24 -19
- package/src/types.ts +41 -3
- package/src/utils/event-stream.ts +35 -5
- package/src/utils/tool-choice-capability.ts +220 -0
|
@@ -51,6 +51,11 @@ import { getOpenAIStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/i
|
|
|
51
51
|
import { parseStreamingJson } from "../utils/json-parse";
|
|
52
52
|
import { resolveRetryBudget } from "../utils/retry-budget";
|
|
53
53
|
import { adaptSchemaForStrict, NO_STRICT, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
|
|
54
|
+
import {
|
|
55
|
+
isForcedToolChoiceUnsupportedError,
|
|
56
|
+
markToolChoiceIncapability,
|
|
57
|
+
resolveToolChoice,
|
|
58
|
+
} from "../utils/tool-choice-capability";
|
|
54
59
|
import { compactGrammarDefinition } from "./grammar";
|
|
55
60
|
import { CODEX_BASE_URL, getCodexAccountId, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "./openai-codex/constants";
|
|
56
61
|
import {
|
|
@@ -175,6 +180,47 @@ interface CodexRequestContext {
|
|
|
175
180
|
rawRequestDump: RawHttpRequestDump;
|
|
176
181
|
}
|
|
177
182
|
|
|
183
|
+
async function retryCodexInitialTransportWithoutToolChoice(
|
|
184
|
+
model: Model<"openai-codex-responses">,
|
|
185
|
+
options: OpenAICodexResponsesOptions | undefined,
|
|
186
|
+
requestSetup: CodexRequestSetup,
|
|
187
|
+
requestContext: CodexRequestContext,
|
|
188
|
+
stream: AssistantMessageEventStream,
|
|
189
|
+
error: unknown,
|
|
190
|
+
): Promise<{
|
|
191
|
+
eventStream: AsyncGenerator<Record<string, unknown>>;
|
|
192
|
+
requestBodyForState: RequestBody;
|
|
193
|
+
transport: CodexTransport;
|
|
194
|
+
}> {
|
|
195
|
+
if (
|
|
196
|
+
!isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(requestContext.transformedBody.tool_choice))
|
|
197
|
+
) {
|
|
198
|
+
throw error;
|
|
199
|
+
}
|
|
200
|
+
const reason = await finalizeErrorMessage(error, requestContext.rawRequestDump);
|
|
201
|
+
markToolChoiceIncapability(model, "auto", reason);
|
|
202
|
+
const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
|
|
203
|
+
stream.push({
|
|
204
|
+
type: "toolChoiceIncapability",
|
|
205
|
+
api: model.api,
|
|
206
|
+
provider: model.provider,
|
|
207
|
+
model: model.id,
|
|
208
|
+
requestedLevel: resolvedToolChoice.requestedLevel,
|
|
209
|
+
resolvedLevel: "auto",
|
|
210
|
+
reason,
|
|
211
|
+
registryKey: resolvedToolChoice.registryKey,
|
|
212
|
+
});
|
|
213
|
+
const next = await openCodexSseTransportWithoutToolChoice(
|
|
214
|
+
model,
|
|
215
|
+
requestContext,
|
|
216
|
+
requestSetup,
|
|
217
|
+
options,
|
|
218
|
+
requestContext.websocketState,
|
|
219
|
+
);
|
|
220
|
+
requestContext.rawRequestDump = { ...requestContext.rawRequestDump, body: next.requestBodyForState };
|
|
221
|
+
return next;
|
|
222
|
+
}
|
|
223
|
+
|
|
178
224
|
interface CodexRequestSetup {
|
|
179
225
|
requestSignal: AbortSignal;
|
|
180
226
|
wrapCodexSseStream: (source: AsyncGenerator<Record<string, unknown>>) => AsyncGenerator<Record<string, unknown>>;
|
|
@@ -601,7 +647,16 @@ async function buildTransformedCodexRequestBody(
|
|
|
601
647
|
if (context.tools && context.tools.length > 0) {
|
|
602
648
|
params.tools = convertOpenAICodexResponsesTools(context.tools, model);
|
|
603
649
|
if (options?.toolChoice) {
|
|
604
|
-
const
|
|
650
|
+
const resolvedToolChoice = resolveToolChoice(model, options.toolChoice);
|
|
651
|
+
if (resolvedToolChoice.degraded && resolvedToolChoice.supportSource === "runtime") {
|
|
652
|
+
logCodexDebug("codex degraded tool_choice after runtime capability discovery", {
|
|
653
|
+
model: model.id,
|
|
654
|
+
requestedLevel: resolvedToolChoice.requestedLevel,
|
|
655
|
+
resolvedLevel: resolvedToolChoice.resolvedLevel,
|
|
656
|
+
reason: resolvedToolChoice.reason,
|
|
657
|
+
});
|
|
658
|
+
}
|
|
659
|
+
const toolChoice = normalizeCodexToolChoice(resolvedToolChoice.resolvedChoice, context.tools, model);
|
|
605
660
|
if (toolChoice) {
|
|
606
661
|
params.tool_choice = toolChoice;
|
|
607
662
|
}
|
|
@@ -754,6 +809,22 @@ async function openCodexSseTransport(
|
|
|
754
809
|
return { eventStream, requestBodyForState: structuredCloneJSON(body), transport: "sse" };
|
|
755
810
|
}
|
|
756
811
|
|
|
812
|
+
async function openCodexSseTransportWithoutToolChoice(
|
|
813
|
+
model: Model<"openai-codex-responses">,
|
|
814
|
+
requestContext: CodexRequestContext,
|
|
815
|
+
requestSetup: CodexRequestSetup,
|
|
816
|
+
options: OpenAICodexResponsesOptions | undefined,
|
|
817
|
+
state: CodexWebSocketSessionState | undefined,
|
|
818
|
+
): Promise<{
|
|
819
|
+
eventStream: AsyncGenerator<Record<string, unknown>>;
|
|
820
|
+
requestBodyForState: RequestBody;
|
|
821
|
+
transport: CodexTransport;
|
|
822
|
+
}> {
|
|
823
|
+
const body = structuredCloneJSON(requestContext.transformedBody);
|
|
824
|
+
delete body.tool_choice;
|
|
825
|
+
return openCodexSseTransport(model, requestContext, requestSetup, options, state, body);
|
|
826
|
+
}
|
|
827
|
+
|
|
757
828
|
async function reopenCodexWebSocketRuntimeStream(
|
|
758
829
|
context: CodexStreamProcessingContext,
|
|
759
830
|
runtime: CodexStreamRuntime,
|
|
@@ -1273,6 +1344,9 @@ async function recoverCodexStreamError(
|
|
|
1273
1344
|
runtime: CodexStreamRuntime,
|
|
1274
1345
|
error: unknown,
|
|
1275
1346
|
): Promise<boolean> {
|
|
1347
|
+
if (await tryRetryWithoutForcedToolChoice(context, runtime, error)) {
|
|
1348
|
+
return true;
|
|
1349
|
+
}
|
|
1276
1350
|
if (await tryReconnectCodexWebSocketOnConnectionLimit(context, runtime, error)) {
|
|
1277
1351
|
return true;
|
|
1278
1352
|
}
|
|
@@ -1288,6 +1362,69 @@ async function recoverCodexStreamError(
|
|
|
1288
1362
|
return false;
|
|
1289
1363
|
}
|
|
1290
1364
|
|
|
1365
|
+
async function tryRetryWithoutForcedToolChoice(
|
|
1366
|
+
context: CodexStreamProcessingContext,
|
|
1367
|
+
runtime: CodexStreamRuntime,
|
|
1368
|
+
error: unknown,
|
|
1369
|
+
): Promise<boolean> {
|
|
1370
|
+
if (
|
|
1371
|
+
runtime.providerRetryAttempt > 0 ||
|
|
1372
|
+
context.output.content.length > 0 ||
|
|
1373
|
+
context.firstTokenTime !== undefined ||
|
|
1374
|
+
context.options?.signal?.aborted ||
|
|
1375
|
+
!isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(runtime.requestBodyForState.tool_choice))
|
|
1376
|
+
) {
|
|
1377
|
+
return false;
|
|
1378
|
+
}
|
|
1379
|
+
|
|
1380
|
+
const reason = await finalizeErrorMessage(error, context.requestContext.rawRequestDump);
|
|
1381
|
+
markToolChoiceIncapability(context.model, "auto", reason);
|
|
1382
|
+
const resolvedToolChoice = resolveToolChoice(context.model, context.options?.toolChoice);
|
|
1383
|
+
context.stream.push({
|
|
1384
|
+
type: "toolChoiceIncapability",
|
|
1385
|
+
api: context.model.api,
|
|
1386
|
+
provider: context.model.provider,
|
|
1387
|
+
model: context.model.id,
|
|
1388
|
+
requestedLevel: resolvedToolChoice.requestedLevel,
|
|
1389
|
+
resolvedLevel: "auto",
|
|
1390
|
+
reason,
|
|
1391
|
+
registryKey: resolvedToolChoice.registryKey,
|
|
1392
|
+
});
|
|
1393
|
+
|
|
1394
|
+
runtime.providerRetryAttempt += 1;
|
|
1395
|
+
runtime.currentItem = null;
|
|
1396
|
+
runtime.currentBlock = null;
|
|
1397
|
+
runtime.sawTerminalEvent = false;
|
|
1398
|
+
runtime.nativeOutputItems.length = 0;
|
|
1399
|
+
resetOutputState(context.output);
|
|
1400
|
+
context.firstTokenTime = undefined;
|
|
1401
|
+
|
|
1402
|
+
const websocketState = context.requestContext.websocketState;
|
|
1403
|
+
if (websocketState) {
|
|
1404
|
+
resetCodexWebSocketAppendState(websocketState);
|
|
1405
|
+
resetCodexSessionMetadata(websocketState);
|
|
1406
|
+
}
|
|
1407
|
+
const next = await openCodexSseTransportWithoutToolChoice(
|
|
1408
|
+
context.model,
|
|
1409
|
+
context.requestContext,
|
|
1410
|
+
context.requestSetup,
|
|
1411
|
+
context.options,
|
|
1412
|
+
websocketState,
|
|
1413
|
+
);
|
|
1414
|
+
runtime.eventStream = next.eventStream;
|
|
1415
|
+
runtime.requestBodyForState = next.requestBodyForState;
|
|
1416
|
+
runtime.transport = next.transport;
|
|
1417
|
+
if (websocketState) {
|
|
1418
|
+
websocketState.lastTransport = next.transport;
|
|
1419
|
+
}
|
|
1420
|
+
context.requestContext.rawRequestDump = { ...context.requestContext.rawRequestDump, body: next.requestBodyForState };
|
|
1421
|
+
return true;
|
|
1422
|
+
}
|
|
1423
|
+
|
|
1424
|
+
function isForcedCodexToolChoice(choice: RequestBody["tool_choice"]): boolean {
|
|
1425
|
+
return !!choice && choice !== "none" && choice !== "auto";
|
|
1426
|
+
}
|
|
1427
|
+
|
|
1291
1428
|
/**
|
|
1292
1429
|
* Handles `websocket_connection_limit_reached` errors by closing the stale connection
|
|
1293
1430
|
* and opening a fresh websocket. If content has already been emitted to the caller,
|
|
@@ -1546,7 +1683,19 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
|
|
|
1546
1683
|
|
|
1547
1684
|
try {
|
|
1548
1685
|
const requestContext = await buildCodexRequestContext(model, context, options, output);
|
|
1549
|
-
|
|
1686
|
+
let initialTransport: Awaited<ReturnType<typeof openInitialCodexEventStream>>;
|
|
1687
|
+
try {
|
|
1688
|
+
initialTransport = await openInitialCodexEventStream(model, options, requestSetup, requestContext);
|
|
1689
|
+
} catch (error) {
|
|
1690
|
+
initialTransport = await retryCodexInitialTransportWithoutToolChoice(
|
|
1691
|
+
model,
|
|
1692
|
+
options,
|
|
1693
|
+
requestSetup,
|
|
1694
|
+
requestContext,
|
|
1695
|
+
stream,
|
|
1696
|
+
error,
|
|
1697
|
+
);
|
|
1698
|
+
}
|
|
1550
1699
|
const runtime = createCodexStreamRuntime({
|
|
1551
1700
|
...initialTransport,
|
|
1552
1701
|
websocketState: requestContext.websocketState,
|
|
@@ -4,12 +4,17 @@ type OpenAIReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh" | "
|
|
|
4
4
|
type ResolvedToolStrictMode = NonNullable<OpenAICompat["toolStrictMode"]> | "mixed";
|
|
5
5
|
|
|
6
6
|
export type ResolvedOpenAICompat = Required<
|
|
7
|
-
Omit<
|
|
7
|
+
Omit<
|
|
8
|
+
OpenAICompat,
|
|
9
|
+
"openRouterRouting" | "vercelGatewayRouting" | "extraBody" | "toolStrictMode" | "toolChoiceSupport"
|
|
10
|
+
>
|
|
8
11
|
> & {
|
|
9
12
|
openRouterRouting?: OpenAICompat["openRouterRouting"];
|
|
10
13
|
vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
|
|
11
14
|
extraBody?: OpenAICompat["extraBody"];
|
|
12
15
|
toolStrictMode: ResolvedToolStrictMode;
|
|
16
|
+
/** Optional explicit capability override; resolved via deriveToolChoiceSupport. */
|
|
17
|
+
toolChoiceSupport?: OpenAICompat["toolChoiceSupport"];
|
|
13
18
|
};
|
|
14
19
|
|
|
15
20
|
function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
|
|
@@ -191,6 +196,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
|
|
191
196
|
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
|
|
192
197
|
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
|
|
193
198
|
supportsToolChoice: !isDirectDeepseekReasoning,
|
|
199
|
+
supportsForcedToolChoice: true,
|
|
194
200
|
maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
|
|
195
201
|
requiresToolResultName: isMistral,
|
|
196
202
|
requiresAssistantAfterToolResult: false,
|
|
@@ -254,6 +260,8 @@ export function resolveOpenAICompat(
|
|
|
254
260
|
reasoningEffortMap: { ...detected.reasoningEffortMap, ...(model.compat.reasoningEffortMap ?? {}) },
|
|
255
261
|
supportsUsageInStreaming: model.compat.supportsUsageInStreaming ?? detected.supportsUsageInStreaming,
|
|
256
262
|
supportsToolChoice: model.compat.supportsToolChoice ?? detected.supportsToolChoice,
|
|
263
|
+
supportsForcedToolChoice: model.compat.supportsForcedToolChoice ?? detected.supportsForcedToolChoice,
|
|
264
|
+
toolChoiceSupport: model.compat.toolChoiceSupport ?? detected.toolChoiceSupport,
|
|
257
265
|
maxTokensField: model.compat.maxTokensField ?? detected.maxTokensField,
|
|
258
266
|
requiresToolResultName: model.compat.requiresToolResultName ?? detected.requiresToolResultName,
|
|
259
267
|
requiresAssistantAfterToolResult:
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { $env, $inheritedEnv, extractHttpStatusFromError } from "@gajae-code/utils";
|
|
1
|
+
import { $credentialEnv, $env, $inheritedEnv, extractHttpStatusFromError, logger } from "@gajae-code/utils";
|
|
2
2
|
import OpenAI from "openai";
|
|
3
3
|
import type {
|
|
4
4
|
ChatCompletionAssistantMessageParam,
|
|
@@ -61,6 +61,12 @@ import { adaptSchemaForStrict, NO_STRICT, toolWireSchema } from "../utils/schema
|
|
|
61
61
|
import { wrapFetchForSseDebug } from "../utils/sse-debug";
|
|
62
62
|
import { type HealedToolCall, modelMayLeakKimiToolCalls, ToolCallHealer } from "../utils/tool-call-healing";
|
|
63
63
|
import { isForcedToolChoice, mapToOpenAICompletionsToolChoice } from "../utils/tool-choice";
|
|
64
|
+
import {
|
|
65
|
+
isForcedToolChoiceUnsupportedError,
|
|
66
|
+
markToolChoiceIncapability,
|
|
67
|
+
resolveToolChoice,
|
|
68
|
+
} from "../utils/tool-choice-capability";
|
|
69
|
+
import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
64
70
|
import {
|
|
65
71
|
buildCopilotDynamicHeaders,
|
|
66
72
|
hasCopilotVisionInput,
|
|
@@ -493,7 +499,25 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
493
499
|
});
|
|
494
500
|
} catch (error) {
|
|
495
501
|
const capturedErrorResponse = getCapturedErrorResponse();
|
|
496
|
-
|
|
502
|
+
const sentForcedToolChoice = isForcedToolChoice(
|
|
503
|
+
(rawRequestDump?.body as { tool_choice?: unknown } | undefined)?.tool_choice,
|
|
504
|
+
);
|
|
505
|
+
if (firstTokenTime === undefined && isForcedToolChoiceUnsupportedError(error, sentForcedToolChoice)) {
|
|
506
|
+
const reason = await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse);
|
|
507
|
+
markToolChoiceIncapability(model, "auto", reason);
|
|
508
|
+
const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
|
|
509
|
+
stream.push({
|
|
510
|
+
type: "toolChoiceIncapability",
|
|
511
|
+
api: model.api,
|
|
512
|
+
provider: model.provider,
|
|
513
|
+
model: model.id,
|
|
514
|
+
requestedLevel: resolvedToolChoice.requestedLevel,
|
|
515
|
+
resolvedLevel: "auto",
|
|
516
|
+
reason,
|
|
517
|
+
registryKey: resolvedToolChoice.registryKey,
|
|
518
|
+
});
|
|
519
|
+
openaiStream = await createCompletionsStream();
|
|
520
|
+
} else if (
|
|
497
521
|
isOpenRouterAnthropicModel(model) &&
|
|
498
522
|
!disableStrictTools &&
|
|
499
523
|
isCompiledGrammarTooLargeStrictError(error, capturedErrorResponse)
|
|
@@ -928,12 +952,12 @@ async function createClient(
|
|
|
928
952
|
clearCapturedErrorResponse: () => void;
|
|
929
953
|
}> {
|
|
930
954
|
if (!apiKey) {
|
|
931
|
-
|
|
955
|
+
apiKey = $credentialEnv("OPENAI_API_KEY");
|
|
956
|
+
if (!apiKey) {
|
|
932
957
|
throw new Error(
|
|
933
958
|
"OpenAI API key is required. Set OPENAI_API_KEY environment variable or pass it as an argument.",
|
|
934
959
|
);
|
|
935
960
|
}
|
|
936
|
-
apiKey = $env.OPENAI_API_KEY;
|
|
937
961
|
}
|
|
938
962
|
const rawApiKey = apiKey;
|
|
939
963
|
|
|
@@ -1166,8 +1190,19 @@ function buildParams(
|
|
|
1166
1190
|
params.tools = [];
|
|
1167
1191
|
}
|
|
1168
1192
|
|
|
1169
|
-
if (options?.toolChoice
|
|
1170
|
-
|
|
1193
|
+
if (options?.toolChoice) {
|
|
1194
|
+
const toolChoice = resolveToolChoice(model, options.toolChoice, compat);
|
|
1195
|
+
if (toolChoice.degraded && toolChoice.supportSource === "runtime") {
|
|
1196
|
+
logger.debug("openai-completions: degraded tool_choice after runtime capability discovery", {
|
|
1197
|
+
model: model.id,
|
|
1198
|
+
requestedLevel: toolChoice.requestedLevel,
|
|
1199
|
+
resolvedLevel: toolChoice.resolvedLevel,
|
|
1200
|
+
reason: toolChoice.reason,
|
|
1201
|
+
});
|
|
1202
|
+
}
|
|
1203
|
+
if (toolChoice.resolvedChoice !== undefined) {
|
|
1204
|
+
params.tool_choice = mapToOpenAICompletionsToolChoice(toolChoice.resolvedChoice);
|
|
1205
|
+
}
|
|
1171
1206
|
}
|
|
1172
1207
|
|
|
1173
1208
|
if (params.tool_choice === "none" && (!Array.isArray(params.tools) || params.tools.length === 0)) {
|
|
@@ -1430,6 +1465,11 @@ export function convertMessages(
|
|
|
1430
1465
|
};
|
|
1431
1466
|
|
|
1432
1467
|
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
|
1468
|
+
// Composer-harness models need anchor/edit discipline pinned ahead of the
|
|
1469
|
+
// host prompt (see composer-discipline.ts for the observed failure modes).
|
|
1470
|
+
if (systemPrompts.length > 0 && isComposerHarnessModel(model.id)) {
|
|
1471
|
+
systemPrompts.unshift(COMPOSER_EDIT_DISCIPLINE_PROMPT);
|
|
1472
|
+
}
|
|
1433
1473
|
if (systemPrompts.length > 0) {
|
|
1434
1474
|
const useDeveloperRole = model.reasoning && compat.supportsDeveloperRole;
|
|
1435
1475
|
const role = useDeveloperRole ? "developer" : "system";
|
|
@@ -1,4 +1,11 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import {
|
|
2
|
+
$credentialEnv,
|
|
3
|
+
$env,
|
|
4
|
+
$inheritedEnv,
|
|
5
|
+
extractHttpStatusFromError,
|
|
6
|
+
logger,
|
|
7
|
+
structuredCloneJSON,
|
|
8
|
+
} from "@gajae-code/utils";
|
|
2
9
|
import OpenAI from "openai";
|
|
3
10
|
import type {
|
|
4
11
|
Tool as OpenAITool,
|
|
@@ -46,6 +53,11 @@ import { resolveRetryBudget } from "../utils/retry-budget";
|
|
|
46
53
|
import { adaptSchemaForStrict, NO_STRICT, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
|
|
47
54
|
import { wrapFetchForSseDebug } from "../utils/sse-debug";
|
|
48
55
|
import { mapToOpenAIResponsesToolChoice, type OpenAIResponsesToolChoice } from "../utils/tool-choice";
|
|
56
|
+
import {
|
|
57
|
+
isForcedToolChoiceUnsupportedError,
|
|
58
|
+
markToolChoiceIncapability,
|
|
59
|
+
resolveToolChoice,
|
|
60
|
+
} from "../utils/tool-choice-capability";
|
|
49
61
|
import {
|
|
50
62
|
buildCopilotDynamicHeaders,
|
|
51
63
|
hasCopilotVisionInput,
|
|
@@ -278,7 +290,31 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
278
290
|
return data;
|
|
279
291
|
},
|
|
280
292
|
{ provider: model.provider, signal: requestSignal },
|
|
281
|
-
)
|
|
293
|
+
).catch(async error => {
|
|
294
|
+
if (!isForcedToolChoiceUnsupportedError(error, isForcedOpenAIResponsesToolChoice(params.tool_choice))) {
|
|
295
|
+
throw error;
|
|
296
|
+
}
|
|
297
|
+
const reason = await finalizeErrorMessage(error, rawRequestDump);
|
|
298
|
+
markToolChoiceIncapability(model, "auto", reason);
|
|
299
|
+
const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
|
|
300
|
+
stream.push({
|
|
301
|
+
type: "toolChoiceIncapability",
|
|
302
|
+
api: model.api,
|
|
303
|
+
provider: model.provider,
|
|
304
|
+
model: model.id,
|
|
305
|
+
requestedLevel: resolvedToolChoice.requestedLevel,
|
|
306
|
+
resolvedLevel: "auto",
|
|
307
|
+
reason,
|
|
308
|
+
registryKey: resolvedToolChoice.registryKey,
|
|
309
|
+
});
|
|
310
|
+
delete params.tool_choice;
|
|
311
|
+
if (rawRequestDump) rawRequestDump.body = params;
|
|
312
|
+
const { data, response, request_id } = await client.responses
|
|
313
|
+
.create(params, { signal: requestSignal })
|
|
314
|
+
.withResponse();
|
|
315
|
+
await notifyProviderResponse(options, response, model, request_id);
|
|
316
|
+
return data;
|
|
317
|
+
});
|
|
282
318
|
const firstEventWatchdog = createWatchdog(
|
|
283
319
|
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
|
|
284
320
|
() => abortTracker.abortLocally(firstEventTimeoutAbortError),
|
|
@@ -363,12 +399,12 @@ function createClient(
|
|
|
363
399
|
baseUrl: string | undefined;
|
|
364
400
|
} {
|
|
365
401
|
if (!apiKey) {
|
|
366
|
-
|
|
402
|
+
apiKey = $credentialEnv("OPENAI_API_KEY");
|
|
403
|
+
if (!apiKey) {
|
|
367
404
|
throw new Error(
|
|
368
405
|
"OpenAI API key is required. Set OPENAI_API_KEY environment variable or pass it as an argument.",
|
|
369
406
|
);
|
|
370
407
|
}
|
|
371
|
-
apiKey = $env.OPENAI_API_KEY;
|
|
372
408
|
}
|
|
373
409
|
const rawApiKey = apiKey;
|
|
374
410
|
|
|
@@ -490,7 +526,16 @@ function buildParams(
|
|
|
490
526
|
if (context.tools) {
|
|
491
527
|
params.tools = convertTools(context.tools, supportsStrictMode(model), model);
|
|
492
528
|
if (options?.toolChoice) {
|
|
493
|
-
|
|
529
|
+
const toolChoice = resolveToolChoice(model, options.toolChoice);
|
|
530
|
+
if (toolChoice.degraded && toolChoice.supportSource === "runtime") {
|
|
531
|
+
logger.debug("openai-responses: degraded tool_choice after runtime capability discovery", {
|
|
532
|
+
model: model.id,
|
|
533
|
+
requestedLevel: toolChoice.requestedLevel,
|
|
534
|
+
resolvedLevel: toolChoice.resolvedLevel,
|
|
535
|
+
reason: toolChoice.reason,
|
|
536
|
+
});
|
|
537
|
+
}
|
|
538
|
+
params.tool_choice = mapOpenAIResponsesToolChoiceForTools(toolChoice.resolvedChoice, context.tools, model);
|
|
494
539
|
}
|
|
495
540
|
// The apply_patch spec §1 marks only `apply_patch` itself as
|
|
496
541
|
// `supports_parallel_tool_calls = false`. OpenAI's Responses API
|
|
@@ -651,6 +696,10 @@ export function mapOpenAIResponsesToolChoiceForTools(
|
|
|
651
696
|
return customTool ? { type: "custom", name: customTool.customWireName ?? customTool.name } : mapped;
|
|
652
697
|
}
|
|
653
698
|
|
|
699
|
+
function isForcedOpenAIResponsesToolChoice(choice: unknown): boolean {
|
|
700
|
+
return !!choice && choice !== "none" && choice !== "auto";
|
|
701
|
+
}
|
|
702
|
+
|
|
654
703
|
/** @internal Exported for tests. */
|
|
655
704
|
export function convertTools(tools: Tool[], strictMode: boolean, model: Model<"openai-responses">): OpenAITool[] {
|
|
656
705
|
const allowFreeform = supportsFreeformApplyPatch(model);
|
|
@@ -6,9 +6,9 @@
|
|
|
6
6
|
* openai) at startup. The loaded module promise is cached so subsequent calls
|
|
7
7
|
* reuse the same import.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
9
|
+
* stream.ts imports its provider stream functions from this module (see the
|
|
10
|
+
* lazy wrappers below), so this file IS the main streaming path's provider
|
|
11
|
+
* loader: heavy SDKs stay out of the CLI startup parse graph.
|
|
12
12
|
*/
|
|
13
13
|
import type {
|
|
14
14
|
Api,
|
|
@@ -390,9 +390,8 @@ function loadBedrockProviderModule(): Promise<LazyProviderModule<"bedrock-conver
|
|
|
390
390
|
// ---------------------------------------------------------------------------
|
|
391
391
|
// Lazy stream function exports
|
|
392
392
|
//
|
|
393
|
-
//
|
|
394
|
-
//
|
|
395
|
-
// providers, the lazy loading will take effect on the main code path.
|
|
393
|
+
// Provider registry code imports these wrappers so the concrete provider modules
|
|
394
|
+
// are loaded on first use instead of during package initialization.
|
|
396
395
|
// ---------------------------------------------------------------------------
|
|
397
396
|
|
|
398
397
|
export const streamAnthropic = createLazyStream(loadAnthropicProviderModule);
|
package/src/stream.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import * as fs from "node:fs";
|
|
2
2
|
import * as os from "node:os";
|
|
3
3
|
import * as path from "node:path";
|
|
4
|
-
import { $
|
|
4
|
+
import { $credentialEnv, $env, $pickCredentialEnv, extractHttpStatusFromError } from "@gajae-code/utils";
|
|
5
5
|
import { getCustomApi } from "./api-registry";
|
|
6
6
|
import type { Effort } from "./model-thinking";
|
|
7
7
|
import {
|
|
@@ -61,7 +61,7 @@ let cachedVertexAdcCredentialsExists: boolean | null = null;
|
|
|
61
61
|
|
|
62
62
|
function hasVertexAdcCredentials(): boolean {
|
|
63
63
|
if (cachedVertexAdcCredentialsExists === null) {
|
|
64
|
-
const gacPath = $
|
|
64
|
+
const gacPath = $credentialEnv("GOOGLE_APPLICATION_CREDENTIALS");
|
|
65
65
|
if (gacPath) {
|
|
66
66
|
cachedVertexAdcCredentialsExists = fs.existsSync(gacPath);
|
|
67
67
|
} else {
|
|
@@ -77,7 +77,7 @@ type KeyResolver = string | (() => string | undefined);
|
|
|
77
77
|
|
|
78
78
|
const serviceProviderMap: Record<string, KeyResolver> = {
|
|
79
79
|
"alibaba-coding-plan": "ALIBABA_CODING_PLAN_API_KEY",
|
|
80
|
-
openai: () => $
|
|
80
|
+
openai: () => $credentialEnv("OPENAI_API_KEY"),
|
|
81
81
|
google: "GEMINI_API_KEY",
|
|
82
82
|
groq: "GROQ_API_KEY",
|
|
83
83
|
cerebras: "CEREBRAS_API_KEY",
|
|
@@ -107,18 +107,18 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
107
107
|
parallel: "PARALLEL_API_KEY",
|
|
108
108
|
kagi: "KAGI_API_KEY",
|
|
109
109
|
// GitHub Copilot uses GitHub personal access token
|
|
110
|
-
"github-copilot": () => $
|
|
110
|
+
"github-copilot": () => $pickCredentialEnv("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN"),
|
|
111
111
|
// Foundry mode optionally switches Anthropic auth to enterprise gateway credentials.
|
|
112
112
|
anthropic: () =>
|
|
113
113
|
isFoundryEnabled()
|
|
114
|
-
? $
|
|
115
|
-
: $
|
|
114
|
+
? $pickCredentialEnv("ANTHROPIC_FOUNDRY_API_KEY", "ANTHROPIC_OAUTH_TOKEN", "ANTHROPIC_API_KEY")
|
|
115
|
+
: $pickCredentialEnv("ANTHROPIC_OAUTH_TOKEN", "ANTHROPIC_API_KEY"),
|
|
116
116
|
"gitlab-duo": "GITLAB_TOKEN",
|
|
117
117
|
// Vertex AI supports either GOOGLE_CLOUD_API_KEY or Application Default Credentials.
|
|
118
118
|
"google-vertex": () => {
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
119
|
+
const googleCloudApiKey = $credentialEnv("GOOGLE_CLOUD_API_KEY");
|
|
120
|
+
if (googleCloudApiKey) return googleCloudApiKey;
|
|
121
|
+
|
|
122
122
|
const hasCredentials = hasVertexAdcCredentials();
|
|
123
123
|
const hasProject = !!($env.GOOGLE_CLOUD_PROJECT || $env.GCLOUD_PROJECT);
|
|
124
124
|
const hasLocation = !!$env.GOOGLE_CLOUD_LOCATION;
|
|
@@ -133,13 +133,18 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
133
133
|
// 4. AWS_CONTAINER_CREDENTIALS_* - ECS/Task IAM role credentials
|
|
134
134
|
// 5. AWS_WEB_IDENTITY_TOKEN_FILE + AWS_ROLE_ARN - IRSA (EKS) web identity
|
|
135
135
|
"amazon-bedrock": () => {
|
|
136
|
+
const awsProfile = $credentialEnv("AWS_PROFILE");
|
|
137
|
+
const awsAccessKeyId = $credentialEnv("AWS_ACCESS_KEY_ID");
|
|
138
|
+
const awsSecretAccessKey = $credentialEnv("AWS_SECRET_ACCESS_KEY");
|
|
139
|
+
const awsBearerToken = $credentialEnv("AWS_BEARER_TOKEN_BEDROCK");
|
|
136
140
|
const hasEcsCredentials =
|
|
137
|
-
!!$
|
|
138
|
-
|
|
141
|
+
!!$credentialEnv("AWS_CONTAINER_CREDENTIALS_RELATIVE_URI") ||
|
|
142
|
+
!!$credentialEnv("AWS_CONTAINER_CREDENTIALS_FULL_URI");
|
|
143
|
+
const hasWebIdentity = !!$credentialEnv("AWS_WEB_IDENTITY_TOKEN_FILE") && !!$credentialEnv("AWS_ROLE_ARN");
|
|
139
144
|
if (
|
|
140
|
-
|
|
141
|
-
(
|
|
142
|
-
|
|
145
|
+
awsProfile ||
|
|
146
|
+
(awsAccessKeyId && awsSecretAccessKey) ||
|
|
147
|
+
awsBearerToken ||
|
|
143
148
|
hasEcsCredentials ||
|
|
144
149
|
hasWebIdentity
|
|
145
150
|
) {
|
|
@@ -148,7 +153,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
148
153
|
},
|
|
149
154
|
synthetic: "SYNTHETIC_API_KEY",
|
|
150
155
|
"cloudflare-ai-gateway": "CLOUDFLARE_AI_GATEWAY_API_KEY",
|
|
151
|
-
huggingface: () => $
|
|
156
|
+
huggingface: () => $pickCredentialEnv("HUGGINGFACE_HUB_TOKEN", "HF_TOKEN"),
|
|
152
157
|
litellm: "LITELLM_API_KEY",
|
|
153
158
|
moonshot: "MOONSHOT_API_KEY",
|
|
154
159
|
nvidia: "NVIDIA_API_KEY",
|
|
@@ -158,7 +163,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
158
163
|
"ollama-cloud": "OLLAMA_CLOUD_API_KEY",
|
|
159
164
|
"llama.cpp": "LLAMA_CPP_API_KEY",
|
|
160
165
|
qianfan: "QIANFAN_API_KEY",
|
|
161
|
-
"qwen-portal": () => $
|
|
166
|
+
"qwen-portal": () => $pickCredentialEnv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
|
|
162
167
|
together: "TOGETHER_API_KEY",
|
|
163
168
|
zenmux: "ZENMUX_API_KEY",
|
|
164
169
|
venice: "VENICE_API_KEY",
|
|
@@ -169,13 +174,13 @@ const serviceProviderMap: Record<string, KeyResolver> = {
|
|
|
169
174
|
/**
|
|
170
175
|
* Get API key for provider from known environment variables, e.g. OPENAI_API_KEY.
|
|
171
176
|
*
|
|
172
|
-
*
|
|
173
|
-
*
|
|
177
|
+
* Provider authentication intentionally excludes cwd/.env values. Project dotenv files are
|
|
178
|
+
* loaded into $env for app/tool execution, but must not silently fund GJC model requests.
|
|
174
179
|
*/
|
|
175
180
|
export function getEnvApiKey(provider: string): string | undefined {
|
|
176
181
|
const resolver = serviceProviderMap[provider];
|
|
177
182
|
if (typeof resolver === "string") {
|
|
178
|
-
return $
|
|
183
|
+
return $credentialEnv(resolver);
|
|
179
184
|
}
|
|
180
185
|
return resolver?.();
|
|
181
186
|
}
|
package/src/types.ts
CHANGED
|
@@ -161,6 +161,18 @@ export type ToolChoice =
|
|
|
161
161
|
| { type: "function"; function: { name: string } }
|
|
162
162
|
| { type: "tool"; name: string };
|
|
163
163
|
|
|
164
|
+
export type ToolChoiceSupport = "none" | "auto" | "required" | "named";
|
|
165
|
+
export type ToolChoiceSupportSource = "static" | "derived" | "runtime";
|
|
166
|
+
|
|
167
|
+
export interface ToolChoiceCompat {
|
|
168
|
+
/** Maximum supported tool_choice level. */
|
|
169
|
+
toolChoiceSupport?: ToolChoiceSupport;
|
|
170
|
+
/** Legacy flag for accepting the tool_choice parameter. */
|
|
171
|
+
supportsToolChoice?: boolean;
|
|
172
|
+
/** Legacy flag for forced tool_choice support. */
|
|
173
|
+
supportsForcedToolChoice?: boolean;
|
|
174
|
+
}
|
|
175
|
+
|
|
164
176
|
// Base options all providers share
|
|
165
177
|
export type CacheRetention = "none" | "short" | "long";
|
|
166
178
|
|
|
@@ -705,13 +717,24 @@ export type AssistantMessageEvent =
|
|
|
705
717
|
contentIndex?: undefined;
|
|
706
718
|
reason: Extract<StopReason, "aborted" | "error">;
|
|
707
719
|
error: AssistantMessage;
|
|
720
|
+
}
|
|
721
|
+
| {
|
|
722
|
+
type: "toolChoiceIncapability";
|
|
723
|
+
contentIndex?: undefined;
|
|
724
|
+
api: string;
|
|
725
|
+
provider: string;
|
|
726
|
+
model: string;
|
|
727
|
+
requestedLevel: ToolChoiceSupport;
|
|
728
|
+
resolvedLevel: ToolChoiceSupport;
|
|
729
|
+
reason: string;
|
|
730
|
+
registryKey: string;
|
|
708
731
|
};
|
|
709
732
|
|
|
710
733
|
/**
|
|
711
734
|
* Compatibility settings for openai-completions API.
|
|
712
735
|
* Use this to override URL-based auto-detection for custom providers.
|
|
713
736
|
*/
|
|
714
|
-
export interface OpenAICompat {
|
|
737
|
+
export interface OpenAICompat extends ToolChoiceCompat {
|
|
715
738
|
/** Whether the provider supports the `store` field. Default: auto-detected from URL. */
|
|
716
739
|
supportsStore?: boolean;
|
|
717
740
|
/** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */
|
|
@@ -757,6 +780,8 @@ export interface OpenAICompat {
|
|
|
757
780
|
requiresAssistantContentForToolCalls?: boolean;
|
|
758
781
|
/** Whether the provider supports the `tool_choice` parameter. Default: true. */
|
|
759
782
|
supportsToolChoice?: boolean;
|
|
783
|
+
/** Whether `tool_choice` may force a tool (`required` / named tool). Default: true. */
|
|
784
|
+
supportsForcedToolChoice?: boolean;
|
|
760
785
|
/**
|
|
761
786
|
* Drop reasoning fields (`reasoning_effort`, OpenRouter `reasoning`) for
|
|
762
787
|
* the request when `tool_choice` forces a tool call. Mirrors the Anthropic
|
|
@@ -789,7 +814,7 @@ export interface OpenAICompat {
|
|
|
789
814
|
* Use this to disable features that strict-by-default Anthropic accepts but
|
|
790
815
|
* that proxy gateways (Vertex AI, AWS Bedrock-style fronts, etc.) reject.
|
|
791
816
|
*/
|
|
792
|
-
export interface AnthropicCompat {
|
|
817
|
+
export interface AnthropicCompat extends ToolChoiceCompat {
|
|
793
818
|
/**
|
|
794
819
|
* Drop the top-level `strict: true` field on tool definitions. Vertex AI's
|
|
795
820
|
* Anthropic-compatible endpoint rejects unknown tool fields with
|
|
@@ -805,6 +830,10 @@ export interface AnthropicCompat {
|
|
|
805
830
|
disableAdaptiveThinking?: boolean;
|
|
806
831
|
/** Whether tools may include Anthropic's per-tool eager_input_streaming flag. Default: true. */
|
|
807
832
|
supportsEagerToolInputStreaming?: boolean;
|
|
833
|
+
/** Whether the provider accepts the `tool_choice` parameter at all. Default: true. */
|
|
834
|
+
supportsToolChoice?: boolean;
|
|
835
|
+
/** Whether `tool_choice` may force a tool (`any` / named `tool`). Default: true except known incompatible Anthropic models. */
|
|
836
|
+
supportsForcedToolChoice?: boolean;
|
|
808
837
|
/** Whether long prompt-cache retention (`ttl: "1h"`) is supported. Default: true for canonical Anthropic API. */
|
|
809
838
|
supportsLongCacheRetention?: boolean;
|
|
810
839
|
}
|
|
@@ -907,7 +936,16 @@ export interface Model<TApi extends Api = any> {
|
|
|
907
936
|
? OpenAICompat
|
|
908
937
|
: TApi extends "anthropic-messages"
|
|
909
938
|
? AnthropicCompat
|
|
910
|
-
:
|
|
939
|
+
: TApi extends
|
|
940
|
+
| "bedrock-converse-stream"
|
|
941
|
+
| "google-generative-ai"
|
|
942
|
+
| "google-gemini-cli"
|
|
943
|
+
| "google-vertex"
|
|
944
|
+
| "ollama-chat"
|
|
945
|
+
| "azure-openai-responses"
|
|
946
|
+
| "openai-codex-responses"
|
|
947
|
+
? ToolChoiceCompat
|
|
948
|
+
: never;
|
|
911
949
|
/**
|
|
912
950
|
* Which shape to use when exposing the OpenAI code backend `apply_patch` tool to this model.
|
|
913
951
|
* Generated catalog policy sets `"freeform"` for first-party GPT-5 Responses
|