@gajae-code/ai 0.4.4 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/CHANGELOG.md +42 -0
  2. package/dist/types/index.d.ts +1 -0
  3. package/dist/types/providers/amazon-bedrock.d.ts +29 -5
  4. package/dist/types/providers/composer-discipline.d.ts +27 -0
  5. package/dist/types/providers/cursor.d.ts +1 -1
  6. package/dist/types/providers/google-gemini-cli.d.ts +1 -1
  7. package/dist/types/providers/google-shared.d.ts +11 -1
  8. package/dist/types/providers/ollama.d.ts +36 -1
  9. package/dist/types/providers/openai-completions-compat.d.ts +3 -1
  10. package/dist/types/providers/register-builtins.d.ts +3 -3
  11. package/dist/types/stream.d.ts +2 -2
  12. package/dist/types/types.d.ts +29 -3
  13. package/dist/types/utils/event-stream.d.ts +6 -1
  14. package/dist/types/utils/tool-choice-capability.d.ts +41 -0
  15. package/package.json +2 -2
  16. package/src/auth-storage.ts +5 -1
  17. package/src/index.ts +1 -0
  18. package/src/model-manager.ts +33 -1
  19. package/src/model-thinking.ts +9 -0
  20. package/src/models.json +92 -0
  21. package/src/models.ts +33 -7
  22. package/src/provider-models/openai-compat.ts +9 -1
  23. package/src/providers/amazon-bedrock.ts +145 -60
  24. package/src/providers/anthropic.ts +89 -10
  25. package/src/providers/azure-openai-responses.ts +44 -3
  26. package/src/providers/composer-discipline.ts +38 -0
  27. package/src/providers/cursor.ts +80 -4
  28. package/src/providers/google-gemini-cli.ts +69 -10
  29. package/src/providers/google-gemini-headers.ts +1 -1
  30. package/src/providers/google-shared.ts +61 -12
  31. package/src/providers/ollama.ts +60 -4
  32. package/src/providers/openai-codex-responses.ts +151 -2
  33. package/src/providers/openai-completions-compat.ts +9 -1
  34. package/src/providers/openai-completions.ts +46 -6
  35. package/src/providers/openai-request-transform.ts +1 -0
  36. package/src/providers/openai-responses.ts +54 -5
  37. package/src/providers/register-builtins.ts +5 -6
  38. package/src/stream.ts +24 -19
  39. package/src/types.ts +41 -3
  40. package/src/utils/event-stream.ts +35 -5
  41. package/src/utils/tool-choice-capability.ts +220 -0
@@ -51,6 +51,11 @@ import { getOpenAIStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/i
51
51
  import { parseStreamingJson } from "../utils/json-parse";
52
52
  import { resolveRetryBudget } from "../utils/retry-budget";
53
53
  import { adaptSchemaForStrict, NO_STRICT, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
54
+ import {
55
+ isForcedToolChoiceUnsupportedError,
56
+ markToolChoiceIncapability,
57
+ resolveToolChoice,
58
+ } from "../utils/tool-choice-capability";
54
59
  import { compactGrammarDefinition } from "./grammar";
55
60
  import { CODEX_BASE_URL, getCodexAccountId, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "./openai-codex/constants";
56
61
  import {
@@ -175,6 +180,47 @@ interface CodexRequestContext {
175
180
  rawRequestDump: RawHttpRequestDump;
176
181
  }
177
182
 
183
+ async function retryCodexInitialTransportWithoutToolChoice(
184
+ model: Model<"openai-codex-responses">,
185
+ options: OpenAICodexResponsesOptions | undefined,
186
+ requestSetup: CodexRequestSetup,
187
+ requestContext: CodexRequestContext,
188
+ stream: AssistantMessageEventStream,
189
+ error: unknown,
190
+ ): Promise<{
191
+ eventStream: AsyncGenerator<Record<string, unknown>>;
192
+ requestBodyForState: RequestBody;
193
+ transport: CodexTransport;
194
+ }> {
195
+ if (
196
+ !isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(requestContext.transformedBody.tool_choice))
197
+ ) {
198
+ throw error;
199
+ }
200
+ const reason = await finalizeErrorMessage(error, requestContext.rawRequestDump);
201
+ markToolChoiceIncapability(model, "auto", reason);
202
+ const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
203
+ stream.push({
204
+ type: "toolChoiceIncapability",
205
+ api: model.api,
206
+ provider: model.provider,
207
+ model: model.id,
208
+ requestedLevel: resolvedToolChoice.requestedLevel,
209
+ resolvedLevel: "auto",
210
+ reason,
211
+ registryKey: resolvedToolChoice.registryKey,
212
+ });
213
+ const next = await openCodexSseTransportWithoutToolChoice(
214
+ model,
215
+ requestContext,
216
+ requestSetup,
217
+ options,
218
+ requestContext.websocketState,
219
+ );
220
+ requestContext.rawRequestDump = { ...requestContext.rawRequestDump, body: next.requestBodyForState };
221
+ return next;
222
+ }
223
+
178
224
  interface CodexRequestSetup {
179
225
  requestSignal: AbortSignal;
180
226
  wrapCodexSseStream: (source: AsyncGenerator<Record<string, unknown>>) => AsyncGenerator<Record<string, unknown>>;
@@ -601,7 +647,16 @@ async function buildTransformedCodexRequestBody(
601
647
  if (context.tools && context.tools.length > 0) {
602
648
  params.tools = convertOpenAICodexResponsesTools(context.tools, model);
603
649
  if (options?.toolChoice) {
604
- const toolChoice = normalizeCodexToolChoice(options.toolChoice, context.tools, model);
650
+ const resolvedToolChoice = resolveToolChoice(model, options.toolChoice);
651
+ if (resolvedToolChoice.degraded && resolvedToolChoice.supportSource === "runtime") {
652
+ logCodexDebug("codex degraded tool_choice after runtime capability discovery", {
653
+ model: model.id,
654
+ requestedLevel: resolvedToolChoice.requestedLevel,
655
+ resolvedLevel: resolvedToolChoice.resolvedLevel,
656
+ reason: resolvedToolChoice.reason,
657
+ });
658
+ }
659
+ const toolChoice = normalizeCodexToolChoice(resolvedToolChoice.resolvedChoice, context.tools, model);
605
660
  if (toolChoice) {
606
661
  params.tool_choice = toolChoice;
607
662
  }
@@ -754,6 +809,22 @@ async function openCodexSseTransport(
754
809
  return { eventStream, requestBodyForState: structuredCloneJSON(body), transport: "sse" };
755
810
  }
756
811
 
812
+ async function openCodexSseTransportWithoutToolChoice(
813
+ model: Model<"openai-codex-responses">,
814
+ requestContext: CodexRequestContext,
815
+ requestSetup: CodexRequestSetup,
816
+ options: OpenAICodexResponsesOptions | undefined,
817
+ state: CodexWebSocketSessionState | undefined,
818
+ ): Promise<{
819
+ eventStream: AsyncGenerator<Record<string, unknown>>;
820
+ requestBodyForState: RequestBody;
821
+ transport: CodexTransport;
822
+ }> {
823
+ const body = structuredCloneJSON(requestContext.transformedBody);
824
+ delete body.tool_choice;
825
+ return openCodexSseTransport(model, requestContext, requestSetup, options, state, body);
826
+ }
827
+
757
828
  async function reopenCodexWebSocketRuntimeStream(
758
829
  context: CodexStreamProcessingContext,
759
830
  runtime: CodexStreamRuntime,
@@ -1273,6 +1344,9 @@ async function recoverCodexStreamError(
1273
1344
  runtime: CodexStreamRuntime,
1274
1345
  error: unknown,
1275
1346
  ): Promise<boolean> {
1347
+ if (await tryRetryWithoutForcedToolChoice(context, runtime, error)) {
1348
+ return true;
1349
+ }
1276
1350
  if (await tryReconnectCodexWebSocketOnConnectionLimit(context, runtime, error)) {
1277
1351
  return true;
1278
1352
  }
@@ -1288,6 +1362,69 @@ async function recoverCodexStreamError(
1288
1362
  return false;
1289
1363
  }
1290
1364
 
1365
+ async function tryRetryWithoutForcedToolChoice(
1366
+ context: CodexStreamProcessingContext,
1367
+ runtime: CodexStreamRuntime,
1368
+ error: unknown,
1369
+ ): Promise<boolean> {
1370
+ if (
1371
+ runtime.providerRetryAttempt > 0 ||
1372
+ context.output.content.length > 0 ||
1373
+ context.firstTokenTime !== undefined ||
1374
+ context.options?.signal?.aborted ||
1375
+ !isForcedToolChoiceUnsupportedError(error, isForcedCodexToolChoice(runtime.requestBodyForState.tool_choice))
1376
+ ) {
1377
+ return false;
1378
+ }
1379
+
1380
+ const reason = await finalizeErrorMessage(error, context.requestContext.rawRequestDump);
1381
+ markToolChoiceIncapability(context.model, "auto", reason);
1382
+ const resolvedToolChoice = resolveToolChoice(context.model, context.options?.toolChoice);
1383
+ context.stream.push({
1384
+ type: "toolChoiceIncapability",
1385
+ api: context.model.api,
1386
+ provider: context.model.provider,
1387
+ model: context.model.id,
1388
+ requestedLevel: resolvedToolChoice.requestedLevel,
1389
+ resolvedLevel: "auto",
1390
+ reason,
1391
+ registryKey: resolvedToolChoice.registryKey,
1392
+ });
1393
+
1394
+ runtime.providerRetryAttempt += 1;
1395
+ runtime.currentItem = null;
1396
+ runtime.currentBlock = null;
1397
+ runtime.sawTerminalEvent = false;
1398
+ runtime.nativeOutputItems.length = 0;
1399
+ resetOutputState(context.output);
1400
+ context.firstTokenTime = undefined;
1401
+
1402
+ const websocketState = context.requestContext.websocketState;
1403
+ if (websocketState) {
1404
+ resetCodexWebSocketAppendState(websocketState);
1405
+ resetCodexSessionMetadata(websocketState);
1406
+ }
1407
+ const next = await openCodexSseTransportWithoutToolChoice(
1408
+ context.model,
1409
+ context.requestContext,
1410
+ context.requestSetup,
1411
+ context.options,
1412
+ websocketState,
1413
+ );
1414
+ runtime.eventStream = next.eventStream;
1415
+ runtime.requestBodyForState = next.requestBodyForState;
1416
+ runtime.transport = next.transport;
1417
+ if (websocketState) {
1418
+ websocketState.lastTransport = next.transport;
1419
+ }
1420
+ context.requestContext.rawRequestDump = { ...context.requestContext.rawRequestDump, body: next.requestBodyForState };
1421
+ return true;
1422
+ }
1423
+
1424
+ function isForcedCodexToolChoice(choice: RequestBody["tool_choice"]): boolean {
1425
+ return !!choice && choice !== "none" && choice !== "auto";
1426
+ }
1427
+
1291
1428
  /**
1292
1429
  * Handles `websocket_connection_limit_reached` errors by closing the stale connection
1293
1430
  * and opening a fresh websocket. If content has already been emitted to the caller,
@@ -1546,7 +1683,19 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
1546
1683
 
1547
1684
  try {
1548
1685
  const requestContext = await buildCodexRequestContext(model, context, options, output);
1549
- const initialTransport = await openInitialCodexEventStream(model, options, requestSetup, requestContext);
1686
+ let initialTransport: Awaited<ReturnType<typeof openInitialCodexEventStream>>;
1687
+ try {
1688
+ initialTransport = await openInitialCodexEventStream(model, options, requestSetup, requestContext);
1689
+ } catch (error) {
1690
+ initialTransport = await retryCodexInitialTransportWithoutToolChoice(
1691
+ model,
1692
+ options,
1693
+ requestSetup,
1694
+ requestContext,
1695
+ stream,
1696
+ error,
1697
+ );
1698
+ }
1550
1699
  const runtime = createCodexStreamRuntime({
1551
1700
  ...initialTransport,
1552
1701
  websocketState: requestContext.websocketState,
@@ -4,12 +4,17 @@ type OpenAIReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh" | "
4
4
  type ResolvedToolStrictMode = NonNullable<OpenAICompat["toolStrictMode"]> | "mixed";
5
5
 
6
6
  export type ResolvedOpenAICompat = Required<
7
- Omit<OpenAICompat, "openRouterRouting" | "vercelGatewayRouting" | "extraBody" | "toolStrictMode">
7
+ Omit<
8
+ OpenAICompat,
9
+ "openRouterRouting" | "vercelGatewayRouting" | "extraBody" | "toolStrictMode" | "toolChoiceSupport"
10
+ >
8
11
  > & {
9
12
  openRouterRouting?: OpenAICompat["openRouterRouting"];
10
13
  vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"];
11
14
  extraBody?: OpenAICompat["extraBody"];
12
15
  toolStrictMode: ResolvedToolStrictMode;
16
+ /** Optional explicit capability override; resolved via deriveToolChoiceSupport. */
17
+ toolChoiceSupport?: OpenAICompat["toolChoiceSupport"];
13
18
  };
14
19
 
15
20
  function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
@@ -191,6 +196,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
191
196
  disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
192
197
  disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter,
193
198
  supportsToolChoice: !isDirectDeepseekReasoning,
199
+ supportsForcedToolChoice: true,
194
200
  maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
195
201
  requiresToolResultName: isMistral,
196
202
  requiresAssistantAfterToolResult: false,
@@ -254,6 +260,8 @@ export function resolveOpenAICompat(
254
260
  reasoningEffortMap: { ...detected.reasoningEffortMap, ...(model.compat.reasoningEffortMap ?? {}) },
255
261
  supportsUsageInStreaming: model.compat.supportsUsageInStreaming ?? detected.supportsUsageInStreaming,
256
262
  supportsToolChoice: model.compat.supportsToolChoice ?? detected.supportsToolChoice,
263
+ supportsForcedToolChoice: model.compat.supportsForcedToolChoice ?? detected.supportsForcedToolChoice,
264
+ toolChoiceSupport: model.compat.toolChoiceSupport ?? detected.toolChoiceSupport,
257
265
  maxTokensField: model.compat.maxTokensField ?? detected.maxTokensField,
258
266
  requiresToolResultName: model.compat.requiresToolResultName ?? detected.requiresToolResultName,
259
267
  requiresAssistantAfterToolResult:
@@ -1,4 +1,4 @@
1
- import { $env, $inheritedEnv, extractHttpStatusFromError } from "@gajae-code/utils";
1
+ import { $credentialEnv, $env, $inheritedEnv, extractHttpStatusFromError, logger } from "@gajae-code/utils";
2
2
  import OpenAI from "openai";
3
3
  import type {
4
4
  ChatCompletionAssistantMessageParam,
@@ -61,6 +61,12 @@ import { adaptSchemaForStrict, NO_STRICT, toolWireSchema } from "../utils/schema
61
61
  import { wrapFetchForSseDebug } from "../utils/sse-debug";
62
62
  import { type HealedToolCall, modelMayLeakKimiToolCalls, ToolCallHealer } from "../utils/tool-call-healing";
63
63
  import { isForcedToolChoice, mapToOpenAICompletionsToolChoice } from "../utils/tool-choice";
64
+ import {
65
+ isForcedToolChoiceUnsupportedError,
66
+ markToolChoiceIncapability,
67
+ resolveToolChoice,
68
+ } from "../utils/tool-choice-capability";
69
+ import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
64
70
  import {
65
71
  buildCopilotDynamicHeaders,
66
72
  hasCopilotVisionInput,
@@ -493,7 +499,25 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
493
499
  });
494
500
  } catch (error) {
495
501
  const capturedErrorResponse = getCapturedErrorResponse();
496
- if (
502
+ const sentForcedToolChoice = isForcedToolChoice(
503
+ (rawRequestDump?.body as { tool_choice?: unknown } | undefined)?.tool_choice,
504
+ );
505
+ if (firstTokenTime === undefined && isForcedToolChoiceUnsupportedError(error, sentForcedToolChoice)) {
506
+ const reason = await finalizeErrorMessage(error, rawRequestDump, capturedErrorResponse);
507
+ markToolChoiceIncapability(model, "auto", reason);
508
+ const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
509
+ stream.push({
510
+ type: "toolChoiceIncapability",
511
+ api: model.api,
512
+ provider: model.provider,
513
+ model: model.id,
514
+ requestedLevel: resolvedToolChoice.requestedLevel,
515
+ resolvedLevel: "auto",
516
+ reason,
517
+ registryKey: resolvedToolChoice.registryKey,
518
+ });
519
+ openaiStream = await createCompletionsStream();
520
+ } else if (
497
521
  isOpenRouterAnthropicModel(model) &&
498
522
  !disableStrictTools &&
499
523
  isCompiledGrammarTooLargeStrictError(error, capturedErrorResponse)
@@ -928,12 +952,12 @@ async function createClient(
928
952
  clearCapturedErrorResponse: () => void;
929
953
  }> {
930
954
  if (!apiKey) {
931
- if (!$env.OPENAI_API_KEY) {
955
+ apiKey = $credentialEnv("OPENAI_API_KEY");
956
+ if (!apiKey) {
932
957
  throw new Error(
933
958
  "OpenAI API key is required. Set OPENAI_API_KEY environment variable or pass it as an argument.",
934
959
  );
935
960
  }
936
- apiKey = $env.OPENAI_API_KEY;
937
961
  }
938
962
  const rawApiKey = apiKey;
939
963
 
@@ -1166,8 +1190,19 @@ function buildParams(
1166
1190
  params.tools = [];
1167
1191
  }
1168
1192
 
1169
- if (options?.toolChoice && compat.supportsToolChoice) {
1170
- params.tool_choice = mapToOpenAICompletionsToolChoice(options.toolChoice);
1193
+ if (options?.toolChoice) {
1194
+ const toolChoice = resolveToolChoice(model, options.toolChoice, compat);
1195
+ if (toolChoice.degraded && toolChoice.supportSource === "runtime") {
1196
+ logger.debug("openai-completions: degraded tool_choice after runtime capability discovery", {
1197
+ model: model.id,
1198
+ requestedLevel: toolChoice.requestedLevel,
1199
+ resolvedLevel: toolChoice.resolvedLevel,
1200
+ reason: toolChoice.reason,
1201
+ });
1202
+ }
1203
+ if (toolChoice.resolvedChoice !== undefined) {
1204
+ params.tool_choice = mapToOpenAICompletionsToolChoice(toolChoice.resolvedChoice);
1205
+ }
1171
1206
  }
1172
1207
 
1173
1208
  if (params.tool_choice === "none" && (!Array.isArray(params.tools) || params.tools.length === 0)) {
@@ -1430,6 +1465,11 @@ export function convertMessages(
1430
1465
  };
1431
1466
 
1432
1467
  const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
1468
+ // Composer-harness models need anchor/edit discipline pinned ahead of the
1469
+ // host prompt (see composer-discipline.ts for the observed failure modes).
1470
+ if (systemPrompts.length > 0 && isComposerHarnessModel(model.id)) {
1471
+ systemPrompts.unshift(COMPOSER_EDIT_DISCIPLINE_PROMPT);
1472
+ }
1433
1473
  if (systemPrompts.length > 0) {
1434
1474
  const useDeveloperRole = model.reasoning && compat.supportsDeveloperRole;
1435
1475
  const role = useDeveloperRole ? "developer" : "system";
@@ -39,6 +39,7 @@ const OPENAI_PROXY_STRIP_HEADERS = [
39
39
  "x-stainless-helper-method",
40
40
  "openai-organization",
41
41
  "openai-project",
42
+ "openai-beta",
42
43
  ] as const;
43
44
 
44
45
  function resolveRequestTransform(
@@ -1,4 +1,11 @@
1
- import { $env, $inheritedEnv, extractHttpStatusFromError, structuredCloneJSON } from "@gajae-code/utils";
1
+ import {
2
+ $credentialEnv,
3
+ $env,
4
+ $inheritedEnv,
5
+ extractHttpStatusFromError,
6
+ logger,
7
+ structuredCloneJSON,
8
+ } from "@gajae-code/utils";
2
9
  import OpenAI from "openai";
3
10
  import type {
4
11
  Tool as OpenAITool,
@@ -46,6 +53,11 @@ import { resolveRetryBudget } from "../utils/retry-budget";
46
53
  import { adaptSchemaForStrict, NO_STRICT, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
47
54
  import { wrapFetchForSseDebug } from "../utils/sse-debug";
48
55
  import { mapToOpenAIResponsesToolChoice, type OpenAIResponsesToolChoice } from "../utils/tool-choice";
56
+ import {
57
+ isForcedToolChoiceUnsupportedError,
58
+ markToolChoiceIncapability,
59
+ resolveToolChoice,
60
+ } from "../utils/tool-choice-capability";
49
61
  import {
50
62
  buildCopilotDynamicHeaders,
51
63
  hasCopilotVisionInput,
@@ -278,7 +290,31 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
278
290
  return data;
279
291
  },
280
292
  { provider: model.provider, signal: requestSignal },
281
- );
293
+ ).catch(async error => {
294
+ if (!isForcedToolChoiceUnsupportedError(error, isForcedOpenAIResponsesToolChoice(params.tool_choice))) {
295
+ throw error;
296
+ }
297
+ const reason = await finalizeErrorMessage(error, rawRequestDump);
298
+ markToolChoiceIncapability(model, "auto", reason);
299
+ const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
300
+ stream.push({
301
+ type: "toolChoiceIncapability",
302
+ api: model.api,
303
+ provider: model.provider,
304
+ model: model.id,
305
+ requestedLevel: resolvedToolChoice.requestedLevel,
306
+ resolvedLevel: "auto",
307
+ reason,
308
+ registryKey: resolvedToolChoice.registryKey,
309
+ });
310
+ delete params.tool_choice;
311
+ if (rawRequestDump) rawRequestDump.body = params;
312
+ const { data, response, request_id } = await client.responses
313
+ .create(params, { signal: requestSignal })
314
+ .withResponse();
315
+ await notifyProviderResponse(options, response, model, request_id);
316
+ return data;
317
+ });
282
318
  const firstEventWatchdog = createWatchdog(
283
319
  options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
284
320
  () => abortTracker.abortLocally(firstEventTimeoutAbortError),
@@ -363,12 +399,12 @@ function createClient(
363
399
  baseUrl: string | undefined;
364
400
  } {
365
401
  if (!apiKey) {
366
- if (!$env.OPENAI_API_KEY) {
402
+ apiKey = $credentialEnv("OPENAI_API_KEY");
403
+ if (!apiKey) {
367
404
  throw new Error(
368
405
  "OpenAI API key is required. Set OPENAI_API_KEY environment variable or pass it as an argument.",
369
406
  );
370
407
  }
371
- apiKey = $env.OPENAI_API_KEY;
372
408
  }
373
409
  const rawApiKey = apiKey;
374
410
 
@@ -490,7 +526,16 @@ function buildParams(
490
526
  if (context.tools) {
491
527
  params.tools = convertTools(context.tools, supportsStrictMode(model), model);
492
528
  if (options?.toolChoice) {
493
- params.tool_choice = mapOpenAIResponsesToolChoiceForTools(options.toolChoice, context.tools, model);
529
+ const toolChoice = resolveToolChoice(model, options.toolChoice);
530
+ if (toolChoice.degraded && toolChoice.supportSource === "runtime") {
531
+ logger.debug("openai-responses: degraded tool_choice after runtime capability discovery", {
532
+ model: model.id,
533
+ requestedLevel: toolChoice.requestedLevel,
534
+ resolvedLevel: toolChoice.resolvedLevel,
535
+ reason: toolChoice.reason,
536
+ });
537
+ }
538
+ params.tool_choice = mapOpenAIResponsesToolChoiceForTools(toolChoice.resolvedChoice, context.tools, model);
494
539
  }
495
540
  // The apply_patch spec §1 marks only `apply_patch` itself as
496
541
  // `supports_parallel_tool_calls = false`. OpenAI's Responses API
@@ -651,6 +696,10 @@ export function mapOpenAIResponsesToolChoiceForTools(
651
696
  return customTool ? { type: "custom", name: customTool.customWireName ?? customTool.name } : mapped;
652
697
  }
653
698
 
699
+ function isForcedOpenAIResponsesToolChoice(choice: unknown): boolean {
700
+ return !!choice && choice !== "none" && choice !== "auto";
701
+ }
702
+
654
703
  /** @internal Exported for tests. */
655
704
  export function convertTools(tools: Tool[], strictMode: boolean, model: Model<"openai-responses">): OpenAITool[] {
656
705
  const allowFreeform = supportsFreeformApplyPatch(model);
@@ -6,9 +6,9 @@
6
6
  * openai) at startup. The loaded module promise is cached so subsequent calls
7
7
  * reuse the same import.
8
8
  *
9
- * NOTE: stream.ts currently imports providers directly, so this file is not yet
10
- * wired into the main streaming path. It provides the infrastructure for lazy
11
- * loading that can be integrated when stream.ts is refactored.
9
+ * stream.ts imports its provider stream functions from this module (see the
10
+ * lazy wrappers below), so this file IS the main streaming path's provider
11
+ * loader: heavy SDKs stay out of the CLI startup parse graph.
12
12
  */
13
13
  import type {
14
14
  Api,
@@ -390,9 +390,8 @@ function loadBedrockProviderModule(): Promise<LazyProviderModule<"bedrock-conver
390
390
  // ---------------------------------------------------------------------------
391
391
  // Lazy stream function exports
392
392
  //
393
- // These use the same names as the direct provider stream functions. When
394
- // stream.ts is updated to import from this module instead of individual
395
- // providers, the lazy loading will take effect on the main code path.
393
+ // Provider registry code imports these wrappers so the concrete provider modules
394
+ // are loaded on first use instead of during package initialization.
396
395
  // ---------------------------------------------------------------------------
397
396
 
398
397
  export const streamAnthropic = createLazyStream(loadAnthropicProviderModule);
package/src/stream.ts CHANGED
@@ -1,7 +1,7 @@
1
1
  import * as fs from "node:fs";
2
2
  import * as os from "node:os";
3
3
  import * as path from "node:path";
4
- import { $env, $inheritedEnv, $pickenv, extractHttpStatusFromError } from "@gajae-code/utils";
4
+ import { $credentialEnv, $env, $pickCredentialEnv, extractHttpStatusFromError } from "@gajae-code/utils";
5
5
  import { getCustomApi } from "./api-registry";
6
6
  import type { Effort } from "./model-thinking";
7
7
  import {
@@ -61,7 +61,7 @@ let cachedVertexAdcCredentialsExists: boolean | null = null;
61
61
 
62
62
  function hasVertexAdcCredentials(): boolean {
63
63
  if (cachedVertexAdcCredentialsExists === null) {
64
- const gacPath = $env.GOOGLE_APPLICATION_CREDENTIALS;
64
+ const gacPath = $credentialEnv("GOOGLE_APPLICATION_CREDENTIALS");
65
65
  if (gacPath) {
66
66
  cachedVertexAdcCredentialsExists = fs.existsSync(gacPath);
67
67
  } else {
@@ -77,7 +77,7 @@ type KeyResolver = string | (() => string | undefined);
77
77
 
78
78
  const serviceProviderMap: Record<string, KeyResolver> = {
79
79
  "alibaba-coding-plan": "ALIBABA_CODING_PLAN_API_KEY",
80
- openai: () => $inheritedEnv("OPENAI_API_KEY") ?? $env.OPENAI_API_KEY,
80
+ openai: () => $credentialEnv("OPENAI_API_KEY"),
81
81
  google: "GEMINI_API_KEY",
82
82
  groq: "GROQ_API_KEY",
83
83
  cerebras: "CEREBRAS_API_KEY",
@@ -107,18 +107,18 @@ const serviceProviderMap: Record<string, KeyResolver> = {
107
107
  parallel: "PARALLEL_API_KEY",
108
108
  kagi: "KAGI_API_KEY",
109
109
  // GitHub Copilot uses GitHub personal access token
110
- "github-copilot": () => $pickenv("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN"),
110
+ "github-copilot": () => $pickCredentialEnv("COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN"),
111
111
  // Foundry mode optionally switches Anthropic auth to enterprise gateway credentials.
112
112
  anthropic: () =>
113
113
  isFoundryEnabled()
114
- ? $pickenv("ANTHROPIC_FOUNDRY_API_KEY", "ANTHROPIC_OAUTH_TOKEN", "ANTHROPIC_API_KEY")
115
- : $pickenv("ANTHROPIC_OAUTH_TOKEN", "ANTHROPIC_API_KEY"),
114
+ ? $pickCredentialEnv("ANTHROPIC_FOUNDRY_API_KEY", "ANTHROPIC_OAUTH_TOKEN", "ANTHROPIC_API_KEY")
115
+ : $pickCredentialEnv("ANTHROPIC_OAUTH_TOKEN", "ANTHROPIC_API_KEY"),
116
116
  "gitlab-duo": "GITLAB_TOKEN",
117
117
  // Vertex AI supports either GOOGLE_CLOUD_API_KEY or Application Default Credentials.
118
118
  "google-vertex": () => {
119
- if ($env.GOOGLE_CLOUD_API_KEY) {
120
- return $env.GOOGLE_CLOUD_API_KEY;
121
- }
119
+ const googleCloudApiKey = $credentialEnv("GOOGLE_CLOUD_API_KEY");
120
+ if (googleCloudApiKey) return googleCloudApiKey;
121
+
122
122
  const hasCredentials = hasVertexAdcCredentials();
123
123
  const hasProject = !!($env.GOOGLE_CLOUD_PROJECT || $env.GCLOUD_PROJECT);
124
124
  const hasLocation = !!$env.GOOGLE_CLOUD_LOCATION;
@@ -133,13 +133,18 @@ const serviceProviderMap: Record<string, KeyResolver> = {
133
133
  // 4. AWS_CONTAINER_CREDENTIALS_* - ECS/Task IAM role credentials
134
134
  // 5. AWS_WEB_IDENTITY_TOKEN_FILE + AWS_ROLE_ARN - IRSA (EKS) web identity
135
135
  "amazon-bedrock": () => {
136
+ const awsProfile = $credentialEnv("AWS_PROFILE");
137
+ const awsAccessKeyId = $credentialEnv("AWS_ACCESS_KEY_ID");
138
+ const awsSecretAccessKey = $credentialEnv("AWS_SECRET_ACCESS_KEY");
139
+ const awsBearerToken = $credentialEnv("AWS_BEARER_TOKEN_BEDROCK");
136
140
  const hasEcsCredentials =
137
- !!$env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI || !!$env.AWS_CONTAINER_CREDENTIALS_FULL_URI;
138
- const hasWebIdentity = !!$env.AWS_WEB_IDENTITY_TOKEN_FILE && !!$env.AWS_ROLE_ARN;
141
+ !!$credentialEnv("AWS_CONTAINER_CREDENTIALS_RELATIVE_URI") ||
142
+ !!$credentialEnv("AWS_CONTAINER_CREDENTIALS_FULL_URI");
143
+ const hasWebIdentity = !!$credentialEnv("AWS_WEB_IDENTITY_TOKEN_FILE") && !!$credentialEnv("AWS_ROLE_ARN");
139
144
  if (
140
- $env.AWS_PROFILE ||
141
- ($env.AWS_ACCESS_KEY_ID && $env.AWS_SECRET_ACCESS_KEY) ||
142
- $env.AWS_BEARER_TOKEN_BEDROCK ||
145
+ awsProfile ||
146
+ (awsAccessKeyId && awsSecretAccessKey) ||
147
+ awsBearerToken ||
143
148
  hasEcsCredentials ||
144
149
  hasWebIdentity
145
150
  ) {
@@ -148,7 +153,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
148
153
  },
149
154
  synthetic: "SYNTHETIC_API_KEY",
150
155
  "cloudflare-ai-gateway": "CLOUDFLARE_AI_GATEWAY_API_KEY",
151
- huggingface: () => $pickenv("HUGGINGFACE_HUB_TOKEN", "HF_TOKEN"),
156
+ huggingface: () => $pickCredentialEnv("HUGGINGFACE_HUB_TOKEN", "HF_TOKEN"),
152
157
  litellm: "LITELLM_API_KEY",
153
158
  moonshot: "MOONSHOT_API_KEY",
154
159
  nvidia: "NVIDIA_API_KEY",
@@ -158,7 +163,7 @@ const serviceProviderMap: Record<string, KeyResolver> = {
158
163
  "ollama-cloud": "OLLAMA_CLOUD_API_KEY",
159
164
  "llama.cpp": "LLAMA_CPP_API_KEY",
160
165
  qianfan: "QIANFAN_API_KEY",
161
- "qwen-portal": () => $pickenv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
166
+ "qwen-portal": () => $pickCredentialEnv("QWEN_OAUTH_TOKEN", "QWEN_PORTAL_API_KEY"),
162
167
  together: "TOGETHER_API_KEY",
163
168
  zenmux: "ZENMUX_API_KEY",
164
169
  venice: "VENICE_API_KEY",
@@ -169,13 +174,13 @@ const serviceProviderMap: Record<string, KeyResolver> = {
169
174
  /**
170
175
  * Get API key for provider from known environment variables, e.g. OPENAI_API_KEY.
171
176
  *
172
- * Will not return API keys for providers that require OAuth tokens.
173
- * Checks Bun.env, then cwd/.env, then ~/.env.
177
+ * Provider authentication intentionally excludes cwd/.env values. Project dotenv files are
178
+ * loaded into $env for app/tool execution, but must not silently fund GJC model requests.
174
179
  */
175
180
  export function getEnvApiKey(provider: string): string | undefined {
176
181
  const resolver = serviceProviderMap[provider];
177
182
  if (typeof resolver === "string") {
178
- return $env[resolver];
183
+ return $credentialEnv(resolver);
179
184
  }
180
185
  return resolver?.();
181
186
  }
package/src/types.ts CHANGED
@@ -161,6 +161,18 @@ export type ToolChoice =
161
161
  | { type: "function"; function: { name: string } }
162
162
  | { type: "tool"; name: string };
163
163
 
164
+ export type ToolChoiceSupport = "none" | "auto" | "required" | "named";
165
+ export type ToolChoiceSupportSource = "static" | "derived" | "runtime";
166
+
167
+ export interface ToolChoiceCompat {
168
+ /** Maximum supported tool_choice level. */
169
+ toolChoiceSupport?: ToolChoiceSupport;
170
+ /** Legacy flag for accepting the tool_choice parameter. */
171
+ supportsToolChoice?: boolean;
172
+ /** Legacy flag for forced tool_choice support. */
173
+ supportsForcedToolChoice?: boolean;
174
+ }
175
+
164
176
  // Base options all providers share
165
177
  export type CacheRetention = "none" | "short" | "long";
166
178
 
@@ -705,13 +717,24 @@ export type AssistantMessageEvent =
705
717
  contentIndex?: undefined;
706
718
  reason: Extract<StopReason, "aborted" | "error">;
707
719
  error: AssistantMessage;
720
+ }
721
+ | {
722
+ type: "toolChoiceIncapability";
723
+ contentIndex?: undefined;
724
+ api: string;
725
+ provider: string;
726
+ model: string;
727
+ requestedLevel: ToolChoiceSupport;
728
+ resolvedLevel: ToolChoiceSupport;
729
+ reason: string;
730
+ registryKey: string;
708
731
  };
709
732
 
710
733
  /**
711
734
  * Compatibility settings for openai-completions API.
712
735
  * Use this to override URL-based auto-detection for custom providers.
713
736
  */
714
- export interface OpenAICompat {
737
+ export interface OpenAICompat extends ToolChoiceCompat {
715
738
  /** Whether the provider supports the `store` field. Default: auto-detected from URL. */
716
739
  supportsStore?: boolean;
717
740
  /** Whether the provider supports the `developer` role (vs `system`). Default: auto-detected from URL. */
@@ -757,6 +780,8 @@ export interface OpenAICompat {
757
780
  requiresAssistantContentForToolCalls?: boolean;
758
781
  /** Whether the provider supports the `tool_choice` parameter. Default: true. */
759
782
  supportsToolChoice?: boolean;
783
+ /** Whether `tool_choice` may force a tool (`required` / named tool). Default: true. */
784
+ supportsForcedToolChoice?: boolean;
760
785
  /**
761
786
  * Drop reasoning fields (`reasoning_effort`, OpenRouter `reasoning`) for
762
787
  * the request when `tool_choice` forces a tool call. Mirrors the Anthropic
@@ -789,7 +814,7 @@ export interface OpenAICompat {
789
814
  * Use this to disable features that strict-by-default Anthropic accepts but
790
815
  * that proxy gateways (Vertex AI, AWS Bedrock-style fronts, etc.) reject.
791
816
  */
792
- export interface AnthropicCompat {
817
+ export interface AnthropicCompat extends ToolChoiceCompat {
793
818
  /**
794
819
  * Drop the top-level `strict: true` field on tool definitions. Vertex AI's
795
820
  * Anthropic-compatible endpoint rejects unknown tool fields with
@@ -805,6 +830,10 @@ export interface AnthropicCompat {
805
830
  disableAdaptiveThinking?: boolean;
806
831
  /** Whether tools may include Anthropic's per-tool eager_input_streaming flag. Default: true. */
807
832
  supportsEagerToolInputStreaming?: boolean;
833
+ /** Whether the provider accepts the `tool_choice` parameter at all. Default: true. */
834
+ supportsToolChoice?: boolean;
835
+ /** Whether `tool_choice` may force a tool (`any` / named `tool`). Default: true except known incompatible Anthropic models. */
836
+ supportsForcedToolChoice?: boolean;
808
837
  /** Whether long prompt-cache retention (`ttl: "1h"`) is supported. Default: true for canonical Anthropic API. */
809
838
  supportsLongCacheRetention?: boolean;
810
839
  }
@@ -907,7 +936,16 @@ export interface Model<TApi extends Api = any> {
907
936
  ? OpenAICompat
908
937
  : TApi extends "anthropic-messages"
909
938
  ? AnthropicCompat
910
- : never;
939
+ : TApi extends
940
+ | "bedrock-converse-stream"
941
+ | "google-generative-ai"
942
+ | "google-gemini-cli"
943
+ | "google-vertex"
944
+ | "ollama-chat"
945
+ | "azure-openai-responses"
946
+ | "openai-codex-responses"
947
+ ? ToolChoiceCompat
948
+ : never;
911
949
  /**
912
950
  * Which shape to use when exposing the OpenAI code backend `apply_patch` tool to this model.
913
951
  * Generated catalog policy sets `"freeform"` for first-party GPT-5 Responses