@gajae-code/ai 0.4.5 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/CHANGELOG.md +25 -1
  2. package/dist/types/index.d.ts +1 -0
  3. package/dist/types/providers/amazon-bedrock.d.ts +29 -5
  4. package/dist/types/providers/composer-discipline.d.ts +27 -0
  5. package/dist/types/providers/cursor.d.ts +1 -1
  6. package/dist/types/providers/google-gemini-cli.d.ts +1 -1
  7. package/dist/types/providers/google-shared.d.ts +11 -1
  8. package/dist/types/providers/ollama.d.ts +36 -1
  9. package/dist/types/providers/openai-completions-compat.d.ts +3 -1
  10. package/dist/types/providers/register-builtins.d.ts +3 -3
  11. package/dist/types/types.d.ts +25 -3
  12. package/dist/types/utils/event-stream.d.ts +6 -1
  13. package/dist/types/utils/tool-choice-capability.d.ts +41 -0
  14. package/package.json +2 -2
  15. package/src/index.ts +1 -0
  16. package/src/model-thinking.ts +9 -0
  17. package/src/models.json +92 -0
  18. package/src/models.ts +33 -7
  19. package/src/provider-models/openai-compat.ts +9 -1
  20. package/src/providers/amazon-bedrock.ts +145 -60
  21. package/src/providers/anthropic.ts +85 -32
  22. package/src/providers/azure-openai-responses.ts +44 -3
  23. package/src/providers/composer-discipline.ts +38 -0
  24. package/src/providers/cursor.ts +10 -3
  25. package/src/providers/google-gemini-cli.ts +69 -10
  26. package/src/providers/google-shared.ts +61 -12
  27. package/src/providers/ollama.ts +60 -4
  28. package/src/providers/openai-codex-responses.ts +151 -2
  29. package/src/providers/openai-completions-compat.ts +9 -1
  30. package/src/providers/openai-completions.ts +46 -6
  31. package/src/providers/openai-request-transform.ts +1 -0
  32. package/src/providers/openai-responses.ts +54 -5
  33. package/src/providers/register-builtins.ts +5 -6
  34. package/src/types.ts +37 -3
  35. package/src/utils/event-stream.ts +35 -5
  36. package/src/utils/tool-choice-capability.ts +220 -0
@@ -25,6 +25,7 @@ import type {
25
25
  ThinkingContent,
26
26
  Tool,
27
27
  ToolCall,
28
+ ToolChoice,
28
29
  ToolResultMessage,
29
30
  } from "../types";
30
31
  import { normalizeToolCallId, resolveCacheRetention } from "../utils";
@@ -33,6 +34,11 @@ import { appendRawHttpRequestDumpFor400, type RawHttpRequestDump, withHttpStatus
33
34
  import { parseStreamingJson } from "../utils/json-parse";
34
35
  import { resolveRetryBudget } from "../utils/retry-budget";
35
36
  import { toolWireSchema } from "../utils/schema/wire";
37
+ import {
38
+ isForcedToolChoiceUnsupportedError,
39
+ markToolChoiceIncapability,
40
+ resolveToolChoice,
41
+ } from "../utils/tool-choice-capability";
36
42
  import { resolveAwsCredentials } from "./aws-credentials";
37
43
  import { decodeEventStream } from "./aws-eventstream";
38
44
  import { signRequest } from "./aws-sigv4";
@@ -43,7 +49,7 @@ export type BedrockThinkingDisplay = "summarized" | "omitted";
43
49
  export interface BedrockOptions extends StreamOptions {
44
50
  region?: string;
45
51
  profile?: string;
46
- toolChoice?: "auto" | "any" | "none" | { type: "tool"; name: string };
52
+ toolChoice?: ToolChoice;
47
53
  /* See https://docs.aws.amazon.com/bedrock/latest/userguide/inference-reasoning.html for supported models. */
48
54
  reasoning?: Effort;
49
55
  /* Custom token budgets per thinking level. Overrides default budgets. */
@@ -191,11 +197,12 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
191
197
 
192
198
  try {
193
199
  const cacheRetention = resolveCacheRetention(options.cacheRetention);
194
- const toolConfig = convertToolConfig(context.tools, options.toolChoice);
200
+ const resolvedToolChoice = resolveToolChoice(model, options.toolChoice);
201
+ const toolConfig = convertToolConfig(context.tools, resolvedToolChoice.resolvedChoice);
195
202
  let additionalModelRequestFields = buildAdditionalModelRequestFields(model, options);
196
203
 
197
204
  // Bedrock rejects thinking + forced tool_choice ("any" or specific tool).
198
- // When tool_choice forces tool use, disable thinking to avoid API errors.
205
+ // When the resolved tool_choice forces tool use, disable thinking to avoid API errors.
199
206
  if (toolConfig?.toolChoice && additionalModelRequestFields) {
200
207
  const tc = toolConfig.toolChoice;
201
208
  if (tc.any || tc.tool) additionalModelRequestFields = undefined;
@@ -254,8 +261,45 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
254
261
  headers: baseHeaders,
255
262
  });
256
263
  const requestHeaders: Record<string, string> = { ...baseHeaders, ...signed };
264
+ const sentForcedToolChoice = Boolean(toolConfig?.toolChoice?.any || toolConfig?.toolChoice?.tool);
265
+ let fallbackRan = false;
266
+ const retryWithoutForcedToolChoice = async (reason: string) => {
267
+ fallbackRan = true;
268
+ markToolChoiceIncapability(model, "auto", reason);
269
+ stream.push({
270
+ type: "toolChoiceIncapability",
271
+ api: output.api,
272
+ provider: model.provider,
273
+ model: model.id,
274
+ requestedLevel: resolvedToolChoice.requestedLevel,
275
+ resolvedLevel: "auto",
276
+ reason,
277
+ registryKey: resolvedToolChoice.registryKey,
278
+ });
279
+ stripBedrockForcedToolChoiceForRetry(commandInput);
280
+ const retryBodyText = JSON.stringify(commandInput);
281
+ const retryBody = new TextEncoder().encode(retryBodyText);
282
+ const retrySigned = await signRequest({
283
+ method: "POST",
284
+ host,
285
+ path: urlPath,
286
+ body: retryBody,
287
+ region,
288
+ service: "bedrock",
289
+ credentials,
290
+ headers: baseHeaders,
291
+ });
292
+ if (rawRequestDump) rawRequestDump.body = commandInput;
293
+ return fetchWithRetry(url, {
294
+ method: "POST",
295
+ headers: { ...baseHeaders, ...retrySigned },
296
+ body: retryBody,
297
+ signal: options.signal,
298
+ maxAttempts: 1,
299
+ });
300
+ };
257
301
 
258
- const response = await fetchWithRetry(url, {
302
+ let response = await fetchWithRetry(url, {
259
303
  method: "POST",
260
304
  headers: requestHeaders,
261
305
  body,
@@ -263,6 +307,18 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
263
307
  maxAttempts: resolveRetryBudget(options.requestMaxRetries, 4) + 1,
264
308
  });
265
309
 
310
+ if (!response.ok && sentForcedToolChoice) {
311
+ const errBody = await response.text().catch(() => "");
312
+ const error = withHttpStatus(
313
+ new Error(`Bedrock HTTP ${response.status}: ${errBody.slice(0, 1000)}`),
314
+ response.status,
315
+ );
316
+ if (firstTokenTime === undefined && !fallbackRan && isForcedToolChoiceUnsupportedError(error, true)) {
317
+ response = await retryWithoutForcedToolChoice(error.message);
318
+ } else {
319
+ throw error;
320
+ }
321
+ }
266
322
  if (!response.ok) {
267
323
  const errBody = await response.text().catch(() => "");
268
324
  throw withHttpStatus(
@@ -273,65 +329,86 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
273
329
  if (!response.body) throw new Error("Bedrock response has no body");
274
330
 
275
331
  // Track first event for the abort/diagnostic path (currently informational).
276
- for await (const message of decodeEventStream(response.body)) {
277
- const messageType = message.headers[":message-type"];
278
- const eventType = message.headers[":event-type"];
279
-
280
- if (messageType === "exception") {
281
- const exceptionType = message.headers[":exception-type"] || "Exception";
282
- const payload = safeParsePayload(message.payload) as { message?: string } | undefined;
283
- const errorMessage = payload?.message || new TextDecoder().decode(message.payload);
284
- const status = exceptionType === "validationException" ? 400 : 0;
285
- const err = new Error(`${exceptionType}: ${errorMessage}`);
286
- throw status ? withHttpStatus(err, status) : err;
287
- }
288
- if (messageType === "error") {
289
- const code = message.headers[":error-code"] || "UnknownError";
290
- const errorMessage = message.headers[":error-message"] || new TextDecoder().decode(message.payload);
291
- throw new Error(`${code}: ${errorMessage}`);
292
- }
293
- if (messageType !== "event") continue;
294
-
295
- const payload = safeParsePayload(message.payload);
296
- if (!payload) continue;
297
-
298
- switch (eventType) {
299
- case "messageStart": {
300
- // no-op: first event marker is implicit by stream entry.
301
- const ev = payload as MessageStartEvent;
302
- if (ev.role !== "assistant") {
303
- throw new Error("Unexpected assistant message start but got user message start instead");
332
+ streamLoop: while (true) {
333
+ for await (const message of decodeEventStream(response.body)) {
334
+ const messageType = message.headers[":message-type"];
335
+ const eventType = message.headers[":event-type"];
336
+
337
+ if (messageType === "exception") {
338
+ const exceptionType = message.headers[":exception-type"] || "Exception";
339
+ const payload = safeParsePayload(message.payload) as { message?: string } | undefined;
340
+ const errorMessage = payload?.message || new TextDecoder().decode(message.payload);
341
+ const status = exceptionType === "validationException" ? 400 : 0;
342
+ const err = new Error(`${exceptionType}: ${errorMessage}`);
343
+ const error = status ? withHttpStatus(err, status) : err;
344
+ if (
345
+ firstTokenTime === undefined &&
346
+ sentForcedToolChoice &&
347
+ !fallbackRan &&
348
+ isForcedToolChoiceUnsupportedError(error, true)
349
+ ) {
350
+ response = await retryWithoutForcedToolChoice(error.message);
351
+ if (!response.ok) {
352
+ const errBody = await response.text().catch(() => "");
353
+ throw withHttpStatus(
354
+ new Error(`Bedrock HTTP ${response.status}: ${errBody.slice(0, 1000)}`),
355
+ response.status,
356
+ );
357
+ }
358
+ if (!response.body) throw new Error("Bedrock response has no body");
359
+ continue streamLoop;
304
360
  }
305
- stream.push({ type: "start", partial: output });
306
- break;
307
- }
308
- case "contentBlockStart": {
309
- if (!firstTokenTime) firstTokenTime = Date.now();
310
- handleContentBlockStart(payload as ContentBlockStartEvent, blocks, output, stream);
311
- break;
312
- }
313
- case "contentBlockDelta": {
314
- if (!firstTokenTime) firstTokenTime = Date.now();
315
- handleContentBlockDelta(payload as ContentBlockDeltaEvent, blocks, output, stream);
316
- break;
317
- }
318
- case "contentBlockStop": {
319
- handleContentBlockStop(payload as ContentBlockStopEvent, blocks, output, stream);
320
- break;
361
+ throw error;
321
362
  }
322
- case "messageStop": {
323
- const ev = payload as MessageStopEvent;
324
- output.stopReason = mapStopReason(ev.stopReason);
325
- break;
363
+ if (messageType === "error") {
364
+ const code = message.headers[":error-code"] || "UnknownError";
365
+ const errorMessage = message.headers[":error-message"] || new TextDecoder().decode(message.payload);
366
+ throw new Error(`${code}: ${errorMessage}`);
326
367
  }
327
- case "metadata": {
328
- handleMetadata(payload as MetadataEvent, model, output);
329
- break;
368
+ if (messageType !== "event") continue;
369
+
370
+ const payload = safeParsePayload(message.payload);
371
+ if (!payload) continue;
372
+
373
+ switch (eventType) {
374
+ case "messageStart": {
375
+ // no-op: first event marker is implicit by stream entry.
376
+ const ev = payload as MessageStartEvent;
377
+ if (ev.role !== "assistant") {
378
+ throw new Error("Unexpected assistant message start but got user message start instead");
379
+ }
380
+ stream.push({ type: "start", partial: output });
381
+ break;
382
+ }
383
+ case "contentBlockStart": {
384
+ if (!firstTokenTime) firstTokenTime = Date.now();
385
+ handleContentBlockStart(payload as ContentBlockStartEvent, blocks, output, stream);
386
+ break;
387
+ }
388
+ case "contentBlockDelta": {
389
+ if (!firstTokenTime) firstTokenTime = Date.now();
390
+ handleContentBlockDelta(payload as ContentBlockDeltaEvent, blocks, output, stream);
391
+ break;
392
+ }
393
+ case "contentBlockStop": {
394
+ handleContentBlockStop(payload as ContentBlockStopEvent, blocks, output, stream);
395
+ break;
396
+ }
397
+ case "messageStop": {
398
+ const ev = payload as MessageStopEvent;
399
+ output.stopReason = mapStopReason(ev.stopReason);
400
+ break;
401
+ }
402
+ case "metadata": {
403
+ handleMetadata(payload as MetadataEvent, model, output);
404
+ break;
405
+ }
406
+ default:
407
+ // Unknown event types (Bedrock may add new ones) — ignore.
408
+ break;
330
409
  }
331
- default:
332
- // Unknown event types (Bedrock may add new ones) — ignore.
333
- break;
334
410
  }
411
+ break;
335
412
  }
336
413
 
337
414
  if (options.signal?.aborted) throw new Error("Request was aborted");
@@ -714,7 +791,14 @@ function convertMessages(
714
791
  return result;
715
792
  }
716
793
 
717
- function convertToolConfig(
794
+ export function stripBedrockForcedToolChoiceForRetry<T extends { toolConfig?: { toolChoice?: unknown } }>(body: T): T {
795
+ if (body.toolConfig) {
796
+ body.toolConfig = { ...body.toolConfig, toolChoice: undefined };
797
+ }
798
+ return body;
799
+ }
800
+
801
+ export function convertToolConfig(
718
802
  tools: Tool[] | undefined,
719
803
  toolChoice: BedrockOptions["toolChoice"],
720
804
  ): WireToolConfig | undefined {
@@ -734,6 +818,7 @@ function convertToolConfig(
734
818
  bedrockToolChoice = { auto: {} };
735
819
  break;
736
820
  case "any":
821
+ case "required":
737
822
  bedrockToolChoice = { any: {} };
738
823
  break;
739
824
  default:
@@ -742,7 +827,7 @@ function convertToolConfig(
742
827
  }
743
828
  }
744
829
 
745
- return { tools: bedrockTools, toolChoice: bedrockToolChoice };
830
+ return bedrockToolChoice ? { tools: bedrockTools, toolChoice: bedrockToolChoice } : { tools: bedrockTools };
746
831
  }
747
832
 
748
833
  function mapStopReason(reason: string | undefined): StopReason {
@@ -65,6 +65,12 @@ import { resolveRetryBudget } from "../utils/retry-budget";
65
65
  import { COMBINATOR_KEYS, NO_STRICT, toolWireSchema } from "../utils/schema";
66
66
  import { spillToDescription } from "../utils/schema/spill";
67
67
  import { notifyRawSseEvent, wrapFetchForSseDebug } from "../utils/sse-debug";
68
+ import {
69
+ isForcedToolChoiceUnsupportedError,
70
+ markToolChoiceIncapability,
71
+ type ResolveToolChoiceResult,
72
+ resolveToolChoice,
73
+ } from "../utils/tool-choice-capability";
68
74
  import {
69
75
  buildCopilotDynamicHeaders,
70
76
  hasCopilotVisionInput,
@@ -875,7 +881,8 @@ async function getAnthropicStreamResponse(
875
881
 
876
882
  function getAnthropicCompat(
877
883
  model: Model<"anthropic-messages">,
878
- ): Required<NonNullable<Model<"anthropic-messages">["compat"]>> {
884
+ ): Required<Omit<NonNullable<Model<"anthropic-messages">["compat"]>, "toolChoiceSupport">> &
885
+ Pick<NonNullable<Model<"anthropic-messages">["compat"]>, "toolChoiceSupport"> {
879
886
  return {
880
887
  disableStrictTools: model.compat?.disableStrictTools ?? false,
881
888
  disableAdaptiveThinking: model.compat?.disableAdaptiveThinking ?? false,
@@ -883,6 +890,7 @@ function getAnthropicCompat(
883
890
  supportsLongCacheRetention: model.compat?.supportsLongCacheRetention ?? true,
884
891
  supportsToolChoice: model.compat?.supportsToolChoice ?? true,
885
892
  supportsForcedToolChoice: model.compat?.supportsForcedToolChoice ?? true,
893
+ toolChoiceSupport: model.compat?.toolChoiceSupport,
886
894
  };
887
895
  }
888
896
 
@@ -1076,8 +1084,10 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1076
1084
  (providerSessionState?.strictToolsDisabled ?? false) || (model.compat?.disableStrictTools ?? false);
1077
1085
  let strictFallbackErrorMessage: string | undefined;
1078
1086
  let dropFastMode = providerSessionState?.fastModeDisabled ?? false;
1087
+ let droppedForcedToolChoice = false;
1079
1088
  const prepareParams = async (paramsOptions?: {
1080
1089
  repairLatestAssistantThinking?: boolean;
1090
+ dropForcedToolChoice?: boolean;
1081
1091
  }): Promise<MessageCreateParamsStreaming> => {
1082
1092
  let nextParams = buildParams(
1083
1093
  model,
@@ -1088,6 +1098,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1088
1098
  disableStrictTools,
1089
1099
  paramsOptions?.repairLatestAssistantThinking === true,
1090
1100
  );
1101
+ if (paramsOptions?.dropForcedToolChoice === true) {
1102
+ delete nextParams.tool_choice;
1103
+ }
1091
1104
  if (disableStrictTools) {
1092
1105
  dropAnthropicStrictTools(nextParams);
1093
1106
  }
@@ -1401,6 +1414,39 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1401
1414
  firstTokenTime = undefined;
1402
1415
  continue;
1403
1416
  }
1417
+ if (
1418
+ !droppedForcedToolChoice &&
1419
+ firstTokenTime === undefined &&
1420
+ isSentForcedAnthropicToolChoice(params.tool_choice) &&
1421
+ isForcedToolChoiceUnsupportedError(streamFailure, true)
1422
+ ) {
1423
+ const message = await finalizeErrorMessage(streamFailure, rawRequestDump);
1424
+ logger.debug("anthropic: forced tool_choice unsupported, retrying with auto tool choice", {
1425
+ model: model.id,
1426
+ error: message,
1427
+ });
1428
+ markToolChoiceIncapability(model, "auto", message);
1429
+ stream.push({
1430
+ type: "toolChoiceIncapability",
1431
+ api: output.api,
1432
+ provider: model.provider,
1433
+ model: model.id,
1434
+ requestedLevel: resolveToolChoice(model, options?.toolChoice).requestedLevel,
1435
+ resolvedLevel: "auto",
1436
+ reason: message,
1437
+ registryKey: resolveToolChoice(model, options?.toolChoice).registryKey,
1438
+ });
1439
+ droppedForcedToolChoice = true;
1440
+ params = await prepareParams({ dropForcedToolChoice: true });
1441
+ providerRetryAttempt = 0;
1442
+ output.content.length = 0;
1443
+ output.responseId = undefined;
1444
+ output.providerPayload = undefined;
1445
+ output.usage = createEmptyUsage(copilotDynamicHeaders?.premiumRequests);
1446
+ output.stopReason = "stop";
1447
+ firstTokenTime = undefined;
1448
+ continue;
1449
+ }
1404
1450
  if (
1405
1451
  !thinkingRepairAttempted &&
1406
1452
  firstTokenTime === undefined &&
@@ -1684,28 +1730,29 @@ function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming)
1684
1730
  }
1685
1731
  }
1686
1732
 
1687
- function isForcedAnthropicToolChoice(toolChoice: NonNullable<AnthropicOptions["toolChoice"]>): boolean {
1688
- if (typeof toolChoice === "string") return toolChoice === "any";
1689
- if ("function" in toolChoice) return true;
1690
- if ("name" in toolChoice && typeof toolChoice.name === "string") return true;
1691
- return toolChoice.type === "tool" || toolChoice.type === "function";
1692
- }
1693
-
1694
- function supportsForcedAnthropicToolChoice(model: Model<"anthropic-messages">): boolean {
1695
- const compat = model.compat;
1696
- if (compat?.supportsToolChoice === false || compat?.supportsForcedToolChoice === false) return false;
1697
-
1698
- // Claude Mythos Preview is documented as accepting tools while rejecting
1699
- // forced tool use; Claude Fable currently returns the same Anthropic 400.
1700
- return !/^claude-(?:fable|mythos)(?:-|$)/i.test(model.id);
1733
+ function mapAnthropicToolChoice(
1734
+ toolChoice: NonNullable<ResolveToolChoiceResult["resolvedChoice"]>,
1735
+ isOAuthToken: boolean,
1736
+ ): NonNullable<MessageCreateParamsStreaming["tool_choice"]> | undefined {
1737
+ if (typeof toolChoice === "string") {
1738
+ if (toolChoice === "required") return { type: "any" };
1739
+ return { type: toolChoice };
1740
+ }
1741
+ if ("function" in toolChoice) {
1742
+ const name = typeof toolChoice.function === "string" ? toolChoice.function : toolChoice.function.name;
1743
+ return { type: "tool", name: isOAuthToken ? applyClaudeToolPrefix(name) : name };
1744
+ }
1745
+ if ("name" in toolChoice && typeof toolChoice.name === "string") {
1746
+ return {
1747
+ ...toolChoice,
1748
+ type: "tool",
1749
+ name: isOAuthToken ? applyClaudeToolPrefix(toolChoice.name) : toolChoice.name,
1750
+ };
1751
+ }
1752
+ return toolChoice as NonNullable<MessageCreateParamsStreaming["tool_choice"]>;
1701
1753
  }
1702
-
1703
- function shouldSendAnthropicToolChoice(
1704
- model: Model<"anthropic-messages">,
1705
- toolChoice: NonNullable<AnthropicOptions["toolChoice"]>,
1706
- ): boolean {
1707
- if (model.compat?.supportsToolChoice === false) return false;
1708
- return !isForcedAnthropicToolChoice(toolChoice) || supportsForcedAnthropicToolChoice(model);
1754
+ function isSentForcedAnthropicToolChoice(toolChoice: MessageCreateParamsStreaming["tool_choice"] | undefined): boolean {
1755
+ return toolChoice?.type === "any" || toolChoice?.type === "tool";
1709
1756
  }
1710
1757
 
1711
1758
  function ensureMaxTokensForThinking(params: MessageCreateParamsStreaming, model: Model<"anthropic-messages">): void {
@@ -2055,16 +2102,22 @@ function buildParams(
2055
2102
  (params as ParamsWithSpeed).speed = "fast";
2056
2103
  }
2057
2104
 
2058
- if (options?.toolChoice && shouldSendAnthropicToolChoice(model, options.toolChoice)) {
2059
- if (typeof options.toolChoice === "string") {
2060
- params.tool_choice = { type: options.toolChoice };
2061
- } else if (isOAuthToken && options.toolChoice.name) {
2062
- params.tool_choice = {
2063
- ...options.toolChoice,
2064
- name: applyClaudeToolPrefix(options.toolChoice.name),
2065
- };
2066
- } else {
2067
- params.tool_choice = options.toolChoice;
2105
+ if (options?.toolChoice) {
2106
+ const resolution = resolveToolChoice(model, options.toolChoice);
2107
+ if (resolution.degraded && resolution.supportSource !== "runtime") {
2108
+ logger.debug("anthropic: degrading tool_choice for model capability", {
2109
+ model: model.id,
2110
+ requestedLevel: resolution.requestedLevel,
2111
+ resolvedLevel: resolution.resolvedLevel,
2112
+ reason: resolution.reason,
2113
+ supportSource: resolution.supportSource,
2114
+ });
2115
+ }
2116
+ if (resolution.resolvedChoice) {
2117
+ const mappedToolChoice = mapAnthropicToolChoice(resolution.resolvedChoice, isOAuthToken);
2118
+ if (mappedToolChoice) {
2119
+ params.tool_choice = mappedToolChoice;
2120
+ }
2068
2121
  }
2069
2122
  }
2070
2123
 
@@ -1,4 +1,4 @@
1
- import { $env, extractHttpStatusFromError } from "@gajae-code/utils";
1
+ import { $env, extractHttpStatusFromError, logger } from "@gajae-code/utils";
2
2
  import { AzureOpenAI } from "openai";
3
3
  import type {
4
4
  Tool as OpenAITool,
@@ -30,6 +30,11 @@ import { resolveRetryBudget } from "../utils/retry-budget";
30
30
  import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
31
31
  import { wrapFetchForSseDebug } from "../utils/sse-debug";
32
32
  import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice";
33
+ import {
34
+ isForcedToolChoiceUnsupportedError,
35
+ markToolChoiceIncapability,
36
+ resolveToolChoice,
37
+ } from "../utils/tool-choice-capability";
33
38
  import { normalizeOpenAIResponsesPromptCacheKey, supportsDeveloperRole } from "./openai-responses";
34
39
  import {
35
40
  appendResponsesToolResultMessages,
@@ -130,7 +135,30 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
130
135
  url: `${baseUrl}/responses`,
131
136
  body: params,
132
137
  };
133
- const openaiStream = await client.responses.create(params, { signal: requestSignal });
138
+ let openaiStream: Awaited<ReturnType<typeof client.responses.create>>;
139
+ try {
140
+ openaiStream = await client.responses.create(params, { signal: requestSignal });
141
+ } catch (error) {
142
+ if (!isForcedToolChoiceUnsupportedError(error, isForcedAzureResponsesToolChoice(params.tool_choice))) {
143
+ throw error;
144
+ }
145
+ const reason = await finalizeErrorMessage(error, rawRequestDump);
146
+ markToolChoiceIncapability(model, "auto", reason);
147
+ const resolvedToolChoice = resolveToolChoice(model, options?.toolChoice);
148
+ stream.push({
149
+ type: "toolChoiceIncapability",
150
+ api: model.api,
151
+ provider: model.provider,
152
+ model: model.id,
153
+ requestedLevel: resolvedToolChoice.requestedLevel,
154
+ resolvedLevel: "auto",
155
+ reason,
156
+ registryKey: resolvedToolChoice.registryKey,
157
+ });
158
+ delete params.tool_choice;
159
+ rawRequestDump = { ...rawRequestDump, body: params };
160
+ openaiStream = await client.responses.create(params, { signal: requestSignal });
161
+ }
134
162
  const firstEventWatchdog = createWatchdog(
135
163
  options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs),
136
164
  () => abortTracker.abortLocally(firstEventTimeoutAbortError),
@@ -278,7 +306,16 @@ function buildParams(
278
306
  if (context.tools) {
279
307
  params.tools = convertTools(context.tools);
280
308
  if (options?.toolChoice) {
281
- params.tool_choice = mapToOpenAIResponsesToolChoice(options.toolChoice);
309
+ const toolChoice = resolveToolChoice(model, options.toolChoice);
310
+ if (toolChoice.degraded && toolChoice.supportSource === "runtime") {
311
+ logger.debug("azure-openai-responses: degraded tool_choice after runtime capability discovery", {
312
+ model: model.id,
313
+ requestedLevel: toolChoice.requestedLevel,
314
+ resolvedLevel: toolChoice.resolvedLevel,
315
+ reason: toolChoice.reason,
316
+ });
317
+ }
318
+ params.tool_choice = mapToOpenAIResponsesToolChoice(toolChoice.resolvedChoice);
282
319
  }
283
320
  }
284
321
 
@@ -287,6 +324,10 @@ function buildParams(
287
324
  return params;
288
325
  }
289
326
 
327
+ function isForcedAzureResponsesToolChoice(choice: AzureOpenAIResponsesSamplingParams["tool_choice"]): boolean {
328
+ return !!choice && choice !== "none" && choice !== "auto";
329
+ }
330
+
290
331
  function convertMessages(
291
332
  model: Model<"azure-openai-responses">,
292
333
  context: Context,
@@ -0,0 +1,38 @@
1
+ /**
2
+ * Anchor/edit discipline for composer-harness models (xai grok-composer-*,
3
+ * cursor composer-*).
4
+ *
5
+ * Composer models are trained on a proprietary coding-agent harness
6
+ * (Cursor / Grok Build) and carry habits that break this agent's hashline
7
+ * edit workflow when driven through a generic provider. Observed in live
8
+ * sessions with grok-composer-2.5-fast:
9
+ *
10
+ * - they print files with shell commands (`sed -n`, `cat`, `grep -n`) or
11
+ * python heredocs whose output carries NO line anchors, then FABRICATE the
12
+ * 2-char anchor hash the edit tool requires (e.g. guessed "617hp" where
13
+ * the file had "617ca" → "Edit rejected: N anchors do not match");
14
+ * - they mutate files out-of-band via python heredocs (pathlib write_text /
15
+ * str.replace), which invalidates every previously seen anchor and defeats
16
+ * the read-cache snapshot that powers stale-anchor recovery;
17
+ * - they arithmetically renumber anchors after their own edits instead of
18
+ * copying them from the latest tool output;
19
+ * - they leak reasoning prose into heredoc bodies, producing shell/python
20
+ * syntax errors.
21
+ *
22
+ * This prompt is the per-request countermeasure, pinned ahead of the host
23
+ * system prompt on both the openai-completions path and the cursor RPC path.
24
+ */
25
+
26
+ /** Matches composer-harness model ids on any provider (xai grok-composer-*, cursor composer-*). */
27
+ export function isComposerHarnessModel(modelId: string): boolean {
28
+ return modelId.toLowerCase().includes("composer");
29
+ }
30
+
31
+ export const COMPOSER_EDIT_DISCIPLINE_PROMPT = `File-editing discipline for this harness (this OVERRIDES contrary habits from your training):
32
+
33
+ - Read file contents ONLY with the provided read/search tools. NEVER print files through shell commands (sed, cat, awk, head, grep) or scripts — that output carries no line anchors, and the edit tool accepts ONLY anchors.
34
+ - Modify files ONLY with the provided edit/write tools. NEVER mutate files through shell redirection, sed -i, or inline python scripts — out-of-band writes invalidate every known anchor and break edit recovery.
35
+ - A line anchor (e.g. "42sr") is a line number plus a 2-char content hash. You CANNOT compute the hash yourself: copy anchors verbatim from the MOST RECENT read/search/edit output of that exact file. NEVER guess, renumber, or arithmetically shift an anchor.
36
+ - After ANY edit to a file (including your own), anchors you saw earlier are stale. Re-read the edited region, or copy the fresh anchors printed in the edit result, before issuing the next edit.
37
+ - If an edit is rejected with "anchors do not match", the rejection message prints the current lines WITH fresh anchors. Retry using exactly those printed anchors.
38
+ - A shell command string must contain only the command itself. NEVER interleave reasoning or commentary into command strings or heredocs.`;
@@ -30,6 +30,7 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
30
30
  import { parseStreamingJson } from "../utils/json-parse";
31
31
  import { formatErrorMessageWithRetryAfter } from "../utils/retry-after";
32
32
  import { toolWireSchema } from "../utils/schema/wire";
33
+ import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
33
34
  import type { McpToolDefinition } from "./cursor/gen/agent_pb";
34
35
  import {
35
36
  AgentClientMessageSchema,
@@ -2284,12 +2285,18 @@ function findLastUserMessageIndex(messages: Message[]): number {
2284
2285
  * When no system prompts are provided, returns a single default greeting so we never emit
2285
2286
  * an empty `rootPromptMessagesJson` head.
2286
2287
  */
2287
- export function buildCursorSystemPromptJsons(systemPrompt: readonly string[] | undefined): string[] {
2288
+ export function buildCursorSystemPromptJsons(systemPrompt: readonly string[] | undefined, modelId?: string): string[] {
2288
2289
  const systemPrompts = normalizeSystemPrompts(systemPrompt);
2289
2290
  if (systemPrompts.length === 0) {
2290
2291
  return [JSON.stringify({ role: "system", content: "You are a helpful assistant." })];
2291
2292
  }
2292
- return systemPrompts.map(content => JSON.stringify({ role: "system", content }));
2293
+ const jsons = systemPrompts.map(content => JSON.stringify({ role: "system", content }));
2294
+ // Composer-harness models need anchor/edit discipline pinned ahead of the
2295
+ // host prompt (see composer-discipline.ts for the observed failure modes).
2296
+ if (modelId !== undefined && isComposerHarnessModel(modelId)) {
2297
+ jsons.unshift(JSON.stringify({ role: "system", content: COMPOSER_EDIT_DISCIPLINE_PROMPT }));
2298
+ }
2299
+ return jsons;
2293
2300
  }
2294
2301
 
2295
2302
  function buildRootPromptMessagesJson(
@@ -2501,7 +2508,7 @@ function buildGrpcRequest(
2501
2508
  } {
2502
2509
  const blobStore = state.blobStore;
2503
2510
 
2504
- const systemPromptIds = buildCursorSystemPromptJsons(context.systemPrompt).map(json =>
2511
+ const systemPromptIds = buildCursorSystemPromptJsons(context.systemPrompt, model.id).map(json =>
2505
2512
  storeCursorBlob(blobStore, new TextEncoder().encode(json)),
2506
2513
  );
2507
2514