@bitkyc08/opencodex 2.46.0 → 2.47.0-preview.20260908

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/README.md +22 -0
  2. package/SPONSORS.md +105 -0
  3. package/gui/dist/assets/index-B8ZZ0ivI.css +1 -0
  4. package/gui/dist/assets/index-CNTWhC5F.js +115 -0
  5. package/gui/dist/index.html +2 -2
  6. package/package.json +2 -1
  7. package/src/adapters/cursor/protobuf-request.ts +34 -21
  8. package/src/adapters/cursor/tool-guidance.ts +2 -2
  9. package/src/adapters/cursor/tool-result-normalize.ts +22 -2
  10. package/src/adapters/exec-tool-result-normalize.ts +87 -0
  11. package/src/adapters/kiro.ts +29 -18
  12. package/src/adapters/responses-code-mode.ts +14 -4
  13. package/src/adapters/tool-catalog-nudge.ts +2 -2
  14. package/src/bridge.ts +40 -24
  15. package/src/claude/inbound.ts +21 -6
  16. package/src/claude/outbound.ts +43 -26
  17. package/src/cli/capabilities.ts +27 -0
  18. package/src/cli/integrations.ts +1 -1
  19. package/src/cli/models-runtime-subcommands.ts +2 -0
  20. package/src/cli/models-runtime.ts +108 -1
  21. package/src/cli/observe.ts +18 -3
  22. package/src/cli/usage-report.ts +6 -1
  23. package/src/clients/config-export.ts +5 -3
  24. package/src/codex/auth-api.ts +36 -0
  25. package/src/codex/catalog/provider-fetch.ts +2 -1
  26. package/src/codex/quota-auto-refresh-state.ts +8 -0
  27. package/src/codex/quota-auto-refresh.ts +157 -27
  28. package/src/codex/warmup.ts +5 -0
  29. package/src/config/provider-validation.ts +70 -1
  30. package/src/config.ts +73 -1
  31. package/src/generated/compatibility-version.json +67 -55
  32. package/src/lib/json-byte-size.ts +61 -0
  33. package/src/lib/provider-outbound.ts +11 -10
  34. package/src/oauth/index.ts +42 -8
  35. package/src/oauth/orcarouter.ts +200 -0
  36. package/src/providers/key-store.ts +22 -0
  37. package/src/providers/registry.ts +78 -29
  38. package/src/responses/citation-markers.ts +56 -24
  39. package/src/responses/reasoning-envelope.ts +53 -19
  40. package/src/server/auth-cors.ts +8 -0
  41. package/src/server/chat-completions.ts +7 -0
  42. package/src/server/chat-native.ts +65 -3
  43. package/src/server/claude-messages.ts +30 -10
  44. package/src/server/effort-policy.ts +183 -1
  45. package/src/server/management/agent-settings-routes.ts +50 -9
  46. package/src/server/management/config-routes.ts +2 -0
  47. package/src/server/management/logs-usage-routes.ts +11 -3
  48. package/src/server/management/model-routes.ts +65 -1
  49. package/src/server/management/model-rows.ts +5 -0
  50. package/src/server/management/provider-routes.ts +99 -6
  51. package/src/server/management/route-registry.ts +2 -0
  52. package/src/server/management/usage-aggregate-cache.ts +14 -7
  53. package/src/server/responses/compact.ts +3 -2
  54. package/src/server/responses/core.ts +16 -2
  55. package/src/server/startup-health-cache.ts +31 -9
  56. package/src/types/config.ts +5 -0
  57. package/src/types/provider.ts +4 -0
  58. package/src/usage/cost.ts +27 -31
  59. package/src/usage/summary.ts +52 -9
  60. package/src/usage/time-range.ts +48 -0
  61. package/src/usage/user-cost-overlays.ts +35 -3
  62. package/src/vision/anthropic-describe.ts +46 -2
  63. package/src/web-search/anthropic-executor.ts +49 -3
  64. package/src/web-search/parse.ts +1 -1
  65. package/gui/dist/assets/index-B72RPblC.js +0 -115
  66. package/gui/dist/assets/index-BFgUC17B.css +0 -1
@@ -16,8 +16,8 @@
16
16
  } catch (e) {}
17
17
  })();
18
18
  </script>
19
- <script type="module" crossorigin src="/assets/index-B72RPblC.js"></script>
20
- <link rel="stylesheet" crossorigin href="/assets/index-BFgUC17B.css">
19
+ <script type="module" crossorigin src="/assets/index-CNTWhC5F.js"></script>
20
+ <link rel="stylesheet" crossorigin href="/assets/index-B8ZZ0ivI.css">
21
21
  </head>
22
22
  <body>
23
23
  <div id="root"></div>
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@bitkyc08/opencodex",
3
- "version": "2.46.0",
3
+ "version": "2.47.0-preview.20260908",
4
4
  "description": "Universal provider proxy for OpenAI Codex & Claude Code — use any LLM with Codex CLI/App/SDK and Claude Code",
5
5
  "type": "module",
6
6
  "main": "./bin/package-main.mjs",
@@ -24,6 +24,7 @@
24
24
  "assets/claude-code-models.gif",
25
25
  "assets/codex-app-picker.png",
26
26
  "README.md",
27
+ "SPONSORS.md",
27
28
  "AGENTS_INSTALL.md",
28
29
  "LICENSE"
29
30
  ],
@@ -58,6 +58,7 @@ import {
58
58
  buildCursorToolDefinitions,
59
59
  cursorToolWireName,
60
60
  cursorRequestHasShellAlias,
61
+ cursorRequestUsesCodeMode,
61
62
  CURSOR_SHELL_ALIAS_SYSTEM_NOTE,
62
63
  OCX_RESPONSES_TOOL_PROVIDER,
63
64
  } from "./tool-definitions";
@@ -222,6 +223,7 @@ function assistantRootText(
222
223
  function rootPromptMessages(
223
224
  request: CursorRunRequest,
224
225
  requestScope: CursorBlobRequestScopeToken,
226
+ codeMode: boolean,
225
227
  /**
226
228
  * Calls indexed from the FULL history. The checkpoint path replays only a suffix of
227
229
  * `rawMessages`, so a result in that suffix can have its originating call before the cut; indexing
@@ -389,10 +391,10 @@ function rootPromptMessages(
389
391
  if (!echoToolResultInRoot) continue;
390
392
  // #1920: the prefix must reflect the NORMALIZED error state (an empty
391
393
  // node_repl result is an error even when the runtime said isError=false).
392
- const prefix = normalizedToolResult(message, contentToText(message.content)).isError ? "[Tool Error]" : "[Tool Result]";
394
+ const prefix = normalizedToolResult(message, contentToText(message.content), codeMode).isError ? "[Tool Error]" : "[Tool Result]";
393
395
  // The bound compares in full-history space: this loop's `i` is already full-history on the
394
396
  // full-replay path, and `knownCallsOffset` re-bases it when only a suffix is replayed.
395
- const text = `${prefix}\n${toolResultToText(message, callBefore(replayedCalls, decodeCursorCallId(message.toolCallId), knownCallsOffset + i))}`;
397
+ const text = `${prefix}\n${toolResultToText(message, callBefore(replayedCalls, decodeCursorCallId(message.toolCallId), knownCallsOffset + i), codeMode)}`;
396
398
  pushDeduped(toolResultRootPayload(text), "toolResult", { messageIndex: i, text }, text);
397
399
  }
398
400
  }
@@ -829,6 +831,7 @@ function countImages(parts: DecodedResultPart[] | undefined): number {
829
831
  */
830
832
  function toolResultContentItems(
831
833
  message: OcxToolResultMessage,
834
+ codeMode: boolean,
832
835
  decoded?: DecodedResultPart[],
833
836
  maxImages = Number.POSITIVE_INFINITY,
834
837
  normalizedText?: NormalizedToolResult,
@@ -839,10 +842,10 @@ function toolResultContentItems(
839
842
  })];
840
843
  if (!parts) {
841
844
  const normalized = normalizedText
842
- ?? normalizedToolResult(message, typeof message.content === "string" ? message.content : "");
845
+ ?? normalizedToolResult(message, typeof message.content === "string" ? message.content : "", codeMode);
843
846
  return textItem(normalized.text);
844
847
  }
845
- const normalized = normalizedText ?? normalizedDecodedTextResult(message, parts);
848
+ const normalized = normalizedText ?? normalizedDecodedTextResult(message, parts, codeMode);
846
849
  if (normalized) {
847
850
  // #1920/#1866: empty or failure-state Computer Use / node_repl results are
848
851
  // normalized before they reach the native wire. Pure-text part arrays use
@@ -1058,8 +1061,9 @@ function toolCallsByCallId(messages: readonly OcxMessage[]): Map<string, Extract
1058
1061
  function toolResultToText(
1059
1062
  message: OcxToolResultMessage,
1060
1063
  call?: Extract<OcxAssistantContentPart, { type: "toolCall" }>,
1064
+ codeMode = false,
1061
1065
  ): string {
1062
- const normalized = normalizedToolResult(message, contentToText(message.content));
1066
+ const normalized = normalizedToolResult(message, contentToText(message.content), codeMode);
1063
1067
  return [
1064
1068
  "[tool_result]",
1065
1069
  `call_id: ${decodeCursorCallId(message.toolCallId)}`,
@@ -1075,12 +1079,16 @@ function toolResultToText(
1075
1079
  * Shared #1920 normalization entry: pure-text results only. Image-bearing or
1076
1080
  * encrypted results pass through untouched (their content is not plain text).
1077
1081
  */
1078
- function normalizedToolResult(message: OcxToolResultMessage, text: string): NormalizedToolResult {
1079
- if (message.containsEncryptedContent) return { text, isError: message.isError };
1082
+ function normalizedToolResult(message: OcxToolResultMessage, text: string, codeMode: boolean): NormalizedToolResult {
1083
+ if (message.containsEncryptedContent
1084
+ || (Array.isArray(message.content) && message.content.some(part => part.type !== "text"))) {
1085
+ return { text, isError: message.isError };
1086
+ }
1080
1087
  return normalizeCursorToolResultText(text, {
1081
1088
  toolName: message.toolName,
1082
1089
  toolNamespace: message.toolNamespace,
1083
1090
  isError: message.isError,
1091
+ codeMode,
1084
1092
  });
1085
1093
  }
1086
1094
 
@@ -1092,9 +1100,10 @@ function normalizedToolResult(message: OcxToolResultMessage, text: string): Norm
1092
1100
  function normalizedDecodedTextResult(
1093
1101
  message: OcxToolResultMessage,
1094
1102
  parts: DecodedResultPart[],
1103
+ codeMode: boolean,
1095
1104
  ): NormalizedToolResult | undefined {
1096
1105
  if (parts.some(part => part.kind !== "text")) return undefined;
1097
- return normalizedToolResult(message, parts.map(part => part.kind === "text" ? part.text : "").join("\n"));
1106
+ return normalizedToolResult(message, parts.map(part => part.kind === "text" ? part.text : "").join("\n"), codeMode);
1098
1107
  }
1099
1108
 
1100
1109
  function argBytes(value: unknown): Uint8Array {
@@ -1109,6 +1118,7 @@ function toolCallStep(
1109
1118
  part: Extract<OcxAssistantContentPart, { type: "toolCall" }>,
1110
1119
  requestScope: CursorBlobRequestScopeToken,
1111
1120
  result?: OcxToolResultMessage,
1121
+ codeMode = false,
1112
1122
  ): Uint8Array {
1113
1123
  const args: Record<string, Uint8Array> = {};
1114
1124
  for (const [key, value] of Object.entries(part.arguments ?? {})) args[key] = argBytes(value);
@@ -1130,7 +1140,7 @@ function toolCallStep(
1130
1140
  providerIdentifier: OCX_RESPONSES_TOOL_PROVIDER,
1131
1141
  args,
1132
1142
  }),
1133
- ...(result ? { result: toolResultPart(result, decodedResult, maxImages) } : {}),
1143
+ ...(result ? { result: toolResultPart(result, codeMode, decodedResult, maxImages) } : {}),
1134
1144
  }),
1135
1145
  },
1136
1146
  }),
@@ -1151,17 +1161,17 @@ function toolCallStep(
1151
1161
  return storeCursorBlob(encoded, requestScope);
1152
1162
  }
1153
1163
 
1154
- function toolResultPart(message: OcxToolResultMessage, decoded?: DecodedResultPart[], maxImages?: number) {
1164
+ function toolResultPart(message: OcxToolResultMessage, codeMode: boolean, decoded?: DecodedResultPart[], maxImages?: number) {
1155
1165
  const parts = decoded ?? decodeResultParts(message);
1156
1166
  const normalized = parts
1157
- ? normalizedDecodedTextResult(message, parts)
1158
- : normalizedToolResult(message, typeof message.content === "string" ? message.content : "");
1167
+ ? normalizedDecodedTextResult(message, parts, codeMode)
1168
+ : normalizedToolResult(message, typeof message.content === "string" ? message.content : "", codeMode);
1159
1169
  return create(McpToolResultSchema, {
1160
1170
  result: {
1161
1171
  case: "success",
1162
1172
  value: create(McpSuccessSchema, {
1163
1173
  isError: normalized?.isError ?? message.isError,
1164
- content: toolResultContentItems(message, parts, maxImages, normalized),
1174
+ content: toolResultContentItems(message, codeMode, parts, maxImages, normalized),
1165
1175
  }),
1166
1176
  },
1167
1177
  });
@@ -1199,6 +1209,7 @@ function lastActionIndex(messages: readonly OcxMessage[] | undefined): number {
1199
1209
  function conversationTurns(
1200
1210
  request: CursorRunRequest,
1201
1211
  requestScope: CursorBlobRequestScopeToken,
1212
+ codeMode: boolean,
1202
1213
  historyMessageStart = 0,
1203
1214
  /** Calls indexed from the FULL history; see {@link rootPromptMessages}. */
1204
1215
  knownCalls?: Map<string, Extract<OcxAssistantContentPart, { type: "toolCall" }>>,
@@ -1269,7 +1280,7 @@ function conversationTurns(
1269
1280
  // #1920/#1866: this external-replay site bypasses toolResultToText, so it
1270
1281
  // must consume the normalizer directly — cursor/grok-4.6 is the exact
1271
1282
  // reported repro path for empty Computer Use results.
1272
- const normalized = normalizedToolResult(message, contentToText(message.content));
1283
+ const normalized = normalizedToolResult(message, contentToText(message.content), codeMode);
1273
1284
  const prefix = normalized.isError ? "[Tool Error]" : "[Tool Result]";
1274
1285
  // Name the invocation here as well, for the same reason the root replay does: a result with
1275
1286
  // no visible originating call reads as an interrupted attempt (devlog 260829 000_rca).
@@ -1285,13 +1296,13 @@ function conversationTurns(
1285
1296
  }
1286
1297
  const priorCall = pendingToolCalls.get(message.toolCallId);
1287
1298
  if (priorCall) {
1288
- current.steps.push(toolCallStep(priorCall, requestScope, message));
1299
+ current.steps.push(toolCallStep(priorCall, requestScope, message, codeMode));
1289
1300
  pendingToolCalls.delete(message.toolCallId);
1290
1301
  } else {
1291
1302
  current.steps.push(storeCursorBlob(toBinary(ConversationStepSchema, create(ConversationStepSchema, {
1292
1303
  message: {
1293
1304
  case: "assistantMessage",
1294
- value: create(AssistantMessageSchema, { text: toolResultToText(message) }),
1305
+ value: create(AssistantMessageSchema, { text: toolResultToText(message, undefined, codeMode) }),
1295
1306
  },
1296
1307
  })), requestScope));
1297
1308
  }
@@ -1368,6 +1379,9 @@ function buildPreparedCursorRunRequest(
1368
1379
  options?: { estimateInputTokens?: boolean },
1369
1380
  ): PreparedCursorRunRequest {
1370
1381
  const rawText = activePromptText(request);
1382
+ // Use the same visible catalog as mcp_tools, including tool_choice, for every history path.
1383
+ const visibleTools = cursorToolsForActivePrompt(request.tools, rawText, request.toolChoice);
1384
+ const codeMode = cursorRequestUsesCodeMode(visibleTools, request.toolChoice);
1371
1385
  const lastRole = request.messages.at(-1)?.role;
1372
1386
  const text = lastRole === "user" || lastRole === "developer"
1373
1387
  ? appendCursorGenericToolUseHint(request.tools, rawText)
@@ -1471,7 +1485,7 @@ function buildPreparedCursorRunRequest(
1471
1485
  // against the raw limit left a band of a few hundred bytes below it where the checkpoint was kept,
1472
1486
  // the suffix budget collapsed, and the newest tool result vanished. Adding `systemBytes` moved the
1473
1487
  // band without closing it. Asking pruning what survived cannot drift from what pruning does.
1474
- const suffixRoots = rootPromptMessages(suffixRequest, requestScope, fullHistoryCalls, suffixStart, carriedRoots);
1488
+ const suffixRoots = rootPromptMessages(suffixRequest, requestScope, codeMode, fullHistoryCalls, suffixStart, carriedRoots);
1475
1489
  const suffixSystemCount = systemPromptBlobs(suffixRequest).length;
1476
1490
  // A tool continuation whose own result did not survive is worthless: that result is the whole
1477
1491
  // reason the turn exists. "Kept SOMETHING" is not enough either — inside the band this fix first
@@ -1543,7 +1557,7 @@ function buildPreparedCursorRunRequest(
1543
1557
  // checkpoint is re-decoded and re-abandoned each turn until TTL, which is wasted work rather
1544
1558
  // than wrong output (audit r8 rounds 3 and 4).
1545
1559
  } else {
1546
- const suffixTurns = conversationTurns(suffixRequest, requestScope, suffixRoots.historyMessageStart, fullHistoryCalls, suffixStart);
1560
+ const suffixTurns = conversationTurns(suffixRequest, requestScope, codeMode, suffixRoots.historyMessageStart, fullHistoryCalls, suffixStart);
1547
1561
  const suffixHistoryIds = suffixRoots.ids.slice(suffixSystemCount);
1548
1562
  const suffixHistorySerialized = suffixRoots.serialized.slice(suffixSystemCount);
1549
1563
  conversationState = create(ConversationStateStructureSchema, {
@@ -1573,10 +1587,10 @@ function buildPreparedCursorRunRequest(
1573
1587
  }
1574
1588
  }
1575
1589
  if (!conversationState) {
1576
- rootPromptMessagesState = rootPromptMessages(request, requestScope);
1590
+ rootPromptMessagesState = rootPromptMessages(request, requestScope, codeMode);
1577
1591
  conversationState = create(ConversationStateStructureSchema, {
1578
1592
  rootPromptMessagesJson: rootPromptMessagesState.ids,
1579
- turns: conversationTurns(request, requestScope, rootPromptMessagesState.historyMessageStart),
1593
+ turns: conversationTurns(request, requestScope, codeMode, rootPromptMessagesState.historyMessageStart),
1580
1594
  todos: [],
1581
1595
  pendingToolCalls: [],
1582
1596
  previousWorkspaceUris: [],
@@ -1590,7 +1604,6 @@ function buildPreparedCursorRunRequest(
1590
1604
  }
1591
1605
  // Hoisted out of the mcp_tools spread below so the estimate can read the same
1592
1606
  // filtered definitions the wire carries. Both helpers are pure.
1593
- const visibleTools = cursorToolsForActivePrompt(request.tools, rawText, request.toolChoice);
1594
1607
  const mcpToolDefs = buildCursorToolDefinitions(visibleTools, request.toolChoice);
1595
1608
  // The envelope is measured HERE, on the final root set, and nowhere else.
1596
1609
  //
@@ -1,5 +1,5 @@
1
1
  import type { OcxRequestOptions, OcxTool } from "../../types";
2
- import { CODE_MODE_RESULT_ECHO_SENTENCE } from "../exec-tool-result-normalize";
2
+ import { CODE_MODE_HOST_CONTRACT_SENTENCE, CODE_MODE_RESULT_ECHO_SENTENCE } from "../exec-tool-result-normalize";
3
3
  import { CODEX_SHELL_BRIDGE_TOOL_NAMES, CODEX_TOOL_SEARCH_TOOL, CODEX_UNIFIED_EXEC_TOOL, clientSemanticToolNameFromCursorWire, cursorRequestAdvertisesApplyPatch, cursorRequestHasExecutionPath, cursorRequestHasShellAlias, cursorRequestUsesCodeMode, cursorToolAllowedByChoice, cursorToolWireName, isCodexShellBridgeToolName, isCursorExecutionPathTool, isCursorStructuredEditToolName } from "./tool-naming";
4
4
 
5
5
  export const CURSOR_SHELL_ALIAS_SYSTEM_NOTE =
@@ -187,7 +187,7 @@ export function buildCursorToolGuidanceSystemNote(
187
187
  ? `\`${CODEX_UNIFIED_EXEC_TOOL}\` is Codex code mode: its body is JavaScript evaluated in a V8 isolate, not a shell command and not Node. Shell, file edits, and MCP are nested helpers called INSIDE that body as \`await tools.<name>(...)\`, for example \`await tools.exec_command({cmd: \"ls\"})\`. Read the tool description and the isolate global \`ALL_TOOLS\` (not \`tools.ALL_TOOLS\`) for helpers this turn provides; absence from the top-level catalog or from \`exec\`'s description is not absence. Those nested helpers are not themselves top-level tools, so do not call \`exec_command\` or \`shell_command\` at the top level here${codeModeOtherTopLevelNames.length > 0 ? `; every other tool this turn lists, including ${quotedNames(codeModeOtherTopLevelNames)}, remains callable at the top level as usual` : ""}. Nested \`tools.apply_patch(input)\` is host-executed: the string must begin exactly with \`*** Begin Patch\` and end with \`*** End Patch\`, each marker line being three asterisks, one space, the two words, then end of line with no further asterisks. OpenCodex does not rewrite JavaScript inside exec, so extra asterisks on a marker line are rejected by Codex before the file is touched.`
188
188
  : undefined,
189
189
  codeMode
190
- ? CODE_MODE_RESULT_ECHO_SENTENCE + " There is no `require`, no `module`, and no filesystem or network globals; reach the host only through the nested helpers."
190
+ ? CODE_MODE_RESULT_ECHO_SENTENCE + " There is no `require`, no `module`, and no filesystem or network globals; reach the host only through the nested helpers. " + CODE_MODE_HOST_CONTRACT_SENTENCE
191
191
  : undefined,
192
192
  codeMode
193
193
  ? "NEVER attempt Cursor-native Shell, Read, Grep, List, or any tool absent from the catalog — they are not executed in this environment and every probe wastes a turn. The exec code cell (with its nested helpers) is the ONLY execution surface; go to it directly on the FIRST attempt and do not narrate switching surfaces."
@@ -10,9 +10,12 @@
10
10
  */
11
11
 
12
12
  import {
13
+ CODE_MODE_HOST_RECOVERY_PREFIX,
13
14
  EMPTY_EXEC_OUTPUT_MESSAGE,
14
15
  EMPTY_EXEC_OUTPUT_REGEX,
15
16
  FAILED_EXEC_OUTPUT_MESSAGE,
17
+ annotateCodeModeHostFailure,
18
+ isCodexCodeModeExecResult,
16
19
  isFailedEmptyExecWrapper,
17
20
  isCodexExecBridgeTool,
18
21
  } from "../exec-tool-result-normalize";
@@ -83,7 +86,13 @@ export interface NormalizedToolResultText {
83
86
  */
84
87
  export function normalizeCursorToolResultText(
85
88
  text: string,
86
- options: { toolName?: string; toolNamespace?: string; isError?: boolean } = {},
89
+ options: {
90
+ toolName?: string;
91
+ toolNamespace?: string;
92
+ isError?: boolean;
93
+ /** True only when the request's visible catalog is Codex code mode. */
94
+ codeMode?: boolean;
95
+ } = {},
87
96
  ): NormalizedToolResultText {
88
97
  const isError = options.isError === true;
89
98
  const computerUse = isNodeReplOrComputerUseTool(options.toolName, options.toolNamespace);
@@ -104,7 +113,18 @@ export function normalizeCursorToolResultText(
104
113
  changed: true,
105
114
  };
106
115
  }
107
- if (!isError) {
116
+ // Replayed guidance and successful wrappers must not enter the legacy substring matcher.
117
+ if (text.includes(CODE_MODE_HOST_RECOVERY_PREFIX)
118
+ || /^(?:Script completed|Command finished|Execution finished)\b/.test(text.trimStart())) {
119
+ return { text, isError, changed: false };
120
+ }
121
+ // The request's visible catalog establishes provenance; the name alone also matches structured
122
+ // exec tools. Host guidance preserves Cursor's original error status.
123
+ if (options.codeMode === true && isCodexCodeModeExecResult(options.toolName, options.toolNamespace)) {
124
+ const hostFailure = annotateCodeModeHostFailure(text, options);
125
+ if (hostFailure !== undefined) return { text: hostFailure, isError, changed: true };
126
+ }
127
+ if (computerUse && !isError) {
108
128
  for (const { marker, guidance } of RUNTIME_FAILURE_GUIDANCE) {
109
129
  if (text.includes(marker)) {
110
130
  return { text: `${text}\n[recovery: ${guidance}]`, isError: true, changed: true };
@@ -115,6 +115,93 @@ export const EMPTY_EXEC_OUTPUT_MESSAGE =
115
115
  export const CODE_MODE_RESULT_ECHO_SENTENCE =
116
116
  "Nothing in the isolate is echoed automatically: a bare trailing `await tools.<name>(...)` or final expression value is DISCARDED, and the cell reports empty output. Pass anything you need to read to `text(...)` (or `notify(...)`) in the same cell — for example `text(JSON.stringify(await tools.exec_command({cmd: 'ls'})))` — and treat an empty result as your own missing `text(...)` call rather than a failed command or lost context.";
117
117
 
118
+ /**
119
+ * Host rules a routed model most often breaks on its first code-mode edit or wait, stated BEFORE
120
+ * the call. Wording tracks the Codex host (0.153.2), probed live on 2026-09-07: a non-string
121
+ * argument to `apply_patch` throws "expects a string input"; a body whose first line is not the
122
+ * bare marker (decorated `*** Begin Patch ***`, a code fence, prose) throws "The first line of the
123
+ * patch must be '*** Begin Patch'" — surrounding newlines are tolerated; ES imports throw
124
+ * "Unsupported import in exec"; a command that outlives `yield_time_ms` returns `session_id` for
125
+ * `write_stdin` polling. xai/grok-4.6 hit the first two, abandoned apply_patch for heredoc writes,
126
+ * blocked a turn in a shell sleep loop, and died once on an import. None of that is repairable in
127
+ * the proxy (devlog/_plan/260905_apply_patch_envelope_gap/010 MODE B); it is a contract the proxy
128
+ * had not stated.
129
+ */
130
+ export const CODE_MODE_HOST_CONTRACT_SENTENCE =
131
+ "Host contract for the nested helpers: `tools.apply_patch(patch)` takes exactly one string, never an object such as `{input: ...}`; the patch text opens with the bare marker line `*** Begin Patch` and closes with the bare marker line `*** End Patch`, written without a code fence, prose, or extra asterisks on those lines (blank lines or indentation around the markers are tolerated; a decorated or missing marker is rejected). The isolate has no `import`, `require`, or module loader; use the globals the exec tool description lists (for example `tools`, `text`, `notify`, `store`/`load`, `ALL_TOOLS`). For a command that may outlive `yield_time_ms`, let `tools.exec_command` return a `session_id` and poll it on later calls with `tools.write_stdin({session_id, chars: \"\"})` instead of blocking a shell in a sleep loop.";
132
+
133
+ /**
134
+ * Post-hoc half of the host contract: the four host strings a routed model reads inside a
135
+ * non-error exec result, each paired with the rule it broke. Matched case-insensitively because
136
+ * the host writes "Unsupported import in exec: <spec>" while Cursor's earlier marker was
137
+ * lowercase; one table, one owner, so this text and the pre-call sentence cannot drift.
138
+ */
139
+ export const CODE_MODE_HOST_FAILURE_GUIDANCE: ReadonlyArray<{ marker: string; guidance: string }> = [
140
+ {
141
+ marker: "expects a string input",
142
+ guidance: "tools.apply_patch takes exactly one string argument; pass the patch text itself, not an object such as {input: ...}.",
143
+ },
144
+ {
145
+ marker: "the first line of the patch must be",
146
+ guidance: "The patch text must open with the bare marker line `*** Begin Patch`: no code fence, prose, or extra asterisks on that line (blank lines or indentation before it are tolerated).",
147
+ },
148
+ {
149
+ marker: "the last line of the patch must be",
150
+ guidance: "The patch text must close with the bare marker line `*** End Patch`: no trailing text or extra asterisks on that line (blank lines after it are tolerated).",
151
+ },
152
+ {
153
+ marker: "unsupported import in exec",
154
+ guidance: "Imports are not available in this exec context; use the injected globals (tools, text, notify, store, load, ALL_TOOLS) instead.",
155
+ },
156
+ ];
157
+
158
+ /** Prefix of every recovery line this module appends; callers use it to recognise replayed annotations. */
159
+ export const CODE_MODE_HOST_RECOVERY_PREFIX = "[recovery: ";
160
+
161
+ // Only a leading failure envelope or a complete host diagnostic establishes error context.
162
+ // Do not search for this prefix inside output: successful source reads can quote any of these.
163
+ const CODE_MODE_HOST_ERROR_PREFIX = /^(?:Script failed(?:[ \t]*(?:\r?\n|$)|:)|Script error:|(?:Error|TypeError|SyntaxError):|tool `apply_patch` expects a string input\b|apply_patch verification failed:|Unsupported import in exec:)/i;
164
+
165
+ /** Namespaces under which Cursor displays Codex's own Responses tools (see cursor/tool-naming.ts). */
166
+ const CODEX_RESPONSES_DISPLAY_NAMESPACES: ReadonlySet<string> = new Set(["opencodex-responses", "mcp__opencodex-responses"]);
167
+ /** Flattened spellings of the same code-mode exec when a client folds the namespace into the name. */
168
+ const CODEX_CODE_MODE_EXEC_ALIASES: ReadonlySet<string> = new Set(["exec", "mcp__opencodex-responses__exec", "mcp_opencodex-responses_exec"]);
169
+
170
+ /**
171
+ * The code-mode `exec` tool by NAME — bare, or under Codex's own `opencodex-responses` display
172
+ * namespace, matched exactly. The four host strings above originate only in that isolate, so flat
173
+ * shell bridges (`exec_command`, `shell`, …) and every other namespace (`mcp__docker`,
174
+ * `mcp__foreign-opencodex-responses`) are excluded: an unrelated server's output that quotes the
175
+ * phrase must not receive Codex guidance. Narrower than `isCodexExecBridgeTool` on purpose; the
176
+ * empty-output repair keeps the wider gate. Callers that KNOW the catalog shape (Kiro's
177
+ * `codeModeExecName`, the Responses body gate) add that check on top; this predicate alone cannot
178
+ * tell a structured tool named `exec` from the freeform one.
179
+ */
180
+ export function isCodexCodeModeExecResult(toolName?: string, toolNamespace?: string): boolean {
181
+ if (!toolName) return false;
182
+ const lower = toolName.toLowerCase();
183
+ if (toolNamespace !== undefined) return CODEX_RESPONSES_DISPLAY_NAMESPACES.has(toolNamespace) && lower === "exec";
184
+ return CODEX_CODE_MODE_EXEC_ALIASES.has(lower);
185
+ }
186
+
187
+ /**
188
+ * Append a one-line recovery hint when a code-mode exec result starts with a host error context
189
+ * and carries a known diagnostic. Successful wrappers and unframed phrase quotations pass through.
190
+ * Returns undefined when the tool/context/marker does not match or a recovery line is already
191
+ * present (a replayed result must not grow a second one). Never touches error status.
192
+ */
193
+ export function annotateCodeModeHostFailure(
194
+ text: string,
195
+ options: { toolName?: string; toolNamespace?: string } = {},
196
+ ): string | undefined {
197
+ if (!isCodexCodeModeExecResult(options.toolName, options.toolNamespace)) return undefined;
198
+ if (text.includes(CODE_MODE_HOST_RECOVERY_PREFIX)) return undefined;
199
+ if (!CODE_MODE_HOST_ERROR_PREFIX.test(text.trimStart())) return undefined;
200
+ const lower = text.toLowerCase();
201
+ const hit = CODE_MODE_HOST_FAILURE_GUIDANCE.find(({ marker }) => lower.includes(marker));
202
+ return hit ? `${text}\n${CODE_MODE_HOST_RECOVERY_PREFIX}${hit.guidance}]` : undefined;
203
+ }
204
+
118
205
  /**
119
206
  * Codex exec / shell-bridge tool names (flat and MCP-prefixed display aliases). An empty result
120
207
  * here is almost always a code-mode cell that never called text()/notify().
@@ -1,6 +1,7 @@
1
1
  import { decodeEventStream } from "../lib/eventstream-decoder";
2
2
  import { estimateTokens } from "../lib/token-estimate";
3
3
  import { debugProviderDiagnostic } from "../lib/debug";
4
+ import { isDebugEnabled } from "../lib/debug-settings";
4
5
  import { resolveKiroApiRegion, resolveKiroRequestProfile } from "../oauth/kiro";
5
6
  import { KIRO_MODEL_CONTEXT_WINDOWS, normalizeKiroModelId } from "../providers/kiro-models";
6
7
  import { modelRecordValue } from "../reasoning-effort";
@@ -44,7 +45,7 @@ import { extractKiroImages, normalizeKiroImages, type KiroImage } from "./kiro-i
44
45
  import { sniffImageDimensions } from "./anthropic-image-guard";
45
46
  import { fetchKiroWithRetry, noteKiroTransientThrottle } from "./kiro-retry";
46
47
  import { convertKiroToolContext } from "./kiro-tools";
47
- import { EMPTY_EXEC_OUTPUT_MESSAGE, normalizeEmptyExecToolResultText } from "./exec-tool-result-normalize";
48
+ import { EMPTY_EXEC_OUTPUT_MESSAGE, annotateCodeModeHostFailure, normalizeEmptyExecToolResultText } from "./exec-tool-result-normalize";
48
49
  import { identifyRoutedModel } from "./identity";
49
50
  import { buildNonOpenAIToolCatalogNudgeFromNames, isBareShellBridgeTool, isCodexCodeModeExecTool } from "./tool-catalog-nudge";
50
51
  import {
@@ -755,11 +756,17 @@ export function buildKiroPayload(
755
756
  // the task instead of calling text()/notify(). Checked before `text.trim()` because the
756
757
  // wrapper form ("Script completed\nWall time ...\nOutput:\n") is non-blank and would
757
758
  // otherwise pass through as if it were real output.
758
- const normalizedExecText = normalizeEmptyExecToolResultText(text, {
759
- toolName: tr.toolName,
760
- toolNamespace: tr.toolNamespace,
761
- });
762
- const resultText = normalizedExecText ?? (text.trim() ? text : KIRO_EMPTY_TOOL_RESULT_MESSAGE);
759
+ const execOptions = { toolName: tr.toolName, toolNamespace: tr.toolNamespace };
760
+ const normalizedExecText = normalizeEmptyExecToolResultText(text, execOptions);
761
+ // A host failure string inside a non-empty exec result gets the rule it broke appended, but
762
+ // only when this request's emitted catalog is genuinely code mode (`codeModeExecName` above):
763
+ // a structured tool named exec, or exec beside a shell bridge, never ran the isolate. This is
764
+ // the only substitution the grouping path below also carries: whitespace and empty/failed
765
+ // wrappers keep their existing raw policy.
766
+ const annotatedExecText = normalizedExecText === undefined && codeModeExecName !== undefined
767
+ ? annotateCodeModeHostFailure(text, execOptions)
768
+ : undefined;
769
+ const resultText = normalizedExecText ?? annotatedExecText ?? (text.trim() ? text : KIRO_EMPTY_TOOL_RESULT_MESSAGE);
763
770
  const images = extractKiroImages(tr.content);
764
771
  const toolUseId = normalizeToolId(tr.toolCallId);
765
772
  const call = priorCalls.get(toolUseId);
@@ -768,7 +775,7 @@ export function buildKiroPayload(
768
775
  }
769
776
  // Keep real whitespace and failed wrappers, but no empty-success wrapper boilerplate.
770
777
  const rawGroupText = text.length > 0 && (!text.trim() || normalizedExecText !== EMPTY_EXEC_OUTPUT_MESSAGE)
771
- ? text : undefined;
778
+ ? (annotatedExecText ?? text) : undefined;
772
779
  const last = turns.at(-1);
773
780
  if (
774
781
  adjacentResult?.rawId === tr.toolCallId
@@ -2114,17 +2121,21 @@ export function createKiroAdapter(provider: OcxProviderConfig): ProviderAdapter
2114
2121
  const rawContextInputEstimate = estimateKiroPayloadInputTokens(built.payload, parsed.modelId);
2115
2122
  const contextInputEstimate = calibrateKiroEstimate(built.conversationId, rawContextInputEstimate);
2116
2123
  const body = JSON.stringify(built.payload);
2117
- debugProviderDiagnostic("kiro", "request", {
2118
- region,
2119
- requestedModel: parsed.modelId,
2120
- completionMode: built.completionMode,
2121
- bodyBytes: new TextEncoder().encode(body).length,
2122
- messageCount: kiroPayloadMessages(parsed).length,
2123
- toolCount: parsed.context.tools?.length ?? 0,
2124
- hasProfileArn: Boolean(profileArn),
2125
- wireClient,
2126
- hasPreviousResponseId: Boolean(parsed.previousResponseId),
2127
- });
2124
+ // Every field below is evaluated before the call, so an unguarded call re-encodes the
2125
+ // whole request body on each request even when provider debug is off. Gate the details.
2126
+ if (isDebugEnabled()) {
2127
+ debugProviderDiagnostic("kiro", "request", {
2128
+ region,
2129
+ requestedModel: parsed.modelId,
2130
+ completionMode: built.completionMode,
2131
+ bodyBytes: new TextEncoder().encode(body).length,
2132
+ messageCount: kiroPayloadMessages(parsed).length,
2133
+ toolCount: parsed.context.tools?.length ?? 0,
2134
+ hasProfileArn: Boolean(profileArn),
2135
+ wireClient,
2136
+ hasPreviousResponseId: Boolean(parsed.previousResponseId),
2137
+ });
2138
+ }
2128
2139
  return {
2129
2140
  request: {
2130
2141
  url: kiroRuntimeEndpoint(provider, region),
@@ -1,6 +1,6 @@
1
1
  import { toolChoiceToolPredicate, type OcxParsedRequest, type OcxProviderConfig } from "../types";
2
2
  import { isOpenAiOperatedResponsesDestination } from "../providers/openai-tiers";
3
- import { CODE_MODE_RESULT_ECHO_SENTENCE, normalizeEmptyExecToolResultText } from "./exec-tool-result-normalize";
3
+ import { CODE_MODE_HOST_CONTRACT_SENTENCE, CODE_MODE_RESULT_ECHO_SENTENCE, annotateCodeModeHostFailure, normalizeEmptyExecToolResultText } from "./exec-tool-result-normalize";
4
4
  import { isBareShellBridgeTool, isCodexCodeModeExecTool } from "./tool-catalog-nudge";
5
5
 
6
6
  function record(value: unknown): value is Record<string, unknown> {
@@ -29,6 +29,14 @@ function withExecInputGuidance(tool: unknown): unknown {
29
29
  } } };
30
30
  }
31
31
 
32
+ /** Append each sentence a replayed instructions string does not already carry, in order. */
33
+ function appendMissing(instructions: string, sentences: readonly string[]): string {
34
+ return sentences.reduce(
35
+ (acc, sentence) => acc.includes(sentence) ? acc : [acc, sentence].filter(Boolean).join("\n\n"),
36
+ instructions,
37
+ );
38
+ }
39
+
32
40
  /** Native routed Responses needs the same first-call/output contract as translated adapters. */
33
41
  export function normalizeResponsesCodeMode(body: unknown, parsed: OcxParsedRequest, provider: OcxProviderConfig): unknown {
34
42
  if (!record(body) || parsed._compactionRequest || isOpenAiOperatedResponsesDestination(provider)) return body;
@@ -42,8 +50,7 @@ export function normalizeResponsesCodeMode(body: unknown, parsed: OcxParsedReque
42
50
  .map(item => item.call_id));
43
51
  return {
44
52
  ...body,
45
- instructions: instructions.includes(CODE_MODE_RESULT_ECHO_SENTENCE)
46
- ? instructions : [instructions, CODE_MODE_RESULT_ECHO_SENTENCE].filter(Boolean).join("\n\n"),
53
+ instructions: appendMissing(instructions, [CODE_MODE_RESULT_ECHO_SENTENCE, CODE_MODE_HOST_CONTRACT_SENTENCE]),
47
54
  ...(Array.isArray(body.tools) ? { tools: body.tools.map(withExecInputGuidance) } : {}),
48
55
  ...(input ? { input: input.map(item => {
49
56
  if (!record(item)) return item;
@@ -52,7 +59,10 @@ export function normalizeResponsesCodeMode(body: unknown, parsed: OcxParsedReque
52
59
  }
53
60
  if ((item.type !== "function_call_output" && item.type !== "custom_tool_call_output") || !execCalls.has(item.call_id)) return item;
54
61
  const text = textOnlyOutput(item.output);
55
- const normalized = text === undefined ? undefined : normalizeEmptyExecToolResultText(text, { toolName: "exec" });
62
+ const normalized = text === undefined
63
+ ? undefined
64
+ : normalizeEmptyExecToolResultText(text, { toolName: "exec" })
65
+ ?? annotateCodeModeHostFailure(text, { toolName: "exec" });
56
66
  return normalized === undefined ? item : { ...item, output: normalized };
57
67
  }) } : {}),
58
68
  };
@@ -5,7 +5,7 @@ import {
5
5
  type OcxTool,
6
6
  type OcxProviderConfig,
7
7
  } from "../types";
8
- import { CODE_MODE_RESULT_ECHO_SENTENCE } from "./exec-tool-result-normalize";
8
+ import { CODE_MODE_HOST_CONTRACT_SENTENCE, CODE_MODE_RESULT_ECHO_SENTENCE } from "./exec-tool-result-normalize";
9
9
 
10
10
  // Tool names that exist only in OTHER agent harnesses (Claude Code and friends). Naming one
11
11
  // here tells a routed model not to call it unless this turn's catalog really lists it.
@@ -121,7 +121,7 @@ export function buildNonOpenAIToolCatalogNudgeFromNames(
121
121
  "Call only listed names with their listed argument keys; do not invent, translate, or rename tools.",
122
122
  "Names mentioned only in instructions, tool descriptions, argument descriptions, or nested helper APIs are not additional top-level tools.",
123
123
  verifiedCodeModeExecName
124
- ? "`" + verifiedCodeModeExecName + "` is Codex code mode: its body is JavaScript evaluated in a V8 isolate. Nested helpers are called INSIDE that body as `await tools.<name>(...)`, for example `await tools.exec_command({cmd: \"ls\"})` or `await tools.codex_app__list_threads({})`. Absence from the top-level catalog or from `" + verifiedCodeModeExecName + "`'s description is not absence: deferred helpers stay callable on `tools.<name>`. Discover them from the isolate global `ALL_TOOLS`, not `tools.ALL_TOOLS`. Do not skip an available nested helper because it is omitted from the listed top-level names. " + CODE_MODE_RESULT_ECHO_SENTENCE + " Nested `tools.apply_patch(input)` is host-executed: the string must begin exactly with `*** Begin Patch` and end with `*** End Patch`, each marker line being three asterisks, one space, the two words, then end of line with no further asterisks. OpenCodex does not rewrite JavaScript inside exec, so extra asterisks on a marker line are rejected by Codex before the file is touched."
124
+ ? "`" + verifiedCodeModeExecName + "` is Codex code mode: its body is JavaScript evaluated in a V8 isolate. Nested helpers are called INSIDE that body as `await tools.<name>(...)`, for example `await tools.exec_command({cmd: \"ls\"})` or `await tools.codex_app__list_threads({})`. Absence from the top-level catalog or from `" + verifiedCodeModeExecName + "`'s description is not absence: deferred helpers stay callable on `tools.<name>`. Discover them from the isolate global `ALL_TOOLS`, not `tools.ALL_TOOLS`. Do not skip an available nested helper because it is omitted from the listed top-level names. " + CODE_MODE_RESULT_ECHO_SENTENCE + " Nested `tools.apply_patch(input)` is host-executed: the string must begin exactly with `*** Begin Patch` and end with `*** End Patch`, each marker line being three asterisks, one space, the two words, then end of line with no further asterisks. OpenCodex does not rewrite JavaScript inside exec, so extra asterisks on a marker line are rejected by Codex before the file is touched. " + CODE_MODE_HOST_CONTRACT_SENTENCE
125
125
  : "If a listed tool exposes nested helpers such as a tools.* API, call the listed parent tool and use those helpers only inside that tool's input.",
126
126
  unavailableNeighborNames.length > 0
127
127
  ? "Do not use neighboring-agent tool names " + quoteNames(unavailableNeighborNames) + " unless this turn's catalog lists those exact names."