@bitkyc08/opencodex 2.54.0 → 2.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (86) hide show
  1. package/gui/dist/assets/{index-CkvITofZ.js → index-VuoiWj9J.js} +10 -10
  2. package/gui/dist/index.html +1 -1
  3. package/package.json +1 -1
  4. package/src/adapters/anthropic-image-codec.ts +57 -0
  5. package/src/adapters/anthropic-image-normalize.ts +28 -1
  6. package/src/adapters/anthropic.ts +68 -6
  7. package/src/adapters/base.ts +8 -0
  8. package/src/adapters/coding-agent/protocol.ts +41 -16
  9. package/src/adapters/cursor/cursor-errors.ts +1 -1
  10. package/src/adapters/cursor/live-transport.ts +5 -1
  11. package/src/adapters/cursor/native-exec-fs.ts +10 -10
  12. package/src/adapters/cursor/native-exec-network.ts +2 -2
  13. package/src/adapters/cursor/native-exec-shell.ts +13 -12
  14. package/src/adapters/cursor/native-exec.ts +51 -10
  15. package/src/adapters/cursor/policy-error.ts +75 -0
  16. package/src/adapters/cursor/protobuf-request.ts +105 -1
  17. package/src/adapters/devin/cloud-direct/catalog.ts +34 -2
  18. package/src/adapters/devin/live-models.ts +33 -2
  19. package/src/adapters/google-wire-compiler.ts +8 -0
  20. package/src/adapters/google.ts +46 -0
  21. package/src/adapters/input-media-guard.ts +45 -0
  22. package/src/adapters/kiro/adapter.ts +8 -0
  23. package/src/adapters/kiro/payload.ts +28 -6
  24. package/src/adapters/kiro-events.ts +25 -6
  25. package/src/adapters/kiro-images.ts +30 -0
  26. package/src/adapters/kiro-retry.ts +8 -0
  27. package/src/adapters/openai-chat.ts +33 -4
  28. package/src/adapters/openai-responses.ts +26 -0
  29. package/src/adapters/registry.ts +4 -0
  30. package/src/bridge.ts +163 -116
  31. package/src/chat/image-parts.ts +151 -0
  32. package/src/chat/inbound.ts +70 -33
  33. package/src/cli/connect.ts +30 -9
  34. package/src/cli/dispatch.ts +7 -3
  35. package/src/cli/index.ts +3 -0
  36. package/src/cli/runtime-api.ts +25 -0
  37. package/src/cli/status.ts +21 -19
  38. package/src/cli/system-restart-client.ts +25 -0
  39. package/src/clients/config-export.ts +14 -4
  40. package/src/codex/app-server-processes.ts +25 -0
  41. package/src/codex/auth-context.ts +8 -0
  42. package/src/codex/autostart-health.ts +36 -2
  43. package/src/codex/catalog/provider-fetch.ts +41 -0
  44. package/src/codex/catalog-auto-refresh.ts +182 -0
  45. package/src/codex/catalog-refresh-status.ts +93 -0
  46. package/src/codex/history-provider.ts +55 -0
  47. package/src/codex/model-entitlements.ts +78 -0
  48. package/src/codex/native-profile-processes.ts +114 -15
  49. package/src/codex/prompt-text-probe.ts +274 -41
  50. package/src/codex/routing-adoption.ts +189 -0
  51. package/src/codex/routing.ts +520 -48
  52. package/src/codex/runtime.ts +249 -7
  53. package/src/combos/failover.ts +45 -0
  54. package/src/config.ts +124 -4
  55. package/src/generated/compatibility-version.json +110 -74
  56. package/src/generated/model-metadata.ts +1 -0
  57. package/src/lib/request-execution-budget.ts +202 -0
  58. package/src/lib/upstream-retry.ts +95 -8
  59. package/src/lib/workflow-budget.ts +172 -0
  60. package/src/oauth/devin.ts +57 -12
  61. package/src/providers/quota.ts +37 -6
  62. package/src/providers/registry.ts +53 -6
  63. package/src/responses/input-media.ts +65 -0
  64. package/src/responses/parser-content.ts +42 -0
  65. package/src/responses/schema.ts +12 -2
  66. package/src/server/audio-live.ts +1 -2
  67. package/src/server/audio-transcriptions.ts +1 -2
  68. package/src/server/auth-cors.ts +1 -1
  69. package/src/server/background-lifecycle.ts +18 -0
  70. package/src/server/chat-completions.ts +23 -8
  71. package/src/server/chat-native.ts +17 -17
  72. package/src/server/index.ts +24 -0
  73. package/src/server/management/request-history-routes.ts +5 -0
  74. package/src/server/request-log.ts +8 -2
  75. package/src/server/responses/compact.ts +51 -3
  76. package/src/server/responses/core.ts +238 -28
  77. package/src/server/search.ts +7 -9
  78. package/src/types/config.ts +51 -6
  79. package/src/usage/log.ts +37 -0
  80. package/src/vision/eligibility.ts +37 -4
  81. package/src/vision/index.ts +1 -0
  82. package/src/vision/plan.ts +45 -10
  83. package/src/web-search/alpha-search.ts +324 -0
  84. package/src/web-search/index.ts +13 -22
  85. package/src/web-search/passthrough-bridge.ts +195 -22
  86. package/src/web-search/sidecar-providers.ts +22 -0
@@ -53,6 +53,8 @@ import {
53
53
  import { clientBytes, execBytes, execStreamCloseBytes, execThrowBytes } from "./native-exec-common";
54
54
  import type { McpToolDefinition } from "./gen/agent_pb";
55
55
  import { OCX_RESPONSES_TOOL_PROVIDER } from "./tool-definitions";
56
+ import { cursorRequestHasExecutionPath, cursorRequestHasShellAlias, cursorToolWireName } from "./tool-naming";
57
+ import type { OcxTool } from "../../types";
56
58
 
57
59
  export type CursorNativeExecDeps = CursorNativeNetworkDeps & CursorNativeToolDeps;
58
60
 
@@ -72,6 +74,45 @@ export interface CursorNativeExecContext extends CursorNativeExecDeps {
72
74
  rejectNativeFileMutations?: boolean;
73
75
  /** The synthetic exact-match edit tools (edit_file / multi_edit) are advertised this request. */
74
76
  structuredEditAvailable?: boolean;
77
+ /** Catalog-aware redirect text for denied native fs/shell attempts (undefined = default bridge wording). */
78
+ nativeExecRedirectHint?: string;
79
+ }
80
+
81
+ const REDIRECT_HINT_MAX_TOOLS = 16;
82
+
83
+ /**
84
+ * Redirect text for Cursor-native fs/shell/fetch attempts when the request catalog carries NO shell
85
+ * bridge or other execution-path tool (an orchestrator client that only exposes delegation tools,
86
+ * for example). The default refusal steers the model to `shell_command` / `exec_command`; when those
87
+ * are not in the catalog some models (kimi-k3 observed) conclude every tool is unavailable and give
88
+ * up instead of using the tools that ARE listed. Name the real catalog instead — the client tools
89
+ * plus any configured MCP tools advertised this turn — and stay neutral about what those tools can
90
+ * do, so a listed file/search/fetch tool is never contradicted.
91
+ */
92
+ export function cursorNativeExecRedirectHint(
93
+ tools: readonly Pick<OcxTool, "namespace" | "name">[] | undefined,
94
+ mcpToolDefs: readonly Pick<McpToolDefinition, "name" | "providerIdentifier">[] = [],
95
+ ): string | undefined {
96
+ const clientTools = tools ?? [];
97
+ if (cursorRequestHasShellAlias(clientTools) || cursorRequestHasExecutionPath(clientTools)) return undefined;
98
+ // Client tools are advertised under OCX_RESPONSES_TOOL_PROVIDER, so the harness shows them as
99
+ // `mcp_<provider>_<wire name>`; configured MCP servers are advertised under their own provider id.
100
+ // A request with no client tools but configured MCP tools still gets those named; a request that
101
+ // advertises nothing at all keeps the default bridge wording.
102
+ const names = [...new Set([
103
+ ...clientTools.map(cursorToolWireName),
104
+ ...mcpToolDefs.map(def => `mcp_${def.providerIdentifier}_${def.name}`),
105
+ ])];
106
+ if (names.length === 0) return undefined;
107
+ const shown = names.slice(0, REDIRECT_HINT_MAX_TOOLS).map(name => `\`${name}\``).join(", ");
108
+ const more = names.length > REDIRECT_HINT_MAX_TOOLS ? ` (+${names.length - REDIRECT_HINT_MAX_TOOLS} more)` : "";
109
+ return (
110
+ `Re-issue this operation NOW through one of the tools listed in this request's catalog: ${shown}${more} `
111
+ + `(the harness displays a \`${OCX_RESPONSES_TOOL_PROVIDER}\` entry as \`mcp_${OCX_RESPONSES_TOOL_PROVIDER}_<name>\`; that is the same tool). `
112
+ + "Cursor-native Read/Glob/Grep/LS/Shell/Write/Fetch are not part of this request's catalog; do not retry them. "
113
+ + "Pick the listed tool that fits the operation — a listed file, search, or fetch tool if there is one, otherwise the listed tool that delegates work to a worker agent. "
114
+ + "Do NOT narrate this redirect, do NOT comment on tool availability, and do NOT re-announce the task — just make the catalog tool call."
115
+ );
75
116
  }
76
117
 
77
118
  export function cursorUnsafeNativeLocalExecEnabled(input: Pick<CursorNativeExecContext, "unsafeAllowNativeLocalExec"> = {}): boolean {
@@ -634,16 +675,16 @@ export async function handleCursorNativeExec(execMsg: ExecServerMessage, deps: C
634
675
  }))];
635
676
  }
636
677
  if (!cursorUnsafeNativeLocalExecEnabled(deps)) {
637
- if (execCase === "readArgs") return [rejectReadExecForPolicy(execMsg)];
638
- if (execCase === "writeArgs") return [rejectWriteExecForPolicy(execMsg)];
639
- if (execCase === "deleteArgs") return [rejectDeleteExecForPolicy(execMsg)];
640
- if (execCase === "lsArgs") return [rejectLsExecForPolicy(execMsg)];
641
- if (execCase === "grepArgs") return [rejectGrepExecForPolicy(execMsg)];
642
- if (execCase === "shellArgs") return [rejectShellExecForPolicy(execMsg)];
643
- if (execCase === "shellStreamArgs") return rejectShellStreamExecForPolicy(execMsg);
644
- if (execCase === "backgroundShellSpawnArgs") return [rejectBackgroundShellSpawnExecForPolicy(execMsg)];
645
- if (execCase === "writeShellStdinArgs") return [rejectWriteShellStdinExecForPolicy(execMsg)];
646
- if (execCase === "fetchArgs") return [rejectFetchExecForPolicy(execMsg)];
678
+ if (execCase === "readArgs") return [rejectReadExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
679
+ if (execCase === "writeArgs") return [rejectWriteExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
680
+ if (execCase === "deleteArgs") return [rejectDeleteExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
681
+ if (execCase === "lsArgs") return [rejectLsExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
682
+ if (execCase === "grepArgs") return [rejectGrepExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
683
+ if (execCase === "shellArgs") return [rejectShellExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
684
+ if (execCase === "shellStreamArgs") return rejectShellStreamExecForPolicy(execMsg, deps.nativeExecRedirectHint);
685
+ if (execCase === "backgroundShellSpawnArgs") return [rejectBackgroundShellSpawnExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
686
+ if (execCase === "writeShellStdinArgs") return [rejectWriteShellStdinExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
687
+ if (execCase === "fetchArgs") return [rejectFetchExecForPolicy(execMsg, deps.nativeExecRedirectHint)];
647
688
  }
648
689
  if (execCase === "readArgs") return [readExec(execMsg)];
649
690
  if (execCase === "writeArgs") return [deps.rejectNativeFileMutations ? rejectWriteExecForApplyPatch(execMsg, deps.structuredEditAvailable === true) : writeExec(execMsg)];
@@ -0,0 +1,75 @@
1
+ import { BinaryReader, WireType } from "@bufbuild/protobuf/wire";
2
+
3
+ const POLICY_TITLE = "Review Data Policy";
4
+ const POLICY_DETAIL = "You must acknowledge Claude Fable 5's data retention policy to use the model.";
5
+ const POLICY_REVIEW_URL = "https://cursor.com/dashboard/restricted_models/claude-fable-5";
6
+ const MAX_VALUE_CHARS = 16_384;
7
+ const MAX_FIELDS = 128;
8
+
9
+ /**
10
+ * Minimal read-only projection of Cursor's aiserver.v1.ErrorDetails:
11
+ * error=1 (MODEL_BLOCKED=58), details=2; CustomErrorDetails title=1, detail=2.
12
+ * Verified against the native CLI schema and the #4508 Connect binary response.
13
+ * Skip buttons, URLs, analytics and dashboardAction rather than interpreting them.
14
+ */
15
+ function isFablePolicyError(bytes: Uint8Array, custom = false): boolean {
16
+ const reader = new BinaryReader(bytes);
17
+ let error: number | undefined;
18
+ let details: Uint8Array | undefined;
19
+ let title: string | undefined;
20
+ let detail: string | undefined;
21
+ let fields = 0;
22
+ while (reader.pos < reader.len) {
23
+ if (++fields > MAX_FIELDS) return false;
24
+ const [field, wire] = reader.tag();
25
+ // Groups are not part of this proto3 projection; avoid recursive unknown-field skips.
26
+ if (wire === WireType.StartGroup || wire === WireType.EndGroup) return false;
27
+ if (!custom && field === 1) {
28
+ if (wire !== WireType.Varint || error !== undefined) return false;
29
+ error = reader.uint32();
30
+ } else if (!custom && field === 2) {
31
+ if (wire !== WireType.LengthDelimited || details !== undefined) return false;
32
+ details = reader.bytes();
33
+ } else if (custom && (field === 1 || field === 2)) {
34
+ if (wire !== WireType.LengthDelimited) return false;
35
+ const value = reader.bytes();
36
+ if (value.length > 256) return false;
37
+ const text = new TextDecoder("utf-8", { fatal: true }).decode(value);
38
+ if (field === 1) {
39
+ if (title !== undefined) return false;
40
+ title = text;
41
+ } else {
42
+ if (detail !== undefined) return false;
43
+ detail = text;
44
+ }
45
+ } else {
46
+ reader.skip(wire);
47
+ }
48
+ }
49
+ return custom
50
+ ? title === POLICY_TITLE && detail === POLICY_DETAIL
51
+ : error === 58 && details !== undefined && isFablePolicyError(details, true);
52
+ }
53
+
54
+ /** Recognize this policy gate, but never forward arbitrary upstream text or consent actions. */
55
+ export function cursorPolicyErrorExplanation(error: unknown): string | undefined {
56
+ if (!error || typeof error !== "object") return undefined;
57
+ const envelope = error as { code?: unknown; details?: unknown };
58
+ if (envelope.code !== "failed_precondition" || !Array.isArray(envelope.details)) return undefined;
59
+ for (const entry of envelope.details.slice(0, 8)) {
60
+ if (!entry || typeof entry !== "object") continue;
61
+ const { type, value } = entry as { type?: unknown; value?: unknown };
62
+ if (type !== "aiserver.v1.ErrorDetails" || typeof value !== "string"
63
+ || value.length === 0 || value.length > MAX_VALUE_CHARS
64
+ || !/^[A-Za-z0-9+/]+={0,2}$/.test(value)) continue;
65
+ try {
66
+ const bytes = Buffer.from(value, "base64");
67
+ if (bytes.toString("base64").replace(/=+$/, "") !== value.replace(/=+$/, "")) continue;
68
+ if (isFablePolicyError(bytes)) {
69
+ // Code-owned copy cannot inject credentials or alter downstream keyword classification.
70
+ return `${POLICY_TITLE}: ${POLICY_DETAIL} Review and accept using the same Cursor account at ${POLICY_REVIEW_URL}, then retry.`;
71
+ }
72
+ } catch { /* Unknown/malformed details retain the existing generic Connect error. */ }
73
+ }
74
+ return undefined;
75
+ }
@@ -325,7 +325,12 @@ function rootPromptMessages(
325
325
  const replacement = rootBlobCandidate(
326
326
  { role: payload.role, content: [{ type: "text", text: marked }] },
327
327
  role,
328
- { ...opts, messageIndex: previous.entry.messageIndex ?? opts.messageIndex },
328
+ // `text` must mirror the payload actually stored, not the unmarked text it was built from.
329
+ // It did not, and every consumer that rebuilds a root from `text` therefore dropped the run
330
+ // note: truncating a collapsed root silently deleted the "produced N times" line, and so did
331
+ // the invocation-argument restoration below. The note is the repetition breaker's per-entry
332
+ // half, so losing it re-primes the self-reinforcing loop the breaker exists to end.
333
+ { ...opts, text: marked, messageIndex: previous.entry.messageIndex ?? opts.messageIndex },
329
334
  );
330
335
  entries[entries.indexOf(previous.entry)] = replacement;
331
336
  replayRuns.set(role, { text: normalized, entry: replacement, length: runLength });
@@ -694,6 +699,20 @@ function rootPromptMessages(
694
699
  historyMessageStart = firstKept?.messageIndex ?? (messages.length);
695
700
  }
696
701
 
702
+ // Refund envelope bytes the assembled set left unused to invocation arguments the per-call cap
703
+ // clipped. Gated on `echoToolResultInRoot`, not `externalModel`: native `composer-2.5` echoes its
704
+ // results into roots without being an external wire model, so the narrower gate would leave the one
705
+ // native model that has clipped invocation lines capped for no reason (#4516).
706
+ if (echoToolResultInRoot && replayedCalls) {
707
+ selected = restoreClippedInvocationArguments(
708
+ selected,
709
+ messages,
710
+ replayedCalls,
711
+ knownCallsOffset,
712
+ carriedRoots.byteLength,
713
+ );
714
+ }
715
+
697
716
  return {
698
717
  ids: selected.map(entry => storeCursorBlob(entry.data, requestScope)),
699
718
  byteLength: selected.reduce((sum, entry) => sum + entry.byteLength, 0),
@@ -967,6 +986,91 @@ function toolInvocationLine(call: Extract<OcxAssistantContentPart, { type: "tool
967
986
  return `invoked: ${namespacedToolName(call.namespace, call.name)} with ${toolCallArgumentsText(call.arguments)}`;
968
987
  }
969
988
 
989
+ /**
990
+ * Second pass over the assembled root set: spend envelope bytes nothing else claimed on invocation
991
+ * arguments the per-call cap clipped.
992
+ *
993
+ * `CURSOR_INVOCATION_ARGUMENTS_BYTE_LIMIT` is charged while the envelope is still being built, so it
994
+ * costs a call 2 KiB whether or not anything else wants those bytes. In a small replay nearly the
995
+ * whole 192-root / 512 KiB envelope goes unused and the cap still bites: a 4,693-byte successful
996
+ * `write_file` lost its tail inside a 6,011-byte replay, and because the result text does not repeat
997
+ * the argument, the model could no longer see what it had just written (#4516).
998
+ *
999
+ * The cap stays, and admission is still decided on its 2 KiB prefix — it is what keeps a 600 KiB
1000
+ * argument from evicting the output it describes. This pass only refunds leftover aggregate bytes,
1001
+ * after every pruning and truncation decision is already final:
1002
+ *
1003
+ * - newest `toolResult` first, because the argument the model is most likely to still need is the
1004
+ * one belonging to the call it just made;
1005
+ * - only out of `spare`, so restoring can never push the envelope past its own limit;
1006
+ * - never for an `outputElided` root, whose own output is already gone — widening the invocation
1007
+ * there would spend the last free bytes describing an answer that is not present;
1008
+ * this guard is load bearing, and it is not reachable the obvious way. Truncation undershoots
1009
+ * its own budget by ~28 bytes, far less than a restoration costs, so a root that was merely
1010
+ * truncated cannot pay. What pays is initiator recovery: after the equal-share pass elides a
1011
+ * trailing run, recovery drops an elided sibling to fit the user turn, and the bytes it frees
1012
+ * become spare. It needs the share to land in a narrow window — wide enough that the clipped
1013
+ * invocation line survives, narrow enough that `output:` does not — and outside it the
1014
+ * clipped-line lookup below declines the root first. `the skip refuses to widen an elided root
1015
+ * even when spare would pay` pins a measured instance;
1016
+ * - never by dropping, shrinking or reordering another root, so nothing pruning chose to keep is
1017
+ * evicted to pay for a wider invocation line.
1018
+ */
1019
+ function restoreClippedInvocationArguments(
1020
+ selected: RootBlobCandidate[],
1021
+ messages: readonly OcxMessage[],
1022
+ replayedCalls: Map<string, Extract<OcxAssistantContentPart, { type: "toolCall" }>>,
1023
+ knownCallsOffset: number,
1024
+ carriedBytes: number,
1025
+ ): RootBlobCandidate[] {
1026
+ let spare = CURSOR_EXTERNAL_ROOT_BYTE_LIMIT
1027
+ - carriedBytes
1028
+ - selected.reduce((sum, entry) => sum + entry.byteLength, 0);
1029
+ if (spare <= 0) return selected;
1030
+ const restored = [...selected];
1031
+ for (let i = restored.length - 1; i >= 0 && spare > 0; i--) {
1032
+ const entry = restored[i];
1033
+ if (!entry || entry.role !== "toolResult" || entry.outputElided === true) continue;
1034
+ if (entry.text === undefined || entry.messageIndex === undefined) continue;
1035
+ const message = messages[entry.messageIndex];
1036
+ if (message?.role !== "toolResult") continue;
1037
+ // Same full-history bound the envelope builder used: `messageIndex` is local to this call's
1038
+ // `rawMessages`, and `knownCallsOffset` re-bases it when only a suffix is replayed.
1039
+ const call = callBefore(
1040
+ replayedCalls,
1041
+ decodeCursorCallId(message.toolCallId),
1042
+ knownCallsOffset + entry.messageIndex,
1043
+ );
1044
+ if (!call) continue;
1045
+ const full = serializeToolCallArguments(call.arguments);
1046
+ if (full === undefined) continue;
1047
+ const clipped = toolCallArgumentsText(call.arguments);
1048
+ if (clipped === full) continue;
1049
+ const name = namespacedToolName(call.namespace, call.name);
1050
+ // Anchored on the preceding newline. `toolResultToText` always emits the invocation after the
1051
+ // `[tool_result]`, `call_id:` and `name:` lines, so the real line is never first — and
1052
+ // `name:` renders the RESULT's tool name, which nothing sanitizes, so an unanchored search could
1053
+ // be satisfied by a crafted tool name and rewrite that header instead of the invocation.
1054
+ const clippedLine = `\ninvoked: ${name} with ${clipped}`;
1055
+ // Absent when truncation already cut through the invocation line itself; there is nothing to
1056
+ // widen in that root, and re-rendering the envelope would undo the output truncation too.
1057
+ if (!entry.text.includes(clippedLine)) continue;
1058
+ // Callback replacement: serialized arguments routinely contain `$&`, `$'` and `$1`, and the
1059
+ // string form of `replace` expands those into the surrounding match instead of inserting them.
1060
+ const widened = entry.text.replace(clippedLine, () => `\ninvoked: ${name} with ${full}`);
1061
+ const candidate = rootBlobCandidate(
1062
+ toolResultRootPayload(widened),
1063
+ "toolResult",
1064
+ { messageIndex: entry.messageIndex, text: widened },
1065
+ );
1066
+ const cost = candidate.byteLength - entry.byteLength;
1067
+ if (cost <= 0 || cost > spare) continue;
1068
+ restored[i] = candidate;
1069
+ spare -= cost;
1070
+ }
1071
+ return restored;
1072
+ }
1073
+
970
1074
  /**
971
1075
  * History position of each indexed call, keyed by the map `toolCallsByCallId` returned.
972
1076
  *
@@ -21,8 +21,10 @@
21
21
  * drift) we silently fall back to the chat path so a transient catalog
22
22
  * outage can't take chat down with it.
23
23
  *
24
- * Schema (verified against the bundled `extension.js`,
25
- * `exa.codeium_common_pb.ClientModelConfig`):
24
+ * Schema (#1/#4/#22 verified against the bundled `extension.js`,
25
+ * `exa.codeium_common_pb.ClientModelConfig`; #18 identified from a live
26
+ * catalog dump against vendor-known windows; #5 corroborated against the
27
+ * public WindsurfAPI `ClientModelConfig` documentation):
26
28
  *
27
29
  * GetCascadeModelConfigsResponse {
28
30
  * #1 client_model_configs: repeated ClientModelConfig
@@ -30,6 +32,8 @@
30
32
  * ClientModelConfig {
31
33
  * #1 label string
32
34
  * #4 disabled bool ← the gate this module reads
35
+ * #5 supports_images bool ← tri-state: absent stays unknown
36
+ * #18 max_input_tokens varint ← per-account context window
33
37
  * #22 model_uid string ← what `GetChatMessage` accepts
34
38
  * }
35
39
  *
@@ -78,6 +82,15 @@ export interface ModelCatalogEntry {
78
82
  * degrades: the caller keeps its static fallback instead of reporting zero.
79
83
  */
80
84
  contextWindow?: number;
85
+ /**
86
+ * Image-input support from `ClientModelConfig` field #5, kept as a
87
+ * tri-state: a present `true` asserts text+image support, a present
88
+ * `false` asserts text-only, and an OMITTED field stays `undefined`
89
+ * (unknown). Deliberately unlike `disabled`, which defaults to false —
90
+ * collapsing "never asserted" into "text-only" was the #1796 regression
91
+ * (see src/providers/antigravity-models.ts).
92
+ */
93
+ supportsImages?: boolean;
81
94
  }
82
95
 
83
96
  export interface CacheEntry {
@@ -113,12 +126,17 @@ export function parseCatalogBuffer(buf: Buffer, apiKey: string, host: string): C
113
126
  let modelUid = '';
114
127
  let disabled = false;
115
128
  let contextWindow = 0;
129
+ let supportsImages: boolean | undefined;
116
130
  for (const sf of iterFields(f.value as Buffer)) {
117
131
  if (sf.num === 1 && sf.wire === 2 && Buffer.isBuffer(sf.value)) {
118
132
  label = (sf.value as Buffer).toString('utf8');
119
133
  } else if (sf.num === 4 && sf.wire === 0) {
120
134
  // #4 = disabled (bool, varint 0/1)
121
135
  disabled = sf.value === 1n;
136
+ } else if (sf.num === 5 && sf.wire === 0) {
137
+ // #5 = supportsImages (bool, varint 0/1). Absent stays unknown — see
138
+ // ModelCatalogEntry; do not default it like disabled.
139
+ supportsImages = sf.value === 1n;
122
140
  } else if (sf.num === 18 && sf.wire === 0) {
123
141
  // #18 = max input tokens. Identified by dumping a live catalog and
124
142
  // reading the varints back against models whose windows are known from
@@ -135,6 +153,7 @@ export function parseCatalogBuffer(buf: Buffer, apiKey: string, host: string): C
135
153
  label: label || modelUid,
136
154
  disabled,
137
155
  ...(contextWindow > 0 ? { contextWindow } : {}),
156
+ ...(supportsImages !== undefined ? { supportsImages } : {}),
138
157
  });
139
158
  }
140
159
  }
@@ -278,6 +297,19 @@ export function clearCachedCatalog(): void {
278
297
  cacheEpoch++;
279
298
  }
280
299
 
300
+ /**
301
+ * Test seam: install a catalog as the live cache entry. Mirrors
302
+ * clearCachedCatalog's invalidation — the in-flight slot is dropped and the
303
+ * epoch bumped — so a fetch racing the seed cannot overwrite it, and a null
304
+ * entry resets the cache between tests.
305
+ */
306
+ export function setCachedCatalogForTests(entry: CacheEntry | null): void {
307
+ cached = entry;
308
+ inFlight = null;
309
+ inFlightKey = null;
310
+ cacheEpoch++;
311
+ }
312
+
281
313
  /**
282
314
  * Tier-disabled error — thrown by the chat pre-flight when the catalog lists
283
315
  * a model as `disabled: true` for this account. The message names the model
@@ -139,7 +139,13 @@ export const DEVIN_MODEL_EFFORTS: Record<string, string[]> = {
139
139
  export const DEVIN_DEFAULT_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
140
140
 
141
141
  export type DevinUsableModelsResult =
142
- | { ok: true; models: string[]; contextWindows: Record<string, number>; efforts: Record<string, string[]> }
142
+ | {
143
+ ok: true;
144
+ models: string[];
145
+ contextWindows: Record<string, number>;
146
+ efforts: Record<string, string[]>;
147
+ inputModalities: Record<string, string[]>;
148
+ }
143
149
  | { ok: false; error: "auth" | "http" | "empty" | "unknown"; detail?: string };
144
150
 
145
151
  /**
@@ -160,6 +166,8 @@ export async function fetchDevinUsableModels(opts: {
160
166
  const contextWindows: Record<string, number> = {};
161
167
  // Effort rungs per base, recovered from the suffixes the collapse strips.
162
168
  const rungs = new Map<string, Set<string>>();
169
+ // supportsImages votes per base; only rows that asserted field #5 vote.
170
+ const imageVotes = new Map<string, { sawTrue: boolean; sawFalse: boolean }>();
163
171
  for (const entry of catalog.byUid.values()) {
164
172
  if (entry.disabled) continue;
165
173
  // Skip internal enum constants (e.g. MODEL_GPT_5_2_LOW, MODEL_PRIVATE_*).
@@ -183,6 +191,15 @@ export async function fetchDevinUsableModels(opts: {
183
191
  const seen = contextWindows[base];
184
192
  contextWindows[base] = seen === undefined ? entry.contextWindow : Math.min(seen, entry.contextWindow);
185
193
  }
194
+ if (entry.supportsImages !== undefined) {
195
+ let votes = imageVotes.get(base);
196
+ if (!votes) {
197
+ votes = { sawTrue: false, sawFalse: false };
198
+ imageVotes.set(base, votes);
199
+ }
200
+ if (entry.supportsImages) votes.sawTrue = true;
201
+ else votes.sawFalse = true;
202
+ }
186
203
  }
187
204
  if (bases.size === 0) return { ok: false, error: "empty" };
188
205
  const efforts: Record<string, string[]> = {};
@@ -191,7 +208,21 @@ export async function fetchDevinUsableModels(opts: {
191
208
  // would draw a picker whose only option is the value already in effect.
192
209
  if (set.size > 1) efforts[base] = sortDevinRungs(set);
193
210
  }
194
- return { ok: true, models: [...bases].sort(), contextWindows, efforts };
211
+ // supportsImages arrives tri-state per catalog row, so the collapse votes:
212
+ // a row that never asserted field #5 abstains, which keeps an unsuffixed
213
+ // unknown row from poisoning a base whose effort variants were measured
214
+ // image-capable. Unanimous measured rows advertise; measured disagreement
215
+ // advertises nothing, because a single measured false is not outvoted by
216
+ // its siblings. One accepted mismatch: resolveWireModelUid prefers the
217
+ // plain UID when the catalog lists it, so a base advertised
218
+ // ["text","image"] on variant evidence can still route a no-effort request
219
+ // to a plain row that never asserted the field.
220
+ const inputModalities: Record<string, string[]> = {};
221
+ for (const [base, votes] of imageVotes) {
222
+ if (votes.sawTrue && votes.sawFalse) continue;
223
+ inputModalities[base] = votes.sawTrue ? ["text", "image"] : ["text"];
224
+ }
225
+ return { ok: true, models: [...bases].sort(), contextWindows, efforts, inputModalities };
195
226
  } catch (error) {
196
227
  const message = error instanceof Error ? error.message : String(error);
197
228
  if (/unauth|401|invalid token|login/i.test(message)) return { ok: false, error: "auth", detail: message };
@@ -149,6 +149,14 @@ function compileGenerationConfig(value: unknown): JsonObject | undefined {
149
149
  const valid = value.responseModalities.filter((m): m is string => typeof m === "string" && ["TEXT", "IMAGE", "AUDIO"].includes(m));
150
150
  if (valid.length > 0) out.responseModalities = valid;
151
151
  }
152
+ // Structured output. This compiler is a whitelist, so without these two the adapter
153
+ // could set a schema and it would still be dropped before the wire.
154
+ if (typeof value.responseMimeType === "string" && value.responseMimeType.length > 0) {
155
+ out.responseMimeType = value.responseMimeType;
156
+ }
157
+ // Carried through unmodified: a caller-authored output schema is not a tool
158
+ // declaration, so sanitizeGeminiToolParameters must not touch it.
159
+ if (isObject(value.responseJsonSchema)) out.responseJsonSchema = value.responseJsonSchema;
152
160
  return Object.keys(out).length > 0 ? out : undefined;
153
161
  }
154
162
 
@@ -789,6 +789,37 @@ export function createGoogleAdapter(provider: OcxProviderConfig): ProviderAdapte
789
789
  : {}),
790
790
 
791
791
  async buildRequest(parsed: OcxParsedRequest) {
792
+ // Structured-output admission runs FIRST, before messagesToGeminiFormat writes
793
+ // lastInjectedCallIds/lastReasoningReplayScope: a refused request must not leave
794
+ // adapter-scoped replay state pointing at call ids that never went out. These
795
+ // refusals are local and precede any fetch, and carry no request content, schema
796
+ // body, URL or credential.
797
+ const requestedTextFormat = parsed.options.textFormat;
798
+ if (requestedTextFormat) {
799
+ if (provider.googleMode === "cloud-code-assist") {
800
+ // Not implemented or verified by opencodex for the Cloud Code Assist envelope,
801
+ // including Claude models served through it. This is not a claim that the
802
+ // upstream cannot do it — silence would return unconstrained prose as success,
803
+ // which is the failure this fix exists to remove.
804
+ throw new Error(
805
+ "google cloud-code-assist structured output is not implemented by opencodex — "
806
+ + "remove response_format or route this model through AI Studio or Vertex",
807
+ );
808
+ }
809
+ if (isImageCapableModel(parsed.modelId)) {
810
+ // An image-output model is configured with responseModalities; constraining the
811
+ // same turn to JSON text is contradictory. Say so rather than dropping the schema.
812
+ throw new Error(
813
+ "google image-capable models cannot combine image output with structured output — "
814
+ + "remove response_format or select a text model",
815
+ );
816
+ }
817
+ if (requestedTextFormat.type === "json_schema" && !requestedTextFormat.schema) {
818
+ // Downgrading a malformed json_schema to bare JSON mode would silently drop the
819
+ // constraint the caller asked for.
820
+ throw new Error("google structured output requires text.format.schema for type json_schema");
821
+ }
822
+ }
792
823
  const routedModelId = provider.googleMode === "cloud-code-assist"
793
824
  ? resolveAntigravityEffortWireModel(
794
825
  parsed.modelId,
@@ -846,6 +877,21 @@ export function createGoogleAdapter(provider: OcxProviderConfig): ProviderAdapte
846
877
  if (!generationConfig.thinkingConfig && isImageCapableModel(parsed.modelId)) {
847
878
  generationConfig.responseModalities = ["TEXT", "IMAGE"];
848
879
  }
880
+ // Structured output travels in generationConfig on generateContent itself.
881
+ // responseJsonSchema takes ordinary JSON Schema (lowercase types), which is what
882
+ // options.textFormat.schema already holds; responseSchema would require Gemini's
883
+ // uppercase typed Schema form, and the docs require omitting it when
884
+ // responseJsonSchema is used. The response type does not change — the model
885
+ // returns text containing the conforming JSON — so response parsing is untouched.
886
+ // The tool-parameter sanitizer is deliberately NOT applied: it narrows a schema
887
+ // to the function-declaration subset and would corrupt a valid output schema.
888
+ const textFormat = parsed.options.textFormat;
889
+ if (textFormat) {
890
+ generationConfig.responseMimeType = "application/json";
891
+ if (textFormat.type === "json_schema" && textFormat.schema) {
892
+ generationConfig.responseJsonSchema = textFormat.schema;
893
+ }
894
+ }
849
895
  if (Object.keys(generationConfig).length > 0) body.generationConfig = generationConfig;
850
896
 
851
897
  const method = parsed.stream ? "streamGenerateContent" : "generateContent";
@@ -0,0 +1,45 @@
1
+ import type { ProviderAdapter } from "./base";
2
+ import { untranslatedInputMediaMessage, untranslatedResponsesInputMedia } from "../responses/input-media";
3
+
4
+ /**
5
+ * Refuse unrepresentable input at the final translated-adapter boundary. The registry
6
+ * applies this after wire resolution; Responses passthrough (including Azure) opts
7
+ * out because it uses the original body rather than the lossy normalized content.
8
+ */
9
+ export function withInputMediaGuard<T extends ProviderAdapter>(adapter: T): T {
10
+ const build = adapter.buildRequest.bind(adapter);
11
+ adapter.buildRequest = (parsed, incoming) => {
12
+ const kind = untranslatedResponsesInputMedia(parsed._rawBody);
13
+ if (kind) throw new Error(untranslatedInputMediaMessage(kind));
14
+ return build(parsed, incoming);
15
+ };
16
+
17
+ const runTurn = adapter.runTurn?.bind(adapter);
18
+ if (runTurn) {
19
+ adapter.runTurn = async (parsed, incoming, emit) => {
20
+ const kind = untranslatedResponsesInputMedia(parsed._rawBody);
21
+ if (kind) {
22
+ emit({
23
+ type: "error",
24
+ status: 400,
25
+ errorType: "invalid_request_error",
26
+ code: "unsupported_input_modality",
27
+ retryable: false,
28
+ message: untranslatedInputMediaMessage(kind),
29
+ });
30
+ return;
31
+ }
32
+ await runTurn(parsed, incoming, emit);
33
+ };
34
+ }
35
+
36
+ const localTerminal = adapter.localTerminal?.bind(adapter);
37
+ if (localTerminal) {
38
+ // This hook is outside the builder's error catch. Decline its success shortcut;
39
+ // the ordinary buildRequest path then returns the established client-safe 400.
40
+ adapter.localTerminal = parsed => untranslatedResponsesInputMedia(parsed._rawBody)
41
+ ? undefined
42
+ : localTerminal(parsed);
43
+ }
44
+ return adapter;
45
+ }
@@ -13,6 +13,7 @@ import type {
13
13
  } from "../../types";
14
14
  import type { ProviderAdapter } from "../base";
15
15
  import type { AdapterFetchContext, AdapterRequest } from "../base";
16
+ import type { RequestExecutionBudget } from "../../lib/request-execution-budget";
16
17
  import { safeKiroHttpErrorMessage } from "../kiro-errors";
17
18
  import { calibrateKiroEstimate } from "../kiro-calibration";
18
19
  import { normalizeKiroImages } from "../kiro-images";
@@ -58,6 +59,9 @@ export function createKiroAdapter(provider: OcxProviderConfig): ProviderAdapter
58
59
  let requestSnapshot: OcxParsedRequest | undefined;
59
60
  let firstRequestBodyBytes = 0;
60
61
  let requestAbortSignal: AbortSignal | undefined;
62
+ // Captured the same way as the abort signal, because the text-fallback rebuild below runs
63
+ // outside the fetchResponse frame and used to construct a context without either (#4546).
64
+ let requestSendBudget: RequestExecutionBudget | undefined;
61
65
 
62
66
  const build = async (
63
67
  parsed: OcxParsedRequest,
@@ -208,6 +212,9 @@ export function createKiroAdapter(provider: OcxProviderConfig): ProviderAdapter
208
212
  abortSignal: requestAbortSignal,
209
213
  returnRawErrors: true,
210
214
  stream: true,
215
+ // The text-fallback rebuild used to construct a fresh context and drop the budget,
216
+ // so everything after the first send escaped the per-request cap.
217
+ ...(requestSendBudget ? { sendBudget: requestSendBudget } : {}),
211
218
  });
212
219
  return {
213
220
  response,
@@ -278,6 +285,7 @@ export function createKiroAdapter(provider: OcxProviderConfig): ProviderAdapter
278
285
  // Keep it for the adapter-owned bounded continuation so cancelling the client turn aborts
279
286
  // both the first Kiro request and its one allowed completion retry.
280
287
  if (ctx?.abortSignal) requestAbortSignal = ctx.abortSignal;
288
+ if (ctx?.sendBudget) requestSendBudget = ctx.sendBudget;
281
289
  return fetchKiroWithRetry(request, ctx);
282
290
  },
283
291
 
@@ -22,7 +22,12 @@ import {
22
22
  } from "../kiro-constants";
23
23
  import { EMPTY_EXEC_OUTPUT_MESSAGE, annotateCodeModeHostFailure, normalizeEmptyExecToolResultText } from "../exec-tool-result-normalize";
24
24
  import { identifyRoutedModel } from "../identity";
25
- import { extractKiroImages, type KiroImage } from "../kiro-images";
25
+ import {
26
+ countKiroUninlinableImages,
27
+ extractKiroImages,
28
+ kiroUninlinableImageMarker,
29
+ type KiroImage,
30
+ } from "../kiro-images";
26
31
  import { convertKiroToolContext } from "../kiro-tools";
27
32
  import { createKiroToolNameRegistry, mapModelId, normalizeToolId, stableConversationId } from "../kiro-wire";
28
33
  import { buildNonOpenAIToolCatalogNudgeFromNames, isBareShellBridgeTool, isCodexCodeModeExecTool } from "../tool-catalog-nudge";
@@ -233,9 +238,13 @@ export function buildKiroPayload(
233
238
  // Original-message adjacency matters even when a turn is collapsed or skipped below.
234
239
  if (msg.role !== "toolResult") finishAdjacentResult();
235
240
  if (msg.role === "user" || msg.role === "developer") {
236
- const text = userContentText((msg as { content: string | OcxContentPart[] }).content);
237
- const images = extractKiroImages((msg as { content: string | OcxContentPart[] }).content);
238
- pushUser(text, images);
241
+ const content = (msg as { content: string | OcxContentPart[] }).content;
242
+ const images = extractKiroImages(content);
243
+ // Kiro inlines base64 bytes only. A remote reference used to vanish with neither
244
+ // bytes nor a trace; attach a bounded, URL-free marker so the loss is visible.
245
+ const marker = kiroUninlinableImageMarker(countKiroUninlinableImages(content));
246
+ const text = userContentText(content);
247
+ pushUser(marker ? (text ? text + "\n" + marker : marker) : text, images);
239
248
  } else if (msg.role === "assistant") {
240
249
  const aMsg = msg as OcxAssistantMessage;
241
250
  const text = (aMsg.content || [])
@@ -281,7 +290,15 @@ export function buildKiroPayload(
281
290
  const annotatedExecText = normalizedExecText === undefined && codeModeExecName !== undefined
282
291
  ? annotateCodeModeHostFailure(text, execOptions)
283
292
  : undefined;
284
- const resultText = normalizedExecText ?? annotatedExecText ?? (text.trim() ? text : KIRO_EMPTY_TOOL_RESULT_MESSAGE);
293
+ const uninlinableMarker = kiroUninlinableImageMarker(countKiroUninlinableImages(tr.content));
294
+ // Appended to the SELECTED result text, not to `text`: when an exec normalization
295
+ // fires, resultText below takes normalizedExecText/annotatedExecText instead, and
296
+ // a marker attached to `text` would be dropped — reinstating the silent loss this
297
+ // exists to remove.
298
+ const chosenText = normalizedExecText ?? annotatedExecText ?? (text.trim() ? text : KIRO_EMPTY_TOOL_RESULT_MESSAGE);
299
+ const resultText = uninlinableMarker
300
+ ? (chosenText ? chosenText + "\n" + uninlinableMarker : uninlinableMarker)
301
+ : chosenText;
285
302
  const images = extractKiroImages(tr.content);
286
303
  const toolUseId = normalizeToolId(tr.toolCallId);
287
304
  const call = priorCalls.get(toolUseId);
@@ -289,8 +306,13 @@ export function buildKiroPayload(
289
306
  throw new Error(`Kiro history contains an orphaned tool result for call ${JSON.stringify(tr.toolCallId)}`);
290
307
  }
291
308
  // Keep real whitespace and failed wrappers, but no empty-success wrapper boilerplate.
292
- const rawGroupText = text.length > 0 && (!text.trim() || normalizedExecText !== EMPTY_EXEC_OUTPUT_MESSAGE)
309
+ const rawGroupBase = text.length > 0 && (!text.trim() || normalizedExecText !== EMPTY_EXEC_OUTPUT_MESSAGE)
293
310
  ? (annotatedExecText ?? text) : undefined;
311
+ // The grouping path rebuilds a collapsed turn's content from these texts, so the
312
+ // marker has to ride along here too or an adjacent-result turn loses it.
313
+ const rawGroupText = uninlinableMarker
314
+ ? (rawGroupBase ? rawGroupBase + "\n" + uninlinableMarker : uninlinableMarker)
315
+ : rawGroupBase;
294
316
  const last = turns.at(-1);
295
317
  if (
296
318
  adjacentResult?.rawId === tr.toolCallId