@bitkyc08/opencodex 2.52.0-preview.20260911 → 2.52.0-preview.20260912

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. package/gui/dist/assets/index-D_t6sCWs.js +115 -0
  2. package/gui/dist/assets/index-EdoPnm9_.css +1 -0
  3. package/gui/dist/index.html +2 -2
  4. package/gui/dist/provider-icons/devin.svg +49 -0
  5. package/gui/dist/provider-icons/omo.svg +42 -0
  6. package/package.json +3 -1
  7. package/src/AGENTS.md +1 -1
  8. package/src/adapters/cline-pass-deepseek-v4-tool-replay.ts +0 -1
  9. package/src/adapters/command-code.ts +0 -1
  10. package/src/adapters/cursor/checkpoint-store.ts +37 -0
  11. package/src/adapters/cursor/live-transport.ts +74 -2
  12. package/src/adapters/cursor.ts +6 -0
  13. package/src/adapters/devin/cloud-direct/auth.ts +264 -0
  14. package/src/adapters/devin/cloud-direct/catalog.ts +306 -0
  15. package/src/adapters/devin/cloud-direct/chat.ts +1274 -0
  16. package/src/adapters/devin/cloud-direct/index.ts +65 -0
  17. package/src/adapters/devin/cloud-direct/metadata.ts +134 -0
  18. package/src/adapters/devin/cloud-direct/wire.ts +206 -0
  19. package/src/adapters/devin/live-models.ts +133 -0
  20. package/src/adapters/devin-cli/acp.ts +204 -0
  21. package/src/adapters/devin-cli/adapter.ts +345 -0
  22. package/src/adapters/devin-cli/binary.ts +69 -0
  23. package/src/adapters/devin-cli/models.ts +57 -0
  24. package/src/adapters/devin.ts +326 -0
  25. package/src/adapters/google.ts +1 -1
  26. package/src/adapters/openai-chat.ts +31 -10
  27. package/src/adapters/openai-responses.ts +16 -1
  28. package/src/adapters/registry.ts +25 -1
  29. package/src/bridge.ts +8 -2
  30. package/src/claude/inbound-cache-stabilize.ts +130 -0
  31. package/src/claude/inbound.ts +45 -5
  32. package/src/cli/account-auth.ts +60 -1
  33. package/src/cli/account-extended.ts +23 -30
  34. package/src/cli/account.ts +2 -1
  35. package/src/cli/capabilities.ts +50 -8
  36. package/src/cli/config-command.ts +2 -2
  37. package/src/cli/dispatch.ts +14 -2
  38. package/src/cli/export-command.ts +11 -25
  39. package/src/cli/help.ts +2 -2
  40. package/src/cli/opencode.ts +5 -0
  41. package/src/cli/registry.ts +15 -4
  42. package/src/clients/config-export/cline.ts +71 -0
  43. package/src/clients/config-export/contracts.ts +10 -1
  44. package/src/clients/config-export/model-metadata.ts +33 -0
  45. package/src/clients/config-export/zcode.ts +23 -12
  46. package/src/clients/config-export.ts +156 -4
  47. package/src/codex/account-store.ts +7 -2
  48. package/src/codex/auth-context.ts +1 -1
  49. package/src/codex/catalog/effort.ts +4 -3
  50. package/src/codex/catalog/metadata.ts +8 -4
  51. package/src/codex/catalog/native-models.ts +24 -4
  52. package/src/codex/catalog/parsing.ts +19 -1
  53. package/src/codex/catalog/provider-fetch.ts +57 -0
  54. package/src/codex/catalog/sync.ts +1 -1
  55. package/src/codex/context-compat.ts +97 -0
  56. package/src/codex/context-owner.ts +201 -0
  57. package/src/codex/history-provider.ts +161 -14
  58. package/src/codex/inject-coordination.ts +3 -2
  59. package/src/codex/inject.ts +128 -9
  60. package/src/codex/pool-rotation.ts +8 -292
  61. package/src/codex/quota.ts +41 -12
  62. package/src/codex/retired-model-migration.ts +41 -0
  63. package/src/codex/routing.ts +327 -22
  64. package/src/codex/warmup.ts +2 -2
  65. package/src/combos/failover.ts +5 -0
  66. package/src/combos/resolve.ts +11 -15
  67. package/src/config.ts +46 -14
  68. package/src/generated/compatibility-version.json +293 -113
  69. package/src/grok/grpc-web.ts +120 -0
  70. package/src/grok/reset-coupon-ledger.ts +139 -0
  71. package/src/grok/reset-coupons.ts +278 -0
  72. package/src/integrations/catalog-refresh.ts +1 -1
  73. package/src/integrations/cline-document.ts +73 -0
  74. package/src/integrations/cline-io.ts +149 -0
  75. package/src/integrations/cline-transaction.ts +42 -0
  76. package/src/integrations/config-io.ts +21 -2
  77. package/src/integrations/journal.ts +43 -1
  78. package/src/integrations/omp-yaml-source.ts +123 -4
  79. package/src/integrations/ownership-policy.ts +7 -0
  80. package/src/integrations/ownership.ts +36 -0
  81. package/src/integrations/registry.ts +27 -0
  82. package/src/integrations/state.ts +5 -2
  83. package/src/integrations/store.ts +6 -0
  84. package/src/integrations/writer.ts +53 -9
  85. package/src/lab/subject/behavior-fingerprint.ts +1 -1
  86. package/src/lib/abort.ts +36 -0
  87. package/src/lib/local-destinations.ts +1 -1
  88. package/src/lib/upstream-retry.ts +36 -1
  89. package/src/oauth/account-quota-rank.ts +11 -0
  90. package/src/oauth/callback-server.ts +10 -5
  91. package/src/oauth/devin/api-base.ts +63 -0
  92. package/src/oauth/devin/login.ts +1 -0
  93. package/src/oauth/devin/register-user.ts +186 -0
  94. package/src/oauth/devin/types.ts +71 -0
  95. package/src/oauth/devin-cli.ts +149 -0
  96. package/src/oauth/devin.ts +170 -0
  97. package/src/oauth/generic-account-failover.ts +169 -1
  98. package/src/oauth/index.ts +20 -1
  99. package/src/oauth/login-cli.ts +19 -5
  100. package/src/oauth/pool-kernel.ts +321 -0
  101. package/src/oauth/pool-settings-capability.ts +127 -9
  102. package/src/oauth/store.ts +7 -3
  103. package/src/oauth/token-guardian.ts +1 -1
  104. package/src/providers/codebuddy-models.ts +0 -3
  105. package/src/providers/command-code-efforts.ts +23 -4
  106. package/src/providers/default-aliases.ts +4 -0
  107. package/src/providers/derive.ts +8 -0
  108. package/src/providers/devin-cli-authmode-migration.ts +57 -0
  109. package/src/providers/key-failover.ts +175 -0
  110. package/src/providers/model-rename-startup.ts +21 -1
  111. package/src/providers/qoder-models.ts +0 -1
  112. package/src/providers/quota-key-accounts.ts +50 -0
  113. package/src/providers/quota-routing-cache.ts +65 -6
  114. package/src/providers/quota-types.ts +7 -0
  115. package/src/providers/quota.ts +113 -30
  116. package/src/providers/registry.ts +237 -77
  117. package/src/providers/stale-context-window-migration.ts +92 -0
  118. package/src/providers/zai-responses-migration.ts +45 -0
  119. package/src/quota/reset-observer.ts +2 -1
  120. package/src/quota/reset-seen-store.ts +9 -1
  121. package/src/remote-control/crypto.ts +442 -0
  122. package/src/remote-control/host.ts +175 -0
  123. package/src/remote-control/index.ts +100 -0
  124. package/src/remote-control/protocol.ts +200 -0
  125. package/src/remote-control/relay.ts +162 -0
  126. package/src/remote-control/workspace-agent-protocol.ts +246 -0
  127. package/src/remote-control/workspace-rpc-framing.ts +128 -0
  128. package/src/remote-control/workspace-tools.ts +237 -0
  129. package/src/remote-control/workspace-utf8.ts +24 -0
  130. package/src/responses/custom-tool-compat.ts +23 -0
  131. package/src/router.ts +6 -1
  132. package/src/routing/compatibility/behavior.ts +3 -0
  133. package/src/server/adapter-resolve.ts +6 -2
  134. package/src/server/auth-cors.ts +71 -6
  135. package/src/server/chat-completions.ts +5 -3
  136. package/src/server/chat-native.ts +9 -0
  137. package/src/server/claude-messages.ts +25 -30
  138. package/src/server/context-history.ts +207 -0
  139. package/src/server/images.ts +28 -1
  140. package/src/server/index.ts +42 -22
  141. package/src/server/live.ts +2 -1
  142. package/src/server/management/agent-settings-routes.ts +7 -1
  143. package/src/server/management/config-routes.ts +3 -3
  144. package/src/server/management/grok-coupon-routes.ts +287 -0
  145. package/src/server/management/integration-routes.ts +6 -1
  146. package/src/server/management/oauth-account-routes.ts +124 -7
  147. package/src/server/management/provider-routes.ts +77 -2
  148. package/src/server/management/route-registry.ts +6 -1
  149. package/src/server/management-api.ts +7 -0
  150. package/src/server/request-log.ts +4 -0
  151. package/src/server/responses/codex-ws-exchange.ts +90 -14
  152. package/src/server/responses/codex-ws-wire.ts +51 -1
  153. package/src/server/responses/compact.ts +8 -1
  154. package/src/server/responses/core.ts +188 -22
  155. package/src/server/responses/ws-upstream.ts +1 -1
  156. package/src/server/responses-undeclared-tool-guard.ts +261 -12
  157. package/src/server/zai-responses-startup.ts +21 -0
  158. package/src/types/config.ts +33 -3
  159. package/src/types/provider.ts +54 -4
  160. package/src/types/request.ts +1 -1
  161. package/src/types/tools.ts +36 -6
  162. package/src/update/job.ts +33 -11
  163. package/src/vision/plan.ts +40 -37
  164. package/src/web-search/index.ts +1 -0
  165. package/gui/dist/assets/index-BoBRSehJ.css +0 -1
  166. package/gui/dist/assets/index-Dx0xv2EA.js +0 -115
@@ -0,0 +1,1274 @@
1
+ /*
2
+ * Derived from rsvedant/opencode-windsurf-auth (src/cloud-direct/), MIT licensed,
3
+ * Copyright (c) 2026 Vedant. The full notice is in ./index.ts.
4
+ */
5
+ /**
6
+ * Cloud-direct streaming chat. Talks to
7
+ * `server.codeium.com/exa.api_server_pb.ApiServerService/GetChatMessage`
8
+ * with no local language_server in the path. Returns an async iterable of
9
+ * CloudChatEvent deltas (text, reasoning, tool calls, usage, finish) so the
10
+ * caller can stream straight into opencodex's internal AdapterEvent model.
11
+ *
12
+ * What this supports:
13
+ * - Single- or multi-turn chat using the prompt-and-history pattern the LS
14
+ * uses (flatten history into one ChatMessagePrompt list)
15
+ * - All free Windsurf/Cognition models (swe-1-7, swe-1-7-lightning, etc.)
16
+ * and any model the user's api_key is entitled to
17
+ * - Streaming (uses Connect-streaming envelope, emits deltas as they arrive)
18
+ * - Tool definitions (encoded via `encodeToolDef`) and tool-call events
19
+ * (tool_call_start, tool_call_args) decoded from the response stream
20
+ * - Usage and finish-reason events for terminal completion
21
+ *
22
+ * Wire-protocol: Connect-RPC streaming over HTTPS with manual protobuf
23
+ * encoding (see `wire.ts`).
24
+ */
25
+
26
+ import * as crypto from 'crypto';
27
+ import * as zlib from 'zlib';
28
+ import {
29
+ encodeMessage,
30
+ encodeString,
31
+ encodeVarintField,
32
+ frameConnectStream,
33
+ iterFields,
34
+ parseConnectFrames,
35
+ } from './wire.js';
36
+ import { buildMetadata } from './metadata.js';
37
+ import { getCachedUserJwt } from './auth.js';
38
+ import { getCachedCatalog, ModelNotAvailableError } from './catalog.js';
39
+ import { anySignal, cancelBodyOnAbort } from '../../../lib/abort.js';
40
+ import { resolveDevinApiBaseUrl } from '../../../oauth/devin/api-base.js';
41
+
42
+ /**
43
+ * Connect-RPC streaming inactivity timeout. If the cloud sends zero bytes
44
+ * for this long after the last chunk, we abort the fetch. The cloud's own
45
+ * idle limit is around 90s on most models; we set ours a little above so
46
+ * we only trigger when the server has genuinely stopped responding.
47
+ */
48
+ const CLOUD_STREAM_IDLE_MS = 120_000;
49
+ /** Time-to-first-byte timeout. */
50
+ const CLOUD_STREAM_TTFB_MS = 60_000;
51
+ /** Maximum acceptable Connect-RPC frame length (16 MB). */
52
+ const MAX_FRAME_LEN = 16 * 1024 * 1024;
53
+
54
+ /**
55
+ * Per-(apiKey, host) session/cascade ID cache. Cloud uses these for
56
+ * server-side context caching across turns of the same conversation; if we
57
+ * mint a fresh sessionId on every call (which we used to), every turn looks
58
+ * like a brand-new session and the prompt-cache hit ratio is zero.
59
+ * Single-process scope is enough: opencode lives in one runtime for a TUI
60
+ * session, and CLI one-shots don't benefit from caching anyway.
61
+ */
62
+ interface SessionIds {
63
+ sessionId: string;
64
+ cascadeId: string;
65
+ }
66
+ /**
67
+ * Bounded the same way the adapter bounds its cascade-id map: a long-running
68
+ * proxy sees one entry per (host, api_key) pair, and nothing ever evicted them.
69
+ */
70
+ const SESSION_CACHE_MAX = 256;
71
+ const sessionCache = new Map<string, SessionIds>();
72
+ function getOrAllocateSessionIds(apiKey: string, host: string, cascadeIdOverride?: string): SessionIds {
73
+ const key = `${host}\x1f${apiKey}`;
74
+ let ids = sessionCache.get(key);
75
+ if (!ids) {
76
+ ids = {
77
+ sessionId: crypto.randomUUID(),
78
+ cascadeId: cascadeIdOverride ?? allocateCascadeId(),
79
+ };
80
+ if (sessionCache.size >= SESSION_CACHE_MAX) {
81
+ const oldest = sessionCache.keys().next().value;
82
+ if (oldest !== undefined) sessionCache.delete(oldest);
83
+ }
84
+ sessionCache.set(key, ids);
85
+ } else if (cascadeIdOverride && ids.cascadeId !== cascadeIdOverride) {
86
+ // Caller explicitly requested a different cascadeId — honor it.
87
+ ids = { sessionId: ids.sessionId, cascadeId: cascadeIdOverride };
88
+ sessionCache.set(key, ids);
89
+ }
90
+ return ids;
91
+ }
92
+
93
+ /** Drop the cached session IDs — call after logout so a new sign-in starts fresh. */
94
+ export function clearSessionIds(): void {
95
+ sessionCache.clear();
96
+ }
97
+
98
+ // ----------------------------------------------------------------------------
99
+ // Per-conversation cascade state — generated client-side; cloud lazy-registers
100
+ // ----------------------------------------------------------------------------
101
+
102
+ /**
103
+ * Allocate a fresh cascade UUID. The cloud lazy-registers cascade_id on first
104
+ * use — confirmed empirically (random UUID accepted, model responded). One
105
+ * cascade_id per opencode-CLI conversation is fine; reuse across turns to
106
+ * preserve server-side context.
107
+ */
108
+ export function allocateCascadeId(): string {
109
+ return crypto.randomUUID();
110
+ }
111
+
112
+ // ----------------------------------------------------------------------------
113
+ // Request encoders
114
+ // ----------------------------------------------------------------------------
115
+
116
+ /**
117
+ * ChatMessagePrompt {
118
+ * #2 source: enum CHAT_MESSAGE_SOURCE_USER=1 / ASSISTANT=2 / SYSTEM=3 / TOOL=4
119
+ * #3 prompt: string (text content)
120
+ * #4 num_tokens: int (rough estimate)
121
+ * #5 safe_for_code_telemetry: bool (1 = ok to log)
122
+ * #10 images: repeated ImageData (multimodal)
123
+ * }
124
+ *
125
+ * ImageData (exa.codeium_common_pb.ImageData) {
126
+ * #1 base64_data: string
127
+ * #2 mime_type: string
128
+ * #3 caption: string (optional)
129
+ * }
130
+ */
131
+ function encodeImageData(img: { mimeType: string; base64Data: string; caption?: string }): Buffer {
132
+ const parts: Buffer[] = [
133
+ encodeString(1, img.base64Data),
134
+ encodeString(2, img.mimeType),
135
+ ];
136
+ if (img.caption) parts.push(encodeString(3, img.caption));
137
+ return Buffer.concat(parts);
138
+ }
139
+
140
+ /**
141
+ * Encode one ChatToolCall sub-message:
142
+ * {#1 id, #2 name, #3 arguments_json}
143
+ * Verified against `exa.codeium_common_pb.ChatToolCall` from extension.js.
144
+ */
145
+ function encodeChatToolCall(tc: { id: string; name: string; arguments: string }): Buffer {
146
+ return Buffer.concat([
147
+ encodeString(1, tc.id),
148
+ encodeString(2, tc.name),
149
+ encodeString(3, tc.arguments),
150
+ ]);
151
+ }
152
+
153
+ function encodeChatMessagePrompt(
154
+ content: ContentPart[],
155
+ source: number,
156
+ opts?: { toolCallId?: string; toolCalls?: Array<{ id: string; name: string; arguments: string }> },
157
+ ): Buffer {
158
+ const textParts = content.filter((p): p is { type: 'text'; text: string } => p.type === 'text');
159
+ const imageParts = content.filter((p): p is { type: 'image'; mimeType: string; base64Data: string; caption?: string } => p.type === 'image');
160
+ const joined = textParts.map((p) => p.text).join('\n');
161
+ const parts: Buffer[] = [
162
+ // #1 message_id. The verified turn-1 capture stamps one on every prompt.
163
+ encodeString(1, crypto.randomUUID()),
164
+ encodeVarintField(2, source),
165
+ encodeString(3, joined),
166
+ ];
167
+ // Tool-result message: attach the id of the call this result answers.
168
+ // Without it, the model can't pair multi-tool conversations.
169
+ if (opts?.toolCallId) {
170
+ parts.push(encodeString(7, opts.toolCallId));
171
+ }
172
+ // Assistant message with tool_calls: encode each as a ChatToolCall.
173
+ if (opts?.toolCalls && opts.toolCalls.length > 0) {
174
+ for (const tc of opts.toolCalls) {
175
+ parts.push(encodeMessage(6, encodeChatToolCall(tc)));
176
+ }
177
+ }
178
+ for (const img of imageParts) {
179
+ parts.push(encodeMessage(10, encodeImageData(img)));
180
+ }
181
+ return Buffer.concat(parts);
182
+ }
183
+
184
+ const SOURCE_BY_ROLE: Record<string, number> = {
185
+ user: 1,
186
+ assistant: 2,
187
+ // NOTE: do not send source=3 (SYSTEM) directly — the Codeium chat backend
188
+ // returns "third-party model provider is experiencing issues" when any
189
+ // ChatMessagePrompt has source=SYSTEM. The captured LS upstream traffic
190
+ // shows the IDE inlines system context into the *user* prompt (source=1)
191
+ // wrapped in <additional_metadata>...</additional_metadata>. We collapse
192
+ // role:'system' messages into the next user turn before building the
193
+ // proto — see `collapseSystemIntoUser` below.
194
+ system: 1,
195
+ tool: 4,
196
+ };
197
+
198
+ /**
199
+ * Collapse OpenAI-style messages so all `role:'system'` entries are inlined
200
+ * into the immediately-following user message, matching the wire format the
201
+ * IDE uses. Cognition's chat backend rejects raw role=system entries.
202
+ *
203
+ * [{system: "S1"}, {system: "S2"}, {user: "U1"}, {assistant: "A1"}, {user: "U2"}]
204
+ *
205
+ * becomes
206
+ *
207
+ * [{user: "<system>\nS1\nS2\n</system>\nU1"}, {assistant: "A1"}, {user: "U2"}]
208
+ *
209
+ * If there's no following user message, the trailing system messages get
210
+ * appended as a synthesized user turn.
211
+ */
212
+ function collapseSystemIntoUser(messages: ChatHistoryItem[]): ChatHistoryItem[] {
213
+ const out: ChatHistoryItem[] = [];
214
+ let pendingSystem: string[] = [];
215
+
216
+ const flushTextOf = (content: ContentPart[]): string =>
217
+ content.filter((p): p is { type: 'text'; text: string } => p.type === 'text')
218
+ .map((p) => p.text).join('\n');
219
+
220
+ for (const m of messages) {
221
+ if (m.role === 'system') {
222
+ const parts = normalizeContent(m.content);
223
+ const text = flushTextOf(parts);
224
+ if (text) pendingSystem.push(text);
225
+ } else if (m.role === 'user' && pendingSystem.length > 0) {
226
+ const userParts = normalizeContent(m.content);
227
+ const userText = flushTextOf(userParts);
228
+ const userImages = userParts.filter((p) => p.type === 'image');
229
+ const wrapped = `<system>\n${pendingSystem.join('\n\n')}\n</system>\n${userText}`;
230
+ const newContent: ContentPart[] = [{ type: 'text', text: wrapped }, ...userImages];
231
+ out.push({ role: 'user', content: newContent });
232
+ pendingSystem = [];
233
+ } else {
234
+ // Flush accumulated system text before any non-system, non-user turn
235
+ // (assistant / tool) so system instructions keep their leading position
236
+ // instead of being deferred to a trailing synthesized user message.
237
+ if (pendingSystem.length > 0) {
238
+ out.push({
239
+ role: 'user',
240
+ content: [{ type: 'text', text: `<system>\n${pendingSystem.join('\n\n')}\n</system>` }],
241
+ });
242
+ pendingSystem = [];
243
+ }
244
+ out.push(m);
245
+ }
246
+ }
247
+ if (pendingSystem.length > 0) {
248
+ // Trailing system messages with no following user turn — convert to a
249
+ // standalone user message so they still reach the model.
250
+ out.push({
251
+ role: 'user',
252
+ content: [{ type: 'text', text: `<system>\n${pendingSystem.join('\n\n')}\n</system>` }],
253
+ });
254
+ }
255
+ return out;
256
+ }
257
+
258
+ /**
259
+ * CompletionConfiguration — mirrors the LS-shipped defaults, lets the caller
260
+ * override the obvious knobs.
261
+ */
262
+ /** Output cap when the caller named none. */
263
+ const DEFAULT_MAX_OUTPUT_TOKENS = 8192;
264
+ /** Context window when the caller named none. */
265
+ const DEFAULT_CONTEXT_WINDOW = 128_000;
266
+
267
+ /**
268
+ * Cognition rejects a temperature of exactly 0 with the same opaque internal
269
+ * error it uses for a malformed request, so a client asking for deterministic
270
+ * output would fail every turn. Clamp to the smallest value the wire accepts
271
+ * rather than silently substituting the service default, which would be a
272
+ * different answer than the caller asked for.
273
+ */
274
+ const MIN_TEMPERATURE = 0.0001;
275
+
276
+ function safeTemperature(value: number | undefined): number {
277
+ if (value === undefined) return 0.7;
278
+ return value <= 0 ? MIN_TEMPERATURE : value;
279
+ }
280
+
281
+ function encodeCompletionConfiguration(opts: {
282
+ maxOutputTokens?: number;
283
+ maxInputTokens?: number;
284
+ temperature?: number;
285
+ topK?: number;
286
+ topP?: number;
287
+ }): Buffer {
288
+ const enc64 = (fieldNum: number, n: number): Buffer => {
289
+ const b = Buffer.alloc(8);
290
+ b.writeDoubleLE(n, 0);
291
+ return Buffer.concat([Buffer.from([(fieldNum << 3) | 1]), b]);
292
+ };
293
+ // Tag map, verified by building the same turn with a working client and
294
+ // diffing the encoded messages field by field: #2 is the OUTPUT cap and #3 is
295
+ // the context window. This layout had those two swapped, so a caller asking
296
+ // for 32 output tokens put 32 into the context-window field and the request
297
+ // came back as an opaque "an internal error occurred" — for every turn, on
298
+ // every account, which is why free and paid failed identically. #6 and #11
299
+ // are not part of the message the service accepts.
300
+ return Buffer.concat([
301
+ encodeVarintField(1, 1),
302
+ encodeVarintField(2, opts.maxOutputTokens ?? DEFAULT_MAX_OUTPUT_TOKENS),
303
+ encodeVarintField(3, opts.maxInputTokens ?? DEFAULT_CONTEXT_WINDOW),
304
+ enc64(5, safeTemperature(opts.temperature)),
305
+ encodeVarintField(7, opts.topK ?? 40),
306
+ enc64(8, opts.topP ?? 1.0),
307
+ ]);
308
+ }
309
+
310
+ /**
311
+ * Multimodal content part — text or image.
312
+ *
313
+ * Text: `{ type: 'text', text: '...' }`
314
+ * Image: `{ type: 'image', mimeType: 'image/png', base64Data: '...' [, caption: '...'] }`
315
+ *
316
+ * Matches the OpenAI/@ai-sdk multimodal message shape — we accept their
317
+ * `image_url: { url: 'data:image/png;base64,...' }` form via {@link parseContent}.
318
+ */
319
+ export type ContentPart =
320
+ | { type: 'text'; text: string }
321
+ | { type: 'image'; mimeType: string; base64Data: string; caption?: string };
322
+
323
+ export interface ChatHistoryItem {
324
+ role: 'user' | 'assistant' | 'system' | 'tool';
325
+ /**
326
+ * Either a plain string or an array of {@link ContentPart}. Plain strings are
327
+ * shorthand for `[{ type: 'text', text: '...' }]`.
328
+ */
329
+ content: string | ContentPart[];
330
+ /**
331
+ * For `role: 'tool'` only — the id of the assistant's preceding tool_call
332
+ * this message answers. Required by the cloud's chat backend to pair
333
+ * tool results with calls; without it, multi-tool conversations can't
334
+ * tell the model which call produced which result. Encoded as
335
+ * ChatMessagePrompt field #7 (verified against the Windsurf bundled
336
+ * extension.js proto schema `exa.chat_pb.ChatMessagePrompt`).
337
+ */
338
+ tool_call_id?: string;
339
+ /**
340
+ * For `role: 'assistant'` only — the tool calls the assistant emitted.
341
+ * Encoded as ChatMessagePrompt field #6 (repeated ChatToolCall, where
342
+ * each ChatToolCall has #1 id, #2 name, #3 arguments_json).
343
+ */
344
+ tool_calls?: Array<{ id: string; name: string; arguments: string }>;
345
+ }
346
+
347
+ /**
348
+ * Normalize ChatHistoryItem content into structured parts. Accepts strings,
349
+ * OpenAI multimodal `[{type:'text',text}, {type:'image_url',image_url}]`, and
350
+ * our own `[{type:'image', mimeType, base64Data}]`.
351
+ */
352
+ function normalizeContent(content: string | ContentPart[] | unknown): ContentPart[] {
353
+ if (typeof content === 'string') return [{ type: 'text', text: content }];
354
+ if (!Array.isArray(content)) return [];
355
+ const out: ContentPart[] = [];
356
+ // Each element may follow our own ContentPart shape, the OpenAI multimodal
357
+ // `image_url` shape, or be malformed — narrow defensively per branch.
358
+ const parts = content as Array<Record<string, unknown>>;
359
+ for (const p of parts) {
360
+ if (!p || typeof p !== 'object') continue;
361
+ if (p.type === 'text' && typeof p.text === 'string') {
362
+ out.push({ type: 'text', text: p.text });
363
+ } else if (p.type === 'image' && typeof p.base64Data === 'string') {
364
+ const mimeType = typeof p.mimeType === 'string' ? p.mimeType : 'image/png';
365
+ const caption = typeof p.caption === 'string' ? p.caption : undefined;
366
+ out.push({ type: 'image', mimeType, base64Data: p.base64Data, caption });
367
+ } else if (p.type === 'image_url' && p.image_url) {
368
+ // OpenAI/@ai-sdk shape — parse data: URL into base64 + mime.
369
+ const imgRef = p.image_url as string | { url?: string };
370
+ const url: string = typeof imgRef === 'string' ? imgRef : (imgRef.url ?? '');
371
+ const m = url.match(/^data:([^;]+);base64,(.+)$/);
372
+ if (m) out.push({ type: 'image', mimeType: m[1], base64Data: m[2] });
373
+ else if (url) out.push({ type: 'text', text: `[image url: ${url}]` });
374
+ }
375
+ }
376
+ return out;
377
+ }
378
+
379
+ export interface ToolDef {
380
+ /** Function name. */
381
+ name: string;
382
+ /** Plain-English description. */
383
+ description: string;
384
+ /** JSON Schema for the function's arguments. */
385
+ parameters: unknown;
386
+ }
387
+
388
+ /**
389
+ * Streaming event emitted by the cloud-direct chat loop.
390
+ *
391
+ * - `text` : incremental visible content from the assistant
392
+ * - `reasoning` : incremental internal thinking (Anthropic-style, kept
393
+ * separate from visible content; @ai-sdk consumers can
394
+ * render in a collapsed/grey region)
395
+ * - `tool_call_*` : function-calling deltas (id+name once, args streamed)
396
+ * - `finish` : stream terminated cleanly with a reason
397
+ * - `usage` : final token-accounting block (input/output/total counts)
398
+ */
399
+ export type CloudChatEvent =
400
+ | { kind: 'text'; text: string }
401
+ | { kind: 'reasoning'; text: string }
402
+ | { kind: 'tool_call_start'; id: string; name: string }
403
+ | {
404
+ kind: 'tool_call_args';
405
+ argsDelta: string;
406
+ /**
407
+ * Tool-call id this delta belongs to, when the cloud surfaced one in
408
+ * this frame. Cognition's wire format only carries id on the START
409
+ * frame today, so most argsDelta events arrive without one — callers
410
+ * route those to the most-recent-start by convention. If Cognition
411
+ * ever interleaves args across calls, the consumer should prefer
412
+ * `id` over the rolling lastToolCallId.
413
+ */
414
+ id?: string;
415
+ }
416
+ // Note: there is no `tool_call_end` event. Cognition's wire format
417
+ // signals the end of a tool call implicitly — args just stop arriving
418
+ // for the current id and either a new `tool_call_start` fires or the
419
+ // stream finishes. Consumers should treat each `tool_call_start` as
420
+ // ending the previous call.
421
+ | { kind: 'finish'; reason: 'stop' | 'tool_calls' | 'length' | 'content_filter' }
422
+ | {
423
+ kind: 'usage';
424
+ promptTokens?: number;
425
+ completionTokens?: number;
426
+ totalTokens?: number;
427
+ /**
428
+ * Tokens served from the cache. Surfaced separately so callers tracking
429
+ * cost can distinguish them from fresh input tokens (Anthropic / OpenAI
430
+ * both bill cache reads cheaper than fresh prompts).
431
+ */
432
+ cachedInputTokens?: number;
433
+ /** Tokens written to the cache on this request (Anthropic-style). */
434
+ cacheCreationInputTokens?: number;
435
+ /** Reasoning tokens (gpt-5-x reasoning models, Claude thinking variants). */
436
+ reasoningTokens?: number;
437
+ };
438
+
439
+ interface BuildArgs {
440
+ apiKey: string;
441
+ userJwt?: string;
442
+ modelUid: string;
443
+ messages: ChatHistoryItem[];
444
+ cascadeId: string;
445
+ /**
446
+ * GetChatMessageRequest #22. Optional because it is omitted on a first turn;
447
+ * the working client only reuses one across a later tool loop.
448
+ */
449
+ promptId?: string;
450
+ sessionId: string;
451
+ requestId: bigint;
452
+ triggerId: string;
453
+ tools?: ToolDef[];
454
+ /** Default 5 = CHAT_MESSAGE_REQUEST_TYPE_CASCADE (matches captured LS body). */
455
+ requestType?: number;
456
+ completionOpts?: {
457
+ maxOutputTokens?: number;
458
+ maxInputTokens?: number;
459
+ temperature?: number;
460
+ topK?: number;
461
+ topP?: number;
462
+ };
463
+ }
464
+
465
+ /**
466
+ * ChatToolDefinition proto, observed in the LS upstream traffic:
467
+ * { #1 name (string), #2 description (string), #3 parameters_schema (JSON string) }
468
+ *
469
+ * Truncation note: Codeium's tool validator rejects very long descriptions
470
+ * with a generic `failed_precondition: "Unable to process request due to an
471
+ * MCP configuration issue."` error. opencode ships some tools (notably `bash`)
472
+ * with ~9.6 KB descriptions packed with examples and rules. We truncate to a
473
+ * conservative `MAX_DESC_LEN` and append an ellipsis so the cloud accepts
474
+ * them. The model still gets the first chunk of the description (where the
475
+ * essential signature lives); detailed examples are sacrificed for
476
+ * compatibility.
477
+ */
478
+ /**
479
+ * The Codeium tool validator rejects any tool whose description hits exactly
480
+ * 7,000 chars (or more) with a misleading `failed_precondition: "Unable to
481
+ * process request due to an MCP configuration issue."` error. Binary-search
482
+ * verified to char-precision:
483
+ * - 6,999 chars → server accepts
484
+ * - 7,000 chars → server returns MCP error
485
+ *
486
+ * The limit is per-description, content-sensitive (plain `a`-repeats up to
487
+ * 20K work fine; the bash description's exact byte at position 6999 trips
488
+ * it). We truncate to the maximum-1 (6,998) for a one-char safety margin.
489
+ *
490
+ * We do NOT need to aggregate-cap — 200K total tool descriptions across 200
491
+ * tools was confirmed to pass server-side. Only per-string length is gated.
492
+ */
493
+ const MAX_TOOL_DESC_LEN = 6998;
494
+
495
+ /**
496
+ * Cognition's cloud enforces a case-sensitive, whitespace-exact exact-phrase
497
+ * blocklist on tool descriptions. Binary-search isolated the trigger to the
498
+ * 7-word phrase "Takes a task_id parameter identifying the task" — verbatim,
499
+ * capital T, single spaces — which causes a `permission_denied` trailer error
500
+ * regardless of model or account tier. Any deviation (lowercase, reword,
501
+ * reorder, extra whitespace) passes. The phrase appears verbatim in Claude
502
+ * Code's built-in TaskOutput tool description.
503
+ *
504
+ * Rewrite known triggers to meaning-preserving forms. This is a
505
+ * Cognition-specific constraint alongside the length limit above; if
506
+ * Cognition adds more blocklisted phrases, extend this table and add a
507
+ * regression test in tests/devin-adapter.test.ts.
508
+ *
509
+ * Not every entry is matched the same way. The Claude Code phrase above is
510
+ * case-sensitive and whitespace-exact, but the two Codex entries below are
511
+ * not: against a live account, lowercasing the first word and doubling an
512
+ * interior space both still produced `permission_denied`, while changing any
513
+ * single word passed. So those two match case-insensitively with flexible
514
+ * whitespace and an optional comma, and the rewrite swaps only the leading
515
+ * verb — the smallest edit measured to clear the filter.
516
+ *
517
+ * These two sentences are Codex's own built-in `exec_command` and
518
+ * `write_stdin` descriptions, verbatim. Every Codex turn carries them, so
519
+ * before this table knew about them the cloud refused literally every request
520
+ * from a Codex client — a bare "hi" included — while the same account
521
+ * answered a hand-built request with an ordinary shell tool. The visible
522
+ * symptom was the adapter's own blocklist message pointing back at this
523
+ * table, which is why they are named here rather than left to the next person
524
+ * to re-bisect.
525
+ */
526
+ const COGNITION_BLOCKLIST_REWRITES: ReadonlyArray<[RegExp, string]> = [
527
+ [/\bTakes a task_id parameter identifying the task\b/g, "Accepts a task_id parameter identifying the task"],
528
+ [
529
+ /\bRuns\s+a\s+command\s+in\s+a\s+PTY,?\s+returning\s+output\s+or\s+a\s+session\s+ID\s+for\s+ongoing\s+interaction\b/gi,
530
+ "Executes a command in a PTY, returning output or a session ID for ongoing interaction",
531
+ ],
532
+ [
533
+ /\bWrites\s+characters\s+to\s+an\s+existing\s+unified\s+exec\s+session\s+and\s+returns\s+recent\s+output\b/gi,
534
+ "Sends characters to an existing unified exec session and returns recent output",
535
+ ],
536
+ ];
537
+
538
+ function sanitizeToolDescriptionForCognition(description: string): string {
539
+ let out = description;
540
+ for (const [pattern, replacement] of COGNITION_BLOCKLIST_REWRITES) {
541
+ out = out.replace(pattern, replacement);
542
+ }
543
+ return out;
544
+ }
545
+
546
+ /** Test-only: exercise the Cognition blocklist rewrite directly. */
547
+ export function sanitizeToolDescriptionForCognitionForTests(description: string): string {
548
+ return sanitizeToolDescriptionForCognition(description);
549
+ }
550
+
551
+ function encodeToolDef(tool: ToolDef): Buffer {
552
+ const rawDesc = sanitizeToolDescriptionForCognition(tool.description ?? '');
553
+ const desc =
554
+ rawDesc.length > MAX_TOOL_DESC_LEN
555
+ ? rawDesc.slice(0, MAX_TOOL_DESC_LEN - 24) + '\n…(truncated for cloud)'
556
+ : rawDesc;
557
+ return Buffer.concat([
558
+ encodeString(1, tool.name),
559
+ encodeString(2, desc),
560
+ encodeString(3, JSON.stringify(tool.parameters ?? {})),
561
+ ]);
562
+ }
563
+
564
+ export function buildGetChatMessageRequestForTests(args: BuildArgs): Buffer {
565
+ return buildGetChatMessageRequest(args);
566
+ }
567
+
568
+ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
569
+ const metadata = buildMetadata({
570
+ apiKey: args.apiKey,
571
+ userJwt: args.userJwt,
572
+ sessionId: args.sessionId,
573
+ requestId: args.requestId,
574
+ triggerId: args.triggerId,
575
+ // GetChatMessage accepts only the calibrated identity shape.
576
+ cloudChatShape: true,
577
+ });
578
+
579
+ // System messages must be inlined into the user turn (Cognition cloud
580
+ // rejects source=3). See `collapseSystemIntoUser` for the format.
581
+ const collapsed = collapseSystemIntoUser(args.messages);
582
+ const promptParts = collapsed.map((m) =>
583
+ encodeMessage(
584
+ 3,
585
+ encodeChatMessagePrompt(
586
+ normalizeContent(m.content),
587
+ SOURCE_BY_ROLE[m.role] ?? 1,
588
+ // Thread tool_call_id (for tool results) + tool_calls (for assistant
589
+ // turns that fired tools) into the proto. Cloud rejects multi-tool
590
+ // conversations otherwise — it can't pair a tool result with the
591
+ // assistant call that produced it.
592
+ {
593
+ toolCallId: m.role === 'tool' ? m.tool_call_id : undefined,
594
+ toolCalls: m.role === 'assistant' ? m.tool_calls : undefined,
595
+ },
596
+ ),
597
+ ),
598
+ );
599
+
600
+ const completion = encodeCompletionConfiguration(args.completionOpts ?? {});
601
+
602
+ const toolParts: Buffer[] = (args.tools ?? []).map((t) =>
603
+ encodeMessage(10, encodeToolDef(t)),
604
+ );
605
+
606
+ // Field layout from mitm capture of the LS:
607
+ // #1 metadata
608
+ // #3 chat_message_prompts (repeated — one element per history turn)
609
+ // #7 request_type (varint enum)
610
+ // #8 completion_configuration
611
+ // #10 tools (repeated ChatToolDefinition)
612
+ // #16 cascade_id (string)
613
+ // #21 chat_model_uid (string)
614
+ // #22 prompt_id (string)
615
+ return Buffer.concat([
616
+ encodeMessage(1, metadata),
617
+ // #2 system_prompt is always written, empty when the caller had none. The
618
+ // system turn is separately collapsed into the first user message because
619
+ // source=SYSTEM is refused; this field is the one the wire expects here.
620
+ encodeString(2, ''),
621
+ ...promptParts,
622
+ encodeVarintField(7, args.requestType ?? 5),
623
+ encodeMessage(8, completion),
624
+ ...toolParts,
625
+ // #15 session model config: { id, turn, 4 }. Present on every verified
626
+ // request.
627
+ encodeMessage(15, Buffer.concat([
628
+ encodeString(1, crypto.randomUUID()),
629
+ encodeVarintField(2, 1),
630
+ encodeVarintField(3, 4),
631
+ ])),
632
+ encodeString(16, args.cascadeId),
633
+ encodeVarintField(20, 1),
634
+ encodeString(21, args.modelUid),
635
+ // #22 is deliberately omitted. It is a user-exchange id that only appears
636
+ // from the second turn onward and is reused across that turn's tool loop; a
637
+ // fresh per-request uuid matches neither shape.
638
+ ]);
639
+ }
640
+
641
+ // ----------------------------------------------------------------------------
642
+ // Response parsing — pull `delta_text` (top-level field #9) out of each frame
643
+ // ----------------------------------------------------------------------------
644
+
645
+ /**
646
+ * Decode a single streaming ChatMessage proto frame into one or more
647
+ * CloudChatEvents. Captured shape (from a tool-using swe-1.6 chat):
648
+ *
649
+ * ChatMessage {
650
+ * #1 bot_id (string)
651
+ * #2 timestamp { seconds, nanos }
652
+ * #5 finish_reason (varint — 10 = "tool_calls" observed, others unknown)
653
+ * #6 ToolCallDelta {
654
+ * #1 id (string, only on first tool-call frame)
655
+ * #2 name (string, only on first tool-call frame)
656
+ * #3 arguments_delta (string, JSON fragment, streamed)
657
+ * }
658
+ * #7 ChatStatus { #6 status_code, #9 model_name }
659
+ * #9 delta_text (string)
660
+ * #12 (fixed64) some_hash
661
+ * #17 (string) message_uuid
662
+ * #28 UsageStats { #1 label, ... }
663
+ * }
664
+ *
665
+ * #9 appears both at top-level (text delta) AND inside #7 (model_name).
666
+ * iterFields walks top-level only, so we don't confuse the two.
667
+ *
668
+ * #5 is the finish_reason. Observed value `10` = tool_calls finish. We map
669
+ * any non-zero to 'tool_calls' for now (and let the caller fall back to
670
+ * 'stop' if no tool_call deltas were emitted).
671
+ */
672
+ function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
673
+ for (const f of iterFields(proto)) {
674
+ if (f.num === 3 && f.wire === 2 && Buffer.isBuffer(f.value)) {
675
+ // Visible delta_text — what the user should SEE in the chat.
676
+ //
677
+ // We previously had this mapping inverted (#3 = thinking, #9 = visible),
678
+ // which produced two compounding bugs in the TUI:
679
+ // 1. The model's CoT was rendered as plain content, so the user saw
680
+ // "The user wants me to X..." instead of the answer.
681
+ // 2. The actual answer (which lives in #3) was silently dropped — so
682
+ // the assistant turn appeared to end after the CoT with nothing
683
+ // after, matching the "model wrote reasoning then went silent"
684
+ // symptom the user reported.
685
+ // Verified live: prompted swe-1.6 with "explain then answer 2+2"; #3
686
+ // streamed "2+2=4 because... 4" while #9 streamed the meta-narration
687
+ // "The user wants me to perform a reasoning task...".
688
+ const s = (f.value as Buffer).toString('utf8');
689
+ if (s) yield { kind: 'text', text: s };
690
+ } else if (f.num === 9 && f.wire === 2 && Buffer.isBuffer(f.value)) {
691
+ // Internal thinking / chain-of-thought. Surface as `reasoning` so
692
+ // @ai-sdk consumers (opencode TUI) render it in a collapsed grey
693
+ // block instead of inline with the answer.
694
+ const s = (f.value as Buffer).toString('utf8');
695
+ if (s) yield { kind: 'reasoning', text: s };
696
+ } else if (f.num === 6 && f.wire === 2 && Buffer.isBuffer(f.value)) {
697
+ let id: string | undefined;
698
+ let name: string | undefined;
699
+ let argsDelta: string | undefined;
700
+ for (const sf of iterFields(f.value as Buffer)) {
701
+ if (sf.wire === 2 && Buffer.isBuffer(sf.value)) {
702
+ const s = (sf.value as Buffer).toString('utf8');
703
+ if (sf.num === 1) id = s;
704
+ else if (sf.num === 2) name = s;
705
+ else if (sf.num === 3) argsDelta = s;
706
+ }
707
+ }
708
+ if (id !== undefined && name !== undefined) {
709
+ yield { kind: 'tool_call_start', id, name };
710
+ }
711
+ if (argsDelta !== undefined) {
712
+ // Pass through `id` when this frame carries one (Cognition only
713
+ // sets it on the start frame today, but defending against future
714
+ // interleaving). Callers should prefer `id` over their rolling
715
+ // lastToolCallId when both are available.
716
+ yield { kind: 'tool_call_args', argsDelta, ...(id !== undefined ? { id } : {}) };
717
+ }
718
+ } else if (f.num === 5 && f.wire === 0) {
719
+ const v = Number(f.value);
720
+ // exa.codeium_common_pb.StopReason → OpenAI finish_reason.
721
+ // Source of truth: Windsurf extension.js sets `setEnumType("StopReason", [...])`
722
+ // 0 UNSPECIFIED → "stop" (no signal — treat as natural end)
723
+ // 1 INCOMPLETE → "length" (request cut short, model wanted more)
724
+ // 2 STOP_PATTERN → "stop" (model emitted its stop sequence — NORMAL)
725
+ // 3 MAX_TOKENS → "length"
726
+ // 4-9 internal → "stop"
727
+ // 10 FUNCTION_CALL → "tool_calls"
728
+ // 11 CONTENT_FILTER → "content_filter"
729
+ // 12 NON_INSERTION → "stop"
730
+ // 13 ERROR → "stop" (errors come as Connect trailer, not via this)
731
+ //
732
+ // We had 2 and 3 swapped previously, which made the model's normal
733
+ // STOP_PATTERN look like "length" → @ai-sdk treated complete responses
734
+ // as truncated. That was the "model wrote reasoning then went silent"
735
+ // symptom the user kept hitting.
736
+ let reason: 'stop' | 'tool_calls' | 'length' | 'content_filter' = 'stop';
737
+ if (v === 10) reason = 'tool_calls';
738
+ else if (v === 11) reason = 'content_filter';
739
+ else if (v === 1 || v === 3) reason = 'length';
740
+ // else stays 'stop' for 0/2/4-9/12/13
741
+ yield { kind: 'finish', reason };
742
+ } else if (f.num === 28 && f.wire === 2 && Buffer.isBuffer(f.value)) {
743
+ const usage = decodeUsageBlock(f.value as Buffer);
744
+ if (usage) yield usage;
745
+ }
746
+ }
747
+ }
748
+
749
+ /**
750
+ * UsageStats block at proto field #28. Captured shape (mitm of a real call):
751
+ *
752
+ * UsageStats {
753
+ * #1 label = "Token Usage"
754
+ * #2 entries [
755
+ * UsageEntry {
756
+ * #1 label = "Input tokens" / "Output tokens" / "Cached tokens" / ...
757
+ * #2 value (fixed32 — IEEE 754 float, OpenAI-style count cast)
758
+ * #3 unit = " tokens"
759
+ * #5 metric_id = "input_tokens" / "output_tokens" / ...
760
+ * },
761
+ * ...
762
+ * ]
763
+ * }
764
+ *
765
+ * We extract the standard input/output counts and synthesize a `total`.
766
+ * Anything else (cached, reasoning_tokens, …) is dropped for v1.
767
+ */
768
+ function decodeUsageBlock(buf: Buffer): CloudChatEvent | null {
769
+ let promptTokens: number | undefined;
770
+ let completionTokens: number | undefined;
771
+ let cachedInputTokens: number | undefined;
772
+ let cacheCreationInputTokens: number | undefined;
773
+ let reasoningTokens: number | undefined;
774
+
775
+ for (const f of iterFields(buf)) {
776
+ // Each UsageEntry lives at field 2 (repeated). Field 1 is the block label
777
+ // ("Token Usage"); skip.
778
+ if (f.num !== 2 || f.wire !== 2 || !Buffer.isBuffer(f.value)) continue;
779
+
780
+ // Observed entry shape:
781
+ // UsageEntry {
782
+ // #4 (sub-message) {
783
+ // #1 label = "Input tokens" / "Output tokens"
784
+ // #2 (fixed32) value (IEEE 754 LE float — count as float)
785
+ // #3 unit = " token"
786
+ // #4 unit_plural = " tokens"
787
+ // }
788
+ // #5 metric_id = "input_tokens" / "output_tokens" / "cached_input_tokens" / ...
789
+ // }
790
+ let entryMetric: string | undefined;
791
+ let entryValue: number | undefined;
792
+ for (const sf of iterFields(f.value as Buffer)) {
793
+ if (sf.num === 5 && sf.wire === 2 && Buffer.isBuffer(sf.value)) {
794
+ entryMetric = (sf.value as Buffer).toString('utf8');
795
+ } else if (sf.num === 4 && sf.wire === 2 && Buffer.isBuffer(sf.value)) {
796
+ // Recurse into the displayed-dimension submessage to pull the fixed32
797
+ // value at its field 2.
798
+ for (const ssf of iterFields(sf.value as Buffer)) {
799
+ if (ssf.num === 2 && ssf.wire === 5 && Buffer.isBuffer(ssf.value)) {
800
+ entryValue = (ssf.value as Buffer).readFloatLE(0);
801
+ break;
802
+ }
803
+ }
804
+ }
805
+ }
806
+ if (entryMetric && entryValue !== undefined && Number.isFinite(entryValue)) {
807
+ const n = Math.round(entryValue);
808
+ if (entryMetric === 'input_tokens') promptTokens = n;
809
+ else if (entryMetric === 'output_tokens') completionTokens = n;
810
+ else if (entryMetric === 'cached_input_tokens' || entryMetric === 'cache_read_input_tokens') {
811
+ cachedInputTokens = (cachedInputTokens ?? 0) + n;
812
+ } else if (entryMetric === 'cache_creation_input_tokens') {
813
+ cacheCreationInputTokens = (cacheCreationInputTokens ?? 0) + n;
814
+ } else if (entryMetric === 'reasoning_tokens' || entryMetric === 'output_reasoning_tokens') {
815
+ reasoningTokens = (reasoningTokens ?? 0) + n;
816
+ }
817
+ }
818
+ }
819
+ if (promptTokens === undefined && completionTokens === undefined) return null;
820
+ // totalTokens reflects what OpenAI's API counts as billable: input +
821
+ // output. Cached / cache-creation / reasoning subtotals are surfaced as
822
+ // additional fields so callers that want a fuller picture (e.g. cost
823
+ // breakdown for reasoning models) can read them, but they're NOT
824
+ // double-counted into total.
825
+ const total = (promptTokens ?? 0) + (completionTokens ?? 0);
826
+ return {
827
+ kind: 'usage',
828
+ promptTokens,
829
+ completionTokens,
830
+ totalTokens: total > 0 ? total : undefined,
831
+ cachedInputTokens,
832
+ cacheCreationInputTokens,
833
+ reasoningTokens,
834
+ };
835
+ }
836
+
837
+ // ----------------------------------------------------------------------------
838
+ // Public API: streamChat
839
+ // ----------------------------------------------------------------------------
840
+
841
+ export interface CloudChatRequest {
842
+ /** Persistent OAuth-issued api_key (`devin-session-token$<JWT>`). */
843
+ apiKey: string;
844
+ /** Pre-resolved API server URL from RegisterUser (falls back to default). */
845
+ apiServerUrl?: string;
846
+ /** Model UID — e.g. `swe-1-6`, `kimi-k2-6`, `claude-opus-4-7-medium`. */
847
+ modelUid: string;
848
+ /** Chat history. */
849
+ messages: ChatHistoryItem[];
850
+ /**
851
+ * Tool definitions available to the model. Cloud encodes these in the
852
+ * GetChatMessage request's `tools` field (proto #10). When set, the model
853
+ * may emit `tool_call_start`/`_args`/`_end` events instead of plain text.
854
+ */
855
+ tools?: ToolDef[];
856
+ /** Cascade ID — reuse across turns of the same conversation. */
857
+ cascadeId?: string;
858
+ /** Optional sampling overrides. */
859
+ completionOpts?: BuildArgs['completionOpts'];
860
+ /** Override request_type (default = 5, CASCADE). */
861
+ requestType?: number;
862
+ /** Abort signal — closes the fetch stream. */
863
+ signal?: AbortSignal;
864
+ }
865
+
866
+ export class CloudChatError extends Error {
867
+ constructor(message: string, public readonly code?: string, public readonly traceId?: string) {
868
+ super(message);
869
+ this.name = 'CloudChatError';
870
+ }
871
+ }
872
+
873
+ const TRACE_ID_RE = /\(trace ID: ([0-9a-f]+)\)/i;
874
+
875
+ /**
876
+ * Stream chat events from the cloud. Yields CloudChatEvent (text deltas, tool
877
+ * call deltas, finish reason). Use `streamChatText` for legacy text-only iteration.
878
+ *
879
+ * On error (auth fail, quota exhausted, malformed request) throws a
880
+ * CloudChatError with the cloud's `code` + `traceId` for diagnostics.
881
+ */
882
+ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<CloudChatEvent> {
883
+ // The api-server host comes from RegisterUser through the credential store.
884
+ // Validate it here too: this request body carries the api_key, so an
885
+ // unallowlisted host is credential exfiltration rather than a wrong endpoint.
886
+ const host = resolveDevinApiBaseUrl(req.apiServerUrl);
887
+ // The hosted chat path does not require the short-lived user_jwt; the working
888
+ // reference omits it by default. Minting it is opt-in so a mint failure or a
889
+ // JWT the chat service does not accept cannot break every turn.
890
+ const userJwt = process.env.OPENCODEX_DEVIN_SEND_USER_JWT === "1"
891
+ ? await getCachedUserJwt(req.apiKey, host, req.signal)
892
+ : undefined;
893
+
894
+ // Pre-flight: consult the per-account model catalog. Cognition's cloud
895
+ // returns an opaque `permission_denied: "an internal error occurred (trace
896
+ // ID: ...)"` for every chat call that targets a model not enabled on the
897
+ // caller's tier — issue #14. The catalog's `disabled` flag is the
898
+ // authoritative source for "can this account run this UID"; we surface a
899
+ // named error here so the user knows why instead of guessing.
900
+ //
901
+ // Best-effort: if the catalog fetch fails (network, auth, schema drift) we
902
+ // pass through to the chat call. The cloud will still surface its own
903
+ // error and the trailer-error path below enriches the message in-place.
904
+ // Treat an empty catalog (schema drift / unexpected response) as "no catalog"
905
+ // so chat passes through instead of failing every request.
906
+ const catalog = await getCachedCatalog(req.apiKey, host, req.signal).catch(() => null);
907
+ if (catalog && catalog.byUid.size > 0) {
908
+ const entry = catalog.byUid.get(req.modelUid);
909
+ if (!entry) {
910
+ throw new ModelNotAvailableError(req.modelUid, req.modelUid, 'not_listed');
911
+ }
912
+ if (entry.disabled) {
913
+ throw new ModelNotAvailableError(req.modelUid, entry.label, 'disabled');
914
+ }
915
+ }
916
+
917
+ // Reuse session + cascade ids across calls for the same (apiKey, host).
918
+ // Without this, every turn looks like a brand-new server-side session
919
+ // and the cloud's prompt cache never hits — significant cost regression
920
+ // for long conversations.
921
+ const sessionIds = getOrAllocateSessionIds(req.apiKey, host, req.cascadeId);
922
+
923
+ const proto = buildGetChatMessageRequest({
924
+ apiKey: req.apiKey,
925
+ userJwt,
926
+ modelUid: req.modelUid,
927
+ messages: req.messages,
928
+ tools: req.tools,
929
+ cascadeId: sessionIds.cascadeId,
930
+ sessionId: sessionIds.sessionId,
931
+ requestId: BigInt(Date.now()),
932
+ triggerId: crypto.randomUUID(),
933
+ requestType: req.requestType,
934
+ completionOpts: req.completionOpts,
935
+ });
936
+ // The request envelope goes up uncompressed. A gzipped GetChatMessage frame is
937
+ // rejected with the same opaque `invalid_argument: an internal error occurred`
938
+ // the short fingerprint produces, and it is one of three things that have to be
939
+ // right together — the other two are the doubled Basic credential and the
940
+ // 732-character Metadata #31.
941
+ const framed = frameConnectStream(proto, false);
942
+ const body = new Blob([new Uint8Array(framed)], { type: "application/connect+proto" });
943
+
944
+ // Compose caller signal with a TTFB timeout. If the cloud takes longer
945
+ // than CLOUD_STREAM_TTFB_MS to start the response, abort. Once any byte
946
+ // arrives we cancel the TTFB timer and start the per-chunk idle timer
947
+ // inside the read loop instead.
948
+ const ttfbController = new AbortController();
949
+ const ttfbTimer = setTimeout(() => ttfbController.abort(new Error(`cloud-direct: time-to-first-byte timeout (${CLOUD_STREAM_TTFB_MS}ms)`)), CLOUD_STREAM_TTFB_MS);
950
+ const ttfbSignal = ttfbController.signal;
951
+ // Compose req.signal + ttfbSignal. AbortSignal.any was added in Node
952
+ // 20.3 / Bun 1.0; our `engines` allows Node ≥18, so on Node 18-20.2 the
953
+ // built-in is missing. The previous fallback `req.signal ?? ttfbSignal`
954
+ // silently discarded one of the two signals (TTFB if caller passed
955
+ // one), defeating the timeout guard. anySignal() is a real polyfill.
956
+ const composed = req.signal ? anySignal([req.signal, ttfbSignal]) : undefined;
957
+ const initialSignal: AbortSignal = composed?.signal ?? ttfbSignal;
958
+
959
+ let resp: Response;
960
+ try {
961
+ resp = await fetch(`${host}/exa.api_server_pb.ApiServerService/GetChatMessage`, {
962
+ method: 'POST',
963
+ headers: {
964
+ 'Content-Type': 'application/connect+proto',
965
+ 'Connect-Protocol-Version': '1',
966
+ 'Connect-Accept-Encoding': 'gzip',
967
+ // The credential is the session token doubled and dash-joined. A single
968
+ // copy is refused with permission_denied. The protobuf body keeps one
969
+ // copy, in Metadata #3.
970
+ Authorization: `Basic ${req.apiKey}-${req.apiKey}`,
971
+ 'User-Agent': 'connect-es/2.0.0',
972
+ Accept: '*/*',
973
+ },
974
+ body,
975
+ redirect: 'error',
976
+ signal: initialSignal,
977
+ });
978
+ } finally {
979
+ clearTimeout(ttfbTimer);
980
+ // The composed signal only guards the headers hop; the body is cancelled
981
+ // through cancelBodyOnAbort below. Detaching here keeps a long-lived caller
982
+ // signal from collecting one listener per turn.
983
+ composed?.cleanup();
984
+ }
985
+
986
+ if (!resp.ok) {
987
+ // The body is not echoed into the message. This error reaches the adapter's
988
+ // error event and /api/logs, and a Connect error can quote the request that
989
+ // produced it - which is the request holding the api_key.
990
+ throw new CloudChatError(`GetChatMessage failed (HTTP ${resp.status})`, undefined);
991
+ }
992
+ if (!resp.body) {
993
+ throw new CloudChatError('GetChatMessage response had no body stream');
994
+ }
995
+
996
+ // Cancel the body when the client goes away. Without this the read loop never
997
+ // observes req.signal after headers arrive: the turn keeps draining until the
998
+ // idle timer fires, and the stream then ends without an EOS trailer, which
999
+ // this function would report as a truncated upstream response rather than as
1000
+ // the cancellation it actually was.
1001
+ const detachBodyCancel = cancelBodyOnAbort(resp.body, req.signal);
1002
+
1003
+ // Incremental parsing. We previously did `pending = Buffer.concat([pending,
1004
+ // chunk])` per chunk — O(n²) over a long stream because every chunk copies
1005
+ // every buffered byte again. Now we keep a queue of arriving chunks with a
1006
+ // running offset; we only `Buffer.concat` when a frame straddles a chunk
1007
+ // boundary, and we slice/drop fully-consumed chunks immediately. For
1008
+ // typical 50-200KB responses this is ~5x faster and produces zero waste.
1009
+ const chunkQueue: Buffer[] = [];
1010
+ let queuedBytes = 0;
1011
+ // Bun + Node ReadableStream readers diverge on the type-level shape
1012
+ // (Bun's includes a `readMany` method); both work the same at runtime.
1013
+ const reader = resp.body.getReader() as ReadableStreamDefaultReader<Uint8Array>;
1014
+ let trailerError: { code?: string; message: string; traceId?: string } | null = null;
1015
+ let sawEos = false;
1016
+
1017
+ /**
1018
+ * Try to read the next `n` bytes from the chunk queue WITHOUT consuming
1019
+ * them. Returns null if not enough buffered.
1020
+ */
1021
+ function peek(n: number): Buffer | null {
1022
+ if (queuedBytes < n) return null;
1023
+ if (chunkQueue.length === 1 && chunkQueue[0].length >= n) {
1024
+ return chunkQueue[0].slice(0, n);
1025
+ }
1026
+ // Cross-chunk peek — concat just the prefix we need.
1027
+ const parts: Buffer[] = [];
1028
+ let remaining = n;
1029
+ for (const c of chunkQueue) {
1030
+ if (remaining <= 0) break;
1031
+ if (c.length <= remaining) {
1032
+ parts.push(c);
1033
+ remaining -= c.length;
1034
+ } else {
1035
+ parts.push(c.slice(0, remaining));
1036
+ remaining = 0;
1037
+ }
1038
+ }
1039
+ return Buffer.concat(parts, n);
1040
+ }
1041
+
1042
+ /** Drop the first `n` bytes from the chunk queue. */
1043
+ function drop(n: number): void {
1044
+ queuedBytes -= n;
1045
+ let remaining = n;
1046
+ while (remaining > 0 && chunkQueue.length > 0) {
1047
+ const head = chunkQueue[0];
1048
+ if (head.length <= remaining) {
1049
+ chunkQueue.shift();
1050
+ remaining -= head.length;
1051
+ } else {
1052
+ chunkQueue[0] = head.slice(remaining);
1053
+ remaining = 0;
1054
+ }
1055
+ }
1056
+ }
1057
+
1058
+ // Track the idle timer at outer scope so the finally block can clear it
1059
+ // regardless of how we exit the read loop (clean done, throw, etc).
1060
+ // Previously this lived inside `try { ... }` and was only cleared on
1061
+ // normal exit — an error path left a 120s timer in the event loop and
1062
+ // the process refused to exit promptly.
1063
+ let idleTimer: ReturnType<typeof setTimeout> | null = null;
1064
+ try {
1065
+ const resetIdle = (): Promise<{ value?: Uint8Array; done: boolean }> => {
1066
+ if (idleTimer) clearTimeout(idleTimer);
1067
+ const idleController = new AbortController();
1068
+ idleTimer = setTimeout(
1069
+ () => idleController.abort(new Error(`cloud-direct: idle timeout (${CLOUD_STREAM_IDLE_MS}ms with no bytes)`)),
1070
+ CLOUD_STREAM_IDLE_MS,
1071
+ );
1072
+ // Race the reader.read() against idle abort. When abort wins, we
1073
+ // also actively `cancel()` the underlying body stream so the
1074
+ // pending read() resolves promptly with done=true instead of
1075
+ // hanging on the now-dead TCP socket until the OS notices.
1076
+ //
1077
+ // Promise-handling carefully: the reader.read() promise can settle
1078
+ // AFTER the outer race rejects (we cancelled, the read eventually
1079
+ // sees the cancellation and either resolves with done=true or
1080
+ // rejects with an abort error). We attach an explicit `.catch(()=>{})`
1081
+ // on the read promise so any post-race rejection doesn't surface as
1082
+ // an unhandled-rejection warning in the host runtime.
1083
+ return new Promise((resolve, reject) => {
1084
+ let settled = false;
1085
+ const settle = (fn: () => void): void => {
1086
+ if (settled) return;
1087
+ settled = true;
1088
+ fn();
1089
+ };
1090
+ const readP = reader.read();
1091
+ // Defensive: swallow any post-race rejection. If the outer promise
1092
+ // already settled via the abort listener, we still need a handler
1093
+ // attached to readP or Node logs an unhandledRejection.
1094
+ readP.catch(() => { /* swallowed; outer promise already rejected */ });
1095
+
1096
+ idleController.signal.addEventListener('abort', () => {
1097
+ try { void resp.body?.cancel(idleController.signal.reason ?? new Error('idle abort')); } catch { /* */ }
1098
+ settle(() => reject(idleController.signal.reason ?? new Error('idle abort')));
1099
+ }, { once: true });
1100
+
1101
+ readP.then(
1102
+ (v) => settle(() => resolve(v)),
1103
+ (e) => settle(() => reject(e)),
1104
+ );
1105
+ });
1106
+ };
1107
+
1108
+ while (true) {
1109
+ const { value, done } = await resetIdle();
1110
+ if (done) break;
1111
+ if (value) {
1112
+ chunkQueue.push(Buffer.from(value));
1113
+ queuedBytes += value.length;
1114
+ }
1115
+
1116
+ // Drain every complete frame currently buffered.
1117
+ while (queuedBytes >= 5) {
1118
+ const header = peek(5);
1119
+ if (!header) break;
1120
+ const flags = header[0];
1121
+ const len = header.readUInt32BE(1);
1122
+ // Cap frame length to prevent memory exhaustion from a corrupt/malicious
1123
+ // length prefix. 16MB is well above any legitimate Connect-RPC frame.
1124
+ if (len > MAX_FRAME_LEN) {
1125
+ throw new CloudChatError(`Connect frame length ${len} exceeds ${MAX_FRAME_LEN} byte cap`);
1126
+ }
1127
+ if (queuedBytes < 5 + len) break; // frame still arriving
1128
+ drop(5);
1129
+ const raw = peek(len) ?? Buffer.alloc(0);
1130
+ drop(len);
1131
+
1132
+ let payload = raw;
1133
+ if (flags & 0x01) {
1134
+ try {
1135
+ // MAX_FRAME_LEN caps the COMPRESSED frame, so without an output cap
1136
+ // a 16 MiB gzip frame can still inflate to gigabytes. The inbound
1137
+ // request path (src/server/request-decompress.ts) already bounds
1138
+ // decompression the same way.
1139
+ payload = zlib.gunzipSync(raw, { maxOutputLength: MAX_FRAME_LEN });
1140
+ } catch (gzipErr) {
1141
+ const code = (gzipErr as NodeJS.ErrnoException).code;
1142
+ if (code === 'ERR_BUFFER_TOO_LARGE') {
1143
+ throw new CloudChatError(`Connect frame inflates past the ${MAX_FRAME_LEN} byte cap`, 'frame_too_large');
1144
+ }
1145
+ // Corrupt compressed frame — surface as a CloudChatError instead
1146
+ // of falling through and re-parsing raw gzip bytes as proto
1147
+ // (which used to misparse silently downstream).
1148
+ throw new CloudChatError(`Connect frame gunzip failed: ${(gzipErr as Error).message}`);
1149
+ }
1150
+ }
1151
+ const eos = (flags & 0x02) !== 0;
1152
+
1153
+ if (eos) {
1154
+ sawEos = true;
1155
+ // Trailer: {} on success, {"error":{code,message}} on failure.
1156
+ const text = payload.toString('utf8');
1157
+ if (text && text.includes('"error"')) {
1158
+ let code: string | undefined;
1159
+ let message = text;
1160
+ try {
1161
+ const j = JSON.parse(text) as { error?: { code?: string; message?: string } };
1162
+ code = j.error?.code;
1163
+ if (j.error?.message) message = j.error.message;
1164
+ } catch { /* keep raw */ }
1165
+ const traceMatch = message.match(TRACE_ID_RE);
1166
+ trailerError = { code, message, traceId: traceMatch?.[1] };
1167
+ }
1168
+ continue;
1169
+ }
1170
+ yield* decodeChatFrame(payload);
1171
+ }
1172
+ }
1173
+ } finally {
1174
+ // Always clear the idle timer. The previous "clear on normal exit
1175
+ // only" path leaked a 120s setTimeout into the event loop on any
1176
+ // throw (idle timeout, gunzip error, trailer error, etc), keeping
1177
+ // the process from exiting promptly.
1178
+ if (idleTimer) clearTimeout(idleTimer);
1179
+ // Cancel the underlying body stream on any non-clean exit so the TCP
1180
+ // connection is released. `releaseLock` alone leaves the body in a
1181
+ // dangling state; we have to call `cancel` on the response body
1182
+ // itself (cancel-via-reader requires holding the lock). Fire and
1183
+ // forget — there's nothing meaningful to do if cancel rejects.
1184
+ try { reader.releaseLock(); } catch { /* */ }
1185
+ try { void resp.body?.cancel(); } catch { /* */ }
1186
+ }
1187
+
1188
+ if (trailerError) {
1189
+ // Cognition uses `permission_denied: "an internal error occurred (trace
1190
+ // ID: …)"` as a catch-all for "your account can't run this model" — same
1191
+ // root cause issue #14 reported. The pre-flight above catches this when
1192
+ // the catalog disagrees with the call, but the catalog can lag (a model
1193
+ // that was enabled at fetch time may have been gated between then and
1194
+ // now) or be missing (network failure caused a fall-through). When the
1195
+ // raw trailer is this exact shape, swap in a message that names the
1196
+ // model and explains the likely cause rather than re-passing
1197
+ // Cognition's opaque text. The cloud's original message is appended in
1198
+ // parens so users (and bug reports) still have it verbatim.
1199
+ // Both codes carry this shape. Cognition uses `invalid_argument` for a
1200
+ // request it could not accept and `permission_denied` for one it would not,
1201
+ // and the message body is the same opaque sentence either way.
1202
+ const isOpaqueDenial =
1203
+ (trailerError.code === 'permission_denied' || trailerError.code === 'invalid_argument') &&
1204
+ /an internal error occurred/i.test(trailerError.message);
1205
+ if (isOpaqueDenial) {
1206
+ const enriched =
1207
+ `Cognition denied this request for model "${req.modelUid}" with the opaque ` +
1208
+ `"an internal error occurred" message, which it uses for both a malformed ` +
1209
+ `request and a refused one. In practice this has meant the request, not ` +
1210
+ `the account: the same sentence came back for every turn until the ` +
1211
+ `CompletionConfiguration tag map was corrected, and a temperature of ` +
1212
+ `exactly 0 still produces it. Check the request before the plan — ` +
1213
+ `tests/providers/devin-hardening.test.ts pins the field layout the ` +
1214
+ `service accepts. If the request is unchanged and this is new, the ` +
1215
+ `account's model access is the next thing to check. ` +
1216
+ `(cloud trace ID: ${trailerError.traceId ?? 'n/a'})`;
1217
+ throw new CloudChatError(enriched, trailerError.code, trailerError.traceId);
1218
+ }
1219
+ // Cognition also returns `permission_denied` when a tool description
1220
+ // contains a blocklisted phrase that the sanitizer above did not catch
1221
+ // (e.g. Cognition added a new phrase). Surface a clear message so the
1222
+ // user knows to check tool descriptions rather than suspect auth/tier.
1223
+ // Only blame the blocklist when tools were actually sent. Asserting it for
1224
+ // every permission_denied sent users to inspect a tool table that had
1225
+ // nothing to do with an ordinary ACL or tier denial.
1226
+ if (trailerError.code === 'permission_denied' && (req.tools?.length ?? 0) > 0) {
1227
+ const enriched =
1228
+ `Cognition denied this request (permission_denied). If tool descriptions ` +
1229
+ `are present, a blocklisted phrase may have triggered this — see the ` +
1230
+ `COGNITION_BLOCKLIST_REWRITES table in cloud-direct/chat.ts. ` +
1231
+ // Keep the cloud's own sentence. Replacing it outright is what made the
1232
+ // two Codex entries in that table expensive to find: the message named
1233
+ // the table but dropped the only text that could have said whether this
1234
+ // was a phrase match at all.
1235
+ `(cloud message: ${trailerError.message}) ` +
1236
+ `(cloud trace ID: ${trailerError.traceId ?? 'n/a'})`;
1237
+ throw new CloudChatError(enriched, trailerError.code, trailerError.traceId);
1238
+ }
1239
+ throw new CloudChatError(trailerError.message, trailerError.code, trailerError.traceId);
1240
+ }
1241
+ // Truncation detection: the cloud always terminates a successful stream
1242
+ // with an EOS trailer. If we hit `done` from the body reader without one,
1243
+ // the connection dropped mid-frame and any bytes still in the queue are
1244
+ // garbage. Previously those leftover bytes were silently discarded and
1245
+ // the consumer saw a clean stop with no error — looked like the model
1246
+ // had finished. Now we surface it.
1247
+ detachBodyCancel();
1248
+ if (req.signal?.aborted) {
1249
+ // The caller cancelled. The missing EOS trailer is the expected consequence
1250
+ // of that cancellation, not evidence that the cloud dropped the response.
1251
+ return;
1252
+ }
1253
+ if (!sawEos) {
1254
+ throw new CloudChatError(
1255
+ `Cloud stream ended without EOS trailer (${queuedBytes} bytes orphaned). ` +
1256
+ `Connection likely dropped mid-response.`,
1257
+ 'truncated_stream',
1258
+ );
1259
+ }
1260
+ }
1261
+
1262
+ /**
1263
+ * Back-compat: yield text content only (drops tool calls). The plugin uses
1264
+ * streamChatEvents directly when it needs to surface tool_calls.
1265
+ */
1266
+ export async function* streamChat(req: CloudChatRequest): AsyncGenerator<string> {
1267
+ for await (const ev of streamChatEvents(req)) {
1268
+ if (ev.kind === 'text') yield ev.text;
1269
+ }
1270
+ }
1271
+
1272
+ // `parseConnectFrames` is no longer needed by streamChat itself, but exported
1273
+ // from wire.ts for one-shot callers + tests.
1274
+ void parseConnectFrames;