@bitkyc08/opencodex 2.52.0-preview.20260911 → 2.52.0-preview.20260912
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/gui/dist/assets/index-D_t6sCWs.js +115 -0
- package/gui/dist/assets/index-EdoPnm9_.css +1 -0
- package/gui/dist/index.html +2 -2
- package/gui/dist/provider-icons/devin.svg +49 -0
- package/gui/dist/provider-icons/omo.svg +42 -0
- package/package.json +3 -1
- package/src/AGENTS.md +1 -1
- package/src/adapters/cline-pass-deepseek-v4-tool-replay.ts +0 -1
- package/src/adapters/command-code.ts +0 -1
- package/src/adapters/cursor/checkpoint-store.ts +37 -0
- package/src/adapters/cursor/live-transport.ts +74 -2
- package/src/adapters/cursor.ts +6 -0
- package/src/adapters/devin/cloud-direct/auth.ts +264 -0
- package/src/adapters/devin/cloud-direct/catalog.ts +306 -0
- package/src/adapters/devin/cloud-direct/chat.ts +1274 -0
- package/src/adapters/devin/cloud-direct/index.ts +65 -0
- package/src/adapters/devin/cloud-direct/metadata.ts +134 -0
- package/src/adapters/devin/cloud-direct/wire.ts +206 -0
- package/src/adapters/devin/live-models.ts +133 -0
- package/src/adapters/devin-cli/acp.ts +204 -0
- package/src/adapters/devin-cli/adapter.ts +345 -0
- package/src/adapters/devin-cli/binary.ts +69 -0
- package/src/adapters/devin-cli/models.ts +57 -0
- package/src/adapters/devin.ts +326 -0
- package/src/adapters/google.ts +1 -1
- package/src/adapters/openai-chat.ts +31 -10
- package/src/adapters/openai-responses.ts +16 -1
- package/src/adapters/registry.ts +25 -1
- package/src/bridge.ts +8 -2
- package/src/claude/inbound-cache-stabilize.ts +130 -0
- package/src/claude/inbound.ts +45 -5
- package/src/cli/account-auth.ts +60 -1
- package/src/cli/account-extended.ts +23 -30
- package/src/cli/account.ts +2 -1
- package/src/cli/capabilities.ts +50 -8
- package/src/cli/config-command.ts +2 -2
- package/src/cli/dispatch.ts +14 -2
- package/src/cli/export-command.ts +11 -25
- package/src/cli/help.ts +2 -2
- package/src/cli/opencode.ts +5 -0
- package/src/cli/registry.ts +15 -4
- package/src/clients/config-export/cline.ts +71 -0
- package/src/clients/config-export/contracts.ts +10 -1
- package/src/clients/config-export/model-metadata.ts +33 -0
- package/src/clients/config-export/zcode.ts +23 -12
- package/src/clients/config-export.ts +156 -4
- package/src/codex/account-store.ts +7 -2
- package/src/codex/auth-context.ts +1 -1
- package/src/codex/catalog/effort.ts +4 -3
- package/src/codex/catalog/metadata.ts +8 -4
- package/src/codex/catalog/native-models.ts +24 -4
- package/src/codex/catalog/parsing.ts +19 -1
- package/src/codex/catalog/provider-fetch.ts +57 -0
- package/src/codex/catalog/sync.ts +1 -1
- package/src/codex/context-compat.ts +97 -0
- package/src/codex/context-owner.ts +201 -0
- package/src/codex/history-provider.ts +161 -14
- package/src/codex/inject-coordination.ts +3 -2
- package/src/codex/inject.ts +128 -9
- package/src/codex/pool-rotation.ts +8 -292
- package/src/codex/quota.ts +41 -12
- package/src/codex/retired-model-migration.ts +41 -0
- package/src/codex/routing.ts +327 -22
- package/src/codex/warmup.ts +2 -2
- package/src/combos/failover.ts +5 -0
- package/src/combos/resolve.ts +11 -15
- package/src/config.ts +46 -14
- package/src/generated/compatibility-version.json +293 -113
- package/src/grok/grpc-web.ts +120 -0
- package/src/grok/reset-coupon-ledger.ts +139 -0
- package/src/grok/reset-coupons.ts +278 -0
- package/src/integrations/catalog-refresh.ts +1 -1
- package/src/integrations/cline-document.ts +73 -0
- package/src/integrations/cline-io.ts +149 -0
- package/src/integrations/cline-transaction.ts +42 -0
- package/src/integrations/config-io.ts +21 -2
- package/src/integrations/journal.ts +43 -1
- package/src/integrations/omp-yaml-source.ts +123 -4
- package/src/integrations/ownership-policy.ts +7 -0
- package/src/integrations/ownership.ts +36 -0
- package/src/integrations/registry.ts +27 -0
- package/src/integrations/state.ts +5 -2
- package/src/integrations/store.ts +6 -0
- package/src/integrations/writer.ts +53 -9
- package/src/lab/subject/behavior-fingerprint.ts +1 -1
- package/src/lib/abort.ts +36 -0
- package/src/lib/local-destinations.ts +1 -1
- package/src/lib/upstream-retry.ts +36 -1
- package/src/oauth/account-quota-rank.ts +11 -0
- package/src/oauth/callback-server.ts +10 -5
- package/src/oauth/devin/api-base.ts +63 -0
- package/src/oauth/devin/login.ts +1 -0
- package/src/oauth/devin/register-user.ts +186 -0
- package/src/oauth/devin/types.ts +71 -0
- package/src/oauth/devin-cli.ts +149 -0
- package/src/oauth/devin.ts +170 -0
- package/src/oauth/generic-account-failover.ts +169 -1
- package/src/oauth/index.ts +20 -1
- package/src/oauth/login-cli.ts +19 -5
- package/src/oauth/pool-kernel.ts +321 -0
- package/src/oauth/pool-settings-capability.ts +127 -9
- package/src/oauth/store.ts +7 -3
- package/src/oauth/token-guardian.ts +1 -1
- package/src/providers/codebuddy-models.ts +0 -3
- package/src/providers/command-code-efforts.ts +23 -4
- package/src/providers/default-aliases.ts +4 -0
- package/src/providers/derive.ts +8 -0
- package/src/providers/devin-cli-authmode-migration.ts +57 -0
- package/src/providers/key-failover.ts +175 -0
- package/src/providers/model-rename-startup.ts +21 -1
- package/src/providers/qoder-models.ts +0 -1
- package/src/providers/quota-key-accounts.ts +50 -0
- package/src/providers/quota-routing-cache.ts +65 -6
- package/src/providers/quota-types.ts +7 -0
- package/src/providers/quota.ts +113 -30
- package/src/providers/registry.ts +237 -77
- package/src/providers/stale-context-window-migration.ts +92 -0
- package/src/providers/zai-responses-migration.ts +45 -0
- package/src/quota/reset-observer.ts +2 -1
- package/src/quota/reset-seen-store.ts +9 -1
- package/src/remote-control/crypto.ts +442 -0
- package/src/remote-control/host.ts +175 -0
- package/src/remote-control/index.ts +100 -0
- package/src/remote-control/protocol.ts +200 -0
- package/src/remote-control/relay.ts +162 -0
- package/src/remote-control/workspace-agent-protocol.ts +246 -0
- package/src/remote-control/workspace-rpc-framing.ts +128 -0
- package/src/remote-control/workspace-tools.ts +237 -0
- package/src/remote-control/workspace-utf8.ts +24 -0
- package/src/responses/custom-tool-compat.ts +23 -0
- package/src/router.ts +6 -1
- package/src/routing/compatibility/behavior.ts +3 -0
- package/src/server/adapter-resolve.ts +6 -2
- package/src/server/auth-cors.ts +71 -6
- package/src/server/chat-completions.ts +5 -3
- package/src/server/chat-native.ts +9 -0
- package/src/server/claude-messages.ts +25 -30
- package/src/server/context-history.ts +207 -0
- package/src/server/images.ts +28 -1
- package/src/server/index.ts +42 -22
- package/src/server/live.ts +2 -1
- package/src/server/management/agent-settings-routes.ts +7 -1
- package/src/server/management/config-routes.ts +3 -3
- package/src/server/management/grok-coupon-routes.ts +287 -0
- package/src/server/management/integration-routes.ts +6 -1
- package/src/server/management/oauth-account-routes.ts +124 -7
- package/src/server/management/provider-routes.ts +77 -2
- package/src/server/management/route-registry.ts +6 -1
- package/src/server/management-api.ts +7 -0
- package/src/server/request-log.ts +4 -0
- package/src/server/responses/codex-ws-exchange.ts +90 -14
- package/src/server/responses/codex-ws-wire.ts +51 -1
- package/src/server/responses/compact.ts +8 -1
- package/src/server/responses/core.ts +188 -22
- package/src/server/responses/ws-upstream.ts +1 -1
- package/src/server/responses-undeclared-tool-guard.ts +261 -12
- package/src/server/zai-responses-startup.ts +21 -0
- package/src/types/config.ts +33 -3
- package/src/types/provider.ts +54 -4
- package/src/types/request.ts +1 -1
- package/src/types/tools.ts +36 -6
- package/src/update/job.ts +33 -11
- package/src/vision/plan.ts +40 -37
- package/src/web-search/index.ts +1 -0
- package/gui/dist/assets/index-BoBRSehJ.css +0 -1
- package/gui/dist/assets/index-Dx0xv2EA.js +0 -115
|
@@ -0,0 +1,1274 @@
|
|
|
1
|
+
/*
|
|
2
|
+
* Derived from rsvedant/opencode-windsurf-auth (src/cloud-direct/), MIT licensed,
|
|
3
|
+
* Copyright (c) 2026 Vedant. The full notice is in ./index.ts.
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Cloud-direct streaming chat. Talks to
|
|
7
|
+
* `server.codeium.com/exa.api_server_pb.ApiServerService/GetChatMessage`
|
|
8
|
+
* with no local language_server in the path. Returns an async iterable of
|
|
9
|
+
* CloudChatEvent deltas (text, reasoning, tool calls, usage, finish) so the
|
|
10
|
+
* caller can stream straight into opencodex's internal AdapterEvent model.
|
|
11
|
+
*
|
|
12
|
+
* What this supports:
|
|
13
|
+
* - Single- or multi-turn chat using the prompt-and-history pattern the LS
|
|
14
|
+
* uses (flatten history into one ChatMessagePrompt list)
|
|
15
|
+
* - All free Windsurf/Cognition models (swe-1-7, swe-1-7-lightning, etc.)
|
|
16
|
+
* and any model the user's api_key is entitled to
|
|
17
|
+
* - Streaming (uses Connect-streaming envelope, emits deltas as they arrive)
|
|
18
|
+
* - Tool definitions (encoded via `encodeToolDef`) and tool-call events
|
|
19
|
+
* (tool_call_start, tool_call_args) decoded from the response stream
|
|
20
|
+
* - Usage and finish-reason events for terminal completion
|
|
21
|
+
*
|
|
22
|
+
* Wire-protocol: Connect-RPC streaming over HTTPS with manual protobuf
|
|
23
|
+
* encoding (see `wire.ts`).
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import * as crypto from 'crypto';
|
|
27
|
+
import * as zlib from 'zlib';
|
|
28
|
+
import {
|
|
29
|
+
encodeMessage,
|
|
30
|
+
encodeString,
|
|
31
|
+
encodeVarintField,
|
|
32
|
+
frameConnectStream,
|
|
33
|
+
iterFields,
|
|
34
|
+
parseConnectFrames,
|
|
35
|
+
} from './wire.js';
|
|
36
|
+
import { buildMetadata } from './metadata.js';
|
|
37
|
+
import { getCachedUserJwt } from './auth.js';
|
|
38
|
+
import { getCachedCatalog, ModelNotAvailableError } from './catalog.js';
|
|
39
|
+
import { anySignal, cancelBodyOnAbort } from '../../../lib/abort.js';
|
|
40
|
+
import { resolveDevinApiBaseUrl } from '../../../oauth/devin/api-base.js';
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Connect-RPC streaming inactivity timeout. If the cloud sends zero bytes
|
|
44
|
+
* for this long after the last chunk, we abort the fetch. The cloud's own
|
|
45
|
+
* idle limit is around 90s on most models; we set ours a little above so
|
|
46
|
+
* we only trigger when the server has genuinely stopped responding.
|
|
47
|
+
*/
|
|
48
|
+
const CLOUD_STREAM_IDLE_MS = 120_000;
|
|
49
|
+
/** Time-to-first-byte timeout. */
|
|
50
|
+
const CLOUD_STREAM_TTFB_MS = 60_000;
|
|
51
|
+
/** Maximum acceptable Connect-RPC frame length (16 MB). */
|
|
52
|
+
const MAX_FRAME_LEN = 16 * 1024 * 1024;
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Per-(apiKey, host) session/cascade ID cache. Cloud uses these for
|
|
56
|
+
* server-side context caching across turns of the same conversation; if we
|
|
57
|
+
* mint a fresh sessionId on every call (which we used to), every turn looks
|
|
58
|
+
* like a brand-new session and the prompt-cache hit ratio is zero.
|
|
59
|
+
* Single-process scope is enough: opencode lives in one runtime for a TUI
|
|
60
|
+
* session, and CLI one-shots don't benefit from caching anyway.
|
|
61
|
+
*/
|
|
62
|
+
interface SessionIds {
|
|
63
|
+
sessionId: string;
|
|
64
|
+
cascadeId: string;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* Bounded the same way the adapter bounds its cascade-id map: a long-running
|
|
68
|
+
* proxy sees one entry per (host, api_key) pair, and nothing ever evicted them.
|
|
69
|
+
*/
|
|
70
|
+
const SESSION_CACHE_MAX = 256;
|
|
71
|
+
const sessionCache = new Map<string, SessionIds>();
|
|
72
|
+
function getOrAllocateSessionIds(apiKey: string, host: string, cascadeIdOverride?: string): SessionIds {
|
|
73
|
+
const key = `${host}\x1f${apiKey}`;
|
|
74
|
+
let ids = sessionCache.get(key);
|
|
75
|
+
if (!ids) {
|
|
76
|
+
ids = {
|
|
77
|
+
sessionId: crypto.randomUUID(),
|
|
78
|
+
cascadeId: cascadeIdOverride ?? allocateCascadeId(),
|
|
79
|
+
};
|
|
80
|
+
if (sessionCache.size >= SESSION_CACHE_MAX) {
|
|
81
|
+
const oldest = sessionCache.keys().next().value;
|
|
82
|
+
if (oldest !== undefined) sessionCache.delete(oldest);
|
|
83
|
+
}
|
|
84
|
+
sessionCache.set(key, ids);
|
|
85
|
+
} else if (cascadeIdOverride && ids.cascadeId !== cascadeIdOverride) {
|
|
86
|
+
// Caller explicitly requested a different cascadeId — honor it.
|
|
87
|
+
ids = { sessionId: ids.sessionId, cascadeId: cascadeIdOverride };
|
|
88
|
+
sessionCache.set(key, ids);
|
|
89
|
+
}
|
|
90
|
+
return ids;
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
/** Drop the cached session IDs — call after logout so a new sign-in starts fresh. */
|
|
94
|
+
export function clearSessionIds(): void {
|
|
95
|
+
sessionCache.clear();
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// ----------------------------------------------------------------------------
|
|
99
|
+
// Per-conversation cascade state — generated client-side; cloud lazy-registers
|
|
100
|
+
// ----------------------------------------------------------------------------
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Allocate a fresh cascade UUID. The cloud lazy-registers cascade_id on first
|
|
104
|
+
* use — confirmed empirically (random UUID accepted, model responded). One
|
|
105
|
+
* cascade_id per opencode-CLI conversation is fine; reuse across turns to
|
|
106
|
+
* preserve server-side context.
|
|
107
|
+
*/
|
|
108
|
+
export function allocateCascadeId(): string {
|
|
109
|
+
return crypto.randomUUID();
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// ----------------------------------------------------------------------------
|
|
113
|
+
// Request encoders
|
|
114
|
+
// ----------------------------------------------------------------------------
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* ChatMessagePrompt {
|
|
118
|
+
* #2 source: enum CHAT_MESSAGE_SOURCE_USER=1 / ASSISTANT=2 / SYSTEM=3 / TOOL=4
|
|
119
|
+
* #3 prompt: string (text content)
|
|
120
|
+
* #4 num_tokens: int (rough estimate)
|
|
121
|
+
* #5 safe_for_code_telemetry: bool (1 = ok to log)
|
|
122
|
+
* #10 images: repeated ImageData (multimodal)
|
|
123
|
+
* }
|
|
124
|
+
*
|
|
125
|
+
* ImageData (exa.codeium_common_pb.ImageData) {
|
|
126
|
+
* #1 base64_data: string
|
|
127
|
+
* #2 mime_type: string
|
|
128
|
+
* #3 caption: string (optional)
|
|
129
|
+
* }
|
|
130
|
+
*/
|
|
131
|
+
function encodeImageData(img: { mimeType: string; base64Data: string; caption?: string }): Buffer {
|
|
132
|
+
const parts: Buffer[] = [
|
|
133
|
+
encodeString(1, img.base64Data),
|
|
134
|
+
encodeString(2, img.mimeType),
|
|
135
|
+
];
|
|
136
|
+
if (img.caption) parts.push(encodeString(3, img.caption));
|
|
137
|
+
return Buffer.concat(parts);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Encode one ChatToolCall sub-message:
|
|
142
|
+
* {#1 id, #2 name, #3 arguments_json}
|
|
143
|
+
* Verified against `exa.codeium_common_pb.ChatToolCall` from extension.js.
|
|
144
|
+
*/
|
|
145
|
+
function encodeChatToolCall(tc: { id: string; name: string; arguments: string }): Buffer {
|
|
146
|
+
return Buffer.concat([
|
|
147
|
+
encodeString(1, tc.id),
|
|
148
|
+
encodeString(2, tc.name),
|
|
149
|
+
encodeString(3, tc.arguments),
|
|
150
|
+
]);
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
function encodeChatMessagePrompt(
|
|
154
|
+
content: ContentPart[],
|
|
155
|
+
source: number,
|
|
156
|
+
opts?: { toolCallId?: string; toolCalls?: Array<{ id: string; name: string; arguments: string }> },
|
|
157
|
+
): Buffer {
|
|
158
|
+
const textParts = content.filter((p): p is { type: 'text'; text: string } => p.type === 'text');
|
|
159
|
+
const imageParts = content.filter((p): p is { type: 'image'; mimeType: string; base64Data: string; caption?: string } => p.type === 'image');
|
|
160
|
+
const joined = textParts.map((p) => p.text).join('\n');
|
|
161
|
+
const parts: Buffer[] = [
|
|
162
|
+
// #1 message_id. The verified turn-1 capture stamps one on every prompt.
|
|
163
|
+
encodeString(1, crypto.randomUUID()),
|
|
164
|
+
encodeVarintField(2, source),
|
|
165
|
+
encodeString(3, joined),
|
|
166
|
+
];
|
|
167
|
+
// Tool-result message: attach the id of the call this result answers.
|
|
168
|
+
// Without it, the model can't pair multi-tool conversations.
|
|
169
|
+
if (opts?.toolCallId) {
|
|
170
|
+
parts.push(encodeString(7, opts.toolCallId));
|
|
171
|
+
}
|
|
172
|
+
// Assistant message with tool_calls: encode each as a ChatToolCall.
|
|
173
|
+
if (opts?.toolCalls && opts.toolCalls.length > 0) {
|
|
174
|
+
for (const tc of opts.toolCalls) {
|
|
175
|
+
parts.push(encodeMessage(6, encodeChatToolCall(tc)));
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
for (const img of imageParts) {
|
|
179
|
+
parts.push(encodeMessage(10, encodeImageData(img)));
|
|
180
|
+
}
|
|
181
|
+
return Buffer.concat(parts);
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
const SOURCE_BY_ROLE: Record<string, number> = {
|
|
185
|
+
user: 1,
|
|
186
|
+
assistant: 2,
|
|
187
|
+
// NOTE: do not send source=3 (SYSTEM) directly — the Codeium chat backend
|
|
188
|
+
// returns "third-party model provider is experiencing issues" when any
|
|
189
|
+
// ChatMessagePrompt has source=SYSTEM. The captured LS upstream traffic
|
|
190
|
+
// shows the IDE inlines system context into the *user* prompt (source=1)
|
|
191
|
+
// wrapped in <additional_metadata>...</additional_metadata>. We collapse
|
|
192
|
+
// role:'system' messages into the next user turn before building the
|
|
193
|
+
// proto — see `collapseSystemIntoUser` below.
|
|
194
|
+
system: 1,
|
|
195
|
+
tool: 4,
|
|
196
|
+
};
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Collapse OpenAI-style messages so all `role:'system'` entries are inlined
|
|
200
|
+
* into the immediately-following user message, matching the wire format the
|
|
201
|
+
* IDE uses. Cognition's chat backend rejects raw role=system entries.
|
|
202
|
+
*
|
|
203
|
+
* [{system: "S1"}, {system: "S2"}, {user: "U1"}, {assistant: "A1"}, {user: "U2"}]
|
|
204
|
+
*
|
|
205
|
+
* becomes
|
|
206
|
+
*
|
|
207
|
+
* [{user: "<system>\nS1\nS2\n</system>\nU1"}, {assistant: "A1"}, {user: "U2"}]
|
|
208
|
+
*
|
|
209
|
+
* If there's no following user message, the trailing system messages get
|
|
210
|
+
* appended as a synthesized user turn.
|
|
211
|
+
*/
|
|
212
|
+
function collapseSystemIntoUser(messages: ChatHistoryItem[]): ChatHistoryItem[] {
|
|
213
|
+
const out: ChatHistoryItem[] = [];
|
|
214
|
+
let pendingSystem: string[] = [];
|
|
215
|
+
|
|
216
|
+
const flushTextOf = (content: ContentPart[]): string =>
|
|
217
|
+
content.filter((p): p is { type: 'text'; text: string } => p.type === 'text')
|
|
218
|
+
.map((p) => p.text).join('\n');
|
|
219
|
+
|
|
220
|
+
for (const m of messages) {
|
|
221
|
+
if (m.role === 'system') {
|
|
222
|
+
const parts = normalizeContent(m.content);
|
|
223
|
+
const text = flushTextOf(parts);
|
|
224
|
+
if (text) pendingSystem.push(text);
|
|
225
|
+
} else if (m.role === 'user' && pendingSystem.length > 0) {
|
|
226
|
+
const userParts = normalizeContent(m.content);
|
|
227
|
+
const userText = flushTextOf(userParts);
|
|
228
|
+
const userImages = userParts.filter((p) => p.type === 'image');
|
|
229
|
+
const wrapped = `<system>\n${pendingSystem.join('\n\n')}\n</system>\n${userText}`;
|
|
230
|
+
const newContent: ContentPart[] = [{ type: 'text', text: wrapped }, ...userImages];
|
|
231
|
+
out.push({ role: 'user', content: newContent });
|
|
232
|
+
pendingSystem = [];
|
|
233
|
+
} else {
|
|
234
|
+
// Flush accumulated system text before any non-system, non-user turn
|
|
235
|
+
// (assistant / tool) so system instructions keep their leading position
|
|
236
|
+
// instead of being deferred to a trailing synthesized user message.
|
|
237
|
+
if (pendingSystem.length > 0) {
|
|
238
|
+
out.push({
|
|
239
|
+
role: 'user',
|
|
240
|
+
content: [{ type: 'text', text: `<system>\n${pendingSystem.join('\n\n')}\n</system>` }],
|
|
241
|
+
});
|
|
242
|
+
pendingSystem = [];
|
|
243
|
+
}
|
|
244
|
+
out.push(m);
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
if (pendingSystem.length > 0) {
|
|
248
|
+
// Trailing system messages with no following user turn — convert to a
|
|
249
|
+
// standalone user message so they still reach the model.
|
|
250
|
+
out.push({
|
|
251
|
+
role: 'user',
|
|
252
|
+
content: [{ type: 'text', text: `<system>\n${pendingSystem.join('\n\n')}\n</system>` }],
|
|
253
|
+
});
|
|
254
|
+
}
|
|
255
|
+
return out;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
/**
|
|
259
|
+
* CompletionConfiguration — mirrors the LS-shipped defaults, lets the caller
|
|
260
|
+
* override the obvious knobs.
|
|
261
|
+
*/
|
|
262
|
+
/** Output cap when the caller named none. */
|
|
263
|
+
const DEFAULT_MAX_OUTPUT_TOKENS = 8192;
|
|
264
|
+
/** Context window when the caller named none. */
|
|
265
|
+
const DEFAULT_CONTEXT_WINDOW = 128_000;
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* Cognition rejects a temperature of exactly 0 with the same opaque internal
|
|
269
|
+
* error it uses for a malformed request, so a client asking for deterministic
|
|
270
|
+
* output would fail every turn. Clamp to the smallest value the wire accepts
|
|
271
|
+
* rather than silently substituting the service default, which would be a
|
|
272
|
+
* different answer than the caller asked for.
|
|
273
|
+
*/
|
|
274
|
+
const MIN_TEMPERATURE = 0.0001;
|
|
275
|
+
|
|
276
|
+
function safeTemperature(value: number | undefined): number {
|
|
277
|
+
if (value === undefined) return 0.7;
|
|
278
|
+
return value <= 0 ? MIN_TEMPERATURE : value;
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
function encodeCompletionConfiguration(opts: {
|
|
282
|
+
maxOutputTokens?: number;
|
|
283
|
+
maxInputTokens?: number;
|
|
284
|
+
temperature?: number;
|
|
285
|
+
topK?: number;
|
|
286
|
+
topP?: number;
|
|
287
|
+
}): Buffer {
|
|
288
|
+
const enc64 = (fieldNum: number, n: number): Buffer => {
|
|
289
|
+
const b = Buffer.alloc(8);
|
|
290
|
+
b.writeDoubleLE(n, 0);
|
|
291
|
+
return Buffer.concat([Buffer.from([(fieldNum << 3) | 1]), b]);
|
|
292
|
+
};
|
|
293
|
+
// Tag map, verified by building the same turn with a working client and
|
|
294
|
+
// diffing the encoded messages field by field: #2 is the OUTPUT cap and #3 is
|
|
295
|
+
// the context window. This layout had those two swapped, so a caller asking
|
|
296
|
+
// for 32 output tokens put 32 into the context-window field and the request
|
|
297
|
+
// came back as an opaque "an internal error occurred" — for every turn, on
|
|
298
|
+
// every account, which is why free and paid failed identically. #6 and #11
|
|
299
|
+
// are not part of the message the service accepts.
|
|
300
|
+
return Buffer.concat([
|
|
301
|
+
encodeVarintField(1, 1),
|
|
302
|
+
encodeVarintField(2, opts.maxOutputTokens ?? DEFAULT_MAX_OUTPUT_TOKENS),
|
|
303
|
+
encodeVarintField(3, opts.maxInputTokens ?? DEFAULT_CONTEXT_WINDOW),
|
|
304
|
+
enc64(5, safeTemperature(opts.temperature)),
|
|
305
|
+
encodeVarintField(7, opts.topK ?? 40),
|
|
306
|
+
enc64(8, opts.topP ?? 1.0),
|
|
307
|
+
]);
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/**
|
|
311
|
+
* Multimodal content part — text or image.
|
|
312
|
+
*
|
|
313
|
+
* Text: `{ type: 'text', text: '...' }`
|
|
314
|
+
* Image: `{ type: 'image', mimeType: 'image/png', base64Data: '...' [, caption: '...'] }`
|
|
315
|
+
*
|
|
316
|
+
* Matches the OpenAI/@ai-sdk multimodal message shape — we accept their
|
|
317
|
+
* `image_url: { url: 'data:image/png;base64,...' }` form via {@link parseContent}.
|
|
318
|
+
*/
|
|
319
|
+
export type ContentPart =
|
|
320
|
+
| { type: 'text'; text: string }
|
|
321
|
+
| { type: 'image'; mimeType: string; base64Data: string; caption?: string };
|
|
322
|
+
|
|
323
|
+
export interface ChatHistoryItem {
|
|
324
|
+
role: 'user' | 'assistant' | 'system' | 'tool';
|
|
325
|
+
/**
|
|
326
|
+
* Either a plain string or an array of {@link ContentPart}. Plain strings are
|
|
327
|
+
* shorthand for `[{ type: 'text', text: '...' }]`.
|
|
328
|
+
*/
|
|
329
|
+
content: string | ContentPart[];
|
|
330
|
+
/**
|
|
331
|
+
* For `role: 'tool'` only — the id of the assistant's preceding tool_call
|
|
332
|
+
* this message answers. Required by the cloud's chat backend to pair
|
|
333
|
+
* tool results with calls; without it, multi-tool conversations can't
|
|
334
|
+
* tell the model which call produced which result. Encoded as
|
|
335
|
+
* ChatMessagePrompt field #7 (verified against the Windsurf bundled
|
|
336
|
+
* extension.js proto schema `exa.chat_pb.ChatMessagePrompt`).
|
|
337
|
+
*/
|
|
338
|
+
tool_call_id?: string;
|
|
339
|
+
/**
|
|
340
|
+
* For `role: 'assistant'` only — the tool calls the assistant emitted.
|
|
341
|
+
* Encoded as ChatMessagePrompt field #6 (repeated ChatToolCall, where
|
|
342
|
+
* each ChatToolCall has #1 id, #2 name, #3 arguments_json).
|
|
343
|
+
*/
|
|
344
|
+
tool_calls?: Array<{ id: string; name: string; arguments: string }>;
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
/**
|
|
348
|
+
* Normalize ChatHistoryItem content into structured parts. Accepts strings,
|
|
349
|
+
* OpenAI multimodal `[{type:'text',text}, {type:'image_url',image_url}]`, and
|
|
350
|
+
* our own `[{type:'image', mimeType, base64Data}]`.
|
|
351
|
+
*/
|
|
352
|
+
function normalizeContent(content: string | ContentPart[] | unknown): ContentPart[] {
|
|
353
|
+
if (typeof content === 'string') return [{ type: 'text', text: content }];
|
|
354
|
+
if (!Array.isArray(content)) return [];
|
|
355
|
+
const out: ContentPart[] = [];
|
|
356
|
+
// Each element may follow our own ContentPart shape, the OpenAI multimodal
|
|
357
|
+
// `image_url` shape, or be malformed — narrow defensively per branch.
|
|
358
|
+
const parts = content as Array<Record<string, unknown>>;
|
|
359
|
+
for (const p of parts) {
|
|
360
|
+
if (!p || typeof p !== 'object') continue;
|
|
361
|
+
if (p.type === 'text' && typeof p.text === 'string') {
|
|
362
|
+
out.push({ type: 'text', text: p.text });
|
|
363
|
+
} else if (p.type === 'image' && typeof p.base64Data === 'string') {
|
|
364
|
+
const mimeType = typeof p.mimeType === 'string' ? p.mimeType : 'image/png';
|
|
365
|
+
const caption = typeof p.caption === 'string' ? p.caption : undefined;
|
|
366
|
+
out.push({ type: 'image', mimeType, base64Data: p.base64Data, caption });
|
|
367
|
+
} else if (p.type === 'image_url' && p.image_url) {
|
|
368
|
+
// OpenAI/@ai-sdk shape — parse data: URL into base64 + mime.
|
|
369
|
+
const imgRef = p.image_url as string | { url?: string };
|
|
370
|
+
const url: string = typeof imgRef === 'string' ? imgRef : (imgRef.url ?? '');
|
|
371
|
+
const m = url.match(/^data:([^;]+);base64,(.+)$/);
|
|
372
|
+
if (m) out.push({ type: 'image', mimeType: m[1], base64Data: m[2] });
|
|
373
|
+
else if (url) out.push({ type: 'text', text: `[image url: ${url}]` });
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
return out;
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
export interface ToolDef {
|
|
380
|
+
/** Function name. */
|
|
381
|
+
name: string;
|
|
382
|
+
/** Plain-English description. */
|
|
383
|
+
description: string;
|
|
384
|
+
/** JSON Schema for the function's arguments. */
|
|
385
|
+
parameters: unknown;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/**
|
|
389
|
+
* Streaming event emitted by the cloud-direct chat loop.
|
|
390
|
+
*
|
|
391
|
+
* - `text` : incremental visible content from the assistant
|
|
392
|
+
* - `reasoning` : incremental internal thinking (Anthropic-style, kept
|
|
393
|
+
* separate from visible content; @ai-sdk consumers can
|
|
394
|
+
* render in a collapsed/grey region)
|
|
395
|
+
* - `tool_call_*` : function-calling deltas (id+name once, args streamed)
|
|
396
|
+
* - `finish` : stream terminated cleanly with a reason
|
|
397
|
+
* - `usage` : final token-accounting block (input/output/total counts)
|
|
398
|
+
*/
|
|
399
|
+
export type CloudChatEvent =
|
|
400
|
+
| { kind: 'text'; text: string }
|
|
401
|
+
| { kind: 'reasoning'; text: string }
|
|
402
|
+
| { kind: 'tool_call_start'; id: string; name: string }
|
|
403
|
+
| {
|
|
404
|
+
kind: 'tool_call_args';
|
|
405
|
+
argsDelta: string;
|
|
406
|
+
/**
|
|
407
|
+
* Tool-call id this delta belongs to, when the cloud surfaced one in
|
|
408
|
+
* this frame. Cognition's wire format only carries id on the START
|
|
409
|
+
* frame today, so most argsDelta events arrive without one — callers
|
|
410
|
+
* route those to the most-recent-start by convention. If Cognition
|
|
411
|
+
* ever interleaves args across calls, the consumer should prefer
|
|
412
|
+
* `id` over the rolling lastToolCallId.
|
|
413
|
+
*/
|
|
414
|
+
id?: string;
|
|
415
|
+
}
|
|
416
|
+
// Note: there is no `tool_call_end` event. Cognition's wire format
|
|
417
|
+
// signals the end of a tool call implicitly — args just stop arriving
|
|
418
|
+
// for the current id and either a new `tool_call_start` fires or the
|
|
419
|
+
// stream finishes. Consumers should treat each `tool_call_start` as
|
|
420
|
+
// ending the previous call.
|
|
421
|
+
| { kind: 'finish'; reason: 'stop' | 'tool_calls' | 'length' | 'content_filter' }
|
|
422
|
+
| {
|
|
423
|
+
kind: 'usage';
|
|
424
|
+
promptTokens?: number;
|
|
425
|
+
completionTokens?: number;
|
|
426
|
+
totalTokens?: number;
|
|
427
|
+
/**
|
|
428
|
+
* Tokens served from the cache. Surfaced separately so callers tracking
|
|
429
|
+
* cost can distinguish them from fresh input tokens (Anthropic / OpenAI
|
|
430
|
+
* both bill cache reads cheaper than fresh prompts).
|
|
431
|
+
*/
|
|
432
|
+
cachedInputTokens?: number;
|
|
433
|
+
/** Tokens written to the cache on this request (Anthropic-style). */
|
|
434
|
+
cacheCreationInputTokens?: number;
|
|
435
|
+
/** Reasoning tokens (gpt-5-x reasoning models, Claude thinking variants). */
|
|
436
|
+
reasoningTokens?: number;
|
|
437
|
+
};
|
|
438
|
+
|
|
439
|
+
interface BuildArgs {
|
|
440
|
+
apiKey: string;
|
|
441
|
+
userJwt?: string;
|
|
442
|
+
modelUid: string;
|
|
443
|
+
messages: ChatHistoryItem[];
|
|
444
|
+
cascadeId: string;
|
|
445
|
+
/**
|
|
446
|
+
* GetChatMessageRequest #22. Optional because it is omitted on a first turn;
|
|
447
|
+
* the working client only reuses one across a later tool loop.
|
|
448
|
+
*/
|
|
449
|
+
promptId?: string;
|
|
450
|
+
sessionId: string;
|
|
451
|
+
requestId: bigint;
|
|
452
|
+
triggerId: string;
|
|
453
|
+
tools?: ToolDef[];
|
|
454
|
+
/** Default 5 = CHAT_MESSAGE_REQUEST_TYPE_CASCADE (matches captured LS body). */
|
|
455
|
+
requestType?: number;
|
|
456
|
+
completionOpts?: {
|
|
457
|
+
maxOutputTokens?: number;
|
|
458
|
+
maxInputTokens?: number;
|
|
459
|
+
temperature?: number;
|
|
460
|
+
topK?: number;
|
|
461
|
+
topP?: number;
|
|
462
|
+
};
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
/**
|
|
466
|
+
* ChatToolDefinition proto, observed in the LS upstream traffic:
|
|
467
|
+
* { #1 name (string), #2 description (string), #3 parameters_schema (JSON string) }
|
|
468
|
+
*
|
|
469
|
+
* Truncation note: Codeium's tool validator rejects very long descriptions
|
|
470
|
+
* with a generic `failed_precondition: "Unable to process request due to an
|
|
471
|
+
* MCP configuration issue."` error. opencode ships some tools (notably `bash`)
|
|
472
|
+
* with ~9.6 KB descriptions packed with examples and rules. We truncate to a
|
|
473
|
+
* conservative `MAX_DESC_LEN` and append an ellipsis so the cloud accepts
|
|
474
|
+
* them. The model still gets the first chunk of the description (where the
|
|
475
|
+
* essential signature lives); detailed examples are sacrificed for
|
|
476
|
+
* compatibility.
|
|
477
|
+
*/
|
|
478
|
+
/**
|
|
479
|
+
* The Codeium tool validator rejects any tool whose description hits exactly
|
|
480
|
+
* 7,000 chars (or more) with a misleading `failed_precondition: "Unable to
|
|
481
|
+
* process request due to an MCP configuration issue."` error. Binary-search
|
|
482
|
+
* verified to char-precision:
|
|
483
|
+
* - 6,999 chars → server accepts
|
|
484
|
+
* - 7,000 chars → server returns MCP error
|
|
485
|
+
*
|
|
486
|
+
* The limit is per-description, content-sensitive (plain `a`-repeats up to
|
|
487
|
+
* 20K work fine; the bash description's exact byte at position 6999 trips
|
|
488
|
+
* it). We truncate to the maximum-1 (6,998) for a one-char safety margin.
|
|
489
|
+
*
|
|
490
|
+
* We do NOT need to aggregate-cap — 200K total tool descriptions across 200
|
|
491
|
+
* tools was confirmed to pass server-side. Only per-string length is gated.
|
|
492
|
+
*/
|
|
493
|
+
const MAX_TOOL_DESC_LEN = 6998;
|
|
494
|
+
|
|
495
|
+
/**
|
|
496
|
+
* Cognition's cloud enforces a case-sensitive, whitespace-exact exact-phrase
|
|
497
|
+
* blocklist on tool descriptions. Binary-search isolated the trigger to the
|
|
498
|
+
* 7-word phrase "Takes a task_id parameter identifying the task" — verbatim,
|
|
499
|
+
* capital T, single spaces — which causes a `permission_denied` trailer error
|
|
500
|
+
* regardless of model or account tier. Any deviation (lowercase, reword,
|
|
501
|
+
* reorder, extra whitespace) passes. The phrase appears verbatim in Claude
|
|
502
|
+
* Code's built-in TaskOutput tool description.
|
|
503
|
+
*
|
|
504
|
+
* Rewrite known triggers to meaning-preserving forms. This is a
|
|
505
|
+
* Cognition-specific constraint alongside the length limit above; if
|
|
506
|
+
* Cognition adds more blocklisted phrases, extend this table and add a
|
|
507
|
+
* regression test in tests/devin-adapter.test.ts.
|
|
508
|
+
*
|
|
509
|
+
* Not every entry is matched the same way. The Claude Code phrase above is
|
|
510
|
+
* case-sensitive and whitespace-exact, but the two Codex entries below are
|
|
511
|
+
* not: against a live account, lowercasing the first word and doubling an
|
|
512
|
+
* interior space both still produced `permission_denied`, while changing any
|
|
513
|
+
* single word passed. So those two match case-insensitively with flexible
|
|
514
|
+
* whitespace and an optional comma, and the rewrite swaps only the leading
|
|
515
|
+
* verb — the smallest edit measured to clear the filter.
|
|
516
|
+
*
|
|
517
|
+
* These two sentences are Codex's own built-in `exec_command` and
|
|
518
|
+
* `write_stdin` descriptions, verbatim. Every Codex turn carries them, so
|
|
519
|
+
* before this table knew about them the cloud refused literally every request
|
|
520
|
+
* from a Codex client — a bare "hi" included — while the same account
|
|
521
|
+
* answered a hand-built request with an ordinary shell tool. The visible
|
|
522
|
+
* symptom was the adapter's own blocklist message pointing back at this
|
|
523
|
+
* table, which is why they are named here rather than left to the next person
|
|
524
|
+
* to re-bisect.
|
|
525
|
+
*/
|
|
526
|
+
const COGNITION_BLOCKLIST_REWRITES: ReadonlyArray<[RegExp, string]> = [
|
|
527
|
+
[/\bTakes a task_id parameter identifying the task\b/g, "Accepts a task_id parameter identifying the task"],
|
|
528
|
+
[
|
|
529
|
+
/\bRuns\s+a\s+command\s+in\s+a\s+PTY,?\s+returning\s+output\s+or\s+a\s+session\s+ID\s+for\s+ongoing\s+interaction\b/gi,
|
|
530
|
+
"Executes a command in a PTY, returning output or a session ID for ongoing interaction",
|
|
531
|
+
],
|
|
532
|
+
[
|
|
533
|
+
/\bWrites\s+characters\s+to\s+an\s+existing\s+unified\s+exec\s+session\s+and\s+returns\s+recent\s+output\b/gi,
|
|
534
|
+
"Sends characters to an existing unified exec session and returns recent output",
|
|
535
|
+
],
|
|
536
|
+
];
|
|
537
|
+
|
|
538
|
+
function sanitizeToolDescriptionForCognition(description: string): string {
|
|
539
|
+
let out = description;
|
|
540
|
+
for (const [pattern, replacement] of COGNITION_BLOCKLIST_REWRITES) {
|
|
541
|
+
out = out.replace(pattern, replacement);
|
|
542
|
+
}
|
|
543
|
+
return out;
|
|
544
|
+
}
|
|
545
|
+
|
|
546
|
+
/** Test-only: exercise the Cognition blocklist rewrite directly. */
|
|
547
|
+
export function sanitizeToolDescriptionForCognitionForTests(description: string): string {
|
|
548
|
+
return sanitizeToolDescriptionForCognition(description);
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
function encodeToolDef(tool: ToolDef): Buffer {
|
|
552
|
+
const rawDesc = sanitizeToolDescriptionForCognition(tool.description ?? '');
|
|
553
|
+
const desc =
|
|
554
|
+
rawDesc.length > MAX_TOOL_DESC_LEN
|
|
555
|
+
? rawDesc.slice(0, MAX_TOOL_DESC_LEN - 24) + '\n…(truncated for cloud)'
|
|
556
|
+
: rawDesc;
|
|
557
|
+
return Buffer.concat([
|
|
558
|
+
encodeString(1, tool.name),
|
|
559
|
+
encodeString(2, desc),
|
|
560
|
+
encodeString(3, JSON.stringify(tool.parameters ?? {})),
|
|
561
|
+
]);
|
|
562
|
+
}
|
|
563
|
+
|
|
564
|
+
export function buildGetChatMessageRequestForTests(args: BuildArgs): Buffer {
|
|
565
|
+
return buildGetChatMessageRequest(args);
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
function buildGetChatMessageRequest(args: BuildArgs): Buffer {
|
|
569
|
+
const metadata = buildMetadata({
|
|
570
|
+
apiKey: args.apiKey,
|
|
571
|
+
userJwt: args.userJwt,
|
|
572
|
+
sessionId: args.sessionId,
|
|
573
|
+
requestId: args.requestId,
|
|
574
|
+
triggerId: args.triggerId,
|
|
575
|
+
// GetChatMessage accepts only the calibrated identity shape.
|
|
576
|
+
cloudChatShape: true,
|
|
577
|
+
});
|
|
578
|
+
|
|
579
|
+
// System messages must be inlined into the user turn (Cognition cloud
|
|
580
|
+
// rejects source=3). See `collapseSystemIntoUser` for the format.
|
|
581
|
+
const collapsed = collapseSystemIntoUser(args.messages);
|
|
582
|
+
const promptParts = collapsed.map((m) =>
|
|
583
|
+
encodeMessage(
|
|
584
|
+
3,
|
|
585
|
+
encodeChatMessagePrompt(
|
|
586
|
+
normalizeContent(m.content),
|
|
587
|
+
SOURCE_BY_ROLE[m.role] ?? 1,
|
|
588
|
+
// Thread tool_call_id (for tool results) + tool_calls (for assistant
|
|
589
|
+
// turns that fired tools) into the proto. Cloud rejects multi-tool
|
|
590
|
+
// conversations otherwise — it can't pair a tool result with the
|
|
591
|
+
// assistant call that produced it.
|
|
592
|
+
{
|
|
593
|
+
toolCallId: m.role === 'tool' ? m.tool_call_id : undefined,
|
|
594
|
+
toolCalls: m.role === 'assistant' ? m.tool_calls : undefined,
|
|
595
|
+
},
|
|
596
|
+
),
|
|
597
|
+
),
|
|
598
|
+
);
|
|
599
|
+
|
|
600
|
+
const completion = encodeCompletionConfiguration(args.completionOpts ?? {});
|
|
601
|
+
|
|
602
|
+
const toolParts: Buffer[] = (args.tools ?? []).map((t) =>
|
|
603
|
+
encodeMessage(10, encodeToolDef(t)),
|
|
604
|
+
);
|
|
605
|
+
|
|
606
|
+
// Field layout from mitm capture of the LS:
|
|
607
|
+
// #1 metadata
|
|
608
|
+
// #3 chat_message_prompts (repeated — one element per history turn)
|
|
609
|
+
// #7 request_type (varint enum)
|
|
610
|
+
// #8 completion_configuration
|
|
611
|
+
// #10 tools (repeated ChatToolDefinition)
|
|
612
|
+
// #16 cascade_id (string)
|
|
613
|
+
// #21 chat_model_uid (string)
|
|
614
|
+
// #22 prompt_id (string)
|
|
615
|
+
return Buffer.concat([
|
|
616
|
+
encodeMessage(1, metadata),
|
|
617
|
+
// #2 system_prompt is always written, empty when the caller had none. The
|
|
618
|
+
// system turn is separately collapsed into the first user message because
|
|
619
|
+
// source=SYSTEM is refused; this field is the one the wire expects here.
|
|
620
|
+
encodeString(2, ''),
|
|
621
|
+
...promptParts,
|
|
622
|
+
encodeVarintField(7, args.requestType ?? 5),
|
|
623
|
+
encodeMessage(8, completion),
|
|
624
|
+
...toolParts,
|
|
625
|
+
// #15 session model config: { id, turn, 4 }. Present on every verified
|
|
626
|
+
// request.
|
|
627
|
+
encodeMessage(15, Buffer.concat([
|
|
628
|
+
encodeString(1, crypto.randomUUID()),
|
|
629
|
+
encodeVarintField(2, 1),
|
|
630
|
+
encodeVarintField(3, 4),
|
|
631
|
+
])),
|
|
632
|
+
encodeString(16, args.cascadeId),
|
|
633
|
+
encodeVarintField(20, 1),
|
|
634
|
+
encodeString(21, args.modelUid),
|
|
635
|
+
// #22 is deliberately omitted. It is a user-exchange id that only appears
|
|
636
|
+
// from the second turn onward and is reused across that turn's tool loop; a
|
|
637
|
+
// fresh per-request uuid matches neither shape.
|
|
638
|
+
]);
|
|
639
|
+
}
|
|
640
|
+
|
|
641
|
+
// ----------------------------------------------------------------------------
|
|
642
|
+
// Response parsing — pull `delta_text` (top-level field #9) out of each frame
|
|
643
|
+
// ----------------------------------------------------------------------------
|
|
644
|
+
|
|
645
|
+
/**
|
|
646
|
+
* Decode a single streaming ChatMessage proto frame into one or more
|
|
647
|
+
* CloudChatEvents. Captured shape (from a tool-using swe-1.6 chat):
|
|
648
|
+
*
|
|
649
|
+
* ChatMessage {
|
|
650
|
+
* #1 bot_id (string)
|
|
651
|
+
* #2 timestamp { seconds, nanos }
|
|
652
|
+
* #5 finish_reason (varint — 10 = "tool_calls" observed, others unknown)
|
|
653
|
+
* #6 ToolCallDelta {
|
|
654
|
+
* #1 id (string, only on first tool-call frame)
|
|
655
|
+
* #2 name (string, only on first tool-call frame)
|
|
656
|
+
* #3 arguments_delta (string, JSON fragment, streamed)
|
|
657
|
+
* }
|
|
658
|
+
* #7 ChatStatus { #6 status_code, #9 model_name }
|
|
659
|
+
* #9 delta_text (string)
|
|
660
|
+
* #12 (fixed64) some_hash
|
|
661
|
+
* #17 (string) message_uuid
|
|
662
|
+
* #28 UsageStats { #1 label, ... }
|
|
663
|
+
* }
|
|
664
|
+
*
|
|
665
|
+
* #9 appears both at top-level (text delta) AND inside #7 (model_name).
|
|
666
|
+
* iterFields walks top-level only, so we don't confuse the two.
|
|
667
|
+
*
|
|
668
|
+
* #5 is the finish_reason. Observed value `10` = tool_calls finish. We map
|
|
669
|
+
* any non-zero to 'tool_calls' for now (and let the caller fall back to
|
|
670
|
+
* 'stop' if no tool_call deltas were emitted).
|
|
671
|
+
*/
|
|
672
|
+
function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
|
|
673
|
+
for (const f of iterFields(proto)) {
|
|
674
|
+
if (f.num === 3 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
675
|
+
// Visible delta_text — what the user should SEE in the chat.
|
|
676
|
+
//
|
|
677
|
+
// We previously had this mapping inverted (#3 = thinking, #9 = visible),
|
|
678
|
+
// which produced two compounding bugs in the TUI:
|
|
679
|
+
// 1. The model's CoT was rendered as plain content, so the user saw
|
|
680
|
+
// "The user wants me to X..." instead of the answer.
|
|
681
|
+
// 2. The actual answer (which lives in #3) was silently dropped — so
|
|
682
|
+
// the assistant turn appeared to end after the CoT with nothing
|
|
683
|
+
// after, matching the "model wrote reasoning then went silent"
|
|
684
|
+
// symptom the user reported.
|
|
685
|
+
// Verified live: prompted swe-1.6 with "explain then answer 2+2"; #3
|
|
686
|
+
// streamed "2+2=4 because... 4" while #9 streamed the meta-narration
|
|
687
|
+
// "The user wants me to perform a reasoning task...".
|
|
688
|
+
const s = (f.value as Buffer).toString('utf8');
|
|
689
|
+
if (s) yield { kind: 'text', text: s };
|
|
690
|
+
} else if (f.num === 9 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
691
|
+
// Internal thinking / chain-of-thought. Surface as `reasoning` so
|
|
692
|
+
// @ai-sdk consumers (opencode TUI) render it in a collapsed grey
|
|
693
|
+
// block instead of inline with the answer.
|
|
694
|
+
const s = (f.value as Buffer).toString('utf8');
|
|
695
|
+
if (s) yield { kind: 'reasoning', text: s };
|
|
696
|
+
} else if (f.num === 6 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
697
|
+
let id: string | undefined;
|
|
698
|
+
let name: string | undefined;
|
|
699
|
+
let argsDelta: string | undefined;
|
|
700
|
+
for (const sf of iterFields(f.value as Buffer)) {
|
|
701
|
+
if (sf.wire === 2 && Buffer.isBuffer(sf.value)) {
|
|
702
|
+
const s = (sf.value as Buffer).toString('utf8');
|
|
703
|
+
if (sf.num === 1) id = s;
|
|
704
|
+
else if (sf.num === 2) name = s;
|
|
705
|
+
else if (sf.num === 3) argsDelta = s;
|
|
706
|
+
}
|
|
707
|
+
}
|
|
708
|
+
if (id !== undefined && name !== undefined) {
|
|
709
|
+
yield { kind: 'tool_call_start', id, name };
|
|
710
|
+
}
|
|
711
|
+
if (argsDelta !== undefined) {
|
|
712
|
+
// Pass through `id` when this frame carries one (Cognition only
|
|
713
|
+
// sets it on the start frame today, but defending against future
|
|
714
|
+
// interleaving). Callers should prefer `id` over their rolling
|
|
715
|
+
// lastToolCallId when both are available.
|
|
716
|
+
yield { kind: 'tool_call_args', argsDelta, ...(id !== undefined ? { id } : {}) };
|
|
717
|
+
}
|
|
718
|
+
} else if (f.num === 5 && f.wire === 0) {
|
|
719
|
+
const v = Number(f.value);
|
|
720
|
+
// exa.codeium_common_pb.StopReason → OpenAI finish_reason.
|
|
721
|
+
// Source of truth: Windsurf extension.js sets `setEnumType("StopReason", [...])`
|
|
722
|
+
// 0 UNSPECIFIED → "stop" (no signal — treat as natural end)
|
|
723
|
+
// 1 INCOMPLETE → "length" (request cut short, model wanted more)
|
|
724
|
+
// 2 STOP_PATTERN → "stop" (model emitted its stop sequence — NORMAL)
|
|
725
|
+
// 3 MAX_TOKENS → "length"
|
|
726
|
+
// 4-9 internal → "stop"
|
|
727
|
+
// 10 FUNCTION_CALL → "tool_calls"
|
|
728
|
+
// 11 CONTENT_FILTER → "content_filter"
|
|
729
|
+
// 12 NON_INSERTION → "stop"
|
|
730
|
+
// 13 ERROR → "stop" (errors come as Connect trailer, not via this)
|
|
731
|
+
//
|
|
732
|
+
// We had 2 and 3 swapped previously, which made the model's normal
|
|
733
|
+
// STOP_PATTERN look like "length" → @ai-sdk treated complete responses
|
|
734
|
+
// as truncated. That was the "model wrote reasoning then went silent"
|
|
735
|
+
// symptom the user kept hitting.
|
|
736
|
+
let reason: 'stop' | 'tool_calls' | 'length' | 'content_filter' = 'stop';
|
|
737
|
+
if (v === 10) reason = 'tool_calls';
|
|
738
|
+
else if (v === 11) reason = 'content_filter';
|
|
739
|
+
else if (v === 1 || v === 3) reason = 'length';
|
|
740
|
+
// else stays 'stop' for 0/2/4-9/12/13
|
|
741
|
+
yield { kind: 'finish', reason };
|
|
742
|
+
} else if (f.num === 28 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
743
|
+
const usage = decodeUsageBlock(f.value as Buffer);
|
|
744
|
+
if (usage) yield usage;
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
}
|
|
748
|
+
|
|
749
|
+
/**
|
|
750
|
+
* UsageStats block at proto field #28. Captured shape (mitm of a real call):
|
|
751
|
+
*
|
|
752
|
+
* UsageStats {
|
|
753
|
+
* #1 label = "Token Usage"
|
|
754
|
+
* #2 entries [
|
|
755
|
+
* UsageEntry {
|
|
756
|
+
* #1 label = "Input tokens" / "Output tokens" / "Cached tokens" / ...
|
|
757
|
+
* #2 value (fixed32 — IEEE 754 float, OpenAI-style count cast)
|
|
758
|
+
* #3 unit = " tokens"
|
|
759
|
+
* #5 metric_id = "input_tokens" / "output_tokens" / ...
|
|
760
|
+
* },
|
|
761
|
+
* ...
|
|
762
|
+
* ]
|
|
763
|
+
* }
|
|
764
|
+
*
|
|
765
|
+
* We extract the standard input/output counts and synthesize a `total`.
|
|
766
|
+
* Anything else (cached, reasoning_tokens, …) is dropped for v1.
|
|
767
|
+
*/
|
|
768
|
+
function decodeUsageBlock(buf: Buffer): CloudChatEvent | null {
|
|
769
|
+
let promptTokens: number | undefined;
|
|
770
|
+
let completionTokens: number | undefined;
|
|
771
|
+
let cachedInputTokens: number | undefined;
|
|
772
|
+
let cacheCreationInputTokens: number | undefined;
|
|
773
|
+
let reasoningTokens: number | undefined;
|
|
774
|
+
|
|
775
|
+
for (const f of iterFields(buf)) {
|
|
776
|
+
// Each UsageEntry lives at field 2 (repeated). Field 1 is the block label
|
|
777
|
+
// ("Token Usage"); skip.
|
|
778
|
+
if (f.num !== 2 || f.wire !== 2 || !Buffer.isBuffer(f.value)) continue;
|
|
779
|
+
|
|
780
|
+
// Observed entry shape:
|
|
781
|
+
// UsageEntry {
|
|
782
|
+
// #4 (sub-message) {
|
|
783
|
+
// #1 label = "Input tokens" / "Output tokens"
|
|
784
|
+
// #2 (fixed32) value (IEEE 754 LE float — count as float)
|
|
785
|
+
// #3 unit = " token"
|
|
786
|
+
// #4 unit_plural = " tokens"
|
|
787
|
+
// }
|
|
788
|
+
// #5 metric_id = "input_tokens" / "output_tokens" / "cached_input_tokens" / ...
|
|
789
|
+
// }
|
|
790
|
+
let entryMetric: string | undefined;
|
|
791
|
+
let entryValue: number | undefined;
|
|
792
|
+
for (const sf of iterFields(f.value as Buffer)) {
|
|
793
|
+
if (sf.num === 5 && sf.wire === 2 && Buffer.isBuffer(sf.value)) {
|
|
794
|
+
entryMetric = (sf.value as Buffer).toString('utf8');
|
|
795
|
+
} else if (sf.num === 4 && sf.wire === 2 && Buffer.isBuffer(sf.value)) {
|
|
796
|
+
// Recurse into the displayed-dimension submessage to pull the fixed32
|
|
797
|
+
// value at its field 2.
|
|
798
|
+
for (const ssf of iterFields(sf.value as Buffer)) {
|
|
799
|
+
if (ssf.num === 2 && ssf.wire === 5 && Buffer.isBuffer(ssf.value)) {
|
|
800
|
+
entryValue = (ssf.value as Buffer).readFloatLE(0);
|
|
801
|
+
break;
|
|
802
|
+
}
|
|
803
|
+
}
|
|
804
|
+
}
|
|
805
|
+
}
|
|
806
|
+
if (entryMetric && entryValue !== undefined && Number.isFinite(entryValue)) {
|
|
807
|
+
const n = Math.round(entryValue);
|
|
808
|
+
if (entryMetric === 'input_tokens') promptTokens = n;
|
|
809
|
+
else if (entryMetric === 'output_tokens') completionTokens = n;
|
|
810
|
+
else if (entryMetric === 'cached_input_tokens' || entryMetric === 'cache_read_input_tokens') {
|
|
811
|
+
cachedInputTokens = (cachedInputTokens ?? 0) + n;
|
|
812
|
+
} else if (entryMetric === 'cache_creation_input_tokens') {
|
|
813
|
+
cacheCreationInputTokens = (cacheCreationInputTokens ?? 0) + n;
|
|
814
|
+
} else if (entryMetric === 'reasoning_tokens' || entryMetric === 'output_reasoning_tokens') {
|
|
815
|
+
reasoningTokens = (reasoningTokens ?? 0) + n;
|
|
816
|
+
}
|
|
817
|
+
}
|
|
818
|
+
}
|
|
819
|
+
if (promptTokens === undefined && completionTokens === undefined) return null;
|
|
820
|
+
// totalTokens reflects what OpenAI's API counts as billable: input +
|
|
821
|
+
// output. Cached / cache-creation / reasoning subtotals are surfaced as
|
|
822
|
+
// additional fields so callers that want a fuller picture (e.g. cost
|
|
823
|
+
// breakdown for reasoning models) can read them, but they're NOT
|
|
824
|
+
// double-counted into total.
|
|
825
|
+
const total = (promptTokens ?? 0) + (completionTokens ?? 0);
|
|
826
|
+
return {
|
|
827
|
+
kind: 'usage',
|
|
828
|
+
promptTokens,
|
|
829
|
+
completionTokens,
|
|
830
|
+
totalTokens: total > 0 ? total : undefined,
|
|
831
|
+
cachedInputTokens,
|
|
832
|
+
cacheCreationInputTokens,
|
|
833
|
+
reasoningTokens,
|
|
834
|
+
};
|
|
835
|
+
}
|
|
836
|
+
|
|
837
|
+
// ----------------------------------------------------------------------------
|
|
838
|
+
// Public API: streamChat
|
|
839
|
+
// ----------------------------------------------------------------------------
|
|
840
|
+
|
|
841
|
+
export interface CloudChatRequest {
|
|
842
|
+
/** Persistent OAuth-issued api_key (`devin-session-token$<JWT>`). */
|
|
843
|
+
apiKey: string;
|
|
844
|
+
/** Pre-resolved API server URL from RegisterUser (falls back to default). */
|
|
845
|
+
apiServerUrl?: string;
|
|
846
|
+
/** Model UID — e.g. `swe-1-6`, `kimi-k2-6`, `claude-opus-4-7-medium`. */
|
|
847
|
+
modelUid: string;
|
|
848
|
+
/** Chat history. */
|
|
849
|
+
messages: ChatHistoryItem[];
|
|
850
|
+
/**
|
|
851
|
+
* Tool definitions available to the model. Cloud encodes these in the
|
|
852
|
+
* GetChatMessage request's `tools` field (proto #10). When set, the model
|
|
853
|
+
* may emit `tool_call_start`/`_args`/`_end` events instead of plain text.
|
|
854
|
+
*/
|
|
855
|
+
tools?: ToolDef[];
|
|
856
|
+
/** Cascade ID — reuse across turns of the same conversation. */
|
|
857
|
+
cascadeId?: string;
|
|
858
|
+
/** Optional sampling overrides. */
|
|
859
|
+
completionOpts?: BuildArgs['completionOpts'];
|
|
860
|
+
/** Override request_type (default = 5, CASCADE). */
|
|
861
|
+
requestType?: number;
|
|
862
|
+
/** Abort signal — closes the fetch stream. */
|
|
863
|
+
signal?: AbortSignal;
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
export class CloudChatError extends Error {
|
|
867
|
+
constructor(message: string, public readonly code?: string, public readonly traceId?: string) {
|
|
868
|
+
super(message);
|
|
869
|
+
this.name = 'CloudChatError';
|
|
870
|
+
}
|
|
871
|
+
}
|
|
872
|
+
|
|
873
|
+
const TRACE_ID_RE = /\(trace ID: ([0-9a-f]+)\)/i;
|
|
874
|
+
|
|
875
|
+
/**
|
|
876
|
+
* Stream chat events from the cloud. Yields CloudChatEvent (text deltas, tool
|
|
877
|
+
* call deltas, finish reason). Use `streamChatText` for legacy text-only iteration.
|
|
878
|
+
*
|
|
879
|
+
* On error (auth fail, quota exhausted, malformed request) throws a
|
|
880
|
+
* CloudChatError with the cloud's `code` + `traceId` for diagnostics.
|
|
881
|
+
*/
|
|
882
|
+
export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<CloudChatEvent> {
|
|
883
|
+
// The api-server host comes from RegisterUser through the credential store.
|
|
884
|
+
// Validate it here too: this request body carries the api_key, so an
|
|
885
|
+
// unallowlisted host is credential exfiltration rather than a wrong endpoint.
|
|
886
|
+
const host = resolveDevinApiBaseUrl(req.apiServerUrl);
|
|
887
|
+
// The hosted chat path does not require the short-lived user_jwt; the working
|
|
888
|
+
// reference omits it by default. Minting it is opt-in so a mint failure or a
|
|
889
|
+
// JWT the chat service does not accept cannot break every turn.
|
|
890
|
+
const userJwt = process.env.OPENCODEX_DEVIN_SEND_USER_JWT === "1"
|
|
891
|
+
? await getCachedUserJwt(req.apiKey, host, req.signal)
|
|
892
|
+
: undefined;
|
|
893
|
+
|
|
894
|
+
// Pre-flight: consult the per-account model catalog. Cognition's cloud
|
|
895
|
+
// returns an opaque `permission_denied: "an internal error occurred (trace
|
|
896
|
+
// ID: ...)"` for every chat call that targets a model not enabled on the
|
|
897
|
+
// caller's tier — issue #14. The catalog's `disabled` flag is the
|
|
898
|
+
// authoritative source for "can this account run this UID"; we surface a
|
|
899
|
+
// named error here so the user knows why instead of guessing.
|
|
900
|
+
//
|
|
901
|
+
// Best-effort: if the catalog fetch fails (network, auth, schema drift) we
|
|
902
|
+
// pass through to the chat call. The cloud will still surface its own
|
|
903
|
+
// error and the trailer-error path below enriches the message in-place.
|
|
904
|
+
// Treat an empty catalog (schema drift / unexpected response) as "no catalog"
|
|
905
|
+
// so chat passes through instead of failing every request.
|
|
906
|
+
const catalog = await getCachedCatalog(req.apiKey, host, req.signal).catch(() => null);
|
|
907
|
+
if (catalog && catalog.byUid.size > 0) {
|
|
908
|
+
const entry = catalog.byUid.get(req.modelUid);
|
|
909
|
+
if (!entry) {
|
|
910
|
+
throw new ModelNotAvailableError(req.modelUid, req.modelUid, 'not_listed');
|
|
911
|
+
}
|
|
912
|
+
if (entry.disabled) {
|
|
913
|
+
throw new ModelNotAvailableError(req.modelUid, entry.label, 'disabled');
|
|
914
|
+
}
|
|
915
|
+
}
|
|
916
|
+
|
|
917
|
+
// Reuse session + cascade ids across calls for the same (apiKey, host).
|
|
918
|
+
// Without this, every turn looks like a brand-new server-side session
|
|
919
|
+
// and the cloud's prompt cache never hits — significant cost regression
|
|
920
|
+
// for long conversations.
|
|
921
|
+
const sessionIds = getOrAllocateSessionIds(req.apiKey, host, req.cascadeId);
|
|
922
|
+
|
|
923
|
+
const proto = buildGetChatMessageRequest({
|
|
924
|
+
apiKey: req.apiKey,
|
|
925
|
+
userJwt,
|
|
926
|
+
modelUid: req.modelUid,
|
|
927
|
+
messages: req.messages,
|
|
928
|
+
tools: req.tools,
|
|
929
|
+
cascadeId: sessionIds.cascadeId,
|
|
930
|
+
sessionId: sessionIds.sessionId,
|
|
931
|
+
requestId: BigInt(Date.now()),
|
|
932
|
+
triggerId: crypto.randomUUID(),
|
|
933
|
+
requestType: req.requestType,
|
|
934
|
+
completionOpts: req.completionOpts,
|
|
935
|
+
});
|
|
936
|
+
// The request envelope goes up uncompressed. A gzipped GetChatMessage frame is
|
|
937
|
+
// rejected with the same opaque `invalid_argument: an internal error occurred`
|
|
938
|
+
// the short fingerprint produces, and it is one of three things that have to be
|
|
939
|
+
// right together — the other two are the doubled Basic credential and the
|
|
940
|
+
// 732-character Metadata #31.
|
|
941
|
+
const framed = frameConnectStream(proto, false);
|
|
942
|
+
const body = new Blob([new Uint8Array(framed)], { type: "application/connect+proto" });
|
|
943
|
+
|
|
944
|
+
// Compose caller signal with a TTFB timeout. If the cloud takes longer
|
|
945
|
+
// than CLOUD_STREAM_TTFB_MS to start the response, abort. Once any byte
|
|
946
|
+
// arrives we cancel the TTFB timer and start the per-chunk idle timer
|
|
947
|
+
// inside the read loop instead.
|
|
948
|
+
const ttfbController = new AbortController();
|
|
949
|
+
const ttfbTimer = setTimeout(() => ttfbController.abort(new Error(`cloud-direct: time-to-first-byte timeout (${CLOUD_STREAM_TTFB_MS}ms)`)), CLOUD_STREAM_TTFB_MS);
|
|
950
|
+
const ttfbSignal = ttfbController.signal;
|
|
951
|
+
// Compose req.signal + ttfbSignal. AbortSignal.any was added in Node
|
|
952
|
+
// 20.3 / Bun 1.0; our `engines` allows Node ≥18, so on Node 18-20.2 the
|
|
953
|
+
// built-in is missing. The previous fallback `req.signal ?? ttfbSignal`
|
|
954
|
+
// silently discarded one of the two signals (TTFB if caller passed
|
|
955
|
+
// one), defeating the timeout guard. anySignal() is a real polyfill.
|
|
956
|
+
const composed = req.signal ? anySignal([req.signal, ttfbSignal]) : undefined;
|
|
957
|
+
const initialSignal: AbortSignal = composed?.signal ?? ttfbSignal;
|
|
958
|
+
|
|
959
|
+
let resp: Response;
|
|
960
|
+
try {
|
|
961
|
+
resp = await fetch(`${host}/exa.api_server_pb.ApiServerService/GetChatMessage`, {
|
|
962
|
+
method: 'POST',
|
|
963
|
+
headers: {
|
|
964
|
+
'Content-Type': 'application/connect+proto',
|
|
965
|
+
'Connect-Protocol-Version': '1',
|
|
966
|
+
'Connect-Accept-Encoding': 'gzip',
|
|
967
|
+
// The credential is the session token doubled and dash-joined. A single
|
|
968
|
+
// copy is refused with permission_denied. The protobuf body keeps one
|
|
969
|
+
// copy, in Metadata #3.
|
|
970
|
+
Authorization: `Basic ${req.apiKey}-${req.apiKey}`,
|
|
971
|
+
'User-Agent': 'connect-es/2.0.0',
|
|
972
|
+
Accept: '*/*',
|
|
973
|
+
},
|
|
974
|
+
body,
|
|
975
|
+
redirect: 'error',
|
|
976
|
+
signal: initialSignal,
|
|
977
|
+
});
|
|
978
|
+
} finally {
|
|
979
|
+
clearTimeout(ttfbTimer);
|
|
980
|
+
// The composed signal only guards the headers hop; the body is cancelled
|
|
981
|
+
// through cancelBodyOnAbort below. Detaching here keeps a long-lived caller
|
|
982
|
+
// signal from collecting one listener per turn.
|
|
983
|
+
composed?.cleanup();
|
|
984
|
+
}
|
|
985
|
+
|
|
986
|
+
if (!resp.ok) {
|
|
987
|
+
// The body is not echoed into the message. This error reaches the adapter's
|
|
988
|
+
// error event and /api/logs, and a Connect error can quote the request that
|
|
989
|
+
// produced it - which is the request holding the api_key.
|
|
990
|
+
throw new CloudChatError(`GetChatMessage failed (HTTP ${resp.status})`, undefined);
|
|
991
|
+
}
|
|
992
|
+
if (!resp.body) {
|
|
993
|
+
throw new CloudChatError('GetChatMessage response had no body stream');
|
|
994
|
+
}
|
|
995
|
+
|
|
996
|
+
// Cancel the body when the client goes away. Without this the read loop never
|
|
997
|
+
// observes req.signal after headers arrive: the turn keeps draining until the
|
|
998
|
+
// idle timer fires, and the stream then ends without an EOS trailer, which
|
|
999
|
+
// this function would report as a truncated upstream response rather than as
|
|
1000
|
+
// the cancellation it actually was.
|
|
1001
|
+
const detachBodyCancel = cancelBodyOnAbort(resp.body, req.signal);
|
|
1002
|
+
|
|
1003
|
+
// Incremental parsing. We previously did `pending = Buffer.concat([pending,
|
|
1004
|
+
// chunk])` per chunk — O(n²) over a long stream because every chunk copies
|
|
1005
|
+
// every buffered byte again. Now we keep a queue of arriving chunks with a
|
|
1006
|
+
// running offset; we only `Buffer.concat` when a frame straddles a chunk
|
|
1007
|
+
// boundary, and we slice/drop fully-consumed chunks immediately. For
|
|
1008
|
+
// typical 50-200KB responses this is ~5x faster and produces zero waste.
|
|
1009
|
+
const chunkQueue: Buffer[] = [];
|
|
1010
|
+
let queuedBytes = 0;
|
|
1011
|
+
// Bun + Node ReadableStream readers diverge on the type-level shape
|
|
1012
|
+
// (Bun's includes a `readMany` method); both work the same at runtime.
|
|
1013
|
+
const reader = resp.body.getReader() as ReadableStreamDefaultReader<Uint8Array>;
|
|
1014
|
+
let trailerError: { code?: string; message: string; traceId?: string } | null = null;
|
|
1015
|
+
let sawEos = false;
|
|
1016
|
+
|
|
1017
|
+
/**
|
|
1018
|
+
* Try to read the next `n` bytes from the chunk queue WITHOUT consuming
|
|
1019
|
+
* them. Returns null if not enough buffered.
|
|
1020
|
+
*/
|
|
1021
|
+
function peek(n: number): Buffer | null {
|
|
1022
|
+
if (queuedBytes < n) return null;
|
|
1023
|
+
if (chunkQueue.length === 1 && chunkQueue[0].length >= n) {
|
|
1024
|
+
return chunkQueue[0].slice(0, n);
|
|
1025
|
+
}
|
|
1026
|
+
// Cross-chunk peek — concat just the prefix we need.
|
|
1027
|
+
const parts: Buffer[] = [];
|
|
1028
|
+
let remaining = n;
|
|
1029
|
+
for (const c of chunkQueue) {
|
|
1030
|
+
if (remaining <= 0) break;
|
|
1031
|
+
if (c.length <= remaining) {
|
|
1032
|
+
parts.push(c);
|
|
1033
|
+
remaining -= c.length;
|
|
1034
|
+
} else {
|
|
1035
|
+
parts.push(c.slice(0, remaining));
|
|
1036
|
+
remaining = 0;
|
|
1037
|
+
}
|
|
1038
|
+
}
|
|
1039
|
+
return Buffer.concat(parts, n);
|
|
1040
|
+
}
|
|
1041
|
+
|
|
1042
|
+
/** Drop the first `n` bytes from the chunk queue. */
|
|
1043
|
+
function drop(n: number): void {
|
|
1044
|
+
queuedBytes -= n;
|
|
1045
|
+
let remaining = n;
|
|
1046
|
+
while (remaining > 0 && chunkQueue.length > 0) {
|
|
1047
|
+
const head = chunkQueue[0];
|
|
1048
|
+
if (head.length <= remaining) {
|
|
1049
|
+
chunkQueue.shift();
|
|
1050
|
+
remaining -= head.length;
|
|
1051
|
+
} else {
|
|
1052
|
+
chunkQueue[0] = head.slice(remaining);
|
|
1053
|
+
remaining = 0;
|
|
1054
|
+
}
|
|
1055
|
+
}
|
|
1056
|
+
}
|
|
1057
|
+
|
|
1058
|
+
// Track the idle timer at outer scope so the finally block can clear it
|
|
1059
|
+
// regardless of how we exit the read loop (clean done, throw, etc).
|
|
1060
|
+
// Previously this lived inside `try { ... }` and was only cleared on
|
|
1061
|
+
// normal exit — an error path left a 120s timer in the event loop and
|
|
1062
|
+
// the process refused to exit promptly.
|
|
1063
|
+
let idleTimer: ReturnType<typeof setTimeout> | null = null;
|
|
1064
|
+
try {
|
|
1065
|
+
const resetIdle = (): Promise<{ value?: Uint8Array; done: boolean }> => {
|
|
1066
|
+
if (idleTimer) clearTimeout(idleTimer);
|
|
1067
|
+
const idleController = new AbortController();
|
|
1068
|
+
idleTimer = setTimeout(
|
|
1069
|
+
() => idleController.abort(new Error(`cloud-direct: idle timeout (${CLOUD_STREAM_IDLE_MS}ms with no bytes)`)),
|
|
1070
|
+
CLOUD_STREAM_IDLE_MS,
|
|
1071
|
+
);
|
|
1072
|
+
// Race the reader.read() against idle abort. When abort wins, we
|
|
1073
|
+
// also actively `cancel()` the underlying body stream so the
|
|
1074
|
+
// pending read() resolves promptly with done=true instead of
|
|
1075
|
+
// hanging on the now-dead TCP socket until the OS notices.
|
|
1076
|
+
//
|
|
1077
|
+
// Promise-handling carefully: the reader.read() promise can settle
|
|
1078
|
+
// AFTER the outer race rejects (we cancelled, the read eventually
|
|
1079
|
+
// sees the cancellation and either resolves with done=true or
|
|
1080
|
+
// rejects with an abort error). We attach an explicit `.catch(()=>{})`
|
|
1081
|
+
// on the read promise so any post-race rejection doesn't surface as
|
|
1082
|
+
// an unhandled-rejection warning in the host runtime.
|
|
1083
|
+
return new Promise((resolve, reject) => {
|
|
1084
|
+
let settled = false;
|
|
1085
|
+
const settle = (fn: () => void): void => {
|
|
1086
|
+
if (settled) return;
|
|
1087
|
+
settled = true;
|
|
1088
|
+
fn();
|
|
1089
|
+
};
|
|
1090
|
+
const readP = reader.read();
|
|
1091
|
+
// Defensive: swallow any post-race rejection. If the outer promise
|
|
1092
|
+
// already settled via the abort listener, we still need a handler
|
|
1093
|
+
// attached to readP or Node logs an unhandledRejection.
|
|
1094
|
+
readP.catch(() => { /* swallowed; outer promise already rejected */ });
|
|
1095
|
+
|
|
1096
|
+
idleController.signal.addEventListener('abort', () => {
|
|
1097
|
+
try { void resp.body?.cancel(idleController.signal.reason ?? new Error('idle abort')); } catch { /* */ }
|
|
1098
|
+
settle(() => reject(idleController.signal.reason ?? new Error('idle abort')));
|
|
1099
|
+
}, { once: true });
|
|
1100
|
+
|
|
1101
|
+
readP.then(
|
|
1102
|
+
(v) => settle(() => resolve(v)),
|
|
1103
|
+
(e) => settle(() => reject(e)),
|
|
1104
|
+
);
|
|
1105
|
+
});
|
|
1106
|
+
};
|
|
1107
|
+
|
|
1108
|
+
while (true) {
|
|
1109
|
+
const { value, done } = await resetIdle();
|
|
1110
|
+
if (done) break;
|
|
1111
|
+
if (value) {
|
|
1112
|
+
chunkQueue.push(Buffer.from(value));
|
|
1113
|
+
queuedBytes += value.length;
|
|
1114
|
+
}
|
|
1115
|
+
|
|
1116
|
+
// Drain every complete frame currently buffered.
|
|
1117
|
+
while (queuedBytes >= 5) {
|
|
1118
|
+
const header = peek(5);
|
|
1119
|
+
if (!header) break;
|
|
1120
|
+
const flags = header[0];
|
|
1121
|
+
const len = header.readUInt32BE(1);
|
|
1122
|
+
// Cap frame length to prevent memory exhaustion from a corrupt/malicious
|
|
1123
|
+
// length prefix. 16MB is well above any legitimate Connect-RPC frame.
|
|
1124
|
+
if (len > MAX_FRAME_LEN) {
|
|
1125
|
+
throw new CloudChatError(`Connect frame length ${len} exceeds ${MAX_FRAME_LEN} byte cap`);
|
|
1126
|
+
}
|
|
1127
|
+
if (queuedBytes < 5 + len) break; // frame still arriving
|
|
1128
|
+
drop(5);
|
|
1129
|
+
const raw = peek(len) ?? Buffer.alloc(0);
|
|
1130
|
+
drop(len);
|
|
1131
|
+
|
|
1132
|
+
let payload = raw;
|
|
1133
|
+
if (flags & 0x01) {
|
|
1134
|
+
try {
|
|
1135
|
+
// MAX_FRAME_LEN caps the COMPRESSED frame, so without an output cap
|
|
1136
|
+
// a 16 MiB gzip frame can still inflate to gigabytes. The inbound
|
|
1137
|
+
// request path (src/server/request-decompress.ts) already bounds
|
|
1138
|
+
// decompression the same way.
|
|
1139
|
+
payload = zlib.gunzipSync(raw, { maxOutputLength: MAX_FRAME_LEN });
|
|
1140
|
+
} catch (gzipErr) {
|
|
1141
|
+
const code = (gzipErr as NodeJS.ErrnoException).code;
|
|
1142
|
+
if (code === 'ERR_BUFFER_TOO_LARGE') {
|
|
1143
|
+
throw new CloudChatError(`Connect frame inflates past the ${MAX_FRAME_LEN} byte cap`, 'frame_too_large');
|
|
1144
|
+
}
|
|
1145
|
+
// Corrupt compressed frame — surface as a CloudChatError instead
|
|
1146
|
+
// of falling through and re-parsing raw gzip bytes as proto
|
|
1147
|
+
// (which used to misparse silently downstream).
|
|
1148
|
+
throw new CloudChatError(`Connect frame gunzip failed: ${(gzipErr as Error).message}`);
|
|
1149
|
+
}
|
|
1150
|
+
}
|
|
1151
|
+
const eos = (flags & 0x02) !== 0;
|
|
1152
|
+
|
|
1153
|
+
if (eos) {
|
|
1154
|
+
sawEos = true;
|
|
1155
|
+
// Trailer: {} on success, {"error":{code,message}} on failure.
|
|
1156
|
+
const text = payload.toString('utf8');
|
|
1157
|
+
if (text && text.includes('"error"')) {
|
|
1158
|
+
let code: string | undefined;
|
|
1159
|
+
let message = text;
|
|
1160
|
+
try {
|
|
1161
|
+
const j = JSON.parse(text) as { error?: { code?: string; message?: string } };
|
|
1162
|
+
code = j.error?.code;
|
|
1163
|
+
if (j.error?.message) message = j.error.message;
|
|
1164
|
+
} catch { /* keep raw */ }
|
|
1165
|
+
const traceMatch = message.match(TRACE_ID_RE);
|
|
1166
|
+
trailerError = { code, message, traceId: traceMatch?.[1] };
|
|
1167
|
+
}
|
|
1168
|
+
continue;
|
|
1169
|
+
}
|
|
1170
|
+
yield* decodeChatFrame(payload);
|
|
1171
|
+
}
|
|
1172
|
+
}
|
|
1173
|
+
} finally {
|
|
1174
|
+
// Always clear the idle timer. The previous "clear on normal exit
|
|
1175
|
+
// only" path leaked a 120s setTimeout into the event loop on any
|
|
1176
|
+
// throw (idle timeout, gunzip error, trailer error, etc), keeping
|
|
1177
|
+
// the process from exiting promptly.
|
|
1178
|
+
if (idleTimer) clearTimeout(idleTimer);
|
|
1179
|
+
// Cancel the underlying body stream on any non-clean exit so the TCP
|
|
1180
|
+
// connection is released. `releaseLock` alone leaves the body in a
|
|
1181
|
+
// dangling state; we have to call `cancel` on the response body
|
|
1182
|
+
// itself (cancel-via-reader requires holding the lock). Fire and
|
|
1183
|
+
// forget — there's nothing meaningful to do if cancel rejects.
|
|
1184
|
+
try { reader.releaseLock(); } catch { /* */ }
|
|
1185
|
+
try { void resp.body?.cancel(); } catch { /* */ }
|
|
1186
|
+
}
|
|
1187
|
+
|
|
1188
|
+
if (trailerError) {
|
|
1189
|
+
// Cognition uses `permission_denied: "an internal error occurred (trace
|
|
1190
|
+
// ID: …)"` as a catch-all for "your account can't run this model" — same
|
|
1191
|
+
// root cause issue #14 reported. The pre-flight above catches this when
|
|
1192
|
+
// the catalog disagrees with the call, but the catalog can lag (a model
|
|
1193
|
+
// that was enabled at fetch time may have been gated between then and
|
|
1194
|
+
// now) or be missing (network failure caused a fall-through). When the
|
|
1195
|
+
// raw trailer is this exact shape, swap in a message that names the
|
|
1196
|
+
// model and explains the likely cause rather than re-passing
|
|
1197
|
+
// Cognition's opaque text. The cloud's original message is appended in
|
|
1198
|
+
// parens so users (and bug reports) still have it verbatim.
|
|
1199
|
+
// Both codes carry this shape. Cognition uses `invalid_argument` for a
|
|
1200
|
+
// request it could not accept and `permission_denied` for one it would not,
|
|
1201
|
+
// and the message body is the same opaque sentence either way.
|
|
1202
|
+
const isOpaqueDenial =
|
|
1203
|
+
(trailerError.code === 'permission_denied' || trailerError.code === 'invalid_argument') &&
|
|
1204
|
+
/an internal error occurred/i.test(trailerError.message);
|
|
1205
|
+
if (isOpaqueDenial) {
|
|
1206
|
+
const enriched =
|
|
1207
|
+
`Cognition denied this request for model "${req.modelUid}" with the opaque ` +
|
|
1208
|
+
`"an internal error occurred" message, which it uses for both a malformed ` +
|
|
1209
|
+
`request and a refused one. In practice this has meant the request, not ` +
|
|
1210
|
+
`the account: the same sentence came back for every turn until the ` +
|
|
1211
|
+
`CompletionConfiguration tag map was corrected, and a temperature of ` +
|
|
1212
|
+
`exactly 0 still produces it. Check the request before the plan — ` +
|
|
1213
|
+
`tests/providers/devin-hardening.test.ts pins the field layout the ` +
|
|
1214
|
+
`service accepts. If the request is unchanged and this is new, the ` +
|
|
1215
|
+
`account's model access is the next thing to check. ` +
|
|
1216
|
+
`(cloud trace ID: ${trailerError.traceId ?? 'n/a'})`;
|
|
1217
|
+
throw new CloudChatError(enriched, trailerError.code, trailerError.traceId);
|
|
1218
|
+
}
|
|
1219
|
+
// Cognition also returns `permission_denied` when a tool description
|
|
1220
|
+
// contains a blocklisted phrase that the sanitizer above did not catch
|
|
1221
|
+
// (e.g. Cognition added a new phrase). Surface a clear message so the
|
|
1222
|
+
// user knows to check tool descriptions rather than suspect auth/tier.
|
|
1223
|
+
// Only blame the blocklist when tools were actually sent. Asserting it for
|
|
1224
|
+
// every permission_denied sent users to inspect a tool table that had
|
|
1225
|
+
// nothing to do with an ordinary ACL or tier denial.
|
|
1226
|
+
if (trailerError.code === 'permission_denied' && (req.tools?.length ?? 0) > 0) {
|
|
1227
|
+
const enriched =
|
|
1228
|
+
`Cognition denied this request (permission_denied). If tool descriptions ` +
|
|
1229
|
+
`are present, a blocklisted phrase may have triggered this — see the ` +
|
|
1230
|
+
`COGNITION_BLOCKLIST_REWRITES table in cloud-direct/chat.ts. ` +
|
|
1231
|
+
// Keep the cloud's own sentence. Replacing it outright is what made the
|
|
1232
|
+
// two Codex entries in that table expensive to find: the message named
|
|
1233
|
+
// the table but dropped the only text that could have said whether this
|
|
1234
|
+
// was a phrase match at all.
|
|
1235
|
+
`(cloud message: ${trailerError.message}) ` +
|
|
1236
|
+
`(cloud trace ID: ${trailerError.traceId ?? 'n/a'})`;
|
|
1237
|
+
throw new CloudChatError(enriched, trailerError.code, trailerError.traceId);
|
|
1238
|
+
}
|
|
1239
|
+
throw new CloudChatError(trailerError.message, trailerError.code, trailerError.traceId);
|
|
1240
|
+
}
|
|
1241
|
+
// Truncation detection: the cloud always terminates a successful stream
|
|
1242
|
+
// with an EOS trailer. If we hit `done` from the body reader without one,
|
|
1243
|
+
// the connection dropped mid-frame and any bytes still in the queue are
|
|
1244
|
+
// garbage. Previously those leftover bytes were silently discarded and
|
|
1245
|
+
// the consumer saw a clean stop with no error — looked like the model
|
|
1246
|
+
// had finished. Now we surface it.
|
|
1247
|
+
detachBodyCancel();
|
|
1248
|
+
if (req.signal?.aborted) {
|
|
1249
|
+
// The caller cancelled. The missing EOS trailer is the expected consequence
|
|
1250
|
+
// of that cancellation, not evidence that the cloud dropped the response.
|
|
1251
|
+
return;
|
|
1252
|
+
}
|
|
1253
|
+
if (!sawEos) {
|
|
1254
|
+
throw new CloudChatError(
|
|
1255
|
+
`Cloud stream ended without EOS trailer (${queuedBytes} bytes orphaned). ` +
|
|
1256
|
+
`Connection likely dropped mid-response.`,
|
|
1257
|
+
'truncated_stream',
|
|
1258
|
+
);
|
|
1259
|
+
}
|
|
1260
|
+
}
|
|
1261
|
+
|
|
1262
|
+
/**
|
|
1263
|
+
* Back-compat: yield text content only (drops tool calls). The plugin uses
|
|
1264
|
+
* streamChatEvents directly when it needs to surface tool_calls.
|
|
1265
|
+
*/
|
|
1266
|
+
export async function* streamChat(req: CloudChatRequest): AsyncGenerator<string> {
|
|
1267
|
+
for await (const ev of streamChatEvents(req)) {
|
|
1268
|
+
if (ev.kind === 'text') yield ev.text;
|
|
1269
|
+
}
|
|
1270
|
+
}
|
|
1271
|
+
|
|
1272
|
+
// `parseConnectFrames` is no longer needed by streamChat itself, but exported
|
|
1273
|
+
// from wire.ts for one-shot callers + tests.
|
|
1274
|
+
void parseConnectFrames;
|