@bitkyc08/opencodex 2.52.0 → 2.53.0-preview.20260913
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/gui/dist/assets/index-BBOZWGB6.css +1 -0
- package/gui/dist/assets/index-D7ynYo2K.js +128 -0
- package/gui/dist/index.html +2 -2
- package/native/remote-workspace-helper/Cargo.lock +130 -0
- package/native/remote-workspace-helper/Cargo.toml +24 -0
- package/native/remote-workspace-helper/src/main.rs +49 -0
- package/native/remote-workspace-helper/src/protocol.rs +246 -0
- package/native/remote-workspace-helper/src/sandbox/macos.rs +19 -0
- package/native/remote-workspace-helper/src/sandbox/mod.rs +77 -0
- package/native/remote-workspace-helper/src/sandbox/windows.rs +15 -0
- package/package.json +6 -1
- package/src/adapters/anthropic-image-normalize.ts +30 -2
- package/src/adapters/anthropic.ts +1 -1
- package/src/adapters/base.ts +8 -2
- package/src/adapters/cursor/cursor-errors.ts +12 -0
- package/src/adapters/cursor/thread-continuity.ts +93 -0
- package/src/adapters/cursor.ts +104 -73
- package/src/adapters/devin/cloud-direct/chat.ts +312 -23
- package/src/adapters/devin/cloud-direct/metadata.ts +31 -2
- package/src/adapters/devin/live-models.ts +70 -3
- package/src/adapters/devin.ts +281 -21
- package/src/adapters/google-wire-compiler.ts +14 -6
- package/src/adapters/google.ts +22 -8
- package/src/adapters/kiro/adapter.ts +316 -0
- package/src/adapters/kiro/conversation.ts +136 -0
- package/src/adapters/kiro/payload.ts +432 -0
- package/src/adapters/kiro/reasoning.ts +56 -0
- package/src/adapters/kiro/stream.ts +1153 -0
- package/src/adapters/kiro/usage.ts +223 -0
- package/src/adapters/kiro/wire.ts +76 -0
- package/src/adapters/kiro.ts +8 -2319
- package/src/adapters/mimo-free.ts +1 -1
- package/src/adapters/openai-chat-images.ts +101 -0
- package/src/adapters/openai-chat.ts +201 -181
- package/src/adapters/openai-responses.ts +92 -224
- package/src/adapters/registry.ts +0 -7
- package/src/adapters/run-turn-queue.ts +13 -6
- package/src/bridge.ts +14 -15
- package/src/chat/inbound.ts +29 -4
- package/src/chat/outbound.ts +145 -107
- package/src/claude/desktop-profile.ts +4 -6
- package/src/cli/account-api.ts +14 -0
- package/src/cli/account-extended.ts +1 -1
- package/src/cli/account-history.ts +60 -0
- package/src/cli/account-main.ts +80 -0
- package/src/cli/account.ts +11 -3
- package/src/cli/capabilities.ts +113 -0
- package/src/cli/catalog.ts +109 -0
- package/src/cli/dispatch.ts +9 -0
- package/src/cli/help.ts +2 -0
- package/src/cli/index.ts +2 -2
- package/src/cli/observe.ts +28 -1
- package/src/cli/opencode.ts +42 -8
- package/src/cli/provider-runtime.ts +11 -1
- package/src/cli/provider.ts +22 -2
- package/src/cli/registry.ts +21 -0
- package/src/cli/remote-workspace.ts +154 -0
- package/src/cli/status.ts +39 -7
- package/src/cli/usage-report.ts +14 -2
- package/src/client/hub-client.ts +34 -0
- package/src/client/hub-state.ts +9 -1
- package/src/codex/account-store.ts +78 -0
- package/src/codex/auth-api.ts +81 -54
- package/src/codex/auth-context.ts +45 -16
- package/src/codex/catalog/effort.ts +1 -1
- package/src/codex/catalog/metadata.ts +3 -6
- package/src/codex/catalog/native-models.ts +4 -4
- package/src/codex/catalog/parsing.ts +2 -20
- package/src/codex/catalog/provider-fetch.ts +10 -1
- package/src/codex/catalog/remote.ts +233 -0
- package/src/codex/catalog/sync.ts +403 -35
- package/src/codex/convergence.ts +1 -1
- package/src/codex/history-manifest.ts +36 -0
- package/src/codex/history-provider.ts +32 -5
- package/src/codex/inject.ts +9 -0
- package/src/codex/main-account.ts +113 -0
- package/src/codex/main-device-reauth-api.ts +89 -0
- package/src/codex/main-device-reauth.ts +217 -0
- package/src/codex/native-residue.ts +9 -2
- package/src/codex/quota-auto-refresh.ts +3 -2
- package/src/codex/quota-capacity.ts +98 -0
- package/src/codex/quota-history.ts +160 -0
- package/src/codex/quota-types.ts +8 -0
- package/src/codex/quota.ts +118 -91
- package/src/codex/refresh.ts +2 -1
- package/src/codex/routing.ts +90 -17
- package/src/codex/sync.ts +33 -4
- package/src/combos/request.ts +19 -1
- package/src/config/multi-agent-surface.ts +61 -0
- package/src/config/provider-validation.ts +176 -0
- package/src/config.ts +213 -11
- package/src/generated/compatibility-version.json +436 -168
- package/src/images/loop.ts +119 -36
- package/src/lib/admission.ts +12 -6
- package/src/lib/redact.ts +7 -0
- package/src/lib/translator-budget.ts +4 -3
- package/src/lib/windows-atomic-replace.ts +1 -0
- package/src/lib/windows-elevation.ts +1 -1
- package/src/oauth/chatgpt-device.ts +62 -5
- package/src/oauth/devin/cli-import.ts +130 -0
- package/src/oauth/devin.ts +63 -8
- package/src/oauth/index.ts +29 -14
- package/src/oauth/kiro.ts +18 -6
- package/src/oauth/login-cli.ts +9 -1
- package/src/oauth/meta-muse-device.ts +464 -0
- package/src/oauth/meta-muse.ts +123 -32
- package/src/oauth/pool-kernel.ts +9 -0
- package/src/oauth/pool-settings-capability.ts +2 -2
- package/src/oauth/store.ts +57 -0
- package/src/oauth/types.ts +31 -0
- package/src/providers/derive.ts +13 -3
- package/src/providers/devin-cli-authmode-migration.ts +57 -35
- package/src/providers/devin-provider-merge-migration.ts +240 -0
- package/src/providers/muse-key-quota.ts +117 -0
- package/src/providers/muse-subscription-usage.ts +14 -2
- package/src/providers/openai-sidecar.ts +25 -3
- package/src/providers/opencode-zen-rate-limit.ts +58 -0
- package/src/providers/provider-id-rewrite.ts +20 -5
- package/src/providers/quota-types.ts +12 -0
- package/src/providers/quota.ts +143 -102
- package/src/providers/reasoning-metadata.ts +543 -0
- package/src/providers/registry.ts +80 -49
- package/src/reasoning-effort.ts +26 -2
- package/src/remote/hub-usage.ts +32 -0
- package/src/remote-control/index.ts +192 -41
- package/src/remote-control/workspace-activation.ts +9 -0
- package/src/remote-control/workspace-agent-connection.ts +366 -0
- package/src/remote-control/workspace-claude-runtime.ts +243 -0
- package/src/remote-control/workspace-codex-runtime.ts +531 -0
- package/src/remote-control/workspace-codex-sandbox.ts +115 -0
- package/src/remote-control/workspace-command-runner.ts +748 -0
- package/src/remote-control/workspace-coordinator.ts +231 -0
- package/src/remote-control/workspace-device.ts +585 -0
- package/src/remote-control/workspace-executable.ts +43 -0
- package/src/remote-control/workspace-executor.ts +397 -0
- package/src/remote-control/workspace-hub.ts +519 -0
- package/src/remote-control/workspace-pi-runtime.ts +382 -0
- package/src/remote-control/workspace-process.ts +129 -0
- package/src/remote-control/workspace-rpc.ts +304 -0
- package/src/remote-control/workspace-runtime.ts +60 -0
- package/src/remote-control/workspace-secret-store.ts +39 -0
- package/src/remote-control/workspace-sessions.ts +799 -0
- package/src/remote-control/workspace-tool-bridge.ts +192 -0
- package/src/responses/code-mode-helper-compat.ts +22 -3
- package/src/responses/hosted-tool-policy.ts +0 -1
- package/src/responses/muse-tool-name-alias.ts +379 -0
- package/src/responses/plaintext-v2-agent-messages.ts +902 -0
- package/src/router.ts +7 -0
- package/src/routing/compatibility/behavior.ts +0 -1
- package/src/server/audio-client.ts +64 -0
- package/src/server/audio-dictation.ts +91 -0
- package/src/server/audio-live.ts +185 -0
- package/src/server/audio-transcriptions.ts +183 -0
- package/src/server/audio-upstream.ts +153 -0
- package/src/server/auth-cors.ts +61 -2
- package/src/server/chat-completions.ts +1 -1
- package/src/server/chat-native-sse.ts +92 -48
- package/src/server/chat-native.ts +37 -15
- package/src/server/hub-usage.ts +57 -0
- package/src/server/images.ts +4 -0
- package/src/server/index.ts +722 -57
- package/src/server/lifecycle.ts +5 -6
- package/src/server/live-call-bindings.ts +60 -0
- package/src/server/live.ts +12 -1
- package/src/server/management/agent-settings-routes.ts +25 -4
- package/src/server/management/api-access.ts +37 -0
- package/src/server/management/api-key-usage.ts +7 -2
- package/src/server/management/config-routes.ts +1 -18
- package/src/server/management/context.ts +15 -0
- package/src/server/management/logs-usage-routes.ts +2 -0
- package/src/server/management/oauth-account-routes.ts +39 -12
- package/src/server/management/provider-routes.ts +125 -2
- package/src/server/management/remote-workspace-routes.ts +140 -0
- package/src/server/management/route-registry.ts +15 -0
- package/src/server/management/usage-aggregate-cache.ts +14 -15
- package/src/server/management/usage-summary-cache.ts +2 -0
- package/src/server/management-api.ts +23 -0
- package/src/server/ports.ts +17 -0
- package/src/server/relay-eager.ts +4 -1
- package/src/server/relay.ts +70 -10
- package/src/server/request-decompress.ts +6 -3
- package/src/server/responses/agent-task-recovery.ts +25 -32
- package/src/server/responses/codex-auth-error.ts +11 -0
- package/src/server/responses/codex-ws-exchange.ts +52 -3
- package/src/server/responses/codex-ws-wire.ts +55 -0
- package/src/server/responses/compact.ts +9 -1
- package/src/server/responses/core.ts +337 -73
- package/src/server/responses/encrypted-payload.ts +45 -2
- package/src/server/responses/ws-upstream.ts +4 -1
- package/src/server/responses-self-named-namespace-scrub.ts +1 -3
- package/src/server/responses-undeclared-tool-guard.ts +1 -1
- package/src/server/search.ts +3 -0
- package/src/server/sse-payload-rewrite.ts +136 -51
- package/src/server/ws-bridge.ts +35 -1
- package/src/service/cli.ts +372 -0
- package/src/service/diagnostics.ts +340 -0
- package/src/service/guards.ts +303 -0
- package/src/service/health.ts +222 -0
- package/src/service/launchd.ts +853 -0
- package/src/service/orchestration.ts +617 -0
- package/src/service/repair.ts +334 -0
- package/src/service/state.ts +363 -0
- package/src/service/systemd.ts +229 -0
- package/src/service/windows-ops.ts +690 -0
- package/src/service/windows-scheduler.ts +769 -0
- package/src/service/windows-taskxml.ts +613 -0
- package/src/service.ts +22 -5550
- package/src/storage/cleanup/db.ts +258 -0
- package/src/storage/cleanup/execute.ts +358 -0
- package/src/storage/cleanup/paths.ts +189 -0
- package/src/storage/cleanup/pending.ts +140 -0
- package/src/storage/cleanup/preview.ts +292 -0
- package/src/storage/cleanup/reconcile.ts +347 -0
- package/src/storage/cleanup/restore.ts +932 -0
- package/src/storage/cleanup/satellite.ts +474 -0
- package/src/storage/cleanup/staging.ts +129 -0
- package/src/storage/cleanup/types.ts +98 -0
- package/src/storage/cleanup.ts +49 -3127
- package/src/types/accounts.ts +2 -0
- package/src/types/config.ts +13 -12
- package/src/types/provider.ts +37 -0
- package/src/types/request.ts +2 -0
- package/src/types/tools.ts +17 -5
- package/src/types.ts +1 -0
- package/src/usage/expected-prices.ts +127 -0
- package/src/usage/log.ts +58 -1
- package/src/vision/eligibility.ts +13 -2
- package/src/web-search/loop.ts +56 -3
- package/gui/dist/assets/index-CWXut3rG.js +0 -115
- package/gui/dist/assets/index-EdoPnm9_.css +0 -1
- package/src/adapters/devin-cli/acp.ts +0 -204
- package/src/adapters/devin-cli/adapter.ts +0 -345
- package/src/adapters/devin-cli/binary.ts +0 -69
- package/src/adapters/devin-cli/models.ts +0 -57
- package/src/oauth/devin-cli.ts +0 -149
- package/src/server/responses-reasoning-summary-rewrite.ts +0 -178
|
@@ -46,16 +46,47 @@ import { resolveDevinApiBaseUrl } from '../../../oauth/devin/api-base.js';
|
|
|
46
46
|
* we only trigger when the server has genuinely stopped responding.
|
|
47
47
|
*/
|
|
48
48
|
const CLOUD_STREAM_IDLE_MS = 120_000;
|
|
49
|
-
/**
|
|
50
|
-
|
|
49
|
+
/**
|
|
50
|
+
* Budget for the response HEADERS, which is not the same thing as a connect
|
|
51
|
+
* timeout. Cognition holds the headers until the model produces its first
|
|
52
|
+
* token, so on a high-effort reasoning model this bounds generation. A 60s
|
|
53
|
+
* value killed live swe-2 high turns at exactly 60000ms with no output while
|
|
54
|
+
* a sibling call on the same account was still alive at 76s, which is the
|
|
55
|
+
* defect this constant exists to record.
|
|
56
|
+
*
|
|
57
|
+
* It has to be at least as generous as the body idle budget above. The cost of
|
|
58
|
+
* the larger value is bounded and understood: a peer that goes silent at the
|
|
59
|
+
* TCP level without sending RST/FIN now hangs for this long instead of 60s. A
|
|
60
|
+
* peer that actually dies still rejects immediately. This timer is the only
|
|
61
|
+
* bound on that case once `timeout: 0` is set on the fetch, so it must not be
|
|
62
|
+
* removed. Override with OPENCODEX_DEVIN_TTFB_MS.
|
|
63
|
+
*/
|
|
64
|
+
const CLOUD_STREAM_HEADERS_DEFAULT_MS = 300_000;
|
|
65
|
+
/** Upper bound for the override, so a stray value cannot wedge a turn forever. */
|
|
66
|
+
const CLOUD_STREAM_HEADERS_MAX_MS = 1_800_000;
|
|
67
|
+
function cloudStreamHeadersMs(): number {
|
|
68
|
+
const raw = process.env.OPENCODEX_DEVIN_TTFB_MS?.trim();
|
|
69
|
+
if (!raw) return CLOUD_STREAM_HEADERS_DEFAULT_MS;
|
|
70
|
+
const parsed = Number(raw);
|
|
71
|
+
if (!Number.isFinite(parsed) || parsed <= 0) return CLOUD_STREAM_HEADERS_DEFAULT_MS;
|
|
72
|
+
return Math.min(parsed, CLOUD_STREAM_HEADERS_MAX_MS);
|
|
73
|
+
}
|
|
74
|
+
/** Test seam for the headers budget; the resolver itself stays private. */
|
|
75
|
+
export const cloudStreamHeadersMsForTests = cloudStreamHeadersMs;
|
|
51
76
|
/** Maximum acceptable Connect-RPC frame length (16 MB). */
|
|
52
77
|
const MAX_FRAME_LEN = 16 * 1024 * 1024;
|
|
53
78
|
|
|
54
79
|
/**
|
|
55
|
-
*
|
|
56
|
-
* server
|
|
57
|
-
|
|
58
|
-
|
|
80
|
+
* PromptCacheOptions.type = EPHEMERAL. Marks the system prefix as a cache entry
|
|
81
|
+
* the server may reuse on the next turn of the same session.
|
|
82
|
+
*/
|
|
83
|
+
const PROMPT_CACHE_EPHEMERAL = 1;
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* Per-identity session/cascade ID cache. Cloud uses these for server-side
|
|
87
|
+
* context caching across turns of the same conversation; if we mint a fresh
|
|
88
|
+
* sessionId on every call (which we used to), every turn looks like a
|
|
89
|
+
* brand-new session and the prompt-cache hit ratio is zero.
|
|
59
90
|
* Single-process scope is enough: opencode lives in one runtime for a TUI
|
|
60
91
|
* session, and CLI one-shots don't benefit from caching anyway.
|
|
61
92
|
*/
|
|
@@ -63,6 +94,18 @@ interface SessionIds {
|
|
|
63
94
|
sessionId: string;
|
|
64
95
|
cascadeId: string;
|
|
65
96
|
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* Cache key for one Devin credential on one host.
|
|
100
|
+
*
|
|
101
|
+
* The credential itself used to be the Map key. Hashing it keeps the raw token
|
|
102
|
+
* out of any structure a heap dump or debugger would walk, and gives the other
|
|
103
|
+
* per-account caches a name they can share. 16 hex is 64 bits, which against a
|
|
104
|
+
* bounded single-process map is not a collision risk worth widening the key for.
|
|
105
|
+
*/
|
|
106
|
+
export function devinCacheIdentity(apiKey: string, host: string): string {
|
|
107
|
+
return crypto.createHash('sha256').update(`${host}\x1f${apiKey}`).digest('hex').slice(0, 16);
|
|
108
|
+
}
|
|
66
109
|
/**
|
|
67
110
|
* Bounded the same way the adapter bounds its cascade-id map: a long-running
|
|
68
111
|
* proxy sees one entry per (host, api_key) pair, and nothing ever evicted them.
|
|
@@ -70,7 +113,7 @@ interface SessionIds {
|
|
|
70
113
|
const SESSION_CACHE_MAX = 256;
|
|
71
114
|
const sessionCache = new Map<string, SessionIds>();
|
|
72
115
|
function getOrAllocateSessionIds(apiKey: string, host: string, cascadeIdOverride?: string): SessionIds {
|
|
73
|
-
const key =
|
|
116
|
+
const key = devinCacheIdentity(apiKey, host);
|
|
74
117
|
let ids = sessionCache.get(key);
|
|
75
118
|
if (!ids) {
|
|
76
119
|
ids = {
|
|
@@ -90,9 +133,21 @@ function getOrAllocateSessionIds(apiKey: string, host: string, cascadeIdOverride
|
|
|
90
133
|
return ids;
|
|
91
134
|
}
|
|
92
135
|
|
|
93
|
-
/**
|
|
94
|
-
|
|
95
|
-
|
|
136
|
+
/**
|
|
137
|
+
* Drop the cached session for ONE identity, after that account signs out or is
|
|
138
|
+
* switched away from.
|
|
139
|
+
*
|
|
140
|
+
* This replaces a global clear(). The proxy serves several accounts from one
|
|
141
|
+
* process, so clearing every entry on a per-provider logout would strip the
|
|
142
|
+
* session and cascade of accounts that were mid-turn. That is why the global
|
|
143
|
+
* version was never safe to call, and why nothing ever called it.
|
|
144
|
+
*
|
|
145
|
+
* A turn already in flight is unaffected: it received its SessionIds object at
|
|
146
|
+
* request start and never re-reads the map, so it finishes on the session it
|
|
147
|
+
* began with and the next turn allocates fresh.
|
|
148
|
+
*/
|
|
149
|
+
export function invalidateSessionIdentity(identity: string): void {
|
|
150
|
+
sessionCache.delete(identity);
|
|
96
151
|
}
|
|
97
152
|
|
|
98
153
|
// ----------------------------------------------------------------------------
|
|
@@ -120,6 +175,9 @@ export function allocateCascadeId(): string {
|
|
|
120
175
|
* #4 num_tokens: int (rough estimate)
|
|
121
176
|
* #5 safe_for_code_telemetry: bool (1 = ok to log)
|
|
122
177
|
* #10 images: repeated ImageData (multimodal)
|
|
178
|
+
* #11 thinking: string (assistant reasoning, replayed)
|
|
179
|
+
* #12 signature: string (opaque attestation for #11)
|
|
180
|
+
* #18 signature_type: string
|
|
123
181
|
* }
|
|
124
182
|
*
|
|
125
183
|
* ImageData (exa.codeium_common_pb.ImageData) {
|
|
@@ -153,7 +211,13 @@ function encodeChatToolCall(tc: { id: string; name: string; arguments: string })
|
|
|
153
211
|
function encodeChatMessagePrompt(
|
|
154
212
|
content: ContentPart[],
|
|
155
213
|
source: number,
|
|
156
|
-
opts?: {
|
|
214
|
+
opts?: {
|
|
215
|
+
toolCallId?: string;
|
|
216
|
+
toolCalls?: Array<{ id: string; name: string; arguments: string }>;
|
|
217
|
+
thinking?: string;
|
|
218
|
+
signature?: string;
|
|
219
|
+
signatureType?: string;
|
|
220
|
+
},
|
|
157
221
|
): Buffer {
|
|
158
222
|
const textParts = content.filter((p): p is { type: 'text'; text: string } => p.type === 'text');
|
|
159
223
|
const imageParts = content.filter((p): p is { type: 'image'; mimeType: string; base64Data: string; caption?: string } => p.type === 'image');
|
|
@@ -178,6 +242,14 @@ function encodeChatMessagePrompt(
|
|
|
178
242
|
for (const img of imageParts) {
|
|
179
243
|
parts.push(encodeMessage(10, encodeImageData(img)));
|
|
180
244
|
}
|
|
245
|
+
// Reasoning replay. This adapter used to assert that Cognition has no
|
|
246
|
+
// reasoning-replay field and drop the assistant's own thinking, so a
|
|
247
|
+
// reasoning model restarted its chain on every turn of a tool loop. Two
|
|
248
|
+
// independent clients of the same service write it here: #11 thinking,
|
|
249
|
+
// #12 signature, #18 signature_type on the assistant prompt.
|
|
250
|
+
if (opts?.thinking) parts.push(encodeString(11, opts.thinking));
|
|
251
|
+
if (opts?.signature) parts.push(encodeString(12, opts.signature));
|
|
252
|
+
if (opts?.signatureType) parts.push(encodeString(18, opts.signatureType));
|
|
181
253
|
return Buffer.concat(parts);
|
|
182
254
|
}
|
|
183
255
|
|
|
@@ -342,6 +414,15 @@ export interface ChatHistoryItem {
|
|
|
342
414
|
* each ChatToolCall has #1 id, #2 name, #3 arguments_json).
|
|
343
415
|
*/
|
|
344
416
|
tool_calls?: Array<{ id: string; name: string; arguments: string }>;
|
|
417
|
+
/**
|
|
418
|
+
* For `role: 'assistant'` only — the model's own reasoning from that turn,
|
|
419
|
+
* replayed so a reasoning model does not restart its chain on the next one.
|
|
420
|
+
* Encoded as ChatMessagePrompt #11 with its #12 signature and #18
|
|
421
|
+
* signature_type.
|
|
422
|
+
*/
|
|
423
|
+
thinking?: string;
|
|
424
|
+
signature?: string;
|
|
425
|
+
signature_type?: string;
|
|
345
426
|
}
|
|
346
427
|
|
|
347
428
|
/**
|
|
@@ -399,6 +480,12 @@ export interface ToolDef {
|
|
|
399
480
|
export type CloudChatEvent =
|
|
400
481
|
| { kind: 'text'; text: string }
|
|
401
482
|
| { kind: 'reasoning'; text: string }
|
|
483
|
+
/**
|
|
484
|
+
* `delta_signature` (#10) — the opaque attestation for the reasoning this
|
|
485
|
+
* turn produced. Without decoding it there is nothing to put in the prompt's
|
|
486
|
+
* #12 on the next turn, so the replay would always be unsigned.
|
|
487
|
+
*/
|
|
488
|
+
| { kind: 'reasoning_signature'; signature: string }
|
|
402
489
|
| { kind: 'tool_call_start'; id: string; name: string }
|
|
403
490
|
| {
|
|
404
491
|
kind: 'tool_call_args';
|
|
@@ -592,6 +679,9 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
|
|
|
592
679
|
{
|
|
593
680
|
toolCallId: m.role === 'tool' ? m.tool_call_id : undefined,
|
|
594
681
|
toolCalls: m.role === 'assistant' ? m.tool_calls : undefined,
|
|
682
|
+
thinking: m.role === 'assistant' ? m.thinking : undefined,
|
|
683
|
+
signature: m.role === 'assistant' ? m.signature : undefined,
|
|
684
|
+
signatureType: m.role === 'assistant' ? m.signature_type : undefined,
|
|
595
685
|
},
|
|
596
686
|
),
|
|
597
687
|
),
|
|
@@ -609,6 +699,7 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
|
|
|
609
699
|
// #7 request_type (varint enum)
|
|
610
700
|
// #8 completion_configuration
|
|
611
701
|
// #10 tools (repeated ChatToolDefinition)
|
|
702
|
+
// #13 prompt_cache_options
|
|
612
703
|
// #16 cascade_id (string)
|
|
613
704
|
// #21 chat_model_uid (string)
|
|
614
705
|
// #22 prompt_id (string)
|
|
@@ -622,6 +713,13 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
|
|
|
622
713
|
encodeVarintField(7, args.requestType ?? 5),
|
|
623
714
|
encodeMessage(8, completion),
|
|
624
715
|
...toolParts,
|
|
716
|
+
// #13 prompt_cache_options: { type: EPHEMERAL }. Reusing a session id is only
|
|
717
|
+
// half of prompt caching — without this the server creates no cache entry and
|
|
718
|
+
// every turn re-reads the whole prefix, which is why the sessionId reuse above
|
|
719
|
+
// was not producing the hit ratio its comment claims. The native client sends
|
|
720
|
+
// it and records real savings; sending it unconditionally matches both the
|
|
721
|
+
// native client and CLIProxyAPIPlus, which places it outside its tools gate.
|
|
722
|
+
encodeMessage(13, encodeVarintField(1, PROMPT_CACHE_EPHEMERAL)),
|
|
625
723
|
// #15 session model config: { id, turn, 4 }. Present on every verified
|
|
626
724
|
// request.
|
|
627
725
|
encodeMessage(15, Buffer.concat([
|
|
@@ -669,7 +767,30 @@ function buildGetChatMessageRequest(args: BuildArgs): Buffer {
|
|
|
669
767
|
* any non-zero to 'tool_calls' for now (and let the caller fall back to
|
|
670
768
|
* 'stop' if no tool_call deltas were emitted).
|
|
671
769
|
*/
|
|
672
|
-
function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
|
|
770
|
+
export function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
|
|
771
|
+
// Field 7 is `ModelUsageStats`, the authoritative per-turn accounting, and
|
|
772
|
+
// field 28 is `response_dimension_groups` — the rows the IDE renders. The
|
|
773
|
+
// decoder below reads 28 because a capture happened to expose metric-looking
|
|
774
|
+
// strings there (`ResponseDimension.uid` is its field 5, which is what the
|
|
775
|
+
// entry walker treats as `metric_id`), and that works only when the service
|
|
776
|
+
// chose to render cache rows. Field 7 carries cache read and cache write
|
|
777
|
+
// unconditionally, which is why a cached Devin turn used to report a bare
|
|
778
|
+
// total with no cached subset.
|
|
779
|
+
//
|
|
780
|
+
// Both fields arrive in the same message and the adapter keeps the last usage
|
|
781
|
+
// event it sees, so this cannot be a plain "decode both": field 7 has to
|
|
782
|
+
// suppress field 28 within the message. It is yielded before the rest of the
|
|
783
|
+
// frame rather than after it, so a frame that also carries finish (field 5)
|
|
784
|
+
// still reports usage ahead of the turn's end, and the order does not depend
|
|
785
|
+
// on where the service happens to place the field.
|
|
786
|
+
let authoritativeUsage: CloudChatEvent | null = null;
|
|
787
|
+
for (const f of iterFields(proto)) {
|
|
788
|
+
if (f.num === 7 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
789
|
+
authoritativeUsage = decodeModelUsageStats(f.value as Buffer);
|
|
790
|
+
if (authoritativeUsage) break;
|
|
791
|
+
}
|
|
792
|
+
}
|
|
793
|
+
if (authoritativeUsage) yield authoritativeUsage;
|
|
673
794
|
for (const f of iterFields(proto)) {
|
|
674
795
|
if (f.num === 3 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
675
796
|
// Visible delta_text — what the user should SEE in the chat.
|
|
@@ -693,6 +814,9 @@ function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
|
|
|
693
814
|
// block instead of inline with the answer.
|
|
694
815
|
const s = (f.value as Buffer).toString('utf8');
|
|
695
816
|
if (s) yield { kind: 'reasoning', text: s };
|
|
817
|
+
} else if (f.num === 10 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
818
|
+
const s = (f.value as Buffer).toString('utf8');
|
|
819
|
+
if (s) yield { kind: 'reasoning_signature', signature: s };
|
|
696
820
|
} else if (f.num === 6 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
697
821
|
let id: string | undefined;
|
|
698
822
|
let name: string | undefined;
|
|
@@ -740,6 +864,7 @@ function* decodeChatFrame(proto: Buffer): Generator<CloudChatEvent> {
|
|
|
740
864
|
// else stays 'stop' for 0/2/4-9/12/13
|
|
741
865
|
yield { kind: 'finish', reason };
|
|
742
866
|
} else if (f.num === 28 && f.wire === 2 && Buffer.isBuffer(f.value)) {
|
|
867
|
+
if (authoritativeUsage) continue;
|
|
743
868
|
const usage = decodeUsageBlock(f.value as Buffer);
|
|
744
869
|
if (usage) yield usage;
|
|
745
870
|
}
|
|
@@ -834,6 +959,68 @@ function decodeUsageBlock(buf: Buffer): CloudChatEvent | null {
|
|
|
834
959
|
};
|
|
835
960
|
}
|
|
836
961
|
|
|
962
|
+
/**
|
|
963
|
+
* `exa.codeium_common_pb.ModelUsageStats` at GetChatMessageResponse field 7.
|
|
964
|
+
*
|
|
965
|
+
* ModelUsageStats {
|
|
966
|
+
* #2 input_tokens uint64
|
|
967
|
+
* #3 output_tokens uint64
|
|
968
|
+
* #4 cache_write_tokens uint64
|
|
969
|
+
* #5 cache_read_tokens uint64
|
|
970
|
+
* }
|
|
971
|
+
*
|
|
972
|
+
* Plain varints, so the field-28 entry walker — which descends a
|
|
973
|
+
* length-delimited sub-message and reads a fixed32 float — cannot read this at
|
|
974
|
+
* all. It needs its own decoder.
|
|
975
|
+
*
|
|
976
|
+
* Whether Cognition's `input_tokens` already includes the cached tokens is not
|
|
977
|
+
* settled. oh-my-pi sums all four into its total, which suggests exclusive, but
|
|
978
|
+
* that is their convention rather than a measurement of this field. Guessing
|
|
979
|
+
* wrong in the inclusive direction is the expensive mistake: `normalizeCostTokens`
|
|
980
|
+
* only rejects `read + write > input`, so an inflated input passes validation and
|
|
981
|
+
* bills cached tokens at the uncached rate.
|
|
982
|
+
*
|
|
983
|
+
* So the shape is derived from the frame instead of assumed. An input that
|
|
984
|
+
* already covers the cache is left alone; one that cannot possibly cover it is
|
|
985
|
+
* folded. Both branches agree on the case that motivated this — a 58k prompt
|
|
986
|
+
* that is 57k cache read and 1k fresh reads as 58k with a 57k cached subset —
|
|
987
|
+
* and neither can emit `read + write > input`. Replace the derivation with a
|
|
988
|
+
* fixed mapping once a live frame settles the question.
|
|
989
|
+
*/
|
|
990
|
+
export function decodeModelUsageStats(buf: Buffer): CloudChatEvent | null {
|
|
991
|
+
let wireInput: number | undefined;
|
|
992
|
+
let output: number | undefined;
|
|
993
|
+
let cacheWrite: number | undefined;
|
|
994
|
+
let cacheRead: number | undefined;
|
|
995
|
+
for (const f of iterFields(buf)) {
|
|
996
|
+
if (f.wire !== 0) continue;
|
|
997
|
+
const n = Number(f.value);
|
|
998
|
+
if (!Number.isFinite(n) || n < 0) continue;
|
|
999
|
+
if (f.num === 2) wireInput = n;
|
|
1000
|
+
else if (f.num === 3) output = n;
|
|
1001
|
+
else if (f.num === 4) cacheWrite = n;
|
|
1002
|
+
else if (f.num === 5) cacheRead = n;
|
|
1003
|
+
}
|
|
1004
|
+
if (wireInput === undefined && output === undefined && cacheRead === undefined && cacheWrite === undefined) {
|
|
1005
|
+
return null;
|
|
1006
|
+
}
|
|
1007
|
+
const read = cacheRead ?? 0;
|
|
1008
|
+
const write = cacheWrite ?? 0;
|
|
1009
|
+
const rawInput = wireInput ?? 0;
|
|
1010
|
+
const promptTokens = rawInput >= read + write ? rawInput : rawInput + read + write;
|
|
1011
|
+
const completionTokens = output ?? 0;
|
|
1012
|
+
const total = promptTokens + completionTokens;
|
|
1013
|
+
return {
|
|
1014
|
+
kind: 'usage',
|
|
1015
|
+
promptTokens,
|
|
1016
|
+
completionTokens,
|
|
1017
|
+
totalTokens: total > 0 ? total : undefined,
|
|
1018
|
+
cachedInputTokens: cacheRead,
|
|
1019
|
+
cacheCreationInputTokens: cacheWrite,
|
|
1020
|
+
reasoningTokens: undefined,
|
|
1021
|
+
};
|
|
1022
|
+
}
|
|
1023
|
+
|
|
837
1024
|
// ----------------------------------------------------------------------------
|
|
838
1025
|
// Public API: streamChat
|
|
839
1026
|
// ----------------------------------------------------------------------------
|
|
@@ -864,7 +1051,18 @@ export interface CloudChatRequest {
|
|
|
864
1051
|
}
|
|
865
1052
|
|
|
866
1053
|
export class CloudChatError extends Error {
|
|
867
|
-
constructor(
|
|
1054
|
+
constructor(
|
|
1055
|
+
message: string,
|
|
1056
|
+
public readonly code?: string,
|
|
1057
|
+
public readonly traceId?: string,
|
|
1058
|
+
/**
|
|
1059
|
+
* Upstream HTTP status, when the failure was a status line rather than a
|
|
1060
|
+
* Connect trailer. Without it the adapter's message reaches
|
|
1061
|
+
* `inferHttpStatusFromAdapterMessage`, which does not parse `HTTP 429`, so
|
|
1062
|
+
* a live rate limit was classified 502 and core's failover never rotated.
|
|
1063
|
+
*/
|
|
1064
|
+
public readonly status?: number,
|
|
1065
|
+
) {
|
|
868
1066
|
super(message);
|
|
869
1067
|
this.name = 'CloudChatError';
|
|
870
1068
|
}
|
|
@@ -872,6 +1070,45 @@ export class CloudChatError extends Error {
|
|
|
872
1070
|
|
|
873
1071
|
const TRACE_ID_RE = /\(trace ID: ([0-9a-f]+)\)/i;
|
|
874
1072
|
|
|
1073
|
+
/**
|
|
1074
|
+
* A quota refusal Cognition delivers as `permission_denied`.
|
|
1075
|
+
*
|
|
1076
|
+
* "Your limit will reset in 13 minutes" and "Reached overall message rate
|
|
1077
|
+
* limit" are caps, not authorization failures. Classified as 403 they invite
|
|
1078
|
+
* the client to retry straight into a live cap; as 429 the proxy backs off and
|
|
1079
|
+
* can rotate.
|
|
1080
|
+
*/
|
|
1081
|
+
const TRAILER_QUOTA_RE = /\b(?:limit will reset|rate limit|quota exceeded|out of credits)\b/i;
|
|
1082
|
+
|
|
1083
|
+
/**
|
|
1084
|
+
* Connect error code to HTTP status.
|
|
1085
|
+
*
|
|
1086
|
+
* Without this only the HTTP status line reached the adapter, so a cap or an
|
|
1087
|
+
* expired credential delivered as an EOS trailer fell through to
|
|
1088
|
+
* `inferHttpStatusFromAdapterMessage` and became a generic 502 — which is not
|
|
1089
|
+
* retryable-with-backoff, not an auth prompt, and not something core's failover
|
|
1090
|
+
* acts on.
|
|
1091
|
+
*/
|
|
1092
|
+
export function connectTrailerHttpStatus(code: string | undefined, message: string): number | undefined {
|
|
1093
|
+
if (code === 'permission_denied' && TRAILER_QUOTA_RE.test(message)) return 429;
|
|
1094
|
+
switch (code) {
|
|
1095
|
+
case 'unauthenticated': return 401;
|
|
1096
|
+
case 'permission_denied': return 403;
|
|
1097
|
+
case 'resource_exhausted': return 429;
|
|
1098
|
+
case 'not_found': return 404;
|
|
1099
|
+
case 'unavailable': return 503;
|
|
1100
|
+
case 'deadline_exceeded': return 504;
|
|
1101
|
+
case 'unimplemented': return 501;
|
|
1102
|
+
case 'invalid_argument':
|
|
1103
|
+
case 'failed_precondition':
|
|
1104
|
+
case 'out_of_range': return 400;
|
|
1105
|
+
case 'internal':
|
|
1106
|
+
case 'unknown':
|
|
1107
|
+
case 'data_loss': return 502;
|
|
1108
|
+
default: return undefined;
|
|
1109
|
+
}
|
|
1110
|
+
}
|
|
1111
|
+
|
|
875
1112
|
/**
|
|
876
1113
|
* Stream chat events from the cloud. Yields CloudChatEvent (text deltas, tool
|
|
877
1114
|
* call deltas, finish reason). Use `streamChatText` for legacy text-only iteration.
|
|
@@ -941,12 +1178,24 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
|
|
|
941
1178
|
const framed = frameConnectStream(proto, false);
|
|
942
1179
|
const body = new Blob([new Uint8Array(framed)], { type: "application/connect+proto" });
|
|
943
1180
|
|
|
944
|
-
// Compose caller signal with a
|
|
945
|
-
//
|
|
946
|
-
//
|
|
947
|
-
//
|
|
1181
|
+
// Compose the caller signal with a deadline on the response HEADERS. The
|
|
1182
|
+
// timer is cleared in the finally below, which runs when `await fetch`
|
|
1183
|
+
// resolves — and fetch resolves on headers, not on the first body byte. An
|
|
1184
|
+
// earlier comment here claimed "once any byte arrives", which was wrong and
|
|
1185
|
+
// hid the defect: Cognition withholds headers until the first token, so this
|
|
1186
|
+
// budget is a generation deadline. Body silence after headers is a separate
|
|
1187
|
+
// budget, the per-chunk idle timer in the read loop below.
|
|
948
1188
|
const ttfbController = new AbortController();
|
|
949
|
-
const
|
|
1189
|
+
const headersMs = cloudStreamHeadersMs();
|
|
1190
|
+
// Abort with no reason and remember that we are the one who fired. Bun rejects
|
|
1191
|
+
// the fetch with its own AbortError rather than handing back `signal.reason`,
|
|
1192
|
+
// so attaching a typed error to abort() would be discarded; the catch below is
|
|
1193
|
+
// what actually produces a classifiable failure.
|
|
1194
|
+
let headersDeadlineFired = false;
|
|
1195
|
+
const ttfbTimer = setTimeout(() => {
|
|
1196
|
+
headersDeadlineFired = true;
|
|
1197
|
+
ttfbController.abort();
|
|
1198
|
+
}, headersMs);
|
|
950
1199
|
const ttfbSignal = ttfbController.signal;
|
|
951
1200
|
// Compose req.signal + ttfbSignal. AbortSignal.any was added in Node
|
|
952
1201
|
// 20.3 / Bun 1.0; our `engines` allows Node ≥18, so on Node 18-20.2 the
|
|
@@ -974,7 +1223,28 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
|
|
|
974
1223
|
body,
|
|
975
1224
|
redirect: 'error',
|
|
976
1225
|
signal: initialSignal,
|
|
977
|
-
|
|
1226
|
+
// Bun applies its own fetch idle timeout (~5 minutes) on top of ours.
|
|
1227
|
+
// Two independent deadlines on the same hop means the shorter one wins
|
|
1228
|
+
// silently and this function can no longer explain its own failure, so
|
|
1229
|
+
// the deadline above is made the single authority. Same reason as
|
|
1230
|
+
// src/server/responses/fetch-helpers.ts.
|
|
1231
|
+
timeout: 0,
|
|
1232
|
+
} as RequestInit);
|
|
1233
|
+
} catch (err) {
|
|
1234
|
+
if (headersDeadlineFired) {
|
|
1235
|
+
// Ours, not the upstream failing. Raised as a typed error with an explicit
|
|
1236
|
+
// status because devinErrorClassification reads CloudChatError.status and
|
|
1237
|
+
// would otherwise return {} for a bare Error, leaving src/lib/errors.ts to
|
|
1238
|
+
// guess from the message text. The message deliberately no longer says
|
|
1239
|
+
// "timeout", so the status is the only thing carrying the classification.
|
|
1240
|
+
throw new CloudChatError(
|
|
1241
|
+
`cloud-direct: no response headers within ${headersMs}ms`,
|
|
1242
|
+
undefined,
|
|
1243
|
+
undefined,
|
|
1244
|
+
504,
|
|
1245
|
+
);
|
|
1246
|
+
}
|
|
1247
|
+
throw err;
|
|
978
1248
|
} finally {
|
|
979
1249
|
clearTimeout(ttfbTimer);
|
|
980
1250
|
// The composed signal only guards the headers hop; the body is cancelled
|
|
@@ -987,7 +1257,11 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
|
|
|
987
1257
|
// The body is not echoed into the message. This error reaches the adapter's
|
|
988
1258
|
// error event and /api/logs, and a Connect error can quote the request that
|
|
989
1259
|
// produced it - which is the request holding the api_key.
|
|
990
|
-
|
|
1260
|
+
//
|
|
1261
|
+
// The status line is carried on the error. A cap or an expired credential
|
|
1262
|
+
// delivered instead as a Connect EOS trailer is mapped by
|
|
1263
|
+
// connectTrailerHttpStatus at the trailer sites below.
|
|
1264
|
+
throw new CloudChatError(`GetChatMessage failed (HTTP ${resp.status})`, undefined, undefined, resp.status);
|
|
991
1265
|
}
|
|
992
1266
|
if (!resp.body) {
|
|
993
1267
|
throw new CloudChatError('GetChatMessage response had no body stream');
|
|
@@ -1214,7 +1488,12 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
|
|
|
1214
1488
|
`service accepts. If the request is unchanged and this is new, the ` +
|
|
1215
1489
|
`account's model access is the next thing to check. ` +
|
|
1216
1490
|
`(cloud trace ID: ${trailerError.traceId ?? 'n/a'})`;
|
|
1217
|
-
throw new CloudChatError(
|
|
1491
|
+
throw new CloudChatError(
|
|
1492
|
+
enriched,
|
|
1493
|
+
trailerError.code,
|
|
1494
|
+
trailerError.traceId,
|
|
1495
|
+
connectTrailerHttpStatus(trailerError.code, trailerError.message),
|
|
1496
|
+
);
|
|
1218
1497
|
}
|
|
1219
1498
|
// Cognition also returns `permission_denied` when a tool description
|
|
1220
1499
|
// contains a blocklisted phrase that the sanitizer above did not catch
|
|
@@ -1234,9 +1513,19 @@ export async function* streamChatEvents(req: CloudChatRequest): AsyncGenerator<C
|
|
|
1234
1513
|
// was a phrase match at all.
|
|
1235
1514
|
`(cloud message: ${trailerError.message}) ` +
|
|
1236
1515
|
`(cloud trace ID: ${trailerError.traceId ?? 'n/a'})`;
|
|
1237
|
-
throw new CloudChatError(
|
|
1516
|
+
throw new CloudChatError(
|
|
1517
|
+
enriched,
|
|
1518
|
+
trailerError.code,
|
|
1519
|
+
trailerError.traceId,
|
|
1520
|
+
connectTrailerHttpStatus(trailerError.code, trailerError.message),
|
|
1521
|
+
);
|
|
1238
1522
|
}
|
|
1239
|
-
throw new CloudChatError(
|
|
1523
|
+
throw new CloudChatError(
|
|
1524
|
+
trailerError.message,
|
|
1525
|
+
trailerError.code,
|
|
1526
|
+
trailerError.traceId,
|
|
1527
|
+
connectTrailerHttpStatus(trailerError.code, trailerError.message),
|
|
1528
|
+
);
|
|
1240
1529
|
}
|
|
1241
1530
|
// Truncation detection: the cloud always terminates a successful stream
|
|
1242
1531
|
// with an EOS trailer. If we hit `done` from the body reader without one,
|
|
@@ -55,6 +55,33 @@ const CLOUD_CHAT_OS = 'windows';
|
|
|
55
55
|
*/
|
|
56
56
|
const DEVICE_FINGERPRINT_BYTES = 366;
|
|
57
57
|
|
|
58
|
+
/** Prefix every Cognition session key carries in `Metadata.api_key`. */
|
|
59
|
+
const DEVIN_SESSION_TOKEN_PREFIX = 'devin-session-token$';
|
|
60
|
+
|
|
61
|
+
/** A bare JWT: three base64url segments. Nothing else is reshaped. */
|
|
62
|
+
const BARE_JWT_PATTERN = /^[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]*$/;
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Restore the `devin-session-token$` prefix on a bare JWT.
|
|
66
|
+
*
|
|
67
|
+
* Cognition reads `Metadata.api_key` as a prefixed session token. A key that
|
|
68
|
+
* arrives without the prefix — a JWT pasted into `apiKey` by hand, or one
|
|
69
|
+
* copied out of the CLI's file without its prefix — is sent verbatim and comes
|
|
70
|
+
* back as an opaque `permission_denied`, which reads as a revoked account
|
|
71
|
+
* rather than as a malformed credential.
|
|
72
|
+
*
|
|
73
|
+
* Only a bare JWT is reshaped. The other key formats this field has carried are
|
|
74
|
+
* not JWTs and must pass through untouched: a Codeium-classic bare UUID, an
|
|
75
|
+
* `sk-ws-01-…` Windsurf key, and a `cog_…` session key would all break if they
|
|
76
|
+
* were prefixed. Anything already containing `$` is left alone for the same
|
|
77
|
+
* reason.
|
|
78
|
+
*/
|
|
79
|
+
export function normalizeDevinSessionToken(apiKey: string): string {
|
|
80
|
+
const trimmed = apiKey.trim();
|
|
81
|
+
if (!trimmed || trimmed.includes('$')) return apiKey;
|
|
82
|
+
return BARE_JWT_PATTERN.test(trimmed) ? `${DEVIN_SESSION_TOKEN_PREFIX}${trimmed}` : apiKey;
|
|
83
|
+
}
|
|
84
|
+
|
|
58
85
|
export interface MetadataInput {
|
|
59
86
|
/** Persistent api_key from OAuth (`devin-session-token$<JWT>`). */
|
|
60
87
|
apiKey: string;
|
|
@@ -100,12 +127,14 @@ function osString(): string {
|
|
|
100
127
|
export function buildMetadata(input: MetadataInput): Buffer {
|
|
101
128
|
const version = input.windsurfVersion ?? WINDSURF_VERSION_STRING;
|
|
102
129
|
const os = input.osName ?? osString();
|
|
130
|
+
// One boundary, so no caller has to remember the prefix rule.
|
|
131
|
+
const apiKey = normalizeDevinSessionToken(input.apiKey);
|
|
103
132
|
if (input.cloudChatShape) {
|
|
104
133
|
const clientVersion = input.windsurfVersion ?? CLOUD_CHAT_CLIENT_VERSION;
|
|
105
134
|
return Buffer.concat([
|
|
106
135
|
encodeString(1, CLOUD_CHAT_CLIENT_NAME),
|
|
107
136
|
encodeString(2, clientVersion),
|
|
108
|
-
encodeString(3,
|
|
137
|
+
encodeString(3, apiKey),
|
|
109
138
|
encodeString(4, 'en'),
|
|
110
139
|
encodeString(5, input.osName ?? CLOUD_CHAT_OS),
|
|
111
140
|
encodeString(7, clientVersion),
|
|
@@ -117,7 +146,7 @@ export function buildMetadata(input: MetadataInput): Buffer {
|
|
|
117
146
|
const parts: Buffer[] = [
|
|
118
147
|
encodeString(1, 'windsurf'), // ide_name
|
|
119
148
|
encodeString(2, version), // extension_version
|
|
120
|
-
encodeString(3,
|
|
149
|
+
encodeString(3, apiKey), // api_key
|
|
121
150
|
encodeString(4, 'en'), // locale
|
|
122
151
|
encodeString(5, os), // os
|
|
123
152
|
encodeString(7, version), // ide_version
|
|
@@ -72,7 +72,7 @@ export const DEVIN_MODEL_CONTEXT_WINDOWS: Record<string, number> = {
|
|
|
72
72
|
* Trailing tokens that the Cognition catalog appends as effort/variant
|
|
73
73
|
* suffixes. Stripped to collapse suffixed UIDs to their base id.
|
|
74
74
|
*/
|
|
75
|
-
const EFFORT_TOKENS = new Set([
|
|
75
|
+
export const EFFORT_TOKENS = new Set([
|
|
76
76
|
"low", "medium", "high", "xhigh", "max", "none", "fast", "priority", "1m",
|
|
77
77
|
]);
|
|
78
78
|
|
|
@@ -85,8 +85,61 @@ export function collapseDevinModelUid(uid: string): string {
|
|
|
85
85
|
return parts.join("-");
|
|
86
86
|
}
|
|
87
87
|
|
|
88
|
+
/**
|
|
89
|
+
* The subset of catalog suffix tokens that are reasoning rungs.
|
|
90
|
+
*
|
|
91
|
+
* `fast`, `priority` and `1m` are service tiers and context variants, not effort.
|
|
92
|
+
* Offering them on a reasoning control would name a setting that does something
|
|
93
|
+
* else, so the collapse keeps stripping them while the ladder ignores them.
|
|
94
|
+
*/
|
|
95
|
+
const REASONING_RUNG_TOKENS = new Set(["none", "low", "medium", "high", "xhigh", "max"]);
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* The reasoning rungs a catalog UID carries, in ladder order.
|
|
99
|
+
*
|
|
100
|
+
* Cognition spells effort as a suffix on the model id, so the variants an account
|
|
101
|
+
* actually has ARE its ladder — and collapseDevinModelUid() was throwing exactly
|
|
102
|
+
* that evidence away. Reading it back is what lets every model advertise the rungs
|
|
103
|
+
* it can really run instead of inheriting the generic six-rung default.
|
|
104
|
+
*/
|
|
105
|
+
export function devinReasoningRungsOf(uid: string): string[] {
|
|
106
|
+
const parts = uid.split("-");
|
|
107
|
+
const rungs: string[] = [];
|
|
108
|
+
while (parts.length > 1 && EFFORT_TOKENS.has(parts[parts.length - 1]!)) {
|
|
109
|
+
const token = parts.pop()!;
|
|
110
|
+
if (REASONING_RUNG_TOKENS.has(token)) rungs.push(token);
|
|
111
|
+
}
|
|
112
|
+
return rungs;
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/** Ladder order for display, matching the Codex rung order. */
|
|
116
|
+
const RUNG_ORDER = ["none", "low", "medium", "high", "xhigh", "max"];
|
|
117
|
+
export function sortDevinRungs(rungs: Iterable<string>): string[] {
|
|
118
|
+
return [...new Set(rungs)].sort((a, b) => RUNG_ORDER.indexOf(a) - RUNG_ORDER.indexOf(b));
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* Degraded-mode ladders, used only before the account catalog is readable.
|
|
123
|
+
*
|
|
124
|
+
* Only measured entries belong here. SWE-2 ships exactly three native lanes
|
|
125
|
+
* (see SWE2_EFFORT in src/adapters/devin.ts); inventing ladders for the rest
|
|
126
|
+
* would advertise rungs nobody verified, and the live catalog replaces this
|
|
127
|
+
* table as soon as a credential is present.
|
|
128
|
+
*/
|
|
129
|
+
export const DEVIN_MODEL_EFFORTS: Record<string, string[]> = {
|
|
130
|
+
"swe-2": ["medium", "high", "max"],
|
|
131
|
+
};
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Provider-level fallback ladder. No `ultra`: Cognition has no such lane, and
|
|
135
|
+
* the Codex catalog re-adds its own top rungs anyway (src/codex/catalog/effort.ts).
|
|
136
|
+
* Clients that key an effort control off this list — the Pi-shaped exports — get
|
|
137
|
+
* a control instead of none.
|
|
138
|
+
*/
|
|
139
|
+
export const DEVIN_DEFAULT_EFFORTS = ["low", "medium", "high", "xhigh", "max"];
|
|
140
|
+
|
|
88
141
|
export type DevinUsableModelsResult =
|
|
89
|
-
| { ok: true; models: string[]; contextWindows: Record<string, number> }
|
|
142
|
+
| { ok: true; models: string[]; contextWindows: Record<string, number>; efforts: Record<string, string[]> }
|
|
90
143
|
| { ok: false; error: "auth" | "http" | "empty" | "unknown"; detail?: string };
|
|
91
144
|
|
|
92
145
|
/**
|
|
@@ -105,6 +158,8 @@ export async function fetchDevinUsableModels(opts: {
|
|
|
105
158
|
if (!catalog) return { ok: false, error: "empty" };
|
|
106
159
|
const bases = new Set<string>();
|
|
107
160
|
const contextWindows: Record<string, number> = {};
|
|
161
|
+
// Effort rungs per base, recovered from the suffixes the collapse strips.
|
|
162
|
+
const rungs = new Map<string, Set<string>>();
|
|
108
163
|
for (const entry of catalog.byUid.values()) {
|
|
109
164
|
if (entry.disabled) continue;
|
|
110
165
|
// Skip internal enum constants (e.g. MODEL_GPT_5_2_LOW, MODEL_PRIVATE_*).
|
|
@@ -112,6 +167,12 @@ export async function fetchDevinUsableModels(opts: {
|
|
|
112
167
|
if (entry.modelUid.startsWith("MODEL_")) continue;
|
|
113
168
|
const base = collapseDevinModelUid(entry.modelUid);
|
|
114
169
|
bases.add(base);
|
|
170
|
+
const found = devinReasoningRungsOf(entry.modelUid);
|
|
171
|
+
if (found.length > 0) {
|
|
172
|
+
let set = rungs.get(base);
|
|
173
|
+
if (!set) { set = new Set(); rungs.set(base, set); }
|
|
174
|
+
for (const rung of found) set.add(rung);
|
|
175
|
+
}
|
|
115
176
|
if (entry.contextWindow && entry.contextWindow > 0) {
|
|
116
177
|
// Variants of one base can disagree: the opt-in `-1m` rows report a
|
|
117
178
|
// larger window than the plain row of the same base, and both collapse
|
|
@@ -124,7 +185,13 @@ export async function fetchDevinUsableModels(opts: {
|
|
|
124
185
|
}
|
|
125
186
|
}
|
|
126
187
|
if (bases.size === 0) return { ok: false, error: "empty" };
|
|
127
|
-
|
|
188
|
+
const efforts: Record<string, string[]> = {};
|
|
189
|
+
for (const [base, set] of rungs) {
|
|
190
|
+
// A single rung is not a choice, so it is not a control. Advertising one
|
|
191
|
+
// would draw a picker whose only option is the value already in effect.
|
|
192
|
+
if (set.size > 1) efforts[base] = sortDevinRungs(set);
|
|
193
|
+
}
|
|
194
|
+
return { ok: true, models: [...bases].sort(), contextWindows, efforts };
|
|
128
195
|
} catch (error) {
|
|
129
196
|
const message = error instanceof Error ? error.message : String(error);
|
|
130
197
|
if (/unauth|401|invalid token|login/i.test(message)) return { ok: false, error: "auth", detail: message };
|