@bitkyc08/opencodex 2.61.0 → 2.63.0-preview.20260923
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/gui/dist/assets/App-EJxiUFMq.js +50 -0
- package/gui/dist/assets/{Tray-CLZh48fM.js → Tray-B03pW-Uf.js} +1 -1
- package/gui/dist/assets/index-BmJwNBHL.js +86 -0
- package/gui/dist/assets/index-DdDunwDb.css +1 -0
- package/gui/dist/assets/{usage-companion-chart-a0N58rRI.js → usage-companion-chart-IftE60UK.js} +1 -1
- package/gui/dist/index.html +2 -2
- package/package.json +1 -1
- package/src/adapters/cursor/catalog.ts +15 -0
- package/src/adapters/cursor/effort-map.ts +7 -0
- package/src/adapters/cursor/envelope-echo.ts +51 -25
- package/src/adapters/cursor/protobuf-request.ts +39 -13
- package/src/adapters/cursor.ts +52 -4
- package/src/adapters/devin/cloud-direct/chat.ts +50 -16
- package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +4 -1
- package/src/adapters/devin/live-models.ts +7 -0
- package/src/adapters/devin.ts +7 -5
- package/src/adapters/kiro/reasoning.ts +5 -0
- package/src/adapters/openai-responses/passthrough.ts +4 -4
- package/src/adapters/openai-responses/tool-output-recovery.ts +5 -3
- package/src/claude/desktop-gateway-state.ts +7 -2
- package/src/claude/desktop-policy.ts +105 -1
- package/src/cli/config-command.ts +22 -7
- package/src/cli/doctor.ts +14 -4
- package/src/codex/catalog/effort.ts +35 -4
- package/src/codex/catalog/metadata.ts +33 -5
- package/src/codex/catalog/native-models.ts +43 -2
- package/src/codex/catalog/pinned-models.ts +37 -0
- package/src/codex/catalog-auto-refresh.ts +6 -0
- package/src/codex/data/roster-pinned-models.json +359 -0
- package/src/codex/data/upstream-models.json +365 -271
- package/src/codex/inject/provider-table.ts +108 -0
- package/src/codex/inject/remove.ts +2 -114
- package/src/codex/model-entitlements.ts +14 -10
- package/src/codex/subagent-defaults.ts +2 -109
- package/src/codex/toml-source-lines.ts +112 -0
- package/src/config/live-reconcile.ts +145 -27
- package/src/config/load-degrade.ts +3 -4
- package/src/config.ts +2 -2
- package/src/generated/compatibility-version.json +83 -67
- package/src/generated/model-metadata.ts +5 -5
- package/src/lab/artifacts/sanitize.ts +60 -19
- package/src/lib/app-owned-memory-stores.ts +7 -0
- package/src/lib/app-owned-memory.ts +13 -2
- package/src/lib/errors.ts +6 -4
- package/src/lib/upstream-retry.ts +35 -5
- package/src/oauth/devin.ts +43 -37
- package/src/providers/codebuddy-models.ts +11 -0
- package/src/providers/kiro-models.ts +9 -0
- package/src/providers/quota/vendor-probes-oauth.ts +9 -2
- package/src/providers/registry/entries-core.ts +19 -8
- package/src/providers/registry/entries-extended.ts +4 -1
- package/src/providers/registry/model-seeds.ts +22 -2
- package/src/responses/bridge-search-replay-cache.ts +20 -10
- package/src/responses/plaintext-v2-agent-messages.ts +10 -1
- package/src/routing/identity-domains.ts +22 -15
- package/src/server/index/websocket-handler.ts +22 -2
- package/src/server/management/agent-settings-routes.ts +19 -10
- package/src/server/management/context.ts +4 -2
- package/src/server/responses/codex-ws-exchange.ts +24 -8
- package/src/server/responses/core-codex-account.ts +4 -0
- package/src/server/responses/core-combo-failure.ts +16 -9
- package/src/server/responses/core-combo.ts +7 -3
- package/src/server/responses/core-options.ts +6 -0
- package/src/server/responses/native-injection-replay.ts +13 -1
- package/src/server/responses/native-injection.ts +66 -7
- package/src/server/responses/native-response-control.ts +6 -2
- package/src/server/responses/native-steering-replay.ts +60 -0
- package/src/server/responses/native-steering.ts +9 -1
- package/src/server/responses/passthrough-delivery.ts +3 -4
- package/src/server/responses/passthrough-dispatch.ts +7 -0
- package/src/server/responses/request-prepare.ts +12 -0
- package/src/server/responses/request-transport.ts +5 -2
- package/src/server/responses/ws-upstream.ts +1 -1
- package/src/server/ws-bridge.ts +17 -1
- package/src/types/request.ts +5 -0
- package/src/usage/expected-prices.ts +38 -8
- package/src/web-search/executor.ts +38 -13
- package/gui/dist/assets/App-CH6C5H7x.js +0 -50
- package/gui/dist/assets/index-_bpvxJu0.css +0 -1
- package/gui/dist/assets/index-wpTOyepx.js +0 -86
|
@@ -20,6 +20,7 @@ import {
|
|
|
20
20
|
sessionIdHeaderFromRequest,
|
|
21
21
|
reasoningReplayConversationIdFromResponsesRequest,
|
|
22
22
|
} from "../request-log-conversation";
|
|
23
|
+
import { resolveContextPrincipal } from "../auth-cors";
|
|
23
24
|
import {
|
|
24
25
|
isShadowSourceModel,
|
|
25
26
|
shadowSourceModelPrefix,
|
|
@@ -407,6 +408,17 @@ export async function prepareResponsesRequest(
|
|
|
407
408
|
parsed._reasoningReplayScope = { clientThreadId: reasoningReplayConversationId };
|
|
408
409
|
}
|
|
409
410
|
}
|
|
411
|
+
if (parsed._reasoningReplayScope) {
|
|
412
|
+
// Scope replay cells to the caller principal. On loopback, admission carries no identity,
|
|
413
|
+
// so resolve it from an opencodex API key the caller volunteered (same rule as context
|
|
414
|
+
// history ownership). A caller that presents none has no principal, and none is invented:
|
|
415
|
+
// every keyless local process would otherwise share one bucket, and a client-visible cell id
|
|
416
|
+
// would become enough to read another caller's retained search result. Without a principal
|
|
417
|
+
// bridgeSearchReplayScope yields no scope, so nothing is recorded or restored for it. The
|
|
418
|
+
// field is always rewritten so an absent principal also clears one a reused holder carried.
|
|
419
|
+
const clientPrincipalId = resolveContextPrincipal(req, config, options.admission);
|
|
420
|
+
parsed._reasoningReplayScope = { ...parsed._reasoningReplayScope, clientPrincipalId };
|
|
421
|
+
}
|
|
410
422
|
// Prefer a pre-populated id (routed Claude) over Responses headers that may be
|
|
411
423
|
// absent or synthetically injected (session_id from prompt_cache_key).
|
|
412
424
|
if (!logCtx.conversationId) {
|
|
@@ -445,6 +445,11 @@ export async function prepareResponsesTransport(
|
|
|
445
445
|
return response;
|
|
446
446
|
}
|
|
447
447
|
const nextAdapter = await refreshDispatchAdapter(requestParsed);
|
|
448
|
+
// Rebind before rebuilding: the rebuild's bridged-search restore and continuation
|
|
449
|
+
// restore key on the serving identity, which must be the refreshed route's, not the
|
|
450
|
+
// credential whose selection just lapsed.
|
|
451
|
+
bindRouteReasoningReplayScope({ parsed: requestParsed, providerName: route.providerName, provider: route.provider,
|
|
452
|
+
adapterName: nextAdapter.name, oauthCredentialSnapshot: replayOAuthCredentialSnapshot });
|
|
448
453
|
const rebuilt = await nextAdapter.buildRequest(requestParsed, {
|
|
449
454
|
headers: requestState.selectedForwardHeaders, translatorBudget,
|
|
450
455
|
...(imageTierBias > 0 ? { imageTierBias } : {}),
|
|
@@ -467,8 +472,6 @@ export async function prepareResponsesTransport(
|
|
|
467
472
|
sameTargetToken = transportToken;
|
|
468
473
|
destination = rebuilt.url;
|
|
469
474
|
dispatchInit = { ...dispatchInit, method: rebuilt.method, headers, body: rebuilt.body };
|
|
470
|
-
bindRouteReasoningReplayScope({ parsed: requestParsed, providerName: route.providerName, provider: route.provider,
|
|
471
|
-
adapterName: nextAdapter.name, oauthCredentialSnapshot: replayOAuthCredentialSnapshot });
|
|
472
475
|
// The next iteration validates synchronously and calls fetch in that same turn.
|
|
473
476
|
}
|
|
474
477
|
throw new Error("OAuth account selection changed repeatedly before dispatch");
|
|
@@ -138,7 +138,7 @@ export function codexWsUpstreamFetch(
|
|
|
138
138
|
// Never infer backend support from a model name or enable controls on a gateway.
|
|
139
139
|
const control = nativeControl?.kind === "injection"
|
|
140
140
|
? ((prepared.canonical || url === OPENAI_API_RESPONSES_URL) && isInjectionRequest(JSON.parse(frameText)) ? nativeControl : undefined)
|
|
141
|
-
:
|
|
141
|
+
: prepared.canonical ? nativeControl : undefined;
|
|
142
142
|
if (control?.kind === "injection" && url === OPENAI_API_RESPONSES_URL) {
|
|
143
143
|
const beta = headers["openai-beta"];
|
|
144
144
|
if (!beta?.split(",").some(value => value.trim() === "responses_multi_agent=v1")) {
|
package/src/server/ws-bridge.ts
CHANGED
|
@@ -228,6 +228,22 @@ function sendProtocolError(ws: ServerWebSocket<WsData>, status: number, message:
|
|
|
228
228
|
sendJsonFrame(ws, buildWsErrorFrame(status, protocolError(message)));
|
|
229
229
|
}
|
|
230
230
|
|
|
231
|
+
/**
|
|
232
|
+
* Report an upstream-pump failure to the client. Errors that carry a structured
|
|
233
|
+
* code (for example the undeclared-tool guard's undeclared_tool_call) keep it so
|
|
234
|
+
* clients see the same rejection identity as the SSE path; everything else stays
|
|
235
|
+
* a generic protocol error.
|
|
236
|
+
*/
|
|
237
|
+
function sendUpstreamError(ws: ServerWebSocket<WsData>, status: number, err: unknown): void {
|
|
238
|
+
const code = err != null && typeof (err as { code?: unknown }).code === "string"
|
|
239
|
+
? (err as { code: string }).code
|
|
240
|
+
: undefined;
|
|
241
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
242
|
+
sendJsonFrame(ws, buildWsErrorFrame(status, code
|
|
243
|
+
? { type: "upstream_error", code, message }
|
|
244
|
+
: protocolError(message)));
|
|
245
|
+
}
|
|
246
|
+
|
|
231
247
|
export async function pumpResponsesSseToWebSocket(
|
|
232
248
|
ws: ServerWebSocket<WsData>,
|
|
233
249
|
sseStream: ReadableStream<Uint8Array>,
|
|
@@ -319,7 +335,7 @@ export async function pumpResponsesSseToWebSocket(
|
|
|
319
335
|
&& !(err instanceof WsSendDroppedError)) {
|
|
320
336
|
reportTerminal("incomplete");
|
|
321
337
|
try {
|
|
322
|
-
|
|
338
|
+
sendUpstreamError(ws, 502, err);
|
|
323
339
|
} catch (sendErr) {
|
|
324
340
|
// If delivery is already dropped, there is no useful error frame left
|
|
325
341
|
// to send. Swallow only that expected transport signal; other failures
|
package/src/types/request.ts
CHANGED
|
@@ -30,6 +30,11 @@ export interface OcxReasoningReplayIdentity {
|
|
|
30
30
|
* the holder, so late tool-call cache writes see the active physical identity.
|
|
31
31
|
*/
|
|
32
32
|
export interface OcxReasoningReplayScopeRef {
|
|
33
|
+
/**
|
|
34
|
+
* Process-local caller principal from resolveContextPrincipal. Absent when the caller presented
|
|
35
|
+
* no identity (keyless loopback); replay state keyed by it then fails closed.
|
|
36
|
+
*/
|
|
37
|
+
readonly clientPrincipalId?: string;
|
|
33
38
|
/**
|
|
34
39
|
* Conversation namespace for replay state. Historically this was always the Codex parent-thread
|
|
35
40
|
* id; headerless Responses callers use a raw sanitized thread/Cursor/session fallback, never the
|
|
@@ -44,6 +44,13 @@ const GEMINI_31_PRO: Cost4 = { input: 2, output: 12, cacheRead: 0.2, cacheWrite:
|
|
|
44
44
|
const GPT56_SOL: Cost4 = { input: 4, output: 20, cacheRead: 0.4, cacheWrite: 5 };
|
|
45
45
|
const GPT6_ASTRA: Cost4 = { input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 };
|
|
46
46
|
const ASTRA_API_PRICING = "https://developers.openai.com/api/docs/models/gpt-6-astra";
|
|
47
|
+
/**
|
|
48
|
+
* GPT-6 Sol and Luna API list prices (released 2026-09-22; the changelog publishes input, cached
|
|
49
|
+
* input and output). Cache write follows the 1.25x-input convention every OpenAI row here uses.
|
|
50
|
+
*/
|
|
51
|
+
const GPT6_SOL: Cost4 = { input: 2, output: 10, cacheRead: 0.2, cacheWrite: 2.5 };
|
|
52
|
+
const GPT6_LUNA: Cost4 = { input: 0.1, output: 0.5, cacheRead: 0.01, cacheWrite: 0.125 };
|
|
53
|
+
const GPT6_API_PRICING = "https://developers.openai.com/api/docs/changelog (2026-09-22: GPT-6 Sol $2 / $0.20 cached / $10; GPT-6 Luna $0.10 / $0.01 cached / $0.50)";
|
|
47
54
|
/**
|
|
48
55
|
* Daybreak aliases. `daybreak-*-latest` never appears in the pricing table itself — only its
|
|
49
56
|
* current snapshot does — so these tuples are the snapshot's published rates and carry
|
|
@@ -101,12 +108,17 @@ const CLAUDE_OPUS_46: Cost4 = { input: 5, output: 25, cacheRead: 0.5, cacheWrite
|
|
|
101
108
|
// (0.25) on Fable 5.1 — NOT the 0.1x (1.00) that Fable 5 and every other family use;
|
|
102
109
|
// the pricing page footnote calls this out explicitly. Verified 2026-09-02.
|
|
103
110
|
const CLAUDE_FABLE_51: Cost4 = { input: 10, output: 50, cacheRead: 0.25, cacheWrite: 12.5 };
|
|
104
|
-
// Opus 5
|
|
105
|
-
//
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
111
|
+
// Opus 5 was first priced from the maintainer's confirmation that it matched Opus 4.6. The
|
|
112
|
+
// pricing page now lists it at that same 5 / 25 / 0.50 / 6.25 tuple (re-verified 2026-09-23).
|
|
113
|
+
const CLAUDE_OPUS_5 = CLAUDE_OPUS_46;
|
|
114
|
+
// Claude Opus 5.5 (claude-opus-5-5, released 2026-09-22): 4 / 20, 5m cache write 5.00. Cache
|
|
115
|
+
// hits are 0.05x base input (0.20), a model-specific footnote on the pricing page, NOT the
|
|
116
|
+
// 0.1x most families use. 1M context and 128K output at one flat rate (no long-context tier).
|
|
117
|
+
const CLAUDE_OPUS_55: Cost4 = { input: 4, output: 20, cacheRead: 0.2, cacheWrite: 5 };
|
|
109
118
|
const ANTHROPIC_PRICING = "https://platform.claude.com/docs/en/about-claude/pricing (official; 5m cache-write tier)";
|
|
119
|
+
const CLAUDE_OPUS_5_SOURCE = `anthropic official Claude Opus 5 ${ANTHROPIC_PRICING}`;
|
|
120
|
+
const CLAUDE_OPUS_55_SOURCE = `anthropic official Claude Opus 5.5 ${ANTHROPIC_PRICING}; cache hit = 0.05x base input`;
|
|
121
|
+
const CURSOR_OPUS_55_PRICING = "https://cursor.com/docs/models/claude-opus-5-5 (Cursor Other Models pool; same list rate as Anthropic, Fast Mode billed separately)";
|
|
110
122
|
|
|
111
123
|
const GEMINI_PRICING = "https://ai.google.dev/gemini-api/docs/pricing (2026-07-22); cacheWrite=0: storage is billed per-hour, not per-token";
|
|
112
124
|
const GEMINI_37_PRICING = "https://ai.google.dev/gemini-api/docs/pricing (2026-08-14); promotional rate through 2026-12-31, rises to 1.50/7.50 on 2027-01-01; cacheWrite=0: storage is billed per-hour, not per-token";
|
|
@@ -190,6 +202,10 @@ export const EXPECTED_PRICE_OVERLAYS: readonly ExpectedPriceOverlay[] = [
|
|
|
190
202
|
{ provider: "openai-apikey", modelId: "gpt-6-astra", cost4: GPT6_ASTRA, source: ASTRA_API_PRICING, verifiedAt: "2026-09-05", status: "verified" },
|
|
191
203
|
// Display estimates use API prices for both login and API-key routes, including cache writes.
|
|
192
204
|
{ provider: "openai", modelId: "gpt-6-astra", cost4: GPT6_ASTRA, source: `API-reference comparison estimate: ${ASTRA_API_PRICING}`, verifiedAt: "2026-09-05", status: "verified-derived" },
|
|
205
|
+
{ provider: "openai-apikey", modelId: "gpt-6-sol", cost4: GPT6_SOL, source: GPT6_API_PRICING, verifiedAt: "2026-09-23", status: "verified" },
|
|
206
|
+
{ provider: "openai-apikey", modelId: "gpt-6-luna", cost4: GPT6_LUNA, source: GPT6_API_PRICING, verifiedAt: "2026-09-23", status: "verified" },
|
|
207
|
+
{ provider: "openai", modelId: "gpt-6-sol", cost4: GPT6_SOL, source: `API-reference comparison estimate: ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
208
|
+
{ provider: "openai", modelId: "gpt-6-luna", cost4: GPT6_LUNA, source: `API-reference comparison estimate: ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
193
209
|
// claude-fable-5-1 now HAS a generated jawcode row, so the two Anthropic surfaces resolve
|
|
194
210
|
// from it and these overlays are the fallback rather than the primary source. They stay:
|
|
195
211
|
// the overlay lookup is keyed by the configured provider id, so an account-pool log label
|
|
@@ -203,9 +219,15 @@ export const EXPECTED_PRICE_OVERLAYS: readonly ExpectedPriceOverlay[] = [
|
|
|
203
219
|
// cost resolution returned null and the Logs `~$` column rendered an em dash. The
|
|
204
220
|
// model-level vendor fallback only searches jawcode metadata, never overlays, so one
|
|
205
221
|
// anthropic row would not cover cursor/kiro — each exposing provider needs its own.
|
|
206
|
-
{ provider: "anthropic", modelId: "claude-opus-5", cost4:
|
|
207
|
-
{ provider: "cursor", modelId: "claude-opus-5", cost4:
|
|
208
|
-
{ provider: "kiro", modelId: "claude-opus-5", cost4:
|
|
222
|
+
{ provider: "anthropic", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_5, source: CLAUDE_OPUS_5_SOURCE, verifiedAt: "2026-09-23", status: "verified" },
|
|
223
|
+
{ provider: "cursor", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_5, source: `${CLAUDE_OPUS_5_SOURCE}; vendor list price applied to the Cursor surface`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
224
|
+
{ provider: "kiro", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_5, source: `${CLAUDE_OPUS_5_SOURCE}; vendor list price applied to the Kiro credit surface`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
225
|
+
// Claude Opus 5.5. The anthropic bundle row wins for the bare provider id; these overlays
|
|
226
|
+
// cover account-label namespaces. Cursor publishes the same list rate on its own model page.
|
|
227
|
+
{ provider: "anthropic", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: CLAUDE_OPUS_55_SOURCE, verifiedAt: "2026-09-23", status: "verified" },
|
|
228
|
+
{ provider: "anthropic-apikey", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: CLAUDE_OPUS_55_SOURCE, verifiedAt: "2026-09-23", status: "verified" },
|
|
229
|
+
// Cursor canonicalizes every Opus 5.5 spelling (thinking/effort/fast suffixes) onto this row.
|
|
230
|
+
{ provider: "cursor", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: CURSOR_OPUS_55_PRICING, verifiedAt: "2026-09-23", status: "verified" },
|
|
209
231
|
// MiniMax M2.1 highspeed — published PAYG price (verified).
|
|
210
232
|
{ provider: "minimax", modelId: "MiniMax-M2.1-highspeed", cost4: MINIMAX_M21_HIGHSPEED, source: MINIMAX_PRICING, verifiedAt: "2026-07-20", status: "verified" },
|
|
211
233
|
{ provider: "minimax-cn", modelId: "MiniMax-M2.1-highspeed", cost4: MINIMAX_M21_HIGHSPEED, source: MINIMAX_PRICING, verifiedAt: "2026-07-20", status: "verified" },
|
|
@@ -371,7 +393,10 @@ export const EXPECTED_PRICE_OVERLAYS: readonly ExpectedPriceOverlay[] = [
|
|
|
371
393
|
{ provider: "devin-cli", modelId: "swe-1-6", cost4: DEVIN_SWE_17, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
372
394
|
{ provider: "devin-cli", modelId: "gpt-5-6-sol", cost4: GPT56_SOL, source: `enterprise list column (self-serve shows discounted 1.2/6); ${DEVIN_PRICING}`, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
373
395
|
{ provider: "devin-cli", modelId: "gpt-6-astra", cost4: GPT6_ASTRA, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
396
|
+
{ provider: "devin-cli", modelId: "gpt-6-sol", cost4: GPT6_SOL, source: `derived: GPT-6 Sol/Luna added 2026-09-23 ahead of Devin's modelCostData table; OpenAI API list price ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
397
|
+
{ provider: "devin-cli", modelId: "gpt-6-luna", cost4: GPT6_LUNA, source: `derived: GPT-6 Sol/Luna added 2026-09-23 ahead of Devin's modelCostData table; OpenAI API list price ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
374
398
|
{ provider: "devin-cli", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_46, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
399
|
+
{ provider: "devin-cli", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: `derived: live Devin catalog lists claude-opus-5-5 but Devin's modelCostData table does not yet; Anthropic list price shown as estimate ${ANTHROPIC_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
375
400
|
{ provider: "devin-cli", modelId: "claude-fable-5-1", cost4: CLAUDE_FABLE_51, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
376
401
|
{ provider: "devin-cli", modelId: "claude-sonnet-5", cost4: DEVIN_SONNET_5, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
377
402
|
{ provider: "devin-cli", modelId: "glm-5-3", cost4: GLM_53, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
@@ -385,7 +410,10 @@ export const EXPECTED_PRICE_OVERLAYS: readonly ExpectedPriceOverlay[] = [
|
|
|
385
410
|
{ provider: "devin", modelId: "gpt-5-6-sol", cost4: GPT56_SOL, source: `enterprise list column (self-serve shows discounted 1.2/6); ${DEVIN_PRICING}`, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
386
411
|
{ provider: "devin", modelId: "gpt-5-6-luna", cost4: GPT56_LUNA, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
387
412
|
{ provider: "devin", modelId: "gpt-5-6-terra", cost4: GPT56_TERRA, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
413
|
+
{ provider: "devin", modelId: "gpt-6-sol", cost4: GPT6_SOL, source: `derived: GPT-6 Sol/Luna added 2026-09-23 ahead of Devin's modelCostData table; OpenAI API list price ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
414
|
+
{ provider: "devin", modelId: "gpt-6-luna", cost4: GPT6_LUNA, source: `derived: GPT-6 Sol/Luna added 2026-09-23 ahead of Devin's modelCostData table; OpenAI API list price ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
388
415
|
{ provider: "devin", modelId: "claude-opus-4-8", cost4: CLAUDE_OPUS_46, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
416
|
+
{ provider: "devin", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: `derived: live Devin catalog lists claude-opus-5-5 but Devin's modelCostData table does not yet; Anthropic list price shown as estimate ${ANTHROPIC_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
|
|
389
417
|
{ provider: "devin", modelId: "claude-fable-5-1", cost4: CLAUDE_FABLE_51, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
390
418
|
{ provider: "devin", modelId: "claude-sonnet-5", cost4: DEVIN_SONNET_5, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
391
419
|
{ provider: "devin", modelId: "glm-5-2", cost4: GLM_52, source: `enterprise list column (self-serve shows an unannounced 0 promo); ${DEVIN_PRICING}`, verifiedAt: "2026-09-13", status: "verified-derived" },
|
|
@@ -563,6 +591,8 @@ const UNIFORM_DOUBLE: Cost4 = { input: 2, output: 2, cacheRead: 2, cacheWrite: 2
|
|
|
563
591
|
const OPENAI_PRICING_DOC = "https://developers.openai.com/api/docs/pricing";
|
|
564
592
|
const OPENAI_CONTEXT_MODELS = [
|
|
565
593
|
"gpt-6-astra",
|
|
594
|
+
"gpt-6-sol",
|
|
595
|
+
"gpt-6-luna",
|
|
566
596
|
"gpt-5.6-sol",
|
|
567
597
|
"gpt-5.6-terra",
|
|
568
598
|
"gpt-5.6-luna",
|
|
@@ -48,13 +48,17 @@ export type SidecarOutcome = WebSearchResult & { error?: string };
|
|
|
48
48
|
*
|
|
49
49
|
* The forward backend throttles burst sidecar traffic, and without a replay the 429 becomes a
|
|
50
50
|
* failed tool result that poisons the query for the whole turn (see failedQueries in loop.ts).
|
|
51
|
-
* 1 initial send + 2 replays
|
|
52
|
-
*
|
|
53
|
-
*
|
|
54
|
-
*
|
|
55
|
-
*
|
|
51
|
+
* 1 initial send + 2 replays, counted as physical sends: connection-reset recovery inside each
|
|
52
|
+
* send draws from the same SIDECAR_MAX_SENDS budget, so the two layers cannot multiply into nine
|
|
53
|
+
* paid requests during a degraded period. Retry-After is honored as a lower bound and capped by
|
|
54
|
+
* RETRY_AFTER_CEILING_MS and the remaining sidecar deadline (an instruction past either
|
|
55
|
+
* ends with the 429 instead of parking the search). Each wait releases the unread 429 body first so sockets do not
|
|
56
|
+
* accumulate under a rate-limit storm. The release itself may take up to a second, so a
|
|
57
|
+
* deadline landing during release or backoff ends with the 429 already in hand rather than
|
|
58
|
+
* a timeout; a caller abort still ends the wait through the shared catch, exactly like an
|
|
59
|
+
* abort during the SSE parse. An exhausted budget likewise ends with the 429 in hand.
|
|
56
60
|
*/
|
|
57
|
-
const
|
|
61
|
+
const SIDECAR_MAX_SENDS = 3;
|
|
58
62
|
const SIDECAR_429_BASE_DELAY_MS = 1_000;
|
|
59
63
|
const SIDECAR_429_MAX_DELAY_MS = 10_000;
|
|
60
64
|
|
|
@@ -98,10 +102,14 @@ export async function runWebSearch(
|
|
|
98
102
|
stream: true,
|
|
99
103
|
};
|
|
100
104
|
const url = `${forwardProvider.baseUrl}/responses`;
|
|
105
|
+
// t0 precedes the deadline timer's start so the remaining-time check stays conservative.
|
|
106
|
+
const t0 = Date.now();
|
|
101
107
|
const linkedSignal = signalWithTimeout(settings.timeoutMs, abortSignal);
|
|
102
108
|
const sidecarExit = sidecarEnter("web-search");
|
|
103
|
-
const t0 = Date.now();
|
|
104
109
|
try {
|
|
110
|
+
// One physical-send budget for the whole search. Each helper call receives only what is left
|
|
111
|
+
// and reports every send it makes, reset retries included.
|
|
112
|
+
let sendsLeft = SIDECAR_MAX_SENDS;
|
|
105
113
|
const sendOnce = () => fetchWithResetRetry(
|
|
106
114
|
// Recovery nests INSIDE the version helper: applyUpstreamRecoveryInit then always receives a
|
|
107
115
|
// defined init, and withUpstreamHttpVersion spreads the result, so `protocol` and the
|
|
@@ -117,10 +125,18 @@ export async function runWebSearch(
|
|
|
117
125
|
// `session_id`, and `x-codex-turn-metadata` to the redirect target.
|
|
118
126
|
redirect: "manual",
|
|
119
127
|
}, recovery), forwardProvider)),
|
|
120
|
-
{
|
|
128
|
+
{
|
|
129
|
+
replaySafe: true,
|
|
130
|
+
abortSignal: linkedSignal.signal,
|
|
131
|
+
label: "web-search-sidecar",
|
|
132
|
+
attempts: sendsLeft,
|
|
133
|
+
onSendsConsumed: sends => { sendsLeft -= sends; },
|
|
134
|
+
},
|
|
121
135
|
);
|
|
122
136
|
let res = await sendOnce();
|
|
123
|
-
|
|
137
|
+
// Checked before the 429 body is released: a budget found spent after the release could only
|
|
138
|
+
// end in a send-budget error, recorded as a connection failure instead of the quota evidence.
|
|
139
|
+
for (let attempt = 0; res.status === 429 && sendsLeft > 0; attempt++) {
|
|
124
140
|
const delay = retryBackoffDelayMs(attempt, {
|
|
125
141
|
baseDelayMs: SIDECAR_429_BASE_DELAY_MS,
|
|
126
142
|
maxDelayMs: SIDECAR_429_MAX_DELAY_MS,
|
|
@@ -129,10 +145,19 @@ export async function runWebSearch(
|
|
|
129
145
|
});
|
|
130
146
|
// A deadline, not a clamp: an instruction past the ceiling ends the search with the
|
|
131
147
|
// 429 instead of parking it at a provider that already said it would refuse.
|
|
132
|
-
if (delay > RETRY_AFTER_CEILING_MS) break;
|
|
133
|
-
console.warn(`[web-search] sidecar HTTP 429 — retrying (${
|
|
134
|
-
|
|
135
|
-
|
|
148
|
+
if (delay > RETRY_AFTER_CEILING_MS || delay >= settings.timeoutMs - (Date.now() - t0)) break;
|
|
149
|
+
console.warn(`[web-search] sidecar HTTP 429 — retrying (send ${SIDECAR_MAX_SENDS - sendsLeft + 1}/${SIDECAR_MAX_SENDS}) after ${delay}ms`);
|
|
150
|
+
try {
|
|
151
|
+
await releaseResponseBodyBestEffort(res.body, linkedSignal.signal);
|
|
152
|
+
await sleepWithAbort(delay, linkedSignal.signal);
|
|
153
|
+
} catch (e) {
|
|
154
|
+
// The release above may consume up to 1s, so the sidecar deadline can land during
|
|
155
|
+
// cleanup or mid-backoff — before the replay is dispatched. The observed 429 is
|
|
156
|
+
// already in hand: end with it rather than laundering it into a timeout. A caller
|
|
157
|
+
// abort (or a non-deadline throw) still propagates to the shared catch below.
|
|
158
|
+
if (!linkedSignal.signal.aborted || linkedSignal.signal.reason === abortSignal?.reason) throw e;
|
|
159
|
+
break;
|
|
160
|
+
}
|
|
136
161
|
res = await sendOnce();
|
|
137
162
|
}
|
|
138
163
|
// Attach the body guard before ANY branch reads it. The success path guarded itself below,
|