@bitkyc08/opencodex 2.61.0 → 2.63.0-preview.20260923

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/gui/dist/assets/App-EJxiUFMq.js +50 -0
  2. package/gui/dist/assets/{Tray-CLZh48fM.js → Tray-B03pW-Uf.js} +1 -1
  3. package/gui/dist/assets/index-BmJwNBHL.js +86 -0
  4. package/gui/dist/assets/index-DdDunwDb.css +1 -0
  5. package/gui/dist/assets/{usage-companion-chart-a0N58rRI.js → usage-companion-chart-IftE60UK.js} +1 -1
  6. package/gui/dist/index.html +2 -2
  7. package/package.json +1 -1
  8. package/src/adapters/cursor/catalog.ts +15 -0
  9. package/src/adapters/cursor/effort-map.ts +7 -0
  10. package/src/adapters/cursor/envelope-echo.ts +51 -25
  11. package/src/adapters/cursor/protobuf-request.ts +39 -13
  12. package/src/adapters/cursor.ts +52 -4
  13. package/src/adapters/devin/cloud-direct/chat.ts +50 -16
  14. package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +4 -1
  15. package/src/adapters/devin/live-models.ts +7 -0
  16. package/src/adapters/devin.ts +7 -5
  17. package/src/adapters/kiro/reasoning.ts +5 -0
  18. package/src/adapters/openai-responses/passthrough.ts +4 -4
  19. package/src/adapters/openai-responses/tool-output-recovery.ts +5 -3
  20. package/src/claude/desktop-gateway-state.ts +7 -2
  21. package/src/claude/desktop-policy.ts +105 -1
  22. package/src/cli/config-command.ts +22 -7
  23. package/src/cli/doctor.ts +14 -4
  24. package/src/codex/catalog/effort.ts +35 -4
  25. package/src/codex/catalog/metadata.ts +33 -5
  26. package/src/codex/catalog/native-models.ts +43 -2
  27. package/src/codex/catalog/pinned-models.ts +37 -0
  28. package/src/codex/catalog-auto-refresh.ts +6 -0
  29. package/src/codex/data/roster-pinned-models.json +359 -0
  30. package/src/codex/data/upstream-models.json +365 -271
  31. package/src/codex/inject/provider-table.ts +108 -0
  32. package/src/codex/inject/remove.ts +2 -114
  33. package/src/codex/model-entitlements.ts +14 -10
  34. package/src/codex/subagent-defaults.ts +2 -109
  35. package/src/codex/toml-source-lines.ts +112 -0
  36. package/src/config/live-reconcile.ts +145 -27
  37. package/src/config/load-degrade.ts +3 -4
  38. package/src/config.ts +2 -2
  39. package/src/generated/compatibility-version.json +83 -67
  40. package/src/generated/model-metadata.ts +5 -5
  41. package/src/lab/artifacts/sanitize.ts +60 -19
  42. package/src/lib/app-owned-memory-stores.ts +7 -0
  43. package/src/lib/app-owned-memory.ts +13 -2
  44. package/src/lib/errors.ts +6 -4
  45. package/src/lib/upstream-retry.ts +35 -5
  46. package/src/oauth/devin.ts +43 -37
  47. package/src/providers/codebuddy-models.ts +11 -0
  48. package/src/providers/kiro-models.ts +9 -0
  49. package/src/providers/quota/vendor-probes-oauth.ts +9 -2
  50. package/src/providers/registry/entries-core.ts +19 -8
  51. package/src/providers/registry/entries-extended.ts +4 -1
  52. package/src/providers/registry/model-seeds.ts +22 -2
  53. package/src/responses/bridge-search-replay-cache.ts +20 -10
  54. package/src/responses/plaintext-v2-agent-messages.ts +10 -1
  55. package/src/routing/identity-domains.ts +22 -15
  56. package/src/server/index/websocket-handler.ts +22 -2
  57. package/src/server/management/agent-settings-routes.ts +19 -10
  58. package/src/server/management/context.ts +4 -2
  59. package/src/server/responses/codex-ws-exchange.ts +24 -8
  60. package/src/server/responses/core-codex-account.ts +4 -0
  61. package/src/server/responses/core-combo-failure.ts +16 -9
  62. package/src/server/responses/core-combo.ts +7 -3
  63. package/src/server/responses/core-options.ts +6 -0
  64. package/src/server/responses/native-injection-replay.ts +13 -1
  65. package/src/server/responses/native-injection.ts +66 -7
  66. package/src/server/responses/native-response-control.ts +6 -2
  67. package/src/server/responses/native-steering-replay.ts +60 -0
  68. package/src/server/responses/native-steering.ts +9 -1
  69. package/src/server/responses/passthrough-delivery.ts +3 -4
  70. package/src/server/responses/passthrough-dispatch.ts +7 -0
  71. package/src/server/responses/request-prepare.ts +12 -0
  72. package/src/server/responses/request-transport.ts +5 -2
  73. package/src/server/responses/ws-upstream.ts +1 -1
  74. package/src/server/ws-bridge.ts +17 -1
  75. package/src/types/request.ts +5 -0
  76. package/src/usage/expected-prices.ts +38 -8
  77. package/src/web-search/executor.ts +38 -13
  78. package/gui/dist/assets/App-CH6C5H7x.js +0 -50
  79. package/gui/dist/assets/index-_bpvxJu0.css +0 -1
  80. package/gui/dist/assets/index-wpTOyepx.js +0 -86
@@ -20,6 +20,7 @@ import {
20
20
  sessionIdHeaderFromRequest,
21
21
  reasoningReplayConversationIdFromResponsesRequest,
22
22
  } from "../request-log-conversation";
23
+ import { resolveContextPrincipal } from "../auth-cors";
23
24
  import {
24
25
  isShadowSourceModel,
25
26
  shadowSourceModelPrefix,
@@ -407,6 +408,17 @@ export async function prepareResponsesRequest(
407
408
  parsed._reasoningReplayScope = { clientThreadId: reasoningReplayConversationId };
408
409
  }
409
410
  }
411
+ if (parsed._reasoningReplayScope) {
412
+ // Scope replay cells to the caller principal. On loopback, admission carries no identity,
413
+ // so resolve it from an opencodex API key the caller volunteered (same rule as context
414
+ // history ownership). A caller that presents none has no principal, and none is invented:
415
+ // every keyless local process would otherwise share one bucket, and a client-visible cell id
416
+ // would become enough to read another caller's retained search result. Without a principal
417
+ // bridgeSearchReplayScope yields no scope, so nothing is recorded or restored for it. The
418
+ // field is always rewritten so an absent principal also clears one a reused holder carried.
419
+ const clientPrincipalId = resolveContextPrincipal(req, config, options.admission);
420
+ parsed._reasoningReplayScope = { ...parsed._reasoningReplayScope, clientPrincipalId };
421
+ }
410
422
  // Prefer a pre-populated id (routed Claude) over Responses headers that may be
411
423
  // absent or synthetically injected (session_id from prompt_cache_key).
412
424
  if (!logCtx.conversationId) {
@@ -445,6 +445,11 @@ export async function prepareResponsesTransport(
445
445
  return response;
446
446
  }
447
447
  const nextAdapter = await refreshDispatchAdapter(requestParsed);
448
+ // Rebind before rebuilding: the rebuild's bridged-search restore and continuation
449
+ // restore key on the serving identity, which must be the refreshed route's, not the
450
+ // credential whose selection just lapsed.
451
+ bindRouteReasoningReplayScope({ parsed: requestParsed, providerName: route.providerName, provider: route.provider,
452
+ adapterName: nextAdapter.name, oauthCredentialSnapshot: replayOAuthCredentialSnapshot });
448
453
  const rebuilt = await nextAdapter.buildRequest(requestParsed, {
449
454
  headers: requestState.selectedForwardHeaders, translatorBudget,
450
455
  ...(imageTierBias > 0 ? { imageTierBias } : {}),
@@ -467,8 +472,6 @@ export async function prepareResponsesTransport(
467
472
  sameTargetToken = transportToken;
468
473
  destination = rebuilt.url;
469
474
  dispatchInit = { ...dispatchInit, method: rebuilt.method, headers, body: rebuilt.body };
470
- bindRouteReasoningReplayScope({ parsed: requestParsed, providerName: route.providerName, provider: route.provider,
471
- adapterName: nextAdapter.name, oauthCredentialSnapshot: replayOAuthCredentialSnapshot });
472
475
  // The next iteration validates synchronously and calls fetch in that same turn.
473
476
  }
474
477
  throw new Error("OAuth account selection changed repeatedly before dispatch");
@@ -138,7 +138,7 @@ export function codexWsUpstreamFetch(
138
138
  // Never infer backend support from a model name or enable controls on a gateway.
139
139
  const control = nativeControl?.kind === "injection"
140
140
  ? ((prepared.canonical || url === OPENAI_API_RESPONSES_URL) && isInjectionRequest(JSON.parse(frameText)) ? nativeControl : undefined)
141
- : (prepared.canonical || url === OPENAI_API_RESPONSES_URL) ? nativeControl : undefined;
141
+ : prepared.canonical ? nativeControl : undefined;
142
142
  if (control?.kind === "injection" && url === OPENAI_API_RESPONSES_URL) {
143
143
  const beta = headers["openai-beta"];
144
144
  if (!beta?.split(",").some(value => value.trim() === "responses_multi_agent=v1")) {
@@ -228,6 +228,22 @@ function sendProtocolError(ws: ServerWebSocket<WsData>, status: number, message:
228
228
  sendJsonFrame(ws, buildWsErrorFrame(status, protocolError(message)));
229
229
  }
230
230
 
231
+ /**
232
+ * Report an upstream-pump failure to the client. Errors that carry a structured
233
+ * code (for example the undeclared-tool guard's undeclared_tool_call) keep it so
234
+ * clients see the same rejection identity as the SSE path; everything else stays
235
+ * a generic protocol error.
236
+ */
237
+ function sendUpstreamError(ws: ServerWebSocket<WsData>, status: number, err: unknown): void {
238
+ const code = err != null && typeof (err as { code?: unknown }).code === "string"
239
+ ? (err as { code: string }).code
240
+ : undefined;
241
+ const message = err instanceof Error ? err.message : String(err);
242
+ sendJsonFrame(ws, buildWsErrorFrame(status, code
243
+ ? { type: "upstream_error", code, message }
244
+ : protocolError(message)));
245
+ }
246
+
231
247
  export async function pumpResponsesSseToWebSocket(
232
248
  ws: ServerWebSocket<WsData>,
233
249
  sseStream: ReadableStream<Uint8Array>,
@@ -319,7 +335,7 @@ export async function pumpResponsesSseToWebSocket(
319
335
  && !(err instanceof WsSendDroppedError)) {
320
336
  reportTerminal("incomplete");
321
337
  try {
322
- sendProtocolError(ws, 502, err instanceof Error ? err.message : String(err));
338
+ sendUpstreamError(ws, 502, err);
323
339
  } catch (sendErr) {
324
340
  // If delivery is already dropped, there is no useful error frame left
325
341
  // to send. Swallow only that expected transport signal; other failures
@@ -30,6 +30,11 @@ export interface OcxReasoningReplayIdentity {
30
30
  * the holder, so late tool-call cache writes see the active physical identity.
31
31
  */
32
32
  export interface OcxReasoningReplayScopeRef {
33
+ /**
34
+ * Process-local caller principal from resolveContextPrincipal. Absent when the caller presented
35
+ * no identity (keyless loopback); replay state keyed by it then fails closed.
36
+ */
37
+ readonly clientPrincipalId?: string;
33
38
  /**
34
39
  * Conversation namespace for replay state. Historically this was always the Codex parent-thread
35
40
  * id; headerless Responses callers use a raw sanitized thread/Cursor/session fallback, never the
@@ -44,6 +44,13 @@ const GEMINI_31_PRO: Cost4 = { input: 2, output: 12, cacheRead: 0.2, cacheWrite:
44
44
  const GPT56_SOL: Cost4 = { input: 4, output: 20, cacheRead: 0.4, cacheWrite: 5 };
45
45
  const GPT6_ASTRA: Cost4 = { input: 10, output: 50, cacheRead: 1, cacheWrite: 12.5 };
46
46
  const ASTRA_API_PRICING = "https://developers.openai.com/api/docs/models/gpt-6-astra";
47
+ /**
48
+ * GPT-6 Sol and Luna API list prices (released 2026-09-22; the changelog publishes input, cached
49
+ * input and output). Cache write follows the 1.25x-input convention every OpenAI row here uses.
50
+ */
51
+ const GPT6_SOL: Cost4 = { input: 2, output: 10, cacheRead: 0.2, cacheWrite: 2.5 };
52
+ const GPT6_LUNA: Cost4 = { input: 0.1, output: 0.5, cacheRead: 0.01, cacheWrite: 0.125 };
53
+ const GPT6_API_PRICING = "https://developers.openai.com/api/docs/changelog (2026-09-22: GPT-6 Sol $2 / $0.20 cached / $10; GPT-6 Luna $0.10 / $0.01 cached / $0.50)";
47
54
  /**
48
55
  * Daybreak aliases. `daybreak-*-latest` never appears in the pricing table itself — only its
49
56
  * current snapshot does — so these tuples are the snapshot's published rates and carry
@@ -101,12 +108,17 @@ const CLAUDE_OPUS_46: Cost4 = { input: 5, output: 25, cacheRead: 0.5, cacheWrite
101
108
  // (0.25) on Fable 5.1 — NOT the 0.1x (1.00) that Fable 5 and every other family use;
102
109
  // the pricing page footnote calls this out explicitly. Verified 2026-09-02.
103
110
  const CLAUDE_FABLE_51: Cost4 = { input: 10, output: 50, cacheRead: 0.25, cacheWrite: 12.5 };
104
- // Opus 5 is priced from the maintainer's confirmation that it matches the previous
105
- // Opus, not from a published Opus 5 page. Hence `verified-derived`, and a source
106
- // string that states the provenance instead of pointing at ANTHROPIC_PRICING.
107
- const CLAUDE_OPUS_5_DERIVED_SOURCE =
108
- "user-confirmed: claude-opus-5 matches Claude Opus 4.6; no separate Anthropic Opus 5 price page verified";
111
+ // Opus 5 was first priced from the maintainer's confirmation that it matched Opus 4.6. The
112
+ // pricing page now lists it at that same 5 / 25 / 0.50 / 6.25 tuple (re-verified 2026-09-23).
113
+ const CLAUDE_OPUS_5 = CLAUDE_OPUS_46;
114
+ // Claude Opus 5.5 (claude-opus-5-5, released 2026-09-22): 4 / 20, 5m cache write 5.00. Cache
115
+ // hits are 0.05x base input (0.20), a model-specific footnote on the pricing page, NOT the
116
+ // 0.1x most families use. 1M context and 128K output at one flat rate (no long-context tier).
117
+ const CLAUDE_OPUS_55: Cost4 = { input: 4, output: 20, cacheRead: 0.2, cacheWrite: 5 };
109
118
  const ANTHROPIC_PRICING = "https://platform.claude.com/docs/en/about-claude/pricing (official; 5m cache-write tier)";
119
+ const CLAUDE_OPUS_5_SOURCE = `anthropic official Claude Opus 5 ${ANTHROPIC_PRICING}`;
120
+ const CLAUDE_OPUS_55_SOURCE = `anthropic official Claude Opus 5.5 ${ANTHROPIC_PRICING}; cache hit = 0.05x base input`;
121
+ const CURSOR_OPUS_55_PRICING = "https://cursor.com/docs/models/claude-opus-5-5 (Cursor Other Models pool; same list rate as Anthropic, Fast Mode billed separately)";
110
122
 
111
123
  const GEMINI_PRICING = "https://ai.google.dev/gemini-api/docs/pricing (2026-07-22); cacheWrite=0: storage is billed per-hour, not per-token";
112
124
  const GEMINI_37_PRICING = "https://ai.google.dev/gemini-api/docs/pricing (2026-08-14); promotional rate through 2026-12-31, rises to 1.50/7.50 on 2027-01-01; cacheWrite=0: storage is billed per-hour, not per-token";
@@ -190,6 +202,10 @@ export const EXPECTED_PRICE_OVERLAYS: readonly ExpectedPriceOverlay[] = [
190
202
  { provider: "openai-apikey", modelId: "gpt-6-astra", cost4: GPT6_ASTRA, source: ASTRA_API_PRICING, verifiedAt: "2026-09-05", status: "verified" },
191
203
  // Display estimates use API prices for both login and API-key routes, including cache writes.
192
204
  { provider: "openai", modelId: "gpt-6-astra", cost4: GPT6_ASTRA, source: `API-reference comparison estimate: ${ASTRA_API_PRICING}`, verifiedAt: "2026-09-05", status: "verified-derived" },
205
+ { provider: "openai-apikey", modelId: "gpt-6-sol", cost4: GPT6_SOL, source: GPT6_API_PRICING, verifiedAt: "2026-09-23", status: "verified" },
206
+ { provider: "openai-apikey", modelId: "gpt-6-luna", cost4: GPT6_LUNA, source: GPT6_API_PRICING, verifiedAt: "2026-09-23", status: "verified" },
207
+ { provider: "openai", modelId: "gpt-6-sol", cost4: GPT6_SOL, source: `API-reference comparison estimate: ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
208
+ { provider: "openai", modelId: "gpt-6-luna", cost4: GPT6_LUNA, source: `API-reference comparison estimate: ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
193
209
  // claude-fable-5-1 now HAS a generated jawcode row, so the two Anthropic surfaces resolve
194
210
  // from it and these overlays are the fallback rather than the primary source. They stay:
195
211
  // the overlay lookup is keyed by the configured provider id, so an account-pool log label
@@ -203,9 +219,15 @@ export const EXPECTED_PRICE_OVERLAYS: readonly ExpectedPriceOverlay[] = [
203
219
  // cost resolution returned null and the Logs `~$` column rendered an em dash. The
204
220
  // model-level vendor fallback only searches jawcode metadata, never overlays, so one
205
221
  // anthropic row would not cover cursor/kiro — each exposing provider needs its own.
206
- { provider: "anthropic", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_46, source: CLAUDE_OPUS_5_DERIVED_SOURCE, verifiedAt: "2026-07-25", status: "verified-derived" },
207
- { provider: "cursor", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_46, source: CLAUDE_OPUS_5_DERIVED_SOURCE, verifiedAt: "2026-07-25", status: "verified-derived" },
208
- { provider: "kiro", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_46, source: CLAUDE_OPUS_5_DERIVED_SOURCE, verifiedAt: "2026-07-25", status: "verified-derived" },
222
+ { provider: "anthropic", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_5, source: CLAUDE_OPUS_5_SOURCE, verifiedAt: "2026-09-23", status: "verified" },
223
+ { provider: "cursor", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_5, source: `${CLAUDE_OPUS_5_SOURCE}; vendor list price applied to the Cursor surface`, verifiedAt: "2026-09-23", status: "verified-derived" },
224
+ { provider: "kiro", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_5, source: `${CLAUDE_OPUS_5_SOURCE}; vendor list price applied to the Kiro credit surface`, verifiedAt: "2026-09-23", status: "verified-derived" },
225
+ // Claude Opus 5.5. The anthropic bundle row wins for the bare provider id; these overlays
226
+ // cover account-label namespaces. Cursor publishes the same list rate on its own model page.
227
+ { provider: "anthropic", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: CLAUDE_OPUS_55_SOURCE, verifiedAt: "2026-09-23", status: "verified" },
228
+ { provider: "anthropic-apikey", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: CLAUDE_OPUS_55_SOURCE, verifiedAt: "2026-09-23", status: "verified" },
229
+ // Cursor canonicalizes every Opus 5.5 spelling (thinking/effort/fast suffixes) onto this row.
230
+ { provider: "cursor", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: CURSOR_OPUS_55_PRICING, verifiedAt: "2026-09-23", status: "verified" },
209
231
  // MiniMax M2.1 highspeed — published PAYG price (verified).
210
232
  { provider: "minimax", modelId: "MiniMax-M2.1-highspeed", cost4: MINIMAX_M21_HIGHSPEED, source: MINIMAX_PRICING, verifiedAt: "2026-07-20", status: "verified" },
211
233
  { provider: "minimax-cn", modelId: "MiniMax-M2.1-highspeed", cost4: MINIMAX_M21_HIGHSPEED, source: MINIMAX_PRICING, verifiedAt: "2026-07-20", status: "verified" },
@@ -371,7 +393,10 @@ export const EXPECTED_PRICE_OVERLAYS: readonly ExpectedPriceOverlay[] = [
371
393
  { provider: "devin-cli", modelId: "swe-1-6", cost4: DEVIN_SWE_17, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
372
394
  { provider: "devin-cli", modelId: "gpt-5-6-sol", cost4: GPT56_SOL, source: `enterprise list column (self-serve shows discounted 1.2/6); ${DEVIN_PRICING}`, verifiedAt: "2026-09-13", status: "verified-derived" },
373
395
  { provider: "devin-cli", modelId: "gpt-6-astra", cost4: GPT6_ASTRA, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
396
+ { provider: "devin-cli", modelId: "gpt-6-sol", cost4: GPT6_SOL, source: `derived: GPT-6 Sol/Luna added 2026-09-23 ahead of Devin's modelCostData table; OpenAI API list price ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
397
+ { provider: "devin-cli", modelId: "gpt-6-luna", cost4: GPT6_LUNA, source: `derived: GPT-6 Sol/Luna added 2026-09-23 ahead of Devin's modelCostData table; OpenAI API list price ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
374
398
  { provider: "devin-cli", modelId: "claude-opus-5", cost4: CLAUDE_OPUS_46, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
399
+ { provider: "devin-cli", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: `derived: live Devin catalog lists claude-opus-5-5 but Devin's modelCostData table does not yet; Anthropic list price shown as estimate ${ANTHROPIC_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
375
400
  { provider: "devin-cli", modelId: "claude-fable-5-1", cost4: CLAUDE_FABLE_51, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
376
401
  { provider: "devin-cli", modelId: "claude-sonnet-5", cost4: DEVIN_SONNET_5, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
377
402
  { provider: "devin-cli", modelId: "glm-5-3", cost4: GLM_53, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
@@ -385,7 +410,10 @@ export const EXPECTED_PRICE_OVERLAYS: readonly ExpectedPriceOverlay[] = [
385
410
  { provider: "devin", modelId: "gpt-5-6-sol", cost4: GPT56_SOL, source: `enterprise list column (self-serve shows discounted 1.2/6); ${DEVIN_PRICING}`, verifiedAt: "2026-09-13", status: "verified-derived" },
386
411
  { provider: "devin", modelId: "gpt-5-6-luna", cost4: GPT56_LUNA, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
387
412
  { provider: "devin", modelId: "gpt-5-6-terra", cost4: GPT56_TERRA, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
413
+ { provider: "devin", modelId: "gpt-6-sol", cost4: GPT6_SOL, source: `derived: GPT-6 Sol/Luna added 2026-09-23 ahead of Devin's modelCostData table; OpenAI API list price ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
414
+ { provider: "devin", modelId: "gpt-6-luna", cost4: GPT6_LUNA, source: `derived: GPT-6 Sol/Luna added 2026-09-23 ahead of Devin's modelCostData table; OpenAI API list price ${GPT6_API_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
388
415
  { provider: "devin", modelId: "claude-opus-4-8", cost4: CLAUDE_OPUS_46, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
416
+ { provider: "devin", modelId: "claude-opus-5-5", cost4: CLAUDE_OPUS_55, source: `derived: live Devin catalog lists claude-opus-5-5 but Devin's modelCostData table does not yet; Anthropic list price shown as estimate ${ANTHROPIC_PRICING}`, verifiedAt: "2026-09-23", status: "verified-derived" },
389
417
  { provider: "devin", modelId: "claude-fable-5-1", cost4: CLAUDE_FABLE_51, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
390
418
  { provider: "devin", modelId: "claude-sonnet-5", cost4: DEVIN_SONNET_5, source: DEVIN_PRICING, verifiedAt: "2026-09-13", status: "verified-derived" },
391
419
  { provider: "devin", modelId: "glm-5-2", cost4: GLM_52, source: `enterprise list column (self-serve shows an unannounced 0 promo); ${DEVIN_PRICING}`, verifiedAt: "2026-09-13", status: "verified-derived" },
@@ -563,6 +591,8 @@ const UNIFORM_DOUBLE: Cost4 = { input: 2, output: 2, cacheRead: 2, cacheWrite: 2
563
591
  const OPENAI_PRICING_DOC = "https://developers.openai.com/api/docs/pricing";
564
592
  const OPENAI_CONTEXT_MODELS = [
565
593
  "gpt-6-astra",
594
+ "gpt-6-sol",
595
+ "gpt-6-luna",
566
596
  "gpt-5.6-sol",
567
597
  "gpt-5.6-terra",
568
598
  "gpt-5.6-luna",
@@ -48,13 +48,17 @@ export type SidecarOutcome = WebSearchResult & { error?: string };
48
48
  *
49
49
  * The forward backend throttles burst sidecar traffic, and without a replay the 429 becomes a
50
50
  * failed tool result that poisons the query for the whole turn (see failedQueries in loop.ts).
51
- * 1 initial send + 2 replays; Retry-After is honored as a lower bound and capped by
52
- * RETRY_AFTER_CEILING_MS (an instruction past the ceiling ends with the 429 instead of
53
- * parking the search). Each wait releases the unread 429 body first so sockets do not
54
- * accumulate under a rate-limit storm. Abort or timeout ends the wait through the existing
55
- * catch, exactly like an abort during the SSE parse.
51
+ * 1 initial send + 2 replays, counted as physical sends: connection-reset recovery inside each
52
+ * send draws from the same SIDECAR_MAX_SENDS budget, so the two layers cannot multiply into nine
53
+ * paid requests during a degraded period. Retry-After is honored as a lower bound and capped by
54
+ * RETRY_AFTER_CEILING_MS and the remaining sidecar deadline (an instruction past either
55
+ * ends with the 429 instead of parking the search). Each wait releases the unread 429 body first so sockets do not
56
+ * accumulate under a rate-limit storm. The release itself may take up to a second, so a
57
+ * deadline landing during release or backoff ends with the 429 already in hand rather than
58
+ * a timeout; a caller abort still ends the wait through the shared catch, exactly like an
59
+ * abort during the SSE parse. An exhausted budget likewise ends with the 429 in hand.
56
60
  */
57
- const SIDECAR_429_MAX_ATTEMPTS = 3;
61
+ const SIDECAR_MAX_SENDS = 3;
58
62
  const SIDECAR_429_BASE_DELAY_MS = 1_000;
59
63
  const SIDECAR_429_MAX_DELAY_MS = 10_000;
60
64
 
@@ -98,10 +102,14 @@ export async function runWebSearch(
98
102
  stream: true,
99
103
  };
100
104
  const url = `${forwardProvider.baseUrl}/responses`;
105
+ // t0 precedes the deadline timer's start so the remaining-time check stays conservative.
106
+ const t0 = Date.now();
101
107
  const linkedSignal = signalWithTimeout(settings.timeoutMs, abortSignal);
102
108
  const sidecarExit = sidecarEnter("web-search");
103
- const t0 = Date.now();
104
109
  try {
110
+ // One physical-send budget for the whole search. Each helper call receives only what is left
111
+ // and reports every send it makes, reset retries included.
112
+ let sendsLeft = SIDECAR_MAX_SENDS;
105
113
  const sendOnce = () => fetchWithResetRetry(
106
114
  // Recovery nests INSIDE the version helper: applyUpstreamRecoveryInit then always receives a
107
115
  // defined init, and withUpstreamHttpVersion spreads the result, so `protocol` and the
@@ -117,10 +125,18 @@ export async function runWebSearch(
117
125
  // `session_id`, and `x-codex-turn-metadata` to the redirect target.
118
126
  redirect: "manual",
119
127
  }, recovery), forwardProvider)),
120
- { replaySafe: true, abortSignal: linkedSignal.signal, label: "web-search-sidecar" },
128
+ {
129
+ replaySafe: true,
130
+ abortSignal: linkedSignal.signal,
131
+ label: "web-search-sidecar",
132
+ attempts: sendsLeft,
133
+ onSendsConsumed: sends => { sendsLeft -= sends; },
134
+ },
121
135
  );
122
136
  let res = await sendOnce();
123
- for (let attempt = 0; res.status === 429 && attempt + 1 < SIDECAR_429_MAX_ATTEMPTS; attempt++) {
137
+ // Checked before the 429 body is released: a budget found spent after the release could only
138
+ // end in a send-budget error, recorded as a connection failure instead of the quota evidence.
139
+ for (let attempt = 0; res.status === 429 && sendsLeft > 0; attempt++) {
124
140
  const delay = retryBackoffDelayMs(attempt, {
125
141
  baseDelayMs: SIDECAR_429_BASE_DELAY_MS,
126
142
  maxDelayMs: SIDECAR_429_MAX_DELAY_MS,
@@ -129,10 +145,19 @@ export async function runWebSearch(
129
145
  });
130
146
  // A deadline, not a clamp: an instruction past the ceiling ends the search with the
131
147
  // 429 instead of parking it at a provider that already said it would refuse.
132
- if (delay > RETRY_AFTER_CEILING_MS) break;
133
- console.warn(`[web-search] sidecar HTTP 429 — retrying (${attempt + 2}/${SIDECAR_429_MAX_ATTEMPTS}) after ${delay}ms`);
134
- await releaseResponseBodyBestEffort(res.body, linkedSignal.signal);
135
- await sleepWithAbort(delay, linkedSignal.signal);
148
+ if (delay > RETRY_AFTER_CEILING_MS || delay >= settings.timeoutMs - (Date.now() - t0)) break;
149
+ console.warn(`[web-search] sidecar HTTP 429 — retrying (send ${SIDECAR_MAX_SENDS - sendsLeft + 1}/${SIDECAR_MAX_SENDS}) after ${delay}ms`);
150
+ try {
151
+ await releaseResponseBodyBestEffort(res.body, linkedSignal.signal);
152
+ await sleepWithAbort(delay, linkedSignal.signal);
153
+ } catch (e) {
154
+ // The release above may consume up to 1s, so the sidecar deadline can land during
155
+ // cleanup or mid-backoff — before the replay is dispatched. The observed 429 is
156
+ // already in hand: end with it rather than laundering it into a timeout. A caller
157
+ // abort (or a non-deadline throw) still propagates to the shared catch below.
158
+ if (!linkedSignal.signal.aborted || linkedSignal.signal.reason === abortSignal?.reason) throw e;
159
+ break;
160
+ }
136
161
  res = await sendOnce();
137
162
  }
138
163
  // Attach the body guard before ANY branch reads it. The success path guarded itself below,