@bitkyc08/opencodex 2.58.0 → 2.59.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -10
- package/gui/dist/assets/index-C5IebErG.js +136 -0
- package/gui/dist/assets/{index-C5-RdDmD.css → index-OESInAjC.css} +1 -1
- package/gui/dist/index.html +2 -2
- package/gui/dist/provider-icons/crusoe.svg +1 -0
- package/gui/dist/provider-icons/opper.svg +3 -0
- package/package.json +1 -1
- package/src/adapters/base.ts +11 -1
- package/src/adapters/cursor/catalog.ts +11 -0
- package/src/adapters/cursor/effort-map.ts +16 -2
- package/src/adapters/cursor/envelope-echo.ts +55 -2
- package/src/adapters/cursor/message-mapper.ts +3 -2
- package/src/adapters/cursor/protobuf-request.ts +8 -5
- package/src/adapters/cursor/request-builder.ts +14 -3
- package/src/adapters/cursor/thread-continuity.ts +105 -31
- package/src/adapters/cursor/tool-guidance.ts +5 -4
- package/src/adapters/cursor.ts +42 -1
- package/src/adapters/devin/cloud-direct/chat.ts +11 -2
- package/src/adapters/devin/cloud-direct/index.ts +7 -0
- package/src/adapters/devin/cloud-direct/stated-reset-retry.ts +103 -0
- package/src/adapters/devin.ts +75 -13
- package/src/adapters/google-antigravity-wire.ts +29 -2
- package/src/adapters/google-http.ts +8 -1
- package/src/adapters/google.ts +23 -4
- package/src/adapters/openai-chat/response-events.ts +61 -0
- package/src/adapters/openai-chat.ts +5 -10
- package/src/adapters/openai-responses/passthrough.ts +10 -1
- package/src/adapters/openai-responses/tool-output-recovery.ts +75 -0
- package/src/adapters/openai-responses/tool-schema.ts +19 -7
- package/src/adapters/responses-tool-schema.ts +76 -46
- package/src/adapters/run-turn-queue.ts +17 -4
- package/src/bridge/response-json.ts +1 -1
- package/src/bridge/sse.ts +165 -24
- package/src/claude/context-windows.ts +22 -0
- package/src/claude/outbound.ts +35 -4
- package/src/cli/account-api.ts +4 -3
- package/src/cli/account-extended.ts +22 -2
- package/src/cli/account-orca-import.ts +63 -0
- package/src/cli/account.ts +32 -4
- package/src/cli/capabilities.ts +40 -0
- package/src/cli/claude.ts +29 -1
- package/src/cli/codex-cli-update.ts +97 -2
- package/src/cli/dispatch.ts +54 -0
- package/src/cli/doctor.ts +197 -2
- package/src/cli/help.ts +4 -1
- package/src/cli/index.ts +88 -20
- package/src/cli/models-runtime.ts +33 -4
- package/src/cli/registry.ts +11 -1
- package/src/cli/runtime-api.ts +44 -0
- package/src/cli/start-args.ts +94 -0
- package/src/cli/system-command.ts +2 -0
- package/src/client/machine-api.ts +4 -3
- package/src/client/machine-listener.ts +14 -1
- package/src/clients/config-export/constants.ts +2 -3
- package/src/clients/config-export.ts +5 -5
- package/src/codex/account-store.ts +81 -5
- package/src/codex/auth-api/pool-quota-probe.ts +14 -3
- package/src/codex/auth-api/routes.ts +17 -2
- package/src/codex/auth-context.ts +16 -12
- package/src/codex/catalog/build-entries.ts +25 -4
- package/src/codex/catalog/derive-entry.ts +8 -1
- package/src/codex/catalog/effort.ts +10 -6
- package/src/codex/catalog/gather-capture.ts +1 -0
- package/src/codex/catalog/model-hints.ts +37 -5
- package/src/codex/catalog/parsing.ts +83 -5
- package/src/codex/catalog/reserve-warn.ts +96 -0
- package/src/codex/catalog/retained-sync.ts +19 -0
- package/src/codex/catalog/routed-gather.ts +42 -3
- package/src/codex/cli-installation-identity.ts +210 -0
- package/src/codex/cli-installation-targets.ts +158 -0
- package/src/codex/convergence.ts +5 -0
- package/src/codex/history-provider.ts +4 -1
- package/src/codex/history-state-open.ts +105 -0
- package/src/codex/inject/config-toml.ts +44 -2
- package/src/codex/inject.ts +3 -2
- package/src/codex/lineage.ts +83 -32
- package/src/codex/loopback-target.ts +31 -0
- package/src/codex/main-account-hard-lock.ts +2 -1
- package/src/codex/main-account.ts +10 -3
- package/src/codex/main-device-reauth.ts +17 -9
- package/src/codex/model-entitlements.ts +60 -1
- package/src/codex/observed-model-denials.ts +137 -0
- package/src/codex/orca-auth-source.ts +94 -0
- package/src/codex/orca-import.ts +219 -0
- package/src/codex/prompt-text-probe.ts +282 -12
- package/src/codex/quota-401-recovery.ts +12 -0
- package/src/codex/quota-types.ts +65 -0
- package/src/codex/quota.ts +24 -19
- package/src/codex/routing/cooldown-math.ts +8 -47
- package/src/codex/routing/pin-drain.ts +57 -0
- package/src/codex/routing.ts +13 -15
- package/src/codex/subagent-model-fallback.ts +94 -0
- package/src/codex/windows-installation-files.ts +224 -0
- package/src/combos/failover.ts +122 -5
- package/src/config/diagnostics.ts +21 -0
- package/src/config/load-degrade.ts +15 -0
- package/src/config/pending-teardown.ts +8 -0
- package/src/config/process-state.ts +36 -3
- package/src/config/provider-relative-send-path.ts +16 -0
- package/src/config/proxy-env.ts +23 -5
- package/src/config/schema/config-schema.ts +21 -0
- package/src/config/schema/leaf-validators.ts +64 -17
- package/src/generated/compatibility-version.json +235 -163
- package/src/generated/model-metadata.ts +1 -1
- package/src/lib/bounded-body.ts +4 -2
- package/src/lib/destination-policy.ts +48 -6
- package/src/lib/errors.ts +3 -15
- package/src/lib/local-destinations.ts +32 -5
- package/src/lib/provider-outbound.ts +3 -3
- package/src/lib/proxy-env.ts +70 -3
- package/src/lib/request-execution-budget.ts +11 -3
- package/src/lib/response-body-inactivity.ts +193 -0
- package/src/lib/retry-delay.ts +69 -0
- package/src/lib/socks5-fetch.ts +631 -0
- package/src/lib/spend-reservation-ledger.ts +115 -9
- package/src/lib/workflow-budget.ts +145 -8
- package/src/oauth/account-quota-rank.ts +72 -15
- package/src/oauth/generic-account-failover.ts +40 -27
- package/src/oauth/orcarouter.ts +15 -2
- package/src/oauth/store.ts +8 -0
- package/src/providers/codex-capacity.ts +9 -0
- package/src/providers/devin-provider-merge-migration.ts +33 -12
- package/src/providers/free-directory.ts +20 -2
- package/src/providers/key-failover.ts +261 -7
- package/src/providers/model-rename-migration.ts +1 -0
- package/src/providers/openai-sidecar.ts +4 -0
- package/src/providers/opencode-go-transport.ts +14 -5
- package/src/providers/quota/report-cache.ts +3 -0
- package/src/providers/registry/entries-extended.ts +96 -0
- package/src/providers/registry/model-seeds.ts +78 -21
- package/src/responses/apply-patch-envelope.ts +44 -11
- package/src/responses/bridge-search-replay-cache.ts +152 -0
- package/src/responses/code-mode-helper-compat.ts +26 -16
- package/src/responses/custom-tool-compat.ts +1 -1
- package/src/responses/hosted-tool-policy.ts +85 -2
- package/src/responses/schema.ts +9 -2
- package/src/server/auth-cors.ts +26 -0
- package/src/server/chat-completions.ts +9 -4
- package/src/server/chat-native-sse.ts +26 -9
- package/src/server/chat-native.ts +10 -4
- package/src/server/claude-messages.ts +24 -2
- package/src/server/gui-static.ts +36 -2
- package/src/server/inbound-body-admission.ts +187 -0
- package/src/server/index.ts +15 -19
- package/src/server/management/api-access.ts +3 -4
- package/src/server/management/config-routes.ts +31 -6
- package/src/server/management/provider-capability-config.ts +35 -7
- package/src/server/management/provider-routes.ts +70 -18
- package/src/server/proxy-liveness.ts +97 -2
- package/src/server/relay.ts +17 -24
- package/src/server/request-log.ts +25 -1
- package/src/server/responses/adapter-continuation.ts +71 -27
- package/src/server/responses/adapter-delivery.ts +39 -8
- package/src/server/responses/adapter-dispatch.ts +52 -24
- package/src/server/responses/compact.ts +60 -11
- package/src/server/responses/core-codex-account.ts +83 -22
- package/src/server/responses/core-normalize.ts +12 -5
- package/src/server/responses/fetch-helpers.ts +68 -2
- package/src/server/responses/passthrough-delivery.ts +10 -1
- package/src/server/responses/passthrough-dispatch.ts +113 -48
- package/src/server/responses/passthrough-execution.ts +11 -1
- package/src/server/responses/request-prepare.ts +29 -0
- package/src/server/responses/request-send-budget.ts +84 -7
- package/src/server/responses/request-sidecar-auth.ts +16 -8
- package/src/server/responses/request-spend.ts +38 -9
- package/src/server/responses/request-transport.ts +13 -10
- package/src/server/responses/run-turn-execution.ts +20 -5
- package/src/server/responses/sidecar-execution.ts +2 -0
- package/src/server/responses/ws-upstream.ts +2 -1
- package/src/server/responses-custom-tool-repair.ts +2 -2
- package/src/server/sse-frame-buffer.ts +12 -10
- package/src/server/sse-payload-rewrite.ts +36 -9
- package/src/server/system-env-shell.ts +5 -1
- package/src/server/system-env.ts +7 -1
- package/src/server/workflow-refusal.ts +56 -2
- package/src/service/cli.ts +16 -6
- package/src/service/guards.ts +10 -0
- package/src/service/health.ts +43 -0
- package/src/service/state.ts +7 -2
- package/src/types/accounts.ts +4 -0
- package/src/types/config.ts +100 -3
- package/src/types/provider.ts +19 -0
- package/src/types/request.ts +7 -1
- package/src/types/wire.ts +9 -1
- package/src/usage/expected-prices.ts +28 -0
- package/src/usage/log.ts +87 -4
- package/src/web-search/passthrough-bridge.ts +39 -5
- package/gui/dist/assets/index-BbrHOIY0.js +0 -128
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Same-target retry of an explicit, pre-output 429 refusal with a stated
|
|
3
|
+
* recovery delay. No event-producing attempt is ever automatically replayed.
|
|
4
|
+
* This is a refusal-specific policy, not a claim that every eventless POST
|
|
5
|
+
* is idempotent; ambiguous transport failures still propagate unchanged.
|
|
6
|
+
*/
|
|
7
|
+
import { parseRetryAfterFromMessage } from '../../../lib/retry-delay.js';
|
|
8
|
+
import { abortError, sleepWithAbort } from '../../../lib/upstream-retry.js';
|
|
9
|
+
import { CloudChatError, streamChatEvents, type CloudChatEvent, type CloudChatRequest } from './chat.js';
|
|
10
|
+
|
|
11
|
+
/** 1 initial attempt plus at most 2 replays. */
|
|
12
|
+
export const STATED_RESET_MAX_REPLAYS = 2;
|
|
13
|
+
/** Default cumulative wait allowance for one invocation (30 minutes). */
|
|
14
|
+
export const STATED_RESET_MAX_WAIT_MS = 1_800_000;
|
|
15
|
+
/** Absolute maximum cumulative allowance, including explicit overrides. */
|
|
16
|
+
export const STATED_RESET_WAIT_CEILING_MS = 3_600_000;
|
|
17
|
+
|
|
18
|
+
function statedResetMaxWaitMs(): number {
|
|
19
|
+
const raw = process.env.OPENCODEX_DEVIN_STATED_RESET_WAIT_MS?.trim();
|
|
20
|
+
if (!raw) return STATED_RESET_MAX_WAIT_MS;
|
|
21
|
+
const parsed = Number(raw);
|
|
22
|
+
if (!Number.isFinite(parsed) || parsed < 0) return STATED_RESET_MAX_WAIT_MS;
|
|
23
|
+
// Zero explicitly disables local waiting. Values above one hour are capped.
|
|
24
|
+
return Math.min(Math.floor(parsed), STATED_RESET_WAIT_CEILING_MS);
|
|
25
|
+
}
|
|
26
|
+
export const statedResetMaxWaitMsForTests = statedResetMaxWaitMs;
|
|
27
|
+
|
|
28
|
+
export interface StatedResetRetryOptions {
|
|
29
|
+
/** Test seam: defaults to the real cloud stream. */
|
|
30
|
+
stream?: (req: CloudChatRequest) => AsyncGenerator<CloudChatEvent>;
|
|
31
|
+
/** Test seam: must either honour the whole delay or reject on cancellation. */
|
|
32
|
+
sleep?: (ms: number, signal?: AbortSignal) => Promise<void>;
|
|
33
|
+
maxReplays?: number;
|
|
34
|
+
/** CUMULATIVE wait allowance, not a fresh allowance on every failure. */
|
|
35
|
+
maxWaitMs?: number;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
function replayLimit(value: number | undefined): number {
|
|
39
|
+
if (value === undefined) return STATED_RESET_MAX_REPLAYS;
|
|
40
|
+
if (!Number.isInteger(value) || value < 0 || value > STATED_RESET_MAX_REPLAYS) {
|
|
41
|
+
throw new RangeError('maxReplays must be an integer from 0 to 2');
|
|
42
|
+
}
|
|
43
|
+
return value;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
function waitLimit(value: number | undefined): number {
|
|
47
|
+
if (value === undefined) return statedResetMaxWaitMs();
|
|
48
|
+
if (!Number.isFinite(value) || value < 0) {
|
|
49
|
+
throw new RangeError('maxWaitMs must be a finite non-negative number');
|
|
50
|
+
}
|
|
51
|
+
return Math.min(Math.floor(value), STATED_RESET_WAIT_CEILING_MS);
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export async function* streamChatEventsWithResetRetry(
|
|
55
|
+
req: CloudChatRequest,
|
|
56
|
+
options?: StatedResetRetryOptions,
|
|
57
|
+
): AsyncGenerator<CloudChatEvent> {
|
|
58
|
+
const stream = options?.stream ?? streamChatEvents;
|
|
59
|
+
const sleep = options?.sleep ?? sleepWithAbort;
|
|
60
|
+
const maxReplays = replayLimit(options?.maxReplays);
|
|
61
|
+
const maxWaitMs = waitLimit(options?.maxWaitMs);
|
|
62
|
+
let replays = 0;
|
|
63
|
+
let waitedMs = 0;
|
|
64
|
+
while (true) {
|
|
65
|
+
// Check again after sleeping: cancellation can race with timer completion.
|
|
66
|
+
// A pre-aborted request must not even enter a custom transport.
|
|
67
|
+
if (req.signal?.aborted) throw abortError(req.signal);
|
|
68
|
+
let yielded = false;
|
|
69
|
+
try {
|
|
70
|
+
for await (const event of stream(req)) {
|
|
71
|
+
// Latch before yielding, so a consumer-injected error is post-output.
|
|
72
|
+
yielded = true;
|
|
73
|
+
yield event;
|
|
74
|
+
}
|
|
75
|
+
return;
|
|
76
|
+
} catch (error) {
|
|
77
|
+
if (req.signal?.aborted) throw abortError(req.signal);
|
|
78
|
+
const waitSec = !yielded
|
|
79
|
+
&& error instanceof CloudChatError
|
|
80
|
+
&& error.status === 429
|
|
81
|
+
? parseRetryAfterFromMessage(error.message)
|
|
82
|
+
: undefined;
|
|
83
|
+
const waitMs = waitSec === undefined ? undefined : waitSec * 1000;
|
|
84
|
+
if (
|
|
85
|
+
waitMs === undefined
|
|
86
|
+
|| replays >= maxReplays
|
|
87
|
+
|| waitMs > maxWaitMs - waitedMs
|
|
88
|
+
) {
|
|
89
|
+
// Never shorten a provider's minimum delay to fit the local budget.
|
|
90
|
+
// Keep the original refusal so outer policy can preserve its metadata.
|
|
91
|
+
throw error;
|
|
92
|
+
}
|
|
93
|
+
replays += 1;
|
|
94
|
+
// Charge the complete scheduled wait once, before sleeping. This is a
|
|
95
|
+
// sleep allowance, not a wall-clock deadline on generation or timer
|
|
96
|
+
// scheduling: waking a few milliseconds late must not reject an already
|
|
97
|
+
// approved one-hour retry. No later wait can spend this allowance again.
|
|
98
|
+
waitedMs += waitMs;
|
|
99
|
+
await sleep(waitMs, req.signal);
|
|
100
|
+
if (req.signal?.aborted) throw abortError(req.signal);
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
}
|
package/src/adapters/devin.ts
CHANGED
|
@@ -9,9 +9,9 @@
|
|
|
9
9
|
import type { AdapterEvent, OcxAssistantMessage, OcxContentPart, OcxMessage, OcxParsedRequest, OcxProviderConfig, OcxTool, OcxToolCall, OcxToolResultMessage, OcxUsage } from "../types";
|
|
10
10
|
import { namespacedToolName } from "../types";
|
|
11
11
|
import type { IncomingMeta, ProviderAdapter } from "./base";
|
|
12
|
-
import {
|
|
12
|
+
import { streamChatEventsWithResetRetry, allocateCascadeId, CloudChatError, type ChatHistoryItem, type ToolDef } from "./devin/cloud-direct";
|
|
13
13
|
import type { ContentPart } from "./devin/cloud-direct/chat";
|
|
14
|
-
import { getCachedCatalog } from "./devin/cloud-direct/catalog";
|
|
14
|
+
import { getCachedCatalog, type CacheEntry } from "./devin/cloud-direct/catalog";
|
|
15
15
|
import { collapseDevinModelUid } from "./devin/live-models";
|
|
16
16
|
import { buildNonOpenAIToolCatalogNudgeForTools } from "./tool-catalog-nudge";
|
|
17
17
|
import { DEVIN_DEFAULT_API_SERVER, resolveDevinApiServer } from "../oauth/devin";
|
|
@@ -155,6 +155,7 @@ async function resolveWireModelUid(
|
|
|
155
155
|
apiKey: string,
|
|
156
156
|
host: string,
|
|
157
157
|
reasoningEffort?: string,
|
|
158
|
+
catalog?: CacheEntry | null,
|
|
158
159
|
): Promise<string> {
|
|
159
160
|
const modelId = normalizeDevinModelId(rawModelId);
|
|
160
161
|
// Explicit effort wins over a suffix the picker already baked into the id, so
|
|
@@ -163,15 +164,18 @@ async function resolveWireModelUid(
|
|
|
163
164
|
const swe2 = resolveSwe2Variant(modelId, reasoningEffort);
|
|
164
165
|
if (swe2) return swe2;
|
|
165
166
|
if (hasEffortSuffix(modelId)) return modelId;
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
167
|
+
// Callers that already read the catalog this turn pass it in; an explicit
|
|
168
|
+
// null records a failed lookup and must not trigger a same-turn retry —
|
|
169
|
+
// failures are not cached, so re-reading would only pay another timeout.
|
|
170
|
+
const entry = catalog !== undefined ? catalog : await getCachedCatalog(apiKey, host);
|
|
171
|
+
if (entry) {
|
|
172
|
+
if (entry.byUid.has(modelId)) return modelId;
|
|
169
173
|
const effort = reasoningEffort && CALLER_EFFORT_VALUES.has(reasoningEffort) ? reasoningEffort : "medium";
|
|
170
174
|
const suffixed = `${modelId}-${effort}`;
|
|
171
|
-
if (
|
|
175
|
+
if (entry.byUid.has(suffixed)) return suffixed;
|
|
172
176
|
// Fall back to any enabled variant of this base model.
|
|
173
|
-
for (const uid of
|
|
174
|
-
if (uid.startsWith(modelId + "-") && !
|
|
177
|
+
for (const uid of entry.byUid.keys()) {
|
|
178
|
+
if (uid.startsWith(modelId + "-") && !entry.byUid.get(uid)?.disabled) return uid;
|
|
175
179
|
}
|
|
176
180
|
}
|
|
177
181
|
// Degraded mode: append the default effort suffix.
|
|
@@ -186,6 +190,46 @@ async function resolveWireModelUid(
|
|
|
186
190
|
*/
|
|
187
191
|
export const resolveWireModelUidForTests = resolveWireModelUid;
|
|
188
192
|
|
|
193
|
+
/**
|
|
194
|
+
* Resolve the INPUT ceiling for the exact UID selected for this turn. Catalog
|
|
195
|
+
* ClientModelConfig #18 and CompletionConfiguration #3 both carry input tokens;
|
|
196
|
+
* the independent output cap is not subtracted here. Smaller operator hints
|
|
197
|
+
* cap live evidence, never enlarge it. No evidence leaves the encoder's 128k
|
|
198
|
+
* fallback intact; an unrelated or opt-in long-context variant is not evidence.
|
|
199
|
+
*/
|
|
200
|
+
function resolveDevinMaxInputTokens(
|
|
201
|
+
provider: OcxProviderConfig,
|
|
202
|
+
modelUid: string,
|
|
203
|
+
liveWindow?: number,
|
|
204
|
+
): number | undefined {
|
|
205
|
+
const positive = (value: unknown): number | undefined =>
|
|
206
|
+
typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : undefined;
|
|
207
|
+
const baseId = collapseDevinModelUid(modelUid);
|
|
208
|
+
const configured = (record: Record<string, number> | undefined): number | undefined => {
|
|
209
|
+
if (!record) return undefined;
|
|
210
|
+
for (const id of [modelUid, baseId]) {
|
|
211
|
+
// Prefer the canonical spelling; retain dotted/case-folded saved hints,
|
|
212
|
+
// matching the model-id normalization used for the inference request.
|
|
213
|
+
const exact = Object.hasOwn(record, id) ? positive(record[id]) : undefined;
|
|
214
|
+
if (exact !== undefined) return exact;
|
|
215
|
+
const matches = Object.entries(record)
|
|
216
|
+
.filter(([key]) => normalizeDevinModelId(key).toLowerCase() === id.toLowerCase())
|
|
217
|
+
.map(([, value]) => positive(value))
|
|
218
|
+
.filter((value): value is number => value !== undefined);
|
|
219
|
+
if (matches.length > 0) return Math.min(...matches);
|
|
220
|
+
}
|
|
221
|
+
return undefined;
|
|
222
|
+
};
|
|
223
|
+
const contextHint = configured(provider.modelContextWindows) ?? positive(provider.contextWindow);
|
|
224
|
+
const inputHint = configured(provider.modelMaxInputTokens);
|
|
225
|
+
const ceilings = [positive(liveWindow), contextHint, inputHint]
|
|
226
|
+
.filter((value): value is number => value !== undefined);
|
|
227
|
+
return ceilings.length > 0 ? Math.min(...ceilings) : undefined;
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
/** Pure test seam; runtime uses the same resolver immediately before dispatch. */
|
|
231
|
+
export const resolveDevinMaxInputTokensForTests = resolveDevinMaxInputTokens;
|
|
232
|
+
|
|
189
233
|
export class DevinMissingCredentialError extends Error {
|
|
190
234
|
constructor() {
|
|
191
235
|
super("Devin live transport requires a Devin API key. Run ocx login devin to sign in with your Cognition/Devin account.");
|
|
@@ -493,7 +537,16 @@ export function createDevinAdapter(
|
|
|
493
537
|
// entry: an EU or FedStart account that used provider.baseUrl would send
|
|
494
538
|
// every RPC to the US server it is not provisioned on.
|
|
495
539
|
const host = resolveDevinApiServer(provider.baseUrl, credentialProviderId);
|
|
496
|
-
|
|
540
|
+
// One catalog read per turn serves model-UID resolution, the input
|
|
541
|
+
// ceiling, and the chat pre-flight inside streamChatEvents. Failures are
|
|
542
|
+
// not cached, so a second read would only pay another fetch timeout on
|
|
543
|
+
// an otherwise valid turn.
|
|
544
|
+
const catalog = await getCachedCatalog(apiKey, host, incoming.abortSignal);
|
|
545
|
+
if (incoming.abortSignal?.aborted) {
|
|
546
|
+
emit({ type: "error", message: DEVIN_CLIENT_CLOSED_MESSAGE, status: 499, retryable: false });
|
|
547
|
+
return;
|
|
548
|
+
}
|
|
549
|
+
const modelUid = await resolveWireModelUid(rawModelId, apiKey, host, parsed.options.reasoning, catalog);
|
|
497
550
|
const returnedToolNames = buildDevinReturnedToolNameMap(parsed.context.tools);
|
|
498
551
|
let openToolId: string | undefined;
|
|
499
552
|
let usage: OcxUsage | undefined;
|
|
@@ -506,17 +559,26 @@ export function createDevinAdapter(
|
|
|
506
559
|
};
|
|
507
560
|
|
|
508
561
|
try {
|
|
509
|
-
|
|
562
|
+
// Read the selected UID's catalog row, not the picker's collapsed base.
|
|
563
|
+
const maxInputTokens = resolveDevinMaxInputTokens(
|
|
564
|
+
provider, modelUid, catalog?.byUid.get(modelUid)?.contextWindow,
|
|
565
|
+
);
|
|
566
|
+
// The reset-retry wrapper waits out a 429 that states its own recovery
|
|
567
|
+
// delay ("limit will reset in 35 seconds") and replays the identical
|
|
568
|
+
// request — but only while zero events have been yielded, so a
|
|
569
|
+
// post-output failure still takes the terminal path untouched.
|
|
570
|
+
for await (const event of streamChatEventsWithResetRetry({
|
|
510
571
|
apiKey,
|
|
511
572
|
apiServerUrl: host,
|
|
512
573
|
modelUid,
|
|
574
|
+
catalog,
|
|
513
575
|
messages: mapOcxMessagesToDevin(parsed),
|
|
514
576
|
tools: mapOcxToolsToDevin(parsed.context.tools),
|
|
515
577
|
cascadeId,
|
|
516
|
-
//
|
|
517
|
-
//
|
|
518
|
-
// that asked for a 4k cap never got one.
|
|
578
|
+
// Input and output ceilings are separate wire fields. Omitting the
|
|
579
|
+
// input hint used to force every model through the 128k default.
|
|
519
580
|
completionOpts: {
|
|
581
|
+
...(maxInputTokens !== undefined ? { maxInputTokens } : {}),
|
|
520
582
|
...(typeof parsed.options.maxOutputTokens === "number" ? { maxOutputTokens: parsed.options.maxOutputTokens } : {}),
|
|
521
583
|
...(typeof parsed.options.temperature === "number" ? { temperature: parsed.options.temperature } : {}),
|
|
522
584
|
...(typeof parsed.options.topP === "number" ? { topP: parsed.options.topP } : {}),
|
|
@@ -97,8 +97,35 @@ export function antigravitySessionId(parsed: OcxParsedRequest): string {
|
|
|
97
97
|
* collision, is the failure mode this function exists to prevent.
|
|
98
98
|
*/
|
|
99
99
|
function clientThreadAnchor(parsed: OcxParsedRequest): string | undefined {
|
|
100
|
-
|
|
101
|
-
|
|
100
|
+
// `_clientThreadId` carries `x-codex-parent-thread-id`, which every parallel child of one
|
|
101
|
+
// parent presents identically, so anchoring on it alone collapsed concurrent children onto a
|
|
102
|
+
// single upstream Cloud Code Assist session (#5033).
|
|
103
|
+
//
|
|
104
|
+
// A thread id is only unique WITHIN its parent, which is why `codexConversationIdentity` keys
|
|
105
|
+
// on both. #5054 anchored on the child alone and therefore only moved the collision: two
|
|
106
|
+
// parents can each have a child of the same id (#5058). The pair is the identity, joined by a
|
|
107
|
+
// NUL so the encoding is injective: no Codex id contains one, so a pair cannot be re-read as a
|
|
108
|
+
// different pair, nor as the parent-only anchor below.
|
|
109
|
+
//
|
|
110
|
+
// Deliberately NOT the general lane key: `codexConversationKeyFor` is an HMAC under a
|
|
111
|
+
// process-random secret, so it changes across a proxy restart — and instability, not sharing,
|
|
112
|
+
// is the failure mode this derivation has to avoid. These ids are Codex's own values and
|
|
113
|
+
// survive both compaction and restart.
|
|
114
|
+
const own = parsed._codexOwnThreadId?.trim();
|
|
115
|
+
const parent = parsed._clientThreadId?.trim();
|
|
116
|
+
if (own && parent) return `codex-thread:${parent}\u0000${own}`;
|
|
117
|
+
// Parent-only clients keep the anchor they already had.
|
|
118
|
+
if (parent) return `codex-thread:${parent}`;
|
|
119
|
+
// A parentless ROOT deliberately omits the parent header. `src/server/context-history.ts` says
|
|
120
|
+
// so in as many words: root model requests use (session-id=root, thread-id=root) and do not
|
|
121
|
+
// fabricate a parent key. It has no pair to key on, and #5054's claim that a root presents
|
|
122
|
+
// `thread-id` equal to its parent was simply wrong.
|
|
123
|
+
//
|
|
124
|
+
// So it keeps the pre-#5054 anchor rather than gaining an own-thread one. That is not a
|
|
125
|
+
// preference: durable Antigravity replay state is keyed by model plus session id, and moving a
|
|
126
|
+
// root's anchor on upgrade strands every signature stored under the old session — the exact
|
|
127
|
+
// instability this derivation exists to avoid, introduced while fixing sharing.
|
|
128
|
+
return undefined;
|
|
102
129
|
}
|
|
103
130
|
|
|
104
131
|
/** A Gemini content part as it appears in an Antigravity request body. */
|
|
@@ -112,7 +112,14 @@ export async function fetchGoogleWithRetry(
|
|
|
112
112
|
} catch (err) {
|
|
113
113
|
if (ctx.abortSignal?.aborted) throw err;
|
|
114
114
|
if (err instanceof SendBudgetExhaustedError) {
|
|
115
|
-
if (pendingResponse)
|
|
115
|
+
if (pendingResponse) {
|
|
116
|
+
// The ladder had already classified this response as retryable and was about to send
|
|
117
|
+
// again; the budget refused. Returning the original response is right — it is a real
|
|
118
|
+
// upstream answer — but it used to leave the log indistinguishable from a request
|
|
119
|
+
// where no retry was ever eligible (#5044).
|
|
120
|
+
ctx.onRecoveryWithheld?.({ reason: "retry-send-budget" });
|
|
121
|
+
return ctx.returnRawErrors ? pendingResponse : normalizeFinalGoogleError(label, pendingResponse, ctx.abortSignal);
|
|
122
|
+
}
|
|
116
123
|
throw err;
|
|
117
124
|
}
|
|
118
125
|
lastError = err;
|
package/src/adapters/google.ts
CHANGED
|
@@ -57,7 +57,6 @@ const GOOGLE_BREVITY_INSTRUCTION = [
|
|
|
57
57
|
|
|
58
58
|
const ANTIGRAVITY_REJECTED_CLAUDE_SDK_PARAGRAPH =
|
|
59
59
|
"You are a Claude agent, built on Anthropic's Claude Agent SDK.";
|
|
60
|
-
|
|
61
60
|
/**
|
|
62
61
|
* CCA Flash generations that reject the Claude-Agent identity paragraph.
|
|
63
62
|
*
|
|
@@ -102,6 +101,20 @@ function stripAntigravityRejectedClaudeSdkParagraph(systemText: string): string
|
|
|
102
101
|
.join("\n\n");
|
|
103
102
|
}
|
|
104
103
|
|
|
104
|
+
/**
|
|
105
|
+
* Strips Claude Code CLI's internal billing header (`x-anthropic-billing-header: ...`)
|
|
106
|
+
* at the start of the system prompt, because Cloud Code Assist / Google Antigravity inspects
|
|
107
|
+
* `systemInstruction` and rejects requests containing Anthropic billing metadata with
|
|
108
|
+
* HTTP 429 RESOURCE_EXHAUSTED.
|
|
109
|
+
*
|
|
110
|
+
* Matching is restricted to the prompt start (`^` without the `/m` multiline flag) so that
|
|
111
|
+
* user prompts discussing billing headers in intermediate lines are never modified, and
|
|
112
|
+
* prompts without a billing header preserve their leading whitespace untouched.
|
|
113
|
+
*/
|
|
114
|
+
function stripAntigravityBillingHeader(systemText: string): string {
|
|
115
|
+
return systemText.replace(/^x-anthropic-billing-header:[^\n]*\n*/, "");
|
|
116
|
+
}
|
|
117
|
+
|
|
105
118
|
/**
|
|
106
119
|
* Documented output ceiling for a Google-surface model, or `undefined` when the id is not
|
|
107
120
|
* recognized.
|
|
@@ -276,6 +289,7 @@ function messagesToGeminiFormat(
|
|
|
276
289
|
parsed: OcxParsedRequest,
|
|
277
290
|
identityModelId: string,
|
|
278
291
|
stripRejectedClaudeSdkParagraph = false,
|
|
292
|
+
isCloudCodeAssist = false,
|
|
279
293
|
): { systemInstruction?: unknown; contents: unknown[]; replayedCallIds: string[] } {
|
|
280
294
|
// Neutralize Codex's GPT-5 identity line (Gemini/Antigravity share this path) so a routed model
|
|
281
295
|
// never misreports as GPT-5/OpenAI, and never leaks the proxy identity upstream.
|
|
@@ -285,9 +299,12 @@ function messagesToGeminiFormat(
|
|
|
285
299
|
...(toolCatalogNudge ? [toolCatalogNudge] : []),
|
|
286
300
|
GOOGLE_BREVITY_INSTRUCTION,
|
|
287
301
|
].join("\n\n"), identityModelId);
|
|
288
|
-
|
|
289
|
-
?
|
|
302
|
+
let systemText = isCloudCodeAssist
|
|
303
|
+
? stripAntigravityBillingHeader(identifiedSystemText)
|
|
290
304
|
: identifiedSystemText;
|
|
305
|
+
if (stripRejectedClaudeSdkParagraph) {
|
|
306
|
+
systemText = stripAntigravityRejectedClaudeSdkParagraph(systemText);
|
|
307
|
+
}
|
|
291
308
|
const systemInstruction = { parts: [{ text: systemText }] };
|
|
292
309
|
|
|
293
310
|
const contents: unknown[] = [];
|
|
@@ -833,12 +850,14 @@ export function createGoogleAdapter(provider: OcxProviderConfig): ProviderAdapte
|
|
|
833
850
|
&& /^gemini-/.test(routedModelId) && !isImageCapableModel(parsed.modelId);
|
|
834
851
|
// AI Studio's `-tiered` spelling is wire-only; CCA aliases may migrate to another generation.
|
|
835
852
|
const identityModelId = provider.googleMode === "cloud-code-assist" ? routedModelId : parsed.modelId;
|
|
836
|
-
const
|
|
853
|
+
const isCloudCodeAssist = provider.googleMode === "cloud-code-assist";
|
|
854
|
+
const stripRejectedClaudeSdkParagraph = isCloudCodeAssist
|
|
837
855
|
&& rejectsClaudeSdkParagraph(parsed.modelId, routedModelId);
|
|
838
856
|
const { systemInstruction, contents, replayedCallIds } = messagesToGeminiFormat(
|
|
839
857
|
parsed,
|
|
840
858
|
identityModelId,
|
|
841
859
|
stripRejectedClaudeSdkParagraph,
|
|
860
|
+
isCloudCodeAssist,
|
|
842
861
|
);
|
|
843
862
|
lastInjectedCallIds = [...replayedCallIds];
|
|
844
863
|
lastReasoningReplayScope = parsed._reasoningReplayScope;
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { diagnoseInvalidToolCalls, isRecord, type InvalidToolCallDiagnostic } from "./tool-call-validation";
|
|
2
|
+
import { TranslatorBudgetExceededError, type TranslatorBudget } from "../../lib/translator-budget";
|
|
2
3
|
import type { AdapterEvent, OcxUsage } from "../../types";
|
|
3
4
|
|
|
4
5
|
export function stopReasonFor(finishReason: unknown): "max_tokens" | "content_filter" | undefined {
|
|
@@ -22,6 +23,9 @@ export interface ReasoningDetailSegment {
|
|
|
22
23
|
text: string;
|
|
23
24
|
}
|
|
24
25
|
|
|
26
|
+
const MAX_REASONING_DETAIL_KEY_BYTES = 1024;
|
|
27
|
+
const MAX_REASONING_DETAIL_SEGMENTS = 1024;
|
|
28
|
+
|
|
25
29
|
/**
|
|
26
30
|
* Structured `reasoning_details` array (MiniMax M-series with `reasoning_split`).
|
|
27
31
|
* Each segment's key scopes cumulative-snapshot tracking: upstream repeats the
|
|
@@ -35,6 +39,11 @@ export function reasoningDetailSegmentsFrom(record: Record<string, unknown>): Re
|
|
|
35
39
|
const item: unknown = raw[i];
|
|
36
40
|
if (!isRecord(item)) continue;
|
|
37
41
|
if (typeof item.text !== "string" || item.text.length === 0) continue;
|
|
42
|
+
// Parsing retains nothing, so it rejects nothing. The non-streaming
|
|
43
|
+
// parseResponse path shares this function and reads only `text`; failing a
|
|
44
|
+
// whole valid response there because an opaque upstream id is long would be
|
|
45
|
+
// a new rejection unrelated to the retention bound. The key cap lives with
|
|
46
|
+
// the map that holds the key; see the tracker below.
|
|
38
47
|
const key = typeof item.id === "string" && item.id.length > 0
|
|
39
48
|
? `id:${item.id}`
|
|
40
49
|
: typeof item.index === "number"
|
|
@@ -45,6 +54,58 @@ export function reasoningDetailSegmentsFrom(record: Record<string, unknown>): Re
|
|
|
45
54
|
return segments;
|
|
46
55
|
}
|
|
47
56
|
|
|
57
|
+
/**
|
|
58
|
+
* Per-stream cumulative-snapshot store for structured `reasoning_details`. Each stream chunk
|
|
59
|
+
* repeats a detail's full text-so-far, so deltas are derived by prefix-diffing per segment key;
|
|
60
|
+
* a piece that does not extend the previous snapshot is appended whole, which keeps incremental
|
|
61
|
+
* senders parseable on the same path. Retained key+text bytes are charged to the translator
|
|
62
|
+
* budget under the `reasoning` kind, and both the key length and the segment count are capped
|
|
63
|
+
* here, where the map actually retains them, so a hostile upstream cannot grow it without bound.
|
|
64
|
+
*/
|
|
65
|
+
export function createReasoningDetailSnapshotTracker(budget: TranslatorBudget): {
|
|
66
|
+
ingest(segment: ReasoningDetailSegment): string | null;
|
|
67
|
+
release(): void;
|
|
68
|
+
} {
|
|
69
|
+
const snapshots = new Map<string, string>();
|
|
70
|
+
const encoder = new TextEncoder();
|
|
71
|
+
let retainedBytes = 0;
|
|
72
|
+
return {
|
|
73
|
+
ingest(segment) {
|
|
74
|
+
const existing = snapshots.get(segment.key);
|
|
75
|
+
if (existing === undefined && encoder.encode(segment.key).byteLength > MAX_REASONING_DETAIL_KEY_BYTES) {
|
|
76
|
+
throw new TranslatorBudgetExceededError("reasoning", MAX_REASONING_DETAIL_KEY_BYTES);
|
|
77
|
+
}
|
|
78
|
+
if (existing === undefined && snapshots.size >= MAX_REASONING_DETAIL_SEGMENTS) {
|
|
79
|
+
throw new TranslatorBudgetExceededError("reasoning", MAX_REASONING_DETAIL_SEGMENTS);
|
|
80
|
+
}
|
|
81
|
+
const prev = existing ?? "";
|
|
82
|
+
if (segment.text === prev) return null;
|
|
83
|
+
const extendsPrev = segment.text.startsWith(prev);
|
|
84
|
+
const next = extendsPrev ? segment.text : prev + segment.text;
|
|
85
|
+
const previousBytes = existing === undefined
|
|
86
|
+
? 0
|
|
87
|
+
: encoder.encode(segment.key).byteLength + encoder.encode(prev).byteLength;
|
|
88
|
+
const nextBytes = encoder.encode(segment.key).byteLength + encoder.encode(next).byteLength;
|
|
89
|
+
const reservation = budget.reserveTransient(nextBytes, { kind: "reasoning" });
|
|
90
|
+
try {
|
|
91
|
+
snapshots.set(segment.key, next);
|
|
92
|
+
reservation.commitRetained();
|
|
93
|
+
budget.releaseRetained(previousBytes, { kind: "reasoning" });
|
|
94
|
+
retainedBytes += nextBytes - previousBytes;
|
|
95
|
+
} catch (error) {
|
|
96
|
+
reservation.release();
|
|
97
|
+
throw error;
|
|
98
|
+
}
|
|
99
|
+
return extendsPrev ? segment.text.slice(prev.length) : segment.text;
|
|
100
|
+
},
|
|
101
|
+
release() {
|
|
102
|
+
budget.releaseRetained(retainedBytes, { kind: "reasoning" });
|
|
103
|
+
retainedBytes = 0;
|
|
104
|
+
snapshots.clear();
|
|
105
|
+
},
|
|
106
|
+
};
|
|
107
|
+
}
|
|
108
|
+
|
|
48
109
|
/** Single-segment `reasoning_details` entry for replaying preserved reasoning (MiniMax wire shape). */
|
|
49
110
|
export function reasoningDetailSegmentForWire(text: string): Record<string, unknown> {
|
|
50
111
|
return { type: "reasoning.text", id: "reasoning-text-1", format: "MiniMax-response-v1", index: 0, text };
|
|
@@ -23,6 +23,7 @@ import {
|
|
|
23
23
|
type InvalidToolCallDiagnostic,
|
|
24
24
|
} from "./openai-chat/tool-call-validation";
|
|
25
25
|
import {
|
|
26
|
+
createReasoningDetailSnapshotTracker,
|
|
26
27
|
invalidChoicesEvent,
|
|
27
28
|
invalidToolCallsEvent,
|
|
28
29
|
reasoningDetailSegmentsFrom,
|
|
@@ -385,7 +386,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd
|
|
|
385
386
|
// full text-so-far, so deltas are derived by prefix-diffing per segment key.
|
|
386
387
|
// A piece that does not extend the previous snapshot is appended whole, which
|
|
387
388
|
// keeps incremental senders parseable on the same path.
|
|
388
|
-
const
|
|
389
|
+
const reasoningDetailTracker = createReasoningDetailSnapshotTracker(budget);
|
|
389
390
|
// Gate on the routed model, not list length: a mixed openai-chat provider
|
|
390
391
|
// can list MiniMax ids without putting every sibling on MiniMax semantics.
|
|
391
392
|
const reasoningDetailsOptIn = modelInList(provider.reasoningDetailsModels, lastRequestedModelId ?? "");
|
|
@@ -448,15 +449,8 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd
|
|
|
448
449
|
const detailSegments = reasoningDetailsOptIn ? reasoningDetailSegmentsFrom(delta) : [];
|
|
449
450
|
if (detailSegments.length > 0) {
|
|
450
451
|
for (const segment of detailSegments) {
|
|
451
|
-
const
|
|
452
|
-
if (
|
|
453
|
-
if (segment.text.startsWith(prev)) {
|
|
454
|
-
reasoningDetailSnapshots.set(segment.key, segment.text);
|
|
455
|
-
yield { type: "reasoning_raw_delta", text: segment.text.slice(prev.length) };
|
|
456
|
-
} else {
|
|
457
|
-
reasoningDetailSnapshots.set(segment.key, prev + segment.text);
|
|
458
|
-
yield { type: "reasoning_raw_delta", text: segment.text };
|
|
459
|
-
}
|
|
452
|
+
const reasoningDelta = reasoningDetailTracker.ingest(segment);
|
|
453
|
+
if (reasoningDelta !== null) yield { type: "reasoning_raw_delta", text: reasoningDelta };
|
|
460
454
|
}
|
|
461
455
|
} else {
|
|
462
456
|
const reasoningText = reasoningTextFrom(delta);
|
|
@@ -692,6 +686,7 @@ export function createOpenAIChatAdapter(provider: OcxProviderConfig): ProviderAd
|
|
|
692
686
|
throw error;
|
|
693
687
|
} finally {
|
|
694
688
|
budget.releaseRetained(bufferBytes, { kind: "live_transient" });
|
|
689
|
+
reasoningDetailTracker.release();
|
|
695
690
|
closeToolCalls();
|
|
696
691
|
reader.releaseLock();
|
|
697
692
|
}
|
|
@@ -36,7 +36,8 @@ import { scrubOcxCompactionItems, stripCanonicalOnlyToolFields, stripCanonicalOn
|
|
|
36
36
|
import { stripCanonicalForwardPromptCacheOptions, stripDeprecatedPromptCacheRetention } from "./prompt-cache";
|
|
37
37
|
import { isPlainObject } from "./internal";
|
|
38
38
|
import { normalizeToolSchemas, promoteClientLoadedTools, stripUnsupportedHostedTools } from "./tool-schema";
|
|
39
|
-
import { annotateEmptyResponsesToolOutputs, backfillWebSearchQueries, normalizeResponsesToolResultAdjacency, repairOrphanedInputItems, repairOversizedReplayCallIds, repairUnidentifiedToolOutputItems } from "./tool-output-recovery";
|
|
39
|
+
import { annotateEmptyResponsesToolOutputs, backfillWebSearchQueries, normalizeResponsesToolResultAdjacency, repairOrphanedInputItems, repairOversizedReplayCallIds, repairUnidentifiedToolOutputItems, restoreBridgedWebSearchCalls } from "./tool-output-recovery";
|
|
40
|
+
import { bridgeSearchReplayScope } from "../../responses/bridge-search-replay-cache";
|
|
40
41
|
import { applyTierDecisionToResponsesBody, normalizeCanonicalForwardContinuationEnvelope, normalizeCanonicalForwardPromptEnvelope, stripCanonicalForwardSamplingParams, stripPreviousResponseId, stripStatefulResponsesParams, stripUnsupportedForwardParams } from "./canonical-forward";
|
|
41
42
|
import { normalizeImageGenClientTools, preferConfiguredHostedTools } from "./image-gen";
|
|
42
43
|
import { stripMuseSparkUnsupportedWebSearchFields, stripOpenAiOnlyWebSearchFields } from "./web-search";
|
|
@@ -321,6 +322,14 @@ export function createResponsesPassthroughAdapter(provider: OcxProviderConfig):
|
|
|
321
322
|
outBody = repairOversizedReplayCallIds(outBody);
|
|
322
323
|
}
|
|
323
324
|
outBody = stripUnsupportedReasoningSummaryDelivery(outBody, parsed.modelId);
|
|
325
|
+
// #4587: on a bridged provider, hand the destination back the search call and result the
|
|
326
|
+
// proxy executed on its behalf, in place of the hosted cell the caller replays. Scoped to
|
|
327
|
+
// this destination and recorded by the bridge itself, so a provider without the opt-in
|
|
328
|
+
// computes no identity and keeps the body reference it already had. This runs before the
|
|
329
|
+
// query backfill below because a restored cell is no longer a web_search_call to repair.
|
|
330
|
+
if (provider.webSearchBridge?.enabled === true) {
|
|
331
|
+
outBody = restoreBridgedWebSearchCalls(outBody, bridgeSearchReplayScope(provider.baseUrl));
|
|
332
|
+
}
|
|
324
333
|
// Repair stored history from before the bridge emitted both keys, in either
|
|
325
334
|
// direction: a conversation that already recorded a web_search_call replays it
|
|
326
335
|
// every turn, and a strict parser rejects the whole request over the missing key —
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { createHash } from "node:crypto";
|
|
2
2
|
import { EMPTY_TOOL_OUTPUT_ANNOTATION, isWhitespaceOnlyTextPartArray } from "../empty-tool-output-annotation";
|
|
3
3
|
import { isPlainObject } from "./internal";
|
|
4
|
+
import { peekBridgeSearchReplay } from "../../responses/bridge-search-replay-cache";
|
|
4
5
|
|
|
5
6
|
const MAX_RESPONSES_CALL_ID_LENGTH = 64;
|
|
6
7
|
|
|
@@ -265,6 +266,80 @@ export function backfillWebSearchQueries(body: unknown): unknown {
|
|
|
265
266
|
return changed ? { ...body, input } : body;
|
|
266
267
|
}
|
|
267
268
|
|
|
269
|
+
/**
|
|
270
|
+
* Give a bridged destination back its own search call and result (issue #4587).
|
|
271
|
+
*
|
|
272
|
+
* When `providers.<name>.webSearchBridge` is armed, the proxy intercepts the destination's
|
|
273
|
+
* `function_call` named `web_search`, runs the search, and shows the CALLER a hosted
|
|
274
|
+
* `web_search_call` cell. The caller stores that cell and replays it on every later turn, so the
|
|
275
|
+
* destination receives an item type it never produced, carrying a query and sources but no result.
|
|
276
|
+
* It typically responds by searching again.
|
|
277
|
+
*
|
|
278
|
+
* This restores the exchange the destination actually had: the cell becomes the destination's own
|
|
279
|
+
* `function_call`, immediately followed by the `function_call_output` the bridge produced for
|
|
280
|
+
* it, in the cell's original position. It runs before the first leg of the next turn is
|
|
281
|
+
* dispatched, which is the only place it can run — by the time the bridge wraps a turn, that
|
|
282
|
+
* turn's first leg is already on the wire.
|
|
283
|
+
*
|
|
284
|
+
* Three things it deliberately does not do:
|
|
285
|
+
* - It never re-runs a search. A missing memo entry means the result is gone, and paying for a
|
|
286
|
+
* second search would answer the model with a different search than its history claims.
|
|
287
|
+
* - It never invents result text. A miss leaves the item exactly as the caller sent it, which is
|
|
288
|
+
* the behaviour every unbridged conversation already has.
|
|
289
|
+
* - It never restores a call id the body already carries. If the history somehow holds that
|
|
290
|
+
* `function_call` too, emitting a second one would be a duplicate the upstream must reject.
|
|
291
|
+
*
|
|
292
|
+
* Entries are scoped to the upstream destination, so a history replayed against a different
|
|
293
|
+
* provider cannot resurrect a call that provider never made. Callers pass `undefined` for any
|
|
294
|
+
* provider without the bridge armed, and the common path then returns the original reference.
|
|
295
|
+
*/
|
|
296
|
+
export function restoreBridgedWebSearchCalls(body: unknown, destinationScope: string | undefined): unknown {
|
|
297
|
+
if (destinationScope === undefined) return body;
|
|
298
|
+
if (!isPlainObject(body) || !Array.isArray(body.input)) return body;
|
|
299
|
+
const input = body.input;
|
|
300
|
+
|
|
301
|
+
// Cheap pre-check: nothing to do for a conversation that carries no hosted search cell at all,
|
|
302
|
+
// which is every turn before the model's first bridged search.
|
|
303
|
+
let hasCell = false;
|
|
304
|
+
for (const item of input) {
|
|
305
|
+
if (isPlainObject(item) && item.type === "web_search_call" && typeof item.id === "string") {
|
|
306
|
+
hasCell = true;
|
|
307
|
+
break;
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
if (!hasCell) return body;
|
|
311
|
+
|
|
312
|
+
const occupiedCallIds = new Set<string>();
|
|
313
|
+
for (const item of input) {
|
|
314
|
+
if (isPlainObject(item) && typeof item.call_id === "string") occupiedCallIds.add(item.call_id);
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
let changed = false;
|
|
318
|
+
const restored: unknown[] = [];
|
|
319
|
+
for (const item of input) {
|
|
320
|
+
if (isPlainObject(item) && item.type === "web_search_call" && typeof item.id === "string") {
|
|
321
|
+
const memo = peekBridgeSearchReplay(destinationScope, item.id);
|
|
322
|
+
if (memo && !occupiedCallIds.has(memo.callId)) {
|
|
323
|
+
changed = true;
|
|
324
|
+
occupiedCallIds.add(memo.callId);
|
|
325
|
+
restored.push({
|
|
326
|
+
type: "function_call",
|
|
327
|
+
...(memo.sourceItemId ? { id: memo.sourceItemId } : {}),
|
|
328
|
+
call_id: memo.callId,
|
|
329
|
+
name: memo.name,
|
|
330
|
+
// The bridge records the complete arguments text from the call's own done frame; the
|
|
331
|
+
// empty-object fallback matches what a continuation leg would have sent.
|
|
332
|
+
arguments: memo.argumentsText.length > 0 ? memo.argumentsText : "{}",
|
|
333
|
+
});
|
|
334
|
+
restored.push({ type: "function_call_output", call_id: memo.callId, output: memo.output });
|
|
335
|
+
continue;
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
restored.push(item);
|
|
339
|
+
}
|
|
340
|
+
return changed ? { ...body, input: restored } : body;
|
|
341
|
+
}
|
|
342
|
+
|
|
268
343
|
export function repairOrphanedInputItems(body: unknown, dropReasoning: boolean, synthesizeMissingCallOutputs = false): unknown {
|
|
269
344
|
if (!isPlainObject(body) || !Array.isArray(body.input)) return body;
|
|
270
345
|
const input = body.input;
|