@bitkyc08/opencodex 2.46.0 → 2.47.0-preview.20260908
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -0
- package/SPONSORS.md +105 -0
- package/gui/dist/assets/index-B8ZZ0ivI.css +1 -0
- package/gui/dist/assets/index-CNTWhC5F.js +115 -0
- package/gui/dist/index.html +2 -2
- package/package.json +2 -1
- package/src/adapters/cursor/protobuf-request.ts +34 -21
- package/src/adapters/cursor/tool-guidance.ts +2 -2
- package/src/adapters/cursor/tool-result-normalize.ts +22 -2
- package/src/adapters/exec-tool-result-normalize.ts +87 -0
- package/src/adapters/kiro.ts +29 -18
- package/src/adapters/responses-code-mode.ts +14 -4
- package/src/adapters/tool-catalog-nudge.ts +2 -2
- package/src/bridge.ts +40 -24
- package/src/claude/inbound.ts +21 -6
- package/src/claude/outbound.ts +43 -26
- package/src/cli/capabilities.ts +27 -0
- package/src/cli/integrations.ts +1 -1
- package/src/cli/models-runtime-subcommands.ts +2 -0
- package/src/cli/models-runtime.ts +108 -1
- package/src/cli/observe.ts +18 -3
- package/src/cli/usage-report.ts +6 -1
- package/src/clients/config-export.ts +5 -3
- package/src/codex/auth-api.ts +36 -0
- package/src/codex/catalog/provider-fetch.ts +2 -1
- package/src/codex/quota-auto-refresh-state.ts +8 -0
- package/src/codex/quota-auto-refresh.ts +157 -27
- package/src/codex/warmup.ts +5 -0
- package/src/config/provider-validation.ts +70 -1
- package/src/config.ts +73 -1
- package/src/generated/compatibility-version.json +67 -55
- package/src/lib/json-byte-size.ts +61 -0
- package/src/lib/provider-outbound.ts +11 -10
- package/src/oauth/index.ts +42 -8
- package/src/oauth/orcarouter.ts +200 -0
- package/src/providers/key-store.ts +22 -0
- package/src/providers/registry.ts +78 -29
- package/src/responses/citation-markers.ts +56 -24
- package/src/responses/reasoning-envelope.ts +53 -19
- package/src/server/auth-cors.ts +8 -0
- package/src/server/chat-completions.ts +7 -0
- package/src/server/chat-native.ts +65 -3
- package/src/server/claude-messages.ts +30 -10
- package/src/server/effort-policy.ts +183 -1
- package/src/server/management/agent-settings-routes.ts +50 -9
- package/src/server/management/config-routes.ts +2 -0
- package/src/server/management/logs-usage-routes.ts +11 -3
- package/src/server/management/model-routes.ts +65 -1
- package/src/server/management/model-rows.ts +5 -0
- package/src/server/management/provider-routes.ts +99 -6
- package/src/server/management/route-registry.ts +2 -0
- package/src/server/management/usage-aggregate-cache.ts +14 -7
- package/src/server/responses/compact.ts +3 -2
- package/src/server/responses/core.ts +16 -2
- package/src/server/startup-health-cache.ts +31 -9
- package/src/types/config.ts +5 -0
- package/src/types/provider.ts +4 -0
- package/src/usage/cost.ts +27 -31
- package/src/usage/summary.ts +52 -9
- package/src/usage/time-range.ts +48 -0
- package/src/usage/user-cost-overlays.ts +35 -3
- package/src/vision/anthropic-describe.ts +46 -2
- package/src/web-search/anthropic-executor.ts +49 -3
- package/src/web-search/parse.ts +1 -1
- package/gui/dist/assets/index-B72RPblC.js +0 -115
- package/gui/dist/assets/index-BFgUC17B.css +0 -1
|
@@ -42,23 +42,24 @@ export function hasCitationMarker(text: string): boolean {
|
|
|
42
42
|
*/
|
|
43
43
|
export function stripCitationMarkers(text: string): string {
|
|
44
44
|
if (!text.includes(CITATION_MARKER_START)) return text;
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
out +=
|
|
60
|
-
|
|
45
|
+
// Walk START-delimited segments exactly like the streaming filter below: a START whose
|
|
46
|
+
// own segment (up to the next START) contains an END within the span bound is a span and
|
|
47
|
+
// is removed; a START that is superseded by another START before any END, or whose span
|
|
48
|
+
// exceeds MAX_CITATION_SPAN_LENGTH, is malformed text and stays verbatim. Pairing an
|
|
49
|
+
// earlier malformed START with a later span's END would delete real answer text and,
|
|
50
|
+
// worse, disagree with what the streaming deltas already emitted (#3843). The bound is
|
|
51
|
+
// shared with the streaming filter for the same reason: a span it has already released
|
|
52
|
+
// as over-bound must not be swallowed here when the END finally arrives.
|
|
53
|
+
let start = text.indexOf(CITATION_MARKER_START);
|
|
54
|
+
let out = text.slice(0, start);
|
|
55
|
+
while (start !== -1) {
|
|
56
|
+
const nextStart = text.indexOf(CITATION_MARKER_START, start + 1);
|
|
57
|
+
const segment = text.slice(start, nextStart === -1 ? text.length : nextStart);
|
|
58
|
+
const end = segment.indexOf(CITATION_MARKER_END, 1);
|
|
59
|
+
out += end === -1 || end + 1 > MAX_CITATION_SPAN_LENGTH ? segment : segment.slice(end + 1);
|
|
60
|
+
start = nextStart;
|
|
61
61
|
}
|
|
62
|
+
return out;
|
|
62
63
|
}
|
|
63
64
|
|
|
64
65
|
export interface CitationMarkerFilter {
|
|
@@ -68,6 +69,18 @@ export interface CitationMarkerFilter {
|
|
|
68
69
|
flush(): string;
|
|
69
70
|
}
|
|
70
71
|
|
|
72
|
+
/**
|
|
73
|
+
* Upper bound on the length of a citation span (START through END inclusive), and therefore
|
|
74
|
+
* on the text the streaming filter withholds for one unterminated START.
|
|
75
|
+
*
|
|
76
|
+
* A real span is `cite` plus a few turn-scoped ids, so it is far under this. Without a
|
|
77
|
+
* bound, a backend that emits a START and never terminates it makes `held` grow for the
|
|
78
|
+
* whole response, and every later delta re-scans that accumulated prefix. The whole-string
|
|
79
|
+
* strip applies the same bound so both paths classify a span identically regardless of how
|
|
80
|
+
* the text was chunked.
|
|
81
|
+
*/
|
|
82
|
+
const MAX_CITATION_SPAN_LENGTH = 4_096;
|
|
83
|
+
|
|
71
84
|
/**
|
|
72
85
|
* Streaming filter.
|
|
73
86
|
*
|
|
@@ -75,6 +88,9 @@ export interface CitationMarkerFilter {
|
|
|
75
88
|
* next — so a stateless per-delta strip would emit the tail of a span it never recognized.
|
|
76
89
|
* This holds back the text from an unterminated START and releases it once the END arrives
|
|
77
90
|
* (removed) or the stream ends (verbatim, so nothing the model actually said is lost).
|
|
91
|
+
*
|
|
92
|
+
* A span that grows past `MAX_CITATION_SPAN_LENGTH` is malformed ordinary text, so
|
|
93
|
+
* it is released verbatim instead of withheld; a later START can still open a valid span.
|
|
78
94
|
*/
|
|
79
95
|
export function createCitationMarkerFilter(): CitationMarkerFilter {
|
|
80
96
|
// Text from an open START that has not been terminated yet.
|
|
@@ -83,13 +99,30 @@ export function createCitationMarkerFilter(): CitationMarkerFilter {
|
|
|
83
99
|
push(delta: string): string {
|
|
84
100
|
const combined = held + delta;
|
|
85
101
|
held = "";
|
|
86
|
-
|
|
87
|
-
if (start === -1) return
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
//
|
|
91
|
-
|
|
92
|
-
|
|
102
|
+
let start = combined.indexOf(CITATION_MARKER_START);
|
|
103
|
+
if (start === -1) return combined;
|
|
104
|
+
let out = combined.slice(0, start);
|
|
105
|
+
// Walk START-delimited segments independently so an earlier malformed START is never
|
|
106
|
+
// paired with a later span's END (the whole-string strip would do exactly that).
|
|
107
|
+
while (start !== -1) {
|
|
108
|
+
const nextStart = combined.indexOf(CITATION_MARKER_START, start + 1);
|
|
109
|
+
const segment = combined.slice(start, nextStart === -1 ? combined.length : nextStart);
|
|
110
|
+
const end = segment.indexOf(CITATION_MARKER_END, 1);
|
|
111
|
+
if (end !== -1 && end + 1 <= MAX_CITATION_SPAN_LENGTH) {
|
|
112
|
+
// A complete span: drop it, keep whatever trails it inside this segment.
|
|
113
|
+
out += segment.slice(end + 1);
|
|
114
|
+
} else if (end === -1 && nextStart === -1 && segment.length <= MAX_CITATION_SPAN_LENGTH) {
|
|
115
|
+
// Only a bounded trailing span can still be completed by a later delta.
|
|
116
|
+
held = segment;
|
|
117
|
+
} else {
|
|
118
|
+
// Superseded by a later START, or over the bound (with or without a late END):
|
|
119
|
+
// ordinary text, emitted verbatim so neither the retained text nor the per-delta
|
|
120
|
+
// rescan grows without limit.
|
|
121
|
+
out += segment;
|
|
122
|
+
}
|
|
123
|
+
start = nextStart;
|
|
124
|
+
}
|
|
125
|
+
return out;
|
|
93
126
|
},
|
|
94
127
|
flush(): string {
|
|
95
128
|
const rest = held;
|
|
@@ -98,4 +131,3 @@ export function createCitationMarkerFilter(): CitationMarkerFilter {
|
|
|
98
131
|
},
|
|
99
132
|
};
|
|
100
133
|
}
|
|
101
|
-
|
|
@@ -12,6 +12,9 @@
|
|
|
12
12
|
* passthrough scrub strips ocxr1 envelopes before native forwarding.
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
|
+
import { createTranslatorBudget, type TranslatorBudget } from "../lib/translator-budget";
|
|
16
|
+
import { jsonUtf8Bytes } from "../lib/json-byte-size";
|
|
17
|
+
|
|
15
18
|
export const OCX_REASONING_PREFIX = "ocxr1:";
|
|
16
19
|
|
|
17
20
|
export interface ReasoningEnvelope {
|
|
@@ -32,30 +35,61 @@ export interface ReasoningEnvelope {
|
|
|
32
35
|
krc?: string;
|
|
33
36
|
}
|
|
34
37
|
|
|
35
|
-
export function encodeReasoningEnvelope(envelope: ReasoningEnvelope): string {
|
|
36
|
-
|
|
38
|
+
export function encodeReasoningEnvelope(envelope: ReasoningEnvelope, budget?: TranslatorBudget): string {
|
|
39
|
+
const activeBudget = budget ?? createTranslatorBudget();
|
|
40
|
+
try {
|
|
41
|
+
const jsonBytes = jsonUtf8Bytes(envelope);
|
|
42
|
+
const base64Bytes = 4 * Math.ceil(jsonBytes / 3);
|
|
43
|
+
// Reserve before materialization: UTF-16 JSON, UTF-8 buffer, base64 string,
|
|
44
|
+
// and the prefixed result may coexist. Returned-value ownership stays with
|
|
45
|
+
// callers, whose existing retained accounting must not be charged twice here.
|
|
46
|
+
const reservation = activeBudget.reserveTransient(
|
|
47
|
+
Math.max(
|
|
48
|
+
3 * jsonBytes + 4 * base64Bytes + 2 * OCX_REASONING_PREFIX.length,
|
|
49
|
+
8 * (OCX_REASONING_PREFIX.length + base64Bytes),
|
|
50
|
+
),
|
|
51
|
+
{ kind: "reasoning" },
|
|
52
|
+
);
|
|
53
|
+
try {
|
|
54
|
+
return OCX_REASONING_PREFIX + Buffer.from(JSON.stringify(envelope), "utf-8").toString("base64");
|
|
55
|
+
} finally {
|
|
56
|
+
reservation.release();
|
|
57
|
+
}
|
|
58
|
+
} finally {
|
|
59
|
+
if (!budget) activeBudget.dispose();
|
|
60
|
+
}
|
|
37
61
|
}
|
|
38
62
|
|
|
39
63
|
/** Decode an ocxr1 envelope; returns null for native (OpenAI-encrypted) blobs or garbage. */
|
|
40
|
-
export function decodeReasoningEnvelope(encryptedContent: string): ReasoningEnvelope | null {
|
|
64
|
+
export function decodeReasoningEnvelope(encryptedContent: string, budget?: TranslatorBudget): ReasoningEnvelope | null {
|
|
41
65
|
if (!encryptedContent.startsWith(OCX_REASONING_PREFIX)) return null;
|
|
66
|
+
const activeBudget = budget ?? createTranslatorBudget();
|
|
42
67
|
try {
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
const
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
const
|
|
50
|
-
|
|
68
|
+
// Also bound already-encoded replay before slicing, decoding, or parsing it.
|
|
69
|
+
// Eight bytes per code unit conservatively covers the string/buffer copies.
|
|
70
|
+
const reservation = activeBudget.reserveTransient(8 * encryptedContent.length, { kind: "reasoning" });
|
|
71
|
+
try {
|
|
72
|
+
const parsed: unknown = JSON.parse(Buffer.from(encryptedContent.slice(OCX_REASONING_PREFIX.length), "base64").toString("utf-8"));
|
|
73
|
+
if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return null;
|
|
74
|
+
const obj = parsed as { sig?: unknown; red?: unknown };
|
|
75
|
+
const envelope: ReasoningEnvelope = {};
|
|
76
|
+
if (typeof obj.sig === "string") envelope.sig = obj.sig;
|
|
77
|
+
if (Array.isArray(obj.red)) {
|
|
78
|
+
const red = obj.red.filter((r): r is string => typeof r === "string");
|
|
79
|
+
if (red.length > 0) envelope.red = red;
|
|
80
|
+
}
|
|
81
|
+
const txt = (parsed as { txt?: unknown }).txt;
|
|
82
|
+
const hasTxt = typeof txt === "string";
|
|
83
|
+
if (hasTxt) envelope.txt = txt;
|
|
84
|
+
const krc = (parsed as { krc?: unknown }).krc;
|
|
85
|
+
if (typeof krc === "string" && krc.length > 0) envelope.krc = krc;
|
|
86
|
+
return envelope.sig || envelope.red || hasTxt || envelope.krc ? envelope : null;
|
|
87
|
+
} catch {
|
|
88
|
+
return null;
|
|
89
|
+
} finally {
|
|
90
|
+
reservation.release();
|
|
51
91
|
}
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
if (hasTxt) envelope.txt = txt;
|
|
55
|
-
const krc = (parsed as { krc?: unknown }).krc;
|
|
56
|
-
if (typeof krc === "string" && krc.length > 0) envelope.krc = krc;
|
|
57
|
-
return envelope.sig || envelope.red || hasTxt || envelope.krc ? envelope : null;
|
|
58
|
-
} catch {
|
|
59
|
-
return null;
|
|
92
|
+
} finally {
|
|
93
|
+
if (!budget) activeBudget.dispose();
|
|
60
94
|
}
|
|
61
95
|
}
|
package/src/server/auth-cors.ts
CHANGED
|
@@ -13,6 +13,7 @@ import {
|
|
|
13
13
|
import {
|
|
14
14
|
apiKeyTransportConfigError,
|
|
15
15
|
booleanRecordConfigError,
|
|
16
|
+
providerReasoningPinsConfigError,
|
|
16
17
|
modelAdapterRecordConfigError,
|
|
17
18
|
nonBlankStringArrayConfigError,
|
|
18
19
|
positiveIntegerConfigError,
|
|
@@ -581,6 +582,8 @@ export function providerManagementConfigError(name: unknown, provider: unknown):
|
|
|
581
582
|
return "provider must be a plain object";
|
|
582
583
|
}
|
|
583
584
|
const raw = provider as Record<string, unknown>;
|
|
585
|
+
const pinsError = providerReasoningPinsConfigError(raw);
|
|
586
|
+
if (pinsError) return pinsError;
|
|
584
587
|
for (const field of FORBIDDEN_PROVIDER_RUNTIME_FIELDS) {
|
|
585
588
|
if (Object.hasOwn(raw, field)) return `provider ${name} must not include runtime field "${field}"`;
|
|
586
589
|
}
|
|
@@ -594,6 +597,9 @@ export function providerManagementConfigError(name: unknown, provider: unknown):
|
|
|
594
597
|
}
|
|
595
598
|
if (seed) seed.codexAccountMode = raw.codexAccountMode;
|
|
596
599
|
const canonicalCandidate = { ...raw };
|
|
600
|
+
// Validated operator overlays do not change the canonical auth/transport seed.
|
|
601
|
+
delete canonicalCandidate.pinnedReasoningEffort;
|
|
602
|
+
delete canonicalCandidate.modelPinnedReasoningEfforts;
|
|
597
603
|
delete canonicalCandidate.responsesSnapshotRepair;
|
|
598
604
|
// modelCosts is a user-owned display overlay, not part of the canonical
|
|
599
605
|
// forward seed; it is validated separately below (providerModelCostsConfigError).
|
|
@@ -829,6 +835,8 @@ const PROVIDER_CONFIG_FIELD_POLICY = {
|
|
|
829
835
|
reasoningEfforts: "editor",
|
|
830
836
|
modelReasoningEfforts: "editor",
|
|
831
837
|
modelDefaultReasoningEfforts: "editor",
|
|
838
|
+
pinnedReasoningEffort: "editor",
|
|
839
|
+
modelPinnedReasoningEfforts: "editor",
|
|
832
840
|
modelSupportsReasoningSummaries: "editor",
|
|
833
841
|
modelSupportsVerbosity: "editor",
|
|
834
842
|
supportsVerbosity: "editor",
|
|
@@ -25,6 +25,8 @@ import { estimateTokens } from "../lib/token-estimate";
|
|
|
25
25
|
import { NoEligiblePolicyCandidateError, UnknownRoutingPolicyError, routeModel } from "../router";
|
|
26
26
|
import { evidenceFromBody } from "../routing/request-evidence";
|
|
27
27
|
import { resolveWireProtocolOverride } from "./adapter-resolve";
|
|
28
|
+
import { resolveOpenCodeGoTransport } from "../providers/opencode-go-transport";
|
|
29
|
+
import { normalizeLogConversationId, sessionLaneIdFromRequest } from "./request-log-conversation";
|
|
28
30
|
import type { OcxConfig } from "../types";
|
|
29
31
|
import { readJsonRequestBody } from "./request-decompress";
|
|
30
32
|
import {
|
|
@@ -136,6 +138,8 @@ async function handleChatCompletionsWithBudget(
|
|
|
136
138
|
let chatNativeRoute: ReturnType<typeof routeModel> | null = null;
|
|
137
139
|
try {
|
|
138
140
|
const route = routeModel(config, chatBody.model as string, evidenceFromBody(chatBody));
|
|
141
|
+
route.provider = resolveOpenCodeGoTransport(route.provider,
|
|
142
|
+
sessionLaneIdFromRequest(req.headers) ?? normalizeLogConversationId(req.headers.get("x-opencode-session")));
|
|
139
143
|
// Settle the wire once so every branch below reads the adapter this model will
|
|
140
144
|
// actually use, not the provider-wide default (#404).
|
|
141
145
|
route.provider = resolveWireProtocolOverride(route.providerName, route.modelId, route.provider, "chat");
|
|
@@ -237,6 +241,9 @@ async function handleChatCompletionsWithBudget(
|
|
|
237
241
|
return chatCompletionsErrorResponse(400, CODEX_RESERVE_HELPER_UNSUPPORTED_MESSAGE, "invalid_request_error");
|
|
238
242
|
}
|
|
239
243
|
const headers = new Headers({ "content-type": "application/json" });
|
|
244
|
+
// Internal bridge metadata; the Go resolver scopes and hashes it before upstream use.
|
|
245
|
+
const openCodeSession = req.headers.get("x-opencode-session");
|
|
246
|
+
if (openCodeSession) headers.set("x-opencode-session", openCodeSession);
|
|
240
247
|
for (const name of FORWARD_HEADERS) {
|
|
241
248
|
if (name === "authorization" && !directRoute) continue;
|
|
242
249
|
const value = req.headers.get(name);
|
|
@@ -6,6 +6,8 @@ import {
|
|
|
6
6
|
collectChatCompletion,
|
|
7
7
|
isChatCompletionsStreamError,
|
|
8
8
|
} from "../chat/outbound";
|
|
9
|
+
import { applyChatEffortCap, chatCollabSurface, effortCapAppliesTo, resolvePinnedEffort, supportedLadderFor } from "./effort-policy";
|
|
10
|
+
import { mapReasoningEffort } from "../reasoning-effort";
|
|
9
11
|
import {
|
|
10
12
|
classifyError,
|
|
11
13
|
cyberPolicyErrorType,
|
|
@@ -60,6 +62,68 @@ type Rec = Record<string, unknown>;
|
|
|
60
62
|
const MAX_NATIVE_CHAT_JSON_BYTES = 32 * 1024 * 1024;
|
|
61
63
|
const MAX_NATIVE_CHAT_ERROR_BYTES = 64 * 1024;
|
|
62
64
|
|
|
65
|
+
const chatEffortSnapshots = new WeakMap<Rec, {
|
|
66
|
+
inputModel: string;
|
|
67
|
+
providerName: string;
|
|
68
|
+
modelId: string;
|
|
69
|
+
present: boolean;
|
|
70
|
+
value: unknown;
|
|
71
|
+
annotation: string | undefined;
|
|
72
|
+
}>();
|
|
73
|
+
|
|
74
|
+
function normalizePinnedChatEffort(options: HandleNativeChatOptions): void {
|
|
75
|
+
const { chatBody, route, config, req, logCtx, requestedModel } = options;
|
|
76
|
+
let snapshot = chatEffortSnapshots.get(chatBody);
|
|
77
|
+
const inputModel = typeof chatBody.model === "string" ? chatBody.model : requestedModel;
|
|
78
|
+
let selector = inputModel;
|
|
79
|
+
if (snapshot) {
|
|
80
|
+
if (snapshot.providerName === route.providerName && snapshot.modelId === route.modelId) {
|
|
81
|
+
logCtx.requestedEffort = snapshot.annotation;
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
84
|
+
if (snapshot.present) chatBody.reasoning_effort = snapshot.value;
|
|
85
|
+
else delete chatBody.reasoning_effort;
|
|
86
|
+
if (selector === snapshot.inputModel || selector === snapshot.modelId) {
|
|
87
|
+
selector = `${route.providerName}/${route.modelId}`;
|
|
88
|
+
}
|
|
89
|
+
} else {
|
|
90
|
+
snapshot = {
|
|
91
|
+
inputModel,
|
|
92
|
+
providerName: route.providerName,
|
|
93
|
+
modelId: route.modelId,
|
|
94
|
+
present: Object.hasOwn(chatBody, "reasoning_effort"),
|
|
95
|
+
value: chatBody.reasoning_effort,
|
|
96
|
+
annotation: undefined,
|
|
97
|
+
};
|
|
98
|
+
chatEffortSnapshots.set(chatBody, snapshot);
|
|
99
|
+
}
|
|
100
|
+
snapshot.inputModel = inputModel;
|
|
101
|
+
snapshot.providerName = route.providerName;
|
|
102
|
+
snapshot.modelId = route.modelId;
|
|
103
|
+
const from = typeof chatBody.reasoning_effort === "string" ? chatBody.reasoning_effort : undefined;
|
|
104
|
+
logCtx.requestedEffort = from;
|
|
105
|
+
// Compaction is normally excluded by native-route eligibility; preserve that boundary here too.
|
|
106
|
+
const pinned = chatBody.compaction_trigger === undefined
|
|
107
|
+
? resolvePinnedEffort(route, selector, config)
|
|
108
|
+
: undefined;
|
|
109
|
+
if (pinned !== undefined) {
|
|
110
|
+
logCtx.requestedEffort = from ? `${from}->${pinned}` : pinned;
|
|
111
|
+
if (pinned === "none") delete chatBody.reasoning_effort;
|
|
112
|
+
else chatBody.reasoning_effort = pinned;
|
|
113
|
+
// The native lane historically passes caller effort through, including with caps set.
|
|
114
|
+
// Only a newly operator-pinned value enters the cap and provider-mapping pipeline.
|
|
115
|
+
if (effortCapAppliesTo(chatCollabSurface(chatBody), req.headers, config)) {
|
|
116
|
+
const capped = applyChatEffortCap(chatBody, req.headers, config, supportedLadderFor(route));
|
|
117
|
+
if (capped) logCtx.requestedEffort = `${logCtx.requestedEffort}->${capped.to}`;
|
|
118
|
+
}
|
|
119
|
+
const effort = typeof chatBody.reasoning_effort === "string" ? chatBody.reasoning_effort : undefined;
|
|
120
|
+
const wireEffort = mapReasoningEffort(route.provider, route.modelId, effort);
|
|
121
|
+
if (wireEffort === undefined) delete chatBody.reasoning_effort;
|
|
122
|
+
else chatBody.reasoning_effort = wireEffort;
|
|
123
|
+
}
|
|
124
|
+
snapshot.annotation = logCtx.requestedEffort;
|
|
125
|
+
}
|
|
126
|
+
|
|
63
127
|
function isRec(value: unknown): value is Rec {
|
|
64
128
|
return value !== null && typeof value === "object" && !Array.isArray(value);
|
|
65
129
|
}
|
|
@@ -147,9 +211,7 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
|
|
|
147
211
|
return chatCompletionsErrorResponse(status, safeMessage, type, code);
|
|
148
212
|
};
|
|
149
213
|
|
|
150
|
-
|
|
151
|
-
? options.chatBody.reasoning_effort
|
|
152
|
-
: undefined;
|
|
214
|
+
normalizePinnedChatEffort(options);
|
|
153
215
|
logCtx.requestedServiceTier = typeof options.chatBody.service_tier === "string"
|
|
154
216
|
? options.chatBody.service_tier
|
|
155
217
|
: undefined;
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
* unchanged. The Responses output (SSE or JSON) is converted back to Anthropic shape.
|
|
8
8
|
*/
|
|
9
9
|
import { FORWARD_HEADERS } from "../adapters/openai-responses";
|
|
10
|
+
import { jsonUtf8Bytes } from "../lib/json-byte-size";
|
|
10
11
|
import { sseFieldValue } from "../lib/sse-decoder";
|
|
11
12
|
import { enforceAnthropicImageLimits, sniffImageDimensions } from "../adapters/anthropic-image-guard";
|
|
12
13
|
import { normalizeAnthropicImages } from "../adapters/anthropic-image-normalize";
|
|
@@ -753,13 +754,13 @@ async function handleClaudeMessagesWithBudget(
|
|
|
753
754
|
};
|
|
754
755
|
delete anthropicBody.thinking;
|
|
755
756
|
}
|
|
756
|
-
const translation = anthropicToResponsesTranslation(anthropicBody, config.claudeCode);
|
|
757
|
+
const translation = anthropicToResponsesTranslation(anthropicBody, config.claudeCode, translatorBudget);
|
|
757
758
|
internalBody = translation.body;
|
|
758
759
|
// The Anthropic translator builds its body from model/input/store/stream plus sampling
|
|
759
760
|
// fields only, so the caller intent is applied to the TRANSLATED body rather than the
|
|
760
761
|
// inbound one.
|
|
761
762
|
if (fastRow) internalBody.service_tier = "priority";
|
|
762
|
-
translatorBudget.chargeRetained(
|
|
763
|
+
translatorBudget.chargeRetained(jsonUtf8Bytes(internalBody), { kind: "request_copies" });
|
|
763
764
|
cacheKeySource = translation.cacheKeySource;
|
|
764
765
|
} catch (err) {
|
|
765
766
|
const overflow = isTranslatorBudgetExceededError(err);
|
|
@@ -862,13 +863,26 @@ async function handleClaudeMessagesWithBudget(
|
|
|
862
863
|
headers.set("session_id", uuidFromHex(internalBody.prompt_cache_key));
|
|
863
864
|
}
|
|
864
865
|
}
|
|
865
|
-
|
|
866
|
-
|
|
867
|
-
|
|
868
|
-
|
|
869
|
-
|
|
870
|
-
|
|
871
|
-
|
|
866
|
+
let internalReq: Request;
|
|
867
|
+
try {
|
|
868
|
+
// The UTF-16 JSON string and the Request's UTF-8 body coexist until dispatch.
|
|
869
|
+
const bodyBytes = jsonUtf8Bytes(internalBody);
|
|
870
|
+
const reservation = translatorBudget.reserveTransient(3 * bodyBytes, { kind: "request_copies" });
|
|
871
|
+
try {
|
|
872
|
+
internalReq = new Request("http://localhost/v1/responses", {
|
|
873
|
+
method: "POST",
|
|
874
|
+
headers,
|
|
875
|
+
body: JSON.stringify(internalBody),
|
|
876
|
+
});
|
|
877
|
+
} finally {
|
|
878
|
+
reservation.release();
|
|
879
|
+
}
|
|
880
|
+
translatorBudget.chargeRetained(bodyBytes, { kind: "request_copies" });
|
|
881
|
+
} catch (err) {
|
|
882
|
+
if (!isTranslatorBudgetExceededError(err)) throw err;
|
|
883
|
+
if (logIds) addFinalRequestLog(logIds.requestId, logIds.start, logCtx, 413, { closeReason: "non_stream" });
|
|
884
|
+
return anthropicErrorResponse(413, "request translation buffer exceeded the safe limit", "request_too_large", "translation_buffer_limit");
|
|
885
|
+
}
|
|
872
886
|
|
|
873
887
|
// Request-log wiring mirrors the /v1/responses route: native passthrough finalizes
|
|
874
888
|
// via the terminal callbacks; routed streams get the Responses-vocabulary log tap
|
|
@@ -1006,7 +1020,13 @@ async function handleClaudeMessagesWithBudget(
|
|
|
1006
1020
|
}
|
|
1007
1021
|
return anthropicErrorResponse(502, error?.message ?? "upstream request failed", "api_error");
|
|
1008
1022
|
}
|
|
1009
|
-
|
|
1023
|
+
let message: Rec;
|
|
1024
|
+
try {
|
|
1025
|
+
message = responsesJsonToAnthropicMessage(json, requestedModel, translatorBudget);
|
|
1026
|
+
} catch (err) {
|
|
1027
|
+
if (!isTranslatorBudgetExceededError(err)) throw err;
|
|
1028
|
+
return anthropicErrorResponse(413, "upstream translation buffer exceeded the safe limit", "request_too_large", "translation_buffer_limit");
|
|
1029
|
+
}
|
|
1010
1030
|
if ((message as Rec).type === "error") {
|
|
1011
1031
|
return new Response(JSON.stringify(message), {
|
|
1012
1032
|
status: 529,
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
*/
|
|
15
15
|
import type { OcxConfig, OcxParsedRequest, OcxProviderConfig } from "../types";
|
|
16
16
|
import { modelInList } from "../types";
|
|
17
|
-
import { codexEffortRank, configuredReasoningEfforts, isCodexReasoningEffort, modelRecordValue } from "../reasoning-effort";
|
|
17
|
+
import { codexEffortRank, configuredReasoningEfforts, isCodexReasoningEffort, isDeclaredReasoningEffort, modelRecordValue } from "../reasoning-effort";
|
|
18
18
|
import { catalogModelEfforts } from "../codex/catalog";
|
|
19
19
|
|
|
20
20
|
/**
|
|
@@ -188,3 +188,185 @@ export function applyEffortCap(
|
|
|
188
188
|
if (raw?.reasoning && typeof raw.reasoning === "object") raw.reasoning.effort = resolved;
|
|
189
189
|
return { from: requested, to: resolved, subagent };
|
|
190
190
|
}
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Resolve any pinned reasoning effort configured for this model or provider.
|
|
194
|
+
* Priority order:
|
|
195
|
+
* 1. Provider model-specific pinned effort (`provider.modelPinnedReasoningEfforts[modelId]`)
|
|
196
|
+
* 2. Provider-wide pinned effort (`provider.pinnedReasoningEffort`)
|
|
197
|
+
* 3. Global config model-specific pinned effort (`config.modelPinnedEfforts[modelId]`)
|
|
198
|
+
* Global keys try the final pre-namespace selector, provider-qualified destination,
|
|
199
|
+
* then bare destination, using modelRecordValue's exact/family/case-fold semantics.
|
|
200
|
+
* The caller removes synthetic effort rows and combo selectors before this boundary.
|
|
201
|
+
*
|
|
202
|
+
* Returns undefined when no valid pinned effort tier is configured.
|
|
203
|
+
*/
|
|
204
|
+
export function resolvePinnedEffort(
|
|
205
|
+
route: { provider: OcxProviderConfig; modelId: string; providerName?: string },
|
|
206
|
+
parsedModelId?: string,
|
|
207
|
+
config?: OcxConfig,
|
|
208
|
+
): string | undefined {
|
|
209
|
+
const prov = route.provider;
|
|
210
|
+
const rawProvModel = modelRecordValue(prov.modelPinnedReasoningEfforts, route.modelId)
|
|
211
|
+
?? (parsedModelId ? modelRecordValue(prov.modelPinnedReasoningEfforts, parsedModelId) : undefined);
|
|
212
|
+
if (rawProvModel && isDeclaredReasoningEffort(rawProvModel)) {
|
|
213
|
+
return rawProvModel;
|
|
214
|
+
}
|
|
215
|
+
if (prov.pinnedReasoningEffort && isDeclaredReasoningEffort(prov.pinnedReasoningEffort)) {
|
|
216
|
+
return prov.pinnedReasoningEffort;
|
|
217
|
+
}
|
|
218
|
+
if (config?.modelPinnedEfforts) {
|
|
219
|
+
const rawGlobal = (parsedModelId ? modelRecordValue(config.modelPinnedEfforts, parsedModelId) : undefined)
|
|
220
|
+
?? (route.providerName ? modelRecordValue(config.modelPinnedEfforts, `${route.providerName}/${route.modelId}`) : undefined)
|
|
221
|
+
?? modelRecordValue(config.modelPinnedEfforts, route.modelId);
|
|
222
|
+
if (rawGlobal && isDeclaredReasoningEffort(rawGlobal)) {
|
|
223
|
+
return rawGlobal;
|
|
224
|
+
}
|
|
225
|
+
}
|
|
226
|
+
return undefined;
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
interface EffortSnapshot {
|
|
230
|
+
selector: string;
|
|
231
|
+
providerName: string;
|
|
232
|
+
modelId: string;
|
|
233
|
+
reasoningPresent: boolean;
|
|
234
|
+
reasoning: OcxParsedRequest["options"]["reasoning"];
|
|
235
|
+
rawEffortPresent: boolean;
|
|
236
|
+
rawEffort: unknown;
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
const effortSnapshots = new WeakMap<OcxParsedRequest, EffortSnapshot>();
|
|
240
|
+
|
|
241
|
+
/** Capture effective synthetic/combo defaults before final model namespace rewriting.
|
|
242
|
+
* A different destination restores effort alone; intervening summary/options edits survive.
|
|
243
|
+
* Credential retries do not change the destination and retain their existing decision.
|
|
244
|
+
*/
|
|
245
|
+
export function prepareEffortNormalization(
|
|
246
|
+
parsed: OcxParsedRequest,
|
|
247
|
+
route: { providerName: string; modelId: string },
|
|
248
|
+
): string {
|
|
249
|
+
const raw = parsed._rawBody as { reasoning?: Record<string, unknown> } | undefined;
|
|
250
|
+
const previous = effortSnapshots.get(parsed);
|
|
251
|
+
if (!previous) {
|
|
252
|
+
effortSnapshots.set(parsed, {
|
|
253
|
+
selector: parsed.modelId,
|
|
254
|
+
providerName: route.providerName,
|
|
255
|
+
modelId: route.modelId,
|
|
256
|
+
reasoningPresent: Object.hasOwn(parsed.options, "reasoning"),
|
|
257
|
+
reasoning: parsed.options.reasoning,
|
|
258
|
+
rawEffortPresent: !!raw?.reasoning && Object.hasOwn(raw.reasoning, "effort"),
|
|
259
|
+
rawEffort: raw?.reasoning?.effort,
|
|
260
|
+
});
|
|
261
|
+
return parsed.modelId;
|
|
262
|
+
}
|
|
263
|
+
if (previous.providerName === route.providerName && previous.modelId === route.modelId) {
|
|
264
|
+
return previous.selector;
|
|
265
|
+
}
|
|
266
|
+
if (previous.reasoningPresent) parsed.options.reasoning = previous.reasoning;
|
|
267
|
+
else delete parsed.options.reasoning;
|
|
268
|
+
if (raw && previous.rawEffortPresent) {
|
|
269
|
+
if (!raw.reasoning || typeof raw.reasoning !== "object") raw.reasoning = {};
|
|
270
|
+
raw.reasoning.effort = previous.rawEffort;
|
|
271
|
+
} else if (raw?.reasoning && typeof raw.reasoning === "object") {
|
|
272
|
+
delete raw.reasoning.effort;
|
|
273
|
+
}
|
|
274
|
+
// An unchanged wire model is the previous destination, not a new requested alias.
|
|
275
|
+
previous.selector = parsed.modelId === previous.modelId || parsed.modelId === previous.selector
|
|
276
|
+
? `${route.providerName}/${route.modelId}`
|
|
277
|
+
: parsed.modelId;
|
|
278
|
+
previous.providerName = route.providerName;
|
|
279
|
+
previous.modelId = route.modelId;
|
|
280
|
+
return previous.selector;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* Detect collaboration surface for a native chat request body.
|
|
285
|
+
* Mirrors Responses collabSurface behavior across function and custom tool representations.
|
|
286
|
+
*/
|
|
287
|
+
export function chatCollabSurface(chatBody: Record<string, unknown>): "v1" | "v2" | null {
|
|
288
|
+
if (!Array.isArray(chatBody.tools)) return null;
|
|
289
|
+
let namespacedSpawn = false;
|
|
290
|
+
let flatSpawn = false;
|
|
291
|
+
let v1Only = false;
|
|
292
|
+
let v2Only = false;
|
|
293
|
+
for (const raw of chatBody.tools) {
|
|
294
|
+
if (!raw || typeof raw !== "object") continue;
|
|
295
|
+
const tool = raw as Record<string, unknown>;
|
|
296
|
+
let name = "";
|
|
297
|
+
let namespace: string | undefined = undefined;
|
|
298
|
+
if (tool.type === "function" && tool.function && typeof tool.function === "object") {
|
|
299
|
+
const fn = tool.function as Record<string, unknown>;
|
|
300
|
+
name = typeof fn.name === "string" ? fn.name : "";
|
|
301
|
+
} else if (tool.type === "custom" && tool.custom && typeof tool.custom === "object") {
|
|
302
|
+
const cust = tool.custom as Record<string, unknown>;
|
|
303
|
+
name = typeof cust.name === "string" ? cust.name : "";
|
|
304
|
+
} else if (typeof tool.name === "string") {
|
|
305
|
+
name = tool.name;
|
|
306
|
+
}
|
|
307
|
+
if (typeof tool.namespace === "string") namespace = tool.namespace;
|
|
308
|
+
if (name === "spawn_agent") {
|
|
309
|
+
if (namespace) namespacedSpawn = true;
|
|
310
|
+
else flatSpawn = true;
|
|
311
|
+
} else if (name === "send_input" || name === "resume_agent" || name === "close_agent") {
|
|
312
|
+
v1Only = true;
|
|
313
|
+
} else if (name === "send_message" || name === "followup_task" || name === "interrupt_agent" || name === "list_agents") {
|
|
314
|
+
v2Only = true;
|
|
315
|
+
}
|
|
316
|
+
}
|
|
317
|
+
if (!namespacedSpawn && !flatSpawn) return null;
|
|
318
|
+
if (namespacedSpawn && flatSpawn) return null;
|
|
319
|
+
if (v1Only && v2Only) return null;
|
|
320
|
+
if (v1Only) return "v1";
|
|
321
|
+
if (v2Only) return "v2";
|
|
322
|
+
return namespacedSpawn ? "v1" : "v2";
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/**
|
|
326
|
+
* Apply effortCap to a native chat completions body when admitted by the collaboration gate.
|
|
327
|
+
*/
|
|
328
|
+
export function applyChatEffortCap(
|
|
329
|
+
chatBody: Record<string, unknown>,
|
|
330
|
+
headers: Headers,
|
|
331
|
+
config: OcxConfig,
|
|
332
|
+
supported?: readonly string[] | undefined,
|
|
333
|
+
): { from: string; to: string; subagent: boolean } | null {
|
|
334
|
+
const subagent = isThreadSpawnRequest(headers);
|
|
335
|
+
const cap = effortCapFor(config, subagent);
|
|
336
|
+
if (!cap) return null;
|
|
337
|
+
const resolved = resolveCappedEffort(cap, supported);
|
|
338
|
+
const requested = typeof chatBody.reasoning_effort === "string" ? chatBody.reasoning_effort : undefined;
|
|
339
|
+
if (resolved === null) {
|
|
340
|
+
if (!requested) return null;
|
|
341
|
+
delete chatBody.reasoning_effort;
|
|
342
|
+
return { from: requested, to: "none", subagent };
|
|
343
|
+
}
|
|
344
|
+
if (!requested || !isCodexReasoningEffort(requested)) return null;
|
|
345
|
+
if (codexEffortRank(requested) <= codexEffortRank(resolved)) return null;
|
|
346
|
+
chatBody.reasoning_effort = resolved;
|
|
347
|
+
return { from: requested, to: resolved, subagent };
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
export function applyPinnedEffort(
|
|
351
|
+
parsed: OcxParsedRequest,
|
|
352
|
+
route: { provider: OcxProviderConfig; modelId: string; providerName?: string },
|
|
353
|
+
config?: OcxConfig,
|
|
354
|
+
selector = effortSnapshots.get(parsed)?.selector ?? parsed.modelId,
|
|
355
|
+
): { from: string | undefined; to: string } | null {
|
|
356
|
+
if (parsed._compactionRequest === true) return null;
|
|
357
|
+
const pinned = resolvePinnedEffort(route, selector, config);
|
|
358
|
+
if (!pinned) return null;
|
|
359
|
+
const requested = parsed.options.reasoning;
|
|
360
|
+
const raw = parsed._rawBody as { reasoning?: { effort?: string } } | undefined;
|
|
361
|
+
const targetEffort = pinned === "none" ? undefined : pinned;
|
|
362
|
+
parsed.options.reasoning = targetEffort;
|
|
363
|
+
if (targetEffort) {
|
|
364
|
+
if (raw && typeof raw === "object") {
|
|
365
|
+
if (!raw.reasoning || typeof raw.reasoning !== "object") raw.reasoning = {};
|
|
366
|
+
raw.reasoning.effort = targetEffort;
|
|
367
|
+
}
|
|
368
|
+
} else if (raw?.reasoning && typeof raw.reasoning === "object") {
|
|
369
|
+
delete raw.reasoning.effort;
|
|
370
|
+
}
|
|
371
|
+
return { from: requested, to: pinned };
|
|
372
|
+
}
|