@bitkyc08/opencodex 2.46.0 → 2.47.0-preview.20260908

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/README.md +22 -0
  2. package/SPONSORS.md +105 -0
  3. package/gui/dist/assets/index-B8ZZ0ivI.css +1 -0
  4. package/gui/dist/assets/index-CNTWhC5F.js +115 -0
  5. package/gui/dist/index.html +2 -2
  6. package/package.json +2 -1
  7. package/src/adapters/cursor/protobuf-request.ts +34 -21
  8. package/src/adapters/cursor/tool-guidance.ts +2 -2
  9. package/src/adapters/cursor/tool-result-normalize.ts +22 -2
  10. package/src/adapters/exec-tool-result-normalize.ts +87 -0
  11. package/src/adapters/kiro.ts +29 -18
  12. package/src/adapters/responses-code-mode.ts +14 -4
  13. package/src/adapters/tool-catalog-nudge.ts +2 -2
  14. package/src/bridge.ts +40 -24
  15. package/src/claude/inbound.ts +21 -6
  16. package/src/claude/outbound.ts +43 -26
  17. package/src/cli/capabilities.ts +27 -0
  18. package/src/cli/integrations.ts +1 -1
  19. package/src/cli/models-runtime-subcommands.ts +2 -0
  20. package/src/cli/models-runtime.ts +108 -1
  21. package/src/cli/observe.ts +18 -3
  22. package/src/cli/usage-report.ts +6 -1
  23. package/src/clients/config-export.ts +5 -3
  24. package/src/codex/auth-api.ts +36 -0
  25. package/src/codex/catalog/provider-fetch.ts +2 -1
  26. package/src/codex/quota-auto-refresh-state.ts +8 -0
  27. package/src/codex/quota-auto-refresh.ts +157 -27
  28. package/src/codex/warmup.ts +5 -0
  29. package/src/config/provider-validation.ts +70 -1
  30. package/src/config.ts +73 -1
  31. package/src/generated/compatibility-version.json +67 -55
  32. package/src/lib/json-byte-size.ts +61 -0
  33. package/src/lib/provider-outbound.ts +11 -10
  34. package/src/oauth/index.ts +42 -8
  35. package/src/oauth/orcarouter.ts +200 -0
  36. package/src/providers/key-store.ts +22 -0
  37. package/src/providers/registry.ts +78 -29
  38. package/src/responses/citation-markers.ts +56 -24
  39. package/src/responses/reasoning-envelope.ts +53 -19
  40. package/src/server/auth-cors.ts +8 -0
  41. package/src/server/chat-completions.ts +7 -0
  42. package/src/server/chat-native.ts +65 -3
  43. package/src/server/claude-messages.ts +30 -10
  44. package/src/server/effort-policy.ts +183 -1
  45. package/src/server/management/agent-settings-routes.ts +50 -9
  46. package/src/server/management/config-routes.ts +2 -0
  47. package/src/server/management/logs-usage-routes.ts +11 -3
  48. package/src/server/management/model-routes.ts +65 -1
  49. package/src/server/management/model-rows.ts +5 -0
  50. package/src/server/management/provider-routes.ts +99 -6
  51. package/src/server/management/route-registry.ts +2 -0
  52. package/src/server/management/usage-aggregate-cache.ts +14 -7
  53. package/src/server/responses/compact.ts +3 -2
  54. package/src/server/responses/core.ts +16 -2
  55. package/src/server/startup-health-cache.ts +31 -9
  56. package/src/types/config.ts +5 -0
  57. package/src/types/provider.ts +4 -0
  58. package/src/usage/cost.ts +27 -31
  59. package/src/usage/summary.ts +52 -9
  60. package/src/usage/time-range.ts +48 -0
  61. package/src/usage/user-cost-overlays.ts +35 -3
  62. package/src/vision/anthropic-describe.ts +46 -2
  63. package/src/web-search/anthropic-executor.ts +49 -3
  64. package/src/web-search/parse.ts +1 -1
  65. package/gui/dist/assets/index-B72RPblC.js +0 -115
  66. package/gui/dist/assets/index-BFgUC17B.css +0 -1
@@ -42,23 +42,24 @@ export function hasCitationMarker(text: string): boolean {
42
42
  */
43
43
  export function stripCitationMarkers(text: string): string {
44
44
  if (!text.includes(CITATION_MARKER_START)) return text;
45
- let out = "";
46
- let index = 0;
47
- for (;;) {
48
- const start = text.indexOf(CITATION_MARKER_START, index);
49
- if (start === -1) {
50
- out += text.slice(index);
51
- return out;
52
- }
53
- const end = text.indexOf(CITATION_MARKER_END, start + 1);
54
- if (end === -1) {
55
- // Unterminated: keep the rest verbatim.
56
- out += text.slice(index);
57
- return out;
58
- }
59
- out += text.slice(index, start);
60
- index = end + 1;
45
+ // Walk START-delimited segments exactly like the streaming filter below: a START whose
46
+ // own segment (up to the next START) contains an END within the span bound is a span and
47
+ // is removed; a START that is superseded by another START before any END, or whose span
48
+ // exceeds MAX_CITATION_SPAN_LENGTH, is malformed text and stays verbatim. Pairing an
49
+ // earlier malformed START with a later span's END would delete real answer text and,
50
+ // worse, disagree with what the streaming deltas already emitted (#3843). The bound is
51
+ // shared with the streaming filter for the same reason: a span it has already released
52
+ // as over-bound must not be swallowed here when the END finally arrives.
53
+ let start = text.indexOf(CITATION_MARKER_START);
54
+ let out = text.slice(0, start);
55
+ while (start !== -1) {
56
+ const nextStart = text.indexOf(CITATION_MARKER_START, start + 1);
57
+ const segment = text.slice(start, nextStart === -1 ? text.length : nextStart);
58
+ const end = segment.indexOf(CITATION_MARKER_END, 1);
59
+ out += end === -1 || end + 1 > MAX_CITATION_SPAN_LENGTH ? segment : segment.slice(end + 1);
60
+ start = nextStart;
61
61
  }
62
+ return out;
62
63
  }
63
64
 
64
65
  export interface CitationMarkerFilter {
@@ -68,6 +69,18 @@ export interface CitationMarkerFilter {
68
69
  flush(): string;
69
70
  }
70
71
 
72
+ /**
73
+ * Upper bound on the length of a citation span (START through END inclusive), and therefore
74
+ * on the text the streaming filter withholds for one unterminated START.
75
+ *
76
+ * A real span is `cite` plus a few turn-scoped ids, so it is far under this. Without a
77
+ * bound, a backend that emits a START and never terminates it makes `held` grow for the
78
+ * whole response, and every later delta re-scans that accumulated prefix. The whole-string
79
+ * strip applies the same bound so both paths classify a span identically regardless of how
80
+ * the text was chunked.
81
+ */
82
+ const MAX_CITATION_SPAN_LENGTH = 4_096;
83
+
71
84
  /**
72
85
  * Streaming filter.
73
86
  *
@@ -75,6 +88,9 @@ export interface CitationMarkerFilter {
75
88
  * next — so a stateless per-delta strip would emit the tail of a span it never recognized.
76
89
  * This holds back the text from an unterminated START and releases it once the END arrives
77
90
  * (removed) or the stream ends (verbatim, so nothing the model actually said is lost).
91
+ *
92
+ * A span that grows past `MAX_CITATION_SPAN_LENGTH` is malformed ordinary text, so
93
+ * it is released verbatim instead of withheld; a later START can still open a valid span.
78
94
  */
79
95
  export function createCitationMarkerFilter(): CitationMarkerFilter {
80
96
  // Text from an open START that has not been terminated yet.
@@ -83,13 +99,30 @@ export function createCitationMarkerFilter(): CitationMarkerFilter {
83
99
  push(delta: string): string {
84
100
  const combined = held + delta;
85
101
  held = "";
86
- const start = combined.lastIndexOf(CITATION_MARKER_START);
87
- if (start === -1) return stripCitationMarkers(combined);
88
- const endAfterStart = combined.indexOf(CITATION_MARKER_END, start + 1);
89
- if (endAfterStart !== -1) return stripCitationMarkers(combined);
90
- // The trailing span is still open: emit everything before it, hold the rest.
91
- held = combined.slice(start);
92
- return stripCitationMarkers(combined.slice(0, start));
102
+ let start = combined.indexOf(CITATION_MARKER_START);
103
+ if (start === -1) return combined;
104
+ let out = combined.slice(0, start);
105
+ // Walk START-delimited segments independently so an earlier malformed START is never
106
+ // paired with a later span's END (the whole-string strip would do exactly that).
107
+ while (start !== -1) {
108
+ const nextStart = combined.indexOf(CITATION_MARKER_START, start + 1);
109
+ const segment = combined.slice(start, nextStart === -1 ? combined.length : nextStart);
110
+ const end = segment.indexOf(CITATION_MARKER_END, 1);
111
+ if (end !== -1 && end + 1 <= MAX_CITATION_SPAN_LENGTH) {
112
+ // A complete span: drop it, keep whatever trails it inside this segment.
113
+ out += segment.slice(end + 1);
114
+ } else if (end === -1 && nextStart === -1 && segment.length <= MAX_CITATION_SPAN_LENGTH) {
115
+ // Only a bounded trailing span can still be completed by a later delta.
116
+ held = segment;
117
+ } else {
118
+ // Superseded by a later START, or over the bound (with or without a late END):
119
+ // ordinary text, emitted verbatim so neither the retained text nor the per-delta
120
+ // rescan grows without limit.
121
+ out += segment;
122
+ }
123
+ start = nextStart;
124
+ }
125
+ return out;
93
126
  },
94
127
  flush(): string {
95
128
  const rest = held;
@@ -98,4 +131,3 @@ export function createCitationMarkerFilter(): CitationMarkerFilter {
98
131
  },
99
132
  };
100
133
  }
101
-
@@ -12,6 +12,9 @@
12
12
  * passthrough scrub strips ocxr1 envelopes before native forwarding.
13
13
  */
14
14
 
15
+ import { createTranslatorBudget, type TranslatorBudget } from "../lib/translator-budget";
16
+ import { jsonUtf8Bytes } from "../lib/json-byte-size";
17
+
15
18
  export const OCX_REASONING_PREFIX = "ocxr1:";
16
19
 
17
20
  export interface ReasoningEnvelope {
@@ -32,30 +35,61 @@ export interface ReasoningEnvelope {
32
35
  krc?: string;
33
36
  }
34
37
 
35
- export function encodeReasoningEnvelope(envelope: ReasoningEnvelope): string {
36
- return OCX_REASONING_PREFIX + Buffer.from(JSON.stringify(envelope), "utf-8").toString("base64");
38
+ export function encodeReasoningEnvelope(envelope: ReasoningEnvelope, budget?: TranslatorBudget): string {
39
+ const activeBudget = budget ?? createTranslatorBudget();
40
+ try {
41
+ const jsonBytes = jsonUtf8Bytes(envelope);
42
+ const base64Bytes = 4 * Math.ceil(jsonBytes / 3);
43
+ // Reserve before materialization: UTF-16 JSON, UTF-8 buffer, base64 string,
44
+ // and the prefixed result may coexist. Returned-value ownership stays with
45
+ // callers, whose existing retained accounting must not be charged twice here.
46
+ const reservation = activeBudget.reserveTransient(
47
+ Math.max(
48
+ 3 * jsonBytes + 4 * base64Bytes + 2 * OCX_REASONING_PREFIX.length,
49
+ 8 * (OCX_REASONING_PREFIX.length + base64Bytes),
50
+ ),
51
+ { kind: "reasoning" },
52
+ );
53
+ try {
54
+ return OCX_REASONING_PREFIX + Buffer.from(JSON.stringify(envelope), "utf-8").toString("base64");
55
+ } finally {
56
+ reservation.release();
57
+ }
58
+ } finally {
59
+ if (!budget) activeBudget.dispose();
60
+ }
37
61
  }
38
62
 
39
63
  /** Decode an ocxr1 envelope; returns null for native (OpenAI-encrypted) blobs or garbage. */
40
- export function decodeReasoningEnvelope(encryptedContent: string): ReasoningEnvelope | null {
64
+ export function decodeReasoningEnvelope(encryptedContent: string, budget?: TranslatorBudget): ReasoningEnvelope | null {
41
65
  if (!encryptedContent.startsWith(OCX_REASONING_PREFIX)) return null;
66
+ const activeBudget = budget ?? createTranslatorBudget();
42
67
  try {
43
- const parsed: unknown = JSON.parse(Buffer.from(encryptedContent.slice(OCX_REASONING_PREFIX.length), "base64").toString("utf-8"));
44
- if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return null;
45
- const obj = parsed as { sig?: unknown; red?: unknown };
46
- const envelope: ReasoningEnvelope = {};
47
- if (typeof obj.sig === "string") envelope.sig = obj.sig;
48
- if (Array.isArray(obj.red)) {
49
- const red = obj.red.filter((r): r is string => typeof r === "string");
50
- if (red.length > 0) envelope.red = red;
68
+ // Also bound already-encoded replay before slicing, decoding, or parsing it.
69
+ // Eight bytes per code unit conservatively covers the string/buffer copies.
70
+ const reservation = activeBudget.reserveTransient(8 * encryptedContent.length, { kind: "reasoning" });
71
+ try {
72
+ const parsed: unknown = JSON.parse(Buffer.from(encryptedContent.slice(OCX_REASONING_PREFIX.length), "base64").toString("utf-8"));
73
+ if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return null;
74
+ const obj = parsed as { sig?: unknown; red?: unknown };
75
+ const envelope: ReasoningEnvelope = {};
76
+ if (typeof obj.sig === "string") envelope.sig = obj.sig;
77
+ if (Array.isArray(obj.red)) {
78
+ const red = obj.red.filter((r): r is string => typeof r === "string");
79
+ if (red.length > 0) envelope.red = red;
80
+ }
81
+ const txt = (parsed as { txt?: unknown }).txt;
82
+ const hasTxt = typeof txt === "string";
83
+ if (hasTxt) envelope.txt = txt;
84
+ const krc = (parsed as { krc?: unknown }).krc;
85
+ if (typeof krc === "string" && krc.length > 0) envelope.krc = krc;
86
+ return envelope.sig || envelope.red || hasTxt || envelope.krc ? envelope : null;
87
+ } catch {
88
+ return null;
89
+ } finally {
90
+ reservation.release();
51
91
  }
52
- const txt = (parsed as { txt?: unknown }).txt;
53
- const hasTxt = typeof txt === "string";
54
- if (hasTxt) envelope.txt = txt;
55
- const krc = (parsed as { krc?: unknown }).krc;
56
- if (typeof krc === "string" && krc.length > 0) envelope.krc = krc;
57
- return envelope.sig || envelope.red || hasTxt || envelope.krc ? envelope : null;
58
- } catch {
59
- return null;
92
+ } finally {
93
+ if (!budget) activeBudget.dispose();
60
94
  }
61
95
  }
@@ -13,6 +13,7 @@ import {
13
13
  import {
14
14
  apiKeyTransportConfigError,
15
15
  booleanRecordConfigError,
16
+ providerReasoningPinsConfigError,
16
17
  modelAdapterRecordConfigError,
17
18
  nonBlankStringArrayConfigError,
18
19
  positiveIntegerConfigError,
@@ -581,6 +582,8 @@ export function providerManagementConfigError(name: unknown, provider: unknown):
581
582
  return "provider must be a plain object";
582
583
  }
583
584
  const raw = provider as Record<string, unknown>;
585
+ const pinsError = providerReasoningPinsConfigError(raw);
586
+ if (pinsError) return pinsError;
584
587
  for (const field of FORBIDDEN_PROVIDER_RUNTIME_FIELDS) {
585
588
  if (Object.hasOwn(raw, field)) return `provider ${name} must not include runtime field "${field}"`;
586
589
  }
@@ -594,6 +597,9 @@ export function providerManagementConfigError(name: unknown, provider: unknown):
594
597
  }
595
598
  if (seed) seed.codexAccountMode = raw.codexAccountMode;
596
599
  const canonicalCandidate = { ...raw };
600
+ // Validated operator overlays do not change the canonical auth/transport seed.
601
+ delete canonicalCandidate.pinnedReasoningEffort;
602
+ delete canonicalCandidate.modelPinnedReasoningEfforts;
597
603
  delete canonicalCandidate.responsesSnapshotRepair;
598
604
  // modelCosts is a user-owned display overlay, not part of the canonical
599
605
  // forward seed; it is validated separately below (providerModelCostsConfigError).
@@ -829,6 +835,8 @@ const PROVIDER_CONFIG_FIELD_POLICY = {
829
835
  reasoningEfforts: "editor",
830
836
  modelReasoningEfforts: "editor",
831
837
  modelDefaultReasoningEfforts: "editor",
838
+ pinnedReasoningEffort: "editor",
839
+ modelPinnedReasoningEfforts: "editor",
832
840
  modelSupportsReasoningSummaries: "editor",
833
841
  modelSupportsVerbosity: "editor",
834
842
  supportsVerbosity: "editor",
@@ -25,6 +25,8 @@ import { estimateTokens } from "../lib/token-estimate";
25
25
  import { NoEligiblePolicyCandidateError, UnknownRoutingPolicyError, routeModel } from "../router";
26
26
  import { evidenceFromBody } from "../routing/request-evidence";
27
27
  import { resolveWireProtocolOverride } from "./adapter-resolve";
28
+ import { resolveOpenCodeGoTransport } from "../providers/opencode-go-transport";
29
+ import { normalizeLogConversationId, sessionLaneIdFromRequest } from "./request-log-conversation";
28
30
  import type { OcxConfig } from "../types";
29
31
  import { readJsonRequestBody } from "./request-decompress";
30
32
  import {
@@ -136,6 +138,8 @@ async function handleChatCompletionsWithBudget(
136
138
  let chatNativeRoute: ReturnType<typeof routeModel> | null = null;
137
139
  try {
138
140
  const route = routeModel(config, chatBody.model as string, evidenceFromBody(chatBody));
141
+ route.provider = resolveOpenCodeGoTransport(route.provider,
142
+ sessionLaneIdFromRequest(req.headers) ?? normalizeLogConversationId(req.headers.get("x-opencode-session")));
139
143
  // Settle the wire once so every branch below reads the adapter this model will
140
144
  // actually use, not the provider-wide default (#404).
141
145
  route.provider = resolveWireProtocolOverride(route.providerName, route.modelId, route.provider, "chat");
@@ -237,6 +241,9 @@ async function handleChatCompletionsWithBudget(
237
241
  return chatCompletionsErrorResponse(400, CODEX_RESERVE_HELPER_UNSUPPORTED_MESSAGE, "invalid_request_error");
238
242
  }
239
243
  const headers = new Headers({ "content-type": "application/json" });
244
+ // Internal bridge metadata; the Go resolver scopes and hashes it before upstream use.
245
+ const openCodeSession = req.headers.get("x-opencode-session");
246
+ if (openCodeSession) headers.set("x-opencode-session", openCodeSession);
240
247
  for (const name of FORWARD_HEADERS) {
241
248
  if (name === "authorization" && !directRoute) continue;
242
249
  const value = req.headers.get(name);
@@ -6,6 +6,8 @@ import {
6
6
  collectChatCompletion,
7
7
  isChatCompletionsStreamError,
8
8
  } from "../chat/outbound";
9
+ import { applyChatEffortCap, chatCollabSurface, effortCapAppliesTo, resolvePinnedEffort, supportedLadderFor } from "./effort-policy";
10
+ import { mapReasoningEffort } from "../reasoning-effort";
9
11
  import {
10
12
  classifyError,
11
13
  cyberPolicyErrorType,
@@ -60,6 +62,68 @@ type Rec = Record<string, unknown>;
60
62
  const MAX_NATIVE_CHAT_JSON_BYTES = 32 * 1024 * 1024;
61
63
  const MAX_NATIVE_CHAT_ERROR_BYTES = 64 * 1024;
62
64
 
65
+ const chatEffortSnapshots = new WeakMap<Rec, {
66
+ inputModel: string;
67
+ providerName: string;
68
+ modelId: string;
69
+ present: boolean;
70
+ value: unknown;
71
+ annotation: string | undefined;
72
+ }>();
73
+
74
+ function normalizePinnedChatEffort(options: HandleNativeChatOptions): void {
75
+ const { chatBody, route, config, req, logCtx, requestedModel } = options;
76
+ let snapshot = chatEffortSnapshots.get(chatBody);
77
+ const inputModel = typeof chatBody.model === "string" ? chatBody.model : requestedModel;
78
+ let selector = inputModel;
79
+ if (snapshot) {
80
+ if (snapshot.providerName === route.providerName && snapshot.modelId === route.modelId) {
81
+ logCtx.requestedEffort = snapshot.annotation;
82
+ return;
83
+ }
84
+ if (snapshot.present) chatBody.reasoning_effort = snapshot.value;
85
+ else delete chatBody.reasoning_effort;
86
+ if (selector === snapshot.inputModel || selector === snapshot.modelId) {
87
+ selector = `${route.providerName}/${route.modelId}`;
88
+ }
89
+ } else {
90
+ snapshot = {
91
+ inputModel,
92
+ providerName: route.providerName,
93
+ modelId: route.modelId,
94
+ present: Object.hasOwn(chatBody, "reasoning_effort"),
95
+ value: chatBody.reasoning_effort,
96
+ annotation: undefined,
97
+ };
98
+ chatEffortSnapshots.set(chatBody, snapshot);
99
+ }
100
+ snapshot.inputModel = inputModel;
101
+ snapshot.providerName = route.providerName;
102
+ snapshot.modelId = route.modelId;
103
+ const from = typeof chatBody.reasoning_effort === "string" ? chatBody.reasoning_effort : undefined;
104
+ logCtx.requestedEffort = from;
105
+ // Compaction is normally excluded by native-route eligibility; preserve that boundary here too.
106
+ const pinned = chatBody.compaction_trigger === undefined
107
+ ? resolvePinnedEffort(route, selector, config)
108
+ : undefined;
109
+ if (pinned !== undefined) {
110
+ logCtx.requestedEffort = from ? `${from}->${pinned}` : pinned;
111
+ if (pinned === "none") delete chatBody.reasoning_effort;
112
+ else chatBody.reasoning_effort = pinned;
113
+ // The native lane historically passes caller effort through, including with caps set.
114
+ // Only a newly operator-pinned value enters the cap and provider-mapping pipeline.
115
+ if (effortCapAppliesTo(chatCollabSurface(chatBody), req.headers, config)) {
116
+ const capped = applyChatEffortCap(chatBody, req.headers, config, supportedLadderFor(route));
117
+ if (capped) logCtx.requestedEffort = `${logCtx.requestedEffort}->${capped.to}`;
118
+ }
119
+ const effort = typeof chatBody.reasoning_effort === "string" ? chatBody.reasoning_effort : undefined;
120
+ const wireEffort = mapReasoningEffort(route.provider, route.modelId, effort);
121
+ if (wireEffort === undefined) delete chatBody.reasoning_effort;
122
+ else chatBody.reasoning_effort = wireEffort;
123
+ }
124
+ snapshot.annotation = logCtx.requestedEffort;
125
+ }
126
+
63
127
  function isRec(value: unknown): value is Rec {
64
128
  return value !== null && typeof value === "object" && !Array.isArray(value);
65
129
  }
@@ -147,9 +211,7 @@ export async function handleNativeChatCompletions(options: HandleNativeChatOptio
147
211
  return chatCompletionsErrorResponse(status, safeMessage, type, code);
148
212
  };
149
213
 
150
- logCtx.requestedEffort = typeof options.chatBody.reasoning_effort === "string"
151
- ? options.chatBody.reasoning_effort
152
- : undefined;
214
+ normalizePinnedChatEffort(options);
153
215
  logCtx.requestedServiceTier = typeof options.chatBody.service_tier === "string"
154
216
  ? options.chatBody.service_tier
155
217
  : undefined;
@@ -7,6 +7,7 @@
7
7
  * unchanged. The Responses output (SSE or JSON) is converted back to Anthropic shape.
8
8
  */
9
9
  import { FORWARD_HEADERS } from "../adapters/openai-responses";
10
+ import { jsonUtf8Bytes } from "../lib/json-byte-size";
10
11
  import { sseFieldValue } from "../lib/sse-decoder";
11
12
  import { enforceAnthropicImageLimits, sniffImageDimensions } from "../adapters/anthropic-image-guard";
12
13
  import { normalizeAnthropicImages } from "../adapters/anthropic-image-normalize";
@@ -753,13 +754,13 @@ async function handleClaudeMessagesWithBudget(
753
754
  };
754
755
  delete anthropicBody.thinking;
755
756
  }
756
- const translation = anthropicToResponsesTranslation(anthropicBody, config.claudeCode);
757
+ const translation = anthropicToResponsesTranslation(anthropicBody, config.claudeCode, translatorBudget);
757
758
  internalBody = translation.body;
758
759
  // The Anthropic translator builds its body from model/input/store/stream plus sampling
759
760
  // fields only, so the caller intent is applied to the TRANSLATED body rather than the
760
761
  // inbound one.
761
762
  if (fastRow) internalBody.service_tier = "priority";
762
- translatorBudget.chargeRetained(new TextEncoder().encode(JSON.stringify(internalBody)).byteLength, { kind: "request_copies" });
763
+ translatorBudget.chargeRetained(jsonUtf8Bytes(internalBody), { kind: "request_copies" });
763
764
  cacheKeySource = translation.cacheKeySource;
764
765
  } catch (err) {
765
766
  const overflow = isTranslatorBudgetExceededError(err);
@@ -862,13 +863,26 @@ async function handleClaudeMessagesWithBudget(
862
863
  headers.set("session_id", uuidFromHex(internalBody.prompt_cache_key));
863
864
  }
864
865
  }
865
- const internalBodyJson = JSON.stringify(internalBody);
866
- translatorBudget.chargeRetained(new TextEncoder().encode(internalBodyJson).byteLength, { kind: "request_copies" });
867
- const internalReq = new Request("http://localhost/v1/responses", {
868
- method: "POST",
869
- headers,
870
- body: internalBodyJson,
871
- });
866
+ let internalReq: Request;
867
+ try {
868
+ // The UTF-16 JSON string and the Request's UTF-8 body coexist until dispatch.
869
+ const bodyBytes = jsonUtf8Bytes(internalBody);
870
+ const reservation = translatorBudget.reserveTransient(3 * bodyBytes, { kind: "request_copies" });
871
+ try {
872
+ internalReq = new Request("http://localhost/v1/responses", {
873
+ method: "POST",
874
+ headers,
875
+ body: JSON.stringify(internalBody),
876
+ });
877
+ } finally {
878
+ reservation.release();
879
+ }
880
+ translatorBudget.chargeRetained(bodyBytes, { kind: "request_copies" });
881
+ } catch (err) {
882
+ if (!isTranslatorBudgetExceededError(err)) throw err;
883
+ if (logIds) addFinalRequestLog(logIds.requestId, logIds.start, logCtx, 413, { closeReason: "non_stream" });
884
+ return anthropicErrorResponse(413, "request translation buffer exceeded the safe limit", "request_too_large", "translation_buffer_limit");
885
+ }
872
886
 
873
887
  // Request-log wiring mirrors the /v1/responses route: native passthrough finalizes
874
888
  // via the terminal callbacks; routed streams get the Responses-vocabulary log tap
@@ -1006,7 +1020,13 @@ async function handleClaudeMessagesWithBudget(
1006
1020
  }
1007
1021
  return anthropicErrorResponse(502, error?.message ?? "upstream request failed", "api_error");
1008
1022
  }
1009
- const message = responsesJsonToAnthropicMessage(json, requestedModel);
1023
+ let message: Rec;
1024
+ try {
1025
+ message = responsesJsonToAnthropicMessage(json, requestedModel, translatorBudget);
1026
+ } catch (err) {
1027
+ if (!isTranslatorBudgetExceededError(err)) throw err;
1028
+ return anthropicErrorResponse(413, "upstream translation buffer exceeded the safe limit", "request_too_large", "translation_buffer_limit");
1029
+ }
1010
1030
  if ((message as Rec).type === "error") {
1011
1031
  return new Response(JSON.stringify(message), {
1012
1032
  status: 529,
@@ -14,7 +14,7 @@
14
14
  */
15
15
  import type { OcxConfig, OcxParsedRequest, OcxProviderConfig } from "../types";
16
16
  import { modelInList } from "../types";
17
- import { codexEffortRank, configuredReasoningEfforts, isCodexReasoningEffort, modelRecordValue } from "../reasoning-effort";
17
+ import { codexEffortRank, configuredReasoningEfforts, isCodexReasoningEffort, isDeclaredReasoningEffort, modelRecordValue } from "../reasoning-effort";
18
18
  import { catalogModelEfforts } from "../codex/catalog";
19
19
 
20
20
  /**
@@ -188,3 +188,185 @@ export function applyEffortCap(
188
188
  if (raw?.reasoning && typeof raw.reasoning === "object") raw.reasoning.effort = resolved;
189
189
  return { from: requested, to: resolved, subagent };
190
190
  }
191
+
192
+ /**
193
+ * Resolve any pinned reasoning effort configured for this model or provider.
194
+ * Priority order:
195
+ * 1. Provider model-specific pinned effort (`provider.modelPinnedReasoningEfforts[modelId]`)
196
+ * 2. Provider-wide pinned effort (`provider.pinnedReasoningEffort`)
197
+ * 3. Global config model-specific pinned effort (`config.modelPinnedEfforts[modelId]`)
198
+ * Global keys try the final pre-namespace selector, provider-qualified destination,
199
+ * then bare destination, using modelRecordValue's exact/family/case-fold semantics.
200
+ * The caller removes synthetic effort rows and combo selectors before this boundary.
201
+ *
202
+ * Returns undefined when no valid pinned effort tier is configured.
203
+ */
204
+ export function resolvePinnedEffort(
205
+ route: { provider: OcxProviderConfig; modelId: string; providerName?: string },
206
+ parsedModelId?: string,
207
+ config?: OcxConfig,
208
+ ): string | undefined {
209
+ const prov = route.provider;
210
+ const rawProvModel = modelRecordValue(prov.modelPinnedReasoningEfforts, route.modelId)
211
+ ?? (parsedModelId ? modelRecordValue(prov.modelPinnedReasoningEfforts, parsedModelId) : undefined);
212
+ if (rawProvModel && isDeclaredReasoningEffort(rawProvModel)) {
213
+ return rawProvModel;
214
+ }
215
+ if (prov.pinnedReasoningEffort && isDeclaredReasoningEffort(prov.pinnedReasoningEffort)) {
216
+ return prov.pinnedReasoningEffort;
217
+ }
218
+ if (config?.modelPinnedEfforts) {
219
+ const rawGlobal = (parsedModelId ? modelRecordValue(config.modelPinnedEfforts, parsedModelId) : undefined)
220
+ ?? (route.providerName ? modelRecordValue(config.modelPinnedEfforts, `${route.providerName}/${route.modelId}`) : undefined)
221
+ ?? modelRecordValue(config.modelPinnedEfforts, route.modelId);
222
+ if (rawGlobal && isDeclaredReasoningEffort(rawGlobal)) {
223
+ return rawGlobal;
224
+ }
225
+ }
226
+ return undefined;
227
+ }
228
+
229
+ interface EffortSnapshot {
230
+ selector: string;
231
+ providerName: string;
232
+ modelId: string;
233
+ reasoningPresent: boolean;
234
+ reasoning: OcxParsedRequest["options"]["reasoning"];
235
+ rawEffortPresent: boolean;
236
+ rawEffort: unknown;
237
+ }
238
+
239
+ const effortSnapshots = new WeakMap<OcxParsedRequest, EffortSnapshot>();
240
+
241
+ /** Capture effective synthetic/combo defaults before final model namespace rewriting.
242
+ * A different destination restores effort alone; intervening summary/options edits survive.
243
+ * Credential retries do not change the destination and retain their existing decision.
244
+ */
245
+ export function prepareEffortNormalization(
246
+ parsed: OcxParsedRequest,
247
+ route: { providerName: string; modelId: string },
248
+ ): string {
249
+ const raw = parsed._rawBody as { reasoning?: Record<string, unknown> } | undefined;
250
+ const previous = effortSnapshots.get(parsed);
251
+ if (!previous) {
252
+ effortSnapshots.set(parsed, {
253
+ selector: parsed.modelId,
254
+ providerName: route.providerName,
255
+ modelId: route.modelId,
256
+ reasoningPresent: Object.hasOwn(parsed.options, "reasoning"),
257
+ reasoning: parsed.options.reasoning,
258
+ rawEffortPresent: !!raw?.reasoning && Object.hasOwn(raw.reasoning, "effort"),
259
+ rawEffort: raw?.reasoning?.effort,
260
+ });
261
+ return parsed.modelId;
262
+ }
263
+ if (previous.providerName === route.providerName && previous.modelId === route.modelId) {
264
+ return previous.selector;
265
+ }
266
+ if (previous.reasoningPresent) parsed.options.reasoning = previous.reasoning;
267
+ else delete parsed.options.reasoning;
268
+ if (raw && previous.rawEffortPresent) {
269
+ if (!raw.reasoning || typeof raw.reasoning !== "object") raw.reasoning = {};
270
+ raw.reasoning.effort = previous.rawEffort;
271
+ } else if (raw?.reasoning && typeof raw.reasoning === "object") {
272
+ delete raw.reasoning.effort;
273
+ }
274
+ // An unchanged wire model is the previous destination, not a new requested alias.
275
+ previous.selector = parsed.modelId === previous.modelId || parsed.modelId === previous.selector
276
+ ? `${route.providerName}/${route.modelId}`
277
+ : parsed.modelId;
278
+ previous.providerName = route.providerName;
279
+ previous.modelId = route.modelId;
280
+ return previous.selector;
281
+ }
282
+
283
+ /**
284
+ * Detect collaboration surface for a native chat request body.
285
+ * Mirrors Responses collabSurface behavior across function and custom tool representations.
286
+ */
287
+ export function chatCollabSurface(chatBody: Record<string, unknown>): "v1" | "v2" | null {
288
+ if (!Array.isArray(chatBody.tools)) return null;
289
+ let namespacedSpawn = false;
290
+ let flatSpawn = false;
291
+ let v1Only = false;
292
+ let v2Only = false;
293
+ for (const raw of chatBody.tools) {
294
+ if (!raw || typeof raw !== "object") continue;
295
+ const tool = raw as Record<string, unknown>;
296
+ let name = "";
297
+ let namespace: string | undefined = undefined;
298
+ if (tool.type === "function" && tool.function && typeof tool.function === "object") {
299
+ const fn = tool.function as Record<string, unknown>;
300
+ name = typeof fn.name === "string" ? fn.name : "";
301
+ } else if (tool.type === "custom" && tool.custom && typeof tool.custom === "object") {
302
+ const cust = tool.custom as Record<string, unknown>;
303
+ name = typeof cust.name === "string" ? cust.name : "";
304
+ } else if (typeof tool.name === "string") {
305
+ name = tool.name;
306
+ }
307
+ if (typeof tool.namespace === "string") namespace = tool.namespace;
308
+ if (name === "spawn_agent") {
309
+ if (namespace) namespacedSpawn = true;
310
+ else flatSpawn = true;
311
+ } else if (name === "send_input" || name === "resume_agent" || name === "close_agent") {
312
+ v1Only = true;
313
+ } else if (name === "send_message" || name === "followup_task" || name === "interrupt_agent" || name === "list_agents") {
314
+ v2Only = true;
315
+ }
316
+ }
317
+ if (!namespacedSpawn && !flatSpawn) return null;
318
+ if (namespacedSpawn && flatSpawn) return null;
319
+ if (v1Only && v2Only) return null;
320
+ if (v1Only) return "v1";
321
+ if (v2Only) return "v2";
322
+ return namespacedSpawn ? "v1" : "v2";
323
+ }
324
+
325
+ /**
326
+ * Apply effortCap to a native chat completions body when admitted by the collaboration gate.
327
+ */
328
+ export function applyChatEffortCap(
329
+ chatBody: Record<string, unknown>,
330
+ headers: Headers,
331
+ config: OcxConfig,
332
+ supported?: readonly string[] | undefined,
333
+ ): { from: string; to: string; subagent: boolean } | null {
334
+ const subagent = isThreadSpawnRequest(headers);
335
+ const cap = effortCapFor(config, subagent);
336
+ if (!cap) return null;
337
+ const resolved = resolveCappedEffort(cap, supported);
338
+ const requested = typeof chatBody.reasoning_effort === "string" ? chatBody.reasoning_effort : undefined;
339
+ if (resolved === null) {
340
+ if (!requested) return null;
341
+ delete chatBody.reasoning_effort;
342
+ return { from: requested, to: "none", subagent };
343
+ }
344
+ if (!requested || !isCodexReasoningEffort(requested)) return null;
345
+ if (codexEffortRank(requested) <= codexEffortRank(resolved)) return null;
346
+ chatBody.reasoning_effort = resolved;
347
+ return { from: requested, to: resolved, subagent };
348
+ }
349
+
350
+ export function applyPinnedEffort(
351
+ parsed: OcxParsedRequest,
352
+ route: { provider: OcxProviderConfig; modelId: string; providerName?: string },
353
+ config?: OcxConfig,
354
+ selector = effortSnapshots.get(parsed)?.selector ?? parsed.modelId,
355
+ ): { from: string | undefined; to: string } | null {
356
+ if (parsed._compactionRequest === true) return null;
357
+ const pinned = resolvePinnedEffort(route, selector, config);
358
+ if (!pinned) return null;
359
+ const requested = parsed.options.reasoning;
360
+ const raw = parsed._rawBody as { reasoning?: { effort?: string } } | undefined;
361
+ const targetEffort = pinned === "none" ? undefined : pinned;
362
+ parsed.options.reasoning = targetEffort;
363
+ if (targetEffort) {
364
+ if (raw && typeof raw === "object") {
365
+ if (!raw.reasoning || typeof raw.reasoning !== "object") raw.reasoning = {};
366
+ raw.reasoning.effort = targetEffort;
367
+ }
368
+ } else if (raw?.reasoning && typeof raw.reasoning === "object") {
369
+ delete raw.reasoning.effort;
370
+ }
371
+ return { from: requested, to: pinned };
372
+ }