@timo972/cc-router 0.9.0 → 0.10.0-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,46 +5,167 @@ import { openAIResponseToAnthropicMessage } from "../protocol/openai-response-to
5
5
  import { createOpenAIStreamToAnthropicNormalizer } from "../protocol/openai-stream-to-anthropic.js";
6
6
  import { encodeSseEvent, parseSseLines } from "../protocol/sse.js";
7
7
  import { forwardOpenAICodexResponse } from "../providers/openai/codex-transport.js";
8
+ import { terminalResponsePayload, usageFromTerminalEvent, usageFromResponseBody, } from "../protocol/openai-responses-collect.js";
8
9
  import { extractAnthropicRouteContext } from "./request-model.js";
10
+ import { stats, applyCodexUsage } from "./stats.js";
11
+ import { extractCodexSessionKey } from "./openai-routing.js";
12
+ import { sendAnthropicNoEligibleResponse } from "./anthropic-routing.js";
13
+ import { mirrorUpstreamHeaders, runOpenAIIngress, } from "./openai-ingress.js";
14
+ const MESSAGES_ENVELOPE = {
15
+ wrap: (type, message) => ({ type: "error", error: { type, message } }),
16
+ sendNoEligible: (error, res, nowMs) => sendAnthropicNoEligibleResponse(error, res, nowMs),
17
+ };
9
18
  function isAnthropicMessagesRequest(value) {
10
19
  return (typeof value === "object" &&
11
20
  value !== null &&
12
21
  Array.isArray(value.messages));
13
22
  }
14
- async function sendOpenAIAsAnthropic(upstream, res, requestedStream) {
23
+ /** Longest upstream error snippet echoed back when the body isn't JSON. */
24
+ const MAX_UPSTREAM_ERROR_MESSAGE_LENGTH = 200;
25
+ /** Anthropic error `type` for a relayed upstream HTTP failure status. */
26
+ function anthropicErrorTypeForStatus(status) {
27
+ if (status === 429)
28
+ return "rate_limit_error";
29
+ if (status === 401)
30
+ return "authentication_error";
31
+ if (status >= 500)
32
+ return "upstream_error";
33
+ return "invalid_request_error";
34
+ }
35
+ /**
36
+ * Best-effort human-readable message for a non-OK upstream response: prefer
37
+ * a JSON `error.message`, else fall back to a bounded, control-character-free
38
+ * snippet of the raw body, else a generic status-only message.
39
+ */
40
+ function extractUpstreamErrorMessage(bodyText, status) {
41
+ const trimmed = bodyText.trim();
42
+ if (trimmed) {
43
+ try {
44
+ const parsed = JSON.parse(trimmed);
45
+ const message = parsed.error?.message;
46
+ if (typeof message === "string" && message)
47
+ return message;
48
+ }
49
+ catch {
50
+ // Not JSON — fall through to a bounded text snippet below.
51
+ }
52
+ const snippet = trimmed.replace(/[\x00-\x1f\x7f]/g, "").slice(0, MAX_UPSTREAM_ERROR_MESSAGE_LENGTH);
53
+ if (snippet)
54
+ return snippet;
55
+ }
56
+ return `Upstream request failed with status ${status}`;
57
+ }
58
+ /**
59
+ * Relay an upstream Codex response in Anthropic Messages shape and report the
60
+ * status the client actually received — which is not always `upstream.status`:
61
+ * the Codex backend signals in-stream failures as a `response.failed`/`error`
62
+ * SSE event on an HTTP 200, and those must not be reported as success.
63
+ */
64
+ async function sendOpenAIAsAnthropic(upstream, res, requestedStream, entry, report) {
65
+ const onUsage = (usage) => applyCodexUsage(entry, usage);
66
+ // A non-OK response is never a Responses payload, whatever its content-type
67
+ // — parsing it as an event stream or a completed response would translate a
68
+ // real upstream failure (401/429/5xx) into an empty "success". Handle it
69
+ // before any content-type dispatch and relay the upstream status verbatim,
70
+ // never a synthesized 502.
71
+ if (!upstream.ok) {
72
+ const bodyText = await upstream.text();
73
+ // Mirror the safe upstream headers so a client can honor the server's
74
+ // backoff: without Retry-After a 429 tells the caller to slow down but not
75
+ // for how long. content-type is skipped for the same reason as the
76
+ // /v1/responses collected path — res.json() below only sets it when unset,
77
+ // so a mirrored value would silently win over the JSON envelope's own.
78
+ mirrorUpstreamHeaders(upstream.headers, (key, value) => {
79
+ if (key.toLowerCase() === "content-type")
80
+ return;
81
+ res.setHeader(key, value);
82
+ });
83
+ res.status(upstream.status).json({
84
+ type: "error",
85
+ error: {
86
+ type: anthropicErrorTypeForStatus(upstream.status),
87
+ message: extractUpstreamErrorMessage(bodyText, upstream.status),
88
+ },
89
+ });
90
+ report.upstreamReportedFailure = true;
91
+ return { statusCode: upstream.status };
92
+ }
15
93
  const contentType = upstream.headers.get("content-type") ?? "";
16
94
  if (contentType.includes("text/event-stream")) {
17
95
  if (requestedStream) {
18
- await sendOpenAIStreamAsAnthropic(upstream, res);
19
- return;
96
+ // Headers are already flushed by the time a mid-stream failure surfaces,
97
+ // so the client keeps the partial stream; reporting 502 here keeps the
98
+ // activity log and error totals honest about what happened.
99
+ const failure = await sendOpenAIStreamAsAnthropic(upstream, res, onUsage, report);
100
+ return { statusCode: failure === undefined ? upstream.status : 502 };
20
101
  }
21
- res.status(upstream.status).json(await collectOpenAIStreamAsAnthropicMessage(upstream));
22
- return;
102
+ const collected = await collectOpenAIStreamAsAnthropicMessage(upstream, report);
103
+ onUsage(collected.usage);
104
+ if (collected.failure !== undefined) {
105
+ // Mirrors collectCodexResponseStream on the /v1/responses path: a stream
106
+ // that ended in failure is a 502, never an empty 200 "success".
107
+ res.status(502).json({
108
+ type: "error",
109
+ error: { type: "upstream_error", message: collected.failure },
110
+ });
111
+ return { statusCode: 502 };
112
+ }
113
+ res.status(upstream.status).json(collected.message);
114
+ return { statusCode: upstream.status };
23
115
  }
24
116
  if (!contentType.includes("application/json")) {
25
117
  res.status(upstream.status);
26
118
  res.setHeader("content-type", contentType || "text/plain");
27
119
  res.send(await upstream.text());
28
- return;
120
+ return { statusCode: upstream.status };
29
121
  }
30
122
  const json = await upstream.json();
123
+ onUsage(usageFromResponseBody(json));
31
124
  res.status(upstream.status).json(openAIResponseToAnthropicMessage(json));
125
+ return { statusCode: upstream.status };
32
126
  }
33
- async function collectOpenAIStreamAsAnthropicMessage(upstream) {
127
+ async function collectOpenAIStreamAsAnthropicMessage(upstream, report) {
34
128
  const reader = upstream.body?.getReader();
35
129
  if (!reader) {
36
- return openAIResponseToAnthropicMessage({ id: "", model: "", output: [], usage: {} });
130
+ // An event-stream response with no body cannot have reached a terminal
131
+ // event. Reporting it as a failure keeps the caller from fabricating an
132
+ // empty 200 — and it is upstream's doing, not a truncation a client
133
+ // disconnect could account for, so the ingress must not write it off as a
134
+ // cancellation when both happen at once.
135
+ report.upstreamReportedFailure = true;
136
+ return {
137
+ message: openAIResponseToAnthropicMessage({ id: "", model: "", output: [], usage: {} }),
138
+ usage: undefined,
139
+ failure: "Upstream response had no body",
140
+ };
37
141
  }
38
142
  const decoder = new TextDecoder();
39
143
  let remainder = "";
40
144
  let id = "";
41
145
  let model = "";
42
146
  let text = "";
147
+ let failure;
148
+ let completed = false;
43
149
  let usage = {};
150
+ let status;
151
+ let incompleteDetails;
44
152
  const applyEvent = (event) => {
45
153
  if (typeof event !== "object" || event === null)
46
154
  return;
47
155
  const openAIEvent = event;
156
+ // Reported the moment it is seen, not when this function returns: the
157
+ // read after it can be cut short by a client disconnect, and losing the
158
+ // verdict there turns a real backend failure into a benign cancellation.
159
+ if (openAIEvent.type === "response.failed") {
160
+ failure = openAIEvent.response?.error?.message ?? "Response failed";
161
+ report.upstreamReportedFailure = true;
162
+ return;
163
+ }
164
+ if (openAIEvent.type === "error") {
165
+ failure = openAIEvent.error?.message ?? "Upstream error event";
166
+ report.upstreamReportedFailure = true;
167
+ return;
168
+ }
48
169
  if (openAIEvent.type === "response.created") {
49
170
  id = openAIEvent.response?.id ?? id;
50
171
  model = openAIEvent.response?.model ?? model;
@@ -54,36 +175,58 @@ async function collectOpenAIStreamAsAnthropicMessage(upstream) {
54
175
  text += openAIEvent.delta ?? "";
55
176
  return;
56
177
  }
57
- if (openAIEvent.type === "response.completed") {
178
+ if (terminalResponsePayload(event) !== undefined) {
58
179
  id = openAIEvent.response?.id ?? id;
59
180
  model = openAIEvent.response?.model ?? model;
60
181
  usage = openAIEvent.response?.usage ?? usage;
182
+ // How the turn ended has to survive into the reconstructed response
183
+ // below: it is what tells the translator to report `max_tokens` rather
184
+ // than an `end_turn` that would make a truncated answer look deliberate.
185
+ status = openAIEvent.response?.status ?? status;
186
+ incompleteDetails = openAIEvent.response?.incomplete_details ?? incompleteDetails;
187
+ completed = true;
61
188
  }
62
189
  };
63
190
  while (true) {
64
191
  const { value, done } = await reader.read();
65
192
  if (done)
66
193
  break;
67
- const parsed = parseSseLines(remainder + decoder.decode(value, { stream: true }));
194
+ const parsed = parseSseLines(remainder + decoder.decode(value, { stream: true }), { tolerant: true });
68
195
  remainder = parsed.remainder;
69
196
  parsed.events.forEach(applyEvent);
70
197
  }
71
198
  const tail = decoder.decode();
72
199
  if (tail || remainder) {
73
- parseSseLines(remainder + tail + "\n").events.forEach(applyEvent);
200
+ parseSseLines(remainder + tail + "\n", { tolerant: true }).events.forEach(applyEvent);
74
201
  }
75
- return openAIResponseToAnthropicMessage({
76
- id,
77
- model,
78
- output: text ? [{
79
- type: "message",
80
- role: "assistant",
81
- content: [{ type: "output_text", text }],
82
- }] : [],
83
- usage,
84
- });
202
+ return {
203
+ message: openAIResponseToAnthropicMessage({
204
+ id,
205
+ model,
206
+ output: text ? [{
207
+ type: "message",
208
+ role: "assistant",
209
+ content: [{ type: "output_text", text }],
210
+ }] : [],
211
+ usage,
212
+ ...(status ? { status } : {}),
213
+ ...(incompleteDetails ? { incomplete_details: incompleteDetails } : {}),
214
+ }),
215
+ usage: usageFromResponseBody({ usage }),
216
+ // Tolerant parsing skips a malformed frame rather than aborting the read,
217
+ // which keeps a bad nonterminal frame from truncating the stream — but it
218
+ // also means a malformed *terminal* frame (`response.completed` or
219
+ // `response.incomplete`) would silently vanish. Without an observed
220
+ // terminal event there is no answer to return, so a stream that ends
221
+ // without one (dropped terminal frame, or an upstream that simply stopped
222
+ // mid-flight) is a failure rather than an empty success. An explicit
223
+ // `response.failed`/`error` message wins, since it says more about what
224
+ // went wrong.
225
+ failure: failure ?? (completed ? undefined : "Upstream stream ended without a terminal response event"),
226
+ };
85
227
  }
86
- async function sendOpenAIStreamAsAnthropic(upstream, res) {
228
+ /** Returns the upstream failure message when the stream ended in one. */
229
+ async function sendOpenAIStreamAsAnthropic(upstream, res, onUsage, report) {
87
230
  res.status(upstream.status);
88
231
  res.setHeader("content-type", "text/event-stream");
89
232
  res.setHeader("cache-control", "no-cache");
@@ -92,40 +235,89 @@ async function sendOpenAIStreamAsAnthropic(upstream, res) {
92
235
  const reader = upstream.body?.getReader();
93
236
  if (!reader) {
94
237
  res.end();
95
- return;
238
+ // No body means no terminal response event was ever possible either —
239
+ // mirror collectOpenAIStreamAsAnthropicMessage's `!reader` case below
240
+ // rather than reporting an empty stream as a success.
241
+ report.upstreamReportedFailure = true;
242
+ return "Upstream response had no body";
96
243
  }
97
244
  const decoder = new TextDecoder();
98
245
  let remainder = "";
246
+ // Usage is applied once, after the stream ends: `applyCodexUsage`
247
+ // accumulates into the process-wide totals, so calling it per terminal event
248
+ // would double-count a stream that carried more than one. Mirrors
249
+ // `createCodexUsageObserver`'s finish()-once contract.
250
+ let totals;
251
+ let failure;
252
+ let completed = false;
253
+ const inspect = (event) => {
254
+ totals = usageFromTerminalEvent(event) ?? totals;
255
+ if (typeof event !== "object" || event === null)
256
+ return;
257
+ const typed = event;
258
+ // Recorded as observed — an aborted read can reject on the very next
259
+ // chunk, and this verdict has to outlive that.
260
+ if (typed.type === "response.failed") {
261
+ failure = typed.response?.error?.message ?? "Response failed";
262
+ report.upstreamReportedFailure = true;
263
+ }
264
+ else if (typed.type === "error") {
265
+ failure = typed.error?.message ?? "Upstream error event";
266
+ report.upstreamReportedFailure = true;
267
+ }
268
+ else if (terminalResponsePayload(event) !== undefined) {
269
+ completed = true;
270
+ }
271
+ };
272
+ const relayEvents = (events) => {
273
+ for (const event of events) {
274
+ inspect(event);
275
+ for (const mapped of normalizer.convert(event)) {
276
+ res.write(encodeSseEvent(mapped));
277
+ }
278
+ }
279
+ };
99
280
  try {
100
281
  while (true) {
282
+ // A client disconnect ends this loop through the upstream fetch: the
283
+ // ingress aborts its signal, which rejects the pending read. That is the
284
+ // cancellation path for every relay here — `sendUpstreamResponse` adds a
285
+ // close-event race on top only because it also relays bodies that may
286
+ // not honour the signal.
101
287
  const { value, done } = await reader.read();
102
288
  if (done)
103
289
  break;
104
- const parsed = parseSseLines(remainder + decoder.decode(value, { stream: true }));
290
+ // Tolerant: one malformed frame must not abort the relay (which would
291
+ // silently truncate the client's stream) nor discard the valid events
292
+ // decoded from the same chunk.
293
+ const parsed = parseSseLines(remainder + decoder.decode(value, { stream: true }), { tolerant: true });
105
294
  remainder = parsed.remainder;
106
- for (const event of parsed.events) {
107
- for (const mapped of normalizer.convert(event)) {
108
- res.write(encodeSseEvent(mapped));
109
- }
110
- }
295
+ relayEvents(parsed.events);
111
296
  }
112
297
  const tail = decoder.decode();
113
298
  if (tail || remainder) {
114
- const parsed = parseSseLines(remainder + tail + "\n");
115
- for (const event of parsed.events) {
116
- for (const mapped of normalizer.convert(event)) {
117
- res.write(encodeSseEvent(mapped));
118
- }
119
- }
299
+ relayEvents(parseSseLines(remainder + tail + "\n", { tolerant: true }).events);
120
300
  }
121
301
  }
122
302
  finally {
123
303
  res.end();
304
+ onUsage?.(totals);
124
305
  }
306
+ // Mirrors collectOpenAIStreamAsAnthropicMessage: tolerant parsing skips a
307
+ // malformed frame rather than aborting the relay, which also means a
308
+ // malformed *terminal* frame (`response.completed` or `response.incomplete`)
309
+ // would silently vanish. Without an observed terminal event the client
310
+ // received a partial answer, not a finished one, so it is reported as a
311
+ // failure (bytes already relayed to the client are unaffected — only the
312
+ // status used for stats/activity changes). An explicit response.failed/error
313
+ // message wins, since it says more about what went wrong.
314
+ return failure ?? (completed ? undefined : "Upstream stream ended without a terminal response event");
125
315
  }
126
316
  export function mountMessagesCrossProviderRoute(app, opts) {
127
317
  const forwardOpenAI = opts.forwardOpenAI ?? forwardOpenAICodexResponse;
128
318
  const prepareOpenAIAccount = opts.prepareOpenAIAccount ?? (async () => true);
319
+ const recordActivity = opts.recordActivity ?? ((entry) => stats.addLog(entry));
320
+ const now = opts.now ?? Date.now;
129
321
  app.post("/v1/messages", express.json({
130
322
  limit: "10mb",
131
323
  verify: (req, _res, buf) => {
@@ -149,34 +341,23 @@ export function mountMessagesCrossProviderRoute(app, opts) {
149
341
  next();
150
342
  return;
151
343
  }
152
- const account = opts.getOpenAIAccount();
153
- if (!account) {
154
- res.status(503).json({
155
- type: "error",
156
- error: {
157
- type: "no_accounts",
158
- message: "No OpenAI subscription accounts are configured",
159
- },
160
- });
161
- return;
162
- }
163
- const ready = await prepareOpenAIAccount(account);
164
- if (!ready) {
165
- res.status(401).json({
166
- type: "error",
167
- error: {
168
- type: "authentication_error",
169
- message: "OpenAI subscription token refresh failed",
170
- },
171
- });
172
- return;
173
- }
174
344
  const body = anthropicToOpenAIResponses(req.body, opts.modelRouting);
175
- const upstream = await forwardOpenAI({
176
- account,
177
- body,
178
- stream: body.stream === true,
345
+ const requestedStream = req.body.stream === true;
346
+ await runOpenAIIngress({
347
+ res,
348
+ sessionKey: extractCodexSessionKey(req, req.body),
349
+ requestedModel: route.upstreamModel,
350
+ path: "/v1/messages",
351
+ openAIRouter: opts.openAIRouter,
352
+ openAIPool: opts.openAIPool,
353
+ prepareOpenAIAccount,
354
+ forwardOpenAI,
355
+ forwardBody: body,
356
+ recordActivity,
357
+ now,
358
+ envelope: MESSAGES_ENVELOPE,
359
+ onUpstreamAuthFailure: opts.onUpstreamAuthFailure,
360
+ relay: (upstream, res, entry, report) => sendOpenAIAsAnthropic(upstream, res, requestedStream, entry, report),
179
361
  });
180
- await sendOpenAIAsAnthropic(upstream, res, req.body.stream === true);
181
362
  });
182
363
  }