@timo972/cc-router 0.9.0 → 0.10.0-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,19 +1,24 @@
1
1
  import express from "express";
2
2
  import { selectRoute } from "../providers/route-selector.js";
3
3
  import { forwardOpenAICodexResponse } from "../providers/openai/codex-transport.js";
4
- import { collectCodexResponseStream } from "../protocol/openai-responses-collect.js";
5
- import { stats } from "./stats.js";
4
+ import { collectCodexResponseStream, createCodexUsageObserver, usageFromResponseBody, } from "../protocol/openai-responses-collect.js";
5
+ import { stats, applyCodexUsage, boundModelId } from "./stats.js";
6
6
  import { logWarn } from "./logger.js";
7
+ import { extractCodexSessionKey, sendOpenAINoEligibleResponse } from "./openai-routing.js";
8
+ import { runOpenAIIngress, mirrorUpstreamHeaders, } from "./openai-ingress.js";
9
+ const RESPONSES_ENVELOPE = {
10
+ wrap: (type, message) => ({ error: { type, message } }),
11
+ sendNoEligible: (error, res, nowMs) => sendOpenAINoEligibleResponse(error, res, nowMs),
12
+ };
7
13
  function isResponsesRequest(value) {
8
14
  return (typeof value === "object" &&
9
15
  value !== null &&
10
16
  typeof value.model === "string" &&
11
17
  Array.isArray(value.input));
12
18
  }
13
- async function sendUpstreamResponse(upstream, res) {
19
+ async function sendUpstreamResponse(upstream, res, onChunk) {
20
+ mirrorUpstreamHeaders(upstream.headers, (key, value) => res.setHeader(key, value));
14
21
  const contentType = upstream.headers.get("content-type");
15
- if (contentType)
16
- res.setHeader("content-type", contentType);
17
22
  res.status(upstream.status);
18
23
  if (!upstream.body) {
19
24
  res.end();
@@ -24,13 +29,34 @@ async function sendUpstreamResponse(upstream, res) {
24
29
  res.flushHeaders?.();
25
30
  }
26
31
  const reader = upstream.body.getReader();
32
+ // Stop relaying the moment there is nobody to relay to. The ingress aborts
33
+ // the upstream fetch on disconnect, which normally makes the pending read
34
+ // reject on its own — but an upstream that has simply gone quiet leaves this
35
+ // loop parked in `read()`, where no amount of polling `res.destroyed` would
36
+ // ever run again. Racing the read against the close event is what actually
37
+ // releases a stalled stream, so the account's upstream slot is not held for
38
+ // a response nobody will receive.
39
+ const DISCONNECTED = Symbol("client-disconnected");
40
+ const disconnected = new Promise(resolve => {
41
+ if (res.destroyed)
42
+ resolve(DISCONNECTED);
43
+ else
44
+ res.once("close", () => resolve(DISCONNECTED));
45
+ });
27
46
  try {
28
47
  while (true) {
29
- const { value, done } = await reader.read();
48
+ const next = await Promise.race([reader.read(), disconnected]);
49
+ if (next === DISCONNECTED) {
50
+ await reader.cancel().catch(() => { });
51
+ break;
52
+ }
53
+ const { value, done } = next;
30
54
  if (done)
31
55
  break;
32
- if (value)
56
+ if (value) {
33
57
  res.write(Buffer.from(value));
58
+ onChunk?.(value);
59
+ }
34
60
  }
35
61
  }
36
62
  finally {
@@ -41,6 +67,7 @@ export function mountResponsesRoutes(app, opts) {
41
67
  const forwardOpenAI = opts.forwardOpenAI ?? forwardOpenAICodexResponse;
42
68
  const prepareOpenAIAccount = opts.prepareOpenAIAccount ?? (async () => true);
43
69
  const recordActivity = opts.recordActivity ?? ((entry) => stats.addLog(entry));
70
+ const now = opts.now ?? Date.now;
44
71
  app.post("/v1/responses", express.json({ limit: "10mb" }), async (req, res) => {
45
72
  if (!isResponsesRequest(req.body)) {
46
73
  res.status(400).json({
@@ -55,9 +82,10 @@ export function mountResponsesRoutes(app, opts) {
55
82
  recordActivity({
56
83
  ts: Date.now(),
57
84
  accountId: "-",
58
- model: req.body.model,
85
+ model: boundModelId(req.body.model),
59
86
  type: "warn",
60
87
  statusCode: 400,
88
+ path: "/v1/responses",
61
89
  details: "store:true rejected — Codex backend is stateless (store:false only)",
62
90
  });
63
91
  logWarn("responses", "store:true is not supported by the Codex backend; rejecting request");
@@ -73,8 +101,9 @@ export function mountResponsesRoutes(app, opts) {
73
101
  recordActivity({
74
102
  ts: Date.now(),
75
103
  accountId: "-",
76
- model: req.body.model,
104
+ model: boundModelId(req.body.model),
77
105
  type: "warn",
106
+ path: "/v1/responses",
78
107
  details: "max_output_tokens ignored — unsupported by the Codex backend",
79
108
  });
80
109
  logWarn("responses", "max_output_tokens is unsupported by the Codex backend and was dropped");
@@ -89,45 +118,77 @@ export function mountResponsesRoutes(app, opts) {
89
118
  });
90
119
  return;
91
120
  }
92
- const account = opts.getOpenAIAccount();
93
- if (!account) {
94
- res.status(503).json({
95
- error: {
96
- type: "no_accounts",
97
- message: "No OpenAI subscription accounts are configured",
98
- },
99
- });
100
- return;
101
- }
102
- const ready = await prepareOpenAIAccount(account);
103
- if (!ready) {
104
- res.status(401).json({
105
- error: {
106
- type: "authentication_error",
107
- message: "OpenAI subscription token refresh failed",
108
- },
109
- });
110
- return;
111
- }
112
- const body = {
113
- ...req.body,
114
- model: route.upstreamModel,
115
- };
116
- const upstream = await forwardOpenAI({
117
- account,
118
- body,
119
- stream: body.stream === true,
121
+ const body = { ...req.body, model: route.upstreamModel };
122
+ await runOpenAIIngress({
123
+ res,
124
+ sessionKey: extractCodexSessionKey(req, req.body),
125
+ requestedModel: route.upstreamModel,
126
+ path: "/v1/responses",
127
+ openAIRouter: opts.openAIRouter,
128
+ openAIPool: opts.openAIPool,
129
+ prepareOpenAIAccount,
130
+ forwardOpenAI,
131
+ forwardBody: body,
132
+ recordActivity,
133
+ now,
134
+ envelope: RESPONSES_ENVELOPE,
135
+ onUpstreamAuthFailure: opts.onUpstreamAuthFailure,
136
+ relay: async (upstream, res, entry, report) => {
137
+ if (body.stream === true) {
138
+ const observer = createCodexUsageObserver();
139
+ // Only an upstream that actually promised a successful event stream
140
+ // can be judged on whether that stream completed. A non-OK response
141
+ // (a plain 401/429/5xx body) has no SSE events to observe, so
142
+ // `failure()` would always report a missing completion — and
143
+ // `sendUpstreamResponse` has already relayed the real status to the
144
+ // client, so reporting 502 here would log a 502 for a client that
145
+ // received a 429 and hide the actual failure from diagnostics.
146
+ const streamed = upstream.ok
147
+ && (upstream.headers.get("content-type") ?? "").includes("text/event-stream");
148
+ // Reported per chunk, not once at the end: this relay can throw
149
+ // (or be cut short) after upstream has already announced a failure,
150
+ // and the verdict has to survive that.
151
+ await sendUpstreamResponse(upstream, res, chunk => {
152
+ observer.push(chunk);
153
+ if (observer.explicitFailure() !== undefined)
154
+ report.upstreamReportedFailure = true;
155
+ });
156
+ applyCodexUsage(entry, observer.finish());
157
+ // Bytes already written to the client are untouched — this only
158
+ // changes the REPORTED status (used for stats/activity/cooldown),
159
+ // matching a stream that upstream answered `200` but that ended in
160
+ // a `response.failed`/`error` SSE event instead of completing.
161
+ const synthesized = streamed && observer.failure() !== undefined;
162
+ // A non-OK upstream has no SSE events to judge, so anything the
163
+ // observer saw belongs to a body that was never an event stream.
164
+ if (!streamed)
165
+ report.upstreamReportedFailure = !upstream.ok;
166
+ return { statusCode: synthesized ? 502 : upstream.status };
167
+ }
168
+ const collected = await collectCodexResponseStream(upstream, () => {
169
+ report.upstreamReportedFailure = true;
170
+ });
171
+ // Mirror upstream headers (e.g. Retry-After, x-codex-*) before sending the
172
+ // collected body, so failure responses reach the client unchanged per the
173
+ // same contract the streaming path already honors via sendUpstreamResponse.
174
+ // content-type is deliberately excluded here: Express's res.json() only sets
175
+ // it when unset, so a mirrored content-type would silently win over the
176
+ // application/json the .json() call below is supposed to set. .json()/.type()
177
+ // remain the single source of truth for content-type, as today.
178
+ mirrorUpstreamHeaders(upstream.headers, (key, value) => {
179
+ if (key.toLowerCase() === "content-type")
180
+ return;
181
+ res.setHeader(key, value);
182
+ });
183
+ if (collected.kind === "json") {
184
+ applyCodexUsage(entry, usageFromResponseBody(collected.body));
185
+ res.status(collected.status).json(collected.body);
186
+ }
187
+ else {
188
+ res.status(collected.status).type(collected.contentType ?? "text/plain").send(collected.body);
189
+ }
190
+ return { statusCode: collected.status };
191
+ },
120
192
  });
121
- if (body.stream === true) {
122
- await sendUpstreamResponse(upstream, res);
123
- return;
124
- }
125
- const collected = await collectCodexResponseStream(upstream);
126
- if (collected.kind === "json") {
127
- res.status(collected.status).json(collected.body);
128
- }
129
- else {
130
- res.status(collected.status).type(collected.contentType ?? "text/plain").send(collected.body);
131
- }
132
193
  });
133
194
  }