ldrouter 1.17.6 → 1.17.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/dist/server/db/migrate.js +4 -4
- package/dist/server/db/schema.js +4 -4
- package/dist/server/providers/codex-refresh.js +6 -1
- package/dist/server/routes/admin/providers.js +8 -8
- package/dist/server/upstream/client.js +10 -4
- package/migrations/0009_raise_timeout_defaults.sql +15 -0
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,19 @@ All notable changes to this project are documented here. The format follows
|
|
|
4
4
|
[Keep a Changelog](https://keepachangelog.com/) and the project adheres to
|
|
5
5
|
[Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [1.17.8] - 2026-09-21
|
|
8
|
+
|
|
9
|
+
### Changed
|
|
10
|
+
|
|
11
|
+
- Raised provider connection, first-token, stream-idle, and total request timeout defaults to 15s, 60s, 300s, and 600s. Existing providers still using the old defaults are upgraded by migration 0009.
|
|
12
|
+
- Distinguish total request timeout errors from stream-idle timeout errors.
|
|
13
|
+
|
|
14
|
+
## [1.17.7] - 2026-09-20
|
|
15
|
+
|
|
16
|
+
### Fixed
|
|
17
|
+
|
|
18
|
+
- **Codex accounts were refreshed at the last minute, in the window where a rotating grant is most fragile.** The refresh lead is a threshold on the *remaining* lifetime, not a token lifetime — `needsCodexRefresh` fires once `expires - now` drops below it — and it sat at 5 minutes, so the exchange ran as the access token expired. Because OpenAI rotates the refresh token on every exchange, that is exactly when another client holding the same grant can invalidate it, and the observed fleet showed every account at `consecutive_failures = 0`: the first failure was already being treated as final. Codex access tokens last 10 days (measured from the fleet: a rotation at 13:39:00 expired at 13:39:00 ten days later), so the lead is now 5 days — half a lifetime of margin, refreshing every 5 days instead of every 10. This matches the 5-day `refreshLeadMs` 9router ships for Codex, and it surfaces a dead grant on day 5 rather than day 10. The lead is deliberately far below the lifetime: any value above it makes `needsCodexRefresh` permanently true and refreshes on *every* request, which multiplies refresh-token rotation by the request count.
|
|
19
|
+
|
|
7
20
|
## [1.17.6] - 2026-09-20
|
|
8
21
|
|
|
9
22
|
### Fixed
|
|
@@ -148,10 +148,10 @@ function buildInitialSchemaSql() {
|
|
|
148
148
|
custom_headers_encrypted TEXT,
|
|
149
149
|
custom_headers_nonce TEXT,
|
|
150
150
|
enabled INTEGER NOT NULL DEFAULT 1,
|
|
151
|
-
connect_timeout_ms INTEGER NOT NULL DEFAULT
|
|
152
|
-
first_token_timeout_ms INTEGER NOT NULL DEFAULT
|
|
153
|
-
stream_idle_timeout_ms INTEGER NOT NULL DEFAULT
|
|
154
|
-
total_timeout_ms INTEGER NOT NULL DEFAULT
|
|
151
|
+
connect_timeout_ms INTEGER NOT NULL DEFAULT 15000,
|
|
152
|
+
first_token_timeout_ms INTEGER NOT NULL DEFAULT 60000,
|
|
153
|
+
stream_idle_timeout_ms INTEGER NOT NULL DEFAULT 300000,
|
|
154
|
+
total_timeout_ms INTEGER NOT NULL DEFAULT 600000,
|
|
155
155
|
max_retries INTEGER NOT NULL DEFAULT 2,
|
|
156
156
|
retry_base_ms INTEGER NOT NULL DEFAULT 500,
|
|
157
157
|
retry_max_ms INTEGER NOT NULL DEFAULT 8000,
|
package/dist/server/db/schema.js
CHANGED
|
@@ -89,10 +89,10 @@ export const providers = sqliteTable('providers', {
|
|
|
89
89
|
customHeadersEncrypted: text('custom_headers_encrypted'),
|
|
90
90
|
customHeadersNonce: text('custom_headers_nonce'),
|
|
91
91
|
enabled: integer('enabled', { mode: 'boolean' }).notNull().notNull().default(true),
|
|
92
|
-
connectTimeoutMs: integer('connect_timeout_ms').notNull().default(
|
|
93
|
-
firstTokenTimeoutMs: integer('first_token_timeout_ms').notNull().default(
|
|
94
|
-
streamIdleTimeoutMs: integer('stream_idle_timeout_ms').notNull().default(
|
|
95
|
-
totalTimeoutMs: integer('total_timeout_ms').notNull().default(
|
|
92
|
+
connectTimeoutMs: integer('connect_timeout_ms').notNull().default(15000),
|
|
93
|
+
firstTokenTimeoutMs: integer('first_token_timeout_ms').notNull().default(60000),
|
|
94
|
+
streamIdleTimeoutMs: integer('stream_idle_timeout_ms').notNull().default(300000),
|
|
95
|
+
totalTimeoutMs: integer('total_timeout_ms').notNull().default(600000),
|
|
96
96
|
maxRetries: integer('max_retries').notNull().default(2),
|
|
97
97
|
retryBaseMs: integer('retry_base_ms').notNull().default(500),
|
|
98
98
|
retryMaxMs: integer('retry_max_ms').notNull().default(8000),
|
|
@@ -28,7 +28,12 @@ const PERMANENT_REFRESH_FAILURES = new Set(['oauth_refresh_failed', 'credential_
|
|
|
28
28
|
export function isPermanentRefreshFailure(error) {
|
|
29
29
|
return PERMANENT_REFRESH_FAILURES.has(error);
|
|
30
30
|
}
|
|
31
|
-
|
|
31
|
+
// Lead is measured against the remaining lifetime, and must stay well under it: a refresh that fires
|
|
32
|
+
// at the last minute races the expiry, and because OpenAI rotates the refresh token on every
|
|
33
|
+
// exchange, that is exactly the moment another client holding the same grant can invalidate it.
|
|
34
|
+
// Codex access tokens last 10 days, so a 5-day lead keeps half the lifetime as margin and refreshes
|
|
35
|
+
// every 5 days. Matches the 5-day `refreshLeadMs` 9router ships for Codex.
|
|
36
|
+
const REFRESH_LEAD_MS = 5 * 24 * 60 * 60 * 1000;
|
|
32
37
|
const REFRESH_TIMEOUT_MS = 10_000;
|
|
33
38
|
const flights = new Map();
|
|
34
39
|
let refreshClient = defaultRefreshClient;
|
|
@@ -19,10 +19,10 @@ const ProviderCreate = z.object({
|
|
|
19
19
|
apiKey: z.string().min(1).max(20000).optional(),
|
|
20
20
|
customHeaders: z.record(z.string(), z.string()).optional(),
|
|
21
21
|
enabled: z.boolean().optional(),
|
|
22
|
-
connectTimeoutMs: z.number().int().min(100).max(
|
|
23
|
-
firstTokenTimeoutMs: z.number().int().min(100).max(
|
|
24
|
-
streamIdleTimeoutMs: z.number().int().min(100).max(
|
|
25
|
-
totalTimeoutMs: z.number().int().min(1000).max(
|
|
22
|
+
connectTimeoutMs: z.number().int().min(100).max(120000).optional(),
|
|
23
|
+
firstTokenTimeoutMs: z.number().int().min(100).max(600000).optional(),
|
|
24
|
+
streamIdleTimeoutMs: z.number().int().min(100).max(900000).optional(),
|
|
25
|
+
totalTimeoutMs: z.number().int().min(1000).max(1800000).optional(),
|
|
26
26
|
maxRetries: z.number().int().min(0).max(8).optional(),
|
|
27
27
|
cbFailureThreshold: z.number().int().min(1).max(50).optional(),
|
|
28
28
|
cbCooldownSeconds: z.number().int().min(1).max(3600).optional(),
|
|
@@ -106,10 +106,10 @@ export async function registerProviderRoutes(app) {
|
|
|
106
106
|
customHeadersEncrypted: headersEnc?.ciphertext ?? null,
|
|
107
107
|
customHeadersNonce: headersEnc?.nonce ?? null,
|
|
108
108
|
enabled: body.enabled ?? true,
|
|
109
|
-
connectTimeoutMs: body.connectTimeoutMs ??
|
|
110
|
-
firstTokenTimeoutMs: body.firstTokenTimeoutMs ??
|
|
111
|
-
streamIdleTimeoutMs: body.streamIdleTimeoutMs ??
|
|
112
|
-
totalTimeoutMs: body.totalTimeoutMs ??
|
|
109
|
+
connectTimeoutMs: body.connectTimeoutMs ?? 15000,
|
|
110
|
+
firstTokenTimeoutMs: body.firstTokenTimeoutMs ?? 60000,
|
|
111
|
+
streamIdleTimeoutMs: body.streamIdleTimeoutMs ?? 300000,
|
|
112
|
+
totalTimeoutMs: body.totalTimeoutMs ?? 600000,
|
|
113
113
|
maxRetries: body.maxRetries ?? 2,
|
|
114
114
|
cbFailureThreshold: body.cbFailureThreshold ?? 5,
|
|
115
115
|
cbCooldownSeconds: body.cbCooldownSeconds ?? 60,
|
|
@@ -148,15 +148,18 @@ function logUpstreamResponse(requestId, status, statusText, headers, durationMs,
|
|
|
148
148
|
*/
|
|
149
149
|
export async function callUpstreamStreaming(cfg, url, payload, onChunk, requestId = '-') {
|
|
150
150
|
const ctl = new AbortController();
|
|
151
|
-
const totalTimer = setTimeout(() => ctl.abort(), cfg.totalTimeoutMs);
|
|
152
151
|
const start = Date.now();
|
|
153
152
|
let ttft = null;
|
|
154
153
|
let firstTokenTimer = null;
|
|
155
154
|
let idleTimer = null;
|
|
155
|
+
// Which watchdog aborted, so the error names the limit that was actually hit: the total
|
|
156
|
+
// timer used to surface as "stream idle timeout", which sent operators tuning the wrong knob.
|
|
157
|
+
let abortedBy = 'total';
|
|
158
|
+
const totalTimer = setTimeout(() => { abortedBy = 'total'; ctl.abort(); }, cfg.totalTimeoutMs);
|
|
156
159
|
const resetIdle = () => {
|
|
157
160
|
if (idleTimer)
|
|
158
161
|
clearTimeout(idleTimer);
|
|
159
|
-
idleTimer = setTimeout(() => ctl.abort(), cfg.streamIdleTimeoutMs);
|
|
162
|
+
idleTimer = setTimeout(() => { abortedBy = 'idle'; ctl.abort(); }, cfg.streamIdleTimeoutMs);
|
|
160
163
|
};
|
|
161
164
|
try {
|
|
162
165
|
const res = await fetch(url, {
|
|
@@ -172,7 +175,7 @@ export async function callUpstreamStreaming(cfg, url, payload, onChunk, requestI
|
|
|
172
175
|
}
|
|
173
176
|
logUpstreamResponse(requestId, res.status, res.statusText, res.headers, Date.now() - start, '');
|
|
174
177
|
// First-token watchdog
|
|
175
|
-
firstTokenTimer = setTimeout(() => ctl.abort(), cfg.firstTokenTimeoutMs);
|
|
178
|
+
firstTokenTimer = setTimeout(() => { abortedBy = 'first_token'; ctl.abort(); }, cfg.firstTokenTimeoutMs);
|
|
176
179
|
resetIdle();
|
|
177
180
|
const reader = res.body.getReader();
|
|
178
181
|
const decoder = new TextDecoder();
|
|
@@ -233,7 +236,10 @@ export async function callUpstreamStreaming(cfg, url, payload, onChunk, requestI
|
|
|
233
236
|
...formatError(e),
|
|
234
237
|
]);
|
|
235
238
|
if (err.name === 'AbortError') {
|
|
236
|
-
if (
|
|
239
|
+
if (abortedBy === 'total') {
|
|
240
|
+
throw new GatewayError('timeout_error', 'Upstream total timeout', { status: 504, cause: e });
|
|
241
|
+
}
|
|
242
|
+
if (abortedBy === 'first_token' && ttft === null) {
|
|
237
243
|
throw new GatewayError('timeout_error', 'Upstream first token timeout', { status: 504, cause: e });
|
|
238
244
|
}
|
|
239
245
|
throw new GatewayError('timeout_error', 'Upstream stream idle timeout', { status: 504, cause: e });
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
-- 0009_raise_timeout_defaults.sql
|
|
2
|
+
-- Raise the upstream timeout defaults. Long reasoning streams sit idle well past 60s between
|
|
3
|
+
-- SSE frames, and the 180s total ceiling cut them off: the production request log showed 55
|
|
4
|
+
-- `timeout_error` rows all ending at exactly 180s with a first token already delivered, i.e.
|
|
5
|
+
-- the total timer firing while the stream was still alive.
|
|
6
|
+
--
|
|
7
|
+
-- New defaults: connect 15s, first token 60s, stream idle 300s, total 600s.
|
|
8
|
+
-- Only rows still on the previous defaults are touched, so an admin-tuned value survives.
|
|
9
|
+
-- The column DEFAULT in `providers` stays at the old values (SQLite cannot ALTER a default
|
|
10
|
+
-- without a full table rebuild) — the application always writes the timeout columns
|
|
11
|
+
-- explicitly, so the column default only applies to raw SQL inserts.
|
|
12
|
+
UPDATE providers SET connect_timeout_ms = 15000 WHERE connect_timeout_ms = 10000;
|
|
13
|
+
UPDATE providers SET first_token_timeout_ms = 60000 WHERE first_token_timeout_ms = 30000;
|
|
14
|
+
UPDATE providers SET stream_idle_timeout_ms = 300000 WHERE stream_idle_timeout_ms = 60000;
|
|
15
|
+
UPDATE providers SET total_timeout_ms = 600000 WHERE total_timeout_ms = 180000;
|