ldrouter 1.17.7 → 1.17.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/dist/server/db/migrate.js +4 -4
- package/dist/server/db/schema.js +4 -4
- package/dist/server/gateway/runner.js +10 -3
- package/dist/server/routes/admin/models.js +3 -2
- package/dist/server/routes/admin/providers.js +8 -8
- package/dist/server/routing/circuit.js +13 -0
- package/dist/server/upstream/client.js +10 -4
- package/migrations/0009_raise_timeout_defaults.sql +15 -0
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,20 @@ All notable changes to this project are documented here. The format follows
|
|
|
4
4
|
[Keep a Changelog](https://keepachangelog.com/) and the project adheres to
|
|
5
5
|
[Semantic Versioning](https://semver.org/).
|
|
6
6
|
|
|
7
|
+
## [1.17.9] - 2026-09-22
|
|
8
|
+
|
|
9
|
+
### Fixed
|
|
10
|
+
|
|
11
|
+
- **A tripped provider circuit breaker locked the provider out until the process restarted.** The breaker's `open -> half_open` decay runs inside `getEffectiveState()`, which is only called from the attempt loop, but both candidate filters read the raw `isOpen()` flag instead — so once the flag was set no candidate ever reached the attempt loop that would have decayed it, and the provider answered `provider circuit is open` to every request until the container was restarted. `health_state` kept reading `healthy` throughout, so nothing in the UI showed the lockout. Live production: `code` and `qo` had each accumulated five consecutive upstream failures (a real upstream `503`) and were both in that state, while the upstream itself answered `200` to the same model and key when called directly. Candidate construction is now cooldown-aware through `circuitBlocks()`, so a candidate is admitted again as soon as the cooldown has elapsed and the half-open probe can close the breaker. Verified on a container against the released image: breaker open with the cooldown elapsed returned a zero-length `200`; with the fix it streams normally again.
|
|
12
|
+
- **The admin "test model" button reported `Stream ended unexpectedly` instead of the actual reason.** The `test-stream` route hijacks the reply, so a `GatewayError` rethrown from its catch block could never reach Fastify's error handler — the response simply ended as `HTTP 200` with a zero-length body and no events, leaving the UI nothing to parse but that placeholder. The route now sends the reason as a `test_error` event. Measured on the released build with the breaker open: status `200`, `content-type` null, `body` 0 bytes; with the fix: `text/event-stream` carrying the real message.
|
|
13
|
+
|
|
14
|
+
## [1.17.8] - 2026-09-21
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
|
|
18
|
+
- Raised provider connection, first-token, stream-idle, and total request timeout defaults to 15s, 60s, 300s, and 600s. Existing providers still using the old defaults are upgraded by migration 0009.
|
|
19
|
+
- Distinguish total request timeout errors from stream-idle timeout errors.
|
|
20
|
+
|
|
7
21
|
## [1.17.7] - 2026-09-20
|
|
8
22
|
|
|
9
23
|
### Fixed
|
|
@@ -148,10 +148,10 @@ function buildInitialSchemaSql() {
|
|
|
148
148
|
custom_headers_encrypted TEXT,
|
|
149
149
|
custom_headers_nonce TEXT,
|
|
150
150
|
enabled INTEGER NOT NULL DEFAULT 1,
|
|
151
|
-
connect_timeout_ms INTEGER NOT NULL DEFAULT
|
|
152
|
-
first_token_timeout_ms INTEGER NOT NULL DEFAULT
|
|
153
|
-
stream_idle_timeout_ms INTEGER NOT NULL DEFAULT
|
|
154
|
-
total_timeout_ms INTEGER NOT NULL DEFAULT
|
|
151
|
+
connect_timeout_ms INTEGER NOT NULL DEFAULT 15000,
|
|
152
|
+
first_token_timeout_ms INTEGER NOT NULL DEFAULT 60000,
|
|
153
|
+
stream_idle_timeout_ms INTEGER NOT NULL DEFAULT 300000,
|
|
154
|
+
total_timeout_ms INTEGER NOT NULL DEFAULT 600000,
|
|
155
155
|
max_retries INTEGER NOT NULL DEFAULT 2,
|
|
156
156
|
retry_base_ms INTEGER NOT NULL DEFAULT 500,
|
|
157
157
|
retry_max_ms INTEGER NOT NULL DEFAULT 8000,
|
package/dist/server/db/schema.js
CHANGED
|
@@ -89,10 +89,10 @@ export const providers = sqliteTable('providers', {
|
|
|
89
89
|
customHeadersEncrypted: text('custom_headers_encrypted'),
|
|
90
90
|
customHeadersNonce: text('custom_headers_nonce'),
|
|
91
91
|
enabled: integer('enabled', { mode: 'boolean' }).notNull().notNull().default(true),
|
|
92
|
-
connectTimeoutMs: integer('connect_timeout_ms').notNull().default(
|
|
93
|
-
firstTokenTimeoutMs: integer('first_token_timeout_ms').notNull().default(
|
|
94
|
-
streamIdleTimeoutMs: integer('stream_idle_timeout_ms').notNull().default(
|
|
95
|
-
totalTimeoutMs: integer('total_timeout_ms').notNull().default(
|
|
92
|
+
connectTimeoutMs: integer('connect_timeout_ms').notNull().default(15000),
|
|
93
|
+
firstTokenTimeoutMs: integer('first_token_timeout_ms').notNull().default(60000),
|
|
94
|
+
streamIdleTimeoutMs: integer('stream_idle_timeout_ms').notNull().default(300000),
|
|
95
|
+
totalTimeoutMs: integer('total_timeout_ms').notNull().default(600000),
|
|
96
96
|
maxRetries: integer('max_retries').notNull().default(2),
|
|
97
97
|
retryBaseMs: integer('retry_base_ms').notNull().default(500),
|
|
98
98
|
retryMaxMs: integer('retry_max_ms').notNull().default(8000),
|
|
@@ -11,7 +11,7 @@ import { qoderAttemptFailure, qoderCredentialsFor } from '../providers/qoder/cre
|
|
|
11
11
|
import { refreshQoderCreditsIfStale } from '../providers/qoder/credits.js';
|
|
12
12
|
import { callQoderNonStreaming, callQoderStreaming } from '../providers/qoder/client.js';
|
|
13
13
|
import { refreshCodexUsageIfStale } from '../providers/codex-autostart.js';
|
|
14
|
-
import { getEffectiveState,
|
|
14
|
+
import { getEffectiveState, recordSuccess, recordFailure, halfOpenProbeAllowed, circuitBlocks } from '../routing/circuit.js';
|
|
15
15
|
import { checkRpm, checkTpm, acquireConcurrent, releaseConcurrent } from '../routing/ratelimit.js';
|
|
16
16
|
import { checkDailyMonthly, consumeUsage } from '../routing/quota.js';
|
|
17
17
|
import { keyAllowedFor } from '../auth/api-key.js';
|
|
@@ -436,7 +436,10 @@ export class GatewayRunner {
|
|
|
436
436
|
enabled: m.enabled,
|
|
437
437
|
providerEnabled: p.enabled,
|
|
438
438
|
upstreamAvailable: m.upstreamAvailable,
|
|
439
|
-
|
|
439
|
+
// Cooldown-aware, not the raw flag: `isOpen()` stays true until some request decays it,
|
|
440
|
+
// so a raw read here refused every candidate forever and the provider was unroutable
|
|
441
|
+
// until a restart.
|
|
442
|
+
circuitOpen: circuitBlocks(m.providerId, p.cbCooldownSeconds),
|
|
440
443
|
capabilities: caps,
|
|
441
444
|
providerType: p.type,
|
|
442
445
|
};
|
|
@@ -479,6 +482,7 @@ export class GatewayRunner {
|
|
|
479
482
|
const providers = db.select().from(schema.providers).all();
|
|
480
483
|
const providerEnabled = new Map(providers.map((p) => [p.id, p.enabled]));
|
|
481
484
|
const providerType = new Map(providers.map((p) => [p.id, p.type]));
|
|
485
|
+
const providerCooldown = new Map(providers.map((p) => [p.id, p.cbCooldownSeconds]));
|
|
482
486
|
// Deliberately unfiltered: every combo member must reach selectCandidates so
|
|
483
487
|
// it can report WHY it was skipped. Pre-filtering here erased the model rows
|
|
484
488
|
// and turned every distinct reason into "model not found".
|
|
@@ -489,7 +493,10 @@ export class GatewayRunner {
|
|
|
489
493
|
enabled: m.enabled,
|
|
490
494
|
providerEnabled: providerEnabled.get(m.providerId),
|
|
491
495
|
upstreamAvailable: m.upstreamAvailable,
|
|
492
|
-
|
|
496
|
+
// Cooldown-aware: a raw `isOpen()` here froze the whole combo on one bad minute,
|
|
497
|
+
// because the flags are computed once per request while the breaker stays open
|
|
498
|
+
// until some request decays it — which this filter prevented from ever happening.
|
|
499
|
+
circuitOpen: circuitBlocks(m.providerId, providerCooldown.get(m.providerId) ?? 0),
|
|
493
500
|
capabilities: safeJson(m.capabilitiesJson),
|
|
494
501
|
providerType: providerType.get(m.providerId),
|
|
495
502
|
}));
|
|
@@ -312,8 +312,9 @@ export async function registerModelRoutes(app) {
|
|
|
312
312
|
catch (e) {
|
|
313
313
|
// Never let a raw non-GatewayError (e.g. MasterKeyError from credential
|
|
314
314
|
// decryption) escape into the global handler as an opaque "Gateway error".
|
|
315
|
-
|
|
316
|
-
|
|
315
|
+
// The reply is hijacked, so a rethrow here cannot reach the error handler:
|
|
316
|
+
// it ends the response having sent nothing, which the UI can only report as
|
|
317
|
+
// "Stream ended unexpectedly". Report the reason over the stream instead.
|
|
317
318
|
send('test_error', { message: e.message });
|
|
318
319
|
}
|
|
319
320
|
finally {
|
|
@@ -19,10 +19,10 @@ const ProviderCreate = z.object({
|
|
|
19
19
|
apiKey: z.string().min(1).max(20000).optional(),
|
|
20
20
|
customHeaders: z.record(z.string(), z.string()).optional(),
|
|
21
21
|
enabled: z.boolean().optional(),
|
|
22
|
-
connectTimeoutMs: z.number().int().min(100).max(
|
|
23
|
-
firstTokenTimeoutMs: z.number().int().min(100).max(
|
|
24
|
-
streamIdleTimeoutMs: z.number().int().min(100).max(
|
|
25
|
-
totalTimeoutMs: z.number().int().min(1000).max(
|
|
22
|
+
connectTimeoutMs: z.number().int().min(100).max(120000).optional(),
|
|
23
|
+
firstTokenTimeoutMs: z.number().int().min(100).max(600000).optional(),
|
|
24
|
+
streamIdleTimeoutMs: z.number().int().min(100).max(900000).optional(),
|
|
25
|
+
totalTimeoutMs: z.number().int().min(1000).max(1800000).optional(),
|
|
26
26
|
maxRetries: z.number().int().min(0).max(8).optional(),
|
|
27
27
|
cbFailureThreshold: z.number().int().min(1).max(50).optional(),
|
|
28
28
|
cbCooldownSeconds: z.number().int().min(1).max(3600).optional(),
|
|
@@ -106,10 +106,10 @@ export async function registerProviderRoutes(app) {
|
|
|
106
106
|
customHeadersEncrypted: headersEnc?.ciphertext ?? null,
|
|
107
107
|
customHeadersNonce: headersEnc?.nonce ?? null,
|
|
108
108
|
enabled: body.enabled ?? true,
|
|
109
|
-
connectTimeoutMs: body.connectTimeoutMs ??
|
|
110
|
-
firstTokenTimeoutMs: body.firstTokenTimeoutMs ??
|
|
111
|
-
streamIdleTimeoutMs: body.streamIdleTimeoutMs ??
|
|
112
|
-
totalTimeoutMs: body.totalTimeoutMs ??
|
|
109
|
+
connectTimeoutMs: body.connectTimeoutMs ?? 15000,
|
|
110
|
+
firstTokenTimeoutMs: body.firstTokenTimeoutMs ?? 60000,
|
|
111
|
+
streamIdleTimeoutMs: body.streamIdleTimeoutMs ?? 300000,
|
|
112
|
+
totalTimeoutMs: body.totalTimeoutMs ?? 600000,
|
|
113
113
|
maxRetries: body.maxRetries ?? 2,
|
|
114
114
|
cbFailureThreshold: body.cbFailureThreshold ?? 5,
|
|
115
115
|
cbCooldownSeconds: body.cbCooldownSeconds ?? 60,
|
|
@@ -35,3 +35,16 @@ export function getEffectiveState(providerId, cooldownSeconds) {
|
|
|
35
35
|
export function halfOpenProbeAllowed(providerId) {
|
|
36
36
|
return circuitState(providerId) === 'half_open';
|
|
37
37
|
}
|
|
38
|
+
/**
|
|
39
|
+
* Should the candidate filter refuse this provider outright?
|
|
40
|
+
*
|
|
41
|
+
* `isOpen()` reports the raw state, which stays `'open'` until a request walks the
|
|
42
|
+
* attempt loop — so a tripped breaker rejected every candidate for the lifetime of the
|
|
43
|
+
* process and made the provider unroutable until a restart (the combo path and the
|
|
44
|
+
* direct-model path both read the flag off the candidate record). A candidate must be
|
|
45
|
+
* allowed through once the cooldown has elapsed so the probe that closes the breaker
|
|
46
|
+
* can actually run; `getEffectiveState` is what performs that open -> half_open decay.
|
|
47
|
+
*/
|
|
48
|
+
export function circuitBlocks(providerId, cooldownSeconds) {
|
|
49
|
+
return getEffectiveState(providerId, cooldownSeconds) === 'open';
|
|
50
|
+
}
|
|
@@ -148,15 +148,18 @@ function logUpstreamResponse(requestId, status, statusText, headers, durationMs,
|
|
|
148
148
|
*/
|
|
149
149
|
export async function callUpstreamStreaming(cfg, url, payload, onChunk, requestId = '-') {
|
|
150
150
|
const ctl = new AbortController();
|
|
151
|
-
const totalTimer = setTimeout(() => ctl.abort(), cfg.totalTimeoutMs);
|
|
152
151
|
const start = Date.now();
|
|
153
152
|
let ttft = null;
|
|
154
153
|
let firstTokenTimer = null;
|
|
155
154
|
let idleTimer = null;
|
|
155
|
+
// Which watchdog aborted, so the error names the limit that was actually hit: the total
|
|
156
|
+
// timer used to surface as "stream idle timeout", which sent operators tuning the wrong knob.
|
|
157
|
+
let abortedBy = 'total';
|
|
158
|
+
const totalTimer = setTimeout(() => { abortedBy = 'total'; ctl.abort(); }, cfg.totalTimeoutMs);
|
|
156
159
|
const resetIdle = () => {
|
|
157
160
|
if (idleTimer)
|
|
158
161
|
clearTimeout(idleTimer);
|
|
159
|
-
idleTimer = setTimeout(() => ctl.abort(), cfg.streamIdleTimeoutMs);
|
|
162
|
+
idleTimer = setTimeout(() => { abortedBy = 'idle'; ctl.abort(); }, cfg.streamIdleTimeoutMs);
|
|
160
163
|
};
|
|
161
164
|
try {
|
|
162
165
|
const res = await fetch(url, {
|
|
@@ -172,7 +175,7 @@ export async function callUpstreamStreaming(cfg, url, payload, onChunk, requestI
|
|
|
172
175
|
}
|
|
173
176
|
logUpstreamResponse(requestId, res.status, res.statusText, res.headers, Date.now() - start, '');
|
|
174
177
|
// First-token watchdog
|
|
175
|
-
firstTokenTimer = setTimeout(() => ctl.abort(), cfg.firstTokenTimeoutMs);
|
|
178
|
+
firstTokenTimer = setTimeout(() => { abortedBy = 'first_token'; ctl.abort(); }, cfg.firstTokenTimeoutMs);
|
|
176
179
|
resetIdle();
|
|
177
180
|
const reader = res.body.getReader();
|
|
178
181
|
const decoder = new TextDecoder();
|
|
@@ -233,7 +236,10 @@ export async function callUpstreamStreaming(cfg, url, payload, onChunk, requestI
|
|
|
233
236
|
...formatError(e),
|
|
234
237
|
]);
|
|
235
238
|
if (err.name === 'AbortError') {
|
|
236
|
-
if (
|
|
239
|
+
if (abortedBy === 'total') {
|
|
240
|
+
throw new GatewayError('timeout_error', 'Upstream total timeout', { status: 504, cause: e });
|
|
241
|
+
}
|
|
242
|
+
if (abortedBy === 'first_token' && ttft === null) {
|
|
237
243
|
throw new GatewayError('timeout_error', 'Upstream first token timeout', { status: 504, cause: e });
|
|
238
244
|
}
|
|
239
245
|
throw new GatewayError('timeout_error', 'Upstream stream idle timeout', { status: 504, cause: e });
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
-- 0009_raise_timeout_defaults.sql
|
|
2
|
+
-- Raise the upstream timeout defaults. Long reasoning streams sit idle well past 60s between
|
|
3
|
+
-- SSE frames, and the 180s total ceiling cut them off: the production request log showed 55
|
|
4
|
+
-- `timeout_error` rows all ending at exactly 180s with a first token already delivered, i.e.
|
|
5
|
+
-- the total timer firing while the stream was still alive.
|
|
6
|
+
--
|
|
7
|
+
-- New defaults: connect 15s, first token 60s, stream idle 300s, total 600s.
|
|
8
|
+
-- Only rows still on the previous defaults are touched, so an admin-tuned value survives.
|
|
9
|
+
-- The column DEFAULT in `providers` stays at the old values (SQLite cannot ALTER a default
|
|
10
|
+
-- without a full table rebuild) — the application always writes the timeout columns
|
|
11
|
+
-- explicitly, so the column default only applies to raw SQL inserts.
|
|
12
|
+
UPDATE providers SET connect_timeout_ms = 15000 WHERE connect_timeout_ms = 10000;
|
|
13
|
+
UPDATE providers SET first_token_timeout_ms = 60000 WHERE first_token_timeout_ms = 30000;
|
|
14
|
+
UPDATE providers SET stream_idle_timeout_ms = 300000 WHERE stream_idle_timeout_ms = 60000;
|
|
15
|
+
UPDATE providers SET total_timeout_ms = 600000 WHERE total_timeout_ms = 180000;
|