free-coding-models 0.5.55 โ 0.5.56
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# Changelog v0.5.56 - 2026-07-25
|
|
2
|
+
|
|
3
|
+
### Fixed
|
|
4
|
+
- ๐ **Router stream-stall failover after partial response** ([#137](https://github.com/vava-nessa/free-coding-models/issues/137)) โ previously, if a model streamed some data then stalled (per-chunk timeout), the client received a truncated response with no automatic retry. Now the router:
|
|
5
|
+
1. Emits a synthetic SSE error event in OpenAI format (`code: fcm_stream_failover`) so the client knows the stream was truncated and that failover is happening.
|
|
6
|
+
2. Returns `failoverToNext: true` to the outer retry loop, which tries the next model on a fresh upstream connection.
|
|
7
|
+
3. The new model's chunks are appended to the same response object โ the client sees one continuous stream with a clear mid-stream error marker.
|
|
8
|
+
4. On a mid-stream failover the router skips `res.writeHead()` (headers were already sent by the previous model) and emits an SSE comment `: fcm-router-failover-from=<oldKey>` so debuggers can see which model served which segment.
|
|
9
|
+
|
|
10
|
+
**Scope of the fix:** only `stream_stall_timeout` and `timeout` errors trigger mid-stream failover. Generic upstream errors (malformed JSON, connection reset, destroy) still close cleanly with no failover โ their partial data is more likely to be invalid, so re-streaming from another model wouldn't help.
|
|
11
|
+
|
|
12
|
+
### Maintenance
|
|
13
|
+
- ๐งช **+1 regression test** (`test/test.js`) locking in the new failover behavior with a stalled-stream mock provider. The existing "does not retry after partial output" test still passes โ it uses `res.destroy()` which is *not* a stall, so it correctly keeps the old behaviour.
|
|
14
|
+
- ๐งช **591/591 tests pass** (`pnpm test`).
|
|
15
|
+
- ๐งน `vite build` succeeds.
|
|
16
|
+
|
|
17
|
+
### How clients see this
|
|
18
|
+
A streaming request that stalls now produces something like:
|
|
19
|
+
```
|
|
20
|
+
data: {"choices":[{"delta":{"content":"partial"}}]}
|
|
21
|
+
|
|
22
|
+
data: {"error":{"message":"Stream truncated by router due to upstream stream_stall_timeout; failing over to next model.","type":"stream_error","code":"fcm_stream_failover","reason":"stream_stall_timeout"}}
|
|
23
|
+
|
|
24
|
+
data: {"choices":[{"delta":{"content":"fresh answer from fallback"}}]}
|
|
25
|
+
|
|
26
|
+
data: [DONE]
|
|
27
|
+
```
|
|
28
|
+
Plus an SSE comment marker on each failover segment.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "free-coding-models",
|
|
3
|
-
"version": "0.5.
|
|
3
|
+
"version": "0.5.56",
|
|
4
4
|
"description": "Find the fastest coding LLM models in seconds โ ping free models from multiple providers, pick the best one for OpenCode, Cursor, or any AI coding assistant.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"nvidia",
|
|
@@ -2194,11 +2194,23 @@ class RouterRuntime {
|
|
|
2194
2194
|
}
|
|
2195
2195
|
|
|
2196
2196
|
if (res.writableEnded) return { done: true }
|
|
2197
|
-
|
|
2198
|
-
|
|
2199
|
-
|
|
2200
|
-
|
|
2201
|
-
|
|
2197
|
+
// ๐ Issue #137: when the previous model sent partial data, headers are
|
|
2198
|
+
// ๐ already on the wire โ re-calling writeHead throws ERR_HTTP_HEADERS_SENT.
|
|
2199
|
+
// ๐ On a mid-stream failover we just append chunks to the existing response.
|
|
2200
|
+
if (!res.headersSent) {
|
|
2201
|
+
res.writeHead(response.status, {
|
|
2202
|
+
...headerEntries(response.headers),
|
|
2203
|
+
'x-fcm-router-model': key,
|
|
2204
|
+
'x-request-id': requestId,
|
|
2205
|
+
})
|
|
2206
|
+
} else {
|
|
2207
|
+
// ๐ Reflect the new model in trailer-ish debug headers. Node won't let
|
|
2208
|
+
// ๐ us add new headers after send, but we still update x-fcm-router-model
|
|
2209
|
+
// ๐ semantics via a leading SSE comment so clients can see the switch.
|
|
2210
|
+
try {
|
|
2211
|
+
res.write(`: fcm-router-failover-from=${key}\n\n`)
|
|
2212
|
+
} catch { /* best-effort */ }
|
|
2213
|
+
}
|
|
2202
2214
|
sentToClient = true
|
|
2203
2215
|
res.write(firstChunkBuffer)
|
|
2204
2216
|
|
|
@@ -2230,6 +2242,12 @@ class RouterRuntime {
|
|
|
2230
2242
|
return { done: true }
|
|
2231
2243
|
}
|
|
2232
2244
|
const reason = error.name === 'AbortError' ? 'timeout' : (error.message || String(error))
|
|
2245
|
+
// ๐ Issue #137: stream-stall timeouts get a special tag so we can
|
|
2246
|
+
// ๐ distinguish them from generic upstream errors below. Only stalls
|
|
2247
|
+
// ๐ should trigger failover after a partial response โ generic errors
|
|
2248
|
+
// ๐ (malformed JSON, network reset, etc.) usually mean the partial
|
|
2249
|
+
// ๐ data is invalid anyway, so closing cleanly is safer.
|
|
2250
|
+
const isStall = reason === 'stream_stall_timeout' || reason === 'timeout'
|
|
2233
2251
|
this.markFailure(key, reason)
|
|
2234
2252
|
if (reason !== 'timeout') {
|
|
2235
2253
|
this.recordRouterError('upstream_stream_error', requestId, { model: key, reason, partial: sentToClient })
|
|
@@ -2238,6 +2256,31 @@ class RouterRuntime {
|
|
|
2238
2256
|
}
|
|
2239
2257
|
this.addRequestLog({ request_id: requestId, model: key, status: 'ERR', latency_ms: null, tokens: 0, failover: attemptIndex > 0, error: reason, stream: true })
|
|
2240
2258
|
if (sentToClient) {
|
|
2259
|
+
if (isStall) {
|
|
2260
|
+
// ๐ Issue #137: failover even after a partial response. Emit a
|
|
2261
|
+
// ๐ synthetic SSE error event in OpenAI format so clients know the
|
|
2262
|
+
// ๐ stream was truncated and that the router is failing over. The
|
|
2263
|
+
// ๐ outer retry loop will then try the next model on a fresh
|
|
2264
|
+
// ๐ upstream connection; its chunks are appended to the same
|
|
2265
|
+
// ๐ response object so the client sees one continuous stream.
|
|
2266
|
+
this.logger.warn(`Stream stall after partial response from ${key}, attempting failover`, { request_id: requestId, reason })
|
|
2267
|
+
if (!res.writableEnded) {
|
|
2268
|
+
try {
|
|
2269
|
+
const errorPayload = JSON.stringify({
|
|
2270
|
+
error: {
|
|
2271
|
+
message: `Stream truncated by router due to upstream ${reason}; failing over to next model.`,
|
|
2272
|
+
type: 'stream_error',
|
|
2273
|
+
code: 'fcm_stream_failover',
|
|
2274
|
+
reason,
|
|
2275
|
+
},
|
|
2276
|
+
})
|
|
2277
|
+
res.write(`data: ${errorPayload}\n\n`)
|
|
2278
|
+
} catch { /* best-effort */ }
|
|
2279
|
+
}
|
|
2280
|
+
return { done: false, failoverToNext: true, reason: `stream_stall_${reason}` }
|
|
2281
|
+
}
|
|
2282
|
+
// ๐ Non-stall errors after partial output: keep existing behaviour
|
|
2283
|
+
// ๐ (close cleanly, no failover) to avoid sending malformed data.
|
|
2241
2284
|
this.logger.warn(`Streaming failure after partial response from ${key}`, { request_id: requestId, reason })
|
|
2242
2285
|
try { if (!res.writableEnded) res.end() } catch {}
|
|
2243
2286
|
return { done: true }
|