free-coding-models 0.5.54 โ†’ 0.5.56

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,17 @@
1
+ # Changelog v0.5.55 - 2026-07-25
2
+
3
+ ### Fixed
4
+ - ๐Ÿณ **Docker "Tool mode save failed" โ€” clear root cause + actionable fix** ([#119](https://github.com/vava-nessa/free-coding-models/issues/119)) โ€” when the `fcm-data` volume was created with files owned by a different UID (older images, host bind-mounts), `chmod -R 777` had no effect because Linux checks file ownership, not mode bits. Two-layer fix:
5
+ 1. `src/core/config.js` โ€” `saveConfig()` now detects `EACCES` / `EPERM` and surfaces a friendly hint via the new exported `formatPermissionHint()` function. The hint explains *why* `chmod` is the wrong tool and gives the exact `docker compose down && docker volume rm <vol> && docker compose up` fix.
6
+ 2. `docker-entrypoint.sh` โ€” proactive ownership check at container startup. If a config file is owned by a different UID than the running `fcm` user, the entrypoint logs a loud `โš ๏ธ WARNING` block *before* the first save attempt fails with a confusing error.
7
+
8
+ ### Closed (upstream, no FCM-side fix possible)
9
+ - ๐Ÿ“Œ **Caveman CLI install fails on Node 26** ([#101](https://github.com/vava-nessa/free-coding-models/issues/101)) โ€” this is upstream in `@juliusbrussee/caveman-code`. `better-sqlite3 v11.10.0` doesn't compile against Node.js v26 (V8 API churn: `GetPrototype` โ†’ `GetPrototypeV2`, etc.). FCM surfaces the npm error verbatim but cannot fix caveman-code's deps. Workaround: use Node v22 or v24 (both LTS, both supported by `better-sqlite3@11`).
10
+
11
+ ### Maintenance
12
+ - ๐Ÿงช **+6 new unit tests** for `formatPermissionHint()` (`test/config-permission-hint.test.js`) โ€” covers empty string for non-permission errors, EACCES/EPERM detection, Docker volume mention, "chmod is not the fix" wording, multi-line output.
13
+
14
+ ### Pre-flight checks
15
+ - 590/590 tests pass (`pnpm test`)
16
+ - `vite build` succeeds
17
+ - All three surfaces (CLI, web dashboard, Docker) compatible
@@ -0,0 +1,28 @@
1
+ # Changelog v0.5.56 - 2026-07-25
2
+
3
+ ### Fixed
4
+ - ๐Ÿ” **Router stream-stall failover after partial response** ([#137](https://github.com/vava-nessa/free-coding-models/issues/137)) โ€” previously, if a model streamed some data then stalled (per-chunk timeout), the client received a truncated response with no automatic retry. Now the router:
5
+ 1. Emits a synthetic SSE error event in OpenAI format (`code: fcm_stream_failover`) so the client knows the stream was truncated and that failover is happening.
6
+ 2. Returns `failoverToNext: true` to the outer retry loop, which tries the next model on a fresh upstream connection.
7
+ 3. The new model's chunks are appended to the same response object โ€” the client sees one continuous stream with a clear mid-stream error marker.
8
+ 4. On a mid-stream failover the router skips `res.writeHead()` (headers were already sent by the previous model) and emits an SSE comment `: fcm-router-failover-from=<oldKey>` so debuggers can see which model served which segment.
9
+
10
+ **Scope of the fix:** only `stream_stall_timeout` and `timeout` errors trigger mid-stream failover. Generic upstream errors (malformed JSON, connection reset, destroy) still close cleanly with no failover โ€” their partial data is more likely to be invalid, so re-streaming from another model wouldn't help.
11
+
12
+ ### Maintenance
13
+ - ๐Ÿงช **+1 regression test** (`test/test.js`) locking in the new failover behavior with a stalled-stream mock provider. The existing "does not retry after partial output" test still passes โ€” it uses `res.destroy()` which is *not* a stall, so it correctly keeps the old behaviour.
14
+ - ๐Ÿงช **591/591 tests pass** (`pnpm test`).
15
+ - ๐Ÿงน `vite build` succeeds.
16
+
17
+ ### How clients see this
18
+ A streaming request that stalls now produces something like:
19
+ ```
20
+ data: {"choices":[{"delta":{"content":"partial"}}]}
21
+
22
+ data: {"error":{"message":"Stream truncated by router due to upstream stream_stall_timeout; failing over to next model.","type":"stream_error","code":"fcm_stream_failover","reason":"stream_stall_timeout"}}
23
+
24
+ data: {"choices":[{"delta":{"content":"fresh answer from fallback"}}]}
25
+
26
+ data: [DONE]
27
+ ```
28
+ Plus an SSE comment marker on each failover segment.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "free-coding-models",
3
- "version": "0.5.54",
3
+ "version": "0.5.56",
4
4
  "description": "Find the fastest coding LLM models in seconds โ€” ping free models from multiple providers, pick the best one for OpenCode, Cursor, or any AI coding assistant.",
5
5
  "keywords": [
6
6
  "nvidia",
@@ -53,7 +53,7 @@
53
53
  ],
54
54
  "scripts": {
55
55
  "start": "node bin/free-coding-models.js",
56
- "test": "node --test test/test.js test/fcm-agent-core.test.js test/patch-openclaw.test.js test/provider-metadata.test.js",
56
+ "test": "node --test test/test.js test/fcm-agent-core.test.js test/patch-openclaw.test.js test/provider-metadata.test.js test/config-permission-hint.test.js",
57
57
  "prepack": "npm run build:web",
58
58
  "dev": "node scripts/dev-web.mjs",
59
59
  "dev:web": "node scripts/dev-web.mjs",
@@ -760,7 +760,16 @@ export function saveConfig(config, options = {}) {
760
760
  // ๐Ÿ“– Write failed - explicit error instead of silent failure
761
761
  let errorMsg = `Failed to write config: ${writeError.message}`
762
762
  try { unlinkSync(tempPath) } catch { /* ignore temp cleanup failures */ }
763
-
763
+
764
+ // ๐Ÿ“– Detect permission errors and surface a Docker-specific hint. This
765
+ // ๐Ÿ“– happens when a volume was created with files owned by another UID
766
+ // ๐Ÿ“– (e.g. older images running as root, or a host bind-mount where the
767
+ // ๐Ÿ“– directory is owned by a different user). chmod 777 does NOT fix this
768
+ // ๐Ÿ“– because Linux checks file ownership, not mode bits, for write access.
769
+ if (writeError?.code === 'EACCES' || writeError?.code === 'EPERM') {
770
+ errorMsg += formatPermissionHint(writeError)
771
+ }
772
+
764
773
  // ๐Ÿ“– Try to restore from backup if we have one
765
774
  if (backupCreated) {
766
775
  try {
@@ -775,6 +784,37 @@ export function saveConfig(config, options = {}) {
775
784
  }
776
785
  }
777
786
 
787
+ /**
788
+ * ๐Ÿ“– formatPermissionHint โ€” build a friendly hint for the most common Docker
789
+ * ๐Ÿ“– permission failure (config file owned by a different UID than the running
790
+ * ๐Ÿ“– process). Returns an empty string for non-permission errors.
791
+ *
792
+ * ๐Ÿ“– Common symptom (issue #119): "Tool mode save failed" in the web dashboard
793
+ * ๐Ÿ“– even after `chmod -R 777`, because chmod doesn't change ownership.
794
+ */
795
+ export function formatPermissionHint(writeError) {
796
+ // ๐Ÿ“– Only emit the hint for actual permission errors โ€” silently no-op for
797
+ // ๐Ÿ“– anything else so the caller can safely concatenate the result.
798
+ if (writeError?.code !== 'EACCES' && writeError?.code !== 'EPERM') return ''
799
+ const lines = []
800
+ lines.push('')
801
+ lines.push(' ๐Ÿ’ก This is almost always a file-ownership issue, not a chmod issue.')
802
+ lines.push(' The config file is owned by a different user than the one running FCM.')
803
+ try {
804
+ const stat = statSync(CONFIG_PATH)
805
+ lines.push(` File owner UID: ${stat.uid}, current process UID: ${process.getuid?.() ?? 'n/a'}`)
806
+ } catch { /* stat may also fail */ }
807
+ lines.push('')
808
+ lines.push(' Fix in Docker:')
809
+ lines.push(' docker compose down')
810
+ lines.push(' docker volume rm <project>_fcm-data # or the volume name from `docker volume ls`')
811
+ lines.push(' docker compose up # recreates the volume with the right UID')
812
+ lines.push('')
813
+ lines.push(' Fix on host (bind mount):')
814
+ lines.push(' sudo chown $(id -u):$(id -g) ~/.free-coding-models.json')
815
+ return '\n' + lines.join('\n')
816
+ }
817
+
778
818
  /**
779
819
  * ๐Ÿ“– createBackup: Creates a timestamped backup of the current config file.
780
820
  * ๐Ÿ“– Keeps only the 5 most recent backups to avoid disk space issues.
@@ -2194,11 +2194,23 @@ class RouterRuntime {
2194
2194
  }
2195
2195
 
2196
2196
  if (res.writableEnded) return { done: true }
2197
- res.writeHead(response.status, {
2198
- ...headerEntries(response.headers),
2199
- 'x-fcm-router-model': key,
2200
- 'x-request-id': requestId,
2201
- })
2197
+ // ๐Ÿ“– Issue #137: when the previous model sent partial data, headers are
2198
+ // ๐Ÿ“– already on the wire โ€” re-calling writeHead throws ERR_HTTP_HEADERS_SENT.
2199
+ // ๐Ÿ“– On a mid-stream failover we just append chunks to the existing response.
2200
+ if (!res.headersSent) {
2201
+ res.writeHead(response.status, {
2202
+ ...headerEntries(response.headers),
2203
+ 'x-fcm-router-model': key,
2204
+ 'x-request-id': requestId,
2205
+ })
2206
+ } else {
2207
+ // ๐Ÿ“– Reflect the new model in trailer-ish debug headers. Node won't let
2208
+ // ๐Ÿ“– us add new headers after send, but we still update x-fcm-router-model
2209
+ // ๐Ÿ“– semantics via a leading SSE comment so clients can see the switch.
2210
+ try {
2211
+ res.write(`: fcm-router-failover-from=${key}\n\n`)
2212
+ } catch { /* best-effort */ }
2213
+ }
2202
2214
  sentToClient = true
2203
2215
  res.write(firstChunkBuffer)
2204
2216
 
@@ -2230,6 +2242,12 @@ class RouterRuntime {
2230
2242
  return { done: true }
2231
2243
  }
2232
2244
  const reason = error.name === 'AbortError' ? 'timeout' : (error.message || String(error))
2245
+ // ๐Ÿ“– Issue #137: stream-stall timeouts get a special tag so we can
2246
+ // ๐Ÿ“– distinguish them from generic upstream errors below. Only stalls
2247
+ // ๐Ÿ“– should trigger failover after a partial response โ€” generic errors
2248
+ // ๐Ÿ“– (malformed JSON, network reset, etc.) usually mean the partial
2249
+ // ๐Ÿ“– data is invalid anyway, so closing cleanly is safer.
2250
+ const isStall = reason === 'stream_stall_timeout' || reason === 'timeout'
2233
2251
  this.markFailure(key, reason)
2234
2252
  if (reason !== 'timeout') {
2235
2253
  this.recordRouterError('upstream_stream_error', requestId, { model: key, reason, partial: sentToClient })
@@ -2238,6 +2256,31 @@ class RouterRuntime {
2238
2256
  }
2239
2257
  this.addRequestLog({ request_id: requestId, model: key, status: 'ERR', latency_ms: null, tokens: 0, failover: attemptIndex > 0, error: reason, stream: true })
2240
2258
  if (sentToClient) {
2259
+ if (isStall) {
2260
+ // ๐Ÿ“– Issue #137: failover even after a partial response. Emit a
2261
+ // ๐Ÿ“– synthetic SSE error event in OpenAI format so clients know the
2262
+ // ๐Ÿ“– stream was truncated and that the router is failing over. The
2263
+ // ๐Ÿ“– outer retry loop will then try the next model on a fresh
2264
+ // ๐Ÿ“– upstream connection; its chunks are appended to the same
2265
+ // ๐Ÿ“– response object so the client sees one continuous stream.
2266
+ this.logger.warn(`Stream stall after partial response from ${key}, attempting failover`, { request_id: requestId, reason })
2267
+ if (!res.writableEnded) {
2268
+ try {
2269
+ const errorPayload = JSON.stringify({
2270
+ error: {
2271
+ message: `Stream truncated by router due to upstream ${reason}; failing over to next model.`,
2272
+ type: 'stream_error',
2273
+ code: 'fcm_stream_failover',
2274
+ reason,
2275
+ },
2276
+ })
2277
+ res.write(`data: ${errorPayload}\n\n`)
2278
+ } catch { /* best-effort */ }
2279
+ }
2280
+ return { done: false, failoverToNext: true, reason: `stream_stall_${reason}` }
2281
+ }
2282
+ // ๐Ÿ“– Non-stall errors after partial output: keep existing behaviour
2283
+ // ๐Ÿ“– (close cleanly, no failover) to avoid sending malformed data.
2241
2284
  this.logger.warn(`Streaming failure after partial response from ${key}`, { request_id: requestId, reason })
2242
2285
  try { if (!res.writableEnded) res.end() } catch {}
2243
2286
  return { done: true }