mslxdff 0.1.5 → 0.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/mslxdff.js CHANGED
@@ -1,19 +1,20 @@
1
1
  #!/usr/bin/env node
2
2
  import { execFile } from "node:child_process";
3
- import { readFileSync, existsSync } from "node:fs";
3
+ import { readFileSync, existsSync, statSync, watch, openSync } from "node:fs";
4
4
  import { fileURLToPath } from "node:url";
5
5
  import { dirname, join } from "node:path";
6
6
  import { startServer, resolvePort } from "../src/server.js";
7
7
  import { createRouter } from "../src/routes.js";
8
8
  import { createUpstreamClient } from "../src/upstream.js";
9
9
  import { createModelsService } from "../src/models.js";
10
- import { loadToken, refreshToken, setPort, getPort, loadGroupsJoined, saveGroupsJoined } from "../src/state.js";
10
+ import { loadToken, refreshToken, setPort, getPort, loadGroupsJoined, saveGroupsJoined, loadModelErrors } from "../src/state.js";
11
11
  import { startDaemon, stopDaemon, writePid, pidFile, logFile, readPid } from "../src/daemon.js";
12
12
  import { createAutoSelector } from "../src/auto.js";
13
13
  import { createPeersService } from "../src/peers.js";
14
14
  import { createGroupsService, createBansService, refreshGroupMembers, syncPeersFromMembers } from "../src/groups.js";
15
- import { logDir, recentCalls, lastError, appendCall, appendError } from "../src/logs.js";
16
- const logs = { appendCall, appendError };
15
+ import { logDir, recentCalls, lastError, appendCall, appendError, appendEvent, eventsFile, recentEvents } from "../src/logs.js";
16
+
17
+ const logs = { appendCall, appendError, appendEvent };
17
18
 
18
19
  const args = process.argv.slice(2);
19
20
  const VERSION = JSON.parse(readFileSync(join(dirname(fileURLToPath(import.meta.url)), "..", "package.json"), "utf8")).version;
@@ -78,8 +79,28 @@ if (args.includes("-model") || args.includes("-models")) {
78
79
  }
79
80
  process.exit(0);
80
81
  }
82
+ if (sub === "status") {
83
+ const statuses = loadModelErrors();
84
+ const cacheFile = join(logDir(), "models.json");
85
+ const cached = readModelsCache(cacheFile);
86
+ const ids = new Set([
87
+ ...(cached?.data || []).map((m) => m.id),
88
+ ...Object.keys(statuses),
89
+ ]);
90
+ for (const id of ids) {
91
+ const e = statuses[id];
92
+ const st = typeof e === "number" ? "error" : e?.status || "normal";
93
+ const at = typeof e === "number" ? e : e?.at;
94
+ const when = at
95
+ ? ` (${new Date(at).toISOString().slice(5, 19).replace("T", " ")})`
96
+ : "";
97
+ const extra = e?.code ? ` HTTP ${e.code}` : "";
98
+ console.log(` ${id} ${st}${when}${extra}`);
99
+ }
100
+ process.exit(0);
101
+ }
81
102
  if (sub !== undefined && sub !== "list") {
82
- console.error("usage: mslxdff -model list | mslxdff -model refresh");
103
+ console.error("usage: mslxdff -model list | mslxdff -model status | mslxdff -model refresh");
83
104
  process.exit(1);
84
105
  }
85
106
  const cacheFile = join(logDir(), "models.json");
@@ -109,6 +130,13 @@ if (args.includes("-model") || args.includes("-models")) {
109
130
  process.exit(0);
110
131
  }
111
132
 
133
+ // -debug: live-follow the daemon's event stream (requests, upstream errors,
134
+ // peer forwards, peer errors, results) for investigation
135
+ if (args.includes("-debug") || args.includes("--debug")) {
136
+ await liveDebug();
137
+ process.exit(0);
138
+ }
139
+
112
140
  // -creategroup <name> | -group create <name> | -group sync | -group leave <name> | -group list |
113
141
  // -addtogroup <leader-host> <name> | -resetban [ip]
114
142
  const createGroupArg = argValue("-creategroup", "--creategroup") || groupIs("create", args);
@@ -230,7 +258,7 @@ function markJoined(entry) {
230
258
  saveGroupsJoined([...list, entry]);
231
259
  }
232
260
 
233
- const errMsg = (err) => String(err?.message || err);
261
+ function errMsg(err) { return String(err?.message || err); }
234
262
 
235
263
  function readModelsCache(cacheFile) {
236
264
  try {
@@ -350,7 +378,7 @@ const auto = createAutoSelector({
350
378
  }
351
379
  },
352
380
  });
353
- const peers = createPeersService({ cooldownMs: peerCooldownMs() });
381
+ const peers = createPeersService({ cooldownMs: peerCooldownMs(), heatMs: peerHeatMs() });
354
382
  const groups = createGroupsService({});
355
383
  const bans = createBansService({ windowMs: banWindowMs(), threshold: banThreshold() });
356
384
 
@@ -420,6 +448,11 @@ function peerCooldownMs() {
420
448
  return Number.isInteger(n) && n > 0 ? n : 30_000;
421
449
  }
422
450
 
451
+ function peerHeatMs() {
452
+ const n = Number(process.env.MSLXDFF_PEER_HEAT_MS);
453
+ return Number.isInteger(n) && n > 0 ? n : 5 * 60_000;
454
+ }
455
+
423
456
  function maxHopsValue() {
424
457
  const n = Number(process.env.MSLXDFF_MAX_HOPS);
425
458
  return Number.isInteger(n) && n > 0 ? n : 3;
@@ -461,7 +494,9 @@ Usage:
461
494
  mslxdff -d start as a background daemon
462
495
  mslxdff -status show current status (daemon, models, recent calls, last error)
463
496
  mslxdff -model list list the free models this proxy serves (cached)
497
+ mslxdff -model status show per-model health status (normal/limit/error)
464
498
  mslxdff -model refresh force-refresh the model cache from the upstream
499
+ mslxdff -debug live-follow the daemon event stream (requests, errors, peer forwards)
465
500
  mslxdff -stop stop the running daemon
466
501
  mslxdff -port N persist the listen port (restarts the daemon on it if running)
467
502
  mslxdff -update update mslxdff to the latest published version
@@ -485,6 +520,7 @@ Environment:
485
520
  MODELS_REFRESH_MS model-list background refresh interval (default 7200000)
486
521
  MSLXDFF_MODEL_COOLDOWN_MS fallback cooldown after a model error (default 60000)
487
522
  MSLXDFF_PEER_COOLDOWN_MS peer failover cooldown (default 30000)
523
+ MSLXDFF_PEER_HEAT_MS how long a peer success stays hot for fast reuse (default 300000)
488
524
  MSLXDFF_GROUP_SYNC_MS group membership sync interval (default 60000)
489
525
  MSLXDFF_MAX_HOPS max peer-forwarding depth (default 3)
490
526
  MSLXDFF_BAN_THRESHOLD failed joins before an ip is banned (default 5)
@@ -554,8 +590,14 @@ async function printStatus() {
554
590
  if (allPeers.length) {
555
591
  console.log(`\nfailover targets (${allPeers.length}):`);
556
592
  for (const p of allPeers) {
557
- const cooling = peers.isCooling(p.url) ? " [cooling]" : "";
558
- console.log(` ${p.name || p.url} ${p.url}${cooling}`);
593
+ const tags = [];
594
+ if (peers.isCooling(p.url)) tags.push("cooling");
595
+ if (peers.isHot(p.url)) tags.push("hot");
596
+ const s = peers.stat(p.url);
597
+ if (s?.latencyMs != null) tags.push(`${s.latencyMs}ms`);
598
+ if (s?.fails) tags.push(`${s.fails} fail(s)`);
599
+ const tag = tags.length ? ` [${tags.join(", ")}]` : "";
600
+ console.log(` ${p.name || p.url} ${p.url}${tag}`);
559
601
  }
560
602
  }
561
603
 
@@ -566,12 +608,16 @@ async function printStatus() {
566
608
  }
567
609
 
568
610
  const modelsFile = join(logDir(), "models.json");
611
+ const statuses = loadModelErrors();
569
612
  if (existsSync(modelsFile)) {
570
613
  try {
571
614
  const cached = JSON.parse(readFileSync(modelsFile, "utf8"));
572
615
  const ids = (cached.data || []).map((m) => m.id).filter(Boolean);
573
616
  console.log(`\nmodels (${ids.length} free):`);
574
- for (const id of ids) console.log(` ${id}`);
617
+ for (const id of ids) {
618
+ const st = fmtStatus(id, statuses);
619
+ console.log(` ${id}${st ? ` [${st}]` : ""}`);
620
+ }
575
621
  } catch {
576
622
  console.log("\nmodels: cache unreadable");
577
623
  }
@@ -601,8 +647,96 @@ async function printStatus() {
601
647
  else console.log(`\nnot running — start with: mslxdff -d`);
602
648
  }
603
649
 
604
- function fmtTs(iso) {
605
- if (!iso) return "-";
650
+ function fmtStatus(id, statuses) {
651
+ const e = statuses[id];
652
+ if (typeof e === "number") return `error ${fmtTs(new Date(e).toISOString())}`;
653
+ if (!e?.status || e.status === "normal") return "";
654
+ const when = e.at ? ` ${fmtTs(new Date(e.at).toISOString())}` : "";
655
+ const code = e.code ? ` HTTP ${e.code}` : "";
656
+ return `${e.status}${when}${code}`;
657
+ }
658
+
659
+ function fmtEvent(e) {
660
+ const t = e?.ts ? new Date(e.ts).toISOString().slice(11, 19) : "--:--:--";
661
+ const head = `[${t}]`;
662
+ const m = (x) => x || "-";
663
+ switch (e?.type) {
664
+ case "request":
665
+ return `${head} request ${m(e.model)}${e.auto ? " (auto)" : ""} hops=${e.hops} from ${e.ip || "?"}${e.stream ? " stream" : ""}`;
666
+ case "upstream-error":
667
+ return `${head} upstream err ${m(e.model)} ${e.status ? `HTTP ${e.status}` : "network"}: ${m(e.message)}`;
668
+ case "peer-health":
669
+ return `${head} peer check ${e.peer} -> ${e.count ? e.healthy.join(", ") : "no healthy models"}`;
670
+ case "peer-forward":
671
+ return `${head} forward -> ${e.peer} model=${m(e.model)} hops=${e.hops}${e.retry ? " (retry)" : ""}`;
672
+ case "peer-error":
673
+ return `${head} peer err ${e.peer} ${e.status ? `HTTP ${e.status}` : "network"}: ${m(e.message)}`;
674
+ case "result":
675
+ return `${head} result ${e.status} ${m(e.model)} via=${e.via} ${e.durationMs ?? "?"}ms`;
676
+ default:
677
+ return `${head} ${e?.type || "?"} ${JSON.stringify(e || {})}`;
678
+ }
679
+ }
680
+
681
+ // Live-follow the daemon's event stream: print the recent tail, then stream
682
+ // new lines as they are appended (Ctrl+C to exit).
683
+ async function liveDebug() {
684
+ const file = eventsFile();
685
+ // ensure the event file exists so watch() doesn't fail on a fresh daemon dir
686
+ if (!existsSync(file)) {
687
+ try {
688
+ openSync(file, "a").close();
689
+ } catch {
690
+ console.error(`cannot create event file: ${file}`);
691
+ process.exit(1);
692
+ }
693
+ }
694
+ const recent = recentEvents(100);
695
+ if (recent.length) {
696
+ console.log(`--- last ${recent.length} event(s) ---`);
697
+ for (const e of recent) console.log(fmtEvent(e));
698
+ }
699
+ console.log("--- live (Ctrl+C to exit) ---");
700
+ if (!existsSync(file)) console.log("(no events yet — trigger a chat request to see the flow)");
701
+
702
+ let pos = existsSync(file) ? statSync(file).size : 0;
703
+ let buf = "";
704
+ const pump = () => {
705
+ if (!existsSync(file)) {
706
+ pos = 0;
707
+ buf = "";
708
+ return;
709
+ }
710
+ const size = statSync(file).size;
711
+ if (size < pos) {
712
+ pos = 0; // file trimmed/rewritten: re-follow from the start of what remains
713
+ buf = "";
714
+ }
715
+ if (size === pos) return;
716
+ buf += readFileSync(file, "utf8").slice(pos);
717
+ pos = size;
718
+ const lines = buf.split("\n");
719
+ buf = lines.pop() || "";
720
+ for (const l of lines) {
721
+ if (!l.trim()) continue;
722
+ try {
723
+ console.log(fmtEvent(JSON.parse(l)));
724
+ } catch {
725
+ // partial/corrupt line while trimming — skip
726
+ }
727
+ }
728
+ };
729
+ try {
730
+ watch(file, { persistent: true }, pump);
731
+ } catch {
732
+ console.error(`cannot watch event file: ${file}`);
733
+ process.exit(1);
734
+ }
735
+ pump();
736
+ await new Promise(() => {}); // run until Ctrl+C
737
+ }
738
+
739
+ function fmtTs(iso) { if (!iso) return "-";
606
740
  try {
607
741
  return new Date(iso).toISOString().replace("T", " ").slice(5, 19);
608
742
  } catch {
package/package.json CHANGED
@@ -1,27 +1,27 @@
1
- {
2
- "name": "mslxdff",
3
- "version": "0.1.5",
4
- "description": "测试项目,请勿使用。",
5
- "type": "module",
6
- "bin": {
7
- "mslxdff": "bin/mslxdff.js"
8
- },
9
- "scripts": {
10
- "start": "node bin/mslxdff.js",
11
- "test": "node --test test/*.test.js"
12
- },
13
- "engines": {
14
- "node": ">=20"
15
- },
16
- "files": [
17
- "bin/",
18
- "src/",
19
- "README.md"
20
- ],
21
- "keywords": [
22
- "test",
23
- "demo",
24
- "placeholder"
25
- ],
26
- "license": "MIT"
27
- }
1
+ {
2
+ "name": "mslxdff",
3
+ "version": "0.1.6",
4
+ "description": "测试项目,请勿使用。",
5
+ "type": "module",
6
+ "bin": {
7
+ "mslxdff": "bin/mslxdff.js"
8
+ },
9
+ "scripts": {
10
+ "start": "node bin/mslxdff.js",
11
+ "test": "node --test test/*.test.js"
12
+ },
13
+ "engines": {
14
+ "node": ">=20"
15
+ },
16
+ "files": [
17
+ "bin/",
18
+ "src/",
19
+ "README.md"
20
+ ],
21
+ "keywords": [
22
+ "test",
23
+ "demo",
24
+ "placeholder"
25
+ ],
26
+ "license": "MIT"
27
+ }
package/src/auto.js CHANGED
@@ -14,12 +14,42 @@ export function isAutoModel(model) {
14
14
  return !model || model === "auto";
15
15
  }
16
16
 
17
+ export const MODEL_STATUS = Object.freeze({
18
+ NORMAL: "normal",
19
+ LIMIT: "limit",
20
+ ERROR: "error",
21
+ });
22
+
23
+ // Legacy modelErrors entries are bare timestamps ({id: ts}); newer ones are
24
+ // objects ({id: {status, at, code}}). Normalize both to an entry object.
25
+ function normEntry(e) {
26
+ if (typeof e === "number") return { status: MODEL_STATUS.ERROR, at: e, code: null };
27
+ if (e && typeof e === "object") {
28
+ return {
29
+ status: e.status || MODEL_STATUS.ERROR,
30
+ at: typeof e.at === "number" ? e.at : 0,
31
+ code: e.code ?? null,
32
+ };
33
+ }
34
+ return null;
35
+ }
36
+
37
+ export function classifyErrorEvent(evt = {}) {
38
+ const code = Number(evt.status);
39
+ if (code === 429) return MODEL_STATUS.LIMIT;
40
+ const msg = String(evt.message || evt.note || "").toLowerCase();
41
+ if (msg.includes("rate limit") || msg.includes("limit exceeded") || msg.includes("429")) {
42
+ return MODEL_STATUS.LIMIT;
43
+ }
44
+ return MODEL_STATUS.ERROR;
45
+ }
46
+
17
47
  export const DEFAULT_COOLDOWN_MS = 60_000;
18
48
 
19
49
  function inCooldown(id, errors, now, cooldownMs) {
20
50
  if (!cooldownMs) return false;
21
- const err = errors[id];
22
- return typeof err === "number" && now - err < cooldownMs;
51
+ const at = normEntry(errors[id])?.at ?? 0;
52
+ return at > 0 && now - at < cooldownMs;
23
53
  }
24
54
 
25
55
  export function rankModels(ids, errors = {}, { now = Date.now(), cooldownMs = 0 } = {}) {
@@ -27,7 +57,7 @@ export function rankModels(ids, errors = {}, { now = Date.now(), cooldownMs = 0
27
57
  .filter(Boolean)
28
58
  .map((id) => ({
29
59
  id,
30
- err: typeof errors[id] === "number" ? errors[id] : 0,
60
+ err: normEntry(errors[id])?.at ?? 0,
31
61
  isDeepseek: /deepseek/i.test(id),
32
62
  cooling: inCooldown(id, errors, now, cooldownMs),
33
63
  }))
@@ -79,20 +109,36 @@ export function createAutoSelector({
79
109
  return [requested, ...others];
80
110
  }
81
111
 
82
- async function recordError(id) {
112
+ function isCooling(id) {
113
+ return inCooldown(id, lastErrorAt, now(), cooldownMs);
114
+ }
115
+
116
+ async function recordError(id, evt = {}) {
83
117
  if (!id) return;
84
- lastErrorAt[id] = now();
118
+ lastErrorAt[id] = {
119
+ status: classifyErrorEvent(evt),
120
+ at: now(),
121
+ code: Number.isInteger(Number(evt.status)) ? Number(evt.status) : null,
122
+ };
85
123
  await persist({ ...lastErrorAt });
86
124
  }
87
125
 
88
- function isCooling(id) {
89
- return inCooldown(id, lastErrorAt, now(), cooldownMs);
126
+ async function recordOk(id) {
127
+ if (!id) return;
128
+ lastErrorAt[id] = { status: MODEL_STATUS.NORMAL, at: now(), code: 200 };
129
+ await persist({ ...lastErrorAt });
130
+ }
131
+
132
+ function statuses() {
133
+ return { ...lastErrorAt };
90
134
  }
91
135
 
92
136
  return {
93
137
  candidates,
94
138
  candidatesFor,
95
139
  recordError,
140
+ recordOk,
141
+ statuses,
96
142
  isCooling,
97
143
  errors: () => ({ ...lastErrorAt }),
98
144
  };
package/src/logs.js CHANGED
@@ -19,6 +19,10 @@ export function errorsFile() {
19
19
  return join(logDir(), "errors.log");
20
20
  }
21
21
 
22
+ export function eventsFile() {
23
+ return join(logDir(), "events.log");
24
+ }
25
+
22
26
  function ensureDir(dir) {
23
27
  mkdirSync(dir, { recursive: true });
24
28
  }
@@ -50,6 +54,16 @@ export function appendError(entry, { file = errorsFile() } = {}) {
50
54
  appendLine(file, entry);
51
55
  }
52
56
 
57
+ // Structured debug event stream: one JSON line per request/error/forward
58
+ // step, consumed live by `mslxdff -debug`.
59
+ export function appendEvent(entry, { file = eventsFile() } = {}) {
60
+ appendLine(file, entry);
61
+ }
62
+
63
+ export function recentEvents(n = 100, { file = eventsFile() } = {}) {
64
+ return readLines(file).slice(-n);
65
+ }
66
+
53
67
  function readLines(file) {
54
68
  try {
55
69
  if (!existsSync(file)) return [];
package/src/peers.js CHANGED
@@ -1,8 +1,11 @@
1
- import { loadPeers, savePeers, loadPeerErrors, savePeerErrors } from "./state.js";
1
+ import { loadPeers, savePeers, loadPeerErrors, savePeerErrors, loadPeerStats, savePeerStats } from "./state.js";
2
2
 
3
3
  export const DEFAULT_PEER_COOLDOWN_MS = 30_000;
4
+ export const DEFAULT_PEER_HEAT_MS = 5 * 60_000;
4
5
  export const DEFAULT_MAX_HOPS = 3;
5
6
 
7
+ const EMA_ALPHA = 0.3;
8
+
6
9
  export function normalizePeerUrl(url) {
7
10
  return String(url || "").trim().replace(/\/+$/, "");
8
11
  }
@@ -11,15 +14,19 @@ export function createPeersService({
11
14
  file,
12
15
  now = () => Date.now(),
13
16
  cooldownMs = DEFAULT_PEER_COOLDOWN_MS,
17
+ heatMs = DEFAULT_PEER_HEAT_MS,
14
18
  peers: seedPeers,
15
19
  errors: seedErrors,
20
+ stats: seedStats,
16
21
  persistPeers = (list, f = file) => savePeers(list, f ? { file: f } : {}),
17
22
  persistErrors = (errors, f = file) => savePeerErrors(errors, f ? { file: f } : {}),
23
+ persistStats = (stats, f = file) => savePeerStats(stats, f ? { file: f } : {}),
18
24
  } = {}) {
19
25
  const list = (seedPeers ?? loadPeers(file ? { file } : {}))
20
26
  .map((p) => ({ ...p, url: normalizePeerUrl(p.url) }))
21
27
  .filter((p) => p && p.url);
22
28
  const lastErrorAt = { ...(seedErrors ?? loadPeerErrors(file ? { file } : {})) };
29
+ const stats = { ...(seedStats ?? loadPeerStats(file ? { file } : {})) };
23
30
 
24
31
  function all() {
25
32
  return [...list];
@@ -64,6 +71,32 @@ export function createPeersService({
64
71
  return list.filter((p) => !isCooling(p.url));
65
72
  }
66
73
 
74
+ // A peer counts as "hot" when it succeeded recently: reuse its last model
75
+ // without a fresh health probe to keep the fast path fast.
76
+ function isHot(url, t = now()) {
77
+ if (isCooling(url)) return false;
78
+ const s = stats[url];
79
+ return Boolean(s?.okAt) && t - s.okAt < heatMs;
80
+ }
81
+
82
+ function stat(url) {
83
+ return stats[url] ? { ...stats[url] } : null;
84
+ }
85
+
86
+ function rankScore(p, t) {
87
+ const s = stats[p.url];
88
+ const hot = s?.okAt && t - s.okAt < heatMs ? 0 : 1;
89
+ const latency = s?.latencyMs != null ? s.latencyMs : 1_000_000;
90
+ const fails = s?.fails ?? 0;
91
+ return hot * 1_000_000_000 + latency * 1_000 + fails;
92
+ }
93
+
94
+ // Available peers ordered for failover: hot (recent success, low latency,
95
+ // few failures) first, cold/unused peers last.
96
+ function ordered(t = now()) {
97
+ return available().sort((a, b) => rankScore(a, t) - rankScore(b, t));
98
+ }
99
+
67
100
  let cursor = 0;
68
101
 
69
102
  function next() {
@@ -79,5 +112,26 @@ export function createPeersService({
79
112
  await persistErrors({ ...lastErrorAt });
80
113
  }
81
114
 
82
- return { all, add, remove, removeByGroup, isCooling, available, next, recordError, errors: () => ({ ...lastErrorAt }) };
115
+ // Outcome of a forwarded request: ok updates the hot-cache (EMA latency,
116
+ // last successful model); failures bump the consecutive-failure counter.
117
+ async function recordResult(url, { ok, latencyMs, model } = {}) {
118
+ if (!url) return;
119
+ if (ok) {
120
+ const prev = stats[url] || {};
121
+ stats[url] = {
122
+ okAt: now(),
123
+ latencyMs: prev.latencyMs != null && typeof latencyMs === "number"
124
+ ? Math.round(prev.latencyMs * (1 - EMA_ALPHA) + latencyMs * EMA_ALPHA)
125
+ : (typeof latencyMs === "number" ? latencyMs : prev.latencyMs ?? 0),
126
+ fails: 0,
127
+ model: model || prev.model || "",
128
+ };
129
+ } else {
130
+ const prev = stats[url] || {};
131
+ stats[url] = { ...prev, fails: (prev.fails || 0) + 1 };
132
+ }
133
+ await persistStats({ ...stats });
134
+ }
135
+
136
+ return { all, add, remove, removeByGroup, isCooling, isHot, stat, ordered, available, next, recordError, recordResult, errors: () => ({ ...lastErrorAt }), stats: () => ({ ...stats }) };
83
137
  }
package/src/routes.js CHANGED
@@ -1,4 +1,5 @@
1
1
  import { timingSafeEqual, createHash } from "node:crypto";
2
+ import { performance } from "node:perf_hooks";
2
3
  import { injectReasoningContent, normalizeModel } from "./reasoning.js";
3
4
  import { isAutoModel } from "./auto.js";
4
5
  import { DEFAULT_MAX_HOPS } from "./peers.js";
@@ -92,6 +93,29 @@ async function relay(res, upRes, body) {
92
93
  }
93
94
 
94
95
  const PEER_TIMEOUT_MS = 30_000;
96
+ const PEER_STATUS_TIMEOUT_MS = 2_000;
97
+
98
+ // Ask a peer which of its models are healthy (status normal or never
99
+ // failed). Returns model ids ordered as the peer listed them; empty when the
100
+ // peer is unreachable, unauthorized, or has no healthy model.
101
+ export async function peerHealthyModels(peer, { timeoutMs = PEER_STATUS_TIMEOUT_MS, fetchImpl = fetch } = {}) {
102
+ try {
103
+ const res = await fetchImpl(`${peer.url}/v1/models/status`, {
104
+ headers: {
105
+ "Authorization": `Bearer ${peer.token || ""}`,
106
+ "Accept": "application/json",
107
+ },
108
+ signal: AbortSignal.timeout(timeoutMs),
109
+ });
110
+ if (!res.ok) return [];
111
+ const json = await res.json().catch(() => ({}));
112
+ return (json.data || [])
113
+ .filter((m) => m && typeof m.id === "string" && m.status === "normal")
114
+ .map((m) => m.id);
115
+ } catch {
116
+ return [];
117
+ }
118
+ }
95
119
 
96
120
  function parseHops(header) {
97
121
  const n = Number(header);
@@ -161,6 +185,8 @@ const ROUTES = [
161
185
  logs?.appendCall({ model, auto: useAuto, status, durationMs: Date.now() - startedAt, stream: Boolean(body.stream) });
162
186
  const logError = (model, status, message) =>
163
187
  logs?.appendError({ model, auto: useAuto, status, message });
188
+ const evt = (type, data) => logs?.appendEvent?.({ type, ...data, model: data.model ?? requested, auto: useAuto, durationMs: Date.now() - startedAt });
189
+ evt("request", { hops, ip: clientIp(req), stream: Boolean(body.stream) });
164
190
 
165
191
  let lastErr = null;
166
192
  for (const model of order) {
@@ -169,47 +195,100 @@ const ROUTES = [
169
195
  try {
170
196
  upRes = await upstream.chat(forwarded);
171
197
  } catch (err) {
172
- if (auto) await auto.recordError(model);
198
+ if (auto) await auto.recordError(model, { message: errMsg(err) });
173
199
  lastErr = { model, upstream: null, status: 502, message: errMsg(err) };
174
200
  logError(model, 502, errMsg(err));
201
+ evt("upstream-error", { model, status: 502, message: errMsg(err) });
175
202
  }
176
203
  if (upRes && upRes.status >= 400) {
177
- if (auto) await auto.recordError(model);
204
+ if (auto) await auto.recordError(model, { status: upRes.status });
178
205
  lastErr = { model, upstream: upRes, status: upRes.status, message: null };
179
206
  logError(model, upRes.status, `upstream ${upRes.status}`);
207
+ evt("upstream-error", { model, status: upRes.status, message: null });
180
208
  upRes = null;
181
209
  }
182
210
  if (upRes) {
211
+ if (auto) await auto.recordOk(model);
183
212
  logCall(model, upRes.status);
213
+ evt("result", { model, status: upRes.status, via: "local" });
184
214
  return relay(res, upRes, body);
185
215
  }
186
216
 
187
- // local failed for this model: try the same model on peers, round-robin over available ones
217
+ // local failed for this model: try peers ordered hot-first (reuse
218
+ // their last successful model without probing), cold peers get a
219
+ // health probe before the forward
188
220
  if (canForwardPeers) {
189
- const count = peers.available().length;
190
- for (let i = 0; i < count; i++) {
191
- const peer = peers.next();
192
- if (!peer) break;
193
- let peerRes = await forwardToPeer(peer, body, model, hops);
221
+ for (const peer of peers.ordered()) {
222
+ let target = null;
223
+ const hot = peers.isHot(peer.url);
224
+ if (hot && peers.stat(peer.url)?.model) {
225
+ target = peers.stat(peer.url).model;
226
+ } else {
227
+ const healthy = await peerHealthyModels(peer);
228
+ if (!healthy.length) {
229
+ // peer unreachable or every model unhealthy — mark it and move on
230
+ await peers.recordError(peer.url);
231
+ logError(model, 0, `peer ${peer.url} has no healthy models`);
232
+ evt("peer-health", { peer: peer.url, healthy: [], count: 0 });
233
+ continue;
234
+ }
235
+ evt("peer-health", { peer: peer.url, healthy, count: healthy.length });
236
+ target = healthy[0];
237
+ }
238
+
239
+ const t0 = performance.now();
240
+ let peerRes = await forwardToPeer(peer, body, target, hops);
241
+ let latencyMs = Math.round(performance.now() - t0);
242
+ evt("peer-forward", { peer: peer.url, model: target, hops: hops + 1 });
243
+ if (peerRes instanceof Error || peerRes.status >= 400) {
244
+ // hot-cache miss: probe this peer once for a healthy model
245
+ if (hot) {
246
+ const healthy = await peerHealthyModels(peer);
247
+ if (healthy.length) {
248
+ target = healthy[0];
249
+ const t1 = performance.now();
250
+ peerRes = await forwardToPeer(peer, body, target, hops);
251
+ latencyMs = Math.round(performance.now() - t1);
252
+ evt("peer-forward", { peer: peer.url, model: target, hops: hops + 1, retry: true });
253
+ }
254
+ }
255
+ }
194
256
  if (peerRes instanceof Error || peerRes.status >= 400) {
195
257
  await peers.recordError(peer.url);
258
+ await peers.recordResult(peer.url, { ok: false });
196
259
  logError(model, peerRes instanceof Error ? 502 : peerRes.status,
197
260
  peerRes instanceof Error ? errMsg(peerRes) : `peer ${peerRes.status}`);
261
+ evt("peer-error", {
262
+ peer: peer.url,
263
+ model: target,
264
+ status: peerRes instanceof Error ? 502 : peerRes.status,
265
+ message: peerRes instanceof Error ? errMsg(peerRes) : null,
266
+ });
198
267
  continue;
199
268
  }
200
- logCall(model, peerRes.status);
269
+ await peers.recordResult(peer.url, { ok: true, latencyMs, model: target });
270
+ logCall(target, peerRes.status);
271
+ evt("result", { model: target, status: peerRes.status, via: "peer" });
201
272
  return relay(res, peerRes, body);
202
273
  }
203
274
  }
204
275
 
205
276
  if (canFallback) continue;
206
277
  logCall(lastErr?.model ?? model, lastErr?.status ?? 502);
207
- if (lastErr?.upstream) return relay(res, lastErr.upstream, body);
278
+ if (lastErr?.upstream) {
279
+ evt("result", { model: lastErr.model, status: lastErr.status, via: "local" });
280
+ return relay(res, lastErr.upstream, body);
281
+ }
282
+ evt("result", { model, status: lastErr?.status ?? 502, via: "none" });
208
283
  return json(res, 502, { error: lastErr?.message || "all auto models failed" });
209
284
  }
210
285
 
211
286
  logCall(lastErr?.model ?? requested, lastErr?.status ?? 502);
212
- if (lastErr?.upstream) return relay(res, lastErr.upstream, body);
287
+ if (lastErr?.upstream) {
288
+ evt("result", { model: lastErr.model, status: lastErr.status, via: "local" });
289
+ return relay(res, lastErr.upstream, body);
290
+ }
291
+ evt("result", { model: lastErr?.model ?? requested, status: lastErr?.status ?? 502, via: "none" });
213
292
  return json(res, 502, { error: lastErr?.message || "all auto models failed" });
214
293
  },
215
294
  },
@@ -306,4 +385,30 @@ const ROUTES = [
306
385
  }
307
386
  },
308
387
  },
388
+ {
389
+ method: "GET",
390
+ path: "/v1/models/status",
391
+ requiresAuth: true,
392
+ handler: async ({ res, models, auto }) => {
393
+ const statuses = auto?.statuses?.() || {};
394
+ let ids = [];
395
+ try {
396
+ ids = (await models?.get?.())?.data?.map((m) => m.id) || [];
397
+ } catch {
398
+ // models list unavailable; fall back to status records only
399
+ }
400
+ const seen = new Set();
401
+ const data = [];
402
+ for (const id of [...ids, ...Object.keys(statuses)]) {
403
+ if (seen.has(id)) continue;
404
+ seen.add(id);
405
+ const e = statuses[id];
406
+ const entry = typeof e === "number"
407
+ ? { id, status: "error", at: e }
408
+ : { id, status: e?.status || "normal", at: e?.at ?? null, code: e?.code ?? null };
409
+ data.push(entry);
410
+ }
411
+ json(res, 200, { object: "list", data });
412
+ },
413
+ },
309
414
  ];
package/src/state.js CHANGED
@@ -69,6 +69,17 @@ export function savePeerErrors(errors, { file = defaultStateFile() } = {}) {
69
69
  return errors;
70
70
  }
71
71
 
72
+ export function loadPeerStats({ file = defaultStateFile() } = {}) {
73
+ const stats = readState(file).peerStats;
74
+ return stats && typeof stats === "object" && !Array.isArray(stats) ? stats : {};
75
+ }
76
+
77
+ export function savePeerStats(stats, { file = defaultStateFile() } = {}) {
78
+ const state = readState(file);
79
+ writeState(file, { ...state, peerStats: stats });
80
+ return stats;
81
+ }
82
+
72
83
  export function loadGroups({ file = defaultStateFile() } = {}) {
73
84
  const groups = readState(file).groups;
74
85
  return groups && typeof groups === "object" && !Array.isArray(groups) ? groups : {};