@chessceo/mcp 0.49.11 → 0.50.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,18 +1,106 @@
1
- // Shared engine-response helpers used across cloud_analyse, auto_evaluate,
2
- // deep_analyse, and get_position_stats. All PGN-shape data comes back
3
- // from the backend in UCI ("g1f3") — this module converts to SAN, extracts
4
- // stored evals from the raw JSON, and derives the eval → NAG mapping.
5
- //
6
- // Extracted from src/index.ts in v0.44 as part of the file split.
1
+ // Shared engine helpers for cloud_analyse, auto_evaluate, deep_analyse and
2
+ // the compact eval attached to position tools. The backend returns one
3
+ // result per position with a block per engine (stockfish, lc0, human), PVs
4
+ // in UCI and WDL in per mille. This module calls it in chunks, converts PVs
5
+ // to SAN and WDL to percent, and turns a result into the stored ceoEval.
7
6
  import { Chess } from "chess.js";
8
7
  import { authedRequest } from "../http.js";
9
- // Convert a UCI move sequence into SAN by walking it move-by-move on
10
- // chess.js from the given starting FEN. LLMs reason far better in SAN
11
- // ("Nf3", "Bxc4") than UCI ("g1f3", "b5c4"), and matches how prep
12
- // discussion is written in the real world. If a move fails to parse
13
- // (illegal from the current position — bug or truncated PV), we
14
- // truncate cleanly rather than throwing so the response still carries
15
- // what we could convert.
8
+ import { rememberEvals } from "./eval_cache.js";
9
+ export const ENGINES = ["stockfish", "lc0", "human"];
10
+ // Engine name → key in the stored ceoEval / [%ceo-eval] escape.
11
+ export const STORED_KEY = { stockfish: "sf", lc0: "lc0", human: "human" };
12
+ // Backend limits (handlers/agent.go): positions per call, and positions ×
13
+ // movetime per call. analysePositions chunks to stay inside both.
14
+ export const MAX_POSITIONS_PER_CALL = 10;
15
+ const MAX_CALL_BUDGET_MS = 150_000;
16
+ export const DEFAULT_MOVETIME_MS = 2000;
17
+ // Validate an `engines` argument. Unknown names throw (better than silently
18
+ // running fewer engines than the caller asked for).
19
+ export function parseEngines(v, fallback) {
20
+ if (v === undefined || v === null)
21
+ return fallback;
22
+ const list = Array.isArray(v) ? v : [v];
23
+ const out = [];
24
+ for (const e of list) {
25
+ const name = String(e).trim().toLowerCase();
26
+ if (!ENGINES.includes(name)) {
27
+ throw new Error(`unknown engine '${e}' (allowed: stockfish, lc0, human)`);
28
+ }
29
+ if (!out.includes(name))
30
+ out.push(name);
31
+ }
32
+ if (out.length === 0)
33
+ throw new Error("engines must list at least one of: stockfish, lc0, human");
34
+ return out;
35
+ }
36
+ // Pull the optional per-engine knobs (multipv, contempt, rentals) out of tool
37
+ // args. Shared by every analysis tool so they all accept the same shape.
38
+ export function analyseOptionsFromArgs(args, engines, movetimeMs) {
39
+ const opts = { engines, movetime_ms: movetimeMs };
40
+ const pickMap = (v) => v && typeof v === "object" && !Array.isArray(v) ? v : undefined;
41
+ const mpv = pickMap(args.multipv);
42
+ if (mpv) {
43
+ const out = {};
44
+ for (const e of ENGINES)
45
+ if (typeof mpv[e] === "number")
46
+ out[e] = mpv[e];
47
+ opts.multipv = out;
48
+ }
49
+ if (typeof args.contempt === "number")
50
+ opts.contempt = args.contempt;
51
+ const rentals = pickMap(args.rentals);
52
+ if (rentals) {
53
+ const out = {};
54
+ for (const e of ENGINES)
55
+ if (typeof rentals[e] === "string" && rentals[e].trim())
56
+ out[e] = rentals[e].trim();
57
+ opts.rentals = out;
58
+ }
59
+ return opts;
60
+ }
61
+ // Run positions through the backend, chunked to its per-call limits. Results
62
+ // come back in input order, raw (UCI PVs, per-mille WDL).
63
+ export async function analysePositions(positions, opts) {
64
+ const movetime = opts.movetime_ms ?? DEFAULT_MOVETIME_MS;
65
+ const perCall = Math.max(1, Math.min(MAX_POSITIONS_PER_CALL, Math.floor(MAX_CALL_BUDGET_MS / movetime)));
66
+ // Every position gets an id; results are matched back by id, not order.
67
+ const withIds = positions.map((p, i) => ({ id: p.id ?? `p${i + 1}`, fen: p.fen }));
68
+ const byId = new Map();
69
+ for (let i = 0; i < withIds.length; i += perCall) {
70
+ const body = {
71
+ positions: withIds.slice(i, i + perCall),
72
+ engines: opts.engines,
73
+ movetime_ms: movetime,
74
+ };
75
+ if (opts.multipv && Object.keys(opts.multipv).length > 0)
76
+ body.multipv = opts.multipv;
77
+ if (typeof opts.contempt === "number")
78
+ body.contempt = opts.contempt;
79
+ if (opts.rentals && Object.keys(opts.rentals).length > 0)
80
+ body.rentals = opts.rentals;
81
+ const raw = (await authedRequest("POST", "/api/agent/cloud-engines/analyse", body));
82
+ if (!raw || !Array.isArray(raw.positions))
83
+ throw new Error("unexpected response from the analyse endpoint");
84
+ const remembered = [];
85
+ for (const r of raw.positions) {
86
+ if (!r || typeof r.id !== "string")
87
+ continue;
88
+ byId.set(r.id, r);
89
+ const ev = resultToStoredEval(convertPositionResult(structuredClone(r)));
90
+ if (ev)
91
+ remembered.push({ fen: r.fen, ev });
92
+ }
93
+ // Remembered per chunk, so a later chunk failing loses nothing already
94
+ // paid for; a later write fills these into the file (eval_cache.ts).
95
+ await rememberEvals(remembered);
96
+ }
97
+ // A position the backend didn't answer comes back with no engines, so
98
+ // callers report it as failed rather than shifting other results.
99
+ return withIds.map(p => byId.get(p.id) ?? { id: p.id, fen: p.fen, engines: {} });
100
+ }
101
+ // Convert a UCI move sequence into SAN by walking it on chess.js from the
102
+ // given FEN. LLMs reason far better in SAN. If a move fails to parse
103
+ // (truncated PV), stop cleanly rather than throw.
16
104
  export function uciLineToSAN(startFen, uciMoves) {
17
105
  const board = new Chess(startFen);
18
106
  const out = [];
@@ -51,131 +139,105 @@ export function uciMoveToSAN(startFen, uci) {
51
139
  return uci;
52
140
  }
53
141
  }
54
- // Fetch a compact cloud eval for `fen`. Returns null on any error — no
55
- // running combo instance, engine failure, network timeout. Callers
56
- // attach the result to their response as `.eval` so the LLM has the
57
- // stockfish + lc0 read without a separate tool call.
58
- export async function fetchCompactEval(fen) {
59
- try {
60
- const raw = await authedRequest("POST", "/api/agent/cloud-engines/analyse", { fen, movetime_ms: 1500, multipv: 1 });
61
- const converted = convertCloudSnapshotResponse(raw, fen);
62
- const stored = analysisToStoredEval(converted);
63
- return storedEvalToCompact(stored, converted);
64
- }
65
- catch {
66
- return null;
67
- }
68
- }
69
- // Trim every PV in a converted cloud-analyse response to `maxPlies`
70
- // and mark each trimmed line with `pv_truncated: true` so the LLM
71
- // sees what happened. Applied ONLY to cloud_analyse (short synchronous
72
- // snapshot); deep_analyse is the explicit "give me the deep line"
73
- // tool and keeps its full PV.
74
- export function capPvsInResponse(converted, maxPlies) {
75
- if (!converted || typeof converted !== "object")
76
- return;
77
- const r = converted;
78
- for (const eng of [r.stockfish, r.lc0]) {
79
- if (!eng || !Array.isArray(eng.lines))
142
+ // In place: PVs and bestMove to SAN, WDL from per mille to percent, and (if
143
+ // pvMaxPlies is set) PVs capped with `pv_truncated` on trimmed lines. PVs
144
+ // past ~6 plies are speculative, and long ones get pasted into add_line as
145
+ // if they were prep; the caller re-analyses the end position to see more.
146
+ export function convertPositionResult(r, pvMaxPlies) {
147
+ for (const e of ENGINES) {
148
+ const block = r.engines?.[e];
149
+ if (!block)
80
150
  continue;
81
- for (const line of eng.lines) {
82
- if (Array.isArray(line.pv) && line.pv.length > maxPlies) {
83
- line.pv = line.pv.slice(0, maxPlies);
84
- line.pv_truncated = true;
151
+ for (const line of block.lines ?? []) {
152
+ if (Array.isArray(line.pv)) {
153
+ line.pv = uciLineToSAN(r.fen, line.pv);
154
+ if (pvMaxPlies && line.pv.length > pvMaxPlies) {
155
+ line.pv = line.pv.slice(0, pvMaxPlies);
156
+ line.pv_truncated = true;
157
+ }
85
158
  }
159
+ if (line.wdl)
160
+ line.wdl = wdlPercent(line.wdl);
86
161
  }
162
+ if (typeof block.bestMove === "string")
163
+ block.bestMove = uciMoveToSAN(r.fen, block.bestMove);
87
164
  }
88
- converted.pv_max_plies = maxPlies;
165
+ return r;
89
166
  }
90
- export function convertCloudSnapshotResponse(raw, startFen) {
91
- if (!raw || typeof raw !== "object")
92
- return raw;
93
- const r = raw;
94
- for (const eng of [r.stockfish, r.lc0]) {
95
- if (!eng)
96
- continue;
97
- if (Array.isArray(eng.lines)) {
98
- for (const line of eng.lines) {
99
- if (Array.isArray(line.pv))
100
- line.pv = uciLineToSAN(startFen, line.pv);
101
- }
102
- }
103
- if (typeof eng.bestMove === "string")
104
- eng.bestMove = uciMoveToSAN(startFen, eng.bestMove);
167
+ // Per mille → whole percent summing to 100 (largest remainder), so a
168
+ // stored W/D/L never reads 101%.
169
+ function wdlPercent(m) {
170
+ const total = m.w + m.d + m.l || 1;
171
+ const exact = [m.w, m.d, m.l].map(v => (v * 100) / total);
172
+ const floor = exact.map(Math.floor);
173
+ let left = 100 - floor.reduce((a, b) => a + b, 0);
174
+ const order = exact.map((v, i) => [v - floor[i], i]).sort((a, b) => b[0] - a[0]);
175
+ for (const [, i] of order) {
176
+ if (left <= 0)
177
+ break;
178
+ floor[i]++;
179
+ left--;
105
180
  }
106
- return raw;
181
+ return { w: floor[0], d: floor[1], l: floor[2] };
107
182
  }
108
- export function analysisToStoredEval(analysis) {
109
- if (!analysis || typeof analysis !== "object")
110
- return null;
111
- const r = analysis;
112
- // Backend returns White-POV cp/mate (engine-ws flips in ParseInfo
113
- // based on side-to-move, cloud_snapshot passes through). Pure
114
- // pass-through here — a previous sign-flip on black-to-move was
115
- // wrong and silently inverted every Black-to-move stored eval.
116
- const engineEval = (block) => {
117
- const line = block?.lines?.[0];
118
- if (!line)
119
- return undefined;
120
- const depth = line.depth ?? block?.depth;
121
- if (typeof line.mate === "number")
122
- return { mate: line.mate, depth };
123
- if (typeof line.scoreCp === "number")
124
- return { cp: line.scoreCp, depth };
125
- return undefined;
126
- };
127
- const sf = engineEval(r.stockfish);
128
- const lc0 = engineEval(r.lc0);
129
- if (!sf && !lc0)
130
- return null;
183
+ // Stored eval from a CONVERTED result (WDL already in percent). Stockfish
184
+ // stores cp/mate; lc0 and human store WDL, falling back to cp/mate if an
185
+ // engine sent no WDL. Engines that errored or returned no line are left out,
186
+ // so a merge keeps whatever was stored for them before.
187
+ export function resultToStoredEval(r) {
131
188
  const ev = {};
132
- if (sf)
133
- ev.sf = sf;
134
- if (lc0)
135
- ev.lc0 = lc0;
136
- ev.nag = nagFromCp(sf?.cp, sf?.mate) ?? nagFromCp(lc0?.cp, lc0?.mate) ?? undefined;
137
- return ev;
189
+ for (const e of ENGINES) {
190
+ const block = r.engines?.[e];
191
+ const line = block?.lines?.[0];
192
+ if (!block || block.error || !line)
193
+ continue;
194
+ const depth = line.depth ?? block.depth;
195
+ let stored;
196
+ if (e !== "stockfish" && line.wdl) {
197
+ stored = { w: line.wdl.w, d: line.wdl.d, l: line.wdl.l, depth };
198
+ }
199
+ else if (typeof line.mate === "number") {
200
+ stored = { mate: line.mate, depth };
201
+ }
202
+ else if (typeof line.scoreCp === "number") {
203
+ stored = { cp: line.scoreCp, depth };
204
+ }
205
+ if (stored)
206
+ ev[STORED_KEY[e]] = stored;
207
+ }
208
+ return ev.sf || ev.lc0 || ev.human ? ev : null;
138
209
  }
139
- export function nagFromCp(cp, mate) {
140
- let effective;
141
- if (typeof mate === "number")
142
- effective = mate > 0 ? 10000 : -10000;
143
- else if (typeof cp === "number")
144
- effective = cp;
145
- else
146
- return null;
147
- const abs = Math.abs(effective);
148
- if (abs < 25)
149
- return "$10";
150
- if (abs < 60)
151
- return effective > 0 ? "$14" : "$15";
152
- if (abs < 130)
153
- return effective > 0 ? "$16" : "$17";
154
- return effective > 0 ? "$18" : "$19";
210
+ // Engines that produced no usable line for this position, with the reason.
211
+ export function engineFailures(r, engines) {
212
+ const out = {};
213
+ for (const e of engines) {
214
+ const block = r.engines?.[e];
215
+ if (!block)
216
+ out[e] = "no result";
217
+ else if (block.error)
218
+ out[e] = block.error;
219
+ else if (!block.lines || block.lines.length === 0)
220
+ out[e] = "no lines";
221
+ }
222
+ return out;
155
223
  }
156
- // Adapter for the compact eval attached to live query responses. Same
157
- // derivation logic; different output shape (needs the .nag + summary
158
- // used by get_position_stats / prep_snapshot).
159
- export function storedEvalToCompact(ev, analysis) {
160
- if (!ev)
161
- return null;
162
- const a = analysis;
163
- const compact = { nag: ev.nag ?? null };
164
- if (ev.sf) {
165
- compact.stockfish = {
166
- cp: ev.sf.cp,
167
- mate: ev.sf.mate,
168
- bestMove: a.stockfish?.bestMove,
169
- pv: a.stockfish?.lines?.[0]?.pv,
170
- };
224
+ export async function fetchCompactEval(fen) {
225
+ try {
226
+ const [raw] = await analysePositions([{ fen }], {
227
+ engines: ["stockfish"],
228
+ movetime_ms: 1500,
229
+ multipv: { stockfish: 1 },
230
+ });
231
+ if (!raw)
232
+ return null;
233
+ const r = convertPositionResult(raw, 6);
234
+ const sf = r.engines.stockfish;
235
+ const line = sf?.lines?.[0];
236
+ if (!sf || sf.error || !line)
237
+ return null;
238
+ return { stockfish: { cp: line.scoreCp, mate: line.mate, bestMove: sf.bestMove, pv: line.pv } };
171
239
  }
172
- if (ev.lc0) {
173
- compact.lc0 = {
174
- cp: ev.lc0.cp,
175
- mate: ev.lc0.mate,
176
- bestMove: a.lc0?.bestMove,
177
- pv: a.lc0?.lines?.[0]?.pv,
178
- };
240
+ catch {
241
+ return null;
179
242
  }
180
- return compact;
181
243
  }
package/dist/index.js CHANGED
@@ -22,10 +22,11 @@ import { TOOLS } from "./tools.js";
22
22
  import { PROMPTS } from "./prompts.js";
23
23
  import { commentAntiPatterns, longLineWarning, noDescribeWarning, noStatsCheckWarning, positionalNagOnIntermediateWarning, positionsDescribed, positionsStatsChecked, } from "./warnings.js";
24
24
  import { authContext, authedRequest, deleteGame, fetchGame, get, makeFileId, restoreGame, } from "./http.js";
25
- import { analysisToStoredEval, capPvsInResponse, convertCloudSnapshotResponse, fetchCompactEval, } from "./analysis/response.js";
25
+ import { fetchCompactEval, } from "./analysis/response.js";
26
+ import { cloudAnalyse, ensureEngines } from "./analysis/cloud.js";
26
27
  import { autoEvaluate, autoEvaluateCancel, autoEvaluateStatus, } from "./analysis/auto.js";
27
28
  import { deepAnalyseCancel, deepAnalyseStart, deepAnalyseStatus, } from "./analysis/deep.js";
28
- import { analyseRouting, getNodeByPath, resolveFromNodeOrFen, storeEvalOnNode, } from "./analysis/file_handle.js";
29
+ import { getNodeByPath, resolveFromNodeOrFen, } from "./analysis/file_handle.js";
29
30
  import { findPositionInCourses, readCourseAtPosition, runSfEval } from "./courses.js";
30
31
  import { applyBatchMutations, applyMutation, argNodeId, } from "./prep/mutations.js";
31
32
  import { listNodes, listTranspositions, readPrepFile, } from "./prep/read.js";
@@ -41,6 +42,8 @@ const AUTHED_TOOLS = new Set([
41
42
  "list_cloud_engines",
42
43
  "stop_cloud_engine",
43
44
  "cloud_analyse",
45
+ "ensure_engines",
46
+ "list_cloud_machine_options",
44
47
  "list_collections",
45
48
  "create_collection",
46
49
  "rename_collection",
@@ -125,11 +128,23 @@ const ENGINE_GUIDE = {
125
128
  stockfish: "Objective calculation, the default cloud engine at about 100 million nodes per second. A few seconds gives a very strong view of the position. Use it on every position, including when steering by lc0 or human. Its score is the objective value with best defense: 0.00 means objectively a draw with best play, not that nothing is happening or that neither side has chances. Many positions that are very hard to play for a human score 0.00. Scores drift toward 0.00 as the search deepens, especially in drawn positions. Use it to check whether a tactic is real, whether a defense holds, and whether a move loses material or the game.",
126
129
  lc0: "Neural-net engine trained on self-play games, not human games. It is better than Stockfish at long-term ideas and at judging which side is practically on top. Never gives 0.00: it always says which side prefers the position, at least a little. Use it when Stockfish gives 0.00 to see who still has the practical edge, and to find where the long-term ideas are. Contempt changes how it values a draw and which side it plays for: contempt at -20 makes Black play for a win, which is useful for finding ideas, including in openings. It is much weaker tactically than Stockfish, so confirm any concrete line with Stockfish.",
127
130
  human: "Neural net trained only on human games, about 11 million rated games. Its moves and evaluations are practical, not objective: the moves it suggests are the ones a human would find and play. It punishes dubious moves less than Stockfish, because human games contain many imperfect moves that still work in practice, so a move with a sound idea behind it can score well here. Use it to predict how a person will respond to an idea, to find ideas (it may like a move that only fails to a refutation no human would find; Stockfish shows whether that refutation exists), and to see how a human would evaluate a position, especially positional or unclear ones. Like lc0 it is weaker in very unusual positions, and its evaluations there can be trusted much less. Stockfish remains the check on tactics and on whether a practical idea is objectively sound.",
128
- combo: "Runs Stockfish and Lc0 together — use when you want both the objective read and the practical/human-feel read on the same position.",
129
131
  };
130
132
  // Modern engines give equality very often, so small numbers are not "equal".
131
133
  // Rough guide, not exact thresholds. Shown alongside ENGINE_GUIDE above.
132
134
  const ENGINE_EVAL_SCALE = "Small numbers are not equality. ±0.00 to ±0.10 is equal in practice (+0.10 is '=' or '+=' at most). +0.20 to +0.40 is real pressure, not 'a bit better'. +0.50 and up is a clear advantage. Stockfish 0.00 with a plus from Lc0 is a practical edge, not equality. Check both engines before calling a position equal.";
135
+ // How to run analysis end to end. Authed response only, with ENGINE_GUIDE.
136
+ const ENGINE_WORKFLOW = [
137
+ "1. Plan: ensure_engines({engines, positions}) shows what is running, what to start, the price and an estimate. It starts nothing.",
138
+ "2. Ask: show the user the price of anything you would start and get a yes. Rentals bill per minute until stopped.",
139
+ "3. Start: start_cloud_engine({machine_type}) per missing engine. Each SKU runs one engine; the engine comes from the SKU. Wait until list_cloud_engines shows it running (static engines are instant, others ~1-5 min).",
140
+ "4. Analyse: cloud_analyse({fens | lines | file_id+node_ids, engines}) for up to 10 positions, each engine on its own rental, in parallel. auto_evaluate({id, engines}) for a whole file or subtree: fills only the engines each node is missing and saves as it goes. deep_analyse({engine, ...}) for one long think.",
141
+ " Nothing analysed is lost: every result is remembered for 24 h, and any later file write (add_move, add_line, apply_mutations, ...) fills nodes reaching an analysed position with that eval (`evals_filled` in the response). So analyse loose lines freely, then add the keepers.",
142
+ "5. Read back: quote_engine_eval gives the stored sf (cp/mate), lc0 and human (win/draw/loss %) per node.",
143
+ "6. Stop: stop_cloud_engine when the work is done, unless the user wants it kept running.",
144
+ ].join("\n");
145
+ // Eval symbols are the writer's call, not the engine's. Recommendation
146
+ // only; nothing in the MCP sets a symbol automatically.
147
+ const EVAL_SYMBOL_GUIDE = "The eval symbol (=, +=, ±, +-) is your call, written for a human reader. The engines inform it; no tool sets it for you. Stockfish 0.00 can still be += when Lc0 or the human engine clearly prefer one side, because the symbol describes the practical picture, not only the objective one. Look at all engines you ran before choosing, and don't put a symbol on a position nobody analysed.";
133
148
  // v0.48: consolidated the five `read_*_guide` / `read_example_prep_files`
134
149
  // tools into ONE `read_docs`. LLM lists what it wants; we return them
135
150
  // in a single response. Also cleaner: enumerating available docs in one
@@ -390,68 +405,26 @@ async function callToolInner(name, args) {
390
405
  ...(resp && typeof resp === "object" ? resp : { options: [] }),
391
406
  engine_guide: ENGINE_GUIDE,
392
407
  eval_scale: ENGINE_EVAL_SCALE,
408
+ eval_symbols: EVAL_SYMBOL_GUIDE,
409
+ workflow: ENGINE_WORKFLOW,
393
410
  };
394
411
  }
412
+ case "ensure_engines":
413
+ return {
414
+ ...await ensureEngines(args),
415
+ engine_guide: ENGINE_GUIDE,
416
+ workflow: ENGINE_WORKFLOW,
417
+ };
395
418
  case "start_cloud_engine":
396
419
  return authedRequest("POST", "/api/agent/cloud-engines", {
397
420
  machineType: String(args.machine_type),
398
- ...(typeof args.engine_type === "string" ? { engineType: args.engine_type } : {}),
399
421
  });
400
422
  case "list_cloud_engines":
401
423
  return authedRequest("GET", "/api/agent/cloud-engines");
402
424
  case "stop_cloud_engine":
403
425
  return authedRequest("DELETE", `/api/agent/cloud-engines/${encodeURIComponent(String(args.contract_id))}`);
404
- case "cloud_analyse": {
405
- const resolved = await resolveFromNodeOrFen(args);
406
- const fen = resolved.fen;
407
- const body = { fen };
408
- if (typeof args.movetime_ms === "number")
409
- body.movetime_ms = args.movetime_ms;
410
- if (typeof args.stockfish_multipv === "number")
411
- body.stockfish_multipv = args.stockfish_multipv;
412
- if (typeof args.lc0_multipv === "number")
413
- body.lc0_multipv = args.lc0_multipv;
414
- if (typeof args.contempt === "number")
415
- body.contempt = args.contempt;
416
- Object.assign(body, analyseRouting(args));
417
- const raw = await authedRequest("POST", "/api/agent/cloud-engines/analyse", body);
418
- const converted = convertCloudSnapshotResponse(raw, fen);
419
- // PV cap: engine PVs beyond ~6 plies are speculative (the tail is
420
- // where the search's confidence collapses — SF at depth 24 has
421
- // seen the first few plies solidly and hedged everything after).
422
- // More importantly, LLMs paste long PVs into `add_line` as if
423
- // they were prepared repertoire. A 15-move PV pasted as a
424
- // variation is one line of engine output through positions
425
- // where both sides had real choices — not a repertoire. Cap the
426
- // affordance: return only what's load-bearing (3 full moves for
427
- // understanding the point), let the caller re-analyse the
428
- // resulting position if they want to see further. Override via
429
- // `pv_max_plies` for the rare case (deep tactics verification).
430
- const pvMaxPlies = typeof args.pv_max_plies === "number" && args.pv_max_plies > 0
431
- ? Math.min(args.pv_max_plies, 40)
432
- : 6;
433
- capPvsInResponse(converted, pvMaxPlies);
434
- // Node-addressed calls: persist the result on the node's ceoEval
435
- // so a later quote_engine_eval can cite this measurement. This is
436
- // the anti-hallucination hinge — prose that says "engines say X
437
- // on node Y" can only trace back to a call actually made against
438
- // node_id=Y, because the store only fires when file_id+node_id
439
- // was supplied and the eval survives via the [%ceo-eval] escape.
440
- if (resolved.file) {
441
- const ev = analysisToStoredEval(converted);
442
- if (ev) {
443
- const stamped = await storeEvalOnNode(resolved.file, ev);
444
- if (stamped.length > 1) {
445
- // Surface the propagation so the LLM sees exactly which
446
- // other nodes now carry this eval (and can skip them for
447
- // re-analysis).
448
- converted.also_stored_on = stamped.slice(1);
449
- }
450
- }
451
- }
452
- converted.eval_scale = ENGINE_EVAL_SCALE;
453
- return converted;
454
- }
426
+ case "cloud_analyse":
427
+ return cloudAnalyse(args, ENGINE_EVAL_SCALE);
455
428
  case "deep_analyse":
456
429
  return deepAnalyseStart(args);
457
430
  case "deep_analyse_status":
@@ -562,7 +535,7 @@ async function callToolInner(name, args) {
562
535
  case "apply_mutations":
563
536
  return applyBatchMutations(args);
564
537
  case "auto_evaluate":
565
- return autoEvaluate(args, applyBatchMutations);
538
+ return autoEvaluate(args);
566
539
  case "auto_evaluate_status":
567
540
  return autoEvaluateStatus(args);
568
541
  case "auto_evaluate_cancel":
@@ -659,72 +632,80 @@ Preparation workflow — follow the steps in order and be explicit about which t
659
632
  Don't just dump data. Reason about it. Cite specific numbers (game counts, win rates, dates) so the user can trust your conclusions.`;
660
633
  };
661
634
  // ── Server wiring ──────────────────────────────────────────────────
662
- const server = new Server({ name: "chessceo-mcp", version: process.env.npm_package_version ?? "0.1.0" }, { capabilities: { tools: {}, prompts: {} } });
663
- server.setRequestHandler(ListToolsRequestSchema, async () => ({ tools: TOOLS }));
664
- server.setRequestHandler(ListPromptsRequestSchema, async () => ({ prompts: PROMPTS }));
665
- server.setRequestHandler(GetPromptRequestSchema, async (req) => {
666
- const { name, arguments: args } = req.params;
667
- const promptArgs = {};
668
- if (args)
669
- for (const [k, v] of Object.entries(args))
670
- promptArgs[k] = String(v);
671
- let text;
672
- switch (name) {
673
- case "prepare_for_game":
674
- text = PREP_WORKFLOW(promptArgs);
675
- break;
676
- case "scout_player": {
677
- const p = promptArgs.player ?? "the player";
678
- text = `Produce a scouting report on ${p}. Steps:
635
+ // One Server per connection. HTTP mode is stateless (a transport per
636
+ // request), and a Server holds one transport at a time, so sharing a
637
+ // single Server made overlapping requests fail with "Already connected
638
+ // to a transport" (two users calling at once, or a client's notification
639
+ // racing its previous request).
640
+ function createServer() {
641
+ const server = new Server({ name: "chessceo-mcp", version: process.env.npm_package_version ?? "0.1.0" }, { capabilities: { tools: {}, prompts: {} } });
642
+ server.setRequestHandler(ListToolsRequestSchema, async () => ({ tools: TOOLS }));
643
+ server.setRequestHandler(ListPromptsRequestSchema, async () => ({ prompts: PROMPTS }));
644
+ server.setRequestHandler(GetPromptRequestSchema, async (req) => {
645
+ const { name, arguments: args } = req.params;
646
+ const promptArgs = {};
647
+ if (args)
648
+ for (const [k, v] of Object.entries(args))
649
+ promptArgs[k] = String(v);
650
+ let text;
651
+ switch (name) {
652
+ case "prepare_for_game":
653
+ text = PREP_WORKFLOW(promptArgs);
654
+ break;
655
+ case "scout_player": {
656
+ const p = promptArgs.player ?? "the player";
657
+ text = `Produce a scouting report on ${p}. Steps:
679
658
  1. \`search_player\` to get their FIDE ID.
680
659
  2. \`get_player_profile\` — pull rating history, career splits by color and time control, opening repertoire, opponent analysis, top events, notable wins and losses.
681
660
  3. Weight the data: recent (last 12-24 months) > older, classical OTB > rapid/blitz > online.
682
661
  4. \`prepare_opponent\` twice (once per colour, or once with two sources), then \`get_prep_position(session_token, node_id="r")\` to summarise their opening choices with actual frequencies and win rates. Filter with \`start_month\` if you only care about their current repertoire.
683
662
  5. Deliver: current strength, characteristic openings, one-sentence style read, biggest wins, biggest losses / recurring weakness. Cite the numbers.`;
684
- break;
685
- }
686
- case "engine_usage_primer":
687
- text = ENGINE_USAGE_DOC;
688
- break;
689
- case "prep_strategy_primer":
690
- text = PREP_STRATEGY_DOC;
691
- break;
692
- case "head_to_head_briefing": {
693
- const a = promptArgs.player_a ?? "player A";
694
- const b = promptArgs.player_b ?? "player B";
695
- text = `Briefing on the ${a} vs ${b} history. Steps:
663
+ break;
664
+ }
665
+ case "engine_usage_primer":
666
+ text = ENGINE_USAGE_DOC;
667
+ break;
668
+ case "prep_strategy_primer":
669
+ text = PREP_STRATEGY_DOC;
670
+ break;
671
+ case "head_to_head_briefing": {
672
+ const a = promptArgs.player_a ?? "player A";
673
+ const b = promptArgs.player_b ?? "player B";
674
+ text = `Briefing on the ${a} vs ${b} history. Steps:
696
675
  1. Resolve both FIDE IDs with \`search_player\`.
697
676
  2. \`get_head_to_head\` for the pair — pull overall + per-color W/D/L (from ${a}'s perspective), splits by time format, first / last meeting, most-played openings between them, average game length.
698
677
  3. Read the pattern: who has the edge, in which colour, in which time format. Which openings decide the meetings? Anything unusual — very drawish, very sharp, big rating gap?
699
678
  4. If either player is currently live in a tournament, note it with \`list_player_live_tournaments\`.
700
679
  5. Deliver a one-paragraph read: score, dominant openings, one-line style clash, current form.`;
701
- break;
680
+ break;
681
+ }
682
+ default:
683
+ throw new Error(`Unknown prompt: ${name}`);
702
684
  }
703
- default:
704
- throw new Error(`Unknown prompt: ${name}`);
705
- }
706
- return {
707
- description: `chessceo prompt: ${name}`,
708
- messages: [
709
- { role: "user", content: { type: "text", text } },
710
- ],
711
- };
712
- });
713
- server.setRequestHandler(CallToolRequestSchema, async (req) => {
714
- const { name, arguments: args } = req.params;
715
- try {
716
- const result = await callTool(name, (args ?? {}));
717
685
  return {
718
- content: [{ type: "text", text: JSON.stringify(result, null, 2) }],
686
+ description: `chessceo prompt: ${name}`,
687
+ messages: [
688
+ { role: "user", content: { type: "text", text } },
689
+ ],
719
690
  };
720
- }
721
- catch (err) {
722
- return {
723
- isError: true,
724
- content: [{ type: "text", text: err instanceof Error ? err.message : String(err) }],
725
- };
726
- }
727
- });
691
+ });
692
+ server.setRequestHandler(CallToolRequestSchema, async (req) => {
693
+ const { name, arguments: args } = req.params;
694
+ try {
695
+ const result = await callTool(name, (args ?? {}));
696
+ return {
697
+ content: [{ type: "text", text: JSON.stringify(result, null, 2) }],
698
+ };
699
+ }
700
+ catch (err) {
701
+ return {
702
+ isError: true,
703
+ content: [{ type: "text", text: err instanceof Error ? err.message : String(err) }],
704
+ };
705
+ }
706
+ });
707
+ return server;
708
+ }
728
709
  // ── Transport selection ────────────────────────────────────────────
729
710
  //
730
711
  // Two modes:
@@ -749,7 +730,7 @@ const arg = (name, def) => {
749
730
  };
750
731
  const transportKind = (arg("transport", process.env.MCP_TRANSPORT ?? "stdio") ?? "stdio").toLowerCase();
751
732
  if (transportKind === "stdio") {
752
- await server.connect(new StdioServerTransport());
733
+ await createServer().connect(new StdioServerTransport());
753
734
  }
754
735
  else if (transportKind === "http" || transportKind === "streamable-http") {
755
736
  const port = Number(arg("http-port", process.env.MCP_HTTP_PORT ?? "8080"));
@@ -847,7 +828,11 @@ else if (transportKind === "http" || transportKind === "streamable-http") {
847
828
  sessionIdGenerator: undefined,
848
829
  enableJsonResponse: true,
849
830
  });
850
- res.on("close", () => transport.close());
831
+ const server = createServer();
832
+ res.on("close", () => {
833
+ void transport.close();
834
+ void server.close();
835
+ });
851
836
  await server.connect(transport);
852
837
  // Forward the caller's Authorization header down to tool handlers so
853
838
  // they can attach it when calling authenticated backend endpoints.