@chessceo/mcp 0.49.11 → 0.50.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/analysis/auto.js +173 -182
- package/dist/analysis/cloud.js +182 -0
- package/dist/analysis/deep.js +39 -66
- package/dist/analysis/file_handle.js +43 -36
- package/dist/analysis/response.js +180 -128
- package/dist/index.js +28 -56
- package/dist/pgn/exporter.js +13 -11
- package/dist/pgn/mutations.js +18 -5
- package/dist/pgn/parser.js +14 -15
- package/dist/pgn/types.js +2 -0
- package/dist/tools.js +75 -100
- package/docs/engine-usage.md +38 -40
- package/docs/pgn-authoring.md +3 -3
- package/docs/summary-authoring.md +1 -1
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -22,10 +22,11 @@ import { TOOLS } from "./tools.js";
|
|
|
22
22
|
import { PROMPTS } from "./prompts.js";
|
|
23
23
|
import { commentAntiPatterns, longLineWarning, noDescribeWarning, noStatsCheckWarning, positionalNagOnIntermediateWarning, positionsDescribed, positionsStatsChecked, } from "./warnings.js";
|
|
24
24
|
import { authContext, authedRequest, deleteGame, fetchGame, get, makeFileId, restoreGame, } from "./http.js";
|
|
25
|
-
import {
|
|
25
|
+
import { fetchCompactEval, } from "./analysis/response.js";
|
|
26
|
+
import { cloudAnalyse, ensureEngines } from "./analysis/cloud.js";
|
|
26
27
|
import { autoEvaluate, autoEvaluateCancel, autoEvaluateStatus, } from "./analysis/auto.js";
|
|
27
28
|
import { deepAnalyseCancel, deepAnalyseStart, deepAnalyseStatus, } from "./analysis/deep.js";
|
|
28
|
-
import {
|
|
29
|
+
import { getNodeByPath, resolveFromNodeOrFen, } from "./analysis/file_handle.js";
|
|
29
30
|
import { findPositionInCourses, readCourseAtPosition, runSfEval } from "./courses.js";
|
|
30
31
|
import { applyBatchMutations, applyMutation, argNodeId, } from "./prep/mutations.js";
|
|
31
32
|
import { listNodes, listTranspositions, readPrepFile, } from "./prep/read.js";
|
|
@@ -41,6 +42,8 @@ const AUTHED_TOOLS = new Set([
|
|
|
41
42
|
"list_cloud_engines",
|
|
42
43
|
"stop_cloud_engine",
|
|
43
44
|
"cloud_analyse",
|
|
45
|
+
"ensure_engines",
|
|
46
|
+
"list_cloud_machine_options",
|
|
44
47
|
"list_collections",
|
|
45
48
|
"create_collection",
|
|
46
49
|
"rename_collection",
|
|
@@ -125,11 +128,22 @@ const ENGINE_GUIDE = {
|
|
|
125
128
|
stockfish: "Objective calculation, the default cloud engine at about 100 million nodes per second. A few seconds gives a very strong view of the position. Use it on every position, including when steering by lc0 or human. Its score is the objective value with best defense: 0.00 means objectively a draw with best play, not that nothing is happening or that neither side has chances. Many positions that are very hard to play for a human score 0.00. Scores drift toward 0.00 as the search deepens, especially in drawn positions. Use it to check whether a tactic is real, whether a defense holds, and whether a move loses material or the game.",
|
|
126
129
|
lc0: "Neural-net engine trained on self-play games, not human games. It is better than Stockfish at long-term ideas and at judging which side is practically on top. Never gives 0.00: it always says which side prefers the position, at least a little. Use it when Stockfish gives 0.00 to see who still has the practical edge, and to find where the long-term ideas are. Contempt changes how it values a draw and which side it plays for: contempt at -20 makes Black play for a win, which is useful for finding ideas, including in openings. It is much weaker tactically than Stockfish, so confirm any concrete line with Stockfish.",
|
|
127
130
|
human: "Neural net trained only on human games, about 11 million rated games. Its moves and evaluations are practical, not objective: the moves it suggests are the ones a human would find and play. It punishes dubious moves less than Stockfish, because human games contain many imperfect moves that still work in practice, so a move with a sound idea behind it can score well here. Use it to predict how a person will respond to an idea, to find ideas (it may like a move that only fails to a refutation no human would find; Stockfish shows whether that refutation exists), and to see how a human would evaluate a position, especially positional or unclear ones. Like lc0 it is weaker in very unusual positions, and its evaluations there can be trusted much less. Stockfish remains the check on tactics and on whether a practical idea is objectively sound.",
|
|
128
|
-
combo: "Runs Stockfish and Lc0 together — use when you want both the objective read and the practical/human-feel read on the same position.",
|
|
129
131
|
};
|
|
130
132
|
// Modern engines give equality very often, so small numbers are not "equal".
|
|
131
133
|
// Rough guide, not exact thresholds. Shown alongside ENGINE_GUIDE above.
|
|
132
134
|
const ENGINE_EVAL_SCALE = "Small numbers are not equality. ±0.00 to ±0.10 is equal in practice (+0.10 is '=' or '+=' at most). +0.20 to +0.40 is real pressure, not 'a bit better'. +0.50 and up is a clear advantage. Stockfish 0.00 with a plus from Lc0 is a practical edge, not equality. Check both engines before calling a position equal.";
|
|
135
|
+
// How to run analysis end to end. Authed response only, with ENGINE_GUIDE.
|
|
136
|
+
const ENGINE_WORKFLOW = [
|
|
137
|
+
"1. Plan: ensure_engines({engines, positions}) shows what is running, what to start, the price and an estimate. It starts nothing.",
|
|
138
|
+
"2. Ask: show the user the price of anything you would start and get a yes. Rentals bill per minute until stopped.",
|
|
139
|
+
"3. Start: start_cloud_engine({machine_type}) per missing engine. Each SKU runs one engine; the engine comes from the SKU. Wait until list_cloud_engines shows it running (static engines are instant, others ~1-5 min).",
|
|
140
|
+
"4. Analyse: cloud_analyse({fens | lines | file_id+node_ids, engines}) for up to 10 positions, each engine on its own rental, in parallel. auto_evaluate({id, engines}) for a whole file or subtree: fills only the engines each node is missing and saves as it goes. deep_analyse({engine, ...}) for one long think.",
|
|
141
|
+
"5. Read back: quote_engine_eval gives the stored sf (cp/mate), lc0 and human (win/draw/loss %) per node.",
|
|
142
|
+
"6. Stop: stop_cloud_engine when the work is done, unless the user wants it kept running.",
|
|
143
|
+
].join("\n");
|
|
144
|
+
// Eval symbols are the writer's call, not the engine's. Recommendation
|
|
145
|
+
// only; nothing in the MCP sets a symbol automatically.
|
|
146
|
+
const EVAL_SYMBOL_GUIDE = "The eval symbol (=, +=, ±, +-) is your call, written for a human reader. The engines inform it; no tool sets it for you. Stockfish 0.00 can still be += when Lc0 or the human engine clearly prefer one side, because the symbol describes the practical picture, not only the objective one. Look at all engines you ran before choosing, and don't put a symbol on a position nobody analysed.";
|
|
133
147
|
// v0.48: consolidated the five `read_*_guide` / `read_example_prep_files`
|
|
134
148
|
// tools into ONE `read_docs`. LLM lists what it wants; we return them
|
|
135
149
|
// in a single response. Also cleaner: enumerating available docs in one
|
|
@@ -390,68 +404,26 @@ async function callToolInner(name, args) {
|
|
|
390
404
|
...(resp && typeof resp === "object" ? resp : { options: [] }),
|
|
391
405
|
engine_guide: ENGINE_GUIDE,
|
|
392
406
|
eval_scale: ENGINE_EVAL_SCALE,
|
|
407
|
+
eval_symbols: EVAL_SYMBOL_GUIDE,
|
|
408
|
+
workflow: ENGINE_WORKFLOW,
|
|
393
409
|
};
|
|
394
410
|
}
|
|
411
|
+
case "ensure_engines":
|
|
412
|
+
return {
|
|
413
|
+
...await ensureEngines(args),
|
|
414
|
+
engine_guide: ENGINE_GUIDE,
|
|
415
|
+
workflow: ENGINE_WORKFLOW,
|
|
416
|
+
};
|
|
395
417
|
case "start_cloud_engine":
|
|
396
418
|
return authedRequest("POST", "/api/agent/cloud-engines", {
|
|
397
419
|
machineType: String(args.machine_type),
|
|
398
|
-
...(typeof args.engine_type === "string" ? { engineType: args.engine_type } : {}),
|
|
399
420
|
});
|
|
400
421
|
case "list_cloud_engines":
|
|
401
422
|
return authedRequest("GET", "/api/agent/cloud-engines");
|
|
402
423
|
case "stop_cloud_engine":
|
|
403
424
|
return authedRequest("DELETE", `/api/agent/cloud-engines/${encodeURIComponent(String(args.contract_id))}`);
|
|
404
|
-
case "cloud_analyse":
|
|
405
|
-
|
|
406
|
-
const fen = resolved.fen;
|
|
407
|
-
const body = { fen };
|
|
408
|
-
if (typeof args.movetime_ms === "number")
|
|
409
|
-
body.movetime_ms = args.movetime_ms;
|
|
410
|
-
if (typeof args.stockfish_multipv === "number")
|
|
411
|
-
body.stockfish_multipv = args.stockfish_multipv;
|
|
412
|
-
if (typeof args.lc0_multipv === "number")
|
|
413
|
-
body.lc0_multipv = args.lc0_multipv;
|
|
414
|
-
if (typeof args.contempt === "number")
|
|
415
|
-
body.contempt = args.contempt;
|
|
416
|
-
Object.assign(body, analyseRouting(args));
|
|
417
|
-
const raw = await authedRequest("POST", "/api/agent/cloud-engines/analyse", body);
|
|
418
|
-
const converted = convertCloudSnapshotResponse(raw, fen);
|
|
419
|
-
// PV cap: engine PVs beyond ~6 plies are speculative (the tail is
|
|
420
|
-
// where the search's confidence collapses — SF at depth 24 has
|
|
421
|
-
// seen the first few plies solidly and hedged everything after).
|
|
422
|
-
// More importantly, LLMs paste long PVs into `add_line` as if
|
|
423
|
-
// they were prepared repertoire. A 15-move PV pasted as a
|
|
424
|
-
// variation is one line of engine output through positions
|
|
425
|
-
// where both sides had real choices — not a repertoire. Cap the
|
|
426
|
-
// affordance: return only what's load-bearing (3 full moves for
|
|
427
|
-
// understanding the point), let the caller re-analyse the
|
|
428
|
-
// resulting position if they want to see further. Override via
|
|
429
|
-
// `pv_max_plies` for the rare case (deep tactics verification).
|
|
430
|
-
const pvMaxPlies = typeof args.pv_max_plies === "number" && args.pv_max_plies > 0
|
|
431
|
-
? Math.min(args.pv_max_plies, 40)
|
|
432
|
-
: 6;
|
|
433
|
-
capPvsInResponse(converted, pvMaxPlies);
|
|
434
|
-
// Node-addressed calls: persist the result on the node's ceoEval
|
|
435
|
-
// so a later quote_engine_eval can cite this measurement. This is
|
|
436
|
-
// the anti-hallucination hinge — prose that says "engines say X
|
|
437
|
-
// on node Y" can only trace back to a call actually made against
|
|
438
|
-
// node_id=Y, because the store only fires when file_id+node_id
|
|
439
|
-
// was supplied and the eval survives via the [%ceo-eval] escape.
|
|
440
|
-
if (resolved.file) {
|
|
441
|
-
const ev = analysisToStoredEval(converted);
|
|
442
|
-
if (ev) {
|
|
443
|
-
const stamped = await storeEvalOnNode(resolved.file, ev);
|
|
444
|
-
if (stamped.length > 1) {
|
|
445
|
-
// Surface the propagation so the LLM sees exactly which
|
|
446
|
-
// other nodes now carry this eval (and can skip them for
|
|
447
|
-
// re-analysis).
|
|
448
|
-
converted.also_stored_on = stamped.slice(1);
|
|
449
|
-
}
|
|
450
|
-
}
|
|
451
|
-
}
|
|
452
|
-
converted.eval_scale = ENGINE_EVAL_SCALE;
|
|
453
|
-
return converted;
|
|
454
|
-
}
|
|
425
|
+
case "cloud_analyse":
|
|
426
|
+
return cloudAnalyse(args, ENGINE_EVAL_SCALE);
|
|
455
427
|
case "deep_analyse":
|
|
456
428
|
return deepAnalyseStart(args);
|
|
457
429
|
case "deep_analyse_status":
|
|
@@ -562,7 +534,7 @@ async function callToolInner(name, args) {
|
|
|
562
534
|
case "apply_mutations":
|
|
563
535
|
return applyBatchMutations(args);
|
|
564
536
|
case "auto_evaluate":
|
|
565
|
-
return autoEvaluate(args
|
|
537
|
+
return autoEvaluate(args);
|
|
566
538
|
case "auto_evaluate_status":
|
|
567
539
|
return autoEvaluateStatus(args);
|
|
568
540
|
case "auto_evaluate_cancel":
|
package/dist/pgn/exporter.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// PrepFile → PGN text. Deterministic — same tree always exports to the
|
|
2
2
|
// same PGN modulo whitespace, so paths stay stable across
|
|
3
3
|
// export→reparse cycles.
|
|
4
|
-
import { codeFromColor } from "./types.js";
|
|
4
|
+
import { codeFromColor, STORED_EVAL_KEYS } from "./types.js";
|
|
5
5
|
// Standard PGN "seven tag roster" order. We emit these first (in the
|
|
6
6
|
// order they appear in tags), then any extra tags in insertion order.
|
|
7
7
|
const STR_ORDER = ["Event", "Site", "Date", "Round", "White", "Black", "Result"];
|
|
@@ -142,22 +142,24 @@ function renderNodeAnnotation(node) {
|
|
|
142
142
|
}
|
|
143
143
|
return bits.join(" ");
|
|
144
144
|
}
|
|
145
|
-
// Serialise a stored eval as `[%ceo-eval sf=+0.20/38 lc0
|
|
146
|
-
// Compact: no PVs (regenerable
|
|
147
|
-
// readability, mate as `MN`/`-MN`, depth as `/N`.
|
|
145
|
+
// Serialise a stored eval as `[%ceo-eval sf=+0.20/38 lc0=W38D45L17/24
|
|
146
|
+
// human=W30D50L20/20]`. Compact: no PVs (regenerable), decimal cp for
|
|
147
|
+
// readability, mate as `MN`/`-MN`, WDL as `W<w>D<d>L<l>`, depth as `/N`.
|
|
148
148
|
function renderCeoEval(ev) {
|
|
149
149
|
const bits = [];
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
bits.push(`nag=${ev.nag}`);
|
|
150
|
+
for (const k of STORED_EVAL_KEYS) {
|
|
151
|
+
const e = ev[k];
|
|
152
|
+
if (e)
|
|
153
|
+
bits.push(`${k}=${formatEngineEval(e)}`);
|
|
154
|
+
}
|
|
156
155
|
return bits.length > 0 ? `[%ceo-eval ${bits.join(" ")}]` : "";
|
|
157
156
|
}
|
|
158
157
|
function formatEngineEval(e) {
|
|
159
158
|
let body;
|
|
160
|
-
if (typeof e.
|
|
159
|
+
if (typeof e.w === "number" && typeof e.d === "number" && typeof e.l === "number") {
|
|
160
|
+
body = `W${e.w}D${e.d}L${e.l}`;
|
|
161
|
+
}
|
|
162
|
+
else if (typeof e.mate === "number") {
|
|
161
163
|
body = e.mate >= 0 ? `+M${e.mate}` : `M${e.mate}`; // "+M5" / "M-5"
|
|
162
164
|
}
|
|
163
165
|
else if (typeof e.cp === "number") {
|
package/dist/pgn/mutations.js
CHANGED
|
@@ -12,6 +12,7 @@ import { Chess } from "chessops/chess";
|
|
|
12
12
|
import { parseSan } from "chessops/san";
|
|
13
13
|
import { makeFen } from "chessops/fen";
|
|
14
14
|
import { cloneOnPath, deriveNodeId, getNode, PathError } from "./paths.js";
|
|
15
|
+
import { STORED_EVAL_KEYS } from "./types.js";
|
|
15
16
|
export class MutationError extends Error {
|
|
16
17
|
}
|
|
17
18
|
// Add a SAN move as a new child under `parentPath`. **Idempotent on
|
|
@@ -162,14 +163,26 @@ export function promoteVariation(file, path) {
|
|
|
162
163
|
parent.children = [promoted, ...rest];
|
|
163
164
|
return { file: { tags: file.tags, root: newRoot }, id: promoted.id };
|
|
164
165
|
}
|
|
165
|
-
//
|
|
166
|
-
//
|
|
166
|
+
// Merge engine evals into the stored ceoEval on the node at `path`: only
|
|
167
|
+
// the engines present in `ev` are replaced, the others are kept, so a
|
|
168
|
+
// Stockfish deep think never wipes a stored human or Lc0 read. Passing null
|
|
169
|
+
// clears every engine.
|
|
167
170
|
export function setCeoEval(file, path, ev) {
|
|
168
171
|
const { root: newRoot, target } = cloneOnPath(file.root, path);
|
|
169
|
-
if (ev === null
|
|
172
|
+
if (ev === null) {
|
|
170
173
|
delete target.ceoEval;
|
|
171
|
-
|
|
172
|
-
|
|
174
|
+
}
|
|
175
|
+
else {
|
|
176
|
+
const merged = { ...(target.ceoEval ?? {}) };
|
|
177
|
+
for (const k of STORED_EVAL_KEYS) {
|
|
178
|
+
if (ev[k])
|
|
179
|
+
merged[k] = ev[k];
|
|
180
|
+
}
|
|
181
|
+
if (merged.sf || merged.lc0 || merged.human)
|
|
182
|
+
target.ceoEval = merged;
|
|
183
|
+
else
|
|
184
|
+
delete target.ceoEval;
|
|
185
|
+
}
|
|
173
186
|
return { file: { tags: file.tags, root: newRoot }, id: target.id };
|
|
174
187
|
}
|
|
175
188
|
// Set the same ceoEval on every path in `paths`. One clone-and-return
|
package/dist/pgn/parser.js
CHANGED
|
@@ -24,10 +24,17 @@ const CEO_EVAL_RE = /\[%ceo-eval\s+([^\]]+)\]/g;
|
|
|
24
24
|
// `{[%eval] [%wdl]}` comment on an imported Lichess PGN got stripped
|
|
25
25
|
// on a set_tag save cycle).
|
|
26
26
|
const KNOWN_CMD_RE = /\[%(?:cal|csl|ceo-eval)\s+[^\]]+\]/g;
|
|
27
|
-
// Parse a `sf=+0.20/38
|
|
28
|
-
// Returns null if the value doesn't parse;
|
|
29
|
-
//
|
|
27
|
+
// Parse a `sf=+0.20/38`, `sf=M-5/22` or `lc0=W38D45L17/24` fragment into
|
|
28
|
+
// its numeric parts. Returns null if the value doesn't parse; depth is
|
|
29
|
+
// optional.
|
|
30
30
|
function parseEngineEvalFragment(raw) {
|
|
31
|
+
const wdl = raw.match(/^W(\d+)D(\d+)L(\d+)(?:\/(\d+))?$/i);
|
|
32
|
+
if (wdl) {
|
|
33
|
+
const out = { w: parseInt(wdl[1], 10), d: parseInt(wdl[2], 10), l: parseInt(wdl[3], 10) };
|
|
34
|
+
if (wdl[4] !== undefined)
|
|
35
|
+
out.depth = parseInt(wdl[4], 10);
|
|
36
|
+
return out;
|
|
37
|
+
}
|
|
31
38
|
const m = raw.match(/^([+-]?)(?:M(-?\d+)|(\d+(?:\.\d+)?))(?:\/(\d+))?$/i);
|
|
32
39
|
if (!m)
|
|
33
40
|
return null;
|
|
@@ -75,22 +82,14 @@ export function parseCommentAnnotations(comment) {
|
|
|
75
82
|
continue;
|
|
76
83
|
const k = entry.slice(0, eq).toLowerCase();
|
|
77
84
|
const v = entry.slice(eq + 1);
|
|
78
|
-
if (k === "sf") {
|
|
85
|
+
if (k === "sf" || k === "lc0" || k === "human") {
|
|
79
86
|
const parsed = parseEngineEvalFragment(v);
|
|
80
87
|
if (parsed)
|
|
81
|
-
ev
|
|
82
|
-
}
|
|
83
|
-
else if (k === "lc0") {
|
|
84
|
-
const parsed = parseEngineEvalFragment(v);
|
|
85
|
-
if (parsed)
|
|
86
|
-
ev.lc0 = parsed;
|
|
87
|
-
}
|
|
88
|
-
else if (k === "nag") {
|
|
89
|
-
if (/^\$\d+$/.test(v))
|
|
90
|
-
ev.nag = v;
|
|
88
|
+
ev[k] = parsed;
|
|
91
89
|
}
|
|
90
|
+
// Anything else (incl. the pre-0.50 `nag=`) is dropped.
|
|
92
91
|
}
|
|
93
|
-
if (ev.sf || ev.lc0 || ev.
|
|
92
|
+
if (ev.sf || ev.lc0 || ev.human)
|
|
94
93
|
ceoEval = ev;
|
|
95
94
|
}
|
|
96
95
|
const text = comment.replace(KNOWN_CMD_RE, "").replace(/\s+/g, " ").trim();
|
package/dist/pgn/types.js
CHANGED
|
@@ -19,6 +19,8 @@
|
|
|
19
19
|
// width, buildIdIndex keeps the first occurrence and logs to stderr
|
|
20
20
|
// rather than throwing — the file stays readable, only mutations
|
|
21
21
|
// against the collided id are ambiguous.
|
|
22
|
+
// Keys of StoredEval in canonical order.
|
|
23
|
+
export const STORED_EVAL_KEYS = ["sf", "lc0", "human"];
|
|
22
24
|
// Named colours as they appear in the parsed tree. The wire format uses
|
|
23
25
|
// single-letter codes (G, R, Y, C, B, O); the tree uses these longer
|
|
24
26
|
// names to match how the frontend represents them.
|
package/dist/tools.js
CHANGED
|
@@ -118,7 +118,7 @@ export const TOOLS = [
|
|
|
118
118
|
"- `gm-classical` — GM classical games (both players ≥2500, real thinking-time). BEST for opening prep — every move is signal, avgElo ~2600 across all listed moves.\n" +
|
|
119
119
|
"- `main` — the whole 11.7M-game DB. Widest coverage but noisiest (includes 1000-Elo blunder-fests in the move stats). Use as fallback when gm-classical's totalCount is too small to be informative.\n\n" +
|
|
120
120
|
"Game movetext is trimmed to the moves AFTER the queried position (using each game's plyNumber). Saves ~70% of the bytes vs full movetext.\n\n" +
|
|
121
|
-
"AUTO-EVAL: if a
|
|
121
|
+
"AUTO-EVAL: if a Stockfish rental is running, the response includes `.eval` with a short Stockfish read (cp/mate White's point of view, best move, PV). Not stored on any node; for a stored eval or other engines use cloud_analyse.",
|
|
122
122
|
inputSchema: {
|
|
123
123
|
type: "object",
|
|
124
124
|
properties: {
|
|
@@ -256,27 +256,34 @@ export const TOOLS = [
|
|
|
256
256
|
required: ["fide_id"],
|
|
257
257
|
},
|
|
258
258
|
},
|
|
259
|
+
{
|
|
260
|
+
name: "ensure_engines",
|
|
261
|
+
description: "Plan cloud-engine work before running it. For each engine you need (`stockfish`, `lc0`, `human`), says whether a rental providing it is already running (with its contract_id), or the cheapest SKU to start and its price. Pass `positions` (how many positions you will analyse) for a time and cost estimate. Starts nothing.\n\n" +
|
|
262
|
+
"CALL FIRST whenever a task needs cloud engines. Then show the user the price of anything to start, get a yes, and start it with start_cloud_engine. The response also carries `engine_guide` (what each engine is for) and `workflow` (the full plan → start → analyse → stop sequence).",
|
|
263
|
+
inputSchema: {
|
|
264
|
+
type: "object",
|
|
265
|
+
properties: {
|
|
266
|
+
engines: { type: "array", items: { type: "string", enum: ["stockfish", "lc0", "human"] }, description: "Engines the task needs. Default [\"stockfish\"]." },
|
|
267
|
+
positions: { type: "integer", minimum: 1, description: "Optional: how many positions you plan to analyse, for the estimate." },
|
|
268
|
+
movetime_ms: { type: "integer", minimum: 100, maximum: 300000, description: "Optional: think time per position for the estimate (default 2000)." },
|
|
269
|
+
},
|
|
270
|
+
},
|
|
271
|
+
},
|
|
259
272
|
{
|
|
260
273
|
name: "list_cloud_machine_options",
|
|
261
|
-
description: "Returns the catalog of cloud-engine
|
|
274
|
+
description: "Returns the catalog of cloud-engine SKUs the user can start, same visibility as the chess.ceo app: SKU (`machineType`), the one engine it runs (`engine`: stockfish, lc0 or human), display name, cost per hour, availability. Also returns `engine_guide`, `eval_scale`, `eval_symbols` and `workflow`. ensure_engines already picks the cheapest SKU per engine; use this when the user wants to choose hardware. SKU strings are NOT guessable from display names; pass them verbatim to start_cloud_engine.",
|
|
262
275
|
inputSchema: { type: "object", properties: {} },
|
|
263
276
|
},
|
|
264
277
|
{
|
|
265
278
|
name: "start_cloud_engine",
|
|
266
|
-
description: "Rent
|
|
267
|
-
"
|
|
268
|
-
"Use list_cloud_engines first to check if the user already has one running; don't start a second instance unless the user asked for it. Requires an MCP token with agent access.",
|
|
279
|
+
description: "Rent one cloud engine on the user's chess.ceo account. Each SKU runs exactly one engine (stockfish, lc0 or human), taken from the SKU. Real money: billed per minute from start until stop_cloud_engine.\n\n" +
|
|
280
|
+
"`machine_type` must be an exact SKU from ensure_engines or list_cloud_machine_options. Show the user the price and get their confirmation first. For several engines, start one rental per engine; cloud_analyse uses them together. Don't start an engine that is already running (ensure_engines / list_cloud_engines show what is).",
|
|
269
281
|
inputSchema: {
|
|
270
282
|
type: "object",
|
|
271
283
|
properties: {
|
|
272
284
|
machine_type: {
|
|
273
285
|
type: "string",
|
|
274
|
-
description: "SKU from
|
|
275
|
-
},
|
|
276
|
-
engine_type: {
|
|
277
|
-
type: "string",
|
|
278
|
-
enum: ["stockfish", "lc0", "combo"],
|
|
279
|
-
description: "Which engine(s) to run. Must match the SKU's own `engine` field from list_cloud_machine_options — e.g. a stockfish-only SKU can only be started as \"stockfish\". Optional; defaults to \"combo\".",
|
|
286
|
+
description: "Exact SKU from ensure_engines or list_cloud_machine_options. Not a display name, not a guess.",
|
|
280
287
|
},
|
|
281
288
|
},
|
|
282
289
|
required: ["machine_type"],
|
|
@@ -284,12 +291,12 @@ export const TOOLS = [
|
|
|
284
291
|
},
|
|
285
292
|
{
|
|
286
293
|
name: "list_cloud_engines",
|
|
287
|
-
description: "List the user's
|
|
294
|
+
description: "List the user's running cloud engines, each with its `contractId` and `engines` (what it provides). A row with empty `engines` is still starting. Use the contractId for stop_cloud_engine, or in `rentals` when two running rentals provide the same engine.",
|
|
288
295
|
inputSchema: { type: "object", properties: {} },
|
|
289
296
|
},
|
|
290
297
|
{
|
|
291
298
|
name: "stop_cloud_engine",
|
|
292
|
-
description: "Destroy a running cloud engine. Billing stops immediately. Use the contract_id from `list_cloud_engines
|
|
299
|
+
description: "Destroy a running cloud engine. Billing stops immediately. Use the contract_id from `list_cloud_engines`, don't guess. Stop rentals when the analysis is done unless the user wants them kept.",
|
|
293
300
|
inputSchema: {
|
|
294
301
|
type: "object",
|
|
295
302
|
properties: {
|
|
@@ -303,65 +310,38 @@ export const TOOLS = [
|
|
|
303
310
|
},
|
|
304
311
|
{
|
|
305
312
|
name: "cloud_analyse",
|
|
306
|
-
description: "
|
|
307
|
-
"
|
|
308
|
-
"
|
|
309
|
-
"
|
|
310
|
-
"
|
|
311
|
-
"
|
|
312
|
-
"
|
|
313
|
-
"
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
type: "integer",
|
|
331
|
-
minimum: 100,
|
|
332
|
-
maximum: 10000,
|
|
333
|
-
description: "Think time in milliseconds (default 2000).",
|
|
334
|
-
},
|
|
335
|
-
stockfish_multipv: {
|
|
336
|
-
type: "integer",
|
|
337
|
-
minimum: 1,
|
|
338
|
-
maximum: 10,
|
|
339
|
-
description: "Stockfish candidate lines (default 2). Kept tight because each extra PV steals search bandwidth from the top choice — SF is the 'what's objectively best' leg, use a low multipv to keep it strong. Raise only when you specifically need SF's take on a wide range of candidates.",
|
|
340
|
-
},
|
|
341
|
-
lc0_multipv: {
|
|
342
|
-
type: "integer",
|
|
343
|
-
minimum: 1,
|
|
344
|
-
maximum: 10,
|
|
345
|
-
description: "Lc0 candidate lines (default 8). Kept wide because multipv doesn't degrade Lc0's strength the way it does Stockfish's — Lc0 is the 'find inspiration / explore practical tries' leg, use a high multipv to get a full slate of ideas.",
|
|
346
|
-
},
|
|
347
|
-
contempt: {
|
|
348
|
-
type: "integer",
|
|
349
|
-
minimum: -100,
|
|
350
|
-
maximum: 100,
|
|
351
|
-
description: "Lc0 contempt bias. Signed 0-100 strength (same scale as the web UI's ContemptStrength slider — server multiplies by 8 to get the internal cp bias). 0 = objective (default). Positive favours White, negative favours Black. Typical: ±15 light nudge, ±30-60 real fighting play, ±80-100 maximum steer. Not applied to Stockfish. See engine_usage_primer for when to use.",
|
|
352
|
-
},
|
|
353
|
-
contract_id: { type: "string", description: "Which of your running rentals to use, from list_cloud_engines. Any engine shape works (combo, stockfish-only, lc0-only). Omit it when only one rental can serve the request; if several can, the error lists their contract ids." },
|
|
354
|
-
engines: {
|
|
355
|
-
type: "array",
|
|
356
|
-
items: { type: "string", enum: ["stockfish", "lc0"] },
|
|
357
|
-
description: "Which engines to run. Default = both. Use `[\"lc0\"]` to skip Stockfish (e.g. while a deep_analyse job is holding the SF slot on the same rental). Use `[\"stockfish\"]` when only the objective read matters. The skipped engine's field is omitted from the response.",
|
|
313
|
+
description: "Analyse up to 10 positions on the engines you choose (`stockfish`, `lc0`, `human`), each engine on whichever running rental provides it. Engines run in parallel per position. Returns, per position, each engine's depth, candidate lines and best move.\n\n" +
|
|
314
|
+
"Positions: `fens` (list), `lines` (list of SAN move sequences from `fen` or the start), `file_id` + `node_ids` (list), or a single `fen` / `moves` / `file_id`+`node_id`. Examples: analyse three FENs with Stockfish → `{fens: [a, b, c], engines: [\"stockfish\"]}`; with the human engine and Stockfish → `engines: [\"stockfish\", \"human\"]`. For a whole file or subtree use auto_evaluate instead.\n\n" +
|
|
315
|
+
"Scores: `scoreCp` and `mate` are White's point of view (+20 = White +0.20; mate +5 = White mates in 5). lc0 and human lines carry `wdl`: win/draw/loss percent, White's point of view. Read the response's `eval_scale` before calling a position equal.\n\n" +
|
|
316
|
+
"GROUNDING: every claim about a position must trace back to engine output from this session. Don't invent evaluations, best moves or variations. When you have no data for a position, run it or say so.\n\n" +
|
|
317
|
+
"Storing: with `file_id`, each result is merged into the `ceoEval` of every node that reaches that position (`stored_on` in the response), keeping other engines' stored reads. quote_engine_eval cites them later.\n\n" +
|
|
318
|
+
"Contempt (`contempt`, lc0 only): signed -100..100, positive favours White. Use it to find ideas (e.g. -20 makes Black play for a win). Never quote a contempt eval as objective.\n\n" +
|
|
319
|
+
"PVs are capped at 6 plies (`pv_truncated: true` when cut). To see further, analyse the position at the end of the line; raise `pv_max_plies` only to verify a forcing line. Don't paste PVs into add_line as prep.\n\n" +
|
|
320
|
+
"Needs running rentals for the engines you ask for (see ensure_engines). Costs money while rentals run; use get_position_stats for casual questions.",
|
|
321
|
+
inputSchema: {
|
|
322
|
+
type: "object",
|
|
323
|
+
properties: {
|
|
324
|
+
engines: { type: "array", items: { type: "string", enum: ["stockfish", "lc0", "human"] }, description: "Engines to run. Default [\"stockfish\"]. Each needs a running rental that provides it." },
|
|
325
|
+
fens: { type: "array", items: { type: "string" }, description: "Positions as FENs (up to 10 in total with the other inputs)." },
|
|
326
|
+
lines: { type: "array", items: { type: "string" }, description: "SAN move sequences, each applied from `fen` (or the start position), e.g. [\"e4 c5 Nf3\", \"e4 e5 Nf3\"]." },
|
|
327
|
+
file_id: { type: "string", description: "Prep file id. With node_ids/node_id the FEN comes from the tree; with any input, results are stored on matching nodes." },
|
|
328
|
+
node_ids: { type: "array", items: { type: "string" }, description: "Node ids inside `file_id`." },
|
|
329
|
+
node_id: { type: "string", description: "One node id inside `file_id`. Root is 'r'." },
|
|
330
|
+
fen: { type: "string", description: "One position, or the start position for `moves` / `lines`." },
|
|
331
|
+
moves: { type: "string", description: "SAN moves applied on top of `fen` (or the start position) for a single position." },
|
|
332
|
+
movetime_ms: { type: "integer", minimum: 100, maximum: 15000, description: "Think time per position in ms (default 2000)." },
|
|
333
|
+
multipv: {
|
|
334
|
+
type: "object",
|
|
335
|
+
properties: { stockfish: { type: "integer", minimum: 1, maximum: 10 }, lc0: { type: "integer", minimum: 1, maximum: 10 }, human: { type: "integer", minimum: 1, maximum: 10 } },
|
|
336
|
+
description: "Candidate lines per engine. Defaults: stockfish 2 (extra lines cost it strength), lc0 and human 8 (wide slate of ideas, no strength cost).",
|
|
358
337
|
},
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
description: "
|
|
338
|
+
contempt: { type: "integer", minimum: -100, maximum: 100, description: "lc0 only. Positive favours White, negative Black. 0 = objective (default)." },
|
|
339
|
+
rentals: {
|
|
340
|
+
type: "object",
|
|
341
|
+
properties: { stockfish: { type: "string" }, lc0: { type: "string" }, human: { type: "string" } },
|
|
342
|
+
description: "Only when two running rentals provide the same engine: engine → contract_id from list_cloud_engines.",
|
|
364
343
|
},
|
|
344
|
+
pv_max_plies: { type: "integer", minimum: 1, maximum: 40, description: "Cap each PV (default 6 plies). Raise only to verify a forcing line." },
|
|
365
345
|
},
|
|
366
346
|
},
|
|
367
347
|
},
|
|
@@ -722,25 +702,27 @@ export const TOOLS = [
|
|
|
722
702
|
},
|
|
723
703
|
{
|
|
724
704
|
name: "auto_evaluate",
|
|
725
|
-
description: "
|
|
726
|
-
"
|
|
727
|
-
"
|
|
728
|
-
"
|
|
729
|
-
"Costs real money — one cloud_analyse per node. A 200-node walk at default movetime is ~5 min of engine time (calls serialise on the per-engine semaphore in the backend).",
|
|
705
|
+
description: "Analyse a whole prep file, or the subtree under `node_id`, on the engines you choose and store the evals on every node. Example: \"analyse the whole tree with Stockfish and the human engine and save the evals\" → `{id, engines: [\"stockfish\", \"human\"]}`.\n\n" +
|
|
706
|
+
"Each node gets only the requested engines it is missing (`only_missing`, default true), so adding an engine later runs just that engine. Transpositions are analysed once and stamped on every node that reaches the position. Stored evals merge per engine; other engines' stored reads are kept. Results are saved after every batch of up to 10 positions.\n\n" +
|
|
707
|
+
"**Async job, returns immediately:** `{ job_id, target_count, already_complete, estimated_seconds }`. Poll auto_evaluate_status(job_id); cancel with auto_evaluate_cancel. Do other work between polls.\n\n" +
|
|
708
|
+
"Needs running rentals for every engine you ask for (ensure_engines). Wall time ≈ target_count × movetime (engines run in parallel per position). Does not set visible NAGs: the eval symbol is your call.",
|
|
730
709
|
inputSchema: {
|
|
731
710
|
type: "object",
|
|
732
|
-
properties: {
|
|
733
|
-
id: { type: "string" },
|
|
711
|
+
properties: {
|
|
712
|
+
id: { type: "string", description: "Prep file id." },
|
|
713
|
+
engines: { type: "array", items: { type: "string", enum: ["stockfish", "lc0", "human"] }, description: "Engines to run. Default [\"stockfish\"]." },
|
|
734
714
|
node_id: { type: "string", description: "Subtree root (default 'r' = whole file)." },
|
|
735
|
-
only_missing: { type: "boolean", description: "
|
|
736
|
-
movetime_ms: { type: "integer", minimum: 500, maximum:
|
|
715
|
+
only_missing: { type: "boolean", description: "Run only the engines a node has no stored eval for (default true). false re-analyses everything." },
|
|
716
|
+
movetime_ms: { type: "integer", minimum: 500, maximum: 15000, description: "Think time per position in ms (default 2000)." },
|
|
717
|
+
contempt: { type: "integer", minimum: -100, maximum: 100, description: "lc0 only. Stored evals then carry the bias; use for idea finding, not for the stored record." },
|
|
718
|
+
rentals: { type: "object", properties: { stockfish: { type: "string" }, lc0: { type: "string" }, human: { type: "string" } }, description: "Only when two running rentals provide the same engine: engine → contract_id." },
|
|
737
719
|
},
|
|
738
720
|
required: ["id"],
|
|
739
721
|
},
|
|
740
722
|
},
|
|
741
723
|
{
|
|
742
724
|
name: "auto_evaluate_status",
|
|
743
|
-
description: "Poll the status of an auto_evaluate job. Response: `{ status: 'running' | 'done' | 'cancelled' | 'error' | 'not_found',
|
|
725
|
+
description: "Poll the status of an auto_evaluate job. Response: `{ status: 'running' | 'done' | 'cancelled' | 'error' | 'not_found', engines, target_count, processed, remaining, evaluated: {engine: n}, failed: {engine: n}, failed_node_ids: {engine: [ids]}, last_error?, aborted_reason?, done }`. Retry failures with cloud_analyse({file_id, node_ids}) or a new auto_evaluate. When `status: 'not_found'` the job either expired (kept ~15 min after completion), never existed, or the MCP restarted since it was created — re-run auto_evaluate.\n\n" +
|
|
744
726
|
"Typical poll cadence: every 3-5 s for small walks, every 10-30 s for large ones. Don't hammer — status is a pure in-memory read but polling doesn't speed the engine up.",
|
|
745
727
|
inputSchema: {
|
|
746
728
|
type: "object",
|
|
@@ -752,7 +734,7 @@ export const TOOLS = [
|
|
|
752
734
|
},
|
|
753
735
|
{
|
|
754
736
|
name: "auto_evaluate_cancel",
|
|
755
|
-
description: "Ask a running auto_evaluate job to stop
|
|
737
|
+
description: "Ask a running auto_evaluate job to stop after the batch in flight. Every batch already analysed is saved in the file. Idempotent — cancelling an already-finished job is a no-op with a clear note in the response.",
|
|
756
738
|
inputSchema: {
|
|
757
739
|
type: "object",
|
|
758
740
|
properties: {
|
|
@@ -763,34 +745,27 @@ export const TOOLS = [
|
|
|
763
745
|
},
|
|
764
746
|
{
|
|
765
747
|
name: "deep_analyse",
|
|
766
|
-
description: "Start
|
|
767
|
-
"Use
|
|
768
|
-
"
|
|
748
|
+
description: "Start one long think (up to 5 min) on one position, on one engine (`engine`, default stockfish). Returns a `job_id` immediately; poll deep_analyse_status, cancel with deep_analyse_cancel. It holds only that engine, so cloud_analyse on the other engines keeps working meanwhile.\n\n" +
|
|
749
|
+
"Use it when a critical position deserves depth: a novelty candidate, a sharp tactic, a hard endgame. Typical movetime: 30_000-60_000 for a careful check, 120_000-300_000 to find the truth.\n\n" +
|
|
750
|
+
"With `file_id`+`node_id`, the result is merged into that node's ceoEval (other engines' stored reads are kept) and stamped on its transpositions.",
|
|
769
751
|
inputSchema: {
|
|
770
752
|
type: "object",
|
|
771
|
-
properties: {
|
|
772
|
-
|
|
753
|
+
properties: {
|
|
754
|
+
engine: { type: "string", enum: ["stockfish", "lc0", "human"], description: "Default stockfish." },
|
|
755
|
+
file_id: { type: "string", description: "Prep file id. With `node_id`, the FEN comes from the tree and the result is stored on the node." },
|
|
773
756
|
node_id: { type: "string", description: "Node id inside `file_id`. Root is 'r'. When set, overrides `fen`/`moves`." },
|
|
774
757
|
fen: { type: "string", description: "Position as FEN. Only used if `node_id` is not set." },
|
|
775
758
|
moves: { type: "string", description: "Optional SAN moves on top of `fen`. Only used if `node_id` is not set." },
|
|
776
|
-
movetime_ms: {
|
|
777
|
-
|
|
778
|
-
|
|
779
|
-
|
|
780
|
-
description: "Think time in ms. Default 60_000 (1 min). Max 300_000 (5 min).",
|
|
781
|
-
},
|
|
782
|
-
multipv: {
|
|
783
|
-
type: "integer",
|
|
784
|
-
minimum: 1,
|
|
785
|
-
maximum: 10,
|
|
786
|
-
description: "Number of candidate lines (default 2). Stockfish gets weaker as multipv grows — each extra PV steals search bandwidth from the top choice — so keep this low unless you specifically want to see several candidates ranked deep.",
|
|
787
|
-
},
|
|
759
|
+
movetime_ms: { type: "integer", minimum: 5_000, maximum: 300_000, description: "Think time in ms. Default 60_000 (1 min). Max 300_000 (5 min)." },
|
|
760
|
+
multipv: { type: "integer", minimum: 1, maximum: 10, description: "Candidate lines. Default 2 for stockfish (extra lines cost it strength), 8 for lc0/human." },
|
|
761
|
+
contempt: { type: "integer", minimum: -100, maximum: 100, description: "lc0 only." },
|
|
762
|
+
rental: { type: "string", description: "Only when two running rentals provide this engine: the contract_id to use." },
|
|
788
763
|
},
|
|
789
764
|
},
|
|
790
765
|
},
|
|
791
766
|
{
|
|
792
767
|
name: "deep_analyse_status",
|
|
793
|
-
description: "Poll a deep_analyse job. Response: `{ status: 'running' | 'done' | 'cancelled' | 'error' | 'not_found', elapsed_ms, movetime_ms, result?, error? }`. `result`
|
|
768
|
+
description: "Poll a deep_analyse job. Response: `{ status: 'running' | 'done' | 'cancelled' | 'error' | 'not_found', elapsed_ms, movetime_ms, result?, error? }`. `result` when done: that engine's block from a cloud_analyse response (`{ depth, bestMove, lines: [{rank, depth, scoreCp?, mate?, wdl?, pv}] }`, PVs in SAN, wdl in percent); `stored_on` lists the nodes it was saved to.\n\n" +
|
|
794
769
|
"Poll cadence: every ~15-30s for long thinks; there's no penalty for polling more often but the engine progresses at its own pace.",
|
|
795
770
|
inputSchema: {
|
|
796
771
|
type: "object",
|
|
@@ -863,7 +838,7 @@ export const TOOLS = [
|
|
|
863
838
|
{
|
|
864
839
|
name: "quote_engine_eval",
|
|
865
840
|
description: "Return the stored engine eval for a node, or null if that node was never analysed. **Call this before writing prose or NAGs that quote engine numbers** — if it returns null, you have no measurement to cite. Do NOT infer an eval for the node from siblings or children; either analyse it (cloud_analyse with node_id) or omit the number from your prose.\n\n" +
|
|
866
|
-
"Response: `{ ceoEval: { sf
|
|
841
|
+
"Response: `{ ceoEval: { sf?: {cp | mate, depth}, lc0?: {w, d, l, depth}, human?: {w, d, l, depth} } | null }`. `cp` is White-POV centipawns (+20 = +0.20); `mate` White-POV moves to mate; `w/d/l` win/draw/loss percent from White's point of view. An engine missing from the object was never run on this position. Files analysed before 0.50 may hold an lc0 `cp` instead of w/d/l.",
|
|
867
842
|
inputSchema: {
|
|
868
843
|
type: "object",
|
|
869
844
|
properties: {
|