@chessceo/mcp 0.49.10 → 0.50.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/analysis/auto.js +173 -182
- package/dist/analysis/cloud.js +182 -0
- package/dist/analysis/deep.js +39 -66
- package/dist/analysis/file_handle.js +43 -36
- package/dist/analysis/response.js +180 -142
- package/dist/index.js +28 -65
- package/dist/pgn/exporter.js +13 -11
- package/dist/pgn/mutations.js +18 -5
- package/dist/pgn/parser.js +14 -15
- package/dist/pgn/types.js +2 -0
- package/dist/tools.js +75 -101
- package/docs/engine-usage.md +38 -40
- package/docs/pgn-authoring.md +3 -3
- package/docs/summary-authoring.md +1 -1
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -22,10 +22,11 @@ import { TOOLS } from "./tools.js";
|
|
|
22
22
|
import { PROMPTS } from "./prompts.js";
|
|
23
23
|
import { commentAntiPatterns, longLineWarning, noDescribeWarning, noStatsCheckWarning, positionalNagOnIntermediateWarning, positionsDescribed, positionsStatsChecked, } from "./warnings.js";
|
|
24
24
|
import { authContext, authedRequest, deleteGame, fetchGame, get, makeFileId, restoreGame, } from "./http.js";
|
|
25
|
-
import {
|
|
25
|
+
import { fetchCompactEval, } from "./analysis/response.js";
|
|
26
|
+
import { cloudAnalyse, ensureEngines } from "./analysis/cloud.js";
|
|
26
27
|
import { autoEvaluate, autoEvaluateCancel, autoEvaluateStatus, } from "./analysis/auto.js";
|
|
27
28
|
import { deepAnalyseCancel, deepAnalyseStart, deepAnalyseStatus, } from "./analysis/deep.js";
|
|
28
|
-
import {
|
|
29
|
+
import { getNodeByPath, resolveFromNodeOrFen, } from "./analysis/file_handle.js";
|
|
29
30
|
import { findPositionInCourses, readCourseAtPosition, runSfEval } from "./courses.js";
|
|
30
31
|
import { applyBatchMutations, applyMutation, argNodeId, } from "./prep/mutations.js";
|
|
31
32
|
import { listNodes, listTranspositions, readPrepFile, } from "./prep/read.js";
|
|
@@ -41,6 +42,8 @@ const AUTHED_TOOLS = new Set([
|
|
|
41
42
|
"list_cloud_engines",
|
|
42
43
|
"stop_cloud_engine",
|
|
43
44
|
"cloud_analyse",
|
|
45
|
+
"ensure_engines",
|
|
46
|
+
"list_cloud_machine_options",
|
|
44
47
|
"list_collections",
|
|
45
48
|
"create_collection",
|
|
46
49
|
"rename_collection",
|
|
@@ -125,11 +128,22 @@ const ENGINE_GUIDE = {
|
|
|
125
128
|
stockfish: "Objective calculation, the default cloud engine at about 100 million nodes per second. A few seconds gives a very strong view of the position. Use it on every position, including when steering by lc0 or human. Its score is the objective value with best defense: 0.00 means objectively a draw with best play, not that nothing is happening or that neither side has chances. Many positions that are very hard to play for a human score 0.00. Scores drift toward 0.00 as the search deepens, especially in drawn positions. Use it to check whether a tactic is real, whether a defense holds, and whether a move loses material or the game.",
|
|
126
129
|
lc0: "Neural-net engine trained on self-play games, not human games. It is better than Stockfish at long-term ideas and at judging which side is practically on top. Never gives 0.00: it always says which side prefers the position, at least a little. Use it when Stockfish gives 0.00 to see who still has the practical edge, and to find where the long-term ideas are. Contempt changes how it values a draw and which side it plays for: contempt at -20 makes Black play for a win, which is useful for finding ideas, including in openings. It is much weaker tactically than Stockfish, so confirm any concrete line with Stockfish.",
|
|
127
130
|
human: "Neural net trained only on human games, about 11 million rated games. Its moves and evaluations are practical, not objective: the moves it suggests are the ones a human would find and play. It punishes dubious moves less than Stockfish, because human games contain many imperfect moves that still work in practice, so a move with a sound idea behind it can score well here. Use it to predict how a person will respond to an idea, to find ideas (it may like a move that only fails to a refutation no human would find; Stockfish shows whether that refutation exists), and to see how a human would evaluate a position, especially positional or unclear ones. Like lc0 it is weaker in very unusual positions, and its evaluations there can be trusted much less. Stockfish remains the check on tactics and on whether a practical idea is objectively sound.",
|
|
128
|
-
combo: "Runs Stockfish and Lc0 together — use when you want both the objective read and the practical/human-feel read on the same position.",
|
|
129
131
|
};
|
|
130
132
|
// Modern engines give equality very often, so small numbers are not "equal".
|
|
131
133
|
// Rough guide, not exact thresholds. Shown alongside ENGINE_GUIDE above.
|
|
132
134
|
const ENGINE_EVAL_SCALE = "Small numbers are not equality. ±0.00 to ±0.10 is equal in practice (+0.10 is '=' or '+=' at most). +0.20 to +0.40 is real pressure, not 'a bit better'. +0.50 and up is a clear advantage. Stockfish 0.00 with a plus from Lc0 is a practical edge, not equality. Check both engines before calling a position equal.";
|
|
135
|
+
// How to run analysis end to end. Authed response only, with ENGINE_GUIDE.
|
|
136
|
+
const ENGINE_WORKFLOW = [
|
|
137
|
+
"1. Plan: ensure_engines({engines, positions}) shows what is running, what to start, the price and an estimate. It starts nothing.",
|
|
138
|
+
"2. Ask: show the user the price of anything you would start and get a yes. Rentals bill per minute until stopped.",
|
|
139
|
+
"3. Start: start_cloud_engine({machine_type}) per missing engine. Each SKU runs one engine; the engine comes from the SKU. Wait until list_cloud_engines shows it running (static engines are instant, others ~1-5 min).",
|
|
140
|
+
"4. Analyse: cloud_analyse({fens | lines | file_id+node_ids, engines}) for up to 10 positions, each engine on its own rental, in parallel. auto_evaluate({id, engines}) for a whole file or subtree: fills only the engines each node is missing and saves as it goes. deep_analyse({engine, ...}) for one long think.",
|
|
141
|
+
"5. Read back: quote_engine_eval gives the stored sf (cp/mate), lc0 and human (win/draw/loss %) per node.",
|
|
142
|
+
"6. Stop: stop_cloud_engine when the work is done, unless the user wants it kept running.",
|
|
143
|
+
].join("\n");
|
|
144
|
+
// Eval symbols are the writer's call, not the engine's. Recommendation
|
|
145
|
+
// only; nothing in the MCP sets a symbol automatically.
|
|
146
|
+
const EVAL_SYMBOL_GUIDE = "The eval symbol (=, +=, ±, +-) is your call, written for a human reader. The engines inform it; no tool sets it for you. Stockfish 0.00 can still be += when Lc0 or the human engine clearly prefer one side, because the symbol describes the practical picture, not only the objective one. Look at all engines you ran before choosing, and don't put a symbol on a position nobody analysed.";
|
|
133
147
|
// v0.48: consolidated the five `read_*_guide` / `read_example_prep_files`
|
|
134
148
|
// tools into ONE `read_docs`. LLM lists what it wants; we return them
|
|
135
149
|
// in a single response. Also cleaner: enumerating available docs in one
|
|
@@ -390,77 +404,26 @@ async function callToolInner(name, args) {
|
|
|
390
404
|
...(resp && typeof resp === "object" ? resp : { options: [] }),
|
|
391
405
|
engine_guide: ENGINE_GUIDE,
|
|
392
406
|
eval_scale: ENGINE_EVAL_SCALE,
|
|
407
|
+
eval_symbols: EVAL_SYMBOL_GUIDE,
|
|
408
|
+
workflow: ENGINE_WORKFLOW,
|
|
393
409
|
};
|
|
394
410
|
}
|
|
411
|
+
case "ensure_engines":
|
|
412
|
+
return {
|
|
413
|
+
...await ensureEngines(args),
|
|
414
|
+
engine_guide: ENGINE_GUIDE,
|
|
415
|
+
workflow: ENGINE_WORKFLOW,
|
|
416
|
+
};
|
|
395
417
|
case "start_cloud_engine":
|
|
396
418
|
return authedRequest("POST", "/api/agent/cloud-engines", {
|
|
397
419
|
machineType: String(args.machine_type),
|
|
398
|
-
...(typeof args.engine_type === "string" ? { engineType: args.engine_type } : {}),
|
|
399
420
|
});
|
|
400
421
|
case "list_cloud_engines":
|
|
401
422
|
return authedRequest("GET", "/api/agent/cloud-engines");
|
|
402
423
|
case "stop_cloud_engine":
|
|
403
424
|
return authedRequest("DELETE", `/api/agent/cloud-engines/${encodeURIComponent(String(args.contract_id))}`);
|
|
404
|
-
case "cloud_analyse":
|
|
405
|
-
|
|
406
|
-
const fen = resolved.fen;
|
|
407
|
-
const body = { fen };
|
|
408
|
-
if (typeof args.movetime_ms === "number")
|
|
409
|
-
body.movetime_ms = args.movetime_ms;
|
|
410
|
-
if (typeof args.stockfish_multipv === "number")
|
|
411
|
-
body.stockfish_multipv = args.stockfish_multipv;
|
|
412
|
-
if (typeof args.lc0_multipv === "number")
|
|
413
|
-
body.lc0_multipv = args.lc0_multipv;
|
|
414
|
-
if (typeof args.contempt === "number")
|
|
415
|
-
body.contempt = args.contempt;
|
|
416
|
-
Object.assign(body, analyseRouting(args));
|
|
417
|
-
// Fan-out: several rentals analyse the same FEN in parallel, each
|
|
418
|
-
// returning the engines it runs; the results are merged into one
|
|
419
|
-
// object (first rental wins per engine field). One call, one
|
|
420
|
-
// position, any mix of rentals.
|
|
421
|
-
const fanOutIds = Array.isArray(args.contract_ids)
|
|
422
|
-
? args.contract_ids.filter((id) => typeof id === "string" && id.trim() !== "").map((id) => id.trim())
|
|
423
|
-
: [];
|
|
424
|
-
const raw = fanOutIds.length > 1
|
|
425
|
-
? mergeRentalAnalyses(await Promise.all(fanOutIds.map((id) => authedRequest("POST", "/api/agent/cloud-engines/analyse", { ...body, contract_id: id }))))
|
|
426
|
-
: await authedRequest("POST", "/api/agent/cloud-engines/analyse", body);
|
|
427
|
-
const converted = convertCloudSnapshotResponse(raw, fen);
|
|
428
|
-
// PV cap: engine PVs beyond ~6 plies are speculative (the tail is
|
|
429
|
-
// where the search's confidence collapses — SF at depth 24 has
|
|
430
|
-
// seen the first few plies solidly and hedged everything after).
|
|
431
|
-
// More importantly, LLMs paste long PVs into `add_line` as if
|
|
432
|
-
// they were prepared repertoire. A 15-move PV pasted as a
|
|
433
|
-
// variation is one line of engine output through positions
|
|
434
|
-
// where both sides had real choices — not a repertoire. Cap the
|
|
435
|
-
// affordance: return only what's load-bearing (3 full moves for
|
|
436
|
-
// understanding the point), let the caller re-analyse the
|
|
437
|
-
// resulting position if they want to see further. Override via
|
|
438
|
-
// `pv_max_plies` for the rare case (deep tactics verification).
|
|
439
|
-
const pvMaxPlies = typeof args.pv_max_plies === "number" && args.pv_max_plies > 0
|
|
440
|
-
? Math.min(args.pv_max_plies, 40)
|
|
441
|
-
: 6;
|
|
442
|
-
capPvsInResponse(converted, pvMaxPlies);
|
|
443
|
-
// Node-addressed calls: persist the result on the node's ceoEval
|
|
444
|
-
// so a later quote_engine_eval can cite this measurement. This is
|
|
445
|
-
// the anti-hallucination hinge — prose that says "engines say X
|
|
446
|
-
// on node Y" can only trace back to a call actually made against
|
|
447
|
-
// node_id=Y, because the store only fires when file_id+node_id
|
|
448
|
-
// was supplied and the eval survives via the [%ceo-eval] escape.
|
|
449
|
-
if (resolved.file) {
|
|
450
|
-
const ev = analysisToStoredEval(converted);
|
|
451
|
-
if (ev) {
|
|
452
|
-
const stamped = await storeEvalOnNode(resolved.file, ev);
|
|
453
|
-
if (stamped.length > 1) {
|
|
454
|
-
// Surface the propagation so the LLM sees exactly which
|
|
455
|
-
// other nodes now carry this eval (and can skip them for
|
|
456
|
-
// re-analysis).
|
|
457
|
-
converted.also_stored_on = stamped.slice(1);
|
|
458
|
-
}
|
|
459
|
-
}
|
|
460
|
-
}
|
|
461
|
-
converted.eval_scale = ENGINE_EVAL_SCALE;
|
|
462
|
-
return converted;
|
|
463
|
-
}
|
|
425
|
+
case "cloud_analyse":
|
|
426
|
+
return cloudAnalyse(args, ENGINE_EVAL_SCALE);
|
|
464
427
|
case "deep_analyse":
|
|
465
428
|
return deepAnalyseStart(args);
|
|
466
429
|
case "deep_analyse_status":
|
|
@@ -571,7 +534,7 @@ async function callToolInner(name, args) {
|
|
|
571
534
|
case "apply_mutations":
|
|
572
535
|
return applyBatchMutations(args);
|
|
573
536
|
case "auto_evaluate":
|
|
574
|
-
return autoEvaluate(args
|
|
537
|
+
return autoEvaluate(args);
|
|
575
538
|
case "auto_evaluate_status":
|
|
576
539
|
return autoEvaluateStatus(args);
|
|
577
540
|
case "auto_evaluate_cancel":
|
package/dist/pgn/exporter.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
// PrepFile → PGN text. Deterministic — same tree always exports to the
|
|
2
2
|
// same PGN modulo whitespace, so paths stay stable across
|
|
3
3
|
// export→reparse cycles.
|
|
4
|
-
import { codeFromColor } from "./types.js";
|
|
4
|
+
import { codeFromColor, STORED_EVAL_KEYS } from "./types.js";
|
|
5
5
|
// Standard PGN "seven tag roster" order. We emit these first (in the
|
|
6
6
|
// order they appear in tags), then any extra tags in insertion order.
|
|
7
7
|
const STR_ORDER = ["Event", "Site", "Date", "Round", "White", "Black", "Result"];
|
|
@@ -142,22 +142,24 @@ function renderNodeAnnotation(node) {
|
|
|
142
142
|
}
|
|
143
143
|
return bits.join(" ");
|
|
144
144
|
}
|
|
145
|
-
// Serialise a stored eval as `[%ceo-eval sf=+0.20/38 lc0
|
|
146
|
-
// Compact: no PVs (regenerable
|
|
147
|
-
// readability, mate as `MN`/`-MN`, depth as `/N`.
|
|
145
|
+
// Serialise a stored eval as `[%ceo-eval sf=+0.20/38 lc0=W38D45L17/24
|
|
146
|
+
// human=W30D50L20/20]`. Compact: no PVs (regenerable), decimal cp for
|
|
147
|
+
// readability, mate as `MN`/`-MN`, WDL as `W<w>D<d>L<l>`, depth as `/N`.
|
|
148
148
|
function renderCeoEval(ev) {
|
|
149
149
|
const bits = [];
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
bits.push(`nag=${ev.nag}`);
|
|
150
|
+
for (const k of STORED_EVAL_KEYS) {
|
|
151
|
+
const e = ev[k];
|
|
152
|
+
if (e)
|
|
153
|
+
bits.push(`${k}=${formatEngineEval(e)}`);
|
|
154
|
+
}
|
|
156
155
|
return bits.length > 0 ? `[%ceo-eval ${bits.join(" ")}]` : "";
|
|
157
156
|
}
|
|
158
157
|
function formatEngineEval(e) {
|
|
159
158
|
let body;
|
|
160
|
-
if (typeof e.
|
|
159
|
+
if (typeof e.w === "number" && typeof e.d === "number" && typeof e.l === "number") {
|
|
160
|
+
body = `W${e.w}D${e.d}L${e.l}`;
|
|
161
|
+
}
|
|
162
|
+
else if (typeof e.mate === "number") {
|
|
161
163
|
body = e.mate >= 0 ? `+M${e.mate}` : `M${e.mate}`; // "+M5" / "M-5"
|
|
162
164
|
}
|
|
163
165
|
else if (typeof e.cp === "number") {
|
package/dist/pgn/mutations.js
CHANGED
|
@@ -12,6 +12,7 @@ import { Chess } from "chessops/chess";
|
|
|
12
12
|
import { parseSan } from "chessops/san";
|
|
13
13
|
import { makeFen } from "chessops/fen";
|
|
14
14
|
import { cloneOnPath, deriveNodeId, getNode, PathError } from "./paths.js";
|
|
15
|
+
import { STORED_EVAL_KEYS } from "./types.js";
|
|
15
16
|
export class MutationError extends Error {
|
|
16
17
|
}
|
|
17
18
|
// Add a SAN move as a new child under `parentPath`. **Idempotent on
|
|
@@ -162,14 +163,26 @@ export function promoteVariation(file, path) {
|
|
|
162
163
|
parent.children = [promoted, ...rest];
|
|
163
164
|
return { file: { tags: file.tags, root: newRoot }, id: promoted.id };
|
|
164
165
|
}
|
|
165
|
-
//
|
|
166
|
-
//
|
|
166
|
+
// Merge engine evals into the stored ceoEval on the node at `path`: only
|
|
167
|
+
// the engines present in `ev` are replaced, the others are kept, so a
|
|
168
|
+
// Stockfish deep think never wipes a stored human or Lc0 read. Passing null
|
|
169
|
+
// clears every engine.
|
|
167
170
|
export function setCeoEval(file, path, ev) {
|
|
168
171
|
const { root: newRoot, target } = cloneOnPath(file.root, path);
|
|
169
|
-
if (ev === null
|
|
172
|
+
if (ev === null) {
|
|
170
173
|
delete target.ceoEval;
|
|
171
|
-
|
|
172
|
-
|
|
174
|
+
}
|
|
175
|
+
else {
|
|
176
|
+
const merged = { ...(target.ceoEval ?? {}) };
|
|
177
|
+
for (const k of STORED_EVAL_KEYS) {
|
|
178
|
+
if (ev[k])
|
|
179
|
+
merged[k] = ev[k];
|
|
180
|
+
}
|
|
181
|
+
if (merged.sf || merged.lc0 || merged.human)
|
|
182
|
+
target.ceoEval = merged;
|
|
183
|
+
else
|
|
184
|
+
delete target.ceoEval;
|
|
185
|
+
}
|
|
173
186
|
return { file: { tags: file.tags, root: newRoot }, id: target.id };
|
|
174
187
|
}
|
|
175
188
|
// Set the same ceoEval on every path in `paths`. One clone-and-return
|
package/dist/pgn/parser.js
CHANGED
|
@@ -24,10 +24,17 @@ const CEO_EVAL_RE = /\[%ceo-eval\s+([^\]]+)\]/g;
|
|
|
24
24
|
// `{[%eval] [%wdl]}` comment on an imported Lichess PGN got stripped
|
|
25
25
|
// on a set_tag save cycle).
|
|
26
26
|
const KNOWN_CMD_RE = /\[%(?:cal|csl|ceo-eval)\s+[^\]]+\]/g;
|
|
27
|
-
// Parse a `sf=+0.20/38
|
|
28
|
-
// Returns null if the value doesn't parse;
|
|
29
|
-
//
|
|
27
|
+
// Parse a `sf=+0.20/38`, `sf=M-5/22` or `lc0=W38D45L17/24` fragment into
|
|
28
|
+
// its numeric parts. Returns null if the value doesn't parse; depth is
|
|
29
|
+
// optional.
|
|
30
30
|
function parseEngineEvalFragment(raw) {
|
|
31
|
+
const wdl = raw.match(/^W(\d+)D(\d+)L(\d+)(?:\/(\d+))?$/i);
|
|
32
|
+
if (wdl) {
|
|
33
|
+
const out = { w: parseInt(wdl[1], 10), d: parseInt(wdl[2], 10), l: parseInt(wdl[3], 10) };
|
|
34
|
+
if (wdl[4] !== undefined)
|
|
35
|
+
out.depth = parseInt(wdl[4], 10);
|
|
36
|
+
return out;
|
|
37
|
+
}
|
|
31
38
|
const m = raw.match(/^([+-]?)(?:M(-?\d+)|(\d+(?:\.\d+)?))(?:\/(\d+))?$/i);
|
|
32
39
|
if (!m)
|
|
33
40
|
return null;
|
|
@@ -75,22 +82,14 @@ export function parseCommentAnnotations(comment) {
|
|
|
75
82
|
continue;
|
|
76
83
|
const k = entry.slice(0, eq).toLowerCase();
|
|
77
84
|
const v = entry.slice(eq + 1);
|
|
78
|
-
if (k === "sf") {
|
|
85
|
+
if (k === "sf" || k === "lc0" || k === "human") {
|
|
79
86
|
const parsed = parseEngineEvalFragment(v);
|
|
80
87
|
if (parsed)
|
|
81
|
-
ev
|
|
82
|
-
}
|
|
83
|
-
else if (k === "lc0") {
|
|
84
|
-
const parsed = parseEngineEvalFragment(v);
|
|
85
|
-
if (parsed)
|
|
86
|
-
ev.lc0 = parsed;
|
|
87
|
-
}
|
|
88
|
-
else if (k === "nag") {
|
|
89
|
-
if (/^\$\d+$/.test(v))
|
|
90
|
-
ev.nag = v;
|
|
88
|
+
ev[k] = parsed;
|
|
91
89
|
}
|
|
90
|
+
// Anything else (incl. the pre-0.50 `nag=`) is dropped.
|
|
92
91
|
}
|
|
93
|
-
if (ev.sf || ev.lc0 || ev.
|
|
92
|
+
if (ev.sf || ev.lc0 || ev.human)
|
|
94
93
|
ceoEval = ev;
|
|
95
94
|
}
|
|
96
95
|
const text = comment.replace(KNOWN_CMD_RE, "").replace(/\s+/g, " ").trim();
|
package/dist/pgn/types.js
CHANGED
|
@@ -19,6 +19,8 @@
|
|
|
19
19
|
// width, buildIdIndex keeps the first occurrence and logs to stderr
|
|
20
20
|
// rather than throwing — the file stays readable, only mutations
|
|
21
21
|
// against the collided id are ambiguous.
|
|
22
|
+
// Keys of StoredEval in canonical order.
|
|
23
|
+
export const STORED_EVAL_KEYS = ["sf", "lc0", "human"];
|
|
22
24
|
// Named colours as they appear in the parsed tree. The wire format uses
|
|
23
25
|
// single-letter codes (G, R, Y, C, B, O); the tree uses these longer
|
|
24
26
|
// names to match how the frontend represents them.
|