@chessceo/mcp 0.49.10 → 0.50.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -22,10 +22,11 @@ import { TOOLS } from "./tools.js";
22
22
  import { PROMPTS } from "./prompts.js";
23
23
  import { commentAntiPatterns, longLineWarning, noDescribeWarning, noStatsCheckWarning, positionalNagOnIntermediateWarning, positionsDescribed, positionsStatsChecked, } from "./warnings.js";
24
24
  import { authContext, authedRequest, deleteGame, fetchGame, get, makeFileId, restoreGame, } from "./http.js";
25
- import { analysisToStoredEval, capPvsInResponse, convertCloudSnapshotResponse, fetchCompactEval, mergeRentalAnalyses, } from "./analysis/response.js";
25
+ import { fetchCompactEval, } from "./analysis/response.js";
26
+ import { cloudAnalyse, ensureEngines } from "./analysis/cloud.js";
26
27
  import { autoEvaluate, autoEvaluateCancel, autoEvaluateStatus, } from "./analysis/auto.js";
27
28
  import { deepAnalyseCancel, deepAnalyseStart, deepAnalyseStatus, } from "./analysis/deep.js";
28
- import { analyseRouting, getNodeByPath, resolveFromNodeOrFen, storeEvalOnNode, } from "./analysis/file_handle.js";
29
+ import { getNodeByPath, resolveFromNodeOrFen, } from "./analysis/file_handle.js";
29
30
  import { findPositionInCourses, readCourseAtPosition, runSfEval } from "./courses.js";
30
31
  import { applyBatchMutations, applyMutation, argNodeId, } from "./prep/mutations.js";
31
32
  import { listNodes, listTranspositions, readPrepFile, } from "./prep/read.js";
@@ -41,6 +42,8 @@ const AUTHED_TOOLS = new Set([
41
42
  "list_cloud_engines",
42
43
  "stop_cloud_engine",
43
44
  "cloud_analyse",
45
+ "ensure_engines",
46
+ "list_cloud_machine_options",
44
47
  "list_collections",
45
48
  "create_collection",
46
49
  "rename_collection",
@@ -125,11 +128,22 @@ const ENGINE_GUIDE = {
125
128
  stockfish: "Objective calculation, the default cloud engine at about 100 million nodes per second. A few seconds gives a very strong view of the position. Use it on every position, including when steering by lc0 or human. Its score is the objective value with best defense: 0.00 means objectively a draw with best play, not that nothing is happening or that neither side has chances. Many positions that are very hard to play for a human score 0.00. Scores drift toward 0.00 as the search deepens, especially in drawn positions. Use it to check whether a tactic is real, whether a defense holds, and whether a move loses material or the game.",
126
129
  lc0: "Neural-net engine trained on self-play games, not human games. It is better than Stockfish at long-term ideas and at judging which side is practically on top. Never gives 0.00: it always says which side prefers the position, at least a little. Use it when Stockfish gives 0.00 to see who still has the practical edge, and to find where the long-term ideas are. Contempt changes how it values a draw and which side it plays for: contempt at -20 makes Black play for a win, which is useful for finding ideas, including in openings. It is much weaker tactically than Stockfish, so confirm any concrete line with Stockfish.",
127
130
  human: "Neural net trained only on human games, about 11 million rated games. Its moves and evaluations are practical, not objective: the moves it suggests are the ones a human would find and play. It punishes dubious moves less than Stockfish, because human games contain many imperfect moves that still work in practice, so a move with a sound idea behind it can score well here. Use it to predict how a person will respond to an idea, to find ideas (it may like a move that only fails to a refutation no human would find; Stockfish shows whether that refutation exists), and to see how a human would evaluate a position, especially positional or unclear ones. Like lc0 it is weaker in very unusual positions, and its evaluations there can be trusted much less. Stockfish remains the check on tactics and on whether a practical idea is objectively sound.",
128
- combo: "Runs Stockfish and Lc0 together — use when you want both the objective read and the practical/human-feel read on the same position.",
129
131
  };
130
132
  // Modern engines give equality very often, so small numbers are not "equal".
131
133
  // Rough guide, not exact thresholds. Shown alongside ENGINE_GUIDE above.
132
134
  const ENGINE_EVAL_SCALE = "Small numbers are not equality. ±0.00 to ±0.10 is equal in practice (+0.10 is '=' or '+=' at most). +0.20 to +0.40 is real pressure, not 'a bit better'. +0.50 and up is a clear advantage. Stockfish 0.00 with a plus from Lc0 is a practical edge, not equality. Check both engines before calling a position equal.";
135
+ // How to run analysis end to end. Authed response only, with ENGINE_GUIDE.
136
+ const ENGINE_WORKFLOW = [
137
+ "1. Plan: ensure_engines({engines, positions}) shows what is running, what to start, the price and an estimate. It starts nothing.",
138
+ "2. Ask: show the user the price of anything you would start and get a yes. Rentals bill per minute until stopped.",
139
+ "3. Start: start_cloud_engine({machine_type}) per missing engine. Each SKU runs one engine; the engine comes from the SKU. Wait until list_cloud_engines shows it running (static engines are instant, others ~1-5 min).",
140
+ "4. Analyse: cloud_analyse({fens | lines | file_id+node_ids, engines}) for up to 10 positions, each engine on its own rental, in parallel. auto_evaluate({id, engines}) for a whole file or subtree: fills only the engines each node is missing and saves as it goes. deep_analyse({engine, ...}) for one long think.",
141
+ "5. Read back: quote_engine_eval gives the stored sf (cp/mate), lc0 and human (win/draw/loss %) per node.",
142
+ "6. Stop: stop_cloud_engine when the work is done, unless the user wants it kept running.",
143
+ ].join("\n");
144
+ // Eval symbols are the writer's call, not the engine's. Recommendation
145
+ // only; nothing in the MCP sets a symbol automatically.
146
+ const EVAL_SYMBOL_GUIDE = "The eval symbol (=, +=, ±, +-) is your call, written for a human reader. The engines inform it; no tool sets it for you. Stockfish 0.00 can still be += when Lc0 or the human engine clearly prefer one side, because the symbol describes the practical picture, not only the objective one. Look at all engines you ran before choosing, and don't put a symbol on a position nobody analysed.";
133
147
  // v0.48: consolidated the five `read_*_guide` / `read_example_prep_files`
134
148
  // tools into ONE `read_docs`. LLM lists what it wants; we return them
135
149
  // in a single response. Also cleaner: enumerating available docs in one
@@ -390,77 +404,26 @@ async function callToolInner(name, args) {
390
404
  ...(resp && typeof resp === "object" ? resp : { options: [] }),
391
405
  engine_guide: ENGINE_GUIDE,
392
406
  eval_scale: ENGINE_EVAL_SCALE,
407
+ eval_symbols: EVAL_SYMBOL_GUIDE,
408
+ workflow: ENGINE_WORKFLOW,
393
409
  };
394
410
  }
411
+ case "ensure_engines":
412
+ return {
413
+ ...await ensureEngines(args),
414
+ engine_guide: ENGINE_GUIDE,
415
+ workflow: ENGINE_WORKFLOW,
416
+ };
395
417
  case "start_cloud_engine":
396
418
  return authedRequest("POST", "/api/agent/cloud-engines", {
397
419
  machineType: String(args.machine_type),
398
- ...(typeof args.engine_type === "string" ? { engineType: args.engine_type } : {}),
399
420
  });
400
421
  case "list_cloud_engines":
401
422
  return authedRequest("GET", "/api/agent/cloud-engines");
402
423
  case "stop_cloud_engine":
403
424
  return authedRequest("DELETE", `/api/agent/cloud-engines/${encodeURIComponent(String(args.contract_id))}`);
404
- case "cloud_analyse": {
405
- const resolved = await resolveFromNodeOrFen(args);
406
- const fen = resolved.fen;
407
- const body = { fen };
408
- if (typeof args.movetime_ms === "number")
409
- body.movetime_ms = args.movetime_ms;
410
- if (typeof args.stockfish_multipv === "number")
411
- body.stockfish_multipv = args.stockfish_multipv;
412
- if (typeof args.lc0_multipv === "number")
413
- body.lc0_multipv = args.lc0_multipv;
414
- if (typeof args.contempt === "number")
415
- body.contempt = args.contempt;
416
- Object.assign(body, analyseRouting(args));
417
- // Fan-out: several rentals analyse the same FEN in parallel, each
418
- // returning the engines it runs; the results are merged into one
419
- // object (first rental wins per engine field). One call, one
420
- // position, any mix of rentals.
421
- const fanOutIds = Array.isArray(args.contract_ids)
422
- ? args.contract_ids.filter((id) => typeof id === "string" && id.trim() !== "").map((id) => id.trim())
423
- : [];
424
- const raw = fanOutIds.length > 1
425
- ? mergeRentalAnalyses(await Promise.all(fanOutIds.map((id) => authedRequest("POST", "/api/agent/cloud-engines/analyse", { ...body, contract_id: id }))))
426
- : await authedRequest("POST", "/api/agent/cloud-engines/analyse", body);
427
- const converted = convertCloudSnapshotResponse(raw, fen);
428
- // PV cap: engine PVs beyond ~6 plies are speculative (the tail is
429
- // where the search's confidence collapses — SF at depth 24 has
430
- // seen the first few plies solidly and hedged everything after).
431
- // More importantly, LLMs paste long PVs into `add_line` as if
432
- // they were prepared repertoire. A 15-move PV pasted as a
433
- // variation is one line of engine output through positions
434
- // where both sides had real choices — not a repertoire. Cap the
435
- // affordance: return only what's load-bearing (3 full moves for
436
- // understanding the point), let the caller re-analyse the
437
- // resulting position if they want to see further. Override via
438
- // `pv_max_plies` for the rare case (deep tactics verification).
439
- const pvMaxPlies = typeof args.pv_max_plies === "number" && args.pv_max_plies > 0
440
- ? Math.min(args.pv_max_plies, 40)
441
- : 6;
442
- capPvsInResponse(converted, pvMaxPlies);
443
- // Node-addressed calls: persist the result on the node's ceoEval
444
- // so a later quote_engine_eval can cite this measurement. This is
445
- // the anti-hallucination hinge — prose that says "engines say X
446
- // on node Y" can only trace back to a call actually made against
447
- // node_id=Y, because the store only fires when file_id+node_id
448
- // was supplied and the eval survives via the [%ceo-eval] escape.
449
- if (resolved.file) {
450
- const ev = analysisToStoredEval(converted);
451
- if (ev) {
452
- const stamped = await storeEvalOnNode(resolved.file, ev);
453
- if (stamped.length > 1) {
454
- // Surface the propagation so the LLM sees exactly which
455
- // other nodes now carry this eval (and can skip them for
456
- // re-analysis).
457
- converted.also_stored_on = stamped.slice(1);
458
- }
459
- }
460
- }
461
- converted.eval_scale = ENGINE_EVAL_SCALE;
462
- return converted;
463
- }
425
+ case "cloud_analyse":
426
+ return cloudAnalyse(args, ENGINE_EVAL_SCALE);
464
427
  case "deep_analyse":
465
428
  return deepAnalyseStart(args);
466
429
  case "deep_analyse_status":
@@ -571,7 +534,7 @@ async function callToolInner(name, args) {
571
534
  case "apply_mutations":
572
535
  return applyBatchMutations(args);
573
536
  case "auto_evaluate":
574
- return autoEvaluate(args, applyBatchMutations);
537
+ return autoEvaluate(args);
575
538
  case "auto_evaluate_status":
576
539
  return autoEvaluateStatus(args);
577
540
  case "auto_evaluate_cancel":
@@ -1,7 +1,7 @@
1
1
  // PrepFile → PGN text. Deterministic — same tree always exports to the
2
2
  // same PGN modulo whitespace, so paths stay stable across
3
3
  // export→reparse cycles.
4
- import { codeFromColor } from "./types.js";
4
+ import { codeFromColor, STORED_EVAL_KEYS } from "./types.js";
5
5
  // Standard PGN "seven tag roster" order. We emit these first (in the
6
6
  // order they appear in tags), then any extra tags in insertion order.
7
7
  const STR_ORDER = ["Event", "Site", "Date", "Round", "White", "Black", "Result"];
@@ -142,22 +142,24 @@ function renderNodeAnnotation(node) {
142
142
  }
143
143
  return bits.join(" ");
144
144
  }
145
- // Serialise a stored eval as `[%ceo-eval sf=+0.20/38 lc0=+0.35/24 nag=$14]`.
146
- // Compact: no PVs (regenerable via cloud_analyse), decimal cp for
147
- // readability, mate as `MN`/`-MN`, depth as `/N`.
145
+ // Serialise a stored eval as `[%ceo-eval sf=+0.20/38 lc0=W38D45L17/24
146
+ // human=W30D50L20/20]`. Compact: no PVs (regenerable), decimal cp for
147
+ // readability, mate as `MN`/`-MN`, WDL as `W<w>D<d>L<l>`, depth as `/N`.
148
148
  function renderCeoEval(ev) {
149
149
  const bits = [];
150
- if (ev.sf)
151
- bits.push(`sf=${formatEngineEval(ev.sf)}`);
152
- if (ev.lc0)
153
- bits.push(`lc0=${formatEngineEval(ev.lc0)}`);
154
- if (ev.nag)
155
- bits.push(`nag=${ev.nag}`);
150
+ for (const k of STORED_EVAL_KEYS) {
151
+ const e = ev[k];
152
+ if (e)
153
+ bits.push(`${k}=${formatEngineEval(e)}`);
154
+ }
156
155
  return bits.length > 0 ? `[%ceo-eval ${bits.join(" ")}]` : "";
157
156
  }
158
157
  function formatEngineEval(e) {
159
158
  let body;
160
- if (typeof e.mate === "number") {
159
+ if (typeof e.w === "number" && typeof e.d === "number" && typeof e.l === "number") {
160
+ body = `W${e.w}D${e.d}L${e.l}`;
161
+ }
162
+ else if (typeof e.mate === "number") {
161
163
  body = e.mate >= 0 ? `+M${e.mate}` : `M${e.mate}`; // "+M5" / "M-5"
162
164
  }
163
165
  else if (typeof e.cp === "number") {
@@ -12,6 +12,7 @@ import { Chess } from "chessops/chess";
12
12
  import { parseSan } from "chessops/san";
13
13
  import { makeFen } from "chessops/fen";
14
14
  import { cloneOnPath, deriveNodeId, getNode, PathError } from "./paths.js";
15
+ import { STORED_EVAL_KEYS } from "./types.js";
15
16
  export class MutationError extends Error {
16
17
  }
17
18
  // Add a SAN move as a new child under `parentPath`. **Idempotent on
@@ -162,14 +163,26 @@ export function promoteVariation(file, path) {
162
163
  parent.children = [promoted, ...rest];
163
164
  return { file: { tags: file.tags, root: newRoot }, id: promoted.id };
164
165
  }
165
- // Replace the stored ceoEval on the node at `path`. Passing null clears.
166
- // This is what auto_evaluate / cloud_analyse(node_id) calls per-node.
166
+ // Merge engine evals into the stored ceoEval on the node at `path`: only
167
+ // the engines present in `ev` are replaced, the others are kept, so a
168
+ // Stockfish deep think never wipes a stored human or Lc0 read. Passing null
169
+ // clears every engine.
167
170
  export function setCeoEval(file, path, ev) {
168
171
  const { root: newRoot, target } = cloneOnPath(file.root, path);
169
- if (ev === null || (!ev.sf && !ev.lc0 && !ev.nag))
172
+ if (ev === null) {
170
173
  delete target.ceoEval;
171
- else
172
- target.ceoEval = ev;
174
+ }
175
+ else {
176
+ const merged = { ...(target.ceoEval ?? {}) };
177
+ for (const k of STORED_EVAL_KEYS) {
178
+ if (ev[k])
179
+ merged[k] = ev[k];
180
+ }
181
+ if (merged.sf || merged.lc0 || merged.human)
182
+ target.ceoEval = merged;
183
+ else
184
+ delete target.ceoEval;
185
+ }
173
186
  return { file: { tags: file.tags, root: newRoot }, id: target.id };
174
187
  }
175
188
  // Set the same ceoEval on every path in `paths`. One clone-and-return
@@ -24,10 +24,17 @@ const CEO_EVAL_RE = /\[%ceo-eval\s+([^\]]+)\]/g;
24
24
  // `{[%eval] [%wdl]}` comment on an imported Lichess PGN got stripped
25
25
  // on a set_tag save cycle).
26
26
  const KNOWN_CMD_RE = /\[%(?:cal|csl|ceo-eval)\s+[^\]]+\]/g;
27
- // Parse a `sf=+0.20/38` or `lc0=M-5/22` fragment into its numeric parts.
28
- // Returns null if the value doesn't parse; the parser tolerates missing
29
- // depth and mate notation.
27
+ // Parse a `sf=+0.20/38`, `sf=M-5/22` or `lc0=W38D45L17/24` fragment into
28
+ // its numeric parts. Returns null if the value doesn't parse; depth is
29
+ // optional.
30
30
  function parseEngineEvalFragment(raw) {
31
+ const wdl = raw.match(/^W(\d+)D(\d+)L(\d+)(?:\/(\d+))?$/i);
32
+ if (wdl) {
33
+ const out = { w: parseInt(wdl[1], 10), d: parseInt(wdl[2], 10), l: parseInt(wdl[3], 10) };
34
+ if (wdl[4] !== undefined)
35
+ out.depth = parseInt(wdl[4], 10);
36
+ return out;
37
+ }
31
38
  const m = raw.match(/^([+-]?)(?:M(-?\d+)|(\d+(?:\.\d+)?))(?:\/(\d+))?$/i);
32
39
  if (!m)
33
40
  return null;
@@ -75,22 +82,14 @@ export function parseCommentAnnotations(comment) {
75
82
  continue;
76
83
  const k = entry.slice(0, eq).toLowerCase();
77
84
  const v = entry.slice(eq + 1);
78
- if (k === "sf") {
85
+ if (k === "sf" || k === "lc0" || k === "human") {
79
86
  const parsed = parseEngineEvalFragment(v);
80
87
  if (parsed)
81
- ev.sf = parsed;
82
- }
83
- else if (k === "lc0") {
84
- const parsed = parseEngineEvalFragment(v);
85
- if (parsed)
86
- ev.lc0 = parsed;
87
- }
88
- else if (k === "nag") {
89
- if (/^\$\d+$/.test(v))
90
- ev.nag = v;
88
+ ev[k] = parsed;
91
89
  }
90
+ // Anything else (incl. the pre-0.50 `nag=`) is dropped.
92
91
  }
93
- if (ev.sf || ev.lc0 || ev.nag)
92
+ if (ev.sf || ev.lc0 || ev.human)
94
93
  ceoEval = ev;
95
94
  }
96
95
  const text = comment.replace(KNOWN_CMD_RE, "").replace(/\s+/g, " ").trim();
package/dist/pgn/types.js CHANGED
@@ -19,6 +19,8 @@
19
19
  // width, buildIdIndex keeps the first occurrence and logs to stderr
20
20
  // rather than throwing — the file stays readable, only mutations
21
21
  // against the collided id are ambiguous.
22
+ // Keys of StoredEval in canonical order.
23
+ export const STORED_EVAL_KEYS = ["sf", "lc0", "human"];
22
24
  // Named colours as they appear in the parsed tree. The wire format uses
23
25
  // single-letter codes (G, R, Y, C, B, O); the tree uses these longer
24
26
  // names to match how the frontend represents them.