@chessceo/mcp 0.44.0 → 0.48.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,242 @@
1
+ // Prep-file reads and list-queries. Everything the LLM calls to inspect
2
+ // a file WITHOUT mutating it:
3
+ // - readPrepFile (with four view modes: compact / full / spine / pgn)
4
+ // - listNodes (cheap filter-based node queries)
5
+ // - listTranspositions (positions occurring 2+ times in the file)
6
+ //
7
+ // Extracted from index.ts in v0.44 as part of the file split. Was
8
+ // originally split off the read-case handler because both readPrepFile
9
+ // and listNodes need the same load-parse pipeline and share the
10
+ // compact / spine / pgn view logic.
11
+ import { fetchGame } from "../http.js";
12
+ import { parsePGN } from "../pgn/parser.js";
13
+ import { exportPGN } from "../pgn/exporter.js";
14
+ import { buildFenIndex, buildIdIndex, positionKey, resolveNodeId, ROOT_ID, } from "../pgn/paths.js";
15
+ import { getNodeByPath } from "../analysis/file_handle.js";
16
+ export async function loadPrepFile(id) {
17
+ const g = await fetchGame(id);
18
+ // Echo the composite id back so read_prep_file responses match the
19
+ // exact id the LLM passed in. The backend returns the raw game_id;
20
+ // recompose so the LLM never sees the split form.
21
+ return { file: parsePGN(g.pgnContent), version: g.version, fileIdEcho: id, pgn: g.pgnContent };
22
+ }
23
+ // Recursively project a PrepNode into the requested view. `depthLeft`
24
+ // null → unlimited; 0 → just the node without children.
25
+ //
26
+ // `fenIndex` (optional) enables the `transposes_to` field — for each
27
+ // node whose position also appears elsewhere in the SAME file, we
28
+ // annotate it with the OTHER occurrences' ids. Pass null (the default)
29
+ // to skip the annotation entirely; passing the map costs one lookup
30
+ // per node projected.
31
+ export function projectNode(node, view, depthLeft, fenIndex = null) {
32
+ const base = {
33
+ id: node.id,
34
+ san: node.san,
35
+ ply: node.ply,
36
+ };
37
+ if (node.nags && node.nags.length > 0)
38
+ base.nags = node.nags;
39
+ if (node.comment)
40
+ base.comment = node.comment;
41
+ if (node.ceoEval)
42
+ base.ceoEval = node.ceoEval;
43
+ if (view === "full") {
44
+ base.fen = node.fen;
45
+ if (node.annotations)
46
+ base.annotations = node.annotations;
47
+ }
48
+ if (fenIndex && node.id !== ROOT_ID) {
49
+ const group = fenIndex.get(positionKey(node.fen));
50
+ if (group && group.length > 1) {
51
+ const others = group.filter(n => n.id !== node.id).map(n => n.id);
52
+ if (others.length > 0)
53
+ base.transposes_to = others;
54
+ }
55
+ }
56
+ // Children handling depends on view + depth budget.
57
+ const showChildren = depthLeft === null || depthLeft > 0;
58
+ const childDepth = depthLeft === null ? null : depthLeft - 1;
59
+ if (showChildren && node.children.length > 0) {
60
+ if (view === "spine") {
61
+ // Only follow children[0] — collapses the tree to the mainline.
62
+ base.children = [projectNode(node.children[0], view, childDepth, fenIndex)];
63
+ }
64
+ else {
65
+ base.children = node.children.map(c => projectNode(c, view, childDepth, fenIndex));
66
+ }
67
+ }
68
+ else {
69
+ base.children = [];
70
+ }
71
+ return base;
72
+ }
73
+ export async function readPrepFile(args) {
74
+ const id = String(args.id);
75
+ const view = (typeof args.view === "string" && ["compact", "full", "spine", "pgn"].includes(args.view))
76
+ ? args.view
77
+ : "compact";
78
+ const startNodeId = typeof args.node_id === "string" && args.node_id.length > 0 ? args.node_id : ROOT_ID;
79
+ const maxDepth = typeof args.max_depth === "number" && args.max_depth >= 0 ? args.max_depth : null;
80
+ const { file, version, fileIdEcho, pgn } = await loadPrepFile(id);
81
+ const idIndex = buildIdIndex(file.root);
82
+ const path = resolveNodeId(idIndex, startNodeId);
83
+ const anchor = getNodeByPath(file.root, path);
84
+ const fenIndex = buildFenIndex(file.root);
85
+ // How many DISTINCT positions in the file appear more than once,
86
+ // and how many nodes are involved. Shown in the header so the LLM
87
+ // sees at a glance whether transpositions matter here before diving
88
+ // into the tree.
89
+ let transGroups = 0;
90
+ let transNodes = 0;
91
+ for (const arr of fenIndex.values()) {
92
+ if (arr.length > 1) {
93
+ transGroups++;
94
+ transNodes += arr.length;
95
+ }
96
+ }
97
+ const header = {
98
+ id: fileIdEcho ?? id,
99
+ version,
100
+ tags: file.tags,
101
+ view,
102
+ node_id: startNodeId,
103
+ max_depth: maxDepth,
104
+ transposition_groups: transGroups,
105
+ transposition_nodes: transNodes,
106
+ };
107
+ if (view === "pgn") {
108
+ // For the root, just return the file's actual PGN as-is. For a
109
+ // subtree, build a mini-Game from the anchor and export it. Keeps
110
+ // formatting identical to what the app renders.
111
+ if (startNodeId === ROOT_ID && (maxDepth === null || maxDepth >= 999)) {
112
+ return { ...header, pgn };
113
+ }
114
+ // Truncate to a subtree with max_depth. Simple: walk the anchor's
115
+ // subtree, produce a synthetic PGN starting from the anchor's FEN.
116
+ const subtreePgn = exportSubtreePgn(file, anchor, maxDepth);
117
+ return { ...header, pgn: subtreePgn };
118
+ }
119
+ return { ...header, tree: projectNode(anchor, view, maxDepth, fenIndex) };
120
+ }
121
+ // Produce a PGN string for a subtree rooted at `anchor`, truncated
122
+ // at `maxDepth` plies below (null = unlimited). Reuses the exporter
123
+ // by building a synthetic PrepFile whose root is a shallow clone of
124
+ // the anchor with its children trimmed to depth.
125
+ function exportSubtreePgn(file, anchor, maxDepth) {
126
+ const trim = (n, depthLeft) => {
127
+ if (depthLeft !== null && depthLeft <= 0)
128
+ return { ...n, children: [] };
129
+ const next = depthLeft === null ? null : depthLeft - 1;
130
+ return { ...n, children: n.children.map(c => trim(c, next)) };
131
+ };
132
+ const trimmedAnchor = trim(anchor, maxDepth);
133
+ // If the anchor IS the root, exporter handles it. If it's an inner
134
+ // node, we set the root's FEN to the anchor's position and hang the
135
+ // trimmed subtree off it. Tags carried over.
136
+ if (anchor.id === ROOT_ID) {
137
+ return exportPGN({ tags: file.tags, root: trimmedAnchor });
138
+ }
139
+ const syntheticRoot = {
140
+ id: ROOT_ID,
141
+ san: null,
142
+ fen: anchor.fen,
143
+ ply: 0,
144
+ children: trimmedAnchor.children,
145
+ };
146
+ const tags = { ...file.tags, FEN: anchor.fen, SetUp: "1" };
147
+ return exportPGN({ tags, root: syntheticRoot });
148
+ }
149
+ export async function listNodes(args) {
150
+ const id = String(args.id);
151
+ const filter = String(args.filter || "");
152
+ const startNodeId = typeof args.node_id === "string" && args.node_id.length > 0 ? args.node_id : ROOT_ID;
153
+ const maxDepth = typeof args.max_depth === "number" && args.max_depth >= 0 ? args.max_depth : null;
154
+ const { file } = await loadPrepFile(id);
155
+ const idIndex = buildIdIndex(file.root);
156
+ const path = resolveNodeId(idIndex, startNodeId);
157
+ const anchor = getNodeByPath(file.root, path);
158
+ const fenIndex = filter === "transpositions" ? buildFenIndex(file.root) : null;
159
+ const hits = [];
160
+ const walk = (node, depthLeft, spineOnly) => {
161
+ // Root has no san — never emit it as a match. Everything else is fair game.
162
+ if (node.id !== ROOT_ID) {
163
+ let include = false;
164
+ let extra = {};
165
+ switch (filter) {
166
+ case "missing_eval":
167
+ include = !node.ceoEval;
168
+ break;
169
+ case "has_comment":
170
+ include = !!(node.comment && node.comment.length > 0);
171
+ if (include)
172
+ extra.comment_preview = (node.comment || "").slice(0, 80);
173
+ break;
174
+ case "has_annotations":
175
+ include = !!(node.annotations && (node.annotations.arrows.length > 0 || node.annotations.highlights.length > 0));
176
+ break;
177
+ case "novelties":
178
+ include = !!(node.nags && node.nags.includes("$146"));
179
+ break;
180
+ case "leaves":
181
+ include = node.children.length === 0;
182
+ break;
183
+ case "mainline":
184
+ include = spineOnly;
185
+ break;
186
+ case "transpositions": {
187
+ const group = fenIndex.get(positionKey(node.fen));
188
+ if (group && group.length > 1) {
189
+ include = true;
190
+ extra.transposes_to = group.filter(n => n.id !== node.id).map(n => n.id);
191
+ }
192
+ break;
193
+ }
194
+ case "all":
195
+ include = true;
196
+ break;
197
+ default:
198
+ throw new Error(`unknown filter: ${filter}`);
199
+ }
200
+ if (include) {
201
+ const hit = { node_id: node.id, san: node.san, ply: node.ply };
202
+ Object.assign(hit, extra);
203
+ hits.push(hit);
204
+ }
205
+ }
206
+ if (depthLeft !== null && depthLeft <= 0)
207
+ return;
208
+ const nextDepth = depthLeft === null ? null : depthLeft - 1;
209
+ if (filter === "mainline" && spineOnly) {
210
+ if (node.children.length > 0)
211
+ walk(node.children[0], nextDepth, true);
212
+ }
213
+ else {
214
+ for (const c of node.children)
215
+ walk(c, nextDepth, filter === "mainline");
216
+ }
217
+ };
218
+ const rootIsSpineForFilter = filter === "mainline";
219
+ walk(anchor, maxDepth, rootIsSpineForFilter);
220
+ return { file_id: id, filter, node_id: startNodeId, max_depth: maxDepth, count: hits.length, nodes: hits };
221
+ }
222
+ // list_transpositions — every position that occurs 2+ times in the
223
+ // file, so the LLM knows where its analysis / prose will double up.
224
+ export async function listTranspositions(args) {
225
+ const id = String(args.id);
226
+ const { file } = await loadPrepFile(id);
227
+ const fenIndex = buildFenIndex(file.root);
228
+ const groups = [];
229
+ for (const [key, arr] of fenIndex.entries()) {
230
+ if (arr.length < 2)
231
+ continue;
232
+ groups.push({
233
+ position_key: key,
234
+ size: arr.length,
235
+ node_ids: arr.map(n => n.id),
236
+ sans: arr.map(n => n.san),
237
+ });
238
+ }
239
+ groups.sort((a, b) => b.size - a.size || a.position_key.localeCompare(b.position_key));
240
+ const nodeCount = groups.reduce((s, g) => s + g.size, 0);
241
+ return { file_id: id, group_count: groups.length, node_count: nodeCount, groups };
242
+ }
@@ -0,0 +1,169 @@
1
+ // Response shape transformations applied to backend payloads before
2
+ // they reach the LLM. Two responsibilities:
3
+ // - Drop internal / debug-only fields the LLM doesn't need
4
+ // (`hash`, `plyNumber`, `relevance`, etc.).
5
+ // - Rewrite backend jargon into LLM-friendly names
6
+ // (`transpositions` → `reachedViaTransposition`, `hotness` →
7
+ // `fashionScore`, UCI moves → SAN).
8
+ //
9
+ // Plus a couple of tool-input adapters that also live here for lack of
10
+ // a better home:
11
+ // - `trimMovesToPly` (used by trimGamesMovetext, exported for reuse).
12
+ // - `normalizeSourceForBackend` (prepare_opponent request adapter).
13
+ //
14
+ // Extracted from index.ts in v0.44 as part of the file split.
15
+ import { uciMoveToSAN } from "./analysis/response.js";
16
+ // Strip cruft the LLM doesn't need from the DB-position response.
17
+ // Called AFTER trimGamesMovetext so plyNumber survives long enough to
18
+ // slice each game's movetext. Also renames the `transpositions` field
19
+ // to something the LLM can parse without knowing chess-DB jargon.
20
+ export function stripPositionResponse(r) {
21
+ if (!r || typeof r !== "object")
22
+ return;
23
+ const t = r;
24
+ delete t.hash; // internal zobrist string
25
+ delete t.source; // internal "database" marker; we overwrite with our own .source
26
+ delete t.totalGames; // duplicates statistics.totalCount often; hasMore covers pagination
27
+ if (Array.isArray(t.moves)) {
28
+ for (const m of t.moves) {
29
+ if (typeof m.transpositions === "number") {
30
+ m.reachedViaTransposition = m.transpositions;
31
+ delete m.transpositions;
32
+ }
33
+ // Backend calls it "hotness" — a 0-100 time-decayed popularity score
34
+ // (recent + played often = high). Rename to something an LLM can read
35
+ // without guessing it means "on a winning streak".
36
+ if (typeof m.hotness === "number") {
37
+ m.fashionScore = m.hotness;
38
+ delete m.hotness;
39
+ }
40
+ }
41
+ }
42
+ if (Array.isArray(t.games)) {
43
+ for (const g of t.games) {
44
+ delete g.gameId;
45
+ delete g.whiteTitle;
46
+ delete g.blackTitle;
47
+ delete g.whiteTeam;
48
+ delete g.blackTeam;
49
+ delete g.round;
50
+ delete g.plyNumber;
51
+ delete g.relevance;
52
+ delete g.site;
53
+ delete g.ply;
54
+ }
55
+ }
56
+ }
57
+ // Trim every game's `moves` field to just the plies AFTER the queried
58
+ // position, using each game's `plyNumber`. Massive token save — a game
59
+ // 80 plies long queried at ply 12 drops to ~68 plies of movetext. Ports
60
+ // the frontend's GamesTable.getMoveDisplay() trim logic.
61
+ export function trimGamesMovetext(response) {
62
+ if (!response || typeof response !== "object")
63
+ return;
64
+ const r = response;
65
+ if (!Array.isArray(r.games))
66
+ return;
67
+ for (const g of r.games) {
68
+ if (typeof g.moves === "string" && typeof g.plyNumber === "number" && g.plyNumber > 0) {
69
+ g.moves = trimMovesToPly(g.moves, g.plyNumber);
70
+ }
71
+ }
72
+ }
73
+ export function trimMovesToPly(moves, plyNumber) {
74
+ // Split into plain SAN tokens, dropping standalone move-number tokens
75
+ // ("1.", "12...") and any glued number prefix on a SAN token ("1.e4").
76
+ // Result markers ("*", "1-0", "0-1", "1/2-1/2") are stripped so they
77
+ // don't get counted as plies.
78
+ const tokens = [];
79
+ for (const chunk of moves.split(/\s+/)) {
80
+ if (!chunk)
81
+ continue;
82
+ const cleaned = chunk.replace(/^\d+\.+/, "");
83
+ if (!cleaned)
84
+ continue;
85
+ if (/^(1-0|0-1|1\/2-1\/2|\*)$/.test(cleaned))
86
+ continue;
87
+ tokens.push(cleaned);
88
+ }
89
+ const remaining = tokens.slice(plyNumber);
90
+ if (remaining.length === 0)
91
+ return "";
92
+ // Reconstruct with move numbering. First move gets "N..." if it's
93
+ // Black's move (starting the slice mid-move-pair), so the reader knows
94
+ // moves were dropped.
95
+ const out = [];
96
+ let ply = plyNumber;
97
+ for (let i = 0; i < remaining.length; i++) {
98
+ const san = remaining[i];
99
+ const moveNumber = Math.floor(ply / 2) + 1;
100
+ if (ply % 2 === 0) {
101
+ out.push(`${moveNumber}. ${san}`);
102
+ }
103
+ else if (i === 0) {
104
+ out.push(`${moveNumber}... ${san}`);
105
+ }
106
+ else {
107
+ out.push(san);
108
+ }
109
+ ply++;
110
+ }
111
+ return out.join(" ");
112
+ }
113
+ // Rewrite availableMoves[].move UCI → SAN. The prep + position-stats
114
+ // endpoints return moves in UCI on the wire — same LLM-readability
115
+ // concern as engine PVs, and the same wrapper-only fix. Passes the
116
+ // response through unchanged if there's no availableMoves array.
117
+ export function convertAvailableMovesToSAN(raw, fen) {
118
+ if (!raw || typeof raw !== "object")
119
+ return raw;
120
+ const r = raw;
121
+ if (!Array.isArray(r.availableMoves))
122
+ return raw;
123
+ for (const m of r.availableMoves) {
124
+ if (typeof m.move === "string" && m.move.length >= 4) {
125
+ m.move = uciMoveToSAN(fen, m.move);
126
+ }
127
+ }
128
+ return raw;
129
+ }
130
+ // Normalize one MCP `prepare_opponent` source into the shape the backend's
131
+ // /api/chess/prep/prepare-multi expects. Handles two impedance mismatches:
132
+ // - snake_case → camelCase (fide_id → fideId, start_month → startMonth, etc.)
133
+ // - the unified `time_control` string → per-source-type filter:
134
+ // * fide / chesscom → timeFormats: ["Classical" | "Rapid" | "Blitz"]
135
+ // * lichess → perfType: "classical" | "rapid" | "blitz" | "bullet"
136
+ // Backend validates required fields per source type, so we don't need to
137
+ // pre-reject missing username/fideId here — it'll come back as a 400 the
138
+ // LLM can act on.
139
+ export function normalizeSourceForBackend(src, idx) {
140
+ const type = typeof src.type === "string" ? src.type : "";
141
+ if (type !== "fide" && type !== "chesscom" && type !== "lichess") {
142
+ throw new Error(`sources[${idx}].type must be one of fide|chesscom|lichess (got ${JSON.stringify(src.type)})`);
143
+ }
144
+ const out = { type };
145
+ if (typeof src.fide_id === "number")
146
+ out.fideId = src.fide_id;
147
+ if (typeof src.username === "string" && src.username.trim() !== "")
148
+ out.username = src.username.trim();
149
+ if (typeof src.color === "string" && (src.color === "white" || src.color === "black"))
150
+ out.color = src.color;
151
+ if (typeof src.start_month === "string" && src.start_month.trim() !== "")
152
+ out.startMonth = src.start_month.trim();
153
+ if (typeof src.end_month === "string" && src.end_month.trim() !== "")
154
+ out.endMonth = src.end_month.trim();
155
+ if (typeof src.exclude_online === "boolean")
156
+ out.excludeOnline = src.exclude_online;
157
+ const tc = typeof src.time_control === "string" ? src.time_control : "";
158
+ if (tc) {
159
+ if (type === "lichess") {
160
+ out.perfType = tc;
161
+ }
162
+ else {
163
+ // fide + chesscom take a titlecased timeFormats array.
164
+ const titled = tc.charAt(0).toUpperCase() + tc.slice(1);
165
+ out.timeFormats = [titled];
166
+ }
167
+ }
168
+ return out;
169
+ }
package/dist/tools.js CHANGED
@@ -837,34 +837,41 @@ export const TOOLS = [
837
837
  },
838
838
  },
839
839
  {
840
- name: "read_engine_usage_guide",
841
- description: "Returns the full chess.ceo engine-usage guide: when to trust Stockfish (objective truth) vs Lc0 (practical eval), how to read disagreements between them, and how to use Lc0 contempt to find non-objective 'practical' ideas. Call this ONCE per session before running expensive `cloud_analyse` calls or when the user asks WHY the engines gave certain scores. Same content is also available as the `engine_usage_primer` prompt (for clients that surface prompts as slash commands), but many clients do not expose prompts to the model — this tool works everywhere.",
842
- inputSchema: { type: "object", properties: {} },
843
- },
844
- {
845
- name: "read_opening_prep_guide",
846
- description: "**CALL WHEN**: the user asks about OPENING PREPARATION — 'prep me against X', 'what should I play vs the Najdorf', 'help me build a repertoire against 1.e4', 'walk this opponent's Sveshnikov'. This guide is chess-and-analysis philosophy, not storage semantics.\n\n" +
847
- "Covers: why win% is one weight not a verdict, why prep is a two-player game with symmetric information (opponent sees your history too), how sample size and recency change the reading, when 'revealed weaknesses' are actionable vs already patched, how to choose between the GM-classical DB and the main DB, when to combine chesscom/lichess sources with FIDE, the three chess.com profile shapes (consistent / eclectic / split-personality), the reversed-colours scarcity trick, how to calibrate surprise (rare secondary lines inside the existing repertoire, not big first-move switches).\n\n" +
848
- "Different tool: `read_prep_files_guide` covers the FILE STORAGE feature (how to list/create/save prep files) — call that only when about to manipulate files, not for opening questions.",
849
- inputSchema: { type: "object", properties: {} },
850
- },
851
- {
852
- name: "read_prep_files_guide",
853
- description: "**CALL WHEN**: you're about to CREATE, LIST, SAVE, or DELETE a prep file — the persistent file storage feature. Not for opening prep philosophy (that's `read_opening_prep_guide`) and not for how to write PGN (that's `read_pgn_authoring_guide`).\n\n" +
854
- "Covers: the AI Prep folder, when to list vs search vs create (avoid duplicate 'Prep vs Firouzja' files), optimistic locking with `version`, naming conventions for the [Event] tag, node-id addressing basics.",
855
- inputSchema: { type: "object", properties: {} },
856
- },
857
- {
858
- name: "read_pgn_authoring_guide",
859
- description: "Returns the guide on how to write correct, useful PGN — mainline discipline, variations as moves (never prose describing moves), NAG symbols including novelty ($146), unclear ($13), compensation ($44) and the standard set, ChessBase arrow/coloured-square syntax ([%cal] / [%csl]), and common pitfalls the parser will reject. Call this ONCE per session before any save_prep_file call, or any time you're producing PGN output for the user.",
860
- inputSchema: { type: "object", properties: {} },
861
- },
862
- {
863
- name: "read_example_prep_files",
864
- description: "**CALL WHEN**: about to write ANY prose commentary in a prep file, ever. Even one comment. Even one variation. This is not optional and not once-per-project — call it early in the session and read the examples before your first `set_comment` or `apply_mutations` batch that includes comments. Log analysis showed <5% of sessions call this despite it being the single biggest quality lift documented in this MCP; that's the mistake this description is trying to fix.\n\n" +
865
- "Why: `read_pgn_authoring_guide` tells you the rules in prose. These files show you the *sound* of them applied by a strong human coach — comment density (short and load-bearing, not verbose), how citations look in-line (`WeiYi-Svidler` not `\"Svidler's choice at the FIDE World Blitz Team, June 2026\"`), when `$146` / `$3` / `$44` earn their place, when a bare `[%csl Rf7]` says everything a sentence would say. LLMs default to florid, restate-what's-visible commentary; reading these once inoculates against that.\n\n" +
866
- "Two files bundled with the MCP (not the user's own): one general opening overview (Italian Fried Liver, both sides, 1600+ audience) and one one-sided repertoire (Najdorf 6.f4 for White, 2200+ audience). Response: `{ overview: <pgn>, repertoire: <pgn> }` — raw PGN with comments, arrows, NAGs, stored evals intact.",
867
- inputSchema: { type: "object", properties: {} },
840
+ name: "read_docs",
841
+ description: "Fetch one or more bundled reference docs / example files in a single call. **Batch what you need in one call rather than reading them one at a time.**\n\n" +
842
+ "**Docs available** — each with when to call:\n\n" +
843
+ " • `engine-usage` — CALL WHEN: about to run `cloud_analyse`, explaining engine disagreements, or asked WHY the engines gave certain scores. Covers Stockfish (objective truth) vs Lc0 (practical eval), how to read disagreements, when to use Lc0 contempt.\n" +
844
+ " • `opening-prep` — CALL WHEN: the user asks about opening preparation ('prep me against X', 'what should I play vs the Najdorf', 'walk this opponent's Sveshnikov'). Covers why win% is one weight not a verdict, prep as a two-player game, sample-size/recency reading, when 'revealed weaknesses' are actionable vs patched, GM-classical vs main DB, chess.com/lichess/FIDE source combining, chess.com profile shapes, reversed-colours scarcity trick, surprise calibration.\n" +
845
+ " • `prep-files` — CALL WHEN: about to CREATE, LIST, SAVE, or DELETE a prep file. Covers list vs search vs create (avoid duplicates), optimistic locking with `version`, naming conventions for the [Event] tag, collection selection, node-id addressing basics.\n" +
846
+ " • `pgn-authoring` — CALL WHEN: about to write any comment / NAG / arrow / variation. Covers mainline discipline, variations as moves (not prose describing moves), NAG placement rules (position NAGs at ENDPOINTS only), the pasted-engine-PV anti-pattern, transposition handling, 'main tabiya' behavior, cover-N-alternatives rule, describe_position grounding.\n" +
847
+ " • `summary-authoring` — CALL WHEN: the user asks for a SUMMARY prep file — the 15-minute-read shape. Covers the two-file convention (reference vs summary), the coach's voice with real jvanf/Peter-Heine-Nielsen examples, endpoint NAG discipline, novelty and move-order-trick callouts, when to CUT branches rather than add them.\n" +
848
+ " • `examples/italian-fried-liver-overview` — bundled reference PGN, general opening overview at ~1600 audience. Shows comment density, NAG discipline, annotation style from a strong human coach.\n" +
849
+ " • `examples/najdorf-6-f4-repertoire` — bundled reference PGN, one-sided repertoire at ~2200 audience. Same purpose.\n\n" +
850
+ "**CALL WHEN in general**: early in the session before any substantive work, with `docs: [\"pgn-authoring\", \"examples/najdorf-6-f4-repertoire\"]` if writing prep, or `[\"opening-prep\", \"engine-usage\"]` if analysing. If you don't know which, ask for `[\"opening-prep\", \"prep-files\", \"pgn-authoring\"]` — three docs in one call is fine.\n\n" +
851
+ "Response: `{ docs: { <name>: <full content> } }`. Content is markdown for guides, raw PGN for examples.",
852
+ inputSchema: {
853
+ type: "object",
854
+ properties: {
855
+ docs: {
856
+ type: "array",
857
+ minItems: 1,
858
+ items: {
859
+ type: "string",
860
+ enum: [
861
+ "engine-usage",
862
+ "opening-prep",
863
+ "prep-files",
864
+ "pgn-authoring",
865
+ "summary-authoring",
866
+ "examples/italian-fried-liver-overview",
867
+ "examples/najdorf-6-f4-repertoire",
868
+ ],
869
+ },
870
+ description: "Names of docs to fetch. See the tool description for the full list + when-to-call hints.",
871
+ },
872
+ },
873
+ required: ["docs"],
874
+ },
868
875
  },
869
876
  {
870
877
  name: "prep_snapshot",
package/dist/warnings.js CHANGED
@@ -54,9 +54,18 @@ export function commentAntiPatterns(comment) {
54
54
  }
55
55
  // Raw centipawn in prose: "+0.35", "-0.20", "+80" (not preceded by move
56
56
  // number). Also "at depth N" or "N nodes" — engine metadata as prose.
57
+ //
58
+ // ceoEval itself is NOT rendered — it's LLM-internal state. But raw cp
59
+ // values in prose are still bad for a different reason: they're opaque
60
+ // decoration. "≈-60" gives the reader no chess signal without knowing
61
+ // the unit (centipawns? spread rank? pawns?) and no scale (is -60
62
+ // slight, meaningful, or losing?). The NAG glyph IS visible and is
63
+ // the intended channel for that judgment — one ⩱ conveys what "-60"
64
+ // fails to convey. Engine metadata like "at depth 24" or "259M nodes"
65
+ // is even weaker: it's process detail, not a claim about the position.
57
66
  if (/(?:^|[^\d.])[+-]\d\.\d\d(?!\d)/.test(comment) || /≈\s*[+\-−]?\d{2,3}\b/.test(comment) ||
58
67
  /\bat depth \d+\b/i.test(comment) || /\b\d{2,3}M nodes\b/.test(comment)) {
59
- warns.push("comment contains raw centipawn values or engine metadata — the app renders ceoEval + NAG glyph next to every node, so these numbers are doubled noise AND opaque (readers can't tell if ≈-60 means eval, spread, or something else). Set the NAG (set_nags) and let the glyph carry the judgment; drop the number from the prose.");
68
+ warns.push("comment contains raw centipawn values or engine metadata — these are opaque to the reader (no clear unit or scale) and the intended channel for the position-quality signal is the NAG glyph, which IS rendered. Set the NAG (set_nags) at the variation's endpoint and let the glyph carry the judgment; drop the number and any \"at depth N\" / \"N nodes\" fragments from the prose.");
60
69
  }
61
70
  // Roster: "N GM games" pattern
62
71
  if (/\b\d{2,4}\s+GM games\b/i.test(comment)) {
@@ -101,6 +110,40 @@ export function noDescribeWarning(node, comment) {
101
110
  noDescribeWarned.add(node.id);
102
111
  return `substantive comment (${comment.length} chars) on a node whose position was never grounded via describe_position this session (id=${node.id}, ${node.san}). LLMs invent captures, miscount pieces, and swap files/ranks when reading FEN strings — describe_position is a pure-computation pass (~1 ms, no engine cost, structural facts + Stockfish's per-term eval breakdown) that reliably prevents this class of hallucination. In live audits, prose accuracy jumps sharply on nodes where describe_position was called first. Call describe_position with file_id+node_id=${node.id} BEFORE writing prose. Warned once per node.`;
103
112
  }
113
+ // Position NAGs on intermediate moves ("everything is ⩲" spam).
114
+ //
115
+ // Position NAGs — `$10` = / `$11` = / `$13` ∞ / `$14` ⩲ / `$15` ⩱ /
116
+ // `$16` ± / `$17` ∓ / `$18` +− / `$19` −+ — are visible glyphs on the
117
+ // move. They belong at variation ENDPOINTS: the reader plays through
118
+ // a line and, at the end, wants to know "so where did we land?"
119
+ // Tagging every mainline move with `$14` (routine slight White edge)
120
+ // turns the movetext into a wall of ⩲ symbols the reader skims past;
121
+ // it also pre-empts the walk-through by hard-coding the verdict at
122
+ // every step. The single leaf NAG carries the same information with
123
+ // none of the noise.
124
+ //
125
+ // Real-world case (Ruy Lopez Bc5 file, 2026-07-30): `$14` set on ~15
126
+ // mainline nodes plus 10+ intermediate move-choice nodes. Nothing
127
+ // signaled where the variation actually converged.
128
+ //
129
+ // Rule this warning encodes: position NAGs on non-leaf nodes are
130
+ // almost always noise. Move-quality NAGs — `$1` !, `$2` ?, `$3` !!,
131
+ // `$4` ??, `$5` !?, `$6` ?!, and `$146` novelty — are FINE at any
132
+ // depth because they're statements about the MOVE, not the resulting
133
+ // position. Warned once per node.
134
+ const positionalNagWarned = new Set();
135
+ export function positionalNagOnIntermediateWarning(node, nags) {
136
+ const positional = new Set(["$10", "$11", "$12", "$13", "$14", "$15", "$16", "$17", "$18", "$19"]);
137
+ const hit = nags.filter(n => positional.has(n));
138
+ if (hit.length === 0)
139
+ return undefined;
140
+ if (node.children.length === 0)
141
+ return undefined; // leaf — legitimate placement
142
+ if (positionalNagWarned.has(node.id))
143
+ return undefined;
144
+ positionalNagWarned.add(node.id);
145
+ return `positional NAG ${hit.join(" ")} set on a non-leaf node (id=${node.id}, ${node.san}, ${node.children.length} children). Position NAGs (=, ⩲, ±, +−) belong at variation ENDPOINTS — the reader plays through the line and at the leaf wants to know how it lands. Marking every intermediate move with the same ⩲ turns the movetext into visual noise the eye skims past AND pre-empts the walk-through. Either move this NAG to the leaf, drop it entirely, or leave it only if THIS specific move is the one that tipped the balance (rare, and worth prose). Move-quality NAGs on intermediate moves — !, ?, !?, ?!, $146 novelty — are FINE, because they're statements about the move not the resulting position. Warned once per node.`;
146
+ }
104
147
  // Compute the "you never DB-checked this parent" warning. Called from
105
148
  // add_move / add_line handlers with the parent node. Returns undefined
106
149
  // when either (a) the parent was checked this session (or is root — the
@@ -40,7 +40,9 @@ All mutations **auto-save** with optimistic locking. Response includes the new `
40
40
 
41
41
  ## Typical build order
42
42
 
43
- 0. **`read_example_prep_files`** — do this once per session, before writing any prose. Not optional. Log analysis shows most sessions skip this and produce documented anti-patterns (long PVs in prose, restating what the app renders, verbose citations). Reading the two bundled reference files once inoculates the LLM against those.
43
+ 0a. **Cloud engine running?** Call `list_cloud_engines` first. Every substantive step below needs Stockfish + Lc0 to be reachable — `cloud_analyse` at critical positions, `auto_evaluate` for the whole tree, engine-derived NAGs at endpoints, describe_position's Stockfish eval-terms breakdown. If the caller has zero running combos: STOP, tell the user prep needs an engine, list options via `list_cloud_machine_options`, get the SKU + explicit confirmation (real money per second), then `start_cloud_engine`. If they already have one, note the contract_id and continue. Never silently write prep without engines — the file ends up with placeholder NAGs the user has no way to distinguish from real ones.
44
+
45
+ 0b. **`read_docs({ docs: ["pgn-authoring", "examples/najdorf-6-f4-repertoire"] })`** — do this once per session, before writing any prose. Not optional. Log analysis shows most sessions skip the example files and produce documented anti-patterns (long PVs in prose, restating what the app renders, verbose citations). Reading the reference PGN once inoculates the LLM against those.
44
46
  1. `read_prep_file` — see what's there. Every node has an `id` you'll pass to the mutation and engine/DB tools. Use `view: "compact"` (default) plus `node_id` + `max_depth` to scope; the full tree of a 500+-node file can blow the token limit.
45
47
  2. `apply_mutations([...])` — one call with your whole intended build (a mix of `add_move` / `add_line` for structure, plus any `set_comment`/`set_annotations` you already know at author time, plus any `set_nags` where you already have a clear judgment — novelty `$146`, `!?` speculative sac, obvious `?` blunder in a sideline you're rejecting).
46
48
  3. `auto_evaluate(id)` — spawns a background job that PERSISTS engine numbers on every node. Does not touch visible NAGs. Cheap way to get every position's Stockfish + Lc0 read baked into the file for later reference. Grab the returned `job_id` and either (a) poll `auto_evaluate_status(job_id)` every ~10-30s until done, or (b) fire and do useful work meanwhile (write more of the tree, walk the opponent's repertoire) and check back later — engine walk-time serialises on the per-combo semaphore, so it takes roughly `target_count × movetime_ms` in wall time.
@@ -214,6 +216,10 @@ This is the single biggest quality problem in current LLM output on this system:
214
216
 
215
217
  The failure mode you're avoiding: writing prep that reads as if the opponent will helpfully play the engine's #1 preference at every ply. They won't; that's the whole point of prep.
216
218
 
219
+ **"Main tabiya" (or "main line", "main try", "Main choice") is the START of analysis, not the end.** Live case (Ruy Lopez Bc5 file, 2026-07-30): the author tagged a position "Main tabiya" and stopped there — despite the DB showing many high-level games with divergent replies from that exact position. If a position is important enough to CALL a tabiya, it's important enough to fully cover: every reply at ~15%+ frequency, or every distinct plan, gets its own branch. Writing "Main tabiya" and leaving it with one continuation is the shape of "I skimmed the DB and picked one" — the exact anti-pattern this system is meant to prevent. Rule of thumb: if you named a node "tabiya" / "main try" / "critical" in prose, the number of children beneath it should be at least the number of DB moves you named as popular.
220
+
221
+ **Cover N alternatives when the DB shows N.** From the same live file: at one move-8 position, `get_position_stats` showed three roughly-equally-played tries; the LLM added exactly one branch. That's not prep, that's a hint. If frequencies were 40 / 30 / 25 / 5, the first three get branches (the 5% one gets skipped or a one-line dismissal); if the top four are all 20-25%, all four get branches. The heuristic isn't "pick the top" — it's "cover what the opponent might actually play." When ceoEvals across the top candidates sit within 0.15 of each other, that's not one Best Move, that's a menu — treat it as one.
222
+
217
223
  ## Course chapters describe COVERAGE, not consensus — always `get_position_stats` at mainline branch points
218
224
 
219
225
  Concrete failure this rule was written to fix: a Modern Defence file made 6.O-O-O the mainline of the entire "5.Qd2 Nd7" branch, wrote a chapter around it, and analysed 15+ plies deep. `get_position_stats` at that position was NEVER called this session. Instead, the LLM ran `find_position_in_courses`, saw So / Kraai / Mihajlov all had chapters titled "5.Qd2 b5 6.O-O-O Bb7" — and cargo-culted that into "6.O-O-O is the main line". It isn't. 6.O-O-O is one of five White tries, and it wasn't even the most-played.
@@ -292,6 +298,12 @@ Where NAGs actually earn their place:
292
298
 
293
299
  Where NAGs are noise: every `$10` "=" glyph on every equal position. If the whole tree is 0.00, that's the *default state* — leave it unmarked and the reader understands.
294
300
 
301
+ **Position NAGs (`$10`-`$19`) belong at variation ENDPOINTS, not on every intermediate move.** This is the rule the Ruy Lopez Bc5 file (2026-07-30) failed: `$14` set on ~15 mainline nodes plus 10+ intermediate move-choice nodes. Result: a wall of ⩲ symbols throughout the movetext, with nothing signalling where the line actually converges.
302
+
303
+ A variation is walkable. The reader plays through the moves and, at the LEAF, wants to know "so where did we land?" One `$14` at the endpoint answers that. Twenty `$14`s along the way don't say more — they say less, because the reader's eye skims past them and the endpoint no longer stands out.
304
+
305
+ Concrete rule: **position NAGs (`$10` = / `$13` ∞ / `$14` ⩲ / `$15` ⩱ / `$16` ± / `$17` ∓ / `$18` +− / `$19` −+) on non-leaf nodes trigger a warning.** Either move the NAG to the leaf, drop it, or leave it only if THIS specific move is the one that TIPPED the balance (rare, and worth prose naming the shift). Move-quality NAGs (`$1` ! / `$2` ? / `$3` !! / `$4` ?? / `$5` !? / `$6` ?! / `$146` novelty) on intermediate moves are FINE at any depth — those are statements about the move, not the resulting position.
306
+
295
307
  When reading back a file, `node.ceoEval.nag` carries the threshold-derived NAG (`$10` / `$14` / `$16` / `$18` etc.) — treat it as a *suggestion*, not an automatic write. Promote it to a visible NAG only when a glyph on that move actually helps the reader.
296
308
 
297
309
  **Persisted evals travel with the file.** Every node with an eval gets `ceoEval: { sf: {cp: 25, depth: 32}, lc0: {cp: 30, depth: 18}, nag: "$14" }` on subsequent `read_prep_file` calls. Stored as `[%ceo-eval sf=+0.25/32 lc0=+0.30/18 nag=$14]` inside the PGN comment (an escape tag the app hides from the board view, just like [%cal] / [%csl]). So a re-read after auto_evaluate gives you every position's number without re-running cloud_analyse. Query tools (get_position_stats, get_prep_position, prep_snapshot) also auto-attach a live `eval` at the request position when a cloud engine is running.