@chessceo/mcp 0.40.0 → 0.43.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -73,8 +73,10 @@ const AUTHED_TOOLS = new Set([
73
73
  "list_cloud_engines",
74
74
  "stop_cloud_engine",
75
75
  "cloud_analyse",
76
+ "list_collections",
76
77
  "list_prep_files",
77
78
  "search_prep_files",
79
+ "find_position_in_files",
78
80
  "read_prep_file",
79
81
  "list_nodes",
80
82
  "list_transpositions",
@@ -174,6 +176,72 @@ async function authedRequest(method, path, body) {
174
176
  }
175
177
  return text.length ? JSON.parse(text) : null;
176
178
  }
179
+ // ── Backend I/O layer ──────────────────────────────────────────────
180
+ //
181
+ // v0.43: MCP surface moved off the single-`/mcp`-folder model onto the
182
+ // full user PGN library. LLM-facing tool ids are opaque composites of
183
+ // the form `<collection_id>:<game_id>` so every existing tool that
184
+ // takes `id` keeps taking `id` — the composite splits back into two
185
+ // pieces at the HTTP layer. Backend routes live under
186
+ // `/api/agent/pgns/*` (same handlers as `/me/pgns/*` browser routes;
187
+ // a path-rewrite middleware in the server main.go maps the two).
188
+ const PGN_BASE = "/api/agent/pgns";
189
+ // Browser handlers wrap successful responses as
190
+ // { success: true, message: "...", data: <actual thing> }
191
+ // via handlers.RespondSuccess. Peel that off; leave anything without
192
+ // the envelope unchanged (some endpoints return raw bodies).
193
+ function unwrap(raw) {
194
+ if (raw &&
195
+ typeof raw === "object" &&
196
+ "data" in raw &&
197
+ raw.success === true) {
198
+ return raw.data;
199
+ }
200
+ return raw;
201
+ }
202
+ // LLM-facing "prep file id" is always "<collection_id>:<game_id>" —
203
+ // makeFileId to compose from a listing, splitFileId at the HTTP edge.
204
+ function makeFileId(collectionId, gameId) {
205
+ return `${collectionId}:${gameId}`;
206
+ }
207
+ function splitFileId(id) {
208
+ const idx = id.indexOf(":");
209
+ if (idx <= 0 || idx === id.length - 1) {
210
+ throw new Error(`invalid prep file id "${id}" — expected "<collection_id>:<game_id>" ` +
211
+ `(get one from list_prep_files, search_prep_files, find_position_in_files, ` +
212
+ `or the create_prep_file return value)`);
213
+ }
214
+ return { collectionId: id.slice(0, idx), gameId: id.slice(idx + 1) };
215
+ }
216
+ // GET one game by composite id. Throws on 404.
217
+ async function fetchGame(id) {
218
+ const { collectionId, gameId } = splitFileId(id);
219
+ const raw = await authedRequest("GET", `${PGN_BASE}/${encodeURIComponent(collectionId)}/games/${encodeURIComponent(gameId)}`);
220
+ const g = unwrap(raw);
221
+ if (!g || typeof g.pgnContent !== "string") {
222
+ throw new Error("prep file missing pgnContent");
223
+ }
224
+ return g;
225
+ }
226
+ // PUT the game body. Returns the saved game (new version).
227
+ async function saveGame(id, pgn, expectedVersion) {
228
+ const { collectionId, gameId } = splitFileId(id);
229
+ const body = { pgnContent: pgn };
230
+ if (typeof expectedVersion === "number")
231
+ body.baseVersion = expectedVersion;
232
+ const raw = await authedRequest("PUT", `${PGN_BASE}/${encodeURIComponent(collectionId)}/games/${encodeURIComponent(gameId)}`, body);
233
+ return unwrap(raw);
234
+ }
235
+ // POST a new game into the given collection. Returns the created game.
236
+ async function createGame(collectionId, pgn) {
237
+ const raw = await authedRequest("POST", `${PGN_BASE}/${encodeURIComponent(collectionId)}/games`, { pgnContent: pgn });
238
+ return unwrap(raw);
239
+ }
240
+ // DELETE (soft) a game by composite id.
241
+ async function deleteGame(id) {
242
+ const { collectionId, gameId } = splitFileId(id);
243
+ await authedRequest("DELETE", `${PGN_BASE}/${encodeURIComponent(collectionId)}/games/${encodeURIComponent(gameId)}`);
244
+ }
177
245
  // ── Tool definitions ───────────────────────────────────────────────
178
246
  //
179
247
  // Descriptions are written for the LLM, not humans — they should hint
@@ -332,7 +400,8 @@ const TOOLS = [
332
400
  },
333
401
  {
334
402
  name: "describe_position",
335
- description: "Everything you need to understand a position in one call. USE BEFORE COMMENTING — pieces get misplaced when reading a FEN, hanging pieces missed, 'the knight on d5' turns out to not exist. ~50-100 ms per call (chess-primitive analysis is instant; the Stockfish leg dominates wall time).\n\n" +
403
+ description: "**CALL WHEN**: about to write ANY comment on a position that describes what's happening on the board — piece activity, structure, plans, weaknesses. This is the single biggest lever for prose quality in the whole system. Live audit: nodes where describe_position was called first produced comments grounded specifically in the position (correct piece squares, real pawn structure, actual weak squares); nodes where it wasn't produced generic pattern-matched prose that confidently named pieces on wrong squares. `set_comment` now emits a warning whenever a substantive comment lands on a node whose position was never grounded via describe_position this session — that warning is telling you to fix a class of hallucination that already showed up in your output. Cheap: chess-primitive analysis is instant, Stockfish leg is ~50-100 ms, no billing.\n\n" +
404
+ "Everything you need to understand a position in one call. Pieces get misplaced when reading a FEN, hanging pieces missed, 'the knight on d5' turns out to not exist.\n\n" +
336
405
  "Returns three layers:\n\n" +
337
406
  "**Board state** — piece placements per colour, material balance in pawn units, contested pieces (attackers + defenders), hanging pieces, checkers if in check, castling rights, en passant, side to move, full LEGAL MOVES list. Use `.legalMoves` when `add_move` rejects an illegal SAN.\n\n" +
338
407
  "**Structural analysis** — chess-concept observations a human sees at a glance:\n" +
@@ -480,6 +549,7 @@ const TOOLS = [
480
549
  "• When they agree → high confidence. When they disagree → look at both scores and reason WHY (Stockfish sharply higher = tactic Lc0 missed; Lc0 higher = long-term positional edge past Stockfish's horizon). Never dismiss either — the disagreement is the signal.\n\n" +
481
550
  "Contempt (`contempt`) skews Lc0 (only Lc0 — Stockfish always stays objective) toward White (positive) or Black (negative). Signed 0-100 strength — same scale as the web UI's ContemptStrength slider (the server multiplies by 8 to produce Lc0's internal cp bias). Typical values: ±15 for a light nudge, ±30-60 for real fighting play, ±80-100 for maximum steer. Use it to find non-objective 'practical' ideas or when the user needs to lean toward fighting/solid lines with a specific colour. Do NOT quote a contempt-biased eval as objective — cross-check with Stockfish.\n\n" +
482
551
  "Also useful: pass `moves` on top of `fen` to explore a variation without computing FENs yourself (e.g. fen='<tabiya>', moves='b4 a5 c3'). And the flip-side-to-move threat check documented in the guide is a great free trick.\n\n" +
552
+ "**PVs are capped at 6 plies by default (3 full moves), and lines that got truncated are marked with `pv_truncated: true`.** This is deliberate: the tail of a PV is where the engine's confidence collapses, AND pasting a long PV into `add_line` as if it were prepared repertoire is the #1 documented anti-pattern of this MCP — a 15-move PV is one line of engine output through positions where both sides had real choices, not a repertoire. To see further, don't raise `pv_max_plies`; instead, walk the tree one branch at a time with a fresh `cloud_analyse` at each position where the opponent has real alternatives — that's what makes it prep instead of pasted output. Only raise the cap when you're verifying a forcing sequence (a mate, a forced tactical resolution), not to build lines.\n\n" +
483
553
  "For the full guide including worked examples, call the `read_engine_usage_guide` tool.\n\n" +
484
554
  "Not for casual questions — this costs real money per second. Use `get_position_stats` for anything that doesn't require deep prep.\n\n" +
485
555
  "**When called with `file_id`+`node_id` (preferred inside a prep file), the resulting eval is auto-stored on that node's `ceoEval` — you can then quote it with quote_engine_eval on any later call.** This is what makes engine attribution trustworthy: prose that says 'engines say X on node Y' can only be true if a call was actually made against node_id=Y.",
@@ -522,17 +592,37 @@ const TOOLS = [
522
592
  items: { type: "string", enum: ["stockfish", "lc0"] },
523
593
  description: "Which engines to run. Default = both. Use `[\"lc0\"]` to skip Stockfish (e.g. while a deep_analyse job is holding the SF slot on the same combo). Use `[\"stockfish\"]` when only the objective read matters. The skipped engine's field is omitted from the response.",
524
594
  },
595
+ pv_max_plies: {
596
+ type: "integer",
597
+ minimum: 1,
598
+ maximum: 40,
599
+ description: "Cap each returned PV to this many plies (default 6 = 3 full moves). PVs beyond ~6 plies are speculative and are the anti-pattern behind pasted-engine-line 'prep' — don't raise unless you're specifically checking a forcing tactic or verifying a mate. When a line was truncated, the response marks it with `pv_truncated: true`.",
600
+ },
525
601
  },
526
602
  },
527
603
  },
528
604
  {
529
- name: "list_prep_files",
530
- description: "List every prep file the user has (all games inside their dedicated AI Prep collection). Returns id, PGN tags (Event, White, Black, Date, etc. — read the Event tag for the user-facing name), size, updated_at. ALWAYS call this before create_prep_file to check for existing coverage — creating a second 'Prep vs Firouzja' when one already exists is a common LLM failure. If the user has many, use search_prep_files with a query to narrow down.",
605
+ name: "list_collections",
606
+ description: "List the user's own PGN collections — every collection they've created, not just prep files. Response: `{collections: [{id, title, icon, folder_path, game_count, updated_at, position_search_enabled}]}`.\n\n" +
607
+ "**Call this before create_prep_file** — the LLM must pick a collection to write into (no default landing folder any more; the old hidden `/mcp` collection was retired in v0.43). Also useful when the user asks a position-shaped question — LLM can then run find_position_in_files to see which of these collections already covers the position.\n\n" +
608
+ "Encrypted collections (client-side-encrypted PGN) are excluded — the server can't read their contents, so they'd be dead weight on this surface.",
531
609
  inputSchema: { type: "object", properties: {} },
532
610
  },
611
+ {
612
+ name: "list_prep_files",
613
+ description: "List the games (prep files) inside one of the user's collections. **Requires `collection_id`** — call `list_collections` first if you don't have one. Returns id (composite `<collection_id>:<game_id>`, opaque to the LLM — pass as-is to read_prep_file / mutation tools), PGN header fields, updated_at.\n\n" +
614
+ "For cross-collection discovery use `search_prep_files` (text) or `find_position_in_files` (position); list_prep_files is the browse-one-collection tool.",
615
+ inputSchema: {
616
+ type: "object",
617
+ properties: {
618
+ collection_id: { type: "string", description: "Collection id from list_collections. Required." },
619
+ },
620
+ required: ["collection_id"],
621
+ },
622
+ },
533
623
  {
534
624
  name: "search_prep_files",
535
- description: "Text search over the user's prep files (matches PGN headers, comments, and content). Use this instead of list_prep_files when you know a keyword — e.g. search_prep_files(query='Firouzja') or search_prep_files(query='Najdorf').",
625
+ description: "Text search over the user's prep files ACROSS ALL their collections (matches PGN headers, comments, and content). Use when you know a keyword — e.g. search_prep_files(query='Firouzja') or search_prep_files(query='Najdorf'). Cheaper than paging list_collections + list_prep_files to find one file by name.",
536
626
  inputSchema: {
537
627
  type: "object",
538
628
  properties: {
@@ -541,6 +631,22 @@ const TOOLS = [
541
631
  required: ["query"],
542
632
  },
543
633
  },
634
+ {
635
+ name: "find_position_in_files",
636
+ description: "Position search across every one of the user's EDITABLE prep files (all their non-encrypted collections). Given a FEN, returns which of the user's files reach that exact position (or a transposition of it — matched by zobrist hash, so move-order variants are found automatically). Recency-sorted.\n\n" +
637
+ "Distinct from `find_position_in_courses`: courses are READ-ONLY reference material (Chessable PGNs, downloaded backups); this searches the user's OWN editable prep. Common workflow: user asks about a position → call this first to see if their existing prep covers it → if yes, extend that file; if no, consider whether to start new prep.\n\n" +
638
+ "Position input: `file_id`+`node_id` (from an already-open prep file), or `fen`, or `moves` from startpos, or `fen`+`moves`.",
639
+ inputSchema: {
640
+ type: "object",
641
+ properties: {
642
+ file_id: { type: "string", description: "Prep file id. With `node_id`, derives FEN from the tree." },
643
+ node_id: { type: "string", description: "Node id inside `file_id`. Root is 'r'." },
644
+ fen: { type: "string", description: "Position as FEN. Only used if `file_id`/`node_id` not set." },
645
+ moves: { type: "string", description: "SAN moves from startpos (or on top of fen)." },
646
+ line: { type: "string", description: "Alias for moves." },
647
+ },
648
+ },
649
+ },
544
650
  {
545
651
  name: "read_prep_file",
546
652
  description: "Read one prep file. Response always includes `id`, `version`, `tags`. The tree/PGN part is controlled by `view` and `node_id`/`max_depth` — large files (500+ nodes) can otherwise blow the LLM's token limit.\n\n" +
@@ -604,17 +710,23 @@ const TOOLS = [
604
710
  },
605
711
  {
606
712
  name: "create_prep_file",
607
- description: "Create a new (empty) prep file. `name` becomes the Event PGN tag. You then extend it with mutation tools (add_move, set_comment, …).\n\n" +
608
- "ALWAYS call list_prep_files (or search_prep_files with the opponent / opening keyword) FIRST — creating a duplicate 'Prep vs Firouzja' when one exists is the #1 LLM failure mode. If a file already covers the topic, add moves to that one instead.",
713
+ description: "Create a new (empty) prep file in the specified collection. `name` becomes the Event PGN tag. You then extend it with mutation tools (add_move, set_comment, …).\n\n" +
714
+ "**collection_id is REQUIRED** — call `list_collections` first to pick where it lives. There is no default landing folder any more (v0.43: the old hidden `/mcp` collection was removed; prep files now live wherever the user organizes them).\n\n" +
715
+ "**Duplicate-check first.** Call `search_prep_files(query=<opponent / opening keyword>)` OR `find_position_in_files(fen=...)` before creating — a second 'Prep vs Firouzja' file when one already exists is a common LLM failure mode. If a file already covers the topic, extend that one instead.\n\n" +
716
+ "Response: `{ok, id, collection_id, version}` — `id` is a composite you pass to every other prep-file tool as `id` or `file_id`.",
609
717
  inputSchema: {
610
718
  type: "object",
611
719
  properties: {
720
+ collection_id: {
721
+ type: "string",
722
+ description: "Collection id from list_collections. Required — this is where the new file lands.",
723
+ },
612
724
  name: {
613
725
  type: "string",
614
726
  description: "User-facing name — becomes the [Event] tag. Example: 'Prep vs Firouzja (Black) 2026-07-23'.",
615
727
  },
616
728
  },
617
- required: ["name"],
729
+ required: ["collection_id", "name"],
618
730
  },
619
731
  },
620
732
  {
@@ -646,7 +758,8 @@ const TOOLS = [
646
758
  {
647
759
  name: "add_line",
648
760
  description: "Append a linear sequence of moves under `parent_id`. Each SAN in the list becomes the mainline child of the previous — one call instead of N add_move calls for a straight variation. If the parent already has other children, this whole line is appended as a variation (promote_variation the first move if you want it as the mainline).\n\n" +
649
- "Auto-saves. Returns `{node_id, line: [{node_id, san}, ...], version}` — `node_id` is the last (leaf) node's id, `line` is every node created in order so you can address any of them next.",
761
+ "**Anti-pattern: pasting an engine PV as a single long `add_line`.** Real prep is a tree, not a line. Almost every position along a variation has more than one plausible move — pasting a 12+-ply engine PV without branching at those points is the #1 documented failure mode of this MCP: it produces a page that reads as prep but ignores every decision the opponent actually gets to make. Long unbranched lines get a warning field in the response starting at ~9 plies and a strong warning at 14+ plies. Rule of thumb: if you added ≥8 plies in one call, at least half of them should have branched. Genuine exceptions exist (forced mates, obligated exchange sequences) — in those cases add a comment naming what makes the sequence forced (`{Every move here is forced by the mate threat.}`), so the reader knows it's forced by chess, not by LLM laziness.\n\n" +
762
+ "Auto-saves. Returns `{node_id, line: [{node_id, san}, ...], version}` — `node_id` is the last (leaf) node's id, `line` is every node created in order so you can address any of them next. When long-and-linear, also includes `warning: \"...\"`.",
650
763
  inputSchema: {
651
764
  type: "object",
652
765
  properties: {
@@ -660,7 +773,10 @@ const TOOLS = [
660
773
  },
661
774
  {
662
775
  name: "set_comment",
663
- description: "Set (or clear, with empty string) the text comment on the node identified by `node_id`. Comments are for plans, prep-signal, and interpretation the app can't derive — NOT for describing moves that should be variations instead. Auto-saves.",
776
+ description: "Set (or clear, with empty string) the text comment on the node identified by `node_id`. Comments are for plans, prep-signal, and interpretation the app can't derive — NOT for describing moves that should be variations instead. Auto-saves.\n\n" +
777
+ "**Two guardrails fire in the response as `warnings: [...]`:**\n" +
778
+ " 1. **Content scan** — comments containing spread lists (`≈50, ≈42, …`), raw centipawn values (`≈−60`, `+0.35`, `at depth 24`), or roster restatement (`146 GM games — Nakamura, …`) are all restating what the app already renders. The warning names the fix (set the NAG and drop the number; label the character not the numbers; cite a specific game instead of a count).\n" +
779
+ " 2. **Ungrounded prose** — substantive comments (≥40 chars) on a node whose position was never passed to `describe_position` this session are prone to hallucinated structural claims (piece on wrong square, invented captures, misidentified pawn structure). Call `describe_position` with `file_id`+`node_id` BEFORE writing prose about the position; the same node's warning clears once the position is described.",
664
780
  inputSchema: {
665
781
  type: "object",
666
782
  properties: {
@@ -1070,6 +1186,98 @@ async function fetchCompactEval(fen) {
1070
1186
  }
1071
1187
  // Rewrite the /api/agent/cloud-engines/analyse response (two engines,
1072
1188
  // each with lines[] and a bestMove) so PVs and bestMove come back in SAN.
1189
+ // Session-lifetime memory of which positions the LLM has actually asked
1190
+ // the DB about via `get_position_stats`. Keyed by the 3-field FEN
1191
+ // (piece placement + side to move + castling — same key used for
1192
+ // transposition detection). Used to warn on `add_move` / `add_line` under
1193
+ // a parent the LLM never DB-checked, which is the exact shape of the
1194
+ // bug where the LLM read course chapters and cargo-culted a "mainline"
1195
+ // that the actual games at the position don't play.
1196
+ //
1197
+ // One MCP server process per user, so this Set is effectively per-user
1198
+ // for the length of a session. Not persisted — a new session starts empty.
1199
+ const positionsStatsChecked = new Set();
1200
+ // Nodes we've ALREADY warned on for the "no stats check" pattern this
1201
+ // session, so repeated adds under the same parent don't spam the LLM.
1202
+ const noStatsWarned = new Set();
1203
+ // Session-lifetime memory of positions the LLM has called
1204
+ // `describe_position` on. Same 3-field FEN key. Live-log audit
1205
+ // (2026-07-27 Modern Defence session): 13 describe_position calls
1206
+ // vs 50+ set_comment ops — most comments were written blind. When
1207
+ // describe_position IS called before commentary, prose accuracy
1208
+ // jumps sharply (user's own observation). This tracks the same way
1209
+ // as positionsStatsChecked and drives the noDescribeWarning below.
1210
+ const positionsDescribed = new Set();
1211
+ // Once-per-node dedup for the describe warning.
1212
+ const noDescribeWarned = new Set();
1213
+ // Detect the anti-patterns the LLM keeps producing in comment prose.
1214
+ // All of these restate what the app already renders elsewhere:
1215
+ // - spread lists ("5.O-O ≈50, 6.h3 ≈42, ...")
1216
+ // - raw centipawn values in prose ("≈-60", "+0.35", "at depth 24")
1217
+ // - long roster restatement ("146 GM games, Nakamura, Kramnik, MVL")
1218
+ // Return an array of warning strings — one per matched category — so the
1219
+ // LLM sees exactly which pattern to remove.
1220
+ function commentAntiPatterns(comment) {
1221
+ if (!comment || typeof comment !== "string")
1222
+ return [];
1223
+ const warns = [];
1224
+ // Spread list: ≈ followed by a 2-3-digit number, appearing 2+ times
1225
+ // (one appearance is a stray, two+ is a comma-separated spread the LLM
1226
+ // pasted from stats output).
1227
+ const spreadMatches = comment.match(/≈\s*[+\-−]?\d{1,3}/g) ?? [];
1228
+ if (spreadMatches.length >= 2) {
1229
+ warns.push("comment contains a spread list (≈ + counts) — the DB viewer already shows sibling counts and fashion scores next to every move, so this is doubled noise. Name the character of the choice instead (\"solid vs sharp\", \"old vs fashionable\") or drop the numbers.");
1230
+ }
1231
+ // Raw centipawn in prose: "+0.35", "-0.20", "+80" (not preceded by move
1232
+ // number). Also "at depth N" or "N nodes" — engine metadata as prose.
1233
+ if (/(?:^|[^\d.])[+-]\d\.\d\d(?!\d)/.test(comment) || /≈\s*[+\-−]?\d{2,3}\b/.test(comment) ||
1234
+ /\bat depth \d+\b/i.test(comment) || /\b\d{2,3}M nodes\b/.test(comment)) {
1235
+ warns.push("comment contains raw centipawn values or engine metadata — the app renders ceoEval + NAG glyph next to every node, so these numbers are doubled noise AND opaque (readers can't tell if ≈-60 means eval, spread, or something else). Set the NAG (set_nags) and let the glyph carry the judgment; drop the number from the prose.");
1236
+ }
1237
+ // Roster: "N GM games" pattern
1238
+ if (/\b\d{2,4}\s+GM games\b/i.test(comment)) {
1239
+ warns.push("comment restates game count — the app shows the count on hover. Either cite a specific game with signal (\"Caruana-Liang, Superbet 2026\") or drop the number.");
1240
+ }
1241
+ return warns;
1242
+ }
1243
+ // Return a warning string when an add_line is suspiciously long-and-linear
1244
+ // (the anti-pattern: LLM pastes a 15-ply engine PV into a single add_line
1245
+ // call as if it were prepared repertoire). Two thresholds so the message
1246
+ // escalates — a 10-ply Berlin mainline is fine, a 20-ply LLM extrapolation
1247
+ // almost never is. Threshold applies at the CALL level, not against
1248
+ // existing tree depth — the anti-pattern is a single tool call adding
1249
+ // many plies at once with no user thought about where the branching should
1250
+ // live.
1251
+ function longLineWarning(sansLength) {
1252
+ if (sansLength >= 14) {
1253
+ return `you added ${sansLength} plies in one call without branching — this is the shape of a pasted engine PV, not a repertoire. Real prep branches at every ply where the opponent has meaningful alternatives. Either (a) delete the tail and rebuild with add_move at each decision point, calling cloud_analyse + get_position_stats to see what actually gets played, or (b) if this really is one forcing sequence (mate combination, tactical winner), add a comment naming what makes it forced. Long unbranched lines with no comment default to "engine PV pasted as prep" in the reader's eyes.`;
1254
+ }
1255
+ if (sansLength >= 9) {
1256
+ return `${sansLength}-ply linear line — check that every ply is a genuine only-move or a documented mainline. If the opponent has real alternatives at any ply (get_position_stats would show 2+ moves with meaningful frequency), that ply should branch instead. Prep is a tree, not a line.`;
1257
+ }
1258
+ return undefined;
1259
+ }
1260
+ // Trim every PV in a converted cloud-analyse response to `maxPlies`
1261
+ // and mark each trimmed line with `pv_truncated: true` so the LLM
1262
+ // sees what happened. Applied ONLY to cloud_analyse (short synchronous
1263
+ // snapshot); deep_analyse is the explicit "give me the deep line"
1264
+ // tool and keeps its full PV.
1265
+ function capPvsInResponse(converted, maxPlies) {
1266
+ if (!converted || typeof converted !== "object")
1267
+ return;
1268
+ const r = converted;
1269
+ for (const eng of [r.stockfish, r.lc0]) {
1270
+ if (!eng || !Array.isArray(eng.lines))
1271
+ continue;
1272
+ for (const line of eng.lines) {
1273
+ if (Array.isArray(line.pv) && line.pv.length > maxPlies) {
1274
+ line.pv = line.pv.slice(0, maxPlies);
1275
+ line.pv_truncated = true;
1276
+ }
1277
+ }
1278
+ }
1279
+ converted.pv_max_plies = maxPlies;
1280
+ }
1073
1281
  function convertCloudSnapshotResponse(raw, startFen) {
1074
1282
  if (!raw || typeof raw !== "object")
1075
1283
  return raw;
@@ -1115,17 +1323,37 @@ function dispatchMutation(file, idIndex, op) {
1115
1323
  };
1116
1324
  const resolve = (id) => resolveNodeId(idIndex, id);
1117
1325
  switch (kind) {
1118
- case "add_move":
1119
- return addMove(file, resolve(nodeIdField("parent_id")), String(op.san));
1326
+ case "add_move": {
1327
+ const parentPath = resolve(nodeIdField("parent_id"));
1328
+ const parent = getNodeByPath(file.root, parentPath);
1329
+ const noStatsWarn = noStatsCheckWarning(parent);
1330
+ const step = addMove(file, parentPath, String(op.san));
1331
+ return { ...step, ...(noStatsWarn ? { warning: noStatsWarn } : {}) };
1332
+ }
1120
1333
  case "add_line": {
1121
1334
  const sans = Array.isArray(op.sans) ? op.sans.map(String) : [];
1122
1335
  const parentPath = resolve(nodeIdField("parent_id"));
1336
+ const parent = getNodeByPath(file.root, parentPath);
1123
1337
  const step = addLine(file, parentPath, sans);
1124
1338
  const lastId = step.line.length > 0 ? step.line[step.line.length - 1].id : nodeIdField("parent_id");
1125
- return { file: step.file, id: lastId, results: step.line };
1339
+ // Same anti-pattern warnings as the standalone add_line case —
1340
+ // long unbranched line + no-stats-check parent are both bugs
1341
+ // whether they land solo or inside a batch.
1342
+ const longLineWarn = longLineWarning(sans.length);
1343
+ const noStatsWarn = noStatsCheckWarning(parent);
1344
+ const warnings = [longLineWarn, noStatsWarn].filter((s) => !!s);
1345
+ return { file: step.file, id: lastId, results: step.line, ...(warnings.length > 0 ? { warnings } : {}) };
1346
+ }
1347
+ case "set_comment": {
1348
+ const commentStr = typeof op.comment === "string" ? op.comment : "";
1349
+ const commentWarns = commentAntiPatterns(commentStr);
1350
+ const targetPath = resolve(nodeIdField("node_id"));
1351
+ const targetNode = getNodeByPath(file.root, targetPath);
1352
+ const describeWarn = noDescribeWarning(targetNode, commentStr);
1353
+ const step = setComment(file, targetPath, commentStr);
1354
+ const all = [...commentWarns, ...(describeWarn ? [describeWarn] : [])];
1355
+ return { ...step, ...(all.length > 0 ? { warnings: all } : {}) };
1126
1356
  }
1127
- case "set_comment":
1128
- return setComment(file, resolve(nodeIdField("node_id")), typeof op.comment === "string" ? op.comment : "");
1129
1357
  case "set_nags":
1130
1358
  return setNags(file, resolve(nodeIdField("node_id")), Array.isArray(op.nags) ? op.nags.map(String) : []);
1131
1359
  case "set_annotations": {
@@ -1157,10 +1385,7 @@ async function applyBatchMutations(args) {
1157
1385
  const mutations = Array.isArray(args.mutations) ? args.mutations : [];
1158
1386
  if (mutations.length === 0)
1159
1387
  throw new Error("mutations array required");
1160
- const raw = await authedRequest("GET", `/api/agent/prep-files/${encodeURIComponent(id)}`);
1161
- const g = raw;
1162
- if (typeof g.pgnContent !== "string")
1163
- throw new Error("prep file missing pgnContent");
1388
+ const g = await fetchGame(id);
1164
1389
  let file = parsePGN(g.pgnContent);
1165
1390
  let idIndex = buildIdIndex(file.root);
1166
1391
  const results = [];
@@ -1170,7 +1395,12 @@ async function applyBatchMutations(args) {
1170
1395
  const step = dispatchMutation(file, idIndex, op);
1171
1396
  file = step.file;
1172
1397
  idIndex = buildIdIndex(file.root);
1173
- results.push({ node_id: step.id, ...(step.results !== undefined ? { line: step.results } : {}) });
1398
+ results.push({
1399
+ node_id: step.id,
1400
+ ...(step.results !== undefined ? { line: step.results } : {}),
1401
+ ...(step.warning ? { warning: step.warning } : {}),
1402
+ ...(step.warnings && step.warnings.length > 0 ? { warnings: step.warnings } : {}),
1403
+ });
1174
1404
  }
1175
1405
  catch (err) {
1176
1406
  const msg = err instanceof Error ? err.message : String(err);
@@ -1179,12 +1409,8 @@ async function applyBatchMutations(args) {
1179
1409
  }
1180
1410
  const newPgn = exportPGN(file);
1181
1411
  const expected = typeof args.expected_version === "number" ? args.expected_version : g.version;
1182
- const saved = await authedRequest("PUT", `/api/agent/prep-files/${encodeURIComponent(id)}`, {
1183
- pgn: newPgn,
1184
- expected_version: expected,
1185
- });
1186
- const savedRow = saved;
1187
- return { ok: true, results, version: savedRow.version };
1412
+ const saved = await saveGame(id, newPgn, expected);
1413
+ return { ok: true, results, version: saved.version };
1188
1414
  }
1189
1415
  const evalJobs = new Map();
1190
1416
  // GC finished jobs after this long so status polling remains useful
@@ -1219,10 +1445,7 @@ async function autoEvaluate(args) {
1219
1445
  : ROOT_ID;
1220
1446
  const onlyMissing = args.only_missing !== false; // default true
1221
1447
  const movetimeMs = typeof args.movetime_ms === "number" ? args.movetime_ms : 1500;
1222
- const raw = await authedRequest("GET", `/api/agent/prep-files/${encodeURIComponent(id)}`);
1223
- const g = raw;
1224
- if (typeof g.pgnContent !== "string")
1225
- throw new Error("prep file missing pgnContent");
1448
+ const g = await fetchGame(id);
1226
1449
  const file = parsePGN(g.pgnContent);
1227
1450
  const idIndex = buildIdIndex(file.root);
1228
1451
  const startPath = resolveNodeId(idIndex, startNodeId);
@@ -1835,10 +2058,7 @@ function storedEvalToCompact(ev, analysis) {
1835
2058
  // resolve node_ids without rebuilding the index itself.
1836
2059
  async function applyMutation(args, mutator) {
1837
2060
  const id = String(args.id);
1838
- const raw = await authedRequest("GET", `/api/agent/prep-files/${encodeURIComponent(id)}`);
1839
- const g = raw;
1840
- if (typeof g.pgnContent !== "string")
1841
- throw new Error("prep file missing pgnContent");
2061
+ const g = await fetchGame(id);
1842
2062
  const file = parsePGN(g.pgnContent);
1843
2063
  const idIndex = buildIdIndex(file.root);
1844
2064
  let result;
@@ -1853,24 +2073,60 @@ async function applyMutation(args, mutator) {
1853
2073
  }
1854
2074
  const newPgn = exportPGN(result.file);
1855
2075
  const expected = typeof args.expected_version === "number" ? args.expected_version : g.version;
1856
- const saved = await authedRequest("PUT", `/api/agent/prep-files/${encodeURIComponent(id)}`, {
1857
- pgn: newPgn,
1858
- expected_version: expected,
1859
- });
1860
- const savedRow = saved;
2076
+ const saved = await saveGame(id, newPgn, expected);
1861
2077
  return {
1862
2078
  ok: true,
1863
2079
  node_id: result.id,
1864
2080
  ...(result.results !== undefined ? { line: result.results } : {}),
1865
- version: savedRow.version,
2081
+ ...(result.warning ? { warning: result.warning } : {}),
2082
+ ...(result.warnings && result.warnings.length > 0 ? { warnings: result.warnings } : {}),
2083
+ version: saved.version,
1866
2084
  };
1867
2085
  }
2086
+ // Compute the "you never called describe_position on this node" warning.
2087
+ // Fires from set_comment when the comment is substantive (>= 40 chars —
2088
+ // anything shorter is a label / pointer, doesn't need structural
2089
+ // grounding). LLMs are unreliable at reading FEN strings and confidently
2090
+ // describe positions that don't match the actual board; describe_position
2091
+ // is a pure-computation grounding pass that reliably fixes this. Warn
2092
+ // once per node.
2093
+ function noDescribeWarning(node, comment) {
2094
+ if (node.id === ROOT_ID)
2095
+ return undefined;
2096
+ if (comment.length < 40)
2097
+ return undefined;
2098
+ const key = positionKey(node.fen);
2099
+ if (positionsDescribed.has(key))
2100
+ return undefined;
2101
+ if (noDescribeWarned.has(node.id))
2102
+ return undefined;
2103
+ noDescribeWarned.add(node.id);
2104
+ return `substantive comment (${comment.length} chars) on a node whose position was never grounded via describe_position this session (id=${node.id}, ${node.san}). LLMs invent captures, miscount pieces, and swap files/ranks when reading FEN strings — describe_position is a pure-computation pass (~1 ms, no engine cost, structural facts + Stockfish's per-term eval breakdown) that reliably prevents this class of hallucination. In live audits, prose accuracy jumps sharply on nodes where describe_position was called first. Call describe_position with file_id+node_id=${node.id} BEFORE writing prose. Warned once per node.`;
2105
+ }
2106
+ // Compute the "you never DB-checked this parent" warning. Called from
2107
+ // add_move / add_line handlers with the parent node. Returns undefined
2108
+ // when either (a) the parent was checked this session (or is root — the
2109
+ // starting position doesn't need a DB check), (b) we already warned on
2110
+ // this parent (dedup so building a big branching subtree isn't spammy),
2111
+ // or (c) the mutator is running against a parent whose position has a
2112
+ // stored ceoEval (implies the LLM has done SOME analytical work here).
2113
+ function noStatsCheckWarning(parent) {
2114
+ if (parent.id === ROOT_ID)
2115
+ return undefined;
2116
+ const key = positionKey(parent.fen);
2117
+ if (positionsStatsChecked.has(key))
2118
+ return undefined;
2119
+ if (noStatsWarned.has(parent.id))
2120
+ return undefined;
2121
+ noStatsWarned.add(parent.id);
2122
+ return `no get_position_stats call for the parent (id=${parent.id}, ${parent.san}) this session. Course chapter titles describe what an author chose to cover, not what practical opponents play — treating "the So chapter says 6.O-O-O" as "the mainline is 6.O-O-O" is the exact pattern this warning exists to catch. Call get_position_stats at this position (via file_id+node_id=${parent.id}) BEFORE deciding which branches belong here; suppress this warning by making that call. Warned once per parent per session.`;
2123
+ }
1868
2124
  async function loadPrepFile(id) {
1869
- const raw = await authedRequest("GET", `/api/agent/prep-files/${encodeURIComponent(id)}`);
1870
- const g = raw;
1871
- if (typeof g.pgnContent !== "string")
1872
- throw new Error("prep file missing pgnContent");
1873
- return { file: parsePGN(g.pgnContent), version: g.version, fileIdEcho: g.id, pgn: g.pgnContent };
2125
+ const g = await fetchGame(id);
2126
+ // Echo the composite id back so read_prep_file responses match the
2127
+ // exact id the LLM passed in. The backend returns the raw game_id;
2128
+ // recompose so the LLM never sees the split form.
2129
+ return { file: parsePGN(g.pgnContent), version: g.version, fileIdEcho: id, pgn: g.pgnContent };
1874
2130
  }
1875
2131
  // Recursively project a PrepNode into the requested view. `depthLeft`
1876
2132
  // null → unlimited; 0 → just the node without children.
@@ -1922,6 +2178,91 @@ function projectNode(node, view, depthLeft, fenIndex = null) {
1922
2178
  }
1923
2179
  return base;
1924
2180
  }
2181
+ async function listCollections(_args) {
2182
+ const raw = await authedRequest("GET", PGN_BASE);
2183
+ const collections = unwrap(raw) ?? [];
2184
+ return {
2185
+ collections: collections.map(c => ({
2186
+ id: c.id,
2187
+ title: c.title,
2188
+ icon: c.icon,
2189
+ folder_path: c.folderPath,
2190
+ game_count: c.gameCount,
2191
+ position_search_enabled: c.positionSearchEnabled,
2192
+ updated_at: c.updatedAt,
2193
+ })),
2194
+ };
2195
+ }
2196
+ // Convert a browser-returned game list row into the LLM shape (composite
2197
+ // id, cleaned field names).
2198
+ function projectGameRow(row) {
2199
+ const collId = row.collectionId ?? "";
2200
+ return {
2201
+ id: collId ? makeFileId(collId, row.id) : row.id,
2202
+ collection_id: collId,
2203
+ collection_title: row.collectionTitle,
2204
+ event: row.event,
2205
+ white: row.white_player,
2206
+ black: row.black_player,
2207
+ eco: row.eco,
2208
+ opening: row.opening,
2209
+ updated_at: row.updated_at,
2210
+ ply: row.ply,
2211
+ };
2212
+ }
2213
+ async function listPrepFiles(args) {
2214
+ const collectionId = typeof args.collection_id === "string" ? args.collection_id.trim() : "";
2215
+ if (!collectionId) {
2216
+ throw new Error("collection_id required — call list_collections to see your options, or search across collections with search_prep_files / find_position_in_files");
2217
+ }
2218
+ // Browser handler at GET /me/pgns/{id}/games returns a paginated list.
2219
+ const raw = await authedRequest("GET", `${PGN_BASE}/${encodeURIComponent(collectionId)}/games?page=1&limit=200`);
2220
+ const data = unwrap(raw);
2221
+ const games = data?.games ?? (Array.isArray(data) ? data : []);
2222
+ return { collection_id: collectionId, prep_files: games.map(projectGameRow) };
2223
+ }
2224
+ async function searchPrepFiles(args) {
2225
+ const q = typeof args.query === "string" ? args.query.trim() : "";
2226
+ if (!q)
2227
+ throw new Error("query required");
2228
+ const raw = await authedRequest("GET", `${PGN_BASE}/games/search?q=${encodeURIComponent(q)}&limit=100`);
2229
+ const data = unwrap(raw);
2230
+ const games = data?.games ?? [];
2231
+ return { query: q, prep_files: games.map(projectGameRow) };
2232
+ }
2233
+ async function findPositionInFiles(args) {
2234
+ // FEN can come from a node handle OR a direct fen/moves/line. Reuse
2235
+ // the same resolver everything else uses.
2236
+ const resolved = await resolveFromNodeOrFen(args);
2237
+ const fen = resolved.fen;
2238
+ const raw = await authedRequest("GET", `${PGN_BASE}/games/search?position=${encodeURIComponent(fen)}&limit=100`);
2239
+ const data = unwrap(raw);
2240
+ const games = data?.games ?? [];
2241
+ return {
2242
+ fen,
2243
+ match_count: games.length,
2244
+ prep_files: games.map(projectGameRow),
2245
+ };
2246
+ }
2247
+ async function createPrepFile(args) {
2248
+ const collectionId = typeof args.collection_id === "string" ? args.collection_id.trim() : "";
2249
+ if (!collectionId) {
2250
+ throw new Error("collection_id required — call list_collections to pick where the new file lives. There is no default landing folder any more (v0.43: the old hidden /mcp collection was removed).");
2251
+ }
2252
+ const name = String(args.name || "").trim();
2253
+ if (!name)
2254
+ throw new Error("name is required");
2255
+ // Seed with a PGN carrying the LLM-chosen name as the Event tag so
2256
+ // subsequent list_prep_files calls display something useful.
2257
+ const seedPgn = `[Event "${name.replace(/\\/g, "\\\\").replace(/"/g, '\\"')}"]\n\n*\n`;
2258
+ const game = await createGame(collectionId, seedPgn);
2259
+ return {
2260
+ ok: true,
2261
+ id: makeFileId(game.collectionId ?? collectionId, game.id),
2262
+ collection_id: game.collectionId ?? collectionId,
2263
+ version: game.version,
2264
+ };
2265
+ }
1925
2266
  async function readPrepFile(args) {
1926
2267
  const id = String(args.id);
1927
2268
  const view = (typeof args.view === "string" && ["compact", "full", "spine", "pgn"].includes(args.view))
@@ -2298,10 +2639,7 @@ async function resolveFromNodeOrFen(args) {
2298
2639
  const fileId = typeof args.file_id === "string" ? args.file_id.trim() : "";
2299
2640
  const nodeId = typeof args.node_id === "string" ? args.node_id.trim() : "";
2300
2641
  if (fileId && nodeId) {
2301
- const raw = await authedRequest("GET", `/api/agent/prep-files/${encodeURIComponent(fileId)}`);
2302
- const g = raw;
2303
- if (typeof g.pgnContent !== "string")
2304
- throw new Error("prep file missing pgnContent");
2642
+ const g = await fetchGame(fileId);
2305
2643
  const parsedFile = parsePGN(g.pgnContent);
2306
2644
  const idIndex = buildIdIndex(parsedFile.root);
2307
2645
  const nodePath = resolveNodeId(idIndex, nodeId);
@@ -2317,18 +2655,15 @@ async function resolveFromNodeOrFen(args) {
2317
2655
  // node_id had been supplied (auto-persist on match).
2318
2656
  const fen = resolveFenFromArgs(args);
2319
2657
  try {
2320
- const raw = await authedRequest("GET", `/api/agent/prep-files/${encodeURIComponent(fileId)}`);
2321
- const g = raw;
2322
- if (typeof g.pgnContent === "string") {
2323
- const parsedFile = parsePGN(g.pgnContent);
2324
- const match = findNodeByFen(parsedFile.root, fen);
2325
- if (match) {
2326
- const idIndex = buildIdIndex(parsedFile.root);
2327
- return {
2328
- fen,
2329
- file: { id: fileId, version: g.version ?? 0, parsedFile, idIndex, nodePath: match.path, fen },
2330
- };
2331
- }
2658
+ const g = await fetchGame(fileId);
2659
+ const parsedFile = parsePGN(g.pgnContent);
2660
+ const match = findNodeByFen(parsedFile.root, fen);
2661
+ if (match) {
2662
+ const idIndex = buildIdIndex(parsedFile.root);
2663
+ return {
2664
+ fen,
2665
+ file: { id: fileId, version: g.version ?? 0, parsedFile, idIndex, nodePath: match.path, fen },
2666
+ };
2332
2667
  }
2333
2668
  }
2334
2669
  catch {
@@ -2389,10 +2724,7 @@ async function storeEvalOnNode(handle, ev) {
2389
2724
  const paths = group.map(n => resolveNodeId(idIndex, n.id));
2390
2725
  const { file: newFile, ids } = setCeoEvalMany(handle.parsedFile, paths, ev);
2391
2726
  const newPgn = exportPGN(newFile);
2392
- await authedRequest("PUT", `/api/agent/prep-files/${encodeURIComponent(handle.id)}`, {
2393
- pgn: newPgn,
2394
- expected_version: handle.version,
2395
- });
2727
+ await saveGame(handle.id, newPgn, handle.version);
2396
2728
  // Ensure the primary node (the one the LLM addressed) comes first.
2397
2729
  const anchorId = anchor.id;
2398
2730
  return [anchorId, ...ids.filter(x => x !== anchorId)];
@@ -2529,6 +2861,10 @@ async function callToolInner(name, args) {
2529
2861
  if (ev)
2530
2862
  converted.eval = ev;
2531
2863
  }
2864
+ // Record that this position was DB-checked this session. Downstream
2865
+ // add_move / add_line under this parent won't fire the "no stats
2866
+ // check" warning. Keyed by 3-field FEN so transpositions count.
2867
+ positionsStatsChecked.add(positionKey(fen));
2532
2868
  return converted;
2533
2869
  }
2534
2870
  case "describe_position": {
@@ -2552,6 +2888,10 @@ async function callToolInner(name, args) {
2552
2888
  if (ev && ev.found === true) {
2553
2889
  merged.engineEvalTerms = { terms: ev.terms, total: ev.total };
2554
2890
  }
2891
+ // Record so set_comment on this node won't fire the "not described"
2892
+ // warning. Keyed by 3-field FEN so a described position is
2893
+ // credited across its transpositions too.
2894
+ positionsDescribed.add(positionKey(resolved.fen));
2555
2895
  return merged;
2556
2896
  }
2557
2897
  case "predict_human_move": {
@@ -2627,6 +2967,21 @@ async function callToolInner(name, args) {
2627
2967
  body.engines = args.engines;
2628
2968
  const raw = await authedRequest("POST", "/api/agent/cloud-engines/analyse", body);
2629
2969
  const converted = convertCloudSnapshotResponse(raw, fen);
2970
+ // PV cap: engine PVs beyond ~6 plies are speculative (the tail is
2971
+ // where the search's confidence collapses — SF at depth 24 has
2972
+ // seen the first few plies solidly and hedged everything after).
2973
+ // More importantly, LLMs paste long PVs into `add_line` as if
2974
+ // they were prepared repertoire. A 15-move PV pasted as a
2975
+ // variation is one line of engine output through positions
2976
+ // where both sides had real choices — not a repertoire. Cap the
2977
+ // affordance: return only what's load-bearing (3 full moves for
2978
+ // understanding the point), let the caller re-analyse the
2979
+ // resulting position if they want to see further. Override via
2980
+ // `pv_max_plies` for the rare case (deep tactics verification).
2981
+ const pvMaxPlies = typeof args.pv_max_plies === "number" && args.pv_max_plies > 0
2982
+ ? Math.min(args.pv_max_plies, 40)
2983
+ : 6;
2984
+ capPvsInResponse(converted, pvMaxPlies);
2630
2985
  // Node-addressed calls: persist the result on the node's ceoEval
2631
2986
  // so a later quote_engine_eval can cite this measurement. This is
2632
2987
  // the anti-hallucination hinge — prose that says "engines say X
@@ -2657,10 +3012,14 @@ async function callToolInner(name, args) {
2657
3012
  return findPositionInCourses(args);
2658
3013
  case "read_course_at_position":
2659
3014
  return readCourseAtPosition(args);
3015
+ case "list_collections":
3016
+ return listCollections(args);
2660
3017
  case "list_prep_files":
2661
- return authedRequest("GET", "/api/agent/prep-files");
3018
+ return listPrepFiles(args);
2662
3019
  case "search_prep_files":
2663
- return authedRequest("GET", `/api/agent/prep-files/search?q=${encodeURIComponent(String(args.query))}`);
3020
+ return searchPrepFiles(args);
3021
+ case "find_position_in_files":
3022
+ return findPositionInFiles(args);
2664
3023
  case "read_prep_file":
2665
3024
  return readPrepFile(args);
2666
3025
  case "list_nodes":
@@ -2668,23 +3027,52 @@ async function callToolInner(name, args) {
2668
3027
  case "list_transpositions":
2669
3028
  return listTranspositions(args);
2670
3029
  case "create_prep_file":
2671
- return authedRequest("POST", "/api/agent/prep-files", {
2672
- name: String(args.name),
2673
- });
2674
- case "delete_prep_file":
2675
- return authedRequest("DELETE", `/api/agent/prep-files/${encodeURIComponent(String(args.id))}`);
3030
+ return createPrepFile(args);
3031
+ case "delete_prep_file": {
3032
+ await deleteGame(String(args.id));
3033
+ return { ok: true };
3034
+ }
2676
3035
  case "add_move":
2677
- return applyMutation(args, (file, idIndex) => addMove(file, resolveNodeId(idIndex, argNodeId(args, "parent_id")), String(args.san)));
2678
- case "add_line":
2679
3036
  return applyMutation(args, (file, idIndex) => {
2680
- const sans = Array.isArray(args.sans) ? args.sans.map(String) : [];
2681
3037
  const parentPath = resolveNodeId(idIndex, argNodeId(args, "parent_id"));
2682
- const step = addLine(file, parentPath, sans);
3038
+ const parent = getNodeByPath(file.root, parentPath);
3039
+ const noStatsWarn = noStatsCheckWarning(parent);
3040
+ const step = addMove(file, parentPath, String(args.san));
3041
+ return { ...step, ...(noStatsWarn ? { warning: noStatsWarn } : {}) };
3042
+ });
3043
+ case "add_line": {
3044
+ const sansArg = Array.isArray(args.sans) ? args.sans.map(String) : [];
3045
+ const longLineWarn = longLineWarning(sansArg.length);
3046
+ return applyMutation(args, (file, idIndex) => {
3047
+ const parentPath = resolveNodeId(idIndex, argNodeId(args, "parent_id"));
3048
+ const parent = getNodeByPath(file.root, parentPath);
3049
+ const noStatsWarn = noStatsCheckWarning(parent);
3050
+ const step = addLine(file, parentPath, sansArg);
2683
3051
  const lastId = step.line.length > 0 ? step.line[step.line.length - 1].id : argNodeId(args, "parent_id");
2684
- return { file: step.file, id: lastId, results: step.line };
3052
+ const combined = [longLineWarn, noStatsWarn].filter((s) => !!s);
3053
+ return {
3054
+ file: step.file,
3055
+ id: lastId,
3056
+ results: step.line,
3057
+ ...(combined.length > 0 ? { warnings: combined } : {}),
3058
+ };
3059
+ });
3060
+ }
3061
+ case "set_comment": {
3062
+ const commentStr = typeof args.comment === "string" ? args.comment : "";
3063
+ const commentWarns = commentAntiPatterns(commentStr);
3064
+ return applyMutation(args, (file, idIndex) => {
3065
+ const targetPath = resolveNodeId(idIndex, argNodeId(args));
3066
+ const targetNode = getNodeByPath(file.root, targetPath);
3067
+ const describeWarn = noDescribeWarning(targetNode, commentStr);
3068
+ const step = setComment(file, targetPath, commentStr);
3069
+ const all = [...commentWarns, ...(describeWarn ? [describeWarn] : [])];
3070
+ return {
3071
+ ...step,
3072
+ ...(all.length > 0 ? { warnings: all } : {}),
3073
+ };
2685
3074
  });
2686
- case "set_comment":
2687
- return applyMutation(args, (file, idIndex) => setComment(file, resolveNodeId(idIndex, argNodeId(args)), typeof args.comment === "string" ? args.comment : ""));
3075
+ }
2688
3076
  case "set_nags":
2689
3077
  return applyMutation(args, (file, idIndex) => setNags(file, resolveNodeId(idIndex, argNodeId(args)), Array.isArray(args.nags) ? args.nags.map(String) : []));
2690
3078
  case "set_annotations": {
@@ -2715,10 +3103,7 @@ async function callToolInner(name, args) {
2715
3103
  case "quote_engine_eval": {
2716
3104
  const fileId = String(args.id);
2717
3105
  const nodeId = argNodeId(args);
2718
- const raw = await authedRequest("GET", `/api/agent/prep-files/${encodeURIComponent(fileId)}`);
2719
- const g = raw;
2720
- if (typeof g.pgnContent !== "string")
2721
- throw new Error("prep file missing pgnContent");
3106
+ const g = await fetchGame(fileId);
2722
3107
  const file = parsePGN(g.pgnContent);
2723
3108
  const idIndex = buildIdIndex(file.root);
2724
3109
  const path = resolveNodeId(idIndex, nodeId);
@@ -191,8 +191,19 @@ Two calls:
191
191
 
192
192
  If the user is specifically preparing to *play* the black side in a must-win, add `contempt=-30` (or up to `-60` for a harder steer) on a follow-up call to see which lines Lc0 finds most fighting for Black. Compare against Stockfish's objective read to make sure the fighting choice isn't just losing.
193
193
 
194
+ ## The PV is not a line to paste
195
+
196
+ `cloud_analyse` caps each PV at 6 plies (3 full moves) by default and marks longer ones `pv_truncated: true`. This is deliberate — the tail of a PV is where the engine's confidence collapses (SF at depth 24 has resolved the first few plies solidly and hedged everything after), and pasting long PVs into `add_line` as prep is the biggest documented anti-pattern of this whole system.
197
+
198
+ **A PV tells you what the engine sees, not what will be played.** A 15-ply PV pasted as a variation is one line of engine output through positions where the opponent had 2-3 real alternatives at almost every ply. That's not prep — that's the engine's preferred game, and no opponent plays the engine's preferred game.
199
+
200
+ **To see further into a line, walk the tree.** Take the position at the tail of your truncated PV, run a fresh `cloud_analyse` on it. That call gives you the multipv candidate set at THAT position — the opponent's actual options — which is what you need to decide whether to branch. Don't raise `pv_max_plies` unless you're verifying a forcing sequence (a mate, an obligated recapture chain).
201
+
202
+ Practical: raise `lc0_multipv` (default 8) to see the candidate spread on the current position, NOT to see further down one PV. If Lc0 shows moves 1-3 within 0.15 of each other, that's a branching point — three responses need coverage, not one PV.
203
+
194
204
  ## What NOT to do
195
205
 
206
+ - **Don't paste PVs as `add_line` variations.** See the section above and `pgn-authoring.md`'s "Prep is a TREE, not a line" section. This is not a stylistic preference — the tool warns you starting at 9 plies and warns hard at 14+, and the truncated `pv_truncated: true` marker is telling you the engine itself doesn't stand behind the tail.
196
207
  - **Don't quote Lc0's contempt-biased eval as objective.** If you tell the user "Lc0 gives Black +0.30 here" without disclosing you set contempt=-30, that's misleading.
197
208
  - **Don't run cloud analysis just for casual questions.** `cloud_analyse` costs the user real money per second. If the question is "is 1.e4 or 1.d4 better?", the free `analyse` (single Stockfish, 2s) or `get_position_stats` (11.7M-game database) is enough.
198
209
  - **Don't ignore the disagreement.** When Stockfish and Lc0 diverge sharply, that's exactly when you should explain *why* to the user — not paper over it.
@@ -155,15 +155,80 @@ Prose is NEVER for:
155
155
  - Move recommendations ("here White should play h4") — add_move it.
156
156
  - Restating the eval a NAG already conveys.
157
157
  - **Describing a sibling variation you already added as a branch.** If move A has variation B added as `add_move(A, sibling)`, don't ALSO write `{if B then ...}` on A. The reader clicks B on the board — the branch is already there.
158
- - **Restating what the app already renders.** The reader opens the file in the app and sees: every sibling move (with count / avg rating / top players from the DB), each node's stored `ceoEval`, the NAG glyphs, the tree structure. Duplicating any of that in prose is pure noise.
158
+ - **Restating what the app already renders.** The reader opens the file in the app and sees: every sibling move (with count / avg rating / top players from the DB), each node's stored `ceoEval`, the NAG glyphs, the tree structure. Duplicating any of that in prose is pure noise. Three specific patterns to never emit:
159
159
 
160
- - ❌ `{5...Bb4 (180/247), 5...d6 (26), 5...Be7 (18), 5...d5 (7), 5...g6 (2)}` — those are the sibling variations, already visible.
161
- - ❌ `{Main move, 180 of 247 games (So, Giri, Karjakin, Duda). Eval −0.20/−0.27.}` — count, top players, eval all shown by the app.
162
- - ✅ `{Main move; the sharpest test is actually 5...d5 (only 7 games but avg 2526) — see the variation.}`
163
- - ✅ `{Popular, but leads to the endgame Black draws — 6.Bd2 is where prep depth matters.}`
160
+ **1. Spread lists.** The database viewer already shows sibling counts and fashion scores next to every move. Pasting them into prose is doubled noise.
161
+
162
+ - ❌ `{Spread: 5.O-O ≈50, 6.h3 ≈42, 6.a4 ≈35, 6.Be3 ≈27, 6.Nbd2 ≈18 — the more White delays castling the less he gets.}` — every number here is one click away.
163
+ - ❌ `{Spread: 5.Nf3 ≈56, 5.Bd3 ≈55, 5.Be3 ≈47, 5.Rb1 ≈35, 5.h3 ≈32.}` — same problem.
164
+ - ✅ `{Two roads: quiet 5.Bd3 keeps position closed, sharp 5.e5 concedes the centre for tempo — cover both.}` — names the *character*, not the numbers.
165
+
166
+ **2. Raw centipawn values in prose.** The app shows `ceoEval` as `+0.20 / −0.27` next to every node and the NAG glyph next to that. If you write `≈−60` or `+0.35` in a comment, no reader knows whether that's a spread number, an eval, or a made-up decoration — and every one of them is visible without your prose.
167
+
168
+ - ❌ `{Best try, but ≈80 — a fifth of a pawn worse than the ...Nd7 move order.}` — cp number, opaque.
169
+ - ❌ `{0.00 at depth 31, 259M nodes.}` — engine metadata masquerading as insight; the eval is already visible.
170
+ - ❌ `{≈−60. White's f5 sacrifice does not work with the centre already liquidated.}` — the number tells the reader nothing; the *reason* is the whole comment.
171
+ - ✅ `{White's f5 sacrifice doesn't work with the centre already liquidated — no target for the pawn.}` — same content, no fake precision.
172
+ - ✅ Or better: set the NAG (`$17` for clear Black advantage) and drop the prose entirely; the glyph carries the judgment.
173
+
174
+ **3. Top-player rosters.** Count + names of the top players are shown on hover in the app. `146 GM games, Nakamura, Kramnik, MVL, So all play this` restates two visible facts.
175
+
176
+ - ❌ `{150 GM games and Caruana's choice against Liang in 2026.}` — count is visible; the specific-game citation is fine on its own if it carries prep signal.
177
+ - ✅ `{Caruana played this against Liang, Superbet 2026 — the current top-choice among elite Black players.}` — same specific-game citation, no restated count.
178
+
179
+ **Positive vocabulary for what the app can't render.** The reader wants labels the DB doesn't provide:
180
+
181
+ - `{The old main line — dominant through 2015, replaced by 6.Bd2 after Ding-Carlsen 2016.}` — historical context, not visible.
182
+ - `{The current fashion — 40+ games since 2024, mostly at 2700+.}` — recency signal condensed to a phrase, not a number list.
183
+ - `{Solid try (holds objectively) vs the sharper 6.Nd5 (small edge but requires memory) — choose based on style.}` — practical framing, uses labels the reader can act on.
184
+ - `{Prophylactic — every White plan is based on Bg5, this pre-empts it.}` — one word (prophylactic / restraint / clamp / breakthrough) does the work of a paragraph.
164
185
 
165
186
  Test: if the reader can see it by looking at the position or clicking a branch, don't write it. Prose is only for plans, prep-signal, or WHY — the layer the app can't derive.
166
187
 
188
+ - **Prose "prevents Y" claims — always show Y as a `?`-tagged variation instead.** When you write `{6.f3 is necessary to prevent ...Bxh3.}` the reader has to trust you that the tactic exists. When you instead add a sibling variation `6.O-O? Bxh3 7.gxh3 …` marked with `?` in the NAG, the reader can play through the refutation themselves. Two benefits: (1) grounds the claim in an actual move sequence, so hallucinated tactics get exposed at authoring time when SAN validation runs; (2) the reader learns the tactic instead of taking your word for it. Rule: any prose of the shape "X because it prevents/avoids/deals with Y" should be either supplemented by or replaced with a `?`-marked variation showing Y.
189
+
190
+ - ❌ `{6.f3 is necessary — 6.O-O? runs into ...Bxh3 winning the exchange.}` — no way to check.
191
+ - ✅ Under 5.Nc3, add both `6.f3` (mainline) AND `6.O-O` as a sibling variation with `set_nags(["$2"])` and continuation `[Bxh3, gxh3, ...]` showing the refutation. The prose on 6.f3 shrinks to `{The 6.O-O branch shows why f3 must come first.}` — six words, refutation is playable.
192
+
193
+ ## Prep is a TREE, not a line — the pasted-engine-PV anti-pattern
194
+
195
+ This is the single biggest quality problem in current LLM output on this system: after a `cloud_analyse` call the LLM sees a PV like `[Nf3, Nc6, Bb5, a6, Ba4, Nf6, O-O, Be7, Re1, b5, Bb3, O-O, ...]` and pastes it into a single `add_line`. 15 moves in one call, no branching, no reason to think the opponent will play any of those specific moves — one line of engine output through positions where both sides had real choices.
196
+
197
+ **The reader can tell.** A pasted engine PV always looks the same: long, straight, no comments, no NAGs, ends in a position with no obvious relevance. It's the shape of the output, not the individual moves, that gives it away.
198
+
199
+ **The fix is branching, not length.** Every ply along a variation is a decision point for whoever's turn it is. If the opponent has more than one plausible move at that ply (`get_position_stats` shows two or more moves with meaningful frequency; `predict_human_move` shows two or more moves with real WDL differences; `cloud_analyse.lc0.lines` shows several evals within 0.2 of each other), then that ply MUST branch — you're not preparing if you only cover one response.
200
+
201
+ **Concrete rules:**
202
+
203
+ - **`cloud_analyse` PVs are capped at 6 plies by default and marked `pv_truncated: true` when longer.** This is not a bug — it's telling you that the engine's confidence tail is not repertoire material. To see further into a line, don't raise `pv_max_plies` — pick the position at the end of the truncated PV and run a fresh `cloud_analyse` on it. That's what makes it prep instead of pasted output.
204
+
205
+ - **`add_line` warns when you pass ≥9 plies** and warns hard at ≥14 plies. Long unbranched lines with no comment default to "engine PV pasted as prep" in the reader's eyes. Rule of thumb: if you added ≥8 plies in one call, at least half of them should have branched.
206
+
207
+ - **Genuine forcing sequences ARE allowed** — a 12-ply mate combination, an obligated exchange sequence where both sides have exactly one reasonable move at every ply. In those cases: (a) write a comment naming what makes it forced (`{Every move here is forced by the mate threat on h7.}`), and (b) don't stop where the engine PV stops — stop where the position becomes evaluatable ("winning endgame, technique wins"; "mate in 3, easy calculation from here").
208
+
209
+ **Correct pattern for building a variation.** At every ply:
210
+
211
+ 1. What are the plausible replies? `get_position_stats` (frequencies), `cloud_analyse` with `lc0_multipv: 8` (candidate spread), `predict_human_move` (what people actually play).
212
+ 2. If ≥2 are plausible → branch. `add_move` each, then recurse on each branch or `add_line` for each of the several straightforward continuations.
213
+ 3. If exactly 1 → continue linearly; note *why* it's the only move in a comment.
214
+
215
+ The failure mode you're avoiding: writing prep that reads as if the opponent will helpfully play the engine's #1 preference at every ply. They won't; that's the whole point of prep.
216
+
217
+ ## Course chapters describe COVERAGE, not consensus — always `get_position_stats` at mainline branch points
218
+
219
+ Concrete failure this rule was written to fix: a Modern Defence file made 6.O-O-O the mainline of the entire "5.Qd2 Nd7" branch, wrote a chapter around it, and analysed 15+ plies deep. `get_position_stats` at that position was NEVER called this session. Instead, the LLM ran `find_position_in_courses`, saw So / Kraai / Mihajlov all had chapters titled "5.Qd2 b5 6.O-O-O Bb7" — and cargo-culted that into "6.O-O-O is the main line". It isn't. 6.O-O-O is one of five White tries, and it wasn't even the most-played.
220
+
221
+ **Course chapter titles tell you what the author decided to cover, not what practical opponents play.** A White repertoire author picks one continuation per branch and writes it up in depth — the chapter is named after what they cover. That is not the same as "this is the mainline in current practice", nor is it the same as "this is the most-played move against 5...Nd7". Five different courses can all pick different 6th moves for the same position; five different courses can all pick the SAME 6th move for coverage reasons (that's the sharpest / most-testing / easiest-to-teach) even though the DB shows practice is split five ways.
222
+
223
+ **Rule:** before committing to a mainline branch — either `add_move` as the first child of a parent that's on the mainline, or `add_line` for a linear continuation you're calling THE way White/Black plays — you must have called `get_position_stats` at that parent position in this session. If you didn't, the mutation will warn (soft). Course reads are for what the authors *say about the move*, not what the mainline *is*.
224
+
225
+ **How the two tools relate at a branching decision:**
226
+
227
+ 1. `get_position_stats(position)` — what moves are actually played, ranked by count / avg rating / fashion. This decides which moves need a branch (any move ≥5% frequency in reasonable practice; any move with unusually high avg rating even if less frequent).
228
+ 2. `find_position_in_courses(position)` → `read_course_at_position(...)` — for each branch you decided to include, what do the authors say about it, and where do they disagree. This decides the *content* of each branch (recommended reply, key concepts, historical context), not which branches exist.
229
+
230
+ Getting these backwards is the failure this section exists to prevent.
231
+
167
232
  ## Transpositions: don't analyse (or comment on) the same position twice
168
233
 
169
234
  Move orders diverge and re-converge constantly. `1.d4 Nf6 2.c4 e6 3.Nc3` and `1.c4 e6 2.Nc3 Nf6 3.d4` land on the same position. The tree model doesn't merge those into one node — it stores both nodes with the same position — so if you're not careful you'll analyse both, quote engine numbers on both, and write two different comments for what is the same chess.
@@ -251,7 +316,9 @@ Contempt scale is signed 0-100 (same as the web UI's ContemptStrength slider). T
251
316
 
252
317
  ### Before you write any commentary: describe_position
253
318
 
254
- LLMs are not reliable at reading FEN strings — you'll swap files/ranks, invent captures, miscount pieces. Before you write a comment describing what's happening in a position, call `describe_position(fen)`. Pure computation (~1 ms, no engine), returns two layers:
319
+ **This is the biggest lever for prose quality in the whole system.** Live audit of a recent session — 13 `describe_position` calls versus 50+ `set_comment` ops. The nodes where `describe_position` was called first produced comments that grounded specifically in the position (correct piece squares, real pawn structure, actual weak squares). The nodes where it wasn't produced generic prose that pattern-matched to similar-*looking* positions and confidently named pieces on wrong squares. This gap is why `set_comment` now emits a warning whenever a substantive comment (≥40 chars) lands on a node whose position was never grounded via `describe_position` this session.
320
+
321
+ LLMs are not reliable at reading FEN strings — you'll swap files/ranks, invent captures, miscount pieces. Before you write a comment describing what's happening in a position, call `describe_position` (with `file_id`+`node_id` inside a prep file). Pure computation (~1 ms, no engine), returns two layers:
255
322
 
256
323
  **Board state** — piece placements, material, contested pieces (attackers + defenders), hanging list, check state, castling, en passant, legal moves. Fixes the *"Black's queen on c7 is defended by the knight on d5"* failure when actually there's no knight on d5 and the queen is on c8.
257
324
 
@@ -4,16 +4,23 @@ You can save chess prep to the user's chess.ceo account and read it back across
4
4
 
5
5
  ## The mental model
6
6
 
7
- The user has **one** collection dedicated to your work — labelled "AI Prep" in their chess.ceo app with a 🤖 icon. Inside it, each **prep file** is one PGN game with variations. You never see the collection itself; the tools operate directly on the files inside it.
7
+ The user has **any number of PGN collections** in their chess.ceo library — you have full access to every non-encrypted one. A **prep file** is one PGN game with variations, inside a collection. Every prep file id is a composite `<collection_id>:<game_id>` — opaque to you, pass it through unchanged to any tool that takes `id` or `file_id`.
8
8
 
9
- File-level tools:
9
+ Discovery tools:
10
10
 
11
- - `list_prep_files` — show me all my prep files
12
- - `search_prep_files(query)` — find by opponent name / opening keyword
13
- - `read_prep_file(id)` — parsed tree (every node carries a stable `id`) + tags + `version`
14
- - `create_prep_file(name)` — new empty file, `name` becomes the [Event] tag
11
+ - `list_collections` — show me all the user's collections (their organizational scheme is theirs; browse before creating)
12
+ - `list_prep_files(collection_id)` — the games inside one collection
13
+ - `search_prep_files(query)` — text search across ALL of the user's collections
14
+ - `find_position_in_files(fen)` — position search across ALL of the user's collections (matched by zobrist hash so move-order variants are found automatically). Distinct from `find_position_in_courses` — that's the user's read-only reference library (Chessable etc.); this is their own editable prep
15
+
16
+ Per-file tools:
17
+
18
+ - `read_prep_file(id)` — parsed tree (every node carries a stable content-derived `id`) + tags + `version`
19
+ - `create_prep_file(collection_id, name)` — new empty file inside the given collection, `name` becomes the [Event] tag
15
20
  - `delete_prep_file(id)` — soft delete (user can restore from app)
16
21
 
22
+ **No default landing folder.** v0.43 removed the old hidden `/mcp` collection — prep files now live wherever the user organizes them. Every `create_prep_file` call REQUIRES `collection_id`; call `list_collections` first if you don't have one. Permanent delete is not exposed on this surface — the user does that from the app.
23
+
17
24
  Mutation tools (edit an existing file — you never touch raw PGN):
18
25
 
19
26
  - `apply_mutations(id, [...])` — batch: N ops in one save. **Primary build tool.**
@@ -25,11 +32,12 @@ Every mutation call takes a `node_id` (or `parent_id` for add-style ops) and aut
25
32
 
26
33
  **Before creating a new file, search for an existing one.** LLMs make three "Prep vs Firouzja" files in a row all the time. Always:
27
34
 
28
- 1. `list_prep_files` (if the user has ≤20-30 files) or `search_prep_files(query=<opponent name>)` for their key term
29
- 2. Read the ones that look relevant
30
- 3. Decide: extend an existing one (save_prep_file) or genuinely start fresh (create_prep_file)
35
+ 1. **Text search first**: `search_prep_files(query=<opponent name or opening keyword>)` — searches across every collection the user owns.
36
+ 2. **Position search when the request is position-shaped** ("prep me against 6.f3 in the Najdorf"): `find_position_in_files(fen=<the specific tabiya>)` — catches files that reach the position via a different move order too. This is often more accurate than text search because file names don't always mention every position they cover.
37
+ 3. Read the ones that look relevant.
38
+ 4. Decide: extend an existing one or genuinely start fresh (`create_prep_file(collection_id, name)`).
31
39
 
32
- Duplicate files are the #1 way to lose your user's trust in this system.
40
+ Duplicate files are the #1 way to lose your user's trust in this system. Two searches (text + position) cost roughly nothing and catch nearly all overlap.
33
41
 
34
42
  ## Before writing any prose: read the examples
35
43
 
@@ -107,6 +115,8 @@ Keep it short enough to fit in a picker (30-40 chars). Long titles get truncated
107
115
  - User asks "I found a novelty in the Najdorf" and a Najdorf file exists → extend.
108
116
  - Rule of thumb: if the user's request semantically overlaps with an existing file's [Event] name or main opening line, extend.
109
117
 
110
- ## Icons and appearance
118
+ ## Appearance
119
+
120
+ Prep files land in whichever collection the user picked (via `create_prep_file(collection_id, name)`). They're first-class citizens in that collection — the user can browse, edit, share, or delete them from the app exactly like manually-created games. Your only visibility signal is the [Event] tag; make it descriptive.
111
121
 
112
- The user sees your files in a collection called "AI Prep" with a 🤖 icon under folder `/mcp` in their chess.ceo app. This is intentional — they can tell at a glance which prep came from you, and they can browse / edit / delete from the app just like their manual work. Your files are first-class citizens on their account.
122
+ The `/mcp` "AI Prep" hidden folder from earlier versions no longer exists. If the user has a collection literally named "AI Prep" it's one they created themselves.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@chessceo/mcp",
3
- "version": "0.40.0",
3
+ "version": "0.43.0",
4
4
  "description": "Model Context Protocol server for chess.ceo — 11.7M+ games, ~1.5M FIDE player profiles, opening preparation, live broadcasts.",
5
5
  "type": "module",
6
6
  "bin": {