@chessceo/mcp 0.43.0 → 0.46.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,121 @@
1
+ // Anti-pattern scanners for LLM output + session-level tracking that
2
+ // drives them. Every function returns either a string (the warning to
3
+ // surface on the LLM's next tool response) or undefined (the mutation
4
+ // looked fine). Callers append to a `warnings: string[]` field on the
5
+ // response the LLM sees.
6
+ //
7
+ // Extracted from index.ts in v0.44. All prior v0.42/v0.42.1 behaviour
8
+ // preserved bit-for-bit — this is a file split, not a rewrite.
9
+ import { positionKey } from "./pgn/paths.js";
10
+ import { ROOT_ID } from "./pgn/paths.js";
11
+ // ── Session tracking sets ──────────────────────────────────────────
12
+ //
13
+ // Every one of these is per-process-lifetime memory (i.e. per-session
14
+ // for stdio callers; per-server-instance for streamable-http). Not
15
+ // persisted; a fresh MCP server starts empty. One MCP server process
16
+ // per user in the common stdio case, so these are effectively per-user.
17
+ // Positions the LLM has asked the DB about via `get_position_stats`.
18
+ // Keyed by the 3-field FEN (piece placement + side to move + castling —
19
+ // same key used for transposition detection). Used to warn on `add_move`
20
+ // / `add_line` under a parent the LLM never DB-checked, which is the
21
+ // exact shape of the bug where the LLM read course chapters and
22
+ // cargo-culted a "mainline" that the actual games at the position don't
23
+ // play.
24
+ export const positionsStatsChecked = new Set();
25
+ // Nodes we've ALREADY warned on for the "no stats check" pattern this
26
+ // session, so repeated adds under the same parent don't spam the LLM.
27
+ export const noStatsWarned = new Set();
28
+ // Positions the LLM has called `describe_position` on. Same 3-field FEN
29
+ // key. Live-log audit (2026-07-27 Modern Defence session): 13
30
+ // describe_position calls vs 50+ set_comment ops — most comments were
31
+ // written blind. When describe_position IS called before commentary,
32
+ // prose accuracy jumps sharply.
33
+ export const positionsDescribed = new Set();
34
+ // Once-per-node dedup for the describe warning.
35
+ export const noDescribeWarned = new Set();
36
+ // ── Scanners ───────────────────────────────────────────────────────
37
+ // Detect the anti-patterns the LLM keeps producing in comment prose.
38
+ // All of these restate what the app already renders elsewhere:
39
+ // - spread lists ("5.O-O ≈50, 6.h3 ≈42, ...")
40
+ // - raw centipawn values in prose ("≈-60", "+0.35", "at depth 24")
41
+ // - long roster restatement ("146 GM games, Nakamura, Kramnik, MVL")
42
+ // Return an array of warning strings — one per matched category — so the
43
+ // LLM sees exactly which pattern to remove.
44
+ export function commentAntiPatterns(comment) {
45
+ if (!comment || typeof comment !== "string")
46
+ return [];
47
+ const warns = [];
48
+ // Spread list: ≈ followed by a 2-3-digit number, appearing 2+ times
49
+ // (one appearance is a stray, two+ is a comma-separated spread the LLM
50
+ // pasted from stats output).
51
+ const spreadMatches = comment.match(/≈\s*[+\-−]?\d{1,3}/g) ?? [];
52
+ if (spreadMatches.length >= 2) {
53
+ warns.push("comment contains a spread list (≈ + counts) — the DB viewer already shows sibling counts and fashion scores next to every move, so this is doubled noise. Name the character of the choice instead (\"solid vs sharp\", \"old vs fashionable\") or drop the numbers.");
54
+ }
55
+ // Raw centipawn in prose: "+0.35", "-0.20", "+80" (not preceded by move
56
+ // number). Also "at depth N" or "N nodes" — engine metadata as prose.
57
+ if (/(?:^|[^\d.])[+-]\d\.\d\d(?!\d)/.test(comment) || /≈\s*[+\-−]?\d{2,3}\b/.test(comment) ||
58
+ /\bat depth \d+\b/i.test(comment) || /\b\d{2,3}M nodes\b/.test(comment)) {
59
+ warns.push("comment contains raw centipawn values or engine metadata — the app renders ceoEval + NAG glyph next to every node, so these numbers are doubled noise AND opaque (readers can't tell if ≈-60 means eval, spread, or something else). Set the NAG (set_nags) and let the glyph carry the judgment; drop the number from the prose.");
60
+ }
61
+ // Roster: "N GM games" pattern
62
+ if (/\b\d{2,4}\s+GM games\b/i.test(comment)) {
63
+ warns.push("comment restates game count — the app shows the count on hover. Either cite a specific game with signal (\"Caruana-Liang, Superbet 2026\") or drop the number.");
64
+ }
65
+ return warns;
66
+ }
67
+ // Return a warning string when an add_line is suspiciously long-and-linear
68
+ // (the anti-pattern: LLM pastes a 15-ply engine PV into a single add_line
69
+ // call as if it were prepared repertoire). Two thresholds so the message
70
+ // escalates — a 10-ply Berlin mainline is fine, a 20-ply LLM extrapolation
71
+ // almost never is. Threshold applies at the CALL level, not against
72
+ // existing tree depth — the anti-pattern is a single tool call adding
73
+ // many plies at once with no user thought about where the branching should
74
+ // live.
75
+ export function longLineWarning(sansLength) {
76
+ if (sansLength >= 14) {
77
+ return `you added ${sansLength} plies in one call without branching — this is the shape of a pasted engine PV, not a repertoire. Real prep branches at every ply where the opponent has meaningful alternatives. Either (a) delete the tail and rebuild with add_move at each decision point, calling cloud_analyse + get_position_stats to see what actually gets played, or (b) if this really is one forcing sequence (mate combination, tactical winner), add a comment naming what makes it forced. Long unbranched lines with no comment default to "engine PV pasted as prep" in the reader's eyes.`;
78
+ }
79
+ if (sansLength >= 9) {
80
+ return `${sansLength}-ply linear line — check that every ply is a genuine only-move or a documented mainline. If the opponent has real alternatives at any ply (get_position_stats would show 2+ moves with meaningful frequency), that ply should branch instead. Prep is a tree, not a line.`;
81
+ }
82
+ return undefined;
83
+ }
84
+ // Compute the "you never called describe_position on this node" warning.
85
+ // Fires from set_comment when the comment is substantive (>= 40 chars —
86
+ // anything shorter is a label / pointer, doesn't need structural
87
+ // grounding). LLMs are unreliable at reading FEN strings and confidently
88
+ // describe positions that don't match the actual board; describe_position
89
+ // is a pure-computation grounding pass that reliably fixes this. Warn
90
+ // once per node.
91
+ export function noDescribeWarning(node, comment) {
92
+ if (node.id === ROOT_ID)
93
+ return undefined;
94
+ if (comment.length < 40)
95
+ return undefined;
96
+ const key = positionKey(node.fen);
97
+ if (positionsDescribed.has(key))
98
+ return undefined;
99
+ if (noDescribeWarned.has(node.id))
100
+ return undefined;
101
+ noDescribeWarned.add(node.id);
102
+ return `substantive comment (${comment.length} chars) on a node whose position was never grounded via describe_position this session (id=${node.id}, ${node.san}). LLMs invent captures, miscount pieces, and swap files/ranks when reading FEN strings — describe_position is a pure-computation pass (~1 ms, no engine cost, structural facts + Stockfish's per-term eval breakdown) that reliably prevents this class of hallucination. In live audits, prose accuracy jumps sharply on nodes where describe_position was called first. Call describe_position with file_id+node_id=${node.id} BEFORE writing prose. Warned once per node.`;
103
+ }
104
+ // Compute the "you never DB-checked this parent" warning. Called from
105
+ // add_move / add_line handlers with the parent node. Returns undefined
106
+ // when either (a) the parent was checked this session (or is root — the
107
+ // starting position doesn't need a DB check), (b) we already warned on
108
+ // this parent (dedup so building a big branching subtree isn't spammy),
109
+ // or (c) the mutator is running against a parent whose position has a
110
+ // stored ceoEval (implies the LLM has done SOME analytical work here).
111
+ export function noStatsCheckWarning(parent) {
112
+ if (parent.id === ROOT_ID)
113
+ return undefined;
114
+ const key = positionKey(parent.fen);
115
+ if (positionsStatsChecked.has(key))
116
+ return undefined;
117
+ if (noStatsWarned.has(parent.id))
118
+ return undefined;
119
+ noStatsWarned.add(parent.id);
120
+ return `no get_position_stats call for the parent (id=${parent.id}, ${parent.san}) this session. Course chapter titles describe what an author chose to cover, not what practical opponents play — treating "the So chapter says 6.O-O-O" as "the mainline is 6.O-O-O" is the exact pattern this warning exists to catch. Call get_position_stats at this position (via file_id+node_id=${parent.id}) BEFORE deciding which branches belong here; suppress this warning by making that call. Warned once per parent per session.`;
121
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@chessceo/mcp",
3
- "version": "0.43.0",
3
+ "version": "0.46.0",
4
4
  "description": "Model Context Protocol server for chess.ceo — 11.7M+ games, ~1.5M FIDE player profiles, opening preparation, live broadcasts.",
5
5
  "type": "module",
6
6
  "bin": {