@chessceo/mcp 0.42.0 → 0.44.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/http.js +150 -0
- package/dist/index.js +175 -1135
- package/dist/prompts.js +42 -0
- package/dist/tools.js +893 -0
- package/dist/warnings.js +121 -0
- package/docs/pgn-authoring.md +3 -1
- package/docs/prep-files-guide.md +22 -12
- package/package.json +1 -1
package/dist/warnings.js
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
// Anti-pattern scanners for LLM output + session-level tracking that
|
|
2
|
+
// drives them. Every function returns either a string (the warning to
|
|
3
|
+
// surface on the LLM's next tool response) or undefined (the mutation
|
|
4
|
+
// looked fine). Callers append to a `warnings: string[]` field on the
|
|
5
|
+
// response the LLM sees.
|
|
6
|
+
//
|
|
7
|
+
// Extracted from index.ts in v0.44. All prior v0.42/v0.42.1 behaviour
|
|
8
|
+
// preserved bit-for-bit — this is a file split, not a rewrite.
|
|
9
|
+
import { positionKey } from "./pgn/paths.js";
|
|
10
|
+
import { ROOT_ID } from "./pgn/paths.js";
|
|
11
|
+
// ── Session tracking sets ──────────────────────────────────────────
|
|
12
|
+
//
|
|
13
|
+
// Every one of these is per-process-lifetime memory (i.e. per-session
|
|
14
|
+
// for stdio callers; per-server-instance for streamable-http). Not
|
|
15
|
+
// persisted; a fresh MCP server starts empty. One MCP server process
|
|
16
|
+
// per user in the common stdio case, so these are effectively per-user.
|
|
17
|
+
// Positions the LLM has asked the DB about via `get_position_stats`.
|
|
18
|
+
// Keyed by the 3-field FEN (piece placement + side to move + castling —
|
|
19
|
+
// same key used for transposition detection). Used to warn on `add_move`
|
|
20
|
+
// / `add_line` under a parent the LLM never DB-checked, which is the
|
|
21
|
+
// exact shape of the bug where the LLM read course chapters and
|
|
22
|
+
// cargo-culted a "mainline" that the actual games at the position don't
|
|
23
|
+
// play.
|
|
24
|
+
export const positionsStatsChecked = new Set();
|
|
25
|
+
// Nodes we've ALREADY warned on for the "no stats check" pattern this
|
|
26
|
+
// session, so repeated adds under the same parent don't spam the LLM.
|
|
27
|
+
export const noStatsWarned = new Set();
|
|
28
|
+
// Positions the LLM has called `describe_position` on. Same 3-field FEN
|
|
29
|
+
// key. Live-log audit (2026-07-27 Modern Defence session): 13
|
|
30
|
+
// describe_position calls vs 50+ set_comment ops — most comments were
|
|
31
|
+
// written blind. When describe_position IS called before commentary,
|
|
32
|
+
// prose accuracy jumps sharply.
|
|
33
|
+
export const positionsDescribed = new Set();
|
|
34
|
+
// Once-per-node dedup for the describe warning.
|
|
35
|
+
export const noDescribeWarned = new Set();
|
|
36
|
+
// ── Scanners ───────────────────────────────────────────────────────
|
|
37
|
+
// Detect the anti-patterns the LLM keeps producing in comment prose.
|
|
38
|
+
// All of these restate what the app already renders elsewhere:
|
|
39
|
+
// - spread lists ("5.O-O ≈50, 6.h3 ≈42, ...")
|
|
40
|
+
// - raw centipawn values in prose ("≈-60", "+0.35", "at depth 24")
|
|
41
|
+
// - long roster restatement ("146 GM games, Nakamura, Kramnik, MVL")
|
|
42
|
+
// Return an array of warning strings — one per matched category — so the
|
|
43
|
+
// LLM sees exactly which pattern to remove.
|
|
44
|
+
export function commentAntiPatterns(comment) {
|
|
45
|
+
if (!comment || typeof comment !== "string")
|
|
46
|
+
return [];
|
|
47
|
+
const warns = [];
|
|
48
|
+
// Spread list: ≈ followed by a 2-3-digit number, appearing 2+ times
|
|
49
|
+
// (one appearance is a stray, two+ is a comma-separated spread the LLM
|
|
50
|
+
// pasted from stats output).
|
|
51
|
+
const spreadMatches = comment.match(/≈\s*[+\-−]?\d{1,3}/g) ?? [];
|
|
52
|
+
if (spreadMatches.length >= 2) {
|
|
53
|
+
warns.push("comment contains a spread list (≈ + counts) — the DB viewer already shows sibling counts and fashion scores next to every move, so this is doubled noise. Name the character of the choice instead (\"solid vs sharp\", \"old vs fashionable\") or drop the numbers.");
|
|
54
|
+
}
|
|
55
|
+
// Raw centipawn in prose: "+0.35", "-0.20", "+80" (not preceded by move
|
|
56
|
+
// number). Also "at depth N" or "N nodes" — engine metadata as prose.
|
|
57
|
+
if (/(?:^|[^\d.])[+-]\d\.\d\d(?!\d)/.test(comment) || /≈\s*[+\-−]?\d{2,3}\b/.test(comment) ||
|
|
58
|
+
/\bat depth \d+\b/i.test(comment) || /\b\d{2,3}M nodes\b/.test(comment)) {
|
|
59
|
+
warns.push("comment contains raw centipawn values or engine metadata — the app renders ceoEval + NAG glyph next to every node, so these numbers are doubled noise AND opaque (readers can't tell if ≈-60 means eval, spread, or something else). Set the NAG (set_nags) and let the glyph carry the judgment; drop the number from the prose.");
|
|
60
|
+
}
|
|
61
|
+
// Roster: "N GM games" pattern
|
|
62
|
+
if (/\b\d{2,4}\s+GM games\b/i.test(comment)) {
|
|
63
|
+
warns.push("comment restates game count — the app shows the count on hover. Either cite a specific game with signal (\"Caruana-Liang, Superbet 2026\") or drop the number.");
|
|
64
|
+
}
|
|
65
|
+
return warns;
|
|
66
|
+
}
|
|
67
|
+
// Return a warning string when an add_line is suspiciously long-and-linear
|
|
68
|
+
// (the anti-pattern: LLM pastes a 15-ply engine PV into a single add_line
|
|
69
|
+
// call as if it were prepared repertoire). Two thresholds so the message
|
|
70
|
+
// escalates — a 10-ply Berlin mainline is fine, a 20-ply LLM extrapolation
|
|
71
|
+
// almost never is. Threshold applies at the CALL level, not against
|
|
72
|
+
// existing tree depth — the anti-pattern is a single tool call adding
|
|
73
|
+
// many plies at once with no user thought about where the branching should
|
|
74
|
+
// live.
|
|
75
|
+
export function longLineWarning(sansLength) {
|
|
76
|
+
if (sansLength >= 14) {
|
|
77
|
+
return `you added ${sansLength} plies in one call without branching — this is the shape of a pasted engine PV, not a repertoire. Real prep branches at every ply where the opponent has meaningful alternatives. Either (a) delete the tail and rebuild with add_move at each decision point, calling cloud_analyse + get_position_stats to see what actually gets played, or (b) if this really is one forcing sequence (mate combination, tactical winner), add a comment naming what makes it forced. Long unbranched lines with no comment default to "engine PV pasted as prep" in the reader's eyes.`;
|
|
78
|
+
}
|
|
79
|
+
if (sansLength >= 9) {
|
|
80
|
+
return `${sansLength}-ply linear line — check that every ply is a genuine only-move or a documented mainline. If the opponent has real alternatives at any ply (get_position_stats would show 2+ moves with meaningful frequency), that ply should branch instead. Prep is a tree, not a line.`;
|
|
81
|
+
}
|
|
82
|
+
return undefined;
|
|
83
|
+
}
|
|
84
|
+
// Compute the "you never called describe_position on this node" warning.
|
|
85
|
+
// Fires from set_comment when the comment is substantive (>= 40 chars —
|
|
86
|
+
// anything shorter is a label / pointer, doesn't need structural
|
|
87
|
+
// grounding). LLMs are unreliable at reading FEN strings and confidently
|
|
88
|
+
// describe positions that don't match the actual board; describe_position
|
|
89
|
+
// is a pure-computation grounding pass that reliably fixes this. Warn
|
|
90
|
+
// once per node.
|
|
91
|
+
export function noDescribeWarning(node, comment) {
|
|
92
|
+
if (node.id === ROOT_ID)
|
|
93
|
+
return undefined;
|
|
94
|
+
if (comment.length < 40)
|
|
95
|
+
return undefined;
|
|
96
|
+
const key = positionKey(node.fen);
|
|
97
|
+
if (positionsDescribed.has(key))
|
|
98
|
+
return undefined;
|
|
99
|
+
if (noDescribeWarned.has(node.id))
|
|
100
|
+
return undefined;
|
|
101
|
+
noDescribeWarned.add(node.id);
|
|
102
|
+
return `substantive comment (${comment.length} chars) on a node whose position was never grounded via describe_position this session (id=${node.id}, ${node.san}). LLMs invent captures, miscount pieces, and swap files/ranks when reading FEN strings — describe_position is a pure-computation pass (~1 ms, no engine cost, structural facts + Stockfish's per-term eval breakdown) that reliably prevents this class of hallucination. In live audits, prose accuracy jumps sharply on nodes where describe_position was called first. Call describe_position with file_id+node_id=${node.id} BEFORE writing prose. Warned once per node.`;
|
|
103
|
+
}
|
|
104
|
+
// Compute the "you never DB-checked this parent" warning. Called from
|
|
105
|
+
// add_move / add_line handlers with the parent node. Returns undefined
|
|
106
|
+
// when either (a) the parent was checked this session (or is root — the
|
|
107
|
+
// starting position doesn't need a DB check), (b) we already warned on
|
|
108
|
+
// this parent (dedup so building a big branching subtree isn't spammy),
|
|
109
|
+
// or (c) the mutator is running against a parent whose position has a
|
|
110
|
+
// stored ceoEval (implies the LLM has done SOME analytical work here).
|
|
111
|
+
export function noStatsCheckWarning(parent) {
|
|
112
|
+
if (parent.id === ROOT_ID)
|
|
113
|
+
return undefined;
|
|
114
|
+
const key = positionKey(parent.fen);
|
|
115
|
+
if (positionsStatsChecked.has(key))
|
|
116
|
+
return undefined;
|
|
117
|
+
if (noStatsWarned.has(parent.id))
|
|
118
|
+
return undefined;
|
|
119
|
+
noStatsWarned.add(parent.id);
|
|
120
|
+
return `no get_position_stats call for the parent (id=${parent.id}, ${parent.san}) this session. Course chapter titles describe what an author chose to cover, not what practical opponents play — treating "the So chapter says 6.O-O-O" as "the mainline is 6.O-O-O" is the exact pattern this warning exists to catch. Call get_position_stats at this position (via file_id+node_id=${parent.id}) BEFORE deciding which branches belong here; suppress this warning by making that call. Warned once per parent per session.`;
|
|
121
|
+
}
|
package/docs/pgn-authoring.md
CHANGED
|
@@ -316,7 +316,9 @@ Contempt scale is signed 0-100 (same as the web UI's ContemptStrength slider). T
|
|
|
316
316
|
|
|
317
317
|
### Before you write any commentary: describe_position
|
|
318
318
|
|
|
319
|
-
|
|
319
|
+
**This is the biggest lever for prose quality in the whole system.** Live audit of a recent session — 13 `describe_position` calls versus 50+ `set_comment` ops. The nodes where `describe_position` was called first produced comments that grounded specifically in the position (correct piece squares, real pawn structure, actual weak squares). The nodes where it wasn't produced generic prose that pattern-matched to similar-*looking* positions and confidently named pieces on wrong squares. This gap is why `set_comment` now emits a warning whenever a substantive comment (≥40 chars) lands on a node whose position was never grounded via `describe_position` this session.
|
|
320
|
+
|
|
321
|
+
LLMs are not reliable at reading FEN strings — you'll swap files/ranks, invent captures, miscount pieces. Before you write a comment describing what's happening in a position, call `describe_position` (with `file_id`+`node_id` inside a prep file). Pure computation (~1 ms, no engine), returns two layers:
|
|
320
322
|
|
|
321
323
|
**Board state** — piece placements, material, contested pieces (attackers + defenders), hanging list, check state, castling, en passant, legal moves. Fixes the *"Black's queen on c7 is defended by the knight on d5"* failure when actually there's no knight on d5 and the queen is on c8.
|
|
322
324
|
|
package/docs/prep-files-guide.md
CHANGED
|
@@ -4,16 +4,23 @@ You can save chess prep to the user's chess.ceo account and read it back across
|
|
|
4
4
|
|
|
5
5
|
## The mental model
|
|
6
6
|
|
|
7
|
-
The user has **
|
|
7
|
+
The user has **any number of PGN collections** in their chess.ceo library — you have full access to every non-encrypted one. A **prep file** is one PGN game with variations, inside a collection. Every prep file id is a composite `<collection_id>:<game_id>` — opaque to you, pass it through unchanged to any tool that takes `id` or `file_id`.
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
Discovery tools:
|
|
10
10
|
|
|
11
|
-
- `
|
|
12
|
-
- `
|
|
13
|
-
- `
|
|
14
|
-
- `
|
|
11
|
+
- `list_collections` — show me all the user's collections (their organizational scheme is theirs; browse before creating)
|
|
12
|
+
- `list_prep_files(collection_id)` — the games inside one collection
|
|
13
|
+
- `search_prep_files(query)` — text search across ALL of the user's collections
|
|
14
|
+
- `find_position_in_files(fen)` — position search across ALL of the user's collections (matched by zobrist hash so move-order variants are found automatically). Distinct from `find_position_in_courses` — that's the user's read-only reference library (Chessable etc.); this is their own editable prep
|
|
15
|
+
|
|
16
|
+
Per-file tools:
|
|
17
|
+
|
|
18
|
+
- `read_prep_file(id)` — parsed tree (every node carries a stable content-derived `id`) + tags + `version`
|
|
19
|
+
- `create_prep_file(collection_id, name)` — new empty file inside the given collection, `name` becomes the [Event] tag
|
|
15
20
|
- `delete_prep_file(id)` — soft delete (user can restore from app)
|
|
16
21
|
|
|
22
|
+
**No default landing folder.** v0.43 removed the old hidden `/mcp` collection — prep files now live wherever the user organizes them. Every `create_prep_file` call REQUIRES `collection_id`; call `list_collections` first if you don't have one. Permanent delete is not exposed on this surface — the user does that from the app.
|
|
23
|
+
|
|
17
24
|
Mutation tools (edit an existing file — you never touch raw PGN):
|
|
18
25
|
|
|
19
26
|
- `apply_mutations(id, [...])` — batch: N ops in one save. **Primary build tool.**
|
|
@@ -25,11 +32,12 @@ Every mutation call takes a `node_id` (or `parent_id` for add-style ops) and aut
|
|
|
25
32
|
|
|
26
33
|
**Before creating a new file, search for an existing one.** LLMs make three "Prep vs Firouzja" files in a row all the time. Always:
|
|
27
34
|
|
|
28
|
-
1.
|
|
29
|
-
2.
|
|
30
|
-
3.
|
|
35
|
+
1. **Text search first**: `search_prep_files(query=<opponent name or opening keyword>)` — searches across every collection the user owns.
|
|
36
|
+
2. **Position search when the request is position-shaped** ("prep me against 6.f3 in the Najdorf"): `find_position_in_files(fen=<the specific tabiya>)` — catches files that reach the position via a different move order too. This is often more accurate than text search because file names don't always mention every position they cover.
|
|
37
|
+
3. Read the ones that look relevant.
|
|
38
|
+
4. Decide: extend an existing one or genuinely start fresh (`create_prep_file(collection_id, name)`).
|
|
31
39
|
|
|
32
|
-
Duplicate files are the #1 way to lose your user's trust in this system.
|
|
40
|
+
Duplicate files are the #1 way to lose your user's trust in this system. Two searches (text + position) cost roughly nothing and catch nearly all overlap.
|
|
33
41
|
|
|
34
42
|
## Before writing any prose: read the examples
|
|
35
43
|
|
|
@@ -107,6 +115,8 @@ Keep it short enough to fit in a picker (30-40 chars). Long titles get truncated
|
|
|
107
115
|
- User asks "I found a novelty in the Najdorf" and a Najdorf file exists → extend.
|
|
108
116
|
- Rule of thumb: if the user's request semantically overlaps with an existing file's [Event] name or main opening line, extend.
|
|
109
117
|
|
|
110
|
-
##
|
|
118
|
+
## Appearance
|
|
119
|
+
|
|
120
|
+
Prep files land in whichever collection the user picked (via `create_prep_file(collection_id, name)`). They're first-class citizens in that collection — the user can browse, edit, share, or delete them from the app exactly like manually-created games. Your only visibility signal is the [Event] tag; make it descriptive.
|
|
111
121
|
|
|
112
|
-
The
|
|
122
|
+
The `/mcp` "AI Prep" hidden folder from earlier versions no longer exists. If the user has a collection literally named "AI Prep" it's one they created themselves.
|
package/package.json
CHANGED