xiangqibench 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- xiangqibench-0.1.0/.gitignore +34 -0
- xiangqibench-0.1.0/CHANGELOG.md +59 -0
- xiangqibench-0.1.0/CITATION.cff +22 -0
- xiangqibench-0.1.0/DATA_CARD.md +92 -0
- xiangqibench-0.1.0/LICENSE +21 -0
- xiangqibench-0.1.0/PKG-INFO +172 -0
- xiangqibench-0.1.0/README.md +130 -0
- xiangqibench-0.1.0/pyproject.toml +79 -0
- xiangqibench-0.1.0/src/xiangqibench/__init__.py +9 -0
- xiangqibench-0.1.0/src/xiangqibench/cases.py +123 -0
- xiangqibench-0.1.0/src/xiangqibench/cli.py +283 -0
- xiangqibench-0.1.0/src/xiangqibench/config.py +203 -0
- xiangqibench-0.1.0/src/xiangqibench/data/cases.jsonl +119 -0
- xiangqibench-0.1.0/src/xiangqibench/data/splits/ablation-gemini-3.1-pro.txt +21 -0
- xiangqibench-0.1.0/src/xiangqibench/data/splits/ablation-gpt-5.5.txt +22 -0
- xiangqibench-0.1.0/src/xiangqibench/defender/__init__.py +18 -0
- xiangqibench-0.1.0/src/xiangqibench/defender/defender.py +124 -0
- xiangqibench-0.1.0/src/xiangqibench/defender/engine.py +298 -0
- xiangqibench-0.1.0/src/xiangqibench/defender/search.py +201 -0
- xiangqibench-0.1.0/src/xiangqibench/example_config.yaml +63 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/__init__.py +6 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/actions.py +265 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/adapter.py +201 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/config.py +62 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/harness.py +239 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/player_state.py +119 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/prompts.py +201 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/protocol.py +137 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/record.py +163 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/simulator.py +213 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/templates/restricted.txt +93 -0
- xiangqibench-0.1.0/src/xiangqibench/harness/templates/sighted.txt +142 -0
- xiangqibench-0.1.0/src/xiangqibench/llm/__init__.py +4 -0
- xiangqibench-0.1.0/src/xiangqibench/llm/base.py +43 -0
- xiangqibench-0.1.0/src/xiangqibench/llm/providers.py +191 -0
- xiangqibench-0.1.0/src/xiangqibench/modes.py +74 -0
- xiangqibench-0.1.0/src/xiangqibench/py.typed +0 -0
- xiangqibench-0.1.0/src/xiangqibench/replay.py +193 -0
- xiangqibench-0.1.0/src/xiangqibench/rules/__init__.py +3 -0
- xiangqibench-0.1.0/src/xiangqibench/rules/xiangqi_env.py +252 -0
- xiangqibench-0.1.0/src/xiangqibench/runner.py +332 -0
- xiangqibench-0.1.0/src/xiangqibench/scoring.py +238 -0
- xiangqibench-0.1.0/tests/conftest.py +30 -0
- xiangqibench-0.1.0/tests/golden/legacy/repl-blind__check_flag.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/legacy/repl-sighted__check_flag.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/legacy/repl-sighted__early_no_command_budget.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/prompts/repl-abl-R-T.json +5 -0
- xiangqibench-0.1.0/tests/golden/prompts/repl-abl-R.json +5 -0
- xiangqibench-0.1.0/tests/golden/prompts/repl-abl-S-NT.json +5 -0
- xiangqibench-0.1.0/tests/golden/prompts/repl-abl-S.json +5 -0
- xiangqibench-0.1.0/tests/golden/prompts/repl-blind.json +5 -0
- xiangqibench-0.1.0/tests/golden/prompts/repl-sighted.json +5 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-abl-R-T__tool.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-abl-R-T__win.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-abl-R__tool.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-abl-R__win.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-abl-S-NT__tool.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-abl-S-NT__win.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-abl-S__tool.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-abl-S__win.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-blind__invalid.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-blind__tool.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-blind__win.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-sighted__invalid.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-sighted__tool.json.gz +0 -0
- xiangqibench-0.1.0/tests/golden/trials/repl-sighted__win.json.gz +0 -0
- xiangqibench-0.1.0/tests/test_cli.py +45 -0
- xiangqibench-0.1.0/tests/test_config.py +60 -0
- xiangqibench-0.1.0/tests/test_conformance.py +102 -0
- xiangqibench-0.1.0/tests/test_engine.py +35 -0
- xiangqibench-0.1.0/tests/test_protocol.py +33 -0
- xiangqibench-0.1.0/tests/test_rules.py +76 -0
- xiangqibench-0.1.0/tests/test_runner.py +103 -0
- xiangqibench-0.1.0/tests/test_scoring.py +55 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Secrets: keys belong in the environment, never in the repository
|
|
2
|
+
.env
|
|
3
|
+
.env.*
|
|
4
|
+
!.env.example
|
|
5
|
+
*.key
|
|
6
|
+
config/*.yaml
|
|
7
|
+
!config/*.example.yaml
|
|
8
|
+
*api*.yaml
|
|
9
|
+
!src/xiangqibench/example_config.yaml
|
|
10
|
+
|
|
11
|
+
# Run outputs
|
|
12
|
+
runs/
|
|
13
|
+
output/
|
|
14
|
+
*.log
|
|
15
|
+
|
|
16
|
+
# Python
|
|
17
|
+
__pycache__/
|
|
18
|
+
*.py[cod]
|
|
19
|
+
*.egg-info/
|
|
20
|
+
build/
|
|
21
|
+
dist/
|
|
22
|
+
.venv/
|
|
23
|
+
venv/
|
|
24
|
+
|
|
25
|
+
# Tooling
|
|
26
|
+
.pytest_cache/
|
|
27
|
+
.ruff_cache/
|
|
28
|
+
.mypy_cache/
|
|
29
|
+
.coverage
|
|
30
|
+
coverage.xml
|
|
31
|
+
htmlcov/
|
|
32
|
+
.DS_Store
|
|
33
|
+
.idea/
|
|
34
|
+
.vscode/
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
|
|
5
|
+
[Semantic Versioning](https://semver.org/).
|
|
6
|
+
|
|
7
|
+
## [0.1.0] — 2026-09-26
|
|
8
|
+
|
|
9
|
+
First public release. It packages the XiangqiBench protocol, the 119 positions, the Pikafish
|
|
10
|
+
defender, and the scoring used in the paper.
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
|
|
14
|
+
- The `xiangqibench` package and CLI, with the subcommands `run`, `score`, `modes`, `prompt`,
|
|
15
|
+
`cases`, `doctor`, `init`, and `play`.
|
|
16
|
+
- Six evaluation modes, selected by `mode:` in the config file or by `--mode`:
|
|
17
|
+
- the paper settings `sighted` and `restricted`;
|
|
18
|
+
- the observation-ablation arms `S`, `S-NT`, `R-T`, and `R`.
|
|
19
|
+
- YAML configuration with `${VAR}` / `${VAR:-default}` expansion from the environment, strict
|
|
20
|
+
key validation, and command-line overrides.
|
|
21
|
+
- Model clients for Chat Completions (OpenAI, OpenRouter, vLLM, SGLang, or any compatible
|
|
22
|
+
server), Azure OpenAI, the OpenAI Responses API, and Anthropic. Calls are retried, and a
|
|
23
|
+
trial whose calls all fail writes no record.
|
|
24
|
+
- The Pikafish defender, with a per-position fallback to a depth-5 rule search. Each defender
|
|
25
|
+
move records its backend, and the engine id and NNUE sha256 are stored with every trial.
|
|
26
|
+
- A resumable, parallel suite runner. Each worker has its own engine, and records are written
|
|
27
|
+
atomically.
|
|
28
|
+
- Scoring: pass@k, pass^k, earliest-three selection per cell, and 95% case-bootstrap intervals
|
|
29
|
+
with the paper's seeds.
|
|
30
|
+
- `xiangqibench.replay.replay_record`, which re-verifies an archived trial against the current
|
|
31
|
+
code.
|
|
32
|
+
- Conformance tests:
|
|
33
|
+
- archived prompts for all six settings, byte for byte;
|
|
34
|
+
- replay of 14 archived trials;
|
|
35
|
+
- rules, protocol, config, scoring, runner, and CLI tests.
|
|
36
|
+
|
|
37
|
+
### Fixed (relative to the code that produced the paper's archive)
|
|
38
|
+
|
|
39
|
+
These fixes change what future runs record. Archived results are scored as they were recorded.
|
|
40
|
+
The effect of each fix on the paper's analyses is described in the paper.
|
|
41
|
+
|
|
42
|
+
- **Check flag.** Feedback used to report `Check: No` after every move, including checking and
|
|
43
|
+
mating moves, and mates were labelled `stalemate`. The flag is now computed on the position
|
|
44
|
+
after the move, and mates are labelled `checkmate`. The trajectory's `in_check` is also set on
|
|
45
|
+
the final, game-ending move. Winners were never affected.
|
|
46
|
+
- **Engine failures.** An engine that failed during search, for example on a network it cannot
|
|
47
|
+
load, silently handed every defender move to the rule fallback. It now stops the run with an
|
|
48
|
+
error, and only positions the engine rejects use the fallback.
|
|
49
|
+
- **Repetition.** A threefold repetition in which one side checked on every one of its moves now
|
|
50
|
+
loses for that side (perpetual check) instead of being a draw.
|
|
51
|
+
- **Forfeit move list.** Forfeit turns appended a copy of the previous move to `moves`, so
|
|
52
|
+
`plies` was one too high for forfeits after at least one move. Forfeits now append nothing.
|
|
53
|
+
- **Statistics.** The simulation counter in `stats` was never incremented. All counters now use
|
|
54
|
+
their documented keys.
|
|
55
|
+
- **Defender timing.** The defender's `duration_ms` now includes the engine search time.
|
|
56
|
+
- **Record size.** Per-call prompt snapshots are off by default. The full message history is
|
|
57
|
+
still stored once per trial.
|
|
58
|
+
|
|
59
|
+
[0.1.0]: https://github.com/floatai/xiangqibench/releases/tag/v0.1.0
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
message: "If you use XiangqiBench, please cite the paper below."
|
|
3
|
+
title: "XiangqiBench"
|
|
4
|
+
type: software
|
|
5
|
+
authors:
|
|
6
|
+
- name: "XiangqiBench authors"
|
|
7
|
+
version: 0.1.0
|
|
8
|
+
date-released: 2026-09-26
|
|
9
|
+
license: MIT
|
|
10
|
+
repository-code: "https://github.com/floatai/xiangqibench"
|
|
11
|
+
keywords:
|
|
12
|
+
- llm-agents
|
|
13
|
+
- benchmark
|
|
14
|
+
- xiangqi
|
|
15
|
+
- chinese-chess
|
|
16
|
+
preferred-citation:
|
|
17
|
+
type: article
|
|
18
|
+
title: "Finding the Move Is Not Winning the Game: XiangqiBench for Closed-Loop Evaluation of LLM Agents"
|
|
19
|
+
authors:
|
|
20
|
+
- name: "XiangqiBench authors"
|
|
21
|
+
year: 2026
|
|
22
|
+
url: "https://github.com/floatai/xiangqibench"
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# Data card: XiangqiBench positions
|
|
2
|
+
|
|
3
|
+
## Summary
|
|
4
|
+
|
|
5
|
+
`src/xiangqibench/data/cases.jsonl` holds 119 composed xiangqi endgames. In every one, Red is to
|
|
6
|
+
move and can force checkmate. Load them with `xiangqibench.load_cases()` or list them with
|
|
7
|
+
`xiangqibench cases`.
|
|
8
|
+
|
|
9
|
+
| | |
|
|
10
|
+
|---|---|
|
|
11
|
+
| Positions | 119 (116 *Shi Qing Ya Qu*, 3 *Jianghu*) |
|
|
12
|
+
| Side to move | Red in all positions (the agent plays Red) |
|
|
13
|
+
| Stored mate distance | 3–11 plies, median 9 |
|
|
14
|
+
| Categories | 106 `threatmate_race`, 13 `forced_mate` |
|
|
15
|
+
| Admission | 117 confirmed by Pikafish (`verified_by: engine`), 2 by a checks-only mate search (`solver_proof`) |
|
|
16
|
+
| sha256 | `3cb4e29148f82cb6d9a881e28336676549225cfe86dd8fb947ccf580f1633d42` |
|
|
17
|
+
|
|
18
|
+
## Provenance
|
|
19
|
+
|
|
20
|
+
- ***Shi Qing Ya Qu*** (适情雅趣) is a Ming-dynasty treatise of composed endgames, printed in
|
|
21
|
+
1570. Digitized entries were taken from
|
|
22
|
+
[xqipu.com](https://www.xqipu.com/canjugupu/1547). Case ids are
|
|
23
|
+
`xq_shi_qing_ya_qu_<n>`, and `raw_index` is the entry's position in that digitization.
|
|
24
|
+
- ***Jianghu*** (江湖) is a classical collection of street endgames. Case ids are
|
|
25
|
+
`xq_jianghu_endgames_<n>`.
|
|
26
|
+
|
|
27
|
+
Both sources are several centuries old and are in the public domain. The historical solution
|
|
28
|
+
text is not included, is never shown to agents, and is not used for scoring. They may still
|
|
29
|
+
appear in pre-training corpora, so first-move agreement can partly reflect recall.
|
|
30
|
+
|
|
31
|
+
## Admission
|
|
32
|
+
|
|
33
|
+
Source solutions were not trusted. Each position was screened with a single-threaded,
|
|
34
|
+
time-limited Pikafish search and admitted when the engine reported a forced mate for Red. The
|
|
35
|
+
engine's mate distance and preferred first move are stored as `engine_mate_plies` and
|
|
36
|
+
`first_winning_move`.
|
|
37
|
+
|
|
38
|
+
Pikafish does not confirm a mate in two compositions, `xq_jianghu_endgames_084` and
|
|
39
|
+
`xq_jianghu_endgames_327`. They were admitted by a checks-only mate search and carry
|
|
40
|
+
`verified_by: solver_proof`. All 119 positions were also checked by hand.
|
|
41
|
+
Admission relies on bounded search, so the positions are *search-supported* rather than formally
|
|
42
|
+
proved.
|
|
43
|
+
|
|
44
|
+
**Categories.** The same search was run with Black to move. If Black then also has a forced mate
|
|
45
|
+
(`defender_threat_in_plies`), the position is a `threatmate_race`: a slow Red move can let Black
|
|
46
|
+
mate first. Otherwise it is a `forced_mate`.
|
|
47
|
+
|
|
48
|
+
## Fields
|
|
49
|
+
|
|
50
|
+
| Field | Meaning |
|
|
51
|
+
|---|---|
|
|
52
|
+
| `id` | Stable case id |
|
|
53
|
+
| `source`, `raw_index`, `name` | Source collection, index in the digitization, original title |
|
|
54
|
+
| `fen` | Start position (FEN, Red to move) |
|
|
55
|
+
| `challenger` | Side played by the agent (`red`) |
|
|
56
|
+
| `category` | `threatmate_race` or `forced_mate` |
|
|
57
|
+
| `win_in_plies` | Stored mate distance (engine or solver) |
|
|
58
|
+
| `engine_mate_plies` | Pikafish mate distance, or `null` if the engine does not confirm a mate |
|
|
59
|
+
| `defender_threat_in_plies` | Black's mate distance if Black were to move, or `null` |
|
|
60
|
+
| `first_winning_move` | Reference root move from the engine (used only for first-move analysis) |
|
|
61
|
+
| `legal_move_count` | Number of legal Red moves at the root |
|
|
62
|
+
| `verified_by` | `engine` or `solver_proof` |
|
|
63
|
+
| `difficulty_score`, `tier`, `piece_theme`, `tags` | Descriptive metadata; not used for scoring |
|
|
64
|
+
|
|
65
|
+
The stored distances are admission metadata. They do not limit the game: every trial may run up
|
|
66
|
+
to 40 plies.
|
|
67
|
+
|
|
68
|
+
## Splits
|
|
69
|
+
|
|
70
|
+
| Split | Cases | Use |
|
|
71
|
+
|---|---|---|
|
|
72
|
+
| `main` | 119 | Leaderboard (paper settings `sighted` and `restricted`) |
|
|
73
|
+
| `ablation-gemini-3.1-pro` | 20 | Observation ablation for Gemini 3.1 Pro |
|
|
74
|
+
| `ablation-gpt-5.5` | 21 | Observation ablation for GPT-5.5 |
|
|
75
|
+
|
|
76
|
+
The ablation subsets are **model-specific**. Each contains positions that the model won at
|
|
77
|
+
least once in the paper's Sighted trials, so the ablation measures how much of a demonstrated
|
|
78
|
+
ability survives each observation change.
|
|
79
|
+
|
|
80
|
+
- For GPT-5.5 the subset contains every such position.
|
|
81
|
+
- For Gemini 3.1 Pro it is a stratified sample by mate distance, with a quota
|
|
82
|
+
{5: 3, 7: 6, 9: 5, 11: 6} and seed 20260926.
|
|
83
|
+
|
|
84
|
+
Ablation numbers are therefore not comparable across models or with the `main` split.
|
|
85
|
+
|
|
86
|
+
## Engine coverage
|
|
87
|
+
|
|
88
|
+
Pikafish refuses to search two positions reachable in the benchmark: the start positions of
|
|
89
|
+
`xq_jianghu_endgames_084` and `xq_jianghu_endgames_327`, and the positions after their first
|
|
90
|
+
move. For such positions, the defender falls back to a depth-5 rule search. Each defender move
|
|
91
|
+
records which backend chose it, so the share of rule-chosen moves can be computed from any set
|
|
92
|
+
of records.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 FloatAI
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: xiangqibench
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Tool-grounded xiangqi endgames for evaluating LLM agents against a Pikafish defender.
|
|
5
|
+
Project-URL: Homepage, https://github.com/floatai/xiangqibench
|
|
6
|
+
Project-URL: Repository, https://github.com/floatai/xiangqibench
|
|
7
|
+
Project-URL: Issues, https://github.com/floatai/xiangqibench/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/floatai/xiangqibench/blob/main/CHANGELOG.md
|
|
9
|
+
Author: XiangqiBench authors
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: agents,benchmark,chinese-chess,evaluation,llm,pikafish,xiangqi
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Games/Entertainment :: Board Games
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Requires-Python: >=3.10
|
|
25
|
+
Requires-Dist: cchess<1.26,>=1.25.5
|
|
26
|
+
Requires-Dist: numpy>=1.24
|
|
27
|
+
Requires-Dist: openai>=1.60
|
|
28
|
+
Requires-Dist: pyyaml>=6.0
|
|
29
|
+
Provides-Extra: all
|
|
30
|
+
Requires-Dist: anthropic>=0.40; extra == 'all'
|
|
31
|
+
Provides-Extra: anthropic
|
|
32
|
+
Requires-Dist: anthropic>=0.40; extra == 'anthropic'
|
|
33
|
+
Provides-Extra: dev
|
|
34
|
+
Requires-Dist: build>=1.2; extra == 'dev'
|
|
35
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
36
|
+
Requires-Dist: pytest-cov>=5; extra == 'dev'
|
|
37
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
38
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
39
|
+
Requires-Dist: twine>=5; extra == 'dev'
|
|
40
|
+
Requires-Dist: types-pyyaml; extra == 'dev'
|
|
41
|
+
Description-Content-Type: text/markdown
|
|
42
|
+
|
|
43
|
+
# XiangqiBench
|
|
44
|
+
|
|
45
|
+
[][paper]
|
|
46
|
+
[](https://pypi.org/project/xiangqibench/)
|
|
47
|
+
[](https://pypi.org/project/xiangqibench/)
|
|
48
|
+
[](https://github.com/floatai/xiangqibench/actions/workflows/ci.yml)
|
|
49
|
+
[](LICENSE)
|
|
50
|
+
|
|
51
|
+
**[Paper][paper]** | **[Data card](DATA_CARD.md)** | **[Changelog](CHANGELOG.md)** | **[Citation](#citation)**
|
|
52
|
+
|
|
53
|
+
This repository contains the official implementation of *Finding the Move Is Not Winning the
|
|
54
|
+
Game: XiangqiBench for Closed-Loop Evaluation of LLM Agents*.
|
|
55
|
+
|
|
56
|
+
XiangqiBench asks an LLM agent to convert 119 composed xiangqi (Chinese chess) endgames into
|
|
57
|
+
checkmate against a Pikafish defender. The agent acts through a small command protocol under
|
|
58
|
+
per-turn budgets and is scored only on whether it actually delivers mate: finding the right
|
|
59
|
+
first move is not enough.
|
|
60
|
+
|
|
61
|
+
## Installation
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
pip install xiangqibench # OpenAI-compatible and Azure endpoints
|
|
65
|
+
pip install "xiangqibench[anthropic]" # adds the native Anthropic client
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
XiangqiBench requires Python 3.10 or later and a [Pikafish](https://github.com/official-pikafish/Pikafish)
|
|
69
|
+
binary for the defender:
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
git clone https://github.com/official-pikafish/Pikafish && make -C Pikafish/src -j build
|
|
73
|
+
export PIKAFISH_PATH=$PWD/Pikafish/src/pikafish
|
|
74
|
+
xiangqibench doctor # checks the engine and prints its build and NNUE hash
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
The paper's defender was Pikafish `fd168f68` with the network whose sha256 begins `a2f41d4d`,
|
|
78
|
+
searched to depth 18 with one thread and a 256 MB hash. Upstream has since replaced that network,
|
|
79
|
+
and older builds cannot load the new one, so build the current Pikafish as shown above. Its
|
|
80
|
+
moves can differ from the paper's defender. Every record stores the engine build and the network
|
|
81
|
+
hash.
|
|
82
|
+
|
|
83
|
+
## Usage
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
export OPENAI_API_KEY=...
|
|
87
|
+
xiangqibench run --model gpt-5.5 --mode sighted --limit 5
|
|
88
|
+
xiangqibench score runs/
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Full runs are configured with a single YAML file; `xiangqibench init` writes a commented
|
|
92
|
+
example. API keys are read from the environment, never from the file.
|
|
93
|
+
|
|
94
|
+
```yaml
|
|
95
|
+
mode: restricted # sighted | restricted | S | S-NT | R-T | R
|
|
96
|
+
model:
|
|
97
|
+
name: qwen3-235b
|
|
98
|
+
provider: openai # openai | azure | openai-responses | anthropic
|
|
99
|
+
base_url: http://localhost:8000/v1
|
|
100
|
+
api_key_env: VLLM_API_KEY
|
|
101
|
+
run:
|
|
102
|
+
trials: 3
|
|
103
|
+
workers: 8
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
```bash
|
|
107
|
+
xiangqibench run -c my_run.yaml # resumable: re-running fills in missing trials
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Command-line flags override the file, and unknown keys are rejected.
|
|
111
|
+
|
|
112
|
+
### Modes
|
|
113
|
+
|
|
114
|
+
| Mode | Observation | Tools |
|
|
115
|
+
|---|---|---|
|
|
116
|
+
| `sighted` | Board, FEN, and legal moves after every ply | `view_board`, `simulate`, `get_legal_moves` |
|
|
117
|
+
| `restricted` | Starting position once, then move diffs only | none |
|
|
118
|
+
| `S`, `S-NT`, `R-T`, `R` | Observation ablations: state push (S/R) × tool access (T/NT) | as named |
|
|
119
|
+
|
|
120
|
+
`sighted` and `restricted` are the paper's two settings. `xiangqibench prompt --mode <mode>`
|
|
121
|
+
prints the exact system prompt for any mode.
|
|
122
|
+
|
|
123
|
+
### Scoring
|
|
124
|
+
|
|
125
|
+
`xiangqibench score` reports pass@k and pass^k over the earliest three scored trials per
|
|
126
|
+
(model, mode, case), with 95% case-bootstrap intervals using the paper's seeds. Trials cut short
|
|
127
|
+
by infrastructure errors are excluded and re-run automatically. Runs that change the standard
|
|
128
|
+
budgets or defender settings are marked `standard: false`.
|
|
129
|
+
|
|
130
|
+
### Python API
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
from xiangqibench import load_config
|
|
134
|
+
from xiangqibench.runner import run_suite
|
|
135
|
+
|
|
136
|
+
report = run_suite(load_config("my_run.yaml"))
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
Any object with a `name` attribute and a `complete(messages) -> Completion` method can be
|
|
140
|
+
evaluated as an agent; see `xiangqibench.runner.play_trial`.
|
|
141
|
+
|
|
142
|
+
## Reproducibility
|
|
143
|
+
|
|
144
|
+
Every trial is stored as one JSON record with the full message history, the move list, the
|
|
145
|
+
resolved configuration, and the defender's identity, including which backend chose each
|
|
146
|
+
defender move. The test suite replays archived trials from the paper against this code and checks
|
|
147
|
+
every environment message and verdict (`xiangqibench.replay`). Known differences from the code
|
|
148
|
+
that produced the paper's archive are listed in the [changelog](CHANGELOG.md).
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
pip install -e ".[dev]" && pytest
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
## Citation
|
|
155
|
+
|
|
156
|
+
```bibtex
|
|
157
|
+
@article{xiangqibench2026,
|
|
158
|
+
title = {Finding the Move Is Not Winning the Game: {XiangqiBench} for Closed-Loop
|
|
159
|
+
Evaluation of {LLM} Agents},
|
|
160
|
+
author = {XiangqiBench authors},
|
|
161
|
+
year = {2026},
|
|
162
|
+
url = {https://github.com/floatai/xiangqibench}
|
|
163
|
+
}
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
## License
|
|
167
|
+
|
|
168
|
+
The code is released under the [MIT License](LICENSE). The historical positions are in the
|
|
169
|
+
public domain. Pikafish is licensed under GPL-3.0; it is not distributed with this package and
|
|
170
|
+
runs as a separate process.
|
|
171
|
+
|
|
172
|
+
[paper]: https://arxiv.org/abs/XXXX.XXXXX
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# XiangqiBench
|
|
2
|
+
|
|
3
|
+
[][paper]
|
|
4
|
+
[](https://pypi.org/project/xiangqibench/)
|
|
5
|
+
[](https://pypi.org/project/xiangqibench/)
|
|
6
|
+
[](https://github.com/floatai/xiangqibench/actions/workflows/ci.yml)
|
|
7
|
+
[](LICENSE)
|
|
8
|
+
|
|
9
|
+
**[Paper][paper]** | **[Data card](DATA_CARD.md)** | **[Changelog](CHANGELOG.md)** | **[Citation](#citation)**
|
|
10
|
+
|
|
11
|
+
This repository contains the official implementation of *Finding the Move Is Not Winning the
|
|
12
|
+
Game: XiangqiBench for Closed-Loop Evaluation of LLM Agents*.
|
|
13
|
+
|
|
14
|
+
XiangqiBench asks an LLM agent to convert 119 composed xiangqi (Chinese chess) endgames into
|
|
15
|
+
checkmate against a Pikafish defender. The agent acts through a small command protocol under
|
|
16
|
+
per-turn budgets and is scored only on whether it actually delivers mate: finding the right
|
|
17
|
+
first move is not enough.
|
|
18
|
+
|
|
19
|
+
## Installation
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
pip install xiangqibench # OpenAI-compatible and Azure endpoints
|
|
23
|
+
pip install "xiangqibench[anthropic]" # adds the native Anthropic client
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
XiangqiBench requires Python 3.10 or later and a [Pikafish](https://github.com/official-pikafish/Pikafish)
|
|
27
|
+
binary for the defender:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
git clone https://github.com/official-pikafish/Pikafish && make -C Pikafish/src -j build
|
|
31
|
+
export PIKAFISH_PATH=$PWD/Pikafish/src/pikafish
|
|
32
|
+
xiangqibench doctor # checks the engine and prints its build and NNUE hash
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
The paper's defender was Pikafish `fd168f68` with the network whose sha256 begins `a2f41d4d`,
|
|
36
|
+
searched to depth 18 with one thread and a 256 MB hash. Upstream has since replaced that network,
|
|
37
|
+
and older builds cannot load the new one, so build the current Pikafish as shown above. Its
|
|
38
|
+
moves can differ from the paper's defender. Every record stores the engine build and the network
|
|
39
|
+
hash.
|
|
40
|
+
|
|
41
|
+
## Usage
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
export OPENAI_API_KEY=...
|
|
45
|
+
xiangqibench run --model gpt-5.5 --mode sighted --limit 5
|
|
46
|
+
xiangqibench score runs/
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Full runs are configured with a single YAML file; `xiangqibench init` writes a commented
|
|
50
|
+
example. API keys are read from the environment, never from the file.
|
|
51
|
+
|
|
52
|
+
```yaml
|
|
53
|
+
mode: restricted # sighted | restricted | S | S-NT | R-T | R
|
|
54
|
+
model:
|
|
55
|
+
name: qwen3-235b
|
|
56
|
+
provider: openai # openai | azure | openai-responses | anthropic
|
|
57
|
+
base_url: http://localhost:8000/v1
|
|
58
|
+
api_key_env: VLLM_API_KEY
|
|
59
|
+
run:
|
|
60
|
+
trials: 3
|
|
61
|
+
workers: 8
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
xiangqibench run -c my_run.yaml # resumable: re-running fills in missing trials
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Command-line flags override the file, and unknown keys are rejected.
|
|
69
|
+
|
|
70
|
+
### Modes
|
|
71
|
+
|
|
72
|
+
| Mode | Observation | Tools |
|
|
73
|
+
|---|---|---|
|
|
74
|
+
| `sighted` | Board, FEN, and legal moves after every ply | `view_board`, `simulate`, `get_legal_moves` |
|
|
75
|
+
| `restricted` | Starting position once, then move diffs only | none |
|
|
76
|
+
| `S`, `S-NT`, `R-T`, `R` | Observation ablations: state push (S/R) × tool access (T/NT) | as named |
|
|
77
|
+
|
|
78
|
+
`sighted` and `restricted` are the paper's two settings. `xiangqibench prompt --mode <mode>`
|
|
79
|
+
prints the exact system prompt for any mode.
|
|
80
|
+
|
|
81
|
+
### Scoring
|
|
82
|
+
|
|
83
|
+
`xiangqibench score` reports pass@k and pass^k over the earliest three scored trials per
|
|
84
|
+
(model, mode, case), with 95% case-bootstrap intervals using the paper's seeds. Trials cut short
|
|
85
|
+
by infrastructure errors are excluded and re-run automatically. Runs that change the standard
|
|
86
|
+
budgets or defender settings are marked `standard: false`.
|
|
87
|
+
|
|
88
|
+
### Python API
|
|
89
|
+
|
|
90
|
+
```python
|
|
91
|
+
from xiangqibench import load_config
|
|
92
|
+
from xiangqibench.runner import run_suite
|
|
93
|
+
|
|
94
|
+
report = run_suite(load_config("my_run.yaml"))
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Any object with a `name` attribute and a `complete(messages) -> Completion` method can be
|
|
98
|
+
evaluated as an agent; see `xiangqibench.runner.play_trial`.
|
|
99
|
+
|
|
100
|
+
## Reproducibility
|
|
101
|
+
|
|
102
|
+
Every trial is stored as one JSON record with the full message history, the move list, the
|
|
103
|
+
resolved configuration, and the defender's identity, including which backend chose each
|
|
104
|
+
defender move. The test suite replays archived trials from the paper against this code and checks
|
|
105
|
+
every environment message and verdict (`xiangqibench.replay`). Known differences from the code
|
|
106
|
+
that produced the paper's archive are listed in the [changelog](CHANGELOG.md).
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
pip install -e ".[dev]" && pytest
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
## Citation
|
|
113
|
+
|
|
114
|
+
```bibtex
|
|
115
|
+
@article{xiangqibench2026,
|
|
116
|
+
title = {Finding the Move Is Not Winning the Game: {XiangqiBench} for Closed-Loop
|
|
117
|
+
Evaluation of {LLM} Agents},
|
|
118
|
+
author = {XiangqiBench authors},
|
|
119
|
+
year = {2026},
|
|
120
|
+
url = {https://github.com/floatai/xiangqibench}
|
|
121
|
+
}
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
## License
|
|
125
|
+
|
|
126
|
+
The code is released under the [MIT License](LICENSE). The historical positions are in the
|
|
127
|
+
public domain. Pikafish is licensed under GPL-3.0; it is not distributed with this package and
|
|
128
|
+
runs as a separate process.
|
|
129
|
+
|
|
130
|
+
[paper]: https://arxiv.org/abs/XXXX.XXXXX
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.24"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "xiangqibench"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Tool-grounded xiangqi endgames for evaluating LLM agents against a Pikafish defender."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
authors = [{ name = "XiangqiBench authors" }]
|
|
14
|
+
keywords = ["llm", "agents", "benchmark", "evaluation", "xiangqi", "chinese-chess", "pikafish"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Intended Audience :: Science/Research",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Programming Language :: Python :: 3.13",
|
|
25
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
26
|
+
"Topic :: Games/Entertainment :: Board Games",
|
|
27
|
+
]
|
|
28
|
+
dependencies = [
|
|
29
|
+
"cchess>=1.25.5,<1.26",
|
|
30
|
+
"numpy>=1.24",
|
|
31
|
+
"pyyaml>=6.0",
|
|
32
|
+
"openai>=1.60",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
[project.optional-dependencies]
|
|
36
|
+
anthropic = ["anthropic>=0.40"]
|
|
37
|
+
all = ["anthropic>=0.40"]
|
|
38
|
+
dev = ["pytest>=8", "pytest-cov>=5", "ruff>=0.6", "mypy>=1.10", "types-PyYAML", "build>=1.2", "twine>=5"]
|
|
39
|
+
|
|
40
|
+
[project.scripts]
|
|
41
|
+
xiangqibench = "xiangqibench.cli:main"
|
|
42
|
+
|
|
43
|
+
[project.urls]
|
|
44
|
+
Homepage = "https://github.com/floatai/xiangqibench"
|
|
45
|
+
Repository = "https://github.com/floatai/xiangqibench"
|
|
46
|
+
Issues = "https://github.com/floatai/xiangqibench/issues"
|
|
47
|
+
Changelog = "https://github.com/floatai/xiangqibench/blob/main/CHANGELOG.md"
|
|
48
|
+
|
|
49
|
+
[tool.hatch.version]
|
|
50
|
+
path = "src/xiangqibench/__init__.py"
|
|
51
|
+
|
|
52
|
+
[tool.hatch.build.targets.wheel]
|
|
53
|
+
packages = ["src/xiangqibench"]
|
|
54
|
+
|
|
55
|
+
[tool.hatch.build.targets.sdist]
|
|
56
|
+
include = ["src/xiangqibench", "tests", "README.md", "CHANGELOG.md", "CITATION.cff", "LICENSE", "DATA_CARD.md"]
|
|
57
|
+
|
|
58
|
+
[tool.pytest.ini_options]
|
|
59
|
+
testpaths = ["tests"]
|
|
60
|
+
addopts = "-ra"
|
|
61
|
+
markers = ["engine: requires a Pikafish binary (set PIKAFISH_PATH)"]
|
|
62
|
+
|
|
63
|
+
[tool.ruff]
|
|
64
|
+
line-length = 110
|
|
65
|
+
target-version = "py310"
|
|
66
|
+
src = ["src", "tests"]
|
|
67
|
+
|
|
68
|
+
[tool.ruff.lint]
|
|
69
|
+
select = ["E", "F", "W", "I", "B", "UP", "SIM", "RUF"]
|
|
70
|
+
ignore = ["RUF001", "RUF002", "RUF003", "E501", "SIM105"]
|
|
71
|
+
|
|
72
|
+
[tool.ruff.lint.per-file-ignores]
|
|
73
|
+
"tests/*" = ["B011"]
|
|
74
|
+
|
|
75
|
+
[tool.mypy]
|
|
76
|
+
# numpy's stubs need >= 3.12; runtime 3.10 compatibility is enforced by ruff (target py310) and CI.
|
|
77
|
+
python_version = "3.12"
|
|
78
|
+
ignore_missing_imports = true
|
|
79
|
+
warn_unused_ignores = true
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""XiangqiBench: tool-grounded xiangqi endgames for evaluating LLM agents."""
|
|
2
|
+
|
|
3
|
+
__version__ = "0.1.0"
|
|
4
|
+
|
|
5
|
+
from xiangqibench.cases import EndgameCase, load_cases
|
|
6
|
+
from xiangqibench.config import Config, load_config
|
|
7
|
+
from xiangqibench.modes import MODES, get_mode
|
|
8
|
+
|
|
9
|
+
__all__ = ["MODES", "Config", "EndgameCase", "__version__", "get_mode", "load_cases", "load_config"]
|