xiangqibench 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. xiangqibench-0.1.0/.gitignore +34 -0
  2. xiangqibench-0.1.0/CHANGELOG.md +59 -0
  3. xiangqibench-0.1.0/CITATION.cff +22 -0
  4. xiangqibench-0.1.0/DATA_CARD.md +92 -0
  5. xiangqibench-0.1.0/LICENSE +21 -0
  6. xiangqibench-0.1.0/PKG-INFO +172 -0
  7. xiangqibench-0.1.0/README.md +130 -0
  8. xiangqibench-0.1.0/pyproject.toml +79 -0
  9. xiangqibench-0.1.0/src/xiangqibench/__init__.py +9 -0
  10. xiangqibench-0.1.0/src/xiangqibench/cases.py +123 -0
  11. xiangqibench-0.1.0/src/xiangqibench/cli.py +283 -0
  12. xiangqibench-0.1.0/src/xiangqibench/config.py +203 -0
  13. xiangqibench-0.1.0/src/xiangqibench/data/cases.jsonl +119 -0
  14. xiangqibench-0.1.0/src/xiangqibench/data/splits/ablation-gemini-3.1-pro.txt +21 -0
  15. xiangqibench-0.1.0/src/xiangqibench/data/splits/ablation-gpt-5.5.txt +22 -0
  16. xiangqibench-0.1.0/src/xiangqibench/defender/__init__.py +18 -0
  17. xiangqibench-0.1.0/src/xiangqibench/defender/defender.py +124 -0
  18. xiangqibench-0.1.0/src/xiangqibench/defender/engine.py +298 -0
  19. xiangqibench-0.1.0/src/xiangqibench/defender/search.py +201 -0
  20. xiangqibench-0.1.0/src/xiangqibench/example_config.yaml +63 -0
  21. xiangqibench-0.1.0/src/xiangqibench/harness/__init__.py +6 -0
  22. xiangqibench-0.1.0/src/xiangqibench/harness/actions.py +265 -0
  23. xiangqibench-0.1.0/src/xiangqibench/harness/adapter.py +201 -0
  24. xiangqibench-0.1.0/src/xiangqibench/harness/config.py +62 -0
  25. xiangqibench-0.1.0/src/xiangqibench/harness/harness.py +239 -0
  26. xiangqibench-0.1.0/src/xiangqibench/harness/player_state.py +119 -0
  27. xiangqibench-0.1.0/src/xiangqibench/harness/prompts.py +201 -0
  28. xiangqibench-0.1.0/src/xiangqibench/harness/protocol.py +137 -0
  29. xiangqibench-0.1.0/src/xiangqibench/harness/record.py +163 -0
  30. xiangqibench-0.1.0/src/xiangqibench/harness/simulator.py +213 -0
  31. xiangqibench-0.1.0/src/xiangqibench/harness/templates/restricted.txt +93 -0
  32. xiangqibench-0.1.0/src/xiangqibench/harness/templates/sighted.txt +142 -0
  33. xiangqibench-0.1.0/src/xiangqibench/llm/__init__.py +4 -0
  34. xiangqibench-0.1.0/src/xiangqibench/llm/base.py +43 -0
  35. xiangqibench-0.1.0/src/xiangqibench/llm/providers.py +191 -0
  36. xiangqibench-0.1.0/src/xiangqibench/modes.py +74 -0
  37. xiangqibench-0.1.0/src/xiangqibench/py.typed +0 -0
  38. xiangqibench-0.1.0/src/xiangqibench/replay.py +193 -0
  39. xiangqibench-0.1.0/src/xiangqibench/rules/__init__.py +3 -0
  40. xiangqibench-0.1.0/src/xiangqibench/rules/xiangqi_env.py +252 -0
  41. xiangqibench-0.1.0/src/xiangqibench/runner.py +332 -0
  42. xiangqibench-0.1.0/src/xiangqibench/scoring.py +238 -0
  43. xiangqibench-0.1.0/tests/conftest.py +30 -0
  44. xiangqibench-0.1.0/tests/golden/legacy/repl-blind__check_flag.json.gz +0 -0
  45. xiangqibench-0.1.0/tests/golden/legacy/repl-sighted__check_flag.json.gz +0 -0
  46. xiangqibench-0.1.0/tests/golden/legacy/repl-sighted__early_no_command_budget.json.gz +0 -0
  47. xiangqibench-0.1.0/tests/golden/prompts/repl-abl-R-T.json +5 -0
  48. xiangqibench-0.1.0/tests/golden/prompts/repl-abl-R.json +5 -0
  49. xiangqibench-0.1.0/tests/golden/prompts/repl-abl-S-NT.json +5 -0
  50. xiangqibench-0.1.0/tests/golden/prompts/repl-abl-S.json +5 -0
  51. xiangqibench-0.1.0/tests/golden/prompts/repl-blind.json +5 -0
  52. xiangqibench-0.1.0/tests/golden/prompts/repl-sighted.json +5 -0
  53. xiangqibench-0.1.0/tests/golden/trials/repl-abl-R-T__tool.json.gz +0 -0
  54. xiangqibench-0.1.0/tests/golden/trials/repl-abl-R-T__win.json.gz +0 -0
  55. xiangqibench-0.1.0/tests/golden/trials/repl-abl-R__tool.json.gz +0 -0
  56. xiangqibench-0.1.0/tests/golden/trials/repl-abl-R__win.json.gz +0 -0
  57. xiangqibench-0.1.0/tests/golden/trials/repl-abl-S-NT__tool.json.gz +0 -0
  58. xiangqibench-0.1.0/tests/golden/trials/repl-abl-S-NT__win.json.gz +0 -0
  59. xiangqibench-0.1.0/tests/golden/trials/repl-abl-S__tool.json.gz +0 -0
  60. xiangqibench-0.1.0/tests/golden/trials/repl-abl-S__win.json.gz +0 -0
  61. xiangqibench-0.1.0/tests/golden/trials/repl-blind__invalid.json.gz +0 -0
  62. xiangqibench-0.1.0/tests/golden/trials/repl-blind__tool.json.gz +0 -0
  63. xiangqibench-0.1.0/tests/golden/trials/repl-blind__win.json.gz +0 -0
  64. xiangqibench-0.1.0/tests/golden/trials/repl-sighted__invalid.json.gz +0 -0
  65. xiangqibench-0.1.0/tests/golden/trials/repl-sighted__tool.json.gz +0 -0
  66. xiangqibench-0.1.0/tests/golden/trials/repl-sighted__win.json.gz +0 -0
  67. xiangqibench-0.1.0/tests/test_cli.py +45 -0
  68. xiangqibench-0.1.0/tests/test_config.py +60 -0
  69. xiangqibench-0.1.0/tests/test_conformance.py +102 -0
  70. xiangqibench-0.1.0/tests/test_engine.py +35 -0
  71. xiangqibench-0.1.0/tests/test_protocol.py +33 -0
  72. xiangqibench-0.1.0/tests/test_rules.py +76 -0
  73. xiangqibench-0.1.0/tests/test_runner.py +103 -0
  74. xiangqibench-0.1.0/tests/test_scoring.py +55 -0
@@ -0,0 +1,34 @@
1
+ # Secrets: keys belong in the environment, never in the repository
2
+ .env
3
+ .env.*
4
+ !.env.example
5
+ *.key
6
+ config/*.yaml
7
+ !config/*.example.yaml
8
+ *api*.yaml
9
+ !src/xiangqibench/example_config.yaml
10
+
11
+ # Run outputs
12
+ runs/
13
+ output/
14
+ *.log
15
+
16
+ # Python
17
+ __pycache__/
18
+ *.py[cod]
19
+ *.egg-info/
20
+ build/
21
+ dist/
22
+ .venv/
23
+ venv/
24
+
25
+ # Tooling
26
+ .pytest_cache/
27
+ .ruff_cache/
28
+ .mypy_cache/
29
+ .coverage
30
+ coverage.xml
31
+ htmlcov/
32
+ .DS_Store
33
+ .idea/
34
+ .vscode/
@@ -0,0 +1,59 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [0.1.0] — 2026-09-26
8
+
9
+ First public release. It packages the XiangqiBench protocol, the 119 positions, the Pikafish
10
+ defender, and the scoring used in the paper.
11
+
12
+ ### Added
13
+
14
+ - The `xiangqibench` package and CLI, with the subcommands `run`, `score`, `modes`, `prompt`,
15
+ `cases`, `doctor`, `init`, and `play`.
16
+ - Six evaluation modes, selected by `mode:` in the config file or by `--mode`:
17
+ - the paper settings `sighted` and `restricted`;
18
+ - the observation-ablation arms `S`, `S-NT`, `R-T`, and `R`.
19
+ - YAML configuration with `${VAR}` / `${VAR:-default}` expansion from the environment, strict
20
+ key validation, and command-line overrides.
21
+ - Model clients for Chat Completions (OpenAI, OpenRouter, vLLM, SGLang, or any compatible
22
+ server), Azure OpenAI, the OpenAI Responses API, and Anthropic. Calls are retried, and a
23
+ trial whose calls all fail writes no record.
24
+ - The Pikafish defender, with a per-position fallback to a depth-5 rule search. Each defender
25
+ move records its backend, and the engine id and NNUE sha256 are stored with every trial.
26
+ - A resumable, parallel suite runner. Each worker has its own engine, and records are written
27
+ atomically.
28
+ - Scoring: pass@k, pass^k, earliest-three selection per cell, and 95% case-bootstrap intervals
29
+ with the paper's seeds.
30
+ - `xiangqibench.replay.replay_record`, which re-verifies an archived trial against the current
31
+ code.
32
+ - Conformance tests:
33
+ - archived prompts for all six settings, byte for byte;
34
+ - replay of 14 archived trials;
35
+ - rules, protocol, config, scoring, runner, and CLI tests.
36
+
37
+ ### Fixed (relative to the code that produced the paper's archive)
38
+
39
+ These fixes change what future runs record. Archived results are scored as they were recorded.
40
+ The effect of each fix on the paper's analyses is described in the paper.
41
+
42
+ - **Check flag.** Feedback used to report `Check: No` after every move, including checking and
43
+ mating moves, and mates were labelled `stalemate`. The flag is now computed on the position
44
+ after the move, and mates are labelled `checkmate`. The trajectory's `in_check` is also set on
45
+ the final, game-ending move. Winners were never affected.
46
+ - **Engine failures.** An engine that failed during search, for example on a network it cannot
47
+ load, silently handed every defender move to the rule fallback. It now stops the run with an
48
+ error, and only positions the engine rejects use the fallback.
49
+ - **Repetition.** A threefold repetition in which one side checked on every one of its moves now
50
+ loses for that side (perpetual check) instead of being a draw.
51
+ - **Forfeit move list.** Forfeit turns appended a copy of the previous move to `moves`, so
52
+ `plies` was one too high for forfeits after at least one move. Forfeits now append nothing.
53
+ - **Statistics.** The simulation counter in `stats` was never incremented. All counters now use
54
+ their documented keys.
55
+ - **Defender timing.** The defender's `duration_ms` now includes the engine search time.
56
+ - **Record size.** Per-call prompt snapshots are off by default. The full message history is
57
+ still stored once per trial.
58
+
59
+ [0.1.0]: https://github.com/floatai/xiangqibench/releases/tag/v0.1.0
@@ -0,0 +1,22 @@
1
+ cff-version: 1.2.0
2
+ message: "If you use XiangqiBench, please cite the paper below."
3
+ title: "XiangqiBench"
4
+ type: software
5
+ authors:
6
+ - name: "XiangqiBench authors"
7
+ version: 0.1.0
8
+ date-released: 2026-09-26
9
+ license: MIT
10
+ repository-code: "https://github.com/floatai/xiangqibench"
11
+ keywords:
12
+ - llm-agents
13
+ - benchmark
14
+ - xiangqi
15
+ - chinese-chess
16
+ preferred-citation:
17
+ type: article
18
+ title: "Finding the Move Is Not Winning the Game: XiangqiBench for Closed-Loop Evaluation of LLM Agents"
19
+ authors:
20
+ - name: "XiangqiBench authors"
21
+ year: 2026
22
+ url: "https://github.com/floatai/xiangqibench"
@@ -0,0 +1,92 @@
1
+ # Data card: XiangqiBench positions
2
+
3
+ ## Summary
4
+
5
+ `src/xiangqibench/data/cases.jsonl` holds 119 composed xiangqi endgames. In every one, Red is to
6
+ move and can force checkmate. Load them with `xiangqibench.load_cases()` or list them with
7
+ `xiangqibench cases`.
8
+
9
+ | | |
10
+ |---|---|
11
+ | Positions | 119 (116 *Shi Qing Ya Qu*, 3 *Jianghu*) |
12
+ | Side to move | Red in all positions (the agent plays Red) |
13
+ | Stored mate distance | 3–11 plies, median 9 |
14
+ | Categories | 106 `threatmate_race`, 13 `forced_mate` |
15
+ | Admission | 117 confirmed by Pikafish (`verified_by: engine`), 2 by a checks-only mate search (`solver_proof`) |
16
+ | sha256 | `3cb4e29148f82cb6d9a881e28336676549225cfe86dd8fb947ccf580f1633d42` |
17
+
18
+ ## Provenance
19
+
20
+ - ***Shi Qing Ya Qu*** (适情雅趣) is a Ming-dynasty treatise of composed endgames, printed in
21
+ 1570. Digitized entries were taken from
22
+ [xqipu.com](https://www.xqipu.com/canjugupu/1547). Case ids are
23
+ `xq_shi_qing_ya_qu_<n>`, and `raw_index` is the entry's position in that digitization.
24
+ - ***Jianghu*** (江湖) is a classical collection of street endgames. Case ids are
25
+ `xq_jianghu_endgames_<n>`.
26
+
27
+ Both sources are several centuries old and are in the public domain. The historical solution
28
+ text is not included, is never shown to agents, and is not used for scoring. They may still
29
+ appear in pre-training corpora, so first-move agreement can partly reflect recall.
30
+
31
+ ## Admission
32
+
33
+ Source solutions were not trusted. Each position was screened with a single-threaded,
34
+ time-limited Pikafish search and admitted when the engine reported a forced mate for Red. The
35
+ engine's mate distance and preferred first move are stored as `engine_mate_plies` and
36
+ `first_winning_move`.
37
+
38
+ Pikafish does not confirm a mate in two compositions, `xq_jianghu_endgames_084` and
39
+ `xq_jianghu_endgames_327`. They were admitted by a checks-only mate search and carry
40
+ `verified_by: solver_proof`. All 119 positions were also checked by hand.
41
+ Admission relies on bounded search, so the positions are *search-supported* rather than formally
42
+ proved.
43
+
44
+ **Categories.** The same search was run with Black to move. If Black then also has a forced mate
45
+ (`defender_threat_in_plies`), the position is a `threatmate_race`: a slow Red move can let Black
46
+ mate first. Otherwise it is a `forced_mate`.
47
+
48
+ ## Fields
49
+
50
+ | Field | Meaning |
51
+ |---|---|
52
+ | `id` | Stable case id |
53
+ | `source`, `raw_index`, `name` | Source collection, index in the digitization, original title |
54
+ | `fen` | Start position (FEN, Red to move) |
55
+ | `challenger` | Side played by the agent (`red`) |
56
+ | `category` | `threatmate_race` or `forced_mate` |
57
+ | `win_in_plies` | Stored mate distance (engine or solver) |
58
+ | `engine_mate_plies` | Pikafish mate distance, or `null` if the engine does not confirm a mate |
59
+ | `defender_threat_in_plies` | Black's mate distance if Black were to move, or `null` |
60
+ | `first_winning_move` | Reference root move from the engine (used only for first-move analysis) |
61
+ | `legal_move_count` | Number of legal Red moves at the root |
62
+ | `verified_by` | `engine` or `solver_proof` |
63
+ | `difficulty_score`, `tier`, `piece_theme`, `tags` | Descriptive metadata; not used for scoring |
64
+
65
+ The stored distances are admission metadata. They do not limit the game: every trial may run up
66
+ to 40 plies.
67
+
68
+ ## Splits
69
+
70
+ | Split | Cases | Use |
71
+ |---|---|---|
72
+ | `main` | 119 | Leaderboard (paper settings `sighted` and `restricted`) |
73
+ | `ablation-gemini-3.1-pro` | 20 | Observation ablation for Gemini 3.1 Pro |
74
+ | `ablation-gpt-5.5` | 21 | Observation ablation for GPT-5.5 |
75
+
76
+ The ablation subsets are **model-specific**. Each contains positions that the model won at
77
+ least once in the paper's Sighted trials, so the ablation measures how much of a demonstrated
78
+ ability survives each observation change.
79
+
80
+ - For GPT-5.5 the subset contains every such position.
81
+ - For Gemini 3.1 Pro it is a stratified sample by mate distance, with a quota
82
+ {5: 3, 7: 6, 9: 5, 11: 6} and seed 20260926.
83
+
84
+ Ablation numbers are therefore not comparable across models or with the `main` split.
85
+
86
+ ## Engine coverage
87
+
88
+ Pikafish refuses to search two positions reachable in the benchmark: the start positions of
89
+ `xq_jianghu_endgames_084` and `xq_jianghu_endgames_327`, and the positions after their first
90
+ move. For such positions, the defender falls back to a depth-5 rule search. Each defender move
91
+ records which backend chose it, so the share of rule-chosen moves can be computed from any set
92
+ of records.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 FloatAI
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,172 @@
1
+ Metadata-Version: 2.5
2
+ Name: xiangqibench
3
+ Version: 0.1.0
4
+ Summary: Tool-grounded xiangqi endgames for evaluating LLM agents against a Pikafish defender.
5
+ Project-URL: Homepage, https://github.com/floatai/xiangqibench
6
+ Project-URL: Repository, https://github.com/floatai/xiangqibench
7
+ Project-URL: Issues, https://github.com/floatai/xiangqibench/issues
8
+ Project-URL: Changelog, https://github.com/floatai/xiangqibench/blob/main/CHANGELOG.md
9
+ Author: XiangqiBench authors
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: agents,benchmark,chinese-chess,evaluation,llm,pikafish,xiangqi
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Topic :: Games/Entertainment :: Board Games
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Requires-Python: >=3.10
25
+ Requires-Dist: cchess<1.26,>=1.25.5
26
+ Requires-Dist: numpy>=1.24
27
+ Requires-Dist: openai>=1.60
28
+ Requires-Dist: pyyaml>=6.0
29
+ Provides-Extra: all
30
+ Requires-Dist: anthropic>=0.40; extra == 'all'
31
+ Provides-Extra: anthropic
32
+ Requires-Dist: anthropic>=0.40; extra == 'anthropic'
33
+ Provides-Extra: dev
34
+ Requires-Dist: build>=1.2; extra == 'dev'
35
+ Requires-Dist: mypy>=1.10; extra == 'dev'
36
+ Requires-Dist: pytest-cov>=5; extra == 'dev'
37
+ Requires-Dist: pytest>=8; extra == 'dev'
38
+ Requires-Dist: ruff>=0.6; extra == 'dev'
39
+ Requires-Dist: twine>=5; extra == 'dev'
40
+ Requires-Dist: types-pyyaml; extra == 'dev'
41
+ Description-Content-Type: text/markdown
42
+
43
+ # XiangqiBench
44
+
45
+ [![Paper](https://img.shields.io/badge/paper-arXiv-b31b1b.svg)][paper]
46
+ [![PyPI](https://img.shields.io/pypi/v/xiangqibench.svg)](https://pypi.org/project/xiangqibench/)
47
+ [![Python](https://img.shields.io/pypi/pyversions/xiangqibench.svg)](https://pypi.org/project/xiangqibench/)
48
+ [![CI](https://github.com/floatai/xiangqibench/actions/workflows/ci.yml/badge.svg)](https://github.com/floatai/xiangqibench/actions/workflows/ci.yml)
49
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
50
+
51
+ **[Paper][paper]** | **[Data card](DATA_CARD.md)** | **[Changelog](CHANGELOG.md)** | **[Citation](#citation)**
52
+
53
+ This repository contains the official implementation of *Finding the Move Is Not Winning the
54
+ Game: XiangqiBench for Closed-Loop Evaluation of LLM Agents*.
55
+
56
+ XiangqiBench asks an LLM agent to convert 119 composed xiangqi (Chinese chess) endgames into
57
+ checkmate against a Pikafish defender. The agent acts through a small command protocol under
58
+ per-turn budgets and is scored only on whether it actually delivers mate: finding the right
59
+ first move is not enough.
60
+
61
+ ## Installation
62
+
63
+ ```bash
64
+ pip install xiangqibench # OpenAI-compatible and Azure endpoints
65
+ pip install "xiangqibench[anthropic]" # adds the native Anthropic client
66
+ ```
67
+
68
+ XiangqiBench requires Python 3.10 or later and a [Pikafish](https://github.com/official-pikafish/Pikafish)
69
+ binary for the defender:
70
+
71
+ ```bash
72
+ git clone https://github.com/official-pikafish/Pikafish && make -C Pikafish/src -j build
73
+ export PIKAFISH_PATH=$PWD/Pikafish/src/pikafish
74
+ xiangqibench doctor # checks the engine and prints its build and NNUE hash
75
+ ```
76
+
77
+ The paper's defender was Pikafish `fd168f68` with the network whose sha256 begins `a2f41d4d`,
78
+ searched to depth 18 with one thread and a 256 MB hash. Upstream has since replaced that network,
79
+ and older builds cannot load the new one, so build the current Pikafish as shown above. Its
80
+ moves can differ from the paper's defender. Every record stores the engine build and the network
81
+ hash.
82
+
83
+ ## Usage
84
+
85
+ ```bash
86
+ export OPENAI_API_KEY=...
87
+ xiangqibench run --model gpt-5.5 --mode sighted --limit 5
88
+ xiangqibench score runs/
89
+ ```
90
+
91
+ Full runs are configured with a single YAML file; `xiangqibench init` writes a commented
92
+ example. API keys are read from the environment, never from the file.
93
+
94
+ ```yaml
95
+ mode: restricted # sighted | restricted | S | S-NT | R-T | R
96
+ model:
97
+ name: qwen3-235b
98
+ provider: openai # openai | azure | openai-responses | anthropic
99
+ base_url: http://localhost:8000/v1
100
+ api_key_env: VLLM_API_KEY
101
+ run:
102
+ trials: 3
103
+ workers: 8
104
+ ```
105
+
106
+ ```bash
107
+ xiangqibench run -c my_run.yaml # resumable: re-running fills in missing trials
108
+ ```
109
+
110
+ Command-line flags override the file, and unknown keys are rejected.
111
+
112
+ ### Modes
113
+
114
+ | Mode | Observation | Tools |
115
+ |---|---|---|
116
+ | `sighted` | Board, FEN, and legal moves after every ply | `view_board`, `simulate`, `get_legal_moves` |
117
+ | `restricted` | Starting position once, then move diffs only | none |
118
+ | `S`, `S-NT`, `R-T`, `R` | Observation ablations: state push (S/R) × tool access (T/NT) | as named |
119
+
120
+ `sighted` and `restricted` are the paper's two settings. `xiangqibench prompt --mode <mode>`
121
+ prints the exact system prompt for any mode.
122
+
123
+ ### Scoring
124
+
125
+ `xiangqibench score` reports pass@k and pass^k over the earliest three scored trials per
126
+ (model, mode, case), with 95% case-bootstrap intervals using the paper's seeds. Trials cut short
127
+ by infrastructure errors are excluded and re-run automatically. Runs that change the standard
128
+ budgets or defender settings are marked `standard: false`.
129
+
130
+ ### Python API
131
+
132
+ ```python
133
+ from xiangqibench import load_config
134
+ from xiangqibench.runner import run_suite
135
+
136
+ report = run_suite(load_config("my_run.yaml"))
137
+ ```
138
+
139
+ Any object with a `name` attribute and a `complete(messages) -> Completion` method can be
140
+ evaluated as an agent; see `xiangqibench.runner.play_trial`.
141
+
142
+ ## Reproducibility
143
+
144
+ Every trial is stored as one JSON record with the full message history, the move list, the
145
+ resolved configuration, and the defender's identity, including which backend chose each
146
+ defender move. The test suite replays archived trials from the paper against this code and checks
147
+ every environment message and verdict (`xiangqibench.replay`). Known differences from the code
148
+ that produced the paper's archive are listed in the [changelog](CHANGELOG.md).
149
+
150
+ ```bash
151
+ pip install -e ".[dev]" && pytest
152
+ ```
153
+
154
+ ## Citation
155
+
156
+ ```bibtex
157
+ @article{xiangqibench2026,
158
+ title = {Finding the Move Is Not Winning the Game: {XiangqiBench} for Closed-Loop
159
+ Evaluation of {LLM} Agents},
160
+ author = {XiangqiBench authors},
161
+ year = {2026},
162
+ url = {https://github.com/floatai/xiangqibench}
163
+ }
164
+ ```
165
+
166
+ ## License
167
+
168
+ The code is released under the [MIT License](LICENSE). The historical positions are in the
169
+ public domain. Pikafish is licensed under GPL-3.0; it is not distributed with this package and
170
+ runs as a separate process.
171
+
172
+ [paper]: https://arxiv.org/abs/XXXX.XXXXX
@@ -0,0 +1,130 @@
1
+ # XiangqiBench
2
+
3
+ [![Paper](https://img.shields.io/badge/paper-arXiv-b31b1b.svg)][paper]
4
+ [![PyPI](https://img.shields.io/pypi/v/xiangqibench.svg)](https://pypi.org/project/xiangqibench/)
5
+ [![Python](https://img.shields.io/pypi/pyversions/xiangqibench.svg)](https://pypi.org/project/xiangqibench/)
6
+ [![CI](https://github.com/floatai/xiangqibench/actions/workflows/ci.yml/badge.svg)](https://github.com/floatai/xiangqibench/actions/workflows/ci.yml)
7
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
8
+
9
+ **[Paper][paper]** | **[Data card](DATA_CARD.md)** | **[Changelog](CHANGELOG.md)** | **[Citation](#citation)**
10
+
11
+ This repository contains the official implementation of *Finding the Move Is Not Winning the
12
+ Game: XiangqiBench for Closed-Loop Evaluation of LLM Agents*.
13
+
14
+ XiangqiBench asks an LLM agent to convert 119 composed xiangqi (Chinese chess) endgames into
15
+ checkmate against a Pikafish defender. The agent acts through a small command protocol under
16
+ per-turn budgets and is scored only on whether it actually delivers mate: finding the right
17
+ first move is not enough.
18
+
19
+ ## Installation
20
+
21
+ ```bash
22
+ pip install xiangqibench # OpenAI-compatible and Azure endpoints
23
+ pip install "xiangqibench[anthropic]" # adds the native Anthropic client
24
+ ```
25
+
26
+ XiangqiBench requires Python 3.10 or later and a [Pikafish](https://github.com/official-pikafish/Pikafish)
27
+ binary for the defender:
28
+
29
+ ```bash
30
+ git clone https://github.com/official-pikafish/Pikafish && make -C Pikafish/src -j build
31
+ export PIKAFISH_PATH=$PWD/Pikafish/src/pikafish
32
+ xiangqibench doctor # checks the engine and prints its build and NNUE hash
33
+ ```
34
+
35
+ The paper's defender was Pikafish `fd168f68` with the network whose sha256 begins `a2f41d4d`,
36
+ searched to depth 18 with one thread and a 256 MB hash. Upstream has since replaced that network,
37
+ and older builds cannot load the new one, so build the current Pikafish as shown above. Its
38
+ moves can differ from the paper's defender. Every record stores the engine build and the network
39
+ hash.
40
+
41
+ ## Usage
42
+
43
+ ```bash
44
+ export OPENAI_API_KEY=...
45
+ xiangqibench run --model gpt-5.5 --mode sighted --limit 5
46
+ xiangqibench score runs/
47
+ ```
48
+
49
+ Full runs are configured with a single YAML file; `xiangqibench init` writes a commented
50
+ example. API keys are read from the environment, never from the file.
51
+
52
+ ```yaml
53
+ mode: restricted # sighted | restricted | S | S-NT | R-T | R
54
+ model:
55
+ name: qwen3-235b
56
+ provider: openai # openai | azure | openai-responses | anthropic
57
+ base_url: http://localhost:8000/v1
58
+ api_key_env: VLLM_API_KEY
59
+ run:
60
+ trials: 3
61
+ workers: 8
62
+ ```
63
+
64
+ ```bash
65
+ xiangqibench run -c my_run.yaml # resumable: re-running fills in missing trials
66
+ ```
67
+
68
+ Command-line flags override the file, and unknown keys are rejected.
69
+
70
+ ### Modes
71
+
72
+ | Mode | Observation | Tools |
73
+ |---|---|---|
74
+ | `sighted` | Board, FEN, and legal moves after every ply | `view_board`, `simulate`, `get_legal_moves` |
75
+ | `restricted` | Starting position once, then move diffs only | none |
76
+ | `S`, `S-NT`, `R-T`, `R` | Observation ablations: state push (S/R) × tool access (T/NT) | as named |
77
+
78
+ `sighted` and `restricted` are the paper's two settings. `xiangqibench prompt --mode <mode>`
79
+ prints the exact system prompt for any mode.
80
+
81
+ ### Scoring
82
+
83
+ `xiangqibench score` reports pass@k and pass^k over the earliest three scored trials per
84
+ (model, mode, case), with 95% case-bootstrap intervals using the paper's seeds. Trials cut short
85
+ by infrastructure errors are excluded and re-run automatically. Runs that change the standard
86
+ budgets or defender settings are marked `standard: false`.
87
+
88
+ ### Python API
89
+
90
+ ```python
91
+ from xiangqibench import load_config
92
+ from xiangqibench.runner import run_suite
93
+
94
+ report = run_suite(load_config("my_run.yaml"))
95
+ ```
96
+
97
+ Any object with a `name` attribute and a `complete(messages) -> Completion` method can be
98
+ evaluated as an agent; see `xiangqibench.runner.play_trial`.
99
+
100
+ ## Reproducibility
101
+
102
+ Every trial is stored as one JSON record with the full message history, the move list, the
103
+ resolved configuration, and the defender's identity, including which backend chose each
104
+ defender move. The test suite replays archived trials from the paper against this code and checks
105
+ every environment message and verdict (`xiangqibench.replay`). Known differences from the code
106
+ that produced the paper's archive are listed in the [changelog](CHANGELOG.md).
107
+
108
+ ```bash
109
+ pip install -e ".[dev]" && pytest
110
+ ```
111
+
112
+ ## Citation
113
+
114
+ ```bibtex
115
+ @article{xiangqibench2026,
116
+ title = {Finding the Move Is Not Winning the Game: {XiangqiBench} for Closed-Loop
117
+ Evaluation of {LLM} Agents},
118
+ author = {XiangqiBench authors},
119
+ year = {2026},
120
+ url = {https://github.com/floatai/xiangqibench}
121
+ }
122
+ ```
123
+
124
+ ## License
125
+
126
+ The code is released under the [MIT License](LICENSE). The historical positions are in the
127
+ public domain. Pikafish is licensed under GPL-3.0; it is not distributed with this package and
128
+ runs as a separate process.
129
+
130
+ [paper]: https://arxiv.org/abs/XXXX.XXXXX
@@ -0,0 +1,79 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.24"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "xiangqibench"
7
+ dynamic = ["version"]
8
+ description = "Tool-grounded xiangqi endgames for evaluating LLM agents against a Pikafish defender."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.10"
13
+ authors = [{ name = "XiangqiBench authors" }]
14
+ keywords = ["llm", "agents", "benchmark", "evaluation", "xiangqi", "chinese-chess", "pikafish"]
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Intended Audience :: Science/Research",
18
+ "License :: OSI Approved :: MIT License",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.10",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Programming Language :: Python :: 3.13",
25
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
26
+ "Topic :: Games/Entertainment :: Board Games",
27
+ ]
28
+ dependencies = [
29
+ "cchess>=1.25.5,<1.26",
30
+ "numpy>=1.24",
31
+ "pyyaml>=6.0",
32
+ "openai>=1.60",
33
+ ]
34
+
35
+ [project.optional-dependencies]
36
+ anthropic = ["anthropic>=0.40"]
37
+ all = ["anthropic>=0.40"]
38
+ dev = ["pytest>=8", "pytest-cov>=5", "ruff>=0.6", "mypy>=1.10", "types-PyYAML", "build>=1.2", "twine>=5"]
39
+
40
+ [project.scripts]
41
+ xiangqibench = "xiangqibench.cli:main"
42
+
43
+ [project.urls]
44
+ Homepage = "https://github.com/floatai/xiangqibench"
45
+ Repository = "https://github.com/floatai/xiangqibench"
46
+ Issues = "https://github.com/floatai/xiangqibench/issues"
47
+ Changelog = "https://github.com/floatai/xiangqibench/blob/main/CHANGELOG.md"
48
+
49
+ [tool.hatch.version]
50
+ path = "src/xiangqibench/__init__.py"
51
+
52
+ [tool.hatch.build.targets.wheel]
53
+ packages = ["src/xiangqibench"]
54
+
55
+ [tool.hatch.build.targets.sdist]
56
+ include = ["src/xiangqibench", "tests", "README.md", "CHANGELOG.md", "CITATION.cff", "LICENSE", "DATA_CARD.md"]
57
+
58
+ [tool.pytest.ini_options]
59
+ testpaths = ["tests"]
60
+ addopts = "-ra"
61
+ markers = ["engine: requires a Pikafish binary (set PIKAFISH_PATH)"]
62
+
63
+ [tool.ruff]
64
+ line-length = 110
65
+ target-version = "py310"
66
+ src = ["src", "tests"]
67
+
68
+ [tool.ruff.lint]
69
+ select = ["E", "F", "W", "I", "B", "UP", "SIM", "RUF"]
70
+ ignore = ["RUF001", "RUF002", "RUF003", "E501", "SIM105"]
71
+
72
+ [tool.ruff.lint.per-file-ignores]
73
+ "tests/*" = ["B011"]
74
+
75
+ [tool.mypy]
76
+ # numpy's stubs need >= 3.12; runtime 3.10 compatibility is enforced by ruff (target py310) and CI.
77
+ python_version = "3.12"
78
+ ignore_missing_imports = true
79
+ warn_unused_ignores = true
@@ -0,0 +1,9 @@
1
+ """XiangqiBench: tool-grounded xiangqi endgames for evaluating LLM agents."""
2
+
3
+ __version__ = "0.1.0"
4
+
5
+ from xiangqibench.cases import EndgameCase, load_cases
6
+ from xiangqibench.config import Config, load_config
7
+ from xiangqibench.modes import MODES, get_mode
8
+
9
+ __all__ = ["MODES", "Config", "EndgameCase", "__version__", "get_mode", "load_cases", "load_config"]