royalelearn 0.5.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- royalelearn-0.5.4/LICENSE +21 -0
- royalelearn-0.5.4/PKG-INFO +20 -0
- royalelearn-0.5.4/README.md +40 -0
- royalelearn-0.5.4/pyproject.toml +75 -0
- royalelearn-0.5.4/royalelearn/__init__.py +60 -0
- royalelearn-0.5.4/royalelearn/__main__.py +15 -0
- royalelearn-0.5.4/royalelearn/api/__init__.py +121 -0
- royalelearn-0.5.4/royalelearn/api/advantage.py +61 -0
- royalelearn-0.5.4/royalelearn/api/buffer.py +179 -0
- royalelearn-0.5.4/royalelearn/api/checkpoint.py +132 -0
- royalelearn-0.5.4/royalelearn/api/ladder.py +179 -0
- royalelearn-0.5.4/royalelearn/api/metrics.py +101 -0
- royalelearn-0.5.4/royalelearn/api/policy.py +189 -0
- royalelearn-0.5.4/royalelearn/api/rollout.py +404 -0
- royalelearn-0.5.4/royalelearn/api/schedule.py +59 -0
- royalelearn-0.5.4/royalelearn/api/update.py +202 -0
- royalelearn-0.5.4/royalelearn/checkpoint.py +547 -0
- royalelearn-0.5.4/royalelearn/cli.py +753 -0
- royalelearn-0.5.4/royalelearn/config.py +1368 -0
- royalelearn-0.5.4/royalelearn/coordinator.py +2908 -0
- royalelearn-0.5.4/royalelearn/determinism.py +146 -0
- royalelearn-0.5.4/royalelearn/errors.py +106 -0
- royalelearn-0.5.4/royalelearn/extensions.py +591 -0
- royalelearn-0.5.4/royalelearn/identity.py +798 -0
- royalelearn-0.5.4/royalelearn/ladder/__init__.py +56 -0
- royalelearn-0.5.4/royalelearn/ladder/actors.py +189 -0
- royalelearn-0.5.4/royalelearn/ladder/evaluate.py +404 -0
- royalelearn-0.5.4/royalelearn/ladder/eviction.py +84 -0
- royalelearn-0.5.4/royalelearn/ladder/farm.py +409 -0
- royalelearn-0.5.4/royalelearn/ladder/gate.py +375 -0
- royalelearn-0.5.4/royalelearn/ladder/matchmaker.py +498 -0
- royalelearn-0.5.4/royalelearn/ladder/pool.py +416 -0
- royalelearn-0.5.4/royalelearn/ladder/rating.py +445 -0
- royalelearn-0.5.4/royalelearn/ladder/results.py +345 -0
- royalelearn-0.5.4/royalelearn/ladder/seat_decks.py +66 -0
- royalelearn-0.5.4/royalelearn/ladder/snapshots.py +267 -0
- royalelearn-0.5.4/royalelearn/learn/__init__.py +97 -0
- royalelearn-0.5.4/royalelearn/learn/actor_critic.py +328 -0
- royalelearn-0.5.4/royalelearn/learn/buffer.py +891 -0
- royalelearn-0.5.4/royalelearn/learn/decode.py +125 -0
- royalelearn-0.5.4/royalelearn/learn/distribution.py +277 -0
- royalelearn-0.5.4/royalelearn/learn/freeze.py +92 -0
- royalelearn-0.5.4/royalelearn/learn/gae.py +192 -0
- royalelearn-0.5.4/royalelearn/learn/inference.py +622 -0
- royalelearn-0.5.4/royalelearn/learn/nets.py +690 -0
- royalelearn-0.5.4/royalelearn/learn/ppo.py +1816 -0
- royalelearn-0.5.4/royalelearn/learn/returns.py +184 -0
- royalelearn-0.5.4/royalelearn/learn/rows.py +63 -0
- royalelearn-0.5.4/royalelearn/learn/schedules.py +299 -0
- royalelearn-0.5.4/royalelearn/learner.py +600 -0
- royalelearn-0.5.4/royalelearn/metrics/__init__.py +34 -0
- royalelearn-0.5.4/royalelearn/metrics/alarms.py +627 -0
- royalelearn-0.5.4/royalelearn/metrics/behaviour.py +231 -0
- royalelearn-0.5.4/royalelearn/metrics/bundle.py +201 -0
- royalelearn-0.5.4/royalelearn/metrics/records.py +647 -0
- royalelearn-0.5.4/royalelearn/metrics/schema.py +1166 -0
- royalelearn-0.5.4/royalelearn/metrics/sinks.py +373 -0
- royalelearn-0.5.4/royalelearn/metrics/viser_sink.py +432 -0
- royalelearn-0.5.4/royalelearn/metrics/wandb_sink.py +141 -0
- royalelearn-0.5.4/royalelearn/obs_layout.py +136 -0
- royalelearn-0.5.4/royalelearn/rewards.py +466 -0
- royalelearn-0.5.4/royalelearn/rollout/__init__.py +37 -0
- royalelearn-0.5.4/royalelearn/rollout/codec.py +661 -0
- royalelearn-0.5.4/royalelearn/rollout/envspec.py +527 -0
- royalelearn-0.5.4/royalelearn/rollout/farm.py +560 -0
- royalelearn-0.5.4/royalelearn/rollout/inline.py +1626 -0
- royalelearn-0.5.4/royalelearn/rollout/layout.py +630 -0
- royalelearn-0.5.4/royalelearn/rollout/plan.py +266 -0
- royalelearn-0.5.4/royalelearn/rollout/preflight.py +787 -0
- royalelearn-0.5.4/royalelearn/rollout/scripted.py +151 -0
- royalelearn-0.5.4/royalelearn/rollout/worker.py +334 -0
- royalelearn-0.5.4/royalelearn/seeding.py +152 -0
- royalelearn-0.5.4/royalelearn/testing.py +686 -0
- royalelearn-0.5.4/royalelearn/version.py +42 -0
- royalelearn-0.5.4/royalelearn.egg-info/PKG-INFO +20 -0
- royalelearn-0.5.4/royalelearn.egg-info/SOURCES.txt +165 -0
- royalelearn-0.5.4/royalelearn.egg-info/dependency_links.txt +1 -0
- royalelearn-0.5.4/royalelearn.egg-info/entry_points.txt +2 -0
- royalelearn-0.5.4/royalelearn.egg-info/requires.txt +15 -0
- royalelearn-0.5.4/royalelearn.egg-info/top_level.txt +1 -0
- royalelearn-0.5.4/setup.cfg +4 -0
- royalelearn-0.5.4/tests/test_ability_buttons.py +152 -0
- royalelearn-0.5.4/tests/test_action_layout.py +109 -0
- royalelearn-0.5.4/tests/test_actor_terms.py +456 -0
- royalelearn-0.5.4/tests/test_adam_floor.py +114 -0
- royalelearn-0.5.4/tests/test_alarm_plants.py +228 -0
- royalelearn-0.5.4/tests/test_alarms.py +611 -0
- royalelearn-0.5.4/tests/test_bench_report.py +108 -0
- royalelearn-0.5.4/tests/test_buffer.py +898 -0
- royalelearn-0.5.4/tests/test_card_identity.py +473 -0
- royalelearn-0.5.4/tests/test_checkpoint.py +394 -0
- royalelearn-0.5.4/tests/test_codec.py +571 -0
- royalelearn-0.5.4/tests/test_config.py +472 -0
- royalelearn-0.5.4/tests/test_control_file.py +93 -0
- royalelearn-0.5.4/tests/test_coordinator.py +945 -0
- royalelearn-0.5.4/tests/test_decode.py +169 -0
- royalelearn-0.5.4/tests/test_dirty_sources.py +202 -0
- royalelearn-0.5.4/tests/test_distribution.py +365 -0
- royalelearn-0.5.4/tests/test_elixir_pricing.py +418 -0
- royalelearn-0.5.4/tests/test_engine_binary_identity.py +223 -0
- royalelearn-0.5.4/tests/test_engine_contract.py +159 -0
- royalelearn-0.5.4/tests/test_env_contract.py +297 -0
- royalelearn-0.5.4/tests/test_env_value_digest.py +133 -0
- royalelearn-0.5.4/tests/test_eval_actors.py +128 -0
- royalelearn-0.5.4/tests/test_eval_dispatch.py +130 -0
- royalelearn-0.5.4/tests/test_eval_farm.py +341 -0
- royalelearn-0.5.4/tests/test_eviction.py +166 -0
- royalelearn-0.5.4/tests/test_examples.py +49 -0
- royalelearn-0.5.4/tests/test_extension_example.py +130 -0
- royalelearn-0.5.4/tests/test_extensions.py +486 -0
- royalelearn-0.5.4/tests/test_factored_head.py +336 -0
- royalelearn-0.5.4/tests/test_farm_shutdown.py +246 -0
- royalelearn-0.5.4/tests/test_fill_policy.py +250 -0
- royalelearn-0.5.4/tests/test_frame_stack.py +149 -0
- royalelearn-0.5.4/tests/test_gae.py +318 -0
- royalelearn-0.5.4/tests/test_game_over_drive.py +30 -0
- royalelearn-0.5.4/tests/test_gate.py +820 -0
- royalelearn-0.5.4/tests/test_hold_lift.py +251 -0
- royalelearn-0.5.4/tests/test_housekeeping_failures.py +407 -0
- royalelearn-0.5.4/tests/test_identity.py +282 -0
- royalelearn-0.5.4/tests/test_inference.py +395 -0
- royalelearn-0.5.4/tests/test_ladder_probe.py +472 -0
- royalelearn-0.5.4/tests/test_layout.py +337 -0
- royalelearn-0.5.4/tests/test_learner.py +545 -0
- royalelearn-0.5.4/tests/test_manual_eval_log.py +60 -0
- royalelearn-0.5.4/tests/test_matchmaker.py +582 -0
- royalelearn-0.5.4/tests/test_metric_names.py +315 -0
- royalelearn-0.5.4/tests/test_metrics.py +812 -0
- royalelearn-0.5.4/tests/test_nets.py +377 -0
- royalelearn-0.5.4/tests/test_no_global_rng.py +127 -0
- royalelearn-0.5.4/tests/test_obs_layout.py +294 -0
- royalelearn-0.5.4/tests/test_package.py +103 -0
- royalelearn-0.5.4/tests/test_policy_probe.py +170 -0
- royalelearn-0.5.4/tests/test_pool_naming.py +116 -0
- royalelearn-0.5.4/tests/test_ppo.py +1450 -0
- royalelearn-0.5.4/tests/test_provenance.py +136 -0
- royalelearn-0.5.4/tests/test_rating.py +266 -0
- royalelearn-0.5.4/tests/test_ratio_precision.py +75 -0
- royalelearn-0.5.4/tests/test_recorder_plumbing.py +206 -0
- royalelearn-0.5.4/tests/test_replay_episode.py +143 -0
- royalelearn-0.5.4/tests/test_results_log.py +151 -0
- royalelearn-0.5.4/tests/test_resume.py +383 -0
- royalelearn-0.5.4/tests/test_returns.py +183 -0
- royalelearn-0.5.4/tests/test_rewards.py +659 -0
- royalelearn-0.5.4/tests/test_rng_roundtrip.py +85 -0
- royalelearn-0.5.4/tests/test_rollout_alignment.py +363 -0
- royalelearn-0.5.4/tests/test_rollout_farm.py +156 -0
- royalelearn-0.5.4/tests/test_rollout_inline.py +429 -0
- royalelearn-0.5.4/tests/test_rollout_invariants.py +189 -0
- royalelearn-0.5.4/tests/test_rollout_reads.py +258 -0
- royalelearn-0.5.4/tests/test_rss_units.py +60 -0
- royalelearn-0.5.4/tests/test_run_directory.py +71 -0
- royalelearn-0.5.4/tests/test_schedules.py +218 -0
- royalelearn-0.5.4/tests/test_scripted_opponents.py +265 -0
- royalelearn-0.5.4/tests/test_seat_decks.py +214 -0
- royalelearn-0.5.4/tests/test_seed_snapshots.py +273 -0
- royalelearn-0.5.4/tests/test_seeding.py +133 -0
- royalelearn-0.5.4/tests/test_segment_names.py +92 -0
- royalelearn-0.5.4/tests/test_shaping_strength.py +303 -0
- royalelearn-0.5.4/tests/test_shell_fences.py +135 -0
- royalelearn-0.5.4/tests/test_snapshots.py +250 -0
- royalelearn-0.5.4/tests/test_spell_identity.py +278 -0
- royalelearn-0.5.4/tests/test_user_code_identity.py +242 -0
- royalelearn-0.5.4/tests/test_viser_protocol.py +124 -0
- royalelearn-0.5.4/tests/test_viser_sink.py +259 -0
- royalelearn-0.5.4/tests/test_vram_device.py +114 -0
- royalelearn-0.5.4/tests/test_worker_hygiene.py +617 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 RoyaleGym contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: royalelearn
|
|
3
|
+
Version: 0.5.4
|
|
4
|
+
Summary: RoyaleLearn: self-play PPO training harness for RoyaleGym Clash Royale environments
|
|
5
|
+
License: MIT
|
|
6
|
+
Requires-Python: >=3.12
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Requires-Dist: royalegym>=0.1.0
|
|
9
|
+
Requires-Dist: numpy>=2.0
|
|
10
|
+
Requires-Dist: msgspec>=0.18
|
|
11
|
+
Provides-Extra: torch
|
|
12
|
+
Requires-Dist: torch>=2.4; extra == "torch"
|
|
13
|
+
Requires-Dist: safetensors>=0.4; extra == "torch"
|
|
14
|
+
Provides-Extra: wandb
|
|
15
|
+
Requires-Dist: wandb; extra == "wandb"
|
|
16
|
+
Provides-Extra: dev
|
|
17
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
18
|
+
Requires-Dist: ruff>=0.16; extra == "dev"
|
|
19
|
+
Requires-Dist: hypothesis>=6.0; extra == "dev"
|
|
20
|
+
Dynamic: license-file
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
<p align="center"><img src="docs/media/logo.png" width="128" alt="The RoyaleLearn logo: a green crown shield with a white graduation cap on it, outlined in gold"></p><h1 align="center">RoyaleLearn</h1>
|
|
2
|
+
|
|
3
|
+
<p align="center"><a href="https://github.com/RoyaleGym/RoyaleLearn/actions/workflows/suite.yml"><img alt="CI" src="https://github.com/RoyaleGym/RoyaleLearn/actions/workflows/suite.yml/badge.svg"></a> <img alt="License" src="https://img.shields.io/github/license/RoyaleGym/RoyaleLearn?style=flat-square&color=555"> <img alt="Python" src="https://img.shields.io/badge/python-3.12%20%7C%203.13%20%7C%203.14-3776AB?style=flat-square&logo=python&logoColor=white"> <a href="https://royalegym.github.io/RoyaleGym/"><img alt="Docs" src="https://img.shields.io/badge/docs-royalegym.github.io-8957e5?style=flat-square&logo=readthedocs&logoColor=white"></a> <a href="https://discord.gg/4D2BS5JBHP"><img alt="Discord" src="https://img.shields.io/discord/1551699576304705647?style=flat-square&logo=discord&logoColor=white&label=discord&color=5865F2"></a> <img alt="Last commit" src="https://img.shields.io/github/last-commit/RoyaleGym/RoyaleLearn?style=flat-square&color=555"></p>
|
|
4
|
+
|
|
5
|
+
Train a Clash Royale bot in Python, on your own machine: you write the reward, it learns to win.
|
|
6
|
+
It is the trainer for the battles RoyaleGym runs, and it uses your NVIDIA graphics card.
|
|
7
|
+
|
|
8
|
+
## Install
|
|
9
|
+
|
|
10
|
+
pip install "royalegym[all]" --find-links https://github.com/RoyaleGym/RoyaleGym/releases/expanded_assets/v0.1.8
|
|
11
|
+
|
|
12
|
+
This is the `[learn]` part, for Python 3.12 to 3.14. On Windows with an NVIDIA card, install PyTorch
|
|
13
|
+
first: [Install](https://royalegym.github.io/RoyaleGym/install/), step 4.
|
|
14
|
+
|
|
15
|
+
## Try it
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
from royalegym import TowerHPReward, make_env
|
|
19
|
+
from royalelearn import Learner
|
|
20
|
+
|
|
21
|
+
def build_env():
|
|
22
|
+
return make_env(reward=TowerHPReward()) # what the bot is paid for
|
|
23
|
+
|
|
24
|
+
if __name__ == "__main__":
|
|
25
|
+
learner = Learner(build_env, save_dir="runs/my_bot")
|
|
26
|
+
learner.learn(total_steps=20_000) # counts all steps so far: raise it, run again to train more
|
|
27
|
+
learner.save("runs/my_bot/bot")
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
It prints one line per update. To watch your bot play, see
|
|
31
|
+
[Watch It Play](https://royalegym.github.io/RoyaleGym/quickstart/#5-watch-it-play).
|
|
32
|
+
|
|
33
|
+
## Next
|
|
34
|
+
|
|
35
|
+
- The docs: [royalegym.github.io/RoyaleGym](https://royalegym.github.io/RoyaleGym/)
|
|
36
|
+
- Quick Start: [the first bot, step by step](https://royalegym.github.io/RoyaleGym/quickstart/)
|
|
37
|
+
- Every setting: [RoyaleLearn](https://royalegym.github.io/RoyaleGym/resources/royalelearn/). How it works inside (Advanced): [the guide](https://royalegym.github.io/RoyaleGym/repos/royalelearn/guide/)
|
|
38
|
+
- Questions: [Discord](https://discord.gg/4D2BS5JBHP)
|
|
39
|
+
|
|
40
|
+
MIT licensed. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "royalelearn"
|
|
7
|
+
version = "0.5.4"
|
|
8
|
+
description = "RoyaleLearn: self-play PPO training harness for RoyaleGym Clash Royale environments"
|
|
9
|
+
requires-python = ">=3.12"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
# royalegym is not on PyPI: it is the sibling checkout ../RoyaleGym, installed editable before
|
|
12
|
+
# this package by the workspace recipe in README.md (RoyaleSim -> RoyaleGym -> RoyaleViser ->
|
|
13
|
+
# RoyaleLearn). pip resolves it from the venv; it never fetches it.
|
|
14
|
+
#
|
|
15
|
+
# The base install is deliberately torch-free: the ABCs, the config tree, the CLI's config and
|
|
16
|
+
# identity commands and the rollout worker all run on numpy and msgspec alone.
|
|
17
|
+
dependencies = [
|
|
18
|
+
"royalegym>=0.1.0",
|
|
19
|
+
"numpy>=2.0",
|
|
20
|
+
"msgspec>=0.18",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
[project.optional-dependencies]
|
|
24
|
+
# The learner: networks, the update, inference, and everything that touches a device.
|
|
25
|
+
# safetensors is what the snapshot archive and the checkpoint store weights in; nothing this
|
|
26
|
+
# package writes is pickled, so it is a requirement of the learner and not an option under it.
|
|
27
|
+
torch = [
|
|
28
|
+
"torch>=2.4",
|
|
29
|
+
"safetensors>=0.4",
|
|
30
|
+
]
|
|
31
|
+
# The optional metrics sink; the JSONL sink is always installed and is never optional.
|
|
32
|
+
wandb = [
|
|
33
|
+
"wandb",
|
|
34
|
+
]
|
|
35
|
+
dev = [
|
|
36
|
+
"pytest>=8.0",
|
|
37
|
+
"ruff>=0.16",
|
|
38
|
+
"hypothesis>=6.0",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[project.scripts]
|
|
42
|
+
royalelearn = "royalelearn.cli:main"
|
|
43
|
+
|
|
44
|
+
[tool.setuptools.packages.find]
|
|
45
|
+
include = ["royalelearn*"]
|
|
46
|
+
|
|
47
|
+
[tool.pytest.ini_options]
|
|
48
|
+
testpaths = ["tests"]
|
|
49
|
+
python_files = ["test_*.py"]
|
|
50
|
+
pythonpath = ["."]
|
|
51
|
+
# The default run is the fast suite. The slow tests and the ones needing a freshly built
|
|
52
|
+
# royalesim are selected explicitly, so a bare `pytest -q` stays a few seconds long.
|
|
53
|
+
addopts = "-ra --strict-markers -m \"not slow and not engine\""
|
|
54
|
+
markers = [
|
|
55
|
+
"slow: takes more than about five seconds",
|
|
56
|
+
"engine: needs a royalesim build that matches the data on disk",
|
|
57
|
+
]
|
|
58
|
+
|
|
59
|
+
[tool.ruff]
|
|
60
|
+
line-length = 100
|
|
61
|
+
target-version = "py312"
|
|
62
|
+
src = ["royalelearn", "tests"]
|
|
63
|
+
|
|
64
|
+
[tool.ruff.lint]
|
|
65
|
+
select = ["E", "F", "W", "I", "B", "UP", "SIM", "RUF"]
|
|
66
|
+
ignore = ["RUF001", "RUF002", "RUF003", "SIM108"]
|
|
67
|
+
|
|
68
|
+
[tool.ruff.lint.per-file-ignores]
|
|
69
|
+
# A VENDORED COPY, kept byte-identical to the original in the other three repos so the four can
|
|
70
|
+
# be diffed against each other. Reformatting it to this repo's line length would make it a local
|
|
71
|
+
# rewrite that only looks like a copy, which is the one thing it must not be.
|
|
72
|
+
"tests/_shell_fences.py" = ["E501"]
|
|
73
|
+
|
|
74
|
+
[tool.ruff.lint.isort]
|
|
75
|
+
known-first-party = ["royalelearn", "royalegym", "royalesim"]
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""royalelearn -- the self-play training harness for RoyaleGym environments.
|
|
2
|
+
|
|
3
|
+
Layers (each user-facing behaviour is an ABC in ``api/`` with a shipped default):
|
|
4
|
+
|
|
5
|
+
RolloutSource api/rollout.py ProcessRolloutSource, InlineRolloutSource
|
|
6
|
+
ActorCritic api/policy.py SeparateActorCritic, SharedTrunkActorCritic
|
|
7
|
+
ObsCodec api/buffer.py SpatialObsCodec
|
|
8
|
+
ExperienceBuffer api/buffer.py RectBuffer
|
|
9
|
+
Matchmaker/Rater api/ladder.py MixMatchmaker, BradleyTerryDavidsonRater
|
|
10
|
+
MetricsSink api/metrics.py JsonlSink, ConsoleSink, WandbSink
|
|
11
|
+
CheckpointStore api/checkpoint.py DirCheckpointStore
|
|
12
|
+
|
|
13
|
+
``LearningCoordinator`` (coordinator.py) is the only place the phases of an iteration are
|
|
14
|
+
ordered. ``docs/harness-spec.md`` specifies all of it; ``docs/design.md`` says why.
|
|
15
|
+
|
|
16
|
+
Importing this package pulls in nothing that needs torch: the ABCs use numpy and msgspec,
|
|
17
|
+
and the concrete learner lives behind the lazy attributes below. That is what lets
|
|
18
|
+
``royalelearn config``, ``royalelearn identity`` and ``royalelearn --help`` work in an
|
|
19
|
+
environment where torch is not installed. Asking for a name that does need torch raises an
|
|
20
|
+
ImportError naming the package that is missing.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import importlib
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
from .version import __version__, git_describe
|
|
29
|
+
|
|
30
|
+
# Public name -> (module, attribute). A None attribute exports the module itself.
|
|
31
|
+
# Resolution is deferred so that the import cost and the torch dependency of a name are
|
|
32
|
+
# paid only by the caller that asks for it.
|
|
33
|
+
_EXPORTS: dict[str, tuple[str, str | None]] = {
|
|
34
|
+
"Learner": (".learner", "Learner"),
|
|
35
|
+
"LearningCoordinator": (".coordinator", "LearningCoordinator"),
|
|
36
|
+
"RunConfig": (".config", "RunConfig"),
|
|
37
|
+
"load_config": (".config", "load_config"),
|
|
38
|
+
"api": (".api", None),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
__all__ = ["__version__", "git_describe", *sorted(_EXPORTS)]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def __getattr__(name: str) -> Any:
|
|
45
|
+
"""Resolve a public name on first use.
|
|
46
|
+
|
|
47
|
+
An unknown name is an AttributeError. A known name whose module cannot be imported
|
|
48
|
+
raises that import's own ImportError unchanged, because its message already names what
|
|
49
|
+
is missing and rewording it would hide which package to install.
|
|
50
|
+
"""
|
|
51
|
+
try:
|
|
52
|
+
module_name, attribute = _EXPORTS[name]
|
|
53
|
+
except KeyError:
|
|
54
|
+
raise AttributeError(f"module 'royalelearn' has no attribute {name!r}") from None
|
|
55
|
+
module = importlib.import_module(module_name, __name__)
|
|
56
|
+
return module if attribute is None else getattr(module, attribute)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def __dir__() -> list[str]:
|
|
60
|
+
return sorted(__all__)
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""``python -m royalelearn``.
|
|
2
|
+
|
|
3
|
+
The module body is one call. Everything the entry point has to do before torch is imported --
|
|
4
|
+
the deterministic cuBLAS workspace and the BLAS thread counts -- ``cli`` does as it is imported,
|
|
5
|
+
which is why this file imports it and nothing else.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import sys
|
|
11
|
+
|
|
12
|
+
from .cli import main
|
|
13
|
+
|
|
14
|
+
if __name__ == "__main__":
|
|
15
|
+
sys.exit(main())
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
"""Every ABC and struct the harness is written against.
|
|
2
|
+
|
|
3
|
+
This subpackage imports numpy and msgspec and nothing else that matters: torch appears in type
|
|
4
|
+
annotations only, behind ``TYPE_CHECKING``. That is what lets a rollout worker, the CLI's
|
|
5
|
+
config and identity commands, and ``import royalelearn`` itself run in an environment with no
|
|
6
|
+
torch installed -- and what keeps a worker's resident memory three hundred megabytes smaller
|
|
7
|
+
than the parent's.
|
|
8
|
+
|
|
9
|
+
The concrete implementations live outside ``api/`` and may import whatever they need:
|
|
10
|
+
|
|
11
|
+
RolloutSource rollout/farm.py, rollout/inline.py
|
|
12
|
+
ActorCritic learn/actor_critic.py
|
|
13
|
+
ObsCodec rollout/codec.py
|
|
14
|
+
ExperienceBuffer learn/buffer.py
|
|
15
|
+
AdvantageEstimator learn/gae.py
|
|
16
|
+
Update learn/ppo.py
|
|
17
|
+
Schedule learn/schedules.py
|
|
18
|
+
Matchmaker, Rater ladder/matchmaker.py, ladder/rating.py
|
|
19
|
+
MetricsSink, Alarm metrics/sinks.py, metrics/alarms.py
|
|
20
|
+
CheckpointStore checkpoint.py
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from .advantage import AdvantageEstimator, AdvantageStats
|
|
26
|
+
from .buffer import CodecTable, ExperienceBuffer, ObsCodec
|
|
27
|
+
from .checkpoint import Checkpointable, CheckpointStore, Manifest, RngState
|
|
28
|
+
from .ladder import (
|
|
29
|
+
ConditionResult,
|
|
30
|
+
EvictionPolicy,
|
|
31
|
+
GateDecision,
|
|
32
|
+
Matchmaker,
|
|
33
|
+
PromotionGate,
|
|
34
|
+
Rater,
|
|
35
|
+
RatingTable,
|
|
36
|
+
SnapshotStore,
|
|
37
|
+
)
|
|
38
|
+
from .metrics import Alarm, AlarmResult, MetricRow, MetricsSink, MetricValue
|
|
39
|
+
from .policy import (
|
|
40
|
+
ActionDistribution,
|
|
41
|
+
Actor,
|
|
42
|
+
ActorCritic,
|
|
43
|
+
ActResult,
|
|
44
|
+
BackpropResult,
|
|
45
|
+
Critic,
|
|
46
|
+
NetworkFactory,
|
|
47
|
+
ObsBatch,
|
|
48
|
+
)
|
|
49
|
+
from .rollout import (
|
|
50
|
+
Assignment,
|
|
51
|
+
Close,
|
|
52
|
+
Defer,
|
|
53
|
+
EnvSpec,
|
|
54
|
+
EpisodeRecord,
|
|
55
|
+
ObsKeySpec,
|
|
56
|
+
Plan,
|
|
57
|
+
RolloutRound,
|
|
58
|
+
RolloutSource,
|
|
59
|
+
SetState,
|
|
60
|
+
SlotPlan,
|
|
61
|
+
Spaces,
|
|
62
|
+
Step,
|
|
63
|
+
WorkerCommand,
|
|
64
|
+
WorkerFailure,
|
|
65
|
+
)
|
|
66
|
+
from .schedule import Schedule, ScheduleState
|
|
67
|
+
from .update import ActorLossTerm, ActorTermInputs, Update, UpdateResult
|
|
68
|
+
|
|
69
|
+
__all__ = [
|
|
70
|
+
"ActResult",
|
|
71
|
+
"ActionDistribution",
|
|
72
|
+
"Actor",
|
|
73
|
+
"ActorCritic",
|
|
74
|
+
"ActorLossTerm",
|
|
75
|
+
"ActorTermInputs",
|
|
76
|
+
"AdvantageEstimator",
|
|
77
|
+
"AdvantageStats",
|
|
78
|
+
"Alarm",
|
|
79
|
+
"AlarmResult",
|
|
80
|
+
"Assignment",
|
|
81
|
+
"BackpropResult",
|
|
82
|
+
"CheckpointStore",
|
|
83
|
+
"Checkpointable",
|
|
84
|
+
"Close",
|
|
85
|
+
"CodecTable",
|
|
86
|
+
"ConditionResult",
|
|
87
|
+
"Critic",
|
|
88
|
+
"Defer",
|
|
89
|
+
"EnvSpec",
|
|
90
|
+
"EpisodeRecord",
|
|
91
|
+
"EvictionPolicy",
|
|
92
|
+
"ExperienceBuffer",
|
|
93
|
+
"GateDecision",
|
|
94
|
+
"Manifest",
|
|
95
|
+
"Matchmaker",
|
|
96
|
+
"MetricRow",
|
|
97
|
+
"MetricValue",
|
|
98
|
+
"MetricsSink",
|
|
99
|
+
"NetworkFactory",
|
|
100
|
+
"ObsBatch",
|
|
101
|
+
"ObsCodec",
|
|
102
|
+
"ObsKeySpec",
|
|
103
|
+
"Plan",
|
|
104
|
+
"PromotionGate",
|
|
105
|
+
"Rater",
|
|
106
|
+
"RatingTable",
|
|
107
|
+
"RngState",
|
|
108
|
+
"RolloutRound",
|
|
109
|
+
"RolloutSource",
|
|
110
|
+
"Schedule",
|
|
111
|
+
"ScheduleState",
|
|
112
|
+
"SetState",
|
|
113
|
+
"SlotPlan",
|
|
114
|
+
"SnapshotStore",
|
|
115
|
+
"Spaces",
|
|
116
|
+
"Step",
|
|
117
|
+
"Update",
|
|
118
|
+
"UpdateResult",
|
|
119
|
+
"WorkerCommand",
|
|
120
|
+
"WorkerFailure",
|
|
121
|
+
]
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""How a return is turned into an advantage.
|
|
2
|
+
|
|
3
|
+
One ABC, because the estimator is the piece most likely to be replaced -- V-trace against a
|
|
4
|
+
frozen pool, a different lambda rule, a per-bucket normalisation -- and because the one thing it
|
|
5
|
+
must never do is carry its recursion across an episode boundary.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from abc import ABC, abstractmethod
|
|
11
|
+
from typing import TYPE_CHECKING
|
|
12
|
+
|
|
13
|
+
import msgspec
|
|
14
|
+
|
|
15
|
+
from .checkpoint import Checkpointable
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING: # pragma: no cover - annotations only
|
|
18
|
+
from torch import Tensor
|
|
19
|
+
|
|
20
|
+
__all__ = ["AdvantageEstimator", "AdvantageStats"]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class AdvantageStats(msgspec.Struct):
|
|
24
|
+
"""What the estimator saw, for the metric row.
|
|
25
|
+
|
|
26
|
+
``reward_scale`` is the divisor the return scaler applied and ``clipped_reward_frac`` is how
|
|
27
|
+
much of the batch the clip bound touched: both are how a scaled reward stays legible after
|
|
28
|
+
the scaling.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
raw_return_mean: float
|
|
32
|
+
raw_return_std: float
|
|
33
|
+
reward_scale: float
|
|
34
|
+
clipped_reward_frac: float
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class AdvantageEstimator(ABC, Checkpointable):
|
|
38
|
+
"""Rewards and values in, advantages and returns out."""
|
|
39
|
+
|
|
40
|
+
@abstractmethod
|
|
41
|
+
def compute(
|
|
42
|
+
self,
|
|
43
|
+
*,
|
|
44
|
+
rewards: Tensor,
|
|
45
|
+
values: Tensor,
|
|
46
|
+
final_values: Tensor,
|
|
47
|
+
terminated: Tensor,
|
|
48
|
+
truncated: Tensor,
|
|
49
|
+
trainable: Tensor,
|
|
50
|
+
gamma: float,
|
|
51
|
+
lam: float,
|
|
52
|
+
) -> tuple[Tensor, Tensor, AdvantageStats]:
|
|
53
|
+
"""``rewards``/``terminated``/``truncated``/``trainable`` are ``(T, R)``; ``values`` is
|
|
54
|
+
``(T+1, R)``; ``final_values`` is ``(T, R)`` and is read only where truncated.
|
|
55
|
+
Returns ``(advantages (T, R), returns (T, R), stats)``.
|
|
56
|
+
|
|
57
|
+
Implementers MUST bootstrap a terminated cell from 0 and a truncated cell from
|
|
58
|
+
``final_values``, and MUST NOT carry the recursion across an episode boundary. A
|
|
59
|
+
truncation is an episode that was cut, not one that was decided, and treating the two
|
|
60
|
+
alike throws away the value of every position a step limit ended.
|
|
61
|
+
"""
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
"""How an observation is stored, and the rectangle it is stored in.
|
|
2
|
+
|
|
3
|
+
The two are one subject. The worker packs a row straight into its final resting place in the
|
|
4
|
+
experience buffer, so there is no intermediate copy to agree about; the learner unpacks it on
|
|
5
|
+
the device inside the same kernel that gathers the frame stack. What the codec may decide is
|
|
6
|
+
how each key is stored, and that is decided from the observation space at preflight rather than
|
|
7
|
+
from a list of plane indices written down here.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from abc import ABC, abstractmethod
|
|
13
|
+
from collections.abc import Callable, Iterator, Sequence
|
|
14
|
+
from typing import TYPE_CHECKING
|
|
15
|
+
|
|
16
|
+
import msgspec
|
|
17
|
+
import numpy as np
|
|
18
|
+
|
|
19
|
+
from .checkpoint import Checkpointable
|
|
20
|
+
|
|
21
|
+
if TYPE_CHECKING: # pragma: no cover - annotations only
|
|
22
|
+
from torch import Tensor
|
|
23
|
+
|
|
24
|
+
from ..learn.buffer import Batch
|
|
25
|
+
from ..rollout.layout import BufferHandle
|
|
26
|
+
from .policy import ObsBatch
|
|
27
|
+
from .rollout import EnvSpec, RolloutRound, SlotPlan
|
|
28
|
+
|
|
29
|
+
__all__ = ["MIN_TABLE_STATES", "CodecTable", "ExperienceBuffer", "ObsCodec"]
|
|
30
|
+
|
|
31
|
+
#: How many real observations a codec table may be decided from. Storage is decided per plane
|
|
32
|
+
#: from what the sample contains, so the sample has to be large enough, and played rather than
|
|
33
|
+
#: idle, to have reached the states in which a plane takes the values that decide it.
|
|
34
|
+
MIN_TABLE_STATES = 1000
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class CodecTable(msgspec.Struct, frozen=True, omit_defaults=True):
|
|
38
|
+
"""How each observation key is stored, decided from ``EnvSpec.obs_space`` at preflight
|
|
39
|
+
rather than from a list of plane indices. Logged, hashed and written into every snapshot.
|
|
40
|
+
|
|
41
|
+
``plane`` is one entry per spatial plane: its name, its storage in
|
|
42
|
+
{"uint8", "float16", "static", "derived"}, and the divisor that takes the stored integer
|
|
43
|
+
back to the value the environment produced. "derived" is reserved: nothing rebuilds such a
|
|
44
|
+
plane yet, so the codec refuses it at bind. Two runs whose tables differ are not comparable,
|
|
45
|
+
and ``digest()`` is what says so.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
plane: tuple[tuple[str, str, float], ...]
|
|
49
|
+
vector: str
|
|
50
|
+
mask: str
|
|
51
|
+
#: How ``card_ids`` is stored: "uint8", exact, or None when the observation has none. It is
|
|
52
|
+
#: omitted from the encoding when None (``omit_defaults``), so every table decided before
|
|
53
|
+
#: card identity existed hashes exactly as it did and no saved run's digest moves.
|
|
54
|
+
ids: str | None = None
|
|
55
|
+
|
|
56
|
+
def digest(self) -> str:
|
|
57
|
+
"""sha256 of the canonical JSON of this table."""
|
|
58
|
+
from ..rollout.envspec import digest_of
|
|
59
|
+
|
|
60
|
+
return digest_of(self)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class ObsCodec(ABC):
|
|
64
|
+
"""Quantisation of one observation row. The worker packs; the learner unpacks on the GPU.
|
|
65
|
+
|
|
66
|
+
Implementers MUST be exact round-trips for the integer-valued channels and MUST declare
|
|
67
|
+
``row_bytes`` as a constant given an ``EnvSpec`` and its ``CodecTable``.
|
|
68
|
+
"""
|
|
69
|
+
|
|
70
|
+
@abstractmethod
|
|
71
|
+
def table(
|
|
72
|
+
self,
|
|
73
|
+
spec: EnvSpec,
|
|
74
|
+
sample: Sequence[dict[str, np.ndarray]],
|
|
75
|
+
*,
|
|
76
|
+
min_states: int = MIN_TABLE_STATES,
|
|
77
|
+
) -> CodecTable:
|
|
78
|
+
"""Decide storage per key from the declared bounds and a sample of real observations.
|
|
79
|
+
|
|
80
|
+
Storage is decided from a sample; EXISTENCE is decided from the declaration. A plane the
|
|
81
|
+
layout does not declare static is stored even when it is constant across the sample,
|
|
82
|
+
because the tower planes are constant in any sample in which no tower falls.
|
|
83
|
+
|
|
84
|
+
Implementers MUST refuse a sample of fewer than ``min_states`` observations. The row
|
|
85
|
+
size of the whole run follows from this one decision, and a sample too small or too
|
|
86
|
+
idle to have reached the states a plane varies in decides it wrongly and in silence.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
@abstractmethod
|
|
90
|
+
def row_bytes(self, spec: EnvSpec) -> int: ...
|
|
91
|
+
|
|
92
|
+
@abstractmethod
|
|
93
|
+
def pack(self, obs: dict[str, np.ndarray], out: memoryview, row: int) -> None: ...
|
|
94
|
+
|
|
95
|
+
@abstractmethod
|
|
96
|
+
def static_planes(self, obs: dict[str, np.ndarray]) -> np.ndarray:
|
|
97
|
+
"""The planes ``EnvSpec.spatial_layout`` declares static; stored once per seat, never
|
|
98
|
+
per row."""
|
|
99
|
+
|
|
100
|
+
@abstractmethod
|
|
101
|
+
def unpack_to_device(self, raw: Tensor, statics: Tensor, out: ObsBatch) -> None:
|
|
102
|
+
"""Dequantise, scatter the static planes in, reshape the stored mask into the mask
|
|
103
|
+
planes, and gather the frame-stack history."""
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
@abstractmethod
|
|
107
|
+
def codec_version(self) -> int:
|
|
108
|
+
"""The RULE's version. The table it produces is data and travels separately."""
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class ExperienceBuffer(ABC, Checkpointable):
|
|
112
|
+
"""A rectangle of ``(T + frame_stack)`` cycles x ``R`` slots: ``T`` collected cycles, one
|
|
113
|
+
bootstrap row, and ``frame_stack - 1`` history rows carried over from the previous
|
|
114
|
+
iteration. Owns the shared-memory block the workers write into.
|
|
115
|
+
|
|
116
|
+
Implementers may assume each ``(cycle, slot)`` cell is written exactly once, by the worker
|
|
117
|
+
that owns that slot; the learner only reads.
|
|
118
|
+
"""
|
|
119
|
+
|
|
120
|
+
@abstractmethod
|
|
121
|
+
def shared_handle(self) -> BufferHandle:
|
|
122
|
+
"""Name, size and the numbers the offsets follow from; picklable, and sent to workers."""
|
|
123
|
+
|
|
124
|
+
@abstractmethod
|
|
125
|
+
def begin_iteration(self, plan: SlotPlan, cycles: int) -> None: ...
|
|
126
|
+
|
|
127
|
+
@abstractmethod
|
|
128
|
+
def record_round(
|
|
129
|
+
self, r: RolloutRound, actions: np.ndarray, log_probs: np.ndarray
|
|
130
|
+
) -> None:
|
|
131
|
+
"""Scalars only: observations are already in place. O(n), no observation copy."""
|
|
132
|
+
|
|
133
|
+
@abstractmethod
|
|
134
|
+
def set_values(self, values: Tensor) -> None:
|
|
135
|
+
"""``(T+1, R)`` float32, from the whole-iteration critic pass."""
|
|
136
|
+
|
|
137
|
+
@abstractmethod
|
|
138
|
+
def set_n_legal(self, counts: Tensor) -> None:
|
|
139
|
+
"""``(T+1, R)`` integer: how many actions each cell's mask left, from the critic's pass.
|
|
140
|
+
|
|
141
|
+
Implementers keep the collected cycles and may drop the bootstrap row, in which no
|
|
142
|
+
action was taken. The count is what tells the update which rows had a choice at all
|
|
143
|
+
without unpacking an observation to find out.
|
|
144
|
+
"""
|
|
145
|
+
|
|
146
|
+
@abstractmethod
|
|
147
|
+
def set_final_values(self, cells: np.ndarray, values: Tensor) -> None:
|
|
148
|
+
"""V(final_obs) for truncated cells; ``cells`` is int64[(k, 2)] of (cycle, slot)."""
|
|
149
|
+
|
|
150
|
+
@abstractmethod
|
|
151
|
+
def set_advantages(self, adv: Tensor, ret: Tensor) -> None:
|
|
152
|
+
"""``(T, R)`` float32 each."""
|
|
153
|
+
|
|
154
|
+
@abstractmethod
|
|
155
|
+
def trainable_mask(self) -> Tensor:
|
|
156
|
+
"""``(T, R)`` bool: which cells reach the update. What decides it is the seat's group
|
|
157
|
+
and the cell's validity, never where the row was written."""
|
|
158
|
+
|
|
159
|
+
@abstractmethod
|
|
160
|
+
def batches(
|
|
161
|
+
self,
|
|
162
|
+
batch_size: int,
|
|
163
|
+
minibatch_size: int,
|
|
164
|
+
epochs: int,
|
|
165
|
+
rng_for_epoch: Callable[[int], np.random.Generator],
|
|
166
|
+
*,
|
|
167
|
+
choice_first: bool = False,
|
|
168
|
+
) -> Iterator[Batch]:
|
|
169
|
+
"""Yield batches; each Batch knows its true sample count and iterates device-resident
|
|
170
|
+
minibatches. A batch never straddles an epoch boundary, and an epoch holds as many whole
|
|
171
|
+
batches as it can fill with the rows over spread one each across them, so every batch is
|
|
172
|
+
at least ``batch_size`` and an epoch is exactly ``n // batch_size`` optimizer steps. An
|
|
173
|
+
epoch with nothing trainable in it yields no batches at all. Each minibatch is weighted by
|
|
174
|
+
its share of its own batch. Gathers per MINIBATCH, never per batch.
|
|
175
|
+
|
|
176
|
+
``choice_first`` reorders each batch's cells so that the ones with more than one legal
|
|
177
|
+
action come first, keeping the permutation's order inside each class. Implementers MUST
|
|
178
|
+
move no cell between batches: it is a reordering, so that a caller skipping the forced
|
|
179
|
+
rows skips whole minibatches of them, and every batch-level denominator is unchanged."""
|