royalelearn 0.5.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. royalelearn-0.5.4/LICENSE +21 -0
  2. royalelearn-0.5.4/PKG-INFO +20 -0
  3. royalelearn-0.5.4/README.md +40 -0
  4. royalelearn-0.5.4/pyproject.toml +75 -0
  5. royalelearn-0.5.4/royalelearn/__init__.py +60 -0
  6. royalelearn-0.5.4/royalelearn/__main__.py +15 -0
  7. royalelearn-0.5.4/royalelearn/api/__init__.py +121 -0
  8. royalelearn-0.5.4/royalelearn/api/advantage.py +61 -0
  9. royalelearn-0.5.4/royalelearn/api/buffer.py +179 -0
  10. royalelearn-0.5.4/royalelearn/api/checkpoint.py +132 -0
  11. royalelearn-0.5.4/royalelearn/api/ladder.py +179 -0
  12. royalelearn-0.5.4/royalelearn/api/metrics.py +101 -0
  13. royalelearn-0.5.4/royalelearn/api/policy.py +189 -0
  14. royalelearn-0.5.4/royalelearn/api/rollout.py +404 -0
  15. royalelearn-0.5.4/royalelearn/api/schedule.py +59 -0
  16. royalelearn-0.5.4/royalelearn/api/update.py +202 -0
  17. royalelearn-0.5.4/royalelearn/checkpoint.py +547 -0
  18. royalelearn-0.5.4/royalelearn/cli.py +753 -0
  19. royalelearn-0.5.4/royalelearn/config.py +1368 -0
  20. royalelearn-0.5.4/royalelearn/coordinator.py +2908 -0
  21. royalelearn-0.5.4/royalelearn/determinism.py +146 -0
  22. royalelearn-0.5.4/royalelearn/errors.py +106 -0
  23. royalelearn-0.5.4/royalelearn/extensions.py +591 -0
  24. royalelearn-0.5.4/royalelearn/identity.py +798 -0
  25. royalelearn-0.5.4/royalelearn/ladder/__init__.py +56 -0
  26. royalelearn-0.5.4/royalelearn/ladder/actors.py +189 -0
  27. royalelearn-0.5.4/royalelearn/ladder/evaluate.py +404 -0
  28. royalelearn-0.5.4/royalelearn/ladder/eviction.py +84 -0
  29. royalelearn-0.5.4/royalelearn/ladder/farm.py +409 -0
  30. royalelearn-0.5.4/royalelearn/ladder/gate.py +375 -0
  31. royalelearn-0.5.4/royalelearn/ladder/matchmaker.py +498 -0
  32. royalelearn-0.5.4/royalelearn/ladder/pool.py +416 -0
  33. royalelearn-0.5.4/royalelearn/ladder/rating.py +445 -0
  34. royalelearn-0.5.4/royalelearn/ladder/results.py +345 -0
  35. royalelearn-0.5.4/royalelearn/ladder/seat_decks.py +66 -0
  36. royalelearn-0.5.4/royalelearn/ladder/snapshots.py +267 -0
  37. royalelearn-0.5.4/royalelearn/learn/__init__.py +97 -0
  38. royalelearn-0.5.4/royalelearn/learn/actor_critic.py +328 -0
  39. royalelearn-0.5.4/royalelearn/learn/buffer.py +891 -0
  40. royalelearn-0.5.4/royalelearn/learn/decode.py +125 -0
  41. royalelearn-0.5.4/royalelearn/learn/distribution.py +277 -0
  42. royalelearn-0.5.4/royalelearn/learn/freeze.py +92 -0
  43. royalelearn-0.5.4/royalelearn/learn/gae.py +192 -0
  44. royalelearn-0.5.4/royalelearn/learn/inference.py +622 -0
  45. royalelearn-0.5.4/royalelearn/learn/nets.py +690 -0
  46. royalelearn-0.5.4/royalelearn/learn/ppo.py +1816 -0
  47. royalelearn-0.5.4/royalelearn/learn/returns.py +184 -0
  48. royalelearn-0.5.4/royalelearn/learn/rows.py +63 -0
  49. royalelearn-0.5.4/royalelearn/learn/schedules.py +299 -0
  50. royalelearn-0.5.4/royalelearn/learner.py +600 -0
  51. royalelearn-0.5.4/royalelearn/metrics/__init__.py +34 -0
  52. royalelearn-0.5.4/royalelearn/metrics/alarms.py +627 -0
  53. royalelearn-0.5.4/royalelearn/metrics/behaviour.py +231 -0
  54. royalelearn-0.5.4/royalelearn/metrics/bundle.py +201 -0
  55. royalelearn-0.5.4/royalelearn/metrics/records.py +647 -0
  56. royalelearn-0.5.4/royalelearn/metrics/schema.py +1166 -0
  57. royalelearn-0.5.4/royalelearn/metrics/sinks.py +373 -0
  58. royalelearn-0.5.4/royalelearn/metrics/viser_sink.py +432 -0
  59. royalelearn-0.5.4/royalelearn/metrics/wandb_sink.py +141 -0
  60. royalelearn-0.5.4/royalelearn/obs_layout.py +136 -0
  61. royalelearn-0.5.4/royalelearn/rewards.py +466 -0
  62. royalelearn-0.5.4/royalelearn/rollout/__init__.py +37 -0
  63. royalelearn-0.5.4/royalelearn/rollout/codec.py +661 -0
  64. royalelearn-0.5.4/royalelearn/rollout/envspec.py +527 -0
  65. royalelearn-0.5.4/royalelearn/rollout/farm.py +560 -0
  66. royalelearn-0.5.4/royalelearn/rollout/inline.py +1626 -0
  67. royalelearn-0.5.4/royalelearn/rollout/layout.py +630 -0
  68. royalelearn-0.5.4/royalelearn/rollout/plan.py +266 -0
  69. royalelearn-0.5.4/royalelearn/rollout/preflight.py +787 -0
  70. royalelearn-0.5.4/royalelearn/rollout/scripted.py +151 -0
  71. royalelearn-0.5.4/royalelearn/rollout/worker.py +334 -0
  72. royalelearn-0.5.4/royalelearn/seeding.py +152 -0
  73. royalelearn-0.5.4/royalelearn/testing.py +686 -0
  74. royalelearn-0.5.4/royalelearn/version.py +42 -0
  75. royalelearn-0.5.4/royalelearn.egg-info/PKG-INFO +20 -0
  76. royalelearn-0.5.4/royalelearn.egg-info/SOURCES.txt +165 -0
  77. royalelearn-0.5.4/royalelearn.egg-info/dependency_links.txt +1 -0
  78. royalelearn-0.5.4/royalelearn.egg-info/entry_points.txt +2 -0
  79. royalelearn-0.5.4/royalelearn.egg-info/requires.txt +15 -0
  80. royalelearn-0.5.4/royalelearn.egg-info/top_level.txt +1 -0
  81. royalelearn-0.5.4/setup.cfg +4 -0
  82. royalelearn-0.5.4/tests/test_ability_buttons.py +152 -0
  83. royalelearn-0.5.4/tests/test_action_layout.py +109 -0
  84. royalelearn-0.5.4/tests/test_actor_terms.py +456 -0
  85. royalelearn-0.5.4/tests/test_adam_floor.py +114 -0
  86. royalelearn-0.5.4/tests/test_alarm_plants.py +228 -0
  87. royalelearn-0.5.4/tests/test_alarms.py +611 -0
  88. royalelearn-0.5.4/tests/test_bench_report.py +108 -0
  89. royalelearn-0.5.4/tests/test_buffer.py +898 -0
  90. royalelearn-0.5.4/tests/test_card_identity.py +473 -0
  91. royalelearn-0.5.4/tests/test_checkpoint.py +394 -0
  92. royalelearn-0.5.4/tests/test_codec.py +571 -0
  93. royalelearn-0.5.4/tests/test_config.py +472 -0
  94. royalelearn-0.5.4/tests/test_control_file.py +93 -0
  95. royalelearn-0.5.4/tests/test_coordinator.py +945 -0
  96. royalelearn-0.5.4/tests/test_decode.py +169 -0
  97. royalelearn-0.5.4/tests/test_dirty_sources.py +202 -0
  98. royalelearn-0.5.4/tests/test_distribution.py +365 -0
  99. royalelearn-0.5.4/tests/test_elixir_pricing.py +418 -0
  100. royalelearn-0.5.4/tests/test_engine_binary_identity.py +223 -0
  101. royalelearn-0.5.4/tests/test_engine_contract.py +159 -0
  102. royalelearn-0.5.4/tests/test_env_contract.py +297 -0
  103. royalelearn-0.5.4/tests/test_env_value_digest.py +133 -0
  104. royalelearn-0.5.4/tests/test_eval_actors.py +128 -0
  105. royalelearn-0.5.4/tests/test_eval_dispatch.py +130 -0
  106. royalelearn-0.5.4/tests/test_eval_farm.py +341 -0
  107. royalelearn-0.5.4/tests/test_eviction.py +166 -0
  108. royalelearn-0.5.4/tests/test_examples.py +49 -0
  109. royalelearn-0.5.4/tests/test_extension_example.py +130 -0
  110. royalelearn-0.5.4/tests/test_extensions.py +486 -0
  111. royalelearn-0.5.4/tests/test_factored_head.py +336 -0
  112. royalelearn-0.5.4/tests/test_farm_shutdown.py +246 -0
  113. royalelearn-0.5.4/tests/test_fill_policy.py +250 -0
  114. royalelearn-0.5.4/tests/test_frame_stack.py +149 -0
  115. royalelearn-0.5.4/tests/test_gae.py +318 -0
  116. royalelearn-0.5.4/tests/test_game_over_drive.py +30 -0
  117. royalelearn-0.5.4/tests/test_gate.py +820 -0
  118. royalelearn-0.5.4/tests/test_hold_lift.py +251 -0
  119. royalelearn-0.5.4/tests/test_housekeeping_failures.py +407 -0
  120. royalelearn-0.5.4/tests/test_identity.py +282 -0
  121. royalelearn-0.5.4/tests/test_inference.py +395 -0
  122. royalelearn-0.5.4/tests/test_ladder_probe.py +472 -0
  123. royalelearn-0.5.4/tests/test_layout.py +337 -0
  124. royalelearn-0.5.4/tests/test_learner.py +545 -0
  125. royalelearn-0.5.4/tests/test_manual_eval_log.py +60 -0
  126. royalelearn-0.5.4/tests/test_matchmaker.py +582 -0
  127. royalelearn-0.5.4/tests/test_metric_names.py +315 -0
  128. royalelearn-0.5.4/tests/test_metrics.py +812 -0
  129. royalelearn-0.5.4/tests/test_nets.py +377 -0
  130. royalelearn-0.5.4/tests/test_no_global_rng.py +127 -0
  131. royalelearn-0.5.4/tests/test_obs_layout.py +294 -0
  132. royalelearn-0.5.4/tests/test_package.py +103 -0
  133. royalelearn-0.5.4/tests/test_policy_probe.py +170 -0
  134. royalelearn-0.5.4/tests/test_pool_naming.py +116 -0
  135. royalelearn-0.5.4/tests/test_ppo.py +1450 -0
  136. royalelearn-0.5.4/tests/test_provenance.py +136 -0
  137. royalelearn-0.5.4/tests/test_rating.py +266 -0
  138. royalelearn-0.5.4/tests/test_ratio_precision.py +75 -0
  139. royalelearn-0.5.4/tests/test_recorder_plumbing.py +206 -0
  140. royalelearn-0.5.4/tests/test_replay_episode.py +143 -0
  141. royalelearn-0.5.4/tests/test_results_log.py +151 -0
  142. royalelearn-0.5.4/tests/test_resume.py +383 -0
  143. royalelearn-0.5.4/tests/test_returns.py +183 -0
  144. royalelearn-0.5.4/tests/test_rewards.py +659 -0
  145. royalelearn-0.5.4/tests/test_rng_roundtrip.py +85 -0
  146. royalelearn-0.5.4/tests/test_rollout_alignment.py +363 -0
  147. royalelearn-0.5.4/tests/test_rollout_farm.py +156 -0
  148. royalelearn-0.5.4/tests/test_rollout_inline.py +429 -0
  149. royalelearn-0.5.4/tests/test_rollout_invariants.py +189 -0
  150. royalelearn-0.5.4/tests/test_rollout_reads.py +258 -0
  151. royalelearn-0.5.4/tests/test_rss_units.py +60 -0
  152. royalelearn-0.5.4/tests/test_run_directory.py +71 -0
  153. royalelearn-0.5.4/tests/test_schedules.py +218 -0
  154. royalelearn-0.5.4/tests/test_scripted_opponents.py +265 -0
  155. royalelearn-0.5.4/tests/test_seat_decks.py +214 -0
  156. royalelearn-0.5.4/tests/test_seed_snapshots.py +273 -0
  157. royalelearn-0.5.4/tests/test_seeding.py +133 -0
  158. royalelearn-0.5.4/tests/test_segment_names.py +92 -0
  159. royalelearn-0.5.4/tests/test_shaping_strength.py +303 -0
  160. royalelearn-0.5.4/tests/test_shell_fences.py +135 -0
  161. royalelearn-0.5.4/tests/test_snapshots.py +250 -0
  162. royalelearn-0.5.4/tests/test_spell_identity.py +278 -0
  163. royalelearn-0.5.4/tests/test_user_code_identity.py +242 -0
  164. royalelearn-0.5.4/tests/test_viser_protocol.py +124 -0
  165. royalelearn-0.5.4/tests/test_viser_sink.py +259 -0
  166. royalelearn-0.5.4/tests/test_vram_device.py +114 -0
  167. royalelearn-0.5.4/tests/test_worker_hygiene.py +617 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 RoyaleGym contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,20 @@
1
+ Metadata-Version: 2.4
2
+ Name: royalelearn
3
+ Version: 0.5.4
4
+ Summary: RoyaleLearn: self-play PPO training harness for RoyaleGym Clash Royale environments
5
+ License: MIT
6
+ Requires-Python: >=3.12
7
+ License-File: LICENSE
8
+ Requires-Dist: royalegym>=0.1.0
9
+ Requires-Dist: numpy>=2.0
10
+ Requires-Dist: msgspec>=0.18
11
+ Provides-Extra: torch
12
+ Requires-Dist: torch>=2.4; extra == "torch"
13
+ Requires-Dist: safetensors>=0.4; extra == "torch"
14
+ Provides-Extra: wandb
15
+ Requires-Dist: wandb; extra == "wandb"
16
+ Provides-Extra: dev
17
+ Requires-Dist: pytest>=8.0; extra == "dev"
18
+ Requires-Dist: ruff>=0.16; extra == "dev"
19
+ Requires-Dist: hypothesis>=6.0; extra == "dev"
20
+ Dynamic: license-file
@@ -0,0 +1,40 @@
1
+ <p align="center"><img src="docs/media/logo.png" width="128" alt="The RoyaleLearn logo: a green crown shield with a white graduation cap on it, outlined in gold"></p><h1 align="center">RoyaleLearn</h1>
2
+
3
+ <p align="center"><a href="https://github.com/RoyaleGym/RoyaleLearn/actions/workflows/suite.yml"><img alt="CI" src="https://github.com/RoyaleGym/RoyaleLearn/actions/workflows/suite.yml/badge.svg"></a> <img alt="License" src="https://img.shields.io/github/license/RoyaleGym/RoyaleLearn?style=flat-square&color=555"> <img alt="Python" src="https://img.shields.io/badge/python-3.12%20%7C%203.13%20%7C%203.14-3776AB?style=flat-square&logo=python&logoColor=white"> <a href="https://royalegym.github.io/RoyaleGym/"><img alt="Docs" src="https://img.shields.io/badge/docs-royalegym.github.io-8957e5?style=flat-square&logo=readthedocs&logoColor=white"></a> <a href="https://discord.gg/4D2BS5JBHP"><img alt="Discord" src="https://img.shields.io/discord/1551699576304705647?style=flat-square&logo=discord&logoColor=white&label=discord&color=5865F2"></a> <img alt="Last commit" src="https://img.shields.io/github/last-commit/RoyaleGym/RoyaleLearn?style=flat-square&color=555"></p>
4
+
5
+ Train a Clash Royale bot in Python, on your own machine: you write the reward, it learns to win.
6
+ It is the trainer for the battles RoyaleGym runs, and it uses your NVIDIA graphics card.
7
+
8
+ ## Install
9
+
10
+ pip install "royalegym[all]" --find-links https://github.com/RoyaleGym/RoyaleGym/releases/expanded_assets/v0.1.8
11
+
12
+ This is the `[learn]` part, for Python 3.12 to 3.14. On Windows with an NVIDIA card, install PyTorch
13
+ first: [Install](https://royalegym.github.io/RoyaleGym/install/), step 4.
14
+
15
+ ## Try it
16
+
17
+ ```python
18
+ from royalegym import TowerHPReward, make_env
19
+ from royalelearn import Learner
20
+
21
+ def build_env():
22
+ return make_env(reward=TowerHPReward()) # what the bot is paid for
23
+
24
+ if __name__ == "__main__":
25
+ learner = Learner(build_env, save_dir="runs/my_bot")
26
+ learner.learn(total_steps=20_000) # counts all steps so far: raise it, run again to train more
27
+ learner.save("runs/my_bot/bot")
28
+ ```
29
+
30
+ It prints one line per update. To watch your bot play, see
31
+ [Watch It Play](https://royalegym.github.io/RoyaleGym/quickstart/#5-watch-it-play).
32
+
33
+ ## Next
34
+
35
+ - The docs: [royalegym.github.io/RoyaleGym](https://royalegym.github.io/RoyaleGym/)
36
+ - Quick Start: [the first bot, step by step](https://royalegym.github.io/RoyaleGym/quickstart/)
37
+ - Every setting: [RoyaleLearn](https://royalegym.github.io/RoyaleGym/resources/royalelearn/). How it works inside (Advanced): [the guide](https://royalegym.github.io/RoyaleGym/repos/royalelearn/guide/)
38
+ - Questions: [Discord](https://discord.gg/4D2BS5JBHP)
39
+
40
+ MIT licensed. See [LICENSE](LICENSE).
@@ -0,0 +1,75 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "royalelearn"
7
+ version = "0.5.4"
8
+ description = "RoyaleLearn: self-play PPO training harness for RoyaleGym Clash Royale environments"
9
+ requires-python = ">=3.12"
10
+ license = { text = "MIT" }
11
+ # royalegym is not on PyPI: it is the sibling checkout ../RoyaleGym, installed editable before
12
+ # this package by the workspace recipe in README.md (RoyaleSim -> RoyaleGym -> RoyaleViser ->
13
+ # RoyaleLearn). pip resolves it from the venv; it never fetches it.
14
+ #
15
+ # The base install is deliberately torch-free: the ABCs, the config tree, the CLI's config and
16
+ # identity commands and the rollout worker all run on numpy and msgspec alone.
17
+ dependencies = [
18
+ "royalegym>=0.1.0",
19
+ "numpy>=2.0",
20
+ "msgspec>=0.18",
21
+ ]
22
+
23
+ [project.optional-dependencies]
24
+ # The learner: networks, the update, inference, and everything that touches a device.
25
+ # safetensors is what the snapshot archive and the checkpoint store weights in; nothing this
26
+ # package writes is pickled, so it is a requirement of the learner and not an option under it.
27
+ torch = [
28
+ "torch>=2.4",
29
+ "safetensors>=0.4",
30
+ ]
31
+ # The optional metrics sink; the JSONL sink is always installed and is never optional.
32
+ wandb = [
33
+ "wandb",
34
+ ]
35
+ dev = [
36
+ "pytest>=8.0",
37
+ "ruff>=0.16",
38
+ "hypothesis>=6.0",
39
+ ]
40
+
41
+ [project.scripts]
42
+ royalelearn = "royalelearn.cli:main"
43
+
44
+ [tool.setuptools.packages.find]
45
+ include = ["royalelearn*"]
46
+
47
+ [tool.pytest.ini_options]
48
+ testpaths = ["tests"]
49
+ python_files = ["test_*.py"]
50
+ pythonpath = ["."]
51
+ # The default run is the fast suite. The slow tests and the ones needing a freshly built
52
+ # royalesim are selected explicitly, so a bare `pytest -q` stays a few seconds long.
53
+ addopts = "-ra --strict-markers -m \"not slow and not engine\""
54
+ markers = [
55
+ "slow: takes more than about five seconds",
56
+ "engine: needs a royalesim build that matches the data on disk",
57
+ ]
58
+
59
+ [tool.ruff]
60
+ line-length = 100
61
+ target-version = "py312"
62
+ src = ["royalelearn", "tests"]
63
+
64
+ [tool.ruff.lint]
65
+ select = ["E", "F", "W", "I", "B", "UP", "SIM", "RUF"]
66
+ ignore = ["RUF001", "RUF002", "RUF003", "SIM108"]
67
+
68
+ [tool.ruff.lint.per-file-ignores]
69
+ # A VENDORED COPY, kept byte-identical to the original in the other three repos so the four can
70
+ # be diffed against each other. Reformatting it to this repo's line length would make it a local
71
+ # rewrite that only looks like a copy, which is the one thing it must not be.
72
+ "tests/_shell_fences.py" = ["E501"]
73
+
74
+ [tool.ruff.lint.isort]
75
+ known-first-party = ["royalelearn", "royalegym", "royalesim"]
@@ -0,0 +1,60 @@
1
+ """royalelearn -- the self-play training harness for RoyaleGym environments.
2
+
3
+ Layers (each user-facing behaviour is an ABC in ``api/`` with a shipped default):
4
+
5
+ RolloutSource api/rollout.py ProcessRolloutSource, InlineRolloutSource
6
+ ActorCritic api/policy.py SeparateActorCritic, SharedTrunkActorCritic
7
+ ObsCodec api/buffer.py SpatialObsCodec
8
+ ExperienceBuffer api/buffer.py RectBuffer
9
+ Matchmaker/Rater api/ladder.py MixMatchmaker, BradleyTerryDavidsonRater
10
+ MetricsSink api/metrics.py JsonlSink, ConsoleSink, WandbSink
11
+ CheckpointStore api/checkpoint.py DirCheckpointStore
12
+
13
+ ``LearningCoordinator`` (coordinator.py) is the only place the phases of an iteration are
14
+ ordered. ``docs/harness-spec.md`` specifies all of it; ``docs/design.md`` says why.
15
+
16
+ Importing this package pulls in nothing that needs torch: the ABCs use numpy and msgspec,
17
+ and the concrete learner lives behind the lazy attributes below. That is what lets
18
+ ``royalelearn config``, ``royalelearn identity`` and ``royalelearn --help`` work in an
19
+ environment where torch is not installed. Asking for a name that does need torch raises an
20
+ ImportError naming the package that is missing.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import importlib
26
+ from typing import Any
27
+
28
+ from .version import __version__, git_describe
29
+
30
+ # Public name -> (module, attribute). A None attribute exports the module itself.
31
+ # Resolution is deferred so that the import cost and the torch dependency of a name are
32
+ # paid only by the caller that asks for it.
33
+ _EXPORTS: dict[str, tuple[str, str | None]] = {
34
+ "Learner": (".learner", "Learner"),
35
+ "LearningCoordinator": (".coordinator", "LearningCoordinator"),
36
+ "RunConfig": (".config", "RunConfig"),
37
+ "load_config": (".config", "load_config"),
38
+ "api": (".api", None),
39
+ }
40
+
41
+ __all__ = ["__version__", "git_describe", *sorted(_EXPORTS)]
42
+
43
+
44
+ def __getattr__(name: str) -> Any:
45
+ """Resolve a public name on first use.
46
+
47
+ An unknown name is an AttributeError. A known name whose module cannot be imported
48
+ raises that import's own ImportError unchanged, because its message already names what
49
+ is missing and rewording it would hide which package to install.
50
+ """
51
+ try:
52
+ module_name, attribute = _EXPORTS[name]
53
+ except KeyError:
54
+ raise AttributeError(f"module 'royalelearn' has no attribute {name!r}") from None
55
+ module = importlib.import_module(module_name, __name__)
56
+ return module if attribute is None else getattr(module, attribute)
57
+
58
+
59
+ def __dir__() -> list[str]:
60
+ return sorted(__all__)
@@ -0,0 +1,15 @@
1
+ """``python -m royalelearn``.
2
+
3
+ The module body is one call. Everything the entry point has to do before torch is imported --
4
+ the deterministic cuBLAS workspace and the BLAS thread counts -- ``cli`` does as it is imported,
5
+ which is why this file imports it and nothing else.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import sys
11
+
12
+ from .cli import main
13
+
14
+ if __name__ == "__main__":
15
+ sys.exit(main())
@@ -0,0 +1,121 @@
1
+ """Every ABC and struct the harness is written against.
2
+
3
+ This subpackage imports numpy and msgspec and nothing else that matters: torch appears in type
4
+ annotations only, behind ``TYPE_CHECKING``. That is what lets a rollout worker, the CLI's
5
+ config and identity commands, and ``import royalelearn`` itself run in an environment with no
6
+ torch installed -- and what keeps a worker's resident memory three hundred megabytes smaller
7
+ than the parent's.
8
+
9
+ The concrete implementations live outside ``api/`` and may import whatever they need:
10
+
11
+ RolloutSource rollout/farm.py, rollout/inline.py
12
+ ActorCritic learn/actor_critic.py
13
+ ObsCodec rollout/codec.py
14
+ ExperienceBuffer learn/buffer.py
15
+ AdvantageEstimator learn/gae.py
16
+ Update learn/ppo.py
17
+ Schedule learn/schedules.py
18
+ Matchmaker, Rater ladder/matchmaker.py, ladder/rating.py
19
+ MetricsSink, Alarm metrics/sinks.py, metrics/alarms.py
20
+ CheckpointStore checkpoint.py
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from .advantage import AdvantageEstimator, AdvantageStats
26
+ from .buffer import CodecTable, ExperienceBuffer, ObsCodec
27
+ from .checkpoint import Checkpointable, CheckpointStore, Manifest, RngState
28
+ from .ladder import (
29
+ ConditionResult,
30
+ EvictionPolicy,
31
+ GateDecision,
32
+ Matchmaker,
33
+ PromotionGate,
34
+ Rater,
35
+ RatingTable,
36
+ SnapshotStore,
37
+ )
38
+ from .metrics import Alarm, AlarmResult, MetricRow, MetricsSink, MetricValue
39
+ from .policy import (
40
+ ActionDistribution,
41
+ Actor,
42
+ ActorCritic,
43
+ ActResult,
44
+ BackpropResult,
45
+ Critic,
46
+ NetworkFactory,
47
+ ObsBatch,
48
+ )
49
+ from .rollout import (
50
+ Assignment,
51
+ Close,
52
+ Defer,
53
+ EnvSpec,
54
+ EpisodeRecord,
55
+ ObsKeySpec,
56
+ Plan,
57
+ RolloutRound,
58
+ RolloutSource,
59
+ SetState,
60
+ SlotPlan,
61
+ Spaces,
62
+ Step,
63
+ WorkerCommand,
64
+ WorkerFailure,
65
+ )
66
+ from .schedule import Schedule, ScheduleState
67
+ from .update import ActorLossTerm, ActorTermInputs, Update, UpdateResult
68
+
69
+ __all__ = [
70
+ "ActResult",
71
+ "ActionDistribution",
72
+ "Actor",
73
+ "ActorCritic",
74
+ "ActorLossTerm",
75
+ "ActorTermInputs",
76
+ "AdvantageEstimator",
77
+ "AdvantageStats",
78
+ "Alarm",
79
+ "AlarmResult",
80
+ "Assignment",
81
+ "BackpropResult",
82
+ "CheckpointStore",
83
+ "Checkpointable",
84
+ "Close",
85
+ "CodecTable",
86
+ "ConditionResult",
87
+ "Critic",
88
+ "Defer",
89
+ "EnvSpec",
90
+ "EpisodeRecord",
91
+ "EvictionPolicy",
92
+ "ExperienceBuffer",
93
+ "GateDecision",
94
+ "Manifest",
95
+ "Matchmaker",
96
+ "MetricRow",
97
+ "MetricValue",
98
+ "MetricsSink",
99
+ "NetworkFactory",
100
+ "ObsBatch",
101
+ "ObsCodec",
102
+ "ObsKeySpec",
103
+ "Plan",
104
+ "PromotionGate",
105
+ "Rater",
106
+ "RatingTable",
107
+ "RngState",
108
+ "RolloutRound",
109
+ "RolloutSource",
110
+ "Schedule",
111
+ "ScheduleState",
112
+ "SetState",
113
+ "SlotPlan",
114
+ "SnapshotStore",
115
+ "Spaces",
116
+ "Step",
117
+ "Update",
118
+ "UpdateResult",
119
+ "WorkerCommand",
120
+ "WorkerFailure",
121
+ ]
@@ -0,0 +1,61 @@
1
+ """How a return is turned into an advantage.
2
+
3
+ One ABC, because the estimator is the piece most likely to be replaced -- V-trace against a
4
+ frozen pool, a different lambda rule, a per-bucket normalisation -- and because the one thing it
5
+ must never do is carry its recursion across an episode boundary.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from abc import ABC, abstractmethod
11
+ from typing import TYPE_CHECKING
12
+
13
+ import msgspec
14
+
15
+ from .checkpoint import Checkpointable
16
+
17
+ if TYPE_CHECKING: # pragma: no cover - annotations only
18
+ from torch import Tensor
19
+
20
+ __all__ = ["AdvantageEstimator", "AdvantageStats"]
21
+
22
+
23
+ class AdvantageStats(msgspec.Struct):
24
+ """What the estimator saw, for the metric row.
25
+
26
+ ``reward_scale`` is the divisor the return scaler applied and ``clipped_reward_frac`` is how
27
+ much of the batch the clip bound touched: both are how a scaled reward stays legible after
28
+ the scaling.
29
+ """
30
+
31
+ raw_return_mean: float
32
+ raw_return_std: float
33
+ reward_scale: float
34
+ clipped_reward_frac: float
35
+
36
+
37
+ class AdvantageEstimator(ABC, Checkpointable):
38
+ """Rewards and values in, advantages and returns out."""
39
+
40
+ @abstractmethod
41
+ def compute(
42
+ self,
43
+ *,
44
+ rewards: Tensor,
45
+ values: Tensor,
46
+ final_values: Tensor,
47
+ terminated: Tensor,
48
+ truncated: Tensor,
49
+ trainable: Tensor,
50
+ gamma: float,
51
+ lam: float,
52
+ ) -> tuple[Tensor, Tensor, AdvantageStats]:
53
+ """``rewards``/``terminated``/``truncated``/``trainable`` are ``(T, R)``; ``values`` is
54
+ ``(T+1, R)``; ``final_values`` is ``(T, R)`` and is read only where truncated.
55
+ Returns ``(advantages (T, R), returns (T, R), stats)``.
56
+
57
+ Implementers MUST bootstrap a terminated cell from 0 and a truncated cell from
58
+ ``final_values``, and MUST NOT carry the recursion across an episode boundary. A
59
+ truncation is an episode that was cut, not one that was decided, and treating the two
60
+ alike throws away the value of every position a step limit ended.
61
+ """
@@ -0,0 +1,179 @@
1
+ """How an observation is stored, and the rectangle it is stored in.
2
+
3
+ The two are one subject. The worker packs a row straight into its final resting place in the
4
+ experience buffer, so there is no intermediate copy to agree about; the learner unpacks it on
5
+ the device inside the same kernel that gathers the frame stack. What the codec may decide is
6
+ how each key is stored, and that is decided from the observation space at preflight rather than
7
+ from a list of plane indices written down here.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from abc import ABC, abstractmethod
13
+ from collections.abc import Callable, Iterator, Sequence
14
+ from typing import TYPE_CHECKING
15
+
16
+ import msgspec
17
+ import numpy as np
18
+
19
+ from .checkpoint import Checkpointable
20
+
21
+ if TYPE_CHECKING: # pragma: no cover - annotations only
22
+ from torch import Tensor
23
+
24
+ from ..learn.buffer import Batch
25
+ from ..rollout.layout import BufferHandle
26
+ from .policy import ObsBatch
27
+ from .rollout import EnvSpec, RolloutRound, SlotPlan
28
+
29
+ __all__ = ["MIN_TABLE_STATES", "CodecTable", "ExperienceBuffer", "ObsCodec"]
30
+
31
+ #: How many real observations a codec table may be decided from. Storage is decided per plane
32
+ #: from what the sample contains, so the sample has to be large enough, and played rather than
33
+ #: idle, to have reached the states in which a plane takes the values that decide it.
34
+ MIN_TABLE_STATES = 1000
35
+
36
+
37
+ class CodecTable(msgspec.Struct, frozen=True, omit_defaults=True):
38
+ """How each observation key is stored, decided from ``EnvSpec.obs_space`` at preflight
39
+ rather than from a list of plane indices. Logged, hashed and written into every snapshot.
40
+
41
+ ``plane`` is one entry per spatial plane: its name, its storage in
42
+ {"uint8", "float16", "static", "derived"}, and the divisor that takes the stored integer
43
+ back to the value the environment produced. "derived" is reserved: nothing rebuilds such a
44
+ plane yet, so the codec refuses it at bind. Two runs whose tables differ are not comparable,
45
+ and ``digest()`` is what says so.
46
+ """
47
+
48
+ plane: tuple[tuple[str, str, float], ...]
49
+ vector: str
50
+ mask: str
51
+ #: How ``card_ids`` is stored: "uint8", exact, or None when the observation has none. It is
52
+ #: omitted from the encoding when None (``omit_defaults``), so every table decided before
53
+ #: card identity existed hashes exactly as it did and no saved run's digest moves.
54
+ ids: str | None = None
55
+
56
+ def digest(self) -> str:
57
+ """sha256 of the canonical JSON of this table."""
58
+ from ..rollout.envspec import digest_of
59
+
60
+ return digest_of(self)
61
+
62
+
63
+ class ObsCodec(ABC):
64
+ """Quantisation of one observation row. The worker packs; the learner unpacks on the GPU.
65
+
66
+ Implementers MUST be exact round-trips for the integer-valued channels and MUST declare
67
+ ``row_bytes`` as a constant given an ``EnvSpec`` and its ``CodecTable``.
68
+ """
69
+
70
+ @abstractmethod
71
+ def table(
72
+ self,
73
+ spec: EnvSpec,
74
+ sample: Sequence[dict[str, np.ndarray]],
75
+ *,
76
+ min_states: int = MIN_TABLE_STATES,
77
+ ) -> CodecTable:
78
+ """Decide storage per key from the declared bounds and a sample of real observations.
79
+
80
+ Storage is decided from a sample; EXISTENCE is decided from the declaration. A plane the
81
+ layout does not declare static is stored even when it is constant across the sample,
82
+ because the tower planes are constant in any sample in which no tower falls.
83
+
84
+ Implementers MUST refuse a sample of fewer than ``min_states`` observations. The row
85
+ size of the whole run follows from this one decision, and a sample too small or too
86
+ idle to have reached the states a plane varies in decides it wrongly and in silence.
87
+ """
88
+
89
+ @abstractmethod
90
+ def row_bytes(self, spec: EnvSpec) -> int: ...
91
+
92
+ @abstractmethod
93
+ def pack(self, obs: dict[str, np.ndarray], out: memoryview, row: int) -> None: ...
94
+
95
+ @abstractmethod
96
+ def static_planes(self, obs: dict[str, np.ndarray]) -> np.ndarray:
97
+ """The planes ``EnvSpec.spatial_layout`` declares static; stored once per seat, never
98
+ per row."""
99
+
100
+ @abstractmethod
101
+ def unpack_to_device(self, raw: Tensor, statics: Tensor, out: ObsBatch) -> None:
102
+ """Dequantise, scatter the static planes in, reshape the stored mask into the mask
103
+ planes, and gather the frame-stack history."""
104
+
105
+ @property
106
+ @abstractmethod
107
+ def codec_version(self) -> int:
108
+ """The RULE's version. The table it produces is data and travels separately."""
109
+
110
+
111
+ class ExperienceBuffer(ABC, Checkpointable):
112
+ """A rectangle of ``(T + frame_stack)`` cycles x ``R`` slots: ``T`` collected cycles, one
113
+ bootstrap row, and ``frame_stack - 1`` history rows carried over from the previous
114
+ iteration. Owns the shared-memory block the workers write into.
115
+
116
+ Implementers may assume each ``(cycle, slot)`` cell is written exactly once, by the worker
117
+ that owns that slot; the learner only reads.
118
+ """
119
+
120
+ @abstractmethod
121
+ def shared_handle(self) -> BufferHandle:
122
+ """Name, size and the numbers the offsets follow from; picklable, and sent to workers."""
123
+
124
+ @abstractmethod
125
+ def begin_iteration(self, plan: SlotPlan, cycles: int) -> None: ...
126
+
127
+ @abstractmethod
128
+ def record_round(
129
+ self, r: RolloutRound, actions: np.ndarray, log_probs: np.ndarray
130
+ ) -> None:
131
+ """Scalars only: observations are already in place. O(n), no observation copy."""
132
+
133
+ @abstractmethod
134
+ def set_values(self, values: Tensor) -> None:
135
+ """``(T+1, R)`` float32, from the whole-iteration critic pass."""
136
+
137
+ @abstractmethod
138
+ def set_n_legal(self, counts: Tensor) -> None:
139
+ """``(T+1, R)`` integer: how many actions each cell's mask left, from the critic's pass.
140
+
141
+ Implementers keep the collected cycles and may drop the bootstrap row, in which no
142
+ action was taken. The count is what tells the update which rows had a choice at all
143
+ without unpacking an observation to find out.
144
+ """
145
+
146
+ @abstractmethod
147
+ def set_final_values(self, cells: np.ndarray, values: Tensor) -> None:
148
+ """V(final_obs) for truncated cells; ``cells`` is int64[(k, 2)] of (cycle, slot)."""
149
+
150
+ @abstractmethod
151
+ def set_advantages(self, adv: Tensor, ret: Tensor) -> None:
152
+ """``(T, R)`` float32 each."""
153
+
154
+ @abstractmethod
155
+ def trainable_mask(self) -> Tensor:
156
+ """``(T, R)`` bool: which cells reach the update. What decides it is the seat's group
157
+ and the cell's validity, never where the row was written."""
158
+
159
+ @abstractmethod
160
+ def batches(
161
+ self,
162
+ batch_size: int,
163
+ minibatch_size: int,
164
+ epochs: int,
165
+ rng_for_epoch: Callable[[int], np.random.Generator],
166
+ *,
167
+ choice_first: bool = False,
168
+ ) -> Iterator[Batch]:
169
+ """Yield batches; each Batch knows its true sample count and iterates device-resident
170
+ minibatches. A batch never straddles an epoch boundary, and an epoch holds as many whole
171
+ batches as it can fill with the rows over spread one each across them, so every batch is
172
+ at least ``batch_size`` and an epoch is exactly ``n // batch_size`` optimizer steps. An
173
+ epoch with nothing trainable in it yields no batches at all. Each minibatch is weighted by
174
+ its share of its own batch. Gathers per MINIBATCH, never per batch.
175
+
176
+ ``choice_first`` reorders each batch's cells so that the ones with more than one legal
177
+ action come first, keeping the permutation's order inside each class. Implementers MUST
178
+ move no cell between batches: it is a reordering, so that a caller skipping the forced
179
+ rows skips whole minibatches of them, and every batch-level denominator is unchanged."""