gauntlet-robotics 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gauntlet_robotics-0.2.0/PKG-INFO +628 -0
- gauntlet_robotics-0.2.0/README.md +546 -0
- gauntlet_robotics-0.2.0/pyproject.toml +887 -0
- gauntlet_robotics-0.2.0/src/gauntlet/__init__.py +60 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/__init__.py +51 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/analyze.py +394 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/cli.py +125 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/fleet_clustering.py +793 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/html.py +102 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/schema.py +104 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/sim_real.py +358 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/templates/__init__.py +1 -0
- gauntlet_robotics-0.2.0/src/gauntlet/aggregate/templates/fleet_report.html.jinja +346 -0
- gauntlet_robotics-0.2.0/src/gauntlet/bisect/__init__.py +64 -0
- gauntlet_robotics-0.2.0/src/gauntlet/bisect/bisect.py +464 -0
- gauntlet_robotics-0.2.0/src/gauntlet/bisect/cli.py +330 -0
- gauntlet_robotics-0.2.0/src/gauntlet/cli.py +3689 -0
- gauntlet_robotics-0.2.0/src/gauntlet/compare/__init__.py +43 -0
- gauntlet_robotics-0.2.0/src/gauntlet/compare/drift_map.py +270 -0
- gauntlet_robotics-0.2.0/src/gauntlet/compare/github_summary.py +139 -0
- gauntlet_robotics-0.2.0/src/gauntlet/dashboard/__init__.py +30 -0
- gauntlet_robotics-0.2.0/src/gauntlet/dashboard/build.py +322 -0
- gauntlet_robotics-0.2.0/src/gauntlet/dashboard/static/__init__.py +1 -0
- gauntlet_robotics-0.2.0/src/gauntlet/dashboard/static/dashboard.css +187 -0
- gauntlet_robotics-0.2.0/src/gauntlet/dashboard/static/dashboard.js +199 -0
- gauntlet_robotics-0.2.0/src/gauntlet/dashboard/templates/__init__.py +1 -0
- gauntlet_robotics-0.2.0/src/gauntlet/dashboard/templates/dashboard.html.jinja +158 -0
- gauntlet_robotics-0.2.0/src/gauntlet/diff/__init__.py +61 -0
- gauntlet_robotics-0.2.0/src/gauntlet/diff/diff.py +457 -0
- gauntlet_robotics-0.2.0/src/gauntlet/diff/paired.py +542 -0
- gauntlet_robotics-0.2.0/src/gauntlet/diff/render.py +229 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/__init__.py +57 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/assets/objects/banana.xml +8 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/assets/objects/bottle.xml +7 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/assets/objects/mug.xml +9 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/assets/objects/screwdriver.xml +7 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/assets/tabletop.xml +122 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/assets/tabletop_stack.xml +66 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/base.py +269 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/color_attack.py +409 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/genesis/__init__.py +80 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/genesis/tabletop_genesis.py +853 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/gym_registration.py +126 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/image_attack.py +397 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/instruction.py +262 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/isaac/__init__.py +62 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/isaac/tabletop_isaac.py +709 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/mobile.py +294 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/perturbation/__init__.py +151 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/perturbation/axes.py +697 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/perturbation/base.py +152 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/pybullet/__init__.py +56 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/pybullet/assets/cube_alt.png +0 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/pybullet/assets/cube_default.png +0 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/pybullet/tabletop_pybullet.py +1235 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/registry.py +127 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/tabletop.py +1533 -0
- gauntlet_robotics-0.2.0/src/gauntlet/env/tabletop_stack.py +613 -0
- gauntlet_robotics-0.2.0/src/gauntlet/monitor/__init__.py +82 -0
- gauntlet_robotics-0.2.0/src/gauntlet/monitor/ae.py +275 -0
- gauntlet_robotics-0.2.0/src/gauntlet/monitor/conformal.py +231 -0
- gauntlet_robotics-0.2.0/src/gauntlet/monitor/entropy.py +85 -0
- gauntlet_robotics-0.2.0/src/gauntlet/monitor/schema.py +102 -0
- gauntlet_robotics-0.2.0/src/gauntlet/monitor/score.py +198 -0
- gauntlet_robotics-0.2.0/src/gauntlet/monitor/train.py +229 -0
- gauntlet_robotics-0.2.0/src/gauntlet/plugins.py +312 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/__init__.py +72 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/base.py +120 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/dt.py +239 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/groot.py +203 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/huggingface.py +278 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/lerobot.py +466 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/pi0.py +250 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/random.py +103 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/rdt.py +177 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/registry.py +219 -0
- gauntlet_robotics-0.2.0/src/gauntlet/policy/scripted.py +99 -0
- gauntlet_robotics-0.2.0/src/gauntlet/py.typed +0 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/__init__.py +112 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/io.py +189 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/pipeline.py +337 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/renderer.py +187 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/renderers/__init__.py +40 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/renderers/gsplat.py +181 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/renderers/nearest_frame.py +160 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/scene_input.py +509 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/scene_to_axis.py +210 -0
- gauntlet_robotics-0.2.0/src/gauntlet/realsim/schema.py +287 -0
- gauntlet_robotics-0.2.0/src/gauntlet/replay/__init__.py +32 -0
- gauntlet_robotics-0.2.0/src/gauntlet/replay/overrides.py +170 -0
- gauntlet_robotics-0.2.0/src/gauntlet/replay/replay.py +262 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/__init__.py +55 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/abstention.py +134 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/analyze.py +682 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/html.py +211 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/junit.py +93 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/schema.py +384 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/sobol_indices.py +182 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/templates/__init__.py +1 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/templates/report.html.jinja +957 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/trajectory_taxonomy.py +529 -0
- gauntlet_robotics-0.2.0/src/gauntlet/report/wilson.py +293 -0
- gauntlet_robotics-0.2.0/src/gauntlet/ros2/__init__.py +78 -0
- gauntlet_robotics-0.2.0/src/gauntlet/ros2/publisher.py +175 -0
- gauntlet_robotics-0.2.0/src/gauntlet/ros2/recorder.py +235 -0
- gauntlet_robotics-0.2.0/src/gauntlet/ros2/schema.py +95 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/__init__.py +72 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/cache.py +373 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/determinism.py +403 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/episode.py +473 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/parquet.py +189 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/provenance.py +268 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/runner.py +927 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/sinks.py +248 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/video.py +222 -0
- gauntlet_robotics-0.2.0/src/gauntlet/runner/worker.py +1128 -0
- gauntlet_robotics-0.2.0/src/gauntlet/security/__init__.py +38 -0
- gauntlet_robotics-0.2.0/src/gauntlet/security/paths.py +147 -0
- gauntlet_robotics-0.2.0/src/gauntlet/security/yaml_guard.py +81 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/__init__.py +47 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/adversarial.py +300 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/lhs.py +179 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/linter.py +389 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/loader.py +318 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/sampling.py +238 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/schema.py +925 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/sobol.py +318 -0
- gauntlet_robotics-0.2.0/src/gauntlet/suite/worst_case.py +403 -0
|
@@ -0,0 +1,628 @@
|
|
|
1
|
+
Metadata-Version: 2.3
|
|
2
|
+
Name: gauntlet-robotics
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: An evaluation harness for learned robot policies.
|
|
5
|
+
Keywords: robotics,evaluation,policy,mujoco,regression-testing
|
|
6
|
+
Author: Gauntlet contributors
|
|
7
|
+
License: MIT
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Dist: mujoco>=3.2,<5
|
|
16
|
+
Requires-Dist: gymnasium>=1.0,<2
|
|
17
|
+
Requires-Dist: pydantic>=2.7,<3
|
|
18
|
+
Requires-Dist: typer>=0.12,<1
|
|
19
|
+
Requires-Dist: numpy>=1.26,<3
|
|
20
|
+
Requires-Dist: pandas>=2.2,<4
|
|
21
|
+
Requires-Dist: jinja2>=3.1,<4
|
|
22
|
+
Requires-Dist: pyyaml>=6.0,<7
|
|
23
|
+
Requires-Dist: rich>=13.7,<17
|
|
24
|
+
Requires-Dist: transformers>=4.40,<5 ; extra == 'dt'
|
|
25
|
+
Requires-Dist: torch>=2.2,<3 ; extra == 'dt'
|
|
26
|
+
Requires-Dist: huggingface-hub>=0.20,<1 ; extra == 'dt'
|
|
27
|
+
Requires-Dist: genesis-world>=0.4,<0.5 ; extra == 'genesis'
|
|
28
|
+
Requires-Dist: torch>=2.2,<3 ; extra == 'genesis'
|
|
29
|
+
Requires-Dist: lerobot>=0.4,<1 ; extra == 'groot'
|
|
30
|
+
Requires-Dist: transformers>=4.40,<5 ; extra == 'groot'
|
|
31
|
+
Requires-Dist: pillow>=10.0,<12 ; extra == 'groot'
|
|
32
|
+
Requires-Dist: torch>=2.2,<3 ; extra == 'hf'
|
|
33
|
+
Requires-Dist: transformers>=4.40,<5 ; extra == 'hf'
|
|
34
|
+
Requires-Dist: timm>=0.9.10,<2 ; extra == 'hf'
|
|
35
|
+
Requires-Dist: tokenizers>=0.19,<1 ; extra == 'hf'
|
|
36
|
+
Requires-Dist: pillow>=10.0,<12 ; extra == 'hf'
|
|
37
|
+
Requires-Dist: isaacsim>=5.0,<6 ; extra == 'isaac'
|
|
38
|
+
Requires-Dist: lerobot[smolvla]>=0.4,<1 ; extra == 'lerobot'
|
|
39
|
+
Requires-Dist: transformers>=4.40,<5 ; extra == 'lerobot'
|
|
40
|
+
Requires-Dist: pillow>=10.0,<12 ; extra == 'lerobot'
|
|
41
|
+
Requires-Dist: mlflow>=2.10,<4 ; extra == 'mlflow'
|
|
42
|
+
Requires-Dist: torch>=2.2,<3 ; extra == 'monitor'
|
|
43
|
+
Requires-Dist: pillow>=10.0,<12 ; extra == 'monitor'
|
|
44
|
+
Requires-Dist: pyarrow>=15.0,<22 ; extra == 'parquet'
|
|
45
|
+
Requires-Dist: lerobot[pi]>=0.4,<1 ; extra == 'pi0'
|
|
46
|
+
Requires-Dist: transformers>=4.40,<5 ; extra == 'pi0'
|
|
47
|
+
Requires-Dist: pillow>=10.0,<12 ; extra == 'pi0'
|
|
48
|
+
Requires-Dist: pybullet>=3.2,<4 ; extra == 'pybullet'
|
|
49
|
+
Requires-Dist: torch>=2.2,<3 ; extra == 'rdt'
|
|
50
|
+
Requires-Dist: transformers>=4.40,<5 ; extra == 'rdt'
|
|
51
|
+
Requires-Dist: huggingface-hub>=0.20,<1 ; extra == 'rdt'
|
|
52
|
+
Requires-Dist: pillow>=10.0,<12 ; extra == 'rdt'
|
|
53
|
+
Requires-Dist: torch>=2.0,<3 ; extra == 'realsim-gsplat'
|
|
54
|
+
Requires-Dist: gsplat>=1.0,<2 ; extra == 'realsim-gsplat'
|
|
55
|
+
Requires-Dist: dtw-python>=1.4,<2 ; extra == 'trajectory-taxonomy'
|
|
56
|
+
Requires-Dist: imageio[ffmpeg]>=2.34,<3 ; extra == 'video'
|
|
57
|
+
Requires-Dist: wandb>=0.16,<1 ; extra == 'wandb'
|
|
58
|
+
Requires-Python: >=3.11
|
|
59
|
+
Project-URL: Changelog, https://github.com/mhussainahmad/gauntlet/blob/main/CHANGELOG.md
|
|
60
|
+
Project-URL: Documentation, https://github.com/mhussainahmad/gauntlet/blob/main/README.md
|
|
61
|
+
Project-URL: Homepage, https://github.com/mhussainahmad/gauntlet
|
|
62
|
+
Project-URL: Issues, https://github.com/mhussainahmad/gauntlet/issues
|
|
63
|
+
Project-URL: Repository, https://github.com/mhussainahmad/gauntlet
|
|
64
|
+
Provides-Extra: dt
|
|
65
|
+
Provides-Extra: genesis
|
|
66
|
+
Provides-Extra: groot
|
|
67
|
+
Provides-Extra: hf
|
|
68
|
+
Provides-Extra: isaac
|
|
69
|
+
Provides-Extra: lerobot
|
|
70
|
+
Provides-Extra: mlflow
|
|
71
|
+
Provides-Extra: monitor
|
|
72
|
+
Provides-Extra: parquet
|
|
73
|
+
Provides-Extra: pi0
|
|
74
|
+
Provides-Extra: pybullet
|
|
75
|
+
Provides-Extra: rdt
|
|
76
|
+
Provides-Extra: realsim-gsplat
|
|
77
|
+
Provides-Extra: ros2
|
|
78
|
+
Provides-Extra: trajectory-taxonomy
|
|
79
|
+
Provides-Extra: video
|
|
80
|
+
Provides-Extra: wandb
|
|
81
|
+
Description-Content-Type: text/markdown
|
|
82
|
+
|
|
83
|
+
# Gauntlet
|
|
84
|
+
|
|
85
|
+
> An evaluation harness for learned robot policies.
|
|
86
|
+
|
|
87
|
+
Gauntlet answers a single question for VLA / diffusion / scripted policies:
|
|
88
|
+
|
|
89
|
+
> *"How does this policy fail, and has the latest checkpoint regressed against the last one?"*
|
|
90
|
+
|
|
91
|
+
It wraps any policy behind a uniform adapter, runs it across a parameterized
|
|
92
|
+
suite of MuJoCo perturbations (lighting, camera pose, textures, clutter,
|
|
93
|
+
initial conditions), and produces a structured report that **breaks failures
|
|
94
|
+
down by axis** instead of hiding them in an aggregate mean.
|
|
95
|
+
|
|
96
|
+
See [`GAUNTLET_SPEC.md`](./GAUNTLET_SPEC.md) for the full design.
|
|
97
|
+
|
|
98
|
+
---
|
|
99
|
+
|
|
100
|
+
## See a real report before installing
|
|
101
|
+
|
|
102
|
+
A public reference benchmark lives at
|
|
103
|
+
**<https://mhussainahmad.github.io/gauntlet/>** — two policies on the
|
|
104
|
+
bundled smoke suite, regenerated on every release, with the
|
|
105
|
+
`gauntlet compare` and `gauntlet diff` deltas surfaced. Open the
|
|
106
|
+
baseline / regressed `report.html` to see what the failure-cluster-first
|
|
107
|
+
layout actually looks like.
|
|
108
|
+
|
|
109
|
+
To regenerate it locally:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
python scripts/generate_reference_benchmark.py --out ./benchmarks/local/
|
|
113
|
+
open ./benchmarks/local/index.html
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
## Install
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
pip install gauntlet-robotics # core (MuJoCo only, torch-free)
|
|
120
|
+
pip install 'gauntlet-robotics[hf]' # + OpenVLA / HuggingFace adapter
|
|
121
|
+
pip install 'gauntlet-robotics[lerobot]' # + SmolVLA / π0 / diffusion adapters
|
|
122
|
+
pip install 'gauntlet-robotics[pybullet]' # + PyBullet backend
|
|
123
|
+
pip install 'gauntlet-robotics[genesis]' # + Genesis backend
|
|
124
|
+
pip install 'gauntlet-robotics[isaac]' # + Isaac Sim backend (CUDA required)
|
|
125
|
+
pip install 'gauntlet-robotics[monitor]' # + runtime drift detector (torch)
|
|
126
|
+
pip install 'gauntlet-robotics[ros2]' # + ROS 2 publish / record (rclpy via system pkg)
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
`uv` users: `uv add gauntlet-robotics` (same extras). Stand-alone CLI:
|
|
130
|
+
`uv tool install gauntlet-robotics` then `gauntlet --help`. Requires Python
|
|
131
|
+
≥3.11.
|
|
132
|
+
|
|
133
|
+
## Status
|
|
134
|
+
|
|
135
|
+
Phase 1 (MVP) and Phase 2 (real-policy adapters + runtime
|
|
136
|
+
observability) are shipped. Phase 1 covers the tabletop env, the
|
|
137
|
+
seven perturbation axes, the parallel Runner, the breakdown-first
|
|
138
|
+
HTML report, and the core `gauntlet run / report / compare` CLI.
|
|
139
|
+
Phase 2 adds the PyBullet / Genesis / Isaac Sim backends, OpenVLA
|
|
140
|
+
and SmolVLA adapters, runtime drift detection (`monitor`), ROS 2
|
|
141
|
+
publishing + recording, multi-camera observations, structured
|
|
142
|
+
per-axis report diffs (`gauntlet diff`), incremental rollout caching,
|
|
143
|
+
and the entry-point-based plugin system for third-party policies
|
|
144
|
+
and envs.
|
|
145
|
+
|
|
146
|
+
Phase 3 (fleet-scale tooling) is **partially shipped**: the
|
|
147
|
+
fleet-wide failure-mode aggregator (`gauntlet aggregate`), the
|
|
148
|
+
self-contained web dashboard, and the real-to-sim scene-ingestion
|
|
149
|
+
input pipeline are all live. The real-to-sim *renderer* itself is
|
|
150
|
+
deferred — `RealSimRenderer` lands as a `typing.Protocol` so a
|
|
151
|
+
gaussian-splatting (or other) renderer plugin can slot in without
|
|
152
|
+
touching the schema.
|
|
153
|
+
|
|
154
|
+
`0.2.0` is the first PyPI release: `pip install gauntlet-robotics`. From this
|
|
155
|
+
release onward, the documented public surface follows
|
|
156
|
+
[Semantic Versioning](https://semver.org/spec/v2.0.0.html) — the
|
|
157
|
+
full contract (which symbols are public, the on-disk schemas, the
|
|
158
|
+
CLI flags, and the deprecation policy) is in
|
|
159
|
+
[`docs/stability.md`](./docs/stability.md). Pin
|
|
160
|
+
`gauntlet>=0.2,<0.3` in your `pyproject.toml` and CI will not break
|
|
161
|
+
on a patch release.
|
|
162
|
+
|
|
163
|
+
## Backends
|
|
164
|
+
|
|
165
|
+
Gauntlet ships four simulator backends. The Suite YAML's `env:` key
|
|
166
|
+
is the dispatch: `tabletop` uses MuJoCo (default, ships in the core
|
|
167
|
+
install); `tabletop-pybullet`, `tabletop-genesis`, and
|
|
168
|
+
`tabletop-isaac` each live behind an optional extra.
|
|
169
|
+
|
|
170
|
+
| `env:` slug | Simulator | Install | Observations |
|
|
171
|
+
|----------------------|-----------|------------------------------------------|--------------|
|
|
172
|
+
| `tabletop` | MuJoCo | `uv sync` (core) | State + render-on-demand |
|
|
173
|
+
| `tabletop-pybullet` | PyBullet | `uv sync --extra pybullet` | State + render-on-demand |
|
|
174
|
+
| `tabletop-genesis` | Genesis | `uv sync --extra genesis` | State + render-on-demand |
|
|
175
|
+
| `tabletop-isaac` | Isaac Sim | `uv sync --extra isaac` (GPU required) | State-only (rendering follow-up) |
|
|
176
|
+
|
|
177
|
+
The four backends share action/observation spaces byte-for-byte and
|
|
178
|
+
the canonical 7 perturbation axes. They are **not** numerically
|
|
179
|
+
identical: same policy + same seed on `tabletop` vs `tabletop-pybullet`
|
|
180
|
+
vs `tabletop-genesis` vs `tabletop-isaac` produces semantically similar
|
|
181
|
+
but numerically different trajectories. Running `gauntlet compare`
|
|
182
|
+
across backends measures simulator drift, not policy regression; the
|
|
183
|
+
CLI requires `--allow-cross-backend` to proceed.
|
|
184
|
+
|
|
185
|
+
The `tabletop-isaac` backend wraps NVIDIA Omniverse Kit and **requires
|
|
186
|
+
a CUDA-capable RTX-class GPU at runtime**. The `[isaac]` extra resolves
|
|
187
|
+
on CPU-only machines but the Kit bootstrap inside `IsaacSimTabletopEnv.__init__`
|
|
188
|
+
fails without a GPU. CI tests use a `sys.modules`-injected fake
|
|
189
|
+
`isaacsim` namespace and do NOT install this extra; live execution
|
|
190
|
+
needs a developer GPU workstation. The state-only first cut declares
|
|
191
|
+
the four cosmetic axes (`lighting_intensity`, `camera_offset_x`,
|
|
192
|
+
`camera_offset_y`, `object_texture`) `VISUAL_ONLY_AXES` so cosmetic-only
|
|
193
|
+
sweeps are rejected at suite-load time on this backend until the
|
|
194
|
+
rendering follow-up RFC lands.
|
|
195
|
+
|
|
196
|
+
Image observations are available on all three backends via
|
|
197
|
+
`render_in_obs=True` / `render_size=(H, W)` on the env constructor
|
|
198
|
+
(`TabletopEnv`, `PyBulletTabletopEnv`, or `GenesisTabletopEnv`).
|
|
199
|
+
PyBullet uses a headless, deterministic TINY rasteriser; Genesis uses
|
|
200
|
+
its default CPU Rasterizer (pyrender-backed). The emitted `obs["image"]`
|
|
201
|
+
Box has shape / dtype / bounds byte-identical across backends, so VLA
|
|
202
|
+
adapters (OpenVLA, SmolVLA) work on any of them by swapping only the
|
|
203
|
+
env factory. Pixel values explicitly differ (different rasterisers —
|
|
204
|
+
semantic parity only). All seven perturbation axes produce observable
|
|
205
|
+
deltas on the rendered image on every backend; `VISUAL_ONLY_AXES` is
|
|
206
|
+
empty everywhere.
|
|
207
|
+
|
|
208
|
+
For multi-view policies (SmolVLA, ACT, Diffusion Policy — anything
|
|
209
|
+
that consumes paired wrist + side + overhead frames), pass
|
|
210
|
+
`cameras=[CameraSpec(...), ...]` to `TabletopEnv` or
|
|
211
|
+
`PyBulletTabletopEnv` instead. Each spec lands in
|
|
212
|
+
`obs["images"][name]`; `obs["image"]` stays populated as an alias to
|
|
213
|
+
the first camera so single-view consumers (the runner's video
|
|
214
|
+
recorder, OpenVLA-style adapters) keep working unchanged. The
|
|
215
|
+
single-camera default (`cameras=None`) is byte-identical to the
|
|
216
|
+
phase-1 contract — see
|
|
217
|
+
[`docs/polish-exploration-multi-camera.md`](./docs/polish-exploration-multi-camera.md)
|
|
218
|
+
for the full design and
|
|
219
|
+
[`examples/evaluate_multi_camera.py`](./examples/evaluate_multi_camera.py)
|
|
220
|
+
for a worked example.
|
|
221
|
+
|
|
222
|
+
See [`docs/phase2-rfc-005-pybullet-adapter.md`](./docs/phase2-rfc-005-pybullet-adapter.md)
|
|
223
|
+
for the full PyBullet backend design,
|
|
224
|
+
[`docs/phase2-rfc-006-pybullet-rendering.md`](./docs/phase2-rfc-006-pybullet-rendering.md)
|
|
225
|
+
for PyBullet's image-observation follow-up,
|
|
226
|
+
[`docs/phase2-rfc-007-genesis-adapter.md`](./docs/phase2-rfc-007-genesis-adapter.md)
|
|
227
|
+
for the Genesis backend design,
|
|
228
|
+
[`docs/phase2-rfc-008-genesis-rendering.md`](./docs/phase2-rfc-008-genesis-rendering.md)
|
|
229
|
+
for the Genesis image-observation follow-up, and
|
|
230
|
+
[`docs/phase2-rfc-009-isaac-sim-adapter.md`](./docs/phase2-rfc-009-isaac-sim-adapter.md)
|
|
231
|
+
for the Isaac Sim backend design.
|
|
232
|
+
|
|
233
|
+
## Quickstart
|
|
234
|
+
|
|
235
|
+
Three commands reproduce the end-to-end example against the bundled
|
|
236
|
+
smoke suite (3 lighting intensities x 2 cube textures x 4 episodes = 24
|
|
237
|
+
rollouts; finishes in seconds on a laptop):
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
uv sync
|
|
241
|
+
uv run gauntlet run examples/suites/tabletop-smoke.yaml --policy random --out out/
|
|
242
|
+
open out/report.html # macOS: open ; Linux: xdg-open ; Windows: start
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
Artefacts land in `out/`: `episodes.json` (one record per rollout),
|
|
246
|
+
`report.json` (analysed breakdowns), and `report.html` — a self-contained
|
|
247
|
+
report leading with the failure-clusters table, then per-axis bar charts,
|
|
248
|
+
then 2D heatmaps of axis combinations. The smoke suite is intentionally
|
|
249
|
+
tiny; for the canonical 4-axis x 144-cell x 1440-rollout shape, swap the
|
|
250
|
+
YAML path for `examples/suites/tabletop-basic-v1.yaml`. See
|
|
251
|
+
[`GAUNTLET_SPEC.md`](./GAUNTLET_SPEC.md) for the full design and
|
|
252
|
+
[`examples/evaluate_random_policy.py`](./examples/evaluate_random_policy.py)
|
|
253
|
+
for the equivalent invocation via the public Python API. For the
|
|
254
|
+
Genesis backend, `uv sync --extra genesis` then
|
|
255
|
+
[`examples/evaluate_random_policy_genesis.py`](./examples/evaluate_random_policy_genesis.py)
|
|
256
|
+
drives the same smoke suite against `tabletop-genesis`.
|
|
257
|
+
|
|
258
|
+
The Suite YAML's `sampling:` key picks the perturbation grid strategy:
|
|
259
|
+
the default `cartesian` enumerates the full Cartesian product of the
|
|
260
|
+
declared axes (the historical behaviour, byte-identical to every
|
|
261
|
+
existing suite), while `latin_hypercube` and `sobol` each draw
|
|
262
|
+
`n_samples` points without enumerating the full grid. For five axes at
|
|
263
|
+
five steps, cartesian = 3,125 cells; LHS or Sobol at `n_samples: 32`
|
|
264
|
+
covers the same hypercube at ~98x fewer rollouts. The two quasi-random
|
|
265
|
+
samplers trade off differently:
|
|
266
|
+
|
|
267
|
+
- `latin_hypercube` (McKay 1979) gives **perfect per-axis marginal
|
|
268
|
+
stratification** — every axis covers exactly `n_samples` distinct
|
|
269
|
+
strata. Joint coverage across axis pairs is essentially random.
|
|
270
|
+
- `sobol` (Joe-Kuo 6.21201 direction numbers, `skip=1`) gives
|
|
271
|
+
**low-discrepancy joint coverage** — Sobol projections onto any
|
|
272
|
+
axis pair are also quasi-uniform, at the cost of slightly worse
|
|
273
|
+
per-axis marginal histograms than LHS.
|
|
274
|
+
|
|
275
|
+
Use Sobol when the failure mode you suspect is a 2-axis (or higher)
|
|
276
|
+
interaction; use LHS when single-axis sweeps are what you need to
|
|
277
|
+
cover. See
|
|
278
|
+
[`examples/suites/tabletop-lhs-smoke.yaml`](./examples/suites/tabletop-lhs-smoke.yaml)
|
|
279
|
+
and [`examples/evaluate_random_policy_lhs.py`](./examples/evaluate_random_policy_lhs.py)
|
|
280
|
+
for an LHS end-to-end demo, and
|
|
281
|
+
[`docs/polish-exploration-sobol-sampler.md`](./docs/polish-exploration-sobol-sampler.md)
|
|
282
|
+
for the Sobol design note (discrepancy targets, direction-number
|
|
283
|
+
table, skip rationale).
|
|
284
|
+
|
|
285
|
+
Once you have multiple runs (different seeds, policy revisions, or
|
|
286
|
+
backends), `gauntlet aggregate <runs-dir> --out fleet/` rolls every
|
|
287
|
+
`report.json` recursively under `<runs-dir>` into a single fleet
|
|
288
|
+
meta-report — `fleet/fleet_report.json` plus a self-contained
|
|
289
|
+
`fleet/fleet_report.html` leading with the persistent failure
|
|
290
|
+
clusters that survive across runs (clusters appearing in at least
|
|
291
|
+
`--persistence-threshold` of the runs, default `0.5`). See
|
|
292
|
+
[`examples/aggregate_runs.py`](./examples/aggregate_runs.py) for the
|
|
293
|
+
equivalent invocation via the Python API and
|
|
294
|
+
[`docs/phase3-rfc-019-fleet-aggregate.md`](./docs/phase3-rfc-019-fleet-aggregate.md)
|
|
295
|
+
for the algorithm.
|
|
296
|
+
|
|
297
|
+
### Using a real VLA
|
|
298
|
+
|
|
299
|
+
- Install the HF extras: `uv sync --extra hf` (pulls torch / transformers / pillow; core installs stay torch-free).
|
|
300
|
+
- See [`examples/evaluate_openvla.py`](./examples/evaluate_openvla.py) for the ≤20-line OpenVLA-7B factory.
|
|
301
|
+
- Image-conditioned policies need a rendered frame — construct `TabletopEnv(render_in_obs=True)` so `obs["image"]` is emitted.
|
|
302
|
+
- **SmolVLA — read the warning first.** `lerobot/smolvla_base` is
|
|
303
|
+
pretrained on the SO-100 / SO-101 follower arm with **6-D
|
|
304
|
+
joint-position** actions; TabletopEnv is a **7-D EE-twist + gripper**
|
|
305
|
+
env. Zero-shot success on the smoke suite is **~0% by embodiment
|
|
306
|
+
mismatch — this is NOT a Gauntlet bug.** The example exists so users
|
|
307
|
+
who already have a TabletopEnv-compatible fine-tune know how to wire
|
|
308
|
+
the adapter (≤20 lines). For a first run that demonstrates the
|
|
309
|
+
harness end-to-end on a policy that solves the env, use
|
|
310
|
+
[the reference benchmark](#see-a-real-report-before-installing)
|
|
311
|
+
instead.
|
|
312
|
+
- With a fine-tune in hand: `uv sync --extra lerobot`;
|
|
313
|
+
see [`examples/evaluate_smolvla.py`](./examples/evaluate_smolvla.py)
|
|
314
|
+
(pass `--action-remap` / override `camera_keys` if your fine-tune
|
|
315
|
+
changed them). For the PyBullet backend,
|
|
316
|
+
`uv sync --extra lerobot --extra pybullet` and run
|
|
317
|
+
[`examples/evaluate_smolvla_pybullet.py`](./examples/evaluate_smolvla_pybullet.py).
|
|
318
|
+
Set `GAUNTLET_SUPPRESS_SMOLVLA_WARNING=1` to silence the runtime
|
|
319
|
+
banner once you've confirmed the embodiment fits.
|
|
320
|
+
|
|
321
|
+
### Runtime drift detection
|
|
322
|
+
|
|
323
|
+
Optional Phase 2 add-on (`[monitor]` extra). Given a reference sweep of a
|
|
324
|
+
known-good policy, fit a small observation autoencoder and score a
|
|
325
|
+
candidate sweep's trajectories against it — per-episode reconstruction
|
|
326
|
+
error + per-dim action-std surface OOD rollouts. `gauntlet run
|
|
327
|
+
--record-trajectories <dir>` dumps per-episode NPZ sidecars;
|
|
328
|
+
`gauntlet monitor train <dir> --out <ae_dir>` fits the AE; `gauntlet
|
|
329
|
+
monitor score <episodes.json> <dir> --ae <ae_dir> --out drift.json`
|
|
330
|
+
writes the sidecar. The three-step workflow is scripted end-to-end in
|
|
331
|
+
[`examples/evaluate_with_drift.py`](./examples/evaluate_with_drift.py);
|
|
332
|
+
`drift.json` is optional and orthogonal to `report.json`.
|
|
333
|
+
|
|
334
|
+
### ROS 2 integration
|
|
335
|
+
|
|
336
|
+
Optional Phase 2 add-on (`[ros2]` extra). Two halves wire gauntlet into a
|
|
337
|
+
ROS 2 graph:
|
|
338
|
+
|
|
339
|
+
- `gauntlet ros2 publish episodes.json --topic /gauntlet/episodes`
|
|
340
|
+
serialises each Episode as JSON inside `std_msgs/msg/String` and
|
|
341
|
+
publishes one message per Episode. Useful for fleet-wide failure-mode
|
|
342
|
+
aggregation across many real robots running gauntlet evaluations.
|
|
343
|
+
- `gauntlet ros2 record --topic /robot/joint_states --out trajectory.jsonl
|
|
344
|
+
--duration 30` subscribes to a real robot's topic and dumps each
|
|
345
|
+
received message to a JSONL file on disk. Useful for the "real robots
|
|
346
|
+
with logging" half of `GAUNTLET_SPEC.md` §7.
|
|
347
|
+
|
|
348
|
+
Because `rclpy` is **not** distributed via PyPI in its official form, the
|
|
349
|
+
`[ros2]` extra is empty — `uv sync --extra ros2` is a no-op beyond the
|
|
350
|
+
dev tooling. Install ROS 2 (Humble or Jazzy) via your system package
|
|
351
|
+
manager, e.g. `sudo apt install ros-humble-rclpy`, or run inside the
|
|
352
|
+
official Docker image (`docker run -it osrf/ros:humble-desktop`), then
|
|
353
|
+
source the relevant `setup.bash` before invoking `gauntlet ros2`. The
|
|
354
|
+
`--dry-run` flag on `gauntlet ros2 publish` short-circuits the rclpy
|
|
355
|
+
import so you can preview the JSON payloads without installing ROS 2.
|
|
356
|
+
|
|
357
|
+
The publisher / recorder API is documented in
|
|
358
|
+
[`docs/phase2-rfc-010-ros2-integration.md`](./docs/phase2-rfc-010-ros2-integration.md);
|
|
359
|
+
see [`examples/publish_episodes_to_ros2.py`](./examples/publish_episodes_to_ros2.py)
|
|
360
|
+
for the equivalent invocation via the public Python API.
|
|
361
|
+
|
|
362
|
+
### Diffing two runs
|
|
363
|
+
|
|
364
|
+
`gauntlet compare a.json b.json` answers a binary question (did `b`
|
|
365
|
+
regress against `a` beyond a threshold?). When you're iterating on a
|
|
366
|
+
checkpoint and want a structured, `git diff`-style breakdown of *what*
|
|
367
|
+
moved — per-axis-value rate deltas, per-cell success-rate flips, and the
|
|
368
|
+
failure-cluster set difference — reach for `gauntlet diff`:
|
|
369
|
+
|
|
370
|
+
```bash
|
|
371
|
+
uv run gauntlet diff out_a/report.json out_b/report.json
|
|
372
|
+
# Or feed episodes.json directly (auto-detected, parity with `compare`):
|
|
373
|
+
uv run gauntlet diff out_a/episodes.json out_b/episodes.json --json | jq
|
|
374
|
+
```
|
|
375
|
+
|
|
376
|
+
Threshold flags `--cell-flip-threshold` (default `0.10`) and
|
|
377
|
+
`--cluster-intensify-threshold` (default `0.5`) gate the per-cell and
|
|
378
|
+
per-cluster surfacings. Default output is human-readable text on stdout;
|
|
379
|
+
`--json` emits the full `ReportDiff` payload for downstream consumption.
|
|
380
|
+
See [`examples/diff_two_runs.py`](./examples/diff_two_runs.py) for the
|
|
381
|
+
equivalent invocation via the public Python API.
|
|
382
|
+
|
|
383
|
+
### Debugging failures with replay
|
|
384
|
+
|
|
385
|
+
Once a run has flagged an episode as failing, `gauntlet replay` re-
|
|
386
|
+
simulates exactly that rollout with the same seed, optionally nudging
|
|
387
|
+
one axis off the original grid:
|
|
388
|
+
|
|
389
|
+
```bash
|
|
390
|
+
uv run gauntlet replay out/episodes.json \
|
|
391
|
+
--suite examples/suites/tabletop-smoke.yaml \
|
|
392
|
+
--policy scripted \
|
|
393
|
+
--episode-id 3:1 \
|
|
394
|
+
--override lighting_intensity=1.2 \
|
|
395
|
+
--out out/replay.json
|
|
396
|
+
```
|
|
397
|
+
|
|
398
|
+
Zero-override replay is bit-identical to the original episode; any
|
|
399
|
+
deviation points at a real reproducibility bug. See
|
|
400
|
+
[`examples/replay_failure.py`](./examples/replay_failure.py) for the
|
|
401
|
+
equivalent library call.
|
|
402
|
+
|
|
403
|
+
### Recording rollout videos
|
|
404
|
+
|
|
405
|
+
Failure analytics are far more actionable when a human can *watch*
|
|
406
|
+
the broken rollout. Opt in to the `[video]` extra to dump one MP4 per
|
|
407
|
+
episode and surface inline `<video>` thumbnails in the failure-
|
|
408
|
+
clusters table of the HTML report:
|
|
409
|
+
|
|
410
|
+
```bash
|
|
411
|
+
uv sync --extra video
|
|
412
|
+
uv run python examples/evaluate_random_policy_with_video.py --out out
|
|
413
|
+
# Open out/report.html — the failure-clusters table now embeds
|
|
414
|
+
# clickable thumbnails of every failed rollout.
|
|
415
|
+
```
|
|
416
|
+
|
|
417
|
+
The `[video]` extra pulls `imageio[ffmpeg]`, which bundles a static
|
|
418
|
+
ffmpeg binary — no system ffmpeg install required. Pass
|
|
419
|
+
`--only-failures` to suppress MP4 writes for successful episodes
|
|
420
|
+
(saves disk on long sweeps). The Runner asserts the env was
|
|
421
|
+
constructed with `render_in_obs=True` when `record_video=True`; the
|
|
422
|
+
example wires that automatically.
|
|
423
|
+
|
|
424
|
+
### Fleet dashboard
|
|
425
|
+
|
|
426
|
+
Once you've accumulated more than a few `report.json` files
|
|
427
|
+
(different policies, different seeds, nightly runs), eyeballing each
|
|
428
|
+
HTML report individually stops scaling. `gauntlet dashboard build`
|
|
429
|
+
materialises a self-contained static SPA that indexes every
|
|
430
|
+
`report.json` under a directory:
|
|
431
|
+
|
|
432
|
+
```bash
|
|
433
|
+
gauntlet dashboard build runs/ --out dashboard-out/
|
|
434
|
+
# Open dashboard-out/index.html via file:// — no web server needed.
|
|
435
|
+
```
|
|
436
|
+
|
|
437
|
+
The Python API is also exposed for notebook / custom-pipeline use:
|
|
438
|
+
|
|
439
|
+
```python
|
|
440
|
+
from pathlib import Path
|
|
441
|
+
from gauntlet.dashboard import build_dashboard
|
|
442
|
+
|
|
443
|
+
build_dashboard(Path("runs/"), Path("dashboard-out/"))
|
|
444
|
+
```
|
|
445
|
+
|
|
446
|
+
The output directory contains exactly three files (`index.html`,
|
|
447
|
+
`dashboard.js`, `dashboard.css`); all run data is embedded as an
|
|
448
|
+
inline JSON literal so the SPA opens straight off the filesystem
|
|
449
|
+
without tripping CORS. The dashboard surfaces an index card
|
|
450
|
+
(n_runs / n_episodes / mean ± std success rate), a per-run table
|
|
451
|
+
filterable by env / suite / policy, a time-series chart of success
|
|
452
|
+
rate keyed off `report.json` mtime, and per-axis aggregate bars
|
|
453
|
+
pooled across the matching runs. Sibling `report.html` files (from
|
|
454
|
+
the originating `gauntlet run`) are auto-linked from each row. See
|
|
455
|
+
[`docs/phase3-rfc-020-web-dashboard.md`](./docs/phase3-rfc-020-web-dashboard.md)
|
|
456
|
+
for the full design.
|
|
457
|
+
|
|
458
|
+
### Real-to-sim scene ingestion
|
|
459
|
+
|
|
460
|
+
The endgame for `GAUNTLET_SPEC.md` §7 is gaussian-splatting
|
|
461
|
+
reconstruction of customer scenes from real-robot camera dumps
|
|
462
|
+
straight into a renderable eval backend. Shipping the renderer
|
|
463
|
+
itself needs `torch` + CUDA + a multi-gigabyte training pipeline,
|
|
464
|
+
which violates spec §6 — so this release lands the *input pipeline*
|
|
465
|
+
and the *renderer extension point* only. A plugin (or a future
|
|
466
|
+
in-tree RFC) implements an actual renderer against the
|
|
467
|
+
`RealSimRenderer` Protocol without touching the schema or the CLI:
|
|
468
|
+
|
|
469
|
+
```bash
|
|
470
|
+
uv run gauntlet realsim ingest <frames-dir> \
|
|
471
|
+
--calib <calib.json> \
|
|
472
|
+
--out <scene-dir>
|
|
473
|
+
|
|
474
|
+
uv run gauntlet realsim info <scene-dir>
|
|
475
|
+
```
|
|
476
|
+
|
|
477
|
+
`ingest` validates the frames + calibration JSON and writes a
|
|
478
|
+
self-contained scene directory (`manifest.json` + frame copies, or
|
|
479
|
+
symlinks via `--symlink`). The manifest carries `Pose` (4x4
|
|
480
|
+
row-major rigid transforms, NeRFStudio / COLMAP `transforms.json`
|
|
481
|
+
convention), `CameraIntrinsics` (pinhole + optional distortion,
|
|
482
|
+
shared by id), and `CameraFrame` rows. `info` prints a one-screen
|
|
483
|
+
manifest summary. The renderer itself is **deferred** — `RealSimRenderer`
|
|
484
|
+
is a `typing.Protocol`, and `register_renderer` / `get_renderer` are a
|
|
485
|
+
module-local registry for plugin renderers. See
|
|
486
|
+
[`docs/phase3-rfc-021-real-to-sim-stub.md`](./docs/phase3-rfc-021-real-to-sim-stub.md)
|
|
487
|
+
for the full design (pose representation, validation rules, plugin
|
|
488
|
+
seam).
|
|
489
|
+
|
|
490
|
+
### Multi-camera observations
|
|
491
|
+
|
|
492
|
+
Multi-view policies (SmolVLA, ACT, Diffusion Policy — anything that
|
|
493
|
+
consumes paired wrist + side + overhead frames) need more than the
|
|
494
|
+
single `obs["image"]` the legacy `render_in_obs=True` path emits.
|
|
495
|
+
Pass `cameras=[CameraSpec(...), ...]` to `TabletopEnv` or
|
|
496
|
+
`PyBulletTabletopEnv` and each spec lands in `obs["images"][name]`:
|
|
497
|
+
|
|
498
|
+
```python
|
|
499
|
+
from gauntlet.env import CameraSpec, TabletopEnv
|
|
500
|
+
|
|
501
|
+
env = TabletopEnv(
|
|
502
|
+
cameras=[
|
|
503
|
+
CameraSpec(name="wrist", pose=(0.0, 0.0, 0.4, 0.0, 0.0, 0.0), size=(96, 96)),
|
|
504
|
+
CameraSpec(name="side", pose=(0.5, 0.0, 0.3, 0.0, 1.2, 0.0), size=(96, 96)),
|
|
505
|
+
],
|
|
506
|
+
)
|
|
507
|
+
obs, _ = env.reset(seed=0)
|
|
508
|
+
wrist = obs["images"]["wrist"] # shape (96, 96, 3), uint8
|
|
509
|
+
```
|
|
510
|
+
|
|
511
|
+
`CameraSpec.pose` is `(x, y, z, rx, ry, rz)` in metres + MuJoCo-XYZ
|
|
512
|
+
Euler radians (looks along local `-Z`); `CameraSpec.size` is `(H, W)`.
|
|
513
|
+
The legacy `obs["image"]` key stays populated as an alias to the
|
|
514
|
+
**first** camera's frame so single-view consumers (the runner's video
|
|
515
|
+
recorder, OpenVLA-style adapters) keep working unchanged. The
|
|
516
|
+
single-camera default (`cameras=None`) is byte-identical to the
|
|
517
|
+
phase-1 contract. See
|
|
518
|
+
[`docs/polish-exploration-multi-camera.md`](./docs/polish-exploration-multi-camera.md)
|
|
519
|
+
for the full design and
|
|
520
|
+
[`examples/evaluate_multi_camera.py`](./examples/evaluate_multi_camera.py)
|
|
521
|
+
for a worked example.
|
|
522
|
+
|
|
523
|
+
## Extending gauntlet
|
|
524
|
+
|
|
525
|
+
Third-party policies and envs plug into gauntlet through Python's
|
|
526
|
+
standard `importlib.metadata` entry-point mechanism — any
|
|
527
|
+
pip-installable package can register itself without modifying
|
|
528
|
+
gauntlet's source. Two groups are read by `gauntlet.plugins`:
|
|
529
|
+
|
|
530
|
+
| Entry-point group | Registers |
|
|
531
|
+
|---------------------|-------------------------------------------------------------------------|
|
|
532
|
+
| `gauntlet.policies` | A class (or zero-arg callable) returning a `gauntlet.policy.base.Policy` |
|
|
533
|
+
| `gauntlet.envs` | A class returning a `gauntlet.env.base.GauntletEnv` |
|
|
534
|
+
|
|
535
|
+
A plugin author writes the adapter, then declares it in their own
|
|
536
|
+
`pyproject.toml`:
|
|
537
|
+
|
|
538
|
+
```toml
|
|
539
|
+
[project.entry-points."gauntlet.policies"]
|
|
540
|
+
sb3 = "my_gauntlet_plugin.sb3_adapter:SBAdapter"
|
|
541
|
+
```
|
|
542
|
+
|
|
543
|
+
After `pip install my-gauntlet-plugin`, `gauntlet run ... --policy sb3`
|
|
544
|
+
resolves through the plugin path. Built-in adapters always win on
|
|
545
|
+
collision; failed entry-point loads are wrapped in a
|
|
546
|
+
`RuntimeWarning` and dropped from the registry — gauntlet itself
|
|
547
|
+
stays operational. See
|
|
548
|
+
[`docs/plugin-development.md`](./docs/plugin-development.md) for the
|
|
549
|
+
full how-to (writing a Policy / Env plugin, constructor-argument
|
|
550
|
+
patterns, testing) and
|
|
551
|
+
[`docs/polish-exploration-plugin-system.md`](./docs/polish-exploration-plugin-system.md)
|
|
552
|
+
for the design note (precedence rules, lazy discovery, collision
|
|
553
|
+
handling).
|
|
554
|
+
|
|
555
|
+
## Development
|
|
556
|
+
|
|
557
|
+
```bash
|
|
558
|
+
# Sync deps (creates .venv, installs everything in pyproject + dev group).
|
|
559
|
+
uv sync
|
|
560
|
+
|
|
561
|
+
# Lint, type-check, test.
|
|
562
|
+
uv run ruff check .
|
|
563
|
+
uv run mypy
|
|
564
|
+
uv run pytest
|
|
565
|
+
```
|
|
566
|
+
|
|
567
|
+
### Property tests
|
|
568
|
+
|
|
569
|
+
The `tests/test_property_*.py` and `tests/test_fuzz_*.py` files use
|
|
570
|
+
[Hypothesis](https://hypothesis.readthedocs.io) to fuzz invariants
|
|
571
|
+
that hand-rolled tests would only sample. They are tagged with the
|
|
572
|
+
`hypothesis_property` pytest marker so a focused run picks them up
|
|
573
|
+
without running the full suite:
|
|
574
|
+
|
|
575
|
+
```bash
|
|
576
|
+
# Run only the property/fuzz tests.
|
|
577
|
+
uv run pytest -m hypothesis_property
|
|
578
|
+
|
|
579
|
+
# Faster CI profile (max_examples=50 instead of the 200 default).
|
|
580
|
+
GAUNTLET_HYPOTHESIS_PROFILE=ci uv run pytest -m hypothesis_property
|
|
581
|
+
```
|
|
582
|
+
|
|
583
|
+
The covered invariants include:
|
|
584
|
+
|
|
585
|
+
* **Perturbation samplers** (`test_property_perturbation.py`,
|
|
586
|
+
`test_property_axis_bounds.py`) — every emitted axis value lies
|
|
587
|
+
inside the declared `[low, high]` envelope; same rng seed produces
|
|
588
|
+
the same value.
|
|
589
|
+
* **Suite YAML round-trip** (`test_fuzz_suite_roundtrip.py`,
|
|
590
|
+
`test_property_suite_loader.py`) — `Suite -> YAML -> Suite` is the
|
|
591
|
+
identity for cartesian / LHS / Sobol suites; insertion order is
|
|
592
|
+
preserved.
|
|
593
|
+
* **Action clipping** (`test_property_action_clipping.py`,
|
|
594
|
+
`test_fuzz_action_space.py`) — finite-but-extreme action vectors
|
|
595
|
+
produce per-step mocap deltas bounded by `MAX_LINEAR_STEP`. NaN/Inf
|
|
596
|
+
inputs are pinned via `pytest.xfail` as a spec for future hardening.
|
|
597
|
+
* **Reset ordering** (`test_property_reset_after_step_ordering.py`)
|
|
598
|
+
— `env.reset(S) -> step* -> env.reset(S)` produces bit-equal
|
|
599
|
+
starting obs regardless of the intervening actions.
|
|
600
|
+
* **Observation NaN/Inf detection**
|
|
601
|
+
(`test_property_observation_validation.py`) — spec-only via
|
|
602
|
+
`pytest.xfail`; pins the gap that `Episode.observation_invalid`
|
|
603
|
+
does not yet exist.
|
|
604
|
+
|
|
605
|
+
## Project layout
|
|
606
|
+
|
|
607
|
+
```
|
|
608
|
+
src/gauntlet/
|
|
609
|
+
policy/ # Policy adapter protocol + reference wrappers (Random, Scripted, HF, LeRobot)
|
|
610
|
+
env/ # Parameterized envs — MuJoCo (core) + PyBullet/Genesis/Isaac (extras)
|
|
611
|
+
suite/ # YAML-defined perturbation grid suites (cartesian / LHS / Sobol)
|
|
612
|
+
runner/ # Parallel rollout orchestration + seed management + cache
|
|
613
|
+
report/ # Per-run failure analysis + HTML/JSON generation
|
|
614
|
+
monitor/ # Runtime drift detection + action-entropy ([monitor] extra)
|
|
615
|
+
replay/ # Single-episode replay with axis overrides
|
|
616
|
+
ros2/ # ROS 2 publisher + recorder ([ros2] extra; rclpy via apt/Docker)
|
|
617
|
+
diff/ # Structured per-axis report deltas powering `gauntlet diff`
|
|
618
|
+
aggregate/ # Fleet-wide failure-mode clustering across many runs
|
|
619
|
+
dashboard/ # Self-contained static SPA indexing every report.json
|
|
620
|
+
realsim/ # Real-to-sim scene ingestion + RealSimRenderer Protocol (renderer deferred)
|
|
621
|
+
plugins.py # Entry-point discovery for third-party policies / envs
|
|
622
|
+
cli.py # gauntlet run / report / compare / diff / aggregate /
|
|
623
|
+
# dashboard / realsim / monitor / replay / ros2
|
|
624
|
+
```
|
|
625
|
+
|
|
626
|
+
## License
|
|
627
|
+
|
|
628
|
+
MIT.
|