fieldtrial 0.1.0.dev0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- fieldtrial-0.2.0/.gitignore +39 -0
- fieldtrial-0.2.0/CHANGELOG.md +181 -0
- fieldtrial-0.2.0/PKG-INFO +204 -0
- fieldtrial-0.2.0/README.md +137 -0
- fieldtrial-0.2.0/pyproject.toml +368 -0
- fieldtrial-0.2.0/src/fieldtrial/__init__.py +13 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/__init__.py +1 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/common.py +20 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/crossover.py +83 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/engine.py +1222 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/ladder.py +103 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/records.py +57 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/results.py +409 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/sequential.py +281 -0
- fieldtrial-0.2.0/src/fieldtrial/analysis/wording.py +400 -0
- fieldtrial-0.2.0/src/fieldtrial/api/__init__.py +1 -0
- fieldtrial-0.2.0/src/fieldtrial/api/schemas.py +213 -0
- fieldtrial-0.2.0/src/fieldtrial/api/v1.py +494 -0
- fieldtrial-0.2.0/src/fieldtrial/capture/__init__.py +1 -0
- fieldtrial-0.2.0/src/fieldtrial/capture/drift.py +193 -0
- fieldtrial-0.2.0/src/fieldtrial/capture/images.py +32 -0
- fieldtrial-0.2.0/src/fieldtrial/capture/recorder.py +208 -0
- fieldtrial-0.2.0/src/fieldtrial/cli/__init__.py +1 -0
- fieldtrial-0.2.0/src/fieldtrial/cli/calc.py +299 -0
- fieldtrial-0.2.0/src/fieldtrial/cli/main.py +42 -0
- fieldtrial-0.2.0/src/fieldtrial/cli/serve.py +112 -0
- fieldtrial-0.2.0/src/fieldtrial/cli/study.py +536 -0
- fieldtrial-0.2.0/src/fieldtrial/client.py +283 -0
- fieldtrial-0.2.0/src/fieldtrial/design/__init__.py +32 -0
- fieldtrial-0.2.0/src/fieldtrial/design/blinding.py +26 -0
- fieldtrial-0.2.0/src/fieldtrial/design/capture_config.py +38 -0
- fieldtrial-0.2.0/src/fieldtrial/design/hashing.py +47 -0
- fieldtrial-0.2.0/src/fieldtrial/design/loader.py +135 -0
- fieldtrial-0.2.0/src/fieldtrial/design/models.py +396 -0
- fieldtrial-0.2.0/src/fieldtrial/design/runner_config.py +171 -0
- fieldtrial-0.2.0/src/fieldtrial/design/schedule.py +170 -0
- fieldtrial-0.2.0/src/fieldtrial/io/__init__.py +1 -0
- fieldtrial-0.2.0/src/fieldtrial/io/csv.py +219 -0
- fieldtrial-0.2.0/src/fieldtrial/io/jsonl.py +17 -0
- fieldtrial-0.2.0/src/fieldtrial/io/lerobot.py +133 -0
- fieldtrial-0.2.0/src/fieldtrial/py.typed +0 -0
- fieldtrial-0.2.0/src/fieldtrial/report/__init__.py +6 -0
- fieldtrial-0.2.0/src/fieldtrial/report/charts.py +462 -0
- fieldtrial-0.2.0/src/fieldtrial/report/html.py +79 -0
- fieldtrial-0.2.0/src/fieldtrial/report/markdown.py +493 -0
- fieldtrial-0.2.0/src/fieldtrial/report/templates/report.css +25 -0
- fieldtrial-0.2.0/src/fieldtrial/report/templates/report.html +280 -0
- fieldtrial-0.2.0/src/fieldtrial/runners/__init__.py +1 -0
- fieldtrial-0.2.0/src/fieldtrial/runners/base.py +98 -0
- fieldtrial-0.2.0/src/fieldtrial/runners/command.py +311 -0
- fieldtrial-0.2.0/src/fieldtrial/runners/manual.py +55 -0
- fieldtrial-0.2.0/src/fieldtrial/runners/openpi_router.py +315 -0
- fieldtrial-0.2.0/src/fieldtrial/runners/sim.py +133 -0
- fieldtrial-0.2.0/src/fieldtrial/services/__init__.py +14 -0
- fieldtrial-0.2.0/src/fieldtrial/services/_context.py +61 -0
- fieldtrial-0.2.0/src/fieldtrial/services/analysis.py +33 -0
- fieldtrial-0.2.0/src/fieldtrial/services/dataset.py +152 -0
- fieldtrial-0.2.0/src/fieldtrial/services/events.py +110 -0
- fieldtrial-0.2.0/src/fieldtrial/services/interim.py +133 -0
- fieldtrial-0.2.0/src/fieldtrial/services/registry.py +95 -0
- fieldtrial-0.2.0/src/fieldtrial/services/rig.py +170 -0
- fieldtrial-0.2.0/src/fieldtrial/services/runner_check.py +96 -0
- fieldtrial-0.2.0/src/fieldtrial/services/session.py +206 -0
- fieldtrial-0.2.0/src/fieldtrial/services/simulate.py +197 -0
- fieldtrial-0.2.0/src/fieldtrial/services/study.py +508 -0
- fieldtrial-0.2.0/src/fieldtrial/services/transfer.py +136 -0
- fieldtrial-0.2.0/src/fieldtrial/services/trial.py +940 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/__init__.py +109 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/_rng.py +72 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/_types.py +237 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/_validation.py +48 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/bayes.py +82 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/compare.py +206 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/crossover.py +247 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/drift.py +80 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/ladder.py +357 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/multiplicity.py +96 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/ordinal.py +139 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/paired.py +244 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/power.py +501 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/proportions.py +160 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/sequential.py +475 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/stratified.py +127 -0
- fieldtrial-0.2.0/src/fieldtrial/stats/timing.py +103 -0
- fieldtrial-0.2.0/src/fieldtrial/store/__init__.py +1 -0
- fieldtrial-0.2.0/src/fieldtrial/store/db.py +46 -0
- fieldtrial-0.2.0/src/fieldtrial/store/ids.py +17 -0
- fieldtrial-0.2.0/src/fieldtrial/store/migrations/__init__.py +0 -0
- fieldtrial-0.2.0/src/fieldtrial/store/migrations/env.py +14 -0
- fieldtrial-0.2.0/src/fieldtrial/store/migrations/script.py.mako +21 -0
- fieldtrial-0.2.0/src/fieldtrial/store/migrations/versions/0001_baseline.py +182 -0
- fieldtrial-0.2.0/src/fieldtrial/store/migrations/versions/0002_idempotency.py +24 -0
- fieldtrial-0.2.0/src/fieldtrial/store/migrations/versions/__init__.py +0 -0
- fieldtrial-0.2.0/src/fieldtrial/store/models.py +188 -0
- fieldtrial-0.2.0/src/fieldtrial/templates/__init__.py +15 -0
- fieldtrial-0.2.0/src/fieldtrial/templates/basic.yaml +58 -0
- fieldtrial-0.2.0/src/fieldtrial/templates/checkpoint-ladder.yaml +45 -0
- fieldtrial-0.2.0/src/fieldtrial/templates/crossover-rounds.yaml +43 -0
- fieldtrial-0.2.0/src/fieldtrial/templates/demo.yaml +59 -0
- fieldtrial-0.2.0/src/fieldtrial/templates/serving-sweep.yaml +41 -0
- fieldtrial-0.2.0/src/fieldtrial/web/__init__.py +1 -0
- fieldtrial-0.2.0/src/fieldtrial/web/app.py +98 -0
- fieldtrial-0.2.0/src/fieldtrial/web/console.py +536 -0
- fieldtrial-0.2.0/src/fieldtrial/web/runners.py +340 -0
- fieldtrial-0.2.0/src/fieldtrial/web/security.py +169 -0
- fieldtrial-0.2.0/src/fieldtrial/web/static/app.css +174 -0
- fieldtrial-0.2.0/src/fieldtrial/web/static/app.js +216 -0
- fieldtrial-0.2.0/src/fieldtrial/web/static/vendor/README.md +11 -0
- fieldtrial-0.2.0/src/fieldtrial/web/static/vendor/htmx-2.0.11.min.js +1 -0
- fieldtrial-0.2.0/src/fieldtrial/web/static/vendor/htmx-LICENSE.txt +13 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/_edit_row.html +38 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/_flash.html +1 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/_panel.html +125 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/_slot_header.html +12 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/_trial_header.html +2 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/_trial_row.html +12 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/base.html +24 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/console.html +27 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/draft.html +13 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/error.html +6 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/history.html +20 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/report.html +54 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/studies.html +19 -0
- fieldtrial-0.2.0/src/fieldtrial/web/templates/study.html +85 -0
- fieldtrial-0.2.0/tests/analysis/test_engine.py +361 -0
- fieldtrial-0.2.0/tests/analysis/test_report.py +72 -0
- fieldtrial-0.2.0/tests/analysis/test_schema.py +19 -0
- fieldtrial-0.2.0/tests/analysis/test_wording.py +210 -0
- fieldtrial-0.2.0/tests/capture/test_drift.py +118 -0
- fieldtrial-0.2.0/tests/capture/test_recorder.py +133 -0
- fieldtrial-0.2.0/tests/conftest.py +50 -0
- fieldtrial-0.2.0/tests/design/snapshots/schedules.json +3580 -0
- fieldtrial-0.2.0/tests/design/test_models.py +280 -0
- fieldtrial-0.2.0/tests/design/test_schedule.py +232 -0
- fieldtrial-0.2.0/tests/fixtures/golden/study.yaml +30 -0
- fieldtrial-0.2.0/tests/fixtures/golden/trials.csv +201 -0
- fieldtrial-0.2.0/tests/io/test_lerobot_links.py +161 -0
- fieldtrial-0.2.0/tests/report/test_html.py +166 -0
- fieldtrial-0.2.0/tests/runners/test_command.py +274 -0
- fieldtrial-0.2.0/tests/runners/test_openpi_router.py +200 -0
- fieldtrial-0.2.0/tests/services/conftest.py +31 -0
- fieldtrial-0.2.0/tests/services/test_console_services.py +272 -0
- fieldtrial-0.2.0/tests/services/test_registry.py +41 -0
- fieldtrial-0.2.0/tests/services/test_rig.py +109 -0
- fieldtrial-0.2.0/tests/services/test_simulate.py +153 -0
- fieldtrial-0.2.0/tests/services/test_store.py +80 -0
- fieldtrial-0.2.0/tests/services/test_study_services.py +164 -0
- fieldtrial-0.2.0/tests/services/test_transfer.py +119 -0
- fieldtrial-0.2.0/tests/services/test_trial_services.py +194 -0
- fieldtrial-0.2.0/tests/services/test_v02_designs.py +238 -0
- fieldtrial-0.2.0/tests/stats/test_bayes.py +71 -0
- fieldtrial-0.2.0/tests/stats/test_compare.py +159 -0
- fieldtrial-0.2.0/tests/stats/test_crossover.py +180 -0
- fieldtrial-0.2.0/tests/stats/test_ladder.py +205 -0
- fieldtrial-0.2.0/tests/stats/test_m2_stats.py +214 -0
- fieldtrial-0.2.0/tests/stats/test_multiplicity.py +91 -0
- fieldtrial-0.2.0/tests/stats/test_paired.py +191 -0
- fieldtrial-0.2.0/tests/stats/test_power.py +252 -0
- fieldtrial-0.2.0/tests/stats/test_proportions.py +166 -0
- fieldtrial-0.2.0/tests/stats/test_sequential.py +218 -0
- fieldtrial-0.2.0/tests/stats/test_slow_properties.py +73 -0
- fieldtrial-0.2.0/tests/test_architecture.py +110 -0
- fieldtrial-0.2.0/tests/test_calc_cli.py +150 -0
- fieldtrial-0.2.0/tests/test_cli.py +48 -0
- fieldtrial-0.2.0/tests/test_examples.py +66 -0
- fieldtrial-0.2.0/tests/test_network_guard.py +27 -0
- fieldtrial-0.2.0/tests/test_release_tools.py +55 -0
- fieldtrial-0.2.0/tests/test_study_cli.py +146 -0
- fieldtrial-0.2.0/tests/web/conftest.py +33 -0
- fieldtrial-0.2.0/tests/web/test_api.py +272 -0
- fieldtrial-0.2.0/tests/web/test_capture_web.py +129 -0
- fieldtrial-0.2.0/tests/web/test_client.py +85 -0
- fieldtrial-0.2.0/tests/web/test_console.py +450 -0
- fieldtrial-0.2.0/tests/web/test_runners_web.py +239 -0
- fieldtrial-0.2.0/tests/web/test_serve.py +117 -0
- fieldtrial-0.2.0/tests/web/test_v02_web.py +109 -0
- fieldtrial-0.1.0.dev0/PKG-INFO +0 -22
- fieldtrial-0.1.0.dev0/README.md +0 -9
- fieldtrial-0.1.0.dev0/pyproject.toml +0 -27
- fieldtrial-0.1.0.dev0/src/fieldtrial/__init__.py +0 -6
- {fieldtrial-0.1.0.dev0 → fieldtrial-0.2.0}/LICENSE +0 -0
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# Secrets: tokens and credentials stay local
|
|
2
|
+
.env
|
|
3
|
+
.env.*
|
|
4
|
+
!.env.example
|
|
5
|
+
.pypirc
|
|
6
|
+
|
|
7
|
+
# Python bytecode
|
|
8
|
+
__pycache__/
|
|
9
|
+
*.py[cod]
|
|
10
|
+
|
|
11
|
+
# Virtual environments (uv)
|
|
12
|
+
.venv/
|
|
13
|
+
venv/
|
|
14
|
+
|
|
15
|
+
# Build artifacts (uv build / hatchling)
|
|
16
|
+
build/
|
|
17
|
+
dist/
|
|
18
|
+
*.egg-info/
|
|
19
|
+
*.whl
|
|
20
|
+
|
|
21
|
+
# Test, lint and type-check caches
|
|
22
|
+
.pytest_cache/
|
|
23
|
+
.hypothesis/
|
|
24
|
+
.mypy_cache/
|
|
25
|
+
.ruff_cache/
|
|
26
|
+
.import_linter_cache/
|
|
27
|
+
.coverage
|
|
28
|
+
.coverage.*
|
|
29
|
+
htmlcov/
|
|
30
|
+
coverage.xml
|
|
31
|
+
|
|
32
|
+
# Docs build (mkdocs)
|
|
33
|
+
/site/
|
|
34
|
+
|
|
35
|
+
# Editors and OS clutter
|
|
36
|
+
.idea/
|
|
37
|
+
.vscode/
|
|
38
|
+
.DS_Store
|
|
39
|
+
Thumbs.db
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this
|
|
6
|
+
project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). Before 1.0,
|
|
7
|
+
minor releases may contain breaking changes.
|
|
8
|
+
|
|
9
|
+
## [Unreleased]
|
|
10
|
+
|
|
11
|
+
## [0.2.0] - 2026-10-02
|
|
12
|
+
|
|
13
|
+
Second release: crossover rounds, checkpoint ladders and group-sequential stopping; real
|
|
14
|
+
blinding with switching runners; evaluation-camera capture, rig drift checks and links to
|
|
15
|
+
LeRobot datasets. Studies, design hashes and results files from 0.1 keep working.
|
|
16
|
+
|
|
17
|
+
### Added
|
|
18
|
+
|
|
19
|
+
- Crossover rounds (`design.type: crossover_rounds`) for tasks whose scene carries over
|
|
20
|
+
between trials. Each arm runs whole rounds in cycles of two, in randomized, balanced
|
|
21
|
+
order. The analysis is a period-adjusted difference with an exact randomization test,
|
|
22
|
+
and its interval inverts the same test. A new `crossover-rounds` template. The console
|
|
23
|
+
shows the round and when to reset the scene.
|
|
24
|
+
- Checkpoint ladders (`analysis.ladder`): Mantel's test of a linear association between
|
|
25
|
+
training step and success, and plateau detection by fixed-sequence non-inferiority
|
|
26
|
+
against the final checkpoint. Without a comparison, the association test is the
|
|
27
|
+
primary analysis. Reports gain a success-by-step chart and a plateau table.
|
|
28
|
+
- Group-sequential stopping (`analysis.stopping: {rule: group_sequential, looks: K}`):
|
|
29
|
+
- Lan–DeMets error-spending boundaries (O'Brien–Fleming or Pocock type).
|
|
30
|
+
- Interim looks via `fieldtrial interim`, the console and `POST /api/v1/studies/{study}/interim`.
|
|
31
|
+
A look reveals only "continue" or "stop" and cancels the remaining trials on a stop.
|
|
32
|
+
- A stage-wise p-value and a repeated confidence interval in the final analysis.
|
|
33
|
+
- The simulator runs planned looks as they come due (`--no-interim` to skip them).
|
|
34
|
+
- Real blinding with switching runners. The console and the REST API start and stop them
|
|
35
|
+
with each trial. `fieldtrial check-runners` checks the setup and names arms by blind
|
|
36
|
+
code only.
|
|
37
|
+
- `command` runner: runs `runners.command.template` for each trial with the arm's
|
|
38
|
+
`policy` and `serving` values. The template is split like a shell would but never
|
|
39
|
+
runs in one, and placeholders are checked at validation.
|
|
40
|
+
- The command gets its own process group and is stopped with SIGINT, then SIGTERM,
|
|
41
|
+
then SIGKILL. Logs go to `logs/`, and the environment carries the blind code only.
|
|
42
|
+
- `launch: per_arm` keeps one process per arm and sends it start and stop lines.
|
|
43
|
+
- With `success_exit_code`, a successful exit preselects the success stage.
|
|
44
|
+
- `openpi_router` runner: a websocket proxy in front of one openpi policy server per
|
|
45
|
+
arm.
|
|
46
|
+
- It refuses arms whose metadata differ, passes `Api-Key` headers through and forwards
|
|
47
|
+
frames unchanged to the current trial's arm.
|
|
48
|
+
- It records request latency per trial. Install with `fieldtrial[openpi]`.
|
|
49
|
+
- A runner that cannot start a trial marks it invalid with the reason, so the slot is
|
|
50
|
+
rescheduled.
|
|
51
|
+
- Reports gain a descriptive runner table per arm: requests, errors, latency and
|
|
52
|
+
abnormal exits.
|
|
53
|
+
- Rig drift checks. Compare a photo of the rig with a reference photo: shift by phase
|
|
54
|
+
correlation, brightness change, and structural similarity (SSIM) after alignment.
|
|
55
|
+
- Run them with `fieldtrial rig-check DIR PHOTO` (or `--camera`), or attach a photo
|
|
56
|
+
when a console session starts.
|
|
57
|
+
- A flagged check shows as a console warning and as a report deviation, and the report
|
|
58
|
+
lists every check.
|
|
59
|
+
- Evaluation-camera capture (`fieldtrial[capture]`: OpenCV and PyAV). With
|
|
60
|
+
`capture.camera` in `study.yaml`, every trial is recorded from Start to Stop and the
|
|
61
|
+
clip is attached to the trial. The rig is checked from the camera at session start and
|
|
62
|
+
every `capture.drift.every_trials` trials. A camera failure never blocks a trial.
|
|
63
|
+
- Links between trials and LeRobot v3.0 dataset episodes (`fieldtrial[lerobot]`:
|
|
64
|
+
pyarrow).
|
|
65
|
+
- `fieldtrial link-episodes DIR DATASET` matches trials to episodes in run order, or
|
|
66
|
+
from a `trial,episode_index` mapping file, and shows the plan by blind code before
|
|
67
|
+
writing.
|
|
68
|
+
- Episodes with DAgger `intervention` flags get intervention counts, and reports gain a
|
|
69
|
+
descriptive dataset-episodes table.
|
|
70
|
+
- A `capture:` section in `study.yaml`, always left out of the design hash.
|
|
71
|
+
- `fieldtrial.stats`:
|
|
72
|
+
- `spending_boundaries`, `constant_boundaries`, `crossing_probabilities`,
|
|
73
|
+
`sequential_test`, `repeated_interval`
|
|
74
|
+
- `trend_test`, `stratified_trend_test`, `plateau`, `plateau_paired`
|
|
75
|
+
- `crossover_test`
|
|
76
|
+
- `Results` gains optional `ladder`, `crossover` and `sequential` blocks, the lists
|
|
77
|
+
`runner`, `rig_checks` and `episodes` (empty by default), and the primary methods
|
|
78
|
+
`ladder`, `crossover` and `group_sequential` (schema version unchanged: additions only).
|
|
79
|
+
|
|
80
|
+
### Changed
|
|
81
|
+
|
|
82
|
+
- `limits.reset: carry_over` is now accepted together with `design.type: crossover_rounds`.
|
|
83
|
+
- Design hashes of existing studies are unchanged: new settings are left out of the hash
|
|
84
|
+
while they are unused.
|
|
85
|
+
- `pillow` is declared as a dependency; it was already required by matplotlib.
|
|
86
|
+
|
|
87
|
+
## [0.2.0rc1] - 2026-10-02
|
|
88
|
+
|
|
89
|
+
Release candidate of 0.2.0, published to PyPI for testing. Its changes are listed under
|
|
90
|
+
0.2.0; the release is identical apart from this changelog.
|
|
91
|
+
|
|
92
|
+
## [0.2.0a2] - 2026-10-02
|
|
93
|
+
|
|
94
|
+
Second v0.2 pre-release: real blinding with the command runner and the openpi router. Its
|
|
95
|
+
changes are listed under 0.2.0.
|
|
96
|
+
|
|
97
|
+
## [0.2.0a1] - 2026-10-02
|
|
98
|
+
|
|
99
|
+
First v0.2 pre-release: crossover rounds, checkpoint ladders and group-sequential stopping.
|
|
100
|
+
Its changes are listed under 0.2.0.
|
|
101
|
+
|
|
102
|
+
## [0.1.0] - 2026-10-02
|
|
103
|
+
|
|
104
|
+
First release. Studies: design, lock, fill and analyze end to end, run them from a
|
|
105
|
+
phone-friendly operator console or your own runtime, and share a self-contained report.
|
|
106
|
+
|
|
107
|
+
### Added
|
|
108
|
+
|
|
109
|
+
- `study.yaml` schema with line-numbered validation errors, and three templates (`basic`,
|
|
110
|
+
`checkpoint-ladder`, `serving-sweep`).
|
|
111
|
+
- Randomized complete block schedules with Williams-balanced arm order and blind codes.
|
|
112
|
+
The schedule is identical on every platform and numpy version for a given seed.
|
|
113
|
+
- Locking and amendments: the design, its hash and the schedule are stored in a SQLite
|
|
114
|
+
database in the study folder; every change is one transaction plus one event-log row.
|
|
115
|
+
- Invalid trials are kept and rescheduled at the end of their block.
|
|
116
|
+
- A simulated runner and auto-operator, and CSV import (all or nothing, `--map`) and
|
|
117
|
+
CSV/JSONL export.
|
|
118
|
+
- The analysis engine: the primary analysis follows the locked design (McNemar with a Tango
|
|
119
|
+
interval, CMH, or Cochran's Q with Holm; a threshold test for single-arm studies), with
|
|
120
|
+
an independent Boschloo/Newcombe sensitivity analysis, stage funnels, time to success,
|
|
121
|
+
per-condition and per-session tables, drift and invalid-trial checks, deviations and
|
|
122
|
+
provenance. The `Results` model is versioned, and its JSON schema is published.
|
|
123
|
+
- Report wording from one tested template per situation, and a Markdown report.
|
|
124
|
+
- New statistics: stage distribution and funnel, Brunner–Munzel on stages, success-time
|
|
125
|
+
curve and bootstrap median time, CMH test, and homogeneity tests for drift.
|
|
126
|
+
- Commands `init`, `validate`, `plan`, `lock`, `amend`, `simulate`, `import`, `export`,
|
|
127
|
+
`status`, `unblind`, `analyze` and `report`.
|
|
128
|
+
- Documentation: "Running a study" and "Analysis and reports".
|
|
129
|
+
- The operator console (`fieldtrial serve`): start a session with the rig checklist, run
|
|
130
|
+
trials by blind code with a timer, label the furthest stage, termination and failure tags,
|
|
131
|
+
undo within 10 seconds, mark invalid trials, correct labels with a logged reason, a live
|
|
132
|
+
mirror screen, keyboard and foot-pedal keys, light and dark themes. `--lan` serves phones
|
|
133
|
+
on the local network behind an access token with a QR code.
|
|
134
|
+
- `fieldtrial demo`: a half-run simulated study in the console.
|
|
135
|
+
- REST API v1 with an OpenAPI document, Server-Sent Events, idempotency keys and optimistic
|
|
136
|
+
concurrency, and `fieldtrial.client`, a dependency-free Python client.
|
|
137
|
+
- Reports list label corrections made after unblinding as deviations.
|
|
138
|
+
- Documentation: "Operator console", "REST API and client" and a phone test checklist.
|
|
139
|
+
- A self-contained HTML report with seven charts (`fieldtrial report`, now the default
|
|
140
|
+
format, and in the console): no scripts and nothing loaded from other hosts.
|
|
141
|
+
- An example re-analysis of Dream Machines' published pi0.5 fine-tuning results.
|
|
142
|
+
- Documentation: quickstart, concepts, guides (planning, comparing checkpoints, ladders,
|
|
143
|
+
serving sweeps, LeRobot, openpi, custom runtimes), and screenshots.
|
|
144
|
+
- `CITATION.cff`.
|
|
145
|
+
|
|
146
|
+
## [0.1.0rc1] - 2026-10-02
|
|
147
|
+
|
|
148
|
+
Release candidate of 0.1.0, published to PyPI for testing. Its changes are listed under
|
|
149
|
+
0.1.0; the release is identical apart from this changelog.
|
|
150
|
+
|
|
151
|
+
## [0.1.0a1] - 2026-10-01
|
|
152
|
+
|
|
153
|
+
First alpha: the statistics core and the calculator commands.
|
|
154
|
+
|
|
155
|
+
### Added
|
|
156
|
+
|
|
157
|
+
- `fieldtrial.stats`, the statistics core (numpy and scipy only):
|
|
158
|
+
- one-arm intervals: Wilson (default), Clopper–Pearson, Jeffreys, Agresti–Coull, and an
|
|
159
|
+
exact binomial test against a threshold
|
|
160
|
+
- two independent arms: Newcombe difference CI, Boschloo exact test (primary), Fisher's
|
|
161
|
+
exact test and the conditional odds ratio
|
|
162
|
+
- paired designs: exact McNemar, Tango score CI, Cochran's Q with pairwise McNemar
|
|
163
|
+
- multiplicity adjustments: Holm, Bonferroni, Benjamini–Hochberg
|
|
164
|
+
- planning: sample size, power and minimum detectable effect (pooled-z, Fleiss
|
|
165
|
+
continuity-corrected, arcsine; unequal allocation), exact power for Boschloo and
|
|
166
|
+
McNemar, and a seeded simulation cross-check
|
|
167
|
+
- Bayesian summaries (descriptive only): P(p_B > p_A) and credible intervals
|
|
168
|
+
- Calculator commands `fieldtrial ci`, `compare`, `paired`, `power`, `mde` and `adjust`,
|
|
169
|
+
each with `--json` output.
|
|
170
|
+
- Documentation: a statistics reference page per method, a command-line page and the API
|
|
171
|
+
reference.
|
|
172
|
+
- Project skeleton: packaging, `fieldtrial --version`, CI and the documentation site.
|
|
173
|
+
|
|
174
|
+
[Unreleased]: https://github.com/rokbenko/fieldtrial/compare/v0.2.0...HEAD
|
|
175
|
+
[0.2.0]: https://github.com/rokbenko/fieldtrial/compare/v0.1.0...v0.2.0
|
|
176
|
+
[0.2.0rc1]: https://github.com/rokbenko/fieldtrial/compare/v0.2.0a2...v0.2.0rc1
|
|
177
|
+
[0.2.0a2]: https://github.com/rokbenko/fieldtrial/compare/v0.2.0a1...v0.2.0a2
|
|
178
|
+
[0.2.0a1]: https://github.com/rokbenko/fieldtrial/compare/v0.1.0...v0.2.0a1
|
|
179
|
+
[0.1.0]: https://github.com/rokbenko/fieldtrial/compare/v0.1.0a1...v0.1.0
|
|
180
|
+
[0.1.0rc1]: https://github.com/rokbenko/fieldtrial/compare/v0.1.0a1...v0.1.0rc1
|
|
181
|
+
[0.1.0a1]: https://github.com/rokbenko/fieldtrial/releases/tag/v0.1.0a1
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: fieldtrial
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Find out whether your robot policy actually got better: statistically rigorous real-world evaluation of robot policies.
|
|
5
|
+
Project-URL: Homepage, https://github.com/rokbenko/fieldtrial
|
|
6
|
+
Project-URL: Repository, https://github.com/rokbenko/fieldtrial
|
|
7
|
+
Project-URL: Issues, https://github.com/rokbenko/fieldtrial/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/rokbenko/fieldtrial/blob/main/CHANGELOG.md
|
|
9
|
+
Author: Rok Benko
|
|
10
|
+
License-Expression: Apache-2.0
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: confidence intervals,hypothesis testing,lerobot,policy evaluation,robot learning,robotics,statistics,vla
|
|
13
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Mathematics
|
|
25
|
+
Classifier: Typing :: Typed
|
|
26
|
+
Requires-Python: >=3.11
|
|
27
|
+
Requires-Dist: alembic>=1.13
|
|
28
|
+
Requires-Dist: fastapi>=0.115
|
|
29
|
+
Requires-Dist: jinja2>=3.1
|
|
30
|
+
Requires-Dist: matplotlib>=3.8
|
|
31
|
+
Requires-Dist: numpy>=1.26
|
|
32
|
+
Requires-Dist: pillow>=10
|
|
33
|
+
Requires-Dist: pydantic>=2.7
|
|
34
|
+
Requires-Dist: python-multipart>=0.0.9
|
|
35
|
+
Requires-Dist: pyyaml>=6.0
|
|
36
|
+
Requires-Dist: qrcode>=7.4
|
|
37
|
+
Requires-Dist: rich>=13.8
|
|
38
|
+
Requires-Dist: scipy>=1.13
|
|
39
|
+
Requires-Dist: sqlalchemy>=2.0.30
|
|
40
|
+
Requires-Dist: sse-starlette>=2.1
|
|
41
|
+
Requires-Dist: typer>=0.15
|
|
42
|
+
Requires-Dist: uvicorn[standard]>=0.30
|
|
43
|
+
Provides-Extra: capture
|
|
44
|
+
Requires-Dist: av>=12; extra == 'capture'
|
|
45
|
+
Requires-Dist: opencv-python-headless>=4.8; extra == 'capture'
|
|
46
|
+
Provides-Extra: dev
|
|
47
|
+
Requires-Dist: httpx>=0.28; extra == 'dev'
|
|
48
|
+
Requires-Dist: hypothesis>=6.130; extra == 'dev'
|
|
49
|
+
Requires-Dist: import-linter>=2.1; extra == 'dev'
|
|
50
|
+
Requires-Dist: mypy>=2.0; extra == 'dev'
|
|
51
|
+
Requires-Dist: pre-commit>=4.0; extra == 'dev'
|
|
52
|
+
Requires-Dist: pytest-cov>=6.0; extra == 'dev'
|
|
53
|
+
Requires-Dist: pytest>=8.3; extra == 'dev'
|
|
54
|
+
Requires-Dist: ruff>=0.16; extra == 'dev'
|
|
55
|
+
Requires-Dist: scipy-stubs>=1.15; extra == 'dev'
|
|
56
|
+
Requires-Dist: statsmodels>=0.14.4; extra == 'dev'
|
|
57
|
+
Requires-Dist: types-pyyaml>=6.0; extra == 'dev'
|
|
58
|
+
Provides-Extra: docs
|
|
59
|
+
Requires-Dist: mkdocs-material>=9.6; extra == 'docs'
|
|
60
|
+
Requires-Dist: mkdocs<2,>=1.6; extra == 'docs'
|
|
61
|
+
Requires-Dist: mkdocstrings[python]>=0.29; extra == 'docs'
|
|
62
|
+
Provides-Extra: lerobot
|
|
63
|
+
Requires-Dist: pyarrow>=15; extra == 'lerobot'
|
|
64
|
+
Provides-Extra: openpi
|
|
65
|
+
Requires-Dist: websockets>=13; extra == 'openpi'
|
|
66
|
+
Description-Content-Type: text/markdown
|
|
67
|
+
|
|
68
|
+
# fieldtrial
|
|
69
|
+
|
|
70
|
+
**Find out whether your robot policy actually got better.**
|
|
71
|
+
|
|
72
|
+
[](https://github.com/rokbenko/fieldtrial/actions/workflows/ci.yml)
|
|
73
|
+
[](https://pypi.org/project/fieldtrial/)
|
|
74
|
+
[](https://github.com/rokbenko/fieldtrial/blob/main/LICENSE)
|
|
75
|
+
|
|
76
|
+
fieldtrial is an open-source Python framework for statistically rigorous, real-world
|
|
77
|
+
evaluation of robot policies. LeRobot trains the policy; fieldtrial tells you whether it
|
|
78
|
+
actually got better.
|
|
79
|
+
|
|
80
|
+
## Why
|
|
81
|
+
|
|
82
|
+
Real-world evaluation is slow, noisy and ad hoc. With 40 rollouts, the 95% confidence
|
|
83
|
+
interval for a success rate is 12 to 15 percentage points wide on each side, so many apparent
|
|
84
|
+
improvements from a fine-tuning run are just noise. fieldtrial helps at every stage:
|
|
85
|
+
|
|
86
|
+
- **Before:** how many rollouts do I need? What is the smallest difference this
|
|
87
|
+
evaluation can detect?
|
|
88
|
+
- **During:** a randomized, blinded schedule, and a phone-friendly operator console that
|
|
89
|
+
records the outcome, the furthest stage reached and the failure mode of each rollout.
|
|
90
|
+
- **After:** the right statistics for the design (exact tests, paired analyses,
|
|
91
|
+
multiplicity control), honest wording, and a self-contained report.
|
|
92
|
+
|
|
93
|
+
## 30-second demo
|
|
94
|
+
|
|
95
|
+
```console
|
|
96
|
+
$ uvx fieldtrial compare 74/80 91/120
|
|
97
|
+
Arm 1: 74/80 = 92.5% (95% CI 84.6%–96.5%)
|
|
98
|
+
Arm 2: 91/120 = 75.8% (95% CI 67.4%–82.6%)
|
|
99
|
+
Difference +16.7 pp, Newcombe 95% CI [+6.2 pp, +26.0 pp]
|
|
100
|
+
Boschloo p = 0.0020 (two-sided); rejects H0 at α = 0.05
|
|
101
|
+
Fisher p = 0.0022
|
|
102
|
+
|
|
103
|
+
$ uvx fieldtrial power --p1 0.76 --p2 0.90 # rollouts needed to detect 76% → 90%
|
|
104
|
+
112 per arm (pooled-z)
|
|
105
|
+
|
|
106
|
+
$ uvx fieldtrial demo # a simulated study in the console
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
`fieldtrial demo` opens a half-run, blinded study with a simulated robot. Run the rest
|
|
110
|
+
from the keyboard (Space, Space, Enter), unblind, and read the report.
|
|
111
|
+
|
|
112
|
+
<p>
|
|
113
|
+
<img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/console-running.png" alt="The operator console on a phone: a trial in progress with its blind code, timer and a large Stop button" width="260">
|
|
114
|
+
<img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/console-label.png" alt="Labelling a trial: furthest stage reached, why it ended, failure tags" width="260">
|
|
115
|
+
</p>
|
|
116
|
+
<p>
|
|
117
|
+
<img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/report.png" alt="The HTML report: summary, success rate per arm with confidence intervals, primary analysis and a forest plot" width="560">
|
|
118
|
+
</p>
|
|
119
|
+
|
|
120
|
+
<!-- A short GIF of a trial run in the console goes here. -->
|
|
121
|
+
|
|
122
|
+
## Quickstart
|
|
123
|
+
|
|
124
|
+
```console
|
|
125
|
+
$ uv tool install fieldtrial # or: pip install fieldtrial
|
|
126
|
+
$ fieldtrial init my-study # writes my-study/study.yaml
|
|
127
|
+
$ fieldtrial plan my-study --baseline 0.75
|
|
128
|
+
$ fieldtrial lock my-study # freezes the design and randomizes the schedule
|
|
129
|
+
$ fieldtrial serve my-study --lan # scan the QR code with a phone at the robot
|
|
130
|
+
$ fieldtrial unblind my-study
|
|
131
|
+
$ fieldtrial report my-study # my-study/reports/report.html
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
A study is one folder: `study.yaml` (the design) and a SQLite database. Everything is
|
|
135
|
+
local and works offline; fieldtrial sends no telemetry.
|
|
136
|
+
|
|
137
|
+
- [Documentation](https://rokbenko.github.io/fieldtrial/): quickstart, concepts, guides
|
|
138
|
+
and a statistics reference.
|
|
139
|
+
- [Re-analysis of Dream Machines' published pi0.5 results](https://github.com/rokbenko/fieldtrial/tree/main/examples/dream-machines-pi05):
|
|
140
|
+
which of 30 published comparisons the data actually resolve.
|
|
141
|
+
|
|
142
|
+
## What you get
|
|
143
|
+
|
|
144
|
+
- **Design:**
|
|
145
|
+
- randomized complete blocks with balanced arm order
|
|
146
|
+
- crossover rounds for tasks whose scene carries over
|
|
147
|
+
- checkpoint ladders
|
|
148
|
+
- planned interim looks with early stopping
|
|
149
|
+
- blind codes, a design hash, locking and logged amendments
|
|
150
|
+
- **Real blinding:** a command runner that launches your rollout command for each arm,
|
|
151
|
+
and an openpi router that sends each trial's requests to that arm's policy server.
|
|
152
|
+
- **Console:** sessions with a rig checklist, a timer, stage and failure-tag labels,
|
|
153
|
+
10-second undo, invalid trials with automatic rescheduling, a live mirror screen,
|
|
154
|
+
keyboard and foot-pedal keys, LAN access with a QR code.
|
|
155
|
+
- **Evidence:** an evaluation camera that records every trial, rig checks against a
|
|
156
|
+
reference photo, and links from trials to LeRobot dataset episodes (with DAgger
|
|
157
|
+
interventions).
|
|
158
|
+
- **Analysis:** the primary test follows from the locked design:
|
|
159
|
+
- exact McNemar with a Tango interval, Cochran–Mantel–Haenszel, or Cochran's Q with Holm
|
|
160
|
+
- group-sequential boundaries
|
|
161
|
+
- the association of success with training step, and plateau detection
|
|
162
|
+
- a period-adjusted crossover test
|
|
163
|
+
|
|
164
|
+
Every analysis also gets an independent-samples sensitivity analysis, stage funnels,
|
|
165
|
+
time to success, drift checks and a list of every deviation from the plan.
|
|
166
|
+
- **Reports:** self-contained HTML with charts, Markdown for pull requests, and a
|
|
167
|
+
versioned JSON results model.
|
|
168
|
+
- **Integration:** a REST API with a dependency-free Python client for custom runtimes,
|
|
169
|
+
CSV import and export, and `fieldtrial.stats` as a library.
|
|
170
|
+
|
|
171
|
+
Every statistical function is tested against an independent reference implementation.
|
|
172
|
+
Reports only describe a difference when the pre-registered test rejects; otherwise they
|
|
173
|
+
say what the study could have detected.
|
|
174
|
+
|
|
175
|
+
## How it fits with LeRobot and openpi
|
|
176
|
+
|
|
177
|
+
fieldtrial complements LeRobot and openpi and never forks them. It is not a training
|
|
178
|
+
framework, a simulation benchmark, a robot driver, a labeling platform or a cloud
|
|
179
|
+
service. In manual mode, fieldtrial schedules and records the trials and you run the
|
|
180
|
+
robot however you like. For real blinding, the `command` runner launches your rollout
|
|
181
|
+
command (for example `lerobot-rollout`) for each arm, and the `openpi_router` runner
|
|
182
|
+
routes an openpi client's traffic to each trial's policy server; see
|
|
183
|
+
[Real blinding with runners](https://rokbenko.github.io/fieldtrial/guides/runners/).
|
|
184
|
+
|
|
185
|
+
## How to cite
|
|
186
|
+
|
|
187
|
+
If fieldtrial helps your research, please cite it (see
|
|
188
|
+
[CITATION.cff](https://github.com/rokbenko/fieldtrial/blob/main/CITATION.cff)):
|
|
189
|
+
|
|
190
|
+
```bibtex
|
|
191
|
+
@software{benko_fieldtrial,
|
|
192
|
+
author = {Benko, Rok},
|
|
193
|
+
title = {fieldtrial: statistically rigorous real-world evaluation for robot policies},
|
|
194
|
+
url = {https://github.com/rokbenko/fieldtrial},
|
|
195
|
+
license = {Apache-2.0}
|
|
196
|
+
}
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
For evaluation practice in general, see Kress-Gazit et al. (2024), *Robot Learning as an
|
|
200
|
+
Empirical Science: Best Practices for Policy Evaluation*, arXiv:2409.09491.
|
|
201
|
+
|
|
202
|
+
## Development
|
|
203
|
+
|
|
204
|
+
See [CONTRIBUTING.md](https://github.com/rokbenko/fieldtrial/blob/main/CONTRIBUTING.md).
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
# fieldtrial
|
|
2
|
+
|
|
3
|
+
**Find out whether your robot policy actually got better.**
|
|
4
|
+
|
|
5
|
+
[](https://github.com/rokbenko/fieldtrial/actions/workflows/ci.yml)
|
|
6
|
+
[](https://pypi.org/project/fieldtrial/)
|
|
7
|
+
[](https://github.com/rokbenko/fieldtrial/blob/main/LICENSE)
|
|
8
|
+
|
|
9
|
+
fieldtrial is an open-source Python framework for statistically rigorous, real-world
|
|
10
|
+
evaluation of robot policies. LeRobot trains the policy; fieldtrial tells you whether it
|
|
11
|
+
actually got better.
|
|
12
|
+
|
|
13
|
+
## Why
|
|
14
|
+
|
|
15
|
+
Real-world evaluation is slow, noisy and ad hoc. With 40 rollouts, the 95% confidence
|
|
16
|
+
interval for a success rate is 12 to 15 percentage points wide on each side, so many apparent
|
|
17
|
+
improvements from a fine-tuning run are just noise. fieldtrial helps at every stage:
|
|
18
|
+
|
|
19
|
+
- **Before:** how many rollouts do I need? What is the smallest difference this
|
|
20
|
+
evaluation can detect?
|
|
21
|
+
- **During:** a randomized, blinded schedule, and a phone-friendly operator console that
|
|
22
|
+
records the outcome, the furthest stage reached and the failure mode of each rollout.
|
|
23
|
+
- **After:** the right statistics for the design (exact tests, paired analyses,
|
|
24
|
+
multiplicity control), honest wording, and a self-contained report.
|
|
25
|
+
|
|
26
|
+
## 30-second demo
|
|
27
|
+
|
|
28
|
+
```console
|
|
29
|
+
$ uvx fieldtrial compare 74/80 91/120
|
|
30
|
+
Arm 1: 74/80 = 92.5% (95% CI 84.6%–96.5%)
|
|
31
|
+
Arm 2: 91/120 = 75.8% (95% CI 67.4%–82.6%)
|
|
32
|
+
Difference +16.7 pp, Newcombe 95% CI [+6.2 pp, +26.0 pp]
|
|
33
|
+
Boschloo p = 0.0020 (two-sided); rejects H0 at α = 0.05
|
|
34
|
+
Fisher p = 0.0022
|
|
35
|
+
|
|
36
|
+
$ uvx fieldtrial power --p1 0.76 --p2 0.90 # rollouts needed to detect 76% → 90%
|
|
37
|
+
112 per arm (pooled-z)
|
|
38
|
+
|
|
39
|
+
$ uvx fieldtrial demo # a simulated study in the console
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
`fieldtrial demo` opens a half-run, blinded study with a simulated robot. Run the rest
|
|
43
|
+
from the keyboard (Space, Space, Enter), unblind, and read the report.
|
|
44
|
+
|
|
45
|
+
<p>
|
|
46
|
+
<img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/console-running.png" alt="The operator console on a phone: a trial in progress with its blind code, timer and a large Stop button" width="260">
|
|
47
|
+
<img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/console-label.png" alt="Labelling a trial: furthest stage reached, why it ended, failure tags" width="260">
|
|
48
|
+
</p>
|
|
49
|
+
<p>
|
|
50
|
+
<img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/report.png" alt="The HTML report: summary, success rate per arm with confidence intervals, primary analysis and a forest plot" width="560">
|
|
51
|
+
</p>
|
|
52
|
+
|
|
53
|
+
<!-- A short GIF of a trial run in the console goes here. -->
|
|
54
|
+
|
|
55
|
+
## Quickstart
|
|
56
|
+
|
|
57
|
+
```console
|
|
58
|
+
$ uv tool install fieldtrial # or: pip install fieldtrial
|
|
59
|
+
$ fieldtrial init my-study # writes my-study/study.yaml
|
|
60
|
+
$ fieldtrial plan my-study --baseline 0.75
|
|
61
|
+
$ fieldtrial lock my-study # freezes the design and randomizes the schedule
|
|
62
|
+
$ fieldtrial serve my-study --lan # scan the QR code with a phone at the robot
|
|
63
|
+
$ fieldtrial unblind my-study
|
|
64
|
+
$ fieldtrial report my-study # my-study/reports/report.html
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
A study is one folder: `study.yaml` (the design) and a SQLite database. Everything is
|
|
68
|
+
local and works offline; fieldtrial sends no telemetry.
|
|
69
|
+
|
|
70
|
+
- [Documentation](https://rokbenko.github.io/fieldtrial/): quickstart, concepts, guides
|
|
71
|
+
and a statistics reference.
|
|
72
|
+
- [Re-analysis of Dream Machines' published pi0.5 results](https://github.com/rokbenko/fieldtrial/tree/main/examples/dream-machines-pi05):
|
|
73
|
+
which of 30 published comparisons the data actually resolve.
|
|
74
|
+
|
|
75
|
+
## What you get
|
|
76
|
+
|
|
77
|
+
- **Design:**
|
|
78
|
+
- randomized complete blocks with balanced arm order
|
|
79
|
+
- crossover rounds for tasks whose scene carries over
|
|
80
|
+
- checkpoint ladders
|
|
81
|
+
- planned interim looks with early stopping
|
|
82
|
+
- blind codes, a design hash, locking and logged amendments
|
|
83
|
+
- **Real blinding:** a command runner that launches your rollout command for each arm,
|
|
84
|
+
and an openpi router that sends each trial's requests to that arm's policy server.
|
|
85
|
+
- **Console:** sessions with a rig checklist, a timer, stage and failure-tag labels,
|
|
86
|
+
10-second undo, invalid trials with automatic rescheduling, a live mirror screen,
|
|
87
|
+
keyboard and foot-pedal keys, LAN access with a QR code.
|
|
88
|
+
- **Evidence:** an evaluation camera that records every trial, rig checks against a
|
|
89
|
+
reference photo, and links from trials to LeRobot dataset episodes (with DAgger
|
|
90
|
+
interventions).
|
|
91
|
+
- **Analysis:** the primary test follows from the locked design:
|
|
92
|
+
- exact McNemar with a Tango interval, Cochran–Mantel–Haenszel, or Cochran's Q with Holm
|
|
93
|
+
- group-sequential boundaries
|
|
94
|
+
- the association of success with training step, and plateau detection
|
|
95
|
+
- a period-adjusted crossover test
|
|
96
|
+
|
|
97
|
+
Every analysis also gets an independent-samples sensitivity analysis, stage funnels,
|
|
98
|
+
time to success, drift checks and a list of every deviation from the plan.
|
|
99
|
+
- **Reports:** self-contained HTML with charts, Markdown for pull requests, and a
|
|
100
|
+
versioned JSON results model.
|
|
101
|
+
- **Integration:** a REST API with a dependency-free Python client for custom runtimes,
|
|
102
|
+
CSV import and export, and `fieldtrial.stats` as a library.
|
|
103
|
+
|
|
104
|
+
Every statistical function is tested against an independent reference implementation.
|
|
105
|
+
Reports only describe a difference when the pre-registered test rejects; otherwise they
|
|
106
|
+
say what the study could have detected.
|
|
107
|
+
|
|
108
|
+
## How it fits with LeRobot and openpi
|
|
109
|
+
|
|
110
|
+
fieldtrial complements LeRobot and openpi and never forks them. It is not a training
|
|
111
|
+
framework, a simulation benchmark, a robot driver, a labeling platform or a cloud
|
|
112
|
+
service. In manual mode, fieldtrial schedules and records the trials and you run the
|
|
113
|
+
robot however you like. For real blinding, the `command` runner launches your rollout
|
|
114
|
+
command (for example `lerobot-rollout`) for each arm, and the `openpi_router` runner
|
|
115
|
+
routes an openpi client's traffic to each trial's policy server; see
|
|
116
|
+
[Real blinding with runners](https://rokbenko.github.io/fieldtrial/guides/runners/).
|
|
117
|
+
|
|
118
|
+
## How to cite
|
|
119
|
+
|
|
120
|
+
If fieldtrial helps your research, please cite it (see
|
|
121
|
+
[CITATION.cff](https://github.com/rokbenko/fieldtrial/blob/main/CITATION.cff)):
|
|
122
|
+
|
|
123
|
+
```bibtex
|
|
124
|
+
@software{benko_fieldtrial,
|
|
125
|
+
author = {Benko, Rok},
|
|
126
|
+
title = {fieldtrial: statistically rigorous real-world evaluation for robot policies},
|
|
127
|
+
url = {https://github.com/rokbenko/fieldtrial},
|
|
128
|
+
license = {Apache-2.0}
|
|
129
|
+
}
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
For evaluation practice in general, see Kress-Gazit et al. (2024), *Robot Learning as an
|
|
133
|
+
Empirical Science: Best Practices for Policy Evaluation*, arXiv:2409.09491.
|
|
134
|
+
|
|
135
|
+
## Development
|
|
136
|
+
|
|
137
|
+
See [CONTRIBUTING.md](https://github.com/rokbenko/fieldtrial/blob/main/CONTRIBUTING.md).
|